Studio: refine CPU footprint accounting and defer the abort memo

Follow-up review on the CPU-only / NUMA hardening:

- Account for MTP memory in the CPU context fit. mtp_engaged alone is a no-op
  once budget_frac is set, so pass the byte-accurate MTP overhead fn; the cap no
  longer ignores the MTP draft KV.
- Fit auto contexts at or below the 32k ceiling too, not only above it, so a
  large small-context GGUF that still exceeds RAM is reduced toward the minimum
  instead of being OS-killed at startup.
- Recompute the NUMA footprint MTP reserve at the post-cap context; the pre-cap
  reserve overstated a capped million-token model and could skip a viable
  interleave.
- Surface the "interleave cannot help" decision (footprint exceeds total RAM
  across all nodes), not only the missing-numactl case.
- Defer recording the scheduler-abort memo until the text-only mmproj fallback
  is ruled out, so a vision model that recovers text-only is not blocked by the
  fail-fast guard on its next load.

Unit tests and the offline simulations updated and passing.
This commit is contained in:
Daniel Han 2026-06-29 07:26:32 +00:00
commit e7ab561c92
3 changed files with 82 additions and 20 deletions

View file

@ -4974,7 +4974,6 @@ class LlamaCppBackend:
)
model_size = None # set in the fit try; used by the APU RAM guard
model_size_fit = None # weights + compute buffer; set in the fit try
_mtp_reserve_bytes = 0 # MTP draft reserve at effective_ctx; set in the fit try
# Layer-fallback min GPUs; raised below on a tensor downgrade. Bound
# before the try so the --fit-on except path still has it (no UnboundLocal).
_layer_min_gpus = 1
@ -5724,9 +5723,12 @@ class LlamaCppBackend:
model_size / (1024**3),
_avail_mib / 1024,
)
# Cap an auto context to a RAM-aware ceiling (explicit -c is honored).
if requested_ctx <= 0 and effective_ctx > _CPU_CTX_AUTO_CEILING:
_cpu_cap = _CPU_CTX_AUTO_CEILING
# Fit an auto context to a RAM-aware ceiling (explicit -c is honored).
# Runs for every auto context, not only above the ceiling: a large
# small-context GGUF can still exceed RAM and must be reduced too.
if requested_ctx <= 0 and effective_ctx > 0:
_ctx_ceiling = min(effective_ctx, _CPU_CTX_AUTO_CEILING)
_cpu_cap = _ctx_ceiling
try:
if _avail_mib and model_size and self._can_estimate_kv():
_budget_b = _avail_mib * _CPU_RAM_BUDGET_FRAC * 1024 * 1024
@ -5736,22 +5738,25 @@ class LlamaCppBackend:
_fixed = model_size_fit or model_size
if _fixed >= _budget_b:
# Footprint alone over budget: _fit_context_to_vram returns
# the ceiling unchanged, but KV at 32k could OOM. Floor to
# the ceiling unchanged, but KV could still OOM. Floor to
# the minimum so the tightest fit gets the smallest context.
_cpu_cap = 4096
else:
_fit = self._fit_context_to_vram(
requested_ctx = _CPU_CTX_AUTO_CEILING,
requested_ctx = _ctx_ceiling,
available_mib = _avail_mib,
model_size_bytes = _fixed,
cache_type_kv = cache_type_kv,
min_ctx = 4096,
n_parallel = n_parallel,
kv_on_gpu = True, # KV lives in the RAM budget we fit
mtp_engaged = True, # flat reserve; no GPU draft here
# mtp_engaged alone is a no-op once budget_frac is set;
# pass the byte-accurate overhead so MTP KV is reserved.
mtp_engaged = _mtp_will_engage,
mtp_overhead_fn = (_mtp_bytes if _mtp_will_engage else None),
budget_frac = _CPU_RAM_BUDGET_FRAC,
)
_cpu_cap = max(4096, min(_CPU_CTX_AUTO_CEILING, _fit))
_cpu_cap = max(4096, min(_ctx_ceiling, _fit))
except Exception as _cap_exc: # best-effort; fall back to ceiling
logger.debug("CPU context-fit failed; using ceiling: %s", _cap_exc)
if _cpu_cap < effective_ctx:
@ -6030,12 +6035,15 @@ class LlamaCppBackend:
_numa_footprint = _resident
if _resident and effective_ctx > 0 and self._can_estimate_kv():
try:
# Recompute MTP at the post-cap context; the pre-cap
# _mtp_reserve_bytes would overstate a capped million-token load.
_numa_mtp = _mtp_bytes(effective_ctx) if _mtp_will_engage else 0
_numa_footprint = (
_resident
+ self._estimate_kv_cache_bytes(
effective_ctx, cache_type_kv, n_parallel = n_parallel
)
+ _mtp_reserve_bytes
+ _numa_mtp
)
except Exception:
_numa_footprint = _resident
@ -6045,7 +6053,12 @@ class LlamaCppBackend:
if not _extra_args_set_any_flag(extra_args, {"--numa"}):
cmd.extend(["--numa", "distribute"])
logger.info("NUMA: %s", _numa.reason)
elif _cpu_only and "numactl` is not installed" in _numa.reason:
elif _cpu_only and (
"numactl` is not installed" in _numa.reason
or "interleave cannot help" in _numa.reason
):
# Actionable: numactl missing, or the footprint exceeds total RAM
# across all nodes (the weights-only preflight can't catch this).
logger.warning("NUMA: %s", _numa.reason)
except Exception as _numa_exc: # never block a load on the NUMA probe
logger.debug("NUMA interleave probe failed: %s", _numa_exc)
@ -6432,15 +6445,16 @@ class LlamaCppBackend:
out = "\n".join(self._stdout_lines[-50:])
# Read the crash code before _kill_process() clears _process.
_crash_rc = self._process.poll() if self._process is not None else None
# Graph-scheduler abort (flash-off retry already failed): memo it so
# the next /load fails fast. Wider slice as the GGML_ASSERT line can
# scroll past the [New LWP] dump (backtrace markers stay in the tail).
if (
# Graph-scheduler abort (flash-off retry already failed)? Capture now (a
# text-only retry overwrites the stdout tail) but memo it only once the
# mmproj fallback is ruled out, so a VLM that recovers text-only is not
# blocked by the fail-fast guard on its next load. Wider slice as the
# GGML_ASSERT line can scroll past the [New LWP] dump.
_was_sched_abort = (
not self._cancel_event.is_set()
and (self._is_signal_crash(_crash_rc) or self._is_abort_exit(_crash_rc))
and self._is_sched_reserve_abort("\n".join(self._stdout_lines[-200:]))
):
LlamaCppBackend._record_sched_reserve_abort(binary, _abort_memo_model)
)
self._kill_process()
# The #6415 split-axis abort is latched earlier (first spawn).
# Skip if a cancel/unload is pending (mirrors the MTP guard).
@ -6470,6 +6484,9 @@ class LlamaCppBackend:
# an OS-killed text-only retry still gets the OOM message.
_retry_rc = self._process.poll() if self._process is not None else None
self._kill_process()
# Fallback exhausted: now memo the original scheduler abort.
if _was_sched_abort:
LlamaCppBackend._record_sched_reserve_abort(binary, _abort_memo_model)
raise RuntimeError(
"Vision projector incompatible with this llama.cpp "
"build, and the text-only retry also failed: "
@ -6481,6 +6498,10 @@ class LlamaCppBackend:
)
)
else:
# No mmproj fallback available/eligible: memo the scheduler abort
# (if that is what crashed) so the next /load fails fast, then raise.
if _was_sched_abort:
LlamaCppBackend._record_sched_reserve_abort(binary, _abort_memo_model)
raise RuntimeError(
self._classify_llama_start_failure(
out,

View file

@ -106,12 +106,29 @@ def test_cpu_context_ceiling_constant_is_sane():
def test_cpu_only_caps_auto_context_but_honors_explicit():
src = _load_model_src()
# Cap only fires for an auto context (requested_ctx <= 0) over the ceiling;
# an explicit -c (requested_ctx > 0) is honored.
assert "requested_ctx <= 0 and effective_ctx > _CPU_CTX_AUTO_CEILING" in src
# The fit runs for any auto context (requested_ctx <= 0), against a ceiling of
# min(native, 32k); an explicit -c (requested_ctx > 0) is honored untouched.
assert "requested_ctx <= 0 and effective_ctx > 0" in src
assert "_ctx_ceiling = min(effective_ctx, _CPU_CTX_AUTO_CEILING)" in src
assert "effective_ctx = _cpu_cap" in src
def test_cpu_context_fit_runs_below_ceiling_too():
"""A large GGUF with a native context already <= 32k must still be RAM-fit, not
skipped, so it can be reduced toward 4096 instead of OS-killed (PR review fix)."""
src = _load_model_src()
# The gate is `> 0`, not `> _CPU_CTX_AUTO_CEILING`, and the fit ceiling is clamped.
assert "effective_ctx > _CPU_CTX_AUTO_CEILING" not in src
assert "requested_ctx = _ctx_ceiling" in src
def test_cpu_context_fit_accounts_for_mtp():
"""mtp_engaged alone is a no-op once budget_frac is set, so the CPU fit must pass the
byte-accurate MTP overhead fn to actually reserve MTP KV (PR review fix)."""
src = _load_model_src()
assert "mtp_overhead_fn = (_mtp_bytes if _mtp_will_engage else None)" in src
def test_cpu_context_cap_reuses_fit_helper_against_ram():
"""The cap reuses _fit_context_to_vram with the system-RAM budget, not VRAM."""
src = _load_model_src()
@ -151,11 +168,20 @@ def test_numa_decision_uses_footprint_not_just_weights():
assert "decide_interleave(_numa_footprint" in src
# Footprint = fitted weights (incl. compute buffer) + KV + MTP reserve.
assert "_resident = model_size_fit or model_size" in src
assert "_mtp_reserve_bytes" in src
# MTP is recomputed at the post-cap context, not the stale pre-cap reserve.
assert "_numa_mtp = _mtp_bytes(effective_ctx) if _mtp_will_engage else 0" in src
# KV must be sized for the launched --parallel slots, not the n_parallel=1 default.
assert "effective_ctx, cache_type_kv, n_parallel = n_parallel" in src
def test_numa_surfaces_total_ram_failure():
"""When the footprint exceeds total RAM across all nodes, decide_interleave returns
an actionable 'interleave cannot help' reason; the caller must surface it, not only
the missing-numactl case (PR review fix)."""
src = _load_model_src()
assert '"interleave cannot help" in _numa.reason' in src
def test_extra_args_forces_cpu_offload_helper():
"""The zero-offload detector: -ngl 0 / --n-gpu-layers 0 / --gpu-layers 0 (last wins)."""
from core.inference.llama_cpp import _extra_args_forces_cpu_offload as f

View file

@ -274,3 +274,18 @@ def test_load_model_records_abort_on_crash():
src = _load_model_src()
assert "_record_sched_reserve_abort(binary, _abort_memo_model)" in src
assert "_is_sched_reserve_abort(" in src
def test_abort_memo_deferred_until_mmproj_fallback_ruled_out():
"""The signature is captured up front but the memo is recorded only on a terminal
raise, after the text-only mmproj fallback is ruled out, so a VLM that recovers
text-only is not blocked by the fail-fast guard next time (PR review fix)."""
src = _load_model_src()
assert "_was_sched_abort = (" in src
assert "if _was_sched_abort:" in src
fn = ast.parse(src).body[0]
strip_line = _call_line(fn, "_strip_mmproj_args")
record_line = _call_line(fn, "_record_sched_reserve_abort")
# Recording happens after the projector strip, i.e. only once the fallback is tried.
assert strip_line is not None and record_line is not None
assert record_line > strip_line