Studio: refine CPU footprint accounting and defer the abort memo
Follow-up review on the CPU-only / NUMA hardening: - Account for MTP memory in the CPU context fit. mtp_engaged alone is a no-op once budget_frac is set, so pass the byte-accurate MTP overhead fn; the cap no longer ignores the MTP draft KV. - Fit auto contexts at or below the 32k ceiling too, not only above it, so a large small-context GGUF that still exceeds RAM is reduced toward the minimum instead of being OS-killed at startup. - Recompute the NUMA footprint MTP reserve at the post-cap context; the pre-cap reserve overstated a capped million-token model and could skip a viable interleave. - Surface the "interleave cannot help" decision (footprint exceeds total RAM across all nodes), not only the missing-numactl case. - Defer recording the scheduler-abort memo until the text-only mmproj fallback is ruled out, so a vision model that recovers text-only is not blocked by the fail-fast guard on its next load. Unit tests and the offline simulations updated and passing.
This commit is contained in:
parent
32c9de2201
commit
e7ab561c92
3 changed files with 82 additions and 20 deletions
|
|
@ -4974,7 +4974,6 @@ class LlamaCppBackend:
|
|||
)
|
||||
model_size = None # set in the fit try; used by the APU RAM guard
|
||||
model_size_fit = None # weights + compute buffer; set in the fit try
|
||||
_mtp_reserve_bytes = 0 # MTP draft reserve at effective_ctx; set in the fit try
|
||||
# Layer-fallback min GPUs; raised below on a tensor downgrade. Bound
|
||||
# before the try so the --fit-on except path still has it (no UnboundLocal).
|
||||
_layer_min_gpus = 1
|
||||
|
|
@ -5724,9 +5723,12 @@ class LlamaCppBackend:
|
|||
model_size / (1024**3),
|
||||
_avail_mib / 1024,
|
||||
)
|
||||
# Cap an auto context to a RAM-aware ceiling (explicit -c is honored).
|
||||
if requested_ctx <= 0 and effective_ctx > _CPU_CTX_AUTO_CEILING:
|
||||
_cpu_cap = _CPU_CTX_AUTO_CEILING
|
||||
# Fit an auto context to a RAM-aware ceiling (explicit -c is honored).
|
||||
# Runs for every auto context, not only above the ceiling: a large
|
||||
# small-context GGUF can still exceed RAM and must be reduced too.
|
||||
if requested_ctx <= 0 and effective_ctx > 0:
|
||||
_ctx_ceiling = min(effective_ctx, _CPU_CTX_AUTO_CEILING)
|
||||
_cpu_cap = _ctx_ceiling
|
||||
try:
|
||||
if _avail_mib and model_size and self._can_estimate_kv():
|
||||
_budget_b = _avail_mib * _CPU_RAM_BUDGET_FRAC * 1024 * 1024
|
||||
|
|
@ -5736,22 +5738,25 @@ class LlamaCppBackend:
|
|||
_fixed = model_size_fit or model_size
|
||||
if _fixed >= _budget_b:
|
||||
# Footprint alone over budget: _fit_context_to_vram returns
|
||||
# the ceiling unchanged, but KV at 32k could OOM. Floor to
|
||||
# the ceiling unchanged, but KV could still OOM. Floor to
|
||||
# the minimum so the tightest fit gets the smallest context.
|
||||
_cpu_cap = 4096
|
||||
else:
|
||||
_fit = self._fit_context_to_vram(
|
||||
requested_ctx = _CPU_CTX_AUTO_CEILING,
|
||||
requested_ctx = _ctx_ceiling,
|
||||
available_mib = _avail_mib,
|
||||
model_size_bytes = _fixed,
|
||||
cache_type_kv = cache_type_kv,
|
||||
min_ctx = 4096,
|
||||
n_parallel = n_parallel,
|
||||
kv_on_gpu = True, # KV lives in the RAM budget we fit
|
||||
mtp_engaged = True, # flat reserve; no GPU draft here
|
||||
# mtp_engaged alone is a no-op once budget_frac is set;
|
||||
# pass the byte-accurate overhead so MTP KV is reserved.
|
||||
mtp_engaged = _mtp_will_engage,
|
||||
mtp_overhead_fn = (_mtp_bytes if _mtp_will_engage else None),
|
||||
budget_frac = _CPU_RAM_BUDGET_FRAC,
|
||||
)
|
||||
_cpu_cap = max(4096, min(_CPU_CTX_AUTO_CEILING, _fit))
|
||||
_cpu_cap = max(4096, min(_ctx_ceiling, _fit))
|
||||
except Exception as _cap_exc: # best-effort; fall back to ceiling
|
||||
logger.debug("CPU context-fit failed; using ceiling: %s", _cap_exc)
|
||||
if _cpu_cap < effective_ctx:
|
||||
|
|
@ -6030,12 +6035,15 @@ class LlamaCppBackend:
|
|||
_numa_footprint = _resident
|
||||
if _resident and effective_ctx > 0 and self._can_estimate_kv():
|
||||
try:
|
||||
# Recompute MTP at the post-cap context; the pre-cap
|
||||
# _mtp_reserve_bytes would overstate a capped million-token load.
|
||||
_numa_mtp = _mtp_bytes(effective_ctx) if _mtp_will_engage else 0
|
||||
_numa_footprint = (
|
||||
_resident
|
||||
+ self._estimate_kv_cache_bytes(
|
||||
effective_ctx, cache_type_kv, n_parallel = n_parallel
|
||||
)
|
||||
+ _mtp_reserve_bytes
|
||||
+ _numa_mtp
|
||||
)
|
||||
except Exception:
|
||||
_numa_footprint = _resident
|
||||
|
|
@ -6045,7 +6053,12 @@ class LlamaCppBackend:
|
|||
if not _extra_args_set_any_flag(extra_args, {"--numa"}):
|
||||
cmd.extend(["--numa", "distribute"])
|
||||
logger.info("NUMA: %s", _numa.reason)
|
||||
elif _cpu_only and "numactl` is not installed" in _numa.reason:
|
||||
elif _cpu_only and (
|
||||
"numactl` is not installed" in _numa.reason
|
||||
or "interleave cannot help" in _numa.reason
|
||||
):
|
||||
# Actionable: numactl missing, or the footprint exceeds total RAM
|
||||
# across all nodes (the weights-only preflight can't catch this).
|
||||
logger.warning("NUMA: %s", _numa.reason)
|
||||
except Exception as _numa_exc: # never block a load on the NUMA probe
|
||||
logger.debug("NUMA interleave probe failed: %s", _numa_exc)
|
||||
|
|
@ -6432,15 +6445,16 @@ class LlamaCppBackend:
|
|||
out = "\n".join(self._stdout_lines[-50:])
|
||||
# Read the crash code before _kill_process() clears _process.
|
||||
_crash_rc = self._process.poll() if self._process is not None else None
|
||||
# Graph-scheduler abort (flash-off retry already failed): memo it so
|
||||
# the next /load fails fast. Wider slice as the GGML_ASSERT line can
|
||||
# scroll past the [New LWP] dump (backtrace markers stay in the tail).
|
||||
if (
|
||||
# Graph-scheduler abort (flash-off retry already failed)? Capture now (a
|
||||
# text-only retry overwrites the stdout tail) but memo it only once the
|
||||
# mmproj fallback is ruled out, so a VLM that recovers text-only is not
|
||||
# blocked by the fail-fast guard on its next load. Wider slice as the
|
||||
# GGML_ASSERT line can scroll past the [New LWP] dump.
|
||||
_was_sched_abort = (
|
||||
not self._cancel_event.is_set()
|
||||
and (self._is_signal_crash(_crash_rc) or self._is_abort_exit(_crash_rc))
|
||||
and self._is_sched_reserve_abort("\n".join(self._stdout_lines[-200:]))
|
||||
):
|
||||
LlamaCppBackend._record_sched_reserve_abort(binary, _abort_memo_model)
|
||||
)
|
||||
self._kill_process()
|
||||
# The #6415 split-axis abort is latched earlier (first spawn).
|
||||
# Skip if a cancel/unload is pending (mirrors the MTP guard).
|
||||
|
|
@ -6470,6 +6484,9 @@ class LlamaCppBackend:
|
|||
# an OS-killed text-only retry still gets the OOM message.
|
||||
_retry_rc = self._process.poll() if self._process is not None else None
|
||||
self._kill_process()
|
||||
# Fallback exhausted: now memo the original scheduler abort.
|
||||
if _was_sched_abort:
|
||||
LlamaCppBackend._record_sched_reserve_abort(binary, _abort_memo_model)
|
||||
raise RuntimeError(
|
||||
"Vision projector incompatible with this llama.cpp "
|
||||
"build, and the text-only retry also failed: "
|
||||
|
|
@ -6481,6 +6498,10 @@ class LlamaCppBackend:
|
|||
)
|
||||
)
|
||||
else:
|
||||
# No mmproj fallback available/eligible: memo the scheduler abort
|
||||
# (if that is what crashed) so the next /load fails fast, then raise.
|
||||
if _was_sched_abort:
|
||||
LlamaCppBackend._record_sched_reserve_abort(binary, _abort_memo_model)
|
||||
raise RuntimeError(
|
||||
self._classify_llama_start_failure(
|
||||
out,
|
||||
|
|
|
|||
|
|
@ -106,12 +106,29 @@ def test_cpu_context_ceiling_constant_is_sane():
|
|||
|
||||
def test_cpu_only_caps_auto_context_but_honors_explicit():
|
||||
src = _load_model_src()
|
||||
# Cap only fires for an auto context (requested_ctx <= 0) over the ceiling;
|
||||
# an explicit -c (requested_ctx > 0) is honored.
|
||||
assert "requested_ctx <= 0 and effective_ctx > _CPU_CTX_AUTO_CEILING" in src
|
||||
# The fit runs for any auto context (requested_ctx <= 0), against a ceiling of
|
||||
# min(native, 32k); an explicit -c (requested_ctx > 0) is honored untouched.
|
||||
assert "requested_ctx <= 0 and effective_ctx > 0" in src
|
||||
assert "_ctx_ceiling = min(effective_ctx, _CPU_CTX_AUTO_CEILING)" in src
|
||||
assert "effective_ctx = _cpu_cap" in src
|
||||
|
||||
|
||||
def test_cpu_context_fit_runs_below_ceiling_too():
|
||||
"""A large GGUF with a native context already <= 32k must still be RAM-fit, not
|
||||
skipped, so it can be reduced toward 4096 instead of OS-killed (PR review fix)."""
|
||||
src = _load_model_src()
|
||||
# The gate is `> 0`, not `> _CPU_CTX_AUTO_CEILING`, and the fit ceiling is clamped.
|
||||
assert "effective_ctx > _CPU_CTX_AUTO_CEILING" not in src
|
||||
assert "requested_ctx = _ctx_ceiling" in src
|
||||
|
||||
|
||||
def test_cpu_context_fit_accounts_for_mtp():
|
||||
"""mtp_engaged alone is a no-op once budget_frac is set, so the CPU fit must pass the
|
||||
byte-accurate MTP overhead fn to actually reserve MTP KV (PR review fix)."""
|
||||
src = _load_model_src()
|
||||
assert "mtp_overhead_fn = (_mtp_bytes if _mtp_will_engage else None)" in src
|
||||
|
||||
|
||||
def test_cpu_context_cap_reuses_fit_helper_against_ram():
|
||||
"""The cap reuses _fit_context_to_vram with the system-RAM budget, not VRAM."""
|
||||
src = _load_model_src()
|
||||
|
|
@ -151,11 +168,20 @@ def test_numa_decision_uses_footprint_not_just_weights():
|
|||
assert "decide_interleave(_numa_footprint" in src
|
||||
# Footprint = fitted weights (incl. compute buffer) + KV + MTP reserve.
|
||||
assert "_resident = model_size_fit or model_size" in src
|
||||
assert "_mtp_reserve_bytes" in src
|
||||
# MTP is recomputed at the post-cap context, not the stale pre-cap reserve.
|
||||
assert "_numa_mtp = _mtp_bytes(effective_ctx) if _mtp_will_engage else 0" in src
|
||||
# KV must be sized for the launched --parallel slots, not the n_parallel=1 default.
|
||||
assert "effective_ctx, cache_type_kv, n_parallel = n_parallel" in src
|
||||
|
||||
|
||||
def test_numa_surfaces_total_ram_failure():
|
||||
"""When the footprint exceeds total RAM across all nodes, decide_interleave returns
|
||||
an actionable 'interleave cannot help' reason; the caller must surface it, not only
|
||||
the missing-numactl case (PR review fix)."""
|
||||
src = _load_model_src()
|
||||
assert '"interleave cannot help" in _numa.reason' in src
|
||||
|
||||
|
||||
def test_extra_args_forces_cpu_offload_helper():
|
||||
"""The zero-offload detector: -ngl 0 / --n-gpu-layers 0 / --gpu-layers 0 (last wins)."""
|
||||
from core.inference.llama_cpp import _extra_args_forces_cpu_offload as f
|
||||
|
|
|
|||
|
|
@ -274,3 +274,18 @@ def test_load_model_records_abort_on_crash():
|
|||
src = _load_model_src()
|
||||
assert "_record_sched_reserve_abort(binary, _abort_memo_model)" in src
|
||||
assert "_is_sched_reserve_abort(" in src
|
||||
|
||||
|
||||
def test_abort_memo_deferred_until_mmproj_fallback_ruled_out():
|
||||
"""The signature is captured up front but the memo is recorded only on a terminal
|
||||
raise, after the text-only mmproj fallback is ruled out, so a VLM that recovers
|
||||
text-only is not blocked by the fail-fast guard next time (PR review fix)."""
|
||||
src = _load_model_src()
|
||||
assert "_was_sched_abort = (" in src
|
||||
assert "if _was_sched_abort:" in src
|
||||
fn = ast.parse(src).body[0]
|
||||
strip_line = _call_line(fn, "_strip_mmproj_args")
|
||||
record_line = _call_line(fn, "_record_sched_reserve_abort")
|
||||
# Recording happens after the projector strip, i.e. only once the fallback is tried.
|
||||
assert strip_line is not None and record_line is not None
|
||||
assert record_line > strip_line
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue