Preflight local-only loads against the resolved path and keep GGUF rows on GGUF snapshots

The load route's sidecar tier probe reached _remote_lora_base's raw HTTP
request and Hub config reads, and the training guard's memory estimation
can call hf model_info; neither honors offline mode, so both now run
against the resolved snapshot path for local-only loads (a local path
resolves from config.json on disk and skips the remote adapter probe).
GGUF cached rows select a GGUF-bearing snapshot (top level or one folder
deep) instead of reusing the safetensors-preferring model helper, so a
mixed repo's safetensors revision cannot become the GGUF row's load target
while the UI reports a quant.
This commit is contained in:
Unsloth 2026-07-27 05:24:22 -07:00
commit 3026ff9e79
4 changed files with 139 additions and 12 deletions

View file

@ -327,7 +327,7 @@ def _scan_cached_gguf() -> list[dict]:
continue
repo_id = repo_info.repo_id
repo_path = Path(repo_info.repo_path)
snapshot_path = _cached_model_snapshot_path(repo_path)
snapshot_path = _cached_gguf_repo_snapshot_path(repo_path)
total_size = _repo_gguf_size_bytes(repo_info)
has_variant_state, variant_state_size = _gguf_variant_state_summary(
repo_id,
@ -552,13 +552,12 @@ def _repo_non_gguf_model_payload(repo_info) -> _CachedNonGgufPayload:
)
def _weightful_snapshot_path(repo_path: Path) -> Optional[Path]:
"""Newest snapshot dir holding config.json plus a safetensors weight.
def _newest_snapshot_where(repo_path: Path, loadable) -> Optional[Path]:
"""Newest snapshots/* dir (by mtime) whose entry names satisfy *loadable*.
Inactive-cache rows emit this snapshot path as their load_id, which the
load consumes directly (it bypasses the repo-id resolver), so a newest
metadata-only revision must not be emitted while an older revision holds
the weights the row was classified from.
Inactive-cache rows emit the selected snapshot path as their load_id,
which the load consumes directly (it bypasses the repo-id resolver), so
the snapshot must actually hold the format the row was classified from.
"""
snapshots = repo_path / "snapshots"
try:
@ -566,14 +565,54 @@ def _weightful_snapshot_path(repo_path: Path) -> Optional[Path]:
except OSError:
return None
def _loadable(rev: Path) -> bool:
def _ok(rev: Path) -> bool:
try:
names = [entry.name for entry in rev.iterdir()]
except OSError:
return False
return "config.json" in names and any(n.endswith(".safetensors") for n in names)
return loadable(names)
candidates = [rev for rev in revisions if _loadable(rev)]
candidates = [rev for rev in revisions if _ok(rev)]
if not candidates:
return None
try:
return max(candidates, key = lambda rev: rev.stat().st_mtime).resolve()
except OSError:
return None
def _weightful_snapshot_path(repo_path: Path) -> Optional[Path]:
"""Newest snapshot holding config.json plus a safetensors weight (the
non-GGUF resolvers' complete-revision predicate)."""
return _newest_snapshot_where(
repo_path,
lambda names: "config.json" in names
and any(n.endswith(".safetensors") for n in names),
)
def _gguf_snapshot_path(repo_path: Path) -> Optional[Path]:
"""Newest snapshot holding a GGUF file (top level or one folder deep,
where multi-quant repos keep per-quant subfolders)."""
direct = _newest_snapshot_where(
repo_path,
lambda names: any(_is_gguf_filename(n.lower()) for n in names),
)
if direct is not None:
return direct
snapshots = repo_path / "snapshots"
try:
revisions = [entry for entry in snapshots.iterdir() if entry.is_dir()]
except OSError:
return None
def _has_nested_gguf(rev: Path) -> bool:
try:
return any(True for _ in rev.glob("*/*.gguf"))
except OSError:
return False
candidates = [rev for rev in revisions if _has_nested_gguf(rev)]
if not candidates:
return None
try:
@ -593,6 +632,19 @@ def _cached_model_snapshot_path(repo_path: Path) -> Optional[Path]:
return path if path.is_dir() else None
def _cached_gguf_repo_snapshot_path(repo_path: Path) -> Optional[Path]:
"""Snapshot path for a GGUF row: a safetensors-bearing revision of a
mixed repo must not become the GGUF row's load target."""
gguf_snapshot = _gguf_snapshot_path(repo_path)
if gguf_snapshot is not None:
return gguf_snapshot
resolved = hf_cache_scan.resolve_hf_cache_realpath(repo_path)
if not resolved:
return None
path = Path(resolved)
return path if path.is_dir() else None
def _read_json_object(path: Path) -> dict:
try:
with open(path, "r", encoding = "utf-8") as f:

View file

@ -4573,7 +4573,13 @@ async def _load_model_impl(
# to match. Off-loop: tier resolution reads configs.
if effective_load_in_4bit and not config.is_gguf:
from utils.transformers_version import latest_tier_active_for
if await asyncio.to_thread(latest_tier_active_for, config.identifier, request.hf_token):
# Local-only loads probe the tier against the resolved snapshot:
# a repo id would reach _remote_lora_base's raw HTTP request and
# Hub config reads, while a local path resolves from config.json
# on disk (non-canonical ids skip the remote adapter probe).
_tier_target = config.path if request.local_files_only else config.identifier
if await asyncio.to_thread(latest_tier_active_for, _tier_target, request.hf_token):
effective_load_in_4bit = False
logger.info(
f"Latest-transformers sidecar active for '{model_log_label}' - "
@ -4595,10 +4601,19 @@ async def _load_model_impl(
# Apply the training coexistence policy before the unload step below
# frees the resident model. Off-loop: the default-mode guard does sync work.
# Local-only non-GGUF loads size against the resolved snapshot so the
# guard's memory estimation reads local files instead of hf model_info
# (which performs no offline-mode check). GGUF sizing already reads
# the cached file under its own local-only handling.
_guard_identifier = (
config.path
if request.local_files_only and not config.is_gguf
else model_identifier
)
await asyncio.to_thread(
_guard_chat_load_against_training,
config,
model_identifier = model_identifier,
model_identifier = _guard_identifier,
hf_token = request.hf_token,
load_in_4bit = effective_load_in_4bit,
max_seq_length = request.max_seq_length,

View file

@ -162,6 +162,41 @@ def test_refless_fallback_picks_newest_snapshot_with_config(tmp_path):
assert Path(resolved).resolve() == new.resolve()
def test_gguf_rows_select_gguf_bearing_snapshot_in_mixed_repos(tmp_path):
"""A mixed repo caching a newer safetensors revision beside an older GGUF
revision: the GGUF row's snapshot selection must return the GGUF-bearing
revision, not the safetensors one the model row prefers."""
import os
import sys
import time
backend_dir = str(Path(__file__).resolve().parent.parent)
if backend_dir not in sys.path:
sys.path.insert(0, backend_dir)
from hub.services.models.cache_inventory import (
_cached_gguf_repo_snapshot_path,
_cached_model_snapshot_path,
)
repo_dir = tmp_path / "models--org--mixed"
gguf_rev = repo_dir / "snapshots" / ("a" * 40)
gguf_rev.mkdir(parents = True)
(gguf_rev / "mixed-Q4_K_M.gguf").write_text("gguf-bytes")
stale = time.time() - 1000
os.utime(gguf_rev, (stale, stale))
st_rev = repo_dir / "snapshots" / ("b" * 40)
st_rev.mkdir(parents = True)
(st_rev / "config.json").write_text("{}")
(st_rev / "model.safetensors").write_text("weights")
gguf_pick = _cached_gguf_repo_snapshot_path(repo_dir)
assert gguf_pick is not None
assert Path(gguf_pick).resolve() == gguf_rev.resolve()
model_pick = _cached_model_snapshot_path(repo_dir)
assert model_pick is not None
assert Path(model_pick).resolve() == st_rev.resolve()
def test_weightless_newest_snapshot_does_not_shadow_complete_older_one(tmp_path):
"""A newest metadata-only revision (config.json, no weights) must not win
over an older revision holding the inventoried safetensors weights: the

View file

@ -1324,6 +1324,31 @@ def test_local_only_gguf_reuse_and_platform_gates_are_authoritative():
assert 'if selected_category == "safetensors":' in inventory
def test_route_preflights_and_gguf_rows_stay_format_true():
"""Round-19 gates. The /load route's sidecar tier probe and training
guard size local-only candidates against the resolved snapshot path (a
repo id would reach _remote_lora_base's raw HTTP request and hf
model_info, neither of which honors offline mode). GGUF cached rows
select a GGUF-bearing snapshot, so a mixed repo's safetensors revision
cannot become the GGUF row's load target."""
route = _read_backend("routes/inference.py")
assert "_tier_target = config.path if request.local_files_only else config.identifier" in route
guard_block = route.split("_guard_identifier = (", 1)[1]
guard_block = guard_block.split("await asyncio.to_thread(", 1)[0]
assert "config.path" in guard_block
assert "request.local_files_only and not config.is_gguf" in guard_block
assert "model_identifier = _guard_identifier," in route
inventory = _read_backend("hub/services/models/cache_inventory.py")
assert "def _newest_snapshot_where(" in inventory
assert "def _gguf_snapshot_path(" in inventory
assert "def _cached_gguf_repo_snapshot_path(" in inventory
gguf_scan = inventory.split("def _scan_cached_gguf(", 1)[1]
gguf_scan = gguf_scan.split("def ", 1)[0]
assert "_cached_gguf_repo_snapshot_path(repo_path)" in gguf_scan
assert "_cached_model_snapshot_path(repo_path)" not in gguf_scan
def test_gguf_background_loads_never_download_companions():
"""A cached GGUF load can still fetch from the Hub through its optional
companions (mmproj, MTP drafter) or a cache-miss main quant. Background