fix: download all GGUF shards for split models (e.g. 7B Q8_0)
LlamaCppBackend.load_model() and precache_helper_gguf() only downloaded the first matching GGUF file. For split models (e.g. 7B Q8_0 with 3 shards), llama-server needs all shards present. Now collects and downloads all matching files.
This commit is contained in:
parent
7f1fd28acd
commit
c2dd0f4cf1
2 changed files with 32 additions and 15 deletions
|
|
@ -249,17 +249,22 @@ class LlamaCppBackend:
|
|||
)
|
||||
|
||||
# Determine the filename from the variant (e.g., "Q4_K_M" -> find matching file)
|
||||
# For split GGUFs (e.g., *-00001-of-00003.gguf) we must download ALL shards.
|
||||
gguf_filename = None
|
||||
gguf_extra_shards: list[str] = []
|
||||
if hf_variant:
|
||||
# Try common naming patterns
|
||||
try:
|
||||
from huggingface_hub import list_repo_files
|
||||
files = list_repo_files(hf_repo, token=hf_token)
|
||||
variant_lower = hf_variant.lower()
|
||||
for f in files:
|
||||
if f.endswith(".gguf") and variant_lower in f.lower():
|
||||
gguf_filename = f
|
||||
break
|
||||
matching = sorted(
|
||||
f for f in files
|
||||
if f.endswith(".gguf") and variant_lower in f.lower()
|
||||
)
|
||||
if matching:
|
||||
gguf_filename = matching[0] # first shard (or single file)
|
||||
gguf_extra_shards = matching[1:] # remaining shards if split
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not list repo files: {e}")
|
||||
|
||||
|
|
@ -269,13 +274,23 @@ class LlamaCppBackend:
|
|||
repo_name = hf_repo.split("/")[-1].replace("-GGUF", "")
|
||||
gguf_filename = f"{repo_name}-{hf_variant}.gguf"
|
||||
|
||||
logger.info(f"Downloading GGUF: {hf_repo}/{gguf_filename}")
|
||||
logger.info(f"Downloading GGUF: {hf_repo}/{gguf_filename}"
|
||||
+ (f" (+{len(gguf_extra_shards)} shards)" if gguf_extra_shards else ""))
|
||||
try:
|
||||
local_path = hf_hub_download(
|
||||
repo_id=hf_repo,
|
||||
filename=gguf_filename,
|
||||
token=hf_token,
|
||||
)
|
||||
# Download remaining shards for split GGUFs — llama-server
|
||||
# auto-discovers them when they are in the same directory.
|
||||
for shard in gguf_extra_shards:
|
||||
logger.info(f"Downloading GGUF shard: {shard}")
|
||||
hf_hub_download(
|
||||
repo_id=hf_repo,
|
||||
filename=shard,
|
||||
token=hf_token,
|
||||
)
|
||||
except Exception as e:
|
||||
raise RuntimeError(
|
||||
f"Failed to download GGUF file '{gguf_filename}' from {hf_repo}: {e}"
|
||||
|
|
|
|||
|
|
@ -47,17 +47,19 @@ def precache_helper_gguf():
|
|||
files = api.list_repo_files(repo, repo_type="model")
|
||||
gguf_files = [f for f in files if f.endswith(".gguf")]
|
||||
|
||||
target = None
|
||||
# Find all GGUF files matching the variant (may be split into shards)
|
||||
variant_lower = variant.lower().replace("-", "_")
|
||||
for f in gguf_files:
|
||||
if variant_lower in f.lower().replace("-", "_"):
|
||||
target = f
|
||||
break
|
||||
matching = sorted(
|
||||
f for f in gguf_files
|
||||
if variant_lower in f.lower().replace("-", "_")
|
||||
)
|
||||
|
||||
if target:
|
||||
logger.info(f"Pre-caching helper GGUF: {repo}/{target}")
|
||||
hf_hub_download(repo_id=repo, filename=target)
|
||||
logger.info(f"Helper GGUF cached: {target}")
|
||||
if matching:
|
||||
logger.info(f"Pre-caching helper GGUF: {repo}/{matching[0]}"
|
||||
+ (f" (+{len(matching) - 1} shards)" if len(matching) > 1 else ""))
|
||||
for target in matching:
|
||||
hf_hub_download(repo_id=repo, filename=target)
|
||||
logger.info(f"Helper GGUF cached: {len(matching)} file(s)")
|
||||
else:
|
||||
logger.warning(f"No GGUF matching variant '{variant}' in {repo}")
|
||||
except Exception as e:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue