Merge branch 'main' into tool-call-confirmation

This commit is contained in:
Daniel Han 2026-05-31 01:49:04 -07:00 committed by GitHub
commit ce3fddbc28
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
40 changed files with 7754 additions and 433 deletions

View file

@ -17,6 +17,7 @@ import struct
import structlog
from loggers import get_logger
import shutil
import signal
import socket
import subprocess
import sys
@ -965,9 +966,6 @@ class LlamaCppBackend:
7. llama-server on PATH (system install)
8. ./bin/llama-server (legacy: extracted binary)
"""
import os
import sys
binary_name = "llama-server.exe" if sys.platform == "win32" else "llama-server"
# 1. Env var — direct path to binary
@ -1238,6 +1236,33 @@ class LlamaCppBackend:
return total
@staticmethod
def _amd_apu_wants_unified_memory() -> bool:
"""True only for AMD unified-memory APUs (gfx1150/gfx1151), where
GGML_CUDA_ENABLE_UNIFIED_MEMORY lets llama.cpp use shared system RAM.
False for discrete AMD, NVIDIA, CPU and macOS (the env hurts discrete
GPUs). ROCm reuses torch.cuda.*; the gcnArchName suffix is stripped."""
try:
import torch
if getattr(torch.version, "hip", None) is None:
return False
if not (hasattr(torch, "cuda") and torch.cuda.is_available()):
return False
for _i in range(torch.cuda.device_count()):
try:
_arch = (
getattr(torch.cuda.get_device_properties(_i), "gcnArchName", "")
or ""
)
except Exception:
continue
if _arch.split(":")[0].strip().lower() in {"gfx1150", "gfx1151"}:
return True
except Exception:
return False
return False
@staticmethod
def _get_gpu_free_memory() -> list[tuple[int, int]]:
"""Query free memory per GPU.
@ -1255,8 +1280,6 @@ class LlamaCppBackend:
Returns list of (gpu_index, free_mib) sorted by index. Empty
list if no supported GPU is reachable.
"""
import os
# ── NVIDIA via nvidia-smi ────────────────────────────────────
try:
result = subprocess.run(
@ -3158,6 +3181,14 @@ class LlamaCppBackend:
env = child_env_without_native_path_secret()
binary_dir = str(Path(binary).parent)
# AMD unified-memory APUs (gfx1150/gfx1151): let llama.cpp use
# shared system RAM. setdefault so a user value wins.
if self._amd_apu_wants_unified_memory():
env.setdefault("GGML_CUDA_ENABLE_UNIFIED_MEMORY", "1")
logger.info(
"AMD unified-memory APU: set GGML_CUDA_ENABLE_UNIFIED_MEMORY=1"
)
if sys.platform == "win32":
# See _build_windows_path_dirs for ordering. #5106.
path_dirs = self._build_windows_path_dirs(
@ -3167,6 +3198,24 @@ class LlamaCppBackend:
)
existing_path = env.get("PATH", "")
env["PATH"] = ";".join(path_dirs) + ";" + existing_path
# ROCm: the llama.cpp prebuilt bundles its own rocblas.dll
# but NOT the Tensile kernel library files it needs
# (rocblas/library/TensileLibrary*.dat + *.hsaco). The
# bundled DLL searches relative to its own location by
# default (i.e. <binary_dir>/rocblas/library/) which does
# not exist, causing a silent crash on the first GEMM.
# ROCBLAS_TENSILE_LIBPATH overrides that search to point at
# the ROCm installation where the kernel files actually are.
_hip_path = os.environ.get(
"HIP_PATH", os.environ.get("ROCM_PATH", "")
)
if _hip_path:
_rocblas_lib = os.path.join(
_hip_path, "bin", "rocblas", "library"
)
if os.path.isdir(_rocblas_lib):
env.setdefault("ROCBLAS_TENSILE_LIBPATH", _rocblas_lib)
else:
# Linux: set LD_LIBRARY_PATH for shared libs next to the binary
# and CUDA runtime libs (libcudart, libcublas, etc.)
@ -3875,10 +3924,6 @@ class LlamaCppBackend:
Falls back to pgrep + /proc/<pid>/exe on Linux when psutil is
not installed.
"""
import os
import signal
import sys
try:
# -- Build the ownership allowlist --------------------------------
# Two kinds of matches: