fix(provision): address Codex review (3 P2s on the aarch64 CUDA provisioner)
- find_nvcc now prefers the highest /usr/local/cuda-<ver> toolkit so a stale unversioned `cuda` symlink or an older nvcc earlier on PATH can't win and rebuild with CUDA 12.x (re-hitting the glibc>=2.41 / Blackwell clash this script avoids); falls back to a PATH nvcc only when no versioned toolkit. - Validate the GPU compute_cap is purely numeric before using it as CMAKE_CUDA_ARCHITECTURES: some WSL GPU-PV / driver combos report "N/A", which CMake rejects (aborting an otherwise-usable build) instead of letting "native" autodetect. - Gate the native-Linux aarch64 provisioner on _SKIP_GGUF_BUILD: when a non-root user declines the sudo prompt (or lacks sudo) for GGUF deps, don't then run a provisioner that does its own sudo apt-get installs.
This commit is contained in:
parent
73bf482f65
commit
6006402fa2
2 changed files with 19 additions and 3 deletions
|
|
@ -50,8 +50,17 @@ if [ "$HAVE_APT" -eq 1 ]; then
|
|||
$SUDO apt-get install -y --no-install-recommends gcc-14 g++-14 >/dev/null 2>&1 || true
|
||||
fi
|
||||
|
||||
# 3. Locate nvcc; install the CUDA toolkit if missing.
|
||||
find_nvcc() { command -v nvcc 2>/dev/null || ls /usr/local/cuda*/bin/nvcc 2>/dev/null | sort -V | tail -1; }
|
||||
# 3. Locate nvcc; install the CUDA toolkit if missing. Prefer the highest
|
||||
# /usr/local/cuda-<ver> toolkit: a stale unversioned `cuda` symlink or an older
|
||||
# nvcc earlier on PATH could otherwise win and rebuild with CUDA 12.x, re-hitting
|
||||
# the glibc>=2.41 / Blackwell clash this script exists to avoid. Fall back to a
|
||||
# PATH nvcc (e.g. conda) only when no versioned system toolkit is present.
|
||||
find_nvcc() {
|
||||
local _v
|
||||
_v="$(ls -d /usr/local/cuda-*/bin/nvcc 2>/dev/null | sort -V | tail -1)"
|
||||
if [ -n "$_v" ]; then printf '%s\n' "$_v"; return 0; fi
|
||||
command -v nvcc 2>/dev/null || ls /usr/local/cuda*/bin/nvcc 2>/dev/null | sort -V | tail -1
|
||||
}
|
||||
NVCC="$(find_nvcc)"
|
||||
if [ -z "$NVCC" ] && [ "$HAVE_APT" -eq 1 ]; then
|
||||
log "CUDA toolkit (nvcc) not found - installing CUDA 13.3 (matches torch cu13x; avoids glibc>=2.41 rsqrt clash)"
|
||||
|
|
@ -99,8 +108,14 @@ HCXX=g++; command -v g++-14 >/dev/null 2>&1 && HCXX=g++-14
|
|||
export CC="$HCC" CXX="$HCXX" CUDAHOSTCXX="$HCXX"
|
||||
|
||||
# 5. CUDA arch from the GPU's compute capability (e.g. "12.1" -> 121). Fallback: native.
|
||||
# Only a purely-numeric capability is a valid CMAKE_CUDA_ARCHITECTURES; some WSL
|
||||
# GPU-PV / driver combos report "N/A", which CMake would reject (aborting an
|
||||
# otherwise-usable build) instead of letting "native" autodetect.
|
||||
CC_CAP="$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader 2>/dev/null | head -1 | tr -d ' .')"
|
||||
if [ -n "$CC_CAP" ]; then CUDA_ARCH="$CC_CAP"; else CUDA_ARCH="native"; fi
|
||||
case "$CC_CAP" in
|
||||
''|*[!0-9]*) CUDA_ARCH="native" ;;
|
||||
*) CUDA_ARCH="$CC_CAP" ;;
|
||||
esac
|
||||
|
||||
# 6. Clone + build into ~/.unsloth/llama.cpp, honoring a UNSLOTH_LLAMA_TAG pin
|
||||
# (same var setup.sh uses) instead of always tracking ggml-org main.
|
||||
|
|
|
|||
|
|
@ -1731,6 +1731,7 @@ if [ "$_HOST_SYSTEM" = "Linux" ] \
|
|||
&& { [ "$_HOST_MACHINE" = "aarch64" ] || [ "$_HOST_MACHINE" = "arm64" ]; } \
|
||||
&& { ! grep -qi microsoft /proc/version 2>/dev/null || [ "${UNSLOTH_WSL_LLAMA_DEFERRED:-0}" != "1" ]; } \
|
||||
&& [ "${UNSLOTH_NO_LLAMA_CUDA:-0}" != "1" ] \
|
||||
&& [ "${_SKIP_GGUF_BUILD:-}" != true ] \
|
||||
&& command -v nvidia-smi >/dev/null 2>&1 \
|
||||
&& nvidia-smi -L 2>/dev/null | awk '/^GPU[[:space:]]+[0-9]+:/{found=1} END{exit !found}' \
|
||||
&& ! _have_cuda_llama_server; then
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue