diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index a57c160a8e..bc3734fde2 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -204,12 +204,11 @@ class LlamaCppBackend: def _get_gpu_free_memory() -> list[tuple[int, int]]: """Query free memory per GPU via nvidia-smi. - Returns list of (gpu_index, free_mib) sorted by index. - Only returns GPUs that are allowed by CUDA_VISIBLE_DEVICES - (if set). Returns empty list if nvidia-smi is not available. + Returns list of (gpu_index, free_mib) for ALL physical GPUs, + ignoring CUDA_VISIBLE_DEVICES. llama-server manages its own GPU + allocation via _select_gpus which sets CUDA_VISIBLE_DEVICES + explicitly for the subprocess. """ - import os - try: result = subprocess.run( [ @@ -224,23 +223,12 @@ class LlamaCppBackend: if result.returncode != 0: return [] - # Parse which GPUs are allowed by existing CUDA_VISIBLE_DEVICES - allowed = None - cvd = os.environ.get("CUDA_VISIBLE_DEVICES") - if cvd is not None and cvd.strip(): - try: - allowed = set(int(x.strip()) for x in cvd.split(",")) - except ValueError: - pass # Non-numeric (e.g., "GPU-uuid"), ignore filter - gpus = [] for line in result.stdout.strip().splitlines(): parts = line.split(",") if len(parts) == 2: idx = int(parts[0].strip()) free_mib = int(parts[1].strip()) - if allowed is not None and idx not in allowed: - continue gpus.append((idx, free_mib)) return gpus except Exception: