diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index eb8387c6ca..e97c5e3457 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -198,8 +198,11 @@ class LlamaCppBackend: """Query free memory per GPU via nvidia-smi. Returns list of (gpu_index, free_mib) sorted by index. - Returns empty list if nvidia-smi is not available. + Only returns GPUs that are allowed by CUDA_VISIBLE_DEVICES + (if set). Returns empty list if nvidia-smi is not available. """ + import os + try: result = subprocess.run( [ @@ -213,12 +216,24 @@ class LlamaCppBackend: ) if result.returncode != 0: return [] + + # Parse which GPUs are allowed by existing CUDA_VISIBLE_DEVICES + allowed = None + cvd = os.environ.get("CUDA_VISIBLE_DEVICES") + if cvd is not None and cvd.strip(): + try: + allowed = set(int(x.strip()) for x in cvd.split(",")) + except ValueError: + pass # Non-numeric (e.g., "GPU-uuid"), ignore filter + gpus = [] for line in result.stdout.strip().splitlines(): parts = line.split(",") if len(parts) == 2: idx = int(parts[0].strip()) free_mib = int(parts[1].strip()) + if allowed is not None and idx not in allowed: + continue gpus.append((idx, free_mib)) return gpus except Exception: