From 71619891e0d17f0075e29a253fa5e57210a82ff4 Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Sat, 25 Jul 2026 01:52:16 -0700 Subject: [PATCH] Fix iGPU-only Vulkan hosts budgeting GGUF against 0 GiB The iGPU exclusion in ggufMemoryTotalGb is right when llama-server is masked to an iGPU beside a dGPU torch can see, but it also catches the plain APU case: a Strix Halo or 890M box on a Vulkan build enumerates one integrated device and nothing else, so the budget collapses to 0 and the Hub renders "0 GiB" on a machine llama-server reports 91 GiB for. Fit still resolves via system RAM, so this is a labelling and fit-surface defect rather than a wrong load, but it reads as OOM on every GGUF row. Fall back to the torch total when the inventory is all-iGPU and torch sees no device beyond the ones probed. The device-count check is what keeps the masked-beside-a-dGPU host on 0, which is the case the exclusion exists for. Tests: the shape was uncovered (the existing cases are all is_igpu False). Added a backend test that an iGPU-only inventory is surfaced with its real total and keeps picks supported, and a source contract guard on the frontend rule, matching how the other frontend fixes here are pinned. Verified: frontend typecheck clean (same 5 pre-existing missing-module errors as main, none new), 211 passed across the vulkan probe, gguf devices, gpu memory mode, chat-load and model-picker contract suites. --- .../backend/tests/test_system_gguf_devices.py | 36 +++++++++++++++++++ studio/frontend/src/hooks/use-gpu-info.ts | 11 +++++- tests/studio/test_model_picker_contracts.py | 20 +++++++++++ 3 files changed, 66 insertions(+), 1 deletion(-) diff --git a/studio/backend/tests/test_system_gguf_devices.py b/studio/backend/tests/test_system_gguf_devices.py index 561d4328c0..b721c63249 100644 --- a/studio/backend/tests/test_system_gguf_devices.py +++ b/studio/backend/tests/test_system_gguf_devices.py @@ -159,3 +159,39 @@ def test_non_vulkan_build_reports_no_gguf_inventory(main_module, monkeypatch): if __name__ == "__main__": raise SystemExit(pytest.main([__file__, "-v"])) + + +_IGPU_ONLY_INVENTORY = [ + { + "index": 0, + "name": "AMD Radeon(TM) 8060S Graphics", + "free_mib": 89 * 1024, + "total_mib": 91 * 1024, + "is_igpu": True, + } +] + + +def test_igpu_only_vulkan_inventory_reports_the_igpu_and_its_total( + main_module, monkeypatch +): + """An APU box on a Vulkan build (Strix Halo) enumerates one integrated device + and no discrete card. The backend must still surface it, flagged is_igpu with + its real total, so the frontend can tell this shape apart from a masked probe + and budget GGUF against the APU pool instead of labelling everything OOM. + """ + info = _gpu_info( + main_module, monkeypatch, is_vulkan = True, inventory = _IGPU_ONLY_INVENTORY + ) + + assert len(info["gguf_devices"]) == 1 + dev = info["gguf_devices"][0] + assert dev["is_igpu"] is True + assert dev["index"] == 0 and dev["index_kind"] == "vulkan" + assert dev["name"] == "AMD Radeon(TM) 8060S Graphics" + # The total is reported, not zeroed: the zeroing is the frontend's budgeting + # decision, and it needs the real number to fall back to. + assert dev["memory_total_gb"] == 91.0 + # A probed device means an ordinal exists, so picks stay supported. + assert info["gguf_gpu_ids_supported"] is True + assert info["gguf_backend_is_vulkan"] is True diff --git a/studio/frontend/src/hooks/use-gpu-info.ts b/studio/frontend/src/hooks/use-gpu-info.ts index 38129be787..5a0a57f06a 100644 --- a/studio/frontend/src/hooks/use-gpu-info.ts +++ b/studio/frontend/src/hooks/use-gpu-info.ts @@ -136,6 +136,13 @@ function toGpuInfo(data: SystemInfoResponse | null): GpuInfo { // see a dGPU llama-server never enumerated, and reusing that total would let // fit checks pass against VRAM /load can't actually place. const isVulkanBuild = gpuData?.gguf_backend_is_vulkan === true; + // An all-iGPU inventory that torch sees no extra device beyond means there is + // no discrete card anywhere (an APU box, e.g. Strix Halo on a Vulkan build). + // The exclusion above would budget 0 and label it "0 GiB"; the APU's pool is + // real and torch reports it, so use that. The device-count check keeps the + // masked-to-iGPU-beside-a-dGPU case on 0, which is what the exclusion is for. + const ggufIgpuOnly = + ggufDevices.length > 0 && ggufDevices.every((d) => d.is_igpu === true); if (!gpuData?.available || !devices.length) { // Torch sees no GPU (training stays CPU-bound / unavailable), but a Vulkan // llama.cpp build may still drive GPUs for GGUF: surface that budget alone. @@ -151,7 +158,9 @@ function toGpuInfo(data: SystemInfoResponse | null): GpuInfo { name: devices[0]?.name ?? "Unknown", memoryTotalGb, ggufMemoryTotalGb: ggufDevices.length - ? ggufDeviceTotalGb + ? ggufIgpuOnly && devices.length <= ggufDevices.length + ? memoryTotalGb + : ggufDeviceTotalGb : isVulkanBuild ? 0 : memoryTotalGb, diff --git a/tests/studio/test_model_picker_contracts.py b/tests/studio/test_model_picker_contracts.py index 635e5a9749..d0c6889c37 100644 --- a/tests/studio/test_model_picker_contracts.py +++ b/tests/studio/test_model_picker_contracts.py @@ -508,3 +508,23 @@ def test_legacy_migration_is_idempotent_and_non_destructive(): # Layer 3: non-overwriting merge skips an existing (or default) key, so even a # forced re-run cannot duplicate or clobber a user's config. assert "if (isDefaultConfig(migrated) || Object.hasOwn(map, key)) {" in src + + +def test_igpu_only_vulkan_budget_falls_back_to_the_torch_total(): + """An APU on a Vulkan build (Strix Halo, Radeon 890M) enumerates one iGPU and + nothing else. The iGPU exclusion in the GGUF budget would make that 0, so the + Hub renders "0 GiB" and every GGUF row labels OOM against a box whose pool is + real and reported by torch. Fall back to the torch total for that shape only; + a masked-to-iGPU-beside-a-dGPU host (torch sees more devices than the probe + enumerated) must stay on 0, which is what the exclusion exists for. + """ + src = " ".join(_read("hooks/use-gpu-info.ts").split()) + # All-iGPU inventory detection, and the device-count guard that keeps the + # mixed masked host on the conservative 0. + assert ( + "const ggufIgpuOnly = ggufDevices.length > 0 && " + "ggufDevices.every((d) => d.is_igpu === true);" in src + ) + assert "ggufIgpuOnly && devices.length <= ggufDevices.length" in src + # Training/safetensors budgets must stay on the torch total regardless. + assert "memoryTotalGb," in src