From 66781abc58578cbfa52e5078bea4dade02209669 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Sat, 20 Jun 2026 04:11:19 +0000 Subject: [PATCH 1/3] Studio: fix llama.cpp update toast tag and reload hint The post-update toast used the job's to_tag, which is the bare bNNNN build number (same as installed_tag), so it showed e.g. "b9726" instead of the full release tag. Use status.latest_tag (e.g. b9726-mix-) to match the tag the banner already shows, falling back to to_tag and then a generic label. Also drop "Reload your model to use it." when there is nothing to reload: only append it when a local model is loaded, since external-provider models do not use llama.cpp. --- .../src/components/llama-update-banner.tsx | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/studio/frontend/src/components/llama-update-banner.tsx b/studio/frontend/src/components/llama-update-banner.tsx index 0383d8a150..0daade512b 100644 --- a/studio/frontend/src/components/llama-update-banner.tsx +++ b/studio/frontend/src/components/llama-update-banner.tsx @@ -2,6 +2,7 @@ // Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 import { Button } from "@/components/ui/button"; +import { isExternalModelId, useChatRuntimeStore } from "@/features/chat"; import { useLlamaUpdateCheck } from "@/hooks/use-llama-update-check"; import { useShowLlamaUpdateBanner } from "@/hooks/use-llama-update-pref"; import { toast } from "@/lib/toast"; @@ -99,9 +100,18 @@ export function LlamaUpdateBanner({ async function handleUpdate() { const result = await apply(); if (result?.ok) { - toast.success( - `llama.cpp updated to ${result.tag ?? "the latest build"}. Reload your model to use it.`, - ); + // Prefer the full release tag (e.g. b9726-mix-); the job's to_tag and + // installed_tag are the bare bNNNN build number. + const updatedTag = status?.latest_tag ?? result.tag ?? "the latest build"; + // Only a loaded local model can be reloaded to pick up the new binary; + // external-provider models do not use llama.cpp at all. + const checkpoint = useChatRuntimeStore.getState().params.checkpoint; + const hasLocalModel = + Boolean(checkpoint) && !isExternalModelId(checkpoint); + const reloadHint = hasLocalModel + ? " Reload your model to use it." + : ""; + toast.success(`llama.cpp updated to ${updatedTag}.${reloadHint}`); } else if (result) { toast.error( `llama.cpp update failed: ${result.error ?? "unknown error"}`, From 6ffa4bfe90ee34f999e7750c8012853778beeabc Mon Sep 17 00:00:00 2001 From: wasimysaid Date: Sun, 21 Jun 2026 12:09:39 +0200 Subject: [PATCH 2/3] Fix/adjust llama update toast for PR #6493 --- studio/backend/routes/llama.py | 1 + studio/backend/tests/test_llama_cpp_update.py | 31 +++++++++++++++++++ studio/backend/tests/test_llama_route.py | 3 +- studio/backend/utils/llama_cpp_update.py | 6 +++- .../src/components/llama-update-banner.tsx | 12 ++----- .../src/features/chat/chat-settings-sheet.tsx | 5 ++- .../src/hooks/use-llama-update-check.ts | 10 +++++- 7 files changed, 54 insertions(+), 14 deletions(-) diff --git a/studio/backend/routes/llama.py b/studio/backend/routes/llama.py index 5559cc0404..123349126d 100644 --- a/studio/backend/routes/llama.py +++ b/studio/backend/routes/llama.py @@ -30,6 +30,7 @@ class LlamaUpdateJob(BaseModel): message: str = "" from_tag: Optional[str] = None to_tag: Optional[str] = None + reload_required: Optional[bool] = None error: Optional[str] = None progress: Optional[float] = Field(None, description = "0..1 while running, 1 on success.") started_at: Optional[str] = None diff --git a/studio/backend/tests/test_llama_cpp_update.py b/studio/backend/tests/test_llama_cpp_update.py index 828d439e0a..1b509fea6c 100644 --- a/studio/backend/tests/test_llama_cpp_update.py +++ b/studio/backend/tests/test_llama_cpp_update.py @@ -445,6 +445,7 @@ def test_start_update_happy_path(monkeypatch, tmp_path): time.sleep(0.05) assert job["state"] == "success", job assert job["to_tag"] == "b9518" + assert job["reload_required"] is False # Installer was invoked with the resolved install dir + latest + repo. assert "--install-dir" in captured["cmd"] assert str(install_dir) in captured["cmd"] @@ -456,6 +457,35 @@ def test_start_update_happy_path(monkeypatch, tmp_path): assert popen_kwargs["env"]["UNSLOTH_PROGRESS_PERCENT_STEP"] == "5" +def test_start_update_reports_full_release_tag(monkeypatch, tmp_path): + install_dir = tmp_path / "llama.cpp" + binary = _write_install(install_dir, "b9595") + monkeypatch.setattr(upd, "_find_binary", lambda: binary) + monkeypatch.setattr(upd, "_installer_script", lambda: tmp_path / "install_llama_prebuilt.py") + monkeypatch.setattr( + freshness, + "_fetch_latest_release_tag", + lambda repo, timeout = 5.0: "b9596-mix-e6f2453", + ) + + def _on_start(cmd): + _write_install(install_dir, "b9596", release_tag = "b9596-mix-e6f2453") + + _patch_installer_popen(monkeypatch, on_start = _on_start) + + res = upd.start_update() + assert res["started"] is True + deadline = time.time() + 10 + while time.time() < deadline: + job = upd.get_update_status()["job"] + if job["state"] in ("success", "error"): + break + time.sleep(0.05) + assert job["state"] == "success", job + assert job["to_tag"] == "b9596-mix-e6f2453" + assert "Updated llama.cpp to b9596-mix-e6f2453." in job["message"] + + def test_start_update_installer_failure_reports_error(monkeypatch, tmp_path): install_dir = tmp_path / "llama.cpp" binary = _write_install(install_dir, "b9493") @@ -674,6 +704,7 @@ def test_update_sets_maintenance_flag_and_unloads(monkeypatch, tmp_path): time.sleep(0.05) assert backend.unloaded is True + assert upd.get_update_status()["job"]["reload_required"] is True assert seen.get("flag_during_install") is True # Cleared in the finally so model loads work again after the swap. assert backend._llama_update_in_progress is False diff --git a/studio/backend/tests/test_llama_route.py b/studio/backend/tests/test_llama_route.py index 333f710b44..0ecfeee018 100644 --- a/studio/backend/tests/test_llama_route.py +++ b/studio/backend/tests/test_llama_route.py @@ -70,10 +70,11 @@ def test_status_response_exposes_source_build(): "installed_at_utc": None, "age_days": None, "source_build": True, - "job": {"state": "idle"}, + "job": {"state": "idle", "reload_required": False}, } model = rl.LlamaUpdateStatusResponse(**payload) assert model.model_dump()["source_build"] is True + assert model.model_dump()["job"]["reload_required"] is False # Extra/unknown keys must not crash the response model. rl.LlamaUpdateStatusResponse(**{**payload, "unexpected": 1}) diff --git a/studio/backend/utils/llama_cpp_update.py b/studio/backend/utils/llama_cpp_update.py index 965f1a970a..4e15fbe074 100644 --- a/studio/backend/utils/llama_cpp_update.py +++ b/studio/backend/utils/llama_cpp_update.py @@ -62,6 +62,7 @@ _job: dict = { "message": "", "from_tag": None, "to_tag": None, + "reload_required": None, "error": None, "progress": None, "started_at": None, @@ -505,7 +506,7 @@ def _run_update(install_dir: Path, repo: str, asset: Optional[str], script: Path except Exception as exc: # pragma: no cover - network defensive logger.debug("llama update: post-install freshness refresh failed", error = str(exc)) new_marker = read_install_marker(_find_binary()) - new_tag = (new_marker or {}).get("tag") or (new_marker or {}).get("release_tag") + new_tag = (new_marker or {}).get("release_tag") or (new_marker or {}).get("tag") with _job_lock: _job.update( @@ -515,6 +516,7 @@ def _run_update(install_dir: Path, repo: str, asset: Optional[str], script: Path + (" Reload your model to use it." if model_was_active else "") ), to_tag = new_tag, + reload_required = model_was_active, error = None, progress = 1.0, finished_at = _utcnow(), @@ -618,6 +620,7 @@ def start_update() -> dict: message = "Downloading and installing the latest llama.cpp prebuilt...", from_tag = from_tag, to_tag = None, + reload_required = None, error = None, progress = 0.0, started_at = _utcnow(), @@ -643,6 +646,7 @@ def _reset_job_for_tests() -> None: message = "", from_tag = None, to_tag = None, + reload_required = None, error = None, progress = None, started_at = None, diff --git a/studio/frontend/src/components/llama-update-banner.tsx b/studio/frontend/src/components/llama-update-banner.tsx index 0daade512b..f53e6e0401 100644 --- a/studio/frontend/src/components/llama-update-banner.tsx +++ b/studio/frontend/src/components/llama-update-banner.tsx @@ -2,7 +2,6 @@ // Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0 import { Button } from "@/components/ui/button"; -import { isExternalModelId, useChatRuntimeStore } from "@/features/chat"; import { useLlamaUpdateCheck } from "@/hooks/use-llama-update-check"; import { useShowLlamaUpdateBanner } from "@/hooks/use-llama-update-pref"; import { toast } from "@/lib/toast"; @@ -100,15 +99,8 @@ export function LlamaUpdateBanner({ async function handleUpdate() { const result = await apply(); if (result?.ok) { - // Prefer the full release tag (e.g. b9726-mix-); the job's to_tag and - // installed_tag are the bare bNNNN build number. - const updatedTag = status?.latest_tag ?? result.tag ?? "the latest build"; - // Only a loaded local model can be reloaded to pick up the new binary; - // external-provider models do not use llama.cpp at all. - const checkpoint = useChatRuntimeStore.getState().params.checkpoint; - const hasLocalModel = - Boolean(checkpoint) && !isExternalModelId(checkpoint); - const reloadHint = hasLocalModel + const updatedTag = result.tag ?? status?.latest_tag ?? "the latest build"; + const reloadHint = result.reloadRequired ? " Reload your model to use it." : ""; toast.success(`llama.cpp updated to ${updatedTag}.${reloadHint}`); diff --git a/studio/frontend/src/features/chat/chat-settings-sheet.tsx b/studio/frontend/src/features/chat/chat-settings-sheet.tsx index 935f7e1bc6..8a7f09062a 100644 --- a/studio/frontend/src/features/chat/chat-settings-sheet.tsx +++ b/studio/frontend/src/features/chat/chat-settings-sheet.tsx @@ -518,8 +518,11 @@ export function ChatSettingsPanel({ const handleMtpUpdate = useCallback(async () => { const result = await applyLlamaUpdate(); if (result.ok) { + const reloadHint = result.reloadRequired + ? " Reload your model to enable MTP." + : ""; toast.success( - `llama.cpp updated to ${result.tag ?? "the latest build"}. Reload your model to enable MTP.`, + `llama.cpp updated to ${result.tag ?? "the latest build"}.${reloadHint}`, ); } else { toast.error(`llama.cpp update failed: ${result.error ?? "unknown error"}`); diff --git a/studio/frontend/src/hooks/use-llama-update-check.ts b/studio/frontend/src/hooks/use-llama-update-check.ts index ccfbb0d377..3b891cddd0 100644 --- a/studio/frontend/src/hooks/use-llama-update-check.ts +++ b/studio/frontend/src/hooks/use-llama-update-check.ts @@ -21,6 +21,7 @@ export interface LlamaUpdateJob { message: string; from_tag: string | null; to_tag: string | null; + reload_required: boolean | null; error: string | null; // Download fraction (0..1) while running, 1 on success, null when unknown. progress: number | null; @@ -52,6 +53,8 @@ function parseStatus(value: unknown): LlamaUpdateStatus | null { message: typeof job.message === "string" ? job.message : "", from_tag: typeof job.from_tag === "string" ? job.from_tag : null, to_tag: typeof job.to_tag === "string" ? job.to_tag : null, + reload_required: + typeof job.reload_required === "boolean" ? job.reload_required : null, error: typeof job.error === "string" ? job.error : null, progress: typeof job.progress === "number" ? job.progress : null, }, @@ -85,6 +88,7 @@ interface UseLlamaUpdateCheckOptions { export interface LlamaApplyResult { ok: boolean; tag?: string | null; + reloadRequired?: boolean | null; error?: string | null; } @@ -126,7 +130,11 @@ export function useLlamaUpdateCheck({ if (s.job.state === "success") { setVisible(false); void refreshHardwareInfo(); - onDone?.({ ok: true, tag: s.job.to_tag }); + onDone?.({ + ok: true, + tag: s.job.to_tag, + reloadRequired: s.job.reload_required, + }); } else if (s.job.state === "error") { // Leave the banner up so the user can retry; clearing applying drops // the "Updating..." state. From 54c7571299d2c9b42af5384126ae34e5f77a58a7 Mon Sep 17 00:00:00 2001 From: wasimysaid Date: Sun, 21 Jun 2026 14:44:30 +0200 Subject: [PATCH 3/3] Tighten comments for PR #6493 --- studio/backend/tests/test_llama_cpp_update.py | 15 +------ studio/backend/utils/llama_cpp_update.py | 19 +++----- .../src/components/llama-update-banner.tsx | 30 ++++--------- .../src/features/chat/chat-settings-sheet.tsx | 3 +- .../src/hooks/use-llama-update-check.ts | 45 +++++-------------- 5 files changed, 29 insertions(+), 83 deletions(-) diff --git a/studio/backend/tests/test_llama_cpp_update.py b/studio/backend/tests/test_llama_cpp_update.py index 1b509fea6c..4ceffbf75b 100644 --- a/studio/backend/tests/test_llama_cpp_update.py +++ b/studio/backend/tests/test_llama_cpp_update.py @@ -84,12 +84,7 @@ def _write_install( asset: str | None = None, release_tag: str | None = None, ) -> str: - """Create a fake prebuilt install tree and return the llama-server path. - - ``asset`` is the bundle filename recorded in the marker; omit it to model an - older marker that predates asset-based ROCm forwarding (backward compat). - ``release_tag`` is the full release tag (e.g. a ``b9596-mix-`` mix - build); defaults to ``tag`` for a plain prebuilt.""" + """Create a fake prebuilt install and return the llama-server path.""" bin_dir = dir_ / "build" / "bin" bin_dir.mkdir(parents = True, exist_ok = True) binary = bin_dir / "llama-server" @@ -436,7 +431,6 @@ def test_start_update_happy_path(monkeypatch, tmp_path): assert res["job"]["from_tag"] == "b9493" assert res["job"]["progress"] == 0.0 - # Wait for the background worker. deadline = time.time() + 10 while time.time() < deadline: job = upd.get_update_status()["job"] @@ -446,14 +440,11 @@ def test_start_update_happy_path(monkeypatch, tmp_path): assert job["state"] == "success", job assert job["to_tag"] == "b9518" assert job["reload_required"] is False - # Installer was invoked with the resolved install dir + latest + repo. assert "--install-dir" in captured["cmd"] assert str(install_dir) in captured["cmd"] assert "--llama-tag" in captured["cmd"] and "latest" in captured["cmd"] assert "unslothai/llama.cpp" in captured["cmd"] - # Progress lines were parsed and success pins progress at 1.0. assert job["progress"] == 1.0 - # The worker asks the installer for fine-grained progress milestones. assert popen_kwargs["env"]["UNSLOTH_PROGRESS_PERCENT_STEP"] == "5" @@ -653,7 +644,7 @@ def test_start_update_installer_missing_refuses(monkeypatch, tmp_path): class _FakeBackend: - """Minimal stand-in for LlamaCppBackend's update-coordination surface.""" + """Fake backend for update coordination.""" def __init__(self): import threading @@ -689,7 +680,6 @@ def test_update_sets_maintenance_flag_and_unloads(monkeypatch, tmp_path): seen = {} def _on_start(cmd): - # The maintenance flag must be set while the installer runs. seen["flag_during_install"] = backend._llama_update_in_progress _write_install(install_dir, "b9518") @@ -706,7 +696,6 @@ def test_update_sets_maintenance_flag_and_unloads(monkeypatch, tmp_path): assert backend.unloaded is True assert upd.get_update_status()["job"]["reload_required"] is True assert seen.get("flag_during_install") is True - # Cleared in the finally so model loads work again after the swap. assert backend._llama_update_in_progress is False diff --git a/studio/backend/utils/llama_cpp_update.py b/studio/backend/utils/llama_cpp_update.py index 4e15fbe074..8648b053d5 100644 --- a/studio/backend/utils/llama_cpp_update.py +++ b/studio/backend/utils/llama_cpp_update.py @@ -417,8 +417,7 @@ def _run_update(install_dir: Path, repo: str, asset: Optional[str], script: Path backend = None model_was_active = False try: - # Maintenance state so no load starts a server from the half-swapped binary - # (and the old binary is freed for the swap). Fails open without a backend. + # Block loads and free the binary while the installer swaps it. try: from routes.inference import get_llama_cpp_backend backend = get_llama_cpp_backend() @@ -432,8 +431,7 @@ def _run_update(install_dir: Path, repo: str, asset: Optional[str], script: Path try: with backend._serial_load_lock: backend._llama_update_in_progress = True - # is_active covers the loading/unhealthy window is_loaded misses - # (a live process also locks the exe on Windows during the swap). + # Active processes can lock the exe on Windows. if getattr(backend, "is_active", False): model_was_active = True backend.unload_model() @@ -452,8 +450,7 @@ def _run_update(install_dir: Path, repo: str, asset: Optional[str], script: Path ] cmd.extend(_rocm_install_args(asset)) logger.info("llama update: installing", cmd = " ".join(cmd)) - # Stream the installer output so download percent lines feed - # job["progress"]; finer milestones via UNSLOTH_PROGRESS_PERCENT_STEP. + # Stream progress lines into job["progress"]. env = dict(os.environ, UNSLOTH_PROGRESS_PERCENT_STEP = "5") proc = subprocess.Popen( cmd, @@ -494,12 +491,8 @@ def _run_update(install_dir: Path, repo: str, asset: Optional[str], script: Path tail = "".join(tail_lines).strip()[-1500:] raise RuntimeError(f"installer exited {returncode}: {tail or 'no output'}") - # New UNSLOTH_PREBUILT_INFO.json is on disk; drop the in-memory AND the - # on-disk freshness caches, then re-prime the 24h disk cache with the - # true newest, so the banner can't linger on a stale same-base value - # after the swap. drop_disk matters when the refresh below can't reach - # GitHub: without it, latest_published_release would replay the stale - # disk value; with it, latest reads as None and the banner fails open. + # Drop stale caches so the banner re-checks the swapped marker. + # If GitHub is offline, latest stays unknown and the banner fails open. reset_caches(drop_disk = True) try: latest_published_release(repo, force_refresh = True) @@ -532,7 +525,7 @@ def _run_update(install_dir: Path, repo: str, asset: Optional[str], script: Path finished_at = _utcnow(), ) finally: - # Lift the maintenance state so model loads work again, success or not. + # Always clear maintenance state. if backend is not None: try: backend._llama_update_in_progress = False diff --git a/studio/frontend/src/components/llama-update-banner.tsx b/studio/frontend/src/components/llama-update-banner.tsx index f53e6e0401..840383de90 100644 --- a/studio/frontend/src/components/llama-update-banner.tsx +++ b/studio/frontend/src/components/llama-update-banner.tsx @@ -8,12 +8,10 @@ import { toast } from "@/lib/toast"; import { cn } from "@/lib/utils"; import { Download } from "lucide-react"; import { type ReactElement, useEffect, useRef, useState } from "react"; -// Backend progress is coarse (5% steps, ~0.9 max) and the extract tail emits no -// signal. Creep toward this cap so the bar keeps moving rather than freezing. +// Creep toward this cap between coarse backend progress updates. const RUNNING_CAP = 0.95; -// Smoothed 0..1 bar progress: eases toward real `progress`, trickles toward a -// ceiling when idle, animates to 100% when `done`. Resets to 0 on each start. +// Smooth coarse backend progress without freezing between milestones. function useSmoothedProgress( active: boolean, progress: number | null, @@ -35,8 +33,7 @@ function useSmoothedProgress( let raf = 0; let last = performance.now(); const tick = (now: number) => { - // rAF timestamps can predate the performance.now() captured above, so - // clamp dt at 0 to keep the first frame from stepping backwards. + // Guard against a first rAF timestamp before the captured start time. const dt = Math.max(0, Math.min((now - last) / 1000, 0.1)); last = now; const current = displayRef.current; @@ -74,18 +71,11 @@ function useSmoothedProgress( interface LlamaUpdateBannerProps { enabled?: boolean; - // false: fill the parent instead of self-anchoring, so banners can stack in a - // shared container. true (default) keeps standalone desktop mounts working. + // false fills a shared stack; true self-anchors. positioned?: boolean; } -/** - * Non-invasive "Update llama.cpp" affordance. Appears bottom-right ~1s after a - * newer prebuilt is detected and stays up until the user explicitly acts on it - * (X, Update, or Remind me later). Clicking Update swaps the prebuilt in place - * via POST /api/llama/update. Can be turned off entirely in Settings -> - * General -> Notifications (on by default). - */ +/** Bottom-right llama.cpp update toast. */ export function LlamaUpdateBanner({ enabled = true, positioned = true, @@ -114,24 +104,20 @@ export function LlamaUpdateBanner({ const show = visible && status != null && (status.update_available || applying); const sizeBytes = status?.update_size_bytes ?? null; - // Round to whole MB; these prebuilts are hundreds of MB. const sizeLabel = sizeBytes && sizeBytes > 0 ? `${Math.round(sizeBytes / (1024 * 1024))} MB` : null; const updateProgress = status?.job.progress ?? null; const jobSucceeded = status?.job.state === "success"; - // Drives the bar so it animates continuously; aria reports the real value. + // Display value animates; aria uses the real progress. const displayProgress = useSmoothedProgress( applying, updateProgress, jobSucceeded, ); - // Render with no enter/exit animation. An opacity/transform transition (in or - // out) promotes a GPU compositing layer whose creation or teardown can flash - // for a frame on real displays, which reads as a flicker on appear and on - // dismiss. A plain conditional mount appears and leaves cleanly. + // Avoid opacity/transform transitions; GPU layer churn can flash. return show ? (