From 1c7bce427e5bcb8ccb3dece9e9fa3a572f6f0139 Mon Sep 17 00:00:00 2001
From: oobabooga <112222186+oobabooga@users.noreply.github.com>
Date: Fri, 17 Jul 2026 07:38:46 -0700
Subject: [PATCH] Revert "Feat/model picker per model config (#6647)"
This reverts commit 8cbdfbe355a83b6cc0706e2ed8ec1c737b71c3f3.
---
studio/backend/hub/schemas/inventory.py | 1 -
.../hub/services/models/cache_inventory.py | 74 +-
studio/backend/main.py | 3 -
studio/backend/picker/__init__.py | 2 -
studio/backend/picker/routes/__init__.py | 6 -
studio/backend/picker/routes/templates.py | 42 -
studio/backend/picker/schemas.py | 32 -
studio/backend/picker/service.py | 361 -------
.../tests/test_model_update_robustness.py | 46 -
studio/backend/tests/test_picker_service.py | 162 ---
studio/backend/utils/models/gguf_metadata.py | 79 --
studio/frontend/src/app/routes/__root.tsx | 7 +
.../assistant-ui}/model-selector.tsx | 130 +--
.../model-selector/folder-browser.tsx | 79 +-
.../model-selector/model-capabilities.ts | 0
.../model-selector/model-delete-action.tsx | 9 +-
.../model-load-settings-action.tsx | 19 +-
.../model-selector/model-update-action.tsx | 25 +-
.../model-selector/model-usage.ts | 3 +-
.../assistant-ui}/model-selector/pickers.tsx | 645 +++++-------
.../model-selector/pill-tabs.tsx | 3 +-
.../model-selector/recommended-fit.ts | 0
.../remembered-load-settings.ts | 69 ++
.../assistant-ui}/model-selector/row-meta.ts | 0
.../model-selector/source-tabs.ts | 0
.../assistant-ui}/model-selector/types.ts | 13 -
.../src/features/chat/api/chat-adapter.ts | 50 +-
.../frontend/src/features/chat/chat-page.tsx | 517 ++++------
.../src/features/chat/chat-settings-sheet.tsx | 957 +++++++++++++++---
.../chat/hooks/use-chat-model-runtime.ts | 105 +-
.../hooks/use-staged-model-preparation.ts | 155 +++
studio/frontend/src/features/chat/index.ts | 18 -
.../lib/apply-inference-status-to-store.ts | 5 +
.../src/features/chat/shared-composer.tsx | 111 +-
.../chat/stores/chat-runtime-store.ts | 186 +++-
.../export/components/export-run-panel.tsx | 71 +-
.../hub/catalog/models-catalog-rows.tsx | 41 +-
.../hub/catalog/on-device-folders-dialog.tsx | 33 +-
.../download-manager-controller.ts | 23 +
.../features/hub/download-manager/index.ts | 1 +
studio/frontend/src/features/hub/hub-page.tsx | 129 ++-
studio/frontend/src/features/hub/index.ts | 58 +-
.../src/features/hub/inventory/api.ts | 2 -
.../src/features/hub/inventory/types.ts | 3 -
.../src/features/hub/inventory/view-models.ts | 9 -
.../model-picker/api/model-metadata.ts | 20 -
.../features/model-picker/api/templates.ts | 51 -
.../chat-template-editor-dialog.tsx | 191 ----
.../components/model-config-page.tsx | 742 --------------
.../components/numeric-value-input.tsx | 113 ---
.../components/sidebar-model-config.tsx | 89 --
.../model-picker/hooks/use-model-defaults.ts | 176 ----
.../src/features/model-picker/index.ts | 30 -
.../inventory/use-chat-picker-inventory.ts | 118 ---
.../model-config/apply-per-model-config.ts | 85 --
.../model-config/model-identity.ts | 69 --
.../model-config/per-model-config.ts | 571 -----------
.../src/features/settings/tabs/chat-tab.tsx | 39 +
.../features/settings/tabs/general-tab.tsx | 3 +-
.../frontend/src/features/training/index.ts | 4 +-
tests/studio/playwright_chat_ui.py | 14 +-
.../test_studio_text_descender_clipping.py | 17 +-
62 files changed, 2133 insertions(+), 4483 deletions(-)
delete mode 100644 studio/backend/picker/__init__.py
delete mode 100644 studio/backend/picker/routes/__init__.py
delete mode 100644 studio/backend/picker/routes/templates.py
delete mode 100644 studio/backend/picker/schemas.py
delete mode 100644 studio/backend/picker/service.py
delete mode 100644 studio/backend/tests/test_picker_service.py
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector.tsx (86%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/folder-browser.tsx (88%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/model-capabilities.ts (100%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/model-delete-action.tsx (90%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/model-load-settings-action.tsx (66%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/model-update-action.tsx (82%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/model-usage.ts (93%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/pickers.tsx (89%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/pill-tabs.tsx (98%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/recommended-fit.ts (100%)
create mode 100644 studio/frontend/src/components/assistant-ui/model-selector/remembered-load-settings.ts
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/row-meta.ts (100%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/source-tabs.ts (100%)
rename studio/frontend/src/{features/model-picker/components => components/assistant-ui}/model-selector/types.ts (76%)
create mode 100644 studio/frontend/src/features/chat/hooks/use-staged-model-preparation.ts
delete mode 100644 studio/frontend/src/features/model-picker/api/model-metadata.ts
delete mode 100644 studio/frontend/src/features/model-picker/api/templates.ts
delete mode 100644 studio/frontend/src/features/model-picker/components/chat-template-editor-dialog.tsx
delete mode 100644 studio/frontend/src/features/model-picker/components/model-config-page.tsx
delete mode 100644 studio/frontend/src/features/model-picker/components/numeric-value-input.tsx
delete mode 100644 studio/frontend/src/features/model-picker/components/sidebar-model-config.tsx
delete mode 100644 studio/frontend/src/features/model-picker/hooks/use-model-defaults.ts
delete mode 100644 studio/frontend/src/features/model-picker/index.ts
delete mode 100644 studio/frontend/src/features/model-picker/inventory/use-chat-picker-inventory.ts
delete mode 100644 studio/frontend/src/features/model-picker/model-config/apply-per-model-config.ts
delete mode 100644 studio/frontend/src/features/model-picker/model-config/model-identity.ts
delete mode 100644 studio/frontend/src/features/model-picker/model-config/per-model-config.ts
diff --git a/studio/backend/hub/schemas/inventory.py b/studio/backend/hub/schemas/inventory.py
index f81c9a3498..ef95efe2f2 100644
--- a/studio/backend/hub/schemas/inventory.py
+++ b/studio/backend/hub/schemas/inventory.py
@@ -160,7 +160,6 @@ class CachedRepoBase(BaseModel):
repo_id: str
size_bytes: int = 0
cache_path: Optional[str] = None
- last_modified: Optional[float] = None
partial: bool = False
partial_transport: Optional[str] = None
inventory_id: Optional[str] = None
diff --git a/studio/backend/hub/services/models/cache_inventory.py b/studio/backend/hub/services/models/cache_inventory.py
index ba3a266bfb..1f38af9381 100644
--- a/studio/backend/hub/services/models/cache_inventory.py
+++ b/studio/backend/hub/services/models/cache_inventory.py
@@ -31,7 +31,6 @@ from hub.services.models.common import (
_is_checkpoint_weight_name,
_is_gguf_filename,
_is_main_gguf_filename,
- _is_mmproj_filename,
_is_transformers_safetensors_weight_name,
_local_inventory_id,
_prefer_complete_larger,
@@ -126,34 +125,6 @@ def _repo_has_gguf_files(repo_info) -> bool:
return _repo_gguf_size_bytes(repo_info) > 0
-def _blob_mtime(file_obj) -> float:
- ts = getattr(file_obj, "blob_last_modified", None)
- if isinstance(ts, (int, float)) and ts > 0:
- return float(ts)
- blob_path = getattr(file_obj, "blob_path", None)
- if blob_path:
- try:
- return float(Path(blob_path).stat().st_mtime)
- except OSError:
- pass
- return 0.0
-
-
-def _repo_gguf_last_modified(repo_info) -> float:
- latest = 0.0
- for revision in repo_info.revisions:
- for f in revision.files:
- if _is_main_gguf_filename(f.file_name):
- latest = max(latest, _blob_mtime(f))
- return latest
-
-
-def _repo_has_mmproj(repo_info) -> bool:
- return any(
- _is_mmproj_filename(f.file_name) for revision in repo_info.revisions for f in revision.files
- )
-
-
def _cached_repo_file_name(file_obj) -> str:
file_path = getattr(file_obj, "file_path", None)
if file_path:
@@ -295,7 +266,6 @@ def _scan_cached_gguf() -> list[dict]:
continue
key = repo_id.lower()
existing = seen_lower.get(key)
- last_modified = _repo_gguf_last_modified(repo_info)
row = {
"repo_id": repo_id,
"size_bytes": max(total_size, variant_state_size),
@@ -305,9 +275,6 @@ def _scan_cached_gguf() -> list[dict]:
# per-variant detail lives on GgufVariantDetail.
"partial_transport": None,
}
- last_modified = max(last_modified, (existing or {}).get("last_modified", 0.0))
- if last_modified > 0:
- row["last_modified"] = last_modified
row.update(
_cache_inventory_fields(
repo_id,
@@ -316,12 +283,8 @@ def _scan_cached_gguf() -> list[dict]:
requires_variant = True,
)
)
- if _repo_has_mmproj(repo_info):
- row["capabilities"]["supports_vision"] = True
if _prefer_cache_row(row, existing):
seen_lower[key] = row
- elif last_modified > existing.get("last_modified", 0.0):
- existing["last_modified"] = last_modified
except Exception as e:
repo_label = getattr(repo_info, "repo_id", "")
logger.warning(f"Skipping cached GGUF repo {repo_label}: {e}")
@@ -349,14 +312,13 @@ class _CachedNonGgufPayload(NamedTuple):
size_bytes: int
has_runnable_weights: bool
model_format: ModelFormat
- last_modified: float
def _repo_non_gguf_model_payload(repo_info) -> _CachedNonGgufPayload:
- all_weight_blobs: dict[str, tuple[int, float]] = {}
- adapter_blobs: dict[str, tuple[int, float]] = {}
- safetensors_blobs: dict[str, tuple[int, float]] = {}
- checkpoint_blobs: dict[str, tuple[int, float]] = {}
+ all_weight_blobs: dict[str, int] = {}
+ adapter_blobs: dict[str, int] = {}
+ safetensors_blobs: dict[str, int] = {}
+ checkpoint_blobs: dict[str, int] = {}
has_config = False
has_adapter_config = False
has_adapter_weights = False
@@ -364,15 +326,12 @@ def _repo_non_gguf_model_payload(repo_info) -> _CachedNonGgufPayload:
has_transformers_safetensors = False
has_checkpoint = False
- def _record_blob(
- target: dict[str, tuple[int, float]], file_obj, rev_id: str, file_name: str
- ) -> None:
+ def _record_blob(target: dict[str, int], file_obj, rev_id: str, file_name: str) -> None:
blob_path = getattr(file_obj, "blob_path", None)
size = int(file_obj.size_on_disk or 0)
key = str(blob_path) if blob_path else f"{rev_id}:{file_name}"
- value = (size, _blob_mtime(file_obj))
- target[key] = value
- all_weight_blobs[key] = value
+ target[key] = size
+ all_weight_blobs[key] = size
for revision in repo_info.revisions:
rev_id = getattr(revision, "commit_hash", None) or str(id(revision))
@@ -416,19 +375,18 @@ def _repo_non_gguf_model_payload(repo_info) -> _CachedNonGgufPayload:
or "unknown"
)
if model_format == "adapter":
- selected_blobs = adapter_blobs
+ size_bytes = sum(adapter_blobs.values())
elif model_format == "safetensors":
- selected_blobs = safetensors_blobs
+ size_bytes = sum(safetensors_blobs.values())
elif model_format == "checkpoint":
- selected_blobs = checkpoint_blobs
+ size_bytes = sum(checkpoint_blobs.values())
else:
- selected_blobs = all_weight_blobs
+ size_bytes = sum(all_weight_blobs.values())
return _CachedNonGgufPayload(
- size_bytes = sum(size for size, _mtime in selected_blobs.values()),
+ size_bytes = size_bytes,
has_runnable_weights = model_format != "unknown",
model_format = model_format,
- last_modified = max((mtime for _size, mtime in selected_blobs.values()), default = 0.0),
)
@@ -550,12 +508,6 @@ def _scan_cached_models() -> list[dict]:
),
**_cached_model_local_metadata(repo_path),
}
- last_modified = max(
- payload.last_modified,
- (existing or {}).get("last_modified", 0.0),
- )
- if last_modified > 0:
- row["last_modified"] = last_modified
row.update(
_cache_inventory_fields(
repo_id,
@@ -565,8 +517,6 @@ def _scan_cached_models() -> list[dict]:
)
if _prefer_cache_row(row, existing):
seen_lower[key] = row
- elif last_modified > existing.get("last_modified", 0.0):
- existing["last_modified"] = last_modified
except Exception as e:
repo_label = getattr(repo_info, "repo_id", "")
logger.warning(f"Skipping cached model repo {repo_label}: {e}")
diff --git a/studio/backend/main.py b/studio/backend/main.py
index bd0d26cf8f..e64048dc00 100644
--- a/studio/backend/main.py
+++ b/studio/backend/main.py
@@ -313,7 +313,6 @@ from hub.routes import (
inventory_router as hub_inventory_router,
datasets_router as hub_datasets_router,
)
-from picker.routes import templates_router as picker_templates_router
from hub.schemas.downloads import TransportCapabilities
from hub.utils.download_registry import (
get_download_transport_capabilities,
@@ -746,7 +745,6 @@ _BODY_PROTECTED_PREFIXES = (
"/v1/completions",
"/p/",
"/api/inference",
- "/api/picker",
"/api/data-recipe",
"/api/datasets",
"/api/hub",
@@ -977,7 +975,6 @@ app.include_router(rag_router, prefix = "/api/rag", tags = ["rag"])
app.include_router(training_history_router, prefix = "/api/train", tags = ["training-history"])
app.include_router(hub_inventory_router, prefix = "/api/hub", tags = ["hub"])
app.include_router(hub_datasets_router, prefix = "/api/hub/datasets", tags = ["hub"])
-app.include_router(picker_templates_router, prefix = "/api/picker", tags = ["picker"])
# Re-wrap client-error responses on the /v1/* surface into OpenAI/Anthropic
# error envelopes; non-/v1 paths keep FastAPI's default {"detail": ...} shape.
diff --git a/studio/backend/picker/__init__.py b/studio/backend/picker/__init__.py
deleted file mode 100644
index 32014236c6..0000000000
--- a/studio/backend/picker/__init__.py
+++ /dev/null
@@ -1,2 +0,0 @@
-# SPDX-License-Identifier: AGPL-3.0-only
-# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
diff --git a/studio/backend/picker/routes/__init__.py b/studio/backend/picker/routes/__init__.py
deleted file mode 100644
index c0e988c8bb..0000000000
--- a/studio/backend/picker/routes/__init__.py
+++ /dev/null
@@ -1,6 +0,0 @@
-# SPDX-License-Identifier: AGPL-3.0-only
-# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
-
-from .templates import router as templates_router
-
-__all__ = ["templates_router"]
diff --git a/studio/backend/picker/routes/templates.py b/studio/backend/picker/routes/templates.py
deleted file mode 100644
index 03707669fa..0000000000
--- a/studio/backend/picker/routes/templates.py
+++ /dev/null
@@ -1,42 +0,0 @@
-# SPDX-License-Identifier: AGPL-3.0-only
-# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
-
-from __future__ import annotations
-
-import asyncio
-from typing import Optional
-
-from fastapi import APIRouter, Body, Depends, Query
-
-from auth.authentication import get_current_subject
-from hub.dependencies import get_hf_token
-
-from ..schemas import (
- ModelTemplateResponse,
- ValidateChatTemplateRequest,
- ValidateChatTemplateResponse,
-)
-from ..service import read_default_chat_template, validate_chat_template
-
-router = APIRouter()
-
-
-@router.post("/validate-chat-template", response_model = ValidateChatTemplateResponse)
-async def validate_chat_template_route(
- body: ValidateChatTemplateRequest = Body(...),
- current_subject: str = Depends(get_current_subject),
-) -> ValidateChatTemplateResponse:
- return await asyncio.to_thread(validate_chat_template, body.template)
-
-
-@router.get("/chat-template/{model_name:path}", response_model = ModelTemplateResponse)
-async def get_default_chat_template_route(
- model_name: str,
- gguf_variant: Optional[str] = Query(None),
- hf_token: Optional[str] = Depends(get_hf_token),
- current_subject: str = Depends(get_current_subject),
-) -> ModelTemplateResponse:
- template = await asyncio.to_thread(
- read_default_chat_template, model_name, hf_token, gguf_variant
- )
- return ModelTemplateResponse(model_name = model_name, chat_template = template)
diff --git a/studio/backend/picker/schemas.py b/studio/backend/picker/schemas.py
deleted file mode 100644
index b4f956188f..0000000000
--- a/studio/backend/picker/schemas.py
+++ /dev/null
@@ -1,32 +0,0 @@
-# SPDX-License-Identifier: AGPL-3.0-only
-# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
-
-from typing import Optional
-
-from pydantic import BaseModel, Field, field_validator
-
-# Mirror the frontend's 64 KiB chat-template contract (per-model-config.ts) at
-# the API boundary so a direct caller cannot make Jinja parse an oversized
-# template. MaxBodyMiddleware only caps the whole request body, not this field.
-MAX_CHAT_TEMPLATE_BYTES = 65_536
-
-
-class ValidateChatTemplateRequest(BaseModel):
- template: str = Field(default = "")
-
- @field_validator("template")
- @classmethod
- def _enforce_template_size(cls, value: str) -> str:
- if len(value.encode("utf-8")) > MAX_CHAT_TEMPLATE_BYTES:
- raise ValueError(f"Chat template exceeds the {MAX_CHAT_TEMPLATE_BYTES}-byte limit.")
- return value
-
-
-class ValidateChatTemplateResponse(BaseModel):
- valid: bool
- error: Optional[str] = None
-
-
-class ModelTemplateResponse(BaseModel):
- model_name: str
- chat_template: Optional[str] = None
diff --git a/studio/backend/picker/service.py b/studio/backend/picker/service.py
deleted file mode 100644
index f5994dc550..0000000000
--- a/studio/backend/picker/service.py
+++ /dev/null
@@ -1,361 +0,0 @@
-# SPDX-License-Identifier: AGPL-3.0-only
-# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
-
-from __future__ import annotations
-
-import json
-import logging
-import os
-import re
-from pathlib import Path
-from typing import Optional
-
-from hub.services.models.folder_browser import (
- _build_browse_allowlist,
- _is_path_inside_allowlist,
-)
-from hub.utils.gguf import iter_hf_cache_snapshots
-from utils.models.gguf_metadata import read_gguf_chat_template
-from utils.models.model_config import (
- _extract_quant_label,
- _is_big_endian_gguf_path,
- _is_mmproj,
- _is_mtp_drafter,
-)
-from utils.paths.path_utils import (
- is_local_path,
- normalize_path,
- resolve_cached_repo_id_case,
-)
-
-from .schemas import ValidateChatTemplateResponse
-
-logger = logging.getLogger(__name__)
-
-_VALID_REPO_ID = re.compile(r"^[A-Za-z0-9._-]+/[A-Za-z0-9._-]+$")
-
-
-def _is_valid_repo_id(repo_id: str) -> bool:
- return bool(_VALID_REPO_ID.fullmatch(repo_id))
-
-
-_TOKENIZER_CONFIG_PATHS = ("tokenizer_config.json", "LLM/tokenizer_config.json")
-_JINJA_TEMPLATE_PATHS = ("chat_template.jinja", "LLM/chat_template.jinja")
-_PROCESSOR_TEMPLATE_PATHS = ("chat_template.json", "LLM/chat_template.json")
-
-
-def _leaf_inside_allowlist(path: Path, allow_roots: Optional[list[Path]]) -> bool:
- # Block symlinked children from escaping the validated directory
- # (realpath-checked). None = trusted caller (HF cache / remote download).
- return allow_roots is None or _is_path_inside_allowlist(path, allow_roots)
-
-
-def validate_chat_template(template: str) -> ValidateChatTemplateResponse:
- text = (template or "").strip()
- if not text:
- return ValidateChatTemplateResponse(valid = True, error = None)
- # Import Jinja lazily: it is optional at runtime (e.g. GGUF-only installs),
- # so a missing dependency must not crash API startup through this module.
- try:
- from jinja2 import TemplateError
- from jinja2.ext import Extension
- from jinja2.sandbox import ImmutableSandboxedEnvironment
- except ImportError:
- return ValidateChatTemplateResponse(valid = True, error = None)
-
- class _GenerationTag(Extension):
- # Accept Transformers' {% generation %}...{% endgeneration %} assistant
- # mask tag so a pasted HF chat template validates (we only parse it).
- tags = {"generation"}
-
- def parse(self, parser):
- next(parser.stream)
- return parser.parse_statements(["name:endgeneration"], drop_needle = True)
-
- try:
- env = ImmutableSandboxedEnvironment(
- trim_blocks = True,
- lstrip_blocks = True,
- extensions = ["jinja2.ext.loopcontrols", _GenerationTag],
- )
- env.parse(text)
- return ValidateChatTemplateResponse(valid = True, error = None)
- except TemplateError as exc:
- message = getattr(exc, "message", None) or str(exc)
- lineno = getattr(exc, "lineno", None)
- if lineno:
- message = f"Line {lineno}: {message}"
- return ValidateChatTemplateResponse(valid = False, error = message)
- except Exception as exc:
- return ValidateChatTemplateResponse(valid = False, error = str(exc))
-
-
-def _chat_template_from_tokenizer_config(config: dict) -> Optional[str]:
- if not isinstance(config, dict):
- return None
- raw = config.get("chat_template")
- if isinstance(raw, str) and raw.strip():
- return raw
- if isinstance(raw, list):
- fallback: Optional[str] = None
- for entry in raw:
- if not isinstance(entry, dict):
- continue
- template = entry.get("template")
- if not isinstance(template, str):
- continue
- if entry.get("name") == "default":
- return template
- if fallback is None:
- fallback = template
- return fallback
- return None
-
-
-def _chat_template_from_jinja_file(
- dir_path: Path, allow_roots: Optional[list[Path]] = None
-) -> Optional[str]:
- for rel in _JINJA_TEMPLATE_PATHS:
- template_file = dir_path / rel
- if not template_file.exists() or not _leaf_inside_allowlist(template_file, allow_roots):
- continue
- try:
- template = template_file.read_text(encoding = "utf-8")
- except Exception:
- continue
- if template.strip():
- return template
- return None
-
-
-def _chat_template_from_processor_payload(payload: object) -> Optional[str]:
- # processor chat_template.json may be the template string itself or a
- # {name: template} map, not only a tokenizer_config-shaped object.
- if isinstance(payload, str):
- return payload if payload.strip() else None
- template = _chat_template_from_tokenizer_config(payload) # type: ignore[arg-type]
- if template:
- return template
- if isinstance(payload, dict):
- # Named-template map: prefer "default", else the first non-empty entry
- # (mirrors the tokenizer-config list fallback).
- default = payload.get("default")
- if isinstance(default, str) and default.strip():
- return default
- for value in payload.values():
- if isinstance(value, str) and value.strip():
- return value
- return None
-
-
-def _chat_template_from_processor_json(
- dir_path: Path, allow_roots: Optional[list[Path]] = None
-) -> Optional[str]:
- for rel in _PROCESSOR_TEMPLATE_PATHS:
- config_file = dir_path / rel
- if not config_file.exists() or not _leaf_inside_allowlist(config_file, allow_roots):
- continue
- try:
- payload = json.loads(config_file.read_text(encoding = "utf-8"))
- except Exception:
- continue
- template = _chat_template_from_processor_payload(payload)
- if template:
- return template
- return None
-
-
-def _chat_template_from_tokenizer_dir(
- dir_path: Path, allow_roots: Optional[list[Path]] = None
-) -> Optional[str]:
- jinja = _chat_template_from_jinja_file(dir_path, allow_roots)
- if jinja:
- return jinja
- for rel in _TOKENIZER_CONFIG_PATHS:
- config_file = dir_path / rel
- if not config_file.exists() or not _leaf_inside_allowlist(config_file, allow_roots):
- continue
- try:
- config = json.loads(config_file.read_text(encoding = "utf-8"))
- except Exception:
- continue
- template = _chat_template_from_tokenizer_config(config)
- if template:
- return template
- return _chat_template_from_processor_json(dir_path, allow_roots)
-
-
-_GGUF_SCAN_MAX_DEPTH = 2
-
-
-def _iter_ggufs(dir_path: Path) -> list[Path]:
- if dir_path == dir_path.parent:
- return []
- root = str(dir_path)
- found: list[Path] = []
- for current, dirs, files in os.walk(root, followlinks = False):
- rel = os.path.relpath(current, root)
- depth = 0 if rel == os.curdir else rel.count(os.sep) + 1
- if depth >= _GGUF_SCAN_MAX_DEPTH:
- dirs[:] = []
- for name in files:
- if not name.lower().endswith(".gguf") or _is_mmproj(name):
- continue
- path = Path(current) / name
- try:
- rel = path.relative_to(dir_path).as_posix()
- except ValueError:
- rel = name
- quant = _extract_quant_label(rel)
- if _is_mtp_drafter(rel) or _is_big_endian_gguf_path(rel, quant):
- continue
- found.append(path)
- return found
-
-
-def _variant_matches(relative_path: str, needle: str) -> bool:
- quant = _extract_quant_label(relative_path).lower()
- if quant == needle:
- return True
- prefix = f"{needle}-"
- if not quant.startswith(prefix):
- return False
- suffix = quant[len(prefix) :]
- if not suffix.endswith("bpw"):
- return False
- value = suffix[:-3]
- return bool(value) and value.replace(".", "", 1).isdigit()
-
-
-def _find_gguf_in_dir(dir_path: Path, gguf_variant: Optional[str]) -> Optional[Path]:
- try:
- ggufs = sorted(_iter_ggufs(dir_path))
- except OSError:
- return None
- if not ggufs:
- return None
- needle = (gguf_variant or "").strip().lower()
- if needle:
- for path in ggufs:
- try:
- relative = path.relative_to(dir_path).as_posix()
- except ValueError:
- relative = path.name
- if _variant_matches(relative, needle):
- return path
- return None
- try:
- return max(ggufs, key = lambda path: path.stat().st_size)
- except OSError:
- return ggufs[0]
-
-
-def _chat_template_from_dir(
- dir_path: Path,
- gguf_variant: Optional[str] = None,
- allow_roots: Optional[list[Path]] = None,
-) -> Optional[str]:
- def from_gguf() -> Optional[str]:
- gguf = _find_gguf_in_dir(dir_path, gguf_variant)
- if gguf is None or not _leaf_inside_allowlist(gguf, allow_roots):
- return None
- return read_gguf_chat_template(str(gguf))
-
- # Sidecar tokenizer files (chat_template.jinja / tokenizer_config.json) are
- # the model author's maintained template and supersede the GGUF's embedded
- # copy, which can be stale. The variant only selects which GGUF to fall back
- # to, so keep tokenizer-first precedence whether or not a variant is given.
- return _chat_template_from_tokenizer_dir(dir_path, allow_roots) or from_gguf()
-
-
-def read_default_chat_template(
- model_name: str,
- hf_token: Optional[str] = None,
- gguf_variant: Optional[str] = None,
-) -> Optional[str]:
- if not isinstance(model_name, str) or not model_name.strip():
- return None
- name = model_name.strip()
-
- if is_local_path(name):
- try:
- target = Path(normalize_path(name)).expanduser()
- allow_roots = _build_browse_allowlist()
- if not _is_path_inside_allowlist(target, allow_roots):
- logger.debug("Refused chat template read outside allowed folders: %s", name)
- return None
- if name.lower().endswith(".gguf"):
- # Prefer a maintained sidecar template (chat_template.jinja /
- # tokenizer_config.json) next to the file over the GGUF's embedded
- # copy, matching the tokenizer-first precedence used for directory
- # and variant selections.
- sidecar = _chat_template_from_tokenizer_dir(target.parent, allow_roots)
- if sidecar:
- return sidecar
- return read_gguf_chat_template(str(target))
- return _chat_template_from_dir(target, gguf_variant, allow_roots)
- except Exception as exc:
- logger.debug("Could not read local chat template for %s: %s", name, exc)
- return None
-
- if not _is_valid_repo_id(name):
- return None
-
- resolved = resolve_cached_repo_id_case(name)
-
- try:
- # Resolve within each cached revision, newest first. A revision's
- # maintained sidecar (chat_template.jinja / tokenizer_config.json)
- # supersedes its own embedded GGUF copy, but a newer revision must not be
- # overridden by an older revision's sidecar, so precedence stays
- # per-snapshot rather than searching all sidecars globally first.
- for snapshot in iter_hf_cache_snapshots(resolved):
- template = _chat_template_from_dir(snapshot, gguf_variant)
- if template:
- return template
- except Exception as exc:
- logger.debug("Could not read cached chat template for %s: %s", resolved, exc)
-
- try:
- from huggingface_hub import hf_hub_download
-
- def _download_text(rel: str) -> Optional[str]:
- try:
- path = hf_hub_download(resolved, rel, token = hf_token)
- return Path(path).read_text(encoding = "utf-8")
- except Exception:
- return None
-
- for rel in _JINJA_TEMPLATE_PATHS:
- template = _download_text(rel)
- if template and template.strip():
- return template
-
- for rel in _TOKENIZER_CONFIG_PATHS:
- raw = _download_text(rel)
- if not raw:
- continue
- try:
- config = json.loads(raw)
- except Exception:
- continue
- template = _chat_template_from_tokenizer_config(config)
- if template:
- return template
-
- for rel in _PROCESSOR_TEMPLATE_PATHS:
- raw = _download_text(rel)
- if not raw:
- continue
- try:
- payload = json.loads(raw)
- except Exception:
- continue
- template = _chat_template_from_processor_payload(payload)
- if template:
- return template
-
- return None
- except Exception as exc:
- logger.debug("Could not fetch chat template for %s: %s", resolved, exc)
- return None
diff --git a/studio/backend/tests/test_model_update_robustness.py b/studio/backend/tests/test_model_update_robustness.py
index 4c5822e662..edf55812e2 100644
--- a/studio/backend/tests/test_model_update_robustness.py
+++ b/studio/backend/tests/test_model_update_robustness.py
@@ -314,7 +314,6 @@ def test_cached_model_scan_keeps_local_safetensors_repo(monkeypatch, tmp_path):
file_name = "model.safetensors",
size_on_disk = 100,
blob_path = str(repo_path / "blobs" / "modelsha"),
- blob_last_modified = 3_000.0,
),
]
)
@@ -337,51 +336,6 @@ def test_cached_model_scan_keeps_local_safetensors_repo(monkeypatch, tmp_path):
assert rows[0]["repo_id"] == "Org/SafeTensorRepo"
assert rows[0]["model_format"] == "safetensors"
assert rows[0]["size_bytes"] == 100
- assert rows[0]["last_modified"] == 3_000.0
-
-
-def test_cached_gguf_scan_keeps_download_timestamp(monkeypatch, tmp_path):
- repo_path = tmp_path / "models--Org--GgufRepo"
- repo = SimpleNamespace(
- repo_id = "Org/GgufRepo",
- repo_type = "model",
- repo_path = repo_path,
- revisions = [
- SimpleNamespace(
- files = [
- SimpleNamespace(
- file_name = "model-Q4_K_M.gguf",
- size_on_disk = 100,
- blob_path = None,
- blob_last_modified = 5_000.0,
- ),
- ]
- )
- ],
- )
- monkeypatch.setattr(
- CI,
- "all_hf_cache_scans",
- lambda: [SimpleNamespace(repos = [repo])],
- )
- monkeypatch.setattr(
- CI.hf_cache_scan,
- "is_gguf_repo_partial",
- lambda *args, **kwargs: False,
- )
- monkeypatch.setattr(
- CI,
- "_gguf_variant_state_summary",
- lambda _repo_id: (False, 0),
- )
-
- rows = CI._scan_cached_gguf()
-
- assert len(rows) == 1
- assert rows[0]["repo_id"] == "Org/GgufRepo"
- assert rows[0]["model_format"] == "gguf"
- assert rows[0]["size_bytes"] == 100
- assert rows[0]["last_modified"] == 5_000.0
# ── hf_hub_download_with_xet_fallback force_download bypass (X2/F2) ───
diff --git a/studio/backend/tests/test_picker_service.py b/studio/backend/tests/test_picker_service.py
deleted file mode 100644
index fc835bf019..0000000000
--- a/studio/backend/tests/test_picker_service.py
+++ /dev/null
@@ -1,162 +0,0 @@
-# SPDX-License-Identifier: AGPL-3.0-only
-# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
-
-import json
-
-from picker.service import (
- _chat_template_from_dir,
- _chat_template_from_tokenizer_config,
- _chat_template_from_tokenizer_dir,
- _find_gguf_in_dir,
- _iter_ggufs,
- read_default_chat_template,
- validate_chat_template,
-)
-
-
-def test_iter_ggufs_skips_gguf_companions(tmp_path):
- mtp_dir = tmp_path / "MTP"
- mtp_dir.mkdir()
- main = tmp_path / "model-Q8_0.gguf"
- main.write_bytes(b"")
- (tmp_path / "mmproj-F16.gguf").write_bytes(b"")
- (tmp_path / "mtp-model-Q8_0.gguf").write_bytes(b"")
- (mtp_dir / "model-Q8_0-MTP.gguf").write_bytes(b"")
- (tmp_path / "model-Q8_0-be.gguf").write_bytes(b"")
-
- assert _iter_ggufs(tmp_path) == [main]
-
-
-def test_find_gguf_in_dir_matches_quant_label(tmp_path):
- mtp_dir = tmp_path / "MTP"
- mtp_dir.mkdir()
- main = tmp_path / "model-Q8_0.gguf"
- main.write_bytes(b"")
- (mtp_dir / "model-Q8_0-MTP.gguf").write_bytes(b"")
- (tmp_path / "model-Q4_K_M.gguf").write_bytes(b"")
-
- assert _find_gguf_in_dir(tmp_path, "Q8_0") == main
- assert _find_gguf_in_dir(tmp_path, "Q4_K") is None
-
-
-def test_find_gguf_in_dir_without_variant_prefers_largest_model(tmp_path):
- smaller = tmp_path / "a-model-Q4_K_M.gguf"
- larger = tmp_path / "z-model-Q8_0.gguf"
- smaller.write_bytes(b"0")
- larger.write_bytes(b"00")
-
- assert _find_gguf_in_dir(tmp_path, None) == larger
-
-
-def test_find_gguf_in_dir_matches_bpw_variant_base_label(tmp_path):
- target = tmp_path / "model-IQ4_XS-3.53bpw.gguf"
- target.write_bytes(b"")
- (tmp_path / "model-Q4_K_M.gguf").write_bytes(b"")
-
- assert _find_gguf_in_dir(tmp_path, "IQ4_XS") == target
- assert _find_gguf_in_dir(tmp_path, "IQ4_XS-3.53bpw") == target
- assert _find_gguf_in_dir(tmp_path, "Q4_K") is None
-
-
-def test_validate_chat_template_accepts_valid_and_empty():
- assert validate_chat_template("{{ messages[0].content }}").valid is True
- assert validate_chat_template("").valid is True
- assert validate_chat_template(" ").valid is True
-
-
-def test_validate_chat_template_reports_syntax_error_with_line():
- result = validate_chat_template("{% if %}{% endif %}")
- assert result.valid is False
- assert result.error is not None
- assert result.error.startswith("Line ")
-
-
-def test_chat_template_from_tokenizer_config_reads_string():
- assert _chat_template_from_tokenizer_config({"chat_template": "HELLO"}) == "HELLO"
- assert _chat_template_from_tokenizer_config({"chat_template": " "}) is None
- assert _chat_template_from_tokenizer_config({}) is None
-
-
-def test_chat_template_from_tokenizer_config_prefers_named_default():
- config = {
- "chat_template": [
- {"name": "tool_use", "template": "TOOL"},
- {"name": "default", "template": "DEFAULT"},
- ]
- }
- assert _chat_template_from_tokenizer_config(config) == "DEFAULT"
-
-
-def test_chat_template_from_tokenizer_config_falls_back_to_first_entry():
- config = {
- "chat_template": [
- {"name": "tool_use", "template": "TOOL"},
- {"name": "other", "template": "OTHER"},
- ]
- }
- assert _chat_template_from_tokenizer_config(config) == "TOOL"
-
-
-def test_chat_template_from_tokenizer_dir_prefers_jinja_file(tmp_path):
- (tmp_path / "chat_template.jinja").write_text("FROM_JINJA", encoding = "utf-8")
- (tmp_path / "tokenizer_config.json").write_text(
- json.dumps({"chat_template": "FROM_CONFIG"}), encoding = "utf-8"
- )
- assert _chat_template_from_tokenizer_dir(tmp_path) == "FROM_JINJA"
-
-
-def test_chat_template_from_tokenizer_dir_reads_tokenizer_config(tmp_path):
- (tmp_path / "tokenizer_config.json").write_text(
- json.dumps({"chat_template": "FROM_CONFIG"}), encoding = "utf-8"
- )
- assert _chat_template_from_tokenizer_dir(tmp_path) == "FROM_CONFIG"
-
-
-def test_chat_template_from_dir_without_variant_prefers_tokenizer(tmp_path):
- (tmp_path / "tokenizer_config.json").write_text(
- json.dumps({"chat_template": "FROM_CONFIG"}), encoding = "utf-8"
- )
- assert _chat_template_from_dir(tmp_path) == "FROM_CONFIG"
-
-
-def test_chat_template_from_dir_with_variant_still_prefers_tokenizer(tmp_path, monkeypatch):
- (tmp_path / "tokenizer_config.json").write_text(
- json.dumps({"chat_template": "FROM_CONFIG"}), encoding = "utf-8"
- )
- (tmp_path / "model-Q4_K_M.gguf").write_bytes(b"")
- monkeypatch.setattr("picker.service.read_gguf_chat_template", lambda _path: "FROM_GGUF")
- # Selecting a variant must not flip precedence to the embedded GGUF template.
- assert _chat_template_from_dir(tmp_path, "Q4_K_M") == "FROM_CONFIG"
-
-
-def test_chat_template_from_dir_with_variant_falls_back_to_gguf(tmp_path, monkeypatch):
- (tmp_path / "model-Q4_K_M.gguf").write_bytes(b"")
- monkeypatch.setattr("picker.service.read_gguf_chat_template", lambda _path: "FROM_GGUF")
- # With no tokenizer sidecar, the embedded GGUF template is still the fallback.
- assert _chat_template_from_dir(tmp_path, "Q4_K_M") == "FROM_GGUF"
-
-
-def test_chat_template_from_dir_returns_none_when_absent(tmp_path):
- assert _chat_template_from_dir(tmp_path) is None
-
-
-def test_read_default_chat_template_direct_gguf_prefers_sidecar(tmp_path, monkeypatch):
- gguf = tmp_path / "model-Q4_K_M.gguf"
- gguf.write_bytes(b"")
- (tmp_path / "tokenizer_config.json").write_text(
- json.dumps({"chat_template": "FROM_CONFIG"}), encoding = "utf-8"
- )
- monkeypatch.setattr("picker.service._build_browse_allowlist", lambda: [tmp_path])
- monkeypatch.setattr("picker.service.read_gguf_chat_template", lambda _path: "FROM_GGUF")
- # A directly selected .gguf file must prefer a maintained sidecar template
- # over its embedded copy, matching directory/variant precedence.
- assert read_default_chat_template(str(gguf)) == "FROM_CONFIG"
-
-
-def test_read_default_chat_template_direct_gguf_falls_back_to_embedded(tmp_path, monkeypatch):
- gguf = tmp_path / "model-Q4_K_M.gguf"
- gguf.write_bytes(b"")
- monkeypatch.setattr("picker.service._build_browse_allowlist", lambda: [tmp_path])
- monkeypatch.setattr("picker.service.read_gguf_chat_template", lambda _path: "FROM_GGUF")
- # With no sidecar next to the file, the embedded GGUF template is the fallback.
- assert read_default_chat_template(str(gguf)) == "FROM_GGUF"
diff --git a/studio/backend/utils/models/gguf_metadata.py b/studio/backend/utils/models/gguf_metadata.py
index 5e25ce1927..c24ec28e1d 100644
--- a/studio/backend/utils/models/gguf_metadata.py
+++ b/studio/backend/utils/models/gguf_metadata.py
@@ -50,8 +50,6 @@ _CACHE_MAX_ENTRIES = 4096
# keyed by (file cache key, wanted key). None = key absent / file unreadable.
_BOOL_CACHE: Dict[Tuple[_CacheKey, str], Optional[bool]] = {}
-_STRING_CACHE: Dict[Tuple[_CacheKey, str], Optional[str]] = {}
-
# Native training context length (``{arch}.context_length``). None = absent /
# unreadable. Lets the UI show the real context ceiling before a model loads.
_CONTEXT_CACHE: Dict[_CacheKey, Optional[int]] = {}
@@ -355,83 +353,6 @@ def _read_gguf_bool(path: str, wanted_key: str) -> Optional[bool]:
return result
-def _parse_gguf_string(path: str, wanted_key: str) -> Optional[str]:
- try:
- with open(path, "rb") as f:
- head = f.read(24)
- if len(head) < 24:
- return None
- magic, _version, _tcount, kv_count = struct.unpack(" 1 << 20:
- break
- kbytes = f.read(klen)
- if len(kbytes) < klen:
- break
- key = kbytes.decode("utf-8", "replace")
- vt_bytes = f.read(4)
- if len(vt_bytes) < 4:
- break
- vtype = struct.unpack(" 1 << 22:
- break
- sbytes = f.read(slen)
- if len(sbytes) < slen:
- break
- return sbytes.decode("utf-8", "replace")
- if not _skip_gguf_value(f, vtype):
- break
- except (struct.error, UnicodeDecodeError):
- break
- except OSError as e:
- logger.debug(f"_parse_gguf_string: cannot open {path}: {e}")
- return None
- except Exception as e:
- logger.debug(f"_parse_gguf_string: parse failure on {path}: {e}")
- return None
- return None
-
-
-def _read_gguf_string(path: str, wanted_key: str) -> Optional[str]:
- fkey = _cache_key(path)
- if fkey is None:
- return None
- ckey = (fkey, wanted_key)
- with _CACHE_LOCK:
- if ckey in _STRING_CACHE:
- return _STRING_CACHE[ckey]
- result = _parse_gguf_string(path, wanted_key)
- with _CACHE_LOCK:
- while len(_STRING_CACHE) >= _CACHE_MAX_ENTRIES:
- try:
- _STRING_CACHE.pop(next(iter(_STRING_CACHE)))
- except StopIteration:
- break
- _STRING_CACHE[ckey] = result
- return result
-
-
-def read_gguf_chat_template(path: str) -> Optional[str]:
- template = _read_gguf_string(path, "tokenizer.chat_template")
- if isinstance(template, str) and template.strip():
- return template
- return None
-
-
def read_mmproj_audio_capability(path: str) -> Optional[bool]:
"""``clip.has_audio_encoder`` from an mmproj GGUF (e.g. Gemma 4's
gemma4ua): ``True``/``False`` if present, ``None`` if absent/unreadable.
diff --git a/studio/frontend/src/app/routes/__root.tsx b/studio/frontend/src/app/routes/__root.tsx
index 6c2505f1ca..ba56ce7525 100644
--- a/studio/frontend/src/app/routes/__root.tsx
+++ b/studio/frontend/src/app/routes/__root.tsx
@@ -195,6 +195,9 @@ function RootLayout() {
chatRuntime.setActiveThreadId(null);
chatRuntime.setActiveProjectId(null);
chatRuntime.setIncognito(false);
+ // Detach the staging UI but keep any in-flight download running, like Hub.
+ if (chatRuntime.pendingSelection)
+ chatRuntime.abandonStagedModel({ keepDownload: true });
void navigate({
to: "/chat",
search: { new: crypto.randomUUID() },
@@ -217,6 +220,10 @@ function RootLayout() {
chatRuntime.setActiveProjectId(null);
chatRuntime.setActiveThreadId(null);
chatRuntime.setIncognito(false);
+ // Leaving chat must not kill an in-flight download: detach the staging UI
+ // but keep the transfer running in the manager, like a Hub download.
+ if (chatRuntime.pendingSelection)
+ chatRuntime.abandonStagedModel({ keepDownload: true });
}, [isChatRoute]);
return (
diff --git a/studio/frontend/src/features/model-picker/components/model-selector.tsx b/studio/frontend/src/components/assistant-ui/model-selector.tsx
similarity index 86%
rename from studio/frontend/src/features/model-picker/components/model-selector.tsx
rename to studio/frontend/src/components/assistant-ui/model-selector.tsx
index 7efcf69162..6bfd1276ac 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector.tsx
+++ b/studio/frontend/src/components/assistant-ui/model-selector.tsx
@@ -3,7 +3,6 @@
"use client";
-import { Input } from "@/components/ui/input";
import {
Popover,
PopoverContent,
@@ -11,7 +10,7 @@ import {
} from "@/components/ui/popover";
import { TooltipProvider } from "@/components/ui/tooltip";
import { usePlatformStore } from "@/config/env";
-import { isCustomProviderType } from "@/features/chat";
+import { isCustomProviderType } from "@/features/chat/external-providers";
import { ChevronDownStandardIcon } from "@/lib/chevron-icons";
import { cn } from "@/lib/utils";
import {
@@ -34,11 +33,7 @@ import {
useRef,
useState,
} from "react";
-import {
- type PerModelConfig,
- resolveInitialConfig,
-} from "../model-config/per-model-config";
-import { ModelConfigPage } from "./model-config-page";
+import { Input } from "../ui/input";
import { HubModelPicker, hasDownloadedModels } from "./model-selector/pickers";
import { PillTabs } from "./model-selector/pill-tabs";
import {
@@ -50,7 +45,6 @@ import type {
ExternalModelOption,
LoraModelOption,
ModelOption,
- ModelPickTarget,
ModelSelectorChangeMeta,
} from "./model-selector/types";
@@ -128,10 +122,6 @@ interface ModelSelectorProps {
value?: string;
defaultValue?: string;
activeGgufVariant?: string | null;
- activeModelConfig?: PerModelConfig | null;
- activeGgufContextLength?: number | null;
- selectedConfig?: PerModelConfig | null;
- selectedGgufVariant?: string | null;
onValueChange?: (value: string, meta: ModelSelectorChangeMeta) => void;
onEject?: () => void;
onFoldersChange?: () => void;
@@ -295,8 +285,7 @@ function saveLastHubSection(section: HubSection): void {
// when they have downloads, else Recommended.
function defaultHubSection(): HubSection {
return (
- loadLastHubSection() ??
- (hasDownloadedModels() ? "downloaded" : "recommended")
+ loadLastHubSection() ?? (hasDownloadedModels() ? "downloaded" : "recommended")
);
}
@@ -319,11 +308,6 @@ function ModelSelectorContent({
loraModels,
externalModels,
value,
- activeGgufVariant,
- activeModelConfig,
- activeGgufContextLength,
- selectedConfig,
- selectedGgufVariant,
onSelect,
onEject,
onFoldersChange,
@@ -339,11 +323,6 @@ function ModelSelectorContent({
loraModels: LoraModelOption[];
externalModels: ExternalModelOption[];
value?: string;
- activeGgufVariant?: string | null;
- activeModelConfig?: PerModelConfig | null;
- activeGgufContextLength?: number | null;
- selectedConfig?: PerModelConfig | null;
- selectedGgufVariant?: string | null;
onSelect: (id: string, meta: ModelSelectorChangeMeta) => void;
onEject?: () => void;
onFoldersChange?: () => void;
@@ -412,10 +391,6 @@ function ModelSelectorContent({
const effectiveHubSection: HubSection =
hubSection === "connected" && !hasExternal ? "recommended" : hubSection;
- const [configTarget, setConfigTarget] = useState(
- null,
- );
-
// The picker below remounts on each open, but this tab state does not, so a
// persisted selection that lands in lora/external after async load would
// reopen on Hub. Re-derive the default tab on the open edge.
@@ -427,9 +402,6 @@ function ModelSelectorContent({
// user has downloads, else their last section.
setHubSection(wantsConnectedDefault ? "connected" : defaultHubSection());
}
- if (!open && wasOpen.current) {
- setConfigTarget(null);
- }
wasOpen.current = open;
}, [
open,
@@ -480,29 +452,6 @@ function ModelSelectorContent({
}
}
- const visibleConfigTarget = open ? configTarget : null;
- const openConfigPage = (id: string, meta: ModelSelectorChangeMeta) => {
- const leaf = id.includes("/") ? id.slice(id.lastIndexOf("/") + 1) : id;
- setConfigTarget({
- id,
- displayName: meta.ggufVariant ? `${leaf} · ${meta.ggufVariant}` : leaf,
- ggufVariant: meta.ggufVariant ?? null,
- isGguf: meta.isGguf ?? Boolean(meta.ggufVariant),
- meta,
- });
- };
- const handlePick = (id: string, meta: ModelSelectorChangeMeta) => {
- if (meta.source === "external") {
- onSelect(id, meta);
- return;
- }
- const resolved = resolveInitialConfig(id, meta.ggufVariant);
- onSelect(id, {
- ...meta,
- ...(resolved.remembered ? { config: resolved.config } : {}),
- });
- };
-
return (
@@ -533,42 +477,6 @@ function ModelSelectorContent({
skipDelayDuration={0}
disableHoverableContent={true}
>
- {visibleConfigTarget ? (
- setConfigTarget(null)}
- onRun={(config) =>
- onSelect(visibleConfigTarget.id, {
- ...visibleConfigTarget.meta,
- config,
- forceReload: true,
- })
- }
- loadedConfig={
- value === visibleConfigTarget.id &&
- (activeGgufVariant ?? null) ===
- (visibleConfigTarget.ggufVariant ?? null)
- ? (activeModelConfig ?? null)
- : null
- }
- loadedContextLength={
- value === visibleConfigTarget.id &&
- (activeGgufVariant ?? null) ===
- (visibleConfigTarget.ggufVariant ?? null)
- ? (activeGgufContextLength ?? null)
- : null
- }
- initialConfig={
- value === visibleConfigTarget.id &&
- (selectedGgufVariant ?? null) ===
- (visibleConfigTarget.ggufVariant ?? null)
- ? (selectedConfig ?? null)
- : null
- }
- />
- ) : (
- <>
{tabs.length > 1 ? (
) : null}
+ {/* Hub renders Eject inline as the last list row; other tabs keep the
+ footer button. */}
{effectiveTab !== "hub" && hasSelection && onEject ? (
-
+
) : null}
- >
- )}
);
@@ -658,10 +565,6 @@ export function ModelSelector({
value,
defaultValue,
activeGgufVariant,
- activeModelConfig,
- activeGgufContextLength,
- selectedConfig,
- selectedGgufVariant,
onValueChange,
onEject,
onFoldersChange,
@@ -790,11 +693,6 @@ export function ModelSelector({
loraModels={loraModels}
externalModels={externalModels}
value={selected}
- activeGgufVariant={activeGgufVariant}
- activeModelConfig={activeModelConfig}
- activeGgufContextLength={activeGgufContextLength}
- selectedConfig={selectedConfig}
- selectedGgufVariant={selectedGgufVariant}
onSelect={handleSelect}
onEject={onEject ? handleEject : undefined}
onFoldersChange={onFoldersChange}
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/folder-browser.tsx b/studio/frontend/src/components/assistant-ui/model-selector/folder-browser.tsx
similarity index 88%
rename from studio/frontend/src/features/model-picker/components/model-selector/folder-browser.tsx
rename to studio/frontend/src/components/assistant-ui/model-selector/folder-browser.tsx
index 023f586781..16cc8a1956 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector/folder-browser.tsx
+++ b/studio/frontend/src/components/assistant-ui/model-selector/folder-browser.tsx
@@ -14,7 +14,10 @@ import {
DialogTitle,
} from "@/components/ui/dialog";
import { Spinner } from "@/components/ui/spinner";
-import { type BrowseFoldersResponse, browseFolders } from "@/features/chat";
+import {
+ type BrowseFoldersResponse,
+ browseFolders,
+} from "@/features/chat/api/chat-api";
import { ChevronUpStandardIcon } from "@/lib/chevron-icons";
import { cn } from "@/lib/utils";
import { Folder02Icon } from "@hugeicons/core-free-icons";
@@ -87,43 +90,47 @@ export function FolderBrowser({
const [error, setError] = useState(null);
const abortRef = useRef(null);
- function navigate(
- target: string | undefined,
- hidden: boolean,
- opts?: { fallbackOnError?: boolean },
- ) {
- abortRef.current?.abort();
- const ctrl = new AbortController();
- abortRef.current = ctrl;
- setLoading(true);
- setError(null);
- // Forward the signal so cancelled navigation aborts the backend
- // enumeration, not just the response.
- browseFolders(target, hidden, ctrl.signal)
- .then((res) => {
- if (ctrl.signal.aborted) return;
- setData(res);
- setPath(res.current);
- })
- .catch((err) => {
- if (ctrl.signal.aborted) return;
- // Surface the error; if the first request (e.g. a bad initialPath)
- // fails, fall back to HOME so the modal stays navigable.
- const message = err instanceof Error ? err.message : String(err);
- setError(message);
- if (opts?.fallbackOnError && target !== undefined) {
- // Re-issue without a target -> backend defaults to HOME.
- // Don't recurse if HOME itself fails (allowlist always has HOME).
- queueMicrotask(() => navigate(undefined, hidden));
- }
- })
- .finally(() => {
- if (!ctrl.signal.aborted) setLoading(false);
- });
- }
+ const navigate = useCallback(
+ (
+ target: string | undefined,
+ hidden: boolean,
+ opts?: { fallbackOnError?: boolean },
+ ) => {
+ abortRef.current?.abort();
+ const ctrl = new AbortController();
+ abortRef.current = ctrl;
+ setLoading(true);
+ setError(null);
+ // Forward the signal so cancelled navigation aborts the backend
+ // enumeration, not just the response.
+ browseFolders(target, hidden, ctrl.signal)
+ .then((res) => {
+ if (ctrl.signal.aborted) return;
+ setData(res);
+ setPath(res.current);
+ })
+ .catch((err) => {
+ if (ctrl.signal.aborted) return;
+ // Surface the error; if the first request (e.g. a bad initialPath)
+ // fails, fall back to HOME so the modal stays navigable.
+ const message = err instanceof Error ? err.message : String(err);
+ setError(message);
+ if (opts?.fallbackOnError && target !== undefined) {
+ // Re-issue without a target -> backend defaults to HOME.
+ // Don't recurse if HOME itself fails (allowlist always has HOME).
+ queueMicrotask(() => navigate(undefined, hidden));
+ }
+ })
+ .finally(() => {
+ if (!ctrl.signal.aborted) setLoading(false);
+ });
+ },
+ [],
+ );
// Fetch only on closed -> open; later navigation is driven by `navigate()`,
// so `path` is deliberately kept out of the dependency list.
+ // eslint-disable-next-line react-hooks/exhaustive-deps
useEffect(() => {
if (!open) return;
// fallbackOnError: recover into HOME if initialPath is bad, rather than
@@ -140,7 +147,7 @@ export function FolderBrowser({
const crumbs = useMemo(
() => (data?.current ? splitBreadcrumb(data.current) : []),
- [data],
+ [data?.current],
);
return (
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/model-capabilities.ts b/studio/frontend/src/components/assistant-ui/model-selector/model-capabilities.ts
similarity index 100%
rename from studio/frontend/src/features/model-picker/components/model-selector/model-capabilities.ts
rename to studio/frontend/src/components/assistant-ui/model-selector/model-capabilities.ts
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/model-delete-action.tsx b/studio/frontend/src/components/assistant-ui/model-selector/model-delete-action.tsx
similarity index 90%
rename from studio/frontend/src/features/model-picker/components/model-selector/model-delete-action.tsx
rename to studio/frontend/src/components/assistant-ui/model-selector/model-delete-action.tsx
index 09bb43abdc..4de96d3648 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector/model-delete-action.tsx
+++ b/studio/frontend/src/components/assistant-ui/model-selector/model-delete-action.tsx
@@ -1,12 +1,12 @@
// SPDX-License-Identifier: AGPL-3.0-only
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
-import { DeleteConfirmDialog } from "@/features/hub";
-import { toast } from "@/lib/toast";
+import { DeleteConfirmDialog } from "@/features/hub/catalog/download-card";
import { cn } from "@/lib/utils";
import { Delete02Icon } from "@hugeicons/core-free-icons";
import { HugeiconsIcon } from "@hugeicons/react";
-import { type ReactNode, useCallback, useState } from "react";
+import { useCallback, useState, type ReactNode } from "react";
+import { toast } from "@/lib/toast";
interface ModelDeleteActionProps {
ariaLabel: string;
@@ -63,8 +63,7 @@ export function ModelDeleteAction({
disabled={disabled}
className={cn(
"shrink-0 rounded-md p-1.5 text-muted-foreground/60 transition-colors hover:bg-destructive/10 hover:text-destructive",
- disabled &&
- "cursor-not-allowed opacity-40 hover:bg-transparent hover:text-muted-foreground/60",
+ disabled && "cursor-not-allowed opacity-40 hover:bg-transparent hover:text-muted-foreground/60",
buttonClassName,
)}
>
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/model-load-settings-action.tsx b/studio/frontend/src/components/assistant-ui/model-selector/model-load-settings-action.tsx
similarity index 66%
rename from studio/frontend/src/features/model-picker/components/model-selector/model-load-settings-action.tsx
rename to studio/frontend/src/components/assistant-ui/model-selector/model-load-settings-action.tsx
index e1d48b5a08..58510762d4 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector/model-load-settings-action.tsx
+++ b/studio/frontend/src/components/assistant-ui/model-selector/model-load-settings-action.tsx
@@ -6,16 +6,24 @@ import {
TooltipContent,
TooltipTrigger,
} from "@/components/ui/tooltip";
+import { useChatRuntimeStore } from "@/features/chat/stores/chat-runtime-store";
import { cn } from "@/lib/utils";
import { Settings02Icon } from "@hugeicons/core-free-icons";
import { HugeiconsIcon } from "@hugeicons/react";
+/** Gear button on a downloaded quant row. Stages the model into the Run
+ * settings sidebar (always, regardless of the Load-on-selection toggle) so the
+ * user can set load options, then click Load model. */
export function ModelLoadSettingsAction({
ariaLabel,
- onConfigure,
+ repoId,
+ quant,
+ maxContext,
}: {
ariaLabel: string;
- onConfigure: () => void;
+ repoId: string;
+ quant: string;
+ maxContext?: number | null;
}) {
return (
@@ -24,7 +32,12 @@ export function ModelLoadSettingsAction({
type="button"
onClick={(e) => {
e.stopPropagation();
- onConfigure();
+ useChatRuntimeStore.getState().stageModel({
+ id: repoId,
+ ggufVariant: quant,
+ isDownloaded: true,
+ contextLength: maxContext ?? null,
+ });
}}
aria-label={ariaLabel}
className={cn(
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/model-update-action.tsx b/studio/frontend/src/components/assistant-ui/model-selector/model-update-action.tsx
similarity index 82%
rename from studio/frontend/src/features/model-picker/components/model-selector/model-update-action.tsx
rename to studio/frontend/src/components/assistant-ui/model-selector/model-update-action.tsx
index b13ed33d04..db7628777a 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector/model-update-action.tsx
+++ b/studio/frontend/src/components/assistant-ui/model-selector/model-update-action.tsx
@@ -1,20 +1,12 @@
// SPDX-License-Identifier: AGPL-3.0-only
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
-import {
- UpdateConfirmDialog,
- ggufVariantsMatch,
- subscribeJobListeners,
-} from "@/features/hub";
+import { subscribeJobListeners } from "@/features/hub/download-manager";
+import { UpdateConfirmDialog } from "@/features/hub/catalog/download-card";
+import { ggufVariantsMatch } from "@/features/hub/lib/model-identity";
import { cn } from "@/lib/utils";
import { RefreshCw } from "lucide-react";
-import {
- type ReactNode,
- useCallback,
- useEffect,
- useRef,
- useState,
-} from "react";
+import { useCallback, useEffect, useRef, useState, type ReactNode } from "react";
import { toast } from "sonner";
interface ModelUpdateActionProps {
@@ -50,10 +42,10 @@ export function ModelUpdateAction({
}: ModelUpdateActionProps) {
const [open, setOpen] = useState(false);
+ // Refresh the caller when this repo+variant's download finishes so the "update available" cue
+ // clears. A ref keeps the subscription stable across renders.
const onUpdatedRef = useRef(onUpdated);
- useEffect(() => {
- onUpdatedRef.current = onUpdated;
- }, [onUpdated]);
+ onUpdatedRef.current = onUpdated;
useEffect(() => {
return subscribeJobListeners("model", repoId, {
onComplete: (completedVariant) => {
@@ -91,8 +83,7 @@ export function ModelUpdateAction({
disabled={disabled}
className={cn(
"shrink-0 rounded-md p-1.5 text-muted-foreground/60 transition-colors hover:bg-amber-500/10 hover:text-amber-700 dark:hover:bg-amber-500/15 dark:hover:text-amber-300",
- disabled &&
- "cursor-not-allowed opacity-40 hover:bg-transparent hover:text-muted-foreground/60",
+ disabled && "cursor-not-allowed opacity-40 hover:bg-transparent hover:text-muted-foreground/60",
buttonClassName,
)}
>
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/model-usage.ts b/studio/frontend/src/components/assistant-ui/model-selector/model-usage.ts
similarity index 93%
rename from studio/frontend/src/features/model-picker/components/model-selector/model-usage.ts
rename to studio/frontend/src/components/assistant-ui/model-selector/model-usage.ts
index c6665e7658..dbcd4b9a1b 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector/model-usage.ts
+++ b/studio/frontend/src/components/assistant-ui/model-selector/model-usage.ts
@@ -40,8 +40,7 @@ export function loadedAt(times: ModelLoadTimes, id: string): number {
export function useModelLoadTimes(currentValue?: string): ModelLoadTimes {
const [times, setTimes] = useState(() => readLoadTimes());
useEffect(() => {
- if (!currentValue) return;
- queueMicrotask(() => setTimes(recordModelLoaded(currentValue)));
+ if (currentValue) setTimes(recordModelLoaded(currentValue));
}, [currentValue]);
return times;
}
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/pickers.tsx b/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx
similarity index 89%
rename from studio/frontend/src/features/model-picker/components/model-selector/pickers.tsx
rename to studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx
index 1569ca0581..8e06181585 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector/pickers.tsx
+++ b/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx
@@ -10,46 +10,49 @@ import {
TooltipTrigger,
} from "@/components/ui/tooltip";
import { usePlatformStore } from "@/config/env";
-import { ApiProviderLogo } from "@/features/chat";
+import { ApiProviderLogo } from "@/features/chat/api-provider-logo";
import {
type ScanFolderInfo,
addScanFolder,
deleteCachedModel,
deleteFineTunedModel,
+ listCachedGguf,
+ listCachedModels,
listGgufVariants,
+ listLocalModels,
listRecommendedFolders,
listScanFolders,
removeScanFolder,
-} from "@/features/chat";
-import { useChatRuntimeStore } from "@/features/chat";
+} from "@/features/chat/api/chat-api";
+import { useChatRuntimeStore } from "@/features/chat/stores/chat-runtime-store";
import type {
CachedGgufRepo,
CachedModelRepo,
- GgufVariantDetail,
LocalModelInfo,
-} from "@/features/chat";
+} from "@/features/chat/api/chat-api";
+import type { GgufVariantDetail } from "@/features/chat/types/api";
+import { DotTag } from "@/features/hub/catalog/dot-tag";
import {
- DotTag,
type HubOption,
HubOptionMenu,
- TrainIcon,
- TransportConflictDialog,
- useHubInfiniteScroll,
-} from "@/features/hub";
+} from "@/features/hub/catalog/hub-option-menu";
+import { TransportConflictDialog } from "@/features/hub/catalog/transport-conflict-dialog";
+import { TrainIcon } from "@/features/hub/components/train-icon";
+import { useHubInfiniteScroll } from "@/features/hub/hooks/use-hub-infinite-scroll";
import {
type HfModelResult,
type HfSortKey,
useHubModelSearch,
-} from "@/features/hub";
+} from "@/features/hub/hooks/use-hub-model-search";
+import { useOnlineStatus } from "@/features/hub/hooks/use-online-status";
+import { isHiddenModelId } from "@/features/hub/lib/hidden-models";
+import { classifyUnslothSupport } from "@/features/hub/lib/unsloth-support";
+import { useHfTokenStore } from "@/features/hub/stores/hf-token-store";
import {
- classifyUnslothSupport,
downloadManager,
- isHiddenModelId,
jobKeyOf,
useDownloadManagerStore,
- useHfTokenStore,
- useOnlineStatus,
-} from "@/features/hub";
+} from "@/features/hub/download-manager";
import { useDebouncedValue, useGpuInfo } from "@/hooks";
import { extractParamLabel } from "@/lib/model-size";
import { toast } from "@/lib/toast";
@@ -82,7 +85,6 @@ import {
useRef,
useState,
} from "react";
-import { useChatPickerInventory } from "../../inventory/use-chat-picker-inventory";
import { FolderBrowser } from "./folder-browser";
import {
type ModelCapabilities,
@@ -90,8 +92,8 @@ import {
hasAnyCapability,
} from "./model-capabilities";
import { ModelDeleteAction } from "./model-delete-action";
-import { ModelLoadSettingsAction } from "./model-load-settings-action";
import { ModelUpdateAction } from "./model-update-action";
+import { ModelLoadSettingsAction } from "./model-load-settings-action";
import {
type ModelLoadTimes,
loadedAt,
@@ -657,7 +659,6 @@ function GgufVariantExpander({
parentOptionKey,
onNavigatePastStart,
onNavigatePastEnd,
- onConfigure,
sourceOverride,
variantActions,
onDevice = false,
@@ -673,7 +674,6 @@ function GgufVariantExpander({
parentOptionKey?: string;
onNavigatePastStart?: () => void;
onNavigatePastEnd?: () => void;
- onConfigure?: (id: string, meta: ModelSelectorChangeMeta) => void;
sourceOverride?: ModelSelectorChangeMeta["source"];
/** Update/delete actions for cached variant rows. Omitted by browse-only
* expanders (Recommended, etc.) that don't manage on-disk variants. */
@@ -715,11 +715,8 @@ function GgufVariantExpander({
useEffect(() => {
let canceled = false;
- queueMicrotask(() => {
- if (canceled) return;
- setLoading(true);
- setError(null);
- });
+ setLoading(true);
+ setError(null);
listGgufVariants(repoId, hfToken)
.then((res) => {
@@ -747,7 +744,7 @@ function GgufVariantExpander({
}, [repoId, refreshKey, hfToken]);
// Covers Unix absolute (/), Windows drive (C:\, D:/), UNC (\\server), relative (./, ../), tilde (~/)
- const isLocalPath = /^(\/|\.{1,2}[\\/]|~[\\/]|[A-Za-z]:[\\/]|\\\\)/.test(
+ const isLocalPath = /^(\/|\.{1,2}[\\\/]|~[\\\/]|[A-Za-z]:[\\\/]|\\\\)/.test(
repoId,
);
@@ -756,7 +753,8 @@ function GgufVariantExpander({
// Only seed the staged context for picks whose weights are already on
// disk. The staging effect short-circuits on a known contextLength
// (pendingHasContext) before starting the download, so attaching it to an
- // undownloaded quant from a partially cached repo would skip the download.
+ // undownloaded quant from a partially cached repo would skip the download
+ // entirely (and, with Load on selection, never load).
const isAvailable = isLocalPath || downloaded === true;
onSelect(repoId, {
source: sourceOverride ?? (isLocalPath ? "local" : "hub"),
@@ -989,8 +987,7 @@ function GgufVariantExpander({
This will update{" "}
{repoId} ({v.quant})
-
- {"."}
+ {"."}
>
)
}
@@ -1003,20 +1000,12 @@ function GgufVariantExpander({
onUpdated={() => setRefreshKey((key) => key + 1)}
/>
)}
- {v.downloaded && onConfigure && (
+ {v.downloaded && (
- onConfigure(repoId, {
- source: sourceOverride ?? (isLocalPath ? "local" : "hub"),
- isLora: false,
- ggufVariant: v.quant,
- isDownloaded: true,
- expectedBytes,
- contextLength: nativeContext,
- isGguf: true,
- })
- }
+ repoId={repoId}
+ quant={v.quant}
+ maxContext={nativeContext}
/>
)}
{v.downloaded && onDeleteVariant && (
@@ -1220,19 +1209,6 @@ function localPathTooltip(name: string, path: string): ReactNode {
);
}
-function localModelMeta(isGguf = false): ModelSelectorChangeMeta {
- return {
- source: "local",
- isLora: false,
- isDownloaded: true,
- ...(isGguf ? { isGguf: true } : {}),
- };
-}
-
-function localDirectGgufMeta(): ModelSelectorChangeMeta {
- return localModelMeta(true);
-}
-
/** Hugging Face address for an online/Hub row, or undefined when the repo id is
* missing so the row shows no (empty) address line on hover. */
function hubRepoUrl(id: string | null | undefined): string | undefined {
@@ -1243,7 +1219,9 @@ function hubRepoUrl(id: string | null | undefined): string | undefined {
/** Whether a local model is an MLX build (name hint). MLX runs on Mac only, so
* callers gate visibility on the host being a Mac. */
function localModelIsMlx(m: LocalModelInfo): boolean {
- return isMlxId(m.id) || isMlxId(m.display_name) || isMlxId(m.model_id ?? "");
+ return (
+ isMlxId(m.id) || isMlxId(m.display_name) || isMlxId(m.model_id ?? "")
+ );
}
/** Whether a local model matches the format toggle (GGUF detected by name/path). */
@@ -1267,7 +1245,6 @@ export function HubModelPicker({
onFoldersChange,
onBrowseHub,
onModelsChange,
- onConfigure,
deleteDisabled = false,
section = "downloaded",
sectionToggle,
@@ -1284,12 +1261,12 @@ export function HubModelPicker({
/** Open the full Hub page to browse more models. */
onBrowseHub?: () => void;
onModelsChange?: (deletedModel?: DeletedModelRef) => void;
- onConfigure?: (id: string, meta: ModelSelectorChangeMeta) => void;
deleteDisabled?: boolean;
/** Section shown when not searching. Search spans all sections. */
section?: "downloaded" | "recommended" | "custom" | "connected";
/** Section toggle rendered under the search bar. */
sectionToggle?: ReactNode;
+ /** Eject the loaded model. Rendered as the last list row when set. */
onEject?: () => void;
}) {
const gpu = useGpuInfo();
@@ -1390,14 +1367,12 @@ export function HubModelPicker({
const setFitOnDeviceOnly = useChatRuntimeStore((s) => s.setFitOnDeviceOnly);
// Repos the user clicked to collapse while expand-by-default is on. Kept in
// memory only, so it resets on reload (and when the setting is toggled).
- const [collapsedGgufState, setCollapsedGgufState] = useState<{
- expandQuantizations: boolean;
- value: Set;
- }>(() => ({ expandQuantizations, value: new Set() }));
- const collapsedGguf =
- collapsedGgufState.expandQuantizations === expandQuantizations
- ? collapsedGgufState.value
- : new Set();
+ const [collapsedGguf, setCollapsedGguf] = useState>(
+ () => new Set(),
+ );
+ useEffect(() => {
+ setCollapsedGguf(new Set());
+ }, [expandQuantizations]);
const isGgufExpanded = useCallback(
(id: string) =>
expandQuantizations ? !collapsedGguf.has(id) : expandedGguf === id,
@@ -1408,15 +1383,11 @@ export function HubModelPicker({
const toggleGgufExpanded = useCallback(
(id: string) => {
if (expandQuantizations) {
- setCollapsedGgufState((prev) => {
- const current =
- prev.expandQuantizations === expandQuantizations
- ? prev.value
- : new Set();
- const next = new Set(current);
+ setCollapsedGguf((prev) => {
+ const next = new Set(prev);
if (next.has(id)) next.delete(id);
else next.add(id);
- return { expandQuantizations, value: next };
+ return next;
});
} else {
setExpandedGguf((prev) => (prev === id ? null : id));
@@ -1476,37 +1447,15 @@ export function HubModelPicker({
});
}, []);
- const pickerInventory = useChatPickerInventory({ enabled: true });
- const { cachedGguf, cachedModels, cachedReady, refreshInventory } =
- pickerInventory;
- const lmStudioModels = useMemo(
- () =>
- sortLmStudio(
- pickerInventory.localModels.filter((m) => m.source === "lmstudio"),
- ),
- [pickerInventory.localModels],
- );
- const localDirModels = useMemo(
- () => pickerInventory.localModels.filter((m) => m.source === "models_dir"),
- [pickerInventory.localModels],
- );
- const customFolderModels = useMemo(
- () => pickerInventory.localModels.filter((m) => m.source === "custom"),
- [pickerInventory.localModels],
- );
- useEffect(() => {
- _cachedGgufCache = cachedGguf;
- _cachedModelsCache = cachedModels;
- _lmStudioCache = lmStudioModels;
- _localDirCache = localDirModels;
- _customFolderCache = customFolderModels;
- }, [
- cachedGguf,
- cachedModels,
- lmStudioModels,
- localDirModels,
- customFolderModels,
- ]);
+ // Cached (downloaded) repos -- module-level cache avoids flashing an
+ // empty "Downloaded" section when the popover re-mounts.
+ const [cachedGguf, setCachedGguf] =
+ useState(_cachedGgufCache);
+ const [cachedModels, setCachedModels] =
+ useState(_cachedModelsCache);
+ const alreadyCached =
+ _cachedGgufCache.length > 0 || _cachedModelsCache.length > 0;
+ const [cachedReady, setCachedReady] = useState(alreadyCached);
const [updateConflictKey, setUpdateConflictKey] = useState(
null,
);
@@ -1530,6 +1479,16 @@ export function HubModelPicker({
setUpdateConflictKey(null);
}, [updateConflictKey]);
+ // LM Studio local models -- module-level cache, same pattern as above.
+ const [lmStudioModels, setLmStudioModels] =
+ useState(_lmStudioCache);
+ // Models found under the local models directory (./models), so they stay
+ // selectable on the On Device tab after leaving the Fine-tuned tab.
+ const [localDirModels, setLocalDirModels] =
+ useState(_localDirCache);
+ const [customFolderModels, setCustomFolderModels] =
+ useState(_customFolderCache);
+
// Custom scan folders management
const [scanFolders, setScanFolders] =
useState(_scanFoldersCache);
@@ -1541,8 +1500,22 @@ export function HubModelPicker({
const [recommendedFolders, setRecommendedFolders] = useState([]);
const refreshLocalModelsList = useCallback(() => {
- void pickerInventory.refreshInventory();
- }, [pickerInventory.refreshInventory]);
+ listLocalModels()
+ .then((res) => {
+ const lm = sortLmStudio(
+ res.models.filter((m) => m.source === "lmstudio"),
+ );
+ _lmStudioCache = lm;
+ setLmStudioModels(lm);
+ const ld = res.models.filter((m) => m.source === "models_dir");
+ _localDirCache = ld;
+ setLocalDirModels(ld);
+ const cf = res.models.filter((m) => m.source === "custom");
+ _customFolderCache = cf;
+ setCustomFolderModels(cf);
+ })
+ .catch(() => {});
+ }, []);
const refreshScanFolders = useCallback(() => {
listScanFolders()
@@ -1621,37 +1594,39 @@ export function HubModelPicker({
);
const refreshCachedLists = useCallback(() => {
- void pickerInventory.refreshInventory();
- }, [pickerInventory.refreshInventory]);
+ listCachedGguf()
+ .then((v) => {
+ _cachedGgufCache = v;
+ setCachedGguf(v);
+ })
+ .catch(() => {});
+ listCachedModels(hfToken || undefined)
+ .then((v) => {
+ _cachedModelsCache = v;
+ setCachedModels(v);
+ })
+ .catch(() => {});
+ refreshLocalModelsList();
+ }, [hfToken, refreshLocalModelsList]);
// Updates run as managed downloads (Downloads panel: progress + Cancel), not a blocking
// call. The worker pulls only changed blobs, so the cached copy stays usable until done.
- const startManagedUpdate = useCallback(
- (repoId: string, variant: string, expectedBytes: number) => {
- return downloadManager
- .requestStart({
- kind: "model",
- repoId,
- variant,
- expectedBytes,
- })
- .then((outcome) => {
- if (outcome === "conflict") {
- setUpdateConflictKey(jobKeyOf("model", repoId, variant));
- } else if (outcome === "busy") {
- // A sibling variant/snapshot for this repo is already downloading,
- // so this update did not start. Say so instead of closing the
- // dialog as if it began and leaving the cached copy stale.
- toast.info("A download for this model is already in progress", {
- description: "Try updating again once it finishes.",
- });
- } else if (outcome === "error") {
- throw new Error("Failed to start update");
- }
- });
- },
- [],
- );
+ const startManagedUpdate = useCallback((repoId: string, variant: string, expectedBytes: number) => {
+ return downloadManager
+ .requestStart({
+ kind: "model",
+ repoId,
+ variant,
+ expectedBytes,
+ })
+ .then((outcome) => {
+ if (outcome === "conflict") {
+ setUpdateConflictKey(jobKeyOf("model", repoId, variant));
+ } else if (outcome === "error") {
+ throw new Error("Failed to start update");
+ }
+ });
+ }, []);
const updateGgufVariant = useCallback(
(repoId: string, quant: string, expectedBytes: number) =>
@@ -1660,15 +1635,36 @@ export function HubModelPicker({
);
useEffect(() => {
+ // Always refresh LM Studio + custom folder models (not gated by alreadyCached).
+ refreshLocalModelsList();
refreshScanFolders();
listRecommendedFolders()
.then(setRecommendedFolders)
.catch(() => {});
- }, [refreshScanFolders]);
- useEffect(() => {
- void refreshInventory();
- }, [refreshInventory]);
+ // Always refetch cached GGUF/model lists. The module-level caches render
+ // instantly with stale data (no spinner flash), but newly downloaded
+ // repos need a fresh backend hit. cachedReady=alreadyCached initially,
+ // so the background refresh is invisible when we already had data.
+ let done = 0;
+ const check = () => {
+ if (++done >= 2) setCachedReady(true);
+ };
+ listCachedGguf()
+ .then((v) => {
+ _cachedGgufCache = v;
+ setCachedGguf(v);
+ })
+ .catch(() => {})
+ .finally(check);
+ listCachedModels(hfToken || undefined)
+ .then((v) => {
+ _cachedModelsCache = v;
+ setCachedModels(v);
+ })
+ .catch(() => {})
+ .finally(check);
+ }, [hfToken, refreshLocalModelsList, refreshScanFolders]);
// Hide downloaded models from the recommended list. Case-insensitive
// since the HF cache lowercases repo IDs.
@@ -1705,8 +1701,7 @@ export function HubModelPicker({
// Chat-only keeps runnable formats: GGUF anywhere, plus MLX/safetensors
// on Mac (matches the empty Recommended view so search stays consistent).
.filter(
- (id) =>
- !chatOnly || isRecommendableFormat(id, isKnownGgufRepo(id), isMac),
+ (id) => !chatOnly || isRecommendableFormat(id, isKnownGgufRepo(id), isMac),
)
.filter((id) => !/-FP8[-.]|FP8-Dynamic/i.test(id));
// Sort: GGUFs first, then hub models
@@ -2057,8 +2052,7 @@ export function HubModelPicker({
// Chat-only keeps runnable formats: GGUF anywhere, plus MLX/safetensors
// on Mac (matches the empty Recommended view so search stays consistent).
.filter(
- (id) =>
- !chatOnly || isRecommendableFormat(id, isKnownGgufRepo(id), isMac),
+ (id) => !chatOnly || isRecommendableFormat(id, isKnownGgufRepo(id), isMac),
)
.filter((id) => !/-FP8[-.]|FP8-Dynamic/i.test(id))
.filter((id) =>
@@ -2514,7 +2508,6 @@ export function HubModelPicker({
onDevice={true}
onHasVision={(v) => reportVision(c.repo_id, v)}
onSelect={onSelect}
- onConfigure={onConfigure}
hfToken={hfToken || undefined}
parentOptionKey={optionKey}
onNavigatePastStart={() => hubModelList.focusOption(optionKey)}
@@ -2524,6 +2517,7 @@ export function HubModelPicker({
variantActions={{
onUpdate: (quant, expectedBytes) =>
updateGgufVariant(c.repo_id, quant, expectedBytes),
+ // Can't update the model that's live in memory under itself.
updateDisabled: loadedModelId === c.repo_id,
onDelete: async (quant) => {
await deleteCachedModel(c.repo_id, quant);
@@ -2568,19 +2562,6 @@ export function HubModelPicker({
className={downloadedRowButtonClassName}
/>
- {onConfigure && (
-
- onConfigure(c.repo_id, {
- source: "hub",
- isLora: false,
- isDownloaded: true,
- isGguf: false,
- })
- }
- />
- )}
+ {/* Clear space for the floating Eject pill when scrolled to the end, so
+ its gap above the last row matches its gap below (applies to every
+ section, including Recommended). */}
-
+
Custom Folders
@@ -3160,70 +3140,54 @@ export function HubModelPicker({
);
return (
+ {/* Floating eject pill: overlaid on the list bottom, outside the scroll
+ so the edge fade never touches it. Only the pill catches clicks. */}
{onEject ? (
- {canConfigure && onConfigure && (
- onConfigure(adapter.id, selectionMeta)}
- />
- )}
{canDelete && (
loraModelList.focusOption(optionKey)}
onNavigatePastEnd={() =>
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/pill-tabs.tsx b/studio/frontend/src/components/assistant-ui/model-selector/pill-tabs.tsx
similarity index 98%
rename from studio/frontend/src/features/model-picker/components/model-selector/pill-tabs.tsx
rename to studio/frontend/src/components/assistant-ui/model-selector/pill-tabs.tsx
index fbc1d5ac91..e6da8a7b74 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector/pill-tabs.tsx
+++ b/studio/frontend/src/components/assistant-ui/model-selector/pill-tabs.tsx
@@ -78,8 +78,7 @@ export function PillTabs({
onValueChange(tabs[next].value);
e.currentTarget.parentElement
?.querySelectorAll('button[role="tab"]')
- .item(next)
- ?.focus();
+ [next]?.focus();
}}
onClick={() => onValueChange(tab.value)}
className={cn(
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/recommended-fit.ts b/studio/frontend/src/components/assistant-ui/model-selector/recommended-fit.ts
similarity index 100%
rename from studio/frontend/src/features/model-picker/components/model-selector/recommended-fit.ts
rename to studio/frontend/src/components/assistant-ui/model-selector/recommended-fit.ts
diff --git a/studio/frontend/src/components/assistant-ui/model-selector/remembered-load-settings.ts b/studio/frontend/src/components/assistant-ui/model-selector/remembered-load-settings.ts
new file mode 100644
index 0000000000..ec75b17f20
--- /dev/null
+++ b/studio/frontend/src/components/assistant-ui/model-selector/remembered-load-settings.ts
@@ -0,0 +1,69 @@
+// SPDX-License-Identifier: AGPL-3.0-only
+// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+
+// Per-model pre-load inference settings, persisted in localStorage so the load
+// dialog can offer "Remember settings for ".
+
+const KEY = "unsloth_load_settings";
+
+export interface RememberedLoadSettings {
+ contextLength: number | null;
+ kvCacheDtype: string | null;
+ speculativeType: string | null;
+ specDraftNMax: number | null;
+ tensorParallel: boolean;
+}
+
+// Storage key for a pick's remembered settings. The remembered knobs are
+// VRAM-budget driven (context override, KV-cache dtype, tensor-parallel), so the
+// right values differ per quant. An HF repo collapses all its GGUF variants into
+// one `id`, so fold the variant in to scope settings per quant. Local .gguf
+// paths key by their file path (already file-specific); native drag-drop files
+// key by display label, so same-named files in different folders share an entry.
+export function rememberedLoadSettingsKey(selection: {
+ id: string;
+ ggufVariant?: string | null;
+}): string {
+ return selection.ggufVariant
+ ? `${selection.id}::${selection.ggufVariant}`
+ : selection.id;
+}
+
+function readAll(): Record {
+ try {
+ return JSON.parse(localStorage.getItem(KEY) ?? "{}");
+ } catch {
+ return {};
+ }
+}
+
+function writeAll(all: Record) {
+ try {
+ localStorage.setItem(KEY, JSON.stringify(all));
+ } catch {
+ // Ignore quota / unavailable storage.
+ }
+}
+
+export function loadRememberedLoadSettings(
+ key: string,
+): RememberedLoadSettings | null {
+ return readAll()[key] ?? null;
+}
+
+export function saveRememberedLoadSettings(
+ key: string,
+ settings: RememberedLoadSettings,
+) {
+ const all = readAll();
+ all[key] = settings;
+ writeAll(all);
+}
+
+export function clearRememberedLoadSettings(key: string) {
+ const all = readAll();
+ if (key in all) {
+ delete all[key];
+ writeAll(all);
+ }
+}
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/row-meta.ts b/studio/frontend/src/components/assistant-ui/model-selector/row-meta.ts
similarity index 100%
rename from studio/frontend/src/features/model-picker/components/model-selector/row-meta.ts
rename to studio/frontend/src/components/assistant-ui/model-selector/row-meta.ts
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/source-tabs.ts b/studio/frontend/src/components/assistant-ui/model-selector/source-tabs.ts
similarity index 100%
rename from studio/frontend/src/features/model-picker/components/model-selector/source-tabs.ts
rename to studio/frontend/src/components/assistant-ui/model-selector/source-tabs.ts
diff --git a/studio/frontend/src/features/model-picker/components/model-selector/types.ts b/studio/frontend/src/components/assistant-ui/model-selector/types.ts
similarity index 76%
rename from studio/frontend/src/features/model-picker/components/model-selector/types.ts
rename to studio/frontend/src/components/assistant-ui/model-selector/types.ts
index dae072236b..6a86515267 100644
--- a/studio/frontend/src/features/model-picker/components/model-selector/types.ts
+++ b/studio/frontend/src/components/assistant-ui/model-selector/types.ts
@@ -2,7 +2,6 @@
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
import type { ReactNode } from "react";
-import type { PerModelConfig } from "../../model-config/per-model-config";
export interface ModelOption {
id: string;
@@ -37,18 +36,6 @@ export interface ModelSelectorChangeMeta {
/** Direct local .gguf file picked without a variant (custom folder / LM
* Studio). Marks it as a GGUF source for the deferred-load staging flow. */
isGguf?: boolean;
- config?: PerModelConfig;
- forceReload?: boolean;
- /** Native path token so an active-model reload can reopen a file-picked GGUF. */
- nativePathToken?: string;
-}
-
-export interface ModelPickTarget {
- id: string;
- displayName: string;
- ggufVariant?: string | null;
- isGguf: boolean;
- meta: ModelSelectorChangeMeta;
}
export interface DeletedModelRef {
diff --git a/studio/frontend/src/features/chat/api/chat-adapter.ts b/studio/frontend/src/features/chat/api/chat-adapter.ts
index d3bdcb6898..0bf46e7343 100644
--- a/studio/frontend/src/features/chat/api/chat-adapter.ts
+++ b/studio/frontend/src/features/chat/api/chat-adapter.ts
@@ -2,7 +2,10 @@
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
import { getAuthToken } from "@/features/auth";
-import { resolveInitialConfig } from "@/features/model-picker";
+import {
+ loadRememberedLoadSettings,
+ rememberedLoadSettingsKey,
+} from "@/components/assistant-ui/model-selector/remembered-load-settings";
import { projectHasSources } from "@/features/rag/api/rag-api";
import { apiUrl } from "@/lib/api-base";
import { parseParamCountB } from "@/lib/model-size";
@@ -1517,25 +1520,27 @@ async function autoLoadSmallestModel(): Promise<{
return false;
}
const currentStore = useChatRuntimeStore.getState();
- const { config } = resolveInitialConfig(candidate.id, candidate.ggufVariant);
+ const remembered = loadRememberedLoadSettings(
+ rememberedLoadSettingsKey({
+ id: candidate.id,
+ ggufVariant: candidate.ggufVariant,
+ }),
+ );
const effectiveMaxSeqLength = resolveLoadMaxSeqLength({
modelId: candidate.id,
ggufVariant: candidate.ggufVariant,
isGguf: candidate.kind === "gguf",
- customContextLength: config.customContextLength,
+ customContextLength: remembered?.contextLength ?? null,
ggufContextLength: null,
currentCheckpoint: currentStore.params.checkpoint,
activeGgufVariant: currentStore.activeGgufVariant,
- maxSeqLength: config.maxSeqLength ?? candidate.maxSeqLength,
+ maxSeqLength: candidate.maxSeqLength,
presetSource: currentStore.activePresetSource,
});
const effectiveSpeculativeType =
- config.speculativeType ?? specSettings.speculativeType;
+ remembered?.speculativeType ?? specSettings.speculativeType;
const effectiveSpecDraftNMax =
- config.specDraftNMax ?? specSettings.specDraftNMax;
- const effectiveChatTemplateOverride = config.chatTemplateOverride?.trim()
- ? config.chatTemplateOverride
- : null;
+ remembered?.specDraftNMax ?? specSettings.specDraftNMax;
if (
!(await canAutoLoad({
model_path: candidate.id,
@@ -1558,18 +1563,12 @@ async function autoLoadSmallestModel(): Promise<{
is_lora: false,
gguf_variant: candidate.ggufVariant,
trust_remote_code: trustRemoteCode,
- chat_template_override: effectiveChatTemplateOverride,
- cache_type_kv: config.kvCacheDtype,
+ cache_type_kv: remembered?.kvCacheDtype ?? null,
speculative_type: effectiveSpeculativeType,
spec_draft_n_max: effectiveSpecDraftNMax,
- tensor_parallel: config.tensorParallel,
+ tensor_parallel: remembered?.tensorParallel ?? false,
});
- // Only persist the global preference when the value came from the global
- // settings. A per-model config's choice must stay load-local, or autoloading
- // a remembered model on startup would rewrite the global default.
- if (config.speculativeType == null) {
- saveSpeculativeType(effectiveSpeculativeType);
- }
+ saveSpeculativeType(effectiveSpeculativeType);
useChatRuntimeStore
.getState()
.setCheckpoint(candidate.id, candidate.ggufVariant ?? undefined);
@@ -1579,9 +1578,6 @@ async function autoLoadSmallestModel(): Promise<{
);
store.setParams({
...store.params,
- ...(candidate.kind === "gguf"
- ? {}
- : { maxSeqLength: effectiveMaxSeqLength }),
maxTokens:
candidate.kind === "gguf"
? loadResp.context_length ?? 131072
@@ -1618,11 +1614,8 @@ async function autoLoadSmallestModel(): Promise<{
tensorParallel: loadResp.tensor_parallel ?? false,
loadedTensorParallel: loadResp.tensor_parallel ?? false,
defaultChatTemplate: loadResp.chat_template ?? null,
- chatTemplateOverride: effectiveChatTemplateOverride,
- loadedChatTemplateOverride: effectiveChatTemplateOverride,
- // Retain the saved requested context so re-saving the config keeps the
- // override; null stays null (auto/VRAM-fit).
- customContextLength: config.customContextLength,
+ chatTemplateOverride: null,
+ loadedChatTemplateOverride: null,
loadedIsMultimodal: isMultimodalResponse(loadResp),
loadedIsDiffusion: loadResp.is_diffusion ?? false,
...resolveLoadedSpeculativeSettings(loadResp),
@@ -1641,9 +1634,8 @@ async function autoLoadSmallestModel(): Promise<{
tensorParallel: loadResp.tensor_parallel ?? false,
loadedTensorParallel: loadResp.tensor_parallel ?? false,
defaultChatTemplate: loadResp.chat_template ?? null,
- chatTemplateOverride: effectiveChatTemplateOverride,
- loadedChatTemplateOverride: effectiveChatTemplateOverride,
- customContextLength: null,
+ chatTemplateOverride: null,
+ loadedChatTemplateOverride: null,
...resolveLoadedSpeculativeSettings(loadResp),
loadedIsMultimodal: isMultimodalResponse(loadResp),
loadedIsDiffusion: loadResp.is_diffusion ?? false,
diff --git a/studio/frontend/src/features/chat/chat-page.tsx b/studio/frontend/src/features/chat/chat-page.tsx
index 731a551c53..380ce0e0ab 100644
--- a/studio/frontend/src/features/chat/chat-page.tsx
+++ b/studio/frontend/src/features/chat/chat-page.tsx
@@ -2,18 +2,16 @@
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
import {
- applyModelLoadConfigToRuntime,
- currentRuntimePerModelConfig,
type DeletedModelRef,
type ExternalModelOption,
type LoraModelOption,
type ModelOption,
ModelSelector,
- type ModelSelectorChangeMeta,
- type PerModelConfig,
- resolveInitialConfig,
- SidebarModelConfig,
-} from "@/features/model-picker";
+} from "@/components/assistant-ui/model-selector";
+import {
+ loadRememberedLoadSettings,
+ rememberedLoadSettingsKey,
+} from "@/components/assistant-ui/model-selector/remembered-load-settings";
import { ProjectComposer, Thread } from "@/components/assistant-ui/thread";
import { CopyableErrorChip } from "@/components/ui/copyable-error-chip";
import {
@@ -29,10 +27,10 @@ import {
} from "@/components/ui/resizable";
import { useSidebar } from "@/components/ui/sidebar";
import { Tooltip, TooltipContent } from "@/components/ui/tooltip";
+import { useLatestRef } from "@/features/hub/hooks/use-latest-ref";
import {
DOWNLOAD_KIND,
downloadManager,
- useRepoDownload,
} from "@/features/hub/download-manager";
import {
type NativeIntent,
@@ -95,6 +93,7 @@ import {
renameChatItem,
useChatSidebarItems,
} from "./hooks/use-chat-sidebar-items";
+import { useStagedModelPreparation } from "./hooks/use-staged-model-preparation";
import {
clearTrainingCompareHandoff,
getTrainingCompareHandoff,
@@ -129,8 +128,10 @@ import {
hasGgufSource,
isDownloadableHubRepo,
loadOptionalBool,
+ pendingSelectionMatches,
useChatRuntimeStore,
} from "./stores/chat-runtime-store";
+import type { PendingModelSelection } from "./stores/chat-runtime-store";
import { useChatPreferencesStore } from "./stores/chat-preferences-store";
import { useExternalProvidersStore } from "./stores/external-providers-store";
import { buildChatTourSteps } from "./tour";
@@ -384,7 +385,6 @@ type CompareModelSelection = {
id: string;
isLora: boolean;
ggufVariant?: string;
- config?: PerModelConfig;
};
function modelMatchesDeleted(
@@ -645,8 +645,6 @@ function GeneralCompareHeader({
loraModels,
externalModels,
value,
- selectedConfig,
- selectedGgufVariant,
onValueChange,
onFoldersChange,
onModelsChange,
@@ -657,11 +655,9 @@ function GeneralCompareHeader({
loraModels: LoraModelOption[];
externalModels: ExternalModelOption[];
value: string;
- selectedConfig?: PerModelConfig | null;
- selectedGgufVariant?: string | null;
onValueChange: (
id: string,
- meta: ModelSelectorChangeMeta,
+ meta: { isLora: boolean; ggufVariant?: string },
) => void;
onFoldersChange?: () => void;
onModelsChange?: (deletedModel?: DeletedModelRef) => void;
@@ -688,8 +684,6 @@ function GeneralCompareHeader({
loraModels={loraModels}
externalModels={externalModels}
value={value}
- selectedConfig={selectedConfig}
- selectedGgufVariant={selectedGgufVariant}
onValueChange={onValueChange}
onFoldersChange={onFoldersChange}
onModelsChange={onModelsChange}
@@ -817,14 +811,11 @@ const GeneralCompareContent = memo(function GeneralCompareContent({
loraModels={loraModels}
externalModels={externalModels}
value={model1.id}
- selectedConfig={model1.config}
- selectedGgufVariant={model1.ggufVariant}
onValueChange={(id, meta) =>
setModel1({
id,
isLora: meta.isLora,
ggufVariant: meta.ggufVariant,
- config: meta.config,
})
}
onFoldersChange={onFoldersChange}
@@ -847,14 +838,11 @@ const GeneralCompareContent = memo(function GeneralCompareContent({
loraModels={loraModels}
externalModels={externalModels}
value={model2.id}
- selectedConfig={model2.config}
- selectedGgufVariant={model2.ggufVariant}
onValueChange={(id, meta) =>
setModel2({
id,
isLora: meta.isLora,
ggufVariant: meta.ggufVariant,
- config: meta.config,
})
}
onFoldersChange={onFoldersChange}
@@ -1248,13 +1236,6 @@ export function validateChatSearch(search: Record): ChatSearch
};
}
-type PendingHubAutoLoad = {
- selection: SelectedModelInput;
- contextKey: string;
- originCheckpoint: string;
- originGgufVariant: string | null;
-};
-
// `search` comes from RootLayout (not useSearch) so ChatPage stays mounted off-route
// (keeping an in-flight generation alive), frozen to the last /chat search. `active`
// is false off-route: close body-portaled surfaces and stop route-specific listeners
@@ -1267,6 +1248,30 @@ export function ChatPage({
const settingsOpen = useChatRuntimeStore((s) => s.settingsPanelOpen);
const setSettingsOpen = useChatRuntimeStore((s) => s.setSettingsPanelOpen);
+ // Deferred-load staging: downloads a staged GGUF (if needed) and reads its
+ // header context so the sheet can show the context slider before the load.
+ // autoLoad picks instead load the cached file as soon as the download ends;
+ // selectModel is defined below, so the load runs through a ref.
+ const autoLoadStagedRef = useRef<
+ ((pending: PendingModelSelection) => void) | null
+ >(null);
+ const stagedDownload = useStagedModelPreparation({
+ onAutoLoad: (pending) => autoLoadStagedRef.current?.(pending),
+ });
+ // Abandon a staged pick: the store action cancels its in-flight download and
+ // reverts the edited knobs, so nothing lingers after the user walks away.
+ const abandonStaged = useCallback(() => {
+ useChatRuntimeStore.getState().abandonStagedModel();
+ }, []);
+ // Detach a staged pick on navigation without cancelling its download: the
+ // transfer keeps running in the manager and lands in cache, like Hub.
+ const detachStaged = useCallback(() => {
+ useChatRuntimeStore.getState().abandonStagedModel({ keepDownload: true });
+ }, []);
+ // Tracks whether the chat page is still mounted, so a staged-load failure that
+ // resolves after the user left chat doesn't resurrect the abandoned pick.
+ const mountedRef = useRef(true);
+ useEffect(() => () => void (mountedRef.current = false), []);
const incognito = useChatRuntimeStore((s) => s.incognito);
const setIncognito = useChatRuntimeStore((s) => s.setIncognito);
const incognitoLabel = incognito
@@ -1358,9 +1363,6 @@ export function ChatPage({
const ggufContextLength = useChatRuntimeStore(
(state) => state.ggufContextLength,
);
- const ggufNativeContextLength = useChatRuntimeStore(
- (state) => state.ggufNativeContextLength,
- );
const contextUsage = useChatRuntimeStore((state) => state.contextUsage);
const modelsFromStore = useChatRuntimeStore((state) => state.models);
const lorasFromStore = useChatRuntimeStore((state) => state.loras);
@@ -1438,82 +1440,37 @@ export function ChatPage({
refreshRef.current = refresh;
selectModelRef.current = selectModel;
}, [refresh, selectModel]);
- const rememberedConfigFor = useCallback(
- (selection: {
- id: string;
- ggufVariant?: string | null;
- source?: string;
- }) => {
- if (selection.source === "external") return null;
- const resolved = resolveInitialConfig(selection.id, selection.ggufVariant);
- return resolved.remembered ? resolved.config : null;
- },
- [],
- );
+ // Load a cached autoLoad pick once its download finishes. The sheet was never
+ // opened, so on a load failure just drop the orphaned staged knobs. The knobs
+ // were already seeded on stage, so keepSpeculative only when a config was
+ // saved -- otherwise the standing speculative preference should win.
+ autoLoadStagedRef.current = (pending) => {
+ const remembered = loadRememberedLoadSettings(
+ rememberedLoadSettingsKey(pending),
+ );
+ void selectModel({
+ ...pending,
+ isDownloaded: true,
+ forceReload: true,
+ keepSpeculative: remembered != null,
+ throwOnError: true,
+ }).catch(() => {
+ const store = useChatRuntimeStore.getState();
+ // selectModel only clears pendingSelection on success, so a failed
+ // auto-load leaves our staged pick (and its edited load knobs) behind.
+ // Abandon it when it is still the active stage; otherwise just revert the
+ // settings if the stage was already cleared by something else.
+ if (pendingSelectionMatches(store.pendingSelection, pending)) {
+ store.abandonStagedModel();
+ } else if (!store.pendingSelection) {
+ store.resetModelSettingsToLoaded();
+ }
+ });
+ };
const isExternalModel = useMemo(
() => isExternalModelId(inferenceParams.checkpoint),
[inferenceParams.checkpoint],
);
- const runtimeCustomContextLength = useChatRuntimeStore(
- (s) => s.customContextLength,
- );
- const runtimeKvCacheDtype = useChatRuntimeStore((s) => s.kvCacheDtype);
- const runtimeSpeculativeType = useChatRuntimeStore((s) => s.speculativeType);
- const runtimeSpecDraftNMax = useChatRuntimeStore((s) => s.specDraftNMax);
- const runtimeTensorParallel = useChatRuntimeStore((s) => s.tensorParallel);
- const runtimeChatTemplateOverride = useChatRuntimeStore(
- (s) => s.chatTemplateOverride,
- );
- const activeModelConfig = useMemo(() => {
- if (!inferenceParams.checkpoint || isExternalModel) return null;
- const activeModelIsGguf =
- activeGgufVariant != null ||
- ggufContextLength != null ||
- inferenceParams.checkpoint.toLowerCase().endsWith(".gguf");
- return {
- customContextLength: runtimeCustomContextLength ?? null,
- maxSeqLength: activeModelIsGguf ? null : inferenceParams.maxSeqLength,
- kvCacheDtype: runtimeKvCacheDtype ?? null,
- speculativeType: runtimeSpeculativeType ?? "auto",
- specDraftNMax: runtimeSpecDraftNMax ?? null,
- tensorParallel: runtimeTensorParallel ?? false,
- chatTemplateOverride: runtimeChatTemplateOverride ?? null,
- };
- }, [
- inferenceParams.checkpoint,
- inferenceParams.maxSeqLength,
- isExternalModel,
- activeGgufVariant,
- ggufContextLength,
- runtimeCustomContextLength,
- runtimeKvCacheDtype,
- runtimeSpeculativeType,
- runtimeSpecDraftNMax,
- runtimeTensorParallel,
- runtimeChatTemplateOverride,
- ]);
- const activeModelIsGguf = useMemo(() => {
- const checkpoint = inferenceParams.checkpoint;
- if (!checkpoint || isExternalModel) return false;
- return (
- activeGgufVariant != null ||
- ggufContextLength != null ||
- checkpoint.toLowerCase().endsWith(".gguf")
- );
- }, [
- inferenceParams.checkpoint,
- isExternalModel,
- activeGgufVariant,
- ggufContextLength,
- ]);
- const activeModelIsLora = useMemo(() => {
- const checkpoint = inferenceParams.checkpoint;
- if (!checkpoint || isExternalModel) return false;
- const model = modelsFromStore.find((entry) => entry.id === checkpoint);
- if (model) return model.isLora;
- const lora = lorasFromStore.find((entry) => entry.id === checkpoint);
- return lora?.exportType === "lora";
- }, [inferenceParams.checkpoint, isExternalModel, modelsFromStore, lorasFromStore]);
const reasoningEnabled = useChatRuntimeStore((s) => s.reasoningEnabled);
const reasoningStyle = useChatRuntimeStore((s) => s.reasoningStyle);
const reasoningEffort = useChatRuntimeStore((s) => s.reasoningEffort);
@@ -1826,21 +1783,75 @@ export function ChatPage({
closeArtifactSurface();
}, [activeThreadId, closeArtifactSurface, selectedArtifact, view]);
- const hasActiveModel = Boolean(inferenceParams.checkpoint);
+ // Abandon a staged (not-yet-loaded) pick when the chat context actually
+ // changes — switching threads, leaving single view, or starting a new chat /
+ // project — so a stale Load button can't resurface in a different context.
+ // New Chat keeps activeThreadId null and only bumps the `new` search nonce, so
+ // the key includes the route identity, not just the thread. Mirrors the
+ // incognito reset pattern. (Route exit is handled in __root.tsx, which runs
+ // after this unmounts.) Clear only on a real change, never on mount: staging
+ // from the Hub sets pendingSelection then navigates here, and clearing on
+ // mount would wipe it. Comparing the previous context (rather than a first-run
+ // flag) is also safe under StrictMode's double-invoke and component remounts.
const chatContextKey = `${view.mode}|${activeThreadId ?? ""}|${search.new ?? ""}|${search.project ?? ""}`;
- const [pendingHubAutoLoad, setPendingHubAutoLoad] =
- useState(null);
+ const chatContextKeyRef = useLatestRef(chatContextKey);
+ const prevChatContextRef = useRef(null);
+ useEffect(() => {
+ const prev = prevChatContextRef.current;
+ prevChatContextRef.current = chatContextKey;
+ if (prev === null || prev === chatContextKey) return;
+ detachStaged();
+ }, [chatContextKey, detachStaged]);
+
+ const hasActiveModel = Boolean(inferenceParams.checkpoint);
+ // Load immediately, or — when "Load on selection" is off — stage the pick so
+ // its load options can be set first. Shared by the main selector, native
+ // drag-drop/picker, and the dropped-file chip (the Hub stages via the store).
const stageOrLoad = useCallback(
async (selection: SelectedModelInput) => {
const store = useChatRuntimeStore.getState();
+ // An un-cached HF repo (GGUF variant or a full non-GGUF snapshot) downloads
+ // through the manager first (global indicator), then auto-loads. Everything
+ // else -- cached picks, local/native files, LoRA, external -- loads now.
const wantManagerDownload =
isDownloadableHubRepo(selection) && !selection.isDownloaded;
+ if (
+ (!hasGgufSource(selection) && !wantManagerDownload) ||
+ (store.loadOnSelection && selection.isDownloaded)
+ ) {
+ // Detach any staged pick first so its edited knobs (e.g. a custom
+ // context length) don't leak into this immediate load -- resolveLoad
+ // reads customContextLength before checking the target is GGUF. Detach
+ // (not abandon) keeps its download running.
+ detachStaged();
+ // Load-on-selection skips the sheet, so seed the saved knobs here the
+ // way the sheet's restore effect would; the switch would otherwise reset
+ // the remembered speculative choice (keepSpeculative below prevents it).
+ const remembered = hasGgufSource(selection)
+ ? loadRememberedLoadSettings(rememberedLoadSettingsKey(selection))
+ : null;
+ if (remembered) store.applyRememberedLoadSettings(remembered);
+ await selectModel(
+ remembered ? { ...selection, keepSpeculative: true } : selection,
+ );
+ return;
+ }
+ // Loads can't queue behind each other, but a download is independent: if
+ // the pick needs downloading, start it in the manager so it runs alongside
+ // the load. Nothing to download (already on device) just waits.
if (store.modelLoading) {
+ // Both an uncached non-GGUF snapshot (wantManagerDownload) and an
+ // uncached remote GGUF quant download through the manager, so either can
+ // run in the background while another model loads. wantManagerDownload
+ // excludes GGUF by design, so the GGUF case is checked separately.
const wantBackgroundDownload =
wantManagerDownload ||
(selection.source === "hub" &&
hasGgufSource(selection) &&
!selection.isDownloaded);
+ // The model currently loading already downloads as part of its own load
+ // (the /load flow fetches before setting the checkpoint), so re-picking
+ // it must not kick off a second transfer against the same cache.
const isLoadingThisPick =
!!loadingModel &&
normalizeModelRef(loadingModel.id) ===
@@ -1851,6 +1862,11 @@ export function ChatPage({
description: "It's downloading as part of the load in progress.",
});
} else if (wantBackgroundDownload) {
+ // Only claim the download started once a job is actually created. A
+ // transport conflict records state that is only resolvable from the
+ // Hub download card, so point the user there instead of showing a
+ // success toast for a transfer that never began; "busy" and "error"
+ // already surface their own toasts.
const outcome = await downloadManager.requestStart({
kind: DOWNLOAD_KIND.MODEL,
repoId: selection.id,
@@ -1867,11 +1883,6 @@ export function ChatPage({
description:
"An earlier partial download used a different transport. Open the Hub tab to resume or restart it.",
});
- } else if (outcome === "busy") {
- toast.info("Download already in progress", {
- description:
- "Another download for this model is still running. Reselect it once that finishes to load it.",
- });
}
} else {
toast.info("Another model is already loading", {
@@ -1880,118 +1891,23 @@ export function ChatPage({
}
return;
}
- const wantManagerStage =
- wantManagerDownload ||
- (selection.source === "hub" &&
- hasGgufSource(selection) &&
- !selection.isDownloaded);
- if (wantManagerStage) {
- setPendingHubAutoLoad({
- selection,
- contextKey: chatContextKey,
- originCheckpoint: store.params.checkpoint,
- originGgufVariant: store.activeGgufVariant,
- });
- return;
- }
- setPendingHubAutoLoad(null);
- const previousConfig = currentRuntimePerModelConfig({
- includeMaxSeqLength: true,
- });
- const hasAppliedConfig = applyModelLoadConfigToRuntime(
- selection.config ?? rememberedConfigFor(selection),
- );
- await selectModel({
- ...selection,
- ...(hasAppliedConfig ? { keepSpeculative: true } : {}),
- previousConfig,
+ // Detach the prior staged pick (keeping its download) before rebinding, so
+ // a second pick downloads alongside the first instead of cancelling it.
+ detachStaged();
+ store.stageModel({
+ id: selection.id,
+ isLora: selection.isLora,
+ ggufVariant: selection.ggufVariant,
+ isDownloaded: selection.isDownloaded,
+ expectedBytes: selection.expectedBytes,
+ nativePathToken: selection.nativePathToken,
+ isGguf: selection.isGguf,
+ isHubRepo: wantManagerDownload || undefined,
+ autoLoad: store.loadOnSelection,
});
},
- [selectModel, loadingModel, rememberedConfigFor, chatContextKey],
+ [detachStaged, selectModel, loadingModel],
);
- useRepoDownload({
- kind: DOWNLOAD_KIND.MODEL,
- repoId: pendingHubAutoLoad?.selection.id ?? "__hub_autoload_idle__",
- activeVariant: pendingHubAutoLoad?.selection.ggufVariant ?? null,
- onComplete: (variant) => {
- const pending = pendingHubAutoLoad;
- if (
- !pending ||
- (pending.selection.ggufVariant ?? null) !== (variant ?? null)
- ) {
- return;
- }
- setPendingHubAutoLoad(null);
- const store = useChatRuntimeStore.getState();
- if (
- !active ||
- pending.contextKey !== chatContextKey ||
- normalizeModelRef(pending.originCheckpoint) !==
- normalizeModelRef(store.params.checkpoint) ||
- pending.originGgufVariant !== store.activeGgufVariant
- ) {
- return;
- }
- void stageOrLoad({ ...pending.selection, isDownloaded: true });
- },
- onError: (variant) => {
- if (
- pendingHubAutoLoad &&
- (pendingHubAutoLoad.selection.ggufVariant ?? null) === (variant ?? null)
- ) {
- setPendingHubAutoLoad(null);
- }
- },
- onCancelled: (variant) => {
- if (
- pendingHubAutoLoad &&
- (pendingHubAutoLoad.selection.ggufVariant ?? null) === (variant ?? null)
- ) {
- setPendingHubAutoLoad(null);
- }
- },
- });
- useEffect(() => {
- const pending = pendingHubAutoLoad;
- if (!pending) return;
- let active = true;
- void (async () => {
- const outcome = await downloadManager.requestStart({
- kind: DOWNLOAD_KIND.MODEL,
- repoId: pending.selection.id,
- variant: pending.selection.ggufVariant ?? null,
- expectedBytes: pending.selection.expectedBytes ?? 0,
- });
- if (!active) return;
- if (outcome === "started") {
- toast.info("Downloading model", {
- description: "It'll load automatically once the download finishes.",
- });
- return;
- }
- if (outcome === "conflict") {
- // Keep pendingHubAutoLoad bound so this surface's cleanup does not wipe
- // the conflict just recorded by requestStart (which the toast points the
- // user to); resolving it from the Hub completes the download and this
- // surface's onComplete auto-loads, mirroring the "started" branch.
- toast.info("Resume this download from the Hub", {
- description:
- "An earlier partial download used a different transport. Open the Hub tab to resume or restart it.",
- });
- return;
- }
- if (outcome === "busy") {
- toast.info("Download already in progress", {
- description:
- "Another download for this model is still running. Reselect it once that finishes to load it.",
- });
- }
- setPendingHubAutoLoad((current) => (current === pending ? null : current));
- })();
- return () => {
- active = false;
- };
- }, [pendingHubAutoLoad]);
const loadNativeModelIntent = useCallback(
async (intent: NativeIntent, loadingDescription: string) => {
const label =
@@ -2004,11 +1920,6 @@ export function ChatPage({
forceReload: true,
throwOnError: true,
});
- // Record when this file lease expires so a later reload can prompt
- // re-selection instead of reusing a token the host has already pruned.
- useChatRuntimeStore.setState({
- activeNativePathExpiresAtMs: intent.path.expiresAtMs ?? null,
- });
useNativeIntentStore.getState().clearModelIntent(intent.id);
},
[stageOrLoad],
@@ -2054,20 +1965,28 @@ export function ChatPage({
const handleCheckpointChange = useCallback(
(
value: string,
- meta?: ModelSelectorChangeMeta,
+ meta?: {
+ source?: string;
+ isLora: boolean;
+ ggufVariant?: string;
+ isDownloaded?: boolean;
+ expectedBytes?: number;
+ isGguf?: boolean;
+ },
) => {
const store = useChatRuntimeStore.getState();
const currentCheckpoint = store.params.checkpoint;
const currentVariant = store.activeGgufVariant;
- if (!value) return;
- setPendingHubAutoLoad(null);
- const isSameLoadedModel =
- value === currentCheckpoint &&
- (meta?.ggufVariant ?? null) === (currentVariant ?? null);
- if (isSameLoadedModel && !meta?.forceReload) {
+ if (
+ !value ||
+ (value === currentCheckpoint &&
+ (meta?.ggufVariant ?? null) === (currentVariant ?? null))
+ )
return;
- }
if (meta?.source === "external" || isExternalModelId(value)) {
+ // Switching to an external model abandons any staged local pick: cancel
+ // its download too (setCheckpoint below only clears the pending + knobs).
+ abandonStaged();
const selectedExternal = parseExternalModelId(value);
const selectedProvider = selectedExternal
? externalProvidersForChat.find(
@@ -2239,17 +2158,19 @@ export function ChatPage({
source: meta?.source,
isLora: meta?.isLora,
ggufVariant: meta?.ggufVariant,
- isDownloaded: meta?.isDownloaded || isSameLoadedModel,
+ isDownloaded: meta?.isDownloaded,
expectedBytes: meta?.expectedBytes,
isGguf: meta?.isGguf,
- config: meta?.config,
- nativePathToken: meta?.nativePathToken,
- forceReload: isSameLoadedModel || undefined,
};
+ // "Load on selection" off: stage the model and open settings so its
+ // load knobs (tensor parallel, context length…) can be set, then it
+ // loads once via the sheet's Load button. The currently loaded model
+ // stays put until the user commits.
await stageOrLoad(selection);
})();
},
[
+ abandonStaged,
activeThreadId,
externalProvidersForChat,
modelsFromStore,
@@ -2257,44 +2178,6 @@ export function ChatPage({
view,
],
);
- const handleReloadActiveModel = useCallback(
- (config: PerModelConfig) => {
- const checkpoint = inferenceParams.checkpoint;
- if (!checkpoint) return;
- const runtime = useChatRuntimeStore.getState();
- const nativeToken = runtime.activeNativePathToken;
- const nativeExpiry = runtime.activeNativePathExpiresAtMs;
- // A file-picked GGUF is reachable only via its native path token, which
- // the desktop host prunes after a TTL. Reusing an expired token makes the
- // reload fail with an opaque error, so prompt the user to re-select the
- // file instead.
- if (nativeToken && nativeExpiry != null && Date.now() >= nativeExpiry) {
- toast.error("This local model file's access has expired.", {
- description: "Re-select the model file to reload it.",
- });
- return;
- }
- handleCheckpointChange(checkpoint, {
- source: "local",
- isLora: activeModelIsLora,
- ggufVariant: activeGgufVariant ?? undefined,
- // Without the native token the reload validates the display label as a
- // repo and fails.
- nativePathToken: nativeToken ?? undefined,
- isGguf: activeModelIsGguf,
- isDownloaded: true,
- config,
- forceReload: true,
- });
- },
- [
- inferenceParams.checkpoint,
- activeGgufVariant,
- activeModelIsLora,
- activeModelIsGguf,
- handleCheckpointChange,
- ],
- );
const handleEject = useCallback(() => {
void (async () => {
if (await ejectModel()) {
@@ -2697,8 +2580,6 @@ export function ChatPage({
externalModels={externalModels}
value={inferenceParams.checkpoint}
activeGgufVariant={activeGgufVariant}
- activeModelConfig={activeModelConfig}
- activeGgufContextLength={ggufContextLength}
onValueChange={handleCheckpointChange}
onEject={handleEject}
onFoldersChange={refreshLocalModels}
@@ -2752,12 +2633,7 @@ export function ChatPage({
- loadNativeModelIntent(
- pendingNativeModelIntent,
- "Loading selected local GGUF model.",
- )
- }
+ onLoad={(selection) => stageOrLoad(selection)}
/>
) : null}
{loadingModel && loadToastDismissed ? (
@@ -2914,22 +2790,13 @@ export function ChatPage({
open={active && settingsOpen}
onOpenChange={(open) => {
setSettingsOpen(open);
+ // Closing the sheet abandons a staged (not-yet-loaded) pick: cancel its
+ // download and revert the staged knobs so nothing lingers as a dirty
+ // edit (or a background download) on the loaded model.
+ if (!open) abandonStaged();
}}
params={inferenceParams}
onParamsChange={setInferenceParams}
- modelConfig={
- view.mode !== "compare" && activeModelConfig && !modelLoading ? (
-
- ) : null
- }
isExternalModel={isExternalModel}
providerCapabilities={activeProviderCapabilities}
activeExternalProvider={activeExternalProvider}
@@ -2941,6 +2808,62 @@ export function ChatPage({
);
}}
externalProviderType={activeExternalProviderType}
+ loadingModel={loadingModel}
+ onReloadModel={() => {
+ const state = useChatRuntimeStore.getState();
+ if (state.params.checkpoint) {
+ selectModel({
+ id: state.params.checkpoint,
+ ggufVariant: state.activeGgufVariant ?? undefined,
+ forceReload: true,
+ isDownloaded: true,
+ loadingDescription: "Reloading with updated chat template.",
+ });
+ }
+ }}
+ onLoadPendingModel={() => {
+ const pending = useChatRuntimeStore.getState().pendingSelection;
+ if (!pending) return;
+ const keyAtLoad = chatContextKey;
+ // forceReload: the staged model isn't loaded yet, so bypass the
+ // same-checkpoint dedupe. keepSpeculative: honor the speculative mode
+ // set on the sidebar.
+ void selectModel({
+ ...pending,
+ forceReload: true,
+ keepSpeculative: true,
+ throwOnError: true,
+ }).catch(() => {
+ // Recoverable failure (expired token, gated repo, OOM…): the pick is
+ // cleared only on success, so it normally stays staged with edited
+ // knobs intact — nothing to restore.
+ const store = useChatRuntimeStore.getState();
+ // Still staged (this pick, or a newer one queued meanwhile): leave it.
+ if (store.pendingSelection) return;
+ // Cleared mid-load (sheet closed / switched chats). Re-stage only if
+ // the staged-load is still wanted: same chat context, sheet still
+ // open, page still mounted.
+ const stillWanted =
+ mountedRef.current &&
+ store.settingsPanelOpen &&
+ chatContextKeyRef.current === keyAtLoad;
+ if (stillWanted) {
+ store.setPendingSelection(pending);
+ } else {
+ // Abandoned (closed the sheet / switched chats / left chat): drop
+ // the orphaned staged knob edits so they don't linger as dirty
+ // settings over the loaded model.
+ store.resetModelSettingsToLoaded();
+ }
+ });
+ }}
+ stagedDownloadFraction={stagedDownload.progress?.fraction ?? null}
+ onCancelStagedDownload={() =>
+ stagedDownload.cancelDownload(
+ useChatRuntimeStore.getState().pendingSelection?.ggufVariant ??
+ null,
+ )
+ }
/>
diff --git a/studio/frontend/src/features/chat/chat-settings-sheet.tsx b/studio/frontend/src/features/chat/chat-settings-sheet.tsx
index a5768024b5..cedd298ecf 100644
--- a/studio/frontend/src/features/chat/chat-settings-sheet.tsx
+++ b/studio/frontend/src/features/chat/chat-settings-sheet.tsx
@@ -1,7 +1,19 @@
// SPDX-License-Identifier: AGPL-3.0-only
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+import {
+ Alert,
+ AlertDescription,
+ AlertTitle,
+} from "@/components/ui/alert";
import { Button } from "@/components/ui/button";
+import { Checkbox } from "@/components/ui/checkbox";
+import {
+ clearRememberedLoadSettings,
+ loadRememberedLoadSettings,
+ rememberedLoadSettingsKey,
+ saveRememberedLoadSettings,
+} from "@/components/assistant-ui/model-selector/remembered-load-settings";
import {
Dialog,
DialogContent,
@@ -17,7 +29,7 @@ import {
DropdownMenuSeparator,
DropdownMenuTrigger,
} from "@/components/ui/dropdown-menu";
-import { InfoHint } from "@/components/ui/info-hint";
+import { Input } from "@/components/ui/input";
import {
InputGroup,
InputGroupAddon,
@@ -38,22 +50,26 @@ import {
SheetTitle,
} from "@/components/ui/sheet";
import { Slider } from "@/components/ui/slider";
+import { Spinner } from "@/components/ui/spinner";
import { Switch } from "@/components/ui/switch";
import { Textarea } from "@/components/ui/textarea";
+import { InfoHint } from "@/components/ui/info-hint";
import { Tooltip, TooltipContent } from "@/components/ui/tooltip";
-import { NumericValueInput, snapToStep } from "@/features/model-picker";
-import { RetrievalSettingsSection } from "@/features/rag";
-import { useLlamaUpdateCheck } from "@/hooks/use-llama-update-check";
import { useIsMobile } from "@/hooks/use-mobile";
-import { ChevronDownStandardIcon } from "@/lib/chevron-icons";
-import { toast } from "@/lib/toast";
+import { useLlamaUpdateCheck } from "@/hooks/use-llama-update-check";
import { cn } from "@/lib/utils";
-import { Edit03Icon, LayoutAlignRightIcon } from "@hugeicons/core-free-icons";
+import {
+ ArrowTurnBackwardIcon,
+ Edit03Icon,
+ LayoutAlignRightIcon,
+} from "@hugeicons/core-free-icons";
+import { ChevronDownStandardIcon } from "@/lib/chevron-icons";
import { HugeiconsIcon } from "@hugeicons/react";
import { Braces, ChevronDown, ExternalLink } from "lucide-react";
import { Tooltip as TooltipPrimitive } from "radix-ui";
import { Fragment, type ReactNode } from "react";
import { useCallback, useEffect, useMemo, useRef, useState } from "react";
+import { toast } from "@/lib/toast";
import { OpenAICodeExecSection } from "./components/openai-code-exec-section";
import { PermissionModeDropdown } from "./permission-mode-select";
import { resyncInferenceStatusAfterServerModelChange } from "./hooks/use-chat-model-runtime";
@@ -61,8 +77,8 @@ import {
type ExternalProviderConfig,
getExternalProviderApiKey,
parseExternalModelId,
- supportsProviderPromptCacheTtl,
supportsProviderPromptCaching,
+ supportsProviderPromptCacheTtl,
} from "./external-providers";
import {
BUILTIN_PRESETS,
@@ -82,7 +98,12 @@ import {
providerSupportsBuiltinCodeExecution,
providerSupportsFastMode,
} from "./provider-capabilities";
-import { useChatRuntimeStore } from "./stores/chat-runtime-store";
+import {
+ isPendingGguf,
+ pendingSelectionMatches,
+ useChatRuntimeStore,
+} from "./stores/chat-runtime-store";
+import { RetrievalSettingsSection } from "@/features/rag/components/retrieval-settings-section";
import type { InferenceParams } from "./types/runtime";
export { defaultInferenceParams, type Preset } from "./presets/preset-policy";
@@ -105,7 +126,7 @@ function getPromptVariablesError(raw: string): string | null {
return null;
}
} catch {
- return 'Use valid JSON, for example { "env": "staging" }.';
+ return "Use valid JSON, for example { \"env\": \"staging\" }.";
}
return "Variables must be a JSON object.";
}
@@ -114,6 +135,111 @@ function hasPromptVariableSyntax(prompt: string): boolean {
return PROMPT_VARIABLE_PATTERN.test(prompt);
}
+/**
+ * Editable numeric value display, shared by every slider value and the Context
+ * Length input. An that looks like text (shows `displayValue ?? value`,
+ * so "Off"/"Max" labels render) until focus, when it swaps to the raw number,
+ * selects it, and accepts free text. Commits on blur/Enter, reverts on Escape.
+ * Clamping happens on commit so typing intermediate values isn't fought.
+ */
+function snapToStep(
+ value: number,
+ step: number,
+ min?: number,
+ max?: number,
+): number {
+ const lo = min ?? Number.NEGATIVE_INFINITY;
+ const hi = max ?? Number.POSITIVE_INFINITY;
+ const clamped = Math.min(Math.max(value, lo), hi);
+ const stepStr = String(step);
+ const decimals = stepStr.includes(".") ? stepStr.split(".")[1].length : 0;
+ const base = Number.isFinite(lo) ? lo : 0;
+ const snapped = base + Math.round((clamped - base) / step) * step;
+ const reclamped = Math.min(Math.max(snapped, lo), hi);
+ return Number(reclamped.toFixed(decimals));
+}
+
+function NumericValueInput({
+ value,
+ min,
+ max,
+ step,
+ onChange,
+ displayValue,
+ className,
+ ariaLabel,
+ size: sizeAttr,
+ disabled = false,
+}: {
+ value: number;
+ min?: number;
+ max?: number;
+ step: number;
+ onChange: (v: number) => void;
+ displayValue?: string;
+ className?: string;
+ ariaLabel?: string;
+ size?: number;
+ disabled?: boolean;
+}) {
+ const [focused, setFocused] = useState(false);
+ const [draft, setDraft] = useState("");
+ const cancelBlurCommitRef = useRef(false);
+
+ const commit = (raw: string) => {
+ const parsed = Number.parseFloat(raw);
+ if (!Number.isFinite(parsed)) {
+ return;
+ }
+ const final = snapToStep(parsed, step, min, max);
+ if (final !== value) {
+ onChange(final);
+ }
+ };
+
+ const displayed = focused ? draft : (displayValue ?? String(value));
+
+ return (
+ {
+ cancelBlurCommitRef.current = false;
+ setDraft(String(value));
+ setFocused(true);
+ // Defer select() so it runs after the value swap above.
+ const target = e.currentTarget;
+ requestAnimationFrame(() => target.select());
+ }}
+ onBlur={() => {
+ if (cancelBlurCommitRef.current) {
+ cancelBlurCommitRef.current = false;
+ } else {
+ commit(draft);
+ }
+ setFocused(false);
+ }}
+ onChange={(e) => setDraft(e.target.value)}
+ onKeyDown={(e) => {
+ if (e.key === "Enter") {
+ e.currentTarget.blur();
+ } else if (e.key === "Escape") {
+ cancelBlurCommitRef.current = true;
+ setDraft(String(value));
+ e.currentTarget.blur();
+ }
+ }}
+ className={cn("panel-number-input", className)}
+ />
+ );
+}
+
function ParamSlider({
label,
value,
@@ -153,7 +279,6 @@ function ParamSlider({
displayValue={displayValue}
ariaLabel={label}
size={valueSize ?? 4}
- className="panel-number-input"
/>
{labelHref ? (
@@ -324,7 +450,6 @@ interface ChatSettingsPanelProps {
onOpenChange?: (open: boolean) => void;
params: InferenceParams;
onParamsChange: (params: InferenceParams) => void;
- modelConfig?: ReactNode;
isExternalModel?: boolean;
/**
* Sampling-param capabilities for the active external provider, or `null` for
@@ -339,6 +464,21 @@ interface ChatSettingsPanelProps {
* Max Tokens floor in the slider.
*/
externalProviderType?: string | null;
+ onReloadModel?: () => void;
+ /** The in-flight load (id + GGUF variant + native path token), or null when
+ * idle. Used to show a loading state for the staged pick only — not for an
+ * unrelated load or a cancel's background unload. */
+ loadingModel?: {
+ id: string;
+ ggufVariant?: string | null;
+ nativePathToken?: string | null;
+ } | null;
+ /** Loads the staged `pendingSelection` (deferred "Load on selection" flow). */
+ onLoadPendingModel?: () => void;
+ /** Download progress (0–1) for a staged GGUF being fetched, or null when idle. */
+ stagedDownloadFraction?: number | null;
+ /** Cancels the in-flight staged download (paired with abandoning the stage). */
+ onCancelStagedDownload?: () => void;
}
export function ChatSettingsPanel({
@@ -346,12 +486,16 @@ export function ChatSettingsPanel({
onOpenChange,
params,
onParamsChange,
- modelConfig = null,
isExternalModel = false,
providerCapabilities = null,
activeExternalProvider = null,
onExternalProviderChange,
externalProviderType = null,
+ onReloadModel,
+ loadingModel = null,
+ onLoadPendingModel,
+ stagedDownloadFraction,
+ onCancelStagedDownload,
}: ChatSettingsPanelProps) {
// Local models show every knob; providerCapabilities is only consulted when
// isExternalModel. Unknown providers fall back to the OpenAI-compat shape via
@@ -366,23 +510,55 @@ export function ChatSettingsPanel({
const showPresencePenalty =
!isExternalModel || Boolean(providerCapabilities?.presencePenalty);
const isMobile = useIsMobile();
- const isLoadedGguf = useChatRuntimeStore((s) => s.activeGgufVariant) != null;
- const currentCheckpoint = params.checkpoint;
- const ggufContextLength = useChatRuntimeStore((s) => s.ggufContextLength);
- // Direct-file / custom-folder GGUFs load without a variant label but still
- // report a GGUF context, so detect them via the context and the checkpoint
- // suffix too (mirrors the chat page's activeModelIsGguf). Otherwise Max Tokens
- // would fall back to params.maxSeqLength instead of the loaded GGUF context.
- const isGguf =
- isLoadedGguf ||
- ggufContextLength != null ||
- (currentCheckpoint?.toLowerCase().endsWith(".gguf") ?? false);
- const ggufMaxContextLength = useChatRuntimeStore(
- (s) => s.ggufMaxContextLength,
+ const pendingSelection = useChatRuntimeStore((s) => s.pendingSelection);
+ // "Loading" only when the in-flight load IS this staged pick (full id + GGUF
+ // variant + native token match), not an unrelated load or a cancel's
+ // background unload. The variant matters: a different quant of the same repo
+ // staged mid-load must not read as this one loading.
+ const stagedLoading =
+ loadingModel != null &&
+ pendingSelectionMatches(pendingSelection, {
+ id: loadingModel.id,
+ ggufVariant: loadingModel.ggufVariant,
+ nativePathToken: loadingModel.nativePathToken,
+ });
+ // Load settings are snapshotted at click time; lock them while loading.
+ const modelControlsDisabled = stagedLoading;
+ const abandonStagedModel = useChatRuntimeStore((s) => s.abandonStagedModel);
+ const resetModelSettingsToLoaded = useChatRuntimeStore(
+ (s) => s.resetModelSettingsToLoaded,
);
- const customContextLength = useChatRuntimeStore((s) => s.customContextLength);
+ // A staged GGUF pick (deferred load) shows the GGUF load knobs so they can be
+ // set before the single load.
+ const pendingIsGguf = isPendingGguf(pendingSelection);
+ // Short, human-readable name for the staged pick (HF ids carry an org prefix;
+ // native picks are already a display label). Drives the "staged, not loaded"
+ // callout so it's obvious the selection hasn't loaded yet.
+ const stagedLabel = (() => {
+ const id = pendingSelection?.id ?? "";
+ const slash = id.lastIndexOf("/");
+ const base = slash >= 0 ? id.slice(slash + 1) : id;
+ return base || id;
+ })();
+ const isLoadedGguf =
+ useChatRuntimeStore((s) => s.activeGgufVariant) != null;
+ // While a pick is staged the sheet configures *that* model, so its GGUF-ness
+ // (not the currently loaded model's) decides whether the GGUF-only controls
+ // show. Otherwise a staged non-GGUF Hub repo would inherit the loaded GGUF's
+ // context/KV/speculative controls.
+ const isGguf = pendingSelection != null ? pendingIsGguf : isLoadedGguf;
+ // The Model section (and Load button) shows for any staged pick, even when the
+ // currently active model is external.
+ const hasModelContent =
+ pendingSelection != null ||
+ (!isExternalModel && (isGguf || Boolean(params.checkpoint)));
const speculativeType = useChatRuntimeStore((s) => s.speculativeType);
+ const setSpeculativeType = useChatRuntimeStore((s) => s.setSpeculativeType);
+ const loadedSpeculativeType = useChatRuntimeStore(
+ (s) => s.loadedSpeculativeType,
+ );
const specFallbackReason = useChatRuntimeStore((s) => s.specFallbackReason);
+ // Only binary fallback states are solved by a newer prebuilt.
const mtpUpdatable =
specFallbackReason === "binary_no_mtp" ||
specFallbackReason === "binary_outdated";
@@ -404,27 +580,43 @@ export function ChatSettingsPanel({
`llama.cpp updated to ${result.tag ?? "the latest build"}.${reloadHint}`,
);
} else {
- toast.error(
- `llama.cpp update failed: ${result.error ?? "unknown error"}`,
- );
+ toast.error(`llama.cpp update failed: ${result.error ?? "unknown error"}`);
}
}, [applyLlamaUpdate]);
- const loadedEffectiveContext = customContextLength ?? ggufContextLength;
- const showSpecFallback =
- !isExternalModel &&
- isLoadedGguf &&
- specFallbackReason != null &&
- (speculativeType === "auto" ||
- speculativeType === "mtp" ||
- speculativeType === "mtp+ngram");
- const showContextVramWarning =
- !isExternalModel &&
- isLoadedGguf &&
- ggufMaxContextLength != null &&
- loadedEffectiveContext != null &&
- loadedEffectiveContext > ggufMaxContextLength;
- const showLoadedDiagnostics = showSpecFallback || showContextVramWarning;
- const hasModelContent = showLoadedDiagnostics;
+ const specDraftNMax = useChatRuntimeStore((s) => s.specDraftNMax);
+ const setSpecDraftNMax = useChatRuntimeStore((s) => s.setSpecDraftNMax);
+ const loadedSpecDraftNMax = useChatRuntimeStore(
+ (s) => s.loadedSpecDraftNMax,
+ );
+ const currentCheckpoint = params.checkpoint;
+ const ggufContextLength = useChatRuntimeStore((s) => s.ggufContextLength);
+ const ggufMaxContextLength = useChatRuntimeStore(
+ (s) => s.ggufMaxContextLength,
+ );
+ const ggufNativeContextLength = useChatRuntimeStore(
+ (s) => s.ggufNativeContextLength,
+ );
+ const kvCacheDtype = useChatRuntimeStore((s) => s.kvCacheDtype);
+ const setKvCacheDtype = useChatRuntimeStore((s) => s.setKvCacheDtype);
+ const applyRememberedLoadSettings = useChatRuntimeStore(
+ (s) => s.applyRememberedLoadSettings,
+ );
+ const loadedKvCacheDtype = useChatRuntimeStore((s) => s.loadedKvCacheDtype);
+ const tensorParallel = useChatRuntimeStore((s) => s.tensorParallel);
+ const setTensorParallel = useChatRuntimeStore((s) => s.setTensorParallel);
+ const loadedTensorParallel = useChatRuntimeStore(
+ (s) => s.loadedTensorParallel,
+ );
+ const chatTemplateOverride = useChatRuntimeStore(
+ (s) => s.chatTemplateOverride,
+ );
+ const loadedChatTemplateOverride = useChatRuntimeStore(
+ (s) => s.loadedChatTemplateOverride,
+ );
+ const customContextLength = useChatRuntimeStore((s) => s.customContextLength);
+ const setCustomContextLength = useChatRuntimeStore(
+ (s) => s.setCustomContextLength,
+ );
const setActivePresetSource = useChatRuntimeStore(
(s) => s.setActivePresetSource,
);
@@ -435,7 +627,49 @@ export function ChatSettingsPanel({
const setActivePreset = useChatRuntimeStore((s) => s.setActivePreset);
const settingsHydrated = useChatRuntimeStore((s) => s.settingsHydrated);
- const baseContext = ggufContextLength;
+ // A staged (not-yet-loaded) GGUF carries its own header context length on
+ // pendingSelection, so the slider can use the staged model's real ceiling
+ // without reading the loaded model's `ggufContextLength`.
+ const stagedContextLength = pendingSelection?.contextLength ?? null;
+ // "Remember settings next time" tick for a staged model. Seeds the store from
+ // the saved per-model settings on stage, so the sheet opens with what was used
+ // last time; the tick reflects whether a saved entry exists.
+ const [remember, setRemember] = useState(false);
+ // Keyed per quant: a different variant of the same repo has its own settings.
+ const pendingKey = pendingSelection
+ ? rememberedLoadSettingsKey(pendingSelection)
+ : null;
+ useEffect(() => {
+ if (!pendingKey) return;
+ const saved = loadRememberedLoadSettings(pendingKey);
+ setRemember(saved != null);
+ if (saved) applyRememberedLoadSettings(saved);
+ }, [pendingKey, applyRememberedLoadSettings]);
+ // While staging, the sheet reflects the STAGED model, so its header context
+ // takes precedence over the loaded model's (which may differ or be larger).
+ const baseContext = pendingIsGguf ? stagedContextLength : ggufContextLength;
+ const baseNativeContext = pendingIsGguf
+ ? stagedContextLength
+ : ggufNativeContextLength;
+ // Context controls render once we actually have a ceiling: for a staged GGUF,
+ // once its header metadata arrives (post-download); otherwise post-load.
+ const showContextControl = pendingIsGguf
+ ? stagedContextLength != null
+ : isLoadedGguf;
+ const stagedDownloading =
+ stagedDownloadFraction != null && stagedDownloadFraction < 1;
+ const ctxDisplayValue = customContextLength ?? baseContext ?? "";
+ const ctxMaxValue = baseNativeContext ?? baseContext ?? null;
+ const kvDirty = kvCacheDtype !== loadedKvCacheDtype;
+ const ctxDirty = customContextLength !== null;
+ const specDirty = speculativeType !== loadedSpeculativeType;
+ const specDraftDirty = specDraftNMax !== loadedSpecDraftNMax;
+ const tpDirty = tensorParallel !== (loadedTensorParallel ?? false);
+ // A saved chat-template override is a reload-time setting too, so surface
+ // Apply for a template-only edit (otherwise it could never be applied).
+ const templateDirty = chatTemplateOverride !== loadedChatTemplateOverride;
+ const modelSettingsDirty =
+ kvDirty || ctxDirty || specDirty || specDraftDirty || tpDirty || templateDirty;
const [presetNameInput, setPresetNameInput] = useState(activePreset);
const [systemPromptEditorOpen, setSystemPromptEditorOpen] = useState(false);
const [systemPromptDraft, setSystemPromptDraft] = useState("");
@@ -461,7 +695,8 @@ export function ChatSettingsPanel({
BUILTIN_PRESETS.find((preset) => preset.name === activePreset) ?? null,
[activePreset],
);
- const hasUnsavedPresetChanges = useMemo(() => {
+ const hasUnsavedPresetChanges = useMemo(
+ () => {
if (activePresetDefinition == null) {
return false;
}
@@ -469,7 +704,9 @@ export function ChatSettingsPanel({
return activePresetSource === "modified";
}
return !isSamePresetConfig(activePresetDefinition.params, params);
- }, [activePresetDefinition, activePresetSource, params]);
+ },
+ [activePresetDefinition, activePresetSource, params],
+ );
const presetSaveState = useMemo(
() =>
getPresetSaveState({
@@ -498,14 +735,6 @@ export function ChatSettingsPanel({
const externalSelection = currentCheckpoint
? parseExternalModelId(currentCheckpoint)
: null;
- const maxTokensMax = isExternalModel
- ? getExternalMaxOutputTokens(
- externalProviderType,
- externalSelection?.modelId,
- )
- : isGguf && baseContext
- ? baseContext
- : Math.max(64, params.maxSeqLength);
const showOpenAICodeExecSection =
activeExternalProvider != null &&
providerSupportsBuiltinCodeExecution(
@@ -588,7 +817,8 @@ export function ChatSettingsPanel({
return;
}
const fallbackPreset =
- BUILTIN_PRESETS.find((preset) => preset.name === "Default") ?? null;
+ BUILTIN_PRESETS.find((preset) => preset.name === "Default") ??
+ null;
const next = customPresets.filter((preset) => preset.name !== name);
setCustomPresets(next);
if (activePreset === name) {
@@ -700,7 +930,7 @@ export function ChatSettingsPanel({
Run settings
-
+ Browse
@@ -308,8 +301,8 @@ export function ExportRunPanel(props: ExportRunPanelProps) {
<>Default: {defaultSaveDirectory}>
) : (
<>
- Paste an absolute path if the folder browser cannot reach
- the drive.
+ Paste an absolute path if the folder browser cannot reach the
+ drive.
>
)}
{/* Format already shows as the status dot, so the pill stays neutral. */}
{formatLabel && {formatLabel}}
- {paramLabel && (
- {paramLabel}
- )}
+ {paramLabel && {paramLabel}}
{quantLabel && (
{quantLabel}
@@ -704,7 +697,9 @@ export const InventoryRow = memo(function InventoryRow({
const compactMarkers =
partialRepoId || unsupported ? (
- {partialRepoId && }
+ {partialRepoId && (
+
+ )}
{unsupported && (
)}
diff --git a/studio/frontend/src/features/hub/catalog/on-device-folders-dialog.tsx b/studio/frontend/src/features/hub/catalog/on-device-folders-dialog.tsx
index 86108bbd80..caa89db196 100644
--- a/studio/frontend/src/features/hub/catalog/on-device-folders-dialog.tsx
+++ b/studio/frontend/src/features/hub/catalog/on-device-folders-dialog.tsx
@@ -1,6 +1,7 @@
// SPDX-License-Identifier: AGPL-3.0-only
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+import { FolderBrowser } from "@/components/assistant-ui/model-selector/folder-browser";
import { Button } from "@/components/ui/button";
import {
Dialog,
@@ -21,11 +22,9 @@ import {
addScanFolder,
listScanFolders,
removeScanFolder,
-} from "@/features/hub";
-import { FolderBrowser } from "@/features/model-picker";
-import { openModelsDir } from "@/features/native-intents";
+} from "@/features/hub/inventory";
+import { openModelsDir } from "@/features/native-intents/api";
import { isTauri } from "@/lib/api-base";
-import { toast } from "@/lib/toast";
import { cn } from "@/lib/utils";
import {
Delete02Icon,
@@ -39,6 +38,7 @@ import {
} from "@hugeicons/core-free-icons";
import { HugeiconsIcon } from "@hugeicons/react";
import { useCallback, useEffect, useMemo, useRef, useState } from "react";
+import { toast } from "@/lib/toast";
function pathTail(path: string): string {
const parts = path.split(/[\\/]/).filter(Boolean);
@@ -123,9 +123,7 @@ export function OnDeviceFoldersDialog({
setPath("");
mutationVersionRef.current += 1;
setFolders((current) => {
- const withoutDuplicate = current.filter(
- (row) => row.id !== folder.id,
- );
+ const withoutDuplicate = current.filter((row) => row.id !== folder.id);
return [...withoutDuplicate, folder];
});
toast.success("Location added", {
@@ -186,12 +184,9 @@ export function OnDeviceFoldersDialog({
overlayClassName="bg-black/20 backdrop-blur-none"
>
-
- On-device locations
-
+ On-device locations
- Hugging Face model folders, GGUF files, and adapters are indexed
- here.
+ Hugging Face model folders, GGUF files, and adapters are indexed here.
@@ -347,7 +342,9 @@ export function OnDeviceFoldersDialog({
-
+
{folder.path}
@@ -375,10 +372,7 @@ export function OnDeviceFoldersDialog({
/>
-
+
Open in file manager
@@ -403,10 +397,7 @@ export function OnDeviceFoldersDialog({
)}
-
+
Remove from list
diff --git a/studio/frontend/src/features/hub/download-manager/download-manager-controller.ts b/studio/frontend/src/features/hub/download-manager/download-manager-controller.ts
index 2cea1d8304..da653ddeb7 100644
--- a/studio/frontend/src/features/hub/download-manager/download-manager-controller.ts
+++ b/studio/frontend/src/features/hub/download-manager/download-manager-controller.ts
@@ -1,10 +1,14 @@
// SPDX-License-Identifier: AGPL-3.0-only
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+import { DOWNLOAD_KIND } from "./constants";
import {
createDownloadManagerInitialState,
+ jobKeyOf,
removeJob,
+ selectActiveJob,
setState,
+ useDownloadManagerStore,
} from "./download-manager-state";
import { resetDownloadApiAdapterState } from "./download-api-adapter";
import {
@@ -65,6 +69,25 @@ export const downloadManager: DownloadManagerController = {
dismiss: removeJob,
};
+/** Cancel the in-flight download for a staged model pick. No-op when nothing is
+ * downloading (e.g. a native/local file that was never fetched). Lets non-React
+ * callers (the chat store's abandon paths) stop a staged transfer without the
+ * useRepoDownload hook. */
+export function cancelStagedModelDownload(
+ pending: { id: string; ggufVariant?: string | null } | null,
+): void {
+ if (!pending) return;
+ const variant = pending.ggufVariant ?? null;
+ const activeJob = selectActiveJob(
+ useDownloadManagerStore.getState(),
+ DOWNLOAD_KIND.MODEL,
+ pending.id,
+ variant,
+ );
+ void downloadManager.cancel(
+ activeJob?.key ?? jobKeyOf(DOWNLOAD_KIND.MODEL, pending.id, variant),
+ );
+}
if (import.meta.hot) {
import.meta.hot.dispose(() => {
diff --git a/studio/frontend/src/features/hub/download-manager/index.ts b/studio/frontend/src/features/hub/download-manager/index.ts
index 60ef3851f8..dd88aaf3f0 100644
--- a/studio/frontend/src/features/hub/download-manager/index.ts
+++ b/studio/frontend/src/features/hub/download-manager/index.ts
@@ -20,6 +20,7 @@ export {
} from "./constants";
export {
__resetDownloadManagerForTests,
+ cancelStagedModelDownload,
clearCompletedInventoryHint,
downloadManager,
hydrateDownloadManager,
diff --git a/studio/frontend/src/features/hub/hub-page.tsx b/studio/frontend/src/features/hub/hub-page.tsx
index 222a74322e..630daa48ad 100644
--- a/studio/frontend/src/features/hub/hub-page.tsx
+++ b/studio/frontend/src/features/hub/hub-page.tsx
@@ -1,32 +1,34 @@
// SPDX-License-Identifier: AGPL-3.0-only
// Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
+import {
+ loadRememberedLoadSettings,
+ rememberedLoadSettingsKey,
+} from "@/components/assistant-ui/model-selector/remembered-load-settings";
+import { hfModelFitsDevice } from "@/components/assistant-ui/model-selector/recommended-fit";
+import { useHubInventory } from "@/features/hub/inventory";
+import { useDebouncedValue } from "@/hooks/use-debounced-value";
+import { useGpuInfo } from "@/hooks/use-gpu-info";
+import {
+ type HfModelSearchChannel,
+ type HfSortDirection,
+ type HfSortKey,
+} from "@/features/hub/hooks/use-hub-model-search";
+import { useOnlineStatus } from "@/features/hub/hooks/use-online-status";
+import { useHubInfiniteScroll } from "@/features/hub/hooks/use-hub-infinite-scroll";
+import { ggufVariantsMatch, modelIdsMatch } from "@/features/hub/lib/model-identity";
+import { cn } from "@/lib/utils";
import { usePlatformStore } from "@/config/env";
+import {
+ hfApiToken,
+ useHfTokenStore,
+} from "@/features/hub/stores/hf-token-store";
import {
getInferenceStatus,
isExternalModelId,
useChatModelRuntime,
useChatRuntimeStore,
} from "@/features/chat";
-import { useHubInventory } from "@/features/hub";
-import type {
- HfModelSearchChannel,
- HfSortDirection,
- HfSortKey,
-} from "@/features/hub";
-import { useOnlineStatus } from "@/features/hub";
-import { useHubInfiniteScroll } from "@/features/hub";
-import { ggufVariantsMatch, modelIdsMatch } from "@/features/hub";
-import { hfApiToken, useHfTokenStore } from "@/features/hub";
-import {
- applyModelLoadConfigToRuntime,
- currentRuntimePerModelConfig,
- hfModelFitsDevice,
- resolveInitialConfig,
-} from "@/features/model-picker";
-import { useDebouncedValue } from "@/hooks/use-debounced-value";
-import { useGpuInfo } from "@/hooks/use-gpu-info";
-import { cn } from "@/lib/utils";
import { useNavigate, useSearch } from "@tanstack/react-router";
import {
useCallback,
@@ -36,17 +38,10 @@ import {
useRef,
useState,
} from "react";
-import { ExternalLinkConfirmDialog } from "./catalog/external-link-confirm-dialog";
import { HubDetailView } from "./catalog/hub-detail-view";
-import { HubFeed } from "./catalog/hub-feed";
import { HubTopBar } from "./catalog/hub-top-bar";
-import {
- ModelsCatalog,
- type ModelsCatalogHandlers,
- type ModelsCatalogPagination,
- type ModelsCatalogState,
-} from "./catalog/models-catalog";
-import { ModelsHeader } from "./catalog/models-header";
+import { HubFeed } from "./catalog/hub-feed";
+import { OwnerScopeToggle } from "./catalog/owner-scope-toggle";
import {
type AllModelsView,
HubListHeader,
@@ -54,9 +49,16 @@ import {
InventorySortControl,
ResultListHeader,
} from "./catalog/models-table";
+import {
+ ModelsCatalog,
+ type ModelsCatalogHandlers,
+ type ModelsCatalogPagination,
+ type ModelsCatalogState,
+} from "./catalog/models-catalog";
+import { ModelsHeader } from "./catalog/models-header";
import { ModelsToolbar } from "./catalog/models-toolbar";
+import { ExternalLinkConfirmDialog } from "./catalog/external-link-confirm-dialog";
import { OnDeviceFoldersDialog } from "./catalog/on-device-folders-dialog";
-import { OwnerScopeToggle } from "./catalog/owner-scope-toggle";
import { useDiscoverSearch } from "./hooks/use-discover-search";
import { useFeedWriteBack } from "./hooks/use-feed-write-back";
import { useHubFeed } from "./hooks/use-hub-feed";
@@ -566,15 +568,15 @@ export function ModelsPage() {
const deferredCapabilityFilter = useDeferredValue(capabilityFilter);
const hasQuery = deferredDebouncedQuery.trim() !== "";
- const mode: DiscoverMode = isModelDiscover
- ? hasQuery
+ const mode: DiscoverMode = !isModelDiscover
+ ? "search"
+ : hasQuery
? "search"
: urlSection != null
? "channel-list"
: sortBrowseActive
? "search"
- : "feed"
- : "search";
+ : "feed";
const isFeedMode = mode === "feed";
const isChannelListMode = mode === "channel-list";
const isSortBrowseMode =
@@ -765,10 +767,7 @@ export function ModelsPage() {
}
return merged;
}, [isFeedMode, feedTrendingRows, filteredDiscoverRows]);
- const feedResults = useMemo(
- () => feedRows.map((row) => row.result),
- [feedRows],
- );
+ const feedResults = useMemo(() => feedRows.map((row) => row.result), [feedRows]);
const selectionDiscoverRows = isFeedMode ? feedRows : discoverRows;
const selectionFilteredDiscoverRows = isFeedMode
? feedRows
@@ -1109,22 +1108,50 @@ export function ModelsPage() {
(opts: ModelLoadOptions, isDownloaded: boolean) => {
if (!selectedModel) return;
const runId = selectedModel.resource.runId;
- const resolvedConfig = resolveInitialConfig(runId, opts.ggufVariant);
- const rememberedConfig = resolvedConfig.remembered
- ? resolvedConfig.config
- : null;
- const previousConfig = currentRuntimePerModelConfig({
- includeMaxSeqLength: true,
- });
- const hasAppliedConfig = applyModelLoadConfigToRuntime(rememberedConfig);
+ // "Load on selection" off: stage GGUF picks instead of loading, so the
+ // chat page's staging flow can read the header and show the load options.
+ // Non-GGUF models have nothing to configure pre-load, so they load now.
+ if (
+ !useChatRuntimeStore.getState().loadOnSelection &&
+ (opts.ggufVariant != null || selectedModel.isGguf)
+ ) {
+ useChatRuntimeStore.getState().stageModel({
+ id: runId,
+ ggufVariant: opts.ggufVariant,
+ isGguf: selectedModel.isGguf,
+ isDownloaded,
+ expectedBytes: opts.expectedBytes,
+ });
+ openNewChat();
+ return;
+ }
+ // Detach any leftover staged pick first so its edited knobs (e.g. a custom
+ // context length) don't leak into this load -- mirrors the chat page's
+ // detachStaged(); keepDownload keeps any staged download running.
+ useChatRuntimeStore.getState().abandonStagedModel({ keepDownload: true });
+ // Load-on-selection skips the chat sheet, so seed this GGUF pick's saved
+ // load knobs here the way the sheet's restore effect would; otherwise the
+ // remembered config is silently ignored on the Hub run path. keepSpeculative
+ // then honors the restored speculative choice across the switch.
+ const remembered =
+ opts.ggufVariant != null || selectedModel.isGguf
+ ? loadRememberedLoadSettings(
+ rememberedLoadSettingsKey({
+ id: runId,
+ ggufVariant: opts.ggufVariant,
+ }),
+ )
+ : null;
+ if (remembered) {
+ useChatRuntimeStore.getState().applyRememberedLoadSettings(remembered);
+ }
void selectModel({
id: runId,
ggufVariant: opts.ggufVariant,
isDownloaded,
expectedBytes: opts.expectedBytes,
- keepSpeculative: hasAppliedConfig,
+ keepSpeculative: remembered != null,
throwOnError: true,
- previousConfig,
})
.then(() => {
// Read fresh: the load is async, so the checkpoint may have changed.
@@ -1348,18 +1375,16 @@ export function ModelsPage() {
);
}
- const ownerToggle = isDatasetMode ? undefined : (
+ const ownerToggle = !isDatasetMode ? (
- );
+ ) : undefined;
// Compact pill so it stays beside the view-mode tabs even in the narrow
// split pane instead of dropping to its own row.
return (
- Chat Template
-
- {readOnly
- ? "Preview the model's chat template. Custom overrides apply to GGUF models for now."
- : "Override the model's chat template with custom Jinja. Applies when the model loads."}
-
-
- KV Cache Dtype
-
- Lower KV cache precision to save VRAM at the cost of some quality.
- f16/bf16 are full precision; q8_0/q5_1/q4_1 are quantized.
-
-
-
-
-
-
-
- Speculative Decoding
-
- Faster generation with no accuracy hit. Auto picks MTP / ngram based
- on the model and platform. Pick a strategy to force it.
-
-
-
-
-
- {isMtp && (
-
-
- Draft Tokens
-
- Max MTP draft tokens per step. Leave blank for the platform
- default (2 on GPU, 3 on CPU/Mac).
-
-