unsloth/studio/backend/utils/paths/storage_roots.py
Roland Tannous 562e54fc6e
Fix HF cache default and show LM Studio models in chat/inference (#4653)
* fix: default HF cache to standard platform path instead of legacy Unsloth cache

* feat: show LM Studio and local models in chat Fine-tuned tab

* feat: show LM Studio models in Hub models tab

* fix: fetch local models after auth refresh completes

* Revert "fix: fetch local models after auth refresh completes"

This reverts commit cfd61f0ac7.

* fix: increase llama-server health check timeout to 600s for large models

* feat: expandable GGUF variant picker for LM Studio local models

* fix: show GGUF variant label for locally loaded LM Studio models

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* fix: show publisher name in LM Studio model labels

* fix: set model_id for loose GGUF files in LM Studio publisher dirs

* fix: show publisher prefix in Fine-tuned tab LM Studio models

* fix: only use model_id for lmstudio source models

* fix: only show LM Studio models in Hub tab on Mac/chat-only mode

* fix: respect XDG_CACHE_HOME, handle Windows paths in isLocalPath, refresh LM Studio on remount

- _setup_cache_env now reads XDG_CACHE_HOME (falls back to ~/.cache)
  instead of hard-coding ~/.cache/huggingface. This follows the standard
  HF cache resolution chain and respects distro/container overrides.

- isLocalPath in GgufVariantExpander uses a regex that covers Windows
  drive letters (C:\, D:/), UNC paths (\\server\share), relative paths
  (./, ../), and tilde (~/) -- not just startsWith("/").

- HubModelPicker.useEffect now calls listLocalModels() before the
  alreadyCached early-return gate so LM Studio models are always
  refreshed on remount. Also seeds useState from _lmStudioCache for
  instant display on re-open.

* fix: add comment explaining isLocalPath regex for Windows/cross-platform paths

* fix: prioritize unsloth publisher in LM Studio model list

* fix: scope unsloth-first sort to LM Studio models on all platforms

* fix: add missing _lmStudioCache module-level declaration

* fix: prioritize unsloth publisher before timestamp sort in LM Studio group

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
Co-authored-by: Daniel Han <danielhanchen@gmail.com>
2026-03-27 06:59:27 -07:00

259 lines
7.1 KiB
Python

# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
from __future__ import annotations
import json
import os
from pathlib import Path
import tempfile
def studio_root() -> Path:
return Path.home() / ".unsloth" / "studio"
def cache_root() -> Path:
"""Central cache directory for all studio downloads (models, datasets, etc.)."""
return Path.home() / ".unsloth" / "studio" / "cache"
def assets_root() -> Path:
return studio_root() / "assets"
def datasets_root() -> Path:
return assets_root() / "datasets"
def dataset_uploads_root() -> Path:
return datasets_root() / "uploads"
def recipe_datasets_root() -> Path:
return datasets_root() / "recipes"
def outputs_root() -> Path:
return studio_root() / "outputs"
def exports_root() -> Path:
return studio_root() / "exports"
def auth_root() -> Path:
return studio_root() / "auth"
def auth_db_path() -> Path:
return auth_root() / "auth.db"
def studio_db_path() -> Path:
return studio_root() / "studio.db"
def tmp_root() -> Path:
return Path(tempfile.gettempdir()) / "unsloth-studio"
def seed_uploads_root() -> Path:
return datasets_root() / "seed-uploads"
def unstructured_seed_cache_root() -> Path:
return tmp_root() / "unstructured-seed-cache"
def unstructured_uploads_root() -> Path:
return datasets_root() / "unstructured-uploads"
def oxc_validator_tmp_root() -> Path:
return tmp_root() / "oxc-validator"
def tensorboard_root() -> Path:
return studio_root() / "runs"
def ensure_dir(path: Path) -> Path:
path.mkdir(parents = True, exist_ok = True)
return path
def legacy_hf_cache_dir() -> Path:
"""Old Unsloth-specific HF hub cache, kept for backward-compat scanning."""
return cache_root() / "huggingface" / "hub"
def hf_default_cache_dir() -> Path:
"""Return the platform default HuggingFace hub cache (ignoring env overrides).
This is the location HF uses when no ``HF_HUB_CACHE`` / ``HF_HOME``
env var is set. We scan it so that models a user downloaded *before*
installing Unsloth Studio are still discovered.
"""
return Path.home() / ".cache" / "huggingface" / "hub"
def lmstudio_model_dirs() -> list[Path]:
"""Return LM Studio model directories that exist on disk."""
dirs: list[Path] = []
seen: set[Path] = set()
def _add(p: Path) -> None:
resolved = p.resolve()
if resolved not in seen and p.is_dir():
seen.add(resolved)
dirs.append(p)
# 1. Check LM Studio settings.json for custom downloads folder
settings_path = Path.home() / ".lmstudio" / "settings.json"
if settings_path.is_file():
try:
with open(settings_path) as f:
settings = json.load(f)
downloads = settings.get("downloadsFolder", "")
if downloads:
_add(Path(downloads).expanduser())
except Exception:
pass
# 2. LM Studio current default models directory (all platforms)
_add(Path.home() / ".lmstudio" / "models")
# 3. Legacy LM Studio cache location
_add(Path.home() / ".cache" / "lm-studio" / "models")
return dirs
def _setup_cache_env() -> None:
"""Set cache environment variables for HuggingFace, uv, and vLLM.
Respects the standard HF cache resolution chain: explicit ``HF_HOME``
/ ``HF_HUB_CACHE`` env vars take priority, then ``XDG_CACHE_HOME``,
then the platform default (``~/.cache/huggingface``). The legacy
Unsloth cache is still *scanned* for models but is never set as the
active download target.
Only sets variables that are not already set by the user, so
explicit overrides (e.g. HF_HOME=/data/hf) are respected.
Works on Linux, macOS, and Windows.
"""
root = cache_root()
xdg_cache = Path(
os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache")
).expanduser()
hf_default = xdg_cache / "huggingface"
defaults: dict[str, str] = {
"HF_HOME": str(hf_default),
"HF_HUB_CACHE": str(hf_default / "hub"),
"HF_XET_CACHE": str(hf_default / "xet"),
"UV_CACHE_DIR": str(root / "uv"),
"VLLM_CACHE_ROOT": str(root / "vllm"),
}
for key, value in defaults.items():
if key not in os.environ:
os.environ[key] = value
Path(value).mkdir(parents = True, exist_ok = True)
def ensure_studio_directories() -> None:
"""Create all standard studio directories on startup."""
for dir_fn in (
studio_root,
assets_root,
datasets_root,
dataset_uploads_root,
recipe_datasets_root,
unstructured_uploads_root,
outputs_root,
exports_root,
auth_root,
tensorboard_root,
):
ensure_dir(dir_fn())
_setup_cache_env()
def _clean_relative_path(
path_value: str, *, strip_prefixes: tuple[str, ...] = ()
) -> Path:
path = Path(path_value).expanduser()
parts = [part for part in path.parts if part not in ("", ".")]
while parts and parts[0] in strip_prefixes:
parts = parts[1:]
return Path(*parts) if parts else Path()
def resolve_under_root(
path_value: str | None,
*,
root: Path,
strip_prefixes: tuple[str, ...] = (),
) -> Path:
if not path_value or not str(path_value).strip():
return root
path = Path(str(path_value).strip()).expanduser()
if path.is_absolute():
return path
cleaned = _clean_relative_path(str(path), strip_prefixes = strip_prefixes)
return root / cleaned
def resolve_output_dir(path_value: str | None = None) -> Path:
return resolve_under_root(
path_value,
root = outputs_root(),
strip_prefixes = ("outputs",),
)
def resolve_export_dir(path_value: str | None = None) -> Path:
return resolve_under_root(
path_value,
root = exports_root(),
strip_prefixes = ("exports",),
)
def resolve_tensorboard_dir(path_value: str | None = None) -> Path:
return resolve_under_root(
path_value,
root = tensorboard_root(),
strip_prefixes = ("runs", "tensorboard"),
)
def resolve_dataset_path(path_value: str) -> Path:
path = Path(path_value).expanduser()
if path.is_absolute():
return path
parts = [part for part in Path(path_value).parts if part not in ("", ".")]
if parts[:2] == ["assets", "datasets"]:
parts = parts[2:]
if parts and parts[0] == "uploads":
cleaned = Path(*parts[1:]) if len(parts) > 1 else Path()
return dataset_uploads_root() / cleaned
if parts and parts[0] == "recipes":
cleaned = Path(*parts[1:]) if len(parts) > 1 else Path()
return recipe_datasets_root() / cleaned
cleaned = Path(*parts) if parts else Path()
candidates = [
dataset_uploads_root() / cleaned,
recipe_datasets_root() / cleaned,
datasets_root() / cleaned,
dataset_uploads_root() / cleaned.name,
recipe_datasets_root() / cleaned.name,
]
for candidate in candidates:
if candidate.exists():
return candidate
return candidates[0]