unsloth/studio/backend/utils/paths/storage_roots.py
Wasim Yousef Said dd283b0605
feat(studio): multi-file unstructured seed upload with better backend extraction (#4468)
* fix(recipe-studio): prevent fitView from zooming to wrong location on recipe load

* feat: add pymupdf/python-docx deps and unstructured uploads storage root

* feat: add POST /seed/upload-unstructured-file endpoint

* feat: add multi-file chunking with source_file column

* feat: update frontend types and API layer for multi-file upload

* feat: round-robin preview rows across source files

Ensures every uploaded file is represented in the preview table
by cycling through sources instead of just taking the first N rows.

* fix: disable OCR, fix auto-load timing, fix persistence on reload

- Disable pymupdf4llm OCR with write_images=False, show_progress=False
- Replace onAllUploaded callback with useEffect that detects uploading→done
  transition (avoids stale closure reading empty file IDs)
- Fix importer to preserve file IDs from saved recipes instead of clearing
  (clearing only happens at share time via sanitizeSeedForShare)

* fix: harden unstructured upload with input validation and state fixes

Validate block_id/file_id with alphanumeric regex to prevent path
traversal, use exact stem match for file deletion, add error handling
for metadata writes and empty files, fix React stale closures and
object mutations in upload loop, and correct validation logic for
unstructured seed resolved_paths.

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* fix: address PR review - legacy path import, share sanitizer, sync effect

Promote legacy source.path into resolved_paths for old unstructured
recipes, clear source.paths in share sanitizer to prevent leaking local
filesystem paths, and gate file sync effect to dialog open transition
so users can actually delete all uploaded files.

* fix: CSV column fix (BOM + whitespace + unnamed index re-save) for #4470

* fix: harden unstructured upload flow and polish dialog UX

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
2026-03-20 13:22:42 -07:00

198 lines
5 KiB
Python

# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
from __future__ import annotations
import os
from pathlib import Path
import tempfile
def studio_root() -> Path:
return Path.home() / ".unsloth" / "studio"
def cache_root() -> Path:
"""Central cache directory for all studio downloads (models, datasets, etc.)."""
return Path.home() / ".unsloth" / "studio" / "cache"
def assets_root() -> Path:
return studio_root() / "assets"
def datasets_root() -> Path:
return assets_root() / "datasets"
def dataset_uploads_root() -> Path:
return datasets_root() / "uploads"
def recipe_datasets_root() -> Path:
return datasets_root() / "recipes"
def outputs_root() -> Path:
return studio_root() / "outputs"
def exports_root() -> Path:
return studio_root() / "exports"
def auth_root() -> Path:
return studio_root() / "auth"
def auth_db_path() -> Path:
return auth_root() / "auth.db"
def tmp_root() -> Path:
return Path(tempfile.gettempdir()) / "unsloth-studio"
def seed_uploads_root() -> Path:
return datasets_root() / "seed-uploads"
def unstructured_seed_cache_root() -> Path:
return tmp_root() / "unstructured-seed-cache"
def unstructured_uploads_root() -> Path:
return datasets_root() / "unstructured-uploads"
def oxc_validator_tmp_root() -> Path:
return tmp_root() / "oxc-validator"
def tensorboard_root() -> Path:
return studio_root() / "runs"
def ensure_dir(path: Path) -> Path:
path.mkdir(parents = True, exist_ok = True)
return path
def _setup_cache_env() -> None:
"""Set cache environment variables for HuggingFace, uv, and vLLM.
Only sets variables that are not already set by the user, so
explicit overrides (e.g. HF_HOME=/data/hf) are respected.
Works on Linux, macOS, and Windows.
"""
root = cache_root()
hf_dir = root / "huggingface"
defaults = {
"HF_HOME": str(hf_dir),
"HF_HUB_CACHE": str(hf_dir / "hub"),
"HF_XET_CACHE": str(hf_dir / "xet"),
"UV_CACHE_DIR": str(root / "uv"),
"VLLM_CACHE_ROOT": str(root / "vllm"),
}
for key, value in defaults.items():
if key not in os.environ:
os.environ[key] = value
Path(value).mkdir(parents = True, exist_ok = True)
def ensure_studio_directories() -> None:
"""Create all standard studio directories on startup."""
for dir_fn in (
studio_root,
assets_root,
datasets_root,
dataset_uploads_root,
recipe_datasets_root,
unstructured_uploads_root,
outputs_root,
exports_root,
auth_root,
tensorboard_root,
):
ensure_dir(dir_fn())
_setup_cache_env()
def _clean_relative_path(
path_value: str, *, strip_prefixes: tuple[str, ...] = ()
) -> Path:
path = Path(path_value).expanduser()
parts = [part for part in path.parts if part not in ("", ".")]
while parts and parts[0] in strip_prefixes:
parts = parts[1:]
return Path(*parts) if parts else Path()
def resolve_under_root(
path_value: str | None,
*,
root: Path,
strip_prefixes: tuple[str, ...] = (),
) -> Path:
if not path_value or not str(path_value).strip():
return root
path = Path(str(path_value).strip()).expanduser()
if path.is_absolute():
return path
cleaned = _clean_relative_path(str(path), strip_prefixes = strip_prefixes)
return root / cleaned
def resolve_output_dir(path_value: str | None = None) -> Path:
return resolve_under_root(
path_value,
root = outputs_root(),
strip_prefixes = ("outputs",),
)
def resolve_export_dir(path_value: str | None = None) -> Path:
return resolve_under_root(
path_value,
root = exports_root(),
strip_prefixes = ("exports",),
)
def resolve_tensorboard_dir(path_value: str | None = None) -> Path:
return resolve_under_root(
path_value,
root = tensorboard_root(),
strip_prefixes = ("runs", "tensorboard"),
)
def resolve_dataset_path(path_value: str) -> Path:
path = Path(path_value).expanduser()
if path.is_absolute():
return path
parts = [part for part in Path(path_value).parts if part not in ("", ".")]
if parts[:2] == ["assets", "datasets"]:
parts = parts[2:]
if parts and parts[0] == "uploads":
cleaned = Path(*parts[1:]) if len(parts) > 1 else Path()
return dataset_uploads_root() / cleaned
if parts and parts[0] == "recipes":
cleaned = Path(*parts[1:]) if len(parts) > 1 else Path()
return recipe_datasets_root() / cleaned
cleaned = Path(*parts) if parts else Path()
candidates = [
dataset_uploads_root() / cleaned,
recipe_datasets_root() / cleaned,
datasets_root() / cleaned,
dataset_uploads_root() / cleaned.name,
recipe_datasets_root() / cleaned.name,
]
for candidate in candidates:
if candidate.exists():
return candidate
return candidates[0]