unsloth/studio/backend/core/__init__.py
Daniel Han f08aef1804 Studio (#4237)
* Rebuild Studio branch on top of main

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* Fix security and code quality issues for Studio PR #4237

- Validate models_dir query param against allowed directory roots
  to prevent path traversal in /api/models/local endpoint
- Replace string startswith() with Path.is_relative_to() for
  frontend path traversal check in serve_frontend
- Sanitize SSE error messages to not leak exception details to
  clients (4 locations in inference.py)
- Bind port-discovery socket to 127.0.0.1 instead of all interfaces
  in llama_cpp backend
- Import datasets_root and resolve_output_dir in embedding training
  function to fix NameError and use managed output directory
- Remove stale .gitignore entries for package-lock.json and test
  directories so tests can be tracked in version control
- Add venv-reexecution logic to ui CLI command matching the studio
  command behavior

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* Move models_dir path validation before try/except block

The HTTPException(403) was inside the try/except Exception handler,
so it would be caught and re-raised as a 500. Moving the validation
before the try block ensures the 403 is returned directly and also
makes the control flow clearer for static analysis (path is validated
before any filesystem operations).

* Use os.path.realpath + startswith for models_dir validation

CodeQL py/path-injection does not recognize Path.is_relative_to() as
a sanitizer. Switched to os.path.realpath + str.startswith which is
a recognized sanitizer pattern in CodeQL's taint analysis. The
startswith check uses root_str + os.sep to prevent prefix collisions
(e.g. /app/models_evil matching /app/models).

* Never pass user input to Path constructor in models_dir validation

CodeQL traces taint through Path(resolved) even after a startswith
barrier guard. Fix: the user-supplied models_dir is only used as a
string for comparison against allowed roots. The Path object passed
to _scan_models_dir comes from the trusted allowed_roots list, not
from user input. This fully breaks the taint chain.

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
2026-03-12 03:36:19 -07:00

134 lines
4.1 KiB
Python

# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
"""
Unified core module for Unsloth backend
Imports are LAZY (via __getattr__) so that training subprocesses can
import core.training.worker without pulling in heavy ML dependencies
like unsloth, transformers, or torch before the version activation
code has a chance to run.
"""
__all__ = [
# Inference
"InferenceBackend",
"get_inference_backend",
# Training
"get_training_backend",
"TrainingBackend",
"TrainingProgress",
# Config
"ModelConfig",
"is_vision_model",
"scan_trained_loras",
"load_model_defaults",
"get_base_model_from_lora",
# Utils
"format_and_template_dataset",
"normalize_path",
"is_local_path",
"is_model_cached",
"without_hf_auth",
"format_error_message",
"get_gpu_memory_info",
"log_gpu_memory",
"get_device",
"is_apple_silicon",
"clear_gpu_cache",
"DeviceType",
]
def __getattr__(name):
# Inference
if name in ("InferenceBackend", "get_inference_backend"):
from .inference import InferenceBackend, get_inference_backend
globals()["InferenceBackend"] = InferenceBackend
globals()["get_inference_backend"] = get_inference_backend
return globals()[name]
# Training
if name in ("TrainingBackend", "get_training_backend", "TrainingProgress"):
from .training import TrainingBackend, get_training_backend, TrainingProgress
globals()["TrainingBackend"] = TrainingBackend
globals()["get_training_backend"] = get_training_backend
globals()["TrainingProgress"] = TrainingProgress
return globals()[name]
# Config (from utils.models)
if name in (
"is_vision_model",
"ModelConfig",
"scan_trained_loras",
"load_model_defaults",
"get_base_model_from_lora",
):
from utils.models import (
is_vision_model,
ModelConfig,
scan_trained_loras,
load_model_defaults,
get_base_model_from_lora,
)
globals()["is_vision_model"] = is_vision_model
globals()["ModelConfig"] = ModelConfig
globals()["scan_trained_loras"] = scan_trained_loras
globals()["load_model_defaults"] = load_model_defaults
globals()["get_base_model_from_lora"] = get_base_model_from_lora
return globals()[name]
# Paths
if name in ("normalize_path", "is_local_path", "is_model_cached"):
from utils.paths import normalize_path, is_local_path, is_model_cached
globals()["normalize_path"] = normalize_path
globals()["is_local_path"] = is_local_path
globals()["is_model_cached"] = is_model_cached
return globals()[name]
# Utils
if name in ("without_hf_auth", "format_error_message"):
from utils.utils import without_hf_auth, format_error_message
globals()["without_hf_auth"] = without_hf_auth
globals()["format_error_message"] = format_error_message
return globals()[name]
# Hardware
if name in (
"get_device",
"is_apple_silicon",
"clear_gpu_cache",
"get_gpu_memory_info",
"log_gpu_memory",
"DeviceType",
):
from utils.hardware import (
get_device,
is_apple_silicon,
clear_gpu_cache,
get_gpu_memory_info,
log_gpu_memory,
DeviceType,
)
globals()["get_device"] = get_device
globals()["is_apple_silicon"] = is_apple_silicon
globals()["clear_gpu_cache"] = clear_gpu_cache
globals()["get_gpu_memory_info"] = get_gpu_memory_info
globals()["log_gpu_memory"] = log_gpu_memory
globals()["DeviceType"] = DeviceType
return globals()[name]
# Datasets
if name == "format_and_template_dataset":
from utils.datasets import format_and_template_dataset
globals()["format_and_template_dataset"] = format_and_template_dataset
return format_and_template_dataset
raise AttributeError(f"module 'core' has no attribute {name!r}")