unsloth/studio/backend/utils/datasets/__init__.py
Daniel Han f08aef1804 Studio (#4237)
* Rebuild Studio branch on top of main

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* Fix security and code quality issues for Studio PR #4237

- Validate models_dir query param against allowed directory roots
  to prevent path traversal in /api/models/local endpoint
- Replace string startswith() with Path.is_relative_to() for
  frontend path traversal check in serve_frontend
- Sanitize SSE error messages to not leak exception details to
  clients (4 locations in inference.py)
- Bind port-discovery socket to 127.0.0.1 instead of all interfaces
  in llama_cpp backend
- Import datasets_root and resolve_output_dir in embedding training
  function to fix NameError and use managed output directory
- Remove stale .gitignore entries for package-lock.json and test
  directories so tests can be tracked in version control
- Add venv-reexecution logic to ui CLI command matching the studio
  command behavior

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* Move models_dir path validation before try/except block

The HTTPException(403) was inside the try/except Exception handler,
so it would be caught and re-raised as a 500. Moving the validation
before the try block ensures the 403 is returned directly and also
makes the control flow clearer for static analysis (path is validated
before any filesystem operations).

* Use os.path.realpath + startswith for models_dir validation

CodeQL py/path-injection does not recognize Path.is_relative_to() as
a sanitizer. Switched to os.path.realpath + str.startswith which is
a recognized sanitizer pattern in CodeQL's taint analysis. The
startswith check uses root_str + os.sep to prevent prefix collisions
(e.g. /app/models_evil matching /app/models).

* Never pass user input to Path constructor in models_dir validation

CodeQL traces taint through Path(resolved) even after a startswith
barrier guard. Fix: the user-supplied models_dir is only used as a
string for comparison against allowed roots. The Path object passed
to _scan_models_dir comes from the trusted allowed_roots list, not
from user input. This fully breaks the taint chain.

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
2026-03-12 03:36:19 -07:00

105 lines
2.8 KiB
Python

# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
"""
Dataset utilities package.
This package provides utilities for dataset format detection, conversion,
and processing for LLM and VLM fine-tuning workflows.
Modules:
- format_detection: Detect dataset formats (Alpaca, ShareGPT, ChatML)
- format_conversion: Convert between dataset formats
- chat_templates: Apply chat templates to datasets
- vlm_processing: Vision-Language Model processing utilities
- data_collators: Custom data collators for training
- model_mappings: Model-to-template mapping constants
"""
# Format detection
from .format_detection import (
detect_dataset_format,
detect_custom_format_heuristic,
detect_multimodal_dataset,
detect_vlm_dataset_structure,
)
# Format conversion
from .format_conversion import (
standardize_chat_format,
convert_chatml_to_alpaca,
convert_alpaca_to_chatml,
convert_to_vlm_format,
convert_llava_to_vlm_format,
convert_sharegpt_with_images_to_vlm_format,
)
# Chat templates
from .chat_templates import (
apply_chat_template_to_dataset,
get_dataset_info_summary,
get_tokenizer_chat_template,
DEFAULT_ALPACA_TEMPLATE,
)
# VLM processing
from .vlm_processing import (
generate_smart_vlm_instruction,
)
# Data collators
from .data_collators import (
DataCollatorSpeechSeq2SeqWithPadding,
DeepSeekOCRDataCollator,
VLMDataCollator,
)
# Model mappings (constants)
from .model_mappings import (
TEMPLATE_TO_MODEL_MAPPER,
MODEL_TO_TEMPLATE_MAPPER,
TEMPLATE_TO_RESPONSES_MAPPER,
)
# Legacy imports from the original dataset_utils.py for backward compatibility
# These functions have not yet been refactored into separate modules
from .dataset_utils import (
check_dataset_format,
format_and_template_dataset,
format_dataset,
)
# Public API
__all__ = [
# Detection
"detect_dataset_format",
"detect_custom_format_heuristic",
"detect_multimodal_dataset",
"detect_vlm_dataset_structure",
# Conversion
"standardize_chat_format",
"convert_chatml_to_alpaca",
"convert_alpaca_to_chatml",
"convert_to_vlm_format",
"convert_llava_to_vlm_format",
"convert_sharegpt_with_images_to_vlm_format",
# Templates
"apply_chat_template_to_dataset",
"get_dataset_info_summary",
"get_tokenizer_chat_template",
"DEFAULT_ALPACA_TEMPLATE",
# VLM
"generate_smart_vlm_instruction",
# Collators
"DataCollatorSpeechSeq2SeqWithPadding",
"DeepSeekOCRDataCollator",
"VLMDataCollator",
# Mappings
"TEMPLATE_TO_MODEL_MAPPER",
"MODEL_TO_TEMPLATE_MAPPER",
"TEMPLATE_TO_RESPONSES_MAPPER",
# Main entry points
"check_dataset_format",
"format_and_template_dataset",
"format_dataset",
]