unsloth/studio/backend/utils/datasets/data_collators.py
Daniel Han f08aef1804 Studio (#4237)
* Rebuild Studio branch on top of main

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* Fix security and code quality issues for Studio PR #4237

- Validate models_dir query param against allowed directory roots
  to prevent path traversal in /api/models/local endpoint
- Replace string startswith() with Path.is_relative_to() for
  frontend path traversal check in serve_frontend
- Sanitize SSE error messages to not leak exception details to
  clients (4 locations in inference.py)
- Bind port-discovery socket to 127.0.0.1 instead of all interfaces
  in llama_cpp backend
- Import datasets_root and resolve_output_dir in embedding training
  function to fix NameError and use managed output directory
- Remove stale .gitignore entries for package-lock.json and test
  directories so tests can be tracked in version control
- Add venv-reexecution logic to ui CLI command matching the studio
  command behavior

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* Move models_dir path validation before try/except block

The HTTPException(403) was inside the try/except Exception handler,
so it would be caught and re-raised as a 500. Moving the validation
before the try block ensures the 403 is returned directly and also
makes the control flow clearer for static analysis (path is validated
before any filesystem operations).

* Use os.path.realpath + startswith for models_dir validation

CodeQL py/path-injection does not recognize Path.is_relative_to() as
a sanitizer. Switched to os.path.realpath + str.startswith which is
a recognized sanitizer pattern in CodeQL's taint analysis. The
startswith check uses root_str + os.sep to prevent prefix collisions
(e.g. /app/models_evil matching /app/models).

* Never pass user input to Path constructor in models_dir validation

CodeQL traces taint through Path(resolved) even after a startswith
barrier guard. Fix: the user-supplied models_dir is only used as a
string for comparison against allowed roots. The Path object passed
to _scan_models_dir comes from the trusted allowed_roots list, not
from user input. This fully breaks the taint chain.

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
2026-03-12 03:36:19 -07:00

203 lines
5.9 KiB
Python

# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
"""
Data collators for dataset processing.
This module contains custom data collators for training,
particularly for VLM/OCR processing.
"""
import torch
from dataclasses import dataclass
from typing import Any, List, Optional, Union
from loggers import get_logger
logger = get_logger(__name__)
@dataclass
class DataCollatorSpeechSeq2SeqWithPadding:
"""
Data collator for Whisper speech-to-text training.
Pads input features (audio) and label sequences (text) separately,
masks padding in labels with -100, and strips leading BOS token.
Mirrors the collator from the Whisper.ipynb notebook.
"""
processor: Any
def __call__(self, features: List[dict]) -> dict:
input_features = [
{"input_features": feature["input_features"]} for feature in features
]
batch = self.processor.feature_extractor.pad(
input_features, return_tensors = "pt"
)
label_features = [{"input_ids": feature["labels"]} for feature in features]
labels_batch = self.processor.tokenizer.pad(label_features, return_tensors = "pt")
labels = labels_batch["input_ids"].masked_fill(
labels_batch.attention_mask.ne(1), -100
)
if (labels[:, 0] == self.processor.tokenizer.bos_token_id).all().cpu().item():
labels = labels[:, 1:]
batch["labels"] = labels
return batch
@dataclass
class DeepSeekOCRDataCollator:
"""
Data collator for DeepSeek OCR VLM training.
Handles:
- Image processing via processor
- Text tokenization
- Proper label masking for instruction fine-tuning
"""
processor: Any # Qwen2VLProcessor or similar
max_length: int = 2048
ignore_index: int = -100
def __call__(self, batch: List[dict]) -> dict:
"""
Collate a batch of samples.
Args:
batch: List of dicts, each with 'messages' containing
[{'role': 'user', 'content': [...]}, {'role': 'assistant', 'content': [...]}]
Returns:
dict with input_ids, attention_mask, labels, pixel_values, etc.
"""
from PIL import Image
# Extract messages and images
all_messages = []
all_images = []
for sample in batch:
messages = sample["messages"]
all_messages.append(messages)
# Extract PIL images from content
for msg in messages:
content = msg.get("content", [])
if isinstance(content, list):
for item in content:
if isinstance(item, dict) and item.get("type") == "image":
img = item.get("image")
if img is not None and hasattr(img, "size"): # PIL Image
all_images.append(img)
# Process with the VL processor
try:
# Qwen2VL style processing
texts = [
self.processor.apply_chat_template(
msgs, tokenize = False, add_generation_prompt = False
)
for msgs in all_messages
]
# Process with images
inputs = self.processor(
text = texts,
images = all_images if all_images else None,
return_tensors = "pt",
padding = True,
truncation = True,
max_length = self.max_length,
)
# Create labels (mask input, keep output)
labels = inputs["input_ids"].clone()
# Simple masking: mask padding tokens
labels[labels == self.processor.tokenizer.pad_token_id] = self.ignore_index
inputs["labels"] = labels
return inputs
except Exception as e:
logger.info(f"⚠️ DeepSeekOCRDataCollator error: {e}")
raise
@dataclass
class VLMDataCollator:
"""
Generic VLM data collator that works with various processors.
Supports:
- Qwen2VL
- LLaVA
- Other VL models with compatible processors
"""
processor: Any
max_length: int = 2048
ignore_index: int = -100
mask_input_tokens: bool = True # Whether to mask user tokens in labels
def __call__(self, batch: List[dict]) -> dict:
"""
Collate a batch of VLM samples.
"""
all_messages = []
all_images = []
for sample in batch:
messages = sample.get("messages", [])
all_messages.append(messages)
# Extract images
for msg in messages:
content = msg.get("content", [])
if isinstance(content, list):
for item in content:
if isinstance(item, dict):
img = item.get("image")
if img is not None:
all_images.append(img)
# Apply chat template
texts = [
self.processor.apply_chat_template(
msgs, tokenize = False, add_generation_prompt = False
)
for msgs in all_messages
]
# Process inputs
inputs = self.processor(
text = texts,
images = all_images if all_images else None,
return_tensors = "pt",
padding = True,
truncation = True,
max_length = self.max_length,
)
# Create labels
labels = inputs["input_ids"].clone()
# Mask padding
if hasattr(self.processor, "tokenizer"):
pad_token_id = self.processor.tokenizer.pad_token_id
else:
pad_token_id = self.processor.pad_token_id
if pad_token_id is not None:
labels[labels == pad_token_id] = self.ignore_index
inputs["labels"] = labels
return inputs