* feat(studio): add S3 dataset configuration foundation (#4539) Add foundational types and configuration for S3 bucket dataset loading: - Add S3Config type to frontend training types - Add S3Config Pydantic model to backend training models - Add "s3" as a DatasetSource option - Add s3Config state and setS3Config action to training config store - Add i18n translations for S3 configuration (English and Chinese) This provides the type definitions and UI text for S3 integration. Full implementation requires boto3 dependency and data loading logic. Refs: #4539 * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Wire S3 config into training pipeline and prevent secrets persistence - Pass s3_config from request into training_kwargs so it flows to training subprocess - Add s3Config to NON_PERSISTED_STATE_KEYS to prevent AWS secrets from being saved to localStorage Addresses code review feedback on PR #5951. * Exclude S3 config from database persistence to protect secrets Filter out s3_config (which contains secret_access_key) from the config_json stored in training_runs table, preventing AWS credentials from being persisted to disk. Addresses P1 security feedback on PR #5951. * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Re-raise HTTPException in start_training and defer s3 DatasetSource widening for PR #5951 * Redact s3_config from W&B run config and accept camelCase S3 credential aliases for PR #5951 * feat(studio): implement S3 dataset loading end-to-end Builds the actual S3 loader on top of the hardened #5951 foundation, turning the 501-gated scaffold into a working dataset source. Backend: - Add core/training/s3_dataset.py: lists and downloads supported dataset files (parquet/json/jsonl/csv) from an S3 bucket to a temp dir, using IAM-role or access-key credentials. boto3 is imported lazily (optional dep). - Wire s3_config into UnslothTrainer.load_and_format_dataset (downloads then reuses the existing local-file path) and thread it through worker.py. - Replace the 501 "not implemented" gate with a boto3-availability guard so S3 works when boto3 is present and fails clearly when it is not. - Add boto3 to studio.txt requirements. - Add tests/test_s3_dataset.py (8 tests) covering download/filtering, collisions, missing-boto3, and S3Config camelCase/IAM validation. Frontend: - Widen DatasetSource to include "s3"; add s3_config to the training payload type and mapper; add an S3 validation branch and selectS3Source store action. - Add s3-config-form.tsx (bucket/region/prefix/keys/IAM toggle) reusing the existing studio.dataset.s3.* i18n strings. - Add a Hugging Face / Local / Amazon S3 source toggle in dataset-section; the S3 config card replaces the dataset combobox when S3 is selected. - Fix DatasetPreviewDialog to accept the widened DatasetSource type. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Fix S3 dataset loader for PR #6222 * Fix S3 dataset edge cases for PR #6222 * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci * Fix S3 IAM payload handling for PR #6222 * Block multimodal S3 datasets for PR #6222 --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com> Co-authored-by: Daniel Han <danielhanchen@gmail.com> Co-authored-by: Ash <ash@MacBook-Pro.local> Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Co-authored-by: wasimysaid <wasimysdev@gmail.com>
96 lines
2.7 KiB
Python
96 lines
2.7 KiB
Python
# SPDX-License-Identifier: AGPL-3.0-only
|
|
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
|
|
|
"""Helpers for validating resumable training outputs."""
|
|
|
|
import json
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
from utils.paths import outputs_root, resolve_output_dir
|
|
|
|
|
|
def _is_under_outputs(path: Path) -> bool:
|
|
resolved = path.resolve(strict = False)
|
|
root = outputs_root().resolve(strict = False)
|
|
try:
|
|
resolved.relative_to(root)
|
|
return True
|
|
except ValueError:
|
|
return False
|
|
|
|
|
|
def has_resume_state(path_value: Optional[str]) -> bool:
|
|
if not path_value:
|
|
return False
|
|
return get_resume_checkpoint_path(path_value) is not None
|
|
|
|
|
|
def _checkpoint_step(path: Path) -> int:
|
|
try:
|
|
return int(path.name.removeprefix("checkpoint-"))
|
|
except ValueError:
|
|
return -1
|
|
|
|
|
|
def get_resume_checkpoint_path(path_value: str) -> Optional[str]:
|
|
path = resolve_output_dir(path_value)
|
|
if not _is_under_outputs(path) or not path.is_dir():
|
|
return None
|
|
if (path / "trainer_state.json").is_file():
|
|
return str(path)
|
|
|
|
checkpoints = [
|
|
child
|
|
for child in path.glob("checkpoint-*")
|
|
if child.is_dir() and (child / "trainer_state.json").is_file()
|
|
]
|
|
if not checkpoints:
|
|
return None
|
|
return str(max(checkpoints, key = _checkpoint_step))
|
|
|
|
|
|
def normalize_resume_output_dir(path_value: str) -> str:
|
|
path = resolve_output_dir(path_value)
|
|
if not _is_under_outputs(path):
|
|
raise ValueError("Resume checkpoint must be inside Studio outputs.")
|
|
return str(path)
|
|
|
|
|
|
def _run_config(run: dict) -> dict:
|
|
raw_config = run.get("config_json")
|
|
if isinstance(raw_config, dict):
|
|
return raw_config
|
|
if not isinstance(raw_config, str) or not raw_config.strip():
|
|
return {}
|
|
try:
|
|
parsed = json.loads(raw_config)
|
|
except (json.JSONDecodeError, TypeError):
|
|
return {}
|
|
return parsed if isinstance(parsed, dict) else {}
|
|
|
|
|
|
def _uses_s3_dataset(run: dict) -> bool:
|
|
config = _run_config(run)
|
|
return config.get("dataset_source") == "s3" or "s3_dataset" in config
|
|
|
|
|
|
def can_resume_run(run: dict) -> bool:
|
|
if run.get("resumed_later"):
|
|
return False
|
|
if _uses_s3_dataset(run):
|
|
return False
|
|
|
|
final_step = run.get("final_step")
|
|
total_steps = run.get("total_steps")
|
|
has_remaining_steps = (
|
|
not isinstance(final_step, int)
|
|
or not isinstance(total_steps, int)
|
|
or total_steps <= 0
|
|
or final_step < total_steps
|
|
)
|
|
return (
|
|
run.get("status") == "stopped"
|
|
and has_remaining_steps
|
|
and has_resume_state(run.get("output_dir"))
|
|
)
|