Text-to-video lands as a SIBLING of the image diffusion backend, not a mode of
it: video pipelines take frame/fps arguments, return frame stacks plus, for
LTX-2, synchronized audio, and persist MP4s -- none of the image module's
img2img/inpaint/ControlNet/LoRA surface applies. The image backend's hardware
and optimisation layers are imported unchanged (device/dtype resolution, memory
planning + offload tiers, attention backends, speed profiles, FBCache), and the
load-token/cancel-event concurrency skeleton is copied verbatim so lifecycle
behaviour cannot diverge.
core/inference/video_families.py: a pure VideoFamily registry (no torch) with
the ltx-2 entry -- LTX2Pipeline + LTX2VideoTransformer3DModel, base
Lightricks/LTX-2, unsloth/LTX-2.3-GGUF as the curated GGUF source, audio on,
frame lattice k*8+1, /32 resolutions with a vertical preset, and measured bf16
component sizes (the Gemma3-27B text encoder outweighs the 19B DiT itself).
MoE fields (transformer_2, guidance_scale_2) are declared now so the Wan2.2
A14B family lands later without churning the schema.
core/inference/video.py: VideoBackend with async begin_load + cache-scan
download progress, GGUF / single-file / full-pipeline loads (the GGUF DiT
assembles onto the base repo exactly like the image path), generation with
frame/size snapping BEFORE latents allocate, per-step progress + ETA and
cooperative cancel via the standard diffusers callback, and MP4 (H.264) export
through diffusers' PyAV encoder with the audio track muxed when the family
produces one. VAE tiling is always on: decoding a 100+ frame clip is the
memory peak, and the frames-aware estimate_video_runtime_mib (new, in
diffusion_memory) feeds the planner where the pixel-area image estimate would
badly undershoot. Loads are gated to unsloth/*, the official Lightricks base
repos, or local paths; PyAV availability is checked at load time so a missing
encoder cannot fail a clip after a multi-minute denoise.
core/inference/video_gallery.py: {id}.mp4 + {id}.json recipe sidecar pairs
under studio_root()/videos (an MP4 has no PNG text chunk to embed the recipe
in), with the image gallery's id/containment guards, newest-first listing that
skips orphans, delete/clear.
gpu_arbiter gains the VIDEO owner: ownership is exclusive, so the existing
evict-the-current-owner already generalises to chat/image/video all evicting
each other. The av (PyAV) dependency joins requirements/studio.txt.
Tests: video family detection/snapping/defaults, backend lifecycle on a faked
torch/diffusers runtime (GGUF assembly, shape snapping, distilled defaults,
cancel/progress, sentinel), gallery roundtrip/containment/orphans. 52 new
tests green plus the arbiter suite.
167 lines
6.1 KiB
Python
167 lines
6.1 KiB
Python
# SPDX-License-Identifier: AGPL-3.0-only
|
|
# Copyright 2026-present the Unsloth AI Inc. team. All rights reserved. See /studio/LICENSE.AGPL-3.0
|
|
|
|
"""Unit tests for the disk-backed video gallery: MP4 + JSON-sidecar round-trips,
|
|
listing order, safe id handling, orphan-pair skipping, and delete/clear."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
|
|
import core.inference.video_gallery as gallery
|
|
|
|
|
|
import pytest
|
|
|
|
|
|
@pytest.fixture(autouse = True)
|
|
def _tmp_gallery(monkeypatch, tmp_path):
|
|
# Point the gallery at a throwaway root instead of ~/.unsloth/studio.
|
|
monkeypatch.setattr(gallery, "studio_root", lambda: tmp_path)
|
|
|
|
|
|
def _mp4(tag = b"\x00\x00\x00\x18ftypmp42"):
|
|
# Not a real container; the gallery treats the bytes as opaque payload.
|
|
return tag
|
|
|
|
|
|
def _meta(**over):
|
|
base = {
|
|
"prompt": "a sloth surfing",
|
|
"negative_prompt": None,
|
|
"width": 1024,
|
|
"height": 576,
|
|
"num_frames": 49,
|
|
"steps": 30,
|
|
"guidance": 6.0,
|
|
"seed": 7,
|
|
"model": "unsloth/some-video-model",
|
|
"created_at": 100.0,
|
|
}
|
|
base.update(over)
|
|
return base
|
|
|
|
|
|
def test_save_writes_pair_and_round_trips():
|
|
record = gallery.save(_mp4(), _meta())
|
|
assert record["id"] and record["url"].endswith(f"{record['id']}/file")
|
|
|
|
# Both files of the pair exist: the mp4 payload and the json recipe sidecar.
|
|
directory = gallery.gallery_dir()
|
|
assert (directory / f"{record['id']}.mp4").is_file()
|
|
sidecar = directory / f"{record['id']}.json"
|
|
assert json.loads(sidecar.read_text(encoding = "utf-8"))["prompt"] == "a sloth surfing"
|
|
|
|
listed = gallery.list_videos()
|
|
assert len(listed) == 1
|
|
assert listed[0]["prompt"] == "a sloth surfing" and listed[0]["seed"] == 7
|
|
# Meta fields survive the sidecar round-trip untouched.
|
|
assert listed[0]["num_frames"] == 49 and listed[0]["model"] == "unsloth/some-video-model"
|
|
|
|
|
|
def test_url_shape():
|
|
record = gallery.save(_mp4(), _meta())
|
|
assert record["url"] == f"/api/inference/video/gallery/{record['id']}/file"
|
|
|
|
|
|
def _save_with_mtime(prompt: str, t: float) -> dict:
|
|
record = gallery.save(_mp4(), _meta(prompt = prompt, created_at = t))
|
|
# Listing orders by mp4 mtime; set it explicitly so a tight test loop can't tie it.
|
|
os.utime(gallery.gallery_dir() / f"{record['id']}.mp4", (t, t))
|
|
return record
|
|
|
|
|
|
def test_list_is_newest_first():
|
|
old = _save_with_mtime("old", 100.0)
|
|
new = _save_with_mtime("new", 200.0)
|
|
assert [r["id"] for r in gallery.list_videos()] == [new["id"], old["id"]]
|
|
|
|
|
|
def test_list_paginates_with_limit_offset():
|
|
# 5 videos, newest (t=4) first.
|
|
for i in range(5):
|
|
_save_with_mtime(f"p{i}", float(i))
|
|
page1 = gallery.list_videos(limit = 2, offset = 0)
|
|
page2 = gallery.list_videos(limit = 2, offset = 2)
|
|
assert [r["prompt"] for r in page1] == ["p4", "p3"]
|
|
assert [r["prompt"] for r in page2] == ["p2", "p1"]
|
|
# limit=None still returns everything from the offset.
|
|
assert len(gallery.list_videos()) == 5
|
|
assert len(gallery.list_videos(offset = 4)) == 1
|
|
|
|
|
|
def test_video_path_rejects_unsafe_ids():
|
|
# Traversal / bad chars / absolute paths never resolve to a path.
|
|
assert gallery.video_path("../../etc/passwd") is None
|
|
assert gallery.video_path("/etc/passwd") is None
|
|
assert gallery.video_path("a/b") is None
|
|
assert gallery.video_path("missing") is None
|
|
|
|
|
|
def test_video_path_returns_mp4_for_saved_id():
|
|
record = gallery.save(_mp4(), _meta())
|
|
path = gallery.video_path(record["id"])
|
|
assert path is not None and path.name == f"{record['id']}.mp4"
|
|
|
|
|
|
def test_delete_removes_both_files():
|
|
record = gallery.save(_mp4(), _meta(prompt = "a"))
|
|
gallery.save(_mp4(), _meta(prompt = "b"))
|
|
directory = gallery.gallery_dir()
|
|
assert gallery.delete(record["id"]) is True
|
|
# Both halves of the pair are gone.
|
|
assert not (directory / f"{record['id']}.mp4").exists()
|
|
assert not (directory / f"{record['id']}.json").exists()
|
|
assert gallery.delete(record["id"]) is False # already gone
|
|
assert len(gallery.list_videos()) == 1
|
|
|
|
|
|
def test_clear_returns_count():
|
|
gallery.save(_mp4(), _meta(prompt = "a"))
|
|
gallery.save(_mp4(), _meta(prompt = "b"))
|
|
assert gallery.clear() == 2
|
|
assert gallery.list_videos() == []
|
|
# No stray sidecars left behind after a clear.
|
|
assert list(gallery.gallery_dir().glob("*.json")) == []
|
|
|
|
|
|
def test_list_skips_orphan_mp4_without_sidecar():
|
|
# An MP4 with no readable json sidecar (a hand-dropped file) is not a record.
|
|
orphan = gallery.gallery_dir() / "orphan.mp4"
|
|
orphan.write_bytes(_mp4())
|
|
gallery.save(_mp4(), _meta(prompt = "ours"))
|
|
listed = gallery.list_videos()
|
|
assert [r["prompt"] for r in listed] == ["ours"]
|
|
|
|
|
|
def test_list_skips_orphan_sidecar_without_mp4():
|
|
# A json sidecar with no MP4 alongside it is never surfaced (listing globs mp4s).
|
|
orphan = gallery.gallery_dir() / "lonely.json"
|
|
orphan.write_text(json.dumps(_meta(prompt = "no video")), encoding = "utf-8")
|
|
gallery.save(_mp4(), _meta(prompt = "ours"))
|
|
listed = gallery.list_videos()
|
|
assert [r["prompt"] for r in listed] == ["ours"]
|
|
|
|
|
|
def test_orphan_mp4_in_window_does_not_drop_valid_videos():
|
|
# An orphan MP4 sorting INTO the requested page must not consume a window slot
|
|
# and drop a valid video that sorts after it: paging is over readable records.
|
|
_save_with_mtime("p2", 100.0)
|
|
orphan = gallery.gallery_dir() / "zzz_orphan.mp4"
|
|
orphan.write_bytes(_mp4()) # newest by mtime (set below), sorts first
|
|
os.utime(orphan, (300.0, 300.0))
|
|
_save_with_mtime("p1", 200.0)
|
|
# First page of 2 must still return both real videos, not [p1] (orphan eating a slot).
|
|
page1 = gallery.list_videos(limit = 2, offset = 0)
|
|
assert [r["prompt"] for r in page1] == ["p1", "p2"]
|
|
|
|
|
|
def test_list_skips_corrupt_sidecar():
|
|
# A sidecar that is not valid JSON is treated as a foreign/orphan mp4 and skipped.
|
|
directory = gallery.gallery_dir()
|
|
(directory / "broken.mp4").write_bytes(_mp4())
|
|
(directory / "broken.json").write_text("{not json", encoding = "utf-8")
|
|
gallery.save(_mp4(), _meta(prompt = "ours"))
|
|
listed = gallery.list_videos()
|
|
assert [r["prompt"] for r in listed] == ["ours"]
|