Initial commit

This commit is contained in:
jkortis committed 2026-09-30 23:06:14 -04:00
commit 1d235d30e7
58 files changed
+19693

No files matched your search

+116
View File
@@ -0,0 +1,116 @@
import struct
from pathlib import Path
import pytest
from media_sorter.analyzer import MediaAnalyzer
@pytest.fixture
def analyzer():
return MediaAnalyzer()
def test_flac_metadata_extraction(tmp_path: Path, analyzer):
flac_file = tmp_path / "test.flac"
# Create minimal synthetic FLAC file
header = bytearray(b"fLaC")
# STREAMINFO block (type 0, length 34, not last: block_hdr = 0x00 0x00 0x00 0x22)
streaminfo_data = bytearray(34)
# sample_rate = 44100, channels = 2, total_samples = 44100 * 10
# byte 10..12: sample rate
streaminfo_data[10] = (44100 >> 12) & 0xFF
streaminfo_data[11] = (44100 >> 4) & 0xFF
streaminfo_data[12] = ((44100 & 0x0F) << 4) | (1 << 1) # 2 channels (bits 1..3 = 1)
# total samples = 441000
total_samples = 441000
streaminfo_data[13] = (total_samples >> 32) & 0x0F
streaminfo_data[14] = (total_samples >> 24) & 0xFF
streaminfo_data[15] = (total_samples >> 16) & 0xFF
streaminfo_data[16] = (total_samples >> 8) & 0xFF
streaminfo_data[17] = total_samples & 0xFF
header.extend(b"\x80\x00\x00\x22") # is_last = 1, type = 0, len = 34
header.extend(streaminfo_data)
flac_file.write_bytes(header)
meta = analyzer.analyze(flac_file)
assert meta.container == "flac"
assert meta.has_audio is True
assert meta.duration_seconds == 10.0
def test_mp3_id3v2_extraction(tmp_path: Path, analyzer):
mp3_file = tmp_path / "test.mp3"
header = bytearray(b"ID3\x03\x00\x00") # ID3v2.3
# Build TIT2 frame (Title: Bohemian Rhapsody)
tit2_val = b"\x00Bohemian Rhapsody"
tit2_frame = b"TIT2" + struct.pack(">I", len(tit2_val)) + b"\x00\x00" + tit2_val
# Build TPE1 frame (Artist: Queen)
tpe1_val = b"\x00Queen"
tpe1_frame = b"TPE1" + struct.pack(">I", len(tpe1_val)) + b"\x00\x00" + tpe1_val
tag_content = tit2_frame + tpe1_frame
tag_len = len(tag_content)
# Syncsafe integer for tag size
b0 = (tag_len >> 21) & 0x7F
b1 = (tag_len >> 14) & 0x7F
b2 = (tag_len >> 7) & 0x7F
b3 = tag_len & 0x7F
header.extend(bytes([b0, b1, b2, b3]))
header.extend(tag_content)
# Add dummy MP3 frame sync bytes
header.extend(b"\xff\xfb\x90\x00")
mp3_file.write_bytes(header)
meta = analyzer.analyze(mp3_file)
assert meta.container == "mp3"
assert meta.has_audio is True
assert meta.tags.get("title") == "Bohemian Rhapsody"
assert meta.tags.get("artist") == "Queen"
def test_jpeg_exif_extraction(tmp_path: Path, analyzer):
jpg_file = tmp_path / "test.jpg"
# SOI marker + APP1 marker with Exif
soi = b"\xff\xd8"
exif_header = b"Exif\x00\x00"
tiff_header = b"II\x2a\x00\x08\x00\x00\x00" # Little endian, IFD at 8
# 1 entry in IFD0: DateTimeOriginal (tag 0x9003)
num_entries = struct.pack("<H", 1)
tag_id = struct.pack("<H", 0x9003)
type_ascii = struct.pack("<H", 2)
count = struct.pack("<I", 20)
val_offset = struct.pack("<I", 22) # offset from tiff_header
dt_str = b"2025:06:15 10:30:00\x00"
tiff_body = tiff_header + num_entries + tag_id + type_ascii + count + val_offset + dt_str
app1_len = struct.pack(">H", len(exif_header) + len(tiff_body) + 2)
app1 = b"\xff\xe1" + app1_len + exif_header + tiff_body
jpg_file.write_bytes(soi + app1)
meta = analyzer.analyze(jpg_file)
assert meta.container == "jpeg"
assert meta.tags.get("datetime_original") == "2025:06:15 10:30:00"
def test_png_dimensions(tmp_path: Path, analyzer):
png_file = tmp_path / "test.png"
sig = b"\x89PNG\r\n\x1a\n"
# IHDR chunk: 13 bytes data (width=1920, height=1080)
ihdr_data = struct.pack(">IIBBBBB", 1920, 1080, 8, 2, 0, 0, 0)
ihdr = struct.pack(">I", 13) + b"IHDR" + ihdr_data + b"\x00\x00\x00\x00"
png_file.write_bytes(sig + ihdr)
meta = analyzer.analyze(png_file)
assert meta.container == "png"
assert meta.width == 1920
assert meta.height == 1080
assert meta.resolution_label == "1080p"
def test_archive_detection(tmp_path: Path, analyzer):
zip_file = tmp_path / "test.zip"
zip_file.write_bytes(b"PK\x03\x04\x14\x00\x00\x00")
meta = analyzer.analyze(zip_file)
assert meta.container == "zip"
assert meta.mime_type == "application/zip"
+312
View File
@@ -0,0 +1,312 @@
from pathlib import Path
import pytest
from media_sorter.analyzer import MediaMetadata, StreamInfo
from media_sorter.classifier import MediaClassifier
from media_sorter.providers import MockMetadataProvider, ProviderResult
from media_sorter.scanner import ScannedFile
from media_sorter.tokenizer import FilenameTokenizer, TokenizedFilename
@pytest.fixture
def classifier():
return MediaClassifier(confidence_threshold=0.75)
def test_classify_tv_show(classifier):
scanned = ScannedFile(path=Path("/downloads/Game.of.Thrones.S01E01.1080p.mkv"), size=1000000, mtime=1000.0)
tokens = TokenizedFilename(
raw_name="Game.of.Thrones.S01E01.1080p.mkv",
title="Game of Thrones",
season=1,
episode=1,
is_episodic=True,
)
meta = MediaMetadata(
path=scanned.path,
mime_type="video/x-matroska",
container="mkv",
duration_seconds=3600,
has_video=True,
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "tv"
assert res.confidence >= 0.75
assert res.needs_quarantine is False
def test_classify_anime(classifier):
scanned = ScannedFile(path=Path("/downloads/[SubsPlease] Jujutsu Kaisen - 01 [1080p].mkv"), size=1000000, mtime=1000.0)
tokens = TokenizedFilename(
raw_name="[SubsPlease] Jujutsu Kaisen - 01 [1080p].mkv",
title="Jujutsu Kaisen",
episode=1,
season=1,
group="SubsPlease",
is_anime=True,
is_episodic=True,
)
meta = MediaMetadata(
path=scanned.path,
mime_type="video/x-matroska",
container="mkv",
duration_seconds=1400,
has_video=True,
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "anime"
assert res.confidence >= 0.75
assert res.needs_quarantine is False
def test_classify_movie(classifier):
scanned = ScannedFile(path=Path("/downloads/Interstellar.2014.1080p.mkv"), size=5000000, mtime=1000.0)
tokens = TokenizedFilename(
raw_name="Interstellar.2014.1080p.mkv",
title="Interstellar",
year=2014,
resolution="1080p",
)
meta = MediaMetadata(
path=scanned.path,
mime_type="video/x-matroska",
container="mkv",
duration_seconds=10140, # ~2.8 hours
has_video=True,
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "movie"
assert res.confidence >= 0.75
assert res.needs_quarantine is False
def test_classify_music(classifier):
scanned = ScannedFile(path=Path("/music/01 - Come Together.flac"), size=30000000, mtime=1000.0)
tokens = TokenizedFilename(
raw_name="01 - Come Together.flac",
title="Come Together",
track=1,
is_music=True,
)
meta = MediaMetadata(
path=scanned.path,
mime_type="audio/flac",
container="flac",
duration_seconds=259,
has_audio=True,
has_video=False,
tags={"artist": "The Beatles", "album": "Abbey Road"},
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "music"
assert res.confidence >= 0.75
assert res.needs_quarantine is False
def test_classify_audiobook(classifier):
scanned = ScannedFile(path=Path("/audiobooks/Dune - Part 01.m4b"), size=50000000, mtime=1000.0)
tokens = TokenizedFilename(raw_name="Dune - Part 01.m4b", title="Dune")
meta = MediaMetadata(
path=scanned.path,
mime_type="audio/mp4",
container="m4b",
duration_seconds=28800, # 8 hours
has_audio=True,
has_video=False,
tags={"narrator": "George Guidall"},
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "audiobook"
assert res.confidence >= 0.75
assert res.needs_quarantine is False
def test_low_confidence_triggers_quarantine(classifier):
# Ambiguous video clip with no year, no episode, no metadata
scanned = ScannedFile(path=Path("/incoming/unknown_recording_xyz.mkv"), size=10000, mtime=1000.0)
tokens = TokenizedFilename(raw_name="unknown_recording_xyz.mkv", title="unknown recording xyz")
meta = MediaMetadata(
path=scanned.path,
mime_type="video/x-matroska",
container="mkv",
duration_seconds=120,
has_video=True,
)
res = classifier.classify(scanned, tokens, meta)
assert res.confidence < 0.75
assert res.needs_quarantine is True
assert res.quarantine_reason is not None
def test_unsupported_format_triggers_quarantine(classifier):
scanned = ScannedFile(path=Path("/incoming/corrupt_data.bin"), size=1000, mtime=1000.0)
tokens = TokenizedFilename(raw_name="corrupt_data.bin")
meta = MediaMetadata(path=scanned.path, mime_type="application/octet-stream", container="bin")
res = classifier.classify(scanned, tokens, meta)
assert res.category == "unknown"
assert res.needs_quarantine is True
def test_classify_with_metadata_provider():
mock_prov = MockMetadataProvider(
mock_data={
"movie:oppenheimer": ProviderResult(
canonical_title="Oppenheimer",
year=2023,
media_type="movie",
confidence_boost=0.15,
),
"tv:the last of us": ProviderResult(
canonical_title="The Last of Us",
year=2023,
media_type="tv",
season=1,
episode=3,
episode_title="Long, Long Time",
confidence_boost=0.20,
),
}
)
prov_classifier = MediaClassifier(confidence_threshold=0.75, provider=mock_prov)
# 1. Movie verified with provider
scanned_m = ScannedFile(path=Path("/downloads/Oppenheimer.mkv"), size=1000000, mtime=1000.0)
tokens_m = TokenizedFilename(raw_name="Oppenheimer.mkv", title="Oppenheimer")
meta_m = MediaMetadata(path=scanned_m.path, mime_type="video/x-matroska", container="mkv", duration_seconds=10800, has_video=True)
res_m = prov_classifier.classify(scanned_m, tokens_m, meta_m)
assert res_m.category == "movie"
assert res_m.provider_result is not None
assert res_m.provider_result.year == 2023
# 2. TV Show verified with provider
scanned_tv = ScannedFile(path=Path("/downloads/The.Last.of.Us.S01E03.mkv"), size=1000000, mtime=1000.0)
tokens_tv = TokenizedFilename(raw_name="The.Last.of.Us.S01E03.mkv", title="The Last of Us", season=1, episode=3, is_episodic=True)
meta_tv = MediaMetadata(path=scanned_tv.path, mime_type="video/x-matroska", container="mkv", duration_seconds=4500, has_video=True)
res_tv = prov_classifier.classify(scanned_tv, tokens_tv, meta_tv)
assert res_tv.category == "tv"
assert res_tv.provider_result is not None
assert res_tv.provider_result.episode_title == "Long, Long Time"
def test_classify_podcast(classifier):
scanned = ScannedFile(path=Path("/podcasts/Hardcore History 2023-05-12 Episode 68.mp3"), size=50000000, mtime=1000.0)
tokens = TokenizedFilename(raw_name="Hardcore History 2023-05-12 Episode 68.mp3", title="Episode 68", date_stamp="2023-05-12")
meta = MediaMetadata(
path=scanned.path,
mime_type="audio/mpeg",
container="mp3",
duration_seconds=14400,
has_audio=True,
has_video=False,
tags={"podcast": "Dan Carlin's Hardcore History"},
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "podcast"
assert res.confidence >= 0.75
def test_classify_photo_and_home_video(classifier):
# Photo test
scanned_p = ScannedFile(path=Path("/photos/IMG_20250615_123456.jpg"), size=4000000, mtime=1000.0)
tokens_p = TokenizedFilename(raw_name="IMG_20250615_123456.jpg", is_photo_or_home_video=True, date_stamp="2025-06-15")
meta_p = MediaMetadata(path=scanned_p.path, mime_type="image/jpeg", container="jpeg", tags={"camera_model": "Pixel 9 Pro"})
res_p = classifier.classify(scanned_p, tokens_p, meta_p)
assert res_p.category == "photo"
assert res_p.confidence >= 0.90
# Home Video test
scanned_v = ScannedFile(path=Path("/home_videos/VID_20250615_140000.mp4"), size=20000000, mtime=1000.0)
tokens_v = TokenizedFilename(raw_name="VID_20250615_140000.mp4", is_photo_or_home_video=True, date_stamp="2025-06-15")
meta_v = MediaMetadata(path=scanned_v.path, mime_type="video/mp4", container="mp4", duration_seconds=120, has_video=True)
res_v = classifier.classify(scanned_v, tokens_v, meta_v)
assert res_v.category == "home_video"
def test_classify_archive(classifier):
scanned = ScannedFile(path=Path("/downloads/Season1_Extras.zip"), size=500000000, mtime=1000.0)
tokens = TokenizedFilename(raw_name="Season1_Extras.zip")
meta = MediaMetadata(path=scanned.path, mime_type="application/zip", container="zip")
res = classifier.classify(scanned, tokens, meta)
assert res.category == "archive"
assert res.confidence >= 0.90
def test_classify_movie_with_hdtv_and_rartv(classifier):
scanned = ScannedFile(path=Path("/downloads/Gladiator.II.2024.1080p.HDTV.x264-[rartv].mkv"), size=4000000000, mtime=1000.0)
tokens = TokenizedFilename(
raw_name="Gladiator.II.2024.1080p.HDTV.x264-[rartv].mkv",
title="Gladiator II",
year=2024,
resolution="1080p",
video_codec="x264",
source="HDTV",
group="rartv",
)
meta = MediaMetadata(
path=scanned.path,
mime_type="video/x-matroska",
container="mkv",
duration_seconds=5000, # ~83 minutes
has_video=True,
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "movie"
assert res.confidence >= 0.75
assert res.needs_quarantine is False
def test_classify_movie_with_apple_tv_tag(classifier):
scanned = ScannedFile(path=Path("/downloads/Wolfs.2024.1080p.Apple.TV.WEB-DL.DDP5.1.Atmos.H.264.mkv"), size=4500000000, mtime=1000.0)
tokens = TokenizedFilename(
raw_name="Wolfs.2024.1080p.Apple.TV.WEB-DL.DDP5.1.Atmos.H.264.mkv",
title="Wolfs",
year=2024,
resolution="1080p",
video_codec="H.264",
source="WEB-DL",
)
meta = MediaMetadata(
path=scanned.path,
mime_type="video/x-matroska",
container="mkv",
duration_seconds=6400,
has_video=True,
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "movie"
assert res.confidence >= 0.75
def test_video_file_with_audio_not_classified_as_music(classifier):
# Video container .mkv with audio track should never be classified as music
scanned = ScannedFile(
path=Path("/downloads/Star.Wars.The.Clone.Wars.S01E01.1080p.BluRay.REMUX.VC-1.DD5.1-NOGRP.mkv"),
size=4500000000,
mtime=1000.0,
)
tokens = TokenizedFilename(
raw_name="Star.Wars.The.Clone.Wars.S01E01.1080p.BluRay.REMUX.VC-1.DD5.1-NOGRP.mkv",
title="Star Wars The Clone Wars",
season=1,
episode=1,
is_episodic=True,
)
meta = MediaMetadata(
path=scanned.path,
mime_type="video/x-matroska",
container="mkv",
has_audio=True,
has_video=False, # e.g. exotic codec in container
)
res = classifier.classify(scanned, tokens, meta)
assert res.category == "tv"
assert res.confidence >= 0.80
assert res.needs_quarantine is False
+75
View File
@@ -0,0 +1,75 @@
import os
import pytest
from pathlib import Path
from media_sorter.config import Settings, ActionType, ConflictPolicy
from media_sorter.db import get_engine, init_db, get_db_session
from media_sorter.models import BatchRecord, Operation, FileRecord, QuarantineRecord
def test_settings_defaults():
settings = Settings()
assert settings.general.dry_run is True
assert settings.general.confidence_threshold == 0.75
assert settings.general.action == ActionType.MOVE
assert settings.conflicts.policy == ConflictPolicy.RENAME_UNIQUE
assert settings.conflicts.allow_overwrite is False
assert "*.txt" in settings.filters.exclude_patterns
from media_sorter.scanner import Scanner
scanner = Scanner()
assert "*.txt" in scanner.exclude_patterns
assert not scanner._matches_filter("info.txt")
assert not scanner._matches_filter("README.TXT")
assert scanner._matches_filter("movie.mkv")
def test_load_yaml_config(tmp_path: Path):
config_file = tmp_path / "test_config.yaml"
config_file.write_text("""
general:
dry_run: false
confidence_threshold: 0.85
storage:
source_dirs:
- "/test/source"
destination_base: "/test/organized"
""", encoding="utf-8")
settings = Settings.load_from_file(config_file)
assert settings.general.dry_run is False
assert settings.general.confidence_threshold == 0.85
assert "/test/source" in settings.storage.source_dirs
def test_database_initialization(tmp_path: Path):
db_file = tmp_path / "test.db"
engine = init_db(db_path=db_file)
assert db_file.exists()
with get_db_session(engine) as session:
batch = BatchRecord(id="test-uuid-1", dry_run=True, status="COMPLETED")
session.add(batch)
with get_db_session(engine) as session:
queried = session.query(BatchRecord).filter_by(id="test-uuid-1").first()
assert queried is not None
assert queried.dry_run is True
assert queried.status == "COMPLETED"
def test_session_factory_caching_and_cleanup(tmp_path: Path):
from media_sorter.db import get_session_factory, _ENGINE_SESSION_FACTORIES
db_file = tmp_path / "test_cache.db"
engine = init_db(db_path=db_file)
factory1 = get_session_factory(engine)
factory2 = get_session_factory(engine)
assert factory1 is factory2
assert engine in _ENGINE_SESSION_FACTORIES
with get_db_session(engine) as session:
batch = BatchRecord(id="cached-1", dry_run=True, status="COMPLETED")
session.add(batch)
# Scoped session registry was cleared via remove()
assert factory1.registry.has() is False
+268
View File
@@ -0,0 +1,268 @@
from pathlib import Path
import pytest
from media_sorter.config import ActionType, ConflictPolicy, Settings
from media_sorter.db import get_db_session, init_db
from media_sorter.executor import MediaExecutor, PlannedOperation
from media_sorter.models import BatchRecord, Operation, OperationStatus
@pytest.fixture
def temp_env(tmp_path: Path):
db_path = tmp_path / "test.db"
engine = init_db(db_path=db_path)
src_dir = tmp_path / "incoming"
src_dir.mkdir()
dst_dir = tmp_path / "organized"
dst_dir.mkdir()
settings = Settings()
settings.database.path = str(db_path)
settings.storage.destination_base = str(dst_dir)
settings.storage.source_dirs = [str(src_dir)]
return settings, engine, src_dir, dst_dir
def test_dry_run_leaves_filesystem_untouched(temp_env):
settings, engine, src_dir, dst_dir = temp_env
test_file = src_dir / "sample_movie.mkv"
test_file.write_text("dummy content")
target_dst = dst_dir / "Movies/sample_movie.mkv"
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
plan = [
PlannedOperation(
src=test_file,
dst=target_dst,
action=ActionType.MOVE,
category="movie",
confidence=0.95,
)
]
report = executor.execute_batch(plan, dry_run=True)
assert report.dry_run is True
assert report.moved_files == 1
# File must still exist at src and not at dst
assert test_file.exists()
assert not target_dst.exists()
def test_live_atomic_move_and_rollback(temp_env):
settings, engine, src_dir, dst_dir = temp_env
settings.general.dry_run = False
test_file = src_dir / "song.flac"
test_file.write_text("flac audio bytes")
target_dst = dst_dir / "Music/Artist/Album/01 - song.flac"
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
plan = [
PlannedOperation(
src=test_file,
dst=target_dst,
action=ActionType.MOVE,
category="music",
confidence=0.95,
)
]
report = executor.execute_batch(plan, dry_run=False)
assert report.moved_files == 1
assert not test_file.exists()
assert target_dst.exists()
assert target_dst.read_text() == "flac audio bytes"
# Now execute rollback
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
reverted = executor.rollback_batch(report.batch_id)
assert reverted == 1
assert test_file.exists()
assert not target_dst.exists()
assert test_file.read_text() == "flac audio bytes"
def test_conflict_rename_unique(temp_env):
settings, engine, src_dir, dst_dir = temp_env
settings.general.dry_run = False
settings.conflicts.policy = ConflictPolicy.RENAME_UNIQUE
# Pre-create existing file at destination
target_dst = dst_dir / "Movies/Avatar (2009)/Avatar (2009).mkv"
target_dst.parent.mkdir(parents=True, exist_ok=True)
target_dst.write_text("original 1080p copy")
# New incoming file
incoming = src_dir / "Avatar.2009.2160p.mkv"
incoming.write_text("new 4k copy")
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
plan = [
PlannedOperation(
src=incoming,
dst=target_dst,
action=ActionType.MOVE,
category="movie",
confidence=0.95,
)
]
report = executor.execute_batch(plan, dry_run=False)
assert report.moved_files == 1
assert target_dst.exists()
assert target_dst.read_text() == "original 1080p copy"
# New file should have been renamed uniquely: Avatar (2009) (1).mkv
unique_dst = dst_dir / "Movies/Avatar (2009)/Avatar (2009) (1).mkv"
assert unique_dst.exists()
assert unique_dst.read_text() == "new 4k copy"
def test_conflict_replace_with_backup_and_rollback(temp_env):
settings, engine, src_dir, dst_dir = temp_env
settings.general.dry_run = False
settings.conflicts.policy = ConflictPolicy.REPLACE_IF_HIGHER_QUALITY
backup_dir = src_dir.parent / ".backup"
settings.conflicts.backup_dir = str(backup_dir)
target_dst = dst_dir / "Movies/Test.mkv"
target_dst.parent.mkdir(parents=True, exist_ok=True)
target_dst.write_text("old version")
incoming = src_dir / "Test.mkv"
incoming.write_text("upgraded high quality version")
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
plan = [
PlannedOperation(
src=incoming,
dst=target_dst,
action=ActionType.MOVE,
category="movie",
confidence=0.95,
)
]
report = executor.execute_batch(plan, dry_run=False)
assert target_dst.read_text() == "upgraded high quality version"
# Rollback should restore the old version from backup!
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
reverted = executor.rollback_batch(report.batch_id)
assert reverted == 1
assert target_dst.read_text() == "old version"
assert incoming.read_text() == "upgraded high quality version"
def test_process_locking(tmp_path: Path):
from media_sorter.executor import acquire_process_lock, ProcessLockError
lock_file = tmp_path / "test.lock"
with acquire_process_lock(lock_file):
# Trying to acquire same lock file concurrently must fail with ProcessLockError
with pytest.raises(ProcessLockError):
with acquire_process_lock(lock_file):
pass
# After exiting the lock block, it should be cleanly re-acquirable
with acquire_process_lock(lock_file):
pass
def test_cleanup_deletes_txt_files_and_removes_dirs(temp_env):
settings, engine, src_dir, dst_dir = temp_env
settings.general.dry_run = False
settings.general.cleanup_empty_dirs = True
# Setup source subdirectory with a movie, a .txt file, and a companion .txt file
sub_dir = src_dir / "Movie.Release.2023"
sub_dir.mkdir(parents=True, exist_ok=True)
movie_file = sub_dir / "movie.mkv"
movie_file.write_text("dummy video")
companion_txt = sub_dir / "movie.txt"
companion_txt.write_text("companion text")
readme_txt = sub_dir / "README.txt"
readme_txt.write_text("torrent info")
root_txt = src_dir / "root_note.txt"
root_txt.write_text("root text file")
target_dst = dst_dir / "Movies/Movie (2023)/movie.mkv"
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
plan = [
PlannedOperation(
src=movie_file,
dst=target_dst,
action=ActionType.MOVE,
category="movie",
confidence=0.95,
)
]
report = executor.execute_batch(plan, dry_run=False)
assert report.moved_files == 1
assert target_dst.exists()
# Verify movie is gone from src
assert not movie_file.exists()
# Verify .txt files were deleted during cleanup
assert not companion_txt.exists()
assert not readme_txt.exists()
assert not root_txt.exists()
# Verify the empty subdirectory was cleaned up (rmdir'd)
assert not sub_dir.exists()
# Verify source root itself was NOT removed
assert src_dir.exists()
def test_rollback_all_batches(temp_env):
settings, engine, src_dir, dst_dir = temp_env
settings.general.dry_run = False
file1 = src_dir / "sample1.mkv"
file1.write_text("file 1")
file2 = src_dir / "sample2.mkv"
file2.write_text("file 2")
dst1 = dst_dir / "Movies/Movie 1/sample1.mkv"
dst2 = dst_dir / "Movies/Movie 2/sample2.mkv"
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
executor.execute_batch(
[PlannedOperation(src=file1, dst=dst1, action=ActionType.MOVE, category="movie", confidence=0.9)],
dry_run=False
)
executor.execute_batch(
[PlannedOperation(src=file2, dst=dst2, action=ActionType.MOVE, category="movie", confidence=0.9)],
dry_run=False
)
assert not file1.exists()
assert not file2.exists()
assert dst1.exists()
assert dst2.exists()
with get_db_session(engine) as session:
executor = MediaExecutor(settings, session)
reverted = executor.rollback_all()
assert reverted == 2
assert file1.exists()
assert file2.exists()
assert not dst1.exists()
assert not dst2.exists()
+253
View File
@@ -0,0 +1,253 @@
from pathlib import Path
import pytest
from fastapi.testclient import TestClient
from media_sorter.config import Settings
from media_sorter.db import get_db_session, init_db
from media_sorter.library import (
clean_show_title,
get_known_shows,
list_library_items,
match_known_show,
record_detected_item,
sync_library_from_disk,
)
from media_sorter.server import cluster_unsure_files, create_app, inspect_downloads_folder
@pytest.fixture
def library_env(tmp_path: Path):
downloads = tmp_path / "downloads"
movies = tmp_path / "movies"
shows = tmp_path / "shows"
db_file = tmp_path / "test.db"
downloads.mkdir()
movies.mkdir()
shows.mkdir()
settings = Settings()
settings.storage.source_dirs = [str(downloads)]
settings.storage.destination_dirs.movies = str(movies)
settings.storage.destination_dirs.tv = str(shows)
settings.database.path = str(db_file)
settings.general.dry_run = False
settings.general.min_file_age_seconds = 0
engine = init_db(db_path=db_file)
test_env_file = tmp_path / ".env"
app = create_app(settings, engine, env_path=test_env_file)
client = TestClient(app)
return client, settings, engine, downloads, movies, shows
def test_library_sync_and_show_memory(library_env):
client, settings, engine, downloads, movies, shows = library_env
# 1. Populate disk with shows and movies
dexter_dir = shows / "Dexter" / "Season 01"
dexter_dir.mkdir(parents=True)
(dexter_dir / "Dexter - S01E01.mkv").write_bytes(b"\x00" * 100)
(dexter_dir / "Dexter - S01E02.mkv").write_bytes(b"\x00" * 100)
breaking_bad_dir = shows / "Breaking Bad" / "Season 01"
breaking_bad_dir.mkdir(parents=True)
(breaking_bad_dir / "Breaking Bad - S01E01.mkv").write_bytes(b"\x00" * 100)
movie_dir = movies / "Inception (2010)"
movie_dir.mkdir(parents=True)
(movie_dir / "Inception (2010).mkv").write_bytes(b"\x00" * 100)
with get_db_session(engine) as session:
sync_res = sync_library_from_disk(session, settings)
assert sync_res["shows_synced"] == 2
assert sync_res["movies_synced"] == 1
# 2. Check list_library_items
all_items = list_library_items(session)
assert all_items["total_shows"] == 2
assert all_items["total_movies"] == 1
shows_only = list_library_items(session, category="tv")
assert len(shows_only["shows"]) == 2
assert len(shows_only["movies"]) == 0
movies_only = list_library_items(session, category="movie")
assert len(movies_only["movies"]) == 1
search_res = list_library_items(session, search="dexter")
assert len(search_res["shows"]) == 1
assert search_res["shows"][0]["title"] == "Dexter"
# 3. Test known shows matching
known_shows = get_known_shows(session)
assert len(known_shows) == 2
matched = match_known_show("Dexter's Kill Room Extra.mkv", known_shows)
assert matched is not None
assert matched["title"] == "Dexter"
matched_bb = match_known_show("Breaking.Bad.Behind.The.Scenes.mp4", known_shows)
assert matched_bb is not None
assert matched_bb["title"] == "Breaking Bad"
assert match_known_show("Unrelated Movie.mkv", known_shows) is None
def test_cluster_unsure_files(library_env):
client, settings, engine, downloads, movies, shows = library_env
unsure_files = [
# Subfolder group (Folder: Dexter Extras)
{"name": "Interview.mkv", "relative_path": "Dexter Extras/Interview.mkv", "detected_type": "other"},
{"name": "Behind Scenes.mkv", "relative_path": "Dexter Extras/Behind Scenes.mkv", "detected_type": "other"},
# Common prefix group (Blood, Guts and Body Parts)
{"name": "Blood, Guts and Body Parts The Blood.mkv", "relative_path": "Blood, Guts and Body Parts The Blood.mkv", "detected_type": "other"},
{"name": "Blood, Guts and Body Parts The Props.mkv", "relative_path": "Blood, Guts and Body Parts The Props.mkv", "detected_type": "other"},
# Same title group (Inception)
{"name": "Inception.1080p.mkv", "relative_path": "Inception.1080p.mkv", "detected_type": "movie"},
{"name": "Inception.720p.mp4", "relative_path": "Inception.720p.mp4", "detected_type": "movie"},
# Standalone single file
{"name": "Solo Movie (2021).mkv", "relative_path": "Solo Movie (2021).mkv", "detected_type": "movie"},
]
groups, singles = cluster_unsure_files(unsure_files, settings)
group_names = [g["group_name"] for g in groups]
assert "Dexter Extras" in group_names
assert any("Blood" in gn for gn in group_names)
single_names = [s["name"] for s in singles]
assert "Solo Movie (2021).mkv" in single_names
def test_show_memory_routing_in_inspection(library_env):
client, settings, engine, downloads, movies, shows = library_env
# Record "Dexter" into library
with get_db_session(engine) as session:
record_detected_item(session, settings, "Dexter", "tv")
# Add a non-standard extra in downloads matching Dexter
dexter_extra = downloads / "Dexter's Kill Room Extra.mkv"
dexter_extra.write_bytes(b"\x00" * 100)
# Inspect downloads with engine
inspection = inspect_downloads_folder(downloads, settings, engine=engine)
assert len(inspection["shows"]) == 1
assert inspection["shows"][0]["show_name"] == "Dexter"
assert len(inspection["shows"][0]["files"]) == 1
def test_library_and_sort_group_api(library_env):
client, settings, engine, downloads, movies, shows = library_env
# 1. API: Rescan Library
dexter_dir = shows / "Dexter" / "Season 01"
dexter_dir.mkdir(parents=True)
(dexter_dir / "Dexter - S01E01.mkv").write_bytes(b"\x00" * 100)
rescan_resp = client.post("/api/library/rescan")
assert rescan_resp.status_code == 200
assert rescan_resp.json()["shows_synced"] >= 1
# 2. API: GET /api/library
lib_resp = client.get("/api/library")
assert lib_resp.status_code == 200
lib_data = lib_resp.json()
assert lib_data["total_shows"] >= 1
# 3. API: POST /api/files/sort-group
group_sub = downloads / "My Special Show"
group_sub.mkdir()
f1 = group_sub / "Episode 1.mkv"
f2 = group_sub / "Episode 2.mkv"
f1.write_bytes(b"\x00" * 100)
f2.write_bytes(b"\x00" * 100)
sort_group_resp = client.post("/api/files/sort-group", json={
"group_name": "My Special Show",
"group_type": "folder",
"category": "tv",
"title": "My Special Show",
"relative_paths": ["My Special Show/Episode 1.mkv", "My Special Show/Episode 2.mkv"],
})
assert sort_group_resp.status_code == 200
sort_data = sort_group_resp.json()
assert sort_data["status"] == "ok"
assert sort_data["moved_files"] == 2
# Verify files moved to shows directory
dest_show = shows / "My Special Show"
assert dest_show.exists()
# Verify newly sorted show was automatically recorded in the library
lib_after = client.get("/api/library?search=Special").json()
assert len(lib_after["shows"]) == 1
assert lib_after["shows"][0]["title"] == "My Special Show"
def test_one_piece_batch_sorting_and_downloads_inspection(library_env):
client, settings, engine, downloads, movies, shows = library_env
# 1. Create simulated One Piece files in downloads (unordered to test natural sorting)
s8 = downloads / "One Piece (0001-1071+Movies+Specials)" / "Season 08 - Water Seven (229-263)"
s8.mkdir(parents=True)
f233 = s8 / "[Anime Time] One Piece - 0233 - Pirate Abduction Incident!.mkv"
f232 = s8 / "[Anime Time] One Piece - 0232 - Galley-La Company!.mkv"
f1171 = downloads / "One.Piece.S01E1171.1080p.CR.WEB-DL.mkv"
f1073 = downloads / "[HatSubs] One Piece 1073 (BD 1080p 10-bit Opus) v2 [194B3FBA].mkv"
f233.write_bytes(b"\x00" * 100)
f232.write_bytes(b"\x00" * 100)
f1171.write_bytes(b"\x00" * 100)
f1073.write_bytes(b"\x00" * 100)
# 2. Inspect downloads folder via /api/files
files_res = client.get("/api/files").json()
dl_data = files_res["downloads"]
assert len(dl_data["shows"]) == 1
op_show = dl_data["shows"][0]
assert op_show["show_name"] == "One Piece"
assert op_show["count"] == 4
# Verify all episodes extracted and ordered naturally (232 before 233)
ep_map = {f["name"]: (f["season"], f["episode"]) for f in op_show["files"]}
assert ep_map["[Anime Time] One Piece - 0232 - Galley-La Company!.mkv"] == (8, 232)
assert ep_map["[Anime Time] One Piece - 0233 - Pirate Abduction Incident!.mkv"] == (8, 233)
assert ep_map["One.Piece.S01E1171.1080p.CR.WEB-DL.mkv"] == (1, 1171)
assert ep_map["[HatSubs] One Piece 1073 (BD 1080p 10-bit Opus) v2 [194B3FBA].mkv"] == (1, 1073)
# Verify natural order: 232 appears before 233
idx_232 = next(i for i, f in enumerate(op_show["files"]) if f["episode"] == 232)
idx_233 = next(i for i, f in enumerate(op_show["files"]) if f["episode"] == 233)
assert idx_232 < idx_233
# 3. Call manual sort batch API
payload = {
"show_name": "One Piece",
"files": [
{"relative_path": str(f232.relative_to(downloads)), "season": 8, "episode": 232},
{"relative_path": str(f233.relative_to(downloads)), "season": 8, "episode": 233},
{"relative_path": str(f1171.relative_to(downloads)), "season": 1, "episode": 1171},
{"relative_path": str(f1073.relative_to(downloads)), "season": 1, "episode": 1073},
]
}
batch_res = client.post("/api/files/manual-sort-batch", json=payload)
assert batch_res.status_code == 200
assert batch_res.json()["status"] == "ok"
assert batch_res.json()["moved_files"] == 4
# 4. Check destinations exist and are correctly named
dest_232 = shows / "One Piece" / "Season 08" / "One Piece - S08E232.mkv"
dest_233 = shows / "One Piece" / "Season 08" / "One Piece - S08E233.mkv"
dest_1073 = shows / "One Piece" / "Season 01" / "One Piece - S01E1073.mkv"
dest_1171 = shows / "One Piece" / "Season 01" / "One Piece - S01E1171.mkv"
assert dest_232.exists(), "Episode 232 must exist in Season 08"
assert dest_233.exists(), "Episode 233 must exist in Season 08"
assert dest_1073.exists(), "Episode 1073 must exist in Season 01"
assert dest_1171.exists(), "Episode 1171 must exist in Season 01"
# 5. Check library records One Piece show
lib = client.get("/api/library?search=One+Piece").json()
assert len(lib["shows"]) == 1
assert lib["shows"][0]["title"] == "One Piece"
+94
View File
@@ -0,0 +1,94 @@
from pathlib import Path
import pytest
from media_sorter.analyzer import MediaMetadata
from media_sorter.classifier import ClassificationResult
from media_sorter.config import Settings
from media_sorter.namer import MediaNamer, sanitize_filename_component
from media_sorter.tokenizer import TokenizedFilename
@pytest.fixture
def settings():
return Settings()
@pytest.fixture
def namer(settings):
return MediaNamer(settings)
def test_sanitize_filename_forbidden_chars():
messy = 'Movie: "The Final Chapter" <Director\'s Cut> | Part 1?.mkv'
cleaned = sanitize_filename_component(messy)
assert ":" not in cleaned
assert '"' not in cleaned
assert "<" not in cleaned
assert ">" not in cleaned
assert "|" not in cleaned
assert "?" not in cleaned
assert cleaned.endswith(".mkv")
def test_sanitize_windows_reserved_names():
res = sanitize_filename_component("CON.mp4")
assert res == "_CON.mp4"
res2 = sanitize_filename_component("nul.txt")
assert res2 == "_nul.txt"
def test_generate_movie_destination(namer):
tokens = TokenizedFilename(
raw_name="The.Matrix.1999.1080p.BluRay.x264.mkv",
title="The Matrix",
year=1999,
resolution="1080p",
video_codec="x264",
)
meta = MediaMetadata(path=Path("The.Matrix.1999.1080p.BluRay.x264.mkv"), mime_type="video/x-matroska", container="mkv")
cls_res = ClassificationResult(category="movie", confidence=0.95, tokens=tokens, metadata=meta)
dest = namer.generate_destination_path(cls_res)
assert "The Matrix (1999)" in str(dest)
assert dest.suffix == ".mkv"
def test_generate_tv_destination(namer):
tokens = TokenizedFilename(
raw_name="Breaking Bad S01E01 Pilot.mkv",
title="Breaking Bad",
season=1,
episode=1,
episode_title="Pilot",
)
meta = MediaMetadata(path=Path("Breaking Bad S01E01 Pilot.mkv"), mime_type="video/x-matroska", container="mkv")
cls_res = ClassificationResult(category="tv", confidence=0.95, tokens=tokens, metadata=meta)
dest = namer.generate_destination_path(cls_res)
assert "Season 01" in str(dest)
assert "Breaking Bad" in str(dest)
assert "S01E01" in str(dest)
def test_generate_quarantine_destination(namer):
meta = MediaMetadata(path=Path("weird_unknown_file.xyz"), mime_type="application/octet-stream", container="xyz")
cls_res = ClassificationResult(
category="unknown",
confidence=0.1,
metadata=meta,
needs_quarantine=True,
quarantine_reason="unrecognized_format",
)
dest = namer.generate_destination_path(cls_res)
assert "Quarantine" in str(dest)
assert "unrecognized-format" in str(dest) or "unrecognized_format" in str(dest)
def test_sidecar_subtitle_matching(namer):
sub_meta = MediaMetadata(path=Path("movie.en.srt"), mime_type="text/plain", container="srt")
cls_res = ClassificationResult(category="subtitle", confidence=0.95, metadata=sub_meta)
primary_dst = Path("/organized/Movies/Inception (2010)/Inception (2010) [1080p].mkv")
sub_dst = namer.generate_destination_path(cls_res, primary_dst_path=primary_dst)
assert sub_dst.parent == primary_dst.parent
assert sub_dst.name == "Inception (2010) [1080p].en.srt"
+112
View File
@@ -0,0 +1,112 @@
from pathlib import Path
import pytest
from media_sorter.db import get_db_session, init_db
from media_sorter.models import QuarantineRecord, QuarantineStatus
from media_sorter.quarantine import QuarantineManager
@pytest.fixture
def session(tmp_path: Path):
db_file = tmp_path / "test.db"
engine = init_db(db_path=db_file)
with get_db_session(engine) as s:
yield s
def test_quarantine_crud(session):
qm = QuarantineManager(session)
# Insert pending item
item = QuarantineRecord(
src="/incoming/ambiguous.avi",
suggested_category="movie",
confidence=0.55,
reason="Missing release year and technical tags",
status=QuarantineStatus.PENDING.value,
)
session.add(item)
session.commit()
pending = qm.list_pending()
assert len(pending) == 1
assert pending[0].src == "/incoming/ambiguous.avi"
# Resolve item
success = qm.resolve_item(pending[0].id, "tv", "/organized/TV/Show/ep.avi")
assert success is True
resolved = qm.get_by_id(pending[0].id)
assert resolved.status == QuarantineStatus.RESOLVED.value
assert resolved.suggested_category == "tv"
assert resolved.resolved_path == "/organized/TV/Show/ep.avi"
# Pending list should now be empty
assert len(qm.list_pending()) == 0
stats = qm.get_statistics()
assert stats["total"] == 1
assert stats["pending"] == 0
assert stats["resolved"] == 1
def test_quarantine_undo(session, tmp_path: Path):
qm = QuarantineManager(session)
# 1. Test unflagging a pending item
src_file = tmp_path / "pending_sample.mkv"
src_file.write_text("dummy")
item1 = QuarantineRecord(
src=str(src_file),
suggested_category="movie",
confidence=0.50,
reason="Low confidence",
status=QuarantineStatus.PENDING.value,
)
session.add(item1)
session.commit()
assert len(qm.list_pending()) == 1
# Undo pending item should delete/unflag it
success = qm.undo_item(item1.id)
assert success is True
assert len(qm.list_pending()) == 0
assert qm.get_by_id(item1.id) is None
# 2. Test undoing a resolved item (moves file back to src)
if src_file.exists():
src_file.unlink()
dst_file = tmp_path / "organized" / "Movie (2020)" / "Movie (2020).mkv"
dst_file.parent.mkdir(parents=True, exist_ok=True)
dst_file.write_text("movie data")
item2 = QuarantineRecord(
src=str(src_file),
suggested_category="movie",
confidence=0.60,
reason="Ambiguous",
status=QuarantineStatus.RESOLVED.value,
resolved_path=str(dst_file),
)
session.add(item2)
session.commit()
assert len(qm.list_resolved()) == 1
assert dst_file.exists()
assert not src_file.exists()
success2 = qm.undo_item(item2.id)
assert success2 is True
# File should be moved back to src
assert src_file.exists()
assert src_file.read_text() == "movie data"
assert not dst_file.exists()
# Record should now be back to PENDING with resolved fields reset
rec2 = qm.get_by_id(item2.id)
assert rec2.status == QuarantineStatus.PENDING.value
assert rec2.resolved_path is None
assert len(qm.list_pending()) == 1
assert len(qm.list_resolved()) == 0
+201
View File
@@ -0,0 +1,201 @@
import os
from pathlib import Path
import socket
import pytest
from fastapi.testclient import TestClient
from media_sorter.config import Settings
from media_sorter.db import get_db_session, init_db
from media_sorter.library import record_detected_item
from media_sorter.models import LibraryItem
from media_sorter.server import create_app
@pytest.fixture
def isolated_web_env(tmp_path: Path):
"""Isolated environment with temporary directories and SQLite database."""
downloads = tmp_path / "downloads"
movies = tmp_path / "movies"
shows = tmp_path / "shows"
db_file = tmp_path / "test.db"
downloads.mkdir()
movies.mkdir()
shows.mkdir()
settings = Settings()
settings.storage.source_dirs = [str(downloads)]
settings.storage.destination_dirs.movies = str(movies)
settings.storage.destination_dirs.tv = str(shows)
settings.database.path = str(db_file)
settings.general.dry_run = False
settings.general.min_file_age_seconds = 0
engine = init_db(db_path=db_file)
test_env_file = tmp_path / ".env"
app = create_app(settings, engine, env_path=test_env_file)
client = TestClient(app)
return client, settings, engine, downloads, movies, shows, test_env_file
# -----------------------------------------------------------------------------
# Test 1: Production Filesystem Safety Trap
# -----------------------------------------------------------------------------
def test_protect_production_filesystem_triggers_on_write(tmp_path: Path):
"""Verify protect_production_filesystem trap triggers RuntimeError on attempted
write or delete in /md0/jdownloads/illegal.txt, /md0/movies1, /md0/tv1.
"""
illegal_download = Path("/md0/jdownloads/illegal.txt")
with pytest.raises(RuntimeError, match="FILESYSTEM SAFETY TRAP"):
illegal_download.write_text("dangerous write")
with pytest.raises(RuntimeError, match="FILESYSTEM SAFETY TRAP"):
illegal_download.write_bytes(b"dangerous bytes")
with pytest.raises(RuntimeError, match="FILESYSTEM SAFETY TRAP"):
illegal_download.unlink()
with pytest.raises(RuntimeError, match="FILESYSTEM SAFETY TRAP"):
os.remove("/md0/jdownloads/illegal.txt")
with pytest.raises(RuntimeError, match="FILESYSTEM SAFETY TRAP"):
open("/md0/jdownloads/illegal.txt", "w")
# Verify protection for /md0/movies1 and /md0/tv1
with pytest.raises(RuntimeError, match="FILESYSTEM SAFETY TRAP"):
Path("/md0/movies1/illegal_movie.mkv").touch()
with pytest.raises(RuntimeError, match="FILESYSTEM SAFETY TRAP"):
os.makedirs("/md0/tv1/illegal_show/Season 01")
# Verify tmp_path operations succeed normally
safe_file = tmp_path / "safe.txt"
safe_file.write_text("allowed content")
assert safe_file.read_text() == "allowed content"
safe_file.unlink()
assert not safe_file.exists()
# -----------------------------------------------------------------------------
# Test 2: Test Environment Isolation
# -----------------------------------------------------------------------------
def test_isolate_test_environment_step_1_mutates_environment():
"""Step 1: mutate CONFIDENCE_THRESHOLD in os.environ and confirm Settings uses it."""
os.environ["CONFIDENCE_THRESHOLD"] = "0.99"
settings = Settings()
assert settings.general.confidence_threshold == 0.99
def test_isolate_test_environment_step_2_reverts_to_clean_defaults():
"""Step 2: verify isolate_test_environment restored os.environ and clean defaults."""
assert os.environ.get("CONFIDENCE_THRESHOLD") != "0.99"
assert "CONFIDENCE_THRESHOLD" not in os.environ
settings = Settings()
assert settings.general.confidence_threshold == 0.75
assert settings.general.worker_count == 4
# -----------------------------------------------------------------------------
# Test 3: /api/explorer/set-destination Latent NameError Fix
# -----------------------------------------------------------------------------
def test_explorer_set_destination_does_not_raise_name_error(isolated_web_env):
"""Verify /api/explorer/set-destination returns 200 without NameError."""
client, settings, engine, downloads, movies, shows, _ = isolated_web_env
resp = client.post(
"/api/explorer/set-destination",
json={"show_idx": 0, "destination": "/custom/destination/folder"},
)
assert resp.status_code == 200
data = resp.json()
assert data.get("status") == "ok"
assert data.get("destination") == "/custom/destination/folder"
assert data.get("success") is True
# Check validation for missing parameters
bad_resp = client.post("/api/explorer/set-destination", json={"show_idx": 0})
assert bad_resp.status_code == 400
# -----------------------------------------------------------------------------
# Test 4: Read-Only File Inspection Does Not Inflate item_count
# -----------------------------------------------------------------------------
def test_get_files_does_not_inflate_library_item_count(isolated_web_env):
"""Verify GET /api/files does not inflate LibraryItem.item_count on repeated calls."""
client, settings, engine, downloads, movies, shows, _ = isolated_web_env
# 1. Place a show episode in downloads
test_file = downloads / "Breaking.Bad.S01E01.1080p.mkv"
test_file.write_text("media bytes")
# Seed an existing library item with initial count 5
with get_db_session(engine) as session:
record_detected_item(
session,
settings,
"Breaking Bad",
"tv",
destination_folder=str(shows / "Breaking Bad"),
delta_count=5,
)
# Query before calling /api/files
with get_db_session(engine) as session:
item_before = (
session.query(LibraryItem).filter_by(title="Breaking Bad", category="tv").first()
)
assert item_before is not None
assert item_before.item_count == 5
# 2. Call GET /api/files repeatedly (5 times)
for _ in range(5):
resp = client.get("/api/files")
assert resp.status_code == 200
# 3. Verify item_count remained 5 and was NOT inflated
with get_db_session(engine) as session:
item_after = (
session.query(LibraryItem).filter_by(title="Breaking Bad", category="tv").first()
)
assert item_after is not None
assert item_after.item_count == 5
# -----------------------------------------------------------------------------
# Test 5: /api/files/scan Alias Routes (GET & POST)
# -----------------------------------------------------------------------------
def test_api_files_scan_alias_get_and_post(isolated_web_env):
"""Verify GET /api/files/scan and POST /api/files/scan return HTTP 200 with
identical data to GET /api/files.
"""
client, settings, engine, downloads, movies, shows, _ = isolated_web_env
(downloads / "Severance.S01E01.mkv").write_text("dummy video")
(movies / "Inception (2010)").mkdir(parents=True, exist_ok=True)
(movies / "Inception (2010)" / "Inception (2010).mkv").write_text("dummy movie")
resp_base = client.get("/api/files")
assert resp_base.status_code == 200
base_data = resp_base.json()
resp_scan_get = client.get("/api/files/scan")
assert resp_scan_get.status_code == 200
assert resp_scan_get.json() == base_data
resp_scan_post = client.post("/api/files/scan")
assert resp_scan_post.status_code == 200
assert resp_scan_post.json() == base_data
# Check top-level contract keys
assert "downloads" in base_data
assert "movies" in base_data
assert "shows" in base_data
# -----------------------------------------------------------------------------
# Bonus Test: External Network Trap
# -----------------------------------------------------------------------------
def test_block_external_network_trap():
"""Verify block_external_network prevents outbound external connections."""
s = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
with pytest.raises(RuntimeError, match="NETWORK ACCESS TRAP"):
s.connect(("8.8.8.8", 53))
+733
View File
@@ -0,0 +1,733 @@
from pathlib import Path
import pytest
from fastapi.testclient import TestClient
from media_sorter.config import Settings
from media_sorter.db import init_db
from media_sorter.server import create_app
@pytest.fixture
def web_env(tmp_path: Path):
downloads = tmp_path / "downloads"
movies = tmp_path / "movies"
shows = tmp_path / "shows"
db_file = tmp_path / "test.db"
downloads.mkdir()
movies.mkdir()
shows.mkdir()
settings = Settings()
settings.storage.source_dirs = [str(downloads)]
settings.storage.destination_dirs.movies = str(movies)
settings.storage.destination_dirs.tv = str(shows)
settings.database.path = str(db_file)
settings.general.dry_run = False
settings.general.min_file_age_seconds = 0
engine = init_db(db_path=db_file)
test_env_file = tmp_path / ".env"
app = create_app(settings, engine, env_path=test_env_file)
client = TestClient(app)
return client, settings, downloads, movies, shows, test_env_file
def test_web_status_and_dashboard(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# 1. HTML Dashboard renders cleanly
resp = client.get("/")
assert resp.status_code == 200
assert "Media Sorter" in resp.text
assert "Folder Explorer" in resp.text
# 2. Status API returns paths
status_resp = client.get("/api/status")
assert status_resp.status_code == 200
data = status_resp.json()
assert data["status"] == "online"
assert data["downloads_dir"] == str(downloads)
assert data["movies_dir"] == str(movies)
assert data["shows_dir"] == str(shows)
def test_web_sample_generation_and_sorting(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# 1. Generate sample downloads
sample_resp = client.post("/api/files/test-sample")
assert sample_resp.status_code == 200
sample_data = sample_resp.json()
assert len(sample_data["files"]) >= 5
# Verify files in downloads folder via API
files_resp = client.get("/api/files")
assert files_resp.status_code == 200
files_data = files_resp.json()
assert len(files_data["downloads"]["files"]) >= 5
# 2. Trigger Sorter Live execution via Web API
run_resp = client.post("/api/run", json={"dry_run": False})
assert run_resp.status_code == 200
run_data = run_resp.json()
assert run_data["moved_files"] >= 3
# Verify files moved to movies and shows
files_after = client.get("/api/files").json()
assert len(files_after["movies"]["files"]) >= 2
assert len(files_after["shows"]["files"]) >= 2
# 3. Trigger Rollback via Web API
rb_resp = client.post("/api/rollback", json={"batch_id": run_data["batch_id"]})
assert rb_resp.status_code == 200
rb_data = rb_resp.json()
assert rb_data["reverted_files"] >= 3
# Verify files restored to downloads folder
files_reverted = client.get("/api/files").json()
assert len(files_reverted["downloads"]["files"]) >= 4
# 4. Test deleting a single download file via Web API
file_to_del = sample_data["files"][0]
del_resp = client.delete(f"/api/files/download?name={file_to_del}")
assert del_resp.status_code == 200
assert del_resp.json()["status"] == "deleted"
files_post_del = client.get("/api/files").json()
assert len(files_post_del["downloads"]["files"]) == len(files_reverted["downloads"]["files"]) - 1
def test_web_settings_update(web_env, tmp_path: Path):
client, settings, downloads, movies, shows, test_env_file = web_env
new_downloads = tmp_path / "new_downloads"
new_movies = tmp_path / "new_movies"
update_payload = {
"downloads_dir": str(new_downloads),
"movies_dir": str(new_movies),
"dry_run": True,
"confidence_threshold": 0.80,
}
resp = client.post("/api/settings", json=update_payload)
assert resp.status_code == 200
# Verify updated settings
settings_resp = client.get("/api/settings")
assert settings_resp.status_code == 200
s_data = settings_resp.json()
assert s_data["downloads_dir"] == str(new_downloads)
assert s_data["movies_dir"] == str(new_movies)
assert s_data["dry_run"] is True
assert s_data["confidence_threshold"] == 0.80
# Verify test_env_file was written
assert test_env_file.is_file()
content = test_env_file.read_text(encoding="utf-8")
assert str(new_downloads) in content
assert str(new_movies) in content
def test_web_restart_endpoint(web_env, monkeypatch):
client, settings, downloads, movies, shows, test_env_file = web_env
# Prevent background task from exiting test process
monkeypatch.setattr("time.sleep", lambda _: None)
monkeypatch.setattr("os._exit", lambda _: None)
monkeypatch.setattr("os.execv", lambda *_: None)
resp = client.post("/api/restart")
assert resp.status_code == 200
assert resp.json()["status"] == "restarting"
def test_web_clear_batches_endpoint(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# Create sample files and run sort
client.post("/api/files/test-sample")
run_resp = client.post("/api/run", json={"dry_run": False})
assert run_resp.status_code == 200
batches = client.get("/api/batches").json()
assert len(batches) >= 1
# Clear batches
clear_resp = client.post("/api/batches/clear")
assert clear_resp.status_code == 200
assert clear_resp.json()["status"] == "cleared"
# Verify batches are now empty
batches_after = client.get("/api/batches").json()
assert len(batches_after) == 0
def test_folder_explorer_excludes_txt_files(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
from media_sorter.server import inspect_downloads_folder, list_files_in_dir
# Create real media files
(downloads / "Movie.2024.1080p.mkv").write_bytes(b"\x1aE\xdf\xa3" + b"\x00" * 100)
(shows / "Show.S01E01.mkv").write_bytes(b"\x1aE\xdf\xa3" + b"\x00" * 100)
# Create various .txt files in downloads and destination folders
(downloads / "Movie.2024.txt").write_text("info text")
(downloads / "readme.txt").write_text("readme text")
(downloads / "NOTES.TXT").write_text("notes text")
sub_dir = downloads / "Show.Release"
sub_dir.mkdir(parents=True, exist_ok=True)
(sub_dir / "tracker.txt").write_text("tracker info")
(shows / "show_notes.txt").write_text("notes")
# 1. inspect_downloads_folder must exclude all .txt files
inspected = inspect_downloads_folder(downloads, settings)
for f in inspected["files"]:
assert not f["name"].lower().endswith(".txt")
for s in inspected["shows"]:
for f in s["files"]:
assert not f["name"].lower().endswith(".txt")
for f in inspected["singles"]:
assert not f["name"].lower().endswith(".txt")
# 2. list_files_in_dir must exclude .txt files
shows_listed = list_files_in_dir(shows)
for f in shows_listed:
assert not f["name"].lower().endswith(".txt")
# 3. GET /api/files endpoint must exclude .txt files from folder explorer response
res = client.get("/api/files")
assert res.status_code == 200
data = res.json()
downloads_data = data["downloads"]
assert any(f["name"] == "Movie.2024.1080p.mkv" for f in downloads_data["files"])
assert not any(f["name"].lower().endswith(".txt") for f in downloads_data["files"])
assert not any(f["name"].lower().endswith(".txt") for f in downloads_data["singles"])
def test_folder_explorer_excludes_srt_files(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
from media_sorter.server import inspect_downloads_folder, list_files_in_dir
# Create real media files alongside .srt subtitles
(downloads / "Avatar.2009.1080p.mkv").write_bytes(b"\x1aE\xdf\xa3" + b"\x00" * 100)
(downloads / "Avatar.2009.1080p.en.srt").write_text("1\n00:00:01 --> 00:00:03\nSubtitles\n")
(downloads / "random_track.SRT").write_text("sub")
sub_dir = downloads / "Show.Episode.Folder"
sub_dir.mkdir(parents=True, exist_ok=True)
(sub_dir / "Episode 01.mkv").write_bytes(b"\x1aE\xdf\xa3" + b"\x00" * 100)
(sub_dir / "Episode 01.srt").write_text("sub")
(shows / "Show.S01E01.en.srt").write_text("sub")
# 1. inspect_downloads_folder must exclude all .srt files
inspected = inspect_downloads_folder(downloads, settings)
for f in inspected["files"]:
assert not f["name"].lower().endswith(".srt")
for s in inspected["shows"]:
for f in s["files"]:
assert not f["name"].lower().endswith(".srt")
for g in inspected["unsure_groups"]:
for f in g["files"]:
assert not f["name"].lower().endswith(".srt")
for f in inspected["singles"]:
assert not f["name"].lower().endswith(".srt")
# 2. list_files_in_dir must exclude .srt files
shows_listed = list_files_in_dir(shows)
for f in shows_listed:
assert not f["name"].lower().endswith(".srt")
# 3. GET /api/files endpoint must exclude .srt files from folder explorer response
res = client.get("/api/files")
assert res.status_code == 200
data = res.json()
downloads_data = data["downloads"]
assert any(f["name"] == "Avatar.2009.1080p.mkv" for f in downloads_data["files"])
assert not any(f["name"].lower().endswith(".srt") for f in downloads_data["files"])
assert not any(f["name"].lower().endswith(".srt") for f in downloads_data["singles"])
assert not any(f["name"].lower().endswith(".srt") for s in downloads_data["shows"] for f in s["files"])
assert not any(f["name"].lower().endswith(".srt") for g in downloads_data["unsure_groups"] for f in g["files"])
def test_delete_download_file_cleans_txt_and_parent_dir(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# Setup a subfolder in downloads with a file, a companion .txt, and an extra .txt
sub_dir = downloads / "Custom.Release.Folder"
sub_dir.mkdir(parents=True, exist_ok=True)
media = sub_dir / "sample.mkv"
media.write_text("sample")
companion_txt = sub_dir / "sample.txt"
companion_txt.write_text("companion")
extra_txt = sub_dir / "release.txt"
extra_txt.write_text("release note")
# Delete sample.mkv via API
del_res = client.delete(f"/api/files/download?name=Custom.Release.Folder/sample.mkv")
assert del_res.status_code == 200
assert del_res.json()["status"] == "deleted"
# Verify media, companion .txt, extra .txt, and empty parent subfolder are removed
assert not media.exists()
assert not companion_txt.exists()
assert not extra_txt.exists()
assert not sub_dir.exists()
assert downloads.exists()
def test_rollback_all_api(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# 1. Generate sample downloads and sort
client.post("/api/files/test-sample")
client.post("/api/run", json={"dry_run": False})
# Add extra file and sort again to form a second batch
f = downloads / "Extra.Film.2022.mkv"
f.write_text("extra movie")
client.post("/api/run", json={"dry_run": False})
# Call rollback all
rb_res = client.post("/api/rollback/all")
assert rb_res.status_code == 200
assert rb_res.json()["status"] == "ok"
assert rb_res.json()["reverted_files"] >= 4
# Verify files restored to downloads
files_res = client.get("/api/files").json()
assert len(files_res["downloads"]["files"]) >= 4
def test_manual_sort_file_api(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# Test sorting as movie with custom title & year
movie_file = downloads / "ambiguous_movie_file.mkv"
movie_file.write_text("movie payload")
res_movie = client.post("/api/files/manual-sort", json={
"relative_path": "ambiguous_movie_file.mkv",
"category": "movie",
"title": "Interstellar",
"year": 2014,
})
assert res_movie.status_code == 200
assert not movie_file.exists()
dest_movie = movies / "Interstellar (2014)" / "Interstellar (2014).mkv"
assert dest_movie.exists()
# Test sorting as TV show with custom title, season, episode
tv_file = downloads / "random_episode.mkv"
tv_file.write_text("tv payload")
res_tv = client.post("/api/files/manual-sort", json={
"relative_path": "random_episode.mkv",
"category": "tv",
"title": "Succession",
"season": 3,
"episode": 5,
})
assert res_tv.status_code == 200
assert not tv_file.exists()
dest_tv = shows / "Succession" / "Season 03" / "Succession - S03E05.mkv"
assert dest_tv.exists()
def test_quarantine_resolve_and_undo_api(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
from media_sorter.db import get_db_session
from media_sorter.models import QuarantineRecord, QuarantineStatus
from media_sorter.quarantine import QuarantineManager
quar_file = downloads / "unknown_sample.xyz"
quar_file.write_text("quarantine payload")
# Manually insert pending record
engine = init_db(db_path=settings.database.path)
with get_db_session(engine) as s:
qm = QuarantineManager(s)
rec = QuarantineRecord(
src=str(quar_file),
suggested_category="movie",
confidence=0.45,
reason="Unrecognized format",
status=QuarantineStatus.PENDING.value,
)
s.add(rec)
s.commit()
item_id = rec.id
# 1. GET /api/quarantine returns pending and resolved
q_res = client.get("/api/quarantine")
assert q_res.status_code == 200
q_data = q_res.json()
assert len(q_data["pending"]) == 1
assert q_data["pending"][0]["id"] == item_id
assert len(q_data["resolved"]) == 0
# 2. POST /api/quarantine/{id}/resolve with custom TV show title
res_resolve = client.post(f"/api/quarantine/{item_id}/resolve", json={
"category": "tv",
"title": "Severance",
"season": 1,
"episode": 1,
})
assert res_resolve.status_code == 200
assert not quar_file.exists()
dest_tv = shows / "Severance" / "Season 01" / "Severance - S01E01.xyz"
assert dest_tv.exists()
# Verify GET /api/quarantine reflects resolved item
q_res2 = client.get("/api/quarantine").json()
assert len(q_res2["pending"]) == 0
assert len(q_res2["resolved"]) == 1
# 3. POST /api/quarantine/{id}/undo restores file to src and marks PENDING
res_undo = client.post(f"/api/quarantine/{item_id}/undo")
assert res_undo.status_code == 200
assert res_undo.json()["status"] == "undone"
assert quar_file.exists()
assert not dest_tv.exists()
q_res3 = client.get("/api/quarantine").json()
assert len(q_res3["pending"]) == 1
assert len(q_res3["resolved"]) == 0
# 4. POST /api/quarantine/{id}/undo on pending item unflags/deletes it
res_unflag = client.post(f"/api/quarantine/{item_id}/undo")
assert res_unflag.status_code == 200
q_res4 = client.get("/api/quarantine").json()
assert len(q_res4["pending"]) == 0
def test_inspect_downloads_movie_subfolder_classified_as_movie(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# 1. Torrent-style movie inside a subfolder
movie_folder = downloads / "The.Dark.Knight.2008.1080p.BluRay.x264-ROVERS"
movie_folder.mkdir(parents=True, exist_ok=True)
movie_file = movie_folder / "The.Dark.Knight.2008.1080p.BluRay.x264-ROVERS.mkv"
movie_file.write_text("movie data")
# 2. Movie inside clean folder
opp_folder = downloads / "Oppenheimer (2023)"
opp_folder.mkdir(parents=True, exist_ok=True)
opp_file = opp_folder / "Oppenheimer.2023.2160p.mkv"
opp_file.write_text("oppenheimer data")
# 3. Legitimate TV show
tv_folder = downloads / "Breaking Bad Season 1"
tv_folder.mkdir(parents=True, exist_ok=True)
tv_file = tv_folder / "Breaking.Bad.S01E01.Pilot.mkv"
tv_file.write_text("tv show data")
# Query folder explorer
files_res = client.get("/api/files").json()
dl_info = files_res["downloads"]
# Movies must be in singles, not in shows
single_rel_paths = [s["relative_path"] for s in dl_info["singles"]]
assert "The.Dark.Knight.2008.1080p.BluRay.x264-ROVERS/The.Dark.Knight.2008.1080p.BluRay.x264-ROVERS.mkv" in single_rel_paths
assert "Oppenheimer (2023)/Oppenheimer.2023.2160p.mkv" in single_rel_paths
for s in dl_info["singles"]:
if "The.Dark.Knight" in s["relative_path"]:
assert s["detected_type"] == "movie"
assert "The Dark Knight" in s["believed_title"]
assert str(movies) in s["believed_destination"]
if "Oppenheimer" in s["relative_path"]:
assert s["detected_type"] == "movie"
assert "Oppenheimer" in s["believed_title"]
assert str(movies) in s["believed_destination"]
# Shows must only contain Breaking Bad, not the movies
show_names = [show["show_name"] for show in dl_info["shows"]]
assert "Breaking Bad" in show_names
assert "The Dark Knight" not in show_names
assert "The.Dark.Knight.2008.1080p.BluRay.x264-ROVERS" not in show_names
assert "Oppenheimer" not in show_names
def test_quarantine_bulk_resolve_and_bulk_undo_api(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
from media_sorter.db import get_db_session
from media_sorter.models import QuarantineRecord, QuarantineStatus
from media_sorter.quarantine import QuarantineManager
f1 = downloads / "Unknown.Movie.2021.1080p.mkv"
f1.write_text("movie1 payload")
f2 = downloads / "Another.Film.2023.720p.mkv"
f2.write_text("movie2 payload")
f3 = downloads / "Mystery.File.xyz"
f3.write_text("mystery payload")
engine = init_db(db_path=settings.database.path)
with get_db_session(engine) as s:
r1 = QuarantineRecord(src=str(f1), suggested_category="movie", confidence=0.4, reason="Low confidence", status=QuarantineStatus.PENDING.value)
r2 = QuarantineRecord(src=str(f2), suggested_category="movie", confidence=0.4, reason="Low confidence", status=QuarantineStatus.PENDING.value)
r3 = QuarantineRecord(src=str(f3), suggested_category="unknown", confidence=0.1, reason="Unrecognized format", status=QuarantineStatus.PENDING.value)
s.add_all([r1, r2, r3])
s.commit()
id1, id2, id3 = r1.id, r2.id, r3.id
# 1. Bulk resolve id1 and id2 as movie
res_b = client.post("/api/quarantine/bulk-resolve", json={
"item_ids": [id1, id2],
"category": "movie"
})
assert res_b.status_code == 200
assert res_b.json()["resolved_count"] == 2
assert not f1.exists()
assert not f2.exists()
assert f3.exists()
# 2. Bulk unflag id3
res_u = client.post("/api/quarantine/bulk-undo", json={
"item_ids": [id3],
"scope": "pending"
})
assert res_u.status_code == 200
assert res_u.json()["undone_count"] == 1
assert f3.exists() # Unflag leaves file in source
# Verify id3 is deleted from quarantine
q_data = client.get("/api/quarantine").json()
assert len(q_data["pending"]) == 0
assert len(q_data["resolved"]) == 2
# 3. Bulk undo all resolved items
res_all_undo = client.post("/api/quarantine/bulk-undo", json={
"scope": "resolved"
})
assert res_all_undo.status_code == 200
assert res_all_undo.json()["undone_count"] == 2
assert f1.exists()
assert f2.exists()
q_data_restored = client.get("/api/quarantine").json()
assert len(q_data_restored["pending"]) == 2
assert len(q_data_restored["resolved"]) == 0
def test_sort_show_endpoint_and_rollback(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# Create episodic show files and an unrelated file
ep1 = downloads / "Naruto Episode 001 Enter Naruto Uzumaki!.mkv"
ep2 = downloads / "Naruto Episode 002 My Name is Konohamaru!.mkv"
movie = downloads / "Inception (2010).mkv"
header = b"\x1aE\xdf\xa3" + b"\x00" * 300
ep1.write_bytes(header)
ep2.write_bytes(header)
movie.write_bytes(header)
# 1. Sort show specifically
res = client.post("/api/files/sort-show", json={
"show_name": "Naruto"
})
assert res.status_code == 200
data = res.json()
assert data["status"] == "ok"
assert data["moved_files"] == 2
assert data["show_name"] == "Naruto"
# Naruto episodes moved, movie remains in downloads
assert not ep1.exists()
assert not ep2.exists()
assert movie.exists()
# Destination directory contains organized show
naruto_dir = shows / "Naruto"
assert naruto_dir.exists()
organized_eps = list(naruto_dir.rglob("*.mkv"))
assert len(organized_eps) == 2
# 2. Rollback the show batch
batch_id = data["batch_id"]
rb_res = client.post("/api/rollback", json={"batch_id": batch_id})
assert rb_res.status_code == 200
assert rb_res.json()["reverted_files"] == 2
# Files restored to downloads
assert ep1.exists()
assert ep2.exists()
# 3. Non-existent show returns 404
err_res = client.post("/api/files/sort-show", json={
"show_name": "NonExistentShow"
})
assert err_res.status_code == 404
def test_dashboard_themes_and_rgb_feature(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
resp = client.get("/")
assert resp.status_code == 200
html = resp.text
# Verify all 30 themes are present in CSS
expected_themes = [
"cyber-dark", "oled-neon", "nord-frost", "dracula", "emerald-matrix",
"solar-sunset", "tokyo-night", "synthwave", "abyssal-ocean", "monokai-pro",
"terminal-crt", "paper-light", "neo-brutalism", "aurora-glass", "catppuccin-mocha",
"rose-pine", "gruvbox-dark", "solarized-dark", "nightowl", "vesper",
"bios-amber", "vapor-glitch", "mossy-stone", "crimson-eclipse", "blueprint-draft",
"neon-forest", "retro-retro", "golden-sand", "deep-space", "candy-cotton",
]
for theme_id in expected_themes:
assert f'[data-theme="{theme_id}"]' in html, f"Missing CSS for theme {theme_id}"
assert f'value="{theme_id}"' in html, f"Missing select option for theme {theme_id}"
assert f"id: '{theme_id}'" in html, f"Missing JS THEMES entry for theme {theme_id}"
# Verify RGB feature controls and styles
assert "rgb-mode" in html
assert "tab-rgb-mode-toggle" in html
assert "tab-rgb-speed-slider" in html
assert "toggleRGBMode" in html
assert "updateRGBSpeed" in html
assert "--rgb-duration" in html
def test_poster_local_storage_boundary_and_security(web_env, tmp_path: Path):
client, settings, downloads, movies, shows, test_env_file = web_env
# 1. Valid poster inside authorized storage root returns 200
valid_poster = shows / "poster.jpg"
valid_poster.write_bytes(b"\xff\xd8\xff\xe0" + b"image_data")
res_valid = client.get(f"/api/poster/local?path={valid_poster}")
assert res_valid.status_code == 200
# 2. Non-existent file inside storage root returns 404
missing_poster = shows / "missing.jpg"
res_missing = client.get(f"/api/poster/local?path={missing_poster}")
assert res_missing.status_code == 404
# 3. Path outside storage roots (traversal attempt) returns 403 Forbidden
outside_dir = tmp_path / "outside_secret"
outside_dir.mkdir()
secret_file = outside_dir / "secret.png"
secret_file.write_bytes(b"secret payload")
res_outside = client.get(f"/api/poster/local?path={secret_file}")
assert res_outside.status_code == 403
# 4. Non-image extension returns 400 Bad Request
text_file = downloads / "notes.txt"
text_file.write_text("not an image")
res_text = client.get(f"/api/poster/local?path={text_file}")
assert res_text.status_code == 400
def test_manual_sort_sanitization_and_reserved_names(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# 1. Path traversal in title is sanitized and contained
target_movie = downloads / "sample_movie.mkv"
target_movie.write_text("movie data")
res_traversal = client.post("/api/files/manual-sort", json={
"relative_path": "sample_movie.mkv",
"category": "movie",
"title": "../../../traversal_title",
"year": 2022,
})
assert res_traversal.status_code == 200
expected_dir = movies / "traversal_title (2022)"
assert expected_dir.exists()
assert (expected_dir / "traversal_title (2022).mkv").exists()
# 2. Windows reserved name in title (e.g. CON, AUX) is prefixed with underscore
target_tv = downloads / "con_show.mkv"
target_tv.write_text("tv data")
res_con = client.post("/api/files/manual-sort", json={
"relative_path": "con_show.mkv",
"category": "tv",
"title": "CON",
"season": 1,
"episode": 2,
})
assert res_con.status_code == 200
expected_tv_dir = shows / "_CON" / "Season 01"
assert expected_tv_dir.exists()
assert (expected_tv_dir / "_CON - S01E02.mkv").exists()
# 3. Pure forbidden characters title is rejected with 400
target_invalid = downloads / "invalid_file.mkv"
target_invalid.write_text("data")
res_bad = client.post("/api/files/manual-sort", json={
"relative_path": "invalid_file.mkv",
"category": "movie",
"title": ":::***???",
})
assert res_bad.status_code == 400
def test_process_locking_concurrency_409(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
from media_sorter.executor import acquire_process_lock
lock_file = settings.get_database_path().with_suffix(".lock")
with acquire_process_lock(lock_file):
# When lock is held, endpoints should return 409 Conflict
res_run = client.post("/api/run", json={"dry_run": True})
assert res_run.status_code == 409
res_rb = client.post("/api/rollback", json={"batch_id": "dummy"})
assert res_rb.status_code == 409
res_rba = client.post("/api/rollback/all")
assert res_rba.status_code == 409
dummy_file = downloads / "dummy.mkv"
dummy_file.write_text("test")
res_ms = client.post("/api/files/manual-sort", json={
"relative_path": "dummy.mkv",
"category": "movie",
"title": "Locked Movie",
})
assert res_ms.status_code == 409
def test_folder_explorer_correct_movie_and_show_buttons(web_env):
client, settings, downloads, movies, shows, test_env_file = web_env
# 1. HTML Dashboard contains correct movie & show buttons in Folder Explorer
resp = client.get("/")
assert resp.status_code == 200
assert "btn-correct-movie-explorer" in resp.text
assert "btn-correct-show-explorer" in resp.text
assert "bulk-btn-correct-movie" in resp.text
assert "bulk-btn-correct-show" in resp.text
assert "correctMovieByIndex" in resp.text
assert "correctShowByIndex" in resp.text
assert "correctMovieForShow" in resp.text
assert "correctMovieForGroup" in resp.text
assert "handleExplorerCorrectMovie" in resp.text
assert "handleExplorerCorrectShow" in resp.text
# 2. Test automatic/manual movie sort on a file in downloads
test_movie_file = downloads / "Unknown.Film.2023.1080p.mkv"
test_movie_file.write_bytes(b"dummy movie data")
sort_res = client.post("/api/files/manual-sort", json={
"relative_path": "Unknown.Film.2023.1080p.mkv",
"category": "movie",
"title": "Oppenheimer",
"year": 2023
})
assert sort_res.status_code == 200
expected_movie_file = movies / "Oppenheimer (2023)" / "Oppenheimer (2023).mkv"
assert expected_movie_file.exists()
# 3. Test automatic/manual show sort on a file in downloads
test_show_file = downloads / "Unknown.Show.S02E05.mkv"
test_show_file.write_bytes(b"dummy show data")
show_res = client.post("/api/files/manual-sort", json={
"relative_path": "Unknown.Show.S02E05.mkv",
"category": "tv",
"title": "Severance",
"season": 2,
"episode": 5
})
assert show_res.status_code == 200
expected_show_file = shows / "Severance" / "Season 02" / "Severance - S02E05.mkv"
assert expected_show_file.exists()
+164
View File
@@ -0,0 +1,164 @@
from pathlib import Path
import pytest
from media_sorter.tokenizer import FilenameTokenizer
@pytest.fixture
def tokenizer():
return FilenameTokenizer()
def test_tv_show_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("Breaking.Bad.S05E14.Ozymandias.1080p.BluRay.x264-ROVERS.mkv"))
assert tokens.is_episodic is True
assert tokens.title == "Breaking Bad"
assert tokens.season == 5
assert tokens.episode == 14
assert tokens.episode_title == "Ozymandias"
assert tokens.resolution == "1080p"
assert tokens.source == "BLURAY"
assert tokens.video_codec == "x264"
assert tokens.group == "ROVERS"
def test_tv_multi_episode(tokenizer):
tokens = tokenizer.tokenize(Path("Stranger.Things.S04E01-E02.Chapter.One.720p.WEB-DL.mkv"))
assert tokens.is_episodic is True
assert tokens.season == 4
assert tokens.episode == 1
assert tokens.multi_episodes == [1, 2]
assert tokens.resolution == "720p"
def test_anime_fansub_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("[SubsPlease] Frieren - Beyond Journey's End - 01 (1080p) [ABCD1234].mkv"))
assert tokens.is_anime is True
assert tokens.is_episodic is True
assert tokens.group == "SubsPlease"
assert "Frieren" in tokens.title
assert tokens.episode == 1
assert tokens.season == 1
def test_movie_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("Inception.2010.2160p.UHD.Remux.HEVC.TrueHD.Atmos-FraMeSToR.mkv"))
assert tokens.is_episodic is False
assert tokens.title == "Inception"
assert tokens.year == 2010
assert tokens.resolution == "2160p"
assert tokens.video_codec == "hevc"
assert tokens.audio_codec == "TRUEHD"
def test_music_track_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("/Music/Daft Punk - Discovery/02 - One More Time.flac"))
assert tokens.is_music is True
assert tokens.track == 2
assert tokens.title == "One More Time"
assert tokens.artist == "Daft Punk"
assert tokens.album == "Discovery"
def test_camera_and_date_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("IMG_20240815_142301.jpg"))
assert tokens.is_photo_or_home_video is True
assert tokens.date_stamp == "2024-08-15"
assert tokens.year == 2024
def test_podcast_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("The Daily - 2026-03-12 - The Sunday Read.mp3"))
assert tokens.artist == "The Daily"
assert tokens.year == 2026
assert tokens.date_stamp == "2026-03-12"
assert tokens.title == "The Sunday Read"
def test_movie_with_dimensions_not_episodic(tokenizer):
tokens = tokenizer.tokenize(Path("Interstellar.1920x1080.mkv"))
assert tokens.is_episodic is False
assert tokens.season is None
assert tokens.episode is None
assert tokens.resolution == "1080p"
assert "Interstellar" in tokens.title
tokens4k = tokenizer.tokenize(Path("Dune.Part.Two.3840x2160.mkv"))
assert tokens4k.is_episodic is False
assert tokens4k.season is None
assert tokens4k.episode is None
assert tokens4k.resolution == "2160p"
def test_movie_bracket_group_year_not_anime(tokenizer):
tokens = tokenizer.tokenize(Path("[YTS.MX] Movie Title - 2024 [1080p].mkv"))
assert tokens.is_anime is False
assert tokens.is_episodic is False
assert tokens.year == 2024
assert tokens.title == "Movie Title"
def test_standalone_episode_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("Naruto Episode 207 The Supposed Sealed Ability.mkv"))
assert tokens.is_episodic is True
assert tokens.title == "Naruto"
assert tokens.season == 1
assert tokens.episode == 207
assert tokens.episode_title == "The Supposed Sealed Ability"
def test_anime_fansub_without_group_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("BLEACH꞉ Sennen Kessen-hen - 27 E89717B7].mkv"))
assert tokens.is_anime is True
assert tokens.is_episodic is True
assert "BLEACH" in tokens.title
assert tokens.episode == 27
assert tokens.season == 1
def test_anime_ending_opening_tokenization(tokenizer):
tokens = tokenizer.tokenize(Path("[A&C] Sword Art Online Alicization S03ED01 [BDrip 1080p] [09F0EA6C].mkv"))
assert tokens.is_episodic is True
assert "Sword Art Online" in tokens.title
assert tokens.season == 3
assert tokens.episode == 1
def test_one_piece_tokenization_varieties(tokenizer):
# 1. Anime Time release with episode title and ancestor season folder
p1 = Path("One Piece (0001-1071+Movies+Specials)/Season 08 - Water Seven (229-263)/[Anime Time] One Piece - 0233 - Pirate Abduction Incident! A Pirate Ship That Can Only Await Its End! [1080p][HEVC 10bit x265][AAC].mkv")
t1 = tokenizer.tokenize(p1)
assert t1.title == "One Piece"
assert t1.season == 8
assert t1.episode == 233
assert "Pirate Abduction" in t1.episode_title
assert t1.group == "Anime Time"
# 2. Adjacent episode in same folder to ensure natural order
p2 = Path("One Piece (0001-1071+Movies+Specials)/Season 08 - Water Seven (229-263)/[Anime Time] One Piece - 0232 - Galley-La Company! A Grand Sight Dock 1!.mkv")
t2 = tokenizer.tokenize(p2)
assert t2.title == "One Piece"
assert t2.season == 8
assert t2.episode == 232
# 3. 4-digit episode S01E1171 WEB-DL
p3 = Path("One.Piece.S01E1171.1080p.CR.WEB-DL.AAC2.0.H.264.mkv")
t3 = tokenizer.tokenize(p3)
assert t3.title == "One Piece"
assert t3.season == 1
assert t3.episode == 1171
# 4. HatSubs release with 4-digit episode and CRC
p4 = Path("[HatSubs] One Piece 1073 (BD 1080p 10-bit Opus) v2 [194B3FBA].mkv")
t4 = tokenizer.tokenize(p4)
assert t4.title == "One Piece"
assert t4.episode == 1073
assert t4.crc32 == "194B3FBA"
# 5. Kaerizaki-Fansub underscore release
p5 = Path("[Kaerizaki-Fansub]_One_Piece_1076_[VOSTFR][FHD_1920x1080].mp4")
t5 = tokenizer.tokenize(p5)
assert t5.title == "One Piece"
assert t5.episode == 1076