Initial commit

This commit is contained in:
jkortis committed 2026-09-30 23:06:14 -04:00
commit 1d235d30e7
58 files changed
+19693

No files matched your search

+102
View File
@@ -0,0 +1,102 @@
import os
import time
from pathlib import Path
import pytest
from media_sorter.config import Settings
from media_sorter.db import get_db_session, init_db
from media_sorter.models import BatchRecord, Operation, QuarantineRecord
from media_sorter.sorter import MediaSorterApp
@pytest.fixture
def library_environment(tmp_path: Path):
incoming = tmp_path / "incoming"
organized = tmp_path / "organized"
db_file = tmp_path / "media_sorter.db"
incoming.mkdir()
organized.mkdir()
settings = Settings()
settings.storage.source_dirs = [str(incoming)]
settings.storage.destination_base = str(organized)
settings.storage.destination_dirs.movies = "Movies"
settings.storage.destination_dirs.tv = "TV Shows"
settings.database.path = str(db_file)
settings.general.min_file_age_seconds = 0 # Process immediately in test
engine = init_db(db_path=db_file)
return settings, engine, incoming, organized
def test_end_to_end_pipeline(library_environment):
settings, engine, incoming, organized = library_environment
# 1. Populate mock media files in incoming directory
movie_file = incoming / "The.Dark.Knight.2008.1080p.BluRay.x264.mkv"
movie_file.write_bytes(b"\x1aE\xdf\xa3" + b"\x00" * 200) # EBML header
sub_file = incoming / "The.Dark.Knight.2008.1080p.BluRay.x264.en.srt"
sub_file.write_text("1\n00:00:01,000 --> 00:00:03,000\nBatman begins.\n")
tv_file = incoming / "The.Wire.S01E01.Target.720p.mkv"
tv_file.write_bytes(b"\x1aE\xdf\xa3" + b"\x00" * 200)
photo_file = incoming / "IMG_20250620_153022.jpg"
photo_file.write_bytes(b"\xff\xd8\xff\xe0" + b"\x00" * 100) # JPEG header
junk_file = incoming / "unrecognized_sample.xyz"
junk_file.write_bytes(b"\x00\x01\x02\x03\x04")
sorter = MediaSorterApp(settings, engine)
# 2. First Run: Dry-Run
dry_report = sorter.run(dry_run=True)
assert dry_report.dry_run is True
assert dry_report.total_files >= 4
# Ensure source files were not modified
assert movie_file.exists()
assert sub_file.exists()
assert tv_file.exists()
assert photo_file.exists()
assert junk_file.exists()
# 3. Second Run: Live Organization
live_report = sorter.run(dry_run=False)
assert live_report.dry_run is False
assert live_report.moved_files >= 3
# Verify Movie and matched subtitle sidecar
movie_dest_dir = organized / "Movies/The Dark Knight (2008)"
assert movie_dest_dir.exists()
dest_movie = list(movie_dest_dir.glob("*.mkv"))[0]
assert "The Dark Knight" in dest_movie.name
# Subtitle should be alongside movie with .en.srt
dest_sub = list(movie_dest_dir.glob("*.srt"))[0]
assert dest_sub.name.endswith(".en.srt")
# Verify TV show organization
tv_season_dir = organized / "TV Shows/The Wire/Season 01"
assert tv_season_dir.exists()
dest_tv = list(tv_season_dir.glob("*.mkv"))[0]
assert "S01E01" in dest_tv.name
# Verify Photo organization
photo_dest_dir = organized / "Photos/2025/2025-06"
assert photo_dest_dir.exists()
# Verify Quarantine of unknown/low confidence file (flagged in place, not moved)
assert junk_file.exists()
from media_sorter.quarantine import QuarantineManager
with get_db_session(engine) as s:
qm = QuarantineManager(s)
pending = qm.list_pending()
assert any("unrecognized_sample.xyz" in q.src for q in pending)
# 4. Third Step: Rollback
reverted = sorter.rollback(live_report.batch_id)
assert reverted >= 3
# Verify files restored to incoming!
assert movie_file.exists()
assert sub_file.exists()
assert tv_file.exists()
@@ -0,0 +1,76 @@
from pathlib import Path
import pytest
from hypothesis import given, settings, strategies as st
from media_sorter.namer import FORBIDDEN_CHARS_PATTERN, RESERVED_NAMES, sanitize_filename_component
from media_sorter.tokenizer import FilenameTokenizer
@pytest.fixture
def tokenizer():
return FilenameTokenizer()
# -----------------------------------------------------------------------------
# Real-World Messy Scene & International Filenames
# -----------------------------------------------------------------------------
MESSY_CASES = [
(
"[HorribleSubs] Shingeki no Kyojin - 59 [1080p].mkv",
{"title": "Shingeki no Kyojin", "episode": 59, "is_anime": True},
),
(
"[SubsPlease] 葬送のフリーレン - 12 (1080p) [98E7B1A2].mkv",
{"title": "葬送のフリーレン", "episode": 12, "is_anime": True},
),
(
"Amélie.2001.PROPER.REMASTERED.1080p.BluRay.x264-CiNEFiLE.mkv",
{"title": "Amélie", "year": 2001, "resolution": "1080p"},
),
(
"Doctor.Who.2005.S01E01.Rose.720p.HDTV.x264-FoV.mkv",
{"title": "Doctor Who", "year": 2005, "season": 1, "episode": 1},
),
(
"Game of Thrones - 1x09 - Baelor [720p HDTV].mkv",
{"title": "Game of Thrones", "season": 1, "episode": 9},
),
(
"Mission.Impossible.Dead.Reckoning.Part.One.2023.2160p.WEB-DL.DDP5.1.Atmos.DV.HDR.H.265-FLUX.mkv",
{"year": 2023, "resolution": "2160p", "video_codec": "h265"},
),
]
@pytest.mark.parametrize("filename,expected", MESSY_CASES)
def test_real_world_messy_filenames(tokenizer, filename, expected):
tokens = tokenizer.tokenize(Path(filename))
for key, val in expected.items():
assert getattr(tokens, key) == val, f"Failed match for {key} in {filename}"
# -----------------------------------------------------------------------------
# Property-Based Fuzz Testing with Hypothesis
# -----------------------------------------------------------------------------
@given(st.text(min_size=1, max_size=500))
@settings(max_examples=150)
def test_sanitize_filename_component_fuzz(input_text):
result = sanitize_filename_component(input_text, max_length=120)
# 1. Result must never be empty
assert len(result) > 0
# 2. Result must contain no forbidden characters
assert not FORBIDDEN_CHARS_PATTERN.search(result)
# 3. Result must not have leading or trailing dots, spaces, or hyphens
assert not result.startswith((" ", ".", "-"))
assert not result.endswith((" ", ".", "-"))
# 4. Result base name must not be a Windows reserved device name
upper_base = result.split(".")[0].upper()
assert upper_base not in RESERVED_NAMES
# 5. Encoded byte length must stay within requested boundary (plus extension tolerance)
assert len(result.encode("utf-8")) <= 140