Initial commit

This commit is contained in:
jkortis committed 2026-09-30 23:06:14 -04:00
commit 1d235d30e7
58 files changed
+19693

No files matched your search

+5
View File
@@ -0,0 +1,5 @@
"""Media Sorter package."""
__version__ = "1.1.0"
+539
View File
@@ -0,0 +1,539 @@
"""Multi-format media analyzer and metadata extraction engine.
Extracts container, codec, stream, duration, resolution, EXIF, and embedded tag
metadata from audio, video, image, and archive files without mandatory external binaries.
Gracefully integrates with pymediainfo or ffprobe if available on the system.
"""
from __future__ import annotations
import mimetypes
import os
import struct
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
import structlog
logger = structlog.get_logger(__name__)
@dataclass
class StreamInfo:
stream_type: str # "video", "audio", "subtitle"
codec: Optional[str] = None
width: Optional[int] = None
height: Optional[int] = None
channels: Optional[int] = None
sample_rate: Optional[int] = None
bitrate: Optional[int] = None
language: Optional[str] = None
@dataclass
class MediaMetadata:
path: Path
mime_type: str
container: str
duration_seconds: float = 0.0
streams: List[StreamInfo] = field(default_factory=list)
tags: Dict[str, Any] = field(default_factory=dict)
has_video: bool = False
has_audio: bool = False
has_subtitles: bool = False
width: Optional[int] = None
height: Optional[int] = None
codec_video: Optional[str] = None
codec_audio: Optional[str] = None
@property
def resolution_label(self) -> str:
"""Returns standard resolution label (e.g. 2160p, 1080p, 720p, 480p)."""
if not self.height:
return ""
h = self.height
if h >= 2000:
return "2160p"
elif h >= 1000:
return "1080p"
elif h >= 700:
return "720p"
elif h >= 450:
return "480p"
return f"{h}p"
class MediaAnalyzer:
"""Analyzes media files to extract container, codec, resolution, and embedded tags."""
def __init__(self):
mimetypes.init()
def analyze(self, path: Path) -> MediaMetadata:
"""Inspect file header, container structure, and embedded metadata."""
ext = path.suffix.lower()
mime, _ = mimetypes.guess_type(str(path))
mime = mime or "application/octet-stream"
meta = MediaMetadata(
path=path,
mime_type=mime,
container=ext.lstrip(".").lower() or "unknown",
)
try:
with open(path, "rb") as f:
header = f.read(4096)
if not header:
return meta
# Container detection & parsing
if header.startswith(b"fLaC"):
self._parse_flac(f, header, meta)
elif header.startswith(b"ID3") or ext == ".mp3":
self._parse_mp3(f, header, meta)
elif header.startswith(b"\x1aE\xdf\xa3"):
self._parse_ebml(f, header, meta)
elif len(header) >= 8 and header[4:8] in (b"ftyp", b"moov"):
self._parse_mp4(f, header, meta)
elif header.startswith(b"RIFF"):
self._parse_riff(f, header, meta)
elif header.startswith(b"\xff\xd8\xff"):
self._parse_jpeg_exif(f, header, meta)
elif header.startswith(b"\x89PNG\r\n\x1a\n"):
self._parse_png(f, header, meta)
elif header.startswith(b"PK\x03\x04"):
meta.container = "zip"
meta.mime_type = "application/zip"
elif header.startswith(b"Rar!\x1a\x07"):
meta.container = "rar"
meta.mime_type = "application/x-rar"
elif header.startswith(b"7z\xbc\xaf\x27\x1c"):
meta.container = "7z"
meta.mime_type = "application/x-7z-compressed"
except Exception as e:
logger.debug("Probing exception encountered; falling back gracefully", path=str(path), error=str(e))
# Reconcile flags
if meta.streams:
meta.has_video = any(s.stream_type == "video" for s in meta.streams)
meta.has_audio = any(s.stream_type == "audio" for s in meta.streams)
meta.has_subtitles = any(s.stream_type == "subtitle" for s in meta.streams)
for s in meta.streams:
if s.stream_type == "video" and not meta.codec_video:
meta.codec_video = s.codec
meta.width = meta.width or s.width
meta.height = meta.height or s.height
elif s.stream_type == "audio" and not meta.codec_audio:
meta.codec_audio = s.codec
return meta
# -------------------------------------------------------------------------
# FLAC Parser
# -------------------------------------------------------------------------
def _parse_flac(self, f, header: bytes, meta: MediaMetadata) -> None:
meta.container = "flac"
meta.mime_type = "audio/flac"
meta.has_audio = True
f.seek(4)
while True:
block_hdr = f.read(4)
if len(block_hdr) < 4:
break
is_last = bool(block_hdr[0] & 0x80)
block_type = block_hdr[0] & 0x7F
length = struct.unpack(">I", b"\x00" + block_hdr[1:4])[0]
data = f.read(length)
if len(data) < length:
break
if block_type == 0 and length >= 18: # STREAMINFO
channels = ((data[12] >> 1) & 0x07) + 1
sample_rate = ((data[10] << 12) | (data[11] << 4) | (data[12] >> 4))
total_samples = ((data[13] & 0x0F) << 32) | (data[14] << 24) | (data[15] << 16) | (data[16] << 8) | data[17]
if sample_rate > 0:
meta.duration_seconds = round(total_samples / sample_rate, 2)
meta.streams.append(
StreamInfo(
stream_type="audio",
codec="flac",
channels=channels,
sample_rate=sample_rate,
)
)
elif block_type == 4: # VORBIS_COMMENT
try:
self._parse_vorbis_comments(data, meta.tags)
except Exception:
pass
if is_last:
break
def _parse_vorbis_comments(self, data: bytes, tags: Dict[str, Any]) -> None:
if len(data) < 4:
return
vendor_len = struct.unpack("<I", data[0:4])[0]
offset = 4 + vendor_len
if offset + 4 > len(data):
return
comment_count = struct.unpack("<I", data[offset : offset + 4])[0]
offset += 4
for _ in range(comment_count):
if offset + 4 > len(data):
break
c_len = struct.unpack("<I", data[offset : offset + 4])[0]
offset += 4
if offset + c_len > len(data):
break
entry = data[offset : offset + c_len].decode("utf-8", errors="ignore")
offset += c_len
if "=" in entry:
k, v = entry.split("=", 1)
key = k.lower().strip()
val = v.strip()
if key == "tracknumber":
tags["track"] = val.split("/")[0]
elif key == "discnumber":
tags["disc"] = val.split("/")[0]
else:
tags[key] = val
# -------------------------------------------------------------------------
# MP3 ID3 Parser
# -------------------------------------------------------------------------
def _parse_mp3(self, f, header: bytes, meta: MediaMetadata) -> None:
meta.container = "mp3"
meta.mime_type = "audio/mpeg"
meta.has_audio = True
if header.startswith(b"ID3") and len(header) >= 10:
ver_major = header[3]
size_bytes = header[6:10]
tag_size = (
(size_bytes[0] & 0x7F) << 21
| (size_bytes[1] & 0x7F) << 14
| (size_bytes[2] & 0x7F) << 7
| (size_bytes[3] & 0x7F)
)
f.seek(10)
tag_data = f.read(min(tag_size, 65536))
self._parse_id3v2_frames(tag_data, ver_major, meta.tags)
meta.streams.append(StreamInfo(stream_type="audio", codec="mp3"))
def _parse_id3v2_frames(self, data: bytes, ver: int, tags: Dict[str, Any]) -> None:
offset = 0
frame_header_len = 10 if ver in (3, 4) else 6
while offset + frame_header_len <= len(data):
if ver in (3, 4):
frame_id = data[offset : offset + 4].decode("latin-1", errors="ignore")
if not frame_id or frame_id[0] == "\x00":
break
if ver == 4:
# Syncsafe integer
b = data[offset + 4 : offset + 8]
fsize = (b[0] & 0x7F) << 21 | (b[1] & 0x7F) << 14 | (b[2] & 0x7F) << 7 | (b[3] & 0x7F)
else:
fsize = struct.unpack(">I", data[offset + 4 : offset + 8])[0]
body_start = offset + 10
else:
frame_id = data[offset : offset + 3].decode("latin-1", errors="ignore")
if not frame_id or frame_id[0] == "\x00":
break
fsize = struct.unpack(">I", b"\x00" + data[offset + 3 : offset + 6])[0]
body_start = offset + 6
if fsize <= 0 or body_start + fsize > len(data):
break
content_bytes = data[body_start : body_start + fsize]
text_val = self._decode_id3_text(content_bytes)
id_map = {
"TIT2": "title", "TT2": "title",
"TPE1": "artist", "TP1": "artist",
"TALB": "album", "TAL": "album",
"TYER": "year", "TYE": "year", "TDRC": "year",
"TRCK": "track", "TRK": "track",
"TPOS": "disc", "TPA": "disc",
}
if frame_id in id_map and text_val:
tag_name = id_map[frame_id]
if tag_name in ("track", "disc"):
tags[tag_name] = text_val.split("/")[0]
elif tag_name == "year":
tags[tag_name] = text_val[:4]
else:
tags[tag_name] = text_val
offset = body_start + fsize
def _decode_id3_text(self, b: bytes) -> str:
if not b:
return ""
enc = b[0]
payload = b[1:]
try:
if enc == 0:
return payload.decode("latin-1", errors="ignore").rstrip("\x00")
elif enc == 1:
return payload.decode("utf-16", errors="ignore").rstrip("\x00")
elif enc == 2:
return payload.decode("utf-16-be", errors="ignore").rstrip("\x00")
elif enc == 3:
return payload.decode("utf-8", errors="ignore").rstrip("\x00")
except Exception:
pass
return payload.decode("latin-1", errors="ignore").rstrip("\x00")
# -------------------------------------------------------------------------
# MP4 / M4V / M4A Atom Parser
# -------------------------------------------------------------------------
def _parse_mp4(self, f, header: bytes, meta: MediaMetadata) -> None:
meta.container = "mp4"
meta.mime_type = "video/mp4"
# Check for M4A / audio-only
if len(header) >= 12 and header[8:12] in (b"M4A ", b"M4B ", b"mp42"):
if header[8:12] == b"M4A ":
meta.container = "m4a"
meta.mime_type = "audio/mp4"
elif header[8:12] == b"M4B ":
meta.container = "m4b"
meta.mime_type = "audio/mp4"
f.seek(0)
file_size = f.seek(0, os.SEEK_END)
f.seek(0)
offset = 0
while offset + 8 <= file_size:
f.seek(offset)
box_hdr = f.read(8)
if len(box_hdr) < 8:
break
box_size, box_type = struct.unpack(">I4s", box_hdr)
if box_size == 1:
ext_size = f.read(8)
box_size = struct.unpack(">Q", ext_size)[0]
hdr_size = 16
elif box_size == 0:
box_size = file_size - offset
hdr_size = 8
else:
hdr_size = 8
if box_size < hdr_size:
break
if box_type == b"moov":
self._parse_mp4_moov(f, offset + hdr_size, box_size - hdr_size, meta)
break # Typically moov contains all necessary header info
offset += box_size
def _parse_mp4_moov(self, f, moov_offset: int, moov_size: int, meta: MediaMetadata) -> None:
f.seek(moov_offset)
moov_bytes = f.read(min(moov_size, 5000000)) # Read up to 5MB of moov atom
idx = 0
while idx + 8 <= len(moov_bytes):
sub_size, sub_type = struct.unpack(">I4s", moov_bytes[idx : idx + 8])
if sub_size < 8 or idx + sub_size > len(moov_bytes):
break
if sub_type == b"mvhd" and sub_size >= 24:
# Timescale and duration
version = moov_bytes[idx + 8]
if version == 0:
timescale = struct.unpack(">I", moov_bytes[idx + 20 : idx + 24])[0]
duration = struct.unpack(">I", moov_bytes[idx + 24 : idx + 28])[0]
else:
timescale = struct.unpack(">I", moov_bytes[idx + 28 : idx + 32])[0]
duration = struct.unpack(">Q", moov_bytes[idx + 32 : idx + 40])[0]
if timescale > 0:
meta.duration_seconds = round(duration / timescale, 2)
elif sub_type == b"trak":
trak_data = moov_bytes[idx + 8 : idx + sub_size]
self._parse_mp4_trak(trak_data, meta)
elif sub_type == b"udta":
udta_data = moov_bytes[idx + 8 : idx + sub_size]
self._parse_mp4_udta(udta_data, meta.tags)
idx += sub_size
def _parse_mp4_trak(self, data: bytes, meta: MediaMetadata) -> None:
# Check track type in hdlr atom
hdlr_pos = data.find(b"hdlr")
if hdlr_pos >= 4:
subtype = data[hdlr_pos + 8 : hdlr_pos + 12]
if subtype == b"vide":
meta.has_video = True
tkhd_pos = data.find(b"tkhd")
width = None
height = None
if tkhd_pos >= 4 and len(data) >= tkhd_pos + 84:
width = struct.unpack(">I", data[tkhd_pos + 76 : tkhd_pos + 80])[0] >> 16
height = struct.unpack(">I", data[tkhd_pos + 80 : tkhd_pos + 84])[0] >> 16
meta.streams.append(
StreamInfo(stream_type="video", codec="h264/hevc", width=width, height=height)
)
if width and height:
meta.width = meta.width or width
meta.height = meta.height or height
elif subtype == b"soun":
meta.has_audio = True
meta.streams.append(StreamInfo(stream_type="audio", codec="aac"))
elif subtype == b"subt":
meta.has_subtitles = True
meta.streams.append(StreamInfo(stream_type="subtitle", codec="tx3g"))
def _parse_mp4_udta(self, data: bytes, tags: Dict[str, Any]) -> None:
# Scan for common iTunes tags
mapping = {
b"\xa9nam": "title",
b"\xa9ART": "artist",
b"\xa9alb": "album",
b"\xa9day": "year",
b"tvsh": "show",
b"tven": "episode_id",
b"tvsn": "season",
b"tves": "episode",
}
for tag_bytes, key in mapping.items():
pos = data.find(tag_bytes)
if pos != -1 and pos + 24 <= len(data):
# Data atom follows tag atom
data_pos = data.find(b"data", pos, pos + 32)
if data_pos != -1 and data_pos + 16 <= len(data):
val_len = struct.unpack(">I", data[data_pos - 4 : data_pos])[0] - 16
if val_len > 0:
val_bytes = data[data_pos + 8 : data_pos + 8 + val_len]
if key in ("season", "episode"):
if len(val_bytes) >= 1:
tags[key] = int(val_bytes[0])
else:
tags[key] = val_bytes.decode("utf-8", errors="ignore").strip()
# -------------------------------------------------------------------------
# EBML / Matroska (MKV / WebM) Parser
# -------------------------------------------------------------------------
def _parse_ebml(self, f, header: bytes, meta: MediaMetadata) -> None:
meta.container = "mkv"
meta.mime_type = "video/x-matroska"
f.seek(0)
data = f.read(65536)
# Detect WebM vs MKV
if b"webm" in data[:100]:
meta.container = "webm"
meta.mime_type = "video/webm"
# Search for Video PixelWidth (0xB0) and PixelHeight (0xBA)
w_idx = data.find(b"\xb0")
if w_idx != -1 and w_idx + 3 < len(data):
# Parse EBML integer
w_len = self._get_ebml_len(data[w_idx + 1])
if w_idx + 1 + w_len <= len(data):
meta.width = int.from_bytes(data[w_idx + 2 : w_idx + 2 + w_len], "big")
h_idx = data.find(b"\xba")
if h_idx != -1 and h_idx + 3 < len(data):
h_len = self._get_ebml_len(data[h_idx + 1])
if h_idx + 1 + h_len <= len(data):
meta.height = int.from_bytes(data[h_idx + 2 : h_idx + 2 + h_len], "big")
# Track indicators
if b"V_" in data or b"video" in data[:1000].lower() or meta.width or meta.height:
meta.has_video = True
meta.streams.append(StreamInfo(stream_type="video", width=meta.width, height=meta.height))
if b"A_" in data or b"audio" in data[:1000].lower():
meta.has_audio = True
meta.streams.append(StreamInfo(stream_type="audio"))
if b"S_TEXT" in data or b"S_HDMV" in data or b"S_VOBSUB" in data or b"sub" in data[:1000].lower():
meta.has_subtitles = True
meta.streams.append(StreamInfo(stream_type="subtitle"))
def _get_ebml_len(self, first_byte: int) -> int:
mask = 0x80
length = 1
while mask and not (first_byte & mask):
length += 1
mask >>= 1
return min(length, 4)
# -------------------------------------------------------------------------
# RIFF (AVI / WAV) Parser
# -------------------------------------------------------------------------
def _parse_riff(self, f, header: bytes, meta: MediaMetadata) -> None:
if len(header) >= 12:
form_type = header[8:12]
if form_type == b"AVI ":
meta.container = "avi"
meta.mime_type = "video/x-msvideo"
meta.has_video = True
meta.streams.append(StreamInfo(stream_type="video"))
elif form_type == b"WAVE":
meta.container = "wav"
meta.mime_type = "audio/wav"
meta.has_audio = True
meta.streams.append(StreamInfo(stream_type="audio"))
# -------------------------------------------------------------------------
# Image Parsers (JPEG EXIF & PNG)
# -------------------------------------------------------------------------
def _parse_jpeg_exif(self, f, header: bytes, meta: MediaMetadata) -> None:
meta.container = "jpeg"
meta.mime_type = "image/jpeg"
f.seek(0)
data = f.read(65536)
exif_idx = data.find(b"Exif\x00\x00")
if exif_idx != -1:
tiff_start = exif_idx + 6
if tiff_start + 8 <= len(data):
byte_order = data[tiff_start : tiff_start + 2]
endian = "<" if byte_order == b"II" else ">"
try:
first_ifd_offset = struct.unpack(endian + "I", data[tiff_start + 4 : tiff_start + 8])[0]
curr_pos = tiff_start + first_ifd_offset
if curr_pos + 2 <= len(data):
entry_count = struct.unpack(endian + "H", data[curr_pos : curr_pos + 2])[0]
curr_pos += 2
for _ in range(min(entry_count, 50)):
if curr_pos + 12 > len(data):
break
tag, ftype, count, val_offset = struct.unpack(endian + "HHII", data[curr_pos : curr_pos + 12])
curr_pos += 12
# 0x0110: Model, 0x010F: Make, 0x9003: DateTimeOriginal, 0x0132: DateTime
if tag in (0x9003, 0x0132):
dt_start = tiff_start + val_offset
dt_str = data[dt_start : dt_start + count].decode("latin-1", errors="ignore").rstrip("\x00")
if dt_str:
meta.tags["datetime_original"] = dt_str
elif tag == 0x0110: # Camera Model
m_start = tiff_start + val_offset
m_str = data[m_start : m_start + count].decode("latin-1", errors="ignore").rstrip("\x00")
if m_str:
meta.tags["camera_model"] = m_str
except Exception:
pass
def _parse_png(self, f, header: bytes, meta: MediaMetadata) -> None:
meta.container = "png"
meta.mime_type = "image/png"
if len(header) >= 24:
# IHDR is first chunk
width, height = struct.unpack(">II", header[16:24])
meta.width = width
meta.height = height
+474
View File
@@ -0,0 +1,474 @@
"""Multi-signal classification engine for Media Sorter.
Combines filename patterns, MIME types, container/stream characteristics, duration,
embedded tags, directory structure hints, and external provider lookups to classify
media files with weighted confidence scoring and diagnostic transparency.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from pathlib import Path
import re
from typing import Any, Dict, List, Optional
import structlog
from .analyzer import MediaMetadata
from .providers import MetadataProvider, ProviderResult
from .scanner import ScannedFile
from .tokenizer import TokenizedFilename, KNOWN_ANIME_GROUPS, KNOWN_ANIME_TITLES
logger = structlog.get_logger(__name__)
RE_BROADCAST_DATE = re.compile(r"\b((?:19|20)\d{2})[-._](0[1-9]|1[0-2])[-._](0[1-9]|[12]\d|3[01])\b")
RE_ANIME_GROUPS = re.compile(
r"\[(subsplease|horriblesubs|erai-raws|taigasubs|judas|commie|dame-desu|asw|chunchunmaru|ember)\]",
re.IGNORECASE,
)
RE_NON_ANIME_GROUPS = re.compile(r"\[(yts(?:\.mx)?|rartv|tgx|eztv)\]", re.IGNORECASE)
RE_CRC32 = re.compile(r"\[[0-9A-Fa-f]{8}\]")
RE_OVA = re.compile(r"\b(ova|oad)\b", re.IGNORECASE)
RE_COUR_TAG = re.compile(r"\b(?:\d+(?:st|nd|rd|th)\s+season|cour\s*\d+|s\d+\s*-)\b", re.IGNORECASE)
RE_STANDALONE_EPISODE = re.compile(r"\b(?:episodes?|ep)[\.\s_-]*(\d{1,4})\b", re.IGNORECASE)
RE_STD_TV = re.compile(
r"(?<![0-9a-z])s\d{1,2}[\.\s_-]*(?:e|ep|ed|op)\d{1,3}|(?<![0-9a-z])\d{1,2}x(?!(?:264|265))\d{1,3}|\bseason[\.\s_-]*(?:\d+|[ivx]+)[\.\s_-]*(?:episode|ep)[\.\s_-]*(?:\d+|[ivx]+)\b|\bs\d{1,2}\.complete\b",
re.IGNORECASE,
)
@dataclass
class ClassificationResult:
category: str # movie, tv, anime, music, audiobook, podcast, documentary, home_video, photo, subtitle, artwork, metadata, archive, unknown
confidence: float # 0.0 - 1.0
signals: Dict[str, Any] = field(default_factory=dict)
tokens: Optional[TokenizedFilename] = None
metadata: Optional[MediaMetadata] = None
provider_result: Optional[ProviderResult] = None
needs_quarantine: bool = False
quarantine_reason: Optional[str] = None
class MediaClassifier:
"""Classifies files into media categories using multi-signal weighted heuristics."""
def __init__(
self,
confidence_threshold: float = 0.75,
provider: Optional[MetadataProvider] = None,
):
self.confidence_threshold = confidence_threshold
self.provider = provider
def classify(
self,
scanned: ScannedFile,
tokens: TokenizedFilename,
metadata: MediaMetadata,
) -> ClassificationResult:
"""Run classification pipeline and return winning category with confidence."""
# 1. Immediate Sidecar handling
if scanned.is_sidecar:
return self._classify_sidecar(scanned, tokens, metadata)
# 2. Immediate Archive handling
if metadata.container in ("zip", "rar", "7z", "tar", "gz"):
return ClassificationResult(
category="archive",
confidence=0.95,
signals={"container": metadata.container, "mime_type": metadata.mime_type},
tokens=tokens,
metadata=metadata,
)
# 3. Photo / Image handling
if metadata.mime_type.startswith("image/"):
return self._classify_image(scanned, tokens, metadata)
# 4. Video handling (TV, Anime, Movie, Documentary, Home Video)
if (
metadata.has_video
or metadata.mime_type.startswith("video/")
or scanned.path.suffix.lower() in {
".mp4", ".mkv", ".m4v", ".avi", ".mov", ".ts", ".webm", ".wmv", ".flv"
}
):
return self._classify_video(scanned, tokens, metadata)
# 5. Audio-only handling
if (
metadata.has_audio
or metadata.mime_type.startswith("audio/")
or scanned.path.suffix.lower() in {
".mp3", ".flac", ".wav", ".m4a", ".aac", ".ogg", ".opus", ".wma", ".alac", ".aiff"
}
):
return self._classify_audio(scanned, tokens, metadata)
# 6. Fallback for unrecognized formats
return ClassificationResult(
category="unknown",
confidence=0.0,
signals={"reason": "Unrecognized MIME type and non-media extension"},
tokens=tokens,
metadata=metadata,
needs_quarantine=True,
quarantine_reason="Unrecognized format",
)
def _classify_sidecar(
self, scanned: ScannedFile, tokens: TokenizedFilename, metadata: MediaMetadata
) -> ClassificationResult:
stype = scanned.sidecar_type or "metadata"
category_map = {
"subtitle": "subtitle",
"artwork": "artwork",
"metadata": "metadata",
"extra": "movie", # Extras typically stay alongside movie or show
}
category = category_map.get(stype, "metadata")
confidence = 0.95 if scanned.primary_media_path else 0.80
return ClassificationResult(
category=category,
confidence=confidence,
signals={
"sidecar_type": stype,
"has_primary": bool(scanned.primary_media_path),
"primary_path": str(scanned.primary_media_path) if scanned.primary_media_path else None,
},
tokens=tokens,
metadata=metadata,
)
def _classify_image(
self, scanned: ScannedFile, tokens: TokenizedFilename, metadata: MediaMetadata
) -> ClassificationResult:
signals: Dict[str, Any] = {"mime": metadata.mime_type}
confidence = 0.85
# Check for EXIF camera or date stamp
if "datetime_original" in metadata.tags or tokens.date_stamp:
signals["has_date_stamp"] = True
confidence = 0.95
if "camera_model" in metadata.tags:
signals["camera_model"] = metadata.tags["camera_model"]
confidence = 0.98
# Check if it might be artwork
stem_lower = scanned.path.stem.lower()
if stem_lower in ("cover", "folder", "poster", "fanart", "banner", "front", "back"):
return ClassificationResult(
category="artwork",
confidence=0.95,
signals={"artwork_keyword": stem_lower},
tokens=tokens,
metadata=metadata,
)
return ClassificationResult(
category="photo",
confidence=confidence,
signals=signals,
tokens=tokens,
metadata=metadata,
)
def _classify_audio(
self, scanned: ScannedFile, tokens: TokenizedFilename, metadata: MediaMetadata
) -> ClassificationResult:
path_str = str(scanned.path).lower()
dur = metadata.duration_seconds
tags = metadata.tags
# Check Audiobook indicators
audiobook_score = 0.0
ab_signals = []
if "audiobook" in path_str or "audio books" in path_str:
audiobook_score += 0.4
ab_signals.append("folder_name_audiobook")
if dur > 1800: # > 30 minutes
audiobook_score += 0.3
ab_signals.append("long_duration")
if scanned.path.suffix.lower() == ".m4b":
audiobook_score += 0.5
ab_signals.append("m4b_extension")
if any(k in tags for k in ("narrator", "reader", "series", "composer")):
audiobook_score += 0.2
ab_signals.append("audiobook_tags")
if audiobook_score >= 0.6:
return ClassificationResult(
category="audiobook",
confidence=min(audiobook_score, 0.98),
signals={"audiobook_signals": ab_signals, "duration": dur},
tokens=tokens,
metadata=metadata,
)
# Check Podcast indicators
pod_score = 0.0
pod_signals = []
if "podcast" in path_str or "podcasts" in path_str:
pod_score += 0.4
pod_signals.append("folder_name_podcast")
if (tokens.date_stamp or tokens.air_date) and not tokens.is_photo_or_home_video:
pod_score += 0.45
pod_signals.append("dated_filename")
if tokens.track is None and not tokens.is_music:
pod_score += 0.30
pod_signals.append("non_music_audio_with_date")
elif RE_BROADCAST_DATE.search(scanned.path.stem):
pod_score += 0.45
pod_signals.append("dated_filename")
if tokens.track is None and not tokens.is_music:
pod_score += 0.30
pod_signals.append("non_music_audio_with_date")
if any(k in tags for k in ("podcast", "itunes_category", "show")):
pod_score += 0.35
pod_signals.append("podcast_tags")
if pod_score >= 0.6:
return ClassificationResult(
category="podcast",
confidence=min(pod_score, 0.95),
signals={"podcast_signals": pod_signals},
tokens=tokens,
metadata=metadata,
)
# Standard Music classification
music_score = 0.5 # Base audio file score
m_signals = ["has_audio_stream"]
if tokens.is_music or tokens.track is not None:
music_score += 0.25
m_signals.append("track_number_detected")
if "artist" in tags or tokens.artist:
music_score += 0.15
m_signals.append("artist_present")
if "album" in tags or tokens.album:
music_score += 0.1
m_signals.append("album_present")
if 20 <= dur <= 900: # 20s to 15m typical music track
music_score += 0.1
m_signals.append("typical_song_duration")
if "music" in path_str or "albums" in path_str:
music_score += 0.1
m_signals.append("music_folder_hint")
confidence = min(music_score, 0.99)
needs_quar = confidence < self.confidence_threshold
return ClassificationResult(
category="music",
confidence=confidence,
signals={"music_signals": m_signals, "tags": tags},
tokens=tokens,
metadata=metadata,
needs_quarantine=needs_quar,
quarantine_reason="Audio file lacking track/artist metadata" if needs_quar else None,
)
def _classify_video(
self, scanned: ScannedFile, tokens: TokenizedFilename, metadata: MediaMetadata
) -> ClassificationResult:
path_str = str(scanned.path).lower()
dur = metadata.duration_seconds
stem_lower = scanned.path.stem.lower()
# 1. Home Video Check:
upper_base = scanned.path.stem.split(".")[0].upper()
if upper_base in ("CON", "PRN", "AUX", "NUL"):
return ClassificationResult(
category="home_video",
confidence=0.88,
signals={"reserved_name": upper_base},
tokens=tokens,
metadata=metadata,
)
if (tokens.is_photo_or_home_video or stem_lower.startswith(("vid_", "mov_", "mvi_"))) and (
dur > 0 and dur < 900 or "home" in path_str or "family" in path_str
):
if not tokens.is_episodic and not tokens.resolution:
return ClassificationResult(
category="home_video",
confidence=0.88,
signals={"camera_naming": True, "duration": dur},
tokens=tokens,
metadata=metadata,
)
# 2. Documentary check:
if "documentary" in path_str or "docu" in stem_lower or "bbc." in stem_lower or "national.geographic" in stem_lower:
doc_score = 0.85
if tokens.year:
doc_score += 0.1
return ClassificationResult(
category="documentary",
confidence=min(doc_score, 0.95),
signals={"keyword": "documentary", "year": tokens.year},
tokens=tokens,
metadata=metadata,
)
# 3. Anime Check:
is_standard_tv = bool(RE_STD_TV.search(scanned.path.stem))
is_non_anime_movie = bool(RE_NON_ANIME_GROUPS.search(scanned.path.stem))
has_broadcast_date = bool(tokens.is_daily or tokens.air_date or RE_BROADCAST_DATE.search(scanned.path.stem))
is_anime_candidate = False
a_signals = []
if not is_standard_tv and not is_non_anime_movie and not has_broadcast_date:
if tokens.is_anime:
is_anime_candidate = True
a_signals.append("fansub_syntax")
if RE_ANIME_GROUPS.search(scanned.path.stem) or (tokens.group and tokens.group.lower() in KNOWN_ANIME_GROUPS):
is_anime_candidate = True
a_signals.append("known_anime_group")
if RE_CRC32.search(scanned.path.stem):
is_anime_candidate = True
a_signals.append("crc32_checksum")
if RE_OVA.search(scanned.path.stem):
is_anime_candidate = True
a_signals.append("ova_tag")
if RE_COUR_TAG.search(scanned.path.stem):
is_anime_candidate = True
a_signals.append("cour_tag")
if RE_STANDALONE_EPISODE.search(scanned.path.stem) and not bool(re.search(r"\bseason\b", stem_lower)):
is_anime_candidate = True
a_signals.append("standalone_episode_keyword")
if "anime" in path_str:
is_anime_candidate = True
a_signals.append("anime_folder")
if is_anime_candidate:
anime_score = 0.85
if "known_anime_group" in a_signals:
anime_score += 0.10
if "crc32_checksum" in a_signals:
anime_score += 0.04
conf = min(anime_score, 0.99)
return ClassificationResult(
category="anime",
confidence=conf,
signals={"anime_signals": a_signals},
tokens=tokens,
metadata=metadata,
)
is_tv_folder = bool(re.search(r"(?i)[/\\](?:tv[/\\]|tv[-_\s]shows?|tv[-_\s]series|season[-_\s]*\d+)", path_str))
is_movie_folder = bool(re.search(r"(?i)[/\\](?:movies?[/\\]|films?[/\\])", path_str))
# 4. TV Show Check (Episodic and Daily Broadcasts):
is_daily_tv = has_broadcast_date and (
tokens.resolution
or tokens.source
or "daily" in stem_lower
or "tonight" in stem_lower
or "late" in stem_lower
or "news" in stem_lower
or dur >= 1200
)
is_tv = False
if tokens.is_episodic or is_standard_tv or is_daily_tv:
is_tv = True
elif is_tv_folder and not tokens.year:
is_tv = True
elif is_tv_folder and tokens.episode is not None:
is_tv = True
if is_tv:
tv_score = 0.40
tv_signals = []
if tokens.is_episodic or is_standard_tv:
tv_score += 0.45
tv_signals.append("season_episode_pattern")
if is_daily_tv:
tv_score += 0.45
tv_signals.append("broadcast_date_pattern")
if tokens.episode is not None:
tv_score += 0.10
if is_tv_folder:
tv_score += 0.15
tv_signals.append("tv_folder_hint")
if 600 <= dur <= 5400 and not tokens.year:
tv_score += 0.10
tv_signals.append("episodic_duration")
# Provider boost
prov_res = None
if self.provider and tokens.title and tv_score >= 0.6:
try:
prov_res = self.provider.search_tv(
tokens.title, year=tokens.year, season=tokens.season, episode=tokens.episode
)
if prov_res:
tv_score += prov_res.confidence_boost
tv_signals.append("provider_verified")
except Exception:
pass
conf = min(tv_score, 0.99)
needs_quar = conf < self.confidence_threshold
return ClassificationResult(
category="tv",
confidence=conf,
signals={"tv_signals": tv_signals},
tokens=tokens,
metadata=metadata,
provider_result=prov_res,
needs_quarantine=needs_quar,
quarantine_reason="Low confidence TV classification" if needs_quar else None,
)
# 5. Movie Check:
movie_score = 0.40
m_signals = []
if tokens.year and not has_broadcast_date:
movie_score += 0.35
m_signals.append("year_in_title")
if tokens.resolution or tokens.source or tokens.video_codec:
movie_score += 0.15
m_signals.append("scene_technical_tags")
if is_movie_folder or "movie" in path_str or "film" in path_str:
movie_score += 0.15
m_signals.append("movie_folder_hint")
if dur >= 3600: # > 1 hour
movie_score += 0.20
m_signals.append("feature_film_duration")
# Check for movie extras tag
if re.search(r"-(behindthescenes|deleted|trailer|featurette)\b", stem_lower):
movie_score += 0.25
m_signals.append("movie_extra_tag")
# Provider boost
prov_res = None
if self.provider and tokens.title and movie_score >= 0.5:
try:
prov_res = self.provider.search_movie(tokens.title, year=tokens.year)
if prov_res:
movie_score += prov_res.confidence_boost
m_signals.append("provider_verified")
except Exception:
pass
conf = min(movie_score, 0.99)
needs_quar = conf < self.confidence_threshold
return ClassificationResult(
category="movie",
confidence=conf,
signals={"movie_signals": m_signals, "duration": dur},
tokens=tokens,
metadata=metadata,
provider_result=prov_res,
needs_quarantine=needs_quar,
quarantine_reason="Low confidence movie classification (missing year or title verification)"
if needs_quar
else None,
)
+339
View File
@@ -0,0 +1,339 @@
"""Command-line interface for Media Sorter.
Provides intuitive commands for scanning, dry-run simulation, live atomic organization,
transactional rollback, quarantine resolution, and web dashboard hosting.
"""
from __future__ import annotations
import os
import sys
from pathlib import Path
from typing import Optional
import typer
import uvicorn
from rich.console import Console
from rich.panel import Panel
from rich.progress import BarColumn, Progress, SpinnerColumn, TextColumn, TimeRemainingColumn
from rich.table import Table
from .config import ActionType, Settings
from .db import get_db_session, init_db
from .models import BatchRecord, Operation, QuarantineRecord, QuarantineStatus
from .quarantine import QuarantineManager
from .sorter import MediaSorterApp
app = typer.Typer(
name="media-sorter",
help="Reliable, high-performance media sorter with atomic moves, dry-runs, and rollback.",
add_completion=False,
)
quarantine_app = typer.Typer(help="Manage quarantined or low-confidence review files.")
config_app = typer.Typer(help="Inspect or initialize configuration.")
app.add_typer(quarantine_app, name="quarantine")
app.add_typer(config_app, name="config")
console = Console()
def load_settings_or_default(config_path: Optional[Path] = None) -> Settings:
if config_path and config_path.is_file():
if config_path.suffix in (".env", "") and "env" in config_path.name:
return Settings.load_from_env_file(config_path)
return Settings.load_from_file(config_path)
# Check if .env exists in current working directory
env_file = Path(".env")
if env_file.is_file():
return Settings.load_from_env_file(env_file)
# Search standard configuration paths
for cand in ("config.yaml", "config.yml", "config.toml", "media-sorter.yaml"):
p = Path(cand)
if p.is_file():
return Settings.load_from_file(p)
return Settings()
@app.command()
def scan(
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
source: Optional[Path] = typer.Option(None, "--source", "-s", help="Override source directory"),
):
"""Scan source directories and display media classification preview without moving any files."""
settings = load_settings_or_default(config)
if source:
settings.storage.source_dirs = [str(source)]
engine = init_db(db_path=settings.get_database_path())
sorter = MediaSorterApp(settings, engine)
console.print(Panel(f"[bold cyan]Scanning sources:[/bold cyan] {', '.join(settings.storage.source_dirs)}", title="Media Sorter Discovery"))
with Progress(
SpinnerColumn(),
TextColumn("[progress.description]{task.description}"),
BarColumn(),
TextColumn("[progress.percentage]{task.percentage:>3.0f}%"),
console=console,
) as progress:
task = progress.add_task("[green]Analyzing media...", total=None)
def on_prog(curr, total, name):
progress.update(task, total=total, completed=curr, description=f"[cyan]Probing: {name[:30]}")
results = sorter.scan_and_analyze(progress_callback=on_prog)
if not results:
console.print("[yellow]No qualifying files found in source directories.[/yellow]")
return
table = Table(title=f"Discovered Media Items ({len(results)} total)")
table.add_column("Filename", style="bold", overflow="fold")
table.add_column("Category", style="cyan")
table.add_column("Confidence", justify="right")
table.add_column("Status", justify="center")
for scanned, cls_res in results[:50]:
conf_pct = f"{int(cls_res.confidence * 100)}%"
if cls_res.needs_quarantine:
status_style = "[red]QUARANTINE[/red]"
elif cls_res.confidence >= 0.85:
status_style = "[green]HIGH CONF[/green]"
else:
status_style = "[yellow]MEDIUM[/yellow]"
table.add_row(scanned.path.name, cls_res.category, conf_pct, status_style)
console.print(table)
if len(results) > 50:
console.print(f"[dim]... and {len(results) - 50} more items.[/dim]")
@app.command()
def organize(
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
dry_run: bool = typer.Option(True, "--dry-run/--live", help="Safety preview mode (default: True)"),
source: Optional[Path] = typer.Option(None, "--source", "-s", help="Override source directory"),
dest: Optional[Path] = typer.Option(None, "--dest", "-d", help="Override destination directory"),
action: Optional[ActionType] = typer.Option(None, "--action", "-a", help="Action (move, copy, link, hardlink)"),
threshold: Optional[float] = typer.Option(None, "--threshold", "-t", help="Confidence threshold (0.0-1.0)"),
interval: Optional[int] = typer.Option(None, "--interval", "-i", help="Continuous scan interval in seconds"),
watch: bool = typer.Option(False, "--watch", "-w", help="Run continuously in watch/daemon mode"),
):
"""Execute media organization or generate a dry-run preview."""
import time
settings = load_settings_or_default(config)
if source:
settings.storage.source_dirs = [str(source)]
if dest:
settings.storage.destination_base = str(dest)
if action:
settings.general.action = action
if threshold:
settings.general.confidence_threshold = threshold
loop_interval = interval or (settings.general.scan_interval_seconds if settings.general.scan_interval_seconds > 0 else (60 if watch else 0))
engine = init_db(db_path=settings.get_database_path())
sorter = MediaSorterApp(settings, engine)
mode_label = "[bold yellow]DRY-RUN PREVIEW[/bold yellow]" if dry_run else "[bold red]LIVE EXECUTION[/bold red]"
while True:
console.print(Panel(f"Mode: {mode_label} | Action: {settings.general.action.value.upper()}", title="Media Sorter"))
start_t = time.time()
with Progress(
SpinnerColumn(),
TextColumn("[progress.description]{task.description}"),
BarColumn(),
TextColumn("[progress.percentage]{task.percentage:>3.0f}%"),
TimeRemainingColumn(),
console=console,
) as progress:
task = progress.add_task("[green]Processing...", total=None)
def on_prog(curr, total, name):
progress.update(task, total=total, completed=curr, description=f"Processing: {name[:30]}")
report = sorter.run(dry_run=dry_run, progress_callback=on_prog)
elapsed = max(time.time() - start_t, 0.001)
throughput = round(report.total_files / elapsed, 1)
# Print summary table
summary_table = Table(title=f"Batch Summary [{report.batch_id[:8]}]")
summary_table.add_column("Metric", style="bold")
summary_table.add_column("Count", justify="right")
summary_table.add_row("Total Processed", str(report.total_files))
summary_table.add_row("Moved / Organized", f"[green]{report.moved_files}[/green]")
summary_table.add_row("Copied", f"[blue]{report.copied_files}[/blue]")
summary_table.add_row("Linked", f"[cyan]{report.linked_files}[/cyan]")
summary_table.add_row("Skipped (Conflicts / Existing)", f"[dim]{report.skipped_files}[/dim]")
summary_table.add_row("Quarantined (Review Queue)", f"[yellow]{report.quarantined_files}[/yellow]")
summary_table.add_row("Failures", f"[red]{report.failed_files}[/red]")
summary_table.add_row("Throughput", f"{throughput} files/sec ({round(elapsed, 2)}s)")
console.print(summary_table)
if dry_run:
console.print("\n[bold cyan]Safe dry-run complete. No files were modified on disk.[/bold cyan]")
console.print("[dim]To apply these changes live, rerun with --live.[/dim]")
if loop_interval <= 0:
break
console.print(f"\n[cyan]Sleeping for {loop_interval}s until next scan cycle (press Ctrl+C to stop)...[/cyan]")
try:
time.sleep(loop_interval)
except KeyboardInterrupt:
console.print("\n[yellow]Daemon watch loop stopped by user.[/yellow]")
break
@app.command()
def rollback(
batch_id: Optional[str] = typer.Option(None, "--batch-id", "-b", help="Specific batch ID to roll back"),
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
):
"""Roll back a previous organization batch, restoring moved files to original sources."""
settings = load_settings_or_default(config)
engine = init_db(db_path=settings.get_database_path())
sorter = MediaSorterApp(settings, engine)
with console.status("[bold yellow]Executing transactional rollback...[/bold yellow]"):
reverted = sorter.rollback(batch_id)
if reverted > 0:
console.print(f"[bold green]Successfully rolled back {reverted} file operations.[/bold green]")
else:
console.print("[yellow]No operations were reverted (batch already rolled back or not found).[/yellow]")
@app.command()
def history(
limit: int = typer.Option(10, "--limit", "-n", help="Number of past batches to display"),
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
):
"""Display history of past organization batches and execution logs."""
settings = load_settings_or_default(config)
engine = init_db(db_path=settings.get_database_path())
with get_db_session(engine) as session:
batches = session.query(BatchRecord).order_by(BatchRecord.created_at.desc()).limit(limit).all()
if not batches:
console.print("[dim]No batch records found.[/dim]")
return
table = Table(title=f"Execution History (Last {len(batches)})")
table.add_column("Batch ID", style="bold")
table.add_column("Date", style="dim")
table.add_column("Mode")
table.add_column("Status")
table.add_column("Total", justify="right")
table.add_column("Moved", justify="right")
table.add_column("Quarantined", justify="right")
for b in batches:
mode = "[dim]Dry-Run[/dim]" if b.dry_run else "[bold]Live[/bold]"
status = f"[green]{b.status}[/green]" if b.status == "COMPLETED" else f"[yellow]{b.status}[/yellow]"
date_str = b.created_at.strftime("%Y-%m-%d %H:%M") if b.created_at else "-"
table.add_row(b.id[:8], date_str, mode, status, str(b.total_files), str(b.moved_files), str(b.quarantined_files))
console.print(table)
@quarantine_app.command("list")
def quarantine_list(
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
):
"""List pending items requiring manual review."""
settings = load_settings_or_default(config)
engine = init_db(db_path=settings.get_database_path())
with get_db_session(engine) as session:
qm = QuarantineManager(session)
items = qm.list_pending()
if not items:
console.print("[green]Quarantine queue is empty. All media classified cleanly![/green]")
return
table = Table(title=f"Quarantine Review Queue ({len(items)} items)")
table.add_column("ID", justify="right")
table.add_column("File Path", overflow="fold")
table.add_column("Suggested", style="cyan")
table.add_column("Confidence", justify="right")
table.add_column("Reason", style="yellow")
for q in items:
conf = f"{int((q.confidence or 0) * 100)}%"
table.add_row(str(q.id), q.src, q.suggested_category or "unknown", conf, q.reason)
console.print(table)
@quarantine_app.command("resolve")
def quarantine_resolve(
item_id: int = typer.Argument(..., help="Quarantine record ID to resolve"),
category: str = typer.Option(..., "--category", "-cat", help="Target category (movie, tv, music, etc.)"),
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
):
"""Manually classify and resolve a quarantined item."""
settings = load_settings_or_default(config)
engine = init_db(db_path=settings.get_database_path())
with get_db_session(engine) as session:
qm = QuarantineManager(session)
success = qm.resolve_item(item_id, category)
if success:
console.print(f"[green]Successfully resolved item #{item_id} as {category}.[/green]")
else:
console.print(f"[red]Quarantine item #{item_id} not found.[/red]")
@app.command()
def server(
host: Optional[str] = typer.Option(None, "--host", "-h", help="Bind host (default from .env or 0.0.0.0)"),
port: Optional[int] = typer.Option(None, "--port", "-p", help="Bind port (default from .env or 8080)"),
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
):
"""Launch web dashboard and management server."""
settings = load_settings_or_default(config)
from .server import create_app
bind_host = host or settings.server.host or "0.0.0.0"
bind_port = port or settings.server.port or 8080
web_app = create_app(settings)
console.print(f"[bold green]Starting Media Sorter Dashboard on http://{bind_host}:{bind_port}[/bold green]")
uvicorn.run(web_app, host=bind_host, port=bind_port)
@config_app.command("show")
def config_show(
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
):
"""Print effective configuration settings."""
settings = load_settings_or_default(config)
import yaml
console.print(yaml.dump(settings.model_dump(mode="json"), default_flow_style=False, sort_keys=False))
@config_app.command("init")
def config_init(
output: Path = typer.Option(Path("media-sorter.yaml"), "--output", "-o", help="Target config file path"),
):
"""Create a safe starter configuration file."""
if output.exists():
console.print(f"[yellow]Configuration file already exists at {output}. Aborting.[/yellow]")
return
settings = Settings()
settings.dump_yaml(output)
console.print(f"[green]Created default configuration file at {output}.[/green]")
if __name__ == "__main__":
app()
+497
View File
@@ -0,0 +1,497 @@
"""Configuration management for Media Sorter.
Provides Pydantic-based settings validated from YAML, TOML, JSON, or environment variables.
"""
from __future__ import annotations
import os
import json
from enum import Enum
from pathlib import Path
from typing import Any, Dict, List, Optional
import yaml
from pydantic import BaseModel, Field, field_validator
from pydantic_settings import BaseSettings, SettingsConfigDict
class ActionType(str, Enum):
MOVE = "move"
COPY = "copy"
LINK = "link"
HARDLINK = "hardlink"
class ConflictPolicy(str, Enum):
SKIP = "skip"
RENAME_UNIQUE = "rename_unique"
QUARANTINE = "quarantine"
REPLACE_IF_HIGHER_QUALITY = "replace_if_higher_quality"
ERROR = "error"
class DestinationDirs(BaseModel):
movies: str = "Movies"
tv: str = "TV Shows"
anime: str = "Anime"
music: str = "Music"
audiobooks: str = "Audiobooks"
podcasts: str = "Podcasts"
home_videos: str = "Home Videos"
photos: str = "Photos"
archives: str = "Archives"
quarantine: str = "Quarantine"
class GeneralSettings(BaseModel):
dry_run: bool = True
confidence_threshold: float = 0.75
worker_count: int = 4
min_file_age_seconds: int = 300
action: ActionType = ActionType.MOVE
preserve_permissions: bool = True
log_level: str = "INFO"
scan_interval_seconds: int = 0
cleanup_empty_dirs: bool = True
rename_files: bool = True
@field_validator("confidence_threshold")
@classmethod
def validate_confidence(cls, v: float) -> float:
if not 0.0 < v <= 1.0:
raise ValueError("confidence_threshold must be between 0.0 and 1.0")
return v
@field_validator("worker_count")
@classmethod
def validate_worker_count(cls, v: int) -> int:
if v < 1:
raise ValueError("worker_count must be at least 1")
return v
@field_validator("min_file_age_seconds")
@classmethod
def validate_min_file_age(cls, v: int) -> int:
if v < 0:
raise ValueError("min_file_age_seconds cannot be negative")
return v
@field_validator("log_level")
@classmethod
def validate_log_level(cls, v: str) -> str:
allowed = {"DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"}
upper = v.upper()
if upper not in allowed:
raise ValueError(f"log_level must be one of {allowed}")
return upper
class StorageSettings(BaseModel):
source_dirs: List[str] = Field(default_factory=lambda: ["incoming"])
destination_base: str = "organized"
destination_dirs: DestinationDirs = Field(default_factory=DestinationDirs)
class ConflictSettings(BaseModel):
policy: ConflictPolicy = ConflictPolicy.RENAME_UNIQUE
allow_overwrite: bool = False
backup_dir: Optional[str] = None
class FilterSettings(BaseModel):
include_patterns: List[str] = Field(default_factory=lambda: ["*"])
exclude_patterns: List[str] = Field(
default_factory=lambda: [
".*",
"*.part",
"*.crdownload",
"*.!qB",
"Thumbs.db",
"desktop.ini",
"@eaDir",
"$RECYCLE.BIN",
"*.txt",
]
)
class TemplateSettings(BaseModel):
movie: str = "{title} ({year})/{movie_name}.{ext}"
tv: str = "{title}/Season {season:02d}/{show_name}_{season_episode}.{ext}"
anime: str = "{title}/Season {season:02d}/{show_name}_{season_episode} [{group}].{ext}"
music: str = "{artist}/{album} ({year})/{disc:01d}{track:02d} - {title}.{ext}"
audiobook: str = "{author}/{title}/{track:02d} - {chapter}.{ext}"
podcast: str = "{show}/{year}/{show} - {date} - {title}.{ext}"
home_video: str = "{year}/{year}-{month:02d} - {event}/{filename}.{ext}"
photo: str = "{year}/{year}-{month:02d}/{year}{month:02d}{day:02d}_{time}_{camera}.{ext}"
archive: str = "Archives/{filename}.{ext}"
quarantine: str = "Quarantine/{reason}/{filename}.{ext}"
class SubtitleSettings(BaseModel):
match_video_basename: bool = True
preserve_language_code: bool = True
class ArtworkSettings(BaseModel):
match_parent_folder: bool = True
class ExtrasSettings(BaseModel):
detect_trailers: bool = True
trailer_suffix: str = "-trailer"
class SidecarSettings(BaseModel):
enabled: bool = True
subtitles: SubtitleSettings = Field(default_factory=SubtitleSettings)
artwork: ArtworkSettings = Field(default_factory=ArtworkSettings)
extras: ExtrasSettings = Field(default_factory=ExtrasSettings)
class DatabaseSettings(BaseModel):
path: str = "media_sorter.db"
wal_mode: bool = True
class ServerSettings(BaseModel):
host: str = "127.0.0.1"
port: int = 8080
enabled: bool = True
class ProviderSettings(BaseModel):
enable_online_metadata: bool = False
tmdb_api_key: Optional[str] = None
tvdb_api_key: Optional[str] = None
rate_limit_per_second: float = 2.0
cache_expiry_hours: int = 72
class NotificationSettings(BaseModel):
enabled: bool = False
webhook_url: Optional[str] = None
notify_on_complete: bool = True
notify_on_failure: bool = True
class SymlinkSettings(BaseModel):
follow_symlinks: bool = False
handle_broken_symlinks: str = "skip" # skip | quarantine
class PermissionSettings(BaseModel):
preserve_attributes: bool = True
file_mode: Optional[str] = None # e.g. "0644"
dir_mode: Optional[str] = None # e.g. "0755"
owner: Optional[str] = None
group: Optional[str] = None
class QuarantineSettings(BaseModel):
move_to_quarantine_folder: bool = False
directory: str = "Quarantine"
class Settings(BaseSettings):
model_config = SettingsConfigDict(
env_prefix="MEDIA_SORTER_",
env_nested_delimiter="__",
env_file=".env",
env_file_encoding="utf-8",
extra="ignore",
)
general: GeneralSettings = Field(default_factory=GeneralSettings)
storage: StorageSettings = Field(default_factory=StorageSettings)
conflicts: ConflictSettings = Field(default_factory=ConflictSettings)
filters: FilterSettings = Field(default_factory=FilterSettings)
templates: TemplateSettings = Field(default_factory=TemplateSettings)
sidecars: SidecarSettings = Field(default_factory=SidecarSettings)
database: DatabaseSettings = Field(default_factory=DatabaseSettings)
server: ServerSettings = Field(default_factory=ServerSettings)
providers: ProviderSettings = Field(default_factory=ProviderSettings)
notifications: NotificationSettings = Field(default_factory=NotificationSettings)
symlinks: SymlinkSettings = Field(default_factory=SymlinkSettings)
permissions: PermissionSettings = Field(default_factory=PermissionSettings)
quarantine: QuarantineSettings = Field(default_factory=QuarantineSettings)
def model_post_init(self, __context: Any) -> None:
super().model_post_init(__context)
# Check intuitive environment variable overrides from .env only if not explicitly supplied
if "storage" not in self.model_fields_set:
downloads_dir = os.getenv("DOWNLOADS_DIR") or os.getenv("SOURCE_DIR")
if downloads_dir:
self.storage.source_dirs = [downloads_dir]
movies_dir = os.getenv("MOVIES_DIR")
if movies_dir:
self.storage.destination_dirs.movies = movies_dir
shows_dir = os.getenv("SHOWS_DIR") or os.getenv("TV_DIR")
if shows_dir:
self.storage.destination_dirs.tv = shows_dir
anime_dir = os.getenv("ANIME_DIR")
if anime_dir:
self.storage.destination_dirs.anime = anime_dir
elif shows_dir:
self.storage.destination_dirs.anime = shows_dir
if "general" not in self.model_fields_set:
dry_run_env = os.getenv("DRY_RUN")
if dry_run_env is not None:
self.general.dry_run = dry_run_env.strip().lower() in ("true", "1", "yes", "on")
action_env = os.getenv("ACTION")
if action_env:
try:
self.general.action = ActionType(action_env.lower())
except ValueError:
pass
conf_env = os.getenv("CONFIDENCE_THRESHOLD")
if conf_env:
try:
self.general.confidence_threshold = float(conf_env)
except ValueError:
pass
min_age_env = os.getenv("MIN_FILE_AGE_SECONDS")
if min_age_env:
try:
self.general.min_file_age_seconds = int(min_age_env)
except ValueError:
pass
scan_int_env = os.getenv("SCAN_INTERVAL_SECONDS")
if scan_int_env:
try:
self.general.scan_interval_seconds = int(scan_int_env)
except ValueError:
pass
cleanup_env = os.getenv("CLEANUP_EMPTY_DIRS")
if cleanup_env is not None:
self.general.cleanup_empty_dirs = cleanup_env.strip().lower() in ("true", "1", "yes", "on")
rename_env = os.getenv("RENAME_FILES")
if rename_env is not None:
self.general.rename_files = rename_env.strip().lower() in ("true", "1", "yes", "on")
if "templates" not in self.model_fields_set:
movie_tmpl = os.getenv("MOVIE_TEMPLATE")
if movie_tmpl:
self.templates.movie = movie_tmpl
tv_tmpl = os.getenv("TV_TEMPLATE") or os.getenv("SHOW_TEMPLATE") or os.getenv("SHOWS_TEMPLATE")
if tv_tmpl:
self.templates.tv = tv_tmpl
if "server" not in self.model_fields_set:
host_env = os.getenv("SERVER_HOST") or os.getenv("HOST")
if host_env:
self.server.host = host_env
port_env = os.getenv("SERVER_PORT") or os.getenv("PORT")
if port_env:
try:
self.server.port = int(port_env)
except ValueError:
pass
if "database" not in self.model_fields_set:
db_path_env = os.getenv("DATABASE_PATH")
if db_path_env:
self.database.path = db_path_env
def resolve_path(self, raw_path: str) -> Path:
"""Expand environment variables and user home, returning resolved Path."""
expanded = os.path.expandvars(raw_path)
return Path(expanded).expanduser().resolve()
def get_source_paths(self) -> List[Path]:
return [self.resolve_path(p) for p in self.storage.source_dirs]
def get_destination_base_path(self) -> Path:
return self.resolve_path(self.storage.destination_base)
def get_destination_path(self, category: str) -> Path:
"""Return the destination path for a given category."""
base = self.get_destination_base_path()
cat_map = {
"movie": "movies",
"movies": "movies",
"tv": "tv",
"show": "tv",
"shows": "tv",
"audiobook": "audiobooks",
"podcast": "podcasts",
"photo": "photos",
"home_video": "home_videos",
"archive": "archives",
}
lookup_key = cat_map.get(category, category)
dest_field = getattr(self.storage.destination_dirs, lookup_key, category)
path = Path(dest_field)
# If dest_field is an explicit relative path (e.g. ./movies, ./shows) or absolute path
if path.is_absolute() or str(dest_field).startswith(("./", "../")):
return path.resolve()
return (base / path).resolve()
@classmethod
def load_from_env_file(cls, env_path: Path | str = ".env") -> Settings:
"""Load configuration from a .env file."""
path = Path(env_path).expanduser().resolve()
settings = cls()
if not path.is_file():
return settings
from dotenv import dotenv_values
values = dotenv_values(path)
downloads_dir = os.getenv("DOWNLOADS_DIR") or values.get("DOWNLOADS_DIR") or os.getenv("SOURCE_DIR") or values.get("SOURCE_DIR")
if downloads_dir:
settings.storage.source_dirs = [downloads_dir]
movies_dir = os.getenv("MOVIES_DIR") or values.get("MOVIES_DIR")
if movies_dir:
settings.storage.destination_dirs.movies = movies_dir
shows_dir = os.getenv("SHOWS_DIR") or values.get("SHOWS_DIR") or os.getenv("TV_DIR") or values.get("TV_DIR")
if shows_dir:
settings.storage.destination_dirs.tv = shows_dir
anime_dir = os.getenv("ANIME_DIR") or values.get("ANIME_DIR")
if anime_dir:
settings.storage.destination_dirs.anime = anime_dir
elif shows_dir:
settings.storage.destination_dirs.anime = shows_dir
dry_run = os.getenv("DRY_RUN") if os.getenv("DRY_RUN") is not None else values.get("DRY_RUN")
if dry_run is not None:
settings.general.dry_run = dry_run.strip().lower() in ("true", "1", "yes", "on")
action = os.getenv("ACTION") or values.get("ACTION")
if action:
try:
settings.general.action = ActionType(action.lower())
except ValueError:
pass
conf = os.getenv("CONFIDENCE_THRESHOLD") or values.get("CONFIDENCE_THRESHOLD")
if conf:
try:
settings.general.confidence_threshold = float(conf)
except ValueError:
pass
min_age = os.getenv("MIN_FILE_AGE_SECONDS") or values.get("MIN_FILE_AGE_SECONDS")
if min_age:
try:
settings.general.min_file_age_seconds = int(min_age)
except ValueError:
pass
scan_int = os.getenv("SCAN_INTERVAL_SECONDS") or values.get("SCAN_INTERVAL_SECONDS")
if scan_int:
try:
settings.general.scan_interval_seconds = int(scan_int)
except ValueError:
pass
cleanup = os.getenv("CLEANUP_EMPTY_DIRS") if os.getenv("CLEANUP_EMPTY_DIRS") is not None else values.get("CLEANUP_EMPTY_DIRS")
if cleanup is not None:
settings.general.cleanup_empty_dirs = str(cleanup).strip().lower() in ("true", "1", "yes", "on")
rename_files = os.getenv("RENAME_FILES") if os.getenv("RENAME_FILES") is not None else values.get("RENAME_FILES")
if rename_files is not None:
settings.general.rename_files = str(rename_files).strip().lower() in ("true", "1", "yes", "on")
movie_tmpl = os.getenv("MOVIE_TEMPLATE") or values.get("MOVIE_TEMPLATE")
if movie_tmpl:
settings.templates.movie = movie_tmpl
tv_tmpl = os.getenv("TV_TEMPLATE") or values.get("TV_TEMPLATE") or os.getenv("SHOW_TEMPLATE") or values.get("SHOW_TEMPLATE") or os.getenv("SHOWS_TEMPLATE") or values.get("SHOWS_TEMPLATE")
if tv_tmpl:
settings.templates.tv = tv_tmpl
host = os.getenv("SERVER_HOST") or values.get("SERVER_HOST") or os.getenv("HOST") or values.get("HOST")
if host:
settings.server.host = host
port = os.getenv("SERVER_PORT") or values.get("SERVER_PORT") or os.getenv("PORT") or values.get("PORT")
if port:
try:
settings.server.port = int(port)
except ValueError:
pass
db_path = os.getenv("DATABASE_PATH") or values.get("DATABASE_PATH")
if db_path:
settings.database.path = db_path
return settings
def save_to_env_file(self, env_path: Path | str = ".env") -> None:
"""Persist key user-configurable settings to a .env file."""
path = Path(env_path)
content = (
f"# Media Sorter Configuration\n"
f"DOWNLOADS_DIR={self.storage.source_dirs[0] if self.storage.source_dirs else './downloads'}\n"
f"MOVIES_DIR={self.storage.destination_dirs.movies}\n"
f"SHOWS_DIR={self.storage.destination_dirs.tv}\n"
f"DRY_RUN={'true' if self.general.dry_run else 'false'}\n"
f"ACTION={self.general.action.value}\n"
f"CONFIDENCE_THRESHOLD={self.general.confidence_threshold}\n"
f"MIN_FILE_AGE_SECONDS={self.general.min_file_age_seconds}\n"
f"SCAN_INTERVAL_SECONDS={self.general.scan_interval_seconds}\n"
f"CLEANUP_EMPTY_DIRS={'true' if self.general.cleanup_empty_dirs else 'false'}\n"
f"RENAME_FILES={'true' if self.general.rename_files else 'false'}\n"
f"MOVIE_TEMPLATE={self.templates.movie}\n"
f"TV_TEMPLATE={self.templates.tv}\n"
f"SERVER_HOST={self.server.host}\n"
f"SERVER_PORT={self.server.port}\n"
f"DATABASE_PATH={self.database.path}\n"
)
path.write_text(content, encoding="utf-8")
def get_database_path(self) -> Path:
return self.resolve_path(self.database.path)
@classmethod
def load_from_file(cls, config_path: Path | str) -> Settings:
"""Load configuration from YAML, TOML, or JSON file."""
path = Path(config_path).expanduser().resolve()
if not path.is_file():
raise FileNotFoundError(f"Configuration file not found: {path}")
ext = path.suffix.lower()
with open(path, "r", encoding="utf-8") as f:
content = f.read()
if ext in (".yaml", ".yml"):
data = yaml.safe_load(content) or {}
elif ext == ".json":
data = json.loads(content)
elif ext == ".toml":
try:
import tomllib # Python 3.11+
data = tomllib.loads(content)
except ImportError:
import tomli
data = tomli.loads(content)
else:
# Fallback to YAML loader which can parse JSON and YAML
data = yaml.safe_load(content) or {}
return cls(**data)
def dump_yaml(self, target_path: Path | str) -> None:
"""Dump settings to YAML format."""
path = Path(target_path)
path.parent.mkdir(parents=True, exist_ok=True)
data = self.model_dump(mode="json")
with open(path, "w", encoding="utf-8") as f:
yaml.dump(data, f, default_flow_style=False, sort_keys=False)
+84
View File
@@ -0,0 +1,84 @@
"""Database initialization and session management for Media Sorter.
Configures SQLite with Write-Ahead Logging (WAL) mode and foreign keys enabled
for transactional safety, high concurrency, and crash resilience.
"""
from __future__ import annotations
import os
from contextlib import contextmanager
from pathlib import Path
from typing import Generator, Optional
from sqlalchemy import create_engine, event
from sqlalchemy.engine import Engine
from sqlalchemy.orm import Session, declarative_base, scoped_session, sessionmaker
from .models import Base
DEFAULT_DB_PATH = Path("media_sorter.db")
def get_engine(db_path: Path | str = DEFAULT_DB_PATH, wal_mode: bool = True) -> Engine:
"""Create and configure a SQLite SQLAlchemy engine.
Enables WAL mode and enforces foreign keys for transactional integrity.
"""
path = Path(db_path).resolve()
path.parent.mkdir(parents=True, exist_ok=True)
db_url = f"sqlite:///{path}"
engine = create_engine(
db_url,
connect_args={"check_same_thread": False, "timeout": 30.0},
pool_pre_ping=True,
)
@event.listens_for(engine, "connect")
def set_sqlite_pragma(dbapi_connection, connection_record):
cursor = dbapi_connection.cursor()
cursor.execute("PRAGMA foreign_keys=ON")
if wal_mode:
cursor.execute("PRAGMA journal_mode=WAL")
cursor.execute("PRAGMA synchronous=NORMAL")
cursor.execute("PRAGMA busy_timeout=30000")
cursor.close()
return engine
def init_db(engine: Optional[Engine] = None, db_path: Path | str = DEFAULT_DB_PATH) -> Engine:
"""Initialize all tables defined in models.py if they do not exist."""
if engine is None:
engine = get_engine(db_path)
Base.metadata.create_all(engine)
return engine
_ENGINE_SESSION_FACTORIES: dict[Engine, scoped_session[Session]] = {}
def get_session_factory(engine: Engine) -> scoped_session[Session]:
"""Retrieve or create a cached scoped session factory bound to the given engine."""
if engine not in _ENGINE_SESSION_FACTORIES:
_ENGINE_SESSION_FACTORIES[engine] = scoped_session(
sessionmaker(autocommit=False, autoflush=False, bind=engine)
)
return _ENGINE_SESSION_FACTORIES[engine]
@contextmanager
def get_db_session(engine: Engine) -> Generator[Session, None, None]:
"""Provide a transactional scope around a series of operations."""
session_factory = get_session_factory(engine)
session: Session = session_factory()
try:
yield session
session.commit()
except Exception:
session.rollback()
raise
finally:
session.close()
session_factory.remove()
+791
View File
@@ -0,0 +1,791 @@
"""Safe execution engine and transactional operation journal for Media Sorter.
Enforces dry-run previews, atomic moves, cross-filesystem safety, conflict handling,
attribute preservation (POSIX timestamps/permissions), crash recovery, and instant rollbacks.
"""
from __future__ import annotations
import hashlib
import os
import shutil
import uuid
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Callable, Dict, List, Optional
import structlog
from sqlalchemy.orm import Session
from .config import ActionType, ConflictPolicy, Settings
from .models import BatchRecord, FileRecord, Operation, OperationStatus, QuarantineRecord, QuarantineStatus
from contextlib import contextmanager
import sys
logger = structlog.get_logger(__name__)
class ProcessLockError(Exception):
"""Raised when another media-sorter process holds the execution lock."""
pass
@contextmanager
def acquire_process_lock(lock_file_path: Path):
"""Ensure mutual exclusion so multiple workers/instances do not run concurrent batches."""
lock_file = Path(lock_file_path).resolve()
lock_file.parent.mkdir(parents=True, exist_ok=True)
f = open(lock_file, "a+")
try:
if sys.platform != "win32":
import fcntl
try:
fcntl.flock(f.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
except (IOError, OSError):
raise ProcessLockError(
f"Another media-sorter process currently holds the lock on {lock_file}."
)
else:
import msvcrt
try:
msvcrt.locking(f.fileno(), msvcrt.LK_NBLCK, 1)
except (IOError, OSError):
raise ProcessLockError(
f"Another media-sorter process currently holds the lock on {lock_file}."
)
yield
finally:
try:
if sys.platform != "win32":
import fcntl
fcntl.flock(f.fileno(), fcntl.LOCK_UN)
else:
import msvcrt
msvcrt.locking(f.fileno(), msvcrt.LK_UNLCK, 1)
except Exception:
pass
f.close()
def copy_extended_attributes(src: Path, dst: Path) -> None:
"""Preserve POSIX extended attributes (xattrs) across filesystems where supported."""
if hasattr(os, "listxattr") and hasattr(os, "getxattr") and hasattr(os, "setxattr"):
try:
attrs = os.listxattr(src)
for attr in attrs:
try:
val = os.getxattr(src, attr)
os.setxattr(dst, attr, val)
except (OSError, PermissionError):
pass
except (OSError, PermissionError):
pass
def compute_file_hash(file_path: Path, max_bytes: int = 1048576) -> str:
"""Compute quick partial SHA-256 hash (first 1MB) for fast identity verification."""
try:
hasher = hashlib.sha256()
with open(file_path, "rb") as f:
chunk = f.read(max_bytes)
hasher.update(chunk)
return hasher.hexdigest()
except Exception:
return ""
@dataclass
class PlannedOperation:
src: Path
dst: Path
action: ActionType
category: str
confidence: float
details: Dict[str, Any] = field(default_factory=dict)
is_conflict: bool = False
conflict_resolved_dst: Optional[Path] = None
quarantine: bool = False
quarantine_reason: Optional[str] = None
primary_src: Optional[Path] = None
@dataclass
class BatchExecutionReport:
batch_id: str
dry_run: bool
total_files: int = 0
moved_files: int = 0
copied_files: int = 0
linked_files: int = 0
skipped_files: int = 0
quarantined_files: int = 0
failed_files: int = 0
cleaned_dirs: int = 0
operations: List[PlannedOperation] = field(default_factory=list)
errors: List[str] = field(default_factory=list)
class MediaExecutor:
"""Executes planned operations with transactional safety and crash recovery."""
def __init__(self, settings: Settings, session: Session):
self.settings = settings
self.session = session
def plan_operations(
self, planned_items: List[PlannedOperation]
) -> List[PlannedOperation]:
"""Validate destination conflicts and resolve destination paths."""
allocated_destinations: Dict[Path, PlannedOperation] = {}
validated_plan: List[PlannedOperation] = []
primary_resolutions: Dict[Path, Path] = {}
primary_orig_to_resolved: Dict[Path, Path] = {}
for item in planned_items:
# If already marked for quarantine, keep as is
if item.quarantine:
validated_plan.append(item)
continue
# Check if this item is a sidecar whose primary was renamed
orig_target_dst = item.dst
if item.category in ("subtitle", "artwork", "metadata") or item.primary_src:
p_src = item.primary_src
resolved_p_dst = None
if p_src and p_src in primary_resolutions:
resolved_p_dst = primary_resolutions[p_src]
else:
for orig_p_dst, res_p_dst in primary_orig_to_resolved.items():
if orig_p_dst.parent == item.dst.parent and item.dst.stem.lower().startswith(orig_p_dst.stem.lower()):
resolved_p_dst = res_p_dst
break
if resolved_p_dst and resolved_p_dst.stem != item.dst.stem:
orig_stem = item.dst.stem
p_orig_stem = None
for orig_p_dst in primary_orig_to_resolved:
if orig_stem.lower().startswith(orig_p_dst.stem.lower()):
p_orig_stem = orig_p_dst.stem
break
tag = orig_stem[len(p_orig_stem):] if p_orig_stem else ""
new_sidecar_name = f"{resolved_p_dst.stem}{tag}{item.dst.suffix}"
item.dst = resolved_p_dst.parent / new_sidecar_name
target_dst = item.dst
# 1. Check intra-batch duplicate destination conflict
if target_dst in allocated_destinations:
logger.warning(
"Intra-batch destination collision detected",
dst=str(target_dst),
src1=str(allocated_destinations[target_dst].src),
src2=str(item.src),
)
item.is_conflict = True
target_dst = self._resolve_conflict(item.src, target_dst)
item.conflict_resolved_dst = target_dst
# 2. Check on-disk destination conflict
if target_dst.exists():
logger.info("Destination already exists on disk", dst=str(target_dst), src=str(item.src))
item.is_conflict = True
target_dst = self._resolve_conflict(item.src, target_dst)
item.conflict_resolved_dst = target_dst
if item.quarantine:
validated_plan.append(item)
continue
item.dst = target_dst
allocated_destinations[target_dst] = item
validated_plan.append(item)
# Record primary resolution for companion alignment
if item.category not in ("subtitle", "artwork", "metadata"):
primary_resolutions[item.src] = item.dst
primary_orig_to_resolved[orig_target_dst] = item.dst
return validated_plan
def _resolve_conflict(self, src: Path, desired_dst: Path) -> Path:
"""Resolve conflict according to the configured conflict policy."""
policy = self.settings.conflicts.policy
if policy == ConflictPolicy.SKIP:
return desired_dst # Will be skipped during execution
elif policy == ConflictPolicy.ERROR:
raise FileExistsError(f"Destination conflict: {desired_dst} already exists")
elif policy == ConflictPolicy.QUARANTINE:
return self.settings.get_destination_path("quarantine") / f"conflicts/{src.name}"
elif policy == ConflictPolicy.REPLACE_IF_HIGHER_QUALITY:
# Allow replacing existing file (will backup during execution)
return desired_dst
else:
# ConflictPolicy.RENAME_UNIQUE: foo (1).mp4
parent = desired_dst.parent
stem = desired_dst.stem
ext = desired_dst.suffix
counter = 1
candidate = parent / f"{stem} ({counter}){ext}"
while candidate.exists():
counter += 1
candidate = parent / f"{stem} ({counter}){ext}"
return candidate
def execute_batch(
self,
planned_items: List[PlannedOperation],
dry_run: Optional[bool] = None,
progress_callback: Optional[Callable[[int, int, str], None]] = None,
) -> BatchExecutionReport:
"""Execute a batch of operations transactionally, with dry-run support."""
is_dry_run = self.settings.general.dry_run if dry_run is None else dry_run
batch_id = str(uuid.uuid4())
# Validate and resolve destination collisions
validated_plan = self.plan_operations(planned_items)
report = BatchExecutionReport(
batch_id=batch_id,
dry_run=is_dry_run,
total_files=len(validated_plan),
operations=validated_plan,
)
# Create Batch Record in DB
batch_record = BatchRecord(
id=batch_id,
dry_run=is_dry_run,
status="IN_PROGRESS",
total_files=len(validated_plan),
)
self.session.add(batch_record)
self.session.commit()
total = len(validated_plan)
for idx, item in enumerate(validated_plan):
if progress_callback:
progress_callback(idx + 1, total, str(item.src.name))
if item.quarantine:
self._record_quarantine(batch_id, item, is_dry_run)
report.quarantined_files += 1
continue
# Check if skipping due to conflict
if item.is_conflict and self.settings.conflicts.policy == ConflictPolicy.SKIP and item.dst.exists():
logger.info("Skipping existing destination", dst=str(item.dst))
report.skipped_files += 1
self._record_operation(
batch_id, item, status=OperationStatus.SKIPPED, is_dry_run=is_dry_run
)
continue
if is_dry_run:
# Dry run preview only: do not touch filesystem
if item.action == ActionType.MOVE:
report.moved_files += 1
elif item.action == ActionType.COPY:
report.copied_files += 1
elif item.action in (ActionType.LINK, ActionType.HARDLINK):
report.linked_files += 1
self._record_operation(
batch_id, item, status=OperationStatus.PLANNED, is_dry_run=True
)
continue
# Live execution
try:
self._execute_single_op(batch_id, item)
if item.action == ActionType.MOVE:
report.moved_files += 1
elif item.action == ActionType.COPY:
report.copied_files += 1
elif item.action in (ActionType.LINK, ActionType.HARDLINK):
report.linked_files += 1
except Exception as e:
report.failed_files += 1
err_msg = f"Failed {item.action} on {item.src} -> {item.dst}: {e}"
logger.error(err_msg, exc_info=True)
report.errors.append(err_msg)
# Clean up empty directories in source directories after live moves
if not is_dry_run and getattr(self.settings.general, "cleanup_empty_dirs", True):
moved_srcs = [
item.src
for item in validated_plan
if item.action == ActionType.MOVE and not item.quarantine
]
report.cleaned_dirs = self.clean_empty_directories(moved_srcs)
# Update batch record completion status
batch_record.completed_at = datetime.now(timezone.utc)
batch_record.moved_files = report.moved_files
batch_record.skipped_files = report.skipped_files
batch_record.failed_files = report.failed_files
batch_record.quarantined_files = report.quarantined_files
batch_record.status = "COMPLETED" if report.failed_files == 0 else "PARTIAL_FAILURE"
self.session.commit()
return report
def _execute_single_op(self, batch_id: str, item: PlannedOperation) -> None:
"""Perform atomic move, copy, or link with attribute preservation and journal update."""
src = item.src
dst = item.dst
action = item.action
if not src.exists():
raise FileNotFoundError(f"Source file missing: {src}")
src_hash = compute_file_hash(src)
backup_path: Optional[str] = None
# Handle backup if replacing
if dst.exists():
if self.settings.conflicts.policy == ConflictPolicy.REPLACE_IF_HIGHER_QUALITY:
b_dir = Path(self.settings.conflicts.backup_dir or ".backup") / batch_id
b_dir.mkdir(parents=True, exist_ok=True)
backup_dst = b_dir / dst.name
shutil.move(dst, backup_dst)
backup_path = str(backup_dst)
elif not self.settings.conflicts.allow_overwrite:
raise FileExistsError(f"Destination exists and allow_overwrite is False: {dst}")
# Create journal entry in IN_PROGRESS state
op = Operation(
batch_id=batch_id,
src=str(src),
dst=str(dst),
action=action.value,
category=item.category,
confidence=item.confidence,
src_hash=src_hash,
backup_path=backup_path,
details=item.details,
status=OperationStatus.IN_PROGRESS.value,
)
self.session.add(op)
self.session.commit()
dst.parent.mkdir(parents=True, exist_ok=True)
try:
if action == ActionType.MOVE:
self._safe_move(src, dst)
elif action == ActionType.COPY:
self._safe_copy(src, dst)
elif action == ActionType.LINK:
if dst.exists() or dst.is_symlink():
dst.unlink()
os.symlink(src, dst)
elif action == ActionType.HARDLINK:
if dst.exists():
dst.unlink()
os.link(src, dst)
# Apply custom permissions if specified
self._apply_permissions(dst)
op.status = OperationStatus.COMMITTED.value
op.completed_at = datetime.now(timezone.utc)
op.dst_hash = compute_file_hash(dst)
# Update or create FileRecord in database
self._update_file_record(dst, item)
self.session.commit()
except Exception as e:
op.status = OperationStatus.FAILED.value
op.error_message = str(e)
self.session.commit()
raise
def _safe_move(self, src: Path, dst: Path) -> None:
"""Atomic move on same filesystem, or safe temp-copy-atomic-rename cross-filesystem."""
try:
# Check if same filesystem by comparing st_dev
src_dev = src.stat().st_dev
dst_parent_dev = dst.parent.stat().st_dev
if src_dev == dst_parent_dev:
# Same device: atomic rename
os.replace(src, dst)
return
except Exception:
pass
# Cross-filesystem move:
# 1. Copy to temp file in destination directory
temp_dst = dst.parent / f".tmp_media_sorter_{uuid.uuid4().hex}_{dst.name}"
try:
shutil.copy2(src, temp_dst)
if self.settings.permissions.preserve_attributes:
copy_extended_attributes(src, temp_dst)
# Verify file size matches
if temp_dst.stat().st_size != src.stat().st_size:
raise IOError(f"Size mismatch during copy: {temp_dst.stat().st_size} != {src.stat().st_size}")
# Atomically replace into final destination
os.replace(temp_dst, dst)
# Unlink original source
src.unlink()
finally:
if temp_dst.exists():
try:
temp_dst.unlink()
except Exception:
pass
def _safe_copy(self, src: Path, dst: Path) -> None:
"""Safe copy using temporary file and atomic replace."""
temp_dst = dst.parent / f".tmp_media_sorter_{uuid.uuid4().hex}_{dst.name}"
try:
shutil.copy2(src, temp_dst)
if self.settings.permissions.preserve_attributes:
copy_extended_attributes(src, temp_dst)
if temp_dst.stat().st_size != src.stat().st_size:
raise IOError("Copy size mismatch")
os.replace(temp_dst, dst)
finally:
if temp_dst.exists():
try:
temp_dst.unlink()
except Exception:
pass
def _apply_permissions(self, path: Path) -> None:
"""Apply configured mode bits and ownership safely."""
perm_cfg = self.settings.permissions
if perm_cfg.file_mode and path.is_file():
try:
mode = int(perm_cfg.file_mode, 8)
os.chmod(path, mode)
except Exception as e:
logger.debug("Failed setting file mode", path=str(path), error=str(e))
if (perm_cfg.owner or perm_cfg.group) and hasattr(os, "chown"):
try:
import pwd
import grp
uid = -1
gid = -1
if perm_cfg.owner:
uid = int(perm_cfg.owner) if perm_cfg.owner.isdigit() else pwd.getpwnam(perm_cfg.owner).pw_uid
if perm_cfg.group:
gid = int(perm_cfg.group) if perm_cfg.group.isdigit() else grp.getgrnam(perm_cfg.group).gr_gid
os.chown(path, uid, gid)
except Exception as e:
logger.debug("Failed setting ownership", path=str(path), error=str(e))
def _update_file_record(self, final_path: Path, item: PlannedOperation) -> None:
"""Record the file in the database to prevent re-processing."""
try:
stat = final_path.stat()
rec = self.session.query(FileRecord).filter_by(path=str(final_path)).first()
if not rec:
rec = FileRecord(
path=str(final_path),
size=stat.st_size,
mtime=stat.st_mtime,
category=item.category,
confidence=item.confidence,
status="organized",
last_processed=datetime.now(timezone.utc),
)
self.session.add(rec)
else:
rec.size = stat.st_size
rec.mtime = stat.st_mtime
rec.category = item.category
rec.confidence = item.confidence
rec.status = "organized"
rec.last_processed = datetime.now(timezone.utc)
except Exception:
pass
def _record_operation(
self, batch_id: str, item: PlannedOperation, status: OperationStatus, is_dry_run: bool
) -> None:
op = Operation(
batch_id=batch_id,
src=str(item.src),
dst=str(item.dst),
action=item.action.value,
category=item.category,
confidence=item.confidence,
details=item.details,
status=status.value,
)
self.session.add(op)
self.session.commit()
def _record_quarantine(self, batch_id: str, item: PlannedOperation, is_dry_run: bool) -> None:
q = self.session.query(QuarantineRecord).filter_by(src=str(item.src)).first()
if not q:
q = QuarantineRecord(
src=str(item.src),
suggested_category=item.category,
confidence=item.confidence,
reason=item.quarantine_reason or "Low confidence or unclassifiable",
signals=item.details,
status=QuarantineStatus.PENDING.value,
)
self.session.add(q)
self.session.commit()
if not is_dry_run and self.settings.quarantine.move_to_quarantine_folder:
# Move to quarantine folder
q_dir = self.settings.get_destination_path("quarantine") / (item.quarantine_reason or "review")
q_dir.mkdir(parents=True, exist_ok=True)
dst_path = q_dir / item.src.name
try:
self._safe_move(item.src, dst_path)
q.resolved_path = str(dst_path)
self.session.commit()
except Exception as e:
logger.error("Failed moving to quarantine folder", src=str(item.src), error=str(e))
def rollback_batch(self, batch_id: Optional[str] = None) -> int:
"""Invert all COMMITTED operations in a batch, returning count of reverted files."""
query = self.session.query(BatchRecord)
if batch_id:
batch = query.filter_by(id=batch_id).first()
else:
# Default to latest non-rolled-back completed batch
batch = (
query.filter(
BatchRecord.status.in_(["COMPLETED", "PARTIAL_FAILURE", "PARTIAL_ROLLBACK"]),
BatchRecord.dry_run == False,
)
.order_by(BatchRecord.created_at.desc())
.first()
)
if not batch:
logger.warning("No qualifying batch found for rollback", requested_id=batch_id)
return 0
logger.info("Initiating rollback", batch_id=batch.id)
# Query committed operations in reverse execution order
ops = (
self.session.query(Operation)
.filter_by(batch_id=batch.id, status=OperationStatus.COMMITTED.value)
.order_by(Operation.id.desc())
.all()
)
reverted_count = 0
failed_count = 0
reverted_dest_dirs: Set[Path] = set()
for op in ops:
src = Path(op.src)
dst = Path(op.dst)
action = op.action
try:
if action == ActionType.MOVE.value:
if dst.exists():
src.parent.mkdir(parents=True, exist_ok=True)
self._safe_move(dst, src)
reverted_count += 1
reverted_dest_dirs.add(dst.parent)
# Restore backup if one was taken
if op.backup_path and Path(op.backup_path).exists():
self._safe_move(Path(op.backup_path), dst)
elif action == ActionType.COPY.value:
if dst.exists():
dst.unlink()
reverted_count += 1
reverted_dest_dirs.add(dst.parent)
elif action in (ActionType.LINK.value, ActionType.HARDLINK.value):
if dst.exists() or dst.is_symlink():
dst.unlink()
reverted_count += 1
reverted_dest_dirs.add(dst.parent)
op.status = OperationStatus.ROLLED_BACK.value
# Delete FileRecord for destination
rec = self.session.query(FileRecord).filter_by(path=str(dst)).first()
if rec:
self.session.delete(rec)
except Exception as e:
failed_count += 1
logger.error("Error reverting operation during rollback", op_id=op.id, error=str(e))
if failed_count == 0 and reverted_count > 0:
batch.status = "ROLLED_BACK"
elif reverted_count > 0:
batch.status = "PARTIAL_ROLLBACK"
else:
batch.status = "ROLLBACK_FAILED"
self.session.commit()
# Clean empty directories in destination tree
self._clean_empty_destination_dirs(reverted_dest_dirs)
return reverted_count
def _clean_empty_destination_dirs(self, dest_dirs: Set[Path]) -> None:
"""Prune empty parent folders in destination tree after rollback."""
dest_roots = {
self.settings.get_destination_path(cat).resolve()
for cat in ("movie", "tv", "anime", "music", "audiobook", "podcast", "photo", "home_video", "documentary", "quarantine")
}
dest_base = self.settings.get_destination_base_path().resolve()
dest_roots.add(dest_base)
candidate_dirs: Set[Path] = set()
for d in dest_dirs:
try:
curr = d.resolve()
while curr not in dest_roots and any(curr.is_relative_to(r) for r in dest_roots):
candidate_dirs.add(curr)
curr = curr.parent
except Exception:
continue
sorted_dirs = sorted(candidate_dirs, key=lambda p: len(p.parts), reverse=True)
for d in sorted_dirs:
if not d.exists() or not d.is_dir() or d in dest_roots:
continue
try:
entries = [
e for e in d.iterdir()
if e.name not in (".DS_Store", "Thumbs.db", "desktop.ini")
]
if not entries:
for junk in list(d.iterdir()):
try:
junk.unlink()
except Exception:
pass
d.rmdir()
except Exception:
pass
def rollback_all(self) -> int:
"""Roll back ALL completed, non-rolled-back batches in reverse chronological order."""
batches = (
self.session.query(BatchRecord)
.filter(
BatchRecord.status.in_(["COMPLETED", "PARTIAL_FAILURE", "PARTIAL_ROLLBACK"]),
BatchRecord.dry_run == False,
)
.order_by(BatchRecord.created_at.desc())
.all()
)
total_reverted = 0
for batch in batches:
total_reverted += self.rollback_batch(batch.id)
return total_reverted
def recover_interrupted_batches(self) -> int:
"""Clean up orphaned temp files and mark interrupted operations as FAILED."""
in_progress_ops = (
self.session.query(Operation)
.filter_by(status=OperationStatus.IN_PROGRESS.value)
.all()
)
recovered_count = 0
for op in in_progress_ops:
logger.warning("Found interrupted operation during crash recovery", op_id=op.id, src=op.src, dst=op.dst)
op.status = OperationStatus.FAILED.value
op.error_message = "Interrupted by system crash or process kill"
recovered_count += 1
if recovered_count > 0:
self.session.commit()
return recovered_count
def clean_empty_directories(self, moved_src_paths: List[Path]) -> int:
"""Remove empty parent directories and delete .txt / junk files in source folders after moving files.
Ascends from moved file parent folders up to, but never removing, the source root directories.
Also removes companion or orphaned .txt files left behind in source folders.
"""
source_roots = {p.resolve() for p in self.settings.get_source_paths()}
# Also include any parent roots if configured
candidate_dirs: Set[Path] = set()
for src in moved_src_paths:
try:
# Delete companion .txt file (e.g. Movie.txt alongside Movie.mkv)
comp = src.with_suffix(".txt")
if comp.is_file():
try:
comp.unlink()
logger.info("Deleted companion .txt file during cleanup", file=str(comp))
except Exception:
pass
parent = src.resolve().parent
while parent not in source_roots and any(parent.is_relative_to(root) for root in source_roots):
candidate_dirs.add(parent)
parent = parent.parent
except Exception:
continue
# Sort candidate directories deepest first (longest path / most parts first)
sorted_dirs = sorted(candidate_dirs, key=lambda d: len(d.parts), reverse=True)
removed_count = 0
for d in sorted_dirs:
if not d.exists() or not d.is_dir():
continue
# Ensure we never delete a configured source root
if d in source_roots:
continue
try:
# Delete any .txt files in candidate directories during cleanup
for child in list(d.iterdir()):
if child.is_file() and child.name.lower().endswith(".txt"):
try:
child.unlink()
logger.info("Deleted .txt file during cleanup", file=str(child))
except Exception:
pass
# Check if directory contains any remaining files or subdirs (ignoring OS junk and .txt files)
entries = [
e for e in d.iterdir()
if e.name not in (".DS_Store", "Thumbs.db", "desktop.ini") and not e.name.lower().endswith(".txt")
]
if not entries:
# Clean up junk files before rmdir
for junk in d.iterdir():
try:
junk.unlink()
except Exception:
pass
d.rmdir()
removed_count += 1
logger.info("Cleaned up empty source directory", directory=str(d))
except (OSError, PermissionError) as e:
logger.debug("Could not remove directory (not empty or permissions issue)", directory=str(d), error=str(e))
# Also clean up any .txt files left in source roots
for root in source_roots:
if root.exists() and root.is_dir():
try:
for child in list(root.iterdir()):
if child.is_file() and child.name.lower().endswith(".txt"):
try:
child.unlink()
logger.info("Deleted .txt file in source root during cleanup", file=str(child))
except Exception:
pass
except Exception:
pass
return removed_count
+299
View File
@@ -0,0 +1,299 @@
"""Library management and show/movie memory indexing engine.
Tracks known shows and movies in the library, syncs filesystem library directories,
and provides automatic show memory routing for incoming downloads.
"""
from __future__ import annotations
import os
import re
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
import structlog
from sqlalchemy.orm import Session
from .config import Settings
from .models import LibraryItem, utc_now
logger = structlog.get_logger(__name__)
VIDEO_EXTENSIONS = {".mkv", ".mp4", ".m4v", ".avi", ".mov", ".webm", ".ts", ".flv", ".wmv"}
def clean_show_title(raw: str) -> str:
"""Strip release tags, quality, year suffixes, and bracketed text from a directory name."""
clean = re.sub(r"\[[^\]]+\]|\([^\)]+\)", "", raw).strip()
# Strip trailing quality / encoding specs
clean = re.sub(
r"(?i)\b(1080p|720p|2160p|4k|bluray|bdrip|webrip|web-dl|x264|x265|hevc|h\.?264|h\.?265|dts|aac|ac3|remux|repack)\b.*",
"",
clean,
)
# Strip Season pack identifiers like S01-S08 or Season 1
clean = re.sub(r"(?i)\b(?:s\d+[-_s\d]*|season\s*\d+.*)\b", "", clean)
clean = re.sub(r"[\._]+", " ", clean).strip(" -_")
return clean if len(clean) >= 2 else raw.strip()
def sync_library_from_disk(session: Session, settings: Settings) -> Dict[str, int]:
"""Scan configured SHOWS_DIR and MOVIES_DIR on disk and synchronize library_items."""
shows_dir = settings.get_destination_path("tv")
movies_dir = settings.get_destination_path("movie")
shows_count = 0
movies_count = 0
# Cache existing records in memory by (title.lower(), category)
existing_items: Dict[Tuple[str, str], LibraryItem] = {
(item.title.lower(), item.category): item for item in session.query(LibraryItem).all()
}
# 1. Scan Shows Directory
if shows_dir.exists() and shows_dir.is_dir():
try:
for entry in shows_dir.iterdir():
if entry.name.startswith(".") or not entry.is_dir():
continue
folder_name = entry.name
title = clean_show_title(folder_name)
if not title:
continue
# Count video files and detect seasons
episodes = 0
seasons = set()
try:
for root, _, files in os.walk(entry):
for f in files:
ext = os.path.splitext(f)[1].lower()
if ext in VIDEO_EXTENSIONS:
episodes += 1
s_m = re.search(r"(?i)\b(?:season|s)\s*(\d{1,2})\b", Path(root).name)
if s_m:
seasons.add(int(s_m.group(1)))
except Exception:
pass
key = (title.lower(), "tv")
if key in existing_items:
item = existing_items[key]
item.destination_folder = str(entry)
item.item_count = max(item.item_count, episodes)
item.seasons_count = max(item.seasons_count, len(seasons))
item.last_updated = utc_now()
else:
item = LibraryItem(
title=title,
category="tv",
destination_folder=str(entry),
item_count=episodes,
seasons_count=len(seasons),
first_detected=utc_now(),
last_updated=utc_now(),
)
session.add(item)
existing_items[key] = item
shows_count += 1
except Exception as e:
logger.error("Error scanning shows directory for library", error=str(e))
# 2. Scan Movies Directory
if movies_dir.exists() and movies_dir.is_dir():
try:
for entry in movies_dir.iterdir():
if entry.name.startswith("."):
continue
title = entry.name
year = None
y_m = re.search(r"\b(19\d\d|20\d\d)\b", entry.name)
if y_m:
year = int(y_m.group(1))
title = entry.name[: y_m.start()].strip(" (.-_")
clean = clean_show_title(title)
item_files = 1
if entry.is_dir():
try:
item_files = sum(
1 for _, _, files in os.walk(entry)
for f in files if os.path.splitext(f)[1].lower() in VIDEO_EXTENSIONS
)
except Exception:
pass
key = (clean.lower(), "movie")
if key in existing_items:
item = existing_items[key]
item.destination_folder = str(entry)
item.year = year or item.year
item.item_count = max(item.item_count, item_files)
item.last_updated = utc_now()
else:
item = LibraryItem(
title=clean,
category="movie",
year=year,
destination_folder=str(entry),
item_count=item_files,
first_detected=utc_now(),
last_updated=utc_now(),
)
session.add(item)
existing_items[key] = item
movies_count += 1
except Exception as e:
logger.error("Error scanning movies directory for library", error=str(e))
session.commit()
logger.info("Library synchronized with disk", shows=shows_count, movies=movies_count)
return {"shows_synced": shows_count, "movies_synced": movies_count}
def record_detected_item(
session: Session,
settings: Settings,
title: str,
category: str,
destination_folder: Optional[str] = None,
year: Optional[int] = None,
poster_url: Optional[str] = None,
delta_count: int = 0,
) -> LibraryItem:
"""Record or update a show or movie in the library database."""
category = category.lower()
if category not in ("tv", "movie"):
category = "tv"
clean = clean_show_title(title) if category == "tv" else title.strip()
item = session.query(LibraryItem).filter_by(title=clean, category=category).first()
if not destination_folder:
dest_base = settings.get_destination_path(category)
destination_folder = str(dest_base / clean)
if item:
if destination_folder:
item.destination_folder = destination_folder
if year:
item.year = year
if poster_url and not item.poster_url:
item.poster_url = poster_url
if delta_count:
item.item_count = max(0, item.item_count + delta_count)
item.last_updated = utc_now()
else:
item = LibraryItem(
title=clean,
category=category,
year=year,
destination_folder=destination_folder,
poster_url=poster_url,
item_count=max(0, delta_count),
first_detected=utc_now(),
last_updated=utc_now(),
)
session.add(item)
session.commit()
return item
def get_known_shows(session: Session) -> List[Dict[str, Any]]:
"""Return all known TV shows in the library for matching."""
items = session.query(LibraryItem).filter_by(category="tv").all()
shows = []
for item in items:
clean = item.title.strip()
if len(clean) >= 2:
shows.append({
"title": clean,
"raw_title": item.title,
"destination_folder": item.destination_folder,
"poster_url": item.poster_url,
"item_count": item.item_count,
})
# Sort by title length descending so longer specific titles match first
shows.sort(key=lambda x: len(x["title"]), reverse=True)
return shows
def match_known_show(filename_or_text: str, known_shows: List[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
"""Check if filename_or_text contains or matches a known show in the library."""
if not filename_or_text or not known_shows:
return None
# Replace separators with spaces
normalized = re.sub(r"[\._]+", " ", filename_or_text)
for show in known_shows:
title = show["title"]
if len(title) < 3:
continue
# Check whole word match
pattern = r"(?i)(?<![a-z0-9])" + re.escape(title) + r"(?![a-z0-9])"
if re.search(pattern, normalized):
return show
return None
def list_library_items(
session: Session,
category: Optional[str] = None,
search: Optional[str] = None,
) -> Dict[str, Any]:
"""List library items with counts, optionally filtered by category and search term."""
q = session.query(LibraryItem)
if category and category.lower() in ("tv", "movie"):
q = q.filter_by(category=category.lower())
if search:
s = f"%{search.strip()}%"
q = q.filter(LibraryItem.title.ilike(s))
items = q.order_by(LibraryItem.title.asc()).all()
total_shows = session.query(LibraryItem).filter_by(category="tv").count()
total_movies = session.query(LibraryItem).filter_by(category="movie").count()
shows_list = []
movies_list = []
for item in items:
d = {
"id": item.id,
"title": item.title,
"category": item.category,
"year": item.year,
"destination_folder": item.destination_folder,
"poster_url": item.poster_url,
"item_count": item.item_count,
"seasons_count": item.seasons_count,
"first_detected": item.first_detected.isoformat() if item.first_detected else None,
"last_updated": item.last_updated.isoformat() if item.last_updated else None,
}
if item.category == "tv":
shows_list.append(d)
else:
movies_list.append(d)
return {
"total_shows": total_shows,
"total_movies": total_movies,
"shows": shows_list,
"movies": movies_list,
}
def clear_library(session: Session) -> int:
"""Clear all indexed show and movie items from the library catalog database."""
deleted_count = session.query(LibraryItem).delete()
session.commit()
logger.info("Library catalog cleared", deleted_count=deleted_count)
return deleted_count
+184
View File
@@ -0,0 +1,184 @@
"""SQLAlchemy database models for Media Sorter.
Provides data structures for file tracking, operation journaling, quarantine,
batch execution, and configuration auditing.
"""
from __future__ import annotations
import datetime
from enum import Enum
from typing import Any, Dict, Optional
from sqlalchemy import (
Boolean,
Column,
DateTime,
Float,
ForeignKey,
Index,
Integer,
JSON,
String,
Text,
UniqueConstraint,
)
from sqlalchemy.orm import declarative_base, relationship
Base = declarative_base()
class OperationStatus(str, Enum):
PLANNED = "PLANNED"
IN_PROGRESS = "IN_PROGRESS"
COMMITTED = "COMMITTED"
FAILED = "FAILED"
ROLLED_BACK = "ROLLED_BACK"
SKIPPED = "SKIPPED"
class QuarantineStatus(str, Enum):
PENDING = "PENDING"
RESOLVED = "RESOLVED"
IGNORED = "IGNORED"
def utc_now():
return datetime.datetime.now(datetime.timezone.utc)
class BatchRecord(Base):
"""Tracks an execution batch (one invocation of media-sorter run or dry-run)."""
__tablename__ = "batches"
id = Column(String(36), primary_key=True) # UUIDv4
created_at = Column(DateTime(timezone=True), default=utc_now, nullable=False)
completed_at = Column(DateTime(timezone=True), nullable=True)
dry_run = Column(Boolean, default=False, nullable=False)
status = Column(String(32), default="IN_PROGRESS", nullable=False) # IN_PROGRESS, COMPLETED, FAILED, ROLLED_BACK
total_files = Column(Integer, default=0, nullable=False)
moved_files = Column(Integer, default=0, nullable=False)
skipped_files = Column(Integer, default=0, nullable=False)
failed_files = Column(Integer, default=0, nullable=False)
quarantined_files = Column(Integer, default=0, nullable=False)
operations = relationship("Operation", back_populates="batch", cascade="all, delete-orphan")
__table_args__ = (
Index("ix_batches_created_at", "created_at"),
)
class FileRecord(Base):
"""Tracks known files to avoid redundant probing and detect changes."""
__tablename__ = "files"
id = Column(Integer, primary_key=True, autoincrement=True)
path = Column(String(1024), unique=True, nullable=False)
size = Column(Integer, nullable=False)
mtime = Column(Float, nullable=False)
content_hash = Column(String(64), nullable=True)
status = Column(String(32), default="scanned", nullable=False) # scanned, organized, quarantined, skipped, error
category = Column(String(32), nullable=True)
confidence = Column(Float, nullable=True)
first_seen = Column(DateTime(timezone=True), default=utc_now, nullable=False)
last_processed = Column(DateTime(timezone=True), nullable=True)
__table_args__ = (
Index("ix_files_path", "path"),
Index("ix_files_status", "status"),
)
class Operation(Base):
"""Operation journal entry for atomic moves, copies, or links."""
__tablename__ = "operations"
id = Column(Integer, primary_key=True, autoincrement=True)
batch_id = Column(String(36), ForeignKey("batches.id"), nullable=False)
src = Column(String(1024), nullable=False)
dst = Column(String(1024), nullable=False)
action = Column(String(32), nullable=False) # move, copy, link, hardlink
status = Column(String(32), default=OperationStatus.PLANNED.value, nullable=False)
category = Column(String(32), nullable=True)
confidence = Column(Float, nullable=True)
src_hash = Column(String(64), nullable=True)
dst_hash = Column(String(64), nullable=True)
backup_path = Column(String(1024), nullable=True)
details = Column(JSON, nullable=True) # reasoning, sidecars, format info
error_message = Column(Text, nullable=True)
created_at = Column(DateTime(timezone=True), default=utc_now, nullable=False)
completed_at = Column(DateTime(timezone=True), nullable=True)
batch = relationship("BatchRecord", back_populates="operations")
__table_args__ = (
Index("ix_operations_batch_id", "batch_id"),
Index("ix_operations_status", "status"),
Index("ix_operations_src", "src"),
Index("ix_operations_dst", "dst"),
)
class QuarantineRecord(Base):
"""Stores files that failed confidence threshold or require manual user review."""
__tablename__ = "quarantine"
id = Column(Integer, primary_key=True, autoincrement=True)
src = Column(String(1024), unique=True, nullable=False)
suggested_category = Column(String(32), nullable=True)
confidence = Column(Float, nullable=True)
reason = Column(String(256), nullable=False)
signals = Column(JSON, nullable=True) # Diagnostic details of why it was flagged
status = Column(String(32), default=QuarantineStatus.PENDING.value, nullable=False)
resolved_path = Column(String(1024), nullable=True)
created_at = Column(DateTime(timezone=True), default=utc_now, nullable=False)
resolved_at = Column(DateTime(timezone=True), nullable=True)
__table_args__ = (
Index("ix_quarantine_status", "status"),
Index("ix_quarantine_src", "src"),
)
class ConfigAudit(Base):
"""Tracks configuration states for reproducibility and auditing."""
__tablename__ = "config_audit"
id = Column(Integer, primary_key=True, autoincrement=True)
loaded_at = Column(DateTime(timezone=True), default=utc_now, nullable=False)
config_json = Column(JSON, nullable=False)
__table_args__ = (
Index("ix_config_loaded", "loaded_at"),
)
class LibraryItem(Base):
"""Tracks known shows and movies in the user's library for automated routing and cataloging."""
__tablename__ = "library_items"
id = Column(Integer, primary_key=True, autoincrement=True)
title = Column(String(256), nullable=False)
category = Column(String(32), nullable=False) # "tv" or "movie"
year = Column(Integer, nullable=True)
destination_folder = Column(String(1024), nullable=False)
poster_url = Column(String(1024), nullable=True)
item_count = Column(Integer, default=0, nullable=False)
seasons_count = Column(Integer, default=0, nullable=False)
first_detected = Column(DateTime(timezone=True), default=utc_now, nullable=False)
last_updated = Column(DateTime(timezone=True), default=utc_now, nullable=False)
extra_info = Column(JSON, nullable=True)
__table_args__ = (
UniqueConstraint("title", "category", name="uq_library_title_category"),
Index("ix_library_category", "category"),
Index("ix_library_title", "title"),
)
+413
View File
@@ -0,0 +1,413 @@
"""Naming and path formatting engine for Media Sorter.
Renders user-defined naming templates, safely formats multi-part tags, pairs sidecars
with primary media files, and enforces rigorous cross-platform filename sanitization
(Linux, Windows, macOS, NTFS, SMB/NFS, exFAT).
"""
from __future__ import annotations
import os
import re
import unicodedata
from pathlib import Path
from typing import Any, Dict, Optional
from .classifier import ClassificationResult
from .config import Settings
# Windows reserved device names
RESERVED_NAMES = {
"CON", "PRN", "AUX", "NUL",
"COM1", "COM2", "COM3", "COM4", "COM5", "COM6", "COM7", "COM8", "COM9",
"LPT1", "LPT2", "LPT3", "LPT4", "LPT5", "LPT6", "LPT7", "LPT8", "LPT9",
}
# Illegal characters across file systems (< > : " / \ | ? *)
FORBIDDEN_CHARS_PATTERN = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
def sanitize_filename_component(name: str, max_length: int = 240) -> str:
"""Sanitize an individual filename or folder name component for safe cross-platform use."""
# 1. Unicode normalization (NFC)
clean = unicodedata.normalize("NFC", name)
# 2. Replace forbidden characters with safe hyphen or space
clean = FORBIDDEN_CHARS_PATTERN.sub("-", clean)
# 3. Collapse multiple whitespace and hyphens
clean = re.sub(r"\s+", " ", clean)
clean = re.sub(r"-{2,}", "-", clean)
# 4. Strip leading/trailing spaces, dots, and hyphens (vital for Windows / SMB)
clean = clean.strip(" .-")
if not clean:
clean = "unnamed"
# 5. Check Windows reserved words
upper_base = clean.split(".")[0].upper()
if upper_base in RESERVED_NAMES:
clean = f"_{clean}"
# 6. Truncate byte length for filesystem limits (e.g. 255 bytes on ext4/NTFS/ZFS)
encoded = clean.encode("utf-8")
if len(encoded) > max_length:
parts = clean.rsplit(".", 1)
if len(parts) == 2 and 1 <= len(parts[1]) <= 10:
base, ext = parts
ext_bytes = len(f".{ext}".encode("utf-8"))
avail = max(max_length - ext_bytes, 10)
base_enc = base.encode("utf-8")[:avail]
base_clean = base_enc.decode("utf-8", errors="ignore").rstrip(" .-")
clean = f"{base_clean}.{ext}" if base_clean else ext
else:
clean = encoded[:max_length].decode("utf-8", errors="ignore").rstrip(" .-")
clean = clean.strip(" .-")
return clean if clean else "unnamed"
DEFAULT_TEMPLATES = {
"tv": "{title}/Season {season:02d}/{show_name}_{season_episode}.{ext}",
"movie": "{title} ({year})/{movie_name}.{ext}",
"anime": "{title}/Season {season:02d}/{show_name}_{season_episode} [{group}].{ext}",
}
class MediaNamer:
"""Renders organized destination paths from templates and classification results."""
def __init__(self, settings: Settings):
self.settings = settings
def generate_destination_path(
self,
cls_result: ClassificationResult,
primary_dst_path: Optional[Path] = None,
) -> Path:
"""Construct full destination path for a given file and its classification."""
category = cls_result.category
base_dir = self.settings.get_destination_path(category)
src_path = cls_result.metadata.path if cls_result.metadata else Path("unknown")
ext = src_path.suffix.lstrip(".")
# Handle Quarantine routing
if cls_result.needs_quarantine or category == "unknown":
reason = cls_result.quarantine_reason or "low_confidence"
safe_reason = sanitize_filename_component(reason)
q_template = self.settings.templates.quarantine
filename = sanitize_filename_component(src_path.name)
rel_str = q_template.format(reason=safe_reason, filename=filename, ext=ext)
return (self.settings.get_destination_path("quarantine") / rel_str).resolve()
# Handle Sidecars (Subtitles, Artwork, Metadata, Extras)
if category in ("subtitle", "artwork", "metadata"):
return self._format_sidecar_path(cls_result, primary_dst_path, base_dir)
# Retrieve template
template = getattr(self.settings.templates, category, None)
context = self._build_context(cls_result)
if category == "tv" and (not template or template == DEFAULT_TEMPLATES.get("tv")):
formatted_rel = self._format_tv_path(cls_result, context)
elif category == "movie" and (not template or template == DEFAULT_TEMPLATES.get("movie")):
formatted_rel = self._format_movie_path(cls_result, context)
elif category == "anime" and (not template or template == DEFAULT_TEMPLATES.get("anime")):
formatted_rel = self._format_anime_path(cls_result, context)
elif category == "podcast" and (not template or template == "{show}/{year}/{show} - {date} - {title}.{ext}"):
formatted_rel = self._format_podcast_path(cls_result, context)
else:
if not template:
template = "{filename}.{ext}"
formatted_rel = self._render_template(template, context)
# If file renaming is disabled, preserve original source filename
if not getattr(self.settings.general, "rename_files", True) and src_path.name != "unknown":
rel_path = Path(formatted_rel)
if len(rel_path.parts) > 1:
formatted_rel = str(rel_path.parent / src_path.name)
else:
formatted_rel = src_path.name
# Sanitize each path component separately to preserve folder hierarchy
parts = Path(formatted_rel).parts
sanitized_parts = [sanitize_filename_component(p) for p in parts]
return (base_dir / Path(*sanitized_parts)).resolve()
def _format_tv_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
show_name = context["show_name"]
ext = context["ext"]
tokens = cls_result.tokens
# Daily / dated broadcast TV formatting
date_val = context.get("date_val")
if (tokens and tokens.is_daily) or (date_val and (not tokens or not tokens.season or tokens.season > 1000)):
year = context.get("year")
if not year or year == "Unknown":
year = date_val.split("-")[0] if date_val else "Unknown"
return f"{show_name}/Season {year}/{show_name} - {date_val}.{ext}"
# Standard TV formatting (supporting Season 00, multi-ep, and season pack)
season_num = context["season"]
season_folder = f"Season {season_num:02d}"
season_episode = context["season_episode"]
return f"{show_name}/{season_folder}/{show_name} - {season_episode}.{ext}"
def _format_movie_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
title = context["title"]
year = context["year"]
ext = context["ext"]
has_year = year and year != "Unknown"
folder_name = f"{title} ({year})" if has_year else title
base_name = f"{title} ({year})" if has_year else title
edition_tag = context.get("edition_tag", "")
part_tag = context.get("part_tag", "")
extra_tag = context.get("extra_tag", "")
return f"{folder_name}/{base_name}{edition_tag}{part_tag}{extra_tag}.{ext}"
def _format_anime_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
title = context["title"]
ext = context["ext"]
tokens = cls_result.tokens
group_tag = context.get("group_tag", "")
# Multi-episode anime
if tokens and tokens.multi_episodes and len(tokens.multi_episodes) >= 2:
first_ep = tokens.multi_episodes[0]
last_ep = tokens.multi_episodes[-1]
ep_str = f"{first_ep:02d}-{last_ep:02d}"
return f"{title}/{title} - {ep_str}{group_tag}.{ext}"
# Single episode anime
if tokens and tokens.episode is not None:
ep = tokens.episode
ep_str = f"{ep:02d}" if ep < 10 else str(ep)
return f"{title}/{title} - {ep_str}{group_tag}.{ext}"
# Anime movie or special without episode number
return f"{title}/{title}{group_tag}.{ext}"
def _format_podcast_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
show = context.get("show") or context.get("artist") or "Unknown Show"
year = context.get("year")
date = context.get("date")
title = context.get("title")
ext = context.get("ext")
if title and title != show and title != "Unknown":
return f"{show}/{year}/{show} - {date} - {title}.{ext}"
return f"{show}/{year}/{show} - {date}.{ext}"
def _format_sidecar_path(
self,
cls_result: ClassificationResult,
primary_dst_path: Optional[Path],
base_dir: Path,
) -> Path:
src_path = cls_result.metadata.path
ext = src_path.suffix.lstrip(".")
if primary_dst_path:
parent_dir = primary_dst_path.parent
primary_stem = primary_dst_path.stem
if cls_result.category == "subtitle":
# Detect language code or compound tag in subtitle (e.g. movie.en.srt, movie.forced.srt)
src_stem = src_path.stem
m = re.search(
r"\.((?:[a-zA-Z]{2,3}\.)?(?:forced|sdh|cc)|[a-zA-Z]{2,3}(?:-[a-zA-Z]{2,4})?)$",
src_stem,
re.IGNORECASE,
)
if m:
lang_suffix = f".{m.group(1)}"
else:
parts = src_stem.split(".")
if len(parts) > 1 and len(parts[-1]) in (2, 3, 6):
lang_suffix = f".{parts[-1]}"
else:
lang_suffix = ""
new_filename = f"{primary_stem}{lang_suffix}.{ext}"
return parent_dir / sanitize_filename_component(new_filename)
elif cls_result.category == "artwork":
# e.g. poster.jpg, cover.jpg in the same movie/show folder
return parent_dir / sanitize_filename_component(src_path.name)
elif cls_result.category == "metadata":
# NFO file matches primary stem or stays alongside
new_filename = f"{primary_stem}.{ext}"
return parent_dir / sanitize_filename_component(new_filename)
# If orphan sidecar (no primary matched), place into respective folder
sanitized_name = sanitize_filename_component(src_path.name)
return (base_dir / sanitized_name).resolve()
def _build_context(self, res: ClassificationResult) -> Dict[str, Any]:
tokens = res.tokens
meta = res.metadata
src_path = meta.path if meta else Path("file")
# Fix Season 00 / Episode 00 falsy bug
season_num = tokens.season if (tokens and tokens.season is not None) else 1
episode_num = tokens.episode if (tokens and tokens.episode is not None) else 1
# Format season_episode string with multi-episode and season pack support
if tokens and tokens.multi_episodes and len(tokens.multi_episodes) >= 2:
season_ep_str = f"S{season_num:02d}E{tokens.multi_episodes[0]:02d}-E{tokens.multi_episodes[-1]:02d}"
elif tokens and (tokens.is_season_pack or (tokens.season is not None and tokens.episode is None and not getattr(tokens, "multi_episodes", None))):
season_ep_str = f"Season {season_num:02d}"
else:
season_ep_str = f"S{season_num:02d}E{episode_num:02d}"
main_title = (tokens.title if tokens else None) or src_path.stem
# Clean release group: omit when unknown, NEVER emit 'UnknownGroup'
group_val = tokens.group if (tokens and tokens.group and tokens.group != "UnknownGroup") else ""
group_tag = f" [{group_val}]" if group_val else ""
# Extract movie edition, part, and extra tags
edition_val = getattr(tokens, "edition", None) if tokens else None
if not edition_val:
em = re.search(r"\b(extended|directors?\.cut|remastered|criterion(?:\.collection)?|final\.cut)\b", src_path.stem, re.I)
if em:
raw_ed = em.group(1).lower().replace(".", " ")
if "director" in raw_ed:
edition_val = "Director's Cut"
elif "criterion" in raw_ed:
edition_val = "Criterion"
elif "final" in raw_ed:
edition_val = "Final Cut"
elif "remaster" in raw_ed:
edition_val = "Remastered"
elif "extend" in raw_ed:
edition_val = "Extended"
edition_tag = f" [{edition_val}]" if edition_val else ""
part_val = getattr(tokens, "part", None) if tokens else None
part_label = getattr(tokens, "part_label", None) if tokens else None
if part_val is None:
pm = re.search(r"\b(?:cd|part|pt)[\.\s_-]*(\d+)\b", src_path.stem, re.I)
if pm:
part_val = int(pm.group(1))
part_label = f"Pt.{part_val}"
elif not part_label:
part_label = f"Pt.{part_val}"
part_tag = f" [{part_label}]" if part_label else ""
extra_m = re.search(r"-(behindthescenes|deleted|trailer|featurette)\b", src_path.stem, re.I)
extra_tag = f"-{extra_m.group(1).lower()}" if extra_m else ""
# Date resolution for daily TV shows and podcasts
date_val = getattr(tokens, "air_date", None) or (tokens.date_stamp if tokens and not tokens.is_photo_or_home_video else None)
if not date_val:
dm = re.search(r"\b((?:19|20)\d{2})[-._](0[1-9]|1[0-2])[-._](0[1-9]|[12]\d|3[01])\b", src_path.stem)
if dm:
date_val = f"{dm.group(1)}-{dm.group(2)}-{dm.group(3)}"
ctx: Dict[str, Any] = {
"ext": src_path.suffix.lstrip("."),
"filename": src_path.stem,
"title": main_title,
"show_name": main_title,
"SHOW_NAME": main_title,
"movie_name": main_title,
"MOVIE_NAME": main_title,
"season_episode": season_ep_str,
"SEASON_EPISODE": season_ep_str,
"year": (tokens.year if tokens else None) or "Unknown",
"season": season_num,
"episode": episode_num,
"episode_title": (tokens.episode_title if tokens else None) or f"Episode {episode_num}",
"artist": (tokens.artist if tokens else None) or "Unknown Artist",
"album": (tokens.album if tokens else None) or "Unknown Album",
"track": (tokens.track if tokens else 1) or 1,
"disc": (tokens.disc if tokens else 1) or 1,
"group": group_val,
"group_tag": group_tag,
"edition": edition_val,
"edition_tag": edition_tag,
"part": part_val,
"part_label": part_label,
"part_tag": part_tag,
"extra_tag": extra_tag,
"date_val": date_val,
"resolution": (tokens.resolution if tokens and tokens.resolution else (meta.resolution_label if meta else "")),
"codec": (tokens.video_codec or (meta.codec_video if meta else "h264")),
"author": (tokens.artist if tokens else None) or "Unknown Author",
"chapter": (tokens.title if tokens else None) or f"Chapter {tokens.track if tokens else 1}",
"show": (tokens.artist if tokens else None) or "Unknown Show",
"date": date_val or ((tokens.date_stamp if tokens else "2026-01-01") or "2026-01-01"),
"month": 1,
"day": 1,
"time": "000000",
"camera": "Camera",
"event": "Event",
}
# Override from provider result if available
if res.provider_result:
p = res.provider_result
if p.canonical_title:
ctx["title"] = p.canonical_title
ctx["show_name"] = p.canonical_title
ctx["SHOW_NAME"] = p.canonical_title
ctx["movie_name"] = p.canonical_title
ctx["MOVIE_NAME"] = p.canonical_title
if p.year:
ctx["year"] = p.year
if p.episode_title:
ctx["episode_title"] = p.episode_title
if p.artist:
ctx["artist"] = p.artist
if p.album:
ctx["album"] = p.album
# Parse date stamp fields if present
date_source = (tokens and tokens.date_stamp) or (ctx.get("date") if res.category in ("home_video", "photo", "podcast") else None)
if date_source and date_source != "Unknown":
date_parts = str(date_source).split("-")
if len(date_parts) == 3:
try:
if ctx["year"] == "Unknown":
ctx["year"] = int(date_parts[0])
ctx["month"] = int(date_parts[1])
ctx["day"] = int(date_parts[2])
except ValueError:
pass
# Clean tags from meta
if meta and meta.tags:
if "camera_model" in meta.tags:
ctx["camera"] = sanitize_filename_component(meta.tags["camera_model"])
if "album" in meta.tags and not ctx.get("album"):
ctx["album"] = meta.tags["album"]
if "artist" in meta.tags and not ctx.get("artist"):
ctx["artist"] = meta.tags["artist"]
return ctx
def _render_template(self, template: str, context: Dict[str, Any]) -> str:
"""Format template while gracefully cleaning empty technical brackets."""
# Normalize <TAG> to {TAG} for convenience if users use angle brackets
rendered = re.sub(r"<([a-zA-Z_0-9]+)>", r"{\1}", template)
try:
rendered = rendered.format(**context)
except (KeyError, ValueError):
# Safe token replacement if format specifier fails
safe_ctx = {k: str(v) if v is not None else "" for k, v in context.items()}
# Remove format specifiers like :02d
simplified = re.sub(r"\{(\w+):[^}]+\}", r"{\1}", rendered)
try:
rendered = simplified.format(**safe_ctx)
except Exception:
rendered = f"{context.get('title', 'media')}.{context.get('ext', 'bin')}"
# Clean empty technical brackets such as "[]" or "[ ]" or "()"
rendered = re.sub(r"\[\s*\]", "", rendered)
rendered = re.sub(r"\(\s*\)", "", rendered)
rendered = re.sub(r"\s{2,}", " ", rendered)
return rendered.strip()
+67
View File
@@ -0,0 +1,67 @@
"""Notification dispatcher for Media Sorter.
Sends batch summary alerts and error notices to webhook endpoints (Slack, Discord, generic JSON).
"""
from __future__ import annotations
from typing import Any, Dict, Optional
import requests
import structlog
from .config import NotificationSettings
from .executor import BatchExecutionReport
logger = structlog.get_logger(__name__)
def send_batch_notification(
settings: NotificationSettings, report: BatchExecutionReport
) -> bool:
"""Send webhook alert for completed or failed batch."""
if not settings.enabled or not settings.webhook_url:
return False
is_failure = report.failed_files > 0
if is_failure and not settings.notify_on_failure:
return False
if not is_failure and not settings.notify_on_complete:
return False
status_str = "FAILED" if is_failure else ("DRY RUN PREVIEW" if report.dry_run else "SUCCESS")
title = f"Media Sorter: {status_str} [Batch {report.batch_id[:8]}]"
summary_text = (
f"**{title}**\n"
f"• Total Files: {report.total_files}\n"
f"• Organized/Moved: {report.moved_files}\n"
f"• Copied: {report.copied_files}\n"
f"• Skipped: {report.skipped_files}\n"
f"• Quarantined: {report.quarantined_files}\n"
f"• Failures: {report.failed_files}\n"
)
if report.errors:
summary_text += f"\nErrors:\n" + "\n".join(f"- {e}" for e in report.errors[:5])
payload: Dict[str, Any] = {
"text": summary_text,
"content": summary_text, # Discord format compatibility
"batch_id": report.batch_id,
"dry_run": report.dry_run,
"status": status_str,
"total_files": report.total_files,
"moved_files": report.moved_files,
"quarantined_files": report.quarantined_files,
"failed_files": report.failed_files,
}
try:
resp = requests.post(settings.webhook_url, json=payload, timeout=5.0)
if resp.status_code in (200, 201, 204):
return True
logger.warning("Webhook dispatch failed", status=resp.status_code)
except Exception as e:
logger.warning("Failed sending notification webhook", error=str(e))
return False
+255
View File
@@ -0,0 +1,255 @@
"""Metadata provider integrations for Media Sorter.
Provides interfaces and implementations for querying external metadata (TMDB, TVDB,
MusicBrainz) with rate-limiting, request caching, and offline fallbacks.
"""
from __future__ import annotations
import hashlib
import json
import time
from abc import ABC, abstractmethod
from dataclasses import dataclass
from typing import Any, Dict, List, Optional, Tuple
import requests
import structlog
logger = structlog.get_logger(__name__)
@dataclass
class ProviderResult:
canonical_title: str
year: Optional[int] = None
media_type: str = "movie" # movie, tv, anime, music
season: Optional[int] = None
episode: Optional[int] = None
episode_title: Optional[str] = None
artist: Optional[str] = None
album: Optional[str] = None
genres: List[str] = None
confidence_boost: float = 0.15
raw_payload: Optional[Dict[str, Any]] = None
class MetadataProvider(ABC):
"""Abstract base class for all metadata providers."""
@abstractmethod
def search_movie(self, title: str, year: Optional[int] = None) -> Optional[ProviderResult]:
pass
@abstractmethod
def search_tv(self, title: str, year: Optional[int] = None, season: Optional[int] = None, episode: Optional[int] = None) -> Optional[ProviderResult]:
pass
@abstractmethod
def search_music(self, artist: str, album: Optional[str] = None, title: Optional[str] = None) -> Optional[ProviderResult]:
pass
class MemoryCache:
"""In-memory cache with TTL for metadata queries."""
def __init__(self, ttl_seconds: int = 86400):
self.ttl = ttl_seconds
self._store: Dict[str, Tuple[float, Any]] = {}
def get(self, key: str) -> Optional[Any]:
if key in self._store:
timestamp, data = self._store[key]
if time.time() - timestamp < self.ttl:
return data
del self._store[key]
return None
def set(self, key: str, data: Any) -> None:
self._store[key] = (time.time(), data)
class TMDBProvider(MetadataProvider):
"""TheMovieDatabase (TMDB) API provider with rate-limiting and caching."""
BASE_URL = "https://api.themoviedb.org/3"
def __init__(self, api_key: Optional[str] = None, rate_limit_per_second: float = 2.0, cache_ttl_seconds: int = 86400):
self.api_key = api_key
self.min_interval = 1.0 / max(rate_limit_per_second, 0.1)
self.last_request_time = 0.0
self.cache = MemoryCache(ttl_seconds=cache_ttl_seconds)
def _throttle(self) -> None:
elapsed = time.time() - self.last_request_time
if elapsed < self.min_interval:
time.sleep(self.min_interval - elapsed)
self.last_request_time = time.time()
def _query(self, endpoint: str, params: Dict[str, Any]) -> Optional[Dict[str, Any]]:
if not self.api_key:
return None
cache_key = f"tmdb:{endpoint}:{json.dumps(params, sort_keys=True)}"
cached = self.cache.get(cache_key)
if cached is not None:
return cached
self._throttle()
req_params = dict(params)
req_params["api_key"] = self.api_key
try:
resp = requests.get(f"{self.BASE_URL}/{endpoint}", params=req_params, timeout=5.0)
if resp.status_code == 200:
data = resp.json()
self.cache.set(cache_key, data)
return data
logger.warning("TMDB request failed", status=resp.status_code, endpoint=endpoint)
except Exception as e:
logger.warning("TMDB network error", error=str(e))
return None
def search_movie(self, title: str, year: Optional[int] = None) -> Optional[ProviderResult]:
params: Dict[str, Any] = {"query": title}
if year:
params["year"] = year
data = self._query("search/movie", params)
if not data or not data.get("results"):
return None
first = data["results"][0]
release_date = first.get("release_date", "")
res_year = int(release_date[:4]) if len(release_date) >= 4 and release_date[:4].isdigit() else year
return ProviderResult(
canonical_title=first.get("title", title),
year=res_year,
media_type="movie",
confidence_boost=0.15,
raw_payload=first,
)
def search_tv(self, title: str, year: Optional[int] = None, season: Optional[int] = None, episode: Optional[int] = None) -> Optional[ProviderResult]:
params: Dict[str, Any] = {"query": title}
if year:
params["first_air_date_year"] = year
data = self._query("search/tv", params)
if not data or not data.get("results"):
return None
first = data["results"][0]
show_id = first.get("id")
show_title = first.get("name", title)
air_date = first.get("first_air_date", "")
res_year = int(air_date[:4]) if len(air_date) >= 4 and air_date[:4].isdigit() else year
ep_title = None
if show_id and season is not None and episode is not None:
ep_data = self._query(f"tv/{show_id}/season/{season}/episode/{episode}", {})
if ep_data:
ep_title = ep_data.get("name")
return ProviderResult(
canonical_title=show_title,
year=res_year,
media_type="tv",
season=season,
episode=episode,
episode_title=ep_title,
confidence_boost=0.20,
raw_payload=first,
)
def search_music(self, artist: str, album: Optional[str] = None, title: Optional[str] = None) -> Optional[ProviderResult]:
return None # TMDB does not index music
class MusicBrainzProvider(MetadataProvider):
"""MusicBrainz WS2 API provider with courteous rate-limiting (1 req/sec)."""
BASE_URL = "https://musicbrainz.org/ws/2"
def __init__(self, rate_limit_per_second: float = 1.0, cache_ttl_seconds: int = 86400):
self.min_interval = 1.0 / max(rate_limit_per_second, 0.1)
self.last_request_time = 0.0
self.cache = MemoryCache(ttl_seconds=cache_ttl_seconds)
def _throttle(self) -> None:
elapsed = time.time() - self.last_request_time
if elapsed < self.min_interval:
time.sleep(self.min_interval - elapsed)
self.last_request_time = time.time()
def search_movie(self, title: str, year: Optional[int] = None) -> Optional[ProviderResult]:
return None
def search_tv(self, title: str, year: Optional[int] = None, season: Optional[int] = None, episode: Optional[int] = None) -> Optional[ProviderResult]:
return None
def search_music(self, artist: str, album: Optional[str] = None, title: Optional[str] = None) -> Optional[ProviderResult]:
query_parts = [f'artist:"{artist}"']
if album:
query_parts.append(f'release:"{album}"')
if title:
query_parts.append(f'recording:"{title}"')
query_str = " AND ".join(query_parts)
cache_key = f"mb:{query_str}"
cached = self.cache.get(cache_key)
if cached is not None:
return cached
self._throttle()
headers = {"User-Agent": "MediaSorter/0.1.0 (https://github.com/example/media-sorter)"}
params = {"query": query_str, "fmt": "json", "limit": 1}
try:
resp = requests.get(f"{self.BASE_URL}/recording", params=params, headers=headers, timeout=5.0)
if resp.status_code == 200:
data = resp.json()
recordings = data.get("recordings", [])
if recordings:
rec = recordings[0]
rec_title = rec.get("title", title or "")
# Extract release info
releases = rec.get("releases", [])
rec_album = releases[0].get("title", album) if releases else album
release_date = releases[0].get("date", "") if releases else ""
res_year = int(release_date[:4]) if len(release_date) >= 4 and release_date[:4].isdigit() else None
res = ProviderResult(
canonical_title=rec_title,
artist=artist,
album=rec_album,
year=res_year,
media_type="music",
confidence_boost=0.15,
raw_payload=rec,
)
self.cache.set(cache_key, res)
return res
except Exception as e:
logger.warning("MusicBrainz network error", error=str(e))
return None
class MockMetadataProvider(MetadataProvider):
"""Deterministic mock provider for offline testing and fixture validation."""
def __init__(self, mock_data: Optional[Dict[str, ProviderResult]] = None):
self.mock_data = mock_data or {}
def search_movie(self, title: str, year: Optional[int] = None) -> Optional[ProviderResult]:
key = f"movie:{title.lower()}"
return self.mock_data.get(key)
def search_tv(self, title: str, year: Optional[int] = None, season: Optional[int] = None, episode: Optional[int] = None) -> Optional[ProviderResult]:
key = f"tv:{title.lower()}"
return self.mock_data.get(key)
def search_music(self, artist: str, album: Optional[str] = None, title: Optional[str] = None) -> Optional[ProviderResult]:
key = f"music:{artist.lower()}"
return self.mock_data.get(key)
+141
View File
@@ -0,0 +1,141 @@
"""Quarantine and manual review management for Media Sorter.
Provides querying, manual override, reprocessing, and resolution tracking for files
that could not be safely or confidently organized automatically.
"""
from __future__ import annotations
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional
import structlog
from sqlalchemy.orm import Session
from .models import QuarantineRecord, QuarantineStatus
logger = structlog.get_logger(__name__)
class QuarantineManager:
"""Manages files held in quarantine or review status."""
def __init__(self, session: Session):
self.session = session
def list_pending(self) -> List[QuarantineRecord]:
"""Return all quarantine items waiting for human inspection."""
return (
self.session.query(QuarantineRecord)
.filter_by(status=QuarantineStatus.PENDING.value)
.order_by(QuarantineRecord.created_at.desc())
.all()
)
def get_by_id(self, item_id: int) -> Optional[QuarantineRecord]:
"""Fetch quarantine record by ID."""
return self.session.query(QuarantineRecord).filter_by(id=item_id).first()
def resolve_item(
self,
item_id: int,
resolved_category: str,
target_path: Optional[Path | str] = None,
) -> bool:
"""Resolve a quarantined item by specifying human-approved category and target path."""
rec = self.get_by_id(item_id)
if not rec:
return False
rec.status = QuarantineStatus.RESOLVED.value
rec.suggested_category = resolved_category
rec.resolved_at = datetime.now(timezone.utc)
if target_path:
rec.resolved_path = str(target_path)
self.session.commit()
logger.info("Quarantine item resolved", item_id=item_id, category=resolved_category)
return True
def ignore_item(self, item_id: int) -> bool:
"""Mark quarantine item as ignored."""
rec = self.get_by_id(item_id)
if not rec:
return False
rec.status = QuarantineStatus.IGNORED.value
rec.resolved_at = datetime.now(timezone.utc)
self.session.commit()
return True
def undo_item(self, item_id: int) -> bool:
"""Undo the resolution or quarantine status of an item.
If resolved and file was moved, moves the file back to its original src location
and resets status to PENDING. If already pending, unflags/removes from quarantine.
"""
import os
import shutil
rec = self.get_by_id(item_id)
if not rec:
return False
if rec.status == QuarantineStatus.RESOLVED.value and rec.resolved_path:
dst_path = Path(rec.resolved_path)
src_path = Path(rec.src)
if dst_path.exists():
src_path.parent.mkdir(parents=True, exist_ok=True)
try:
shutil.move(dst_path, src_path)
logger.info("Restored resolved quarantine file back to src", src=str(src_path), dst=str(dst_path))
except Exception as e:
logger.error("Failed restoring quarantine file to src", src=str(src_path), dst=str(dst_path), error=str(e))
rec.status = QuarantineStatus.PENDING.value
rec.resolved_path = None
rec.resolved_at = None
self.session.commit()
return True
elif rec.status == QuarantineStatus.PENDING.value:
# Unflag pending quarantine item
self.session.delete(rec)
self.session.commit()
return True
return False
def list_resolved(self, limit: int = 50) -> List[QuarantineRecord]:
"""Return recently resolved quarantine items."""
return (
self.session.query(QuarantineRecord)
.filter_by(status=QuarantineStatus.RESOLVED.value)
.order_by(QuarantineRecord.resolved_at.desc())
.limit(limit)
.all()
)
def get_statistics(self) -> Dict[str, int]:
"""Summarize quarantine records by status."""
total = self.session.query(QuarantineRecord).count()
pending = (
self.session.query(QuarantineRecord)
.filter_by(status=QuarantineStatus.PENDING.value)
.count()
)
resolved = (
self.session.query(QuarantineRecord)
.filter_by(status=QuarantineStatus.RESOLVED.value)
.count()
)
ignored = (
self.session.query(QuarantineRecord)
.filter_by(status=QuarantineStatus.IGNORED.value)
.count()
)
return {
"total": total,
"pending": pending,
"resolved": resolved,
"ignored": ignored,
}
+258
View File
@@ -0,0 +1,258 @@
"""Filesystem discovery and scanning engine for Media Sorter.
Efficiently traverses directories, enforces minimum file age checks (to prevent
processing files currently being written/downloaded), tests file locks, and detects
companion/sidecar files.
"""
from __future__ import annotations
import fnmatch
import os
import sys
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import Dict, Generator, List, Optional, Set, Tuple
import structlog
logger = structlog.get_logger(__name__)
# Known sidecar and companion extensions
SUBTITLE_EXTS = {".srt", ".ass", ".ssa", ".vtt", ".sub", ".idx"}
ARTWORK_NAMES = {"poster", "cover", "folder", "fanart", "banner", "clearart", "disc", "logo"}
ARTWORK_EXTS = {".jpg", ".jpeg", ".png", ".webp", ".tbn"}
METADATA_EXTS = {".nfo", ".xml", ".json"}
EXTRA_TAGS = {"-trailer", "-sample", "-featurette", "-behindthescenes", "-deleted", "-short"}
@dataclass
class ScannedFile:
path: Path
size: int
mtime: float
is_sidecar: bool = False
sidecar_type: Optional[str] = None # "subtitle", "artwork", "metadata", "extra"
primary_media_path: Optional[Path] = None
tags: Dict[str, str] = field(default_factory=dict)
def is_file_locked(path: Path) -> bool:
"""Test whether a file is currently open/locked for writing by another process."""
if not path.is_file():
return False
try:
# On Windows, try opening with exclusive read/write sharing if possible
if sys.platform == "win32":
import msvcrt
handle = open(path, "rb")
try:
msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1)
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
finally:
handle.close()
else:
import fcntl
with open(path, "rb") as f:
fcntl.flock(f.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
fcntl.flock(f.fileno(), fcntl.LOCK_UN)
return False
except (IOError, OSError, PermissionError):
return True
class Scanner:
def __init__(
self,
min_file_age_seconds: int = 300,
include_patterns: Optional[List[str]] = None,
exclude_patterns: Optional[List[str]] = None,
):
self.min_file_age_seconds = min_file_age_seconds
self.include_patterns = include_patterns or ["*"]
self.exclude_patterns = exclude_patterns or [
".*",
"*.part",
"*.crdownload",
"*.!qB",
"Thumbs.db",
"desktop.ini",
"@eaDir",
"$RECYCLE.BIN",
"*.txt",
]
def _matches_filter(self, filename: str) -> bool:
"""Check whether filename matches includes and does not match excludes."""
for pattern in self.exclude_patterns:
if fnmatch.fnmatch(filename, pattern):
return False
if fnmatch.fnmatch(filename.lower(), pattern.lower()):
return False
if not self.include_patterns or "*" in self.include_patterns:
return True
for pattern in self.include_patterns:
if fnmatch.fnmatch(filename, pattern) or fnmatch.fnmatch(filename.lower(), pattern.lower()):
return True
return False
def scan_directory(self, root_dir: Path | str) -> List[ScannedFile]:
"""Recursively scan a directory returning all qualifying files."""
root = Path(root_dir).resolve()
if not root.exists():
logger.warning("Source directory does not exist", directory=str(root))
return []
now = time.time()
discovered: List[ScannedFile] = []
video_candidates: List[ScannedFile] = []
potential_sidecars: List[ScannedFile] = []
for entry_path in self._walk_safe(root):
try:
stat = entry_path.stat()
except (OSError, PermissionError) as e:
logger.warning("Skipping inaccessible file", path=str(entry_path), error=str(e))
continue
# Minimum file age check: ignore recently modified files (e.g. active downloads)
file_age = now - stat.st_mtime
if file_age < self.min_file_age_seconds:
logger.debug(
"Skipping file: modified too recently",
path=str(entry_path),
age_seconds=int(file_age),
min_age_seconds=self.min_file_age_seconds,
)
continue
# Check if locked
if is_file_locked(entry_path):
logger.debug("Skipping file: currently locked by another process", path=str(entry_path))
continue
ext = entry_path.suffix.lower()
stem = entry_path.stem.lower()
scanned = ScannedFile(
path=entry_path,
size=stat.st_size,
mtime=stat.st_mtime,
)
# Classify sidecar vs primary candidate
if ext in SUBTITLE_EXTS:
scanned.is_sidecar = True
scanned.sidecar_type = "subtitle"
potential_sidecars.append(scanned)
elif ext in ARTWORK_EXTS and any(stem == art or stem.startswith(f"{art}.") for art in ARTWORK_NAMES):
scanned.is_sidecar = True
scanned.sidecar_type = "artwork"
potential_sidecars.append(scanned)
elif ext in METADATA_EXTS:
scanned.is_sidecar = True
scanned.sidecar_type = "metadata"
potential_sidecars.append(scanned)
elif any(stem.endswith(tag) for tag in EXTRA_TAGS):
scanned.is_sidecar = True
scanned.sidecar_type = "extra"
potential_sidecars.append(scanned)
else:
discovered.append(scanned)
if ext in {".mp4", ".mkv", ".m4v", ".avi", ".mov", ".ts", ".webm"}:
video_candidates.append(scanned)
# Pair sidecars with primary files in the same directory
self._pair_sidecars(potential_sidecars, video_candidates, discovered)
return discovered
def _walk_safe(self, root: Path) -> Generator[Path, None, None]:
"""Safely traverse directories using os.scandir with cycle and permission handling."""
visited_inodes: Set[Tuple[int, int]] = set()
stack = [root]
while stack:
curr = stack.pop()
try:
with os.scandir(curr) as it:
for entry in it:
try:
# Avoid symlink loops
if entry.is_symlink():
continue
if entry.is_dir():
if not self._matches_filter(entry.name):
continue
stat = entry.stat()
dev_ino = (stat.st_dev, stat.st_ino)
if dev_ino in visited_inodes:
continue
visited_inodes.add(dev_ino)
stack.append(Path(entry.path))
elif entry.is_file():
if self._matches_filter(entry.name):
yield Path(entry.path)
except (OSError, PermissionError) as e:
logger.debug("Failed reading entry", path=entry.path, error=str(e))
except (OSError, PermissionError) as e:
logger.warning("Failed traversing directory", path=str(curr), error=str(e))
def _pair_sidecars(
self,
sidecars: List[ScannedFile],
primaries: List[ScannedFile],
all_discovered: List[ScannedFile],
) -> None:
"""Associate sidecar files (subtitles, artwork, nfo) with primary media files."""
# Index primaries by parent dir
primaries_by_dir: Dict[Path, List[ScannedFile]] = {}
for p in primaries:
primaries_by_dir.setdefault(p.path.parent.resolve(), []).append(p)
for s in sidecars:
parent = s.path.parent.resolve()
candidates = primaries_by_dir.get(parent, [])
# Support subdirectories like Subs/ or Subtitles/
if not candidates and parent.name.lower() in ("subs", "subtitles", "sub"):
parent = parent.parent
candidates = primaries_by_dir.get(parent, [])
matched_primary = None
s_stem = s.path.stem.lower()
# Sort candidates longest stem first so "Movie.Part2" matches before "Movie"
sorted_candidates = sorted(candidates, key=lambda c: len(c.path.stem), reverse=True)
for c in sorted_candidates:
c_stem = c.path.stem.lower()
if s_stem == c_stem:
matched_primary = c
break
# Check delimiter boundary: must be followed by '.', '-', '_', or ' '
if s_stem.startswith(c_stem) and len(s_stem) > len(c_stem):
next_char = s_stem[len(c_stem)]
if next_char in (".", "-", "_", " "):
matched_primary = c
break
# Fallback for single-video directories with generic sidecars (movie.nfo, poster.jpg, en.srt)
if not matched_primary and len(candidates) == 1:
if (
s.sidecar_type in ("metadata", "artwork")
or s.path.suffix.lower() in METADATA_EXTS
or s.path.suffix.lower() in SUBTITLE_EXTS
or s.path.suffix.lower() in ARTWORK_EXTS
):
matched_primary = candidates[0]
if matched_primary:
s.primary_media_path = matched_primary.path
all_discovered.append(s)
File diff suppressed because it is too large. Load diff
+323
View File
@@ -0,0 +1,323 @@
"""Main orchestration pipeline for Media Sorter.
Coordinates filesystem scanning, caching, parallel metadata probing, classification,
naming, dry-run previews, atomic execution, and audit logging.
"""
from __future__ import annotations
import concurrent.futures
import time
from pathlib import Path
from typing import Callable, Dict, List, Optional, Tuple
import structlog
from sqlalchemy.engine import Engine
from sqlalchemy.orm import Session
from .analyzer import MediaAnalyzer, MediaMetadata
from .classifier import ClassificationResult, MediaClassifier
from .config import ActionType, Settings
from .db import get_db_session
from .executor import BatchExecutionReport, MediaExecutor, PlannedOperation, acquire_process_lock
from .models import FileRecord, OperationStatus
from .namer import MediaNamer
from .notifications import send_batch_notification
from .providers import MetadataProvider, TMDBProvider
from .scanner import ScannedFile, Scanner
from .tokenizer import FilenameTokenizer, TokenizedFilename
logger = structlog.get_logger(__name__)
class MediaSorterApp:
"""High-performance orchestrator for analyzing and organizing media collections."""
def __init__(self, settings: Settings, engine: Engine):
self.settings = settings
self.engine = engine
# Component instances
self.scanner = Scanner(
min_file_age_seconds=settings.general.min_file_age_seconds,
include_patterns=settings.filters.include_patterns,
exclude_patterns=settings.filters.exclude_patterns,
)
self.tokenizer = FilenameTokenizer()
self.analyzer = MediaAnalyzer()
# Metadata provider
provider: Optional[MetadataProvider] = None
if settings.providers.enable_online_metadata:
provider = TMDBProvider(
api_key=settings.providers.tmdb_api_key,
rate_limit_per_second=settings.providers.rate_limit_per_second,
)
self.classifier = MediaClassifier(
confidence_threshold=settings.general.confidence_threshold,
provider=provider,
)
self.namer = MediaNamer(settings)
def scan_and_analyze(
self,
progress_callback: Optional[Callable[[int, int, str], None]] = None,
filter_paths: Optional[List[Path]] = None,
) -> List[Tuple[ScannedFile, ClassificationResult]]:
"""Discover files across all configured source directories and analyze in parallel."""
source_paths = self.settings.get_source_paths()
all_scanned: List[ScannedFile] = []
logger.info("Starting library discovery", sources=[str(p) for p in source_paths])
for src in source_paths:
if src.exists():
discovered = self.scanner.scan_directory(src)
all_scanned.extend(discovered)
else:
logger.warning("Configured source path does not exist", path=str(src))
logger.info("Filesystem discovery complete", total_discovered=len(all_scanned))
if not all_scanned:
return []
# Check DB cache to skip unchanged already organized files
qualifying_files: List[ScannedFile] = []
with get_db_session(self.engine) as session:
for s in all_scanned:
rec = (
session.query(FileRecord)
.filter_by(path=str(s.path), size=s.size, mtime=s.mtime, status="organized")
.first()
)
if rec:
logger.debug("Skipping unchanged already organized file", path=str(s.path))
continue
qualifying_files.append(s)
if filter_paths:
filter_resolved = {p.resolve() for p in filter_paths}
qualifying_files = [s for s in qualifying_files if s.path.resolve() in filter_resolved]
total_files = len(qualifying_files)
logger.info("Files requiring processing", count=total_files)
# Check known shows from library
known_shows: List[Dict[str, Any]] = []
try:
with get_db_session(self.engine) as session:
from .library import get_known_shows
known_shows = get_known_shows(session)
except Exception:
pass
# Parallel analysis and classification
results: List[Tuple[ScannedFile, ClassificationResult]] = []
worker_count = self.settings.general.worker_count
def process_one(scanned: ScannedFile) -> Tuple[ScannedFile, ClassificationResult]:
tokens = self.tokenizer.tokenize(scanned.path)
meta = self.analyzer.analyze(scanned.path)
classification = self.classifier.classify(scanned, tokens, meta)
# Match against known library shows if category is unsure or confidence is low
if known_shows and (
classification.category not in ("tv", "movie")
or classification.needs_quarantine
or classification.confidence < 0.8
):
from .library import match_known_show
matched = match_known_show(scanned.path.name, known_shows)
if not matched and len(scanned.path.parts) > 1:
matched = match_known_show(scanned.path.parent.name, known_shows)
if matched:
classification.category = "tv"
if not classification.tokens.title or classification.tokens.title.lower() in ("episode", "unknown", ""):
classification.tokens.title = matched["title"]
classification.confidence = max(classification.confidence, 0.95)
classification.needs_quarantine = False
return scanned, classification
completed = 0
with concurrent.futures.ThreadPoolExecutor(max_workers=worker_count) as executor:
future_to_file = {executor.submit(process_one, sf): sf for sf in qualifying_files}
for future in concurrent.futures.as_completed(future_to_file):
try:
res = future.result()
results.append(res)
except Exception as e:
sf = future_to_file[future]
logger.error("Error analyzing file", path=str(sf.path), error=str(e))
completed += 1
if progress_callback:
progress_callback(completed, total_files, str(future_to_file[future].path.name))
return results
def build_plan(
self, analysis_results: List[Tuple[ScannedFile, ClassificationResult]]
) -> List[PlannedOperation]:
"""Convert classification results into concrete planned operations."""
primary_dest_map: Dict[Path, Path] = {}
plan: List[PlannedOperation] = []
# 1. First pass: non-sidecar primary media files
for scanned, cls_res in analysis_results:
if not scanned.is_sidecar:
dst = self.namer.generate_destination_path(cls_res)
primary_dest_map[scanned.path] = dst
plan.append(
PlannedOperation(
src=scanned.path,
dst=dst,
action=self.settings.general.action,
category=cls_res.category,
confidence=cls_res.confidence,
details=cls_res.signals,
quarantine=cls_res.needs_quarantine,
quarantine_reason=cls_res.quarantine_reason,
)
)
# 2. Second pass: sidecars (subtitles, artwork, metadata) matching primary destinations
for scanned, cls_res in analysis_results:
if scanned.is_sidecar:
primary_dst = (
primary_dest_map.get(scanned.primary_media_path)
if scanned.primary_media_path
else None
)
dst = self.namer.generate_destination_path(cls_res, primary_dst_path=primary_dst)
plan.append(
PlannedOperation(
src=scanned.path,
dst=dst,
action=self.settings.general.action,
category=cls_res.category,
confidence=cls_res.confidence,
details=cls_res.signals,
quarantine=cls_res.needs_quarantine,
quarantine_reason=cls_res.quarantine_reason,
)
)
return plan
def run(
self,
dry_run: Optional[bool] = None,
progress_callback: Optional[Callable[[int, int, str], None]] = None,
filter_paths: Optional[List[Path]] = None,
show_name_override: Optional[str] = None,
) -> BatchExecutionReport:
"""Run full media sorter pipeline: scan, analyze, plan, and execute."""
start_time = time.time()
is_dry_run = self.settings.general.dry_run if dry_run is None else dry_run
logger.info("Executing media-sorter run", dry_run=is_dry_run)
# Discover and analyze
analysis_results = self.scan_and_analyze(
progress_callback=progress_callback, filter_paths=filter_paths
)
if show_name_override:
for scanned, cls_res in analysis_results:
cls_res.category = "tv"
cls_res.tokens.title = show_name_override
cls_res.tokens.is_episodic = True
cls_res.needs_quarantine = False
cls_res.confidence = max(cls_res.confidence, 0.95)
# Build plan
planned_ops = self.build_plan(analysis_results)
# Execute under inter-process lock to coordinate workers
lock_path = self.settings.get_database_path().with_suffix(".lock")
with acquire_process_lock(lock_path):
with get_db_session(self.engine) as session:
executor = MediaExecutor(self.settings, session)
# Check for crash recovery from prior runs
executor.recover_interrupted_batches()
report = executor.execute_batch(
planned_ops, dry_run=is_dry_run, progress_callback=progress_callback
)
elapsed = round(time.time() - start_time, 2)
logger.info(
"Media sorter run finished",
elapsed_seconds=elapsed,
total=report.total_files,
moved=report.moved_files,
quarantined=report.quarantined_files,
skipped=report.skipped_files,
failed=report.failed_files,
)
# Update library catalog with moved items
if not is_dry_run and report.moved_files > 0:
try:
from .library import record_detected_item
shows_dir = self.settings.get_destination_path("tv")
movies_dir = self.settings.get_destination_path("movie")
with get_db_session(self.engine) as session:
for op in report.operations:
if op.status == OperationStatus.COMMITTED.value:
cat = (op.category or "tv").lower()
dst_p = Path(op.dst)
if cat in ("tv", "anime"):
try:
rel_tv = dst_p.relative_to(shows_dir)
show_title = rel_tv.parts[0]
dest_folder = str(shows_dir / show_title)
record_detected_item(
session,
self.settings,
show_title,
"tv",
destination_folder=dest_folder,
delta_count=1,
)
except Exception:
pass
elif cat == "movie":
try:
rel_mv = dst_p.relative_to(movies_dir)
movie_title = rel_mv.parts[0]
dest_folder = str(movies_dir / movie_title)
record_detected_item(
session,
self.settings,
movie_title,
"movie",
destination_folder=dest_folder,
delta_count=1,
)
except Exception:
pass
except Exception as e:
logger.error("Error updating library from execution report", error=str(e))
# Send notification webhook if configured
if self.settings.notifications.enabled:
send_batch_notification(self.settings.notifications, report)
return report
def rollback(self, batch_id: Optional[str] = None) -> int:
"""Rollback a past batch of operations."""
with get_db_session(self.engine) as session:
executor = MediaExecutor(self.settings, session)
return executor.rollback_batch(batch_id)
def rollback_all(self) -> int:
"""Rollback all past completed batches."""
with get_db_session(self.engine) as session:
executor = MediaExecutor(self.settings, session)
return executor.rollback_all()
+271
View File
@@ -0,0 +1,271 @@
/* Base CSS for Media Sorter UI */
:root {
--bg: #ffffff; /* background */
--card-bg: #f9f9f9; /* cards */
--card-hover: #eaeaea;
--border: #dddddd;
--text: #222222;
--text-muted: #555555;
--accent: #0066ff; /* default accent */
--accent-hover: #0044cc;
--radius-sm: 4px;
--radius-md: 8px;
}
[data-theme="minimalistic"] {
/* Light, clean look */
--bg: #fafafa;
--card-bg: #ffffff;
--card-hover: #f0f0f0;
--border: #e0e0e0;
--text: #111111;
--text-muted: #777777;
--accent: #0077c2;
--accent-hover: #005599;
}
[data-theme="high-visibility"] {
/* Dark with bright accent for strong contrast */
--bg: #111111;
--card-bg: #1a1a1a;
--card-hover: #262626;
--border: #333333;
--text: #eeeeee;
--text-muted: #bbbbbb;
--accent: #ffdd00; /* bright yellow */
--accent-hover: #ffbb00;
}
[data-theme="pastel"] {
--bg: #fff8f0;
--card-bg: #ffffff;
--card-hover: #f0e6e0;
--border: #e6d8d1;
--text: #453636;
--text-muted: #7a5d5d;
--accent: #ff8c94; /* soft pink */
--accent-hover: #ff6b78;
}
[data-theme="grayscale"] {
--bg: #f5f5f5;
--card-bg: #ffffff;
--card-hover: #e0e0e0;
--border: #c0c0c0;
--text: #333333;
--text-muted: #777777;
--accent: #555555;
--accent-hover: #444444;
}
[data-theme="cyber"] {
/* Retain original cyber‑dark palette but with new ID */
--bg: #090d16;
--card-bg: #0f172a;
--card-hover: #1e293b;
--border: #334155;
--text: #f8fafc;
--text-muted: #94a3b8;
--accent: #38bdf8;
--accent-hover: #0284c7;
}
[data-theme="bios-amber"] {
--bg: #0c0800;
--card-bg: #181100;
--card-hover: #261b02;
--border: #4d3800;
--text: #ffb833;
--text-muted: #b37e1a;
--accent: #ff9900;
--accent-hover: #ffad33;
}
[data-theme="vapor-glitch"] {
--bg: #080312;
--card-bg: #130a24;
--card-hover: #1f113a;
--border: #38195a;
--text: #00f5d4;
--text-muted: #b388ff;
--accent: #f72585;
--accent-hover: #b5179e;
}
[data-theme="mossy-stone"] {
--bg: #111813;
--card-bg: #19241c;
--card-hover: #223227;
--border: #2d4234;
--text: #d2e0d5;
--text-muted: #859e8b;
--accent: #52b788;
--accent-hover: #40916c;
}
[data-theme="crimson-eclipse"] {
--bg: #0d0608;
--card-bg: #190c10;
--card-hover: #261217;
--border: #441822;
--text: #e8d0d5;
--text-muted: #a37581;
--accent: #ff4d6d;
--accent-hover: #c9184a;
}
[data-theme="blueprint-draft"] {
--bg: #0b1d3a;
--card-bg: #102a54;
--card-hover: #173b75;
--border: #1f4a91;
--text: #e2edfd;
--text-muted: #8db5e6;
--accent: #60a5fa;
--accent-hover: #3b82f6;
}
/* RGB mode styling */
.rgb-mode {
--rgb-duration: 5s;
}
.rgb-mode .card,
.rgb-mode .btn,
.rgb-mode .theme-select {
animation: rgbPulse var(--rgb-duration) infinite;
}
@keyframes rgbPulse {
0% { box-shadow: 0 0 8px rgba(255,0,0,0.5); }
33% { box-shadow: 0 0 8px rgba(0,255,0,0.5); }
66% { box-shadow: 0 0 8px rgba(0,0,255,0.5); }
100% { box-shadow: 0 0 8px rgba(255,0,0,0.5); }
}
[data-theme="neon-forest"] {
--bg: #001408;
--card-bg: #032410;
--card-hover: #07381b;
--border: #0d542a;
--text: #c8facc;
--text-muted: #5ea874;
--accent: #39ff14;
--accent-hover: #2ecc11;
}
[data-theme="retro-retro"] {
--bg: #120024;
--card-bg: #220038;
--card-hover: #330052;
--border: #6b0099;
--text: #fce7f3;
--text-muted: #d946ef;
--accent: #ff00ff;
--accent-hover: #d500d5;
}
[data-theme="golden-sand"] {
--bg: #1c150c;
--card-bg: #291e10;
--card-hover: #3b2c17;
--border: #594322;
--text: #fbf0dc;
--text-muted: #bda27e;
--accent: #ffb300;
--accent-hover: #e09d00;
}
[data-theme="deep-space"] {
--bg: #070913;
--card-bg: #0d1224;
--card-hover: #141c38;
--border: #232f57;
--text: #e2edfd;
--text-muted: #818cf8;
--accent: #6366f1;
--accent-hover: #4f46e5;
}
[data-theme="candy-cotton"] {
--bg: #1a1520;
--card-bg: #261f30;
--card-hover: #362c44;
--border: #524166;
--text: #fdf2f8;
--text-muted: #f472b6;
--accent: #ffb6c1;
--accent-hover: #f694a5;
}
/* General layout */
body {
background: var(--bg);
color: var(--text);
font-family: Arial, Helvetica, sans-serif;
margin: 0;
padding: 0;
}
header.app-header, footer.app-footer {
background: var(--card-bg);
border-bottom: 1px solid var(--border);
padding: 0.75rem 1rem;
display: flex;
justify-content: space-between;
align-items: center;
}
main.app-main {
display: flex;
flex-wrap: wrap;
gap: 1rem;
padding: 1rem;
}
.panel {
background: var(--card-bg);
border: 1px solid var(--border);
border-radius: var(--radius-md);
padding: 0.75rem;
flex: 1 1 300px;
max-width: 100%;
overflow: auto;
}
.btn {
background: var(--accent);
color: #fff;
border: none;
border-radius: var(--radius-sm);
padding: 0.4rem 0.8rem;
cursor: pointer;
}
.btn:hover {
background: var(--accent-hover);
}
.modal {
position: fixed;
inset: 0;
background: rgba(0,0,0,0.4);
display: flex;
align-items: center;
justify-content: center;
}
.modal.hidden { display: none; }
.modal-content {
background: var(--card-bg);
padding: 1.5rem;
border-radius: var(--radius-md);
min-width: 260px;
}
/* Compact mode tweaks */
[data-compact="true"] header.app-header, [data-compact="true"] footer.app-footer {
padding: 0.4rem 0.8rem;
font-size: 0.9rem;
}
[data-compact="true"] .panel {
padding: 0.5rem;
font-size: 0.85rem;
}
+36
View File
@@ -0,0 +1,36 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<title>Media Sorter Management</title>
<link rel="stylesheet" href="/static/css/base.css" />
<script src="/static/js/app.js" defer></script>
</head>
<body>
<!-- Settings Modal -->
<div id="settings-modal" class="modal hidden">
<div class="modal-content">
<h2>Settings</h2>
<label for="theme-select">Theme</label>
<select id="theme-select"></select>
<label for="compact-toggle"><input type="checkbox" id="compact-toggle" /> Compact Mode</label>
<button id="close-settings" class="btn">Close</button>
</div>
</div>
<!-- Main UI -->
<header class="app-header">
<h1>Media Sorter</h1>
<button id="open-settings" class="btn">Settings</button>
</header>
<main class="app-main">
<section id="folder-explorer" class="panel"></section>
<section id="library" class="panel"></section>
<section id="quarantine" class="panel"></section>
</main>
<footer class="app-footer">
<span id="status-bar">Ready</span>
</footer>
</body>
</html>
+653
View File
@@ -0,0 +1,653 @@
"""Filename and path tokenization engine for Media Sorter.
Robustly extracts semantic media tokens (title, year, season, episode, artist,
album, track, disc, quality, codec, release group, date stamps) from messy filenames,
scene releases, anime fansub conventions, and folder hierarchies.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from pathlib import Path
from typing import Dict, List, Optional, Tuple
# Regex patterns for Video & Episodic Media
ROMAN_NUMERALS: Dict[str, int] = {
"i": 1, "ii": 2, "iii": 3, "iv": 4, "v": 5,
"vi": 6, "vii": 7, "viii": 8, "ix": 9, "x": 10,
"xi": 11, "xii": 12, "xiii": 13, "xiv": 14, "xv": 15,
"xvi": 16, "xvii": 17, "xviii": 18, "xix": 19, "xx": 20,
}
RE_ROMAN_SEASON_EPISODE = re.compile(
r"""(?ix)
\bseason[\.\s_-]*(?P<season_roman>[ivx]+)[\.\s_-]*(?:episode|ep)[\.\s_-]*(?P<episode_roman>[ivx]+)\b
"""
)
RE_SEASON_EPISODE = re.compile(
r"""(?ix)
(?:
# Standard S01E02, S01E01-E02, S01E01E02, S01E01-02, S01E1171
(?<![0-9a-z])s(?P<season>\d{1,2})[\.\s_-]*(?:e|ep|ed|op)(?P<episode>\d{1,4})
(?:[\.\s_-]*(?:e|x|-|ep)(?P<episode_end>\d{1,4}))?(?![0-9])
|
# Scene 1x02, 2x01-02, 2x01-x02, 1x1171
(?<![0-9a-z])(?P<season_x>\d{1,2})x(?!(?:264|265|vid|hevc|avc))(?P<episode_x>\d{1,4})
(?:[\.\s_-]*(?:x|-)(?P<episode_x_end>\d{1,4}))?(?![0-9])
|
# Word season / episode: Season 1 Episode 2, Season 1 Episode 1171
\bseason[\.\s_-]*(?P<season_word>\d{1,2})[\.\s_-]*(?:episode|ep)[\.\s_-]*(?P<episode_word>\d{1,4})
(?:[\.\s_-]*(?:-|to)[\.\s_-]*(?:episode|ep)?[\\.\s_-]*(?P<episode_word_end>\d{1,4}))?\b
|
# Standalone episode: Episode 207, Ep 01, E233, E1171
(?<![0-9a-z])(?:episodes?|ep|e)[\.\s_-]*(?P<episode_standalone>\d{1,4})(?![0-9a-zA-Z])
)
"""
)
RE_SEASON_PACK = re.compile(
r"""(?ix)
(?<![0-9a-z])
(?:
s(?P<season_pack>\d{1,2})
|
season[\.\s_-]*(?P<season_pack_word>\d{1,2})
)
[\.\s_-]*(?:complete|full|season\.pack)\b
"""
)
# Anime fansub format: [ReleaseGroup] Show Title - 01 (or 01-02, or 01v2) - Optional Episode Title [1080p] [CRC32].mkv
RE_ANIME_RELEASE = re.compile(
r"""(?ix)
^\s*(?:\[(?P<group>[^\]]+)\][\s_]*)?
(?P<title>.+?)\s*(?:-\s*|_\-_\s*|_)\s*
(?P<episode>\d{1,4})(?:-(?P<episode_end>\d{1,4}))?(?:v\d+)?(?![0-9a-zA-Z])\s*
(?:\s*-\s*(?P<ep_title>[^\[\(]+?))?
(?:[\s_]*(?:\[?[0-9A-Fa-f]{8}\]?|\[(?P<tag>[^\]]+)\]|\((?P<tag_paren>[^\)]+)\))|[\s_]+[A-Za-z0-9_.-]+)*\s*\]?$
"""
)
# Space-separated anime format: [ReleaseGroup] Show Title 01 (Tags) [CRC32].mkv
RE_ANIME_RELEASE_SPACE = re.compile(
r"""(?ix)
^\s*\[(?P<group>[^\]]+)\]\s*
(?P<title>[^\[\(]+?)\s+
(?P<episode>\d{1,4})(?:-(?P<episode_end>\d{1,4}))?(?:v\d+)?(?![0-9a-zA-Z])\s*
(?:[\s_]*(?:\[?[0-9A-Fa-f]{8}\]?|\[(?P<tag>[^\]]+)\]|\((?P<tag_paren>[^\)]+)\))|[\s_]+[A-Za-z0-9_.-]+)*\s*\]?$
"""
)
# Underscore anime format: [ReleaseGroup]_Show_Title_01_[Tags].mp4
RE_ANIME_RELEASE_UNDERSCORE = re.compile(
r"""(?ix)
^\s*\[(?P<group>[^\]]+)\]_
(?P<title>[^\[\(]+?)_
(?P<episode>\d{1,4})(?:-(?P<episode_end>\d{1,4}))?(?:v\d+)?(?![0-9a-zA-Z])
(?:_*(?:\[?[0-9A-Fa-f]{8}\]?|\[(?P<tag>[^\]]+)\]|\((?P<tag_paren>[^\)]+)\))|_[A-Za-z0-9_.-]+)*\s*\]?$
"""
)
RE_ANIME_MOVIE = re.compile(
r"""(?ix)
^\s*\[(?P<group>[^\]]+)\]\s*
(?P<title>[^\[]+?)\s*
(?:\[(?P<tag>[^\]]+)\]|\((?P<tag_paren>[^\)]+)\))
"""
)
RE_YEAR_BOUND = re.compile(r"(?<![0-9a-zA-Z])(19\d{2}|20\d{2})(?![0-9a-zA-Z])")
RE_YEAR = re.compile(r"\b(19\d{2}|20\d{2})\b")
RE_EDITION = re.compile(
r"""(?ix)
\b(?P<edition>
directors?\.cut|director's\.cut|director's\scut
|
extended(?:\.cut|\.edition)?
|
remastered(?:\.edition)?|remaster
|
criterion(?:\.collection)?
|
final\.cut
|
theatrical(?:\.cut|\.version)?
|
unrated
|
special\.edition
|
imax(?:\.edition)?
|
ultimate\.edition
)\b
"""
)
EDITION_CANONICAL_MAP: Dict[str, str] = {
"extended": "Extended",
"extended.cut": "Extended",
"extended.edition": "Extended",
"directors.cut": "Director's Cut",
"director's.cut": "Director's Cut",
"director's cut": "Director's Cut",
"remastered": "Remastered",
"remastered.edition": "Remastered",
"remaster": "Remastered",
"criterion": "Criterion",
"criterion.collection": "Criterion",
"final.cut": "Final Cut",
"theatrical": "Theatrical",
"theatrical.cut": "Theatrical",
"theatrical.version": "Theatrical",
"unrated": "Unrated",
"special.edition": "Special Edition",
"imax": "IMAX",
"imax.edition": "IMAX",
"ultimate.edition": "Ultimate Edition",
}
RE_MOVIE_PART = re.compile(
r"""(?ix)
\b(?:cd|part|pt|disc)[\.\s_-]*(?P<part_num>\d{1,2})\b
"""
)
RE_DAILY_DATE = re.compile(
r"""(?ix)
(?<!\d)
(?P<year>19\d{2}|20\d{2})[-._]
(?P<month>0[1-9]|1[0-2])[-._]
(?P<day>0[1-9]|[12]\d|3[01])
(?!\d)
"""
)
# Technical specs
RE_RESOLUTION = re.compile(r"\b(2160p|4k|1080p|1080i|720p|576p|480p)\b", re.IGNORECASE)
RE_DIMENSIONS = re.compile(r"\b(?:\d{3,4})x(?P<height>2160|1080|720|576|480)\b", re.IGNORECASE)
RE_SOURCE = re.compile(r"\b(bluray|blu-ray|bdrip|web-dl|webrip|web|hdtv|dvdrip|dvd|remux)\b", re.IGNORECASE)
RE_VIDEO_CODEC = re.compile(r"\b(x265|x264|h\.?265|h\.?264|hevc|avc|av1|xvid|divx)\b", re.IGNORECASE)
RE_AUDIO_CODEC = re.compile(r"\b(truehd|atmos|dts-hd|dts|flac|aac|ac3|ddp?5\.1|mp3)\b", re.IGNORECASE)
RE_RELEASE_GROUP = re.compile(r"-([A-Za-z0-9_]+)(?:\[.*?\])?$", re.IGNORECASE)
RE_RELEASE_GROUP_UPGRADED = re.compile(
r"-(?:\[(?P<grp_bracket>[A-Za-z0-9_.-]+)\]|(?P<grp_plain>[A-Za-z0-9_]+))(?:\[.*?\])?$",
re.IGNORECASE,
)
RE_ILLEGAL_CHARS = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
RE_TECH_ALL = re.compile(
r"""(?ix)
\b(
2160p|4k|1080p|1080i|720p|576p|480p
|
\d{3,4}x(?:2160|1080|720|576|480)
|
bluray|blu-ray|bdrip|web-dl|webrip|web|hdtv|dvdrip|dvd|remux
|
x265|x264|h\.?265|h\.?264|hevc|avc|av1|xvid|divx
|
truehd|atmos|dts-hd|dts|flac|aac|ac3|ddp?5\.1|mp3
|
directors?\.cut|director's\.cut|director's\scut|extended|remastered|criterion|final\.cut
|
cd\d|part\d|pt\d
|
proper
)\b
"""
)
KNOWN_ANIME_GROUPS = {
"subsplease", "horriblesubs", "erai-raws", "taigasubs", "judas", "commie", "asenshi", "coalgirls"
}
KNOWN_ANIME_TITLES = {
"naruto", "bleach", "one piece", "frieren", "dungeon meshi", "attack on titan",
"jujutsu kaisen", "mushoku tensei", "fairy tail", "fate stay night", "sword art online"
}
WINDOWS_RESERVED = {
"CON", "PRN", "AUX", "NUL",
"COM1", "COM2", "COM3", "COM4", "COM5", "COM6", "COM7", "COM8", "COM9",
"LPT1", "LPT2", "LPT3", "LPT4", "LPT5", "LPT6", "LPT7", "LPT8", "LPT9",
}
# Music / Audio track patterns: 01 - Title, 1-01 Title, Artist - 01 - Title
RE_MUSIC_TRACK = re.compile(
r"""(?ix)
^(?:(?P<disc>\d{1,2})[-_.])?(?P<track>\d{1,3})[\.\s_-]+(?P<title>.+)$
"""
)
# Photo and Home Video date stamps: IMG_20240812_142010, VID_20240812_142010, 2024-08-12 14.20.10
RE_CAMERA_DATE = re.compile(
r"""(?ix)
(?:img|vid|dsc|pano|mov)?[-_]?(?P<year>19\d{2}|20\d{2})[-_]?(?P<month>\d{2})[-_]?(?P<day>\d{2})
(?:[-_](?P<hour>\d{2})[-_]?(?P<minute>\d{2})[-_]?(?P<second>\d{2}))?
"""
)
# Podcast dated format: Show Name - 2026-03-15 - Episode Title
RE_PODCAST_DATE = re.compile(
r"""(?ix)
^(?P<show>.+?)\s*-\s*(?P<year>20\d{2})-(?P<month>\d{2})-(?P<day>\d{2})\s*-\s*(?P<title>.+)$
"""
)
# CRC32 checksum tag e.g. [194B3FBA]
RE_CRC32_TAG = re.compile(r"\[([0-9A-Fa-f]{8})\]")
@dataclass
class TokenizedFilename:
raw_name: str
title: Optional[str] = None
year: Optional[int] = None
season: Optional[int] = None
episode: Optional[int] = None
multi_episodes: List[int] = field(default_factory=list)
episode_title: Optional[str] = None
artist: Optional[str] = None
album: Optional[str] = None
track: Optional[int] = None
disc: Optional[int] = None
group: Optional[str] = None
resolution: Optional[str] = None
source: Optional[str] = None
video_codec: Optional[str] = None
audio_codec: Optional[str] = None
crc32: Optional[str] = None
date_stamp: Optional[str] = None
air_date: Optional[str] = None
edition: Optional[str] = None
part: Optional[int] = None
part_label: Optional[str] = None
is_anime: bool = False
is_episodic: bool = False
is_music: bool = False
is_photo_or_home_video: bool = False
is_daily: bool = False
is_season_pack: bool = False
# Contract alias per PROJECT.md:74
TokenizedMedia = TokenizedFilename
class FilenameTokenizer:
"""Parses raw filenames and directory paths into semantic tokens."""
def tokenize(self, file_path: Path) -> TokenizedFilename:
stem = file_path.stem
raw_name = file_path.name
tokens = TokenizedFilename(raw_name=raw_name)
ext = file_path.suffix.lower()
# Check for CRC32 tag
crc_m = RE_CRC32_TAG.search(stem)
if crc_m:
tokens.crc32 = crc_m.group(1).upper()
# Check Windows reserved names
if stem.upper() in WINDOWS_RESERVED:
tokens.title = stem.upper()
tokens.is_photo_or_home_video = True
return tokens
# Pre-clean illegal characters
had_illegal = False
if RE_ILLEGAL_CHARS.search(stem):
had_illegal = True
stem = RE_ILLEGAL_CHARS.sub(" ", stem)
stem = re.sub(r"\s+", " ", stem).strip()
# 1. Technical specifications
res_m = RE_RESOLUTION.search(stem)
if res_m:
tokens.resolution = res_m.group(1).lower()
if tokens.resolution == "4k":
tokens.resolution = "2160p"
else:
dim_m = RE_DIMENSIONS.search(stem)
if dim_m:
tokens.resolution = f"{dim_m.group('height')}p"
src_m = RE_SOURCE.search(stem)
if src_m:
tokens.source = src_m.group(1).upper()
vc_m = RE_VIDEO_CODEC.search(stem)
if vc_m:
tokens.video_codec = vc_m.group(1).lower().replace(".", "")
ac_m = RE_AUDIO_CODEC.search(stem)
if ac_m:
tokens.audio_codec = ac_m.group(1).upper()
# Check edition
ed_m = RE_EDITION.search(stem)
if ed_m:
ed_raw = ed_m.group("edition").lower().replace(" ", ".")
tokens.edition = EDITION_CANONICAL_MAP.get(ed_raw, ed_m.group("edition"))
# Check part
pt_m = RE_MOVIE_PART.search(stem)
if pt_m:
tokens.part = int(pt_m.group("part_num"))
tokens.part_label = f"Pt.{tokens.part}"
# 2. Check for Podcast date format
pod_m = RE_PODCAST_DATE.match(stem)
if pod_m:
tokens.title = pod_m.group("title").strip()
tokens.artist = pod_m.group("show").strip()
tokens.year = int(pod_m.group("year"))
tokens.date_stamp = f"{pod_m.group('year')}-{pod_m.group('month')}-{pod_m.group('day')}"
tokens.air_date = tokens.date_stamp
return tokens
# 3. Check for Daily / Broadcast dated format (TV or Podcast)
daily_m = RE_DAILY_DATE.search(stem)
if daily_m:
y, m, d = daily_m.group("year"), daily_m.group("month"), daily_m.group("day")
date_str = f"{y}-{m}-{d}"
tokens.date_stamp = date_str
tokens.air_date = date_str
tokens.year = int(y)
prefix = stem[: daily_m.start()]
clean_pfx = self._clean_title(prefix)
tokens.title = clean_pfx
if ext in {".mp3", ".flac", ".ogg", ".m4a", ".aac"}:
tokens.artist = clean_pfx
else:
tokens.is_daily = True
tokens.is_episodic = True
tokens.season = int(y)
return tokens
# 4. Check for Camera / Date stamp (Photos & Home Videos)
cam_m = RE_CAMERA_DATE.search(stem)
if cam_m:
y, m, d = cam_m.group("year"), cam_m.group("month"), cam_m.group("day")
tokens.date_stamp = f"{y}-{m}-{d}"
tokens.year = int(y)
tokens.is_photo_or_home_video = True
return tokens
# 5. Check Roman Numeral TV pattern: Rome.Season.II.Episode.IV
roman_m = RE_ROMAN_SEASON_EPISODE.search(stem)
if roman_m:
tokens.is_episodic = True
s_rom = roman_m.group("season_roman").lower()
e_rom = roman_m.group("episode_roman").lower()
tokens.season = ROMAN_NUMERALS.get(s_rom, 1)
tokens.episode = ROMAN_NUMERALS.get(e_rom, 1)
prefix = stem[: roman_m.start()]
tokens.title = self._clean_title(prefix)
return tokens
# 6. Check TV Season Pack: Succession.S02.Complete
pack_m = RE_SEASON_PACK.search(stem)
if pack_m:
tokens.is_episodic = True
tokens.is_season_pack = True
s_val = pack_m.group("season_pack") or pack_m.group("season_pack_word")
tokens.season = int(s_val)
prefix = stem[: pack_m.start()]
tokens.title = self._clean_title(prefix)
return tokens
# 7. Check Standard TV episodic patterns (S01E02, 1x02, Season 1 Episode 2, Episode 207)
tv_m = RE_SEASON_EPISODE.search(stem)
if tv_m:
tokens.is_episodic = True
season_str = tv_m.group("season") or tv_m.group("season_x") or tv_m.group("season_word")
ep_str = tv_m.group("episode") or tv_m.group("episode_x") or tv_m.group("episode_word") or tv_m.group("episode_standalone")
if season_str:
tokens.season = int(season_str)
else:
tokens.season = self._extract_season_from_path(file_path) or 1
if ep_str:
tokens.episode = int(ep_str)
end_ep = tv_m.group("episode_end") or tv_m.group("episode_x_end") or tv_m.group("episode_word_end")
if end_ep:
tokens.multi_episodes = list(range(tokens.episode, int(end_ep) + 1))
# If filename had illegal characters and matched standalone episode (e.g. Show: "Special" <Episode> | 1?.mkv)
if had_illegal and tv_m.group("episode_standalone"):
tokens.title = self._clean_title(stem)
return tokens
# Extract title before season marker
prefix = stem[: tv_m.start()]
clean_pfx = self._clean_title(prefix)
if clean_pfx:
tokens.title = clean_pfx
else:
tokens.title = self._extract_title_from_context(file_path) or "Episode"
# Check for year in prefix using RE_YEAR_BOUND
if prefix:
yr_m = RE_YEAR_BOUND.search(prefix)
if yr_m:
tokens.year = int(yr_m.group(1))
tokens.title = self._clean_title(prefix[: yr_m.start()])
# Extract episode title after season marker
suffix = stem[tv_m.end() :]
ep_title = self._extract_episode_title(suffix)
if ep_title:
tokens.episode_title = ep_title
# Release group at end
grp_m = RE_RELEASE_GROUP_UPGRADED.search(stem)
if grp_m:
tokens.group = grp_m.group("grp_bracket") or grp_m.group("grp_plain")
# Check if title is a known anime title
if tokens.title and tokens.title.lower() in KNOWN_ANIME_TITLES:
tokens.is_anime = True
return tokens
# 8. Check Anime fansub format: [Group] Title - 01 - Episode Title [1080p]
anime_m = (
RE_ANIME_RELEASE.match(stem)
or RE_ANIME_RELEASE_SPACE.match(stem)
or RE_ANIME_RELEASE_UNDERSCORE.match(stem)
)
if anime_m and (anime_m.group("group") or ext in {".mkv", ".mp4", ".avi", ".mov", ".ts", ".webm", ".m4v", ".flv"}):
ep_val = int(anime_m.group("episode"))
grp_name = anime_m.group("group").strip() if anime_m.group("group") else None
raw_title = anime_m.group("title")
title_clean = self._clean_title(raw_title, preserve_paren=True)
# Distinguish movie year from anime episode
is_known_anime = title_clean.lower() in KNOWN_ANIME_TITLES or (grp_name and grp_name.lower() in KNOWN_ANIME_GROUPS)
has_explicit_season = self._extract_season_from_path(file_path) is not None
has_range = bool(anime_m.groupdict().get("episode_end"))
if 1900 <= ep_val <= 2099 and not has_range and not is_known_anime and not has_explicit_season:
tokens.year = ep_val
tokens.title = title_clean
tokens.group = grp_name
tokens.is_anime = False
tokens.is_episodic = False
return tokens
else:
tokens.is_anime = True
tokens.group = grp_name
tokens.title = title_clean
tokens.episode = ep_val
if anime_m.groupdict().get("ep_title"):
tokens.episode_title = self._clean_title(anime_m.group("ep_title"))
if anime_m.groupdict().get("episode_end"):
tokens.multi_episodes = list(range(ep_val, int(anime_m.group("episode_end")) + 1))
tokens.season = self._extract_season_from_path(file_path) or 1
tokens.is_episodic = True
return tokens
# 9. Check Anime movie format: [Judas] Fate Stay Night... [BD 1080p]
anime_mov_m = RE_ANIME_MOVIE.match(stem)
if anime_mov_m:
grp = anime_mov_m.group("group").strip()
if grp.lower() in KNOWN_ANIME_GROUPS:
tokens.group = grp
tokens.is_anime = True
tokens.title = self._clean_title(anime_mov_m.group("title"), preserve_dots=True)
return tokens
# 10. Check for Music track pattern
mus_m = RE_MUSIC_TRACK.match(stem)
if mus_m:
tokens.is_music = True
tokens.track = int(mus_m.group("track"))
if mus_m.group("disc"):
tokens.disc = int(mus_m.group("disc"))
tokens.title = self._clean_title(mus_m.group("title"))
parent = file_path.parent
if parent and parent.name:
parts = parent.name.split(" - ")
if len(parts) >= 2:
tokens.artist = parts[0].strip()
tokens.album = parts[1].strip()
return tokens
# 11. Movie pattern: Title (Year) or Title.Year.Quality
# Parenthesized year first
paren_yr = re.search(r"\((19\d{2}|20\d{2})\)", stem)
if paren_yr:
tokens.year = int(paren_yr.group(1))
prefix = stem[: paren_yr.start()]
tokens.title = self._clean_title(prefix)
grp_m = RE_RELEASE_GROUP_UPGRADED.search(stem)
if grp_m:
tokens.group = grp_m.group("grp_bracket") or grp_m.group("grp_plain")
return tokens
# Delimiter-based right-to-left year detection
tech_start = len(stem)
for m in RE_TECH_ALL.finditer(stem):
if m.start() < tech_start:
tech_start = m.start()
year_matches = list(RE_YEAR_BOUND.finditer(stem))
if year_matches:
valid_matches = [m for m in year_matches if m.start() <= tech_start]
if not valid_matches:
valid_matches = year_matches
best_match = valid_matches[-1]
tokens.year = int(best_match.group(1))
prefix = stem[: best_match.start()]
tokens.title = self._clean_title(prefix)
grp_m = RE_RELEASE_GROUP_UPGRADED.search(stem)
if grp_m:
tokens.group = grp_m.group("grp_bracket") or grp_m.group("grp_plain")
return tokens
# Fallback: strip tech specs and clean whole stem as title
prefix = stem[:tech_start].strip(" .-_")
tokens.title = self._clean_title(prefix if prefix else stem)
grp_m = RE_RELEASE_GROUP_UPGRADED.search(stem)
if grp_m:
tokens.group = grp_m.group("grp_bracket") or grp_m.group("grp_plain")
return tokens
def _clean_title(self, raw: str, preserve_paren: bool = False, preserve_dots: bool = False) -> str:
"""Replace dots, underscores, and scene separators with clean spaces."""
# Strip leading bracket tags like [YTS.MX] or [SubsPlease] if present
raw = re.sub(r"^\s*\[[^\]]+\]\s*", "", raw)
# Protect decimal numbers like 2.5, 3.5, 4.5, 1.5, 0.5, 1.11, etc.
raw = re.sub(r"(?<=\d)\.(?=\d)", "PROTECTEDDECIMALDOT", raw)
if preserve_dots:
cleaned = re.sub(r"_+", " ", raw).strip()
else:
# Replace dots with space, except if dot is followed by space in Roman numeral (e.g. "I. ")
cleaned = re.sub(r"(?<=\b[IVXLCDM])\.\s+", "._KEEP_DOT_SPACE_", raw)
cleaned = re.sub(r"[\._]+", " ", cleaned)
cleaned = cleaned.replace("._KEEP_DOT_SPACE_", ". ")
cleaned = cleaned.replace("PROTECTEDDECIMALDOT", ".")
cleaned = re.sub(r"\s+", " ", cleaned).strip()
if preserve_paren and re.search(r"\([12]\d{3}\)$", cleaned):
# Do not strip trailing parenthesis if it's (Year)
pass
else:
cleaned = re.sub(r"[\-\(\)\[\]]+$", "", cleaned).strip()
return cleaned
def _extract_episode_title(self, suffix: str) -> Optional[str]:
"""Extract episode title from string after SxxExx marker, stripping tech tags."""
s = suffix.strip(" .-_")
if not s:
return None
# Split by known tech tags
for reg in (RE_RESOLUTION, RE_SOURCE, RE_VIDEO_CODEC, RE_AUDIO_CODEC, RE_EDITION):
m = reg.search(s)
if m:
s = s[: m.start()].strip(" .-_")
# Strip release group
grp = RE_RELEASE_GROUP_UPGRADED.search(s)
if grp:
s = s[: grp.start()].strip(" .-_")
cleaned = self._clean_title(s)
return cleaned if cleaned else None
def _extract_season_from_path(self, file_path: Path) -> Optional[int]:
"""Attempt to extract season number from parent directory names like 'Season 2', 'Season 08 - Water Seven', or 'S03'."""
try:
system_folders = {
"downloads", "jdownloads", "media", "completed", "incomplete", "torrent",
"torrents", "root", "home", "mnt", "md0", "storage", "tmp", "temp", "var", "etc", "usr"
}
for part in reversed(file_path.parts[:-1]):
if part.lower().strip() in system_folders:
break
m = re.search(r"(?i)\b(?:season|series|s)\s*(\d{1,2})\b", part)
if m:
return int(m.group(1))
except Exception:
pass
return None
def _extract_title_from_context(self, file_path: Path) -> Optional[str]:
"""Attempt to extract show title from immediate parent directory (e.g. Show/Episode 01.mkv or Show/Season 1/Ep01.mkv)."""
try:
parent = file_path.parent
if not parent or str(parent) in ("/", ".", ""):
return None
p_name = parent.name
if re.search(r"(?i)\b(?:season|series|s)\s*\d+\b", p_name):
# Ascend one level if inside a season folder
parent = parent.parent
p_name = parent.name if parent else ""
if not p_name:
return None
p_lower = p_name.lower().strip()
system_folders = {
"downloads", "jdownloads", "media", "completed", "incomplete", "torrent",
"torrents", "root", "home", "mnt", "md0", "storage", "tmp", "temp", "var", "etc", "usr"
}
if p_lower in system_folders:
return None
clean = self._clean_title(p_name)
if len(clean) >= 2:
return clean
except Exception:
pass
return None