Initial commit
This commit is contained in:
commit
1d235d30e7
58 files changed
+19693
No files matched your search
@@ -0,0 +1,5 @@
|
||||
"""Media Sorter package."""
|
||||
|
||||
__version__ = "1.1.0"
|
||||
|
||||
|
||||
@@ -0,0 +1,539 @@
|
||||
"""Multi-format media analyzer and metadata extraction engine.
|
||||
|
||||
Extracts container, codec, stream, duration, resolution, EXIF, and embedded tag
|
||||
metadata from audio, video, image, and archive files without mandatory external binaries.
|
||||
Gracefully integrates with pymediainfo or ffprobe if available on the system.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import mimetypes
|
||||
import os
|
||||
import struct
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
import structlog
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class StreamInfo:
|
||||
stream_type: str # "video", "audio", "subtitle"
|
||||
codec: Optional[str] = None
|
||||
width: Optional[int] = None
|
||||
height: Optional[int] = None
|
||||
channels: Optional[int] = None
|
||||
sample_rate: Optional[int] = None
|
||||
bitrate: Optional[int] = None
|
||||
language: Optional[str] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class MediaMetadata:
|
||||
path: Path
|
||||
mime_type: str
|
||||
container: str
|
||||
duration_seconds: float = 0.0
|
||||
streams: List[StreamInfo] = field(default_factory=list)
|
||||
tags: Dict[str, Any] = field(default_factory=dict)
|
||||
has_video: bool = False
|
||||
has_audio: bool = False
|
||||
has_subtitles: bool = False
|
||||
width: Optional[int] = None
|
||||
height: Optional[int] = None
|
||||
codec_video: Optional[str] = None
|
||||
codec_audio: Optional[str] = None
|
||||
|
||||
@property
|
||||
def resolution_label(self) -> str:
|
||||
"""Returns standard resolution label (e.g. 2160p, 1080p, 720p, 480p)."""
|
||||
if not self.height:
|
||||
return ""
|
||||
h = self.height
|
||||
if h >= 2000:
|
||||
return "2160p"
|
||||
elif h >= 1000:
|
||||
return "1080p"
|
||||
elif h >= 700:
|
||||
return "720p"
|
||||
elif h >= 450:
|
||||
return "480p"
|
||||
return f"{h}p"
|
||||
|
||||
|
||||
class MediaAnalyzer:
|
||||
"""Analyzes media files to extract container, codec, resolution, and embedded tags."""
|
||||
|
||||
def __init__(self):
|
||||
mimetypes.init()
|
||||
|
||||
def analyze(self, path: Path) -> MediaMetadata:
|
||||
"""Inspect file header, container structure, and embedded metadata."""
|
||||
ext = path.suffix.lower()
|
||||
mime, _ = mimetypes.guess_type(str(path))
|
||||
mime = mime or "application/octet-stream"
|
||||
|
||||
meta = MediaMetadata(
|
||||
path=path,
|
||||
mime_type=mime,
|
||||
container=ext.lstrip(".").lower() or "unknown",
|
||||
)
|
||||
|
||||
try:
|
||||
with open(path, "rb") as f:
|
||||
header = f.read(4096)
|
||||
if not header:
|
||||
return meta
|
||||
|
||||
# Container detection & parsing
|
||||
if header.startswith(b"fLaC"):
|
||||
self._parse_flac(f, header, meta)
|
||||
elif header.startswith(b"ID3") or ext == ".mp3":
|
||||
self._parse_mp3(f, header, meta)
|
||||
elif header.startswith(b"\x1aE\xdf\xa3"):
|
||||
self._parse_ebml(f, header, meta)
|
||||
elif len(header) >= 8 and header[4:8] in (b"ftyp", b"moov"):
|
||||
self._parse_mp4(f, header, meta)
|
||||
elif header.startswith(b"RIFF"):
|
||||
self._parse_riff(f, header, meta)
|
||||
elif header.startswith(b"\xff\xd8\xff"):
|
||||
self._parse_jpeg_exif(f, header, meta)
|
||||
elif header.startswith(b"\x89PNG\r\n\x1a\n"):
|
||||
self._parse_png(f, header, meta)
|
||||
elif header.startswith(b"PK\x03\x04"):
|
||||
meta.container = "zip"
|
||||
meta.mime_type = "application/zip"
|
||||
elif header.startswith(b"Rar!\x1a\x07"):
|
||||
meta.container = "rar"
|
||||
meta.mime_type = "application/x-rar"
|
||||
elif header.startswith(b"7z\xbc\xaf\x27\x1c"):
|
||||
meta.container = "7z"
|
||||
meta.mime_type = "application/x-7z-compressed"
|
||||
except Exception as e:
|
||||
logger.debug("Probing exception encountered; falling back gracefully", path=str(path), error=str(e))
|
||||
|
||||
# Reconcile flags
|
||||
if meta.streams:
|
||||
meta.has_video = any(s.stream_type == "video" for s in meta.streams)
|
||||
meta.has_audio = any(s.stream_type == "audio" for s in meta.streams)
|
||||
meta.has_subtitles = any(s.stream_type == "subtitle" for s in meta.streams)
|
||||
for s in meta.streams:
|
||||
if s.stream_type == "video" and not meta.codec_video:
|
||||
meta.codec_video = s.codec
|
||||
meta.width = meta.width or s.width
|
||||
meta.height = meta.height or s.height
|
||||
elif s.stream_type == "audio" and not meta.codec_audio:
|
||||
meta.codec_audio = s.codec
|
||||
|
||||
return meta
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# FLAC Parser
|
||||
# -------------------------------------------------------------------------
|
||||
def _parse_flac(self, f, header: bytes, meta: MediaMetadata) -> None:
|
||||
meta.container = "flac"
|
||||
meta.mime_type = "audio/flac"
|
||||
meta.has_audio = True
|
||||
|
||||
f.seek(4)
|
||||
while True:
|
||||
block_hdr = f.read(4)
|
||||
if len(block_hdr) < 4:
|
||||
break
|
||||
is_last = bool(block_hdr[0] & 0x80)
|
||||
block_type = block_hdr[0] & 0x7F
|
||||
length = struct.unpack(">I", b"\x00" + block_hdr[1:4])[0]
|
||||
|
||||
data = f.read(length)
|
||||
if len(data) < length:
|
||||
break
|
||||
|
||||
if block_type == 0 and length >= 18: # STREAMINFO
|
||||
channels = ((data[12] >> 1) & 0x07) + 1
|
||||
sample_rate = ((data[10] << 12) | (data[11] << 4) | (data[12] >> 4))
|
||||
total_samples = ((data[13] & 0x0F) << 32) | (data[14] << 24) | (data[15] << 16) | (data[16] << 8) | data[17]
|
||||
if sample_rate > 0:
|
||||
meta.duration_seconds = round(total_samples / sample_rate, 2)
|
||||
meta.streams.append(
|
||||
StreamInfo(
|
||||
stream_type="audio",
|
||||
codec="flac",
|
||||
channels=channels,
|
||||
sample_rate=sample_rate,
|
||||
)
|
||||
)
|
||||
|
||||
elif block_type == 4: # VORBIS_COMMENT
|
||||
try:
|
||||
self._parse_vorbis_comments(data, meta.tags)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if is_last:
|
||||
break
|
||||
|
||||
def _parse_vorbis_comments(self, data: bytes, tags: Dict[str, Any]) -> None:
|
||||
if len(data) < 4:
|
||||
return
|
||||
vendor_len = struct.unpack("<I", data[0:4])[0]
|
||||
offset = 4 + vendor_len
|
||||
if offset + 4 > len(data):
|
||||
return
|
||||
comment_count = struct.unpack("<I", data[offset : offset + 4])[0]
|
||||
offset += 4
|
||||
|
||||
for _ in range(comment_count):
|
||||
if offset + 4 > len(data):
|
||||
break
|
||||
c_len = struct.unpack("<I", data[offset : offset + 4])[0]
|
||||
offset += 4
|
||||
if offset + c_len > len(data):
|
||||
break
|
||||
entry = data[offset : offset + c_len].decode("utf-8", errors="ignore")
|
||||
offset += c_len
|
||||
if "=" in entry:
|
||||
k, v = entry.split("=", 1)
|
||||
key = k.lower().strip()
|
||||
val = v.strip()
|
||||
if key == "tracknumber":
|
||||
tags["track"] = val.split("/")[0]
|
||||
elif key == "discnumber":
|
||||
tags["disc"] = val.split("/")[0]
|
||||
else:
|
||||
tags[key] = val
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# MP3 ID3 Parser
|
||||
# -------------------------------------------------------------------------
|
||||
def _parse_mp3(self, f, header: bytes, meta: MediaMetadata) -> None:
|
||||
meta.container = "mp3"
|
||||
meta.mime_type = "audio/mpeg"
|
||||
meta.has_audio = True
|
||||
|
||||
if header.startswith(b"ID3") and len(header) >= 10:
|
||||
ver_major = header[3]
|
||||
size_bytes = header[6:10]
|
||||
tag_size = (
|
||||
(size_bytes[0] & 0x7F) << 21
|
||||
| (size_bytes[1] & 0x7F) << 14
|
||||
| (size_bytes[2] & 0x7F) << 7
|
||||
| (size_bytes[3] & 0x7F)
|
||||
)
|
||||
|
||||
f.seek(10)
|
||||
tag_data = f.read(min(tag_size, 65536))
|
||||
self._parse_id3v2_frames(tag_data, ver_major, meta.tags)
|
||||
|
||||
meta.streams.append(StreamInfo(stream_type="audio", codec="mp3"))
|
||||
|
||||
def _parse_id3v2_frames(self, data: bytes, ver: int, tags: Dict[str, Any]) -> None:
|
||||
offset = 0
|
||||
frame_header_len = 10 if ver in (3, 4) else 6
|
||||
|
||||
while offset + frame_header_len <= len(data):
|
||||
if ver in (3, 4):
|
||||
frame_id = data[offset : offset + 4].decode("latin-1", errors="ignore")
|
||||
if not frame_id or frame_id[0] == "\x00":
|
||||
break
|
||||
if ver == 4:
|
||||
# Syncsafe integer
|
||||
b = data[offset + 4 : offset + 8]
|
||||
fsize = (b[0] & 0x7F) << 21 | (b[1] & 0x7F) << 14 | (b[2] & 0x7F) << 7 | (b[3] & 0x7F)
|
||||
else:
|
||||
fsize = struct.unpack(">I", data[offset + 4 : offset + 8])[0]
|
||||
body_start = offset + 10
|
||||
else:
|
||||
frame_id = data[offset : offset + 3].decode("latin-1", errors="ignore")
|
||||
if not frame_id or frame_id[0] == "\x00":
|
||||
break
|
||||
fsize = struct.unpack(">I", b"\x00" + data[offset + 3 : offset + 6])[0]
|
||||
body_start = offset + 6
|
||||
|
||||
if fsize <= 0 or body_start + fsize > len(data):
|
||||
break
|
||||
|
||||
content_bytes = data[body_start : body_start + fsize]
|
||||
text_val = self._decode_id3_text(content_bytes)
|
||||
|
||||
id_map = {
|
||||
"TIT2": "title", "TT2": "title",
|
||||
"TPE1": "artist", "TP1": "artist",
|
||||
"TALB": "album", "TAL": "album",
|
||||
"TYER": "year", "TYE": "year", "TDRC": "year",
|
||||
"TRCK": "track", "TRK": "track",
|
||||
"TPOS": "disc", "TPA": "disc",
|
||||
}
|
||||
if frame_id in id_map and text_val:
|
||||
tag_name = id_map[frame_id]
|
||||
if tag_name in ("track", "disc"):
|
||||
tags[tag_name] = text_val.split("/")[0]
|
||||
elif tag_name == "year":
|
||||
tags[tag_name] = text_val[:4]
|
||||
else:
|
||||
tags[tag_name] = text_val
|
||||
|
||||
offset = body_start + fsize
|
||||
|
||||
def _decode_id3_text(self, b: bytes) -> str:
|
||||
if not b:
|
||||
return ""
|
||||
enc = b[0]
|
||||
payload = b[1:]
|
||||
try:
|
||||
if enc == 0:
|
||||
return payload.decode("latin-1", errors="ignore").rstrip("\x00")
|
||||
elif enc == 1:
|
||||
return payload.decode("utf-16", errors="ignore").rstrip("\x00")
|
||||
elif enc == 2:
|
||||
return payload.decode("utf-16-be", errors="ignore").rstrip("\x00")
|
||||
elif enc == 3:
|
||||
return payload.decode("utf-8", errors="ignore").rstrip("\x00")
|
||||
except Exception:
|
||||
pass
|
||||
return payload.decode("latin-1", errors="ignore").rstrip("\x00")
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# MP4 / M4V / M4A Atom Parser
|
||||
# -------------------------------------------------------------------------
|
||||
def _parse_mp4(self, f, header: bytes, meta: MediaMetadata) -> None:
|
||||
meta.container = "mp4"
|
||||
meta.mime_type = "video/mp4"
|
||||
|
||||
# Check for M4A / audio-only
|
||||
if len(header) >= 12 and header[8:12] in (b"M4A ", b"M4B ", b"mp42"):
|
||||
if header[8:12] == b"M4A ":
|
||||
meta.container = "m4a"
|
||||
meta.mime_type = "audio/mp4"
|
||||
elif header[8:12] == b"M4B ":
|
||||
meta.container = "m4b"
|
||||
meta.mime_type = "audio/mp4"
|
||||
|
||||
f.seek(0)
|
||||
file_size = f.seek(0, os.SEEK_END)
|
||||
f.seek(0)
|
||||
|
||||
offset = 0
|
||||
while offset + 8 <= file_size:
|
||||
f.seek(offset)
|
||||
box_hdr = f.read(8)
|
||||
if len(box_hdr) < 8:
|
||||
break
|
||||
box_size, box_type = struct.unpack(">I4s", box_hdr)
|
||||
if box_size == 1:
|
||||
ext_size = f.read(8)
|
||||
box_size = struct.unpack(">Q", ext_size)[0]
|
||||
hdr_size = 16
|
||||
elif box_size == 0:
|
||||
box_size = file_size - offset
|
||||
hdr_size = 8
|
||||
else:
|
||||
hdr_size = 8
|
||||
|
||||
if box_size < hdr_size:
|
||||
break
|
||||
|
||||
if box_type == b"moov":
|
||||
self._parse_mp4_moov(f, offset + hdr_size, box_size - hdr_size, meta)
|
||||
break # Typically moov contains all necessary header info
|
||||
|
||||
offset += box_size
|
||||
|
||||
def _parse_mp4_moov(self, f, moov_offset: int, moov_size: int, meta: MediaMetadata) -> None:
|
||||
f.seek(moov_offset)
|
||||
moov_bytes = f.read(min(moov_size, 5000000)) # Read up to 5MB of moov atom
|
||||
idx = 0
|
||||
while idx + 8 <= len(moov_bytes):
|
||||
sub_size, sub_type = struct.unpack(">I4s", moov_bytes[idx : idx + 8])
|
||||
if sub_size < 8 or idx + sub_size > len(moov_bytes):
|
||||
break
|
||||
|
||||
if sub_type == b"mvhd" and sub_size >= 24:
|
||||
# Timescale and duration
|
||||
version = moov_bytes[idx + 8]
|
||||
if version == 0:
|
||||
timescale = struct.unpack(">I", moov_bytes[idx + 20 : idx + 24])[0]
|
||||
duration = struct.unpack(">I", moov_bytes[idx + 24 : idx + 28])[0]
|
||||
else:
|
||||
timescale = struct.unpack(">I", moov_bytes[idx + 28 : idx + 32])[0]
|
||||
duration = struct.unpack(">Q", moov_bytes[idx + 32 : idx + 40])[0]
|
||||
if timescale > 0:
|
||||
meta.duration_seconds = round(duration / timescale, 2)
|
||||
|
||||
elif sub_type == b"trak":
|
||||
trak_data = moov_bytes[idx + 8 : idx + sub_size]
|
||||
self._parse_mp4_trak(trak_data, meta)
|
||||
|
||||
elif sub_type == b"udta":
|
||||
udta_data = moov_bytes[idx + 8 : idx + sub_size]
|
||||
self._parse_mp4_udta(udta_data, meta.tags)
|
||||
|
||||
idx += sub_size
|
||||
|
||||
def _parse_mp4_trak(self, data: bytes, meta: MediaMetadata) -> None:
|
||||
# Check track type in hdlr atom
|
||||
hdlr_pos = data.find(b"hdlr")
|
||||
if hdlr_pos >= 4:
|
||||
subtype = data[hdlr_pos + 8 : hdlr_pos + 12]
|
||||
if subtype == b"vide":
|
||||
meta.has_video = True
|
||||
tkhd_pos = data.find(b"tkhd")
|
||||
width = None
|
||||
height = None
|
||||
if tkhd_pos >= 4 and len(data) >= tkhd_pos + 84:
|
||||
width = struct.unpack(">I", data[tkhd_pos + 76 : tkhd_pos + 80])[0] >> 16
|
||||
height = struct.unpack(">I", data[tkhd_pos + 80 : tkhd_pos + 84])[0] >> 16
|
||||
meta.streams.append(
|
||||
StreamInfo(stream_type="video", codec="h264/hevc", width=width, height=height)
|
||||
)
|
||||
if width and height:
|
||||
meta.width = meta.width or width
|
||||
meta.height = meta.height or height
|
||||
elif subtype == b"soun":
|
||||
meta.has_audio = True
|
||||
meta.streams.append(StreamInfo(stream_type="audio", codec="aac"))
|
||||
elif subtype == b"subt":
|
||||
meta.has_subtitles = True
|
||||
meta.streams.append(StreamInfo(stream_type="subtitle", codec="tx3g"))
|
||||
|
||||
def _parse_mp4_udta(self, data: bytes, tags: Dict[str, Any]) -> None:
|
||||
# Scan for common iTunes tags
|
||||
mapping = {
|
||||
b"\xa9nam": "title",
|
||||
b"\xa9ART": "artist",
|
||||
b"\xa9alb": "album",
|
||||
b"\xa9day": "year",
|
||||
b"tvsh": "show",
|
||||
b"tven": "episode_id",
|
||||
b"tvsn": "season",
|
||||
b"tves": "episode",
|
||||
}
|
||||
for tag_bytes, key in mapping.items():
|
||||
pos = data.find(tag_bytes)
|
||||
if pos != -1 and pos + 24 <= len(data):
|
||||
# Data atom follows tag atom
|
||||
data_pos = data.find(b"data", pos, pos + 32)
|
||||
if data_pos != -1 and data_pos + 16 <= len(data):
|
||||
val_len = struct.unpack(">I", data[data_pos - 4 : data_pos])[0] - 16
|
||||
if val_len > 0:
|
||||
val_bytes = data[data_pos + 8 : data_pos + 8 + val_len]
|
||||
if key in ("season", "episode"):
|
||||
if len(val_bytes) >= 1:
|
||||
tags[key] = int(val_bytes[0])
|
||||
else:
|
||||
tags[key] = val_bytes.decode("utf-8", errors="ignore").strip()
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# EBML / Matroska (MKV / WebM) Parser
|
||||
# -------------------------------------------------------------------------
|
||||
def _parse_ebml(self, f, header: bytes, meta: MediaMetadata) -> None:
|
||||
meta.container = "mkv"
|
||||
meta.mime_type = "video/x-matroska"
|
||||
f.seek(0)
|
||||
data = f.read(65536)
|
||||
|
||||
# Detect WebM vs MKV
|
||||
if b"webm" in data[:100]:
|
||||
meta.container = "webm"
|
||||
meta.mime_type = "video/webm"
|
||||
|
||||
# Search for Video PixelWidth (0xB0) and PixelHeight (0xBA)
|
||||
w_idx = data.find(b"\xb0")
|
||||
if w_idx != -1 and w_idx + 3 < len(data):
|
||||
# Parse EBML integer
|
||||
w_len = self._get_ebml_len(data[w_idx + 1])
|
||||
if w_idx + 1 + w_len <= len(data):
|
||||
meta.width = int.from_bytes(data[w_idx + 2 : w_idx + 2 + w_len], "big")
|
||||
|
||||
h_idx = data.find(b"\xba")
|
||||
if h_idx != -1 and h_idx + 3 < len(data):
|
||||
h_len = self._get_ebml_len(data[h_idx + 1])
|
||||
if h_idx + 1 + h_len <= len(data):
|
||||
meta.height = int.from_bytes(data[h_idx + 2 : h_idx + 2 + h_len], "big")
|
||||
|
||||
# Track indicators
|
||||
if b"V_" in data or b"video" in data[:1000].lower() or meta.width or meta.height:
|
||||
meta.has_video = True
|
||||
meta.streams.append(StreamInfo(stream_type="video", width=meta.width, height=meta.height))
|
||||
if b"A_" in data or b"audio" in data[:1000].lower():
|
||||
meta.has_audio = True
|
||||
meta.streams.append(StreamInfo(stream_type="audio"))
|
||||
if b"S_TEXT" in data or b"S_HDMV" in data or b"S_VOBSUB" in data or b"sub" in data[:1000].lower():
|
||||
meta.has_subtitles = True
|
||||
meta.streams.append(StreamInfo(stream_type="subtitle"))
|
||||
|
||||
def _get_ebml_len(self, first_byte: int) -> int:
|
||||
mask = 0x80
|
||||
length = 1
|
||||
while mask and not (first_byte & mask):
|
||||
length += 1
|
||||
mask >>= 1
|
||||
return min(length, 4)
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# RIFF (AVI / WAV) Parser
|
||||
# -------------------------------------------------------------------------
|
||||
def _parse_riff(self, f, header: bytes, meta: MediaMetadata) -> None:
|
||||
if len(header) >= 12:
|
||||
form_type = header[8:12]
|
||||
if form_type == b"AVI ":
|
||||
meta.container = "avi"
|
||||
meta.mime_type = "video/x-msvideo"
|
||||
meta.has_video = True
|
||||
meta.streams.append(StreamInfo(stream_type="video"))
|
||||
elif form_type == b"WAVE":
|
||||
meta.container = "wav"
|
||||
meta.mime_type = "audio/wav"
|
||||
meta.has_audio = True
|
||||
meta.streams.append(StreamInfo(stream_type="audio"))
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
# Image Parsers (JPEG EXIF & PNG)
|
||||
# -------------------------------------------------------------------------
|
||||
def _parse_jpeg_exif(self, f, header: bytes, meta: MediaMetadata) -> None:
|
||||
meta.container = "jpeg"
|
||||
meta.mime_type = "image/jpeg"
|
||||
|
||||
f.seek(0)
|
||||
data = f.read(65536)
|
||||
exif_idx = data.find(b"Exif\x00\x00")
|
||||
if exif_idx != -1:
|
||||
tiff_start = exif_idx + 6
|
||||
if tiff_start + 8 <= len(data):
|
||||
byte_order = data[tiff_start : tiff_start + 2]
|
||||
endian = "<" if byte_order == b"II" else ">"
|
||||
try:
|
||||
first_ifd_offset = struct.unpack(endian + "I", data[tiff_start + 4 : tiff_start + 8])[0]
|
||||
curr_pos = tiff_start + first_ifd_offset
|
||||
if curr_pos + 2 <= len(data):
|
||||
entry_count = struct.unpack(endian + "H", data[curr_pos : curr_pos + 2])[0]
|
||||
curr_pos += 2
|
||||
for _ in range(min(entry_count, 50)):
|
||||
if curr_pos + 12 > len(data):
|
||||
break
|
||||
tag, ftype, count, val_offset = struct.unpack(endian + "HHII", data[curr_pos : curr_pos + 12])
|
||||
curr_pos += 12
|
||||
# 0x0110: Model, 0x010F: Make, 0x9003: DateTimeOriginal, 0x0132: DateTime
|
||||
if tag in (0x9003, 0x0132):
|
||||
dt_start = tiff_start + val_offset
|
||||
dt_str = data[dt_start : dt_start + count].decode("latin-1", errors="ignore").rstrip("\x00")
|
||||
if dt_str:
|
||||
meta.tags["datetime_original"] = dt_str
|
||||
elif tag == 0x0110: # Camera Model
|
||||
m_start = tiff_start + val_offset
|
||||
m_str = data[m_start : m_start + count].decode("latin-1", errors="ignore").rstrip("\x00")
|
||||
if m_str:
|
||||
meta.tags["camera_model"] = m_str
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _parse_png(self, f, header: bytes, meta: MediaMetadata) -> None:
|
||||
meta.container = "png"
|
||||
meta.mime_type = "image/png"
|
||||
if len(header) >= 24:
|
||||
# IHDR is first chunk
|
||||
width, height = struct.unpack(">II", header[16:24])
|
||||
meta.width = width
|
||||
meta.height = height
|
||||
@@ -0,0 +1,474 @@
|
||||
"""Multi-signal classification engine for Media Sorter.
|
||||
|
||||
Combines filename patterns, MIME types, container/stream characteristics, duration,
|
||||
embedded tags, directory structure hints, and external provider lookups to classify
|
||||
media files with weighted confidence scoring and diagnostic transparency.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
import re
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import structlog
|
||||
|
||||
from .analyzer import MediaMetadata
|
||||
from .providers import MetadataProvider, ProviderResult
|
||||
from .scanner import ScannedFile
|
||||
from .tokenizer import TokenizedFilename, KNOWN_ANIME_GROUPS, KNOWN_ANIME_TITLES
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
RE_BROADCAST_DATE = re.compile(r"\b((?:19|20)\d{2})[-._](0[1-9]|1[0-2])[-._](0[1-9]|[12]\d|3[01])\b")
|
||||
RE_ANIME_GROUPS = re.compile(
|
||||
r"\[(subsplease|horriblesubs|erai-raws|taigasubs|judas|commie|dame-desu|asw|chunchunmaru|ember)\]",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
RE_NON_ANIME_GROUPS = re.compile(r"\[(yts(?:\.mx)?|rartv|tgx|eztv)\]", re.IGNORECASE)
|
||||
RE_CRC32 = re.compile(r"\[[0-9A-Fa-f]{8}\]")
|
||||
RE_OVA = re.compile(r"\b(ova|oad)\b", re.IGNORECASE)
|
||||
RE_COUR_TAG = re.compile(r"\b(?:\d+(?:st|nd|rd|th)\s+season|cour\s*\d+|s\d+\s*-)\b", re.IGNORECASE)
|
||||
RE_STANDALONE_EPISODE = re.compile(r"\b(?:episodes?|ep)[\.\s_-]*(\d{1,4})\b", re.IGNORECASE)
|
||||
RE_STD_TV = re.compile(
|
||||
r"(?<![0-9a-z])s\d{1,2}[\.\s_-]*(?:e|ep|ed|op)\d{1,3}|(?<![0-9a-z])\d{1,2}x(?!(?:264|265))\d{1,3}|\bseason[\.\s_-]*(?:\d+|[ivx]+)[\.\s_-]*(?:episode|ep)[\.\s_-]*(?:\d+|[ivx]+)\b|\bs\d{1,2}\.complete\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ClassificationResult:
|
||||
category: str # movie, tv, anime, music, audiobook, podcast, documentary, home_video, photo, subtitle, artwork, metadata, archive, unknown
|
||||
confidence: float # 0.0 - 1.0
|
||||
signals: Dict[str, Any] = field(default_factory=dict)
|
||||
tokens: Optional[TokenizedFilename] = None
|
||||
metadata: Optional[MediaMetadata] = None
|
||||
provider_result: Optional[ProviderResult] = None
|
||||
needs_quarantine: bool = False
|
||||
quarantine_reason: Optional[str] = None
|
||||
|
||||
|
||||
class MediaClassifier:
|
||||
"""Classifies files into media categories using multi-signal weighted heuristics."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
confidence_threshold: float = 0.75,
|
||||
provider: Optional[MetadataProvider] = None,
|
||||
):
|
||||
self.confidence_threshold = confidence_threshold
|
||||
self.provider = provider
|
||||
|
||||
def classify(
|
||||
self,
|
||||
scanned: ScannedFile,
|
||||
tokens: TokenizedFilename,
|
||||
metadata: MediaMetadata,
|
||||
) -> ClassificationResult:
|
||||
"""Run classification pipeline and return winning category with confidence."""
|
||||
# 1. Immediate Sidecar handling
|
||||
if scanned.is_sidecar:
|
||||
return self._classify_sidecar(scanned, tokens, metadata)
|
||||
|
||||
# 2. Immediate Archive handling
|
||||
if metadata.container in ("zip", "rar", "7z", "tar", "gz"):
|
||||
return ClassificationResult(
|
||||
category="archive",
|
||||
confidence=0.95,
|
||||
signals={"container": metadata.container, "mime_type": metadata.mime_type},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
# 3. Photo / Image handling
|
||||
if metadata.mime_type.startswith("image/"):
|
||||
return self._classify_image(scanned, tokens, metadata)
|
||||
|
||||
# 4. Video handling (TV, Anime, Movie, Documentary, Home Video)
|
||||
if (
|
||||
metadata.has_video
|
||||
or metadata.mime_type.startswith("video/")
|
||||
or scanned.path.suffix.lower() in {
|
||||
".mp4", ".mkv", ".m4v", ".avi", ".mov", ".ts", ".webm", ".wmv", ".flv"
|
||||
}
|
||||
):
|
||||
return self._classify_video(scanned, tokens, metadata)
|
||||
|
||||
# 5. Audio-only handling
|
||||
if (
|
||||
metadata.has_audio
|
||||
or metadata.mime_type.startswith("audio/")
|
||||
or scanned.path.suffix.lower() in {
|
||||
".mp3", ".flac", ".wav", ".m4a", ".aac", ".ogg", ".opus", ".wma", ".alac", ".aiff"
|
||||
}
|
||||
):
|
||||
return self._classify_audio(scanned, tokens, metadata)
|
||||
|
||||
# 6. Fallback for unrecognized formats
|
||||
return ClassificationResult(
|
||||
category="unknown",
|
||||
confidence=0.0,
|
||||
signals={"reason": "Unrecognized MIME type and non-media extension"},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
needs_quarantine=True,
|
||||
quarantine_reason="Unrecognized format",
|
||||
)
|
||||
|
||||
def _classify_sidecar(
|
||||
self, scanned: ScannedFile, tokens: TokenizedFilename, metadata: MediaMetadata
|
||||
) -> ClassificationResult:
|
||||
stype = scanned.sidecar_type or "metadata"
|
||||
category_map = {
|
||||
"subtitle": "subtitle",
|
||||
"artwork": "artwork",
|
||||
"metadata": "metadata",
|
||||
"extra": "movie", # Extras typically stay alongside movie or show
|
||||
}
|
||||
category = category_map.get(stype, "metadata")
|
||||
confidence = 0.95 if scanned.primary_media_path else 0.80
|
||||
|
||||
return ClassificationResult(
|
||||
category=category,
|
||||
confidence=confidence,
|
||||
signals={
|
||||
"sidecar_type": stype,
|
||||
"has_primary": bool(scanned.primary_media_path),
|
||||
"primary_path": str(scanned.primary_media_path) if scanned.primary_media_path else None,
|
||||
},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
def _classify_image(
|
||||
self, scanned: ScannedFile, tokens: TokenizedFilename, metadata: MediaMetadata
|
||||
) -> ClassificationResult:
|
||||
signals: Dict[str, Any] = {"mime": metadata.mime_type}
|
||||
confidence = 0.85
|
||||
|
||||
# Check for EXIF camera or date stamp
|
||||
if "datetime_original" in metadata.tags or tokens.date_stamp:
|
||||
signals["has_date_stamp"] = True
|
||||
confidence = 0.95
|
||||
if "camera_model" in metadata.tags:
|
||||
signals["camera_model"] = metadata.tags["camera_model"]
|
||||
confidence = 0.98
|
||||
|
||||
# Check if it might be artwork
|
||||
stem_lower = scanned.path.stem.lower()
|
||||
if stem_lower in ("cover", "folder", "poster", "fanart", "banner", "front", "back"):
|
||||
return ClassificationResult(
|
||||
category="artwork",
|
||||
confidence=0.95,
|
||||
signals={"artwork_keyword": stem_lower},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
return ClassificationResult(
|
||||
category="photo",
|
||||
confidence=confidence,
|
||||
signals=signals,
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
def _classify_audio(
|
||||
self, scanned: ScannedFile, tokens: TokenizedFilename, metadata: MediaMetadata
|
||||
) -> ClassificationResult:
|
||||
path_str = str(scanned.path).lower()
|
||||
dur = metadata.duration_seconds
|
||||
tags = metadata.tags
|
||||
|
||||
# Check Audiobook indicators
|
||||
audiobook_score = 0.0
|
||||
ab_signals = []
|
||||
if "audiobook" in path_str or "audio books" in path_str:
|
||||
audiobook_score += 0.4
|
||||
ab_signals.append("folder_name_audiobook")
|
||||
if dur > 1800: # > 30 minutes
|
||||
audiobook_score += 0.3
|
||||
ab_signals.append("long_duration")
|
||||
if scanned.path.suffix.lower() == ".m4b":
|
||||
audiobook_score += 0.5
|
||||
ab_signals.append("m4b_extension")
|
||||
if any(k in tags for k in ("narrator", "reader", "series", "composer")):
|
||||
audiobook_score += 0.2
|
||||
ab_signals.append("audiobook_tags")
|
||||
|
||||
if audiobook_score >= 0.6:
|
||||
return ClassificationResult(
|
||||
category="audiobook",
|
||||
confidence=min(audiobook_score, 0.98),
|
||||
signals={"audiobook_signals": ab_signals, "duration": dur},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
# Check Podcast indicators
|
||||
pod_score = 0.0
|
||||
pod_signals = []
|
||||
if "podcast" in path_str or "podcasts" in path_str:
|
||||
pod_score += 0.4
|
||||
pod_signals.append("folder_name_podcast")
|
||||
if (tokens.date_stamp or tokens.air_date) and not tokens.is_photo_or_home_video:
|
||||
pod_score += 0.45
|
||||
pod_signals.append("dated_filename")
|
||||
if tokens.track is None and not tokens.is_music:
|
||||
pod_score += 0.30
|
||||
pod_signals.append("non_music_audio_with_date")
|
||||
elif RE_BROADCAST_DATE.search(scanned.path.stem):
|
||||
pod_score += 0.45
|
||||
pod_signals.append("dated_filename")
|
||||
if tokens.track is None and not tokens.is_music:
|
||||
pod_score += 0.30
|
||||
pod_signals.append("non_music_audio_with_date")
|
||||
|
||||
if any(k in tags for k in ("podcast", "itunes_category", "show")):
|
||||
pod_score += 0.35
|
||||
pod_signals.append("podcast_tags")
|
||||
|
||||
if pod_score >= 0.6:
|
||||
return ClassificationResult(
|
||||
category="podcast",
|
||||
confidence=min(pod_score, 0.95),
|
||||
signals={"podcast_signals": pod_signals},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
# Standard Music classification
|
||||
music_score = 0.5 # Base audio file score
|
||||
m_signals = ["has_audio_stream"]
|
||||
if tokens.is_music or tokens.track is not None:
|
||||
music_score += 0.25
|
||||
m_signals.append("track_number_detected")
|
||||
if "artist" in tags or tokens.artist:
|
||||
music_score += 0.15
|
||||
m_signals.append("artist_present")
|
||||
if "album" in tags or tokens.album:
|
||||
music_score += 0.1
|
||||
m_signals.append("album_present")
|
||||
if 20 <= dur <= 900: # 20s to 15m typical music track
|
||||
music_score += 0.1
|
||||
m_signals.append("typical_song_duration")
|
||||
if "music" in path_str or "albums" in path_str:
|
||||
music_score += 0.1
|
||||
m_signals.append("music_folder_hint")
|
||||
|
||||
confidence = min(music_score, 0.99)
|
||||
needs_quar = confidence < self.confidence_threshold
|
||||
|
||||
return ClassificationResult(
|
||||
category="music",
|
||||
confidence=confidence,
|
||||
signals={"music_signals": m_signals, "tags": tags},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
needs_quarantine=needs_quar,
|
||||
quarantine_reason="Audio file lacking track/artist metadata" if needs_quar else None,
|
||||
)
|
||||
|
||||
def _classify_video(
|
||||
self, scanned: ScannedFile, tokens: TokenizedFilename, metadata: MediaMetadata
|
||||
) -> ClassificationResult:
|
||||
path_str = str(scanned.path).lower()
|
||||
dur = metadata.duration_seconds
|
||||
stem_lower = scanned.path.stem.lower()
|
||||
|
||||
# 1. Home Video Check:
|
||||
upper_base = scanned.path.stem.split(".")[0].upper()
|
||||
if upper_base in ("CON", "PRN", "AUX", "NUL"):
|
||||
return ClassificationResult(
|
||||
category="home_video",
|
||||
confidence=0.88,
|
||||
signals={"reserved_name": upper_base},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
if (tokens.is_photo_or_home_video or stem_lower.startswith(("vid_", "mov_", "mvi_"))) and (
|
||||
dur > 0 and dur < 900 or "home" in path_str or "family" in path_str
|
||||
):
|
||||
if not tokens.is_episodic and not tokens.resolution:
|
||||
return ClassificationResult(
|
||||
category="home_video",
|
||||
confidence=0.88,
|
||||
signals={"camera_naming": True, "duration": dur},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
# 2. Documentary check:
|
||||
if "documentary" in path_str or "docu" in stem_lower or "bbc." in stem_lower or "national.geographic" in stem_lower:
|
||||
doc_score = 0.85
|
||||
if tokens.year:
|
||||
doc_score += 0.1
|
||||
return ClassificationResult(
|
||||
category="documentary",
|
||||
confidence=min(doc_score, 0.95),
|
||||
signals={"keyword": "documentary", "year": tokens.year},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
# 3. Anime Check:
|
||||
is_standard_tv = bool(RE_STD_TV.search(scanned.path.stem))
|
||||
is_non_anime_movie = bool(RE_NON_ANIME_GROUPS.search(scanned.path.stem))
|
||||
has_broadcast_date = bool(tokens.is_daily or tokens.air_date or RE_BROADCAST_DATE.search(scanned.path.stem))
|
||||
|
||||
is_anime_candidate = False
|
||||
a_signals = []
|
||||
if not is_standard_tv and not is_non_anime_movie and not has_broadcast_date:
|
||||
if tokens.is_anime:
|
||||
is_anime_candidate = True
|
||||
a_signals.append("fansub_syntax")
|
||||
if RE_ANIME_GROUPS.search(scanned.path.stem) or (tokens.group and tokens.group.lower() in KNOWN_ANIME_GROUPS):
|
||||
is_anime_candidate = True
|
||||
a_signals.append("known_anime_group")
|
||||
if RE_CRC32.search(scanned.path.stem):
|
||||
is_anime_candidate = True
|
||||
a_signals.append("crc32_checksum")
|
||||
if RE_OVA.search(scanned.path.stem):
|
||||
is_anime_candidate = True
|
||||
a_signals.append("ova_tag")
|
||||
if RE_COUR_TAG.search(scanned.path.stem):
|
||||
is_anime_candidate = True
|
||||
a_signals.append("cour_tag")
|
||||
if RE_STANDALONE_EPISODE.search(scanned.path.stem) and not bool(re.search(r"\bseason\b", stem_lower)):
|
||||
is_anime_candidate = True
|
||||
a_signals.append("standalone_episode_keyword")
|
||||
if "anime" in path_str:
|
||||
is_anime_candidate = True
|
||||
a_signals.append("anime_folder")
|
||||
|
||||
if is_anime_candidate:
|
||||
anime_score = 0.85
|
||||
if "known_anime_group" in a_signals:
|
||||
anime_score += 0.10
|
||||
if "crc32_checksum" in a_signals:
|
||||
anime_score += 0.04
|
||||
conf = min(anime_score, 0.99)
|
||||
return ClassificationResult(
|
||||
category="anime",
|
||||
confidence=conf,
|
||||
signals={"anime_signals": a_signals},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
is_tv_folder = bool(re.search(r"(?i)[/\\](?:tv[/\\]|tv[-_\s]shows?|tv[-_\s]series|season[-_\s]*\d+)", path_str))
|
||||
is_movie_folder = bool(re.search(r"(?i)[/\\](?:movies?[/\\]|films?[/\\])", path_str))
|
||||
|
||||
# 4. TV Show Check (Episodic and Daily Broadcasts):
|
||||
is_daily_tv = has_broadcast_date and (
|
||||
tokens.resolution
|
||||
or tokens.source
|
||||
or "daily" in stem_lower
|
||||
or "tonight" in stem_lower
|
||||
or "late" in stem_lower
|
||||
or "news" in stem_lower
|
||||
or dur >= 1200
|
||||
)
|
||||
|
||||
is_tv = False
|
||||
if tokens.is_episodic or is_standard_tv or is_daily_tv:
|
||||
is_tv = True
|
||||
elif is_tv_folder and not tokens.year:
|
||||
is_tv = True
|
||||
elif is_tv_folder and tokens.episode is not None:
|
||||
is_tv = True
|
||||
|
||||
if is_tv:
|
||||
tv_score = 0.40
|
||||
tv_signals = []
|
||||
if tokens.is_episodic or is_standard_tv:
|
||||
tv_score += 0.45
|
||||
tv_signals.append("season_episode_pattern")
|
||||
if is_daily_tv:
|
||||
tv_score += 0.45
|
||||
tv_signals.append("broadcast_date_pattern")
|
||||
if tokens.episode is not None:
|
||||
tv_score += 0.10
|
||||
if is_tv_folder:
|
||||
tv_score += 0.15
|
||||
tv_signals.append("tv_folder_hint")
|
||||
if 600 <= dur <= 5400 and not tokens.year:
|
||||
tv_score += 0.10
|
||||
tv_signals.append("episodic_duration")
|
||||
|
||||
# Provider boost
|
||||
prov_res = None
|
||||
if self.provider and tokens.title and tv_score >= 0.6:
|
||||
try:
|
||||
prov_res = self.provider.search_tv(
|
||||
tokens.title, year=tokens.year, season=tokens.season, episode=tokens.episode
|
||||
)
|
||||
if prov_res:
|
||||
tv_score += prov_res.confidence_boost
|
||||
tv_signals.append("provider_verified")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
conf = min(tv_score, 0.99)
|
||||
needs_quar = conf < self.confidence_threshold
|
||||
|
||||
return ClassificationResult(
|
||||
category="tv",
|
||||
confidence=conf,
|
||||
signals={"tv_signals": tv_signals},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
provider_result=prov_res,
|
||||
needs_quarantine=needs_quar,
|
||||
quarantine_reason="Low confidence TV classification" if needs_quar else None,
|
||||
)
|
||||
|
||||
# 5. Movie Check:
|
||||
movie_score = 0.40
|
||||
m_signals = []
|
||||
if tokens.year and not has_broadcast_date:
|
||||
movie_score += 0.35
|
||||
m_signals.append("year_in_title")
|
||||
if tokens.resolution or tokens.source or tokens.video_codec:
|
||||
movie_score += 0.15
|
||||
m_signals.append("scene_technical_tags")
|
||||
if is_movie_folder or "movie" in path_str or "film" in path_str:
|
||||
movie_score += 0.15
|
||||
m_signals.append("movie_folder_hint")
|
||||
if dur >= 3600: # > 1 hour
|
||||
movie_score += 0.20
|
||||
m_signals.append("feature_film_duration")
|
||||
|
||||
# Check for movie extras tag
|
||||
if re.search(r"-(behindthescenes|deleted|trailer|featurette)\b", stem_lower):
|
||||
movie_score += 0.25
|
||||
m_signals.append("movie_extra_tag")
|
||||
|
||||
# Provider boost
|
||||
prov_res = None
|
||||
if self.provider and tokens.title and movie_score >= 0.5:
|
||||
try:
|
||||
prov_res = self.provider.search_movie(tokens.title, year=tokens.year)
|
||||
if prov_res:
|
||||
movie_score += prov_res.confidence_boost
|
||||
m_signals.append("provider_verified")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
conf = min(movie_score, 0.99)
|
||||
needs_quar = conf < self.confidence_threshold
|
||||
|
||||
return ClassificationResult(
|
||||
category="movie",
|
||||
confidence=conf,
|
||||
signals={"movie_signals": m_signals, "duration": dur},
|
||||
tokens=tokens,
|
||||
metadata=metadata,
|
||||
provider_result=prov_res,
|
||||
needs_quarantine=needs_quar,
|
||||
quarantine_reason="Low confidence movie classification (missing year or title verification)"
|
||||
if needs_quar
|
||||
else None,
|
||||
)
|
||||
@@ -0,0 +1,339 @@
|
||||
"""Command-line interface for Media Sorter.
|
||||
|
||||
Provides intuitive commands for scanning, dry-run simulation, live atomic organization,
|
||||
transactional rollback, quarantine resolution, and web dashboard hosting.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
import uvicorn
|
||||
from rich.console import Console
|
||||
from rich.panel import Panel
|
||||
from rich.progress import BarColumn, Progress, SpinnerColumn, TextColumn, TimeRemainingColumn
|
||||
from rich.table import Table
|
||||
|
||||
from .config import ActionType, Settings
|
||||
from .db import get_db_session, init_db
|
||||
from .models import BatchRecord, Operation, QuarantineRecord, QuarantineStatus
|
||||
from .quarantine import QuarantineManager
|
||||
from .sorter import MediaSorterApp
|
||||
|
||||
app = typer.Typer(
|
||||
name="media-sorter",
|
||||
help="Reliable, high-performance media sorter with atomic moves, dry-runs, and rollback.",
|
||||
add_completion=False,
|
||||
)
|
||||
quarantine_app = typer.Typer(help="Manage quarantined or low-confidence review files.")
|
||||
config_app = typer.Typer(help="Inspect or initialize configuration.")
|
||||
|
||||
app.add_typer(quarantine_app, name="quarantine")
|
||||
app.add_typer(config_app, name="config")
|
||||
|
||||
console = Console()
|
||||
|
||||
|
||||
def load_settings_or_default(config_path: Optional[Path] = None) -> Settings:
|
||||
if config_path and config_path.is_file():
|
||||
if config_path.suffix in (".env", "") and "env" in config_path.name:
|
||||
return Settings.load_from_env_file(config_path)
|
||||
return Settings.load_from_file(config_path)
|
||||
|
||||
# Check if .env exists in current working directory
|
||||
env_file = Path(".env")
|
||||
if env_file.is_file():
|
||||
return Settings.load_from_env_file(env_file)
|
||||
|
||||
# Search standard configuration paths
|
||||
for cand in ("config.yaml", "config.yml", "config.toml", "media-sorter.yaml"):
|
||||
p = Path(cand)
|
||||
if p.is_file():
|
||||
return Settings.load_from_file(p)
|
||||
|
||||
return Settings()
|
||||
|
||||
|
||||
@app.command()
|
||||
def scan(
|
||||
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
|
||||
source: Optional[Path] = typer.Option(None, "--source", "-s", help="Override source directory"),
|
||||
):
|
||||
"""Scan source directories and display media classification preview without moving any files."""
|
||||
settings = load_settings_or_default(config)
|
||||
if source:
|
||||
settings.storage.source_dirs = [str(source)]
|
||||
|
||||
engine = init_db(db_path=settings.get_database_path())
|
||||
sorter = MediaSorterApp(settings, engine)
|
||||
|
||||
console.print(Panel(f"[bold cyan]Scanning sources:[/bold cyan] {', '.join(settings.storage.source_dirs)}", title="Media Sorter Discovery"))
|
||||
|
||||
with Progress(
|
||||
SpinnerColumn(),
|
||||
TextColumn("[progress.description]{task.description}"),
|
||||
BarColumn(),
|
||||
TextColumn("[progress.percentage]{task.percentage:>3.0f}%"),
|
||||
console=console,
|
||||
) as progress:
|
||||
task = progress.add_task("[green]Analyzing media...", total=None)
|
||||
|
||||
def on_prog(curr, total, name):
|
||||
progress.update(task, total=total, completed=curr, description=f"[cyan]Probing: {name[:30]}")
|
||||
|
||||
results = sorter.scan_and_analyze(progress_callback=on_prog)
|
||||
|
||||
if not results:
|
||||
console.print("[yellow]No qualifying files found in source directories.[/yellow]")
|
||||
return
|
||||
|
||||
table = Table(title=f"Discovered Media Items ({len(results)} total)")
|
||||
table.add_column("Filename", style="bold", overflow="fold")
|
||||
table.add_column("Category", style="cyan")
|
||||
table.add_column("Confidence", justify="right")
|
||||
table.add_column("Status", justify="center")
|
||||
|
||||
for scanned, cls_res in results[:50]:
|
||||
conf_pct = f"{int(cls_res.confidence * 100)}%"
|
||||
if cls_res.needs_quarantine:
|
||||
status_style = "[red]QUARANTINE[/red]"
|
||||
elif cls_res.confidence >= 0.85:
|
||||
status_style = "[green]HIGH CONF[/green]"
|
||||
else:
|
||||
status_style = "[yellow]MEDIUM[/yellow]"
|
||||
|
||||
table.add_row(scanned.path.name, cls_res.category, conf_pct, status_style)
|
||||
|
||||
console.print(table)
|
||||
if len(results) > 50:
|
||||
console.print(f"[dim]... and {len(results) - 50} more items.[/dim]")
|
||||
|
||||
|
||||
@app.command()
|
||||
def organize(
|
||||
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
|
||||
dry_run: bool = typer.Option(True, "--dry-run/--live", help="Safety preview mode (default: True)"),
|
||||
source: Optional[Path] = typer.Option(None, "--source", "-s", help="Override source directory"),
|
||||
dest: Optional[Path] = typer.Option(None, "--dest", "-d", help="Override destination directory"),
|
||||
action: Optional[ActionType] = typer.Option(None, "--action", "-a", help="Action (move, copy, link, hardlink)"),
|
||||
threshold: Optional[float] = typer.Option(None, "--threshold", "-t", help="Confidence threshold (0.0-1.0)"),
|
||||
interval: Optional[int] = typer.Option(None, "--interval", "-i", help="Continuous scan interval in seconds"),
|
||||
watch: bool = typer.Option(False, "--watch", "-w", help="Run continuously in watch/daemon mode"),
|
||||
):
|
||||
"""Execute media organization or generate a dry-run preview."""
|
||||
import time
|
||||
settings = load_settings_or_default(config)
|
||||
if source:
|
||||
settings.storage.source_dirs = [str(source)]
|
||||
if dest:
|
||||
settings.storage.destination_base = str(dest)
|
||||
if action:
|
||||
settings.general.action = action
|
||||
if threshold:
|
||||
settings.general.confidence_threshold = threshold
|
||||
|
||||
loop_interval = interval or (settings.general.scan_interval_seconds if settings.general.scan_interval_seconds > 0 else (60 if watch else 0))
|
||||
|
||||
engine = init_db(db_path=settings.get_database_path())
|
||||
sorter = MediaSorterApp(settings, engine)
|
||||
|
||||
mode_label = "[bold yellow]DRY-RUN PREVIEW[/bold yellow]" if dry_run else "[bold red]LIVE EXECUTION[/bold red]"
|
||||
|
||||
while True:
|
||||
console.print(Panel(f"Mode: {mode_label} | Action: {settings.general.action.value.upper()}", title="Media Sorter"))
|
||||
start_t = time.time()
|
||||
|
||||
with Progress(
|
||||
SpinnerColumn(),
|
||||
TextColumn("[progress.description]{task.description}"),
|
||||
BarColumn(),
|
||||
TextColumn("[progress.percentage]{task.percentage:>3.0f}%"),
|
||||
TimeRemainingColumn(),
|
||||
console=console,
|
||||
) as progress:
|
||||
task = progress.add_task("[green]Processing...", total=None)
|
||||
|
||||
def on_prog(curr, total, name):
|
||||
progress.update(task, total=total, completed=curr, description=f"Processing: {name[:30]}")
|
||||
|
||||
report = sorter.run(dry_run=dry_run, progress_callback=on_prog)
|
||||
|
||||
elapsed = max(time.time() - start_t, 0.001)
|
||||
throughput = round(report.total_files / elapsed, 1)
|
||||
|
||||
# Print summary table
|
||||
summary_table = Table(title=f"Batch Summary [{report.batch_id[:8]}]")
|
||||
summary_table.add_column("Metric", style="bold")
|
||||
summary_table.add_column("Count", justify="right")
|
||||
|
||||
summary_table.add_row("Total Processed", str(report.total_files))
|
||||
summary_table.add_row("Moved / Organized", f"[green]{report.moved_files}[/green]")
|
||||
summary_table.add_row("Copied", f"[blue]{report.copied_files}[/blue]")
|
||||
summary_table.add_row("Linked", f"[cyan]{report.linked_files}[/cyan]")
|
||||
summary_table.add_row("Skipped (Conflicts / Existing)", f"[dim]{report.skipped_files}[/dim]")
|
||||
summary_table.add_row("Quarantined (Review Queue)", f"[yellow]{report.quarantined_files}[/yellow]")
|
||||
summary_table.add_row("Failures", f"[red]{report.failed_files}[/red]")
|
||||
summary_table.add_row("Throughput", f"{throughput} files/sec ({round(elapsed, 2)}s)")
|
||||
|
||||
console.print(summary_table)
|
||||
|
||||
if dry_run:
|
||||
console.print("\n[bold cyan]Safe dry-run complete. No files were modified on disk.[/bold cyan]")
|
||||
console.print("[dim]To apply these changes live, rerun with --live.[/dim]")
|
||||
|
||||
if loop_interval <= 0:
|
||||
break
|
||||
|
||||
console.print(f"\n[cyan]Sleeping for {loop_interval}s until next scan cycle (press Ctrl+C to stop)...[/cyan]")
|
||||
try:
|
||||
time.sleep(loop_interval)
|
||||
except KeyboardInterrupt:
|
||||
console.print("\n[yellow]Daemon watch loop stopped by user.[/yellow]")
|
||||
break
|
||||
|
||||
|
||||
@app.command()
|
||||
def rollback(
|
||||
batch_id: Optional[str] = typer.Option(None, "--batch-id", "-b", help="Specific batch ID to roll back"),
|
||||
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
|
||||
):
|
||||
"""Roll back a previous organization batch, restoring moved files to original sources."""
|
||||
settings = load_settings_or_default(config)
|
||||
engine = init_db(db_path=settings.get_database_path())
|
||||
sorter = MediaSorterApp(settings, engine)
|
||||
|
||||
with console.status("[bold yellow]Executing transactional rollback...[/bold yellow]"):
|
||||
reverted = sorter.rollback(batch_id)
|
||||
|
||||
if reverted > 0:
|
||||
console.print(f"[bold green]Successfully rolled back {reverted} file operations.[/bold green]")
|
||||
else:
|
||||
console.print("[yellow]No operations were reverted (batch already rolled back or not found).[/yellow]")
|
||||
|
||||
|
||||
@app.command()
|
||||
def history(
|
||||
limit: int = typer.Option(10, "--limit", "-n", help="Number of past batches to display"),
|
||||
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
|
||||
):
|
||||
"""Display history of past organization batches and execution logs."""
|
||||
settings = load_settings_or_default(config)
|
||||
engine = init_db(db_path=settings.get_database_path())
|
||||
|
||||
with get_db_session(engine) as session:
|
||||
batches = session.query(BatchRecord).order_by(BatchRecord.created_at.desc()).limit(limit).all()
|
||||
if not batches:
|
||||
console.print("[dim]No batch records found.[/dim]")
|
||||
return
|
||||
|
||||
table = Table(title=f"Execution History (Last {len(batches)})")
|
||||
table.add_column("Batch ID", style="bold")
|
||||
table.add_column("Date", style="dim")
|
||||
table.add_column("Mode")
|
||||
table.add_column("Status")
|
||||
table.add_column("Total", justify="right")
|
||||
table.add_column("Moved", justify="right")
|
||||
table.add_column("Quarantined", justify="right")
|
||||
|
||||
for b in batches:
|
||||
mode = "[dim]Dry-Run[/dim]" if b.dry_run else "[bold]Live[/bold]"
|
||||
status = f"[green]{b.status}[/green]" if b.status == "COMPLETED" else f"[yellow]{b.status}[/yellow]"
|
||||
date_str = b.created_at.strftime("%Y-%m-%d %H:%M") if b.created_at else "-"
|
||||
table.add_row(b.id[:8], date_str, mode, status, str(b.total_files), str(b.moved_files), str(b.quarantined_files))
|
||||
|
||||
console.print(table)
|
||||
|
||||
|
||||
@quarantine_app.command("list")
|
||||
def quarantine_list(
|
||||
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
|
||||
):
|
||||
"""List pending items requiring manual review."""
|
||||
settings = load_settings_or_default(config)
|
||||
engine = init_db(db_path=settings.get_database_path())
|
||||
|
||||
with get_db_session(engine) as session:
|
||||
qm = QuarantineManager(session)
|
||||
items = qm.list_pending()
|
||||
if not items:
|
||||
console.print("[green]Quarantine queue is empty. All media classified cleanly![/green]")
|
||||
return
|
||||
|
||||
table = Table(title=f"Quarantine Review Queue ({len(items)} items)")
|
||||
table.add_column("ID", justify="right")
|
||||
table.add_column("File Path", overflow="fold")
|
||||
table.add_column("Suggested", style="cyan")
|
||||
table.add_column("Confidence", justify="right")
|
||||
table.add_column("Reason", style="yellow")
|
||||
|
||||
for q in items:
|
||||
conf = f"{int((q.confidence or 0) * 100)}%"
|
||||
table.add_row(str(q.id), q.src, q.suggested_category or "unknown", conf, q.reason)
|
||||
|
||||
console.print(table)
|
||||
|
||||
|
||||
@quarantine_app.command("resolve")
|
||||
def quarantine_resolve(
|
||||
item_id: int = typer.Argument(..., help="Quarantine record ID to resolve"),
|
||||
category: str = typer.Option(..., "--category", "-cat", help="Target category (movie, tv, music, etc.)"),
|
||||
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
|
||||
):
|
||||
"""Manually classify and resolve a quarantined item."""
|
||||
settings = load_settings_or_default(config)
|
||||
engine = init_db(db_path=settings.get_database_path())
|
||||
|
||||
with get_db_session(engine) as session:
|
||||
qm = QuarantineManager(session)
|
||||
success = qm.resolve_item(item_id, category)
|
||||
if success:
|
||||
console.print(f"[green]Successfully resolved item #{item_id} as {category}.[/green]")
|
||||
else:
|
||||
console.print(f"[red]Quarantine item #{item_id} not found.[/red]")
|
||||
|
||||
|
||||
@app.command()
|
||||
def server(
|
||||
host: Optional[str] = typer.Option(None, "--host", "-h", help="Bind host (default from .env or 0.0.0.0)"),
|
||||
port: Optional[int] = typer.Option(None, "--port", "-p", help="Bind port (default from .env or 8080)"),
|
||||
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
|
||||
):
|
||||
"""Launch web dashboard and management server."""
|
||||
settings = load_settings_or_default(config)
|
||||
from .server import create_app
|
||||
bind_host = host or settings.server.host or "0.0.0.0"
|
||||
bind_port = port or settings.server.port or 8080
|
||||
web_app = create_app(settings)
|
||||
console.print(f"[bold green]Starting Media Sorter Dashboard on http://{bind_host}:{bind_port}[/bold green]")
|
||||
uvicorn.run(web_app, host=bind_host, port=bind_port)
|
||||
|
||||
|
||||
@config_app.command("show")
|
||||
def config_show(
|
||||
config: Optional[Path] = typer.Option(None, "--config", "-c", help="Path to configuration file"),
|
||||
):
|
||||
"""Print effective configuration settings."""
|
||||
settings = load_settings_or_default(config)
|
||||
import yaml
|
||||
console.print(yaml.dump(settings.model_dump(mode="json"), default_flow_style=False, sort_keys=False))
|
||||
|
||||
|
||||
@config_app.command("init")
|
||||
def config_init(
|
||||
output: Path = typer.Option(Path("media-sorter.yaml"), "--output", "-o", help="Target config file path"),
|
||||
):
|
||||
"""Create a safe starter configuration file."""
|
||||
if output.exists():
|
||||
console.print(f"[yellow]Configuration file already exists at {output}. Aborting.[/yellow]")
|
||||
return
|
||||
settings = Settings()
|
||||
settings.dump_yaml(output)
|
||||
console.print(f"[green]Created default configuration file at {output}.[/green]")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app()
|
||||
@@ -0,0 +1,497 @@
|
||||
"""Configuration management for Media Sorter.
|
||||
|
||||
Provides Pydantic-based settings validated from YAML, TOML, JSON, or environment variables.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import json
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import yaml
|
||||
from pydantic import BaseModel, Field, field_validator
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
|
||||
class ActionType(str, Enum):
|
||||
MOVE = "move"
|
||||
COPY = "copy"
|
||||
LINK = "link"
|
||||
HARDLINK = "hardlink"
|
||||
|
||||
|
||||
class ConflictPolicy(str, Enum):
|
||||
SKIP = "skip"
|
||||
RENAME_UNIQUE = "rename_unique"
|
||||
QUARANTINE = "quarantine"
|
||||
REPLACE_IF_HIGHER_QUALITY = "replace_if_higher_quality"
|
||||
ERROR = "error"
|
||||
|
||||
|
||||
class DestinationDirs(BaseModel):
|
||||
movies: str = "Movies"
|
||||
tv: str = "TV Shows"
|
||||
anime: str = "Anime"
|
||||
music: str = "Music"
|
||||
audiobooks: str = "Audiobooks"
|
||||
podcasts: str = "Podcasts"
|
||||
home_videos: str = "Home Videos"
|
||||
photos: str = "Photos"
|
||||
archives: str = "Archives"
|
||||
quarantine: str = "Quarantine"
|
||||
|
||||
|
||||
class GeneralSettings(BaseModel):
|
||||
dry_run: bool = True
|
||||
confidence_threshold: float = 0.75
|
||||
worker_count: int = 4
|
||||
min_file_age_seconds: int = 300
|
||||
action: ActionType = ActionType.MOVE
|
||||
preserve_permissions: bool = True
|
||||
log_level: str = "INFO"
|
||||
scan_interval_seconds: int = 0
|
||||
cleanup_empty_dirs: bool = True
|
||||
rename_files: bool = True
|
||||
|
||||
@field_validator("confidence_threshold")
|
||||
@classmethod
|
||||
def validate_confidence(cls, v: float) -> float:
|
||||
if not 0.0 < v <= 1.0:
|
||||
raise ValueError("confidence_threshold must be between 0.0 and 1.0")
|
||||
return v
|
||||
|
||||
@field_validator("worker_count")
|
||||
@classmethod
|
||||
def validate_worker_count(cls, v: int) -> int:
|
||||
if v < 1:
|
||||
raise ValueError("worker_count must be at least 1")
|
||||
return v
|
||||
|
||||
@field_validator("min_file_age_seconds")
|
||||
@classmethod
|
||||
def validate_min_file_age(cls, v: int) -> int:
|
||||
if v < 0:
|
||||
raise ValueError("min_file_age_seconds cannot be negative")
|
||||
return v
|
||||
|
||||
@field_validator("log_level")
|
||||
@classmethod
|
||||
def validate_log_level(cls, v: str) -> str:
|
||||
allowed = {"DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL"}
|
||||
upper = v.upper()
|
||||
if upper not in allowed:
|
||||
raise ValueError(f"log_level must be one of {allowed}")
|
||||
return upper
|
||||
|
||||
|
||||
class StorageSettings(BaseModel):
|
||||
source_dirs: List[str] = Field(default_factory=lambda: ["incoming"])
|
||||
destination_base: str = "organized"
|
||||
destination_dirs: DestinationDirs = Field(default_factory=DestinationDirs)
|
||||
|
||||
|
||||
class ConflictSettings(BaseModel):
|
||||
policy: ConflictPolicy = ConflictPolicy.RENAME_UNIQUE
|
||||
allow_overwrite: bool = False
|
||||
backup_dir: Optional[str] = None
|
||||
|
||||
|
||||
class FilterSettings(BaseModel):
|
||||
include_patterns: List[str] = Field(default_factory=lambda: ["*"])
|
||||
exclude_patterns: List[str] = Field(
|
||||
default_factory=lambda: [
|
||||
".*",
|
||||
"*.part",
|
||||
"*.crdownload",
|
||||
"*.!qB",
|
||||
"Thumbs.db",
|
||||
"desktop.ini",
|
||||
"@eaDir",
|
||||
"$RECYCLE.BIN",
|
||||
"*.txt",
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
class TemplateSettings(BaseModel):
|
||||
movie: str = "{title} ({year})/{movie_name}.{ext}"
|
||||
tv: str = "{title}/Season {season:02d}/{show_name}_{season_episode}.{ext}"
|
||||
anime: str = "{title}/Season {season:02d}/{show_name}_{season_episode} [{group}].{ext}"
|
||||
music: str = "{artist}/{album} ({year})/{disc:01d}{track:02d} - {title}.{ext}"
|
||||
audiobook: str = "{author}/{title}/{track:02d} - {chapter}.{ext}"
|
||||
podcast: str = "{show}/{year}/{show} - {date} - {title}.{ext}"
|
||||
home_video: str = "{year}/{year}-{month:02d} - {event}/{filename}.{ext}"
|
||||
photo: str = "{year}/{year}-{month:02d}/{year}{month:02d}{day:02d}_{time}_{camera}.{ext}"
|
||||
archive: str = "Archives/{filename}.{ext}"
|
||||
quarantine: str = "Quarantine/{reason}/{filename}.{ext}"
|
||||
|
||||
|
||||
class SubtitleSettings(BaseModel):
|
||||
match_video_basename: bool = True
|
||||
preserve_language_code: bool = True
|
||||
|
||||
|
||||
class ArtworkSettings(BaseModel):
|
||||
match_parent_folder: bool = True
|
||||
|
||||
|
||||
class ExtrasSettings(BaseModel):
|
||||
detect_trailers: bool = True
|
||||
trailer_suffix: str = "-trailer"
|
||||
|
||||
|
||||
class SidecarSettings(BaseModel):
|
||||
enabled: bool = True
|
||||
subtitles: SubtitleSettings = Field(default_factory=SubtitleSettings)
|
||||
artwork: ArtworkSettings = Field(default_factory=ArtworkSettings)
|
||||
extras: ExtrasSettings = Field(default_factory=ExtrasSettings)
|
||||
|
||||
|
||||
class DatabaseSettings(BaseModel):
|
||||
path: str = "media_sorter.db"
|
||||
wal_mode: bool = True
|
||||
|
||||
|
||||
class ServerSettings(BaseModel):
|
||||
host: str = "127.0.0.1"
|
||||
port: int = 8080
|
||||
enabled: bool = True
|
||||
|
||||
|
||||
class ProviderSettings(BaseModel):
|
||||
enable_online_metadata: bool = False
|
||||
tmdb_api_key: Optional[str] = None
|
||||
tvdb_api_key: Optional[str] = None
|
||||
rate_limit_per_second: float = 2.0
|
||||
cache_expiry_hours: int = 72
|
||||
|
||||
|
||||
class NotificationSettings(BaseModel):
|
||||
enabled: bool = False
|
||||
webhook_url: Optional[str] = None
|
||||
notify_on_complete: bool = True
|
||||
notify_on_failure: bool = True
|
||||
|
||||
|
||||
class SymlinkSettings(BaseModel):
|
||||
follow_symlinks: bool = False
|
||||
handle_broken_symlinks: str = "skip" # skip | quarantine
|
||||
|
||||
|
||||
class PermissionSettings(BaseModel):
|
||||
preserve_attributes: bool = True
|
||||
file_mode: Optional[str] = None # e.g. "0644"
|
||||
dir_mode: Optional[str] = None # e.g. "0755"
|
||||
owner: Optional[str] = None
|
||||
group: Optional[str] = None
|
||||
|
||||
|
||||
class QuarantineSettings(BaseModel):
|
||||
move_to_quarantine_folder: bool = False
|
||||
directory: str = "Quarantine"
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
model_config = SettingsConfigDict(
|
||||
env_prefix="MEDIA_SORTER_",
|
||||
env_nested_delimiter="__",
|
||||
env_file=".env",
|
||||
env_file_encoding="utf-8",
|
||||
extra="ignore",
|
||||
)
|
||||
|
||||
general: GeneralSettings = Field(default_factory=GeneralSettings)
|
||||
storage: StorageSettings = Field(default_factory=StorageSettings)
|
||||
conflicts: ConflictSettings = Field(default_factory=ConflictSettings)
|
||||
filters: FilterSettings = Field(default_factory=FilterSettings)
|
||||
templates: TemplateSettings = Field(default_factory=TemplateSettings)
|
||||
sidecars: SidecarSettings = Field(default_factory=SidecarSettings)
|
||||
database: DatabaseSettings = Field(default_factory=DatabaseSettings)
|
||||
server: ServerSettings = Field(default_factory=ServerSettings)
|
||||
providers: ProviderSettings = Field(default_factory=ProviderSettings)
|
||||
notifications: NotificationSettings = Field(default_factory=NotificationSettings)
|
||||
symlinks: SymlinkSettings = Field(default_factory=SymlinkSettings)
|
||||
permissions: PermissionSettings = Field(default_factory=PermissionSettings)
|
||||
quarantine: QuarantineSettings = Field(default_factory=QuarantineSettings)
|
||||
|
||||
def model_post_init(self, __context: Any) -> None:
|
||||
super().model_post_init(__context)
|
||||
# Check intuitive environment variable overrides from .env only if not explicitly supplied
|
||||
if "storage" not in self.model_fields_set:
|
||||
downloads_dir = os.getenv("DOWNLOADS_DIR") or os.getenv("SOURCE_DIR")
|
||||
if downloads_dir:
|
||||
self.storage.source_dirs = [downloads_dir]
|
||||
|
||||
movies_dir = os.getenv("MOVIES_DIR")
|
||||
if movies_dir:
|
||||
self.storage.destination_dirs.movies = movies_dir
|
||||
|
||||
shows_dir = os.getenv("SHOWS_DIR") or os.getenv("TV_DIR")
|
||||
if shows_dir:
|
||||
self.storage.destination_dirs.tv = shows_dir
|
||||
|
||||
anime_dir = os.getenv("ANIME_DIR")
|
||||
if anime_dir:
|
||||
self.storage.destination_dirs.anime = anime_dir
|
||||
elif shows_dir:
|
||||
self.storage.destination_dirs.anime = shows_dir
|
||||
|
||||
if "general" not in self.model_fields_set:
|
||||
dry_run_env = os.getenv("DRY_RUN")
|
||||
if dry_run_env is not None:
|
||||
self.general.dry_run = dry_run_env.strip().lower() in ("true", "1", "yes", "on")
|
||||
|
||||
action_env = os.getenv("ACTION")
|
||||
if action_env:
|
||||
try:
|
||||
self.general.action = ActionType(action_env.lower())
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
conf_env = os.getenv("CONFIDENCE_THRESHOLD")
|
||||
if conf_env:
|
||||
try:
|
||||
self.general.confidence_threshold = float(conf_env)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
min_age_env = os.getenv("MIN_FILE_AGE_SECONDS")
|
||||
if min_age_env:
|
||||
try:
|
||||
self.general.min_file_age_seconds = int(min_age_env)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
scan_int_env = os.getenv("SCAN_INTERVAL_SECONDS")
|
||||
if scan_int_env:
|
||||
try:
|
||||
self.general.scan_interval_seconds = int(scan_int_env)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
cleanup_env = os.getenv("CLEANUP_EMPTY_DIRS")
|
||||
if cleanup_env is not None:
|
||||
self.general.cleanup_empty_dirs = cleanup_env.strip().lower() in ("true", "1", "yes", "on")
|
||||
|
||||
rename_env = os.getenv("RENAME_FILES")
|
||||
if rename_env is not None:
|
||||
self.general.rename_files = rename_env.strip().lower() in ("true", "1", "yes", "on")
|
||||
|
||||
if "templates" not in self.model_fields_set:
|
||||
movie_tmpl = os.getenv("MOVIE_TEMPLATE")
|
||||
if movie_tmpl:
|
||||
self.templates.movie = movie_tmpl
|
||||
|
||||
tv_tmpl = os.getenv("TV_TEMPLATE") or os.getenv("SHOW_TEMPLATE") or os.getenv("SHOWS_TEMPLATE")
|
||||
if tv_tmpl:
|
||||
self.templates.tv = tv_tmpl
|
||||
|
||||
if "server" not in self.model_fields_set:
|
||||
host_env = os.getenv("SERVER_HOST") or os.getenv("HOST")
|
||||
if host_env:
|
||||
self.server.host = host_env
|
||||
|
||||
port_env = os.getenv("SERVER_PORT") or os.getenv("PORT")
|
||||
if port_env:
|
||||
try:
|
||||
self.server.port = int(port_env)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
if "database" not in self.model_fields_set:
|
||||
db_path_env = os.getenv("DATABASE_PATH")
|
||||
if db_path_env:
|
||||
self.database.path = db_path_env
|
||||
|
||||
def resolve_path(self, raw_path: str) -> Path:
|
||||
"""Expand environment variables and user home, returning resolved Path."""
|
||||
expanded = os.path.expandvars(raw_path)
|
||||
return Path(expanded).expanduser().resolve()
|
||||
|
||||
def get_source_paths(self) -> List[Path]:
|
||||
return [self.resolve_path(p) for p in self.storage.source_dirs]
|
||||
|
||||
def get_destination_base_path(self) -> Path:
|
||||
return self.resolve_path(self.storage.destination_base)
|
||||
|
||||
def get_destination_path(self, category: str) -> Path:
|
||||
"""Return the destination path for a given category."""
|
||||
base = self.get_destination_base_path()
|
||||
cat_map = {
|
||||
"movie": "movies",
|
||||
"movies": "movies",
|
||||
"tv": "tv",
|
||||
"show": "tv",
|
||||
"shows": "tv",
|
||||
"audiobook": "audiobooks",
|
||||
"podcast": "podcasts",
|
||||
"photo": "photos",
|
||||
"home_video": "home_videos",
|
||||
"archive": "archives",
|
||||
}
|
||||
lookup_key = cat_map.get(category, category)
|
||||
dest_field = getattr(self.storage.destination_dirs, lookup_key, category)
|
||||
path = Path(dest_field)
|
||||
# If dest_field is an explicit relative path (e.g. ./movies, ./shows) or absolute path
|
||||
if path.is_absolute() or str(dest_field).startswith(("./", "../")):
|
||||
return path.resolve()
|
||||
return (base / path).resolve()
|
||||
|
||||
@classmethod
|
||||
def load_from_env_file(cls, env_path: Path | str = ".env") -> Settings:
|
||||
"""Load configuration from a .env file."""
|
||||
path = Path(env_path).expanduser().resolve()
|
||||
settings = cls()
|
||||
if not path.is_file():
|
||||
return settings
|
||||
|
||||
from dotenv import dotenv_values
|
||||
values = dotenv_values(path)
|
||||
|
||||
downloads_dir = os.getenv("DOWNLOADS_DIR") or values.get("DOWNLOADS_DIR") or os.getenv("SOURCE_DIR") or values.get("SOURCE_DIR")
|
||||
if downloads_dir:
|
||||
settings.storage.source_dirs = [downloads_dir]
|
||||
|
||||
movies_dir = os.getenv("MOVIES_DIR") or values.get("MOVIES_DIR")
|
||||
if movies_dir:
|
||||
settings.storage.destination_dirs.movies = movies_dir
|
||||
|
||||
shows_dir = os.getenv("SHOWS_DIR") or values.get("SHOWS_DIR") or os.getenv("TV_DIR") or values.get("TV_DIR")
|
||||
if shows_dir:
|
||||
settings.storage.destination_dirs.tv = shows_dir
|
||||
|
||||
anime_dir = os.getenv("ANIME_DIR") or values.get("ANIME_DIR")
|
||||
if anime_dir:
|
||||
settings.storage.destination_dirs.anime = anime_dir
|
||||
elif shows_dir:
|
||||
settings.storage.destination_dirs.anime = shows_dir
|
||||
|
||||
dry_run = os.getenv("DRY_RUN") if os.getenv("DRY_RUN") is not None else values.get("DRY_RUN")
|
||||
if dry_run is not None:
|
||||
settings.general.dry_run = dry_run.strip().lower() in ("true", "1", "yes", "on")
|
||||
|
||||
action = os.getenv("ACTION") or values.get("ACTION")
|
||||
if action:
|
||||
try:
|
||||
settings.general.action = ActionType(action.lower())
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
conf = os.getenv("CONFIDENCE_THRESHOLD") or values.get("CONFIDENCE_THRESHOLD")
|
||||
if conf:
|
||||
try:
|
||||
settings.general.confidence_threshold = float(conf)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
min_age = os.getenv("MIN_FILE_AGE_SECONDS") or values.get("MIN_FILE_AGE_SECONDS")
|
||||
if min_age:
|
||||
try:
|
||||
settings.general.min_file_age_seconds = int(min_age)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
scan_int = os.getenv("SCAN_INTERVAL_SECONDS") or values.get("SCAN_INTERVAL_SECONDS")
|
||||
if scan_int:
|
||||
try:
|
||||
settings.general.scan_interval_seconds = int(scan_int)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
cleanup = os.getenv("CLEANUP_EMPTY_DIRS") if os.getenv("CLEANUP_EMPTY_DIRS") is not None else values.get("CLEANUP_EMPTY_DIRS")
|
||||
if cleanup is not None:
|
||||
settings.general.cleanup_empty_dirs = str(cleanup).strip().lower() in ("true", "1", "yes", "on")
|
||||
|
||||
rename_files = os.getenv("RENAME_FILES") if os.getenv("RENAME_FILES") is not None else values.get("RENAME_FILES")
|
||||
if rename_files is not None:
|
||||
settings.general.rename_files = str(rename_files).strip().lower() in ("true", "1", "yes", "on")
|
||||
|
||||
movie_tmpl = os.getenv("MOVIE_TEMPLATE") or values.get("MOVIE_TEMPLATE")
|
||||
if movie_tmpl:
|
||||
settings.templates.movie = movie_tmpl
|
||||
|
||||
tv_tmpl = os.getenv("TV_TEMPLATE") or values.get("TV_TEMPLATE") or os.getenv("SHOW_TEMPLATE") or values.get("SHOW_TEMPLATE") or os.getenv("SHOWS_TEMPLATE") or values.get("SHOWS_TEMPLATE")
|
||||
if tv_tmpl:
|
||||
settings.templates.tv = tv_tmpl
|
||||
|
||||
host = os.getenv("SERVER_HOST") or values.get("SERVER_HOST") or os.getenv("HOST") or values.get("HOST")
|
||||
if host:
|
||||
settings.server.host = host
|
||||
|
||||
port = os.getenv("SERVER_PORT") or values.get("SERVER_PORT") or os.getenv("PORT") or values.get("PORT")
|
||||
if port:
|
||||
try:
|
||||
settings.server.port = int(port)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
db_path = os.getenv("DATABASE_PATH") or values.get("DATABASE_PATH")
|
||||
if db_path:
|
||||
settings.database.path = db_path
|
||||
|
||||
return settings
|
||||
|
||||
def save_to_env_file(self, env_path: Path | str = ".env") -> None:
|
||||
"""Persist key user-configurable settings to a .env file."""
|
||||
path = Path(env_path)
|
||||
content = (
|
||||
f"# Media Sorter Configuration\n"
|
||||
f"DOWNLOADS_DIR={self.storage.source_dirs[0] if self.storage.source_dirs else './downloads'}\n"
|
||||
f"MOVIES_DIR={self.storage.destination_dirs.movies}\n"
|
||||
f"SHOWS_DIR={self.storage.destination_dirs.tv}\n"
|
||||
f"DRY_RUN={'true' if self.general.dry_run else 'false'}\n"
|
||||
f"ACTION={self.general.action.value}\n"
|
||||
f"CONFIDENCE_THRESHOLD={self.general.confidence_threshold}\n"
|
||||
f"MIN_FILE_AGE_SECONDS={self.general.min_file_age_seconds}\n"
|
||||
f"SCAN_INTERVAL_SECONDS={self.general.scan_interval_seconds}\n"
|
||||
f"CLEANUP_EMPTY_DIRS={'true' if self.general.cleanup_empty_dirs else 'false'}\n"
|
||||
f"RENAME_FILES={'true' if self.general.rename_files else 'false'}\n"
|
||||
f"MOVIE_TEMPLATE={self.templates.movie}\n"
|
||||
f"TV_TEMPLATE={self.templates.tv}\n"
|
||||
f"SERVER_HOST={self.server.host}\n"
|
||||
f"SERVER_PORT={self.server.port}\n"
|
||||
f"DATABASE_PATH={self.database.path}\n"
|
||||
)
|
||||
path.write_text(content, encoding="utf-8")
|
||||
|
||||
def get_database_path(self) -> Path:
|
||||
return self.resolve_path(self.database.path)
|
||||
|
||||
@classmethod
|
||||
def load_from_file(cls, config_path: Path | str) -> Settings:
|
||||
"""Load configuration from YAML, TOML, or JSON file."""
|
||||
path = Path(config_path).expanduser().resolve()
|
||||
if not path.is_file():
|
||||
raise FileNotFoundError(f"Configuration file not found: {path}")
|
||||
|
||||
ext = path.suffix.lower()
|
||||
with open(path, "r", encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
|
||||
if ext in (".yaml", ".yml"):
|
||||
data = yaml.safe_load(content) or {}
|
||||
elif ext == ".json":
|
||||
data = json.loads(content)
|
||||
elif ext == ".toml":
|
||||
try:
|
||||
import tomllib # Python 3.11+
|
||||
data = tomllib.loads(content)
|
||||
except ImportError:
|
||||
import tomli
|
||||
data = tomli.loads(content)
|
||||
else:
|
||||
# Fallback to YAML loader which can parse JSON and YAML
|
||||
data = yaml.safe_load(content) or {}
|
||||
|
||||
return cls(**data)
|
||||
|
||||
def dump_yaml(self, target_path: Path | str) -> None:
|
||||
"""Dump settings to YAML format."""
|
||||
path = Path(target_path)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
data = self.model_dump(mode="json")
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
yaml.dump(data, f, default_flow_style=False, sort_keys=False)
|
||||
@@ -0,0 +1,84 @@
|
||||
"""Database initialization and session management for Media Sorter.
|
||||
|
||||
Configures SQLite with Write-Ahead Logging (WAL) mode and foreign keys enabled
|
||||
for transactional safety, high concurrency, and crash resilience.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
from typing import Generator, Optional
|
||||
|
||||
from sqlalchemy import create_engine, event
|
||||
from sqlalchemy.engine import Engine
|
||||
from sqlalchemy.orm import Session, declarative_base, scoped_session, sessionmaker
|
||||
|
||||
from .models import Base
|
||||
|
||||
DEFAULT_DB_PATH = Path("media_sorter.db")
|
||||
|
||||
|
||||
def get_engine(db_path: Path | str = DEFAULT_DB_PATH, wal_mode: bool = True) -> Engine:
|
||||
"""Create and configure a SQLite SQLAlchemy engine.
|
||||
|
||||
Enables WAL mode and enforces foreign keys for transactional integrity.
|
||||
"""
|
||||
path = Path(db_path).resolve()
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
db_url = f"sqlite:///{path}"
|
||||
|
||||
engine = create_engine(
|
||||
db_url,
|
||||
connect_args={"check_same_thread": False, "timeout": 30.0},
|
||||
pool_pre_ping=True,
|
||||
)
|
||||
|
||||
@event.listens_for(engine, "connect")
|
||||
def set_sqlite_pragma(dbapi_connection, connection_record):
|
||||
cursor = dbapi_connection.cursor()
|
||||
cursor.execute("PRAGMA foreign_keys=ON")
|
||||
if wal_mode:
|
||||
cursor.execute("PRAGMA journal_mode=WAL")
|
||||
cursor.execute("PRAGMA synchronous=NORMAL")
|
||||
cursor.execute("PRAGMA busy_timeout=30000")
|
||||
cursor.close()
|
||||
|
||||
return engine
|
||||
|
||||
|
||||
def init_db(engine: Optional[Engine] = None, db_path: Path | str = DEFAULT_DB_PATH) -> Engine:
|
||||
"""Initialize all tables defined in models.py if they do not exist."""
|
||||
if engine is None:
|
||||
engine = get_engine(db_path)
|
||||
Base.metadata.create_all(engine)
|
||||
return engine
|
||||
|
||||
|
||||
_ENGINE_SESSION_FACTORIES: dict[Engine, scoped_session[Session]] = {}
|
||||
|
||||
|
||||
def get_session_factory(engine: Engine) -> scoped_session[Session]:
|
||||
"""Retrieve or create a cached scoped session factory bound to the given engine."""
|
||||
if engine not in _ENGINE_SESSION_FACTORIES:
|
||||
_ENGINE_SESSION_FACTORIES[engine] = scoped_session(
|
||||
sessionmaker(autocommit=False, autoflush=False, bind=engine)
|
||||
)
|
||||
return _ENGINE_SESSION_FACTORIES[engine]
|
||||
|
||||
|
||||
@contextmanager
|
||||
def get_db_session(engine: Engine) -> Generator[Session, None, None]:
|
||||
"""Provide a transactional scope around a series of operations."""
|
||||
session_factory = get_session_factory(engine)
|
||||
session: Session = session_factory()
|
||||
try:
|
||||
yield session
|
||||
session.commit()
|
||||
except Exception:
|
||||
session.rollback()
|
||||
raise
|
||||
finally:
|
||||
session.close()
|
||||
session_factory.remove()
|
||||
@@ -0,0 +1,791 @@
|
||||
"""Safe execution engine and transactional operation journal for Media Sorter.
|
||||
|
||||
Enforces dry-run previews, atomic moves, cross-filesystem safety, conflict handling,
|
||||
attribute preservation (POSIX timestamps/permissions), crash recovery, and instant rollbacks.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
import shutil
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
|
||||
import structlog
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from .config import ActionType, ConflictPolicy, Settings
|
||||
from .models import BatchRecord, FileRecord, Operation, OperationStatus, QuarantineRecord, QuarantineStatus
|
||||
|
||||
from contextlib import contextmanager
|
||||
import sys
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
|
||||
class ProcessLockError(Exception):
|
||||
"""Raised when another media-sorter process holds the execution lock."""
|
||||
pass
|
||||
|
||||
|
||||
@contextmanager
|
||||
def acquire_process_lock(lock_file_path: Path):
|
||||
"""Ensure mutual exclusion so multiple workers/instances do not run concurrent batches."""
|
||||
lock_file = Path(lock_file_path).resolve()
|
||||
lock_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
f = open(lock_file, "a+")
|
||||
try:
|
||||
if sys.platform != "win32":
|
||||
import fcntl
|
||||
try:
|
||||
fcntl.flock(f.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
except (IOError, OSError):
|
||||
raise ProcessLockError(
|
||||
f"Another media-sorter process currently holds the lock on {lock_file}."
|
||||
)
|
||||
else:
|
||||
import msvcrt
|
||||
try:
|
||||
msvcrt.locking(f.fileno(), msvcrt.LK_NBLCK, 1)
|
||||
except (IOError, OSError):
|
||||
raise ProcessLockError(
|
||||
f"Another media-sorter process currently holds the lock on {lock_file}."
|
||||
)
|
||||
yield
|
||||
finally:
|
||||
try:
|
||||
if sys.platform != "win32":
|
||||
import fcntl
|
||||
fcntl.flock(f.fileno(), fcntl.LOCK_UN)
|
||||
else:
|
||||
import msvcrt
|
||||
msvcrt.locking(f.fileno(), msvcrt.LK_UNLCK, 1)
|
||||
except Exception:
|
||||
pass
|
||||
f.close()
|
||||
|
||||
|
||||
def copy_extended_attributes(src: Path, dst: Path) -> None:
|
||||
"""Preserve POSIX extended attributes (xattrs) across filesystems where supported."""
|
||||
if hasattr(os, "listxattr") and hasattr(os, "getxattr") and hasattr(os, "setxattr"):
|
||||
try:
|
||||
attrs = os.listxattr(src)
|
||||
for attr in attrs:
|
||||
try:
|
||||
val = os.getxattr(src, attr)
|
||||
os.setxattr(dst, attr, val)
|
||||
except (OSError, PermissionError):
|
||||
pass
|
||||
except (OSError, PermissionError):
|
||||
pass
|
||||
|
||||
|
||||
def compute_file_hash(file_path: Path, max_bytes: int = 1048576) -> str:
|
||||
"""Compute quick partial SHA-256 hash (first 1MB) for fast identity verification."""
|
||||
try:
|
||||
hasher = hashlib.sha256()
|
||||
with open(file_path, "rb") as f:
|
||||
chunk = f.read(max_bytes)
|
||||
hasher.update(chunk)
|
||||
return hasher.hexdigest()
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlannedOperation:
|
||||
src: Path
|
||||
dst: Path
|
||||
action: ActionType
|
||||
category: str
|
||||
confidence: float
|
||||
details: Dict[str, Any] = field(default_factory=dict)
|
||||
is_conflict: bool = False
|
||||
conflict_resolved_dst: Optional[Path] = None
|
||||
quarantine: bool = False
|
||||
quarantine_reason: Optional[str] = None
|
||||
primary_src: Optional[Path] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class BatchExecutionReport:
|
||||
batch_id: str
|
||||
dry_run: bool
|
||||
total_files: int = 0
|
||||
moved_files: int = 0
|
||||
copied_files: int = 0
|
||||
linked_files: int = 0
|
||||
skipped_files: int = 0
|
||||
quarantined_files: int = 0
|
||||
failed_files: int = 0
|
||||
cleaned_dirs: int = 0
|
||||
operations: List[PlannedOperation] = field(default_factory=list)
|
||||
errors: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
class MediaExecutor:
|
||||
"""Executes planned operations with transactional safety and crash recovery."""
|
||||
|
||||
def __init__(self, settings: Settings, session: Session):
|
||||
self.settings = settings
|
||||
self.session = session
|
||||
|
||||
def plan_operations(
|
||||
self, planned_items: List[PlannedOperation]
|
||||
) -> List[PlannedOperation]:
|
||||
"""Validate destination conflicts and resolve destination paths."""
|
||||
allocated_destinations: Dict[Path, PlannedOperation] = {}
|
||||
validated_plan: List[PlannedOperation] = []
|
||||
primary_resolutions: Dict[Path, Path] = {}
|
||||
primary_orig_to_resolved: Dict[Path, Path] = {}
|
||||
|
||||
for item in planned_items:
|
||||
# If already marked for quarantine, keep as is
|
||||
if item.quarantine:
|
||||
validated_plan.append(item)
|
||||
continue
|
||||
|
||||
# Check if this item is a sidecar whose primary was renamed
|
||||
orig_target_dst = item.dst
|
||||
if item.category in ("subtitle", "artwork", "metadata") or item.primary_src:
|
||||
p_src = item.primary_src
|
||||
resolved_p_dst = None
|
||||
if p_src and p_src in primary_resolutions:
|
||||
resolved_p_dst = primary_resolutions[p_src]
|
||||
else:
|
||||
for orig_p_dst, res_p_dst in primary_orig_to_resolved.items():
|
||||
if orig_p_dst.parent == item.dst.parent and item.dst.stem.lower().startswith(orig_p_dst.stem.lower()):
|
||||
resolved_p_dst = res_p_dst
|
||||
break
|
||||
|
||||
if resolved_p_dst and resolved_p_dst.stem != item.dst.stem:
|
||||
orig_stem = item.dst.stem
|
||||
p_orig_stem = None
|
||||
for orig_p_dst in primary_orig_to_resolved:
|
||||
if orig_stem.lower().startswith(orig_p_dst.stem.lower()):
|
||||
p_orig_stem = orig_p_dst.stem
|
||||
break
|
||||
tag = orig_stem[len(p_orig_stem):] if p_orig_stem else ""
|
||||
new_sidecar_name = f"{resolved_p_dst.stem}{tag}{item.dst.suffix}"
|
||||
item.dst = resolved_p_dst.parent / new_sidecar_name
|
||||
|
||||
target_dst = item.dst
|
||||
|
||||
# 1. Check intra-batch duplicate destination conflict
|
||||
if target_dst in allocated_destinations:
|
||||
logger.warning(
|
||||
"Intra-batch destination collision detected",
|
||||
dst=str(target_dst),
|
||||
src1=str(allocated_destinations[target_dst].src),
|
||||
src2=str(item.src),
|
||||
)
|
||||
item.is_conflict = True
|
||||
target_dst = self._resolve_conflict(item.src, target_dst)
|
||||
item.conflict_resolved_dst = target_dst
|
||||
|
||||
# 2. Check on-disk destination conflict
|
||||
if target_dst.exists():
|
||||
logger.info("Destination already exists on disk", dst=str(target_dst), src=str(item.src))
|
||||
item.is_conflict = True
|
||||
target_dst = self._resolve_conflict(item.src, target_dst)
|
||||
item.conflict_resolved_dst = target_dst
|
||||
|
||||
if item.quarantine:
|
||||
validated_plan.append(item)
|
||||
continue
|
||||
|
||||
item.dst = target_dst
|
||||
allocated_destinations[target_dst] = item
|
||||
validated_plan.append(item)
|
||||
|
||||
# Record primary resolution for companion alignment
|
||||
if item.category not in ("subtitle", "artwork", "metadata"):
|
||||
primary_resolutions[item.src] = item.dst
|
||||
primary_orig_to_resolved[orig_target_dst] = item.dst
|
||||
|
||||
return validated_plan
|
||||
|
||||
def _resolve_conflict(self, src: Path, desired_dst: Path) -> Path:
|
||||
"""Resolve conflict according to the configured conflict policy."""
|
||||
policy = self.settings.conflicts.policy
|
||||
|
||||
if policy == ConflictPolicy.SKIP:
|
||||
return desired_dst # Will be skipped during execution
|
||||
elif policy == ConflictPolicy.ERROR:
|
||||
raise FileExistsError(f"Destination conflict: {desired_dst} already exists")
|
||||
elif policy == ConflictPolicy.QUARANTINE:
|
||||
return self.settings.get_destination_path("quarantine") / f"conflicts/{src.name}"
|
||||
elif policy == ConflictPolicy.REPLACE_IF_HIGHER_QUALITY:
|
||||
# Allow replacing existing file (will backup during execution)
|
||||
return desired_dst
|
||||
else:
|
||||
# ConflictPolicy.RENAME_UNIQUE: foo (1).mp4
|
||||
parent = desired_dst.parent
|
||||
stem = desired_dst.stem
|
||||
ext = desired_dst.suffix
|
||||
counter = 1
|
||||
candidate = parent / f"{stem} ({counter}){ext}"
|
||||
while candidate.exists():
|
||||
counter += 1
|
||||
candidate = parent / f"{stem} ({counter}){ext}"
|
||||
return candidate
|
||||
|
||||
def execute_batch(
|
||||
self,
|
||||
planned_items: List[PlannedOperation],
|
||||
dry_run: Optional[bool] = None,
|
||||
progress_callback: Optional[Callable[[int, int, str], None]] = None,
|
||||
) -> BatchExecutionReport:
|
||||
"""Execute a batch of operations transactionally, with dry-run support."""
|
||||
is_dry_run = self.settings.general.dry_run if dry_run is None else dry_run
|
||||
batch_id = str(uuid.uuid4())
|
||||
|
||||
# Validate and resolve destination collisions
|
||||
validated_plan = self.plan_operations(planned_items)
|
||||
|
||||
report = BatchExecutionReport(
|
||||
batch_id=batch_id,
|
||||
dry_run=is_dry_run,
|
||||
total_files=len(validated_plan),
|
||||
operations=validated_plan,
|
||||
)
|
||||
|
||||
# Create Batch Record in DB
|
||||
batch_record = BatchRecord(
|
||||
id=batch_id,
|
||||
dry_run=is_dry_run,
|
||||
status="IN_PROGRESS",
|
||||
total_files=len(validated_plan),
|
||||
)
|
||||
self.session.add(batch_record)
|
||||
self.session.commit()
|
||||
|
||||
total = len(validated_plan)
|
||||
for idx, item in enumerate(validated_plan):
|
||||
if progress_callback:
|
||||
progress_callback(idx + 1, total, str(item.src.name))
|
||||
|
||||
if item.quarantine:
|
||||
self._record_quarantine(batch_id, item, is_dry_run)
|
||||
report.quarantined_files += 1
|
||||
continue
|
||||
|
||||
# Check if skipping due to conflict
|
||||
if item.is_conflict and self.settings.conflicts.policy == ConflictPolicy.SKIP and item.dst.exists():
|
||||
logger.info("Skipping existing destination", dst=str(item.dst))
|
||||
report.skipped_files += 1
|
||||
self._record_operation(
|
||||
batch_id, item, status=OperationStatus.SKIPPED, is_dry_run=is_dry_run
|
||||
)
|
||||
continue
|
||||
|
||||
if is_dry_run:
|
||||
# Dry run preview only: do not touch filesystem
|
||||
if item.action == ActionType.MOVE:
|
||||
report.moved_files += 1
|
||||
elif item.action == ActionType.COPY:
|
||||
report.copied_files += 1
|
||||
elif item.action in (ActionType.LINK, ActionType.HARDLINK):
|
||||
report.linked_files += 1
|
||||
|
||||
self._record_operation(
|
||||
batch_id, item, status=OperationStatus.PLANNED, is_dry_run=True
|
||||
)
|
||||
continue
|
||||
|
||||
# Live execution
|
||||
try:
|
||||
self._execute_single_op(batch_id, item)
|
||||
if item.action == ActionType.MOVE:
|
||||
report.moved_files += 1
|
||||
elif item.action == ActionType.COPY:
|
||||
report.copied_files += 1
|
||||
elif item.action in (ActionType.LINK, ActionType.HARDLINK):
|
||||
report.linked_files += 1
|
||||
except Exception as e:
|
||||
report.failed_files += 1
|
||||
err_msg = f"Failed {item.action} on {item.src} -> {item.dst}: {e}"
|
||||
logger.error(err_msg, exc_info=True)
|
||||
report.errors.append(err_msg)
|
||||
|
||||
# Clean up empty directories in source directories after live moves
|
||||
if not is_dry_run and getattr(self.settings.general, "cleanup_empty_dirs", True):
|
||||
moved_srcs = [
|
||||
item.src
|
||||
for item in validated_plan
|
||||
if item.action == ActionType.MOVE and not item.quarantine
|
||||
]
|
||||
report.cleaned_dirs = self.clean_empty_directories(moved_srcs)
|
||||
|
||||
# Update batch record completion status
|
||||
batch_record.completed_at = datetime.now(timezone.utc)
|
||||
batch_record.moved_files = report.moved_files
|
||||
batch_record.skipped_files = report.skipped_files
|
||||
batch_record.failed_files = report.failed_files
|
||||
batch_record.quarantined_files = report.quarantined_files
|
||||
batch_record.status = "COMPLETED" if report.failed_files == 0 else "PARTIAL_FAILURE"
|
||||
self.session.commit()
|
||||
|
||||
return report
|
||||
|
||||
def _execute_single_op(self, batch_id: str, item: PlannedOperation) -> None:
|
||||
"""Perform atomic move, copy, or link with attribute preservation and journal update."""
|
||||
src = item.src
|
||||
dst = item.dst
|
||||
action = item.action
|
||||
|
||||
if not src.exists():
|
||||
raise FileNotFoundError(f"Source file missing: {src}")
|
||||
|
||||
src_hash = compute_file_hash(src)
|
||||
backup_path: Optional[str] = None
|
||||
|
||||
# Handle backup if replacing
|
||||
if dst.exists():
|
||||
if self.settings.conflicts.policy == ConflictPolicy.REPLACE_IF_HIGHER_QUALITY:
|
||||
b_dir = Path(self.settings.conflicts.backup_dir or ".backup") / batch_id
|
||||
b_dir.mkdir(parents=True, exist_ok=True)
|
||||
backup_dst = b_dir / dst.name
|
||||
shutil.move(dst, backup_dst)
|
||||
backup_path = str(backup_dst)
|
||||
elif not self.settings.conflicts.allow_overwrite:
|
||||
raise FileExistsError(f"Destination exists and allow_overwrite is False: {dst}")
|
||||
|
||||
# Create journal entry in IN_PROGRESS state
|
||||
op = Operation(
|
||||
batch_id=batch_id,
|
||||
src=str(src),
|
||||
dst=str(dst),
|
||||
action=action.value,
|
||||
category=item.category,
|
||||
confidence=item.confidence,
|
||||
src_hash=src_hash,
|
||||
backup_path=backup_path,
|
||||
details=item.details,
|
||||
status=OperationStatus.IN_PROGRESS.value,
|
||||
)
|
||||
self.session.add(op)
|
||||
self.session.commit()
|
||||
|
||||
dst.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
try:
|
||||
if action == ActionType.MOVE:
|
||||
self._safe_move(src, dst)
|
||||
elif action == ActionType.COPY:
|
||||
self._safe_copy(src, dst)
|
||||
elif action == ActionType.LINK:
|
||||
if dst.exists() or dst.is_symlink():
|
||||
dst.unlink()
|
||||
os.symlink(src, dst)
|
||||
elif action == ActionType.HARDLINK:
|
||||
if dst.exists():
|
||||
dst.unlink()
|
||||
os.link(src, dst)
|
||||
|
||||
# Apply custom permissions if specified
|
||||
self._apply_permissions(dst)
|
||||
|
||||
op.status = OperationStatus.COMMITTED.value
|
||||
op.completed_at = datetime.now(timezone.utc)
|
||||
op.dst_hash = compute_file_hash(dst)
|
||||
|
||||
# Update or create FileRecord in database
|
||||
self._update_file_record(dst, item)
|
||||
|
||||
self.session.commit()
|
||||
|
||||
except Exception as e:
|
||||
op.status = OperationStatus.FAILED.value
|
||||
op.error_message = str(e)
|
||||
self.session.commit()
|
||||
raise
|
||||
|
||||
def _safe_move(self, src: Path, dst: Path) -> None:
|
||||
"""Atomic move on same filesystem, or safe temp-copy-atomic-rename cross-filesystem."""
|
||||
try:
|
||||
# Check if same filesystem by comparing st_dev
|
||||
src_dev = src.stat().st_dev
|
||||
dst_parent_dev = dst.parent.stat().st_dev
|
||||
if src_dev == dst_parent_dev:
|
||||
# Same device: atomic rename
|
||||
os.replace(src, dst)
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Cross-filesystem move:
|
||||
# 1. Copy to temp file in destination directory
|
||||
temp_dst = dst.parent / f".tmp_media_sorter_{uuid.uuid4().hex}_{dst.name}"
|
||||
try:
|
||||
shutil.copy2(src, temp_dst)
|
||||
if self.settings.permissions.preserve_attributes:
|
||||
copy_extended_attributes(src, temp_dst)
|
||||
# Verify file size matches
|
||||
if temp_dst.stat().st_size != src.stat().st_size:
|
||||
raise IOError(f"Size mismatch during copy: {temp_dst.stat().st_size} != {src.stat().st_size}")
|
||||
# Atomically replace into final destination
|
||||
os.replace(temp_dst, dst)
|
||||
# Unlink original source
|
||||
src.unlink()
|
||||
finally:
|
||||
if temp_dst.exists():
|
||||
try:
|
||||
temp_dst.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _safe_copy(self, src: Path, dst: Path) -> None:
|
||||
"""Safe copy using temporary file and atomic replace."""
|
||||
temp_dst = dst.parent / f".tmp_media_sorter_{uuid.uuid4().hex}_{dst.name}"
|
||||
try:
|
||||
shutil.copy2(src, temp_dst)
|
||||
if self.settings.permissions.preserve_attributes:
|
||||
copy_extended_attributes(src, temp_dst)
|
||||
if temp_dst.stat().st_size != src.stat().st_size:
|
||||
raise IOError("Copy size mismatch")
|
||||
os.replace(temp_dst, dst)
|
||||
finally:
|
||||
if temp_dst.exists():
|
||||
try:
|
||||
temp_dst.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _apply_permissions(self, path: Path) -> None:
|
||||
"""Apply configured mode bits and ownership safely."""
|
||||
perm_cfg = self.settings.permissions
|
||||
if perm_cfg.file_mode and path.is_file():
|
||||
try:
|
||||
mode = int(perm_cfg.file_mode, 8)
|
||||
os.chmod(path, mode)
|
||||
except Exception as e:
|
||||
logger.debug("Failed setting file mode", path=str(path), error=str(e))
|
||||
|
||||
if (perm_cfg.owner or perm_cfg.group) and hasattr(os, "chown"):
|
||||
try:
|
||||
import pwd
|
||||
import grp
|
||||
uid = -1
|
||||
gid = -1
|
||||
if perm_cfg.owner:
|
||||
uid = int(perm_cfg.owner) if perm_cfg.owner.isdigit() else pwd.getpwnam(perm_cfg.owner).pw_uid
|
||||
if perm_cfg.group:
|
||||
gid = int(perm_cfg.group) if perm_cfg.group.isdigit() else grp.getgrnam(perm_cfg.group).gr_gid
|
||||
os.chown(path, uid, gid)
|
||||
except Exception as e:
|
||||
logger.debug("Failed setting ownership", path=str(path), error=str(e))
|
||||
|
||||
def _update_file_record(self, final_path: Path, item: PlannedOperation) -> None:
|
||||
"""Record the file in the database to prevent re-processing."""
|
||||
try:
|
||||
stat = final_path.stat()
|
||||
rec = self.session.query(FileRecord).filter_by(path=str(final_path)).first()
|
||||
if not rec:
|
||||
rec = FileRecord(
|
||||
path=str(final_path),
|
||||
size=stat.st_size,
|
||||
mtime=stat.st_mtime,
|
||||
category=item.category,
|
||||
confidence=item.confidence,
|
||||
status="organized",
|
||||
last_processed=datetime.now(timezone.utc),
|
||||
)
|
||||
self.session.add(rec)
|
||||
else:
|
||||
rec.size = stat.st_size
|
||||
rec.mtime = stat.st_mtime
|
||||
rec.category = item.category
|
||||
rec.confidence = item.confidence
|
||||
rec.status = "organized"
|
||||
rec.last_processed = datetime.now(timezone.utc)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _record_operation(
|
||||
self, batch_id: str, item: PlannedOperation, status: OperationStatus, is_dry_run: bool
|
||||
) -> None:
|
||||
op = Operation(
|
||||
batch_id=batch_id,
|
||||
src=str(item.src),
|
||||
dst=str(item.dst),
|
||||
action=item.action.value,
|
||||
category=item.category,
|
||||
confidence=item.confidence,
|
||||
details=item.details,
|
||||
status=status.value,
|
||||
)
|
||||
self.session.add(op)
|
||||
self.session.commit()
|
||||
|
||||
def _record_quarantine(self, batch_id: str, item: PlannedOperation, is_dry_run: bool) -> None:
|
||||
q = self.session.query(QuarantineRecord).filter_by(src=str(item.src)).first()
|
||||
if not q:
|
||||
q = QuarantineRecord(
|
||||
src=str(item.src),
|
||||
suggested_category=item.category,
|
||||
confidence=item.confidence,
|
||||
reason=item.quarantine_reason or "Low confidence or unclassifiable",
|
||||
signals=item.details,
|
||||
status=QuarantineStatus.PENDING.value,
|
||||
)
|
||||
self.session.add(q)
|
||||
self.session.commit()
|
||||
|
||||
if not is_dry_run and self.settings.quarantine.move_to_quarantine_folder:
|
||||
# Move to quarantine folder
|
||||
q_dir = self.settings.get_destination_path("quarantine") / (item.quarantine_reason or "review")
|
||||
q_dir.mkdir(parents=True, exist_ok=True)
|
||||
dst_path = q_dir / item.src.name
|
||||
try:
|
||||
self._safe_move(item.src, dst_path)
|
||||
q.resolved_path = str(dst_path)
|
||||
self.session.commit()
|
||||
except Exception as e:
|
||||
logger.error("Failed moving to quarantine folder", src=str(item.src), error=str(e))
|
||||
|
||||
def rollback_batch(self, batch_id: Optional[str] = None) -> int:
|
||||
"""Invert all COMMITTED operations in a batch, returning count of reverted files."""
|
||||
query = self.session.query(BatchRecord)
|
||||
if batch_id:
|
||||
batch = query.filter_by(id=batch_id).first()
|
||||
else:
|
||||
# Default to latest non-rolled-back completed batch
|
||||
batch = (
|
||||
query.filter(
|
||||
BatchRecord.status.in_(["COMPLETED", "PARTIAL_FAILURE", "PARTIAL_ROLLBACK"]),
|
||||
BatchRecord.dry_run == False,
|
||||
)
|
||||
.order_by(BatchRecord.created_at.desc())
|
||||
.first()
|
||||
)
|
||||
|
||||
if not batch:
|
||||
logger.warning("No qualifying batch found for rollback", requested_id=batch_id)
|
||||
return 0
|
||||
|
||||
logger.info("Initiating rollback", batch_id=batch.id)
|
||||
# Query committed operations in reverse execution order
|
||||
ops = (
|
||||
self.session.query(Operation)
|
||||
.filter_by(batch_id=batch.id, status=OperationStatus.COMMITTED.value)
|
||||
.order_by(Operation.id.desc())
|
||||
.all()
|
||||
)
|
||||
|
||||
reverted_count = 0
|
||||
failed_count = 0
|
||||
reverted_dest_dirs: Set[Path] = set()
|
||||
|
||||
for op in ops:
|
||||
src = Path(op.src)
|
||||
dst = Path(op.dst)
|
||||
action = op.action
|
||||
|
||||
try:
|
||||
if action == ActionType.MOVE.value:
|
||||
if dst.exists():
|
||||
src.parent.mkdir(parents=True, exist_ok=True)
|
||||
self._safe_move(dst, src)
|
||||
reverted_count += 1
|
||||
reverted_dest_dirs.add(dst.parent)
|
||||
|
||||
# Restore backup if one was taken
|
||||
if op.backup_path and Path(op.backup_path).exists():
|
||||
self._safe_move(Path(op.backup_path), dst)
|
||||
|
||||
elif action == ActionType.COPY.value:
|
||||
if dst.exists():
|
||||
dst.unlink()
|
||||
reverted_count += 1
|
||||
reverted_dest_dirs.add(dst.parent)
|
||||
|
||||
elif action in (ActionType.LINK.value, ActionType.HARDLINK.value):
|
||||
if dst.exists() or dst.is_symlink():
|
||||
dst.unlink()
|
||||
reverted_count += 1
|
||||
reverted_dest_dirs.add(dst.parent)
|
||||
|
||||
op.status = OperationStatus.ROLLED_BACK.value
|
||||
|
||||
# Delete FileRecord for destination
|
||||
rec = self.session.query(FileRecord).filter_by(path=str(dst)).first()
|
||||
if rec:
|
||||
self.session.delete(rec)
|
||||
except Exception as e:
|
||||
failed_count += 1
|
||||
logger.error("Error reverting operation during rollback", op_id=op.id, error=str(e))
|
||||
|
||||
if failed_count == 0 and reverted_count > 0:
|
||||
batch.status = "ROLLED_BACK"
|
||||
elif reverted_count > 0:
|
||||
batch.status = "PARTIAL_ROLLBACK"
|
||||
else:
|
||||
batch.status = "ROLLBACK_FAILED"
|
||||
|
||||
self.session.commit()
|
||||
|
||||
# Clean empty directories in destination tree
|
||||
self._clean_empty_destination_dirs(reverted_dest_dirs)
|
||||
|
||||
return reverted_count
|
||||
|
||||
def _clean_empty_destination_dirs(self, dest_dirs: Set[Path]) -> None:
|
||||
"""Prune empty parent folders in destination tree after rollback."""
|
||||
dest_roots = {
|
||||
self.settings.get_destination_path(cat).resolve()
|
||||
for cat in ("movie", "tv", "anime", "music", "audiobook", "podcast", "photo", "home_video", "documentary", "quarantine")
|
||||
}
|
||||
dest_base = self.settings.get_destination_base_path().resolve()
|
||||
dest_roots.add(dest_base)
|
||||
|
||||
candidate_dirs: Set[Path] = set()
|
||||
for d in dest_dirs:
|
||||
try:
|
||||
curr = d.resolve()
|
||||
while curr not in dest_roots and any(curr.is_relative_to(r) for r in dest_roots):
|
||||
candidate_dirs.add(curr)
|
||||
curr = curr.parent
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
sorted_dirs = sorted(candidate_dirs, key=lambda p: len(p.parts), reverse=True)
|
||||
for d in sorted_dirs:
|
||||
if not d.exists() or not d.is_dir() or d in dest_roots:
|
||||
continue
|
||||
try:
|
||||
entries = [
|
||||
e for e in d.iterdir()
|
||||
if e.name not in (".DS_Store", "Thumbs.db", "desktop.ini")
|
||||
]
|
||||
if not entries:
|
||||
for junk in list(d.iterdir()):
|
||||
try:
|
||||
junk.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
d.rmdir()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def rollback_all(self) -> int:
|
||||
"""Roll back ALL completed, non-rolled-back batches in reverse chronological order."""
|
||||
batches = (
|
||||
self.session.query(BatchRecord)
|
||||
.filter(
|
||||
BatchRecord.status.in_(["COMPLETED", "PARTIAL_FAILURE", "PARTIAL_ROLLBACK"]),
|
||||
BatchRecord.dry_run == False,
|
||||
)
|
||||
.order_by(BatchRecord.created_at.desc())
|
||||
.all()
|
||||
)
|
||||
total_reverted = 0
|
||||
for batch in batches:
|
||||
total_reverted += self.rollback_batch(batch.id)
|
||||
return total_reverted
|
||||
|
||||
|
||||
def recover_interrupted_batches(self) -> int:
|
||||
"""Clean up orphaned temp files and mark interrupted operations as FAILED."""
|
||||
in_progress_ops = (
|
||||
self.session.query(Operation)
|
||||
.filter_by(status=OperationStatus.IN_PROGRESS.value)
|
||||
.all()
|
||||
)
|
||||
recovered_count = 0
|
||||
|
||||
for op in in_progress_ops:
|
||||
logger.warning("Found interrupted operation during crash recovery", op_id=op.id, src=op.src, dst=op.dst)
|
||||
op.status = OperationStatus.FAILED.value
|
||||
op.error_message = "Interrupted by system crash or process kill"
|
||||
recovered_count += 1
|
||||
|
||||
if recovered_count > 0:
|
||||
self.session.commit()
|
||||
|
||||
return recovered_count
|
||||
|
||||
def clean_empty_directories(self, moved_src_paths: List[Path]) -> int:
|
||||
"""Remove empty parent directories and delete .txt / junk files in source folders after moving files.
|
||||
|
||||
Ascends from moved file parent folders up to, but never removing, the source root directories.
|
||||
Also removes companion or orphaned .txt files left behind in source folders.
|
||||
"""
|
||||
source_roots = {p.resolve() for p in self.settings.get_source_paths()}
|
||||
# Also include any parent roots if configured
|
||||
candidate_dirs: Set[Path] = set()
|
||||
for src in moved_src_paths:
|
||||
try:
|
||||
# Delete companion .txt file (e.g. Movie.txt alongside Movie.mkv)
|
||||
comp = src.with_suffix(".txt")
|
||||
if comp.is_file():
|
||||
try:
|
||||
comp.unlink()
|
||||
logger.info("Deleted companion .txt file during cleanup", file=str(comp))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
parent = src.resolve().parent
|
||||
while parent not in source_roots and any(parent.is_relative_to(root) for root in source_roots):
|
||||
candidate_dirs.add(parent)
|
||||
parent = parent.parent
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
# Sort candidate directories deepest first (longest path / most parts first)
|
||||
sorted_dirs = sorted(candidate_dirs, key=lambda d: len(d.parts), reverse=True)
|
||||
removed_count = 0
|
||||
|
||||
for d in sorted_dirs:
|
||||
if not d.exists() or not d.is_dir():
|
||||
continue
|
||||
# Ensure we never delete a configured source root
|
||||
if d in source_roots:
|
||||
continue
|
||||
try:
|
||||
# Delete any .txt files in candidate directories during cleanup
|
||||
for child in list(d.iterdir()):
|
||||
if child.is_file() and child.name.lower().endswith(".txt"):
|
||||
try:
|
||||
child.unlink()
|
||||
logger.info("Deleted .txt file during cleanup", file=str(child))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Check if directory contains any remaining files or subdirs (ignoring OS junk and .txt files)
|
||||
entries = [
|
||||
e for e in d.iterdir()
|
||||
if e.name not in (".DS_Store", "Thumbs.db", "desktop.ini") and not e.name.lower().endswith(".txt")
|
||||
]
|
||||
if not entries:
|
||||
# Clean up junk files before rmdir
|
||||
for junk in d.iterdir():
|
||||
try:
|
||||
junk.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
d.rmdir()
|
||||
removed_count += 1
|
||||
logger.info("Cleaned up empty source directory", directory=str(d))
|
||||
except (OSError, PermissionError) as e:
|
||||
logger.debug("Could not remove directory (not empty or permissions issue)", directory=str(d), error=str(e))
|
||||
|
||||
# Also clean up any .txt files left in source roots
|
||||
for root in source_roots:
|
||||
if root.exists() and root.is_dir():
|
||||
try:
|
||||
for child in list(root.iterdir()):
|
||||
if child.is_file() and child.name.lower().endswith(".txt"):
|
||||
try:
|
||||
child.unlink()
|
||||
logger.info("Deleted .txt file in source root during cleanup", file=str(child))
|
||||
except Exception:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return removed_count
|
||||
@@ -0,0 +1,299 @@
|
||||
"""Library management and show/movie memory indexing engine.
|
||||
|
||||
Tracks known shows and movies in the library, syncs filesystem library directories,
|
||||
and provides automatic show memory routing for incoming downloads.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
import structlog
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from .config import Settings
|
||||
from .models import LibraryItem, utc_now
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
VIDEO_EXTENSIONS = {".mkv", ".mp4", ".m4v", ".avi", ".mov", ".webm", ".ts", ".flv", ".wmv"}
|
||||
|
||||
|
||||
def clean_show_title(raw: str) -> str:
|
||||
"""Strip release tags, quality, year suffixes, and bracketed text from a directory name."""
|
||||
clean = re.sub(r"\[[^\]]+\]|\([^\)]+\)", "", raw).strip()
|
||||
# Strip trailing quality / encoding specs
|
||||
clean = re.sub(
|
||||
r"(?i)\b(1080p|720p|2160p|4k|bluray|bdrip|webrip|web-dl|x264|x265|hevc|h\.?264|h\.?265|dts|aac|ac3|remux|repack)\b.*",
|
||||
"",
|
||||
clean,
|
||||
)
|
||||
# Strip Season pack identifiers like S01-S08 or Season 1
|
||||
clean = re.sub(r"(?i)\b(?:s\d+[-_s\d]*|season\s*\d+.*)\b", "", clean)
|
||||
clean = re.sub(r"[\._]+", " ", clean).strip(" -_")
|
||||
return clean if len(clean) >= 2 else raw.strip()
|
||||
|
||||
|
||||
def sync_library_from_disk(session: Session, settings: Settings) -> Dict[str, int]:
|
||||
"""Scan configured SHOWS_DIR and MOVIES_DIR on disk and synchronize library_items."""
|
||||
shows_dir = settings.get_destination_path("tv")
|
||||
movies_dir = settings.get_destination_path("movie")
|
||||
|
||||
shows_count = 0
|
||||
movies_count = 0
|
||||
|
||||
# Cache existing records in memory by (title.lower(), category)
|
||||
existing_items: Dict[Tuple[str, str], LibraryItem] = {
|
||||
(item.title.lower(), item.category): item for item in session.query(LibraryItem).all()
|
||||
}
|
||||
|
||||
# 1. Scan Shows Directory
|
||||
if shows_dir.exists() and shows_dir.is_dir():
|
||||
try:
|
||||
for entry in shows_dir.iterdir():
|
||||
if entry.name.startswith(".") or not entry.is_dir():
|
||||
continue
|
||||
|
||||
folder_name = entry.name
|
||||
title = clean_show_title(folder_name)
|
||||
if not title:
|
||||
continue
|
||||
|
||||
# Count video files and detect seasons
|
||||
episodes = 0
|
||||
seasons = set()
|
||||
try:
|
||||
for root, _, files in os.walk(entry):
|
||||
for f in files:
|
||||
ext = os.path.splitext(f)[1].lower()
|
||||
if ext in VIDEO_EXTENSIONS:
|
||||
episodes += 1
|
||||
s_m = re.search(r"(?i)\b(?:season|s)\s*(\d{1,2})\b", Path(root).name)
|
||||
if s_m:
|
||||
seasons.add(int(s_m.group(1)))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
key = (title.lower(), "tv")
|
||||
if key in existing_items:
|
||||
item = existing_items[key]
|
||||
item.destination_folder = str(entry)
|
||||
item.item_count = max(item.item_count, episodes)
|
||||
item.seasons_count = max(item.seasons_count, len(seasons))
|
||||
item.last_updated = utc_now()
|
||||
else:
|
||||
item = LibraryItem(
|
||||
title=title,
|
||||
category="tv",
|
||||
destination_folder=str(entry),
|
||||
item_count=episodes,
|
||||
seasons_count=len(seasons),
|
||||
first_detected=utc_now(),
|
||||
last_updated=utc_now(),
|
||||
)
|
||||
session.add(item)
|
||||
existing_items[key] = item
|
||||
shows_count += 1
|
||||
except Exception as e:
|
||||
logger.error("Error scanning shows directory for library", error=str(e))
|
||||
|
||||
# 2. Scan Movies Directory
|
||||
if movies_dir.exists() and movies_dir.is_dir():
|
||||
try:
|
||||
for entry in movies_dir.iterdir():
|
||||
if entry.name.startswith("."):
|
||||
continue
|
||||
|
||||
title = entry.name
|
||||
year = None
|
||||
y_m = re.search(r"\b(19\d\d|20\d\d)\b", entry.name)
|
||||
if y_m:
|
||||
year = int(y_m.group(1))
|
||||
title = entry.name[: y_m.start()].strip(" (.-_")
|
||||
|
||||
clean = clean_show_title(title)
|
||||
item_files = 1
|
||||
if entry.is_dir():
|
||||
try:
|
||||
item_files = sum(
|
||||
1 for _, _, files in os.walk(entry)
|
||||
for f in files if os.path.splitext(f)[1].lower() in VIDEO_EXTENSIONS
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
key = (clean.lower(), "movie")
|
||||
if key in existing_items:
|
||||
item = existing_items[key]
|
||||
item.destination_folder = str(entry)
|
||||
item.year = year or item.year
|
||||
item.item_count = max(item.item_count, item_files)
|
||||
item.last_updated = utc_now()
|
||||
else:
|
||||
item = LibraryItem(
|
||||
title=clean,
|
||||
category="movie",
|
||||
year=year,
|
||||
destination_folder=str(entry),
|
||||
item_count=item_files,
|
||||
first_detected=utc_now(),
|
||||
last_updated=utc_now(),
|
||||
)
|
||||
session.add(item)
|
||||
existing_items[key] = item
|
||||
movies_count += 1
|
||||
except Exception as e:
|
||||
logger.error("Error scanning movies directory for library", error=str(e))
|
||||
|
||||
session.commit()
|
||||
logger.info("Library synchronized with disk", shows=shows_count, movies=movies_count)
|
||||
return {"shows_synced": shows_count, "movies_synced": movies_count}
|
||||
|
||||
|
||||
def record_detected_item(
|
||||
session: Session,
|
||||
settings: Settings,
|
||||
title: str,
|
||||
category: str,
|
||||
destination_folder: Optional[str] = None,
|
||||
year: Optional[int] = None,
|
||||
poster_url: Optional[str] = None,
|
||||
delta_count: int = 0,
|
||||
) -> LibraryItem:
|
||||
"""Record or update a show or movie in the library database."""
|
||||
category = category.lower()
|
||||
if category not in ("tv", "movie"):
|
||||
category = "tv"
|
||||
|
||||
clean = clean_show_title(title) if category == "tv" else title.strip()
|
||||
item = session.query(LibraryItem).filter_by(title=clean, category=category).first()
|
||||
|
||||
if not destination_folder:
|
||||
dest_base = settings.get_destination_path(category)
|
||||
destination_folder = str(dest_base / clean)
|
||||
|
||||
if item:
|
||||
if destination_folder:
|
||||
item.destination_folder = destination_folder
|
||||
if year:
|
||||
item.year = year
|
||||
if poster_url and not item.poster_url:
|
||||
item.poster_url = poster_url
|
||||
if delta_count:
|
||||
item.item_count = max(0, item.item_count + delta_count)
|
||||
item.last_updated = utc_now()
|
||||
else:
|
||||
item = LibraryItem(
|
||||
title=clean,
|
||||
category=category,
|
||||
year=year,
|
||||
destination_folder=destination_folder,
|
||||
poster_url=poster_url,
|
||||
item_count=max(0, delta_count),
|
||||
first_detected=utc_now(),
|
||||
last_updated=utc_now(),
|
||||
)
|
||||
session.add(item)
|
||||
|
||||
session.commit()
|
||||
return item
|
||||
|
||||
|
||||
def get_known_shows(session: Session) -> List[Dict[str, Any]]:
|
||||
"""Return all known TV shows in the library for matching."""
|
||||
items = session.query(LibraryItem).filter_by(category="tv").all()
|
||||
shows = []
|
||||
for item in items:
|
||||
clean = item.title.strip()
|
||||
if len(clean) >= 2:
|
||||
shows.append({
|
||||
"title": clean,
|
||||
"raw_title": item.title,
|
||||
"destination_folder": item.destination_folder,
|
||||
"poster_url": item.poster_url,
|
||||
"item_count": item.item_count,
|
||||
})
|
||||
# Sort by title length descending so longer specific titles match first
|
||||
shows.sort(key=lambda x: len(x["title"]), reverse=True)
|
||||
return shows
|
||||
|
||||
|
||||
def match_known_show(filename_or_text: str, known_shows: List[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
|
||||
"""Check if filename_or_text contains or matches a known show in the library."""
|
||||
if not filename_or_text or not known_shows:
|
||||
return None
|
||||
|
||||
# Replace separators with spaces
|
||||
normalized = re.sub(r"[\._]+", " ", filename_or_text)
|
||||
|
||||
for show in known_shows:
|
||||
title = show["title"]
|
||||
if len(title) < 3:
|
||||
continue
|
||||
|
||||
# Check whole word match
|
||||
pattern = r"(?i)(?<![a-z0-9])" + re.escape(title) + r"(?![a-z0-9])"
|
||||
if re.search(pattern, normalized):
|
||||
return show
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def list_library_items(
|
||||
session: Session,
|
||||
category: Optional[str] = None,
|
||||
search: Optional[str] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""List library items with counts, optionally filtered by category and search term."""
|
||||
q = session.query(LibraryItem)
|
||||
if category and category.lower() in ("tv", "movie"):
|
||||
q = q.filter_by(category=category.lower())
|
||||
|
||||
if search:
|
||||
s = f"%{search.strip()}%"
|
||||
q = q.filter(LibraryItem.title.ilike(s))
|
||||
|
||||
items = q.order_by(LibraryItem.title.asc()).all()
|
||||
|
||||
total_shows = session.query(LibraryItem).filter_by(category="tv").count()
|
||||
total_movies = session.query(LibraryItem).filter_by(category="movie").count()
|
||||
|
||||
shows_list = []
|
||||
movies_list = []
|
||||
|
||||
for item in items:
|
||||
d = {
|
||||
"id": item.id,
|
||||
"title": item.title,
|
||||
"category": item.category,
|
||||
"year": item.year,
|
||||
"destination_folder": item.destination_folder,
|
||||
"poster_url": item.poster_url,
|
||||
"item_count": item.item_count,
|
||||
"seasons_count": item.seasons_count,
|
||||
"first_detected": item.first_detected.isoformat() if item.first_detected else None,
|
||||
"last_updated": item.last_updated.isoformat() if item.last_updated else None,
|
||||
}
|
||||
if item.category == "tv":
|
||||
shows_list.append(d)
|
||||
else:
|
||||
movies_list.append(d)
|
||||
|
||||
return {
|
||||
"total_shows": total_shows,
|
||||
"total_movies": total_movies,
|
||||
"shows": shows_list,
|
||||
"movies": movies_list,
|
||||
}
|
||||
|
||||
|
||||
def clear_library(session: Session) -> int:
|
||||
"""Clear all indexed show and movie items from the library catalog database."""
|
||||
deleted_count = session.query(LibraryItem).delete()
|
||||
session.commit()
|
||||
logger.info("Library catalog cleared", deleted_count=deleted_count)
|
||||
return deleted_count
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
"""SQLAlchemy database models for Media Sorter.
|
||||
|
||||
Provides data structures for file tracking, operation journaling, quarantine,
|
||||
batch execution, and configuration auditing.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime
|
||||
from enum import Enum
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from sqlalchemy import (
|
||||
Boolean,
|
||||
Column,
|
||||
DateTime,
|
||||
Float,
|
||||
ForeignKey,
|
||||
Index,
|
||||
Integer,
|
||||
JSON,
|
||||
String,
|
||||
Text,
|
||||
UniqueConstraint,
|
||||
)
|
||||
from sqlalchemy.orm import declarative_base, relationship
|
||||
|
||||
Base = declarative_base()
|
||||
|
||||
|
||||
class OperationStatus(str, Enum):
|
||||
PLANNED = "PLANNED"
|
||||
IN_PROGRESS = "IN_PROGRESS"
|
||||
COMMITTED = "COMMITTED"
|
||||
FAILED = "FAILED"
|
||||
ROLLED_BACK = "ROLLED_BACK"
|
||||
SKIPPED = "SKIPPED"
|
||||
|
||||
|
||||
class QuarantineStatus(str, Enum):
|
||||
PENDING = "PENDING"
|
||||
RESOLVED = "RESOLVED"
|
||||
IGNORED = "IGNORED"
|
||||
|
||||
|
||||
def utc_now():
|
||||
return datetime.datetime.now(datetime.timezone.utc)
|
||||
|
||||
|
||||
class BatchRecord(Base):
|
||||
"""Tracks an execution batch (one invocation of media-sorter run or dry-run)."""
|
||||
|
||||
__tablename__ = "batches"
|
||||
|
||||
id = Column(String(36), primary_key=True) # UUIDv4
|
||||
created_at = Column(DateTime(timezone=True), default=utc_now, nullable=False)
|
||||
completed_at = Column(DateTime(timezone=True), nullable=True)
|
||||
dry_run = Column(Boolean, default=False, nullable=False)
|
||||
status = Column(String(32), default="IN_PROGRESS", nullable=False) # IN_PROGRESS, COMPLETED, FAILED, ROLLED_BACK
|
||||
total_files = Column(Integer, default=0, nullable=False)
|
||||
moved_files = Column(Integer, default=0, nullable=False)
|
||||
skipped_files = Column(Integer, default=0, nullable=False)
|
||||
failed_files = Column(Integer, default=0, nullable=False)
|
||||
quarantined_files = Column(Integer, default=0, nullable=False)
|
||||
|
||||
operations = relationship("Operation", back_populates="batch", cascade="all, delete-orphan")
|
||||
|
||||
__table_args__ = (
|
||||
Index("ix_batches_created_at", "created_at"),
|
||||
)
|
||||
|
||||
|
||||
class FileRecord(Base):
|
||||
"""Tracks known files to avoid redundant probing and detect changes."""
|
||||
|
||||
__tablename__ = "files"
|
||||
|
||||
id = Column(Integer, primary_key=True, autoincrement=True)
|
||||
path = Column(String(1024), unique=True, nullable=False)
|
||||
size = Column(Integer, nullable=False)
|
||||
mtime = Column(Float, nullable=False)
|
||||
content_hash = Column(String(64), nullable=True)
|
||||
status = Column(String(32), default="scanned", nullable=False) # scanned, organized, quarantined, skipped, error
|
||||
category = Column(String(32), nullable=True)
|
||||
confidence = Column(Float, nullable=True)
|
||||
first_seen = Column(DateTime(timezone=True), default=utc_now, nullable=False)
|
||||
last_processed = Column(DateTime(timezone=True), nullable=True)
|
||||
|
||||
__table_args__ = (
|
||||
Index("ix_files_path", "path"),
|
||||
Index("ix_files_status", "status"),
|
||||
)
|
||||
|
||||
|
||||
class Operation(Base):
|
||||
"""Operation journal entry for atomic moves, copies, or links."""
|
||||
|
||||
__tablename__ = "operations"
|
||||
|
||||
id = Column(Integer, primary_key=True, autoincrement=True)
|
||||
batch_id = Column(String(36), ForeignKey("batches.id"), nullable=False)
|
||||
src = Column(String(1024), nullable=False)
|
||||
dst = Column(String(1024), nullable=False)
|
||||
action = Column(String(32), nullable=False) # move, copy, link, hardlink
|
||||
status = Column(String(32), default=OperationStatus.PLANNED.value, nullable=False)
|
||||
category = Column(String(32), nullable=True)
|
||||
confidence = Column(Float, nullable=True)
|
||||
src_hash = Column(String(64), nullable=True)
|
||||
dst_hash = Column(String(64), nullable=True)
|
||||
backup_path = Column(String(1024), nullable=True)
|
||||
details = Column(JSON, nullable=True) # reasoning, sidecars, format info
|
||||
error_message = Column(Text, nullable=True)
|
||||
created_at = Column(DateTime(timezone=True), default=utc_now, nullable=False)
|
||||
completed_at = Column(DateTime(timezone=True), nullable=True)
|
||||
|
||||
batch = relationship("BatchRecord", back_populates="operations")
|
||||
|
||||
__table_args__ = (
|
||||
Index("ix_operations_batch_id", "batch_id"),
|
||||
Index("ix_operations_status", "status"),
|
||||
Index("ix_operations_src", "src"),
|
||||
Index("ix_operations_dst", "dst"),
|
||||
)
|
||||
|
||||
|
||||
class QuarantineRecord(Base):
|
||||
"""Stores files that failed confidence threshold or require manual user review."""
|
||||
|
||||
__tablename__ = "quarantine"
|
||||
|
||||
id = Column(Integer, primary_key=True, autoincrement=True)
|
||||
src = Column(String(1024), unique=True, nullable=False)
|
||||
suggested_category = Column(String(32), nullable=True)
|
||||
confidence = Column(Float, nullable=True)
|
||||
reason = Column(String(256), nullable=False)
|
||||
signals = Column(JSON, nullable=True) # Diagnostic details of why it was flagged
|
||||
status = Column(String(32), default=QuarantineStatus.PENDING.value, nullable=False)
|
||||
resolved_path = Column(String(1024), nullable=True)
|
||||
created_at = Column(DateTime(timezone=True), default=utc_now, nullable=False)
|
||||
resolved_at = Column(DateTime(timezone=True), nullable=True)
|
||||
|
||||
__table_args__ = (
|
||||
Index("ix_quarantine_status", "status"),
|
||||
Index("ix_quarantine_src", "src"),
|
||||
)
|
||||
|
||||
|
||||
class ConfigAudit(Base):
|
||||
"""Tracks configuration states for reproducibility and auditing."""
|
||||
|
||||
__tablename__ = "config_audit"
|
||||
|
||||
id = Column(Integer, primary_key=True, autoincrement=True)
|
||||
loaded_at = Column(DateTime(timezone=True), default=utc_now, nullable=False)
|
||||
config_json = Column(JSON, nullable=False)
|
||||
|
||||
__table_args__ = (
|
||||
Index("ix_config_loaded", "loaded_at"),
|
||||
)
|
||||
|
||||
|
||||
class LibraryItem(Base):
|
||||
"""Tracks known shows and movies in the user's library for automated routing and cataloging."""
|
||||
|
||||
__tablename__ = "library_items"
|
||||
|
||||
id = Column(Integer, primary_key=True, autoincrement=True)
|
||||
title = Column(String(256), nullable=False)
|
||||
category = Column(String(32), nullable=False) # "tv" or "movie"
|
||||
year = Column(Integer, nullable=True)
|
||||
destination_folder = Column(String(1024), nullable=False)
|
||||
poster_url = Column(String(1024), nullable=True)
|
||||
item_count = Column(Integer, default=0, nullable=False)
|
||||
seasons_count = Column(Integer, default=0, nullable=False)
|
||||
first_detected = Column(DateTime(timezone=True), default=utc_now, nullable=False)
|
||||
last_updated = Column(DateTime(timezone=True), default=utc_now, nullable=False)
|
||||
extra_info = Column(JSON, nullable=True)
|
||||
|
||||
__table_args__ = (
|
||||
UniqueConstraint("title", "category", name="uq_library_title_category"),
|
||||
Index("ix_library_category", "category"),
|
||||
Index("ix_library_title", "title"),
|
||||
)
|
||||
|
||||
@@ -0,0 +1,413 @@
|
||||
"""Naming and path formatting engine for Media Sorter.
|
||||
|
||||
Renders user-defined naming templates, safely formats multi-part tags, pairs sidecars
|
||||
with primary media files, and enforces rigorous cross-platform filename sanitization
|
||||
(Linux, Windows, macOS, NTFS, SMB/NFS, exFAT).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from .classifier import ClassificationResult
|
||||
from .config import Settings
|
||||
|
||||
# Windows reserved device names
|
||||
RESERVED_NAMES = {
|
||||
"CON", "PRN", "AUX", "NUL",
|
||||
"COM1", "COM2", "COM3", "COM4", "COM5", "COM6", "COM7", "COM8", "COM9",
|
||||
"LPT1", "LPT2", "LPT3", "LPT4", "LPT5", "LPT6", "LPT7", "LPT8", "LPT9",
|
||||
}
|
||||
|
||||
# Illegal characters across file systems (< > : " / \ | ? *)
|
||||
FORBIDDEN_CHARS_PATTERN = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
|
||||
|
||||
|
||||
def sanitize_filename_component(name: str, max_length: int = 240) -> str:
|
||||
"""Sanitize an individual filename or folder name component for safe cross-platform use."""
|
||||
# 1. Unicode normalization (NFC)
|
||||
clean = unicodedata.normalize("NFC", name)
|
||||
|
||||
# 2. Replace forbidden characters with safe hyphen or space
|
||||
clean = FORBIDDEN_CHARS_PATTERN.sub("-", clean)
|
||||
|
||||
# 3. Collapse multiple whitespace and hyphens
|
||||
clean = re.sub(r"\s+", " ", clean)
|
||||
clean = re.sub(r"-{2,}", "-", clean)
|
||||
|
||||
# 4. Strip leading/trailing spaces, dots, and hyphens (vital for Windows / SMB)
|
||||
clean = clean.strip(" .-")
|
||||
|
||||
if not clean:
|
||||
clean = "unnamed"
|
||||
|
||||
# 5. Check Windows reserved words
|
||||
upper_base = clean.split(".")[0].upper()
|
||||
if upper_base in RESERVED_NAMES:
|
||||
clean = f"_{clean}"
|
||||
|
||||
# 6. Truncate byte length for filesystem limits (e.g. 255 bytes on ext4/NTFS/ZFS)
|
||||
encoded = clean.encode("utf-8")
|
||||
if len(encoded) > max_length:
|
||||
parts = clean.rsplit(".", 1)
|
||||
if len(parts) == 2 and 1 <= len(parts[1]) <= 10:
|
||||
base, ext = parts
|
||||
ext_bytes = len(f".{ext}".encode("utf-8"))
|
||||
avail = max(max_length - ext_bytes, 10)
|
||||
base_enc = base.encode("utf-8")[:avail]
|
||||
base_clean = base_enc.decode("utf-8", errors="ignore").rstrip(" .-")
|
||||
clean = f"{base_clean}.{ext}" if base_clean else ext
|
||||
else:
|
||||
clean = encoded[:max_length].decode("utf-8", errors="ignore").rstrip(" .-")
|
||||
|
||||
clean = clean.strip(" .-")
|
||||
return clean if clean else "unnamed"
|
||||
|
||||
|
||||
DEFAULT_TEMPLATES = {
|
||||
"tv": "{title}/Season {season:02d}/{show_name}_{season_episode}.{ext}",
|
||||
"movie": "{title} ({year})/{movie_name}.{ext}",
|
||||
"anime": "{title}/Season {season:02d}/{show_name}_{season_episode} [{group}].{ext}",
|
||||
}
|
||||
|
||||
|
||||
class MediaNamer:
|
||||
"""Renders organized destination paths from templates and classification results."""
|
||||
|
||||
def __init__(self, settings: Settings):
|
||||
self.settings = settings
|
||||
|
||||
def generate_destination_path(
|
||||
self,
|
||||
cls_result: ClassificationResult,
|
||||
primary_dst_path: Optional[Path] = None,
|
||||
) -> Path:
|
||||
"""Construct full destination path for a given file and its classification."""
|
||||
category = cls_result.category
|
||||
base_dir = self.settings.get_destination_path(category)
|
||||
src_path = cls_result.metadata.path if cls_result.metadata else Path("unknown")
|
||||
ext = src_path.suffix.lstrip(".")
|
||||
|
||||
# Handle Quarantine routing
|
||||
if cls_result.needs_quarantine or category == "unknown":
|
||||
reason = cls_result.quarantine_reason or "low_confidence"
|
||||
safe_reason = sanitize_filename_component(reason)
|
||||
q_template = self.settings.templates.quarantine
|
||||
filename = sanitize_filename_component(src_path.name)
|
||||
rel_str = q_template.format(reason=safe_reason, filename=filename, ext=ext)
|
||||
return (self.settings.get_destination_path("quarantine") / rel_str).resolve()
|
||||
|
||||
# Handle Sidecars (Subtitles, Artwork, Metadata, Extras)
|
||||
if category in ("subtitle", "artwork", "metadata"):
|
||||
return self._format_sidecar_path(cls_result, primary_dst_path, base_dir)
|
||||
|
||||
# Retrieve template
|
||||
template = getattr(self.settings.templates, category, None)
|
||||
context = self._build_context(cls_result)
|
||||
|
||||
if category == "tv" and (not template or template == DEFAULT_TEMPLATES.get("tv")):
|
||||
formatted_rel = self._format_tv_path(cls_result, context)
|
||||
elif category == "movie" and (not template or template == DEFAULT_TEMPLATES.get("movie")):
|
||||
formatted_rel = self._format_movie_path(cls_result, context)
|
||||
elif category == "anime" and (not template or template == DEFAULT_TEMPLATES.get("anime")):
|
||||
formatted_rel = self._format_anime_path(cls_result, context)
|
||||
elif category == "podcast" and (not template or template == "{show}/{year}/{show} - {date} - {title}.{ext}"):
|
||||
formatted_rel = self._format_podcast_path(cls_result, context)
|
||||
else:
|
||||
if not template:
|
||||
template = "{filename}.{ext}"
|
||||
formatted_rel = self._render_template(template, context)
|
||||
|
||||
# If file renaming is disabled, preserve original source filename
|
||||
if not getattr(self.settings.general, "rename_files", True) and src_path.name != "unknown":
|
||||
rel_path = Path(formatted_rel)
|
||||
if len(rel_path.parts) > 1:
|
||||
formatted_rel = str(rel_path.parent / src_path.name)
|
||||
else:
|
||||
formatted_rel = src_path.name
|
||||
|
||||
# Sanitize each path component separately to preserve folder hierarchy
|
||||
parts = Path(formatted_rel).parts
|
||||
sanitized_parts = [sanitize_filename_component(p) for p in parts]
|
||||
return (base_dir / Path(*sanitized_parts)).resolve()
|
||||
|
||||
def _format_tv_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
|
||||
show_name = context["show_name"]
|
||||
ext = context["ext"]
|
||||
tokens = cls_result.tokens
|
||||
|
||||
# Daily / dated broadcast TV formatting
|
||||
date_val = context.get("date_val")
|
||||
if (tokens and tokens.is_daily) or (date_val and (not tokens or not tokens.season or tokens.season > 1000)):
|
||||
year = context.get("year")
|
||||
if not year or year == "Unknown":
|
||||
year = date_val.split("-")[0] if date_val else "Unknown"
|
||||
return f"{show_name}/Season {year}/{show_name} - {date_val}.{ext}"
|
||||
|
||||
# Standard TV formatting (supporting Season 00, multi-ep, and season pack)
|
||||
season_num = context["season"]
|
||||
season_folder = f"Season {season_num:02d}"
|
||||
season_episode = context["season_episode"]
|
||||
return f"{show_name}/{season_folder}/{show_name} - {season_episode}.{ext}"
|
||||
|
||||
def _format_movie_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
|
||||
title = context["title"]
|
||||
year = context["year"]
|
||||
ext = context["ext"]
|
||||
has_year = year and year != "Unknown"
|
||||
folder_name = f"{title} ({year})" if has_year else title
|
||||
base_name = f"{title} ({year})" if has_year else title
|
||||
|
||||
edition_tag = context.get("edition_tag", "")
|
||||
part_tag = context.get("part_tag", "")
|
||||
extra_tag = context.get("extra_tag", "")
|
||||
return f"{folder_name}/{base_name}{edition_tag}{part_tag}{extra_tag}.{ext}"
|
||||
|
||||
def _format_anime_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
|
||||
title = context["title"]
|
||||
ext = context["ext"]
|
||||
tokens = cls_result.tokens
|
||||
group_tag = context.get("group_tag", "")
|
||||
|
||||
# Multi-episode anime
|
||||
if tokens and tokens.multi_episodes and len(tokens.multi_episodes) >= 2:
|
||||
first_ep = tokens.multi_episodes[0]
|
||||
last_ep = tokens.multi_episodes[-1]
|
||||
ep_str = f"{first_ep:02d}-{last_ep:02d}"
|
||||
return f"{title}/{title} - {ep_str}{group_tag}.{ext}"
|
||||
|
||||
# Single episode anime
|
||||
if tokens and tokens.episode is not None:
|
||||
ep = tokens.episode
|
||||
ep_str = f"{ep:02d}" if ep < 10 else str(ep)
|
||||
return f"{title}/{title} - {ep_str}{group_tag}.{ext}"
|
||||
|
||||
# Anime movie or special without episode number
|
||||
return f"{title}/{title}{group_tag}.{ext}"
|
||||
|
||||
def _format_podcast_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
|
||||
show = context.get("show") or context.get("artist") or "Unknown Show"
|
||||
year = context.get("year")
|
||||
date = context.get("date")
|
||||
title = context.get("title")
|
||||
ext = context.get("ext")
|
||||
if title and title != show and title != "Unknown":
|
||||
return f"{show}/{year}/{show} - {date} - {title}.{ext}"
|
||||
return f"{show}/{year}/{show} - {date}.{ext}"
|
||||
|
||||
def _format_sidecar_path(
|
||||
self,
|
||||
cls_result: ClassificationResult,
|
||||
primary_dst_path: Optional[Path],
|
||||
base_dir: Path,
|
||||
) -> Path:
|
||||
src_path = cls_result.metadata.path
|
||||
ext = src_path.suffix.lstrip(".")
|
||||
|
||||
if primary_dst_path:
|
||||
parent_dir = primary_dst_path.parent
|
||||
primary_stem = primary_dst_path.stem
|
||||
|
||||
if cls_result.category == "subtitle":
|
||||
# Detect language code or compound tag in subtitle (e.g. movie.en.srt, movie.forced.srt)
|
||||
src_stem = src_path.stem
|
||||
m = re.search(
|
||||
r"\.((?:[a-zA-Z]{2,3}\.)?(?:forced|sdh|cc)|[a-zA-Z]{2,3}(?:-[a-zA-Z]{2,4})?)$",
|
||||
src_stem,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if m:
|
||||
lang_suffix = f".{m.group(1)}"
|
||||
else:
|
||||
parts = src_stem.split(".")
|
||||
if len(parts) > 1 and len(parts[-1]) in (2, 3, 6):
|
||||
lang_suffix = f".{parts[-1]}"
|
||||
else:
|
||||
lang_suffix = ""
|
||||
new_filename = f"{primary_stem}{lang_suffix}.{ext}"
|
||||
return parent_dir / sanitize_filename_component(new_filename)
|
||||
|
||||
elif cls_result.category == "artwork":
|
||||
# e.g. poster.jpg, cover.jpg in the same movie/show folder
|
||||
return parent_dir / sanitize_filename_component(src_path.name)
|
||||
|
||||
elif cls_result.category == "metadata":
|
||||
# NFO file matches primary stem or stays alongside
|
||||
new_filename = f"{primary_stem}.{ext}"
|
||||
return parent_dir / sanitize_filename_component(new_filename)
|
||||
|
||||
# If orphan sidecar (no primary matched), place into respective folder
|
||||
sanitized_name = sanitize_filename_component(src_path.name)
|
||||
return (base_dir / sanitized_name).resolve()
|
||||
|
||||
def _build_context(self, res: ClassificationResult) -> Dict[str, Any]:
|
||||
tokens = res.tokens
|
||||
meta = res.metadata
|
||||
src_path = meta.path if meta else Path("file")
|
||||
|
||||
# Fix Season 00 / Episode 00 falsy bug
|
||||
season_num = tokens.season if (tokens and tokens.season is not None) else 1
|
||||
episode_num = tokens.episode if (tokens and tokens.episode is not None) else 1
|
||||
|
||||
# Format season_episode string with multi-episode and season pack support
|
||||
if tokens and tokens.multi_episodes and len(tokens.multi_episodes) >= 2:
|
||||
season_ep_str = f"S{season_num:02d}E{tokens.multi_episodes[0]:02d}-E{tokens.multi_episodes[-1]:02d}"
|
||||
elif tokens and (tokens.is_season_pack or (tokens.season is not None and tokens.episode is None and not getattr(tokens, "multi_episodes", None))):
|
||||
season_ep_str = f"Season {season_num:02d}"
|
||||
else:
|
||||
season_ep_str = f"S{season_num:02d}E{episode_num:02d}"
|
||||
|
||||
main_title = (tokens.title if tokens else None) or src_path.stem
|
||||
|
||||
# Clean release group: omit when unknown, NEVER emit 'UnknownGroup'
|
||||
group_val = tokens.group if (tokens and tokens.group and tokens.group != "UnknownGroup") else ""
|
||||
group_tag = f" [{group_val}]" if group_val else ""
|
||||
|
||||
# Extract movie edition, part, and extra tags
|
||||
edition_val = getattr(tokens, "edition", None) if tokens else None
|
||||
if not edition_val:
|
||||
em = re.search(r"\b(extended|directors?\.cut|remastered|criterion(?:\.collection)?|final\.cut)\b", src_path.stem, re.I)
|
||||
if em:
|
||||
raw_ed = em.group(1).lower().replace(".", " ")
|
||||
if "director" in raw_ed:
|
||||
edition_val = "Director's Cut"
|
||||
elif "criterion" in raw_ed:
|
||||
edition_val = "Criterion"
|
||||
elif "final" in raw_ed:
|
||||
edition_val = "Final Cut"
|
||||
elif "remaster" in raw_ed:
|
||||
edition_val = "Remastered"
|
||||
elif "extend" in raw_ed:
|
||||
edition_val = "Extended"
|
||||
|
||||
edition_tag = f" [{edition_val}]" if edition_val else ""
|
||||
|
||||
part_val = getattr(tokens, "part", None) if tokens else None
|
||||
part_label = getattr(tokens, "part_label", None) if tokens else None
|
||||
if part_val is None:
|
||||
pm = re.search(r"\b(?:cd|part|pt)[\.\s_-]*(\d+)\b", src_path.stem, re.I)
|
||||
if pm:
|
||||
part_val = int(pm.group(1))
|
||||
part_label = f"Pt.{part_val}"
|
||||
elif not part_label:
|
||||
part_label = f"Pt.{part_val}"
|
||||
|
||||
part_tag = f" [{part_label}]" if part_label else ""
|
||||
|
||||
extra_m = re.search(r"-(behindthescenes|deleted|trailer|featurette)\b", src_path.stem, re.I)
|
||||
extra_tag = f"-{extra_m.group(1).lower()}" if extra_m else ""
|
||||
|
||||
# Date resolution for daily TV shows and podcasts
|
||||
date_val = getattr(tokens, "air_date", None) or (tokens.date_stamp if tokens and not tokens.is_photo_or_home_video else None)
|
||||
if not date_val:
|
||||
dm = re.search(r"\b((?:19|20)\d{2})[-._](0[1-9]|1[0-2])[-._](0[1-9]|[12]\d|3[01])\b", src_path.stem)
|
||||
if dm:
|
||||
date_val = f"{dm.group(1)}-{dm.group(2)}-{dm.group(3)}"
|
||||
|
||||
ctx: Dict[str, Any] = {
|
||||
"ext": src_path.suffix.lstrip("."),
|
||||
"filename": src_path.stem,
|
||||
"title": main_title,
|
||||
"show_name": main_title,
|
||||
"SHOW_NAME": main_title,
|
||||
"movie_name": main_title,
|
||||
"MOVIE_NAME": main_title,
|
||||
"season_episode": season_ep_str,
|
||||
"SEASON_EPISODE": season_ep_str,
|
||||
"year": (tokens.year if tokens else None) or "Unknown",
|
||||
"season": season_num,
|
||||
"episode": episode_num,
|
||||
"episode_title": (tokens.episode_title if tokens else None) or f"Episode {episode_num}",
|
||||
"artist": (tokens.artist if tokens else None) or "Unknown Artist",
|
||||
"album": (tokens.album if tokens else None) or "Unknown Album",
|
||||
"track": (tokens.track if tokens else 1) or 1,
|
||||
"disc": (tokens.disc if tokens else 1) or 1,
|
||||
"group": group_val,
|
||||
"group_tag": group_tag,
|
||||
"edition": edition_val,
|
||||
"edition_tag": edition_tag,
|
||||
"part": part_val,
|
||||
"part_label": part_label,
|
||||
"part_tag": part_tag,
|
||||
"extra_tag": extra_tag,
|
||||
"date_val": date_val,
|
||||
"resolution": (tokens.resolution if tokens and tokens.resolution else (meta.resolution_label if meta else "")),
|
||||
"codec": (tokens.video_codec or (meta.codec_video if meta else "h264")),
|
||||
"author": (tokens.artist if tokens else None) or "Unknown Author",
|
||||
"chapter": (tokens.title if tokens else None) or f"Chapter {tokens.track if tokens else 1}",
|
||||
"show": (tokens.artist if tokens else None) or "Unknown Show",
|
||||
"date": date_val or ((tokens.date_stamp if tokens else "2026-01-01") or "2026-01-01"),
|
||||
"month": 1,
|
||||
"day": 1,
|
||||
"time": "000000",
|
||||
"camera": "Camera",
|
||||
"event": "Event",
|
||||
}
|
||||
|
||||
# Override from provider result if available
|
||||
if res.provider_result:
|
||||
p = res.provider_result
|
||||
if p.canonical_title:
|
||||
ctx["title"] = p.canonical_title
|
||||
ctx["show_name"] = p.canonical_title
|
||||
ctx["SHOW_NAME"] = p.canonical_title
|
||||
ctx["movie_name"] = p.canonical_title
|
||||
ctx["MOVIE_NAME"] = p.canonical_title
|
||||
if p.year:
|
||||
ctx["year"] = p.year
|
||||
if p.episode_title:
|
||||
ctx["episode_title"] = p.episode_title
|
||||
if p.artist:
|
||||
ctx["artist"] = p.artist
|
||||
if p.album:
|
||||
ctx["album"] = p.album
|
||||
|
||||
# Parse date stamp fields if present
|
||||
date_source = (tokens and tokens.date_stamp) or (ctx.get("date") if res.category in ("home_video", "photo", "podcast") else None)
|
||||
if date_source and date_source != "Unknown":
|
||||
date_parts = str(date_source).split("-")
|
||||
if len(date_parts) == 3:
|
||||
try:
|
||||
if ctx["year"] == "Unknown":
|
||||
ctx["year"] = int(date_parts[0])
|
||||
ctx["month"] = int(date_parts[1])
|
||||
ctx["day"] = int(date_parts[2])
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
# Clean tags from meta
|
||||
if meta and meta.tags:
|
||||
if "camera_model" in meta.tags:
|
||||
ctx["camera"] = sanitize_filename_component(meta.tags["camera_model"])
|
||||
if "album" in meta.tags and not ctx.get("album"):
|
||||
ctx["album"] = meta.tags["album"]
|
||||
if "artist" in meta.tags and not ctx.get("artist"):
|
||||
ctx["artist"] = meta.tags["artist"]
|
||||
|
||||
return ctx
|
||||
|
||||
def _render_template(self, template: str, context: Dict[str, Any]) -> str:
|
||||
"""Format template while gracefully cleaning empty technical brackets."""
|
||||
# Normalize <TAG> to {TAG} for convenience if users use angle brackets
|
||||
rendered = re.sub(r"<([a-zA-Z_0-9]+)>", r"{\1}", template)
|
||||
try:
|
||||
rendered = rendered.format(**context)
|
||||
except (KeyError, ValueError):
|
||||
# Safe token replacement if format specifier fails
|
||||
safe_ctx = {k: str(v) if v is not None else "" for k, v in context.items()}
|
||||
# Remove format specifiers like :02d
|
||||
simplified = re.sub(r"\{(\w+):[^}]+\}", r"{\1}", rendered)
|
||||
try:
|
||||
rendered = simplified.format(**safe_ctx)
|
||||
except Exception:
|
||||
rendered = f"{context.get('title', 'media')}.{context.get('ext', 'bin')}"
|
||||
|
||||
# Clean empty technical brackets such as "[]" or "[ ]" or "()"
|
||||
rendered = re.sub(r"\[\s*\]", "", rendered)
|
||||
rendered = re.sub(r"\(\s*\)", "", rendered)
|
||||
rendered = re.sub(r"\s{2,}", " ", rendered)
|
||||
return rendered.strip()
|
||||
@@ -0,0 +1,67 @@
|
||||
"""Notification dispatcher for Media Sorter.
|
||||
|
||||
Sends batch summary alerts and error notices to webhook endpoints (Slack, Discord, generic JSON).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
import requests
|
||||
import structlog
|
||||
|
||||
from .config import NotificationSettings
|
||||
from .executor import BatchExecutionReport
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
|
||||
def send_batch_notification(
|
||||
settings: NotificationSettings, report: BatchExecutionReport
|
||||
) -> bool:
|
||||
"""Send webhook alert for completed or failed batch."""
|
||||
if not settings.enabled or not settings.webhook_url:
|
||||
return False
|
||||
|
||||
is_failure = report.failed_files > 0
|
||||
if is_failure and not settings.notify_on_failure:
|
||||
return False
|
||||
if not is_failure and not settings.notify_on_complete:
|
||||
return False
|
||||
|
||||
status_str = "FAILED" if is_failure else ("DRY RUN PREVIEW" if report.dry_run else "SUCCESS")
|
||||
title = f"Media Sorter: {status_str} [Batch {report.batch_id[:8]}]"
|
||||
|
||||
summary_text = (
|
||||
f"**{title}**\n"
|
||||
f"• Total Files: {report.total_files}\n"
|
||||
f"• Organized/Moved: {report.moved_files}\n"
|
||||
f"• Copied: {report.copied_files}\n"
|
||||
f"• Skipped: {report.skipped_files}\n"
|
||||
f"• Quarantined: {report.quarantined_files}\n"
|
||||
f"• Failures: {report.failed_files}\n"
|
||||
)
|
||||
if report.errors:
|
||||
summary_text += f"\nErrors:\n" + "\n".join(f"- {e}" for e in report.errors[:5])
|
||||
|
||||
payload: Dict[str, Any] = {
|
||||
"text": summary_text,
|
||||
"content": summary_text, # Discord format compatibility
|
||||
"batch_id": report.batch_id,
|
||||
"dry_run": report.dry_run,
|
||||
"status": status_str,
|
||||
"total_files": report.total_files,
|
||||
"moved_files": report.moved_files,
|
||||
"quarantined_files": report.quarantined_files,
|
||||
"failed_files": report.failed_files,
|
||||
}
|
||||
|
||||
try:
|
||||
resp = requests.post(settings.webhook_url, json=payload, timeout=5.0)
|
||||
if resp.status_code in (200, 201, 204):
|
||||
return True
|
||||
logger.warning("Webhook dispatch failed", status=resp.status_code)
|
||||
except Exception as e:
|
||||
logger.warning("Failed sending notification webhook", error=str(e))
|
||||
|
||||
return False
|
||||
@@ -0,0 +1,255 @@
|
||||
"""Metadata provider integrations for Media Sorter.
|
||||
|
||||
Provides interfaces and implementations for querying external metadata (TMDB, TVDB,
|
||||
MusicBrainz) with rate-limiting, request caching, and offline fallbacks.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import time
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
import requests
|
||||
import structlog
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ProviderResult:
|
||||
canonical_title: str
|
||||
year: Optional[int] = None
|
||||
media_type: str = "movie" # movie, tv, anime, music
|
||||
season: Optional[int] = None
|
||||
episode: Optional[int] = None
|
||||
episode_title: Optional[str] = None
|
||||
artist: Optional[str] = None
|
||||
album: Optional[str] = None
|
||||
genres: List[str] = None
|
||||
confidence_boost: float = 0.15
|
||||
raw_payload: Optional[Dict[str, Any]] = None
|
||||
|
||||
|
||||
class MetadataProvider(ABC):
|
||||
"""Abstract base class for all metadata providers."""
|
||||
|
||||
@abstractmethod
|
||||
def search_movie(self, title: str, year: Optional[int] = None) -> Optional[ProviderResult]:
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def search_tv(self, title: str, year: Optional[int] = None, season: Optional[int] = None, episode: Optional[int] = None) -> Optional[ProviderResult]:
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def search_music(self, artist: str, album: Optional[str] = None, title: Optional[str] = None) -> Optional[ProviderResult]:
|
||||
pass
|
||||
|
||||
|
||||
class MemoryCache:
|
||||
"""In-memory cache with TTL for metadata queries."""
|
||||
|
||||
def __init__(self, ttl_seconds: int = 86400):
|
||||
self.ttl = ttl_seconds
|
||||
self._store: Dict[str, Tuple[float, Any]] = {}
|
||||
|
||||
def get(self, key: str) -> Optional[Any]:
|
||||
if key in self._store:
|
||||
timestamp, data = self._store[key]
|
||||
if time.time() - timestamp < self.ttl:
|
||||
return data
|
||||
del self._store[key]
|
||||
return None
|
||||
|
||||
def set(self, key: str, data: Any) -> None:
|
||||
self._store[key] = (time.time(), data)
|
||||
|
||||
|
||||
class TMDBProvider(MetadataProvider):
|
||||
"""TheMovieDatabase (TMDB) API provider with rate-limiting and caching."""
|
||||
|
||||
BASE_URL = "https://api.themoviedb.org/3"
|
||||
|
||||
def __init__(self, api_key: Optional[str] = None, rate_limit_per_second: float = 2.0, cache_ttl_seconds: int = 86400):
|
||||
self.api_key = api_key
|
||||
self.min_interval = 1.0 / max(rate_limit_per_second, 0.1)
|
||||
self.last_request_time = 0.0
|
||||
self.cache = MemoryCache(ttl_seconds=cache_ttl_seconds)
|
||||
|
||||
def _throttle(self) -> None:
|
||||
elapsed = time.time() - self.last_request_time
|
||||
if elapsed < self.min_interval:
|
||||
time.sleep(self.min_interval - elapsed)
|
||||
self.last_request_time = time.time()
|
||||
|
||||
def _query(self, endpoint: str, params: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
if not self.api_key:
|
||||
return None
|
||||
|
||||
cache_key = f"tmdb:{endpoint}:{json.dumps(params, sort_keys=True)}"
|
||||
cached = self.cache.get(cache_key)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
self._throttle()
|
||||
req_params = dict(params)
|
||||
req_params["api_key"] = self.api_key
|
||||
|
||||
try:
|
||||
resp = requests.get(f"{self.BASE_URL}/{endpoint}", params=req_params, timeout=5.0)
|
||||
if resp.status_code == 200:
|
||||
data = resp.json()
|
||||
self.cache.set(cache_key, data)
|
||||
return data
|
||||
logger.warning("TMDB request failed", status=resp.status_code, endpoint=endpoint)
|
||||
except Exception as e:
|
||||
logger.warning("TMDB network error", error=str(e))
|
||||
return None
|
||||
|
||||
def search_movie(self, title: str, year: Optional[int] = None) -> Optional[ProviderResult]:
|
||||
params: Dict[str, Any] = {"query": title}
|
||||
if year:
|
||||
params["year"] = year
|
||||
|
||||
data = self._query("search/movie", params)
|
||||
if not data or not data.get("results"):
|
||||
return None
|
||||
|
||||
first = data["results"][0]
|
||||
release_date = first.get("release_date", "")
|
||||
res_year = int(release_date[:4]) if len(release_date) >= 4 and release_date[:4].isdigit() else year
|
||||
|
||||
return ProviderResult(
|
||||
canonical_title=first.get("title", title),
|
||||
year=res_year,
|
||||
media_type="movie",
|
||||
confidence_boost=0.15,
|
||||
raw_payload=first,
|
||||
)
|
||||
|
||||
def search_tv(self, title: str, year: Optional[int] = None, season: Optional[int] = None, episode: Optional[int] = None) -> Optional[ProviderResult]:
|
||||
params: Dict[str, Any] = {"query": title}
|
||||
if year:
|
||||
params["first_air_date_year"] = year
|
||||
|
||||
data = self._query("search/tv", params)
|
||||
if not data or not data.get("results"):
|
||||
return None
|
||||
|
||||
first = data["results"][0]
|
||||
show_id = first.get("id")
|
||||
show_title = first.get("name", title)
|
||||
air_date = first.get("first_air_date", "")
|
||||
res_year = int(air_date[:4]) if len(air_date) >= 4 and air_date[:4].isdigit() else year
|
||||
|
||||
ep_title = None
|
||||
if show_id and season is not None and episode is not None:
|
||||
ep_data = self._query(f"tv/{show_id}/season/{season}/episode/{episode}", {})
|
||||
if ep_data:
|
||||
ep_title = ep_data.get("name")
|
||||
|
||||
return ProviderResult(
|
||||
canonical_title=show_title,
|
||||
year=res_year,
|
||||
media_type="tv",
|
||||
season=season,
|
||||
episode=episode,
|
||||
episode_title=ep_title,
|
||||
confidence_boost=0.20,
|
||||
raw_payload=first,
|
||||
)
|
||||
|
||||
def search_music(self, artist: str, album: Optional[str] = None, title: Optional[str] = None) -> Optional[ProviderResult]:
|
||||
return None # TMDB does not index music
|
||||
|
||||
|
||||
class MusicBrainzProvider(MetadataProvider):
|
||||
"""MusicBrainz WS2 API provider with courteous rate-limiting (1 req/sec)."""
|
||||
|
||||
BASE_URL = "https://musicbrainz.org/ws/2"
|
||||
|
||||
def __init__(self, rate_limit_per_second: float = 1.0, cache_ttl_seconds: int = 86400):
|
||||
self.min_interval = 1.0 / max(rate_limit_per_second, 0.1)
|
||||
self.last_request_time = 0.0
|
||||
self.cache = MemoryCache(ttl_seconds=cache_ttl_seconds)
|
||||
|
||||
def _throttle(self) -> None:
|
||||
elapsed = time.time() - self.last_request_time
|
||||
if elapsed < self.min_interval:
|
||||
time.sleep(self.min_interval - elapsed)
|
||||
self.last_request_time = time.time()
|
||||
|
||||
def search_movie(self, title: str, year: Optional[int] = None) -> Optional[ProviderResult]:
|
||||
return None
|
||||
|
||||
def search_tv(self, title: str, year: Optional[int] = None, season: Optional[int] = None, episode: Optional[int] = None) -> Optional[ProviderResult]:
|
||||
return None
|
||||
|
||||
def search_music(self, artist: str, album: Optional[str] = None, title: Optional[str] = None) -> Optional[ProviderResult]:
|
||||
query_parts = [f'artist:"{artist}"']
|
||||
if album:
|
||||
query_parts.append(f'release:"{album}"')
|
||||
if title:
|
||||
query_parts.append(f'recording:"{title}"')
|
||||
|
||||
query_str = " AND ".join(query_parts)
|
||||
cache_key = f"mb:{query_str}"
|
||||
cached = self.cache.get(cache_key)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
self._throttle()
|
||||
headers = {"User-Agent": "MediaSorter/0.1.0 (https://github.com/example/media-sorter)"}
|
||||
params = {"query": query_str, "fmt": "json", "limit": 1}
|
||||
|
||||
try:
|
||||
resp = requests.get(f"{self.BASE_URL}/recording", params=params, headers=headers, timeout=5.0)
|
||||
if resp.status_code == 200:
|
||||
data = resp.json()
|
||||
recordings = data.get("recordings", [])
|
||||
if recordings:
|
||||
rec = recordings[0]
|
||||
rec_title = rec.get("title", title or "")
|
||||
# Extract release info
|
||||
releases = rec.get("releases", [])
|
||||
rec_album = releases[0].get("title", album) if releases else album
|
||||
release_date = releases[0].get("date", "") if releases else ""
|
||||
res_year = int(release_date[:4]) if len(release_date) >= 4 and release_date[:4].isdigit() else None
|
||||
|
||||
res = ProviderResult(
|
||||
canonical_title=rec_title,
|
||||
artist=artist,
|
||||
album=rec_album,
|
||||
year=res_year,
|
||||
media_type="music",
|
||||
confidence_boost=0.15,
|
||||
raw_payload=rec,
|
||||
)
|
||||
self.cache.set(cache_key, res)
|
||||
return res
|
||||
except Exception as e:
|
||||
logger.warning("MusicBrainz network error", error=str(e))
|
||||
return None
|
||||
|
||||
|
||||
class MockMetadataProvider(MetadataProvider):
|
||||
"""Deterministic mock provider for offline testing and fixture validation."""
|
||||
|
||||
def __init__(self, mock_data: Optional[Dict[str, ProviderResult]] = None):
|
||||
self.mock_data = mock_data or {}
|
||||
|
||||
def search_movie(self, title: str, year: Optional[int] = None) -> Optional[ProviderResult]:
|
||||
key = f"movie:{title.lower()}"
|
||||
return self.mock_data.get(key)
|
||||
|
||||
def search_tv(self, title: str, year: Optional[int] = None, season: Optional[int] = None, episode: Optional[int] = None) -> Optional[ProviderResult]:
|
||||
key = f"tv:{title.lower()}"
|
||||
return self.mock_data.get(key)
|
||||
|
||||
def search_music(self, artist: str, album: Optional[str] = None, title: Optional[str] = None) -> Optional[ProviderResult]:
|
||||
key = f"music:{artist.lower()}"
|
||||
return self.mock_data.get(key)
|
||||
@@ -0,0 +1,141 @@
|
||||
"""Quarantine and manual review management for Media Sorter.
|
||||
|
||||
Provides querying, manual override, reprocessing, and resolution tracking for files
|
||||
that could not be safely or confidently organized automatically.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import structlog
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from .models import QuarantineRecord, QuarantineStatus
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
|
||||
class QuarantineManager:
|
||||
"""Manages files held in quarantine or review status."""
|
||||
|
||||
def __init__(self, session: Session):
|
||||
self.session = session
|
||||
|
||||
def list_pending(self) -> List[QuarantineRecord]:
|
||||
"""Return all quarantine items waiting for human inspection."""
|
||||
return (
|
||||
self.session.query(QuarantineRecord)
|
||||
.filter_by(status=QuarantineStatus.PENDING.value)
|
||||
.order_by(QuarantineRecord.created_at.desc())
|
||||
.all()
|
||||
)
|
||||
|
||||
def get_by_id(self, item_id: int) -> Optional[QuarantineRecord]:
|
||||
"""Fetch quarantine record by ID."""
|
||||
return self.session.query(QuarantineRecord).filter_by(id=item_id).first()
|
||||
|
||||
def resolve_item(
|
||||
self,
|
||||
item_id: int,
|
||||
resolved_category: str,
|
||||
target_path: Optional[Path | str] = None,
|
||||
) -> bool:
|
||||
"""Resolve a quarantined item by specifying human-approved category and target path."""
|
||||
rec = self.get_by_id(item_id)
|
||||
if not rec:
|
||||
return False
|
||||
|
||||
rec.status = QuarantineStatus.RESOLVED.value
|
||||
rec.suggested_category = resolved_category
|
||||
rec.resolved_at = datetime.now(timezone.utc)
|
||||
if target_path:
|
||||
rec.resolved_path = str(target_path)
|
||||
|
||||
self.session.commit()
|
||||
logger.info("Quarantine item resolved", item_id=item_id, category=resolved_category)
|
||||
return True
|
||||
|
||||
def ignore_item(self, item_id: int) -> bool:
|
||||
"""Mark quarantine item as ignored."""
|
||||
rec = self.get_by_id(item_id)
|
||||
if not rec:
|
||||
return False
|
||||
|
||||
rec.status = QuarantineStatus.IGNORED.value
|
||||
rec.resolved_at = datetime.now(timezone.utc)
|
||||
self.session.commit()
|
||||
return True
|
||||
|
||||
def undo_item(self, item_id: int) -> bool:
|
||||
"""Undo the resolution or quarantine status of an item.
|
||||
|
||||
If resolved and file was moved, moves the file back to its original src location
|
||||
and resets status to PENDING. If already pending, unflags/removes from quarantine.
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
rec = self.get_by_id(item_id)
|
||||
if not rec:
|
||||
return False
|
||||
|
||||
if rec.status == QuarantineStatus.RESOLVED.value and rec.resolved_path:
|
||||
dst_path = Path(rec.resolved_path)
|
||||
src_path = Path(rec.src)
|
||||
if dst_path.exists():
|
||||
src_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
try:
|
||||
shutil.move(dst_path, src_path)
|
||||
logger.info("Restored resolved quarantine file back to src", src=str(src_path), dst=str(dst_path))
|
||||
except Exception as e:
|
||||
logger.error("Failed restoring quarantine file to src", src=str(src_path), dst=str(dst_path), error=str(e))
|
||||
rec.status = QuarantineStatus.PENDING.value
|
||||
rec.resolved_path = None
|
||||
rec.resolved_at = None
|
||||
self.session.commit()
|
||||
return True
|
||||
elif rec.status == QuarantineStatus.PENDING.value:
|
||||
# Unflag pending quarantine item
|
||||
self.session.delete(rec)
|
||||
self.session.commit()
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def list_resolved(self, limit: int = 50) -> List[QuarantineRecord]:
|
||||
"""Return recently resolved quarantine items."""
|
||||
return (
|
||||
self.session.query(QuarantineRecord)
|
||||
.filter_by(status=QuarantineStatus.RESOLVED.value)
|
||||
.order_by(QuarantineRecord.resolved_at.desc())
|
||||
.limit(limit)
|
||||
.all()
|
||||
)
|
||||
|
||||
def get_statistics(self) -> Dict[str, int]:
|
||||
"""Summarize quarantine records by status."""
|
||||
total = self.session.query(QuarantineRecord).count()
|
||||
pending = (
|
||||
self.session.query(QuarantineRecord)
|
||||
.filter_by(status=QuarantineStatus.PENDING.value)
|
||||
.count()
|
||||
)
|
||||
resolved = (
|
||||
self.session.query(QuarantineRecord)
|
||||
.filter_by(status=QuarantineStatus.RESOLVED.value)
|
||||
.count()
|
||||
)
|
||||
ignored = (
|
||||
self.session.query(QuarantineRecord)
|
||||
.filter_by(status=QuarantineStatus.IGNORED.value)
|
||||
.count()
|
||||
)
|
||||
|
||||
return {
|
||||
"total": total,
|
||||
"pending": pending,
|
||||
"resolved": resolved,
|
||||
"ignored": ignored,
|
||||
}
|
||||
@@ -0,0 +1,258 @@
|
||||
"""Filesystem discovery and scanning engine for Media Sorter.
|
||||
|
||||
Efficiently traverses directories, enforces minimum file age checks (to prevent
|
||||
processing files currently being written/downloaded), tests file locks, and detects
|
||||
companion/sidecar files.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import fnmatch
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Dict, Generator, List, Optional, Set, Tuple
|
||||
|
||||
import structlog
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
# Known sidecar and companion extensions
|
||||
SUBTITLE_EXTS = {".srt", ".ass", ".ssa", ".vtt", ".sub", ".idx"}
|
||||
ARTWORK_NAMES = {"poster", "cover", "folder", "fanart", "banner", "clearart", "disc", "logo"}
|
||||
ARTWORK_EXTS = {".jpg", ".jpeg", ".png", ".webp", ".tbn"}
|
||||
METADATA_EXTS = {".nfo", ".xml", ".json"}
|
||||
EXTRA_TAGS = {"-trailer", "-sample", "-featurette", "-behindthescenes", "-deleted", "-short"}
|
||||
|
||||
|
||||
@dataclass
|
||||
class ScannedFile:
|
||||
path: Path
|
||||
size: int
|
||||
mtime: float
|
||||
is_sidecar: bool = False
|
||||
sidecar_type: Optional[str] = None # "subtitle", "artwork", "metadata", "extra"
|
||||
primary_media_path: Optional[Path] = None
|
||||
tags: Dict[str, str] = field(default_factory=dict)
|
||||
|
||||
|
||||
def is_file_locked(path: Path) -> bool:
|
||||
"""Test whether a file is currently open/locked for writing by another process."""
|
||||
if not path.is_file():
|
||||
return False
|
||||
|
||||
try:
|
||||
# On Windows, try opening with exclusive read/write sharing if possible
|
||||
if sys.platform == "win32":
|
||||
import msvcrt
|
||||
handle = open(path, "rb")
|
||||
try:
|
||||
msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1)
|
||||
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
|
||||
finally:
|
||||
handle.close()
|
||||
else:
|
||||
import fcntl
|
||||
with open(path, "rb") as f:
|
||||
fcntl.flock(f.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
fcntl.flock(f.fileno(), fcntl.LOCK_UN)
|
||||
return False
|
||||
except (IOError, OSError, PermissionError):
|
||||
return True
|
||||
|
||||
|
||||
class Scanner:
|
||||
def __init__(
|
||||
self,
|
||||
min_file_age_seconds: int = 300,
|
||||
include_patterns: Optional[List[str]] = None,
|
||||
exclude_patterns: Optional[List[str]] = None,
|
||||
):
|
||||
self.min_file_age_seconds = min_file_age_seconds
|
||||
self.include_patterns = include_patterns or ["*"]
|
||||
self.exclude_patterns = exclude_patterns or [
|
||||
".*",
|
||||
"*.part",
|
||||
"*.crdownload",
|
||||
"*.!qB",
|
||||
"Thumbs.db",
|
||||
"desktop.ini",
|
||||
"@eaDir",
|
||||
"$RECYCLE.BIN",
|
||||
"*.txt",
|
||||
]
|
||||
|
||||
def _matches_filter(self, filename: str) -> bool:
|
||||
"""Check whether filename matches includes and does not match excludes."""
|
||||
for pattern in self.exclude_patterns:
|
||||
if fnmatch.fnmatch(filename, pattern):
|
||||
return False
|
||||
if fnmatch.fnmatch(filename.lower(), pattern.lower()):
|
||||
return False
|
||||
|
||||
if not self.include_patterns or "*" in self.include_patterns:
|
||||
return True
|
||||
|
||||
for pattern in self.include_patterns:
|
||||
if fnmatch.fnmatch(filename, pattern) or fnmatch.fnmatch(filename.lower(), pattern.lower()):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
def scan_directory(self, root_dir: Path | str) -> List[ScannedFile]:
|
||||
"""Recursively scan a directory returning all qualifying files."""
|
||||
root = Path(root_dir).resolve()
|
||||
if not root.exists():
|
||||
logger.warning("Source directory does not exist", directory=str(root))
|
||||
return []
|
||||
|
||||
now = time.time()
|
||||
discovered: List[ScannedFile] = []
|
||||
video_candidates: List[ScannedFile] = []
|
||||
potential_sidecars: List[ScannedFile] = []
|
||||
|
||||
for entry_path in self._walk_safe(root):
|
||||
try:
|
||||
stat = entry_path.stat()
|
||||
except (OSError, PermissionError) as e:
|
||||
logger.warning("Skipping inaccessible file", path=str(entry_path), error=str(e))
|
||||
continue
|
||||
|
||||
# Minimum file age check: ignore recently modified files (e.g. active downloads)
|
||||
file_age = now - stat.st_mtime
|
||||
if file_age < self.min_file_age_seconds:
|
||||
logger.debug(
|
||||
"Skipping file: modified too recently",
|
||||
path=str(entry_path),
|
||||
age_seconds=int(file_age),
|
||||
min_age_seconds=self.min_file_age_seconds,
|
||||
)
|
||||
continue
|
||||
|
||||
# Check if locked
|
||||
if is_file_locked(entry_path):
|
||||
logger.debug("Skipping file: currently locked by another process", path=str(entry_path))
|
||||
continue
|
||||
|
||||
ext = entry_path.suffix.lower()
|
||||
stem = entry_path.stem.lower()
|
||||
|
||||
scanned = ScannedFile(
|
||||
path=entry_path,
|
||||
size=stat.st_size,
|
||||
mtime=stat.st_mtime,
|
||||
)
|
||||
|
||||
# Classify sidecar vs primary candidate
|
||||
if ext in SUBTITLE_EXTS:
|
||||
scanned.is_sidecar = True
|
||||
scanned.sidecar_type = "subtitle"
|
||||
potential_sidecars.append(scanned)
|
||||
elif ext in ARTWORK_EXTS and any(stem == art or stem.startswith(f"{art}.") for art in ARTWORK_NAMES):
|
||||
scanned.is_sidecar = True
|
||||
scanned.sidecar_type = "artwork"
|
||||
potential_sidecars.append(scanned)
|
||||
elif ext in METADATA_EXTS:
|
||||
scanned.is_sidecar = True
|
||||
scanned.sidecar_type = "metadata"
|
||||
potential_sidecars.append(scanned)
|
||||
elif any(stem.endswith(tag) for tag in EXTRA_TAGS):
|
||||
scanned.is_sidecar = True
|
||||
scanned.sidecar_type = "extra"
|
||||
potential_sidecars.append(scanned)
|
||||
else:
|
||||
discovered.append(scanned)
|
||||
if ext in {".mp4", ".mkv", ".m4v", ".avi", ".mov", ".ts", ".webm"}:
|
||||
video_candidates.append(scanned)
|
||||
|
||||
# Pair sidecars with primary files in the same directory
|
||||
self._pair_sidecars(potential_sidecars, video_candidates, discovered)
|
||||
return discovered
|
||||
|
||||
def _walk_safe(self, root: Path) -> Generator[Path, None, None]:
|
||||
"""Safely traverse directories using os.scandir with cycle and permission handling."""
|
||||
visited_inodes: Set[Tuple[int, int]] = set()
|
||||
stack = [root]
|
||||
|
||||
while stack:
|
||||
curr = stack.pop()
|
||||
try:
|
||||
with os.scandir(curr) as it:
|
||||
for entry in it:
|
||||
try:
|
||||
# Avoid symlink loops
|
||||
if entry.is_symlink():
|
||||
continue
|
||||
|
||||
if entry.is_dir():
|
||||
if not self._matches_filter(entry.name):
|
||||
continue
|
||||
stat = entry.stat()
|
||||
dev_ino = (stat.st_dev, stat.st_ino)
|
||||
if dev_ino in visited_inodes:
|
||||
continue
|
||||
visited_inodes.add(dev_ino)
|
||||
stack.append(Path(entry.path))
|
||||
elif entry.is_file():
|
||||
if self._matches_filter(entry.name):
|
||||
yield Path(entry.path)
|
||||
except (OSError, PermissionError) as e:
|
||||
logger.debug("Failed reading entry", path=entry.path, error=str(e))
|
||||
except (OSError, PermissionError) as e:
|
||||
logger.warning("Failed traversing directory", path=str(curr), error=str(e))
|
||||
|
||||
def _pair_sidecars(
|
||||
self,
|
||||
sidecars: List[ScannedFile],
|
||||
primaries: List[ScannedFile],
|
||||
all_discovered: List[ScannedFile],
|
||||
) -> None:
|
||||
"""Associate sidecar files (subtitles, artwork, nfo) with primary media files."""
|
||||
# Index primaries by parent dir
|
||||
primaries_by_dir: Dict[Path, List[ScannedFile]] = {}
|
||||
for p in primaries:
|
||||
primaries_by_dir.setdefault(p.path.parent.resolve(), []).append(p)
|
||||
|
||||
for s in sidecars:
|
||||
parent = s.path.parent.resolve()
|
||||
candidates = primaries_by_dir.get(parent, [])
|
||||
|
||||
# Support subdirectories like Subs/ or Subtitles/
|
||||
if not candidates and parent.name.lower() in ("subs", "subtitles", "sub"):
|
||||
parent = parent.parent
|
||||
candidates = primaries_by_dir.get(parent, [])
|
||||
|
||||
matched_primary = None
|
||||
s_stem = s.path.stem.lower()
|
||||
|
||||
# Sort candidates longest stem first so "Movie.Part2" matches before "Movie"
|
||||
sorted_candidates = sorted(candidates, key=lambda c: len(c.path.stem), reverse=True)
|
||||
|
||||
for c in sorted_candidates:
|
||||
c_stem = c.path.stem.lower()
|
||||
if s_stem == c_stem:
|
||||
matched_primary = c
|
||||
break
|
||||
# Check delimiter boundary: must be followed by '.', '-', '_', or ' '
|
||||
if s_stem.startswith(c_stem) and len(s_stem) > len(c_stem):
|
||||
next_char = s_stem[len(c_stem)]
|
||||
if next_char in (".", "-", "_", " "):
|
||||
matched_primary = c
|
||||
break
|
||||
|
||||
# Fallback for single-video directories with generic sidecars (movie.nfo, poster.jpg, en.srt)
|
||||
if not matched_primary and len(candidates) == 1:
|
||||
if (
|
||||
s.sidecar_type in ("metadata", "artwork")
|
||||
or s.path.suffix.lower() in METADATA_EXTS
|
||||
or s.path.suffix.lower() in SUBTITLE_EXTS
|
||||
or s.path.suffix.lower() in ARTWORK_EXTS
|
||||
):
|
||||
matched_primary = candidates[0]
|
||||
|
||||
if matched_primary:
|
||||
s.primary_media_path = matched_primary.path
|
||||
|
||||
all_discovered.append(s)
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,323 @@
|
||||
"""Main orchestration pipeline for Media Sorter.
|
||||
|
||||
Coordinates filesystem scanning, caching, parallel metadata probing, classification,
|
||||
naming, dry-run previews, atomic execution, and audit logging.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import concurrent.futures
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Callable, Dict, List, Optional, Tuple
|
||||
|
||||
import structlog
|
||||
from sqlalchemy.engine import Engine
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from .analyzer import MediaAnalyzer, MediaMetadata
|
||||
from .classifier import ClassificationResult, MediaClassifier
|
||||
from .config import ActionType, Settings
|
||||
from .db import get_db_session
|
||||
from .executor import BatchExecutionReport, MediaExecutor, PlannedOperation, acquire_process_lock
|
||||
from .models import FileRecord, OperationStatus
|
||||
from .namer import MediaNamer
|
||||
from .notifications import send_batch_notification
|
||||
from .providers import MetadataProvider, TMDBProvider
|
||||
from .scanner import ScannedFile, Scanner
|
||||
from .tokenizer import FilenameTokenizer, TokenizedFilename
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
|
||||
class MediaSorterApp:
|
||||
"""High-performance orchestrator for analyzing and organizing media collections."""
|
||||
|
||||
def __init__(self, settings: Settings, engine: Engine):
|
||||
self.settings = settings
|
||||
self.engine = engine
|
||||
|
||||
# Component instances
|
||||
self.scanner = Scanner(
|
||||
min_file_age_seconds=settings.general.min_file_age_seconds,
|
||||
include_patterns=settings.filters.include_patterns,
|
||||
exclude_patterns=settings.filters.exclude_patterns,
|
||||
)
|
||||
self.tokenizer = FilenameTokenizer()
|
||||
self.analyzer = MediaAnalyzer()
|
||||
|
||||
# Metadata provider
|
||||
provider: Optional[MetadataProvider] = None
|
||||
if settings.providers.enable_online_metadata:
|
||||
provider = TMDBProvider(
|
||||
api_key=settings.providers.tmdb_api_key,
|
||||
rate_limit_per_second=settings.providers.rate_limit_per_second,
|
||||
)
|
||||
|
||||
self.classifier = MediaClassifier(
|
||||
confidence_threshold=settings.general.confidence_threshold,
|
||||
provider=provider,
|
||||
)
|
||||
self.namer = MediaNamer(settings)
|
||||
|
||||
def scan_and_analyze(
|
||||
self,
|
||||
progress_callback: Optional[Callable[[int, int, str], None]] = None,
|
||||
filter_paths: Optional[List[Path]] = None,
|
||||
) -> List[Tuple[ScannedFile, ClassificationResult]]:
|
||||
"""Discover files across all configured source directories and analyze in parallel."""
|
||||
source_paths = self.settings.get_source_paths()
|
||||
all_scanned: List[ScannedFile] = []
|
||||
|
||||
logger.info("Starting library discovery", sources=[str(p) for p in source_paths])
|
||||
for src in source_paths:
|
||||
if src.exists():
|
||||
discovered = self.scanner.scan_directory(src)
|
||||
all_scanned.extend(discovered)
|
||||
else:
|
||||
logger.warning("Configured source path does not exist", path=str(src))
|
||||
|
||||
logger.info("Filesystem discovery complete", total_discovered=len(all_scanned))
|
||||
|
||||
if not all_scanned:
|
||||
return []
|
||||
|
||||
# Check DB cache to skip unchanged already organized files
|
||||
qualifying_files: List[ScannedFile] = []
|
||||
with get_db_session(self.engine) as session:
|
||||
for s in all_scanned:
|
||||
rec = (
|
||||
session.query(FileRecord)
|
||||
.filter_by(path=str(s.path), size=s.size, mtime=s.mtime, status="organized")
|
||||
.first()
|
||||
)
|
||||
if rec:
|
||||
logger.debug("Skipping unchanged already organized file", path=str(s.path))
|
||||
continue
|
||||
qualifying_files.append(s)
|
||||
|
||||
if filter_paths:
|
||||
filter_resolved = {p.resolve() for p in filter_paths}
|
||||
qualifying_files = [s for s in qualifying_files if s.path.resolve() in filter_resolved]
|
||||
|
||||
total_files = len(qualifying_files)
|
||||
logger.info("Files requiring processing", count=total_files)
|
||||
|
||||
# Check known shows from library
|
||||
known_shows: List[Dict[str, Any]] = []
|
||||
try:
|
||||
with get_db_session(self.engine) as session:
|
||||
from .library import get_known_shows
|
||||
known_shows = get_known_shows(session)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Parallel analysis and classification
|
||||
results: List[Tuple[ScannedFile, ClassificationResult]] = []
|
||||
worker_count = self.settings.general.worker_count
|
||||
|
||||
def process_one(scanned: ScannedFile) -> Tuple[ScannedFile, ClassificationResult]:
|
||||
tokens = self.tokenizer.tokenize(scanned.path)
|
||||
meta = self.analyzer.analyze(scanned.path)
|
||||
classification = self.classifier.classify(scanned, tokens, meta)
|
||||
|
||||
# Match against known library shows if category is unsure or confidence is low
|
||||
if known_shows and (
|
||||
classification.category not in ("tv", "movie")
|
||||
or classification.needs_quarantine
|
||||
or classification.confidence < 0.8
|
||||
):
|
||||
from .library import match_known_show
|
||||
matched = match_known_show(scanned.path.name, known_shows)
|
||||
if not matched and len(scanned.path.parts) > 1:
|
||||
matched = match_known_show(scanned.path.parent.name, known_shows)
|
||||
if matched:
|
||||
classification.category = "tv"
|
||||
if not classification.tokens.title or classification.tokens.title.lower() in ("episode", "unknown", ""):
|
||||
classification.tokens.title = matched["title"]
|
||||
classification.confidence = max(classification.confidence, 0.95)
|
||||
classification.needs_quarantine = False
|
||||
|
||||
return scanned, classification
|
||||
|
||||
completed = 0
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=worker_count) as executor:
|
||||
future_to_file = {executor.submit(process_one, sf): sf for sf in qualifying_files}
|
||||
for future in concurrent.futures.as_completed(future_to_file):
|
||||
try:
|
||||
res = future.result()
|
||||
results.append(res)
|
||||
except Exception as e:
|
||||
sf = future_to_file[future]
|
||||
logger.error("Error analyzing file", path=str(sf.path), error=str(e))
|
||||
completed += 1
|
||||
if progress_callback:
|
||||
progress_callback(completed, total_files, str(future_to_file[future].path.name))
|
||||
|
||||
return results
|
||||
|
||||
def build_plan(
|
||||
self, analysis_results: List[Tuple[ScannedFile, ClassificationResult]]
|
||||
) -> List[PlannedOperation]:
|
||||
"""Convert classification results into concrete planned operations."""
|
||||
primary_dest_map: Dict[Path, Path] = {}
|
||||
plan: List[PlannedOperation] = []
|
||||
|
||||
# 1. First pass: non-sidecar primary media files
|
||||
for scanned, cls_res in analysis_results:
|
||||
if not scanned.is_sidecar:
|
||||
dst = self.namer.generate_destination_path(cls_res)
|
||||
primary_dest_map[scanned.path] = dst
|
||||
|
||||
plan.append(
|
||||
PlannedOperation(
|
||||
src=scanned.path,
|
||||
dst=dst,
|
||||
action=self.settings.general.action,
|
||||
category=cls_res.category,
|
||||
confidence=cls_res.confidence,
|
||||
details=cls_res.signals,
|
||||
quarantine=cls_res.needs_quarantine,
|
||||
quarantine_reason=cls_res.quarantine_reason,
|
||||
)
|
||||
)
|
||||
|
||||
# 2. Second pass: sidecars (subtitles, artwork, metadata) matching primary destinations
|
||||
for scanned, cls_res in analysis_results:
|
||||
if scanned.is_sidecar:
|
||||
primary_dst = (
|
||||
primary_dest_map.get(scanned.primary_media_path)
|
||||
if scanned.primary_media_path
|
||||
else None
|
||||
)
|
||||
dst = self.namer.generate_destination_path(cls_res, primary_dst_path=primary_dst)
|
||||
|
||||
plan.append(
|
||||
PlannedOperation(
|
||||
src=scanned.path,
|
||||
dst=dst,
|
||||
action=self.settings.general.action,
|
||||
category=cls_res.category,
|
||||
confidence=cls_res.confidence,
|
||||
details=cls_res.signals,
|
||||
quarantine=cls_res.needs_quarantine,
|
||||
quarantine_reason=cls_res.quarantine_reason,
|
||||
)
|
||||
)
|
||||
|
||||
return plan
|
||||
|
||||
def run(
|
||||
self,
|
||||
dry_run: Optional[bool] = None,
|
||||
progress_callback: Optional[Callable[[int, int, str], None]] = None,
|
||||
filter_paths: Optional[List[Path]] = None,
|
||||
show_name_override: Optional[str] = None,
|
||||
) -> BatchExecutionReport:
|
||||
"""Run full media sorter pipeline: scan, analyze, plan, and execute."""
|
||||
start_time = time.time()
|
||||
is_dry_run = self.settings.general.dry_run if dry_run is None else dry_run
|
||||
|
||||
logger.info("Executing media-sorter run", dry_run=is_dry_run)
|
||||
|
||||
# Discover and analyze
|
||||
analysis_results = self.scan_and_analyze(
|
||||
progress_callback=progress_callback, filter_paths=filter_paths
|
||||
)
|
||||
|
||||
if show_name_override:
|
||||
for scanned, cls_res in analysis_results:
|
||||
cls_res.category = "tv"
|
||||
cls_res.tokens.title = show_name_override
|
||||
cls_res.tokens.is_episodic = True
|
||||
cls_res.needs_quarantine = False
|
||||
cls_res.confidence = max(cls_res.confidence, 0.95)
|
||||
|
||||
# Build plan
|
||||
planned_ops = self.build_plan(analysis_results)
|
||||
|
||||
# Execute under inter-process lock to coordinate workers
|
||||
lock_path = self.settings.get_database_path().with_suffix(".lock")
|
||||
with acquire_process_lock(lock_path):
|
||||
with get_db_session(self.engine) as session:
|
||||
executor = MediaExecutor(self.settings, session)
|
||||
# Check for crash recovery from prior runs
|
||||
executor.recover_interrupted_batches()
|
||||
|
||||
report = executor.execute_batch(
|
||||
planned_ops, dry_run=is_dry_run, progress_callback=progress_callback
|
||||
)
|
||||
|
||||
elapsed = round(time.time() - start_time, 2)
|
||||
logger.info(
|
||||
"Media sorter run finished",
|
||||
elapsed_seconds=elapsed,
|
||||
total=report.total_files,
|
||||
moved=report.moved_files,
|
||||
quarantined=report.quarantined_files,
|
||||
skipped=report.skipped_files,
|
||||
failed=report.failed_files,
|
||||
)
|
||||
|
||||
# Update library catalog with moved items
|
||||
if not is_dry_run and report.moved_files > 0:
|
||||
try:
|
||||
from .library import record_detected_item
|
||||
shows_dir = self.settings.get_destination_path("tv")
|
||||
movies_dir = self.settings.get_destination_path("movie")
|
||||
with get_db_session(self.engine) as session:
|
||||
for op in report.operations:
|
||||
if op.status == OperationStatus.COMMITTED.value:
|
||||
cat = (op.category or "tv").lower()
|
||||
dst_p = Path(op.dst)
|
||||
if cat in ("tv", "anime"):
|
||||
try:
|
||||
rel_tv = dst_p.relative_to(shows_dir)
|
||||
show_title = rel_tv.parts[0]
|
||||
dest_folder = str(shows_dir / show_title)
|
||||
record_detected_item(
|
||||
session,
|
||||
self.settings,
|
||||
show_title,
|
||||
"tv",
|
||||
destination_folder=dest_folder,
|
||||
delta_count=1,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
elif cat == "movie":
|
||||
try:
|
||||
rel_mv = dst_p.relative_to(movies_dir)
|
||||
movie_title = rel_mv.parts[0]
|
||||
dest_folder = str(movies_dir / movie_title)
|
||||
record_detected_item(
|
||||
session,
|
||||
self.settings,
|
||||
movie_title,
|
||||
"movie",
|
||||
destination_folder=dest_folder,
|
||||
delta_count=1,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as e:
|
||||
logger.error("Error updating library from execution report", error=str(e))
|
||||
|
||||
# Send notification webhook if configured
|
||||
if self.settings.notifications.enabled:
|
||||
send_batch_notification(self.settings.notifications, report)
|
||||
|
||||
return report
|
||||
|
||||
def rollback(self, batch_id: Optional[str] = None) -> int:
|
||||
"""Rollback a past batch of operations."""
|
||||
with get_db_session(self.engine) as session:
|
||||
executor = MediaExecutor(self.settings, session)
|
||||
return executor.rollback_batch(batch_id)
|
||||
|
||||
def rollback_all(self) -> int:
|
||||
"""Rollback all past completed batches."""
|
||||
with get_db_session(self.engine) as session:
|
||||
executor = MediaExecutor(self.settings, session)
|
||||
return executor.rollback_all()
|
||||
|
||||
@@ -0,0 +1,271 @@
|
||||
/* Base CSS for Media Sorter UI */
|
||||
:root {
|
||||
--bg: #ffffff; /* background */
|
||||
--card-bg: #f9f9f9; /* cards */
|
||||
--card-hover: #eaeaea;
|
||||
--border: #dddddd;
|
||||
--text: #222222;
|
||||
--text-muted: #555555;
|
||||
--accent: #0066ff; /* default accent */
|
||||
--accent-hover: #0044cc;
|
||||
--radius-sm: 4px;
|
||||
--radius-md: 8px;
|
||||
}
|
||||
|
||||
[data-theme="minimalistic"] {
|
||||
/* Light, clean look */
|
||||
--bg: #fafafa;
|
||||
--card-bg: #ffffff;
|
||||
--card-hover: #f0f0f0;
|
||||
--border: #e0e0e0;
|
||||
--text: #111111;
|
||||
--text-muted: #777777;
|
||||
--accent: #0077c2;
|
||||
--accent-hover: #005599;
|
||||
}
|
||||
|
||||
[data-theme="high-visibility"] {
|
||||
/* Dark with bright accent for strong contrast */
|
||||
--bg: #111111;
|
||||
--card-bg: #1a1a1a;
|
||||
--card-hover: #262626;
|
||||
--border: #333333;
|
||||
--text: #eeeeee;
|
||||
--text-muted: #bbbbbb;
|
||||
--accent: #ffdd00; /* bright yellow */
|
||||
--accent-hover: #ffbb00;
|
||||
}
|
||||
|
||||
[data-theme="pastel"] {
|
||||
--bg: #fff8f0;
|
||||
--card-bg: #ffffff;
|
||||
--card-hover: #f0e6e0;
|
||||
--border: #e6d8d1;
|
||||
--text: #453636;
|
||||
--text-muted: #7a5d5d;
|
||||
--accent: #ff8c94; /* soft pink */
|
||||
--accent-hover: #ff6b78;
|
||||
}
|
||||
|
||||
[data-theme="grayscale"] {
|
||||
--bg: #f5f5f5;
|
||||
--card-bg: #ffffff;
|
||||
--card-hover: #e0e0e0;
|
||||
--border: #c0c0c0;
|
||||
--text: #333333;
|
||||
--text-muted: #777777;
|
||||
--accent: #555555;
|
||||
--accent-hover: #444444;
|
||||
}
|
||||
|
||||
[data-theme="cyber"] {
|
||||
/* Retain original cyber‑dark palette but with new ID */
|
||||
--bg: #090d16;
|
||||
--card-bg: #0f172a;
|
||||
--card-hover: #1e293b;
|
||||
--border: #334155;
|
||||
--text: #f8fafc;
|
||||
--text-muted: #94a3b8;
|
||||
--accent: #38bdf8;
|
||||
--accent-hover: #0284c7;
|
||||
}
|
||||
|
||||
[data-theme="bios-amber"] {
|
||||
--bg: #0c0800;
|
||||
--card-bg: #181100;
|
||||
--card-hover: #261b02;
|
||||
--border: #4d3800;
|
||||
--text: #ffb833;
|
||||
--text-muted: #b37e1a;
|
||||
--accent: #ff9900;
|
||||
--accent-hover: #ffad33;
|
||||
}
|
||||
|
||||
[data-theme="vapor-glitch"] {
|
||||
--bg: #080312;
|
||||
--card-bg: #130a24;
|
||||
--card-hover: #1f113a;
|
||||
--border: #38195a;
|
||||
--text: #00f5d4;
|
||||
--text-muted: #b388ff;
|
||||
--accent: #f72585;
|
||||
--accent-hover: #b5179e;
|
||||
}
|
||||
|
||||
[data-theme="mossy-stone"] {
|
||||
--bg: #111813;
|
||||
--card-bg: #19241c;
|
||||
--card-hover: #223227;
|
||||
--border: #2d4234;
|
||||
--text: #d2e0d5;
|
||||
--text-muted: #859e8b;
|
||||
--accent: #52b788;
|
||||
--accent-hover: #40916c;
|
||||
}
|
||||
|
||||
[data-theme="crimson-eclipse"] {
|
||||
--bg: #0d0608;
|
||||
--card-bg: #190c10;
|
||||
--card-hover: #261217;
|
||||
--border: #441822;
|
||||
--text: #e8d0d5;
|
||||
--text-muted: #a37581;
|
||||
--accent: #ff4d6d;
|
||||
--accent-hover: #c9184a;
|
||||
}
|
||||
|
||||
[data-theme="blueprint-draft"] {
|
||||
--bg: #0b1d3a;
|
||||
--card-bg: #102a54;
|
||||
--card-hover: #173b75;
|
||||
--border: #1f4a91;
|
||||
--text: #e2edfd;
|
||||
--text-muted: #8db5e6;
|
||||
--accent: #60a5fa;
|
||||
--accent-hover: #3b82f6;
|
||||
}
|
||||
|
||||
/* RGB mode styling */
|
||||
.rgb-mode {
|
||||
--rgb-duration: 5s;
|
||||
}
|
||||
.rgb-mode .card,
|
||||
.rgb-mode .btn,
|
||||
.rgb-mode .theme-select {
|
||||
animation: rgbPulse var(--rgb-duration) infinite;
|
||||
}
|
||||
@keyframes rgbPulse {
|
||||
0% { box-shadow: 0 0 8px rgba(255,0,0,0.5); }
|
||||
33% { box-shadow: 0 0 8px rgba(0,255,0,0.5); }
|
||||
66% { box-shadow: 0 0 8px rgba(0,0,255,0.5); }
|
||||
100% { box-shadow: 0 0 8px rgba(255,0,0,0.5); }
|
||||
}
|
||||
|
||||
[data-theme="neon-forest"] {
|
||||
--bg: #001408;
|
||||
--card-bg: #032410;
|
||||
--card-hover: #07381b;
|
||||
--border: #0d542a;
|
||||
--text: #c8facc;
|
||||
--text-muted: #5ea874;
|
||||
--accent: #39ff14;
|
||||
--accent-hover: #2ecc11;
|
||||
}
|
||||
|
||||
[data-theme="retro-retro"] {
|
||||
--bg: #120024;
|
||||
--card-bg: #220038;
|
||||
--card-hover: #330052;
|
||||
--border: #6b0099;
|
||||
--text: #fce7f3;
|
||||
--text-muted: #d946ef;
|
||||
--accent: #ff00ff;
|
||||
--accent-hover: #d500d5;
|
||||
}
|
||||
|
||||
[data-theme="golden-sand"] {
|
||||
--bg: #1c150c;
|
||||
--card-bg: #291e10;
|
||||
--card-hover: #3b2c17;
|
||||
--border: #594322;
|
||||
--text: #fbf0dc;
|
||||
--text-muted: #bda27e;
|
||||
--accent: #ffb300;
|
||||
--accent-hover: #e09d00;
|
||||
}
|
||||
|
||||
[data-theme="deep-space"] {
|
||||
--bg: #070913;
|
||||
--card-bg: #0d1224;
|
||||
--card-hover: #141c38;
|
||||
--border: #232f57;
|
||||
--text: #e2edfd;
|
||||
--text-muted: #818cf8;
|
||||
--accent: #6366f1;
|
||||
--accent-hover: #4f46e5;
|
||||
}
|
||||
|
||||
[data-theme="candy-cotton"] {
|
||||
--bg: #1a1520;
|
||||
--card-bg: #261f30;
|
||||
--card-hover: #362c44;
|
||||
--border: #524166;
|
||||
--text: #fdf2f8;
|
||||
--text-muted: #f472b6;
|
||||
--accent: #ffb6c1;
|
||||
--accent-hover: #f694a5;
|
||||
}
|
||||
|
||||
/* General layout */
|
||||
body {
|
||||
background: var(--bg);
|
||||
color: var(--text);
|
||||
font-family: Arial, Helvetica, sans-serif;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
}
|
||||
|
||||
header.app-header, footer.app-footer {
|
||||
background: var(--card-bg);
|
||||
border-bottom: 1px solid var(--border);
|
||||
padding: 0.75rem 1rem;
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
align-items: center;
|
||||
}
|
||||
|
||||
main.app-main {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 1rem;
|
||||
padding: 1rem;
|
||||
}
|
||||
|
||||
.panel {
|
||||
background: var(--card-bg);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: var(--radius-md);
|
||||
padding: 0.75rem;
|
||||
flex: 1 1 300px;
|
||||
max-width: 100%;
|
||||
overflow: auto;
|
||||
}
|
||||
|
||||
.btn {
|
||||
background: var(--accent);
|
||||
color: #fff;
|
||||
border: none;
|
||||
border-radius: var(--radius-sm);
|
||||
padding: 0.4rem 0.8rem;
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.btn:hover {
|
||||
background: var(--accent-hover);
|
||||
}
|
||||
|
||||
.modal {
|
||||
position: fixed;
|
||||
inset: 0;
|
||||
background: rgba(0,0,0,0.4);
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
}
|
||||
.modal.hidden { display: none; }
|
||||
.modal-content {
|
||||
background: var(--card-bg);
|
||||
padding: 1.5rem;
|
||||
border-radius: var(--radius-md);
|
||||
min-width: 260px;
|
||||
}
|
||||
|
||||
/* Compact mode tweaks */
|
||||
[data-compact="true"] header.app-header, [data-compact="true"] footer.app-footer {
|
||||
padding: 0.4rem 0.8rem;
|
||||
font-size: 0.9rem;
|
||||
}
|
||||
[data-compact="true"] .panel {
|
||||
padding: 0.5rem;
|
||||
font-size: 0.85rem;
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<title>Media Sorter Management</title>
|
||||
<link rel="stylesheet" href="/static/css/base.css" />
|
||||
<script src="/static/js/app.js" defer></script>
|
||||
</head>
|
||||
<body>
|
||||
<!-- Settings Modal -->
|
||||
<div id="settings-modal" class="modal hidden">
|
||||
<div class="modal-content">
|
||||
<h2>Settings</h2>
|
||||
<label for="theme-select">Theme</label>
|
||||
<select id="theme-select"></select>
|
||||
<label for="compact-toggle"><input type="checkbox" id="compact-toggle" /> Compact Mode</label>
|
||||
<button id="close-settings" class="btn">Close</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Main UI -->
|
||||
<header class="app-header">
|
||||
<h1>Media Sorter</h1>
|
||||
<button id="open-settings" class="btn">Settings</button>
|
||||
</header>
|
||||
<main class="app-main">
|
||||
<section id="folder-explorer" class="panel"></section>
|
||||
<section id="library" class="panel"></section>
|
||||
<section id="quarantine" class="panel"></section>
|
||||
</main>
|
||||
<footer class="app-footer">
|
||||
<span id="status-bar">Ready</span>
|
||||
</footer>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,653 @@
|
||||
"""Filename and path tokenization engine for Media Sorter.
|
||||
|
||||
Robustly extracts semantic media tokens (title, year, season, episode, artist,
|
||||
album, track, disc, quality, codec, release group, date stamps) from messy filenames,
|
||||
scene releases, anime fansub conventions, and folder hierarchies.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
# Regex patterns for Video & Episodic Media
|
||||
ROMAN_NUMERALS: Dict[str, int] = {
|
||||
"i": 1, "ii": 2, "iii": 3, "iv": 4, "v": 5,
|
||||
"vi": 6, "vii": 7, "viii": 8, "ix": 9, "x": 10,
|
||||
"xi": 11, "xii": 12, "xiii": 13, "xiv": 14, "xv": 15,
|
||||
"xvi": 16, "xvii": 17, "xviii": 18, "xix": 19, "xx": 20,
|
||||
}
|
||||
|
||||
RE_ROMAN_SEASON_EPISODE = re.compile(
|
||||
r"""(?ix)
|
||||
\bseason[\.\s_-]*(?P<season_roman>[ivx]+)[\.\s_-]*(?:episode|ep)[\.\s_-]*(?P<episode_roman>[ivx]+)\b
|
||||
"""
|
||||
)
|
||||
|
||||
RE_SEASON_EPISODE = re.compile(
|
||||
r"""(?ix)
|
||||
(?:
|
||||
# Standard S01E02, S01E01-E02, S01E01E02, S01E01-02, S01E1171
|
||||
(?<![0-9a-z])s(?P<season>\d{1,2})[\.\s_-]*(?:e|ep|ed|op)(?P<episode>\d{1,4})
|
||||
(?:[\.\s_-]*(?:e|x|-|ep)(?P<episode_end>\d{1,4}))?(?![0-9])
|
||||
|
|
||||
# Scene 1x02, 2x01-02, 2x01-x02, 1x1171
|
||||
(?<![0-9a-z])(?P<season_x>\d{1,2})x(?!(?:264|265|vid|hevc|avc))(?P<episode_x>\d{1,4})
|
||||
(?:[\.\s_-]*(?:x|-)(?P<episode_x_end>\d{1,4}))?(?![0-9])
|
||||
|
|
||||
# Word season / episode: Season 1 Episode 2, Season 1 Episode 1171
|
||||
\bseason[\.\s_-]*(?P<season_word>\d{1,2})[\.\s_-]*(?:episode|ep)[\.\s_-]*(?P<episode_word>\d{1,4})
|
||||
(?:[\.\s_-]*(?:-|to)[\.\s_-]*(?:episode|ep)?[\\.\s_-]*(?P<episode_word_end>\d{1,4}))?\b
|
||||
|
|
||||
# Standalone episode: Episode 207, Ep 01, E233, E1171
|
||||
(?<![0-9a-z])(?:episodes?|ep|e)[\.\s_-]*(?P<episode_standalone>\d{1,4})(?![0-9a-zA-Z])
|
||||
)
|
||||
"""
|
||||
)
|
||||
|
||||
RE_SEASON_PACK = re.compile(
|
||||
r"""(?ix)
|
||||
(?<![0-9a-z])
|
||||
(?:
|
||||
s(?P<season_pack>\d{1,2})
|
||||
|
|
||||
season[\.\s_-]*(?P<season_pack_word>\d{1,2})
|
||||
)
|
||||
[\.\s_-]*(?:complete|full|season\.pack)\b
|
||||
"""
|
||||
)
|
||||
|
||||
# Anime fansub format: [ReleaseGroup] Show Title - 01 (or 01-02, or 01v2) - Optional Episode Title [1080p] [CRC32].mkv
|
||||
RE_ANIME_RELEASE = re.compile(
|
||||
r"""(?ix)
|
||||
^\s*(?:\[(?P<group>[^\]]+)\][\s_]*)?
|
||||
(?P<title>.+?)\s*(?:-\s*|_\-_\s*|_)\s*
|
||||
(?P<episode>\d{1,4})(?:-(?P<episode_end>\d{1,4}))?(?:v\d+)?(?![0-9a-zA-Z])\s*
|
||||
(?:\s*-\s*(?P<ep_title>[^\[\(]+?))?
|
||||
(?:[\s_]*(?:\[?[0-9A-Fa-f]{8}\]?|\[(?P<tag>[^\]]+)\]|\((?P<tag_paren>[^\)]+)\))|[\s_]+[A-Za-z0-9_.-]+)*\s*\]?$
|
||||
"""
|
||||
)
|
||||
|
||||
# Space-separated anime format: [ReleaseGroup] Show Title 01 (Tags) [CRC32].mkv
|
||||
RE_ANIME_RELEASE_SPACE = re.compile(
|
||||
r"""(?ix)
|
||||
^\s*\[(?P<group>[^\]]+)\]\s*
|
||||
(?P<title>[^\[\(]+?)\s+
|
||||
(?P<episode>\d{1,4})(?:-(?P<episode_end>\d{1,4}))?(?:v\d+)?(?![0-9a-zA-Z])\s*
|
||||
(?:[\s_]*(?:\[?[0-9A-Fa-f]{8}\]?|\[(?P<tag>[^\]]+)\]|\((?P<tag_paren>[^\)]+)\))|[\s_]+[A-Za-z0-9_.-]+)*\s*\]?$
|
||||
"""
|
||||
)
|
||||
|
||||
# Underscore anime format: [ReleaseGroup]_Show_Title_01_[Tags].mp4
|
||||
RE_ANIME_RELEASE_UNDERSCORE = re.compile(
|
||||
r"""(?ix)
|
||||
^\s*\[(?P<group>[^\]]+)\]_
|
||||
(?P<title>[^\[\(]+?)_
|
||||
(?P<episode>\d{1,4})(?:-(?P<episode_end>\d{1,4}))?(?:v\d+)?(?![0-9a-zA-Z])
|
||||
(?:_*(?:\[?[0-9A-Fa-f]{8}\]?|\[(?P<tag>[^\]]+)\]|\((?P<tag_paren>[^\)]+)\))|_[A-Za-z0-9_.-]+)*\s*\]?$
|
||||
"""
|
||||
)
|
||||
|
||||
RE_ANIME_MOVIE = re.compile(
|
||||
r"""(?ix)
|
||||
^\s*\[(?P<group>[^\]]+)\]\s*
|
||||
(?P<title>[^\[]+?)\s*
|
||||
(?:\[(?P<tag>[^\]]+)\]|\((?P<tag_paren>[^\)]+)\))
|
||||
"""
|
||||
)
|
||||
|
||||
RE_YEAR_BOUND = re.compile(r"(?<![0-9a-zA-Z])(19\d{2}|20\d{2})(?![0-9a-zA-Z])")
|
||||
RE_YEAR = re.compile(r"\b(19\d{2}|20\d{2})\b")
|
||||
|
||||
RE_EDITION = re.compile(
|
||||
r"""(?ix)
|
||||
\b(?P<edition>
|
||||
directors?\.cut|director's\.cut|director's\scut
|
||||
|
|
||||
extended(?:\.cut|\.edition)?
|
||||
|
|
||||
remastered(?:\.edition)?|remaster
|
||||
|
|
||||
criterion(?:\.collection)?
|
||||
|
|
||||
final\.cut
|
||||
|
|
||||
theatrical(?:\.cut|\.version)?
|
||||
|
|
||||
unrated
|
||||
|
|
||||
special\.edition
|
||||
|
|
||||
imax(?:\.edition)?
|
||||
|
|
||||
ultimate\.edition
|
||||
)\b
|
||||
"""
|
||||
)
|
||||
|
||||
EDITION_CANONICAL_MAP: Dict[str, str] = {
|
||||
"extended": "Extended",
|
||||
"extended.cut": "Extended",
|
||||
"extended.edition": "Extended",
|
||||
"directors.cut": "Director's Cut",
|
||||
"director's.cut": "Director's Cut",
|
||||
"director's cut": "Director's Cut",
|
||||
"remastered": "Remastered",
|
||||
"remastered.edition": "Remastered",
|
||||
"remaster": "Remastered",
|
||||
"criterion": "Criterion",
|
||||
"criterion.collection": "Criterion",
|
||||
"final.cut": "Final Cut",
|
||||
"theatrical": "Theatrical",
|
||||
"theatrical.cut": "Theatrical",
|
||||
"theatrical.version": "Theatrical",
|
||||
"unrated": "Unrated",
|
||||
"special.edition": "Special Edition",
|
||||
"imax": "IMAX",
|
||||
"imax.edition": "IMAX",
|
||||
"ultimate.edition": "Ultimate Edition",
|
||||
}
|
||||
|
||||
RE_MOVIE_PART = re.compile(
|
||||
r"""(?ix)
|
||||
\b(?:cd|part|pt|disc)[\.\s_-]*(?P<part_num>\d{1,2})\b
|
||||
"""
|
||||
)
|
||||
|
||||
RE_DAILY_DATE = re.compile(
|
||||
r"""(?ix)
|
||||
(?<!\d)
|
||||
(?P<year>19\d{2}|20\d{2})[-._]
|
||||
(?P<month>0[1-9]|1[0-2])[-._]
|
||||
(?P<day>0[1-9]|[12]\d|3[01])
|
||||
(?!\d)
|
||||
"""
|
||||
)
|
||||
|
||||
# Technical specs
|
||||
RE_RESOLUTION = re.compile(r"\b(2160p|4k|1080p|1080i|720p|576p|480p)\b", re.IGNORECASE)
|
||||
RE_DIMENSIONS = re.compile(r"\b(?:\d{3,4})x(?P<height>2160|1080|720|576|480)\b", re.IGNORECASE)
|
||||
RE_SOURCE = re.compile(r"\b(bluray|blu-ray|bdrip|web-dl|webrip|web|hdtv|dvdrip|dvd|remux)\b", re.IGNORECASE)
|
||||
RE_VIDEO_CODEC = re.compile(r"\b(x265|x264|h\.?265|h\.?264|hevc|avc|av1|xvid|divx)\b", re.IGNORECASE)
|
||||
RE_AUDIO_CODEC = re.compile(r"\b(truehd|atmos|dts-hd|dts|flac|aac|ac3|ddp?5\.1|mp3)\b", re.IGNORECASE)
|
||||
RE_RELEASE_GROUP = re.compile(r"-([A-Za-z0-9_]+)(?:\[.*?\])?$", re.IGNORECASE)
|
||||
RE_RELEASE_GROUP_UPGRADED = re.compile(
|
||||
r"-(?:\[(?P<grp_bracket>[A-Za-z0-9_.-]+)\]|(?P<grp_plain>[A-Za-z0-9_]+))(?:\[.*?\])?$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
RE_ILLEGAL_CHARS = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
|
||||
|
||||
RE_TECH_ALL = re.compile(
|
||||
r"""(?ix)
|
||||
\b(
|
||||
2160p|4k|1080p|1080i|720p|576p|480p
|
||||
|
|
||||
\d{3,4}x(?:2160|1080|720|576|480)
|
||||
|
|
||||
bluray|blu-ray|bdrip|web-dl|webrip|web|hdtv|dvdrip|dvd|remux
|
||||
|
|
||||
x265|x264|h\.?265|h\.?264|hevc|avc|av1|xvid|divx
|
||||
|
|
||||
truehd|atmos|dts-hd|dts|flac|aac|ac3|ddp?5\.1|mp3
|
||||
|
|
||||
directors?\.cut|director's\.cut|director's\scut|extended|remastered|criterion|final\.cut
|
||||
|
|
||||
cd\d|part\d|pt\d
|
||||
|
|
||||
proper
|
||||
)\b
|
||||
"""
|
||||
)
|
||||
|
||||
KNOWN_ANIME_GROUPS = {
|
||||
"subsplease", "horriblesubs", "erai-raws", "taigasubs", "judas", "commie", "asenshi", "coalgirls"
|
||||
}
|
||||
|
||||
KNOWN_ANIME_TITLES = {
|
||||
"naruto", "bleach", "one piece", "frieren", "dungeon meshi", "attack on titan",
|
||||
"jujutsu kaisen", "mushoku tensei", "fairy tail", "fate stay night", "sword art online"
|
||||
}
|
||||
|
||||
WINDOWS_RESERVED = {
|
||||
"CON", "PRN", "AUX", "NUL",
|
||||
"COM1", "COM2", "COM3", "COM4", "COM5", "COM6", "COM7", "COM8", "COM9",
|
||||
"LPT1", "LPT2", "LPT3", "LPT4", "LPT5", "LPT6", "LPT7", "LPT8", "LPT9",
|
||||
}
|
||||
|
||||
# Music / Audio track patterns: 01 - Title, 1-01 Title, Artist - 01 - Title
|
||||
RE_MUSIC_TRACK = re.compile(
|
||||
r"""(?ix)
|
||||
^(?:(?P<disc>\d{1,2})[-_.])?(?P<track>\d{1,3})[\.\s_-]+(?P<title>.+)$
|
||||
"""
|
||||
)
|
||||
|
||||
# Photo and Home Video date stamps: IMG_20240812_142010, VID_20240812_142010, 2024-08-12 14.20.10
|
||||
RE_CAMERA_DATE = re.compile(
|
||||
r"""(?ix)
|
||||
(?:img|vid|dsc|pano|mov)?[-_]?(?P<year>19\d{2}|20\d{2})[-_]?(?P<month>\d{2})[-_]?(?P<day>\d{2})
|
||||
(?:[-_](?P<hour>\d{2})[-_]?(?P<minute>\d{2})[-_]?(?P<second>\d{2}))?
|
||||
"""
|
||||
)
|
||||
|
||||
# Podcast dated format: Show Name - 2026-03-15 - Episode Title
|
||||
RE_PODCAST_DATE = re.compile(
|
||||
r"""(?ix)
|
||||
^(?P<show>.+?)\s*-\s*(?P<year>20\d{2})-(?P<month>\d{2})-(?P<day>\d{2})\s*-\s*(?P<title>.+)$
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
# CRC32 checksum tag e.g. [194B3FBA]
|
||||
RE_CRC32_TAG = re.compile(r"\[([0-9A-Fa-f]{8})\]")
|
||||
|
||||
|
||||
@dataclass
|
||||
class TokenizedFilename:
|
||||
raw_name: str
|
||||
title: Optional[str] = None
|
||||
year: Optional[int] = None
|
||||
season: Optional[int] = None
|
||||
episode: Optional[int] = None
|
||||
multi_episodes: List[int] = field(default_factory=list)
|
||||
episode_title: Optional[str] = None
|
||||
artist: Optional[str] = None
|
||||
album: Optional[str] = None
|
||||
track: Optional[int] = None
|
||||
disc: Optional[int] = None
|
||||
group: Optional[str] = None
|
||||
resolution: Optional[str] = None
|
||||
source: Optional[str] = None
|
||||
video_codec: Optional[str] = None
|
||||
audio_codec: Optional[str] = None
|
||||
crc32: Optional[str] = None
|
||||
date_stamp: Optional[str] = None
|
||||
air_date: Optional[str] = None
|
||||
edition: Optional[str] = None
|
||||
part: Optional[int] = None
|
||||
part_label: Optional[str] = None
|
||||
is_anime: bool = False
|
||||
is_episodic: bool = False
|
||||
is_music: bool = False
|
||||
is_photo_or_home_video: bool = False
|
||||
is_daily: bool = False
|
||||
is_season_pack: bool = False
|
||||
|
||||
|
||||
# Contract alias per PROJECT.md:74
|
||||
TokenizedMedia = TokenizedFilename
|
||||
|
||||
|
||||
class FilenameTokenizer:
|
||||
"""Parses raw filenames and directory paths into semantic tokens."""
|
||||
|
||||
def tokenize(self, file_path: Path) -> TokenizedFilename:
|
||||
stem = file_path.stem
|
||||
raw_name = file_path.name
|
||||
tokens = TokenizedFilename(raw_name=raw_name)
|
||||
ext = file_path.suffix.lower()
|
||||
|
||||
# Check for CRC32 tag
|
||||
crc_m = RE_CRC32_TAG.search(stem)
|
||||
if crc_m:
|
||||
tokens.crc32 = crc_m.group(1).upper()
|
||||
|
||||
# Check Windows reserved names
|
||||
if stem.upper() in WINDOWS_RESERVED:
|
||||
tokens.title = stem.upper()
|
||||
tokens.is_photo_or_home_video = True
|
||||
return tokens
|
||||
|
||||
# Pre-clean illegal characters
|
||||
had_illegal = False
|
||||
if RE_ILLEGAL_CHARS.search(stem):
|
||||
had_illegal = True
|
||||
stem = RE_ILLEGAL_CHARS.sub(" ", stem)
|
||||
stem = re.sub(r"\s+", " ", stem).strip()
|
||||
|
||||
# 1. Technical specifications
|
||||
res_m = RE_RESOLUTION.search(stem)
|
||||
if res_m:
|
||||
tokens.resolution = res_m.group(1).lower()
|
||||
if tokens.resolution == "4k":
|
||||
tokens.resolution = "2160p"
|
||||
else:
|
||||
dim_m = RE_DIMENSIONS.search(stem)
|
||||
if dim_m:
|
||||
tokens.resolution = f"{dim_m.group('height')}p"
|
||||
|
||||
src_m = RE_SOURCE.search(stem)
|
||||
if src_m:
|
||||
tokens.source = src_m.group(1).upper()
|
||||
|
||||
vc_m = RE_VIDEO_CODEC.search(stem)
|
||||
if vc_m:
|
||||
tokens.video_codec = vc_m.group(1).lower().replace(".", "")
|
||||
|
||||
ac_m = RE_AUDIO_CODEC.search(stem)
|
||||
if ac_m:
|
||||
tokens.audio_codec = ac_m.group(1).upper()
|
||||
|
||||
# Check edition
|
||||
ed_m = RE_EDITION.search(stem)
|
||||
if ed_m:
|
||||
ed_raw = ed_m.group("edition").lower().replace(" ", ".")
|
||||
tokens.edition = EDITION_CANONICAL_MAP.get(ed_raw, ed_m.group("edition"))
|
||||
|
||||
# Check part
|
||||
pt_m = RE_MOVIE_PART.search(stem)
|
||||
if pt_m:
|
||||
tokens.part = int(pt_m.group("part_num"))
|
||||
tokens.part_label = f"Pt.{tokens.part}"
|
||||
|
||||
# 2. Check for Podcast date format
|
||||
pod_m = RE_PODCAST_DATE.match(stem)
|
||||
if pod_m:
|
||||
tokens.title = pod_m.group("title").strip()
|
||||
tokens.artist = pod_m.group("show").strip()
|
||||
tokens.year = int(pod_m.group("year"))
|
||||
tokens.date_stamp = f"{pod_m.group('year')}-{pod_m.group('month')}-{pod_m.group('day')}"
|
||||
tokens.air_date = tokens.date_stamp
|
||||
return tokens
|
||||
|
||||
# 3. Check for Daily / Broadcast dated format (TV or Podcast)
|
||||
daily_m = RE_DAILY_DATE.search(stem)
|
||||
if daily_m:
|
||||
y, m, d = daily_m.group("year"), daily_m.group("month"), daily_m.group("day")
|
||||
date_str = f"{y}-{m}-{d}"
|
||||
tokens.date_stamp = date_str
|
||||
tokens.air_date = date_str
|
||||
tokens.year = int(y)
|
||||
prefix = stem[: daily_m.start()]
|
||||
clean_pfx = self._clean_title(prefix)
|
||||
tokens.title = clean_pfx
|
||||
if ext in {".mp3", ".flac", ".ogg", ".m4a", ".aac"}:
|
||||
tokens.artist = clean_pfx
|
||||
else:
|
||||
tokens.is_daily = True
|
||||
tokens.is_episodic = True
|
||||
tokens.season = int(y)
|
||||
return tokens
|
||||
|
||||
# 4. Check for Camera / Date stamp (Photos & Home Videos)
|
||||
cam_m = RE_CAMERA_DATE.search(stem)
|
||||
if cam_m:
|
||||
y, m, d = cam_m.group("year"), cam_m.group("month"), cam_m.group("day")
|
||||
tokens.date_stamp = f"{y}-{m}-{d}"
|
||||
tokens.year = int(y)
|
||||
tokens.is_photo_or_home_video = True
|
||||
return tokens
|
||||
|
||||
# 5. Check Roman Numeral TV pattern: Rome.Season.II.Episode.IV
|
||||
roman_m = RE_ROMAN_SEASON_EPISODE.search(stem)
|
||||
if roman_m:
|
||||
tokens.is_episodic = True
|
||||
s_rom = roman_m.group("season_roman").lower()
|
||||
e_rom = roman_m.group("episode_roman").lower()
|
||||
tokens.season = ROMAN_NUMERALS.get(s_rom, 1)
|
||||
tokens.episode = ROMAN_NUMERALS.get(e_rom, 1)
|
||||
prefix = stem[: roman_m.start()]
|
||||
tokens.title = self._clean_title(prefix)
|
||||
return tokens
|
||||
|
||||
# 6. Check TV Season Pack: Succession.S02.Complete
|
||||
pack_m = RE_SEASON_PACK.search(stem)
|
||||
if pack_m:
|
||||
tokens.is_episodic = True
|
||||
tokens.is_season_pack = True
|
||||
s_val = pack_m.group("season_pack") or pack_m.group("season_pack_word")
|
||||
tokens.season = int(s_val)
|
||||
prefix = stem[: pack_m.start()]
|
||||
tokens.title = self._clean_title(prefix)
|
||||
return tokens
|
||||
|
||||
# 7. Check Standard TV episodic patterns (S01E02, 1x02, Season 1 Episode 2, Episode 207)
|
||||
tv_m = RE_SEASON_EPISODE.search(stem)
|
||||
if tv_m:
|
||||
tokens.is_episodic = True
|
||||
season_str = tv_m.group("season") or tv_m.group("season_x") or tv_m.group("season_word")
|
||||
ep_str = tv_m.group("episode") or tv_m.group("episode_x") or tv_m.group("episode_word") or tv_m.group("episode_standalone")
|
||||
if season_str:
|
||||
tokens.season = int(season_str)
|
||||
else:
|
||||
tokens.season = self._extract_season_from_path(file_path) or 1
|
||||
|
||||
if ep_str:
|
||||
tokens.episode = int(ep_str)
|
||||
|
||||
end_ep = tv_m.group("episode_end") or tv_m.group("episode_x_end") or tv_m.group("episode_word_end")
|
||||
if end_ep:
|
||||
tokens.multi_episodes = list(range(tokens.episode, int(end_ep) + 1))
|
||||
|
||||
# If filename had illegal characters and matched standalone episode (e.g. Show: "Special" <Episode> | 1?.mkv)
|
||||
if had_illegal and tv_m.group("episode_standalone"):
|
||||
tokens.title = self._clean_title(stem)
|
||||
return tokens
|
||||
|
||||
# Extract title before season marker
|
||||
prefix = stem[: tv_m.start()]
|
||||
clean_pfx = self._clean_title(prefix)
|
||||
if clean_pfx:
|
||||
tokens.title = clean_pfx
|
||||
else:
|
||||
tokens.title = self._extract_title_from_context(file_path) or "Episode"
|
||||
|
||||
# Check for year in prefix using RE_YEAR_BOUND
|
||||
if prefix:
|
||||
yr_m = RE_YEAR_BOUND.search(prefix)
|
||||
if yr_m:
|
||||
tokens.year = int(yr_m.group(1))
|
||||
tokens.title = self._clean_title(prefix[: yr_m.start()])
|
||||
|
||||
# Extract episode title after season marker
|
||||
suffix = stem[tv_m.end() :]
|
||||
ep_title = self._extract_episode_title(suffix)
|
||||
if ep_title:
|
||||
tokens.episode_title = ep_title
|
||||
|
||||
# Release group at end
|
||||
grp_m = RE_RELEASE_GROUP_UPGRADED.search(stem)
|
||||
if grp_m:
|
||||
tokens.group = grp_m.group("grp_bracket") or grp_m.group("grp_plain")
|
||||
|
||||
# Check if title is a known anime title
|
||||
if tokens.title and tokens.title.lower() in KNOWN_ANIME_TITLES:
|
||||
tokens.is_anime = True
|
||||
|
||||
return tokens
|
||||
|
||||
# 8. Check Anime fansub format: [Group] Title - 01 - Episode Title [1080p]
|
||||
anime_m = (
|
||||
RE_ANIME_RELEASE.match(stem)
|
||||
or RE_ANIME_RELEASE_SPACE.match(stem)
|
||||
or RE_ANIME_RELEASE_UNDERSCORE.match(stem)
|
||||
)
|
||||
if anime_m and (anime_m.group("group") or ext in {".mkv", ".mp4", ".avi", ".mov", ".ts", ".webm", ".m4v", ".flv"}):
|
||||
ep_val = int(anime_m.group("episode"))
|
||||
grp_name = anime_m.group("group").strip() if anime_m.group("group") else None
|
||||
raw_title = anime_m.group("title")
|
||||
title_clean = self._clean_title(raw_title, preserve_paren=True)
|
||||
|
||||
# Distinguish movie year from anime episode
|
||||
is_known_anime = title_clean.lower() in KNOWN_ANIME_TITLES or (grp_name and grp_name.lower() in KNOWN_ANIME_GROUPS)
|
||||
has_explicit_season = self._extract_season_from_path(file_path) is not None
|
||||
has_range = bool(anime_m.groupdict().get("episode_end"))
|
||||
|
||||
if 1900 <= ep_val <= 2099 and not has_range and not is_known_anime and not has_explicit_season:
|
||||
tokens.year = ep_val
|
||||
tokens.title = title_clean
|
||||
tokens.group = grp_name
|
||||
tokens.is_anime = False
|
||||
tokens.is_episodic = False
|
||||
return tokens
|
||||
else:
|
||||
tokens.is_anime = True
|
||||
tokens.group = grp_name
|
||||
tokens.title = title_clean
|
||||
tokens.episode = ep_val
|
||||
if anime_m.groupdict().get("ep_title"):
|
||||
tokens.episode_title = self._clean_title(anime_m.group("ep_title"))
|
||||
if anime_m.groupdict().get("episode_end"):
|
||||
tokens.multi_episodes = list(range(ep_val, int(anime_m.group("episode_end")) + 1))
|
||||
tokens.season = self._extract_season_from_path(file_path) or 1
|
||||
tokens.is_episodic = True
|
||||
return tokens
|
||||
|
||||
# 9. Check Anime movie format: [Judas] Fate Stay Night... [BD 1080p]
|
||||
anime_mov_m = RE_ANIME_MOVIE.match(stem)
|
||||
if anime_mov_m:
|
||||
grp = anime_mov_m.group("group").strip()
|
||||
if grp.lower() in KNOWN_ANIME_GROUPS:
|
||||
tokens.group = grp
|
||||
tokens.is_anime = True
|
||||
tokens.title = self._clean_title(anime_mov_m.group("title"), preserve_dots=True)
|
||||
return tokens
|
||||
|
||||
# 10. Check for Music track pattern
|
||||
mus_m = RE_MUSIC_TRACK.match(stem)
|
||||
if mus_m:
|
||||
tokens.is_music = True
|
||||
tokens.track = int(mus_m.group("track"))
|
||||
if mus_m.group("disc"):
|
||||
tokens.disc = int(mus_m.group("disc"))
|
||||
tokens.title = self._clean_title(mus_m.group("title"))
|
||||
|
||||
parent = file_path.parent
|
||||
if parent and parent.name:
|
||||
parts = parent.name.split(" - ")
|
||||
if len(parts) >= 2:
|
||||
tokens.artist = parts[0].strip()
|
||||
tokens.album = parts[1].strip()
|
||||
return tokens
|
||||
|
||||
# 11. Movie pattern: Title (Year) or Title.Year.Quality
|
||||
# Parenthesized year first
|
||||
paren_yr = re.search(r"\((19\d{2}|20\d{2})\)", stem)
|
||||
if paren_yr:
|
||||
tokens.year = int(paren_yr.group(1))
|
||||
prefix = stem[: paren_yr.start()]
|
||||
tokens.title = self._clean_title(prefix)
|
||||
grp_m = RE_RELEASE_GROUP_UPGRADED.search(stem)
|
||||
if grp_m:
|
||||
tokens.group = grp_m.group("grp_bracket") or grp_m.group("grp_plain")
|
||||
return tokens
|
||||
|
||||
# Delimiter-based right-to-left year detection
|
||||
tech_start = len(stem)
|
||||
for m in RE_TECH_ALL.finditer(stem):
|
||||
if m.start() < tech_start:
|
||||
tech_start = m.start()
|
||||
|
||||
year_matches = list(RE_YEAR_BOUND.finditer(stem))
|
||||
if year_matches:
|
||||
valid_matches = [m for m in year_matches if m.start() <= tech_start]
|
||||
if not valid_matches:
|
||||
valid_matches = year_matches
|
||||
best_match = valid_matches[-1]
|
||||
tokens.year = int(best_match.group(1))
|
||||
prefix = stem[: best_match.start()]
|
||||
tokens.title = self._clean_title(prefix)
|
||||
grp_m = RE_RELEASE_GROUP_UPGRADED.search(stem)
|
||||
if grp_m:
|
||||
tokens.group = grp_m.group("grp_bracket") or grp_m.group("grp_plain")
|
||||
return tokens
|
||||
|
||||
# Fallback: strip tech specs and clean whole stem as title
|
||||
prefix = stem[:tech_start].strip(" .-_")
|
||||
tokens.title = self._clean_title(prefix if prefix else stem)
|
||||
grp_m = RE_RELEASE_GROUP_UPGRADED.search(stem)
|
||||
if grp_m:
|
||||
tokens.group = grp_m.group("grp_bracket") or grp_m.group("grp_plain")
|
||||
return tokens
|
||||
|
||||
def _clean_title(self, raw: str, preserve_paren: bool = False, preserve_dots: bool = False) -> str:
|
||||
"""Replace dots, underscores, and scene separators with clean spaces."""
|
||||
# Strip leading bracket tags like [YTS.MX] or [SubsPlease] if present
|
||||
raw = re.sub(r"^\s*\[[^\]]+\]\s*", "", raw)
|
||||
|
||||
# Protect decimal numbers like 2.5, 3.5, 4.5, 1.5, 0.5, 1.11, etc.
|
||||
raw = re.sub(r"(?<=\d)\.(?=\d)", "PROTECTEDDECIMALDOT", raw)
|
||||
|
||||
if preserve_dots:
|
||||
cleaned = re.sub(r"_+", " ", raw).strip()
|
||||
else:
|
||||
# Replace dots with space, except if dot is followed by space in Roman numeral (e.g. "I. ")
|
||||
cleaned = re.sub(r"(?<=\b[IVXLCDM])\.\s+", "._KEEP_DOT_SPACE_", raw)
|
||||
cleaned = re.sub(r"[\._]+", " ", cleaned)
|
||||
cleaned = cleaned.replace("._KEEP_DOT_SPACE_", ". ")
|
||||
|
||||
cleaned = cleaned.replace("PROTECTEDDECIMALDOT", ".")
|
||||
cleaned = re.sub(r"\s+", " ", cleaned).strip()
|
||||
|
||||
if preserve_paren and re.search(r"\([12]\d{3}\)$", cleaned):
|
||||
# Do not strip trailing parenthesis if it's (Year)
|
||||
pass
|
||||
else:
|
||||
cleaned = re.sub(r"[\-\(\)\[\]]+$", "", cleaned).strip()
|
||||
|
||||
return cleaned
|
||||
|
||||
def _extract_episode_title(self, suffix: str) -> Optional[str]:
|
||||
"""Extract episode title from string after SxxExx marker, stripping tech tags."""
|
||||
s = suffix.strip(" .-_")
|
||||
if not s:
|
||||
return None
|
||||
# Split by known tech tags
|
||||
for reg in (RE_RESOLUTION, RE_SOURCE, RE_VIDEO_CODEC, RE_AUDIO_CODEC, RE_EDITION):
|
||||
m = reg.search(s)
|
||||
if m:
|
||||
s = s[: m.start()].strip(" .-_")
|
||||
|
||||
# Strip release group
|
||||
grp = RE_RELEASE_GROUP_UPGRADED.search(s)
|
||||
if grp:
|
||||
s = s[: grp.start()].strip(" .-_")
|
||||
|
||||
cleaned = self._clean_title(s)
|
||||
return cleaned if cleaned else None
|
||||
|
||||
def _extract_season_from_path(self, file_path: Path) -> Optional[int]:
|
||||
"""Attempt to extract season number from parent directory names like 'Season 2', 'Season 08 - Water Seven', or 'S03'."""
|
||||
try:
|
||||
system_folders = {
|
||||
"downloads", "jdownloads", "media", "completed", "incomplete", "torrent",
|
||||
"torrents", "root", "home", "mnt", "md0", "storage", "tmp", "temp", "var", "etc", "usr"
|
||||
}
|
||||
for part in reversed(file_path.parts[:-1]):
|
||||
if part.lower().strip() in system_folders:
|
||||
break
|
||||
m = re.search(r"(?i)\b(?:season|series|s)\s*(\d{1,2})\b", part)
|
||||
if m:
|
||||
return int(m.group(1))
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
def _extract_title_from_context(self, file_path: Path) -> Optional[str]:
|
||||
"""Attempt to extract show title from immediate parent directory (e.g. Show/Episode 01.mkv or Show/Season 1/Ep01.mkv)."""
|
||||
try:
|
||||
parent = file_path.parent
|
||||
if not parent or str(parent) in ("/", ".", ""):
|
||||
return None
|
||||
p_name = parent.name
|
||||
if re.search(r"(?i)\b(?:season|series|s)\s*\d+\b", p_name):
|
||||
# Ascend one level if inside a season folder
|
||||
parent = parent.parent
|
||||
p_name = parent.name if parent else ""
|
||||
if not p_name:
|
||||
return None
|
||||
p_lower = p_name.lower().strip()
|
||||
system_folders = {
|
||||
"downloads", "jdownloads", "media", "completed", "incomplete", "torrent",
|
||||
"torrents", "root", "home", "mnt", "md0", "storage", "tmp", "temp", "var", "etc", "usr"
|
||||
}
|
||||
if p_lower in system_folders:
|
||||
return None
|
||||
clean = self._clean_title(p_name)
|
||||
if len(clean) >= 2:
|
||||
return clean
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
Reference in new issue
Block a user