Initial commit
This commit is contained in:
commit
1d235d30e7
58 files changed
+19693
No files matched your search
@@ -0,0 +1,413 @@
|
||||
"""Naming and path formatting engine for Media Sorter.
|
||||
|
||||
Renders user-defined naming templates, safely formats multi-part tags, pairs sidecars
|
||||
with primary media files, and enforces rigorous cross-platform filename sanitization
|
||||
(Linux, Windows, macOS, NTFS, SMB/NFS, exFAT).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from .classifier import ClassificationResult
|
||||
from .config import Settings
|
||||
|
||||
# Windows reserved device names
|
||||
RESERVED_NAMES = {
|
||||
"CON", "PRN", "AUX", "NUL",
|
||||
"COM1", "COM2", "COM3", "COM4", "COM5", "COM6", "COM7", "COM8", "COM9",
|
||||
"LPT1", "LPT2", "LPT3", "LPT4", "LPT5", "LPT6", "LPT7", "LPT8", "LPT9",
|
||||
}
|
||||
|
||||
# Illegal characters across file systems (< > : " / \ | ? *)
|
||||
FORBIDDEN_CHARS_PATTERN = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
|
||||
|
||||
|
||||
def sanitize_filename_component(name: str, max_length: int = 240) -> str:
|
||||
"""Sanitize an individual filename or folder name component for safe cross-platform use."""
|
||||
# 1. Unicode normalization (NFC)
|
||||
clean = unicodedata.normalize("NFC", name)
|
||||
|
||||
# 2. Replace forbidden characters with safe hyphen or space
|
||||
clean = FORBIDDEN_CHARS_PATTERN.sub("-", clean)
|
||||
|
||||
# 3. Collapse multiple whitespace and hyphens
|
||||
clean = re.sub(r"\s+", " ", clean)
|
||||
clean = re.sub(r"-{2,}", "-", clean)
|
||||
|
||||
# 4. Strip leading/trailing spaces, dots, and hyphens (vital for Windows / SMB)
|
||||
clean = clean.strip(" .-")
|
||||
|
||||
if not clean:
|
||||
clean = "unnamed"
|
||||
|
||||
# 5. Check Windows reserved words
|
||||
upper_base = clean.split(".")[0].upper()
|
||||
if upper_base in RESERVED_NAMES:
|
||||
clean = f"_{clean}"
|
||||
|
||||
# 6. Truncate byte length for filesystem limits (e.g. 255 bytes on ext4/NTFS/ZFS)
|
||||
encoded = clean.encode("utf-8")
|
||||
if len(encoded) > max_length:
|
||||
parts = clean.rsplit(".", 1)
|
||||
if len(parts) == 2 and 1 <= len(parts[1]) <= 10:
|
||||
base, ext = parts
|
||||
ext_bytes = len(f".{ext}".encode("utf-8"))
|
||||
avail = max(max_length - ext_bytes, 10)
|
||||
base_enc = base.encode("utf-8")[:avail]
|
||||
base_clean = base_enc.decode("utf-8", errors="ignore").rstrip(" .-")
|
||||
clean = f"{base_clean}.{ext}" if base_clean else ext
|
||||
else:
|
||||
clean = encoded[:max_length].decode("utf-8", errors="ignore").rstrip(" .-")
|
||||
|
||||
clean = clean.strip(" .-")
|
||||
return clean if clean else "unnamed"
|
||||
|
||||
|
||||
DEFAULT_TEMPLATES = {
|
||||
"tv": "{title}/Season {season:02d}/{show_name}_{season_episode}.{ext}",
|
||||
"movie": "{title} ({year})/{movie_name}.{ext}",
|
||||
"anime": "{title}/Season {season:02d}/{show_name}_{season_episode} [{group}].{ext}",
|
||||
}
|
||||
|
||||
|
||||
class MediaNamer:
|
||||
"""Renders organized destination paths from templates and classification results."""
|
||||
|
||||
def __init__(self, settings: Settings):
|
||||
self.settings = settings
|
||||
|
||||
def generate_destination_path(
|
||||
self,
|
||||
cls_result: ClassificationResult,
|
||||
primary_dst_path: Optional[Path] = None,
|
||||
) -> Path:
|
||||
"""Construct full destination path for a given file and its classification."""
|
||||
category = cls_result.category
|
||||
base_dir = self.settings.get_destination_path(category)
|
||||
src_path = cls_result.metadata.path if cls_result.metadata else Path("unknown")
|
||||
ext = src_path.suffix.lstrip(".")
|
||||
|
||||
# Handle Quarantine routing
|
||||
if cls_result.needs_quarantine or category == "unknown":
|
||||
reason = cls_result.quarantine_reason or "low_confidence"
|
||||
safe_reason = sanitize_filename_component(reason)
|
||||
q_template = self.settings.templates.quarantine
|
||||
filename = sanitize_filename_component(src_path.name)
|
||||
rel_str = q_template.format(reason=safe_reason, filename=filename, ext=ext)
|
||||
return (self.settings.get_destination_path("quarantine") / rel_str).resolve()
|
||||
|
||||
# Handle Sidecars (Subtitles, Artwork, Metadata, Extras)
|
||||
if category in ("subtitle", "artwork", "metadata"):
|
||||
return self._format_sidecar_path(cls_result, primary_dst_path, base_dir)
|
||||
|
||||
# Retrieve template
|
||||
template = getattr(self.settings.templates, category, None)
|
||||
context = self._build_context(cls_result)
|
||||
|
||||
if category == "tv" and (not template or template == DEFAULT_TEMPLATES.get("tv")):
|
||||
formatted_rel = self._format_tv_path(cls_result, context)
|
||||
elif category == "movie" and (not template or template == DEFAULT_TEMPLATES.get("movie")):
|
||||
formatted_rel = self._format_movie_path(cls_result, context)
|
||||
elif category == "anime" and (not template or template == DEFAULT_TEMPLATES.get("anime")):
|
||||
formatted_rel = self._format_anime_path(cls_result, context)
|
||||
elif category == "podcast" and (not template or template == "{show}/{year}/{show} - {date} - {title}.{ext}"):
|
||||
formatted_rel = self._format_podcast_path(cls_result, context)
|
||||
else:
|
||||
if not template:
|
||||
template = "{filename}.{ext}"
|
||||
formatted_rel = self._render_template(template, context)
|
||||
|
||||
# If file renaming is disabled, preserve original source filename
|
||||
if not getattr(self.settings.general, "rename_files", True) and src_path.name != "unknown":
|
||||
rel_path = Path(formatted_rel)
|
||||
if len(rel_path.parts) > 1:
|
||||
formatted_rel = str(rel_path.parent / src_path.name)
|
||||
else:
|
||||
formatted_rel = src_path.name
|
||||
|
||||
# Sanitize each path component separately to preserve folder hierarchy
|
||||
parts = Path(formatted_rel).parts
|
||||
sanitized_parts = [sanitize_filename_component(p) for p in parts]
|
||||
return (base_dir / Path(*sanitized_parts)).resolve()
|
||||
|
||||
def _format_tv_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
|
||||
show_name = context["show_name"]
|
||||
ext = context["ext"]
|
||||
tokens = cls_result.tokens
|
||||
|
||||
# Daily / dated broadcast TV formatting
|
||||
date_val = context.get("date_val")
|
||||
if (tokens and tokens.is_daily) or (date_val and (not tokens or not tokens.season or tokens.season > 1000)):
|
||||
year = context.get("year")
|
||||
if not year or year == "Unknown":
|
||||
year = date_val.split("-")[0] if date_val else "Unknown"
|
||||
return f"{show_name}/Season {year}/{show_name} - {date_val}.{ext}"
|
||||
|
||||
# Standard TV formatting (supporting Season 00, multi-ep, and season pack)
|
||||
season_num = context["season"]
|
||||
season_folder = f"Season {season_num:02d}"
|
||||
season_episode = context["season_episode"]
|
||||
return f"{show_name}/{season_folder}/{show_name} - {season_episode}.{ext}"
|
||||
|
||||
def _format_movie_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
|
||||
title = context["title"]
|
||||
year = context["year"]
|
||||
ext = context["ext"]
|
||||
has_year = year and year != "Unknown"
|
||||
folder_name = f"{title} ({year})" if has_year else title
|
||||
base_name = f"{title} ({year})" if has_year else title
|
||||
|
||||
edition_tag = context.get("edition_tag", "")
|
||||
part_tag = context.get("part_tag", "")
|
||||
extra_tag = context.get("extra_tag", "")
|
||||
return f"{folder_name}/{base_name}{edition_tag}{part_tag}{extra_tag}.{ext}"
|
||||
|
||||
def _format_anime_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
|
||||
title = context["title"]
|
||||
ext = context["ext"]
|
||||
tokens = cls_result.tokens
|
||||
group_tag = context.get("group_tag", "")
|
||||
|
||||
# Multi-episode anime
|
||||
if tokens and tokens.multi_episodes and len(tokens.multi_episodes) >= 2:
|
||||
first_ep = tokens.multi_episodes[0]
|
||||
last_ep = tokens.multi_episodes[-1]
|
||||
ep_str = f"{first_ep:02d}-{last_ep:02d}"
|
||||
return f"{title}/{title} - {ep_str}{group_tag}.{ext}"
|
||||
|
||||
# Single episode anime
|
||||
if tokens and tokens.episode is not None:
|
||||
ep = tokens.episode
|
||||
ep_str = f"{ep:02d}" if ep < 10 else str(ep)
|
||||
return f"{title}/{title} - {ep_str}{group_tag}.{ext}"
|
||||
|
||||
# Anime movie or special without episode number
|
||||
return f"{title}/{title}{group_tag}.{ext}"
|
||||
|
||||
def _format_podcast_path(self, cls_result: ClassificationResult, context: Dict[str, Any]) -> str:
|
||||
show = context.get("show") or context.get("artist") or "Unknown Show"
|
||||
year = context.get("year")
|
||||
date = context.get("date")
|
||||
title = context.get("title")
|
||||
ext = context.get("ext")
|
||||
if title and title != show and title != "Unknown":
|
||||
return f"{show}/{year}/{show} - {date} - {title}.{ext}"
|
||||
return f"{show}/{year}/{show} - {date}.{ext}"
|
||||
|
||||
def _format_sidecar_path(
|
||||
self,
|
||||
cls_result: ClassificationResult,
|
||||
primary_dst_path: Optional[Path],
|
||||
base_dir: Path,
|
||||
) -> Path:
|
||||
src_path = cls_result.metadata.path
|
||||
ext = src_path.suffix.lstrip(".")
|
||||
|
||||
if primary_dst_path:
|
||||
parent_dir = primary_dst_path.parent
|
||||
primary_stem = primary_dst_path.stem
|
||||
|
||||
if cls_result.category == "subtitle":
|
||||
# Detect language code or compound tag in subtitle (e.g. movie.en.srt, movie.forced.srt)
|
||||
src_stem = src_path.stem
|
||||
m = re.search(
|
||||
r"\.((?:[a-zA-Z]{2,3}\.)?(?:forced|sdh|cc)|[a-zA-Z]{2,3}(?:-[a-zA-Z]{2,4})?)$",
|
||||
src_stem,
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if m:
|
||||
lang_suffix = f".{m.group(1)}"
|
||||
else:
|
||||
parts = src_stem.split(".")
|
||||
if len(parts) > 1 and len(parts[-1]) in (2, 3, 6):
|
||||
lang_suffix = f".{parts[-1]}"
|
||||
else:
|
||||
lang_suffix = ""
|
||||
new_filename = f"{primary_stem}{lang_suffix}.{ext}"
|
||||
return parent_dir / sanitize_filename_component(new_filename)
|
||||
|
||||
elif cls_result.category == "artwork":
|
||||
# e.g. poster.jpg, cover.jpg in the same movie/show folder
|
||||
return parent_dir / sanitize_filename_component(src_path.name)
|
||||
|
||||
elif cls_result.category == "metadata":
|
||||
# NFO file matches primary stem or stays alongside
|
||||
new_filename = f"{primary_stem}.{ext}"
|
||||
return parent_dir / sanitize_filename_component(new_filename)
|
||||
|
||||
# If orphan sidecar (no primary matched), place into respective folder
|
||||
sanitized_name = sanitize_filename_component(src_path.name)
|
||||
return (base_dir / sanitized_name).resolve()
|
||||
|
||||
def _build_context(self, res: ClassificationResult) -> Dict[str, Any]:
|
||||
tokens = res.tokens
|
||||
meta = res.metadata
|
||||
src_path = meta.path if meta else Path("file")
|
||||
|
||||
# Fix Season 00 / Episode 00 falsy bug
|
||||
season_num = tokens.season if (tokens and tokens.season is not None) else 1
|
||||
episode_num = tokens.episode if (tokens and tokens.episode is not None) else 1
|
||||
|
||||
# Format season_episode string with multi-episode and season pack support
|
||||
if tokens and tokens.multi_episodes and len(tokens.multi_episodes) >= 2:
|
||||
season_ep_str = f"S{season_num:02d}E{tokens.multi_episodes[0]:02d}-E{tokens.multi_episodes[-1]:02d}"
|
||||
elif tokens and (tokens.is_season_pack or (tokens.season is not None and tokens.episode is None and not getattr(tokens, "multi_episodes", None))):
|
||||
season_ep_str = f"Season {season_num:02d}"
|
||||
else:
|
||||
season_ep_str = f"S{season_num:02d}E{episode_num:02d}"
|
||||
|
||||
main_title = (tokens.title if tokens else None) or src_path.stem
|
||||
|
||||
# Clean release group: omit when unknown, NEVER emit 'UnknownGroup'
|
||||
group_val = tokens.group if (tokens and tokens.group and tokens.group != "UnknownGroup") else ""
|
||||
group_tag = f" [{group_val}]" if group_val else ""
|
||||
|
||||
# Extract movie edition, part, and extra tags
|
||||
edition_val = getattr(tokens, "edition", None) if tokens else None
|
||||
if not edition_val:
|
||||
em = re.search(r"\b(extended|directors?\.cut|remastered|criterion(?:\.collection)?|final\.cut)\b", src_path.stem, re.I)
|
||||
if em:
|
||||
raw_ed = em.group(1).lower().replace(".", " ")
|
||||
if "director" in raw_ed:
|
||||
edition_val = "Director's Cut"
|
||||
elif "criterion" in raw_ed:
|
||||
edition_val = "Criterion"
|
||||
elif "final" in raw_ed:
|
||||
edition_val = "Final Cut"
|
||||
elif "remaster" in raw_ed:
|
||||
edition_val = "Remastered"
|
||||
elif "extend" in raw_ed:
|
||||
edition_val = "Extended"
|
||||
|
||||
edition_tag = f" [{edition_val}]" if edition_val else ""
|
||||
|
||||
part_val = getattr(tokens, "part", None) if tokens else None
|
||||
part_label = getattr(tokens, "part_label", None) if tokens else None
|
||||
if part_val is None:
|
||||
pm = re.search(r"\b(?:cd|part|pt)[\.\s_-]*(\d+)\b", src_path.stem, re.I)
|
||||
if pm:
|
||||
part_val = int(pm.group(1))
|
||||
part_label = f"Pt.{part_val}"
|
||||
elif not part_label:
|
||||
part_label = f"Pt.{part_val}"
|
||||
|
||||
part_tag = f" [{part_label}]" if part_label else ""
|
||||
|
||||
extra_m = re.search(r"-(behindthescenes|deleted|trailer|featurette)\b", src_path.stem, re.I)
|
||||
extra_tag = f"-{extra_m.group(1).lower()}" if extra_m else ""
|
||||
|
||||
# Date resolution for daily TV shows and podcasts
|
||||
date_val = getattr(tokens, "air_date", None) or (tokens.date_stamp if tokens and not tokens.is_photo_or_home_video else None)
|
||||
if not date_val:
|
||||
dm = re.search(r"\b((?:19|20)\d{2})[-._](0[1-9]|1[0-2])[-._](0[1-9]|[12]\d|3[01])\b", src_path.stem)
|
||||
if dm:
|
||||
date_val = f"{dm.group(1)}-{dm.group(2)}-{dm.group(3)}"
|
||||
|
||||
ctx: Dict[str, Any] = {
|
||||
"ext": src_path.suffix.lstrip("."),
|
||||
"filename": src_path.stem,
|
||||
"title": main_title,
|
||||
"show_name": main_title,
|
||||
"SHOW_NAME": main_title,
|
||||
"movie_name": main_title,
|
||||
"MOVIE_NAME": main_title,
|
||||
"season_episode": season_ep_str,
|
||||
"SEASON_EPISODE": season_ep_str,
|
||||
"year": (tokens.year if tokens else None) or "Unknown",
|
||||
"season": season_num,
|
||||
"episode": episode_num,
|
||||
"episode_title": (tokens.episode_title if tokens else None) or f"Episode {episode_num}",
|
||||
"artist": (tokens.artist if tokens else None) or "Unknown Artist",
|
||||
"album": (tokens.album if tokens else None) or "Unknown Album",
|
||||
"track": (tokens.track if tokens else 1) or 1,
|
||||
"disc": (tokens.disc if tokens else 1) or 1,
|
||||
"group": group_val,
|
||||
"group_tag": group_tag,
|
||||
"edition": edition_val,
|
||||
"edition_tag": edition_tag,
|
||||
"part": part_val,
|
||||
"part_label": part_label,
|
||||
"part_tag": part_tag,
|
||||
"extra_tag": extra_tag,
|
||||
"date_val": date_val,
|
||||
"resolution": (tokens.resolution if tokens and tokens.resolution else (meta.resolution_label if meta else "")),
|
||||
"codec": (tokens.video_codec or (meta.codec_video if meta else "h264")),
|
||||
"author": (tokens.artist if tokens else None) or "Unknown Author",
|
||||
"chapter": (tokens.title if tokens else None) or f"Chapter {tokens.track if tokens else 1}",
|
||||
"show": (tokens.artist if tokens else None) or "Unknown Show",
|
||||
"date": date_val or ((tokens.date_stamp if tokens else "2026-01-01") or "2026-01-01"),
|
||||
"month": 1,
|
||||
"day": 1,
|
||||
"time": "000000",
|
||||
"camera": "Camera",
|
||||
"event": "Event",
|
||||
}
|
||||
|
||||
# Override from provider result if available
|
||||
if res.provider_result:
|
||||
p = res.provider_result
|
||||
if p.canonical_title:
|
||||
ctx["title"] = p.canonical_title
|
||||
ctx["show_name"] = p.canonical_title
|
||||
ctx["SHOW_NAME"] = p.canonical_title
|
||||
ctx["movie_name"] = p.canonical_title
|
||||
ctx["MOVIE_NAME"] = p.canonical_title
|
||||
if p.year:
|
||||
ctx["year"] = p.year
|
||||
if p.episode_title:
|
||||
ctx["episode_title"] = p.episode_title
|
||||
if p.artist:
|
||||
ctx["artist"] = p.artist
|
||||
if p.album:
|
||||
ctx["album"] = p.album
|
||||
|
||||
# Parse date stamp fields if present
|
||||
date_source = (tokens and tokens.date_stamp) or (ctx.get("date") if res.category in ("home_video", "photo", "podcast") else None)
|
||||
if date_source and date_source != "Unknown":
|
||||
date_parts = str(date_source).split("-")
|
||||
if len(date_parts) == 3:
|
||||
try:
|
||||
if ctx["year"] == "Unknown":
|
||||
ctx["year"] = int(date_parts[0])
|
||||
ctx["month"] = int(date_parts[1])
|
||||
ctx["day"] = int(date_parts[2])
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
# Clean tags from meta
|
||||
if meta and meta.tags:
|
||||
if "camera_model" in meta.tags:
|
||||
ctx["camera"] = sanitize_filename_component(meta.tags["camera_model"])
|
||||
if "album" in meta.tags and not ctx.get("album"):
|
||||
ctx["album"] = meta.tags["album"]
|
||||
if "artist" in meta.tags and not ctx.get("artist"):
|
||||
ctx["artist"] = meta.tags["artist"]
|
||||
|
||||
return ctx
|
||||
|
||||
def _render_template(self, template: str, context: Dict[str, Any]) -> str:
|
||||
"""Format template while gracefully cleaning empty technical brackets."""
|
||||
# Normalize <TAG> to {TAG} for convenience if users use angle brackets
|
||||
rendered = re.sub(r"<([a-zA-Z_0-9]+)>", r"{\1}", template)
|
||||
try:
|
||||
rendered = rendered.format(**context)
|
||||
except (KeyError, ValueError):
|
||||
# Safe token replacement if format specifier fails
|
||||
safe_ctx = {k: str(v) if v is not None else "" for k, v in context.items()}
|
||||
# Remove format specifiers like :02d
|
||||
simplified = re.sub(r"\{(\w+):[^}]+\}", r"{\1}", rendered)
|
||||
try:
|
||||
rendered = simplified.format(**safe_ctx)
|
||||
except Exception:
|
||||
rendered = f"{context.get('title', 'media')}.{context.get('ext', 'bin')}"
|
||||
|
||||
# Clean empty technical brackets such as "[]" or "[ ]" or "()"
|
||||
rendered = re.sub(r"\[\s*\]", "", rendered)
|
||||
rendered = re.sub(r"\(\s*\)", "", rendered)
|
||||
rendered = re.sub(r"\s{2,}", " ", rendered)
|
||||
return rendered.strip()
|
||||
Reference in new issue
Block a user