Add TheTVDB episode title matching and release v1.0.3.

Match filenames to TheTVDB episode order (default, aired, DVD, etc.) to fix chaotic SxxExx numbering while preserving titles.

Co-authored-by: Cursor <[email protected]>
This commit is contained in:
Bulk Renamer
2026-07-03 16:47:37 -05:00
co-authored by Cursor
parent 59dba20c1e
commit 085d6dc296
8 changed files with 609 additions and 5 deletions
+159
View File
@@ -0,0 +1,159 @@
"""
Match episode titles from filenames against a reference episode list (e.g. TheTVDB).
"""
from __future__ import annotations
import re
from difflib import SequenceMatcher
from typing import Optional
from .tvdb_client import TvdbEpisode
DEFAULT_EPISODE_PATTERN = r"(.*?[Ss]\d+[Ee])(\d+)(-[Ee](\d+))?(.*)"
def normalize_title(title: str) -> str:
"""Lowercase, strip punctuation, collapse whitespace for fuzzy comparison."""
t = title.lower()
t = re.sub(r"[^\w\s]", " ", t, flags=re.UNICODE)
return re.sub(r"\s+", " ", t).strip()
def parse_episode_stem(stem: str, pattern: str = DEFAULT_EPISODE_PATTERN) -> Optional[dict]:
"""
Parse SxxExx-style filename stem.
Returns dict with prefix, episode numbers, title rest, and span — or None if no match.
"""
try:
m = re.match(pattern, stem)
except re.error:
return None
if not m:
return None
prefix, old_first_s, _range_dash_e, old_second_s, rest = (
m.group(1),
m.group(2),
m.group(3),
m.group(4),
m.group(5),
)
try:
old_first = int(old_first_s)
except ValueError:
return None
span = 1
if old_second_s is not None:
try:
old_second = int(old_second_s)
except ValueError:
return None
span = old_second - old_first + 1
if span < 1:
span = 1
title = rest.strip()
if title.startswith("-") or title.startswith(""):
title = title[1:].strip()
if title.startswith("_"):
title = title[1:].strip()
return {
"prefix": prefix,
"old_first": old_first,
"span": span,
"title": title,
}
def rewrite_episode_number(
stem: str,
new_first_ep: int,
padding: int = 2,
pattern: str = DEFAULT_EPISODE_PATTERN,
) -> str:
"""Replace episode number block in stem, preserving prefix, span, and title."""
parsed = parse_episode_stem(stem, pattern)
if not parsed:
return stem
try:
m = re.match(pattern, stem)
except re.error:
return stem
if not m:
return stem
prefix = m.group(1)
rest = m.group(5)
span = parsed["span"]
pad = max(1, padding)
e1 = str(new_first_ep).zfill(pad)
if span <= 1:
return f"{prefix}{e1}{rest}"
e2 = str(new_first_ep + span - 1).zfill(pad)
return f"{prefix}{e1}-E{e2}{rest}"
def _similarity(a: str, b: str) -> float:
if not a or not b:
return 0.0
if a == b:
return 1.0
return SequenceMatcher(None, a, b).ratio()
def match_filenames_to_episodes(
filenames: list[str],
episodes: list[TvdbEpisode],
pattern: str = DEFAULT_EPISODE_PATTERN,
min_score: float = 0.72,
) -> tuple[dict[str, int], list[str], list[str]]:
"""
Match filenames to TheTVDB episode numbers by title.
Returns:
mapping: original filename -> correct episode number
unmatched_files: filenames that could not be matched
notes: human-readable match details for UI
"""
file_entries: list[tuple[str, str, str]] = []
for name in filenames:
stem = name.rsplit(".", 1)[0] if "." in name and not name.startswith(".") else name
parsed = parse_episode_stem(stem, pattern)
if not parsed or not parsed["title"]:
continue
norm = normalize_title(parsed["title"])
if norm:
file_entries.append((name, norm, parsed["title"]))
ep_entries = [(ep.number, normalize_title(ep.name), ep.name) for ep in episodes]
pairs: list[tuple[float, str, int, str, str]] = []
for fname, fnorm, raw_title in file_entries:
for ep_num, enorm, ep_name in ep_entries:
score = _similarity(fnorm, enorm)
pairs.append((score, fname, ep_num, raw_title, ep_name))
pairs.sort(key=lambda x: (-x[0], x[1], x[2]))
mapping: dict[str, int] = {}
used_files: set[str] = set()
used_eps: set[int] = set()
notes: list[str] = []
for score, fname, ep_num, raw_title, ep_name in pairs:
if score < min_score:
break
if fname in used_files or ep_num in used_eps:
continue
mapping[fname] = ep_num
used_files.add(fname)
used_eps.add(ep_num)
pct = int(round(score * 100))
notes.append(f"{fname}: E{ep_num:02d} ← “{ep_name}” ({pct}% match, file title “{raw_title}”)")
unmatched_files = [
name for name in filenames
if name not in mapping
and parse_episode_stem(
name.rsplit(".", 1)[0] if "." in name and not name.startswith(".") else name,
pattern,
)
]
return mapping, unmatched_files, notes