Files
HS-Rename/engine/episode_match.py
T

520 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
Match episode titles from filenames against a reference episode list (e.g. TheTVDB).
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from difflib import SequenceMatcher
from pathlib import Path
from typing import Optional
from .tvdb_client import TvdbEpisode
# S01E05 - Title or S01E05-E06 - Title
PATTERN_SXXEXX = re.compile(
r"^(.*?)([Ss])(\d+)([Ee])(\d+)(-[Ee](\d+))?(.*)$",
)
# Show Name 04x01 Title, 4x01 - Title, etc.
PATTERN_NXNN = re.compile(
r"^(.*?)(\d{1,2})[xX](\d{1,4})(?:([\s._-]+)(.+))?$",
)
DEFAULT_EPISODE_PATTERN = PATTERN_SXXEXX.pattern
@dataclass(frozen=True)
class EpisodeTarget:
season: int
episode: int
episode_end: Optional[int] = None
@property
def span(self) -> int:
if self.episode_end is not None and self.episode_end > self.episode:
return self.episode_end - self.episode + 1
return 1
def format_code(self, padding: int = 2) -> str:
pad = max(1, padding)
s = str(self.season).zfill(pad)
e1 = str(self.episode).zfill(pad)
if self.episode_end is not None and self.episode_end > self.episode:
e2 = str(self.episode_end).zfill(pad)
return f"S{s}E{e1}-E{e2}"
return f"S{s}E{e1}"
def target_to_tuple(target: EpisodeTarget) -> tuple[int, ...]:
"""Plain tuple safe to pass through Qt signals."""
if target.episode_end is not None and target.episode_end > target.episode:
return (target.season, target.episode, target.episode_end)
return (target.season, target.episode)
def split_combined_title(name: str) -> list[str]:
"""Split a combined-order episode title like 'Ep A/Ep B' into parts."""
return [p.strip() for p in name.replace(" / ", "/").split("/") if p.strip()]
def normalize_title(title: str) -> str:
"""Lowercase, strip punctuation, collapse whitespace for fuzzy comparison."""
t = title.lower()
t = re.sub(r"[^\w\s]", " ", t, flags=re.UNICODE)
t = re.sub(r"\s+", " ", t).strip()
if t.startswith("the "):
t = t[4:]
return t
def _clean_title(rest: str) -> str:
title = rest.strip()
for prefix in ("- ", " ", "_ ", ". "):
if title.startswith(prefix):
title = title[len(prefix) :].strip()
if title.startswith("-") or title.startswith(""):
title = title[1:].strip()
if title.startswith("_"):
title = title[1:].strip()
title = re.sub(
r"\(\s*(?:2160p|1080p|720p|480p|4k|uhd|hd|sd|web[- ]?dl|bluray|dvdrip)\s*\)",
"",
title,
flags=re.IGNORECASE,
)
title = re.sub(r"\s+", " ", title).strip()
return title
def parse_episode_stem(stem: str, pattern: str = DEFAULT_EPISODE_PATTERN) -> Optional[dict]:
"""
Parse common TV filename stems (S01E05 or 04x01 styles).
Returns dict with format, season, episode, title, padding hints — or None.
"""
del pattern # legacy param; auto-detect formats instead
m = PATTERN_SXXEXX.match(stem)
if m:
prefix, s_letter, season_s, e_letter, ep_s, _range_block, ep2_s, rest = m.groups()
try:
season = int(season_s)
old_first = int(ep_s)
except ValueError:
return None
span = 1
if ep2_s is not None:
try:
old_second = int(ep2_s)
except ValueError:
return None
span = old_second - old_first + 1
if span < 1:
span = 1
title = _clean_title(rest)
return {
"format": "sxxexx",
"prefix": prefix,
"s_letter": s_letter,
"e_letter": e_letter,
"season": season,
"season_pad": len(season_s),
"episode_pad": len(ep_s),
"old_first": old_first,
"span": span,
"title": title,
"suffix": rest,
}
m = PATTERN_NXNN.match(stem)
if m:
prefix, season_s, ep_s, sep, title_part = m.groups()
try:
season = int(season_s)
episode = int(ep_s)
except ValueError:
return None
title = _clean_title(title_part or "")
sep = sep or " "
if title and not sep.strip():
sep = " "
return {
"format": "nxnn",
"prefix": prefix,
"season": season,
"season_pad": len(season_s),
"episode_pad": len(ep_s),
"old_first": episode,
"span": 1,
"title": title,
"title_sep": sep,
}
return None
def rewrite_episode_stem(
stem: str,
target: EpisodeTarget,
padding: int = 2,
pattern: str = DEFAULT_EPISODE_PATTERN,
) -> str:
"""Replace season/episode block in stem, preserving layout and title."""
parsed = parse_episode_stem(stem, pattern)
if not parsed:
return stem
pad = max(1, padding)
new_season = target.season
new_ep = target.episode
if target.episode_end is not None and target.episode_end > target.episode:
span = target.episode_end - target.episode + 1
else:
span = parsed["span"]
title = parsed.get("title") or ""
if parsed["format"] == "nxnn":
s_pad = max(parsed["season_pad"], len(str(new_season)))
e_pad = max(parsed["episode_pad"], pad, len(str(new_ep)))
block = f"{new_season:0{s_pad}d}x{new_ep:0{e_pad}d}"
if title:
return f"{parsed['prefix']}{block}{parsed['title_sep']}{title}"
return f"{parsed['prefix']}{block}"
s_pad = max(parsed["season_pad"], len(str(new_season)))
e_pad = max(parsed["episode_pad"], pad, len(str(new_ep)))
e1 = str(new_ep).zfill(e_pad)
head = (
f"{parsed['prefix']}{parsed['s_letter']}{new_season:0{s_pad}d}{parsed['e_letter']}"
)
if span <= 1:
return f"{head}{e1}{parsed['suffix']}"
e2 = str(new_ep + span - 1).zfill(e_pad)
range_prefix = f"-{parsed['e_letter']}"
return f"{head}{e1}{range_prefix}{e2}{parsed['suffix']}"
def rewrite_episode_number(
stem: str,
new_first_ep: int,
padding: int = 2,
pattern: str = DEFAULT_EPISODE_PATTERN,
) -> str:
"""Legacy helper: episode only, keep season from filename."""
parsed = parse_episode_stem(stem, pattern)
if not parsed:
return stem
return rewrite_episode_stem(
stem,
EpisodeTarget(season=parsed["season"], episode=new_first_ep),
padding=padding,
pattern=pattern,
)
def _similarity(a: str, b: str) -> float:
if not a or not b:
return 0.0
if a == b:
return 1.0
return SequenceMatcher(None, a, b).ratio()
def _title_match_score(fnorm: str, enorm: str) -> float:
"""Fuzzy title match; handles filenames that use a shortened episode title."""
if not fnorm or not enorm:
return 0.0
if fnorm == enorm:
return 1.0
score = _similarity(fnorm, enorm)
if len(fnorm) >= 4 and (fnorm in enorm or enorm.startswith(fnorm)):
score = max(score, 0.78)
ftokens = [t for t in fnorm.split() if len(t) > 2]
if ftokens:
etokens = set(enorm.split())
overlap = sum(1 for t in ftokens if t in etokens) / len(ftokens)
if overlap >= 0.75:
score = max(score, 0.66 + overlap * 0.3)
return score
def _episode_number_boost(parsed: dict, season: int, ep_num: int, title_score: float) -> float:
"""Nudge score up when NxNN/SxxExx in the filename agrees with this episode."""
if title_score < 0.45:
return 0.0
if parsed.get("season") == season and parsed.get("old_first") == ep_num:
return 0.12
return 0.0
def _coerce_target(
value: EpisodeTarget | tuple[int, ...] | int,
parsed: dict,
) -> EpisodeTarget:
if isinstance(value, EpisodeTarget):
return value
if isinstance(value, tuple):
if len(value) >= 3:
return EpisodeTarget(
season=int(value[0]),
episode=int(value[1]),
episode_end=int(value[2]) if value[2] is not None else None,
)
if len(value) >= 2:
return EpisodeTarget(season=int(value[0]), episode=int(value[1]))
return EpisodeTarget(season=parsed["season"], episode=int(value[0]))
return EpisodeTarget(season=parsed["season"], episode=int(value))
def resolve_combined_to_official(
combined_ep: TvdbEpisode,
official_episodes: list[TvdbEpisode],
) -> Optional[EpisodeTarget]:
"""Map a combined-order episode to official aired SxxExx(-Exx) numbers."""
parts = split_combined_title(combined_ep.name)
if not parts:
return None
matched: list[int] = []
season = combined_ep.season_number
season_official = [ep for ep in official_episodes if ep.season_number == season]
for part in parts:
pn = normalize_title(part)
best_num: Optional[int] = None
best_score = 0.0
for ep in season_official:
score = _similarity(pn, normalize_title(ep.name))
if score > best_score:
best_score = score
best_num = ep.number
if best_num is not None and best_score >= 0.72:
matched.append(best_num)
if not matched:
return None
start, end = min(matched), max(matched)
return EpisodeTarget(
season=season,
episode=start,
episode_end=end if end > start else None,
)
def _combined_title_variants(name: str) -> list[str]:
variants = [normalize_title(name.replace("/", " ")), normalize_title(name)]
for part in split_combined_title(name):
variants.append(normalize_title(part))
return variants
def _combined_match_score(fnorm: str, combined_name: str, file_title: str = "") -> float:
variants = _combined_title_variants(combined_name)
best = max((_title_match_score(fnorm, v) for v in variants), default=0.0)
if file_title and _dual_title_matches_combined(file_title, combined_name):
best = max(best, 0.96)
return best
def _dual_title_matches_combined(file_title: str, combined_name: str) -> bool:
"""True when the filename lists both stories in a combined-order pair."""
file_parts = [p.strip() for p in re.split(r"\s+-\s+", file_title) if p.strip()]
combined_parts = split_combined_title(combined_name)
if len(file_parts) < 2 or len(combined_parts) < 2:
return False
matched = 0
for fp in file_parts[: len(combined_parts)]:
fn = normalize_title(fp)
if max(_similarity(fn, normalize_title(cp)) for cp in combined_parts) >= 0.72:
matched += 1
return matched >= 2
def _combined_allowed_for_file(
parsed: dict,
target: EpisodeTarget,
combined_name: str = "",
) -> bool:
"""Treat as multi-episode when the episode tag or dual title fits the aired range."""
if target.span <= 1:
return True
if parsed.get("span", 1) > 1:
return True
file_title = parsed.get("title") or ""
if combined_name and _dual_title_matches_combined(file_title, combined_name):
return True
file_ep = parsed.get("old_first")
if file_ep is None:
return False
end = target.episode_end if target.episode_end is not None else target.episode
return file_ep == target.episode or file_ep == end
def _range_overlaps(
season: int,
start: int,
end: int,
used_ranges: list[tuple[int, int, int]],
) -> bool:
for s, a, b in used_ranges:
if s != season:
continue
if not (end < a or start > b):
return True
return False
def _target_range(target: EpisodeTarget) -> tuple[int, int, int]:
end = target.episode_end if target.episode_end is not None else target.episode
return target.season, target.episode, end
def _apply_season_hint(score: float, target: EpisodeTarget, parsed: dict) -> float:
"""Prefer matches in the same season as the filename's episode code."""
file_season = parsed.get("season")
if file_season is None:
return score
if target.season == file_season:
return score + 0.08
return score * 0.5
def match_filenames_to_episodes(
filenames: list[str],
episodes: list[TvdbEpisode],
pattern: str = DEFAULT_EPISODE_PATTERN,
min_score: float = 0.65,
season_filter: int = 0,
official_episodes: Optional[list[TvdbEpisode]] = None,
combined_episodes: Optional[list[TvdbEpisode]] = None,
) -> tuple[dict[str, EpisodeTarget], list[str], list[str]]:
"""
Match filenames to TheTVDB episodes by title.
season_filter: 0 = use all episodes; else only episodes from that season.
official_episodes + combined_episodes: when set, also match combined-order titles
and map to official aired numbers as Jellyfin multi-episode ranges (S01E01-E02).
Returns mapping filename -> target, unmatched list, notes.
"""
if season_filter > 0:
episodes = [ep for ep in episodes if ep.season_number == season_filter]
official = official_episodes or episodes
if season_filter > 0:
official = [ep for ep in official if ep.season_number == season_filter]
combined = combined_episodes
if combined and season_filter > 0:
combined = [ep for ep in combined if ep.season_number == season_filter]
file_entries: list[tuple[str, str, str, dict]] = []
for name in filenames:
base = Path(name).name
stem = base.rsplit(".", 1)[0] if "." in base and not base.startswith(".") else base
parsed = parse_episode_stem(stem, pattern)
if not parsed:
continue
if season_filter > 0 and parsed["season"] != season_filter:
continue
norm = normalize_title(parsed["title"]) if parsed["title"] else ""
if norm or parsed.get("span", 1) > 1:
file_entries.append((name, norm, parsed["title"], parsed))
ep_entries = [
(ep.season_number, ep.number, normalize_title(ep.name), ep.name)
for ep in episodes
]
combined_entries: list[tuple[TvdbEpisode, EpisodeTarget, list[str]]] = []
if combined:
for cep in combined:
target = resolve_combined_to_official(cep, official)
if target is None:
continue
combined_entries.append((cep, target, _combined_title_variants(cep.name)))
pairs: list[tuple[float, str, EpisodeTarget, str, str]] = []
for fname, fnorm, raw_title, parsed in file_entries:
file_season = parsed.get("season")
if season_filter == 0 and file_season is not None:
season_eps = [e for e in ep_entries if e[0] == file_season]
season_combined = [
(cep, target, variants)
for cep, target, variants in combined_entries
if target.season == file_season
]
else:
season_eps = ep_entries
season_combined = combined_entries
for season, ep_num, enorm, ep_name in season_eps:
if not fnorm:
continue
title_score = _title_match_score(fnorm, enorm)
score = title_score + _episode_number_boost(parsed, season, ep_num, title_score)
score = _apply_season_hint(
score,
EpisodeTarget(season=season, episode=ep_num),
parsed,
)
pairs.append(
(
score,
fname,
EpisodeTarget(season=season, episode=ep_num),
raw_title,
ep_name,
)
)
for _cep, target, variants in season_combined:
if not fnorm:
continue
if not _combined_allowed_for_file(parsed, target, _cep.name):
continue
best = _combined_match_score(fnorm, _cep.name, raw_title)
best = _apply_season_hint(best, target, parsed)
pairs.append((best, fname, target, raw_title, _cep.name))
pairs.sort(key=lambda x: (-x[0], -x[2].span, x[1], x[2].season, x[2].episode))
mapping: dict[str, EpisodeTarget] = {}
used_files: set[str] = set()
used_ranges: list[tuple[int, int, int]] = []
notes: list[str] = []
multi_file_names = {
fname for fname, _fnorm, _raw, parsed in file_entries if parsed.get("span", 1) > 1
}
def _assign_pairs(candidates: list[tuple[float, str, EpisodeTarget, str, str]]) -> None:
for score, fname, target, raw_title, ep_name in candidates:
if score < min_score:
break
if fname in used_files:
continue
season, start, end = _target_range(target)
if _range_overlaps(season, start, end, used_ranges):
continue
mapping[fname] = target
used_files.add(fname)
used_ranges.append((season, start, end))
pct = min(100, int(round(score * 100)))
code = target.format_code()
notes.append(
f"{fname}: {code} ← “{ep_name}” ({pct}% match, file title “{raw_title}”)"
)
multi_pairs = [p for p in pairs if p[1] in multi_file_names]
other_pairs = [p for p in pairs if p[1] not in multi_file_names]
_assign_pairs(multi_pairs)
_assign_pairs(other_pairs)
unmatched_files = [
name for name in filenames
if name not in mapping
and parse_episode_stem(
(
Path(name).name.rsplit(".", 1)[0]
if "." in Path(name).name and not Path(name).name.startswith(".")
else Path(name).name
),
pattern,
)
]
return mapping, unmatched_files, notes