From daf0f55499c828debd3f9f62acc2d7298cc0439e Mon Sep 17 00:00:00 2001 From: Urban Date: Sat, 18 Jul 2026 22:18:19 +0200 Subject: [PATCH] matching script working --- scripts/match_filenames.py | 1367 +++++++++++++++++++++++++++++ scripts/schema.sql | 21 + scripts/test_duration_matching.py | 270 ++++++ 3 files changed, 1658 insertions(+) create mode 100755 scripts/match_filenames.py create mode 100755 scripts/test_duration_matching.py diff --git a/scripts/match_filenames.py b/scripts/match_filenames.py new file mode 100755 index 0000000..fd310f9 --- /dev/null +++ b/scripts/match_filenames.py @@ -0,0 +1,1367 @@ +#!/usr/bin/env python3 + +""" +Match local video filenames against canonical WCX movie names. + +The script is deliberately conservative and read-only. + +Matching sources: + +- Current canonical name from movie.name +- Verified aliases from movie_name_alias +- Current duration from movie.duration_seconds +- Historical durations from movie_history.duration_seconds +- Actual file duration read with ffprobe + +Important principles: + +- Duration never creates a candidate without a name or alias match. +- A close duration may resolve an otherwise ambiguous name match. +- The least-wrong duration does not win when all candidates have poor matches. +- The current canonical movie name is always displayed. +- Files and database contents are never modified. + +Examples: + + match_filenames.py /storage/disk1/X + + match_filenames.py /storage/disk1/X --recursive + + match_filenames.py /storage/disk1/X --ending mp4,avi + + match_filenames.py /storage/disk1/X \ + --ending mp4 \ + --ending avi \ + --debug +""" + +import argparse +import shutil +import sqlite3 +import subprocess +import sys +import unicodedata +from dataclasses import dataclass, replace +from pathlib import Path + + +DEFAULT_DATABASE_FILE = Path( + "/storage/disk1/WCX/database/wcx.db" +) + +VIDEO_EXTENSIONS = { + ".avi", + ".m4v", + ".mkv", + ".mov", + ".mp4", + ".mpeg", + ".mpg", + ".ts", + ".webm", + ".wmv", +} + +MIN_NAME_SCORE = 100 +MIN_NAME_MARGIN = 15 + +# At least this score is considered credible duration support. +MIN_CREDIBLE_DURATION_SCORE = 10 + +# Difference between duration scores required to resolve several name matches. +MIN_DURATION_SCORE_MARGIN = 15 + + +@dataclass(frozen=True) +class MatchName: + """ + One searchable name belonging to a movie. + + source is normally: + current + manual + history + csv + """ + + display_name: str + normalized_name: str + normalized_persons: tuple[str, ...] + source: str + + +@dataclass(frozen=True) +class DurationVersion: + duration_seconds: int + source: str + archived_at: str | None = None + + +@dataclass(frozen=True) +class Movie: + movie_id: str + name: str + match_names: tuple[MatchName, ...] + duration_versions: tuple[DurationVersion, ...] + + +@dataclass(frozen=True) +class DurationMatch: + score: int + classification: str + matched_duration: int | None + difference_seconds: int | None + source: str | None + archived_at: str | None + + +@dataclass(frozen=True) +class MatchCandidate: + movie: Movie + matched_name: MatchName + + name_score: int + duration_score: int + total_score: int + + name_reason: str + duration_reason: str + + full_name_occurrences: int + all_persons_matched: bool + duration_match: DurationMatch + + +@dataclass(frozen=True) +class MatchResult: + status: str + best: MatchCandidate | None + margin: int + reason: str + detected_movies: tuple[str, ...] + + +def parse_arguments() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description=( + "Match local video filenames against canonical WCX movie names." + ) + ) + + parser.add_argument( + "directory", + type=Path, + help="Directory containing video files.", + ) + + parser.add_argument( + "--database", + type=Path, + default=DEFAULT_DATABASE_FILE, + help=f"SQLite database; default: {DEFAULT_DATABASE_FILE}", + ) + + parser.add_argument( + "--recursive", + action="store_true", + help="Search recursively below the input directory.", + ) + + parser.add_argument( + "--ending", + action="append", + help=( + "Video file endings to include. Examples: " + "'--ending mp4,avi' or '--ending mp4 --ending avi'. " + "A leading dot is optional. " + "Default: all supported video endings." + ), + ) + + parser.add_argument( + "--limit", + type=int, + help="Analyze at most N files.", + ) + + parser.add_argument( + "--debug", + action="store_true", + help="Show candidates, scores, durations and classification details.", + ) + + return parser.parse_args() + + +def normalize_text(value: str) -> str: + decomposed = unicodedata.normalize("NFKD", value) + + without_diacritics = "".join( + character + for character in decomposed + if not unicodedata.combining(character) + ) + + return "".join( + character.casefold() + for character in without_diacritics + if character.isalnum() + ) + + +def split_person_names(value: str) -> tuple[str, ...]: + persons = [] + + for part in value.split("+"): + normalized = normalize_text(part) + + if normalized: + persons.append(normalized) + + return tuple(persons) + + +def normalize_duration(value: object) -> int | None: + if value is None: + return None + + try: + duration = int(value) + except (TypeError, ValueError): + return None + + if duration <= 0: + return None + + return duration + + +def parse_video_extensions( + ending_arguments: list[str] | None, +) -> set[str]: + """ + Parse --ending arguments. + + Supported examples: + + --ending mp4,avi + --ending .mp4,.avi + --ending mp4 --ending avi + + Without --ending, all extensions in VIDEO_EXTENSIONS are used. + """ + if not ending_arguments: + return set(VIDEO_EXTENSIONS) + + extensions: set[str] = set() + + for argument in ending_arguments: + for value in argument.split(","): + extension = value.strip().casefold() + + if not extension: + continue + + if not extension.startswith("."): + extension = f".{extension}" + + extensions.add(extension) + + if not extensions: + raise ValueError( + "--ending did not contain any valid file endings." + ) + + return extensions + + +def load_movies(database_file: Path) -> list[Movie]: + if not database_file.is_file(): + raise FileNotFoundError( + f"Database file does not exist: {database_file}" + ) + + with sqlite3.connect(database_file) as connection: + connection.row_factory = sqlite3.Row + + movie_rows = connection.execute( + """ + SELECT + id, + name, + duration_seconds + FROM movie + WHERE name IS NOT NULL + AND TRIM(name) <> '' + ORDER BY name, id + """ + ).fetchall() + + alias_rows = connection.execute( + """ + SELECT + movie_id, + alias, + normalized_alias, + source + FROM movie_name_alias + WHERE alias IS NOT NULL + AND TRIM(alias) <> '' + ORDER BY movie_id, alias + """ + ).fetchall() + + history_rows = connection.execute( + """ + SELECT + movie_id, + duration_seconds, + archived_at + FROM movie_history + WHERE duration_seconds IS NOT NULL + AND duration_seconds > 0 + ORDER BY movie_id, archived_at + """ + ).fetchall() + + aliases_by_movie: dict[str, list[MatchName]] = {} + + for row in alias_rows: + movie_id = str(row["movie_id"]) + alias = str(row["alias"]) + + normalized_alias = ( + str(row["normalized_alias"]).strip() + if row["normalized_alias"] + else normalize_text(alias) + ) + + if not normalized_alias: + continue + + aliases_by_movie.setdefault( + movie_id, + [], + ).append( + MatchName( + display_name=alias, + normalized_name=normalized_alias, + normalized_persons=split_person_names(alias), + source=str(row["source"] or "alias"), + ) + ) + + historical_durations_by_movie: dict[ + str, + list[DurationVersion], + ] = {} + + for row in history_rows: + duration = normalize_duration( + row["duration_seconds"] + ) + + if duration is None: + continue + + movie_id = str(row["movie_id"]) + + historical_durations_by_movie.setdefault( + movie_id, + [], + ).append( + DurationVersion( + duration_seconds=duration, + source="historical", + archived_at=( + str(row["archived_at"]) + if row["archived_at"] is not None + else None + ), + ) + ) + + movies = [] + + for row in movie_rows: + movie_id = str(row["id"]) + current_name = str(row["name"]) + + current_match_name = MatchName( + display_name=current_name, + normalized_name=normalize_text(current_name), + normalized_persons=split_person_names(current_name), + source="current", + ) + + match_names = [current_match_name] + + seen_normalized_names = { + current_match_name.normalized_name + } + + for alias in aliases_by_movie.get(movie_id, []): + if alias.normalized_name in seen_normalized_names: + continue + + seen_normalized_names.add( + alias.normalized_name + ) + + match_names.append(alias) + + duration_versions: list[DurationVersion] = [] + seen_durations: set[int] = set() + + current_duration = normalize_duration( + row["duration_seconds"] + ) + + if current_duration is not None: + duration_versions.append( + DurationVersion( + duration_seconds=current_duration, + source="current", + ) + ) + + seen_durations.add(current_duration) + + for historical in historical_durations_by_movie.get( + movie_id, + [], + ): + if historical.duration_seconds in seen_durations: + continue + + seen_durations.add( + historical.duration_seconds + ) + + duration_versions.append(historical) + + movies.append( + Movie( + movie_id=movie_id, + name=current_name, + match_names=tuple(match_names), + duration_versions=tuple(duration_versions), + ) + ) + + return movies + + +def should_ignore_file(path: Path) -> bool: + """ + Ignore common operating-system metadata and resource-fork files. + + macOS commonly creates files such as: + + ._Video.mp4 + .DS_Store + + AppleDouble files may have a valid video extension but are not videos. + """ + name = path.name + + if name.startswith("._"): + return True + + if name in { + ".DS_Store", + "Thumbs.db", + "desktop.ini", + }: + return True + + return False + + +def find_video_files( + directory: Path, + recursive: bool, + extensions: set[str], +) -> list[Path]: + if not directory.is_dir(): + raise NotADirectoryError( + f"Input directory does not exist: {directory}" + ) + + iterator = ( + directory.rglob("*") + if recursive + else directory.glob("*") + ) + + files = [] + + for path in iterator: + if not path.is_file(): + continue + + if should_ignore_file(path): + continue + + if path.suffix.casefold() not in extensions: + continue + + files.append(path) + + return sorted( + files, + key=lambda path: str(path).casefold(), + ) + + +def read_file_duration(file_path: Path) -> int | None: + result = subprocess.run( + [ + "ffprobe", + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + str(file_path), + ], + capture_output=True, + text=True, + check=False, + ) + + if result.returncode != 0: + return None + + output = result.stdout.strip() + + if not output: + return None + + try: + duration = round(float(output)) + except ValueError: + return None + + return duration if duration > 0 else None + + +def calculate_duration_score( + difference_seconds: int, + reference_duration: int, +) -> int: + relative_difference = ( + difference_seconds / reference_duration + if reference_duration > 0 + else 1.0 + ) + + if difference_seconds <= 15: + return 50 + + if difference_seconds <= 60: + return 40 + + if difference_seconds <= 180: + return 25 + + if relative_difference <= 0.05: + return 20 + + if difference_seconds <= 600: + return 10 + + return 0 + + +def classify_poor_duration( + file_duration: int, + reference_duration: int, +) -> str: + if file_duration < reference_duration: + return "shorter_than_known_version" + + if file_duration > reference_duration: + return "longer_than_known_version" + + return "duration_mismatch" + + +def match_duration( + file_duration: int | None, + movie: Movie, +) -> DurationMatch: + if file_duration is None: + return DurationMatch( + score=0, + classification="file_duration_unknown", + matched_duration=None, + difference_seconds=None, + source=None, + archived_at=None, + ) + + if not movie.duration_versions: + return DurationMatch( + score=0, + classification="database_duration_unknown", + matched_duration=None, + difference_seconds=None, + source=None, + archived_at=None, + ) + + scored_versions = [] + + for version in movie.duration_versions: + difference = abs( + file_duration - version.duration_seconds + ) + + score = calculate_duration_score( + difference_seconds=difference, + reference_duration=version.duration_seconds, + ) + + scored_versions.append( + ( + score, + -difference, + version.source == "current", + version, + difference, + ) + ) + + ( + score, + _negative_difference, + _prefer_current, + best_version, + difference, + ) = max(scored_versions) + + if score > 0: + classification = ( + "current_version_match" + if best_version.source == "current" + else "historical_version_match" + ) + else: + classification = classify_poor_duration( + file_duration=file_duration, + reference_duration=best_version.duration_seconds, + ) + + return DurationMatch( + score=score, + classification=classification, + matched_duration=best_version.duration_seconds, + difference_seconds=difference, + source=best_version.source, + archived_at=best_version.archived_at, + ) + + +def score_match_name( + normalized_filename: str, + movie: Movie, + match_name: MatchName, +) -> MatchCandidate | None: + full_name_occurrences = normalized_filename.count( + match_name.normalized_name + ) + + matched_persons = [ + person + for person in match_name.normalized_persons + if person in normalized_filename + ] + + all_persons_matched = ( + bool(match_name.normalized_persons) + and len(matched_persons) + == len(match_name.normalized_persons) + ) + + if full_name_occurrences == 0 and not all_persons_matched: + return None + + name_score = 0 + reasons = [] + + if full_name_occurrences: + name_score += 80 + + if match_name.source == "current": + reasons.append("exact current-name match") + else: + reasons.append( + f"exact alias match: {match_name.display_name}" + ) + + if full_name_occurrences > 1: + repetition_bonus = min( + 15, + (full_name_occurrences - 1) * 5, + ) + + name_score += repetition_bonus + + reasons.append( + f"matched name occurs " + f"{full_name_occurrences} times" + ) + + if all_persons_matched: + name_score += 20 + reasons.append("all persons matched") + + if len(match_name.normalized_persons) > 1: + name_score += 10 + reasons.append("multi-person name matched") + + specificity_bonus = min( + 10, + len(match_name.normalized_name) // 5, + ) + + name_score += specificity_bonus + + reasons.append( + f"specificity bonus {specificity_bonus}" + ) + + empty_duration_match = DurationMatch( + score=0, + classification="not_evaluated", + matched_duration=None, + difference_seconds=None, + source=None, + archived_at=None, + ) + + return MatchCandidate( + movie=movie, + matched_name=match_name, + name_score=name_score, + duration_score=0, + total_score=name_score, + name_reason=", ".join(reasons), + duration_reason="duration not evaluated", + full_name_occurrences=full_name_occurrences, + all_persons_matched=all_persons_matched, + duration_match=empty_duration_match, + ) + + +def best_name_candidate_for_movie( + normalized_filename: str, + movie: Movie, +) -> MatchCandidate | None: + candidates = [] + + for match_name in movie.match_names: + candidate = score_match_name( + normalized_filename=normalized_filename, + movie=movie, + match_name=match_name, + ) + + if candidate is not None: + candidates.append(candidate) + + if not candidates: + return None + + return max( + candidates, + key=lambda candidate: ( + candidate.name_score, + candidate.matched_name.source == "current", + len(candidate.matched_name.normalized_persons), + len(candidate.matched_name.normalized_name), + ), + ) + + +def describe_duration_match( + duration_match: DurationMatch, +) -> str: + classification = duration_match.classification + + if classification == "file_duration_unknown": + return "file duration unavailable" + + if classification == "database_duration_unknown": + return "database duration unavailable" + + if duration_match.matched_duration is None: + return classification + + difference = duration_match.difference_seconds or 0 + + if classification == "current_version_match": + return ( + f"current duration match, difference {difference}s" + ) + + if classification == "historical_version_match": + archived = ( + f", archived {duration_match.archived_at}" + if duration_match.archived_at + else "" + ) + + return ( + f"historical duration match, " + f"difference {difference}s{archived}" + ) + + return ( + f"{classification}, difference {difference}s" + ) + + +def apply_duration_score( + candidate: MatchCandidate, + file_duration: int | None, +) -> MatchCandidate: + duration_match = match_duration( + file_duration=file_duration, + movie=candidate.movie, + ) + + return replace( + candidate, + duration_score=duration_match.score, + total_score=( + candidate.name_score + + duration_match.score + ), + duration_reason=describe_duration_match( + duration_match + ), + duration_match=duration_match, + ) + + +def find_candidates( + filename: str, + movies: list[Movie], + file_duration: int | None, +) -> list[MatchCandidate]: + filename_stem = Path(filename).stem + normalized_filename = normalize_text(filename_stem) + + candidates = [] + + for movie in movies: + candidate = best_name_candidate_for_movie( + normalized_filename=normalized_filename, + movie=movie, + ) + + if candidate is None: + continue + + candidates.append( + apply_duration_score( + candidate=candidate, + file_duration=file_duration, + ) + ) + + return sorted( + candidates, + key=lambda candidate: ( + candidate.total_score, + candidate.duration_score, + candidate.name_score, + len(candidate.matched_name.normalized_persons), + len(candidate.matched_name.normalized_name), + ), + reverse=True, + ) + + +def get_detected_movies( + candidates: list[MatchCandidate], +) -> dict[str, str]: + detected: dict[str, str] = {} + + for candidate in candidates: + if candidate.full_name_occurrences == 0: + continue + + detected[candidate.movie.movie_id] = ( + candidate.movie.name + ) + + return detected + + +def candidate_covers_detected_movies( + candidate: MatchCandidate, + candidates: list[MatchCandidate], + detected_movie_ids: set[str], +) -> bool: + candidate_persons = set( + candidate.matched_name.normalized_persons + ) + + if len(candidate_persons) <= 1: + return False + + detected_names: set[str] = set() + + for other in candidates: + if other.movie.movie_id not in detected_movie_ids: + continue + + if other.full_name_occurrences == 0: + continue + + if len(other.matched_name.normalized_persons) != 1: + continue + + detected_names.add( + other.matched_name.normalized_persons[0] + ) + + return ( + bool(detected_names) + and detected_names.issubset(candidate_persons) + ) + + +def calculate_total_margin( + candidates: list[MatchCandidate], +) -> int: + if not candidates: + return 0 + + if len(candidates) == 1: + return candidates[0].total_score + + return ( + candidates[0].total_score + - candidates[1].total_score + ) + + +def resolve_with_duration( + candidates: list[MatchCandidate], +) -> MatchCandidate | None: + """ + Resolve several exact name candidates using duration. + + Exactly one candidate must have credible duration support, or the best + duration-supported candidate must have a clear duration-score margin. + """ + credible = [ + candidate + for candidate in candidates + if candidate.duration_score + >= MIN_CREDIBLE_DURATION_SCORE + ] + + if not credible: + return None + + credible = sorted( + credible, + key=lambda candidate: ( + candidate.duration_score, + candidate.total_score, + candidate.name_score, + ), + reverse=True, + ) + + if len(credible) == 1: + return credible[0] + + duration_margin = ( + credible[0].duration_score + - credible[1].duration_score + ) + + if duration_margin >= MIN_DURATION_SCORE_MARGIN: + return credible[0] + + return None + + +def classify_candidates( + candidates: list[MatchCandidate], +) -> MatchResult: + if not candidates: + return MatchResult( + status="unmatched", + best=None, + margin=0, + reason="no exact database name or alias found", + detected_movies=(), + ) + + best = candidates[0] + margin = calculate_total_margin(candidates) + + detected = get_detected_movies(candidates) + detected_movie_ids = set(detected) + + if len(detected_movie_ids) > 1: + covering_candidates = [ + candidate + for candidate in candidates + if candidate_covers_detected_movies( + candidate=candidate, + candidates=candidates, + detected_movie_ids=detected_movie_ids, + ) + ] + + if covering_candidates: + covering_candidates.sort( + key=lambda candidate: ( + candidate.total_score, + candidate.duration_score, + candidate.name_score, + ), + reverse=True, + ) + + covering_best = covering_candidates[0] + + other_scores = [ + candidate.total_score + for candidate in candidates + if candidate.movie.movie_id + != covering_best.movie.movie_id + ] + + covering_margin = ( + covering_best.total_score + - max(other_scores, default=0) + ) + + if ( + covering_best.name_score >= MIN_NAME_SCORE + and covering_margin >= MIN_NAME_MARGIN + ): + return MatchResult( + status="matched", + best=covering_best, + margin=covering_margin, + reason=( + "multi-person database entry covers all " + "detected movie names" + ), + detected_movies=tuple( + sorted( + detected.values(), + key=str.casefold, + ) + ), + ) + + duration_winner = resolve_with_duration( + candidates + ) + + if duration_winner is not None: + other_scores = [ + candidate.total_score + for candidate in candidates + if candidate.movie.movie_id + != duration_winner.movie.movie_id + ] + + winner_margin = ( + duration_winner.total_score + - max(other_scores, default=0) + ) + + return MatchResult( + status="matched", + best=duration_winner, + margin=winner_margin, + reason=( + "multiple names matched, but duration " + "clearly supports one candidate" + ), + detected_movies=tuple( + sorted( + detected.values(), + key=str.casefold, + ) + ), + ) + + return MatchResult( + status="ambiguous", + best=best, + margin=margin, + reason=( + "multiple distinct database movies matched; " + "duration does not clearly resolve them" + ), + detected_movies=tuple( + sorted( + detected.values(), + key=str.casefold, + ) + ), + ) + + if ( + best.name_score >= MIN_NAME_SCORE + and ( + len(candidates) == 1 + or margin >= MIN_NAME_MARGIN + ) + ): + return MatchResult( + status="matched", + best=best, + margin=margin, + reason=( + "best candidate exceeds name and margin thresholds" + ), + detected_movies=tuple( + sorted( + detected.values(), + key=str.casefold, + ) + ), + ) + + return MatchResult( + status="ambiguous", + best=best, + margin=margin, + reason="score or margin is insufficient", + detected_movies=tuple( + sorted( + detected.values(), + key=str.casefold, + ) + ), + ) + + +def format_duration(seconds: int | None) -> str: + if seconds is None: + return "unknown" + + hours, remainder = divmod(seconds, 3600) + minutes, seconds = divmod(remainder, 60) + + if hours: + return f"{hours}:{minutes:02d}:{seconds:02d}" + + return f"{minutes}:{seconds:02d}" + + +def format_match_source( + candidate: MatchCandidate, +) -> str: + if candidate.matched_name.source == "current": + return "current name" + + return ( + f"{candidate.matched_name.source} alias " + f"{candidate.matched_name.display_name!r}" + ) + + +def print_result( + path: Path, + file_duration: int | None, + result: MatchResult, + debug: bool, + candidates: list[MatchCandidate], +) -> None: + if result.best is None: + name = "" + score = 0 + version = "-" + else: + name = result.best.movie.name + score = result.best.total_score + version = result.best.duration_match.classification + + print( + f"{path.name:<55} " + f"{name:<35} " + f"{score:>5} " + f"{result.status:<10} " + f"{version}" + ) + + if not debug: + return + + print( + f" file duration: " + f"{format_duration(file_duration)}" + ) + + if not candidates: + print(" no candidates") + print(f" reason: {result.reason}") + return + + if result.detected_movies: + print( + " detected movies: " + + ", ".join(result.detected_movies) + ) + + for position, candidate in enumerate( + candidates[:5], + start=1, + ): + matched_duration = ( + candidate.duration_match.matched_duration + ) + + print( + f" {position}. " + f"{candidate.movie.name} " + f"[name={candidate.name_score}, " + f"duration={candidate.duration_score}, " + f"total={candidate.total_score}]" + ) + + print( + f" via {format_match_source(candidate)}" + ) + + print( + f" name: {candidate.name_reason}" + ) + + print( + f" duration: " + f"{candidate.duration_reason}; " + f"reference=" + f"{format_duration(matched_duration)}" + ) + + print(f" margin: {result.margin}") + print(f" reason: {result.reason}") + + +def main() -> None: + args = parse_arguments() + + if args.limit is not None and args.limit <= 0: + raise ValueError( + "--limit must be greater than zero." + ) + + if shutil.which("ffprobe") is None: + raise FileNotFoundError( + "ffprobe was not found in PATH." + ) + + extensions = parse_video_extensions( + args.ending + ) + + movies = load_movies(args.database) + + video_files = find_video_files( + directory=args.directory, + recursive=args.recursive, + extensions=extensions, + ) + + if args.limit is not None: + video_files = video_files[:args.limit] + + print( + f"{'Filename':<55} " + f"{'Likely database name':<35} " + f"{'Score':>5} " + f"{'Status':<10} " + f"Version" + ) + + print( + f"{'-' * 55} " + f"{'-' * 35} " + f"{'-' * 5} " + f"{'-' * 10} " + f"{'-' * 26}" + ) + + matched = 0 + ambiguous = 0 + unmatched = 0 + ffprobe_failures = 0 + historical_matches = 0 + + for video_file in video_files: + file_duration = read_file_duration( + video_file + ) + + if file_duration is None: + ffprobe_failures += 1 + + candidates = find_candidates( + filename=video_file.name, + movies=movies, + file_duration=file_duration, + ) + + result = classify_candidates(candidates) + + if result.status == "matched": + matched += 1 + elif result.status == "ambiguous": + ambiguous += 1 + else: + unmatched += 1 + + if ( + result.best is not None + and result.best.duration_match.classification + == "historical_version_match" + ): + historical_matches += 1 + + print_result( + path=video_file, + file_duration=file_duration, + result=result, + debug=args.debug, + candidates=candidates, + ) + + print() + print("Summary") + print("-------") + print(f"Files analyzed: {len(video_files)}") + print(f"Matched: {matched}") + print(f"Ambiguous: {ambiguous}") + print(f"Unmatched: {unmatched}") + print(f"Historical versions: {historical_matches}") + print(f"ffprobe failures: {ffprobe_failures}") + + +if __name__ == "__main__": + try: + main() + except ( + FileNotFoundError, + NotADirectoryError, + ValueError, + sqlite3.Error, + ) as error: + print(f"Error: {error}", file=sys.stderr) + sys.exit(1) diff --git a/scripts/schema.sql b/scripts/schema.sql index 9eaf8b6..fe5a567 100644 --- a/scripts/schema.sql +++ b/scripts/schema.sql @@ -61,3 +61,24 @@ CREATE INDEX IF NOT EXISTS idx_movie_history_movie_id CREATE INDEX IF NOT EXISTS idx_movie_history_archived_at ON movie_history(archived_at); + +CREATE TABLE IF NOT EXISTS movie_name_alias ( + alias_id INTEGER PRIMARY KEY AUTOINCREMENT, + + movie_id TEXT NOT NULL, + alias TEXT NOT NULL, + normalized_alias TEXT NOT NULL, + source TEXT NOT NULL DEFAULT 'manual', + + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP, + + FOREIGN KEY (movie_id) REFERENCES movie(id), + + UNIQUE (movie_id, normalized_alias) +); + +CREATE INDEX IF NOT EXISTS idx_movie_name_alias_movie_id + ON movie_name_alias(movie_id); + +CREATE INDEX IF NOT EXISTS idx_movie_name_alias_normalized + ON movie_name_alias(normalized_alias); diff --git a/scripts/test_duration_matching.py b/scripts/test_duration_matching.py new file mode 100755 index 0000000..450145a --- /dev/null +++ b/scripts/test_duration_matching.py @@ -0,0 +1,270 @@ +#!/usr/bin/env python3 + +""" +Small local test for matching a file duration against current and historical +movie durations. + +This script does not modify the database or any files. +""" + +import sqlite3 +import subprocess +import sys +from pathlib import Path + + +DATABASE_FILE = Path( + "/storage/disk1/WCX/database/wcx.db" +) + + +def read_file_duration(file_path: Path) -> int: + result = subprocess.run( + [ + "ffprobe", + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + str(file_path), + ], + capture_output=True, + text=True, + ) + + if result.returncode != 0: + raise RuntimeError( + result.stderr.strip() + or f"ffprobe failed for {file_path}" + ) + + try: + return round(float(result.stdout.strip())) + except ValueError as error: + raise RuntimeError( + f"Invalid ffprobe duration: {result.stdout!r}" + ) from error + + +def format_duration(seconds: int | None) -> str: + if seconds is None: + return "unknown" + + hours, remainder = divmod(seconds, 3600) + minutes, seconds = divmod(remainder, 60) + + if hours: + return f"{hours}:{minutes:02d}:{seconds:02d}" + + return f"{minutes}:{seconds:02d}" + + +def load_movie_durations( + connection: sqlite3.Connection, + movie_name: str, +) -> tuple[str, int | None, list[int]]: + current = connection.execute( + """ + SELECT id, duration_seconds + FROM movie + WHERE name = ? + """, + (movie_name,), + ).fetchone() + + if current is None: + raise ValueError( + f"Movie not found: {movie_name}" + ) + + movie_id = str(current[0]) + current_duration = current[1] + + historical_rows = connection.execute( + """ + SELECT DISTINCT duration_seconds + FROM movie_history + WHERE movie_id = ? + AND duration_seconds IS NOT NULL + ORDER BY duration_seconds + """, + (movie_id,), + ).fetchall() + + historical_durations = [ + int(row[0]) + for row in historical_rows + if row[0] != current_duration + ] + + return ( + movie_id, + current_duration, + historical_durations, + ) + + +def classify_duration( + file_duration: int, + current_duration: int | None, + historical_durations: list[int], +) -> tuple[str, int | None, int | None]: + """ + Return: + classification + closest matching duration + absolute difference in seconds + """ + candidates: list[tuple[str, int]] = [] + + if current_duration is not None: + candidates.append( + ("current_version_match", current_duration) + ) + + for duration in historical_durations: + candidates.append( + ("historical_version_match", duration) + ) + + if not candidates: + return "unknown_version", None, None + + classification, closest_duration = min( + candidates, + key=lambda item: abs(file_duration - item[1]), + ) + + difference = abs( + file_duration - closest_duration + ) + + relative_difference = ( + difference / closest_duration + if closest_duration > 0 + else 1.0 + ) + + close_match = ( + difference <= 180 + or relative_difference <= 0.05 + ) + + if close_match: + return ( + classification, + closest_duration, + difference, + ) + + if file_duration < closest_duration: + return ( + "possible_truncated_version", + closest_duration, + difference, + ) + + if file_duration > closest_duration: + return ( + "possible_extended_version", + closest_duration, + difference, + ) + + return ( + "duration_mismatch", + closest_duration, + difference, + ) + + +def main() -> None: + if len(sys.argv) != 3: + raise SystemExit( + "Usage: test_duration_matching.py " + " " + ) + + file_path = Path(sys.argv[1]) + movie_name = sys.argv[2] + + if not file_path.is_file(): + raise FileNotFoundError( + f"Video file not found: {file_path}" + ) + + file_duration = read_file_duration( + file_path + ) + + with sqlite3.connect(DATABASE_FILE) as connection: + ( + movie_id, + current_duration, + historical_durations, + ) = load_movie_durations( + connection, + movie_name, + ) + + ( + classification, + matched_duration, + difference, + ) = classify_duration( + file_duration=file_duration, + current_duration=current_duration, + historical_durations=historical_durations, + ) + + print(f"Movie ID: {movie_id}") + print(f"Movie name: {movie_name}") + print( + f"File duration: " + f"{format_duration(file_duration)} " + f"({file_duration} seconds)" + ) + print( + f"Current duration: " + f"{format_duration(current_duration)}" + ) + + if historical_durations: + print("Historical durations:") + + for duration in historical_durations: + print( + f" {format_duration(duration)} " + f"({duration} seconds)" + ) + else: + print("Historical durations: none") + + print( + f"Closest duration: " + f"{format_duration(matched_duration)}" + ) + print( + f"Difference: " + f"{format_duration(difference)}" + ) + print( + f"Classification: " + f"{classification}" + ) + + +if __name__ == "__main__": + try: + main() + except ( + FileNotFoundError, + RuntimeError, + ValueError, + sqlite3.Error, + ) as error: + print(f"Error: {error}", file=sys.stderr) + sys.exit(1) +