#!/usr/bin/env python3 """ identity.py — provider-agnostic match keys. The pipeline output and the ground truth may not share one id space: an output actor can carry only tmdb/jellyfin ids while X-Ray/MovieNet key on IMDb nm-ids. We represent each actor by the *set* of every key we can derive, and treat two actors as the same iff their key sets intersect. Namespacing each key by its provider prevents cross-provider collisions (e.g. an nm-number equalling a tmdb-number). """ from __future__ import annotations import re import unicodedata def norm_name(name: str | None) -> str | None: """Lowercased, accent-stripped, punctuation-free name for fuzzy fallback match.""" if not name: return None s = unicodedata.normalize("NFKD", name) s = "".join(c for c in s if not unicodedata.combining(c)) s = re.sub(r"[^a-z0-9 ]+", "", s.lower()).strip() s = re.sub(r"\s+", " ", s) return s or None def keys_for(imdb_id: str | None = None, tmdb_id: str | None = None, jellyfin_id: str | None = None, name: str | None = None, crosswalk=None) -> set[str]: """All identity tokens for one actor. Empty strings are ignored. If `crosswalk` (a CrosswalkTable) is given and no imdb_id is present, resolve tmdb_id → imdb_id through it so a tmdb-only actor still gets an exact `imdb:` key — turning the fuzzy name join into an exact id join. See tmdb_imdb_map.py. """ keys: set[str] = set() imdb = imdb_id.strip() if (imdb_id and imdb_id.strip()) else None if not imdb and crosswalk is not None and tmdb_id: imdb = crosswalk.imdb_for(tmdb_id) if imdb: keys.add(f"imdb:{imdb}") if tmdb_id and str(tmdb_id).strip(): keys.add(f"tmdb:{str(tmdb_id).strip()}") if jellyfin_id and jellyfin_id.strip(): keys.add(f"jf:{jellyfin_id.strip()}") nn = norm_name(name) if nn: keys.add(f"name:{nn}") return keys