#!/usr/bin/env python3 """ cast_restrict.py — produce a per-film gallery restricted to its credited cast. Benchmark arm: instead of matching a face against the WHOLE gallery (2418 actors, risking cross-film misIDs like naming Archie Yates in a film he's not in), restrict the matcher's candidate set to the title's credited cast (from Jellyfin — the top ~15 billed actors, exactly what run_from_jellyfin.py does in production). Filters a gallery to actors whose jellyfin_id is in the film's cast set, writing a small gallery JSON the replay can load. Actors are kept if their jellyfin_id (or, as a fallback, normalized name) matches the cast. Used by the full-vs-restricted bake-off. Cached per (gallery, film) so a DE sweep reuses the restricted gallery. """ from __future__ import annotations import json import sys import tempfile from pathlib import Path REPO = Path(__file__).resolve().parent.parent.parent sys.path.insert(0, str(REPO / "scripts" / "validation")) from identity import norm_name # noqa: E402 _CACHE: dict = {} def restricted_gallery_path(gallery_path: str, cast_jellyfin_ids: set[str], cast_names: set[str] | None = None) -> str: """Write (once, cached) a gallery filtered to the film's credited cast; return path. Matches gallery actors to the cast by jellyfin_id first, then normalized name.""" key = (gallery_path, frozenset(cast_jellyfin_ids)) if key in _CACHE: return _CACHE[key] gal = json.loads(Path(gallery_path).read_text()) names = {norm_name(n) for n in (cast_names or set())} kept = [] for a in gal["actors"]: jid = a.get("jellyfin_id", "") if (jid and jid in cast_jellyfin_ids) or (names and norm_name(a["name"]) in names): kept.append(a) tf = tempfile.NamedTemporaryFile("w", suffix=".json", delete=False, prefix="castgal_") json.dump({"actors": kept}, tf) tf.close() _CACHE[key] = tf.name return tf.name def load_casts(casts_json: str) -> dict[str, list[str]]: """film name → [jellyfin person id, ...] from jellyfin_casts.json.""" return json.loads(Path(casts_json).read_text())