Optimizer (scripts/optimizer/): replay.py runs the real C++ tracker/matcher/ scene_tracker chain over a dumped-embeddings HDF5 via sae_kpn, so a threshold sweep never re-decodes video or re-embeds faces. optimize.py drives scipy's differential_evolution over the knob space, with DE-level parallelism (multiple population candidates evaluated concurrently via a ThreadPoolExecutor) on top of per-film replay parallelism. second_score.py is the per-second X-Ray scoring metric (TPI/FPI/FN, out-of-cast misID weighted 10x, fair recall masked to gallery-known cast) that superseded an earlier scene-union metric. dump_error_frames.py / dump_scene_montage.py extract annotated video frames (bounding boxes, TPI/FPI/FN captions, onscreen-vs-offscreen split) for visual review of a replay against ground truth. Gallery utilities: cast_restrict.py, gallery_membership.py, fetch_missing_actors.py, reembed_gallery.py. scripts/validation/: X-Ray ground-truth loading and provider-agnostic identity matching (identity.py's keys_for — an actor is the union of every id we can derive, since pipeline output and ground truth don't share one id space). scripts/artifacts/: push/pull scripts for the Gitea generic package registry — galleries, montage frames, and experiment data (manifests/trajectories/results) are pushed there instead of committed, since none are needed to run the app, only benchmarks. Versioned by git short-SHA. scripts/docs/: MkDocs site build (build_site.sh) and the calibration-curve comparison chart (calibration_chart.py, matplotlib, reads each gallery's embedded calibration). Gallery-building scripts (make_jellyfin_gallery.py, make_gallery.py, filter_gallery.py, run_from_jellyfin.py, movienet_eval.py, movienet_prep.py, sae_gallery.py) updated to read/write HDF5 galleries exclusively, matching the engine-side format switch. run_from_jellyfin.py and the optimizer no longer carry movie source paths in shared manifests (some source filenames include scene-release tags) — resolved locally via a gitignored file-lut.json instead.
211 lines
8.5 KiB
Python
Executable File
211 lines
8.5 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""make_gallery.py — fetch actor images for a movie and build gallery.h5.
|
|
|
|
Fetches the cast from TMDB, downloads actor profile images, embeds them via
|
|
the sae_embed module (SCRFD + ArcFace, same models as scene_analyze, loaded
|
|
once), then writes gallery.h5.
|
|
|
|
Requirements:
|
|
pip install requests Pillow
|
|
|
|
Usage:
|
|
# By IMDB movie ID (most natural — resolves to TMDB automatically):
|
|
python scripts/make_gallery.py \\
|
|
--tmdb-key YOUR_KEY \\
|
|
--imdb-id tt0137523 \\
|
|
--output gallery.h5
|
|
|
|
# Or directly with a TMDB movie ID:
|
|
python scripts/make_gallery.py \\
|
|
--tmdb-key YOUR_KEY \\
|
|
--movie-id 550 \\
|
|
--output gallery.h5
|
|
|
|
# Additional options:
|
|
# --build-dir build/ build dir containing sae_embed module
|
|
# --models-dir models/ directory with ONNX models
|
|
# --images-per-actor 3 profile images to download per actor
|
|
# --image-dir /tmp/gallery_imgs where to cache downloaded images
|
|
|
|
Get a free TMDB API key at: https://www.themoviedb.org/settings/api
|
|
"""
|
|
|
|
import argparse
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from sae_embed_loader import load_embedder
|
|
from sae_gallery import download_images, save_gallery, wikidata_image_urls
|
|
from sae_tmdb import TMDB_IMG, tmdb_get, tmdb_id_from_imdb
|
|
|
|
|
|
def fetch_cast(movie_id: int, key: str) -> list[dict]:
|
|
"""Return list of {id, name, imdb_id, profile_images: [...url...]}."""
|
|
credits = tmdb_get(f"/movie/{movie_id}/credits", key)
|
|
cast = credits.get("cast", [])
|
|
|
|
actors = []
|
|
for member in cast:
|
|
person_id = member["id"]
|
|
|
|
# Get IMDB ID for this person
|
|
ext = tmdb_get(f"/person/{person_id}/external_ids", key)
|
|
imdb_id = ext.get("imdb_id") or ""
|
|
|
|
# Get profile images (sorted by vote_average desc by TMDB)
|
|
images_data = tmdb_get(f"/person/{person_id}/images", key)
|
|
profiles = images_data.get("profiles", [])
|
|
image_urls = [TMDB_IMG + p["file_path"] for p in profiles if p.get("file_path")]
|
|
|
|
if not image_urls:
|
|
image_urls = wikidata_image_urls(imdb_id)
|
|
if image_urls:
|
|
print(f" [info] no TMDB images for {member['name']}, "
|
|
f"found {len(image_urls)} via Wikidata", file=sys.stderr)
|
|
|
|
if not image_urls:
|
|
print(f" [warn] no images for {member['name']}", file=sys.stderr)
|
|
|
|
actors.append({
|
|
"id": person_id,
|
|
"name": member["name"],
|
|
"imdb_id": imdb_id,
|
|
"tmdb_id": str(person_id),
|
|
"profile_images": image_urls,
|
|
})
|
|
time.sleep(0.05) # be polite to TMDB
|
|
|
|
return actors
|
|
|
|
|
|
# ── Gallery assembly ─────────────────────────────────────────────────────────
|
|
|
|
def build_gallery(movie_id: int, key: str, embedder,
|
|
images_per_actor: int,
|
|
image_root: Path) -> tuple[dict, list[dict]]:
|
|
"""Fetch cast, download images, embed, return (gallery dict, actors needing more images)."""
|
|
print(f"Fetching cast for TMDB movie {movie_id}…", file=sys.stderr)
|
|
actors = fetch_cast(movie_id, key)
|
|
print(f"Found {len(actors)} cast member(s)", file=sys.stderr)
|
|
|
|
gallery_actors = []
|
|
missing = []
|
|
|
|
for actor in actors:
|
|
safe_name = actor["name"].replace(" ", "_")
|
|
dir_id = actor["imdb_id"] or f"tmdb_{actor['tmdb_id']}"
|
|
actor_dir = image_root / f"{dir_id}_{safe_name}"
|
|
|
|
print(f"\n{actor['name']} ({dir_id})", file=sys.stderr)
|
|
|
|
if not actor["profile_images"]:
|
|
print(" no images found, skipping", file=sys.stderr)
|
|
missing.append({"name": actor["name"], "imdb_id": actor["imdb_id"],
|
|
"tmdb_id": actor["tmdb_id"], "reason": "no images found"})
|
|
continue
|
|
|
|
image_paths = download_images(actor["profile_images"], actor_dir, images_per_actor)
|
|
if not image_paths:
|
|
print(" no images downloaded, skipping", file=sys.stderr)
|
|
missing.append({"name": actor["name"], "imdb_id": actor["imdb_id"],
|
|
"tmdb_id": actor["tmdb_id"], "reason": "download failed"})
|
|
continue
|
|
|
|
print(f" embedding {len(image_paths)} image(s)…", file=sys.stderr)
|
|
|
|
embeddings = []
|
|
source_images = []
|
|
for path in image_paths:
|
|
res = embedder.embed(str(path))
|
|
if not res.ok:
|
|
print(f" [skip] {path.name}: {res.error}", file=sys.stderr)
|
|
continue
|
|
embeddings.append(res.embedding)
|
|
source_images.append(path.name)
|
|
print(f" [ok] {path.name} conf={res.confidence:.2f}",
|
|
file=sys.stderr)
|
|
|
|
if not embeddings:
|
|
print(" no valid embeddings, skipping actor", file=sys.stderr)
|
|
missing.append({"name": actor["name"], "imdb_id": actor["imdb_id"],
|
|
"tmdb_id": actor["tmdb_id"], "reason": "no valid embeddings"})
|
|
continue
|
|
|
|
gallery_actors.append({
|
|
"imdb_id": actor["imdb_id"],
|
|
"tmdb_id": actor["tmdb_id"],
|
|
"jellyfin_id": "",
|
|
"name": actor["name"],
|
|
"source_images": source_images,
|
|
"embeddings": embeddings,
|
|
})
|
|
print(f" → {len(embeddings)} embedding(s) stored", file=sys.stderr)
|
|
|
|
return {"actors": gallery_actors}, missing
|
|
|
|
|
|
# ── Entry point ───────────────────────────────────────────────────────────────
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description="Fetch TMDB cast images and build gallery.h5 via sae_embed")
|
|
parser.add_argument("--tmdb-key", required=True,
|
|
help="TMDB Bearer token (API Read Access Token from themoviedb.org/settings/api)")
|
|
group = parser.add_mutually_exclusive_group(required=True)
|
|
group.add_argument("--imdb-id",
|
|
help="IMDB movie ID, e.g. tt0137523 — looked up via TMDB automatically")
|
|
group.add_argument("--movie-id", type=int,
|
|
help="TMDB movie ID (alternative to --imdb-id)")
|
|
parser.add_argument("--output", required=True, help="Output gallery.h5 path")
|
|
parser.add_argument("--build-dir", default="build",
|
|
help="Build directory containing the sae_embed module (default: build)")
|
|
parser.add_argument("--models-dir", default="models",
|
|
help="Directory containing ONNX models (default: models/)")
|
|
parser.add_argument("--arcface", default=None,
|
|
help="Path to ArcFace ONNX model (overrides --models-dir selection)")
|
|
parser.add_argument("--images-per-actor",type=int, default=3,
|
|
help="Profile images to download per actor (default: 3)")
|
|
parser.add_argument("--image-dir", default=None,
|
|
help="Where to store downloaded images (default: <output_dir>/images)")
|
|
parser.add_argument("--keep-images", action="store_true",
|
|
help="Do not delete downloaded images after embedding")
|
|
args = parser.parse_args()
|
|
|
|
# Resolve paths
|
|
output = Path(args.output)
|
|
image_root = Path(args.image_dir) if args.image_dir else output.parent / "images"
|
|
|
|
embedder = load_embedder(args.build_dir, args.models_dir, args.arcface)
|
|
|
|
# Resolve movie ID
|
|
movie_id = args.movie_id
|
|
if movie_id is None:
|
|
print(f"Resolving IMDB ID {args.imdb_id} → TMDB…", file=sys.stderr)
|
|
movie_id = tmdb_id_from_imdb(args.imdb_id, args.tmdb_key)
|
|
print(f"TMDB movie ID: {movie_id}", file=sys.stderr)
|
|
|
|
# Build gallery
|
|
gallery, missing = build_gallery(
|
|
movie_id = movie_id,
|
|
key = args.tmdb_key,
|
|
embedder = embedder,
|
|
images_per_actor = args.images_per_actor,
|
|
image_root = image_root,
|
|
)
|
|
|
|
n_actors = len(gallery["actors"])
|
|
n_embeddings = sum(len(a["embeddings"]) for a in gallery["actors"])
|
|
print(f"\nGallery: {n_actors} actors, {n_embeddings} total embeddings",
|
|
file=sys.stderr)
|
|
|
|
if n_actors == 0:
|
|
sys.exit("No actors could be processed — check models and images.")
|
|
|
|
save_gallery(gallery, missing, output)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|