Optimizer (scripts/optimizer/): replay.py runs the real C++ tracker/matcher/ scene_tracker chain over a dumped-embeddings HDF5 via sae_kpn, so a threshold sweep never re-decodes video or re-embeds faces. optimize.py drives scipy's differential_evolution over the knob space, with DE-level parallelism (multiple population candidates evaluated concurrently via a ThreadPoolExecutor) on top of per-film replay parallelism. second_score.py is the per-second X-Ray scoring metric (TPI/FPI/FN, out-of-cast misID weighted 10x, fair recall masked to gallery-known cast) that superseded an earlier scene-union metric. dump_error_frames.py / dump_scene_montage.py extract annotated video frames (bounding boxes, TPI/FPI/FN captions, onscreen-vs-offscreen split) for visual review of a replay against ground truth. Gallery utilities: cast_restrict.py, gallery_membership.py, fetch_missing_actors.py, reembed_gallery.py. scripts/validation/: X-Ray ground-truth loading and provider-agnostic identity matching (identity.py's keys_for — an actor is the union of every id we can derive, since pipeline output and ground truth don't share one id space). scripts/artifacts/: push/pull scripts for the Gitea generic package registry — galleries, montage frames, and experiment data (manifests/trajectories/results) are pushed there instead of committed, since none are needed to run the app, only benchmarks. Versioned by git short-SHA. scripts/docs/: MkDocs site build (build_site.sh) and the calibration-curve comparison chart (calibration_chart.py, matplotlib, reads each gallery's embedded calibration). Gallery-building scripts (make_jellyfin_gallery.py, make_gallery.py, filter_gallery.py, run_from_jellyfin.py, movienet_eval.py, movienet_prep.py, sae_gallery.py) updated to read/write HDF5 galleries exclusively, matching the engine-side format switch. run_from_jellyfin.py and the optimizer no longer carry movie source paths in shared manifests (some source filenames include scene-release tags) — resolved locally via a gitignored file-lut.json instead.
155 lines
6.3 KiB
Python
155 lines
6.3 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
fetch_missing_actors.py — close the gallery coverage gap.
|
|
|
|
X-Ray credits ~67% of each film's cast that our gallery never had a reference
|
|
embedding for, making those actors unrecoverable FNs no threshold can fix. This
|
|
fetches images for those missing actors (by IMDb nm id → TMDB profile photos),
|
|
embeds them with the SAME SCRFD+ArcFace models (sae_embed), and writes gallery
|
|
entries. Merge the result into the baseline to make those actors recognisable.
|
|
|
|
nm → TMDB person → /person/{id}/images profile photos → download → embed.
|
|
|
|
Usage:
|
|
python scripts/optimizer/fetch_missing_actors.py \
|
|
--missing missing_actors.json \
|
|
--out gallery_missing.json \
|
|
[--images-per-actor 3] [--build-dir build]
|
|
# TMDB_API_KEY from env/.env
|
|
|
|
Then merge:
|
|
python scripts/optimizer/fetch_missing_actors.py --merge \
|
|
gallery_arcface_w600k_r50.json gallery_missing.json \
|
|
--out gallery_augmented.json
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import sys
|
|
import tempfile
|
|
from pathlib import Path
|
|
|
|
REPO = Path(__file__).resolve().parent.parent.parent
|
|
sys.path.insert(0, str(REPO / "scripts"))
|
|
import sae_env # noqa: E402 loads .env
|
|
from sae_tmdb import tmdb_get, tmdb_person_for_imdb, TMDB_IMG # noqa: E402
|
|
from sae_gallery import download_images, wikidata_image_urls # noqa: E402
|
|
from sae_embed_loader import load_embedder # noqa: E402
|
|
|
|
|
|
def profile_urls_for_imdb(imdb_id: str, token: str, n: int) -> tuple[str | None, list[str]]:
|
|
"""(tmdb_person_id, [image_url,...]) via /find then /person/{id}/images."""
|
|
data = tmdb_get(f"/find/{imdb_id}", token, external_source="imdb_id")
|
|
people = data.get("person_results", [])
|
|
if not people:
|
|
return None, []
|
|
pid = str(people[0]["id"])
|
|
imgs = tmdb_get(f"/person/{pid}/images", token)
|
|
profiles = imgs.get("profiles", [])[:n]
|
|
return pid, [TMDB_IMG + p["file_path"] for p in profiles if p.get("file_path")]
|
|
|
|
|
|
def fetch(missing_path, out_path, token, build_dir, models_dir, arcface,
|
|
images_per_actor, use_wikidata=False):
|
|
missing = json.loads(Path(missing_path).read_text())
|
|
src = "TMDB + Wikidata fallback" if use_wikidata else "TMDB"
|
|
print(f"[fetch] {len(missing)} missing actors to resolve via {src}", file=sys.stderr)
|
|
embedder = load_embedder(build_dir, models_dir, arcface)
|
|
|
|
img_root = Path(tempfile.mkdtemp(prefix="missing_gallery_"))
|
|
actors = []
|
|
n_resolved = n_no_tmdb = n_no_img = n_no_face = 0
|
|
n_via_wikidata = 0
|
|
|
|
for i, m in enumerate(missing, 1):
|
|
nm, name = m["imdb_id"], m.get("name", "")
|
|
tmdb_id, urls = None, []
|
|
try:
|
|
tmdb_id, urls = profile_urls_for_imdb(nm, token, images_per_actor)
|
|
except Exception as e:
|
|
print(f" [{i}] {name}: TMDB error {e}", file=sys.stderr)
|
|
# Wikidata fallback: keyed cleanly by IMDb nm (P345→P18 Commons photo),
|
|
# recovers on-camera character actors TMDB's film-centric DB misses.
|
|
if (not urls) and use_wikidata:
|
|
wiki_urls = wikidata_image_urls(nm)[:images_per_actor]
|
|
if wiki_urls:
|
|
urls = wiki_urls
|
|
n_via_wikidata += 1
|
|
if not urls:
|
|
if tmdb_id is None:
|
|
n_no_tmdb += 1
|
|
else:
|
|
n_no_img += 1
|
|
continue
|
|
dest = img_root / nm
|
|
dest.mkdir(parents=True, exist_ok=True)
|
|
paths = download_images(urls, dest, images_per_actor)
|
|
embeddings = []
|
|
for p in paths:
|
|
res = embedder.embed(str(p))
|
|
if res.ok:
|
|
embeddings.append(list(res.embedding))
|
|
if not embeddings:
|
|
n_no_face += 1
|
|
continue
|
|
actors.append({"imdb_id": nm, "tmdb_id": str(tmdb_id) if tmdb_id else "",
|
|
"jellyfin_id": "", "name": name,
|
|
"embeddings": embeddings, "source_images": []})
|
|
n_resolved += 1
|
|
if i % 20 == 0 or i == len(missing):
|
|
print(f" [{i}/{len(missing)}] resolved={n_resolved} "
|
|
f"(wiki={n_via_wikidata}) no_tmdb={n_no_tmdb} no_img={n_no_img} "
|
|
f"no_face={n_no_face}", file=sys.stderr)
|
|
|
|
Path(out_path).write_text(json.dumps({"actors": actors}, indent=2))
|
|
n_emb = sum(len(a["embeddings"]) for a in actors)
|
|
print(f"\n[fetch] recovered {n_resolved}/{len(missing)} actors "
|
|
f"({n_via_wikidata} via Wikidata), {n_emb} embeddings → {out_path}",
|
|
file=sys.stderr)
|
|
print(f"[fetch] unrecoverable: no_tmdb={n_no_tmdb} no_img={n_no_img} "
|
|
f"no_face={n_no_face}", file=sys.stderr)
|
|
|
|
|
|
def merge(base_path, add_path, out_path):
|
|
base = json.loads(Path(base_path).read_text())
|
|
add = json.loads(Path(add_path).read_text())
|
|
have = {a.get("imdb_id") for a in base["actors"] if a.get("imdb_id")}
|
|
added = [a for a in add["actors"] if a.get("imdb_id") not in have]
|
|
base["actors"].extend(added)
|
|
Path(out_path).write_text(json.dumps(base, indent=2))
|
|
print(f"[merge] {len(base['actors'])-len(added)} + {len(added)} = "
|
|
f"{len(base['actors'])} actors → {out_path}", file=sys.stderr)
|
|
|
|
|
|
def main():
|
|
p = argparse.ArgumentParser(description=__doc__,
|
|
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
p.add_argument("--merge", nargs=2, metavar=("BASE", "ADD"),
|
|
help="merge ADD gallery into BASE → --out")
|
|
p.add_argument("--missing")
|
|
p.add_argument("--out", required=True)
|
|
p.add_argument("--tmdb-key", default=os.environ.get("TMDB_API_KEY"))
|
|
p.add_argument("--build-dir", default=str(REPO / "build"))
|
|
p.add_argument("--models-dir", default=str(REPO / "models"))
|
|
p.add_argument("--arcface", default=None)
|
|
p.add_argument("--images-per-actor", type=int, default=3)
|
|
p.add_argument("--wikidata", action="store_true",
|
|
help="fall back to Wikidata (P345→P18 Commons photo) when TMDB has no image")
|
|
args = p.parse_args()
|
|
|
|
if args.merge:
|
|
merge(args.merge[0], args.merge[1], args.out)
|
|
return
|
|
if not args.missing:
|
|
sys.exit("--missing required (or use --merge)")
|
|
if not args.tmdb_key:
|
|
sys.exit("no TMDB key — set TMDB_API_KEY")
|
|
fetch(args.missing, args.out, args.tmdb_key, args.build_dir, args.models_dir,
|
|
args.arcface, args.images_per_actor, use_wikidata=args.wikidata)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|