feat(tooling): X-Ray threshold optimizer, gallery utilities, artifact registry, docs build

Optimizer (scripts/optimizer/): replay.py runs the real C++ tracker/matcher/
scene_tracker chain over a dumped-embeddings HDF5 via sae_kpn, so a threshold
sweep never re-decodes video or re-embeds faces. optimize.py drives scipy's
differential_evolution over the knob space, with DE-level parallelism
(multiple population candidates evaluated concurrently via a ThreadPoolExecutor)
on top of per-film replay parallelism. second_score.py is the per-second X-Ray
scoring metric (TPI/FPI/FN, out-of-cast misID weighted 10x, fair recall masked
to gallery-known cast) that superseded an earlier scene-union metric.
dump_error_frames.py / dump_scene_montage.py extract annotated video frames
(bounding boxes, TPI/FPI/FN captions, onscreen-vs-offscreen split) for visual
review of a replay against ground truth. Gallery utilities: cast_restrict.py,
gallery_membership.py, fetch_missing_actors.py, reembed_gallery.py.

scripts/validation/: X-Ray ground-truth loading and provider-agnostic identity
matching (identity.py's keys_for — an actor is the union of every id we can
derive, since pipeline output and ground truth don't share one id space).

scripts/artifacts/: push/pull scripts for the Gitea generic package registry —
galleries, montage frames, and experiment data (manifests/trajectories/results)
are pushed there instead of committed, since none are needed to run the app,
only benchmarks. Versioned by git short-SHA.

scripts/docs/: MkDocs site build (build_site.sh) and the calibration-curve
comparison chart (calibration_chart.py, matplotlib, reads each gallery's
embedded calibration).

Gallery-building scripts (make_jellyfin_gallery.py, make_gallery.py,
filter_gallery.py, run_from_jellyfin.py, movienet_eval.py, movienet_prep.py,
sae_gallery.py) updated to read/write HDF5 galleries exclusively, matching the
engine-side format switch. run_from_jellyfin.py and the optimizer no longer
carry movie source paths in shared manifests (some source filenames include
scene-release tags) — resolved locally via a gitignored file-lut.json instead.
This commit is contained in:
2026-07-19 19:06:48 +02:00
parent 26139ffe8a
commit 6f0ad83a55
31 changed files with 3411 additions and 47 deletions
+154
View File
@@ -0,0 +1,154 @@
#!/usr/bin/env python3
"""
fetch_missing_actors.py — close the gallery coverage gap.
X-Ray credits ~67% of each film's cast that our gallery never had a reference
embedding for, making those actors unrecoverable FNs no threshold can fix. This
fetches images for those missing actors (by IMDb nm id → TMDB profile photos),
embeds them with the SAME SCRFD+ArcFace models (sae_embed), and writes gallery
entries. Merge the result into the baseline to make those actors recognisable.
nm → TMDB person → /person/{id}/images profile photos → download → embed.
Usage:
python scripts/optimizer/fetch_missing_actors.py \
--missing missing_actors.json \
--out gallery_missing.json \
[--images-per-actor 3] [--build-dir build]
# TMDB_API_KEY from env/.env
Then merge:
python scripts/optimizer/fetch_missing_actors.py --merge \
gallery_arcface_w600k_r50.json gallery_missing.json \
--out gallery_augmented.json
"""
from __future__ import annotations
import argparse
import json
import os
import sys
import tempfile
from pathlib import Path
REPO = Path(__file__).resolve().parent.parent.parent
sys.path.insert(0, str(REPO / "scripts"))
import sae_env # noqa: E402 loads .env
from sae_tmdb import tmdb_get, tmdb_person_for_imdb, TMDB_IMG # noqa: E402
from sae_gallery import download_images, wikidata_image_urls # noqa: E402
from sae_embed_loader import load_embedder # noqa: E402
def profile_urls_for_imdb(imdb_id: str, token: str, n: int) -> tuple[str | None, list[str]]:
"""(tmdb_person_id, [image_url,...]) via /find then /person/{id}/images."""
data = tmdb_get(f"/find/{imdb_id}", token, external_source="imdb_id")
people = data.get("person_results", [])
if not people:
return None, []
pid = str(people[0]["id"])
imgs = tmdb_get(f"/person/{pid}/images", token)
profiles = imgs.get("profiles", [])[:n]
return pid, [TMDB_IMG + p["file_path"] for p in profiles if p.get("file_path")]
def fetch(missing_path, out_path, token, build_dir, models_dir, arcface,
images_per_actor, use_wikidata=False):
missing = json.loads(Path(missing_path).read_text())
src = "TMDB + Wikidata fallback" if use_wikidata else "TMDB"
print(f"[fetch] {len(missing)} missing actors to resolve via {src}", file=sys.stderr)
embedder = load_embedder(build_dir, models_dir, arcface)
img_root = Path(tempfile.mkdtemp(prefix="missing_gallery_"))
actors = []
n_resolved = n_no_tmdb = n_no_img = n_no_face = 0
n_via_wikidata = 0
for i, m in enumerate(missing, 1):
nm, name = m["imdb_id"], m.get("name", "")
tmdb_id, urls = None, []
try:
tmdb_id, urls = profile_urls_for_imdb(nm, token, images_per_actor)
except Exception as e:
print(f" [{i}] {name}: TMDB error {e}", file=sys.stderr)
# Wikidata fallback: keyed cleanly by IMDb nm (P345→P18 Commons photo),
# recovers on-camera character actors TMDB's film-centric DB misses.
if (not urls) and use_wikidata:
wiki_urls = wikidata_image_urls(nm)[:images_per_actor]
if wiki_urls:
urls = wiki_urls
n_via_wikidata += 1
if not urls:
if tmdb_id is None:
n_no_tmdb += 1
else:
n_no_img += 1
continue
dest = img_root / nm
dest.mkdir(parents=True, exist_ok=True)
paths = download_images(urls, dest, images_per_actor)
embeddings = []
for p in paths:
res = embedder.embed(str(p))
if res.ok:
embeddings.append(list(res.embedding))
if not embeddings:
n_no_face += 1
continue
actors.append({"imdb_id": nm, "tmdb_id": str(tmdb_id) if tmdb_id else "",
"jellyfin_id": "", "name": name,
"embeddings": embeddings, "source_images": []})
n_resolved += 1
if i % 20 == 0 or i == len(missing):
print(f" [{i}/{len(missing)}] resolved={n_resolved} "
f"(wiki={n_via_wikidata}) no_tmdb={n_no_tmdb} no_img={n_no_img} "
f"no_face={n_no_face}", file=sys.stderr)
Path(out_path).write_text(json.dumps({"actors": actors}, indent=2))
n_emb = sum(len(a["embeddings"]) for a in actors)
print(f"\n[fetch] recovered {n_resolved}/{len(missing)} actors "
f"({n_via_wikidata} via Wikidata), {n_emb} embeddings → {out_path}",
file=sys.stderr)
print(f"[fetch] unrecoverable: no_tmdb={n_no_tmdb} no_img={n_no_img} "
f"no_face={n_no_face}", file=sys.stderr)
def merge(base_path, add_path, out_path):
base = json.loads(Path(base_path).read_text())
add = json.loads(Path(add_path).read_text())
have = {a.get("imdb_id") for a in base["actors"] if a.get("imdb_id")}
added = [a for a in add["actors"] if a.get("imdb_id") not in have]
base["actors"].extend(added)
Path(out_path).write_text(json.dumps(base, indent=2))
print(f"[merge] {len(base['actors'])-len(added)} + {len(added)} = "
f"{len(base['actors'])} actors → {out_path}", file=sys.stderr)
def main():
p = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
p.add_argument("--merge", nargs=2, metavar=("BASE", "ADD"),
help="merge ADD gallery into BASE → --out")
p.add_argument("--missing")
p.add_argument("--out", required=True)
p.add_argument("--tmdb-key", default=os.environ.get("TMDB_API_KEY"))
p.add_argument("--build-dir", default=str(REPO / "build"))
p.add_argument("--models-dir", default=str(REPO / "models"))
p.add_argument("--arcface", default=None)
p.add_argument("--images-per-actor", type=int, default=3)
p.add_argument("--wikidata", action="store_true",
help="fall back to Wikidata (P345→P18 Commons photo) when TMDB has no image")
args = p.parse_args()
if args.merge:
merge(args.merge[0], args.merge[1], args.out)
return
if not args.missing:
sys.exit("--missing required (or use --merge)")
if not args.tmdb_key:
sys.exit("no TMDB key — set TMDB_API_KEY")
fetch(args.missing, args.out, args.tmdb_key, args.build_dir, args.models_dir,
args.arcface, args.images_per_actor, use_wikidata=args.wikidata)
if __name__ == "__main__":
main()