docs: deploy from 0bd2747

This commit is contained in:
2026-07-21 08:56:42 +02:00
parent 36f75ba199
commit 34b1f2c58f
24 changed files with 2066 additions and 1897 deletions
+52 -52
View File
@@ -232,6 +232,25 @@
<li class="md-tabs__item">
<a href="/dtourolle/scene-actor-extraction/methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
<li class="md-tabs__item">
@@ -259,26 +278,7 @@
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item">
<a href="/dtourolle/scene-actor-extraction/optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
Full Experiment Log
</a>
</li>
@@ -382,6 +382,33 @@
<li class="md-nav__item">
<a href="/dtourolle/scene-actor-extraction/methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -396,10 +423,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" >
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="0">
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
@@ -417,8 +444,8 @@
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_2">
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span>
@@ -561,34 +588,7 @@
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="/dtourolle/scene-actor-extraction/optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
Full Experiment Log
Binary file not shown.

Before

Width:  |  Height:  |  Size: 68 KiB

After

Width:  |  Height:  |  Size: 69 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 155 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 176 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 161 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 142 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 174 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 124 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 111 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 134 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 118 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 129 KiB

After

Width:  |  Height:  |  Size: 103 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 40 KiB

+236 -118
View File
@@ -13,7 +13,7 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/best-model/">
<link rel="prev" href="..">
<link rel="prev" href="../methodology/">
<link rel="next" href="../gallery-scope/">
@@ -243,6 +243,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -272,26 +291,7 @@
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
Full Experiment Log
</a>
</li>
@@ -395,6 +395,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -414,10 +441,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked>
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="">
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="">
@@ -435,8 +462,8 @@
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2">
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span>
@@ -524,10 +551,10 @@
</li>
<li class="md-nav__item">
<a href="#second-signal-f1-on-the-actual-benchmark" class="md-nav__link">
<a href="#second-signal-held-out-f1" class="md-nav__link">
<span class="md-ellipsis">
Second signal: F1 on the actual benchmark
Second signal: held-out F1
</span>
</a>
@@ -535,10 +562,21 @@
</li>
<li class="md-nav__item">
<a href="#caveat-model-choice-is-an-operational-change" class="md-nav__link">
<a href="#full-training-matrix-picture" class="md-nav__link">
<span class="md-ellipsis">
Caveat: model choice is an operational change
Full training-matrix picture
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#operational-note" class="md-nav__link">
<span class="md-ellipsis">
Operational note
</span>
</a>
@@ -659,34 +697,7 @@
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
Full Experiment Log
@@ -764,10 +775,10 @@
</li>
<li class="md-nav__item">
<a href="#second-signal-f1-on-the-actual-benchmark" class="md-nav__link">
<a href="#second-signal-held-out-f1" class="md-nav__link">
<span class="md-ellipsis">
Second signal: F1 on the actual benchmark
Second signal: held-out F1
</span>
</a>
@@ -775,10 +786,21 @@
</li>
<li class="md-nav__item">
<a href="#caveat-model-choice-is-an-operational-change" class="md-nav__link">
<a href="#full-training-matrix-picture" class="md-nav__link">
<span class="md-ellipsis">
Caveat: model choice is an operational change
Full training-matrix picture
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#operational-note" class="md-nav__link">
<span class="md-ellipsis">
Operational note
</span>
</a>
@@ -803,31 +825,35 @@
<h1 id="which-embedding-model-is-best">Which embedding model is best?<a class="headerlink" href="#which-embedding-model-is-best" title="Permanent link">&para;</a></h1>
<p>Four candidates went into the bake-off: three ArcFace variants (w600k-R50,
R18, w600k-MBF) and LVFace-B (Glint360K), a Vision-Transformer embedder that's
a drop-in replacement for ArcFace's <code>[N,3,112,112]</code> input / 512-d output. The
open question: is LVFace (455MB) actually better, or just the biggest?</p>
<p>Three ArcFace variants (w600k-R50, R18, w600k-MBF) and LVFace-B (Glint360K,
455MB) were compared. r50 is excluded from the training/held-out comparison
below; its gallery has roughly 30% fewer reference images per actor than the
other three on the identical source photos, which confounds a direct score
comparison (see <a href="../model-bakeoff/">the full experiment log</a> for detail). It
remains in the calibration comparison, which does not depend on the gallery
image count.</p>
<h2 id="first-signal-calibration-curves">First signal: calibration curves<a class="headerlink" href="#first-signal-calibration-curves" title="Permanent link">&para;</a></h2>
<p>Each gallery carries a fitted Platt sigmoid <code>P(match | cosine similarity) =
σ(a·sim + b)</code>, embedded directly in the gallery's HDF5 file
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/gallery_calibration.hpp"><code>src/gallery/gallery_calibration.hpp</code></a>). This is a property of the embedding
space alone computed from intra/inter-actor reference-image pairs, no
tracking or scene logic involved — so it's a clean first read on discriminative
power before running a single benchmark.</p>
σ(a·sim + b)</code>, stored directly in the gallery HDF5
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/gallery/gallery_calibration.hpp"><code>src/gallery/gallery_calibration.hpp</code></a>).
This is a property of the embedding space alone, computed from intra- and
inter-actor reference-image pairs with no tracking or scene logic involved,
so it is a clean first read on discriminative power before running a
benchmark.</p>
<p><img alt="Calibrated P(match|similarity) for all four models" src="../assets/images/calibration_curves.png" /></p>
<table>
<thead>
<tr>
<th>model</th>
<th><code>a</code> (steepness)</th>
<th>a (steepness)</th>
<th>boundary at P=0.5</th>
</tr>
</thead>
<tbody>
<tr>
<td><strong>LVFace-B Glint360K</strong></td>
<td><strong>17.7</strong></td>
<td><strong>sim 0.228</strong></td>
<td>LVFace-B Glint360K</td>
<td>17.7</td>
<td>sim 0.228</td>
</tr>
<tr>
<td>ArcFace w600k-MBF</td>
@@ -846,13 +872,123 @@ power before running a single benchmark.</p>
</tr>
</tbody>
</table>
<p>LVFace has both the steepest transition and the lowest decision boundary — it
separates same-actor from different-actor reference pairs more confidently, at
a <em>lower</em> similarity threshold, than any ArcFace variant. That's a genuine
head start before the tracking/scoring pipeline is even involved.</p>
<h2 id="second-signal-f1-on-the-actual-benchmark">Second signal: F1 on the actual benchmark<a class="headerlink" href="#second-signal-f1-on-the-actual-benchmark" title="Permanent link">&para;</a></h2>
<p>Best full-gallery (no cast-restriction) result per model, from the 16-combo
bake-off matrix (<a href="../model-bakeoff/">full experiment log</a>):</p>
<p>LVFace has both the steepest transition and the lowest decision boundary,
separating same-actor from different-actor reference pairs more confidently
at a lower similarity than any ArcFace variant.</p>
<h2 id="second-signal-held-out-f1">Second signal: held-out F1<a class="headerlink" href="#second-signal-held-out-f1" title="Permanent link">&para;</a></h2>
<p>Each model's own tuned <code>full_exp</code> config, replayed against the 5 films the
optimizer never saw and scored the same way:</p>
<table>
<thead>
<tr>
<th>film</th>
<th>LVFace F1</th>
<th>mbf F1</th>
<th>r18 F1</th>
</tr>
</thead>
<tbody>
<tr>
<td>Benny &amp; Joon</td>
<td>83.0%</td>
<td>78.5%</td>
<td>77.1%</td>
</tr>
<tr>
<td>Lovelace</td>
<td>77.5%</td>
<td>73.7%</td>
<td>72.2%</td>
</tr>
<tr>
<td>Valerian and the City of a Thousand Planets</td>
<td>74.1%</td>
<td>70.2%</td>
<td>71.0%</td>
</tr>
<tr>
<td>Downton Abbey: A New Era</td>
<td>56.2%</td>
<td>55.0%</td>
<td>53.0%</td>
</tr>
<tr>
<td>The Many Saints of Newark</td>
<td>46.3%</td>
<td>44.5%</td>
<td>42.1%</td>
</tr>
<tr>
<td><strong>macro average</strong></td>
<td><strong>67.4%</strong></td>
<td><strong>64.4%</strong></td>
<td><strong>63.1%</strong></td>
</tr>
</tbody>
</table>
<p>LVFace scores highest on all 5 held-out films; the ranking never flips
between models. Total misID count across the 5 films: LVFace 1032, mbf
2197, r18 1224. LVFace has less than half mbf's misID total and still
scores higher on every film.</p>
<p>Held-out results are stronger evidence than training results, because
training numbers can reflect what the optimizer was tuned to fit rather
than general performance. On training data, the ordering is not as clean:</p>
<table>
<thead>
<tr>
<th>film</th>
<th>LVFace F1</th>
<th>mbf F1</th>
<th>r18 F1</th>
<th>best</th>
</tr>
</thead>
<tbody>
<tr>
<td>Café Society</td>
<td>68.1%</td>
<td>62.2%</td>
<td>60.1%</td>
<td>LVFace</td>
</tr>
<tr>
<td>Lord of War</td>
<td>75.6%</td>
<td>77.2%</td>
<td>75.6%</td>
<td>mbf</td>
</tr>
<tr>
<td>Scarface</td>
<td>71.5%</td>
<td>68.6%</td>
<td>64.1%</td>
<td>LVFace</td>
</tr>
<tr>
<td>Sound of Metal</td>
<td>78.8%</td>
<td>76.5%</td>
<td>71.6%</td>
<td>LVFace</td>
</tr>
</tbody>
</table>
<p>mbf beats LVFace on Lord of War (77.2% vs 75.6%), the only film in either
table where LVFace does not score highest. LVFace's training-set macro
average (75.3%, see <a href="../model-bakeoff/">the full experiment log</a>) is not a
uniform win across every film it contributes to; the held-out result, where
LVFace wins all 5 films outright, is the stronger claim.</p>
<p>This reverses an earlier, superseded benchmarking pass that used a
scene-union metric and found the three models statistically
indistinguishable (around 85% each), concluding LVFace was not worth its
size. That metric masked out-of-cast false positives behind a
gallery-intersect-cast recall filter; the per-second metric used here does
not.</p>
<h2 id="full-training-matrix-picture">Full training-matrix picture<a class="headerlink" href="#full-training-matrix-picture" title="Permanent link">&para;</a></h2>
<p><img alt="All 12 combos ranked by training-set F1" src="../assets/images/rep4_matrix_f1.png" /></p>
<p>Best full-gallery combo per model (all three are <code>full_exp</code>), from the
training matrix in <a href="../model-bakeoff/">the full experiment log</a>:</p>
<table>
<thead>
<tr>
@@ -865,18 +1001,18 @@ bake-off matrix (<a href="../model-bakeoff/">full experiment log</a>):</p>
</thead>
<tbody>
<tr>
<td><strong>LVFace-B Glint360K</strong></td>
<td><strong>75.3%</strong></td>
<td>LVFace-B Glint360K</td>
<td>75.3%</td>
<td>89.7%</td>
<td><strong>65.4%</strong></td>
<td>65.4%</td>
<td>232</td>
</tr>
<tr>
<td>ArcFace w600k-MBF</td>
<td>74.2%</td>
<td>87.4%</td>
<td>64.4%</td>
<td>57</td>
<td>72.0%</td>
<td>87.7%</td>
<td>61.4%</td>
<td>240</td>
</tr>
<tr>
<td>ArcFace R18</td>
@@ -885,36 +1021,18 @@ bake-off matrix (<a href="../model-bakeoff/">full experiment log</a>):</p>
<td>57.7%</td>
<td>242</td>
</tr>
<tr>
<td>ArcFace w600k-R50</td>
<td>68.5%</td>
<td>94.0%</td>
<td>54.1%</td>
<td>150</td>
</tr>
</tbody>
</table>
<p>The full 16-combo picture makes the model ordering visible at a glance — LVFace
(yellow) tops both the restricted and full columns, and R18 (green) props up
the bottom of the full-gallery ranking:</p>
<p><img alt="All 16 bake-off combos ranked by training-set F1" src="../assets/images/rep4_matrix_f1.png" /></p>
<p>LVFace wins outright, with the highest recall of any full-mode combo. This
reverses an earlier conclusion from a prior (superseded) benchmarking pass
using a scene-union metric, which found the three models statistically
indistinguishable (~85% each) and concluded LVFace wasn't worth its size — that
metric hid out-of-cast false positives behind a gallery∩cast recall mask (see
<a href="../optimizer-experiments/">the prior optimizer round</a>); the per-second metric
used here does not.</p>
<p>Held-out validation (5 films never seen by the optimizer) confirms LVFace's
lead holds up out of sample — see the
<a href="../lvface-deep-dive/">LVFace deep dive</a> for the full breakdown, including
where it fails.</p>
<h2 id="caveat-model-choice-is-an-operational-change">Caveat: model choice is an operational change<a class="headerlink" href="#caveat-model-choice-is-an-operational-change" title="Permanent link">&para;</a></h2>
<p>Switching the default embedder isn't just flipping a config value — the
gallery itself is model-specific (embeddings from different models aren't
comparable), so any existing gallery built against ArcFace w600k-R50 needs to
be rebuilt from source images against LVFace before the new default takes
effect. <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/reembed_gallery.py"><code>scripts/optimizer/reembed_gallery.py</code></a>
<p>LVFace leads within both the restricted and full gallery modes, visible
directly in the chart above without reading the table. The three models'
misID counts on the full gallery are nearly identical (232/240/242); LVFace's
lead here is a precision-and-recall lead, not a misID one.</p>
<h2 id="operational-note">Operational note<a class="headerlink" href="#operational-note" title="Permanent link">&para;</a></h2>
<p>Switching the default embedder is not a config change alone; the gallery
is model-specific, since embeddings from different models are not
comparable. Any existing gallery built against a different model must be
rebuilt from source images before the new default takes effect.
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/optimizer/reembed_gallery.py"><code>scripts/optimizer/reembed_gallery.py</code></a>
does this from a reference gallery's cached source images without
re-downloading anything.</p>
@@ -952,7 +1070,7 @@ re-downloading anything.</p>
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href=".." class="md-footer__link md-footer__link--prev" aria-label="Previous: Home">
<a href="../methodology/" class="md-footer__link md-footer__link--prev" aria-label="Previous: How We Score Against X-Ray">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
@@ -962,7 +1080,7 @@ re-downloading anything.</p>
Previous
</span>
<div class="md-ellipsis">
Home
How We Score Against X-Ray
</div>
</div>
</a>
+121 -118
View File
@@ -80,7 +80,7 @@
<div data-md-component="skip">
<a href="#whole-gallery-vs-limited-cast-restricted-gallery" class="md-skip">
<a href="#whole-gallery-vs-cast-restricted-gallery" class="md-skip">
Skip to content
</a>
@@ -243,6 +243,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -272,26 +291,7 @@
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
Full Experiment Log
</a>
</li>
@@ -395,6 +395,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -414,10 +441,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked>
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="">
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="">
@@ -435,8 +462,8 @@
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2">
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span>
@@ -541,10 +568,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#the-result" class="md-nav__link">
<a href="#result" class="md-nav__link">
<span class="md-ellipsis">
The result
Result
</span>
</a>
@@ -552,10 +579,10 @@
</li>
<li class="md-nav__item">
<a href="#why-this-isnt-the-shipped-default" class="md-nav__link">
<a href="#why-this-is-not-the-shipped-default" class="md-nav__link">
<span class="md-ellipsis">
Why this isn't the shipped default
Why this is not the shipped default
</span>
</a>
@@ -648,34 +675,7 @@
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
Full Experiment Log
@@ -742,10 +742,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#the-result" class="md-nav__link">
<a href="#result" class="md-nav__link">
<span class="md-ellipsis">
The result
Result
</span>
</a>
@@ -753,10 +753,10 @@
</li>
<li class="md-nav__item">
<a href="#why-this-isnt-the-shipped-default" class="md-nav__link">
<a href="#why-this-is-not-the-shipped-default" class="md-nav__link">
<span class="md-ellipsis">
Why this isn't the shipped default
Why this is not the shipped default
</span>
</a>
@@ -780,14 +780,15 @@
<h1 id="whole-gallery-vs-limited-cast-restricted-gallery">Whole gallery vs. limited (cast-restricted) gallery<a class="headerlink" href="#whole-gallery-vs-limited-cast-restricted-gallery" title="Permanent link">&para;</a></h1>
<p>Two ways to run the matcher: <strong>full</strong> scores every detected face against the
entire library gallery (2418 actors across the 9-film benchmark set); <strong>restricted</strong>
pre-filters each film's gallery down to just its Jellyfin-credited cast (typically
~15 top-billed actors) before the matcher ever runs.</p>
<h2 id="the-result">The result<a class="headerlink" href="#the-result" title="Permanent link">&para;</a></h2>
<p>Averaged across all 4 models and both expansion settings, on the 4 bake-off training
films:</p>
<h1 id="whole-gallery-vs-cast-restricted-gallery">Whole gallery vs. cast-restricted gallery<a class="headerlink" href="#whole-gallery-vs-cast-restricted-gallery" title="Permanent link">&para;</a></h1>
<p>Two ways to run the matcher. Full mode scores every detected face against
the entire 2418-actor gallery. Restricted mode pre-filters each film's
gallery down to just its Jellyfin-credited cast (typically around 15
top-billed actors) before the matcher runs.</p>
<h2 id="result">Result<a class="headerlink" href="#result" title="Permanent link">&para;</a></h2>
<p>Averaged across the 3 compared models (r50 excluded, see
<a href="../model-bakeoff/">the full experiment log</a>) and both expansion settings, on
the 4 training films:</p>
<table>
<thead>
<tr>
@@ -795,68 +796,70 @@ films:</p>
<th>F1</th>
<th>P</th>
<th>R</th>
<th>total misID (8 evals)</th>
<th>total misID</th>
</tr>
</thead>
<tbody>
<tr>
<td>full</td>
<td>71.2%</td>
<td>91.1%</td>
<td>59.0%</td>
<td>1073</td>
<td>71.1%</td>
<td>89.6%</td>
<td>59.6%</td>
<td>1121</td>
</tr>
<tr>
<td><strong>restricted</strong></td>
<td><strong>74.5%</strong></td>
<td>92.2%</td>
<td><strong>62.9%</strong></td>
<td><strong>329</strong></td>
<td>restricted</td>
<td>75.9%</td>
<td>90.4%</td>
<td>65.6%</td>
<td>299</td>
</tr>
</tbody>
</table>
<p>This is not a precision/recall trade — restriction wins on every axis at once:
<strong>+3.3pp F1, +3.9pp recall, and less than a third the total misIDs.</strong> Fewer
<p>Restriction improves every metric at once, not a precision/recall trade:
+4.8pp F1, +6.0pp recall, roughly a quarter the total misIDs. Fewer
candidates in the matcher's search space means fewer opportunities for a
look-alike false match (an actor who happens to share enough facial structure
with someone in the film, but isn't actually in it), and the recall gain shows
it isn't costing real detections to get there.</p>
<p>Per-model, every single model's best-scoring combo in the full 16-way matrix is
a <code>restricted</code> variant — visible directly in the ranking below (filled dots =
restricted, open = full; the filled dots cluster at the top for every color):</p>
<p><img alt="All 16 bake-off combos — filled dots (restricted) dominate the top" src="../assets/images/rep4_matrix_f1.png" /></p>
<p>See the full table in the
<a href="../model-bakeoff/">bake-off experiment log</a>. Two
combos hit <strong>zero</strong> true out-of-cast misidentifications:
<code>arcface_w600k_mbf_restricted_exp</code> (F1 76.5%) and, in full mode,
<code>LVFace-B_Glint360K_full_noexp</code> (F1 72.4%) — restriction isn't the only way to
reach misid=0, but it's the more reliable one.</p>
<h2 id="why-this-isnt-the-shipped-default">Why this isn't the shipped default<a class="headerlink" href="#why-this-isnt-the-shipped-default" title="Permanent link">&para;</a></h2>
<p>Cast-restriction is implemented today only as an <strong>offline optimizer technique</strong>
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/cast_restrict.py"><code>scripts/optimizer/cast_restrict.py</code></a>):
it pre-builds a filtered gallery file
per film, using Jellyfin's own cast list, before the benchmark ever calls the
matcher. There's no runtime "restrict matching to this title's credited cast"
switch in the shipped application<code>scene_analyze</code> always matches against
whatever single gallery file it's given.</p>
<p>Building that as a real feature would need, at minimum:</p>
lookalike false match, and the recall gain shows this does not cost real
detections.</p>
<p>Every model's best-scoring combo in the training matrix uses the
restricted gallery:</p>
<p><img alt="All combos ranked by training-set F1, filled dots are restricted" src="../assets/images/rep4_matrix_f1.png" /></p>
<p>See <a href="../model-bakeoff/">the full experiment log</a> for the complete table. One
combo reaches zero true out-of-cast misidentifications,
<code>arcface_w600k_mbf_restricted_exp</code> (F1 76.2%), and it is a restricted one,
consistent with restriction, not expansion, being what suppresses cross-film
confusions.</p>
<p>The restriction effect (+4.8pp averaged across models) is larger than the
model-choice effect: LVFace beats r18 by 6.2pp in full mode but beats mbf by
3.3pp. Restriction is the single strongest lever in the matrix.</p>
<h2 id="why-this-is-not-the-shipped-default">Why this is not the shipped default<a class="headerlink" href="#why-this-is-not-the-shipped-default" title="Permanent link">&para;</a></h2>
<p>Cast restriction is implemented today only as an offline optimizer
technique
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/optimizer/cast_restrict.py"><code>scripts/optimizer/cast_restrict.py</code></a>):
it pre-builds a filtered gallery file per film using Jellyfin's cast list
before the benchmark calls the matcher. There is no runtime "restrict to
this title's credited cast" switch in the shipped application;
<code>scene_analyze</code> always matches against whatever single gallery file it is
given.</p>
<p>Building this as a real feature requires:</p>
<ul>
<li>A live Jellyfin cast lookup at analysis time (the title is already known
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a>
already does this same lookup for its own
<code>filter_gallery</code>-based restriction path, just not wired into <code>scene_analyze</code>
itself as a first-class option).</li>
<li>A decision on the <em>fallback</em>: what happens to a real, uncredited cameo
(see the Germar Terrell Gardner case in the LVFace deep-dive) if the gallery
never includes them at all?</li>
<li>Regenerating the restricted-gallery cache whenever the title's Jellyfin cast
list changes.</li>
<li>A live Jellyfin cast lookup at analysis time. The title is already known,
and <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a>
already performs this lookup for its own <code>filter_gallery</code>-based
restriction path; it is not wired into <code>scene_analyze</code> as a first-class
option.</li>
<li>A decision on the fallback case: what happens to a real, uncredited
cameo (see the Germar Terrell Gardner and Talia Balsam cases in the
<a href="../lvface-deep-dive/#where-lvface-beat-x-ray">LVFace deep dive</a>) if the
restricted gallery never includes them at all.</li>
<li>Regenerating the restricted-gallery cache whenever a title's Jellyfin
cast list changes.</li>
</ul>
<p>This is why the shipped <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/config.hpp"><code>src/config.hpp</code></a>
defaults use the <code>full</code>-mode winner
(<code>LVFace-B_Glint360K_full_exp</code>, F1 75.3% training / 67.4% held-out macro) rather
than the higher-scoring <code>restricted_exp</code> (78.3%) — the 78.3% number describes a
capability the app doesn't have yet, not what actually ships.</p>
<p>The shipped <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/config.hpp"><code>src/config.hpp</code></a> defaults use
the full-mode winner (<code>LVFace-B_Glint360K_full_exp</code>, F1 75.3% training,
67.4% held-out macro) rather than the higher-scoring <code>restricted_exp</code>
(78.3%), because 78.3% describes a capability the application does not
have yet.</p>
+110 -108
View File
@@ -14,7 +14,7 @@
<link rel="next" href="best-model/">
<link rel="next" href="methodology/">
@@ -243,6 +243,25 @@
<li class="md-tabs__item">
<a href="methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
<li class="md-tabs__item">
@@ -270,26 +289,7 @@
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item">
<a href="optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
Full Experiment Log
</a>
</li>
@@ -427,10 +427,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#start-here-four-questions-this-bake-off-answers" class="md-nav__link">
<a href="#findings" class="md-nav__link">
<span class="md-ellipsis">
Start here — four questions this bake-off answers
Findings
</span>
</a>
@@ -438,10 +438,10 @@
</li>
<li class="md-nav__item">
<a href="#the-full-technical-log" class="md-nav__link">
<a href="#full-experiment-log" class="md-nav__link">
<span class="md-ellipsis">
The full technical log
Full experiment log
</span>
</a>
@@ -473,6 +473,33 @@
<li class="md-nav__item">
<a href="methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -487,10 +514,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" >
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="0">
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
@@ -508,8 +535,8 @@
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_2">
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span>
@@ -652,34 +679,7 @@
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
Full Experiment Log
@@ -746,10 +746,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#start-here-four-questions-this-bake-off-answers" class="md-nav__link">
<a href="#findings" class="md-nav__link">
<span class="md-ellipsis">
Start here — four questions this bake-off answers
Findings
</span>
</a>
@@ -757,10 +757,10 @@
</li>
<li class="md-nav__item">
<a href="#the-full-technical-log" class="md-nav__link">
<a href="#full-experiment-log" class="md-nav__link">
<span class="md-ellipsis">
The full technical log
Full experiment log
</span>
</a>
@@ -796,79 +796,81 @@
<h1 id="scene-actor-extraction">scene-actor-extraction<a class="headerlink" href="#scene-actor-extraction" title="Permanent link">&para;</a></h1>
<p>A face-recognition pipeline that finds when each actor appears on screen in a
film or TV episode built on <a href="https://gitea.tourolle.paris/dtourolle/KPN">KPN++</a>
(a C++20 Kahn Process Network library) for the detect track match → scene
pipeline, with a Jellyfin-integrated gallery and an X-Ray-validated optimizer.</p>
<p>This is a perfect X-Ray second, on a film the optimizer never saw:</p>
<p>A face-recognition pipeline that finds when each actor appears on screen in
a film or TV episode, built on <a href="https://gitea.tourolle.paris/dtourolle/KPN">KPN++</a>
(a C++20 Kahn Process Network library) for the detect, track, match, and
scene pipeline, with a Jellyfin-integrated gallery and an X-Ray-validated
optimizer.</p>
<p>This is a correctly scored second from a held-out film, one the optimizer
never saw during tuning:</p>
<p><img alt="A perfect X-Ray second: three faces named at 100%, two more correctly carried off-screen" src="assets/images/lovelace_perfect_second.jpg" /></p>
<p>Every visible face named at 100% Chris Noth, Hank Azaria, Bobby Cannavale —
the background extra honestly left unnamed, and the two credited cast without
a visible face correctly carried as present off-screen by the tracker's
presence windows. That's the pipeline exactly reproducing Amazon X-Ray's
record for this second.</p>
<p>It doesn't always go like that: the hardest held-out film scores 46% F1, and
the report is honest about <em>why</em> one tunable trade (extinction bridging at
hard cuts), one structural ceiling (X-Ray credits people whose faces never
appear), and a few cases where the pipeline is right and X-Ray is wrong. The
evidence for all of it is in the pages below.</p>
<h2 id="start-here-four-questions-this-bake-off-answers">Start here — four questions this bake-off answers<a class="headerlink" href="#start-here-four-questions-this-bake-off-answers" title="Permanent link">&para;</a></h2>
<p>Every visible face is named at 100% confidence (Chris Noth, Hank Azaria,
Bobby Cannavale), the background extra is correctly left unnamed, and the
two credited cast members without a visible face are correctly reported
present but not visible. This matches Amazon X-Ray's own record for this
second exactly.</p>
<p>Results are not uniform across films. The hardest held-out film scores 46%
F1. This report documents why: one tunable trade (extinction bridging at
hard cuts), one structural limit (X-Ray credits people whose faces never
appear on screen), and a small number of cases where the pipeline is
correct and X-Ray's ground truth is not. Read
<a href="methodology/">how we score against X-Ray</a> first. X-Ray's ground truth is
scene-level; the pipeline's output is per-second. That difference shapes
every finding below.</p>
<h2 id="findings">Findings<a class="headerlink" href="#findings" title="Permanent link">&para;</a></h2>
<div class="grid cards">
<ul>
<li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M18 2c-.9 0-2 1-2 2H8c0-1-1.1-2-2-2H2v9c0 1 1 2 2 2h2.2c.4 2 1.7 3.7 4.8 4v2.08C8 19.54 8 22 8 22h8s0-2.46-3-2.92V17c3.1-.3 4.4-2 4.8-4H20c1 0 2-1 2-2V2zM6 11H4V4h2zm14 0h-2V4h2z"/></svg></span> <strong><a href="best-model/">Which model is best?</a></strong></p>
<hr />
<p>Calibration curves first (discriminative power, independent of any
threshold), then F1 on the actual benchmark. LVFace-B Glint360K wins
both.</p>
<p>Calibration curves first, independent of any threshold, then held-out
F1 across three models. LVFace-B Glint360K wins both, and wins on every
held-out film.</p>
</li>
<li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M14 12v7.88c.04.3-.06.62-.29.83a.996.996 0 0 1-1.41 0l-2.01-2.01a.99.99 0 0 1-.29-.83V12h-.03L4.21 4.62a1 1 0 0 1 .17-1.4c.19-.14.4-.22.62-.22h14c.22 0 .43.08.62.22a1 1 0 0 1 .17 1.4L14.03 12z"/></svg></span> <strong><a href="gallery-scope/">Whole vs. cast-restricted gallery</a></strong></p>
<hr />
<p>Restricting the matcher to a film's credited cast is a clean win on
every axis (+3.3pp F1, less than a third the misIDs) — but isn't a
shipped runtime feature yet.</p>
<p>Restricting the matcher to a film's credited cast improves F1,
recall, and misID rate at once, but is not a shipped runtime feature
yet.</p>
</li>
<li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m12 0-.66.03 3.81 3.81L16.5 2.5c3.25 1.57 5.59 4.74 5.95 8.5h1.5C23.44 4.84 18.29 0 12 0m0 4c-1.93 0-3.5 1.57-3.5 3.5S10.07 11 12 11s3.5-1.57 3.5-3.5S13.93 4 12 4M.05 13C.56 19.16 5.71 24 12 24l.66-.03-3.81-3.81L7.5 21.5c-3.25-1.56-5.59-4.74-5.95-8.5zM12 13c-3.87 0-7 1.57-7 3.5V18h14v-1.5c0-1.93-3.13-3.5-7-3.5"/></svg></span> <strong><a href="pose-expansion/">Does pose expansion help?</a></strong></p>
<hr />
<p>A convincing training-set effect that didn't reproduce on 5 held-out
films once two methodology bugs were caught and fixed. An honest null
result, not a forced narrative.</p>
<p>A training-set effect that did not reproduce on 5 held-out films once
two methodology bugs in the comparison harness were found and fixed.</p>
</li>
<li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M18 16h-.58l-.81-.81A7.07 7.07 0 0 0 18 11c0-3.87-3.13-7-7-7-1.5 0-3 .5-4.21 1.4-3.09 2.32-3.72 6.71-1.4 9.8s6.71 3.72 9.8 1.4l.81.81V18l5 5 2-2zm-7 0c-2.76 0-5-2.24-5-5s2.24-5 5-5 5 2.24 5 5-2.24 5-5 5M3 6 1 8V1h7L6 3H3zm18-5v7l-2-2V3h-3l-2-2zM6 19l2 2H1v-7l2 2v3z"/></svg></span> <strong><a href="lvface-deep-dive/">Deep dive: LVFace-B Glint360K</a></strong></p>
<hr />
<p>The held-out generalization gap, how the error budget decomposes
(extinction bridging at hard cuts, X-Ray's scene-membership vs.
on-screen-face ceiling), and the frames where the pipeline is right
and the ground truth is wrong.</p>
<p>The held-out generalization gap, the two mechanisms behind its errors,
and every distinct case where it names someone outside the film's
credited cast.</p>
</li>
</ul>
</div>
<h2 id="the-full-technical-log">The full technical log<a class="headerlink" href="#the-full-technical-log" title="Permanent link">&para;</a></h2>
<h2 id="full-experiment-log">Full experiment log<a class="headerlink" href="#full-experiment-log" title="Permanent link">&para;</a></h2>
<ul>
<li><strong><a href="model-bakeoff/">Model bake-off + threshold re-tune</a></strong>
the complete experiment log behind the four pages above: the ROCm teardown
deadlock root cause and fix, DE concurrency tuning, the full 16-combo
results table, and every caveat. This is where the shipped
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/config.hpp"><code>src/config.hpp</code></a> defaults come from.</li>
<li><strong><a href="optimizer-experiments/">Optimizer experiments (prior round)</a></strong> — the
earlier scene-union-metric tuning pass, superseded by the per-second metric
used in the bake-off but kept for the ground-truth/architecture background.</li>
<li><strong><a href="service-conversion/">Service conversion (proposal)</a></strong> — design sketch
for a native idle-GPU worker gated on screen lock, not yet built.</li>
<li><strong><a href="model-bakeoff/">Full experiment log</a></strong>: the complete log behind the
four pages above, including how replaying against cached embeddings
inside the same KPN network makes a full model and configuration
comparison practical, the full results table, and every caveat. This is
where the shipped <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/config.hpp"><code>src/config.hpp</code></a>
defaults come from.</li>
<li><strong><a href="service-conversion/">Service conversion (proposal)</a></strong>: design
sketch for a native idle-GPU worker gated on screen lock, not yet built.</li>
</ul>
<h2 id="reproducing-the-benchmarks">Reproducing the benchmarks<a class="headerlink" href="#reproducing-the-benchmarks" title="Permanent link">&para;</a></h2>
<p>Gallery <code>.h5</code> files, embedding dumps, the X-Ray corpus, montage frame images,
and DE trajectories are not committed to this repository — they're pushed to
the Gitea package registry and pulled on demand:</p>
<p>Gallery <code>.h5</code> files, embedding dumps, the X-Ray corpus, montage frame
images, and DE trajectories are not committed to this repository. They are
pushed to the Gitea package registry and pulled on demand:</p>
<div class="language-bash highlight"><pre><span></span><code><span id="__span-0-1"><a id="__codelineno-0-1" name="__codelineno-0-1" href="#__codelineno-0-1"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>galleries
</span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>experiment-data
</span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>montage-frames<span class="w"> </span>&lt;film-slug&gt;
</span></code></pre></div>
<p>See <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/artifacts/push_artifacts.sh"><code>scripts/artifacts/push_artifacts.sh</code></a>
for the upload side (requires a <code>GITEA_TOKEN</code> with package write scope).</p>
<p>See <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/artifacts/push_artifacts.sh"><code>scripts/artifacts/push_artifacts.sh</code></a>
for the upload side, which requires a <code>GITEA_TOKEN</code> with package write
scope.</p>
@@ -905,13 +907,13 @@ for the upload side (requires a <code>GITEA_TOKEN</code> with package write scop
<a href="best-model/" class="md-footer__link md-footer__link--next" aria-label="Next: Best Model">
<a href="methodology/" class="md-footer__link md-footer__link--next" aria-label="Next: How We Score Against X-Ray">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Best Model
How We Score Against X-Ray
</div>
</div>
<div class="md-footer__button md-icon">
+481 -196
View File
@@ -243,6 +243,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -272,26 +291,7 @@
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
Full Experiment Log
</a>
</li>
@@ -395,6 +395,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -414,10 +441,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked>
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="">
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="">
@@ -435,8 +462,8 @@
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2">
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span>
@@ -597,10 +624,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#what-good-looks-like" class="md-nav__link">
<a href="#baseline-correctly-scored-seconds" class="md-nav__link">
<span class="md-ellipsis">
What good looks like
Baseline: correctly scored seconds
</span>
</a>
@@ -619,10 +646,10 @@
</li>
<li class="md-nav__item">
<a href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" class="md-nav__link">
<a href="#mechanism-1-extinction-bridging" class="md-nav__link">
<span class="md-ellipsis">
Mechanism 1: extinction bridging — usually right, wrong at hard cuts
Mechanism 1: extinction bridging
</span>
</a>
@@ -638,6 +665,78 @@
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#every-distinct-out-of-cast-name" class="md-nav__link">
<span class="md-ellipsis">
Every distinct out-of-cast name
</span>
</a>
<nav class="md-nav" aria-label="Every distinct out-of-cast name">
<ul class="md-nav__list">
<li class="md-nav__item">
<a href="#the-many-saints-of-newark-4-names" class="md-nav__link">
<span class="md-ellipsis">
The Many Saints of Newark: 4 names
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#lord-of-war-3-names" class="md-nav__link">
<span class="md-ellipsis">
Lord of War: 3 names
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#lovelace-1-name" class="md-nav__link">
<span class="md-ellipsis">
Lovelace: 1 name
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#scarface-1-name" class="md-nav__link">
<span class="md-ellipsis">
Scarface: 1 name
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#summary-of-the-nine" class="md-nav__link">
<span class="md-ellipsis">
Summary of the nine
</span>
</a>
</li>
</ul>
</nav>
</li>
<li class="md-nav__item">
@@ -692,34 +791,7 @@
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
Full Experiment Log
@@ -786,10 +858,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#what-good-looks-like" class="md-nav__link">
<a href="#baseline-correctly-scored-seconds" class="md-nav__link">
<span class="md-ellipsis">
What good looks like
Baseline: correctly scored seconds
</span>
</a>
@@ -808,10 +880,10 @@
</li>
<li class="md-nav__item">
<a href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" class="md-nav__link">
<a href="#mechanism-1-extinction-bridging" class="md-nav__link">
<span class="md-ellipsis">
Mechanism 1: extinction bridging — usually right, wrong at hard cuts
Mechanism 1: extinction bridging
</span>
</a>
@@ -827,6 +899,78 @@
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#every-distinct-out-of-cast-name" class="md-nav__link">
<span class="md-ellipsis">
Every distinct out-of-cast name
</span>
</a>
<nav class="md-nav" aria-label="Every distinct out-of-cast name">
<ul class="md-nav__list">
<li class="md-nav__item">
<a href="#the-many-saints-of-newark-4-names" class="md-nav__link">
<span class="md-ellipsis">
The Many Saints of Newark: 4 names
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#lord-of-war-3-names" class="md-nav__link">
<span class="md-ellipsis">
Lord of War: 3 names
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#lovelace-1-name" class="md-nav__link">
<span class="md-ellipsis">
Lovelace: 1 name
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#scarface-1-name" class="md-nav__link">
<span class="md-ellipsis">
Scarface: 1 name
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#summary-of-the-nine" class="md-nav__link">
<span class="md-ellipsis">
Summary of the nine
</span>
</a>
</li>
</ul>
</nav>
</li>
<li class="md-nav__item">
@@ -869,43 +1013,44 @@
<h1 id="deep-dive-lvface-b-glint360k">Deep dive: LVFace-B Glint360K<a class="headerlink" href="#deep-dive-lvface-b-glint360k" title="Permanent link">&para;</a></h1>
<p>LVFace won the model bake-off (see <a href="../best-model/">Which model is best?</a>) and is
the shipped default embedder. This page is the honest accounting of how it
actually performs — what a good second looks like, where the errors actually
come from, and two cases where the ground truth itself is wrong and LVFace is
right.</p>
<p>LVFace won the model comparison (see <a href="../best-model/">Which model is best?</a>)
and is the shipped default embedder. This page reports how it performs in
detail: a baseline of correct output, the two mechanisms behind its errors,
and every distinct case where it names someone who is not in the film's
credited cast.</p>
<p>Read <a href="../methodology/">How we score against X-Ray</a> first. X-Ray's ground truth
is scene-level, not per-frame. A name marked correct in the Offscreen column
below is the pipeline correctly reporting scene membership, not a workaround.</p>
<div class="admonition note">
<p class="admonition-title">How to read the frames on this page</p>
<p>The top is the film frame, with a box and name on every face the pipeline
identified. The bottom panels are the per-second verdict against X-Ray:
<strong>Onscreen</strong> lists faces named in the frame, <strong>Offscreen</strong> lists cast
X-Ray marks present in the scene without a visible face — presence
carried by the tracker's windows, not by a detection. Colors are the
score: <span style="color:#0ca30c"><strong>green</strong></span> = correct (TPI),
<span style="color:#eb6834"><strong>orange</strong></span> = wrong (FPI),
<span style="color:#3987e5"><strong>blue</strong></span> = missed (FN).</p>
<p>The top of each image is the film frame, with a box and name on every
face the pipeline matched to a real detection. The panels below are the
per-second result against X-Ray. <strong>Onscreen</strong> lists names attached to a
visible face this second. <strong>Offscreen</strong> lists names the pipeline reports
present without a currently visible face. Colors mark the verdict:
<span style="color:#0ca30c"><strong>green</strong></span> correct (TPI),
<span style="color:#eb6834"><strong>orange</strong></span> wrong (FPI),
<span style="color:#3987e5"><strong>blue</strong></span> missed (FN).</p>
</div>
<h2 id="what-good-looks-like">What good looks like<a class="headerlink" href="#what-good-looks-like" title="Permanent link">&para;</a></h2>
<h2 id="baseline-correctly-scored-seconds">Baseline: correctly scored seconds<a class="headerlink" href="#baseline-correctly-scored-seconds" title="Permanent link">&para;</a></h2>
<p><img alt="Wedding couple correctly identified, Downton Abbey: A New Era" src="../assets/images/downton_wedding_couple.jpg" /></p>
<p>Six faces on screen, all six named correctly including Penelope Wilton at the
edge of the pews and a half-occluded Michelle Dockery — while thirteen more
cast members X-Ray marks present in the scene are correctly carried as
"Offscreen" by their presence windows. One miss in the whole frame: Maggie
Smith (blue). Score for this second: 0.86.</p>
<p>Six faces on screen, all six named correctly, including Penelope Wilton at
the edge of the pews and a partly occluded Michelle Dockery. Thirteen more
cast members X-Ray lists as present in the scene are correctly reported
Offscreen. One miss: Maggie Smith (blue). Score for this second: 0.86.</p>
<p><img alt="19 of 20 correct in the funeral crowd" src="../assets/images/downton_funeral_19of20.jpg" /></p>
<p>The same film's funeral gathering: mourning dress, hats, half the faces turned.
<strong>Nineteen of the twenty cast X-Ray lists for this scene are scored correctly</strong>
seven named on screen at up to 100% confidence, twelve more correctly held
as present off-screen.</p>
<p>And the pipeline doesn't need the face to be <em>real</em>:</p>
<p>The same film's funeral scene: dark clothing, hats, half the faces turned
away. Nineteen of the twenty cast members X-Ray lists for this scene score
correct: seven named on screen at up to 100% confidence, twelve more reported
correctly as present but not visible.</p>
<p><img alt="Herbie Hancock identified on an in-fiction video call" src="../assets/images/valerian_screen_call.jpg" /></p>
<p>That's Herbie Hancock at 98% — as a face on a <em>screen inside the movie</em>, over a
sci-fi HUD overlay, during a video call in Valerian. A face is a face, whether
it's in the room or on the bridge's comms display.</p>
<p>The pipeline does not require a live face. This is Herbie Hancock at 98%
confidence, identified from a face displayed on a screen inside the film, on
a video call under a science-fiction HUD overlay.</p>
<h2 id="training-vs-held-out-the-generalization-gap">Training vs. held-out: the generalization gap<a class="headerlink" href="#training-vs-held-out-the-generalization-gap" title="Permanent link">&para;</a></h2>
<p>The shipped config (<code>prob_threshold=0.754, anneal_sec=35.54,
extinction_sec=57.43, expand_gallery=true</code>) was tuned against 4 films. Scored
against the 5 films the optimizer never saw:</p>
<p>The shipped config (<code>prob_threshold=0.754</code>, <code>anneal_sec=35.54</code>,
<code>extinction_sec=57.43</code>, <code>expand_gallery=true</code>) was tuned on 4 films. Scored
on the 5 films the optimizer never saw:</p>
<p><img alt="Held-out per-film F1 vs. the training-set fit" src="../assets/images/holdout_f1_by_film.png" /></p>
<table>
<thead>
@@ -962,18 +1107,18 @@ against the 5 films the optimizer never saw:</p>
<td>80084</td>
</tr>
<tr>
<td><strong>The Many Saints of Newark</strong></td>
<td><strong>46.3%</strong></td>
<td><strong>54.7%</strong></td>
<td>The Many Saints of Newark</td>
<td>46.3%</td>
<td>54.7%</td>
<td>40.1%</td>
<td>15922</td>
<td>4394</td>
<td><strong>974</strong></td>
<td>974</td>
<td>23791</td>
</tr>
<tr>
<td><strong>macro average</strong></td>
<td><strong>67.4%</strong></td>
<td>macro average</td>
<td>67.4%</td>
<td>85.8%</td>
<td>57.0%</td>
<td></td>
@@ -983,112 +1128,252 @@ against the 5 films the optimizer never saw:</p>
</tr>
</tbody>
</table>
<p><strong>67.4% held-out vs. 75.3% on training</strong> — an ~8pp drop, and a <strong>37pp spread
between the best and worst held-out film</strong>. The config does not generalize
uniformly, and the spread traces to two mechanisms, both visible frame by
frame below.</p>
<h2 id="mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts">Mechanism 1: extinction bridging — usually right, wrong at hard cuts<a class="headerlink" href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" title="Permanent link">&para;</a></h2>
<p>The extinction window keeps an identity alive through seconds where no face is
detectable. <strong>Most of the time this is exactly what you want</strong>, and it's where
a lot of the TPI count comes from:</p>
<p>The <code>P</code> column is misID-weighted (each out-of-film name counts 10x in the
denominator; see <a href="../methodology/#precision-recall-and-the-misid-weighting">methodology</a>).
That weighting is why Many Saints reads 54.7% here despite naming mostly real,
present faces: its raw (unweighted) precision is <strong>78.4%</strong>, and the gap is
entirely its 974 misIDs paying the 10x penalty. The three zero-misID films
(Benny &amp; Joon, Downton, Valerian) have identical weighted and raw precision;
Lovelace, with 58 misIDs, sits 3pp below its raw 93.3%.</p>
<p>Held-out F1 is 67.4%, against 75.3% on training, an 8pp drop. The spread
between the best and worst held-out film is 37pp. This is not unique to
LVFace: <a href="../model-bakeoff/#held-out-validation-all-3-models">the full experiment log</a>
shows mbf and r18 with the same shape of spread on the same films, at a
uniformly lower level. Two mechanisms explain the spread. Both are shown
below with frame-level evidence.</p>
<h2 id="mechanism-1-extinction-bridging">Mechanism 1: extinction bridging<a class="headerlink" href="#mechanism-1-extinction-bridging" title="Permanent link">&para;</a></h2>
<p>The extinction window keeps a name reported as present for up to
<code>extinction_sec</code> after its last real detection. This is deliberate: most
gaps in face visibility are short (a turned head, an occlusion, a cut to a
reaction shot), and the window bridges them.</p>
<p><img alt="Two faces on screen, six more correctly bridged" src="../assets/images/lovelace_polygraph_bridged.jpg" /></p>
<p>Lovelace's polygraph scene: only Eric Roberts and Amanda Seyfried have visible
faces, but X-Ray lists eight cast present — and all eight score green, the
other six correctly carried by presence windows through a scene where the
camera never shows them. A perfect second, and the extinction/anneal machinery
is <em>why</em>.</p>
<p>The same mechanism has a failure case: a hard cut into long faceless footage.
Both Many Saints of Newark (974 misIDs) and Downton Abbey (FN=80084, the worst
recall of the five) are dominated by it — verified directly against the raw
per-frame stream and the HDF5 dump's own detection counts, not inferred from
the score alone. <strong>This is not a malfunction</strong>: the tracker is doing exactly
what its window is for; the footage just stops cooperating. In the debug
overlay (which draws a bridged identity's last-known bbox, unlike the shipped
output, which emits presence windows and no boxes at all) the bridged state is
visible spatially:</p>
<p><img alt="Debug overlay: bridged identities drawn at their last-known positions" src="../assets/images/many_saints_ghost_fpi.jpg" />
<em>Debug-overlay rendering (<code>dump_error_frames.py --raw</code>): "Jon Bernthal", "Joey
Diaz" and "Billy Magnussen" are extinction-bridged identities from the previous
shot, drawn frozen over the wall and the hanging plates. Frame
<code>many_saints/fpi/fpi_t03543.jpg</code>, <code>montage-frames</code> artifact package.</em></p>
<p>The cost is measurable, not just visible. Downton Abbey's hard cut into its
closing credits, plotting the dump's own per-second <code>face_count</code> (detector
output, independent of the tracker) against what the tracker reports:</p>
<p>Lovelace's polygraph scene: only Eric Roberts and Amanda Seyfried have
visible faces. X-Ray lists eight cast members present. All eight score
correct; the other six are reported Offscreen through a stretch where the
camera never shows them. The extinction window is why.</p>
<p>The same mechanism fails at a hard cut into a long stretch with no faces at
all. Downton Abbey's recall (39.4%, the worst of the five held-out films) is
dominated by this failure. It is verified directly against the raw
per-frame stream and the dump's own detection counts, not inferred from the
score. Plotting the dump's per-second <code>face_count</code> (detector output,
independent of the tracker) against what the tracker reports, through
Downton Abbey's hard cut into its closing credits:</p>
<p><img alt="Detector vs. tracker through Downton Abbey's cut to credits" src="../assets/images/downton_ghost_timeline.png" /></p>
<p>From the cut onward the detector sees <strong>zero faces for nearly a minute</strong> — and
the tracker keeps reporting the last shot's 15 identities the whole time
(verified for Hugh Bonneville: bbox <code>(1743.2, 0.0, 171.3, 317.8)</code>, unchanged to
the pixel, at every sampled second for 57+ seconds). The staircase at the right
edge is the extinction window expiring actor by actor. That plateau is
<code>SceneTrackerFunc::active_[actor_idx].last_bbox</code>
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/nodes/scene_tracker_node.hpp"><code>src/nodes/scene_tracker_node.hpp</code></a>)
re-emitted as designed: <code>extinction_sec=57.4</code> was tuned long because bridging
wins on most footage (see the polygraph frame above) — the training films just
never contained a faceless stretch long enough to show the cost side, and the
held-out set did.</p>
<p>The same track-continuation machinery has one milder spatial artifact, worth
knowing when reading these frames:</p>
<p><img alt="Two labels on one face after a shot/reverse-shot cut" src="../assets/images/cafe_society_rapid_cut.jpg" />
<em>Café Society (a training film), a shot/reverse-shot dialog: that is Steve
Carell wearing both his own label and Jesse Eisenberg's.</em></p>
<p>At a rapid cut, the previous shot's track can linger for a beat at nearly the
same screen position the new face occupies — here Jesse Eisenberg's box from
the counter-shot lands on Steve Carell. Note what the score panel says,
though: both actors are green, because both <em>are</em> present in this dialog
scene per X-Ray. The spatial label is briefly wrong; the per-second presence
claim — the thing the pipeline actually ships — is right. It's the same trade
as the extinction window: track continuation smooths over cuts, and 1 fps
sampling occasionally catches the seam.</p>
<p>From the cut onward the detector reports zero faces for close to a minute.
The tracker continues reporting the previous shot's 15 identities for the
same span (verified for Hugh Bonneville: bbox <code>(1743.2, 0.0, 171.3, 317.8)</code>,
unchanged to the pixel, at every sampled second for 57 seconds). The
staircase at the right edge is the extinction window expiring, actor by
actor. This is <code>SceneTrackerFunc::active_[actor_idx].last_bbox</code>
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/nodes/scene_tracker_node.hpp"><code>src/nodes/scene_tracker_node.hpp</code></a>)
re-emitted as designed. <code>extinction_sec=57.4</code> was tuned long because
bridging is correct on most footage, as in the polygraph scene above. The
training films did not contain a faceless stretch long enough to expose the
cost side; the held-out set did.</p>
<p>The extinction window is a scoring concept, not something drawn on screen.
The shipped output is presence windows with no bounding boxes. Even the
debug overlay used for this report never draws a box for a bridged name: a
name inside its extinction window with no current detection appears only as
a name in the Offscreen column, the same as every correctly bridged name
above.</p>
<p>A related, smaller effect shows up at rapid cuts:</p>
<p><img alt="Two labels on one face after a shot/reverse-shot cut" src="../assets/images/cafe_society_rapid_cut.jpg" /></p>
<p>Café Society (a training film), a shot/reverse-shot dialog. The box on Steve
Carell's face carries two labels: his own, and Jesse Eisenberg's, left over
from the counter-shot a moment earlier. Both names score correct, because
both actors are present in this scene per X-Ray. The box position is
briefly wrong; the presence claim, which is what the pipeline ships, is
right.</p>
<h2 id="mechanism-2-the-face-vs-presence-ceiling">Mechanism 2: the face-vs-presence ceiling<a class="headerlink" href="#mechanism-2-the-face-vs-presence-ceiling" title="Permanent link">&para;</a></h2>
<p>Downton Abbey's recall didn't collapse because faces were misread — it
collapsed because for most of its 80084 FN-seconds there was <strong>no face to
read</strong>:</p>
<p>Downton Abbey's recall did not collapse because faces were misread. It
collapsed because for most of its 80084 false-negative seconds there was no
face to read.</p>
<p><img alt="22 cast credited, nobody facing the camera" src="../assets/images/downton_crew_fn.jpg" /></p>
<p>A newsreel crew hauls equipment through the hall: X-Ray credits 22 cast as
present in this scene; not one face looks at the camera. Eight are still
scored green (windows bridging from adjacent shots) — the other fourteen are
blue FNs that no face-recognition pipeline could ever recover. X-Ray encodes
<em>scene membership</em>; the pipeline measures <em>on-screen faces</em>. In ensemble films
those two definitions diverge massively, and that gap — not identification
error — is most of what the FN column counts.</p>
<p><img alt="Presence without a detectable face, The Many Saints of Newark" src="../assets/images/many_saints_outofcast_fpi.jpg" /></p>
<p>Same ceiling from the other side: Michela De Rossi in frame but turned away,
five cast correctly bridged as offscreen (green), four blue FNs — and one
orange we'll come back to below.</p>
<p>A newsreel crew moves equipment through the hall. X-Ray credits 22 cast
members as present in this scene. None face the camera. Eight still score
correct, carried by presence windows from adjacent shots. The other fourteen
are missed, and no face-recognition system can recover them, because there
is no face in the frame. X-Ray records scene membership; the pipeline
measures visible faces. In ensemble scenes these two quantities diverge, and
that gap accounts for most of the false-negative count.</p>
<h2 id="every-distinct-out-of-cast-name">Every distinct out-of-cast name<a class="headerlink" href="#every-distinct-out-of-cast-name" title="Permanent link">&para;</a></h2>
<p>Many Saints of Newark has the largest misID count of any held-out film: 974
seconds, weighted. Rather than characterize this from a single frame, the
raw replay stream was searched directly for every name the pipeline reports
that is not in the film's credited cast. The same search was run on all 9
films in the benchmark, one rule applied uniformly: <strong>find the first second
each distinct out-of-cast name appears, and render that exact second.</strong></p>
<p>Five films produce no such name anywhere in their runtime: Benny &amp; Joon,
Café Society, Downton Abbey, Sound of Metal, Valerian. Zero out-of-cast
names across their entire length. Four films produce nine distinct names
between them, shown below in full, not a sample.</p>
<h3 id="the-many-saints-of-newark-4-names">The Many Saints of Newark: 4 names<a class="headerlink" href="#the-many-saints-of-newark-4-names" title="Permanent link">&para;</a></h3>
<p><img alt="Germar Terrell Gardner, first out-of-cast name in Many Saints" src="../assets/images/many_saints_fpi_gardner.jpg" /></p>
<p>Germar Terrell Gardner, t=848s, 78% confidence. A real, clearly visible
background actor. He is not in X-Ray's cast list for this film, but he is
credited in Jellyfin's independent cast metadata (see
<a href="#where-lvface-beat-x-ray">Where LVFace beat X-Ray</a> below). This is a
ground-truth gap, not a model error.</p>
<p><img alt="Archie Yates, second out-of-cast name in Many Saints" src="../assets/images/many_saints_fpi_yates.jpg" /></p>
<p>Archie Yates, t=2521s, 78% confidence. A real detected face, a genuine
lookalike confusion.</p>
<p><img alt="Zooey Deschanel, third out-of-cast name in Many Saints" src="../assets/images/many_saints_fpi_deschanel.jpg" /></p>
<p>Zooey Deschanel, t=2819s, 99% confidence. A real detected face at a dinner
table, high-confidence lookalike confusion.</p>
<p><img alt="Talia Balsam, fourth out-of-cast name in Many Saints" src="../assets/images/many_saints_fpi_balsam.jpg" /></p>
<p>Talia Balsam, t=4551s, 93% confidence. A real detected face. Talia Balsam
plays Mrs. Jarecki, a guidance counselor, in this film; she is confirmed
on screen by direct inspection of the frame. She does not appear in X-Ray's
<code>people.csv</code> for this title. This is a second ground-truth gap in the same
film, not a model error.</p>
<p>Two of these four names are ground-truth gaps (Gardner, Balsam), not
misidentifications. The other two (Yates, Deschanel) are genuine embedding
errors on real faces.</p>
<h3 id="lord-of-war-3-names">Lord of War: 3 names<a class="headerlink" href="#lord-of-war-3-names" title="Permanent link">&para;</a></h3>
<p><img alt="David Shumbris, first out-of-cast name in Lord of War" src="../assets/images/lord_of_war_fpi_shumbris.jpg" /></p>
<p>David Shumbris, t=418s, 81% confidence. A real face in a dim, low-detail
shot under a train track. A genuine lookalike confusion in poor lighting.</p>
<p><img alt="Ronald Reagan, second out-of-cast name in Lord of War" src="../assets/images/lord_of_war_fpi_reagan_photo.jpg" /></p>
<p>Ronald Reagan, t=1003s, 100% confidence. This is not a lookalike confusion.
The detected face is a photograph of Reagan appearing within the shot, not a
living actor. The detector and matcher both did their job correctly on the
image content in front of them; the error is that a photograph inside the
scene is not the same thing as an actor present in the scene, and the
pipeline has no way to draw that distinction from a face crop alone.</p>
<p><img alt="Lance Reddick, third out-of-cast name in Lord of War" src="../assets/images/lord_of_war_fpi_reddick.jpg" /></p>
<p>Lance Reddick, t=6424s, 78% confidence. A small, distant, low-detail face at
the edge of frame. A marginal, low-confidence lookalike confusion.</p>
<h3 id="lovelace-1-name">Lovelace: 1 name<a class="headerlink" href="#lovelace-1-name" title="Permanent link">&para;</a></h3>
<p><img alt="Chloë Sevigny, out-of-cast name in Lovelace" src="../assets/images/lovelace_fpi_sevigny.jpg" /></p>
<p>Chloë Sevigny, t=2451s, 100% confidence. Two boxes are drawn on the same
face: one correctly labeled Amanda Seyfried, one incorrectly labeled Chloë
Sevigny, both at 100%. A single detection producing two competing high-
confidence identities on the same crop.</p>
<h3 id="scarface-1-name">Scarface: 1 name<a class="headerlink" href="#scarface-1-name" title="Permanent link">&para;</a></h3>
<p><img alt="Kirstie Alley, out-of-cast name in Scarface" src="../assets/images/scarface_fpi_alley.jpg" /></p>
<p>Kirstie Alley, t=2451s, 89% confidence. Al Pacino is correctly identified in
the foreground at 100%; a background face in the same shot is wrongly
labeled Kirstie Alley. (The t=2451s here and the Lovelace Chloë Sevigny case
above landing on the identical second is a genuine coincidence, verified from
each film's raw stream by <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/docs/first_fpi_frames.py"><code>first_fpi_frames.py</code></a>,
not a transcription slip, two unrelated films whose <em>first</em> out-of-cast name
happens to fall at the same timestamp.)</p>
<h3 id="summary-of-the-nine">Summary of the nine<a class="headerlink" href="#summary-of-the-nine" title="Permanent link">&para;</a></h3>
<table>
<thead>
<tr>
<th>film</th>
<th>name</th>
<th>t (s)</th>
<th>confidence</th>
<th>classification</th>
</tr>
</thead>
<tbody>
<tr>
<td>Many Saints of Newark</td>
<td>Germar Terrell Gardner</td>
<td>848</td>
<td>78%</td>
<td>ground-truth gap</td>
</tr>
<tr>
<td>Many Saints of Newark</td>
<td>Archie Yates</td>
<td>2521</td>
<td>78%</td>
<td>lookalike confusion</td>
</tr>
<tr>
<td>Many Saints of Newark</td>
<td>Zooey Deschanel</td>
<td>2819</td>
<td>99%</td>
<td>lookalike confusion</td>
</tr>
<tr>
<td>Many Saints of Newark</td>
<td>Talia Balsam</td>
<td>4551</td>
<td>93%</td>
<td>ground-truth gap</td>
</tr>
<tr>
<td>Lord of War</td>
<td>David Shumbris</td>
<td>418</td>
<td>81%</td>
<td>lookalike confusion</td>
</tr>
<tr>
<td>Lord of War</td>
<td>Ronald Reagan</td>
<td>1003</td>
<td>100%</td>
<td>photo-in-frame</td>
</tr>
<tr>
<td>Lord of War</td>
<td>Lance Reddick</td>
<td>6424</td>
<td>78%</td>
<td>lookalike confusion, marginal</td>
</tr>
<tr>
<td>Lovelace</td>
<td>Chloë Sevigny</td>
<td>2451</td>
<td>100%</td>
<td>lookalike confusion</td>
</tr>
<tr>
<td>Scarface</td>
<td>Kirstie Alley</td>
<td>2451</td>
<td>89%</td>
<td>lookalike confusion</td>
</tr>
</tbody>
</table>
<p>Of nine distinct out-of-cast names across four films, two are ground-truth
gaps, one is a photograph misread as a person, and six are genuine
embedding-space confusions on real detected faces. None trace to extinction
bridging: every one of these nine is a fresh detection on a real face crop
at the second it first appears.</p>
<h2 id="where-lvface-beat-x-ray">Where LVFace beat X-Ray<a class="headerlink" href="#where-lvface-beat-x-ray" title="Permanent link">&para;</a></h2>
<p>Not every orange in these frames is actually wrong.
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>
scores strictly against X-Ray — but X-Ray itself has holes, and the pipeline
found two kinds.</p>
<p>Not every name marked wrong is actually wrong.
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>
scores strictly against X-Ray, and X-Ray has gaps of its own.</p>
<p><img alt="LVFace correctly identifies Germar Terrell Gardner, uncredited by X-Ray" src="../assets/images/germar_beats_xray.jpg" /></p>
<p>Germar Terrell Gardner — a real, clean, high-confidence detection — is counted
as an out-of-cast misID because he doesn't appear in X-Ray's <code>people.csv</code> for
The Many Saints of Newark at all. But Jellyfin's independent cast metadata
<em>does</em> credit him for this exact film (cross-checked via
<code>experiments/manifests/jellyfin_casts.json</code> from the <code>experiment-data</code> artifact
package, a completely separate data source from X-Ray). That's also him in
orange in the frame above — every one of those "errors" is the pipeline being
right about a person X-Ray forgot.</p>
<p>Germar Terrell Gardner, the same name from the table above, does not appear
in X-Ray's <code>people.csv</code> for The Many Saints of Newark. Jellyfin's
independent cast metadata does credit him for this film (cross-checked
against <code>experiments/manifests/jellyfin_casts.json</code> from the
<code>experiment-data</code> artifact package, a data source entirely separate from
X-Ray). Talia Balsam is the same case: confirmed on screen, absent from
X-Ray's cast list for this title.</p>
<p><img alt="Robert Patrick, clearly on screen, scored wrong by a ground-truth gap" src="../assets/images/lovelace_robert_patrick_fpi.jpg" /></p>
<p>And it isn't only uncredited bit-parts. That is <strong>Robert Patrick</strong> — top-billed
in Lovelace, unmistakably on screen, reading his newspaper, identified at
100% scored orange because X-Ray's people-in-scene list for <em>this scene</em>
doesn't include him. The identification is flawless; the ground truth missed
an actor sitting in the middle of the frame.</p>
<p>This doesn't mean every flagged misID is secretly correct — Many Saints'
974-count total is still overwhelmingly extinction bridging at cuts, not
uncredited cameos. But the X-Ray corpus is a convenient, large-scale ground
truth, not a perfect one, and the misID/FPI numbers in these tables carry an
irreducible noise floor from ground-truth gaps in both directions.</p>
<p>This extends past uncredited background actors. This is Robert Patrick,
top-billed in Lovelace, clearly on screen reading a newspaper, identified at
100%. The frame is scored wrong because X-Ray's people-in-scene list for
this specific scene omits him, despite crediting him elsewhere in the film.
The identification is correct; the ground truth is missing an entry.</p>
<p>X-Ray is a large, convenient ground truth. It is not a complete one. The
misID and FPI counts reported throughout this document include some fixed
amount of noise from gaps in X-Ray itself, in both directions.</p>
<h2 id="summary">Summary<a class="headerlink" href="#summary" title="Permanent link">&para;</a></h2>
<p>LVFace is the right default: it wins the model comparison outright, it names
19 of 20 correctly across a hat-heavy funeral crowd, and it recognises a face
on a screen inside the movie. Its error budget decomposes into two understood
mechanisms extinction bridging at hard cuts (a tunable trade, not a bug) and
the face-vs-presence ceiling baked into X-Ray's semantics — plus a nonzero
slice where the pipeline is right and the ground truth is wrong. The held-out
generalization gap (75.3% → 67.4%) is real and should be treated as the honest
expected performance, not the training-set number.</p>
<p>LVFace wins the model comparison on every held-out film. It correctly names
19 of 20 people in a crowded funeral scene and correctly identifies a face
displayed on a screen inside the film. Its errors resolve into two
mechanisms: extinction bridging, which is correct on most footage and fails
specifically at hard cuts into long faceless stretches, and the
face-versus-presence ceiling, where X-Ray credits scene membership for
people whose faces never appear on screen. Of the nine distinct
out-of-cast identifications found across the benchmark, two trace to gaps in
X-Ray's own cast data, one is a photograph misread as a person, and six are
genuine lookalike confusions on real faces. The held-out generalization gap,
75.3% training to 67.4% held-out, is real and should be treated as the
expected operating point, not the training-set figure.</p>
@@ -1141,13 +1426,13 @@ expected performance, not the training-set number.</p>
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--next" aria-label="Next: Model Bake-off &amp; Re-tune (full log)">
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--next" aria-label="Next: Full Experiment Log">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Model Bake-off & Re-tune (full log)
Full Experiment Log
</div>
</div>
<div class="md-footer__button md-icon">
@@ -10,13 +10,13 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/optimizer-experiments/">
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/methodology/">
<link rel="prev" href="../model-bakeoff/">
<link rel="prev" href="..">
<link rel="next" href="../service-conversion/">
<link rel="next" href="../best-model/">
@@ -27,7 +27,7 @@
<title>Optimizer Experiments (prior round) - scene-actor-extraction</title>
<title>How We Score Against X-Ray - scene-actor-extraction</title>
@@ -80,7 +80,7 @@
<div data-md-component="skip">
<a href="#threshold-optimization-against-amazon-x-ray-experiment-log" class="md-skip">
<a href="#how-we-score-against-x-ray" class="md-skip">
Skip to content
</a>
@@ -114,7 +114,7 @@
<div class="md-header__topic" data-md-component="header-topic">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
How We Score Against X-Ray
</span>
</div>
@@ -245,6 +245,27 @@
<li class="md-tabs__item md-tabs__item--active">
<a href="./" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
<li class="md-tabs__item">
<a href="../best-model/" class="md-tabs__link">
@@ -270,28 +291,7 @@
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item md-tabs__item--active">
<a href="./" class="md-tabs__link">
Optimizer Experiments (prior round)
Full Experiment Log
</a>
</li>
@@ -397,6 +397,135 @@
<li class="md-nav__item md-nav__item--active">
<input class="md-nav__toggle md-toggle" type="checkbox" id="__toc">
<label class="md-nav__link md-nav__link--active" for="__toc">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
<span class="md-nav__icon md-icon"></span>
</label>
<a href="./" class="md-nav__link md-nav__link--active">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
<nav class="md-nav md-nav--secondary" aria-label="Table of contents">
<label class="md-nav__title" for="__toc">
<span class="md-nav__icon md-icon"></span>
Table of contents
</label>
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#what-amazon-x-ray-records" class="md-nav__link">
<span class="md-ellipsis">
What Amazon X-Ray records
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#why-an-offscreen-name-can-be-scored-correct" class="md-nav__link">
<span class="md-ellipsis">
Why an offscreen name can be scored correct
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#what-this-resolves-and-what-it-does-not" class="md-nav__link">
<span class="md-ellipsis">
What this resolves and what it does not
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#precision-recall-and-the-misid-weighting" class="md-nav__link">
<span class="md-ellipsis">
Precision, recall, and the misID weighting
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#reproduce" class="md-nav__link">
<span class="md-ellipsis">
Reproduce
</span>
</a>
</li>
</ul>
</nav>
</li>
@@ -409,10 +538,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" >
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="0">
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
@@ -430,8 +559,8 @@
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_2">
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span>
@@ -574,7 +703,7 @@
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
Full Experiment Log
@@ -593,157 +722,6 @@
<li class="md-nav__item md-nav__item--active">
<input class="md-nav__toggle md-toggle" type="checkbox" id="__toc">
<label class="md-nav__link md-nav__link--active" for="__toc">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
</span>
<span class="md-nav__icon md-icon"></span>
</label>
<a href="./" class="md-nav__link md-nav__link--active">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
</span>
</a>
<nav class="md-nav md-nav--secondary" aria-label="Table of contents">
<label class="md-nav__title" for="__toc">
<span class="md-nav__icon md-icon"></span>
Table of contents
</label>
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#tldr-what-changed" class="md-nav__link">
<span class="md-ellipsis">
TL;DR — what changed
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#ground-truth" class="md-nav__link">
<span class="md-ellipsis">
Ground truth
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#the-scoring-metric-evolved-through-review" class="md-nav__link">
<span class="md-ellipsis">
The scoring metric (evolved through review)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#the-gallery-coverage-gap" class="md-nav__link">
<span class="md-ellipsis">
The gallery coverage gap
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#optimizer" class="md-nav__link">
<span class="md-ellipsis">
Optimizer
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#replay-architecture-how-the-sweep-is-cheap" class="md-nav__link">
<span class="md-ellipsis">
Replay architecture (how the sweep is cheap)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#reproduce" class="md-nav__link">
<span class="md-ellipsis">
Reproduce
</span>
</a>
</li>
</ul>
</nav>
</li>
<li class="md-nav__item">
<a href="../service-conversion/" class="md-nav__link">
@@ -792,10 +770,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#tldr-what-changed" class="md-nav__link">
<a href="#what-amazon-x-ray-records" class="md-nav__link">
<span class="md-ellipsis">
TL;DR — what changed
What Amazon X-Ray records
</span>
</a>
@@ -803,10 +781,10 @@
</li>
<li class="md-nav__item">
<a href="#ground-truth" class="md-nav__link">
<a href="#why-an-offscreen-name-can-be-scored-correct" class="md-nav__link">
<span class="md-ellipsis">
Ground truth
Why an offscreen name can be scored correct
</span>
</a>
@@ -814,10 +792,10 @@
</li>
<li class="md-nav__item">
<a href="#the-scoring-metric-evolved-through-review" class="md-nav__link">
<a href="#what-this-resolves-and-what-it-does-not" class="md-nav__link">
<span class="md-ellipsis">
The scoring metric (evolved through review)
What this resolves and what it does not
</span>
</a>
@@ -825,32 +803,10 @@
</li>
<li class="md-nav__item">
<a href="#the-gallery-coverage-gap" class="md-nav__link">
<a href="#precision-recall-and-the-misid-weighting" class="md-nav__link">
<span class="md-ellipsis">
The gallery coverage gap
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#optimizer" class="md-nav__link">
<span class="md-ellipsis">
Optimizer
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#replay-architecture-how-the-sweep-is-cheap" class="md-nav__link">
<span class="md-ellipsis">
Replay architecture (how the sweep is cheap)
Precision, recall, and the misID weighting
</span>
</a>
@@ -885,161 +841,115 @@
<h1 id="threshold-optimization-against-amazon-x-ray-experiment-log">Threshold optimization against Amazon X-Ray — experiment log<a class="headerlink" href="#threshold-optimization-against-amazon-x-ray-experiment-log" title="Permanent link">&para;</a></h1>
<p>Record of the July 2026 work that tuned the pipeline's recognition/tracking defaults
against ground-truth per-scene actor presence, and the tooling built to do it.</p>
<h2 id="tldr-what-changed">TL;DR — what changed<a class="headerlink" href="#tldr-what-changed" title="Permanent link">&para;</a></h2>
<table>
<thead>
<tr>
<th>knob</th>
<th>old default</th>
<th>new default</th>
<th>why</th>
</tr>
</thead>
<tbody>
<tr>
<td><code>prob_threshold</code></td>
<td>0.99</td>
<td><strong>0.76</strong></td>
<td>0.99 was far too strict — halved recall for a fraction of a precision point. DE optimum, tightly converged.</td>
</tr>
<tr>
<td><code>extinction_sec</code></td>
<td>5.0</td>
<td><strong>1.5</strong></td>
<td>Long extinction smears presence into later scenes → FPs. DE converged tightly low.</td>
</tr>
<tr>
<td><code>anneal_sec</code></td>
<td>10.0</td>
<td>10.0 (unchanged)</td>
<td>DE found it <strong>insensitive</strong> (F1 flat ±0.3pp across 326s) — kept the round default.</td>
</tr>
<tr>
<td><code>detector_conf</code></td>
<td>0.5</td>
<td>0.5 (unchanged)</td>
<td>Sweep showed raising it only trades recall for precision at a net F1 loss — near-threshold detections are real faces, not phantoms.</td>
</tr>
</tbody>
</table>
<p>Net effect on the 9-film benchmark (strict per-scene, augmented gallery):
recall <strong>58% → ~72%</strong>, F1 <strong>70% → ~76%</strong>, precision ~85%, at no meaningful precision cost.</p>
<h2 id="ground-truth">Ground truth<a class="headerlink" href="#ground-truth" title="Permanent link">&para;</a></h2>
<p>Public scene-level <strong>Amazon X-Ray</strong> dataset (Zenodo DOI 10.5281/zenodo.17659734,
CC-BY-4.0): per movie, <code>people.csv</code> (name_id/person/character), <code>scenes.csv</code>
(scene/start/end ms), <code>people_in_scenes.csv</code>. Films matched to the library by an
<strong>authoritative Jellyfin ID join</strong> (query <code>/Items?IncludeItemTypes=Movie&amp;Fields=
ProviderIds,Path</code>, join Imdb/Tmdb against X-Ray metadata) — NOT fuzzy title matching,
which collides badly (TV episodes vs same-named films). 9 genuine films with source
video on disk: Benny &amp; Joon, Café Society, Downton Abbey: A New Era, Lord of War,
Lovelace, The Many Saints of Newark, Scarface, Sound of Metal, Valerian.</p>
<h2 id="the-scoring-metric-evolved-through-review">The scoring metric (evolved through review)<a class="headerlink" href="#the-scoring-metric-evolved-through-review" title="Permanent link">&para;</a></h2>
<p>Comparison unit is the <strong>X-Ray scene</strong>, not sampled timepoints. For each scene
<code>[start,end]</code>: predicted set = <strong>union</strong> of actors detected anywhere in the span;
GT set = actors X-Ray lists for that scene. Per scene TP/FP/FN, then:</p>
<ul>
<li><strong>Precision: STRICT.</strong> Any predicted actor not in the scene's X-Ray set is an FP,
<em>including out-of-cast confusions</em> (no gallery∩cast masking). An earlier
timepoint-sampled, cast-masked metric HID ~570 such FPs across 9 films and let the
optimizer drive <code>prob_threshold</code> to the 0.50 floor — a metric artifact. Counting
them is essential.</li>
<li><strong>Recall: FAIR.</strong> FN counts only X-Ray cast members <strong>who are in the gallery</strong>. 67%
of X-Ray cast (261/392) have no gallery reference embedding and can never be
recognised — counting them as misses penalises coverage, not the threshold. Both
<code>recall</code> (fair) and <code>recall_strict</code> (all) are reported.</li>
<li><strong>Aggregation:</strong> per-scene F1 → <strong>duration-weighted average within a movie</strong> (long
scenes count more) → <strong>equal-weight mean across movies</strong> (macro; each film counts
the same regardless of length). This is the DE objective.</li>
</ul>
<p>Implemented in <code>scripts/optimizer/scene_score.py</code> — since <strong>removed</strong> along
with this metric; its per-second successor is
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>
(see the <a href="../model-bakeoff/">bake-off round</a>).</p>
<h2 id="the-gallery-coverage-gap">The gallery coverage gap<a class="headerlink" href="#the-gallery-coverage-gap" title="Permanent link">&para;</a></h2>
<p>Diagnosing low recall: only <strong>131 of 392</strong> X-Ray cast were in the gallery (33%). Every
in-gallery actor HAD embeddings (gallery well-formed) — the gap was pure coverage.
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/fetch_missing_actors.py"><code>scripts/optimizer/fetch_missing_actors.py</code></a>
recovers missing actors:
<code>nm-id → TMDB /find external_ids → /person/{id}/images → download → embed (sae_embed)</code>,
with a <code>--wikidata</code> fallback (P345→P18 Commons photo).</p>
<ul>
<li><strong>TMDB recovered 143/261</strong> (55%). 0 face-detection failures; the rest had no TMDB
person (60) or no profile photo (58). Coverage 33% → <strong>70%</strong>.</li>
<li><strong>Wikidata fallback: 0/118</strong> of the TMDB failures — only 4 even had a Commons photo,
none yielded a detectable face. → <strong>TheTVDB not worth pursuing</strong>: these remaining
actors are obscure enough that no image source covers them, AND (see below) most are
off-camera anyway.</li>
</ul>
<p><strong>Coverage vs detectability.</strong> Adding references lifted recall (58→68% at fixed config)
but modestly. Per-film drill-down (Lord of War: 12 actors recovered, only 1 had a
detectable on-camera face) showed most missing cast are a <strong>detectability gap</strong> — X-Ray
credits them as cast-in-scene (incl. off-camera/background), but their face never
appears clearly for the pipeline to detect. This is a fundamental ceiling of a
face-recognition pipeline vs X-Ray's presence semantics, not a fixable gap.</p>
<h2 id="optimizer">Optimizer<a class="headerlink" href="#optimizer" title="Permanent link">&para;</a></h2>
<p><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/optimize.py"><code>scripts/optimizer/optimize.py</code></a>
— scipy <code>differential_evolution</code> over the knob space,
each candidate = full replay of all films through the <strong>real</strong> C++ nodes (see the
KPN replay architecture below) scored by the metric above. Global objective (one
config for all films, not per-film).</p>
<p><strong>Convergence stability (augmented gallery, 233 evals):</strong></p>
<table>
<thead>
<tr>
<th>knob</th>
<th>top-20 range</th>
<th>verdict</th>
</tr>
</thead>
<tbody>
<tr>
<td><code>prob_threshold</code></td>
<td>0.690.83 (σ 0.05)</td>
<td>TIGHT — trust 0.76</td>
</tr>
<tr>
<td><code>extinction_sec</code></td>
<td>1.02.2 (σ 0.33)</td>
<td>TIGHT — trust 1.5</td>
</tr>
<tr>
<td><code>anneal_sec</code></td>
<td>3.126.3 (σ 6.4)</td>
<td>LOOSE — insensitive, not hard-coded</td>
</tr>
</tbody>
</table>
<p>F1 varied only 0.3pp across the top-20 → objective is flat near the optimum, so only
the tightly-converged knobs were adopted as defaults.</p>
<h2 id="replay-architecture-how-the-sweep-is-cheap">Replay architecture (how the sweep is cheap)<a class="headerlink" href="#replay-architecture-how-the-sweep-is-cheap" title="Permanent link">&para;</a></h2>
<p>The optimizer never re-decodes video. <code>scene_analyze --dump-embeddings out.h5</code> runs the
expensive half once (decode→detect→align→embed) and dumps per-frame face embeddings
+ metadata to HDF5 (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/SCHEMA.md"><code>scripts/optimizer/SCHEMA.md</code></a>).
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/replay.py"><code>scripts/optimizer/replay.py</code></a> then
replays that dump through the <strong>real</strong> C++ <code>face_tracker → identity_matcher →
scene_tracker</code> assembled in a Python KPN network (<code>sae_kpn</code> nanobind module), varying
Config knobs freely — no GPU embedding, no decode. Verified BYTE-EXACT against
<code>scene_analyze</code>'s own output. The dumps are gallery-independent, so testing the
augmented gallery needed no re-dump. <code>detector_conf</code> is replayable UPWARD only (the
dump floor is 0.5).</p>
<h2 id="reproduce">Reproduce<a class="headerlink" href="#reproduce" title="Permanent link">&para;</a></h2>
<div class="language-bash highlight"><pre><span></span><code><span id="__span-0-1"><a id="__codelineno-0-1" name="__codelineno-0-1" href="#__codelineno-0-1"></a><span class="c1"># 1. dump (once per film, needs video)</span>
</span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a>scene_analyze<span class="w"> </span>--movie<span class="w"> </span>&lt;f&gt;<span class="w"> </span>--gallery<span class="w"> </span>gallery.json<span class="w"> </span>--dump-embeddings<span class="w"> </span>dump.h5<span class="w"> </span>--fps<span class="w"> </span><span class="m">1</span>
</span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a><span class="c1"># 2. build films manifest by Jellyfin ID join (see scripts/optimizer notes)</span>
</span><span id="__span-0-4"><a id="__codelineno-0-4" name="__codelineno-0-4" href="#__codelineno-0-4"></a><span class="c1"># 3. optimize</span>
</span><span id="__span-0-5"><a id="__codelineno-0-5" name="__codelineno-0-5" href="#__codelineno-0-5"></a>python<span class="w"> </span>scripts/optimizer/optimize.py<span class="w"> </span>--manifest<span class="w"> </span>films.json<span class="w"> </span>--gallery<span class="w"> </span>gallery.json<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-6"><a id="__codelineno-0-6" name="__codelineno-0-6" href="#__codelineno-0-6"></a><span class="w"> </span>--params<span class="w"> </span>prob_threshold:0.5:0.999<span class="w"> </span>anneal_sec:1:30<span class="w"> </span>extinction_sec:1:15<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-7"><a id="__codelineno-0-7" name="__codelineno-0-7" href="#__codelineno-0-7"></a><span class="w"> </span>--popsize<span class="w"> </span><span class="m">8</span><span class="w"> </span>--maxiter<span class="w"> </span><span class="m">20</span><span class="w"> </span>--trajectory<span class="w"> </span>traj.jsonl<span class="w"> </span>--out<span class="w"> </span>opt.json
</span><span id="__span-0-8"><a id="__codelineno-0-8" name="__codelineno-0-8" href="#__codelineno-0-8"></a><span class="c1"># 4. score a fixed config / validate on a held-out set</span>
</span><span id="__span-0-9"><a id="__codelineno-0-9" name="__codelineno-0-9" href="#__codelineno-0-9"></a><span class="c1"># (historical: score_config.py and scene_score.py were removed with the</span>
</span><span id="__span-0-10"><a id="__codelineno-0-10" name="__codelineno-0-10" href="#__codelineno-0-10"></a><span class="c1"># scene-union metric — use scripts/optimizer/second_score.py, per-second)</span>
</span><span id="__span-0-11"><a id="__codelineno-0-11" name="__codelineno-0-11" href="#__codelineno-0-11"></a>python<span class="w"> </span>scripts/optimizer/second_score.py<span class="w"> </span>--help
<h1 id="how-we-score-against-x-ray">How we score against X-Ray<a class="headerlink" href="#how-we-score-against-x-ray" title="Permanent link">&para;</a></h1>
<p>Every number in this report, every F1 and misID count, comes from one
comparison. The comparison has a mismatch at its core that shapes nearly
every finding in this report: the ground truth is scene-level, the
pipeline's output is per-second, and the two do not mean the same thing.
This page documents that comparison once, so the findings pages can rely on
it without re-explaining it.</p>
<h2 id="what-amazon-x-ray-records">What Amazon X-Ray records<a class="headerlink" href="#what-amazon-x-ray-records" title="Permanent link">&para;</a></h2>
<p>X-Ray ships three tables per film: <code>scenes.csv</code> (a list of <code>[start, end]</code>
timespans), <code>people_in_scenes.csv</code> (which actors are credited in each
scene), and <code>people.csv</code> (actor identities). There is no per-frame or
per-second annotation anywhere in X-Ray. A scene might run 45 seconds, and
X-Ray records one cast list for the entire span, not "on screen from
second 12 to second 30."</p>
<p>To compare this against per-second predictions, <code>second_score.py</code> expands
every scene into per-second ground truth by copying the whole scene's cast
list onto every second inside it:</p>
<div class="language-python highlight"><pre><span></span><code><span id="__span-0-1"><a id="__codelineno-0-1" name="__codelineno-0-1" href="#__codelineno-0-1"></a><span class="k">for</span> <span class="n">sn</span><span class="p">,</span> <span class="p">(</span><span class="n">t0</span><span class="p">,</span> <span class="n">t1</span><span class="p">)</span> <span class="ow">in</span> <span class="n">spans</span><span class="o">.</span><span class="n">items</span><span class="p">():</span>
</span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a> <span class="n">cast</span> <span class="o">=</span> <span class="n">scene_cast</span><span class="o">.</span><span class="n">get</span><span class="p">(</span><span class="n">sn</span><span class="p">,</span> <span class="p">[])</span>
</span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a> <span class="k">for</span> <span class="n">t</span> <span class="ow">in</span> <span class="nb">range</span><span class="p">(</span><span class="nb">int</span><span class="p">(</span><span class="n">t0</span><span class="p">),</span> <span class="nb">int</span><span class="p">(</span><span class="n">t1</span><span class="p">)):</span>
</span><span id="__span-0-4"><a id="__codelineno-0-4" name="__codelineno-0-4" href="#__codelineno-0-4"></a> <span class="n">timeline</span><span class="p">[</span><span class="n">t</span><span class="p">]</span> <span class="o">=</span> <span class="n">cast</span>
</span></code></pre></div>
<p>Superseded by the <a href="../model-bakeoff/">model bake-off + re-tune</a>, which
replaced this round's scene-union metric with per-second scoring.</p>
<p>That is the entire mechanism. If X-Ray credits five actors to a 30-second
scene, all five count as ground truth present for all 30 seconds, including
seconds where only one of them is on screen. This is not a simplification
introduced by the pipeline; it is the only reading of X-Ray's data that is
possible, because X-Ray itself does not record anything finer-grained.</p>
<h2 id="why-an-offscreen-name-can-be-scored-correct">Why an offscreen name can be scored correct<a class="headerlink" href="#why-an-offscreen-name-can-be-scored-correct" title="Permanent link">&para;</a></h2>
<p>A name listed under Offscreen with a correct (green) label is not the
pipeline guessing or padding its score. It is the pipeline correctly
answering the question X-Ray actually asks: is this actor part of this
scene. It answers that question using a presence window (<code>[start, end]</code>,
held open across cuts by <code>anneal_sec</code> and <code>extinction_sec</code>), which matches
X-Ray's scene-level semantics more closely than a raw per-frame detection
would.</p>
<p>A system that only reported "this actor is visible in this exact frame"
would score worse against X-Ray's scene-level ground truth, producing a
false negative every time the camera cuts away from a character who is
still present in the scene. Not because it is wrong about the world, but
because it would be answering a stricter, different question than the one
X-Ray's data supports. The presence-window design exists specifically to
answer X-Ray's actual question.</p>
<h2 id="what-this-resolves-and-what-it-does-not">What this resolves and what it does not<a class="headerlink" href="#what-this-resolves-and-what-it-does-not" title="Permanent link">&para;</a></h2>
<p>This resolves the semantic mismatch between a scene and an instant. It does
not resolve two other limitations, both discussed in the
<a href="../lvface-deep-dive/">LVFace deep dive</a>.</p>
<p><strong>The face-vs-presence ceiling.</strong> X-Ray credits scene membership regardless
of whether a face is ever visible: background crew, characters shot from
behind, voice-only presence. No amount of bridging recovers a face that
never appears on screen. This is a hard ceiling on recall, not a defect.</p>
<p><strong>Extinction bridging can overshoot.</strong> The same presence-window mechanism
that correctly answers "still in this scene" during a normal cut can also
bridge across a scene boundary it has no way to detect. A hard cut into a
different scene with no faces, such as closing credits, carries the
previous scene's identities forward until the window expires. This is the
mechanism behind Downton Abbey's recall collapse, documented in the deep
dive.</p>
<h2 id="precision-recall-and-the-misid-weighting">Precision, recall, and the misID weighting<a class="headerlink" href="#precision-recall-and-the-misid-weighting" title="Permanent link">&para;</a></h2>
<p>Per sampled second <code>t</code>:</p>
<p><strong>TPI</strong> (true positive instances): actors both X-Ray and the pipeline agree
are present.</p>
<p><strong>FPI</strong> (false positive instances): actors the pipeline reports that are
not in X-Ray's cast for this second. Split into two categories:</p>
<ul>
<li><strong>FPI_incast</strong>: the actor is in the film's cast, just not credited to
this particular scene. A timing or boundary slip.</li>
<li><strong>FPI_misid</strong>: the actor is not in the film's cast at all. A genuine
wrong-identity error, weighted 10x in the precision objective, because
naming someone who is not even in the film is a categorically worse
error than a few seconds of scene-boundary slop.</li>
</ul>
<div class="admonition note">
<p class="admonition-title">Every headline <code>P</code> and <code>F1</code> is misID-weighted</p>
<p>The precision reported throughout this report, and therefore the F1
derived from it, puts each <code>FPI_misid</code> into the denominator <strong>10 times</strong>
(<code>precision = TPI / (TPI + FPI_incast + 10·FPI_misid)</code>,
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/optimizer/second_score.py"><code>second_score.py</code></a>).
This is deliberate: the whole point is to punish naming an out-of-film
actor far harder than a scene-boundary slip. But it means the <code>P</code> column
is not raw precision, and a misID-heavy film's <code>P</code> is depressed
super-linearly. <code>second_score.py</code> also emits an unweighted <code>precision_raw</code>
(always ≥ the weighted <code>P</code>); where the gap matters, The Many Saints of
Newark, weighted <code>P</code> 54.7% vs. raw 78.4%, the <a href="../lvface-deep-dive/">LVFace deep dive</a>
reports both. When comparing <code>P</code> across films, remember you are comparing a
quantity that penalizes misIDs, not just a hit rate.</p>
</div>
<p><strong>FN</strong> (false negatives): actors X-Ray lists that the pipeline never
reports, counted only for actors who have a gallery reference embedding.
Across the 9-film benchmark, coverage of X-Ray's credited cast ranges from
20% to 79% by film (see
<a href="../model-bakeoff/#gallery-coverage-per-film">the full experiment log</a>); an
actor with no reference photo can never be recognized regardless of model
quality, and counting them as a miss would penalize gallery coverage, not
recognition accuracy.</p>
<p>Two further numbers are reported alongside F1:</p>
<p><strong>agreement_rate</strong>: mean per-second Jaccard overlap
(<code>|Pred ∩ GT| / |Pred GT|</code>), partial credit. Naming 2 of 3 present actors
scores 2/3, not 0.</p>
<p><strong>exact_match_rate</strong>: the fraction of sampled seconds where the pipeline's
named set exactly equals X-Ray's, no partial credit. Far harsher, and
dominated by recall, since any single missed actor zeroes that second.</p>
<h2 id="reproduce">Reproduce<a class="headerlink" href="#reproduce" title="Permanent link">&para;</a></h2>
<div class="language-bash highlight"><pre><span></span><code><span id="__span-1-1"><a id="__codelineno-1-1" name="__codelineno-1-1" href="#__codelineno-1-1"></a>python3<span class="w"> </span>scripts/optimizer/second_score.py<span class="w"> </span><span class="se">\</span>
</span><span id="__span-1-2"><a id="__codelineno-1-2" name="__codelineno-1-2" href="#__codelineno-1-2"></a><span class="w"> </span>--pred<span class="w"> </span>pred.json<span class="w"> </span>--xray<span class="w"> </span>experiments/xray/.../&lt;xray_dir&gt;<span class="w"> </span><span class="se">\</span>
</span><span id="__span-1-3"><a id="__codelineno-1-3" name="__codelineno-1-3" href="#__codelineno-1-3"></a><span class="w"> </span>--gallery<span class="w"> </span>experiments/galleries/gallery_LVFace-B_Glint360K.h5
</span></code></pre></div>
<p>See also <a href="../model-bakeoff/">the full experiment log</a> for how <code>pred.json</code> is
produced, and the <a href="../lvface-deep-dive/">LVFace deep dive</a> for what these
mechanisms look like frame by frame.</p>
@@ -1075,7 +985,7 @@ replaced this round's scene-union metric with per-second scoring.</p>
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Model Bake-off &amp; Re-tune (full log)">
<a href=".." class="md-footer__link md-footer__link--prev" aria-label="Previous: Home">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
@@ -1085,20 +995,20 @@ replaced this round's scene-union metric with per-second scoring.</p>
Previous
</span>
<div class="md-ellipsis">
Model Bake-off & Re-tune (full log)
Home
</div>
</div>
</a>
<a href="../service-conversion/" class="md-footer__link md-footer__link--next" aria-label="Next: Service Conversion (proposal)">
<a href="../best-model/" class="md-footer__link md-footer__link--next" aria-label="Next: Best Model">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Service Conversion (proposal)
Best Model
</div>
</div>
<div class="md-footer__button md-icon">
+543 -702
View File
File diff suppressed because it is too large Load Diff
+164 -154
View File
@@ -80,7 +80,7 @@
<div data-md-component="skip">
<a href="#pose-expansion-does-learning-new-poses-mid-film-help" class="md-skip">
<a href="#pose-expansion-does-promoting-new-poses-mid-film-help" class="md-skip">
Skip to content
</a>
@@ -243,6 +243,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -272,26 +291,7 @@
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
Full Experiment Log
</a>
</li>
@@ -395,6 +395,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -414,10 +441,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked>
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="">
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="">
@@ -435,8 +462,8 @@
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2">
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span>
@@ -569,10 +596,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#the-training-set-signal" class="md-nav__link">
<a href="#training-set-signal" class="md-nav__link">
<span class="md-ellipsis">
The training-set signal
Training-set signal
</span>
</a>
@@ -580,10 +607,10 @@
</li>
<li class="md-nav__item">
<a href="#held-out-test-does-it-reproduce" class="md-nav__link">
<a href="#held-out-test" class="md-nav__link">
<span class="md-ellipsis">
Held-out test: does it reproduce?
Held-out test
</span>
</a>
@@ -591,10 +618,10 @@
</li>
<li class="md-nav__item">
<a href="#two-bugs-this-required-catching-this-sections-own-methodology" class="md-nav__link">
<a href="#two-methodology-bugs-caught-during-this-check" class="md-nav__link">
<span class="md-ellipsis">
Two bugs this required catching (this section's own methodology)
Two methodology bugs caught during this check
</span>
</a>
@@ -602,10 +629,10 @@
</li>
<li class="md-nav__item">
<a href="#what-this-means" class="md-nav__link">
<a href="#conclusion" class="md-nav__link">
<span class="md-ellipsis">
What this means
Conclusion
</span>
</a>
@@ -670,34 +697,7 @@
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
Full Experiment Log
@@ -764,10 +764,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#the-training-set-signal" class="md-nav__link">
<a href="#training-set-signal" class="md-nav__link">
<span class="md-ellipsis">
The training-set signal
Training-set signal
</span>
</a>
@@ -775,10 +775,10 @@
</li>
<li class="md-nav__item">
<a href="#held-out-test-does-it-reproduce" class="md-nav__link">
<a href="#held-out-test" class="md-nav__link">
<span class="md-ellipsis">
Held-out test: does it reproduce?
Held-out test
</span>
</a>
@@ -786,10 +786,10 @@
</li>
<li class="md-nav__item">
<a href="#two-bugs-this-required-catching-this-sections-own-methodology" class="md-nav__link">
<a href="#two-methodology-bugs-caught-during-this-check" class="md-nav__link">
<span class="md-ellipsis">
Two bugs this required catching (this section's own methodology)
Two methodology bugs caught during this check
</span>
</a>
@@ -797,10 +797,10 @@
</li>
<li class="md-nav__item">
<a href="#what-this-means" class="md-nav__link">
<a href="#conclusion" class="md-nav__link">
<span class="md-ellipsis">
What this means
Conclusion
</span>
</a>
@@ -824,16 +824,20 @@
<h1 id="pose-expansion-does-learning-new-poses-mid-film-help">Pose expansion: does "learning" new poses mid-film help?<a class="headerlink" href="#pose-expansion-does-learning-new-poses-mid-film-help" title="Permanent link">&para;</a></h1>
<p><code>expand_gallery</code> (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/track_gallery.hpp"><code>src/gallery/track_gallery.hpp</code></a>)
promotes a confidently-identified
track's novel-pose reference views into a per-film, in-memory gallery annex — the
idea being that once the pipeline is sure who someone is, a pose it hasn't seen
before (turned head, different lighting) becomes a free extra reference for
recognising that actor again later in the same film, without touching the baked
gallery.</p>
<h2 id="the-training-set-signal">The training-set signal<a class="headerlink" href="#the-training-set-signal" title="Permanent link">&para;</a></h2>
<p>Averaged across all 4 models, on the 4 films used for optimization:</p>
<h1 id="pose-expansion-does-promoting-new-poses-mid-film-help">Pose expansion: does promoting new poses mid-film help?<a class="headerlink" href="#pose-expansion-does-promoting-new-poses-mid-film-help" title="Permanent link">&para;</a></h1>
<p><code>expand_gallery</code>
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/gallery/track_gallery.hpp"><code>src/gallery/track_gallery.hpp</code></a>)
promotes a confidently identified track's novel-pose reference views into a
per-film, in-memory gallery annex. The idea: once the pipeline is confident
about an identity, a pose it has not seen before (turned head, different
lighting) becomes an extra reference for recognizing that actor again later
in the same film, without touching the baked gallery.</p>
<h2 id="training-set-signal">Training-set signal<a class="headerlink" href="#training-set-signal" title="Permanent link">&para;</a></h2>
<p>Averaged across the 3 compared models (r50 excluded), on the 4 films used
for optimization. These are the corrected, full-coverage figures, see the
<a href="../model-bakeoff/#a-scoring-bug-worth-recording-dropped-film-evaluations">dropped-film note</a>
in the experiment log for why an earlier version of this table overstated the
full-mode misID jump (209 → 864) that was itself partly a truncation artifact:</p>
<table>
<thead>
<tr>
@@ -848,54 +852,54 @@ gallery.</p>
<tr>
<td>full</td>
<td>off</td>
<td>71.2%</td>
<td>58.3%</td>
<td>209</td>
<td>70.0%</td>
<td>57.6%</td>
<td>407</td>
</tr>
<tr>
<td>full</td>
<td><strong>on</strong></td>
<td>71.2%</td>
<td>59.7%</td>
<td><strong>864</strong></td>
<td>on</td>
<td>72.1%</td>
<td>61.5%</td>
<td>714</td>
</tr>
<tr>
<td>restricted</td>
<td>off</td>
<td>73.6%</td>
<td>61.3%</td>
<td>194</td>
<td>75.1%</td>
<td>63.9%</td>
<td>179</td>
</tr>
<tr>
<td>restricted</td>
<td><strong>on</strong></td>
<td><strong>75.4%</strong></td>
<td><strong>64.5%</strong></td>
<td>135</td>
<td>on</td>
<td>76.7%</td>
<td>67.2%</td>
<td>120</td>
</tr>
</tbody>
</table>
<p>In <code>restricted</code> mode (matcher's candidate set capped to the film's own credited
cast) expansion looked like a clean win: +1.8pp F1, +3.2pp recall, misID actually
lower. In <code>full</code> mode it looked flat-to-costly: ~0 F1 change, recall +1.4pp, but
misID roughly quadrupled (209 → 864) — see the
<a href="../model-bakeoff/">bake-off experiment log</a> for the per-model breakdown. That's the number that motivated this page: <strong>does turning
expansion on actually change what gets recognised, frame by frame, or is the
aggregate F1 shift something else?</strong></p>
<h2 id="held-out-test-does-it-reproduce">Held-out test: does it reproduce?<a class="headerlink" href="#held-out-test-does-it-reproduce" title="Permanent link">&para;</a></h2>
<p>Same model + same tuned config, <code>expand_gallery</code> toggled on vs. off, nothing else
changed full gallery mode, per-second scoring against X-Ray. This isolates
expansion from every other variable (config, model, threshold) that differs
between the training-set <code>exp</code>/<code>noexp</code> rows above.</p>
<p><strong>LVFace-B Glint360K, all 5 held-out films</strong> (films never seen by the optimizer):</p>
<p>In restricted mode, expansion looks like a clean win: +1.6pp F1, +3.3pp
recall, lower misID. In full mode it looks like a recall-for-misID trade:
+2.1pp F1, +3.9pp recall, but misID rises from 407 to 714. See
<a href="../model-bakeoff/">the full experiment log</a> for the per-model breakdown.
This asymmetry motivated the question below: does turning expansion on
change what gets recognized frame by frame, or is the aggregate F1 shift
coming from something else.</p>
<h2 id="held-out-test">Held-out test<a class="headerlink" href="#held-out-test" title="Permanent link">&para;</a></h2>
<p>Same model, same tuned config, <code>expand_gallery</code> toggled on vs. off, nothing
else changed, full gallery mode, per-second scoring against X-Ray. This
isolates expansion from every other variable that differs between the
training-set rows above.</p>
<p>LVFace-B Glint360K, all 5 held-out films:</p>
<table>
<thead>
<tr>
<th>film</th>
<th>F1 (exp)</th>
<th>F1 (noexp)</th>
<th>TPI Δ</th>
<th>FN Δ</th>
<th>TPI delta</th>
<th>FN delta</th>
</tr>
</thead>
<tbody>
@@ -936,54 +940,60 @@ between the training-set <code>exp</code>/<code>noexp</code> rows above.</p>
</tr>
</tbody>
</table>
<p><strong>ArcFace R18</strong> (Benny &amp; Joon, r18's own tuned config): F1 77.1% for both, TPI/FN
identical, FPI differs by 2 (noise).</p>
<p><strong>Every film, both models tested: F1 within 0.10.2pp, TPI/FN swings in the tens
out of tens of thousands.</strong> That's noise, not a signal — expansion made no
measurable difference to per-second onscreen identification anywhere it was
tested on unseen data.</p>
<h2 id="two-bugs-this-required-catching-this-sections-own-methodology">Two bugs this required catching (this section's own methodology)<a class="headerlink" href="#two-bugs-this-required-catching-this-sections-own-methodology" title="Permanent link">&para;</a></h2>
<p>Getting to the clean table above took two wrong turns, both worth recording
since they're exactly the kind of error that produces a false positive "look,
expansion helped!" finding:</p>
<p>ArcFace R18, Benny &amp; Joon, r18's own tuned config: F1 77.1% for both, TPI
and FN identical, FPI differs by 2.</p>
<p>Every film, both models tested: F1 differs by 0.1-0.2pp, TPI/FN swings are
in the tens out of tens of thousands. This is noise, not a signal.
Expansion made no measurable difference to per-second on-screen
identification on any held-out film tested.</p>
<h2 id="two-methodology-bugs-caught-during-this-check">Two methodology bugs caught during this check<a class="headerlink" href="#two-methodology-bugs-caught-during-this-check" title="Permanent link">&para;</a></h2>
<p>Getting to the table above required catching two wrong turns, both worth
recording because they are exactly the kind of error that produces a false
positive "expansion helped" finding.</p>
<ol>
<li><strong>Timeout truncation.</strong> The first Downton Abbey <code>exp</code> replay was cut off by a
60s subprocess timeout at ~76% through the film (5589 of 7368 expected
seconds) — a genuinely large, silent data loss that showed up as a large,
convincing-looking TPI gap (47938 vs 52032) purely because one run had a
quarter of the film missing. Caught by comparing <code>n_seconds</code> between runs
before trusting any score delta; fixed by re-running with a longer timeout.</li>
<li><strong>Bbox-matching bug.</strong> An early per-second raw-annotation diff matched each
<code>exp</code> detection to the <em>first</em> <code>noexp</code> detection with IoU &gt; 0.5, not the
<em>best</em>-overlapping one. With 3 faces close together in frame, this produced
spurious "disagreements" (e.g. "exp says Aidan Quinn, noexp says Johnny
Depp" at the same seconds) that vanished entirely once the match picked the
true best-IoU candidate — both configs had actually output the exact same
three names at the exact same three boxes.</li>
<li><strong>Timeout truncation.</strong> The first Downton Abbey <code>exp</code> replay was cut off
by a 60-second subprocess timeout at about 76% through the film (5589 of
7368 expected seconds). This silent data loss produced a large,
convincing-looking TPI gap (47938 vs 52032) purely because one run was
missing a quarter of the film. Caught by comparing <code>n_seconds</code> between
runs before trusting any score delta; fixed by re-running with a longer
timeout.</li>
<li><strong>Bbox-matching bug.</strong> An early per-second raw-annotation diff matched
each <code>exp</code> detection to the first <code>noexp</code> detection with IoU above 0.5,
not the best-overlapping one. With 3 faces close together in frame, this
produced spurious disagreements (for example "exp says Aidan Quinn,
noexp says Johnny Depp" at the same second) that vanished once the match
used the best-IoU candidate instead of the first one. Both configs had
actually output the same three names at the same three boxes.</li>
</ol>
<p>Both bugs independently pointed toward "expansion is doing something," and both
were artifacts of the comparison harness, not the pipeline. Worth remembering
when a before/after diff looks dramatic: check that the two runs actually cover
the same seconds, and match entities by best overlap, not first-found.</p>
<h2 id="what-this-means">What this means<a class="headerlink" href="#what-this-means" title="Permanent link">&para;</a></h2>
<p>The training-set aggregate effect (particularly the ~4x misID increase in full
mode) doesn't reproduce on held-out data — at minimum it's far smaller than the
training-set numbers suggested, and plausibly it's sampling variation from only
4 training films rather than a real, generalizable mechanism. This doesn't mean
<code>expand_gallery</code> never does anything (the mechanism is real — see
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/track_gallery.hpp"><code>track_gallery.hpp</code></a>'s
promotion logging: tracks <em>do</em> get confirmed and views <em>do</em>
get promoted into the annex on every film tested), only that <strong>whatever effect
it has on final per-second identification was too small to detect against 5
held-out films</strong> with this scoring method. A cleaner test would need either many
more held-out films or a metric that can see the annex's direct contribution
(e.g. tagging which reference embedding won each match), neither of which this
pass had budget for.</p>
<p><strong>Practical takeaway</strong>: don't treat the training-set <code>exp</code> vs <code>noexp</code> numbers in
the <a href="../model-bakeoff/">bake-off experiment log</a> as proof that expansion
changes real-world behavior
in either direction — on the evidence gathered so far, it doesn't move the
needle enough to see.</p>
<p>Both bugs independently pointed toward "expansion is doing something," and
both were artifacts of the comparison harness, not the pipeline. Before
trusting a dramatic before/after diff, check that both runs cover the same
seconds and that entities are matched by best overlap, not first found.</p>
<h2 id="conclusion">Conclusion<a class="headerlink" href="#conclusion" title="Permanent link">&para;</a></h2>
<p>The training-set aggregate effect, particularly the full-mode misID
increase, does not reproduce on held-out data. At minimum it
is far smaller than the training-set numbers suggested; it may be sampling
variation from only 4 training films rather than a generalizable
mechanism. Note the same <em>class</em> of harness bug appears twice in this
investigation, the timeout truncation in bug #1 above, and the dropped-film
aggregation that inflated the raw training-set misID figures. Both make an
inert config look consequential; both are reasons to distrust a dramatic
training-set delta until it survives on held-out films, which this one did
not. This does not mean <code>expand_gallery</code> never does anything: the
mechanism is real, and
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/gallery/track_gallery.hpp"><code>track_gallery.hpp</code></a>'s
promotion logging confirms tracks get confirmed and views get promoted
into the annex on every film tested. It means whatever effect expansion
has on final per-second identification was too small to detect against 5
held-out films with this scoring method. A cleaner test would need either
more held-out films or a metric that can see the annex's direct
contribution, such as tagging which reference embedding won each match;
neither was in scope for this pass.</p>
<p>Do not treat the training-set exp/noexp numbers in
<a href="../model-bakeoff/">the full experiment log</a> as proof that expansion changes
real-world behavior in either direction. On the evidence gathered so far,
it does not move the needle enough to see.</p>
File diff suppressed because one or more lines are too long
+59 -59
View File
@@ -13,7 +13,7 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/service-conversion/">
<link rel="prev" href="../optimizer-experiments/">
<link rel="prev" href="../model-bakeoff/">
@@ -241,6 +241,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
<li class="md-tabs__item">
@@ -268,26 +287,7 @@
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
Full Experiment Log
</a>
</li>
@@ -393,6 +393,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -407,10 +434,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" >
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="0">
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
@@ -428,8 +455,8 @@
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_2">
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span>
@@ -572,34 +599,7 @@
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
Full Experiment Log
@@ -1035,7 +1035,7 @@ lock-gating, not new pipeline logic.</p>
</tr>
<tr>
<td>Backend selection</td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/CMakeLists.txt"><code>CMakeLists.txt</code></a> (<code>SAE_INFERENCE_BACKEND</code>, <code>SAE_GEMM_BACKEND</code>)</td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/CMakeLists.txt"><code>CMakeLists.txt</code></a> (<code>SAE_INFERENCE_BACKEND</code>, <code>SAE_GEMM_BACKEND</code>)</td>
<td>ORT/TRT + ROCm/CUDA, chosen <strong>at build time</strong></td>
</tr>
<tr>
@@ -1045,7 +1045,7 @@ lock-gating, not new pipeline logic.</p>
</tr>
<tr>
<td>Worker loop</td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a><code>--worker</code></td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a><code>--worker</code></td>
<td>Poll Pending → run <code>scene_analyze</code> → push results</td>
</tr>
<tr>
@@ -1055,12 +1055,12 @@ lock-gating, not new pipeline logic.</p>
</tr>
<tr>
<td>Incremental gallery</td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/make_jellyfin_gallery.py"><code>scripts/make_jellyfin_gallery.py</code></a><code>--merge</code></td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/make_jellyfin_gallery.py"><code>scripts/make_jellyfin_gallery.py</code></a><code>--merge</code></td>
<td>Embeds only cast not already in the gallery</td>
</tr>
<tr>
<td>Secrets loader</td>
<td><code>.env</code> via <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/sae_env.py"><code>scripts/sae_env.py</code></a></td>
<td><code>.env</code> via <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/sae_env.py"><code>scripts/sae_env.py</code></a></td>
<td><code>JELLYFIN_URL</code>, <code>JELLYFIN_API_KEY</code>, <code>TMDB_API_KEY</code></td>
</tr>
</tbody>
@@ -1276,7 +1276,7 @@ same filesystem and GPU as everything else on the box.</p>
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../optimizer-experiments/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Optimizer Experiments (prior round)">
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Full Experiment Log">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
@@ -1286,7 +1286,7 @@ same filesystem and GPU as everything else on the box.</p>
Previous
</span>
<div class="md-ellipsis">
Optimizer Experiments (prior round)
Full Experiment Log
</div>
</div>
</a>
+11 -11
View File
@@ -2,34 +2,34 @@
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/</loc>
<lastmod>2026-07-19</lastmod>
<lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/best-model/</loc>
<lastmod>2026-07-19</lastmod>
<lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/gallery-scope/</loc>
<lastmod>2026-07-19</lastmod>
<lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/lvface-deep-dive/</loc>
<lastmod>2026-07-19</lastmod>
<lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/methodology/</loc>
<lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/model-bakeoff/</loc>
<lastmod>2026-07-19</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/optimizer-experiments/</loc>
<lastmod>2026-07-19</lastmod>
<lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/pose-expansion/</loc>
<lastmod>2026-07-19</lastmod>
<lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/service-conversion/</loc>
<lastmod>2026-07-19</lastmod>
<lastmod>2026-07-21</lastmod>
</url>
</urlset>
BIN
View File
Binary file not shown.