docs: deploy from 0bd2747

This commit is contained in:
2026-07-21 08:56:42 +02:00
parent 36f75ba199
commit 34b1f2c58f
24 changed files with 2066 additions and 1897 deletions
+52 -52
View File
@@ -232,6 +232,25 @@
<li class="md-tabs__item">
<a href="/dtourolle/scene-actor-extraction/methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
<li class="md-tabs__item"> <li class="md-tabs__item">
@@ -259,26 +278,7 @@
Model Bake-off & Re-tune (full log) Full Experiment Log
</a>
</li>
<li class="md-tabs__item">
<a href="/dtourolle/scene-actor-extraction/optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
</a> </a>
</li> </li>
@@ -382,6 +382,33 @@
<li class="md-nav__item">
<a href="/dtourolle/scene-actor-extraction/methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -396,10 +423,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" > <input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="0"> <label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
@@ -417,8 +444,8 @@
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
</label> </label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="false"> <nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_2"> <label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
@@ -561,34 +588,7 @@
<span class="md-ellipsis"> <span class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
</span>
</a>
</li>
<li class="md-nav__item">
<a href="/dtourolle/scene-actor-extraction/optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
Binary file not shown.

Before

Width:  |  Height:  |  Size: 68 KiB

After

Width:  |  Height:  |  Size: 69 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 155 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 176 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 161 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 142 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 174 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 124 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 111 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 134 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 118 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 129 KiB

After

Width:  |  Height:  |  Size: 103 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 40 KiB

+236 -118
View File
@@ -13,7 +13,7 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/best-model/"> <link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/best-model/">
<link rel="prev" href=".."> <link rel="prev" href="../methodology/">
<link rel="next" href="../gallery-scope/"> <link rel="next" href="../gallery-scope/">
@@ -242,6 +242,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -272,26 +291,7 @@
Model Bake-off & Re-tune (full log) Full Experiment Log
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
</a> </a>
</li> </li>
@@ -393,6 +393,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -414,10 +441,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked> <input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex=""> <label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="">
@@ -435,8 +462,8 @@
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
</label> </label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true"> <nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2"> <label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
@@ -524,10 +551,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#second-signal-f1-on-the-actual-benchmark" class="md-nav__link"> <a href="#second-signal-held-out-f1" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Second signal: F1 on the actual benchmark Second signal: held-out F1
</span> </span>
</a> </a>
@@ -535,10 +562,21 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#caveat-model-choice-is-an-operational-change" class="md-nav__link"> <a href="#full-training-matrix-picture" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Caveat: model choice is an operational change Full training-matrix picture
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#operational-note" class="md-nav__link">
<span class="md-ellipsis">
Operational note
</span> </span>
</a> </a>
@@ -659,34 +697,7 @@
<span class="md-ellipsis"> <span class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
@@ -764,10 +775,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#second-signal-f1-on-the-actual-benchmark" class="md-nav__link"> <a href="#second-signal-held-out-f1" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Second signal: F1 on the actual benchmark Second signal: held-out F1
</span> </span>
</a> </a>
@@ -775,10 +786,21 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#caveat-model-choice-is-an-operational-change" class="md-nav__link"> <a href="#full-training-matrix-picture" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Caveat: model choice is an operational change Full training-matrix picture
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#operational-note" class="md-nav__link">
<span class="md-ellipsis">
Operational note
</span> </span>
</a> </a>
@@ -803,31 +825,35 @@
<h1 id="which-embedding-model-is-best">Which embedding model is best?<a class="headerlink" href="#which-embedding-model-is-best" title="Permanent link">&para;</a></h1> <h1 id="which-embedding-model-is-best">Which embedding model is best?<a class="headerlink" href="#which-embedding-model-is-best" title="Permanent link">&para;</a></h1>
<p>Four candidates went into the bake-off: three ArcFace variants (w600k-R50, <p>Three ArcFace variants (w600k-R50, R18, w600k-MBF) and LVFace-B (Glint360K,
R18, w600k-MBF) and LVFace-B (Glint360K), a Vision-Transformer embedder that's 455MB) were compared. r50 is excluded from the training/held-out comparison
a drop-in replacement for ArcFace's <code>[N,3,112,112]</code> input / 512-d output. The below; its gallery has roughly 30% fewer reference images per actor than the
open question: is LVFace (455MB) actually better, or just the biggest?</p> other three on the identical source photos, which confounds a direct score
comparison (see <a href="../model-bakeoff/">the full experiment log</a> for detail). It
remains in the calibration comparison, which does not depend on the gallery
image count.</p>
<h2 id="first-signal-calibration-curves">First signal: calibration curves<a class="headerlink" href="#first-signal-calibration-curves" title="Permanent link">&para;</a></h2> <h2 id="first-signal-calibration-curves">First signal: calibration curves<a class="headerlink" href="#first-signal-calibration-curves" title="Permanent link">&para;</a></h2>
<p>Each gallery carries a fitted Platt sigmoid <code>P(match | cosine similarity) = <p>Each gallery carries a fitted Platt sigmoid <code>P(match | cosine similarity) =
σ(a·sim + b)</code>, embedded directly in the gallery's HDF5 file σ(a·sim + b)</code>, stored directly in the gallery HDF5
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/gallery_calibration.hpp"><code>src/gallery/gallery_calibration.hpp</code></a>). This is a property of the embedding (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/gallery/gallery_calibration.hpp"><code>src/gallery/gallery_calibration.hpp</code></a>).
space alone computed from intra/inter-actor reference-image pairs, no This is a property of the embedding space alone, computed from intra- and
tracking or scene logic involved — so it's a clean first read on discriminative inter-actor reference-image pairs with no tracking or scene logic involved,
power before running a single benchmark.</p> so it is a clean first read on discriminative power before running a
benchmark.</p>
<p><img alt="Calibrated P(match|similarity) for all four models" src="../assets/images/calibration_curves.png" /></p> <p><img alt="Calibrated P(match|similarity) for all four models" src="../assets/images/calibration_curves.png" /></p>
<table> <table>
<thead> <thead>
<tr> <tr>
<th>model</th> <th>model</th>
<th><code>a</code> (steepness)</th> <th>a (steepness)</th>
<th>boundary at P=0.5</th> <th>boundary at P=0.5</th>
</tr> </tr>
</thead> </thead>
<tbody> <tbody>
<tr> <tr>
<td><strong>LVFace-B Glint360K</strong></td> <td>LVFace-B Glint360K</td>
<td><strong>17.7</strong></td> <td>17.7</td>
<td><strong>sim 0.228</strong></td> <td>sim 0.228</td>
</tr> </tr>
<tr> <tr>
<td>ArcFace w600k-MBF</td> <td>ArcFace w600k-MBF</td>
@@ -846,13 +872,123 @@ power before running a single benchmark.</p>
</tr> </tr>
</tbody> </tbody>
</table> </table>
<p>LVFace has both the steepest transition and the lowest decision boundary — it <p>LVFace has both the steepest transition and the lowest decision boundary,
separates same-actor from different-actor reference pairs more confidently, at separating same-actor from different-actor reference pairs more confidently
a <em>lower</em> similarity threshold, than any ArcFace variant. That's a genuine at a lower similarity than any ArcFace variant.</p>
head start before the tracking/scoring pipeline is even involved.</p> <h2 id="second-signal-held-out-f1">Second signal: held-out F1<a class="headerlink" href="#second-signal-held-out-f1" title="Permanent link">&para;</a></h2>
<h2 id="second-signal-f1-on-the-actual-benchmark">Second signal: F1 on the actual benchmark<a class="headerlink" href="#second-signal-f1-on-the-actual-benchmark" title="Permanent link">&para;</a></h2> <p>Each model's own tuned <code>full_exp</code> config, replayed against the 5 films the
<p>Best full-gallery (no cast-restriction) result per model, from the 16-combo optimizer never saw and scored the same way:</p>
bake-off matrix (<a href="../model-bakeoff/">full experiment log</a>):</p> <table>
<thead>
<tr>
<th>film</th>
<th>LVFace F1</th>
<th>mbf F1</th>
<th>r18 F1</th>
</tr>
</thead>
<tbody>
<tr>
<td>Benny &amp; Joon</td>
<td>83.0%</td>
<td>78.5%</td>
<td>77.1%</td>
</tr>
<tr>
<td>Lovelace</td>
<td>77.5%</td>
<td>73.7%</td>
<td>72.2%</td>
</tr>
<tr>
<td>Valerian and the City of a Thousand Planets</td>
<td>74.1%</td>
<td>70.2%</td>
<td>71.0%</td>
</tr>
<tr>
<td>Downton Abbey: A New Era</td>
<td>56.2%</td>
<td>55.0%</td>
<td>53.0%</td>
</tr>
<tr>
<td>The Many Saints of Newark</td>
<td>46.3%</td>
<td>44.5%</td>
<td>42.1%</td>
</tr>
<tr>
<td><strong>macro average</strong></td>
<td><strong>67.4%</strong></td>
<td><strong>64.4%</strong></td>
<td><strong>63.1%</strong></td>
</tr>
</tbody>
</table>
<p>LVFace scores highest on all 5 held-out films; the ranking never flips
between models. Total misID count across the 5 films: LVFace 1032, mbf
2197, r18 1224. LVFace has less than half mbf's misID total and still
scores higher on every film.</p>
<p>Held-out results are stronger evidence than training results, because
training numbers can reflect what the optimizer was tuned to fit rather
than general performance. On training data, the ordering is not as clean:</p>
<table>
<thead>
<tr>
<th>film</th>
<th>LVFace F1</th>
<th>mbf F1</th>
<th>r18 F1</th>
<th>best</th>
</tr>
</thead>
<tbody>
<tr>
<td>Café Society</td>
<td>68.1%</td>
<td>62.2%</td>
<td>60.1%</td>
<td>LVFace</td>
</tr>
<tr>
<td>Lord of War</td>
<td>75.6%</td>
<td>77.2%</td>
<td>75.6%</td>
<td>mbf</td>
</tr>
<tr>
<td>Scarface</td>
<td>71.5%</td>
<td>68.6%</td>
<td>64.1%</td>
<td>LVFace</td>
</tr>
<tr>
<td>Sound of Metal</td>
<td>78.8%</td>
<td>76.5%</td>
<td>71.6%</td>
<td>LVFace</td>
</tr>
</tbody>
</table>
<p>mbf beats LVFace on Lord of War (77.2% vs 75.6%), the only film in either
table where LVFace does not score highest. LVFace's training-set macro
average (75.3%, see <a href="../model-bakeoff/">the full experiment log</a>) is not a
uniform win across every film it contributes to; the held-out result, where
LVFace wins all 5 films outright, is the stronger claim.</p>
<p>This reverses an earlier, superseded benchmarking pass that used a
scene-union metric and found the three models statistically
indistinguishable (around 85% each), concluding LVFace was not worth its
size. That metric masked out-of-cast false positives behind a
gallery-intersect-cast recall filter; the per-second metric used here does
not.</p>
<h2 id="full-training-matrix-picture">Full training-matrix picture<a class="headerlink" href="#full-training-matrix-picture" title="Permanent link">&para;</a></h2>
<p><img alt="All 12 combos ranked by training-set F1" src="../assets/images/rep4_matrix_f1.png" /></p>
<p>Best full-gallery combo per model (all three are <code>full_exp</code>), from the
training matrix in <a href="../model-bakeoff/">the full experiment log</a>:</p>
<table> <table>
<thead> <thead>
<tr> <tr>
@@ -865,18 +1001,18 @@ bake-off matrix (<a href="../model-bakeoff/">full experiment log</a>):</p>
</thead> </thead>
<tbody> <tbody>
<tr> <tr>
<td><strong>LVFace-B Glint360K</strong></td> <td>LVFace-B Glint360K</td>
<td><strong>75.3%</strong></td> <td>75.3%</td>
<td>89.7%</td> <td>89.7%</td>
<td><strong>65.4%</strong></td> <td>65.4%</td>
<td>232</td> <td>232</td>
</tr> </tr>
<tr> <tr>
<td>ArcFace w600k-MBF</td> <td>ArcFace w600k-MBF</td>
<td>74.2%</td> <td>72.0%</td>
<td>87.4%</td> <td>87.7%</td>
<td>64.4%</td> <td>61.4%</td>
<td>57</td> <td>240</td>
</tr> </tr>
<tr> <tr>
<td>ArcFace R18</td> <td>ArcFace R18</td>
@@ -885,36 +1021,18 @@ bake-off matrix (<a href="../model-bakeoff/">full experiment log</a>):</p>
<td>57.7%</td> <td>57.7%</td>
<td>242</td> <td>242</td>
</tr> </tr>
<tr>
<td>ArcFace w600k-R50</td>
<td>68.5%</td>
<td>94.0%</td>
<td>54.1%</td>
<td>150</td>
</tr>
</tbody> </tbody>
</table> </table>
<p>The full 16-combo picture makes the model ordering visible at a glance — LVFace <p>LVFace leads within both the restricted and full gallery modes, visible
(yellow) tops both the restricted and full columns, and R18 (green) props up directly in the chart above without reading the table. The three models'
the bottom of the full-gallery ranking:</p> misID counts on the full gallery are nearly identical (232/240/242); LVFace's
<p><img alt="All 16 bake-off combos ranked by training-set F1" src="../assets/images/rep4_matrix_f1.png" /></p> lead here is a precision-and-recall lead, not a misID one.</p>
<p>LVFace wins outright, with the highest recall of any full-mode combo. This <h2 id="operational-note">Operational note<a class="headerlink" href="#operational-note" title="Permanent link">&para;</a></h2>
reverses an earlier conclusion from a prior (superseded) benchmarking pass <p>Switching the default embedder is not a config change alone; the gallery
using a scene-union metric, which found the three models statistically is model-specific, since embeddings from different models are not
indistinguishable (~85% each) and concluded LVFace wasn't worth its size — that comparable. Any existing gallery built against a different model must be
metric hid out-of-cast false positives behind a gallery∩cast recall mask (see rebuilt from source images before the new default takes effect.
<a href="../optimizer-experiments/">the prior optimizer round</a>); the per-second metric <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/optimizer/reembed_gallery.py"><code>scripts/optimizer/reembed_gallery.py</code></a>
used here does not.</p>
<p>Held-out validation (5 films never seen by the optimizer) confirms LVFace's
lead holds up out of sample — see the
<a href="../lvface-deep-dive/">LVFace deep dive</a> for the full breakdown, including
where it fails.</p>
<h2 id="caveat-model-choice-is-an-operational-change">Caveat: model choice is an operational change<a class="headerlink" href="#caveat-model-choice-is-an-operational-change" title="Permanent link">&para;</a></h2>
<p>Switching the default embedder isn't just flipping a config value — the
gallery itself is model-specific (embeddings from different models aren't
comparable), so any existing gallery built against ArcFace w600k-R50 needs to
be rebuilt from source images against LVFace before the new default takes
effect. <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/reembed_gallery.py"><code>scripts/optimizer/reembed_gallery.py</code></a>
does this from a reference gallery's cached source images without does this from a reference gallery's cached source images without
re-downloading anything.</p> re-downloading anything.</p>
@@ -952,7 +1070,7 @@ re-downloading anything.</p>
<nav class="md-footer__inner md-grid" aria-label="Footer" > <nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href=".." class="md-footer__link md-footer__link--prev" aria-label="Previous: Home"> <a href="../methodology/" class="md-footer__link md-footer__link--prev" aria-label="Previous: How We Score Against X-Ray">
<div class="md-footer__button md-icon"> <div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg> <svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
@@ -962,7 +1080,7 @@ re-downloading anything.</p>
Previous Previous
</span> </span>
<div class="md-ellipsis"> <div class="md-ellipsis">
Home How We Score Against X-Ray
</div> </div>
</div> </div>
</a> </a>
+121 -118
View File
@@ -80,7 +80,7 @@
<div data-md-component="skip"> <div data-md-component="skip">
<a href="#whole-gallery-vs-limited-cast-restricted-gallery" class="md-skip"> <a href="#whole-gallery-vs-cast-restricted-gallery" class="md-skip">
Skip to content Skip to content
</a> </a>
@@ -242,6 +242,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -272,26 +291,7 @@
Model Bake-off & Re-tune (full log) Full Experiment Log
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
</a> </a>
</li> </li>
@@ -393,6 +393,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -414,10 +441,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked> <input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex=""> <label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="">
@@ -435,8 +462,8 @@
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
</label> </label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true"> <nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2"> <label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
@@ -541,10 +568,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#the-result" class="md-nav__link"> <a href="#result" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
The result Result
</span> </span>
</a> </a>
@@ -552,10 +579,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#why-this-isnt-the-shipped-default" class="md-nav__link"> <a href="#why-this-is-not-the-shipped-default" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Why this isn't the shipped default Why this is not the shipped default
</span> </span>
</a> </a>
@@ -648,34 +675,7 @@
<span class="md-ellipsis"> <span class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
@@ -742,10 +742,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#the-result" class="md-nav__link"> <a href="#result" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
The result Result
</span> </span>
</a> </a>
@@ -753,10 +753,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#why-this-isnt-the-shipped-default" class="md-nav__link"> <a href="#why-this-is-not-the-shipped-default" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Why this isn't the shipped default Why this is not the shipped default
</span> </span>
</a> </a>
@@ -780,14 +780,15 @@
<h1 id="whole-gallery-vs-limited-cast-restricted-gallery">Whole gallery vs. limited (cast-restricted) gallery<a class="headerlink" href="#whole-gallery-vs-limited-cast-restricted-gallery" title="Permanent link">&para;</a></h1> <h1 id="whole-gallery-vs-cast-restricted-gallery">Whole gallery vs. cast-restricted gallery<a class="headerlink" href="#whole-gallery-vs-cast-restricted-gallery" title="Permanent link">&para;</a></h1>
<p>Two ways to run the matcher: <strong>full</strong> scores every detected face against the <p>Two ways to run the matcher. Full mode scores every detected face against
entire library gallery (2418 actors across the 9-film benchmark set); <strong>restricted</strong> the entire 2418-actor gallery. Restricted mode pre-filters each film's
pre-filters each film's gallery down to just its Jellyfin-credited cast (typically gallery down to just its Jellyfin-credited cast (typically around 15
~15 top-billed actors) before the matcher ever runs.</p> top-billed actors) before the matcher runs.</p>
<h2 id="the-result">The result<a class="headerlink" href="#the-result" title="Permanent link">&para;</a></h2> <h2 id="result">Result<a class="headerlink" href="#result" title="Permanent link">&para;</a></h2>
<p>Averaged across all 4 models and both expansion settings, on the 4 bake-off training <p>Averaged across the 3 compared models (r50 excluded, see
films:</p> <a href="../model-bakeoff/">the full experiment log</a>) and both expansion settings, on
the 4 training films:</p>
<table> <table>
<thead> <thead>
<tr> <tr>
@@ -795,68 +796,70 @@ films:</p>
<th>F1</th> <th>F1</th>
<th>P</th> <th>P</th>
<th>R</th> <th>R</th>
<th>total misID (8 evals)</th> <th>total misID</th>
</tr> </tr>
</thead> </thead>
<tbody> <tbody>
<tr> <tr>
<td>full</td> <td>full</td>
<td>71.2%</td> <td>71.1%</td>
<td>91.1%</td> <td>89.6%</td>
<td>59.0%</td> <td>59.6%</td>
<td>1073</td> <td>1121</td>
</tr> </tr>
<tr> <tr>
<td><strong>restricted</strong></td> <td>restricted</td>
<td><strong>74.5%</strong></td> <td>75.9%</td>
<td>92.2%</td> <td>90.4%</td>
<td><strong>62.9%</strong></td> <td>65.6%</td>
<td><strong>329</strong></td> <td>299</td>
</tr> </tr>
</tbody> </tbody>
</table> </table>
<p>This is not a precision/recall trade — restriction wins on every axis at once: <p>Restriction improves every metric at once, not a precision/recall trade:
<strong>+3.3pp F1, +3.9pp recall, and less than a third the total misIDs.</strong> Fewer +4.8pp F1, +6.0pp recall, roughly a quarter the total misIDs. Fewer
candidates in the matcher's search space means fewer opportunities for a candidates in the matcher's search space means fewer opportunities for a
look-alike false match (an actor who happens to share enough facial structure lookalike false match, and the recall gain shows this does not cost real
with someone in the film, but isn't actually in it), and the recall gain shows detections.</p>
it isn't costing real detections to get there.</p> <p>Every model's best-scoring combo in the training matrix uses the
<p>Per-model, every single model's best-scoring combo in the full 16-way matrix is restricted gallery:</p>
a <code>restricted</code> variant — visible directly in the ranking below (filled dots = <p><img alt="All combos ranked by training-set F1, filled dots are restricted" src="../assets/images/rep4_matrix_f1.png" /></p>
restricted, open = full; the filled dots cluster at the top for every color):</p> <p>See <a href="../model-bakeoff/">the full experiment log</a> for the complete table. One
<p><img alt="All 16 bake-off combos — filled dots (restricted) dominate the top" src="../assets/images/rep4_matrix_f1.png" /></p> combo reaches zero true out-of-cast misidentifications,
<p>See the full table in the <code>arcface_w600k_mbf_restricted_exp</code> (F1 76.2%), and it is a restricted one,
<a href="../model-bakeoff/">bake-off experiment log</a>. Two consistent with restriction, not expansion, being what suppresses cross-film
combos hit <strong>zero</strong> true out-of-cast misidentifications: confusions.</p>
<code>arcface_w600k_mbf_restricted_exp</code> (F1 76.5%) and, in full mode, <p>The restriction effect (+4.8pp averaged across models) is larger than the
<code>LVFace-B_Glint360K_full_noexp</code> (F1 72.4%) — restriction isn't the only way to model-choice effect: LVFace beats r18 by 6.2pp in full mode but beats mbf by
reach misid=0, but it's the more reliable one.</p> 3.3pp. Restriction is the single strongest lever in the matrix.</p>
<h2 id="why-this-isnt-the-shipped-default">Why this isn't the shipped default<a class="headerlink" href="#why-this-isnt-the-shipped-default" title="Permanent link">&para;</a></h2> <h2 id="why-this-is-not-the-shipped-default">Why this is not the shipped default<a class="headerlink" href="#why-this-is-not-the-shipped-default" title="Permanent link">&para;</a></h2>
<p>Cast-restriction is implemented today only as an <strong>offline optimizer technique</strong> <p>Cast restriction is implemented today only as an offline optimizer
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/cast_restrict.py"><code>scripts/optimizer/cast_restrict.py</code></a>): technique
it pre-builds a filtered gallery file (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/optimizer/cast_restrict.py"><code>scripts/optimizer/cast_restrict.py</code></a>):
per film, using Jellyfin's own cast list, before the benchmark ever calls the it pre-builds a filtered gallery file per film using Jellyfin's cast list
matcher. There's no runtime "restrict matching to this title's credited cast" before the benchmark calls the matcher. There is no runtime "restrict to
switch in the shipped application<code>scene_analyze</code> always matches against this title's credited cast" switch in the shipped application;
whatever single gallery file it's given.</p> <code>scene_analyze</code> always matches against whatever single gallery file it is
<p>Building that as a real feature would need, at minimum:</p> given.</p>
<p>Building this as a real feature requires:</p>
<ul> <ul>
<li>A live Jellyfin cast lookup at analysis time (the title is already known <li>A live Jellyfin cast lookup at analysis time. The title is already known,
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a> and <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a>
already does this same lookup for its own already performs this lookup for its own <code>filter_gallery</code>-based
<code>filter_gallery</code>-based restriction path, just not wired into <code>scene_analyze</code> restriction path; it is not wired into <code>scene_analyze</code> as a first-class
itself as a first-class option).</li> option.</li>
<li>A decision on the <em>fallback</em>: what happens to a real, uncredited cameo <li>A decision on the fallback case: what happens to a real, uncredited
(see the Germar Terrell Gardner case in the LVFace deep-dive) if the gallery cameo (see the Germar Terrell Gardner and Talia Balsam cases in the
never includes them at all?</li> <a href="../lvface-deep-dive/#where-lvface-beat-x-ray">LVFace deep dive</a>) if the
<li>Regenerating the restricted-gallery cache whenever the title's Jellyfin cast restricted gallery never includes them at all.</li>
list changes.</li> <li>Regenerating the restricted-gallery cache whenever a title's Jellyfin
cast list changes.</li>
</ul> </ul>
<p>This is why the shipped <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/config.hpp"><code>src/config.hpp</code></a> <p>The shipped <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/config.hpp"><code>src/config.hpp</code></a> defaults use
defaults use the <code>full</code>-mode winner the full-mode winner (<code>LVFace-B_Glint360K_full_exp</code>, F1 75.3% training,
(<code>LVFace-B_Glint360K_full_exp</code>, F1 75.3% training / 67.4% held-out macro) rather 67.4% held-out macro) rather than the higher-scoring <code>restricted_exp</code>
than the higher-scoring <code>restricted_exp</code> (78.3%) — the 78.3% number describes a (78.3%), because 78.3% describes a capability the application does not
capability the app doesn't have yet, not what actually ships.</p> have yet.</p>
+110 -108
View File
@@ -14,7 +14,7 @@
<link rel="next" href="best-model/"> <link rel="next" href="methodology/">
@@ -243,6 +243,25 @@
<li class="md-tabs__item">
<a href="methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
<li class="md-tabs__item"> <li class="md-tabs__item">
@@ -270,26 +289,7 @@
Model Bake-off & Re-tune (full log) Full Experiment Log
</a>
</li>
<li class="md-tabs__item">
<a href="optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
</a> </a>
</li> </li>
@@ -427,10 +427,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#start-here-four-questions-this-bake-off-answers" class="md-nav__link"> <a href="#findings" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Start here — four questions this bake-off answers Findings
</span> </span>
</a> </a>
@@ -438,10 +438,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#the-full-technical-log" class="md-nav__link"> <a href="#full-experiment-log" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
The full technical log Full experiment log
</span> </span>
</a> </a>
@@ -473,6 +473,33 @@
<li class="md-nav__item">
<a href="methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -487,10 +514,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" > <input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="0"> <label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
@@ -508,8 +535,8 @@
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
</label> </label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="false"> <nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_2"> <label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
@@ -652,34 +679,7 @@
<span class="md-ellipsis"> <span class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
</span>
</a>
</li>
<li class="md-nav__item">
<a href="optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
@@ -746,10 +746,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#start-here-four-questions-this-bake-off-answers" class="md-nav__link"> <a href="#findings" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Start here — four questions this bake-off answers Findings
</span> </span>
</a> </a>
@@ -757,10 +757,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#the-full-technical-log" class="md-nav__link"> <a href="#full-experiment-log" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
The full technical log Full experiment log
</span> </span>
</a> </a>
@@ -796,79 +796,81 @@
<h1 id="scene-actor-extraction">scene-actor-extraction<a class="headerlink" href="#scene-actor-extraction" title="Permanent link">&para;</a></h1> <h1 id="scene-actor-extraction">scene-actor-extraction<a class="headerlink" href="#scene-actor-extraction" title="Permanent link">&para;</a></h1>
<p>A face-recognition pipeline that finds when each actor appears on screen in a <p>A face-recognition pipeline that finds when each actor appears on screen in
film or TV episode built on <a href="https://gitea.tourolle.paris/dtourolle/KPN">KPN++</a> a film or TV episode, built on <a href="https://gitea.tourolle.paris/dtourolle/KPN">KPN++</a>
(a C++20 Kahn Process Network library) for the detect track match → scene (a C++20 Kahn Process Network library) for the detect, track, match, and
pipeline, with a Jellyfin-integrated gallery and an X-Ray-validated optimizer.</p> scene pipeline, with a Jellyfin-integrated gallery and an X-Ray-validated
<p>This is a perfect X-Ray second, on a film the optimizer never saw:</p> optimizer.</p>
<p>This is a correctly scored second from a held-out film, one the optimizer
never saw during tuning:</p>
<p><img alt="A perfect X-Ray second: three faces named at 100%, two more correctly carried off-screen" src="assets/images/lovelace_perfect_second.jpg" /></p> <p><img alt="A perfect X-Ray second: three faces named at 100%, two more correctly carried off-screen" src="assets/images/lovelace_perfect_second.jpg" /></p>
<p>Every visible face named at 100% Chris Noth, Hank Azaria, Bobby Cannavale — <p>Every visible face is named at 100% confidence (Chris Noth, Hank Azaria,
the background extra honestly left unnamed, and the two credited cast without Bobby Cannavale), the background extra is correctly left unnamed, and the
a visible face correctly carried as present off-screen by the tracker's two credited cast members without a visible face are correctly reported
presence windows. That's the pipeline exactly reproducing Amazon X-Ray's present but not visible. This matches Amazon X-Ray's own record for this
record for this second.</p> second exactly.</p>
<p>It doesn't always go like that: the hardest held-out film scores 46% F1, and <p>Results are not uniform across films. The hardest held-out film scores 46%
the report is honest about <em>why</em> one tunable trade (extinction bridging at F1. This report documents why: one tunable trade (extinction bridging at
hard cuts), one structural ceiling (X-Ray credits people whose faces never hard cuts), one structural limit (X-Ray credits people whose faces never
appear), and a few cases where the pipeline is right and X-Ray is wrong. The appear on screen), and a small number of cases where the pipeline is
evidence for all of it is in the pages below.</p> correct and X-Ray's ground truth is not. Read
<h2 id="start-here-four-questions-this-bake-off-answers">Start here — four questions this bake-off answers<a class="headerlink" href="#start-here-four-questions-this-bake-off-answers" title="Permanent link">&para;</a></h2> <a href="methodology/">how we score against X-Ray</a> first. X-Ray's ground truth is
scene-level; the pipeline's output is per-second. That difference shapes
every finding below.</p>
<h2 id="findings">Findings<a class="headerlink" href="#findings" title="Permanent link">&para;</a></h2>
<div class="grid cards"> <div class="grid cards">
<ul> <ul>
<li> <li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M18 2c-.9 0-2 1-2 2H8c0-1-1.1-2-2-2H2v9c0 1 1 2 2 2h2.2c.4 2 1.7 3.7 4.8 4v2.08C8 19.54 8 22 8 22h8s0-2.46-3-2.92V17c3.1-.3 4.4-2 4.8-4H20c1 0 2-1 2-2V2zM6 11H4V4h2zm14 0h-2V4h2z"/></svg></span> <strong><a href="best-model/">Which model is best?</a></strong></p> <p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M18 2c-.9 0-2 1-2 2H8c0-1-1.1-2-2-2H2v9c0 1 1 2 2 2h2.2c.4 2 1.7 3.7 4.8 4v2.08C8 19.54 8 22 8 22h8s0-2.46-3-2.92V17c3.1-.3 4.4-2 4.8-4H20c1 0 2-1 2-2V2zM6 11H4V4h2zm14 0h-2V4h2z"/></svg></span> <strong><a href="best-model/">Which model is best?</a></strong></p>
<hr /> <hr />
<p>Calibration curves first (discriminative power, independent of any <p>Calibration curves first, independent of any threshold, then held-out
threshold), then F1 on the actual benchmark. LVFace-B Glint360K wins F1 across three models. LVFace-B Glint360K wins both, and wins on every
both.</p> held-out film.</p>
</li> </li>
<li> <li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M14 12v7.88c.04.3-.06.62-.29.83a.996.996 0 0 1-1.41 0l-2.01-2.01a.99.99 0 0 1-.29-.83V12h-.03L4.21 4.62a1 1 0 0 1 .17-1.4c.19-.14.4-.22.62-.22h14c.22 0 .43.08.62.22a1 1 0 0 1 .17 1.4L14.03 12z"/></svg></span> <strong><a href="gallery-scope/">Whole vs. cast-restricted gallery</a></strong></p> <p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M14 12v7.88c.04.3-.06.62-.29.83a.996.996 0 0 1-1.41 0l-2.01-2.01a.99.99 0 0 1-.29-.83V12h-.03L4.21 4.62a1 1 0 0 1 .17-1.4c.19-.14.4-.22.62-.22h14c.22 0 .43.08.62.22a1 1 0 0 1 .17 1.4L14.03 12z"/></svg></span> <strong><a href="gallery-scope/">Whole vs. cast-restricted gallery</a></strong></p>
<hr /> <hr />
<p>Restricting the matcher to a film's credited cast is a clean win on <p>Restricting the matcher to a film's credited cast improves F1,
every axis (+3.3pp F1, less than a third the misIDs) — but isn't a recall, and misID rate at once, but is not a shipped runtime feature
shipped runtime feature yet.</p> yet.</p>
</li> </li>
<li> <li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m12 0-.66.03 3.81 3.81L16.5 2.5c3.25 1.57 5.59 4.74 5.95 8.5h1.5C23.44 4.84 18.29 0 12 0m0 4c-1.93 0-3.5 1.57-3.5 3.5S10.07 11 12 11s3.5-1.57 3.5-3.5S13.93 4 12 4M.05 13C.56 19.16 5.71 24 12 24l.66-.03-3.81-3.81L7.5 21.5c-3.25-1.56-5.59-4.74-5.95-8.5zM12 13c-3.87 0-7 1.57-7 3.5V18h14v-1.5c0-1.93-3.13-3.5-7-3.5"/></svg></span> <strong><a href="pose-expansion/">Does pose expansion help?</a></strong></p> <p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m12 0-.66.03 3.81 3.81L16.5 2.5c3.25 1.57 5.59 4.74 5.95 8.5h1.5C23.44 4.84 18.29 0 12 0m0 4c-1.93 0-3.5 1.57-3.5 3.5S10.07 11 12 11s3.5-1.57 3.5-3.5S13.93 4 12 4M.05 13C.56 19.16 5.71 24 12 24l.66-.03-3.81-3.81L7.5 21.5c-3.25-1.56-5.59-4.74-5.95-8.5zM12 13c-3.87 0-7 1.57-7 3.5V18h14v-1.5c0-1.93-3.13-3.5-7-3.5"/></svg></span> <strong><a href="pose-expansion/">Does pose expansion help?</a></strong></p>
<hr /> <hr />
<p>A convincing training-set effect that didn't reproduce on 5 held-out <p>A training-set effect that did not reproduce on 5 held-out films once
films once two methodology bugs were caught and fixed. An honest null two methodology bugs in the comparison harness were found and fixed.</p>
result, not a forced narrative.</p>
</li> </li>
<li> <li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M18 16h-.58l-.81-.81A7.07 7.07 0 0 0 18 11c0-3.87-3.13-7-7-7-1.5 0-3 .5-4.21 1.4-3.09 2.32-3.72 6.71-1.4 9.8s6.71 3.72 9.8 1.4l.81.81V18l5 5 2-2zm-7 0c-2.76 0-5-2.24-5-5s2.24-5 5-5 5 2.24 5 5-2.24 5-5 5M3 6 1 8V1h7L6 3H3zm18-5v7l-2-2V3h-3l-2-2zM6 19l2 2H1v-7l2 2v3z"/></svg></span> <strong><a href="lvface-deep-dive/">Deep dive: LVFace-B Glint360K</a></strong></p> <p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M18 16h-.58l-.81-.81A7.07 7.07 0 0 0 18 11c0-3.87-3.13-7-7-7-1.5 0-3 .5-4.21 1.4-3.09 2.32-3.72 6.71-1.4 9.8s6.71 3.72 9.8 1.4l.81.81V18l5 5 2-2zm-7 0c-2.76 0-5-2.24-5-5s2.24-5 5-5 5 2.24 5 5-2.24 5-5 5M3 6 1 8V1h7L6 3H3zm18-5v7l-2-2V3h-3l-2-2zM6 19l2 2H1v-7l2 2v3z"/></svg></span> <strong><a href="lvface-deep-dive/">Deep dive: LVFace-B Glint360K</a></strong></p>
<hr /> <hr />
<p>The held-out generalization gap, how the error budget decomposes <p>The held-out generalization gap, the two mechanisms behind its errors,
(extinction bridging at hard cuts, X-Ray's scene-membership vs. and every distinct case where it names someone outside the film's
on-screen-face ceiling), and the frames where the pipeline is right credited cast.</p>
and the ground truth is wrong.</p>
</li> </li>
</ul> </ul>
</div> </div>
<h2 id="the-full-technical-log">The full technical log<a class="headerlink" href="#the-full-technical-log" title="Permanent link">&para;</a></h2> <h2 id="full-experiment-log">Full experiment log<a class="headerlink" href="#full-experiment-log" title="Permanent link">&para;</a></h2>
<ul> <ul>
<li><strong><a href="model-bakeoff/">Model bake-off + threshold re-tune</a></strong> <li><strong><a href="model-bakeoff/">Full experiment log</a></strong>: the complete log behind the
the complete experiment log behind the four pages above: the ROCm teardown four pages above, including how replaying against cached embeddings
deadlock root cause and fix, DE concurrency tuning, the full 16-combo inside the same KPN network makes a full model and configuration
results table, and every caveat. This is where the shipped comparison practical, the full results table, and every caveat. This is
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/config.hpp"><code>src/config.hpp</code></a> defaults come from.</li> where the shipped <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/config.hpp"><code>src/config.hpp</code></a>
<li><strong><a href="optimizer-experiments/">Optimizer experiments (prior round)</a></strong> — the defaults come from.</li>
earlier scene-union-metric tuning pass, superseded by the per-second metric <li><strong><a href="service-conversion/">Service conversion (proposal)</a></strong>: design
used in the bake-off but kept for the ground-truth/architecture background.</li> sketch for a native idle-GPU worker gated on screen lock, not yet built.</li>
<li><strong><a href="service-conversion/">Service conversion (proposal)</a></strong> — design sketch
for a native idle-GPU worker gated on screen lock, not yet built.</li>
</ul> </ul>
<h2 id="reproducing-the-benchmarks">Reproducing the benchmarks<a class="headerlink" href="#reproducing-the-benchmarks" title="Permanent link">&para;</a></h2> <h2 id="reproducing-the-benchmarks">Reproducing the benchmarks<a class="headerlink" href="#reproducing-the-benchmarks" title="Permanent link">&para;</a></h2>
<p>Gallery <code>.h5</code> files, embedding dumps, the X-Ray corpus, montage frame images, <p>Gallery <code>.h5</code> files, embedding dumps, the X-Ray corpus, montage frame
and DE trajectories are not committed to this repository — they're pushed to images, and DE trajectories are not committed to this repository. They are
the Gitea package registry and pulled on demand:</p> pushed to the Gitea package registry and pulled on demand:</p>
<div class="language-bash highlight"><pre><span></span><code><span id="__span-0-1"><a id="__codelineno-0-1" name="__codelineno-0-1" href="#__codelineno-0-1"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>galleries <div class="language-bash highlight"><pre><span></span><code><span id="__span-0-1"><a id="__codelineno-0-1" name="__codelineno-0-1" href="#__codelineno-0-1"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>galleries
</span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>experiment-data </span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>experiment-data
</span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>montage-frames<span class="w"> </span>&lt;film-slug&gt; </span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>montage-frames<span class="w"> </span>&lt;film-slug&gt;
</span></code></pre></div> </span></code></pre></div>
<p>See <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/artifacts/push_artifacts.sh"><code>scripts/artifacts/push_artifacts.sh</code></a> <p>See <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/artifacts/push_artifacts.sh"><code>scripts/artifacts/push_artifacts.sh</code></a>
for the upload side (requires a <code>GITEA_TOKEN</code> with package write scope).</p> for the upload side, which requires a <code>GITEA_TOKEN</code> with package write
scope.</p>
@@ -905,13 +907,13 @@ for the upload side (requires a <code>GITEA_TOKEN</code> with package write scop
<a href="best-model/" class="md-footer__link md-footer__link--next" aria-label="Next: Best Model"> <a href="methodology/" class="md-footer__link md-footer__link--next" aria-label="Next: How We Score Against X-Ray">
<div class="md-footer__title"> <div class="md-footer__title">
<span class="md-footer__direction"> <span class="md-footer__direction">
Next Next
</span> </span>
<div class="md-ellipsis"> <div class="md-ellipsis">
Best Model How We Score Against X-Ray
</div> </div>
</div> </div>
<div class="md-footer__button md-icon"> <div class="md-footer__button md-icon">
+481 -196
View File
@@ -242,6 +242,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -272,26 +291,7 @@
Model Bake-off & Re-tune (full log) Full Experiment Log
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
</a> </a>
</li> </li>
@@ -393,6 +393,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -414,10 +441,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked> <input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex=""> <label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="">
@@ -435,8 +462,8 @@
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
</label> </label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true"> <nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2"> <label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
@@ -597,10 +624,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#what-good-looks-like" class="md-nav__link"> <a href="#baseline-correctly-scored-seconds" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
What good looks like Baseline: correctly scored seconds
</span> </span>
</a> </a>
@@ -619,10 +646,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" class="md-nav__link"> <a href="#mechanism-1-extinction-bridging" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Mechanism 1: extinction bridging — usually right, wrong at hard cuts Mechanism 1: extinction bridging
</span> </span>
</a> </a>
@@ -638,6 +665,78 @@
</span> </span>
</a> </a>
</li>
<li class="md-nav__item">
<a href="#every-distinct-out-of-cast-name" class="md-nav__link">
<span class="md-ellipsis">
Every distinct out-of-cast name
</span>
</a>
<nav class="md-nav" aria-label="Every distinct out-of-cast name">
<ul class="md-nav__list">
<li class="md-nav__item">
<a href="#the-many-saints-of-newark-4-names" class="md-nav__link">
<span class="md-ellipsis">
The Many Saints of Newark: 4 names
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#lord-of-war-3-names" class="md-nav__link">
<span class="md-ellipsis">
Lord of War: 3 names
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#lovelace-1-name" class="md-nav__link">
<span class="md-ellipsis">
Lovelace: 1 name
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#scarface-1-name" class="md-nav__link">
<span class="md-ellipsis">
Scarface: 1 name
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#summary-of-the-nine" class="md-nav__link">
<span class="md-ellipsis">
Summary of the nine
</span>
</a>
</li>
</ul>
</nav>
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
@@ -692,34 +791,7 @@
<span class="md-ellipsis"> <span class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
@@ -786,10 +858,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#what-good-looks-like" class="md-nav__link"> <a href="#baseline-correctly-scored-seconds" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
What good looks like Baseline: correctly scored seconds
</span> </span>
</a> </a>
@@ -808,10 +880,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" class="md-nav__link"> <a href="#mechanism-1-extinction-bridging" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Mechanism 1: extinction bridging — usually right, wrong at hard cuts Mechanism 1: extinction bridging
</span> </span>
</a> </a>
@@ -827,6 +899,78 @@
</span> </span>
</a> </a>
</li>
<li class="md-nav__item">
<a href="#every-distinct-out-of-cast-name" class="md-nav__link">
<span class="md-ellipsis">
Every distinct out-of-cast name
</span>
</a>
<nav class="md-nav" aria-label="Every distinct out-of-cast name">
<ul class="md-nav__list">
<li class="md-nav__item">
<a href="#the-many-saints-of-newark-4-names" class="md-nav__link">
<span class="md-ellipsis">
The Many Saints of Newark: 4 names
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#lord-of-war-3-names" class="md-nav__link">
<span class="md-ellipsis">
Lord of War: 3 names
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#lovelace-1-name" class="md-nav__link">
<span class="md-ellipsis">
Lovelace: 1 name
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#scarface-1-name" class="md-nav__link">
<span class="md-ellipsis">
Scarface: 1 name
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#summary-of-the-nine" class="md-nav__link">
<span class="md-ellipsis">
Summary of the nine
</span>
</a>
</li>
</ul>
</nav>
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
@@ -869,43 +1013,44 @@
<h1 id="deep-dive-lvface-b-glint360k">Deep dive: LVFace-B Glint360K<a class="headerlink" href="#deep-dive-lvface-b-glint360k" title="Permanent link">&para;</a></h1> <h1 id="deep-dive-lvface-b-glint360k">Deep dive: LVFace-B Glint360K<a class="headerlink" href="#deep-dive-lvface-b-glint360k" title="Permanent link">&para;</a></h1>
<p>LVFace won the model bake-off (see <a href="../best-model/">Which model is best?</a>) and is <p>LVFace won the model comparison (see <a href="../best-model/">Which model is best?</a>)
the shipped default embedder. This page is the honest accounting of how it and is the shipped default embedder. This page reports how it performs in
actually performs — what a good second looks like, where the errors actually detail: a baseline of correct output, the two mechanisms behind its errors,
come from, and two cases where the ground truth itself is wrong and LVFace is and every distinct case where it names someone who is not in the film's
right.</p> credited cast.</p>
<p>Read <a href="../methodology/">How we score against X-Ray</a> first. X-Ray's ground truth
is scene-level, not per-frame. A name marked correct in the Offscreen column
below is the pipeline correctly reporting scene membership, not a workaround.</p>
<div class="admonition note"> <div class="admonition note">
<p class="admonition-title">How to read the frames on this page</p> <p class="admonition-title">How to read the frames on this page</p>
<p>The top is the film frame, with a box and name on every face the pipeline <p>The top of each image is the film frame, with a box and name on every
identified. The bottom panels are the per-second verdict against X-Ray: face the pipeline matched to a real detection. The panels below are the
<strong>Onscreen</strong> lists faces named in the frame, <strong>Offscreen</strong> lists cast per-second result against X-Ray. <strong>Onscreen</strong> lists names attached to a
X-Ray marks present in the scene without a visible face — presence visible face this second. <strong>Offscreen</strong> lists names the pipeline reports
carried by the tracker's windows, not by a detection. Colors are the present without a currently visible face. Colors mark the verdict:
score: <span style="color:#0ca30c"><strong>green</strong></span> = correct (TPI), <span style="color:#0ca30c"><strong>green</strong></span> correct (TPI),
<span style="color:#eb6834"><strong>orange</strong></span> = wrong (FPI), <span style="color:#eb6834"><strong>orange</strong></span> wrong (FPI),
<span style="color:#3987e5"><strong>blue</strong></span> = missed (FN).</p> <span style="color:#3987e5"><strong>blue</strong></span> missed (FN).</p>
</div> </div>
<h2 id="what-good-looks-like">What good looks like<a class="headerlink" href="#what-good-looks-like" title="Permanent link">&para;</a></h2> <h2 id="baseline-correctly-scored-seconds">Baseline: correctly scored seconds<a class="headerlink" href="#baseline-correctly-scored-seconds" title="Permanent link">&para;</a></h2>
<p><img alt="Wedding couple correctly identified, Downton Abbey: A New Era" src="../assets/images/downton_wedding_couple.jpg" /></p> <p><img alt="Wedding couple correctly identified, Downton Abbey: A New Era" src="../assets/images/downton_wedding_couple.jpg" /></p>
<p>Six faces on screen, all six named correctly including Penelope Wilton at the <p>Six faces on screen, all six named correctly, including Penelope Wilton at
edge of the pews and a half-occluded Michelle Dockery — while thirteen more the edge of the pews and a partly occluded Michelle Dockery. Thirteen more
cast members X-Ray marks present in the scene are correctly carried as cast members X-Ray lists as present in the scene are correctly reported
"Offscreen" by their presence windows. One miss in the whole frame: Maggie Offscreen. One miss: Maggie Smith (blue). Score for this second: 0.86.</p>
Smith (blue). Score for this second: 0.86.</p>
<p><img alt="19 of 20 correct in the funeral crowd" src="../assets/images/downton_funeral_19of20.jpg" /></p> <p><img alt="19 of 20 correct in the funeral crowd" src="../assets/images/downton_funeral_19of20.jpg" /></p>
<p>The same film's funeral gathering: mourning dress, hats, half the faces turned. <p>The same film's funeral scene: dark clothing, hats, half the faces turned
<strong>Nineteen of the twenty cast X-Ray lists for this scene are scored correctly</strong> away. Nineteen of the twenty cast members X-Ray lists for this scene score
seven named on screen at up to 100% confidence, twelve more correctly held correct: seven named on screen at up to 100% confidence, twelve more reported
as present off-screen.</p> correctly as present but not visible.</p>
<p>And the pipeline doesn't need the face to be <em>real</em>:</p>
<p><img alt="Herbie Hancock identified on an in-fiction video call" src="../assets/images/valerian_screen_call.jpg" /></p> <p><img alt="Herbie Hancock identified on an in-fiction video call" src="../assets/images/valerian_screen_call.jpg" /></p>
<p>That's Herbie Hancock at 98% — as a face on a <em>screen inside the movie</em>, over a <p>The pipeline does not require a live face. This is Herbie Hancock at 98%
sci-fi HUD overlay, during a video call in Valerian. A face is a face, whether confidence, identified from a face displayed on a screen inside the film, on
it's in the room or on the bridge's comms display.</p> a video call under a science-fiction HUD overlay.</p>
<h2 id="training-vs-held-out-the-generalization-gap">Training vs. held-out: the generalization gap<a class="headerlink" href="#training-vs-held-out-the-generalization-gap" title="Permanent link">&para;</a></h2> <h2 id="training-vs-held-out-the-generalization-gap">Training vs. held-out: the generalization gap<a class="headerlink" href="#training-vs-held-out-the-generalization-gap" title="Permanent link">&para;</a></h2>
<p>The shipped config (<code>prob_threshold=0.754, anneal_sec=35.54, <p>The shipped config (<code>prob_threshold=0.754</code>, <code>anneal_sec=35.54</code>,
extinction_sec=57.43, expand_gallery=true</code>) was tuned against 4 films. Scored <code>extinction_sec=57.43</code>, <code>expand_gallery=true</code>) was tuned on 4 films. Scored
against the 5 films the optimizer never saw:</p> on the 5 films the optimizer never saw:</p>
<p><img alt="Held-out per-film F1 vs. the training-set fit" src="../assets/images/holdout_f1_by_film.png" /></p> <p><img alt="Held-out per-film F1 vs. the training-set fit" src="../assets/images/holdout_f1_by_film.png" /></p>
<table> <table>
<thead> <thead>
@@ -962,18 +1107,18 @@ against the 5 films the optimizer never saw:</p>
<td>80084</td> <td>80084</td>
</tr> </tr>
<tr> <tr>
<td><strong>The Many Saints of Newark</strong></td> <td>The Many Saints of Newark</td>
<td><strong>46.3%</strong></td> <td>46.3%</td>
<td><strong>54.7%</strong></td> <td>54.7%</td>
<td>40.1%</td> <td>40.1%</td>
<td>15922</td> <td>15922</td>
<td>4394</td> <td>4394</td>
<td><strong>974</strong></td> <td>974</td>
<td>23791</td> <td>23791</td>
</tr> </tr>
<tr> <tr>
<td><strong>macro average</strong></td> <td>macro average</td>
<td><strong>67.4%</strong></td> <td>67.4%</td>
<td>85.8%</td> <td>85.8%</td>
<td>57.0%</td> <td>57.0%</td>
<td></td> <td></td>
@@ -983,112 +1128,252 @@ against the 5 films the optimizer never saw:</p>
</tr> </tr>
</tbody> </tbody>
</table> </table>
<p><strong>67.4% held-out vs. 75.3% on training</strong> — an ~8pp drop, and a <strong>37pp spread <p>The <code>P</code> column is misID-weighted (each out-of-film name counts 10x in the
between the best and worst held-out film</strong>. The config does not generalize denominator; see <a href="../methodology/#precision-recall-and-the-misid-weighting">methodology</a>).
uniformly, and the spread traces to two mechanisms, both visible frame by That weighting is why Many Saints reads 54.7% here despite naming mostly real,
frame below.</p> present faces: its raw (unweighted) precision is <strong>78.4%</strong>, and the gap is
<h2 id="mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts">Mechanism 1: extinction bridging — usually right, wrong at hard cuts<a class="headerlink" href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" title="Permanent link">&para;</a></h2> entirely its 974 misIDs paying the 10x penalty. The three zero-misID films
<p>The extinction window keeps an identity alive through seconds where no face is (Benny &amp; Joon, Downton, Valerian) have identical weighted and raw precision;
detectable. <strong>Most of the time this is exactly what you want</strong>, and it's where Lovelace, with 58 misIDs, sits 3pp below its raw 93.3%.</p>
a lot of the TPI count comes from:</p> <p>Held-out F1 is 67.4%, against 75.3% on training, an 8pp drop. The spread
between the best and worst held-out film is 37pp. This is not unique to
LVFace: <a href="../model-bakeoff/#held-out-validation-all-3-models">the full experiment log</a>
shows mbf and r18 with the same shape of spread on the same films, at a
uniformly lower level. Two mechanisms explain the spread. Both are shown
below with frame-level evidence.</p>
<h2 id="mechanism-1-extinction-bridging">Mechanism 1: extinction bridging<a class="headerlink" href="#mechanism-1-extinction-bridging" title="Permanent link">&para;</a></h2>
<p>The extinction window keeps a name reported as present for up to
<code>extinction_sec</code> after its last real detection. This is deliberate: most
gaps in face visibility are short (a turned head, an occlusion, a cut to a
reaction shot), and the window bridges them.</p>
<p><img alt="Two faces on screen, six more correctly bridged" src="../assets/images/lovelace_polygraph_bridged.jpg" /></p> <p><img alt="Two faces on screen, six more correctly bridged" src="../assets/images/lovelace_polygraph_bridged.jpg" /></p>
<p>Lovelace's polygraph scene: only Eric Roberts and Amanda Seyfried have visible <p>Lovelace's polygraph scene: only Eric Roberts and Amanda Seyfried have
faces, but X-Ray lists eight cast present — and all eight score green, the visible faces. X-Ray lists eight cast members present. All eight score
other six correctly carried by presence windows through a scene where the correct; the other six are reported Offscreen through a stretch where the
camera never shows them. A perfect second, and the extinction/anneal machinery camera never shows them. The extinction window is why.</p>
is <em>why</em>.</p> <p>The same mechanism fails at a hard cut into a long stretch with no faces at
<p>The same mechanism has a failure case: a hard cut into long faceless footage. all. Downton Abbey's recall (39.4%, the worst of the five held-out films) is
Both Many Saints of Newark (974 misIDs) and Downton Abbey (FN=80084, the worst dominated by this failure. It is verified directly against the raw
recall of the five) are dominated by it — verified directly against the raw per-frame stream and the dump's own detection counts, not inferred from the
per-frame stream and the HDF5 dump's own detection counts, not inferred from score. Plotting the dump's per-second <code>face_count</code> (detector output,
the score alone. <strong>This is not a malfunction</strong>: the tracker is doing exactly independent of the tracker) against what the tracker reports, through
what its window is for; the footage just stops cooperating. In the debug Downton Abbey's hard cut into its closing credits:</p>
overlay (which draws a bridged identity's last-known bbox, unlike the shipped
output, which emits presence windows and no boxes at all) the bridged state is
visible spatially:</p>
<p><img alt="Debug overlay: bridged identities drawn at their last-known positions" src="../assets/images/many_saints_ghost_fpi.jpg" />
<em>Debug-overlay rendering (<code>dump_error_frames.py --raw</code>): "Jon Bernthal", "Joey
Diaz" and "Billy Magnussen" are extinction-bridged identities from the previous
shot, drawn frozen over the wall and the hanging plates. Frame
<code>many_saints/fpi/fpi_t03543.jpg</code>, <code>montage-frames</code> artifact package.</em></p>
<p>The cost is measurable, not just visible. Downton Abbey's hard cut into its
closing credits, plotting the dump's own per-second <code>face_count</code> (detector
output, independent of the tracker) against what the tracker reports:</p>
<p><img alt="Detector vs. tracker through Downton Abbey's cut to credits" src="../assets/images/downton_ghost_timeline.png" /></p> <p><img alt="Detector vs. tracker through Downton Abbey's cut to credits" src="../assets/images/downton_ghost_timeline.png" /></p>
<p>From the cut onward the detector sees <strong>zero faces for nearly a minute</strong> — and <p>From the cut onward the detector reports zero faces for close to a minute.
the tracker keeps reporting the last shot's 15 identities the whole time The tracker continues reporting the previous shot's 15 identities for the
(verified for Hugh Bonneville: bbox <code>(1743.2, 0.0, 171.3, 317.8)</code>, unchanged to same span (verified for Hugh Bonneville: bbox <code>(1743.2, 0.0, 171.3, 317.8)</code>,
the pixel, at every sampled second for 57+ seconds). The staircase at the right unchanged to the pixel, at every sampled second for 57 seconds). The
edge is the extinction window expiring actor by actor. That plateau is staircase at the right edge is the extinction window expiring, actor by
<code>SceneTrackerFunc::active_[actor_idx].last_bbox</code> actor. This is <code>SceneTrackerFunc::active_[actor_idx].last_bbox</code>
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/nodes/scene_tracker_node.hpp"><code>src/nodes/scene_tracker_node.hpp</code></a>) (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/nodes/scene_tracker_node.hpp"><code>src/nodes/scene_tracker_node.hpp</code></a>)
re-emitted as designed: <code>extinction_sec=57.4</code> was tuned long because bridging re-emitted as designed. <code>extinction_sec=57.4</code> was tuned long because
wins on most footage (see the polygraph frame above) — the training films just bridging is correct on most footage, as in the polygraph scene above. The
never contained a faceless stretch long enough to show the cost side, and the training films did not contain a faceless stretch long enough to expose the
held-out set did.</p> cost side; the held-out set did.</p>
<p>The same track-continuation machinery has one milder spatial artifact, worth <p>The extinction window is a scoring concept, not something drawn on screen.
knowing when reading these frames:</p> The shipped output is presence windows with no bounding boxes. Even the
<p><img alt="Two labels on one face after a shot/reverse-shot cut" src="../assets/images/cafe_society_rapid_cut.jpg" /> debug overlay used for this report never draws a box for a bridged name: a
<em>Café Society (a training film), a shot/reverse-shot dialog: that is Steve name inside its extinction window with no current detection appears only as
Carell wearing both his own label and Jesse Eisenberg's.</em></p> a name in the Offscreen column, the same as every correctly bridged name
<p>At a rapid cut, the previous shot's track can linger for a beat at nearly the above.</p>
same screen position the new face occupies — here Jesse Eisenberg's box from <p>A related, smaller effect shows up at rapid cuts:</p>
the counter-shot lands on Steve Carell. Note what the score panel says, <p><img alt="Two labels on one face after a shot/reverse-shot cut" src="../assets/images/cafe_society_rapid_cut.jpg" /></p>
though: both actors are green, because both <em>are</em> present in this dialog <p>Café Society (a training film), a shot/reverse-shot dialog. The box on Steve
scene per X-Ray. The spatial label is briefly wrong; the per-second presence Carell's face carries two labels: his own, and Jesse Eisenberg's, left over
claim — the thing the pipeline actually ships — is right. It's the same trade from the counter-shot a moment earlier. Both names score correct, because
as the extinction window: track continuation smooths over cuts, and 1 fps both actors are present in this scene per X-Ray. The box position is
sampling occasionally catches the seam.</p> briefly wrong; the presence claim, which is what the pipeline ships, is
right.</p>
<h2 id="mechanism-2-the-face-vs-presence-ceiling">Mechanism 2: the face-vs-presence ceiling<a class="headerlink" href="#mechanism-2-the-face-vs-presence-ceiling" title="Permanent link">&para;</a></h2> <h2 id="mechanism-2-the-face-vs-presence-ceiling">Mechanism 2: the face-vs-presence ceiling<a class="headerlink" href="#mechanism-2-the-face-vs-presence-ceiling" title="Permanent link">&para;</a></h2>
<p>Downton Abbey's recall didn't collapse because faces were misread — it <p>Downton Abbey's recall did not collapse because faces were misread. It
collapsed because for most of its 80084 FN-seconds there was <strong>no face to collapsed because for most of its 80084 false-negative seconds there was no
read</strong>:</p> face to read.</p>
<p><img alt="22 cast credited, nobody facing the camera" src="../assets/images/downton_crew_fn.jpg" /></p> <p><img alt="22 cast credited, nobody facing the camera" src="../assets/images/downton_crew_fn.jpg" /></p>
<p>A newsreel crew hauls equipment through the hall: X-Ray credits 22 cast as <p>A newsreel crew moves equipment through the hall. X-Ray credits 22 cast
present in this scene; not one face looks at the camera. Eight are still members as present in this scene. None face the camera. Eight still score
scored green (windows bridging from adjacent shots) — the other fourteen are correct, carried by presence windows from adjacent shots. The other fourteen
blue FNs that no face-recognition pipeline could ever recover. X-Ray encodes are missed, and no face-recognition system can recover them, because there
<em>scene membership</em>; the pipeline measures <em>on-screen faces</em>. In ensemble films is no face in the frame. X-Ray records scene membership; the pipeline
those two definitions diverge massively, and that gap — not identification measures visible faces. In ensemble scenes these two quantities diverge, and
error — is most of what the FN column counts.</p> that gap accounts for most of the false-negative count.</p>
<p><img alt="Presence without a detectable face, The Many Saints of Newark" src="../assets/images/many_saints_outofcast_fpi.jpg" /></p> <h2 id="every-distinct-out-of-cast-name">Every distinct out-of-cast name<a class="headerlink" href="#every-distinct-out-of-cast-name" title="Permanent link">&para;</a></h2>
<p>Same ceiling from the other side: Michela De Rossi in frame but turned away, <p>Many Saints of Newark has the largest misID count of any held-out film: 974
five cast correctly bridged as offscreen (green), four blue FNs — and one seconds, weighted. Rather than characterize this from a single frame, the
orange we'll come back to below.</p> raw replay stream was searched directly for every name the pipeline reports
that is not in the film's credited cast. The same search was run on all 9
films in the benchmark, one rule applied uniformly: <strong>find the first second
each distinct out-of-cast name appears, and render that exact second.</strong></p>
<p>Five films produce no such name anywhere in their runtime: Benny &amp; Joon,
Café Society, Downton Abbey, Sound of Metal, Valerian. Zero out-of-cast
names across their entire length. Four films produce nine distinct names
between them, shown below in full, not a sample.</p>
<h3 id="the-many-saints-of-newark-4-names">The Many Saints of Newark: 4 names<a class="headerlink" href="#the-many-saints-of-newark-4-names" title="Permanent link">&para;</a></h3>
<p><img alt="Germar Terrell Gardner, first out-of-cast name in Many Saints" src="../assets/images/many_saints_fpi_gardner.jpg" /></p>
<p>Germar Terrell Gardner, t=848s, 78% confidence. A real, clearly visible
background actor. He is not in X-Ray's cast list for this film, but he is
credited in Jellyfin's independent cast metadata (see
<a href="#where-lvface-beat-x-ray">Where LVFace beat X-Ray</a> below). This is a
ground-truth gap, not a model error.</p>
<p><img alt="Archie Yates, second out-of-cast name in Many Saints" src="../assets/images/many_saints_fpi_yates.jpg" /></p>
<p>Archie Yates, t=2521s, 78% confidence. A real detected face, a genuine
lookalike confusion.</p>
<p><img alt="Zooey Deschanel, third out-of-cast name in Many Saints" src="../assets/images/many_saints_fpi_deschanel.jpg" /></p>
<p>Zooey Deschanel, t=2819s, 99% confidence. A real detected face at a dinner
table, high-confidence lookalike confusion.</p>
<p><img alt="Talia Balsam, fourth out-of-cast name in Many Saints" src="../assets/images/many_saints_fpi_balsam.jpg" /></p>
<p>Talia Balsam, t=4551s, 93% confidence. A real detected face. Talia Balsam
plays Mrs. Jarecki, a guidance counselor, in this film; she is confirmed
on screen by direct inspection of the frame. She does not appear in X-Ray's
<code>people.csv</code> for this title. This is a second ground-truth gap in the same
film, not a model error.</p>
<p>Two of these four names are ground-truth gaps (Gardner, Balsam), not
misidentifications. The other two (Yates, Deschanel) are genuine embedding
errors on real faces.</p>
<h3 id="lord-of-war-3-names">Lord of War: 3 names<a class="headerlink" href="#lord-of-war-3-names" title="Permanent link">&para;</a></h3>
<p><img alt="David Shumbris, first out-of-cast name in Lord of War" src="../assets/images/lord_of_war_fpi_shumbris.jpg" /></p>
<p>David Shumbris, t=418s, 81% confidence. A real face in a dim, low-detail
shot under a train track. A genuine lookalike confusion in poor lighting.</p>
<p><img alt="Ronald Reagan, second out-of-cast name in Lord of War" src="../assets/images/lord_of_war_fpi_reagan_photo.jpg" /></p>
<p>Ronald Reagan, t=1003s, 100% confidence. This is not a lookalike confusion.
The detected face is a photograph of Reagan appearing within the shot, not a
living actor. The detector and matcher both did their job correctly on the
image content in front of them; the error is that a photograph inside the
scene is not the same thing as an actor present in the scene, and the
pipeline has no way to draw that distinction from a face crop alone.</p>
<p><img alt="Lance Reddick, third out-of-cast name in Lord of War" src="../assets/images/lord_of_war_fpi_reddick.jpg" /></p>
<p>Lance Reddick, t=6424s, 78% confidence. A small, distant, low-detail face at
the edge of frame. A marginal, low-confidence lookalike confusion.</p>
<h3 id="lovelace-1-name">Lovelace: 1 name<a class="headerlink" href="#lovelace-1-name" title="Permanent link">&para;</a></h3>
<p><img alt="Chloë Sevigny, out-of-cast name in Lovelace" src="../assets/images/lovelace_fpi_sevigny.jpg" /></p>
<p>Chloë Sevigny, t=2451s, 100% confidence. Two boxes are drawn on the same
face: one correctly labeled Amanda Seyfried, one incorrectly labeled Chloë
Sevigny, both at 100%. A single detection producing two competing high-
confidence identities on the same crop.</p>
<h3 id="scarface-1-name">Scarface: 1 name<a class="headerlink" href="#scarface-1-name" title="Permanent link">&para;</a></h3>
<p><img alt="Kirstie Alley, out-of-cast name in Scarface" src="../assets/images/scarface_fpi_alley.jpg" /></p>
<p>Kirstie Alley, t=2451s, 89% confidence. Al Pacino is correctly identified in
the foreground at 100%; a background face in the same shot is wrongly
labeled Kirstie Alley. (The t=2451s here and the Lovelace Chloë Sevigny case
above landing on the identical second is a genuine coincidence, verified from
each film's raw stream by <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/docs/first_fpi_frames.py"><code>first_fpi_frames.py</code></a>,
not a transcription slip, two unrelated films whose <em>first</em> out-of-cast name
happens to fall at the same timestamp.)</p>
<h3 id="summary-of-the-nine">Summary of the nine<a class="headerlink" href="#summary-of-the-nine" title="Permanent link">&para;</a></h3>
<table>
<thead>
<tr>
<th>film</th>
<th>name</th>
<th>t (s)</th>
<th>confidence</th>
<th>classification</th>
</tr>
</thead>
<tbody>
<tr>
<td>Many Saints of Newark</td>
<td>Germar Terrell Gardner</td>
<td>848</td>
<td>78%</td>
<td>ground-truth gap</td>
</tr>
<tr>
<td>Many Saints of Newark</td>
<td>Archie Yates</td>
<td>2521</td>
<td>78%</td>
<td>lookalike confusion</td>
</tr>
<tr>
<td>Many Saints of Newark</td>
<td>Zooey Deschanel</td>
<td>2819</td>
<td>99%</td>
<td>lookalike confusion</td>
</tr>
<tr>
<td>Many Saints of Newark</td>
<td>Talia Balsam</td>
<td>4551</td>
<td>93%</td>
<td>ground-truth gap</td>
</tr>
<tr>
<td>Lord of War</td>
<td>David Shumbris</td>
<td>418</td>
<td>81%</td>
<td>lookalike confusion</td>
</tr>
<tr>
<td>Lord of War</td>
<td>Ronald Reagan</td>
<td>1003</td>
<td>100%</td>
<td>photo-in-frame</td>
</tr>
<tr>
<td>Lord of War</td>
<td>Lance Reddick</td>
<td>6424</td>
<td>78%</td>
<td>lookalike confusion, marginal</td>
</tr>
<tr>
<td>Lovelace</td>
<td>Chloë Sevigny</td>
<td>2451</td>
<td>100%</td>
<td>lookalike confusion</td>
</tr>
<tr>
<td>Scarface</td>
<td>Kirstie Alley</td>
<td>2451</td>
<td>89%</td>
<td>lookalike confusion</td>
</tr>
</tbody>
</table>
<p>Of nine distinct out-of-cast names across four films, two are ground-truth
gaps, one is a photograph misread as a person, and six are genuine
embedding-space confusions on real detected faces. None trace to extinction
bridging: every one of these nine is a fresh detection on a real face crop
at the second it first appears.</p>
<h2 id="where-lvface-beat-x-ray">Where LVFace beat X-Ray<a class="headerlink" href="#where-lvface-beat-x-ray" title="Permanent link">&para;</a></h2> <h2 id="where-lvface-beat-x-ray">Where LVFace beat X-Ray<a class="headerlink" href="#where-lvface-beat-x-ray" title="Permanent link">&para;</a></h2>
<p>Not every orange in these frames is actually wrong. <p>Not every name marked wrong is actually wrong.
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a> <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>
scores strictly against X-Ray — but X-Ray itself has holes, and the pipeline scores strictly against X-Ray, and X-Ray has gaps of its own.</p>
found two kinds.</p>
<p><img alt="LVFace correctly identifies Germar Terrell Gardner, uncredited by X-Ray" src="../assets/images/germar_beats_xray.jpg" /></p> <p><img alt="LVFace correctly identifies Germar Terrell Gardner, uncredited by X-Ray" src="../assets/images/germar_beats_xray.jpg" /></p>
<p>Germar Terrell Gardner — a real, clean, high-confidence detection — is counted <p>Germar Terrell Gardner, the same name from the table above, does not appear
as an out-of-cast misID because he doesn't appear in X-Ray's <code>people.csv</code> for in X-Ray's <code>people.csv</code> for The Many Saints of Newark. Jellyfin's
The Many Saints of Newark at all. But Jellyfin's independent cast metadata independent cast metadata does credit him for this film (cross-checked
<em>does</em> credit him for this exact film (cross-checked via against <code>experiments/manifests/jellyfin_casts.json</code> from the
<code>experiments/manifests/jellyfin_casts.json</code> from the <code>experiment-data</code> artifact <code>experiment-data</code> artifact package, a data source entirely separate from
package, a completely separate data source from X-Ray). That's also him in X-Ray). Talia Balsam is the same case: confirmed on screen, absent from
orange in the frame above — every one of those "errors" is the pipeline being X-Ray's cast list for this title.</p>
right about a person X-Ray forgot.</p>
<p><img alt="Robert Patrick, clearly on screen, scored wrong by a ground-truth gap" src="../assets/images/lovelace_robert_patrick_fpi.jpg" /></p> <p><img alt="Robert Patrick, clearly on screen, scored wrong by a ground-truth gap" src="../assets/images/lovelace_robert_patrick_fpi.jpg" /></p>
<p>And it isn't only uncredited bit-parts. That is <strong>Robert Patrick</strong> — top-billed <p>This extends past uncredited background actors. This is Robert Patrick,
in Lovelace, unmistakably on screen, reading his newspaper, identified at top-billed in Lovelace, clearly on screen reading a newspaper, identified at
100% scored orange because X-Ray's people-in-scene list for <em>this scene</em> 100%. The frame is scored wrong because X-Ray's people-in-scene list for
doesn't include him. The identification is flawless; the ground truth missed this specific scene omits him, despite crediting him elsewhere in the film.
an actor sitting in the middle of the frame.</p> The identification is correct; the ground truth is missing an entry.</p>
<p>This doesn't mean every flagged misID is secretly correct — Many Saints' <p>X-Ray is a large, convenient ground truth. It is not a complete one. The
974-count total is still overwhelmingly extinction bridging at cuts, not misID and FPI counts reported throughout this document include some fixed
uncredited cameos. But the X-Ray corpus is a convenient, large-scale ground amount of noise from gaps in X-Ray itself, in both directions.</p>
truth, not a perfect one, and the misID/FPI numbers in these tables carry an
irreducible noise floor from ground-truth gaps in both directions.</p>
<h2 id="summary">Summary<a class="headerlink" href="#summary" title="Permanent link">&para;</a></h2> <h2 id="summary">Summary<a class="headerlink" href="#summary" title="Permanent link">&para;</a></h2>
<p>LVFace is the right default: it wins the model comparison outright, it names <p>LVFace wins the model comparison on every held-out film. It correctly names
19 of 20 correctly across a hat-heavy funeral crowd, and it recognises a face 19 of 20 people in a crowded funeral scene and correctly identifies a face
on a screen inside the movie. Its error budget decomposes into two understood displayed on a screen inside the film. Its errors resolve into two
mechanisms extinction bridging at hard cuts (a tunable trade, not a bug) and mechanisms: extinction bridging, which is correct on most footage and fails
the face-vs-presence ceiling baked into X-Ray's semantics — plus a nonzero specifically at hard cuts into long faceless stretches, and the
slice where the pipeline is right and the ground truth is wrong. The held-out face-versus-presence ceiling, where X-Ray credits scene membership for
generalization gap (75.3% → 67.4%) is real and should be treated as the honest people whose faces never appear on screen. Of the nine distinct
expected performance, not the training-set number.</p> out-of-cast identifications found across the benchmark, two trace to gaps in
X-Ray's own cast data, one is a photograph misread as a person, and six are
genuine lookalike confusions on real faces. The held-out generalization gap,
75.3% training to 67.4% held-out, is real and should be treated as the
expected operating point, not the training-set figure.</p>
@@ -1141,13 +1426,13 @@ expected performance, not the training-set number.</p>
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--next" aria-label="Next: Model Bake-off &amp; Re-tune (full log)"> <a href="../model-bakeoff/" class="md-footer__link md-footer__link--next" aria-label="Next: Full Experiment Log">
<div class="md-footer__title"> <div class="md-footer__title">
<span class="md-footer__direction"> <span class="md-footer__direction">
Next Next
</span> </span>
<div class="md-ellipsis"> <div class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
</div> </div>
</div> </div>
<div class="md-footer__button md-icon"> <div class="md-footer__button md-icon">
@@ -10,13 +10,13 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/optimizer-experiments/"> <link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/methodology/">
<link rel="prev" href="../model-bakeoff/"> <link rel="prev" href="..">
<link rel="next" href="../service-conversion/"> <link rel="next" href="../best-model/">
@@ -27,7 +27,7 @@
<title>Optimizer Experiments (prior round) - scene-actor-extraction</title> <title>How We Score Against X-Ray - scene-actor-extraction</title>
@@ -80,7 +80,7 @@
<div data-md-component="skip"> <div data-md-component="skip">
<a href="#threshold-optimization-against-amazon-x-ray-experiment-log" class="md-skip"> <a href="#how-we-score-against-x-ray" class="md-skip">
Skip to content Skip to content
</a> </a>
@@ -114,7 +114,7 @@
<div class="md-header__topic" data-md-component="header-topic"> <div class="md-header__topic" data-md-component="header-topic">
<span class="md-ellipsis"> <span class="md-ellipsis">
Optimizer Experiments (prior round) How We Score Against X-Ray
</span> </span>
</div> </div>
@@ -242,6 +242,27 @@
<li class="md-tabs__item md-tabs__item--active">
<a href="./" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -270,28 +291,7 @@
Model Bake-off & Re-tune (full log) Full Experiment Log
</a>
</li>
<li class="md-tabs__item md-tabs__item--active">
<a href="./" class="md-tabs__link">
Optimizer Experiments (prior round)
</a> </a>
</li> </li>
@@ -393,6 +393,135 @@
<li class="md-nav__item md-nav__item--active">
<input class="md-nav__toggle md-toggle" type="checkbox" id="__toc">
<label class="md-nav__link md-nav__link--active" for="__toc">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
<span class="md-nav__icon md-icon"></span>
</label>
<a href="./" class="md-nav__link md-nav__link--active">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
<nav class="md-nav md-nav--secondary" aria-label="Table of contents">
<label class="md-nav__title" for="__toc">
<span class="md-nav__icon md-icon"></span>
Table of contents
</label>
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#what-amazon-x-ray-records" class="md-nav__link">
<span class="md-ellipsis">
What Amazon X-Ray records
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#why-an-offscreen-name-can-be-scored-correct" class="md-nav__link">
<span class="md-ellipsis">
Why an offscreen name can be scored correct
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#what-this-resolves-and-what-it-does-not" class="md-nav__link">
<span class="md-ellipsis">
What this resolves and what it does not
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#precision-recall-and-the-misid-weighting" class="md-nav__link">
<span class="md-ellipsis">
Precision, recall, and the misID weighting
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#reproduce" class="md-nav__link">
<span class="md-ellipsis">
Reproduce
</span>
</a>
</li>
</ul>
</nav>
</li>
@@ -409,10 +538,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" > <input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="0"> <label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
@@ -430,8 +559,8 @@
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
</label> </label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="false"> <nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_2"> <label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
@@ -574,7 +703,7 @@
<span class="md-ellipsis"> <span class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
@@ -591,157 +720,6 @@
<li class="md-nav__item md-nav__item--active">
<input class="md-nav__toggle md-toggle" type="checkbox" id="__toc">
<label class="md-nav__link md-nav__link--active" for="__toc">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
</span>
<span class="md-nav__icon md-icon"></span>
</label>
<a href="./" class="md-nav__link md-nav__link--active">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
</span>
</a>
<nav class="md-nav md-nav--secondary" aria-label="Table of contents">
<label class="md-nav__title" for="__toc">
<span class="md-nav__icon md-icon"></span>
Table of contents
</label>
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#tldr-what-changed" class="md-nav__link">
<span class="md-ellipsis">
TL;DR — what changed
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#ground-truth" class="md-nav__link">
<span class="md-ellipsis">
Ground truth
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#the-scoring-metric-evolved-through-review" class="md-nav__link">
<span class="md-ellipsis">
The scoring metric (evolved through review)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#the-gallery-coverage-gap" class="md-nav__link">
<span class="md-ellipsis">
The gallery coverage gap
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#optimizer" class="md-nav__link">
<span class="md-ellipsis">
Optimizer
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#replay-architecture-how-the-sweep-is-cheap" class="md-nav__link">
<span class="md-ellipsis">
Replay architecture (how the sweep is cheap)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#reproduce" class="md-nav__link">
<span class="md-ellipsis">
Reproduce
</span>
</a>
</li>
</ul>
</nav>
</li>
<li class="md-nav__item"> <li class="md-nav__item">
@@ -792,10 +770,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#tldr-what-changed" class="md-nav__link"> <a href="#what-amazon-x-ray-records" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
TL;DR — what changed What Amazon X-Ray records
</span> </span>
</a> </a>
@@ -803,10 +781,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#ground-truth" class="md-nav__link"> <a href="#why-an-offscreen-name-can-be-scored-correct" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Ground truth Why an offscreen name can be scored correct
</span> </span>
</a> </a>
@@ -814,10 +792,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#the-scoring-metric-evolved-through-review" class="md-nav__link"> <a href="#what-this-resolves-and-what-it-does-not" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
The scoring metric (evolved through review) What this resolves and what it does not
</span> </span>
</a> </a>
@@ -825,32 +803,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#the-gallery-coverage-gap" class="md-nav__link"> <a href="#precision-recall-and-the-misid-weighting" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
The gallery coverage gap Precision, recall, and the misID weighting
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#optimizer" class="md-nav__link">
<span class="md-ellipsis">
Optimizer
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#replay-architecture-how-the-sweep-is-cheap" class="md-nav__link">
<span class="md-ellipsis">
Replay architecture (how the sweep is cheap)
</span> </span>
</a> </a>
@@ -885,161 +841,115 @@
<h1 id="threshold-optimization-against-amazon-x-ray-experiment-log">Threshold optimization against Amazon X-Ray — experiment log<a class="headerlink" href="#threshold-optimization-against-amazon-x-ray-experiment-log" title="Permanent link">&para;</a></h1> <h1 id="how-we-score-against-x-ray">How we score against X-Ray<a class="headerlink" href="#how-we-score-against-x-ray" title="Permanent link">&para;</a></h1>
<p>Record of the July 2026 work that tuned the pipeline's recognition/tracking defaults <p>Every number in this report, every F1 and misID count, comes from one
against ground-truth per-scene actor presence, and the tooling built to do it.</p> comparison. The comparison has a mismatch at its core that shapes nearly
<h2 id="tldr-what-changed">TL;DR — what changed<a class="headerlink" href="#tldr-what-changed" title="Permanent link">&para;</a></h2> every finding in this report: the ground truth is scene-level, the
<table> pipeline's output is per-second, and the two do not mean the same thing.
<thead> This page documents that comparison once, so the findings pages can rely on
<tr> it without re-explaining it.</p>
<th>knob</th> <h2 id="what-amazon-x-ray-records">What Amazon X-Ray records<a class="headerlink" href="#what-amazon-x-ray-records" title="Permanent link">&para;</a></h2>
<th>old default</th> <p>X-Ray ships three tables per film: <code>scenes.csv</code> (a list of <code>[start, end]</code>
<th>new default</th> timespans), <code>people_in_scenes.csv</code> (which actors are credited in each
<th>why</th> scene), and <code>people.csv</code> (actor identities). There is no per-frame or
</tr> per-second annotation anywhere in X-Ray. A scene might run 45 seconds, and
</thead> X-Ray records one cast list for the entire span, not "on screen from
<tbody> second 12 to second 30."</p>
<tr> <p>To compare this against per-second predictions, <code>second_score.py</code> expands
<td><code>prob_threshold</code></td> every scene into per-second ground truth by copying the whole scene's cast
<td>0.99</td> list onto every second inside it:</p>
<td><strong>0.76</strong></td> <div class="language-python highlight"><pre><span></span><code><span id="__span-0-1"><a id="__codelineno-0-1" name="__codelineno-0-1" href="#__codelineno-0-1"></a><span class="k">for</span> <span class="n">sn</span><span class="p">,</span> <span class="p">(</span><span class="n">t0</span><span class="p">,</span> <span class="n">t1</span><span class="p">)</span> <span class="ow">in</span> <span class="n">spans</span><span class="o">.</span><span class="n">items</span><span class="p">():</span>
<td>0.99 was far too strict — halved recall for a fraction of a precision point. DE optimum, tightly converged.</td> </span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a> <span class="n">cast</span> <span class="o">=</span> <span class="n">scene_cast</span><span class="o">.</span><span class="n">get</span><span class="p">(</span><span class="n">sn</span><span class="p">,</span> <span class="p">[])</span>
</tr> </span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a> <span class="k">for</span> <span class="n">t</span> <span class="ow">in</span> <span class="nb">range</span><span class="p">(</span><span class="nb">int</span><span class="p">(</span><span class="n">t0</span><span class="p">),</span> <span class="nb">int</span><span class="p">(</span><span class="n">t1</span><span class="p">)):</span>
<tr> </span><span id="__span-0-4"><a id="__codelineno-0-4" name="__codelineno-0-4" href="#__codelineno-0-4"></a> <span class="n">timeline</span><span class="p">[</span><span class="n">t</span><span class="p">]</span> <span class="o">=</span> <span class="n">cast</span>
<td><code>extinction_sec</code></td>
<td>5.0</td>
<td><strong>1.5</strong></td>
<td>Long extinction smears presence into later scenes → FPs. DE converged tightly low.</td>
</tr>
<tr>
<td><code>anneal_sec</code></td>
<td>10.0</td>
<td>10.0 (unchanged)</td>
<td>DE found it <strong>insensitive</strong> (F1 flat ±0.3pp across 326s) — kept the round default.</td>
</tr>
<tr>
<td><code>detector_conf</code></td>
<td>0.5</td>
<td>0.5 (unchanged)</td>
<td>Sweep showed raising it only trades recall for precision at a net F1 loss — near-threshold detections are real faces, not phantoms.</td>
</tr>
</tbody>
</table>
<p>Net effect on the 9-film benchmark (strict per-scene, augmented gallery):
recall <strong>58% → ~72%</strong>, F1 <strong>70% → ~76%</strong>, precision ~85%, at no meaningful precision cost.</p>
<h2 id="ground-truth">Ground truth<a class="headerlink" href="#ground-truth" title="Permanent link">&para;</a></h2>
<p>Public scene-level <strong>Amazon X-Ray</strong> dataset (Zenodo DOI 10.5281/zenodo.17659734,
CC-BY-4.0): per movie, <code>people.csv</code> (name_id/person/character), <code>scenes.csv</code>
(scene/start/end ms), <code>people_in_scenes.csv</code>. Films matched to the library by an
<strong>authoritative Jellyfin ID join</strong> (query <code>/Items?IncludeItemTypes=Movie&amp;Fields=
ProviderIds,Path</code>, join Imdb/Tmdb against X-Ray metadata) — NOT fuzzy title matching,
which collides badly (TV episodes vs same-named films). 9 genuine films with source
video on disk: Benny &amp; Joon, Café Society, Downton Abbey: A New Era, Lord of War,
Lovelace, The Many Saints of Newark, Scarface, Sound of Metal, Valerian.</p>
<h2 id="the-scoring-metric-evolved-through-review">The scoring metric (evolved through review)<a class="headerlink" href="#the-scoring-metric-evolved-through-review" title="Permanent link">&para;</a></h2>
<p>Comparison unit is the <strong>X-Ray scene</strong>, not sampled timepoints. For each scene
<code>[start,end]</code>: predicted set = <strong>union</strong> of actors detected anywhere in the span;
GT set = actors X-Ray lists for that scene. Per scene TP/FP/FN, then:</p>
<ul>
<li><strong>Precision: STRICT.</strong> Any predicted actor not in the scene's X-Ray set is an FP,
<em>including out-of-cast confusions</em> (no gallery∩cast masking). An earlier
timepoint-sampled, cast-masked metric HID ~570 such FPs across 9 films and let the
optimizer drive <code>prob_threshold</code> to the 0.50 floor — a metric artifact. Counting
them is essential.</li>
<li><strong>Recall: FAIR.</strong> FN counts only X-Ray cast members <strong>who are in the gallery</strong>. 67%
of X-Ray cast (261/392) have no gallery reference embedding and can never be
recognised — counting them as misses penalises coverage, not the threshold. Both
<code>recall</code> (fair) and <code>recall_strict</code> (all) are reported.</li>
<li><strong>Aggregation:</strong> per-scene F1 → <strong>duration-weighted average within a movie</strong> (long
scenes count more) → <strong>equal-weight mean across movies</strong> (macro; each film counts
the same regardless of length). This is the DE objective.</li>
</ul>
<p>Implemented in <code>scripts/optimizer/scene_score.py</code> — since <strong>removed</strong> along
with this metric; its per-second successor is
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>
(see the <a href="../model-bakeoff/">bake-off round</a>).</p>
<h2 id="the-gallery-coverage-gap">The gallery coverage gap<a class="headerlink" href="#the-gallery-coverage-gap" title="Permanent link">&para;</a></h2>
<p>Diagnosing low recall: only <strong>131 of 392</strong> X-Ray cast were in the gallery (33%). Every
in-gallery actor HAD embeddings (gallery well-formed) — the gap was pure coverage.
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/fetch_missing_actors.py"><code>scripts/optimizer/fetch_missing_actors.py</code></a>
recovers missing actors:
<code>nm-id → TMDB /find external_ids → /person/{id}/images → download → embed (sae_embed)</code>,
with a <code>--wikidata</code> fallback (P345→P18 Commons photo).</p>
<ul>
<li><strong>TMDB recovered 143/261</strong> (55%). 0 face-detection failures; the rest had no TMDB
person (60) or no profile photo (58). Coverage 33% → <strong>70%</strong>.</li>
<li><strong>Wikidata fallback: 0/118</strong> of the TMDB failures — only 4 even had a Commons photo,
none yielded a detectable face. → <strong>TheTVDB not worth pursuing</strong>: these remaining
actors are obscure enough that no image source covers them, AND (see below) most are
off-camera anyway.</li>
</ul>
<p><strong>Coverage vs detectability.</strong> Adding references lifted recall (58→68% at fixed config)
but modestly. Per-film drill-down (Lord of War: 12 actors recovered, only 1 had a
detectable on-camera face) showed most missing cast are a <strong>detectability gap</strong> — X-Ray
credits them as cast-in-scene (incl. off-camera/background), but their face never
appears clearly for the pipeline to detect. This is a fundamental ceiling of a
face-recognition pipeline vs X-Ray's presence semantics, not a fixable gap.</p>
<h2 id="optimizer">Optimizer<a class="headerlink" href="#optimizer" title="Permanent link">&para;</a></h2>
<p><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/optimize.py"><code>scripts/optimizer/optimize.py</code></a>
— scipy <code>differential_evolution</code> over the knob space,
each candidate = full replay of all films through the <strong>real</strong> C++ nodes (see the
KPN replay architecture below) scored by the metric above. Global objective (one
config for all films, not per-film).</p>
<p><strong>Convergence stability (augmented gallery, 233 evals):</strong></p>
<table>
<thead>
<tr>
<th>knob</th>
<th>top-20 range</th>
<th>verdict</th>
</tr>
</thead>
<tbody>
<tr>
<td><code>prob_threshold</code></td>
<td>0.690.83 (σ 0.05)</td>
<td>TIGHT — trust 0.76</td>
</tr>
<tr>
<td><code>extinction_sec</code></td>
<td>1.02.2 (σ 0.33)</td>
<td>TIGHT — trust 1.5</td>
</tr>
<tr>
<td><code>anneal_sec</code></td>
<td>3.126.3 (σ 6.4)</td>
<td>LOOSE — insensitive, not hard-coded</td>
</tr>
</tbody>
</table>
<p>F1 varied only 0.3pp across the top-20 → objective is flat near the optimum, so only
the tightly-converged knobs were adopted as defaults.</p>
<h2 id="replay-architecture-how-the-sweep-is-cheap">Replay architecture (how the sweep is cheap)<a class="headerlink" href="#replay-architecture-how-the-sweep-is-cheap" title="Permanent link">&para;</a></h2>
<p>The optimizer never re-decodes video. <code>scene_analyze --dump-embeddings out.h5</code> runs the
expensive half once (decode→detect→align→embed) and dumps per-frame face embeddings
+ metadata to HDF5 (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/SCHEMA.md"><code>scripts/optimizer/SCHEMA.md</code></a>).
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/replay.py"><code>scripts/optimizer/replay.py</code></a> then
replays that dump through the <strong>real</strong> C++ <code>face_tracker → identity_matcher →
scene_tracker</code> assembled in a Python KPN network (<code>sae_kpn</code> nanobind module), varying
Config knobs freely — no GPU embedding, no decode. Verified BYTE-EXACT against
<code>scene_analyze</code>'s own output. The dumps are gallery-independent, so testing the
augmented gallery needed no re-dump. <code>detector_conf</code> is replayable UPWARD only (the
dump floor is 0.5).</p>
<h2 id="reproduce">Reproduce<a class="headerlink" href="#reproduce" title="Permanent link">&para;</a></h2>
<div class="language-bash highlight"><pre><span></span><code><span id="__span-0-1"><a id="__codelineno-0-1" name="__codelineno-0-1" href="#__codelineno-0-1"></a><span class="c1"># 1. dump (once per film, needs video)</span>
</span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a>scene_analyze<span class="w"> </span>--movie<span class="w"> </span>&lt;f&gt;<span class="w"> </span>--gallery<span class="w"> </span>gallery.json<span class="w"> </span>--dump-embeddings<span class="w"> </span>dump.h5<span class="w"> </span>--fps<span class="w"> </span><span class="m">1</span>
</span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a><span class="c1"># 2. build films manifest by Jellyfin ID join (see scripts/optimizer notes)</span>
</span><span id="__span-0-4"><a id="__codelineno-0-4" name="__codelineno-0-4" href="#__codelineno-0-4"></a><span class="c1"># 3. optimize</span>
</span><span id="__span-0-5"><a id="__codelineno-0-5" name="__codelineno-0-5" href="#__codelineno-0-5"></a>python<span class="w"> </span>scripts/optimizer/optimize.py<span class="w"> </span>--manifest<span class="w"> </span>films.json<span class="w"> </span>--gallery<span class="w"> </span>gallery.json<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-6"><a id="__codelineno-0-6" name="__codelineno-0-6" href="#__codelineno-0-6"></a><span class="w"> </span>--params<span class="w"> </span>prob_threshold:0.5:0.999<span class="w"> </span>anneal_sec:1:30<span class="w"> </span>extinction_sec:1:15<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-7"><a id="__codelineno-0-7" name="__codelineno-0-7" href="#__codelineno-0-7"></a><span class="w"> </span>--popsize<span class="w"> </span><span class="m">8</span><span class="w"> </span>--maxiter<span class="w"> </span><span class="m">20</span><span class="w"> </span>--trajectory<span class="w"> </span>traj.jsonl<span class="w"> </span>--out<span class="w"> </span>opt.json
</span><span id="__span-0-8"><a id="__codelineno-0-8" name="__codelineno-0-8" href="#__codelineno-0-8"></a><span class="c1"># 4. score a fixed config / validate on a held-out set</span>
</span><span id="__span-0-9"><a id="__codelineno-0-9" name="__codelineno-0-9" href="#__codelineno-0-9"></a><span class="c1"># (historical: score_config.py and scene_score.py were removed with the</span>
</span><span id="__span-0-10"><a id="__codelineno-0-10" name="__codelineno-0-10" href="#__codelineno-0-10"></a><span class="c1"># scene-union metric — use scripts/optimizer/second_score.py, per-second)</span>
</span><span id="__span-0-11"><a id="__codelineno-0-11" name="__codelineno-0-11" href="#__codelineno-0-11"></a>python<span class="w"> </span>scripts/optimizer/second_score.py<span class="w"> </span>--help
</span></code></pre></div> </span></code></pre></div>
<p>Superseded by the <a href="../model-bakeoff/">model bake-off + re-tune</a>, which <p>That is the entire mechanism. If X-Ray credits five actors to a 30-second
replaced this round's scene-union metric with per-second scoring.</p> scene, all five count as ground truth present for all 30 seconds, including
seconds where only one of them is on screen. This is not a simplification
introduced by the pipeline; it is the only reading of X-Ray's data that is
possible, because X-Ray itself does not record anything finer-grained.</p>
<h2 id="why-an-offscreen-name-can-be-scored-correct">Why an offscreen name can be scored correct<a class="headerlink" href="#why-an-offscreen-name-can-be-scored-correct" title="Permanent link">&para;</a></h2>
<p>A name listed under Offscreen with a correct (green) label is not the
pipeline guessing or padding its score. It is the pipeline correctly
answering the question X-Ray actually asks: is this actor part of this
scene. It answers that question using a presence window (<code>[start, end]</code>,
held open across cuts by <code>anneal_sec</code> and <code>extinction_sec</code>), which matches
X-Ray's scene-level semantics more closely than a raw per-frame detection
would.</p>
<p>A system that only reported "this actor is visible in this exact frame"
would score worse against X-Ray's scene-level ground truth, producing a
false negative every time the camera cuts away from a character who is
still present in the scene. Not because it is wrong about the world, but
because it would be answering a stricter, different question than the one
X-Ray's data supports. The presence-window design exists specifically to
answer X-Ray's actual question.</p>
<h2 id="what-this-resolves-and-what-it-does-not">What this resolves and what it does not<a class="headerlink" href="#what-this-resolves-and-what-it-does-not" title="Permanent link">&para;</a></h2>
<p>This resolves the semantic mismatch between a scene and an instant. It does
not resolve two other limitations, both discussed in the
<a href="../lvface-deep-dive/">LVFace deep dive</a>.</p>
<p><strong>The face-vs-presence ceiling.</strong> X-Ray credits scene membership regardless
of whether a face is ever visible: background crew, characters shot from
behind, voice-only presence. No amount of bridging recovers a face that
never appears on screen. This is a hard ceiling on recall, not a defect.</p>
<p><strong>Extinction bridging can overshoot.</strong> The same presence-window mechanism
that correctly answers "still in this scene" during a normal cut can also
bridge across a scene boundary it has no way to detect. A hard cut into a
different scene with no faces, such as closing credits, carries the
previous scene's identities forward until the window expires. This is the
mechanism behind Downton Abbey's recall collapse, documented in the deep
dive.</p>
<h2 id="precision-recall-and-the-misid-weighting">Precision, recall, and the misID weighting<a class="headerlink" href="#precision-recall-and-the-misid-weighting" title="Permanent link">&para;</a></h2>
<p>Per sampled second <code>t</code>:</p>
<p><strong>TPI</strong> (true positive instances): actors both X-Ray and the pipeline agree
are present.</p>
<p><strong>FPI</strong> (false positive instances): actors the pipeline reports that are
not in X-Ray's cast for this second. Split into two categories:</p>
<ul>
<li><strong>FPI_incast</strong>: the actor is in the film's cast, just not credited to
this particular scene. A timing or boundary slip.</li>
<li><strong>FPI_misid</strong>: the actor is not in the film's cast at all. A genuine
wrong-identity error, weighted 10x in the precision objective, because
naming someone who is not even in the film is a categorically worse
error than a few seconds of scene-boundary slop.</li>
</ul>
<div class="admonition note">
<p class="admonition-title">Every headline <code>P</code> and <code>F1</code> is misID-weighted</p>
<p>The precision reported throughout this report, and therefore the F1
derived from it, puts each <code>FPI_misid</code> into the denominator <strong>10 times</strong>
(<code>precision = TPI / (TPI + FPI_incast + 10·FPI_misid)</code>,
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/optimizer/second_score.py"><code>second_score.py</code></a>).
This is deliberate: the whole point is to punish naming an out-of-film
actor far harder than a scene-boundary slip. But it means the <code>P</code> column
is not raw precision, and a misID-heavy film's <code>P</code> is depressed
super-linearly. <code>second_score.py</code> also emits an unweighted <code>precision_raw</code>
(always ≥ the weighted <code>P</code>); where the gap matters, The Many Saints of
Newark, weighted <code>P</code> 54.7% vs. raw 78.4%, the <a href="../lvface-deep-dive/">LVFace deep dive</a>
reports both. When comparing <code>P</code> across films, remember you are comparing a
quantity that penalizes misIDs, not just a hit rate.</p>
</div>
<p><strong>FN</strong> (false negatives): actors X-Ray lists that the pipeline never
reports, counted only for actors who have a gallery reference embedding.
Across the 9-film benchmark, coverage of X-Ray's credited cast ranges from
20% to 79% by film (see
<a href="../model-bakeoff/#gallery-coverage-per-film">the full experiment log</a>); an
actor with no reference photo can never be recognized regardless of model
quality, and counting them as a miss would penalize gallery coverage, not
recognition accuracy.</p>
<p>Two further numbers are reported alongside F1:</p>
<p><strong>agreement_rate</strong>: mean per-second Jaccard overlap
(<code>|Pred ∩ GT| / |Pred GT|</code>), partial credit. Naming 2 of 3 present actors
scores 2/3, not 0.</p>
<p><strong>exact_match_rate</strong>: the fraction of sampled seconds where the pipeline's
named set exactly equals X-Ray's, no partial credit. Far harsher, and
dominated by recall, since any single missed actor zeroes that second.</p>
<h2 id="reproduce">Reproduce<a class="headerlink" href="#reproduce" title="Permanent link">&para;</a></h2>
<div class="language-bash highlight"><pre><span></span><code><span id="__span-1-1"><a id="__codelineno-1-1" name="__codelineno-1-1" href="#__codelineno-1-1"></a>python3<span class="w"> </span>scripts/optimizer/second_score.py<span class="w"> </span><span class="se">\</span>
</span><span id="__span-1-2"><a id="__codelineno-1-2" name="__codelineno-1-2" href="#__codelineno-1-2"></a><span class="w"> </span>--pred<span class="w"> </span>pred.json<span class="w"> </span>--xray<span class="w"> </span>experiments/xray/.../&lt;xray_dir&gt;<span class="w"> </span><span class="se">\</span>
</span><span id="__span-1-3"><a id="__codelineno-1-3" name="__codelineno-1-3" href="#__codelineno-1-3"></a><span class="w"> </span>--gallery<span class="w"> </span>experiments/galleries/gallery_LVFace-B_Glint360K.h5
</span></code></pre></div>
<p>See also <a href="../model-bakeoff/">the full experiment log</a> for how <code>pred.json</code> is
produced, and the <a href="../lvface-deep-dive/">LVFace deep dive</a> for what these
mechanisms look like frame by frame.</p>
@@ -1075,7 +985,7 @@ replaced this round's scene-union metric with per-second scoring.</p>
<nav class="md-footer__inner md-grid" aria-label="Footer" > <nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Model Bake-off &amp; Re-tune (full log)"> <a href=".." class="md-footer__link md-footer__link--prev" aria-label="Previous: Home">
<div class="md-footer__button md-icon"> <div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg> <svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
@@ -1085,20 +995,20 @@ replaced this round's scene-union metric with per-second scoring.</p>
Previous Previous
</span> </span>
<div class="md-ellipsis"> <div class="md-ellipsis">
Model Bake-off & Re-tune (full log) Home
</div> </div>
</div> </div>
</a> </a>
<a href="../service-conversion/" class="md-footer__link md-footer__link--next" aria-label="Next: Service Conversion (proposal)"> <a href="../best-model/" class="md-footer__link md-footer__link--next" aria-label="Next: Best Model">
<div class="md-footer__title"> <div class="md-footer__title">
<span class="md-footer__direction"> <span class="md-footer__direction">
Next Next
</span> </span>
<div class="md-ellipsis"> <div class="md-ellipsis">
Service Conversion (proposal) Best Model
</div> </div>
</div> </div>
<div class="md-footer__button md-icon"> <div class="md-footer__button md-icon">
+549 -708
View File
File diff suppressed because it is too large Load Diff
+164 -154
View File
@@ -80,7 +80,7 @@
<div data-md-component="skip"> <div data-md-component="skip">
<a href="#pose-expansion-does-learning-new-poses-mid-film-help" class="md-skip"> <a href="#pose-expansion-does-promoting-new-poses-mid-film-help" class="md-skip">
Skip to content Skip to content
</a> </a>
@@ -242,6 +242,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
@@ -272,26 +291,7 @@
Model Bake-off & Re-tune (full log) Full Experiment Log
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
</a> </a>
</li> </li>
@@ -393,6 +393,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -414,10 +441,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked> <input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex=""> <label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="">
@@ -435,8 +462,8 @@
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
</label> </label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true"> <nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2"> <label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
@@ -569,10 +596,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#the-training-set-signal" class="md-nav__link"> <a href="#training-set-signal" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
The training-set signal Training-set signal
</span> </span>
</a> </a>
@@ -580,10 +607,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#held-out-test-does-it-reproduce" class="md-nav__link"> <a href="#held-out-test" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Held-out test: does it reproduce? Held-out test
</span> </span>
</a> </a>
@@ -591,10 +618,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#two-bugs-this-required-catching-this-sections-own-methodology" class="md-nav__link"> <a href="#two-methodology-bugs-caught-during-this-check" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Two bugs this required catching (this section's own methodology) Two methodology bugs caught during this check
</span> </span>
</a> </a>
@@ -602,10 +629,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#what-this-means" class="md-nav__link"> <a href="#conclusion" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
What this means Conclusion
</span> </span>
</a> </a>
@@ -670,34 +697,7 @@
<span class="md-ellipsis"> <span class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
@@ -764,10 +764,10 @@
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix> <ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#the-training-set-signal" class="md-nav__link"> <a href="#training-set-signal" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
The training-set signal Training-set signal
</span> </span>
</a> </a>
@@ -775,10 +775,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#held-out-test-does-it-reproduce" class="md-nav__link"> <a href="#held-out-test" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Held-out test: does it reproduce? Held-out test
</span> </span>
</a> </a>
@@ -786,10 +786,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#two-bugs-this-required-catching-this-sections-own-methodology" class="md-nav__link"> <a href="#two-methodology-bugs-caught-during-this-check" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
Two bugs this required catching (this section's own methodology) Two methodology bugs caught during this check
</span> </span>
</a> </a>
@@ -797,10 +797,10 @@
</li> </li>
<li class="md-nav__item"> <li class="md-nav__item">
<a href="#what-this-means" class="md-nav__link"> <a href="#conclusion" class="md-nav__link">
<span class="md-ellipsis"> <span class="md-ellipsis">
What this means Conclusion
</span> </span>
</a> </a>
@@ -824,16 +824,20 @@
<h1 id="pose-expansion-does-learning-new-poses-mid-film-help">Pose expansion: does "learning" new poses mid-film help?<a class="headerlink" href="#pose-expansion-does-learning-new-poses-mid-film-help" title="Permanent link">&para;</a></h1> <h1 id="pose-expansion-does-promoting-new-poses-mid-film-help">Pose expansion: does promoting new poses mid-film help?<a class="headerlink" href="#pose-expansion-does-promoting-new-poses-mid-film-help" title="Permanent link">&para;</a></h1>
<p><code>expand_gallery</code> (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/track_gallery.hpp"><code>src/gallery/track_gallery.hpp</code></a>) <p><code>expand_gallery</code>
promotes a confidently-identified (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/gallery/track_gallery.hpp"><code>src/gallery/track_gallery.hpp</code></a>)
track's novel-pose reference views into a per-film, in-memory gallery annex — the promotes a confidently identified track's novel-pose reference views into a
idea being that once the pipeline is sure who someone is, a pose it hasn't seen per-film, in-memory gallery annex. The idea: once the pipeline is confident
before (turned head, different lighting) becomes a free extra reference for about an identity, a pose it has not seen before (turned head, different
recognising that actor again later in the same film, without touching the baked lighting) becomes an extra reference for recognizing that actor again later
gallery.</p> in the same film, without touching the baked gallery.</p>
<h2 id="the-training-set-signal">The training-set signal<a class="headerlink" href="#the-training-set-signal" title="Permanent link">&para;</a></h2> <h2 id="training-set-signal">Training-set signal<a class="headerlink" href="#training-set-signal" title="Permanent link">&para;</a></h2>
<p>Averaged across all 4 models, on the 4 films used for optimization:</p> <p>Averaged across the 3 compared models (r50 excluded), on the 4 films used
for optimization. These are the corrected, full-coverage figures, see the
<a href="../model-bakeoff/#a-scoring-bug-worth-recording-dropped-film-evaluations">dropped-film note</a>
in the experiment log for why an earlier version of this table overstated the
full-mode misID jump (209 → 864) that was itself partly a truncation artifact:</p>
<table> <table>
<thead> <thead>
<tr> <tr>
@@ -848,54 +852,54 @@ gallery.</p>
<tr> <tr>
<td>full</td> <td>full</td>
<td>off</td> <td>off</td>
<td>71.2%</td> <td>70.0%</td>
<td>58.3%</td> <td>57.6%</td>
<td>209</td> <td>407</td>
</tr> </tr>
<tr> <tr>
<td>full</td> <td>full</td>
<td><strong>on</strong></td> <td>on</td>
<td>71.2%</td> <td>72.1%</td>
<td>59.7%</td> <td>61.5%</td>
<td><strong>864</strong></td> <td>714</td>
</tr> </tr>
<tr> <tr>
<td>restricted</td> <td>restricted</td>
<td>off</td> <td>off</td>
<td>73.6%</td> <td>75.1%</td>
<td>61.3%</td> <td>63.9%</td>
<td>194</td> <td>179</td>
</tr> </tr>
<tr> <tr>
<td>restricted</td> <td>restricted</td>
<td><strong>on</strong></td> <td>on</td>
<td><strong>75.4%</strong></td> <td>76.7%</td>
<td><strong>64.5%</strong></td> <td>67.2%</td>
<td>135</td> <td>120</td>
</tr> </tr>
</tbody> </tbody>
</table> </table>
<p>In <code>restricted</code> mode (matcher's candidate set capped to the film's own credited <p>In restricted mode, expansion looks like a clean win: +1.6pp F1, +3.3pp
cast) expansion looked like a clean win: +1.8pp F1, +3.2pp recall, misID actually recall, lower misID. In full mode it looks like a recall-for-misID trade:
lower. In <code>full</code> mode it looked flat-to-costly: ~0 F1 change, recall +1.4pp, but +2.1pp F1, +3.9pp recall, but misID rises from 407 to 714. See
misID roughly quadrupled (209 → 864) — see the <a href="../model-bakeoff/">the full experiment log</a> for the per-model breakdown.
<a href="../model-bakeoff/">bake-off experiment log</a> for the per-model breakdown. That's the number that motivated this page: <strong>does turning This asymmetry motivated the question below: does turning expansion on
expansion on actually change what gets recognised, frame by frame, or is the change what gets recognized frame by frame, or is the aggregate F1 shift
aggregate F1 shift something else?</strong></p> coming from something else.</p>
<h2 id="held-out-test-does-it-reproduce">Held-out test: does it reproduce?<a class="headerlink" href="#held-out-test-does-it-reproduce" title="Permanent link">&para;</a></h2> <h2 id="held-out-test">Held-out test<a class="headerlink" href="#held-out-test" title="Permanent link">&para;</a></h2>
<p>Same model + same tuned config, <code>expand_gallery</code> toggled on vs. off, nothing else <p>Same model, same tuned config, <code>expand_gallery</code> toggled on vs. off, nothing
changed full gallery mode, per-second scoring against X-Ray. This isolates else changed, full gallery mode, per-second scoring against X-Ray. This
expansion from every other variable (config, model, threshold) that differs isolates expansion from every other variable that differs between the
between the training-set <code>exp</code>/<code>noexp</code> rows above.</p> training-set rows above.</p>
<p><strong>LVFace-B Glint360K, all 5 held-out films</strong> (films never seen by the optimizer):</p> <p>LVFace-B Glint360K, all 5 held-out films:</p>
<table> <table>
<thead> <thead>
<tr> <tr>
<th>film</th> <th>film</th>
<th>F1 (exp)</th> <th>F1 (exp)</th>
<th>F1 (noexp)</th> <th>F1 (noexp)</th>
<th>TPI Δ</th> <th>TPI delta</th>
<th>FN Δ</th> <th>FN delta</th>
</tr> </tr>
</thead> </thead>
<tbody> <tbody>
@@ -936,54 +940,60 @@ between the training-set <code>exp</code>/<code>noexp</code> rows above.</p>
</tr> </tr>
</tbody> </tbody>
</table> </table>
<p><strong>ArcFace R18</strong> (Benny &amp; Joon, r18's own tuned config): F1 77.1% for both, TPI/FN <p>ArcFace R18, Benny &amp; Joon, r18's own tuned config: F1 77.1% for both, TPI
identical, FPI differs by 2 (noise).</p> and FN identical, FPI differs by 2.</p>
<p><strong>Every film, both models tested: F1 within 0.10.2pp, TPI/FN swings in the tens <p>Every film, both models tested: F1 differs by 0.1-0.2pp, TPI/FN swings are
out of tens of thousands.</strong> That's noise, not a signal — expansion made no in the tens out of tens of thousands. This is noise, not a signal.
measurable difference to per-second onscreen identification anywhere it was Expansion made no measurable difference to per-second on-screen
tested on unseen data.</p> identification on any held-out film tested.</p>
<h2 id="two-bugs-this-required-catching-this-sections-own-methodology">Two bugs this required catching (this section's own methodology)<a class="headerlink" href="#two-bugs-this-required-catching-this-sections-own-methodology" title="Permanent link">&para;</a></h2> <h2 id="two-methodology-bugs-caught-during-this-check">Two methodology bugs caught during this check<a class="headerlink" href="#two-methodology-bugs-caught-during-this-check" title="Permanent link">&para;</a></h2>
<p>Getting to the clean table above took two wrong turns, both worth recording <p>Getting to the table above required catching two wrong turns, both worth
since they're exactly the kind of error that produces a false positive "look, recording because they are exactly the kind of error that produces a false
expansion helped!" finding:</p> positive "expansion helped" finding.</p>
<ol> <ol>
<li><strong>Timeout truncation.</strong> The first Downton Abbey <code>exp</code> replay was cut off by a <li><strong>Timeout truncation.</strong> The first Downton Abbey <code>exp</code> replay was cut off
60s subprocess timeout at ~76% through the film (5589 of 7368 expected by a 60-second subprocess timeout at about 76% through the film (5589 of
seconds) — a genuinely large, silent data loss that showed up as a large, 7368 expected seconds). This silent data loss produced a large,
convincing-looking TPI gap (47938 vs 52032) purely because one run had a convincing-looking TPI gap (47938 vs 52032) purely because one run was
quarter of the film missing. Caught by comparing <code>n_seconds</code> between runs missing a quarter of the film. Caught by comparing <code>n_seconds</code> between
before trusting any score delta; fixed by re-running with a longer timeout.</li> runs before trusting any score delta; fixed by re-running with a longer
<li><strong>Bbox-matching bug.</strong> An early per-second raw-annotation diff matched each timeout.</li>
<code>exp</code> detection to the <em>first</em> <code>noexp</code> detection with IoU &gt; 0.5, not the <li><strong>Bbox-matching bug.</strong> An early per-second raw-annotation diff matched
<em>best</em>-overlapping one. With 3 faces close together in frame, this produced each <code>exp</code> detection to the first <code>noexp</code> detection with IoU above 0.5,
spurious "disagreements" (e.g. "exp says Aidan Quinn, noexp says Johnny not the best-overlapping one. With 3 faces close together in frame, this
Depp" at the same seconds) that vanished entirely once the match picked the produced spurious disagreements (for example "exp says Aidan Quinn,
true best-IoU candidate — both configs had actually output the exact same noexp says Johnny Depp" at the same second) that vanished once the match
three names at the exact same three boxes.</li> used the best-IoU candidate instead of the first one. Both configs had
actually output the same three names at the same three boxes.</li>
</ol> </ol>
<p>Both bugs independently pointed toward "expansion is doing something," and both <p>Both bugs independently pointed toward "expansion is doing something," and
were artifacts of the comparison harness, not the pipeline. Worth remembering both were artifacts of the comparison harness, not the pipeline. Before
when a before/after diff looks dramatic: check that the two runs actually cover trusting a dramatic before/after diff, check that both runs cover the same
the same seconds, and match entities by best overlap, not first-found.</p> seconds and that entities are matched by best overlap, not first found.</p>
<h2 id="what-this-means">What this means<a class="headerlink" href="#what-this-means" title="Permanent link">&para;</a></h2> <h2 id="conclusion">Conclusion<a class="headerlink" href="#conclusion" title="Permanent link">&para;</a></h2>
<p>The training-set aggregate effect (particularly the ~4x misID increase in full <p>The training-set aggregate effect, particularly the full-mode misID
mode) doesn't reproduce on held-out data — at minimum it's far smaller than the increase, does not reproduce on held-out data. At minimum it
training-set numbers suggested, and plausibly it's sampling variation from only is far smaller than the training-set numbers suggested; it may be sampling
4 training films rather than a real, generalizable mechanism. This doesn't mean variation from only 4 training films rather than a generalizable
<code>expand_gallery</code> never does anything (the mechanism is real — see mechanism. Note the same <em>class</em> of harness bug appears twice in this
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/track_gallery.hpp"><code>track_gallery.hpp</code></a>'s investigation, the timeout truncation in bug #1 above, and the dropped-film
promotion logging: tracks <em>do</em> get confirmed and views <em>do</em> aggregation that inflated the raw training-set misID figures. Both make an
get promoted into the annex on every film tested), only that <strong>whatever effect inert config look consequential; both are reasons to distrust a dramatic
it has on final per-second identification was too small to detect against 5 training-set delta until it survives on held-out films, which this one did
held-out films</strong> with this scoring method. A cleaner test would need either many not. This does not mean <code>expand_gallery</code> never does anything: the
more held-out films or a metric that can see the annex's direct contribution mechanism is real, and
(e.g. tagging which reference embedding won each match), neither of which this <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/src/gallery/track_gallery.hpp"><code>track_gallery.hpp</code></a>'s
pass had budget for.</p> promotion logging confirms tracks get confirmed and views get promoted
<p><strong>Practical takeaway</strong>: don't treat the training-set <code>exp</code> vs <code>noexp</code> numbers in into the annex on every film tested. It means whatever effect expansion
the <a href="../model-bakeoff/">bake-off experiment log</a> as proof that expansion has on final per-second identification was too small to detect against 5
changes real-world behavior held-out films with this scoring method. A cleaner test would need either
in either direction — on the evidence gathered so far, it doesn't move the more held-out films or a metric that can see the annex's direct
needle enough to see.</p> contribution, such as tagging which reference embedding won each match;
neither was in scope for this pass.</p>
<p>Do not treat the training-set exp/noexp numbers in
<a href="../model-bakeoff/">the full experiment log</a> as proof that expansion changes
real-world behavior in either direction. On the evidence gathered so far,
it does not move the needle enough to see.</p>
File diff suppressed because one or more lines are too long
+59 -59
View File
@@ -13,7 +13,7 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/service-conversion/"> <link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/service-conversion/">
<link rel="prev" href="../optimizer-experiments/"> <link rel="prev" href="../model-bakeoff/">
@@ -241,6 +241,25 @@
<li class="md-tabs__item">
<a href="../methodology/" class="md-tabs__link">
How We Score Against X-Ray
</a>
</li>
<li class="md-tabs__item"> <li class="md-tabs__item">
@@ -268,26 +287,7 @@
Model Bake-off & Re-tune (full log) Full Experiment Log
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
</a> </a>
</li> </li>
@@ -393,6 +393,33 @@
<li class="md-nav__item">
<a href="../methodology/" class="md-nav__link">
<span class="md-ellipsis">
How We Score Against X-Ray
</span>
</a>
</li>
@@ -407,10 +434,10 @@
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" > <input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="0"> <label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
@@ -428,8 +455,8 @@
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
</label> </label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="false"> <nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
<label class="md-nav__title" for="__nav_2"> <label class="md-nav__title" for="__nav_3">
<span class="md-nav__icon md-icon"></span> <span class="md-nav__icon md-icon"></span>
@@ -572,34 +599,7 @@
<span class="md-ellipsis"> <span class="md-ellipsis">
Model Bake-off & Re-tune (full log) Full Experiment Log
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
@@ -1035,7 +1035,7 @@ lock-gating, not new pipeline logic.</p>
</tr> </tr>
<tr> <tr>
<td>Backend selection</td> <td>Backend selection</td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/CMakeLists.txt"><code>CMakeLists.txt</code></a> (<code>SAE_INFERENCE_BACKEND</code>, <code>SAE_GEMM_BACKEND</code>)</td> <td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/CMakeLists.txt"><code>CMakeLists.txt</code></a> (<code>SAE_INFERENCE_BACKEND</code>, <code>SAE_GEMM_BACKEND</code>)</td>
<td>ORT/TRT + ROCm/CUDA, chosen <strong>at build time</strong></td> <td>ORT/TRT + ROCm/CUDA, chosen <strong>at build time</strong></td>
</tr> </tr>
<tr> <tr>
@@ -1045,7 +1045,7 @@ lock-gating, not new pipeline logic.</p>
</tr> </tr>
<tr> <tr>
<td>Worker loop</td> <td>Worker loop</td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a><code>--worker</code></td> <td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a><code>--worker</code></td>
<td>Poll Pending → run <code>scene_analyze</code> → push results</td> <td>Poll Pending → run <code>scene_analyze</code> → push results</td>
</tr> </tr>
<tr> <tr>
@@ -1055,12 +1055,12 @@ lock-gating, not new pipeline logic.</p>
</tr> </tr>
<tr> <tr>
<td>Incremental gallery</td> <td>Incremental gallery</td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/make_jellyfin_gallery.py"><code>scripts/make_jellyfin_gallery.py</code></a><code>--merge</code></td> <td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/make_jellyfin_gallery.py"><code>scripts/make_jellyfin_gallery.py</code></a><code>--merge</code></td>
<td>Embeds only cast not already in the gallery</td> <td>Embeds only cast not already in the gallery</td>
</tr> </tr>
<tr> <tr>
<td>Secrets loader</td> <td>Secrets loader</td>
<td><code>.env</code> via <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/sae_env.py"><code>scripts/sae_env.py</code></a></td> <td><code>.env</code> via <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/0bd27470698c45cab21935d636a04512360e1008/scripts/sae_env.py"><code>scripts/sae_env.py</code></a></td>
<td><code>JELLYFIN_URL</code>, <code>JELLYFIN_API_KEY</code>, <code>TMDB_API_KEY</code></td> <td><code>JELLYFIN_URL</code>, <code>JELLYFIN_API_KEY</code>, <code>TMDB_API_KEY</code></td>
</tr> </tr>
</tbody> </tbody>
@@ -1276,7 +1276,7 @@ same filesystem and GPU as everything else on the box.</p>
<nav class="md-footer__inner md-grid" aria-label="Footer" > <nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../optimizer-experiments/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Optimizer Experiments (prior round)"> <a href="../model-bakeoff/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Full Experiment Log">
<div class="md-footer__button md-icon"> <div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg> <svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
@@ -1286,7 +1286,7 @@ same filesystem and GPU as everything else on the box.</p>
Previous Previous
</span> </span>
<div class="md-ellipsis"> <div class="md-ellipsis">
Optimizer Experiments (prior round) Full Experiment Log
</div> </div>
</div> </div>
</a> </a>
+11 -11
View File
@@ -2,34 +2,34 @@
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"> <urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
<url> <url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/</loc> <loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/</loc>
<lastmod>2026-07-19</lastmod> <lastmod>2026-07-21</lastmod>
</url> </url>
<url> <url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/best-model/</loc> <loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/best-model/</loc>
<lastmod>2026-07-19</lastmod> <lastmod>2026-07-21</lastmod>
</url> </url>
<url> <url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/gallery-scope/</loc> <loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/gallery-scope/</loc>
<lastmod>2026-07-19</lastmod> <lastmod>2026-07-21</lastmod>
</url> </url>
<url> <url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/lvface-deep-dive/</loc> <loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/lvface-deep-dive/</loc>
<lastmod>2026-07-19</lastmod> <lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/methodology/</loc>
<lastmod>2026-07-21</lastmod>
</url> </url>
<url> <url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/model-bakeoff/</loc> <loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/model-bakeoff/</loc>
<lastmod>2026-07-19</lastmod> <lastmod>2026-07-21</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/optimizer-experiments/</loc>
<lastmod>2026-07-19</lastmod>
</url> </url>
<url> <url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/pose-expansion/</loc> <loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/pose-expansion/</loc>
<lastmod>2026-07-19</lastmod> <lastmod>2026-07-21</lastmod>
</url> </url>
<url> <url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/service-conversion/</loc> <loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/service-conversion/</loc>
<lastmod>2026-07-19</lastmod> <lastmod>2026-07-21</lastmod>
</url> </url>
</urlset> </urlset>
BIN
View File
Binary file not shown.