docs: deploy from 4b55579

This commit is contained in:
2026-07-19 22:28:14 +02:00
parent 252625330f
commit 36f75ba199
29 changed files with 1024 additions and 260 deletions
+43 -25
View File
@@ -16,7 +16,7 @@
<link rel="icon" href="/assets/images/favicon.png">
<link rel="icon" href="/dtourolle/scene-actor-extraction/assets/images/favicon.png">
<meta name="generator" content="mkdocs-1.6.1, mkdocs-material-9.7.7">
@@ -25,10 +25,10 @@
<link rel="stylesheet" href="/assets/stylesheets/main.ec1eaa64.min.css">
<link rel="stylesheet" href="/dtourolle/scene-actor-extraction/assets/stylesheets/main.ec1eaa64.min.css">
<link rel="stylesheet" href="/assets/stylesheets/palette.ab4e12ef.min.css">
<link rel="stylesheet" href="/dtourolle/scene-actor-extraction/assets/stylesheets/palette.ab4e12ef.min.css">
@@ -47,7 +47,9 @@
<script>__md_scope=new URL("/",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
<link rel="stylesheet" href="/dtourolle/scene-actor-extraction/stylesheets/extra.css">
<script>__md_scope=new URL("/dtourolle/scene-actor-extraction/",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -63,7 +65,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -81,7 +83,7 @@
<header class="md-header" data-md-component="header">
<nav class="md-header__inner md-grid" aria-label="Header">
<a href="/." title="scene-actor-extraction" class="md-header__button md-logo" aria-label="scene-actor-extraction" data-md-component="logo">
<a href="/dtourolle/scene-actor-extraction/." title="scene-actor-extraction" class="md-header__button md-logo" aria-label="scene-actor-extraction" data-md-component="logo">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 8a3 3 0 0 0 3-3 3 3 0 0 0-3-3 3 3 0 0 0-3 3 3 3 0 0 0 3 3m0 3.54C9.64 9.35 6.5 8 3 8v11c3.5 0 6.64 1.35 9 3.54 2.36-2.19 5.5-3.54 9-3.54V8c-3.5 0-6.64 1.35-9 3.54"/></svg>
@@ -114,7 +116,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -198,7 +214,7 @@
<li class="md-tabs__item">
<a href="/." class="md-tabs__link">
<a href="/dtourolle/scene-actor-extraction/." class="md-tabs__link">
@@ -219,7 +235,7 @@
<li class="md-tabs__item">
<a href="/best-model/" class="md-tabs__link">
<a href="/dtourolle/scene-actor-extraction/best-model/" class="md-tabs__link">
@@ -237,13 +253,13 @@
<li class="md-tabs__item">
<a href="/rep4-optimizer-results/" class="md-tabs__link">
<a href="/dtourolle/scene-actor-extraction/model-bakeoff/" class="md-tabs__link">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -256,7 +272,7 @@
<li class="md-tabs__item">
<a href="/optimizer-experiments/" class="md-tabs__link">
<a href="/dtourolle/scene-actor-extraction/optimizer-experiments/" class="md-tabs__link">
@@ -275,7 +291,7 @@
<li class="md-tabs__item">
<a href="/service-conversion/" class="md-tabs__link">
<a href="/dtourolle/scene-actor-extraction/service-conversion/" class="md-tabs__link">
@@ -310,7 +326,7 @@
<nav class="md-nav md-nav--primary md-nav--lifted" aria-label="Navigation" data-md-level="0">
<label class="md-nav__title" for="__drawer">
<a href="/." title="scene-actor-extraction" class="md-nav__button md-logo" aria-label="scene-actor-extraction" data-md-component="logo">
<a href="/dtourolle/scene-actor-extraction/." title="scene-actor-extraction" class="md-nav__button md-logo" aria-label="scene-actor-extraction" data-md-component="logo">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 8a3 3 0 0 0 3-3 3 3 0 0 0-3-3 3 3 0 0 0-3 3 3 3 0 0 0 3 3m0 3.54C9.64 9.35 6.5 8 3 8v11c3.5 0 6.64 1.35 9 3.54 2.36-2.19 5.5-3.54 9-3.54V8c-3.5 0-6.64 1.35-9 3.54"/></svg>
@@ -340,7 +356,7 @@
<li class="md-nav__item">
<a href="/." class="md-nav__link">
<a href="/dtourolle/scene-actor-extraction/." class="md-nav__link">
@@ -419,7 +435,7 @@
<li class="md-nav__item">
<a href="/best-model/" class="md-nav__link">
<a href="/dtourolle/scene-actor-extraction/best-model/" class="md-nav__link">
@@ -447,7 +463,7 @@
<li class="md-nav__item">
<a href="/gallery-scope/" class="md-nav__link">
<a href="/dtourolle/scene-actor-extraction/gallery-scope/" class="md-nav__link">
@@ -475,7 +491,7 @@
<li class="md-nav__item">
<a href="/pose-expansion/" class="md-nav__link">
<a href="/dtourolle/scene-actor-extraction/pose-expansion/" class="md-nav__link">
@@ -503,7 +519,7 @@
<li class="md-nav__item">
<a href="/lvface-deep-dive/" class="md-nav__link">
<a href="/dtourolle/scene-actor-extraction/lvface-deep-dive/" class="md-nav__link">
@@ -538,14 +554,14 @@
<li class="md-nav__item">
<a href="/rep4-optimizer-results/" class="md-nav__link">
<a href="/dtourolle/scene-actor-extraction/model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -565,7 +581,7 @@
<li class="md-nav__item">
<a href="/optimizer-experiments/" class="md-nav__link">
<a href="/dtourolle/scene-actor-extraction/optimizer-experiments/" class="md-nav__link">
@@ -592,7 +608,7 @@
<li class="md-nav__item">
<a href="/service-conversion/" class="md-nav__link">
<a href="/dtourolle/scene-actor-extraction/service-conversion/" class="md-nav__link">
@@ -660,6 +676,8 @@
<footer class="md-footer">
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -685,10 +703,10 @@
<script id="__config" type="application/json">{"annotate": null, "base": "/", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "/assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": "/dtourolle/scene-actor-extraction/", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "/dtourolle/scene-actor-extraction/assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="/assets/javascripts/bundle.d7400e89.min.js"></script>
<script src="/dtourolle/scene-actor-extraction/assets/javascripts/bundle.d7400e89.min.js"></script>
</body>
Binary file not shown.

After

Width:  |  Height:  |  Size: 200 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 137 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 137 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 219 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 164 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 68 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 201 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 83 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 239 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 234 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 163 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 174 KiB

After

Width:  |  Height:  |  Size: 118 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 141 KiB

+59
View File
@@ -0,0 +1,59 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 980 260" font-family="-apple-system,BlinkMacSystemFont,Segoe UI,sans-serif">
<style>
.box { fill:#1f6feb; stroke:#0d4f9e; stroke-width:1.5; }
.box-debug { fill:#57534e; stroke:#3a3733; stroke-width:1.5; }
.lbl { fill:#ffffff; font-size:13px; text-anchor:middle; dominant-baseline:middle; }
.arrow { stroke:#57534e; stroke-width:2; marker-end:url(#arrowhead); fill:none; }
.arrow-debug { stroke:#a8a29e; stroke-width:2; stroke-dasharray:4,3; marker-end:url(#arrowhead-debug); fill:none; }
.caption { fill:#57534e; font-size:11px; text-anchor:middle; }
</style>
<defs>
<marker id="arrowhead" markerWidth="8" markerHeight="8" refX="7" refY="4" orient="auto">
<path d="M0,0 L8,4 L0,8 Z" fill="#57534e"/>
</marker>
<marker id="arrowhead-debug" markerWidth="8" markerHeight="8" refX="7" refY="4" orient="auto">
<path d="M0,0 L8,4 L0,8 Z" fill="#a8a29e"/>
</marker>
</defs>
<!-- top row: main chain -->
<rect class="box" x="10" y="40" width="120" height="50" rx="6"/>
<text class="lbl" x="70" y="65">frame_source</text>
<rect class="box" x="160" y="40" width="120" height="50" rx="6"/>
<text class="lbl" x="220" y="65">face_detector</text>
<rect class="box" x="310" y="40" width="120" height="50" rx="6"/>
<text class="lbl" x="370" y="65">face_aligner</text>
<rect class="box" x="460" y="40" width="110" height="50" rx="6"/>
<text class="lbl" x="515" y="65">embedder</text>
<rect class="box" x="600" y="40" width="120" height="50" rx="6"/>
<text class="lbl" x="660" y="65">face_tracker</text>
<rect class="box" x="750" y="40" width="130" height="50" rx="6"/>
<text class="lbl" x="815" y="65">identity_matcher</text>
<line class="arrow" x1="130" y1="65" x2="158" y2="65"/>
<line class="arrow" x1="280" y1="65" x2="308" y2="65"/>
<line class="arrow" x1="430" y1="65" x2="458" y2="65"/>
<line class="arrow" x1="570" y1="65" x2="598" y2="65"/>
<line class="arrow" x1="720" y1="65" x2="748" y2="65"/>
<!-- second row: scene_tracker + result_sink -->
<line class="arrow" x1="815" y1="90" x2="815" y2="150"/>
<rect class="box" x="720" y="150" width="130" height="50" rx="6"/>
<text class="lbl" x="785" y="175">scene_tracker</text>
<line class="arrow" x1="718" y1="175" x2="582" y2="175"/>
<rect class="box" x="452" y="150" width="120" height="50" rx="6"/>
<text class="lbl" x="512" y="175">result_sink</text>
<!-- debug fan-out -->
<line class="arrow-debug" x1="855" y1="90" x2="920" y2="148"/>
<rect class="box-debug" x="855" y="150" width="120" height="50" rx="6"/>
<text class="lbl" x="915" y="168">debug_renderer /</text>
<text class="lbl" x="915" y="184">preview (opt-in)</text>
<text class="caption" x="490" y="230">solid = always-on data path &#183; dashed = optional debug/preview fan-out from identity_matcher's output</text>
</svg>

After

Width:  |  Height:  |  Size: 2.8 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 129 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 165 KiB

+79 -14
View File
@@ -10,6 +10,8 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/best-model/">
<link rel="prev" href="..">
@@ -51,6 +53,8 @@
<link rel="stylesheet" href="../stylesheets/extra.css">
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -67,7 +71,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -123,7 +127,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -248,13 +266,13 @@
<li class="md-tabs__item">
<a href="../rep4-optimizer-results/" class="md-tabs__link">
<a href="../model-bakeoff/" class="md-tabs__link">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -634,14 +652,14 @@
<li class="md-nav__item">
<a href="../rep4-optimizer-results/" class="md-nav__link">
<a href="../model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -792,7 +810,7 @@ open question: is LVFace (455MB) actually better, or just the biggest?</p>
<h2 id="first-signal-calibration-curves">First signal: calibration curves<a class="headerlink" href="#first-signal-calibration-curves" title="Permanent link">&para;</a></h2>
<p>Each gallery carries a fitted Platt sigmoid <code>P(match | cosine similarity) =
σ(a·sim + b)</code>, embedded directly in the gallery's HDF5 file
(<code>src/gallery/gallery_calibration.hpp</code>). This is a property of the embedding
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/gallery_calibration.hpp"><code>src/gallery/gallery_calibration.hpp</code></a>). This is a property of the embedding
space alone — computed from intra/inter-actor reference-image pairs, no
tracking or scene logic involved — so it's a clean first read on discriminative
power before running a single benchmark.</p>
@@ -834,7 +852,7 @@ a <em>lower</em> similarity threshold, than any ArcFace variant. That's a genuin
head start before the tracking/scoring pipeline is even involved.</p>
<h2 id="second-signal-f1-on-the-actual-benchmark">Second signal: F1 on the actual benchmark<a class="headerlink" href="#second-signal-f1-on-the-actual-benchmark" title="Permanent link">&para;</a></h2>
<p>Best full-gallery (no cast-restriction) result per model, from the 16-combo
rep4 matrix (<code>rep4-optimizer-results.md</code>):</p>
bake-off matrix (<a href="../model-bakeoff/">full experiment log</a>):</p>
<table>
<thead>
<tr>
@@ -876,22 +894,29 @@ rep4 matrix (<code>rep4-optimizer-results.md</code>):</p>
</tr>
</tbody>
</table>
<p>The full 16-combo picture makes the model ordering visible at a glance — LVFace
(yellow) tops both the restricted and full columns, and R18 (green) props up
the bottom of the full-gallery ranking:</p>
<p><img alt="All 16 bake-off combos ranked by training-set F1" src="../assets/images/rep4_matrix_f1.png" /></p>
<p>LVFace wins outright, with the highest recall of any full-mode combo. This
reverses an earlier conclusion from a prior (superseded) benchmarking pass
using a scene-union metric, which found the three models statistically
indistinguishable (~85% each) and concluded LVFace wasn't worth its size — that
metric hid out-of-cast false positives behind a gallery∩cast recall mask (see
<code>optimizer-experiments.md</code>); the per-second metric used here does not.</p>
<a href="../optimizer-experiments/">the prior optimizer round</a>); the per-second metric
used here does not.</p>
<p>Held-out validation (5 films never seen by the optimizer) confirms LVFace's
lead holds up out of sample — see the deep-dive page for the full breakdown,
including where it fails.</p>
lead holds up out of sample — see the
<a href="../lvface-deep-dive/">LVFace deep dive</a> for the full breakdown, including
where it fails.</p>
<h2 id="caveat-model-choice-is-an-operational-change">Caveat: model choice is an operational change<a class="headerlink" href="#caveat-model-choice-is-an-operational-change" title="Permanent link">&para;</a></h2>
<p>Switching the default embedder isn't just flipping a config value — the
gallery itself is model-specific (embeddings from different models aren't
comparable), so any existing gallery built against ArcFace w600k-R50 needs to
be rebuilt from source images against LVFace before the new default takes
effect. <code>scripts/optimizer/reembed_gallery.py</code> does this from a reference
gallery's cached source images without re-downloading anything.</p>
effect. <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/reembed_gallery.py"><code>scripts/optimizer/reembed_gallery.py</code></a>
does this from a reference gallery's cached source images without
re-downloading anything.</p>
@@ -922,6 +947,46 @@ gallery's cached source images without re-downloading anything.</p>
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href=".." class="md-footer__link md-footer__link--prev" aria-label="Previous: Home">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</div>
<div class="md-footer__title">
<span class="md-footer__direction">
Previous
</span>
<div class="md-ellipsis">
Home
</div>
</div>
</a>
<a href="../gallery-scope/" class="md-footer__link md-footer__link--next" aria-label="Next: Gallery Scope (Full vs. Limited)">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Gallery Scope (Full vs. Limited)
</div>
</div>
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -947,7 +1012,7 @@ gallery's cached source images without re-downloading anything.</p>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
+77 -12
View File
@@ -10,6 +10,8 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/gallery-scope/">
<link rel="prev" href="../best-model/">
@@ -51,6 +53,8 @@
<link rel="stylesheet" href="../stylesheets/extra.css">
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -67,7 +71,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -123,7 +127,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -248,13 +266,13 @@
<li class="md-tabs__item">
<a href="../rep4-optimizer-results/" class="md-tabs__link">
<a href="../model-bakeoff/" class="md-tabs__link">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -623,14 +641,14 @@
<li class="md-nav__item">
<a href="../rep4-optimizer-results/" class="md-nav__link">
<a href="../model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -768,7 +786,7 @@ entire library gallery (2418 actors across the 9-film benchmark set); <strong>re
pre-filters each film's gallery down to just its Jellyfin-credited cast (typically
~15 top-billed actors) before the matcher ever runs.</p>
<h2 id="the-result">The result<a class="headerlink" href="#the-result" title="Permanent link">&para;</a></h2>
<p>Averaged across all 4 models and both expansion settings, on the 4 rep4 training
<p>Averaged across all 4 models and both expansion settings, on the 4 bake-off training
films:</p>
<table>
<thead>
@@ -804,14 +822,19 @@ look-alike false match (an actor who happens to share enough facial structure
with someone in the film, but isn't actually in it), and the recall gain shows
it isn't costing real detections to get there.</p>
<p>Per-model, every single model's best-scoring combo in the full 16-way matrix is
a <code>restricted</code> variant — see the full table in <code>rep4-optimizer-results.md</code>. Two
a <code>restricted</code> variant — visible directly in the ranking below (filled dots =
restricted, open = full; the filled dots cluster at the top for every color):</p>
<p><img alt="All 16 bake-off combos — filled dots (restricted) dominate the top" src="../assets/images/rep4_matrix_f1.png" /></p>
<p>See the full table in the
<a href="../model-bakeoff/">bake-off experiment log</a>. Two
combos hit <strong>zero</strong> true out-of-cast misidentifications:
<code>arcface_w600k_mbf_restricted_exp</code> (F1 76.5%) and, in full mode,
<code>LVFace-B_Glint360K_full_noexp</code> (F1 72.4%) — restriction isn't the only way to
reach misid=0, but it's the more reliable one.</p>
<h2 id="why-this-isnt-the-shipped-default">Why this isn't the shipped default<a class="headerlink" href="#why-this-isnt-the-shipped-default" title="Permanent link">&para;</a></h2>
<p>Cast-restriction is implemented today only as an <strong>offline optimizer technique</strong>
(<code>scripts/optimizer/cast_restrict.py</code>): it pre-builds a filtered gallery file
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/cast_restrict.py"><code>scripts/optimizer/cast_restrict.py</code></a>):
it pre-builds a filtered gallery file
per film, using Jellyfin's own cast list, before the benchmark ever calls the
matcher. There's no runtime "restrict matching to this title's credited cast"
switch in the shipped application — <code>scene_analyze</code> always matches against
@@ -819,7 +842,8 @@ whatever single gallery file it's given.</p>
<p>Building that as a real feature would need, at minimum:</p>
<ul>
<li>A live Jellyfin cast lookup at analysis time (the title is already known —
<code>run_from_jellyfin.py</code> already does this same lookup for its own
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a>
already does this same lookup for its own
<code>filter_gallery</code>-based restriction path, just not wired into <code>scene_analyze</code>
itself as a first-class option).</li>
<li>A decision on the <em>fallback</em>: what happens to a real, uncredited cameo
@@ -828,7 +852,8 @@ whatever single gallery file it's given.</p>
<li>Regenerating the restricted-gallery cache whenever the title's Jellyfin cast
list changes.</li>
</ul>
<p>This is why the shipped <code>src/config.hpp</code> defaults use the <code>full</code>-mode winner
<p>This is why the shipped <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/config.hpp"><code>src/config.hpp</code></a>
defaults use the <code>full</code>-mode winner
(<code>LVFace-B_Glint360K_full_exp</code>, F1 75.3% training / 67.4% held-out macro) rather
than the higher-scoring <code>restricted_exp</code> (78.3%) — the 78.3% number describes a
capability the app doesn't have yet, not what actually ships.</p>
@@ -862,6 +887,46 @@ capability the app doesn't have yet, not what actually ships.</p>
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../best-model/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Best Model">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</div>
<div class="md-footer__title">
<span class="md-footer__direction">
Previous
</span>
<div class="md-ellipsis">
Best Model
</div>
</div>
</a>
<a href="../pose-expansion/" class="md-footer__link md-footer__link--next" aria-label="Next: Pose Expansion">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Pose Expansion
</div>
</div>
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -887,7 +952,7 @@ capability the app doesn't have yet, not what actually ships.</p>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
+99 -28
View File
@@ -10,6 +10,8 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/">
<link rel="next" href="best-model/">
@@ -49,6 +51,8 @@
<link rel="stylesheet" href="stylesheets/extra.css">
<script>__md_scope=new URL(".",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -65,7 +69,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -121,7 +125,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -246,13 +264,13 @@
<li class="md-tabs__item">
<a href="rep4-optimizer-results/" class="md-tabs__link">
<a href="model-bakeoff/" class="md-tabs__link">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -627,14 +645,14 @@
<li class="md-nav__item">
<a href="rep4-optimizer-results/" class="md-nav__link">
<a href="model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -782,35 +800,64 @@
film or TV episode — built on <a href="https://gitea.tourolle.paris/dtourolle/KPN">KPN++</a>
(a C++20 Kahn Process Network library) for the detect → track → match → scene
pipeline, with a Jellyfin-integrated gallery and an X-Ray-validated optimizer.</p>
<p>This is a perfect X-Ray second, on a film the optimizer never saw:</p>
<p><img alt="A perfect X-Ray second: three faces named at 100%, two more correctly carried off-screen" src="assets/images/lovelace_perfect_second.jpg" /></p>
<p>Every visible face named at 100% — Chris Noth, Hank Azaria, Bobby Cannavale —
the background extra honestly left unnamed, and the two credited cast without
a visible face correctly carried as present off-screen by the tracker's
presence windows. That's the pipeline exactly reproducing Amazon X-Ray's
record for this second.</p>
<p>It doesn't always go like that: the hardest held-out film scores 46% F1, and
the report is honest about <em>why</em> — one tunable trade (extinction bridging at
hard cuts), one structural ceiling (X-Ray credits people whose faces never
appear), and a few cases where the pipeline is right and X-Ray is wrong. The
evidence for all of it is in the pages below.</p>
<h2 id="start-here-four-questions-this-bake-off-answers">Start here — four questions this bake-off answers<a class="headerlink" href="#start-here-four-questions-this-bake-off-answers" title="Permanent link">&para;</a></h2>
<div class="grid cards">
<ul>
<li><strong><a href="best-model/">Which model is best?</a></strong> — calibration curves first
(discriminative power, independent of any threshold), then F1 on the actual
benchmark. LVFace-B Glint360K wins both.</li>
<li><strong><a href="gallery-scope/">Whole gallery vs. limited (cast-restricted) gallery</a></strong>
restricting the matcher to a film's credited cast is a clean win on every
axis (+3.3pp F1, less than a third the misIDs), but isn't a shipped runtime
feature yet.</li>
<li><strong><a href="pose-expansion/">Does pose expansion help?</a></strong> — a real training-set
effect that didn't reproduce on 5 held-out films once two methodology bugs
were caught and fixed. An honest null result, not a forced narrative.</li>
<li><strong><a href="lvface-deep-dive/">Deep dive: LVFace-B Glint360K</a></strong> — the winning
model's held-out generalization gap, its two real failure modes (frozen-bbox
"ghost tracks"), and one case where it correctly identified an actor that
the X-Ray ground truth itself failed to credit.</li>
<li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M18 2c-.9 0-2 1-2 2H8c0-1-1.1-2-2-2H2v9c0 1 1 2 2 2h2.2c.4 2 1.7 3.7 4.8 4v2.08C8 19.54 8 22 8 22h8s0-2.46-3-2.92V17c3.1-.3 4.4-2 4.8-4H20c1 0 2-1 2-2V2zM6 11H4V4h2zm14 0h-2V4h2z"/></svg></span> <strong><a href="best-model/">Which model is best?</a></strong></p>
<hr />
<p>Calibration curves first (discriminative power, independent of any
threshold), then F1 on the actual benchmark. LVFace-B Glint360K wins
both.</p>
</li>
<li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M14 12v7.88c.04.3-.06.62-.29.83a.996.996 0 0 1-1.41 0l-2.01-2.01a.99.99 0 0 1-.29-.83V12h-.03L4.21 4.62a1 1 0 0 1 .17-1.4c.19-.14.4-.22.62-.22h14c.22 0 .43.08.62.22a1 1 0 0 1 .17 1.4L14.03 12z"/></svg></span> <strong><a href="gallery-scope/">Whole vs. cast-restricted gallery</a></strong></p>
<hr />
<p>Restricting the matcher to a film's credited cast is a clean win on
every axis (+3.3pp F1, less than a third the misIDs) — but isn't a
shipped runtime feature yet.</p>
</li>
<li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m12 0-.66.03 3.81 3.81L16.5 2.5c3.25 1.57 5.59 4.74 5.95 8.5h1.5C23.44 4.84 18.29 0 12 0m0 4c-1.93 0-3.5 1.57-3.5 3.5S10.07 11 12 11s3.5-1.57 3.5-3.5S13.93 4 12 4M.05 13C.56 19.16 5.71 24 12 24l.66-.03-3.81-3.81L7.5 21.5c-3.25-1.56-5.59-4.74-5.95-8.5zM12 13c-3.87 0-7 1.57-7 3.5V18h14v-1.5c0-1.93-3.13-3.5-7-3.5"/></svg></span> <strong><a href="pose-expansion/">Does pose expansion help?</a></strong></p>
<hr />
<p>A convincing training-set effect that didn't reproduce on 5 held-out
films once two methodology bugs were caught and fixed. An honest null
result, not a forced narrative.</p>
</li>
<li>
<p><span class="twemoji lg middle"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M18 16h-.58l-.81-.81A7.07 7.07 0 0 0 18 11c0-3.87-3.13-7-7-7-1.5 0-3 .5-4.21 1.4-3.09 2.32-3.72 6.71-1.4 9.8s6.71 3.72 9.8 1.4l.81.81V18l5 5 2-2zm-7 0c-2.76 0-5-2.24-5-5s2.24-5 5-5 5 2.24 5 5-2.24 5-5 5M3 6 1 8V1h7L6 3H3zm18-5v7l-2-2V3h-3l-2-2zM6 19l2 2H1v-7l2 2v3z"/></svg></span> <strong><a href="lvface-deep-dive/">Deep dive: LVFace-B Glint360K</a></strong></p>
<hr />
<p>The held-out generalization gap, how the error budget decomposes
(extinction bridging at hard cuts, X-Ray's scene-membership vs.
on-screen-face ceiling), and the frames where the pipeline is right
and the ground truth is wrong.</p>
</li>
</ul>
</div>
<h2 id="the-full-technical-log">The full technical log<a class="headerlink" href="#the-full-technical-log" title="Permanent link">&para;</a></h2>
<ul>
<li><strong><a href="rep4-optimizer-results/">Rep4 model bake-off + threshold re-tune</a></strong>
<li><strong><a href="model-bakeoff/">Model bake-off + threshold re-tune</a></strong>
the complete experiment log behind the four pages above: the ROCm teardown
deadlock root cause and fix, DE concurrency tuning, the full 16-combo
results table, and every caveat. This is where the shipped <code>src/config.hpp</code>
defaults come from.</li>
results table, and every caveat. This is where the shipped
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/config.hpp"><code>src/config.hpp</code></a> defaults come from.</li>
<li><strong><a href="optimizer-experiments/">Optimizer experiments (prior round)</a></strong> — the
earlier scene-union-metric tuning pass, superseded by the per-second metric
used in rep4 but kept for the ground-truth/architecture background.</li>
used in the bake-off but kept for the ground-truth/architecture background.</li>
<li><strong><a href="service-conversion/">Service conversion (proposal)</a></strong> — design sketch
for an idle-GPU Docker worker, not yet built.</li>
for a native idle-GPU worker gated on screen lock, not yet built.</li>
</ul>
<h2 id="reproducing-the-benchmarks">Reproducing the benchmarks<a class="headerlink" href="#reproducing-the-benchmarks" title="Permanent link">&para;</a></h2>
<p>Gallery <code>.h5</code> files, embedding dumps, the X-Ray corpus, montage frame images,
@@ -820,8 +867,8 @@ the Gitea package registry and pulled on demand:</p>
</span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>experiment-data
</span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a>scripts/artifacts/pull_artifacts.sh<span class="w"> </span>montage-frames<span class="w"> </span>&lt;film-slug&gt;
</span></code></pre></div>
<p>See <code>scripts/artifacts/push_artifacts.sh</code> for the upload side (requires a
<code>GITEA_TOKEN</code> with package write scope).</p>
<p>See <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/artifacts/push_artifacts.sh"><code>scripts/artifacts/push_artifacts.sh</code></a>
for the upload side (requires a <code>GITEA_TOKEN</code> with package write scope).</p>
@@ -852,6 +899,30 @@ the Gitea package registry and pulled on demand:</p>
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="best-model/" class="md-footer__link md-footer__link--next" aria-label="Next: Best Model">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Best Model
</div>
</div>
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -877,7 +948,7 @@ the Gitea package registry and pulled on demand:</p>
<script id="__config" type="application/json">{"annotate": null, "base": ".", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": ".", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="assets/javascripts/bundle.d7400e89.min.js"></script>
+227 -70
View File
@@ -10,11 +10,13 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/lvface-deep-dive/">
<link rel="prev" href="../pose-expansion/">
<link rel="next" href="../rep4-optimizer-results/">
<link rel="next" href="../model-bakeoff/">
@@ -51,6 +53,8 @@
<link rel="stylesheet" href="../stylesheets/extra.css">
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -67,7 +71,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -123,7 +127,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -248,13 +266,13 @@
<li class="md-tabs__item">
<a href="../rep4-optimizer-results/" class="md-tabs__link">
<a href="../model-bakeoff/" class="md-tabs__link">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -578,6 +596,17 @@
</label>
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#what-good-looks-like" class="md-nav__link">
<span class="md-ellipsis">
What good looks like
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#training-vs-held-out-the-generalization-gap" class="md-nav__link">
<span class="md-ellipsis">
@@ -590,10 +619,10 @@
</li>
<li class="md-nav__item">
<a href="#failure-mode-1-frozen-bbox-ghost-tracks" class="md-nav__link">
<a href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" class="md-nav__link">
<span class="md-ellipsis">
Failure mode 1: frozen-bbox "ghost tracks"
Mechanism 1: extinction bridging — usually right, wrong at hard cuts
</span>
</a>
@@ -601,10 +630,10 @@
</li>
<li class="md-nav__item">
<a href="#failure-mode-2-a-genuine-misid-for-contrast" class="md-nav__link">
<a href="#mechanism-2-the-face-vs-presence-ceiling" class="md-nav__link">
<span class="md-ellipsis">
Failure mode 2: a genuine misID (for contrast)
Mechanism 2: the face-vs-presence ceiling
</span>
</a>
@@ -656,14 +685,14 @@
<li class="md-nav__item">
<a href="../rep4-optimizer-results/" class="md-nav__link">
<a href="../model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -756,6 +785,17 @@
</label>
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#what-good-looks-like" class="md-nav__link">
<span class="md-ellipsis">
What good looks like
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#training-vs-held-out-the-generalization-gap" class="md-nav__link">
<span class="md-ellipsis">
@@ -768,10 +808,10 @@
</li>
<li class="md-nav__item">
<a href="#failure-mode-1-frozen-bbox-ghost-tracks" class="md-nav__link">
<a href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" class="md-nav__link">
<span class="md-ellipsis">
Failure mode 1: frozen-bbox "ghost tracks"
Mechanism 1: extinction bridging — usually right, wrong at hard cuts
</span>
</a>
@@ -779,10 +819,10 @@
</li>
<li class="md-nav__item">
<a href="#failure-mode-2-a-genuine-misid-for-contrast" class="md-nav__link">
<a href="#mechanism-2-the-face-vs-presence-ceiling" class="md-nav__link">
<span class="md-ellipsis">
Failure mode 2: a genuine misID (for contrast)
Mechanism 2: the face-vs-presence ceiling
</span>
</a>
@@ -829,14 +869,44 @@
<h1 id="deep-dive-lvface-b-glint360k">Deep dive: LVFace-B Glint360K<a class="headerlink" href="#deep-dive-lvface-b-glint360k" title="Permanent link">&para;</a></h1>
<p>LVFace won the model bake-off (see <code>best-model.md</code>) and is the shipped default
embedder. This page is the honest accounting of how it actually performs —
including where it's wrong, and one case where the ground truth itself is
wrong and LVFace is right.</p>
<p>LVFace won the model bake-off (see <a href="../best-model/">Which model is best?</a>) and is
the shipped default embedder. This page is the honest accounting of how it
actually performs — what a good second looks like, where the errors actually
come from, and two cases where the ground truth itself is wrong and LVFace is
right.</p>
<div class="admonition note">
<p class="admonition-title">How to read the frames on this page</p>
<p>The top is the film frame, with a box and name on every face the pipeline
identified. The bottom panels are the per-second verdict against X-Ray:
<strong>Onscreen</strong> lists faces named in the frame, <strong>Offscreen</strong> lists cast
X-Ray marks present in the scene without a visible face — presence
carried by the tracker's windows, not by a detection. Colors are the
score: <span style="color:#0ca30c"><strong>green</strong></span> = correct (TPI),
<span style="color:#eb6834"><strong>orange</strong></span> = wrong (FPI),
<span style="color:#3987e5"><strong>blue</strong></span> = missed (FN).</p>
</div>
<h2 id="what-good-looks-like">What good looks like<a class="headerlink" href="#what-good-looks-like" title="Permanent link">&para;</a></h2>
<p><img alt="Wedding couple correctly identified, Downton Abbey: A New Era" src="../assets/images/downton_wedding_couple.jpg" /></p>
<p>Six faces on screen, all six named correctly — including Penelope Wilton at the
edge of the pews and a half-occluded Michelle Dockery — while thirteen more
cast members X-Ray marks present in the scene are correctly carried as
"Offscreen" by their presence windows. One miss in the whole frame: Maggie
Smith (blue). Score for this second: 0.86.</p>
<p><img alt="19 of 20 correct in the funeral crowd" src="../assets/images/downton_funeral_19of20.jpg" /></p>
<p>The same film's funeral gathering: mourning dress, hats, half the faces turned.
<strong>Nineteen of the twenty cast X-Ray lists for this scene are scored correctly</strong>
— seven named on screen at up to 100% confidence, twelve more correctly held
as present off-screen.</p>
<p>And the pipeline doesn't need the face to be <em>real</em>:</p>
<p><img alt="Herbie Hancock identified on an in-fiction video call" src="../assets/images/valerian_screen_call.jpg" /></p>
<p>That's Herbie Hancock at 98% — as a face on a <em>screen inside the movie</em>, over a
sci-fi HUD overlay, during a video call in Valerian. A face is a face, whether
it's in the room or on the bridge's comms display.</p>
<h2 id="training-vs-held-out-the-generalization-gap">Training vs. held-out: the generalization gap<a class="headerlink" href="#training-vs-held-out-the-generalization-gap" title="Permanent link">&para;</a></h2>
<p>The shipped config (<code>prob_threshold=0.754, anneal_sec=35.54,
extinction_sec=57.43, expand_gallery=true</code>) was tuned against 4 films. Scored
against the 5 films the optimizer never saw:</p>
<p><img alt="Held-out per-film F1 vs. the training-set fit" src="../assets/images/holdout_f1_by_film.png" /></p>
<table>
<thead>
<tr>
@@ -915,61 +985,108 @@ against the 5 films the optimizer never saw:</p>
</table>
<p><strong>67.4% held-out vs. 75.3% on training</strong> — an ~8pp drop, and a <strong>37pp spread
between the best and worst held-out film</strong>. The config does not generalize
uniformly; two films are outright failure cases, for two different reasons.</p>
<h2 id="failure-mode-1-frozen-bbox-ghost-tracks">Failure mode 1: frozen-bbox "ghost tracks"<a class="headerlink" href="#failure-mode-1-frozen-bbox-ghost-tracks" title="Permanent link">&para;</a></h2>
<p>Both Many Saints of Newark (974 misIDs) and Downton Abbey (FN=80084, the worst
recall of the five) trace to the same root cause, verified directly against
the raw per-frame stream and the HDF5 dump's own detection counts — not
inferred from the score alone.</p>
<p><img alt="Frozen ghost boxes over background, The Many Saints of Newark" src="../assets/images/many_saints_ghost_fpi.jpg" /></p>
<p>At this second, three of the four labeled boxes ("Jon Bernthal", "Joey Diaz",
"Billy Magnussen") sit over empty background — a blurred wall, hanging
plates — with no face in them. The real face in frame carries a second,
colliding label from another frozen box.</p>
<p><img alt="15 ghost boxes over a blank title card, Downton Abbey: A New Era" src="../assets/images/downton_abbey_ghost_fpi.jpg" /></p>
<p>This is the starkest case: <strong>15 actors named, all wrong, over a completely
blank closing title card.</strong> Confirmed against the dump directly: <code>face_count</code>
is 0 from this point onward (no detector output at all), yet the same 15
identities keep appearing with the <em>exact same bounding box, unchanged to the
pixel</em>, for 57+ consecutive seconds.</p>
<p>This is <code>SceneTrackerFunc::active_[actor_idx].last_bbox</code>
(<code>src/nodes/scene_tracker_node.hpp</code>) being re-emitted unchanged — the
extinction state machine working exactly as coded, not a bug. The film cuts
from a packed group shot straight into 40+ seconds of blank titles/credits,
and <code>extinction_sec=57.4</code> is comfortably long enough to bridge that entire gap
without expiring, so the tracker faithfully reports "last known position" for
a cast that is no longer on screen at all. <code>extinction_sec</code> was tuned toward
long windows specifically because they bridge real gaps (occlusion, a turned
face) in most training footage — this is the cost side of that trade,
surfacing only when a film has a long enough faceless stretch to expose it.</p>
<h2 id="failure-mode-2-a-genuine-misid-for-contrast">Failure mode 2: a genuine misID (for contrast)<a class="headerlink" href="#failure-mode-2-a-genuine-misid-for-contrast" title="Permanent link">&para;</a></h2>
<p>Not every held-out failure is a ghost. This is a real face, correctly
detected, confidently misidentified:</p>
<p><em>(same many_saints_ghost_fpi.jpg frame above also shows Leslie Odom Jr.'s box
carrying a second, colliding "Michael Gandolfini" label — two real tracks'
frozen positions happening to overlap, not a detection error.)</em></p>
uniformly, and the spread traces to two mechanisms, both visible frame by
frame below.</p>
<h2 id="mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts">Mechanism 1: extinction bridging — usually right, wrong at hard cuts<a class="headerlink" href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" title="Permanent link">&para;</a></h2>
<p>The extinction window keeps an identity alive through seconds where no face is
detectable. <strong>Most of the time this is exactly what you want</strong>, and it's where
a lot of the TPI count comes from:</p>
<p><img alt="Two faces on screen, six more correctly bridged" src="../assets/images/lovelace_polygraph_bridged.jpg" /></p>
<p>Lovelace's polygraph scene: only Eric Roberts and Amanda Seyfried have visible
faces, but X-Ray lists eight cast present — and all eight score green, the
other six correctly carried by presence windows through a scene where the
camera never shows them. A perfect second, and the extinction/anneal machinery
is <em>why</em>.</p>
<p>The same mechanism has a failure case: a hard cut into long faceless footage.
Both Many Saints of Newark (974 misIDs) and Downton Abbey (FN=80084, the worst
recall of the five) are dominated by it — verified directly against the raw
per-frame stream and the HDF5 dump's own detection counts, not inferred from
the score alone. <strong>This is not a malfunction</strong>: the tracker is doing exactly
what its window is for; the footage just stops cooperating. In the debug
overlay (which draws a bridged identity's last-known bbox, unlike the shipped
output, which emits presence windows and no boxes at all) the bridged state is
visible spatially:</p>
<p><img alt="Debug overlay: bridged identities drawn at their last-known positions" src="../assets/images/many_saints_ghost_fpi.jpg" />
<em>Debug-overlay rendering (<code>dump_error_frames.py --raw</code>): "Jon Bernthal", "Joey
Diaz" and "Billy Magnussen" are extinction-bridged identities from the previous
shot, drawn frozen over the wall and the hanging plates. Frame
<code>many_saints/fpi/fpi_t03543.jpg</code>, <code>montage-frames</code> artifact package.</em></p>
<p>The cost is measurable, not just visible. Downton Abbey's hard cut into its
closing credits, plotting the dump's own per-second <code>face_count</code> (detector
output, independent of the tracker) against what the tracker reports:</p>
<p><img alt="Detector vs. tracker through Downton Abbey's cut to credits" src="../assets/images/downton_ghost_timeline.png" /></p>
<p>From the cut onward the detector sees <strong>zero faces for nearly a minute</strong> — and
the tracker keeps reporting the last shot's 15 identities the whole time
(verified for Hugh Bonneville: bbox <code>(1743.2, 0.0, 171.3, 317.8)</code>, unchanged to
the pixel, at every sampled second for 57+ seconds). The staircase at the right
edge is the extinction window expiring actor by actor. That plateau is
<code>SceneTrackerFunc::active_[actor_idx].last_bbox</code>
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/nodes/scene_tracker_node.hpp"><code>src/nodes/scene_tracker_node.hpp</code></a>)
re-emitted as designed: <code>extinction_sec=57.4</code> was tuned long because bridging
wins on most footage (see the polygraph frame above) — the training films just
never contained a faceless stretch long enough to show the cost side, and the
held-out set did.</p>
<p>The same track-continuation machinery has one milder spatial artifact, worth
knowing when reading these frames:</p>
<p><img alt="Two labels on one face after a shot/reverse-shot cut" src="../assets/images/cafe_society_rapid_cut.jpg" />
<em>Café Society (a training film), a shot/reverse-shot dialog: that is Steve
Carell wearing both his own label and Jesse Eisenberg's.</em></p>
<p>At a rapid cut, the previous shot's track can linger for a beat at nearly the
same screen position the new face occupies — here Jesse Eisenberg's box from
the counter-shot lands on Steve Carell. Note what the score panel says,
though: both actors are green, because both <em>are</em> present in this dialog
scene per X-Ray. The spatial label is briefly wrong; the per-second presence
claim — the thing the pipeline actually ships — is right. It's the same trade
as the extinction window: track continuation smooths over cuts, and 1 fps
sampling occasionally catches the seam.</p>
<h2 id="mechanism-2-the-face-vs-presence-ceiling">Mechanism 2: the face-vs-presence ceiling<a class="headerlink" href="#mechanism-2-the-face-vs-presence-ceiling" title="Permanent link">&para;</a></h2>
<p>Downton Abbey's recall didn't collapse because faces were misread — it
collapsed because for most of its 80084 FN-seconds there was <strong>no face to
read</strong>:</p>
<p><img alt="22 cast credited, nobody facing the camera" src="../assets/images/downton_crew_fn.jpg" /></p>
<p>A newsreel crew hauls equipment through the hall: X-Ray credits 22 cast as
present in this scene; not one face looks at the camera. Eight are still
scored green (windows bridging from adjacent shots) — the other fourteen are
blue FNs that no face-recognition pipeline could ever recover. X-Ray encodes
<em>scene membership</em>; the pipeline measures <em>on-screen faces</em>. In ensemble films
those two definitions diverge massively, and that gap — not identification
error — is most of what the FN column counts.</p>
<p><img alt="Presence without a detectable face, The Many Saints of Newark" src="../assets/images/many_saints_outofcast_fpi.jpg" /></p>
<p>Same ceiling from the other side: Michela De Rossi in frame but turned away,
five cast correctly bridged as offscreen (green), four blue FNs — and one
orange we'll come back to below.</p>
<h2 id="where-lvface-beat-x-ray">Where LVFace beat X-Ray<a class="headerlink" href="#where-lvface-beat-x-ray" title="Permanent link">&para;</a></h2>
<p>Not every "misID" is actually wrong. <code>second_score.py</code> counts a name as a true
out-of-cast misID whenever the named actor isn't in X-Ray's credited cast list
for the film at all — but X-Ray's cast list is itself incomplete.</p>
<p>Not every orange in these frames is actually wrong.
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>
scores strictly against X-Ray — but X-Ray itself has holes, and the pipeline
found two kinds.</p>
<p><img alt="LVFace correctly identifies Germar Terrell Gardner, uncredited by X-Ray" src="../assets/images/germar_beats_xray.jpg" /></p>
<p>Germar Terrell Gardner — a real, clean, high-confidence detection — is counted
as a misID here because he doesn't appear in X-Ray's <code>people.csv</code> for The Many
Saints of Newark at all. But Jellyfin's independent cast metadata <em>does</em> credit
him for this exact film (cross-checked via <code>experiments/manifests/
jellyfin_casts.json</code>, a completely separate data source from X-Ray). This
isn't a lookalike error or a gallery mixup — it's the pipeline correctly
recognising a real cast member that one ground-truth source happened to omit.</p>
as an out-of-cast misID because he doesn't appear in X-Ray's <code>people.csv</code> for
The Many Saints of Newark at all. But Jellyfin's independent cast metadata
<em>does</em> credit him for this exact film (cross-checked via
<code>experiments/manifests/jellyfin_casts.json</code> from the <code>experiment-data</code> artifact
package, a completely separate data source from X-Ray). That's also him in
orange in the frame above — every one of those "errors" is the pipeline being
right about a person X-Ray forgot.</p>
<p><img alt="Robert Patrick, clearly on screen, scored wrong by a ground-truth gap" src="../assets/images/lovelace_robert_patrick_fpi.jpg" /></p>
<p>And it isn't only uncredited bit-parts. That is <strong>Robert Patrick</strong> — top-billed
in Lovelace, unmistakably on screen, reading his newspaper, identified at
100% — scored orange because X-Ray's people-in-scene list for <em>this scene</em>
doesn't include him. The identification is flawless; the ground truth missed
an actor sitting in the middle of the frame.</p>
<p>This doesn't mean every flagged misID is secretly correct — Many Saints'
974-count total is still overwhelmingly the frozen-bbox failure mode above,
not uncredited-but-real cameos. But it's a reminder that the X-Ray corpus is a
convenient, large-scale ground truth, not a perfect one, and the "misID" number
in any of these tables has some irreducible noise floor from ground-truth gaps
in the other direction too.</p>
974-count total is still overwhelmingly extinction bridging at cuts, not
uncredited cameos. But the X-Ray corpus is a convenient, large-scale ground
truth, not a perfect one, and the misID/FPI numbers in these tables carry an
irreducible noise floor from ground-truth gaps in both directions.</p>
<h2 id="summary">Summary<a class="headerlink" href="#summary" title="Permanent link">&para;</a></h2>
<p>LVFace is the right default: it wins the model comparison outright, and its
failures are traceable, understood, and mostly attributable to one tunable
knob (<code>extinction_sec</code>) rather than the embedder itself. The held-out
<p>LVFace is the right default: it wins the model comparison outright, it names
19 of 20 correctly across a hat-heavy funeral crowd, and it recognises a face
on a screen inside the movie. Its error budget decomposes into two understood
mechanisms — extinction bridging at hard cuts (a tunable trade, not a bug) and
the face-vs-presence ceiling baked into X-Ray's semantics — plus a nonzero
slice where the pipeline is right and the ground truth is wrong. The held-out
generalization gap (75.3% → 67.4%) is real and should be treated as the honest
expected performance, not the training-set number.</p>
@@ -1002,6 +1119,46 @@ expected performance, not the training-set number.</p>
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../pose-expansion/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Pose Expansion">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</div>
<div class="md-footer__title">
<span class="md-footer__direction">
Previous
</span>
<div class="md-ellipsis">
Pose Expansion
</div>
</div>
</a>
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--next" aria-label="Next: Model Bake-off &amp; Re-tune (full log)">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</div>
</div>
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -1027,7 +1184,7 @@ expected performance, not the training-set number.</p>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
@@ -10,6 +10,8 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/model-bakeoff/">
<link rel="prev" href="../lvface-deep-dive/">
@@ -25,7 +27,7 @@
<title>Rep4 Bake-off & Re-tune (full log) - scene-actor-extraction</title>
<title>Model Bake-off & Re-tune (full log) - scene-actor-extraction</title>
@@ -51,6 +53,8 @@
<link rel="stylesheet" href="../stylesheets/extra.css">
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -67,7 +71,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -76,7 +80,7 @@
<div data-md-component="skip">
<a href="#rep4-model-bake-off-threshold-re-tune-experiment-log-2026-07-1819" class="md-skip">
<a href="#model-bake-off-threshold-re-tune-experiment-log-2026-07-1819" class="md-skip">
Skip to content
</a>
@@ -110,7 +114,7 @@
<div class="md-header__topic" data-md-component="header-topic">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</span>
</div>
@@ -123,7 +127,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -254,7 +272,7 @@
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -565,7 +583,7 @@
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -583,7 +601,7 @@
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -664,10 +682,10 @@
</li>
<li class="md-nav__item">
<a href="#training-films-rep4-and-validation-set" class="md-nav__link">
<a href="#training-films-and-validation-set" class="md-nav__link">
<span class="md-ellipsis">
Training films (rep4) and validation set
Training films and validation set
</span>
</a>
@@ -906,10 +924,10 @@
</li>
<li class="md-nav__item">
<a href="#training-films-rep4-and-validation-set" class="md-nav__link">
<a href="#training-films-and-validation-set" class="md-nav__link">
<span class="md-ellipsis">
Training films (rep4) and validation set
Training films and validation set
</span>
</a>
@@ -1021,12 +1039,16 @@
<h1 id="rep4-model-bake-off-threshold-re-tune-experiment-log-2026-07-1819">Rep4 model bake-off + threshold re-tune — experiment log (2026-07-18/19)<a class="headerlink" href="#rep4-model-bake-off-threshold-re-tune-experiment-log-2026-07-1819" title="Permanent link">&para;</a></h1>
<p>Follow-on to <code>docs/optimizer-experiments.md</code>, which used an older, since-superseded
scene-union metric. This round uses the <strong>per-second</strong> metric
(<code>scripts/optimizer/second_score.py</code>) and answers three questions in one 16-run
matrix: which embedding model is best, does cast-restriction cut misIDs, and does
<h1 id="model-bake-off-threshold-re-tune-experiment-log-2026-07-1819">Model bake-off + threshold re-tune — experiment log (2026-07-18/19)<a class="headerlink" href="#model-bake-off-threshold-re-tune-experiment-log-2026-07-1819" title="Permanent link">&para;</a></h1>
<p>Follow-on to <a href="../optimizer-experiments/">the prior optimizer round</a>, which used
an older, since-superseded scene-union metric. This round uses the <strong>per-second</strong> metric
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>)
and answers three questions in one 16-run matrix: which embedding model is best, does cast-restriction cut misIDs, and does
per-film gallery expansion help.</p>
<p>(This was the fourth optimizer campaign against the X-Ray corpus, so its on-disk
artifacts carry an internal <code>rep4_</code> prefix — <code>experiments/results/rep4_best_*.json</code>,
<code>experiments/trajectories/rep4_*.jsonl</code>, and the manifests referenced below. The
earlier campaigns used the superseded scene-union metric and were discarded.)</p>
<h2 id="why-this-experiment-and-what-it-actually-delivered">Why this experiment, and what it actually delivered<a class="headerlink" href="#why-this-experiment-and-what-it-actually-delivered" title="Permanent link">&para;</a></h2>
<p>Four goals going in, and an honest read on each after held-out validation (see
below):</p>
@@ -1093,12 +1115,13 @@ below):</p>
</tr>
</tbody>
</table>
<p>These are the <strong><code>LVFace-B_Glint360K_full_exp</code></strong> winning values — the best result that
<p>These are the <strong><code>LVFace-B_Glint360K_full_exp</code></strong> winning values, applied to
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/config.hpp"><code>src/config.hpp</code></a> — the best result that
uses only features already live in the running app (full gallery, no cast
restriction; see below for why restricted mode isn't applied even though it scored
higher).</p>
<h2 id="why-re-run-at-all">Why re-run at all<a class="headerlink" href="#why-re-run-at-all" title="Permanent link">&para;</a></h2>
<p><code>docs/optimizer-experiments.md</code>'s scene-union metric hid out-of-cast false positives
<p><a href="../optimizer-experiments/">The prior round</a>'s scene-union metric hid out-of-cast false positives
behind a gallery∩cast recall mask, and the earlier 9-film benchmark was scene-level
(union over a whole X-Ray scene), not a fair per-timepoint comparison. SESSION_STATE
flagged the old scene-metric bake-off numbers (R50≈LVFace≈MBF ~85%) as superseded.
@@ -1106,17 +1129,20 @@ This round uses <code>second_score.py</code>: uniform per-second sampling, GT =
cast at time <em>t</em>, pred = actors whose presence window covers <em>t</em>, FPI weighted 10×
when the named actor isn't in the film's cast at all (true misID) vs. an in-cast
timing slip. FN only counts gallery-known cast (fair recall — 67% of X-Ray cast have
no reference embedding, see <code>gallery-coverage-gap</code> memory).</p>
no reference embedding; see
<a href="../optimizer-experiments/#the-gallery-coverage-gap">the prior round's gallery-coverage-gap analysis</a>).</p>
<h2 id="the-deadlock-that-was-blocking-all-of-this">The deadlock that was blocking all of this<a class="headerlink" href="#the-deadlock-that-was-blocking-all-of-this" title="Permanent link">&para;</a></h2>
<p>Every replay in this line of work goes through <code>scripts/optimizer/replay.py</code>, which
runs the real C++ tracker/matcher/scene_tracker nodes inside a Python-assembled KPN
network. Before this session, every subprocess replay <strong>timed out at 45s, 100% of
<p>Every replay in this line of work goes through
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/replay.py"><code>scripts/optimizer/replay.py</code></a>,
which runs the real C++ tracker/matcher/scene_tracker nodes inside a
Python-assembled KPN network. Before this session, every subprocess replay <strong>timed out at 45s, 100% of
the time</strong> — not the documented ~20-30% ROCm rocBLAS-GEMM driver flake, but a plain
logic bug: <code>replay.py</code>'s CLI called <code>replay(net, ..., stop=False)</code> to <em>skip</em>
<code>net.stop()</code> (trying to dodge the GEMM deadlock), planning to <code>os._exit(0)</code>
immediately after. But:</p>
<ul>
<li><code>PyNode::stop()</code> (<code>external/KPN/include/kpn/python/bindings.hpp</code>) is the <em>only</em>
<li><code>PyNode::stop()</code> (<a href="https://gitea.tourolle.paris/dtourolle/KPN/raw/commit/4b6e498ba7e70a34cc0b57638f9e56a43b7f41ae/include/kpn/python/bindings.hpp"><code>include/kpn/python/bindings.hpp</code></a>
in the KPN++ submodule) is the <em>only</em>
code that sets <code>stop_flag_ = true</code> before joining the node's worker thread.</li>
<li>The source node's <code>run_loop()</code> has <code>while (!stop_flag_)</code> as its only exit
condition (it has no input channels, so it never sees a channel-closed signal
@@ -1136,7 +1162,9 @@ same trace are normal ROCm runtime housekeeping, not evidence of a wedged GPU ke
(down from a guaranteed 45s timeout), and a full DE sweep producing real, sensible
F1/precision/recall instead of flat 0.0%.</p>
<h2 id="concurrency-tuning">Concurrency tuning<a class="headerlink" href="#concurrency-tuning" title="Permanent link">&para;</a></h2>
<p>With the deadlock fixed, <code>optimize.py</code> was extended with DE-level parallelism —
<p>With the deadlock fixed,
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/optimize.py"><code>scripts/optimizer/optimize.py</code></a>
was extended with DE-level parallelism —
<code>differential_evolution(..., workers=ThreadPoolExecutor.map)</code> — so multiple
population candidates evaluate concurrently, each spawning its own per-film replay
subprocesses (<code>REPLAY_WORKERS</code>). Total concurrent GPU replay processes ≈
@@ -1171,9 +1199,9 @@ subprocesses (<code>REPLAY_WORKERS</code>). Total concurrent GPU replay processe
actually being garbage — a dangerous failure mode, not a crash. <strong>8 concurrent is the
practical ceiling</strong> on this GPU (gfx1100) for this workload. The matrix ran at
<code>REPLAY_WORKERS=4 DE_WORKERS=2</code>.</p>
<h2 id="training-films-rep4-and-validation-set">Training films (rep4) and validation set<a class="headerlink" href="#training-films-rep4-and-validation-set" title="Permanent link">&para;</a></h2>
<h2 id="training-films-and-validation-set">Training films and validation set<a class="headerlink" href="#training-films-and-validation-set" title="Permanent link">&para;</a></h2>
<p>9 films total have dumped embeddings across all 4 models. 4 were used for
optimization (rep4), leaving 5 held out for validation:</p>
optimization, leaving 5 held out for validation:</p>
<ul>
<li><strong>Lord of War</strong> (64-cast, "clean")</li>
<li><strong>Scarface</strong> (67-cast, "ensemble/lookalike")</li>
@@ -1183,7 +1211,7 @@ optimization (rep4), leaving 5 held out for validation:</p>
Downton Abbey or The Many Saints of Newark)</li>
</ul>
<p>Held out: Benny &amp; Joon, Downton Abbey: A New Era, Lovelace, The Many Saints of
Newark, Valerian and the City of a Thousand Planets. The rep4 numbers below are
Newark, Valerian and the City of a Thousand Planets. The training-set numbers below are
training-set fit — see "Held-out validation" further down for the real
generalization test.</p>
<h2 id="search-space-and-de-settings">Search space and DE settings<a class="headerlink" href="#search-space-and-de-settings" title="Permanent link">&para;</a></h2>
@@ -1375,9 +1403,14 @@ includes in-cast timing slips.</p>
</table>
<p>† old, narrower anneal/extinction bounds (see above) — not directly comparable to
the other 12 on those two params.</p>
<p>The same 16 results as a picture — the two headline effects are visible without
reading a single row: filled (restricted) dots stack the top of the ranking for
every model color, and yellow (LVFace) leads within both scopes:</p>
<p><img alt="All 16 bake-off combos ranked by training-set F1" src="../assets/images/rep4_matrix_f1.png" /></p>
<h2 id="calibration-curves-discriminative-power-independent-of-the-threshold">Calibration curves — discriminative power, independent of the threshold<a class="headerlink" href="#calibration-curves-discriminative-power-independent-of-the-threshold" title="Permanent link">&para;</a></h2>
<p>Each model's gallery carries a fitted Platt sigmoid <code>P(match | sim) = σ(a·sim + b)</code>
(embedded directly in the gallery HDF5, see <code>src/gallery/gallery_calibration.hpp</code>).
(embedded directly in the gallery HDF5, see
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/gallery_calibration.hpp"><code>src/gallery/gallery_calibration.hpp</code></a>).
Plotting all four side by side shows discriminative power directly, independent of
whatever <code>prob_threshold</code> a particular run happened to use:</p>
<p><img alt="Calibrated P(match|similarity) for all four models" src="../assets/images/calibration_curves.png" /></p>
@@ -1385,8 +1418,10 @@ whatever <code>prob_threshold</code> a particular run happened to use:</p>
variants) and the lowest P=0.5 decision boundary (similarity 0.23 vs. 0.270.31) —
it separates same-actor from different-actor pairs more confidently at a lower
similarity, consistent with it winning the full-gallery F1 comparison below.
Generated by <code>scripts/docs/calibration_chart.py</code> (requires each gallery to have
been calibrated at least once — run any replay against it first).</p>
Generated by
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/docs/calibration_chart.py"><code>scripts/docs/calibration_chart.py</code></a>
(requires each gallery to have been calibrated at least once — run any replay
against it first).</p>
<h2 id="two-effects-in-isolation-gallery-scope-and-pose-expansion">Two effects in isolation: gallery scope, and pose expansion<a class="headerlink" href="#two-effects-in-isolation-gallery-scope-and-pose-expansion" title="Permanent link">&para;</a></h2>
<p>The matrix crosses two independent variables — averaging across all 4 models
isolates each one from model choice:</p>
@@ -1427,7 +1462,8 @@ This is the single cleanest signal in the whole matrix — stronger than the mod
choice itself — which is exactly why cast-restriction becoming a real runtime
feature (not just an optimizer trick) is the top item in Caveats below.</p>
<p><strong>Pose expansion (promoting a confidently-identified track's novel-pose views into
a per-film gallery annex — <code>track_gallery.hpp</code>)</strong> is smaller and interacts with
a per-film gallery annex — <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/track_gallery.hpp"><code>src/gallery/track_gallery.hpp</code></a>)</strong>
is smaller and interacts with
scope rather than acting independently:</p>
<table>
<thead>
@@ -1518,6 +1554,19 @@ support in the app yet (see Caveats).</p>
Decided not to chase further this round (diminishing-returns judgment call) — flag
for a future sweep if it matters.</li>
</ul>
<p>The ceiling-pinning is visible in the raw search itself. Every one of the 512
DE evaluations for the winning combo, plotted over the
<code>prob_threshold</code> × <code>extinction_sec</code> plane:</p>
<p><img alt="DE search landscape: 512 evaluations over prob_threshold × extinction_sec" src="../assets/images/de_search_landscape.png" /></p>
<p>The dark band hugging the top edge <em>is</em> the finding: nearly everything scoring
well sits at <code>extinction_sec</code> ≥ 50, across a wide range of thresholds, and the
population converged into a dense cloud around the optimum (threshold ~0.700.80,
extinction pinned at the 60s bound). Short extinction windows (bottom half) are
uniformly pale — under a strict threshold there is simply no good configuration
down there. Generated by
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/docs/experiment_charts.py"><code>scripts/docs/experiment_charts.py</code></a>
from the DE trajectories (<code>experiments/trajectories/*.jsonl</code>, part of the
<code>experiment-data</code> artifact package).</p>
<h2 id="held-out-validation-the-number-that-actually-matters">Held-out validation — the number that actually matters<a class="headerlink" href="#held-out-validation-the-number-that-actually-matters" title="Permanent link">&para;</a></h2>
<p>The 16-combo matrix above is training-set fit. This is the real test: the shipped
config (<code>LVFace-B_Glint360K_full_exp</code><code>prob_threshold=0.754, anneal_sec=35.5,
@@ -1607,11 +1656,13 @@ Saints of Newark, Valerian and the City of a Thousand Planets), scored the same
</tr>
</tbody>
</table>
<p><img alt="Held-out per-film F1 vs. the training-set fit" src="../assets/images/holdout_f1_by_film.png" /></p>
<p><strong>67.4% held out vs. 75.3% on training</strong> — an ~8pp drop, and a much more informative
number than the training-set F1 alone: a <strong>37pp spread between best and worst film</strong>
(83.0% vs 46.3%). The config does not generalize uniformly.</p>
<p>Two films are outright failure cases, and rendering bounding boxes + names on the
extracted frames (<code>replay.py --raw-out</code> + <code>dump_error_frames.py --raw</code>, see
extracted frames (<code>replay.py --raw-out</code> +
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/dump_error_frames.py"><code>dump_error_frames.py</code></a><code>--raw</code>, see
Reproduce) turned what looked like a same-scene misidentification into something
more precise and more damning:</p>
<ul>
@@ -1626,28 +1677,34 @@ more precise and more damning:</p>
happen to overlap.</li>
</ul>
<p><img alt="Frozen ghost boxes over background, The Many Saints of Newark" src="../assets/images/many_saints_ghost_fpi.jpg" />
<em><code>experiments/results/holdout/frames/many_saints/fpi/fpi_t03543.jpg</code></em></p>
<em>Frame <code>many_saints/fpi/fpi_t03543.jpg</code> from the <code>montage-frames</code> artifact
package (<code>scripts/artifacts/pull_artifacts.sh montage-frames
Many_Saints_of_Newark</code>).</em></p>
<ul>
<li><strong>Downton Abbey: A New Era</strong> (large ensemble, 36-cast) has high precision (97.8%)
but recall collapses to 39.4% (FN=80084, by far the largest of the 5). The frame
below is the starkest evidence in this whole experiment: <strong>the matcher named 15
actors — all of them wrong — over a completely blank closing title card with no
faces on screen at all.</strong></li>
but recall collapses to 39.4% (FN=80084, by far the largest of the 5). Its
starkest failure happens where there is nothing to see at all: the film's hard
cut into its closing credits, where <strong>the matcher kept reporting 15 actors —
all wrong — for nearly a minute of faceless screen.</strong></li>
</ul>
<p><img alt="15 ghost boxes over a blank title card, Downton Abbey: A New Era" src="../assets/images/downton_abbey_ghost_fpi.jpg" />
<em><code>experiments/results/holdout/frames/downton_abbey/fpi/fpi_t07242.jpg</code></em></p>
<p>Both are the same mechanism, verified directly against the HDF5 dump and the raw
per-frame stream (not just inferred from the screenshot): at the Downton Abbey
title card (t=7242), the dump's own <code>face_count</code> is <strong>0 from t≈7240 onward</strong> — no
detector output at all, confirmed independently of the pipeline. Yet all 15 "wrong"
actors are still marked visible, each with the <em>exact same bbox, unchanged to the
pixel</em>, repeated every single frame back to t=7222 (verified for Hugh Bonneville:
<code>(1743.2, 0.0, 171.3, 317.8)</code> at every sampled second from 7222 through 7279+). That
is <code>SceneTrackerFunc</code>'s <code>active_[actor_idx].last_bbox</code> (<code>scene_tracker_node.hpp</code>)
being re-emitted unchanged — <strong>this is the extinction state machine working exactly
as coded</strong>, not a bug in the logic. The film cuts from a packed group shot straight
into ~40+ seconds of blank titles/credits with zero faces, and <code>extinction_sec=57.4</code>
is comfortably long enough to bridge that entire gap without expiring, so the
<p>Both are the same mechanism, and it can be <em>measured</em>, not just screenshotted.
Plotting the dump's own per-second <code>face_count</code> (detector output, independent
of the tracker) against the number of actors the tracker reports, through
Downton Abbey's cut to credits:</p>
<p><img alt="Detector face_count vs. tracker-reported actors through the cut to credits" src="../assets/images/downton_ghost_timeline.png" /></p>
<p>From the cut onward the detector sees <strong>zero faces</strong> — yet the tracker holds a
perfectly flat plateau of 15 reported identities for 56 seconds, each with the
<em>exact same bbox, unchanged to the pixel</em> (verified for Hugh Bonneville:
<code>(1743.2, 0.0, 171.3, 317.8)</code> at every sampled second from 7222 through 7279+).
The staircase on the right edge is the extinction window finally expiring,
actor by actor. That plateau is <code>SceneTrackerFunc</code>'s
<code>active_[actor_idx].last_bbox</code>
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/nodes/scene_tracker_node.hpp"><code>src/nodes/scene_tracker_node.hpp</code></a>)
being re-emitted
unchanged — <strong>the extinction state machine working exactly as coded</strong>, not a
bug in the logic. The film cuts from a packed group shot straight into ~40+
seconds of blank titles/credits with zero faces, and <code>extinction_sec=57.4</code> is
comfortably long enough to bridge that entire gap without expiring, so the
tracker faithfully keeps reporting "last known position" for a cast that is no
longer on screen at all.</p>
<p>This reframes the "long extinction window wins" DE-search pattern (see above): it
@@ -1658,11 +1715,14 @@ precisely the failure mode the <em>original</em> short-extinction-window default
was chosen to avoid. The training-set films apparently didn't have a long enough
faceless stretch after a confirmed identity to expose this; the held-out set did.</p>
<p>Frames for all three films (<code>benny_joon</code>, <code>many_saints</code>, <code>downton_abbey</code> — one strong
performer, two failure cases) are under <code>experiments/results/holdout/frames/</code>, each
performer, two failure cases) are under <code>experiments/results/holdout/frames/</code>
(not committed — pull per film with <code>scripts/artifacts/pull_artifacts.sh
montage-frames &lt;film-slug&gt;</code>), each
with a <code>manifest.json</code> listing the bucket (<code>best</code>/<code>fpi</code>/<code>fn</code>), timestamp, and
predicted vs. ground-truth actors for every dumped frame. Frames are annotated with
bounding boxes + name/confidence (green = identified, orange = unknown), matching
<code>debug_renderer_node.hpp</code>'s colour convention. Generated by
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/nodes/debug_renderer_node.hpp"><code>src/nodes/debug_renderer_node.hpp</code></a>'s
colour convention. Generated by
<code>scripts/optimizer/dump_error_frames.py --raw &lt;replay.py --raw-out output&gt;</code> (see
Reproduce).</p>
<p><code>dump_error_frames.py --interval-sec 600</code> also supports a per-N-second sweep
@@ -1710,24 +1770,28 @@ across a film's runtime rather than only at its most extreme seconds.</p>
</span><span id="__span-0-16"><a id="__codelineno-0-16" name="__codelineno-0-16" href="#__codelineno-0-16"></a><span class="w"> </span>--out<span class="w"> </span>pred.json<span class="w"> </span>--raw-out<span class="w"> </span>raw.jsonl<span class="w"> </span>--prob-threshold<span class="w"> </span><span class="m">0</span>.754<span class="w"> </span>--anneal-sec<span class="w"> </span><span class="m">35</span>.54<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-17"><a id="__codelineno-0-17" name="__codelineno-0-17" href="#__codelineno-0-17"></a><span class="w"> </span>--extinction-sec<span class="w"> </span><span class="m">57</span>.43<span class="w"> </span>--expand-gallery
</span><span id="__span-0-18"><a id="__codelineno-0-18" name="__codelineno-0-18" href="#__codelineno-0-18"></a>
</span><span id="__span-0-19"><a id="__codelineno-0-19" name="__codelineno-0-19" href="#__codelineno-0-19"></a><span class="c1"># dump example frames (best-agreement / FPI / FN) for visual inspection, annotated</span>
</span><span id="__span-0-20"><a id="__codelineno-0-20" name="__codelineno-0-20" href="#__codelineno-0-20"></a><span class="c1"># with bounding boxes + names (--raw is optional; omit for unannotated frames)</span>
</span><span id="__span-0-21"><a id="__codelineno-0-21" name="__codelineno-0-21" href="#__codelineno-0-21"></a>python3<span class="w"> </span>scripts/optimizer/dump_error_frames.py<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-22"><a id="__codelineno-0-22" name="__codelineno-0-22" href="#__codelineno-0-22"></a><span class="w"> </span>--pred<span class="w"> </span>pred.json<span class="w"> </span>--raw<span class="w"> </span>raw.jsonl<span class="w"> </span>--xray<span class="w"> </span>experiments/xray/.../&lt;xray_dir&gt;<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-23"><a id="__codelineno-0-23" name="__codelineno-0-23" href="#__codelineno-0-23"></a><span class="w"> </span>--movie<span class="w"> </span><span class="s2">&quot;&lt;path to source video&gt;&quot;</span><span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-24"><a id="__codelineno-0-24" name="__codelineno-0-24" href="#__codelineno-0-24"></a><span class="w"> </span>--gallery<span class="w"> </span>experiments/galleries/gallery_LVFace-B_Glint360K.h5<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-25"><a id="__codelineno-0-25" name="__codelineno-0-25" href="#__codelineno-0-25"></a><span class="w"> </span>--out-dir<span class="w"> </span>experiments/results/holdout/frames/&lt;name&gt;<span class="w"> </span>--n-per-bucket<span class="w"> </span><span class="m">4</span>
</span><span id="__span-0-26"><a id="__codelineno-0-26" name="__codelineno-0-26" href="#__codelineno-0-26"></a>
</span><span id="__span-0-27"><a id="__codelineno-0-27" name="__codelineno-0-27" href="#__codelineno-0-27"></a><span class="c1"># or: one best + one worst frame per 10-minute window across the whole film</span>
</span><span id="__span-0-28"><a id="__codelineno-0-28" name="__codelineno-0-28" href="#__codelineno-0-28"></a>python3<span class="w"> </span>scripts/optimizer/dump_error_frames.py<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-29"><a id="__codelineno-0-29" name="__codelineno-0-29" href="#__codelineno-0-29"></a><span class="w"> </span>--pred<span class="w"> </span>pred.json<span class="w"> </span>--raw<span class="w"> </span>raw.jsonl<span class="w"> </span>--xray<span class="w"> </span>experiments/xray/.../&lt;xray_dir&gt;<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-30"><a id="__codelineno-0-30" name="__codelineno-0-30" href="#__codelineno-0-30"></a><span class="w"> </span>--movie<span class="w"> </span><span class="s2">&quot;&lt;path to source video&gt;&quot;</span><span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-31"><a id="__codelineno-0-31" name="__codelineno-0-31" href="#__codelineno-0-31"></a><span class="w"> </span>--gallery<span class="w"> </span>experiments/galleries/gallery_LVFace-B_Glint360K.h5<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-32"><a id="__codelineno-0-32" name="__codelineno-0-32" href="#__codelineno-0-32"></a><span class="w"> </span>--out-dir<span class="w"> </span>experiments/results/holdout/frames/&lt;name&gt;_intervals<span class="w"> </span>--interval-sec<span class="w"> </span><span class="m">600</span>
</span><span id="__span-0-19"><a id="__codelineno-0-19" name="__codelineno-0-19" href="#__codelineno-0-19"></a><span class="c1"># regenerate the report&#39;s charts (16-combo ranking, DE landscape, held-out</span>
</span><span id="__span-0-20"><a id="__codelineno-0-20" name="__codelineno-0-20" href="#__codelineno-0-20"></a><span class="c1"># per-film F1, Downton ghost timeline) from the artifacts under experiments/</span>
</span><span id="__span-0-21"><a id="__codelineno-0-21" name="__codelineno-0-21" href="#__codelineno-0-21"></a>python3<span class="w"> </span>scripts/docs/experiment_charts.py<span class="w"> </span>--out-dir<span class="w"> </span>docs/assets/images
</span><span id="__span-0-22"><a id="__codelineno-0-22" name="__codelineno-0-22" href="#__codelineno-0-22"></a>
</span><span id="__span-0-23"><a id="__codelineno-0-23" name="__codelineno-0-23" href="#__codelineno-0-23"></a><span class="c1"># dump example frames (best-agreement / FPI / FN) for visual inspection, annotated</span>
</span><span id="__span-0-24"><a id="__codelineno-0-24" name="__codelineno-0-24" href="#__codelineno-0-24"></a><span class="c1"># with bounding boxes + names (--raw is optional; omit for unannotated frames)</span>
</span><span id="__span-0-25"><a id="__codelineno-0-25" name="__codelineno-0-25" href="#__codelineno-0-25"></a>python3<span class="w"> </span>scripts/optimizer/dump_error_frames.py<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-26"><a id="__codelineno-0-26" name="__codelineno-0-26" href="#__codelineno-0-26"></a><span class="w"> </span>--pred<span class="w"> </span>pred.json<span class="w"> </span>--raw<span class="w"> </span>raw.jsonl<span class="w"> </span>--xray<span class="w"> </span>experiments/xray/.../&lt;xray_dir&gt;<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-27"><a id="__codelineno-0-27" name="__codelineno-0-27" href="#__codelineno-0-27"></a><span class="w"> </span>--movie<span class="w"> </span><span class="s2">&quot;&lt;path to source video&gt;&quot;</span><span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-28"><a id="__codelineno-0-28" name="__codelineno-0-28" href="#__codelineno-0-28"></a><span class="w"> </span>--gallery<span class="w"> </span>experiments/galleries/gallery_LVFace-B_Glint360K.h5<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-29"><a id="__codelineno-0-29" name="__codelineno-0-29" href="#__codelineno-0-29"></a><span class="w"> </span>--out-dir<span class="w"> </span>experiments/results/holdout/frames/&lt;name&gt;<span class="w"> </span>--n-per-bucket<span class="w"> </span><span class="m">4</span>
</span><span id="__span-0-30"><a id="__codelineno-0-30" name="__codelineno-0-30" href="#__codelineno-0-30"></a>
</span><span id="__span-0-31"><a id="__codelineno-0-31" name="__codelineno-0-31" href="#__codelineno-0-31"></a><span class="c1"># or: one best + one worst frame per 10-minute window across the whole film</span>
</span><span id="__span-0-32"><a id="__codelineno-0-32" name="__codelineno-0-32" href="#__codelineno-0-32"></a>python3<span class="w"> </span>scripts/optimizer/dump_error_frames.py<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-33"><a id="__codelineno-0-33" name="__codelineno-0-33" href="#__codelineno-0-33"></a><span class="w"> </span>--pred<span class="w"> </span>pred.json<span class="w"> </span>--raw<span class="w"> </span>raw.jsonl<span class="w"> </span>--xray<span class="w"> </span>experiments/xray/.../&lt;xray_dir&gt;<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-34"><a id="__codelineno-0-34" name="__codelineno-0-34" href="#__codelineno-0-34"></a><span class="w"> </span>--movie<span class="w"> </span><span class="s2">&quot;&lt;path to source video&gt;&quot;</span><span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-35"><a id="__codelineno-0-35" name="__codelineno-0-35" href="#__codelineno-0-35"></a><span class="w"> </span>--gallery<span class="w"> </span>experiments/galleries/gallery_LVFace-B_Glint360K.h5<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-36"><a id="__codelineno-0-36" name="__codelineno-0-36" href="#__codelineno-0-36"></a><span class="w"> </span>--out-dir<span class="w"> </span>experiments/results/holdout/frames/&lt;name&gt;_intervals<span class="w"> </span>--interval-sec<span class="w"> </span><span class="m">600</span>
</span></code></pre></div>
<p>See also: <code>docs/optimizer-experiments.md</code> (prior round, superseded metric),
<code>experiments/SESSION_STATE.md</code>, and memory: kpn-python-replay-optimizer,
gallery-coverage-gap, xray-validation-results, per-scene-presence-eval-design.</p>
<p>See also: <a href="../optimizer-experiments/">the prior optimizer round</a> (superseded
metric) and the session log
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/experiments/SESSION_STATE.md"><code>experiments/SESSION_STATE.md</code></a>.</p>
@@ -1758,6 +1822,46 @@ gallery-coverage-gap, xray-validation-results, per-scene-presence-eval-design.</
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../lvface-deep-dive/" class="md-footer__link md-footer__link--prev" aria-label="Previous: LVFace Deep Dive">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</div>
<div class="md-footer__title">
<span class="md-footer__direction">
Previous
</span>
<div class="md-ellipsis">
LVFace Deep Dive
</div>
</div>
</a>
<a href="../optimizer-experiments/" class="md-footer__link md-footer__link--next" aria-label="Next: Optimizer Experiments (prior round)">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Optimizer Experiments (prior round)
</div>
</div>
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -1783,7 +1887,7 @@ gallery-coverage-gap, xray-validation-results, per-scene-presence-eval-design.</
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
+81 -15
View File
@@ -10,8 +10,10 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/optimizer-experiments/">
<link rel="prev" href="../rep4-optimizer-results/">
<link rel="prev" href="../model-bakeoff/">
<link rel="next" href="../service-conversion/">
@@ -51,6 +53,8 @@
<link rel="stylesheet" href="../stylesheets/extra.css">
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -67,7 +71,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -123,7 +127,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -246,13 +264,13 @@
<li class="md-tabs__item">
<a href="../rep4-optimizer-results/" class="md-tabs__link">
<a href="../model-bakeoff/" class="md-tabs__link">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -549,14 +567,14 @@
<li class="md-nav__item">
<a href="../rep4-optimizer-results/" class="md-nav__link">
<a href="../model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -936,11 +954,15 @@ GT set = actors X-Ray lists for that scene. Per scene TP/FP/FN, then:</p>
scenes count more) → <strong>equal-weight mean across movies</strong> (macro; each film counts
the same regardless of length). This is the DE objective.</li>
</ul>
<p>Implemented in <code>scripts/optimizer/scene_score.py</code>.</p>
<p>Implemented in <code>scripts/optimizer/scene_score.py</code> — since <strong>removed</strong> along
with this metric; its per-second successor is
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>
(see the <a href="../model-bakeoff/">bake-off round</a>).</p>
<h2 id="the-gallery-coverage-gap">The gallery coverage gap<a class="headerlink" href="#the-gallery-coverage-gap" title="Permanent link">&para;</a></h2>
<p>Diagnosing low recall: only <strong>131 of 392</strong> X-Ray cast were in the gallery (33%). Every
in-gallery actor HAD embeddings (gallery well-formed) — the gap was pure coverage.
<code>scripts/optimizer/fetch_missing_actors.py</code> recovers missing actors:
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/fetch_missing_actors.py"><code>scripts/optimizer/fetch_missing_actors.py</code></a>
recovers missing actors:
<code>nm-id → TMDB /find external_ids → /person/{id}/images → download → embed (sae_embed)</code>,
with a <code>--wikidata</code> fallback (P345→P18 Commons photo).</p>
<ul>
@@ -958,7 +980,8 @@ credits them as cast-in-scene (incl. off-camera/background), but their face neve
appears clearly for the pipeline to detect. This is a fundamental ceiling of a
face-recognition pipeline vs X-Ray's presence semantics, not a fixable gap.</p>
<h2 id="optimizer">Optimizer<a class="headerlink" href="#optimizer" title="Permanent link">&para;</a></h2>
<p><code>scripts/optimizer/optimize.py</code>scipy <code>differential_evolution</code> over the knob space,
<p><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/optimize.py"><code>scripts/optimizer/optimize.py</code></a>
— scipy <code>differential_evolution</code> over the knob space,
each candidate = full replay of all films through the <strong>real</strong> C++ nodes (see the
KPN replay architecture below) scored by the metric above. Global objective (one
config for all films, not per-film).</p>
@@ -994,7 +1017,8 @@ the tightly-converged knobs were adopted as defaults.</p>
<h2 id="replay-architecture-how-the-sweep-is-cheap">Replay architecture (how the sweep is cheap)<a class="headerlink" href="#replay-architecture-how-the-sweep-is-cheap" title="Permanent link">&para;</a></h2>
<p>The optimizer never re-decodes video. <code>scene_analyze --dump-embeddings out.h5</code> runs the
expensive half once (decode→detect→align→embed) and dumps per-frame face embeddings
+ metadata to HDF5 (<code>scripts/optimizer/SCHEMA.md</code>). <code>scripts/optimizer/replay.py</code> then
+ metadata to HDF5 (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/SCHEMA.md"><code>scripts/optimizer/SCHEMA.md</code></a>).
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/replay.py"><code>scripts/optimizer/replay.py</code></a> then
replays that dump through the <strong>real</strong> C++ <code>face_tracker → identity_matcher →
scene_tracker</code> assembled in a Python KPN network (<code>sae_kpn</code> nanobind module), varying
Config knobs freely — no GPU embedding, no decode. Verified BYTE-EXACT against
@@ -1010,10 +1034,12 @@ dump floor is 0.5).</p>
</span><span id="__span-0-6"><a id="__codelineno-0-6" name="__codelineno-0-6" href="#__codelineno-0-6"></a><span class="w"> </span>--params<span class="w"> </span>prob_threshold:0.5:0.999<span class="w"> </span>anneal_sec:1:30<span class="w"> </span>extinction_sec:1:15<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-7"><a id="__codelineno-0-7" name="__codelineno-0-7" href="#__codelineno-0-7"></a><span class="w"> </span>--popsize<span class="w"> </span><span class="m">8</span><span class="w"> </span>--maxiter<span class="w"> </span><span class="m">20</span><span class="w"> </span>--trajectory<span class="w"> </span>traj.jsonl<span class="w"> </span>--out<span class="w"> </span>opt.json
</span><span id="__span-0-8"><a id="__codelineno-0-8" name="__codelineno-0-8" href="#__codelineno-0-8"></a><span class="c1"># 4. score a fixed config / validate on a held-out set</span>
</span><span id="__span-0-9"><a id="__codelineno-0-9" name="__codelineno-0-9" href="#__codelineno-0-9"></a>python<span class="w"> </span>scripts/optimizer/score_config.py<span class="w"> </span>--manifest<span class="w"> </span>heldout.json<span class="w"> </span>--gallery<span class="w"> </span>gallery.json<span class="w"> </span><span class="se">\</span>
</span><span id="__span-0-10"><a id="__codelineno-0-10" name="__codelineno-0-10" href="#__codelineno-0-10"></a><span class="w"> </span>--config<span class="w"> </span><span class="s1">&#39;{&quot;prob_threshold&quot;:0.76,&quot;extinction_sec&quot;:1.5,&quot;anneal_sec&quot;:10}&#39;</span>
</span><span id="__span-0-9"><a id="__codelineno-0-9" name="__codelineno-0-9" href="#__codelineno-0-9"></a><span class="c1"># (historical: score_config.py and scene_score.py were removed with the</span>
</span><span id="__span-0-10"><a id="__codelineno-0-10" name="__codelineno-0-10" href="#__codelineno-0-10"></a><span class="c1"># scene-union metric — use scripts/optimizer/second_score.py, per-second)</span>
</span><span id="__span-0-11"><a id="__codelineno-0-11" name="__codelineno-0-11" href="#__codelineno-0-11"></a>python<span class="w"> </span>scripts/optimizer/second_score.py<span class="w"> </span>--help
</span></code></pre></div>
<p>See also memory: kpn-python-replay-optimizer, gallery-coverage-gap, xray-validation-*.</p>
<p>Superseded by the <a href="../model-bakeoff/">model bake-off + re-tune</a>, which
replaced this round's scene-union metric with per-second scoring.</p>
@@ -1044,6 +1070,46 @@ dump floor is 0.5).</p>
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Model Bake-off &amp; Re-tune (full log)">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</div>
<div class="md-footer__title">
<span class="md-footer__direction">
Previous
</span>
<div class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</div>
</div>
</a>
<a href="../service-conversion/" class="md-footer__link md-footer__link--next" aria-label="Next: Service Conversion (proposal)">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Service Conversion (proposal)
</div>
</div>
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -1069,7 +1135,7 @@ dump floor is 0.5).</p>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
+73 -12
View File
@@ -10,6 +10,8 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/pose-expansion/">
<link rel="prev" href="../gallery-scope/">
@@ -51,6 +53,8 @@
<link rel="stylesheet" href="../stylesheets/extra.css">
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -67,7 +71,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -123,7 +127,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -248,13 +266,13 @@
<li class="md-tabs__item">
<a href="../rep4-optimizer-results/" class="md-tabs__link">
<a href="../model-bakeoff/" class="md-tabs__link">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -645,14 +663,14 @@
<li class="md-nav__item">
<a href="../rep4-optimizer-results/" class="md-nav__link">
<a href="../model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -807,7 +825,8 @@
<h1 id="pose-expansion-does-learning-new-poses-mid-film-help">Pose expansion: does "learning" new poses mid-film help?<a class="headerlink" href="#pose-expansion-does-learning-new-poses-mid-film-help" title="Permanent link">&para;</a></h1>
<p><code>expand_gallery</code> (<code>src/gallery/track_gallery.hpp</code>) promotes a confidently-identified
<p><code>expand_gallery</code> (<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/track_gallery.hpp"><code>src/gallery/track_gallery.hpp</code></a>)
promotes a confidently-identified
track's novel-pose reference views into a per-film, in-memory gallery annex — the
idea being that once the pipeline is sure who someone is, a pose it hasn't seen
before (turned head, different lighting) becomes a free extra reference for
@@ -859,8 +878,8 @@ gallery.</p>
<p>In <code>restricted</code> mode (matcher's candidate set capped to the film's own credited
cast) expansion looked like a clean win: +1.8pp F1, +3.2pp recall, misID actually
lower. In <code>full</code> mode it looked flat-to-costly: ~0 F1 change, recall +1.4pp, but
misID roughly quadrupled (209 → 864) — see <code>rep4-optimizer-results.md</code> for the
per-model breakdown. That's the number that motivated this page: <strong>does turning
misID roughly quadrupled (209 → 864) — see the
<a href="../model-bakeoff/">bake-off experiment log</a> for the per-model breakdown. That's the number that motivated this page: <strong>does turning
expansion on actually change what gets recognised, frame by frame, or is the
aggregate F1 shift something else?</strong></p>
<h2 id="held-out-test-does-it-reproduce">Held-out test: does it reproduce?<a class="headerlink" href="#held-out-test-does-it-reproduce" title="Permanent link">&para;</a></h2>
@@ -952,7 +971,8 @@ mode) doesn't reproduce on held-out data — at minimum it's far smaller than th
training-set numbers suggested, and plausibly it's sampling variation from only
4 training films rather than a real, generalizable mechanism. This doesn't mean
<code>expand_gallery</code> never does anything (the mechanism is real — see
<code>track_gallery.hpp</code>'s promotion logging: tracks <em>do</em> get confirmed and views <em>do</em>
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/gallery/track_gallery.hpp"><code>track_gallery.hpp</code></a>'s
promotion logging: tracks <em>do</em> get confirmed and views <em>do</em>
get promoted into the annex on every film tested), only that <strong>whatever effect
it has on final per-second identification was too small to detect against 5
held-out films</strong> with this scoring method. A cleaner test would need either many
@@ -960,7 +980,8 @@ more held-out films or a metric that can see the annex's direct contribution
(e.g. tagging which reference embedding won each match), neither of which this
pass had budget for.</p>
<p><strong>Practical takeaway</strong>: don't treat the training-set <code>exp</code> vs <code>noexp</code> numbers in
<code>rep4-optimizer-results.md</code> as proof that expansion changes real-world behavior
the <a href="../model-bakeoff/">bake-off experiment log</a> as proof that expansion
changes real-world behavior
in either direction — on the evidence gathered so far, it doesn't move the
needle enough to see.</p>
@@ -993,6 +1014,46 @@ needle enough to see.</p>
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../gallery-scope/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Gallery Scope (Full vs. Limited)">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</div>
<div class="md-footer__title">
<span class="md-footer__direction">
Previous
</span>
<div class="md-ellipsis">
Gallery Scope (Full vs. Limited)
</div>
</div>
</a>
<a href="../lvface-deep-dive/" class="md-footer__link md-footer__link--next" aria-label="Next: LVFace Deep Dive">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
LVFace Deep Dive
</div>
</div>
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -1018,7 +1079,7 @@ needle enough to see.</p>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
File diff suppressed because one or more lines are too long
+53 -11
View File
@@ -10,6 +10,8 @@
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/service-conversion/">
<link rel="prev" href="../optimizer-experiments/">
@@ -49,6 +51,8 @@
<link rel="stylesheet" href="../stylesheets/extra.css">
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
@@ -65,7 +69,7 @@
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo">
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
@@ -121,7 +125,21 @@
<input class="md-option" data-md-color-media="" data-md-color-scheme="slate" data-md-color-primary="indigo" data-md-color-accent="indigo" aria-hidden="true" type="radio" name="__palette" id="__palette_0">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
@@ -244,13 +262,13 @@
<li class="md-tabs__item">
<a href="../rep4-optimizer-results/" class="md-tabs__link">
<a href="../model-bakeoff/" class="md-tabs__link">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
</a>
</li>
@@ -547,14 +565,14 @@
<li class="md-nav__item">
<a href="../rep4-optimizer-results/" class="md-nav__link">
<a href="../model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Rep4 Bake-off & Re-tune (full log)
Model Bake-off & Re-tune (full log)
@@ -1017,7 +1035,7 @@ lock-gating, not new pipeline logic.</p>
</tr>
<tr>
<td>Backend selection</td>
<td><code>CMakeLists.txt</code> (<code>SAE_INFERENCE_BACKEND</code>, <code>SAE_GEMM_BACKEND</code>)</td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/CMakeLists.txt"><code>CMakeLists.txt</code></a> (<code>SAE_INFERENCE_BACKEND</code>, <code>SAE_GEMM_BACKEND</code>)</td>
<td>ORT/TRT + ROCm/CUDA, chosen <strong>at build time</strong></td>
</tr>
<tr>
@@ -1027,7 +1045,7 @@ lock-gating, not new pipeline logic.</p>
</tr>
<tr>
<td>Worker loop</td>
<td><code>run_from_jellyfin.py --worker</code></td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/run_from_jellyfin.py"><code>scripts/run_from_jellyfin.py</code></a><code>--worker</code></td>
<td>Poll Pending → run <code>scene_analyze</code> → push results</td>
</tr>
<tr>
@@ -1037,12 +1055,12 @@ lock-gating, not new pipeline logic.</p>
</tr>
<tr>
<td>Incremental gallery</td>
<td><code>make_jellyfin_gallery.py --merge</code></td>
<td><a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/make_jellyfin_gallery.py"><code>scripts/make_jellyfin_gallery.py</code></a><code>--merge</code></td>
<td>Embeds only cast not already in the gallery</td>
</tr>
<tr>
<td>Secrets loader</td>
<td><code>.env</code> via <code>sae_env.py</code></td>
<td><code>.env</code> via <a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/sae_env.py"><code>scripts/sae_env.py</code></a></td>
<td><code>JELLYFIN_URL</code>, <code>JELLYFIN_API_KEY</code>, <code>TMDB_API_KEY</code></td>
</tr>
</tbody>
@@ -1253,6 +1271,30 @@ same filesystem and GPU as everything else on the box.</p>
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../optimizer-experiments/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Optimizer Experiments (prior round)">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</div>
<div class="md-footer__title">
<span class="md-footer__direction">
Previous
</span>
<div class="md-ellipsis">
Optimizer Experiments (prior round)
</div>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
@@ -1278,7 +1320,7 @@ same filesystem and GPU as everything else on the box.</p>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
+32
View File
@@ -1,3 +1,35 @@
<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/</loc>
<lastmod>2026-07-19</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/best-model/</loc>
<lastmod>2026-07-19</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/gallery-scope/</loc>
<lastmod>2026-07-19</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/lvface-deep-dive/</loc>
<lastmod>2026-07-19</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/model-bakeoff/</loc>
<lastmod>2026-07-19</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/optimizer-experiments/</loc>
<lastmod>2026-07-19</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/pose-expansion/</loc>
<lastmod>2026-07-19</lastmod>
</url>
<url>
<loc>https://pages.tourolle.paris/dtourolle/scene-actor-extraction/service-conversion/</loc>
<lastmod>2026-07-19</lastmod>
</url>
</urlset>
BIN
View File
Binary file not shown.
+24
View File
@@ -0,0 +1,24 @@
/* Frames and charts presented as cards */
.md-typeset img {
border-radius: 6px;
}
.md-typeset p > img:only-child {
box-shadow: 0 2px 12px rgba(0, 0, 0, 0.35);
}
/* Captions: an italic-only paragraph immediately after an image reads as a
figure caption — centered, small, muted. Falls back to plain italics in
browsers without :has(). */
.md-typeset p:has(> img:only-child) + p > em:only-child {
display: block;
text-align: center;
font-size: 0.72rem;
line-height: 1.5;
color: var(--md-default-fg-color--light);
margin-top: -0.4rem;
}
/* Slightly tighter hero image spacing on the landing page */
.md-typeset h1 + p + p > img:only-child {
margin-top: 0.2rem;
}