Files
scene-actor-extraction/lvface-deep-dive/index.html
T
2026-07-19 22:28:14 +02:00

1194 lines
37 KiB
HTML

<!doctype html>
<html lang="en" class="no-js">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<meta name="description" content="Face-recognition pipeline for finding on-screen actor presence in film/TV, built on KPN++">
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/lvface-deep-dive/">
<link rel="prev" href="../pose-expansion/">
<link rel="next" href="../model-bakeoff/">
<link rel="icon" href="../assets/images/favicon.png">
<meta name="generator" content="mkdocs-1.6.1, mkdocs-material-9.7.7">
<title>LVFace Deep Dive - scene-actor-extraction</title>
<link rel="stylesheet" href="../assets/stylesheets/main.ec1eaa64.min.css">
<link rel="stylesheet" href="../assets/stylesheets/palette.ab4e12ef.min.css">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link rel="stylesheet" href="https://fonts.googleapis.com/css?family=Roboto:300,300i,400,400i,700,700i%7CRoboto+Mono:400,400i,700,700i&display=fallback">
<style>:root{--md-text-font:"Roboto";--md-code-font:"Roboto Mono"}</style>
<link rel="stylesheet" href="../stylesheets/extra.css">
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
</head>
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
<input class="md-toggle" data-md-toggle="search" type="checkbox" id="__search" autocomplete="off">
<label class="md-overlay" for="__drawer"></label>
<div data-md-component="skip">
<a href="#deep-dive-lvface-b-glint360k" class="md-skip">
Skip to content
</a>
</div>
<div data-md-component="announce">
</div>
<header class="md-header" data-md-component="header">
<nav class="md-header__inner md-grid" aria-label="Header">
<a href=".." title="scene-actor-extraction" class="md-header__button md-logo" aria-label="scene-actor-extraction" data-md-component="logo">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 8a3 3 0 0 0 3-3 3 3 0 0 0-3-3 3 3 0 0 0-3 3 3 3 0 0 0 3 3m0 3.54C9.64 9.35 6.5 8 3 8v11c3.5 0 6.64 1.35 9 3.54 2.36-2.19 5.5-3.54 9-3.54V8c-3.5 0-6.64 1.35-9 3.54"/></svg>
</a>
<label class="md-header__button md-icon" for="__drawer">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M3 6h18v2H3zm0 5h18v2H3zm0 5h18v2H3z"/></svg>
</label>
<div class="md-header__title" data-md-component="header-title">
<div class="md-header__ellipsis">
<div class="md-header__topic">
<span class="md-ellipsis">
scene-actor-extraction
</span>
</div>
<div class="md-header__topic" data-md-component="header-topic">
<span class="md-ellipsis">
LVFace Deep Dive
</span>
</div>
</div>
</div>
<form class="md-header__option" data-md-component="palette">
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
</label>
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
</label>
</form>
<script>var palette=__md_get("__palette");if(palette&&palette.color){if("(prefers-color-scheme)"===palette.color.media){var media=matchMedia("(prefers-color-scheme: light)"),input=document.querySelector(media.matches?"[data-md-color-media='(prefers-color-scheme: light)']":"[data-md-color-media='(prefers-color-scheme: dark)']");palette.color.media=input.getAttribute("data-md-color-media"),palette.color.scheme=input.getAttribute("data-md-color-scheme"),palette.color.primary=input.getAttribute("data-md-color-primary"),palette.color.accent=input.getAttribute("data-md-color-accent")}for(var[key,value]of Object.entries(palette.color))document.body.setAttribute("data-md-color-"+key,value)}</script>
<label class="md-header__button md-icon" for="__search">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M9.5 3A6.5 6.5 0 0 1 16 9.5c0 1.61-.59 3.09-1.56 4.23l.27.27h.79l5 5-1.5 1.5-5-5v-.79l-.27-.27A6.52 6.52 0 0 1 9.5 16 6.5 6.5 0 0 1 3 9.5 6.5 6.5 0 0 1 9.5 3m0 2C7 5 5 7 5 9.5S7 14 9.5 14 14 12 14 9.5 12 5 9.5 5"/></svg>
</label>
<div class="md-search" data-md-component="search" role="dialog">
<label class="md-search__overlay" for="__search"></label>
<div class="md-search__inner" role="search">
<form class="md-search__form" name="search">
<input type="text" class="md-search__input" name="query" aria-label="Search" placeholder="Search" autocapitalize="off" autocorrect="off" autocomplete="off" spellcheck="false" data-md-component="search-query" required>
<label class="md-search__icon md-icon" for="__search">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M9.5 3A6.5 6.5 0 0 1 16 9.5c0 1.61-.59 3.09-1.56 4.23l.27.27h.79l5 5-1.5 1.5-5-5v-.79l-.27-.27A6.52 6.52 0 0 1 9.5 16 6.5 6.5 0 0 1 3 9.5 6.5 6.5 0 0 1 9.5 3m0 2C7 5 5 7 5 9.5S7 14 9.5 14 14 12 14 9.5 12 5 9.5 5"/></svg>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</label>
<nav class="md-search__options" aria-label="Search">
<button type="reset" class="md-search__icon md-icon" title="Clear" aria-label="Clear" tabindex="-1">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M19 6.41 17.59 5 12 10.59 6.41 5 5 6.41 10.59 12 5 17.59 6.41 19 12 13.41 17.59 19 19 17.59 13.41 12z"/></svg>
</button>
</nav>
</form>
<div class="md-search__output">
<div class="md-search__scrollwrap" tabindex="0" data-md-scrollfix>
<div class="md-search-result" data-md-component="search-result">
<div class="md-search-result__meta">
Initializing search
</div>
<ol class="md-search-result__list" role="presentation"></ol>
</div>
</div>
</div>
</div>
</div>
<div class="md-header__source">
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction" title="Go to repository" class="md-source" data-md-component="source">
<div class="md-source__icon md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 448 512"><!--! Font Awesome Free 7.1.0 by @fontawesome - https://fontawesome.com License - https://fontawesome.com/license/free (Icons: CC BY 4.0, Fonts: SIL OFL 1.1, Code: MIT License) Copyright 2025 Fonticons, Inc.--><path d="M439.6 236.1 244 40.5c-5.4-5.5-12.8-8.5-20.4-8.5s-15 3-20.4 8.4L162.5 81l51.5 51.5c27.1-9.1 52.7 16.8 43.4 43.7l49.7 49.7c34.2-11.8 61.2 31 35.5 56.7-26.5 26.5-70.2-2.9-56-37.3L240.3 199v121.9c25.3 12.5 22.3 41.8 9.1 55-6.4 6.4-15.2 10.1-24.3 10.1s-17.8-3.6-24.3-10.1c-17.6-17.6-11.1-46.9 11.2-56v-123c-20.8-8.5-24.6-30.7-18.6-45L142.6 101 8.5 235.1C3 240.6 0 247.9 0 255.5s3 15 8.5 20.4l195.6 195.7c5.4 5.4 12.7 8.4 20.4 8.4s15-3 20.4-8.4l194.7-194.7c5.4-5.4 8.4-12.8 8.4-20.4s-3-15-8.4-20.4"/></svg>
</div>
<div class="md-source__repository">
dtourolle/scene-actor-extraction
</div>
</a>
</div>
</nav>
</header>
<div class="md-container" data-md-component="container">
<nav class="md-tabs" aria-label="Tabs" data-md-component="tabs">
<div class="md-grid">
<ul class="md-tabs__list">
<li class="md-tabs__item">
<a href=".." class="md-tabs__link">
Home
</a>
</li>
<li class="md-tabs__item md-tabs__item--active">
<a href="../best-model/" class="md-tabs__link">
Findings
</a>
</li>
<li class="md-tabs__item">
<a href="../model-bakeoff/" class="md-tabs__link">
Model Bake-off & Re-tune (full log)
</a>
</li>
<li class="md-tabs__item">
<a href="../optimizer-experiments/" class="md-tabs__link">
Optimizer Experiments (prior round)
</a>
</li>
<li class="md-tabs__item">
<a href="../service-conversion/" class="md-tabs__link">
Service Conversion (proposal)
</a>
</li>
</ul>
</div>
</nav>
<main class="md-main" data-md-component="main">
<div class="md-main__inner md-grid">
<div class="md-sidebar md-sidebar--primary" data-md-component="sidebar" data-md-type="navigation" >
<div class="md-sidebar__scrollwrap">
<div class="md-sidebar__inner">
<nav class="md-nav md-nav--primary md-nav--lifted" aria-label="Navigation" data-md-level="0">
<label class="md-nav__title" for="__drawer">
<a href=".." title="scene-actor-extraction" class="md-nav__button md-logo" aria-label="scene-actor-extraction" data-md-component="logo">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 8a3 3 0 0 0 3-3 3 3 0 0 0-3-3 3 3 0 0 0-3 3 3 3 0 0 0 3 3m0 3.54C9.64 9.35 6.5 8 3 8v11c3.5 0 6.64 1.35 9 3.54 2.36-2.19 5.5-3.54 9-3.54V8c-3.5 0-6.64 1.35-9 3.54"/></svg>
</a>
scene-actor-extraction
</label>
<div class="md-nav__source">
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction" title="Go to repository" class="md-source" data-md-component="source">
<div class="md-source__icon md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 448 512"><!--! Font Awesome Free 7.1.0 by @fontawesome - https://fontawesome.com License - https://fontawesome.com/license/free (Icons: CC BY 4.0, Fonts: SIL OFL 1.1, Code: MIT License) Copyright 2025 Fonticons, Inc.--><path d="M439.6 236.1 244 40.5c-5.4-5.5-12.8-8.5-20.4-8.5s-15 3-20.4 8.4L162.5 81l51.5 51.5c27.1-9.1 52.7 16.8 43.4 43.7l49.7 49.7c34.2-11.8 61.2 31 35.5 56.7-26.5 26.5-70.2-2.9-56-37.3L240.3 199v121.9c25.3 12.5 22.3 41.8 9.1 55-6.4 6.4-15.2 10.1-24.3 10.1s-17.8-3.6-24.3-10.1c-17.6-17.6-11.1-46.9 11.2-56v-123c-20.8-8.5-24.6-30.7-18.6-45L142.6 101 8.5 235.1C3 240.6 0 247.9 0 255.5s3 15 8.5 20.4l195.6 195.7c5.4 5.4 12.7 8.4 20.4 8.4s15-3 20.4-8.4l194.7-194.7c5.4-5.4 8.4-12.8 8.4-20.4s-3-15-8.4-20.4"/></svg>
</div>
<div class="md-source__repository">
dtourolle/scene-actor-extraction
</div>
</a>
</div>
<ul class="md-nav__list" data-md-scrollfix>
<li class="md-nav__item">
<a href=".." class="md-nav__link">
<span class="md-ellipsis">
Home
</span>
</a>
</li>
<li class="md-nav__item md-nav__item--active md-nav__item--section md-nav__item--nested">
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_2" checked>
<label class="md-nav__link" for="__nav_2" id="__nav_2_label" tabindex="">
<span class="md-ellipsis">
Findings
</span>
<span class="md-nav__icon md-icon"></span>
</label>
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_2_label" aria-expanded="true">
<label class="md-nav__title" for="__nav_2">
<span class="md-nav__icon md-icon"></span>
Findings
</label>
<ul class="md-nav__list" data-md-scrollfix>
<li class="md-nav__item">
<a href="../best-model/" class="md-nav__link">
<span class="md-ellipsis">
Best Model
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../gallery-scope/" class="md-nav__link">
<span class="md-ellipsis">
Gallery Scope (Full vs. Limited)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../pose-expansion/" class="md-nav__link">
<span class="md-ellipsis">
Pose Expansion
</span>
</a>
</li>
<li class="md-nav__item md-nav__item--active">
<input class="md-nav__toggle md-toggle" type="checkbox" id="__toc">
<label class="md-nav__link md-nav__link--active" for="__toc">
<span class="md-ellipsis">
LVFace Deep Dive
</span>
<span class="md-nav__icon md-icon"></span>
</label>
<a href="./" class="md-nav__link md-nav__link--active">
<span class="md-ellipsis">
LVFace Deep Dive
</span>
</a>
<nav class="md-nav md-nav--secondary" aria-label="Table of contents">
<label class="md-nav__title" for="__toc">
<span class="md-nav__icon md-icon"></span>
Table of contents
</label>
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#what-good-looks-like" class="md-nav__link">
<span class="md-ellipsis">
What good looks like
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#training-vs-held-out-the-generalization-gap" class="md-nav__link">
<span class="md-ellipsis">
Training vs. held-out: the generalization gap
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" class="md-nav__link">
<span class="md-ellipsis">
Mechanism 1: extinction bridging — usually right, wrong at hard cuts
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#mechanism-2-the-face-vs-presence-ceiling" class="md-nav__link">
<span class="md-ellipsis">
Mechanism 2: the face-vs-presence ceiling
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#where-lvface-beat-x-ray" class="md-nav__link">
<span class="md-ellipsis">
Where LVFace beat X-Ray
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#summary" class="md-nav__link">
<span class="md-ellipsis">
Summary
</span>
</a>
</li>
</ul>
</nav>
</li>
</ul>
</nav>
</li>
<li class="md-nav__item">
<a href="../model-bakeoff/" class="md-nav__link">
<span class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../optimizer-experiments/" class="md-nav__link">
<span class="md-ellipsis">
Optimizer Experiments (prior round)
</span>
</a>
</li>
<li class="md-nav__item">
<a href="../service-conversion/" class="md-nav__link">
<span class="md-ellipsis">
Service Conversion (proposal)
</span>
</a>
</li>
</ul>
</nav>
</div>
</div>
</div>
<div class="md-sidebar md-sidebar--secondary" data-md-component="sidebar" data-md-type="toc" >
<div class="md-sidebar__scrollwrap">
<div class="md-sidebar__inner">
<nav class="md-nav md-nav--secondary" aria-label="Table of contents">
<label class="md-nav__title" for="__toc">
<span class="md-nav__icon md-icon"></span>
Table of contents
</label>
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
<li class="md-nav__item">
<a href="#what-good-looks-like" class="md-nav__link">
<span class="md-ellipsis">
What good looks like
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#training-vs-held-out-the-generalization-gap" class="md-nav__link">
<span class="md-ellipsis">
Training vs. held-out: the generalization gap
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" class="md-nav__link">
<span class="md-ellipsis">
Mechanism 1: extinction bridging — usually right, wrong at hard cuts
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#mechanism-2-the-face-vs-presence-ceiling" class="md-nav__link">
<span class="md-ellipsis">
Mechanism 2: the face-vs-presence ceiling
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#where-lvface-beat-x-ray" class="md-nav__link">
<span class="md-ellipsis">
Where LVFace beat X-Ray
</span>
</a>
</li>
<li class="md-nav__item">
<a href="#summary" class="md-nav__link">
<span class="md-ellipsis">
Summary
</span>
</a>
</li>
</ul>
</nav>
</div>
</div>
</div>
<div class="md-content" data-md-component="content">
<article class="md-content__inner md-typeset">
<h1 id="deep-dive-lvface-b-glint360k">Deep dive: LVFace-B Glint360K<a class="headerlink" href="#deep-dive-lvface-b-glint360k" title="Permanent link">&para;</a></h1>
<p>LVFace won the model bake-off (see <a href="../best-model/">Which model is best?</a>) and is
the shipped default embedder. This page is the honest accounting of how it
actually performs — what a good second looks like, where the errors actually
come from, and two cases where the ground truth itself is wrong and LVFace is
right.</p>
<div class="admonition note">
<p class="admonition-title">How to read the frames on this page</p>
<p>The top is the film frame, with a box and name on every face the pipeline
identified. The bottom panels are the per-second verdict against X-Ray:
<strong>Onscreen</strong> lists faces named in the frame, <strong>Offscreen</strong> lists cast
X-Ray marks present in the scene without a visible face — presence
carried by the tracker's windows, not by a detection. Colors are the
score: <span style="color:#0ca30c"><strong>green</strong></span> = correct (TPI),
<span style="color:#eb6834"><strong>orange</strong></span> = wrong (FPI),
<span style="color:#3987e5"><strong>blue</strong></span> = missed (FN).</p>
</div>
<h2 id="what-good-looks-like">What good looks like<a class="headerlink" href="#what-good-looks-like" title="Permanent link">&para;</a></h2>
<p><img alt="Wedding couple correctly identified, Downton Abbey: A New Era" src="../assets/images/downton_wedding_couple.jpg" /></p>
<p>Six faces on screen, all six named correctly — including Penelope Wilton at the
edge of the pews and a half-occluded Michelle Dockery — while thirteen more
cast members X-Ray marks present in the scene are correctly carried as
"Offscreen" by their presence windows. One miss in the whole frame: Maggie
Smith (blue). Score for this second: 0.86.</p>
<p><img alt="19 of 20 correct in the funeral crowd" src="../assets/images/downton_funeral_19of20.jpg" /></p>
<p>The same film's funeral gathering: mourning dress, hats, half the faces turned.
<strong>Nineteen of the twenty cast X-Ray lists for this scene are scored correctly</strong>
— seven named on screen at up to 100% confidence, twelve more correctly held
as present off-screen.</p>
<p>And the pipeline doesn't need the face to be <em>real</em>:</p>
<p><img alt="Herbie Hancock identified on an in-fiction video call" src="../assets/images/valerian_screen_call.jpg" /></p>
<p>That's Herbie Hancock at 98% — as a face on a <em>screen inside the movie</em>, over a
sci-fi HUD overlay, during a video call in Valerian. A face is a face, whether
it's in the room or on the bridge's comms display.</p>
<h2 id="training-vs-held-out-the-generalization-gap">Training vs. held-out: the generalization gap<a class="headerlink" href="#training-vs-held-out-the-generalization-gap" title="Permanent link">&para;</a></h2>
<p>The shipped config (<code>prob_threshold=0.754, anneal_sec=35.54,
extinction_sec=57.43, expand_gallery=true</code>) was tuned against 4 films. Scored
against the 5 films the optimizer never saw:</p>
<p><img alt="Held-out per-film F1 vs. the training-set fit" src="../assets/images/holdout_f1_by_film.png" /></p>
<table>
<thead>
<tr>
<th>film</th>
<th>F1</th>
<th>P</th>
<th>R</th>
<th>TPI</th>
<th>FPI</th>
<th>misid</th>
<th>FN</th>
</tr>
</thead>
<tbody>
<tr>
<td>Benny &amp; Joon</td>
<td>83.0%</td>
<td>89.1%</td>
<td>77.7%</td>
<td>15125</td>
<td>1846</td>
<td>0</td>
<td>4337</td>
</tr>
<tr>
<td>Lovelace</td>
<td>77.5%</td>
<td>90.3%</td>
<td>67.9%</td>
<td>14990</td>
<td>1085</td>
<td>58</td>
<td>7085</td>
</tr>
<tr>
<td>Valerian and the City of a Thousand Planets</td>
<td>74.1%</td>
<td>97.1%</td>
<td>60.0%</td>
<td>18663</td>
<td>548</td>
<td>0</td>
<td>12467</td>
</tr>
<tr>
<td>Downton Abbey: A New Era</td>
<td>56.2%</td>
<td>97.8%</td>
<td>39.4%</td>
<td>52027</td>
<td>1173</td>
<td>0</td>
<td>80084</td>
</tr>
<tr>
<td><strong>The Many Saints of Newark</strong></td>
<td><strong>46.3%</strong></td>
<td><strong>54.7%</strong></td>
<td>40.1%</td>
<td>15922</td>
<td>4394</td>
<td><strong>974</strong></td>
<td>23791</td>
</tr>
<tr>
<td><strong>macro average</strong></td>
<td><strong>67.4%</strong></td>
<td>85.8%</td>
<td>57.0%</td>
<td></td>
<td></td>
<td></td>
<td></td>
</tr>
</tbody>
</table>
<p><strong>67.4% held-out vs. 75.3% on training</strong> — an ~8pp drop, and a <strong>37pp spread
between the best and worst held-out film</strong>. The config does not generalize
uniformly, and the spread traces to two mechanisms, both visible frame by
frame below.</p>
<h2 id="mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts">Mechanism 1: extinction bridging — usually right, wrong at hard cuts<a class="headerlink" href="#mechanism-1-extinction-bridging-usually-right-wrong-at-hard-cuts" title="Permanent link">&para;</a></h2>
<p>The extinction window keeps an identity alive through seconds where no face is
detectable. <strong>Most of the time this is exactly what you want</strong>, and it's where
a lot of the TPI count comes from:</p>
<p><img alt="Two faces on screen, six more correctly bridged" src="../assets/images/lovelace_polygraph_bridged.jpg" /></p>
<p>Lovelace's polygraph scene: only Eric Roberts and Amanda Seyfried have visible
faces, but X-Ray lists eight cast present — and all eight score green, the
other six correctly carried by presence windows through a scene where the
camera never shows them. A perfect second, and the extinction/anneal machinery
is <em>why</em>.</p>
<p>The same mechanism has a failure case: a hard cut into long faceless footage.
Both Many Saints of Newark (974 misIDs) and Downton Abbey (FN=80084, the worst
recall of the five) are dominated by it — verified directly against the raw
per-frame stream and the HDF5 dump's own detection counts, not inferred from
the score alone. <strong>This is not a malfunction</strong>: the tracker is doing exactly
what its window is for; the footage just stops cooperating. In the debug
overlay (which draws a bridged identity's last-known bbox, unlike the shipped
output, which emits presence windows and no boxes at all) the bridged state is
visible spatially:</p>
<p><img alt="Debug overlay: bridged identities drawn at their last-known positions" src="../assets/images/many_saints_ghost_fpi.jpg" />
<em>Debug-overlay rendering (<code>dump_error_frames.py --raw</code>): "Jon Bernthal", "Joey
Diaz" and "Billy Magnussen" are extinction-bridged identities from the previous
shot, drawn frozen over the wall and the hanging plates. Frame
<code>many_saints/fpi/fpi_t03543.jpg</code>, <code>montage-frames</code> artifact package.</em></p>
<p>The cost is measurable, not just visible. Downton Abbey's hard cut into its
closing credits, plotting the dump's own per-second <code>face_count</code> (detector
output, independent of the tracker) against what the tracker reports:</p>
<p><img alt="Detector vs. tracker through Downton Abbey's cut to credits" src="../assets/images/downton_ghost_timeline.png" /></p>
<p>From the cut onward the detector sees <strong>zero faces for nearly a minute</strong> — and
the tracker keeps reporting the last shot's 15 identities the whole time
(verified for Hugh Bonneville: bbox <code>(1743.2, 0.0, 171.3, 317.8)</code>, unchanged to
the pixel, at every sampled second for 57+ seconds). The staircase at the right
edge is the extinction window expiring actor by actor. That plateau is
<code>SceneTrackerFunc::active_[actor_idx].last_bbox</code>
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/src/nodes/scene_tracker_node.hpp"><code>src/nodes/scene_tracker_node.hpp</code></a>)
re-emitted as designed: <code>extinction_sec=57.4</code> was tuned long because bridging
wins on most footage (see the polygraph frame above) — the training films just
never contained a faceless stretch long enough to show the cost side, and the
held-out set did.</p>
<p>The same track-continuation machinery has one milder spatial artifact, worth
knowing when reading these frames:</p>
<p><img alt="Two labels on one face after a shot/reverse-shot cut" src="../assets/images/cafe_society_rapid_cut.jpg" />
<em>Café Society (a training film), a shot/reverse-shot dialog: that is Steve
Carell wearing both his own label and Jesse Eisenberg's.</em></p>
<p>At a rapid cut, the previous shot's track can linger for a beat at nearly the
same screen position the new face occupies — here Jesse Eisenberg's box from
the counter-shot lands on Steve Carell. Note what the score panel says,
though: both actors are green, because both <em>are</em> present in this dialog
scene per X-Ray. The spatial label is briefly wrong; the per-second presence
claim — the thing the pipeline actually ships — is right. It's the same trade
as the extinction window: track continuation smooths over cuts, and 1 fps
sampling occasionally catches the seam.</p>
<h2 id="mechanism-2-the-face-vs-presence-ceiling">Mechanism 2: the face-vs-presence ceiling<a class="headerlink" href="#mechanism-2-the-face-vs-presence-ceiling" title="Permanent link">&para;</a></h2>
<p>Downton Abbey's recall didn't collapse because faces were misread — it
collapsed because for most of its 80084 FN-seconds there was <strong>no face to
read</strong>:</p>
<p><img alt="22 cast credited, nobody facing the camera" src="../assets/images/downton_crew_fn.jpg" /></p>
<p>A newsreel crew hauls equipment through the hall: X-Ray credits 22 cast as
present in this scene; not one face looks at the camera. Eight are still
scored green (windows bridging from adjacent shots) — the other fourteen are
blue FNs that no face-recognition pipeline could ever recover. X-Ray encodes
<em>scene membership</em>; the pipeline measures <em>on-screen faces</em>. In ensemble films
those two definitions diverge massively, and that gap — not identification
error — is most of what the FN column counts.</p>
<p><img alt="Presence without a detectable face, The Many Saints of Newark" src="../assets/images/many_saints_outofcast_fpi.jpg" /></p>
<p>Same ceiling from the other side: Michela De Rossi in frame but turned away,
five cast correctly bridged as offscreen (green), four blue FNs — and one
orange we'll come back to below.</p>
<h2 id="where-lvface-beat-x-ray">Where LVFace beat X-Ray<a class="headerlink" href="#where-lvface-beat-x-ray" title="Permanent link">&para;</a></h2>
<p>Not every orange in these frames is actually wrong.
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/4b5557974bef8783bacc375c0869e8f589d1b0a3/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>
scores strictly against X-Ray — but X-Ray itself has holes, and the pipeline
found two kinds.</p>
<p><img alt="LVFace correctly identifies Germar Terrell Gardner, uncredited by X-Ray" src="../assets/images/germar_beats_xray.jpg" /></p>
<p>Germar Terrell Gardner — a real, clean, high-confidence detection — is counted
as an out-of-cast misID because he doesn't appear in X-Ray's <code>people.csv</code> for
The Many Saints of Newark at all. But Jellyfin's independent cast metadata
<em>does</em> credit him for this exact film (cross-checked via
<code>experiments/manifests/jellyfin_casts.json</code> from the <code>experiment-data</code> artifact
package, a completely separate data source from X-Ray). That's also him in
orange in the frame above — every one of those "errors" is the pipeline being
right about a person X-Ray forgot.</p>
<p><img alt="Robert Patrick, clearly on screen, scored wrong by a ground-truth gap" src="../assets/images/lovelace_robert_patrick_fpi.jpg" /></p>
<p>And it isn't only uncredited bit-parts. That is <strong>Robert Patrick</strong> — top-billed
in Lovelace, unmistakably on screen, reading his newspaper, identified at
100% — scored orange because X-Ray's people-in-scene list for <em>this scene</em>
doesn't include him. The identification is flawless; the ground truth missed
an actor sitting in the middle of the frame.</p>
<p>This doesn't mean every flagged misID is secretly correct — Many Saints'
974-count total is still overwhelmingly extinction bridging at cuts, not
uncredited cameos. But the X-Ray corpus is a convenient, large-scale ground
truth, not a perfect one, and the misID/FPI numbers in these tables carry an
irreducible noise floor from ground-truth gaps in both directions.</p>
<h2 id="summary">Summary<a class="headerlink" href="#summary" title="Permanent link">&para;</a></h2>
<p>LVFace is the right default: it wins the model comparison outright, it names
19 of 20 correctly across a hat-heavy funeral crowd, and it recognises a face
on a screen inside the movie. Its error budget decomposes into two understood
mechanisms — extinction bridging at hard cuts (a tunable trade, not a bug) and
the face-vs-presence ceiling baked into X-Ray's semantics — plus a nonzero
slice where the pipeline is right and the ground truth is wrong. The held-out
generalization gap (75.3% → 67.4%) is real and should be treated as the honest
expected performance, not the training-set number.</p>
</article>
</div>
<script>var target=document.getElementById(location.hash.slice(1));target&&target.name&&(target.checked=target.name.startsWith("__tabbed_"))</script>
</div>
<button type="button" class="md-top md-icon" data-md-component="top" hidden>
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M13 20h-2V8l-5.5 5.5-1.42-1.42L12 4.16l7.92 7.92-1.42 1.42L13 8z"/></svg>
Back to top
</button>
</main>
<footer class="md-footer">
<nav class="md-footer__inner md-grid" aria-label="Footer" >
<a href="../pose-expansion/" class="md-footer__link md-footer__link--prev" aria-label="Previous: Pose Expansion">
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
</div>
<div class="md-footer__title">
<span class="md-footer__direction">
Previous
</span>
<div class="md-ellipsis">
Pose Expansion
</div>
</div>
</a>
<a href="../model-bakeoff/" class="md-footer__link md-footer__link--next" aria-label="Next: Model Bake-off &amp; Re-tune (full log)">
<div class="md-footer__title">
<span class="md-footer__direction">
Next
</span>
<div class="md-ellipsis">
Model Bake-off & Re-tune (full log)
</div>
</div>
<div class="md-footer__button md-icon">
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
</div>
</a>
</nav>
<div class="md-footer-meta md-typeset">
<div class="md-footer-meta__inner md-grid">
<div class="md-copyright">
Made with
<a href="https://squidfunk.github.io/mkdocs-material/" target="_blank" rel="noopener">
Material for MkDocs
</a>
</div>
</div>
</div>
</footer>
</div>
<div class="md-dialog" data-md-component="dialog">
<div class="md-dialog__inner md-typeset"></div>
</div>
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
</body>
</html>