1738 lines
54 KiB
HTML
1738 lines
54 KiB
HTML
|
||
<!doctype html>
|
||
<html lang="en" class="no-js">
|
||
<head>
|
||
|
||
<meta charset="utf-8">
|
||
<meta name="viewport" content="width=device-width,initial-scale=1">
|
||
|
||
<meta name="description" content="Face-recognition pipeline for finding on-screen actor presence in film/TV, built on KPN++">
|
||
|
||
|
||
|
||
<link rel="canonical" href="https://pages.tourolle.paris/dtourolle/scene-actor-extraction/model-bakeoff/">
|
||
|
||
|
||
<link rel="prev" href="../lvface-deep-dive/">
|
||
|
||
|
||
<link rel="next" href="../service-conversion/">
|
||
|
||
|
||
|
||
|
||
|
||
<link rel="icon" href="../assets/images/favicon.png">
|
||
<meta name="generator" content="mkdocs-1.6.1, mkdocs-material-9.7.7">
|
||
|
||
|
||
|
||
<title>Full Experiment Log - scene-actor-extraction</title>
|
||
|
||
|
||
|
||
<link rel="stylesheet" href="../assets/stylesheets/main.ec1eaa64.min.css">
|
||
|
||
|
||
<link rel="stylesheet" href="../assets/stylesheets/palette.ab4e12ef.min.css">
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||
<link rel="stylesheet" href="https://fonts.googleapis.com/css?family=Roboto:300,300i,400,400i,700,700i%7CRoboto+Mono:400,400i,700,700i&display=fallback">
|
||
<style>:root{--md-text-font:"Roboto";--md-code-font:"Roboto Mono"}</style>
|
||
|
||
|
||
|
||
<link rel="stylesheet" href="../stylesheets/extra.css">
|
||
|
||
<script>__md_scope=new URL("..",location),__md_hash=e=>[...e].reduce(((e,_)=>(e<<5)-e+_.charCodeAt(0)),0),__md_get=(e,_=localStorage,t=__md_scope)=>JSON.parse(_.getItem(t.pathname+"."+e)),__md_set=(e,_,t=localStorage,a=__md_scope)=>{try{t.setItem(a.pathname+"."+e,JSON.stringify(_))}catch(e){}}</script>
|
||
|
||
|
||
|
||
|
||
|
||
</head>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<body dir="ltr" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber">
|
||
|
||
|
||
<input class="md-toggle" data-md-toggle="drawer" type="checkbox" id="__drawer" autocomplete="off">
|
||
<input class="md-toggle" data-md-toggle="search" type="checkbox" id="__search" autocomplete="off">
|
||
<label class="md-overlay" for="__drawer"></label>
|
||
<div data-md-component="skip">
|
||
|
||
|
||
<a href="#full-experiment-log" class="md-skip">
|
||
Skip to content
|
||
</a>
|
||
|
||
</div>
|
||
<div data-md-component="announce">
|
||
|
||
</div>
|
||
|
||
|
||
|
||
|
||
<header class="md-header" data-md-component="header">
|
||
<nav class="md-header__inner md-grid" aria-label="Header">
|
||
<a href=".." title="scene-actor-extraction" class="md-header__button md-logo" aria-label="scene-actor-extraction" data-md-component="logo">
|
||
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 8a3 3 0 0 0 3-3 3 3 0 0 0-3-3 3 3 0 0 0-3 3 3 3 0 0 0 3 3m0 3.54C9.64 9.35 6.5 8 3 8v11c3.5 0 6.64 1.35 9 3.54 2.36-2.19 5.5-3.54 9-3.54V8c-3.5 0-6.64 1.35-9 3.54"/></svg>
|
||
|
||
</a>
|
||
<label class="md-header__button md-icon" for="__drawer">
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M3 6h18v2H3zm0 5h18v2H3zm0 5h18v2H3z"/></svg>
|
||
</label>
|
||
<div class="md-header__title" data-md-component="header-title">
|
||
<div class="md-header__ellipsis">
|
||
<div class="md-header__topic">
|
||
<span class="md-ellipsis">
|
||
scene-actor-extraction
|
||
</span>
|
||
</div>
|
||
<div class="md-header__topic" data-md-component="header-topic">
|
||
<span class="md-ellipsis">
|
||
|
||
Full Experiment Log
|
||
|
||
</span>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
|
||
|
||
<form class="md-header__option" data-md-component="palette">
|
||
|
||
|
||
|
||
|
||
<input class="md-option" data-md-color-media="(prefers-color-scheme: dark)" data-md-color-scheme="slate" data-md-color-primary="black" data-md-color-accent="amber" aria-label="Switch to light mode" type="radio" name="__palette" id="__palette_0">
|
||
|
||
<label class="md-header__button md-icon" title="Switch to light mode" for="__palette_1" hidden>
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 7a5 5 0 0 1 5 5 5 5 0 0 1-5 5 5 5 0 0 1-5-5 5 5 0 0 1 5-5m0 2a3 3 0 0 0-3 3 3 3 0 0 0 3 3 3 3 0 0 0 3-3 3 3 0 0 0-3-3m0-7 2.39 3.42C13.65 5.15 12.84 5 12 5s-1.65.15-2.39.42zM3.34 7l4.16-.35A7.2 7.2 0 0 0 5.94 8.5c-.44.74-.69 1.5-.83 2.29zm.02 10 1.76-3.77a7.131 7.131 0 0 0 2.38 4.14zM20.65 7l-1.77 3.79a7.02 7.02 0 0 0-2.38-4.15zm-.01 10-4.14.36c.59-.51 1.12-1.14 1.54-1.86.42-.73.69-1.5.83-2.29zM12 22l-2.41-3.44c.74.27 1.55.44 2.41.44.82 0 1.63-.17 2.37-.44z"/></svg>
|
||
</label>
|
||
|
||
|
||
|
||
|
||
|
||
<input class="md-option" data-md-color-media="(prefers-color-scheme: light)" data-md-color-scheme="default" data-md-color-primary="black" data-md-color-accent="indigo" aria-label="Switch to dark mode" type="radio" name="__palette" id="__palette_1">
|
||
|
||
<label class="md-header__button md-icon" title="Switch to dark mode" for="__palette_0" hidden>
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="m17.75 4.09-2.53 1.94.91 3.06-2.63-1.81-2.63 1.81.91-3.06-2.53-1.94L12.44 4l1.06-3 1.06 3zm3.5 6.91-1.64 1.25.59 1.98-1.7-1.17-1.7 1.17.59-1.98L15.75 11l2.06-.05L18.5 9l.69 1.95zm-2.28 4.95c.83-.08 1.72 1.1 1.19 1.85-.32.45-.66.87-1.08 1.27C15.17 23 8.84 23 4.94 19.07c-3.91-3.9-3.91-10.24 0-14.14.4-.4.82-.76 1.27-1.08.75-.53 1.93.36 1.85 1.19-.27 2.86.69 5.83 2.89 8.02a9.96 9.96 0 0 0 8.02 2.89m-1.64 2.02a12.08 12.08 0 0 1-7.8-3.47c-2.17-2.19-3.33-5-3.49-7.82-2.81 3.14-2.7 7.96.31 10.98 3.02 3.01 7.84 3.12 10.98.31"/></svg>
|
||
</label>
|
||
|
||
|
||
</form>
|
||
|
||
|
||
|
||
<script>var palette=__md_get("__palette");if(palette&&palette.color){if("(prefers-color-scheme)"===palette.color.media){var media=matchMedia("(prefers-color-scheme: light)"),input=document.querySelector(media.matches?"[data-md-color-media='(prefers-color-scheme: light)']":"[data-md-color-media='(prefers-color-scheme: dark)']");palette.color.media=input.getAttribute("data-md-color-media"),palette.color.scheme=input.getAttribute("data-md-color-scheme"),palette.color.primary=input.getAttribute("data-md-color-primary"),palette.color.accent=input.getAttribute("data-md-color-accent")}for(var[key,value]of Object.entries(palette.color))document.body.setAttribute("data-md-color-"+key,value)}</script>
|
||
|
||
|
||
|
||
|
||
|
||
<label class="md-header__button md-icon" for="__search">
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M9.5 3A6.5 6.5 0 0 1 16 9.5c0 1.61-.59 3.09-1.56 4.23l.27.27h.79l5 5-1.5 1.5-5-5v-.79l-.27-.27A6.52 6.52 0 0 1 9.5 16 6.5 6.5 0 0 1 3 9.5 6.5 6.5 0 0 1 9.5 3m0 2C7 5 5 7 5 9.5S7 14 9.5 14 14 12 14 9.5 12 5 9.5 5"/></svg>
|
||
</label>
|
||
<div class="md-search" data-md-component="search" role="dialog">
|
||
<label class="md-search__overlay" for="__search"></label>
|
||
<div class="md-search__inner" role="search">
|
||
<form class="md-search__form" name="search">
|
||
<input type="text" class="md-search__input" name="query" aria-label="Search" placeholder="Search" autocapitalize="off" autocorrect="off" autocomplete="off" spellcheck="false" data-md-component="search-query" required>
|
||
<label class="md-search__icon md-icon" for="__search">
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M9.5 3A6.5 6.5 0 0 1 16 9.5c0 1.61-.59 3.09-1.56 4.23l.27.27h.79l5 5-1.5 1.5-5-5v-.79l-.27-.27A6.52 6.52 0 0 1 9.5 16 6.5 6.5 0 0 1 3 9.5 6.5 6.5 0 0 1 9.5 3m0 2C7 5 5 7 5 9.5S7 14 9.5 14 14 12 14 9.5 12 5 9.5 5"/></svg>
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
|
||
</label>
|
||
<nav class="md-search__options" aria-label="Search">
|
||
|
||
<button type="reset" class="md-search__icon md-icon" title="Clear" aria-label="Clear" tabindex="-1">
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M19 6.41 17.59 5 12 10.59 6.41 5 5 6.41 10.59 12 5 17.59 6.41 19 12 13.41 17.59 19 19 17.59 13.41 12z"/></svg>
|
||
</button>
|
||
</nav>
|
||
|
||
</form>
|
||
<div class="md-search__output">
|
||
<div class="md-search__scrollwrap" tabindex="0" data-md-scrollfix>
|
||
<div class="md-search-result" data-md-component="search-result">
|
||
<div class="md-search-result__meta">
|
||
Initializing search
|
||
</div>
|
||
<ol class="md-search-result__list" role="presentation"></ol>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
|
||
|
||
|
||
<div class="md-header__source">
|
||
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction" title="Go to repository" class="md-source" data-md-component="source">
|
||
<div class="md-source__icon md-icon">
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 448 512"><!--! Font Awesome Free 7.1.0 by @fontawesome - https://fontawesome.com License - https://fontawesome.com/license/free (Icons: CC BY 4.0, Fonts: SIL OFL 1.1, Code: MIT License) Copyright 2025 Fonticons, Inc.--><path d="M439.6 236.1 244 40.5c-5.4-5.5-12.8-8.5-20.4-8.5s-15 3-20.4 8.4L162.5 81l51.5 51.5c27.1-9.1 52.7 16.8 43.4 43.7l49.7 49.7c34.2-11.8 61.2 31 35.5 56.7-26.5 26.5-70.2-2.9-56-37.3L240.3 199v121.9c25.3 12.5 22.3 41.8 9.1 55-6.4 6.4-15.2 10.1-24.3 10.1s-17.8-3.6-24.3-10.1c-17.6-17.6-11.1-46.9 11.2-56v-123c-20.8-8.5-24.6-30.7-18.6-45L142.6 101 8.5 235.1C3 240.6 0 247.9 0 255.5s3 15 8.5 20.4l195.6 195.7c5.4 5.4 12.7 8.4 20.4 8.4s15-3 20.4-8.4l194.7-194.7c5.4-5.4 8.4-12.8 8.4-20.4s-3-15-8.4-20.4"/></svg>
|
||
</div>
|
||
<div class="md-source__repository">
|
||
dtourolle/scene-actor-extraction
|
||
</div>
|
||
</a>
|
||
</div>
|
||
|
||
</nav>
|
||
|
||
</header>
|
||
|
||
<div class="md-container" data-md-component="container">
|
||
|
||
|
||
|
||
|
||
|
||
<nav class="md-tabs" aria-label="Tabs" data-md-component="tabs">
|
||
<div class="md-grid">
|
||
<ul class="md-tabs__list">
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-tabs__item">
|
||
<a href=".." class="md-tabs__link">
|
||
|
||
|
||
|
||
|
||
|
||
Home
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-tabs__item">
|
||
<a href="../methodology/" class="md-tabs__link">
|
||
|
||
|
||
|
||
|
||
|
||
How We Score Against X-Ray
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-tabs__item">
|
||
<a href="../best-model/" class="md-tabs__link">
|
||
|
||
|
||
|
||
Findings
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-tabs__item md-tabs__item--active">
|
||
<a href="./" class="md-tabs__link">
|
||
|
||
|
||
|
||
|
||
|
||
Full Experiment Log
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-tabs__item">
|
||
<a href="../service-conversion/" class="md-tabs__link">
|
||
|
||
|
||
|
||
|
||
|
||
Service Conversion (proposal)
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
</ul>
|
||
</div>
|
||
</nav>
|
||
|
||
|
||
|
||
<main class="md-main" data-md-component="main">
|
||
<div class="md-main__inner md-grid">
|
||
|
||
|
||
|
||
<div class="md-sidebar md-sidebar--primary" data-md-component="sidebar" data-md-type="navigation" >
|
||
<div class="md-sidebar__scrollwrap">
|
||
<div class="md-sidebar__inner">
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<nav class="md-nav md-nav--primary md-nav--lifted" aria-label="Navigation" data-md-level="0">
|
||
<label class="md-nav__title" for="__drawer">
|
||
<a href=".." title="scene-actor-extraction" class="md-nav__button md-logo" aria-label="scene-actor-extraction" data-md-component="logo">
|
||
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M12 8a3 3 0 0 0 3-3 3 3 0 0 0-3-3 3 3 0 0 0-3 3 3 3 0 0 0 3 3m0 3.54C9.64 9.35 6.5 8 3 8v11c3.5 0 6.64 1.35 9 3.54 2.36-2.19 5.5-3.54 9-3.54V8c-3.5 0-6.64 1.35-9 3.54"/></svg>
|
||
|
||
</a>
|
||
scene-actor-extraction
|
||
</label>
|
||
|
||
<div class="md-nav__source">
|
||
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction" title="Go to repository" class="md-source" data-md-component="source">
|
||
<div class="md-source__icon md-icon">
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 448 512"><!--! Font Awesome Free 7.1.0 by @fontawesome - https://fontawesome.com License - https://fontawesome.com/license/free (Icons: CC BY 4.0, Fonts: SIL OFL 1.1, Code: MIT License) Copyright 2025 Fonticons, Inc.--><path d="M439.6 236.1 244 40.5c-5.4-5.5-12.8-8.5-20.4-8.5s-15 3-20.4 8.4L162.5 81l51.5 51.5c27.1-9.1 52.7 16.8 43.4 43.7l49.7 49.7c34.2-11.8 61.2 31 35.5 56.7-26.5 26.5-70.2-2.9-56-37.3L240.3 199v121.9c25.3 12.5 22.3 41.8 9.1 55-6.4 6.4-15.2 10.1-24.3 10.1s-17.8-3.6-24.3-10.1c-17.6-17.6-11.1-46.9 11.2-56v-123c-20.8-8.5-24.6-30.7-18.6-45L142.6 101 8.5 235.1C3 240.6 0 247.9 0 255.5s3 15 8.5 20.4l195.6 195.7c5.4 5.4 12.7 8.4 20.4 8.4s15-3 20.4-8.4l194.7-194.7c5.4-5.4 8.4-12.8 8.4-20.4s-3-15-8.4-20.4"/></svg>
|
||
</div>
|
||
<div class="md-source__repository">
|
||
dtourolle/scene-actor-extraction
|
||
</div>
|
||
</a>
|
||
</div>
|
||
|
||
<ul class="md-nav__list" data-md-scrollfix>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item">
|
||
<a href=".." class="md-nav__link">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
Home
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item">
|
||
<a href="../methodology/" class="md-nav__link">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
How We Score Against X-Ray
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item md-nav__item--nested">
|
||
|
||
|
||
|
||
<input class="md-nav__toggle md-toggle " type="checkbox" id="__nav_3" >
|
||
|
||
|
||
<label class="md-nav__link" for="__nav_3" id="__nav_3_label" tabindex="0">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
Findings
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
<span class="md-nav__icon md-icon"></span>
|
||
</label>
|
||
|
||
<nav class="md-nav" data-md-level="1" aria-labelledby="__nav_3_label" aria-expanded="false">
|
||
<label class="md-nav__title" for="__nav_3">
|
||
<span class="md-nav__icon md-icon"></span>
|
||
|
||
|
||
Findings
|
||
|
||
|
||
</label>
|
||
<ul class="md-nav__list" data-md-scrollfix>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item">
|
||
<a href="../best-model/" class="md-nav__link">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
Best Model
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item">
|
||
<a href="../gallery-scope/" class="md-nav__link">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
Gallery Scope (Full vs. Limited)
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item">
|
||
<a href="../pose-expansion/" class="md-nav__link">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
Pose Expansion
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item">
|
||
<a href="../lvface-deep-dive/" class="md-nav__link">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
LVFace Deep Dive
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
</ul>
|
||
</nav>
|
||
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item md-nav__item--active">
|
||
|
||
<input class="md-nav__toggle md-toggle" type="checkbox" id="__toc">
|
||
|
||
|
||
|
||
|
||
|
||
<label class="md-nav__link md-nav__link--active" for="__toc">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
Full Experiment Log
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
<span class="md-nav__icon md-icon"></span>
|
||
</label>
|
||
|
||
<a href="./" class="md-nav__link md-nav__link--active">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
Full Experiment Log
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
</a>
|
||
|
||
|
||
|
||
<nav class="md-nav md-nav--secondary" aria-label="Table of contents">
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<label class="md-nav__title" for="__toc">
|
||
<span class="md-nav__icon md-icon"></span>
|
||
Table of contents
|
||
</label>
|
||
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#why-replay-makes-this-affordable" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Why replay makes this affordable
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#search-space" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Search space
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#training-films-and-held-out-films" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Training films and held-out films
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#gallery-coverage-per-film" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Gallery coverage per film
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#training-results-3-models-2-gallery-modes-2-expansion-settings" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Training results, 3 models × 2 gallery modes × 2 expansion settings
|
||
|
||
</span>
|
||
</a>
|
||
|
||
<nav class="md-nav" aria-label="Training results, 3 models × 2 gallery modes × 2 expansion settings">
|
||
<ul class="md-nav__list">
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#a-scoring-bug-worth-recording-dropped-film-evaluations" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
A scoring bug worth recording: dropped-film evaluations
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#per-film-training-breakdown" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Per-film training breakdown
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
</ul>
|
||
</nav>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#held-out-validation-all-3-models" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Held-out validation, all 3 models
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#two-effects-in-isolation-gallery-scope-and-pose-expansion" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Two effects in isolation: gallery scope and pose expansion
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#calibration-curves" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Calibration curves
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#extinction-and-anneal-window-search" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Extinction and anneal window search
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#caveats" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Caveats
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#reproduce" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Reproduce
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
</ul>
|
||
|
||
</nav>
|
||
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li class="md-nav__item">
|
||
<a href="../service-conversion/" class="md-nav__link">
|
||
|
||
|
||
|
||
<span class="md-ellipsis">
|
||
|
||
|
||
Service Conversion (proposal)
|
||
|
||
|
||
|
||
</span>
|
||
|
||
|
||
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
</ul>
|
||
</nav>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
|
||
|
||
|
||
<div class="md-sidebar md-sidebar--secondary" data-md-component="sidebar" data-md-type="toc" >
|
||
<div class="md-sidebar__scrollwrap">
|
||
<div class="md-sidebar__inner">
|
||
|
||
|
||
<nav class="md-nav md-nav--secondary" aria-label="Table of contents">
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<label class="md-nav__title" for="__toc">
|
||
<span class="md-nav__icon md-icon"></span>
|
||
Table of contents
|
||
</label>
|
||
<ul class="md-nav__list" data-md-component="toc" data-md-scrollfix>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#why-replay-makes-this-affordable" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Why replay makes this affordable
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#search-space" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Search space
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#training-films-and-held-out-films" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Training films and held-out films
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#gallery-coverage-per-film" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Gallery coverage per film
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#training-results-3-models-2-gallery-modes-2-expansion-settings" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Training results, 3 models × 2 gallery modes × 2 expansion settings
|
||
|
||
</span>
|
||
</a>
|
||
|
||
<nav class="md-nav" aria-label="Training results, 3 models × 2 gallery modes × 2 expansion settings">
|
||
<ul class="md-nav__list">
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#a-scoring-bug-worth-recording-dropped-film-evaluations" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
A scoring bug worth recording: dropped-film evaluations
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#per-film-training-breakdown" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Per-film training breakdown
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
</ul>
|
||
</nav>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#held-out-validation-all-3-models" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Held-out validation, all 3 models
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#two-effects-in-isolation-gallery-scope-and-pose-expansion" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Two effects in isolation: gallery scope and pose expansion
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#calibration-curves" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Calibration curves
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#extinction-and-anneal-window-search" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Extinction and anneal window search
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#caveats" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Caveats
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
<li class="md-nav__item">
|
||
<a href="#reproduce" class="md-nav__link">
|
||
<span class="md-ellipsis">
|
||
|
||
Reproduce
|
||
|
||
</span>
|
||
</a>
|
||
|
||
</li>
|
||
|
||
</ul>
|
||
|
||
</nav>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
|
||
|
||
|
||
<div class="md-content" data-md-component="content">
|
||
|
||
<article class="md-content__inner md-typeset">
|
||
|
||
|
||
|
||
|
||
|
||
<h1 id="full-experiment-log">Full experiment log<a class="headerlink" href="#full-experiment-log" title="Permanent link">¶</a></h1>
|
||
<p>This page reports how the pipeline performs across three questions: which
|
||
embedding model is best, whether restricting the gallery to a film's
|
||
credited cast helps, and whether promoting confidently identified poses into
|
||
a per-film gallery annex helps. It also documents the replay architecture
|
||
that made testing all three questions in one pass practical, and every
|
||
caveat needed to trust the numbers.</p>
|
||
<p>Read <a href="../methodology/">How we score against X-Ray</a> first for what F1,
|
||
precision, recall, and misID mean in this report. All numbers below use the
|
||
per-second metric
|
||
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/e5885977dfd04741a0d1bb7aaf0270d9c40a2fa8/scripts/optimizer/second_score.py"><code>scripts/optimizer/second_score.py</code></a>).</p>
|
||
<p>r50 (ArcFace w600k-R50) is excluded from the detailed comparison below. Its
|
||
gallery was built with roughly 30% fewer reference images per actor than the
|
||
other three models on the identical source photos (10808 vs 15055 total
|
||
embeddings across the same 2418 actors), which confounds any direct
|
||
comparison of its scores against the others. It remains in the
|
||
<a href="../best-model/#first-signal-calibration-curves">calibration curve comparison</a>,
|
||
which does not depend on the training benchmark.</p>
|
||
<h2 id="why-replay-makes-this-affordable">Why replay makes this affordable<a class="headerlink" href="#why-replay-makes-this-affordable" title="Permanent link">¶</a></h2>
|
||
<p>Decoding video and running face detection, alignment, and embedding is the
|
||
expensive part of this pipeline. Everything downstream of that (tracking,
|
||
identity matching, scene aggregation) is cheap. KPN++'s node/network
|
||
structure means those two stages are separate components connected by
|
||
typed channels, so the expensive stage can run once per film, cache its
|
||
output, and the cheap stage can be re-run against that cache as many times
|
||
as needed with different Config values.</p>
|
||
<p><code>scene_analyze --dump-embeddings out.h5</code> runs the expensive half once per
|
||
film and writes per-frame face detections and embeddings to HDF5
|
||
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/e5885977dfd04741a0d1bb7aaf0270d9c40a2fa8/scripts/optimizer/SCHEMA.md"><code>scripts/optimizer/SCHEMA.md</code></a>).
|
||
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/e5885977dfd04741a0d1bb7aaf0270d9c40a2fa8/scripts/optimizer/replay.py"><code>scripts/optimizer/replay.py</code></a>
|
||
then re-assembles the real C++ <code>face_tracker</code>, <code>identity_matcher</code>, and
|
||
<code>scene_tracker</code> nodes into a Python-driven KPN network and replays a
|
||
film's cached embeddings through them, varying <code>prob_threshold</code>,
|
||
<code>anneal_sec</code>, <code>extinction_sec</code>, and <code>expand_gallery</code> freely. No GPU
|
||
inference and no video decode happen during a replay; each one completes
|
||
in seconds. This is what makes a 512-evaluation differential-evolution
|
||
search per model, per gallery mode, per expansion setting, tractable, and
|
||
what made the full held-out validation across three models in this report
|
||
possible in one session rather than requiring three full re-encodes of the
|
||
benchmark set.</p>
|
||
<p><code>optimize.py</code> runs <code>differential_evolution</code> over this replay function as its
|
||
objective, with DE-level parallelism (multiple candidate configs evaluated
|
||
concurrently, each spawning its own replay subprocesses) on top of it. The
|
||
practical ceiling on this machine's GPU was 8 concurrent replay processes;
|
||
9 silently degraded every score to 0.0% (well-formed output, wrong numbers,
|
||
not a crash), so <code>optimize.py</code> was run at <code>REPLAY_WORKERS=4 DE_WORKERS=2</code>.</p>
|
||
<h2 id="search-space">Search space<a class="headerlink" href="#search-space" title="Permanent link">¶</a></h2>
|
||
<p><code>popsize=10, maxiter=15</code> per combo (3 parameters, up to 512 evaluations,
|
||
usually stopping earlier on DE's convergence tolerance).
|
||
<code>anneal_sec</code>/<code>extinction_sec</code> bounds were widened from 1-30/1-15 to 1-60/1-60
|
||
partway through the sweep. r50's 4 combos finished before the widening and
|
||
used the old, narrower bounds; this is one more reason r50 is excluded from
|
||
direct comparison here.</p>
|
||
<h2 id="training-films-and-held-out-films">Training films and held-out films<a class="headerlink" href="#training-films-and-held-out-films" title="Permanent link">¶</a></h2>
|
||
<p>9 films have dumped embeddings across all 4 models. 4 were used for
|
||
optimization:</p>
|
||
<ul>
|
||
<li>Café Society (62-cast)</li>
|
||
<li>Lord of War (64-cast)</li>
|
||
<li>Scarface (67-cast)</li>
|
||
<li>Sound of Metal (14-cast)</li>
|
||
</ul>
|
||
<p>5 were held out, never seen by any optimizer run:</p>
|
||
<ul>
|
||
<li>Benny & Joon</li>
|
||
<li>Downton Abbey: A New Era</li>
|
||
<li>Lovelace</li>
|
||
<li>The Many Saints of Newark</li>
|
||
<li>Valerian and the City of a Thousand Planets</li>
|
||
</ul>
|
||
<h2 id="gallery-coverage-per-film">Gallery coverage per film<a class="headerlink" href="#gallery-coverage-per-film" title="Permanent link">¶</a></h2>
|
||
<p>The gallery has reference embeddings for 2418 actors, but coverage of any
|
||
given film's credited cast varies widely. This was previously reported as
|
||
one flat number (67% of X-Ray cast lacking a reference embedding, averaged
|
||
across the whole benchmark); the per-film breakdown is:</p>
|
||
<table>
|
||
<thead>
|
||
<tr>
|
||
<th>film</th>
|
||
<th>cast credited</th>
|
||
<th>in gallery</th>
|
||
<th>coverage</th>
|
||
</tr>
|
||
</thead>
|
||
<tbody>
|
||
<tr>
|
||
<td>Lord of War</td>
|
||
<td>64</td>
|
||
<td>13</td>
|
||
<td>20.3%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Scarface</td>
|
||
<td>67</td>
|
||
<td>15</td>
|
||
<td>22.4%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>The Many Saints of Newark</td>
|
||
<td>48</td>
|
||
<td>13</td>
|
||
<td>27.1%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Café Society</td>
|
||
<td>62</td>
|
||
<td>17</td>
|
||
<td>27.4%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Lovelace</td>
|
||
<td>42</td>
|
||
<td>15</td>
|
||
<td>35.7%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Valerian and the City of a Thousand Planets</td>
|
||
<td>36</td>
|
||
<td>13</td>
|
||
<td>36.1%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Benny & Joon</td>
|
||
<td>23</td>
|
||
<td>12</td>
|
||
<td>52.2%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Downton Abbey: A New Era</td>
|
||
<td>36</td>
|
||
<td>22</td>
|
||
<td>61.1%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Sound of Metal</td>
|
||
<td>14</td>
|
||
<td>11</td>
|
||
<td>78.6%</td>
|
||
</tr>
|
||
</tbody>
|
||
</table>
|
||
<p>Two training films (Lord of War, Scarface) have the worst coverage in the
|
||
set, 20-22%. Their training-set F1 numbers below are partly capped by
|
||
missing references, not purely by model quality. Downton Abbey has 61%
|
||
coverage, the second-best in the benchmark, yet the worst held-out recall
|
||
of any film (39.4%, LVFace). Its recall problem is not primarily a coverage
|
||
problem; it is the extinction-bridging failure documented in the
|
||
<a href="../lvface-deep-dive/#mechanism-1-extinction-bridging">LVFace deep dive</a>.
|
||
Reproduce with <code>scripts/docs/gallery_coverage_per_film.py</code>.</p>
|
||
<h2 id="training-results-3-models-2-gallery-modes-2-expansion-settings">Training results, 3 models × 2 gallery modes × 2 expansion settings<a class="headerlink" href="#training-results-3-models-2-gallery-modes-2-expansion-settings" title="Permanent link">¶</a></h2>
|
||
<p>Ranked by F1. misid = FPI_misid, the count of true wrong-actor
|
||
identifications (naming someone not in the film's cast at all), distinct
|
||
from FPI, which also includes in-cast timing slips.</p>
|
||
<p>Each combo's row is its best <strong>full-coverage</strong> evaluation: the highest-F1 DE
|
||
evaluation in which all 4 training films replayed without a timeout (see
|
||
<a href="#a-scoring-bug-worth-recording-dropped-film-evaluations">Dropped-film scoring</a>
|
||
below for why this qualifier is load-bearing and not the same as <code>argmax F1</code>
|
||
over the raw sweep).</p>
|
||
<table>
|
||
<thead>
|
||
<tr>
|
||
<th>combo</th>
|
||
<th>F1</th>
|
||
<th>P</th>
|
||
<th>R</th>
|
||
<th>TPI</th>
|
||
<th>FPI</th>
|
||
<th>misid</th>
|
||
<th>FN</th>
|
||
</tr>
|
||
</thead>
|
||
<tbody>
|
||
<tr>
|
||
<td>LVFace-B_Glint360K_restricted_exp</td>
|
||
<td>78.3%</td>
|
||
<td>91.0%</td>
|
||
<td>68.9%</td>
|
||
<td>42830</td>
|
||
<td>3782</td>
|
||
<td>60</td>
|
||
<td>19492</td>
|
||
</tr>
|
||
<tr>
|
||
<td>LVFace-B_Glint360K_restricted_noexp</td>
|
||
<td>76.7%</td>
|
||
<td>91.5%</td>
|
||
<td>66.2%</td>
|
||
<td>41149</td>
|
||
<td>3400</td>
|
||
<td>59</td>
|
||
<td>21173</td>
|
||
</tr>
|
||
<tr>
|
||
<td>arcface_w600k_mbf_restricted_exp</td>
|
||
<td>76.2%</td>
|
||
<td>90.0%</td>
|
||
<td>66.2%</td>
|
||
<td>64328</td>
|
||
<td>7480</td>
|
||
<td>0</td>
|
||
<td>33234</td>
|
||
</tr>
|
||
<tr>
|
||
<td>arcface_r18_restricted_exp</td>
|
||
<td>75.5%</td>
|
||
<td>87.6%</td>
|
||
<td>66.5%</td>
|
||
<td>41399</td>
|
||
<td>5666</td>
|
||
<td>60</td>
|
||
<td>20923</td>
|
||
</tr>
|
||
<tr>
|
||
<td>LVFace-B_Glint360K_full_exp</td>
|
||
<td>75.3%</td>
|
||
<td>89.7%</td>
|
||
<td>65.4%</td>
|
||
<td>47757</td>
|
||
<td>3407</td>
|
||
<td>232</td>
|
||
<td>26966</td>
|
||
</tr>
|
||
<tr>
|
||
<td>arcface_w600k_mbf_restricted_noexp</td>
|
||
<td>75.0%</td>
|
||
<td>91.1%</td>
|
||
<td>63.9%</td>
|
||
<td>39752</td>
|
||
<td>3465</td>
|
||
<td>60</td>
|
||
<td>22570</td>
|
||
</tr>
|
||
<tr>
|
||
<td>arcface_r18_restricted_noexp</td>
|
||
<td>73.5%</td>
|
||
<td>91.3%</td>
|
||
<td>61.7%</td>
|
||
<td>38299</td>
|
||
<td>3220</td>
|
||
<td>60</td>
|
||
<td>24023</td>
|
||
</tr>
|
||
<tr>
|
||
<td>LVFace-B_Glint360K_full_noexp</td>
|
||
<td>72.3%</td>
|
||
<td>88.3%</td>
|
||
<td>61.8%</td>
|
||
<td>40363</td>
|
||
<td>3503</td>
|
||
<td>244</td>
|
||
<td>25850</td>
|
||
</tr>
|
||
<tr>
|
||
<td>arcface_w600k_mbf_full_exp</td>
|
||
<td>72.0%</td>
|
||
<td>87.7%</td>
|
||
<td>61.4%</td>
|
||
<td>39875</td>
|
||
<td>3729</td>
|
||
<td>240</td>
|
||
<td>26338</td>
|
||
</tr>
|
||
<tr>
|
||
<td>arcface_w600k_mbf_full_noexp</td>
|
||
<td>71.0%</td>
|
||
<td>93.2%</td>
|
||
<td>57.9%</td>
|
||
<td>41699</td>
|
||
<td>2472</td>
|
||
<td>56</td>
|
||
<td>33024</td>
|
||
</tr>
|
||
<tr>
|
||
<td>arcface_r18_full_exp</td>
|
||
<td>69.1%</td>
|
||
<td>87.6%</td>
|
||
<td>57.7%</td>
|
||
<td>37342</td>
|
||
<td>3119</td>
|
||
<td>242</td>
|
||
<td>28871</td>
|
||
</tr>
|
||
<tr>
|
||
<td>arcface_r18_full_noexp</td>
|
||
<td>66.6%</td>
|
||
<td>91.3%</td>
|
||
<td>53.1%</td>
|
||
<td>34314</td>
|
||
<td>2362</td>
|
||
<td>107</td>
|
||
<td>31899</td>
|
||
</tr>
|
||
</tbody>
|
||
</table>
|
||
<p><img alt="All combos ranked by training-set F1" src="../assets/images/rep4_matrix_f1.png" /></p>
|
||
<p>The two clearest patterns: every model's best-scoring combo uses the
|
||
restricted gallery, and LVFace leads within both gallery modes. <code>full_exp</code>
|
||
(the shipped combination) is the best-scoring option that uses only
|
||
features the running application currently supports; restriction is not
|
||
wired into the application yet (see
|
||
<a href="../gallery-scope/">Whole vs. cast-restricted gallery</a>).</p>
|
||
<h3 id="a-scoring-bug-worth-recording-dropped-film-evaluations">A scoring bug worth recording: dropped-film evaluations<a class="headerlink" href="#a-scoring-bug-worth-recording-dropped-film-evaluations" title="Permanent link">¶</a></h3>
|
||
<p>The numbers above are corrected ones. The raw <code>rep4_best_*.json</code> files, and an
|
||
earlier version of this table, reported a different <code>arcface_w600k_mbf_full_noexp</code>
|
||
row: <strong>74.2% F1 at TPI 12645</strong>, a third the TPI of every sibling combo. That was
|
||
not a better config; it was an artifact of how the optimizer aggregates.</p>
|
||
<p><code>optimize.py</code> builds each candidate's score from only the films whose replay
|
||
subprocess returned (<code>per_film = [m for m in ex.map(_one, films) if m is not
|
||
None]</code>), then <strong>averages</strong> F1/precision/recall and <strong>sums</strong> TPI/FPI/misID over
|
||
just those survivors. When a film's replay times out (the sweep ran near the
|
||
8-process concurrency ceiling, so this happened intermittently), that film
|
||
silently drops from both. A candidate whose hardest film timed out is therefore
|
||
scored on an easier subset, and differential evolution, maximizing that score,
|
||
will happily converge onto exactly such a candidate. For <code>mbf_full_noexp</code> the
|
||
reported winner was one of 7 evaluations (out of 512) whose TPI had collapsed to
|
||
a partial-film subset; its median-coverage evaluations sit around 51686 TPI.</p>
|
||
<p>The fix here was to re-derive each combo's best row from its DE trajectory
|
||
(<code>experiments/trajectories/rep4_*.jsonl</code>), keeping only evaluations within 30% of
|
||
that combo's median TPI (full 4-film coverage) before taking the best F1. This
|
||
needs no re-running, the honest best configuration was already in the sweep,
|
||
just not the one <code>argmax F1</code> selected. Three combos moved: <code>mbf_full_noexp</code>
|
||
74.2% → <strong>71.0%</strong>, <code>LVFace_full_noexp</code> 72.4% → <strong>72.3%</strong> (and its misID, 0 → 244,
|
||
was itself a dropped-film artifact), <code>mbf_restricted_exp</code> 76.5% → <strong>76.2%</strong>. The
|
||
shipped LVFace <code>full_exp</code> winner was unaffected, its reported evaluation already
|
||
had full coverage (TPI 47757 ≈ median). <code>experiment_charts.py</code> applies the same
|
||
<code>clean_best</code> filter, so every figure on this page matches the corrected table.
|
||
The underlying <code>optimize.py</code> aggregation is also being fixed so a dropped-film
|
||
evaluation can never be selected as a winner again.</p>
|
||
<h3 id="per-film-training-breakdown">Per-film training breakdown<a class="headerlink" href="#per-film-training-breakdown" title="Permanent link">¶</a></h3>
|
||
<p>The 75.3% LVFace training figure is a macro average across 4 films, not a
|
||
uniform result:</p>
|
||
<table>
|
||
<thead>
|
||
<tr>
|
||
<th>film</th>
|
||
<th>LVFace F1</th>
|
||
<th>mbf F1</th>
|
||
<th>r18 F1</th>
|
||
<th>best model</th>
|
||
</tr>
|
||
</thead>
|
||
<tbody>
|
||
<tr>
|
||
<td>Café Society</td>
|
||
<td>68.1%</td>
|
||
<td>62.2%</td>
|
||
<td>60.1%</td>
|
||
<td>LVFace</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Lord of War</td>
|
||
<td>75.6%</td>
|
||
<td>77.2%</td>
|
||
<td>75.6%</td>
|
||
<td>mbf</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Scarface</td>
|
||
<td>71.5%</td>
|
||
<td>68.6%</td>
|
||
<td>64.1%</td>
|
||
<td>LVFace</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Sound of Metal</td>
|
||
<td>78.8%</td>
|
||
<td>76.5%</td>
|
||
<td>71.6%</td>
|
||
<td>LVFace</td>
|
||
</tr>
|
||
</tbody>
|
||
</table>
|
||
<p>LVFace does not win every training film. mbf scores higher on Lord of War
|
||
(77.2% vs 75.6%). LVFace's own training-film range is 68.1% to 78.8%, a
|
||
10.7pp spread, smaller than the 37pp spread seen on held-out films but real.
|
||
Reproduce with <code>scripts/docs/run_holdout_all_models.py --films training</code>.</p>
|
||
<h2 id="held-out-validation-all-3-models">Held-out validation, all 3 models<a class="headerlink" href="#held-out-validation-all-3-models" title="Permanent link">¶</a></h2>
|
||
<p>The training matrix above is training-set fit. Each model's own tuned
|
||
<code>full_exp</code> config was replayed against the 5 held-out films, scored the
|
||
same way:</p>
|
||
<table>
|
||
<thead>
|
||
<tr>
|
||
<th>film</th>
|
||
<th>LVFace F1</th>
|
||
<th>mbf F1</th>
|
||
<th>r18 F1</th>
|
||
</tr>
|
||
</thead>
|
||
<tbody>
|
||
<tr>
|
||
<td>Benny & Joon</td>
|
||
<td>83.0%</td>
|
||
<td>78.5%</td>
|
||
<td>77.1%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Lovelace</td>
|
||
<td>77.5%</td>
|
||
<td>73.7%</td>
|
||
<td>72.2%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Valerian and the City of a Thousand Planets</td>
|
||
<td>74.1%</td>
|
||
<td>70.2%</td>
|
||
<td>71.0%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Downton Abbey: A New Era</td>
|
||
<td>56.2%</td>
|
||
<td>55.0%</td>
|
||
<td>53.0%</td>
|
||
</tr>
|
||
<tr>
|
||
<td>The Many Saints of Newark</td>
|
||
<td>46.3%</td>
|
||
<td>44.5%</td>
|
||
<td>42.1%</td>
|
||
</tr>
|
||
<tr>
|
||
<td><strong>macro average</strong></td>
|
||
<td><strong>67.4%</strong></td>
|
||
<td><strong>64.4%</strong></td>
|
||
<td><strong>63.1%</strong></td>
|
||
</tr>
|
||
</tbody>
|
||
</table>
|
||
<p>LVFace scores highest on every one of the 5 held-out films; the ranking
|
||
never flips. Total misIDs across the 5 films: LVFace 1032, mbf 2197, r18
|
||
1224. LVFace has less than half mbf's misID count while also scoring
|
||
higher on every film. This directly confirms the model choice out of
|
||
sample; it is not inferred from the training numbers alone. See the
|
||
<a href="../lvface-deep-dive/">LVFace deep dive</a> for frame-level detail on where and
|
||
why LVFace still fails on the two worst films. Reproduce with
|
||
<code>scripts/docs/run_holdout_all_models.py</code>.</p>
|
||
<h2 id="two-effects-in-isolation-gallery-scope-and-pose-expansion">Two effects in isolation: gallery scope and pose expansion<a class="headerlink" href="#two-effects-in-isolation-gallery-scope-and-pose-expansion" title="Permanent link">¶</a></h2>
|
||
<p>Averaging across the 3 compared models (r50 excluded) isolates each variable
|
||
from model choice.</p>
|
||
<p><strong>Gallery scope</strong>, averaged over both expansion settings and all 3 models
|
||
(6 evaluations per row):</p>
|
||
<table>
|
||
<thead>
|
||
<tr>
|
||
<th>scope</th>
|
||
<th>F1</th>
|
||
<th>P</th>
|
||
<th>R</th>
|
||
<th>total misID</th>
|
||
</tr>
|
||
</thead>
|
||
<tbody>
|
||
<tr>
|
||
<td>full</td>
|
||
<td>71.1%</td>
|
||
<td>89.6%</td>
|
||
<td>59.6%</td>
|
||
<td>1121</td>
|
||
</tr>
|
||
<tr>
|
||
<td>restricted</td>
|
||
<td>75.9%</td>
|
||
<td>90.4%</td>
|
||
<td>65.6%</td>
|
||
<td>299</td>
|
||
</tr>
|
||
</tbody>
|
||
</table>
|
||
<p>Restriction improves every metric at once. This is not a precision/recall
|
||
trade: +4.8pp F1, +6.0pp recall, and roughly a quarter the misIDs. Fewer
|
||
candidates in the matcher's search space means fewer opportunities for a
|
||
lookalike false match, and the recall gain shows this does not cost real
|
||
detections. Restriction is currently an offline optimizer technique, not a
|
||
runtime feature of the application; see
|
||
<a href="../gallery-scope/">Whole vs. cast-restricted gallery</a> for what building it
|
||
into the application would require.</p>
|
||
<p><strong>Pose expansion</strong> (promoting a confidently identified track's novel-pose
|
||
views into a per-film gallery annex,
|
||
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/e5885977dfd04741a0d1bb7aaf0270d9c40a2fa8/src/gallery/track_gallery.hpp"><code>src/gallery/track_gallery.hpp</code></a>):</p>
|
||
<table>
|
||
<thead>
|
||
<tr>
|
||
<th>scope</th>
|
||
<th>expansion</th>
|
||
<th>F1</th>
|
||
<th>R</th>
|
||
<th>misID</th>
|
||
</tr>
|
||
</thead>
|
||
<tbody>
|
||
<tr>
|
||
<td>full</td>
|
||
<td>off</td>
|
||
<td>70.0%</td>
|
||
<td>57.6%</td>
|
||
<td>407</td>
|
||
</tr>
|
||
<tr>
|
||
<td>full</td>
|
||
<td>on</td>
|
||
<td>72.1%</td>
|
||
<td>61.5%</td>
|
||
<td>714</td>
|
||
</tr>
|
||
<tr>
|
||
<td>restricted</td>
|
||
<td>off</td>
|
||
<td>75.1%</td>
|
||
<td>63.9%</td>
|
||
<td>179</td>
|
||
</tr>
|
||
<tr>
|
||
<td>restricted</td>
|
||
<td>on</td>
|
||
<td>76.7%</td>
|
||
<td>67.2%</td>
|
||
<td>120</td>
|
||
</tr>
|
||
</tbody>
|
||
</table>
|
||
<p>In restricted mode, expansion is a clean win: +1.6pp F1, +3.3pp recall,
|
||
misID drops. The annex only competes against the film's own roughly 15-actor
|
||
cast, so a new pose of a known actor is unlikely to be confused with someone
|
||
else. In full mode, expansion buys +2.1pp F1 and +3.9pp recall but at a real
|
||
cost: misID rises from 407 to 714 as the same new-pose view now competes
|
||
against the full 2418-actor gallery, where a confidently learned pose is more
|
||
likely to match the wrong person. On the full gallery it is a recall-vs-misID
|
||
trade, not a free gain. This training-set effect
|
||
did not reproduce on held-out data; see
|
||
<a href="../pose-expansion/">Does pose expansion help?</a> for the full held-out test
|
||
and the two methodology bugs caught while checking it.</p>
|
||
<h2 id="calibration-curves">Calibration curves<a class="headerlink" href="#calibration-curves" title="Permanent link">¶</a></h2>
|
||
<p>Each gallery carries a fitted Platt sigmoid <code>P(match | sim) = σ(a·sim + b)</code>,
|
||
stored directly in the gallery HDF5
|
||
(<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/e5885977dfd04741a0d1bb7aaf0270d9c40a2fa8/src/gallery/gallery_calibration.hpp"><code>src/gallery/gallery_calibration.hpp</code></a>).
|
||
This measures discriminative power independent of whatever
|
||
<code>prob_threshold</code> a given run used:</p>
|
||
<p><img alt="Calibrated P(match|similarity) for all four models" src="../assets/images/calibration_curves.png" /></p>
|
||
<p>LVFace has the steepest curve (<code>a=17.7</code> vs 15.3-16.2 for the ArcFace
|
||
variants) and the lowest P=0.5 decision boundary (similarity 0.23 vs
|
||
0.27-0.31), separating same-actor from different-actor pairs more
|
||
confidently at a lower similarity than any ArcFace variant tested,
|
||
including r50. Generated by
|
||
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/e5885977dfd04741a0d1bb7aaf0270d9c40a2fa8/scripts/docs/calibration_chart.py"><code>scripts/docs/calibration_chart.py</code></a>.</p>
|
||
<h2 id="extinction-and-anneal-window-search">Extinction and anneal window search<a class="headerlink" href="#extinction-and-anneal-window-search" title="Permanent link">¶</a></h2>
|
||
<p>Every one of the 512 DE evaluations for the winning LVFace <code>full_exp</code>
|
||
combo, plotted over the <code>prob_threshold</code> × <code>extinction_sec</code> plane:</p>
|
||
<p><img alt="DE search landscape: 512 evaluations over prob_threshold × extinction_sec" src="../assets/images/de_search_landscape.png" /></p>
|
||
<p>Nearly everything scoring well sits at <code>extinction_sec</code> above 50, across a
|
||
wide range of thresholds. Short extinction windows are uniformly weaker:
|
||
under a strict threshold, there is no good configuration in that region of
|
||
the search space. The optimizer converged with <code>anneal_sec=59.2,
|
||
extinction_sec=59.2</code>, about 99% of the widened 60s bound, which raises an
|
||
open question not resolved in this round: does performance keep improving
|
||
past 60s, or does it plateau there. Not chased further this pass.</p>
|
||
<h2 id="caveats">Caveats<a class="headerlink" href="#caveats" title="Permanent link">¶</a></h2>
|
||
<ul>
|
||
<li>r50's 4 combos used the older, narrower search bounds (1-30/1-15 instead
|
||
of 1-60/1-60) and are further confounded by its thinner gallery. Excluded
|
||
from all comparisons above except calibration.</li>
|
||
<li>The shipped defaults use <code>full_exp</code> (75.3% training F1), not the
|
||
higher-scoring <code>restricted_exp</code> (78.3%), because cast restriction is not
|
||
a runtime feature of the application yet.</li>
|
||
<li><code>expand_gallery</code> is mode-dependent, not a free win. Averaged across models
|
||
on the full gallery it trades misIDs for recall (see the pose-expansion
|
||
table). For LVFace specifically, though, <code>full_exp</code> beats <code>full_noexp</code> on
|
||
every axis at once (F1 75.3 vs 72.3, precision 89.7 vs 88.3, recall 65.4 vs
|
||
61.8, misID 232 vs 244), so the shipped <code>full_exp</code> is a clean choice for
|
||
this model, not an F1-vs-safety trade. (An earlier version of this page
|
||
reported <code>full_noexp</code> at 72.4% with zero misIDs and higher precision, which
|
||
made it look like the safer option; that was the dropped-film artifact
|
||
described above, not a real property of the config.)</li>
|
||
<li>Switching the default model is an operational change: any gallery built
|
||
from a different model's embeddings must be rebuilt before the new
|
||
default takes effect.</li>
|
||
</ul>
|
||
<h2 id="reproduce">Reproduce<a class="headerlink" href="#reproduce" title="Permanent link">¶</a></h2>
|
||
<div class="language-bash highlight"><pre><span></span><code><span id="__span-0-1"><a id="__codelineno-0-1" name="__codelineno-0-1" href="#__codelineno-0-1"></a><span class="c1"># 4-film training matrix, all 4 models × 2 gallery modes × 2 expansion settings</span>
|
||
</span><span id="__span-0-2"><a id="__codelineno-0-2" name="__codelineno-0-2" href="#__codelineno-0-2"></a>bash<span class="w"> </span>experiments/run_rep4_subprocess.sh
|
||
</span><span id="__span-0-3"><a id="__codelineno-0-3" name="__codelineno-0-3" href="#__codelineno-0-3"></a>
|
||
</span><span id="__span-0-4"><a id="__codelineno-0-4" name="__codelineno-0-4" href="#__codelineno-0-4"></a><span class="c1"># single combo</span>
|
||
</span><span id="__span-0-5"><a id="__codelineno-0-5" name="__codelineno-0-5" href="#__codelineno-0-5"></a><span class="nv">SAE_EXPAND</span><span class="o">=</span><span class="m">1</span><span class="w"> </span><span class="nv">REPLAY_WORKERS</span><span class="o">=</span><span class="m">4</span><span class="w"> </span><span class="nv">DE_WORKERS</span><span class="o">=</span><span class="m">2</span><span class="w"> </span>python3<span class="w"> </span>scripts/optimizer/optimize.py<span class="w"> </span><span class="se">\</span>
|
||
</span><span id="__span-0-6"><a id="__codelineno-0-6" name="__codelineno-0-6" href="#__codelineno-0-6"></a><span class="w"> </span>--manifest<span class="w"> </span>experiments/manifests/rep4_LVFace-B_Glint360K_full.json<span class="w"> </span><span class="se">\</span>
|
||
</span><span id="__span-0-7"><a id="__codelineno-0-7" name="__codelineno-0-7" href="#__codelineno-0-7"></a><span class="w"> </span>--gallery<span class="w"> </span>experiments/galleries/gallery_LVFace-B_Glint360K.h5<span class="w"> </span><span class="se">\</span>
|
||
</span><span id="__span-0-8"><a id="__codelineno-0-8" name="__codelineno-0-8" href="#__codelineno-0-8"></a><span class="w"> </span>--params<span class="w"> </span>prob_threshold:0.5:0.999<span class="w"> </span>anneal_sec:1:60<span class="w"> </span>extinction_sec:1:60<span class="w"> </span><span class="se">\</span>
|
||
</span><span id="__span-0-9"><a id="__codelineno-0-9" name="__codelineno-0-9" href="#__codelineno-0-9"></a><span class="w"> </span>--popsize<span class="w"> </span><span class="m">10</span><span class="w"> </span>--maxiter<span class="w"> </span><span class="m">15</span><span class="w"> </span>--trajectory<span class="w"> </span>traj.jsonl<span class="w"> </span>--out<span class="w"> </span>best.json
|
||
</span><span id="__span-0-10"><a id="__codelineno-0-10" name="__codelineno-0-10" href="#__codelineno-0-10"></a>
|
||
</span><span id="__span-0-11"><a id="__codelineno-0-11" name="__codelineno-0-11" href="#__codelineno-0-11"></a><span class="c1"># held-out validation, all 3 models, 5 films</span>
|
||
</span><span id="__span-0-12"><a id="__codelineno-0-12" name="__codelineno-0-12" href="#__codelineno-0-12"></a>python3<span class="w"> </span>scripts/docs/run_holdout_all_models.py<span class="w"> </span>--out<span class="w"> </span>docs_data/holdout_all_models.json
|
||
</span><span id="__span-0-13"><a id="__codelineno-0-13" name="__codelineno-0-13" href="#__codelineno-0-13"></a>
|
||
</span><span id="__span-0-14"><a id="__codelineno-0-14" name="__codelineno-0-14" href="#__codelineno-0-14"></a><span class="c1"># per-film training breakdown, all 3 models, 4 films</span>
|
||
</span><span id="__span-0-15"><a id="__codelineno-0-15" name="__codelineno-0-15" href="#__codelineno-0-15"></a>python3<span class="w"> </span>scripts/docs/run_holdout_all_models.py<span class="w"> </span>--films<span class="w"> </span>training<span class="w"> </span>--out<span class="w"> </span>docs_data/training_per_film.json
|
||
</span><span id="__span-0-16"><a id="__codelineno-0-16" name="__codelineno-0-16" href="#__codelineno-0-16"></a>
|
||
</span><span id="__span-0-17"><a id="__codelineno-0-17" name="__codelineno-0-17" href="#__codelineno-0-17"></a><span class="c1"># gallery coverage per film</span>
|
||
</span><span id="__span-0-18"><a id="__codelineno-0-18" name="__codelineno-0-18" href="#__codelineno-0-18"></a>python3<span class="w"> </span>scripts/docs/gallery_coverage_per_film.py<span class="w"> </span>--out<span class="w"> </span>docs_data/gallery_coverage_per_film.json
|
||
</span><span id="__span-0-19"><a id="__codelineno-0-19" name="__codelineno-0-19" href="#__codelineno-0-19"></a>
|
||
</span><span id="__span-0-20"><a id="__codelineno-0-20" name="__codelineno-0-20" href="#__codelineno-0-20"></a><span class="c1"># regenerate this page's charts from experiments/ artifacts</span>
|
||
</span><span id="__span-0-21"><a id="__codelineno-0-21" name="__codelineno-0-21" href="#__codelineno-0-21"></a>python3<span class="w"> </span>scripts/docs/experiment_charts.py<span class="w"> </span>--out-dir<span class="w"> </span>docs/assets/images
|
||
</span><span id="__span-0-22"><a id="__codelineno-0-22" name="__codelineno-0-22" href="#__codelineno-0-22"></a>
|
||
</span><span id="__span-0-23"><a id="__codelineno-0-23" name="__codelineno-0-23" href="#__codelineno-0-23"></a><span class="c1"># one frame per distinct out-of-cast name across all 9 films (used in the deep dive)</span>
|
||
</span><span id="__span-0-24"><a id="__codelineno-0-24" name="__codelineno-0-24" href="#__codelineno-0-24"></a>python3<span class="w"> </span>scripts/docs/first_fpi_frames.py
|
||
</span></code></pre></div>
|
||
<p>See also the session log
|
||
<a href="https://gitea.tourolle.paris/dtourolle/scene-actor-extraction/raw/commit/e5885977dfd04741a0d1bb7aaf0270d9c40a2fa8/experiments/SESSION_STATE.md"><code>experiments/SESSION_STATE.md</code></a>.</p>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
</article>
|
||
</div>
|
||
|
||
|
||
<script>var target=document.getElementById(location.hash.slice(1));target&&target.name&&(target.checked=target.name.startsWith("__tabbed_"))</script>
|
||
</div>
|
||
|
||
<button type="button" class="md-top md-icon" data-md-component="top" hidden>
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M13 20h-2V8l-5.5 5.5-1.42-1.42L12 4.16l7.92 7.92-1.42 1.42L13 8z"/></svg>
|
||
Back to top
|
||
</button>
|
||
|
||
</main>
|
||
|
||
<footer class="md-footer">
|
||
|
||
|
||
|
||
<nav class="md-footer__inner md-grid" aria-label="Footer" >
|
||
|
||
|
||
<a href="../lvface-deep-dive/" class="md-footer__link md-footer__link--prev" aria-label="Previous: LVFace Deep Dive">
|
||
<div class="md-footer__button md-icon">
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M20 11v2H8l5.5 5.5-1.42 1.42L4.16 12l7.92-7.92L13.5 5.5 8 11z"/></svg>
|
||
</div>
|
||
<div class="md-footer__title">
|
||
<span class="md-footer__direction">
|
||
Previous
|
||
</span>
|
||
<div class="md-ellipsis">
|
||
LVFace Deep Dive
|
||
</div>
|
||
</div>
|
||
</a>
|
||
|
||
|
||
|
||
<a href="../service-conversion/" class="md-footer__link md-footer__link--next" aria-label="Next: Service Conversion (proposal)">
|
||
<div class="md-footer__title">
|
||
<span class="md-footer__direction">
|
||
Next
|
||
</span>
|
||
<div class="md-ellipsis">
|
||
Service Conversion (proposal)
|
||
</div>
|
||
</div>
|
||
<div class="md-footer__button md-icon">
|
||
|
||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 24 24"><path d="M4 11v2h12l-5.5 5.5 1.42 1.42L19.84 12l-7.92-7.92L10.5 5.5 16 11z"/></svg>
|
||
</div>
|
||
</a>
|
||
|
||
</nav>
|
||
|
||
|
||
<div class="md-footer-meta md-typeset">
|
||
<div class="md-footer-meta__inner md-grid">
|
||
<div class="md-copyright">
|
||
|
||
|
||
Made with
|
||
<a href="https://squidfunk.github.io/mkdocs-material/" target="_blank" rel="noopener">
|
||
Material for MkDocs
|
||
</a>
|
||
|
||
</div>
|
||
|
||
</div>
|
||
</div>
|
||
</footer>
|
||
|
||
</div>
|
||
<div class="md-dialog" data-md-component="dialog">
|
||
<div class="md-dialog__inner md-typeset"></div>
|
||
</div>
|
||
|
||
|
||
|
||
|
||
|
||
<script id="__config" type="application/json">{"annotate": null, "base": "..", "features": ["navigation.tabs", "navigation.sections", "navigation.top", "navigation.footer", "content.code.copy", "content.code.annotate"], "search": "../assets/javascripts/workers/search.2c215733.min.js", "tags": null, "translations": {"clipboard.copied": "Copied to clipboard", "clipboard.copy": "Copy to clipboard", "search.result.more.one": "1 more on this page", "search.result.more.other": "# more on this page", "search.result.none": "No matching documents", "search.result.one": "1 matching document", "search.result.other": "# matching documents", "search.result.placeholder": "Type to start searching", "search.result.term.missing": "Missing", "select.version": "Select version"}, "version": null}</script>
|
||
|
||
|
||
<script src="../assets/javascripts/bundle.d7400e89.min.js"></script>
|
||
|
||
|
||
</body>
|
||
</html> |