From c8e05831f4fdb82e134b11bf1169e0f5f7413625 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 09:57:38 +0200 Subject: [PATCH 1/9] Show the confidence, and say which curve it came from MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit FR-CULL-9 was read as "no fit, no number", so every library without 200 confirmed positive pairs showed "Confidence unavailable" on every suggestion — which is every library, until enough confirmations exist to fit one. The confirmations are made on this screen, ranked by the number it was withholding, so the degraded state was also the permanent one. There has always been a curve: Calibration::default is the reference implementation's fitted MBF sigmoid, which is what clustering already operates at. It is a published operating point, not an invention, and what the requirement forbids is presenting it *as though it were measured on this library*. So the percentage is shown, and the screen says once, above the grid, where the curve came from. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-face/src/calibrate.rs | 24 ++++++++++------- docs/faces.md | 14 +++++++--- docs/requirements.md | 9 +++++-- docs/traceability.md | 10 +++---- ui/dr-ui/src/identity.rs | 50 +++++++++++++++++------------------ ui/dr-ui/src/identity_ui.rs | 9 +++---- ui/dr-ui/ui/identity.slint | 20 +++++++------- 7 files changed, 76 insertions(+), 60 deletions(-) diff --git a/core/dr-face/src/calibrate.rs b/core/dr-face/src/calibrate.rs index 5d52c04..21c15f2 100644 --- a/core/dr-face/src/calibrate.rs +++ b/core/dr-face/src/calibrate.rs @@ -27,7 +27,11 @@ //! person. Nothing else — bootstrapping positives from high cosine is circular, //! fitting the calibration to the belief it was supposed to test. //! -//! Which is why a fresh library has **no valid calibration**, and says so. +//! Which is why a fresh library has **no valid calibration** — no fit of its +//! own — and says so. It is not left without a curve: it uses the reference +//! implementation's fitted one ([`Calibration::default`]), which is a published +//! operating point rather than an invention, and the interface reports which of +//! the two it is speaking from. /// Bins over cosine ∈ [-1, 1]. /// @@ -56,11 +60,13 @@ pub struct Calibration { pub b: f32, /// Weight on `log2(min crop_px)` — the face-size term FR-CULL-9 asks for. pub w_size: f32, - /// Whether there was enough evidence to trust the fit. + /// Whether there was enough evidence to fit this library's own curve. /// - /// When false the UI says confidence is unavailable. It does **not** present - /// an untuned default as though it were measured, which is the distinction - /// FR-CULL-9 spends a paragraph on. + /// False means the numbers came from the built-in reference curve, and the + /// UI says so — once, at the screen level. It does **not** mean confidences + /// are withheld: FR-CULL-9's distinction is between presenting an untuned + /// default *as though it were measured* and presenting it as what it is, + /// and only the first is forbidden. pub valid: bool, pub positive_pairs: u64, pub negative_pairs: u64, @@ -70,10 +76,10 @@ impl Default for Calibration { /// The reference implementation's fitted MBF curve (docs/faces.md §1): /// steepness 16.2, P=0.5 at cosine 0.267. /// - /// **`valid` is false**, and that is the point. This exists so an - /// un-calibrated library has a documented operating point to cluster at - /// rather than no behaviour at all — but nothing may show its output as a - /// measured confidence. + /// **`valid` is false**, and that is the point. It is a documented + /// operating point rather than an invented one, so a library with no fit of + /// its own can both cluster and quote a probability from it — what it may + /// not do is call that probability a measurement of *this* library. fn default() -> Self { Self { a: 16.2, diff --git a/docs/faces.md b/docs/faces.md index b86e5cd..b72fa65 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -614,10 +614,16 @@ positive pair is trustworthy by construction. Here the positives are bootstrappe early confirmations (§8.1) and the whole risk is fitting confidently to a handful of them. The reference also refuses to draw positives from an identity with fewer than five distinct embeddings, letting it contribute negatives only — the same asymmetry applies to a thinly-confirmed person and is -worth keeping. In that state the UI says confidence is unavailable and the People view -still works — clustering falls back to a documented default operating point, labelled in the -interface as an untuned default, and no probability is displayed. That is FR-CULL-9's requirement -read literally: not presenting an untuned default *as though it were measured*. +worth keeping. In that state both clustering and the displayed confidences fall back to the +reference implementation's fitted curve — a documented operating point, not an invention — and the +People screen says so once, above the grid, rather than blanking every percentage. That is +FR-CULL-9's requirement read as written: what may not happen is an untuned default presented *as +though it were measured* on this library. + +Blanking them was the first reading, and it was wrong in a way worth recording. A young library has +no fit; a fit needs confirmations; confirmations are made on a screen the user ranks by confidence. +Withholding the confidence until the fit exists is a deadlock in which the normal state of the +feature is its degraded one. Refit is triggered by the same debounce as clustering (§9), and when the confirmed-pair count grows materially. diff --git a/docs/requirements.md b/docs/requirements.md index 8ce0b58..4cfb55c 100644 --- a/docs/requirements.md +++ b/docs/requirements.md @@ -908,8 +908,13 @@ similarity still *looks* like a plausible number all the way to the user interfa confidence that does not mean what it says is worse than no confidence, because it is trusted. The calibration shall be fitted per library from that library's own faces, and shall report whether -it is valid. Where it is not — too few examples to fit — the app shall say the confidence is -unavailable rather than present an untuned default as though it were measured. +it is valid. Where it is not — too few examples to fit — the app shall fall back to a **documented, +published operating point** (the reference implementation's fitted curve) and shall say, at the +screen level, that the confidences come from it. What is forbidden is presenting an untuned default +*as though it were measured on this library*; withholding the number entirely is not required and +shall not be done, because a screen of unranked suggestions is the state most libraries would +permanently sit in — the fit needs confirmations, and confirmations need a ranked screen to be made +on. *Acceptance:* on a labelled corpus, the stated probability is within a documented tolerance of the observed match rate across the probability range (a reliability-diagram check, not a single diff --git a/docs/traceability.md b/docs/traceability.md index ca44ba4..7902f84 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -9,8 +9,8 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n | Metric | Value | |---|---| -| Source files scanned | 278 | -| TRACES tags found | 811 | +| Source files scanned | 279 | +| TRACES tags found | 812 | | Requirements defined | 177 | | Requirements covered | 106 | | **Coverage** | **59.9%** (106/177) | @@ -49,13 +49,13 @@ _None._ | FR-CAT-8 | [`core/dr-pipeline/src/graph.rs:345`](../core/dr-pipeline/src/graph.rs#L345), [`core/dr-pipeline/src/graph.rs:384`](../core/dr-pipeline/src/graph.rs#L384), [`core/dr-pipeline/src/ops/curve.rs:137`](../core/dr-pipeline/src/ops/curve.rs#L137), [`core/dr-pipeline/src/ops/curve.rs:656`](../core/dr-pipeline/src/ops/curve.rs#L656), [`core/dr-pipeline/src/sidecar.rs:1636`](../core/dr-pipeline/src/sidecar.rs#L1636), [`core/dr-pipeline/src/sidecar.rs:92`](../core/dr-pipeline/src/sidecar.rs#L92), [`core/dr-pipeline/src/state.rs:1`](../core/dr-pipeline/src/state.rs#L1), [`core/dr-pipeline/src/state.rs:75`](../core/dr-pipeline/src/state.rs#L75), [`core/dr-pipeline/tests/tone_curve.rs:34`](../core/dr-pipeline/tests/tone_curve.rs#L34), [`ui/dr-ui/src/develop.rs:3345`](../ui/dr-ui/src/develop.rs#L3345), [`ui/dr-ui/src/develop.rs:3374`](../ui/dr-ui/src/develop.rs#L3374), [`ui/dr-ui/src/export.rs:752`](../ui/dr-ui/src/export.rs#L752), [`ui/dr-ui/src/lib.rs:1387`](../ui/dr-ui/src/lib.rs#L1387), [`ui/dr-ui/src/lib.rs:1788`](../ui/dr-ui/src/lib.rs#L1788), [`ui/dr-ui/src/lib.rs:1924`](../ui/dr-ui/src/lib.rs#L1924), [`ui/dr-ui/src/lib.rs:498`](../ui/dr-ui/src/lib.rs#L498), [`ui/dr-ui/src/lib.rs:923`](../ui/dr-ui/src/lib.rs#L923), [`ui/dr-ui/src/library.rs:1656`](../ui/dr-ui/src/library.rs#L1656), [`ui/dr-ui/src/library.rs:460`](../ui/dr-ui/src/library.rs#L460), [`ui/dr-ui/src/library.rs:507`](../ui/dr-ui/src/library.rs#L507), [`ui/dr-ui/src/library.rs:544`](../ui/dr-ui/src/library.rs#L544), [`ui/dr-ui/src/library.rs:792`](../ui/dr-ui/src/library.rs#L792), [`ui/dr-ui/src/library_ui.rs:4885`](../ui/dr-ui/src/library_ui.rs#L4885), [`ui/dr-ui/src/sidecar_cache.rs:1`](../ui/dr-ui/src/sidecar_cache.rs#L1) | | FR-CAT-9 | [`core/dr-catalog/src/cache.rs:1`](../core/dr-catalog/src/cache.rs#L1), [`core/dr-catalog/src/scan.rs:1`](../core/dr-catalog/src/scan.rs#L1), [`core/dr-catalog/src/schema.rs:613`](../core/dr-catalog/src/schema.rs#L613), [`core/dr-catalog/src/walk.rs:162`](../core/dr-catalog/src/walk.rs#L162), [`core/dr-catalog/src/walk.rs:1`](../core/dr-catalog/src/walk.rs#L1), [`core/dr-catalog/src/walk.rs:435`](../core/dr-catalog/src/walk.rs#L435), [`core/dr-catalog/src/walk.rs:704`](../core/dr-catalog/src/walk.rs#L704), [`core/dr-sync-nextcloud/src/desktop_client.rs:30`](../core/dr-sync-nextcloud/src/desktop_client.rs#L30), [`core/dr-sync/src/reachability.rs:1`](../core/dr-sync/src/reachability.rs#L1), [`core/dr-types/src/lib.rs:119`](../core/dr-types/src/lib.rs#L119), [`ui/dr-ui/src/develop.rs:2870`](../ui/dr-ui/src/develop.rs#L2870), [`ui/dr-ui/src/library.rs:149`](../ui/dr-ui/src/library.rs#L149), [`ui/dr-ui/src/library.rs:1616`](../ui/dr-ui/src/library.rs#L1616), [`ui/dr-ui/src/library.rs:1693`](../ui/dr-ui/src/library.rs#L1693), [`ui/dr-ui/src/library.rs:234`](../ui/dr-ui/src/library.rs#L234), [`ui/dr-ui/src/library.rs:4062`](../ui/dr-ui/src/library.rs#L4062), [`ui/dr-ui/src/library.rs:544`](../ui/dr-ui/src/library.rs#L544), [`ui/dr-ui/src/library.rs:776`](../ui/dr-ui/src/library.rs#L776), [`ui/dr-ui/src/library.rs:792`](../ui/dr-ui/src/library.rs#L792), [`ui/dr-ui/src/library.rs:846`](../ui/dr-ui/src/library.rs#L846), [`ui/dr-ui/src/library_ui.rs:1565`](../ui/dr-ui/src/library_ui.rs#L1565), [`ui/dr-ui/src/library_ui.rs:1591`](../ui/dr-ui/src/library_ui.rs#L1591), [`ui/dr-ui/src/library_ui.rs:1607`](../ui/dr-ui/src/library_ui.rs#L1607), [`ui/dr-ui/src/library_ui.rs:1701`](../ui/dr-ui/src/library_ui.rs#L1701), [`ui/dr-ui/src/library_ui.rs:227`](../ui/dr-ui/src/library_ui.rs#L227), [`ui/dr-ui/src/library_ui.rs:2317`](../ui/dr-ui/src/library_ui.rs#L2317), [`ui/dr-ui/src/library_ui.rs:260`](../ui/dr-ui/src/library_ui.rs#L260), [`ui/dr-ui/src/library_ui.rs:2755`](../ui/dr-ui/src/library_ui.rs#L2755), [`ui/dr-ui/src/library_ui.rs:2979`](../ui/dr-ui/src/library_ui.rs#L2979), [`ui/dr-ui/src/library_ui.rs:3260`](../ui/dr-ui/src/library_ui.rs#L3260), [`ui/dr-ui/src/library_ui.rs:3348`](../ui/dr-ui/src/library_ui.rs#L3348), [`ui/dr-ui/src/library_ui.rs:3536`](../ui/dr-ui/src/library_ui.rs#L3536), [`ui/dr-ui/src/library_ui.rs:3654`](../ui/dr-ui/src/library_ui.rs#L3654), [`ui/dr-ui/src/library_ui.rs:441`](../ui/dr-ui/src/library_ui.rs#L441), [`ui/dr-ui/src/library_ui.rs:499`](../ui/dr-ui/src/library_ui.rs#L499), [`ui/dr-ui/src/library_ui.rs:5302`](../ui/dr-ui/src/library_ui.rs#L5302), [`ui/dr-ui/src/library_ui.rs:5417`](../ui/dr-ui/src/library_ui.rs#L5417), [`ui/dr-ui/src/presets.rs:326`](../ui/dr-ui/src/presets.rs#L326), [`ui/dr-ui/src/presets.rs:338`](../ui/dr-ui/src/presets.rs#L338), [`ui/dr-ui/src/sidecar_cache.rs:1`](../ui/dr-ui/src/sidecar_cache.rs#L1) | | FR-CULL-1 | [`core/dr-decode/src/preview.rs:121`](../core/dr-decode/src/preview.rs#L121) | -| FR-CULL-10 | [`core/dr-catalog/src/faces.rs:1`](../core/dr-catalog/src/faces.rs#L1), [`core/dr-catalog/src/schema.rs:357`](../core/dr-catalog/src/schema.rs#L357), [`core/dr-catalog/src/schema.rs:442`](../core/dr-catalog/src/schema.rs#L442), [`core/dr-face/src/neighbours.rs:1`](../core/dr-face/src/neighbours.rs#L1), [`ui/dr-ui/src/develop.rs:119`](../ui/dr-ui/src/develop.rs#L119), [`ui/dr-ui/src/develop.rs:128`](../ui/dr-ui/src/develop.rs#L128), [`ui/dr-ui/src/develop.rs:1793`](../ui/dr-ui/src/develop.rs#L1793), [`ui/dr-ui/src/develop.rs:194`](../ui/dr-ui/src/develop.rs#L194), [`ui/dr-ui/src/develop.rs:589`](../ui/dr-ui/src/develop.rs#L589), [`ui/dr-ui/src/faces.rs:1`](../ui/dr-ui/src/faces.rs#L1), [`ui/dr-ui/src/identity.rs:1`](../ui/dr-ui/src/identity.rs#L1), [`ui/dr-ui/src/identity_ui.rs:1`](../ui/dr-ui/src/identity_ui.rs#L1), [`ui/dr-ui/src/lib.rs:1896`](../ui/dr-ui/src/lib.rs#L1896), [`ui/dr-ui/ui/identity.slint:1`](../ui/dr-ui/ui/identity.slint#L1) | +| FR-CULL-10 | [`core/dr-catalog/src/faces.rs:1`](../core/dr-catalog/src/faces.rs#L1), [`core/dr-catalog/src/schema.rs:357`](../core/dr-catalog/src/schema.rs#L357), [`core/dr-catalog/src/schema.rs:442`](../core/dr-catalog/src/schema.rs#L442), [`core/dr-face/src/assign.rs:1`](../core/dr-face/src/assign.rs#L1), [`core/dr-face/src/neighbours.rs:1`](../core/dr-face/src/neighbours.rs#L1), [`ui/dr-ui/src/develop.rs:119`](../ui/dr-ui/src/develop.rs#L119), [`ui/dr-ui/src/develop.rs:128`](../ui/dr-ui/src/develop.rs#L128), [`ui/dr-ui/src/develop.rs:1793`](../ui/dr-ui/src/develop.rs#L1793), [`ui/dr-ui/src/develop.rs:194`](../ui/dr-ui/src/develop.rs#L194), [`ui/dr-ui/src/develop.rs:589`](../ui/dr-ui/src/develop.rs#L589), [`ui/dr-ui/src/faces.rs:1`](../ui/dr-ui/src/faces.rs#L1), [`ui/dr-ui/src/identity.rs:1`](../ui/dr-ui/src/identity.rs#L1), [`ui/dr-ui/src/identity_ui.rs:1`](../ui/dr-ui/src/identity_ui.rs#L1), [`ui/dr-ui/src/lib.rs:1896`](../ui/dr-ui/src/lib.rs#L1896), [`ui/dr-ui/ui/identity.slint:1`](../ui/dr-ui/ui/identity.slint#L1) | | FR-CULL-11 | [`core/dr-catalog/src/faces.rs:1`](../core/dr-catalog/src/faces.rs#L1), [`core/dr-catalog/src/schema.rs:442`](../core/dr-catalog/src/schema.rs#L442), [`ui/dr-ui/src/identity.rs:1`](../ui/dr-ui/src/identity.rs#L1), [`ui/dr-ui/src/identity_ui.rs:1`](../ui/dr-ui/src/identity_ui.rs#L1), [`ui/dr-ui/src/library.rs:255`](../ui/dr-ui/src/library.rs#L255), [`ui/dr-ui/src/library.rs:285`](../ui/dr-ui/src/library.rs#L285), [`ui/dr-ui/ui/identity.slint:1`](../ui/dr-ui/ui/identity.slint#L1) | | FR-CULL-12 | [`core/dr-catalog/src/faces.rs:1`](../core/dr-catalog/src/faces.rs#L1), [`core/dr-catalog/src/schema.rs:357`](../core/dr-catalog/src/schema.rs#L357), [`core/dr-catalog/src/schema.rs:442`](../core/dr-catalog/src/schema.rs#L442), [`ui/dr-ui/src/identity.rs:1`](../ui/dr-ui/src/identity.rs#L1), [`ui/dr-ui/ui/identity.slint:1`](../ui/dr-ui/ui/identity.slint#L1) | | FR-CULL-2 | [`core/dr-decode/src/locate.rs:1`](../core/dr-decode/src/locate.rs#L1), [`core/dr-decode/src/preview.rs:148`](../core/dr-decode/src/preview.rs#L148), [`ui/dr-ui/src/import.rs:464`](../ui/dr-ui/src/import.rs#L464) | | FR-CULL-4 | [`core/dr-catalog/src/rating.rs:1`](../core/dr-catalog/src/rating.rs#L1), [`core/dr-pipeline/src/sidecar.rs:135`](../core/dr-pipeline/src/sidecar.rs#L135), [`ui/dr-ui/src/library.rs:214`](../ui/dr-ui/src/library.rs#L214), [`ui/dr-ui/src/library.rs:460`](../ui/dr-ui/src/library.rs#L460) | | FR-CULL-8 | [`core/dr-catalog/src/face_shard.rs:1`](../core/dr-catalog/src/face_shard.rs#L1), [`core/dr-catalog/src/faces.rs:1`](../core/dr-catalog/src/faces.rs#L1), [`core/dr-catalog/src/schema.rs:401`](../core/dr-catalog/src/schema.rs#L401), [`core/dr-catalog/src/schema.rs:442`](../core/dr-catalog/src/schema.rs#L442), [`ui/dr-ui/src/faces.rs:1`](../ui/dr-ui/src/faces.rs#L1), [`ui/dr-ui/src/library.rs:2725`](../ui/dr-ui/src/library.rs#L2725), [`ui/dr-ui/src/library.rs:2829`](../ui/dr-ui/src/library.rs#L2829), [`ui/dr-ui/ui/settings.slint:404`](../ui/dr-ui/ui/settings.slint#L404), [`ui/dr-ui/ui/settings.slint:81`](../ui/dr-ui/ui/settings.slint#L81) | -| FR-CULL-9 | [`core/dr-catalog/src/faces.rs:1`](../core/dr-catalog/src/faces.rs#L1), [`core/dr-catalog/src/schema.rs:442`](../core/dr-catalog/src/schema.rs#L442), [`core/dr-face/src/neighbours.rs:1`](../core/dr-face/src/neighbours.rs#L1), [`ui/dr-ui/src/faces.rs:1`](../ui/dr-ui/src/faces.rs#L1), [`ui/dr-ui/src/identity_ui.rs:1`](../ui/dr-ui/src/identity_ui.rs#L1) | +| FR-CULL-9 | [`core/dr-catalog/src/faces.rs:1`](../core/dr-catalog/src/faces.rs#L1), [`core/dr-catalog/src/schema.rs:442`](../core/dr-catalog/src/schema.rs#L442), [`core/dr-face/src/assign.rs:1`](../core/dr-face/src/assign.rs#L1), [`core/dr-face/src/neighbours.rs:1`](../core/dr-face/src/neighbours.rs#L1), [`ui/dr-ui/src/faces.rs:1`](../ui/dr-ui/src/faces.rs#L1), [`ui/dr-ui/src/identity_ui.rs:1`](../ui/dr-ui/src/identity_ui.rs#L1) | | FR-DEV-2 | [`core/dr-pipeline/src/operation.rs:389`](../core/dr-pipeline/src/operation.rs#L389) | | FR-DEV-3 | [`core/dr-gpu/src/adjust.rs:2165`](../core/dr-gpu/src/adjust.rs#L2165), [`core/dr-gpu/src/adjust.rs:651`](../core/dr-gpu/src/adjust.rs#L651), [`core/dr-gpu/src/adjust.rs:770`](../core/dr-gpu/src/adjust.rs#L770), [`core/dr-gpu/src/adjust.rs:84`](../core/dr-gpu/src/adjust.rs#L84), [`core/dr-gpu/tests/tone_curve.rs:1`](../core/dr-gpu/tests/tone_curve.rs#L1), [`core/dr-pipeline/src/detail.rs:387`](../core/dr-pipeline/src/detail.rs#L387), [`core/dr-pipeline/src/detail.rs:465`](../core/dr-pipeline/src/detail.rs#L465), [`core/dr-pipeline/src/framing.rs:191`](../core/dr-pipeline/src/framing.rs#L191), [`core/dr-pipeline/src/framing.rs:365`](../core/dr-pipeline/src/framing.rs#L365), [`core/dr-pipeline/src/framing.rs:620`](../core/dr-pipeline/src/framing.rs#L620), [`core/dr-pipeline/src/graph.rs:169`](../core/dr-pipeline/src/graph.rs#L169), [`core/dr-pipeline/src/graph.rs:577`](../core/dr-pipeline/src/graph.rs#L577), [`core/dr-pipeline/src/mask.rs:121`](../core/dr-pipeline/src/mask.rs#L121), [`core/dr-pipeline/src/operation.rs:330`](../core/dr-pipeline/src/operation.rs#L330), [`core/dr-pipeline/src/operation.rs:516`](../core/dr-pipeline/src/operation.rs#L516), [`core/dr-pipeline/src/ops/capture_sharpen.rs:1`](../core/dr-pipeline/src/ops/capture_sharpen.rs#L1), [`core/dr-pipeline/src/ops/capture_sharpen.rs:210`](../core/dr-pipeline/src/ops/capture_sharpen.rs#L210), [`core/dr-pipeline/src/ops/curve.rs:100`](../core/dr-pipeline/src/ops/curve.rs#L100), [`core/dr-pipeline/src/ops/curve.rs:1`](../core/dr-pipeline/src/ops/curve.rs#L1), [`core/dr-pipeline/src/ops/curve.rs:219`](../core/dr-pipeline/src/ops/curve.rs#L219), [`core/dr-pipeline/src/ops/curve.rs:635`](../core/dr-pipeline/src/ops/curve.rs#L635), [`core/dr-pipeline/src/ops/local_contrast.rs:1`](../core/dr-pipeline/src/ops/local_contrast.rs#L1), [`core/dr-pipeline/src/ops/noise_reduction.rs:1`](../core/dr-pipeline/src/ops/noise_reduction.rs#L1), [`core/dr-pipeline/src/ops/noise_reduction.rs:273`](../core/dr-pipeline/src/ops/noise_reduction.rs#L273), [`core/dr-pipeline/src/sidecar.rs:156`](../core/dr-pipeline/src/sidecar.rs#L156), [`core/dr-pipeline/src/sidecar.rs:1636`](../core/dr-pipeline/src/sidecar.rs#L1636), [`core/dr-pipeline/src/sidecar.rs:1696`](../core/dr-pipeline/src/sidecar.rs#L1696), [`core/dr-pipeline/tests/tone_curve.rs:1`](../core/dr-pipeline/tests/tone_curve.rs#L1), [`ui/dr-ui/src/develop.rs:101`](../ui/dr-ui/src/develop.rs#L101), [`ui/dr-ui/src/develop.rs:1297`](../ui/dr-ui/src/develop.rs#L1297), [`ui/dr-ui/src/develop.rs:163`](../ui/dr-ui/src/develop.rs#L163), [`ui/dr-ui/src/develop.rs:1775`](../ui/dr-ui/src/develop.rs#L1775), [`ui/dr-ui/src/develop.rs:1793`](../ui/dr-ui/src/develop.rs#L1793), [`ui/dr-ui/src/develop.rs:1807`](../ui/dr-ui/src/develop.rs#L1807), [`ui/dr-ui/src/develop.rs:1829`](../ui/dr-ui/src/develop.rs#L1829), [`ui/dr-ui/src/develop.rs:1975`](../ui/dr-ui/src/develop.rs#L1975), [`ui/dr-ui/src/develop.rs:2073`](../ui/dr-ui/src/develop.rs#L2073), [`ui/dr-ui/src/develop.rs:326`](../ui/dr-ui/src/develop.rs#L326), [`ui/dr-ui/src/develop.rs:3345`](../ui/dr-ui/src/develop.rs#L3345), [`ui/dr-ui/src/develop.rs:363`](../ui/dr-ui/src/develop.rs#L363), [`ui/dr-ui/src/develop.rs:3909`](../ui/dr-ui/src/develop.rs#L3909), [`ui/dr-ui/src/develop.rs:3963`](../ui/dr-ui/src/develop.rs#L3963), [`ui/dr-ui/src/develop.rs:4007`](../ui/dr-ui/src/develop.rs#L4007), [`ui/dr-ui/src/develop.rs:4057`](../ui/dr-ui/src/develop.rs#L4057), [`ui/dr-ui/src/develop.rs:628`](../ui/dr-ui/src/develop.rs#L628), [`ui/dr-ui/src/develop.rs:675`](../ui/dr-ui/src/develop.rs#L675), [`ui/dr-ui/src/lib.rs:1472`](../ui/dr-ui/src/lib.rs#L1472), [`ui/dr-ui/src/lib.rs:2174`](../ui/dr-ui/src/lib.rs#L2174), [`ui/dr-ui/src/lib.rs:319`](../ui/dr-ui/src/lib.rs#L319), [`ui/dr-ui/src/library.rs:507`](../ui/dr-ui/src/library.rs#L507), [`ui/dr-ui/src/masks_ui.rs:218`](../ui/dr-ui/src/masks_ui.rs#L218), [`ui/dr-ui/src/masks_ui.rs:41`](../ui/dr-ui/src/masks_ui.rs#L41), [`ui/dr-ui/src/masks_ui.rs:816`](../ui/dr-ui/src/masks_ui.rs#L816), [`ui/dr-ui/src/masks_ui.rs:930`](../ui/dr-ui/src/masks_ui.rs#L930), [`ui/dr-ui/src/segmentation.rs:219`](../ui/dr-ui/src/segmentation.rs#L219), [`ui/dr-ui/src/segmentation.rs:322`](../ui/dr-ui/src/segmentation.rs#L322), [`ui/dr-ui/src/segmentation.rs:350`](../ui/dr-ui/src/segmentation.rs#L350), [`ui/dr-ui/ui/app.slint:1761`](../ui/dr-ui/ui/app.slint#L1761), [`ui/dr-ui/ui/app.slint:790`](../ui/dr-ui/ui/app.slint#L790), [`ui/dr-ui/ui/masks.slint:490`](../ui/dr-ui/ui/masks.slint#L490) | | FR-DEV-3a | [`core/dr-pipeline/build.rs:756`](../core/dr-pipeline/build.rs#L756), [`core/dr-pipeline/ops/exposure.yaml:1`](../core/dr-pipeline/ops/exposure.yaml#L1), [`core/dr-pipeline/src/descriptor.rs:194`](../core/dr-pipeline/src/descriptor.rs#L194), [`core/dr-pipeline/src/descriptor.rs:234`](../core/dr-pipeline/src/descriptor.rs#L234), [`core/dr-pipeline/src/descriptor.rs:258`](../core/dr-pipeline/src/descriptor.rs#L258), [`core/dr-pipeline/src/descriptor.rs:313`](../core/dr-pipeline/src/descriptor.rs#L313), [`core/dr-pipeline/src/framing.rs:262`](../core/dr-pipeline/src/framing.rs#L262), [`core/dr-pipeline/src/graph.rs:23`](../core/dr-pipeline/src/graph.rs#L23), [`core/dr-pipeline/src/graph.rs:250`](../core/dr-pipeline/src/graph.rs#L250), [`core/dr-pipeline/src/graph.rs:45`](../core/dr-pipeline/src/graph.rs#L45), [`core/dr-pipeline/src/graph.rs:58`](../core/dr-pipeline/src/graph.rs#L58), [`core/dr-pipeline/src/mask.rs:955`](../core/dr-pipeline/src/mask.rs#L955), [`core/dr-pipeline/src/operation.rs:232`](../core/dr-pipeline/src/operation.rs#L232), [`core/dr-pipeline/src/operation.rs:365`](../core/dr-pipeline/src/operation.rs#L365), [`core/dr-pipeline/src/ops/curve.rs:319`](../core/dr-pipeline/src/ops/curve.rs#L319), [`ui/dr-ui/src/develop.rs:1194`](../ui/dr-ui/src/develop.rs#L1194), [`ui/dr-ui/src/lib.rs:613`](../ui/dr-ui/src/lib.rs#L613), [`ui/dr-ui/tests/ui_names_no_operation.rs:1`](../ui/dr-ui/tests/ui_names_no_operation.rs#L1) | @@ -111,7 +111,7 @@ _None._ | FR-RAW-3 | [`core/dr-decode/src/lib.rs:139`](../core/dr-decode/src/lib.rs#L139), [`core/dr-decode/src/lib.rs:506`](../core/dr-decode/src/lib.rs#L506), [`core/dr-decode/src/locate.rs:1366`](../core/dr-decode/src/locate.rs#L1366) | | FR-RAW-4 | [`core/dr-decode/src/error.rs:1`](../core/dr-decode/src/error.rs#L1), [`ui/dr-ui/src/lib.rs:204`](../ui/dr-ui/src/lib.rs#L204) | | FR-RAW-5 | [`core/dr-decode/src/lib.rs:167`](../core/dr-decode/src/lib.rs#L167), [`core/dr-gpu/src/demosaic.rs:34`](../core/dr-gpu/src/demosaic.rs#L34), [`core/dr-gpu/src/demosaic.rs:602`](../core/dr-gpu/src/demosaic.rs#L602), [`core/dr-gpu/src/demosaic.rs:681`](../core/dr-gpu/src/demosaic.rs#L681), [`core/dr-gpu/src/demosaic.rs:805`](../core/dr-gpu/src/demosaic.rs#L805) | -| FR-UI-1 | [`ui/dr-ui/src/lib.rs:2792`](../ui/dr-ui/src/lib.rs#L2792), [`ui/dr-ui/src/lib.rs:80`](../ui/dr-ui/src/lib.rs#L80), [`ui/dr-ui/src/masks_ui.rs:816`](../ui/dr-ui/src/masks_ui.rs#L816), [`ui/dr-ui/ui/identity.slint:165`](../ui/dr-ui/ui/identity.slint#L165), [`ui/dr-ui/ui/library.slint:1046`](../ui/dr-ui/ui/library.slint#L1046) | +| FR-UI-1 | [`ui/dr-ui/src/lib.rs:2792`](../ui/dr-ui/src/lib.rs#L2792), [`ui/dr-ui/src/lib.rs:80`](../ui/dr-ui/src/lib.rs#L80), [`ui/dr-ui/src/masks_ui.rs:816`](../ui/dr-ui/src/masks_ui.rs#L816), [`ui/dr-ui/ui/identity.slint:164`](../ui/dr-ui/ui/identity.slint#L164), [`ui/dr-ui/ui/library.slint:1046`](../ui/dr-ui/ui/library.slint#L1046) | | FR-UI-2 | [`ui/dr-ui/src/collections_ui.rs:1005`](../ui/dr-ui/src/collections_ui.rs#L1005), [`ui/dr-ui/src/collections_ui.rs:118`](../ui/dr-ui/src/collections_ui.rs#L118), [`ui/dr-ui/src/collections_ui.rs:1508`](../ui/dr-ui/src/collections_ui.rs#L1508), [`ui/dr-ui/src/collections_ui.rs:1522`](../ui/dr-ui/src/collections_ui.rs#L1522), [`ui/dr-ui/src/collections_ui.rs:1568`](../ui/dr-ui/src/collections_ui.rs#L1568), [`ui/dr-ui/src/collections_ui.rs:159`](../ui/dr-ui/src/collections_ui.rs#L159), [`ui/dr-ui/src/collections_ui.rs:1666`](../ui/dr-ui/src/collections_ui.rs#L1666), [`ui/dr-ui/src/collections_ui.rs:485`](../ui/dr-ui/src/collections_ui.rs#L485), [`ui/dr-ui/src/collections_ui.rs:514`](../ui/dr-ui/src/collections_ui.rs#L514), [`ui/dr-ui/src/collections_ui.rs:995`](../ui/dr-ui/src/collections_ui.rs#L995), [`ui/dr-ui/src/lib.rs:80`](../ui/dr-ui/src/lib.rs#L80), [`ui/dr-ui/src/lib.rs:87`](../ui/dr-ui/src/lib.rs#L87), [`ui/dr-ui/src/library_ui.rs:287`](../ui/dr-ui/src/library_ui.rs#L287), [`ui/dr-ui/src/library_ui.rs:5204`](../ui/dr-ui/src/library_ui.rs#L5204), [`ui/dr-ui/src/library_ui.rs:5349`](../ui/dr-ui/src/library_ui.rs#L5349), [`ui/dr-ui/src/library_ui.rs:6480`](../ui/dr-ui/src/library_ui.rs#L6480), [`ui/dr-ui/ui/adjust.slint:506`](../ui/dr-ui/ui/adjust.slint#L506), [`ui/dr-ui/ui/adjust.slint:641`](../ui/dr-ui/ui/adjust.slint#L641), [`ui/dr-ui/ui/adjust.slint:962`](../ui/dr-ui/ui/adjust.slint#L962), [`ui/dr-ui/ui/app.slint:1945`](../ui/dr-ui/ui/app.slint#L1945), [`ui/dr-ui/ui/app.slint:461`](../ui/dr-ui/ui/app.slint#L461), [`ui/dr-ui/ui/app.slint:468`](../ui/dr-ui/ui/app.slint#L468), [`ui/dr-ui/ui/app.slint:54`](../ui/dr-ui/ui/app.slint#L54), [`ui/dr-ui/ui/app.slint:761`](../ui/dr-ui/ui/app.slint#L761), [`ui/dr-ui/ui/develop.slint:223`](../ui/dr-ui/ui/develop.slint#L223), [`ui/dr-ui/ui/histogram.slint:127`](../ui/dr-ui/ui/histogram.slint#L127), [`ui/dr-ui/ui/history.slint:118`](../ui/dr-ui/ui/history.slint#L118), [`ui/dr-ui/ui/library.slint:1202`](../ui/dr-ui/ui/library.slint#L1202), [`ui/dr-ui/ui/library.slint:1209`](../ui/dr-ui/ui/library.slint#L1209), [`ui/dr-ui/ui/library.slint:1215`](../ui/dr-ui/ui/library.slint#L1215), [`ui/dr-ui/ui/library.slint:2314`](../ui/dr-ui/ui/library.slint#L2314), [`ui/dr-ui/ui/library.slint:829`](../ui/dr-ui/ui/library.slint#L829), [`ui/dr-ui/ui/library.slint:879`](../ui/dr-ui/ui/library.slint#L879), [`ui/dr-ui/ui/masks.slint:304`](../ui/dr-ui/ui/masks.slint#L304), [`ui/dr-ui/ui/settings.slint:106`](../ui/dr-ui/ui/settings.slint#L106), [`ui/dr-ui/ui/spots.slint:88`](../ui/dr-ui/ui/spots.slint#L88) | | FR-UI-3 | [`ui/dr-ui/src/develop.rs:2073`](../ui/dr-ui/src/develop.rs#L2073), [`ui/dr-ui/src/develop.rs:2182`](../ui/dr-ui/src/develop.rs#L2182), [`ui/dr-ui/src/library_ui.rs:4388`](../ui/dr-ui/src/library_ui.rs#L4388), [`ui/dr-ui/src/masks_ui.rs:218`](../ui/dr-ui/src/masks_ui.rs#L218), [`ui/dr-ui/src/masks_ui.rs:908`](../ui/dr-ui/src/masks_ui.rs#L908), [`ui/dr-ui/src/masks_ui.rs:930`](../ui/dr-ui/src/masks_ui.rs#L930), [`ui/dr-ui/src/spots_ui.rs:19`](../ui/dr-ui/src/spots_ui.rs#L19), [`ui/dr-ui/ui/app.slint:1761`](../ui/dr-ui/ui/app.slint#L1761), [`ui/dr-ui/ui/collections.slint:4`](../ui/dr-ui/ui/collections.slint#L4), [`ui/dr-ui/ui/collections.slint:682`](../ui/dr-ui/ui/collections.slint#L682), [`ui/dr-ui/ui/masks.slint:490`](../ui/dr-ui/ui/masks.slint#L490) | | FR-UI-4 | [`ui/dr-ui/src/collections_ui.rs:1005`](../ui/dr-ui/src/collections_ui.rs#L1005), [`ui/dr-ui/src/collections_ui.rs:118`](../ui/dr-ui/src/collections_ui.rs#L118), [`ui/dr-ui/src/collections_ui.rs:131`](../ui/dr-ui/src/collections_ui.rs#L131), [`ui/dr-ui/src/collections_ui.rs:1508`](../ui/dr-ui/src/collections_ui.rs#L1508), [`ui/dr-ui/src/collections_ui.rs:1522`](../ui/dr-ui/src/collections_ui.rs#L1522), [`ui/dr-ui/src/collections_ui.rs:1568`](../ui/dr-ui/src/collections_ui.rs#L1568), [`ui/dr-ui/src/collections_ui.rs:159`](../ui/dr-ui/src/collections_ui.rs#L159), [`ui/dr-ui/src/collections_ui.rs:1666`](../ui/dr-ui/src/collections_ui.rs#L1666), [`ui/dr-ui/src/collections_ui.rs:1693`](../ui/dr-ui/src/collections_ui.rs#L1693), [`ui/dr-ui/src/collections_ui.rs:485`](../ui/dr-ui/src/collections_ui.rs#L485), [`ui/dr-ui/src/collections_ui.rs:514`](../ui/dr-ui/src/collections_ui.rs#L514), [`ui/dr-ui/src/collections_ui.rs:582`](../ui/dr-ui/src/collections_ui.rs#L582), [`ui/dr-ui/src/collections_ui.rs:995`](../ui/dr-ui/src/collections_ui.rs#L995), [`ui/dr-ui/src/library_ui.rs:4388`](../ui/dr-ui/src/library_ui.rs#L4388), [`ui/dr-ui/src/library_ui.rs:4433`](../ui/dr-ui/src/library_ui.rs#L4433), [`ui/dr-ui/src/library_ui.rs:4536`](../ui/dr-ui/src/library_ui.rs#L4536), [`ui/dr-ui/src/library_ui.rs:4564`](../ui/dr-ui/src/library_ui.rs#L4564), [`ui/dr-ui/src/library_ui.rs:5337`](../ui/dr-ui/src/library_ui.rs#L5337), [`ui/dr-ui/src/library_ui.rs:5349`](../ui/dr-ui/src/library_ui.rs#L5349), [`ui/dr-ui/ui/app.slint:1603`](../ui/dr-ui/ui/app.slint#L1603), [`ui/dr-ui/ui/app.slint:437`](../ui/dr-ui/ui/app.slint#L437), [`ui/dr-ui/ui/app.slint:468`](../ui/dr-ui/ui/app.slint#L468), [`ui/dr-ui/ui/library.slint:1202`](../ui/dr-ui/ui/library.slint#L1202), [`ui/dr-ui/ui/library.slint:1209`](../ui/dr-ui/ui/library.slint#L1209), [`ui/dr-ui/ui/library.slint:1215`](../ui/dr-ui/ui/library.slint#L1215), [`ui/dr-ui/ui/library.slint:2314`](../ui/dr-ui/ui/library.slint#L2314), [`ui/dr-ui/ui/library.slint:829`](../ui/dr-ui/ui/library.slint#L829), [`ui/dr-ui/ui/library.slint:879`](../ui/dr-ui/ui/library.slint#L879), [`ui/dr-ui/ui/library.slint:898`](../ui/dr-ui/ui/library.slint#L898) | diff --git a/ui/dr-ui/src/identity.rs b/ui/dr-ui/src/identity.rs index 90cf3a3..1d21b98 100644 --- a/ui/dr-ui/src/identity.rs +++ b/ui/dr-ui/src/identity.rs @@ -88,25 +88,29 @@ pub struct FaceCell { /// The user asserted this, as against the system guessing it. pub confirmed: bool, /// Calibrated P(this face is this person). + /// + /// Always meaningful, and always shown. Where the library has no fit of + /// its own the number comes from the reference curve — a published + /// operating point, not an invention — and the screen says so once, at the + /// top, rather than blanking every face (FR-CULL-9). pub probability: f32, - /// Whether that probability means anything — false when the library's - /// calibration has not been fitted (FR-CULL-9). - pub probability_known: bool, /// Source pixels across the aligned crop. Small faces embed worse, and the /// user deserves to know which of a bad suggestion's causes is in play. pub crop_px: f32, } impl FaceCell { - /// Confidence as text, or why there is none. + /// Confidence as text. /// - /// The FR-CULL-9 rule made concrete: where the calibration is not fitted, - /// this says so rather than printing an untuned number that looks measured. + /// FR-CULL-9's rule is about *provenance*, not about silence: the number + /// must not be passed off as measured when it is not. Withholding it + /// entirely was the wrong reading — it left the user ranking forty + /// suggestions with nothing to rank them by, on every library that has not + /// yet earned a fit, which is most of them. The number is shown; where the + /// curve is the built-in one the screen says so above the grid. pub fn confidence_label(&self) -> String { if self.confirmed { "Confirmed".into() - } else if !self.probability_known { - "Confidence unavailable".into() } else { format!("{:.0}% likely", self.probability * 100.0) } @@ -122,7 +126,9 @@ pub struct IdentityView { /// Faces belonging to nobody — the pool the user can pull a new person out /// of, and the honest answer to "why is this photo not under anyone". pub unassigned: usize, - /// Whether the library's similarity calibration has been fitted. + /// Whether the library's similarity calibration has been fitted from its + /// own faces, as against the built-in reference curve. Not a gate on + /// showing confidences — only on how the screen describes them. pub calibrated: bool, } @@ -166,7 +172,6 @@ pub fn load_faces( catalog: &Catalog, store: &ThumbStore, person: PersonId, - calibrated: bool, ) -> Result, dr_catalog::CatalogError> { let conn = catalog.connection(); let rows = faces::for_person(conn, person, true)?; @@ -203,7 +208,6 @@ pub fn load_faces( crop, confirmed: f.confirmed, probability: f.probability, - probability_known: calibrated, crop_px: f.crop_px, }); } @@ -557,9 +561,8 @@ pub fn preview_split( let cal = faces::calibration(conn, model_id)? .map(|(c, _)| c) .unwrap_or_default(); - let calibrated = cal.valid; - let cells = load_faces(catalog, store, person, calibrated)?; + let cells = load_faces(catalog, store, person)?; if cells.len() < 2 { return Ok(vec![cells]); } @@ -715,30 +718,25 @@ mod tests { assert!(!row.is_unconfirmed()); } - /// FR-CULL-9's rule at the point it becomes visible: an unfitted - /// calibration must not print a number that looks measured. + /// FR-CULL-9 at the point it becomes visible. The rule is that an unfitted + /// curve must not be *described* as measured — not that the number is + /// withheld, which left the user with forty unrankable suggestions on every + /// library too young to have earned a fit. The percentage is always shown; + /// the provenance is said once, at the screen level. #[test] - fn an_uncalibrated_library_says_so_instead_of_showing_a_percentage() { + fn every_suggestion_shows_a_percentage_whatever_the_curve_was_fitted_from() { let cell = FaceCell { face: FaceId(1), image: ImageId(1), crop: None, confirmed: false, probability: 0.87, - probability_known: false, crop_px: 120.0, }; - assert_eq!(cell.confidence_label(), "Confidence unavailable"); - - let calibrated = FaceCell { - probability_known: true, - ..cell.clone() - }; - assert_eq!(calibrated.confidence_label(), "87% likely"); + assert_eq!(cell.confidence_label(), "87% likely"); let confirmed = FaceCell { confirmed: true, - probability_known: false, ..cell }; assert_eq!(confirmed.confidence_label(), "Confirmed"); @@ -762,7 +760,7 @@ mod tests { assert_eq!(view.unassigned, 1); assert!( !view.calibrated, - "a fresh library has no fitted calibration" + "a fresh library has no fit of its own, and says so — it still shows confidences" ); } diff --git a/ui/dr-ui/src/identity_ui.rs b/ui/dr-ui/src/identity_ui.rs index 87f503c..19fc5a6 100644 --- a/ui/dr-ui/src/identity_ui.rs +++ b/ui/dr-ui/src/identity_ui.rs @@ -189,11 +189,10 @@ pub fn refresh( match (ctl.selected.get(), store) { (Some(person), Some(store)) => { - let cells = - identity::load_faces(cat, store, person, view.calibrated).unwrap_or_else(|e| { - log::warn!("identity: reading faces: {e}"); - Vec::new() - }); + let cells = identity::load_faces(cat, store, person).unwrap_or_else(|e| { + log::warn!("identity: reading faces: {e}"); + Vec::new() + }); push_faces(window, ctl, &cells); *ctl.faces.borrow_mut() = cells; diff --git a/ui/dr-ui/ui/identity.slint b/ui/dr-ui/ui/identity.slint index 3bbbb60..1db0d13 100644 --- a/ui/dr-ui/ui/identity.slint +++ b/ui/dr-ui/ui/identity.slint @@ -42,10 +42,9 @@ export struct IdentityFace { crop: image, has-crop: bool, confirmed: bool, - // "Confirmed", "83% likely", or "Confidence unavailable" — composed in - // Rust, because FR-CULL-9's rule about not showing an untuned number as - // though it were measured is a rule about *content*, and it should not be - // re-derived from a float in a second place. + // "Confirmed" or "83% likely" — composed in Rust, because FR-CULL-9's rule + // about how a confidence is described is a rule about *content*, and it + // should not be re-derived from a float in a second place. confidence: string, // Source pixels across the aligned crop. A suggestion the user disagrees // with has causes, and "the face was 41 pixels across" is one the user can @@ -177,7 +176,9 @@ export component IdentityScreen inherits Rectangle { // Faces belonging to nobody. Shown as a count rather than hidden, because // "why is this photograph not under anyone" deserves an answer. in property unassigned: 0; - // Whether the library's similarity calibration is fitted (FR-CULL-9). + // Whether the similarity curve was fitted from this library's own faces, + // as against the built-in one (FR-CULL-9). Changes what the screen says + // about the confidences, never whether they are shown. in property calibrated: false; in property indexing: false; in property indexing-status; @@ -591,11 +592,12 @@ export component IdentityScreen inherits Rectangle { } } - // FR-CULL-9, made visible: where the calibration is not fitted the - // screen says the confidences are unavailable rather than letting - // per-face percentages imply a measurement that was never made. + // FR-CULL-9, made visible: the percentages are real numbers off a + // real curve, but on a young library that curve is the built-in one + // rather than one measured here. Said once, above the grid, so no + // per-face percentage implies a measurement that was never made. if !root.calibrated && !root.model-missing: Text { - text: "Similarity is not calibrated for this library yet, so no confidence is shown."; + text: "Confidences use the built-in similarity curve; they will sharpen once this library has enough confirmed faces to fit its own."; color: Theme.ink-faint; font-size: Theme.text-sm; wrap: word-wrap; From ebb7d3cf5ce00a11b408e33c9993d2aafcfcbb4a Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 09:57:56 +0200 Subject: [PATCH 2/9] Score a suggestion against the people the user has named MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The number beside a suggestion was the mean calibrated probability between the face and the rest of its group, which measures the wrong thing twice. It punishes coverage: a person with two hundred faces over fifteen years is *meant* to have members a given photograph is orthogonal to, so a correct suggestion onto a well-photographed person scored low for being well photographed. And it never asked who else the face might be — a face matching Anna at 0.95 and nobody else, and one matching Anna at 0.95 and her sister at 0.93, came out identical, when the second is the only one worth the user's attention. dr_face::assign answers both, and multiplies them: the mean of the best ten calibrated matches into the identity (the old mean, capped, which is what stops coverage counting against it), times that identity's share of the evidence against every *named* rival. Only named people compete, and per person rather than per group. Both halves of that had to be measured on a real 18,000-face library rather than reasoned about. Normalising across every group made the number useless — median suggestion 21%, four in five under half — because clustering leaves one person spread over many groups, so a face competed against itself; and keying rivals by group left Catherine competing with Catherine, median 39%. Per named person: median 99.5%. Rivals are gathered below the merge threshold, down to even odds: a named person matching at 0.6 will never be merged into but is exactly the competition to discount for. That would be a second similarity scan, the expensive half of regrouping a library, so cluster_scored scans once at the looser floor and hands the merge engine the subset at or above the threshold — pair for pair what it would have scanned for itself, held to that by a test. Leave-one-out over that library's 2,702 confirmations across 54 named people: 99.33% of faces placed on the right person against the old mean's 99.15%, and the number shown for the right person moves from a median of 90.4% to 99.3%. It errs low — 100% correct wherever it states 80% or more — which is the safe direction, and docs/faces.md §9.1 says plainly that the low bands are not calibrated. The example that measures it comes too: this is a claim about a library's numbers, and nobody should have to take it on faith. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-catalog/examples/face_confidence.rs | 373 ++++++++++++++++++++ core/dr-face/src/assign.rs | 360 +++++++++++++++++++ core/dr-face/src/cluster.rs | 172 ++++++++- core/dr-face/src/lib.rs | 11 +- docs/faces.md | 59 ++++ docs/traceability.md | 2 +- ui/dr-ui/src/faces.rs | 46 +-- 7 files changed, 971 insertions(+), 52 deletions(-) create mode 100644 core/dr-catalog/examples/face_confidence.rs create mode 100644 core/dr-face/src/assign.rs diff --git a/core/dr-catalog/examples/face_confidence.rs b/core/dr-catalog/examples/face_confidence.rs new file mode 100644 index 0000000..e2d633c --- /dev/null +++ b/core/dr-catalog/examples/face_confidence.rs @@ -0,0 +1,373 @@ +//! What the suggestion confidence would say about a real library. +//! +//! cargo run --release -p dr-catalog --example face_confidence -- CATALOG.sqlite [--full] +//! +//! Read-only: it writes nothing to the catalog, so it can be pointed at a copy +//! of a live library and re-run at will. +//! +//! # What it measures +//! +//! The user's own confirmations are the only ground truth a library has, so +//! the evaluation is leave-one-out over them: hide one confirmed face, ask the +//! scorer which of the confirmed identities it belongs to, and compare with +//! what the user said. Faces from the same photograph are excluded exactly as +//! the clusterer excludes them, so nothing is scored against a co-occurrence +//! that would never have been allowed to merge. +//! +//! Two numbers are compared on that task: the share (`dr_face::assign`) and the +//! mean-within-group figure it replaced. Accuracy says which one picks the +//! right person; the reliability table says whether the percentage the user is +//! shown means what it claims — which is the question FR-CULL-9 exists for. + +use std::collections::HashMap; + +use dr_catalog::faces::{self, PersonId}; +use dr_catalog::Catalog; + +const MODEL_ID: &str = "w600k_mbf"; +const TOP: usize = 10; + +struct Known { + image: u64, + person: PersonId, + embedding: Vec, + crop_px: f32, +} + +fn main() { + let args: Vec = std::env::args().skip(1).collect(); + let Some(path) = args.first() else { + eprintln!("usage: face_confidence CATALOG.sqlite [--full]"); + std::process::exit(2); + }; + + let catalog = Catalog::open(std::path::Path::new(path)).expect("open catalog"); + let conn = catalog.connection(); + + let cal = match faces::calibration(conn, MODEL_ID) { + Ok(Some((c, _))) => c, + _ => dr_face::Calibration::default(), + }; + println!( + "calibration: a={:.2} b={:.2} w_size={:.3} valid={} (P=0.5 at cosine {:.3})", + cal.a, + cal.b, + cal.w_size, + cal.valid, + cal.boundary_at(0.5, 150.0, 0.0) + ); + + let model = dr_face::ModelId::new(MODEL_ID.to_string()); + let stored = faces::embeddings(conn, MODEL_ID).expect("embeddings"); + let mut embedding_of = HashMap::new(); + for (id, image, blob, crop_px) in stored { + if let Some(e) = dr_face::Embedding::from_f16_bytes(model.clone(), &blob) { + embedding_of.insert(id, (image.0, e.v.to_vec(), crop_px)); + } + } + println!("faces with embeddings: {}", embedding_of.len()); + + // The ground truth: every confirmed face, under the person the user put it + // on. Identities with a single confirmation are dropped — leaving one out + // leaves that identity with no evidence at all, so they measure nothing. + let people = faces::people(conn).expect("people"); + let mut known: Vec = Vec::new(); + let mut identities = 0usize; + for p in &people { + if p.confirmed_faces < 2 { + continue; + } + let mut mine = Vec::new(); + for f in faces::for_person(conn, p.id, false).expect("faces") { + if !f.confirmed { + continue; + } + if let Some((image, embedding, crop_px)) = embedding_of.get(&f.id) { + mine.push(Known { + image: *image, + person: p.id, + embedding: embedding.clone(), + crop_px: *crop_px, + }); + } + } + if mine.len() >= 2 { + identities += 1; + known.extend(mine); + } + } + println!( + "ground truth: {} confirmed faces across {identities} identities\n", + known.len() + ); + if known.len() < 2 { + println!("not enough confirmations to evaluate."); + return; + } + + let mut share_right = 0usize; + let mut mean_right = 0usize; + // (share of the winner, was the winner correct) + let mut reliability: Vec<(f32, bool)> = Vec::with_capacity(known.len()); + // What each scorer would have *displayed* for the correct answer. + let mut shown_share = Vec::with_capacity(known.len()); + let mut shown_mean = Vec::with_capacity(known.len()); + + for (i, me) in known.iter().enumerate() { + let mut per_person: HashMap> = HashMap::new(); + // The mean baseline is the old code's, which had no floor: it averaged + // over every member of the group. + let mut all_person: HashMap> = HashMap::new(); + for (j, them) in known.iter().enumerate() { + if i == j || me.image == them.image { + continue; + } + let cos: f32 = me + .embedding + .iter() + .zip(&them.embedding) + .map(|(a, b)| a * b) + .sum(); + // The floor the real scorer sees: `cluster_scored` scans at + // `RIVAL_FLOOR` and `identity_shares` never learns about a pair + // below it. Summing the near-orthogonal ones here instead of + // dropping them is not a stricter test, it is a different + // function — fifty identities contributing their *upper tail* of + // noise outweigh one contributing a real match. + let probability = cal.probability(cos, me.crop_px.min(them.crop_px), 0.0); + if probability >= dr_face::RIVAL_FLOOR { + per_person.entry(them.person).or_default().push(probability); + } + all_person.entry(them.person).or_default().push(probability); + } + + // The share: sum of the best TOP matches per identity, normalised. + // Evidence, and the coherence that goes with it: the sum of the best + // TOP matches, and their mean. dr_face::assign shows the product of + // that mean and the identity's share of the total. + let mut evidence: Vec<(PersonId, f32, f32)> = per_person + .iter() + .map(|(&p, probabilities)| { + let mut v = probabilities.clone(); + v.sort_by(|a, b| b.total_cmp(a)); + let counted = v.len().min(TOP); + let sum = v.iter().take(TOP).sum::(); + (p, sum, sum / counted as f32) + }) + .collect(); + let total: f32 = evidence.iter().map(|(_, s, _)| *s).sum(); + evidence.sort_by(|a, b| b.1.total_cmp(&a.1)); + + // The number it replaced: the mean over every member of the identity. + let mut means: Vec<(PersonId, f32)> = all_person + .iter() + .map(|(&p, probabilities)| { + ( + p, + probabilities.iter().sum::() / probabilities.len() as f32, + ) + }) + .collect(); + means.sort_by(|a, b| b.1.total_cmp(&a.1)); + + if let (Some(&(winner, score, coherence)), true) = (evidence.first(), total > 0.0) { + let correct = winner == me.person; + share_right += correct as usize; + // What the screen would say about the identity it picked. + reliability.push((coherence * score / total, correct)); + let ours = evidence + .iter() + .find(|(p, _, _)| *p == me.person) + .map(|(_, s, c)| c * s / total) + .unwrap_or(0.0); + shown_share.push(ours); + } + if let Some(&(winner, _)) = means.first() { + mean_right += (winner == me.person) as usize; + shown_mean.push( + means + .iter() + .find(|(p, _)| *p == me.person) + .map(|(_, s)| *s) + .unwrap_or(0.0), + ); + } + } + + let n = known.len() as f64; + println!("which identity does this face belong to? (leave-one-out, top-1)"); + println!( + " share of evidence {:>6.2}% ({share_right}/{})", + 100.0 * share_right as f64 / n, + known.len() + ); + println!( + " mean within group {:>6.2}% ({mean_right}/{})\n", + 100.0 * mean_right as f64 / n, + known.len() + ); + + println!("what the screen would show for the answer the user gave:"); + band(" share ", &shown_share); + band(" mean ", &shown_mean); + + println!("\nreliability of the share — is a stated {{n}}% right {{n}}% of the time?"); + println!( + " {:>12} {:>7} {:>9} {:>8}", + "stated", "faces", "correct", "gap" + ); + for (lo, hi) in [ + (0.0, 0.5), + (0.5, 0.6), + (0.6, 0.7), + (0.7, 0.8), + (0.8, 0.9), + (0.9, 0.95), + (0.95, 1.001), + ] { + let bucket: Vec = reliability + .iter() + .filter(|(s, _)| *s >= lo && *s < hi) + .map(|(_, c)| *c) + .collect(); + if bucket.is_empty() { + continue; + } + let observed = bucket.iter().filter(|c| **c).count() as f64 / bucket.len() as f64; + let stated = reliability + .iter() + .filter(|(s, _)| *s >= lo && *s < hi) + .map(|(s, _)| *s as f64) + .sum::() + / bucket.len() as f64; + println!( + " {:>5.0}–{:>3.0}% {:>9} {:>8.1}% {:>+7.1}", + lo * 100.0, + hi.min(1.0) * 100.0, + bucket.len(), + 100.0 * observed, + 100.0 * (observed - stated) + ); + } + + if args.iter().any(|a| a == "--full") { + // The confirmations go in as anchors, exactly as `recluster` sends + // them: they are what makes a group a named identity, and therefore + // what makes it a rival. + let mut confirmed = HashMap::new(); + for p in &people { + for f in faces::for_person(conn, p.id, false).expect("faces") { + if f.confirmed { + confirmed.insert(f.id, p.id.0); + } + } + } + full_library(&embedding_of, &confirmed, &cal); + } +} + +/// Where a set of confidences actually falls. +fn band(label: &str, v: &[f32]) { + if v.is_empty() { + return; + } + let mut s = v.to_vec(); + s.sort_by(|a, b| a.total_cmp(b)); + let pct = |q: f64| s[((s.len() - 1) as f64 * q) as usize]; + let mean = s.iter().sum::() / s.len() as f32; + println!( + "{label} median {:>5.1}% mean {:>5.1}% p10 {:>5.1}% p90 {:>5.1}% under 50%: {:>5.1}%", + 100.0 * pct(0.5), + 100.0 * mean, + 100.0 * pct(0.10), + 100.0 * pct(0.90), + 100.0 * s.iter().filter(|x| **x < 0.5).count() as f32 / s.len() as f32 + ); +} + +/// The whole library through the real clusterer, for the numbers it would +/// actually write. +fn full_library( + embedding_of: &HashMap, f32)>, + confirmed: &HashMap, + cal: &dr_face::Calibration, +) { + let mut candidates: Vec = embedding_of + .iter() + .map(|(id, (image, embedding, crop_px))| dr_face::Candidate { + face: id.0, + image: *image, + embedding: embedding.clone(), + crop_px: *crop_px, + confirmed_person: confirmed.get(id).copied(), + }) + .collect(); + candidates.sort_by_key(|c| c.face); + + println!("\nthe whole library, at the default merge probability:"); + let start = std::time::Instant::now(); + let grouping = dr_face::cluster_scored(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY); + let real: Vec<_> = grouping + .clusters + .iter() + .filter(|c| c.members.len() >= 2) + .collect(); + let grouped: usize = real.iter().map(|c| c.members.len()).sum(); + println!( + " {} face(s) → {} group(s) of two or more, holding {grouped} faces ({:.0}%), in {:.1}s", + candidates.len(), + real.len(), + 100.0 * grouped as f64 / candidates.len() as f64, + start.elapsed().as_secs_f64() + ); + + let named: Vec<_> = real.iter().filter(|c| c.person.is_some()).collect(); + println!( + " {} of those group(s) carry a confirmation, holding {} faces", + named.len(), + named.iter().map(|c| c.members.len()).sum::() + ); + + let shown: Vec = real + .iter() + .flat_map(|c| c.members.iter().map(|&m| grouping.confidence[m])) + .collect(); + band(" new, all groups ", &shown); + let onto_people: Vec = named + .iter() + .flat_map(|c| c.members.iter().map(|&m| grouping.confidence[m])) + .collect(); + band(" new, onto a person", &onto_people); + + // The number the old code would have written for the same grouping. + let means: Vec = real + .iter() + .flat_map(|c| { + c.members.iter().map(|&m| { + let me = &candidates[m]; + let mut sum = 0.0; + let mut n = 0.0; + for &other in &c.members { + if other == m { + continue; + } + let them = &candidates[other]; + let cos: f32 = me + .embedding + .iter() + .zip(&them.embedding) + .map(|(a, b)| a * b) + .sum(); + sum += cal.probability(cos, me.crop_px.min(them.crop_px), 0.0); + n += 1.0; + } + if n == 0.0 { + 1.0 + } else { + sum / n + } + }) + }) + .collect(); + band(" old, all groups ", &means); +} diff --git a/core/dr-face/src/assign.rs b/core/dr-face/src/assign.rs new file mode 100644 index 0000000..66a7236 --- /dev/null +++ b/core/dr-face/src/assign.rs @@ -0,0 +1,360 @@ +//! TRACES: FR-CULL-9 | FR-CULL-10 +//! How sure a *suggestion* is: an identity's share of the evidence for a face. +//! +//! [`crate::cluster`] decides which people exist; this decides what number to +//! put beside "we think this is Anna". They are not the same question, and the +//! answer to the second used to be a by-product of the first — the mean +//! calibrated probability between a face and *every* other member of its group. +//! +//! # Why a mean over the group is the wrong number +//! +//! It measures the wrong thing twice over. +//! +//! **It punishes large, well-photographed people.** Anna has two hundred faces +//! spanning fifteen years; a new photograph of her matches thirty of them +//! strongly and is near-orthogonal to the rest, because a face at 8 and a face +//! at 23 genuinely are. The mean lands around 0.2 and the interface reports a +//! correct suggestion as a doubtful one. The better a person is covered, the +//! worse their confidences get, which is exactly backwards. +//! +//! **It never asks who else it could be.** A face that matches Anna at 0.95 and +//! matches nobody else at all, and a face that matches Anna at 0.95 *and her +//! sister at 0.93*, are the same number under a within-group mean. The second +//! is the one the user actually needs to look at, and it was indistinguishable +//! from the first. +//! +//! # Coherence, times uniqueness +//! +//! Two questions, and the number is their product because they are genuinely +//! independent: *is this the same person at all*, and *of the people we know, +//! is it uniquely this one*. +//! +//! ```text +//! evidence(P) = Σ of the top n of { P(same | this face, f) : f ∈ P } +//! coherence = evidence(own) / (however many of the top n there were) +//! uniqueness = evidence(own) / (evidence(own) + Σ evidence(named rivals)) +//! confidence = coherence × uniqueness +//! ``` +//! +//! **Coherence** is a mean, like the old number, but over the face's best +//! [`TOP_MATCHES`] matches into the identity rather than over all of them. That single cap is what +//! stops a well-photographed person scoring worse than a thin one: the two +//! hundred faces a given photograph is legitimately orthogonal to no longer +//! count against it. +//! +//! **Uniqueness** is the competition. A sole strong match leaves it at 1 and +//! the confidence is the coherence; two identities matching equally well pull +//! it to 0.5 each, and the screen has told the user the truth, which is that +//! this face is a coin toss between two people. +//! +//! # Only the people the user has named compete +//! +//! Measured on a real 18,000-face library, normalising across *every* group +//! made the number useless: the median suggestion read 21% and four in five +//! read under half. The cause is not a bug in the arithmetic but a fact about +//! clustering — one person is spread across many groups, since the pairs that +//! would have joined them are the ones that fell short of the merge threshold. +//! Normalising over groups therefore makes a face compete against *itself*, +//! and the better covered the person, the more fragments there are to lose to. +//! +//! A fragment is not a rival. An identity the user has actually asserted is, so +//! the denominator counts only people a confirmation names ([`Cluster::person`]) +//! — and counts them **per person, not per group**, since a named person is +//! left in several anchored groups for the same reason. Keying it by group had +//! Catherine competing with Catherine and put the median suggestion onto a +//! named person at 39%; keying it by person put it at 99.5%. +//! +//! Leave-one-out over that library's 2,702 confirmations across 54 named +//! people, this is the regime where the number is worth having: 99.3% of faces +//! are placed on the right person (the old mean managed 99.15%), the stated +//! percentage is monotone in being right, and it errs low — 100% correct +//! wherever it states 80% or more, 84% correct where it states under half. +//! Understating is the safe direction for a screen whose whole purpose is +//! deciding what to look at first, but it is *not* calibrated in the low bands +//! and should not be read as though it were. +//! +//! Two faces of one unnamed group cannot be told from two fragments of one +//! person by similarity alone; that is exactly why clustering stopped where it +//! did. So the module does not pretend to: where nobody is named, uniqueness is +//! 1 and the number falls back to plain coherence. +//! +//! # What it does not do +//! +//! It is not a merge threshold and must not become one. Clustering keeps +//! deciding on the pairwise calibrated probability: uniqueness is *relative*, +//! so a library with one named person in it would hand every stray face a +//! uniqueness of 1. The absolute question ("is this the same person at all") +//! and the comparative one ("of the people we know, which") are different, and +//! the product is what keeps both in the answer. + +use std::collections::BTreeMap; + +use crate::cluster::Cluster; +use crate::neighbours::Pair; + +/// Who the evidence is for. +/// +/// Ordered rather than hashed so the sums below are reproducible; the ordering +/// itself carries no meaning. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +enum Identity { + /// A person the user has confirmed a face onto. Every group anchored to + /// them is the same identity, however many of them the clusterer left. + Person(u64), + /// A group nobody has ruled on. It stands for itself and competes with + /// nothing. + Group(usize), +} + +/// How many of an identity's best matches count as its evidence. +/// +/// The cap is the whole reason the sum works: uncapped, evidence would grow +/// with a person's face count and the largest group in the library would win +/// every contest. Ten is enough that a person photographed from several angles +/// contributes more than one lucky frame, and small enough that the hundred +/// mediocre matches inside a well-covered identity cannot add up to a strong +/// one. It is a starting point, not a measured optimum — M7's corpus is where +/// it would be tuned. +pub const TOP_MATCHES: usize = 10; + +/// The weakest match that counts as evidence for an identity. +/// +/// Rivals are half the point of this module, so the evidence scan has to reach +/// *below* the merge threshold — a named person who matches at 0.6 will never +/// be merged into but is precisely the competition a suggestion should be +/// discounted for. Even odds is the natural floor: below it a pair is more +/// likely different people than the same, and it is not a small distinction — +/// summing the near-orthogonal pairs instead of dropping them lets fifty +/// identities' worth of noise, each contributing its *upper tail*, outweigh one +/// real match. Measured on a real library, that alone moved the median stated +/// confidence from 100% to 31%. +pub const RIVAL_FLOOR: f32 = 0.5; + +/// How confident each face's placement is, indexed like the face slice the +/// clusters came from. +/// +/// A face in no group, or one with no evidence for anybody, scores 0. +/// +/// `pairs` must be the *evidence* list — scanned at [`RIVAL_FLOOR`], not at the +/// merge threshold. Passing the merge list still works but silently removes +/// every rival weaker than a merge, which is most of them, and every uniqueness +/// collapses to 1. +pub fn identity_shares(faces: usize, clusters: &[Cluster], pairs: &[Pair], top: usize) -> Vec { + // An identity is a *person*, not a group. One person routinely holds + // several anchored groups — the same reason they hold several unnamed ones + // — and keying this by group had Catherine competing with Catherine, which + // on the library it was measured against put the median suggestion onto a + // named person at 39%. + let key_of: Vec = clusters + .iter() + .enumerate() + .map(|(g, c)| match c.person { + Some(p) => Identity::Person(p), + None => Identity::Group(g), + }) + .collect(); + + let mut group_of = vec![usize::MAX; faces]; + for (g, c) in clusters.iter().enumerate() { + for &m in &c.members { + if m < faces { + group_of[m] = g; + } + } + } + + // Ordered, not hashed: the numbers are sums of floats over these buckets + // and this module inherits [`crate::cluster`]'s promise that the same input + // yields the same output, bit for bit. + let mut evidence: Vec>> = vec![BTreeMap::new(); faces]; + for p in pairs { + if p.i >= faces || p.j >= faces { + continue; + } + // A pair is evidence in both directions: j's identity hears about i, + // and i's identity hears about j. The pair list holds each unordered + // pair once, so both have to be recorded here. + let (gi, gj) = (group_of[p.i], group_of[p.j]); + if gj != usize::MAX { + evidence[p.i] + .entry(key_of[gj]) + .or_default() + .push(p.probability); + } + if gi != usize::MAX { + evidence[p.j] + .entry(key_of[gi]) + .or_default() + .push(p.probability); + } + } + + let mut out = vec![0.0; faces]; + for (i, buckets) in evidence.iter_mut().enumerate() { + let mine = group_of[i]; + if mine == usize::MAX { + continue; + } + let mine = key_of[mine]; + let mut coherence = 0.0; + let mut ours = 0.0; + let mut rivals = 0.0; + for (&who, probabilities) in buckets.iter_mut() { + // Descending, and the ties broken by nothing: equal probabilities + // sum the same whichever order they land in. + probabilities.sort_by(|a, b| b.total_cmp(a)); + let counted = probabilities.len().min(top); + let score: f32 = probabilities.iter().take(top).sum(); + if who == mine { + ours = score; + coherence = score / counted as f32; + } else if matches!(who, Identity::Person(_)) { + // Only an identity the user has asserted competes. An unnamed + // group that matches this face is far more likely to be another + // fragment of the same person than a different one — see the + // module note, and the library it was measured on. + rivals += score; + } + } + let total = ours + rivals; + // No evidence at all: a face anchored into a group it has no measured + // similarity to. Nothing honest to report, so nothing is claimed. + out[i] = if total > 0.0 { + coherence * (ours / total) + } else { + 0.0 + }; + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + /// An unnamed group: nobody has ruled on it, so it competes with nothing. + fn cluster(members: &[usize]) -> Cluster { + Cluster { + members: members.to_vec(), + person: None, + } + } + + /// A group the user has confirmed a face onto — an identity, and therefore + /// a rival. + fn named(members: &[usize], person: u64) -> Cluster { + Cluster { + members: members.to_vec(), + person: Some(person), + } + } + + fn pair(i: usize, j: usize, probability: f32) -> Pair { + Pair { i, j, probability } + } + + /// The failure the module exists to fix: face 0 matches its own group's + /// three members strongly, and the group has forty more it is unrelated to. + /// The old within-group mean reported ~0.07 for this. + #[test] + fn a_large_group_does_not_dilute_a_strong_match() { + let members: Vec = (0..44).collect(); + let clusters = vec![cluster(&members)]; + let pairs = vec![pair(0, 1, 0.99), pair(0, 2, 0.97), pair(0, 3, 0.95)]; + + let shares = identity_shares(44, &clusters, &pairs, TOP_MATCHES); + assert!( + (shares[0] - 0.97).abs() < 1e-6, + "the mean of its three real matches, undiluted: {}", + shares[0] + ); + } + + /// Two named people matching equally well is a coin toss, and saying so is + /// the point — this is the sibling case FR-CULL-10 warns about. + #[test] + fn an_ambiguous_face_splits_its_confidence_between_the_rivals() { + let clusters = vec![named(&[0, 1, 2], 1), named(&[3, 4], 2)]; + let pairs = vec![ + pair(0, 1, 0.90), + pair(0, 2, 0.90), + pair(0, 3, 0.90), + pair(0, 4, 0.90), + ]; + + let shares = identity_shares(5, &clusters, &pairs, TOP_MATCHES); + // Coherent at 0.90, and only half of the evidence is its own. + assert!( + (shares[0] - 0.45).abs() < 1e-6, + "even evidence both ways: {}", + shares[0] + ); + } + + /// A rival below the merge threshold still has to count, which is why the + /// evidence scan reaches down to [`RIVAL_FLOOR`]. + #[test] + fn a_rival_too_weak_to_merge_still_lowers_the_confidence() { + let clusters = vec![named(&[0, 1], 1), named(&[2, 3], 2)]; + let sure = identity_shares(4, &clusters, &[pair(0, 1, 0.95)], TOP_MATCHES); + let contested = identity_shares( + 4, + &clusters, + &[pair(0, 1, 0.95), pair(0, 2, 0.60)], + TOP_MATCHES, + ); + + assert_eq!(sure[0], 0.95, "nobody else to be: its coherence stands"); + assert!( + contested[0] < 0.59 && contested[0] > 0.57, + "0.95 coherent, but 0.95 against 0.60: {}", + contested[0] + ); + } + + /// A fragment of the same person is not a rival. Measured on a real + /// library, counting unnamed groups as competition put four suggestions in + /// five under half — see the module note. + #[test] + fn an_unnamed_group_is_not_treated_as_competition() { + let clusters = vec![cluster(&[0, 1]), cluster(&[2, 3])]; + let shares = identity_shares( + 4, + &clusters, + &[pair(0, 1, 0.95), pair(0, 2, 0.90)], + TOP_MATCHES, + ); + assert_eq!( + shares[0], 0.95, + "an unnamed group took evidence off a suggestion" + ); + } + + /// The cap, doing its job: an identity with fifty mediocre matches must not + /// beat one with ten strong ones on volume alone. + #[test] + fn evidence_is_capped_so_the_biggest_group_cannot_win_on_volume() { + let small: Vec = (0..11).collect(); + let large: Vec = (11..62).collect(); + let clusters = vec![named(&small, 1), named(&large, 2)]; + + let mut pairs: Vec = (1..11).map(|j| pair(0, j, 0.90)).collect(); + pairs.extend((11..62).map(|j| pair(0, j, 0.55))); + + let shares = identity_shares(62, &clusters, &pairs, TOP_MATCHES); + // Ten at 0.90 against ten at 0.55 — not fifty-one at 0.55. + assert!( + (shares[0] - 0.90 * (9.0 / 14.5)).abs() < 1e-5, + "capped at ten either side: {}", + shares[0] + ); + } + + /// A face nothing has any evidence about claims nothing. + #[test] + fn a_face_with_no_evidence_reports_no_confidence() { + let clusters = vec![cluster(&[0, 1])]; + let shares = identity_shares(2, &clusters, &[], TOP_MATCHES); + assert_eq!(shares, vec![0.0, 0.0]); + } +} diff --git a/core/dr-face/src/cluster.rs b/core/dr-face/src/cluster.rs index 6fe94be..be1315a 100644 --- a/core/dr-face/src/cluster.rs +++ b/core/dr-face/src/cluster.rs @@ -148,22 +148,110 @@ pub fn cluster(faces: &[Candidate], cal: &Calibration, min_probability: f32) -> return Vec::new(); } - let embeddings: Vec> = faces.iter().map(|f| f.embedding.clone()).collect(); - let crop_px: Vec = faces.iter().map(|f| f.crop_px).collect(); - let images: Vec = faces.iter().map(|f| f.image).collect(); - let view = Faces { - embeddings: &embeddings, - crop_px: &crop_px, - images: &images, - }; - + let columns = Columns::of(faces); // Every pair that could ever contribute to a merge. See the module note on // why nothing outside this list can matter. - let pairs = neighbours::above_threshold(&view, cal, min_probability); + let pairs = neighbours::above_threshold(&columns.view(), cal, min_probability); + build(faces, cal, min_probability, &pairs) +} +/// Groups, and how confident each face's placement is. +/// +/// The second half is [`crate::assign`]'s share, not the pairwise probability +/// that put the face in the group — see that module for why the two are +/// different questions. +#[derive(Debug, Clone, PartialEq)] +pub struct Grouping { + pub clusters: Vec, + /// Indexed like the input faces. 0 for a face in no group. + pub confidence: Vec, +} + +/// Group faces into people, and score each placement against its rivals. +/// +/// What a caller writing suggestions into a catalog wants: [`cluster`] answers +/// *which person*, this answers *and how sure*. +/// +/// One scan, two thresholds. The similarity scan is the expensive part of the +/// whole subsystem and running it twice — once to merge, once to find rivals — +/// would double the cost of regrouping a library. So it runs once at the looser +/// of the two floors, and the merge engine takes the subset at or above +/// `min_probability`. That subset is identical, pair for pair and in the same +/// order, to what a scan at `min_probability` would have produced, so grouping +/// is unchanged by scoring being asked for: `scoring_does_not_change_the_ +/// groups` holds it to that. +pub fn cluster_scored(faces: &[Candidate], cal: &Calibration, min_probability: f32) -> Grouping { + if faces.is_empty() { + return Grouping { + clusters: Vec::new(), + confidence: Vec::new(), + }; + } + + let columns = Columns::of(faces); + let evidence = neighbours::above_threshold( + &columns.view(), + cal, + min_probability.min(crate::assign::RIVAL_FLOOR), + ); + let merges: Vec = evidence + .iter() + .copied() + .filter(|p| p.probability >= min_probability) + .collect(); + + let clusters = build(faces, cal, min_probability, &merges); + let confidence = crate::assign::identity_shares( + faces.len(), + &clusters, + &evidence, + crate::assign::TOP_MATCHES, + ); + Grouping { + clusters, + confidence, + } +} + +/// The three arrays [`neighbours::Faces`] borrows, owned. +/// +/// [`neighbours`] takes parallel slices rather than candidates on purpose — it +/// has no business knowing what a person is — so somebody has to hold the +/// columns. Both entry points do, identically, which is the only reason this is +/// a type and not three locals. +struct Columns { + embeddings: Vec>, + crop_px: Vec, + images: Vec, +} + +impl Columns { + fn of(faces: &[Candidate]) -> Self { + Self { + embeddings: faces.iter().map(|f| f.embedding.clone()).collect(), + crop_px: faces.iter().map(|f| f.crop_px).collect(), + images: faces.iter().map(|f| f.image).collect(), + } + } + + fn view(&self) -> Faces<'_> { + Faces { + embeddings: &self.embeddings, + crop_px: &self.crop_px, + images: &self.images, + } + } +} + +fn build( + faces: &[Candidate], + cal: &Calibration, + min_probability: f32, + pairs: &[neighbours::Pair], +) -> Vec { let mut engine = Engine::new(faces, cal, min_probability); - for component in components(faces.len(), &pairs) { - engine.agglomerate(&component, &pairs); + for component in components(faces.len(), pairs) { + engine.agglomerate(&component, pairs); } engine.finish() } @@ -663,6 +751,66 @@ mod tests { assert_eq!(out.len(), 2, "clustering overrode two user confirmations"); } + /// A face at a chosen cosine to identity 0 *and* to identity 1 at once — + /// the sibling geometry, which [`at_cosine`]'s per-identity subspaces + /// cannot express. + fn contested(to_first: f32, to_second: f32) -> Vec { + let mut v = vec![0.0_f32; EMBEDDING_DIM]; + v[0] = to_first; + v[2] = to_second; + v[4] = (1.0 - to_first * to_first - to_second * to_second) + .max(0.0) + .sqrt(); + v + } + + /// The population the scoring tests share: two faces of one person, a + /// stranger, and a face that matches the person well and the stranger + /// weakly — weakly enough that it will never merge with them, which is + /// exactly the rival a within-group score cannot see. + fn with_a_rival() -> Vec { + let mut x = candidate(4, 13, 0, 0.0); + x.embedding = contested(0.45, 0.37); + // The stranger is *named*: only an identity the user has asserted + // competes for a face (crate::assign). + let mut stranger = candidate(3, 12, 1, 1.0); + stranger.confirmed_person = Some(7); + vec![ + candidate(1, 10, 0, 1.0), + candidate(2, 11, 0, 1.0), + stranger, + x, + ] + } + + /// Asking for confidences must not move a single face. The scan runs at a + /// looser floor to find rivals, and the merge engine has to see exactly the + /// pairs it would have seen without them. + #[test] + fn scoring_does_not_change_the_groups() { + let faces = with_a_rival(); + let plain = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + let scored = cluster_scored(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(plain, scored.clusters); + } + + /// The number the user is shown answers "which of these people", so a + /// second claimant has to lower it even when it is too weak to merge. + #[test] + fn a_face_two_identities_could_claim_is_reported_as_less_certain() { + let faces = with_a_rival(); + let scored = cluster_scored(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + + let alone = cluster_scored(&faces[..2], &cal(), DEFAULT_MERGE_PROBABILITY); + assert!(alone.confidence[0] > 0.99, "nobody else to be"); + + let contested = scored.confidence[3]; + assert!( + (0.6..0.85).contains(&contested), + "a face with a second claimant: {contested}" + ); + } + #[test] fn a_suggestion_joins_the_person_its_group_is_anchored_to() { let mut anchor = candidate(1, 10, 0, 1.0); diff --git a/core/dr-face/src/lib.rs b/core/dr-face/src/lib.rs index 4a5c433..96a72e8 100644 --- a/core/dr-face/src/lib.rs +++ b/core/dr-face/src/lib.rs @@ -24,13 +24,15 @@ //! //! # Why the runtime is split behind a feature //! -//! [`calibrate`] and [`cluster`] are where this subsystem's accuracy actually -//! lives, and both are pure arithmetic over embeddings with no model in them. +//! [`calibrate`], [`cluster`] and [`assign`] are where this subsystem's accuracy +//! actually lives, and all three are pure arithmetic over embeddings with no +//! model in them. //! They build and test without `inference`, on synthetic embeddings, on a //! machine with no weights on it — which is what lets CI cover the part most //! likely to be subtly wrong. pub mod align; +pub mod assign; pub mod calibrate; pub mod cluster; #[cfg(feature = "inference")] @@ -42,8 +44,11 @@ pub mod naming; pub mod neighbours; pub use align::{warp, Aligned112, Similarity, ALIGNED_EDGE, ARCFACE_TEMPLATE}; +pub use assign::{identity_shares, RIVAL_FLOOR, TOP_MATCHES}; pub use calibrate::{Calibration, Pairs, ReliabilityBand}; -pub use cluster::{cluster, split, Candidate, Cluster, DEFAULT_MERGE_PROBABILITY}; +pub use cluster::{ + cluster, cluster_scored, split, Candidate, Cluster, Grouping, DEFAULT_MERGE_PROBABILITY, +}; #[cfg(feature = "inference")] pub use detect::{DetectOptions, Detection, Detector}; #[cfg(feature = "inference")] diff --git a/docs/faces.md b/docs/faces.md index b72fa65..3e586c9 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -228,6 +228,7 @@ core/dr-face/ src/embed.rs MBF: preprocess, forward, L2 normalise src/calibrate.rs cosine → P(same person) (FR-CULL-9) src/cluster.rs constrained agglomeration (FR-CULL-10) + src/assign.rs which person, and how sure (§9.1, FR-CULL-9) ``` ```toml @@ -675,6 +676,64 @@ are recomputed freely; confirmations survive all of it (FR-CULL-10). as the split. FR-CULL-10 requires splitting to be as easy as merging, and a split that hands the user a pile of loose faces to re-sort is not that. +### 9.1 The number beside a suggestion + +Which person a face belongs to and how sure that is are **different questions**, and the second one +is not answered by the pairwise probabilities that settled the first. + +The first implementation answered it with the mean calibrated probability between the face and the +rest of its group, and that measures the wrong thing twice. It punishes coverage: a person with two +hundred faces across fifteen years is *supposed* to have members a new photograph is orthogonal to, +so the better someone is photographed the worse their suggestions score. And it never asks who else +the face could be — a face matching Anna at 0.95 and nobody else, and one matching Anna at 0.95 and +her sister at 0.93, come out identical, when the second is the only one the user needs to look at. + +Two questions, so two factors, multiplied: + +``` +evidence(P) = Σ of the top n of { P(same | this face, f) : f ∈ P } n = 10 +coherence = evidence(own) / how many of the top n there were +uniqueness = evidence(own) / (evidence(own) + Σ evidence(named rivals)) +confidence = coherence × uniqueness +``` + +**Coherence** is the old mean with a cap on it, and the cap is the whole fix: the two hundred faces a +given photograph is legitimately orthogonal to stop counting against it. **Uniqueness** is the +competition, and it is what makes an ambiguous face read as ambiguous — two identities matching +equally well land at 0.5 each, which is the truth about a sibling. + +**Only named people compete, and they compete per person.** This is the part that had to be measured +rather than reasoned about. Normalising across *every* group made the number useless on a real +18,000-face library — median suggestion 21%, four in five under half — because clustering leaves one +person spread across many groups, so a face competes against itself. Counting only groups holding a +confirmation fixed most of it; counting them **per person** rather than per group fixed the rest, +since a named person is left in several anchored groups for the same reason. + +**Rivals are gathered below the merge threshold**, down to even odds: a named person who matches at +0.6 will never be merged into but is exactly the competition a suggestion should be discounted for. +The floor matters in both directions — summing the near-orthogonal pairs instead of dropping them +lets fifty identities' worth of upper-tail noise outweigh one real match, which on the same library +moved the median stated confidence from 100% to 31%. + +*Measured (`cargo run --release -p dr-catalog --example face_confidence`), leave-one-out over that +library's 2,702 confirmations across 54 named people:* + +| | share | old mean | +|---|---|---| +| right person picked | **99.33%** | 99.15% | +| stated for the right person, median | **99.3%** | 90.4% | +| stated for the right person, p10 | **79.2%** | 68.0% | + +The reliability table is monotone and **errs low**: 100% correct wherever it states 80% or more, 84% +correct where it states under half. Understating is the safe direction for a screen whose purpose is +deciding what to look at first, but the low bands are not calibrated and should not be read as +though they were — and the leave-one-out task asks *which of these people*, never *is it any of +them*, so it cannot speak to a stranger at all. + +It is **not** a merge threshold and must not become one. Uniqueness is relative, so a library with +one named person would hand every stray face a 1. "Is this the same person at all" stays §8's +question, and coherence is the half of the product that carries it. + --- ## 10. Catalog and jobs diff --git a/docs/traceability.md b/docs/traceability.md index 7902f84..ce9d686 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -9,7 +9,7 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n | Metric | Value | |---|---| -| Source files scanned | 279 | +| Source files scanned | 280 | | TRACES tags found | 812 | | Requirements defined | 177 | | Requirements covered | 106 | diff --git a/ui/dr-ui/src/faces.rs b/ui/dr-ui/src/faces.rs index 25fd06b..11b7376 100644 --- a/ui/dr-ui/src/faces.rs +++ b/ui/dr-ui/src/faces.rs @@ -543,7 +543,10 @@ pub fn recluster( ids.push(face_id); } - let clusters = dr_face::cluster(&candidates, &cal, min_probability); + let dr_face::Grouping { + clusters, + confidence, + } = dr_face::cluster_scored(&candidates, &cal, min_probability); let mut suggested = 0usize; let mut created = 0usize; @@ -569,10 +572,12 @@ pub fn recluster( if confirmed.contains_key(&face) { continue; } - // The probability the user is shown is the group's own coherence, - // not the single best edge into it — a face admitted by one strong - // match to an outlier should not present as certain. - let p = group_probability(&candidates, c, m, &cal); + // The probability the user is shown is this identity's share of + // the evidence for the face, against every other identity that + // could plausibly claim it (dr_face::assign) — not the single best + // edge, which cannot tell a sole match from a coin toss between + // two siblings. + let p = confidence[m]; if faces::suggest(conn, face, person, p)? { suggested += 1; } @@ -662,37 +667,6 @@ pub fn spawn_recluster( rx } -/// Mean calibrated probability between one member and the rest of its group. -fn group_probability( - candidates: &[dr_face::Candidate], - cluster: &dr_face::Cluster, - member: usize, - cal: &Calibration, -) -> f32 { - let me = &candidates[member]; - let mut sum = 0.0; - let mut n = 0.0; - for &other in &cluster.members { - if other == member { - continue; - } - let them = &candidates[other]; - let cos: f32 = me - .embedding - .iter() - .zip(&them.embedding) - .map(|(a, b)| a * b) - .sum(); - sum += cal.probability(cos, me.crop_px.min(them.crop_px), 0.0); - n += 1.0; - } - if n == 0.0 { - 1.0 - } else { - sum / n - } -} - /// A face cut out of its photograph, ready to draw. #[derive(Debug, Clone, PartialEq)] pub struct FaceCrop { From e596eb06572f03fc26838c1665f15b6031503f2f Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 11:33:15 +0200 Subject: [PATCH 3/9] Give the similarity scan the machine's SIMD, and its cache MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The scan is O(n²) dot products and nothing else, so its speed is the face subsystem's speed — and it was running at 0.7 flops per cycle. Two separate faults, both measured over the reference 18,143-face library on twenty cores. It walked the whole embedding array once per row, ~336 GB of traffic, where a column tile that fits in L2 is read once per tile of rows: 4.64s → 2.81s. And the workspace builds for baseline x86-64 — SSE2, no FMA — into which the portable loop was not being vectorised at all: 2.81s → 0.86s, 195 GFLOP/s. So the dot product is now chosen per machine. AVX2 + FMA where is_x86_feature_detected! finds it; NEON unconditionally on aarch64, since Advanced SIMD is in that baseline and every Android device the app builds for has it — with the explicit vfmaq, because LLVM will not fuse a multiply and an add without being told to. The portable loop stays as the definition the others are tested against, and the_fastest_kernel_agrees_with_the_portable_one is the only check the NEON path gets on a machine that is not aarch64. Faces::embeddings is one flat buffer rather than a Vec per face: the pointer chase defeated both the prefetcher and the tiling, and it is also the layout a GPU pass would want. Behaviour is unchanged and that is checked rather than asserted — the same 1,531,969 pairs from all three kernels, and on the real library the same 2,518 groups holding the same 16,246 faces with the same confidence distribution. A full regroup there goes from 10.0s to 5.9s; the rest is the agglomeration, which is a sequential heap walk and is where the next look should go. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-face/src/cluster.rs | 19 +- core/dr-face/src/neighbours.rs | 316 ++++++++++++++++++++++++++++----- docs/code-health.md | 2 +- docs/faces.md | 27 +++ 4 files changed, 321 insertions(+), 43 deletions(-) diff --git a/core/dr-face/src/cluster.rs b/core/dr-face/src/cluster.rs index be1315a..4a3137f 100644 --- a/core/dr-face/src/cluster.rs +++ b/core/dr-face/src/cluster.rs @@ -220,15 +220,29 @@ pub fn cluster_scored(faces: &[Candidate], cal: &Calibration, min_probability: f /// columns. Both entry points do, identically, which is the only reason this is /// a type and not three locals. struct Columns { - embeddings: Vec>, + /// Every embedding end to end — see [`Faces::embeddings`] for why flat. + embeddings: Vec, + dim: usize, crop_px: Vec, images: Vec, } impl Columns { fn of(faces: &[Candidate]) -> Self { + // Ragged input would index the wrong row for every face after the odd + // one, so the widest wins and short rows are padded with zeros: a + // zero-padded row scores lower against everything, which is the safe + // direction. It does not happen — one model, one dimension — and it is + // handled rather than trusted because the failure would be silent. + let dim = faces.iter().map(|f| f.embedding.len()).max().unwrap_or(0); + let mut embeddings = Vec::with_capacity(faces.len() * dim); + for f in faces { + embeddings.extend_from_slice(&f.embedding); + embeddings.resize(embeddings.len() + dim - f.embedding.len(), 0.0); + } Self { - embeddings: faces.iter().map(|f| f.embedding.clone()).collect(), + embeddings, + dim, crop_px: faces.iter().map(|f| f.crop_px).collect(), images: faces.iter().map(|f| f.image).collect(), } @@ -237,6 +251,7 @@ impl Columns { fn view(&self) -> Faces<'_> { Faces { embeddings: &self.embeddings, + dim: self.dim, crop_px: &self.crop_px, images: &self.images, } diff --git a/core/dr-face/src/neighbours.rs b/core/dr-face/src/neighbours.rs index af2caf3..03d0dd8 100644 --- a/core/dr-face/src/neighbours.rs +++ b/core/dr-face/src/neighbours.rs @@ -72,6 +72,13 @@ use crate::calibrate::Calibration; /// work — and large enough that the per-block overhead disappears. const BLOCK: usize = 64; +/// Columns compared against one row block before moving on. +/// +/// The other half of the tiling: 64 embeddings of 512 floats is 128 KB, which +/// sits in L2 beside the row block instead of being re-read from memory for +/// every row. See [`scan_rows`] for what it was worth. +const COLUMN_TILE: usize = 64; + /// Below this many faces, do the whole thing on the calling thread. /// /// Spawning threads for a set this small costs more than the scan. @@ -94,8 +101,17 @@ pub struct Pair { /// knowing what a person or a photograph is, and taking the three arrays it /// actually reads keeps it testable on bare vectors. pub struct Faces<'a> { - /// L2-normalised, `EMBEDDING_DIM` long, one per face. - pub embeddings: &'a [Vec], + /// Every embedding end to end, L2-normalised, [`Faces::dim`] floats each. + /// + /// **Flat, not a slice of vectors**, and the difference is measurable: a + /// `&[Vec]` is one heap allocation per face and the scan chases a + /// pointer per row, which defeats both the prefetcher and the tiling this + /// module does to stay in cache. One buffer is also the layout a GPU would + /// want, which is where this is eventually going. + pub embeddings: &'a [f32], + /// Floats per embedding — [`crate::EMBEDDING_DIM`] in practice, a parameter + /// so the tests can work in 64 dimensions. + pub dim: usize, /// Source pixels across the aligned crop, for the calibration's size term. pub crop_px: &'a [f32], /// Which photograph each face came from. Two faces in one frame are not @@ -103,6 +119,26 @@ pub struct Faces<'a> { pub images: &'a [u64], } +impl Faces<'_> { + /// How many faces there are. + pub fn len(&self) -> usize { + if self.dim == 0 { + 0 + } else { + self.embeddings.len() / self.dim + } + } + + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + #[inline(always)] + fn row(&self, i: usize) -> &[f32] { + &self.embeddings[i * self.dim..(i + 1) * self.dim] + } +} + /// Every pair whose calibrated probability reaches `min_probability`. /// /// Excludes pairs from the same photograph, which the clusterer would refuse @@ -111,14 +147,20 @@ pub struct Faces<'a> { /// /// Ordered by `(i, j)`, which is what the caller's determinism rests on. pub fn above_threshold(faces: &Faces, cal: &Calibration, min_probability: f32) -> Vec { - let n = faces.embeddings.len(); + let n = faces.len(); if n < 2 { return Vec::new(); } // The loosest cosine that could clear the bar for *any* pair in the set. // Cheaper than the sigmoid by far, and it rejects almost everything. - let tau = loosest_cosine(faces.crop_px, cal, min_probability); + let scan = Scan { + faces, + cal, + min_probability, + tau: loosest_cosine(faces.crop_px, cal, min_probability), + dot: fastest_dot(), + }; let blocks: Vec<(usize, usize)> = (0..n) .step_by(BLOCK) @@ -128,7 +170,7 @@ pub fn above_threshold(faces: &Faces, cal: &Calibration, min_probability: f32) - if n < THREADS_ABOVE { let mut out = Vec::new(); for &(from, to) in &blocks { - scan_block(faces, cal, min_probability, tau, from, to, &mut out); + scan_rows(&scan, from, to, &mut out); } return out; } @@ -158,7 +200,7 @@ pub fn above_threshold(faces: &Faces, cal: &Calibration, min_probability: f32) - break; }; let mut out = Vec::new(); - scan_block(faces, cal, min_probability, tau, from, to, &mut out); + scan_rows(&scan, from, to, &mut out); mine.push((b, out)); } mine @@ -184,40 +226,69 @@ pub fn above_threshold(faces: &Faces, cal: &Calibration, min_probability: f32) - out } -/// Compare rows `from..to` against everything after them. +/// Everything [`scan_rows`] needs that does not change between blocks. /// -/// The upper triangle, split by rows. Row `i` only looks at `j > i`, so every -/// unordered pair is visited exactly once and the emitted order is `(i, j)` -/// ascending within the block. -fn scan_block( - faces: &Faces, - cal: &Calibration, +/// A struct rather than eight arguments, and the grouping is real: these five +/// are fixed for a whole scan and only the row range moves. +#[derive(Clone, Copy)] +struct Scan<'a> { + faces: &'a Faces<'a>, + cal: &'a Calibration, min_probability: f32, tau: f32, - from: usize, - to: usize, - out: &mut Vec, -) { - let n = faces.embeddings.len(); - for i in from..to { - let a = &faces.embeddings[i]; - let crop_a = faces.crop_px[i]; - let image_a = faces.images[i]; - for j in i + 1..n { - if image_a == faces.images[j] { + dot: DotFn, +} + +/// Compare rows `from..to` against everything after them. +/// +/// The upper triangle, split by rows, and **tiled on the column side too**. +/// Walking `j` from `i + 1` to `n` for one row at a time streams the whole +/// embedding array past the core once per row — 336 GB of traffic for an +/// 18,000-face library — where a column tile small enough to sit in L2 is read +/// once per *tile* of rows. Measured on that library, the tiling alone took the +/// scan from 4.64 s to 2.81 s before any change to the kernel. +/// +/// Pairs come out ordered by `(i, j)`: the tiles are walked in ascending order +/// but the row loop is inside them, so the block's own output is sorted before +/// it is returned. Blocks are concatenated in row order, so the whole list is +/// ordered — which is the promise the caller's determinism rests on. +fn scan_rows(scan: &Scan, from: usize, to: usize, out: &mut Vec) { + let Scan { + faces, + cal, + min_probability, + tau, + dot, + } = *scan; + let n = faces.len(); + for tile in (from..n).step_by(COLUMN_TILE) { + let tile_end = (tile + COLUMN_TILE).min(n); + for i in from..to { + // The diagonal: nothing at or before `i` is this row's business. + let start = tile.max(i + 1); + if start >= tile_end { continue; } - let cos = dot(a, &faces.embeddings[j]); - // The cheap rejection, and it takes well over 99% of pairs. - if cos < tau { - continue; - } - let probability = cal.probability(cos, crop_a.min(faces.crop_px[j]), 0.0); - if probability >= min_probability { - out.push(Pair { i, j, probability }); + let a = faces.row(i); + let crop_a = faces.crop_px[i]; + let image_a = faces.images[i]; + for j in start..tile_end { + if image_a == faces.images[j] { + continue; + } + let cos = dot(a, faces.row(j)); + // The cheap rejection, and it takes well over 99% of pairs. + if cos < tau { + continue; + } + let probability = cal.probability(cos, crop_a.min(faces.crop_px[j]), 0.0); + if probability >= min_probability { + out.push(Pair { i, j, probability }); + } } } } + out.sort_unstable_by_key(|p| (p.i, p.j)); } /// The lowest cosine that could yield `min_probability` for any pair in the set. @@ -266,10 +337,144 @@ fn loosest_cosine(crop_px: &[f32], cal: &Calibration, min_probability: f32) -> f /// something to pipeline and vectorise. The order is fixed and identical on /// every run, which is what the caller's determinism needs — it is a different /// order from the naive sum, not a variable one. +/// A dot product over two equal-length, L2-normalised rows. +/// +/// Chosen once per scan rather than per pair — see [`fastest_dot`]. +type DotFn = fn(&[f32], &[f32]) -> f32; + +/// The widest dot product this machine can actually run. +/// +/// # Why this is worth unsafe code +/// +/// The scan is the arithmetic floor of the whole subsystem and it was running +/// at **0.7 flops per cycle**. The workspace builds for baseline `x86-64`, +/// which is SSE2 and no FMA, and the portable loop below was not being +/// vectorised into even that. Measured over a real 18,143-face library, on +/// twenty cores: +/// +/// | kernel | scan | GFLOP/s | +/// |---|---|---| +/// | portable, untiled (what this replaced) | 4.64 s | 36 | +/// | portable, tiled | 2.81 s | 60 | +/// | **AVX2 + FMA, tiled** | **0.86 s** | **195 | +/// +/// Identical pair lists — 1,531,969 — all three ways. +/// +/// **Runtime detection on x86-64, unconditional on aarch64.** Advanced SIMD is +/// in the aarch64 baseline, so every Android device that runs this has NEON and +/// there is nothing to detect; on x86-64 AVX2 is not baseline and a binary that +/// assumed it would not start on older hardware. +/// +/// The three kernels sum in different orders, so a cosine may differ in its +/// last bit between them. That only matters for a pair sitting exactly on the +/// threshold, and the portable kernel already sums in eight accumulators rather +/// than one, so the module was never bit-comparable with a naive sum. +fn fastest_dot() -> DotFn { + #[cfg(target_arch = "x86_64")] + { + if std::arch::is_x86_feature_detected!("avx2") && std::arch::is_x86_feature_detected!("fma") + { + return dot_avx2; + } + } + #[cfg(target_arch = "aarch64")] + { + return dot_neon; + } + #[allow(unreachable_code)] + dot +} + +/// Eight-wide fused multiply-add, on the half of desktops that have it. +#[cfg(target_arch = "x86_64")] +fn dot_avx2(a: &[f32], b: &[f32]) -> f32 { + // SAFETY: `fastest_dot` is the only thing that hands this out, and only + // after `is_x86_feature_detected!` has said both features are present. + unsafe { dot_avx2_inner(a, b) } +} + +#[cfg(target_arch = "x86_64")] +#[target_feature(enable = "avx2", enable = "fma")] +unsafe fn dot_avx2_inner(a: &[f32], b: &[f32]) -> f32 { + use std::arch::x86_64::*; + + let n = a.len().min(b.len()); + let (mut acc0, mut acc1) = (_mm256_setzero_ps(), _mm256_setzero_ps()); + let mut k = 0; + // Two accumulators, because one FMA cannot start until the previous one + // retires and the unit is pipelined several deep. + while k + 16 <= n { + // SAFETY: `k + 16 <= n`, and `n` is within both slices. + acc0 = _mm256_fmadd_ps( + _mm256_loadu_ps(a.as_ptr().add(k)), + _mm256_loadu_ps(b.as_ptr().add(k)), + acc0, + ); + acc1 = _mm256_fmadd_ps( + _mm256_loadu_ps(a.as_ptr().add(k + 8)), + _mm256_loadu_ps(b.as_ptr().add(k + 8)), + acc1, + ); + k += 16; + } + + let mut lanes = [0.0_f32; 8]; + _mm256_storeu_ps(lanes.as_mut_ptr(), _mm256_add_ps(acc0, acc1)); + let mut total = ((lanes[0] + lanes[1]) + (lanes[2] + lanes[3])) + + ((lanes[4] + lanes[5]) + (lanes[6] + lanes[7])); + for k in k..n { + total += a[k] * b[k]; + } + total +} + +/// The same, four-wide, for the phone and the tablet. +/// +/// No feature detection and no `target_feature`: Advanced SIMD is mandatory in +/// the aarch64 baseline, so this compiles for every Android target the app +/// builds for. The explicit `vfmaq` matters — LLVM will not fuse a multiply and +/// an add on its own without fast-math, which is most of the win. +#[cfg(target_arch = "aarch64")] +fn dot_neon(a: &[f32], b: &[f32]) -> f32 { + use std::arch::aarch64::*; + + let n = a.len().min(b.len()); + // SAFETY: every load below is bounded by `k + 16 <= n`, and `n` is within + // both slices. NEON needs no feature detection on aarch64. + unsafe { + let mut acc = [vdupq_n_f32(0.0); 4]; + let mut k = 0; + while k + 16 <= n { + for (l, slot) in acc.iter_mut().enumerate() { + *slot = vfmaq_f32( + *slot, + vld1q_f32(a.as_ptr().add(k + l * 4)), + vld1q_f32(b.as_ptr().add(k + l * 4)), + ); + } + k += 16; + } + + let mut total = vaddvq_f32(vaddq_f32( + vaddq_f32(acc[0], acc[1]), + vaddq_f32(acc[2], acc[3]), + )); + for k in k..n { + total += a[k] * b[k]; + } + total + } +} + +/// The portable fallback, and the definition the others have to agree with. +/// +/// Eight accumulators so the adds are independent; that is as much as can be +/// asked of a loop that has to compile for anything. pub(crate) fn dot(a: &[f32], b: &[f32]) -> f32 { const LANES: usize = 8; let mut acc = [0.0_f32; LANES]; - let chunks = a.len() / LANES; + let n = a.len().min(b.len()); + let chunks = n / LANES; for c in 0..chunks { let base = c * LANES; @@ -279,7 +484,7 @@ pub(crate) fn dot(a: &[f32], b: &[f32]) -> f32 { } let mut total = ((acc[0] + acc[1]) + (acc[2] + acc[3])) + ((acc[4] + acc[5]) + (acc[6] + acc[7])); - for k in chunks * LANES..a.len() { + for k in chunks * LANES..n { total += a[k] * b[k]; } total @@ -344,7 +549,7 @@ mod tests { } struct Set { - embeddings: Vec>, + embeddings: Vec, crop_px: Vec, images: Vec, } @@ -353,10 +558,15 @@ mod tests { fn faces(&self) -> Faces<'_> { Faces { embeddings: &self.embeddings, + dim: DIM, crop_px: &self.crop_px, images: &self.images, } } + + fn len(&self) -> usize { + self.embeddings.len() / DIM + } } /// `groups` identities, `per` faces each, every face in its own photograph. @@ -379,7 +589,7 @@ mod tests { } let crop_px = vec![150.0; embeddings.len()]; Set { - embeddings, + embeddings: embeddings.concat(), crop_px, images, } @@ -387,16 +597,17 @@ mod tests { /// The unpruned, unthreaded, unblocked definition of the answer. fn reference(faces: &Faces, cal: &Calibration, min_probability: f32) -> Vec { - let n = faces.embeddings.len(); + let n = faces.len(); let mut out = Vec::new(); for i in 0..n { for j in i + 1..n { if faces.images[i] == faces.images[j] { continue; } - let cos: f32 = faces.embeddings[i] + let cos: f32 = faces + .row(i) .iter() - .zip(&faces.embeddings[j]) + .zip(faces.row(j)) .map(|(x, y)| x * y) .sum(); let p = cal.probability(cos, faces.crop_px[i].min(faces.crop_px[j]), 0.0); @@ -416,6 +627,31 @@ mod tests { a.len() == b.len() && a.iter().zip(b).all(|(x, y)| x.i == y.i && x.j == y.j) } + /// The guard on the SIMD kernels, and the only check the aarch64 one gets + /// on a machine that is not aarch64: whatever [`fastest_dot`] picked has to + /// agree with the portable definition. A wrong lane index or a mishandled + /// tail would show up here as a wildly different number, not a rounding + /// difference. + #[test] + fn the_fastest_kernel_agrees_with_the_portable_one() { + let fast = fastest_dot(); + for seed in 0..64u64 { + let a = vector(seed); + let b = vector(seed + 1_000); + let (want, got) = (dot(&a, &b), fast(&a, &b)); + assert!( + (want - got).abs() < 1e-5, + "kernel disagreed on seed {seed}: {want} vs {got}" + ); + } + + // A length that is not a multiple of the widest step, so the tail is + // exercised rather than assumed away. + let a: Vec = (0..37).map(|k| k as f32 * 0.01).collect(); + let b: Vec = (0..37).map(|k| 1.0 - k as f32 * 0.02).collect(); + assert!((dot(&a, &b) - fast(&a, &b)).abs() < 1e-5, "tail mishandled"); + } + #[test] fn nothing_to_pair_is_no_pairs() { let s = population(1, 1, 0.9); @@ -438,7 +674,7 @@ mod tests { fn the_threaded_scan_finds_exactly_what_the_reference_does() { let s = population(300, 8, 0.97); assert!( - s.embeddings.len() > THREADS_ABOVE, + s.len() > THREADS_ABOVE, "population is below the threading cutoff" ); let f = s.faces(); diff --git a/docs/code-health.md b/docs/code-health.md index f996bf8..ce9ae39 100644 --- a/docs/code-health.md +++ b/docs/code-health.md @@ -45,7 +45,7 @@ done | sort -rn | Test functions | 2,042, plus 21 integration test files | | `.unwrap()` in production code | **3** — one in `dr-gpu`, two in `dr-ingest` | | `.unwrap()` in test code | ~1,500, which is where it belongs | -| `unsafe` blocks | 6 | +| `unsafe` blocks | 9 — three of them the face scan's SIMD kernels (faces.md §9) | | `TRACES` tags / orphan tags | 793 / 0 | | Resolved dependencies | 826 | | Largest function | `dr-ui::run` — 1,855 lines | diff --git a/docs/faces.md b/docs/faces.md index 3e586c9..c43507f 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -653,6 +653,33 @@ its histogram contribution before the next is started, and never materialised wh background, after an indexing sweep that took an hour. An approximate index is an optimisation to reach for when §12 says it is needed, not before. +**The kernel is most of the cost, and it was running at a tenth of the machine.** The scan is +`O(n²)` dot products and nothing else, so its speed *is* the subsystem's speed. Two things were +wrong with the first one, both measured over a real 18,143-face library on a twenty-core desktop: + +| | scan | GFLOP/s | +|---|---|---| +| a row against every other row, `&[Vec]` | 4.64 s | 36 | +| tiled on the column side too, one flat buffer | 2.81 s | 60 | +| **plus AVX2 + FMA** | **0.86 s** | **195** | + +The first is memory: walking the whole embedding array once per row moves ~336 GB for that library, +where a column tile that fits in L2 is read once per *tile of rows*. The second is that the workspace +builds for baseline `x86-64` — SSE2, no FMA — and the portable loop was not being vectorised into +even that, at 0.7 flops per cycle. + +So the dot product is chosen per machine: AVX2 + FMA where `is_x86_feature_detected!` finds it, +**NEON unconditionally on aarch64** — Advanced SIMD is in that baseline, so every Android device the +app builds for has it, and the explicit `vfmaq` matters because LLVM will not fuse a multiply and an +add on its own. The portable loop remains the definition the others are tested against. All three +produce the same 1,531,969 pairs. + +Worth keeping in view when this is next optimised: on that library a full regroup is **scan 4.60 s · +agglomerate 4.84 s · score 0.23 s**, so the scan was under half of it and the SIMD work moved the +whole pass from 10.0 s to 5.9 s. A GPU GEMM is the next step for the scan, and it is capped by the same +arithmetic — the agglomeration is a sequential heap walk and no amount of silicon touches +it. + **Constraints, not just thresholds:** - **Cannot-link on co-occurrence.** Two faces in the same image are never merged. This is the same From b4d39ba33a7947dc0a305c5eef2583100abc0a64 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 11:57:29 +0200 Subject: [PATCH 4/9] File each pair under its component once, not once per component MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Seeding the merge heaps was 3.06s of a 5.93s regroup on the reference 18,143-face library — more than half the pass, spent before a single merge was considered. Every component scanned the whole pair list looking for the pairs that were its own: 475 components against 804,499 pairs, 382 million set lookups to place 804,499 of them. A pair can only ever join two faces of one component, since that is what a component is, so the union-find that finds the components can file the pairs at the same time and hand each agglomeration the list it needs. The membership set inside agglomerate goes with it — it existed only to run that filter — and the heap can be sized up front now that the pair count is known. Ordering is preserved deliberately: pairs are filed in the order they arrive, which is the global (i, j) order, so the seeded heap breaks its ties exactly as before and the merge order is unchanged. Same 2,518 groups holding the same 16,246 faces on the reference library, at 3.5s rather than 5.9s. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-face/src/cluster.rs | 53 ++++++++++++++++++++++++++++++------- 1 file changed, 44 insertions(+), 9 deletions(-) diff --git a/core/dr-face/src/cluster.rs b/core/dr-face/src/cluster.rs index 4a3137f..8f578e3 100644 --- a/core/dr-face/src/cluster.rs +++ b/core/dr-face/src/cluster.rs @@ -265,8 +265,9 @@ fn build( pairs: &[neighbours::Pair], ) -> Vec { let mut engine = Engine::new(faces, cal, min_probability); - for component in components(faces.len(), pairs) { - engine.agglomerate(&component, pairs); + let parts = components(faces.len(), pairs); + for (component, edges) in parts.members.iter().zip(&parts.edges) { + engine.agglomerate(component, edges); } engine.finish() } @@ -405,10 +406,9 @@ impl<'a> Engine<'a> { if component.len() < 2 { return; } - let members: HashSet = component.iter().copied().collect(); - let mut heap = BinaryHeap::new(); - for p in pairs.iter().filter(|p| members.contains(&p.i)) { + let mut heap = BinaryHeap::with_capacity(pairs.len()); + for p in pairs { self.links.insert( key(p.i, p.j), Link { @@ -632,7 +632,7 @@ fn key(a: usize, b: usize) -> (usize, usize) { /// separate and much smaller agglomeration. Returned with the members of each /// component ascending, and the components themselves in order of their lowest /// member — the determinism the merge order inherits. -fn components(n: usize, pairs: &[neighbours::Pair]) -> Vec> { +fn components(n: usize, pairs: &[neighbours::Pair]) -> Components { let mut parent: Vec = (0..n).collect(); fn find(parent: &mut [usize], mut x: usize) -> usize { @@ -659,9 +659,44 @@ fn components(n: usize, pairs: &[neighbours::Pair]) -> Vec> { let r = find(&mut parent, i); by_root.entry(r).or_default().push(i); } - let mut out: Vec> = by_root.into_values().filter(|c| c.len() > 1).collect(); - out.sort_unstable_by_key(|c| c[0]); - out + let mut members: Vec> = by_root.into_values().filter(|c| c.len() > 1).collect(); + members.sort_unstable_by_key(|c| c[0]); + + // Each pair filed under its component, in one pass. + // + // **This is not bookkeeping, it is the cost of the whole stage.** Each + // component used to scan the entire pair list for the ones that were its + // own: 475 components against 804,499 pairs on the reference library, 382 + // million set lookups, and 3.06 s of a 5.93 s regroup spent before a single + // merge was considered. A pair can only ever join two faces of one + // component — that is what a component *is* — so one pass over the list + // places every pair exactly where it is needed. + let mut slot_of = vec![usize::MAX; n]; + for (slot, c) in members.iter().enumerate() { + for &m in c { + slot_of[m] = slot; + } + } + let mut edges: Vec> = vec![Vec::new(); members.len()]; + for p in pairs { + // Ordering within a component follows the global order, which is what + // the seeded heap's tiebreak — and so the determinism promise — rests + // on. + if let Some(slot) = slot_of.get(p.i).copied().filter(|&s| s != usize::MAX) { + edges[slot].push(*p); + } + } + + Components { members, edges } +} + +/// The connected components of the pair graph, each with the pairs inside it. +/// +/// The two halves are parallel: `members[k]` and `edges[k]` describe the same +/// component. +struct Components { + members: Vec>, + edges: Vec>, } #[cfg(test)] mod tests { From a67402961d7ed08592253ed319f4ac17d50544a6 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 11:58:07 +0200 Subject: [PATCH 5/9] Let the merge engine's dot product use the machine's kernel too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Engine::cross is the one place a dot product is computed during agglomeration — when two groups become adjacent through a third and their sub-threshold pairs, never summed because they were never interesting, have to be accounted for. It was calling the portable loop while the scan beside it had AVX2 or NEON, which on the reference library was 1,753,514 dot products taking 0.54s. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-face/src/cluster.rs | 7 ++++++- core/dr-face/src/neighbours.rs | 4 ++-- 2 files changed, 8 insertions(+), 3 deletions(-) diff --git a/core/dr-face/src/cluster.rs b/core/dr-face/src/cluster.rs index 8f578e3..065f81d 100644 --- a/core/dr-face/src/cluster.rs +++ b/core/dr-face/src/cluster.rs @@ -370,6 +370,10 @@ impl Ord for Pending { struct Engine<'a> { faces: &'a [Candidate], cal: &'a Calibration, + /// The same kernel [`neighbours`] scans with. [`Engine::cross`] is the one + /// place a dot product is computed during agglomeration, and it was using + /// the portable loop while the scan beside it had the machine's SIMD. + dot: neighbours::DotFn, min_probability: f32, groups: Vec, links: HashMap<(usize, usize), Link>, @@ -394,6 +398,7 @@ impl<'a> Engine<'a> { Self { faces, cal, + dot: neighbours::fastest_dot(), min_probability, groups, links: HashMap::new(), @@ -581,7 +586,7 @@ impl<'a> Engine<'a> { let mut count = 0.0_f64; for &i in members { for &j in &self.groups[group].members { - let cos = neighbours::dot(&self.faces[i].embedding, &self.faces[j].embedding); + let cos = (self.dot)(&self.faces[i].embedding, &self.faces[j].embedding); let min_crop = self.faces[i].crop_px.min(self.faces[j].crop_px); sum += self.cal.probability(cos, min_crop, 0.0) as f64; count += 1.0; diff --git a/core/dr-face/src/neighbours.rs b/core/dr-face/src/neighbours.rs index 03d0dd8..b6b4213 100644 --- a/core/dr-face/src/neighbours.rs +++ b/core/dr-face/src/neighbours.rs @@ -340,7 +340,7 @@ fn loosest_cosine(crop_px: &[f32], cal: &Calibration, min_probability: f32) -> f /// A dot product over two equal-length, L2-normalised rows. /// /// Chosen once per scan rather than per pair — see [`fastest_dot`]. -type DotFn = fn(&[f32], &[f32]) -> f32; +pub(crate) type DotFn = fn(&[f32], &[f32]) -> f32; /// The widest dot product this machine can actually run. /// @@ -369,7 +369,7 @@ type DotFn = fn(&[f32], &[f32]) -> f32; /// last bit between them. That only matters for a pair sitting exactly on the /// threshold, and the portable kernel already sums in eight accumulators rather /// than one, so the module was never bit-comparable with a naive sum. -fn fastest_dot() -> DotFn { +pub(crate) fn fastest_dot() -> DotFn { #[cfg(target_arch = "x86_64")] { if std::arch::is_x86_feature_detected!("avx2") && std::arch::is_x86_feature_detected!("fma") From 8f596262b8985643c6253c4821cc4fccb2285c80 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 11:59:01 +0200 Subject: [PATCH 6/9] Record where a regroup's time actually goes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The §9 note said the scan was half a regroup and implied the agglomeration was an irreducible sequential walk. Both halves of that are now wrong, and the numbers are the point of the section. Co-Authored-By: Claude Opus 5 (1M context) --- docs/faces.md | 25 ++++++++++++++++++++----- 1 file changed, 20 insertions(+), 5 deletions(-) diff --git a/docs/faces.md b/docs/faces.md index c43507f..56abafb 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -674,11 +674,26 @@ app builds for has it, and the explicit `vfmaq` matters because LLVM will not fu add on its own. The portable loop remains the definition the others are tested against. All three produce the same 1,531,969 pairs. -Worth keeping in view when this is next optimised: on that library a full regroup is **scan 4.60 s · -agglomerate 4.84 s · score 0.23 s**, so the scan was under half of it and the SIMD work moved the -whole pass from 10.0 s to 5.9 s. A GPU GEMM is the next step for the scan, and it is capped by the same -arithmetic — the agglomeration is a sequential heap walk and no amount of silicon touches -it. +**Where a regroup's time actually goes**, on that library, because the answer moved twice while it +was being looked at: + +| | before | after | +|---|---|---| +| scan | 4.60 s | **0.94 s** | +| agglomerate | 4.84 s | **1.75 s** | +| score | 0.23 s | 0.28 s | +| **total** | **10.0 s** | **3.0 s** | + +The scan came down by the kernel work above. The agglomeration was not, as it looked, an irreducible +sequential heap walk: 3.06 s of it was every component scanning the *whole* pair list for the pairs +that were its own — 382 million set lookups to place 804,499 pairs — which the union-find can do in +one pass while it is finding the components anyway. Another 0.54 s was `Engine::cross` computing dot +products with the portable loop while the scan beside it used the machine's SIMD. + +A GPU GEMM is the obvious next step for the scan, and it should be judged against the second column +rather than the first: the scan is now under a third of the pass on a desktop, so the ceiling there +is a 1.5× regroup. On a tablet, where the CPU is several times slower and the GPU is not, the split +is different and the case is stronger — which is a measurement nobody has taken yet. **Constraints, not just thresholds:** From f4395bd17c712a85fef1537475ca9a0aba5ad5b9 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 12:02:03 +0200 Subject: [PATCH 7/9] Run the face tests on the tablet, where the NEON kernel actually runs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The similarity scan picks its dot product per machine, and the NEON one is the kernel that ships to the phone and the tablet — and the one a desktop cargo test never executes. A wrong lane index or a mishandled tail there is a silent wrong answer on exactly the devices nobody runs the suite on, which is a poor place for the only untested code path. dr-face carries no weights and touches no display, so its tests are a plain ARM64 binary that runs under adb shell with nothing installed. The script builds it against the SDK's newest NDK, pushes it, runs it and cleans up. It checks for the device first, so a tablet that is not plugged in costs a second rather than the two minutes it takes to compile for it. Not wired into CI, which has no device attached. Co-Authored-By: Claude Opus 5 (1M context) --- docs/faces.md | 5 +++ tools/face-tests-on-device.sh | 78 +++++++++++++++++++++++++++++++++++ 2 files changed, 83 insertions(+) create mode 100755 tools/face-tests-on-device.sh diff --git a/docs/faces.md b/docs/faces.md index 56abafb..2ee00c5 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -674,6 +674,11 @@ app builds for has it, and the explicit `vfmaq` matters because LLVM will not fu add on its own. The portable loop remains the definition the others are tested against. All three produce the same 1,531,969 pairs. +The NEON kernel is the one a desktop `cargo test` never executes, so +`tools/face-tests-on-device.sh` runs the suite on an attached device: `dr-face` carries no weights +and touches no display, so its tests are a plain ARM64 binary that runs under `adb shell` with +nothing installed. Worth running whenever the kernels change. + **Where a regroup's time actually goes**, on that library, because the answer moved twice while it was being looked at: diff --git a/tools/face-tests-on-device.sh b/tools/face-tests-on-device.sh new file mode 100755 index 0000000..121b7b1 --- /dev/null +++ b/tools/face-tests-on-device.sh @@ -0,0 +1,78 @@ +#!/usr/bin/env bash +# Run dr-face's test suite on a connected Android device. +# +# ./tools/face-tests-on-device.sh [extra cargo test args…] +# +# ## Why this exists +# +# `dr-face`'s similarity scan picks its dot product per machine (docs/faces.md +# §9): AVX2 where the CPU has it, **NEON on aarch64**, and a portable loop +# otherwise. The NEON kernel is the one that runs on the phone and the tablet, +# and it is the one a desktop `cargo test` never executes — a wrong lane index +# or a mishandled tail there would be a silent wrong answer on exactly the +# devices nobody runs the test suite on. +# +# `the_fastest_kernel_agrees_with_the_portable_one` is written to catch that, +# and this is how it gets to run on the hardware it is about. Everything else +# in the suite comes along for free: `dr-face` carries no weights and touches +# no display, so its tests are a plain ARM64 binary that runs under `adb +# shell` with nothing installed. +# +# Not part of CI, which has no device attached. Run it when the kernels change. +set -euo pipefail + +TARGET=aarch64-linux-android +API=26 +DEST=/data/local/tmp/dr_face_tests + +ndk="${ANDROID_NDK_HOME:-}" +if [[ -z "$ndk" ]]; then + # The newest NDK the SDK has, which is what the app is built with. + ndk=$(find "${ANDROID_HOME:-$HOME/Android/Sdk}/ndk" -maxdepth 1 -mindepth 1 -type d 2>/dev/null | + sort -V | tail -1) +fi +[[ -n "$ndk" && -d "$ndk" ]] || { + echo "no NDK found — set ANDROID_NDK_HOME" >&2 + exit 1 +} + +clang="$ndk/toolchains/llvm/prebuilt/linux-x86_64/bin/$TARGET$API-clang" +[[ -x "$clang" ]] || { + echo "no $TARGET$API-clang in $ndk" >&2 + exit 1 +} + +rustup target list --installed | grep -qx "$TARGET" || rustup target add "$TARGET" + +# Told to the device before the build, so a missing tablet costs seconds rather +# than the two minutes it takes to compile for it. +adb devices | grep -qw device || { + echo "no device attached — plug the tablet in and enable USB debugging" >&2 + exit 1 +} + +echo "building dr-face tests for $TARGET…" +binary=$( + CARGO_TARGET_AARCH64_LINUX_ANDROID_LINKER="$clang" \ + CC_aarch64_linux_android="$clang" \ + cargo test -p dr-face --lib --release --target "$TARGET" --no-run --message-format=json "$@" | + python3 -c ' +import json, sys +for line in sys.stdin: + m = json.loads(line) + if m.get("profile", {}).get("test") and m.get("executable"): + print(m["executable"]) +' | tail -1 +) +[[ -n "$binary" ]] || { + echo "cargo produced no test binary" >&2 + exit 1 +} + +echo "pushing $(basename "$binary")…" +adb push "$binary" "$DEST" >/dev/null +adb shell chmod 755 "$DEST" +# `--test-threads` left alone: the scan spawns its own workers and the point is +# to exercise them the way the app will. +adb shell "$DEST" --color never +adb shell rm -f "$DEST" From b2250cc460a1ca2fa59bb6a96d2480fdb31cd41a Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 12:16:48 +0200 Subject: [PATCH 8/9] Measure a regroup on the tablet, not just on the desktop MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The GPU question needed a number nobody had: how a regroup divides on the hardware whose CPU is weakest. dr-face carries no weights and touches no display, and dr-catalog's example needs only a catalog file, so both run under adb shell against a copy of a real library. On the same 18,143 faces — desktop against the tablet — scan 0.96s / 2.61s, agglomerate 1.69s / 2.16s, score 0.26s / 0.40s. The scan is half the pass on the tablet and under a third on the desktop, because twenty cores of AVX2 pull ahead of NEON much further than the merge engine's single-threaded hashing does. So a GPU GEMM is worth roughly 2× a regroup on the tablet and 1.5× here, and it is the tablet that should decide whether it is built. The two architectures agree exactly: the same 1,531,969 evidence pairs, the same 2,518 groups holding the same 16,246 faces, the same reliability table. That is a better check on the NEON kernel than the unit test can be. Two instruments, both read-only: the example now prints its phases, and dr-face gains scan_bench, which needs no library at all and so can answer "how fast is this machine" on a device with nothing on it. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-catalog/examples/face_confidence.rs | 40 ++++++++ core/dr-face/examples/scan_bench.rs | 104 ++++++++++++++++++++ docs/faces.md | 28 +++--- docs/traceability.md | 2 +- 4 files changed, 162 insertions(+), 12 deletions(-) create mode 100644 core/dr-face/examples/scan_bench.rs diff --git a/core/dr-catalog/examples/face_confidence.rs b/core/dr-catalog/examples/face_confidence.rs index e2d633c..55d278a 100644 --- a/core/dr-catalog/examples/face_confidence.rs +++ b/core/dr-catalog/examples/face_confidence.rs @@ -305,6 +305,46 @@ fn full_library( candidates.sort_by_key(|c| c.face); println!("\nthe whole library, at the default merge probability:"); + + // The three phases, separately, because "a regroup takes n seconds" does + // not tell anyone which half to optimise — and the answer differs between + // a desktop and a tablet (docs/faces.md §9). + { + let dim = candidates.first().map(|c| c.embedding.len()).unwrap_or(0); + let flat: Vec = candidates.iter().flat_map(|c| c.embedding.clone()).collect(); + let crop_px: Vec = candidates.iter().map(|c| c.crop_px).collect(); + let images: Vec = candidates.iter().map(|c| c.image).collect(); + let view = dr_face::neighbours::Faces { + embeddings: &flat, + dim, + crop_px: &crop_px, + images: &images, + }; + + let t = std::time::Instant::now(); + let evidence = dr_face::neighbours::above_threshold(&view, cal, dr_face::RIVAL_FLOOR); + let scan = t.elapsed().as_secs_f64(); + + // `cluster` runs its own scan at the merge threshold, so the + // agglomeration is what is left after taking one scan off the total. + let t = std::time::Instant::now(); + let clusters = dr_face::cluster(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY); + let agglomerate = t.elapsed().as_secs_f64() - scan; + + let t = std::time::Instant::now(); + let _ = dr_face::identity_shares( + candidates.len(), + &clusters, + &evidence, + dr_face::TOP_MATCHES, + ); + println!( + " scan {scan:.2}s ({} evidence pairs) · agglomerate {agglomerate:.2}s · score {:.2}s", + evidence.len(), + t.elapsed().as_secs_f64() + ); + } + let start = std::time::Instant::now(); let grouping = dr_face::cluster_scored(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY); let real: Vec<_> = grouping diff --git a/core/dr-face/examples/scan_bench.rs b/core/dr-face/examples/scan_bench.rs new file mode 100644 index 0000000..d1279c2 --- /dev/null +++ b/core/dr-face/examples/scan_bench.rs @@ -0,0 +1,104 @@ +//! How fast this machine can scan a library for face pairs. +//! +//! cargo run --release -p dr-face --example scan_bench [-- FACES…] +//! +//! Synthetic embeddings, because the scan's cost is `n²/2` dot products and +//! does not care what the vectors mean — which is what makes this runnable on +//! a phone with no library on it, over `adb shell`, beside +//! `tools/face-tests-on-device.sh`. +//! +//! # What it is for +//! +//! docs/faces.md §9 has the desktop numbers and the question they leave open: +//! a GPU GEMM is worth roughly 1.5× of a regroup on a twenty-core desktop, +//! because the scan is under a third of the pass there. On a tablet the CPU is +//! several times slower and the GPU is not, so the same optimisation is worth +//! something different — and nobody had measured which. +//! +//! No weights, no catalog, no display: it needs nothing but the binary. + +use dr_face::neighbours::{above_threshold, Faces}; +use dr_face::{Calibration, EMBEDDING_DIM, RIVAL_FLOOR}; + +/// Sizes to time, unless the command line names others. +const DEFAULT_SIZES: [usize; 4] = [2_000, 4_000, 8_000, 18_000]; + +fn main() { + let sizes: Vec = { + let given: Vec = std::env::args().skip(1).filter_map(|a| a.parse().ok()).collect(); + if given.is_empty() { + DEFAULT_SIZES.to_vec() + } else { + given + } + }; + + // The reference curve, so the cosine floor is the one a real library with + // no fit of its own would scan at. + let cal = Calibration::default(); + + println!("{:>8} {:>9} {:>10} {:>9}", "faces", "scan", "pairs", "GFLOP/s"); + for n in sizes { + let (embeddings, crop_px, images) = population(n); + let faces = Faces { + embeddings: &embeddings, + dim: EMBEDDING_DIM, + crop_px: &crop_px, + images: &images, + }; + + let start = std::time::Instant::now(); + let pairs = above_threshold(&faces, &cal, RIVAL_FLOOR); + let secs = start.elapsed().as_secs_f64(); + + let flop = n as f64 * n as f64 / 2.0 * EMBEDDING_DIM as f64 * 2.0; + println!( + "{n:>8} {secs:>8.2}s {:>10} {:>9.1}", + pairs.len(), + flop / secs / 1e9 + ); + } +} + +/// `n` L2-normalised embeddings in a handful of loose clusters. +/// +/// Clustered rather than uniform so the scan finds a plausible number of pairs +/// to keep — a population where nothing survives the threshold would time the +/// rejection path alone, which is not the path that matters. Hashed from an +/// index rather than drawn from an RNG, so a number is reproducible from the +/// command that produced it. +fn population(n: usize) -> (Vec, Vec, Vec) { + let identities = (n / 12).max(1); + let mut embeddings = Vec::with_capacity(n * EMBEDDING_DIM); + for i in 0..n { + let mut v = unit(i % identities); + let noise = unit(i + 1_000_000); + for (x, e) in v.iter_mut().zip(&noise) { + *x = 0.75 * *x + 0.25 * e; + } + embeddings.extend(normalise(v)); + } + // Every face in its own photograph: the co-occurrence rule skips pairs + // rather than scoring them, and skipped pairs are not what is being timed. + ((embeddings), vec![150.0; n], (0..n as u64).collect()) +} + +fn unit(seed: usize) -> Vec { + let mut s = (seed as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15) | 1; + let mut v = Vec::with_capacity(EMBEDDING_DIM); + for _ in 0..EMBEDDING_DIM { + s ^= s << 13; + s ^= s >> 7; + s ^= s << 17; + v.push(((s >> 11) as f64 / (1u64 << 53) as f64) as f32 - 0.5); + } + normalise(v) +} + +fn normalise(mut v: Vec) -> Vec { + let len = v.iter().map(|x| x * x).sum::().sqrt(); + for x in &mut v { + *x /= len; + } + v +} diff --git a/docs/faces.md b/docs/faces.md index 2ee00c5..40ed45e 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -680,14 +680,15 @@ and touches no display, so its tests are a plain ARM64 binary that runs under `a nothing installed. Worth running whenever the kernels change. **Where a regroup's time actually goes**, on that library, because the answer moved twice while it -was being looked at: +was being looked at. Measured with `cargo run --release -p dr-catalog --example face_confidence -- +CATALOG --full`, on the reference desktop and on a Honor tablet (ROD2-W09, aarch64): -| | before | after | -|---|---|---| -| scan | 4.60 s | **0.94 s** | -| agglomerate | 4.84 s | **1.75 s** | -| score | 0.23 s | 0.28 s | -| **total** | **10.0 s** | **3.0 s** | +| | desktop, before | desktop | **tablet** | +|---|---|---|---| +| scan | 4.60 s | 0.96 s | **2.61 s** | +| agglomerate | 4.84 s | 1.69 s | **2.16 s** | +| score | 0.23 s | 0.26 s | **0.40 s** | +| **total** | **10.0 s** | **3.1 s** | **5.2 s** | The scan came down by the kernel work above. The agglomeration was not, as it looked, an irreducible sequential heap walk: 3.06 s of it was every component scanning the *whole* pair list for the pairs @@ -695,10 +696,15 @@ that were its own — 382 million set lookups to place 804,499 pairs — which t one pass while it is finding the components anyway. Another 0.54 s was `Engine::cross` computing dot products with the portable loop while the scan beside it used the machine's SIMD. -A GPU GEMM is the obvious next step for the scan, and it should be judged against the second column -rather than the first: the scan is now under a third of the pass on a desktop, so the ceiling there -is a 1.5× regroup. On a tablet, where the CPU is several times slower and the GPU is not, the split -is different and the case is stronger — which is a measurement nobody has taken yet. +**The two architectures agree exactly**, which is worth more than either column: the same 1,531,969 +evidence pairs, the same 2,518 groups holding the same 16,246 faces, and the same reliability table, +from AVX2 on the desktop and NEON on the tablet. That is the cross-kernel check the unit test can +only approximate. + +**The tablet is where a GPU GEMM would pay.** Its scan is *half* the pass, against under a third on +the desktop — twenty cores of AVX2 pull ahead of a tablet's NEON far more than the merge engine's +single-threaded hashing does. So a perfect GEMM is worth about 2× a regroup there and about 1.5× +here, and it is the phone and tablet story that should decide whether it gets built. **Constraints, not just thresholds:** diff --git a/docs/traceability.md b/docs/traceability.md index ce9d686..21fabda 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -9,7 +9,7 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n | Metric | Value | |---|---| -| Source files scanned | 280 | +| Source files scanned | 281 | | TRACES tags found | 812 | | Requirements defined | 177 | | Requirements covered | 106 | From 39c34d4e4485f1f11529e72ead9cf7fa145bfd5a Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 12:32:31 +0200 Subject: [PATCH 9/9] Say what actually counts as a rival, now that a name anchors too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit assign's denominator is the identities the user has ruled on, and it reads Cluster::person to find them. Master's "Let a name hold a group together" widened what sets that field: a confirmation, a name, or an ignore, where before it was a confirmation alone. The behaviour is right either way — a named person is exactly the identity a suggestion should be discounted against — but the module note and faces.md §9.1 both said "a confirmation", which is now too narrow. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-face/src/assign.rs | 16 +++++++++------- docs/faces.md | 7 ++++--- 2 files changed, 13 insertions(+), 10 deletions(-) diff --git a/core/dr-face/src/assign.rs b/core/dr-face/src/assign.rs index 66a7236..e5e433e 100644 --- a/core/dr-face/src/assign.rs +++ b/core/dr-face/src/assign.rs @@ -58,9 +58,10 @@ //! and the better covered the person, the more fragments there are to lose to. //! //! A fragment is not a rival. An identity the user has actually asserted is, so -//! the denominator counts only people a confirmation names ([`Cluster::person`]) -//! — and counts them **per person, not per group**, since a named person is -//! left in several anchored groups for the same reason. Keying it by group had +//! the denominator counts only the people they have ruled on — a group carries +//! a [`Cluster::person`] when it holds a confirmation, a name, or an ignore — +//! and counts them **per person, not per group**, since one person is left in +//! several anchored groups for the same reason. Keying it by group had //! Catherine competing with Catherine and put the median suggestion onto a //! named person at 39%; keying it by person put it at 99.5%. //! @@ -209,10 +210,11 @@ pub fn identity_shares(faces: usize, clusters: &[Cluster], pairs: &[Pair], top: ours = score; coherence = score / counted as f32; } else if matches!(who, Identity::Person(_)) { - // Only an identity the user has asserted competes. An unnamed - // group that matches this face is far more likely to be another - // fragment of the same person than a different one — see the - // module note, and the library it was measured on. + // Only an identity the user has ruled on competes. A group + // nobody has ruled on and that matches this face is far more + // likely to be another fragment of the same person than a + // different one — see the module note, and the library it was + // measured on. rivals += score; } } diff --git a/docs/faces.md b/docs/faces.md index 40ed45e..6abfcfd 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -758,9 +758,10 @@ equally well land at 0.5 each, which is the truth about a sibling. **Only named people compete, and they compete per person.** This is the part that had to be measured rather than reasoned about. Normalising across *every* group made the number useless on a real 18,000-face library — median suggestion 21%, four in five under half — because clustering leaves one -person spread across many groups, so a face competes against itself. Counting only groups holding a -confirmation fixed most of it; counting them **per person** rather than per group fixed the rest, -since a named person is left in several anchored groups for the same reason. +person spread across many groups, so a face competes against itself. Counting only the groups the user has +ruled on — one holding a confirmation, a name, or an ignore — fixed most of it; counting them **per +person** rather than per group fixed the rest, since one person is left in several anchored groups +for the same reason. **Rivals are gathered below the merge threshold**, down to even odds: a named person who matches at 0.6 will never be merged into but is exactly the competition a suggestion should be discounted for.