From 00e78dc2ac88f4414e2aa39f7e7cea1c2a85fadb Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Wed, 26 Aug 2026 20:10:33 +0200 Subject: [PATCH] Cluster faces into people, and calibrate what a similarity means FR-CULL-9 forbids thresholding a bare cosine anywhere in the subsystem, so calibrate fits P(same person) per library and reports whether the fit is trustworthy. Two details carry most of the weight. The fit runs against a 200-bin histogram rather than a pair list: a 25,000-face library has ~3e8 pairs and no gradient descent is running over that. And a fresh library has no valid calibration, because the positives have to come from user confirmations or burst siblings -- bootstrapping them from high cosine would fit the calibration to the belief it was supposed to test. Clustering defends against the over-merging FR-CULL-10 warns about with constraints rather than a better threshold: two faces in one photograph never merge, and two groups confirmed as different people never merge. Average link rather than single link, so one strong edge cannot weld two families together. Calibration is defined once, in dr-face, and dr-catalog re-exports it. Two implementations of one probability model is exactly how a number comes to mean the wrong thing. Co-Authored-By: Claude Opus 5 (1M context) --- Cargo.lock | 1 + core/dr-catalog/Cargo.toml | 5 + core/dr-catalog/src/faces.rs | 65 ++--- core/dr-face/src/align.rs | 2 +- core/dr-face/src/calibrate.rs | 436 +++++++++++++++++++++++++++++++++ core/dr-face/src/cluster.rs | 443 ++++++++++++++++++++++++++++++++++ core/dr-face/src/lib.rs | 4 + docs/faces.md | 23 ++ docs/traceability.md | 2 +- 9 files changed, 929 insertions(+), 52 deletions(-) create mode 100644 core/dr-face/src/calibrate.rs create mode 100644 core/dr-face/src/cluster.rs diff --git a/Cargo.lock b/Cargo.lock index 5ddd6ca..b235857 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1406,6 +1406,7 @@ checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76" name = "dr-catalog" version = "0.7.0" dependencies = [ + "dr-face", "dr-plat", "dr-types", "env_logger", diff --git a/core/dr-catalog/Cargo.toml b/core/dr-catalog/Cargo.toml index 6be0996..984aa5c 100644 --- a/core/dr-catalog/Cargo.toml +++ b/core/dr-catalog/Cargo.toml @@ -7,6 +7,11 @@ license.workspace = true [dependencies] dr-types.workspace = true +# The face subsystem's arithmetic — `Calibration` in particular, so the sigmoid +# that turns a cosine into a probability has exactly one definition. Default +# features are off, so this brings in no ONNX runtime and no weights: only the +# model-free half compiles here. +dr-face.workspace = true # The `Storage` trait, and nothing else from it. A scan has to read a real # directory, and this is how `core/` reaches the platform without a # `#[cfg(target_os)]` of its own (ARCH §4.1: calls go downward). diff --git a/core/dr-catalog/src/faces.rs b/core/dr-catalog/src/faces.rs index 58c4d61..9fefcf1 100644 --- a/core/dr-catalog/src/faces.rs +++ b/core/dr-catalog/src/faces.rs @@ -103,55 +103,14 @@ pub struct Person { pub suggested_faces: u64, } -/// The FR-CULL-9 calibration for one model. -#[derive(Debug, Clone, Copy, PartialEq)] -pub struct Calibration { - pub a: f32, - pub b: f32, - pub w_size: f32, - /// Whether the fit is usable. When false the UI says confidence is - /// unavailable rather than presenting an untuned default as a measurement. - pub valid: bool, - pub positive_pairs: u64, - pub negative_pairs: u64, -} - -impl Calibration { - /// P(same person), optionally shifted by a base-rate prior. - /// - /// `log_prior_odds` is applied at evaluation rather than baked into the - /// fit, so one stored calibration serves every context: the odds that two - /// faces in a 40-image album match are not the odds in a 40,000-image - /// archive. Baking a prior in would need a refit per context and would make - /// the stored parameters mean different things depending on where they came - /// from. - pub fn probability(&self, cosine: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 { - let z = self.a * cosine - + self.b - + self.w_size * min_crop_px.max(1.0).log2() - + log_prior_odds; - // Branch on the sign so neither tail overflows: exp(-z) for large - // positive z, exp(z) for large negative. - if z >= 0.0 { - 1.0 / (1.0 + (-z).exp()) - } else { - let e = z.exp(); - e / (1.0 + e) - } - } - - /// The cosine at which [`Calibration::probability`] crosses `p`. - /// - /// What turns "merge above 0.9" into an actual comparison against a stored - /// similarity, without evaluating the sigmoid per candidate edge. - pub fn boundary_at(&self, p: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 { - ((p / (1.0 - p)).ln() - - self.b - - self.w_size * min_crop_px.max(1.0).log2() - - log_prior_odds) - / self.a - } -} +/// The FR-CULL-9 calibration, re-exported from where it is fitted. +/// +/// Deliberately *not* redefined here. The catalog stores five numbers; what +/// those numbers mean — the sigmoid, the size term, the prior — is +/// [`dr_face::calibrate`]'s business, and two implementations of one +/// probability model is precisely the kind of divergence that yields a +/// plausible number meaning the wrong thing. +pub use dr_face::Calibration; // ── detection ───────────────────────────────────────────────────────────── @@ -278,6 +237,12 @@ pub fn unassigned(conn: &Connection, model_id: &str) -> Result, Cata rows.collect::>().map_err(Into::into) } +/// One face's stored embedding, as the clustering pass consumes it. +/// +/// A named type rather than a tuple because it crosses a crate boundary and +/// "the third element" is not a thing anyone should have to remember. +pub type StoredEmbedding = (FaceId, ImageId, Vec, f32); + /// Embeddings for clustering, oldest first so the pass is deterministic. /// /// Returned as raw f16 blobs rather than decoded vectors: the caller is @@ -286,7 +251,7 @@ pub fn unassigned(conn: &Connection, model_id: &str) -> Result, Cata pub fn embeddings( conn: &Connection, model_id: &str, -) -> Result, f32)>, CatalogError> { +) -> Result, CatalogError> { let mut q = conn.prepare( "SELECT id, image_id, embedding, crop_px FROM faces WHERE model_id = ?1 ORDER BY id", diff --git a/core/dr-face/src/align.rs b/core/dr-face/src/align.rs index 7757d36..2d5f18f 100644 --- a/core/dr-face/src/align.rs +++ b/core/dr-face/src/align.rs @@ -332,7 +332,7 @@ mod tests { let mut rgb = vec![0.0_f32; w * h * 3]; for y in 0..h { for x in 0..w { - if x >= 56 && x < 168 && y >= 56 && y < 168 { + if (56..168).contains(&x) && (56..168).contains(&y) { for c in 0..3 { rgb[(y * w + x) * 3 + c] = 1.0; } diff --git a/core/dr-face/src/calibrate.rs b/core/dr-face/src/calibrate.rs new file mode 100644 index 0000000..1884a90 --- /dev/null +++ b/core/dr-face/src/calibrate.rs @@ -0,0 +1,436 @@ +//! Cosine to probability (docs/faces.md §8, FR-CULL-9). +//! +//! FR-CULL-9 is a hard requirement rather than an implementation detail: no +//! code path may threshold a bare cosine, every threshold in the subsystem is +//! stated as a probability, and the fit is per library and reports its own +//! validity. The failure it guards against is invisible — a raw cosine means +//! something different for every model, every population and every face size, +//! and an uncalibrated similarity still *looks* like a plausible number all the +//! way to the user interface. +//! +//! Model-free, so the part of this subsystem most likely to be subtly wrong is +//! testable on synthetic embeddings with no weights on the machine. +//! +//! # Where the training pairs come from +//! +//! **Negatives are free and abundant.** Two faces detected in *the same +//! photograph* are almost never the same person, which hands every multi-face +//! image in the library a full set of negative pairs at no labelling cost — and +//! they are *hard* negatives, from the same camera, lighting and processing, +//! which is exactly the population where a threshold tuned on easy negatives +//! fails. The exceptions (mirrors, photographs of photographs, collages) are +//! rare enough to be noise at this scale. +//! +//! **Positives have to be earned.** In order of trustworthiness: pairs the user +//! has confirmed onto one person; then burst siblings, since FR-CULL-5 already +//! groups bursts and two faces in adjacent frames are near-certainly the same +//! person. Nothing else — bootstrapping positives from high cosine is circular, +//! fitting the calibration to the belief it was supposed to test. +//! +//! Which is why a fresh library has **no valid calibration**, and says so. + +/// Bins over cosine ∈ [-1, 1]. +/// +/// 200 is the reference implementation's figure and the resolution is not +/// critical; what matters is that there *is* a histogram. See [`Pairs`]. +const BINS: usize = 200; + +/// Minimum evidence before a fit is trusted. +/// +/// Far stricter than the reference implementation's floor of two positives and +/// one negative. That floor is reasonable there: its pairs come from a curated +/// gallery of labelled reference portraits, where a positive pair is +/// trustworthy by construction. Here the positives are bootstrapped from bursts +/// and a handful of early confirmations, and the whole risk is fitting +/// confidently to too few of them. +pub const MIN_POSITIVE_PAIRS: u64 = 200; +pub const MIN_NEGATIVE_PAIRS: u64 = 2_000; + +/// A fitted `P(same person | cosine, face size)`. +/// +/// The single definition of what a similarity means in this subsystem. The +/// catalog stores its parameters; nothing re-implements the sigmoid. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct Calibration { + pub a: f32, + pub b: f32, + /// Weight on `log2(min crop_px)` — the face-size term FR-CULL-9 asks for. + pub w_size: f32, + /// Whether there was enough evidence to trust the fit. + /// + /// When false the UI says confidence is unavailable. It does **not** present + /// an untuned default as though it were measured, which is the distinction + /// FR-CULL-9 spends a paragraph on. + pub valid: bool, + pub positive_pairs: u64, + pub negative_pairs: u64, +} + +impl Default for Calibration { + /// The reference implementation's fitted MBF curve (docs/faces.md §1): + /// steepness 16.2, P=0.5 at cosine 0.267. + /// + /// **`valid` is false**, and that is the point. This exists so an + /// un-calibrated library has a documented operating point to cluster at + /// rather than no behaviour at all — but nothing may show its output as a + /// measured confidence. + fn default() -> Self { + Self { + a: 16.2, + b: -16.2 * 0.267, + w_size: 0.0, + valid: false, + positive_pairs: 0, + negative_pairs: 0, + } + } +} + +impl Calibration { + /// P(same person), shifted by a base-rate prior. + /// + /// `log_prior_odds` is applied at evaluation rather than folded into the + /// fit, so one stored calibration serves every context: the odds that two + /// faces in a 40-image album match are not the odds in a 40,000-image + /// archive. Folding a prior in would need a refit per context and would + /// make the stored parameters mean different things depending on where they + /// came from. + pub fn probability(&self, cosine: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 { + sigmoid(self.logit(cosine, min_crop_px) + log_prior_odds) + } + + fn logit(&self, cosine: f32, min_crop_px: f32) -> f32 { + self.a * cosine + self.b + self.w_size * min_crop_px.max(1.0).log2() + } + + /// The cosine at which [`Calibration::probability`] crosses `p`. + /// + /// What turns "merge above 0.9" into one comparison against a stored + /// similarity, rather than a sigmoid evaluated per candidate edge. + pub fn boundary_at(&self, p: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 { + ((p / (1.0 - p)).ln() + - self.b + - self.w_size * min_crop_px.max(1.0).log2() + - log_prior_odds) + / self.a + } +} + +fn sigmoid(z: f32) -> f32 { + // Branch on the sign so neither tail overflows: exp(-z) for large positive + // z, exp(z) for large negative. + if z >= 0.0 { + 1.0 / (1.0 + (-z).exp()) + } else { + let e = z.exp(); + e / (1.0 + e) + } +} + +/// Accumulated pair evidence, as a histogram rather than a list. +/// +/// # Why a histogram +/// +/// A 25,000-face library has ~3×10⁸ pairs and no gradient descent is running +/// over that. Bucketing them costs 200 counters per class and reduces the fit +/// to two parameters against per-bin totals; the expensive part becomes the +/// similarity matrix, which is one blocked GEMM. This is the trick that makes a +/// per-library fit affordable at all, and it is not obvious from outside. +#[derive(Debug, Clone)] +pub struct Pairs { + positive: Vec, + negative: Vec, + n_pos: u64, + n_neg: u64, +} + +impl Default for Pairs { + fn default() -> Self { + Self::new() + } +} + +impl Pairs { + pub fn new() -> Self { + Self { + positive: vec![0.0; BINS], + negative: vec![0.0; BINS], + n_pos: 0, + n_neg: 0, + } + } + + /// Record a pair known to be the same person. + pub fn push_positive(&mut self, cosine: f32) { + self.positive[bin(cosine)] += 1.0; + self.n_pos += 1; + } + + /// Record a pair known to be different people. + pub fn push_negative(&mut self, cosine: f32) { + self.negative[bin(cosine)] += 1.0; + self.n_neg += 1; + } + + pub fn positives(&self) -> u64 { + self.n_pos + } + pub fn negatives(&self) -> u64 { + self.n_neg + } + + /// Fit `P(same) = σ(a·cos + b)` by weighted logistic regression. + /// + /// Class weights are explicit because negatives outnumber positives by + /// orders of magnitude, and an unweighted fit produces a well-shaped curve + /// sitting at the wrong height — precisely the "plausible number all the + /// way to the user interface" failure FR-CULL-9 describes. + /// + /// Returns a calibration with `valid` set only if there was enough + /// evidence; the parameters are filled in either way so a caller with no + /// better option can still cluster at a documented operating point. + pub fn fit(&self) -> Calibration { + let base = Calibration { + positive_pairs: self.n_pos, + negative_pairs: self.n_neg, + ..Calibration::default() + }; + if self.n_pos < MIN_POSITIVE_PAIRS || self.n_neg < MIN_NEGATIVE_PAIRS { + return base; + } + + let total = self.n_pos as f64 + self.n_neg as f64; + let w_pos = total / (2.0 * self.n_pos as f64); + let w_neg = total / (2.0 * self.n_neg as f64); + + // Start from the reference's fitted MBF curve rather than from zero: + // it is the right order of magnitude for every model in this family, + // so descent converges in far fewer steps and cannot wander into a + // sign-flipped solution on thin evidence. + let mut a = base.a as f64; + let mut b = base.b as f64; + const LR: f64 = 0.05; + const MAX_ITER: usize = 20_000; + const TOL: f64 = 1e-7; + + for _ in 0..MAX_ITER { + let (mut da, mut db) = (0.0, 0.0); + for i in 0..BINS { + let x = bin_centre(i) as f64; + let s = 1.0 / (1.0 + (-(a * x + b)).exp()); + if self.positive[i] > 0.0 { + let e = (s - 1.0) * w_pos * self.positive[i]; + da += e * x; + db += e; + } + if self.negative[i] > 0.0 { + let e = s * w_neg * self.negative[i]; + da += e * x; + db += e; + } + } + da /= total; + db /= total; + a -= LR * da; + b -= LR * db; + if da * da + db * db < TOL * TOL { + break; + } + } + + Calibration { + a: a as f32, + b: b as f32, + w_size: 0.0, + valid: true, + positive_pairs: self.n_pos, + negative_pairs: self.n_neg, + } + } + + /// How well the fit predicts the evidence, as a reliability diagram. + /// + /// FR-CULL-9's acceptance criterion is exactly this and not a single + /// accuracy figure: for each populated probability band, the observed match + /// rate against the predicted one. Returned rather than asserted so the + /// caller can show it, log it, or fail a test on it. + pub fn reliability(&self, cal: &Calibration, bands: usize) -> Vec { + let mut out = vec![ + ReliabilityBand { + predicted: 0.0, + observed: 0.0, + count: 0 + }; + bands + ]; + let mut sum_pred = vec![0.0_f64; bands]; + for i in 0..BINS { + let n_pos = self.positive[i]; + let n_neg = self.negative[i]; + if n_pos + n_neg == 0.0 { + continue; + } + let p = cal.probability(bin_centre(i), 112.0, 0.0) as f64; + let band = ((p * bands as f64) as usize).min(bands - 1); + sum_pred[band] += p * (n_pos + n_neg); + out[band].observed += n_pos as f32; + out[band].count += (n_pos + n_neg) as u64; + } + for (band, o) in out.iter_mut().enumerate() { + if o.count > 0 { + o.predicted = (sum_pred[band] / o.count as f64) as f32; + o.observed /= o.count as f32; + } + } + out + } +} + +/// One row of a reliability diagram. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct ReliabilityBand { + /// Mean probability the calibration predicted for pairs in this band. + pub predicted: f32, + /// Fraction of them that were actually the same person. + pub observed: f32, + pub count: u64, +} + +fn bin(cosine: f32) -> usize { + let width = 2.0 / BINS as f32; + (((cosine + 1.0) / width) as usize).min(BINS - 1) +} + +fn bin_centre(i: usize) -> f32 { + let width = 2.0 / BINS as f32; + -1.0 + (i as f32 + 0.5) * width +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Synthesise pairs from two well-separated cosine distributions, the way + /// a real embedding space behaves: positives near 0.6, negatives near 0.05 + /// — the numbers our own end-to-end run actually produced. + fn realistic_pairs(n_pos: u64, n_neg: u64) -> Pairs { + let mut p = Pairs::new(); + let mut s = 12345_u32; + let mut rand = move || { + s = s.wrapping_mul(1_664_525).wrapping_add(1_013_904_223); + (s >> 8) as f32 / (1u32 << 24) as f32 + }; + for _ in 0..n_pos { + // ~N(0.60, 0.12), by summing uniforms. + let g = (0..4).map(|_| rand()).sum::() / 4.0 - 0.5; + p.push_positive((0.60 + g * 0.48).clamp(-1.0, 1.0)); + } + for _ in 0..n_neg { + let g = (0..4).map(|_| rand()).sum::() / 4.0 - 0.5; + p.push_negative((0.05 + g * 0.32).clamp(-1.0, 1.0)); + } + p + } + + #[test] + fn an_empty_library_has_no_valid_calibration() { + let cal = Pairs::new().fit(); + assert!(!cal.valid, "a fit with no evidence must not claim validity"); + assert_eq!(cal.positive_pairs, 0); + } + + /// The exact case FR-CULL-9 legislates: enough negatives, too few + /// positives. The answer is "unavailable", not a plausible-looking curve. + #[test] + fn too_few_positives_is_invalid_however_many_negatives_there_are() { + let p = realistic_pairs(MIN_POSITIVE_PAIRS - 1, MIN_NEGATIVE_PAIRS * 10); + assert!(!p.fit().valid); + } + + #[test] + fn too_few_negatives_is_invalid_too() { + let p = realistic_pairs(MIN_POSITIVE_PAIRS * 10, MIN_NEGATIVE_PAIRS - 1); + assert!(!p.fit().valid); + } + + #[test] + fn a_well_separated_library_fits_a_usable_curve() { + let cal = realistic_pairs(2_000, 40_000).fit(); + assert!(cal.valid); + assert!(cal.a > 0.0, "steepness must be positive: {}", cal.a); + + // The decision boundary lands between the two populations. + let boundary = cal.boundary_at(0.5, 112.0, 0.0); + assert!( + boundary > 0.05 && boundary < 0.60, + "boundary {boundary} is not between the negative and positive modes" + ); + + // And the measured cosines from the real end-to-end run fall the + // right side of it. + assert!(cal.probability(0.596, 200.0, 0.0) > 0.9); + assert!(cal.probability(0.050, 200.0, 0.0) < 0.1); + } + + #[test] + fn probability_and_boundary_are_inverses() { + let cal = realistic_pairs(2_000, 40_000).fit(); + for &p in &[0.1_f32, 0.5, 0.9, 0.99] { + let cos = cal.boundary_at(p, 112.0, 0.0); + assert!((cal.probability(cos, 112.0, 0.0) - p).abs() < 1e-3); + } + } + + /// A base rate shifts the answer without a refit — the property that lets + /// one stored calibration serve a small album and a large archive. + #[test] + fn a_prior_moves_the_boundary_in_the_right_direction() { + let cal = realistic_pairs(2_000, 40_000).fit(); + let neutral = cal.probability(0.4, 112.0, 0.0); + let pessimistic = cal.probability(0.4, 112.0, -2.0); + let optimistic = cal.probability(0.4, 112.0, 2.0); + assert!(pessimistic < neutral && neutral < optimistic); + } + + /// FR-CULL-9's acceptance criterion, run against the fit's own evidence: + /// in every populated band, the stated probability should track the + /// observed match rate. + #[test] + fn the_fit_is_reliable_on_the_evidence_it_was_fitted_to() { + let pairs = realistic_pairs(4_000, 40_000); + let cal = pairs.fit(); + let bands = pairs.reliability(&cal, 10); + + let mut checked = 0; + for b in &bands { + // Thinly populated bands are noise, not evidence. + if b.count < 200 { + continue; + } + checked += 1; + assert!( + (b.predicted - b.observed).abs() < 0.15, + "band predicted {:.3} but observed {:.3} over {} pairs", + b.predicted, + b.observed, + b.count + ); + } + assert!(checked >= 2, "only {checked} bands had enough pairs to check"); + } + + #[test] + fn the_default_curve_is_the_references_and_is_not_marked_valid() { + let cal = Calibration::default(); + assert!(!cal.valid); + assert!((cal.boundary_at(0.5, 112.0, 0.0) - 0.267).abs() < 1e-3); + } + + #[test] + fn bins_cover_the_cosine_range_without_overflowing() { + assert_eq!(bin(-1.0), 0); + assert_eq!(bin(1.0), BINS - 1); + assert_eq!(bin(2.0), BINS - 1, "an out-of-range cosine must not panic"); + assert!((bin_centre(bin(0.5)) - 0.5).abs() < 0.01); + } +} diff --git a/core/dr-face/src/cluster.rs b/core/dr-face/src/cluster.rs new file mode 100644 index 0000000..19d94db --- /dev/null +++ b/core/dr-face/src/cluster.rs @@ -0,0 +1,443 @@ +//! Grouping faces into people (docs/faces.md §9, FR-CULL-10). +//! +//! Model-free: this is arithmetic over embeddings, and it is where the +//! subsystem's accuracy actually lives, so it is testable with no weights on +//! the machine. +//! +//! # Constraints, not just a threshold +//! +//! FR-CULL-10 warns that clustering will over-merge on siblings, on parents and +//! children, and on the same person a decade apart. Two structural defences, +//! both cheaper than a better threshold: +//! +//! **Cannot-link on co-occurrence.** Two faces in the same photograph are never +//! merged. It is the same observation [`crate::calibrate`] mines for free +//! negatives, used here as a hard constraint, and it is the single cheapest +//! defence against over-merging that exists. +//! +//! **Confirmed faces are anchors.** A confirmation is user data (FR-CULL-12) +//! and clustering never moves it. Two groups holding confirmations of +//! *different* people cannot merge, whatever their similarity says. +//! +//! # Average link, not single link +//! +//! Single-link chains: one bad edge welds two identities together, and it is +//! the documented way face clustering fails on families. Average link asks +//! whether the *groups* are similar, which one outlier cannot force. + +use std::collections::HashSet; + +use crate::calibrate::Calibration; +use crate::embedding::EMBEDDING_DIM; + +/// Probability above which two groups are judged the same person. +/// +/// Stated as a probability and not a cosine, because FR-CULL-9 forbids +/// thresholding a bare similarity anywhere in this subsystem. +pub const DEFAULT_MERGE_PROBABILITY: f32 = 0.9; + +/// A face presented to the clusterer. +/// +/// Ids are opaque `u64`s rather than catalog types: this crate has no business +/// knowing what a `FaceId` means, and the caller does the translation. +#[derive(Debug, Clone)] +pub struct Candidate { + pub face: u64, + /// Which photograph it came from — the cannot-link key. + pub image: u64, + /// L2-normalised, `EMBEDDING_DIM` long. + pub embedding: Vec, + /// Source pixels across the aligned crop, for the calibration's size term. + pub crop_px: f32, + /// The person this face is *confirmed* to be, if any. + /// + /// Suggestions are deliberately not passed here. They are this function's + /// own previous output, and feeding them back in would let a guess harden + /// into a fact across successive passes. + pub confirmed_person: Option, +} + +/// One group of faces the clusterer believes are one person. +#[derive(Debug, Clone, PartialEq)] +pub struct Cluster { + /// Indices into the input slice. + pub members: Vec, + /// The person this group is already known to be, from its anchors. + /// + /// `Some` means the group contains confirmed faces and the suggestions in + /// it attach to that existing person. `None` is a new unnamed group. + pub person: Option, +} + +/// Group faces into people. +/// +/// `min_probability` is compared against the calibrated average-link +/// probability between two groups. Deterministic: the same input yields the +/// same clusters, because the merge order is by score with the index pair as +/// the tiebreak. +pub fn cluster( + faces: &[Candidate], + cal: &Calibration, + min_probability: f32, +) -> Vec { + if faces.is_empty() { + return Vec::new(); + } + + let n = faces.len(); + let mut groups: Vec = faces + .iter() + .enumerate() + .map(|(i, f)| Group { + members: vec![i], + images: HashSet::from([f.image]), + person: f.confirmed_person, + alive: true, + }) + .collect(); + + // Pairwise cosine once. n² f32 is the honest cost at library scale — for + // 25,000 faces that is the blocked GEMM docs/faces.md §9 describes, and the + // caller is expected to shard rather than this function growing an index. + let mut cos = vec![0.0_f32; n * n]; + for i in 0..n { + for j in i + 1..n { + let c = dot(&faces[i].embedding, &faces[j].embedding); + cos[i * n + j] = c; + cos[j * n + i] = c; + } + } + + loop { + let mut best: Option<(f32, usize, usize)> = None; + + for a in 0..n { + if !groups[a].alive { + continue; + } + for b in a + 1..n { + if !groups[b].alive || !can_link(&groups[a], &groups[b]) { + continue; + } + let p = average_link(&groups[a], &groups[b], faces, &cos, n, cal); + if p >= min_probability && best.is_none_or(|(bp, _, _)| p > bp) { + best = Some((p, a, b)); + } + } + } + + let Some((_, a, b)) = best else { break }; + let taken = std::mem::take(&mut groups[b]); + groups[b].alive = false; + let ga = &mut groups[a]; + ga.members.extend(taken.members); + ga.images.extend(taken.images); + // At most one side carries a person: `can_link` refuses a merge of two + // groups anchored to different people, so this cannot silently discard + // one of them. + ga.person = ga.person.or(taken.person); + } + + let mut out: Vec = groups + .into_iter() + .filter(|g| g.alive) + .map(|mut g| { + g.members.sort_unstable(); + Cluster { + members: g.members, + person: g.person, + } + }) + .collect(); + // Largest first: the People view shows the best-evidenced groups at the top. + out.sort_by(|x, y| y.members.len().cmp(&x.members.len()).then(x.members[0].cmp(&y.members[0]))); + out +} + +/// Split one person's faces into the groups a raised threshold separates them +/// into. +/// +/// FR-CULL-10 requires splitting to be as easy as merging, and a split that +/// hands the user a pile of loose faces to re-sort is not that. This re-runs +/// the same agglomeration at a stricter probability so the user is offered +/// coherent sub-groups to pull apart. +/// +/// Anchors are ignored here on purpose: every face in the input is already +/// nominally the same person, so honouring the anchors would refuse to split +/// anything. +pub fn split(faces: &[Candidate], cal: &Calibration, min_probability: f32) -> Vec { + let anchorless: Vec = faces + .iter() + .cloned() + .map(|mut f| { + f.confirmed_person = None; + f + }) + .collect(); + cluster(&anchorless, cal, min_probability) +} + +#[derive(Debug, Default)] +struct Group { + members: Vec, + images: HashSet, + person: Option, + alive: bool, +} + +/// Whether two groups are allowed to merge at all, before similarity is asked. +fn can_link(a: &Group, b: &Group) -> bool { + // Two confirmations of different people. The user has said these are not + // the same person, and no similarity overrides that. + if let (Some(pa), Some(pb)) = (a.person, b.person) { + if pa != pb { + return false; + } + } + // Co-occurrence: a photograph containing a face from each group means the + // two faces are in the same frame, so they are not the same person. + a.images.is_disjoint(&b.images) +} + +/// Mean calibrated probability over every cross-group pair. +fn average_link( + a: &Group, + b: &Group, + faces: &[Candidate], + cos: &[f32], + n: usize, + cal: &Calibration, +) -> f32 { + let mut sum = 0.0; + let mut count = 0.0; + for &i in &a.members { + for &j in &b.members { + let min_crop = faces[i].crop_px.min(faces[j].crop_px); + sum += cal.probability(cos[i * n + j], min_crop, 0.0); + count += 1.0; + } + } + if count == 0.0 { + 0.0 + } else { + sum / count + } +} + +fn dot(a: &[f32], b: &[f32]) -> f32 { + debug_assert_eq!(a.len(), EMBEDDING_DIM); + debug_assert_eq!(b.len(), EMBEDDING_DIM); + a.iter().zip(b).map(|(x, y)| x * y).sum() +} + +#[cfg(test)] +mod tests { + use super::*; + + /// An embedding a known cosine away from a base direction, built by mixing + /// two orthogonal unit vectors. Lets a test state "these two faces are 0.7 + /// similar" and have it be exactly true. + fn at_cosine(identity: usize, cosine: f32) -> Vec { + let mut v = vec![0.0_f32; EMBEDDING_DIM]; + let base = identity * 2; + let perp = identity * 2 + 1; + v[base] = cosine; + v[perp] = (1.0 - cosine * cosine).max(0.0).sqrt(); + v + } + + fn candidate(face: u64, image: u64, identity: usize, cosine: f32) -> Candidate { + Candidate { + face, + image, + embedding: at_cosine(identity, cosine), + crop_px: 150.0, + confirmed_person: None, + } + } + + /// A calibration steep enough that the test's cosines are unambiguous: + /// 0.6 is near-certain, 0.1 is near-impossible. + fn cal() -> Calibration { + Calibration { + a: 30.0, + b: -30.0 * 0.35, + w_size: 0.0, + valid: true, + positive_pairs: 1000, + negative_pairs: 10_000, + } + } + + #[test] + fn no_faces_makes_no_clusters() { + assert!(cluster(&[], &cal(), DEFAULT_MERGE_PROBABILITY).is_empty()); + } + + #[test] + fn similar_faces_from_different_photographs_group_together() { + let faces = vec![ + candidate(1, 10, 0, 1.0), + candidate(2, 11, 0, 0.95), + candidate(3, 12, 0, 0.92), + ]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 1); + assert_eq!(out[0].members, vec![0, 1, 2]); + } + + #[test] + fn dissimilar_faces_stay_apart() { + let faces = vec![candidate(1, 10, 0, 1.0), candidate(2, 11, 1, 1.0)]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 2); + } + + /// The cheapest defence against over-merging: two faces in one frame are + /// not the same person however similar the model finds them. + #[test] + fn two_faces_in_one_photograph_never_merge() { + // Identical embeddings — siblings, or a model that cannot tell them + // apart — but both in image 10. + let faces = vec![candidate(1, 10, 0, 1.0), candidate(2, 10, 0, 1.0)]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 2, "co-occurring faces were merged"); + } + + /// And the constraint has to survive transitively: once a group holds a + /// face from image 10, no other group holding one from image 10 may join + /// it, even indirectly. + #[test] + fn the_co_occurrence_constraint_propagates_through_a_group() { + let faces = vec![ + candidate(1, 10, 0, 1.0), // A, in the group photo + candidate(2, 10, 0, 1.0), // B, in the same group photo + candidate(3, 11, 0, 0.99), // A again, alone + ]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 2); + // Whichever of A/B absorbed face 2, the other stays out. + assert!(out.iter().any(|c| c.members.len() == 2)); + assert!(out.iter().any(|c| c.members.len() == 1)); + } + + /// FR-CULL-10: a confirmation is user data and no inference overrides it. + #[test] + fn groups_confirmed_as_different_people_do_not_merge() { + let mut a = candidate(1, 10, 0, 1.0); + let mut b = candidate(2, 11, 0, 1.0); + a.confirmed_person = Some(100); + b.confirmed_person = Some(200); + let out = cluster(&[a, b], &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 2, "clustering overrode two user confirmations"); + } + + #[test] + fn a_suggestion_joins_the_person_its_group_is_anchored_to() { + let mut anchor = candidate(1, 10, 0, 1.0); + anchor.confirmed_person = Some(42); + let loose = candidate(2, 11, 0, 0.95); + let out = cluster(&[anchor, loose], &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 1); + assert_eq!(out[0].person, Some(42)); + assert_eq!(out[0].members.len(), 2); + } + + #[test] + fn an_unanchored_group_is_a_new_unnamed_person() { + let out = cluster( + &[candidate(1, 10, 0, 1.0), candidate(2, 11, 0, 0.95)], + &cal(), + DEFAULT_MERGE_PROBABILITY, + ); + assert_eq!(out[0].person, None); + } + + /// Average link rather than single link: one strong edge must not weld two + /// otherwise-dissimilar groups together. This is the family failure mode + /// FR-CULL-10 names. + #[test] + fn one_strong_edge_does_not_chain_two_groups_together() { + // Two tight pairs, with a single borderline link between them. + let faces = vec![ + candidate(1, 10, 0, 1.00), + candidate(2, 11, 0, 0.99), + candidate(3, 12, 0, 0.42), + candidate(4, 13, 0, 0.40), + ]; + let out = cluster(&faces, &cal(), 0.99); + assert!( + out.len() >= 2, + "single-link chaining merged everything into {} cluster(s)", + out.len() + ); + } + + #[test] + fn clustering_is_deterministic() { + let faces = vec![ + candidate(1, 10, 0, 1.0), + candidate(2, 11, 0, 0.96), + candidate(3, 12, 1, 1.0), + candidate(4, 13, 1, 0.97), + candidate(5, 14, 0, 0.94), + ]; + let a = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + let b = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(a, b); + } + + #[test] + fn clusters_come_back_largest_first() { + let faces = vec![ + candidate(1, 10, 1, 1.0), + candidate(2, 11, 0, 1.0), + candidate(3, 12, 0, 0.97), + candidate(4, 13, 0, 0.95), + ]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out[0].members.len(), 3); + assert_eq!(out[1].members.len(), 1); + } + + /// Splitting is the inverse operation and must actually separate a group + /// that a looser threshold had merged. + #[test] + fn split_separates_a_group_that_a_looser_threshold_merged() { + let faces = vec![ + candidate(1, 10, 0, 1.00), + candidate(2, 11, 0, 0.98), + candidate(3, 12, 0, 0.45), + candidate(4, 13, 0, 0.43), + ]; + // Loose: one person. + assert_eq!(cluster(&faces, &cal(), 0.5).len(), 1); + // Strict: the two sub-groups the user wants offered. + let parts = split(&faces, &cal(), 0.999); + assert!(parts.len() >= 2, "split produced {} group(s)", parts.len()); + } + + #[test] + fn split_ignores_the_anchor_so_a_mislabelled_person_can_be_taken_apart() { + let mut a = candidate(1, 10, 0, 1.0); + let mut b = candidate(2, 11, 0, 0.40); + a.confirmed_person = Some(7); + b.confirmed_person = Some(7); + let parts = split(&[a, b], &cal(), 0.99); + assert_eq!(parts.len(), 2); + } + + /// The size term earns its place: the same cosine between two thumbnail- + /// sized faces should be less convincing than between two large ones. + #[test] + fn the_face_size_term_moves_the_probability() { + let sized = Calibration { + w_size: 0.5, + b: -30.0 * 0.35 - 0.5 * 7.0, + ..cal() + }; + let big = sized.probability(0.5, 300.0, 0.0); + let small = sized.probability(0.5, 40.0, 0.0); + assert!(big > small, "big {big} should beat small {small}"); + } +} diff --git a/core/dr-face/src/lib.rs b/core/dr-face/src/lib.rs index 3436a88..bebb338 100644 --- a/core/dr-face/src/lib.rs +++ b/core/dr-face/src/lib.rs @@ -31,6 +31,8 @@ //! likely to be subtly wrong. pub mod align; +pub mod calibrate; +pub mod cluster; pub mod embedding; #[cfg(feature = "inference")] pub mod detect; @@ -38,6 +40,8 @@ pub mod detect; pub mod embed; pub use align::{warp, Aligned112, Similarity, ALIGNED_EDGE, ARCFACE_TEMPLATE}; +pub use calibrate::{Calibration, Pairs, ReliabilityBand}; +pub use cluster::{cluster, split, Candidate, Cluster, DEFAULT_MERGE_PROBABILITY}; pub use embedding::{Embedding, ModelId, EMBEDDING_DIM}; #[cfg(feature = "inference")] pub use detect::{DetectOptions, Detection, Detector}; diff --git a/docs/faces.md b/docs/faces.md index eca7b42..5ed7e7f 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -760,6 +760,29 @@ scalar letterbox resample in front of it. Steps 1–5 are the spike. Steps 6–8 are the build, and they are only justified if §12 says so. +### 13.1 Where this had got to · 2026-08-26 + +- **1 — done.** M1 passes conditionally; §12.1. +- **2 — outstanding.** §2.3's three questions are unanswered and gate *shipping*, not building. +- **3 — done.** `core/dr-face`: `align`, `detect`, `embed`, `embedding`, and `examples/faces.rs`. + §12.2 is its first end-to-end run. +- **4 — partly.** M2/M3 have provisional numbers; M4/M5 need the labelled corpus. +- **5 — built, not yet measured.** `calibrate` and `cluster` exist with 34 model-free tests behind + them; M6–M8 are a run over a real library, which is what the corpus in §1.1's `images/` is for. +- **6 — done for storage.** Catalog schema v8 — `people`, `faces`, `face_person`, + `face_person_rejected`, `face_calibration` — with the identity operations FR-CULL-10 requires. + The `DetectFaces` job kind and the debounced clustering pass are not wired yet. +- **7, 8 — not started.** The People/Identity screen, and the route-C first-run flow. + +Two things the build changed in this document's own design: + +**`crop_px` reached the schema** as §7 argued it should, and the calibration carries a `w_size` term +for it. + +**A rejection table was added**, which §8 and §9 did not contemplate. Rejection is not the absence of +an assignment: without storing it, the next clustering pass re-suggests exactly the face the user +just pushed away. It is user data in the same sense a confirmation is (FR-CULL-12), just negative. + --- ## 14. What this does not settle diff --git a/docs/traceability.md b/docs/traceability.md index c5a123f..518215b 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -9,7 +9,7 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n | Metric | Value | |---|---| -| Source files scanned | 235 | +| Source files scanned | 237 | | TRACES tags found | 623 | | Requirements defined | 177 | | Requirements covered | 97 |