diff --git a/Cargo.lock b/Cargo.lock index 5ddd6ca..b235857 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1406,6 +1406,7 @@ checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76" name = "dr-catalog" version = "0.7.0" dependencies = [ + "dr-face", "dr-plat", "dr-types", "env_logger", diff --git a/core/dr-catalog/Cargo.toml b/core/dr-catalog/Cargo.toml index 6be0996..984aa5c 100644 --- a/core/dr-catalog/Cargo.toml +++ b/core/dr-catalog/Cargo.toml @@ -7,6 +7,11 @@ license.workspace = true [dependencies] dr-types.workspace = true +# The face subsystem's arithmetic — `Calibration` in particular, so the sigmoid +# that turns a cosine into a probability has exactly one definition. Default +# features are off, so this brings in no ONNX runtime and no weights: only the +# model-free half compiles here. +dr-face.workspace = true # The `Storage` trait, and nothing else from it. A scan has to read a real # directory, and this is how `core/` reaches the platform without a # `#[cfg(target_os)]` of its own (ARCH §4.1: calls go downward). diff --git a/core/dr-catalog/src/faces.rs b/core/dr-catalog/src/faces.rs index 58c4d61..9fefcf1 100644 --- a/core/dr-catalog/src/faces.rs +++ b/core/dr-catalog/src/faces.rs @@ -103,55 +103,14 @@ pub struct Person { pub suggested_faces: u64, } -/// The FR-CULL-9 calibration for one model. -#[derive(Debug, Clone, Copy, PartialEq)] -pub struct Calibration { - pub a: f32, - pub b: f32, - pub w_size: f32, - /// Whether the fit is usable. When false the UI says confidence is - /// unavailable rather than presenting an untuned default as a measurement. - pub valid: bool, - pub positive_pairs: u64, - pub negative_pairs: u64, -} - -impl Calibration { - /// P(same person), optionally shifted by a base-rate prior. - /// - /// `log_prior_odds` is applied at evaluation rather than baked into the - /// fit, so one stored calibration serves every context: the odds that two - /// faces in a 40-image album match are not the odds in a 40,000-image - /// archive. Baking a prior in would need a refit per context and would make - /// the stored parameters mean different things depending on where they came - /// from. - pub fn probability(&self, cosine: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 { - let z = self.a * cosine - + self.b - + self.w_size * min_crop_px.max(1.0).log2() - + log_prior_odds; - // Branch on the sign so neither tail overflows: exp(-z) for large - // positive z, exp(z) for large negative. - if z >= 0.0 { - 1.0 / (1.0 + (-z).exp()) - } else { - let e = z.exp(); - e / (1.0 + e) - } - } - - /// The cosine at which [`Calibration::probability`] crosses `p`. - /// - /// What turns "merge above 0.9" into an actual comparison against a stored - /// similarity, without evaluating the sigmoid per candidate edge. - pub fn boundary_at(&self, p: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 { - ((p / (1.0 - p)).ln() - - self.b - - self.w_size * min_crop_px.max(1.0).log2() - - log_prior_odds) - / self.a - } -} +/// The FR-CULL-9 calibration, re-exported from where it is fitted. +/// +/// Deliberately *not* redefined here. The catalog stores five numbers; what +/// those numbers mean — the sigmoid, the size term, the prior — is +/// [`dr_face::calibrate`]'s business, and two implementations of one +/// probability model is precisely the kind of divergence that yields a +/// plausible number meaning the wrong thing. +pub use dr_face::Calibration; // ── detection ───────────────────────────────────────────────────────────── @@ -278,6 +237,12 @@ pub fn unassigned(conn: &Connection, model_id: &str) -> Result, Cata rows.collect::>().map_err(Into::into) } +/// One face's stored embedding, as the clustering pass consumes it. +/// +/// A named type rather than a tuple because it crosses a crate boundary and +/// "the third element" is not a thing anyone should have to remember. +pub type StoredEmbedding = (FaceId, ImageId, Vec, f32); + /// Embeddings for clustering, oldest first so the pass is deterministic. /// /// Returned as raw f16 blobs rather than decoded vectors: the caller is @@ -286,7 +251,7 @@ pub fn unassigned(conn: &Connection, model_id: &str) -> Result, Cata pub fn embeddings( conn: &Connection, model_id: &str, -) -> Result, f32)>, CatalogError> { +) -> Result, CatalogError> { let mut q = conn.prepare( "SELECT id, image_id, embedding, crop_px FROM faces WHERE model_id = ?1 ORDER BY id", diff --git a/core/dr-face/src/align.rs b/core/dr-face/src/align.rs index 7757d36..2d5f18f 100644 --- a/core/dr-face/src/align.rs +++ b/core/dr-face/src/align.rs @@ -332,7 +332,7 @@ mod tests { let mut rgb = vec![0.0_f32; w * h * 3]; for y in 0..h { for x in 0..w { - if x >= 56 && x < 168 && y >= 56 && y < 168 { + if (56..168).contains(&x) && (56..168).contains(&y) { for c in 0..3 { rgb[(y * w + x) * 3 + c] = 1.0; } diff --git a/core/dr-face/src/calibrate.rs b/core/dr-face/src/calibrate.rs new file mode 100644 index 0000000..1884a90 --- /dev/null +++ b/core/dr-face/src/calibrate.rs @@ -0,0 +1,436 @@ +//! Cosine to probability (docs/faces.md §8, FR-CULL-9). +//! +//! FR-CULL-9 is a hard requirement rather than an implementation detail: no +//! code path may threshold a bare cosine, every threshold in the subsystem is +//! stated as a probability, and the fit is per library and reports its own +//! validity. The failure it guards against is invisible — a raw cosine means +//! something different for every model, every population and every face size, +//! and an uncalibrated similarity still *looks* like a plausible number all the +//! way to the user interface. +//! +//! Model-free, so the part of this subsystem most likely to be subtly wrong is +//! testable on synthetic embeddings with no weights on the machine. +//! +//! # Where the training pairs come from +//! +//! **Negatives are free and abundant.** Two faces detected in *the same +//! photograph* are almost never the same person, which hands every multi-face +//! image in the library a full set of negative pairs at no labelling cost — and +//! they are *hard* negatives, from the same camera, lighting and processing, +//! which is exactly the population where a threshold tuned on easy negatives +//! fails. The exceptions (mirrors, photographs of photographs, collages) are +//! rare enough to be noise at this scale. +//! +//! **Positives have to be earned.** In order of trustworthiness: pairs the user +//! has confirmed onto one person; then burst siblings, since FR-CULL-5 already +//! groups bursts and two faces in adjacent frames are near-certainly the same +//! person. Nothing else — bootstrapping positives from high cosine is circular, +//! fitting the calibration to the belief it was supposed to test. +//! +//! Which is why a fresh library has **no valid calibration**, and says so. + +/// Bins over cosine ∈ [-1, 1]. +/// +/// 200 is the reference implementation's figure and the resolution is not +/// critical; what matters is that there *is* a histogram. See [`Pairs`]. +const BINS: usize = 200; + +/// Minimum evidence before a fit is trusted. +/// +/// Far stricter than the reference implementation's floor of two positives and +/// one negative. That floor is reasonable there: its pairs come from a curated +/// gallery of labelled reference portraits, where a positive pair is +/// trustworthy by construction. Here the positives are bootstrapped from bursts +/// and a handful of early confirmations, and the whole risk is fitting +/// confidently to too few of them. +pub const MIN_POSITIVE_PAIRS: u64 = 200; +pub const MIN_NEGATIVE_PAIRS: u64 = 2_000; + +/// A fitted `P(same person | cosine, face size)`. +/// +/// The single definition of what a similarity means in this subsystem. The +/// catalog stores its parameters; nothing re-implements the sigmoid. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct Calibration { + pub a: f32, + pub b: f32, + /// Weight on `log2(min crop_px)` — the face-size term FR-CULL-9 asks for. + pub w_size: f32, + /// Whether there was enough evidence to trust the fit. + /// + /// When false the UI says confidence is unavailable. It does **not** present + /// an untuned default as though it were measured, which is the distinction + /// FR-CULL-9 spends a paragraph on. + pub valid: bool, + pub positive_pairs: u64, + pub negative_pairs: u64, +} + +impl Default for Calibration { + /// The reference implementation's fitted MBF curve (docs/faces.md §1): + /// steepness 16.2, P=0.5 at cosine 0.267. + /// + /// **`valid` is false**, and that is the point. This exists so an + /// un-calibrated library has a documented operating point to cluster at + /// rather than no behaviour at all — but nothing may show its output as a + /// measured confidence. + fn default() -> Self { + Self { + a: 16.2, + b: -16.2 * 0.267, + w_size: 0.0, + valid: false, + positive_pairs: 0, + negative_pairs: 0, + } + } +} + +impl Calibration { + /// P(same person), shifted by a base-rate prior. + /// + /// `log_prior_odds` is applied at evaluation rather than folded into the + /// fit, so one stored calibration serves every context: the odds that two + /// faces in a 40-image album match are not the odds in a 40,000-image + /// archive. Folding a prior in would need a refit per context and would + /// make the stored parameters mean different things depending on where they + /// came from. + pub fn probability(&self, cosine: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 { + sigmoid(self.logit(cosine, min_crop_px) + log_prior_odds) + } + + fn logit(&self, cosine: f32, min_crop_px: f32) -> f32 { + self.a * cosine + self.b + self.w_size * min_crop_px.max(1.0).log2() + } + + /// The cosine at which [`Calibration::probability`] crosses `p`. + /// + /// What turns "merge above 0.9" into one comparison against a stored + /// similarity, rather than a sigmoid evaluated per candidate edge. + pub fn boundary_at(&self, p: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 { + ((p / (1.0 - p)).ln() + - self.b + - self.w_size * min_crop_px.max(1.0).log2() + - log_prior_odds) + / self.a + } +} + +fn sigmoid(z: f32) -> f32 { + // Branch on the sign so neither tail overflows: exp(-z) for large positive + // z, exp(z) for large negative. + if z >= 0.0 { + 1.0 / (1.0 + (-z).exp()) + } else { + let e = z.exp(); + e / (1.0 + e) + } +} + +/// Accumulated pair evidence, as a histogram rather than a list. +/// +/// # Why a histogram +/// +/// A 25,000-face library has ~3×10⁸ pairs and no gradient descent is running +/// over that. Bucketing them costs 200 counters per class and reduces the fit +/// to two parameters against per-bin totals; the expensive part becomes the +/// similarity matrix, which is one blocked GEMM. This is the trick that makes a +/// per-library fit affordable at all, and it is not obvious from outside. +#[derive(Debug, Clone)] +pub struct Pairs { + positive: Vec, + negative: Vec, + n_pos: u64, + n_neg: u64, +} + +impl Default for Pairs { + fn default() -> Self { + Self::new() + } +} + +impl Pairs { + pub fn new() -> Self { + Self { + positive: vec![0.0; BINS], + negative: vec![0.0; BINS], + n_pos: 0, + n_neg: 0, + } + } + + /// Record a pair known to be the same person. + pub fn push_positive(&mut self, cosine: f32) { + self.positive[bin(cosine)] += 1.0; + self.n_pos += 1; + } + + /// Record a pair known to be different people. + pub fn push_negative(&mut self, cosine: f32) { + self.negative[bin(cosine)] += 1.0; + self.n_neg += 1; + } + + pub fn positives(&self) -> u64 { + self.n_pos + } + pub fn negatives(&self) -> u64 { + self.n_neg + } + + /// Fit `P(same) = σ(a·cos + b)` by weighted logistic regression. + /// + /// Class weights are explicit because negatives outnumber positives by + /// orders of magnitude, and an unweighted fit produces a well-shaped curve + /// sitting at the wrong height — precisely the "plausible number all the + /// way to the user interface" failure FR-CULL-9 describes. + /// + /// Returns a calibration with `valid` set only if there was enough + /// evidence; the parameters are filled in either way so a caller with no + /// better option can still cluster at a documented operating point. + pub fn fit(&self) -> Calibration { + let base = Calibration { + positive_pairs: self.n_pos, + negative_pairs: self.n_neg, + ..Calibration::default() + }; + if self.n_pos < MIN_POSITIVE_PAIRS || self.n_neg < MIN_NEGATIVE_PAIRS { + return base; + } + + let total = self.n_pos as f64 + self.n_neg as f64; + let w_pos = total / (2.0 * self.n_pos as f64); + let w_neg = total / (2.0 * self.n_neg as f64); + + // Start from the reference's fitted MBF curve rather than from zero: + // it is the right order of magnitude for every model in this family, + // so descent converges in far fewer steps and cannot wander into a + // sign-flipped solution on thin evidence. + let mut a = base.a as f64; + let mut b = base.b as f64; + const LR: f64 = 0.05; + const MAX_ITER: usize = 20_000; + const TOL: f64 = 1e-7; + + for _ in 0..MAX_ITER { + let (mut da, mut db) = (0.0, 0.0); + for i in 0..BINS { + let x = bin_centre(i) as f64; + let s = 1.0 / (1.0 + (-(a * x + b)).exp()); + if self.positive[i] > 0.0 { + let e = (s - 1.0) * w_pos * self.positive[i]; + da += e * x; + db += e; + } + if self.negative[i] > 0.0 { + let e = s * w_neg * self.negative[i]; + da += e * x; + db += e; + } + } + da /= total; + db /= total; + a -= LR * da; + b -= LR * db; + if da * da + db * db < TOL * TOL { + break; + } + } + + Calibration { + a: a as f32, + b: b as f32, + w_size: 0.0, + valid: true, + positive_pairs: self.n_pos, + negative_pairs: self.n_neg, + } + } + + /// How well the fit predicts the evidence, as a reliability diagram. + /// + /// FR-CULL-9's acceptance criterion is exactly this and not a single + /// accuracy figure: for each populated probability band, the observed match + /// rate against the predicted one. Returned rather than asserted so the + /// caller can show it, log it, or fail a test on it. + pub fn reliability(&self, cal: &Calibration, bands: usize) -> Vec { + let mut out = vec![ + ReliabilityBand { + predicted: 0.0, + observed: 0.0, + count: 0 + }; + bands + ]; + let mut sum_pred = vec![0.0_f64; bands]; + for i in 0..BINS { + let n_pos = self.positive[i]; + let n_neg = self.negative[i]; + if n_pos + n_neg == 0.0 { + continue; + } + let p = cal.probability(bin_centre(i), 112.0, 0.0) as f64; + let band = ((p * bands as f64) as usize).min(bands - 1); + sum_pred[band] += p * (n_pos + n_neg); + out[band].observed += n_pos as f32; + out[band].count += (n_pos + n_neg) as u64; + } + for (band, o) in out.iter_mut().enumerate() { + if o.count > 0 { + o.predicted = (sum_pred[band] / o.count as f64) as f32; + o.observed /= o.count as f32; + } + } + out + } +} + +/// One row of a reliability diagram. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct ReliabilityBand { + /// Mean probability the calibration predicted for pairs in this band. + pub predicted: f32, + /// Fraction of them that were actually the same person. + pub observed: f32, + pub count: u64, +} + +fn bin(cosine: f32) -> usize { + let width = 2.0 / BINS as f32; + (((cosine + 1.0) / width) as usize).min(BINS - 1) +} + +fn bin_centre(i: usize) -> f32 { + let width = 2.0 / BINS as f32; + -1.0 + (i as f32 + 0.5) * width +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Synthesise pairs from two well-separated cosine distributions, the way + /// a real embedding space behaves: positives near 0.6, negatives near 0.05 + /// — the numbers our own end-to-end run actually produced. + fn realistic_pairs(n_pos: u64, n_neg: u64) -> Pairs { + let mut p = Pairs::new(); + let mut s = 12345_u32; + let mut rand = move || { + s = s.wrapping_mul(1_664_525).wrapping_add(1_013_904_223); + (s >> 8) as f32 / (1u32 << 24) as f32 + }; + for _ in 0..n_pos { + // ~N(0.60, 0.12), by summing uniforms. + let g = (0..4).map(|_| rand()).sum::() / 4.0 - 0.5; + p.push_positive((0.60 + g * 0.48).clamp(-1.0, 1.0)); + } + for _ in 0..n_neg { + let g = (0..4).map(|_| rand()).sum::() / 4.0 - 0.5; + p.push_negative((0.05 + g * 0.32).clamp(-1.0, 1.0)); + } + p + } + + #[test] + fn an_empty_library_has_no_valid_calibration() { + let cal = Pairs::new().fit(); + assert!(!cal.valid, "a fit with no evidence must not claim validity"); + assert_eq!(cal.positive_pairs, 0); + } + + /// The exact case FR-CULL-9 legislates: enough negatives, too few + /// positives. The answer is "unavailable", not a plausible-looking curve. + #[test] + fn too_few_positives_is_invalid_however_many_negatives_there_are() { + let p = realistic_pairs(MIN_POSITIVE_PAIRS - 1, MIN_NEGATIVE_PAIRS * 10); + assert!(!p.fit().valid); + } + + #[test] + fn too_few_negatives_is_invalid_too() { + let p = realistic_pairs(MIN_POSITIVE_PAIRS * 10, MIN_NEGATIVE_PAIRS - 1); + assert!(!p.fit().valid); + } + + #[test] + fn a_well_separated_library_fits_a_usable_curve() { + let cal = realistic_pairs(2_000, 40_000).fit(); + assert!(cal.valid); + assert!(cal.a > 0.0, "steepness must be positive: {}", cal.a); + + // The decision boundary lands between the two populations. + let boundary = cal.boundary_at(0.5, 112.0, 0.0); + assert!( + boundary > 0.05 && boundary < 0.60, + "boundary {boundary} is not between the negative and positive modes" + ); + + // And the measured cosines from the real end-to-end run fall the + // right side of it. + assert!(cal.probability(0.596, 200.0, 0.0) > 0.9); + assert!(cal.probability(0.050, 200.0, 0.0) < 0.1); + } + + #[test] + fn probability_and_boundary_are_inverses() { + let cal = realistic_pairs(2_000, 40_000).fit(); + for &p in &[0.1_f32, 0.5, 0.9, 0.99] { + let cos = cal.boundary_at(p, 112.0, 0.0); + assert!((cal.probability(cos, 112.0, 0.0) - p).abs() < 1e-3); + } + } + + /// A base rate shifts the answer without a refit — the property that lets + /// one stored calibration serve a small album and a large archive. + #[test] + fn a_prior_moves_the_boundary_in_the_right_direction() { + let cal = realistic_pairs(2_000, 40_000).fit(); + let neutral = cal.probability(0.4, 112.0, 0.0); + let pessimistic = cal.probability(0.4, 112.0, -2.0); + let optimistic = cal.probability(0.4, 112.0, 2.0); + assert!(pessimistic < neutral && neutral < optimistic); + } + + /// FR-CULL-9's acceptance criterion, run against the fit's own evidence: + /// in every populated band, the stated probability should track the + /// observed match rate. + #[test] + fn the_fit_is_reliable_on_the_evidence_it_was_fitted_to() { + let pairs = realistic_pairs(4_000, 40_000); + let cal = pairs.fit(); + let bands = pairs.reliability(&cal, 10); + + let mut checked = 0; + for b in &bands { + // Thinly populated bands are noise, not evidence. + if b.count < 200 { + continue; + } + checked += 1; + assert!( + (b.predicted - b.observed).abs() < 0.15, + "band predicted {:.3} but observed {:.3} over {} pairs", + b.predicted, + b.observed, + b.count + ); + } + assert!(checked >= 2, "only {checked} bands had enough pairs to check"); + } + + #[test] + fn the_default_curve_is_the_references_and_is_not_marked_valid() { + let cal = Calibration::default(); + assert!(!cal.valid); + assert!((cal.boundary_at(0.5, 112.0, 0.0) - 0.267).abs() < 1e-3); + } + + #[test] + fn bins_cover_the_cosine_range_without_overflowing() { + assert_eq!(bin(-1.0), 0); + assert_eq!(bin(1.0), BINS - 1); + assert_eq!(bin(2.0), BINS - 1, "an out-of-range cosine must not panic"); + assert!((bin_centre(bin(0.5)) - 0.5).abs() < 0.01); + } +} diff --git a/core/dr-face/src/cluster.rs b/core/dr-face/src/cluster.rs new file mode 100644 index 0000000..19d94db --- /dev/null +++ b/core/dr-face/src/cluster.rs @@ -0,0 +1,443 @@ +//! Grouping faces into people (docs/faces.md §9, FR-CULL-10). +//! +//! Model-free: this is arithmetic over embeddings, and it is where the +//! subsystem's accuracy actually lives, so it is testable with no weights on +//! the machine. +//! +//! # Constraints, not just a threshold +//! +//! FR-CULL-10 warns that clustering will over-merge on siblings, on parents and +//! children, and on the same person a decade apart. Two structural defences, +//! both cheaper than a better threshold: +//! +//! **Cannot-link on co-occurrence.** Two faces in the same photograph are never +//! merged. It is the same observation [`crate::calibrate`] mines for free +//! negatives, used here as a hard constraint, and it is the single cheapest +//! defence against over-merging that exists. +//! +//! **Confirmed faces are anchors.** A confirmation is user data (FR-CULL-12) +//! and clustering never moves it. Two groups holding confirmations of +//! *different* people cannot merge, whatever their similarity says. +//! +//! # Average link, not single link +//! +//! Single-link chains: one bad edge welds two identities together, and it is +//! the documented way face clustering fails on families. Average link asks +//! whether the *groups* are similar, which one outlier cannot force. + +use std::collections::HashSet; + +use crate::calibrate::Calibration; +use crate::embedding::EMBEDDING_DIM; + +/// Probability above which two groups are judged the same person. +/// +/// Stated as a probability and not a cosine, because FR-CULL-9 forbids +/// thresholding a bare similarity anywhere in this subsystem. +pub const DEFAULT_MERGE_PROBABILITY: f32 = 0.9; + +/// A face presented to the clusterer. +/// +/// Ids are opaque `u64`s rather than catalog types: this crate has no business +/// knowing what a `FaceId` means, and the caller does the translation. +#[derive(Debug, Clone)] +pub struct Candidate { + pub face: u64, + /// Which photograph it came from — the cannot-link key. + pub image: u64, + /// L2-normalised, `EMBEDDING_DIM` long. + pub embedding: Vec, + /// Source pixels across the aligned crop, for the calibration's size term. + pub crop_px: f32, + /// The person this face is *confirmed* to be, if any. + /// + /// Suggestions are deliberately not passed here. They are this function's + /// own previous output, and feeding them back in would let a guess harden + /// into a fact across successive passes. + pub confirmed_person: Option, +} + +/// One group of faces the clusterer believes are one person. +#[derive(Debug, Clone, PartialEq)] +pub struct Cluster { + /// Indices into the input slice. + pub members: Vec, + /// The person this group is already known to be, from its anchors. + /// + /// `Some` means the group contains confirmed faces and the suggestions in + /// it attach to that existing person. `None` is a new unnamed group. + pub person: Option, +} + +/// Group faces into people. +/// +/// `min_probability` is compared against the calibrated average-link +/// probability between two groups. Deterministic: the same input yields the +/// same clusters, because the merge order is by score with the index pair as +/// the tiebreak. +pub fn cluster( + faces: &[Candidate], + cal: &Calibration, + min_probability: f32, +) -> Vec { + if faces.is_empty() { + return Vec::new(); + } + + let n = faces.len(); + let mut groups: Vec = faces + .iter() + .enumerate() + .map(|(i, f)| Group { + members: vec![i], + images: HashSet::from([f.image]), + person: f.confirmed_person, + alive: true, + }) + .collect(); + + // Pairwise cosine once. n² f32 is the honest cost at library scale — for + // 25,000 faces that is the blocked GEMM docs/faces.md §9 describes, and the + // caller is expected to shard rather than this function growing an index. + let mut cos = vec![0.0_f32; n * n]; + for i in 0..n { + for j in i + 1..n { + let c = dot(&faces[i].embedding, &faces[j].embedding); + cos[i * n + j] = c; + cos[j * n + i] = c; + } + } + + loop { + let mut best: Option<(f32, usize, usize)> = None; + + for a in 0..n { + if !groups[a].alive { + continue; + } + for b in a + 1..n { + if !groups[b].alive || !can_link(&groups[a], &groups[b]) { + continue; + } + let p = average_link(&groups[a], &groups[b], faces, &cos, n, cal); + if p >= min_probability && best.is_none_or(|(bp, _, _)| p > bp) { + best = Some((p, a, b)); + } + } + } + + let Some((_, a, b)) = best else { break }; + let taken = std::mem::take(&mut groups[b]); + groups[b].alive = false; + let ga = &mut groups[a]; + ga.members.extend(taken.members); + ga.images.extend(taken.images); + // At most one side carries a person: `can_link` refuses a merge of two + // groups anchored to different people, so this cannot silently discard + // one of them. + ga.person = ga.person.or(taken.person); + } + + let mut out: Vec = groups + .into_iter() + .filter(|g| g.alive) + .map(|mut g| { + g.members.sort_unstable(); + Cluster { + members: g.members, + person: g.person, + } + }) + .collect(); + // Largest first: the People view shows the best-evidenced groups at the top. + out.sort_by(|x, y| y.members.len().cmp(&x.members.len()).then(x.members[0].cmp(&y.members[0]))); + out +} + +/// Split one person's faces into the groups a raised threshold separates them +/// into. +/// +/// FR-CULL-10 requires splitting to be as easy as merging, and a split that +/// hands the user a pile of loose faces to re-sort is not that. This re-runs +/// the same agglomeration at a stricter probability so the user is offered +/// coherent sub-groups to pull apart. +/// +/// Anchors are ignored here on purpose: every face in the input is already +/// nominally the same person, so honouring the anchors would refuse to split +/// anything. +pub fn split(faces: &[Candidate], cal: &Calibration, min_probability: f32) -> Vec { + let anchorless: Vec = faces + .iter() + .cloned() + .map(|mut f| { + f.confirmed_person = None; + f + }) + .collect(); + cluster(&anchorless, cal, min_probability) +} + +#[derive(Debug, Default)] +struct Group { + members: Vec, + images: HashSet, + person: Option, + alive: bool, +} + +/// Whether two groups are allowed to merge at all, before similarity is asked. +fn can_link(a: &Group, b: &Group) -> bool { + // Two confirmations of different people. The user has said these are not + // the same person, and no similarity overrides that. + if let (Some(pa), Some(pb)) = (a.person, b.person) { + if pa != pb { + return false; + } + } + // Co-occurrence: a photograph containing a face from each group means the + // two faces are in the same frame, so they are not the same person. + a.images.is_disjoint(&b.images) +} + +/// Mean calibrated probability over every cross-group pair. +fn average_link( + a: &Group, + b: &Group, + faces: &[Candidate], + cos: &[f32], + n: usize, + cal: &Calibration, +) -> f32 { + let mut sum = 0.0; + let mut count = 0.0; + for &i in &a.members { + for &j in &b.members { + let min_crop = faces[i].crop_px.min(faces[j].crop_px); + sum += cal.probability(cos[i * n + j], min_crop, 0.0); + count += 1.0; + } + } + if count == 0.0 { + 0.0 + } else { + sum / count + } +} + +fn dot(a: &[f32], b: &[f32]) -> f32 { + debug_assert_eq!(a.len(), EMBEDDING_DIM); + debug_assert_eq!(b.len(), EMBEDDING_DIM); + a.iter().zip(b).map(|(x, y)| x * y).sum() +} + +#[cfg(test)] +mod tests { + use super::*; + + /// An embedding a known cosine away from a base direction, built by mixing + /// two orthogonal unit vectors. Lets a test state "these two faces are 0.7 + /// similar" and have it be exactly true. + fn at_cosine(identity: usize, cosine: f32) -> Vec { + let mut v = vec![0.0_f32; EMBEDDING_DIM]; + let base = identity * 2; + let perp = identity * 2 + 1; + v[base] = cosine; + v[perp] = (1.0 - cosine * cosine).max(0.0).sqrt(); + v + } + + fn candidate(face: u64, image: u64, identity: usize, cosine: f32) -> Candidate { + Candidate { + face, + image, + embedding: at_cosine(identity, cosine), + crop_px: 150.0, + confirmed_person: None, + } + } + + /// A calibration steep enough that the test's cosines are unambiguous: + /// 0.6 is near-certain, 0.1 is near-impossible. + fn cal() -> Calibration { + Calibration { + a: 30.0, + b: -30.0 * 0.35, + w_size: 0.0, + valid: true, + positive_pairs: 1000, + negative_pairs: 10_000, + } + } + + #[test] + fn no_faces_makes_no_clusters() { + assert!(cluster(&[], &cal(), DEFAULT_MERGE_PROBABILITY).is_empty()); + } + + #[test] + fn similar_faces_from_different_photographs_group_together() { + let faces = vec![ + candidate(1, 10, 0, 1.0), + candidate(2, 11, 0, 0.95), + candidate(3, 12, 0, 0.92), + ]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 1); + assert_eq!(out[0].members, vec![0, 1, 2]); + } + + #[test] + fn dissimilar_faces_stay_apart() { + let faces = vec![candidate(1, 10, 0, 1.0), candidate(2, 11, 1, 1.0)]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 2); + } + + /// The cheapest defence against over-merging: two faces in one frame are + /// not the same person however similar the model finds them. + #[test] + fn two_faces_in_one_photograph_never_merge() { + // Identical embeddings — siblings, or a model that cannot tell them + // apart — but both in image 10. + let faces = vec![candidate(1, 10, 0, 1.0), candidate(2, 10, 0, 1.0)]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 2, "co-occurring faces were merged"); + } + + /// And the constraint has to survive transitively: once a group holds a + /// face from image 10, no other group holding one from image 10 may join + /// it, even indirectly. + #[test] + fn the_co_occurrence_constraint_propagates_through_a_group() { + let faces = vec![ + candidate(1, 10, 0, 1.0), // A, in the group photo + candidate(2, 10, 0, 1.0), // B, in the same group photo + candidate(3, 11, 0, 0.99), // A again, alone + ]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 2); + // Whichever of A/B absorbed face 2, the other stays out. + assert!(out.iter().any(|c| c.members.len() == 2)); + assert!(out.iter().any(|c| c.members.len() == 1)); + } + + /// FR-CULL-10: a confirmation is user data and no inference overrides it. + #[test] + fn groups_confirmed_as_different_people_do_not_merge() { + let mut a = candidate(1, 10, 0, 1.0); + let mut b = candidate(2, 11, 0, 1.0); + a.confirmed_person = Some(100); + b.confirmed_person = Some(200); + let out = cluster(&[a, b], &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 2, "clustering overrode two user confirmations"); + } + + #[test] + fn a_suggestion_joins_the_person_its_group_is_anchored_to() { + let mut anchor = candidate(1, 10, 0, 1.0); + anchor.confirmed_person = Some(42); + let loose = candidate(2, 11, 0, 0.95); + let out = cluster(&[anchor, loose], &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out.len(), 1); + assert_eq!(out[0].person, Some(42)); + assert_eq!(out[0].members.len(), 2); + } + + #[test] + fn an_unanchored_group_is_a_new_unnamed_person() { + let out = cluster( + &[candidate(1, 10, 0, 1.0), candidate(2, 11, 0, 0.95)], + &cal(), + DEFAULT_MERGE_PROBABILITY, + ); + assert_eq!(out[0].person, None); + } + + /// Average link rather than single link: one strong edge must not weld two + /// otherwise-dissimilar groups together. This is the family failure mode + /// FR-CULL-10 names. + #[test] + fn one_strong_edge_does_not_chain_two_groups_together() { + // Two tight pairs, with a single borderline link between them. + let faces = vec![ + candidate(1, 10, 0, 1.00), + candidate(2, 11, 0, 0.99), + candidate(3, 12, 0, 0.42), + candidate(4, 13, 0, 0.40), + ]; + let out = cluster(&faces, &cal(), 0.99); + assert!( + out.len() >= 2, + "single-link chaining merged everything into {} cluster(s)", + out.len() + ); + } + + #[test] + fn clustering_is_deterministic() { + let faces = vec![ + candidate(1, 10, 0, 1.0), + candidate(2, 11, 0, 0.96), + candidate(3, 12, 1, 1.0), + candidate(4, 13, 1, 0.97), + candidate(5, 14, 0, 0.94), + ]; + let a = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + let b = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(a, b); + } + + #[test] + fn clusters_come_back_largest_first() { + let faces = vec![ + candidate(1, 10, 1, 1.0), + candidate(2, 11, 0, 1.0), + candidate(3, 12, 0, 0.97), + candidate(4, 13, 0, 0.95), + ]; + let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(out[0].members.len(), 3); + assert_eq!(out[1].members.len(), 1); + } + + /// Splitting is the inverse operation and must actually separate a group + /// that a looser threshold had merged. + #[test] + fn split_separates_a_group_that_a_looser_threshold_merged() { + let faces = vec![ + candidate(1, 10, 0, 1.00), + candidate(2, 11, 0, 0.98), + candidate(3, 12, 0, 0.45), + candidate(4, 13, 0, 0.43), + ]; + // Loose: one person. + assert_eq!(cluster(&faces, &cal(), 0.5).len(), 1); + // Strict: the two sub-groups the user wants offered. + let parts = split(&faces, &cal(), 0.999); + assert!(parts.len() >= 2, "split produced {} group(s)", parts.len()); + } + + #[test] + fn split_ignores_the_anchor_so_a_mislabelled_person_can_be_taken_apart() { + let mut a = candidate(1, 10, 0, 1.0); + let mut b = candidate(2, 11, 0, 0.40); + a.confirmed_person = Some(7); + b.confirmed_person = Some(7); + let parts = split(&[a, b], &cal(), 0.99); + assert_eq!(parts.len(), 2); + } + + /// The size term earns its place: the same cosine between two thumbnail- + /// sized faces should be less convincing than between two large ones. + #[test] + fn the_face_size_term_moves_the_probability() { + let sized = Calibration { + w_size: 0.5, + b: -30.0 * 0.35 - 0.5 * 7.0, + ..cal() + }; + let big = sized.probability(0.5, 300.0, 0.0); + let small = sized.probability(0.5, 40.0, 0.0); + assert!(big > small, "big {big} should beat small {small}"); + } +} diff --git a/core/dr-face/src/lib.rs b/core/dr-face/src/lib.rs index 3436a88..bebb338 100644 --- a/core/dr-face/src/lib.rs +++ b/core/dr-face/src/lib.rs @@ -31,6 +31,8 @@ //! likely to be subtly wrong. pub mod align; +pub mod calibrate; +pub mod cluster; pub mod embedding; #[cfg(feature = "inference")] pub mod detect; @@ -38,6 +40,8 @@ pub mod detect; pub mod embed; pub use align::{warp, Aligned112, Similarity, ALIGNED_EDGE, ARCFACE_TEMPLATE}; +pub use calibrate::{Calibration, Pairs, ReliabilityBand}; +pub use cluster::{cluster, split, Candidate, Cluster, DEFAULT_MERGE_PROBABILITY}; pub use embedding::{Embedding, ModelId, EMBEDDING_DIM}; #[cfg(feature = "inference")] pub use detect::{DetectOptions, Detection, Detector}; diff --git a/docs/faces.md b/docs/faces.md index eca7b42..5ed7e7f 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -760,6 +760,29 @@ scalar letterbox resample in front of it. Steps 1–5 are the spike. Steps 6–8 are the build, and they are only justified if §12 says so. +### 13.1 Where this had got to · 2026-08-26 + +- **1 — done.** M1 passes conditionally; §12.1. +- **2 — outstanding.** §2.3's three questions are unanswered and gate *shipping*, not building. +- **3 — done.** `core/dr-face`: `align`, `detect`, `embed`, `embedding`, and `examples/faces.rs`. + §12.2 is its first end-to-end run. +- **4 — partly.** M2/M3 have provisional numbers; M4/M5 need the labelled corpus. +- **5 — built, not yet measured.** `calibrate` and `cluster` exist with 34 model-free tests behind + them; M6–M8 are a run over a real library, which is what the corpus in §1.1's `images/` is for. +- **6 — done for storage.** Catalog schema v8 — `people`, `faces`, `face_person`, + `face_person_rejected`, `face_calibration` — with the identity operations FR-CULL-10 requires. + The `DetectFaces` job kind and the debounced clustering pass are not wired yet. +- **7, 8 — not started.** The People/Identity screen, and the route-C first-run flow. + +Two things the build changed in this document's own design: + +**`crop_px` reached the schema** as §7 argued it should, and the calibration carries a `w_size` term +for it. + +**A rejection table was added**, which §8 and §9 did not contemplate. Rejection is not the absence of +an assignment: without storing it, the next clustering pass re-suggests exactly the face the user +just pushed away. It is user data in the same sense a confirmation is (FR-CULL-12), just negative. + --- ## 14. What this does not settle diff --git a/docs/traceability.md b/docs/traceability.md index c5a123f..518215b 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -9,7 +9,7 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n | Metric | Value | |---|---| -| Source files scanned | 235 | +| Source files scanned | 237 | | TRACES tags found | 623 | | Requirements defined | 177 | | Requirements covered | 97 |