diff --git a/core/dr-catalog/examples/face_confidence.rs b/core/dr-catalog/examples/face_confidence.rs new file mode 100644 index 0000000..e2d633c --- /dev/null +++ b/core/dr-catalog/examples/face_confidence.rs @@ -0,0 +1,373 @@ +//! What the suggestion confidence would say about a real library. +//! +//! cargo run --release -p dr-catalog --example face_confidence -- CATALOG.sqlite [--full] +//! +//! Read-only: it writes nothing to the catalog, so it can be pointed at a copy +//! of a live library and re-run at will. +//! +//! # What it measures +//! +//! The user's own confirmations are the only ground truth a library has, so +//! the evaluation is leave-one-out over them: hide one confirmed face, ask the +//! scorer which of the confirmed identities it belongs to, and compare with +//! what the user said. Faces from the same photograph are excluded exactly as +//! the clusterer excludes them, so nothing is scored against a co-occurrence +//! that would never have been allowed to merge. +//! +//! Two numbers are compared on that task: the share (`dr_face::assign`) and the +//! mean-within-group figure it replaced. Accuracy says which one picks the +//! right person; the reliability table says whether the percentage the user is +//! shown means what it claims — which is the question FR-CULL-9 exists for. + +use std::collections::HashMap; + +use dr_catalog::faces::{self, PersonId}; +use dr_catalog::Catalog; + +const MODEL_ID: &str = "w600k_mbf"; +const TOP: usize = 10; + +struct Known { + image: u64, + person: PersonId, + embedding: Vec, + crop_px: f32, +} + +fn main() { + let args: Vec = std::env::args().skip(1).collect(); + let Some(path) = args.first() else { + eprintln!("usage: face_confidence CATALOG.sqlite [--full]"); + std::process::exit(2); + }; + + let catalog = Catalog::open(std::path::Path::new(path)).expect("open catalog"); + let conn = catalog.connection(); + + let cal = match faces::calibration(conn, MODEL_ID) { + Ok(Some((c, _))) => c, + _ => dr_face::Calibration::default(), + }; + println!( + "calibration: a={:.2} b={:.2} w_size={:.3} valid={} (P=0.5 at cosine {:.3})", + cal.a, + cal.b, + cal.w_size, + cal.valid, + cal.boundary_at(0.5, 150.0, 0.0) + ); + + let model = dr_face::ModelId::new(MODEL_ID.to_string()); + let stored = faces::embeddings(conn, MODEL_ID).expect("embeddings"); + let mut embedding_of = HashMap::new(); + for (id, image, blob, crop_px) in stored { + if let Some(e) = dr_face::Embedding::from_f16_bytes(model.clone(), &blob) { + embedding_of.insert(id, (image.0, e.v.to_vec(), crop_px)); + } + } + println!("faces with embeddings: {}", embedding_of.len()); + + // The ground truth: every confirmed face, under the person the user put it + // on. Identities with a single confirmation are dropped — leaving one out + // leaves that identity with no evidence at all, so they measure nothing. + let people = faces::people(conn).expect("people"); + let mut known: Vec = Vec::new(); + let mut identities = 0usize; + for p in &people { + if p.confirmed_faces < 2 { + continue; + } + let mut mine = Vec::new(); + for f in faces::for_person(conn, p.id, false).expect("faces") { + if !f.confirmed { + continue; + } + if let Some((image, embedding, crop_px)) = embedding_of.get(&f.id) { + mine.push(Known { + image: *image, + person: p.id, + embedding: embedding.clone(), + crop_px: *crop_px, + }); + } + } + if mine.len() >= 2 { + identities += 1; + known.extend(mine); + } + } + println!( + "ground truth: {} confirmed faces across {identities} identities\n", + known.len() + ); + if known.len() < 2 { + println!("not enough confirmations to evaluate."); + return; + } + + let mut share_right = 0usize; + let mut mean_right = 0usize; + // (share of the winner, was the winner correct) + let mut reliability: Vec<(f32, bool)> = Vec::with_capacity(known.len()); + // What each scorer would have *displayed* for the correct answer. + let mut shown_share = Vec::with_capacity(known.len()); + let mut shown_mean = Vec::with_capacity(known.len()); + + for (i, me) in known.iter().enumerate() { + let mut per_person: HashMap> = HashMap::new(); + // The mean baseline is the old code's, which had no floor: it averaged + // over every member of the group. + let mut all_person: HashMap> = HashMap::new(); + for (j, them) in known.iter().enumerate() { + if i == j || me.image == them.image { + continue; + } + let cos: f32 = me + .embedding + .iter() + .zip(&them.embedding) + .map(|(a, b)| a * b) + .sum(); + // The floor the real scorer sees: `cluster_scored` scans at + // `RIVAL_FLOOR` and `identity_shares` never learns about a pair + // below it. Summing the near-orthogonal ones here instead of + // dropping them is not a stricter test, it is a different + // function — fifty identities contributing their *upper tail* of + // noise outweigh one contributing a real match. + let probability = cal.probability(cos, me.crop_px.min(them.crop_px), 0.0); + if probability >= dr_face::RIVAL_FLOOR { + per_person.entry(them.person).or_default().push(probability); + } + all_person.entry(them.person).or_default().push(probability); + } + + // The share: sum of the best TOP matches per identity, normalised. + // Evidence, and the coherence that goes with it: the sum of the best + // TOP matches, and their mean. dr_face::assign shows the product of + // that mean and the identity's share of the total. + let mut evidence: Vec<(PersonId, f32, f32)> = per_person + .iter() + .map(|(&p, probabilities)| { + let mut v = probabilities.clone(); + v.sort_by(|a, b| b.total_cmp(a)); + let counted = v.len().min(TOP); + let sum = v.iter().take(TOP).sum::(); + (p, sum, sum / counted as f32) + }) + .collect(); + let total: f32 = evidence.iter().map(|(_, s, _)| *s).sum(); + evidence.sort_by(|a, b| b.1.total_cmp(&a.1)); + + // The number it replaced: the mean over every member of the identity. + let mut means: Vec<(PersonId, f32)> = all_person + .iter() + .map(|(&p, probabilities)| { + ( + p, + probabilities.iter().sum::() / probabilities.len() as f32, + ) + }) + .collect(); + means.sort_by(|a, b| b.1.total_cmp(&a.1)); + + if let (Some(&(winner, score, coherence)), true) = (evidence.first(), total > 0.0) { + let correct = winner == me.person; + share_right += correct as usize; + // What the screen would say about the identity it picked. + reliability.push((coherence * score / total, correct)); + let ours = evidence + .iter() + .find(|(p, _, _)| *p == me.person) + .map(|(_, s, c)| c * s / total) + .unwrap_or(0.0); + shown_share.push(ours); + } + if let Some(&(winner, _)) = means.first() { + mean_right += (winner == me.person) as usize; + shown_mean.push( + means + .iter() + .find(|(p, _)| *p == me.person) + .map(|(_, s)| *s) + .unwrap_or(0.0), + ); + } + } + + let n = known.len() as f64; + println!("which identity does this face belong to? (leave-one-out, top-1)"); + println!( + " share of evidence {:>6.2}% ({share_right}/{})", + 100.0 * share_right as f64 / n, + known.len() + ); + println!( + " mean within group {:>6.2}% ({mean_right}/{})\n", + 100.0 * mean_right as f64 / n, + known.len() + ); + + println!("what the screen would show for the answer the user gave:"); + band(" share ", &shown_share); + band(" mean ", &shown_mean); + + println!("\nreliability of the share — is a stated {{n}}% right {{n}}% of the time?"); + println!( + " {:>12} {:>7} {:>9} {:>8}", + "stated", "faces", "correct", "gap" + ); + for (lo, hi) in [ + (0.0, 0.5), + (0.5, 0.6), + (0.6, 0.7), + (0.7, 0.8), + (0.8, 0.9), + (0.9, 0.95), + (0.95, 1.001), + ] { + let bucket: Vec = reliability + .iter() + .filter(|(s, _)| *s >= lo && *s < hi) + .map(|(_, c)| *c) + .collect(); + if bucket.is_empty() { + continue; + } + let observed = bucket.iter().filter(|c| **c).count() as f64 / bucket.len() as f64; + let stated = reliability + .iter() + .filter(|(s, _)| *s >= lo && *s < hi) + .map(|(s, _)| *s as f64) + .sum::() + / bucket.len() as f64; + println!( + " {:>5.0}–{:>3.0}% {:>9} {:>8.1}% {:>+7.1}", + lo * 100.0, + hi.min(1.0) * 100.0, + bucket.len(), + 100.0 * observed, + 100.0 * (observed - stated) + ); + } + + if args.iter().any(|a| a == "--full") { + // The confirmations go in as anchors, exactly as `recluster` sends + // them: they are what makes a group a named identity, and therefore + // what makes it a rival. + let mut confirmed = HashMap::new(); + for p in &people { + for f in faces::for_person(conn, p.id, false).expect("faces") { + if f.confirmed { + confirmed.insert(f.id, p.id.0); + } + } + } + full_library(&embedding_of, &confirmed, &cal); + } +} + +/// Where a set of confidences actually falls. +fn band(label: &str, v: &[f32]) { + if v.is_empty() { + return; + } + let mut s = v.to_vec(); + s.sort_by(|a, b| a.total_cmp(b)); + let pct = |q: f64| s[((s.len() - 1) as f64 * q) as usize]; + let mean = s.iter().sum::() / s.len() as f32; + println!( + "{label} median {:>5.1}% mean {:>5.1}% p10 {:>5.1}% p90 {:>5.1}% under 50%: {:>5.1}%", + 100.0 * pct(0.5), + 100.0 * mean, + 100.0 * pct(0.10), + 100.0 * pct(0.90), + 100.0 * s.iter().filter(|x| **x < 0.5).count() as f32 / s.len() as f32 + ); +} + +/// The whole library through the real clusterer, for the numbers it would +/// actually write. +fn full_library( + embedding_of: &HashMap, f32)>, + confirmed: &HashMap, + cal: &dr_face::Calibration, +) { + let mut candidates: Vec = embedding_of + .iter() + .map(|(id, (image, embedding, crop_px))| dr_face::Candidate { + face: id.0, + image: *image, + embedding: embedding.clone(), + crop_px: *crop_px, + confirmed_person: confirmed.get(id).copied(), + }) + .collect(); + candidates.sort_by_key(|c| c.face); + + println!("\nthe whole library, at the default merge probability:"); + let start = std::time::Instant::now(); + let grouping = dr_face::cluster_scored(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY); + let real: Vec<_> = grouping + .clusters + .iter() + .filter(|c| c.members.len() >= 2) + .collect(); + let grouped: usize = real.iter().map(|c| c.members.len()).sum(); + println!( + " {} face(s) → {} group(s) of two or more, holding {grouped} faces ({:.0}%), in {:.1}s", + candidates.len(), + real.len(), + 100.0 * grouped as f64 / candidates.len() as f64, + start.elapsed().as_secs_f64() + ); + + let named: Vec<_> = real.iter().filter(|c| c.person.is_some()).collect(); + println!( + " {} of those group(s) carry a confirmation, holding {} faces", + named.len(), + named.iter().map(|c| c.members.len()).sum::() + ); + + let shown: Vec = real + .iter() + .flat_map(|c| c.members.iter().map(|&m| grouping.confidence[m])) + .collect(); + band(" new, all groups ", &shown); + let onto_people: Vec = named + .iter() + .flat_map(|c| c.members.iter().map(|&m| grouping.confidence[m])) + .collect(); + band(" new, onto a person", &onto_people); + + // The number the old code would have written for the same grouping. + let means: Vec = real + .iter() + .flat_map(|c| { + c.members.iter().map(|&m| { + let me = &candidates[m]; + let mut sum = 0.0; + let mut n = 0.0; + for &other in &c.members { + if other == m { + continue; + } + let them = &candidates[other]; + let cos: f32 = me + .embedding + .iter() + .zip(&them.embedding) + .map(|(a, b)| a * b) + .sum(); + sum += cal.probability(cos, me.crop_px.min(them.crop_px), 0.0); + n += 1.0; + } + if n == 0.0 { + 1.0 + } else { + sum / n + } + }) + }) + .collect(); + band(" old, all groups ", &means); +} diff --git a/core/dr-face/src/assign.rs b/core/dr-face/src/assign.rs new file mode 100644 index 0000000..66a7236 --- /dev/null +++ b/core/dr-face/src/assign.rs @@ -0,0 +1,360 @@ +//! TRACES: FR-CULL-9 | FR-CULL-10 +//! How sure a *suggestion* is: an identity's share of the evidence for a face. +//! +//! [`crate::cluster`] decides which people exist; this decides what number to +//! put beside "we think this is Anna". They are not the same question, and the +//! answer to the second used to be a by-product of the first — the mean +//! calibrated probability between a face and *every* other member of its group. +//! +//! # Why a mean over the group is the wrong number +//! +//! It measures the wrong thing twice over. +//! +//! **It punishes large, well-photographed people.** Anna has two hundred faces +//! spanning fifteen years; a new photograph of her matches thirty of them +//! strongly and is near-orthogonal to the rest, because a face at 8 and a face +//! at 23 genuinely are. The mean lands around 0.2 and the interface reports a +//! correct suggestion as a doubtful one. The better a person is covered, the +//! worse their confidences get, which is exactly backwards. +//! +//! **It never asks who else it could be.** A face that matches Anna at 0.95 and +//! matches nobody else at all, and a face that matches Anna at 0.95 *and her +//! sister at 0.93*, are the same number under a within-group mean. The second +//! is the one the user actually needs to look at, and it was indistinguishable +//! from the first. +//! +//! # Coherence, times uniqueness +//! +//! Two questions, and the number is their product because they are genuinely +//! independent: *is this the same person at all*, and *of the people we know, +//! is it uniquely this one*. +//! +//! ```text +//! evidence(P) = Σ of the top n of { P(same | this face, f) : f ∈ P } +//! coherence = evidence(own) / (however many of the top n there were) +//! uniqueness = evidence(own) / (evidence(own) + Σ evidence(named rivals)) +//! confidence = coherence × uniqueness +//! ``` +//! +//! **Coherence** is a mean, like the old number, but over the face's best +//! [`TOP_MATCHES`] matches into the identity rather than over all of them. That single cap is what +//! stops a well-photographed person scoring worse than a thin one: the two +//! hundred faces a given photograph is legitimately orthogonal to no longer +//! count against it. +//! +//! **Uniqueness** is the competition. A sole strong match leaves it at 1 and +//! the confidence is the coherence; two identities matching equally well pull +//! it to 0.5 each, and the screen has told the user the truth, which is that +//! this face is a coin toss between two people. +//! +//! # Only the people the user has named compete +//! +//! Measured on a real 18,000-face library, normalising across *every* group +//! made the number useless: the median suggestion read 21% and four in five +//! read under half. The cause is not a bug in the arithmetic but a fact about +//! clustering — one person is spread across many groups, since the pairs that +//! would have joined them are the ones that fell short of the merge threshold. +//! Normalising over groups therefore makes a face compete against *itself*, +//! and the better covered the person, the more fragments there are to lose to. +//! +//! A fragment is not a rival. An identity the user has actually asserted is, so +//! the denominator counts only people a confirmation names ([`Cluster::person`]) +//! — and counts them **per person, not per group**, since a named person is +//! left in several anchored groups for the same reason. Keying it by group had +//! Catherine competing with Catherine and put the median suggestion onto a +//! named person at 39%; keying it by person put it at 99.5%. +//! +//! Leave-one-out over that library's 2,702 confirmations across 54 named +//! people, this is the regime where the number is worth having: 99.3% of faces +//! are placed on the right person (the old mean managed 99.15%), the stated +//! percentage is monotone in being right, and it errs low — 100% correct +//! wherever it states 80% or more, 84% correct where it states under half. +//! Understating is the safe direction for a screen whose whole purpose is +//! deciding what to look at first, but it is *not* calibrated in the low bands +//! and should not be read as though it were. +//! +//! Two faces of one unnamed group cannot be told from two fragments of one +//! person by similarity alone; that is exactly why clustering stopped where it +//! did. So the module does not pretend to: where nobody is named, uniqueness is +//! 1 and the number falls back to plain coherence. +//! +//! # What it does not do +//! +//! It is not a merge threshold and must not become one. Clustering keeps +//! deciding on the pairwise calibrated probability: uniqueness is *relative*, +//! so a library with one named person in it would hand every stray face a +//! uniqueness of 1. The absolute question ("is this the same person at all") +//! and the comparative one ("of the people we know, which") are different, and +//! the product is what keeps both in the answer. + +use std::collections::BTreeMap; + +use crate::cluster::Cluster; +use crate::neighbours::Pair; + +/// Who the evidence is for. +/// +/// Ordered rather than hashed so the sums below are reproducible; the ordering +/// itself carries no meaning. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +enum Identity { + /// A person the user has confirmed a face onto. Every group anchored to + /// them is the same identity, however many of them the clusterer left. + Person(u64), + /// A group nobody has ruled on. It stands for itself and competes with + /// nothing. + Group(usize), +} + +/// How many of an identity's best matches count as its evidence. +/// +/// The cap is the whole reason the sum works: uncapped, evidence would grow +/// with a person's face count and the largest group in the library would win +/// every contest. Ten is enough that a person photographed from several angles +/// contributes more than one lucky frame, and small enough that the hundred +/// mediocre matches inside a well-covered identity cannot add up to a strong +/// one. It is a starting point, not a measured optimum — M7's corpus is where +/// it would be tuned. +pub const TOP_MATCHES: usize = 10; + +/// The weakest match that counts as evidence for an identity. +/// +/// Rivals are half the point of this module, so the evidence scan has to reach +/// *below* the merge threshold — a named person who matches at 0.6 will never +/// be merged into but is precisely the competition a suggestion should be +/// discounted for. Even odds is the natural floor: below it a pair is more +/// likely different people than the same, and it is not a small distinction — +/// summing the near-orthogonal pairs instead of dropping them lets fifty +/// identities' worth of noise, each contributing its *upper tail*, outweigh one +/// real match. Measured on a real library, that alone moved the median stated +/// confidence from 100% to 31%. +pub const RIVAL_FLOOR: f32 = 0.5; + +/// How confident each face's placement is, indexed like the face slice the +/// clusters came from. +/// +/// A face in no group, or one with no evidence for anybody, scores 0. +/// +/// `pairs` must be the *evidence* list — scanned at [`RIVAL_FLOOR`], not at the +/// merge threshold. Passing the merge list still works but silently removes +/// every rival weaker than a merge, which is most of them, and every uniqueness +/// collapses to 1. +pub fn identity_shares(faces: usize, clusters: &[Cluster], pairs: &[Pair], top: usize) -> Vec { + // An identity is a *person*, not a group. One person routinely holds + // several anchored groups — the same reason they hold several unnamed ones + // — and keying this by group had Catherine competing with Catherine, which + // on the library it was measured against put the median suggestion onto a + // named person at 39%. + let key_of: Vec = clusters + .iter() + .enumerate() + .map(|(g, c)| match c.person { + Some(p) => Identity::Person(p), + None => Identity::Group(g), + }) + .collect(); + + let mut group_of = vec![usize::MAX; faces]; + for (g, c) in clusters.iter().enumerate() { + for &m in &c.members { + if m < faces { + group_of[m] = g; + } + } + } + + // Ordered, not hashed: the numbers are sums of floats over these buckets + // and this module inherits [`crate::cluster`]'s promise that the same input + // yields the same output, bit for bit. + let mut evidence: Vec>> = vec![BTreeMap::new(); faces]; + for p in pairs { + if p.i >= faces || p.j >= faces { + continue; + } + // A pair is evidence in both directions: j's identity hears about i, + // and i's identity hears about j. The pair list holds each unordered + // pair once, so both have to be recorded here. + let (gi, gj) = (group_of[p.i], group_of[p.j]); + if gj != usize::MAX { + evidence[p.i] + .entry(key_of[gj]) + .or_default() + .push(p.probability); + } + if gi != usize::MAX { + evidence[p.j] + .entry(key_of[gi]) + .or_default() + .push(p.probability); + } + } + + let mut out = vec![0.0; faces]; + for (i, buckets) in evidence.iter_mut().enumerate() { + let mine = group_of[i]; + if mine == usize::MAX { + continue; + } + let mine = key_of[mine]; + let mut coherence = 0.0; + let mut ours = 0.0; + let mut rivals = 0.0; + for (&who, probabilities) in buckets.iter_mut() { + // Descending, and the ties broken by nothing: equal probabilities + // sum the same whichever order they land in. + probabilities.sort_by(|a, b| b.total_cmp(a)); + let counted = probabilities.len().min(top); + let score: f32 = probabilities.iter().take(top).sum(); + if who == mine { + ours = score; + coherence = score / counted as f32; + } else if matches!(who, Identity::Person(_)) { + // Only an identity the user has asserted competes. An unnamed + // group that matches this face is far more likely to be another + // fragment of the same person than a different one — see the + // module note, and the library it was measured on. + rivals += score; + } + } + let total = ours + rivals; + // No evidence at all: a face anchored into a group it has no measured + // similarity to. Nothing honest to report, so nothing is claimed. + out[i] = if total > 0.0 { + coherence * (ours / total) + } else { + 0.0 + }; + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + /// An unnamed group: nobody has ruled on it, so it competes with nothing. + fn cluster(members: &[usize]) -> Cluster { + Cluster { + members: members.to_vec(), + person: None, + } + } + + /// A group the user has confirmed a face onto — an identity, and therefore + /// a rival. + fn named(members: &[usize], person: u64) -> Cluster { + Cluster { + members: members.to_vec(), + person: Some(person), + } + } + + fn pair(i: usize, j: usize, probability: f32) -> Pair { + Pair { i, j, probability } + } + + /// The failure the module exists to fix: face 0 matches its own group's + /// three members strongly, and the group has forty more it is unrelated to. + /// The old within-group mean reported ~0.07 for this. + #[test] + fn a_large_group_does_not_dilute_a_strong_match() { + let members: Vec = (0..44).collect(); + let clusters = vec![cluster(&members)]; + let pairs = vec![pair(0, 1, 0.99), pair(0, 2, 0.97), pair(0, 3, 0.95)]; + + let shares = identity_shares(44, &clusters, &pairs, TOP_MATCHES); + assert!( + (shares[0] - 0.97).abs() < 1e-6, + "the mean of its three real matches, undiluted: {}", + shares[0] + ); + } + + /// Two named people matching equally well is a coin toss, and saying so is + /// the point — this is the sibling case FR-CULL-10 warns about. + #[test] + fn an_ambiguous_face_splits_its_confidence_between_the_rivals() { + let clusters = vec![named(&[0, 1, 2], 1), named(&[3, 4], 2)]; + let pairs = vec![ + pair(0, 1, 0.90), + pair(0, 2, 0.90), + pair(0, 3, 0.90), + pair(0, 4, 0.90), + ]; + + let shares = identity_shares(5, &clusters, &pairs, TOP_MATCHES); + // Coherent at 0.90, and only half of the evidence is its own. + assert!( + (shares[0] - 0.45).abs() < 1e-6, + "even evidence both ways: {}", + shares[0] + ); + } + + /// A rival below the merge threshold still has to count, which is why the + /// evidence scan reaches down to [`RIVAL_FLOOR`]. + #[test] + fn a_rival_too_weak_to_merge_still_lowers_the_confidence() { + let clusters = vec![named(&[0, 1], 1), named(&[2, 3], 2)]; + let sure = identity_shares(4, &clusters, &[pair(0, 1, 0.95)], TOP_MATCHES); + let contested = identity_shares( + 4, + &clusters, + &[pair(0, 1, 0.95), pair(0, 2, 0.60)], + TOP_MATCHES, + ); + + assert_eq!(sure[0], 0.95, "nobody else to be: its coherence stands"); + assert!( + contested[0] < 0.59 && contested[0] > 0.57, + "0.95 coherent, but 0.95 against 0.60: {}", + contested[0] + ); + } + + /// A fragment of the same person is not a rival. Measured on a real + /// library, counting unnamed groups as competition put four suggestions in + /// five under half — see the module note. + #[test] + fn an_unnamed_group_is_not_treated_as_competition() { + let clusters = vec![cluster(&[0, 1]), cluster(&[2, 3])]; + let shares = identity_shares( + 4, + &clusters, + &[pair(0, 1, 0.95), pair(0, 2, 0.90)], + TOP_MATCHES, + ); + assert_eq!( + shares[0], 0.95, + "an unnamed group took evidence off a suggestion" + ); + } + + /// The cap, doing its job: an identity with fifty mediocre matches must not + /// beat one with ten strong ones on volume alone. + #[test] + fn evidence_is_capped_so_the_biggest_group_cannot_win_on_volume() { + let small: Vec = (0..11).collect(); + let large: Vec = (11..62).collect(); + let clusters = vec![named(&small, 1), named(&large, 2)]; + + let mut pairs: Vec = (1..11).map(|j| pair(0, j, 0.90)).collect(); + pairs.extend((11..62).map(|j| pair(0, j, 0.55))); + + let shares = identity_shares(62, &clusters, &pairs, TOP_MATCHES); + // Ten at 0.90 against ten at 0.55 — not fifty-one at 0.55. + assert!( + (shares[0] - 0.90 * (9.0 / 14.5)).abs() < 1e-5, + "capped at ten either side: {}", + shares[0] + ); + } + + /// A face nothing has any evidence about claims nothing. + #[test] + fn a_face_with_no_evidence_reports_no_confidence() { + let clusters = vec![cluster(&[0, 1])]; + let shares = identity_shares(2, &clusters, &[], TOP_MATCHES); + assert_eq!(shares, vec![0.0, 0.0]); + } +} diff --git a/core/dr-face/src/cluster.rs b/core/dr-face/src/cluster.rs index 6fe94be..be1315a 100644 --- a/core/dr-face/src/cluster.rs +++ b/core/dr-face/src/cluster.rs @@ -148,22 +148,110 @@ pub fn cluster(faces: &[Candidate], cal: &Calibration, min_probability: f32) -> return Vec::new(); } - let embeddings: Vec> = faces.iter().map(|f| f.embedding.clone()).collect(); - let crop_px: Vec = faces.iter().map(|f| f.crop_px).collect(); - let images: Vec = faces.iter().map(|f| f.image).collect(); - let view = Faces { - embeddings: &embeddings, - crop_px: &crop_px, - images: &images, - }; - + let columns = Columns::of(faces); // Every pair that could ever contribute to a merge. See the module note on // why nothing outside this list can matter. - let pairs = neighbours::above_threshold(&view, cal, min_probability); + let pairs = neighbours::above_threshold(&columns.view(), cal, min_probability); + build(faces, cal, min_probability, &pairs) +} +/// Groups, and how confident each face's placement is. +/// +/// The second half is [`crate::assign`]'s share, not the pairwise probability +/// that put the face in the group — see that module for why the two are +/// different questions. +#[derive(Debug, Clone, PartialEq)] +pub struct Grouping { + pub clusters: Vec, + /// Indexed like the input faces. 0 for a face in no group. + pub confidence: Vec, +} + +/// Group faces into people, and score each placement against its rivals. +/// +/// What a caller writing suggestions into a catalog wants: [`cluster`] answers +/// *which person*, this answers *and how sure*. +/// +/// One scan, two thresholds. The similarity scan is the expensive part of the +/// whole subsystem and running it twice — once to merge, once to find rivals — +/// would double the cost of regrouping a library. So it runs once at the looser +/// of the two floors, and the merge engine takes the subset at or above +/// `min_probability`. That subset is identical, pair for pair and in the same +/// order, to what a scan at `min_probability` would have produced, so grouping +/// is unchanged by scoring being asked for: `scoring_does_not_change_the_ +/// groups` holds it to that. +pub fn cluster_scored(faces: &[Candidate], cal: &Calibration, min_probability: f32) -> Grouping { + if faces.is_empty() { + return Grouping { + clusters: Vec::new(), + confidence: Vec::new(), + }; + } + + let columns = Columns::of(faces); + let evidence = neighbours::above_threshold( + &columns.view(), + cal, + min_probability.min(crate::assign::RIVAL_FLOOR), + ); + let merges: Vec = evidence + .iter() + .copied() + .filter(|p| p.probability >= min_probability) + .collect(); + + let clusters = build(faces, cal, min_probability, &merges); + let confidence = crate::assign::identity_shares( + faces.len(), + &clusters, + &evidence, + crate::assign::TOP_MATCHES, + ); + Grouping { + clusters, + confidence, + } +} + +/// The three arrays [`neighbours::Faces`] borrows, owned. +/// +/// [`neighbours`] takes parallel slices rather than candidates on purpose — it +/// has no business knowing what a person is — so somebody has to hold the +/// columns. Both entry points do, identically, which is the only reason this is +/// a type and not three locals. +struct Columns { + embeddings: Vec>, + crop_px: Vec, + images: Vec, +} + +impl Columns { + fn of(faces: &[Candidate]) -> Self { + Self { + embeddings: faces.iter().map(|f| f.embedding.clone()).collect(), + crop_px: faces.iter().map(|f| f.crop_px).collect(), + images: faces.iter().map(|f| f.image).collect(), + } + } + + fn view(&self) -> Faces<'_> { + Faces { + embeddings: &self.embeddings, + crop_px: &self.crop_px, + images: &self.images, + } + } +} + +fn build( + faces: &[Candidate], + cal: &Calibration, + min_probability: f32, + pairs: &[neighbours::Pair], +) -> Vec { let mut engine = Engine::new(faces, cal, min_probability); - for component in components(faces.len(), &pairs) { - engine.agglomerate(&component, &pairs); + for component in components(faces.len(), pairs) { + engine.agglomerate(&component, pairs); } engine.finish() } @@ -663,6 +751,66 @@ mod tests { assert_eq!(out.len(), 2, "clustering overrode two user confirmations"); } + /// A face at a chosen cosine to identity 0 *and* to identity 1 at once — + /// the sibling geometry, which [`at_cosine`]'s per-identity subspaces + /// cannot express. + fn contested(to_first: f32, to_second: f32) -> Vec { + let mut v = vec![0.0_f32; EMBEDDING_DIM]; + v[0] = to_first; + v[2] = to_second; + v[4] = (1.0 - to_first * to_first - to_second * to_second) + .max(0.0) + .sqrt(); + v + } + + /// The population the scoring tests share: two faces of one person, a + /// stranger, and a face that matches the person well and the stranger + /// weakly — weakly enough that it will never merge with them, which is + /// exactly the rival a within-group score cannot see. + fn with_a_rival() -> Vec { + let mut x = candidate(4, 13, 0, 0.0); + x.embedding = contested(0.45, 0.37); + // The stranger is *named*: only an identity the user has asserted + // competes for a face (crate::assign). + let mut stranger = candidate(3, 12, 1, 1.0); + stranger.confirmed_person = Some(7); + vec![ + candidate(1, 10, 0, 1.0), + candidate(2, 11, 0, 1.0), + stranger, + x, + ] + } + + /// Asking for confidences must not move a single face. The scan runs at a + /// looser floor to find rivals, and the merge engine has to see exactly the + /// pairs it would have seen without them. + #[test] + fn scoring_does_not_change_the_groups() { + let faces = with_a_rival(); + let plain = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + let scored = cluster_scored(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + assert_eq!(plain, scored.clusters); + } + + /// The number the user is shown answers "which of these people", so a + /// second claimant has to lower it even when it is too weak to merge. + #[test] + fn a_face_two_identities_could_claim_is_reported_as_less_certain() { + let faces = with_a_rival(); + let scored = cluster_scored(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); + + let alone = cluster_scored(&faces[..2], &cal(), DEFAULT_MERGE_PROBABILITY); + assert!(alone.confidence[0] > 0.99, "nobody else to be"); + + let contested = scored.confidence[3]; + assert!( + (0.6..0.85).contains(&contested), + "a face with a second claimant: {contested}" + ); + } + #[test] fn a_suggestion_joins_the_person_its_group_is_anchored_to() { let mut anchor = candidate(1, 10, 0, 1.0); diff --git a/core/dr-face/src/lib.rs b/core/dr-face/src/lib.rs index 4a5c433..96a72e8 100644 --- a/core/dr-face/src/lib.rs +++ b/core/dr-face/src/lib.rs @@ -24,13 +24,15 @@ //! //! # Why the runtime is split behind a feature //! -//! [`calibrate`] and [`cluster`] are where this subsystem's accuracy actually -//! lives, and both are pure arithmetic over embeddings with no model in them. +//! [`calibrate`], [`cluster`] and [`assign`] are where this subsystem's accuracy +//! actually lives, and all three are pure arithmetic over embeddings with no +//! model in them. //! They build and test without `inference`, on synthetic embeddings, on a //! machine with no weights on it — which is what lets CI cover the part most //! likely to be subtly wrong. pub mod align; +pub mod assign; pub mod calibrate; pub mod cluster; #[cfg(feature = "inference")] @@ -42,8 +44,11 @@ pub mod naming; pub mod neighbours; pub use align::{warp, Aligned112, Similarity, ALIGNED_EDGE, ARCFACE_TEMPLATE}; +pub use assign::{identity_shares, RIVAL_FLOOR, TOP_MATCHES}; pub use calibrate::{Calibration, Pairs, ReliabilityBand}; -pub use cluster::{cluster, split, Candidate, Cluster, DEFAULT_MERGE_PROBABILITY}; +pub use cluster::{ + cluster, cluster_scored, split, Candidate, Cluster, Grouping, DEFAULT_MERGE_PROBABILITY, +}; #[cfg(feature = "inference")] pub use detect::{DetectOptions, Detection, Detector}; #[cfg(feature = "inference")] diff --git a/docs/faces.md b/docs/faces.md index b72fa65..3e586c9 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -228,6 +228,7 @@ core/dr-face/ src/embed.rs MBF: preprocess, forward, L2 normalise src/calibrate.rs cosine → P(same person) (FR-CULL-9) src/cluster.rs constrained agglomeration (FR-CULL-10) + src/assign.rs which person, and how sure (§9.1, FR-CULL-9) ``` ```toml @@ -675,6 +676,64 @@ are recomputed freely; confirmations survive all of it (FR-CULL-10). as the split. FR-CULL-10 requires splitting to be as easy as merging, and a split that hands the user a pile of loose faces to re-sort is not that. +### 9.1 The number beside a suggestion + +Which person a face belongs to and how sure that is are **different questions**, and the second one +is not answered by the pairwise probabilities that settled the first. + +The first implementation answered it with the mean calibrated probability between the face and the +rest of its group, and that measures the wrong thing twice. It punishes coverage: a person with two +hundred faces across fifteen years is *supposed* to have members a new photograph is orthogonal to, +so the better someone is photographed the worse their suggestions score. And it never asks who else +the face could be — a face matching Anna at 0.95 and nobody else, and one matching Anna at 0.95 and +her sister at 0.93, come out identical, when the second is the only one the user needs to look at. + +Two questions, so two factors, multiplied: + +``` +evidence(P) = Σ of the top n of { P(same | this face, f) : f ∈ P } n = 10 +coherence = evidence(own) / how many of the top n there were +uniqueness = evidence(own) / (evidence(own) + Σ evidence(named rivals)) +confidence = coherence × uniqueness +``` + +**Coherence** is the old mean with a cap on it, and the cap is the whole fix: the two hundred faces a +given photograph is legitimately orthogonal to stop counting against it. **Uniqueness** is the +competition, and it is what makes an ambiguous face read as ambiguous — two identities matching +equally well land at 0.5 each, which is the truth about a sibling. + +**Only named people compete, and they compete per person.** This is the part that had to be measured +rather than reasoned about. Normalising across *every* group made the number useless on a real +18,000-face library — median suggestion 21%, four in five under half — because clustering leaves one +person spread across many groups, so a face competes against itself. Counting only groups holding a +confirmation fixed most of it; counting them **per person** rather than per group fixed the rest, +since a named person is left in several anchored groups for the same reason. + +**Rivals are gathered below the merge threshold**, down to even odds: a named person who matches at +0.6 will never be merged into but is exactly the competition a suggestion should be discounted for. +The floor matters in both directions — summing the near-orthogonal pairs instead of dropping them +lets fifty identities' worth of upper-tail noise outweigh one real match, which on the same library +moved the median stated confidence from 100% to 31%. + +*Measured (`cargo run --release -p dr-catalog --example face_confidence`), leave-one-out over that +library's 2,702 confirmations across 54 named people:* + +| | share | old mean | +|---|---|---| +| right person picked | **99.33%** | 99.15% | +| stated for the right person, median | **99.3%** | 90.4% | +| stated for the right person, p10 | **79.2%** | 68.0% | + +The reliability table is monotone and **errs low**: 100% correct wherever it states 80% or more, 84% +correct where it states under half. Understating is the safe direction for a screen whose purpose is +deciding what to look at first, but the low bands are not calibrated and should not be read as +though they were — and the leave-one-out task asks *which of these people*, never *is it any of +them*, so it cannot speak to a stranger at all. + +It is **not** a merge threshold and must not become one. Uniqueness is relative, so a library with +one named person would hand every stray face a 1. "Is this the same person at all" stays §8's +question, and coherence is the half of the product that carries it. + --- ## 10. Catalog and jobs diff --git a/docs/traceability.md b/docs/traceability.md index 7902f84..ce9d686 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -9,7 +9,7 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n | Metric | Value | |---|---| -| Source files scanned | 279 | +| Source files scanned | 280 | | TRACES tags found | 812 | | Requirements defined | 177 | | Requirements covered | 106 | diff --git a/ui/dr-ui/src/faces.rs b/ui/dr-ui/src/faces.rs index 25fd06b..11b7376 100644 --- a/ui/dr-ui/src/faces.rs +++ b/ui/dr-ui/src/faces.rs @@ -543,7 +543,10 @@ pub fn recluster( ids.push(face_id); } - let clusters = dr_face::cluster(&candidates, &cal, min_probability); + let dr_face::Grouping { + clusters, + confidence, + } = dr_face::cluster_scored(&candidates, &cal, min_probability); let mut suggested = 0usize; let mut created = 0usize; @@ -569,10 +572,12 @@ pub fn recluster( if confirmed.contains_key(&face) { continue; } - // The probability the user is shown is the group's own coherence, - // not the single best edge into it — a face admitted by one strong - // match to an outlier should not present as certain. - let p = group_probability(&candidates, c, m, &cal); + // The probability the user is shown is this identity's share of + // the evidence for the face, against every other identity that + // could plausibly claim it (dr_face::assign) — not the single best + // edge, which cannot tell a sole match from a coin toss between + // two siblings. + let p = confidence[m]; if faces::suggest(conn, face, person, p)? { suggested += 1; } @@ -662,37 +667,6 @@ pub fn spawn_recluster( rx } -/// Mean calibrated probability between one member and the rest of its group. -fn group_probability( - candidates: &[dr_face::Candidate], - cluster: &dr_face::Cluster, - member: usize, - cal: &Calibration, -) -> f32 { - let me = &candidates[member]; - let mut sum = 0.0; - let mut n = 0.0; - for &other in &cluster.members { - if other == member { - continue; - } - let them = &candidates[other]; - let cos: f32 = me - .embedding - .iter() - .zip(&them.embedding) - .map(|(a, b)| a * b) - .sum(); - sum += cal.probability(cos, me.crop_px.min(them.crop_px), 0.0); - n += 1.0; - } - if n == 0.0 { - 1.0 - } else { - sum / n - } -} - /// A face cut out of its photograph, ready to draw. #[derive(Debug, Clone, PartialEq)] pub struct FaceCrop {