//! Grouping faces into people (docs/faces.md §9, FR-CULL-10). //! //! Model-free: this is arithmetic over embeddings, and it is where the //! subsystem's accuracy actually lives, so it is testable with no weights on //! the machine. //! //! # Constraints, not just a threshold //! //! FR-CULL-10 warns that clustering will over-merge on siblings, on parents and //! children, and on the same person a decade apart. Two structural defences, //! both cheaper than a better threshold: //! //! **Cannot-link on co-occurrence.** Two faces in the same photograph are never //! merged. It is the same observation [`crate::calibrate`] mines for free //! negatives, used here as a hard constraint, and it is the single cheapest //! defence against over-merging that exists. //! //! **Confirmed faces are anchors.** A confirmation is user data (FR-CULL-12) //! and clustering never moves it. Two groups holding confirmations of //! *different* people cannot merge, whatever their similarity says. //! //! # Average link, not single link //! //! Single-link chains: one bad edge welds two identities together, and it is //! the documented way face clustering fails on families. Average link asks //! whether the *groups* are similar, which one outlier cannot force. use std::collections::HashSet; use crate::calibrate::Calibration; use crate::embedding::EMBEDDING_DIM; /// Probability above which two groups are judged the same person. /// /// Stated as a probability and not a cosine, because FR-CULL-9 forbids /// thresholding a bare similarity anywhere in this subsystem. pub const DEFAULT_MERGE_PROBABILITY: f32 = 0.9; /// A face presented to the clusterer. /// /// Ids are opaque `u64`s rather than catalog types: this crate has no business /// knowing what a `FaceId` means, and the caller does the translation. #[derive(Debug, Clone)] pub struct Candidate { pub face: u64, /// Which photograph it came from — the cannot-link key. pub image: u64, /// L2-normalised, `EMBEDDING_DIM` long. pub embedding: Vec, /// Source pixels across the aligned crop, for the calibration's size term. pub crop_px: f32, /// The person this face is *confirmed* to be, if any. /// /// Suggestions are deliberately not passed here. They are this function's /// own previous output, and feeding them back in would let a guess harden /// into a fact across successive passes. pub confirmed_person: Option, } /// One group of faces the clusterer believes are one person. #[derive(Debug, Clone, PartialEq)] pub struct Cluster { /// Indices into the input slice. pub members: Vec, /// The person this group is already known to be, from its anchors. /// /// `Some` means the group contains confirmed faces and the suggestions in /// it attach to that existing person. `None` is a new unnamed group. pub person: Option, } /// Group faces into people. /// /// `min_probability` is compared against the calibrated average-link /// probability between two groups. Deterministic: the same input yields the /// same clusters, because the merge order is by score with the index pair as /// the tiebreak. pub fn cluster( faces: &[Candidate], cal: &Calibration, min_probability: f32, ) -> Vec { if faces.is_empty() { return Vec::new(); } let n = faces.len(); let mut groups: Vec = faces .iter() .enumerate() .map(|(i, f)| Group { members: vec![i], images: HashSet::from([f.image]), person: f.confirmed_person, alive: true, }) .collect(); // Pairwise cosine once. n² f32 is the honest cost at library scale — for // 25,000 faces that is the blocked GEMM docs/faces.md §9 describes, and the // caller is expected to shard rather than this function growing an index. let mut cos = vec![0.0_f32; n * n]; for i in 0..n { for j in i + 1..n { let c = dot(&faces[i].embedding, &faces[j].embedding); cos[i * n + j] = c; cos[j * n + i] = c; } } loop { let mut best: Option<(f32, usize, usize)> = None; for a in 0..n { if !groups[a].alive { continue; } for b in a + 1..n { if !groups[b].alive || !can_link(&groups[a], &groups[b]) { continue; } let p = average_link(&groups[a], &groups[b], faces, &cos, n, cal); if p >= min_probability && best.is_none_or(|(bp, _, _)| p > bp) { best = Some((p, a, b)); } } } let Some((_, a, b)) = best else { break }; let taken = std::mem::take(&mut groups[b]); groups[b].alive = false; let ga = &mut groups[a]; ga.members.extend(taken.members); ga.images.extend(taken.images); // At most one side carries a person: `can_link` refuses a merge of two // groups anchored to different people, so this cannot silently discard // one of them. ga.person = ga.person.or(taken.person); } let mut out: Vec = groups .into_iter() .filter(|g| g.alive) .map(|mut g| { g.members.sort_unstable(); Cluster { members: g.members, person: g.person, } }) .collect(); // Largest first: the People view shows the best-evidenced groups at the top. out.sort_by(|x, y| y.members.len().cmp(&x.members.len()).then(x.members[0].cmp(&y.members[0]))); out } /// Split one person's faces into the groups a raised threshold separates them /// into. /// /// FR-CULL-10 requires splitting to be as easy as merging, and a split that /// hands the user a pile of loose faces to re-sort is not that. This re-runs /// the same agglomeration at a stricter probability so the user is offered /// coherent sub-groups to pull apart. /// /// Anchors are ignored here on purpose: every face in the input is already /// nominally the same person, so honouring the anchors would refuse to split /// anything. pub fn split(faces: &[Candidate], cal: &Calibration, min_probability: f32) -> Vec { let anchorless: Vec = faces .iter() .cloned() .map(|mut f| { f.confirmed_person = None; f }) .collect(); cluster(&anchorless, cal, min_probability) } #[derive(Debug, Default)] struct Group { members: Vec, images: HashSet, person: Option, alive: bool, } /// Whether two groups are allowed to merge at all, before similarity is asked. fn can_link(a: &Group, b: &Group) -> bool { // Two confirmations of different people. The user has said these are not // the same person, and no similarity overrides that. if let (Some(pa), Some(pb)) = (a.person, b.person) { if pa != pb { return false; } } // Co-occurrence: a photograph containing a face from each group means the // two faces are in the same frame, so they are not the same person. a.images.is_disjoint(&b.images) } /// Mean calibrated probability over every cross-group pair. fn average_link( a: &Group, b: &Group, faces: &[Candidate], cos: &[f32], n: usize, cal: &Calibration, ) -> f32 { let mut sum = 0.0; let mut count = 0.0; for &i in &a.members { for &j in &b.members { let min_crop = faces[i].crop_px.min(faces[j].crop_px); sum += cal.probability(cos[i * n + j], min_crop, 0.0); count += 1.0; } } if count == 0.0 { 0.0 } else { sum / count } } fn dot(a: &[f32], b: &[f32]) -> f32 { debug_assert_eq!(a.len(), EMBEDDING_DIM); debug_assert_eq!(b.len(), EMBEDDING_DIM); a.iter().zip(b).map(|(x, y)| x * y).sum() } #[cfg(test)] mod tests { use super::*; /// An embedding a known cosine away from a base direction, built by mixing /// two orthogonal unit vectors. Lets a test state "these two faces are 0.7 /// similar" and have it be exactly true. fn at_cosine(identity: usize, cosine: f32) -> Vec { let mut v = vec![0.0_f32; EMBEDDING_DIM]; let base = identity * 2; let perp = identity * 2 + 1; v[base] = cosine; v[perp] = (1.0 - cosine * cosine).max(0.0).sqrt(); v } fn candidate(face: u64, image: u64, identity: usize, cosine: f32) -> Candidate { Candidate { face, image, embedding: at_cosine(identity, cosine), crop_px: 150.0, confirmed_person: None, } } /// A calibration steep enough that the test's cosines are unambiguous: /// 0.6 is near-certain, 0.1 is near-impossible. fn cal() -> Calibration { Calibration { a: 30.0, b: -30.0 * 0.35, w_size: 0.0, valid: true, positive_pairs: 1000, negative_pairs: 10_000, } } #[test] fn no_faces_makes_no_clusters() { assert!(cluster(&[], &cal(), DEFAULT_MERGE_PROBABILITY).is_empty()); } #[test] fn similar_faces_from_different_photographs_group_together() { let faces = vec![ candidate(1, 10, 0, 1.0), candidate(2, 11, 0, 0.95), candidate(3, 12, 0, 0.92), ]; let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); assert_eq!(out.len(), 1); assert_eq!(out[0].members, vec![0, 1, 2]); } #[test] fn dissimilar_faces_stay_apart() { let faces = vec![candidate(1, 10, 0, 1.0), candidate(2, 11, 1, 1.0)]; let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); assert_eq!(out.len(), 2); } /// The cheapest defence against over-merging: two faces in one frame are /// not the same person however similar the model finds them. #[test] fn two_faces_in_one_photograph_never_merge() { // Identical embeddings — siblings, or a model that cannot tell them // apart — but both in image 10. let faces = vec![candidate(1, 10, 0, 1.0), candidate(2, 10, 0, 1.0)]; let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); assert_eq!(out.len(), 2, "co-occurring faces were merged"); } /// And the constraint has to survive transitively: once a group holds a /// face from image 10, no other group holding one from image 10 may join /// it, even indirectly. #[test] fn the_co_occurrence_constraint_propagates_through_a_group() { let faces = vec![ candidate(1, 10, 0, 1.0), // A, in the group photo candidate(2, 10, 0, 1.0), // B, in the same group photo candidate(3, 11, 0, 0.99), // A again, alone ]; let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); assert_eq!(out.len(), 2); // Whichever of A/B absorbed face 2, the other stays out. assert!(out.iter().any(|c| c.members.len() == 2)); assert!(out.iter().any(|c| c.members.len() == 1)); } /// FR-CULL-10: a confirmation is user data and no inference overrides it. #[test] fn groups_confirmed_as_different_people_do_not_merge() { let mut a = candidate(1, 10, 0, 1.0); let mut b = candidate(2, 11, 0, 1.0); a.confirmed_person = Some(100); b.confirmed_person = Some(200); let out = cluster(&[a, b], &cal(), DEFAULT_MERGE_PROBABILITY); assert_eq!(out.len(), 2, "clustering overrode two user confirmations"); } #[test] fn a_suggestion_joins_the_person_its_group_is_anchored_to() { let mut anchor = candidate(1, 10, 0, 1.0); anchor.confirmed_person = Some(42); let loose = candidate(2, 11, 0, 0.95); let out = cluster(&[anchor, loose], &cal(), DEFAULT_MERGE_PROBABILITY); assert_eq!(out.len(), 1); assert_eq!(out[0].person, Some(42)); assert_eq!(out[0].members.len(), 2); } #[test] fn an_unanchored_group_is_a_new_unnamed_person() { let out = cluster( &[candidate(1, 10, 0, 1.0), candidate(2, 11, 0, 0.95)], &cal(), DEFAULT_MERGE_PROBABILITY, ); assert_eq!(out[0].person, None); } /// Average link rather than single link: one strong edge must not weld two /// otherwise-dissimilar groups together. This is the family failure mode /// FR-CULL-10 names. #[test] fn one_strong_edge_does_not_chain_two_groups_together() { // Two tight pairs, with a single borderline link between them. let faces = vec![ candidate(1, 10, 0, 1.00), candidate(2, 11, 0, 0.99), candidate(3, 12, 0, 0.42), candidate(4, 13, 0, 0.40), ]; let out = cluster(&faces, &cal(), 0.99); assert!( out.len() >= 2, "single-link chaining merged everything into {} cluster(s)", out.len() ); } #[test] fn clustering_is_deterministic() { let faces = vec![ candidate(1, 10, 0, 1.0), candidate(2, 11, 0, 0.96), candidate(3, 12, 1, 1.0), candidate(4, 13, 1, 0.97), candidate(5, 14, 0, 0.94), ]; let a = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); let b = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); assert_eq!(a, b); } #[test] fn clusters_come_back_largest_first() { let faces = vec![ candidate(1, 10, 1, 1.0), candidate(2, 11, 0, 1.0), candidate(3, 12, 0, 0.97), candidate(4, 13, 0, 0.95), ]; let out = cluster(&faces, &cal(), DEFAULT_MERGE_PROBABILITY); assert_eq!(out[0].members.len(), 3); assert_eq!(out[1].members.len(), 1); } /// Splitting is the inverse operation and must actually separate a group /// that a looser threshold had merged. #[test] fn split_separates_a_group_that_a_looser_threshold_merged() { let faces = vec![ candidate(1, 10, 0, 1.00), candidate(2, 11, 0, 0.98), candidate(3, 12, 0, 0.45), candidate(4, 13, 0, 0.43), ]; // Loose: one person. assert_eq!(cluster(&faces, &cal(), 0.5).len(), 1); // Strict: the two sub-groups the user wants offered. let parts = split(&faces, &cal(), 0.999); assert!(parts.len() >= 2, "split produced {} group(s)", parts.len()); } #[test] fn split_ignores_the_anchor_so_a_mislabelled_person_can_be_taken_apart() { let mut a = candidate(1, 10, 0, 1.0); let mut b = candidate(2, 11, 0, 0.40); a.confirmed_person = Some(7); b.confirmed_person = Some(7); let parts = split(&[a, b], &cal(), 0.99); assert_eq!(parts.len(), 2); } /// The size term earns its place: the same cosine between two thumbnail- /// sized faces should be less convincing than between two large ones. #[test] fn the_face_size_term_moves_the_probability() { let sized = Calibration { w_size: 0.5, b: -30.0 * 0.35 - 0.5 * 7.0, ..cal() }; let big = sized.probability(0.5, 300.0, 0.0); let small = sized.probability(0.5, 40.0, 0.0); assert!(big > small, "big {big} should beat small {small}"); } }