//! How fast this machine can scan a library for face pairs. //! //! cargo run --release -p dr-face --example scan_bench [-- FACES…] //! //! Synthetic embeddings, because the scan's cost is `n²/2` dot products and //! does not care what the vectors mean — which is what makes this runnable on //! a phone with no library on it, over `adb shell`, beside //! `tools/face-tests-on-device.sh`. //! //! # What it is for //! //! docs/faces.md §9 has the desktop numbers and the question they leave open: //! a GPU GEMM is worth roughly 1.5× of a regroup on a twenty-core desktop, //! because the scan is under a third of the pass there. On a tablet the CPU is //! several times slower and the GPU is not, so the same optimisation is worth //! something different — and nobody had measured which. //! //! No weights, no catalog, no display: it needs nothing but the binary. use dr_face::neighbours::{above_threshold, Faces}; use dr_face::{Calibration, EMBEDDING_DIM, RIVAL_FLOOR}; /// Sizes to time, unless the command line names others. const DEFAULT_SIZES: [usize; 4] = [2_000, 4_000, 8_000, 18_000]; fn main() { let sizes: Vec = { let given: Vec = std::env::args() .skip(1) .filter_map(|a| a.parse().ok()) .collect(); if given.is_empty() { DEFAULT_SIZES.to_vec() } else { given } }; // The reference curve, so the cosine floor is the one a real library with // no fit of its own would scan at. let cal = Calibration::default(); println!( "{:>8} {:>9} {:>10} {:>9}", "faces", "scan", "pairs", "GFLOP/s" ); for n in sizes { let (embeddings, crop_px, images) = population(n); let gallery = vec![true; n]; let faces = Faces { embeddings: &embeddings, dim: EMBEDDING_DIM, crop_px: &crop_px, images: &images, gallery: &gallery, }; let start = std::time::Instant::now(); let pairs = above_threshold(&faces, &cal, RIVAL_FLOOR); let secs = start.elapsed().as_secs_f64(); let flop = n as f64 * n as f64 / 2.0 * EMBEDDING_DIM as f64 * 2.0; println!( "{n:>8} {secs:>8.2}s {:>10} {:>9.1}", pairs.len(), flop / secs / 1e9 ); } } /// `n` L2-normalised embeddings in a handful of loose clusters. /// /// Clustered rather than uniform so the scan finds a plausible number of pairs /// to keep — a population where nothing survives the threshold would time the /// rejection path alone, which is not the path that matters. Hashed from an /// index rather than drawn from an RNG, so a number is reproducible from the /// command that produced it. fn population(n: usize) -> (Vec, Vec, Vec) { let identities = (n / 12).max(1); let mut embeddings = Vec::with_capacity(n * EMBEDDING_DIM); for i in 0..n { let mut v = unit(i % identities); let noise = unit(i + 1_000_000); for (x, e) in v.iter_mut().zip(&noise) { *x = 0.75 * *x + 0.25 * e; } embeddings.extend(normalise(v)); } // Every face in its own photograph: the co-occurrence rule skips pairs // rather than scoring them, and skipped pairs are not what is being timed. ((embeddings), vec![150.0; n], (0..n as u64).collect()) } fn unit(seed: usize) -> Vec { let mut s = (seed as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15) | 1; let mut v = Vec::with_capacity(EMBEDDING_DIM); for _ in 0..EMBEDDING_DIM { s ^= s << 13; s ^= s >> 7; s ^= s << 17; v.push(((s >> 11) as f64 / (1u64 << 53) as f64) as f32 - 0.5); } normalise(v) } fn normalise(mut v: Vec) -> Vec { let len = v.iter().map(|x| x * x).sum::().sqrt(); for x in &mut v { *x /= len; } v }