From b2250cc460a1ca2fa59bb6a96d2480fdb31cd41a Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 29 Aug 2026 12:16:48 +0200 Subject: [PATCH] Measure a regroup on the tablet, not just on the desktop MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The GPU question needed a number nobody had: how a regroup divides on the hardware whose CPU is weakest. dr-face carries no weights and touches no display, and dr-catalog's example needs only a catalog file, so both run under adb shell against a copy of a real library. On the same 18,143 faces — desktop against the tablet — scan 0.96s / 2.61s, agglomerate 1.69s / 2.16s, score 0.26s / 0.40s. The scan is half the pass on the tablet and under a third on the desktop, because twenty cores of AVX2 pull ahead of NEON much further than the merge engine's single-threaded hashing does. So a GPU GEMM is worth roughly 2× a regroup on the tablet and 1.5× here, and it is the tablet that should decide whether it is built. The two architectures agree exactly: the same 1,531,969 evidence pairs, the same 2,518 groups holding the same 16,246 faces, the same reliability table. That is a better check on the NEON kernel than the unit test can be. Two instruments, both read-only: the example now prints its phases, and dr-face gains scan_bench, which needs no library at all and so can answer "how fast is this machine" on a device with nothing on it. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-catalog/examples/face_confidence.rs | 40 ++++++++ core/dr-face/examples/scan_bench.rs | 104 ++++++++++++++++++++ docs/faces.md | 28 +++--- docs/traceability.md | 2 +- 4 files changed, 162 insertions(+), 12 deletions(-) create mode 100644 core/dr-face/examples/scan_bench.rs diff --git a/core/dr-catalog/examples/face_confidence.rs b/core/dr-catalog/examples/face_confidence.rs index e2d633c..55d278a 100644 --- a/core/dr-catalog/examples/face_confidence.rs +++ b/core/dr-catalog/examples/face_confidence.rs @@ -305,6 +305,46 @@ fn full_library( candidates.sort_by_key(|c| c.face); println!("\nthe whole library, at the default merge probability:"); + + // The three phases, separately, because "a regroup takes n seconds" does + // not tell anyone which half to optimise — and the answer differs between + // a desktop and a tablet (docs/faces.md §9). + { + let dim = candidates.first().map(|c| c.embedding.len()).unwrap_or(0); + let flat: Vec = candidates.iter().flat_map(|c| c.embedding.clone()).collect(); + let crop_px: Vec = candidates.iter().map(|c| c.crop_px).collect(); + let images: Vec = candidates.iter().map(|c| c.image).collect(); + let view = dr_face::neighbours::Faces { + embeddings: &flat, + dim, + crop_px: &crop_px, + images: &images, + }; + + let t = std::time::Instant::now(); + let evidence = dr_face::neighbours::above_threshold(&view, cal, dr_face::RIVAL_FLOOR); + let scan = t.elapsed().as_secs_f64(); + + // `cluster` runs its own scan at the merge threshold, so the + // agglomeration is what is left after taking one scan off the total. + let t = std::time::Instant::now(); + let clusters = dr_face::cluster(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY); + let agglomerate = t.elapsed().as_secs_f64() - scan; + + let t = std::time::Instant::now(); + let _ = dr_face::identity_shares( + candidates.len(), + &clusters, + &evidence, + dr_face::TOP_MATCHES, + ); + println!( + " scan {scan:.2}s ({} evidence pairs) · agglomerate {agglomerate:.2}s · score {:.2}s", + evidence.len(), + t.elapsed().as_secs_f64() + ); + } + let start = std::time::Instant::now(); let grouping = dr_face::cluster_scored(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY); let real: Vec<_> = grouping diff --git a/core/dr-face/examples/scan_bench.rs b/core/dr-face/examples/scan_bench.rs new file mode 100644 index 0000000..d1279c2 --- /dev/null +++ b/core/dr-face/examples/scan_bench.rs @@ -0,0 +1,104 @@ +//! How fast this machine can scan a library for face pairs. +//! +//! cargo run --release -p dr-face --example scan_bench [-- FACES…] +//! +//! Synthetic embeddings, because the scan's cost is `n²/2` dot products and +//! does not care what the vectors mean — which is what makes this runnable on +//! a phone with no library on it, over `adb shell`, beside +//! `tools/face-tests-on-device.sh`. +//! +//! # What it is for +//! +//! docs/faces.md §9 has the desktop numbers and the question they leave open: +//! a GPU GEMM is worth roughly 1.5× of a regroup on a twenty-core desktop, +//! because the scan is under a third of the pass there. On a tablet the CPU is +//! several times slower and the GPU is not, so the same optimisation is worth +//! something different — and nobody had measured which. +//! +//! No weights, no catalog, no display: it needs nothing but the binary. + +use dr_face::neighbours::{above_threshold, Faces}; +use dr_face::{Calibration, EMBEDDING_DIM, RIVAL_FLOOR}; + +/// Sizes to time, unless the command line names others. +const DEFAULT_SIZES: [usize; 4] = [2_000, 4_000, 8_000, 18_000]; + +fn main() { + let sizes: Vec = { + let given: Vec = std::env::args().skip(1).filter_map(|a| a.parse().ok()).collect(); + if given.is_empty() { + DEFAULT_SIZES.to_vec() + } else { + given + } + }; + + // The reference curve, so the cosine floor is the one a real library with + // no fit of its own would scan at. + let cal = Calibration::default(); + + println!("{:>8} {:>9} {:>10} {:>9}", "faces", "scan", "pairs", "GFLOP/s"); + for n in sizes { + let (embeddings, crop_px, images) = population(n); + let faces = Faces { + embeddings: &embeddings, + dim: EMBEDDING_DIM, + crop_px: &crop_px, + images: &images, + }; + + let start = std::time::Instant::now(); + let pairs = above_threshold(&faces, &cal, RIVAL_FLOOR); + let secs = start.elapsed().as_secs_f64(); + + let flop = n as f64 * n as f64 / 2.0 * EMBEDDING_DIM as f64 * 2.0; + println!( + "{n:>8} {secs:>8.2}s {:>10} {:>9.1}", + pairs.len(), + flop / secs / 1e9 + ); + } +} + +/// `n` L2-normalised embeddings in a handful of loose clusters. +/// +/// Clustered rather than uniform so the scan finds a plausible number of pairs +/// to keep — a population where nothing survives the threshold would time the +/// rejection path alone, which is not the path that matters. Hashed from an +/// index rather than drawn from an RNG, so a number is reproducible from the +/// command that produced it. +fn population(n: usize) -> (Vec, Vec, Vec) { + let identities = (n / 12).max(1); + let mut embeddings = Vec::with_capacity(n * EMBEDDING_DIM); + for i in 0..n { + let mut v = unit(i % identities); + let noise = unit(i + 1_000_000); + for (x, e) in v.iter_mut().zip(&noise) { + *x = 0.75 * *x + 0.25 * e; + } + embeddings.extend(normalise(v)); + } + // Every face in its own photograph: the co-occurrence rule skips pairs + // rather than scoring them, and skipped pairs are not what is being timed. + ((embeddings), vec![150.0; n], (0..n as u64).collect()) +} + +fn unit(seed: usize) -> Vec { + let mut s = (seed as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15) | 1; + let mut v = Vec::with_capacity(EMBEDDING_DIM); + for _ in 0..EMBEDDING_DIM { + s ^= s << 13; + s ^= s >> 7; + s ^= s << 17; + v.push(((s >> 11) as f64 / (1u64 << 53) as f64) as f32 - 0.5); + } + normalise(v) +} + +fn normalise(mut v: Vec) -> Vec { + let len = v.iter().map(|x| x * x).sum::().sqrt(); + for x in &mut v { + *x /= len; + } + v +} diff --git a/docs/faces.md b/docs/faces.md index 2ee00c5..40ed45e 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -680,14 +680,15 @@ and touches no display, so its tests are a plain ARM64 binary that runs under `a nothing installed. Worth running whenever the kernels change. **Where a regroup's time actually goes**, on that library, because the answer moved twice while it -was being looked at: +was being looked at. Measured with `cargo run --release -p dr-catalog --example face_confidence -- +CATALOG --full`, on the reference desktop and on a Honor tablet (ROD2-W09, aarch64): -| | before | after | -|---|---|---| -| scan | 4.60 s | **0.94 s** | -| agglomerate | 4.84 s | **1.75 s** | -| score | 0.23 s | 0.28 s | -| **total** | **10.0 s** | **3.0 s** | +| | desktop, before | desktop | **tablet** | +|---|---|---|---| +| scan | 4.60 s | 0.96 s | **2.61 s** | +| agglomerate | 4.84 s | 1.69 s | **2.16 s** | +| score | 0.23 s | 0.26 s | **0.40 s** | +| **total** | **10.0 s** | **3.1 s** | **5.2 s** | The scan came down by the kernel work above. The agglomeration was not, as it looked, an irreducible sequential heap walk: 3.06 s of it was every component scanning the *whole* pair list for the pairs @@ -695,10 +696,15 @@ that were its own — 382 million set lookups to place 804,499 pairs — which t one pass while it is finding the components anyway. Another 0.54 s was `Engine::cross` computing dot products with the portable loop while the scan beside it used the machine's SIMD. -A GPU GEMM is the obvious next step for the scan, and it should be judged against the second column -rather than the first: the scan is now under a third of the pass on a desktop, so the ceiling there -is a 1.5× regroup. On a tablet, where the CPU is several times slower and the GPU is not, the split -is different and the case is stronger — which is a measurement nobody has taken yet. +**The two architectures agree exactly**, which is worth more than either column: the same 1,531,969 +evidence pairs, the same 2,518 groups holding the same 16,246 faces, and the same reliability table, +from AVX2 on the desktop and NEON on the tablet. That is the cross-kernel check the unit test can +only approximate. + +**The tablet is where a GPU GEMM would pay.** Its scan is *half* the pass, against under a third on +the desktop — twenty cores of AVX2 pull ahead of a tablet's NEON far more than the merge engine's +single-threaded hashing does. So a perfect GEMM is worth about 2× a regroup there and about 1.5× +here, and it is the phone and tablet story that should decide whether it gets built. **Constraints, not just thresholds:** diff --git a/docs/traceability.md b/docs/traceability.md index ce9d686..21fabda 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -9,7 +9,7 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n | Metric | Value | |---|---| -| Source files scanned | 280 | +| Source files scanned | 281 | | TRACES tags found | 812 | | Requirements defined | 177 | | Requirements covered | 106 |