//! The face-indexing batch job, off the GUI. //! //! Checks every library image for a face-detection run marker, and optionally //! indexes whatever is missing one. //! //! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR [--run DET.onnx EMB.onnx] //! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR --cluster //! //! # Why this exists beside the button in the Identity screen //! //! Indexing a real library is hours of work (docs/faces.md §12.2), and the //! cases where that is worth starting — an overnight pass, a fresh import, a //! machine left running — are exactly the ones where holding a window open is //! the wrong shape. The check half is useful on its own: it is cheap, it //! answers "has face recognition been over all of this", and it distinguishes //! *not yet indexed* from *waiting on a proxy*, which are different problems //! with different fixes. //! //! The models must have had their input dims frozen first; see //! `tools/fix-face-model-shapes.sh`. use std::path::PathBuf; use dr_catalog::Catalog; use dr_thumbs::ThumbStore; use dr_ui::faces::{self, FaceSweepMessage}; const MODEL_ID: &str = "w600k_mbf"; fn main() { env_logger::init(); let args: Vec = std::env::args().skip(1).collect(); if args.len() < 2 { eprintln!( "usage: face_index CATALOG.db THUMBS_DIR [--run DETECTOR.onnx EMBEDDER.onnx]\n\ \n\ With no --run this only reports; nothing is written.\n\ --cluster groups what is indexed; --tune compares thresholds without writing.\n\ --quality DET.onnx EMB.onnx reports face size and sharpness, also without writing." ); std::process::exit(2); } let catalog_path = PathBuf::from(&args[0]); let store_dir = PathBuf::from(&args[1]); let catalog = match Catalog::open(&catalog_path) { Ok(c) => c, Err(e) => { eprintln!("cannot open catalog {}: {e}", catalog_path.display()); std::process::exit(1); } }; let store = match ThumbStore::open(&store_dir) { Ok(s) => s, Err(e) => { eprintln!("cannot open thumbnail store {}: {e}", store_dir.display()); std::process::exit(1); } }; // The registry a device without the eye models would run: what this // example counts as work is what the app would. let repairs = dr_ui::repairs::registry( dr_ui::repairs::Scope::Outstanding, MODEL_ID, dr_types::FaceDetector::Scrfd500m, dr_ui::repairs::Capabilities { gpu: true, face_models: true, eye_models: false, }, ); let audit = match faces::audit(&catalog, &store, MODEL_ID, &repairs) { Ok(a) => a, Err(e) => { eprintln!("coverage check failed: {e}"); std::process::exit(1); } }; println!("model {MODEL_ID}"); println!("images {}", audit.coverage.images); println!( "indexed {} ({:.1}%)", audit.coverage.indexed, audit.coverage.fraction() * 100.0 ); println!("faces {}", audit.coverage.faces); println!( "no faces {} (indexed, nothing found — the common case)", audit.coverage.without_faces ); println!("outstanding {}", audit.coverage.outstanding()); println!(" ready to index {}", audit.ready); println!(" awaiting proxy {}", audit.awaiting_proxy); // Grouping is a separate step from indexing on purpose: it is a // whole-library operation over the embeddings detection produced, and it is // worth running *after* a sweep rather than during one (catalog.md §10.2). if args.iter().any(|a| a == "--cluster") { // The engine's own defaults, not this device's settings file: a batch // job run over a library on a server has no business inheriting the // dials somebody moved on their laptop. let grouping = dr_types::settings::FaceSettings::default(); match dr_ui::faces::recluster(&catalog, MODEL_ID, &grouping) { Ok((suggested, created)) => { println!("\nclustering: {suggested} suggestion(s), {created} new group(s)"); report_people(&catalog); } Err(e) => { eprintln!("clustering failed: {e}"); std::process::exit(1); } } return; } // Tuning, and deliberately read-only: it answers "what would this // threshold do to my library" without writing a single suggestion, which // is the only way to compare several without each one polluting the next. if args.iter().any(|a| a == "--tune") { tune_thresholds(&catalog); return; } // Quality tuning, and read-only like `--tune`. Runs the real detector over // real proxies with **both gates disabled**, so the distribution it prints // is of everything the detector finds rather than of what survives the // current settings — which is the only way to see what a threshold would // actually remove. if let Some(i) = args.iter().position(|a| a == "--quality") { let (Some(detector), Some(embedder)) = (args.get(i + 1), args.get(i + 2)) else { eprintln!("--quality needs both a detector and an embedder"); std::process::exit(2); }; report_quality( &catalog, &store, std::path::Path::new(detector), std::path::Path::new(embedder), ); return; } let run = args.iter().position(|a| a == "--run"); let Some(i) = run else { if audit.coverage.is_complete() { println!("\nnothing outstanding."); } else { println!("\npass --run DETECTOR.onnx EMBEDDER.onnx to index the outstanding images."); } return; }; let (Some(detector), Some(embedder)) = (args.get(i + 1), args.get(i + 2)) else { eprintln!("--run needs both a detector and an embedder"); std::process::exit(2); }; if audit.ready == 0 { println!("\nnothing ready to index."); if audit.awaiting_proxy > 0 { // Worth saying plainly: running this again will not help, because // the blocker is in the thumbnail store rather than here. println!( "{} image(s) are waiting on a proxy — run the thumbnail sweep first.", audit.awaiting_proxy ); } return; } // Say it here rather than letting the pass return an empty result and // print "done: 0 image(s)". That is the shape of report this whole change // exists to stop producing. if faces::FACE_TIER.edge() < dr_face::MIN_CROP_EDGE { println!( "\n--run cannot index from the local store any more.\n\ \n\ Face crops must be sampled from a buffer longer than {}px and the\n\ store's largest tier is {}px, so every image here would be refused.\n\ The measurement behind that floor is docs/faces.md §7b.\n\ \n\ Index from the app's Identity screen instead: that pass renders at\n\ native resolution, which is what the crop needs.", dr_face::MIN_CROP_EDGE - 1, faces::FACE_TIER.edge(), ); return; } println!("\nindexing {} image(s)…", audit.ready); let rx = faces::spawn_store_face_sweep( catalog_path, store_dir, dr_ui::FaceModelPaths { detector: PathBuf::from(detector), embedder: PathBuf::from(embedder), // A measurement tool for the detector and embedder; the eye // models are the library pass's business. eyes: None, }, MODEL_ID.to_string(), dr_face::DetectOptions::default(), ); let mut seen = 0usize; let mut total = 0usize; let start = std::time::Instant::now(); for msg in rx { match msg { FaceSweepMessage::Total(n) => total = n, FaceSweepMessage::Indexed { faces, .. } => { seen += 1; // One line per image would be thousands of lines; one per // twenty-five is enough to see it moving and to estimate. if seen.is_multiple_of(25) || faces > 0 { let rate = seen as f64 / start.elapsed().as_secs_f64().max(1e-6); println!( " {seen}/{total} {:.2} img/s ~{:.0} min left", rate, (total.saturating_sub(seen)) as f64 / rate.max(1e-6) / 60.0 ); } } FaceSweepMessage::Failed { images } => seen += images, FaceSweepMessage::Finished { images, faces, failed, } => { println!( "\ndone: {images} image(s), {faces} face(s), {failed} failed, in {:.0}s", start.elapsed().as_secs_f64() ); } } } if let Ok(a) = faces::audit(&catalog, &store, MODEL_ID, &repairs) { println!("{}", a.summary()); } println!("\nrun again with --cluster to group these faces into people."); } /// What the clustering proposed, largest group first. fn report_people(catalog: &Catalog) { let Ok(people) = dr_catalog::faces::people(catalog.connection()) else { return; }; if people.is_empty() { println!("no groups — too few faces, or none similar enough to group."); return; } println!("\n{} group(s):", people.len()); for p in people.iter().take(30) { let name = if p.name.is_empty() { "(unnamed)".to_string() } else { p.name.clone() }; println!( " {name:<24} {} confirmed, {} suggested", p.confirmed_faces, p.suggested_faces ); } if people.len() > 30 { println!(" … and {} more", people.len() - 30); } } /// What several merge thresholds would each do to this library. /// /// The default 0.9 is a *probability*, and the cosine it lands on depends on /// the calibration — so "is 0.9 too tight" is not a question anyone can answer /// from the number alone. This runs the real clusterer over the real /// embeddings at a range of thresholds and prints what each one produces, which /// is the only honest way to choose. /// /// Nothing is written. Run it, read the table, then pass the number you want. fn tune_thresholds(catalog: &Catalog) { use dr_catalog::faces; let conn = catalog.connection(); let cal = match faces::calibration(conn, MODEL_ID) { Ok(Some((c, _))) => c, _ => dr_face::Calibration::default(), }; let stored = match faces::embeddings(conn, MODEL_ID) { Ok(s) => s, Err(e) => { eprintln!("cannot read embeddings: {e}"); std::process::exit(1); } }; if stored.is_empty() { println!("no faces indexed yet — nothing to tune."); return; } let model = dr_face::ModelId::new(MODEL_ID.to_string()); let mut candidates = Vec::with_capacity(stored.len()); for f in stored { let Some(emb) = dr_face::Embedding::from_f16_bytes(model.clone(), &f.embedding) else { continue; }; candidates.push(dr_face::Candidate { face: f.face.0, image: f.image.0, embedding: emb.v.to_vec(), crop_px: f.crop_px, quality: f.quality, confirmed_person: None, }); } println!( "\n{} face(s), calibration valid: {}", candidates.len(), cal.valid ); println!( "\n{:>6} {:>7} {:>7} {:>7} {:>7} {:>8} {:>7}", "P", "cosine", "groups", "grouped", "largest", "in groups", "time" ); println!("{}", "-".repeat(60)); for p in [0.99_f32, 0.97, 0.95, 0.9, 0.85, 0.8, 0.75, 0.7, 0.6, 0.5] { let start = std::time::Instant::now(); let clusters = dr_face::cluster(&candidates, &cal, p); let elapsed = start.elapsed(); // A group of one is not a person, and `recluster` discards those, so // the interesting figures count only the real groups. let real: Vec<_> = clusters.iter().filter(|c| c.members.len() >= 2).collect(); let grouped: usize = real.iter().map(|c| c.members.len()).sum(); let largest = real.first().map(|c| c.members.len()).unwrap_or(0); println!( "{p:>6.2} {:>7.3} {:>7} {:>7} {:>7} {:>7.0}% {:>6.2}s", cal.boundary_at(p, 150.0, 0.0), real.len(), grouped, largest, 100.0 * grouped as f64 / candidates.len() as f64, elapsed.as_secs_f64(), ); } println!( "\nA looser threshold makes bigger groups and merges people who are not\n\ the same; a tighter one splits one person across several. The largest\n\ group is the tell: when it starts growing much faster than the rest,\n\ identities are being welded together." ); } /// How many images to sample for the quality report. /// /// Enough for the distribution to settle, few enough to finish while the user /// is watching: detection is ~100ms an image, so this is a couple of minutes. const QUALITY_SAMPLE: usize = 600; /// What the detector finds, before either quality gate is applied. /// /// The two floors — face size and sharpness — are not independent: a face /// smaller than the embedder's 112-pixel input was upsampled to reach it, and /// upsampling invents no edges, so small faces score low on sharpness even when /// the original was crisp. Choosing either number without seeing the other is /// how you end up with one gate doing nothing and the other doing too much. /// /// So this prints them together, over the real library, with nothing filtered. fn report_quality( catalog: &Catalog, store: &ThumbStore, detector: &std::path::Path, embedder: &std::path::Path, ) { let mut det = match dr_face::Detector::from_path(detector) { Ok(d) => d, Err(e) => { eprintln!("cannot load the detector: {e}"); std::process::exit(1); } }; // Loaded but unused: the point is to fail here, before a two-minute scan, // if the pair the user passed is not the pair indexing would use. if let Err(e) = dr_face::Embedder::from_path(embedder, dr_face::ModelId::new(MODEL_ID.to_string())) { eprintln!("cannot load the embedder: {e}"); std::process::exit(1); } // Everything the detector can find: no size floor, no sharpness floor. let options = dr_face::DetectOptions { min_face_px: 0.0, min_source_px: 0.0, min_sharpness: 0.0, ..Default::default() }; let mut stmt = match catalog.connection().prepare( "SELECT r.file_id FROM remote r JOIN images i ON i.id = r.image_id WHERE r.file_id IS NOT NULL AND i.trashed_at IS NULL ORDER BY i.id", ) { Ok(s) => s, Err(e) => { eprintln!("cannot list images: {e}"); std::process::exit(1); } }; let file_ids: Vec = stmt .query_map([], |r| r.get::<_, i64>(0)) .into_iter() .flatten() .filter_map(Result::ok) .map(|v| v as u64) .filter(|id| store.contains(*id, faces::FACE_TIER)) .take(QUALITY_SAMPLE) .collect(); if file_ids.is_empty() { println!("no proxies on disk to measure — browse the library first."); return; } println!("\nmeasuring {} image(s)…", file_ids.len()); // (source_px, sharpness) per detected face. let mut found: Vec<(f32, f32)> = Vec::new(); let mut images = 0usize; for id in &file_ids { let Ok(Some(thumb)) = store.get(*id, faces::FACE_TIER) else { continue; }; let Ok((w, h, rgba)) = dr_thumbs::codec::decode_rgba(&thumb.bytes) else { continue; }; let rgb: Vec = rgba .chunks_exact(4) .flat_map(|p| { [ p[0] as f32 / 255.0, p[1] as f32 / 255.0, p[2] as f32 / 255.0, ] }) .collect(); let Ok(dets) = det.detect(&rgb, w as usize, h as usize, &options) else { continue; }; images += 1; for d in &dets { if let Some(a) = dr_face::warp(&rgb, w as usize, h as usize, &d.landmarks) { found.push((a.source_px(), a.sharpness())); } } if images.is_multiple_of(50) { println!( " {images}/{} images, {} face(s)", file_ids.len(), found.len() ); } } if found.is_empty() { println!("no faces found in the sample."); return; } let pct = |v: &mut Vec, p: f64| -> f32 { v.sort_by(|a, b| a.total_cmp(b)); v[(((v.len() - 1) as f64) * p) as usize] }; let mut sizes: Vec = found.iter().map(|f| f.0).collect(); let mut sharps: Vec = found.iter().map(|f| f.1).collect(); println!("\n{} face(s) in {images} image(s)\n", found.len()); println!( "{:>12} {:>8} {:>10}", "percentile", "size px", "sharpness" ); println!("{}", "-".repeat(34)); for p in [0.01, 0.05, 0.10, 0.25, 0.50, 0.75, 0.90, 0.99] { println!( "{:>11.0}% {:>8.0} {:>10.4}", p * 100.0, pct(&mut sizes, p), pct(&mut sharps, p) ); } // What each candidate pair would remove. Cumulative, because the gates are // applied together and their overlap is the whole question. println!( "\n{:>8} {:>10} {:>9} {:>9} {:>9}", "min crop", "min sharp", "size cut", "blur cut", "kept" ); println!("{}", "-".repeat(52)); for (min_px, min_sharp) in [ (0.0_f32, 0.0_f32), (32.0, 0.0), (32.0, 0.002), (32.0, 0.005), (32.0, 0.010), (32.0, 0.020), (48.0, 0.005), (64.0, 0.010), (64.0, 0.020), ] { let by_size = found.iter().filter(|f| f.0 < min_px).count(); let by_blur = found .iter() .filter(|f| f.0 >= min_px && f.1 < min_sharp) .count(); let kept = found.len() - by_size - by_blur; println!( "{min_px:>8.0} {min_sharp:>10.3} {:>8.0}% {:>8.0}% {:>8.0}%", 100.0 * by_size as f64 / found.len() as f64, 100.0 * by_blur as f64 / found.len() as f64, 100.0 * kept as f64 / found.len() as f64, ); } println!( "\n`size cut` is what the size floor removes; `blur cut` is what the\n\ sharpness floor removes *of what the size floor left*, so the two\n\ columns do not double-count. A sharpness floor that cuts almost\n\ nothing once the size floor is in place is a floor that is not\n\ earning its place." ); }