//! What one click on the Identity screen costs, off the GUI. //! //! cargo run --release -p dr-ui --example identity_bench -- CATALOG.sqlite THUMBS_DIR //! //! Times each read the screen performs after a confirm, a reject or a //! split — the people rail, the selected person's faces with their crops, //! the coverage line — and the two batch writes, against a *copy* of a real //! catalog. It writes to the catalog it is given (the batch operations are //! the point), so never hand it the library's own file. //! //! The figures are wall-clock on this machine and this library, for reading //! side by side before and after a change; they are not a gate. use std::path::PathBuf; use std::time::{Duration, Instant}; use dr_catalog::faces::{self, PersonId}; use dr_catalog::Catalog; use dr_thumbs::ThumbStore; use dr_ui::identity; use dr_ui::repairs::{self, Capabilities, Scope}; const MODEL_ID: &str = "scrfd_10g+w600k_mbf"; fn main() { let args: Vec = std::env::args().skip(1).collect(); if args.len() < 2 { eprintln!("usage: identity_bench CATALOG.sqlite THUMBS_DIR"); std::process::exit(2); } let catalog = Catalog::open(&PathBuf::from(&args[0])).expect("catalog"); let store = ThumbStore::open(&PathBuf::from(&args[1])).expect("thumbs"); let conn = catalog.connection(); // The person with the most faces: the worst case for the face grid, and // the one a user is likeliest to be confirming through. let (biggest, n_faces): (i64, i64) = conn .query_row( "SELECT person_id, COUNT(*) c FROM face_person GROUP BY 1 ORDER BY c DESC LIMIT 1", [], |r| Ok((r.get(0)?, r.get(1)?)), ) .expect("a person"); let biggest = PersonId(biggest as u64); println!("largest person {biggest:?} holds {n_faces} faces\n"); // ── the reads a click triggers ──────────────────────────────────── let detector = dr_types::FaceDetector::for_model_id(MODEL_ID).unwrap_or_default(); let registry = repairs::registry( Scope::Outstanding, MODEL_ID, detector, Capabilities { gpu: true, face_models: true, eye_models: true, }, ); time("load_people", 5, || { identity::load_people(&catalog, MODEL_ID).unwrap(); }); time("load_faces (largest person)", 5, || { identity::load_faces(&catalog, &store, biggest).unwrap(); }); time("audit (coverage line)", 5, || { dr_ui::faces::audit(&catalog, &store, MODEL_ID, ®istry).unwrap(); }); // ── the batch writes ────────────────────────────────────────────── // Every face of the largest person is demoted to a suggestion, then // confirmed back in one call, so the measurement covers the whole group. let ids: Vec = { let mut q = conn .prepare("SELECT face_id FROM face_person WHERE person_id = ?1") .unwrap(); q.query_map([biggest.0 as i64], |r| r.get(0)) .unwrap() .collect::>() .unwrap() }; let demote = |conn: &rusqlite::Connection| { conn.execute( "UPDATE face_person SET confirmed = 0, probability = 0.5 WHERE person_id = ?1", [biggest.0 as i64], ) .unwrap(); }; demote(conn); time("confirm_all (largest person)", 3, || { demote(conn); identity::confirm_all(&catalog, biggest).unwrap(); }); // Split the group onto a new person and fold it straight back, so the // catalog ends where it started apart from the redirect rows. let members: Vec = ids.iter().map(|&i| faces::FaceId(i as u64)).collect(); time("split_off (largest person, all faces)", 3, || { let new = identity::split_off(&catalog, biggest, &members, "").unwrap(); faces::merge_people(conn, biggest, new).unwrap(); }); // The split rejects every face from `biggest`; a merge back does not // clear that, so clear it here or a later run measures a different table. conn.execute( "DELETE FROM face_person_rejected WHERE person_id = ?1", [biggest.0 as i64], ) .unwrap(); } /// Run `f` a few times and print the best wall-clock, the median, and the /// best CPU time. The best is what the code costs, the median is what the /// user waits — and the CPU figure is the one to compare across runs, since /// this path is single-threaded and a build running on the same machine /// doubles the wall clock without touching it. fn time(label: &str, runs: usize, mut f: impl FnMut()) { let mut wall: Vec = Vec::with_capacity(runs); let mut cpu: Vec = Vec::with_capacity(runs); for _ in 0..runs { let c = cpu_now(); let t = Instant::now(); f(); wall.push(t.elapsed()); cpu.push(cpu_now().saturating_sub(c)); } wall.sort(); cpu.sort(); println!( "{label:42} best {:8.1} ms median {:8.1} ms cpu {:8.1} ms", wall[0].as_secs_f64() * 1e3, wall[runs / 2].as_secs_f64() * 1e3, cpu[0].as_secs_f64() * 1e3 ); } /// This thread's time on a CPU so far, from the scheduler's own account. /// /// `/proc/self/schedstat` is the main thread's; the bench runs everything on /// it. Zero where the file is missing, which only makes the CPU column /// useless rather than the run. fn cpu_now() -> Duration { std::fs::read_to_string("/proc/self/schedstat") .ok() .and_then(|s| s.split_whitespace().next()?.parse::().ok()) .map(Duration::from_nanos) .unwrap_or_default() }