Files
DarkRoom/ui/dr-ui/examples/face_index.rs
T
dtourolleandClaude Opus 5 7275c020d7 Group people at the threshold the library actually supports
0.90 left a third of the reference library ungrouped: 1,213 of 1,813 faces in a
group, and the rest sitting alone in a screen that had nothing to offer for
them.

"Is 0.90 too tight" is not answerable from the number. It is a probability, and
which cosine it lands on depends on the calibration — so the first half of this
is a way to ask the question properly. `face_index --tune` runs the real
clusterer over the real embeddings at ten thresholds and prints what each one
produces. It writes nothing; comparing thresholds by applying them would have
each one pollute the next.

On the reference library:

      P   cosine   groups  grouped  largest
   0.95    0.449      311      62%       51
   0.90    0.403      316      67%       51
   0.85    0.374      318      70%       57
   0.80    0.353      328      74%       69
   0.75    0.335      327      77%       69
   0.70    0.319      326      79%       81
   0.50    0.267      303      85%       90

The count of *groups* is the signal, not the count of grouped faces. Loosening
from 0.95 makes it climb: real people are being assembled out of fragments. It
peaks at 0.80 and then falls — and a falling group count while the grouped faces
keep rising is the shape of over-merging, separate identities being welded
together. That is the FR-CULL-10 failure, and the one the user cannot undo by
hand.

So 0.80: the loosest setting still building people rather than melting them
together. A third more of the library gets grouped than at 0.90, and the largest
group grows by eighteen faces rather than by forty.

The table is one library, and the doc comment says so — `--tune` reruns it on
any other.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-27 21:17:16 +02:00

298 lines
10 KiB
Rust

//! The face-indexing batch job, off the GUI.
//!
//! Checks every library image for a face-detection run marker, and optionally
//! indexes whatever is missing one.
//!
//! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR [--run DET.onnx EMB.onnx]
//! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR --cluster
//!
//! # Why this exists beside the button in the Identity screen
//!
//! Indexing a real library is hours of work (docs/faces.md §12.2), and the
//! cases where that is worth starting — an overnight pass, a fresh import, a
//! machine left running — are exactly the ones where holding a window open is
//! the wrong shape. The check half is useful on its own: it is cheap, it
//! answers "has face recognition been over all of this", and it distinguishes
//! *not yet indexed* from *waiting on a proxy*, which are different problems
//! with different fixes.
//!
//! The models must have had their input dims frozen first; see
//! `tools/fix-face-model-shapes.sh`.
use std::path::PathBuf;
use dr_catalog::Catalog;
use dr_thumbs::ThumbStore;
use dr_ui::faces::{self, FaceSweepMessage};
const MODEL_ID: &str = "w600k_mbf";
fn main() {
env_logger::init();
let args: Vec<String> = std::env::args().skip(1).collect();
if args.len() < 2 {
eprintln!(
"usage: face_index CATALOG.db THUMBS_DIR [--run DETECTOR.onnx EMBEDDER.onnx]\n\
\n\
With no --run this only reports; nothing is written.\n\
--cluster groups what is indexed; --tune compares thresholds without writing."
);
std::process::exit(2);
}
let catalog_path = PathBuf::from(&args[0]);
let store_dir = PathBuf::from(&args[1]);
let catalog = match Catalog::open(&catalog_path) {
Ok(c) => c,
Err(e) => {
eprintln!("cannot open catalog {}: {e}", catalog_path.display());
std::process::exit(1);
}
};
let store = match ThumbStore::open(&store_dir) {
Ok(s) => s,
Err(e) => {
eprintln!("cannot open thumbnail store {}: {e}", store_dir.display());
std::process::exit(1);
}
};
let audit = match faces::audit(&catalog, &store, MODEL_ID) {
Ok(a) => a,
Err(e) => {
eprintln!("coverage check failed: {e}");
std::process::exit(1);
}
};
println!("model {MODEL_ID}");
println!("images {}", audit.coverage.images);
println!(
"indexed {} ({:.1}%)",
audit.coverage.indexed,
audit.coverage.fraction() * 100.0
);
println!("faces {}", audit.coverage.faces);
println!(
"no faces {} (indexed, nothing found — the common case)",
audit.coverage.without_faces
);
println!("outstanding {}", audit.coverage.outstanding());
println!(" ready to index {}", audit.ready);
println!(" awaiting proxy {}", audit.awaiting_proxy);
// Grouping is a separate step from indexing on purpose: it is a
// whole-library operation over the embeddings detection produced, and it is
// worth running *after* a sweep rather than during one (catalog.md §10.2).
if args.iter().any(|a| a == "--cluster") {
match dr_ui::faces::recluster(&catalog, MODEL_ID, dr_face::DEFAULT_MERGE_PROBABILITY) {
Ok((suggested, created)) => {
println!("\nclustering: {suggested} suggestion(s), {created} new group(s)");
report_people(&catalog);
}
Err(e) => {
eprintln!("clustering failed: {e}");
std::process::exit(1);
}
}
return;
}
// Tuning, and deliberately read-only: it answers "what would this
// threshold do to my library" without writing a single suggestion, which
// is the only way to compare several without each one polluting the next.
if args.iter().any(|a| a == "--tune") {
tune_thresholds(&catalog);
return;
}
let run = args.iter().position(|a| a == "--run");
let Some(i) = run else {
if audit.coverage.is_complete() {
println!("\nnothing outstanding.");
} else {
println!("\npass --run DETECTOR.onnx EMBEDDER.onnx to index the outstanding images.");
}
return;
};
let (Some(detector), Some(embedder)) = (args.get(i + 1), args.get(i + 2)) else {
eprintln!("--run needs both a detector and an embedder");
std::process::exit(2);
};
if audit.ready == 0 {
println!("\nnothing ready to index.");
if audit.awaiting_proxy > 0 {
// Worth saying plainly: running this again will not help, because
// the blocker is in the thumbnail store rather than here.
println!(
"{} image(s) are waiting on a proxy — run the thumbnail sweep first.",
audit.awaiting_proxy
);
}
return;
}
println!("\nindexing {} image(s)…", audit.ready);
let rx = faces::spawn_store_face_sweep(
catalog_path,
store_dir,
PathBuf::from(detector),
PathBuf::from(embedder),
MODEL_ID.to_string(),
dr_face::DetectOptions::default(),
);
let mut seen = 0usize;
let mut total = 0usize;
let start = std::time::Instant::now();
for msg in rx {
match msg {
FaceSweepMessage::Total(n) => total = n,
FaceSweepMessage::Indexed { faces, .. } => {
seen += 1;
// One line per image would be thousands of lines; one per
// twenty-five is enough to see it moving and to estimate.
if seen.is_multiple_of(25) || faces > 0 {
let rate = seen as f64 / start.elapsed().as_secs_f64().max(1e-6);
println!(
" {seen}/{total} {:.2} img/s ~{:.0} min left",
rate,
(total.saturating_sub(seen)) as f64 / rate.max(1e-6) / 60.0
);
}
}
FaceSweepMessage::Finished {
images,
faces,
failed,
} => {
println!(
"\ndone: {images} image(s), {faces} face(s), {failed} failed, in {:.0}s",
start.elapsed().as_secs_f64()
);
}
}
}
if let Ok(a) = faces::audit(&catalog, &store, MODEL_ID) {
println!("{}", a.summary());
}
println!("\nrun again with --cluster to group these faces into people.");
}
/// What the clustering proposed, largest group first.
fn report_people(catalog: &Catalog) {
let Ok(people) = dr_catalog::faces::people(catalog.connection()) else {
return;
};
if people.is_empty() {
println!("no groups — too few faces, or none similar enough to group.");
return;
}
println!("\n{} group(s):", people.len());
for p in people.iter().take(30) {
let name = if p.name.is_empty() {
"(unnamed)".to_string()
} else {
p.name.clone()
};
println!(
" {name:<24} {} confirmed, {} suggested",
p.confirmed_faces, p.suggested_faces
);
}
if people.len() > 30 {
println!(" … and {} more", people.len() - 30);
}
}
/// What several merge thresholds would each do to this library.
///
/// The default 0.9 is a *probability*, and the cosine it lands on depends on
/// the calibration — so "is 0.9 too tight" is not a question anyone can answer
/// from the number alone. This runs the real clusterer over the real
/// embeddings at a range of thresholds and prints what each one produces, which
/// is the only honest way to choose.
///
/// Nothing is written. Run it, read the table, then pass the number you want.
fn tune_thresholds(catalog: &Catalog) {
use dr_catalog::faces;
let conn = catalog.connection();
let cal = match faces::calibration(conn, MODEL_ID) {
Ok(Some((c, _))) => c,
_ => dr_face::Calibration::default(),
};
let stored = match faces::embeddings(conn, MODEL_ID) {
Ok(s) => s,
Err(e) => {
eprintln!("cannot read embeddings: {e}");
std::process::exit(1);
}
};
if stored.is_empty() {
println!("no faces indexed yet — nothing to tune.");
return;
}
let model = dr_face::ModelId::new(MODEL_ID.to_string());
let mut candidates = Vec::with_capacity(stored.len());
for (face_id, image_id, blob, crop_px) in stored {
let Some(emb) = dr_face::Embedding::from_f16_bytes(model.clone(), &blob) else {
continue;
};
candidates.push(dr_face::Candidate {
face: face_id.0,
image: image_id.0,
embedding: emb.v.to_vec(),
crop_px,
confirmed_person: None,
});
}
println!(
"\n{} face(s), calibration valid: {}",
candidates.len(),
cal.valid
);
println!(
"\n{:>6} {:>7} {:>7} {:>7} {:>7} {:>8} {:>7}",
"P", "cosine", "groups", "grouped", "largest", "in groups", "time"
);
println!("{}", "-".repeat(60));
for p in [0.99_f32, 0.97, 0.95, 0.9, 0.85, 0.8, 0.75, 0.7, 0.6, 0.5] {
let start = std::time::Instant::now();
let clusters = dr_face::cluster(&candidates, &cal, p);
let elapsed = start.elapsed();
// A group of one is not a person, and `recluster` discards those, so
// the interesting figures count only the real groups.
let real: Vec<_> = clusters.iter().filter(|c| c.members.len() >= 2).collect();
let grouped: usize = real.iter().map(|c| c.members.len()).sum();
let largest = real.first().map(|c| c.members.len()).unwrap_or(0);
println!(
"{p:>6.2} {:>7.3} {:>7} {:>7} {:>7} {:>7.0}% {:>6.2}s",
cal.boundary_at(p, 150.0, 0.0),
real.len(),
grouped,
largest,
100.0 * grouped as f64 / candidates.len() as f64,
elapsed.as_secs_f64(),
);
}
println!(
"\nA looser threshold makes bigger groups and merges people who are not\n\
the same; a tighter one splits one person across several. The largest\n\
group is the tell: when it starts growing much faster than the rest,\n\
identities are being welded together."
);
}