Files
dtourolle 84fade99ec Put the developer docs under docs/dev and index the folder for users first
docs/ had 26 developer documents flat beside the manual, and the two
audiences are very differently sized: most readers want the manual and
the gesture reference, a few want the register, the designs and the
measurements. The manual and gestures.md stay at the top; everything for
someone changing the code moves to docs/dev/, and the two documents that
name their own successors — the v0.1 milestone and the UI-refinement plan
— go to docs/dev/archive/ rather than being deleted, since both are still
cited. docs/README.md is the index, users first.

Every reference follows: code comments, Cargo manifests, the workflows,
the pre-commit hook, the bench and traceability tools (which locate the
repo root by docs/dev/requirements.md now), packaging, the Docker READMEs,
CLAUDE.md, CONTRIBUTING.md and the README. The matrix links one level
deeper and is regenerated. Links out of the moved documents into the tree
gain a level; a link checker over every Markdown file finds none broken.
2026-09-20 21:16:03 +02:00

540 lines
19 KiB
Rust

//! The face-indexing batch job, off the GUI.
//!
//! Checks every library image for a face-detection run marker, and optionally
//! indexes whatever is missing one.
//!
//! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR [--run DET.onnx EMB.onnx]
//! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR --cluster
//!
//! # Why this exists beside the button in the Identity screen
//!
//! Indexing a real library is hours of work (docs/dev/faces.md §12.2), and the
//! cases where that is worth starting — an overnight pass, a fresh import, a
//! machine left running — are exactly the ones where holding a window open is
//! the wrong shape. The check half is useful on its own: it is cheap, it
//! answers "has face recognition been over all of this", and it distinguishes
//! *not yet indexed* from *waiting on a proxy*, which are different problems
//! with different fixes.
//!
//! The models must have had their input dims frozen first; see
//! `tools/fix-face-model-shapes.sh`.
use std::path::PathBuf;
use dr_catalog::Catalog;
use dr_thumbs::ThumbStore;
use dr_ui::faces::{self, FaceSweepMessage};
const MODEL_ID: &str = "w600k_mbf";
fn main() {
env_logger::init();
let args: Vec<String> = std::env::args().skip(1).collect();
if args.len() < 2 {
eprintln!(
"usage: face_index CATALOG.db THUMBS_DIR [--run DETECTOR.onnx EMBEDDER.onnx]\n\
\n\
With no --run this only reports; nothing is written.\n\
--cluster groups what is indexed; --tune compares thresholds without writing.\n\
--quality DET.onnx EMB.onnx reports face size and sharpness, also without writing."
);
std::process::exit(2);
}
let catalog_path = PathBuf::from(&args[0]);
let store_dir = PathBuf::from(&args[1]);
let catalog = match Catalog::open(&catalog_path) {
Ok(c) => c,
Err(e) => {
eprintln!("cannot open catalog {}: {e}", catalog_path.display());
std::process::exit(1);
}
};
let store = match ThumbStore::open(&store_dir) {
Ok(s) => s,
Err(e) => {
eprintln!("cannot open thumbnail store {}: {e}", store_dir.display());
std::process::exit(1);
}
};
// The registry a device without the eye models would run: what this
// example counts as work is what the app would.
let repairs = dr_ui::repairs::registry(
dr_ui::repairs::Scope::Outstanding,
MODEL_ID,
dr_types::FaceDetector::Scrfd500m,
dr_ui::repairs::Capabilities {
gpu: true,
face_models: true,
eye_models: false,
},
);
let audit = match faces::audit(&catalog, &store, MODEL_ID, &repairs) {
Ok(a) => a,
Err(e) => {
eprintln!("coverage check failed: {e}");
std::process::exit(1);
}
};
println!("model {MODEL_ID}");
println!("images {}", audit.coverage.images);
println!(
"indexed {} ({:.1}%)",
audit.coverage.indexed,
audit.coverage.fraction() * 100.0
);
println!("faces {}", audit.coverage.faces);
println!(
"no faces {} (indexed, nothing found — the common case)",
audit.coverage.without_faces
);
println!("outstanding {}", audit.coverage.outstanding());
println!(" ready to index {}", audit.ready);
println!(" awaiting proxy {}", audit.awaiting_proxy);
// Grouping is a separate step from indexing on purpose: it is a
// whole-library operation over the embeddings detection produced, and it is
// worth running *after* a sweep rather than during one (catalog.md §10.2).
if args.iter().any(|a| a == "--cluster") {
// The engine's own defaults, not this device's settings file: a batch
// job run over a library on a server has no business inheriting the
// dials somebody moved on their laptop.
let grouping = dr_types::settings::FaceSettings::default();
match dr_ui::faces::recluster(&catalog, MODEL_ID, &grouping) {
Ok((suggested, created)) => {
println!("\nclustering: {suggested} suggestion(s), {created} new group(s)");
report_people(&catalog);
}
Err(e) => {
eprintln!("clustering failed: {e}");
std::process::exit(1);
}
}
return;
}
// Tuning, and deliberately read-only: it answers "what would this
// threshold do to my library" without writing a single suggestion, which
// is the only way to compare several without each one polluting the next.
if args.iter().any(|a| a == "--tune") {
tune_thresholds(&catalog);
return;
}
// Quality tuning, and read-only like `--tune`. Runs the real detector over
// real proxies with **both gates disabled**, so the distribution it prints
// is of everything the detector finds rather than of what survives the
// current settings — which is the only way to see what a threshold would
// actually remove.
if let Some(i) = args.iter().position(|a| a == "--quality") {
let (Some(detector), Some(embedder)) = (args.get(i + 1), args.get(i + 2)) else {
eprintln!("--quality needs both a detector and an embedder");
std::process::exit(2);
};
report_quality(
&catalog,
&store,
std::path::Path::new(detector),
std::path::Path::new(embedder),
);
return;
}
let run = args.iter().position(|a| a == "--run");
let Some(i) = run else {
if audit.coverage.is_complete() {
println!("\nnothing outstanding.");
} else {
println!("\npass --run DETECTOR.onnx EMBEDDER.onnx to index the outstanding images.");
}
return;
};
let (Some(detector), Some(embedder)) = (args.get(i + 1), args.get(i + 2)) else {
eprintln!("--run needs both a detector and an embedder");
std::process::exit(2);
};
if audit.ready == 0 {
println!("\nnothing ready to index.");
if audit.awaiting_proxy > 0 {
// Worth saying plainly: running this again will not help, because
// the blocker is in the thumbnail store rather than here.
println!(
"{} image(s) are waiting on a proxy — run the thumbnail sweep first.",
audit.awaiting_proxy
);
}
return;
}
// Say it here rather than letting the pass return an empty result and
// print "done: 0 image(s)". That is the shape of report this whole change
// exists to stop producing.
if faces::FACE_TIER.edge() < dr_face::MIN_CROP_EDGE {
println!(
"\n--run cannot index from the local store any more.\n\
\n\
Face crops must be sampled from a buffer longer than {}px and the\n\
store's largest tier is {}px, so every image here would be refused.\n\
The measurement behind that floor is docs/dev/faces.md §7b.\n\
\n\
Index from the app's Identity screen instead: that pass renders at\n\
native resolution, which is what the crop needs.",
dr_face::MIN_CROP_EDGE - 1,
faces::FACE_TIER.edge(),
);
return;
}
println!("\nindexing {} image(s)…", audit.ready);
let rx = faces::spawn_store_face_sweep(
catalog_path,
store_dir,
dr_ui::FaceModelPaths {
detector: PathBuf::from(detector),
embedder: PathBuf::from(embedder),
// A measurement tool for the detector and embedder; the eye
// models are the library pass's business.
eyes: None,
},
MODEL_ID.to_string(),
dr_face::DetectOptions::default(),
);
let mut seen = 0usize;
let mut total = 0usize;
let start = std::time::Instant::now();
for msg in rx {
match msg {
FaceSweepMessage::Total(n) => total = n,
FaceSweepMessage::Indexed { faces, .. } => {
seen += 1;
// One line per image would be thousands of lines; one per
// twenty-five is enough to see it moving and to estimate.
if seen.is_multiple_of(25) || faces > 0 {
let rate = seen as f64 / start.elapsed().as_secs_f64().max(1e-6);
println!(
" {seen}/{total} {:.2} img/s ~{:.0} min left",
rate,
(total.saturating_sub(seen)) as f64 / rate.max(1e-6) / 60.0
);
}
}
FaceSweepMessage::Failed { images } => seen += images,
FaceSweepMessage::Finished {
images,
faces,
failed,
} => {
println!(
"\ndone: {images} image(s), {faces} face(s), {failed} failed, in {:.0}s",
start.elapsed().as_secs_f64()
);
}
}
}
if let Ok(a) = faces::audit(&catalog, &store, MODEL_ID, &repairs) {
println!("{}", a.summary());
}
println!("\nrun again with --cluster to group these faces into people.");
}
/// What the clustering proposed, largest group first.
fn report_people(catalog: &Catalog) {
let Ok(people) = dr_catalog::faces::people(catalog.connection()) else {
return;
};
if people.is_empty() {
println!("no groups — too few faces, or none similar enough to group.");
return;
}
println!("\n{} group(s):", people.len());
for p in people.iter().take(30) {
let name = if p.name.is_empty() {
"(unnamed)".to_string()
} else {
p.name.clone()
};
println!(
" {name:<24} {} confirmed, {} suggested",
p.confirmed_faces, p.suggested_faces
);
}
if people.len() > 30 {
println!(" … and {} more", people.len() - 30);
}
}
/// What several merge thresholds would each do to this library.
///
/// The default 0.9 is a *probability*, and the cosine it lands on depends on
/// the calibration — so "is 0.9 too tight" is not a question anyone can answer
/// from the number alone. This runs the real clusterer over the real
/// embeddings at a range of thresholds and prints what each one produces, which
/// is the only honest way to choose.
///
/// Nothing is written. Run it, read the table, then pass the number you want.
fn tune_thresholds(catalog: &Catalog) {
use dr_catalog::faces;
let conn = catalog.connection();
let cal = match faces::calibration(conn, MODEL_ID) {
Ok(Some((c, _))) => c,
_ => dr_face::Calibration::default(),
};
let stored = match faces::embeddings(conn, MODEL_ID) {
Ok(s) => s,
Err(e) => {
eprintln!("cannot read embeddings: {e}");
std::process::exit(1);
}
};
if stored.is_empty() {
println!("no faces indexed yet — nothing to tune.");
return;
}
let model = dr_face::ModelId::new(MODEL_ID.to_string());
let mut candidates = Vec::with_capacity(stored.len());
for f in stored {
let Some(emb) = dr_face::Embedding::from_f16_bytes(model.clone(), &f.embedding) else {
continue;
};
candidates.push(dr_face::Candidate {
face: f.face.0,
image: f.image.0,
embedding: emb.v.to_vec(),
crop_px: f.crop_px,
quality: f.quality,
confirmed_person: None,
});
}
println!(
"\n{} face(s), calibration valid: {}",
candidates.len(),
cal.valid
);
println!(
"\n{:>6} {:>7} {:>7} {:>7} {:>7} {:>8} {:>7}",
"P", "cosine", "groups", "grouped", "largest", "in groups", "time"
);
println!("{}", "-".repeat(60));
for p in [0.99_f32, 0.97, 0.95, 0.9, 0.85, 0.8, 0.75, 0.7, 0.6, 0.5] {
let start = std::time::Instant::now();
let clusters = dr_face::cluster(&candidates, &cal, p);
let elapsed = start.elapsed();
// A group of one is not a person, and `recluster` discards those, so
// the interesting figures count only the real groups.
let real: Vec<_> = clusters.iter().filter(|c| c.members.len() >= 2).collect();
let grouped: usize = real.iter().map(|c| c.members.len()).sum();
let largest = real.first().map(|c| c.members.len()).unwrap_or(0);
println!(
"{p:>6.2} {:>7.3} {:>7} {:>7} {:>7} {:>7.0}% {:>6.2}s",
cal.boundary_at(p, 150.0, 0.0),
real.len(),
grouped,
largest,
100.0 * grouped as f64 / candidates.len() as f64,
elapsed.as_secs_f64(),
);
}
println!(
"\nA looser threshold makes bigger groups and merges people who are not\n\
the same; a tighter one splits one person across several. The largest\n\
group is the tell: when it starts growing much faster than the rest,\n\
identities are being welded together."
);
}
/// How many images to sample for the quality report.
///
/// Enough for the distribution to settle, few enough to finish while the user
/// is watching: detection is ~100ms an image, so this is a couple of minutes.
const QUALITY_SAMPLE: usize = 600;
/// What the detector finds, before either quality gate is applied.
///
/// The two floors — face size and sharpness — are not independent: a face
/// smaller than the embedder's 112-pixel input was upsampled to reach it, and
/// upsampling invents no edges, so small faces score low on sharpness even when
/// the original was crisp. Choosing either number without seeing the other is
/// how you end up with one gate doing nothing and the other doing too much.
///
/// So this prints them together, over the real library, with nothing filtered.
fn report_quality(
catalog: &Catalog,
store: &ThumbStore,
detector: &std::path::Path,
embedder: &std::path::Path,
) {
let mut det = match dr_face::Detector::from_path(detector) {
Ok(d) => d,
Err(e) => {
eprintln!("cannot load the detector: {e}");
std::process::exit(1);
}
};
// Loaded but unused: the point is to fail here, before a two-minute scan,
// if the pair the user passed is not the pair indexing would use.
if let Err(e) =
dr_face::Embedder::from_path(embedder, dr_face::ModelId::new(MODEL_ID.to_string()))
{
eprintln!("cannot load the embedder: {e}");
std::process::exit(1);
}
// Everything the detector can find: no size floor, no sharpness floor.
let options = dr_face::DetectOptions {
min_face_px: 0.0,
min_source_px: 0.0,
min_sharpness: 0.0,
..Default::default()
};
let mut stmt = match catalog.connection().prepare(
"SELECT r.file_id FROM remote r
JOIN images i ON i.id = r.image_id
WHERE r.file_id IS NOT NULL AND i.trashed_at IS NULL
ORDER BY i.id",
) {
Ok(s) => s,
Err(e) => {
eprintln!("cannot list images: {e}");
std::process::exit(1);
}
};
let file_ids: Vec<u64> = stmt
.query_map([], |r| r.get::<_, i64>(0))
.into_iter()
.flatten()
.filter_map(Result::ok)
.map(|v| v as u64)
.filter(|id| store.contains(*id, faces::FACE_TIER))
.take(QUALITY_SAMPLE)
.collect();
if file_ids.is_empty() {
println!("no proxies on disk to measure — browse the library first.");
return;
}
println!("\nmeasuring {} image(s)…", file_ids.len());
// (source_px, sharpness) per detected face.
let mut found: Vec<(f32, f32)> = Vec::new();
let mut images = 0usize;
for id in &file_ids {
let Ok(Some(thumb)) = store.get(*id, faces::FACE_TIER) else {
continue;
};
let Ok((w, h, rgba)) = dr_thumbs::codec::decode_rgba(&thumb.bytes) else {
continue;
};
let rgb: Vec<f32> = rgba
.chunks_exact(4)
.flat_map(|p| {
[
p[0] as f32 / 255.0,
p[1] as f32 / 255.0,
p[2] as f32 / 255.0,
]
})
.collect();
let Ok(dets) = det.detect(&rgb, w as usize, h as usize, &options) else {
continue;
};
images += 1;
for d in &dets {
if let Some(a) = dr_face::warp(&rgb, w as usize, h as usize, &d.landmarks) {
found.push((a.source_px(), a.sharpness()));
}
}
if images.is_multiple_of(50) {
println!(
" {images}/{} images, {} face(s)",
file_ids.len(),
found.len()
);
}
}
if found.is_empty() {
println!("no faces found in the sample.");
return;
}
let pct = |v: &mut Vec<f32>, p: f64| -> f32 {
v.sort_by(|a, b| a.total_cmp(b));
v[(((v.len() - 1) as f64) * p) as usize]
};
let mut sizes: Vec<f32> = found.iter().map(|f| f.0).collect();
let mut sharps: Vec<f32> = found.iter().map(|f| f.1).collect();
println!("\n{} face(s) in {images} image(s)\n", found.len());
println!(
"{:>12} {:>8} {:>10}",
"percentile", "size px", "sharpness"
);
println!("{}", "-".repeat(34));
for p in [0.01, 0.05, 0.10, 0.25, 0.50, 0.75, 0.90, 0.99] {
println!(
"{:>11.0}% {:>8.0} {:>10.4}",
p * 100.0,
pct(&mut sizes, p),
pct(&mut sharps, p)
);
}
// What each candidate pair would remove. Cumulative, because the gates are
// applied together and their overlap is the whole question.
println!(
"\n{:>8} {:>10} {:>9} {:>9} {:>9}",
"min crop", "min sharp", "size cut", "blur cut", "kept"
);
println!("{}", "-".repeat(52));
for (min_px, min_sharp) in [
(0.0_f32, 0.0_f32),
(32.0, 0.0),
(32.0, 0.002),
(32.0, 0.005),
(32.0, 0.010),
(32.0, 0.020),
(48.0, 0.005),
(64.0, 0.010),
(64.0, 0.020),
] {
let by_size = found.iter().filter(|f| f.0 < min_px).count();
let by_blur = found
.iter()
.filter(|f| f.0 >= min_px && f.1 < min_sharp)
.count();
let kept = found.len() - by_size - by_blur;
println!(
"{min_px:>8.0} {min_sharp:>10.3} {:>8.0}% {:>8.0}% {:>8.0}%",
100.0 * by_size as f64 / found.len() as f64,
100.0 * by_blur as f64 / found.len() as f64,
100.0 * kept as f64 / found.len() as f64,
);
}
println!(
"\n`size cut` is what the size floor removes; `blur cut` is what the\n\
sharpness floor removes *of what the size floor left*, so the two\n\
columns do not double-count. A sharpness floor that cuts almost\n\
nothing once the size floor is in place is a floor that is not\n\
earning its place."
);
}