The embedder's raw output has a length, and the length is a reading of how recognisable the crop was: a blur, an occlusion or a hard profile comes out short. Normalising threw it away. A short vector sits near the middle of the sphere and matches a little of everyone, which is how one bad crop bridges two people in a grouping pass. So the length is kept — the store now holds the raw vector, re-normalised on load, with the length beside it as `faces.quality` — and a face under MIN_GALLERY_QUALITY (14) is a probe: measured against the gallery and placed where it fits, but never what another face is measured against. Two probes are never paired, and a probe is nobody's evidence for a confidence. The People screen shows the number as "Quality 17.3", dimmed below the floor. Faces indexed before this stored unit vectors and have no reading; they are admitted to the gallery, and schema V14 forgets the run marker of every image holding one so the next indexing pass measures them. A peer's unmeasured shard faces are not adopted, or a sync would write that marker back.
523 lines
18 KiB
Rust
523 lines
18 KiB
Rust
//! The face-indexing batch job, off the GUI.
|
|
//!
|
|
//! Checks every library image for a face-detection run marker, and optionally
|
|
//! indexes whatever is missing one.
|
|
//!
|
|
//! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR [--run DET.onnx EMB.onnx]
|
|
//! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR --cluster
|
|
//!
|
|
//! # Why this exists beside the button in the Identity screen
|
|
//!
|
|
//! Indexing a real library is hours of work (docs/faces.md §12.2), and the
|
|
//! cases where that is worth starting — an overnight pass, a fresh import, a
|
|
//! machine left running — are exactly the ones where holding a window open is
|
|
//! the wrong shape. The check half is useful on its own: it is cheap, it
|
|
//! answers "has face recognition been over all of this", and it distinguishes
|
|
//! *not yet indexed* from *waiting on a proxy*, which are different problems
|
|
//! with different fixes.
|
|
//!
|
|
//! The models must have had their input dims frozen first; see
|
|
//! `tools/fix-face-model-shapes.sh`.
|
|
|
|
use std::path::PathBuf;
|
|
|
|
use dr_catalog::Catalog;
|
|
use dr_thumbs::ThumbStore;
|
|
use dr_ui::faces::{self, FaceSweepMessage};
|
|
|
|
const MODEL_ID: &str = "w600k_mbf";
|
|
|
|
fn main() {
|
|
env_logger::init();
|
|
|
|
let args: Vec<String> = std::env::args().skip(1).collect();
|
|
if args.len() < 2 {
|
|
eprintln!(
|
|
"usage: face_index CATALOG.db THUMBS_DIR [--run DETECTOR.onnx EMBEDDER.onnx]\n\
|
|
\n\
|
|
With no --run this only reports; nothing is written.\n\
|
|
--cluster groups what is indexed; --tune compares thresholds without writing.\n\
|
|
--quality DET.onnx EMB.onnx reports face size and sharpness, also without writing."
|
|
);
|
|
std::process::exit(2);
|
|
}
|
|
|
|
let catalog_path = PathBuf::from(&args[0]);
|
|
let store_dir = PathBuf::from(&args[1]);
|
|
|
|
let catalog = match Catalog::open(&catalog_path) {
|
|
Ok(c) => c,
|
|
Err(e) => {
|
|
eprintln!("cannot open catalog {}: {e}", catalog_path.display());
|
|
std::process::exit(1);
|
|
}
|
|
};
|
|
let store = match ThumbStore::open(&store_dir) {
|
|
Ok(s) => s,
|
|
Err(e) => {
|
|
eprintln!("cannot open thumbnail store {}: {e}", store_dir.display());
|
|
std::process::exit(1);
|
|
}
|
|
};
|
|
|
|
let audit = match faces::audit(&catalog, &store, MODEL_ID) {
|
|
Ok(a) => a,
|
|
Err(e) => {
|
|
eprintln!("coverage check failed: {e}");
|
|
std::process::exit(1);
|
|
}
|
|
};
|
|
|
|
println!("model {MODEL_ID}");
|
|
println!("images {}", audit.coverage.images);
|
|
println!(
|
|
"indexed {} ({:.1}%)",
|
|
audit.coverage.indexed,
|
|
audit.coverage.fraction() * 100.0
|
|
);
|
|
println!("faces {}", audit.coverage.faces);
|
|
println!(
|
|
"no faces {} (indexed, nothing found — the common case)",
|
|
audit.coverage.without_faces
|
|
);
|
|
println!("outstanding {}", audit.coverage.outstanding());
|
|
println!(" ready to index {}", audit.ready);
|
|
println!(" awaiting proxy {}", audit.awaiting_proxy);
|
|
|
|
// Grouping is a separate step from indexing on purpose: it is a
|
|
// whole-library operation over the embeddings detection produced, and it is
|
|
// worth running *after* a sweep rather than during one (catalog.md §10.2).
|
|
if args.iter().any(|a| a == "--cluster") {
|
|
// The engine's own defaults, not this device's settings file: a batch
|
|
// job run over a library on a server has no business inheriting the
|
|
// dials somebody moved on their laptop.
|
|
let grouping = dr_types::settings::FaceSettings::default();
|
|
match dr_ui::faces::recluster(&catalog, MODEL_ID, &grouping) {
|
|
Ok((suggested, created)) => {
|
|
println!("\nclustering: {suggested} suggestion(s), {created} new group(s)");
|
|
report_people(&catalog);
|
|
}
|
|
Err(e) => {
|
|
eprintln!("clustering failed: {e}");
|
|
std::process::exit(1);
|
|
}
|
|
}
|
|
return;
|
|
}
|
|
|
|
// Tuning, and deliberately read-only: it answers "what would this
|
|
// threshold do to my library" without writing a single suggestion, which
|
|
// is the only way to compare several without each one polluting the next.
|
|
if args.iter().any(|a| a == "--tune") {
|
|
tune_thresholds(&catalog);
|
|
return;
|
|
}
|
|
|
|
// Quality tuning, and read-only like `--tune`. Runs the real detector over
|
|
// real proxies with **both gates disabled**, so the distribution it prints
|
|
// is of everything the detector finds rather than of what survives the
|
|
// current settings — which is the only way to see what a threshold would
|
|
// actually remove.
|
|
if let Some(i) = args.iter().position(|a| a == "--quality") {
|
|
let (Some(detector), Some(embedder)) = (args.get(i + 1), args.get(i + 2)) else {
|
|
eprintln!("--quality needs both a detector and an embedder");
|
|
std::process::exit(2);
|
|
};
|
|
report_quality(
|
|
&catalog,
|
|
&store,
|
|
std::path::Path::new(detector),
|
|
std::path::Path::new(embedder),
|
|
);
|
|
return;
|
|
}
|
|
|
|
let run = args.iter().position(|a| a == "--run");
|
|
let Some(i) = run else {
|
|
if audit.coverage.is_complete() {
|
|
println!("\nnothing outstanding.");
|
|
} else {
|
|
println!("\npass --run DETECTOR.onnx EMBEDDER.onnx to index the outstanding images.");
|
|
}
|
|
return;
|
|
};
|
|
|
|
let (Some(detector), Some(embedder)) = (args.get(i + 1), args.get(i + 2)) else {
|
|
eprintln!("--run needs both a detector and an embedder");
|
|
std::process::exit(2);
|
|
};
|
|
|
|
if audit.ready == 0 {
|
|
println!("\nnothing ready to index.");
|
|
if audit.awaiting_proxy > 0 {
|
|
// Worth saying plainly: running this again will not help, because
|
|
// the blocker is in the thumbnail store rather than here.
|
|
println!(
|
|
"{} image(s) are waiting on a proxy — run the thumbnail sweep first.",
|
|
audit.awaiting_proxy
|
|
);
|
|
}
|
|
return;
|
|
}
|
|
|
|
// Say it here rather than letting the pass return an empty result and
|
|
// print "done: 0 image(s)". That is the shape of report this whole change
|
|
// exists to stop producing.
|
|
if faces::FACE_TIER.edge() < dr_face::MIN_CROP_EDGE {
|
|
println!(
|
|
"\n--run cannot index from the local store any more.\n\
|
|
\n\
|
|
Face crops must be sampled from a buffer longer than {}px and the\n\
|
|
store's largest tier is {}px, so every image here would be refused.\n\
|
|
The measurement behind that floor is docs/faces.md §7b.\n\
|
|
\n\
|
|
Index from the app's Identity screen instead: that pass renders at\n\
|
|
native resolution, which is what the crop needs.",
|
|
dr_face::MIN_CROP_EDGE - 1,
|
|
faces::FACE_TIER.edge(),
|
|
);
|
|
return;
|
|
}
|
|
|
|
println!("\nindexing {} image(s)…", audit.ready);
|
|
let rx = faces::spawn_store_face_sweep(
|
|
catalog_path,
|
|
store_dir,
|
|
PathBuf::from(detector),
|
|
PathBuf::from(embedder),
|
|
MODEL_ID.to_string(),
|
|
dr_face::DetectOptions::default(),
|
|
);
|
|
|
|
let mut seen = 0usize;
|
|
let mut total = 0usize;
|
|
let start = std::time::Instant::now();
|
|
for msg in rx {
|
|
match msg {
|
|
FaceSweepMessage::Total(n) => total = n,
|
|
FaceSweepMessage::Indexed { faces, .. } => {
|
|
seen += 1;
|
|
// One line per image would be thousands of lines; one per
|
|
// twenty-five is enough to see it moving and to estimate.
|
|
if seen.is_multiple_of(25) || faces > 0 {
|
|
let rate = seen as f64 / start.elapsed().as_secs_f64().max(1e-6);
|
|
println!(
|
|
" {seen}/{total} {:.2} img/s ~{:.0} min left",
|
|
rate,
|
|
(total.saturating_sub(seen)) as f64 / rate.max(1e-6) / 60.0
|
|
);
|
|
}
|
|
}
|
|
FaceSweepMessage::Failed { images } => seen += images,
|
|
FaceSweepMessage::Finished {
|
|
images,
|
|
faces,
|
|
failed,
|
|
} => {
|
|
println!(
|
|
"\ndone: {images} image(s), {faces} face(s), {failed} failed, in {:.0}s",
|
|
start.elapsed().as_secs_f64()
|
|
);
|
|
}
|
|
}
|
|
}
|
|
|
|
if let Ok(a) = faces::audit(&catalog, &store, MODEL_ID) {
|
|
println!("{}", a.summary());
|
|
}
|
|
println!("\nrun again with --cluster to group these faces into people.");
|
|
}
|
|
|
|
/// What the clustering proposed, largest group first.
|
|
fn report_people(catalog: &Catalog) {
|
|
let Ok(people) = dr_catalog::faces::people(catalog.connection()) else {
|
|
return;
|
|
};
|
|
if people.is_empty() {
|
|
println!("no groups — too few faces, or none similar enough to group.");
|
|
return;
|
|
}
|
|
println!("\n{} group(s):", people.len());
|
|
for p in people.iter().take(30) {
|
|
let name = if p.name.is_empty() {
|
|
"(unnamed)".to_string()
|
|
} else {
|
|
p.name.clone()
|
|
};
|
|
println!(
|
|
" {name:<24} {} confirmed, {} suggested",
|
|
p.confirmed_faces, p.suggested_faces
|
|
);
|
|
}
|
|
if people.len() > 30 {
|
|
println!(" … and {} more", people.len() - 30);
|
|
}
|
|
}
|
|
|
|
/// What several merge thresholds would each do to this library.
|
|
///
|
|
/// The default 0.9 is a *probability*, and the cosine it lands on depends on
|
|
/// the calibration — so "is 0.9 too tight" is not a question anyone can answer
|
|
/// from the number alone. This runs the real clusterer over the real
|
|
/// embeddings at a range of thresholds and prints what each one produces, which
|
|
/// is the only honest way to choose.
|
|
///
|
|
/// Nothing is written. Run it, read the table, then pass the number you want.
|
|
fn tune_thresholds(catalog: &Catalog) {
|
|
use dr_catalog::faces;
|
|
|
|
let conn = catalog.connection();
|
|
let cal = match faces::calibration(conn, MODEL_ID) {
|
|
Ok(Some((c, _))) => c,
|
|
_ => dr_face::Calibration::default(),
|
|
};
|
|
|
|
let stored = match faces::embeddings(conn, MODEL_ID) {
|
|
Ok(s) => s,
|
|
Err(e) => {
|
|
eprintln!("cannot read embeddings: {e}");
|
|
std::process::exit(1);
|
|
}
|
|
};
|
|
if stored.is_empty() {
|
|
println!("no faces indexed yet — nothing to tune.");
|
|
return;
|
|
}
|
|
|
|
let model = dr_face::ModelId::new(MODEL_ID.to_string());
|
|
let mut candidates = Vec::with_capacity(stored.len());
|
|
for f in stored {
|
|
let Some(emb) = dr_face::Embedding::from_f16_bytes(model.clone(), &f.embedding) else {
|
|
continue;
|
|
};
|
|
candidates.push(dr_face::Candidate {
|
|
face: f.face.0,
|
|
image: f.image.0,
|
|
embedding: emb.v.to_vec(),
|
|
crop_px: f.crop_px,
|
|
quality: f.quality,
|
|
confirmed_person: None,
|
|
});
|
|
}
|
|
|
|
println!(
|
|
"\n{} face(s), calibration valid: {}",
|
|
candidates.len(),
|
|
cal.valid
|
|
);
|
|
println!(
|
|
"\n{:>6} {:>7} {:>7} {:>7} {:>7} {:>8} {:>7}",
|
|
"P", "cosine", "groups", "grouped", "largest", "in groups", "time"
|
|
);
|
|
println!("{}", "-".repeat(60));
|
|
|
|
for p in [0.99_f32, 0.97, 0.95, 0.9, 0.85, 0.8, 0.75, 0.7, 0.6, 0.5] {
|
|
let start = std::time::Instant::now();
|
|
let clusters = dr_face::cluster(&candidates, &cal, p);
|
|
let elapsed = start.elapsed();
|
|
|
|
// A group of one is not a person, and `recluster` discards those, so
|
|
// the interesting figures count only the real groups.
|
|
let real: Vec<_> = clusters.iter().filter(|c| c.members.len() >= 2).collect();
|
|
let grouped: usize = real.iter().map(|c| c.members.len()).sum();
|
|
let largest = real.first().map(|c| c.members.len()).unwrap_or(0);
|
|
|
|
println!(
|
|
"{p:>6.2} {:>7.3} {:>7} {:>7} {:>7} {:>7.0}% {:>6.2}s",
|
|
cal.boundary_at(p, 150.0, 0.0),
|
|
real.len(),
|
|
grouped,
|
|
largest,
|
|
100.0 * grouped as f64 / candidates.len() as f64,
|
|
elapsed.as_secs_f64(),
|
|
);
|
|
}
|
|
|
|
println!(
|
|
"\nA looser threshold makes bigger groups and merges people who are not\n\
|
|
the same; a tighter one splits one person across several. The largest\n\
|
|
group is the tell: when it starts growing much faster than the rest,\n\
|
|
identities are being welded together."
|
|
);
|
|
}
|
|
|
|
/// How many images to sample for the quality report.
|
|
///
|
|
/// Enough for the distribution to settle, few enough to finish while the user
|
|
/// is watching: detection is ~100ms an image, so this is a couple of minutes.
|
|
const QUALITY_SAMPLE: usize = 600;
|
|
|
|
/// What the detector finds, before either quality gate is applied.
|
|
///
|
|
/// The two floors — face size and sharpness — are not independent: a face
|
|
/// smaller than the embedder's 112-pixel input was upsampled to reach it, and
|
|
/// upsampling invents no edges, so small faces score low on sharpness even when
|
|
/// the original was crisp. Choosing either number without seeing the other is
|
|
/// how you end up with one gate doing nothing and the other doing too much.
|
|
///
|
|
/// So this prints them together, over the real library, with nothing filtered.
|
|
fn report_quality(
|
|
catalog: &Catalog,
|
|
store: &ThumbStore,
|
|
detector: &std::path::Path,
|
|
embedder: &std::path::Path,
|
|
) {
|
|
let mut det = match dr_face::Detector::from_path(detector) {
|
|
Ok(d) => d,
|
|
Err(e) => {
|
|
eprintln!("cannot load the detector: {e}");
|
|
std::process::exit(1);
|
|
}
|
|
};
|
|
// Loaded but unused: the point is to fail here, before a two-minute scan,
|
|
// if the pair the user passed is not the pair indexing would use.
|
|
if let Err(e) =
|
|
dr_face::Embedder::from_path(embedder, dr_face::ModelId::new(MODEL_ID.to_string()))
|
|
{
|
|
eprintln!("cannot load the embedder: {e}");
|
|
std::process::exit(1);
|
|
}
|
|
|
|
// Everything the detector can find: no size floor, no sharpness floor.
|
|
let options = dr_face::DetectOptions {
|
|
min_face_px: 0.0,
|
|
min_source_px: 0.0,
|
|
min_sharpness: 0.0,
|
|
..Default::default()
|
|
};
|
|
|
|
let mut stmt = match catalog.connection().prepare(
|
|
"SELECT r.file_id FROM remote r
|
|
JOIN images i ON i.id = r.image_id
|
|
WHERE r.file_id IS NOT NULL AND i.trashed_at IS NULL
|
|
ORDER BY i.id",
|
|
) {
|
|
Ok(s) => s,
|
|
Err(e) => {
|
|
eprintln!("cannot list images: {e}");
|
|
std::process::exit(1);
|
|
}
|
|
};
|
|
let file_ids: Vec<u64> = stmt
|
|
.query_map([], |r| r.get::<_, i64>(0))
|
|
.into_iter()
|
|
.flatten()
|
|
.filter_map(Result::ok)
|
|
.map(|v| v as u64)
|
|
.filter(|id| store.contains(*id, faces::FACE_TIER))
|
|
.take(QUALITY_SAMPLE)
|
|
.collect();
|
|
|
|
if file_ids.is_empty() {
|
|
println!("no proxies on disk to measure — browse the library first.");
|
|
return;
|
|
}
|
|
println!("\nmeasuring {} image(s)…", file_ids.len());
|
|
|
|
// (source_px, sharpness) per detected face.
|
|
let mut found: Vec<(f32, f32)> = Vec::new();
|
|
let mut images = 0usize;
|
|
for id in &file_ids {
|
|
let Ok(Some(thumb)) = store.get(*id, faces::FACE_TIER) else {
|
|
continue;
|
|
};
|
|
let Ok((w, h, rgba)) = dr_thumbs::codec::decode_rgba(&thumb.bytes) else {
|
|
continue;
|
|
};
|
|
let rgb: Vec<f32> = rgba
|
|
.chunks_exact(4)
|
|
.flat_map(|p| {
|
|
[
|
|
p[0] as f32 / 255.0,
|
|
p[1] as f32 / 255.0,
|
|
p[2] as f32 / 255.0,
|
|
]
|
|
})
|
|
.collect();
|
|
|
|
let Ok(dets) = det.detect(&rgb, w as usize, h as usize, &options) else {
|
|
continue;
|
|
};
|
|
images += 1;
|
|
for d in &dets {
|
|
if let Some(a) = dr_face::warp(&rgb, w as usize, h as usize, &d.landmarks) {
|
|
found.push((a.source_px(), a.sharpness()));
|
|
}
|
|
}
|
|
if images.is_multiple_of(50) {
|
|
println!(
|
|
" {images}/{} images, {} face(s)",
|
|
file_ids.len(),
|
|
found.len()
|
|
);
|
|
}
|
|
}
|
|
|
|
if found.is_empty() {
|
|
println!("no faces found in the sample.");
|
|
return;
|
|
}
|
|
|
|
let pct = |v: &mut Vec<f32>, p: f64| -> f32 {
|
|
v.sort_by(|a, b| a.total_cmp(b));
|
|
v[(((v.len() - 1) as f64) * p) as usize]
|
|
};
|
|
let mut sizes: Vec<f32> = found.iter().map(|f| f.0).collect();
|
|
let mut sharps: Vec<f32> = found.iter().map(|f| f.1).collect();
|
|
|
|
println!("\n{} face(s) in {images} image(s)\n", found.len());
|
|
println!(
|
|
"{:>12} {:>8} {:>10}",
|
|
"percentile", "size px", "sharpness"
|
|
);
|
|
println!("{}", "-".repeat(34));
|
|
for p in [0.01, 0.05, 0.10, 0.25, 0.50, 0.75, 0.90, 0.99] {
|
|
println!(
|
|
"{:>11.0}% {:>8.0} {:>10.4}",
|
|
p * 100.0,
|
|
pct(&mut sizes, p),
|
|
pct(&mut sharps, p)
|
|
);
|
|
}
|
|
|
|
// What each candidate pair would remove. Cumulative, because the gates are
|
|
// applied together and their overlap is the whole question.
|
|
println!(
|
|
"\n{:>8} {:>10} {:>9} {:>9} {:>9}",
|
|
"min crop", "min sharp", "size cut", "blur cut", "kept"
|
|
);
|
|
println!("{}", "-".repeat(52));
|
|
for (min_px, min_sharp) in [
|
|
(0.0_f32, 0.0_f32),
|
|
(32.0, 0.0),
|
|
(32.0, 0.002),
|
|
(32.0, 0.005),
|
|
(32.0, 0.010),
|
|
(32.0, 0.020),
|
|
(48.0, 0.005),
|
|
(64.0, 0.010),
|
|
(64.0, 0.020),
|
|
] {
|
|
let by_size = found.iter().filter(|f| f.0 < min_px).count();
|
|
let by_blur = found
|
|
.iter()
|
|
.filter(|f| f.0 >= min_px && f.1 < min_sharp)
|
|
.count();
|
|
let kept = found.len() - by_size - by_blur;
|
|
println!(
|
|
"{min_px:>8.0} {min_sharp:>10.3} {:>8.0}% {:>8.0}% {:>8.0}%",
|
|
100.0 * by_size as f64 / found.len() as f64,
|
|
100.0 * by_blur as f64 / found.len() as f64,
|
|
100.0 * kept as f64 / found.len() as f64,
|
|
);
|
|
}
|
|
|
|
println!(
|
|
"\n`size cut` is what the size floor removes; `blur cut` is what the\n\
|
|
sharpness floor removes *of what the size floor left*, so the two\n\
|
|
columns do not double-count. A sharpness floor that cuts almost\n\
|
|
nothing once the size floor is in place is a floor that is not\n\
|
|
earning its place."
|
|
);
|
|
}
|