Measure what the cheapest SCRFD actually costs in faces
§1 chose scrfd_500m on FLOPs and never measured the recall it gave up. A dr-ui example now runs several detectors over the same sample of stored proxies, matches boxes by IoU against the first, buckets the result by face size, times each, and writes contact sheets of the disagreements in both directions — because a count of extra faces says nothing until someone has looked at whether they are faces. Over 400 proxies from the reference library: 2.5G finds 14% more faces for 12% more time, 10G a further 12% for 3.1× the time. The extras are small real faces. The 86 faces only 500M found are a dog a dozen times, a stop sign, a wheel and the backs of heads. Recorded in faces.md §12.3.
This commit is contained in:
@@ -0,0 +1,427 @@
|
||||
//! TRACES: FR-CULL-8
|
||||
//! Compare face detectors over the same proxies.
|
||||
//!
|
||||
//! cargo run --release -p dr-ui --example face_detectors -- \
|
||||
//! CATALOG.db THUMBS_DIR BASELINE.onnx CANDIDATE.onnx [CANDIDATE.onnx…] \
|
||||
//! [--sample N] [--sheet DIR]
|
||||
//!
|
||||
//! Runs every detector over the same sample of stored proxies and reports, per
|
||||
//! detector, how long it took and how many faces it found by size; then, per
|
||||
//! candidate, which of those faces the baseline also found and which it did
|
||||
//! not. `--sheet` writes a contact sheet of the disagreements in each direction,
|
||||
//! because a count of "extra faces" says nothing until someone has looked at
|
||||
//! whether they are faces.
|
||||
//!
|
||||
//! # What it measures, and what it cannot
|
||||
//!
|
||||
//! docs/faces.md §12 M4 is recall against hand-labelled faces. There are no
|
||||
//! labels here, so this is the cheaper question that decides whether M4 is
|
||||
//! worth the labelling: *do the detectors disagree, where, and does the
|
||||
//! disagreement look like faces*. A candidate whose extras are all real faces
|
||||
//! under 20 px has found the group shots the baseline lost; one whose extras
|
||||
//! are ears and door handles has found nothing.
|
||||
//!
|
||||
//! Both size gates are off, so what is counted is what the detector *emits*
|
||||
//! above its confidence, not what indexing would keep. The confidence floor
|
||||
//! is the production one, because a detector that only wins below it has not
|
||||
//! won anything the pipeline would see.
|
||||
//!
|
||||
//! The models must have had their input dims frozen first; see
|
||||
//! `tools/fix-face-model-shapes.sh`.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::time::Instant;
|
||||
|
||||
use dr_catalog::Catalog;
|
||||
use dr_face::{DetectOptions, Detection, Detector};
|
||||
use dr_thumbs::ThumbStore;
|
||||
use dr_ui::faces;
|
||||
|
||||
const DEFAULT_SAMPLE: usize = 400;
|
||||
|
||||
/// Two boxes are the same face above this overlap.
|
||||
const MATCH_IOU: f32 = 0.5;
|
||||
|
||||
/// Size buckets, on the box's shorter edge in **proxy pixels**. The proxy is
|
||||
/// 1024 on its long edge and the detector sees it letterboxed to 640, so the
|
||||
/// model's own view is 0.625 of these — the smallest bucket is a face under
|
||||
/// 10 px to the network.
|
||||
const BUCKETS: [(f32, &str); 4] = [
|
||||
(16.0, "<16"),
|
||||
(32.0, "16-32"),
|
||||
(64.0, "32-64"),
|
||||
(f32::INFINITY, ">=64"),
|
||||
];
|
||||
|
||||
/// Tiles on a contact sheet: this many per row, this wide.
|
||||
const SHEET_COLS: usize = 10;
|
||||
const SHEET_TILE: usize = 112;
|
||||
const SHEET_MAX: usize = 100;
|
||||
|
||||
fn main() {
|
||||
env_logger::init();
|
||||
|
||||
let args: Vec<String> = std::env::args().skip(1).collect();
|
||||
let mut sample = DEFAULT_SAMPLE;
|
||||
let mut sheet: Option<PathBuf> = None;
|
||||
let mut positional: Vec<String> = Vec::new();
|
||||
let mut i = 0;
|
||||
while i < args.len() {
|
||||
match args[i].as_str() {
|
||||
"--sample" => {
|
||||
sample = args
|
||||
.get(i + 1)
|
||||
.and_then(|s| s.parse().ok())
|
||||
.unwrap_or_else(|| usage());
|
||||
i += 2;
|
||||
}
|
||||
"--sheet" => {
|
||||
sheet = Some(PathBuf::from(args.get(i + 1).unwrap_or_else(|| usage())));
|
||||
i += 2;
|
||||
}
|
||||
other => {
|
||||
positional.push(other.to_string());
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
if positional.len() < 4 {
|
||||
usage();
|
||||
}
|
||||
|
||||
let catalog = Catalog::open(Path::new(&positional[0])).unwrap_or_else(|e| {
|
||||
eprintln!("cannot open catalog {}: {e}", positional[0]);
|
||||
std::process::exit(1);
|
||||
});
|
||||
let store = ThumbStore::open(Path::new(&positional[1])).unwrap_or_else(|e| {
|
||||
eprintln!("cannot open thumbnail store {}: {e}", positional[1]);
|
||||
std::process::exit(1);
|
||||
});
|
||||
|
||||
let mut detectors: Vec<(String, Detector)> = positional[2..]
|
||||
.iter()
|
||||
.map(|p| {
|
||||
let label = Path::new(p)
|
||||
.file_stem()
|
||||
.map(|s| s.to_string_lossy().into_owned())
|
||||
.unwrap_or_else(|| p.clone());
|
||||
let t = Instant::now();
|
||||
let det = Detector::from_path(p).unwrap_or_else(|e| {
|
||||
eprintln!("cannot load {p}: {e}");
|
||||
std::process::exit(1);
|
||||
});
|
||||
println!(
|
||||
"{label:<20} loaded in {:>5.0} ms strides {:?}",
|
||||
t.elapsed().as_secs_f64() * 1e3,
|
||||
det.strides()
|
||||
);
|
||||
(label, det)
|
||||
})
|
||||
.collect();
|
||||
|
||||
let file_ids = sample_ids(&catalog, &store, sample);
|
||||
if file_ids.is_empty() {
|
||||
println!("no proxies on disk to measure — browse the library first.");
|
||||
return;
|
||||
}
|
||||
println!(
|
||||
"\n{} image(s), evenly spaced through the library\n",
|
||||
file_ids.len()
|
||||
);
|
||||
|
||||
// Gates off, confidence as shipped: see the module note.
|
||||
let options = DetectOptions {
|
||||
min_face_px: 0.0,
|
||||
min_source_px: 0.0,
|
||||
min_sharpness: 0.0,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
// Per detector: per-image timings, and per-image detections.
|
||||
let n = detectors.len();
|
||||
let mut times: Vec<Vec<f64>> = vec![Vec::new(); n];
|
||||
let mut found: Vec<Vec<Vec<Detection>>> = vec![Vec::new(); n];
|
||||
// The decoded proxies the sheet will cut from, kept only when asked for.
|
||||
let mut images: Vec<(u32, u32, Vec<u8>)> = Vec::new();
|
||||
let mut done = 0usize;
|
||||
|
||||
for id in &file_ids {
|
||||
let Ok(Some(thumb)) = store.get(*id, faces::FACE_TIER) else {
|
||||
continue;
|
||||
};
|
||||
let Ok((w, h, rgba)) = dr_thumbs::codec::decode_rgba(&thumb.bytes) else {
|
||||
continue;
|
||||
};
|
||||
let rgb: Vec<f32> = rgba
|
||||
.chunks_exact(4)
|
||||
.flat_map(|p| [p[0], p[1], p[2]].map(|c| c as f32 / 255.0))
|
||||
.collect();
|
||||
|
||||
for (k, (_, det)) in detectors.iter_mut().enumerate() {
|
||||
let t = Instant::now();
|
||||
let dets = det
|
||||
.detect(&rgb, w as usize, h as usize, &options)
|
||||
.unwrap_or_default();
|
||||
times[k].push(t.elapsed().as_secs_f64() * 1e3);
|
||||
found[k].push(dets);
|
||||
}
|
||||
if sheet.is_some() {
|
||||
images.push((w, h, rgba));
|
||||
}
|
||||
done += 1;
|
||||
if done.is_multiple_of(50) {
|
||||
println!(" {done}/{} images", file_ids.len());
|
||||
}
|
||||
}
|
||||
|
||||
println!("\n{done} image(s) measured\n");
|
||||
|
||||
// ---- per detector: speed and what it emits -------------------------
|
||||
|
||||
println!(
|
||||
"{:<20} {:>8} {:>8} {:>7} {}",
|
||||
"detector",
|
||||
"mean ms",
|
||||
"p50 ms",
|
||||
"faces",
|
||||
BUCKETS
|
||||
.iter()
|
||||
.map(|(_, l)| format!("{l:>7}"))
|
||||
.collect::<String>()
|
||||
);
|
||||
for k in 0..n {
|
||||
let mut t = times[k].clone();
|
||||
t.sort_by(|a, b| a.total_cmp(b));
|
||||
let mean = t.iter().sum::<f64>() / t.len().max(1) as f64;
|
||||
let p50 = t.get(t.len() / 2).copied().unwrap_or(0.0);
|
||||
let all: Vec<&Detection> = found[k].iter().flatten().collect();
|
||||
let counts = bucket_counts(all.iter().copied());
|
||||
println!(
|
||||
"{:<20} {mean:>8.1} {p50:>8.1} {:>7} {}",
|
||||
detectors[k].0,
|
||||
all.len(),
|
||||
counts.iter().map(|c| format!("{c:>7}")).collect::<String>()
|
||||
);
|
||||
}
|
||||
|
||||
// ---- per candidate: agreement with the baseline ----------------------
|
||||
|
||||
let (base_label, _) = &detectors[0];
|
||||
for k in 1..n {
|
||||
let label = &detectors[k].0;
|
||||
// (image index, detection) for each side of the disagreement.
|
||||
let mut matched: Vec<&Detection> = Vec::new();
|
||||
let mut only_candidate: Vec<(usize, &Detection)> = Vec::new();
|
||||
let mut only_baseline: Vec<(usize, &Detection)> = Vec::new();
|
||||
|
||||
for (img, (base, cand)) in found[0].iter().zip(&found[k]).enumerate() {
|
||||
let mut base_used = vec![false; base.len()];
|
||||
for c in cand {
|
||||
let best = base
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(bi, _)| !base_used[*bi])
|
||||
.map(|(bi, b)| (bi, iou(b, c)))
|
||||
.filter(|(_, v)| *v >= MATCH_IOU)
|
||||
.max_by(|a, b| a.1.total_cmp(&b.1));
|
||||
match best {
|
||||
Some((bi, _)) => {
|
||||
base_used[bi] = true;
|
||||
matched.push(c);
|
||||
}
|
||||
None => only_candidate.push((img, c)),
|
||||
}
|
||||
}
|
||||
for (bi, b) in base.iter().enumerate() {
|
||||
if !base_used[bi] {
|
||||
only_baseline.push((img, b));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
println!("\n{label} against {base_label}:");
|
||||
println!(
|
||||
"{:<28} {:>7} {}",
|
||||
"",
|
||||
"faces",
|
||||
BUCKETS
|
||||
.iter()
|
||||
.map(|(_, l)| format!("{l:>7}"))
|
||||
.collect::<String>()
|
||||
);
|
||||
for (name, set) in [
|
||||
("both found", matched.clone()),
|
||||
(
|
||||
"candidate only",
|
||||
only_candidate.iter().map(|(_, d)| *d).collect(),
|
||||
),
|
||||
(
|
||||
"baseline only",
|
||||
only_baseline.iter().map(|(_, d)| *d).collect(),
|
||||
),
|
||||
] {
|
||||
let counts = bucket_counts(set.iter().copied());
|
||||
println!(
|
||||
" {name:<26} {:>7} {} median conf {:.2}",
|
||||
set.len(),
|
||||
counts.iter().map(|c| format!("{c:>7}")).collect::<String>(),
|
||||
median_confidence(&set)
|
||||
);
|
||||
}
|
||||
|
||||
if let Some(dir) = &sheet {
|
||||
std::fs::create_dir_all(dir).expect("create sheet dir");
|
||||
for (suffix, set) in [("extra", &only_candidate), ("missed", &only_baseline)] {
|
||||
let path = dir.join(format!("{label}-{suffix}.jpg"));
|
||||
match write_sheet(&path, &images, set) {
|
||||
Ok(n) => println!(" {suffix:<26} {n} tile(s) -> {}", path.display()),
|
||||
Err(e) => eprintln!(" {suffix}: {e}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn usage() -> ! {
|
||||
eprintln!(
|
||||
"usage: face_detectors CATALOG.db THUMBS_DIR BASELINE.onnx CANDIDATE.onnx [CANDIDATE.onnx…] \
|
||||
[--sample N] [--sheet DIR]"
|
||||
);
|
||||
std::process::exit(2);
|
||||
}
|
||||
|
||||
/// Up to `n` file ids with a proxy on disk, evenly spaced through the library
|
||||
/// rather than its first `n` — the first `n` are one trip.
|
||||
fn sample_ids(catalog: &Catalog, store: &ThumbStore, n: usize) -> Vec<u64> {
|
||||
let mut stmt = catalog
|
||||
.connection()
|
||||
.prepare(
|
||||
"SELECT r.file_id FROM remote r
|
||||
JOIN images i ON i.id = r.image_id
|
||||
WHERE r.file_id IS NOT NULL AND i.trashed_at IS NULL
|
||||
ORDER BY i.id",
|
||||
)
|
||||
.expect("list images");
|
||||
let with_proxy: Vec<u64> = stmt
|
||||
.query_map([], |r| r.get::<_, i64>(0))
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(Result::ok)
|
||||
.map(|v| v as u64)
|
||||
.filter(|id| store.contains(*id, faces::FACE_TIER))
|
||||
.collect();
|
||||
if with_proxy.len() <= n {
|
||||
return with_proxy;
|
||||
}
|
||||
let step = with_proxy.len() as f64 / n as f64;
|
||||
(0..n)
|
||||
.map(|i| with_proxy[(i as f64 * step) as usize])
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn iou(a: &Detection, b: &Detection) -> f32 {
|
||||
let x0 = a.bbox.0.max(b.bbox.0);
|
||||
let y0 = a.bbox.1.max(b.bbox.1);
|
||||
let x1 = a.bbox.2.min(b.bbox.2);
|
||||
let y1 = a.bbox.3.min(b.bbox.3);
|
||||
let inter = (x1 - x0).max(0.0) * (y1 - y0).max(0.0);
|
||||
let union = a.width() * a.height() + b.width() * b.height() - inter;
|
||||
if union <= 0.0 {
|
||||
0.0
|
||||
} else {
|
||||
inter / union
|
||||
}
|
||||
}
|
||||
|
||||
fn bucket_counts<'a>(dets: impl Iterator<Item = &'a Detection>) -> [usize; BUCKETS.len()] {
|
||||
let mut counts = [0usize; BUCKETS.len()];
|
||||
for d in dets {
|
||||
let edge = d.width().min(d.height());
|
||||
let b = BUCKETS
|
||||
.iter()
|
||||
.position(|(limit, _)| edge < *limit)
|
||||
.unwrap_or(BUCKETS.len() - 1);
|
||||
counts[b] += 1;
|
||||
}
|
||||
counts
|
||||
}
|
||||
|
||||
fn median_confidence(dets: &[&Detection]) -> f32 {
|
||||
if dets.is_empty() {
|
||||
return 0.0;
|
||||
}
|
||||
let mut c: Vec<f32> = dets.iter().map(|d| d.confidence).collect();
|
||||
c.sort_by(|a, b| a.total_cmp(b));
|
||||
c[c.len() / 2]
|
||||
}
|
||||
|
||||
/// A grid of face tiles, each the box enlarged by half again so there is
|
||||
/// context to judge by, resampled to a fixed tile whatever its source size.
|
||||
/// Smallest faces first: those are the ones the question is about.
|
||||
fn write_sheet(
|
||||
path: &Path,
|
||||
images: &[(u32, u32, Vec<u8>)],
|
||||
set: &[(usize, &Detection)],
|
||||
) -> Result<usize, String> {
|
||||
if set.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
let mut ordered: Vec<&(usize, &Detection)> = set.iter().collect();
|
||||
ordered.sort_by(|a, b| {
|
||||
let ea = a.1.width().min(a.1.height());
|
||||
let eb = b.1.width().min(b.1.height());
|
||||
ea.total_cmp(&eb)
|
||||
});
|
||||
ordered.truncate(SHEET_MAX);
|
||||
|
||||
let rows = ordered.len().div_ceil(SHEET_COLS);
|
||||
let (sw, sh) = (SHEET_COLS * SHEET_TILE, rows * SHEET_TILE);
|
||||
let mut sheet = vec![0u8; sw * sh * 4];
|
||||
|
||||
for (i, (img, d)) in ordered.iter().enumerate() {
|
||||
let (w, h, rgba) = &images[*img];
|
||||
let (w, h) = (*w as usize, *h as usize);
|
||||
let cx = (d.bbox.0 + d.bbox.2) * 0.5;
|
||||
let cy = (d.bbox.1 + d.bbox.3) * 0.5;
|
||||
let half = d.width().max(d.height()) * 0.75;
|
||||
let (tx0, ty0) = ((i % SHEET_COLS) * SHEET_TILE, (i / SHEET_COLS) * SHEET_TILE);
|
||||
for ty in 0..SHEET_TILE {
|
||||
for tx in 0..SHEET_TILE {
|
||||
let sx = cx - half + (tx as f32 + 0.5) / SHEET_TILE as f32 * half * 2.0;
|
||||
let sy = cy - half + (ty as f32 + 0.5) / SHEET_TILE as f32 * half * 2.0;
|
||||
let px = bilinear(rgba, w, h, sx, sy);
|
||||
let o = ((ty0 + ty) * sw + tx0 + tx) * 4;
|
||||
sheet[o..o + 4].copy_from_slice(&px);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let bytes = dr_thumbs::codec::encode_rgba(sw as u32, sh as u32, &sheet)
|
||||
.map_err(|e| format!("encode: {e}"))?;
|
||||
std::fs::write(path, bytes).map_err(|e| format!("write {}: {e}", path.display()))?;
|
||||
Ok(ordered.len())
|
||||
}
|
||||
|
||||
/// RGBA sample at a continuous position; black outside the image.
|
||||
fn bilinear(rgba: &[u8], w: usize, h: usize, x: f32, y: f32) -> [u8; 4] {
|
||||
if x < 0.0 || y < 0.0 || x >= (w - 1) as f32 || y >= (h - 1) as f32 {
|
||||
return [0, 0, 0, 255];
|
||||
}
|
||||
let (x0, y0) = (x.floor() as usize, y.floor() as usize);
|
||||
let (fx, fy) = (x - x0 as f32, y - y0 as f32);
|
||||
let at = |xx: usize, yy: usize| &rgba[(yy * w + xx) * 4..(yy * w + xx) * 4 + 4];
|
||||
let (p00, p10, p01, p11) = (
|
||||
at(x0, y0),
|
||||
at(x0 + 1, y0),
|
||||
at(x0, y0 + 1),
|
||||
at(x0 + 1, y0 + 1),
|
||||
);
|
||||
let mut out = [0u8; 4];
|
||||
for c in 0..3 {
|
||||
let top = p00[c] as f32 * (1.0 - fx) + p10[c] as f32 * fx;
|
||||
let bot = p01[c] as f32 * (1.0 - fx) + p11[c] as f32 * fx;
|
||||
out[c] = (top * (1.0 - fy) + bot * fy).round() as u8;
|
||||
}
|
||||
out[3] = 255;
|
||||
out
|
||||
}
|
||||
Reference in New Issue
Block a user