Cut the eye box from a landmark contour, and refuse eyes that cannot be read

SCRFD's eye point places a face, not an eye: on turned and smiling heads
the classifier's window had the eye in a corner, and two model-free ways
of re-centring it — the darkest blob, the most contrasty window — both
lost open eyes (19 → 15 and 19 → 9 of 25). Three landmark models were
then run over the same faces; Face Mesh V2 and InsightFace's 2d106det
tied at 22 of 25 and 2d106det ships, being the cheapest by far and under
the grant the detector and embedder already carry. The eye box is the
tight bounding box of its ten lid points, cut upright from the native
render, which is what the classifier was trained on.

The larger change is that the reading now carries, per eye, the source
pixels across the box and the sharpness of the patch — because the
commonest wrong answer on the reference library was a soft eye read as
closed, and a classifier shown a smear will always say something. An eye
under either floor, or narrower than six tenths of its partner (the far
eye of a turned head, whose contour collapses), is not asked; a face with
no readable eye is a fourth state, Unreadable, that no filter drops. On
twenty native renders the one real blink is caught, the laughing faces
are closed, the profiles are judged on the near eye, and the one thing
left beyond any floor is a face with a pot held over it.
This commit is contained in:
2026-09-19 14:04:08 +02:00
parent b908d861e0
commit f5956707e7
7 changed files with 690 additions and 311 deletions
+56 -76
View File
@@ -1,17 +1,15 @@
//! Detect the faces in a JPEG and read each one's eyes (docs/faces.md §17).
//!
//! The thing worth looking at is whether the eye windows land on eyes — so
//! with `--dump DIR` the crops the classifiers were shown are written out as
//! PPMs, one per eye and one per head, named by image and face.
//! The thing worth looking at is whether the eye boxes land on eyes and
//! whether soft ones are refused — so with `--dump DIR` the crops the
//! classifiers were shown are written out as PPMs, one per eye and one per
//! head framing, named by image and face, and every line carries the
//! numbers the readability floors are set from.
//!
//! cargo run -p dr-face --features inference --example eyes -- \
//! DET.onnx OCEC.onnx SGC.onnx [--dump DIR] [--eye W,H] [--head X,Y,W,H] \
//! photo.jpg [photo.jpg ...]
//! DET.onnx 2D106DET.onnx OCEC.onnx SGC.onnx [--dump DIR] photo.jpg [photo.jpg ...]
//!
//! `--eye` and `--head` try other crop windows, in template units; they are
//! how `EYE_WINDOW` and `SUNGLASSES_WINDOWS` were chosen.
//!
//! All three models must have had their dynamic dims pinned first; see
//! All four models must have had their dynamic dims pinned first; see
//! `tools/fix-face-model-shapes.sh`.
use std::path::{Path, PathBuf};
@@ -27,40 +25,9 @@ fn main() {
args.remove(i);
PathBuf::from(args.remove(i))
});
// `--head X,Y,W,H` tries a single head window, in template units, in
// place of the shipped pair.
let head_windows: Vec<(f32, f32, f32, f32)> = args
.iter()
.position(|a| a == "--head")
.map(|i| {
args.remove(i);
let spec = args.remove(i);
let v: Vec<f32> = spec
.split(',')
.map(|s| s.parse().expect("--head number"))
.collect();
assert_eq!(v.len(), 4, "--head wants X,Y,W,H");
vec![(v[0], v[1], v[2], v[3])]
})
.unwrap_or_else(|| align::SUNGLASSES_WINDOWS.to_vec());
// `--eye W,H` tries another eye window, in template units.
let eye_window = args
.iter()
.position(|a| a == "--eye")
.map(|i| {
args.remove(i);
let spec = args.remove(i);
let v: Vec<f32> = spec
.split(',')
.map(|s| s.parse().expect("--eye number"))
.collect();
assert_eq!(v.len(), 2, "--eye wants W,H");
(v[0], v[1])
})
.unwrap_or(align::EYE_WINDOW);
if args.len() < 4 {
if args.len() < 5 {
eprintln!(
"usage: eyes DET.onnx OCEC.onnx SGC.onnx [--dump DIR] [--eye W,H] [--head X,Y,W,H] IMAGE.jpg [IMAGE.jpg ...]"
"usage: eyes DET.onnx 2D106DET.onnx OCEC.onnx SGC.onnx [--dump DIR] IMAGE.jpg [IMAGE.jpg ...]"
);
std::process::exit(2);
}
@@ -70,11 +37,11 @@ fn main() {
let t = Instant::now();
let mut detector = Detector::from_path(&args[0]).expect("load detector");
let mut models = EyeModels::from_paths(&args[1], &args[2]).expect("load eye models");
let mut models = EyeModels::from_paths(&args[1], &args[2], &args[3]).expect("load eye models");
println!("loaded the models in {:?}", t.elapsed());
let opts = DetectOptions::default();
for path in &args[3..] {
for path in &args[4..] {
let (rgb, w, h) = match load_jpeg(path) {
Ok(v) => v,
Err(e) => {
@@ -92,46 +59,59 @@ fn main() {
for (i, d) in dets.iter().enumerate() {
let px = Pixels::RgbF32(&rgb);
let (Some(eyes), Some(head)) = (
align::eye_patches_in(px, w, h, &d.landmarks, eye_window),
align::head_views_in(px, w, h, &d.landmarks, &head_windows),
) else {
println!(" [{i}] degenerate landmarks, skipped");
let t = Instant::now();
let reading = models
.read(px, w, h, d.bbox, &d.landmarks)
.expect("read eyes");
let ms = t.elapsed().as_secs_f64() * 1e3;
let Some(r) = reading else {
println!(" [{i}] nothing to cut, skipped");
continue;
};
let t = Instant::now();
let reading = models.read(&eyes, &head).expect("classify");
let ms = t.elapsed().as_secs_f64() * 1e3;
println!(
" [{i}] conf {:.2} box {:.0}×{:.0} right {:.3} left {:.3} sunglasses {:.3} → {:?} ({ms:.1} ms)",
" [{i}] conf {:.2} box {:.0}×{:.0} right {:.3} ({:.0}px, sharp {:.3}) left {:.3} ({:.0}px, sharp {:.3}) sunglasses {:.3} → {:?} ({ms:.1} ms)",
d.confidence,
d.width(),
d.height(),
reading.right_open,
reading.left_open,
reading.sunglasses,
reading.state(),
r.right.open,
r.right.px,
r.right.sharpness,
r.left.open,
r.left.px,
r.left.sharpness,
r.sunglasses,
r.state(),
);
if let Some(dir) = &dump {
write_ppm(
&dir.join(format!("{stem}-{i}-right.ppm")),
eyes.right.pixels(),
align::EYE_PATCH_WIDTH,
align::EYE_PATCH_HEIGHT,
);
write_ppm(
&dir.join(format!("{stem}-{i}-left.ppm")),
eyes.left.pixels(),
align::EYE_PATCH_WIDTH,
align::EYE_PATCH_HEIGHT,
);
for (n, view) in head.views().enumerate() {
write_ppm(
&dir.join(format!("{stem}-{i}-head{n}.ppm")),
view,
align::SUNGLASSES_EDGE,
align::SUNGLASSES_EDGE,
);
// The same crops `EyeModels::read` cut, cut again for the
// sheet: the reading itself carries numbers, not pixels.
if let Some(lm) = models
.landmarks
.landmarks(px, w, h, d.bbox)
.expect("landmarks")
{
for (name, contour) in [("right", lm.right_eye()), ("left", lm.left_eye())] {
if let Some(patch) =
align::eye_box(&contour).and_then(|b| align::eye_patch(px, w, h, b))
{
write_ppm(
&dir.join(format!("{stem}-{i}-{name}.ppm")),
patch.pixels(),
align::EYE_PATCH_WIDTH,
align::EYE_PATCH_HEIGHT,
);
}
}
}
if let Some(head) = align::head_views(px, w, h, &d.landmarks) {
for (n, view) in head.views().enumerate() {
write_ppm(
&dir.join(format!("{stem}-{i}-head{n}.ppm")),
view,
align::SUNGLASSES_EDGE,
align::SUNGLASSES_EDGE,
);
}
}
}
}