Read each face's eyes, and whether sunglasses hide them

Two MIT classifiers from the same author as the reference pipeline's
whole-body detector: OCEC answers P(open) for one 40×24 eye, SGC
P(sunglasses) for a 48×48 head. Both load in tract once their batch
dimension is pinned by tools/fix-face-model-shapes.sh, like the embedder.

The crops come through the same fitted similarity the aligned face does,
so an eye window is a constant in template units rather than a second
warp, and a tilted head yields an upright eye. Measured on 60 proxies
from the reference library: the eye window plateaus at 22×11, the S
variant beats M and L (which overfit their own domain), and for
sunglasses the aligned face beats a head framing but the higher of the
two catches 11 of 12 pairs against 9 for either alone.

The reading keeps both eyes and the sunglasses number apart, because a
wink averages to the least informative value and a lens of dark glass
draws a confident answer from the eye classifier — over a woman in
sunglasses it read the right eye 0.97 open. Sunglasses take precedence,
and a face behind them is neither open nor a blink.
This commit is contained in:
2026-09-19 14:03:31 +02:00
parent 2481904016
commit 6b51726322
7 changed files with 931 additions and 22 deletions
+170
View File
@@ -0,0 +1,170 @@
//! Detect the faces in a JPEG and read each one's eyes (docs/faces.md §17).
//!
//! The thing worth looking at is whether the eye windows land on eyes — so
//! with `--dump DIR` the crops the classifiers were shown are written out as
//! PPMs, one per eye and one per head, named by image and face.
//!
//! cargo run -p dr-face --features inference --example eyes -- \
//! DET.onnx OCEC.onnx SGC.onnx [--dump DIR] [--eye W,H] [--head X,Y,W,H] \
//! photo.jpg [photo.jpg ...]
//!
//! `--eye` and `--head` try other crop windows, in template units; they are
//! how `EYE_WINDOW` and `SUNGLASSES_WINDOWS` were chosen.
//!
//! All three models must have had their dynamic dims pinned first; see
//! `tools/fix-face-model-shapes.sh`.
use std::path::{Path, PathBuf};
use std::time::Instant;
use dr_face::{align, DetectOptions, Detector, EyeModels, Pixels};
fn main() {
env_logger::init();
let mut args: Vec<String> = std::env::args().skip(1).collect();
let dump = args.iter().position(|a| a == "--dump").map(|i| {
args.remove(i);
PathBuf::from(args.remove(i))
});
// `--head X,Y,W,H` tries a single head window, in template units, in
// place of the shipped pair.
let head_windows: Vec<(f32, f32, f32, f32)> = args
.iter()
.position(|a| a == "--head")
.map(|i| {
args.remove(i);
let spec = args.remove(i);
let v: Vec<f32> = spec
.split(',')
.map(|s| s.parse().expect("--head number"))
.collect();
assert_eq!(v.len(), 4, "--head wants X,Y,W,H");
vec![(v[0], v[1], v[2], v[3])]
})
.unwrap_or_else(|| align::SUNGLASSES_WINDOWS.to_vec());
// `--eye W,H` tries another eye window, in template units.
let eye_window = args
.iter()
.position(|a| a == "--eye")
.map(|i| {
args.remove(i);
let spec = args.remove(i);
let v: Vec<f32> = spec
.split(',')
.map(|s| s.parse().expect("--eye number"))
.collect();
assert_eq!(v.len(), 2, "--eye wants W,H");
(v[0], v[1])
})
.unwrap_or(align::EYE_WINDOW);
if args.len() < 4 {
eprintln!(
"usage: eyes DET.onnx OCEC.onnx SGC.onnx [--dump DIR] [--eye W,H] [--head X,Y,W,H] IMAGE.jpg [IMAGE.jpg ...]"
);
std::process::exit(2);
}
if let Some(d) = &dump {
std::fs::create_dir_all(d).expect("dump dir");
}
let t = Instant::now();
let mut detector = Detector::from_path(&args[0]).expect("load detector");
let mut models = EyeModels::from_paths(&args[1], &args[2]).expect("load eye models");
println!("loaded the models in {:?}", t.elapsed());
let opts = DetectOptions::default();
for path in &args[3..] {
let (rgb, w, h) = match load_jpeg(path) {
Ok(v) => v,
Err(e) => {
println!("{path}: {e}");
continue;
}
};
let dets = detector.detect(&rgb, w, h, &opts).expect("detect");
println!("\n{path} ({w}×{h}) {} face(s)", dets.len());
let stem = Path::new(path)
.file_stem()
.map(|s| s.to_string_lossy().into_owned())
.unwrap_or_default();
for (i, d) in dets.iter().enumerate() {
let px = Pixels::RgbF32(&rgb);
let (Some(eyes), Some(head)) = (
align::eye_patches_in(px, w, h, &d.landmarks, eye_window),
align::head_views_in(px, w, h, &d.landmarks, &head_windows),
) else {
println!(" [{i}] degenerate landmarks, skipped");
continue;
};
let t = Instant::now();
let reading = models.read(&eyes, &head).expect("classify");
let ms = t.elapsed().as_secs_f64() * 1e3;
println!(
" [{i}] conf {:.2} box {:.0}×{:.0} right {:.3} left {:.3} sunglasses {:.3} → {:?} ({ms:.1} ms)",
d.confidence,
d.width(),
d.height(),
reading.right_open,
reading.left_open,
reading.sunglasses,
reading.state(),
);
if let Some(dir) = &dump {
write_ppm(
&dir.join(format!("{stem}-{i}-right.ppm")),
eyes.right.pixels(),
align::EYE_PATCH_WIDTH,
align::EYE_PATCH_HEIGHT,
);
write_ppm(
&dir.join(format!("{stem}-{i}-left.ppm")),
eyes.left.pixels(),
align::EYE_PATCH_WIDTH,
align::EYE_PATCH_HEIGHT,
);
for (n, view) in head.views().enumerate() {
write_ppm(
&dir.join(format!("{stem}-{i}-head{n}.ppm")),
view,
align::SUNGLASSES_EDGE,
align::SUNGLASSES_EDGE,
);
}
}
}
}
}
fn write_ppm(path: &Path, rgb: &[f32], w: usize, h: usize) {
let mut out = format!("P6\n{w} {h}\n255\n").into_bytes();
out.extend(
rgb.iter()
.map(|v| (v.clamp(0.0, 1.0) * 255.0).round() as u8),
);
std::fs::write(path, out).expect("write ppm");
}
/// Decode to the tightly packed `f32` RGB `0.0..=1.0` the crate expects.
fn load_jpeg(path: &str) -> Result<(Vec<f32>, usize, usize), String> {
let bytes = std::fs::read(path).map_err(|e| e.to_string())?;
let mut dec = zune_jpeg::JpegDecoder::new(&bytes);
let px = dec.decode().map_err(|e| e.to_string())?;
let info = dec.info().ok_or("no jpeg header")?;
let (w, h) = (info.width as usize, info.height as usize);
let rgb: Vec<f32> = match px.len() / (w * h) {
3 => px.iter().map(|&v| v as f32 / 255.0).collect(),
1 => px
.iter()
.flat_map(|&v| {
let g = v as f32 / 255.0;
[g, g, g]
})
.collect(),
n => return Err(format!("{n} components per pixel, expected 1 or 3")),
};
Ok((rgb, w, h))
}