SCRFD's eye point places a face, not an eye: on turned and smiling heads the classifier's window had the eye in a corner, and two model-free ways of re-centring it — the darkest blob, the most contrasty window — both lost open eyes (19 → 15 and 19 → 9 of 25). Three landmark models were then run over the same faces; Face Mesh V2 and InsightFace's 2d106det tied at 22 of 25 and 2d106det ships, being the cheapest by far and under the grant the detector and embedder already carry. The eye box is the tight bounding box of its ten lid points, cut upright from the native render, which is what the classifier was trained on. The larger change is that the reading now carries, per eye, the source pixels across the box and the sharpness of the patch — because the commonest wrong answer on the reference library was a soft eye read as closed, and a classifier shown a smear will always say something. An eye under either floor, or narrower than six tenths of its partner (the far eye of a turned head, whose contour collapses), is not asked; a face with no readable eye is a fourth state, Unreadable, that no filter drops. On twenty native renders the one real blink is caught, the laughing faces are closed, the profiles are judged on the near eye, and the one thing left beyond any floor is a face with a pot held over it.
151 lines
5.6 KiB
Rust
151 lines
5.6 KiB
Rust
//! Detect the faces in a JPEG and read each one's eyes (docs/faces.md §17).
|
||
//!
|
||
//! The thing worth looking at is whether the eye boxes land on eyes and
|
||
//! whether soft ones are refused — so with `--dump DIR` the crops the
|
||
//! classifiers were shown are written out as PPMs, one per eye and one per
|
||
//! head framing, named by image and face, and every line carries the
|
||
//! numbers the readability floors are set from.
|
||
//!
|
||
//! cargo run -p dr-face --features inference --example eyes -- \
|
||
//! DET.onnx 2D106DET.onnx OCEC.onnx SGC.onnx [--dump DIR] photo.jpg [photo.jpg ...]
|
||
//!
|
||
//! All four models must have had their dynamic dims pinned first; see
|
||
//! `tools/fix-face-model-shapes.sh`.
|
||
|
||
use std::path::{Path, PathBuf};
|
||
use std::time::Instant;
|
||
|
||
use dr_face::{align, DetectOptions, Detector, EyeModels, Pixels};
|
||
|
||
fn main() {
|
||
env_logger::init();
|
||
|
||
let mut args: Vec<String> = std::env::args().skip(1).collect();
|
||
let dump = args.iter().position(|a| a == "--dump").map(|i| {
|
||
args.remove(i);
|
||
PathBuf::from(args.remove(i))
|
||
});
|
||
if args.len() < 5 {
|
||
eprintln!(
|
||
"usage: eyes DET.onnx 2D106DET.onnx OCEC.onnx SGC.onnx [--dump DIR] IMAGE.jpg [IMAGE.jpg ...]"
|
||
);
|
||
std::process::exit(2);
|
||
}
|
||
if let Some(d) = &dump {
|
||
std::fs::create_dir_all(d).expect("dump dir");
|
||
}
|
||
|
||
let t = Instant::now();
|
||
let mut detector = Detector::from_path(&args[0]).expect("load detector");
|
||
let mut models = EyeModels::from_paths(&args[1], &args[2], &args[3]).expect("load eye models");
|
||
println!("loaded the models in {:?}", t.elapsed());
|
||
|
||
let opts = DetectOptions::default();
|
||
for path in &args[4..] {
|
||
let (rgb, w, h) = match load_jpeg(path) {
|
||
Ok(v) => v,
|
||
Err(e) => {
|
||
println!("{path}: {e}");
|
||
continue;
|
||
}
|
||
};
|
||
let dets = detector.detect(&rgb, w, h, &opts).expect("detect");
|
||
println!("\n{path} ({w}×{h}) {} face(s)", dets.len());
|
||
|
||
let stem = Path::new(path)
|
||
.file_stem()
|
||
.map(|s| s.to_string_lossy().into_owned())
|
||
.unwrap_or_default();
|
||
|
||
for (i, d) in dets.iter().enumerate() {
|
||
let px = Pixels::RgbF32(&rgb);
|
||
let t = Instant::now();
|
||
let reading = models
|
||
.read(px, w, h, d.bbox, &d.landmarks)
|
||
.expect("read eyes");
|
||
let ms = t.elapsed().as_secs_f64() * 1e3;
|
||
let Some(r) = reading else {
|
||
println!(" [{i}] nothing to cut, skipped");
|
||
continue;
|
||
};
|
||
println!(
|
||
" [{i}] conf {:.2} box {:.0}×{:.0} right {:.3} ({:.0}px, sharp {:.3}) left {:.3} ({:.0}px, sharp {:.3}) sunglasses {:.3} → {:?} ({ms:.1} ms)",
|
||
d.confidence,
|
||
d.width(),
|
||
d.height(),
|
||
r.right.open,
|
||
r.right.px,
|
||
r.right.sharpness,
|
||
r.left.open,
|
||
r.left.px,
|
||
r.left.sharpness,
|
||
r.sunglasses,
|
||
r.state(),
|
||
);
|
||
if let Some(dir) = &dump {
|
||
// The same crops `EyeModels::read` cut, cut again for the
|
||
// sheet: the reading itself carries numbers, not pixels.
|
||
if let Some(lm) = models
|
||
.landmarks
|
||
.landmarks(px, w, h, d.bbox)
|
||
.expect("landmarks")
|
||
{
|
||
for (name, contour) in [("right", lm.right_eye()), ("left", lm.left_eye())] {
|
||
if let Some(patch) =
|
||
align::eye_box(&contour).and_then(|b| align::eye_patch(px, w, h, b))
|
||
{
|
||
write_ppm(
|
||
&dir.join(format!("{stem}-{i}-{name}.ppm")),
|
||
patch.pixels(),
|
||
align::EYE_PATCH_WIDTH,
|
||
align::EYE_PATCH_HEIGHT,
|
||
);
|
||
}
|
||
}
|
||
}
|
||
if let Some(head) = align::head_views(px, w, h, &d.landmarks) {
|
||
for (n, view) in head.views().enumerate() {
|
||
write_ppm(
|
||
&dir.join(format!("{stem}-{i}-head{n}.ppm")),
|
||
view,
|
||
align::SUNGLASSES_EDGE,
|
||
align::SUNGLASSES_EDGE,
|
||
);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
fn write_ppm(path: &Path, rgb: &[f32], w: usize, h: usize) {
|
||
let mut out = format!("P6\n{w} {h}\n255\n").into_bytes();
|
||
out.extend(
|
||
rgb.iter()
|
||
.map(|v| (v.clamp(0.0, 1.0) * 255.0).round() as u8),
|
||
);
|
||
std::fs::write(path, out).expect("write ppm");
|
||
}
|
||
|
||
/// Decode to the tightly packed `f32` RGB `0.0..=1.0` the crate expects.
|
||
fn load_jpeg(path: &str) -> Result<(Vec<f32>, usize, usize), String> {
|
||
let bytes = std::fs::read(path).map_err(|e| e.to_string())?;
|
||
let mut dec = zune_jpeg::JpegDecoder::new(&bytes);
|
||
let px = dec.decode().map_err(|e| e.to_string())?;
|
||
let info = dec.info().ok_or("no jpeg header")?;
|
||
let (w, h) = (info.width as usize, info.height as usize);
|
||
|
||
let rgb: Vec<f32> = match px.len() / (w * h) {
|
||
3 => px.iter().map(|&v| v as f32 / 255.0).collect(),
|
||
1 => px
|
||
.iter()
|
||
.flat_map(|&v| {
|
||
let g = v as f32 / 255.0;
|
||
[g, g, g]
|
||
})
|
||
.collect(),
|
||
n => return Err(format!("{n} components per pixel, expected 1 or 3")),
|
||
};
|
||
Ok((rgb, w, h))
|
||
}
|