Read each face's eyes, and whether sunglasses hide them
Two MIT classifiers from the same author as the reference pipeline's whole-body detector: OCEC answers P(open) for one 40×24 eye, SGC P(sunglasses) for a 48×48 head. Both load in tract once their batch dimension is pinned by tools/fix-face-model-shapes.sh, like the embedder. The crops come through the same fitted similarity the aligned face does, so an eye window is a constant in template units rather than a second warp, and a tilted head yields an upright eye. Measured on 60 proxies from the reference library: the eye window plateaus at 22×11, the S variant beats M and L (which overfit their own domain), and for sunglasses the aligned face beats a head framing but the higher of the two catches 11 of 12 pairs against 9 for either alone. The reading keeps both eyes and the sunglasses number apart, because a wink averages to the least informative value and a lens of dark glass draws a confident answer from the eye classifier — over a woman in sunglasses it read the right eye 0.97 open. Sunglasses take precedence, and a face behind them is neither open nor a blink.
This commit is contained in:
+398
-14
@@ -351,27 +351,278 @@ pub fn warp_pixels(
|
||||
let m = fit_similarity(landmarks, &ARCFACE_TEMPLATE)?;
|
||||
|
||||
let e = ALIGNED_EDGE;
|
||||
let mut pixels = vec![0.0_f32; e * e * 3];
|
||||
for v in 0..e {
|
||||
for u in 0..e {
|
||||
// Pixel centres, so the transform is not off by half a pixel —
|
||||
// which is small enough to survive review and large enough to
|
||||
// matter on a 40-pixel face.
|
||||
let (x, y) = m.invert(u as f32 + 0.5, v as f32 + 0.5);
|
||||
let (x, y) = (x - 0.5, y - 0.5);
|
||||
let out = (v * e + u) * 3;
|
||||
sample_bilinear(px, width, height, x, y, &mut pixels[out..out + 3]);
|
||||
}
|
||||
}
|
||||
|
||||
let window = TemplateWindow {
|
||||
x: 0.0,
|
||||
y: 0.0,
|
||||
w: e as f32,
|
||||
h: e as f32,
|
||||
};
|
||||
Some(Aligned112 {
|
||||
pixels,
|
||||
pixels: sample_window(px, width, height, &m, &window, e, e),
|
||||
// The warp maps `scale` source pixels to one destination pixel, so the
|
||||
// crop spans 112/scale of the source.
|
||||
source_px: ALIGNED_EDGE as f32 / m.scale(),
|
||||
})
|
||||
}
|
||||
|
||||
/// A rectangle in **template** coordinates — the 112-unit frame
|
||||
/// [`ARCFACE_TEMPLATE`] is written in — that a crop is sampled from.
|
||||
///
|
||||
/// Every crop this module makes is one of these resampled through the same
|
||||
/// fitted similarity: the aligned face is the window `(0, 0, 112, 112)`, an
|
||||
/// eye is a small window around its template point, a head is a window larger
|
||||
/// than the face. Stating them all in one frame is what lets a second crop be
|
||||
/// added as a constant rather than a second warp, and what keeps them
|
||||
/// consistent with each other — the eye window sits where the eye landmark
|
||||
/// lands *after* alignment, so a tilted face gets an upright eye.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
struct TemplateWindow {
|
||||
x: f32,
|
||||
y: f32,
|
||||
w: f32,
|
||||
h: f32,
|
||||
}
|
||||
|
||||
/// Resample `window` of the template frame into an `out_w × out_h` RGB buffer.
|
||||
///
|
||||
/// Bilinear, from the source, in one step — the property [`warp`] insists on,
|
||||
/// and every crop through here inherits it. The output pixel `(u, v)` is placed
|
||||
/// at its centre in the window, taken back through `m` to source coordinates,
|
||||
/// and sampled there; the window's aspect is **not** preserved when it differs
|
||||
/// from the output's, which is deliberate for the eye classifier (it was
|
||||
/// trained on detector boxes resized the same way) and moot for the others.
|
||||
fn sample_window(
|
||||
px: Pixels<'_>,
|
||||
width: usize,
|
||||
height: usize,
|
||||
m: &Similarity,
|
||||
window: &TemplateWindow,
|
||||
out_w: usize,
|
||||
out_h: usize,
|
||||
) -> Vec<f32> {
|
||||
let mut pixels = vec![0.0_f32; out_w * out_h * 3];
|
||||
let sx = window.w / out_w as f32;
|
||||
let sy = window.h / out_h as f32;
|
||||
for v in 0..out_h {
|
||||
for u in 0..out_w {
|
||||
// Pixel centres, so the transform is not off by half a pixel —
|
||||
// which is small enough to survive review and large enough to
|
||||
// matter on a 40-pixel face.
|
||||
let tx = window.x + (u as f32 + 0.5) * sx;
|
||||
let ty = window.y + (v as f32 + 0.5) * sy;
|
||||
let (x, y) = m.invert(tx, ty);
|
||||
let (x, y) = (x - 0.5, y - 0.5);
|
||||
let out = (v * out_w + u) * 3;
|
||||
sample_bilinear(px, width, height, x, y, &mut pixels[out..out + 3]);
|
||||
}
|
||||
}
|
||||
pixels
|
||||
}
|
||||
|
||||
// ── eyes ──────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Width of an eye crop as the classifier reads it, in pixels. Fixed by the
|
||||
/// OCEC input (`docs/faces.md` §17): 40 wide, 24 high.
|
||||
pub const EYE_PATCH_WIDTH: usize = 40;
|
||||
/// Height of an eye crop as the classifier reads it, in pixels.
|
||||
pub const EYE_PATCH_HEIGHT: usize = 24;
|
||||
|
||||
/// The window read around each eye, in template units: width and height.
|
||||
///
|
||||
/// The classifier was trained on the *eye* boxes of a whole-body detector —
|
||||
/// tight boxes round the palpebral fissure, on the reference footage about
|
||||
/// twice as wide as they are high — and this is that box expressed in the
|
||||
/// aligned frame, where the two eyes sit 35 template units apart. A human eye
|
||||
/// is close to half the interocular distance wide, so the first guess was
|
||||
/// 20×10; measured over 25 clearly open-eyed faces from the reference
|
||||
/// library (`examples/eyes.rs --eye`), recall was flat from 20×10 to 34×17
|
||||
/// and fell off below it, and 22×11 was the best of the plateau. docs/faces.md
|
||||
/// §17 has the table.
|
||||
pub const EYE_WINDOW: (f32, f32) = (22.0, 11.0);
|
||||
|
||||
/// One eye, resampled to the classifier's input.
|
||||
///
|
||||
/// Constructible only by [`eye_patches`], for the reason [`Aligned112`] is
|
||||
/// only constructible by [`warp`]: the classifier accepting a plain buffer
|
||||
/// would accept any 40×24 of anything, and its answer would still be a
|
||||
/// plausible probability.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct EyePatch {
|
||||
/// `24 × 40 × 3`, row-major RGB in `0.0..=1.0`.
|
||||
pixels: Vec<f32>,
|
||||
}
|
||||
|
||||
impl EyePatch {
|
||||
pub fn pixels(&self) -> &[f32] {
|
||||
&self.pixels
|
||||
}
|
||||
}
|
||||
|
||||
/// Both eyes of one face, in the detector's landmark order.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct EyePatches {
|
||||
/// The subject's **right** eye — image-left, landmark 0.
|
||||
pub right: EyePatch,
|
||||
/// The subject's **left** eye — image-right, landmark 1.
|
||||
pub left: EyePatch,
|
||||
}
|
||||
|
||||
/// Cut both eyes out of the source, aligned, at the classifier's size.
|
||||
///
|
||||
/// The same similarity [`warp`] fits, so the eyes come out upright whatever
|
||||
/// the head's tilt, and the same one-step bilinear sampling from the native
|
||||
/// buffer, so a large face gives the classifier real pixels rather than a
|
||||
/// re-enlargement of the 112-pixel crop. A face too small for the window to
|
||||
/// hold a real eye is not refused here: the classifier was trained down to
|
||||
/// eyes a dozen pixels across, and the caller's size gate has already spoken.
|
||||
pub fn eye_patches(
|
||||
px: Pixels<'_>,
|
||||
width: usize,
|
||||
height: usize,
|
||||
landmarks: &[(f32, f32); 5],
|
||||
) -> Option<EyePatches> {
|
||||
eye_patches_in(px, width, height, landmarks, EYE_WINDOW)
|
||||
}
|
||||
|
||||
/// [`eye_patches`] over a window other than [`EYE_WINDOW`].
|
||||
///
|
||||
/// For measuring the window, which is how [`EYE_WINDOW`] was chosen
|
||||
/// (`examples/eyes.rs --eye`); production callers use the constant.
|
||||
pub fn eye_patches_in(
|
||||
px: Pixels<'_>,
|
||||
width: usize,
|
||||
height: usize,
|
||||
landmarks: &[(f32, f32); 5],
|
||||
window: (f32, f32),
|
||||
) -> Option<EyePatches> {
|
||||
if !px.fits(width, height) {
|
||||
return None;
|
||||
}
|
||||
let m = fit_similarity(landmarks, &ARCFACE_TEMPLATE)?;
|
||||
let (ww, wh) = window;
|
||||
let eye = |i: usize| {
|
||||
let (cx, cy) = ARCFACE_TEMPLATE[i];
|
||||
let window = TemplateWindow {
|
||||
x: cx - ww / 2.0,
|
||||
y: cy - wh / 2.0,
|
||||
w: ww,
|
||||
h: wh,
|
||||
};
|
||||
EyePatch {
|
||||
pixels: sample_window(
|
||||
px,
|
||||
width,
|
||||
height,
|
||||
&m,
|
||||
&window,
|
||||
EYE_PATCH_WIDTH,
|
||||
EYE_PATCH_HEIGHT,
|
||||
),
|
||||
}
|
||||
};
|
||||
Some(EyePatches {
|
||||
right: eye(0),
|
||||
left: eye(1),
|
||||
})
|
||||
}
|
||||
|
||||
// ── sunglasses ────────────────────────────────────────────────────────────
|
||||
|
||||
/// Edge of the crop the sunglasses classifier reads. Fixed by the SGC input:
|
||||
/// 48×48.
|
||||
pub const SUNGLASSES_EDGE: usize = 48;
|
||||
|
||||
/// The windows read for the sunglasses classifier, in template units:
|
||||
/// `(x, y, w, h)`.
|
||||
///
|
||||
/// **Two framings, and the classifier's answer is the higher of the two.**
|
||||
/// It was trained on a whole-body detector's *head* boxes, and a head box
|
||||
/// is not reproducible from five landmarks: how much hair and hat it took in
|
||||
/// depended on the person. So it is shown the face twice — once as the
|
||||
/// aligned crop itself, once shifted up and widened to take in hair and
|
||||
/// hat at the cost of the chin, which is roughly where a head box falls —
|
||||
/// and a pair of sunglasses counts if it looks like one in either.
|
||||
///
|
||||
/// Measured over 12 faces in sunglasses and 28 with plainly visible eyes
|
||||
/// from the reference library (`examples/eyes.rs --head`), at the 0.5
|
||||
/// threshold:
|
||||
///
|
||||
/// | window | sunglasses found | clear eyes kept |
|
||||
/// |---|---|---|
|
||||
/// | the aligned face, `(0, 0, 112, 112)` | 9 | 28 |
|
||||
/// | a head, `(-5, -14, 122, 122)` | 6 | 27 |
|
||||
/// | a larger head, `(-30, -55, 172, 190)` | 6 | 25 |
|
||||
/// | **the higher of the first two** | **11** | 27 |
|
||||
///
|
||||
/// The face-tight crop alone was the best single framing, which was not the
|
||||
/// expectation; the head framing found the sunglasses under a cap that the
|
||||
/// face crop missed. The one clear-eyed face the pair loses wears a cap and
|
||||
/// clear glasses, at 0.68. Erring towards "sunglasses" is the safe direction
|
||||
/// for what this feeds: a face called sunglasses is left alone by the
|
||||
/// eyes-open filter, where a pair of sunglasses missed hands the eye
|
||||
/// classifier a lens to guess at (docs/faces.md §17).
|
||||
pub const SUNGLASSES_WINDOWS: [(f32, f32, f32, f32); 2] =
|
||||
[(0.0, 0.0, 112.0, 112.0), (-5.0, -14.0, 122.0, 122.0)];
|
||||
|
||||
/// The framings of one face the sunglasses classifier is shown.
|
||||
///
|
||||
/// A newtype for the reason [`EyePatch`] is one.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct HeadViews {
|
||||
/// Each `48 × 48 × 3`, row-major RGB in `0.0..=1.0`.
|
||||
views: Vec<Vec<f32>>,
|
||||
}
|
||||
|
||||
impl HeadViews {
|
||||
pub fn views(&self) -> impl Iterator<Item = &[f32]> {
|
||||
self.views.iter().map(Vec::as_slice)
|
||||
}
|
||||
}
|
||||
|
||||
/// Cut the [`SUNGLASSES_WINDOWS`] out of the source, aligned, at the
|
||||
/// classifier's size.
|
||||
pub fn head_views(
|
||||
px: Pixels<'_>,
|
||||
width: usize,
|
||||
height: usize,
|
||||
landmarks: &[(f32, f32); 5],
|
||||
) -> Option<HeadViews> {
|
||||
head_views_in(px, width, height, landmarks, &SUNGLASSES_WINDOWS)
|
||||
}
|
||||
|
||||
/// [`head_views`] over windows other than [`SUNGLASSES_WINDOWS`].
|
||||
///
|
||||
/// For measuring them, which is how the constant was chosen
|
||||
/// (`examples/eyes.rs --head`); production callers use the constant.
|
||||
pub fn head_views_in(
|
||||
px: Pixels<'_>,
|
||||
width: usize,
|
||||
height: usize,
|
||||
landmarks: &[(f32, f32); 5],
|
||||
windows: &[(f32, f32, f32, f32)],
|
||||
) -> Option<HeadViews> {
|
||||
if !px.fits(width, height) || windows.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let m = fit_similarity(landmarks, &ARCFACE_TEMPLATE)?;
|
||||
let views = windows
|
||||
.iter()
|
||||
.map(|&(x, y, w, h)| {
|
||||
let window = TemplateWindow { x, y, w, h };
|
||||
sample_window(
|
||||
px,
|
||||
width,
|
||||
height,
|
||||
&m,
|
||||
&window,
|
||||
SUNGLASSES_EDGE,
|
||||
SUNGLASSES_EDGE,
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
Some(HeadViews { views })
|
||||
}
|
||||
|
||||
fn sample_bilinear(px: Pixels<'_>, w: usize, h: usize, x: f32, y: f32, out: &mut [f32]) {
|
||||
let x0 = x.floor();
|
||||
let y0 = y.floor();
|
||||
@@ -489,6 +740,139 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// A source whose red channel is its x coordinate and green its y, so a
|
||||
/// crop's mean colour says where in the source it was taken from.
|
||||
fn coordinate_image(w: usize, h: usize) -> Vec<f32> {
|
||||
let mut rgb = vec![0.0_f32; w * h * 3];
|
||||
for y in 0..h {
|
||||
for x in 0..w {
|
||||
rgb[(y * w + x) * 3] = x as f32 / w as f32;
|
||||
rgb[(y * w + x) * 3 + 1] = y as f32 / h as f32;
|
||||
}
|
||||
}
|
||||
rgb
|
||||
}
|
||||
|
||||
fn mean_channel(px: &[f32], c: usize) -> f32 {
|
||||
let n = px.len() / 3;
|
||||
px.chunks_exact(3).map(|p| p[c]).sum::<f32>() / n as f32
|
||||
}
|
||||
|
||||
/// The eye windows are cut where the landmarks say the eyes are, in the
|
||||
/// detector's order — subject's right (image-left) first.
|
||||
#[test]
|
||||
fn eye_patches_are_cut_around_each_eye_landmark() {
|
||||
let (w, h) = (224, 224);
|
||||
let rgb = coordinate_image(w, h);
|
||||
// Pure translation by (56, 56): template (x, y) is source (x+56, y+56).
|
||||
let lm = shifted_scaled(1.0, 56.0, 56.0, 0.0);
|
||||
let eyes = eye_patches(Pixels::RgbF32(&rgb), w, h, &lm).unwrap();
|
||||
assert_eq!(
|
||||
eyes.right.pixels().len(),
|
||||
EYE_PATCH_WIDTH * EYE_PATCH_HEIGHT * 3
|
||||
);
|
||||
|
||||
for (patch, (tx, ty)) in [
|
||||
(&eyes.right, ARCFACE_TEMPLATE[0]),
|
||||
(&eyes.left, ARCFACE_TEMPLATE[1]),
|
||||
] {
|
||||
let want_x = (tx + 56.0) / w as f32;
|
||||
let want_y = (ty + 56.0) / h as f32;
|
||||
let got_x = mean_channel(patch.pixels(), 0);
|
||||
let got_y = mean_channel(patch.pixels(), 1);
|
||||
assert!((got_x - want_x).abs() < 0.01, "x {got_x} vs {want_x}");
|
||||
assert!((got_y - want_y).abs() < 0.01, "y {got_y} vs {want_y}");
|
||||
}
|
||||
// And the two are distinct eyes, the right one image-left of the left.
|
||||
assert!(mean_channel(eyes.right.pixels(), 0) < mean_channel(eyes.left.pixels(), 0));
|
||||
}
|
||||
|
||||
/// The window is wider than it is high in the source, and is resampled to
|
||||
/// the classifier's 40×24 without keeping that aspect — the red channel
|
||||
/// spans `EYE_WINDOW.0` source pixels across 40 output columns.
|
||||
#[test]
|
||||
fn an_eye_patch_spans_the_window_it_was_asked_for() {
|
||||
let (w, h) = (224, 224);
|
||||
let rgb = coordinate_image(w, h);
|
||||
let lm = shifted_scaled(1.0, 56.0, 56.0, 0.0);
|
||||
let eyes = eye_patches(Pixels::RgbF32(&rgb), w, h, &lm).unwrap();
|
||||
let px = eyes.right.pixels();
|
||||
let row = |v: usize| &px[v * EYE_PATCH_WIDTH * 3..(v + 1) * EYE_PATCH_WIDTH * 3];
|
||||
let first = row(0)[0];
|
||||
let last = row(0)[(EYE_PATCH_WIDTH - 1) * 3];
|
||||
let span = (last - first) * w as f32;
|
||||
// 39 pixel-centre steps across a 20-unit window.
|
||||
let want = EYE_WINDOW.0 * (EYE_PATCH_WIDTH as f32 - 1.0) / EYE_PATCH_WIDTH as f32;
|
||||
assert!((span - want).abs() < 0.1, "span {span} vs {want}");
|
||||
}
|
||||
|
||||
/// A tilted face yields upright eyes: the patch's rows run along the
|
||||
/// interocular line, not along the image's x axis.
|
||||
#[test]
|
||||
fn eye_patches_follow_the_heads_tilt() {
|
||||
let (w, h) = (300, 300);
|
||||
let rgb = coordinate_image(w, h);
|
||||
let rot = 0.5_f32;
|
||||
let lm = shifted_scaled(1.0, 100.0, 60.0, rot);
|
||||
let eyes = eye_patches(Pixels::RgbF32(&rgb), w, h, &lm).unwrap();
|
||||
let px = eyes.left.pixels();
|
||||
// Walking one output row moves along the rotated x axis, so both
|
||||
// source coordinates change, in the ratio the rotation dictates.
|
||||
let a = &px[0..3];
|
||||
let b = &px[(EYE_PATCH_WIDTH - 1) * 3..EYE_PATCH_WIDTH * 3];
|
||||
let dx = (b[0] - a[0]) * w as f32;
|
||||
let dy = (b[1] - a[1]) * h as f32;
|
||||
let angle = dy.atan2(dx);
|
||||
assert!(
|
||||
(angle - rot).abs() < 0.02,
|
||||
"row runs at {angle}, want {rot}"
|
||||
);
|
||||
}
|
||||
|
||||
/// The second sunglasses framing takes in more than the face — it starts
|
||||
/// above the template's top edge and ends below its bottom — and the
|
||||
/// first is the aligned face itself.
|
||||
#[test]
|
||||
fn the_head_views_are_the_face_and_a_wider_framing_of_it() {
|
||||
let (w, h) = (300, 300);
|
||||
let rgb = coordinate_image(w, h);
|
||||
let lm = shifted_scaled(1.0, 100.0, 100.0, 0.0);
|
||||
let head = head_views(Pixels::RgbF32(&rgb), w, h, &lm).unwrap();
|
||||
let views: Vec<&[f32]> = head.views().collect();
|
||||
let face = warp(&rgb, w, h, &lm).unwrap();
|
||||
assert_eq!(views.len(), SUNGLASSES_WINDOWS.len());
|
||||
for v in &views {
|
||||
assert_eq!(v.len(), SUNGLASSES_EDGE * SUNGLASSES_EDGE * 3);
|
||||
}
|
||||
|
||||
// The face view samples the same region as the aligned crop.
|
||||
assert!((mean_channel(views[0], 0) - mean_channel(face.pixels(), 0)).abs() < 0.01);
|
||||
assert!((mean_channel(views[0], 1) - mean_channel(face.pixels(), 1)).abs() < 0.01);
|
||||
|
||||
let (x, y, ww, hh) = SUNGLASSES_WINDOWS[1];
|
||||
assert!(
|
||||
x < 0.0 && y < 0.0,
|
||||
"the window starts outside the face crop"
|
||||
);
|
||||
assert!(x + ww > ALIGNED_EDGE as f32, "and is wider than it");
|
||||
assert!(y + hh < ALIGNED_EDGE as f32, "but stops short of the chin");
|
||||
// Centred horizontally on the face, so the two share a mean x.
|
||||
assert!((mean_channel(views[1], 0) - mean_channel(face.pixels(), 0)).abs() < 0.01);
|
||||
// Its first row lies above the face's first row.
|
||||
assert!(views[1][1] < face.pixels()[1]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn degenerate_landmarks_yield_no_eye_or_head_crop() {
|
||||
let rgb = vec![0.5_f32; 64 * 64 * 3];
|
||||
let degenerate = [(50.0, 50.0); 5];
|
||||
assert!(eye_patches(Pixels::RgbF32(&rgb), 64, 64, °enerate).is_none());
|
||||
assert!(head_views(Pixels::RgbF32(&rgb), 64, 64, °enerate).is_none());
|
||||
// And a buffer that is not the size it claims.
|
||||
let lm = shifted_scaled(1.0, 0.0, 0.0, 0.0);
|
||||
assert!(eye_patches(Pixels::RgbF32(&rgb), 60, 60, &lm).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn out_of_bounds_samples_read_black_rather_than_wrapping() {
|
||||
let rgb = vec![1.0_f32; 32 * 32 * 3];
|
||||
|
||||
Reference in New Issue
Block a user