//! Five-point face alignment (docs/faces.md §5). //! //! ArcFace embeddings are trained on faces warped to a canonical 112×112 //! arrangement. Feeding the model a plain bounding-box crop *works* — it //! produces 512 numbers, they are unit-norm, and cosine similarities between //! them look entirely reasonable. They are just much worse, and nothing in the //! system reports it. //! //! That is the whole reason this module exists, and the reason [`Aligned112`] //! is a newtype only [`warp`] can construct: the mistake is not one a reviewer //! catches, so the type system catches it instead. //! //! Model-free, so it builds and tests without the `inference` feature. /// Canonical landmark positions for a 112×112 ArcFace crop. /// /// # The naming is a trap; the order is not /// /// Point 0 sits at x=38 on a 112-wide canvas — left of centre *in the image*, /// which is the subject's **right** eye. Both namings are in circulation and /// they are opposite, so the array is written in the detector's order and the /// comment says whose left is whose: /// /// ```text /// 0 subject's right eye (image-left) /// 1 subject's left eye (image-right) /// 2 nose tip /// 3 subject's right mouth corner /// 4 subject's left mouth corner /// ``` /// /// SCRFD emits its five points in this same order, so the correct amount of /// reordering between detector and template is **none**. A detector with a /// different order carries its own permutation beside its model id rather than /// this constant growing an assumption. pub const ARCFACE_TEMPLATE: [(f32, f32); 5] = [ (38.2946, 51.6963), (73.5318, 51.5014), (56.0252, 71.7366), (41.5493, 92.3655), (70.7299, 92.2041), ]; /// Edge of the aligned crop, in pixels. Fixed by the embedder's input. pub const ALIGNED_EDGE: usize = 112; /// A face warped to [`ARCFACE_TEMPLATE`], ready for the embedder. /// /// Constructible only by [`warp`]. That is the point: an `Embedder` that took /// a plain `&[f32]` would accept an unaligned bounding-box crop and silently /// return worse embeddings, which is a failure no test of the embedder itself /// would catch. pub struct Aligned112 { /// `112 × 112 × 3`, row-major RGB in `0.0..=1.0`. pixels: Vec, /// Source pixels across the crop before warping — `crop_px` in the catalog. /// /// Carried here rather than recomputed later because the scale factor is /// known exactly at warp time and only approximately from the box /// afterwards. §7: it is the honest quality signal, and a feature in the /// calibration. source_px: f32, } impl Aligned112 { pub fn pixels(&self) -> &[f32] { &self.pixels } /// Source pixels spanned by the 112-pixel crop. /// /// Below ~112 the face was upsampled to reach the embedder and the /// embedding is correspondingly weaker; above it, downsampled and healthy. pub fn source_px(&self) -> f32 { self.source_px } /// How sharp the face the embedder is about to see actually is. /// /// # Why size is not enough /// /// A face can be large and useless. A subject walking through a half-second /// exposure, a frame focused on the person behind them, a hand-held shot at /// 1/15 — all yield a big box, a confident detection and five landmarks in /// plausible places. The embedding that comes back is not *wrong* in any /// way the system can see: it is unit-norm and its cosines look ordinary. /// It is simply an embedding of a blur, and blurs resemble each other more /// than they resemble the people they were, so they cluster together and /// bridge identities that have nothing to do with one another. /// /// That is the failure this exists to prevent, and it is the same class of /// fault as the unaligned-crop one the [`Aligned112`] newtype guards /// against: plausible output, no error, worse results, nothing reported. /// /// # The measure /// /// Variance of the Laplacian — the standard blur metric — **divided by the /// variance of the luma it was taken over**. The division is what makes it /// usable here. Raw Laplacian variance scales with contrast, so a sharp /// face in flat, hazy or backlit light scores like a blurred one in hard /// light, and a threshold on it would quietly throw away every face shot /// against a bright sky. The ratio asks the question that actually matters /// — *how much of this crop's variation is edges rather than broad /// gradients* — and is invariant to exposure and contrast. /// /// Computed on luma over the interior, so the 3x3 kernel never needs a /// border rule. Returns 0.0 for a crop with no variation at all, which is /// a flat patch and correctly unusable rather than infinitely sharp. /// /// # This is not independent of size /// /// A face smaller than 112 pixels was *upsampled* to reach the embedder, /// and upsampling invents no edges — so a small face scores low here even /// when the original was perfectly sharp. That is not a flaw to correct: it /// is the honest statement that the embedder is looking at a soft image. /// The size floor and this one overlap deliberately, and /// `face_index --quality` prints the joint distribution so the two are /// chosen together rather than each in ignorance of the other. pub fn sharpness(&self) -> f32 { laplacian_ratio(&self.pixels, ALIGNED_EDGE, ALIGNED_EDGE) } } /// Variance of the four-neighbour Laplacian over the variance of the luma, /// for a `w × h` RGB crop — the measure [`Aligned112::sharpness`] describes, /// shared with [`EyePatch::sharpness`]. fn laplacian_ratio(pixels: &[f32], w: usize, h: usize) -> f32 { let luma: Vec = pixels .chunks_exact(3) .map(|p| 0.2126 * p[0] + 0.7152 * p[1] + 0.0722 * p[2]) .collect(); let (mut lap_sum, mut lap_sq) = (0.0_f64, 0.0_f64); let (mut lum_sum, mut lum_sq) = (0.0_f64, 0.0_f64); let mut n = 0.0_f64; for y in 1..h.saturating_sub(1) { for x in 1..w.saturating_sub(1) { let i = y * w + x; // Four-neighbour Laplacian. The 8-neighbour form is more // sensitive to diagonal detail and also to noise, which on a // high-ISO frame is exactly the thing that must not read as // sharpness. let lap = 4.0 * luma[i] - luma[i - 1] - luma[i + 1] - luma[i - w] - luma[i + w]; let lap = lap as f64; lap_sum += lap; lap_sq += lap * lap; let l = luma[i] as f64; lum_sum += l; lum_sq += l * l; n += 1.0; } } if n == 0.0 { return 0.0; } let lap_var = (lap_sq / n - (lap_sum / n).powi(2)).max(0.0); let lum_var = (lum_sq / n - (lum_sum / n).powi(2)).max(0.0); // A crop with no luma variation has no edges to find either, so the // ratio is 0/0. Zero is the right answer: nothing there is a face. if lum_var <= 1e-9 { return 0.0; } (lap_var / lum_var) as f32 } /// A similarity transform: rotation, uniform scale, translation. /// /// Stored as the four independent parameters rather than a 2×3 matrix so that /// [`Similarity::scale`] is readable without a decomposition. #[derive(Debug, Clone, Copy, PartialEq)] pub struct Similarity { a: f32, b: f32, tx: f32, ty: f32, } impl Similarity { /// `x' = a·x − b·y + tx`, `y' = b·x + a·y + ty`. pub fn apply(&self, x: f32, y: f32) -> (f32, f32) { ( self.a * x - self.b * y + self.tx, self.b * x + self.a * y + self.ty, ) } /// Uniform scale factor — destination pixels per source pixel. pub fn scale(&self) -> f32 { (self.a * self.a + self.b * self.b).sqrt() } fn invert(&self, u: f32, v: f32) -> (f32, f32) { let det = self.a * self.a + self.b * self.b; let du = u - self.tx; let dv = v - self.ty; ( (self.a * du + self.b * dv) / det, (-self.b * du + self.a * dv) / det, ) } } /// Least-squares similarity transform from `src` onto `dst`. /// /// # Why least squares and not RANSAC /// /// The reference C++ implementation (docs/faces.md §1.1) fits this with /// OpenCV's `estimateAffinePartial2D` under RANSAC. RANSAC over five points is /// a strange fit: the minimal sample for a similarity is two, so it can discard /// landmarks it judges outliers and solve from a subset — and on a profile face /// the "outlier" is as likely to be the correct geometry as the wrong one. /// InsightFace's own pipeline uses plain least squares over all five points, /// which cannot silently drop anything, and that is what this is. /// /// # The closed form /// /// A 2-D similarity is linear in its four parameters: /// /// ```text /// x' = a·x − b·y + tx /// y' = b·x + a·y + ty /// ``` /// /// so this is an ordinary linear least-squares problem, not an SVD one. /// Centring both point sets kills `tx`/`ty` from the normal equations and /// leaves `a` and `b` as two dot products over a common denominator — which is /// why there is no matrix decomposition anywhere in this function. /// /// Returns `None` when the source points are degenerate (coincident or /// collinear to within f32), which does happen: a detector firing on a /// motion-blurred profile can put all five landmarks on a line. pub fn fit_similarity(src: &[(f32, f32); 5], dst: &[(f32, f32); 5]) -> Option { let n = 5.0_f32; let (mut sx, mut sy, mut dx, mut dy) = (0.0, 0.0, 0.0, 0.0); for i in 0..5 { sx += src[i].0; sy += src[i].1; dx += dst[i].0; dy += dst[i].1; } let (sx, sy, dx, dy) = (sx / n, sy / n, dx / n, dy / n); let mut var = 0.0_f32; let mut num_a = 0.0_f32; let mut num_b = 0.0_f32; for i in 0..5 { let (px, py) = (src[i].0 - sx, src[i].1 - sy); let (qx, qy) = (dst[i].0 - dx, dst[i].1 - dy); var += px * px + py * py; num_a += px * qx + py * qy; num_b += px * qy - py * qx; } // Degenerate: every landmark on one point. Collinear input still solves, // but with a scale that can be absurd, so the caller's sanity check on // `scale()` is what catches that case. if var <= f32::EPSILON { return None; } let a = num_a / var; let b = num_b / var; if !a.is_finite() || !b.is_finite() || (a * a + b * b) <= f32::EPSILON { return None; } Some(Similarity { a, b, tx: dx - (a * sx - b * sy), ty: dy - (b * sx + a * sy), }) } /// Warp a face onto the canonical 112×112 arrangement. /// /// `rgb` is tightly packed `f32` RGB in `0.0..=1.0`, row-major — the same /// convention `dr-segment` uses, so both read the same proxy. /// /// Sampling is bilinear **from the source in one step**: never crop-then-warp, /// which resamples twice and throws away detail the warp could have used. /// Pixels falling outside the source read as black. pub fn warp( rgb: &[f32], width: usize, height: usize, landmarks: &[(f32, f32); 5], ) -> Option { warp_pixels(Pixels::RgbF32(rgb), width, height, landmarks) } /// TRACES: FR-CULL-8 /// What the warp may sample, in whichever layout the caller already holds. /// /// # Why the 8-bit variant exists /// /// FR-CULL-8 requires the crop to come from the **native** render, and a native /// render is large: a 24 MP frame is 96 MB as `RGBA8` and 288 MB converted to /// the `f32` RGB this module was originally written against. Converting the /// whole frame to sample 112×112 from it is three hundred megabytes allocated /// to read about forty thousand pixels, per image, on a pass that runs over a /// whole library — and on Android it is NFR-RES-2's budget spent outright. /// /// So the warp reads whatever the caller has instead. It touches so few pixels /// that the per-sample conversion is free, and the buffer never has to be /// duplicated in another layout. #[derive(Debug, Clone, Copy)] pub enum Pixels<'a> { /// Tightly packed `f32` RGB in `0.0..=1.0`, row-major. RgbF32(&'a [f32]), /// Tightly packed 8-bit RGBA, row-major. Alpha is ignored: a face crop has /// no use for it and carrying it would change what the embedder receives. Rgba8(&'a [u8]), } impl Pixels<'_> { /// Whether the buffer is the size `width × height` implies. fn fits(&self, width: usize, height: usize) -> bool { match self { Pixels::RgbF32(v) => v.len() == width * height * 3, Pixels::Rgba8(v) => v.len() == width * height * 4, } } /// One channel of one pixel, as `0.0..=1.0`. Outside the buffer reads black. /// /// Public because the face *crop* stored for the People screen is cut from /// the same buffer by the same caller, and it should not need a second /// copy of this to do it. pub fn channel(&self, w: usize, h: usize, x: isize, y: isize, c: usize) -> f32 { if x < 0 || y < 0 || x >= w as isize || y >= h as isize { return 0.0; } let i = y as usize * w + x as usize; match self { Pixels::RgbF32(v) => v[i * 3 + c], Pixels::Rgba8(v) => v[i * 4 + c] as f32 / 255.0, } } } /// [`warp`], over any layout [`Pixels`] describes. pub fn warp_pixels( px: Pixels<'_>, width: usize, height: usize, landmarks: &[(f32, f32); 5], ) -> Option { if !px.fits(width, height) { return None; } let m = fit_similarity(landmarks, &ARCFACE_TEMPLATE)?; let e = ALIGNED_EDGE; let window = TemplateWindow { x: 0.0, y: 0.0, w: e as f32, h: e as f32, }; Some(Aligned112 { pixels: sample_window(px, width, height, &m, &window, e, e), // The warp maps `scale` source pixels to one destination pixel, so the // crop spans 112/scale of the source. source_px: ALIGNED_EDGE as f32 / m.scale(), }) } /// A rectangle in **template** coordinates — the 112-unit frame /// [`ARCFACE_TEMPLATE`] is written in — that a crop is sampled from. /// /// Every crop this module makes is one of these resampled through the same /// fitted similarity: the aligned face is the window `(0, 0, 112, 112)`, an /// eye is a small window around its template point, a head is a window larger /// than the face. Stating them all in one frame is what lets a second crop be /// added as a constant rather than a second warp, and what keeps them /// consistent with each other — the eye window sits where the eye landmark /// lands *after* alignment, so a tilted face gets an upright eye. #[derive(Debug, Clone, Copy, PartialEq)] struct TemplateWindow { x: f32, y: f32, w: f32, h: f32, } /// Resample `window` of the template frame into an `out_w × out_h` RGB buffer. /// /// Bilinear, from the source, in one step — the property [`warp`] insists on, /// and every crop through here inherits it. The output pixel `(u, v)` is placed /// at its centre in the window, taken back through `m` to source coordinates, /// and sampled there; the window's aspect is **not** preserved when it differs /// from the output's, which is deliberate for the eye classifier (it was /// trained on detector boxes resized the same way) and moot for the others. fn sample_window( px: Pixels<'_>, width: usize, height: usize, m: &Similarity, window: &TemplateWindow, out_w: usize, out_h: usize, ) -> Vec { let mut pixels = vec![0.0_f32; out_w * out_h * 3]; let sx = window.w / out_w as f32; let sy = window.h / out_h as f32; for v in 0..out_h { for u in 0..out_w { // Pixel centres, so the transform is not off by half a pixel — // which is small enough to survive review and large enough to // matter on a 40-pixel face. let tx = window.x + (u as f32 + 0.5) * sx; let ty = window.y + (v as f32 + 0.5) * sy; let (x, y) = m.invert(tx, ty); let (x, y) = (x - 0.5, y - 0.5); let out = (v * out_w + u) * 3; sample_bilinear(px, width, height, x, y, &mut pixels[out..out + 3]); } } pixels } // ── eyes ────────────────────────────────────────────────────────────────── /// Width of an eye crop as the classifier reads it, in pixels. Fixed by the /// OCEC input (`docs/faces.md` §17): 40 wide, 24 high. pub const EYE_PATCH_WIDTH: usize = 40; /// Height of an eye crop as the classifier reads it, in pixels. pub const EYE_PATCH_HEIGHT: usize = 24; /// How much an eye's box is grown beyond its lid contour, as a fraction of /// its width and height on each side. /// /// The classifier was trained on a whole-body detector's *eye* boxes — tight /// round the palpebral fissure — and measured on 25 open-eyed faces from the /// reference library, a tight box is what it wants: 22 of 25 read open at /// 0 and 0.1, 18 at 0.4, 14 at 0.6 (docs/faces.md §17.2). A tenth, so a /// contour landing a pixel short of the lashes still holds them. pub const EYE_BOX_MARGIN: f32 = 0.1; /// Height a shut eye's box is given, as a fraction of its width. /// /// A closed eye's contour has no height. The box is given the height an /// open eye of the same width would have, so the classifier sees the same /// framing either way — which is what it was trained on. pub const EYE_BOX_MIN_ASPECT: f32 = 0.4; /// The box round an eye's lid contour, in the contour's own coordinates: /// `(x, y, w, h)`. /// /// Model-free: the contour is whatever the landmark model gave for the ten /// (or so) points on the lids, in source pixels. `None` for an empty /// contour or one with no width, which is what a hidden eye's collapsed /// contour can come to. pub fn eye_box(contour: &[(f32, f32)]) -> Option<(f32, f32, f32, f32)> { let (mut x0, mut y0, mut x1, mut y1) = (f32::MAX, f32::MAX, f32::MIN, f32::MIN); for &(x, y) in contour { x0 = x0.min(x); y0 = y0.min(y); x1 = x1.max(x); y1 = y1.max(y); } let w = x1 - x0; if contour.is_empty() || !(w > 0.0) { return None; } let h = (y1 - y0).max(w * EYE_BOX_MIN_ASPECT); let cy = (y0 + y1) / 2.0; let (mx, my) = (w * EYE_BOX_MARGIN, h * EYE_BOX_MARGIN); Some((x0 - mx, cy - h / 2.0 - my, w + 2.0 * mx, h + 2.0 * my)) } /// One eye, resampled to the classifier's input. /// /// Constructible only by [`eye_patch`], for the reason [`Aligned112`] is /// only constructible by [`warp`]: the classifier accepting a plain buffer /// would accept any 40×24 of anything, and its answer would still be a /// plausible probability. #[derive(Debug, Clone, PartialEq)] pub struct EyePatch { /// `24 × 40 × 3`, row-major RGB in `0.0..=1.0`. pixels: Vec, /// Source pixels across the box the patch was cut from. source_px: f32, } impl EyePatch { pub fn pixels(&self) -> &[f32] { &self.pixels } /// Source pixels across the eye box — how much eye there was to read. /// /// The classifier was trained down to eyes a dozen pixels wide, and /// below that a crop is an interpolation of nothing; `crate::eyes` draws /// the line. Zero when the box had no width, which is a hidden eye. pub fn source_px(&self) -> f32 { self.source_px } /// How sharp the eye the classifier is about to see actually is — /// [`Aligned112::sharpness`]'s measure, over the patch. /// /// The reason it exists is the reason the face's does: a soft eye is /// not a closed one, but a classifier shown a smear says "closed" with /// the same confidence it says anything, and the only defence is to /// not ask. A face sharp enough to embed can still hold an eye too soft /// to read — it is a fortieth of the face — so the measure is taken /// here and not inherited from the crop. pub fn sharpness(&self) -> f32 { laplacian_ratio(&self.pixels, EYE_PATCH_WIDTH, EYE_PATCH_HEIGHT) } } /// Cut an eye out of the source at the classifier's size, from an /// axis-aligned box in source pixels — [`eye_box`]'s, as a rule. /// /// Upright and from the frame, not through the face's alignment: the /// classifier's training crops were detector boxes, and a landmark model's /// contour already says where the eye is on a tilted head. Bilinear in one /// step from the native buffer, so a large face gives real pixels; the /// box's aspect is not preserved, which is what the training resize did. pub fn eye_patch( px: Pixels<'_>, width: usize, height: usize, bbox: (f32, f32, f32, f32), ) -> Option { let pixels = crop_box(px, width, height, bbox, EYE_PATCH_WIDTH, EYE_PATCH_HEIGHT)?; Some(EyePatch { pixels, source_px: bbox.2, }) } // ── sunglasses ──────────────────────────────────────────────────────────── /// Edge of the crop the sunglasses classifier reads. Fixed by the SGC input: /// 48×48. pub const SUNGLASSES_EDGE: usize = 48; /// The windows read for the sunglasses classifier, in template units: /// `(x, y, w, h)`. /// /// **Two framings, and the classifier's answer is the higher of the two.** /// It was trained on a whole-body detector's *head* boxes, and a head box /// is not reproducible from five landmarks: how much hair and hat it took in /// depended on the person. So it is shown the face twice — once as the /// aligned crop itself, once shifted up and widened to take in hair and /// hat at the cost of the chin, which is roughly where a head box falls — /// and a pair of sunglasses counts if it looks like one in either. /// /// Measured over 12 faces in sunglasses and 28 with plainly visible eyes /// from the reference library (`examples/eyes.rs --head`), at the 0.5 /// threshold: /// /// | window | sunglasses found | clear eyes kept | /// |---|---|---| /// | the aligned face, `(0, 0, 112, 112)` | 9 | 28 | /// | a head, `(-5, -14, 122, 122)` | 6 | 27 | /// | a larger head, `(-30, -55, 172, 190)` | 6 | 25 | /// | **the higher of the first two** | **11** | 27 | /// /// The face-tight crop alone was the best single framing, which was not the /// expectation; the head framing found the sunglasses under a cap that the /// face crop missed. The one clear-eyed face the pair loses wears a cap and /// clear glasses, at 0.68. Erring towards "sunglasses" is the safe direction /// for what this feeds: a face called sunglasses is left alone by the /// eyes-open filter, where a pair of sunglasses missed hands the eye /// classifier a lens to guess at (docs/faces.md §17). pub const SUNGLASSES_WINDOWS: [(f32, f32, f32, f32); 2] = [(0.0, 0.0, 112.0, 112.0), (-5.0, -14.0, 122.0, 122.0)]; /// The framings of one face the sunglasses classifier is shown. /// /// A newtype for the reason [`EyePatch`] is one. #[derive(Debug, Clone, PartialEq)] pub struct HeadViews { /// Each `48 × 48 × 3`, row-major RGB in `0.0..=1.0`. views: Vec>, } impl HeadViews { pub fn views(&self) -> impl Iterator { self.views.iter().map(Vec::as_slice) } } /// Cut the [`SUNGLASSES_WINDOWS`] out of the source, aligned, at the /// classifier's size. pub fn head_views( px: Pixels<'_>, width: usize, height: usize, landmarks: &[(f32, f32); 5], ) -> Option { head_views_in(px, width, height, landmarks, &SUNGLASSES_WINDOWS) } /// [`head_views`] over windows other than [`SUNGLASSES_WINDOWS`]. /// /// For measuring them, which is how the constant was chosen /// (`examples/eyes.rs --head`); production callers use the constant. pub fn head_views_in( px: Pixels<'_>, width: usize, height: usize, landmarks: &[(f32, f32); 5], windows: &[(f32, f32, f32, f32)], ) -> Option { if !px.fits(width, height) || windows.is_empty() { return None; } let m = fit_similarity(landmarks, &ARCFACE_TEMPLATE)?; let views = windows .iter() .map(|&(x, y, w, h)| { let window = TemplateWindow { x, y, w, h }; sample_window( px, width, height, &m, &window, SUNGLASSES_EDGE, SUNGLASSES_EDGE, ) }) .collect(); Some(HeadViews { views }) } /// An axis-aligned crop of the source, resampled to `out_w × out_h` RGB. /// /// `(x, y, w, h)` in source pixels; the aspect is not preserved when it /// differs from the output's. Bilinear in one step, like every crop here; /// pixels outside the source read black. What a landmark model trained on /// detector boxes wants — upright, from the frame — as against the aligned /// windows above. pub fn crop_box( px: Pixels<'_>, width: usize, height: usize, (x, y, w, h): (f32, f32, f32, f32), out_w: usize, out_h: usize, ) -> Option> { if !px.fits(width, height) || w <= 0.0 || h <= 0.0 { return None; } let identity = Similarity { a: 1.0, b: 0.0, tx: 0.0, ty: 0.0, }; let window = TemplateWindow { x, y, w, h }; Some(sample_window( px, width, height, &identity, &window, out_w, out_h, )) } fn sample_bilinear(px: Pixels<'_>, w: usize, h: usize, x: f32, y: f32, out: &mut [f32]) { let x0 = x.floor(); let y0 = y.floor(); let fx = x - x0; let fy = y - y0; let x0 = x0 as isize; let y0 = y0 as isize; for (c, o) in out.iter_mut().enumerate() { let get = |xi: isize, yi: isize| -> f32 { px.channel(w, h, xi, yi, c) }; let top = get(x0, y0) * (1.0 - fx) + get(x0 + 1, y0) * fx; let bot = get(x0, y0 + 1) * (1.0 - fx) + get(x0 + 1, y0 + 1) * fx; *o = top * (1.0 - fy) + bot * fy; } } #[cfg(test)] mod tests { use super::*; fn shifted_scaled(scale: f32, dx: f32, dy: f32, rot: f32) -> [(f32, f32); 5] { let (s, c) = (rot.sin(), rot.cos()); let mut out = [(0.0, 0.0); 5]; for (i, &(x, y)) in ARCFACE_TEMPLATE.iter().enumerate() { out[i] = (scale * (c * x - s * y) + dx, scale * (s * x + c * y) + dy); } out } #[test] fn template_onto_itself_is_the_identity() { let m = fit_similarity(&ARCFACE_TEMPLATE, &ARCFACE_TEMPLATE).unwrap(); for &(x, y) in &ARCFACE_TEMPLATE { let (u, v) = m.apply(x, y); assert!((u - x).abs() < 1e-3, "{u} vs {x}"); assert!((v - y).abs() < 1e-3, "{v} vs {y}"); } assert!((m.scale() - 1.0).abs() < 1e-4); } /// The property that matters: whatever similarity the face was seen under, /// the fit must undo it and land the landmarks back on the template. This /// is the test that fails if the transform is ever "simplified" into an /// affine or a bare scale-and-translate. #[test] fn any_similarity_of_the_template_maps_back_onto_it() { for &(scale, dx, dy, rot) in &[ (1.0_f32, 0.0_f32, 0.0_f32, 0.0_f32), (2.5, 100.0, -40.0, 0.0), (0.4, -12.0, 300.0, 0.6), (1.7, 5.0, 5.0, -1.2), ] { let observed = shifted_scaled(scale, dx, dy, rot); let m = fit_similarity(&observed, &ARCFACE_TEMPLATE).unwrap(); for (i, &(tx, ty)) in ARCFACE_TEMPLATE.iter().enumerate() { let (u, v) = m.apply(observed[i].0, observed[i].1); assert!( (u - tx).abs() < 1e-2 && (v - ty).abs() < 1e-2, "scale={scale} rot={rot}: point {i} landed at ({u}, {v}), want ({tx}, {ty})" ); } assert!( (m.scale() - 1.0 / scale).abs() < 1e-3, "scale {} should invert {scale}", m.scale() ); } } #[test] fn coincident_landmarks_are_rejected_rather_than_producing_a_crop() { let degenerate = [(50.0, 50.0); 5]; assert!(fit_similarity(°enerate, &ARCFACE_TEMPLATE).is_none()); let rgb = vec![0.5_f32; 64 * 64 * 3]; assert!(warp(&rgb, 64, 64, °enerate).is_none()); } #[test] fn source_px_reports_the_face_size_the_embedder_actually_saw() { let rgb = vec![0.5_f32; 400 * 400 * 3]; // A face twice the template's size spans 224 source pixels. let big = shifted_scaled(2.0, 80.0, 80.0, 0.0); let a = warp(&rgb, 400, 400, &big).unwrap(); assert!((a.source_px() - 224.0).abs() < 0.5, "{}", a.source_px()); // Half-size: 56 source pixels upsampled to 112, which §7 calls the // degraded bucket. let small = shifted_scaled(0.5, 10.0, 10.0, 0.0); let a = warp(&rgb, 400, 400, &small).unwrap(); assert!((a.source_px() - 56.0).abs() < 0.5, "{}", a.source_px()); } /// A white square on black, warped by a transform that should centre it: /// checks the sampler's geometry rather than the fit's algebra. #[test] fn warp_resamples_the_right_pixels() { let (w, h) = (224, 224); let mut rgb = vec![0.0_f32; w * h * 3]; for y in 0..h { for x in 0..w { if (56..168).contains(&x) && (56..168).contains(&y) { for c in 0..3 { rgb[(y * w + x) * 3 + c] = 1.0; } } } } // Landmarks placed so the fit is a pure translation of (56, 56): // the white square maps exactly onto the 112×112 output. let lm = shifted_scaled(1.0, 56.0, 56.0, 0.0); let a = warp(&rgb, w, h, &lm).unwrap(); let px = a.pixels(); for (i, v) in px.iter().enumerate() { assert!((v - 1.0).abs() < 1e-3, "pixel {i} is {v}, expected white"); } } /// A source whose red channel is its x coordinate and green its y, so a /// crop's mean colour says where in the source it was taken from. fn coordinate_image(w: usize, h: usize) -> Vec { let mut rgb = vec![0.0_f32; w * h * 3]; for y in 0..h { for x in 0..w { rgb[(y * w + x) * 3] = x as f32 / w as f32; rgb[(y * w + x) * 3 + 1] = y as f32 / h as f32; } } rgb } fn mean_channel(px: &[f32], c: usize) -> f32 { let n = px.len() / 3; px.chunks_exact(3).map(|p| p[c]).sum::() / n as f32 } /// The box is the contour's bounds, grown by the margin, and a shut /// eye's flat contour is given an open eye's height. #[test] fn an_eye_box_holds_its_contour_with_a_margin() { let open = [(100.0, 50.0), (110.0, 46.0), (120.0, 50.0), (110.0, 54.0)]; let (x, y, w, h) = eye_box(&open).unwrap(); assert!((w - 20.0 * (1.0 + 2.0 * EYE_BOX_MARGIN)).abs() < 1e-4); assert!((h - 8.0 * (1.0 + 2.0 * EYE_BOX_MARGIN)).abs() < 1e-4); assert!((x + w / 2.0 - 110.0).abs() < 1e-4); assert!((y + h / 2.0 - 50.0).abs() < 1e-4); let shut = [(100.0, 50.0), (110.0, 50.0), (120.0, 50.0)]; let (_, _, w2, h2) = eye_box(&shut).unwrap(); assert!((w2 - w).abs() < 1e-4, "same width"); assert!((h2 - 20.0 * EYE_BOX_MIN_ASPECT * (1.0 + 2.0 * EYE_BOX_MARGIN)).abs() < 1e-4); assert!(eye_box(&[]).is_none()); assert!(eye_box(&[(5.0, 5.0), (5.0, 9.0)]).is_none(), "no width"); } /// The patch is cut from the box it was given, upright, and knows how /// many source pixels it spans. #[test] fn an_eye_patch_is_the_box_resampled() { let (w, h) = (200, 200); let rgb = coordinate_image(w, h); let bbox = (60.0, 90.0, 30.0, 12.0); let eye = eye_patch(Pixels::RgbF32(&rgb), w, h, bbox).unwrap(); assert_eq!(eye.pixels().len(), EYE_PATCH_WIDTH * EYE_PATCH_HEIGHT * 3); assert_eq!(eye.source_px(), 30.0); let cx = mean_channel(eye.pixels(), 0) * w as f32; let cy = mean_channel(eye.pixels(), 1) * h as f32; assert!((cx - 75.0).abs() < 0.6, "{cx}"); assert!((cy - 96.0).abs() < 0.6, "{cy}"); // No width, or a buffer that is not the size it claims: nothing. assert!(eye_patch(Pixels::RgbF32(&rgb), w, h, (60.0, 90.0, 0.0, 12.0)).is_none()); assert!(eye_patch(Pixels::RgbF32(&rgb), 190, 200, bbox).is_none()); } /// A soft eye scores lower than the same eye sharp, on the patch itself. #[test] fn an_eye_patchs_sharpness_falls_with_blur() { let edge = 120; let sharp = image( edge, |x, y| if (x / 5 + y / 5) % 2 == 0 { 0.9 } else { 0.1 }, ); let soft = blur(&blur(&sharp, edge), edge); let bbox = (20.0, 40.0, 40.0, 24.0); let a = eye_patch(Pixels::RgbF32(&sharp), edge, edge, bbox) .unwrap() .sharpness(); let b = eye_patch(Pixels::RgbF32(&soft), edge, edge, bbox) .unwrap() .sharpness(); assert!(a > b * 2.0, "sharp {a} should clearly beat blurred {b}"); } /// The second sunglasses framing takes in more than the face — it starts /// above the template's top edge and ends below its bottom — and the /// first is the aligned face itself. #[test] fn the_head_views_are_the_face_and_a_wider_framing_of_it() { let (w, h) = (300, 300); let rgb = coordinate_image(w, h); let lm = shifted_scaled(1.0, 100.0, 100.0, 0.0); let head = head_views(Pixels::RgbF32(&rgb), w, h, &lm).unwrap(); let views: Vec<&[f32]> = head.views().collect(); let face = warp(&rgb, w, h, &lm).unwrap(); assert_eq!(views.len(), SUNGLASSES_WINDOWS.len()); for v in &views { assert_eq!(v.len(), SUNGLASSES_EDGE * SUNGLASSES_EDGE * 3); } // The face view samples the same region as the aligned crop. assert!((mean_channel(views[0], 0) - mean_channel(face.pixels(), 0)).abs() < 0.01); assert!((mean_channel(views[0], 1) - mean_channel(face.pixels(), 1)).abs() < 0.01); let (x, y, ww, hh) = SUNGLASSES_WINDOWS[1]; assert!( x < 0.0 && y < 0.0, "the window starts outside the face crop" ); assert!(x + ww > ALIGNED_EDGE as f32, "and is wider than it"); assert!(y + hh < ALIGNED_EDGE as f32, "but stops short of the chin"); // Centred horizontally on the face, so the two share a mean x. assert!((mean_channel(views[1], 0) - mean_channel(face.pixels(), 0)).abs() < 0.01); // Its first row lies above the face's first row. assert!(views[1][1] < face.pixels()[1]); } #[test] fn degenerate_landmarks_yield_no_head_crop() { let rgb = vec![0.5_f32; 64 * 64 * 3]; let degenerate = [(50.0, 50.0); 5]; assert!(head_views(Pixels::RgbF32(&rgb), 64, 64, °enerate).is_none()); // And a buffer that is not the size it claims. let lm = shifted_scaled(1.0, 0.0, 0.0, 0.0); assert!(head_views(Pixels::RgbF32(&rgb), 60, 60, &lm).is_none()); } #[test] fn out_of_bounds_samples_read_black_rather_than_wrapping() { let rgb = vec![1.0_f32; 32 * 32 * 3]; // Face far outside the image: every sample is out of bounds. let lm = shifted_scaled(1.0, 5000.0, 5000.0, 0.0); let a = warp(&rgb, 32, 32, &lm).unwrap(); assert!(a.pixels().iter().all(|&v| v == 0.0)); } // ── sharpness ───────────────────────────────────────────────────────── /// An image of `edge` square, filled by `f(x, y) -> luma`. fn image(edge: usize, f: impl Fn(usize, usize) -> f32) -> Vec { let mut v = Vec::with_capacity(edge * edge * 3); for y in 0..edge { for x in 0..edge { let l = f(x, y); v.extend_from_slice(&[l, l, l]); } } v } /// One box-blur pass, which is enough to move the metric a long way. fn blur(rgb: &[f32], edge: usize) -> Vec { let mut out = rgb.to_vec(); for y in 1..edge - 1 { for x in 1..edge - 1 { for c in 0..3 { let mut sum = 0.0; for dy in -1isize..=1 { for dx in -1isize..=1 { let i = (((y as isize + dy) as usize) * edge + ((x as isize + dx) as usize)) * 3 + c; sum += rgb[i]; } } out[(y * edge + x) * 3 + c] = sum / 9.0; } } } out } /// Landmarks placing the template into a larger image at scale 1, so the /// warp resamples one-to-one and the metric sees the source detail. fn centred(edge: usize) -> [(f32, f32); 5] { let off = (edge as f32 - ALIGNED_EDGE as f32) / 2.0; shifted_scaled(1.0, off, off, 0.0) } #[test] fn a_blurred_face_scores_lower_than_a_sharp_one() { let edge = 200; let sharp = image( edge, |x, y| if (x / 3 + y / 3) % 2 == 0 { 0.9 } else { 0.1 }, ); let soft = blur(&blur(&sharp, edge), edge); let a = warp(&sharp, edge, edge, ¢red(edge)) .unwrap() .sharpness(); let b = warp(&soft, edge, edge, ¢red(edge)).unwrap().sharpness(); assert!(a > b * 2.0, "sharp {a} should clearly beat blurred {b}"); } /// The reason for dividing by luma variance. A sharp face photographed /// against a bright sky is low-contrast, and a raw Laplacian variance would /// reject it as blurred — which would quietly throw away every backlit /// portrait in the library. #[test] fn both_pixel_layouts_warp_to_the_same_crop() { // The 8-bit path exists so a native render need not be converted to // f32 whole; it has to agree with the path it replaces to within the // quantisation it introduces. let (w, h) = (64usize, 64usize); let mut rgba = vec![0u8; w * h * 4]; let mut rgb = vec![0.0f32; w * h * 3]; for y in 0..h { for x in 0..w { let v = [ (x * 4 % 256) as u8, (y * 4 % 256) as u8, ((x + y) % 256) as u8, ]; for c in 0..3 { rgba[(y * w + x) * 4 + c] = v[c]; rgb[(y * w + x) * 3 + c] = v[c] as f32 / 255.0; } rgba[(y * w + x) * 4 + 3] = 255; } } let lm = shifted_scaled(0.35, 32.0, 32.0, 0.2); let a = warp_pixels(Pixels::RgbF32(&rgb), w, h, &lm).unwrap(); let b = warp_pixels(Pixels::Rgba8(&rgba), w, h, &lm).unwrap(); assert_eq!(a.source_px(), b.source_px()); for (x, y) in a.pixels().iter().zip(b.pixels()) { assert!((x - y).abs() < 1e-6, "{x} vs {y}"); } } #[test] fn sharpness_survives_the_contrast_being_halved() { let edge = 200; let full = image( edge, |x, y| if (x / 3 + y / 3) % 2 == 0 { 0.9 } else { 0.1 }, ); // Same detail, half the contrast, lifted so it does not clip. let flat = image( edge, |x, y| { if (x / 3 + y / 3) % 2 == 0 { 0.55 } else { 0.45 } }, ); let a = warp(&full, edge, edge, ¢red(edge)).unwrap().sharpness(); let b = warp(&flat, edge, edge, ¢red(edge)).unwrap().sharpness(); let ratio = a / b; assert!( (0.5..2.0).contains(&ratio), "contrast changed the score {ratio}x ({a} vs {b})" ); } #[test] fn a_flat_crop_has_no_sharpness() { let edge = 200; let flat = image(edge, |_, _| 0.5); assert_eq!( warp(&flat, edge, edge, ¢red(edge)).unwrap().sharpness(), 0.0 ); } /// Upsampling invents no detail, so a face that had to be stretched to /// reach the embedder scores lower than the same face at full size. That /// overlap with the size floor is deliberate and documented; this pins it /// so a future change cannot quietly remove it. #[test] fn an_upsampled_face_scores_lower_than_the_same_face_at_full_size() { let edge = 200; let src = image( edge, |x, y| if (x / 3 + y / 3) % 2 == 0 { 0.9 } else { 0.1 }, ); let full = warp(&src, edge, edge, ¢red(edge)).unwrap(); // Half scale: the crop spans 56 source pixels and is stretched to 112. let off = (edge as f32 - ALIGNED_EDGE as f32 / 2.0) / 2.0; let small = warp(&src, edge, edge, &shifted_scaled(0.5, off, off, 0.0)).unwrap(); assert!(small.source_px() < full.source_px()); assert!( small.sharpness() < full.sharpness(), "upsampled {} should be softer than full {}", small.sharpness(), full.sharpness() ); } }