//! TRACES: FR-CULL-8a //! Dense facial landmarks — InsightFace's `2d106det` (docs/dev/faces.md §17.2). //! //! SCRFD's five points place a face; they do not place an eye. Its eye //! point is loose enough that a window centred on it left the eye in a //! corner on turned and smiling heads, and two model-free ways of //! re-centring it made things worse. So a second model draws the eye's lid //! contour, and the eye box is cut from that. //! //! **Why this one.** Three were measured on the same faces — MediaPipe Face //! Mesh V2, PIPNet and this — and tied on what the eye classifier made of //! their boxes (22 of 25 open eyes read open, against 19 from the SCRFD //! point). This is the cheapest of the three by a wide margin (5 MB, 106 //! points, ~24 ms in tract), and it is under the grant the detector and //! embedder already carry rather than a new one to read. //! //! # Pre-processing //! //! Ported from InsightFace's `landmark.py`: a square crop centred on the //! detector box, 1.5× its longer edge, resized to 192; **RGB in 0..255** //! (the graph carries its own `bn_data` normalisation, so `input_mean` is //! 0 and `input_std` 1); 106 `(x, y)` in −1..1 mapped back through //! `(p + 1) · 96`. The graph's batch dimension is the literal `None` and //! is pinned to 1 by `tools/fix-face-model-shapes.sh`, like the embedder's. //! //! # The layout //! //! Checked by drawing the points on the reference faces rather than taken //! from a diagram: the subject's right eye (image-left) is points 33–42, //! the left 87–96, ten each round the lids. use ndarray::Array4; use crate::align::crop_box; use crate::{FaceError, Pixels}; use dr_inference_engine::{Form, Model, Role}; /// The graph's input edge, in pixels. pub const INPUT_EDGE: usize = 192; /// How many points the model returns. pub const POINTS: usize = 106; /// The crop's edge as a multiple of the detector box's longer edge. const CROP_SCALE: f32 = 1.5; /// The span of the frame, in long-edge units, the packed form covers: a /// quarter of the frame outside each edge. pub const PACKED_RANGE: (f32, f32) = (-0.25, 1.25); /// Bytes the packed form of one face's landmarks takes. pub const PACKED_BYTES: usize = POINTS * 4; /// Point indices of the subject's right eye's lid contour (image-left). pub const RIGHT_EYE: [usize; 10] = [33, 34, 35, 36, 37, 38, 39, 40, 41, 42]; /// Point indices of the subject's left eye's lid contour (image-right). pub const LEFT_EYE: [usize; 10] = [87, 88, 89, 90, 91, 92, 93, 94, 95, 96]; /// The 106 points of one face, in **source pixels**. #[derive(Debug, Clone, PartialEq)] pub struct Landmarks { pub points: [(f32, f32); POINTS], } impl Landmarks { /// Storage form: `106 × (x, y)` as little-endian **`u16` fixed point** /// over the frame, 424 bytes. /// /// Each coordinate is normalised by `long_edge` like the five points the /// catalog already keeps, then mapped over [`PACKED_RANGE`] — a quarter /// of the frame either side of it, because a landmark on a face at the /// edge does land outside the image — onto 0..65535. That is 0.14 source /// pixels on a 6000-pixel frame. `f16` would be the same size and worse: /// its three significant figures near 1.0 are six pixels at that scale, /// and the eye contour this is kept for is drawn to the pixel. pub fn to_packed_bytes(&self, long_edge: f32) -> Vec { let (lo, hi) = PACKED_RANGE; let pack = |v: f32| -> [u8; 2] { let t = ((v / long_edge - lo) / (hi - lo)).clamp(0.0, 1.0); ((t * 65535.0).round() as u16).to_le_bytes() }; let mut out = Vec::with_capacity(POINTS * 4); for &(x, y) in &self.points { out.extend_from_slice(&pack(x)); out.extend_from_slice(&pack(y)); } out } /// [`Self::to_packed_bytes`] read back, into source pixels of a frame /// with this `long_edge`. `None` for a blob of the wrong length. pub fn from_packed_bytes(bytes: &[u8], long_edge: f32) -> Option { if bytes.len() != POINTS * 4 { return None; } let (lo, hi) = PACKED_RANGE; let unpack = |b: &[u8]| -> f32 { let t = u16::from_le_bytes([b[0], b[1]]) as f32 / 65535.0; (t * (hi - lo) + lo) * long_edge }; let mut points = [(0.0_f32, 0.0_f32); POINTS]; for (i, p) in points.iter_mut().enumerate() { let at = i * 4; *p = (unpack(&bytes[at..at + 2]), unpack(&bytes[at + 2..at + 4])); } Some(Self { points }) } /// The lid contour of the subject's right eye. pub fn right_eye(&self) -> [(f32, f32); 10] { RIGHT_EYE.map(|i| self.points[i]) } /// The lid contour of the subject's left eye. pub fn left_eye(&self) -> [(f32, f32); 10] { LEFT_EYE.map(|i| self.points[i]) } } /// A loaded `2d106det` graph. pub struct Landmarker { session: Model, } impl Landmarker { /// The graph at `path`, or the `.a16w8.onnx` sibling beside it when the /// device's backend runs that (the Hexagon, inference.md §1.5: 0.25 px /// from f32 in the 192 crop, where int8 moved the points by 1.5). pub fn from_path(path: impl AsRef) -> Result { let (path, form) = dr_inference_engine::resolve_model(Role::Landmarks, path.as_ref()); let bytes = std::fs::read(path).map_err(FaceError::ModelRead)?; Self::from_bytes_in(&bytes, form) } pub fn from_bytes(bytes: &[u8]) -> Result { Self::from_bytes_in(bytes, Form::F32) } /// `bytes` in a stated numeric form; the output keeps its meaning. pub fn from_bytes_in(bytes: &[u8], form: Form) -> Result { let model = dr_inference_engine::open(Role::Landmarks, form, bytes)?; let acquired = model.acquire()?; let session = acquired.lock(); let input = session.inputs().first().ok_or(FaceError::WrongModel { expected: "2d106det", detail: "model has no inputs".into(), })?; let shape: Option> = input.dtype().tensor_shape().map(|s| s.to_vec()); let want = [1, 3, INPUT_EDGE as i64, INPUT_EDGE as i64]; if shape.as_deref() != Some(&want[..]) { return Err(FaceError::WrongModel { expected: "2d106det", detail: format!( "input '{}' is {:?}, expected {:?} (batch pinned to 1)", input.name(), shape, want ), }); } let out = session.outputs().first().ok_or(FaceError::WrongModel { expected: "2d106det", detail: "model has no outputs".into(), })?; let last: Option = out.dtype().tensor_shape().and_then(|d| d.last().copied()); if last != Some((POINTS * 2) as i64) { return Err(FaceError::WrongModel { expected: "2d106det", detail: format!( "output '{}' is {:?}-wide, expected {}", out.name(), last, POINTS * 2 ), }); } drop(session); drop(acquired); Ok(Self { session: model }) } /// The landmarks of the face in `bbox` — `(x0, y0, x1, y1)` in source /// pixels, the detector's box — read from the source. /// /// `None` for a box with no area or a buffer that is not the size it /// claims, as every crop here. pub fn landmarks( &mut self, px: Pixels<'_>, width: usize, height: usize, bbox: (f32, f32, f32, f32), ) -> Result, FaceError> { let (w, h) = (bbox.2 - bbox.0, bbox.3 - bbox.1); let side = w.max(h) * CROP_SCALE; let (cx, cy) = ((bbox.0 + bbox.2) / 2.0, (bbox.1 + bbox.3) / 2.0); let (x0, y0) = (cx - side / 2.0, cy - side / 2.0); let Some(crop) = crop_box( px, width, height, (x0, y0, side, side), INPUT_EDGE, INPUT_EDGE, ) else { return Ok(None); }; let e = INPUT_EDGE; let mut input = Array4::::zeros((1, 3, e, e)); for y in 0..e { for x in 0..e { for c in 0..3 { input[[0, c, y, x]] = crop[(y * e + x) * 3 + c] * 255.0; } } } let acquired = self.session.acquire()?; let mut session = acquired.lock(); let outputs = session .run(ort::inputs![ ort::value::Tensor::from_array(input).map_err(FaceError::Inference)? ]) .map_err(FaceError::Inference)?; let (_, data) = outputs[0] .try_extract_tensor::() .map_err(FaceError::Inference)?; if data.len() < POINTS * 2 { return Err(FaceError::WrongModel { expected: "2d106det", detail: format!("got {} values, expected {}", data.len(), POINTS * 2), }); } // −1..1 in the crop → crop pixels → source pixels. let scale = side / e as f32; let half = e as f32 / 2.0; let mut points = [(0.0_f32, 0.0_f32); POINTS]; for (i, p) in points.iter_mut().enumerate() { let (u, v) = ((data[2 * i] + 1.0) * half, (data[2 * i + 1] + 1.0) * half); *p = (x0 + u * scale, y0 + v * scale); } Ok(Some(Landmarks { points })) } } #[cfg(test)] mod tests { use super::*; /// Packed and unpacked, every point comes back within a fifth of a /// source pixel on a 6000-pixel frame — including one outside the /// image, which a face at the edge does produce. #[test] fn dense_landmarks_round_trip_through_their_packed_bytes() { let mut points = [(0.0_f32, 0.0_f32); POINTS]; for (i, p) in points.iter_mut().enumerate() { *p = (i as f32 * 37.3 - 200.0, 5900.0 - i as f32 * 11.1); } let lm = Landmarks { points }; let bytes = lm.to_packed_bytes(6000.0); assert_eq!(bytes.len(), PACKED_BYTES); assert_eq!(PACKED_BYTES, 424); let back = Landmarks::from_packed_bytes(&bytes, 6000.0).unwrap(); for (a, b) in lm.points.iter().zip(back.points.iter()) { assert!((a.0 - b.0).abs() < 0.2, "{} vs {}", a.0, b.0); assert!((a.1 - b.1).abs() < 0.2, "{} vs {}", a.1, b.1); } assert!(Landmarks::from_packed_bytes(&bytes[..100], 6000.0).is_none()); } }