From 72410f39c608e1da4649977d15fd7d481492fbe2 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Wed, 26 Aug 2026 19:51:02 +0200 Subject: [PATCH] Answer M1: tract loads both face graphs once their dims are pinned Neither InsightFace export parses as shipped -- SCRFD fails at its input node, ArcFace at the first Conv -- which is the same wall dr-segment hit on YOLO's dynamic export. Both load cleanly with the input dims frozen, so the pure-Rust runtime holds for the face pipeline too. tools/fix-face-model-shapes.sh does the freezing, and exists so the artefact is reproducible rather than a binary someone once produced. It takes two forms because the two graphs need different ones: ArcFace's batch is a named dim_param, SCRFD's H and W are dynamic but unnamed. Also notes YuNet loading with no intervention, which matters for the licence question in faces.md 2.3. Co-Authored-By: Claude Opus 5 (1M context) --- Cargo.lock | 13 ++++ Cargo.toml | 5 ++ core/dr-face/Cargo.toml | 43 +++++++++++++ core/dr-face/examples/probe.rs | 66 ++++++++++++++++++++ core/dr-face/src/detect.rs | 87 ++++++++++++++++++++++++++ core/dr-face/src/lib.rs | 83 +++++++++++++++++++++++++ docs/faces.md | 28 +++++++++ docs/traceability.md | 2 +- tools/fix-face-model-shapes.sh | 109 +++++++++++++++++++++++++++++++++ 9 files changed, 435 insertions(+), 1 deletion(-) create mode 100644 core/dr-face/Cargo.toml create mode 100644 core/dr-face/examples/probe.rs create mode 100644 core/dr-face/src/detect.rs create mode 100644 core/dr-face/src/lib.rs create mode 100755 tools/fix-face-model-shapes.sh diff --git a/Cargo.lock b/Cargo.lock index 0b13f47..5ddd6ca 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1447,6 +1447,19 @@ dependencies = [ "zune-jpeg 0.4.21", ] +[[package]] +name = "dr-face" +version = "0.7.0" +dependencies = [ + "env_logger", + "log", + "ndarray", + "ort", + "ort-tract", + "thiserror 2.0.20", + "zune-jpeg 0.4.21", +] + [[package]] name = "dr-film" version = "0.7.0" diff --git a/Cargo.toml b/Cargo.toml index e15768c..fcd0b34 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -6,6 +6,7 @@ members = [ "core/dr-thumbs", "core/dr-decode", "core/dr-export", + "core/dr-face", "core/dr-film", "core/dr-ingest", "core/dr-gpu", @@ -35,6 +36,10 @@ dr-catalog = { path = "core/dr-catalog" } dr-thumbs = { path = "core/dr-thumbs" } dr-decode = { path = "core/dr-decode" } dr-export = { path = "core/dr-export" } +# Stated explicitly for the same reason as `dr-segment` below: no dependant +# should drag in an ONNX runtime by accident. Members opt in with +# `features = ["inference"]`. +dr-face = { path = "core/dr-face", default-features = false } dr-film = { path = "core/dr-film" } dr-ingest = { path = "core/dr-ingest" } dr-gpu = { path = "core/dr-gpu" } diff --git a/core/dr-face/Cargo.toml b/core/dr-face/Cargo.toml new file mode 100644 index 0000000..b1d54c0 --- /dev/null +++ b/core/dr-face/Cargo.toml @@ -0,0 +1,43 @@ +[package] +name = "dr-face" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true + +[dependencies] +thiserror.workspace = true +log.workspace = true + +# Inference. `ort` is the API; **tract is the engine** — see the workspace +# manifest, and docs/faces.md §3, for why the C++ ONNX Runtime is not linked. +ort = { workspace = true, optional = true } +ort-tract = { workspace = true, optional = true } +ndarray = { workspace = true, optional = true } + +[dev-dependencies] +zune-jpeg.workspace = true +env_logger.workspace = true +# The M1 probe drives `ort` directly so it can print the raw load error. +ort = { workspace = true } +ort-tract = { workspace = true } + +[[example]] +name = "probe" +required-features = ["inference"] + +[features] +# Nothing on by default, and in particular **no `embedded-model`**: the weights +# are not a build input and never become one (docs/faces.md §2.2). A feature +# flag that *could* embed them is a flag someone eventually sets in a packaging +# script, and the InsightFace grant does not survive that. +default = [] + +# The ONNX runtime, and the two stages that need it. +# +# Separable because the accuracy of this subsystem lives in `calibrate` and +# `cluster`, which are arithmetic over embeddings with no model in them. They +# must be testable against synthetic embeddings on a machine with no weights on +# it — a test suite that needs a research-licensed download is a test suite +# that does not run in CI. +inference = ["dep:ort", "dep:ort-tract", "dep:ndarray"] diff --git a/core/dr-face/examples/probe.rs b/core/dr-face/examples/probe.rs new file mode 100644 index 0000000..29f82bf --- /dev/null +++ b/core/dr-face/examples/probe.rs @@ -0,0 +1,66 @@ +//! M1 (docs/faces.md §12) — will tract load these graphs at all? +//! +//! The one measurement everything else in the face subsystem is conditional +//! on. `det_500m.onnx` has a dynamic H/W input, which is exactly what tract +//! failed on for YOLO26n-seg, so a plain "no" here is the expected outcome and +//! the interesting part is the error it gives. +//! +//! cargo run -p dr-face --features inference --example probe -- MODEL... + +fn main() { + env_logger::init(); + + let paths: Vec = std::env::args().skip(1).collect(); + if paths.is_empty() { + eprintln!("usage: probe MODEL.onnx [MODEL.onnx ...]"); + std::process::exit(2); + } + + let mut failures = 0; + for path in &paths { + println!("\n=== {path} ==="); + let bytes = match std::fs::read(path) { + Ok(b) => b, + Err(e) => { + println!(" UNREADABLE: {e}"); + failures += 1; + continue; + } + }; + println!(" {} bytes", bytes.len()); + + dr_face::install_backend_for_probe(); + + let session = ort::session::Session::builder() + .and_then(|mut b| b.commit_from_memory(&bytes)); + + match session { + Err(e) => { + println!(" LOAD FAILED: {e}"); + failures += 1; + } + Ok(s) => { + println!(" LOADED"); + for i in s.inputs() { + println!( + " in {:<24} {:?}", + i.name(), + i.dtype().tensor_shape() + ); + } + for o in s.outputs() { + println!( + " out {:<24} {:?}", + o.name(), + o.dtype().tensor_shape() + ); + } + } + } + } + + println!("\n{} of {} failed", failures, paths.len()); + if failures > 0 { + std::process::exit(1); + } +} diff --git a/core/dr-face/src/detect.rs b/core/dr-face/src/detect.rs new file mode 100644 index 0000000..ae9f447 --- /dev/null +++ b/core/dr-face/src/detect.rs @@ -0,0 +1,87 @@ +//! SCRFD face detection (docs/faces.md §4). +//! +//! For now: enough of the loader to answer M1 — whether tract will parse these +//! graphs at all — plus the load-time shape validation that keeps a YuNet file +//! from being decoded as an SCRFD one. + +use crate::{install_backend, FaceError}; + +/// A loaded SCRFD graph. +pub struct Detector { + session: ort::session::Session, + /// Feature-map count: 3 for strides {8,16,32}, 4 for {8,16,32,64}. + /// + /// Discovered from the output count rather than assumed, because both + /// exports exist and hardcoding 3 silently ignores the largest faces in a + /// four-stride model. + strides: usize, +} + +/// Strides, in the order SCRFD emits them. +pub const ALL_STRIDES: [usize; 4] = [8, 16, 32, 64]; + +/// Anchors per feature-map location. +pub const ANCHORS: usize = 2; + +impl Detector { + pub fn from_path(path: impl AsRef) -> Result { + let bytes = std::fs::read(path).map_err(FaceError::ModelRead)?; + Self::from_bytes(&bytes) + } + + pub fn from_bytes(bytes: &[u8]) -> Result { + install_backend(); + + let session = ort::session::Session::builder() + .map_err(FaceError::Inference)? + .commit_from_memory(bytes) + .map_err(FaceError::Inference)?; + + let n_out = session.outputs().len(); + if n_out % 3 != 0 || !(9..=12).contains(&n_out) { + return Err(FaceError::WrongModel { + expected: "InsightFace SCRFD", + detail: format!("expected 9 or 12 outputs, got {n_out}"), + }); + } + let strides = n_out / 3; + + // The check that actually distinguishes the models: score, box and + // landmark groups end in 1, 4 and 10 respectively. YuNet also has + // twelve outputs, so the count alone proves nothing. + for (group, expected_last) in [1_i64, 4, 10].into_iter().enumerate() { + for s in 0..strides { + let idx = group * strides + s; + let out = &session.outputs()[idx]; + let last: Option = + out.dtype().tensor_shape().and_then(|d| d.last().copied()); + if last != Some(expected_last) { + return Err(FaceError::WrongModel { + expected: "InsightFace SCRFD", + detail: format!( + "output '{}' last dim is {:?}, expected {expected_last}", + out.name(), last + ), + }); + } + } + } + + Ok(Self { session, strides }) + } + + /// Number of stride levels this graph emits. + pub fn strides(&self) -> &'static [usize] { + &ALL_STRIDES[..self.strides] + } + + /// The graph's declared input shape, for diagnosing a dynamic export. + pub fn input_shape(&self) -> Option> { + self.session + .inputs() + .first()? + .dtype() + .tensor_shape() + .map(|s| s.to_vec()) + } +} diff --git a/core/dr-face/src/lib.rs b/core/dr-face/src/lib.rs new file mode 100644 index 0000000..ba8e6bf --- /dev/null +++ b/core/dr-face/src/lib.rs @@ -0,0 +1,83 @@ +//! Faces and identity (S14, docs/faces.md). +//! +//! Two models, run over the proxy tier, producing per face a box, five +//! landmarks, a confidence and a 512-d embedding (FR-CULL-8) — and then the +//! arithmetic that turns embeddings into people (FR-CULL-9, FR-CULL-10). +//! +//! Like `dr-segment`, this crate is **device-free**: no GPU adapter, no +//! Slint, nothing that needs a display. Unlike `dr-segment`, it carries **no +//! weights at all**, and the absence is deliberate — see [`the licence +//! note`](#the-weights-are-not-in-this-repository) below. +//! +//! # The weights are not in this repository +//! +//! The models this crate is built for — SCRFD-500MF and ArcFace/MobileFaceNet +//! — are InsightFace's, and their pretrained weights carry a **non-commercial +//! research-only** grant. That is incompatible with GPL-3.0-or-later and with +//! every channel DarkRoom ships through, so the weights cannot be committed +//! here the way `dr-segment`'s can, and there is no `embedded-model` feature +//! for a packaging script to switch on. The application obtains a model at +//! runtime; this crate takes bytes and never fetches anything. +//! +//! docs/faces.md §2 is the full reading, including what would have to change +//! for that to stop being true. +//! +//! # Why the runtime is split behind a feature +//! +//! [`calibrate`] and [`cluster`] are where this subsystem's accuracy actually +//! lives, and both are pure arithmetic over embeddings with no model in them. +//! They build and test without `inference`, on synthetic embeddings, on a +//! machine with no weights on it — which is what lets CI cover the part most +//! likely to be subtly wrong. + +#[cfg(feature = "inference")] +pub mod detect; + +/// What can go wrong between an image and a face. +#[derive(Debug, thiserror::Error)] +pub enum FaceError { + #[error("could not read model file: {0}")] + ModelRead(#[source] std::io::Error), + + #[cfg(feature = "inference")] + #[error("inference failed: {0}")] + Inference(#[source] ort::Error), + + /// The graph is not the one this decoder was written for. + /// + /// Worth a distinct variant rather than a generic failure: the models in + /// this space have interchangeable *shapes* and incompatible *layouts* + /// (a YuNet export also has twelve outputs), so the failure this catches + /// is not a crash but a page of plausible numbers. + #[error("model does not look like {expected}: {detail}")] + WrongModel { + expected: &'static str, + detail: String, + }, + + #[error("image buffer is {got} floats, expected {expected} (RGB, three per pixel)")] + ImageShape { expected: usize, got: usize }, +} + +/// Install tract as `ort`'s backend. +/// +/// Idempotent, and it must happen before any other `ort` call: with +/// `alternative-backend` there is no linked runtime to fall back on, so an +/// un-set API is a panic rather than a slow path. Same helper as +/// `dr-segment::semantic`, for the same reason. +#[cfg(feature = "inference")] +pub(crate) fn install_backend() { + use std::sync::Once; + static ONCE: Once = Once::new(); + ONCE.call_once(|| { + let _ = ort::set_api(ort_tract::api()); + }); +} + +/// [`install_backend`] for the M1 probe example, which drives `ort` directly +/// rather than through [`detect::Detector`] so it can report the raw error. +#[cfg(feature = "inference")] +#[doc(hidden)] +pub fn install_backend_for_probe() { + install_backend(); +} diff --git a/docs/faces.md b/docs/faces.md index a5988eb..da1dc50 100644 --- a/docs/faces.md +++ b/docs/faces.md @@ -678,6 +678,34 @@ argument: | **M9** | YuNet as a drop-in detector: M4 and M6, re-run | §2.3's question 3, and the model file is already on disk. If the answer is "close enough", half the licence problem disappears. Needs its own decode path — its outputs are not SCRFD's. | | **M10** | Peak RSS during an indexing sweep | NFR-RES-2. Two loaded graphs plus a proxy plus a batch of crops, on a phone. | +### 12.1 M1 result — **PASS, conditionally** · 2026-08-26 + +Measured, not extrapolated. Both InsightFace graphs **fail to load in tract as shipped**, exactly as +§4.1 predicted and for the reason it gave: + +``` +scrfd_500m_bnkps.onnx Translating node #0 "input.1" Source ToTypedTranslator +arcface_w600k_mbf.onnx Failed analyse for node #139 "Conv_0" ConvHir +``` + +Both **load cleanly once their input dimensions are pinned** — SCRFD's unnamed H/W to 640, ArcFace's +`None` batch to 1 — by `tools/fix-face-model-shapes.sh`, which rewrites the declared dims and touches +no weights. The frozen SCRFD reports the layout §4.2 specifies, which is the second half of the +answer: nine outputs, three strides, last dims 1/4/10, and `12800 = 80 × 80 × 2` confirming two +anchors per location at stride 8. + +Two things worth carrying forward: + +**SCRFD's outputs were already static.** The export was made at 640 and only its input forgot to say +so, so pinning to 640 is not a choice this project is making — it is the shape the graph was always +going to run at. §12's "320 as a faster option" would need a different export, not a different flag. + +**YuNet loads with no intervention at all**, at a fixed `[1, 3, 640, 640]`, with twelve outputs in +three strides — `cls`/`obj`/`bbox`/`kps`, which is a *different layout* from SCRFD's and confirms why +§4.2's load-time check has to look at shapes rather than count outputs. Combined with its permissive +licence (§2.3) that makes M9 more interesting than it looked: the permissive detector is also the one +with no shape-fixing step in front of it. + --- ## 13. Order diff --git a/docs/traceability.md b/docs/traceability.md index 5b25898..f28b982 100644 --- a/docs/traceability.md +++ b/docs/traceability.md @@ -9,7 +9,7 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n | Metric | Value | |---|---| -| Source files scanned | 227 | +| Source files scanned | 230 | | TRACES tags found | 621 | | Requirements defined | 177 | | Requirements covered | 91 | diff --git a/tools/fix-face-model-shapes.sh b/tools/fix-face-model-shapes.sh new file mode 100755 index 0000000..49acf9d --- /dev/null +++ b/tools/fix-face-model-shapes.sh @@ -0,0 +1,109 @@ +#!/usr/bin/env bash +# Freeze the input dimensions of the face models so tract can parse them. +# +# ./tools/fix-face-model-shapes.sh IN.onnx OUT.onnx --input NAME=1,3,640,640 +# ./tools/fix-face-model-shapes.sh IN.onnx OUT.onnx --dim NAME=1 +# +# The two the face pipeline needs, verified 2026-08-26 (docs/faces.md §12 M1): +# +# ... det_500m.onnx scrfd_500m_640.onnx --input input.1=1,3,640,640 +# ... w600k_mbf.onnx arcface_mbf_b1.onnx --dim None=1 +# +# ## Why this exists +# +# The InsightFace exports declare dynamic input dimensions — SCRFD's H and W, +# ArcFace's batch N. **tract cannot parse either graph in that form**, failing +# at the input node and at the first Conv respectively: +# +# scrfd_500m_bnkps.onnx Translating node #0 "input.1" Source ToTypedTranslator +# arcface_w600k_mbf.onnx Failed analyse for node #139 "Conv_0" ConvHir +# +# Both load cleanly once the dims are pinned. This is the same wall dr-segment +# hit, which is why `tools/export-seg-model.sh` passes `dynamic=False`; here we +# cannot re-export from PyTorch, because the weights are InsightFace's and the +# training code is not in the loop, so the dims are rewritten in the ONNX file +# instead. +# +# `make_dynamic_shape_fixed` only edits the declared dimension; it does not +# retrain, requantise, or change a single weight. The output is numerically the +# same graph with one shape pinned. SCRFD's *outputs* were already static — the +# export was made at 640 and only its input forgot to say so — which is why 640 +# is not a free choice here. +# +# ## Why it is a script and not a build step +# +# Same reason as the segmentation export: the model is not a build input +# (docs/faces.md §2.2 — the weights are never committed, because InsightFace's +# grant is non-commercial). This runs once, wherever the user's model lives, +# and the app loads the result. It exists so the transformation is reproducible +# rather than a binary someone once produced and nobody can regenerate. +# +# Requires `uv`. Everything else is fetched into a throwaway venv, in /var/tmp +# rather than /tmp — /tmp here is a tmpfs, and onnxruntime is not small. +set -euo pipefail + +if [ "$#" -lt 4 ]; then + sed -n '2,10p' "$0" >&2 + exit 2 +fi + +IN="$1"; shift +OUT="$1"; shift + +[ -f "$IN" ] || { echo "no such model: $IN" >&2; exit 1; } + +WORK="$(mktemp -d -p /var/tmp fix-face-shapes.XXXXXX)" +trap 'rm -rf "${WORK}"' EXIT + +echo "==> venv in ${WORK}" +uv venv --python 3.12 "${WORK}/venv" >/dev/null +VIRTUAL_ENV="${WORK}/venv" uv pip install --quiet onnx onnxruntime + +# Two forms, because the two models need different ones — and which one a graph +# needs is not a matter of taste: +# +# --dim NAME=VALUE for a *named* symbolic dimension. +# --input NAME=D,D,D,D for a dimension that is dynamic but unnamed. +# +# ArcFace declares its batch as the literal dim_param "None", so `--dim` binds +# it. SCRFD's H and W carry no dim_param at all, so there is no name to bind +# and the whole input shape has to be restated. Reaching for `--dim` first and +# getting a silent no-op is the half-hour worth skipping. +CUR="$IN" +STEP=0 +while [ "$#" -gt 0 ]; do + FLAG="$1"; shift + PAIR="${1:-}"; shift || true + NAME="${PAIR%%=*}" + VAL="${PAIR#*=}" + STEP=$((STEP + 1)) + NEXT="${WORK}/step${STEP}.onnx" + case "$FLAG" in + --dim) + echo "==> dim_param ${NAME} := ${VAL}" + "${WORK}/venv/bin/python" -m onnxruntime.tools.make_dynamic_shape_fixed \ + --dim_param "${NAME}" --dim_value "${VAL}" "${CUR}" "${NEXT}" + ;; + --input) + echo "==> input ${NAME} := ${VAL}" + "${WORK}/venv/bin/python" -m onnxruntime.tools.make_dynamic_shape_fixed \ + --input_name "${NAME}" --input_shape "${VAL}" "${CUR}" "${NEXT}" + ;; + *) + echo "unknown flag ${FLAG} (want --dim or --input)" >&2 + exit 2 + ;; + esac + CUR="${NEXT}" +done + +cp "${CUR}" "${OUT}" +echo "==> wrote ${OUT}" +"${WORK}/venv/bin/python" - "$OUT" <<'PY' +import sys, onnx +m = onnx.load(sys.argv[1]) +for vi in list(m.graph.input) + list(m.graph.output): + dims = [d.dim_value if d.HasField("dim_value") else (d.dim_param or "?") + for d in vi.type.tensor_type.shape.dim] + print(f" {vi.name:<24} {dims}") +PY