From 05508741af99d6790636361a8e01a9c732be9421 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 19 Sep 2026 15:23:31 +0200 Subject: [PATCH] Start the inference engine from both apps and show its choice in Settings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The desktop names where a package may have put libonnxruntime — an override variable, beside the executable, the package's own library directory, the Flatpak prefix, the system library directory — and Android points at the APK's native library directory, which is also what Qualcomm's DSP loader must be told for the Hexagon skel. Android starts the engine at the end of the model unpack rather than at launch, because the probe fingerprints the model files and a first launch has none until then. The About panel gains an Inference row beside Graphics, re-read every two seconds while the probe runs and engines land, and faces.model_id carries the detector's form: an int8 detector finds a different set of faces and is a different population (docs/inference.md §7). A low-memory signal drops every idle session with the GPU caches. The APK assembly bundles ONNX Runtime and the Qualcomm HTP libraries from Maven, fetched by tools/fetch-android-runtime.sh with their published checksums; RUNTIME_DIR=none builds the tract-only APK, which is a slower app and not a broken one. The desktop packages carry no runtime yet. Two probe fixes from the first desktop run: the floor must not be built with CPU fallback disabled, and a versioned libonnxruntime.so is a runtime too. On the reference desktop the probe now loads ONNX Runtime 1.30, measures 30 ms on the CPU provider, and selects TensorRT at 1.5 ms. --- Cargo.lock | 1 + apps/darkroom-android/src/lib.rs | 21 ++++++ apps/darkroom-desktop/src/main.rs | 36 ++++++++++ core/dr-face/src/classify.rs | 33 +++++---- core/dr-face/src/landmarks.rs | 22 +++--- core/dr-inference-engine/src/api.rs | 41 +++++++++-- core/dr-inference-engine/src/engines.rs | 41 +++++++---- core/dr-inference-engine/src/lib.rs | 6 ++ core/dr-inference-engine/src/probe.rs | 6 ++ core/dr-inference-engine/src/session.rs | 2 +- core/dr-segment/src/lib.rs | 2 + core/dr-segment/src/semantic.rs | 7 ++ core/dr-types/src/settings.rs | 14 ++++ docker/android/assemble-apk.sh | 26 ++++++- tools/fetch-android-runtime.sh | 96 +++++++++++++++++++++++++ ui/dr-ui/Cargo.toml | 3 + ui/dr-ui/src/identity_ui.rs | 2 +- ui/dr-ui/src/inference.rs | 91 +++++++++++++++++++++++ ui/dr-ui/src/lib.rs | 33 ++++++++- ui/dr-ui/src/library.rs | 7 ++ ui/dr-ui/src/library_ui.rs | 4 +- ui/dr-ui/ui/app.slint | 4 ++ ui/dr-ui/ui/settings.slint | 24 +++++++ 23 files changed, 465 insertions(+), 57 deletions(-) create mode 100755 tools/fetch-android-runtime.sh create mode 100644 ui/dr-ui/src/inference.rs diff --git a/Cargo.lock b/Cargo.lock index 67edfcc..be2dcaa 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1688,6 +1688,7 @@ dependencies = [ "dr-face", "dr-film", "dr-gpu", + "dr-inference-engine", "dr-ingest", "dr-lens", "dr-pano", diff --git a/apps/darkroom-android/src/lib.rs b/apps/darkroom-android/src/lib.rs index dd62545..e10e57d 100644 --- a/apps/darkroom-android/src/lib.rs +++ b/apps/darkroom-android/src/lib.rs @@ -400,6 +400,27 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) { "bundled models ready: {copied} bytes copied in {} ms", started.elapsed().as_millis() ); + + // Now, and not at launch: the probe fingerprints the model files, and + // on a first launch they were not on disk until this line. The runtime + // is in the APK's native library directory beside `libdarkroom.so`, + // which is also where Qualcomm's DSP loader has to be pointed for the + // Hexagon skel (docs/inference.md §3, §8). + dr_ui::inference::init(native_library_dir().into_iter().collect()); +} + +/// The directory the system unpacked this APK's native libraries into. +/// +/// Read from where the loader put *this* library rather than asked of the +/// activity: `android-activity` does not expose `nativeLibraryDir`, and the +/// answer is in `/proc/self/maps` for free. +#[cfg(target_os = "android")] +fn native_library_dir() -> Option { + let maps = std::fs::read_to_string("/proc/self/maps").ok()?; + maps.lines() + .filter_map(|l| l.split_whitespace().nth(5)) + .find(|p| p.ends_with("/libdarkroom.so")) + .and_then(|p| std::path::Path::new(p).parent().map(Into::into)) } /// TRACES: FR-PLAT-AND-6 diff --git a/apps/darkroom-desktop/src/main.rs b/apps/darkroom-desktop/src/main.rs index 07d6655..1b9933b 100644 --- a/apps/darkroom-desktop/src/main.rs +++ b/apps/darkroom-desktop/src/main.rs @@ -61,6 +61,11 @@ fn main() -> anyhow::Result<()> { eprintln!("usage: darkroom-desktop ..."); } + // Before the window: the probe runs on its own thread and the first + // frame does not wait for it, but the models a background job asks for + // should already know where the runtime is (docs/inference.md §4). + dr_ui::inference::init(runtime_dirs()); + dr_ui::run(paths)?; // Skip Rust's normal static/thread-local teardown on the way out: a @@ -70,3 +75,34 @@ fn main() -> anyhow::Result<()> { // destruction" when the window is closed. std::process::exit(0); } + +/// Where a desktop package may have put `libonnxruntime`, most specific +/// first. None of these existing is the tract build, which is a complete +/// application and not an error (docs/inference.md §3). +/// +/// `DARKROOM_ORT_DIR` is for a developer pointing at a runtime that is not +/// installed — the wheel's `capi` directory, say. Then beside the executable +/// and in the package's private library directory, for a package that +/// bundles its own; then the Flatpak prefix; then the system library +/// directory, for a distribution that ships ONNX Runtime as a package of its +/// own. A system copy whose GPU providers do not load is not a problem: the +/// probe builds a real session before believing a provider. +fn runtime_dirs() -> Vec { + let mut dirs = Vec::new(); + if let Some(dir) = std::env::var_os("DARKROOM_ORT_DIR") { + dirs.push(PathBuf::from(dir)); + } + if let Ok(exe) = std::env::current_exe() { + if let Some(bin) = exe.parent() { + dirs.push(bin.to_path_buf()); + dirs.push(bin.join("../lib/darkroom")); + } + } + #[cfg(target_os = "linux")] + dirs.extend([ + PathBuf::from("/app/lib/darkroom"), + PathBuf::from("/usr/lib/darkroom"), + PathBuf::from("/usr/lib"), + ]); + dirs +} diff --git a/core/dr-face/src/classify.rs b/core/dr-face/src/classify.rs index b21f4ec..de4234c 100644 --- a/core/dr-face/src/classify.rs +++ b/core/dr-face/src/classify.rs @@ -40,16 +40,17 @@ use crate::align::{ }; use crate::eyes::{Eye, EyeReading}; use crate::landmarks::{Landmarker, Landmarks}; -use crate::{install_backend, FaceError, Pixels}; +use crate::{FaceError, Pixels}; +use dr_inference_engine::{Form, Model, Role}; /// A loaded OCEC graph. pub struct EyeClassifier { - session: ort::session::Session, + session: Model, } /// A loaded SGC graph. pub struct SunglassesClassifier { - session: ort::session::Session, + session: Model, } /// Open a single-input, single-output classifier and check it is the shape @@ -63,12 +64,10 @@ fn open_classifier( bytes: &[u8], expected: &'static str, (h, w): (usize, usize), -) -> Result { - install_backend(); - let session = ort::session::Session::builder() - .map_err(FaceError::Inference)? - .commit_from_memory(bytes) - .map_err(FaceError::Inference)?; +) -> Result { + let model = dr_inference_engine::open(Role::EyeClassifier, Form::F32, bytes)?; + let acquired = model.acquire()?; + let session = acquired.lock(); let input = session.inputs().first().ok_or(FaceError::WrongModel { expected, @@ -93,7 +92,9 @@ fn open_classifier( detail: format!("{} outputs, expected one", session.outputs().len()), }); } - Ok(session) + drop(session); + drop(acquired); + Ok(model) } /// Lay a `h × w` RGB crop out as the `[1, 3, h, w]` tensor both graphs take. @@ -110,11 +111,9 @@ fn to_nchw(pixels: &[f32], h: usize, w: usize) -> Array4 { } /// Run a one-number classifier and read its sigmoid back, clamped. -fn run_scalar( - session: &mut ort::session::Session, - input: Array4, - expected: &'static str, -) -> Result { +fn run_scalar(model: &Model, input: Array4, expected: &'static str) -> Result { + let acquired = model.acquire()?; + let mut session = acquired.lock(); let outputs = session .run(ort::inputs![ ort::value::Tensor::from_array(input).map_err(FaceError::Inference)? @@ -149,7 +148,7 @@ impl EyeClassifier { /// P(open) for one eye. pub fn classify(&mut self, eye: &EyePatch) -> Result { let input = to_nchw(eye.pixels(), EYE_PATCH_HEIGHT, EYE_PATCH_WIDTH); - run_scalar(&mut self.session, input, "OCEC") + run_scalar(&self.session, input, "OCEC") } } @@ -170,7 +169,7 @@ impl SunglassesClassifier { let mut best = 0.0_f32; for view in head.views() { let input = to_nchw(view, SUNGLASSES_EDGE, SUNGLASSES_EDGE); - best = best.max(run_scalar(&mut self.session, input, "SGC")?); + best = best.max(run_scalar(&self.session, input, "SGC")?); } Ok(best) } diff --git a/core/dr-face/src/landmarks.rs b/core/dr-face/src/landmarks.rs index 2321a07..fdd857e 100644 --- a/core/dr-face/src/landmarks.rs +++ b/core/dr-face/src/landmarks.rs @@ -32,7 +32,8 @@ use ndarray::Array4; use crate::align::crop_box; -use crate::{install_backend, FaceError, Pixels}; +use crate::{FaceError, Pixels}; +use dr_inference_engine::{Form, Model, Role}; /// The graph's input edge, in pixels. pub const INPUT_EDGE: usize = 192; @@ -118,7 +119,7 @@ impl Landmarks { /// A loaded `2d106det` graph. pub struct Landmarker { - session: ort::session::Session, + session: Model, } impl Landmarker { @@ -128,11 +129,9 @@ impl Landmarker { } pub fn from_bytes(bytes: &[u8]) -> Result { - install_backend(); - let session = ort::session::Session::builder() - .map_err(FaceError::Inference)? - .commit_from_memory(bytes) - .map_err(FaceError::Inference)?; + let model = dr_inference_engine::open(Role::Landmarks, Form::F32, bytes)?; + let acquired = model.acquire()?; + let session = acquired.lock(); let input = session.inputs().first().ok_or(FaceError::WrongModel { expected: "2d106det", @@ -167,7 +166,9 @@ impl Landmarker { ), }); } - Ok(Self { session }) + drop(session); + drop(acquired); + Ok(Self { session: model }) } /// The landmarks of the face in `bbox` — `(x0, y0, x1, y1)` in source @@ -206,8 +207,9 @@ impl Landmarker { } } } - let outputs = self - .session + let acquired = self.session.acquire()?; + let mut session = acquired.lock(); + let outputs = session .run(ort::inputs![ ort::value::Tensor::from_array(input).map_err(FaceError::Inference)? ]) diff --git a/core/dr-inference-engine/src/api.rs b/core/dr-inference-engine/src/api.rs index d18c527..e8a895f 100644 --- a/core/dr-inference-engine/src/api.rs +++ b/core/dr-inference-engine/src/api.rs @@ -38,10 +38,16 @@ pub fn runtime() -> Runtime { } /// Install a table if none is installed yet — tract, since no directories -/// were named. What a test or an example gets. +/// were named. What a test or an example gets, unless `DARKROOM_ORT_DIR` +/// names a runtime: the same variable the desktop honours, so an example +/// can be pointed at the runtime the app uses without learning `init`. pub fn ensure_installed() { if RUNTIME.get().is_none() { - install(&[]); + let dirs: Vec = std::env::var_os("DARKROOM_ORT_DIR") + .map(PathBuf::from) + .into_iter() + .collect(); + install(&dirs); } } @@ -93,11 +99,7 @@ fn load_native(dir: &std::path::Path) -> Result { let path = if dir.as_os_str().is_empty() { PathBuf::from(name) } else { - let p = dir.join(name); - if !p.is_file() { - return Err("not present".into()); - } - p + find_library(dir, name).ok_or("not present")? }; // SAFETY: the library's initialisers are ONNX Runtime's own; the symbol @@ -140,3 +142,28 @@ fn load_native(dir: &std::path::Path) -> Result { Ok(Runtime::OnnxRuntime { path, version }) } } + +/// `libonnxruntime.so` in `dir`, or a versioned spelling of it — +/// `libonnxruntime.so.1.30.0` is what the Python wheel ships, and a package +/// that installs only the versioned file is not wrong. +#[cfg(feature = "native")] +fn find_library(dir: &std::path::Path, name: &str) -> Option { + let exact = dir.join(name); + if exact.is_file() { + return Some(exact); + } + let prefix = format!("{name}."); + let mut versioned: Vec = std::fs::read_dir(dir) + .ok()? + .filter_map(|e| e.ok()) + .map(|e| e.path()) + .filter(|p| { + p.is_file() + && p.file_name() + .and_then(|n| n.to_str()) + .is_some_and(|n| n.starts_with(&prefix)) + }) + .collect(); + versioned.sort(); + versioned.pop() +} diff --git a/core/dr-inference-engine/src/engines.rs b/core/dr-inference-engine/src/engines.rs index d565758..12df295 100644 --- a/core/dr-inference-engine/src/engines.rs +++ b/core/dr-inference-engine/src/engines.rs @@ -10,6 +10,11 @@ use std::path::PathBuf; use crate::{state, Config, Form, Rung}; +enum Source { + File(PathBuf), + Bytes(&'static [u8]), +} + /// 64-bit FNV-1a. A cache key, not a checksum: two model files that collide /// here would have to also be the same size and the same role, and the cost /// of that is a rebuilt engine. @@ -47,34 +52,42 @@ pub fn run() { // Smallest first, so the detector — the one that runs per image — is // ready soonest (§6 step 3). - let mut jobs: Vec<(crate::Role, PathBuf, u64)> = cfg + let mut jobs: Vec<(crate::Role, Source, u64)> = cfg .models .iter() - .filter(|(role, _)| rung.form(*role) != Form::F32 || rung != Rung::Hexagon) .filter_map(|(role, path)| { let (path, form) = crate::resolve_model(*role, path); (form == rung.form(*role)).then(|| { let size = std::fs::metadata(&path).map(|m| m.len()).unwrap_or(0); - (*role, path, size) + (*role, Source::File(path), size) }) }) + .chain(cfg.embedded.iter().filter_map(|(role, bytes)| { + // An embedded model has no int8 sibling to offer a rung that + // wants one; it runs on that rung's fallback. + (rung.form(*role) == Form::F32).then_some(( + *role, + Source::Bytes(bytes), + bytes.len() as u64, + )) + })) .collect(); jobs.sort_by_key(|j| j.2); state().lock().unwrap().wanted = jobs.len(); - for (role, path, _) in jobs { - let Ok(bytes) = std::fs::read(&path) else { - continue; + for (role, source, _) in jobs { + let (bytes, name) = match &source { + Source::File(path) => match std::fs::read(path) { + Ok(b) => (b, path.display().to_string()), + Err(_) => continue, + }, + Source::Bytes(b) => (b.to_vec(), format!("embedded {role:?}")), }; let key = key(rung, &bytes); if state().lock().unwrap().cache.compiled.contains(&key) { continue; } - log::info!( - "inference: compiling {} for {}", - path.display(), - rung.label() - ); + log::info!("inference: compiling {name} for {}", rung.label()); let started = std::time::Instant::now(); match crate::session::build(rung, role, &bytes, &cfg, false) { Ok(session) => { @@ -83,8 +96,7 @@ pub fn run() { s.cache.compiled.insert(key); crate::probe::write_cache(&s.config, &s.cache); log::info!( - "inference: {} ready on {} in {:.1} s", - path.display(), + "inference: {name} ready on {} in {:.1} s", rung.label(), started.elapsed().as_secs_f64() ); @@ -94,8 +106,7 @@ pub fn run() { // their engine. A corrected model file changes the hash and // is retried. log::warn!( - "inference: {} will not compile for {}: {e}", - path.display(), + "inference: {name} will not compile for {}: {e}", rung.label() ); } diff --git a/core/dr-inference-engine/src/lib.rs b/core/dr-inference-engine/src/lib.rs index 2eac684..0cb706e 100644 --- a/core/dr-inference-engine/src/lib.rs +++ b/core/dr-inference-engine/src/lib.rs @@ -35,6 +35,10 @@ pub enum Role { Embedder, Segmenter, Scene, + /// The dense landmark model behind the eye reading (docs/faces.md §7c). + Landmarks, + /// The eye-state and sunglasses classifiers, a few hundred kilobytes. + EyeClassifier, } /// Which numeric form of a model a session was built from. @@ -114,6 +118,8 @@ pub struct Config { /// The canonical model files on this device, so engines can be compiled /// ahead of the first request for them. pub models: Vec<(Role, PathBuf)>, + /// Models compiled into the binary, for the same reason. + pub embedded: Vec<(Role, &'static [u8])>, /// The highest rung the user allows; `None` is "the best that works". pub ceiling: Option, /// ONNX Runtime's intra-op pool; 0 picks from the core count. diff --git a/core/dr-inference-engine/src/probe.rs b/core/dr-inference-engine/src/probe.rs index ffff81b..0701769 100644 --- a/core/dr-inference-engine/src/probe.rs +++ b/core/dr-inference-engine/src/probe.rs @@ -201,6 +201,12 @@ fn fingerprint(runtime: &Runtime, cfg: &Config) -> String { }, device_identity(), ]; + for (role, bytes) in &cfg.embedded { + parts.push(format!( + "{role:?} embedded {:016x}", + crate::engines::hash(bytes) + )); + } for (role, path) in &cfg.models { let hash = std::fs::read(path) .map(|b| crate::engines::hash(&b)) diff --git a/core/dr-inference-engine/src/session.rs b/core/dr-inference-engine/src/session.rs index f899f6b..485ae82 100644 --- a/core/dr-inference-engine/src/session.rs +++ b/core/dr-inference-engine/src/session.rs @@ -20,7 +20,7 @@ pub fn build( let mut b = Session::builder()? .with_optimization_level(GraphOptimizationLevel::Level3)? .with_intra_threads(threads(cfg))?; - if strict { + if strict && rung != Rung::Cpu { b = b.with_config_entry("session.disable_cpu_ep_fallback", "1")?; } // A Hexagon session loads the compiled context when there is one and diff --git a/core/dr-segment/src/lib.rs b/core/dr-segment/src/lib.rs index 7dbbb25..b800f97 100644 --- a/core/dr-segment/src/lib.rs +++ b/core/dr-segment/src/lib.rs @@ -71,6 +71,8 @@ pub use refine::{ }; #[cfg(feature = "semantic")] pub use scene::{Category, Scene, SceneModel}; +#[cfg(feature = "embedded-model")] +pub use semantic::embedded_model_bytes; #[cfg(feature = "semantic")] pub use semantic::{Instance, SemanticModel, SemanticOptions, Tiling}; diff --git a/core/dr-segment/src/semantic.rs b/core/dr-segment/src/semantic.rs index e64fa15..35cd523 100644 --- a/core/dr-segment/src/semantic.rs +++ b/core/dr-segment/src/semantic.rs @@ -208,6 +208,13 @@ const EMBEDDED_MODEL: &[u8] = include_bytes!("../../../models/segment/yolo26n-se #[cfg(feature = "embedded-model")] const EMBEDDED_CLASSES: &str = include_str!("../../../models/segment/yolo26n-seg.classes.json"); +/// The bytes of the model that ships with this crate, for whoever compiles +/// engines ahead of the first request (docs/inference.md §6). +#[cfg(feature = "embedded-model")] +pub fn embedded_model_bytes() -> &'static [u8] { + EMBEDDED_MODEL +} + impl SemanticModel { /// Load the model that ships with this crate. #[cfg(feature = "embedded-model")] diff --git a/core/dr-types/src/settings.rs b/core/dr-types/src/settings.rs index c6b04d0..022f19c 100644 --- a/core/dr-types/src/settings.rs +++ b/core/dr-types/src/settings.rs @@ -260,6 +260,20 @@ impl FaceDetector { } } + /// The id when the detector runs in its int8 form (docs/inference.md §7). + /// + /// A different detector: it finds a different set of faces, so it is a + /// different population of detections. The embedder half is unchanged, + /// because the embedder never runs in int8, and `embedder_of` keeps the + /// two spellings' vectors in one space. + pub fn model_id_int8(self) -> &'static str { + match self { + FaceDetector::Scrfd500m => "scrfd_500m_i8+w600k_mbf", + FaceDetector::Scrfd2_5g => "scrfd_2.5g_i8+w600k_mbf", + FaceDetector::Scrfd10g => "scrfd_10g_i8+w600k_mbf", + } + } + /// The detector that writes under a pipeline id, if it is one of these. /// /// The inverse of [`Self::model_id`]. `None` for an id from another diff --git a/docker/android/assemble-apk.sh b/docker/android/assemble-apk.sh index 351facd..adfe128 100755 --- a/docker/android/assemble-apk.sh +++ b/docker/android/assemble-apk.sh @@ -17,6 +17,8 @@ # JNILIBS where cargo-ndk wrote the .so (default: $TARGET_DIR/jniLibs) # OUT output directory (default: $TARGET_DIR/apk) # KEYSTORE signing keystore (default: $TARGET_DIR/debug.keystore) +# RUNTIME_DIR the inference runtime (default: $TARGET_DIR/runtime, +# fetched by tools/fetch-android-runtime.sh) # ABI Android ABI (default: arm64-v8a) # RUST_TARGET Rust target triple (default: aarch64-linux-android) # @@ -43,6 +45,7 @@ TARGET_DIR="$(cd "${TARGET_DIR}" && pwd)" JNILIBS="${JNILIBS:-${TARGET_DIR}/jniLibs}" OUT="${OUT:-${TARGET_DIR}/apk}" KEYSTORE="${KEYSTORE:-${TARGET_DIR}/debug.keystore}" +RUNTIME_DIR="${RUNTIME_DIR:-${TARGET_DIR}/runtime}" # Release signing is selected by supplying a password, not by a flag, so there # is no way to ask for a release build and silently get a debug one. @@ -255,6 +258,25 @@ fi cp "${SO}" "${OUT}/staging/lib/${ABI}/libdarkroom.so" cp "${DEX}" "${OUT}/staging/classes.dex" +# The inference runtime (docs/inference.md §3): ONNX Runtime and Qualcomm's +# Hexagon backend, beside libdarkroom.so so the app finds them in its own +# native library directory. The build links none of it — the app dlopens +# `libonnxruntime.so` at launch and runs on tract if it is not there — so an +# APK without these is a slower app, not a broken one, and `RUNTIME_DIR=none` +# builds exactly that. 174 MB for the default set; the script says which +# Hexagon generations that buys. +if [[ "${RUNTIME_DIR}" != "none" ]]; then + if [[ ! -f "${RUNTIME_DIR}/lib/libonnxruntime.so" ]]; then + "${REPO}/tools/fetch-android-runtime.sh" "${RUNTIME_DIR}" + fi + cp "${RUNTIME_DIR}"/lib/*.so "${OUT}/staging/lib/${ABI}/" + mkdir -p "${OUT}/staging/assets/licences" + cp "${RUNTIME_DIR}"/QNN-*.* "${OUT}/staging/assets/licences/" 2>/dev/null || true + echo " runtime: $(ls "${RUNTIME_DIR}/lib" | wc -l) libraries from ${RUNTIME_DIR}/lib" +else + echo " runtime: none (tract only)" +fi + # The models. Android has no other route to one — app-private storage is not # user-reachable and the in-app fetch is unbuilt (docs/faces.md §2.2a) — so # they go in the APK and `android_main` unpacks them on first launch. The @@ -274,7 +296,7 @@ cp "${DEX}" "${OUT}/staging/classes.dex" # # Cleared first: a previous run that died between staging and cleanup would # otherwise leave models in the APK that are no longer in the tree. -rm -rf "${OUT}/staging/assets" +rm -rf "${OUT}/staging/assets/models" mkdir -p "${OUT}/staging/assets/models" _bundled="" for _dir in face scene; do @@ -315,7 +337,7 @@ fi # install times sane. cd "${OUT}/staging" cp "${OUT}/base.apk" "${OUT}/unaligned.apk" -zip -q -0 -X "${OUT}/unaligned.apk" "lib/${ABI}/libdarkroom.so" +zip -q -0 -X "${OUT}/unaligned.apk" lib/"${ABI}"/*.so zip -q -X "${OUT}/unaligned.apk" classes.dex # Stored, not deflated: an ONNX graph is mostly incompressible float data, so # deflating it buys a few percent and costs the whole file being inflated into diff --git a/tools/fetch-android-runtime.sh b/tools/fetch-android-runtime.sh new file mode 100755 index 0000000..5b11351 --- /dev/null +++ b/tools/fetch-android-runtime.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash +# Fetch the inference runtime the Android APK carries (docs/inference.md §3). +# +# ./tools/fetch-android-runtime.sh [DEST] +# +# Two Maven artefacts, pinned to each other by ONNX Runtime's own POM: +# +# com.microsoft.onnxruntime:onnxruntime-android-qnn MIT +# com.qualcomm.qti:qnn-runtime Qualcomm AI Engine Direct SDK licence +# +# The first is ONNX Runtime built with the CPU, QNN, XNNPACK, NNAPI and WebGPU +# providers; the second is Qualcomm's HTP backend — the ARM-side compiler and +# the per-generation Hexagon "skel" the DSP loads. Both ship as AARs whose +# `jni/arm64-v8a/` is what an APK's `lib/arm64-v8a/` wants, so this script +# unpacks exactly that and nothing else, plus the licence texts, which travel +# with the libraries (§3.1). +# +# ## What is and is not taken from the Qualcomm package +# +# Every HTP generation has its own skel and stub, 16 MB a pair. `QNN_HTP_ARCHS` +# names the ones to ship; the default is the generations in devices sold since +# the 8 Gen 2 (V73 — the MagicPad 2's 8s Gen 3 is one), through the 8 Elite +# (V79) and its successor (V81). V66–V69 are 2020–2021 silicon and are left +# out, which is 50 MB the tablet never loads. `libQnnHtpPrepare.so`, the +# on-device graph compiler, is 84 MB and cannot be left out: it is what turns +# an int8 ONNX graph into something the DSP runs, once per device (§5). +# +# `libQnnGpu.so` and `libQnnDsp*.so` are not taken: the Adreno rung measured +# slower than the Hexagon everywhere the Hexagon exists (§1.1), and the cDSP +# path is the pre-HTP generation. +# +# ## Why a script and not a checked-in copy +# +# 150 MB of vendor binaries in LFS, re-fetched by every clone, for files that +# Maven serves with checksums. The cache directory keeps them across builds; +# CI's is the target cache it already mounts. +set -euo pipefail + +ORT_VERSION="${ORT_VERSION:-1.29.0}" +QNN_VERSION="${QNN_VERSION:-2.42.0}" +QNN_HTP_ARCHS="${QNN_HTP_ARCHS:-73 75 79 81}" +DEST="${1:-${PWD}/target-android/runtime}" +MAVEN="https://repo1.maven.org/maven2" + +fetch() { + # A Maven artefact and its SHA-1, verified before anything is unpacked. + local group="$1" artefact="$2" version="$3" out="$4" + local url="${MAVEN}/${group//.//}/${artefact}/${version}/${artefact}-${version}.aar" + if [[ -f "${out}" && -f "${out}.sha1" ]] \ + && [[ "$(sha1sum "${out}" | cut -d' ' -f1)" == "$(cut -c1-40 "${out}.sha1")" ]]; then + return + fi + echo "==> fetching ${artefact} ${version}" + curl -fsSL -o "${out}.sha1" "${url}.sha1" + curl -fsSL -o "${out}" "${url}" + [[ "$(sha1sum "${out}" | cut -d' ' -f1)" == "$(cut -c1-40 "${out}.sha1")" ]] || { + echo "error: ${artefact}-${version}.aar does not match its published SHA-1" >&2 + rm -f "${out}" + exit 1 + } +} + +mkdir -p "${DEST}/aar" "${DEST}/lib" +fetch com.microsoft.onnxruntime onnxruntime-android-qnn "${ORT_VERSION}" \ + "${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar" +fetch com.qualcomm.qti qnn-runtime "${QNN_VERSION}" \ + "${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar" + +# The ONNX Runtime POM names the QNN version it was built against; a pair +# that disagrees loads and then fails at the first graph, which is the kind +# of failure the probe would only report as "Hexagon failed". +pom="${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.pom" +[[ -f "${pom}" ]] || curl -fsSL -o "${pom}" \ + "${MAVEN}/com/microsoft/onnxruntime/onnxruntime-android-qnn/${ORT_VERSION}/onnxruntime-android-qnn-${ORT_VERSION}.pom" +wanted="$(sed -n '/qnn-runtime<\/artifactId>/{n;s/.*\(.*\)<\/version>.*/\1/p}' "${pom}")" +if [[ -n "${wanted}" && "${wanted}" != "${QNN_VERSION}" ]]; then + echo "error: ONNX Runtime ${ORT_VERSION} was built against QNN ${wanted}, not ${QNN_VERSION}" >&2 + exit 1 +fi + +rm -rf "${DEST}/lib" +mkdir -p "${DEST}/lib" +unzip -q -o -j "${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar" \ + 'jni/arm64-v8a/libonnxruntime.so' -d "${DEST}/lib" +members=(jni/arm64-v8a/libQnnHtp.so jni/arm64-v8a/libQnnHtpPrepare.so jni/arm64-v8a/libQnnSystem.so) +for arch in ${QNN_HTP_ARCHS}; do + members+=("jni/arm64-v8a/libQnnHtpV${arch}Skel.so" "jni/arm64-v8a/libQnnHtpV${arch}Stub.so") +done +unzip -q -o -j "${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar" "${members[@]}" -d "${DEST}/lib" +unzip -q -o -j "${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar" 'LICENSE.pdf' 'NOTICE.txt' -d "${DEST}" 2>/dev/null || true +mv -f "${DEST}/LICENSE.pdf" "${DEST}/QNN-LICENSE.pdf" 2>/dev/null || true +mv -f "${DEST}/NOTICE.txt" "${DEST}/QNN-NOTICE.txt" 2>/dev/null || true + +echo "==> runtime in ${DEST}/lib:" +du -sh "${DEST}/lib" | cut -f1 | sed 's/^/ /' +ls "${DEST}/lib" | sed 's/^/ /' diff --git a/ui/dr-ui/Cargo.toml b/ui/dr-ui/Cargo.toml index 6cf1407..f048684 100644 --- a/ui/dr-ui/Cargo.toml +++ b/ui/dr-ui/Cargo.toml @@ -49,6 +49,9 @@ dr-catalog.workspace = true # The face pipeline, with the ONNX runtime: this is the layer that actually # runs the models over the library (docs/faces.md). dr-face = { workspace = true, features = ["inference"] } +# The engine behind both. `native` here means the *app* may look for a +# runtime file; the build stays C-free either way (docs/inference.md §3). +dr-inference-engine = { workspace = true, features = ["native"] } dr-thumbs.workspace = true # The library module writes scan results straight into the catalog, so it # needs the same SQLite types dr-catalog exposes. diff --git a/ui/dr-ui/src/identity_ui.rs b/ui/dr-ui/src/identity_ui.rs index 2afdc70..73ef482 100644 --- a/ui/dr-ui/src/identity_ui.rs +++ b/ui/dr-ui/src/identity_ui.rs @@ -27,7 +27,7 @@ use crate::{AppWindow, IdentityFace, IdentityPerson}; /// use rather than captured once, because the page can change it while the /// screen is open. pub fn model_id(settings: &crate::settings_ui::SettingsController) -> String { - settings.snapshot().faces.detector.model_id().to_string() + crate::inference::model_id(settings.snapshot().faces.detector).to_string() } /// Screen state that outlives a single callback. diff --git a/ui/dr-ui/src/inference.rs b/ui/dr-ui/src/inference.rs new file mode 100644 index 0000000..ffb021c --- /dev/null +++ b/ui/dr-ui/src/inference.rs @@ -0,0 +1,91 @@ +//! The app's side of `dr-inference-engine` (docs/inference.md §8). +//! +//! What lives here is what only the app knows: where the runtime file might +//! be, where the disposable cache goes, which model files this device has, +//! and how the engine's status becomes a line on the settings page. What +//! runs the models does not. + +use std::path::PathBuf; + +use dr_inference_engine::{Form, Role, Status}; +use dr_types::FaceDetector; + +/// Start the engine: choose the runtime, probe in the background, compile +/// engines for whatever this device turns out to have. +/// +/// `runtime_dirs` is where the platform put `libonnxruntime`: an empty list +/// is the tract build. Called once, after the models are on disk — on +/// Android that is the end of `install_bundled_models`, since the probe +/// fingerprints the model files and a probe before they land would be a +/// probe of nothing. +pub fn init(runtime_dirs: Vec) { + let dir = crate::library::shared_face_models_dir(); + let mut models: Vec<(Role, PathBuf)> = FaceDetector::ALL + .iter() + .map(|d| (Role::Detector, dir.join(d.file_name()))) + .collect(); + models.push((Role::Embedder, dir.join("arcface_mbf_b1.onnx"))); + models.push((Role::Scene, dir.join("yolo26s-sem-ade20k.onnx"))); + models.push((Role::Landmarks, dir.join(crate::library::LANDMARK_MODEL))); + models.push((Role::EyeClassifier, dir.join(crate::library::EYE_MODEL))); + models.push(( + Role::EyeClassifier, + dir.join(crate::library::SUNGLASSES_MODEL), + )); + models.retain(|(_, p)| p.is_file()); + + dr_inference_engine::init(dr_inference_engine::Config { + runtime_dirs, + cache_dir: crate::library::inference_cache_dir(), + models, + embedded: vec![(Role::Segmenter, dr_segment::embedded_model_bytes())], + ceiling: None, + threads: 0, + decay: std::time::Duration::ZERO, + }); + + // A low-memory signal drops every session nobody is mid-run with; the + // next use loads again. Same tier as the GPU caches: rebuilt from data + // the process still holds, and on a mobile GPU or NPU the largest pool. + crate::memory::evict_at(crate::memory::Tier::Gpu, dr_inference_engine::release_all); +} + +/// Which form the current backend loads `detector` in, given the files on +/// this device — the fact `faces.model_id` has to carry (§7). +/// +/// Reads the shared directory only. An account-private model directory can +/// override the file `library::face_models` loads, but not which form the +/// backend wants, and the int8 sibling is something a packager ships, not +/// something a user drops in. +pub fn detector_form(detector: FaceDetector) -> Form { + let canonical = crate::library::shared_face_models_dir().join(detector.file_name()); + dr_inference_engine::resolve_model(Role::Detector, &canonical).1 +} + +/// The `faces.model_id` this device indexes under with `detector`. +pub fn model_id(detector: FaceDetector) -> &'static str { + match detector_form(detector) { + Form::F32 => detector.model_id(), + Form::Int8 => detector.model_id_int8(), + } +} + +/// The two lines the About panel shows: what is running the models, and +/// why or how far along. +pub fn about_lines() -> (String, String) { + let status: Status = dr_inference_engine::status(); + let line = status.line(); + let detail = if status.probing { + "Checking what this device can run the models on…".to_string() + } else if status.engines.1 > 0 && status.engines.0 < status.engines.1 { + format!( + "Preparing {} engines · {} of {}", + status.rung.label(), + status.engines.0, + status.engines.1 + ) + } else { + status.reason + }; + (line, detail) +} diff --git a/ui/dr-ui/src/lib.rs b/ui/dr-ui/src/lib.rs index 6891f76..5d9ae7a 100644 --- a/ui/dr-ui/src/lib.rs +++ b/ui/dr-ui/src/lib.rs @@ -39,6 +39,7 @@ pub mod identity; mod identity_ui; mod import; mod import_ui; +pub mod inference; mod labels; mod library; mod library_ui; @@ -1209,6 +1210,32 @@ pub fn run(paths: Vec) -> Result<()> { None => window.set_backend("NO GPU".into()), } + // TRACES: FR-INF-1 + // What the models run on. Re-read every two seconds because the answer + // changes twice after launch — when the probe reports and as each + // engine lands — and the page is open for longer than either takes. + { + let set = |w: &AppWindow| { + let (line, detail) = inference::about_lines(); + w.set_inference_backend(line.into()); + w.set_inference_detail(detail.into()); + }; + set(&window); + let weak = window.as_weak(); + let timer = Rc::new(slint::Timer::default()); + let held = timer.clone(); + timer.start( + slint::TimerMode::Repeated, + std::time::Duration::from_secs(2), + move || { + let _keep = &held; + if let Some(w) = weak.upgrade() { + set(&w); + } + }, + ); + } + // TRACES: NFR-OPS-1 // The diagnostics bundle, wired as the two presses the requirement // describes. Preparing gathers the log and the crash records into memory @@ -1638,7 +1665,7 @@ pub fn run(paths: Vec) -> Result<()> { library.set_fetch_ahead(stored.cache.fetch_ahead); library.set_write_xmp_sidecars(stored.library.write_xmp_sidecars); library.set_timeline_bars(stored.library.timeline_bars); - library.set_face_model_id(stored.faces.detector.model_id()); + library.set_face_model_id(inference::model_id(stored.faces.detector)); } // --- the export folder picker ---------------------------------------- @@ -1790,7 +1817,7 @@ pub fn run(paths: Vec) -> Result<()> { // file is installed, and how much of the library that // pipeline has covered — which for a freshly chosen one is // nothing, and saying so is the point. - lib.set_face_model_id(s.faces.detector.model_id()); + lib.set_face_model_id(inference::model_id(s.faces.detector)); if let Some(w) = weak.upgrade() { refresh_face_status(&w, &lib, s.faces.detector); } @@ -3933,7 +3960,7 @@ fn refresh_face_status( window, &library.catalog(), store.as_ref(), - detector.model_id(), + inference::model_id(detector), models.as_ref().is_some_and(|m| m.eyes.is_some()), ); window.set_identity_model_missing(models.is_none()); diff --git a/ui/dr-ui/src/library.rs b/ui/dr-ui/src/library.rs index 755579f..b232847 100644 --- a/ui/dr-ui/src/library.rs +++ b/ui/dr-ui/src/library.rs @@ -5191,6 +5191,13 @@ impl FaceModelPaths { } } +/// Where the inference engine keeps what it derives per device: the probe +/// result and compiled engines (docs/inference.md §4, §5). A peer of +/// `thumbs`, not of the catalog: disposable, regenerable, never synced. +pub fn inference_cache_dir() -> PathBuf { + data_root().join("inference") +} + /// The detector and embedder files, if both are present — and the eye /// models beside them, if those are. /// diff --git a/ui/dr-ui/src/library_ui.rs b/ui/dr-ui/src/library_ui.rs index 71bcd75..b01abd6 100644 --- a/ui/dr-ui/src/library_ui.rs +++ b/ui/dr-ui/src/library_ui.rs @@ -484,7 +484,9 @@ impl LibraryController { dr_types::LibrarySettings::default().write_xmp_sidecars, ), timeline_bars: std::cell::Cell::new(dr_types::LibrarySettings::default().timeline_bars), - face_model_id: RefCell::new(dr_types::FaceDetector::default().model_id().to_string()), + face_model_id: RefCell::new( + crate::inference::model_id(dr_types::FaceDetector::default()).to_string(), + ), }) } diff --git a/ui/dr-ui/ui/app.slint b/ui/dr-ui/ui/app.slint index d389d80..cb0805c 100644 --- a/ui/dr-ui/ui/app.slint +++ b/ui/dr-ui/ui/app.slint @@ -58,6 +58,8 @@ export component AppWindow inherits Window { in property canvas; in property adapter: "detecting…"; in property backend: "—"; + in property inference-backend: "detecting…"; + in property inference-detail: ""; in property fps: 0; /// TRACES: FR-DSP-8 /// The display showing the canvas and the colour transform it is getting. @@ -1322,6 +1324,8 @@ in property panel-visible: true; adapter: root.adapter; backend: root.backend; + inference-backend: root.inference-backend; + inference-detail: root.inference-detail; fps: root.fps; layout-class: root.layout-class; app-version: root.app-version; diff --git a/ui/dr-ui/ui/settings.slint b/ui/dr-ui/ui/settings.slint index 886bc5e..110f348 100644 --- a/ui/dr-ui/ui/settings.slint +++ b/ui/dr-ui/ui/settings.slint @@ -137,6 +137,10 @@ export component SettingsPage inherits Rectangle { /// something up, not where they are looked at all day. in property adapter; in property backend; + /// What runs the neural models and how it was chosen — the two lines + /// `dr_ui::inference::about_lines` produces (docs/inference.md §4). + in property inference-backend; + in property inference-detail; in property fps; in property layout-class; in property app-version; @@ -872,6 +876,26 @@ export component SettingsPage inherits Rectangle { } } + // TRACES: FR-INF-1 + // The backend the models run on, and why. Beside + // Graphics because it is the same kind of fact: a + // property of this device, chosen by measurement, + // that a bug report about a slow index should quote. + HorizontalLayout { + spacing: Theme.gap; + Label { text: "Inference"; } + Value { + text: root.inference-backend; + horizontal-stretch: 1; + overflow: elide; + } + } + + if root.inference-detail != "": Caption { + text: root.inference-detail; + wrap: word-wrap; + } + HorizontalLayout { spacing: Theme.gap; Label { text: "Frame rate"; }