diff --git a/Cargo.lock b/Cargo.lock index 67edfcc..be2dcaa 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1688,6 +1688,7 @@ dependencies = [ "dr-face", "dr-film", "dr-gpu", + "dr-inference-engine", "dr-ingest", "dr-lens", "dr-pano", diff --git a/apps/darkroom-android/src/lib.rs b/apps/darkroom-android/src/lib.rs index dd62545..e10e57d 100644 --- a/apps/darkroom-android/src/lib.rs +++ b/apps/darkroom-android/src/lib.rs @@ -400,6 +400,27 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) { "bundled models ready: {copied} bytes copied in {} ms", started.elapsed().as_millis() ); + + // Now, and not at launch: the probe fingerprints the model files, and + // on a first launch they were not on disk until this line. The runtime + // is in the APK's native library directory beside `libdarkroom.so`, + // which is also where Qualcomm's DSP loader has to be pointed for the + // Hexagon skel (docs/inference.md §3, §8). + dr_ui::inference::init(native_library_dir().into_iter().collect()); +} + +/// The directory the system unpacked this APK's native libraries into. +/// +/// Read from where the loader put *this* library rather than asked of the +/// activity: `android-activity` does not expose `nativeLibraryDir`, and the +/// answer is in `/proc/self/maps` for free. +#[cfg(target_os = "android")] +fn native_library_dir() -> Option { + let maps = std::fs::read_to_string("/proc/self/maps").ok()?; + maps.lines() + .filter_map(|l| l.split_whitespace().nth(5)) + .find(|p| p.ends_with("/libdarkroom.so")) + .and_then(|p| std::path::Path::new(p).parent().map(Into::into)) } /// TRACES: FR-PLAT-AND-6 diff --git a/apps/darkroom-desktop/src/main.rs b/apps/darkroom-desktop/src/main.rs index 07d6655..1b9933b 100644 --- a/apps/darkroom-desktop/src/main.rs +++ b/apps/darkroom-desktop/src/main.rs @@ -61,6 +61,11 @@ fn main() -> anyhow::Result<()> { eprintln!("usage: darkroom-desktop ..."); } + // Before the window: the probe runs on its own thread and the first + // frame does not wait for it, but the models a background job asks for + // should already know where the runtime is (docs/inference.md §4). + dr_ui::inference::init(runtime_dirs()); + dr_ui::run(paths)?; // Skip Rust's normal static/thread-local teardown on the way out: a @@ -70,3 +75,34 @@ fn main() -> anyhow::Result<()> { // destruction" when the window is closed. std::process::exit(0); } + +/// Where a desktop package may have put `libonnxruntime`, most specific +/// first. None of these existing is the tract build, which is a complete +/// application and not an error (docs/inference.md §3). +/// +/// `DARKROOM_ORT_DIR` is for a developer pointing at a runtime that is not +/// installed — the wheel's `capi` directory, say. Then beside the executable +/// and in the package's private library directory, for a package that +/// bundles its own; then the Flatpak prefix; then the system library +/// directory, for a distribution that ships ONNX Runtime as a package of its +/// own. A system copy whose GPU providers do not load is not a problem: the +/// probe builds a real session before believing a provider. +fn runtime_dirs() -> Vec { + let mut dirs = Vec::new(); + if let Some(dir) = std::env::var_os("DARKROOM_ORT_DIR") { + dirs.push(PathBuf::from(dir)); + } + if let Ok(exe) = std::env::current_exe() { + if let Some(bin) = exe.parent() { + dirs.push(bin.to_path_buf()); + dirs.push(bin.join("../lib/darkroom")); + } + } + #[cfg(target_os = "linux")] + dirs.extend([ + PathBuf::from("/app/lib/darkroom"), + PathBuf::from("/usr/lib/darkroom"), + PathBuf::from("/usr/lib"), + ]); + dirs +} diff --git a/core/dr-face/src/classify.rs b/core/dr-face/src/classify.rs index b21f4ec..de4234c 100644 --- a/core/dr-face/src/classify.rs +++ b/core/dr-face/src/classify.rs @@ -40,16 +40,17 @@ use crate::align::{ }; use crate::eyes::{Eye, EyeReading}; use crate::landmarks::{Landmarker, Landmarks}; -use crate::{install_backend, FaceError, Pixels}; +use crate::{FaceError, Pixels}; +use dr_inference_engine::{Form, Model, Role}; /// A loaded OCEC graph. pub struct EyeClassifier { - session: ort::session::Session, + session: Model, } /// A loaded SGC graph. pub struct SunglassesClassifier { - session: ort::session::Session, + session: Model, } /// Open a single-input, single-output classifier and check it is the shape @@ -63,12 +64,10 @@ fn open_classifier( bytes: &[u8], expected: &'static str, (h, w): (usize, usize), -) -> Result { - install_backend(); - let session = ort::session::Session::builder() - .map_err(FaceError::Inference)? - .commit_from_memory(bytes) - .map_err(FaceError::Inference)?; +) -> Result { + let model = dr_inference_engine::open(Role::EyeClassifier, Form::F32, bytes)?; + let acquired = model.acquire()?; + let session = acquired.lock(); let input = session.inputs().first().ok_or(FaceError::WrongModel { expected, @@ -93,7 +92,9 @@ fn open_classifier( detail: format!("{} outputs, expected one", session.outputs().len()), }); } - Ok(session) + drop(session); + drop(acquired); + Ok(model) } /// Lay a `h × w` RGB crop out as the `[1, 3, h, w]` tensor both graphs take. @@ -110,11 +111,9 @@ fn to_nchw(pixels: &[f32], h: usize, w: usize) -> Array4 { } /// Run a one-number classifier and read its sigmoid back, clamped. -fn run_scalar( - session: &mut ort::session::Session, - input: Array4, - expected: &'static str, -) -> Result { +fn run_scalar(model: &Model, input: Array4, expected: &'static str) -> Result { + let acquired = model.acquire()?; + let mut session = acquired.lock(); let outputs = session .run(ort::inputs![ ort::value::Tensor::from_array(input).map_err(FaceError::Inference)? @@ -149,7 +148,7 @@ impl EyeClassifier { /// P(open) for one eye. pub fn classify(&mut self, eye: &EyePatch) -> Result { let input = to_nchw(eye.pixels(), EYE_PATCH_HEIGHT, EYE_PATCH_WIDTH); - run_scalar(&mut self.session, input, "OCEC") + run_scalar(&self.session, input, "OCEC") } } @@ -170,7 +169,7 @@ impl SunglassesClassifier { let mut best = 0.0_f32; for view in head.views() { let input = to_nchw(view, SUNGLASSES_EDGE, SUNGLASSES_EDGE); - best = best.max(run_scalar(&mut self.session, input, "SGC")?); + best = best.max(run_scalar(&self.session, input, "SGC")?); } Ok(best) } diff --git a/core/dr-face/src/landmarks.rs b/core/dr-face/src/landmarks.rs index 2321a07..fdd857e 100644 --- a/core/dr-face/src/landmarks.rs +++ b/core/dr-face/src/landmarks.rs @@ -32,7 +32,8 @@ use ndarray::Array4; use crate::align::crop_box; -use crate::{install_backend, FaceError, Pixels}; +use crate::{FaceError, Pixels}; +use dr_inference_engine::{Form, Model, Role}; /// The graph's input edge, in pixels. pub const INPUT_EDGE: usize = 192; @@ -118,7 +119,7 @@ impl Landmarks { /// A loaded `2d106det` graph. pub struct Landmarker { - session: ort::session::Session, + session: Model, } impl Landmarker { @@ -128,11 +129,9 @@ impl Landmarker { } pub fn from_bytes(bytes: &[u8]) -> Result { - install_backend(); - let session = ort::session::Session::builder() - .map_err(FaceError::Inference)? - .commit_from_memory(bytes) - .map_err(FaceError::Inference)?; + let model = dr_inference_engine::open(Role::Landmarks, Form::F32, bytes)?; + let acquired = model.acquire()?; + let session = acquired.lock(); let input = session.inputs().first().ok_or(FaceError::WrongModel { expected: "2d106det", @@ -167,7 +166,9 @@ impl Landmarker { ), }); } - Ok(Self { session }) + drop(session); + drop(acquired); + Ok(Self { session: model }) } /// The landmarks of the face in `bbox` — `(x0, y0, x1, y1)` in source @@ -206,8 +207,9 @@ impl Landmarker { } } } - let outputs = self - .session + let acquired = self.session.acquire()?; + let mut session = acquired.lock(); + let outputs = session .run(ort::inputs![ ort::value::Tensor::from_array(input).map_err(FaceError::Inference)? ]) diff --git a/core/dr-inference-engine/src/api.rs b/core/dr-inference-engine/src/api.rs index d18c527..e8a895f 100644 --- a/core/dr-inference-engine/src/api.rs +++ b/core/dr-inference-engine/src/api.rs @@ -38,10 +38,16 @@ pub fn runtime() -> Runtime { } /// Install a table if none is installed yet — tract, since no directories -/// were named. What a test or an example gets. +/// were named. What a test or an example gets, unless `DARKROOM_ORT_DIR` +/// names a runtime: the same variable the desktop honours, so an example +/// can be pointed at the runtime the app uses without learning `init`. pub fn ensure_installed() { if RUNTIME.get().is_none() { - install(&[]); + let dirs: Vec = std::env::var_os("DARKROOM_ORT_DIR") + .map(PathBuf::from) + .into_iter() + .collect(); + install(&dirs); } } @@ -93,11 +99,7 @@ fn load_native(dir: &std::path::Path) -> Result { let path = if dir.as_os_str().is_empty() { PathBuf::from(name) } else { - let p = dir.join(name); - if !p.is_file() { - return Err("not present".into()); - } - p + find_library(dir, name).ok_or("not present")? }; // SAFETY: the library's initialisers are ONNX Runtime's own; the symbol @@ -140,3 +142,28 @@ fn load_native(dir: &std::path::Path) -> Result { Ok(Runtime::OnnxRuntime { path, version }) } } + +/// `libonnxruntime.so` in `dir`, or a versioned spelling of it — +/// `libonnxruntime.so.1.30.0` is what the Python wheel ships, and a package +/// that installs only the versioned file is not wrong. +#[cfg(feature = "native")] +fn find_library(dir: &std::path::Path, name: &str) -> Option { + let exact = dir.join(name); + if exact.is_file() { + return Some(exact); + } + let prefix = format!("{name}."); + let mut versioned: Vec = std::fs::read_dir(dir) + .ok()? + .filter_map(|e| e.ok()) + .map(|e| e.path()) + .filter(|p| { + p.is_file() + && p.file_name() + .and_then(|n| n.to_str()) + .is_some_and(|n| n.starts_with(&prefix)) + }) + .collect(); + versioned.sort(); + versioned.pop() +} diff --git a/core/dr-inference-engine/src/engines.rs b/core/dr-inference-engine/src/engines.rs index d565758..12df295 100644 --- a/core/dr-inference-engine/src/engines.rs +++ b/core/dr-inference-engine/src/engines.rs @@ -10,6 +10,11 @@ use std::path::PathBuf; use crate::{state, Config, Form, Rung}; +enum Source { + File(PathBuf), + Bytes(&'static [u8]), +} + /// 64-bit FNV-1a. A cache key, not a checksum: two model files that collide /// here would have to also be the same size and the same role, and the cost /// of that is a rebuilt engine. @@ -47,34 +52,42 @@ pub fn run() { // Smallest first, so the detector — the one that runs per image — is // ready soonest (§6 step 3). - let mut jobs: Vec<(crate::Role, PathBuf, u64)> = cfg + let mut jobs: Vec<(crate::Role, Source, u64)> = cfg .models .iter() - .filter(|(role, _)| rung.form(*role) != Form::F32 || rung != Rung::Hexagon) .filter_map(|(role, path)| { let (path, form) = crate::resolve_model(*role, path); (form == rung.form(*role)).then(|| { let size = std::fs::metadata(&path).map(|m| m.len()).unwrap_or(0); - (*role, path, size) + (*role, Source::File(path), size) }) }) + .chain(cfg.embedded.iter().filter_map(|(role, bytes)| { + // An embedded model has no int8 sibling to offer a rung that + // wants one; it runs on that rung's fallback. + (rung.form(*role) == Form::F32).then_some(( + *role, + Source::Bytes(bytes), + bytes.len() as u64, + )) + })) .collect(); jobs.sort_by_key(|j| j.2); state().lock().unwrap().wanted = jobs.len(); - for (role, path, _) in jobs { - let Ok(bytes) = std::fs::read(&path) else { - continue; + for (role, source, _) in jobs { + let (bytes, name) = match &source { + Source::File(path) => match std::fs::read(path) { + Ok(b) => (b, path.display().to_string()), + Err(_) => continue, + }, + Source::Bytes(b) => (b.to_vec(), format!("embedded {role:?}")), }; let key = key(rung, &bytes); if state().lock().unwrap().cache.compiled.contains(&key) { continue; } - log::info!( - "inference: compiling {} for {}", - path.display(), - rung.label() - ); + log::info!("inference: compiling {name} for {}", rung.label()); let started = std::time::Instant::now(); match crate::session::build(rung, role, &bytes, &cfg, false) { Ok(session) => { @@ -83,8 +96,7 @@ pub fn run() { s.cache.compiled.insert(key); crate::probe::write_cache(&s.config, &s.cache); log::info!( - "inference: {} ready on {} in {:.1} s", - path.display(), + "inference: {name} ready on {} in {:.1} s", rung.label(), started.elapsed().as_secs_f64() ); @@ -94,8 +106,7 @@ pub fn run() { // their engine. A corrected model file changes the hash and // is retried. log::warn!( - "inference: {} will not compile for {}: {e}", - path.display(), + "inference: {name} will not compile for {}: {e}", rung.label() ); } diff --git a/core/dr-inference-engine/src/lib.rs b/core/dr-inference-engine/src/lib.rs index 2eac684..0cb706e 100644 --- a/core/dr-inference-engine/src/lib.rs +++ b/core/dr-inference-engine/src/lib.rs @@ -35,6 +35,10 @@ pub enum Role { Embedder, Segmenter, Scene, + /// The dense landmark model behind the eye reading (docs/faces.md §7c). + Landmarks, + /// The eye-state and sunglasses classifiers, a few hundred kilobytes. + EyeClassifier, } /// Which numeric form of a model a session was built from. @@ -114,6 +118,8 @@ pub struct Config { /// The canonical model files on this device, so engines can be compiled /// ahead of the first request for them. pub models: Vec<(Role, PathBuf)>, + /// Models compiled into the binary, for the same reason. + pub embedded: Vec<(Role, &'static [u8])>, /// The highest rung the user allows; `None` is "the best that works". pub ceiling: Option, /// ONNX Runtime's intra-op pool; 0 picks from the core count. diff --git a/core/dr-inference-engine/src/probe.rs b/core/dr-inference-engine/src/probe.rs index ffff81b..0701769 100644 --- a/core/dr-inference-engine/src/probe.rs +++ b/core/dr-inference-engine/src/probe.rs @@ -201,6 +201,12 @@ fn fingerprint(runtime: &Runtime, cfg: &Config) -> String { }, device_identity(), ]; + for (role, bytes) in &cfg.embedded { + parts.push(format!( + "{role:?} embedded {:016x}", + crate::engines::hash(bytes) + )); + } for (role, path) in &cfg.models { let hash = std::fs::read(path) .map(|b| crate::engines::hash(&b)) diff --git a/core/dr-inference-engine/src/session.rs b/core/dr-inference-engine/src/session.rs index f899f6b..485ae82 100644 --- a/core/dr-inference-engine/src/session.rs +++ b/core/dr-inference-engine/src/session.rs @@ -20,7 +20,7 @@ pub fn build( let mut b = Session::builder()? .with_optimization_level(GraphOptimizationLevel::Level3)? .with_intra_threads(threads(cfg))?; - if strict { + if strict && rung != Rung::Cpu { b = b.with_config_entry("session.disable_cpu_ep_fallback", "1")?; } // A Hexagon session loads the compiled context when there is one and diff --git a/core/dr-segment/src/lib.rs b/core/dr-segment/src/lib.rs index 7dbbb25..b800f97 100644 --- a/core/dr-segment/src/lib.rs +++ b/core/dr-segment/src/lib.rs @@ -71,6 +71,8 @@ pub use refine::{ }; #[cfg(feature = "semantic")] pub use scene::{Category, Scene, SceneModel}; +#[cfg(feature = "embedded-model")] +pub use semantic::embedded_model_bytes; #[cfg(feature = "semantic")] pub use semantic::{Instance, SemanticModel, SemanticOptions, Tiling}; diff --git a/core/dr-segment/src/semantic.rs b/core/dr-segment/src/semantic.rs index e64fa15..35cd523 100644 --- a/core/dr-segment/src/semantic.rs +++ b/core/dr-segment/src/semantic.rs @@ -208,6 +208,13 @@ const EMBEDDED_MODEL: &[u8] = include_bytes!("../../../models/segment/yolo26n-se #[cfg(feature = "embedded-model")] const EMBEDDED_CLASSES: &str = include_str!("../../../models/segment/yolo26n-seg.classes.json"); +/// The bytes of the model that ships with this crate, for whoever compiles +/// engines ahead of the first request (docs/inference.md §6). +#[cfg(feature = "embedded-model")] +pub fn embedded_model_bytes() -> &'static [u8] { + EMBEDDED_MODEL +} + impl SemanticModel { /// Load the model that ships with this crate. #[cfg(feature = "embedded-model")] diff --git a/core/dr-types/src/settings.rs b/core/dr-types/src/settings.rs index c6b04d0..022f19c 100644 --- a/core/dr-types/src/settings.rs +++ b/core/dr-types/src/settings.rs @@ -260,6 +260,20 @@ impl FaceDetector { } } + /// The id when the detector runs in its int8 form (docs/inference.md §7). + /// + /// A different detector: it finds a different set of faces, so it is a + /// different population of detections. The embedder half is unchanged, + /// because the embedder never runs in int8, and `embedder_of` keeps the + /// two spellings' vectors in one space. + pub fn model_id_int8(self) -> &'static str { + match self { + FaceDetector::Scrfd500m => "scrfd_500m_i8+w600k_mbf", + FaceDetector::Scrfd2_5g => "scrfd_2.5g_i8+w600k_mbf", + FaceDetector::Scrfd10g => "scrfd_10g_i8+w600k_mbf", + } + } + /// The detector that writes under a pipeline id, if it is one of these. /// /// The inverse of [`Self::model_id`]. `None` for an id from another diff --git a/docker/android/assemble-apk.sh b/docker/android/assemble-apk.sh index 351facd..adfe128 100755 --- a/docker/android/assemble-apk.sh +++ b/docker/android/assemble-apk.sh @@ -17,6 +17,8 @@ # JNILIBS where cargo-ndk wrote the .so (default: $TARGET_DIR/jniLibs) # OUT output directory (default: $TARGET_DIR/apk) # KEYSTORE signing keystore (default: $TARGET_DIR/debug.keystore) +# RUNTIME_DIR the inference runtime (default: $TARGET_DIR/runtime, +# fetched by tools/fetch-android-runtime.sh) # ABI Android ABI (default: arm64-v8a) # RUST_TARGET Rust target triple (default: aarch64-linux-android) # @@ -43,6 +45,7 @@ TARGET_DIR="$(cd "${TARGET_DIR}" && pwd)" JNILIBS="${JNILIBS:-${TARGET_DIR}/jniLibs}" OUT="${OUT:-${TARGET_DIR}/apk}" KEYSTORE="${KEYSTORE:-${TARGET_DIR}/debug.keystore}" +RUNTIME_DIR="${RUNTIME_DIR:-${TARGET_DIR}/runtime}" # Release signing is selected by supplying a password, not by a flag, so there # is no way to ask for a release build and silently get a debug one. @@ -255,6 +258,25 @@ fi cp "${SO}" "${OUT}/staging/lib/${ABI}/libdarkroom.so" cp "${DEX}" "${OUT}/staging/classes.dex" +# The inference runtime (docs/inference.md §3): ONNX Runtime and Qualcomm's +# Hexagon backend, beside libdarkroom.so so the app finds them in its own +# native library directory. The build links none of it — the app dlopens +# `libonnxruntime.so` at launch and runs on tract if it is not there — so an +# APK without these is a slower app, not a broken one, and `RUNTIME_DIR=none` +# builds exactly that. 174 MB for the default set; the script says which +# Hexagon generations that buys. +if [[ "${RUNTIME_DIR}" != "none" ]]; then + if [[ ! -f "${RUNTIME_DIR}/lib/libonnxruntime.so" ]]; then + "${REPO}/tools/fetch-android-runtime.sh" "${RUNTIME_DIR}" + fi + cp "${RUNTIME_DIR}"/lib/*.so "${OUT}/staging/lib/${ABI}/" + mkdir -p "${OUT}/staging/assets/licences" + cp "${RUNTIME_DIR}"/QNN-*.* "${OUT}/staging/assets/licences/" 2>/dev/null || true + echo " runtime: $(ls "${RUNTIME_DIR}/lib" | wc -l) libraries from ${RUNTIME_DIR}/lib" +else + echo " runtime: none (tract only)" +fi + # The models. Android has no other route to one — app-private storage is not # user-reachable and the in-app fetch is unbuilt (docs/faces.md §2.2a) — so # they go in the APK and `android_main` unpacks them on first launch. The @@ -274,7 +296,7 @@ cp "${DEX}" "${OUT}/staging/classes.dex" # # Cleared first: a previous run that died between staging and cleanup would # otherwise leave models in the APK that are no longer in the tree. -rm -rf "${OUT}/staging/assets" +rm -rf "${OUT}/staging/assets/models" mkdir -p "${OUT}/staging/assets/models" _bundled="" for _dir in face scene; do @@ -315,7 +337,7 @@ fi # install times sane. cd "${OUT}/staging" cp "${OUT}/base.apk" "${OUT}/unaligned.apk" -zip -q -0 -X "${OUT}/unaligned.apk" "lib/${ABI}/libdarkroom.so" +zip -q -0 -X "${OUT}/unaligned.apk" lib/"${ABI}"/*.so zip -q -X "${OUT}/unaligned.apk" classes.dex # Stored, not deflated: an ONNX graph is mostly incompressible float data, so # deflating it buys a few percent and costs the whole file being inflated into diff --git a/tools/fetch-android-runtime.sh b/tools/fetch-android-runtime.sh new file mode 100755 index 0000000..5b11351 --- /dev/null +++ b/tools/fetch-android-runtime.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash +# Fetch the inference runtime the Android APK carries (docs/inference.md §3). +# +# ./tools/fetch-android-runtime.sh [DEST] +# +# Two Maven artefacts, pinned to each other by ONNX Runtime's own POM: +# +# com.microsoft.onnxruntime:onnxruntime-android-qnn MIT +# com.qualcomm.qti:qnn-runtime Qualcomm AI Engine Direct SDK licence +# +# The first is ONNX Runtime built with the CPU, QNN, XNNPACK, NNAPI and WebGPU +# providers; the second is Qualcomm's HTP backend — the ARM-side compiler and +# the per-generation Hexagon "skel" the DSP loads. Both ship as AARs whose +# `jni/arm64-v8a/` is what an APK's `lib/arm64-v8a/` wants, so this script +# unpacks exactly that and nothing else, plus the licence texts, which travel +# with the libraries (§3.1). +# +# ## What is and is not taken from the Qualcomm package +# +# Every HTP generation has its own skel and stub, 16 MB a pair. `QNN_HTP_ARCHS` +# names the ones to ship; the default is the generations in devices sold since +# the 8 Gen 2 (V73 — the MagicPad 2's 8s Gen 3 is one), through the 8 Elite +# (V79) and its successor (V81). V66–V69 are 2020–2021 silicon and are left +# out, which is 50 MB the tablet never loads. `libQnnHtpPrepare.so`, the +# on-device graph compiler, is 84 MB and cannot be left out: it is what turns +# an int8 ONNX graph into something the DSP runs, once per device (§5). +# +# `libQnnGpu.so` and `libQnnDsp*.so` are not taken: the Adreno rung measured +# slower than the Hexagon everywhere the Hexagon exists (§1.1), and the cDSP +# path is the pre-HTP generation. +# +# ## Why a script and not a checked-in copy +# +# 150 MB of vendor binaries in LFS, re-fetched by every clone, for files that +# Maven serves with checksums. The cache directory keeps them across builds; +# CI's is the target cache it already mounts. +set -euo pipefail + +ORT_VERSION="${ORT_VERSION:-1.29.0}" +QNN_VERSION="${QNN_VERSION:-2.42.0}" +QNN_HTP_ARCHS="${QNN_HTP_ARCHS:-73 75 79 81}" +DEST="${1:-${PWD}/target-android/runtime}" +MAVEN="https://repo1.maven.org/maven2" + +fetch() { + # A Maven artefact and its SHA-1, verified before anything is unpacked. + local group="$1" artefact="$2" version="$3" out="$4" + local url="${MAVEN}/${group//.//}/${artefact}/${version}/${artefact}-${version}.aar" + if [[ -f "${out}" && -f "${out}.sha1" ]] \ + && [[ "$(sha1sum "${out}" | cut -d' ' -f1)" == "$(cut -c1-40 "${out}.sha1")" ]]; then + return + fi + echo "==> fetching ${artefact} ${version}" + curl -fsSL -o "${out}.sha1" "${url}.sha1" + curl -fsSL -o "${out}" "${url}" + [[ "$(sha1sum "${out}" | cut -d' ' -f1)" == "$(cut -c1-40 "${out}.sha1")" ]] || { + echo "error: ${artefact}-${version}.aar does not match its published SHA-1" >&2 + rm -f "${out}" + exit 1 + } +} + +mkdir -p "${DEST}/aar" "${DEST}/lib" +fetch com.microsoft.onnxruntime onnxruntime-android-qnn "${ORT_VERSION}" \ + "${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar" +fetch com.qualcomm.qti qnn-runtime "${QNN_VERSION}" \ + "${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar" + +# The ONNX Runtime POM names the QNN version it was built against; a pair +# that disagrees loads and then fails at the first graph, which is the kind +# of failure the probe would only report as "Hexagon failed". +pom="${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.pom" +[[ -f "${pom}" ]] || curl -fsSL -o "${pom}" \ + "${MAVEN}/com/microsoft/onnxruntime/onnxruntime-android-qnn/${ORT_VERSION}/onnxruntime-android-qnn-${ORT_VERSION}.pom" +wanted="$(sed -n '/qnn-runtime<\/artifactId>/{n;s/.*\(.*\)<\/version>.*/\1/p}' "${pom}")" +if [[ -n "${wanted}" && "${wanted}" != "${QNN_VERSION}" ]]; then + echo "error: ONNX Runtime ${ORT_VERSION} was built against QNN ${wanted}, not ${QNN_VERSION}" >&2 + exit 1 +fi + +rm -rf "${DEST}/lib" +mkdir -p "${DEST}/lib" +unzip -q -o -j "${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar" \ + 'jni/arm64-v8a/libonnxruntime.so' -d "${DEST}/lib" +members=(jni/arm64-v8a/libQnnHtp.so jni/arm64-v8a/libQnnHtpPrepare.so jni/arm64-v8a/libQnnSystem.so) +for arch in ${QNN_HTP_ARCHS}; do + members+=("jni/arm64-v8a/libQnnHtpV${arch}Skel.so" "jni/arm64-v8a/libQnnHtpV${arch}Stub.so") +done +unzip -q -o -j "${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar" "${members[@]}" -d "${DEST}/lib" +unzip -q -o -j "${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar" 'LICENSE.pdf' 'NOTICE.txt' -d "${DEST}" 2>/dev/null || true +mv -f "${DEST}/LICENSE.pdf" "${DEST}/QNN-LICENSE.pdf" 2>/dev/null || true +mv -f "${DEST}/NOTICE.txt" "${DEST}/QNN-NOTICE.txt" 2>/dev/null || true + +echo "==> runtime in ${DEST}/lib:" +du -sh "${DEST}/lib" | cut -f1 | sed 's/^/ /' +ls "${DEST}/lib" | sed 's/^/ /' diff --git a/ui/dr-ui/Cargo.toml b/ui/dr-ui/Cargo.toml index 6cf1407..f048684 100644 --- a/ui/dr-ui/Cargo.toml +++ b/ui/dr-ui/Cargo.toml @@ -49,6 +49,9 @@ dr-catalog.workspace = true # The face pipeline, with the ONNX runtime: this is the layer that actually # runs the models over the library (docs/faces.md). dr-face = { workspace = true, features = ["inference"] } +# The engine behind both. `native` here means the *app* may look for a +# runtime file; the build stays C-free either way (docs/inference.md §3). +dr-inference-engine = { workspace = true, features = ["native"] } dr-thumbs.workspace = true # The library module writes scan results straight into the catalog, so it # needs the same SQLite types dr-catalog exposes. diff --git a/ui/dr-ui/src/identity_ui.rs b/ui/dr-ui/src/identity_ui.rs index 2afdc70..73ef482 100644 --- a/ui/dr-ui/src/identity_ui.rs +++ b/ui/dr-ui/src/identity_ui.rs @@ -27,7 +27,7 @@ use crate::{AppWindow, IdentityFace, IdentityPerson}; /// use rather than captured once, because the page can change it while the /// screen is open. pub fn model_id(settings: &crate::settings_ui::SettingsController) -> String { - settings.snapshot().faces.detector.model_id().to_string() + crate::inference::model_id(settings.snapshot().faces.detector).to_string() } /// Screen state that outlives a single callback. diff --git a/ui/dr-ui/src/inference.rs b/ui/dr-ui/src/inference.rs new file mode 100644 index 0000000..ffb021c --- /dev/null +++ b/ui/dr-ui/src/inference.rs @@ -0,0 +1,91 @@ +//! The app's side of `dr-inference-engine` (docs/inference.md §8). +//! +//! What lives here is what only the app knows: where the runtime file might +//! be, where the disposable cache goes, which model files this device has, +//! and how the engine's status becomes a line on the settings page. What +//! runs the models does not. + +use std::path::PathBuf; + +use dr_inference_engine::{Form, Role, Status}; +use dr_types::FaceDetector; + +/// Start the engine: choose the runtime, probe in the background, compile +/// engines for whatever this device turns out to have. +/// +/// `runtime_dirs` is where the platform put `libonnxruntime`: an empty list +/// is the tract build. Called once, after the models are on disk — on +/// Android that is the end of `install_bundled_models`, since the probe +/// fingerprints the model files and a probe before they land would be a +/// probe of nothing. +pub fn init(runtime_dirs: Vec) { + let dir = crate::library::shared_face_models_dir(); + let mut models: Vec<(Role, PathBuf)> = FaceDetector::ALL + .iter() + .map(|d| (Role::Detector, dir.join(d.file_name()))) + .collect(); + models.push((Role::Embedder, dir.join("arcface_mbf_b1.onnx"))); + models.push((Role::Scene, dir.join("yolo26s-sem-ade20k.onnx"))); + models.push((Role::Landmarks, dir.join(crate::library::LANDMARK_MODEL))); + models.push((Role::EyeClassifier, dir.join(crate::library::EYE_MODEL))); + models.push(( + Role::EyeClassifier, + dir.join(crate::library::SUNGLASSES_MODEL), + )); + models.retain(|(_, p)| p.is_file()); + + dr_inference_engine::init(dr_inference_engine::Config { + runtime_dirs, + cache_dir: crate::library::inference_cache_dir(), + models, + embedded: vec![(Role::Segmenter, dr_segment::embedded_model_bytes())], + ceiling: None, + threads: 0, + decay: std::time::Duration::ZERO, + }); + + // A low-memory signal drops every session nobody is mid-run with; the + // next use loads again. Same tier as the GPU caches: rebuilt from data + // the process still holds, and on a mobile GPU or NPU the largest pool. + crate::memory::evict_at(crate::memory::Tier::Gpu, dr_inference_engine::release_all); +} + +/// Which form the current backend loads `detector` in, given the files on +/// this device — the fact `faces.model_id` has to carry (§7). +/// +/// Reads the shared directory only. An account-private model directory can +/// override the file `library::face_models` loads, but not which form the +/// backend wants, and the int8 sibling is something a packager ships, not +/// something a user drops in. +pub fn detector_form(detector: FaceDetector) -> Form { + let canonical = crate::library::shared_face_models_dir().join(detector.file_name()); + dr_inference_engine::resolve_model(Role::Detector, &canonical).1 +} + +/// The `faces.model_id` this device indexes under with `detector`. +pub fn model_id(detector: FaceDetector) -> &'static str { + match detector_form(detector) { + Form::F32 => detector.model_id(), + Form::Int8 => detector.model_id_int8(), + } +} + +/// The two lines the About panel shows: what is running the models, and +/// why or how far along. +pub fn about_lines() -> (String, String) { + let status: Status = dr_inference_engine::status(); + let line = status.line(); + let detail = if status.probing { + "Checking what this device can run the models on…".to_string() + } else if status.engines.1 > 0 && status.engines.0 < status.engines.1 { + format!( + "Preparing {} engines · {} of {}", + status.rung.label(), + status.engines.0, + status.engines.1 + ) + } else { + status.reason + }; + (line, detail) +} diff --git a/ui/dr-ui/src/lib.rs b/ui/dr-ui/src/lib.rs index 6891f76..5d9ae7a 100644 --- a/ui/dr-ui/src/lib.rs +++ b/ui/dr-ui/src/lib.rs @@ -39,6 +39,7 @@ pub mod identity; mod identity_ui; mod import; mod import_ui; +pub mod inference; mod labels; mod library; mod library_ui; @@ -1209,6 +1210,32 @@ pub fn run(paths: Vec) -> Result<()> { None => window.set_backend("NO GPU".into()), } + // TRACES: FR-INF-1 + // What the models run on. Re-read every two seconds because the answer + // changes twice after launch — when the probe reports and as each + // engine lands — and the page is open for longer than either takes. + { + let set = |w: &AppWindow| { + let (line, detail) = inference::about_lines(); + w.set_inference_backend(line.into()); + w.set_inference_detail(detail.into()); + }; + set(&window); + let weak = window.as_weak(); + let timer = Rc::new(slint::Timer::default()); + let held = timer.clone(); + timer.start( + slint::TimerMode::Repeated, + std::time::Duration::from_secs(2), + move || { + let _keep = &held; + if let Some(w) = weak.upgrade() { + set(&w); + } + }, + ); + } + // TRACES: NFR-OPS-1 // The diagnostics bundle, wired as the two presses the requirement // describes. Preparing gathers the log and the crash records into memory @@ -1638,7 +1665,7 @@ pub fn run(paths: Vec) -> Result<()> { library.set_fetch_ahead(stored.cache.fetch_ahead); library.set_write_xmp_sidecars(stored.library.write_xmp_sidecars); library.set_timeline_bars(stored.library.timeline_bars); - library.set_face_model_id(stored.faces.detector.model_id()); + library.set_face_model_id(inference::model_id(stored.faces.detector)); } // --- the export folder picker ---------------------------------------- @@ -1790,7 +1817,7 @@ pub fn run(paths: Vec) -> Result<()> { // file is installed, and how much of the library that // pipeline has covered — which for a freshly chosen one is // nothing, and saying so is the point. - lib.set_face_model_id(s.faces.detector.model_id()); + lib.set_face_model_id(inference::model_id(s.faces.detector)); if let Some(w) = weak.upgrade() { refresh_face_status(&w, &lib, s.faces.detector); } @@ -3933,7 +3960,7 @@ fn refresh_face_status( window, &library.catalog(), store.as_ref(), - detector.model_id(), + inference::model_id(detector), models.as_ref().is_some_and(|m| m.eyes.is_some()), ); window.set_identity_model_missing(models.is_none()); diff --git a/ui/dr-ui/src/library.rs b/ui/dr-ui/src/library.rs index 755579f..b232847 100644 --- a/ui/dr-ui/src/library.rs +++ b/ui/dr-ui/src/library.rs @@ -5191,6 +5191,13 @@ impl FaceModelPaths { } } +/// Where the inference engine keeps what it derives per device: the probe +/// result and compiled engines (docs/inference.md §4, §5). A peer of +/// `thumbs`, not of the catalog: disposable, regenerable, never synced. +pub fn inference_cache_dir() -> PathBuf { + data_root().join("inference") +} + /// The detector and embedder files, if both are present — and the eye /// models beside them, if those are. /// diff --git a/ui/dr-ui/src/library_ui.rs b/ui/dr-ui/src/library_ui.rs index 71bcd75..b01abd6 100644 --- a/ui/dr-ui/src/library_ui.rs +++ b/ui/dr-ui/src/library_ui.rs @@ -484,7 +484,9 @@ impl LibraryController { dr_types::LibrarySettings::default().write_xmp_sidecars, ), timeline_bars: std::cell::Cell::new(dr_types::LibrarySettings::default().timeline_bars), - face_model_id: RefCell::new(dr_types::FaceDetector::default().model_id().to_string()), + face_model_id: RefCell::new( + crate::inference::model_id(dr_types::FaceDetector::default()).to_string(), + ), }) } diff --git a/ui/dr-ui/ui/app.slint b/ui/dr-ui/ui/app.slint index d389d80..cb0805c 100644 --- a/ui/dr-ui/ui/app.slint +++ b/ui/dr-ui/ui/app.slint @@ -58,6 +58,8 @@ export component AppWindow inherits Window { in property canvas; in property adapter: "detecting…"; in property backend: "—"; + in property inference-backend: "detecting…"; + in property inference-detail: ""; in property fps: 0; /// TRACES: FR-DSP-8 /// The display showing the canvas and the colour transform it is getting. @@ -1322,6 +1324,8 @@ in property panel-visible: true; adapter: root.adapter; backend: root.backend; + inference-backend: root.inference-backend; + inference-detail: root.inference-detail; fps: root.fps; layout-class: root.layout-class; app-version: root.app-version; diff --git a/ui/dr-ui/ui/settings.slint b/ui/dr-ui/ui/settings.slint index 886bc5e..110f348 100644 --- a/ui/dr-ui/ui/settings.slint +++ b/ui/dr-ui/ui/settings.slint @@ -137,6 +137,10 @@ export component SettingsPage inherits Rectangle { /// something up, not where they are looked at all day. in property adapter; in property backend; + /// What runs the neural models and how it was chosen — the two lines + /// `dr_ui::inference::about_lines` produces (docs/inference.md §4). + in property inference-backend; + in property inference-detail; in property fps; in property layout-class; in property app-version; @@ -872,6 +876,26 @@ export component SettingsPage inherits Rectangle { } } + // TRACES: FR-INF-1 + // The backend the models run on, and why. Beside + // Graphics because it is the same kind of fact: a + // property of this device, chosen by measurement, + // that a bug report about a slow index should quote. + HorizontalLayout { + spacing: Theme.gap; + Label { text: "Inference"; } + Value { + text: root.inference-backend; + horizontal-stretch: 1; + overflow: elide; + } + } + + if root.inference-detail != "": Caption { + text: root.inference-detail; + wrap: word-wrap; + } + HorizontalLayout { spacing: Theme.gap; Label { text: "Frame rate"; }