//! The app's side of `dr-inference-engine` (docs/dev/inference.md §8). //! //! What lives here is what only the app knows: where the runtime file might //! be, where the disposable cache goes, which model files this device has, //! and how the engine's status becomes a line on the settings page. What //! runs the models does not. use std::path::PathBuf; use dr_inference_engine::{Form, Role, Status}; use dr_types::FaceDetector; /// Start the engine: choose the runtime, probe in the background, compile /// engines for whatever this device turns out to have. /// /// `runtime_dirs` is where the platform put `libonnxruntime`: an empty list /// is the tract build. Called once, after the models are on disk — on /// Android that is the end of `install_bundled_models`, since the probe /// fingerprints the model files and a probe before they land would be a /// probe of nothing. pub fn init(runtime_dirs: Vec) { // Each file where the app will actually load it from — the user's // shared directory, else the package's — so a fresh install with models // only under `/usr/share` probes and compiles for them rather than // finding nothing and settling on the CPU. let mut wanted: Vec<(Role, &str)> = FaceDetector::ALL .iter() .map(|d| (Role::Detector, d.file_name())) .collect(); wanted.extend([ (Role::Embedder, "arcface_mbf_b1.onnx"), (Role::Scene, "yolo26s-sem-ade20k.onnx"), (Role::Landmarks, crate::library::LANDMARK_MODEL), (Role::EyeClassifier, crate::library::EYE_MODEL), (Role::EyeClassifier, crate::library::SUNGLASSES_MODEL), (Role::Inpainter, crate::library::INPAINT_MODEL), ]); let models: Vec<(Role, PathBuf)> = wanted .into_iter() .filter_map(|(role, name)| Some((role, crate::library::shared_model(name)?))) .collect(); dr_inference_engine::init(dr_inference_engine::Config { runtime_dirs, cache_dir: crate::library::inference_cache_dir(), models, embedded: { let [landscape, portrait] = dr_pano::xfeat::embedded_model_bytes(); vec![ (Role::Segmenter, dr_segment::embedded_model_bytes()), (Role::Keypoints, landscape), (Role::Keypoints, portrait), ] }, ceiling: None, threads: 0, decay: std::time::Duration::ZERO, }); // A low-memory signal drops every session nobody is mid-run with; the // next use loads again. Same tier as the GPU caches: rebuilt from data // the process still holds, and on a mobile GPU or NPU the largest pool. crate::memory::evict_at(crate::memory::Tier::Gpu, dr_inference_engine::release_all); } /// Where a person can put a runtime by hand: `runtime/` beside the models, /// searched before any system library. The system copy on the reference /// desktop is built without TensorRT and against the wrong cuDNN, and a /// working one is four files from the `onnxruntime-gpu` wheel; this is /// where they go, and `tools/fetch-desktop-runtime.sh` puts them there. pub fn user_runtime_dir() -> PathBuf { crate::library::shared_face_models_dir() .parent() .map(|p| p.join("runtime")) .unwrap_or_else(|| PathBuf::from("runtime")) } /// Which form the current backend loads `detector` in, given the files on /// this device — the fact `faces.model_id` has to carry (§7). /// /// Reads the shared and system directories only. An account-private model /// directory can override the file `library::face_models` loads, but not /// which form the backend wants, and the int8 sibling is something a /// packager ships, not something a user drops in. pub fn detector_form(detector: FaceDetector) -> Form { let canonical = crate::library::shared_model(detector.file_name()) .unwrap_or_else(|| crate::library::shared_face_models_dir().join(detector.file_name())); dr_inference_engine::resolve_model(Role::Detector, &canonical).1 } /// The `faces.model_id` this device indexes under with `detector`. pub fn model_id(detector: FaceDetector) -> &'static str { match detector_form(detector) { Form::F32 => detector.model_id(), Form::Int8 => detector.model_id_int8(), } } /// The two lines the About panel shows: what is running the models, and /// why or how far along. pub fn about_lines() -> (String, String) { let status: Status = dr_inference_engine::status(); let line = status.line(); let detail = if status.probing { "Checking what this device can run the models on…".to_string() } else if status.engines.1 > 0 && status.engines.0 < status.engines.1 { format!( "Preparing {} engines · {} of {}", status.rung.label(), status.engines.0, status.engines.1 ) } else if status.failed.is_empty() { status.reason } else { // Every rung that was tried and why it lost, not only the first: // "TensorRT: not enabled in this build" says nothing about why CUDA // was not taken instead. status .failed .iter() .map(|(rung, why)| format!("{}: {why}", rung.label())) .collect::>() .join(" · ") }; (line, detail) }