//! Compiled engines: what a rung builds once per device, and the thread that //! builds them before anyone asks (docs/dev/inference.md §5, §6). //! //! TensorRT keeps its own engine cache keyed by graph hash; QNN writes a //! context model. Both are opaque to this crate, which tracks only *that* a //! model compiled — by the hash of its bytes — so [`crate::open`] can tell a //! request whether to expect the rung or its fallback. use std::path::PathBuf; use crate::{state, Config, Rung}; enum Source { File(PathBuf), Bytes(&'static [u8]), } /// 64-bit FNV-1a. A cache key, not a checksum: two model files that collide /// here would have to also be the same size and the same role, and the cost /// of that is a rebuilt engine. pub fn hash(bytes: &[u8]) -> u64 { let mut h = 0xcbf2_9ce4_8422_2325u64; for &b in bytes { h ^= b as u64; h = h.wrapping_mul(0x0000_0100_0000_01b3); } h } /// The cache entry for `bytes` compiled on `rung`. pub fn key(rung: Rung, bytes: &[u8]) -> String { key_of(rung, hash(bytes)) } /// The same, from a hash already taken. pub fn key_of(rung: Rung, hash: u64) -> String { format!("{}:{:016x}", rung.label(), hash) } /// Where QNN's compiled context for `bytes` lives. pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf { cfg.cache_dir .join("qnn") .join(format!("{:016x}_ctx.onnx", hash(bytes))) } /// Where CoreML compiles `bytes` to: one directory per model, because /// CoreML's own cache key leaves out the weights of a model loaded from /// memory (`session::coreml`), and one per runtime version, which wrote it. pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf { model_dir(cfg, "coreml", bytes) } /// Where OpenVINO compiles `bytes` to, at one precision. OpenVINO hashes /// the model it is given, weights included, but a key that leaves out /// what is being varied has cost a day before (CLAUDE.md, "Providers"), /// and a directory per model and precision costs nothing: the precision /// is a compile option, and the two forms are different programs. pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf { model_dir( cfg, if fp16 { "openvino/fp16" } else { "openvino/f32" }, bytes, ) } /// Where TensorRT keeps the engine for a whole-frame model. Its own /// directory per model: ONNX Runtime's engine cache key leaves the input /// shape out, and served one export's engine to another of the same graph /// with a different shape when the denoiser was first cut into pieces /// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are /// exactly that pair. pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf { model_dir(cfg, "tensorrt-whole", bytes) } /// `///`: one per /// model, and one per runtime version, which wrote it. fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf { let runtime = match crate::api::runtime() { crate::Runtime::OnnxRuntime { version, .. } => version, crate::Runtime::Tract => "tract".into(), }; cfg.cache_dir .join(provider) .join(runtime) .join(format!("{:016x}", hash(bytes))) } /// After the probe: compile every configured model the selected rung can /// take, smallest first, recording each as it lands. pub fn run() { let (rung, cfg) = { let s = state().lock().unwrap(); (crate::current_rung(&s), s.config.clone()) }; if !rung.compiles() { return; } // Smallest first, so the detector — the one that runs per image — is // ready soonest (§6 step 3). let mut jobs: Vec<(crate::Role, Source, u64)> = cfg .models .iter() .filter(|(role, _)| rung.serves(*role)) .filter_map(|(role, path)| { let (path, form) = crate::resolve_model(*role, path); (form == rung.form(*role)).then(|| { let size = std::fs::metadata(&path).map(|m| m.len()).unwrap_or(0); (*role, Source::File(path), size) }) }) .chain(cfg.embedded.iter().filter_map(|(role, form, bytes)| { // The embedded form the rung wants, if the build carries it; // a build without it runs that model on the rung's fallback. (rung.serves(*role) && rung.form(*role) == *form).then_some(( *role, Source::Bytes(bytes), bytes.len() as u64, )) })) .collect(); jobs.sort_by_key(|j| j.2); state().lock().unwrap().wanted = jobs.len(); for (role, source, _) in jobs { let (bytes, name) = match &source { Source::File(path) => match std::fs::read(path) { Ok(b) => (b, path.display().to_string()), Err(_) => continue, }, Source::Bytes(b) => (b.to_vec(), format!("embedded {role:?}")), }; let key = key(rung, &bytes); { let s = state().lock().unwrap(); if s.cache.compiled.contains(&key) || s.cache.refused.contains(&key) { continue; } } log::info!("inference: compiling {name} for {}", rung.label()); let started = std::time::Instant::now(); let built = match crate::probe::attempt(&cfg, &key, || { crate::session::build(rung, role, &bytes, &cfg) }) { Ok(built) => built, Err(_) => { // Refused: the process died inside this compile before. let mut s = state().lock().unwrap(); s.cache.refused.insert(key); crate::probe::write_cache(&s.config, &s.cache); continue; } }; match built { Ok(session) => { drop(session); let mut s = state().lock().unwrap(); s.cache.compiled.insert(key); crate::probe::write_cache(&s.config, &s.cache); log::info!( "inference: {name} ready on {} in {:.1} s", rung.label(), started.elapsed().as_secs_f64() ); } Err(e) => { // This model stays on the fallback; the others still get // their engine. A corrected model file changes the hash and // is retried. log::warn!( "inference: {name} will not compile for {}: {e}", rung.label() ); } } } }