Role::WholeDenoiser is the denoise network exported with any height and
width, for a whole frame instead of 1408 tiles whose borders are thrown
away. It is served only where a new size costs nothing: the CUDA
provider, and TensorRT through an optimisation profile from 256 to
4608 x 6656, tuned for the 6D's frame with Best's border. Everywhere
else whole_frame_limit() says None and the fixed tiles run.
ort's TensorRT builder has no profile options, so the engine registers
through the runtime's V2 options with the names 1.30 reads
(trt_profile_{min,opt,max}_shapes). Without a profile a dynamic input
compiled an engine per size at run time, 156 s on the first frame. The
engine lives in its own directory per model: ORT's cache key leaves the
shape out, and the fixed 1408 export and its any-size sibling are the
same graph.
184 lines
6.6 KiB
Rust
184 lines
6.6 KiB
Rust
//! Compiled engines: what a rung builds once per device, and the thread that
|
|
//! builds them before anyone asks (docs/dev/inference.md §5, §6).
|
|
//!
|
|
//! TensorRT keeps its own engine cache keyed by graph hash; QNN writes a
|
|
//! context model. Both are opaque to this crate, which tracks only *that* a
|
|
//! model compiled — by the hash of its bytes — so [`crate::open`] can tell a
|
|
//! request whether to expect the rung or its fallback.
|
|
|
|
use std::path::PathBuf;
|
|
|
|
use crate::{state, Config, Rung};
|
|
|
|
enum Source {
|
|
File(PathBuf),
|
|
Bytes(&'static [u8]),
|
|
}
|
|
|
|
/// 64-bit FNV-1a. A cache key, not a checksum: two model files that collide
|
|
/// here would have to also be the same size and the same role, and the cost
|
|
/// of that is a rebuilt engine.
|
|
pub fn hash(bytes: &[u8]) -> u64 {
|
|
let mut h = 0xcbf2_9ce4_8422_2325u64;
|
|
for &b in bytes {
|
|
h ^= b as u64;
|
|
h = h.wrapping_mul(0x0000_0100_0000_01b3);
|
|
}
|
|
h
|
|
}
|
|
|
|
/// The cache entry for `bytes` compiled on `rung`.
|
|
pub fn key(rung: Rung, bytes: &[u8]) -> String {
|
|
key_of(rung, hash(bytes))
|
|
}
|
|
|
|
/// The same, from a hash already taken.
|
|
pub fn key_of(rung: Rung, hash: u64) -> String {
|
|
format!("{}:{:016x}", rung.label(), hash)
|
|
}
|
|
|
|
/// Where QNN's compiled context for `bytes` lives.
|
|
pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
|
cfg.cache_dir
|
|
.join("qnn")
|
|
.join(format!("{:016x}_ctx.onnx", hash(bytes)))
|
|
}
|
|
|
|
/// Where CoreML compiles `bytes` to: one directory per model, because
|
|
/// CoreML's own cache key leaves out the weights of a model loaded from
|
|
/// memory (`session::coreml`), and one per runtime version, which wrote it.
|
|
pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
|
model_dir(cfg, "coreml", bytes)
|
|
}
|
|
|
|
/// Where OpenVINO compiles `bytes` to, at one precision. OpenVINO hashes
|
|
/// the model it is given, weights included, but a key that leaves out
|
|
/// what is being varied has cost a day before (CLAUDE.md, "Providers"),
|
|
/// and a directory per model and precision costs nothing: the precision
|
|
/// is a compile option, and the two forms are different programs.
|
|
pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
|
|
model_dir(
|
|
cfg,
|
|
if fp16 {
|
|
"openvino/fp16"
|
|
} else {
|
|
"openvino/f32"
|
|
},
|
|
bytes,
|
|
)
|
|
}
|
|
|
|
/// Where TensorRT keeps the engine for a whole-frame model. Its own
|
|
/// directory per model: ONNX Runtime's engine cache key leaves the input
|
|
/// shape out, and served one export's engine to another of the same graph
|
|
/// with a different shape when the denoiser was first cut into pieces
|
|
/// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are
|
|
/// exactly that pair.
|
|
pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
|
model_dir(cfg, "tensorrt-whole", bytes)
|
|
}
|
|
|
|
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
|
|
/// model, and one per runtime version, which wrote it.
|
|
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
|
|
let runtime = match crate::api::runtime() {
|
|
crate::Runtime::OnnxRuntime { version, .. } => version,
|
|
crate::Runtime::Tract => "tract".into(),
|
|
};
|
|
cfg.cache_dir
|
|
.join(provider)
|
|
.join(runtime)
|
|
.join(format!("{:016x}", hash(bytes)))
|
|
}
|
|
|
|
/// After the probe: compile every configured model the selected rung can
|
|
/// take, smallest first, recording each as it lands.
|
|
pub fn run() {
|
|
let (rung, cfg) = {
|
|
let s = state().lock().unwrap();
|
|
(crate::current_rung(&s), s.config.clone())
|
|
};
|
|
if !rung.compiles() {
|
|
return;
|
|
}
|
|
|
|
// Smallest first, so the detector — the one that runs per image — is
|
|
// ready soonest (§6 step 3).
|
|
let mut jobs: Vec<(crate::Role, Source, u64)> = cfg
|
|
.models
|
|
.iter()
|
|
.filter(|(role, _)| rung.serves(*role))
|
|
.filter_map(|(role, path)| {
|
|
let (path, form) = crate::resolve_model(*role, path);
|
|
(form == rung.form(*role)).then(|| {
|
|
let size = std::fs::metadata(&path).map(|m| m.len()).unwrap_or(0);
|
|
(*role, Source::File(path), size)
|
|
})
|
|
})
|
|
.chain(cfg.embedded.iter().filter_map(|(role, form, bytes)| {
|
|
// The embedded form the rung wants, if the build carries it;
|
|
// a build without it runs that model on the rung's fallback.
|
|
(rung.serves(*role) && rung.form(*role) == *form).then_some((
|
|
*role,
|
|
Source::Bytes(bytes),
|
|
bytes.len() as u64,
|
|
))
|
|
}))
|
|
.collect();
|
|
jobs.sort_by_key(|j| j.2);
|
|
state().lock().unwrap().wanted = jobs.len();
|
|
|
|
for (role, source, _) in jobs {
|
|
let (bytes, name) = match &source {
|
|
Source::File(path) => match std::fs::read(path) {
|
|
Ok(b) => (b, path.display().to_string()),
|
|
Err(_) => continue,
|
|
},
|
|
Source::Bytes(b) => (b.to_vec(), format!("embedded {role:?}")),
|
|
};
|
|
let key = key(rung, &bytes);
|
|
{
|
|
let s = state().lock().unwrap();
|
|
if s.cache.compiled.contains(&key) || s.cache.refused.contains(&key) {
|
|
continue;
|
|
}
|
|
}
|
|
log::info!("inference: compiling {name} for {}", rung.label());
|
|
let started = std::time::Instant::now();
|
|
let built = match crate::probe::attempt(&cfg, &key, || {
|
|
crate::session::build(rung, role, &bytes, &cfg)
|
|
}) {
|
|
Ok(built) => built,
|
|
Err(_) => {
|
|
// Refused: the process died inside this compile before.
|
|
let mut s = state().lock().unwrap();
|
|
s.cache.refused.insert(key);
|
|
crate::probe::write_cache(&s.config, &s.cache);
|
|
continue;
|
|
}
|
|
};
|
|
match built {
|
|
Ok(session) => {
|
|
drop(session);
|
|
let mut s = state().lock().unwrap();
|
|
s.cache.compiled.insert(key);
|
|
crate::probe::write_cache(&s.config, &s.cache);
|
|
log::info!(
|
|
"inference: {name} ready on {} in {:.1} s",
|
|
rung.label(),
|
|
started.elapsed().as_secs_f64()
|
|
);
|
|
}
|
|
Err(e) => {
|
|
// This model stays on the fallback; the others still get
|
|
// their engine. A corrected model file changes the hash and
|
|
// is retried.
|
|
log::warn!(
|
|
"inference: {name} will not compile for {}: {e}",
|
|
rung.label()
|
|
);
|
|
}
|
|
}
|
|
}
|
|
}
|