Files
DarkRoom/core/dr-inference-engine/src/engines.rs
T
dtourolle 56f4180347 Run a denoiser of any input size on TensorRT and CUDA
Role::WholeDenoiser is the denoise network exported with any height and
width, for a whole frame instead of 1408 tiles whose borders are thrown
away. It is served only where a new size costs nothing: the CUDA
provider, and TensorRT through an optimisation profile from 256 to
4608 x 6656, tuned for the 6D's frame with Best's border. Everywhere
else whole_frame_limit() says None and the fixed tiles run.

ort's TensorRT builder has no profile options, so the engine registers
through the runtime's V2 options with the names 1.30 reads
(trt_profile_{min,opt,max}_shapes). Without a profile a dynamic input
compiled an engine per size at run time, 156 s on the first frame. The
engine lives in its own directory per model: ORT's cache key leaves the
shape out, and the fixed 1408 export and its any-size sibling are the
same graph.
2026-10-06 21:41:31 -04:00

184 lines
6.6 KiB
Rust

//! Compiled engines: what a rung builds once per device, and the thread that
//! builds them before anyone asks (docs/dev/inference.md §5, §6).
//!
//! TensorRT keeps its own engine cache keyed by graph hash; QNN writes a
//! context model. Both are opaque to this crate, which tracks only *that* a
//! model compiled — by the hash of its bytes — so [`crate::open`] can tell a
//! request whether to expect the rung or its fallback.
use std::path::PathBuf;
use crate::{state, Config, Rung};
enum Source {
File(PathBuf),
Bytes(&'static [u8]),
}
/// 64-bit FNV-1a. A cache key, not a checksum: two model files that collide
/// here would have to also be the same size and the same role, and the cost
/// of that is a rebuilt engine.
pub fn hash(bytes: &[u8]) -> u64 {
let mut h = 0xcbf2_9ce4_8422_2325u64;
for &b in bytes {
h ^= b as u64;
h = h.wrapping_mul(0x0000_0100_0000_01b3);
}
h
}
/// The cache entry for `bytes` compiled on `rung`.
pub fn key(rung: Rung, bytes: &[u8]) -> String {
key_of(rung, hash(bytes))
}
/// The same, from a hash already taken.
pub fn key_of(rung: Rung, hash: u64) -> String {
format!("{}:{:016x}", rung.label(), hash)
}
/// Where QNN's compiled context for `bytes` lives.
pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf {
cfg.cache_dir
.join("qnn")
.join(format!("{:016x}_ctx.onnx", hash(bytes)))
}
/// Where CoreML compiles `bytes` to: one directory per model, because
/// CoreML's own cache key leaves out the weights of a model loaded from
/// memory (`session::coreml`), and one per runtime version, which wrote it.
pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
model_dir(cfg, "coreml", bytes)
}
/// Where OpenVINO compiles `bytes` to, at one precision. OpenVINO hashes
/// the model it is given, weights included, but a key that leaves out
/// what is being varied has cost a day before (CLAUDE.md, "Providers"),
/// and a directory per model and precision costs nothing: the precision
/// is a compile option, and the two forms are different programs.
pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
model_dir(
cfg,
if fp16 {
"openvino/fp16"
} else {
"openvino/f32"
},
bytes,
)
}
/// Where TensorRT keeps the engine for a whole-frame model. Its own
/// directory per model: ONNX Runtime's engine cache key leaves the input
/// shape out, and served one export's engine to another of the same graph
/// with a different shape when the denoiser was first cut into pieces
/// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are
/// exactly that pair.
pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
model_dir(cfg, "tensorrt-whole", bytes)
}
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
/// model, and one per runtime version, which wrote it.
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
let runtime = match crate::api::runtime() {
crate::Runtime::OnnxRuntime { version, .. } => version,
crate::Runtime::Tract => "tract".into(),
};
cfg.cache_dir
.join(provider)
.join(runtime)
.join(format!("{:016x}", hash(bytes)))
}
/// After the probe: compile every configured model the selected rung can
/// take, smallest first, recording each as it lands.
pub fn run() {
let (rung, cfg) = {
let s = state().lock().unwrap();
(crate::current_rung(&s), s.config.clone())
};
if !rung.compiles() {
return;
}
// Smallest first, so the detector — the one that runs per image — is
// ready soonest (§6 step 3).
let mut jobs: Vec<(crate::Role, Source, u64)> = cfg
.models
.iter()
.filter(|(role, _)| rung.serves(*role))
.filter_map(|(role, path)| {
let (path, form) = crate::resolve_model(*role, path);
(form == rung.form(*role)).then(|| {
let size = std::fs::metadata(&path).map(|m| m.len()).unwrap_or(0);
(*role, Source::File(path), size)
})
})
.chain(cfg.embedded.iter().filter_map(|(role, form, bytes)| {
// The embedded form the rung wants, if the build carries it;
// a build without it runs that model on the rung's fallback.
(rung.serves(*role) && rung.form(*role) == *form).then_some((
*role,
Source::Bytes(bytes),
bytes.len() as u64,
))
}))
.collect();
jobs.sort_by_key(|j| j.2);
state().lock().unwrap().wanted = jobs.len();
for (role, source, _) in jobs {
let (bytes, name) = match &source {
Source::File(path) => match std::fs::read(path) {
Ok(b) => (b, path.display().to_string()),
Err(_) => continue,
},
Source::Bytes(b) => (b.to_vec(), format!("embedded {role:?}")),
};
let key = key(rung, &bytes);
{
let s = state().lock().unwrap();
if s.cache.compiled.contains(&key) || s.cache.refused.contains(&key) {
continue;
}
}
log::info!("inference: compiling {name} for {}", rung.label());
let started = std::time::Instant::now();
let built = match crate::probe::attempt(&cfg, &key, || {
crate::session::build(rung, role, &bytes, &cfg)
}) {
Ok(built) => built,
Err(_) => {
// Refused: the process died inside this compile before.
let mut s = state().lock().unwrap();
s.cache.refused.insert(key);
crate::probe::write_cache(&s.config, &s.cache);
continue;
}
};
match built {
Ok(session) => {
drop(session);
let mut s = state().lock().unwrap();
s.cache.compiled.insert(key);
crate::probe::write_cache(&s.config, &s.cache);
log::info!(
"inference: {name} ready on {} in {:.1} s",
rung.label(),
started.elapsed().as_secs_f64()
);
}
Err(e) => {
// This model stays on the fallback; the others still get
// their engine. A corrected model file changes the hash and
// is retried.
log::warn!(
"inference: {name} will not compile for {}: {e}",
rung.label()
);
}
}
}
}