diff --git a/core/dr-inference-engine/src/engines.rs b/core/dr-inference-engine/src/engines.rs index 6539298..f9197db 100644 --- a/core/dr-inference-engine/src/engines.rs +++ b/core/dr-inference-engine/src/engines.rs @@ -68,6 +68,16 @@ pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf { ) } +/// Where TensorRT keeps the engine for a whole-frame model. Its own +/// directory per model: ONNX Runtime's engine cache key leaves the input +/// shape out, and served one export's engine to another of the same graph +/// with a different shape when the denoiser was first cut into pieces +/// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are +/// exactly that pair. +pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf { + model_dir(cfg, "tensorrt-whole", bytes) +} + /// `///`: one per /// model, and one per runtime version, which wrote it. fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf { diff --git a/core/dr-inference-engine/src/lib.rs b/core/dr-inference-engine/src/lib.rs index f91ff23..f04cb99 100644 --- a/core/dr-inference-engine/src/lib.rs +++ b/core/dr-inference-engine/src/lib.rs @@ -53,6 +53,31 @@ pub enum Role { /// levels cannot hold the shadow steps it exists to recover — so the /// Hexagon takes it with 16-bit activations and weights (§1.5). Denoiser, + /// The same denoise networks exported with any height and width, run + /// over a whole frame — or the fewest large tiles that fit — instead of + /// 1408² tiles whose borders are thrown away (docs/dev/denoise.md §14). + /// Served only where a size the graph was not compiled for costs + /// nothing: TensorRT, through an optimisation profile up to + /// [`WHOLE_FRAME_MAX`], and the CUDA provider. Everywhere else the + /// fixed-tile [`Role::Denoiser`] runs; see [`whole_frame_limit`]. + WholeDenoiser, +} + +/// The largest input, rows × columns, a [`Role::WholeDenoiser`] session +/// takes: TensorRT's optimisation profile is built up to it, and the tiler +/// cuts a larger frame into tiles no bigger. A 20 MP 6D frame with Best's +/// reflected border is 4160 × 5984. +pub const WHOLE_FRAME_MAX: (usize, usize) = (4608, 6656); + +/// The input size TensorRT tunes a whole-frame engine for: the 6D's frame +/// with Best's border, the frame the reference measurements are of. +pub const WHOLE_FRAME_OPT: (usize, usize) = (4160, 5984); + +/// Whether the selected rung runs [`Role::WholeDenoiser`], and if so the +/// largest input it takes. `None` means run the fixed tiles. +pub fn whole_frame_limit() -> Option<(usize, usize)> { + let rung = current_rung(&state().lock().unwrap()); + rung.serves(Role::WholeDenoiser).then_some(WHOLE_FRAME_MAX) } /// Which numeric form of a model a session was built from. @@ -177,7 +202,8 @@ impl Rung { Role::Keypoints => Form::Int8, Role::Detector | Role::Landmarks => Form::A16W8, Role::Segmenter | Role::Scene | Role::Inpainter | Role::Denoiser => Form::A16W16, - Role::Embedder | Role::EyeClassifier => Form::F32, + // Not served there at all: the Hexagon takes fixed shapes. + Role::Embedder | Role::EyeClassifier | Role::WholeDenoiser => Form::F32, }, _ => Form::F32, } @@ -193,6 +219,13 @@ impl Rung { /// the Neural Engine is fp16, and which unit runs a graph is CoreML's /// choice. fn serves(self, role: Role) -> bool { + // Any input size only where a new size costs nothing. MIGraphX, + // OpenVINO and CoreML compile per shape, the Hexagon takes fixed + // shapes only, and the CPU could but would hold gigabytes of f32 + // activations for a whole frame of Best. + if role == Role::WholeDenoiser { + return matches!(self, Rung::TensorRt | Rung::Cuda); + } match self { Rung::Hexagon => !matches!(role, Role::Embedder | Role::EyeClassifier), Rung::CoreMl => role != Role::Embedder, diff --git a/core/dr-inference-engine/src/session.rs b/core/dr-inference-engine/src/session.rs index 1b09930..dcd4271 100644 --- a/core/dr-inference-engine/src/session.rs +++ b/core/dr-inference-engine/src/session.rs @@ -58,6 +58,9 @@ fn build_with( // write, or the directory CoreML or OpenVINO compiles into. let per_model = match rung { Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)), + Rung::TensorRt if role == Role::WholeDenoiser => { + Some(crate::engines::tensorrt_whole_dir(cfg, bytes)) + } Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))), _ if ready => None, _ => context.clone(), @@ -135,6 +138,11 @@ fn providers( Rung::Cuda => { Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?) } + Rung::TensorRt if role == Role::WholeDenoiser => { + let mut b = b; + tensorrt_whole(&mut b, per_model.expect("a whole-frame engine directory"))?; + Ok(b.with_execution_providers([ep::CUDA::default().build()])?) + } Rung::TensorRt => { let cache = cfg.cache_dir.join("tensorrt"); let _ = std::fs::create_dir_all(&cache); @@ -251,6 +259,72 @@ fn migraphx( ) } +/// TensorRT for a whole-frame model: one engine for every input size up to +/// [`crate::WHOLE_FRAME_MAX`], kept in its own directory. +/// +/// `ort`'s builder has no profile options, so this registers through the +/// runtime's TensorRT V2 options, with the names 1.30 reads +/// (`tensorrt_execution_provider_info.cc`): `trt_profile_{min,opt,max}_shapes`. +/// Without a profile a dynamic input compiles a new engine per size at run +/// time — 156 s on the first frame, measured — so the profile is the +/// difference between a whole-frame engine and a stall. fp16, as for every +/// role but the embedder (§7); the denoiser measured 0.00 dB from f32. +#[cfg(not(target_os = "android"))] +fn tensorrt_whole( + b: &mut ort::session::builder::SessionBuilder, + cache: &std::path::Path, +) -> ort::Result<()> { + use ort::AsPointer; + use std::ffi::CString; + let _ = std::fs::create_dir_all(cache); + let shapes = |(h, w): (usize, usize)| format!("mosaic:1x1x{h}x{w},sigma:1x1x{h}x{w}"); + let dir = cache.to_string_lossy().into_owned(); + let options = [ + ("trt_fp16_enable", "1".to_string()), + ("trt_engine_cache_enable", "1".to_string()), + ("trt_engine_cache_path", dir.clone()), + ("trt_timing_cache_enable", "1".to_string()), + ("trt_timing_cache_path", dir), + ("trt_max_workspace_size", (1u64 << 30).to_string()), + ("trt_profile_min_shapes", shapes((256, 256))), + ("trt_profile_opt_shapes", shapes(crate::WHOLE_FRAME_OPT)), + ("trt_profile_max_shapes", shapes(crate::WHOLE_FRAME_MAX)), + ]; + let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string())); + let keys = options + .iter() + .map(|(k, _)| cstr(k)) + .collect::>>()?; + let values = options + .iter() + .map(|(_, v)| cstr(v)) + .collect::>>()?; + let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect(); + let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect(); + let api = ort::api(); + // SAFETY: the documented create / update / append / release sequence + // `ort`'s own TensorRT builder makes, over arrays that outlive it; the + // runtime copies the options into the session before the release. + unsafe { + let mut trt: *mut ort::sys::OrtTensorRTProviderOptionsV2 = std::ptr::null_mut(); + ort::Error::result_from_status((api.CreateTensorRTProviderOptions)(&mut trt))?; + let result = ort::Error::result_from_status((api.UpdateTensorRTProviderOptions)( + trt, + key_ptrs.as_ptr(), + value_ptrs.as_ptr(), + keys.len(), + )) + .and_then(|()| { + ort::Error::result_from_status((api.SessionOptionsAppendExecutionProvider_TensorRT_V2)( + b.ptr_mut(), + trt, + )) + }); + (api.ReleaseTensorRTProviderOptions)(trt); + result + } +} + /// OpenVINO on the GPU, compiling into `cache`. /// /// The option names are those `openvino_provider_factory.cc` reads at 1.24,