Run a denoiser of any input size on TensorRT and CUDA

Role::WholeDenoiser is the denoise network exported with any height and
width, for a whole frame instead of 1408 tiles whose borders are thrown
away. It is served only where a new size costs nothing: the CUDA
provider, and TensorRT through an optimisation profile from 256 to
4608 x 6656, tuned for the 6D's frame with Best's border. Everywhere
else whole_frame_limit() says None and the fixed tiles run.

ort's TensorRT builder has no profile options, so the engine registers
through the runtime's V2 options with the names 1.30 reads
(trt_profile_{min,opt,max}_shapes). Without a profile a dynamic input
compiled an engine per size at run time, 156 s on the first frame. The
engine lives in its own directory per model: ORT's cache key leaves the
shape out, and the fixed 1408 export and its any-size sibling are the
same graph.
This commit is contained in:
2026-10-06 21:41:31 -04:00
parent 417cba8b4d
commit 56f4180347
3 changed files with 118 additions and 1 deletions
+10
View File
@@ -68,6 +68,16 @@ pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
) )
} }
/// Where TensorRT keeps the engine for a whole-frame model. Its own
/// directory per model: ONNX Runtime's engine cache key leaves the input
/// shape out, and served one export's engine to another of the same graph
/// with a different shape when the denoiser was first cut into pieces
/// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are
/// exactly that pair.
pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
model_dir(cfg, "tensorrt-whole", bytes)
}
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per /// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
/// model, and one per runtime version, which wrote it. /// model, and one per runtime version, which wrote it.
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf { fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
+34 -1
View File
@@ -53,6 +53,31 @@ pub enum Role {
/// levels cannot hold the shadow steps it exists to recover — so the /// levels cannot hold the shadow steps it exists to recover — so the
/// Hexagon takes it with 16-bit activations and weights (§1.5). /// Hexagon takes it with 16-bit activations and weights (§1.5).
Denoiser, Denoiser,
/// The same denoise networks exported with any height and width, run
/// over a whole frame — or the fewest large tiles that fit — instead of
/// 1408² tiles whose borders are thrown away (docs/dev/denoise.md §14).
/// Served only where a size the graph was not compiled for costs
/// nothing: TensorRT, through an optimisation profile up to
/// [`WHOLE_FRAME_MAX`], and the CUDA provider. Everywhere else the
/// fixed-tile [`Role::Denoiser`] runs; see [`whole_frame_limit`].
WholeDenoiser,
}
/// The largest input, rows × columns, a [`Role::WholeDenoiser`] session
/// takes: TensorRT's optimisation profile is built up to it, and the tiler
/// cuts a larger frame into tiles no bigger. A 20 MP 6D frame with Best's
/// reflected border is 4160 × 5984.
pub const WHOLE_FRAME_MAX: (usize, usize) = (4608, 6656);
/// The input size TensorRT tunes a whole-frame engine for: the 6D's frame
/// with Best's border, the frame the reference measurements are of.
pub const WHOLE_FRAME_OPT: (usize, usize) = (4160, 5984);
/// Whether the selected rung runs [`Role::WholeDenoiser`], and if so the
/// largest input it takes. `None` means run the fixed tiles.
pub fn whole_frame_limit() -> Option<(usize, usize)> {
let rung = current_rung(&state().lock().unwrap());
rung.serves(Role::WholeDenoiser).then_some(WHOLE_FRAME_MAX)
} }
/// Which numeric form of a model a session was built from. /// Which numeric form of a model a session was built from.
@@ -177,7 +202,8 @@ impl Rung {
Role::Keypoints => Form::Int8, Role::Keypoints => Form::Int8,
Role::Detector | Role::Landmarks => Form::A16W8, Role::Detector | Role::Landmarks => Form::A16W8,
Role::Segmenter | Role::Scene | Role::Inpainter | Role::Denoiser => Form::A16W16, Role::Segmenter | Role::Scene | Role::Inpainter | Role::Denoiser => Form::A16W16,
Role::Embedder | Role::EyeClassifier => Form::F32, // Not served there at all: the Hexagon takes fixed shapes.
Role::Embedder | Role::EyeClassifier | Role::WholeDenoiser => Form::F32,
}, },
_ => Form::F32, _ => Form::F32,
} }
@@ -193,6 +219,13 @@ impl Rung {
/// the Neural Engine is fp16, and which unit runs a graph is CoreML's /// the Neural Engine is fp16, and which unit runs a graph is CoreML's
/// choice. /// choice.
fn serves(self, role: Role) -> bool { fn serves(self, role: Role) -> bool {
// Any input size only where a new size costs nothing. MIGraphX,
// OpenVINO and CoreML compile per shape, the Hexagon takes fixed
// shapes only, and the CPU could but would hold gigabytes of f32
// activations for a whole frame of Best.
if role == Role::WholeDenoiser {
return matches!(self, Rung::TensorRt | Rung::Cuda);
}
match self { match self {
Rung::Hexagon => !matches!(role, Role::Embedder | Role::EyeClassifier), Rung::Hexagon => !matches!(role, Role::Embedder | Role::EyeClassifier),
Rung::CoreMl => role != Role::Embedder, Rung::CoreMl => role != Role::Embedder,
+74
View File
@@ -58,6 +58,9 @@ fn build_with(
// write, or the directory CoreML or OpenVINO compiles into. // write, or the directory CoreML or OpenVINO compiles into.
let per_model = match rung { let per_model = match rung {
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)), Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
Rung::TensorRt if role == Role::WholeDenoiser => {
Some(crate::engines::tensorrt_whole_dir(cfg, bytes))
}
Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))), Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))),
_ if ready => None, _ if ready => None,
_ => context.clone(), _ => context.clone(),
@@ -135,6 +138,11 @@ fn providers(
Rung::Cuda => { Rung::Cuda => {
Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?) Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?)
} }
Rung::TensorRt if role == Role::WholeDenoiser => {
let mut b = b;
tensorrt_whole(&mut b, per_model.expect("a whole-frame engine directory"))?;
Ok(b.with_execution_providers([ep::CUDA::default().build()])?)
}
Rung::TensorRt => { Rung::TensorRt => {
let cache = cfg.cache_dir.join("tensorrt"); let cache = cfg.cache_dir.join("tensorrt");
let _ = std::fs::create_dir_all(&cache); let _ = std::fs::create_dir_all(&cache);
@@ -251,6 +259,72 @@ fn migraphx(
) )
} }
/// TensorRT for a whole-frame model: one engine for every input size up to
/// [`crate::WHOLE_FRAME_MAX`], kept in its own directory.
///
/// `ort`'s builder has no profile options, so this registers through the
/// runtime's TensorRT V2 options, with the names 1.30 reads
/// (`tensorrt_execution_provider_info.cc`): `trt_profile_{min,opt,max}_shapes`.
/// Without a profile a dynamic input compiles a new engine per size at run
/// time — 156 s on the first frame, measured — so the profile is the
/// difference between a whole-frame engine and a stall. fp16, as for every
/// role but the embedder (§7); the denoiser measured 0.00 dB from f32.
#[cfg(not(target_os = "android"))]
fn tensorrt_whole(
b: &mut ort::session::builder::SessionBuilder,
cache: &std::path::Path,
) -> ort::Result<()> {
use ort::AsPointer;
use std::ffi::CString;
let _ = std::fs::create_dir_all(cache);
let shapes = |(h, w): (usize, usize)| format!("mosaic:1x1x{h}x{w},sigma:1x1x{h}x{w}");
let dir = cache.to_string_lossy().into_owned();
let options = [
("trt_fp16_enable", "1".to_string()),
("trt_engine_cache_enable", "1".to_string()),
("trt_engine_cache_path", dir.clone()),
("trt_timing_cache_enable", "1".to_string()),
("trt_timing_cache_path", dir),
("trt_max_workspace_size", (1u64 << 30).to_string()),
("trt_profile_min_shapes", shapes((256, 256))),
("trt_profile_opt_shapes", shapes(crate::WHOLE_FRAME_OPT)),
("trt_profile_max_shapes", shapes(crate::WHOLE_FRAME_MAX)),
];
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
let keys = options
.iter()
.map(|(k, _)| cstr(k))
.collect::<ort::Result<Vec<_>>>()?;
let values = options
.iter()
.map(|(_, v)| cstr(v))
.collect::<ort::Result<Vec<_>>>()?;
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
let api = ort::api();
// SAFETY: the documented create / update / append / release sequence
// `ort`'s own TensorRT builder makes, over arrays that outlive it; the
// runtime copies the options into the session before the release.
unsafe {
let mut trt: *mut ort::sys::OrtTensorRTProviderOptionsV2 = std::ptr::null_mut();
ort::Error::result_from_status((api.CreateTensorRTProviderOptions)(&mut trt))?;
let result = ort::Error::result_from_status((api.UpdateTensorRTProviderOptions)(
trt,
key_ptrs.as_ptr(),
value_ptrs.as_ptr(),
keys.len(),
))
.and_then(|()| {
ort::Error::result_from_status((api.SessionOptionsAppendExecutionProvider_TensorRT_V2)(
b.ptr_mut(),
trt,
))
});
(api.ReleaseTensorRTProviderOptions)(trt);
result
}
}
/// OpenVINO on the GPU, compiling into `cache`. /// OpenVINO on the GPU, compiling into `cache`.
/// ///
/// The option names are those `openvino_provider_factory.cc` reads at 1.24, /// The option names are those `openvino_provider_factory.cc` reads at 1.24,