Run a denoiser of any input size on TensorRT and CUDA
Role::WholeDenoiser is the denoise network exported with any height and
width, for a whole frame instead of 1408 tiles whose borders are thrown
away. It is served only where a new size costs nothing: the CUDA
provider, and TensorRT through an optimisation profile from 256 to
4608 x 6656, tuned for the 6D's frame with Best's border. Everywhere
else whole_frame_limit() says None and the fixed tiles run.
ort's TensorRT builder has no profile options, so the engine registers
through the runtime's V2 options with the names 1.30 reads
(trt_profile_{min,opt,max}_shapes). Without a profile a dynamic input
compiled an engine per size at run time, 156 s on the first frame. The
engine lives in its own directory per model: ORT's cache key leaves the
shape out, and the fixed 1408 export and its any-size sibling are the
same graph.
This commit is contained in:
@@ -68,6 +68,16 @@ pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Where TensorRT keeps the engine for a whole-frame model. Its own
|
||||||
|
/// directory per model: ONNX Runtime's engine cache key leaves the input
|
||||||
|
/// shape out, and served one export's engine to another of the same graph
|
||||||
|
/// with a different shape when the denoiser was first cut into pieces
|
||||||
|
/// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are
|
||||||
|
/// exactly that pair.
|
||||||
|
pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||||
|
model_dir(cfg, "tensorrt-whole", bytes)
|
||||||
|
}
|
||||||
|
|
||||||
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
|
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
|
||||||
/// model, and one per runtime version, which wrote it.
|
/// model, and one per runtime version, which wrote it.
|
||||||
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
|
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
|
||||||
|
|||||||
@@ -53,6 +53,31 @@ pub enum Role {
|
|||||||
/// levels cannot hold the shadow steps it exists to recover — so the
|
/// levels cannot hold the shadow steps it exists to recover — so the
|
||||||
/// Hexagon takes it with 16-bit activations and weights (§1.5).
|
/// Hexagon takes it with 16-bit activations and weights (§1.5).
|
||||||
Denoiser,
|
Denoiser,
|
||||||
|
/// The same denoise networks exported with any height and width, run
|
||||||
|
/// over a whole frame — or the fewest large tiles that fit — instead of
|
||||||
|
/// 1408² tiles whose borders are thrown away (docs/dev/denoise.md §14).
|
||||||
|
/// Served only where a size the graph was not compiled for costs
|
||||||
|
/// nothing: TensorRT, through an optimisation profile up to
|
||||||
|
/// [`WHOLE_FRAME_MAX`], and the CUDA provider. Everywhere else the
|
||||||
|
/// fixed-tile [`Role::Denoiser`] runs; see [`whole_frame_limit`].
|
||||||
|
WholeDenoiser,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The largest input, rows × columns, a [`Role::WholeDenoiser`] session
|
||||||
|
/// takes: TensorRT's optimisation profile is built up to it, and the tiler
|
||||||
|
/// cuts a larger frame into tiles no bigger. A 20 MP 6D frame with Best's
|
||||||
|
/// reflected border is 4160 × 5984.
|
||||||
|
pub const WHOLE_FRAME_MAX: (usize, usize) = (4608, 6656);
|
||||||
|
|
||||||
|
/// The input size TensorRT tunes a whole-frame engine for: the 6D's frame
|
||||||
|
/// with Best's border, the frame the reference measurements are of.
|
||||||
|
pub const WHOLE_FRAME_OPT: (usize, usize) = (4160, 5984);
|
||||||
|
|
||||||
|
/// Whether the selected rung runs [`Role::WholeDenoiser`], and if so the
|
||||||
|
/// largest input it takes. `None` means run the fixed tiles.
|
||||||
|
pub fn whole_frame_limit() -> Option<(usize, usize)> {
|
||||||
|
let rung = current_rung(&state().lock().unwrap());
|
||||||
|
rung.serves(Role::WholeDenoiser).then_some(WHOLE_FRAME_MAX)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Which numeric form of a model a session was built from.
|
/// Which numeric form of a model a session was built from.
|
||||||
@@ -177,7 +202,8 @@ impl Rung {
|
|||||||
Role::Keypoints => Form::Int8,
|
Role::Keypoints => Form::Int8,
|
||||||
Role::Detector | Role::Landmarks => Form::A16W8,
|
Role::Detector | Role::Landmarks => Form::A16W8,
|
||||||
Role::Segmenter | Role::Scene | Role::Inpainter | Role::Denoiser => Form::A16W16,
|
Role::Segmenter | Role::Scene | Role::Inpainter | Role::Denoiser => Form::A16W16,
|
||||||
Role::Embedder | Role::EyeClassifier => Form::F32,
|
// Not served there at all: the Hexagon takes fixed shapes.
|
||||||
|
Role::Embedder | Role::EyeClassifier | Role::WholeDenoiser => Form::F32,
|
||||||
},
|
},
|
||||||
_ => Form::F32,
|
_ => Form::F32,
|
||||||
}
|
}
|
||||||
@@ -193,6 +219,13 @@ impl Rung {
|
|||||||
/// the Neural Engine is fp16, and which unit runs a graph is CoreML's
|
/// the Neural Engine is fp16, and which unit runs a graph is CoreML's
|
||||||
/// choice.
|
/// choice.
|
||||||
fn serves(self, role: Role) -> bool {
|
fn serves(self, role: Role) -> bool {
|
||||||
|
// Any input size only where a new size costs nothing. MIGraphX,
|
||||||
|
// OpenVINO and CoreML compile per shape, the Hexagon takes fixed
|
||||||
|
// shapes only, and the CPU could but would hold gigabytes of f32
|
||||||
|
// activations for a whole frame of Best.
|
||||||
|
if role == Role::WholeDenoiser {
|
||||||
|
return matches!(self, Rung::TensorRt | Rung::Cuda);
|
||||||
|
}
|
||||||
match self {
|
match self {
|
||||||
Rung::Hexagon => !matches!(role, Role::Embedder | Role::EyeClassifier),
|
Rung::Hexagon => !matches!(role, Role::Embedder | Role::EyeClassifier),
|
||||||
Rung::CoreMl => role != Role::Embedder,
|
Rung::CoreMl => role != Role::Embedder,
|
||||||
|
|||||||
@@ -58,6 +58,9 @@ fn build_with(
|
|||||||
// write, or the directory CoreML or OpenVINO compiles into.
|
// write, or the directory CoreML or OpenVINO compiles into.
|
||||||
let per_model = match rung {
|
let per_model = match rung {
|
||||||
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
|
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
|
||||||
|
Rung::TensorRt if role == Role::WholeDenoiser => {
|
||||||
|
Some(crate::engines::tensorrt_whole_dir(cfg, bytes))
|
||||||
|
}
|
||||||
Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))),
|
Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))),
|
||||||
_ if ready => None,
|
_ if ready => None,
|
||||||
_ => context.clone(),
|
_ => context.clone(),
|
||||||
@@ -135,6 +138,11 @@ fn providers(
|
|||||||
Rung::Cuda => {
|
Rung::Cuda => {
|
||||||
Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?)
|
Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?)
|
||||||
}
|
}
|
||||||
|
Rung::TensorRt if role == Role::WholeDenoiser => {
|
||||||
|
let mut b = b;
|
||||||
|
tensorrt_whole(&mut b, per_model.expect("a whole-frame engine directory"))?;
|
||||||
|
Ok(b.with_execution_providers([ep::CUDA::default().build()])?)
|
||||||
|
}
|
||||||
Rung::TensorRt => {
|
Rung::TensorRt => {
|
||||||
let cache = cfg.cache_dir.join("tensorrt");
|
let cache = cfg.cache_dir.join("tensorrt");
|
||||||
let _ = std::fs::create_dir_all(&cache);
|
let _ = std::fs::create_dir_all(&cache);
|
||||||
@@ -251,6 +259,72 @@ fn migraphx(
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// TensorRT for a whole-frame model: one engine for every input size up to
|
||||||
|
/// [`crate::WHOLE_FRAME_MAX`], kept in its own directory.
|
||||||
|
///
|
||||||
|
/// `ort`'s builder has no profile options, so this registers through the
|
||||||
|
/// runtime's TensorRT V2 options, with the names 1.30 reads
|
||||||
|
/// (`tensorrt_execution_provider_info.cc`): `trt_profile_{min,opt,max}_shapes`.
|
||||||
|
/// Without a profile a dynamic input compiles a new engine per size at run
|
||||||
|
/// time — 156 s on the first frame, measured — so the profile is the
|
||||||
|
/// difference between a whole-frame engine and a stall. fp16, as for every
|
||||||
|
/// role but the embedder (§7); the denoiser measured 0.00 dB from f32.
|
||||||
|
#[cfg(not(target_os = "android"))]
|
||||||
|
fn tensorrt_whole(
|
||||||
|
b: &mut ort::session::builder::SessionBuilder,
|
||||||
|
cache: &std::path::Path,
|
||||||
|
) -> ort::Result<()> {
|
||||||
|
use ort::AsPointer;
|
||||||
|
use std::ffi::CString;
|
||||||
|
let _ = std::fs::create_dir_all(cache);
|
||||||
|
let shapes = |(h, w): (usize, usize)| format!("mosaic:1x1x{h}x{w},sigma:1x1x{h}x{w}");
|
||||||
|
let dir = cache.to_string_lossy().into_owned();
|
||||||
|
let options = [
|
||||||
|
("trt_fp16_enable", "1".to_string()),
|
||||||
|
("trt_engine_cache_enable", "1".to_string()),
|
||||||
|
("trt_engine_cache_path", dir.clone()),
|
||||||
|
("trt_timing_cache_enable", "1".to_string()),
|
||||||
|
("trt_timing_cache_path", dir),
|
||||||
|
("trt_max_workspace_size", (1u64 << 30).to_string()),
|
||||||
|
("trt_profile_min_shapes", shapes((256, 256))),
|
||||||
|
("trt_profile_opt_shapes", shapes(crate::WHOLE_FRAME_OPT)),
|
||||||
|
("trt_profile_max_shapes", shapes(crate::WHOLE_FRAME_MAX)),
|
||||||
|
];
|
||||||
|
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
|
||||||
|
let keys = options
|
||||||
|
.iter()
|
||||||
|
.map(|(k, _)| cstr(k))
|
||||||
|
.collect::<ort::Result<Vec<_>>>()?;
|
||||||
|
let values = options
|
||||||
|
.iter()
|
||||||
|
.map(|(_, v)| cstr(v))
|
||||||
|
.collect::<ort::Result<Vec<_>>>()?;
|
||||||
|
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||||
|
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||||
|
let api = ort::api();
|
||||||
|
// SAFETY: the documented create / update / append / release sequence
|
||||||
|
// `ort`'s own TensorRT builder makes, over arrays that outlive it; the
|
||||||
|
// runtime copies the options into the session before the release.
|
||||||
|
unsafe {
|
||||||
|
let mut trt: *mut ort::sys::OrtTensorRTProviderOptionsV2 = std::ptr::null_mut();
|
||||||
|
ort::Error::result_from_status((api.CreateTensorRTProviderOptions)(&mut trt))?;
|
||||||
|
let result = ort::Error::result_from_status((api.UpdateTensorRTProviderOptions)(
|
||||||
|
trt,
|
||||||
|
key_ptrs.as_ptr(),
|
||||||
|
value_ptrs.as_ptr(),
|
||||||
|
keys.len(),
|
||||||
|
))
|
||||||
|
.and_then(|()| {
|
||||||
|
ort::Error::result_from_status((api.SessionOptionsAppendExecutionProvider_TensorRT_V2)(
|
||||||
|
b.ptr_mut(),
|
||||||
|
trt,
|
||||||
|
))
|
||||||
|
});
|
||||||
|
(api.ReleaseTensorRTProviderOptions)(trt);
|
||||||
|
result
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// OpenVINO on the GPU, compiling into `cache`.
|
/// OpenVINO on the GPU, compiling into `cache`.
|
||||||
///
|
///
|
||||||
/// The option names are those `openvino_provider_factory.cc` reads at 1.24,
|
/// The option names are those `openvino_provider_factory.cc` reads at 1.24,
|
||||||
|
|||||||
Reference in New Issue
Block a user