Feed the denoise network without making it wait for the CPU

A 20 MP frame spent 0.32 s outside the network: each tile's mosaic and
sigma gathered on one thread, then its 24 MB output copied out of the
runtime and back into the frame, all in series with the device. Tiles are
now gathered on every core by a producer thread one tile ahead, so the
gather overlaps the run; the centre is written back across cores; and the
tile interface hands its inputs over and lends its output, so neither
side is copied. With a stand-in network that does nothing, the tiler's own
time falls to 0.14 s at the 1408 tile and 0.09 s at 2048. The exactness
and Bayer-phase tests are unchanged and pass.
This commit is contained in:
2026-10-04 07:30:09 -04:00
parent 0c9d564586
commit 14f08a565f
2 changed files with 204 additions and 56 deletions
+12 -4
View File
@@ -44,15 +44,21 @@ impl TileNet for OnnxNet {
self.tile
}
fn run(&mut self, mosaic: &[f32], sigma: &[f32]) -> Result<Vec<f32>, DenoiseError> {
fn run(
&mut self,
mosaic: Vec<f32>,
sigma: Vec<f32>,
write: &mut dyn FnMut(&[f32]),
) -> Result<(), DenoiseError> {
let n = self.tile;
let shape = ndarray::IxDyn(&[1, 1, n, n]);
// The vectors become the tensors: no copy on the way in.
let m = ort::value::Tensor::from_array(
ndarray::Array::from_shape_vec(shape.clone(), mosaic.to_vec())
ndarray::Array::from_shape_vec(shape.clone(), mosaic)
.map_err(|e| DenoiseError::Model(e.to_string()))?,
)?;
let s = ort::value::Tensor::from_array(
ndarray::Array::from_shape_vec(shape, sigma.to_vec())
ndarray::Array::from_shape_vec(shape, sigma)
.map_err(|e| DenoiseError::Model(e.to_string()))?,
)?;
let acquired = self.model.acquire()?;
@@ -65,6 +71,8 @@ impl TileNet for OnnxNet {
"output is {dims:?}, expected [1, 3, {n}, {n}]"
)));
}
Ok(data.to_vec())
// And none on the way out: the frame is written from the runtime's buffer.
write(data);
Ok(())
}
}