Feed the denoise network without making it wait for the CPU
A 20 MP frame spent 0.32 s outside the network: each tile's mosaic and sigma gathered on one thread, then its 24 MB output copied out of the runtime and back into the frame, all in series with the device. Tiles are now gathered on every core by a producer thread one tile ahead, so the gather overlaps the run; the centre is written back across cores; and the tile interface hands its inputs over and lends its output, so neither side is copied. With a stand-in network that does nothing, the tiler's own time falls to 0.14 s at the 1408 tile and 0.09 s at 2048. The exactness and Bayer-phase tests are unchanged and pass.
This commit is contained in:
@@ -44,15 +44,21 @@ impl TileNet for OnnxNet {
|
||||
self.tile
|
||||
}
|
||||
|
||||
fn run(&mut self, mosaic: &[f32], sigma: &[f32]) -> Result<Vec<f32>, DenoiseError> {
|
||||
fn run(
|
||||
&mut self,
|
||||
mosaic: Vec<f32>,
|
||||
sigma: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), DenoiseError> {
|
||||
let n = self.tile;
|
||||
let shape = ndarray::IxDyn(&[1, 1, n, n]);
|
||||
// The vectors become the tensors: no copy on the way in.
|
||||
let m = ort::value::Tensor::from_array(
|
||||
ndarray::Array::from_shape_vec(shape.clone(), mosaic.to_vec())
|
||||
ndarray::Array::from_shape_vec(shape.clone(), mosaic)
|
||||
.map_err(|e| DenoiseError::Model(e.to_string()))?,
|
||||
)?;
|
||||
let s = ort::value::Tensor::from_array(
|
||||
ndarray::Array::from_shape_vec(shape, sigma.to_vec())
|
||||
ndarray::Array::from_shape_vec(shape, sigma)
|
||||
.map_err(|e| DenoiseError::Model(e.to_string()))?,
|
||||
)?;
|
||||
let acquired = self.model.acquire()?;
|
||||
@@ -65,6 +71,8 @@ impl TileNet for OnnxNet {
|
||||
"output is {dims:?}, expected [1, 3, {n}, {n}]"
|
||||
)));
|
||||
}
|
||||
Ok(data.to_vec())
|
||||
// And none on the way out: the frame is written from the runtime's buffer.
|
||||
write(data);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user