Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8aa10cd249 | ||
|
|
c2cfacd7d3 | ||
|
|
a9271c4850 | ||
|
|
4b71ef0947 | ||
|
|
1e1aa1442b | ||
|
|
3761281dd1 | ||
|
|
9cba420fd5 | ||
|
|
56f4180347 | ||
|
|
417cba8b4d |
Generated
+26
-26
@@ -1265,7 +1265,7 @@ checksum = "f27ae1dd37df86211c42e150270f82743308803d90a6f6e6651cd730d5e1732f"
|
||||
|
||||
[[package]]
|
||||
name = "darkroom-android"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"android_logger",
|
||||
"dr-plat",
|
||||
@@ -1278,7 +1278,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "darkroom-desktop"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-plat",
|
||||
@@ -1454,7 +1454,7 @@ checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76"
|
||||
|
||||
[[package]]
|
||||
name = "dr-bench"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-catalog",
|
||||
@@ -1471,7 +1471,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-catalog"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-face",
|
||||
"dr-plat",
|
||||
@@ -1486,7 +1486,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-decode"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"env_logger",
|
||||
@@ -1500,7 +1500,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-denoise"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-gpu",
|
||||
@@ -1517,7 +1517,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-export"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-gpu",
|
||||
@@ -1536,7 +1536,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-face"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-inference-engine",
|
||||
"env_logger",
|
||||
@@ -1549,7 +1549,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-film"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"log",
|
||||
"serde",
|
||||
@@ -1558,7 +1558,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-gpu"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
"dr-decode",
|
||||
@@ -1576,7 +1576,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-inference-engine"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"libloading",
|
||||
@@ -1591,7 +1591,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-ingest"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-plat",
|
||||
"dr-types",
|
||||
@@ -1603,7 +1603,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-lens"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"lensfun",
|
||||
"log",
|
||||
@@ -1611,7 +1611,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-pano"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-inference-engine",
|
||||
@@ -1625,7 +1625,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-pipeline"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"log",
|
||||
@@ -1634,7 +1634,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-plat"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"android-native-keyring-store",
|
||||
"dr-types",
|
||||
@@ -1650,7 +1650,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-preset-xmp"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-pipeline",
|
||||
"log",
|
||||
@@ -1660,7 +1660,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-segment"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-inference-engine",
|
||||
"env_logger",
|
||||
@@ -1673,7 +1673,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-plat",
|
||||
@@ -1687,7 +1687,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync-folder"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-sync",
|
||||
@@ -1699,7 +1699,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync-nextcloud"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-decode",
|
||||
@@ -1721,7 +1721,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-thumbs"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"jpeg-encoder",
|
||||
@@ -1733,7 +1733,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-types"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -1742,7 +1742,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-ui"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"async-trait",
|
||||
@@ -1792,7 +1792,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-xmp"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"log",
|
||||
@@ -7126,7 +7126,7 @@ checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3"
|
||||
|
||||
[[package]]
|
||||
name = "traceability"
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"proc-macro2",
|
||||
|
||||
+1
-1
@@ -33,7 +33,7 @@ members = [
|
||||
exclude = ["third_party"]
|
||||
|
||||
[workspace.package]
|
||||
version = "0.23.0"
|
||||
version = "0.24.0"
|
||||
edition = "2021"
|
||||
rust-version = "1.92"
|
||||
license = "GPL-3.0-or-later"
|
||||
|
||||
@@ -201,7 +201,7 @@ controls, its place in the chain and its tests.
|
||||
|
||||
## Where it stands
|
||||
|
||||
**0.23.0**, thirty-eight tagged releases in. 193 numbered requirements in
|
||||
**0.24.0**, thirty-nine tagged releases in. 193 numbered requirements in
|
||||
scope, 85% of them claimed by code and [traced to it](docs/dev/traceability.md);
|
||||
the rest are written down rather than merely absent.
|
||||
|
||||
|
||||
@@ -338,7 +338,7 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
// sibling when the probe chose that rung and ignores it otherwise. The
|
||||
// segmenter's and XFeat's forms are compiled into the binary instead,
|
||||
// beside their f32 graphs.
|
||||
const BUNDLED: [(&std::ffi::CStr, &str); 23] = [
|
||||
const BUNDLED: [(&std::ffi::CStr, &str); 21] = [
|
||||
(c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"),
|
||||
(
|
||||
c"models/scrfd_500m_640.a16w8.onnx",
|
||||
@@ -379,15 +379,10 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
c"models/mosaic-fast-1408.a16w16.onnx",
|
||||
"mosaic-fast-1408.a16w16.onnx",
|
||||
),
|
||||
(c"models/mosaic-medium-1408.onnx", "mosaic-medium-1408.onnx"),
|
||||
(c"models/mosaic-hq-1408.onnx", "mosaic-hq-1408.onnx"),
|
||||
(
|
||||
c"models/mosaic-medium-1408.a16w16.onnx",
|
||||
"mosaic-medium-1408.a16w16.onnx",
|
||||
),
|
||||
(c"models/mosaic-best-1408.onnx", "mosaic-best-1408.onnx"),
|
||||
(
|
||||
c"models/mosaic-best-1408.a16w16.onnx",
|
||||
"mosaic-best-1408.a16w16.onnx",
|
||||
c"models/mosaic-hq-1408.a16w16.onnx",
|
||||
"mosaic-hq-1408.a16w16.onnx",
|
||||
),
|
||||
];
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
//!
|
||||
//! ```sh
|
||||
//! DARKROOM_ORT_DIR=~/.local/share/darkroom/runtime \
|
||||
//! cargo run --release -p dr-denoise --features native --example denoise_raw -- IMG.CR2 out [fast|medium|best]
|
||||
//! cargo run --release -p dr-denoise --features native --example denoise_raw -- IMG.CR2 out [fast|best]
|
||||
//! ```
|
||||
//!
|
||||
//! Decode, the app's hot-pixel pass, the frame's noise from its best source,
|
||||
@@ -11,7 +11,9 @@
|
||||
//! camera RGB — for comparison with the training repo's own path
|
||||
//! (`tools/compare_rust.py` in darkroom-denoise). `DARKROOM_ORT_DIR` points
|
||||
//! at an ONNX Runtime build; the engine's cache goes to `DR_ENGINE_CACHE` or
|
||||
//! a temporary directory.
|
||||
//! a temporary directory. The whole-frame network (`mosaic-hq.onnx` beside
|
||||
//! the fixed file) runs where the rung takes any size; `DR_PLAN=tiles` keeps
|
||||
//! the 1408² tiles anyway, to compare the two.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::time::{Duration, Instant};
|
||||
@@ -23,21 +25,26 @@ fn main() {
|
||||
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("warn")).init();
|
||||
let mut args = std::env::args().skip(1);
|
||||
let (Some(input), Some(out)) = (args.next(), args.next()) else {
|
||||
eprintln!("usage: denoise_raw RAW OUT_PREFIX [fast|medium|best]");
|
||||
eprintln!("usage: denoise_raw RAW OUT_PREFIX [fast|best]");
|
||||
std::process::exit(2);
|
||||
};
|
||||
let shipped = match args.next().as_deref() {
|
||||
None | Some("best") => dr_denoise::BEST,
|
||||
Some("medium") => dr_denoise::MEDIUM,
|
||||
Some("fast") => dr_denoise::FAST,
|
||||
Some(other) => {
|
||||
eprintln!("no network called {other}: fast, medium or best");
|
||||
eprintln!("no network called {other}: fast or best");
|
||||
std::process::exit(2);
|
||||
}
|
||||
};
|
||||
let model = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("../../models/denoise")
|
||||
.join(shipped.file);
|
||||
let whole = model.with_file_name(shipped.whole);
|
||||
let tiles_only = std::env::var("DR_PLAN").is_ok_and(|p| p == "tiles");
|
||||
let mut models = vec![(Role::Denoiser, model.clone())];
|
||||
if whole.is_file() && !tiles_only {
|
||||
models.push((Role::WholeDenoiser, whole));
|
||||
}
|
||||
let cache = std::env::var_os("DR_ENGINE_CACHE")
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| std::env::temp_dir().join("dr-denoise-engines"));
|
||||
@@ -48,7 +55,7 @@ fn main() {
|
||||
.into_iter()
|
||||
.collect(),
|
||||
cache_dir: cache,
|
||||
models: vec![(Role::Denoiser, model.clone())],
|
||||
models,
|
||||
embedded: Vec::new(),
|
||||
ceiling: None,
|
||||
threads: 0,
|
||||
@@ -101,10 +108,20 @@ fn main() {
|
||||
noise.col
|
||||
);
|
||||
|
||||
let mut net = OnnxNet::from_path(&model, shipped.halo).expect("model");
|
||||
let mut net = if tiles_only {
|
||||
OnnxNet::open_tiled(&model, shipped)
|
||||
} else {
|
||||
OnnxNet::open(&model, shipped)
|
||||
}
|
||||
.expect("model");
|
||||
println!(
|
||||
"rung {}",
|
||||
net.rung().map(|r| r.label()).unwrap_or("?")
|
||||
"rung {} · {}",
|
||||
net.rung().map(|r| r.label()).unwrap_or("?"),
|
||||
if net.whole_frame() {
|
||||
"whole frame"
|
||||
} else {
|
||||
"1408² tiles"
|
||||
}
|
||||
);
|
||||
let t = Instant::now();
|
||||
let rgb = dr_denoise::denoise(&raw, &noise, &mut net, &mut |done, total| {
|
||||
|
||||
+14
-13
@@ -25,33 +25,34 @@ pub mod tile;
|
||||
use dr_decode::RawImage;
|
||||
|
||||
pub use noise::{NoiseModel, Source};
|
||||
pub use tile::{TileNet, HALO};
|
||||
pub use tile::{Sizes, TileNet, HALO};
|
||||
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// A network the app ships in `models/denoise/`: its file, and the context
|
||||
/// it needs past a tile's kept centre (docs/dev/denoise.md §13).
|
||||
/// A network the app ships in `models/denoise/`: its fixed-tile file, the
|
||||
/// same network with any height and width for a whole frame (§14), and the
|
||||
/// context it needs past a tile's kept centre (§13).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Shipped {
|
||||
pub file: &'static str,
|
||||
pub whole: &'static str,
|
||||
pub halo: usize,
|
||||
}
|
||||
|
||||
/// The smallest student: 0.9 M parameters, 11 GMAC a megapixel.
|
||||
pub const FAST: Shipped = Shipped {
|
||||
file: "mosaic-fast-1408.onnx",
|
||||
whole: "mosaic-fast.onnx",
|
||||
halo: HALO,
|
||||
};
|
||||
/// A student of the mixture with the first release's shape: 3.2 M
|
||||
/// parameters, 48 GMAC a megapixel.
|
||||
pub const MEDIUM: Shipped = Shipped {
|
||||
file: "mosaic-medium-1408.onnx",
|
||||
halo: HALO,
|
||||
};
|
||||
/// The mixture: a flat expert, an edge expert and the gate that blends them.
|
||||
/// It reaches further than either, so it keeps a smaller centre of each tile.
|
||||
/// One network of the first release's shape, 3.2 M parameters and 48 GMAC a
|
||||
/// megapixel, taught by the mixture of experts that was Best until 0.24:
|
||||
/// its edges at a third of its work (denoise.md §15). A new file name, not
|
||||
/// the old Medium's or Best's: the result cache keys a model by its name
|
||||
/// and size, and this one is byte for byte the old Medium's size.
|
||||
pub const BEST: Shipped = Shipped {
|
||||
file: "mosaic-best-1408.onnx",
|
||||
halo: 256,
|
||||
file: "mosaic-hq-1408.onnx",
|
||||
whole: "mosaic-hq.onnx",
|
||||
halo: HALO,
|
||||
};
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
|
||||
+72
-16
@@ -10,30 +10,80 @@
|
||||
//! lost 5–9 dB (docs/dev/inference.md §1.5). That sibling is the same network
|
||||
//! with the Bayer packing spelled `SpaceToDepth`, which QNN can hold and the
|
||||
//! 6-D reshape it replaces it cannot.
|
||||
//!
|
||||
//! Each network also ships with any height and width (`mosaic-hq.onnx`
|
||||
//! beside `mosaic-hq-1408.onnx`, darkroom-denoise `tools/export_whole.py`,
|
||||
//! identical to the fixed file at 1408²). Where the rung takes any size, the
|
||||
//! frame runs whole instead of in tiles whose borders are thrown away — a
|
||||
//! 1408² tile keeps 1024², 1.89 photosites computed for each one kept
|
||||
//! (denoise.md §14).
|
||||
|
||||
use crate::tile::TileNet;
|
||||
use crate::DenoiseError;
|
||||
use dr_inference_engine::{Model, Role};
|
||||
use crate::tile::{Sizes, TileNet};
|
||||
use crate::{DenoiseError, Shipped};
|
||||
use dr_inference_engine::{Form, Model, Role};
|
||||
|
||||
/// The edge of the tile the shipped export takes.
|
||||
/// The edge of the tile the shipped fixed-shape export takes.
|
||||
pub const TILE: usize = 1408;
|
||||
|
||||
/// What a whole-frame input's sides must be multiples of: the networks pack
|
||||
/// 2×2 and halve three times, so a side is a whole number of positions at
|
||||
/// their coarsest level only in steps of 16.
|
||||
pub const ALIGN: usize = 16;
|
||||
|
||||
pub struct OnnxNet {
|
||||
model: Model,
|
||||
tile: usize,
|
||||
sizes: Sizes,
|
||||
halo: usize,
|
||||
}
|
||||
|
||||
impl OnnxNet {
|
||||
/// The network at `path`, which needs `halo` photosites of context
|
||||
/// ([`crate::Shipped::halo`]).
|
||||
pub fn from_path(path: &std::path::Path, halo: usize) -> Result<Self, DenoiseError> {
|
||||
/// The network `shipped`, whose fixed-tile file is at `path`.
|
||||
///
|
||||
/// On a rung that runs any input size (TensorRT, the CUDA provider —
|
||||
/// [`dr_inference_engine::whole_frame_limit`]) and with the any-size
|
||||
/// export installed beside it, the whole-frame network: the frame in one
|
||||
/// call, or the fewest large tiles that fit (§14). Its output is the
|
||||
/// fixed tiles' to rounding. Everywhere else, and if the whole-frame
|
||||
/// model will not open, the 1408² tiles.
|
||||
pub fn open(path: &std::path::Path, shipped: Shipped) -> Result<Self, DenoiseError> {
|
||||
if let Some(max) = dr_inference_engine::whole_frame_limit() {
|
||||
let whole = path.with_file_name(shipped.whole);
|
||||
if whole.is_file() {
|
||||
let opened = std::fs::read(&whole)
|
||||
.map_err(DenoiseError::from)
|
||||
.and_then(|bytes| {
|
||||
Ok(dr_inference_engine::open(
|
||||
Role::WholeDenoiser,
|
||||
Form::F32,
|
||||
&bytes,
|
||||
)?)
|
||||
});
|
||||
match opened {
|
||||
Ok(model) => {
|
||||
return Ok(OnnxNet {
|
||||
model,
|
||||
sizes: Sizes::Any { align: ALIGN, max },
|
||||
halo: shipped.halo,
|
||||
})
|
||||
}
|
||||
Err(e) => log::warn!(
|
||||
"learned denoise: {} will not open ({e}); running 1408² tiles",
|
||||
whole.display()
|
||||
),
|
||||
}
|
||||
}
|
||||
}
|
||||
Self::open_tiled(path, shipped)
|
||||
}
|
||||
|
||||
/// The fixed-tile network at `path`, whatever the rung: 1408² tiles.
|
||||
pub fn open_tiled(path: &std::path::Path, shipped: Shipped) -> Result<Self, DenoiseError> {
|
||||
let (path, form) = dr_inference_engine::resolve_model(Role::Denoiser, path);
|
||||
let bytes = std::fs::read(&path)?;
|
||||
Ok(OnnxNet {
|
||||
model: dr_inference_engine::open(Role::Denoiser, form, &bytes)?,
|
||||
tile: TILE,
|
||||
halo,
|
||||
sizes: Sizes::Square(TILE),
|
||||
halo: shipped.halo,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -41,11 +91,16 @@ impl OnnxNet {
|
||||
pub fn rung(&self) -> Result<dr_inference_engine::Rung, DenoiseError> {
|
||||
Ok(self.model.acquire()?.rung())
|
||||
}
|
||||
|
||||
/// Whether this is the whole-frame network.
|
||||
pub fn whole_frame(&self) -> bool {
|
||||
matches!(self.sizes, Sizes::Any { .. })
|
||||
}
|
||||
}
|
||||
|
||||
impl TileNet for OnnxNet {
|
||||
fn tile(&self) -> usize {
|
||||
self.tile
|
||||
fn sizes(&self) -> Sizes {
|
||||
self.sizes
|
||||
}
|
||||
|
||||
fn halo(&self) -> usize {
|
||||
@@ -54,12 +109,13 @@ impl TileNet for OnnxNet {
|
||||
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
mosaic: Vec<f32>,
|
||||
sigma: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), DenoiseError> {
|
||||
let n = self.tile;
|
||||
let shape = ndarray::IxDyn(&[1, 1, n, n]);
|
||||
let shape = ndarray::IxDyn(&[1, 1, rows, cols]);
|
||||
// The vectors become the tensors: no copy on the way in.
|
||||
let m = ort::value::Tensor::from_array(
|
||||
ndarray::Array::from_shape_vec(shape.clone(), mosaic)
|
||||
@@ -74,9 +130,9 @@ impl TileNet for OnnxNet {
|
||||
let outputs = session.run(ort::inputs!["mosaic" => m, "sigma" => s])?;
|
||||
let (shape, data) = outputs[0].try_extract_tensor::<f32>()?;
|
||||
let dims: Vec<i64> = shape.iter().copied().collect();
|
||||
if dims != [1, 3, n as i64, n as i64] {
|
||||
if dims != [1, 3, rows as i64, cols as i64] {
|
||||
return Err(DenoiseError::Model(format!(
|
||||
"output is {dims:?}, expected [1, 3, {n}, {n}]"
|
||||
"output is {dims:?}, expected [1, 3, {rows}, {cols}]"
|
||||
)));
|
||||
}
|
||||
// And none on the way out: the frame is written from the runtime's buffer.
|
||||
|
||||
+363
-76
@@ -1,12 +1,18 @@
|
||||
//! TRACES: FR-DEV-3g
|
||||
//! A whole frame through a fixed-shape network, exactly (denoise.md §3.4).
|
||||
//! A whole frame through a network, in tiles, exactly (denoise.md §3.4, §14).
|
||||
//!
|
||||
//! The network sees `TILE_IN`² photosites and its output is exact in the
|
||||
//! central `TILE_IN − 2·HALO`: the halo is wider than its receptive field
|
||||
//! (185 photosites, counted from the layers), so a tile's centre equals the
|
||||
//! whole frame's at the same place. The frame is extended by reflection
|
||||
//! about its edge photosites, which keeps every photosite's CFA colour, so
|
||||
//! edge tiles see real context too.
|
||||
//! A tile's output is exact in its centre: past a halo wider than the
|
||||
//! network's receptive field (185 photosites for a single network, more for
|
||||
//! the mixture), a tile's centre equals the whole frame's at the same place.
|
||||
//! The frame is extended by reflection about its edge photosites, which
|
||||
//! keeps every photosite's CFA colour, so edge tiles see real context too.
|
||||
//!
|
||||
//! **Tile sizes.** A fixed-shape network takes one square ([`Sizes::Square`],
|
||||
//! 1408², of which Best keeps 896²). A network exported with any height and
|
||||
//! width ([`Sizes::Any`]) takes the frame whole when it is small enough, and
|
||||
//! otherwise the fewest equal tiles that are: [`plan`] picks the grid that
|
||||
//! computes the fewest photosites. If the first tile of a plan fails — a
|
||||
//! GPU out of memory — the limit is halved and the frame planned again.
|
||||
//!
|
||||
//! **Phase.** The network was trained on RGGB. A frame whose pattern starts
|
||||
//! on another colour is read from one photosite up and/or left — the
|
||||
@@ -20,16 +26,26 @@ use dr_decode::CfaPattern;
|
||||
/// [`TileNet::halo`].
|
||||
pub const HALO: usize = 192;
|
||||
|
||||
/// A fixed-shape network: `mosaic` and `sigma`, `n×n` RGGB, in; `3×n×n`
|
||||
/// The tiles a network takes.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Sizes {
|
||||
/// One square, `n` photosites a side.
|
||||
Square(usize),
|
||||
/// Any rectangle whose sides are multiples of `align`, at most `max`
|
||||
/// (rows, columns).
|
||||
Any { align: usize, max: (usize, usize) },
|
||||
}
|
||||
|
||||
/// A network: `mosaic` and `sigma`, `rows×cols` RGGB, in; `3×rows×cols`
|
||||
/// planar linear camera RGB out.
|
||||
///
|
||||
/// The inputs are handed over, and the output is lent to `write` rather than
|
||||
/// returned: a 1408² tile is 24 MB of output, and copying it out of the
|
||||
/// runtime's buffer and back into the frame was a measurable share of a
|
||||
/// frame's time.
|
||||
/// returned: a 1408² tile is 24 MB of output and a whole frame 300 MB, and
|
||||
/// copying it out of the runtime's buffer and back into the frame was a
|
||||
/// measurable share of a frame's time.
|
||||
pub trait TileNet {
|
||||
/// The edge `n` of the square tile the network takes.
|
||||
fn tile(&self) -> usize;
|
||||
/// The tile sizes it takes.
|
||||
fn sizes(&self) -> Sizes;
|
||||
/// Photosites of context it needs past a tile's kept centre: at least
|
||||
/// its receptive field. [`HALO`] unless the network says otherwise.
|
||||
fn halo(&self) -> usize {
|
||||
@@ -37,12 +53,87 @@ pub trait TileNet {
|
||||
}
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
mosaic: Vec<f32>,
|
||||
sigma: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), crate::DenoiseError>;
|
||||
}
|
||||
|
||||
/// How a frame is cut: every tile `rows × cols` in, keeping its centre
|
||||
/// `core.0 × core.1` past the halo, on a `grid.0 × grid.1` grid.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Plan {
|
||||
pub rows: usize,
|
||||
pub cols: usize,
|
||||
pub core: (usize, usize),
|
||||
pub grid: (usize, usize),
|
||||
}
|
||||
|
||||
impl Plan {
|
||||
/// Photosites the network computes for the frame.
|
||||
pub fn work(&self) -> usize {
|
||||
self.grid.0 * self.grid.1 * self.rows * self.cols
|
||||
}
|
||||
}
|
||||
|
||||
/// The tiles for an `uh × uw` frame (in the network's phase) at `sizes`, or
|
||||
/// `None` when no tile fits.
|
||||
///
|
||||
/// Square tiles are today's grid. Any-size tiles are equal on each axis, so
|
||||
/// one call shape serves the frame — TensorRT's profile tunes for one, and
|
||||
/// the CUDA provider searches its algorithms once per shape — and the grid
|
||||
/// is the one with the least work: one tile whenever the frame and its
|
||||
/// halo fit under `max`.
|
||||
pub fn plan(uh: usize, uw: usize, halo: usize, sizes: Sizes) -> Option<Plan> {
|
||||
match sizes {
|
||||
Sizes::Square(n) => {
|
||||
if n <= 2 * halo || !(n - 2 * halo).is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
let core = n - 2 * halo;
|
||||
Some(Plan {
|
||||
rows: n,
|
||||
cols: n,
|
||||
core: (core, core),
|
||||
grid: (uh.div_ceil(core), uw.div_ceil(core)),
|
||||
})
|
||||
}
|
||||
Sizes::Any { align, max } => {
|
||||
// An even align keeps every tile origin on an even photosite,
|
||||
// so every tile starts on red.
|
||||
let align = align.max(2).next_multiple_of(2);
|
||||
let axis = |extent: usize, tiles: usize, limit: usize| {
|
||||
let size = (extent.div_ceil(tiles) + 2 * halo).next_multiple_of(align);
|
||||
let core = size.checked_sub(2 * halo)?;
|
||||
(size <= limit && core > 0 && core.is_multiple_of(2)).then_some((size, core))
|
||||
};
|
||||
let mut best: Option<Plan> = None;
|
||||
for gy in 1..=16 {
|
||||
let Some((rows, cy)) = axis(uh, gy, max.0) else {
|
||||
continue;
|
||||
};
|
||||
for gx in 1..=16 {
|
||||
let Some((cols, cx)) = axis(uw, gx, max.1) else {
|
||||
continue;
|
||||
};
|
||||
let p = Plan {
|
||||
rows,
|
||||
cols,
|
||||
core: (cy, cx),
|
||||
grid: (uh.div_ceil(cy), uw.div_ceil(cx)),
|
||||
};
|
||||
if best.is_none_or(|b| p.work() < b.work()) {
|
||||
best = Some(p);
|
||||
}
|
||||
}
|
||||
}
|
||||
best
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Index into `0..n` by reflection about the end photosites, any distance
|
||||
/// out: …2 1 [0 1 2 … n−1] n−2 n−3…, period `2(n−1)`. Parity is kept, which
|
||||
/// is what keeps a CFA colour.
|
||||
@@ -71,7 +162,9 @@ pub fn rggb_offset(p: CfaPattern) -> Option<(usize, usize)> {
|
||||
/// `sigma(colour, value)`, and return `h×w` interleaved RGB.
|
||||
///
|
||||
/// `progress(done, total)` is called after each tile and stops the run by
|
||||
/// returning `false`, in which case the result is `Ok(None)`.
|
||||
/// returning `false`, in which case the result is `Ok(None)`. An any-size
|
||||
/// network whose first tile fails is planned again with tiles half that
|
||||
/// size, until a tile would keep no centre; then the failure is returned.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn run_tiled(
|
||||
net: &mut dyn TileNet,
|
||||
@@ -85,40 +178,103 @@ pub fn run_tiled(
|
||||
let (dy, dx) = rggb_offset(pattern).ok_or_else(|| {
|
||||
crate::DenoiseError::Unsupported(format!("{pattern:?} is not a Bayer pattern"))
|
||||
})?;
|
||||
let (n, halo) = (net.tile(), net.halo());
|
||||
if n <= 2 * halo || !(n - 2 * halo).is_multiple_of(2) {
|
||||
return Err(crate::DenoiseError::Model(format!(
|
||||
"tile {n} leaves no even centre past a {halo} halo"
|
||||
)));
|
||||
let halo = net.halo();
|
||||
let (uh, uw) = (h + dy, w + dx);
|
||||
let mut sizes = net.sizes();
|
||||
loop {
|
||||
let plan = plan(uh, uw, halo, sizes).ok_or_else(|| {
|
||||
crate::DenoiseError::Model(format!(
|
||||
"no tile of {sizes:?} keeps a centre past a {halo} halo"
|
||||
))
|
||||
})?;
|
||||
match run_plan(net, plan, h, w, (dy, dx), halo, at, sigma, progress) {
|
||||
Err(Failed { error, first: true }) => {
|
||||
// The first call of a size is where a GPU runs out of
|
||||
// memory. Halve the larger kept centre of the tile that
|
||||
// failed — not the limit, which may be far above it, and not
|
||||
// the tile, half of which may be all halo — and plan again,
|
||||
// until no smaller tile keeps a centre.
|
||||
let Sizes::Any { align, .. } = sizes else {
|
||||
return Err(error);
|
||||
};
|
||||
let (cr, cc) = plan.core;
|
||||
let smaller = if cr >= cc {
|
||||
(cr / 2 + 2 * halo, plan.cols)
|
||||
} else {
|
||||
(plan.rows, cc / 2 + 2 * halo)
|
||||
};
|
||||
let next = Sizes::Any {
|
||||
align,
|
||||
max: smaller,
|
||||
};
|
||||
if self::plan(uh, uw, halo, next).is_none() {
|
||||
return Err(error);
|
||||
}
|
||||
let core = n - 2 * halo;
|
||||
log::warn!(
|
||||
"learned denoise: a {}×{} tile failed ({error}); trying tiles up to {}×{}",
|
||||
plan.rows,
|
||||
plan.cols,
|
||||
smaller.0,
|
||||
smaller.1
|
||||
);
|
||||
sizes = next;
|
||||
}
|
||||
Err(Failed { error, .. }) => return Err(error),
|
||||
Ok(done) => return Ok(done),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A run that stopped on an error, and whether it was the plan's first call.
|
||||
struct Failed {
|
||||
error: crate::DenoiseError,
|
||||
first: bool,
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn run_plan(
|
||||
net: &mut dyn TileNet,
|
||||
plan: Plan,
|
||||
h: usize,
|
||||
w: usize,
|
||||
(dy, dx): (usize, usize),
|
||||
halo: usize,
|
||||
at: &(dyn Fn(usize, usize) -> f32 + Sync),
|
||||
sigma: &(dyn Fn(usize, f32) -> f32 + Sync),
|
||||
progress: &mut dyn FnMut(usize, usize) -> bool,
|
||||
) -> Result<Option<Vec<f32>>, Failed> {
|
||||
let Plan {
|
||||
rows: nr,
|
||||
cols: nc,
|
||||
core: (cr, cc),
|
||||
grid: (ty, tx),
|
||||
} = plan;
|
||||
// In unified coordinates the frame spans u ∈ [dy, dy + h), v ∈ [dx, dx + w).
|
||||
let (uh, uw) = (h + dy, w + dx);
|
||||
let (ty, tx) = (uh.div_ceil(core), uw.div_ceil(core));
|
||||
let total = ty * tx;
|
||||
let origins: Vec<(usize, usize)> = (0..ty)
|
||||
.flat_map(|i| (0..tx).map(move |j| (i * core, j * core)))
|
||||
.flat_map(|i| (0..tx).map(move |j| (i * cr, j * cc)))
|
||||
.collect();
|
||||
let threads = std::thread::available_parallelism().map_or(1, |n| n.get());
|
||||
|
||||
// One tile's mosaic and σ, gathered on every core: rows are independent.
|
||||
let gather = |u0: usize, v0: usize| {
|
||||
let mut mos = vec![0.0f32; n * n];
|
||||
let mut sig = vec![0.0f32; n * n];
|
||||
let rows_per = n.div_ceil(threads).max(1);
|
||||
let mut mos = vec![0.0f32; nr * nc];
|
||||
let mut sig = vec![0.0f32; nr * nc];
|
||||
let rows_per = nr.div_ceil(threads).max(1);
|
||||
std::thread::scope(|scope| {
|
||||
for (chunk, (m, s)) in mos
|
||||
.chunks_mut(rows_per * n)
|
||||
.zip(sig.chunks_mut(rows_per * n))
|
||||
.chunks_mut(rows_per * nc)
|
||||
.zip(sig.chunks_mut(rows_per * nc))
|
||||
.enumerate()
|
||||
{
|
||||
scope.spawn(move || {
|
||||
for (i, (mrow, srow)) in m.chunks_mut(n).zip(s.chunks_mut(n)).enumerate() {
|
||||
for (i, (mrow, srow)) in m.chunks_mut(nc).zip(s.chunks_mut(nc)).enumerate() {
|
||||
let r = chunk * rows_per + i;
|
||||
// Unified row u = u0 + r − halo; frame row y = u − dy, reflected.
|
||||
let u = u0 as isize + r as isize - halo as isize;
|
||||
let y = reflect(u - dy as isize, h);
|
||||
for c in 0..n {
|
||||
for c in 0..nc {
|
||||
let v = v0 as isize + c as isize - halo as isize;
|
||||
let x = reflect(v - dx as isize, w);
|
||||
let val = at(y, x);
|
||||
@@ -138,7 +294,14 @@ pub fn run_tiled(
|
||||
// most two tiles' inputs alive.
|
||||
let mut out = vec![0.0f32; h * w * 3];
|
||||
let stop = std::sync::atomic::AtomicBool::new(false);
|
||||
std::thread::scope(|scope| -> Result<Option<()>, crate::DenoiseError> {
|
||||
let fail = |error, k: usize, stop: &std::sync::atomic::AtomicBool| {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
Failed {
|
||||
error,
|
||||
first: k == 0,
|
||||
}
|
||||
};
|
||||
std::thread::scope(|scope| -> Result<Option<()>, Failed> {
|
||||
let (tx_tiles, rx_tiles) = std::sync::mpsc::sync_channel(1);
|
||||
let (origins, stop, gather) = (&origins, &stop, &gather);
|
||||
scope.spawn(move || {
|
||||
@@ -156,18 +319,15 @@ pub fn run_tiled(
|
||||
break;
|
||||
};
|
||||
let mut wrong = None;
|
||||
let ran = net.run(mos, sig, &mut |rgb: &[f32]| {
|
||||
if rgb.len() != 3 * n * n {
|
||||
let ran = net.run(nr, nc, mos, sig, &mut |rgb: &[f32]| {
|
||||
if rgb.len() != 3 * nr * nc {
|
||||
wrong = Some(rgb.len());
|
||||
return;
|
||||
}
|
||||
// The tile's centre back into the frame: the frame rows it covers,
|
||||
// split across cores (each row is written by one thread only).
|
||||
let (y_lo, y_hi) = (
|
||||
(u0 + dy.saturating_sub(u0)).max(dy) - dy,
|
||||
(u0 + core).min(uh) - dy,
|
||||
);
|
||||
let (x_lo, x_hi) = ((v0.max(dx)) - dx, (v0 + core).min(uw) - dx);
|
||||
let (y_lo, y_hi) = (u0.max(dy) - dy, (u0 + cr).min(uh) - dy);
|
||||
let (x_lo, x_hi) = (v0.max(dx) - dx, (v0 + cc).min(uw) - dx);
|
||||
if y_hi > y_lo && x_hi > x_lo {
|
||||
let rows = &mut out[y_lo * w * 3..y_hi * w * 3];
|
||||
let per = (y_hi - y_lo).div_ceil(threads).max(1);
|
||||
@@ -181,7 +341,7 @@ pub fn run_tiled(
|
||||
for x in x_lo..x_hi {
|
||||
let c = x + dx + halo - v0;
|
||||
for ch in 0..3 {
|
||||
row[x * 3 + ch] = rgb[ch * n * n + r * n + c];
|
||||
row[x * 3 + ch] = rgb[ch * nr * nc + r * nc + c];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -191,14 +351,18 @@ pub fn run_tiled(
|
||||
}
|
||||
});
|
||||
if let Err(e) = ran {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
return Err(e);
|
||||
while rx_tiles.try_recv().is_ok() {}
|
||||
return Err(fail(e, k, stop));
|
||||
}
|
||||
if let Some(len) = wrong {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
return Err(crate::DenoiseError::Model(format!(
|
||||
"network returned {len} values for a {n}² tile"
|
||||
)));
|
||||
while rx_tiles.try_recv().is_ok() {}
|
||||
return Err(fail(
|
||||
crate::DenoiseError::Model(format!(
|
||||
"network returned {len} values for a {nr}×{nc} tile"
|
||||
)),
|
||||
k,
|
||||
stop,
|
||||
));
|
||||
}
|
||||
if !progress(k + 1, total) {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
@@ -236,37 +400,65 @@ mod tests {
|
||||
/// is its 2×2 quad's (R, mean G, B), averaged over the quads within
|
||||
/// `reach` quads. Purely a function of the tile, like the real one.
|
||||
struct BoxNet {
|
||||
n: usize,
|
||||
sizes: Sizes,
|
||||
reach: usize,
|
||||
/// Fails any call with more photosites than this, as a GPU out of
|
||||
/// memory does.
|
||||
fails_above: usize,
|
||||
calls: Vec<(usize, usize)>,
|
||||
}
|
||||
|
||||
fn square(n: usize, reach: usize) -> BoxNet {
|
||||
BoxNet {
|
||||
sizes: Sizes::Square(n),
|
||||
reach,
|
||||
fails_above: usize::MAX,
|
||||
calls: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
fn any(max: (usize, usize), reach: usize) -> BoxNet {
|
||||
BoxNet {
|
||||
sizes: Sizes::Any { align: 16, max },
|
||||
reach,
|
||||
fails_above: usize::MAX,
|
||||
calls: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
impl TileNet for BoxNet {
|
||||
fn tile(&self) -> usize {
|
||||
self.n
|
||||
fn sizes(&self) -> Sizes {
|
||||
self.sizes
|
||||
}
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
m: Vec<f32>,
|
||||
_s: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), crate::DenoiseError> {
|
||||
let n = self.n;
|
||||
let q = n / 2;
|
||||
self.calls.push((rows, cols));
|
||||
if rows * cols > self.fails_above {
|
||||
return Err(crate::DenoiseError::Model("out of memory".into()));
|
||||
}
|
||||
let (qr, qc) = (rows / 2, cols / 2);
|
||||
let quad = |qy: usize, qx: usize| {
|
||||
let (y, x) = (2 * qy, 2 * qx);
|
||||
[
|
||||
m[y * n + x],
|
||||
0.5 * (m[y * n + x + 1] + m[(y + 1) * n + x]),
|
||||
m[(y + 1) * n + x + 1],
|
||||
m[y * cols + x],
|
||||
0.5 * (m[y * cols + x + 1] + m[(y + 1) * cols + x]),
|
||||
m[(y + 1) * cols + x + 1],
|
||||
]
|
||||
};
|
||||
let mut out = vec![0.0; 3 * n * n];
|
||||
for qy in 0..q {
|
||||
for qx in 0..q {
|
||||
let plane = rows * cols;
|
||||
let mut out = vec![0.0; 3 * plane];
|
||||
for qy in 0..qr {
|
||||
for qx in 0..qc {
|
||||
let mut acc = [0.0f32; 3];
|
||||
let mut cnt = 0.0;
|
||||
for a in qy.saturating_sub(self.reach)..(qy + self.reach + 1).min(q) {
|
||||
for b in qx.saturating_sub(self.reach)..(qx + self.reach + 1).min(q) {
|
||||
for a in qy.saturating_sub(self.reach)..(qy + self.reach + 1).min(qr) {
|
||||
for b in qx.saturating_sub(self.reach)..(qx + self.reach + 1).min(qc) {
|
||||
let v = quad(a, b);
|
||||
for c in 0..3 {
|
||||
acc[c] += v[c];
|
||||
@@ -276,7 +468,7 @@ mod tests {
|
||||
}
|
||||
for (dy, dx) in [(0, 0), (0, 1), (1, 0), (1, 1)] {
|
||||
for c in 0..3 {
|
||||
out[c * n * n + (2 * qy + dy) * n + 2 * qx + dx] = acc[c] / cnt;
|
||||
out[c * plane + (2 * qy + dy) * cols + 2 * qx + dx] = acc[c] / cnt;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -306,10 +498,7 @@ mod tests {
|
||||
] {
|
||||
let (h, w) = (300, 410);
|
||||
let at = field(p);
|
||||
let mut net = BoxNet {
|
||||
n: 2 * HALO + 64,
|
||||
reach: 0,
|
||||
};
|
||||
let mut net = square(2 * HALO + 64, 0);
|
||||
let out = run_tiled(&mut net, h, w, p, &at, &|_, _| 0.01, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
@@ -336,14 +525,8 @@ mod tests {
|
||||
let (h, w) = (230, 170);
|
||||
for p in [CfaPattern::Rggb, CfaPattern::Bggr] {
|
||||
let at = |y: usize, x: usize| ((y * 7919 + x * 104729) % 1000) as f32 / 1000.0;
|
||||
let mut small = BoxNet {
|
||||
n: 2 * HALO + 32,
|
||||
reach: 20,
|
||||
};
|
||||
let mut big = BoxNet {
|
||||
n: 2 * HALO + 256,
|
||||
reach: 20,
|
||||
};
|
||||
let mut small = square(2 * HALO + 32, 20);
|
||||
let mut big = square(2 * HALO + 256, 20);
|
||||
let a = run_tiled(&mut small, h, w, p, &at, &|_, _| 0.0, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
@@ -359,12 +542,114 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// The same frame through square tiles, one whole-frame call, a grid of
|
||||
/// any-size tiles, and a network that runs out of memory on the whole
|
||||
/// frame and is planned again: one answer.
|
||||
#[test]
|
||||
fn any_size_tiles_give_the_square_tiles_answer() {
|
||||
let (h, w) = (230, 170);
|
||||
for p in [
|
||||
CfaPattern::Rggb,
|
||||
CfaPattern::Grbg,
|
||||
CfaPattern::Gbrg,
|
||||
CfaPattern::Bggr,
|
||||
] {
|
||||
let at = |y: usize, x: usize| ((y * 7919 + x * 104729) % 1000) as f32 / 1000.0;
|
||||
let run = |net: &mut BoxNet| {
|
||||
run_tiled(net, h, w, p, &at, &|_, _| 0.0, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap()
|
||||
};
|
||||
// A reach of 6 quads is well inside the halo, and keeps a
|
||||
// debug-build test of four phases short.
|
||||
let want = run(&mut square(2 * HALO + 32, 6));
|
||||
|
||||
let mut whole = any((4096, 4096), 6);
|
||||
let got = run(&mut whole);
|
||||
assert_eq!(whole.calls.len(), 1, "the frame fits: one call");
|
||||
assert_eq!(got, want, "{p:?}: whole frame");
|
||||
|
||||
let mut grid = any((2 * HALO + 96, 2 * HALO + 64), 6);
|
||||
let got = run(&mut grid);
|
||||
assert!(grid.calls.len() > 1);
|
||||
assert!(
|
||||
grid.calls.windows(2).all(|c| c[0] == c[1]),
|
||||
"one call shape"
|
||||
);
|
||||
assert_eq!(got, want, "{p:?}: a grid of any-size tiles");
|
||||
|
||||
let mut tight = any((4096, 4096), 6);
|
||||
tight.fails_above = (2 * HALO + 200) * (2 * HALO + 200);
|
||||
let got = run(&mut tight);
|
||||
assert_eq!(got, want, "{p:?}: planned again after a failure");
|
||||
assert!(tight.calls.len() > 2, "the whole frame failed, then tiles");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_plan_is_one_tile_when_the_frame_fits_and_the_least_work_when_not() {
|
||||
// A 6D frame with Best's halo, under the whole-frame limit: one call.
|
||||
let one = plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 6656),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(one.grid, (1, 1));
|
||||
assert_eq!((one.rows, one.cols), (4160, 5984));
|
||||
assert!(one.core.0 >= 3648 && one.core.1 >= 5472);
|
||||
// Too wide for one: the cheapest grid, every tile within the limit.
|
||||
let two = plan(
|
||||
3648,
|
||||
8192,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 6656),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert!(two.cols <= 6656 && two.rows <= 4608);
|
||||
assert_eq!(two.grid, (1, 2));
|
||||
// And always less work than today's 1408 squares.
|
||||
let squares = plan(3648, 5472, 256, Sizes::Square(1408)).unwrap();
|
||||
assert_eq!(squares.grid, (5, 7));
|
||||
assert!(one.work() * 2 < squares.work());
|
||||
// The whole-frame engine's limit on a 6 GB card: two tiles, each
|
||||
// within it, and still under half the work of the 1408 squares.
|
||||
let halves = plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 3328),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(halves.grid, (1, 2));
|
||||
assert_eq!((halves.rows, halves.cols), (4160, 3248));
|
||||
assert!(halves.work() * 2 < squares.work());
|
||||
// A limit no tile fits under.
|
||||
assert!(plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (400, 400)
|
||||
}
|
||||
)
|
||||
.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_cancelled_run_returns_nothing() {
|
||||
let mut net = BoxNet {
|
||||
n: 2 * HALO + 32,
|
||||
reach: 0,
|
||||
};
|
||||
let mut net = square(2 * HALO + 32, 0);
|
||||
let r = run_tiled(
|
||||
&mut net,
|
||||
100,
|
||||
@@ -389,11 +674,13 @@ mod timing {
|
||||
struct Null(usize, Vec<f32>);
|
||||
|
||||
impl TileNet for Null {
|
||||
fn tile(&self) -> usize {
|
||||
self.0
|
||||
fn sizes(&self) -> Sizes {
|
||||
Sizes::Square(self.0)
|
||||
}
|
||||
fn run(
|
||||
&mut self,
|
||||
_rows: usize,
|
||||
_cols: usize,
|
||||
m: Vec<f32>,
|
||||
_s: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
|
||||
@@ -68,6 +68,18 @@ pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
|
||||
)
|
||||
}
|
||||
|
||||
/// Where TensorRT keeps the engine for a whole-frame model. Its own
|
||||
/// directory per model: ONNX Runtime's engine cache key leaves the input
|
||||
/// shape out, and served one export's engine to another of the same graph
|
||||
/// with a different shape when the denoiser was first cut into pieces
|
||||
/// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are
|
||||
/// exactly that pair. The profile's largest shape is in the name for the
|
||||
/// same reason: an engine built for one range is not the next one's.
|
||||
pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
let (h, w) = crate::WHOLE_FRAME_MAX;
|
||||
model_dir(cfg, &format!("tensorrt-whole-{h}x{w}"), bytes)
|
||||
}
|
||||
|
||||
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
|
||||
/// model, and one per runtime version, which wrote it.
|
||||
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
|
||||
|
||||
@@ -21,6 +21,8 @@ use serde::{Deserialize, Serialize};
|
||||
|
||||
mod api;
|
||||
mod engines;
|
||||
// Read only when a runtime is loaded from disk (`api::install_best`).
|
||||
#[cfg_attr(not(feature = "native"), allow(dead_code))]
|
||||
mod hardware;
|
||||
mod probe;
|
||||
mod session;
|
||||
@@ -51,6 +53,36 @@ pub enum Role {
|
||||
/// levels cannot hold the shadow steps it exists to recover — so the
|
||||
/// Hexagon takes it with 16-bit activations and weights (§1.5).
|
||||
Denoiser,
|
||||
/// The same denoise networks exported with any height and width, run
|
||||
/// over a whole frame — or the fewest large tiles that fit — instead of
|
||||
/// 1408² tiles whose borders are thrown away (docs/dev/denoise.md §14).
|
||||
/// Served only where a size the graph was not compiled for costs
|
||||
/// nothing: TensorRT, through an optimisation profile up to
|
||||
/// [`WHOLE_FRAME_MAX`], and the CUDA provider. Everywhere else the
|
||||
/// fixed-tile [`Role::Denoiser`] runs; see [`whole_frame_limit`].
|
||||
WholeDenoiser,
|
||||
}
|
||||
|
||||
/// The largest input, rows × columns, a [`Role::WholeDenoiser`] session
|
||||
/// takes: TensorRT's optimisation profile is built up to it, and the tiler
|
||||
/// cuts a larger frame into the fewest tiles no bigger.
|
||||
///
|
||||
/// Sized for a 6 GB card. TensorRT plans its memory for the profile's
|
||||
/// largest shape, and at 4608 × 6656 (a whole 6D frame with Best's border
|
||||
/// and room to spare) it asked for 4.9–5.9 GB and could not build on the
|
||||
/// RTX 3050. At 15 MP a 6D frame is two tiles of 4160 × 3248: 27 MP of
|
||||
/// work for 20 MP kept, against 49 MP in 1408² tiles.
|
||||
pub const WHOLE_FRAME_MAX: (usize, usize) = (4608, 3328);
|
||||
|
||||
/// The input size TensorRT tunes a whole-frame engine for: half a 6D frame
|
||||
/// with Best's border, the tile the reference measurements run.
|
||||
pub const WHOLE_FRAME_OPT: (usize, usize) = (4160, 3248);
|
||||
|
||||
/// Whether the selected rung runs [`Role::WholeDenoiser`], and if so the
|
||||
/// largest input it takes. `None` means run the fixed tiles.
|
||||
pub fn whole_frame_limit() -> Option<(usize, usize)> {
|
||||
let rung = current_rung(&state().lock().unwrap());
|
||||
rung.serves(Role::WholeDenoiser).then_some(WHOLE_FRAME_MAX)
|
||||
}
|
||||
|
||||
/// Which numeric form of a model a session was built from.
|
||||
@@ -175,7 +207,8 @@ impl Rung {
|
||||
Role::Keypoints => Form::Int8,
|
||||
Role::Detector | Role::Landmarks => Form::A16W8,
|
||||
Role::Segmenter | Role::Scene | Role::Inpainter | Role::Denoiser => Form::A16W16,
|
||||
Role::Embedder | Role::EyeClassifier => Form::F32,
|
||||
// Not served there at all: the Hexagon takes fixed shapes.
|
||||
Role::Embedder | Role::EyeClassifier | Role::WholeDenoiser => Form::F32,
|
||||
},
|
||||
_ => Form::F32,
|
||||
}
|
||||
@@ -191,6 +224,13 @@ impl Rung {
|
||||
/// the Neural Engine is fp16, and which unit runs a graph is CoreML's
|
||||
/// choice.
|
||||
fn serves(self, role: Role) -> bool {
|
||||
// Any input size only where a new size costs nothing. MIGraphX,
|
||||
// OpenVINO and CoreML compile per shape, the Hexagon takes fixed
|
||||
// shapes only, and the CPU could but would hold gigabytes of f32
|
||||
// activations for a whole frame of Best.
|
||||
if role == Role::WholeDenoiser {
|
||||
return matches!(self, Rung::TensorRt | Rung::Cuda);
|
||||
}
|
||||
match self {
|
||||
Rung::Hexagon => !matches!(role, Role::Embedder | Role::EyeClassifier),
|
||||
Rung::CoreMl => role != Role::Embedder,
|
||||
|
||||
@@ -58,6 +58,9 @@ fn build_with(
|
||||
// write, or the directory CoreML or OpenVINO compiles into.
|
||||
let per_model = match rung {
|
||||
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
|
||||
Rung::TensorRt if role == Role::WholeDenoiser => {
|
||||
Some(crate::engines::tensorrt_whole_dir(cfg, bytes))
|
||||
}
|
||||
Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))),
|
||||
_ if ready => None,
|
||||
_ => context.clone(),
|
||||
@@ -135,6 +138,11 @@ fn providers(
|
||||
Rung::Cuda => {
|
||||
Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?)
|
||||
}
|
||||
Rung::TensorRt if role == Role::WholeDenoiser => {
|
||||
let mut b = b;
|
||||
tensorrt_whole(&mut b, per_model.expect("a whole-frame engine directory"))?;
|
||||
Ok(b.with_execution_providers([ep::CUDA::default().build()])?)
|
||||
}
|
||||
Rung::TensorRt => {
|
||||
let cache = cfg.cache_dir.join("tensorrt");
|
||||
let _ = std::fs::create_dir_all(&cache);
|
||||
@@ -251,6 +259,72 @@ fn migraphx(
|
||||
)
|
||||
}
|
||||
|
||||
/// TensorRT for a whole-frame model: one engine for every input size up to
|
||||
/// [`crate::WHOLE_FRAME_MAX`], kept in its own directory.
|
||||
///
|
||||
/// `ort`'s builder has no profile options, so this registers through the
|
||||
/// runtime's TensorRT V2 options, with the names 1.30 reads
|
||||
/// (`tensorrt_execution_provider_info.cc`): `trt_profile_{min,opt,max}_shapes`.
|
||||
/// Without a profile a dynamic input compiles a new engine per size at run
|
||||
/// time — 156 s on the first frame, measured — so the profile is the
|
||||
/// difference between a whole-frame engine and a stall. fp16, as for every
|
||||
/// role but the embedder (§7); the denoiser measured 0.00 dB from f32.
|
||||
#[cfg(not(target_os = "android"))]
|
||||
fn tensorrt_whole(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
cache: &std::path::Path,
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let _ = std::fs::create_dir_all(cache);
|
||||
let shapes = |(h, w): (usize, usize)| format!("mosaic:1x1x{h}x{w},sigma:1x1x{h}x{w}");
|
||||
let dir = cache.to_string_lossy().into_owned();
|
||||
let options = [
|
||||
("trt_fp16_enable", "1".to_string()),
|
||||
("trt_engine_cache_enable", "1".to_string()),
|
||||
("trt_engine_cache_path", dir.clone()),
|
||||
("trt_timing_cache_enable", "1".to_string()),
|
||||
("trt_timing_cache_path", dir),
|
||||
("trt_max_workspace_size", (1u64 << 30).to_string()),
|
||||
("trt_profile_min_shapes", shapes((256, 256))),
|
||||
("trt_profile_opt_shapes", shapes(crate::WHOLE_FRAME_OPT)),
|
||||
("trt_profile_max_shapes", shapes(crate::WHOLE_FRAME_MAX)),
|
||||
];
|
||||
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
|
||||
let keys = options
|
||||
.iter()
|
||||
.map(|(k, _)| cstr(k))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let values = options
|
||||
.iter()
|
||||
.map(|(_, v)| cstr(v))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
let api = ort::api();
|
||||
// SAFETY: the documented create / update / append / release sequence
|
||||
// `ort`'s own TensorRT builder makes, over arrays that outlive it; the
|
||||
// runtime copies the options into the session before the release.
|
||||
unsafe {
|
||||
let mut trt: *mut ort::sys::OrtTensorRTProviderOptionsV2 = std::ptr::null_mut();
|
||||
ort::Error::result_from_status((api.CreateTensorRTProviderOptions)(&mut trt))?;
|
||||
let result = ort::Error::result_from_status((api.UpdateTensorRTProviderOptions)(
|
||||
trt,
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
))
|
||||
.and_then(|()| {
|
||||
ort::Error::result_from_status((api.SessionOptionsAppendExecutionProvider_TensorRT_V2)(
|
||||
b.ptr_mut(),
|
||||
trt,
|
||||
))
|
||||
});
|
||||
(api.ReleaseTensorRTProviderOptions)(trt);
|
||||
result
|
||||
}
|
||||
}
|
||||
|
||||
/// OpenVINO on the GPU, compiling into `cache`.
|
||||
///
|
||||
/// The option names are those `openvino_provider_factory.cc` reads at 1.24,
|
||||
|
||||
@@ -2216,13 +2216,13 @@ mod tests {
|
||||
g.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
learned_denoise::Method::Medium.index(),
|
||||
learned_denoise::Method::Fast.index(),
|
||||
);
|
||||
let state = g.state();
|
||||
let mut h = EditGraph::default_chain();
|
||||
h.set_denoise_available(true);
|
||||
let _ = h.set_state(&state);
|
||||
assert_eq!(h.denoise_method(), learned_denoise::Method::Medium);
|
||||
assert_eq!(h.denoise_method(), learned_denoise::Method::Fast);
|
||||
assert!((h.denoise_grain() - 0.4).abs() < 1e-6);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -35,23 +35,27 @@ pub const GRAIN: ParamId = ParamId("grain");
|
||||
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// The demosaics a photograph can be developed with, in the order the
|
||||
/// sidecar numbers them. Three networks that trade time for quality — the
|
||||
/// same training, distilled into smaller students (docs/dev/denoise.md §13)
|
||||
/// — and the classical demosaic, which is no network at all.
|
||||
/// sidecar numbers them. Two networks that trade time for quality
|
||||
/// (docs/dev/denoise.md §15) and the classical demosaic, which is no network
|
||||
/// at all.
|
||||
///
|
||||
/// Until 0.24 there were four — Bilinear, Fast, Medium, Best — and the
|
||||
/// sidecar keeps their numbers: 2, which was Medium, is now Best, and 3,
|
||||
/// which was Best, is past the end and reads as the default, which is
|
||||
/// Best. Both land on the network that replaced them, with no migration.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||
pub enum Method {
|
||||
/// The classical demosaic: the noise stays.
|
||||
Bilinear,
|
||||
/// The smallest student: a quarter of the medium network's work.
|
||||
/// The smallest student: a quarter of Best's work.
|
||||
Fast,
|
||||
/// One network the size of the first release's.
|
||||
Medium,
|
||||
/// Two experts, one for flat areas and one for edges, and a gate.
|
||||
/// One network of the first release's size, taught by the mixture of
|
||||
/// experts it replaced: the mixture's edges at a third of its work.
|
||||
Best,
|
||||
}
|
||||
|
||||
impl Method {
|
||||
pub const ALL: [Method; 4] = [Method::Bilinear, Method::Fast, Method::Medium, Method::Best];
|
||||
pub const ALL: [Method; 3] = [Method::Bilinear, Method::Fast, Method::Best];
|
||||
pub const DEFAULT: Method = Method::Best;
|
||||
|
||||
/// The sidecar's number for it.
|
||||
@@ -92,7 +96,6 @@ pub(crate) static DESCRIPTOR: LazyLock<Arc<OpDescriptor>> = LazyLock::new(|| {
|
||||
vec![
|
||||
LocalizedKey("param.learned_denoise.method.bilinear"),
|
||||
LocalizedKey("param.learned_denoise.method.fast"),
|
||||
LocalizedKey("param.learned_denoise.method.medium"),
|
||||
LocalizedKey("param.learned_denoise.method.best"),
|
||||
],
|
||||
)
|
||||
@@ -117,3 +120,20 @@ pub(crate) static DESCRIPTOR: LazyLock<Arc<OpDescriptor>> = LazyLock::new(|| {
|
||||
pub fn descriptor() -> Arc<OpDescriptor> {
|
||||
DESCRIPTOR.clone()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// An edit saved before 0.24 stored Medium as 2 and Best as 3. Both
|
||||
/// now name the network that replaced them, and nothing reads as Fast
|
||||
/// or Bilinear that did not before.
|
||||
#[test]
|
||||
fn the_retired_methods_read_as_best() {
|
||||
assert_eq!(Method::from_index(0.0), Method::Bilinear);
|
||||
assert_eq!(Method::from_index(1.0), Method::Fast);
|
||||
assert_eq!(Method::from_index(2.0), Method::Best, "Medium, before 0.24");
|
||||
assert_eq!(Method::from_index(3.0), Method::Best, "Best, before 0.24");
|
||||
assert_eq!(Method::Best.index(), 2.0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -323,6 +323,10 @@ for _dir in face scene inpaint denoise; do
|
||||
for f in "${ASSETS}"/*; do
|
||||
case "$(basename "${f}")" in
|
||||
README.md) continue ;;
|
||||
# The denoisers' any-size exports run whole frames on TensorRT
|
||||
# and CUDA (denoise.md §14); the Hexagon takes fixed shapes, and
|
||||
# 16 MB of graphs it never loads stay out of the APK.
|
||||
mosaic-fast.onnx | mosaic-hq.onnx) continue ;;
|
||||
esac
|
||||
cp "${f}" "${OUT}/staging/assets/models/"
|
||||
_bundled="${_bundled} $(basename "${f}")"
|
||||
|
||||
@@ -526,3 +526,104 @@ reads the cache.
|
||||
**Packaging.** All six files in the APK (`BUNDLED`, 23 entries, +44.6 MB, ~41 MB compressed); the
|
||||
three f32 networks in the Arch package and the Windows installer, which stage `models/denoise` by
|
||||
directory.
|
||||
|
||||
## 14. A whole frame, not 1408² tiles (after 0.23.0)
|
||||
|
||||
A fixed 1408² tile is exact only past its halo, and Best's halo is 256: of every 1408² it computes
|
||||
it keeps 896², 2.47 photosites of work for each one kept (Medium and Fast keep 1024², 1.89×). On a
|
||||
GPU the network can instead run over the whole frame and its reflected border in one call, which is
|
||||
exact by the same argument (§3.4) and wastes only the border.
|
||||
|
||||
**The networks** are re-exported with any height and width (`mosaic-{best,medium,fast}.onnx` beside
|
||||
the 1408 files; darkroom-denoise `tools/export_whole.py`), from the checkpoints the shipped files
|
||||
came from. The tool refuses unless each matches its 1408 file at 1408² (max |Δ| = 0 for all three),
|
||||
matches torch at 592 × 848, and equals tiled inference over the reflected frame in f64 (≤ 7e-16).
|
||||
The APK leaves them out: the Hexagon takes fixed shapes.
|
||||
|
||||
**Where they run.** `Role::WholeDenoiser` is served by TensorRT and the CUDA provider only, the rungs
|
||||
where a new input size costs nothing at run time; MIGraphX, OpenVINO and CoreML compile per shape,
|
||||
the Hexagon takes fixed shapes, and the CPU would hold gigabytes of f32 activations. Everywhere else
|
||||
`whole_frame_limit()` is `None` and the 1408² tiles run as before. TensorRT gets an optimisation
|
||||
profile up to `WHOLE_FRAME_MAX` (4608 × 3328) — without one a dynamic input compiles a new engine per
|
||||
size at run time — through the runtime's V2 options, since `ort`'s builder has none, and keeps the
|
||||
engine in a directory per model and profile (ONNX Runtime's cache key leaves the shape out).
|
||||
|
||||
**The limit is the card's memory.** TensorRT plans its memory for the profile's largest shape. A
|
||||
profile up to a whole 6D frame with Best's border (4608 × 6656) asked for 4.9–5.9 GB and would not
|
||||
build on the 6 GB RTX 3050. At 15 MP the tiler (`tile::plan`) cuts the frame into the fewest equal
|
||||
tiles under the limit: a 6D frame is two of 4160 × 3248, 27 MP of work for 20 MP kept, against 49 MP
|
||||
in 1408² tiles. If a plan's first call fails, as a GPU out of memory does, its kept centre is halved
|
||||
and the frame planned again.
|
||||
|
||||
**Measured** 2026-10-06 on `_MG_8862` (6D, ISO 8000, 20 MP), RTX 3050 Laptop, TensorRT fp16, P3 /
|
||||
5001 MHz, another session's paused training holding 1.3 GB:
|
||||
|
||||
| Best | Network time | Against the tiles |
|
||||
|---|---|---|
|
||||
| 1408² tiles | 2.60 s | — |
|
||||
| Whole frame, two 4160 × 3248 tiles | **1.37 s** | max \|Δ\| 0.0029, mean 1.1e-5 — fp16's own spread (GPU tiles against CPU tiles: 0.0025) |
|
||||
|
||||
The first build of the whole-frame engine took 28 minutes, in the background at first launch, with
|
||||
the 1408² tiles serving meanwhile — against about 3 minutes for the fixed one; the profile's range
|
||||
is what it tunes across. A cached engine loads in about a second.
|
||||
|
||||
In PyTorch fp16 the same network over the whole 20 MP frame in one call took 3.5× less than in
|
||||
tiles, so a card that holds a whole frame gains more than the 6 GB one does; `WHOLE_FRAME_MAX` is a
|
||||
constant sized for 6 GB until the limit follows the card's memory.
|
||||
|
||||
## 15. Best becomes one network (0.24)
|
||||
|
||||
The photographer's goal for 0.24 was Best's quality in under a second on the laptop. Whole frames
|
||||
(§14) took the mixture from 2.60 s to 1.37 s and no further on a 6 GB card, so the other half was
|
||||
a single network that holds the mixture's quality at a third of its work. Methods are now
|
||||
`Bilinear`, `Fast` and `Best`; Medium and the mixture are retired.
|
||||
|
||||
**The network** is `fb-combo` (darkroom-denoise, 2026-10-07): §11's shape (32-64-128-192, blocks
|
||||
1-1-2-2, 3.2 M parameters, 48 GMAC/MP, halo 192), 20 000 steps from `fb-edges2` ← `student-m`,
|
||||
taught by the mixture at a half share, with 10 % drawn scenes and 25 % crops from the edge-rich
|
||||
cells of the training frames (branch `edge-sampling`). Scored on real photographs — the chart
|
||||
overstated the mixture's lead (a chart-sharp network was softer than Medium on real edges) — on the
|
||||
validation crops in the top quarter for sharp detail:
|
||||
|
||||
| | Edge PSNR, ISO 1600 / 6400 / 25600 | Sharpness kept | Smooth areas | Held-out PSNR, ISO 400 / 1600 / 6400 / 25600 | Chart edge |
|
||||
|---|---|---|---|---|---|
|
||||
| Mixture (Best to 0.23) | 30.71 / 29.93 / 28.55 | 0.899 / 0.868 / 0.782 | 42.61 / 41.71 / 40.12 | 40.69 / 39.84 / 38.47 / 36.71 | 0.82 |
|
||||
| `fb-combo` (Best from 0.24) | 30.67 / 29.87 / 28.49 | 0.902 / 0.874 / 0.792 | 42.54 / 41.57 / 39.85 | 40.64 / 39.78 / 38.38 / 36.54 | 0.89 |
|
||||
| Medium (to 0.23) | 30.43 / 29.68 / 28.38 | 0.896 / 0.862 / 0.776 | 42.59 / 41.69 / 40.08 | 40.59 / 39.75 / 38.40 / 36.65 | 1.30 |
|
||||
|
||||
Edges within 0.04–0.06 dB and more sharpness kept at every ISO; the known shortfall is smooth areas
|
||||
at ISO 25600, 0.27 dB. The photographer took it as it stood at 20 000 of a planned 30 000 steps.
|
||||
Others tried on the way, each short of the mixture on real photographs: `fb-sharp` (drawn scenes,
|
||||
chart-sharp but Medium's real edges), `fb-edges` (half edge-rich crops: edges close, ISO 25600
|
||||
flats −0.24 dB), `fb-edges2` (a quarter: 0.03–0.14 dB short everywhere, chart 1.33–1.47), and a
|
||||
from-scratch 24-48-96-128 between Fast and Medium.
|
||||
|
||||
**Files.** `mosaic-hq-1408.onnx`, `mosaic-hq.onnx` (any size) and `mosaic-hq-1408.a16w16.onnx` for
|
||||
the Hexagon. A new name, not Medium's or Best's: the result cache keys a model by name and size,
|
||||
and this one is byte for byte Medium's size. The tablet form lost 0.00 dB in simulated QDQ at every
|
||||
ISO and at most 0.09 dB across the ×0.5–×4 noise bracket (A16W8 0.08 / 0.26 dB; int8 −10.6 dB);
|
||||
not yet confirmed on the tablet itself.
|
||||
|
||||
**Saved edits** keep their numbers: 2, which was Medium, is now Best; 3, which was Best, is past the
|
||||
end and reads as the default, Best. Both land on the new network with no migration.
|
||||
|
||||
**Measured** 2026-10-07, `_MG_8862`, RTX 3050 Laptop, TensorRT fp16, P3 / 5001 MHz, nothing else on
|
||||
the card:
|
||||
|
||||
| Best | Network time | Peak GPU memory |
|
||||
|---|---|---|
|
||||
| mixture, 1408² tiles (0.23) | 2.60 s | — |
|
||||
| mixture, whole frame (§14) | 1.37 s | — |
|
||||
| `fb-combo`, 1408² tiles | 0.95 s | 0.55 GB |
|
||||
| `fb-combo`, whole frame (two 4160 × 3248) | **0.51–0.54 s** | 1.75 GB |
|
||||
|
||||
Decode and the hot-pixel pass add 0.4–0.5 s, so a photograph is about a second end to end. Whole
|
||||
frame against tiles: max |Δ| 0.0029, 90 dB apart — fp16's spread. The whole-frame engine's first
|
||||
build took 12 minutes (the mixture's 28); from the cache it loads in about a second, so the session
|
||||
keeps the engine's ordinary 30 s idle decay rather than unloading after each photograph: at 1.75 GB
|
||||
it fits beside the develop view on a 6 GB card, and an unload would cost the next photograph a
|
||||
second.
|
||||
|
||||
The manual's close-up for Best is still the mixture's render, which this network matches to within
|
||||
the table above; it is re-recorded with the next pass of `tools/manual/record.sh`.
|
||||
|
||||
|
||||
File diff suppressed because one or more lines are too long
+14
-14
@@ -215,22 +215,22 @@ colour that come with it, while keeping the fine detail. Look at it at
|
||||
|
||||
`Method` chooses how:
|
||||
|
||||
- `Best`, the default: two networks, one for smooth areas and one for
|
||||
edges, blended where each is better. The cleanest skies and the sharpest
|
||||
lettering, and the slowest.
|
||||
- `Medium`: one network taught by `Best`. Nearly as clean in smooth areas,
|
||||
a little softer on hard edges, in about a third of the time.
|
||||
- `Fast`: a smaller one, taught the same way. Visibly noisier at very high
|
||||
ISO than the other two, but still far cleaner than none, and quick.
|
||||
- `Best`, the default: clean skies and sharp lettering, edges kept as
|
||||
crisp as the camera recorded them.
|
||||
- `Fast`: a smaller network, taught the same way. Visibly noisier at very
|
||||
high ISO than `Best`, but still far cleaner than none, and quicker.
|
||||
- `Bilinear`: the camera's ordinary conversion, noise and all.
|
||||
|
||||
A photograph last edited with `Medium`, which earlier versions offered,
|
||||
opens with `Best`.
|
||||
|
||||
The photograph shows the camera's ordinary conversion while the network
|
||||
works, with its progress in the bar at the top, and changes when it is
|
||||
done — on a laptop's graphics card, about two and a half seconds for a
|
||||
20-megapixel photograph with `Best` and under one with the other two;
|
||||
longer on a processor alone or on the tablet. The first photograph after
|
||||
installing waits a few minutes more while the graphics card prepares each
|
||||
network, once. The result is kept, so a photograph opened again,
|
||||
done — on a laptop's graphics card, about a second for a 20-megapixel
|
||||
photograph with `Best`, reading the file included; longer on a processor
|
||||
alone or on the tablet. After installing, the graphics card spends up to a
|
||||
quarter of an hour preparing each network, once, in the background; the
|
||||
photographs developed meanwhile take a little longer. The result is kept, so a photograph opened again,
|
||||
or exported, does not wait a second time, and switching back to a method
|
||||
already used is quick.
|
||||
`Strength` eases it off: below 100 % it puts back some of what was removed,
|
||||
@@ -241,8 +241,8 @@ The lamp and railing of a night frame at ISO 8000, at 1:1, by each method:
|
||||
| Bilinear | Fast |
|
||||
|---|---|
|
||||
|  |  |
|
||||
| **Medium** | **Best** |
|
||||
|  |  |
|
||||
| **Best** | |
|
||||
|  | |
|
||||
|
||||
It works on raw files from any camera with the usual colour pattern of
|
||||
red, green and blue squares — not on JPEGs, and not yet on Fujifilm's
|
||||
|
||||
+13
-14
@@ -296,22 +296,21 @@ colour that come with it, while keeping the fine detail. Look at it at
|
||||
1:1, where noise lives.</p>
|
||||
<p><code>Method</code> chooses how:</p>
|
||||
<ul>
|
||||
<li><code>Best</code>, the default: two networks, one for smooth areas and one for
|
||||
edges, blended where each is better. The cleanest skies and the sharpest
|
||||
lettering, and the slowest.</li>
|
||||
<li><code>Medium</code>: one network taught by <code>Best</code>. Nearly as clean in smooth areas,
|
||||
a little softer on hard edges, in about a third of the time.</li>
|
||||
<li><code>Fast</code>: a smaller one, taught the same way. Visibly noisier at very high
|
||||
ISO than the other two, but still far cleaner than none, and quick.</li>
|
||||
<li><code>Best</code>, the default: clean skies and sharp lettering, edges kept as
|
||||
crisp as the camera recorded them.</li>
|
||||
<li><code>Fast</code>: a smaller network, taught the same way. Visibly noisier at very
|
||||
high ISO than <code>Best</code>, but still far cleaner than none, and quicker.</li>
|
||||
<li><code>Bilinear</code>: the camera's ordinary conversion, noise and all.</li>
|
||||
</ul>
|
||||
<p>A photograph last edited with <code>Medium</code>, which earlier versions offered,
|
||||
opens with <code>Best</code>.</p>
|
||||
<p>The photograph shows the camera's ordinary conversion while the network
|
||||
works, with its progress in the bar at the top, and changes when it is
|
||||
done — on a laptop's graphics card, about two and a half seconds for a
|
||||
20-megapixel photograph with <code>Best</code> and under one with the other two;
|
||||
longer on a processor alone or on the tablet. The first photograph after
|
||||
installing waits a few minutes more while the graphics card prepares each
|
||||
network, once. The result is kept, so a photograph opened again,
|
||||
done — on a laptop's graphics card, about a second for a 20-megapixel
|
||||
photograph with <code>Best</code>, reading the file included; longer on a processor
|
||||
alone or on the tablet. After installing, the graphics card spends up to a
|
||||
quarter of an hour preparing each network, once, in the background; the
|
||||
photographs developed meanwhile take a little longer. The result is kept, so a photograph opened again,
|
||||
or exported, does not wait a second time, and switching back to a method
|
||||
already used is quick.
|
||||
<code>Strength</code> eases it off: below 100 % it puts back some of what was removed,
|
||||
@@ -319,8 +318,8 @@ as grain without colour, for a picture that does not look too smooth.</p>
|
||||
<p>The lamp and railing of a night frame at ISO 8000, at 1:1, by each method:</p>
|
||||
<table><thead><tr><th>Bilinear</th><th>Fast</th></tr></thead><tbody>
|
||||
<tr><td><img src="media/develop-denoise-bilinear.png" alt="The railing and the lamp at ISO 8000, as the camera recorded them" /></td><td><img src="media/develop-denoise-fast.png" alt="The same, with the Fast network" /></td></tr>
|
||||
<tr><td><strong>Medium</strong></td><td><strong>Best</strong></td></tr>
|
||||
<tr><td><img src="media/develop-denoise-medium.png" alt="The same, with the Medium network" /></td><td><img src="media/develop-denoise-best.png" alt="The same, with the Best network" /></td></tr>
|
||||
<tr><td><strong>Best</strong></td><td></td></tr>
|
||||
<tr><td><img src="media/develop-denoise-best.png" alt="The same, with the Best network" /></td><td></td></tr>
|
||||
</tbody></table>
|
||||
<p>It works on raw files from any camera with the usual colour pattern of
|
||||
red, green and blue squares — not on JPEGs, and not yet on Fujifilm's
|
||||
|
||||
Binary file not shown.
+2
-2
@@ -134,9 +134,9 @@ declined, and the InsightFace grant of D13).
|
||||
|
||||
| File | Source | Trained on | Used by |
|
||||
|---|---|---|---|
|
||||
| `denoise/mosaic-best-1408.onnx` | trained in the `darkroom-denoise` repository (2026-10-04, run `final`, 30 000 steps, from the experts of runs `m2` and `edges-100`) | 1,701 of the maintainer's own base-ISO raws and 6,000 synthetic scenes the repository draws itself, with the Canon EOS 6D's measured noise added | the learned demosaic and denoise, Best (FR-DEV-3g) |
|
||||
| `denoise/mosaic-medium-1408.onnx` | distilled from `final` in the same repository (2026-10-04, run `student-m`, 20 000 steps, from `m2`) | the same | Medium |
|
||||
| `denoise/mosaic-hq-1408.onnx` | trained in the `darkroom-denoise` repository (2026-10-07, run `fb-combo`, 20 000 steps, from `fb-edges2` ← `student-m`), taught by the mixture of experts that was Best until 0.24 (run `final`) at a half share, with 10 % drawn scenes and 25 % crops from the edge-rich parts of the training frames | 1,701 of the maintainer's own base-ISO raws and 6,000 synthetic scenes the repository draws itself, with the Canon EOS 6D's measured noise added | the learned demosaic and denoise, Best (FR-DEV-3g) |
|
||||
| `denoise/mosaic-fast-1408.onnx` | distilled from `final` (2026-10-04, run `student-s`, 30 000 steps, from scratch) | the same | Fast |
|
||||
| `denoise/mosaic-{hq,fast}.onnx` | the two networks above with any height and width, by `tools/export_whole.py` in the same repository from the same checkpoints; identical to the 1408 files at 1408² | the same | the same methods, a whole frame at a time on a GPU (denoise.md §14) |
|
||||
|
||||
U-Nets of plain 3×3 convolutions, ReLU, strided and transposed
|
||||
convolutions and additive skips — no third-party architecture code or
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+10
-7
@@ -4,7 +4,7 @@
|
||||
# makes `makepkg -si` in this directory install what you are actually working
|
||||
# on. Swap `source` for a tagged tarball when there is something to release.
|
||||
pkgname=darkroom
|
||||
pkgver=0.23.0
|
||||
pkgver=0.24.0
|
||||
# Back to 1 with the version: a new pkgver is a new archive name, so there is
|
||||
# nothing for makepkg to reuse and nothing for a release number to disambiguate.
|
||||
pkgrel=1
|
||||
@@ -132,14 +132,17 @@ package() {
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/migan-512.onnx"
|
||||
|
||||
# The learned demosaic and denoise, one network per method (the project's
|
||||
# own weights, GPL — models/LICENCE.md). Same pointer check, same
|
||||
# directory.
|
||||
for _net in fast medium best; do
|
||||
_src="models/denoise/mosaic-${_net}-1408.onnx"
|
||||
# own weights, GPL — models/LICENCE.md), each at the fixed 1408 tile and
|
||||
# with any height and width for a whole frame on a GPU (denoise.md §14).
|
||||
# Same pointer check, same directory.
|
||||
for _net in fast hq; do
|
||||
for _file in "mosaic-${_net}-1408.onnx" "mosaic-${_net}.onnx"; do
|
||||
_src="models/denoise/${_file}"
|
||||
if [[ "$(stat -c%s "${_src}")" -lt 100000 ]]; then
|
||||
echo "error: the ${_net} denoise model is an LFS pointer — run: git lfs pull" >&2
|
||||
echo "error: ${_file} is an LFS pointer — run: git lfs pull" >&2
|
||||
return 1
|
||||
fi
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/mosaic-${_net}-1408.onnx"
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/${_file}"
|
||||
done
|
||||
done
|
||||
}
|
||||
|
||||
@@ -844,7 +844,7 @@ def develop_zoom():
|
||||
|
||||
# The methods in the order the scene visits them: `Best` is what the
|
||||
# photograph opens with, then each smaller network, then none.
|
||||
DENOISE_METHODS = ['Best', 'Medium', 'Fast', 'Bilinear']
|
||||
DENOISE_METHODS = ['Best', 'Fast', 'Bilinear']
|
||||
DENOISE_CLOSE_UP = 560 # pixels of canvas, square, around the lamp at 1:1
|
||||
|
||||
|
||||
|
||||
@@ -42,10 +42,8 @@ TABLE = {
|
||||
"xfeat-1024": dict(dir="keypoints", form="int8", feed="xfeat", rewrites=["unfold", "resize"]),
|
||||
"xfeat-768": dict(dir="keypoints", form="int8", feed="xfeat", rewrites=["unfold", "resize"]),
|
||||
"mosaic-fast-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
"mosaic-medium-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
# Exported with its packs as SpaceToDepth already; the rewrite finds
|
||||
# nothing to do.
|
||||
"mosaic-best-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
# Best since 0.24: one network, exported as Fast is, so the same rewrite.
|
||||
"mosaic-hq-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
# Produce the Hexagon's form of each model (docs/dev/inference.md §1.5, §5).
|
||||
#
|
||||
# ./tools/quantise-models.sh PHOTO_DIR [MODEL ...]
|
||||
# ./tools/quantise-models.sh --ranges RANGES.json mosaic-medium-1408
|
||||
# ./tools/quantise-models.sh --ranges RANGES.json mosaic-hq-1408
|
||||
#
|
||||
# Writes `<stem>.<form>.onnx` beside each canonical file under models/: a QDQ
|
||||
# graph from QNN's own quantisation config, per-channel weights, in the form
|
||||
|
||||
@@ -179,7 +179,7 @@ impl DevelopSession {
|
||||
iso: self.denoise.iso,
|
||||
cache_key: self.denoise.file_hash.as_ref().map(|h| h.key(&model)),
|
||||
model,
|
||||
halo: net.halo,
|
||||
net,
|
||||
cancel,
|
||||
})
|
||||
}
|
||||
@@ -316,8 +316,8 @@ struct Work {
|
||||
profile: Option<Vec<(f32, f32)>>,
|
||||
iso: Option<u32>,
|
||||
model: std::path::PathBuf,
|
||||
/// The context `model` needs past a tile's kept centre.
|
||||
halo: usize,
|
||||
/// The network `model` is: its context and its whole-frame sibling.
|
||||
net: dr_denoise::Shipped,
|
||||
cancel: Arc<AtomicBool>,
|
||||
cache_key: Option<String>,
|
||||
}
|
||||
@@ -351,8 +351,8 @@ impl Work {
|
||||
.map_err(|e| e.to_string())?;
|
||||
let noise = dr_denoise::noise::for_frame_with(&raw, self.profile.as_deref(), self.iso)
|
||||
.ok_or("this photograph gives no way to measure its noise")?;
|
||||
let mut net = dr_denoise::onnx::OnnxNet::from_path(&self.model, self.halo)
|
||||
.map_err(|e| e.to_string())?;
|
||||
let mut net =
|
||||
dr_denoise::onnx::OnnxNet::open(&self.model, self.net).map_err(|e| e.to_string())?;
|
||||
let rung = net
|
||||
.rung()
|
||||
.map(|r| r.label().to_string())
|
||||
|
||||
@@ -35,9 +35,11 @@ pub fn init(runtime_dirs: Vec<PathBuf>) {
|
||||
(Role::EyeClassifier, crate::library::SUNGLASSES_MODEL),
|
||||
(Role::Inpainter, crate::library::INPAINT_MODEL),
|
||||
]);
|
||||
wanted.extend(
|
||||
[dr_denoise::FAST, dr_denoise::MEDIUM, dr_denoise::BEST].map(|n| (Role::Denoiser, n.file)),
|
||||
);
|
||||
let denoisers = [dr_denoise::FAST, dr_denoise::BEST];
|
||||
wanted.extend(denoisers.map(|n| (Role::Denoiser, n.file)));
|
||||
// Their any-size siblings, which the engine compiles only on a rung
|
||||
// that runs whole frames (TensorRT; denoise.md §14).
|
||||
wanted.extend(denoisers.map(|n| (Role::WholeDenoiser, n.whole)));
|
||||
let models: Vec<(Role, PathBuf)> = wanted
|
||||
.into_iter()
|
||||
.filter_map(|(role, name)| Some((role, crate::library::shared_model(name)?)))
|
||||
|
||||
@@ -227,7 +227,6 @@ fn catalogued(key: &str) -> Option<&'static str> {
|
||||
"param.learned_denoise.method" => "Method",
|
||||
"param.learned_denoise.method.bilinear" => "Bilinear",
|
||||
"param.learned_denoise.method.fast" => "Fast",
|
||||
"param.learned_denoise.method.medium" => "Medium",
|
||||
"param.learned_denoise.method.best" => "Best",
|
||||
// How strongly: 100 % is the network's result, and less puts the
|
||||
// removed noise's brightness back as grain.
|
||||
|
||||
@@ -399,7 +399,6 @@ pub fn denoise_network(
|
||||
match method {
|
||||
Method::Bilinear => None,
|
||||
Method::Fast => Some(dr_denoise::FAST),
|
||||
Method::Medium => Some(dr_denoise::MEDIUM),
|
||||
Method::Best => Some(dr_denoise::BEST),
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user