Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8aa10cd249 | ||
|
|
c2cfacd7d3 | ||
|
|
a9271c4850 | ||
|
|
4b71ef0947 | ||
|
|
1e1aa1442b | ||
|
|
3761281dd1 | ||
|
|
9cba420fd5 | ||
|
|
56f4180347 | ||
|
|
417cba8b4d | ||
|
|
7966bf2dd8 | ||
|
|
4822991bec | ||
|
|
4b33754482 | ||
|
|
b3999bbc0c | ||
|
|
43402bfe5b | ||
|
|
cb97ebe7ac | ||
|
|
2ced6f114f | ||
|
|
85dee4375b | ||
|
|
87c405eb46 | ||
|
|
555ec0efb3 | ||
|
|
0291b80672 | ||
|
|
ef71bb3289 |
@@ -499,6 +499,12 @@ jobs:
|
||||
WANT=$(ls docs/manual/media | wc -l)
|
||||
GOT=$(ls "$INST/manual/media" | wc -l)
|
||||
[ "$GOT" = "$WANT" ] || { echo "FAIL: expected $WANT manual pictures, installed $GOT"; exit 1; }
|
||||
# Both bundled runtimes, each with its provider beside it
|
||||
# (tools/fetch-bundled-runtimes.sh).
|
||||
for f in openvino/onnxruntime.dll openvino/onnxruntime_providers_openvino.dll \
|
||||
openvino/openvino.dll webgpu/onnxruntime.dll webgpu/dxcompiler.dll; do
|
||||
[ -f "$INST/runtimes/$f" ] || { echo "FAIL: runtimes/$f not installed"; exit 1; }
|
||||
done
|
||||
wine reg query 'HKCU\Software\Microsoft\Windows\CurrentVersion\Uninstall\DarkRoom' 2>/dev/null \
|
||||
| grep -q DisplayVersion || { echo "FAIL: no uninstall registry key"; exit 1; }
|
||||
wine "$INST/darkroom.exe" --version 2>/dev/null | grep -q '^darkroom-desktop ' \
|
||||
|
||||
Generated
+26
-26
@@ -1265,7 +1265,7 @@ checksum = "f27ae1dd37df86211c42e150270f82743308803d90a6f6e6651cd730d5e1732f"
|
||||
|
||||
[[package]]
|
||||
name = "darkroom-android"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"android_logger",
|
||||
"dr-plat",
|
||||
@@ -1278,7 +1278,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "darkroom-desktop"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-plat",
|
||||
@@ -1454,7 +1454,7 @@ checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76"
|
||||
|
||||
[[package]]
|
||||
name = "dr-bench"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-catalog",
|
||||
@@ -1471,7 +1471,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-catalog"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-face",
|
||||
"dr-plat",
|
||||
@@ -1486,7 +1486,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-decode"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"env_logger",
|
||||
@@ -1500,7 +1500,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-denoise"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-gpu",
|
||||
@@ -1517,7 +1517,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-export"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-gpu",
|
||||
@@ -1536,7 +1536,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-face"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-inference-engine",
|
||||
"env_logger",
|
||||
@@ -1549,7 +1549,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-film"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"log",
|
||||
"serde",
|
||||
@@ -1558,7 +1558,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-gpu"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
"dr-decode",
|
||||
@@ -1576,7 +1576,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-inference-engine"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"libloading",
|
||||
@@ -1591,7 +1591,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-ingest"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-plat",
|
||||
"dr-types",
|
||||
@@ -1603,7 +1603,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-lens"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"lensfun",
|
||||
"log",
|
||||
@@ -1611,7 +1611,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-pano"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-inference-engine",
|
||||
@@ -1625,7 +1625,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-pipeline"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"log",
|
||||
@@ -1634,7 +1634,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-plat"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"android-native-keyring-store",
|
||||
"dr-types",
|
||||
@@ -1650,7 +1650,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-preset-xmp"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-pipeline",
|
||||
"log",
|
||||
@@ -1660,7 +1660,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-segment"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-inference-engine",
|
||||
"env_logger",
|
||||
@@ -1673,7 +1673,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-plat",
|
||||
@@ -1687,7 +1687,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync-folder"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-sync",
|
||||
@@ -1699,7 +1699,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync-nextcloud"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-decode",
|
||||
@@ -1721,7 +1721,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-thumbs"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"jpeg-encoder",
|
||||
@@ -1733,7 +1733,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-types"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -1742,7 +1742,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-ui"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"async-trait",
|
||||
@@ -1792,7 +1792,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-xmp"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"log",
|
||||
@@ -7126,7 +7126,7 @@ checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3"
|
||||
|
||||
[[package]]
|
||||
name = "traceability"
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"proc-macro2",
|
||||
|
||||
+1
-1
@@ -33,7 +33,7 @@ members = [
|
||||
exclude = ["third_party"]
|
||||
|
||||
[workspace.package]
|
||||
version = "0.22.1"
|
||||
version = "0.24.0"
|
||||
edition = "2021"
|
||||
rust-version = "1.92"
|
||||
license = "GPL-3.0-or-later"
|
||||
|
||||
@@ -201,7 +201,7 @@ controls, its place in the chain and its tests.
|
||||
|
||||
## Where it stands
|
||||
|
||||
**0.22.1**, thirty-seven tagged releases in. 193 numbered requirements in
|
||||
**0.24.0**, thirty-nine tagged releases in. 193 numbered requirements in
|
||||
scope, 85% of them claimed by code and [traced to it](docs/dev/traceability.md);
|
||||
the rest are written down rather than merely absent.
|
||||
|
||||
|
||||
@@ -338,7 +338,7 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
// sibling when the probe chose that rung and ignores it otherwise. The
|
||||
// segmenter's and XFeat's forms are compiled into the binary instead,
|
||||
// beside their f32 graphs.
|
||||
const BUNDLED: [(&std::ffi::CStr, &str); 23] = [
|
||||
const BUNDLED: [(&std::ffi::CStr, &str); 21] = [
|
||||
(c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"),
|
||||
(
|
||||
c"models/scrfd_500m_640.a16w8.onnx",
|
||||
@@ -379,15 +379,10 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
c"models/mosaic-fast-1408.a16w16.onnx",
|
||||
"mosaic-fast-1408.a16w16.onnx",
|
||||
),
|
||||
(c"models/mosaic-medium-1408.onnx", "mosaic-medium-1408.onnx"),
|
||||
(c"models/mosaic-hq-1408.onnx", "mosaic-hq-1408.onnx"),
|
||||
(
|
||||
c"models/mosaic-medium-1408.a16w16.onnx",
|
||||
"mosaic-medium-1408.a16w16.onnx",
|
||||
),
|
||||
(c"models/mosaic-best-1408.onnx", "mosaic-best-1408.onnx"),
|
||||
(
|
||||
c"models/mosaic-best-1408.a16w16.onnx",
|
||||
"mosaic-best-1408.a16w16.onnx",
|
||||
c"models/mosaic-hq-1408.a16w16.onnx",
|
||||
"mosaic-hq-1408.a16w16.onnx",
|
||||
),
|
||||
];
|
||||
|
||||
@@ -451,8 +446,13 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
// on a first launch they were not on disk until this line. The runtime
|
||||
// is in the APK's native library directory beside `libdarkroom.so`,
|
||||
// which is also where Qualcomm's DSP loader has to be pointed for the
|
||||
// Hexagon skel (docs/dev/inference.md §3, §8).
|
||||
dr_ui::inference::init(native_library_dir().into_iter().collect());
|
||||
// Hexagon skel (docs/dev/inference.md §3, §8). Two of them: the QNN
|
||||
// build, and the generic WebGPU build by its file name, which the
|
||||
// engine opens only when the first does not fit the SoC (§3.2).
|
||||
let runtimes = native_library_dir()
|
||||
.map(|dir| vec![dir.clone(), dir.join("libonnxruntime_generic.so")])
|
||||
.unwrap_or_default();
|
||||
dr_ui::inference::init(runtimes);
|
||||
}
|
||||
|
||||
/// The directory the system unpacked this APK's native libraries into.
|
||||
|
||||
@@ -106,24 +106,37 @@ fn main() -> anyhow::Result<()> {
|
||||
/// providers, or against the wrong cuDNN — and a system copy whose providers
|
||||
/// do not load is not a problem, only a slower app: the probe builds a real
|
||||
/// session before believing a provider.
|
||||
///
|
||||
/// The order breaks ties only. The engine opens every runtime on this list
|
||||
/// and loads the one whose providers fit the GPU (inference.md §3.2), so a
|
||||
/// package's bundled builds — `runtimes/openvino` and `runtimes/webgpu`
|
||||
/// beside each place a package installs to, from
|
||||
/// `tools/fetch-bundled-runtimes.sh` — sit beside a CUDA or ROCm runtime
|
||||
/// without hiding it.
|
||||
fn runtime_dirs() -> Vec<PathBuf> {
|
||||
// A place a package installs to, and the bundled runtimes under it.
|
||||
fn packaged(dirs: &mut Vec<PathBuf>, base: PathBuf) {
|
||||
dirs.push(base.join("runtimes/openvino"));
|
||||
dirs.push(base.join("runtimes/webgpu"));
|
||||
dirs.push(base);
|
||||
}
|
||||
let mut dirs = Vec::new();
|
||||
if let Some(dir) = std::env::var_os("DARKROOM_ORT_DIR") {
|
||||
dirs.push(PathBuf::from(dir));
|
||||
}
|
||||
if let Ok(exe) = std::env::current_exe() {
|
||||
if let Some(bin) = exe.parent() {
|
||||
dirs.push(bin.to_path_buf());
|
||||
dirs.push(bin.join("../lib/darkroom"));
|
||||
packaged(&mut dirs, bin.to_path_buf());
|
||||
packaged(&mut dirs, bin.join("../lib/darkroom"));
|
||||
}
|
||||
}
|
||||
dirs.push(dr_ui::inference::user_runtime_dir());
|
||||
#[cfg(target_os = "linux")]
|
||||
dirs.extend([
|
||||
PathBuf::from("/app/lib/darkroom"),
|
||||
PathBuf::from("/usr/lib/darkroom"),
|
||||
PathBuf::from("/usr/lib"),
|
||||
]);
|
||||
{
|
||||
packaged(&mut dirs, PathBuf::from("/app/lib/darkroom"));
|
||||
packaged(&mut dirs, PathBuf::from("/usr/lib/darkroom"));
|
||||
dirs.push(PathBuf::from("/usr/lib"));
|
||||
}
|
||||
// An app bundle keeps its libraries in `Contents/Frameworks`, beside
|
||||
// the `Contents/MacOS` the executable is in; then Homebrew's
|
||||
// `onnxruntime`, Apple silicon's prefix before Intel's. Homebrew's build
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
//!
|
||||
//! ```sh
|
||||
//! DARKROOM_ORT_DIR=~/.local/share/darkroom/runtime \
|
||||
//! cargo run --release -p dr-denoise --features native --example denoise_raw -- IMG.CR2 out [fast|medium|best]
|
||||
//! cargo run --release -p dr-denoise --features native --example denoise_raw -- IMG.CR2 out [fast|best]
|
||||
//! ```
|
||||
//!
|
||||
//! Decode, the app's hot-pixel pass, the frame's noise from its best source,
|
||||
@@ -11,7 +11,9 @@
|
||||
//! camera RGB — for comparison with the training repo's own path
|
||||
//! (`tools/compare_rust.py` in darkroom-denoise). `DARKROOM_ORT_DIR` points
|
||||
//! at an ONNX Runtime build; the engine's cache goes to `DR_ENGINE_CACHE` or
|
||||
//! a temporary directory.
|
||||
//! a temporary directory. The whole-frame network (`mosaic-hq.onnx` beside
|
||||
//! the fixed file) runs where the rung takes any size; `DR_PLAN=tiles` keeps
|
||||
//! the 1408² tiles anyway, to compare the two.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::time::{Duration, Instant};
|
||||
@@ -23,21 +25,26 @@ fn main() {
|
||||
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("warn")).init();
|
||||
let mut args = std::env::args().skip(1);
|
||||
let (Some(input), Some(out)) = (args.next(), args.next()) else {
|
||||
eprintln!("usage: denoise_raw RAW OUT_PREFIX [fast|medium|best]");
|
||||
eprintln!("usage: denoise_raw RAW OUT_PREFIX [fast|best]");
|
||||
std::process::exit(2);
|
||||
};
|
||||
let shipped = match args.next().as_deref() {
|
||||
None | Some("best") => dr_denoise::BEST,
|
||||
Some("medium") => dr_denoise::MEDIUM,
|
||||
Some("fast") => dr_denoise::FAST,
|
||||
Some(other) => {
|
||||
eprintln!("no network called {other}: fast, medium or best");
|
||||
eprintln!("no network called {other}: fast or best");
|
||||
std::process::exit(2);
|
||||
}
|
||||
};
|
||||
let model = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("../../models/denoise")
|
||||
.join(shipped.file);
|
||||
let whole = model.with_file_name(shipped.whole);
|
||||
let tiles_only = std::env::var("DR_PLAN").is_ok_and(|p| p == "tiles");
|
||||
let mut models = vec![(Role::Denoiser, model.clone())];
|
||||
if whole.is_file() && !tiles_only {
|
||||
models.push((Role::WholeDenoiser, whole));
|
||||
}
|
||||
let cache = std::env::var_os("DR_ENGINE_CACHE")
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| std::env::temp_dir().join("dr-denoise-engines"));
|
||||
@@ -48,7 +55,7 @@ fn main() {
|
||||
.into_iter()
|
||||
.collect(),
|
||||
cache_dir: cache,
|
||||
models: vec![(Role::Denoiser, model.clone())],
|
||||
models,
|
||||
embedded: Vec::new(),
|
||||
ceiling: None,
|
||||
threads: 0,
|
||||
@@ -101,10 +108,20 @@ fn main() {
|
||||
noise.col
|
||||
);
|
||||
|
||||
let mut net = OnnxNet::from_path(&model, shipped.halo).expect("model");
|
||||
let mut net = if tiles_only {
|
||||
OnnxNet::open_tiled(&model, shipped)
|
||||
} else {
|
||||
OnnxNet::open(&model, shipped)
|
||||
}
|
||||
.expect("model");
|
||||
println!(
|
||||
"rung {}",
|
||||
net.rung().map(|r| r.label()).unwrap_or("?")
|
||||
"rung {} · {}",
|
||||
net.rung().map(|r| r.label()).unwrap_or("?"),
|
||||
if net.whole_frame() {
|
||||
"whole frame"
|
||||
} else {
|
||||
"1408² tiles"
|
||||
}
|
||||
);
|
||||
let t = Instant::now();
|
||||
let rgb = dr_denoise::denoise(&raw, &noise, &mut net, &mut |done, total| {
|
||||
|
||||
+14
-13
@@ -25,33 +25,34 @@ pub mod tile;
|
||||
use dr_decode::RawImage;
|
||||
|
||||
pub use noise::{NoiseModel, Source};
|
||||
pub use tile::{TileNet, HALO};
|
||||
pub use tile::{Sizes, TileNet, HALO};
|
||||
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// A network the app ships in `models/denoise/`: its file, and the context
|
||||
/// it needs past a tile's kept centre (docs/dev/denoise.md §13).
|
||||
/// A network the app ships in `models/denoise/`: its fixed-tile file, the
|
||||
/// same network with any height and width for a whole frame (§14), and the
|
||||
/// context it needs past a tile's kept centre (§13).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Shipped {
|
||||
pub file: &'static str,
|
||||
pub whole: &'static str,
|
||||
pub halo: usize,
|
||||
}
|
||||
|
||||
/// The smallest student: 0.9 M parameters, 11 GMAC a megapixel.
|
||||
pub const FAST: Shipped = Shipped {
|
||||
file: "mosaic-fast-1408.onnx",
|
||||
whole: "mosaic-fast.onnx",
|
||||
halo: HALO,
|
||||
};
|
||||
/// A student of the mixture with the first release's shape: 3.2 M
|
||||
/// parameters, 48 GMAC a megapixel.
|
||||
pub const MEDIUM: Shipped = Shipped {
|
||||
file: "mosaic-medium-1408.onnx",
|
||||
halo: HALO,
|
||||
};
|
||||
/// The mixture: a flat expert, an edge expert and the gate that blends them.
|
||||
/// It reaches further than either, so it keeps a smaller centre of each tile.
|
||||
/// One network of the first release's shape, 3.2 M parameters and 48 GMAC a
|
||||
/// megapixel, taught by the mixture of experts that was Best until 0.24:
|
||||
/// its edges at a third of its work (denoise.md §15). A new file name, not
|
||||
/// the old Medium's or Best's: the result cache keys a model by its name
|
||||
/// and size, and this one is byte for byte the old Medium's size.
|
||||
pub const BEST: Shipped = Shipped {
|
||||
file: "mosaic-best-1408.onnx",
|
||||
halo: 256,
|
||||
file: "mosaic-hq-1408.onnx",
|
||||
whole: "mosaic-hq.onnx",
|
||||
halo: HALO,
|
||||
};
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
|
||||
+72
-16
@@ -10,30 +10,80 @@
|
||||
//! lost 5–9 dB (docs/dev/inference.md §1.5). That sibling is the same network
|
||||
//! with the Bayer packing spelled `SpaceToDepth`, which QNN can hold and the
|
||||
//! 6-D reshape it replaces it cannot.
|
||||
//!
|
||||
//! Each network also ships with any height and width (`mosaic-hq.onnx`
|
||||
//! beside `mosaic-hq-1408.onnx`, darkroom-denoise `tools/export_whole.py`,
|
||||
//! identical to the fixed file at 1408²). Where the rung takes any size, the
|
||||
//! frame runs whole instead of in tiles whose borders are thrown away — a
|
||||
//! 1408² tile keeps 1024², 1.89 photosites computed for each one kept
|
||||
//! (denoise.md §14).
|
||||
|
||||
use crate::tile::TileNet;
|
||||
use crate::DenoiseError;
|
||||
use dr_inference_engine::{Model, Role};
|
||||
use crate::tile::{Sizes, TileNet};
|
||||
use crate::{DenoiseError, Shipped};
|
||||
use dr_inference_engine::{Form, Model, Role};
|
||||
|
||||
/// The edge of the tile the shipped export takes.
|
||||
/// The edge of the tile the shipped fixed-shape export takes.
|
||||
pub const TILE: usize = 1408;
|
||||
|
||||
/// What a whole-frame input's sides must be multiples of: the networks pack
|
||||
/// 2×2 and halve three times, so a side is a whole number of positions at
|
||||
/// their coarsest level only in steps of 16.
|
||||
pub const ALIGN: usize = 16;
|
||||
|
||||
pub struct OnnxNet {
|
||||
model: Model,
|
||||
tile: usize,
|
||||
sizes: Sizes,
|
||||
halo: usize,
|
||||
}
|
||||
|
||||
impl OnnxNet {
|
||||
/// The network at `path`, which needs `halo` photosites of context
|
||||
/// ([`crate::Shipped::halo`]).
|
||||
pub fn from_path(path: &std::path::Path, halo: usize) -> Result<Self, DenoiseError> {
|
||||
/// The network `shipped`, whose fixed-tile file is at `path`.
|
||||
///
|
||||
/// On a rung that runs any input size (TensorRT, the CUDA provider —
|
||||
/// [`dr_inference_engine::whole_frame_limit`]) and with the any-size
|
||||
/// export installed beside it, the whole-frame network: the frame in one
|
||||
/// call, or the fewest large tiles that fit (§14). Its output is the
|
||||
/// fixed tiles' to rounding. Everywhere else, and if the whole-frame
|
||||
/// model will not open, the 1408² tiles.
|
||||
pub fn open(path: &std::path::Path, shipped: Shipped) -> Result<Self, DenoiseError> {
|
||||
if let Some(max) = dr_inference_engine::whole_frame_limit() {
|
||||
let whole = path.with_file_name(shipped.whole);
|
||||
if whole.is_file() {
|
||||
let opened = std::fs::read(&whole)
|
||||
.map_err(DenoiseError::from)
|
||||
.and_then(|bytes| {
|
||||
Ok(dr_inference_engine::open(
|
||||
Role::WholeDenoiser,
|
||||
Form::F32,
|
||||
&bytes,
|
||||
)?)
|
||||
});
|
||||
match opened {
|
||||
Ok(model) => {
|
||||
return Ok(OnnxNet {
|
||||
model,
|
||||
sizes: Sizes::Any { align: ALIGN, max },
|
||||
halo: shipped.halo,
|
||||
})
|
||||
}
|
||||
Err(e) => log::warn!(
|
||||
"learned denoise: {} will not open ({e}); running 1408² tiles",
|
||||
whole.display()
|
||||
),
|
||||
}
|
||||
}
|
||||
}
|
||||
Self::open_tiled(path, shipped)
|
||||
}
|
||||
|
||||
/// The fixed-tile network at `path`, whatever the rung: 1408² tiles.
|
||||
pub fn open_tiled(path: &std::path::Path, shipped: Shipped) -> Result<Self, DenoiseError> {
|
||||
let (path, form) = dr_inference_engine::resolve_model(Role::Denoiser, path);
|
||||
let bytes = std::fs::read(&path)?;
|
||||
Ok(OnnxNet {
|
||||
model: dr_inference_engine::open(Role::Denoiser, form, &bytes)?,
|
||||
tile: TILE,
|
||||
halo,
|
||||
sizes: Sizes::Square(TILE),
|
||||
halo: shipped.halo,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -41,11 +91,16 @@ impl OnnxNet {
|
||||
pub fn rung(&self) -> Result<dr_inference_engine::Rung, DenoiseError> {
|
||||
Ok(self.model.acquire()?.rung())
|
||||
}
|
||||
|
||||
/// Whether this is the whole-frame network.
|
||||
pub fn whole_frame(&self) -> bool {
|
||||
matches!(self.sizes, Sizes::Any { .. })
|
||||
}
|
||||
}
|
||||
|
||||
impl TileNet for OnnxNet {
|
||||
fn tile(&self) -> usize {
|
||||
self.tile
|
||||
fn sizes(&self) -> Sizes {
|
||||
self.sizes
|
||||
}
|
||||
|
||||
fn halo(&self) -> usize {
|
||||
@@ -54,12 +109,13 @@ impl TileNet for OnnxNet {
|
||||
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
mosaic: Vec<f32>,
|
||||
sigma: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), DenoiseError> {
|
||||
let n = self.tile;
|
||||
let shape = ndarray::IxDyn(&[1, 1, n, n]);
|
||||
let shape = ndarray::IxDyn(&[1, 1, rows, cols]);
|
||||
// The vectors become the tensors: no copy on the way in.
|
||||
let m = ort::value::Tensor::from_array(
|
||||
ndarray::Array::from_shape_vec(shape.clone(), mosaic)
|
||||
@@ -74,9 +130,9 @@ impl TileNet for OnnxNet {
|
||||
let outputs = session.run(ort::inputs!["mosaic" => m, "sigma" => s])?;
|
||||
let (shape, data) = outputs[0].try_extract_tensor::<f32>()?;
|
||||
let dims: Vec<i64> = shape.iter().copied().collect();
|
||||
if dims != [1, 3, n as i64, n as i64] {
|
||||
if dims != [1, 3, rows as i64, cols as i64] {
|
||||
return Err(DenoiseError::Model(format!(
|
||||
"output is {dims:?}, expected [1, 3, {n}, {n}]"
|
||||
"output is {dims:?}, expected [1, 3, {rows}, {cols}]"
|
||||
)));
|
||||
}
|
||||
// And none on the way out: the frame is written from the runtime's buffer.
|
||||
|
||||
+363
-76
@@ -1,12 +1,18 @@
|
||||
//! TRACES: FR-DEV-3g
|
||||
//! A whole frame through a fixed-shape network, exactly (denoise.md §3.4).
|
||||
//! A whole frame through a network, in tiles, exactly (denoise.md §3.4, §14).
|
||||
//!
|
||||
//! The network sees `TILE_IN`² photosites and its output is exact in the
|
||||
//! central `TILE_IN − 2·HALO`: the halo is wider than its receptive field
|
||||
//! (185 photosites, counted from the layers), so a tile's centre equals the
|
||||
//! whole frame's at the same place. The frame is extended by reflection
|
||||
//! about its edge photosites, which keeps every photosite's CFA colour, so
|
||||
//! edge tiles see real context too.
|
||||
//! A tile's output is exact in its centre: past a halo wider than the
|
||||
//! network's receptive field (185 photosites for a single network, more for
|
||||
//! the mixture), a tile's centre equals the whole frame's at the same place.
|
||||
//! The frame is extended by reflection about its edge photosites, which
|
||||
//! keeps every photosite's CFA colour, so edge tiles see real context too.
|
||||
//!
|
||||
//! **Tile sizes.** A fixed-shape network takes one square ([`Sizes::Square`],
|
||||
//! 1408², of which Best keeps 896²). A network exported with any height and
|
||||
//! width ([`Sizes::Any`]) takes the frame whole when it is small enough, and
|
||||
//! otherwise the fewest equal tiles that are: [`plan`] picks the grid that
|
||||
//! computes the fewest photosites. If the first tile of a plan fails — a
|
||||
//! GPU out of memory — the limit is halved and the frame planned again.
|
||||
//!
|
||||
//! **Phase.** The network was trained on RGGB. A frame whose pattern starts
|
||||
//! on another colour is read from one photosite up and/or left — the
|
||||
@@ -20,16 +26,26 @@ use dr_decode::CfaPattern;
|
||||
/// [`TileNet::halo`].
|
||||
pub const HALO: usize = 192;
|
||||
|
||||
/// A fixed-shape network: `mosaic` and `sigma`, `n×n` RGGB, in; `3×n×n`
|
||||
/// The tiles a network takes.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Sizes {
|
||||
/// One square, `n` photosites a side.
|
||||
Square(usize),
|
||||
/// Any rectangle whose sides are multiples of `align`, at most `max`
|
||||
/// (rows, columns).
|
||||
Any { align: usize, max: (usize, usize) },
|
||||
}
|
||||
|
||||
/// A network: `mosaic` and `sigma`, `rows×cols` RGGB, in; `3×rows×cols`
|
||||
/// planar linear camera RGB out.
|
||||
///
|
||||
/// The inputs are handed over, and the output is lent to `write` rather than
|
||||
/// returned: a 1408² tile is 24 MB of output, and copying it out of the
|
||||
/// runtime's buffer and back into the frame was a measurable share of a
|
||||
/// frame's time.
|
||||
/// returned: a 1408² tile is 24 MB of output and a whole frame 300 MB, and
|
||||
/// copying it out of the runtime's buffer and back into the frame was a
|
||||
/// measurable share of a frame's time.
|
||||
pub trait TileNet {
|
||||
/// The edge `n` of the square tile the network takes.
|
||||
fn tile(&self) -> usize;
|
||||
/// The tile sizes it takes.
|
||||
fn sizes(&self) -> Sizes;
|
||||
/// Photosites of context it needs past a tile's kept centre: at least
|
||||
/// its receptive field. [`HALO`] unless the network says otherwise.
|
||||
fn halo(&self) -> usize {
|
||||
@@ -37,12 +53,87 @@ pub trait TileNet {
|
||||
}
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
mosaic: Vec<f32>,
|
||||
sigma: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), crate::DenoiseError>;
|
||||
}
|
||||
|
||||
/// How a frame is cut: every tile `rows × cols` in, keeping its centre
|
||||
/// `core.0 × core.1` past the halo, on a `grid.0 × grid.1` grid.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Plan {
|
||||
pub rows: usize,
|
||||
pub cols: usize,
|
||||
pub core: (usize, usize),
|
||||
pub grid: (usize, usize),
|
||||
}
|
||||
|
||||
impl Plan {
|
||||
/// Photosites the network computes for the frame.
|
||||
pub fn work(&self) -> usize {
|
||||
self.grid.0 * self.grid.1 * self.rows * self.cols
|
||||
}
|
||||
}
|
||||
|
||||
/// The tiles for an `uh × uw` frame (in the network's phase) at `sizes`, or
|
||||
/// `None` when no tile fits.
|
||||
///
|
||||
/// Square tiles are today's grid. Any-size tiles are equal on each axis, so
|
||||
/// one call shape serves the frame — TensorRT's profile tunes for one, and
|
||||
/// the CUDA provider searches its algorithms once per shape — and the grid
|
||||
/// is the one with the least work: one tile whenever the frame and its
|
||||
/// halo fit under `max`.
|
||||
pub fn plan(uh: usize, uw: usize, halo: usize, sizes: Sizes) -> Option<Plan> {
|
||||
match sizes {
|
||||
Sizes::Square(n) => {
|
||||
if n <= 2 * halo || !(n - 2 * halo).is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
let core = n - 2 * halo;
|
||||
Some(Plan {
|
||||
rows: n,
|
||||
cols: n,
|
||||
core: (core, core),
|
||||
grid: (uh.div_ceil(core), uw.div_ceil(core)),
|
||||
})
|
||||
}
|
||||
Sizes::Any { align, max } => {
|
||||
// An even align keeps every tile origin on an even photosite,
|
||||
// so every tile starts on red.
|
||||
let align = align.max(2).next_multiple_of(2);
|
||||
let axis = |extent: usize, tiles: usize, limit: usize| {
|
||||
let size = (extent.div_ceil(tiles) + 2 * halo).next_multiple_of(align);
|
||||
let core = size.checked_sub(2 * halo)?;
|
||||
(size <= limit && core > 0 && core.is_multiple_of(2)).then_some((size, core))
|
||||
};
|
||||
let mut best: Option<Plan> = None;
|
||||
for gy in 1..=16 {
|
||||
let Some((rows, cy)) = axis(uh, gy, max.0) else {
|
||||
continue;
|
||||
};
|
||||
for gx in 1..=16 {
|
||||
let Some((cols, cx)) = axis(uw, gx, max.1) else {
|
||||
continue;
|
||||
};
|
||||
let p = Plan {
|
||||
rows,
|
||||
cols,
|
||||
core: (cy, cx),
|
||||
grid: (uh.div_ceil(cy), uw.div_ceil(cx)),
|
||||
};
|
||||
if best.is_none_or(|b| p.work() < b.work()) {
|
||||
best = Some(p);
|
||||
}
|
||||
}
|
||||
}
|
||||
best
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Index into `0..n` by reflection about the end photosites, any distance
|
||||
/// out: …2 1 [0 1 2 … n−1] n−2 n−3…, period `2(n−1)`. Parity is kept, which
|
||||
/// is what keeps a CFA colour.
|
||||
@@ -71,7 +162,9 @@ pub fn rggb_offset(p: CfaPattern) -> Option<(usize, usize)> {
|
||||
/// `sigma(colour, value)`, and return `h×w` interleaved RGB.
|
||||
///
|
||||
/// `progress(done, total)` is called after each tile and stops the run by
|
||||
/// returning `false`, in which case the result is `Ok(None)`.
|
||||
/// returning `false`, in which case the result is `Ok(None)`. An any-size
|
||||
/// network whose first tile fails is planned again with tiles half that
|
||||
/// size, until a tile would keep no centre; then the failure is returned.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn run_tiled(
|
||||
net: &mut dyn TileNet,
|
||||
@@ -85,40 +178,103 @@ pub fn run_tiled(
|
||||
let (dy, dx) = rggb_offset(pattern).ok_or_else(|| {
|
||||
crate::DenoiseError::Unsupported(format!("{pattern:?} is not a Bayer pattern"))
|
||||
})?;
|
||||
let (n, halo) = (net.tile(), net.halo());
|
||||
if n <= 2 * halo || !(n - 2 * halo).is_multiple_of(2) {
|
||||
return Err(crate::DenoiseError::Model(format!(
|
||||
"tile {n} leaves no even centre past a {halo} halo"
|
||||
)));
|
||||
let halo = net.halo();
|
||||
let (uh, uw) = (h + dy, w + dx);
|
||||
let mut sizes = net.sizes();
|
||||
loop {
|
||||
let plan = plan(uh, uw, halo, sizes).ok_or_else(|| {
|
||||
crate::DenoiseError::Model(format!(
|
||||
"no tile of {sizes:?} keeps a centre past a {halo} halo"
|
||||
))
|
||||
})?;
|
||||
match run_plan(net, plan, h, w, (dy, dx), halo, at, sigma, progress) {
|
||||
Err(Failed { error, first: true }) => {
|
||||
// The first call of a size is where a GPU runs out of
|
||||
// memory. Halve the larger kept centre of the tile that
|
||||
// failed — not the limit, which may be far above it, and not
|
||||
// the tile, half of which may be all halo — and plan again,
|
||||
// until no smaller tile keeps a centre.
|
||||
let Sizes::Any { align, .. } = sizes else {
|
||||
return Err(error);
|
||||
};
|
||||
let (cr, cc) = plan.core;
|
||||
let smaller = if cr >= cc {
|
||||
(cr / 2 + 2 * halo, plan.cols)
|
||||
} else {
|
||||
(plan.rows, cc / 2 + 2 * halo)
|
||||
};
|
||||
let next = Sizes::Any {
|
||||
align,
|
||||
max: smaller,
|
||||
};
|
||||
if self::plan(uh, uw, halo, next).is_none() {
|
||||
return Err(error);
|
||||
}
|
||||
log::warn!(
|
||||
"learned denoise: a {}×{} tile failed ({error}); trying tiles up to {}×{}",
|
||||
plan.rows,
|
||||
plan.cols,
|
||||
smaller.0,
|
||||
smaller.1
|
||||
);
|
||||
sizes = next;
|
||||
}
|
||||
Err(Failed { error, .. }) => return Err(error),
|
||||
Ok(done) => return Ok(done),
|
||||
}
|
||||
}
|
||||
let core = n - 2 * halo;
|
||||
}
|
||||
|
||||
/// A run that stopped on an error, and whether it was the plan's first call.
|
||||
struct Failed {
|
||||
error: crate::DenoiseError,
|
||||
first: bool,
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn run_plan(
|
||||
net: &mut dyn TileNet,
|
||||
plan: Plan,
|
||||
h: usize,
|
||||
w: usize,
|
||||
(dy, dx): (usize, usize),
|
||||
halo: usize,
|
||||
at: &(dyn Fn(usize, usize) -> f32 + Sync),
|
||||
sigma: &(dyn Fn(usize, f32) -> f32 + Sync),
|
||||
progress: &mut dyn FnMut(usize, usize) -> bool,
|
||||
) -> Result<Option<Vec<f32>>, Failed> {
|
||||
let Plan {
|
||||
rows: nr,
|
||||
cols: nc,
|
||||
core: (cr, cc),
|
||||
grid: (ty, tx),
|
||||
} = plan;
|
||||
// In unified coordinates the frame spans u ∈ [dy, dy + h), v ∈ [dx, dx + w).
|
||||
let (uh, uw) = (h + dy, w + dx);
|
||||
let (ty, tx) = (uh.div_ceil(core), uw.div_ceil(core));
|
||||
let total = ty * tx;
|
||||
let origins: Vec<(usize, usize)> = (0..ty)
|
||||
.flat_map(|i| (0..tx).map(move |j| (i * core, j * core)))
|
||||
.flat_map(|i| (0..tx).map(move |j| (i * cr, j * cc)))
|
||||
.collect();
|
||||
let threads = std::thread::available_parallelism().map_or(1, |n| n.get());
|
||||
|
||||
// One tile's mosaic and σ, gathered on every core: rows are independent.
|
||||
let gather = |u0: usize, v0: usize| {
|
||||
let mut mos = vec![0.0f32; n * n];
|
||||
let mut sig = vec![0.0f32; n * n];
|
||||
let rows_per = n.div_ceil(threads).max(1);
|
||||
let mut mos = vec![0.0f32; nr * nc];
|
||||
let mut sig = vec![0.0f32; nr * nc];
|
||||
let rows_per = nr.div_ceil(threads).max(1);
|
||||
std::thread::scope(|scope| {
|
||||
for (chunk, (m, s)) in mos
|
||||
.chunks_mut(rows_per * n)
|
||||
.zip(sig.chunks_mut(rows_per * n))
|
||||
.chunks_mut(rows_per * nc)
|
||||
.zip(sig.chunks_mut(rows_per * nc))
|
||||
.enumerate()
|
||||
{
|
||||
scope.spawn(move || {
|
||||
for (i, (mrow, srow)) in m.chunks_mut(n).zip(s.chunks_mut(n)).enumerate() {
|
||||
for (i, (mrow, srow)) in m.chunks_mut(nc).zip(s.chunks_mut(nc)).enumerate() {
|
||||
let r = chunk * rows_per + i;
|
||||
// Unified row u = u0 + r − halo; frame row y = u − dy, reflected.
|
||||
let u = u0 as isize + r as isize - halo as isize;
|
||||
let y = reflect(u - dy as isize, h);
|
||||
for c in 0..n {
|
||||
for c in 0..nc {
|
||||
let v = v0 as isize + c as isize - halo as isize;
|
||||
let x = reflect(v - dx as isize, w);
|
||||
let val = at(y, x);
|
||||
@@ -138,7 +294,14 @@ pub fn run_tiled(
|
||||
// most two tiles' inputs alive.
|
||||
let mut out = vec![0.0f32; h * w * 3];
|
||||
let stop = std::sync::atomic::AtomicBool::new(false);
|
||||
std::thread::scope(|scope| -> Result<Option<()>, crate::DenoiseError> {
|
||||
let fail = |error, k: usize, stop: &std::sync::atomic::AtomicBool| {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
Failed {
|
||||
error,
|
||||
first: k == 0,
|
||||
}
|
||||
};
|
||||
std::thread::scope(|scope| -> Result<Option<()>, Failed> {
|
||||
let (tx_tiles, rx_tiles) = std::sync::mpsc::sync_channel(1);
|
||||
let (origins, stop, gather) = (&origins, &stop, &gather);
|
||||
scope.spawn(move || {
|
||||
@@ -156,18 +319,15 @@ pub fn run_tiled(
|
||||
break;
|
||||
};
|
||||
let mut wrong = None;
|
||||
let ran = net.run(mos, sig, &mut |rgb: &[f32]| {
|
||||
if rgb.len() != 3 * n * n {
|
||||
let ran = net.run(nr, nc, mos, sig, &mut |rgb: &[f32]| {
|
||||
if rgb.len() != 3 * nr * nc {
|
||||
wrong = Some(rgb.len());
|
||||
return;
|
||||
}
|
||||
// The tile's centre back into the frame: the frame rows it covers,
|
||||
// split across cores (each row is written by one thread only).
|
||||
let (y_lo, y_hi) = (
|
||||
(u0 + dy.saturating_sub(u0)).max(dy) - dy,
|
||||
(u0 + core).min(uh) - dy,
|
||||
);
|
||||
let (x_lo, x_hi) = ((v0.max(dx)) - dx, (v0 + core).min(uw) - dx);
|
||||
let (y_lo, y_hi) = (u0.max(dy) - dy, (u0 + cr).min(uh) - dy);
|
||||
let (x_lo, x_hi) = (v0.max(dx) - dx, (v0 + cc).min(uw) - dx);
|
||||
if y_hi > y_lo && x_hi > x_lo {
|
||||
let rows = &mut out[y_lo * w * 3..y_hi * w * 3];
|
||||
let per = (y_hi - y_lo).div_ceil(threads).max(1);
|
||||
@@ -181,7 +341,7 @@ pub fn run_tiled(
|
||||
for x in x_lo..x_hi {
|
||||
let c = x + dx + halo - v0;
|
||||
for ch in 0..3 {
|
||||
row[x * 3 + ch] = rgb[ch * n * n + r * n + c];
|
||||
row[x * 3 + ch] = rgb[ch * nr * nc + r * nc + c];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -191,14 +351,18 @@ pub fn run_tiled(
|
||||
}
|
||||
});
|
||||
if let Err(e) = ran {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
return Err(e);
|
||||
while rx_tiles.try_recv().is_ok() {}
|
||||
return Err(fail(e, k, stop));
|
||||
}
|
||||
if let Some(len) = wrong {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
return Err(crate::DenoiseError::Model(format!(
|
||||
"network returned {len} values for a {n}² tile"
|
||||
)));
|
||||
while rx_tiles.try_recv().is_ok() {}
|
||||
return Err(fail(
|
||||
crate::DenoiseError::Model(format!(
|
||||
"network returned {len} values for a {nr}×{nc} tile"
|
||||
)),
|
||||
k,
|
||||
stop,
|
||||
));
|
||||
}
|
||||
if !progress(k + 1, total) {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
@@ -236,37 +400,65 @@ mod tests {
|
||||
/// is its 2×2 quad's (R, mean G, B), averaged over the quads within
|
||||
/// `reach` quads. Purely a function of the tile, like the real one.
|
||||
struct BoxNet {
|
||||
n: usize,
|
||||
sizes: Sizes,
|
||||
reach: usize,
|
||||
/// Fails any call with more photosites than this, as a GPU out of
|
||||
/// memory does.
|
||||
fails_above: usize,
|
||||
calls: Vec<(usize, usize)>,
|
||||
}
|
||||
|
||||
fn square(n: usize, reach: usize) -> BoxNet {
|
||||
BoxNet {
|
||||
sizes: Sizes::Square(n),
|
||||
reach,
|
||||
fails_above: usize::MAX,
|
||||
calls: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
fn any(max: (usize, usize), reach: usize) -> BoxNet {
|
||||
BoxNet {
|
||||
sizes: Sizes::Any { align: 16, max },
|
||||
reach,
|
||||
fails_above: usize::MAX,
|
||||
calls: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
impl TileNet for BoxNet {
|
||||
fn tile(&self) -> usize {
|
||||
self.n
|
||||
fn sizes(&self) -> Sizes {
|
||||
self.sizes
|
||||
}
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
m: Vec<f32>,
|
||||
_s: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), crate::DenoiseError> {
|
||||
let n = self.n;
|
||||
let q = n / 2;
|
||||
self.calls.push((rows, cols));
|
||||
if rows * cols > self.fails_above {
|
||||
return Err(crate::DenoiseError::Model("out of memory".into()));
|
||||
}
|
||||
let (qr, qc) = (rows / 2, cols / 2);
|
||||
let quad = |qy: usize, qx: usize| {
|
||||
let (y, x) = (2 * qy, 2 * qx);
|
||||
[
|
||||
m[y * n + x],
|
||||
0.5 * (m[y * n + x + 1] + m[(y + 1) * n + x]),
|
||||
m[(y + 1) * n + x + 1],
|
||||
m[y * cols + x],
|
||||
0.5 * (m[y * cols + x + 1] + m[(y + 1) * cols + x]),
|
||||
m[(y + 1) * cols + x + 1],
|
||||
]
|
||||
};
|
||||
let mut out = vec![0.0; 3 * n * n];
|
||||
for qy in 0..q {
|
||||
for qx in 0..q {
|
||||
let plane = rows * cols;
|
||||
let mut out = vec![0.0; 3 * plane];
|
||||
for qy in 0..qr {
|
||||
for qx in 0..qc {
|
||||
let mut acc = [0.0f32; 3];
|
||||
let mut cnt = 0.0;
|
||||
for a in qy.saturating_sub(self.reach)..(qy + self.reach + 1).min(q) {
|
||||
for b in qx.saturating_sub(self.reach)..(qx + self.reach + 1).min(q) {
|
||||
for a in qy.saturating_sub(self.reach)..(qy + self.reach + 1).min(qr) {
|
||||
for b in qx.saturating_sub(self.reach)..(qx + self.reach + 1).min(qc) {
|
||||
let v = quad(a, b);
|
||||
for c in 0..3 {
|
||||
acc[c] += v[c];
|
||||
@@ -276,7 +468,7 @@ mod tests {
|
||||
}
|
||||
for (dy, dx) in [(0, 0), (0, 1), (1, 0), (1, 1)] {
|
||||
for c in 0..3 {
|
||||
out[c * n * n + (2 * qy + dy) * n + 2 * qx + dx] = acc[c] / cnt;
|
||||
out[c * plane + (2 * qy + dy) * cols + 2 * qx + dx] = acc[c] / cnt;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -306,10 +498,7 @@ mod tests {
|
||||
] {
|
||||
let (h, w) = (300, 410);
|
||||
let at = field(p);
|
||||
let mut net = BoxNet {
|
||||
n: 2 * HALO + 64,
|
||||
reach: 0,
|
||||
};
|
||||
let mut net = square(2 * HALO + 64, 0);
|
||||
let out = run_tiled(&mut net, h, w, p, &at, &|_, _| 0.01, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
@@ -336,14 +525,8 @@ mod tests {
|
||||
let (h, w) = (230, 170);
|
||||
for p in [CfaPattern::Rggb, CfaPattern::Bggr] {
|
||||
let at = |y: usize, x: usize| ((y * 7919 + x * 104729) % 1000) as f32 / 1000.0;
|
||||
let mut small = BoxNet {
|
||||
n: 2 * HALO + 32,
|
||||
reach: 20,
|
||||
};
|
||||
let mut big = BoxNet {
|
||||
n: 2 * HALO + 256,
|
||||
reach: 20,
|
||||
};
|
||||
let mut small = square(2 * HALO + 32, 20);
|
||||
let mut big = square(2 * HALO + 256, 20);
|
||||
let a = run_tiled(&mut small, h, w, p, &at, &|_, _| 0.0, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
@@ -359,12 +542,114 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// The same frame through square tiles, one whole-frame call, a grid of
|
||||
/// any-size tiles, and a network that runs out of memory on the whole
|
||||
/// frame and is planned again: one answer.
|
||||
#[test]
|
||||
fn any_size_tiles_give_the_square_tiles_answer() {
|
||||
let (h, w) = (230, 170);
|
||||
for p in [
|
||||
CfaPattern::Rggb,
|
||||
CfaPattern::Grbg,
|
||||
CfaPattern::Gbrg,
|
||||
CfaPattern::Bggr,
|
||||
] {
|
||||
let at = |y: usize, x: usize| ((y * 7919 + x * 104729) % 1000) as f32 / 1000.0;
|
||||
let run = |net: &mut BoxNet| {
|
||||
run_tiled(net, h, w, p, &at, &|_, _| 0.0, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap()
|
||||
};
|
||||
// A reach of 6 quads is well inside the halo, and keeps a
|
||||
// debug-build test of four phases short.
|
||||
let want = run(&mut square(2 * HALO + 32, 6));
|
||||
|
||||
let mut whole = any((4096, 4096), 6);
|
||||
let got = run(&mut whole);
|
||||
assert_eq!(whole.calls.len(), 1, "the frame fits: one call");
|
||||
assert_eq!(got, want, "{p:?}: whole frame");
|
||||
|
||||
let mut grid = any((2 * HALO + 96, 2 * HALO + 64), 6);
|
||||
let got = run(&mut grid);
|
||||
assert!(grid.calls.len() > 1);
|
||||
assert!(
|
||||
grid.calls.windows(2).all(|c| c[0] == c[1]),
|
||||
"one call shape"
|
||||
);
|
||||
assert_eq!(got, want, "{p:?}: a grid of any-size tiles");
|
||||
|
||||
let mut tight = any((4096, 4096), 6);
|
||||
tight.fails_above = (2 * HALO + 200) * (2 * HALO + 200);
|
||||
let got = run(&mut tight);
|
||||
assert_eq!(got, want, "{p:?}: planned again after a failure");
|
||||
assert!(tight.calls.len() > 2, "the whole frame failed, then tiles");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_plan_is_one_tile_when_the_frame_fits_and_the_least_work_when_not() {
|
||||
// A 6D frame with Best's halo, under the whole-frame limit: one call.
|
||||
let one = plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 6656),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(one.grid, (1, 1));
|
||||
assert_eq!((one.rows, one.cols), (4160, 5984));
|
||||
assert!(one.core.0 >= 3648 && one.core.1 >= 5472);
|
||||
// Too wide for one: the cheapest grid, every tile within the limit.
|
||||
let two = plan(
|
||||
3648,
|
||||
8192,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 6656),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert!(two.cols <= 6656 && two.rows <= 4608);
|
||||
assert_eq!(two.grid, (1, 2));
|
||||
// And always less work than today's 1408 squares.
|
||||
let squares = plan(3648, 5472, 256, Sizes::Square(1408)).unwrap();
|
||||
assert_eq!(squares.grid, (5, 7));
|
||||
assert!(one.work() * 2 < squares.work());
|
||||
// The whole-frame engine's limit on a 6 GB card: two tiles, each
|
||||
// within it, and still under half the work of the 1408 squares.
|
||||
let halves = plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 3328),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(halves.grid, (1, 2));
|
||||
assert_eq!((halves.rows, halves.cols), (4160, 3248));
|
||||
assert!(halves.work() * 2 < squares.work());
|
||||
// A limit no tile fits under.
|
||||
assert!(plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (400, 400)
|
||||
}
|
||||
)
|
||||
.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_cancelled_run_returns_nothing() {
|
||||
let mut net = BoxNet {
|
||||
n: 2 * HALO + 32,
|
||||
reach: 0,
|
||||
};
|
||||
let mut net = square(2 * HALO + 32, 0);
|
||||
let r = run_tiled(
|
||||
&mut net,
|
||||
100,
|
||||
@@ -389,11 +674,13 @@ mod timing {
|
||||
struct Null(usize, Vec<f32>);
|
||||
|
||||
impl TileNet for Null {
|
||||
fn tile(&self) -> usize {
|
||||
self.0
|
||||
fn sizes(&self) -> Sizes {
|
||||
Sizes::Square(self.0)
|
||||
}
|
||||
fn run(
|
||||
&mut self,
|
||||
_rows: usize,
|
||||
_cols: usize,
|
||||
m: Vec<f32>,
|
||||
_s: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
|
||||
@@ -24,6 +24,14 @@ enum Ep {
|
||||
Cpu,
|
||||
MiGraphX,
|
||||
MiGraphXFp16,
|
||||
OpenVinoCpu,
|
||||
OpenVinoGpu,
|
||||
OpenVinoGpuFp16,
|
||||
OpenVinoNpu,
|
||||
/// Dawn's low-power adapter: the integrated GPU on a hybrid machine.
|
||||
WebGpuLow,
|
||||
/// Dawn's high-performance adapter: the discrete one, if there is one.
|
||||
WebGpuHigh,
|
||||
}
|
||||
|
||||
impl Ep {
|
||||
@@ -32,20 +40,113 @@ impl Ep {
|
||||
Ep::Cpu => "CPU",
|
||||
Ep::MiGraphX => "MIGraphX f32",
|
||||
Ep::MiGraphXFp16 => "MIGraphX fp16",
|
||||
Ep::OpenVinoCpu => "OpenVINO CPU",
|
||||
Ep::OpenVinoGpu => "OpenVINO GPU",
|
||||
Ep::OpenVinoGpuFp16 => "OpenVINO GPU16",
|
||||
Ep::OpenVinoNpu => "OpenVINO NPU",
|
||||
Ep::WebGpuLow => "WebGPU low",
|
||||
Ep::WebGpuHigh => "WebGPU high",
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a second build reads what the first one compiled.
|
||||
fn caches(self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Ep::MiGraphX
|
||||
| Ep::MiGraphXFp16
|
||||
| Ep::OpenVinoGpu
|
||||
| Ep::OpenVinoGpuFp16
|
||||
| Ep::OpenVinoNpu
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
fn build(ep: Ep, bytes: &[u8], threads: usize, cache: &Path) -> ort::Result<ort::session::Session> {
|
||||
let mut b = ort::session::Session::builder()?.with_intra_threads(threads)?;
|
||||
let dir = |sub: &str| {
|
||||
let d = cache.join(sub);
|
||||
let _ = std::fs::create_dir_all(&d);
|
||||
d.to_string_lossy().into_owned()
|
||||
};
|
||||
// `GPU` is OpenVINO's first OpenCL GPU, which on a hybrid laptop can be
|
||||
// the discrete NVIDIA one; DARKROOM_OV_GPU=GPU.1 names another.
|
||||
let gpu = std::env::var("DARKROOM_OV_GPU").unwrap_or_else(|_| "GPU".into());
|
||||
match ep {
|
||||
Ep::Cpu => {}
|
||||
Ep::MiGraphX => migraphx(&mut b, false, &cache.join("f32"))?,
|
||||
Ep::MiGraphXFp16 => migraphx(&mut b, true, &cache.join("fp16"))?,
|
||||
// Option names as `openvino_provider_factory.cc` reads them at 1.24.
|
||||
Ep::OpenVinoCpu => append(&mut b, c"OpenVINO", &[("device_type", "CPU".into())])?,
|
||||
Ep::OpenVinoGpu => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[
|
||||
("device_type", gpu.clone()),
|
||||
("precision", "FP32".into()),
|
||||
("cache_dir", dir("ov-gpu-f32")),
|
||||
],
|
||||
)?,
|
||||
Ep::OpenVinoGpuFp16 => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[
|
||||
("device_type", gpu.clone()),
|
||||
("precision", "FP16".into()),
|
||||
("cache_dir", dir("ov-gpu-fp16")),
|
||||
],
|
||||
)?,
|
||||
Ep::OpenVinoNpu => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[("device_type", "NPU".into()), ("cache_dir", dir("ov-npu"))],
|
||||
)?,
|
||||
// `webgpu_provider_options.h` at 1.27; the runtime prefixes the key.
|
||||
Ep::WebGpuLow => append(
|
||||
&mut b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "low-power".into())],
|
||||
)?,
|
||||
Ep::WebGpuHigh => append(
|
||||
&mut b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "high-performance".into())],
|
||||
)?,
|
||||
}
|
||||
b.commit_from_memory(bytes)
|
||||
}
|
||||
|
||||
/// Any provider through the generic key/value entry point.
|
||||
fn append(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
name: &std::ffi::CStr,
|
||||
options: &[(&str, String)],
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let keys: Vec<CString> = options
|
||||
.iter()
|
||||
.map(|(k, _)| CString::new(*k).unwrap())
|
||||
.collect();
|
||||
let values: Vec<CString> = options
|
||||
.iter()
|
||||
.map(|(_, v)| CString::new(v.as_bytes()).unwrap())
|
||||
.collect();
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
// SAFETY: as `migraphx` below.
|
||||
unsafe {
|
||||
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
|
||||
b.ptr_mut(),
|
||||
name.as_ptr(),
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
);
|
||||
ort::Error::result_from_status(status)
|
||||
}
|
||||
}
|
||||
|
||||
/// Register MIGraphX through the generic key/value API. `ort`'s own
|
||||
/// builder fills the legacy `OrtMIGraphXProviderOptions`, which 1.29 reads
|
||||
/// for its precision flags and nothing else: the model cache directory —
|
||||
@@ -83,19 +184,29 @@ fn migraphx(
|
||||
|
||||
/// Median of `runs` timed runs over zeros, in milliseconds, after warm-ups.
|
||||
fn time(session: &mut ort::session::Session, warmups: usize, runs: usize) -> Result<f64, String> {
|
||||
let shape: Vec<usize> = session.inputs()[0]
|
||||
.dtype()
|
||||
.tensor_shape()
|
||||
.ok_or("input is not a tensor")?
|
||||
.iter()
|
||||
.map(|&d| if d > 0 { d as usize } else { 1 })
|
||||
.collect();
|
||||
let zeros = vec![0f32; shape.iter().product()];
|
||||
// Zeros for every input, not just the first: the denoiser takes
|
||||
// `mosaic` and `sigma`. A dynamic dimension is read as 1.
|
||||
let mut inputs = Vec::new();
|
||||
for input in session.inputs() {
|
||||
let shape: Vec<usize> = input
|
||||
.dtype()
|
||||
.tensor_shape()
|
||||
.ok_or("input is not a tensor")?
|
||||
.iter()
|
||||
.map(|&d| if d > 0 { d as usize } else { 1 })
|
||||
.collect();
|
||||
let zeros = vec![0f32; shape.iter().product()];
|
||||
inputs.push((input.name().to_string(), shape, zeros));
|
||||
}
|
||||
let once = |s: &mut ort::session::Session| -> Result<f64, String> {
|
||||
let input = ort::value::Tensor::from_array((shape.clone(), zeros.clone()))
|
||||
.map_err(|e| e.to_string())?;
|
||||
let mut values = Vec::with_capacity(inputs.len());
|
||||
for (name, shape, zeros) in &inputs {
|
||||
let value = ort::value::Tensor::from_array((shape.clone(), zeros.clone()))
|
||||
.map_err(|e| e.to_string())?;
|
||||
values.push((name.clone(), ort::session::SessionInputValue::from(value)));
|
||||
}
|
||||
let t = Instant::now();
|
||||
let out = s.run(ort::inputs![input]).map_err(|e| e.to_string())?;
|
||||
let out = s.run(values).map_err(|e| e.to_string())?;
|
||||
let _ = out[0]
|
||||
.try_extract_tensor::<f32>()
|
||||
.map_err(|e| e.to_string())?;
|
||||
@@ -155,13 +266,35 @@ fn main() {
|
||||
// A compiling provider is built twice: the second build reads the
|
||||
// program the first wrote, and its time is what a launch after the
|
||||
// first costs.
|
||||
let plan = [
|
||||
(Ep::Cpu, false),
|
||||
(Ep::MiGraphX, false),
|
||||
(Ep::MiGraphX, true),
|
||||
(Ep::MiGraphXFp16, false),
|
||||
(Ep::MiGraphXFp16, true),
|
||||
];
|
||||
// DARKROOM_EPS narrows the list (`cpu,openvino,webgpu,migraphx`);
|
||||
// a runtime without a provider fails its build in a millisecond
|
||||
// anyway, so the default is all of them.
|
||||
let wanted = std::env::var("DARKROOM_EPS").unwrap_or_default();
|
||||
let on = |family: &str| wanted.is_empty() || wanted.split(',').any(|w| w == family);
|
||||
let mut plan = Vec::new();
|
||||
for (family, eps) in [
|
||||
("cpu", &[Ep::Cpu][..]),
|
||||
("migraphx", &[Ep::MiGraphX, Ep::MiGraphXFp16][..]),
|
||||
(
|
||||
"openvino",
|
||||
&[
|
||||
Ep::OpenVinoCpu,
|
||||
Ep::OpenVinoGpu,
|
||||
Ep::OpenVinoGpuFp16,
|
||||
Ep::OpenVinoNpu,
|
||||
][..],
|
||||
),
|
||||
("webgpu", &[Ep::WebGpuLow, Ep::WebGpuHigh][..]),
|
||||
] {
|
||||
if on(family) {
|
||||
for &ep in eps {
|
||||
plan.push((ep, false));
|
||||
if ep.caches() {
|
||||
plan.push((ep, true));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (ep, cached) in plan {
|
||||
let started = Instant::now();
|
||||
match build(ep, &bytes, threads, &cache) {
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
//! cargo run --release -p dr-inference-engine --features native,tract \
|
||||
//! --example ladder -- CACHE_DIR models/face/scrfd_500m_640.onnx [ROLE=MODEL.onnx ...]
|
||||
//!
|
||||
//! `DARKROOM_ORT_DIRS=a:b:c` offers several runtimes, as the app's search
|
||||
//! list does, and shows which the engine chose for this device's GPU.
|
||||
//!
|
||||
//! A bare path is a `Detector`; `denoiser=…`, `scene=…`, `inpainter=…`,
|
||||
//! `landmarks=…` (any `Role`, lower case) says otherwise, so a device can
|
||||
//! show each role taking its own form (inference.md §1.5). Each is opened
|
||||
@@ -31,9 +34,16 @@ fn main() {
|
||||
std::process::exit(2);
|
||||
}
|
||||
|
||||
// DARKROOM_ORT_DIRS lists several, colon-separated, as the app's search
|
||||
// does: the engine loads the one that fits the GPU (§3.2).
|
||||
let runtime_dirs: Vec<PathBuf> = std::env::var_os("DARKROOM_ORT_DIR")
|
||||
.map(PathBuf::from)
|
||||
.into_iter()
|
||||
.chain(
|
||||
std::env::var_os("DARKROOM_ORT_DIRS")
|
||||
.map(|v| std::env::split_paths(&v).collect::<Vec<_>>())
|
||||
.unwrap_or_default(),
|
||||
)
|
||||
.collect();
|
||||
let started = Instant::now();
|
||||
dr_inference_engine::init(dr_inference_engine::Config {
|
||||
|
||||
@@ -51,17 +51,25 @@ pub fn ensure_installed() {
|
||||
}
|
||||
}
|
||||
|
||||
/// Look for `libonnxruntime` in `dirs`, in order, and hand `ort` the first
|
||||
/// table that loads; otherwise tract. Once per process.
|
||||
/// Find every `libonnxruntime` in `dirs`, hand `ort` the table of the one
|
||||
/// that best fits this device's GPUs, and fall to tract if none loads.
|
||||
/// Once per process.
|
||||
///
|
||||
/// Best fit, not first found (§3.2): a device can hold several runtimes —
|
||||
/// the package's OpenVINO build, a CUDA build the user fetched, the
|
||||
/// distribution's ROCm build — and each carries one vendor's providers.
|
||||
/// Between equals, the earlier directory wins, as it always has, and a
|
||||
/// runtime that fits perfectly ends the search: the APK's QNN build on a
|
||||
/// Qualcomm tablet is found first, and the generic build beside it is
|
||||
/// never opened there.
|
||||
/// `DARKROOM_ORT_DIR`, when it loads, wins outright: it is how a person
|
||||
/// says which runtime they mean.
|
||||
pub fn install(dirs: &[PathBuf]) -> Runtime {
|
||||
RUNTIME
|
||||
.get_or_init(|| {
|
||||
#[cfg(feature = "native")]
|
||||
for dir in dirs {
|
||||
match load_native(dir) {
|
||||
Ok(rt) => return rt,
|
||||
Err(e) => log::info!("inference: no runtime in {}: {e}", dir.display()),
|
||||
}
|
||||
if let Some(rt) = install_best(dirs) {
|
||||
return rt;
|
||||
}
|
||||
#[cfg(not(feature = "native"))]
|
||||
let _ = dirs;
|
||||
@@ -70,6 +78,103 @@ pub fn install(dirs: &[PathBuf]) -> Runtime {
|
||||
.clone()
|
||||
}
|
||||
|
||||
/// A runtime opened to read its providers, not yet handed to `ort`.
|
||||
#[cfg(feature = "native")]
|
||||
struct Found {
|
||||
lib: libloading::Library,
|
||||
api: *const ort_sys::OrtApi,
|
||||
path: PathBuf,
|
||||
version: String,
|
||||
providers: Vec<String>,
|
||||
}
|
||||
|
||||
#[cfg(feature = "native")]
|
||||
fn install_best(dirs: &[PathBuf]) -> Option<Runtime> {
|
||||
let named = std::env::var_os("DARKROOM_ORT_DIR").map(PathBuf::from);
|
||||
let gpus = crate::hardware::detect();
|
||||
let mut found: Vec<Found> = Vec::new();
|
||||
let mut seen = std::collections::HashSet::new();
|
||||
for dir in dirs {
|
||||
match open_native(dir) {
|
||||
Ok(f) => {
|
||||
// `bin/../lib/darkroom` and `/usr/lib/darkroom` are one file.
|
||||
if !seen.insert(std::fs::canonicalize(&f.path).unwrap_or(f.path.clone())) {
|
||||
std::mem::forget(f.lib);
|
||||
continue;
|
||||
}
|
||||
log::info!(
|
||||
"inference: ONNX Runtime {} at {} offers {}",
|
||||
f.version,
|
||||
f.path.display(),
|
||||
f.providers.join(", ")
|
||||
);
|
||||
if named.as_deref() == Some(dir.as_path()) {
|
||||
found.clear();
|
||||
found.push(f);
|
||||
break;
|
||||
}
|
||||
let perfect = gpus.score(&f.providers) >= crate::hardware::PERFECT;
|
||||
found.push(f);
|
||||
if perfect {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(e) => log::info!("inference: no runtime in {}: {e}", dir.display()),
|
||||
}
|
||||
}
|
||||
let best = (0..found.len())
|
||||
.max_by_key(|&i| (gpus.score(&found[i].providers), std::cmp::Reverse(i)))?;
|
||||
let chosen = found.swap_remove(best);
|
||||
// The others stay mapped. Unloading a C++ runtime after its static
|
||||
// constructors ran is a crash at exit waiting to happen, and an
|
||||
// unused mapping costs address space, not memory.
|
||||
for other in found {
|
||||
std::mem::forget(other.lib);
|
||||
}
|
||||
log::info!("inference: chose {} for {gpus:?}", chosen.path.display());
|
||||
|
||||
// SAFETY: the table came from this library's `OrtGetApiBase`, and the
|
||||
// library is leaked below, so every pointer in the copy stays valid for
|
||||
// the life of the process.
|
||||
if !ort::set_api(unsafe { (*chosen.api).clone() }) {
|
||||
log::warn!("inference: an API table was already installed");
|
||||
std::mem::forget(chosen.lib);
|
||||
return None;
|
||||
}
|
||||
std::mem::forget(chosen.lib);
|
||||
|
||||
// Qualcomm's DSP loader finds the Hexagon skel through this variable,
|
||||
// and only through it; the runtime's own directory is where the APK
|
||||
// put it. Harmless anywhere else.
|
||||
#[cfg(target_os = "android")]
|
||||
if let Some(dir) = chosen.path.parent().filter(|d| !d.as_os_str().is_empty()) {
|
||||
std::env::set_var("ADSP_LIBRARY_PATH", dir);
|
||||
}
|
||||
|
||||
// Windows looks for a provider's own dependencies — OpenVINO's DLLs,
|
||||
// which Intel's build leaves beside it — on the DLL search path, not in
|
||||
// the provider's directory. Intel's Python shim prepends to `PATH` for
|
||||
// the same reason; so does this, before any provider loads.
|
||||
#[cfg(target_os = "windows")]
|
||||
if let Some(dir) = chosen.path.parent() {
|
||||
let old = std::env::var_os("PATH").unwrap_or_default();
|
||||
let dirs = std::iter::once(dir.to_path_buf()).chain(std::env::split_paths(&old));
|
||||
if let Ok(path) = std::env::join_paths(dirs) {
|
||||
std::env::set_var("PATH", path);
|
||||
}
|
||||
}
|
||||
|
||||
log::info!(
|
||||
"inference: ONNX Runtime {} from {}",
|
||||
chosen.version,
|
||||
chosen.path.display()
|
||||
);
|
||||
Some(Runtime::OnnxRuntime {
|
||||
path: chosen.path,
|
||||
version: chosen.version,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(feature = "tract")]
|
||||
fn install_tract() -> Runtime {
|
||||
let _ = ort::set_api(ort_tract::api());
|
||||
@@ -85,8 +190,12 @@ fn install_tract() -> Runtime {
|
||||
Runtime::Tract
|
||||
}
|
||||
|
||||
/// Open the runtime in `dir` and read what it offers. `dir` may also name
|
||||
/// the library itself — Android has two runtimes and one directory, so the
|
||||
/// second goes by its file name — and an empty path is the bare name
|
||||
/// through the system loader, which on Android is the APK's own copy.
|
||||
#[cfg(feature = "native")]
|
||||
fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
fn open_native(dir: &std::path::Path) -> Result<Found, String> {
|
||||
let name = if cfg!(target_os = "windows") {
|
||||
"onnxruntime.dll"
|
||||
} else if cfg!(any(target_os = "macos", target_os = "ios")) {
|
||||
@@ -94,18 +203,21 @@ fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
} else {
|
||||
"libonnxruntime.so"
|
||||
};
|
||||
// An empty dir means the bare name: the system loader's search, which on
|
||||
// Android includes the APK's own native libraries.
|
||||
let is_library = dir.file_name().and_then(|n| n.to_str()).is_some_and(|n| {
|
||||
n.contains("onnxruntime")
|
||||
&& (n.ends_with(".so") || n.ends_with(".dll") || n.ends_with(".dylib"))
|
||||
});
|
||||
let path = if dir.as_os_str().is_empty() {
|
||||
PathBuf::from(name)
|
||||
} else if is_library {
|
||||
dir.to_path_buf()
|
||||
} else {
|
||||
find_library(dir, name).ok_or("not present")?
|
||||
};
|
||||
|
||||
// SAFETY: the library's initialisers are ONNX Runtime's own; the symbol
|
||||
// is the documented entry point with the documented signature; the table
|
||||
// is copied out and the library handle is leaked, so every pointer in
|
||||
// the copy stays valid for the life of the process.
|
||||
// is the documented entry point with the documented signature. The
|
||||
// table pointer is valid while `lib` is, which the caller keeps.
|
||||
unsafe {
|
||||
let lib = libloading::Library::new(&path).map_err(|e| e.to_string())?;
|
||||
let get_base: libloading::Symbol<
|
||||
@@ -125,24 +237,45 @@ fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
ort_sys::ORT_API_VERSION
|
||||
));
|
||||
}
|
||||
if !ort::set_api((*api).clone()) {
|
||||
return Err("an API table was already installed".into());
|
||||
}
|
||||
std::mem::forget(lib);
|
||||
|
||||
// Qualcomm's DSP loader finds the Hexagon skel through this variable,
|
||||
// and only through it; the runtime's own directory is where the APK
|
||||
// put it. Harmless anywhere else.
|
||||
#[cfg(target_os = "android")]
|
||||
if !dir.as_os_str().is_empty() {
|
||||
std::env::set_var("ADSP_LIBRARY_PATH", dir);
|
||||
}
|
||||
|
||||
log::info!("inference: ONNX Runtime {version} from {}", path.display());
|
||||
Ok(Runtime::OnnxRuntime { path, version })
|
||||
let providers = available_providers(api);
|
||||
Ok(Found {
|
||||
lib,
|
||||
api,
|
||||
path,
|
||||
version,
|
||||
providers,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// The providers compiled into the runtime behind `api` — not the ones this
|
||||
/// device can run, which is the probe's question.
|
||||
///
|
||||
/// # Safety
|
||||
/// `api` must be a live table from `GetApi`.
|
||||
#[cfg(feature = "native")]
|
||||
unsafe fn available_providers(api: *const ort_sys::OrtApi) -> Vec<String> {
|
||||
let mut list: *mut *mut std::ffi::c_char = std::ptr::null_mut();
|
||||
let mut n: std::ffi::c_int = 0;
|
||||
let status = ((*api).GetAvailableProviders)(&mut list, &mut n);
|
||||
if !status.0.is_null() {
|
||||
((*api).ReleaseStatus)(status.0);
|
||||
return Vec::new();
|
||||
}
|
||||
let names = (0..n.max(0) as usize)
|
||||
.map(|i| {
|
||||
std::ffi::CStr::from_ptr(*list.add(i))
|
||||
.to_string_lossy()
|
||||
.into_owned()
|
||||
})
|
||||
.collect();
|
||||
let status = ((*api).ReleaseAvailableProviders)(list, n);
|
||||
if !status.0.is_null() {
|
||||
((*api).ReleaseStatus)(status.0);
|
||||
}
|
||||
names
|
||||
}
|
||||
|
||||
/// `libonnxruntime.so` in `dir`, or a versioned spelling of it —
|
||||
/// `libonnxruntime.so.1.30.0` is what the Python wheel ships, and a package
|
||||
/// that installs only the versioned file is not wrong.
|
||||
|
||||
@@ -48,12 +48,47 @@ pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
/// CoreML's own cache key leaves out the weights of a model loaded from
|
||||
/// memory (`session::coreml`), and one per runtime version, which wrote it.
|
||||
pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
model_dir(cfg, "coreml", bytes)
|
||||
}
|
||||
|
||||
/// Where OpenVINO compiles `bytes` to, at one precision. OpenVINO hashes
|
||||
/// the model it is given, weights included, but a key that leaves out
|
||||
/// what is being varied has cost a day before (CLAUDE.md, "Providers"),
|
||||
/// and a directory per model and precision costs nothing: the precision
|
||||
/// is a compile option, and the two forms are different programs.
|
||||
pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
|
||||
model_dir(
|
||||
cfg,
|
||||
if fp16 {
|
||||
"openvino/fp16"
|
||||
} else {
|
||||
"openvino/f32"
|
||||
},
|
||||
bytes,
|
||||
)
|
||||
}
|
||||
|
||||
/// Where TensorRT keeps the engine for a whole-frame model. Its own
|
||||
/// directory per model: ONNX Runtime's engine cache key leaves the input
|
||||
/// shape out, and served one export's engine to another of the same graph
|
||||
/// with a different shape when the denoiser was first cut into pieces
|
||||
/// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are
|
||||
/// exactly that pair. The profile's largest shape is in the name for the
|
||||
/// same reason: an engine built for one range is not the next one's.
|
||||
pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
let (h, w) = crate::WHOLE_FRAME_MAX;
|
||||
model_dir(cfg, &format!("tensorrt-whole-{h}x{w}"), bytes)
|
||||
}
|
||||
|
||||
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
|
||||
/// model, and one per runtime version, which wrote it.
|
||||
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
|
||||
let runtime = match crate::api::runtime() {
|
||||
crate::Runtime::OnnxRuntime { version, .. } => version,
|
||||
crate::Runtime::Tract => "tract".into(),
|
||||
};
|
||||
cfg.cache_dir
|
||||
.join("coreml")
|
||||
.join(provider)
|
||||
.join(runtime)
|
||||
.join(format!("{:016x}", hash(bytes)))
|
||||
}
|
||||
|
||||
@@ -0,0 +1,176 @@
|
||||
//! Which GPUs this device has, as far as choosing a runtime needs to know
|
||||
//! (docs/dev/inference.md §3.2).
|
||||
//!
|
||||
//! A runtime carries one vendor's providers — Intel's build has OpenVINO,
|
||||
//! the `onnxruntime-gpu` wheel CUDA and TensorRT, a ROCm build MIGraphX,
|
||||
//! Microsoft's WebGPU build the generic rung — and only one runtime loads
|
||||
//! per process. These checks are what lets `api` load the one that fits
|
||||
//! when a device has several installed. They read files, never a driver:
|
||||
//! a wrong answer costs a slower rung, which the probe still measures, and
|
||||
//! a driver call at start-up could cost the launch.
|
||||
|
||||
/// What a runtime's providers are scored against.
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||
pub struct Gpus {
|
||||
pub nvidia: bool,
|
||||
/// An AMD GPU with the ROCm kernel interface, which MIGraphX needs.
|
||||
pub amd_rocm: bool,
|
||||
pub intel: bool,
|
||||
pub qualcomm: bool,
|
||||
}
|
||||
|
||||
/// The score of a runtime whose vendor rung matches the device's GPU.
|
||||
/// Nothing beats it, so the search stops there.
|
||||
pub const PERFECT: u32 = 3;
|
||||
|
||||
impl Gpus {
|
||||
/// How well a runtime offering `providers` fits this device. The vendor
|
||||
/// rungs score above OpenVINO because a machine with an Intel iGPU and
|
||||
/// an NVIDIA or AMD card wants the card; the generic rung scores above
|
||||
/// a CPU-only build because it carries the same CPU provider and might
|
||||
/// beat it.
|
||||
pub fn score(&self, providers: &[String]) -> u32 {
|
||||
providers
|
||||
.iter()
|
||||
.map(|p| match p.as_str() {
|
||||
"TensorrtExecutionProvider" | "CUDAExecutionProvider" if self.nvidia => PERFECT,
|
||||
"MIGraphXExecutionProvider" if self.amd_rocm => PERFECT,
|
||||
"QNNExecutionProvider" if self.qualcomm => PERFECT,
|
||||
"CoreMLExecutionProvider" => PERFECT,
|
||||
"OpenVINOExecutionProvider" if self.intel => 2,
|
||||
"WebGpuExecutionProvider" => 1,
|
||||
_ => 0,
|
||||
})
|
||||
.max()
|
||||
.unwrap_or(0)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
pub fn detect() -> Gpus {
|
||||
use std::path::Path;
|
||||
// Every DRM card's PCI vendor: an Intel iGPU is `0x8086` whether or
|
||||
// not its compute driver is installed, which the probe finds out.
|
||||
let vendors: Vec<String> = std::fs::read_dir("/sys/class/drm")
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.filter(|e| {
|
||||
let name = e.file_name();
|
||||
let name = name.to_string_lossy();
|
||||
name.starts_with("card") && !name.contains('-')
|
||||
})
|
||||
.filter_map(|e| std::fs::read_to_string(e.path().join("device/vendor")).ok())
|
||||
.map(|v| v.trim().to_string())
|
||||
.collect();
|
||||
Gpus {
|
||||
nvidia: Path::new("/proc/driver/nvidia/version").exists(),
|
||||
amd_rocm: Path::new("/dev/kfd").exists(),
|
||||
intel: vendors.iter().any(|v| v == "0x8086"),
|
||||
qualcomm: false,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "windows")]
|
||||
pub fn detect() -> Gpus {
|
||||
use std::path::PathBuf;
|
||||
let root = std::env::var_os("SystemRoot")
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| PathBuf::from(r"C:\Windows"));
|
||||
let system32 = root.join("System32");
|
||||
// Intel's DCH graphics driver, integrated and Arc alike, installs
|
||||
// from `iigd_dch.inf`; its package directory is the evidence.
|
||||
let intel = std::fs::read_dir(system32.join(r"DriverStore\FileRepository"))
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.any(|e| e.file_name().to_string_lossy().starts_with("iigd_dch"));
|
||||
Gpus {
|
||||
nvidia: system32.join("nvcuda.dll").exists(),
|
||||
amd_rocm: false,
|
||||
intel,
|
||||
qualcomm: false,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
pub fn detect() -> Gpus {
|
||||
// Fail-safe: only a device that names another vendor is not Qualcomm.
|
||||
// `ro.soc.manufacturer` exists from Android 12, and a property or file
|
||||
// the app cannot read reads as nothing; nothing keeps the QNN build
|
||||
// first, as 0.22 had it, where a Qualcomm device mistaken for another
|
||||
// would trade its NPU for the generic rung. Qualcomm's FastRPC library,
|
||||
// which the Hexagon path loads anyway, overrules a name.
|
||||
let soc = crate::probe::system_property("ro.soc.manufacturer");
|
||||
let fastrpc = [
|
||||
"/vendor/lib64/libcdsprpc.so",
|
||||
"/system/vendor/lib64/libcdsprpc.so",
|
||||
]
|
||||
.iter()
|
||||
.any(|p| std::path::Path::new(p).exists());
|
||||
Gpus {
|
||||
qualcomm: qualcomm_soc(&soc) || fastrpc,
|
||||
..Gpus::default()
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `ro.soc.manufacturer` leaves the device Qualcomm's: it says so,
|
||||
/// or it says nothing.
|
||||
#[cfg(any(target_os = "android", test))]
|
||||
fn qualcomm_soc(manufacturer: &str) -> bool {
|
||||
let m = manufacturer.trim();
|
||||
m.is_empty() || m.eq_ignore_ascii_case("QTI") || m.eq_ignore_ascii_case("Qualcomm")
|
||||
}
|
||||
|
||||
#[cfg(not(any(target_os = "linux", target_os = "windows", target_os = "android")))]
|
||||
pub fn detect() -> Gpus {
|
||||
Gpus::default()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn offers(p: &[&str]) -> Vec<String> {
|
||||
p.iter().map(|s| s.to_string()).collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_a_named_other_vendor_is_not_qualcomm() {
|
||||
assert!(qualcomm_soc("QTI"));
|
||||
assert!(qualcomm_soc("Qualcomm"));
|
||||
// Unreadable, or older than Android 12: the QNN build stays first.
|
||||
assert!(qualcomm_soc(""));
|
||||
assert!(!qualcomm_soc("Mediatek"));
|
||||
assert!(!qualcomm_soc("Google"));
|
||||
assert!(!qualcomm_soc("Samsung"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_card_beats_the_integrated_gpu_and_both_beat_the_generic_rung() {
|
||||
let cpu = offers(&["CPUExecutionProvider"]);
|
||||
let nvidia = offers(&[
|
||||
"TensorrtExecutionProvider",
|
||||
"CUDAExecutionProvider",
|
||||
"CPUExecutionProvider",
|
||||
]);
|
||||
let intel = offers(&["OpenVINOExecutionProvider", "CPUExecutionProvider"]);
|
||||
let webgpu = offers(&["WebGpuExecutionProvider", "CPUExecutionProvider"]);
|
||||
let laptop = Gpus {
|
||||
nvidia: true,
|
||||
intel: true,
|
||||
..Gpus::default()
|
||||
};
|
||||
assert!(laptop.score(&nvidia) > laptop.score(&intel));
|
||||
assert!(laptop.score(&intel) > laptop.score(&webgpu));
|
||||
assert!(laptop.score(&webgpu) > laptop.score(&cpu));
|
||||
// No Intel GPU: Intel's build is worth no more than a CPU build to
|
||||
// this device, and the generic rung is worth more.
|
||||
let amd_on_windows = Gpus::default();
|
||||
assert_eq!(amd_on_windows.score(&intel), amd_on_windows.score(&cpu));
|
||||
assert!(amd_on_windows.score(&webgpu) > amd_on_windows.score(&intel));
|
||||
// A ROCm build on a machine without ROCm is a CPU build.
|
||||
let rocm = offers(&["MIGraphXExecutionProvider", "CPUExecutionProvider"]);
|
||||
assert_eq!(amd_on_windows.score(&rocm), 0);
|
||||
}
|
||||
}
|
||||
@@ -21,6 +21,9 @@ use serde::{Deserialize, Serialize};
|
||||
|
||||
mod api;
|
||||
mod engines;
|
||||
// Read only when a runtime is loaded from disk (`api::install_best`).
|
||||
#[cfg_attr(not(feature = "native"), allow(dead_code))]
|
||||
mod hardware;
|
||||
mod probe;
|
||||
mod session;
|
||||
|
||||
@@ -50,6 +53,36 @@ pub enum Role {
|
||||
/// levels cannot hold the shadow steps it exists to recover — so the
|
||||
/// Hexagon takes it with 16-bit activations and weights (§1.5).
|
||||
Denoiser,
|
||||
/// The same denoise networks exported with any height and width, run
|
||||
/// over a whole frame — or the fewest large tiles that fit — instead of
|
||||
/// 1408² tiles whose borders are thrown away (docs/dev/denoise.md §14).
|
||||
/// Served only where a size the graph was not compiled for costs
|
||||
/// nothing: TensorRT, through an optimisation profile up to
|
||||
/// [`WHOLE_FRAME_MAX`], and the CUDA provider. Everywhere else the
|
||||
/// fixed-tile [`Role::Denoiser`] runs; see [`whole_frame_limit`].
|
||||
WholeDenoiser,
|
||||
}
|
||||
|
||||
/// The largest input, rows × columns, a [`Role::WholeDenoiser`] session
|
||||
/// takes: TensorRT's optimisation profile is built up to it, and the tiler
|
||||
/// cuts a larger frame into the fewest tiles no bigger.
|
||||
///
|
||||
/// Sized for a 6 GB card. TensorRT plans its memory for the profile's
|
||||
/// largest shape, and at 4608 × 6656 (a whole 6D frame with Best's border
|
||||
/// and room to spare) it asked for 4.9–5.9 GB and could not build on the
|
||||
/// RTX 3050. At 15 MP a 6D frame is two tiles of 4160 × 3248: 27 MP of
|
||||
/// work for 20 MP kept, against 49 MP in 1408² tiles.
|
||||
pub const WHOLE_FRAME_MAX: (usize, usize) = (4608, 3328);
|
||||
|
||||
/// The input size TensorRT tunes a whole-frame engine for: half a 6D frame
|
||||
/// with Best's border, the tile the reference measurements run.
|
||||
pub const WHOLE_FRAME_OPT: (usize, usize) = (4160, 3248);
|
||||
|
||||
/// Whether the selected rung runs [`Role::WholeDenoiser`], and if so the
|
||||
/// largest input it takes. `None` means run the fixed tiles.
|
||||
pub fn whole_frame_limit() -> Option<(usize, usize)> {
|
||||
let rung = current_rung(&state().lock().unwrap());
|
||||
rung.serves(Role::WholeDenoiser).then_some(WHOLE_FRAME_MAX)
|
||||
}
|
||||
|
||||
/// Which numeric form of a model a session was built from.
|
||||
@@ -110,6 +143,16 @@ pub enum Rung {
|
||||
/// embedder stays on the CPU, as on the Hexagon: the Neural Engine
|
||||
/// computes in fp16 (§7).
|
||||
CoreMl,
|
||||
/// Intel, through OpenVINO on the integrated or Arc GPU. Desktop only.
|
||||
/// Compiles a program per model, as MIGraphX does, so the CPU is its
|
||||
/// fallback; fp16 on the same terms as TensorRT (§7).
|
||||
OpenVino,
|
||||
/// Any other GPU, through ONNX Runtime's WebGPU provider: Dawn on
|
||||
/// Vulkan, D3D12 or Metal. The generic rung, for a GPU no vendor rung
|
||||
/// covers. Measured slower than the CPU on every GPU it has been timed
|
||||
/// on (§1), so it is on the ladder for the GPUs it has not, and the
|
||||
/// probe's clock is what keeps it off the rest.
|
||||
WebGpu,
|
||||
}
|
||||
|
||||
impl Rung {
|
||||
@@ -121,6 +164,8 @@ impl Rung {
|
||||
Rung::MiGraphX => "MIGraphX",
|
||||
Rung::Hexagon => "Hexagon NPU",
|
||||
Rung::CoreMl => "CoreML",
|
||||
Rung::OpenVino => "OpenVINO",
|
||||
Rung::WebGpu => "WebGPU",
|
||||
}
|
||||
}
|
||||
|
||||
@@ -129,7 +174,13 @@ impl Rung {
|
||||
fn fallback(self) -> Rung {
|
||||
match self {
|
||||
Rung::TensorRt => Rung::Cuda,
|
||||
Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::Cuda | Rung::Cpu => Rung::Cpu,
|
||||
Rung::MiGraphX
|
||||
| Rung::Hexagon
|
||||
| Rung::CoreMl
|
||||
| Rung::OpenVino
|
||||
| Rung::WebGpu
|
||||
| Rung::Cuda
|
||||
| Rung::Cpu => Rung::Cpu,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -137,7 +188,7 @@ impl Rung {
|
||||
fn compiles(self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::OpenVino
|
||||
)
|
||||
}
|
||||
|
||||
@@ -156,7 +207,8 @@ impl Rung {
|
||||
Role::Keypoints => Form::Int8,
|
||||
Role::Detector | Role::Landmarks => Form::A16W8,
|
||||
Role::Segmenter | Role::Scene | Role::Inpainter | Role::Denoiser => Form::A16W16,
|
||||
Role::Embedder | Role::EyeClassifier => Form::F32,
|
||||
// Not served there at all: the Hexagon takes fixed shapes.
|
||||
Role::Embedder | Role::EyeClassifier | Role::WholeDenoiser => Form::F32,
|
||||
},
|
||||
_ => Form::F32,
|
||||
}
|
||||
@@ -172,6 +224,13 @@ impl Rung {
|
||||
/// the Neural Engine is fp16, and which unit runs a graph is CoreML's
|
||||
/// choice.
|
||||
fn serves(self, role: Role) -> bool {
|
||||
// Any input size only where a new size costs nothing. MIGraphX,
|
||||
// OpenVINO and CoreML compile per shape, the Hexagon takes fixed
|
||||
// shapes only, and the CPU could but would hold gigabytes of f32
|
||||
// activations for a whole frame of Best.
|
||||
if role == Role::WholeDenoiser {
|
||||
return matches!(self, Rung::TensorRt | Rung::Cuda);
|
||||
}
|
||||
match self {
|
||||
Rung::Hexagon => !matches!(role, Role::Embedder | Role::EyeClassifier),
|
||||
Rung::CoreMl => role != Role::Embedder,
|
||||
@@ -230,7 +289,7 @@ impl Status {
|
||||
pub fn line(&self) -> String {
|
||||
let form = match self.rung {
|
||||
Rung::Hexagon => " · quantised",
|
||||
Rung::TensorRt | Rung::MiGraphX => " · fp16",
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::OpenVino => " · fp16",
|
||||
_ => "",
|
||||
};
|
||||
format!("{}{} · {}", self.rung.label(), form, self.runtime.label())
|
||||
@@ -725,6 +784,8 @@ mod tests {
|
||||
Rung::TensorRt,
|
||||
Rung::MiGraphX,
|
||||
Rung::CoreMl,
|
||||
Rung::OpenVino,
|
||||
Rung::WebGpu,
|
||||
] {
|
||||
assert_eq!(rung.form(Role::Detector), F32);
|
||||
}
|
||||
@@ -765,6 +826,29 @@ mod tests {
|
||||
assert_eq!(on(&s, Role::Embedder), Rung::Cpu);
|
||||
}
|
||||
|
||||
/// OpenVINO compiles a program per model, so a request waits on the CPU
|
||||
/// until the engine thread has built it; WebGPU builds in the session
|
||||
/// and serves at once. Both take every role in f32 graphs.
|
||||
#[test]
|
||||
fn openvino_waits_for_its_program_and_webgpu_does_not() {
|
||||
let hash = engines::hash(b"detector");
|
||||
let mut s = State {
|
||||
config: Config::default(),
|
||||
cache: Cache::default(),
|
||||
probing: false,
|
||||
wanted: 0,
|
||||
};
|
||||
let on = |s: &State, rung| effective_rung(s, rung, Role::Detector, Form::F32, hash);
|
||||
assert_eq!(on(&s, Rung::OpenVino), Rung::Cpu);
|
||||
s.cache
|
||||
.compiled
|
||||
.insert(engines::key_of(Rung::OpenVino, hash));
|
||||
assert_eq!(on(&s, Rung::OpenVino), Rung::OpenVino);
|
||||
assert_eq!(on(&s, Rung::WebGpu), Rung::WebGpu);
|
||||
let embedder = effective_rung(&s, Rung::WebGpu, Role::Embedder, Form::F32, hash);
|
||||
assert_eq!(embedder, Rung::WebGpu);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_status_reports_only_the_rungs_above_the_selection() {
|
||||
let _serial = serial();
|
||||
|
||||
@@ -14,18 +14,27 @@ use crate::{api::Runtime, state, Cache, Config, Form, Role, Rung};
|
||||
|
||||
/// The rungs to try on this platform, best first, under the user's ceiling.
|
||||
fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
|
||||
// WebGPU is the generic rung (§2): it is reached only on a runtime that
|
||||
// carries it, which `api` loads where no vendor's runtime fits the
|
||||
// device, and kept only where it beats the CPU.
|
||||
#[cfg(target_os = "android")]
|
||||
let all = [Rung::Hexagon];
|
||||
let all = [Rung::Hexagon, Rung::WebGpu];
|
||||
// Unmeasured (§2 ⁵): it is on the ladder because the probe's clock and
|
||||
// `attempt` make a wrong guess cost one slow or failed probe, not a
|
||||
// slow or crashing app.
|
||||
#[cfg(target_os = "macos")]
|
||||
let all = [Rung::CoreMl];
|
||||
// A desktop has one vendor's GPU; the other vendor's providers are
|
||||
// "not enabled in this build" or a library that fails to load, and
|
||||
// either answer arrives in milliseconds.
|
||||
// A runtime carries one vendor's providers, chosen for this device's
|
||||
// GPU (`api`); the others are "not enabled in this build", and that
|
||||
// answer arrives in milliseconds.
|
||||
#[cfg(not(any(target_os = "android", target_os = "macos")))]
|
||||
let all = [Rung::TensorRt, Rung::Cuda, Rung::MiGraphX];
|
||||
let all = [
|
||||
Rung::TensorRt,
|
||||
Rung::Cuda,
|
||||
Rung::MiGraphX,
|
||||
Rung::OpenVino,
|
||||
Rung::WebGpu,
|
||||
];
|
||||
all.into_iter()
|
||||
.filter(|r| ceiling.is_none_or(|c| *r <= c))
|
||||
.collect()
|
||||
@@ -157,7 +166,16 @@ pub fn run(runtime: Runtime) {
|
||||
}
|
||||
if cache.rung.is_none() {
|
||||
cache.rung = Some(Rung::Cpu);
|
||||
cache.reason = match cache.failed.first() {
|
||||
// The rung that tried and lost, not the first one the runtime was
|
||||
// never built with: "WebGPU 150 ms, slower than the CPU" says why
|
||||
// this device is on the CPU, "TensorRT not enabled" does not.
|
||||
let tried = cache
|
||||
.failed
|
||||
.iter()
|
||||
.rev()
|
||||
.find(|(_, why)| !why.contains("in this build"))
|
||||
.or(cache.failed.first());
|
||||
cache.reason = match tried {
|
||||
Some((r, why)) => format!("{} {}", r.label(), first_line(why)),
|
||||
None => "the only rung on this platform".into(),
|
||||
};
|
||||
@@ -237,9 +255,14 @@ fn probe_model(cfg: &Config) -> Option<(Role, PathBuf)> {
|
||||
smallest(Some(Role::Detector)).or_else(|| smallest(None))
|
||||
}
|
||||
|
||||
/// Build, run once for the engine, then time three runs; the median in
|
||||
/// milliseconds and, for a compiling rung, the cache key of the engine this
|
||||
/// just built.
|
||||
/// Build, warm up, then time seven runs; the median in milliseconds and,
|
||||
/// for a compiling rung, the cache key of the engine this just built.
|
||||
///
|
||||
/// Three warm-ups, not one: an idle integrated GPU takes a few runs to
|
||||
/// raise its clock. With one, the Iris Xe's OpenVINO lost to the CPU on
|
||||
/// the smallest detector in two probes of three, where warm it is 5.8 ms
|
||||
/// against 9.5 (§1.6). The smallest detector is a GPU's worst case; the
|
||||
/// clock must not also be.
|
||||
fn time_rung(
|
||||
rung: Rung,
|
||||
role: Role,
|
||||
@@ -297,11 +320,15 @@ fn time_rung(
|
||||
.map_err(|e| e.to_string())?;
|
||||
Ok(t.elapsed().as_secs_f64() * 1e3)
|
||||
};
|
||||
run(&mut session)?;
|
||||
let mut times = [run(&mut session)?, run(&mut session)?, run(&mut session)?];
|
||||
for _ in 0..3 {
|
||||
run(&mut session)?;
|
||||
}
|
||||
let mut times = (0..7)
|
||||
.map(|_| run(&mut session))
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
times.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
||||
let key = rung.compiles().then(|| crate::engines::key(rung, &bytes));
|
||||
Ok((times[1], key))
|
||||
Ok((times[times.len() / 2], key))
|
||||
}
|
||||
|
||||
/// The part of a provider's error a person can act on. ONNX Runtime's
|
||||
@@ -382,19 +409,35 @@ fn providers_beside(runtime: &Path) -> String {
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn device_identity() -> String {
|
||||
// The NVIDIA driver's version line, or the ROCm release the AMD stack
|
||||
// came from (`rocm-core` writes it; the kernel driver has no version
|
||||
// of its own). Absent means neither.
|
||||
// The NVIDIA driver's version line, the ROCm release the AMD stack came
|
||||
// from (`rocm-core` writes it; the kernel driver has no version of its
|
||||
// own), and the OpenCL drivers registered — OpenVINO reaches the GPU
|
||||
// through one, and installing Intel's is what makes the Iris Xe a rung.
|
||||
let mut parts = Vec::new();
|
||||
if let Some(line) = std::fs::read_to_string("/proc/driver/nvidia/version")
|
||||
.ok()
|
||||
.and_then(|s| s.lines().next().map(str::to_string))
|
||||
{
|
||||
return line;
|
||||
parts.push(line);
|
||||
}
|
||||
if let Ok(rocm) = std::fs::read_to_string("/opt/rocm/.info/version") {
|
||||
return format!("rocm {}", rocm.trim());
|
||||
parts.push(format!("rocm {}", rocm.trim()));
|
||||
}
|
||||
let mut icds: Vec<String> = std::fs::read_dir("/etc/OpenCL/vendors")
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.file_name().to_string_lossy().into_owned())
|
||||
.collect();
|
||||
icds.sort();
|
||||
if !icds.is_empty() {
|
||||
parts.push(format!("opencl {}", icds.join(" ")));
|
||||
}
|
||||
if parts.is_empty() {
|
||||
"no nvidia driver, no rocm, no opencl".into()
|
||||
} else {
|
||||
parts.join("; ")
|
||||
}
|
||||
"no nvidia driver, no rocm".into()
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
@@ -409,7 +452,7 @@ fn device_identity() -> String {
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
fn system_property(name: &str) -> String {
|
||||
pub(crate) fn system_property(name: &str) -> String {
|
||||
extern "C" {
|
||||
fn __system_property_get(
|
||||
name: *const std::ffi::c_char,
|
||||
|
||||
@@ -55,9 +55,13 @@ fn build_with(
|
||||
let context = (rung == Rung::Hexagon).then(|| crate::engines::context_path(cfg, bytes));
|
||||
let ready = context.as_ref().is_some_and(|p| p.is_file());
|
||||
// What the rung keeps for this model: the context the Hexagon is to
|
||||
// write, or the directory CoreML compiles into.
|
||||
// write, or the directory CoreML or OpenVINO compiles into.
|
||||
let per_model = match rung {
|
||||
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
|
||||
Rung::TensorRt if role == Role::WholeDenoiser => {
|
||||
Some(crate::engines::tensorrt_whole_dir(cfg, bytes))
|
||||
}
|
||||
Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))),
|
||||
_ if ready => None,
|
||||
_ => context.clone(),
|
||||
};
|
||||
@@ -101,6 +105,13 @@ fn with_runtime_log(
|
||||
.with_log_level(level)?)
|
||||
}
|
||||
|
||||
/// Whether `role` runs in fp16 on a rung that offers it: everything but the
|
||||
/// embedder, whose comparability across devices is worth more than its
|
||||
/// fraction of a millisecond (§7).
|
||||
fn fp16(role: Role) -> bool {
|
||||
role != Role::Embedder
|
||||
}
|
||||
|
||||
/// The intra-op pool: what the config says, else the cores less two for
|
||||
/// the compositor and the decoder (§9). tract ignores it.
|
||||
fn threads(cfg: &Config) -> usize {
|
||||
@@ -127,6 +138,11 @@ fn providers(
|
||||
Rung::Cuda => {
|
||||
Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?)
|
||||
}
|
||||
Rung::TensorRt if role == Role::WholeDenoiser => {
|
||||
let mut b = b;
|
||||
tensorrt_whole(&mut b, per_model.expect("a whole-frame engine directory"))?;
|
||||
Ok(b.with_execution_providers([ep::CUDA::default().build()])?)
|
||||
}
|
||||
Rung::TensorRt => {
|
||||
let cache = cfg.cache_dir.join("tensorrt");
|
||||
let _ = std::fs::create_dir_all(&cache);
|
||||
@@ -137,7 +153,7 @@ fn providers(
|
||||
// (NFR-RES-2). CUDA behind it takes any node TensorRT declines.
|
||||
Ok(b.with_execution_providers([
|
||||
ep::TensorRT::default()
|
||||
.with_fp16(role != Role::Embedder)
|
||||
.with_fp16(fp16(role))
|
||||
.with_engine_cache(true)
|
||||
.with_engine_cache_path(&cache)
|
||||
.with_timing_cache(true)
|
||||
@@ -154,7 +170,7 @@ fn providers(
|
||||
// directory, keyed on the graph, the GPU and its own version
|
||||
// but not the precision: hence one directory per precision.
|
||||
// The CPU takes any node it declines.
|
||||
let fp16 = role != Role::Embedder;
|
||||
let fp16 = fp16(role);
|
||||
let cache = cfg
|
||||
.cache_dir
|
||||
.join("migraphx")
|
||||
@@ -164,6 +180,16 @@ fn providers(
|
||||
migraphx(&mut b, fp16, &cache)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::OpenVino => {
|
||||
let mut b = b;
|
||||
openvino(&mut b, fp16(role), per_model)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::WebGpu => {
|
||||
let mut b = b;
|
||||
webgpu(&mut b)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::Hexagon => unreachable!("the Hexagon rung is not on a desktop ladder"),
|
||||
}
|
||||
}
|
||||
@@ -219,15 +245,145 @@ fn migraphx(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
fp16: bool,
|
||||
cache: &std::path::Path,
|
||||
) -> ort::Result<()> {
|
||||
append(
|
||||
b,
|
||||
c"MIGraphX",
|
||||
&[
|
||||
("migraphx_fp16_enable", if fp16 { "1" } else { "0" }.into()),
|
||||
(
|
||||
"migraphx_model_cache_dir",
|
||||
cache.to_string_lossy().into_owned(),
|
||||
),
|
||||
],
|
||||
)
|
||||
}
|
||||
|
||||
/// TensorRT for a whole-frame model: one engine for every input size up to
|
||||
/// [`crate::WHOLE_FRAME_MAX`], kept in its own directory.
|
||||
///
|
||||
/// `ort`'s builder has no profile options, so this registers through the
|
||||
/// runtime's TensorRT V2 options, with the names 1.30 reads
|
||||
/// (`tensorrt_execution_provider_info.cc`): `trt_profile_{min,opt,max}_shapes`.
|
||||
/// Without a profile a dynamic input compiles a new engine per size at run
|
||||
/// time — 156 s on the first frame, measured — so the profile is the
|
||||
/// difference between a whole-frame engine and a stall. fp16, as for every
|
||||
/// role but the embedder (§7); the denoiser measured 0.00 dB from f32.
|
||||
#[cfg(not(target_os = "android"))]
|
||||
fn tensorrt_whole(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
cache: &std::path::Path,
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let keys = [c"migraphx_fp16_enable", c"migraphx_model_cache_dir"];
|
||||
let values = [
|
||||
CString::new(if fp16 { "1" } else { "0" }).unwrap(),
|
||||
CString::new(cache.to_string_lossy().as_bytes())
|
||||
.map_err(|e| ort::Error::new(e.to_string()))?,
|
||||
let _ = std::fs::create_dir_all(cache);
|
||||
let shapes = |(h, w): (usize, usize)| format!("mosaic:1x1x{h}x{w},sigma:1x1x{h}x{w}");
|
||||
let dir = cache.to_string_lossy().into_owned();
|
||||
let options = [
|
||||
("trt_fp16_enable", "1".to_string()),
|
||||
("trt_engine_cache_enable", "1".to_string()),
|
||||
("trt_engine_cache_path", dir.clone()),
|
||||
("trt_timing_cache_enable", "1".to_string()),
|
||||
("trt_timing_cache_path", dir),
|
||||
("trt_max_workspace_size", (1u64 << 30).to_string()),
|
||||
("trt_profile_min_shapes", shapes((256, 256))),
|
||||
("trt_profile_opt_shapes", shapes(crate::WHOLE_FRAME_OPT)),
|
||||
("trt_profile_max_shapes", shapes(crate::WHOLE_FRAME_MAX)),
|
||||
];
|
||||
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
|
||||
let keys = options
|
||||
.iter()
|
||||
.map(|(k, _)| cstr(k))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let values = options
|
||||
.iter()
|
||||
.map(|(_, v)| cstr(v))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
let api = ort::api();
|
||||
// SAFETY: the documented create / update / append / release sequence
|
||||
// `ort`'s own TensorRT builder makes, over arrays that outlive it; the
|
||||
// runtime copies the options into the session before the release.
|
||||
unsafe {
|
||||
let mut trt: *mut ort::sys::OrtTensorRTProviderOptionsV2 = std::ptr::null_mut();
|
||||
ort::Error::result_from_status((api.CreateTensorRTProviderOptions)(&mut trt))?;
|
||||
let result = ort::Error::result_from_status((api.UpdateTensorRTProviderOptions)(
|
||||
trt,
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
))
|
||||
.and_then(|()| {
|
||||
ort::Error::result_from_status((api.SessionOptionsAppendExecutionProvider_TensorRT_V2)(
|
||||
b.ptr_mut(),
|
||||
trt,
|
||||
))
|
||||
});
|
||||
(api.ReleaseTensorRTProviderOptions)(trt);
|
||||
result
|
||||
}
|
||||
}
|
||||
|
||||
/// OpenVINO on the GPU, compiling into `cache`.
|
||||
///
|
||||
/// The option names are those `openvino_provider_factory.cc` reads at 1.24,
|
||||
/// the version of Intel's `onnxruntime-openvino` build. `GPU` is OpenVINO's
|
||||
/// first OpenCL GPU: the Intel one on a hybrid laptop with both drivers
|
||||
/// installed, but an NVIDIA card through its OpenCL when Intel's is absent
|
||||
/// — slower than the CPU there, and rejected by the probe's clock. The
|
||||
/// precision is always named: the GPU plugin's own default is fp16, and
|
||||
/// the embedder must not get it (§7).
|
||||
#[cfg(not(target_os = "android"))]
|
||||
fn openvino(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
fp16: bool,
|
||||
cache: Option<&std::path::Path>,
|
||||
) -> ort::Result<()> {
|
||||
let mut options = vec![
|
||||
("device_type", "GPU".to_string()),
|
||||
("precision", if fp16 { "FP16" } else { "FP32" }.into()),
|
||||
];
|
||||
if let Some(dir) = cache {
|
||||
let _ = std::fs::create_dir_all(dir);
|
||||
options.push(("cache_dir", dir.to_string_lossy().into_owned()));
|
||||
}
|
||||
append(b, c"OpenVINO", &options)
|
||||
}
|
||||
|
||||
/// WebGPU on the high-performance adapter: the discrete GPU where there is
|
||||
/// one, since the integrated one on a machine with both is the one this
|
||||
/// rung is least likely to beat the CPU on. The key is as
|
||||
/// `webgpu_provider_options.h` spells it, without the `ep.<name>.` prefix
|
||||
/// the runtime adds.
|
||||
fn webgpu(b: &mut ort::session::builder::SessionBuilder) -> ort::Result<()> {
|
||||
append(
|
||||
b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "high-performance".to_string())],
|
||||
)
|
||||
}
|
||||
|
||||
/// Register the provider `name` with `options` through the runtime's
|
||||
/// generic key/value entry point, which takes every provider by its short
|
||||
/// name and reads options at the runtime's own version — not at the
|
||||
/// version `ort`'s builders were written against (CLAUDE.md, "Providers").
|
||||
fn append(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
name: &std::ffi::CStr,
|
||||
options: &[(&str, String)],
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
|
||||
let keys = options
|
||||
.iter()
|
||||
.map(|(k, _)| cstr(k))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let values = options
|
||||
.iter()
|
||||
.map(|(_, v)| cstr(v))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
// SAFETY: the documented C call over arrays that outlive it; the
|
||||
@@ -235,7 +391,7 @@ fn migraphx(
|
||||
unsafe {
|
||||
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
|
||||
b.ptr_mut(),
|
||||
c"MIGraphX".as_ptr(),
|
||||
name.as_ptr(),
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
@@ -277,7 +433,12 @@ fn providers(
|
||||
.build()
|
||||
.error_on_failure()])?)
|
||||
}
|
||||
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl => {
|
||||
Rung::WebGpu => {
|
||||
let mut b = b;
|
||||
webgpu(&mut b)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl | Rung::OpenVino => {
|
||||
unreachable!("no desktop rung on Android")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2216,13 +2216,13 @@ mod tests {
|
||||
g.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
learned_denoise::Method::Medium.index(),
|
||||
learned_denoise::Method::Fast.index(),
|
||||
);
|
||||
let state = g.state();
|
||||
let mut h = EditGraph::default_chain();
|
||||
h.set_denoise_available(true);
|
||||
let _ = h.set_state(&state);
|
||||
assert_eq!(h.denoise_method(), learned_denoise::Method::Medium);
|
||||
assert_eq!(h.denoise_method(), learned_denoise::Method::Fast);
|
||||
assert!((h.denoise_grain() - 0.4).abs() < 1e-6);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -35,23 +35,27 @@ pub const GRAIN: ParamId = ParamId("grain");
|
||||
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// The demosaics a photograph can be developed with, in the order the
|
||||
/// sidecar numbers them. Three networks that trade time for quality — the
|
||||
/// same training, distilled into smaller students (docs/dev/denoise.md §13)
|
||||
/// — and the classical demosaic, which is no network at all.
|
||||
/// sidecar numbers them. Two networks that trade time for quality
|
||||
/// (docs/dev/denoise.md §15) and the classical demosaic, which is no network
|
||||
/// at all.
|
||||
///
|
||||
/// Until 0.24 there were four — Bilinear, Fast, Medium, Best — and the
|
||||
/// sidecar keeps their numbers: 2, which was Medium, is now Best, and 3,
|
||||
/// which was Best, is past the end and reads as the default, which is
|
||||
/// Best. Both land on the network that replaced them, with no migration.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||
pub enum Method {
|
||||
/// The classical demosaic: the noise stays.
|
||||
Bilinear,
|
||||
/// The smallest student: a quarter of the medium network's work.
|
||||
/// The smallest student: a quarter of Best's work.
|
||||
Fast,
|
||||
/// One network the size of the first release's.
|
||||
Medium,
|
||||
/// Two experts, one for flat areas and one for edges, and a gate.
|
||||
/// One network of the first release's size, taught by the mixture of
|
||||
/// experts it replaced: the mixture's edges at a third of its work.
|
||||
Best,
|
||||
}
|
||||
|
||||
impl Method {
|
||||
pub const ALL: [Method; 4] = [Method::Bilinear, Method::Fast, Method::Medium, Method::Best];
|
||||
pub const ALL: [Method; 3] = [Method::Bilinear, Method::Fast, Method::Best];
|
||||
pub const DEFAULT: Method = Method::Best;
|
||||
|
||||
/// The sidecar's number for it.
|
||||
@@ -92,7 +96,6 @@ pub(crate) static DESCRIPTOR: LazyLock<Arc<OpDescriptor>> = LazyLock::new(|| {
|
||||
vec![
|
||||
LocalizedKey("param.learned_denoise.method.bilinear"),
|
||||
LocalizedKey("param.learned_denoise.method.fast"),
|
||||
LocalizedKey("param.learned_denoise.method.medium"),
|
||||
LocalizedKey("param.learned_denoise.method.best"),
|
||||
],
|
||||
)
|
||||
@@ -117,3 +120,20 @@ pub(crate) static DESCRIPTOR: LazyLock<Arc<OpDescriptor>> = LazyLock::new(|| {
|
||||
pub fn descriptor() -> Arc<OpDescriptor> {
|
||||
DESCRIPTOR.clone()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// An edit saved before 0.24 stored Medium as 2 and Best as 3. Both
|
||||
/// now name the network that replaced them, and nothing reads as Fast
|
||||
/// or Bilinear that did not before.
|
||||
#[test]
|
||||
fn the_retired_methods_read_as_best() {
|
||||
assert_eq!(Method::from_index(0.0), Method::Bilinear);
|
||||
assert_eq!(Method::from_index(1.0), Method::Fast);
|
||||
assert_eq!(Method::from_index(2.0), Method::Best, "Medium, before 0.24");
|
||||
assert_eq!(Method::from_index(3.0), Method::Best, "Best, before 0.24");
|
||||
assert_eq!(Method::Best.index(), 2.0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -263,10 +263,12 @@ cp "${DEX}" "${OUT}/staging/classes.dex"
|
||||
# native library directory. The build links none of it — the app dlopens
|
||||
# `libonnxruntime.so` at launch and runs on tract if it is not there — so an
|
||||
# APK without these is a slower app, not a broken one, and `RUNTIME_DIR=none`
|
||||
# builds exactly that. 174 MB for the default set; the script says which
|
||||
# Hexagon generations that buys.
|
||||
# builds exactly that. 206 MB for the default set — 32 MB of it the generic
|
||||
# WebGPU build for a phone without a Qualcomm SoC; the script says which
|
||||
# Hexagon generations the rest buys.
|
||||
if [[ "${RUNTIME_DIR}" != "none" ]]; then
|
||||
if [[ ! -f "${RUNTIME_DIR}/lib/libonnxruntime.so" ]]; then
|
||||
if [[ ! -f "${RUNTIME_DIR}/lib/libonnxruntime.so" \
|
||||
|| ! -f "${RUNTIME_DIR}/lib/libonnxruntime_generic.so" ]]; then
|
||||
"${REPO}/tools/fetch-android-runtime.sh" "${RUNTIME_DIR}"
|
||||
fi
|
||||
cp "${RUNTIME_DIR}"/lib/*.so "${OUT}/staging/lib/${ABI}/"
|
||||
@@ -321,6 +323,10 @@ for _dir in face scene inpaint denoise; do
|
||||
for f in "${ASSETS}"/*; do
|
||||
case "$(basename "${f}")" in
|
||||
README.md) continue ;;
|
||||
# The denoisers' any-size exports run whole frames on TensorRT
|
||||
# and CUDA (denoise.md §14); the Hexagon takes fixed shapes, and
|
||||
# 16 MB of graphs it never loads stay out of the APK.
|
||||
mosaic-fast.onnx | mosaic-hq.onnx) continue ;;
|
||||
esac
|
||||
cp "${f}" "${OUT}/staging/assets/models/"
|
||||
_bundled="${_bundled} $(basename "${f}")"
|
||||
|
||||
@@ -31,6 +31,9 @@ ENV DEBIAN_FRONTEND=noninteractive \
|
||||
# ---------------------------------------------------------------------------
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl git git-lfs \
|
||||
# The bundled ONNX Runtime builds arrive as wheels, which are zips
|
||||
# (tools/fetch-bundled-runtimes.sh).
|
||||
unzip \
|
||||
# A *host* C compiler as well as the cross one: build scripts and
|
||||
# proc-macros are compiled for Linux and linked with `cc`, whatever
|
||||
# the target. Without it the very first build script fails with
|
||||
|
||||
@@ -90,6 +90,13 @@ for f in "${REPO}/docs/manual/media"/*; do
|
||||
done
|
||||
echo "==> staged the manual and $(ls "${STAGE}/manual/media" | wc -l) picture(s)"
|
||||
|
||||
# The two ONNX Runtime builds the app chooses between at launch
|
||||
# (docs/dev/inference.md §3.2): Intel's OpenVINO build for an Intel GPU and
|
||||
# the WebGPU build — D3D12 — for any other, both carrying the CPU provider.
|
||||
# Beside the executable under `runtimes\`, where darkroom-desktop looks.
|
||||
"${REPO}/tools/fetch-bundled-runtimes.sh" windows "${STAGE}/runtimes"
|
||||
echo "==> staged $(ls "${STAGE}/runtimes"/* | wc -l) runtime file(s)"
|
||||
|
||||
# One installer in the output directory, the one just built. The directory
|
||||
# is cached between CI runs, so after a version bump a glob over it would find
|
||||
# two and the smoke test would hand Wine both names as one path.
|
||||
|
||||
@@ -526,3 +526,104 @@ reads the cache.
|
||||
**Packaging.** All six files in the APK (`BUNDLED`, 23 entries, +44.6 MB, ~41 MB compressed); the
|
||||
three f32 networks in the Arch package and the Windows installer, which stage `models/denoise` by
|
||||
directory.
|
||||
|
||||
## 14. A whole frame, not 1408² tiles (after 0.23.0)
|
||||
|
||||
A fixed 1408² tile is exact only past its halo, and Best's halo is 256: of every 1408² it computes
|
||||
it keeps 896², 2.47 photosites of work for each one kept (Medium and Fast keep 1024², 1.89×). On a
|
||||
GPU the network can instead run over the whole frame and its reflected border in one call, which is
|
||||
exact by the same argument (§3.4) and wastes only the border.
|
||||
|
||||
**The networks** are re-exported with any height and width (`mosaic-{best,medium,fast}.onnx` beside
|
||||
the 1408 files; darkroom-denoise `tools/export_whole.py`), from the checkpoints the shipped files
|
||||
came from. The tool refuses unless each matches its 1408 file at 1408² (max |Δ| = 0 for all three),
|
||||
matches torch at 592 × 848, and equals tiled inference over the reflected frame in f64 (≤ 7e-16).
|
||||
The APK leaves them out: the Hexagon takes fixed shapes.
|
||||
|
||||
**Where they run.** `Role::WholeDenoiser` is served by TensorRT and the CUDA provider only, the rungs
|
||||
where a new input size costs nothing at run time; MIGraphX, OpenVINO and CoreML compile per shape,
|
||||
the Hexagon takes fixed shapes, and the CPU would hold gigabytes of f32 activations. Everywhere else
|
||||
`whole_frame_limit()` is `None` and the 1408² tiles run as before. TensorRT gets an optimisation
|
||||
profile up to `WHOLE_FRAME_MAX` (4608 × 3328) — without one a dynamic input compiles a new engine per
|
||||
size at run time — through the runtime's V2 options, since `ort`'s builder has none, and keeps the
|
||||
engine in a directory per model and profile (ONNX Runtime's cache key leaves the shape out).
|
||||
|
||||
**The limit is the card's memory.** TensorRT plans its memory for the profile's largest shape. A
|
||||
profile up to a whole 6D frame with Best's border (4608 × 6656) asked for 4.9–5.9 GB and would not
|
||||
build on the 6 GB RTX 3050. At 15 MP the tiler (`tile::plan`) cuts the frame into the fewest equal
|
||||
tiles under the limit: a 6D frame is two of 4160 × 3248, 27 MP of work for 20 MP kept, against 49 MP
|
||||
in 1408² tiles. If a plan's first call fails, as a GPU out of memory does, its kept centre is halved
|
||||
and the frame planned again.
|
||||
|
||||
**Measured** 2026-10-06 on `_MG_8862` (6D, ISO 8000, 20 MP), RTX 3050 Laptop, TensorRT fp16, P3 /
|
||||
5001 MHz, another session's paused training holding 1.3 GB:
|
||||
|
||||
| Best | Network time | Against the tiles |
|
||||
|---|---|---|
|
||||
| 1408² tiles | 2.60 s | — |
|
||||
| Whole frame, two 4160 × 3248 tiles | **1.37 s** | max \|Δ\| 0.0029, mean 1.1e-5 — fp16's own spread (GPU tiles against CPU tiles: 0.0025) |
|
||||
|
||||
The first build of the whole-frame engine took 28 minutes, in the background at first launch, with
|
||||
the 1408² tiles serving meanwhile — against about 3 minutes for the fixed one; the profile's range
|
||||
is what it tunes across. A cached engine loads in about a second.
|
||||
|
||||
In PyTorch fp16 the same network over the whole 20 MP frame in one call took 3.5× less than in
|
||||
tiles, so a card that holds a whole frame gains more than the 6 GB one does; `WHOLE_FRAME_MAX` is a
|
||||
constant sized for 6 GB until the limit follows the card's memory.
|
||||
|
||||
## 15. Best becomes one network (0.24)
|
||||
|
||||
The photographer's goal for 0.24 was Best's quality in under a second on the laptop. Whole frames
|
||||
(§14) took the mixture from 2.60 s to 1.37 s and no further on a 6 GB card, so the other half was
|
||||
a single network that holds the mixture's quality at a third of its work. Methods are now
|
||||
`Bilinear`, `Fast` and `Best`; Medium and the mixture are retired.
|
||||
|
||||
**The network** is `fb-combo` (darkroom-denoise, 2026-10-07): §11's shape (32-64-128-192, blocks
|
||||
1-1-2-2, 3.2 M parameters, 48 GMAC/MP, halo 192), 20 000 steps from `fb-edges2` ← `student-m`,
|
||||
taught by the mixture at a half share, with 10 % drawn scenes and 25 % crops from the edge-rich
|
||||
cells of the training frames (branch `edge-sampling`). Scored on real photographs — the chart
|
||||
overstated the mixture's lead (a chart-sharp network was softer than Medium on real edges) — on the
|
||||
validation crops in the top quarter for sharp detail:
|
||||
|
||||
| | Edge PSNR, ISO 1600 / 6400 / 25600 | Sharpness kept | Smooth areas | Held-out PSNR, ISO 400 / 1600 / 6400 / 25600 | Chart edge |
|
||||
|---|---|---|---|---|---|
|
||||
| Mixture (Best to 0.23) | 30.71 / 29.93 / 28.55 | 0.899 / 0.868 / 0.782 | 42.61 / 41.71 / 40.12 | 40.69 / 39.84 / 38.47 / 36.71 | 0.82 |
|
||||
| `fb-combo` (Best from 0.24) | 30.67 / 29.87 / 28.49 | 0.902 / 0.874 / 0.792 | 42.54 / 41.57 / 39.85 | 40.64 / 39.78 / 38.38 / 36.54 | 0.89 |
|
||||
| Medium (to 0.23) | 30.43 / 29.68 / 28.38 | 0.896 / 0.862 / 0.776 | 42.59 / 41.69 / 40.08 | 40.59 / 39.75 / 38.40 / 36.65 | 1.30 |
|
||||
|
||||
Edges within 0.04–0.06 dB and more sharpness kept at every ISO; the known shortfall is smooth areas
|
||||
at ISO 25600, 0.27 dB. The photographer took it as it stood at 20 000 of a planned 30 000 steps.
|
||||
Others tried on the way, each short of the mixture on real photographs: `fb-sharp` (drawn scenes,
|
||||
chart-sharp but Medium's real edges), `fb-edges` (half edge-rich crops: edges close, ISO 25600
|
||||
flats −0.24 dB), `fb-edges2` (a quarter: 0.03–0.14 dB short everywhere, chart 1.33–1.47), and a
|
||||
from-scratch 24-48-96-128 between Fast and Medium.
|
||||
|
||||
**Files.** `mosaic-hq-1408.onnx`, `mosaic-hq.onnx` (any size) and `mosaic-hq-1408.a16w16.onnx` for
|
||||
the Hexagon. A new name, not Medium's or Best's: the result cache keys a model by name and size,
|
||||
and this one is byte for byte Medium's size. The tablet form lost 0.00 dB in simulated QDQ at every
|
||||
ISO and at most 0.09 dB across the ×0.5–×4 noise bracket (A16W8 0.08 / 0.26 dB; int8 −10.6 dB);
|
||||
not yet confirmed on the tablet itself.
|
||||
|
||||
**Saved edits** keep their numbers: 2, which was Medium, is now Best; 3, which was Best, is past the
|
||||
end and reads as the default, Best. Both land on the new network with no migration.
|
||||
|
||||
**Measured** 2026-10-07, `_MG_8862`, RTX 3050 Laptop, TensorRT fp16, P3 / 5001 MHz, nothing else on
|
||||
the card:
|
||||
|
||||
| Best | Network time | Peak GPU memory |
|
||||
|---|---|---|
|
||||
| mixture, 1408² tiles (0.23) | 2.60 s | — |
|
||||
| mixture, whole frame (§14) | 1.37 s | — |
|
||||
| `fb-combo`, 1408² tiles | 0.95 s | 0.55 GB |
|
||||
| `fb-combo`, whole frame (two 4160 × 3248) | **0.51–0.54 s** | 1.75 GB |
|
||||
|
||||
Decode and the hot-pixel pass add 0.4–0.5 s, so a photograph is about a second end to end. Whole
|
||||
frame against tiles: max |Δ| 0.0029, 90 dB apart — fp16's spread. The whole-frame engine's first
|
||||
build took 12 minutes (the mixture's 28); from the cache it loads in about a second, so the session
|
||||
keeps the engine's ordinary 30 s idle decay rather than unloading after each photograph: at 1.75 GB
|
||||
it fits beside the develop view on a 6 GB card, and an unload would cost the next photograph a
|
||||
second.
|
||||
|
||||
The manual's close-up for Best is still the mixture's render, which this network matches to within
|
||||
the table above; it is re-recorded with the next pass of `tools/manual/record.sh`.
|
||||
|
||||
|
||||
+88
-6
@@ -194,6 +194,45 @@ int8's matches land 0.45 px from f32's in the overlaps — the spread f32 shows
|
||||
|
||||
---
|
||||
|
||||
### 1.6 Intel Iris Xe, and the generic rung · 2026-10-04
|
||||
|
||||
The RTX 3050 laptop's other GPU: Raptor Lake-P's Iris Xe (96 EU), Intel's `onnxruntime-openvino`
|
||||
1.24.1 (OpenVINO 2025.4.1) and Microsoft's `onnxruntime-webgpu` 1.27.0, both PyPI wheels, through
|
||||
`ep_probe`. Three warm-ups, the median of 15 runs. Another build shared the CPU during the run, so
|
||||
the CPU columns are a little pessimistic; the GPU columns are not.
|
||||
|
||||
| Model | ORT CPU f32 | OpenVINO CPU | OpenVINO GPU f32 | **OpenVINO GPU fp16** | WebGPU (Iris Xe) |
|
||||
|---|---|---|---|---|---|
|
||||
| scrfd_500m (Fast) | 9.5 | 10.8 | 7.3 | **5.8** | 24.2 |
|
||||
| scrfd_2.5g (Balanced) | 18.5 | 16.2 | 15.5 | **11.1** | 40.8 |
|
||||
| scrfd_10g (Thorough) | 58.5 | 72.4 | 38.7 | **23.0** | 84.5 |
|
||||
| arcface_mbf (per face) | 9.5 | 11.7 | **3.0** | 2.4 | 56.6 |
|
||||
| 2d106det (landmarks) | 10.8 | 2.0 | 2.4 | **1.9** | 42.7 |
|
||||
| yolo26s-sem-ade20k | 57.0 | 47.4 | 26.0 | **16.7** | 53.0 |
|
||||
| xfeat-1024 | 23.5 | 18.5 | 19.9 | **17.5** | 34.2 |
|
||||
| migan-512 (per tile) | 330 | ✗ ¹ | 89.7 | **57.2** | 275 |
|
||||
| mosaic-fast-1408 (per tile) | 159 | 107 | 68.1 | **40.0** | 188 ² |
|
||||
| mosaic-best-1408 (per tile) | 1109 | 1670 | 947 | **604** | 1034 ² |
|
||||
|
||||
¹ OpenVINO's CPU plugin refuses the graph at initialisation. Not shipped (§3.2), so moot.
|
||||
² A later run, after `ep_probe` learned to feed the denoiser's two inputs, under heavier load: the
|
||||
CPU provider took 256 and 1034 ms in that run, so WebGPU beat it by a quarter on mosaic-fast and tied
|
||||
on mosaic-best — the only rows where it is not well behind.
|
||||
|
||||
- **OpenVINO on the Iris Xe beats ONNX Runtime's CPU provider on every model**, 1.3× on XFeat to
|
||||
5.8× on MI-GAN, with a 1–3 s compile per graph and 0.1–0.4 s from its cache. It is the Intel rung.
|
||||
fp16 is worth 1.3–1.7× over f32 here, against 1.1–1.35× on MIGraphX.
|
||||
- **Its "GPU" is OpenCL's first GPU, not Intel's.** Before `intel-compute-runtime` was installed
|
||||
the only OpenCL driver was NVIDIA's, and `device_type=GPU` ran on the RTX 3050 — slower than the
|
||||
CPU, which is the probe's to catch. Read the process's maps for `libigdrcl` before believing a
|
||||
number is the iGPU's.
|
||||
- **WebGPU is slower than the CPU on the Iris Xe** on everything but MI-GAN, as it was on the
|
||||
Adreno (§1.1), and on the RTX 3050 through Vulkan too. It is on the ladder anyway, as the generic
|
||||
rung (§2): for GPUs no vendor rung covers — an AMD card on Windows or without ROCm, a Mali — where
|
||||
it is unmeasured, and the probe's clock decides.
|
||||
- **OpenVINO's CPU plugin is not a better floor.** It wins on some graphs and loses on scrfd_10g,
|
||||
the embedder and mosaic-best, and refuses MI-GAN.
|
||||
|
||||
## 2. The shape of the answer
|
||||
|
||||
A **ladder per platform**, walked at start-up, with the first rung that builds a real session
|
||||
@@ -202,10 +241,11 @@ winning:
|
||||
| Platform | 1st | 2nd | 3rd | Floor |
|
||||
|---|---|---|---|---|
|
||||
| Android, Qualcomm with a Hexagon the shipped QNN skel covers (V68–V81) | QNN HTP, each model's quantised form (§1.5) | ORT CPU, f32 model | — | tract |
|
||||
| Android, any other SoC | ORT CPU, f32 | — | — | tract |
|
||||
| Android, any other SoC ⁶ | WebGPU (Vulkan), f32 | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, NVIDIA GPU | TensorRT, f32 model, fp16 engine | CUDA provider, f32 | ORT CPU, f32 | tract |
|
||||
| Linux, AMD GPU with ROCm | MIGraphX, f32 model, fp16 program | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, no GPU stack | ORT CPU, f32 | — | — | tract |
|
||||
| Linux / Windows, Intel GPU | OpenVINO, f32 model, fp16 program (§1.6) | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, any other GPU ⁶ | WebGPU (Vulkan / D3D12), f32 | ORT CPU, f32 | — | tract |
|
||||
| macOS ⁵ | CoreML, f32 model, ML Program | ORT CPU, f32 | — | tract |
|
||||
|
||||
⁵ **Unmeasured**, and the one exception to the rule below: nobody here has a Mac. The rung is on
|
||||
@@ -215,11 +255,17 @@ down is refused on the third launch (§4, `attempt`). The embedder stays on the
|
||||
macOS log that shows a probe line is this row's measurement; [macos.md](macos.md) says what to
|
||||
ask for.
|
||||
|
||||
⁶ **The generic rung, unmeasured where it is meant to help.** WebGPU lost to the CPU on every GPU
|
||||
it has been timed on — the Adreno, the Iris Xe, the RTX 3050 (§1.1, §1.6) — none of which it serves
|
||||
here, since each has its own rung. It is on the ladder for the GPUs that have none, on the same
|
||||
terms as CoreML: a WebGPU that is slower than the CPU is rejected by §4's clock, one that errors is
|
||||
recorded as failed. Its first measurement on an AMD card without ROCm, or a Mali, is this row's.
|
||||
|
||||
Deliberately **not** on any ladder, with the measurement that excluded each: NNAPI (no driver),
|
||||
XNNPACK (slower than CPU, aborts on SCRFD), WebGPU (slower than CPU), the Adreno through QNN (works,
|
||||
but never where the Hexagon does not also), CUDA int8 (slower than CUDA f32), the ROCm provider
|
||||
(gone: §1.3). A rung is added to this table by a measurement on this page, not by a provider
|
||||
existing.
|
||||
XNNPACK (slower than CPU, aborts on SCRFD), the Adreno through QNN (works, but never where the
|
||||
Hexagon does not also), CUDA int8 (slower than CUDA f32), the ROCm provider (gone: §1.3),
|
||||
OpenVINO's CPU plugin as a floor (§1.6). A rung is added to this table by a measurement on this
|
||||
page, not by a provider existing — the two footnoted rows are the exceptions, and say so.
|
||||
|
||||
The AMD ladder has no middle rung. TensorRT falls back to the CUDA provider while its engines
|
||||
compile; MIGraphX has no such twin, so its fallback is the CPU provider, and the minute or two of
|
||||
@@ -295,6 +341,40 @@ it.
|
||||
|
||||
Both positions are D13 territory and are recorded there (§12).
|
||||
|
||||
Two more runtimes ship in every desktop package since 0.23 (§3.2), and their parts are all
|
||||
redistributable:
|
||||
|
||||
| Component | Licence | Shipped |
|
||||
|---|---|---|
|
||||
| Intel's `onnxruntime-openvino` build, with OpenVINO 2025.4.1 and oneTBB | MIT; Apache-2.0; Apache-2.0 | Linux and Windows packages, texts beside the libraries |
|
||||
| Microsoft's WebGPU build (Dawn inside); on Windows the DirectX shader compiler | MIT; LLVM / MIT | Linux and Windows packages |
|
||||
| Microsoft's stock `onnxruntime-android` (WebGPU) | MIT | The APK, as `libonnxruntime_generic.so` |
|
||||
|
||||
### 3.2 Several runtimes, one per process
|
||||
|
||||
A runtime carries one vendor's providers: Intel's build has OpenVINO, the `onnxruntime-gpu` wheel
|
||||
CUDA and TensorRT, a ROCm build MIGraphX, Microsoft's WebGPU build the generic rung, the APK's QNN
|
||||
build the Hexagon. No prebuilt carries two vendors, and `set_api` takes one table per process.
|
||||
So a device that may hold several — the package's OpenVINO and WebGPU builds, a CUDA build the user
|
||||
fetched, the distribution's ROCm build — has to choose which to load *before* the probe, and
|
||||
cannot choose by trying.
|
||||
|
||||
`api::install` opens every runtime on the search list, asks each for `GetAvailableProviders`, and
|
||||
loads the one that scores highest against the GPUs `hardware::detect` reads from files: a vendor
|
||||
rung on its own vendor's GPU (NVIDIA driver, `/dev/kfd`, a Qualcomm SoC, macOS) above OpenVINO on
|
||||
an Intel GPU (PCI vendor `0x8086`; on Windows Intel's DCH driver package) above WebGPU above a
|
||||
CPU-only build. Equal scores keep the search order, a perfect fit ends the search — the APK's QNN
|
||||
build is listed first, so on a Qualcomm device the generic build is never opened — and
|
||||
`DARKROOM_ORT_DIR` wins outright. The losers stay mapped: unloading a C++ runtime whose static
|
||||
constructors ran is a crash at exit waiting to happen.
|
||||
|
||||
The desktop packages install the two bundled builds under `runtimes/openvino` and
|
||||
`runtimes/webgpu` beside each place a package installs to, from
|
||||
`tools/fetch-bundled-runtimes.sh` (PyPI wheels pinned by SHA-256, pruned to the native libraries:
|
||||
81 + 31 MB on Linux, 67 + 42 MB on Windows). On Windows the chosen runtime's directory is put on
|
||||
`PATH`, because Intel's build leaves OpenVINO's DLLs for the loader to find there. The Flatpak has
|
||||
no Intel OpenCL driver in its sandbox, so an Intel machine there settles on the CPU.
|
||||
|
||||
---
|
||||
|
||||
## 4. Selection — the probe, its cache, and what it may not do
|
||||
@@ -603,6 +683,8 @@ device are comparable. *Acceptance:* M3.
|
||||
**D13 — updated.** The runtime half is reopened to the extent of §3: the Rust build stays C-free
|
||||
under `alternative-backend`; packages may install a dynamically loaded ONNX Runtime and, per §3.1,
|
||||
the Qualcomm QNN runtime; the NVIDIA libraries are not bundled. The licensing half is unchanged.
|
||||
Since 0.23 every desktop package bundles two runtimes — Intel's OpenVINO build and the WebGPU
|
||||
build — and the APK a second, generic one; the engine loads the one that fits the GPU (§3.2).
|
||||
|
||||
---
|
||||
|
||||
|
||||
File diff suppressed because one or more lines are too long
+14
-14
@@ -215,22 +215,22 @@ colour that come with it, while keeping the fine detail. Look at it at
|
||||
|
||||
`Method` chooses how:
|
||||
|
||||
- `Best`, the default: two networks, one for smooth areas and one for
|
||||
edges, blended where each is better. The cleanest skies and the sharpest
|
||||
lettering, and the slowest.
|
||||
- `Medium`: one network taught by `Best`. Nearly as clean in smooth areas,
|
||||
a little softer on hard edges, in about a third of the time.
|
||||
- `Fast`: a smaller one, taught the same way. Visibly noisier at very high
|
||||
ISO than the other two, but still far cleaner than none, and quick.
|
||||
- `Best`, the default: clean skies and sharp lettering, edges kept as
|
||||
crisp as the camera recorded them.
|
||||
- `Fast`: a smaller network, taught the same way. Visibly noisier at very
|
||||
high ISO than `Best`, but still far cleaner than none, and quicker.
|
||||
- `Bilinear`: the camera's ordinary conversion, noise and all.
|
||||
|
||||
A photograph last edited with `Medium`, which earlier versions offered,
|
||||
opens with `Best`.
|
||||
|
||||
The photograph shows the camera's ordinary conversion while the network
|
||||
works, with its progress in the bar at the top, and changes when it is
|
||||
done — on a laptop's graphics card, about two and a half seconds for a
|
||||
20-megapixel photograph with `Best` and under one with the other two;
|
||||
longer on a processor alone or on the tablet. The first photograph after
|
||||
installing waits a few minutes more while the graphics card prepares each
|
||||
network, once. The result is kept, so a photograph opened again,
|
||||
done — on a laptop's graphics card, about a second for a 20-megapixel
|
||||
photograph with `Best`, reading the file included; longer on a processor
|
||||
alone or on the tablet. After installing, the graphics card spends up to a
|
||||
quarter of an hour preparing each network, once, in the background; the
|
||||
photographs developed meanwhile take a little longer. The result is kept, so a photograph opened again,
|
||||
or exported, does not wait a second time, and switching back to a method
|
||||
already used is quick.
|
||||
`Strength` eases it off: below 100 % it puts back some of what was removed,
|
||||
@@ -241,8 +241,8 @@ The lamp and railing of a night frame at ISO 8000, at 1:1, by each method:
|
||||
| Bilinear | Fast |
|
||||
|---|---|
|
||||
|  |  |
|
||||
| **Medium** | **Best** |
|
||||
|  |  |
|
||||
| **Best** | |
|
||||
|  | |
|
||||
|
||||
It works on raw files from any camera with the usual colour pattern of
|
||||
red, green and blue squares — not on JPEGs, and not yet on Fujifilm's
|
||||
|
||||
+13
-14
@@ -296,22 +296,21 @@ colour that come with it, while keeping the fine detail. Look at it at
|
||||
1:1, where noise lives.</p>
|
||||
<p><code>Method</code> chooses how:</p>
|
||||
<ul>
|
||||
<li><code>Best</code>, the default: two networks, one for smooth areas and one for
|
||||
edges, blended where each is better. The cleanest skies and the sharpest
|
||||
lettering, and the slowest.</li>
|
||||
<li><code>Medium</code>: one network taught by <code>Best</code>. Nearly as clean in smooth areas,
|
||||
a little softer on hard edges, in about a third of the time.</li>
|
||||
<li><code>Fast</code>: a smaller one, taught the same way. Visibly noisier at very high
|
||||
ISO than the other two, but still far cleaner than none, and quick.</li>
|
||||
<li><code>Best</code>, the default: clean skies and sharp lettering, edges kept as
|
||||
crisp as the camera recorded them.</li>
|
||||
<li><code>Fast</code>: a smaller network, taught the same way. Visibly noisier at very
|
||||
high ISO than <code>Best</code>, but still far cleaner than none, and quicker.</li>
|
||||
<li><code>Bilinear</code>: the camera's ordinary conversion, noise and all.</li>
|
||||
</ul>
|
||||
<p>A photograph last edited with <code>Medium</code>, which earlier versions offered,
|
||||
opens with <code>Best</code>.</p>
|
||||
<p>The photograph shows the camera's ordinary conversion while the network
|
||||
works, with its progress in the bar at the top, and changes when it is
|
||||
done — on a laptop's graphics card, about two and a half seconds for a
|
||||
20-megapixel photograph with <code>Best</code> and under one with the other two;
|
||||
longer on a processor alone or on the tablet. The first photograph after
|
||||
installing waits a few minutes more while the graphics card prepares each
|
||||
network, once. The result is kept, so a photograph opened again,
|
||||
done — on a laptop's graphics card, about a second for a 20-megapixel
|
||||
photograph with <code>Best</code>, reading the file included; longer on a processor
|
||||
alone or on the tablet. After installing, the graphics card spends up to a
|
||||
quarter of an hour preparing each network, once, in the background; the
|
||||
photographs developed meanwhile take a little longer. The result is kept, so a photograph opened again,
|
||||
or exported, does not wait a second time, and switching back to a method
|
||||
already used is quick.
|
||||
<code>Strength</code> eases it off: below 100 % it puts back some of what was removed,
|
||||
@@ -319,8 +318,8 @@ as grain without colour, for a picture that does not look too smooth.</p>
|
||||
<p>The lamp and railing of a night frame at ISO 8000, at 1:1, by each method:</p>
|
||||
<table><thead><tr><th>Bilinear</th><th>Fast</th></tr></thead><tbody>
|
||||
<tr><td><img src="media/develop-denoise-bilinear.png" alt="The railing and the lamp at ISO 8000, as the camera recorded them" /></td><td><img src="media/develop-denoise-fast.png" alt="The same, with the Fast network" /></td></tr>
|
||||
<tr><td><strong>Medium</strong></td><td><strong>Best</strong></td></tr>
|
||||
<tr><td><img src="media/develop-denoise-medium.png" alt="The same, with the Medium network" /></td><td><img src="media/develop-denoise-best.png" alt="The same, with the Best network" /></td></tr>
|
||||
<tr><td><strong>Best</strong></td><td></td></tr>
|
||||
<tr><td><img src="media/develop-denoise-best.png" alt="The same, with the Best network" /></td><td></td></tr>
|
||||
</tbody></table>
|
||||
<p>It works on raw files from any camera with the usual colour pattern of
|
||||
red, green and blue squares — not on JPEGs, and not yet on Fujifilm's
|
||||
|
||||
Binary file not shown.
+2
-2
@@ -134,9 +134,9 @@ declined, and the InsightFace grant of D13).
|
||||
|
||||
| File | Source | Trained on | Used by |
|
||||
|---|---|---|---|
|
||||
| `denoise/mosaic-best-1408.onnx` | trained in the `darkroom-denoise` repository (2026-10-04, run `final`, 30 000 steps, from the experts of runs `m2` and `edges-100`) | 1,701 of the maintainer's own base-ISO raws and 6,000 synthetic scenes the repository draws itself, with the Canon EOS 6D's measured noise added | the learned demosaic and denoise, Best (FR-DEV-3g) |
|
||||
| `denoise/mosaic-medium-1408.onnx` | distilled from `final` in the same repository (2026-10-04, run `student-m`, 20 000 steps, from `m2`) | the same | Medium |
|
||||
| `denoise/mosaic-hq-1408.onnx` | trained in the `darkroom-denoise` repository (2026-10-07, run `fb-combo`, 20 000 steps, from `fb-edges2` ← `student-m`), taught by the mixture of experts that was Best until 0.24 (run `final`) at a half share, with 10 % drawn scenes and 25 % crops from the edge-rich parts of the training frames | 1,701 of the maintainer's own base-ISO raws and 6,000 synthetic scenes the repository draws itself, with the Canon EOS 6D's measured noise added | the learned demosaic and denoise, Best (FR-DEV-3g) |
|
||||
| `denoise/mosaic-fast-1408.onnx` | distilled from `final` (2026-10-04, run `student-s`, 30 000 steps, from scratch) | the same | Fast |
|
||||
| `denoise/mosaic-{hq,fast}.onnx` | the two networks above with any height and width, by `tools/export_whole.py` in the same repository from the same checkpoints; identical to the 1408 files at 1408² | the same | the same methods, a whole frame at a time on a GPU (denoise.md §14) |
|
||||
|
||||
U-Nets of plain 3×3 convolutions, ReLU, strided and transposed
|
||||
convolutions and additive skips — no third-party architecture code or
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+26
-16
@@ -4,7 +4,7 @@
|
||||
# makes `makepkg -si` in this directory install what you are actually working
|
||||
# on. Swap `source` for a tagged tarball when there is something to release.
|
||||
pkgname=darkroom
|
||||
pkgver=0.22.1
|
||||
pkgver=0.24.0
|
||||
# Back to 1 with the version: a new pkgver is a new archive name, so there is
|
||||
# nothing for makepkg to reuse and nothing for a release number to disambiguate.
|
||||
pkgrel=1
|
||||
@@ -15,14 +15,17 @@ license=('GPL-3.0-or-later')
|
||||
# Runtime: Vulkan for wgpu, and a Secret Service implementation for the
|
||||
# Nextcloud credentials (FR-NC-2) — gnome-keyring or kwallet both provide it.
|
||||
depends=('vulkan-icd-loader' 'fontconfig' 'libxkbcommon')
|
||||
makedepends=('cargo' 'git')
|
||||
# ONNX Runtime is loaded from /usr/lib at launch if a package put it there
|
||||
# (docs/inference.md §3): the CPU build is 8–10× the built-in tract, the
|
||||
# ROCm build adds the MIGraphX rung on an AMD GPU. Neither is required.
|
||||
makedepends=('cargo' 'git' 'curl' 'unzip')
|
||||
# Two ONNX Runtime builds ship in /usr/lib/darkroom/runtimes — Intel's
|
||||
# OpenVINO build and the generic WebGPU one, both 8–10× the built-in tract
|
||||
# on the CPU alone — and the app opens every runtime it finds and keeps the
|
||||
# one that fits the GPU (docs/dev/inference.md §3.2). The ROCm build in
|
||||
# /usr/lib adds the MIGraphX rung on an AMD GPU and outranks both there;
|
||||
# Intel's OpenCL driver is what lets OpenVINO reach an Intel GPU.
|
||||
optdepends=('gnome-keyring: store Nextcloud credentials'
|
||||
'kwallet: store Nextcloud credentials'
|
||||
'onnxruntime-cpu: run the neural models on every core'
|
||||
'onnxruntime-rocm: run the neural models on an AMD GPU')
|
||||
'onnxruntime-rocm: run the neural models on an AMD GPU'
|
||||
'intel-compute-runtime: run the neural models on an Intel GPU')
|
||||
options=('!lto') # the workspace sets its own LTO in Cargo.toml
|
||||
|
||||
_repo="$(cd "${startdir}/.." && pwd)"
|
||||
@@ -59,6 +62,10 @@ package() {
|
||||
|
||||
install -Dm644 "README.md" "${pkgdir}/usr/share/doc/${pkgname}/README.md"
|
||||
|
||||
# The bundled runtimes, where darkroom-desktop's search finds them
|
||||
# (`runtimes/` under /usr/lib/darkroom). Pinned wheels, checked by hash.
|
||||
./tools/fetch-bundled-runtimes.sh linux "${pkgdir}/usr/lib/darkroom/runtimes"
|
||||
|
||||
# The manual: the rendered page and its pictures, where the app's Help
|
||||
# opens it (dr_ui::manual, through dr_plat::system_data_dirs). Offline by
|
||||
# design — the help sheet's "See it" links land here, on a machine that
|
||||
@@ -125,14 +132,17 @@ package() {
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/migan-512.onnx"
|
||||
|
||||
# The learned demosaic and denoise, one network per method (the project's
|
||||
# own weights, GPL — models/LICENCE.md). Same pointer check, same
|
||||
# directory.
|
||||
for _net in fast medium best; do
|
||||
_src="models/denoise/mosaic-${_net}-1408.onnx"
|
||||
if [[ "$(stat -c%s "${_src}")" -lt 100000 ]]; then
|
||||
echo "error: the ${_net} denoise model is an LFS pointer — run: git lfs pull" >&2
|
||||
return 1
|
||||
fi
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/mosaic-${_net}-1408.onnx"
|
||||
# own weights, GPL — models/LICENCE.md), each at the fixed 1408 tile and
|
||||
# with any height and width for a whole frame on a GPU (denoise.md §14).
|
||||
# Same pointer check, same directory.
|
||||
for _net in fast hq; do
|
||||
for _file in "mosaic-${_net}-1408.onnx" "mosaic-${_net}.onnx"; do
|
||||
_src="models/denoise/${_file}"
|
||||
if [[ "$(stat -c%s "${_src}")" -lt 100000 ]]; then
|
||||
echo "error: ${_file} is an LFS pointer — run: git lfs pull" >&2
|
||||
return 1
|
||||
fi
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/${_file}"
|
||||
done
|
||||
done
|
||||
}
|
||||
|
||||
@@ -185,6 +185,15 @@ modules:
|
||||
install -Dm644 "models/scene/$m" "/app/share/darkroom/models/$m"
|
||||
done
|
||||
|
||||
# The two runtimes the app chooses between at launch
|
||||
# (docs/dev/inference.md §3.2): Intel's OpenVINO build and the generic
|
||||
# WebGPU one, each with ONNX Runtime's CPU provider — 8–10× the
|
||||
# built-in tract on any machine. Pinned wheels, fetched over the
|
||||
# build's network. In the sandbox the OpenVINO rung finds no Intel
|
||||
# OpenCL driver and the probe settles on the CPU; WebGPU reaches the
|
||||
# GPU through the runtime's Vulkan.
|
||||
- ./tools/fetch-bundled-runtimes.sh linux /app/lib/darkroom/runtimes
|
||||
|
||||
- install -Dm644 README.md /app/share/doc/darkroom/README.md
|
||||
|
||||
sources:
|
||||
|
||||
@@ -84,6 +84,12 @@ Section "DarkRoom" SecMain
|
||||
SetOutPath "$INSTDIR\manual"
|
||||
File /r "${STAGE}\manual\*"
|
||||
|
||||
; ONNX Runtime, twice: Intel's OpenVINO build and the WebGPU build. The
|
||||
; application opens both and keeps the one that fits the GPU
|
||||
; (docs/dev/inference.md §3.2); each also runs the models on every core.
|
||||
SetOutPath "$INSTDIR\runtimes"
|
||||
File /r "${STAGE}\runtimes\*"
|
||||
|
||||
WriteUninstaller "$INSTDIR\uninstall.exe"
|
||||
|
||||
; Add/Remove Programs. HKCU, to match the per-user install.
|
||||
@@ -123,6 +129,7 @@ Section "Uninstall"
|
||||
Delete "$INSTDIR\uninstall.exe"
|
||||
RMDir /r "$INSTDIR\models"
|
||||
RMDir /r "$INSTDIR\manual"
|
||||
RMDir /r "$INSTDIR\runtimes"
|
||||
RMDir "$INSTDIR"
|
||||
|
||||
Delete "$SMPROGRAMS\${NAME}\${NAME}.lnk"
|
||||
|
||||
@@ -3,17 +3,23 @@
|
||||
#
|
||||
# ./tools/fetch-android-runtime.sh [DEST]
|
||||
#
|
||||
# Two Maven artefacts, pinned to each other by ONNX Runtime's own POM:
|
||||
# Three Maven artefacts, the first two pinned to each other by ONNX Runtime's
|
||||
# own POM:
|
||||
#
|
||||
# com.microsoft.onnxruntime:onnxruntime-android-qnn MIT
|
||||
# com.qualcomm.qti:qnn-runtime Qualcomm AI Engine Direct SDK licence
|
||||
# com.microsoft.onnxruntime:onnxruntime-android MIT
|
||||
#
|
||||
# The first is ONNX Runtime built with the CPU, QNN, XNNPACK, NNAPI and WebGPU
|
||||
# providers; the second is Qualcomm's HTP backend — the ARM-side compiler and
|
||||
# the per-generation Hexagon "skel" the DSP loads. Both ship as AARs whose
|
||||
# `jni/arm64-v8a/` is what an APK's `lib/arm64-v8a/` wants, so this script
|
||||
# unpacks exactly that and nothing else, plus the licence texts, which travel
|
||||
# with the libraries (§3.1).
|
||||
# The first is ONNX Runtime built with the CPU and QNN providers; the second
|
||||
# is Qualcomm's HTP backend — the ARM-side compiler and the per-generation
|
||||
# Hexagon "skel" the DSP loads. The third is Microsoft's stock build, which
|
||||
# carries WebGPU (Dawn on Vulkan) and the QNN build does not: the generic
|
||||
# rung for a phone without a Qualcomm SoC (§3.2). It lands as
|
||||
# `libonnxruntime_generic.so` beside the QNN build; the engine opens the QNN
|
||||
# build first, and on a Qualcomm device stops there. All three ship as AARs
|
||||
# whose `jni/arm64-v8a/` is what an APK's `lib/arm64-v8a/` wants, so this
|
||||
# script unpacks exactly that and nothing else, plus the licence texts, which
|
||||
# travel with the libraries (§3.1).
|
||||
#
|
||||
# ## What is and is not taken from the Qualcomm package
|
||||
#
|
||||
@@ -65,6 +71,8 @@ fetch com.microsoft.onnxruntime onnxruntime-android-qnn "${ORT_VERSION}" \
|
||||
"${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar"
|
||||
fetch com.qualcomm.qti qnn-runtime "${QNN_VERSION}" \
|
||||
"${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar"
|
||||
fetch com.microsoft.onnxruntime onnxruntime-android "${ORT_VERSION}" \
|
||||
"${DEST}/aar/onnxruntime-android-${ORT_VERSION}.aar"
|
||||
|
||||
# The ONNX Runtime POM names the QNN version it was built against; a pair
|
||||
# that disagrees loads and then fails at the first graph, which is the kind
|
||||
@@ -82,6 +90,8 @@ rm -rf "${DEST}/lib"
|
||||
mkdir -p "${DEST}/lib"
|
||||
unzip -q -o -j "${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar" \
|
||||
'jni/arm64-v8a/libonnxruntime.so' -d "${DEST}/lib"
|
||||
unzip -q -o -p "${DEST}/aar/onnxruntime-android-${ORT_VERSION}.aar" \
|
||||
'jni/arm64-v8a/libonnxruntime.so' > "${DEST}/lib/libonnxruntime_generic.so"
|
||||
members=(jni/arm64-v8a/libQnnHtp.so jni/arm64-v8a/libQnnHtpPrepare.so jni/arm64-v8a/libQnnSystem.so)
|
||||
for arch in ${QNN_HTP_ARCHS}; do
|
||||
members+=("jni/arm64-v8a/libQnnHtpV${arch}Skel.so" "jni/arm64-v8a/libQnnHtpV${arch}Stub.so")
|
||||
|
||||
Executable
+132
@@ -0,0 +1,132 @@
|
||||
#!/usr/bin/env bash
|
||||
# Fetch the two ONNX Runtime builds a desktop package ships
|
||||
# (docs/dev/inference.md §3.2), each into its own directory:
|
||||
#
|
||||
# DEST/openvino Intel's build: the OpenVINO rung on an Intel GPU
|
||||
# DEST/webgpu Microsoft's WebGPU build: the generic rung on any other
|
||||
#
|
||||
# ./tools/fetch-bundled-runtimes.sh {linux|windows} DEST
|
||||
#
|
||||
# Only one runtime loads per process; the app opens every one it finds and
|
||||
# keeps the one that fits the device's GPU (`dr_inference_engine::api`).
|
||||
# Both carry ONNX Runtime's CPU provider, which is the floor either way.
|
||||
# A CUDA or ROCm runtime is never bundled (§3.1) — the user's own, found
|
||||
# beside these, outranks both on its vendor's GPU.
|
||||
#
|
||||
# The wheels are PyPI's, pinned by SHA-256: the option names the engine
|
||||
# sets were read from these versions' source (CLAUDE.md, "Providers").
|
||||
# Only the native libraries are kept — not the Python bindings, not
|
||||
# OpenVINO's CPU plugin (ONNX Runtime's CPU provider is the floor), not
|
||||
# the duplicate versioned copies a wheel holds as files.
|
||||
#
|
||||
# Licences: ONNX Runtime MIT, OpenVINO Apache-2.0, oneTBB Apache-2.0, the
|
||||
# DirectX shader compiler (Windows WebGPU) LLVM/MIT; the texts go beside
|
||||
# the libraries.
|
||||
set -euo pipefail
|
||||
|
||||
PLATFORM="${1:?usage: fetch-bundled-runtimes.sh linux|windows DEST}"
|
||||
DEST="${2:?usage: fetch-bundled-runtimes.sh linux|windows DEST}"
|
||||
PYPI="https://files.pythonhosted.org/packages"
|
||||
|
||||
case "${PLATFORM}" in
|
||||
linux)
|
||||
ORT_OPENVINO="${PYPI}/08/07/f225999919f56506b603aaa3ff837ad563ab26f86906ed7fa7e5abcd849e/onnxruntime_openvino-1.24.1-cp313-cp313-manylinux_2_28_x86_64.whl"
|
||||
ORT_OPENVINO_SHA=2c3bb73e68ac27f4891af8a595c1faf574ec68b772e6583c90a0b997a1822782
|
||||
ORT_WEBGPU="${PYPI}/7a/4e/782b2457b863e1748866b82d918323e61c2d0108a01906a278e7a86a3a55/onnxruntime_webgpu-1.27.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl"
|
||||
ORT_WEBGPU_SHA=3eb30b487d2b428d2e5c1ac538177ab6336cce9b204197135dc5a16753b0258e
|
||||
# The Linux wheel carries OpenVINO itself, under the names the provider
|
||||
# was linked against.
|
||||
OPENVINO_KEEP=(libonnxruntime.so.1.24.1 libonnxruntime_providers_shared.so
|
||||
libonnxruntime_providers_openvino.so libopenvino.so.2541
|
||||
libopenvino_onnx_frontend.so.2541 libopenvino_intel_gpu_plugin.so
|
||||
libtbb.so.12 libtbbmalloc.so)
|
||||
WEBGPU_KEEP=(libonnxruntime.so.1.27.0 libonnxruntime_providers_shared.so)
|
||||
;;
|
||||
windows)
|
||||
ORT_OPENVINO="${PYPI}/3e/92/46ae2cd565961a89189900f385bb2f13a9fa731ea4674001d23720fbb1e0/onnxruntime_openvino-1.24.1-cp313-cp313-win_amd64.whl"
|
||||
ORT_OPENVINO_SHA=434bf49aa71393c577a456c9d76c98e6d6958a833fa0876793e3d5437b5a511a
|
||||
ORT_WEBGPU="${PYPI}/dd/f3/6294f9617e97035771593d604729a420a848150054cc1c422a47a4915412/onnxruntime_webgpu-1.27.0-cp313-cp313-win_amd64.whl"
|
||||
ORT_WEBGPU_SHA=c45377099fcf23ae87427eb52e2b1d35415cabb4e3419dc645f9ee08f730e9aa
|
||||
# The Windows wheel leaves OpenVINO to the `openvino` wheel; its DLLs
|
||||
# go in the same directory, which the engine puts on the DLL search
|
||||
# path when it chooses this runtime.
|
||||
OPENVINO_LIBS="${PYPI}/3c/e5/da52a86cc5f1c86871002712429cdcca0c0dbff12dfbce730b05db60340b/openvino-2025.4.1-20426-cp313-cp313-win_amd64.whl"
|
||||
OPENVINO_LIBS_SHA=a28eef35e3ed497c3238eb8f3d1ee90647c449707a8b0a7630758cd15555d8dd
|
||||
OPENVINO_KEEP=(onnxruntime.dll onnxruntime_providers_shared.dll
|
||||
onnxruntime_providers_openvino.dll openvino.dll
|
||||
openvino_onnx_frontend.dll openvino_intel_gpu_plugin.dll
|
||||
tbb12.dll tbbmalloc.dll)
|
||||
WEBGPU_KEEP=(onnxruntime.dll onnxruntime_providers_shared.dll dxcompiler.dll dxil.dll)
|
||||
;;
|
||||
*)
|
||||
echo "error: platform is linux or windows, not ${PLATFORM}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
WORK="$(mktemp -d)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
|
||||
# fetch URL SHA256 DIR: download a wheel, check it, unpack it into DIR.
|
||||
fetch() {
|
||||
local url="$1" sha="$2" dir="$3" whl
|
||||
whl="${WORK}/$(basename "${url}")"
|
||||
echo "==> $(basename "${url}")"
|
||||
curl -fsSL -o "${whl}" "${url}"
|
||||
echo "${sha} ${whl}" | sha256sum -c --quiet - || {
|
||||
echo "error: $(basename "${url}") does not match its pinned SHA-256" >&2
|
||||
exit 1
|
||||
}
|
||||
mkdir -p "${dir}"
|
||||
unzip -q -o "${whl}" -d "${dir}"
|
||||
}
|
||||
|
||||
# keep FROM TO NAMES...: copy only NAMES, every one of which must exist.
|
||||
keep() {
|
||||
local from="$1" to="$2" name
|
||||
shift 2
|
||||
mkdir -p "${to}"
|
||||
for name in "$@"; do
|
||||
local src
|
||||
src="$(find "${from}" -name "${name}" -type f | head -1)"
|
||||
[[ -n "${src}" ]] || {
|
||||
echo "error: ${name} is not in the wheel" >&2
|
||||
exit 1
|
||||
}
|
||||
install -m755 "${src}" "${to}/${name}"
|
||||
done
|
||||
}
|
||||
|
||||
fetch "${ORT_OPENVINO}" "${ORT_OPENVINO_SHA}" "${WORK}/openvino"
|
||||
if [[ -n "${OPENVINO_LIBS:-}" ]]; then
|
||||
fetch "${OPENVINO_LIBS}" "${OPENVINO_LIBS_SHA}" "${WORK}/openvino"
|
||||
fi
|
||||
fetch "${ORT_WEBGPU}" "${ORT_WEBGPU_SHA}" "${WORK}/webgpu"
|
||||
|
||||
rm -rf "${DEST}/openvino" "${DEST}/webgpu"
|
||||
keep "${WORK}/openvino" "${DEST}/openvino" "${OPENVINO_KEEP[@]}"
|
||||
keep "${WORK}/webgpu" "${DEST}/webgpu" "${WEBGPU_KEEP[@]}"
|
||||
# The wheels' own licence and notice files — ONNX Runtime keeps its in the
|
||||
# package directory, OpenVINO in `dist-info` — beside what they cover,
|
||||
# each under the name of the directory it came from: two wheels unpacked
|
||||
# into one directory both carry a `LICENSE`.
|
||||
for rt in openvino webgpu; do
|
||||
find "${WORK}/${rt}" -maxdepth 3 -type f \
|
||||
\( -iname 'LICENSE*' -o -iname 'NOTICE*' -o -iname 'ThirdPartyNotices*' \) |
|
||||
while IFS= read -r f; do
|
||||
wheel="$(basename "$(dirname "${f}")" .dist-info)"
|
||||
[[ "${wheel}" == licenses ]] && wheel="$(basename "$(dirname "$(dirname "${f}")")" .dist-info)"
|
||||
install -m644 "${f}" "${DEST}/${rt}/${wheel}.$(basename "${f}")"
|
||||
done
|
||||
done
|
||||
|
||||
# The Linux wheel bundles OpenVINO without its licence; Apache-2.0 asks
|
||||
# for the text beside the binaries. From the tag the libraries were built at.
|
||||
if [[ "${PLATFORM}" == linux ]]; then
|
||||
curl -fsSL -o "${DEST}/openvino/openvino-2025.4.1.LICENSE" \
|
||||
"https://raw.githubusercontent.com/openvinotoolkit/openvino/2025.4.1/LICENSE"
|
||||
echo "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4 ${DEST}/openvino/openvino-2025.4.1.LICENSE" |
|
||||
sha256sum -c --quiet -
|
||||
fi
|
||||
|
||||
du -sh "${DEST}/openvino" "${DEST}/webgpu"
|
||||
@@ -844,7 +844,7 @@ def develop_zoom():
|
||||
|
||||
# The methods in the order the scene visits them: `Best` is what the
|
||||
# photograph opens with, then each smaller network, then none.
|
||||
DENOISE_METHODS = ['Best', 'Medium', 'Fast', 'Bilinear']
|
||||
DENOISE_METHODS = ['Best', 'Fast', 'Bilinear']
|
||||
DENOISE_CLOSE_UP = 560 # pixels of canvas, square, around the lamp at 1:1
|
||||
|
||||
|
||||
|
||||
@@ -42,10 +42,8 @@ TABLE = {
|
||||
"xfeat-1024": dict(dir="keypoints", form="int8", feed="xfeat", rewrites=["unfold", "resize"]),
|
||||
"xfeat-768": dict(dir="keypoints", form="int8", feed="xfeat", rewrites=["unfold", "resize"]),
|
||||
"mosaic-fast-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
"mosaic-medium-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
# Exported with its packs as SpaceToDepth already; the rewrite finds
|
||||
# nothing to do.
|
||||
"mosaic-best-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
# Best since 0.24: one network, exported as Fast is, so the same rewrite.
|
||||
"mosaic-hq-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
# Produce the Hexagon's form of each model (docs/dev/inference.md §1.5, §5).
|
||||
#
|
||||
# ./tools/quantise-models.sh PHOTO_DIR [MODEL ...]
|
||||
# ./tools/quantise-models.sh --ranges RANGES.json mosaic-medium-1408
|
||||
# ./tools/quantise-models.sh --ranges RANGES.json mosaic-hq-1408
|
||||
#
|
||||
# Writes `<stem>.<form>.onnx` beside each canonical file under models/: a QDQ
|
||||
# graph from QNN's own quantisation config, per-channel weights, in the form
|
||||
|
||||
@@ -179,7 +179,7 @@ impl DevelopSession {
|
||||
iso: self.denoise.iso,
|
||||
cache_key: self.denoise.file_hash.as_ref().map(|h| h.key(&model)),
|
||||
model,
|
||||
halo: net.halo,
|
||||
net,
|
||||
cancel,
|
||||
})
|
||||
}
|
||||
@@ -316,8 +316,8 @@ struct Work {
|
||||
profile: Option<Vec<(f32, f32)>>,
|
||||
iso: Option<u32>,
|
||||
model: std::path::PathBuf,
|
||||
/// The context `model` needs past a tile's kept centre.
|
||||
halo: usize,
|
||||
/// The network `model` is: its context and its whole-frame sibling.
|
||||
net: dr_denoise::Shipped,
|
||||
cancel: Arc<AtomicBool>,
|
||||
cache_key: Option<String>,
|
||||
}
|
||||
@@ -351,8 +351,8 @@ impl Work {
|
||||
.map_err(|e| e.to_string())?;
|
||||
let noise = dr_denoise::noise::for_frame_with(&raw, self.profile.as_deref(), self.iso)
|
||||
.ok_or("this photograph gives no way to measure its noise")?;
|
||||
let mut net = dr_denoise::onnx::OnnxNet::from_path(&self.model, self.halo)
|
||||
.map_err(|e| e.to_string())?;
|
||||
let mut net =
|
||||
dr_denoise::onnx::OnnxNet::open(&self.model, self.net).map_err(|e| e.to_string())?;
|
||||
let rung = net
|
||||
.rung()
|
||||
.map(|r| r.label().to_string())
|
||||
|
||||
@@ -35,9 +35,11 @@ pub fn init(runtime_dirs: Vec<PathBuf>) {
|
||||
(Role::EyeClassifier, crate::library::SUNGLASSES_MODEL),
|
||||
(Role::Inpainter, crate::library::INPAINT_MODEL),
|
||||
]);
|
||||
wanted.extend(
|
||||
[dr_denoise::FAST, dr_denoise::MEDIUM, dr_denoise::BEST].map(|n| (Role::Denoiser, n.file)),
|
||||
);
|
||||
let denoisers = [dr_denoise::FAST, dr_denoise::BEST];
|
||||
wanted.extend(denoisers.map(|n| (Role::Denoiser, n.file)));
|
||||
// Their any-size siblings, which the engine compiles only on a rung
|
||||
// that runs whole frames (TensorRT; denoise.md §14).
|
||||
wanted.extend(denoisers.map(|n| (Role::WholeDenoiser, n.whole)));
|
||||
let models: Vec<(Role, PathBuf)> = wanted
|
||||
.into_iter()
|
||||
.filter_map(|(role, name)| Some((role, crate::library::shared_model(name)?)))
|
||||
|
||||
@@ -227,7 +227,6 @@ fn catalogued(key: &str) -> Option<&'static str> {
|
||||
"param.learned_denoise.method" => "Method",
|
||||
"param.learned_denoise.method.bilinear" => "Bilinear",
|
||||
"param.learned_denoise.method.fast" => "Fast",
|
||||
"param.learned_denoise.method.medium" => "Medium",
|
||||
"param.learned_denoise.method.best" => "Best",
|
||||
// How strongly: 100 % is the network's result, and less puts the
|
||||
// removed noise's brightness back as grain.
|
||||
|
||||
@@ -399,7 +399,6 @@ pub fn denoise_network(
|
||||
match method {
|
||||
Method::Bilinear => None,
|
||||
Method::Fast => Some(dr_denoise::FAST),
|
||||
Method::Medium => Some(dr_denoise::MEDIUM),
|
||||
Method::Best => Some(dr_denoise::BEST),
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user