Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8aa10cd249 | ||
|
|
c2cfacd7d3 | ||
|
|
a9271c4850 | ||
|
|
4b71ef0947 | ||
|
|
1e1aa1442b | ||
|
|
3761281dd1 | ||
|
|
9cba420fd5 | ||
|
|
56f4180347 | ||
|
|
417cba8b4d | ||
|
|
7966bf2dd8 | ||
|
|
4822991bec | ||
|
|
4b33754482 | ||
|
|
b3999bbc0c | ||
|
|
43402bfe5b | ||
|
|
cb97ebe7ac | ||
|
|
2ced6f114f | ||
|
|
85dee4375b | ||
|
|
87c405eb46 | ||
|
|
555ec0efb3 | ||
|
|
0291b80672 | ||
|
|
ef71bb3289 | ||
|
|
08b7d23e86 | ||
|
|
a116325991 | ||
|
|
f20e481358 | ||
|
|
16a5957aa7 | ||
|
|
5736a21a3a | ||
|
|
b6ca7be185 | ||
|
|
06422a07db | ||
|
|
14f08a565f | ||
|
|
0c9d564586 | ||
|
|
75e12441fd | ||
|
|
3f8f909e41 | ||
|
|
5ffd54ba43 | ||
|
|
5a8c3e4c40 | ||
|
|
0e6ac09fd5 | ||
|
|
948f6c3ed2 | ||
|
|
8f9e59b9fa | ||
|
|
83f0461ce7 | ||
|
|
26e50ae723 | ||
|
|
25dc0d0179 | ||
|
|
a36ec98b36 | ||
|
|
c343ac79d3 | ||
|
|
2e7f14dafe |
@@ -487,10 +487,11 @@ jobs:
|
||||
wine "$SETUP" /S 2>/dev/null
|
||||
INST=$(echo "$HOME"/.wine/drive_c/users/*/AppData/Local/Programs/DarkRoom)
|
||||
ls "$INST"
|
||||
# As many files as package.sh stages: everything but the READMEs in
|
||||
# the directories it copies. A literal here went stale the first
|
||||
# time a model was added.
|
||||
WANT=$(find models/face models/scene models/inpaint models/denoise -maxdepth 1 -type f ! -name README.md | wc -l)
|
||||
# As many files as package.sh stages: everything but the READMEs and
|
||||
# the Hexagon's quantised siblings in the directories it copies. A
|
||||
# literal here went stale the first time a model was added.
|
||||
WANT=$(find models/face models/scene models/inpaint models/denoise -maxdepth 1 -type f ! -name README.md \
|
||||
! -name '*.int8.onnx' ! -name '*.a16w8.onnx' ! -name '*.a16w16.onnx' | wc -l)
|
||||
GOT=$(ls "$INST/models" | wc -l)
|
||||
[ "$GOT" = "$WANT" ] || { echo "FAIL: expected $WANT model files, installed $GOT"; exit 1; }
|
||||
# The manual, and every picture it shows, counted the same way.
|
||||
@@ -498,6 +499,12 @@ jobs:
|
||||
WANT=$(ls docs/manual/media | wc -l)
|
||||
GOT=$(ls "$INST/manual/media" | wc -l)
|
||||
[ "$GOT" = "$WANT" ] || { echo "FAIL: expected $WANT manual pictures, installed $GOT"; exit 1; }
|
||||
# Both bundled runtimes, each with its provider beside it
|
||||
# (tools/fetch-bundled-runtimes.sh).
|
||||
for f in openvino/onnxruntime.dll openvino/onnxruntime_providers_openvino.dll \
|
||||
openvino/openvino.dll webgpu/onnxruntime.dll webgpu/dxcompiler.dll; do
|
||||
[ -f "$INST/runtimes/$f" ] || { echo "FAIL: runtimes/$f not installed"; exit 1; }
|
||||
done
|
||||
wine reg query 'HKCU\Software\Microsoft\Windows\CurrentVersion\Uninstall\DarkRoom' 2>/dev/null \
|
||||
| grep -q DisplayVersion || { echo "FAIL: no uninstall registry key"; exit 1; }
|
||||
wine "$INST/darkroom.exe" --version 2>/dev/null | grep -q '^darkroom-desktop ' \
|
||||
|
||||
Generated
+27
-26
@@ -1265,7 +1265,7 @@ checksum = "f27ae1dd37df86211c42e150270f82743308803d90a6f6e6651cd730d5e1732f"
|
||||
|
||||
[[package]]
|
||||
name = "darkroom-android"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"android_logger",
|
||||
"dr-plat",
|
||||
@@ -1278,7 +1278,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "darkroom-desktop"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-plat",
|
||||
@@ -1454,7 +1454,7 @@ checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76"
|
||||
|
||||
[[package]]
|
||||
name = "dr-bench"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-catalog",
|
||||
@@ -1471,7 +1471,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-catalog"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-face",
|
||||
"dr-plat",
|
||||
@@ -1486,7 +1486,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-decode"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"env_logger",
|
||||
@@ -1500,7 +1500,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-denoise"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-gpu",
|
||||
@@ -1517,7 +1517,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-export"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-gpu",
|
||||
@@ -1536,7 +1536,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-face"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-inference-engine",
|
||||
"env_logger",
|
||||
@@ -1549,7 +1549,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-film"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"log",
|
||||
"serde",
|
||||
@@ -1558,7 +1558,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-gpu"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
"dr-decode",
|
||||
@@ -1576,7 +1576,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-inference-engine"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"libloading",
|
||||
@@ -1591,7 +1591,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-ingest"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-plat",
|
||||
"dr-types",
|
||||
@@ -1603,7 +1603,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-lens"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"lensfun",
|
||||
"log",
|
||||
@@ -1611,7 +1611,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-pano"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-inference-engine",
|
||||
@@ -1625,7 +1625,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-pipeline"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"log",
|
||||
@@ -1634,7 +1634,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-plat"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"android-native-keyring-store",
|
||||
"dr-types",
|
||||
@@ -1650,7 +1650,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-preset-xmp"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-pipeline",
|
||||
"log",
|
||||
@@ -1660,7 +1660,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-segment"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-inference-engine",
|
||||
"env_logger",
|
||||
@@ -1673,7 +1673,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-plat",
|
||||
@@ -1687,7 +1687,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync-folder"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-sync",
|
||||
@@ -1699,7 +1699,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync-nextcloud"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-decode",
|
||||
@@ -1721,7 +1721,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-thumbs"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"jpeg-encoder",
|
||||
@@ -1733,7 +1733,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-types"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -1742,7 +1742,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-ui"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"async-trait",
|
||||
@@ -1768,6 +1768,7 @@ dependencies = [
|
||||
"dr-types",
|
||||
"dr-xmp",
|
||||
"env_logger",
|
||||
"half",
|
||||
"i-slint-backend-testing",
|
||||
"jni 0.22.4",
|
||||
"log",
|
||||
@@ -1791,7 +1792,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-xmp"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"log",
|
||||
@@ -7125,7 +7126,7 @@ checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3"
|
||||
|
||||
[[package]]
|
||||
name = "traceability"
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"proc-macro2",
|
||||
|
||||
+9
-1
@@ -33,7 +33,7 @@ members = [
|
||||
exclude = ["third_party"]
|
||||
|
||||
[workspace.package]
|
||||
version = "0.21.0"
|
||||
version = "0.24.0"
|
||||
edition = "2021"
|
||||
rust-version = "1.92"
|
||||
license = "GPL-3.0-or-later"
|
||||
@@ -278,6 +278,14 @@ opt-level = 0
|
||||
lto = "thin"
|
||||
codegen-units = 1
|
||||
|
||||
# Except dr-ui. Slint expands the `.slint` files into ~27 MB of Rust
|
||||
# (`out/app.rs`), and at one codegen unit LLVM optimises all of it on a single
|
||||
# thread: 13.5 minutes of a release build with the other cores idle. The code
|
||||
# it holds is UI glue — property bindings and callbacks — not the image work,
|
||||
# which lives in the crates above that keep the single unit.
|
||||
[profile.release.package.dr-ui]
|
||||
codegen-units = 16
|
||||
|
||||
# A release build that can say where it panicked: line tables, so a crash
|
||||
# record's backtrace (`dr_plat::crash`) reads `file.rs:123` rather than bare
|
||||
# addresses. The macOS build uses it (docs/dev/macos.md) — no one here can
|
||||
|
||||
@@ -201,7 +201,7 @@ controls, its place in the chain and its tests.
|
||||
|
||||
## Where it stands
|
||||
|
||||
**0.21.0**, thirty-five tagged releases in. 193 numbered requirements in
|
||||
**0.24.0**, thirty-nine tagged releases in. 193 numbered requirements in
|
||||
scope, 85% of them claimed by code and [traced to it](docs/dev/traceability.md);
|
||||
the rest are written down rather than merely absent.
|
||||
|
||||
|
||||
@@ -73,6 +73,14 @@
|
||||
android:requestLegacyExternalStorage="true"
|
||||
android:supportsRtl="true">
|
||||
|
||||
<!-- The DSP's RPC library, which QNN's Hexagon stub loads. From API 31
|
||||
an app's linker namespace refuses a vendor library the manifest
|
||||
does not name, and QNN then fails to create its device
|
||||
(QNN_DEVICE_ERROR_INVALID_CONFIG) before it reaches the DSP:
|
||||
every model ran on the CPU on 0.22.0. Not required, so a device
|
||||
without one still installs and stays on the CPU. -->
|
||||
<uses-native-library android:name="libcdsprpc.so" android:required="false" />
|
||||
|
||||
<!-- NativeActivity rather than a Kotlin Activity: android-activity's
|
||||
glue loads libdarkroom.so and calls android_main. `android.app.lib_name`
|
||||
is how it learns which library to load, and must match [lib].name.
|
||||
|
||||
@@ -332,27 +332,38 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
// eyes-open filter has something to read, and a tablet has no other way
|
||||
// to get them either.
|
||||
//
|
||||
// The int8 forms beside the three detectors are what the Hexagon runs
|
||||
// (docs/dev/inference.md §5); the engine loads the sibling when the probe
|
||||
// chose that rung and ignores it otherwise.
|
||||
const BUNDLED: [(&std::ffi::CStr, &str); 15] = [
|
||||
// The quantised siblings — `.a16w8.onnx`, `.a16w16.onnx` — are what the
|
||||
// Hexagon runs (docs/dev/inference.md §1.5), each in the narrowest form
|
||||
// that held that model's accuracy on the tablet; the engine loads the
|
||||
// sibling when the probe chose that rung and ignores it otherwise. The
|
||||
// segmenter's and XFeat's forms are compiled into the binary instead,
|
||||
// beside their f32 graphs.
|
||||
const BUNDLED: [(&std::ffi::CStr, &str); 21] = [
|
||||
(c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"),
|
||||
(
|
||||
c"models/scrfd_500m_640.int8.onnx",
|
||||
"scrfd_500m_640.int8.onnx",
|
||||
c"models/scrfd_500m_640.a16w8.onnx",
|
||||
"scrfd_500m_640.a16w8.onnx",
|
||||
),
|
||||
(c"models/scrfd_2.5g_640.onnx", "scrfd_2.5g_640.onnx"),
|
||||
(
|
||||
c"models/scrfd_2.5g_640.int8.onnx",
|
||||
"scrfd_2.5g_640.int8.onnx",
|
||||
c"models/scrfd_2.5g_640.a16w8.onnx",
|
||||
"scrfd_2.5g_640.a16w8.onnx",
|
||||
),
|
||||
(c"models/scrfd_10g_640.onnx", "scrfd_10g_640.onnx"),
|
||||
(c"models/scrfd_10g_640.int8.onnx", "scrfd_10g_640.int8.onnx"),
|
||||
(
|
||||
c"models/scrfd_10g_640.a16w8.onnx",
|
||||
"scrfd_10g_640.a16w8.onnx",
|
||||
),
|
||||
(c"models/arcface_mbf_b1.onnx", "arcface_mbf_b1.onnx"),
|
||||
(c"models/2d106det_b1.onnx", "2d106det_b1.onnx"),
|
||||
(c"models/2d106det_b1.a16w8.onnx", "2d106det_b1.a16w8.onnx"),
|
||||
(c"models/ocec_s_b1.onnx", "ocec_s_b1.onnx"),
|
||||
(c"models/sgc_l_48_b1.onnx", "sgc_l_48_b1.onnx"),
|
||||
(c"models/yolo26s-sem-ade20k.onnx", "yolo26s-sem-ade20k.onnx"),
|
||||
(
|
||||
c"models/yolo26s-sem-ade20k.a16w16.onnx",
|
||||
"yolo26s-sem-ade20k.a16w16.onnx",
|
||||
),
|
||||
(
|
||||
c"models/yolo26s-sem-ade20k.classes.json",
|
||||
"yolo26s-sem-ade20k.classes.json",
|
||||
@@ -360,7 +371,19 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
(c"models/categories.txt", "categories.txt"),
|
||||
// The panorama border filler (FR-MRG-4); MIT, 28 MB.
|
||||
(c"models/migan-512.onnx", "migan-512.onnx"),
|
||||
(c"models/mosaic-1408.onnx", "mosaic-1408.onnx"),
|
||||
(c"models/migan-512.a16w16.onnx", "migan-512.a16w16.onnx"),
|
||||
// The learned demosaic and denoise, one network per method
|
||||
// (FR-DEV-3g), each with the 16-bit form the Hexagon runs.
|
||||
(c"models/mosaic-fast-1408.onnx", "mosaic-fast-1408.onnx"),
|
||||
(
|
||||
c"models/mosaic-fast-1408.a16w16.onnx",
|
||||
"mosaic-fast-1408.a16w16.onnx",
|
||||
),
|
||||
(c"models/mosaic-hq-1408.onnx", "mosaic-hq-1408.onnx"),
|
||||
(
|
||||
c"models/mosaic-hq-1408.a16w16.onnx",
|
||||
"mosaic-hq-1408.a16w16.onnx",
|
||||
),
|
||||
];
|
||||
|
||||
let dir = dr_ui::shared_face_models_dir();
|
||||
@@ -423,8 +446,13 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
// on a first launch they were not on disk until this line. The runtime
|
||||
// is in the APK's native library directory beside `libdarkroom.so`,
|
||||
// which is also where Qualcomm's DSP loader has to be pointed for the
|
||||
// Hexagon skel (docs/dev/inference.md §3, §8).
|
||||
dr_ui::inference::init(native_library_dir().into_iter().collect());
|
||||
// Hexagon skel (docs/dev/inference.md §3, §8). Two of them: the QNN
|
||||
// build, and the generic WebGPU build by its file name, which the
|
||||
// engine opens only when the first does not fit the SoC (§3.2).
|
||||
let runtimes = native_library_dir()
|
||||
.map(|dir| vec![dir.clone(), dir.join("libonnxruntime_generic.so")])
|
||||
.unwrap_or_default();
|
||||
dr_ui::inference::init(runtimes);
|
||||
}
|
||||
|
||||
/// The directory the system unpacked this APK's native libraries into.
|
||||
@@ -510,6 +538,19 @@ mod tests {
|
||||
Some(value.to_string())
|
||||
}
|
||||
|
||||
/// From API 31 the linker refuses a vendor library the manifest does not
|
||||
/// name, and QNN cannot create its Hexagon device without the DSP's RPC
|
||||
/// library: 0.22.0 ran every model on the CPU for want of this line.
|
||||
#[test]
|
||||
fn the_npu_can_reach_the_dsp() {
|
||||
assert!(
|
||||
manifest().contains(
|
||||
r#"<uses-native-library android:name="libcdsprpc.so" android:required="false" />"#
|
||||
),
|
||||
"libcdsprpc.so must be declared, and not required"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_gallery_can_open_a_photograph_in_this_app() {
|
||||
let manifest = manifest();
|
||||
|
||||
@@ -106,24 +106,37 @@ fn main() -> anyhow::Result<()> {
|
||||
/// providers, or against the wrong cuDNN — and a system copy whose providers
|
||||
/// do not load is not a problem, only a slower app: the probe builds a real
|
||||
/// session before believing a provider.
|
||||
///
|
||||
/// The order breaks ties only. The engine opens every runtime on this list
|
||||
/// and loads the one whose providers fit the GPU (inference.md §3.2), so a
|
||||
/// package's bundled builds — `runtimes/openvino` and `runtimes/webgpu`
|
||||
/// beside each place a package installs to, from
|
||||
/// `tools/fetch-bundled-runtimes.sh` — sit beside a CUDA or ROCm runtime
|
||||
/// without hiding it.
|
||||
fn runtime_dirs() -> Vec<PathBuf> {
|
||||
// A place a package installs to, and the bundled runtimes under it.
|
||||
fn packaged(dirs: &mut Vec<PathBuf>, base: PathBuf) {
|
||||
dirs.push(base.join("runtimes/openvino"));
|
||||
dirs.push(base.join("runtimes/webgpu"));
|
||||
dirs.push(base);
|
||||
}
|
||||
let mut dirs = Vec::new();
|
||||
if let Some(dir) = std::env::var_os("DARKROOM_ORT_DIR") {
|
||||
dirs.push(PathBuf::from(dir));
|
||||
}
|
||||
if let Ok(exe) = std::env::current_exe() {
|
||||
if let Some(bin) = exe.parent() {
|
||||
dirs.push(bin.to_path_buf());
|
||||
dirs.push(bin.join("../lib/darkroom"));
|
||||
packaged(&mut dirs, bin.to_path_buf());
|
||||
packaged(&mut dirs, bin.join("../lib/darkroom"));
|
||||
}
|
||||
}
|
||||
dirs.push(dr_ui::inference::user_runtime_dir());
|
||||
#[cfg(target_os = "linux")]
|
||||
dirs.extend([
|
||||
PathBuf::from("/app/lib/darkroom"),
|
||||
PathBuf::from("/usr/lib/darkroom"),
|
||||
PathBuf::from("/usr/lib"),
|
||||
]);
|
||||
{
|
||||
packaged(&mut dirs, PathBuf::from("/app/lib/darkroom"));
|
||||
packaged(&mut dirs, PathBuf::from("/usr/lib/darkroom"));
|
||||
dirs.push(PathBuf::from("/usr/lib"));
|
||||
}
|
||||
// An app bundle keeps its libraries in `Contents/Frameworks`, beside
|
||||
// the `Contents/MacOS` the executable is in; then Homebrew's
|
||||
// `onnxruntime`, Apple silicon's prefix before Intel's. Homebrew's build
|
||||
|
||||
@@ -2,16 +2,18 @@
|
||||
//!
|
||||
//! ```sh
|
||||
//! DARKROOM_ORT_DIR=~/.local/share/darkroom/runtime \
|
||||
//! cargo run --release -p dr-denoise --features native --example denoise_raw -- IMG.CR2 out
|
||||
//! cargo run --release -p dr-denoise --features native --example denoise_raw -- IMG.CR2 out [fast|best]
|
||||
//! ```
|
||||
//!
|
||||
//! Decode, the app's hot-pixel pass, the frame's noise from its best source,
|
||||
//! then the shipped network under the inference engine on whatever rung this
|
||||
//! then one of the shipped networks (`best` unless named) under the inference engine on whatever rung this
|
||||
//! machine probes to. Writes `out.npy` — the active area, `h×w×3` f32 linear
|
||||
//! camera RGB — for comparison with the training repo's own path
|
||||
//! (`tools/compare_rust.py` in darkroom-denoise). `DARKROOM_ORT_DIR` points
|
||||
//! at an ONNX Runtime build; the engine's cache goes to `DR_ENGINE_CACHE` or
|
||||
//! a temporary directory.
|
||||
//! a temporary directory. The whole-frame network (`mosaic-hq.onnx` beside
|
||||
//! the fixed file) runs where the rung takes any size; `DR_PLAN=tiles` keeps
|
||||
//! the 1408² tiles anyway, to compare the two.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::time::{Duration, Instant};
|
||||
@@ -23,11 +25,26 @@ fn main() {
|
||||
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("warn")).init();
|
||||
let mut args = std::env::args().skip(1);
|
||||
let (Some(input), Some(out)) = (args.next(), args.next()) else {
|
||||
eprintln!("usage: denoise_raw RAW OUT_PREFIX");
|
||||
eprintln!("usage: denoise_raw RAW OUT_PREFIX [fast|best]");
|
||||
std::process::exit(2);
|
||||
};
|
||||
let model =
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../models/denoise/mosaic-1408.onnx");
|
||||
let shipped = match args.next().as_deref() {
|
||||
None | Some("best") => dr_denoise::BEST,
|
||||
Some("fast") => dr_denoise::FAST,
|
||||
Some(other) => {
|
||||
eprintln!("no network called {other}: fast or best");
|
||||
std::process::exit(2);
|
||||
}
|
||||
};
|
||||
let model = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("../../models/denoise")
|
||||
.join(shipped.file);
|
||||
let whole = model.with_file_name(shipped.whole);
|
||||
let tiles_only = std::env::var("DR_PLAN").is_ok_and(|p| p == "tiles");
|
||||
let mut models = vec![(Role::Denoiser, model.clone())];
|
||||
if whole.is_file() && !tiles_only {
|
||||
models.push((Role::WholeDenoiser, whole));
|
||||
}
|
||||
let cache = std::env::var_os("DR_ENGINE_CACHE")
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| std::env::temp_dir().join("dr-denoise-engines"));
|
||||
@@ -38,7 +55,7 @@ fn main() {
|
||||
.into_iter()
|
||||
.collect(),
|
||||
cache_dir: cache,
|
||||
models: vec![(Role::Denoiser, model.clone())],
|
||||
models,
|
||||
embedded: Vec::new(),
|
||||
ceiling: None,
|
||||
threads: 0,
|
||||
@@ -91,10 +108,20 @@ fn main() {
|
||||
noise.col
|
||||
);
|
||||
|
||||
let mut net = OnnxNet::from_path(&model).expect("model");
|
||||
let mut net = if tiles_only {
|
||||
OnnxNet::open_tiled(&model, shipped)
|
||||
} else {
|
||||
OnnxNet::open(&model, shipped)
|
||||
}
|
||||
.expect("model");
|
||||
println!(
|
||||
"rung {}",
|
||||
net.rung().map(|r| r.label()).unwrap_or("?")
|
||||
"rung {} · {}",
|
||||
net.rung().map(|r| r.label()).unwrap_or("?"),
|
||||
if net.whole_frame() {
|
||||
"whole frame"
|
||||
} else {
|
||||
"1408² tiles"
|
||||
}
|
||||
);
|
||||
let t = Instant::now();
|
||||
let rgb = dr_denoise::denoise(&raw, &noise, &mut net, &mut |done, total| {
|
||||
|
||||
@@ -19,12 +19,41 @@
|
||||
pub mod noise;
|
||||
#[cfg(feature = "onnx")]
|
||||
pub mod onnx;
|
||||
pub mod repair;
|
||||
pub mod tile;
|
||||
|
||||
use dr_decode::RawImage;
|
||||
|
||||
pub use noise::{NoiseModel, Source};
|
||||
pub use tile::{TileNet, HALO};
|
||||
pub use tile::{Sizes, TileNet, HALO};
|
||||
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// A network the app ships in `models/denoise/`: its fixed-tile file, the
|
||||
/// same network with any height and width for a whole frame (§14), and the
|
||||
/// context it needs past a tile's kept centre (§13).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Shipped {
|
||||
pub file: &'static str,
|
||||
pub whole: &'static str,
|
||||
pub halo: usize,
|
||||
}
|
||||
|
||||
/// The smallest student: 0.9 M parameters, 11 GMAC a megapixel.
|
||||
pub const FAST: Shipped = Shipped {
|
||||
file: "mosaic-fast-1408.onnx",
|
||||
whole: "mosaic-fast.onnx",
|
||||
halo: HALO,
|
||||
};
|
||||
/// One network of the first release's shape, 3.2 M parameters and 48 GMAC a
|
||||
/// megapixel, taught by the mixture of experts that was Best until 0.24:
|
||||
/// its edges at a third of its work (denoise.md §15). A new file name, not
|
||||
/// the old Medium's or Best's: the result cache keys a model by its name
|
||||
/// and size, and this one is byte for byte the old Medium's size.
|
||||
pub const BEST: Shipped = Shipped {
|
||||
file: "mosaic-hq-1408.onnx",
|
||||
whole: "mosaic-hq.onnx",
|
||||
halo: HALO,
|
||||
};
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum DenoiseError {
|
||||
@@ -67,12 +96,23 @@ pub fn denoise(
|
||||
}
|
||||
let active = noise::active(raw);
|
||||
let (h, w) = (active.h, active.w);
|
||||
// The active area laid out once, then the noise-aware repair the model
|
||||
// was trained behind (see `repair`).
|
||||
let mut mosaic: Vec<f32> = (0..h * w).map(|i| active.at(i / w, i % w)).collect();
|
||||
let pattern = raw.cfa_pattern;
|
||||
let repaired = repair::repair(&mut mosaic, h, w, repair::REPAIR_K, &|y, x, v| {
|
||||
noise.sigma(pattern.colour_at(x as u32, y as u32) as usize, v)
|
||||
});
|
||||
log::info!(
|
||||
"learned denoise: {repaired} photosites beyond {}σ of every neighbour repaired",
|
||||
repair::REPAIR_K
|
||||
);
|
||||
tile::run_tiled(
|
||||
net,
|
||||
h,
|
||||
w,
|
||||
raw.cfa_pattern,
|
||||
&|y, x| active.at(y, x),
|
||||
&|y, x| mosaic[y * w + x],
|
||||
&|c, v| noise.sigma(c, v),
|
||||
progress,
|
||||
)
|
||||
|
||||
+94
-18
@@ -5,27 +5,85 @@
|
||||
//! returns `rgb`, `1×3×1408×1408` (darkroom-denoise `denoise/export.py`,
|
||||
//! fixed shape because every model the engine runs is). The engine picks the
|
||||
//! rung: fp16 on TensorRT and MIGraphX, which measured 0.00 dB from f32; f32
|
||||
//! on CUDA and the CPU; never the Hexagon, where int8 lost 6–9 dB.
|
||||
//! on CUDA and the CPU; on the Hexagon the `.a16w16.onnx` sibling, 16-bit
|
||||
//! activations and weights, 0.00 dB from f32 on the tablet itself where int8
|
||||
//! lost 5–9 dB (docs/dev/inference.md §1.5). That sibling is the same network
|
||||
//! with the Bayer packing spelled `SpaceToDepth`, which QNN can hold and the
|
||||
//! 6-D reshape it replaces it cannot.
|
||||
//!
|
||||
//! Each network also ships with any height and width (`mosaic-hq.onnx`
|
||||
//! beside `mosaic-hq-1408.onnx`, darkroom-denoise `tools/export_whole.py`,
|
||||
//! identical to the fixed file at 1408²). Where the rung takes any size, the
|
||||
//! frame runs whole instead of in tiles whose borders are thrown away — a
|
||||
//! 1408² tile keeps 1024², 1.89 photosites computed for each one kept
|
||||
//! (denoise.md §14).
|
||||
|
||||
use crate::tile::TileNet;
|
||||
use crate::DenoiseError;
|
||||
use dr_inference_engine::{Model, Role};
|
||||
use crate::tile::{Sizes, TileNet};
|
||||
use crate::{DenoiseError, Shipped};
|
||||
use dr_inference_engine::{Form, Model, Role};
|
||||
|
||||
/// The edge of the tile the shipped export takes.
|
||||
/// The edge of the tile the shipped fixed-shape export takes.
|
||||
pub const TILE: usize = 1408;
|
||||
|
||||
/// What a whole-frame input's sides must be multiples of: the networks pack
|
||||
/// 2×2 and halve three times, so a side is a whole number of positions at
|
||||
/// their coarsest level only in steps of 16.
|
||||
pub const ALIGN: usize = 16;
|
||||
|
||||
pub struct OnnxNet {
|
||||
model: Model,
|
||||
tile: usize,
|
||||
sizes: Sizes,
|
||||
halo: usize,
|
||||
}
|
||||
|
||||
impl OnnxNet {
|
||||
pub fn from_path(path: &std::path::Path) -> Result<Self, DenoiseError> {
|
||||
/// The network `shipped`, whose fixed-tile file is at `path`.
|
||||
///
|
||||
/// On a rung that runs any input size (TensorRT, the CUDA provider —
|
||||
/// [`dr_inference_engine::whole_frame_limit`]) and with the any-size
|
||||
/// export installed beside it, the whole-frame network: the frame in one
|
||||
/// call, or the fewest large tiles that fit (§14). Its output is the
|
||||
/// fixed tiles' to rounding. Everywhere else, and if the whole-frame
|
||||
/// model will not open, the 1408² tiles.
|
||||
pub fn open(path: &std::path::Path, shipped: Shipped) -> Result<Self, DenoiseError> {
|
||||
if let Some(max) = dr_inference_engine::whole_frame_limit() {
|
||||
let whole = path.with_file_name(shipped.whole);
|
||||
if whole.is_file() {
|
||||
let opened = std::fs::read(&whole)
|
||||
.map_err(DenoiseError::from)
|
||||
.and_then(|bytes| {
|
||||
Ok(dr_inference_engine::open(
|
||||
Role::WholeDenoiser,
|
||||
Form::F32,
|
||||
&bytes,
|
||||
)?)
|
||||
});
|
||||
match opened {
|
||||
Ok(model) => {
|
||||
return Ok(OnnxNet {
|
||||
model,
|
||||
sizes: Sizes::Any { align: ALIGN, max },
|
||||
halo: shipped.halo,
|
||||
})
|
||||
}
|
||||
Err(e) => log::warn!(
|
||||
"learned denoise: {} will not open ({e}); running 1408² tiles",
|
||||
whole.display()
|
||||
),
|
||||
}
|
||||
}
|
||||
}
|
||||
Self::open_tiled(path, shipped)
|
||||
}
|
||||
|
||||
/// The fixed-tile network at `path`, whatever the rung: 1408² tiles.
|
||||
pub fn open_tiled(path: &std::path::Path, shipped: Shipped) -> Result<Self, DenoiseError> {
|
||||
let (path, form) = dr_inference_engine::resolve_model(Role::Denoiser, path);
|
||||
let bytes = std::fs::read(&path)?;
|
||||
Ok(OnnxNet {
|
||||
model: dr_inference_engine::open(Role::Denoiser, form, &bytes)?,
|
||||
tile: TILE,
|
||||
sizes: Sizes::Square(TILE),
|
||||
halo: shipped.halo,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -33,22 +91,38 @@ impl OnnxNet {
|
||||
pub fn rung(&self) -> Result<dr_inference_engine::Rung, DenoiseError> {
|
||||
Ok(self.model.acquire()?.rung())
|
||||
}
|
||||
|
||||
/// Whether this is the whole-frame network.
|
||||
pub fn whole_frame(&self) -> bool {
|
||||
matches!(self.sizes, Sizes::Any { .. })
|
||||
}
|
||||
}
|
||||
|
||||
impl TileNet for OnnxNet {
|
||||
fn tile(&self) -> usize {
|
||||
self.tile
|
||||
fn sizes(&self) -> Sizes {
|
||||
self.sizes
|
||||
}
|
||||
|
||||
fn run(&mut self, mosaic: &[f32], sigma: &[f32]) -> Result<Vec<f32>, DenoiseError> {
|
||||
let n = self.tile;
|
||||
let shape = ndarray::IxDyn(&[1, 1, n, n]);
|
||||
fn halo(&self) -> usize {
|
||||
self.halo
|
||||
}
|
||||
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
mosaic: Vec<f32>,
|
||||
sigma: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), DenoiseError> {
|
||||
let shape = ndarray::IxDyn(&[1, 1, rows, cols]);
|
||||
// The vectors become the tensors: no copy on the way in.
|
||||
let m = ort::value::Tensor::from_array(
|
||||
ndarray::Array::from_shape_vec(shape.clone(), mosaic.to_vec())
|
||||
ndarray::Array::from_shape_vec(shape.clone(), mosaic)
|
||||
.map_err(|e| DenoiseError::Model(e.to_string()))?,
|
||||
)?;
|
||||
let s = ort::value::Tensor::from_array(
|
||||
ndarray::Array::from_shape_vec(shape, sigma.to_vec())
|
||||
ndarray::Array::from_shape_vec(shape, sigma)
|
||||
.map_err(|e| DenoiseError::Model(e.to_string()))?,
|
||||
)?;
|
||||
let acquired = self.model.acquire()?;
|
||||
@@ -56,11 +130,13 @@ impl TileNet for OnnxNet {
|
||||
let outputs = session.run(ort::inputs!["mosaic" => m, "sigma" => s])?;
|
||||
let (shape, data) = outputs[0].try_extract_tensor::<f32>()?;
|
||||
let dims: Vec<i64> = shape.iter().copied().collect();
|
||||
if dims != [1, 3, n as i64, n as i64] {
|
||||
if dims != [1, 3, rows as i64, cols as i64] {
|
||||
return Err(DenoiseError::Model(format!(
|
||||
"output is {dims:?}, expected [1, 3, {n}, {n}]"
|
||||
"output is {dims:?}, expected [1, 3, {rows}, {cols}]"
|
||||
)));
|
||||
}
|
||||
Ok(data.to_vec())
|
||||
// And none on the way out: the frame is written from the runtime's buffer.
|
||||
write(data);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
//! TRACES: FR-DEV-3g
|
||||
//! Hot and dead photosites, judged against the noise, before the network.
|
||||
//!
|
||||
//! The app's own pass (`dr_gpu::Demosaicer::repair_hot_pixels`) runs first and
|
||||
//! takes the gross defects. At high ISO it leaves thousands of photosites per
|
||||
//! 6D frame more than 8σ beyond every neighbour, which the network turns into
|
||||
//! specks. This second pass uses that pass's two tests with the threshold in
|
||||
//! units of the photosite's own σ from the noise model:
|
||||
//!
|
||||
//! - beyond every same-colour neighbour (two photosites away, the 3×3 of its
|
||||
//! plane) by more than `k·σ`, and
|
||||
//! - beyond every adjacent photosite, whatever its colour, by more than
|
||||
//! `k·σ` **and** by a factor of two — what keeps a real point of light,
|
||||
//! which lights its neighbours through the lens and the anti-aliasing
|
||||
//! filter. A margin in σ alone is not enough: on a bright star 8σ is a
|
||||
//! sliver of the signal, and the star would be flattened.
|
||||
//!
|
||||
//! A hot one becomes its brightest same-colour neighbour, a dead one its
|
||||
//! darkest. The shipped model was trained on input repaired exactly so
|
||||
//! (darkroom-denoise `denoise/repair.py`, `--repair-k 8`): the threshold
|
||||
//! belongs to the model, and changes with it. Neighbours off the frame are
|
||||
//! the nearest photosite on it, as the training code reads them.
|
||||
|
||||
/// The threshold the shipped model was trained with, in σ.
|
||||
pub const REPAIR_K: f32 = 8.0;
|
||||
|
||||
/// Repair `mosaic` (`h×w`, row-major, normalised) in place; `sigma(y, x, v)`
|
||||
/// is the photosite's σ. Returns how many photosites changed.
|
||||
pub fn repair(
|
||||
mosaic: &mut [f32],
|
||||
h: usize,
|
||||
w: usize,
|
||||
k: f32,
|
||||
sigma: &(dyn Fn(usize, usize, f32) -> f32 + Sync),
|
||||
) -> usize {
|
||||
let copy = mosaic.to_vec();
|
||||
let original = ©
|
||||
let at = |y: isize, x: isize| {
|
||||
let y = y.clamp(0, h as isize - 1) as usize;
|
||||
let x = x.clamp(0, w as isize - 1) as usize;
|
||||
original[y * w + x]
|
||||
};
|
||||
let threads = std::thread::available_parallelism().map_or(1, |n| n.get());
|
||||
let rows_per = h.div_ceil(threads).max(1);
|
||||
let mut counts = vec![0usize; h.div_ceil(rows_per)];
|
||||
std::thread::scope(|scope| {
|
||||
for ((chunk, rows), count) in mosaic
|
||||
.chunks_mut(rows_per * w)
|
||||
.enumerate()
|
||||
.zip(counts.iter_mut())
|
||||
{
|
||||
let at = &at;
|
||||
scope.spawn(move || {
|
||||
for (i, row) in rows.chunks_mut(w).enumerate() {
|
||||
let y = chunk * rows_per + i;
|
||||
for (x, out) in row.iter_mut().enumerate() {
|
||||
let v = original[y * w + x];
|
||||
let (yi, xi) = (y as isize, x as isize);
|
||||
let (mut s_hi, mut s_lo) = (f32::MIN, f32::MAX);
|
||||
let (mut a_hi, mut a_lo) = (f32::MIN, f32::MAX);
|
||||
for dy in -1isize..=1 {
|
||||
for dx in -1isize..=1 {
|
||||
if dy == 0 && dx == 0 {
|
||||
continue;
|
||||
}
|
||||
let s = at(yi + 2 * dy, xi + 2 * dx);
|
||||
s_hi = s_hi.max(s);
|
||||
s_lo = s_lo.min(s);
|
||||
let a = at(yi + dy, xi + dx);
|
||||
a_hi = a_hi.max(a);
|
||||
a_lo = a_lo.min(a);
|
||||
}
|
||||
}
|
||||
let t = k * sigma(y, x, v);
|
||||
if v - s_hi > t && v - a_hi > t && a_hi < 0.5 * v {
|
||||
*out = s_hi;
|
||||
*count += 1;
|
||||
} else if s_lo - v > t && a_lo - v > t && v < 0.5 * a_lo {
|
||||
*out = s_lo;
|
||||
*count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
counts.iter().sum()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
const N: usize = 16;
|
||||
|
||||
fn flat(level: f32) -> Vec<f32> {
|
||||
vec![level; N * N]
|
||||
}
|
||||
|
||||
fn run(m: &mut [f32]) -> usize {
|
||||
repair(m, N, N, REPAIR_K, &|_, _, _| 0.01)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_hot_photosite_becomes_its_brightest_same_colour_neighbour() {
|
||||
let mut m = flat(0.1);
|
||||
m[8 * N + 8] = 0.5; // 40σ above everything around it
|
||||
m[8 * N + 10] = 0.12; // a same-colour neighbour, a little brighter
|
||||
assert_eq!(run(&mut m), 1);
|
||||
assert_eq!(m[8 * N + 8], 0.12);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_dead_photosite_in_a_lit_area_is_repaired() {
|
||||
let mut m = flat(0.5);
|
||||
m[5 * N + 5] = 0.0;
|
||||
assert_eq!(run(&mut m), 1);
|
||||
assert_eq!(m[5 * N + 5], 0.5);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_point_of_real_light_is_kept() {
|
||||
// Light through a lens lands on a patch: its adjacent photosites are
|
||||
// lit too, so the second test refuses it.
|
||||
let mut m = flat(0.1);
|
||||
for dy in 0..3 {
|
||||
for dx in 0..3 {
|
||||
m[(7 + dy) * N + 7 + dx] = if (dy, dx) == (1, 1) { 0.9 } else { 0.6 };
|
||||
}
|
||||
}
|
||||
let before = m.clone();
|
||||
assert_eq!(run(&mut m), 0);
|
||||
assert_eq!(m, before);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn noise_within_the_threshold_is_left_alone() {
|
||||
let mut m: Vec<f32> = (0..N * N)
|
||||
.map(|i| 0.1 + 0.005 * ((i * 7919 % 13) as f32 - 6.0) / 6.0)
|
||||
.collect();
|
||||
let before = m.clone();
|
||||
assert_eq!(run(&mut m), 0);
|
||||
assert_eq!(m, before);
|
||||
}
|
||||
}
|
||||
+534
-100
@@ -1,12 +1,18 @@
|
||||
//! TRACES: FR-DEV-3g
|
||||
//! A whole frame through a fixed-shape network, exactly (denoise.md §3.4).
|
||||
//! A whole frame through a network, in tiles, exactly (denoise.md §3.4, §14).
|
||||
//!
|
||||
//! The network sees `TILE_IN`² photosites and its output is exact in the
|
||||
//! central `TILE_IN − 2·HALO`: the halo is wider than its receptive field
|
||||
//! (185 photosites, counted from the layers), so a tile's centre equals the
|
||||
//! whole frame's at the same place. The frame is extended by reflection
|
||||
//! about its edge photosites, which keeps every photosite's CFA colour, so
|
||||
//! edge tiles see real context too.
|
||||
//! A tile's output is exact in its centre: past a halo wider than the
|
||||
//! network's receptive field (185 photosites for a single network, more for
|
||||
//! the mixture), a tile's centre equals the whole frame's at the same place.
|
||||
//! The frame is extended by reflection about its edge photosites, which
|
||||
//! keeps every photosite's CFA colour, so edge tiles see real context too.
|
||||
//!
|
||||
//! **Tile sizes.** A fixed-shape network takes one square ([`Sizes::Square`],
|
||||
//! 1408², of which Best keeps 896²). A network exported with any height and
|
||||
//! width ([`Sizes::Any`]) takes the frame whole when it is small enough, and
|
||||
//! otherwise the fewest equal tiles that are: [`plan`] picks the grid that
|
||||
//! computes the fewest photosites. If the first tile of a plan fails — a
|
||||
//! GPU out of memory — the limit is halved and the frame planned again.
|
||||
//!
|
||||
//! **Phase.** The network was trained on RGGB. A frame whose pattern starts
|
||||
//! on another colour is read from one photosite up and/or left — the
|
||||
@@ -15,15 +21,117 @@
|
||||
|
||||
use dr_decode::CfaPattern;
|
||||
|
||||
/// Photosites of context beyond a tile's kept centre, on every side.
|
||||
/// Photosites of context beyond a tile's kept centre, on every side, for a
|
||||
/// single network; a mixture reaches further and says so through
|
||||
/// [`TileNet::halo`].
|
||||
pub const HALO: usize = 192;
|
||||
|
||||
/// A fixed-shape network: `mosaic` and `sigma`, `n×n` RGGB, in; `3×n×n`
|
||||
/// The tiles a network takes.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Sizes {
|
||||
/// One square, `n` photosites a side.
|
||||
Square(usize),
|
||||
/// Any rectangle whose sides are multiples of `align`, at most `max`
|
||||
/// (rows, columns).
|
||||
Any { align: usize, max: (usize, usize) },
|
||||
}
|
||||
|
||||
/// A network: `mosaic` and `sigma`, `rows×cols` RGGB, in; `3×rows×cols`
|
||||
/// planar linear camera RGB out.
|
||||
///
|
||||
/// The inputs are handed over, and the output is lent to `write` rather than
|
||||
/// returned: a 1408² tile is 24 MB of output and a whole frame 300 MB, and
|
||||
/// copying it out of the runtime's buffer and back into the frame was a
|
||||
/// measurable share of a frame's time.
|
||||
pub trait TileNet {
|
||||
/// The edge `n` of the square tile the network takes.
|
||||
fn tile(&self) -> usize;
|
||||
fn run(&mut self, mosaic: &[f32], sigma: &[f32]) -> Result<Vec<f32>, crate::DenoiseError>;
|
||||
/// The tile sizes it takes.
|
||||
fn sizes(&self) -> Sizes;
|
||||
/// Photosites of context it needs past a tile's kept centre: at least
|
||||
/// its receptive field. [`HALO`] unless the network says otherwise.
|
||||
fn halo(&self) -> usize {
|
||||
HALO
|
||||
}
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
mosaic: Vec<f32>,
|
||||
sigma: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), crate::DenoiseError>;
|
||||
}
|
||||
|
||||
/// How a frame is cut: every tile `rows × cols` in, keeping its centre
|
||||
/// `core.0 × core.1` past the halo, on a `grid.0 × grid.1` grid.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Plan {
|
||||
pub rows: usize,
|
||||
pub cols: usize,
|
||||
pub core: (usize, usize),
|
||||
pub grid: (usize, usize),
|
||||
}
|
||||
|
||||
impl Plan {
|
||||
/// Photosites the network computes for the frame.
|
||||
pub fn work(&self) -> usize {
|
||||
self.grid.0 * self.grid.1 * self.rows * self.cols
|
||||
}
|
||||
}
|
||||
|
||||
/// The tiles for an `uh × uw` frame (in the network's phase) at `sizes`, or
|
||||
/// `None` when no tile fits.
|
||||
///
|
||||
/// Square tiles are today's grid. Any-size tiles are equal on each axis, so
|
||||
/// one call shape serves the frame — TensorRT's profile tunes for one, and
|
||||
/// the CUDA provider searches its algorithms once per shape — and the grid
|
||||
/// is the one with the least work: one tile whenever the frame and its
|
||||
/// halo fit under `max`.
|
||||
pub fn plan(uh: usize, uw: usize, halo: usize, sizes: Sizes) -> Option<Plan> {
|
||||
match sizes {
|
||||
Sizes::Square(n) => {
|
||||
if n <= 2 * halo || !(n - 2 * halo).is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
let core = n - 2 * halo;
|
||||
Some(Plan {
|
||||
rows: n,
|
||||
cols: n,
|
||||
core: (core, core),
|
||||
grid: (uh.div_ceil(core), uw.div_ceil(core)),
|
||||
})
|
||||
}
|
||||
Sizes::Any { align, max } => {
|
||||
// An even align keeps every tile origin on an even photosite,
|
||||
// so every tile starts on red.
|
||||
let align = align.max(2).next_multiple_of(2);
|
||||
let axis = |extent: usize, tiles: usize, limit: usize| {
|
||||
let size = (extent.div_ceil(tiles) + 2 * halo).next_multiple_of(align);
|
||||
let core = size.checked_sub(2 * halo)?;
|
||||
(size <= limit && core > 0 && core.is_multiple_of(2)).then_some((size, core))
|
||||
};
|
||||
let mut best: Option<Plan> = None;
|
||||
for gy in 1..=16 {
|
||||
let Some((rows, cy)) = axis(uh, gy, max.0) else {
|
||||
continue;
|
||||
};
|
||||
for gx in 1..=16 {
|
||||
let Some((cols, cx)) = axis(uw, gx, max.1) else {
|
||||
continue;
|
||||
};
|
||||
let p = Plan {
|
||||
rows,
|
||||
cols,
|
||||
core: (cy, cx),
|
||||
grid: (uh.div_ceil(cy), uw.div_ceil(cx)),
|
||||
};
|
||||
if best.is_none_or(|b| p.work() < b.work()) {
|
||||
best = Some(p);
|
||||
}
|
||||
}
|
||||
}
|
||||
best
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Index into `0..n` by reflection about the end photosites, any distance
|
||||
@@ -54,83 +162,218 @@ pub fn rggb_offset(p: CfaPattern) -> Option<(usize, usize)> {
|
||||
/// `sigma(colour, value)`, and return `h×w` interleaved RGB.
|
||||
///
|
||||
/// `progress(done, total)` is called after each tile and stops the run by
|
||||
/// returning `false`, in which case the result is `Ok(None)`.
|
||||
/// returning `false`, in which case the result is `Ok(None)`. An any-size
|
||||
/// network whose first tile fails is planned again with tiles half that
|
||||
/// size, until a tile would keep no centre; then the failure is returned.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn run_tiled(
|
||||
net: &mut dyn TileNet,
|
||||
h: usize,
|
||||
w: usize,
|
||||
pattern: CfaPattern,
|
||||
at: &dyn Fn(usize, usize) -> f32,
|
||||
sigma: &dyn Fn(usize, f32) -> f32,
|
||||
at: &(dyn Fn(usize, usize) -> f32 + Sync),
|
||||
sigma: &(dyn Fn(usize, f32) -> f32 + Sync),
|
||||
progress: &mut dyn FnMut(usize, usize) -> bool,
|
||||
) -> Result<Option<Vec<f32>>, crate::DenoiseError> {
|
||||
let (dy, dx) = rggb_offset(pattern).ok_or_else(|| {
|
||||
crate::DenoiseError::Unsupported(format!("{pattern:?} is not a Bayer pattern"))
|
||||
})?;
|
||||
let n = net.tile();
|
||||
if n <= 2 * HALO || !(n - 2 * HALO).is_multiple_of(2) {
|
||||
return Err(crate::DenoiseError::Model(format!(
|
||||
"tile {n} leaves no even centre past a {HALO} halo"
|
||||
)));
|
||||
let halo = net.halo();
|
||||
let (uh, uw) = (h + dy, w + dx);
|
||||
let mut sizes = net.sizes();
|
||||
loop {
|
||||
let plan = plan(uh, uw, halo, sizes).ok_or_else(|| {
|
||||
crate::DenoiseError::Model(format!(
|
||||
"no tile of {sizes:?} keeps a centre past a {halo} halo"
|
||||
))
|
||||
})?;
|
||||
match run_plan(net, plan, h, w, (dy, dx), halo, at, sigma, progress) {
|
||||
Err(Failed { error, first: true }) => {
|
||||
// The first call of a size is where a GPU runs out of
|
||||
// memory. Halve the larger kept centre of the tile that
|
||||
// failed — not the limit, which may be far above it, and not
|
||||
// the tile, half of which may be all halo — and plan again,
|
||||
// until no smaller tile keeps a centre.
|
||||
let Sizes::Any { align, .. } = sizes else {
|
||||
return Err(error);
|
||||
};
|
||||
let (cr, cc) = plan.core;
|
||||
let smaller = if cr >= cc {
|
||||
(cr / 2 + 2 * halo, plan.cols)
|
||||
} else {
|
||||
(plan.rows, cc / 2 + 2 * halo)
|
||||
};
|
||||
let next = Sizes::Any {
|
||||
align,
|
||||
max: smaller,
|
||||
};
|
||||
if self::plan(uh, uw, halo, next).is_none() {
|
||||
return Err(error);
|
||||
}
|
||||
log::warn!(
|
||||
"learned denoise: a {}×{} tile failed ({error}); trying tiles up to {}×{}",
|
||||
plan.rows,
|
||||
plan.cols,
|
||||
smaller.0,
|
||||
smaller.1
|
||||
);
|
||||
sizes = next;
|
||||
}
|
||||
Err(Failed { error, .. }) => return Err(error),
|
||||
Ok(done) => return Ok(done),
|
||||
}
|
||||
}
|
||||
let core = n - 2 * HALO;
|
||||
}
|
||||
|
||||
/// A run that stopped on an error, and whether it was the plan's first call.
|
||||
struct Failed {
|
||||
error: crate::DenoiseError,
|
||||
first: bool,
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn run_plan(
|
||||
net: &mut dyn TileNet,
|
||||
plan: Plan,
|
||||
h: usize,
|
||||
w: usize,
|
||||
(dy, dx): (usize, usize),
|
||||
halo: usize,
|
||||
at: &(dyn Fn(usize, usize) -> f32 + Sync),
|
||||
sigma: &(dyn Fn(usize, f32) -> f32 + Sync),
|
||||
progress: &mut dyn FnMut(usize, usize) -> bool,
|
||||
) -> Result<Option<Vec<f32>>, Failed> {
|
||||
let Plan {
|
||||
rows: nr,
|
||||
cols: nc,
|
||||
core: (cr, cc),
|
||||
grid: (ty, tx),
|
||||
} = plan;
|
||||
// In unified coordinates the frame spans u ∈ [dy, dy + h), v ∈ [dx, dx + w).
|
||||
let (uh, uw) = (h + dy, w + dx);
|
||||
let (ty, tx) = (uh.div_ceil(core), uw.div_ceil(core));
|
||||
let total = ty * tx;
|
||||
let origins: Vec<(usize, usize)> = (0..ty)
|
||||
.flat_map(|i| (0..tx).map(move |j| (i * cr, j * cc)))
|
||||
.collect();
|
||||
let threads = std::thread::available_parallelism().map_or(1, |n| n.get());
|
||||
|
||||
// One tile's mosaic and σ, gathered on every core: rows are independent.
|
||||
let gather = |u0: usize, v0: usize| {
|
||||
let mut mos = vec![0.0f32; nr * nc];
|
||||
let mut sig = vec![0.0f32; nr * nc];
|
||||
let rows_per = nr.div_ceil(threads).max(1);
|
||||
std::thread::scope(|scope| {
|
||||
for (chunk, (m, s)) in mos
|
||||
.chunks_mut(rows_per * nc)
|
||||
.zip(sig.chunks_mut(rows_per * nc))
|
||||
.enumerate()
|
||||
{
|
||||
scope.spawn(move || {
|
||||
for (i, (mrow, srow)) in m.chunks_mut(nc).zip(s.chunks_mut(nc)).enumerate() {
|
||||
let r = chunk * rows_per + i;
|
||||
// Unified row u = u0 + r − halo; frame row y = u − dy, reflected.
|
||||
let u = u0 as isize + r as isize - halo as isize;
|
||||
let y = reflect(u - dy as isize, h);
|
||||
for c in 0..nc {
|
||||
let v = v0 as isize + c as isize - halo as isize;
|
||||
let x = reflect(v - dx as isize, w);
|
||||
let val = at(y, x);
|
||||
mrow[c] = val;
|
||||
// RGGB colour of the tile position (r, c).
|
||||
srow[c] = sigma([[0, 1], [1, 2]][r & 1][c & 1], val);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
(mos, sig)
|
||||
};
|
||||
|
||||
// Pipelined: the next tile is gathered while the network runs this one,
|
||||
// so the device does not wait on the CPU. A channel of one keeps at
|
||||
// most two tiles' inputs alive.
|
||||
let mut out = vec![0.0f32; h * w * 3];
|
||||
let mut mos = vec![0.0f32; n * n];
|
||||
let mut sig = vec![0.0f32; n * n];
|
||||
// RGGB colour of unified position (u, v).
|
||||
let colour = |u: usize, v: usize| [[0, 1], [1, 2]][u & 1][v & 1];
|
||||
for (k, (i, j)) in (0..ty)
|
||||
.flat_map(|i| (0..tx).map(move |j| (i, j)))
|
||||
.enumerate()
|
||||
{
|
||||
let (u0, v0) = (i * core, j * core);
|
||||
for r in 0..n {
|
||||
// Unified row u = u0 + r − HALO; frame row y = u − dy, reflected.
|
||||
let u = u0 as isize + r as isize - HALO as isize;
|
||||
let y = reflect(u - dy as isize, h);
|
||||
for c in 0..n {
|
||||
let v = v0 as isize + c as isize - HALO as isize;
|
||||
let x = reflect(v - dx as isize, w);
|
||||
let val = at(y, x);
|
||||
mos[r * n + c] = val;
|
||||
sig[r * n + c] = sigma(colour(r, c), val);
|
||||
}
|
||||
let stop = std::sync::atomic::AtomicBool::new(false);
|
||||
let fail = |error, k: usize, stop: &std::sync::atomic::AtomicBool| {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
Failed {
|
||||
error,
|
||||
first: k == 0,
|
||||
}
|
||||
let rgb = net.run(&mos, &sig)?;
|
||||
if rgb.len() != 3 * n * n {
|
||||
return Err(crate::DenoiseError::Model(format!(
|
||||
"network returned {} values for a {n}² tile",
|
||||
rgb.len()
|
||||
)));
|
||||
}
|
||||
for r in HALO..HALO + core {
|
||||
let u = u0 + r - HALO;
|
||||
if u < dy || u >= uh {
|
||||
continue;
|
||||
}
|
||||
let y = u - dy;
|
||||
for c in HALO..HALO + core {
|
||||
let v = v0 + c - HALO;
|
||||
if v < dx || v >= uw {
|
||||
continue;
|
||||
};
|
||||
std::thread::scope(|scope| -> Result<Option<()>, Failed> {
|
||||
let (tx_tiles, rx_tiles) = std::sync::mpsc::sync_channel(1);
|
||||
let (origins, stop, gather) = (&origins, &stop, &gather);
|
||||
scope.spawn(move || {
|
||||
for &(u0, v0) in origins {
|
||||
if stop.load(std::sync::atomic::Ordering::Relaxed) {
|
||||
break;
|
||||
}
|
||||
let x = v - dx;
|
||||
let o = (y * w + x) * 3;
|
||||
for ch in 0..3 {
|
||||
out[o + ch] = rgb[ch * n * n + r * n + c];
|
||||
if tx_tiles.send((u0, v0, gather(u0, v0))).is_err() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
});
|
||||
for k in 0..total {
|
||||
let Ok((u0, v0, (mos, sig))) = rx_tiles.recv() else {
|
||||
break;
|
||||
};
|
||||
let mut wrong = None;
|
||||
let ran = net.run(nr, nc, mos, sig, &mut |rgb: &[f32]| {
|
||||
if rgb.len() != 3 * nr * nc {
|
||||
wrong = Some(rgb.len());
|
||||
return;
|
||||
}
|
||||
// The tile's centre back into the frame: the frame rows it covers,
|
||||
// split across cores (each row is written by one thread only).
|
||||
let (y_lo, y_hi) = (u0.max(dy) - dy, (u0 + cr).min(uh) - dy);
|
||||
let (x_lo, x_hi) = (v0.max(dx) - dx, (v0 + cc).min(uw) - dx);
|
||||
if y_hi > y_lo && x_hi > x_lo {
|
||||
let rows = &mut out[y_lo * w * 3..y_hi * w * 3];
|
||||
let per = (y_hi - y_lo).div_ceil(threads).max(1);
|
||||
std::thread::scope(|scope| {
|
||||
for (chunk, block) in rows.chunks_mut(per * w * 3).enumerate() {
|
||||
scope.spawn(move || {
|
||||
for (i, row) in block.chunks_mut(w * 3).enumerate() {
|
||||
let y = y_lo + chunk * per + i;
|
||||
// Tile row of frame row y: u = y + dy = u0 + r − halo.
|
||||
let r = y + dy + halo - u0;
|
||||
for x in x_lo..x_hi {
|
||||
let c = x + dx + halo - v0;
|
||||
for ch in 0..3 {
|
||||
row[x * 3 + ch] = rgb[ch * nr * nc + r * nc + c];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
if let Err(e) = ran {
|
||||
while rx_tiles.try_recv().is_ok() {}
|
||||
return Err(fail(e, k, stop));
|
||||
}
|
||||
if let Some(len) = wrong {
|
||||
while rx_tiles.try_recv().is_ok() {}
|
||||
return Err(fail(
|
||||
crate::DenoiseError::Model(format!(
|
||||
"network returned {len} values for a {nr}×{nc} tile"
|
||||
)),
|
||||
k,
|
||||
stop,
|
||||
));
|
||||
}
|
||||
if !progress(k + 1, total) {
|
||||
stop.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
// Drain so the producer is not left blocked on a full channel.
|
||||
while rx_tiles.try_recv().is_ok() {}
|
||||
return Ok(None);
|
||||
}
|
||||
}
|
||||
if !progress(k + 1, total) {
|
||||
return Ok(None);
|
||||
}
|
||||
}
|
||||
Ok(Some(out))
|
||||
Ok(Some(()))
|
||||
})
|
||||
.map(|done| done.map(|()| out))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -157,32 +400,65 @@ mod tests {
|
||||
/// is its 2×2 quad's (R, mean G, B), averaged over the quads within
|
||||
/// `reach` quads. Purely a function of the tile, like the real one.
|
||||
struct BoxNet {
|
||||
n: usize,
|
||||
sizes: Sizes,
|
||||
reach: usize,
|
||||
/// Fails any call with more photosites than this, as a GPU out of
|
||||
/// memory does.
|
||||
fails_above: usize,
|
||||
calls: Vec<(usize, usize)>,
|
||||
}
|
||||
|
||||
fn square(n: usize, reach: usize) -> BoxNet {
|
||||
BoxNet {
|
||||
sizes: Sizes::Square(n),
|
||||
reach,
|
||||
fails_above: usize::MAX,
|
||||
calls: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
fn any(max: (usize, usize), reach: usize) -> BoxNet {
|
||||
BoxNet {
|
||||
sizes: Sizes::Any { align: 16, max },
|
||||
reach,
|
||||
fails_above: usize::MAX,
|
||||
calls: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
impl TileNet for BoxNet {
|
||||
fn tile(&self) -> usize {
|
||||
self.n
|
||||
fn sizes(&self) -> Sizes {
|
||||
self.sizes
|
||||
}
|
||||
fn run(&mut self, m: &[f32], _s: &[f32]) -> Result<Vec<f32>, crate::DenoiseError> {
|
||||
let n = self.n;
|
||||
let q = n / 2;
|
||||
fn run(
|
||||
&mut self,
|
||||
rows: usize,
|
||||
cols: usize,
|
||||
m: Vec<f32>,
|
||||
_s: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), crate::DenoiseError> {
|
||||
self.calls.push((rows, cols));
|
||||
if rows * cols > self.fails_above {
|
||||
return Err(crate::DenoiseError::Model("out of memory".into()));
|
||||
}
|
||||
let (qr, qc) = (rows / 2, cols / 2);
|
||||
let quad = |qy: usize, qx: usize| {
|
||||
let (y, x) = (2 * qy, 2 * qx);
|
||||
[
|
||||
m[y * n + x],
|
||||
0.5 * (m[y * n + x + 1] + m[(y + 1) * n + x]),
|
||||
m[(y + 1) * n + x + 1],
|
||||
m[y * cols + x],
|
||||
0.5 * (m[y * cols + x + 1] + m[(y + 1) * cols + x]),
|
||||
m[(y + 1) * cols + x + 1],
|
||||
]
|
||||
};
|
||||
let mut out = vec![0.0; 3 * n * n];
|
||||
for qy in 0..q {
|
||||
for qx in 0..q {
|
||||
let plane = rows * cols;
|
||||
let mut out = vec![0.0; 3 * plane];
|
||||
for qy in 0..qr {
|
||||
for qx in 0..qc {
|
||||
let mut acc = [0.0f32; 3];
|
||||
let mut cnt = 0.0;
|
||||
for a in qy.saturating_sub(self.reach)..(qy + self.reach + 1).min(q) {
|
||||
for b in qx.saturating_sub(self.reach)..(qx + self.reach + 1).min(q) {
|
||||
for a in qy.saturating_sub(self.reach)..(qy + self.reach + 1).min(qr) {
|
||||
for b in qx.saturating_sub(self.reach)..(qx + self.reach + 1).min(qc) {
|
||||
let v = quad(a, b);
|
||||
for c in 0..3 {
|
||||
acc[c] += v[c];
|
||||
@@ -192,12 +468,13 @@ mod tests {
|
||||
}
|
||||
for (dy, dx) in [(0, 0), (0, 1), (1, 0), (1, 1)] {
|
||||
for c in 0..3 {
|
||||
out[c * n * n + (2 * qy + dy) * n + 2 * qx + dx] = acc[c] / cnt;
|
||||
out[c * plane + (2 * qy + dy) * cols + 2 * qx + dx] = acc[c] / cnt;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(out)
|
||||
write(&out);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -221,10 +498,7 @@ mod tests {
|
||||
] {
|
||||
let (h, w) = (300, 410);
|
||||
let at = field(p);
|
||||
let mut net = BoxNet {
|
||||
n: 2 * HALO + 64,
|
||||
reach: 0,
|
||||
};
|
||||
let mut net = square(2 * HALO + 64, 0);
|
||||
let out = run_tiled(&mut net, h, w, p, &at, &|_, _| 0.01, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
@@ -251,14 +525,8 @@ mod tests {
|
||||
let (h, w) = (230, 170);
|
||||
for p in [CfaPattern::Rggb, CfaPattern::Bggr] {
|
||||
let at = |y: usize, x: usize| ((y * 7919 + x * 104729) % 1000) as f32 / 1000.0;
|
||||
let mut small = BoxNet {
|
||||
n: 2 * HALO + 32,
|
||||
reach: 20,
|
||||
};
|
||||
let mut big = BoxNet {
|
||||
n: 2 * HALO + 256,
|
||||
reach: 20,
|
||||
};
|
||||
let mut small = square(2 * HALO + 32, 20);
|
||||
let mut big = square(2 * HALO + 256, 20);
|
||||
let a = run_tiled(&mut small, h, w, p, &at, &|_, _| 0.0, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap();
|
||||
@@ -274,12 +542,114 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// The same frame through square tiles, one whole-frame call, a grid of
|
||||
/// any-size tiles, and a network that runs out of memory on the whole
|
||||
/// frame and is planned again: one answer.
|
||||
#[test]
|
||||
fn any_size_tiles_give_the_square_tiles_answer() {
|
||||
let (h, w) = (230, 170);
|
||||
for p in [
|
||||
CfaPattern::Rggb,
|
||||
CfaPattern::Grbg,
|
||||
CfaPattern::Gbrg,
|
||||
CfaPattern::Bggr,
|
||||
] {
|
||||
let at = |y: usize, x: usize| ((y * 7919 + x * 104729) % 1000) as f32 / 1000.0;
|
||||
let run = |net: &mut BoxNet| {
|
||||
run_tiled(net, h, w, p, &at, &|_, _| 0.0, &mut |_, _| true)
|
||||
.unwrap()
|
||||
.unwrap()
|
||||
};
|
||||
// A reach of 6 quads is well inside the halo, and keeps a
|
||||
// debug-build test of four phases short.
|
||||
let want = run(&mut square(2 * HALO + 32, 6));
|
||||
|
||||
let mut whole = any((4096, 4096), 6);
|
||||
let got = run(&mut whole);
|
||||
assert_eq!(whole.calls.len(), 1, "the frame fits: one call");
|
||||
assert_eq!(got, want, "{p:?}: whole frame");
|
||||
|
||||
let mut grid = any((2 * HALO + 96, 2 * HALO + 64), 6);
|
||||
let got = run(&mut grid);
|
||||
assert!(grid.calls.len() > 1);
|
||||
assert!(
|
||||
grid.calls.windows(2).all(|c| c[0] == c[1]),
|
||||
"one call shape"
|
||||
);
|
||||
assert_eq!(got, want, "{p:?}: a grid of any-size tiles");
|
||||
|
||||
let mut tight = any((4096, 4096), 6);
|
||||
tight.fails_above = (2 * HALO + 200) * (2 * HALO + 200);
|
||||
let got = run(&mut tight);
|
||||
assert_eq!(got, want, "{p:?}: planned again after a failure");
|
||||
assert!(tight.calls.len() > 2, "the whole frame failed, then tiles");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_plan_is_one_tile_when_the_frame_fits_and_the_least_work_when_not() {
|
||||
// A 6D frame with Best's halo, under the whole-frame limit: one call.
|
||||
let one = plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 6656),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(one.grid, (1, 1));
|
||||
assert_eq!((one.rows, one.cols), (4160, 5984));
|
||||
assert!(one.core.0 >= 3648 && one.core.1 >= 5472);
|
||||
// Too wide for one: the cheapest grid, every tile within the limit.
|
||||
let two = plan(
|
||||
3648,
|
||||
8192,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 6656),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert!(two.cols <= 6656 && two.rows <= 4608);
|
||||
assert_eq!(two.grid, (1, 2));
|
||||
// And always less work than today's 1408 squares.
|
||||
let squares = plan(3648, 5472, 256, Sizes::Square(1408)).unwrap();
|
||||
assert_eq!(squares.grid, (5, 7));
|
||||
assert!(one.work() * 2 < squares.work());
|
||||
// The whole-frame engine's limit on a 6 GB card: two tiles, each
|
||||
// within it, and still under half the work of the 1408 squares.
|
||||
let halves = plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (4608, 3328),
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(halves.grid, (1, 2));
|
||||
assert_eq!((halves.rows, halves.cols), (4160, 3248));
|
||||
assert!(halves.work() * 2 < squares.work());
|
||||
// A limit no tile fits under.
|
||||
assert!(plan(
|
||||
3648,
|
||||
5472,
|
||||
256,
|
||||
Sizes::Any {
|
||||
align: 16,
|
||||
max: (400, 400)
|
||||
}
|
||||
)
|
||||
.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_cancelled_run_returns_nothing() {
|
||||
let mut net = BoxNet {
|
||||
n: 2 * HALO + 32,
|
||||
reach: 0,
|
||||
};
|
||||
let mut net = square(2 * HALO + 32, 0);
|
||||
let r = run_tiled(
|
||||
&mut net,
|
||||
100,
|
||||
@@ -293,3 +663,67 @@ mod tests {
|
||||
assert!(r.is_none());
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod timing {
|
||||
use super::*;
|
||||
|
||||
/// A network that answers instantly with an output of the right size,
|
||||
/// so what is timed is the tiler alone: gathering each tile's mosaic and
|
||||
/// σ, and writing its centre back.
|
||||
struct Null(usize, Vec<f32>);
|
||||
|
||||
impl TileNet for Null {
|
||||
fn sizes(&self) -> Sizes {
|
||||
Sizes::Square(self.0)
|
||||
}
|
||||
fn run(
|
||||
&mut self,
|
||||
_rows: usize,
|
||||
_cols: usize,
|
||||
m: Vec<f32>,
|
||||
_s: Vec<f32>,
|
||||
write: &mut dyn FnMut(&[f32]),
|
||||
) -> Result<(), crate::DenoiseError> {
|
||||
// Stands for the runtime's own output buffer: allocated once.
|
||||
if self.1.len() != 3 * m.len() {
|
||||
self.1 = vec![m[0]; 3 * m.len()];
|
||||
}
|
||||
write(&self.1);
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// `cargo test --release -p dr-denoise tiler_overhead -- --ignored --nocapture`
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn tiler_overhead_on_a_6d_frame() {
|
||||
let (h, w) = (3648, 5472);
|
||||
let frame: Vec<f32> = (0..h * w).map(|i| (i % 977) as f32 / 977.0).collect();
|
||||
let at = |y: usize, x: usize| frame[y * w + x];
|
||||
let sigma = |_c: usize, v: f32| (0.001 * v + 1e-5).sqrt();
|
||||
for n in [1408usize, 2048] {
|
||||
let mut net = Null(n, Vec::new());
|
||||
let t = std::time::Instant::now();
|
||||
let mut tiles = 0;
|
||||
run_tiled(
|
||||
&mut net,
|
||||
h,
|
||||
w,
|
||||
CfaPattern::Rggb,
|
||||
&at,
|
||||
&sigma,
|
||||
&mut |_, total| {
|
||||
tiles = total;
|
||||
true
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
let s = t.elapsed().as_secs_f64();
|
||||
println!(
|
||||
"tile {n}: {tiles} tiles, tiler alone {s:.2} s ({:.0} ms a tile)",
|
||||
s / tiles as f64 * 1e3
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -137,8 +137,8 @@ impl Detection {
|
||||
/// A loaded SCRFD graph.
|
||||
pub struct Detector {
|
||||
session: Model,
|
||||
/// f32 or int8 — the int8 form finds a different set of faces and is a
|
||||
/// different detector in `model_id` (docs/dev/inference.md §7).
|
||||
/// f32 or a quantised form — which finds a different set of faces and is
|
||||
/// a different detector in `model_id` (docs/dev/inference.md §7).
|
||||
form: Form,
|
||||
/// Feature-map count: 3 for strides {8,16,32}, 4 for {8,16,32,64}.
|
||||
///
|
||||
@@ -155,7 +155,7 @@ impl Detector {
|
||||
}
|
||||
|
||||
/// Load the canonical f32 file at `path`, or the form the device's
|
||||
/// backend wants instead — the `.int8.onnx` beside it on a Hexagon —
|
||||
/// backend wants instead — the `.a16w8.onnx` beside it on a Hexagon —
|
||||
/// which [`Detector::form`] then reports.
|
||||
pub fn from_path(path: impl AsRef<std::path::Path>) -> Result<Self, FaceError> {
|
||||
let (path, form) = dr_inference_engine::resolve_model(Role::Detector, path.as_ref());
|
||||
|
||||
@@ -123,13 +123,22 @@ pub struct Landmarker {
|
||||
}
|
||||
|
||||
impl Landmarker {
|
||||
/// The graph at `path`, or the `.a16w8.onnx` sibling beside it when the
|
||||
/// device's backend runs that (the Hexagon, inference.md §1.5: 0.25 px
|
||||
/// from f32 in the 192 crop, where int8 moved the points by 1.5).
|
||||
pub fn from_path(path: impl AsRef<std::path::Path>) -> Result<Self, FaceError> {
|
||||
let (path, form) = dr_inference_engine::resolve_model(Role::Landmarks, path.as_ref());
|
||||
let bytes = std::fs::read(path).map_err(FaceError::ModelRead)?;
|
||||
Self::from_bytes(&bytes)
|
||||
Self::from_bytes_in(&bytes, form)
|
||||
}
|
||||
|
||||
pub fn from_bytes(bytes: &[u8]) -> Result<Self, FaceError> {
|
||||
let model = dr_inference_engine::open(Role::Landmarks, Form::F32, bytes)?;
|
||||
Self::from_bytes_in(bytes, Form::F32)
|
||||
}
|
||||
|
||||
/// `bytes` in a stated numeric form; the output keeps its meaning.
|
||||
pub fn from_bytes_in(bytes: &[u8], form: Form) -> Result<Self, FaceError> {
|
||||
let model = dr_inference_engine::open(Role::Landmarks, form, bytes)?;
|
||||
let acquired = model.acquire()?;
|
||||
let session = acquired.lock();
|
||||
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
//! List each RAW file's hot and dead photosite candidates (docs/dev/sensor-health.md).
|
||||
//!
|
||||
//! Reads paths on stdin and prints one JSON line per file: its capture
|
||||
//! conditions and every photosite [`Demosaicer::find_hot_pixels`] flags, as
|
||||
//! `[x, y, value, hot]` in sensor coordinates. Which candidates are defects
|
||||
//! is a question across frames, so this answers nothing on its own.
|
||||
//!
|
||||
//! ```sh
|
||||
//! find ~/Pictures -name '*.CR2' | cargo run --release -p dr-gpu --example sensor_scan
|
||||
//! ```
|
||||
|
||||
use std::io::{BufRead, Write};
|
||||
|
||||
use dr_gpu::{Demosaicer, GpuContext};
|
||||
|
||||
fn main() {
|
||||
if let Some(list) = std::env::args().skip_while(|a| a != "--probe").nth(1) {
|
||||
return probe(&list);
|
||||
}
|
||||
let ctx = pollster::block_on(GpuContext::new_headless()).expect("a GPU for the hot-pixel pass");
|
||||
let demosaicer = Demosaicer::new(&ctx).expect("demosaicer");
|
||||
for line in std::io::stdin().lock().lines() {
|
||||
let path = line.expect("stdin");
|
||||
match scan(&path, &demosaicer) {
|
||||
Ok(json) => println!("{json}"),
|
||||
Err(e) => eprintln!("fail\t{path}\t{e}"),
|
||||
}
|
||||
std::io::stdout().flush().ok();
|
||||
}
|
||||
}
|
||||
|
||||
fn scan(path: &str, demosaicer: &Demosaicer) -> Result<String, String> {
|
||||
let bytes = std::fs::read(path).map_err(|e| e.to_string())?;
|
||||
let raw = dr_decode::decode(&bytes).map_err(|e| e.to_string())?;
|
||||
let meta = dr_decode::metadata(&bytes).map_err(|e| e.to_string())?;
|
||||
let sites = demosaicer
|
||||
.find_hot_pixels(&raw)
|
||||
.map_err(|e| e.to_string())?;
|
||||
let opt = |v: Option<f64>| v.map_or("null".to_string(), |v| v.to_string());
|
||||
let list: Vec<String> = sites
|
||||
.iter()
|
||||
.map(|s| {
|
||||
let v = raw.data[(s.y * raw.width + s.x) as usize];
|
||||
format!("[{},{},{},{}]", s.x, s.y, v, u8::from(s.hot))
|
||||
})
|
||||
.collect();
|
||||
Ok(format!(
|
||||
"{{\"path\":{:?},\"model\":{:?},\"captured\":{},\"iso\":{},\"shutter\":{},\"white\":{},\"black\":{:?},\"crop\":[{},{},{},{}],\"sites\":[{}]}}",
|
||||
path,
|
||||
format!("{} {}", raw.make, raw.model),
|
||||
meta.captured_at.map_or("null".to_string(), |t| t.to_string()),
|
||||
opt(meta.iso.map(f64::from)),
|
||||
opt(meta.shutter.map(f64::from)),
|
||||
raw.white_level,
|
||||
raw.black_level,
|
||||
raw.crop.x,
|
||||
raw.crop.y,
|
||||
raw.crop.width,
|
||||
raw.crop.height,
|
||||
list.join(","),
|
||||
))
|
||||
}
|
||||
|
||||
/// `--probe COORDS`: for each path on stdin, each `x y` line of COORDS as
|
||||
/// `[value, same-colour neighbour max, median]` over black, on the CPU. A
|
||||
/// probe asks whether a photosite stood out in a frame where it would have
|
||||
/// been visible, which the scan's verdict cannot say: a frame that does not
|
||||
/// flag a defect may only have been too bright around it.
|
||||
fn probe(list: &str) {
|
||||
let coords: Vec<(u32, u32)> = std::fs::read_to_string(list)
|
||||
.expect("coords")
|
||||
.lines()
|
||||
.filter_map(|l| {
|
||||
let mut it = l.split_whitespace().map(|v| v.parse().ok());
|
||||
Some((it.next()??, it.next()??))
|
||||
})
|
||||
.collect();
|
||||
for line in std::io::stdin().lock().lines() {
|
||||
let path = line.expect("stdin");
|
||||
let Ok(bytes) = std::fs::read(&path) else {
|
||||
continue;
|
||||
};
|
||||
let (Ok(raw), Ok(meta)) = (dr_decode::decode(&bytes), dr_decode::metadata(&bytes)) else {
|
||||
continue;
|
||||
};
|
||||
let w = raw.width as i64;
|
||||
let at = |x: i64, y: i64| {
|
||||
let cell = (((y - raw.crop.y as i64) & 1) * 2 + ((x - raw.crop.x as i64) & 1)) as usize;
|
||||
raw.data[(y * w + x) as usize].saturating_sub(raw.black_level[cell])
|
||||
};
|
||||
let rows: Vec<String> = coords
|
||||
.iter()
|
||||
.map(|&(x, y)| {
|
||||
let (x, y) = (x as i64, y as i64);
|
||||
let mut n: Vec<u16> = Vec::new();
|
||||
for dy in [-2i64, 0, 2] {
|
||||
for dx in [-2i64, 0, 2] {
|
||||
if (dx, dy) != (0, 0) {
|
||||
n.push(at(x + dx, y + dy));
|
||||
}
|
||||
}
|
||||
}
|
||||
n.sort_unstable();
|
||||
format!("[{},{},{}]", at(x, y), n[n.len() - 1], n[n.len() / 2])
|
||||
})
|
||||
.collect();
|
||||
println!(
|
||||
"{{\"path\":{:?},\"captured\":{},\"iso\":{},\"shutter\":{},\"range\":{},\"p\":[{}]}}",
|
||||
path,
|
||||
meta.captured_at.unwrap_or(0),
|
||||
meta.iso.unwrap_or(0),
|
||||
meta.shutter.unwrap_or(0.0),
|
||||
raw.white_level - raw.black_level[0],
|
||||
rows.join(","),
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1071,6 +1071,46 @@ impl Demosaicer {
|
||||
if raw.samples_per_pixel != 1 {
|
||||
return Ok(0);
|
||||
}
|
||||
let words = self.hot_pixel_words(raw)?;
|
||||
let mut changed = 0;
|
||||
for (i, v) in raw.data.iter_mut().enumerate() {
|
||||
let new = unpack_sample(&words, i);
|
||||
changed += usize::from(new != *v);
|
||||
*v = new;
|
||||
}
|
||||
Ok(changed)
|
||||
}
|
||||
|
||||
/// The photosites [`Self::repair_hot_pixels`] would replace, in sensor
|
||||
/// coordinates, without replacing them.
|
||||
///
|
||||
/// For the sensor health record (docs/dev/sensor-health.md): one frame's
|
||||
/// verdict is a candidate list, not a defect map — a single photosite of a
|
||||
/// star that passes both tests reads the same as a hot one. Which of them
|
||||
/// is the sensor is decided across frames, by who keeps coming back.
|
||||
pub fn find_hot_pixels(&self, raw: &RawImage) -> Result<Vec<Photosite>, GpuError> {
|
||||
if raw.samples_per_pixel != 1 {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let words = self.hot_pixel_words(raw)?;
|
||||
let stride = raw.width.max(1);
|
||||
Ok(raw
|
||||
.data
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(i, &v)| {
|
||||
let new = unpack_sample(&words, i);
|
||||
(new != v).then(|| Photosite {
|
||||
x: i as u32 % stride,
|
||||
y: i as u32 / stride,
|
||||
hot: new < v,
|
||||
})
|
||||
})
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// The hot-pixel pass over `raw`, read back as packed words.
|
||||
fn hot_pixel_words(&self, raw: &RawImage) -> Result<Vec<u32>, GpuError> {
|
||||
let (width, height) = (raw.crop.width.max(1), raw.crop.height.max(1));
|
||||
let xtrans_tile = raw
|
||||
.cfa_pattern
|
||||
@@ -1122,18 +1162,30 @@ impl Demosaicer {
|
||||
.map_err(|e| GpuError::Readback(e.to_string()))?;
|
||||
let words: Vec<u32> = bytemuck::cast_slice(&slice.get_mapped_range()).to_vec();
|
||||
readback.unmap();
|
||||
|
||||
let mut changed = 0;
|
||||
for (i, v) in raw.data.iter_mut().enumerate() {
|
||||
let w = words[i / 2];
|
||||
let new = if i % 2 == 0 { w & 0xFFFF } else { w >> 16 } as u16;
|
||||
changed += usize::from(new != *v);
|
||||
*v = new;
|
||||
}
|
||||
Ok(changed)
|
||||
Ok(words)
|
||||
}
|
||||
}
|
||||
|
||||
/// One photosite the hot-pixel pass judged defective.
|
||||
#[derive(Copy, Clone, Debug, PartialEq, Eq, Hash)]
|
||||
pub struct Photosite {
|
||||
/// Sensor coordinates: the full readout, masked border included.
|
||||
pub x: u32,
|
||||
pub y: u32,
|
||||
/// Read far above its neighbourhood; otherwise far below (dead).
|
||||
pub hot: bool,
|
||||
}
|
||||
|
||||
/// Sample `i` of a readout packed by [`pack_samples`].
|
||||
fn unpack_sample(words: &[u32], i: usize) -> u16 {
|
||||
let w = words[i / 2];
|
||||
(if i.is_multiple_of(2) {
|
||||
w & 0xFFFF
|
||||
} else {
|
||||
w >> 16
|
||||
}) as u16
|
||||
}
|
||||
|
||||
const IDENTITY_3X3: [f32; 9] = [1.0, 0.0, 0.0, 0.0, 1.0, 0.0, 0.0, 0.0, 1.0];
|
||||
|
||||
/// Pack u16 samples two per u32, little-endian within the word.
|
||||
|
||||
@@ -38,7 +38,7 @@ pub use adjust::AdjustPass;
|
||||
// rather than an implementation detail: a detail pass is guaranteed linear,
|
||||
// unclipped, full internal precision (FR-DEV-2), and anyone reasoning about
|
||||
// VRAM at 24 MP needs to know what an intermediate costs.
|
||||
pub use demosaic::{DemosaicedImage, Demosaicer};
|
||||
pub use demosaic::{DemosaicedImage, Demosaicer, Photosite};
|
||||
pub use detail::INTERMEDIATE_FORMAT as DETAIL_INTERMEDIATE_FORMAT;
|
||||
pub use error::GpuError;
|
||||
pub use focus::{FocusPeakPass, FocusPeaking, PeakColour, PeakSensitivity};
|
||||
|
||||
@@ -18,7 +18,7 @@ use dr_decode::{CfaPattern, CropRect, RawImage};
|
||||
use dr_gpu::{AdjustPass, Demosaicer, GpuContext};
|
||||
use dr_pipeline::descriptor::{Attribute, LocalizedKey, OpDescriptor, OpId, ParamId};
|
||||
use dr_pipeline::operation::{Operation, Stage, Uniform};
|
||||
use dr_pipeline::ops::camera_profile::{apply_reference, CameraProfile, APPLY, LOOK};
|
||||
use dr_pipeline::ops::camera_profile::{apply_reference, CameraProfile, APPLY, LOOK, PROFILE_LOOK};
|
||||
use dr_types::{HueSatTable, ProfileOrigin, ProfileTables, Transfer};
|
||||
|
||||
const SIZE: u32 = 16;
|
||||
@@ -151,6 +151,14 @@ fn render(ctx: &GpuContext, raw: &RawImage, op: CameraProfile) -> Vec<[u8; 3]> {
|
||||
pixels.chunks_exact(4).map(|p| [p[0], p[1], p[2]]).collect()
|
||||
}
|
||||
|
||||
/// The profile at the strength it states — the look table on, as the
|
||||
/// reference applies it at 1.0. Not the default, which leaves it off (D21).
|
||||
fn as_stated() -> CameraProfile {
|
||||
let mut op = CameraProfile::new();
|
||||
op.set_param(LOOK, PROFILE_LOOK);
|
||||
op
|
||||
}
|
||||
|
||||
fn encode(c: [f32; 3]) -> [i32; 3] {
|
||||
c.map(|v| (Transfer::Srgb.encode(v.clamp(0.0, 1.0)) * 255.0).round() as i32)
|
||||
}
|
||||
@@ -183,8 +191,12 @@ fn the_shader_agrees_with_the_cpu_reference() {
|
||||
return;
|
||||
};
|
||||
let tables = strong_tables();
|
||||
let got = render(&ctx, &frame(Some(tables.clone())), CameraProfile::new());
|
||||
assert_agrees(&got, |c| apply_reference(&tables, c, 1.0), "at defaults");
|
||||
let got = render(&ctx, &frame(Some(tables.clone())), as_stated());
|
||||
assert_agrees(
|
||||
&got,
|
||||
|c| apply_reference(&tables, c, 1.0),
|
||||
"as the profile states it",
|
||||
);
|
||||
|
||||
let mut doubled = CameraProfile::new();
|
||||
doubled.set_param(LOOK, 200.0);
|
||||
@@ -279,7 +291,7 @@ fn the_libraries_adobe_standard_renders_as_the_reference_does() {
|
||||
let tables = dr_decode::dcp::embedded_in(&bytes)
|
||||
.expect("Adobe Standard")
|
||||
.tables(5000.0, ProfileOrigin::Embedded);
|
||||
let got = render(&ctx, &frame(Some(tables.clone())), CameraProfile::new());
|
||||
let got = render(&ctx, &frame(Some(tables.clone())), as_stated());
|
||||
for (i, (c, g)) in colours().into_iter().zip(&got).enumerate() {
|
||||
let want = encode(apply_reference(&tables, c, 1.0));
|
||||
let g = g.map(i32::from);
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
//! see of a defect it missed is the coloured cross the demosaic makes of it.
|
||||
|
||||
use dr_decode::{CfaPattern, CropRect, RawImage};
|
||||
use dr_gpu::{AdjustPass, Demosaicer, GpuContext};
|
||||
use dr_gpu::{AdjustPass, Demosaicer, GpuContext, Photosite};
|
||||
use dr_pipeline::EditGraph;
|
||||
|
||||
const SIZE: u32 = 36;
|
||||
@@ -171,3 +171,43 @@ fn the_repaired_mosaic_reads_back_with_only_the_defect_changed() {
|
||||
let others = (0..raw.data.len()).filter(|&i| i != at);
|
||||
assert!(others.into_iter().all(|i| raw.data[i] == before.data[i]));
|
||||
}
|
||||
|
||||
/// Finding without repairing (docs/dev/sensor-health.md): the same verdict as
|
||||
/// the repair, as sensor coordinates, with the frame left as it was. The
|
||||
/// sensor health record builds on this, so it must name exactly the
|
||||
/// photosites the repair would change — the hot one and the dead one, and
|
||||
/// not the star.
|
||||
#[test]
|
||||
fn finding_names_what_the_repair_would_change_and_changes_nothing() {
|
||||
let Some(ctx) = ctx() else {
|
||||
eprintln!("no GPU adapter; skipping");
|
||||
return;
|
||||
};
|
||||
let d = Demosaicer::new(&ctx).expect("demosaicer");
|
||||
let mut set = vec![(MIDDLE, MIDDLE, WHITE), (9, 25, 0)];
|
||||
for dy in 0..3 {
|
||||
for dx in 0..3 {
|
||||
set.push((4 + dx, 4 + dy, WHITE));
|
||||
}
|
||||
}
|
||||
let raw = frame(CfaPattern::Rggb, 1600, &set);
|
||||
let mut found = d.find_hot_pixels(&raw).expect("find");
|
||||
found.sort_by_key(|p| (p.y, p.x));
|
||||
assert_eq!(
|
||||
found,
|
||||
vec![
|
||||
Photosite {
|
||||
x: MIDDLE,
|
||||
y: MIDDLE,
|
||||
hot: true
|
||||
},
|
||||
Photosite {
|
||||
x: 9,
|
||||
y: 25,
|
||||
hot: false
|
||||
},
|
||||
]
|
||||
);
|
||||
let mut repaired = raw.clone();
|
||||
assert_eq!(d.repair_hot_pixels(&mut repaired).expect("repair"), 2);
|
||||
}
|
||||
|
||||
@@ -24,6 +24,14 @@ enum Ep {
|
||||
Cpu,
|
||||
MiGraphX,
|
||||
MiGraphXFp16,
|
||||
OpenVinoCpu,
|
||||
OpenVinoGpu,
|
||||
OpenVinoGpuFp16,
|
||||
OpenVinoNpu,
|
||||
/// Dawn's low-power adapter: the integrated GPU on a hybrid machine.
|
||||
WebGpuLow,
|
||||
/// Dawn's high-performance adapter: the discrete one, if there is one.
|
||||
WebGpuHigh,
|
||||
}
|
||||
|
||||
impl Ep {
|
||||
@@ -32,20 +40,113 @@ impl Ep {
|
||||
Ep::Cpu => "CPU",
|
||||
Ep::MiGraphX => "MIGraphX f32",
|
||||
Ep::MiGraphXFp16 => "MIGraphX fp16",
|
||||
Ep::OpenVinoCpu => "OpenVINO CPU",
|
||||
Ep::OpenVinoGpu => "OpenVINO GPU",
|
||||
Ep::OpenVinoGpuFp16 => "OpenVINO GPU16",
|
||||
Ep::OpenVinoNpu => "OpenVINO NPU",
|
||||
Ep::WebGpuLow => "WebGPU low",
|
||||
Ep::WebGpuHigh => "WebGPU high",
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a second build reads what the first one compiled.
|
||||
fn caches(self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Ep::MiGraphX
|
||||
| Ep::MiGraphXFp16
|
||||
| Ep::OpenVinoGpu
|
||||
| Ep::OpenVinoGpuFp16
|
||||
| Ep::OpenVinoNpu
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
fn build(ep: Ep, bytes: &[u8], threads: usize, cache: &Path) -> ort::Result<ort::session::Session> {
|
||||
let mut b = ort::session::Session::builder()?.with_intra_threads(threads)?;
|
||||
let dir = |sub: &str| {
|
||||
let d = cache.join(sub);
|
||||
let _ = std::fs::create_dir_all(&d);
|
||||
d.to_string_lossy().into_owned()
|
||||
};
|
||||
// `GPU` is OpenVINO's first OpenCL GPU, which on a hybrid laptop can be
|
||||
// the discrete NVIDIA one; DARKROOM_OV_GPU=GPU.1 names another.
|
||||
let gpu = std::env::var("DARKROOM_OV_GPU").unwrap_or_else(|_| "GPU".into());
|
||||
match ep {
|
||||
Ep::Cpu => {}
|
||||
Ep::MiGraphX => migraphx(&mut b, false, &cache.join("f32"))?,
|
||||
Ep::MiGraphXFp16 => migraphx(&mut b, true, &cache.join("fp16"))?,
|
||||
// Option names as `openvino_provider_factory.cc` reads them at 1.24.
|
||||
Ep::OpenVinoCpu => append(&mut b, c"OpenVINO", &[("device_type", "CPU".into())])?,
|
||||
Ep::OpenVinoGpu => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[
|
||||
("device_type", gpu.clone()),
|
||||
("precision", "FP32".into()),
|
||||
("cache_dir", dir("ov-gpu-f32")),
|
||||
],
|
||||
)?,
|
||||
Ep::OpenVinoGpuFp16 => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[
|
||||
("device_type", gpu.clone()),
|
||||
("precision", "FP16".into()),
|
||||
("cache_dir", dir("ov-gpu-fp16")),
|
||||
],
|
||||
)?,
|
||||
Ep::OpenVinoNpu => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[("device_type", "NPU".into()), ("cache_dir", dir("ov-npu"))],
|
||||
)?,
|
||||
// `webgpu_provider_options.h` at 1.27; the runtime prefixes the key.
|
||||
Ep::WebGpuLow => append(
|
||||
&mut b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "low-power".into())],
|
||||
)?,
|
||||
Ep::WebGpuHigh => append(
|
||||
&mut b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "high-performance".into())],
|
||||
)?,
|
||||
}
|
||||
b.commit_from_memory(bytes)
|
||||
}
|
||||
|
||||
/// Any provider through the generic key/value entry point.
|
||||
fn append(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
name: &std::ffi::CStr,
|
||||
options: &[(&str, String)],
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let keys: Vec<CString> = options
|
||||
.iter()
|
||||
.map(|(k, _)| CString::new(*k).unwrap())
|
||||
.collect();
|
||||
let values: Vec<CString> = options
|
||||
.iter()
|
||||
.map(|(_, v)| CString::new(v.as_bytes()).unwrap())
|
||||
.collect();
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
// SAFETY: as `migraphx` below.
|
||||
unsafe {
|
||||
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
|
||||
b.ptr_mut(),
|
||||
name.as_ptr(),
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
);
|
||||
ort::Error::result_from_status(status)
|
||||
}
|
||||
}
|
||||
|
||||
/// Register MIGraphX through the generic key/value API. `ort`'s own
|
||||
/// builder fills the legacy `OrtMIGraphXProviderOptions`, which 1.29 reads
|
||||
/// for its precision flags and nothing else: the model cache directory —
|
||||
@@ -83,19 +184,29 @@ fn migraphx(
|
||||
|
||||
/// Median of `runs` timed runs over zeros, in milliseconds, after warm-ups.
|
||||
fn time(session: &mut ort::session::Session, warmups: usize, runs: usize) -> Result<f64, String> {
|
||||
let shape: Vec<usize> = session.inputs()[0]
|
||||
.dtype()
|
||||
.tensor_shape()
|
||||
.ok_or("input is not a tensor")?
|
||||
.iter()
|
||||
.map(|&d| if d > 0 { d as usize } else { 1 })
|
||||
.collect();
|
||||
let zeros = vec![0f32; shape.iter().product()];
|
||||
// Zeros for every input, not just the first: the denoiser takes
|
||||
// `mosaic` and `sigma`. A dynamic dimension is read as 1.
|
||||
let mut inputs = Vec::new();
|
||||
for input in session.inputs() {
|
||||
let shape: Vec<usize> = input
|
||||
.dtype()
|
||||
.tensor_shape()
|
||||
.ok_or("input is not a tensor")?
|
||||
.iter()
|
||||
.map(|&d| if d > 0 { d as usize } else { 1 })
|
||||
.collect();
|
||||
let zeros = vec![0f32; shape.iter().product()];
|
||||
inputs.push((input.name().to_string(), shape, zeros));
|
||||
}
|
||||
let once = |s: &mut ort::session::Session| -> Result<f64, String> {
|
||||
let input = ort::value::Tensor::from_array((shape.clone(), zeros.clone()))
|
||||
.map_err(|e| e.to_string())?;
|
||||
let mut values = Vec::with_capacity(inputs.len());
|
||||
for (name, shape, zeros) in &inputs {
|
||||
let value = ort::value::Tensor::from_array((shape.clone(), zeros.clone()))
|
||||
.map_err(|e| e.to_string())?;
|
||||
values.push((name.clone(), ort::session::SessionInputValue::from(value)));
|
||||
}
|
||||
let t = Instant::now();
|
||||
let out = s.run(ort::inputs![input]).map_err(|e| e.to_string())?;
|
||||
let out = s.run(values).map_err(|e| e.to_string())?;
|
||||
let _ = out[0]
|
||||
.try_extract_tensor::<f32>()
|
||||
.map_err(|e| e.to_string())?;
|
||||
@@ -155,13 +266,35 @@ fn main() {
|
||||
// A compiling provider is built twice: the second build reads the
|
||||
// program the first wrote, and its time is what a launch after the
|
||||
// first costs.
|
||||
let plan = [
|
||||
(Ep::Cpu, false),
|
||||
(Ep::MiGraphX, false),
|
||||
(Ep::MiGraphX, true),
|
||||
(Ep::MiGraphXFp16, false),
|
||||
(Ep::MiGraphXFp16, true),
|
||||
];
|
||||
// DARKROOM_EPS narrows the list (`cpu,openvino,webgpu,migraphx`);
|
||||
// a runtime without a provider fails its build in a millisecond
|
||||
// anyway, so the default is all of them.
|
||||
let wanted = std::env::var("DARKROOM_EPS").unwrap_or_default();
|
||||
let on = |family: &str| wanted.is_empty() || wanted.split(',').any(|w| w == family);
|
||||
let mut plan = Vec::new();
|
||||
for (family, eps) in [
|
||||
("cpu", &[Ep::Cpu][..]),
|
||||
("migraphx", &[Ep::MiGraphX, Ep::MiGraphXFp16][..]),
|
||||
(
|
||||
"openvino",
|
||||
&[
|
||||
Ep::OpenVinoCpu,
|
||||
Ep::OpenVinoGpu,
|
||||
Ep::OpenVinoGpuFp16,
|
||||
Ep::OpenVinoNpu,
|
||||
][..],
|
||||
),
|
||||
("webgpu", &[Ep::WebGpuLow, Ep::WebGpuHigh][..]),
|
||||
] {
|
||||
if on(family) {
|
||||
for &ep in eps {
|
||||
plan.push((ep, false));
|
||||
if ep.caches() {
|
||||
plan.push((ep, true));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (ep, cached) in plan {
|
||||
let started = Instant::now();
|
||||
match build(ep, &bytes, threads, &cache) {
|
||||
|
||||
@@ -4,12 +4,17 @@
|
||||
//!
|
||||
//! DARKROOM_ORT_DIR=/usr/lib \
|
||||
//! cargo run --release -p dr-inference-engine --features native,tract \
|
||||
//! --example ladder -- CACHE_DIR models/face/scrfd_500m_640.onnx [MODEL.onnx ...]
|
||||
//! --example ladder -- CACHE_DIR models/face/scrfd_500m_640.onnx [ROLE=MODEL.onnx ...]
|
||||
//!
|
||||
//! Every model named is a `Detector` for the config's purposes, which is
|
||||
//! enough to see the rung taken, the engines compiled and a session land
|
||||
//! on it. Delete `CACHE_DIR` to see the first run again; keep it to see the
|
||||
//! second.
|
||||
//! `DARKROOM_ORT_DIRS=a:b:c` offers several runtimes, as the app's search
|
||||
//! list does, and shows which the engine chose for this device's GPU.
|
||||
//!
|
||||
//! A bare path is a `Detector`; `denoiser=…`, `scene=…`, `inpainter=…`,
|
||||
//! `landmarks=…` (any `Role`, lower case) says otherwise, so a device can
|
||||
//! show each role taking its own form (inference.md §1.5). Each is opened
|
||||
//! through `resolve_model`, as the app opens it, and the line says which
|
||||
//! form and which rung it landed on. Delete `CACHE_DIR` to see the first
|
||||
//! run again; keep it to see the second.
|
||||
|
||||
use std::path::PathBuf;
|
||||
use std::time::{Duration, Instant};
|
||||
@@ -17,7 +22,10 @@ use std::time::{Duration, Instant};
|
||||
fn main() {
|
||||
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
|
||||
let mut args = std::env::args_os().skip(1).map(PathBuf::from);
|
||||
let (Some(cache_dir), models) = (args.next(), args.collect::<Vec<_>>()) else {
|
||||
let (Some(cache_dir), models) = (
|
||||
args.next(),
|
||||
args.map(|a| role_and_path(&a)).collect::<Vec<_>>(),
|
||||
) else {
|
||||
eprintln!("usage: ladder CACHE_DIR MODEL.onnx [MODEL.onnx ...]");
|
||||
std::process::exit(2);
|
||||
};
|
||||
@@ -26,18 +34,22 @@ fn main() {
|
||||
std::process::exit(2);
|
||||
}
|
||||
|
||||
// DARKROOM_ORT_DIRS lists several, colon-separated, as the app's search
|
||||
// does: the engine loads the one that fits the GPU (§3.2).
|
||||
let runtime_dirs: Vec<PathBuf> = std::env::var_os("DARKROOM_ORT_DIR")
|
||||
.map(PathBuf::from)
|
||||
.into_iter()
|
||||
.chain(
|
||||
std::env::var_os("DARKROOM_ORT_DIRS")
|
||||
.map(|v| std::env::split_paths(&v).collect::<Vec<_>>())
|
||||
.unwrap_or_default(),
|
||||
)
|
||||
.collect();
|
||||
let started = Instant::now();
|
||||
dr_inference_engine::init(dr_inference_engine::Config {
|
||||
runtime_dirs,
|
||||
cache_dir: cache_dir.clone(),
|
||||
models: models
|
||||
.iter()
|
||||
.map(|p| (dr_inference_engine::Role::Detector, p.clone()))
|
||||
.collect(),
|
||||
models: models.clone(),
|
||||
embedded: Vec::new(),
|
||||
ceiling: None,
|
||||
threads: 0,
|
||||
@@ -80,21 +92,39 @@ fn main() {
|
||||
std::thread::sleep(Duration::from_millis(500));
|
||||
}
|
||||
|
||||
for path in &models {
|
||||
let bytes = std::fs::read(path).expect("read model");
|
||||
for (role, path) in &models {
|
||||
let (path, form) = dr_inference_engine::resolve_model(*role, path);
|
||||
let bytes = std::fs::read(&path).expect("read model");
|
||||
let t = Instant::now();
|
||||
let model = dr_inference_engine::open(
|
||||
dr_inference_engine::Role::Detector,
|
||||
dr_inference_engine::Form::F32,
|
||||
&bytes,
|
||||
)
|
||||
.expect("open model");
|
||||
let model = dr_inference_engine::open(*role, form, &bytes).expect("open model");
|
||||
let acquired = model.acquire().expect("acquire session");
|
||||
println!(
|
||||
"{} on {} in {:.2} s",
|
||||
"{role:?}: {} ({form:?}) on {} in {:.2} s",
|
||||
path.file_name().unwrap().to_string_lossy(),
|
||||
acquired.rung().label(),
|
||||
t.elapsed().as_secs_f64()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// `denoiser=path` → (Denoiser, path); a bare path is a detector.
|
||||
fn role_and_path(arg: &std::path::Path) -> (dr_inference_engine::Role, PathBuf) {
|
||||
use dr_inference_engine::Role::*;
|
||||
let s = arg.to_string_lossy();
|
||||
let Some((name, path)) = s.split_once('=') else {
|
||||
return (Detector, arg.to_path_buf());
|
||||
};
|
||||
let role = match name {
|
||||
"detector" => Detector,
|
||||
"embedder" => Embedder,
|
||||
"segmenter" => Segmenter,
|
||||
"scene" => Scene,
|
||||
"landmarks" => Landmarks,
|
||||
"eyes" => EyeClassifier,
|
||||
"keypoints" => Keypoints,
|
||||
"inpainter" => Inpainter,
|
||||
"denoiser" => Denoiser,
|
||||
other => panic!("no role {other:?}"),
|
||||
};
|
||||
(role, PathBuf::from(path))
|
||||
}
|
||||
|
||||
@@ -51,17 +51,25 @@ pub fn ensure_installed() {
|
||||
}
|
||||
}
|
||||
|
||||
/// Look for `libonnxruntime` in `dirs`, in order, and hand `ort` the first
|
||||
/// table that loads; otherwise tract. Once per process.
|
||||
/// Find every `libonnxruntime` in `dirs`, hand `ort` the table of the one
|
||||
/// that best fits this device's GPUs, and fall to tract if none loads.
|
||||
/// Once per process.
|
||||
///
|
||||
/// Best fit, not first found (§3.2): a device can hold several runtimes —
|
||||
/// the package's OpenVINO build, a CUDA build the user fetched, the
|
||||
/// distribution's ROCm build — and each carries one vendor's providers.
|
||||
/// Between equals, the earlier directory wins, as it always has, and a
|
||||
/// runtime that fits perfectly ends the search: the APK's QNN build on a
|
||||
/// Qualcomm tablet is found first, and the generic build beside it is
|
||||
/// never opened there.
|
||||
/// `DARKROOM_ORT_DIR`, when it loads, wins outright: it is how a person
|
||||
/// says which runtime they mean.
|
||||
pub fn install(dirs: &[PathBuf]) -> Runtime {
|
||||
RUNTIME
|
||||
.get_or_init(|| {
|
||||
#[cfg(feature = "native")]
|
||||
for dir in dirs {
|
||||
match load_native(dir) {
|
||||
Ok(rt) => return rt,
|
||||
Err(e) => log::info!("inference: no runtime in {}: {e}", dir.display()),
|
||||
}
|
||||
if let Some(rt) = install_best(dirs) {
|
||||
return rt;
|
||||
}
|
||||
#[cfg(not(feature = "native"))]
|
||||
let _ = dirs;
|
||||
@@ -70,6 +78,103 @@ pub fn install(dirs: &[PathBuf]) -> Runtime {
|
||||
.clone()
|
||||
}
|
||||
|
||||
/// A runtime opened to read its providers, not yet handed to `ort`.
|
||||
#[cfg(feature = "native")]
|
||||
struct Found {
|
||||
lib: libloading::Library,
|
||||
api: *const ort_sys::OrtApi,
|
||||
path: PathBuf,
|
||||
version: String,
|
||||
providers: Vec<String>,
|
||||
}
|
||||
|
||||
#[cfg(feature = "native")]
|
||||
fn install_best(dirs: &[PathBuf]) -> Option<Runtime> {
|
||||
let named = std::env::var_os("DARKROOM_ORT_DIR").map(PathBuf::from);
|
||||
let gpus = crate::hardware::detect();
|
||||
let mut found: Vec<Found> = Vec::new();
|
||||
let mut seen = std::collections::HashSet::new();
|
||||
for dir in dirs {
|
||||
match open_native(dir) {
|
||||
Ok(f) => {
|
||||
// `bin/../lib/darkroom` and `/usr/lib/darkroom` are one file.
|
||||
if !seen.insert(std::fs::canonicalize(&f.path).unwrap_or(f.path.clone())) {
|
||||
std::mem::forget(f.lib);
|
||||
continue;
|
||||
}
|
||||
log::info!(
|
||||
"inference: ONNX Runtime {} at {} offers {}",
|
||||
f.version,
|
||||
f.path.display(),
|
||||
f.providers.join(", ")
|
||||
);
|
||||
if named.as_deref() == Some(dir.as_path()) {
|
||||
found.clear();
|
||||
found.push(f);
|
||||
break;
|
||||
}
|
||||
let perfect = gpus.score(&f.providers) >= crate::hardware::PERFECT;
|
||||
found.push(f);
|
||||
if perfect {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(e) => log::info!("inference: no runtime in {}: {e}", dir.display()),
|
||||
}
|
||||
}
|
||||
let best = (0..found.len())
|
||||
.max_by_key(|&i| (gpus.score(&found[i].providers), std::cmp::Reverse(i)))?;
|
||||
let chosen = found.swap_remove(best);
|
||||
// The others stay mapped. Unloading a C++ runtime after its static
|
||||
// constructors ran is a crash at exit waiting to happen, and an
|
||||
// unused mapping costs address space, not memory.
|
||||
for other in found {
|
||||
std::mem::forget(other.lib);
|
||||
}
|
||||
log::info!("inference: chose {} for {gpus:?}", chosen.path.display());
|
||||
|
||||
// SAFETY: the table came from this library's `OrtGetApiBase`, and the
|
||||
// library is leaked below, so every pointer in the copy stays valid for
|
||||
// the life of the process.
|
||||
if !ort::set_api(unsafe { (*chosen.api).clone() }) {
|
||||
log::warn!("inference: an API table was already installed");
|
||||
std::mem::forget(chosen.lib);
|
||||
return None;
|
||||
}
|
||||
std::mem::forget(chosen.lib);
|
||||
|
||||
// Qualcomm's DSP loader finds the Hexagon skel through this variable,
|
||||
// and only through it; the runtime's own directory is where the APK
|
||||
// put it. Harmless anywhere else.
|
||||
#[cfg(target_os = "android")]
|
||||
if let Some(dir) = chosen.path.parent().filter(|d| !d.as_os_str().is_empty()) {
|
||||
std::env::set_var("ADSP_LIBRARY_PATH", dir);
|
||||
}
|
||||
|
||||
// Windows looks for a provider's own dependencies — OpenVINO's DLLs,
|
||||
// which Intel's build leaves beside it — on the DLL search path, not in
|
||||
// the provider's directory. Intel's Python shim prepends to `PATH` for
|
||||
// the same reason; so does this, before any provider loads.
|
||||
#[cfg(target_os = "windows")]
|
||||
if let Some(dir) = chosen.path.parent() {
|
||||
let old = std::env::var_os("PATH").unwrap_or_default();
|
||||
let dirs = std::iter::once(dir.to_path_buf()).chain(std::env::split_paths(&old));
|
||||
if let Ok(path) = std::env::join_paths(dirs) {
|
||||
std::env::set_var("PATH", path);
|
||||
}
|
||||
}
|
||||
|
||||
log::info!(
|
||||
"inference: ONNX Runtime {} from {}",
|
||||
chosen.version,
|
||||
chosen.path.display()
|
||||
);
|
||||
Some(Runtime::OnnxRuntime {
|
||||
path: chosen.path,
|
||||
version: chosen.version,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(feature = "tract")]
|
||||
fn install_tract() -> Runtime {
|
||||
let _ = ort::set_api(ort_tract::api());
|
||||
@@ -85,8 +190,12 @@ fn install_tract() -> Runtime {
|
||||
Runtime::Tract
|
||||
}
|
||||
|
||||
/// Open the runtime in `dir` and read what it offers. `dir` may also name
|
||||
/// the library itself — Android has two runtimes and one directory, so the
|
||||
/// second goes by its file name — and an empty path is the bare name
|
||||
/// through the system loader, which on Android is the APK's own copy.
|
||||
#[cfg(feature = "native")]
|
||||
fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
fn open_native(dir: &std::path::Path) -> Result<Found, String> {
|
||||
let name = if cfg!(target_os = "windows") {
|
||||
"onnxruntime.dll"
|
||||
} else if cfg!(any(target_os = "macos", target_os = "ios")) {
|
||||
@@ -94,18 +203,21 @@ fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
} else {
|
||||
"libonnxruntime.so"
|
||||
};
|
||||
// An empty dir means the bare name: the system loader's search, which on
|
||||
// Android includes the APK's own native libraries.
|
||||
let is_library = dir.file_name().and_then(|n| n.to_str()).is_some_and(|n| {
|
||||
n.contains("onnxruntime")
|
||||
&& (n.ends_with(".so") || n.ends_with(".dll") || n.ends_with(".dylib"))
|
||||
});
|
||||
let path = if dir.as_os_str().is_empty() {
|
||||
PathBuf::from(name)
|
||||
} else if is_library {
|
||||
dir.to_path_buf()
|
||||
} else {
|
||||
find_library(dir, name).ok_or("not present")?
|
||||
};
|
||||
|
||||
// SAFETY: the library's initialisers are ONNX Runtime's own; the symbol
|
||||
// is the documented entry point with the documented signature; the table
|
||||
// is copied out and the library handle is leaked, so every pointer in
|
||||
// the copy stays valid for the life of the process.
|
||||
// is the documented entry point with the documented signature. The
|
||||
// table pointer is valid while `lib` is, which the caller keeps.
|
||||
unsafe {
|
||||
let lib = libloading::Library::new(&path).map_err(|e| e.to_string())?;
|
||||
let get_base: libloading::Symbol<
|
||||
@@ -125,24 +237,45 @@ fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
ort_sys::ORT_API_VERSION
|
||||
));
|
||||
}
|
||||
if !ort::set_api((*api).clone()) {
|
||||
return Err("an API table was already installed".into());
|
||||
}
|
||||
std::mem::forget(lib);
|
||||
|
||||
// Qualcomm's DSP loader finds the Hexagon skel through this variable,
|
||||
// and only through it; the runtime's own directory is where the APK
|
||||
// put it. Harmless anywhere else.
|
||||
#[cfg(target_os = "android")]
|
||||
if !dir.as_os_str().is_empty() {
|
||||
std::env::set_var("ADSP_LIBRARY_PATH", dir);
|
||||
}
|
||||
|
||||
log::info!("inference: ONNX Runtime {version} from {}", path.display());
|
||||
Ok(Runtime::OnnxRuntime { path, version })
|
||||
let providers = available_providers(api);
|
||||
Ok(Found {
|
||||
lib,
|
||||
api,
|
||||
path,
|
||||
version,
|
||||
providers,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// The providers compiled into the runtime behind `api` — not the ones this
|
||||
/// device can run, which is the probe's question.
|
||||
///
|
||||
/// # Safety
|
||||
/// `api` must be a live table from `GetApi`.
|
||||
#[cfg(feature = "native")]
|
||||
unsafe fn available_providers(api: *const ort_sys::OrtApi) -> Vec<String> {
|
||||
let mut list: *mut *mut std::ffi::c_char = std::ptr::null_mut();
|
||||
let mut n: std::ffi::c_int = 0;
|
||||
let status = ((*api).GetAvailableProviders)(&mut list, &mut n);
|
||||
if !status.0.is_null() {
|
||||
((*api).ReleaseStatus)(status.0);
|
||||
return Vec::new();
|
||||
}
|
||||
let names = (0..n.max(0) as usize)
|
||||
.map(|i| {
|
||||
std::ffi::CStr::from_ptr(*list.add(i))
|
||||
.to_string_lossy()
|
||||
.into_owned()
|
||||
})
|
||||
.collect();
|
||||
let status = ((*api).ReleaseAvailableProviders)(list, n);
|
||||
if !status.0.is_null() {
|
||||
((*api).ReleaseStatus)(status.0);
|
||||
}
|
||||
names
|
||||
}
|
||||
|
||||
/// `libonnxruntime.so` in `dir`, or a versioned spelling of it —
|
||||
/// `libonnxruntime.so.1.30.0` is what the Python wheel ships, and a package
|
||||
/// that installs only the versioned file is not wrong.
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
|
||||
use std::path::PathBuf;
|
||||
|
||||
use crate::{state, Config, Form, Rung};
|
||||
use crate::{state, Config, Rung};
|
||||
|
||||
enum Source {
|
||||
File(PathBuf),
|
||||
@@ -48,12 +48,47 @@ pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
/// CoreML's own cache key leaves out the weights of a model loaded from
|
||||
/// memory (`session::coreml`), and one per runtime version, which wrote it.
|
||||
pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
model_dir(cfg, "coreml", bytes)
|
||||
}
|
||||
|
||||
/// Where OpenVINO compiles `bytes` to, at one precision. OpenVINO hashes
|
||||
/// the model it is given, weights included, but a key that leaves out
|
||||
/// what is being varied has cost a day before (CLAUDE.md, "Providers"),
|
||||
/// and a directory per model and precision costs nothing: the precision
|
||||
/// is a compile option, and the two forms are different programs.
|
||||
pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
|
||||
model_dir(
|
||||
cfg,
|
||||
if fp16 {
|
||||
"openvino/fp16"
|
||||
} else {
|
||||
"openvino/f32"
|
||||
},
|
||||
bytes,
|
||||
)
|
||||
}
|
||||
|
||||
/// Where TensorRT keeps the engine for a whole-frame model. Its own
|
||||
/// directory per model: ONNX Runtime's engine cache key leaves the input
|
||||
/// shape out, and served one export's engine to another of the same graph
|
||||
/// with a different shape when the denoiser was first cut into pieces
|
||||
/// (2026-10-04) — the fixed 1408² denoiser and its any-size sibling are
|
||||
/// exactly that pair. The profile's largest shape is in the name for the
|
||||
/// same reason: an engine built for one range is not the next one's.
|
||||
pub fn tensorrt_whole_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
let (h, w) = crate::WHOLE_FRAME_MAX;
|
||||
model_dir(cfg, &format!("tensorrt-whole-{h}x{w}"), bytes)
|
||||
}
|
||||
|
||||
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
|
||||
/// model, and one per runtime version, which wrote it.
|
||||
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
|
||||
let runtime = match crate::api::runtime() {
|
||||
crate::Runtime::OnnxRuntime { version, .. } => version,
|
||||
crate::Runtime::Tract => "tract".into(),
|
||||
};
|
||||
cfg.cache_dir
|
||||
.join("coreml")
|
||||
.join(provider)
|
||||
.join(runtime)
|
||||
.join(format!("{:016x}", hash(bytes)))
|
||||
}
|
||||
@@ -82,10 +117,10 @@ pub fn run() {
|
||||
(*role, Source::File(path), size)
|
||||
})
|
||||
})
|
||||
.chain(cfg.embedded.iter().filter_map(|(role, bytes)| {
|
||||
// An embedded model has no int8 sibling to offer a rung that
|
||||
// wants one; it runs on that rung's fallback.
|
||||
(rung.serves(*role) && rung.form(*role) == Form::F32).then_some((
|
||||
.chain(cfg.embedded.iter().filter_map(|(role, form, bytes)| {
|
||||
// The embedded form the rung wants, if the build carries it;
|
||||
// a build without it runs that model on the rung's fallback.
|
||||
(rung.serves(*role) && rung.form(*role) == *form).then_some((
|
||||
*role,
|
||||
Source::Bytes(bytes),
|
||||
bytes.len() as u64,
|
||||
|
||||
@@ -0,0 +1,176 @@
|
||||
//! Which GPUs this device has, as far as choosing a runtime needs to know
|
||||
//! (docs/dev/inference.md §3.2).
|
||||
//!
|
||||
//! A runtime carries one vendor's providers — Intel's build has OpenVINO,
|
||||
//! the `onnxruntime-gpu` wheel CUDA and TensorRT, a ROCm build MIGraphX,
|
||||
//! Microsoft's WebGPU build the generic rung — and only one runtime loads
|
||||
//! per process. These checks are what lets `api` load the one that fits
|
||||
//! when a device has several installed. They read files, never a driver:
|
||||
//! a wrong answer costs a slower rung, which the probe still measures, and
|
||||
//! a driver call at start-up could cost the launch.
|
||||
|
||||
/// What a runtime's providers are scored against.
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||
pub struct Gpus {
|
||||
pub nvidia: bool,
|
||||
/// An AMD GPU with the ROCm kernel interface, which MIGraphX needs.
|
||||
pub amd_rocm: bool,
|
||||
pub intel: bool,
|
||||
pub qualcomm: bool,
|
||||
}
|
||||
|
||||
/// The score of a runtime whose vendor rung matches the device's GPU.
|
||||
/// Nothing beats it, so the search stops there.
|
||||
pub const PERFECT: u32 = 3;
|
||||
|
||||
impl Gpus {
|
||||
/// How well a runtime offering `providers` fits this device. The vendor
|
||||
/// rungs score above OpenVINO because a machine with an Intel iGPU and
|
||||
/// an NVIDIA or AMD card wants the card; the generic rung scores above
|
||||
/// a CPU-only build because it carries the same CPU provider and might
|
||||
/// beat it.
|
||||
pub fn score(&self, providers: &[String]) -> u32 {
|
||||
providers
|
||||
.iter()
|
||||
.map(|p| match p.as_str() {
|
||||
"TensorrtExecutionProvider" | "CUDAExecutionProvider" if self.nvidia => PERFECT,
|
||||
"MIGraphXExecutionProvider" if self.amd_rocm => PERFECT,
|
||||
"QNNExecutionProvider" if self.qualcomm => PERFECT,
|
||||
"CoreMLExecutionProvider" => PERFECT,
|
||||
"OpenVINOExecutionProvider" if self.intel => 2,
|
||||
"WebGpuExecutionProvider" => 1,
|
||||
_ => 0,
|
||||
})
|
||||
.max()
|
||||
.unwrap_or(0)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
pub fn detect() -> Gpus {
|
||||
use std::path::Path;
|
||||
// Every DRM card's PCI vendor: an Intel iGPU is `0x8086` whether or
|
||||
// not its compute driver is installed, which the probe finds out.
|
||||
let vendors: Vec<String> = std::fs::read_dir("/sys/class/drm")
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.filter(|e| {
|
||||
let name = e.file_name();
|
||||
let name = name.to_string_lossy();
|
||||
name.starts_with("card") && !name.contains('-')
|
||||
})
|
||||
.filter_map(|e| std::fs::read_to_string(e.path().join("device/vendor")).ok())
|
||||
.map(|v| v.trim().to_string())
|
||||
.collect();
|
||||
Gpus {
|
||||
nvidia: Path::new("/proc/driver/nvidia/version").exists(),
|
||||
amd_rocm: Path::new("/dev/kfd").exists(),
|
||||
intel: vendors.iter().any(|v| v == "0x8086"),
|
||||
qualcomm: false,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "windows")]
|
||||
pub fn detect() -> Gpus {
|
||||
use std::path::PathBuf;
|
||||
let root = std::env::var_os("SystemRoot")
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| PathBuf::from(r"C:\Windows"));
|
||||
let system32 = root.join("System32");
|
||||
// Intel's DCH graphics driver, integrated and Arc alike, installs
|
||||
// from `iigd_dch.inf`; its package directory is the evidence.
|
||||
let intel = std::fs::read_dir(system32.join(r"DriverStore\FileRepository"))
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.any(|e| e.file_name().to_string_lossy().starts_with("iigd_dch"));
|
||||
Gpus {
|
||||
nvidia: system32.join("nvcuda.dll").exists(),
|
||||
amd_rocm: false,
|
||||
intel,
|
||||
qualcomm: false,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
pub fn detect() -> Gpus {
|
||||
// Fail-safe: only a device that names another vendor is not Qualcomm.
|
||||
// `ro.soc.manufacturer` exists from Android 12, and a property or file
|
||||
// the app cannot read reads as nothing; nothing keeps the QNN build
|
||||
// first, as 0.22 had it, where a Qualcomm device mistaken for another
|
||||
// would trade its NPU for the generic rung. Qualcomm's FastRPC library,
|
||||
// which the Hexagon path loads anyway, overrules a name.
|
||||
let soc = crate::probe::system_property("ro.soc.manufacturer");
|
||||
let fastrpc = [
|
||||
"/vendor/lib64/libcdsprpc.so",
|
||||
"/system/vendor/lib64/libcdsprpc.so",
|
||||
]
|
||||
.iter()
|
||||
.any(|p| std::path::Path::new(p).exists());
|
||||
Gpus {
|
||||
qualcomm: qualcomm_soc(&soc) || fastrpc,
|
||||
..Gpus::default()
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `ro.soc.manufacturer` leaves the device Qualcomm's: it says so,
|
||||
/// or it says nothing.
|
||||
#[cfg(any(target_os = "android", test))]
|
||||
fn qualcomm_soc(manufacturer: &str) -> bool {
|
||||
let m = manufacturer.trim();
|
||||
m.is_empty() || m.eq_ignore_ascii_case("QTI") || m.eq_ignore_ascii_case("Qualcomm")
|
||||
}
|
||||
|
||||
#[cfg(not(any(target_os = "linux", target_os = "windows", target_os = "android")))]
|
||||
pub fn detect() -> Gpus {
|
||||
Gpus::default()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn offers(p: &[&str]) -> Vec<String> {
|
||||
p.iter().map(|s| s.to_string()).collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_a_named_other_vendor_is_not_qualcomm() {
|
||||
assert!(qualcomm_soc("QTI"));
|
||||
assert!(qualcomm_soc("Qualcomm"));
|
||||
// Unreadable, or older than Android 12: the QNN build stays first.
|
||||
assert!(qualcomm_soc(""));
|
||||
assert!(!qualcomm_soc("Mediatek"));
|
||||
assert!(!qualcomm_soc("Google"));
|
||||
assert!(!qualcomm_soc("Samsung"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_card_beats_the_integrated_gpu_and_both_beat_the_generic_rung() {
|
||||
let cpu = offers(&["CPUExecutionProvider"]);
|
||||
let nvidia = offers(&[
|
||||
"TensorrtExecutionProvider",
|
||||
"CUDAExecutionProvider",
|
||||
"CPUExecutionProvider",
|
||||
]);
|
||||
let intel = offers(&["OpenVINOExecutionProvider", "CPUExecutionProvider"]);
|
||||
let webgpu = offers(&["WebGpuExecutionProvider", "CPUExecutionProvider"]);
|
||||
let laptop = Gpus {
|
||||
nvidia: true,
|
||||
intel: true,
|
||||
..Gpus::default()
|
||||
};
|
||||
assert!(laptop.score(&nvidia) > laptop.score(&intel));
|
||||
assert!(laptop.score(&intel) > laptop.score(&webgpu));
|
||||
assert!(laptop.score(&webgpu) > laptop.score(&cpu));
|
||||
// No Intel GPU: Intel's build is worth no more than a CPU build to
|
||||
// this device, and the generic rung is worth more.
|
||||
let amd_on_windows = Gpus::default();
|
||||
assert_eq!(amd_on_windows.score(&intel), amd_on_windows.score(&cpu));
|
||||
assert!(amd_on_windows.score(&webgpu) > amd_on_windows.score(&intel));
|
||||
// A ROCm build on a machine without ROCm is a CPU build.
|
||||
let rocm = offers(&["MIGraphXExecutionProvider", "CPUExecutionProvider"]);
|
||||
assert_eq!(amd_on_windows.score(&rocm), 0);
|
||||
}
|
||||
}
|
||||
@@ -21,6 +21,9 @@ use serde::{Deserialize, Serialize};
|
||||
|
||||
mod api;
|
||||
mod engines;
|
||||
// Read only when a runtime is loaded from disk (`api::install_best`).
|
||||
#[cfg_attr(not(feature = "native"), allow(dead_code))]
|
||||
mod hardware;
|
||||
mod probe;
|
||||
mod session;
|
||||
|
||||
@@ -42,25 +45,76 @@ pub enum Role {
|
||||
/// XFeat, the panorama keypoint detector (docs/dev/panorama.md).
|
||||
Keypoints,
|
||||
/// MI-GAN, the panorama border filler (docs/dev/panorama.md §12). Plain
|
||||
/// convolutions, so any rung serves it; fp16 on TensorRT and int8 on
|
||||
/// the Hexagon are the point of it.
|
||||
/// convolutions, so any rung serves it; fp16 on TensorRT and 16-bit
|
||||
/// activations on the Hexagon (int8 changes the fill, §1.5).
|
||||
Inpainter,
|
||||
/// The learned demosaic and denoise on the raw mosaic (docs/dev/denoise.md).
|
||||
/// fp16 costs it nothing measurable; int8 costs 6–9 dB, because 256
|
||||
/// levels cannot hold the shadow steps it exists to recover — so the
|
||||
/// Hexagon does not take it.
|
||||
/// Hexagon takes it with 16-bit activations and weights (§1.5).
|
||||
Denoiser,
|
||||
/// The same denoise networks exported with any height and width, run
|
||||
/// over a whole frame — or the fewest large tiles that fit — instead of
|
||||
/// 1408² tiles whose borders are thrown away (docs/dev/denoise.md §14).
|
||||
/// Served only where a size the graph was not compiled for costs
|
||||
/// nothing: TensorRT, through an optimisation profile up to
|
||||
/// [`WHOLE_FRAME_MAX`], and the CUDA provider. Everywhere else the
|
||||
/// fixed-tile [`Role::Denoiser`] runs; see [`whole_frame_limit`].
|
||||
WholeDenoiser,
|
||||
}
|
||||
|
||||
/// The largest input, rows × columns, a [`Role::WholeDenoiser`] session
|
||||
/// takes: TensorRT's optimisation profile is built up to it, and the tiler
|
||||
/// cuts a larger frame into the fewest tiles no bigger.
|
||||
///
|
||||
/// Sized for a 6 GB card. TensorRT plans its memory for the profile's
|
||||
/// largest shape, and at 4608 × 6656 (a whole 6D frame with Best's border
|
||||
/// and room to spare) it asked for 4.9–5.9 GB and could not build on the
|
||||
/// RTX 3050. At 15 MP a 6D frame is two tiles of 4160 × 3248: 27 MP of
|
||||
/// work for 20 MP kept, against 49 MP in 1408² tiles.
|
||||
pub const WHOLE_FRAME_MAX: (usize, usize) = (4608, 3328);
|
||||
|
||||
/// The input size TensorRT tunes a whole-frame engine for: half a 6D frame
|
||||
/// with Best's border, the tile the reference measurements run.
|
||||
pub const WHOLE_FRAME_OPT: (usize, usize) = (4160, 3248);
|
||||
|
||||
/// Whether the selected rung runs [`Role::WholeDenoiser`], and if so the
|
||||
/// largest input it takes. `None` means run the fixed tiles.
|
||||
pub fn whole_frame_limit() -> Option<(usize, usize)> {
|
||||
let rung = current_rung(&state().lock().unwrap());
|
||||
rung.serves(Role::WholeDenoiser).then_some(WHOLE_FRAME_MAX)
|
||||
}
|
||||
|
||||
/// Which numeric form of a model a session was built from.
|
||||
///
|
||||
/// `Int8` is a different network from `F32` for a detector — it finds a
|
||||
/// different set of faces — which is why [`form_suffix`] exists and why a
|
||||
/// The quantised forms are QDQ graphs, per-channel weights, as QNN's HTP
|
||||
/// takes them (docs/dev/inference.md §1.5): `Int8` is 8-bit activations and
|
||||
/// weights, `A16W8` 16-bit activations with 8-bit weights, `A16W16` 16-bit
|
||||
/// both. The Hexagon accepts no float tensor at all, so these are the
|
||||
/// whole menu; which one a role gets is [`Rung::form`], measured per model.
|
||||
///
|
||||
/// A quantised detector is a different network from the f32 one — it finds
|
||||
/// a different set of faces — which is why [`form_suffix`] exists and why a
|
||||
/// caller appends it to `model_id`.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)]
|
||||
pub enum Form {
|
||||
F32,
|
||||
Int8,
|
||||
A16W8,
|
||||
A16W16,
|
||||
}
|
||||
|
||||
impl Form {
|
||||
/// The infix of the sibling file that holds this form:
|
||||
/// `scrfd_500m_640.a16w8.onnx` beside `scrfd_500m_640.onnx`.
|
||||
pub fn file_tag(self) -> Option<&'static str> {
|
||||
match self {
|
||||
Form::F32 => None,
|
||||
Form::Int8 => Some("int8"),
|
||||
Form::A16W8 => Some("a16w8"),
|
||||
Form::A16W16 => Some("a16w16"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A rung of the ladder (§2). Ordered: a user override names the highest rung
|
||||
@@ -81,7 +135,7 @@ pub enum Rung {
|
||||
/// removed in ONNX Runtime 1.23, so there is no non-compiling AMD rung
|
||||
/// to fall back to: this one falls back to the CPU.
|
||||
MiGraphX,
|
||||
/// Qualcomm's Hexagon NPU through QNN, int8 models only. Android only.
|
||||
/// Qualcomm's Hexagon NPU through QNN, quantised models only. Android only.
|
||||
Hexagon,
|
||||
/// Apple, through CoreML: the Neural Engine, the GPU or the CPU, as
|
||||
/// CoreML schedules it. macOS only. Compiles an ML Program per model on
|
||||
@@ -89,6 +143,16 @@ pub enum Rung {
|
||||
/// embedder stays on the CPU, as on the Hexagon: the Neural Engine
|
||||
/// computes in fp16 (§7).
|
||||
CoreMl,
|
||||
/// Intel, through OpenVINO on the integrated or Arc GPU. Desktop only.
|
||||
/// Compiles a program per model, as MIGraphX does, so the CPU is its
|
||||
/// fallback; fp16 on the same terms as TensorRT (§7).
|
||||
OpenVino,
|
||||
/// Any other GPU, through ONNX Runtime's WebGPU provider: Dawn on
|
||||
/// Vulkan, D3D12 or Metal. The generic rung, for a GPU no vendor rung
|
||||
/// covers. Measured slower than the CPU on every GPU it has been timed
|
||||
/// on (§1), so it is on the ladder for the GPUs it has not, and the
|
||||
/// probe's clock is what keeps it off the rest.
|
||||
WebGpu,
|
||||
}
|
||||
|
||||
impl Rung {
|
||||
@@ -100,6 +164,8 @@ impl Rung {
|
||||
Rung::MiGraphX => "MIGraphX",
|
||||
Rung::Hexagon => "Hexagon NPU",
|
||||
Rung::CoreMl => "CoreML",
|
||||
Rung::OpenVino => "OpenVINO",
|
||||
Rung::WebGpu => "WebGPU",
|
||||
}
|
||||
}
|
||||
|
||||
@@ -108,7 +174,13 @@ impl Rung {
|
||||
fn fallback(self) -> Rung {
|
||||
match self {
|
||||
Rung::TensorRt => Rung::Cuda,
|
||||
Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::Cuda | Rung::Cpu => Rung::Cpu,
|
||||
Rung::MiGraphX
|
||||
| Rung::Hexagon
|
||||
| Rung::CoreMl
|
||||
| Rung::OpenVino
|
||||
| Rung::WebGpu
|
||||
| Rung::Cuda
|
||||
| Rung::Cpu => Rung::Cpu,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -116,28 +188,51 @@ impl Rung {
|
||||
fn compiles(self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::OpenVino
|
||||
)
|
||||
}
|
||||
|
||||
/// The model form this rung wants for a role.
|
||||
fn form(self, _role: Role) -> Form {
|
||||
///
|
||||
/// On the Hexagon, the narrowest form that held each model's accuracy
|
||||
/// on the tablet itself (§1.5): int8 lost 5% of the detector's faces at
|
||||
/// 40–80 px, moved the landmarks by 1.5 px and the segmenter's scores
|
||||
/// to nothing, and the denoiser by 6–9 dB, so those take 16-bit
|
||||
/// activations; the segmenter, scene model, filler and denoiser also
|
||||
/// needed 16-bit weights. Only XFeat keeps int8: its panorama alignment
|
||||
/// moved by no more than f32's own refits do.
|
||||
pub fn form(self, role: Role) -> Form {
|
||||
match self {
|
||||
Rung::Hexagon => Form::Int8,
|
||||
Rung::Hexagon => match role {
|
||||
Role::Keypoints => Form::Int8,
|
||||
Role::Detector | Role::Landmarks => Form::A16W8,
|
||||
Role::Segmenter | Role::Scene | Role::Inpainter | Role::Denoiser => Form::A16W16,
|
||||
// Not served there at all: the Hexagon takes fixed shapes.
|
||||
Role::Embedder | Role::EyeClassifier | Role::WholeDenoiser => Form::F32,
|
||||
},
|
||||
_ => Form::F32,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether this rung runs `role` at all. The Hexagon takes int8 graphs
|
||||
/// only, and the embedder is never int8 (§7) — it runs on the CPU
|
||||
/// beside a detector on the NPU, so its vectors compare across devices.
|
||||
/// Nor is the denoiser: its int8 form failed the 0.5 dB gate by 6–9 dB
|
||||
/// (denoise.md §8), so it runs on the CPU there too. CoreML is kept off
|
||||
/// the embedder for the same reason as the Hexagon: the Neural Engine is
|
||||
/// fp16, and which unit runs a graph is CoreML's choice.
|
||||
/// Whether this rung runs `role` at all. The Hexagon takes quantised
|
||||
/// graphs only, and the embedder is never quantised (§7) — it runs on
|
||||
/// the CPU beside a detector on the NPU, so its vectors compare across
|
||||
/// devices; at A16W16 it still missed the 0.999 cosine gate. The eye
|
||||
/// classifiers stay on the CPU too: a millisecond there, and the two
|
||||
/// share one role while only one of them held its readings quantised.
|
||||
/// CoreML is kept off the embedder for the same reason as the Hexagon:
|
||||
/// the Neural Engine is fp16, and which unit runs a graph is CoreML's
|
||||
/// choice.
|
||||
fn serves(self, role: Role) -> bool {
|
||||
// Any input size only where a new size costs nothing. MIGraphX,
|
||||
// OpenVINO and CoreML compile per shape, the Hexagon takes fixed
|
||||
// shapes only, and the CPU could but would hold gigabytes of f32
|
||||
// activations for a whole frame of Best.
|
||||
if role == Role::WholeDenoiser {
|
||||
return matches!(self, Rung::TensorRt | Rung::Cuda);
|
||||
}
|
||||
match self {
|
||||
Rung::Hexagon => !matches!(role, Role::Embedder | Role::Denoiser),
|
||||
Rung::Hexagon => !matches!(role, Role::Embedder | Role::EyeClassifier),
|
||||
Rung::CoreMl => role != Role::Embedder,
|
||||
_ => true,
|
||||
}
|
||||
@@ -161,8 +256,10 @@ pub struct Config {
|
||||
/// The canonical model files on this device, so engines can be compiled
|
||||
/// ahead of the first request for them.
|
||||
pub models: Vec<(Role, PathBuf)>,
|
||||
/// Models compiled into the binary, for the same reason.
|
||||
pub embedded: Vec<(Role, &'static [u8])>,
|
||||
/// Models compiled into the binary, for the same reason, each with the
|
||||
/// form it is. A build that embeds a quantised sibling lists it here
|
||||
/// beside the f32 graph, and the compile step takes the one the rung wants.
|
||||
pub embedded: Vec<(Role, Form, &'static [u8])>,
|
||||
/// The highest rung the user allows; `None` is "the best that works".
|
||||
pub ceiling: Option<Rung>,
|
||||
/// ONNX Runtime's intra-op pool; 0 picks from the core count.
|
||||
@@ -188,11 +285,11 @@ pub struct Status {
|
||||
}
|
||||
|
||||
impl Status {
|
||||
/// "Hexagon NPU · int8 · ONNX Runtime 1.29" — the settings row's text.
|
||||
/// "Hexagon NPU · quantised · ONNX Runtime 1.29" — the settings row's text.
|
||||
pub fn line(&self) -> String {
|
||||
let form = match self.rung {
|
||||
Rung::Hexagon => " · int8",
|
||||
Rung::TensorRt | Rung::MiGraphX => " · fp16",
|
||||
Rung::Hexagon => " · quantised",
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::OpenVino => " · fp16",
|
||||
_ => "",
|
||||
};
|
||||
format!("{}{} · {}", self.rung.label(), form, self.runtime.label())
|
||||
@@ -374,6 +471,12 @@ struct Cache {
|
||||
/// still reads.
|
||||
#[serde(default)]
|
||||
refused: BTreeSet<String>,
|
||||
/// Probes run under this fingerprint (`probe::run`): a fall-back to the
|
||||
/// CPU is re-probed until there have been `RETRIES`. Defaulted, so a
|
||||
/// cache from 0.22.0 or before — which may hold exactly such a verdict —
|
||||
/// probes again.
|
||||
#[serde(default)]
|
||||
attempts: u32,
|
||||
}
|
||||
|
||||
struct State {
|
||||
@@ -464,26 +567,44 @@ fn current_rung(s: &State) -> Rung {
|
||||
|
||||
/// The file to load for `role` under the current selection, and its form.
|
||||
///
|
||||
/// A rung that wants int8 gets the `.int8.onnx` sibling of the canonical file
|
||||
/// if it exists; otherwise the canonical file, on the rung's fallback. A
|
||||
/// caller adds [`form_suffix`] to the `model_id` it records.
|
||||
/// A rung that wants a quantised form gets that sibling of the canonical
|
||||
/// file (`<stem>.a16w8.onnx` and so on, [`Form::file_tag`]) if it exists;
|
||||
/// otherwise the canonical file, on the rung's fallback. A caller adds
|
||||
/// [`form_suffix`] to the `model_id` it records.
|
||||
pub fn resolve_model(role: Role, canonical: &Path) -> (PathBuf, Form) {
|
||||
let rung = current_rung(&state().lock().unwrap());
|
||||
if rung.serves(role) && rung.form(role) == Form::Int8 {
|
||||
let sibling = int8_sibling(canonical);
|
||||
let want = rung.form(role);
|
||||
if rung.serves(role) && want != Form::F32 {
|
||||
let sibling = form_sibling(canonical, want);
|
||||
if sibling.is_file() {
|
||||
return (sibling, Form::Int8);
|
||||
return (sibling, want);
|
||||
}
|
||||
}
|
||||
(canonical.to_path_buf(), Form::F32)
|
||||
}
|
||||
|
||||
fn int8_sibling(canonical: &Path) -> PathBuf {
|
||||
/// The same choice for a model compiled into the binary: of the forms
|
||||
/// `offered`, the one the current rung wants for `role`, else the f32 one.
|
||||
/// `offered` must hold an `F32` entry.
|
||||
pub fn choose_embedded(role: Role, offered: &[(Form, &'static [u8])]) -> (&'static [u8], Form) {
|
||||
let rung = current_rung(&state().lock().unwrap());
|
||||
let want = rung.form(role);
|
||||
let pick = |form| offered.iter().find(|(f, _)| *f == form);
|
||||
let (form, bytes) = (rung.serves(role).then(|| pick(want)).flatten())
|
||||
.or_else(|| pick(Form::F32))
|
||||
.expect("an embedded model offers its f32 form");
|
||||
(bytes, *form)
|
||||
}
|
||||
|
||||
fn form_sibling(canonical: &Path, form: Form) -> PathBuf {
|
||||
let stem = canonical
|
||||
.file_stem()
|
||||
.map(|s| s.to_string_lossy().into_owned())
|
||||
.unwrap_or_default();
|
||||
canonical.with_file_name(format!("{stem}.int8.onnx"))
|
||||
match form.file_tag() {
|
||||
Some(tag) => canonical.with_file_name(format!("{stem}.{tag}.onnx")),
|
||||
None => canonical.to_path_buf(),
|
||||
}
|
||||
}
|
||||
|
||||
/// What a form appends to a detector's `model_id` (§7).
|
||||
@@ -491,6 +612,8 @@ pub fn form_suffix(form: Form) -> &'static str {
|
||||
match form {
|
||||
Form::F32 => "",
|
||||
Form::Int8 => "_i8",
|
||||
Form::A16W8 => "_a16",
|
||||
Form::A16W16 => "_a16w16",
|
||||
}
|
||||
}
|
||||
|
||||
@@ -520,8 +643,8 @@ pub fn open(role: Role, form: Form, bytes: &[u8]) -> Result<Model, Error> {
|
||||
fn effective_rung(s: &State, selected: Rung, role: Role, form: Form, hash: u64) -> Rung {
|
||||
let mut rung = selected;
|
||||
if !rung.serves(role) || rung.form(role) != form {
|
||||
// The embedder on a Hexagon device, or an f32 detector where the int8
|
||||
// sibling was missing: neither can go to the NPU.
|
||||
// The embedder on a Hexagon device, or an f32 detector where the
|
||||
// quantised sibling was missing: neither can go to the NPU.
|
||||
rung = rung.fallback();
|
||||
}
|
||||
if rung.compiles() && !s.cache.compiled.contains(&engines::key_of(rung, hash)) {
|
||||
@@ -612,9 +735,10 @@ mod tests {
|
||||
#[test]
|
||||
fn the_hexagon_never_takes_the_embedder() {
|
||||
assert!(!Rung::Hexagon.serves(Role::Embedder));
|
||||
assert!(!Rung::Hexagon.serves(Role::Denoiser));
|
||||
assert!(!Rung::Hexagon.serves(Role::EyeClassifier));
|
||||
assert!(Rung::Hexagon.serves(Role::Detector));
|
||||
assert_eq!(Rung::Hexagon.form(Role::Detector), Form::Int8);
|
||||
assert!(Rung::Hexagon.serves(Role::Denoiser));
|
||||
assert_eq!(Rung::Hexagon.form(Role::Detector), Form::A16W8);
|
||||
// A detector offered in f32 on a Hexagon device lands on the CPU.
|
||||
let s = State {
|
||||
config: Config::default(),
|
||||
@@ -625,36 +749,59 @@ mod tests {
|
||||
probing: false,
|
||||
wanted: 0,
|
||||
};
|
||||
let on = |role, form| effective_rung(&s, Rung::Hexagon, role, form, engines::hash(b""));
|
||||
assert_eq!(on(Role::Embedder, Form::F32), Rung::Cpu);
|
||||
assert_eq!(on(Role::Detector, Form::F32), Rung::Cpu);
|
||||
// A form other than the one the role wants is not the NPU's either:
|
||||
// an int8 detector left over from an older install stays off it.
|
||||
assert_eq!(on(Role::Detector, Form::Int8), Rung::Cpu);
|
||||
// The wanted form whose context is not compiled yet: also the CPU.
|
||||
assert_eq!(on(Role::Detector, Form::A16W8), Rung::Cpu);
|
||||
}
|
||||
|
||||
/// The form each role gets on the Hexagon is the one measured to hold
|
||||
/// its accuracy there (§1.5); a change to this table is a change to
|
||||
/// what the tablet computes, and must come with a measurement.
|
||||
#[test]
|
||||
fn each_role_has_its_measured_form_on_the_hexagon() {
|
||||
use Form::*;
|
||||
for (role, form) in [
|
||||
(Role::Detector, A16W8),
|
||||
(Role::Landmarks, A16W8),
|
||||
(Role::Segmenter, A16W16),
|
||||
(Role::Scene, A16W16),
|
||||
(Role::Inpainter, A16W16),
|
||||
(Role::Denoiser, A16W16),
|
||||
(Role::Keypoints, Int8),
|
||||
(Role::Embedder, F32),
|
||||
(Role::EyeClassifier, F32),
|
||||
] {
|
||||
assert_eq!(Rung::Hexagon.form(role), form, "{role:?}");
|
||||
}
|
||||
for rung in [
|
||||
Rung::Cpu,
|
||||
Rung::Cuda,
|
||||
Rung::TensorRt,
|
||||
Rung::MiGraphX,
|
||||
Rung::CoreMl,
|
||||
Rung::OpenVino,
|
||||
Rung::WebGpu,
|
||||
] {
|
||||
assert_eq!(rung.form(Role::Detector), F32);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_form_lives_in_its_tagged_sibling() {
|
||||
let canonical = Path::new("/m/scrfd_500m_640.onnx");
|
||||
assert_eq!(form_sibling(canonical, Form::F32), canonical);
|
||||
assert_eq!(
|
||||
effective_rung(
|
||||
&s,
|
||||
Rung::Hexagon,
|
||||
Role::Embedder,
|
||||
Form::F32,
|
||||
engines::hash(b"")
|
||||
),
|
||||
Rung::Cpu
|
||||
form_sibling(canonical, Form::A16W8),
|
||||
Path::new("/m/scrfd_500m_640.a16w8.onnx")
|
||||
);
|
||||
assert_eq!(
|
||||
effective_rung(
|
||||
&s,
|
||||
Rung::Hexagon,
|
||||
Role::Detector,
|
||||
Form::F32,
|
||||
engines::hash(b"")
|
||||
),
|
||||
Rung::Cpu
|
||||
);
|
||||
// An int8 detector whose context is not compiled yet: also the CPU.
|
||||
assert_eq!(
|
||||
effective_rung(
|
||||
&s,
|
||||
Rung::Hexagon,
|
||||
Role::Detector,
|
||||
Form::Int8,
|
||||
engines::hash(b"")
|
||||
),
|
||||
Rung::Cpu
|
||||
form_sibling(canonical, Form::Int8),
|
||||
Path::new("/m/scrfd_500m_640.int8.onnx")
|
||||
);
|
||||
}
|
||||
|
||||
@@ -679,6 +826,29 @@ mod tests {
|
||||
assert_eq!(on(&s, Role::Embedder), Rung::Cpu);
|
||||
}
|
||||
|
||||
/// OpenVINO compiles a program per model, so a request waits on the CPU
|
||||
/// until the engine thread has built it; WebGPU builds in the session
|
||||
/// and serves at once. Both take every role in f32 graphs.
|
||||
#[test]
|
||||
fn openvino_waits_for_its_program_and_webgpu_does_not() {
|
||||
let hash = engines::hash(b"detector");
|
||||
let mut s = State {
|
||||
config: Config::default(),
|
||||
cache: Cache::default(),
|
||||
probing: false,
|
||||
wanted: 0,
|
||||
};
|
||||
let on = |s: &State, rung| effective_rung(s, rung, Role::Detector, Form::F32, hash);
|
||||
assert_eq!(on(&s, Rung::OpenVino), Rung::Cpu);
|
||||
s.cache
|
||||
.compiled
|
||||
.insert(engines::key_of(Rung::OpenVino, hash));
|
||||
assert_eq!(on(&s, Rung::OpenVino), Rung::OpenVino);
|
||||
assert_eq!(on(&s, Rung::WebGpu), Rung::WebGpu);
|
||||
let embedder = effective_rung(&s, Rung::WebGpu, Role::Embedder, Form::F32, hash);
|
||||
assert_eq!(embedder, Rung::WebGpu);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_status_reports_only_the_rungs_above_the_selection() {
|
||||
let _serial = serial();
|
||||
|
||||
@@ -14,23 +14,64 @@ use crate::{api::Runtime, state, Cache, Config, Form, Role, Rung};
|
||||
|
||||
/// The rungs to try on this platform, best first, under the user's ceiling.
|
||||
fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
|
||||
// WebGPU is the generic rung (§2): it is reached only on a runtime that
|
||||
// carries it, which `api` loads where no vendor's runtime fits the
|
||||
// device, and kept only where it beats the CPU.
|
||||
#[cfg(target_os = "android")]
|
||||
let all = [Rung::Hexagon];
|
||||
let all = [Rung::Hexagon, Rung::WebGpu];
|
||||
// Unmeasured (§2 ⁵): it is on the ladder because the probe's clock and
|
||||
// `attempt` make a wrong guess cost one slow or failed probe, not a
|
||||
// slow or crashing app.
|
||||
#[cfg(target_os = "macos")]
|
||||
let all = [Rung::CoreMl];
|
||||
// A desktop has one vendor's GPU; the other vendor's providers are
|
||||
// "not enabled in this build" or a library that fails to load, and
|
||||
// either answer arrives in milliseconds.
|
||||
// A runtime carries one vendor's providers, chosen for this device's
|
||||
// GPU (`api`); the others are "not enabled in this build", and that
|
||||
// answer arrives in milliseconds.
|
||||
#[cfg(not(any(target_os = "android", target_os = "macos")))]
|
||||
let all = [Rung::TensorRt, Rung::Cuda, Rung::MiGraphX];
|
||||
let all = [
|
||||
Rung::TensorRt,
|
||||
Rung::Cuda,
|
||||
Rung::MiGraphX,
|
||||
Rung::OpenVino,
|
||||
Rung::WebGpu,
|
||||
];
|
||||
all.into_iter()
|
||||
.filter(|r| ceiling.is_none_or(|c| *r <= c))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Probes under one fingerprint that may end on the CPU after an
|
||||
/// accelerator failed or lost, before that answer is kept.
|
||||
const RETRIES: u32 = 3;
|
||||
|
||||
/// What a cached probe result is good for.
|
||||
#[derive(Debug, PartialEq)]
|
||||
enum Reuse {
|
||||
/// Use it as it is.
|
||||
Keep,
|
||||
/// Probe again: it fell back to the CPU after this many probes.
|
||||
Again(u32),
|
||||
/// Another device, runtime or model set: probe from the start.
|
||||
Fresh,
|
||||
}
|
||||
|
||||
/// The CPU because an accelerator failed or lost is asked again on the next
|
||||
/// launches, a few times: a failure can be a moment's (QNN could not create
|
||||
/// its device on 0.22.0's first launch after the update), and keeping it for
|
||||
/// good left the tablet's every model on the CPU. Bounded, so a wedged
|
||||
/// driver costs a few launches, not all.
|
||||
fn reuse(cached: &Cache, fingerprint: &str) -> Reuse {
|
||||
if cached.fingerprint != fingerprint || cached.rung.is_none() {
|
||||
return Reuse::Fresh;
|
||||
}
|
||||
let fell_back = cached.rung == Some(Rung::Cpu) && !cached.failed.is_empty();
|
||||
if fell_back && cached.attempts < RETRIES {
|
||||
Reuse::Again(cached.attempts)
|
||||
} else {
|
||||
Reuse::Keep
|
||||
}
|
||||
}
|
||||
|
||||
/// The probe body. Sets the cache and clears `probing` when done; never
|
||||
/// panics out, because a failed probe is a result (the floor) and not an
|
||||
/// error.
|
||||
@@ -38,20 +79,33 @@ pub fn run(runtime: Runtime) {
|
||||
let cfg = state().lock().unwrap().config.clone();
|
||||
let fingerprint = fingerprint(&runtime, &cfg);
|
||||
|
||||
let mut attempts = 0;
|
||||
if let Some(cached) = read_cache(&cfg) {
|
||||
if cached.fingerprint == fingerprint && cached.rung.is_some() {
|
||||
log::info!(
|
||||
"inference: cached selection {} ({})",
|
||||
cached.rung.unwrap().label(),
|
||||
cached.reason
|
||||
);
|
||||
finish(cached);
|
||||
return;
|
||||
match reuse(&cached, &fingerprint) {
|
||||
Reuse::Keep => {
|
||||
log::info!(
|
||||
"inference: cached selection {} ({})",
|
||||
cached.rung.map_or("?", |r| r.label()),
|
||||
cached.reason
|
||||
);
|
||||
finish(cached);
|
||||
return;
|
||||
}
|
||||
Reuse::Again(n) => {
|
||||
attempts = n;
|
||||
log::info!(
|
||||
"inference: probing again after falling back to the CPU ({}), attempt {} of {RETRIES}",
|
||||
cached.reason,
|
||||
n + 1
|
||||
);
|
||||
}
|
||||
Reuse::Fresh => {}
|
||||
}
|
||||
}
|
||||
|
||||
let mut cache = Cache {
|
||||
fingerprint,
|
||||
attempts: attempts + 1,
|
||||
..Cache::default()
|
||||
};
|
||||
|
||||
@@ -112,7 +166,16 @@ pub fn run(runtime: Runtime) {
|
||||
}
|
||||
if cache.rung.is_none() {
|
||||
cache.rung = Some(Rung::Cpu);
|
||||
cache.reason = match cache.failed.first() {
|
||||
// The rung that tried and lost, not the first one the runtime was
|
||||
// never built with: "WebGPU 150 ms, slower than the CPU" says why
|
||||
// this device is on the CPU, "TensorRT not enabled" does not.
|
||||
let tried = cache
|
||||
.failed
|
||||
.iter()
|
||||
.rev()
|
||||
.find(|(_, why)| !why.contains("in this build"))
|
||||
.or(cache.failed.first());
|
||||
cache.reason = match tried {
|
||||
Some((r, why)) => format!("{} {}", r.label(), first_line(why)),
|
||||
None => "the only rung on this platform".into(),
|
||||
};
|
||||
@@ -174,10 +237,9 @@ pub fn attempt<T>(cfg: &Config, what: &str, f: impl FnOnce() -> T) -> Result<T,
|
||||
}
|
||||
|
||||
/// The smallest detector, or the smallest model of any role if there is
|
||||
/// none. A ~2 MB detector is the cheapest real test of a provider, and the
|
||||
/// detector is the role the int8 forms exist for — the eye classifiers are
|
||||
/// smaller still, and a Hexagon probed with one would fail for want of a
|
||||
/// form nobody ships.
|
||||
/// none. A ~2 MB detector is the cheapest real test of a provider, and
|
||||
/// every rung serves it — the eye classifiers are smaller still, but the
|
||||
/// Hexagon does not take them, and a probe with one would fail it for that.
|
||||
fn probe_model(cfg: &Config) -> Option<(Role, PathBuf)> {
|
||||
let smallest = |want: Option<Role>| {
|
||||
cfg.models
|
||||
@@ -193,9 +255,14 @@ fn probe_model(cfg: &Config) -> Option<(Role, PathBuf)> {
|
||||
smallest(Some(Role::Detector)).or_else(|| smallest(None))
|
||||
}
|
||||
|
||||
/// Build, run once for the engine, then time three runs; the median in
|
||||
/// milliseconds and, for a compiling rung, the cache key of the engine this
|
||||
/// just built.
|
||||
/// Build, warm up, then time seven runs; the median in milliseconds and,
|
||||
/// for a compiling rung, the cache key of the engine this just built.
|
||||
///
|
||||
/// Three warm-ups, not one: an idle integrated GPU takes a few runs to
|
||||
/// raise its clock. With one, the Iris Xe's OpenVINO lost to the CPU on
|
||||
/// the smallest detector in two probes of three, where warm it is 5.8 ms
|
||||
/// against 9.5 (§1.6). The smallest detector is a GPU's worst case; the
|
||||
/// clock must not also be.
|
||||
fn time_rung(
|
||||
rung: Rung,
|
||||
role: Role,
|
||||
@@ -203,20 +270,18 @@ fn time_rung(
|
||||
cfg: &Config,
|
||||
) -> Result<(f64, Option<String>), String> {
|
||||
let want = rung.form(role);
|
||||
let path = match want {
|
||||
Form::Int8 => {
|
||||
let p = crate::int8_sibling(canonical);
|
||||
if !p.is_file() {
|
||||
return Err(format!("no int8 form of {}", canonical.display()));
|
||||
}
|
||||
p
|
||||
}
|
||||
Form::F32 => canonical.to_path_buf(),
|
||||
};
|
||||
let path = crate::form_sibling(canonical, want);
|
||||
if want != Form::F32 && !path.is_file() {
|
||||
return Err(format!(
|
||||
"no {} form of {}",
|
||||
want.file_tag().unwrap_or("f32"),
|
||||
canonical.display()
|
||||
));
|
||||
}
|
||||
let bytes = std::fs::read(&path).map_err(|e| e.to_string())?;
|
||||
let started = Instant::now();
|
||||
let mut session =
|
||||
crate::session::build(rung, role, &bytes, cfg).map_err(|e| first_line(&e.to_string()))?;
|
||||
let mut session = crate::session::build_probe(rung, role, &bytes, cfg)
|
||||
.map_err(|e| first_line(&e.to_string()))?;
|
||||
log::info!(
|
||||
"inference: {} session built in {:.1} s",
|
||||
rung.label(),
|
||||
@@ -255,11 +320,15 @@ fn time_rung(
|
||||
.map_err(|e| e.to_string())?;
|
||||
Ok(t.elapsed().as_secs_f64() * 1e3)
|
||||
};
|
||||
run(&mut session)?;
|
||||
let mut times = [run(&mut session)?, run(&mut session)?, run(&mut session)?];
|
||||
for _ in 0..3 {
|
||||
run(&mut session)?;
|
||||
}
|
||||
let mut times = (0..7)
|
||||
.map(|_| run(&mut session))
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
times.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
||||
let key = rung.compiles().then(|| crate::engines::key(rung, &bytes));
|
||||
Ok((times[1], key))
|
||||
Ok((times[times.len() / 2], key))
|
||||
}
|
||||
|
||||
/// The part of a provider's error a person can act on. ONNX Runtime's
|
||||
@@ -295,9 +364,9 @@ fn fingerprint(runtime: &Runtime, cfg: &Config) -> String {
|
||||
},
|
||||
device_identity(),
|
||||
];
|
||||
for (role, bytes) in &cfg.embedded {
|
||||
for (role, form, bytes) in &cfg.embedded {
|
||||
parts.push(format!(
|
||||
"{role:?} embedded {:016x}",
|
||||
"{role:?} embedded {form:?} {:016x}",
|
||||
crate::engines::hash(bytes)
|
||||
));
|
||||
}
|
||||
@@ -306,9 +375,13 @@ fn fingerprint(runtime: &Runtime, cfg: &Config) -> String {
|
||||
.map(|b| crate::engines::hash(&b))
|
||||
.unwrap_or(0);
|
||||
parts.push(format!("{role:?} {hash:016x}"));
|
||||
let int8 = crate::int8_sibling(path);
|
||||
if let Ok(b) = std::fs::read(&int8) {
|
||||
parts.push(format!("{role:?} int8 {:016x}", crate::engines::hash(&b)));
|
||||
for form in [Form::Int8, Form::A16W8, Form::A16W16] {
|
||||
if let Ok(b) = std::fs::read(crate::form_sibling(path, form)) {
|
||||
parts.push(format!(
|
||||
"{role:?} {form:?} {:016x}",
|
||||
crate::engines::hash(&b)
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
parts.join("\n")
|
||||
@@ -336,19 +409,35 @@ fn providers_beside(runtime: &Path) -> String {
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn device_identity() -> String {
|
||||
// The NVIDIA driver's version line, or the ROCm release the AMD stack
|
||||
// came from (`rocm-core` writes it; the kernel driver has no version
|
||||
// of its own). Absent means neither.
|
||||
// The NVIDIA driver's version line, the ROCm release the AMD stack came
|
||||
// from (`rocm-core` writes it; the kernel driver has no version of its
|
||||
// own), and the OpenCL drivers registered — OpenVINO reaches the GPU
|
||||
// through one, and installing Intel's is what makes the Iris Xe a rung.
|
||||
let mut parts = Vec::new();
|
||||
if let Some(line) = std::fs::read_to_string("/proc/driver/nvidia/version")
|
||||
.ok()
|
||||
.and_then(|s| s.lines().next().map(str::to_string))
|
||||
{
|
||||
return line;
|
||||
parts.push(line);
|
||||
}
|
||||
if let Ok(rocm) = std::fs::read_to_string("/opt/rocm/.info/version") {
|
||||
return format!("rocm {}", rocm.trim());
|
||||
parts.push(format!("rocm {}", rocm.trim()));
|
||||
}
|
||||
let mut icds: Vec<String> = std::fs::read_dir("/etc/OpenCL/vendors")
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.file_name().to_string_lossy().into_owned())
|
||||
.collect();
|
||||
icds.sort();
|
||||
if !icds.is_empty() {
|
||||
parts.push(format!("opencl {}", icds.join(" ")));
|
||||
}
|
||||
if parts.is_empty() {
|
||||
"no nvidia driver, no rocm, no opencl".into()
|
||||
} else {
|
||||
parts.join("; ")
|
||||
}
|
||||
"no nvidia driver, no rocm".into()
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
@@ -363,7 +452,7 @@ fn device_identity() -> String {
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
fn system_property(name: &str) -> String {
|
||||
pub(crate) fn system_property(name: &str) -> String {
|
||||
extern "C" {
|
||||
fn __system_property_get(
|
||||
name: *const std::ffi::c_char,
|
||||
@@ -493,4 +582,34 @@ mod tests {
|
||||
died_inside(&cfg, "probe TensorRT", 2);
|
||||
assert_eq!(attempt(&cfg, "probe CUDA", || 7), Ok(7));
|
||||
}
|
||||
|
||||
/// The tablet's cache after 0.22.0's first launch, as 0.22.0 wrote it:
|
||||
/// no `attempts`, the Hexagon "rejected", the CPU selected.
|
||||
const TABLET: &str = r#"{"fingerprint":"f","rung":"Cpu","reason":"Hexagon NPU 28.5 ms, slower than the CPU's 19.4 ms","compiled":[],"failed":[["Hexagon","28.5 ms, slower than the CPU's 19.4 ms"]]}"#;
|
||||
|
||||
#[test]
|
||||
fn a_fall_back_to_the_cpu_is_probed_again_a_few_times() {
|
||||
let mut cache: Cache = serde_json::from_str(TABLET).unwrap();
|
||||
assert_eq!(cache.attempts, 0, "a 0.22.0 cache reads as never retried");
|
||||
assert_eq!(reuse(&cache, "f"), Reuse::Again(0));
|
||||
cache.attempts = RETRIES - 1;
|
||||
assert_eq!(reuse(&cache, "f"), Reuse::Again(RETRIES - 1));
|
||||
cache.attempts = RETRIES;
|
||||
assert_eq!(reuse(&cache, "f"), Reuse::Keep, "then it is kept");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_accelerator_chosen_or_a_cpu_only_device_is_kept() {
|
||||
let mut cache: Cache = serde_json::from_str(TABLET).unwrap();
|
||||
cache.rung = Some(Rung::Hexagon);
|
||||
assert_eq!(reuse(&cache, "f"), Reuse::Keep);
|
||||
cache.rung = Some(Rung::Cpu);
|
||||
cache.failed.clear();
|
||||
assert_eq!(
|
||||
reuse(&cache, "f"),
|
||||
Reuse::Keep,
|
||||
"nothing failed: the only rung"
|
||||
);
|
||||
assert_eq!(reuse(&cache, "other"), Reuse::Fresh);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,30 @@ use crate::{Config, Role, Rung};
|
||||
/// its clock instead (§4): a provider that hands real work to the CPU is
|
||||
/// slower than the CPU floor and rejected by the same measurement.
|
||||
pub fn build(rung: Rung, role: Role, bytes: &[u8], cfg: &Config) -> ort::Result<Session> {
|
||||
build_with(rung, role, bytes, cfg, false)
|
||||
}
|
||||
|
||||
/// [`build`] for the probe: on the Hexagon, a session that cannot put the
|
||||
/// whole graph on the NPU fails instead of running the rest on the CPU.
|
||||
///
|
||||
/// The probe times a rung by its session, and a QNN provider that could not
|
||||
/// create its device still builds one — with every node on the CPU behind
|
||||
/// it. 0.22.0's first launch on the tablet timed that (28.5 ms against the
|
||||
/// CPU's own 19.4) and put every model on the CPU. Only the probe is strict:
|
||||
/// some shipped graphs keep a few nodes on the CPU on purpose
|
||||
/// (`tools/quantise-models.py`, `float_nodes`), and the probe's detector is
|
||||
/// not one of them.
|
||||
pub fn build_probe(rung: Rung, role: Role, bytes: &[u8], cfg: &Config) -> ort::Result<Session> {
|
||||
build_with(rung, role, bytes, cfg, rung == Rung::Hexagon)
|
||||
}
|
||||
|
||||
fn build_with(
|
||||
rung: Rung,
|
||||
role: Role,
|
||||
bytes: &[u8],
|
||||
cfg: &Config,
|
||||
strict: bool,
|
||||
) -> ort::Result<Session> {
|
||||
// No optimisation level named. ONNX Runtime's default is already its
|
||||
// fullest, and on tract any level but "disabled" means `into_optimized`,
|
||||
// whose optimiser divides by zero inside yolo26n-seg (tract-data
|
||||
@@ -22,15 +46,22 @@ pub fn build(rung: Rung, role: Role, bytes: &[u8], cfg: &Config) -> ort::Result<
|
||||
if crate::api::runtime().is_native() {
|
||||
b = with_runtime_log(b)?;
|
||||
}
|
||||
if strict {
|
||||
b = b.with_config_entry("session.disable_cpu_ep_fallback", "1")?;
|
||||
}
|
||||
// A Hexagon session loads the compiled context when there is one and
|
||||
// compiles it from the model when there is not; the engine thread is
|
||||
// what makes the second case rare (§6).
|
||||
let context = (rung == Rung::Hexagon).then(|| crate::engines::context_path(cfg, bytes));
|
||||
let ready = context.as_ref().is_some_and(|p| p.is_file());
|
||||
// What the rung keeps for this model: the context the Hexagon is to
|
||||
// write, or the directory CoreML compiles into.
|
||||
// write, or the directory CoreML or OpenVINO compiles into.
|
||||
let per_model = match rung {
|
||||
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
|
||||
Rung::TensorRt if role == Role::WholeDenoiser => {
|
||||
Some(crate::engines::tensorrt_whole_dir(cfg, bytes))
|
||||
}
|
||||
Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))),
|
||||
_ if ready => None,
|
||||
_ => context.clone(),
|
||||
};
|
||||
@@ -74,6 +105,13 @@ fn with_runtime_log(
|
||||
.with_log_level(level)?)
|
||||
}
|
||||
|
||||
/// Whether `role` runs in fp16 on a rung that offers it: everything but the
|
||||
/// embedder, whose comparability across devices is worth more than its
|
||||
/// fraction of a millisecond (§7).
|
||||
fn fp16(role: Role) -> bool {
|
||||
role != Role::Embedder
|
||||
}
|
||||
|
||||
/// The intra-op pool: what the config says, else the cores less two for
|
||||
/// the compositor and the decoder (§9). tract ignores it.
|
||||
fn threads(cfg: &Config) -> usize {
|
||||
@@ -100,6 +138,11 @@ fn providers(
|
||||
Rung::Cuda => {
|
||||
Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?)
|
||||
}
|
||||
Rung::TensorRt if role == Role::WholeDenoiser => {
|
||||
let mut b = b;
|
||||
tensorrt_whole(&mut b, per_model.expect("a whole-frame engine directory"))?;
|
||||
Ok(b.with_execution_providers([ep::CUDA::default().build()])?)
|
||||
}
|
||||
Rung::TensorRt => {
|
||||
let cache = cfg.cache_dir.join("tensorrt");
|
||||
let _ = std::fs::create_dir_all(&cache);
|
||||
@@ -110,7 +153,7 @@ fn providers(
|
||||
// (NFR-RES-2). CUDA behind it takes any node TensorRT declines.
|
||||
Ok(b.with_execution_providers([
|
||||
ep::TensorRT::default()
|
||||
.with_fp16(role != Role::Embedder)
|
||||
.with_fp16(fp16(role))
|
||||
.with_engine_cache(true)
|
||||
.with_engine_cache_path(&cache)
|
||||
.with_timing_cache(true)
|
||||
@@ -127,7 +170,7 @@ fn providers(
|
||||
// directory, keyed on the graph, the GPU and its own version
|
||||
// but not the precision: hence one directory per precision.
|
||||
// The CPU takes any node it declines.
|
||||
let fp16 = role != Role::Embedder;
|
||||
let fp16 = fp16(role);
|
||||
let cache = cfg
|
||||
.cache_dir
|
||||
.join("migraphx")
|
||||
@@ -137,6 +180,16 @@ fn providers(
|
||||
migraphx(&mut b, fp16, &cache)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::OpenVino => {
|
||||
let mut b = b;
|
||||
openvino(&mut b, fp16(role), per_model)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::WebGpu => {
|
||||
let mut b = b;
|
||||
webgpu(&mut b)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::Hexagon => unreachable!("the Hexagon rung is not on a desktop ladder"),
|
||||
}
|
||||
}
|
||||
@@ -192,15 +245,145 @@ fn migraphx(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
fp16: bool,
|
||||
cache: &std::path::Path,
|
||||
) -> ort::Result<()> {
|
||||
append(
|
||||
b,
|
||||
c"MIGraphX",
|
||||
&[
|
||||
("migraphx_fp16_enable", if fp16 { "1" } else { "0" }.into()),
|
||||
(
|
||||
"migraphx_model_cache_dir",
|
||||
cache.to_string_lossy().into_owned(),
|
||||
),
|
||||
],
|
||||
)
|
||||
}
|
||||
|
||||
/// TensorRT for a whole-frame model: one engine for every input size up to
|
||||
/// [`crate::WHOLE_FRAME_MAX`], kept in its own directory.
|
||||
///
|
||||
/// `ort`'s builder has no profile options, so this registers through the
|
||||
/// runtime's TensorRT V2 options, with the names 1.30 reads
|
||||
/// (`tensorrt_execution_provider_info.cc`): `trt_profile_{min,opt,max}_shapes`.
|
||||
/// Without a profile a dynamic input compiles a new engine per size at run
|
||||
/// time — 156 s on the first frame, measured — so the profile is the
|
||||
/// difference between a whole-frame engine and a stall. fp16, as for every
|
||||
/// role but the embedder (§7); the denoiser measured 0.00 dB from f32.
|
||||
#[cfg(not(target_os = "android"))]
|
||||
fn tensorrt_whole(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
cache: &std::path::Path,
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let keys = [c"migraphx_fp16_enable", c"migraphx_model_cache_dir"];
|
||||
let values = [
|
||||
CString::new(if fp16 { "1" } else { "0" }).unwrap(),
|
||||
CString::new(cache.to_string_lossy().as_bytes())
|
||||
.map_err(|e| ort::Error::new(e.to_string()))?,
|
||||
let _ = std::fs::create_dir_all(cache);
|
||||
let shapes = |(h, w): (usize, usize)| format!("mosaic:1x1x{h}x{w},sigma:1x1x{h}x{w}");
|
||||
let dir = cache.to_string_lossy().into_owned();
|
||||
let options = [
|
||||
("trt_fp16_enable", "1".to_string()),
|
||||
("trt_engine_cache_enable", "1".to_string()),
|
||||
("trt_engine_cache_path", dir.clone()),
|
||||
("trt_timing_cache_enable", "1".to_string()),
|
||||
("trt_timing_cache_path", dir),
|
||||
("trt_max_workspace_size", (1u64 << 30).to_string()),
|
||||
("trt_profile_min_shapes", shapes((256, 256))),
|
||||
("trt_profile_opt_shapes", shapes(crate::WHOLE_FRAME_OPT)),
|
||||
("trt_profile_max_shapes", shapes(crate::WHOLE_FRAME_MAX)),
|
||||
];
|
||||
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
|
||||
let keys = options
|
||||
.iter()
|
||||
.map(|(k, _)| cstr(k))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let values = options
|
||||
.iter()
|
||||
.map(|(_, v)| cstr(v))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
let api = ort::api();
|
||||
// SAFETY: the documented create / update / append / release sequence
|
||||
// `ort`'s own TensorRT builder makes, over arrays that outlive it; the
|
||||
// runtime copies the options into the session before the release.
|
||||
unsafe {
|
||||
let mut trt: *mut ort::sys::OrtTensorRTProviderOptionsV2 = std::ptr::null_mut();
|
||||
ort::Error::result_from_status((api.CreateTensorRTProviderOptions)(&mut trt))?;
|
||||
let result = ort::Error::result_from_status((api.UpdateTensorRTProviderOptions)(
|
||||
trt,
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
))
|
||||
.and_then(|()| {
|
||||
ort::Error::result_from_status((api.SessionOptionsAppendExecutionProvider_TensorRT_V2)(
|
||||
b.ptr_mut(),
|
||||
trt,
|
||||
))
|
||||
});
|
||||
(api.ReleaseTensorRTProviderOptions)(trt);
|
||||
result
|
||||
}
|
||||
}
|
||||
|
||||
/// OpenVINO on the GPU, compiling into `cache`.
|
||||
///
|
||||
/// The option names are those `openvino_provider_factory.cc` reads at 1.24,
|
||||
/// the version of Intel's `onnxruntime-openvino` build. `GPU` is OpenVINO's
|
||||
/// first OpenCL GPU: the Intel one on a hybrid laptop with both drivers
|
||||
/// installed, but an NVIDIA card through its OpenCL when Intel's is absent
|
||||
/// — slower than the CPU there, and rejected by the probe's clock. The
|
||||
/// precision is always named: the GPU plugin's own default is fp16, and
|
||||
/// the embedder must not get it (§7).
|
||||
#[cfg(not(target_os = "android"))]
|
||||
fn openvino(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
fp16: bool,
|
||||
cache: Option<&std::path::Path>,
|
||||
) -> ort::Result<()> {
|
||||
let mut options = vec![
|
||||
("device_type", "GPU".to_string()),
|
||||
("precision", if fp16 { "FP16" } else { "FP32" }.into()),
|
||||
];
|
||||
if let Some(dir) = cache {
|
||||
let _ = std::fs::create_dir_all(dir);
|
||||
options.push(("cache_dir", dir.to_string_lossy().into_owned()));
|
||||
}
|
||||
append(b, c"OpenVINO", &options)
|
||||
}
|
||||
|
||||
/// WebGPU on the high-performance adapter: the discrete GPU where there is
|
||||
/// one, since the integrated one on a machine with both is the one this
|
||||
/// rung is least likely to beat the CPU on. The key is as
|
||||
/// `webgpu_provider_options.h` spells it, without the `ep.<name>.` prefix
|
||||
/// the runtime adds.
|
||||
fn webgpu(b: &mut ort::session::builder::SessionBuilder) -> ort::Result<()> {
|
||||
append(
|
||||
b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "high-performance".to_string())],
|
||||
)
|
||||
}
|
||||
|
||||
/// Register the provider `name` with `options` through the runtime's
|
||||
/// generic key/value entry point, which takes every provider by its short
|
||||
/// name and reads options at the runtime's own version — not at the
|
||||
/// version `ort`'s builders were written against (CLAUDE.md, "Providers").
|
||||
fn append(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
name: &std::ffi::CStr,
|
||||
options: &[(&str, String)],
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
|
||||
let keys = options
|
||||
.iter()
|
||||
.map(|(k, _)| cstr(k))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let values = options
|
||||
.iter()
|
||||
.map(|(_, v)| cstr(v))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
// SAFETY: the documented C call over arrays that outlive it; the
|
||||
@@ -208,7 +391,7 @@ fn migraphx(
|
||||
unsafe {
|
||||
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
|
||||
b.ptr_mut(),
|
||||
c"MIGraphX".as_ptr(),
|
||||
name.as_ptr(),
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
@@ -250,7 +433,12 @@ fn providers(
|
||||
.build()
|
||||
.error_on_failure()])?)
|
||||
}
|
||||
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl => {
|
||||
Rung::WebGpu => {
|
||||
let mut b = b;
|
||||
webgpu(&mut b)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl | Rung::OpenVino => {
|
||||
unreachable!("no desktop rung on Android")
|
||||
}
|
||||
}
|
||||
|
||||
+12
-1
@@ -13,8 +13,13 @@ const MODELS: &[&str] = &[
|
||||
"../../models/keypoints/xfeat-768.onnx",
|
||||
];
|
||||
|
||||
const QUANTISED: &[&str] = &[
|
||||
"../../models/keypoints/xfeat-1024.int8.onnx",
|
||||
"../../models/keypoints/xfeat-768.int8.onnx",
|
||||
];
|
||||
|
||||
fn main() {
|
||||
for m in MODELS {
|
||||
for m in MODELS.iter().chain(QUANTISED) {
|
||||
println!("cargo:rerun-if-changed={m}");
|
||||
}
|
||||
println!("cargo:rerun-if-changed=build.rs");
|
||||
@@ -26,6 +31,12 @@ fn main() {
|
||||
for model in MODELS.iter().copied() {
|
||||
check(model);
|
||||
}
|
||||
// The Hexagon's int8 forms ride only in an Android build.
|
||||
if std::env::var("CARGO_CFG_TARGET_OS").as_deref() == Ok("android") {
|
||||
for model in QUANTISED.iter().copied() {
|
||||
check(model);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn check(model: &str) {
|
||||
|
||||
@@ -30,7 +30,8 @@ pub struct MiGan {
|
||||
|
||||
impl MiGan {
|
||||
/// From the model file, in whichever form the engine's rung wants
|
||||
/// (`resolve_model` picks an int8 sibling for the Hexagon).
|
||||
/// (`resolve_model` picks the `.a16w16.onnx` sibling on the Hexagon:
|
||||
/// int8 moved the fill 16 dB from f32's, 16-bit about 41).
|
||||
pub fn from_path(path: &std::path::Path) -> Result<Self, PanoError> {
|
||||
use dr_inference_engine::{resolve_model, Role};
|
||||
let (path, form) = resolve_model(Role::Inpainter, path);
|
||||
|
||||
@@ -36,18 +36,49 @@ pub struct XFeat {
|
||||
pub options: DecodeOptions,
|
||||
}
|
||||
|
||||
/// The bytes of both exports compiled into the binary, for whoever compiles
|
||||
/// engines ahead of the first request (docs/dev/inference.md §6).
|
||||
/// The Hexagon's forms (docs/dev/inference.md §1.5): int8, from the same
|
||||
/// network spelled for the HTP (the unfold as SpaceToDepth, the bilinear
|
||||
/// resizes as matrix products). Only Android has a Hexagon.
|
||||
#[cfg(all(feature = "embedded-model", target_os = "android"))]
|
||||
const EMBEDDED_LANDSCAPE_INT8: &[u8] =
|
||||
include_bytes!("../../../models/keypoints/xfeat-1024.int8.onnx");
|
||||
#[cfg(all(feature = "embedded-model", target_os = "android"))]
|
||||
const EMBEDDED_PORTRAIT_INT8: &[u8] =
|
||||
include_bytes!("../../../models/keypoints/xfeat-768.int8.onnx");
|
||||
|
||||
/// Every form of both exports compiled into the binary, landscape then
|
||||
/// portrait, for whoever compiles engines ahead of the first request
|
||||
/// (docs/dev/inference.md §6).
|
||||
#[cfg(feature = "embedded-model")]
|
||||
pub fn embedded_model_bytes() -> [&'static [u8]; 2] {
|
||||
[EMBEDDED_LANDSCAPE, EMBEDDED_PORTRAIT]
|
||||
pub fn embedded_models() -> [Vec<(dr_inference_engine::Form, &'static [u8])>; 2] {
|
||||
use dr_inference_engine::Form;
|
||||
#[allow(unused_mut)]
|
||||
let mut forms = [
|
||||
vec![(Form::F32, EMBEDDED_LANDSCAPE)],
|
||||
vec![(Form::F32, EMBEDDED_PORTRAIT)],
|
||||
];
|
||||
#[cfg(target_os = "android")]
|
||||
{
|
||||
forms[0].push((Form::Int8, EMBEDDED_LANDSCAPE_INT8));
|
||||
forms[1].push((Form::Int8, EMBEDDED_PORTRAIT_INT8));
|
||||
}
|
||||
forms
|
||||
}
|
||||
|
||||
impl XFeat {
|
||||
/// The weights compiled into the binary.
|
||||
/// The weights compiled into the binary, in the form the device's
|
||||
/// backend runs.
|
||||
#[cfg(feature = "embedded-model")]
|
||||
pub fn embedded() -> Result<Self, PanoError> {
|
||||
Self::from_bytes(EMBEDDED_LANDSCAPE, EMBEDDED_PORTRAIT)
|
||||
use dr_inference_engine::{choose_embedded, open, Role};
|
||||
let [l, p] = embedded_models();
|
||||
let (l, lf) = choose_embedded(Role::Keypoints, &l);
|
||||
let (p, pf) = choose_embedded(Role::Keypoints, &p);
|
||||
Ok(XFeat {
|
||||
landscape: open(Role::Keypoints, lf, l)?,
|
||||
portrait: open(Role::Keypoints, pf, p)?,
|
||||
options: DecodeOptions::default(),
|
||||
})
|
||||
}
|
||||
|
||||
/// From the two exports on disk.
|
||||
|
||||
@@ -25,7 +25,7 @@ vibrance.vibrance = 10
|
||||
blacks_whites.blacks = -8
|
||||
clarity.amount = 12
|
||||
contrast.contrast = 18
|
||||
vibrance.vibrance = 18
|
||||
vibrance.vibrance = 36
|
||||
|
||||
[preset Recover the sky]
|
||||
blacks_whites.whites = -10
|
||||
|
||||
@@ -15,38 +15,38 @@ drpl 1
|
||||
|
||||
[preset Blue sky]
|
||||
colour_mixer.azure_lum = -20
|
||||
colour_mixer.azure_sat = 25
|
||||
colour_mixer.azure_sat = 18
|
||||
colour_mixer.blue_lum = -15
|
||||
colour_mixer.blue_sat = 20
|
||||
colour_mixer.blue_sat = 14
|
||||
highlights_shadows.highlights = -15
|
||||
|
||||
[preset Deep blue sky]
|
||||
colour_mixer.azure_hue = 10
|
||||
colour_mixer.azure_lum = -30
|
||||
colour_mixer.azure_sat = 35
|
||||
colour_mixer.azure_sat = 22
|
||||
colour_mixer.blue_lum = -25
|
||||
colour_mixer.blue_sat = 30
|
||||
colour_mixer.cyan_sat = 10
|
||||
colour_mixer.blue_sat = 19
|
||||
colour_mixer.cyan_sat = 6
|
||||
highlights_shadows.highlights = -30
|
||||
|
||||
[preset Polariser]
|
||||
colour_mixer.azure_hue = 10
|
||||
colour_mixer.azure_lum = -35
|
||||
colour_mixer.azure_sat = 40
|
||||
colour_mixer.azure_sat = 29
|
||||
colour_mixer.blue_lum = -30
|
||||
colour_mixer.blue_sat = 35
|
||||
colour_mixer.blue_sat = 26
|
||||
colour_mixer.cyan_lum = -10
|
||||
colour_mixer.cyan_sat = 15
|
||||
colour_mixer.cyan_sat = 11
|
||||
dehaze.amount = 20
|
||||
highlights_shadows.highlights = -35
|
||||
vibrance.vibrance = 10
|
||||
vibrance.vibrance = 7
|
||||
|
||||
[preset Blue sky, golden land]
|
||||
colour_mixer.azure_lum = -20
|
||||
colour_mixer.azure_sat = 25
|
||||
colour_mixer.azure_sat = 16
|
||||
colour_mixer.blue_lum = -15
|
||||
colour_mixer.blue_sat = 20
|
||||
colour_mixer.orange_sat = 12
|
||||
colour_mixer.blue_sat = 13
|
||||
colour_mixer.orange_sat = 8
|
||||
colour_mixer.yellow_hue = -10
|
||||
colour_mixer.yellow_sat = 15
|
||||
colour_mixer.yellow_sat = 10
|
||||
highlights_shadows.highlights = -20
|
||||
|
||||
@@ -9,18 +9,27 @@ drpl 1
|
||||
#
|
||||
# Each changes only what it names (FR-DEV-6), so a corrected exposure or
|
||||
# white balance survives applying one.
|
||||
#
|
||||
# How much colour each adds is measured, not guessed: mean CIELAB chroma on
|
||||
# raws rendered with the default (DNG reference) rendering, as a ratio to that
|
||||
# rendering. For scale, the photographer's earlier exports of the same kind of
|
||||
# raws sit at 1.14 with no look applied and 1.27 with their everyday look.
|
||||
# Vivid 1.30 and Vivid warm 1.30 sit just above that; Vivid landscape 1.38;
|
||||
# Vivid, strong 1.45; Vivid portrait 1.15, with its skin bands held down as
|
||||
# written. Tuned by scaling each preset's colour values together, never its
|
||||
# tone ones.
|
||||
|
||||
[preset Vivid]
|
||||
contrast.contrast = 10
|
||||
saturation.saturation = 8
|
||||
vibrance.vibrance = 30
|
||||
saturation.saturation = 11
|
||||
vibrance.vibrance = 42
|
||||
|
||||
[preset Vivid, strong]
|
||||
blacks_whites.blacks = -10
|
||||
clarity.amount = 8
|
||||
contrast.contrast = 18
|
||||
saturation.saturation = 15
|
||||
vibrance.vibrance = 45
|
||||
saturation.saturation = 19
|
||||
vibrance.vibrance = 57
|
||||
|
||||
# Foliage and sky: green and chartreuse for leaves and grass, azure and blue
|
||||
# for sky and water, a little yellow for dry grass and stone. The skin bands
|
||||
@@ -29,37 +38,37 @@ vibrance.vibrance = 45
|
||||
[preset Vivid landscape]
|
||||
clarity.amount = 10
|
||||
colour_mixer.azure_lum = -10
|
||||
colour_mixer.azure_sat = 20
|
||||
colour_mixer.azure_sat = 25
|
||||
colour_mixer.blue_lum = -10
|
||||
colour_mixer.blue_sat = 15
|
||||
colour_mixer.chartreuse_sat = 15
|
||||
colour_mixer.green_sat = 20
|
||||
colour_mixer.yellow_sat = 10
|
||||
colour_mixer.blue_sat = 20
|
||||
colour_mixer.chartreuse_sat = 20
|
||||
colour_mixer.green_sat = 25
|
||||
colour_mixer.yellow_sat = 13
|
||||
contrast.contrast = 12
|
||||
saturation.saturation = 5
|
||||
vibrance.vibrance = 25
|
||||
saturation.saturation = 7
|
||||
vibrance.vibrance = 32
|
||||
|
||||
# Golden hour: oranges and yellows up and a warm cast laid over the
|
||||
# highlights only, so shadows stay clean rather than muddy.
|
||||
[preset Vivid warm]
|
||||
colour_grading.highlight_hue = 45
|
||||
colour_grading.highlight_strength = 12
|
||||
colour_mixer.orange_sat = 15
|
||||
colour_mixer.red_sat = 8
|
||||
colour_mixer.yellow_sat = 15
|
||||
colour_mixer.orange_sat = 17
|
||||
colour_mixer.red_sat = 9
|
||||
colour_mixer.yellow_sat = 17
|
||||
contrast.contrast = 8
|
||||
vibrance.vibrance = 25
|
||||
vibrance.vibrance = 29
|
||||
|
||||
# People: everything around the subject gets richer while skin does not.
|
||||
# Vibrance already protects skin; the orange and red bands are then held a
|
||||
# little below where they started, because a face is the one colour every
|
||||
# viewer knows the right value of.
|
||||
[preset Vivid portrait]
|
||||
colour_mixer.azure_sat = 10
|
||||
colour_mixer.blue_sat = 12
|
||||
colour_mixer.green_sat = 12
|
||||
colour_mixer.azure_sat = 14
|
||||
colour_mixer.blue_sat = 17
|
||||
colour_mixer.green_sat = 17
|
||||
colour_mixer.orange_sat = -10
|
||||
colour_mixer.red_sat = -5
|
||||
contrast.contrast = 6
|
||||
saturation.saturation = -5
|
||||
vibrance.vibrance = 25
|
||||
vibrance.vibrance = 35
|
||||
|
||||
+135
-32
@@ -176,12 +176,12 @@ pub struct EditGraph {
|
||||
/// lens profile, so not in the state; it only decides whether the
|
||||
/// switch below is offered.
|
||||
denoise_available: bool,
|
||||
/// Whether the learned denoise replaces the demosaic. An edit: published
|
||||
/// as [`crate::learned_denoise`], captured, stored and undone with the
|
||||
/// rest (FR-DEV-3c).
|
||||
denoise_applied: bool,
|
||||
/// How much of the removed noise's brightness to put back, 0–100.
|
||||
denoise_grain: f32,
|
||||
/// Which demosaic develops the photograph: a network, or the classical
|
||||
/// one. An edit: published as [`crate::learned_denoise`], captured,
|
||||
/// stored and undone with the rest (FR-DEV-3c).
|
||||
denoise_method: crate::learned_denoise::Method,
|
||||
/// How strongly to denoise, 0–100; what is not taken goes back as grain.
|
||||
denoise_strength: f32,
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-3f
|
||||
@@ -239,8 +239,8 @@ impl EditGraph {
|
||||
lens_profile: None,
|
||||
lens_profile_applied: true,
|
||||
denoise_available: false,
|
||||
denoise_applied: false,
|
||||
denoise_grain: 0.0,
|
||||
denoise_method: crate::learned_denoise::Method::DEFAULT,
|
||||
denoise_strength: 100.0,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -425,13 +425,19 @@ impl EditGraph {
|
||||
/// photograph that cannot take it is harmless and does nothing, as a
|
||||
/// lens switch with no profile does.
|
||||
pub fn denoise_applied(&self) -> bool {
|
||||
self.denoise_applied
|
||||
self.denoise_method.learned()
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// The grain to keep, 0–1.
|
||||
/// Which demosaic is asked for.
|
||||
pub fn denoise_method(&self) -> crate::learned_denoise::Method {
|
||||
self.denoise_method
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// The grain to keep, 0–1: what the strength does not take.
|
||||
pub fn denoise_grain(&self) -> f32 {
|
||||
self.denoise_grain / 100.0
|
||||
(100.0 - self.denoise_strength) / 100.0
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-3
|
||||
@@ -625,7 +631,7 @@ impl EditGraph {
|
||||
OpCapability {
|
||||
id: desc.id,
|
||||
label: desc.label,
|
||||
active: self.denoise_applied,
|
||||
active: self.denoise_applied(),
|
||||
params: desc
|
||||
.params
|
||||
.iter()
|
||||
@@ -643,10 +649,12 @@ impl EditGraph {
|
||||
}
|
||||
});
|
||||
|
||||
switch
|
||||
// The learned denoise first: it decides what every control below
|
||||
// is applied to, so it heads the panel (docs/dev/denoise.md §7).
|
||||
denoise
|
||||
.into_iter()
|
||||
.chain(switch)
|
||||
.chain(warps)
|
||||
.chain(denoise)
|
||||
.chain(ops)
|
||||
.chain(std::iter::once(framing))
|
||||
.collect()
|
||||
@@ -741,8 +749,8 @@ impl EditGraph {
|
||||
// Derived from the file, like the profile above.
|
||||
denoise_available: _,
|
||||
// Edits, in the state through `capabilities` like the lens switch.
|
||||
denoise_applied: _,
|
||||
denoise_grain: _,
|
||||
denoise_method: _,
|
||||
denoise_strength: _,
|
||||
masks,
|
||||
film,
|
||||
spots,
|
||||
@@ -816,9 +824,25 @@ impl EditGraph {
|
||||
pub fn set_param(&mut self, op: OpId, param: ParamId, value: f32) {
|
||||
if op == crate::learned_denoise::ID {
|
||||
match param {
|
||||
p if p == crate::learned_denoise::APPLY => self.denoise_applied = value != 0.0,
|
||||
p if p == crate::learned_denoise::METHOD => {
|
||||
self.denoise_method = crate::learned_denoise::Method::from_index(value)
|
||||
}
|
||||
// 0.21 and 0.22's switch (see `APPLY`): off is the classical
|
||||
// demosaic, on is a network — the one already chosen, if any.
|
||||
p if p == crate::learned_denoise::APPLY => {
|
||||
use crate::learned_denoise::Method;
|
||||
if value == 0.0 {
|
||||
self.denoise_method = Method::Bilinear;
|
||||
} else if !self.denoise_method.learned() {
|
||||
self.denoise_method = Method::DEFAULT;
|
||||
}
|
||||
}
|
||||
p if p == crate::learned_denoise::STRENGTH => {
|
||||
self.denoise_strength = value.clamp(0.0, 100.0)
|
||||
}
|
||||
// 0.21.0's grain, the strength's inverse (see `GRAIN`).
|
||||
p if p == crate::learned_denoise::GRAIN => {
|
||||
self.denoise_grain = value.clamp(0.0, 100.0)
|
||||
self.denoise_strength = 100.0 - value.clamp(0.0, 100.0)
|
||||
}
|
||||
_ => log::warn!("unknown parameter {param} on {op}; ignoring"),
|
||||
}
|
||||
@@ -883,10 +907,12 @@ impl EditGraph {
|
||||
pub fn param(&self, op: OpId, param: ParamId) -> Option<f32> {
|
||||
if op == crate::learned_denoise::ID {
|
||||
return match param {
|
||||
p if p == crate::learned_denoise::METHOD => Some(self.denoise_method.index()),
|
||||
p if p == crate::learned_denoise::APPLY => {
|
||||
Some(if self.denoise_applied { 1.0 } else { 0.0 })
|
||||
Some(if self.denoise_applied() { 1.0 } else { 0.0 })
|
||||
}
|
||||
p if p == crate::learned_denoise::GRAIN => Some(self.denoise_grain),
|
||||
p if p == crate::learned_denoise::STRENGTH => Some(self.denoise_strength),
|
||||
p if p == crate::learned_denoise::GRAIN => Some(100.0 - self.denoise_strength),
|
||||
_ => None,
|
||||
};
|
||||
}
|
||||
@@ -936,10 +962,10 @@ impl EditGraph {
|
||||
// a reset does not change which lens took the photograph. What returns
|
||||
// to default is the answer to whether to use it, which is on.
|
||||
self.set_lens_profile_applied(true);
|
||||
// The learned denoise returns to off; whether it is available is the
|
||||
// file's and stays.
|
||||
self.denoise_applied = false;
|
||||
self.denoise_grain = 0.0;
|
||||
// The learned denoise returns to its default network; whether it is
|
||||
// available is the file's and stays.
|
||||
self.denoise_method = crate::learned_denoise::Method::DEFAULT;
|
||||
self.denoise_strength = 100.0;
|
||||
}
|
||||
|
||||
/// Set the crop rectangle. Clamped to keep it inside the frame.
|
||||
@@ -2098,28 +2124,105 @@ mod tests {
|
||||
.into_iter()
|
||||
.find(|c| c.id == learned_denoise::ID)
|
||||
.expect("offered");
|
||||
assert!(!cap.active, "off until asked for");
|
||||
g.set_param(learned_denoise::ID, learned_denoise::APPLY, 1.0);
|
||||
g.set_param(learned_denoise::ID, learned_denoise::GRAIN, 30.0);
|
||||
assert!(g.denoise_applied());
|
||||
assert!(cap.active, "on by default");
|
||||
assert_eq!(g.denoise_method(), learned_denoise::Method::Best);
|
||||
assert_eq!(g.denoise_grain(), 0.0, "at full strength");
|
||||
assert_eq!(cap.id, g.capabilities()[0].id, "and first in the panel");
|
||||
g.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
learned_denoise::Method::Bilinear.index(),
|
||||
);
|
||||
g.set_param(learned_denoise::ID, learned_denoise::STRENGTH, 70.0);
|
||||
assert!(!g.denoise_applied());
|
||||
assert!((g.denoise_grain() - 0.3).abs() < 1e-6);
|
||||
g.reset();
|
||||
assert!(!g.denoise_applied());
|
||||
assert!(g.denoise_applied(), "reset is back to on");
|
||||
assert_eq!(g.denoise_grain(), 0.0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_untouched_raw_writes_nothing_and_develops_through_the_best() {
|
||||
// TRACES: FR-DEV-3g
|
||||
use crate::learned_denoise::{self, Method};
|
||||
let mut g = EditGraph::default_chain();
|
||||
g.set_denoise_available(true);
|
||||
assert_eq!(g.denoise_method(), Method::Best);
|
||||
let stored = |g: &EditGraph| {
|
||||
crate::Preset::capture_params(g)
|
||||
.params()
|
||||
.keys()
|
||||
.any(|(op, _)| op == learned_denoise::ID.0)
|
||||
};
|
||||
assert!(!stored(&g), "the default is not written");
|
||||
g.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
Method::Fast.index(),
|
||||
);
|
||||
assert!(stored(&g), "a choice is");
|
||||
assert!(
|
||||
!crate::Preset::capture_params(&g).params().contains_key(&(
|
||||
learned_denoise::ID.0.into(),
|
||||
learned_denoise::APPLY.0.into()
|
||||
)),
|
||||
"and the old switch never is"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_edit_saved_with_the_switch_keeps_its_look() {
|
||||
// TRACES: FR-DEV-3g
|
||||
// 0.21 and 0.22 stored on or off; off is the classical demosaic, and
|
||||
// on keeps a network already chosen.
|
||||
use crate::learned_denoise::{self, Method};
|
||||
let mut g = EditGraph::default_chain();
|
||||
g.set_param(learned_denoise::ID, learned_denoise::APPLY, 0.0);
|
||||
assert_eq!(g.denoise_method(), Method::Bilinear);
|
||||
g.set_param(learned_denoise::ID, learned_denoise::APPLY, 1.0);
|
||||
assert_eq!(g.denoise_method(), Method::DEFAULT);
|
||||
g.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
Method::Fast.index(),
|
||||
);
|
||||
g.set_param(learned_denoise::ID, learned_denoise::APPLY, 1.0);
|
||||
assert_eq!(g.denoise_method(), Method::Fast);
|
||||
// A number from a newer build with more methods is the default.
|
||||
g.set_param(learned_denoise::ID, learned_denoise::METHOD, 9.0);
|
||||
assert_eq!(g.denoise_method(), Method::DEFAULT);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_edit_saved_with_grain_keeps_its_look() {
|
||||
// TRACES: FR-DEV-3g
|
||||
// 0.21.0 stored the grain kept rather than the strength.
|
||||
use crate::learned_denoise;
|
||||
let mut g = EditGraph::default_chain();
|
||||
g.set_param(learned_denoise::ID, learned_denoise::GRAIN, 25.0);
|
||||
assert_eq!(
|
||||
g.param(learned_denoise::ID, learned_denoise::STRENGTH),
|
||||
Some(75.0)
|
||||
);
|
||||
assert!((g.denoise_grain() - 0.25).abs() < 1e-6);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_learned_denoise_travels_in_the_state() {
|
||||
use crate::learned_denoise;
|
||||
let mut g = EditGraph::default_chain();
|
||||
g.set_denoise_available(true);
|
||||
g.set_param(learned_denoise::ID, learned_denoise::APPLY, 1.0);
|
||||
g.set_param(learned_denoise::ID, learned_denoise::GRAIN, 40.0);
|
||||
g.set_param(learned_denoise::ID, learned_denoise::STRENGTH, 60.0);
|
||||
g.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
learned_denoise::Method::Fast.index(),
|
||||
);
|
||||
let state = g.state();
|
||||
let mut h = EditGraph::default_chain();
|
||||
h.set_denoise_available(true);
|
||||
let _ = h.set_state(&state);
|
||||
assert!(h.denoise_applied());
|
||||
assert_eq!(h.denoise_method(), learned_denoise::Method::Fast);
|
||||
assert!((h.denoise_grain() - 0.4).abs() < 1e-6);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
//! TRACES: FR-DEV-3g
|
||||
//! The learned denoise's settings: whether to use it, and how much grain to
|
||||
//! keep.
|
||||
//! The learned denoise's settings: which network develops the photograph,
|
||||
//! if any, and how much grain to keep.
|
||||
//!
|
||||
//! Not an [`crate::operation::Operation`]: the learned stage replaces the
|
||||
//! demosaic and runs once per photograph, off the render path
|
||||
@@ -17,24 +17,95 @@ use crate::descriptor::{Attribute, LocalizedKey, OpDescriptor, ParamDescriptor,
|
||||
use crate::{OpId, ParamId};
|
||||
|
||||
pub const ID: OpId = OpId("learned_denoise");
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// Which demosaic develops the photograph, a [`Method`] by index.
|
||||
pub const METHOD: ParamId = ParamId("method");
|
||||
/// What 0.21 and 0.22 stored instead of [`METHOD`]: on or off. Still read —
|
||||
/// off is [`Method::Bilinear`], on is the default network — so an edit saved
|
||||
/// by those releases keeps its look; never written, and not offered.
|
||||
pub const APPLY: ParamId = ParamId("apply");
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// How strongly to denoise, 0–100: 100 is the network's result as it is, and
|
||||
/// lower puts the removed noise's brightness back as grain.
|
||||
pub const STRENGTH: ParamId = ParamId("strength");
|
||||
/// What 0.21.0 stored instead of [`STRENGTH`]: the grain kept, its inverse.
|
||||
/// Still read, so an edit saved by that release keeps its look; never
|
||||
/// written, and not offered as a control.
|
||||
pub const GRAIN: ParamId = ParamId("grain");
|
||||
|
||||
/// Off by default: it costs seconds per photograph and replaces the
|
||||
/// demosaic, which is the photographer's call. Grain 0 is the network's
|
||||
/// result as it is.
|
||||
/// TRACES: FR-DEV-3g
|
||||
/// The demosaics a photograph can be developed with, in the order the
|
||||
/// sidecar numbers them. Two networks that trade time for quality
|
||||
/// (docs/dev/denoise.md §15) and the classical demosaic, which is no network
|
||||
/// at all.
|
||||
///
|
||||
/// Until 0.24 there were four — Bilinear, Fast, Medium, Best — and the
|
||||
/// sidecar keeps their numbers: 2, which was Medium, is now Best, and 3,
|
||||
/// which was Best, is past the end and reads as the default, which is
|
||||
/// Best. Both land on the network that replaced them, with no migration.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||
pub enum Method {
|
||||
/// The classical demosaic: the noise stays.
|
||||
Bilinear,
|
||||
/// The smallest student: a quarter of Best's work.
|
||||
Fast,
|
||||
/// One network of the first release's size, taught by the mixture of
|
||||
/// experts it replaced: the mixture's edges at a third of its work.
|
||||
Best,
|
||||
}
|
||||
|
||||
impl Method {
|
||||
pub const ALL: [Method; 3] = [Method::Bilinear, Method::Fast, Method::Best];
|
||||
pub const DEFAULT: Method = Method::Best;
|
||||
|
||||
/// The sidecar's number for it.
|
||||
pub fn index(self) -> f32 {
|
||||
Self::ALL.iter().position(|m| *m == self).unwrap_or(0) as f32
|
||||
}
|
||||
|
||||
/// The method a stored number names; out of range is the default, as
|
||||
/// from a newer build with more of them.
|
||||
pub fn from_index(value: f32) -> Method {
|
||||
let i = value.round();
|
||||
if i >= 0.0 && (i as usize) < Self::ALL.len() {
|
||||
Self::ALL[i as usize]
|
||||
} else {
|
||||
Self::DEFAULT
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a network runs at all.
|
||||
pub fn learned(self) -> bool {
|
||||
self != Method::Bilinear
|
||||
}
|
||||
}
|
||||
|
||||
/// The best network by default, at full strength: every Bayer raw is
|
||||
/// developed from the learned demosaic, and the choice and the slider are
|
||||
/// there to take it back, trade it for time, or ease it off. It costs seconds per photograph the first time, while
|
||||
/// the classical demosaic shows; the result is cached, so a photograph
|
||||
/// reopened or exported does not pay again (docs/dev/denoise.md §7).
|
||||
pub(crate) static DESCRIPTOR: LazyLock<Arc<OpDescriptor>> = LazyLock::new(|| {
|
||||
Arc::new(OpDescriptor {
|
||||
id: ID,
|
||||
label: LocalizedKey("op.learned_denoise"),
|
||||
params: vec![
|
||||
ParamDescriptor::switch("apply", "param.learned_denoise.apply"),
|
||||
ParamDescriptor::choice(
|
||||
"method",
|
||||
"param.learned_denoise.method",
|
||||
vec![
|
||||
LocalizedKey("param.learned_denoise.method.bilinear"),
|
||||
LocalizedKey("param.learned_denoise.method.fast"),
|
||||
LocalizedKey("param.learned_denoise.method.best"),
|
||||
],
|
||||
)
|
||||
.with_default(Method::DEFAULT.index()),
|
||||
ParamDescriptor::scalar(
|
||||
"grain",
|
||||
"param.learned_denoise.grain",
|
||||
"strength",
|
||||
"param.learned_denoise.strength",
|
||||
0.0,
|
||||
100.0,
|
||||
0.0,
|
||||
100.0,
|
||||
Unit::Percent,
|
||||
Scale::Linear,
|
||||
0,
|
||||
@@ -49,3 +120,20 @@ pub(crate) static DESCRIPTOR: LazyLock<Arc<OpDescriptor>> = LazyLock::new(|| {
|
||||
pub fn descriptor() -> Arc<OpDescriptor> {
|
||||
DESCRIPTOR.clone()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// An edit saved before 0.24 stored Medium as 2 and Best as 3. Both
|
||||
/// now name the network that replaced them, and nothing reads as Fast
|
||||
/// or Bilinear that did not before.
|
||||
#[test]
|
||||
fn the_retired_methods_read_as_best() {
|
||||
assert_eq!(Method::from_index(0.0), Method::Bilinear);
|
||||
assert_eq!(Method::from_index(1.0), Method::Fast);
|
||||
assert_eq!(Method::from_index(2.0), Method::Best, "Medium, before 0.24");
|
||||
assert_eq!(Method::from_index(3.0), Method::Best, "Best, before 0.24");
|
||||
assert_eq!(Method::Best.index(), 2.0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -47,9 +47,18 @@ pub const LOOK: ParamId = ParamId("look");
|
||||
|
||||
/// The look's strength at which the LookTable is applied as the profile
|
||||
/// states it, in percent.
|
||||
pub const DEFAULT_LOOK: f32 = 100.0;
|
||||
pub const PROFILE_LOOK: f32 = 100.0;
|
||||
/// The look's default strength: off.
|
||||
///
|
||||
/// Measured, not chosen. Against the photographer's earlier exports with no
|
||||
/// look applied, the default rendering scores the same with the table at 100,
|
||||
/// 50 or 0 (held-out MSE 140, 140, 143), and is 9 % more colourful without
|
||||
/// it: the table desaturates near-neutral tones, which is exactly where the
|
||||
/// default rendering was short of those exports. The table stays one slider
|
||||
/// away for anyone who wants the profile's look.
|
||||
pub const DEFAULT_LOOK: f32 = 0.0;
|
||||
/// Twice the profile's look.
|
||||
pub const MAX_LOOK: f32 = 200.0;
|
||||
pub const MAX_LOOK: f32 = 2.0 * PROFILE_LOOK;
|
||||
|
||||
/// Entries of the buffer's header, before the entries themselves: one
|
||||
/// `vec4` describing each table — `(hue divisions, saturation divisions,
|
||||
|
||||
@@ -39,6 +39,16 @@ use crate::ops::helpers;
|
||||
|
||||
pub const ID: OpId = OpId("colour_mixer");
|
||||
|
||||
/// How much a raised saturation band adds to a muted colour, per unit of its
|
||||
/// value; the push tapers linearly to nothing at full saturation.
|
||||
///
|
||||
/// Measured, not chosen. Fitted against the photographer's earlier exports
|
||||
/// (two looks, ~90 photographs), a sky band raised by 58 there lifted muted
|
||||
/// sky blues about 2.1×; at 3.0 a band here does about the same at the same
|
||||
/// value, so imported values stay inside the slider's range (dr-preset-xmp
|
||||
/// carries the per-band factors). Lowering saturation is unaffected.
|
||||
pub const SAT_GAIN: f32 = 3.0;
|
||||
|
||||
/// The twelve bands, in hue order starting at red.
|
||||
///
|
||||
/// Twelve rather than Lightroom's eight: the extra bands fall between the
|
||||
@@ -333,9 +343,16 @@ impl Operation for ColourMixer {
|
||||
// Only the bands the user actually touched contribute code. A single
|
||||
// adjusted band therefore costs one weight evaluation rather than
|
||||
// twelve — the composition property applied within an operation.
|
||||
let mut lines = String::from(
|
||||
let lines = String::from(
|
||||
"\
|
||||
let hcl = rgb_to_hcl(c);
|
||||
// Bands are matched, and saturation judged, on display-encoded values: in
|
||||
// scene-linear light a muted colour reads as strongly saturated and its hue
|
||||
// sits away from where it is seen, so a band set on what the photograph
|
||||
// shows would land on other colours.
|
||||
let lin = max(c, vec3<f32>(0.0));
|
||||
let e = pow(lin, vec3<f32>(1.0 / 2.2));
|
||||
let SAT_GAIN = @SAT_GAIN@;
|
||||
let hcl = rgb_to_hcl(e);
|
||||
let hue = hcl.x;
|
||||
let chroma = hcl.y;
|
||||
let hi = hcl.z;
|
||||
@@ -350,6 +367,8 @@ if (chroma > 0.0001) {
|
||||
",
|
||||
);
|
||||
|
||||
let mut lines = lines.replace("@SAT_GAIN@", &format!("{SAT_GAIN:.4}"));
|
||||
|
||||
for (b, band) in BANDS.iter().enumerate() {
|
||||
let v = self.values[b];
|
||||
if v.iter().all(|x| *x == 0.0) {
|
||||
@@ -384,10 +403,17 @@ if (chroma > 0.0001) {
|
||||
// yellow-green to green, not enough to turn it blue by accident.
|
||||
let new_hue = hue + d_hue * 30.0;
|
||||
|
||||
// Saturation scales chroma; luminance scales the whole colour.
|
||||
let new_chroma = clamp(chroma * (1.0 + d_sat), 0.0, hi);
|
||||
c = hue_to_rgb_scale(new_hue, new_chroma, hi);
|
||||
c = c * exp2(d_lum);
|
||||
// Saturation. Raising it pushes muted colours hardest and tapers to
|
||||
// nothing at full saturation, as the eye expects a mixer to; the
|
||||
// gain makes a value deliver the strength it names, measured
|
||||
// against the photographer's earlier exports. Lowering it scales
|
||||
// every colour alike, so -100 is grey.
|
||||
let sat = chroma / max(hi, 0.00001);
|
||||
let gain = select(1.0 + d_sat, 1.0 + d_sat * SAT_GAIN * (1.0 - sat), d_sat > 0.0);
|
||||
let new_chroma = clamp(chroma * gain, 0.0, hi);
|
||||
let shifted = hue_to_rgb_scale(new_hue, new_chroma, hi);
|
||||
// Back to linear light, where luminance scales the whole colour.
|
||||
c = pow(shifted, vec3<f32>(2.2)) * exp2(d_lum);
|
||||
}
|
||||
}
|
||||
c = max(c, vec3<f32>(0.0));",
|
||||
|
||||
@@ -675,6 +675,56 @@ impl PresetLibrary {
|
||||
self.unknown.values().map(Vec::len).sum()
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-6
|
||||
/// Merge two copies of a library that both descend from `base`.
|
||||
///
|
||||
/// For keeping one library on several devices: `base` is what the last
|
||||
/// exchange left both sides holding, `ours` this device's copy now and
|
||||
/// `theirs` the server's. Each name is decided on its own:
|
||||
///
|
||||
/// - Changed on one side only — added, edited or deleted — that side's
|
||||
/// answer stands. This is why the base is needed at all: without it a
|
||||
/// preset deleted here and one added there look the same, and a
|
||||
/// deletion would come back on every exchange.
|
||||
/// - Changed on both sides to the same thing, nothing to decide.
|
||||
/// - Deleted on one side and edited on the other, the edit stands. A
|
||||
/// preset is work, and an absence is not.
|
||||
/// - Edited on both sides differently, ours stands. Either answer loses
|
||||
/// one edit; this one at least converges, since the other device takes
|
||||
/// ours on its next exchange as an edit made on one side only.
|
||||
///
|
||||
/// A missing `base` is an empty one, which can only add: a device's first
|
||||
/// exchange is a union of the two libraries, never a deletion.
|
||||
pub fn merge(base: &Self, ours: &Self, theirs: &Self) -> Self {
|
||||
let mut out = Self::default();
|
||||
let names: std::collections::BTreeSet<&str> = ours.names().chain(theirs.names()).collect();
|
||||
for name in names {
|
||||
let (b, o, t) = (base.entry(name), ours.entry(name), theirs.entry(name));
|
||||
let chosen = if o == t || t == b {
|
||||
o
|
||||
} else if o == b {
|
||||
t
|
||||
} else {
|
||||
o.or(t)
|
||||
};
|
||||
if let Some((preset, unknown)) = chosen {
|
||||
out.presets.insert(name.to_string(), preset.clone());
|
||||
if let Some(lines) = unknown {
|
||||
out.unknown.insert(name.to_string(), lines.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// One name's preset together with the lines kept beside it, which are
|
||||
/// part of what that preset is when two copies are compared.
|
||||
fn entry(&self, name: &str) -> Option<(&Preset, Option<&Vec<String>>)> {
|
||||
self.presets
|
||||
.get(name)
|
||||
.map(|preset| (preset, self.unknown.get(name)))
|
||||
}
|
||||
|
||||
/// Serialise to the on-disk form.
|
||||
///
|
||||
/// Deterministic, like the sidecar's: the same library always produces the
|
||||
@@ -1479,6 +1529,97 @@ mod tests {
|
||||
assert_eq!(other.to_text(), named().to_text());
|
||||
}
|
||||
|
||||
/// A one-parameter preset, so two of them differ by their value.
|
||||
fn exposure(ev: f32) -> Preset {
|
||||
let mut params = BTreeMap::new();
|
||||
params.insert(("exposure".to_string(), "exposure".to_string()), ev);
|
||||
Preset::from_params(params)
|
||||
}
|
||||
|
||||
fn library_of(entries: &[(&str, f32)]) -> PresetLibrary {
|
||||
let mut lib = PresetLibrary::default();
|
||||
for (name, ev) in entries {
|
||||
lib.insert(name, exposure(*ev)).unwrap();
|
||||
}
|
||||
lib
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_first_merge_is_the_union_of_both_libraries() {
|
||||
let ours = library_of(&[("Mine", 1.0), ("Both", 0.5)]);
|
||||
let theirs = library_of(&[("Theirs", 2.0), ("Both", 0.5)]);
|
||||
let merged = PresetLibrary::merge(&PresetLibrary::default(), &ours, &theirs);
|
||||
assert_eq!(
|
||||
merged.names().collect::<Vec<_>>(),
|
||||
vec!["Both", "Mine", "Theirs"]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_deletion_on_either_side_is_kept_rather_than_undone() {
|
||||
// The case the base exists for: without it, the deleted preset is
|
||||
// indistinguishable from one the other side has just added.
|
||||
let base = library_of(&[("Gone here", 1.0), ("Gone there", 2.0)]);
|
||||
let ours = library_of(&[("Gone there", 2.0)]);
|
||||
let theirs = library_of(&[("Gone here", 1.0)]);
|
||||
assert!(PresetLibrary::merge(&base, &ours, &theirs).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_edit_on_one_side_reaches_the_other() {
|
||||
let base = library_of(&[("Warm", 1.0)]);
|
||||
let ours = library_of(&[("Warm", 1.0)]);
|
||||
let theirs = library_of(&[("Warm", 1.5)]);
|
||||
let merged = PresetLibrary::merge(&base, &ours, &theirs);
|
||||
assert_eq!(merged.get("Warm"), Some(&exposure(1.5)));
|
||||
// And the other way round.
|
||||
let merged = PresetLibrary::merge(&base, &theirs, &ours);
|
||||
assert_eq!(merged.get("Warm"), Some(&exposure(1.5)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_edit_outlives_a_deletion_made_elsewhere() {
|
||||
let base = library_of(&[("Warm", 1.0)]);
|
||||
let edited = library_of(&[("Warm", 1.5)]);
|
||||
let deleted = PresetLibrary::default();
|
||||
assert_eq!(
|
||||
PresetLibrary::merge(&base, &edited, &deleted).get("Warm"),
|
||||
Some(&exposure(1.5))
|
||||
);
|
||||
assert_eq!(
|
||||
PresetLibrary::merge(&base, &deleted, &edited).get("Warm"),
|
||||
Some(&exposure(1.5))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn two_different_edits_keep_ours_and_then_converge() {
|
||||
let base = library_of(&[("Warm", 1.0)]);
|
||||
let here = library_of(&[("Warm", 1.5)]);
|
||||
let there = library_of(&[("Warm", 0.5)]);
|
||||
let pushed = PresetLibrary::merge(&base, &here, &there);
|
||||
assert_eq!(pushed.get("Warm"), Some(&exposure(1.5)));
|
||||
// The other device's next exchange: its base is what it last pushed,
|
||||
// its own copy is unchanged since, and the server holds ours.
|
||||
let settled = PresetLibrary::merge(&there, &there, &pushed);
|
||||
assert_eq!(settled, pushed);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn lines_this_build_cannot_read_travel_with_their_preset() {
|
||||
let text = format!(
|
||||
"drpl {LIBRARY_FORMAT_VERSION}\n\n[preset Future]\nexposure.exposure = 0.5\n\
|
||||
something_new_entirely\n"
|
||||
);
|
||||
let theirs = PresetLibrary::parse(&text).unwrap();
|
||||
let merged = PresetLibrary::merge(
|
||||
&PresetLibrary::default(),
|
||||
&PresetLibrary::default(),
|
||||
&theirs,
|
||||
);
|
||||
assert!(merged.to_text().contains("something_new_entirely"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_neutral_preset_is_storable_and_survives_the_round_trip() {
|
||||
// The empty preset is the "clear these forty frames" action, so it has
|
||||
|
||||
+188
-49
@@ -83,35 +83,43 @@ const MAPPINGS: &[Mapping] = &[
|
||||
param: "exposure",
|
||||
convert: Convert::Direct,
|
||||
},
|
||||
// The five tone sliders do not mean the same thing in the two applications,
|
||||
// whatever their shared ±100 suggests. The factors were fitted against the
|
||||
// library's own Lightroom 6 exports and their raws (darkroom-lrfit, 2026-10):
|
||||
// each photograph's sliders carried across as `slider × factor`, one factor
|
||||
// per slider, on ~90 exports with no look applied. Contrast is ours at a
|
||||
// tenth — at −100 ours flattens a frame to grey — and our shadows need
|
||||
// nearly twice Lightroom's number. Whites barely appears in those exports;
|
||||
// every fit put it under 1 but none agreed where, so 0.5 is a hedge.
|
||||
Mapping {
|
||||
crs: "Contrast2012",
|
||||
op: "contrast",
|
||||
param: "contrast",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(0.1),
|
||||
},
|
||||
Mapping {
|
||||
crs: "Highlights2012",
|
||||
op: "highlights_shadows",
|
||||
param: "highlights",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(1.4),
|
||||
},
|
||||
Mapping {
|
||||
crs: "Shadows2012",
|
||||
op: "highlights_shadows",
|
||||
param: "shadows",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(1.9),
|
||||
},
|
||||
Mapping {
|
||||
crs: "Whites2012",
|
||||
op: "blacks_whites",
|
||||
param: "whites",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(0.5),
|
||||
},
|
||||
Mapping {
|
||||
crs: "Blacks2012",
|
||||
op: "blacks_whites",
|
||||
param: "blacks",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(1.25),
|
||||
},
|
||||
Mapping {
|
||||
crs: "Clarity2012",
|
||||
@@ -139,12 +147,21 @@ const MAPPINGS: &[Mapping] = &[
|
||||
},
|
||||
// TRACES: FR-DEV-6
|
||||
// Lightroom's HSL panel: eight bands, each ±100 for hue, saturation and
|
||||
// luminance, onto the colour mixer's twelve. Lightroom's bands sit where
|
||||
// ours do except two, matched to the nearest of ours by hue: Aqua (180°)
|
||||
// is our cyan, Purple (270°) our violet; chartreuse, spring, azure and
|
||||
// rose have no Lightroom counterpart and are left alone. One for one, as
|
||||
// a first translation — the band widths differ, and `lr-fit`'s
|
||||
// measurement against Lightroom's own output may yet scale these.
|
||||
// luminance, onto the colour mixer's twelve.
|
||||
//
|
||||
// Hue and luminance go to the band of the same hue, one for one: Aqua
|
||||
// (180°) is our cyan, Purple (270°) our violet.
|
||||
//
|
||||
// Saturation is measured. Fitted against the library's Lightroom 6 exports
|
||||
// of two looks and their raws (darkroom-lrfit, hsl_map_fit), Lightroom's
|
||||
// saturation bands are about 45° wide either side on our hue wheel, wider
|
||||
// than ours, and do not all have our strength: each is shared between
|
||||
// two or three of our bands with the factors below. Aqua sits at 187°
|
||||
// and reaches into azure, where skies are; Blue at 251° reaches violet;
|
||||
// Orange, where skin is, carries only ~0.4 — Lightroom's Orange is
|
||||
// gentle. Red, Purple and Magenta barely appear in those exports and
|
||||
// take the common gain; Green is capped where its few pixels would push
|
||||
// it further. Values add when two Lightroom bands share one of ours.
|
||||
Mapping {
|
||||
crs: "HueAdjustmentRed",
|
||||
op: "colour_mixer",
|
||||
@@ -155,7 +172,19 @@ const MAPPINGS: &[Mapping] = &[
|
||||
crs: "SaturationAdjustmentRed",
|
||||
op: "colour_mixer",
|
||||
param: "red_sat",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(1.02),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentRed",
|
||||
op: "colour_mixer",
|
||||
param: "orange_sat",
|
||||
convert: Convert::Scale(0.34),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentRed",
|
||||
op: "colour_mixer",
|
||||
param: "rose_sat",
|
||||
convert: Convert::Scale(0.34),
|
||||
},
|
||||
Mapping {
|
||||
crs: "LuminanceAdjustmentRed",
|
||||
@@ -173,7 +202,19 @@ const MAPPINGS: &[Mapping] = &[
|
||||
crs: "SaturationAdjustmentOrange",
|
||||
op: "colour_mixer",
|
||||
param: "orange_sat",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(0.39),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentOrange",
|
||||
op: "colour_mixer",
|
||||
param: "red_sat",
|
||||
convert: Convert::Scale(0.13),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentOrange",
|
||||
op: "colour_mixer",
|
||||
param: "yellow_sat",
|
||||
convert: Convert::Scale(0.13),
|
||||
},
|
||||
Mapping {
|
||||
crs: "LuminanceAdjustmentOrange",
|
||||
@@ -191,7 +232,19 @@ const MAPPINGS: &[Mapping] = &[
|
||||
crs: "SaturationAdjustmentYellow",
|
||||
op: "colour_mixer",
|
||||
param: "yellow_sat",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(0.87),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentYellow",
|
||||
op: "colour_mixer",
|
||||
param: "orange_sat",
|
||||
convert: Convert::Scale(0.29),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentYellow",
|
||||
op: "colour_mixer",
|
||||
param: "chartreuse_sat",
|
||||
convert: Convert::Scale(0.29),
|
||||
},
|
||||
Mapping {
|
||||
crs: "LuminanceAdjustmentYellow",
|
||||
@@ -209,7 +262,19 @@ const MAPPINGS: &[Mapping] = &[
|
||||
crs: "SaturationAdjustmentGreen",
|
||||
op: "colour_mixer",
|
||||
param: "green_sat",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(1.2),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentGreen",
|
||||
op: "colour_mixer",
|
||||
param: "chartreuse_sat",
|
||||
convert: Convert::Scale(0.4),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentGreen",
|
||||
op: "colour_mixer",
|
||||
param: "spring_sat",
|
||||
convert: Convert::Scale(0.4),
|
||||
},
|
||||
Mapping {
|
||||
crs: "LuminanceAdjustmentGreen",
|
||||
@@ -227,7 +292,19 @@ const MAPPINGS: &[Mapping] = &[
|
||||
crs: "SaturationAdjustmentAqua",
|
||||
op: "colour_mixer",
|
||||
param: "cyan_sat",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(0.9),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentAqua",
|
||||
op: "colour_mixer",
|
||||
param: "azure_sat",
|
||||
convert: Convert::Scale(0.51),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentAqua",
|
||||
op: "colour_mixer",
|
||||
param: "spring_sat",
|
||||
convert: Convert::Scale(0.19),
|
||||
},
|
||||
Mapping {
|
||||
crs: "LuminanceAdjustmentAqua",
|
||||
@@ -245,7 +322,19 @@ const MAPPINGS: &[Mapping] = &[
|
||||
crs: "SaturationAdjustmentBlue",
|
||||
op: "colour_mixer",
|
||||
param: "blue_sat",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(0.75),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentBlue",
|
||||
op: "colour_mixer",
|
||||
param: "violet_sat",
|
||||
convert: Convert::Scale(0.56),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentBlue",
|
||||
op: "colour_mixer",
|
||||
param: "azure_sat",
|
||||
convert: Convert::Scale(0.09),
|
||||
},
|
||||
Mapping {
|
||||
crs: "LuminanceAdjustmentBlue",
|
||||
@@ -263,7 +352,19 @@ const MAPPINGS: &[Mapping] = &[
|
||||
crs: "SaturationAdjustmentPurple",
|
||||
op: "colour_mixer",
|
||||
param: "violet_sat",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(1.26),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentPurple",
|
||||
op: "colour_mixer",
|
||||
param: "blue_sat",
|
||||
convert: Convert::Scale(0.42),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentPurple",
|
||||
op: "colour_mixer",
|
||||
param: "magenta_sat",
|
||||
convert: Convert::Scale(0.42),
|
||||
},
|
||||
Mapping {
|
||||
crs: "LuminanceAdjustmentPurple",
|
||||
@@ -281,7 +382,19 @@ const MAPPINGS: &[Mapping] = &[
|
||||
crs: "SaturationAdjustmentMagenta",
|
||||
op: "colour_mixer",
|
||||
param: "magenta_sat",
|
||||
convert: Convert::Direct,
|
||||
convert: Convert::Scale(1.05),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentMagenta",
|
||||
op: "colour_mixer",
|
||||
param: "violet_sat",
|
||||
convert: Convert::Scale(0.35),
|
||||
},
|
||||
Mapping {
|
||||
crs: "SaturationAdjustmentMagenta",
|
||||
op: "colour_mixer",
|
||||
param: "rose_sat",
|
||||
convert: Convert::Scale(0.35),
|
||||
},
|
||||
Mapping {
|
||||
crs: "LuminanceAdjustmentMagenta",
|
||||
@@ -458,7 +571,12 @@ pub fn read_xmp(text: &str) -> Result<Import, ImportError> {
|
||||
// Out-of-range values are left as they are: `EditGraph::set_param`
|
||||
// clamps when the preset is applied, and clamping here as well would
|
||||
// mean two places to be wrong about a range.
|
||||
params.insert((mapping.op.to_string(), mapping.param.to_string()), value);
|
||||
//
|
||||
// Added rather than set: one of Lightroom's HSL bands is shared
|
||||
// between two or three of ours, and two of its bands can share one.
|
||||
*params
|
||||
.entry((mapping.op.to_string(), mapping.param.to_string()))
|
||||
.or_insert(0.0) += value;
|
||||
}
|
||||
|
||||
let skipped = KNOWN_UNSUPPORTED
|
||||
@@ -561,13 +679,19 @@ mod tests {
|
||||
let import = read_embedded(&file).expect("an edit");
|
||||
let p = import.preset.params();
|
||||
let get = |op: &str, param: &str| p.get(&(op.to_string(), param.to_string())).copied();
|
||||
assert_eq!(get("colour_mixer", "blue_sat"), Some(58.0));
|
||||
assert_eq!(get("colour_mixer", "cyan_sat"), Some(50.0));
|
||||
assert_eq!(get("colour_mixer", "violet_sat"), Some(23.0));
|
||||
// Saturation is shared out by the measured factors; values add.
|
||||
let near = |got: Option<f32>, want: f32| {
|
||||
let got = got.expect("set");
|
||||
assert!((got - want).abs() < 1e-3, "{got} vs {want}");
|
||||
};
|
||||
near(get("colour_mixer", "blue_sat"), 58.0 * 0.75 + 23.0 * 0.42);
|
||||
near(get("colour_mixer", "cyan_sat"), 50.0 * 0.9);
|
||||
near(get("colour_mixer", "azure_sat"), 50.0 * 0.51 + 58.0 * 0.09);
|
||||
near(get("colour_mixer", "violet_sat"), 58.0 * 0.56 + 23.0 * 1.26);
|
||||
assert_eq!(get("colour_mixer", "red_hue"), Some(-5.0));
|
||||
assert_eq!(get("colour_mixer", "green_lum"), Some(7.0));
|
||||
assert_eq!(get("highlights_shadows", "highlights"), Some(-40.0));
|
||||
assert_eq!(get("blacks_whites", "blacks"), Some(-20.0));
|
||||
assert_eq!(get("highlights_shadows", "highlights"), Some(-40.0 * 1.4));
|
||||
assert_eq!(get("blacks_whites", "blacks"), Some(-20.0 * 1.25));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -586,9 +710,12 @@ mod tests {
|
||||
let import = read_embedded(&bytes).expect("Lightroom's edit");
|
||||
let p = import.preset.params();
|
||||
let get = |op: &str, param: &str| p.get(&(op.to_string(), param.to_string())).copied();
|
||||
assert_eq!(get("colour_mixer", "blue_sat"), Some(58.0));
|
||||
assert_eq!(get("colour_mixer", "cyan_sat"), Some(50.0));
|
||||
assert_eq!(get("highlights_shadows", "highlights"), Some(-40.0));
|
||||
// The house look's sky: Aqua and Blue land in cyan, azure and blue.
|
||||
for (param, at_least) in [("cyan_sat", 40.0), ("azure_sat", 25.0), ("blue_sat", 40.0)] {
|
||||
let v = get("colour_mixer", param).unwrap_or(0.0);
|
||||
assert!(v >= at_least, "{param} {v}");
|
||||
}
|
||||
assert_eq!(get("highlights_shadows", "highlights"), Some(-40.0 * 1.4));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -620,36 +747,48 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_two_mappings_claim_the_same_key_or_the_same_target() {
|
||||
let mut keys: Vec<&str> = MAPPINGS.iter().map(|m| m.crs).collect();
|
||||
keys.sort_unstable();
|
||||
let before = keys.len();
|
||||
keys.dedup();
|
||||
assert_eq!(before, keys.len(), "two mappings read the same crs key");
|
||||
fn no_two_mappings_repeat_a_key_and_target() {
|
||||
// A saturation band may be shared between several of ours, and two of
|
||||
// Lightroom's may share one of ours (their values add); but the same
|
||||
// key written twice to the same target would count it twice.
|
||||
let mut pairs: Vec<(&str, &str, &str)> =
|
||||
MAPPINGS.iter().map(|m| (m.crs, m.op, m.param)).collect();
|
||||
pairs.sort_unstable();
|
||||
let before = pairs.len();
|
||||
pairs.dedup();
|
||||
assert_eq!(before, pairs.len(), "a key is written twice to one target");
|
||||
|
||||
let mut targets: Vec<(&str, &str)> = MAPPINGS.iter().map(|m| (m.op, m.param)).collect();
|
||||
targets.sort_unstable();
|
||||
let before = targets.len();
|
||||
targets.dedup();
|
||||
assert_eq!(
|
||||
before,
|
||||
targets.len(),
|
||||
"two mappings write the same parameter"
|
||||
);
|
||||
// Only the HSL saturation bands are shared; every other key has one
|
||||
// home, so a slip in the table cannot fan a slider out unnoticed.
|
||||
let mut single: Vec<&str> = MAPPINGS
|
||||
.iter()
|
||||
.map(|m| m.crs)
|
||||
.filter(|k| !k.starts_with("SaturationAdjustment"))
|
||||
.collect();
|
||||
single.sort_unstable();
|
||||
let before = single.len();
|
||||
single.dedup();
|
||||
assert_eq!(before, single.len(), "two mappings read the same crs key");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_settings_that_share_a_convention_come_across_unchanged() {
|
||||
let import = read_xmp(ATTRIBUTE_FORM).unwrap();
|
||||
assert_eq!(value(&import, "exposure", "exposure"), Some(0.75));
|
||||
assert_eq!(value(&import, "contrast", "contrast"), Some(25.0));
|
||||
assert_eq!(value(&import, "contrast", "contrast"), Some(25.0 * 0.1));
|
||||
assert_eq!(
|
||||
value(&import, "highlights_shadows", "highlights"),
|
||||
Some(-40.0)
|
||||
Some(-40.0 * 1.4)
|
||||
);
|
||||
assert_eq!(
|
||||
value(&import, "highlights_shadows", "shadows"),
|
||||
Some(30.0 * 1.9)
|
||||
);
|
||||
assert_eq!(value(&import, "blacks_whites", "whites"), Some(10.0 * 0.5));
|
||||
assert_eq!(
|
||||
value(&import, "blacks_whites", "blacks"),
|
||||
Some(-15.0 * 1.25)
|
||||
);
|
||||
assert_eq!(value(&import, "highlights_shadows", "shadows"), Some(30.0));
|
||||
assert_eq!(value(&import, "blacks_whites", "whites"), Some(10.0));
|
||||
assert_eq!(value(&import, "blacks_whites", "blacks"), Some(-15.0));
|
||||
assert_eq!(value(&import, "clarity", "amount"), Some(12.0));
|
||||
assert_eq!(value(&import, "texture", "amount"), Some(8.0));
|
||||
assert_eq!(value(&import, "vibrance", "vibrance"), Some(20.0));
|
||||
@@ -690,7 +829,7 @@ mod tests {
|
||||
// does not say which shape it used.
|
||||
let import = read_xmp(ELEMENT_FORM).unwrap();
|
||||
assert_eq!(value(&import, "exposure", "exposure"), Some(0.75));
|
||||
assert_eq!(value(&import, "contrast", "contrast"), Some(25.0));
|
||||
assert_eq!(value(&import, "contrast", "contrast"), Some(25.0 * 0.1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -13,9 +13,11 @@
|
||||
use std::path::Path;
|
||||
|
||||
const MODEL: &str = "../../models/segment/yolo26n-seg.onnx";
|
||||
const QUANTISED: &str = "../../models/segment/yolo26n-seg.a16w16.onnx";
|
||||
|
||||
fn main() {
|
||||
println!("cargo:rerun-if-changed={MODEL}");
|
||||
println!("cargo:rerun-if-changed={QUANTISED}");
|
||||
println!("cargo:rerun-if-changed=build.rs");
|
||||
|
||||
// Only the embedded path needs the file present; a build without it is
|
||||
@@ -24,10 +26,18 @@ fn main() {
|
||||
return;
|
||||
}
|
||||
|
||||
let path = Path::new(MODEL);
|
||||
check(MODEL);
|
||||
// The Hexagon's quantised form rides only in an Android build.
|
||||
if std::env::var("CARGO_CFG_TARGET_OS").as_deref() == Ok("android") {
|
||||
check(QUANTISED);
|
||||
}
|
||||
}
|
||||
|
||||
fn check(model: &str) {
|
||||
let path = Path::new(model);
|
||||
let Ok(bytes) = std::fs::read(path) else {
|
||||
panic!(
|
||||
"\n\n{MODEL} is missing.\n\
|
||||
"\n\n{model} is missing.\n\
|
||||
It ships in Git LFS. Run `git lfs install && git lfs pull`, or build \
|
||||
with `--no-default-features` for a watershed-only build.\n"
|
||||
);
|
||||
@@ -40,7 +50,7 @@ fn main() {
|
||||
// happens in practice.
|
||||
if bytes.starts_with(b"version https://git-lfs") {
|
||||
panic!(
|
||||
"\n\n{MODEL} is a Git LFS pointer, not the model ({} bytes).\n\
|
||||
"\n\n{model} is a Git LFS pointer, not the model ({} bytes).\n\
|
||||
Run `git lfs install && git lfs pull` to fetch the real file.\n",
|
||||
bytes.len()
|
||||
);
|
||||
@@ -50,7 +60,7 @@ fn main() {
|
||||
// export is ~11 MB; anything under a megabyte is a truncated checkout.
|
||||
if bytes.len() < 1_000_000 {
|
||||
panic!(
|
||||
"\n\n{MODEL} is only {} bytes — expected ~11 MB.\n\
|
||||
"\n\n{model} is only {} bytes — expected several MB.\n\
|
||||
The checkout looks incomplete; try `git lfs pull`.\n",
|
||||
bytes.len()
|
||||
);
|
||||
|
||||
@@ -72,7 +72,7 @@ pub use refine::{
|
||||
#[cfg(feature = "semantic")]
|
||||
pub use scene::{Category, Scene, SceneModel};
|
||||
#[cfg(feature = "embedded-model")]
|
||||
pub use semantic::embedded_model_bytes;
|
||||
pub use semantic::embedded_models;
|
||||
#[cfg(feature = "semantic")]
|
||||
pub use semantic::{Instance, SemanticModel, SemanticOptions, Tiling};
|
||||
|
||||
|
||||
@@ -123,21 +123,29 @@ impl SceneModel {
|
||||
classes: impl AsRef<std::path::Path>,
|
||||
categories: impl AsRef<std::path::Path>,
|
||||
) -> Result<Self, SegmentError> {
|
||||
// The form the device's backend runs: the `.a16w16.onnx` sibling on
|
||||
// the Hexagon (attention left in float, inference.md §1.5), else this.
|
||||
let (model, form) =
|
||||
dr_inference_engine::resolve_model(dr_inference_engine::Role::Scene, model.as_ref());
|
||||
let bytes = std::fs::read(model).map_err(SegmentError::ModelRead)?;
|
||||
let classes = std::fs::read_to_string(classes).map_err(SegmentError::ModelRead)?;
|
||||
let categories = std::fs::read_to_string(categories).map_err(SegmentError::ModelRead)?;
|
||||
let classes = crate::semantic::parse_classes(&classes);
|
||||
let categories = parse_categories(&categories, &classes)?;
|
||||
Self::from_bytes(&bytes, categories)
|
||||
Self::from_bytes_in(&bytes, form, categories)
|
||||
}
|
||||
|
||||
pub fn from_bytes(bytes: &[u8], categories: Vec<Category>) -> Result<Self, SegmentError> {
|
||||
// f32, as for `SemanticModel`; see there.
|
||||
let session = dr_inference_engine::open(
|
||||
dr_inference_engine::Role::Scene,
|
||||
dr_inference_engine::Form::F32,
|
||||
bytes,
|
||||
)?;
|
||||
Self::from_bytes_in(bytes, dr_inference_engine::Form::F32, categories)
|
||||
}
|
||||
|
||||
/// `bytes` in a stated numeric form; the outputs keep their shape.
|
||||
pub fn from_bytes_in(
|
||||
bytes: &[u8],
|
||||
form: dr_inference_engine::Form,
|
||||
categories: Vec<Category>,
|
||||
) -> Result<Self, SegmentError> {
|
||||
let session = dr_inference_engine::open(dr_inference_engine::Role::Scene, form, bytes)?;
|
||||
|
||||
Ok(Self {
|
||||
session,
|
||||
|
||||
@@ -208,18 +208,32 @@ const EMBEDDED_MODEL: &[u8] = include_bytes!("../../../models/segment/yolo26n-se
|
||||
#[cfg(feature = "embedded-model")]
|
||||
const EMBEDDED_CLASSES: &str = include_str!("../../../models/segment/yolo26n-seg.classes.json");
|
||||
|
||||
/// The bytes of the model that ships with this crate, for whoever compiles
|
||||
/// The Hexagon's form (docs/dev/inference.md §1.5): 16-bit activations and
|
||||
/// weights, the rows' tail left in float. Only Android has a Hexagon, so only
|
||||
/// Android carries it.
|
||||
#[cfg(all(feature = "embedded-model", target_os = "android"))]
|
||||
const EMBEDDED_A16W16: &[u8] = include_bytes!("../../../models/segment/yolo26n-seg.a16w16.onnx");
|
||||
|
||||
/// Every form of the model that ships with this crate, for whoever compiles
|
||||
/// engines ahead of the first request (docs/dev/inference.md §6).
|
||||
#[cfg(feature = "embedded-model")]
|
||||
pub fn embedded_model_bytes() -> &'static [u8] {
|
||||
EMBEDDED_MODEL
|
||||
pub fn embedded_models() -> Vec<(dr_inference_engine::Form, &'static [u8])> {
|
||||
#[allow(unused_mut)]
|
||||
let mut forms = vec![(dr_inference_engine::Form::F32, EMBEDDED_MODEL)];
|
||||
#[cfg(target_os = "android")]
|
||||
forms.push((dr_inference_engine::Form::A16W16, EMBEDDED_A16W16));
|
||||
forms
|
||||
}
|
||||
|
||||
impl SemanticModel {
|
||||
/// Load the model that ships with this crate.
|
||||
/// Load the model that ships with this crate, in the form the device's
|
||||
/// backend runs.
|
||||
#[cfg(feature = "embedded-model")]
|
||||
pub fn embedded() -> Result<Self, SegmentError> {
|
||||
Self::from_bytes(EMBEDDED_MODEL, parse_classes(EMBEDDED_CLASSES))
|
||||
let forms = embedded_models();
|
||||
let (bytes, form) =
|
||||
dr_inference_engine::choose_embedded(dr_inference_engine::Role::Segmenter, &forms);
|
||||
Self::from_bytes_in(bytes, form, parse_classes(EMBEDDED_CLASSES))
|
||||
}
|
||||
|
||||
/// Load a model from an ONNX file, with `classes` supplying its vocabulary.
|
||||
@@ -236,14 +250,18 @@ impl SemanticModel {
|
||||
}
|
||||
|
||||
pub fn from_bytes(bytes: &[u8], classes: Vec<Arc<str>>) -> Result<Self, SegmentError> {
|
||||
// The f32 graph on whatever the device's backend is. An int8 form
|
||||
// for the Hexagon waits on docs/dev/inference.md §10 M7 — the mask
|
||||
// boundary has to be measured before it moves.
|
||||
let session = dr_inference_engine::open(
|
||||
dr_inference_engine::Role::Segmenter,
|
||||
dr_inference_engine::Form::F32,
|
||||
bytes,
|
||||
)?;
|
||||
Self::from_bytes_in(bytes, dr_inference_engine::Form::F32, classes)
|
||||
}
|
||||
|
||||
/// `bytes` in a stated numeric form. The quantised one keeps the same
|
||||
/// outputs (the rows' tail stays float), so decoding does not change; the
|
||||
/// masks it draws were measured against f32's (inference.md §1.5).
|
||||
pub fn from_bytes_in(
|
||||
bytes: &[u8],
|
||||
form: dr_inference_engine::Form,
|
||||
classes: Vec<Arc<str>>,
|
||||
) -> Result<Self, SegmentError> {
|
||||
let session = dr_inference_engine::open(dr_inference_engine::Role::Segmenter, form, bytes)?;
|
||||
|
||||
Ok(Self { session, classes })
|
||||
}
|
||||
|
||||
@@ -275,13 +275,26 @@ impl FaceDetector {
|
||||
}
|
||||
}
|
||||
|
||||
/// Both ids this detector writes under — the f32 form and the int8 one —
|
||||
/// for a question that is about the detector and not about which form
|
||||
/// of it a device happened to run: "has the chosen detector been over
|
||||
/// this image", asked by a re-index that must not ping-pong between a
|
||||
/// desktop that runs it in f32 and a tablet that runs it on the Hexagon.
|
||||
pub fn model_ids(self) -> [&'static str; 2] {
|
||||
[self.model_id(), self.model_id_int8()]
|
||||
/// The id when the detector runs with 16-bit activations and 8-bit
|
||||
/// weights, the Hexagon's form since the int8 one lost faces at 40–80 px
|
||||
/// (docs/dev/inference.md §1.5). Different again from both, for the same
|
||||
/// reason as [`Self::model_id_int8`]; the embedder half is unchanged.
|
||||
pub fn model_id_a16w8(self) -> &'static str {
|
||||
match self {
|
||||
FaceDetector::Scrfd500m => "scrfd_500m_a16+w600k_mbf",
|
||||
FaceDetector::Scrfd2_5g => "scrfd_2.5g_a16+w600k_mbf",
|
||||
FaceDetector::Scrfd10g => "scrfd_10g_a16+w600k_mbf",
|
||||
}
|
||||
}
|
||||
|
||||
/// Every id this detector writes under — the f32 form and each quantised
|
||||
/// one a device has run — for a question that is about the detector and
|
||||
/// not about which form of it a device happened to run: "has the chosen
|
||||
/// detector been over this image", asked by a re-index that must not
|
||||
/// ping-pong between a desktop that runs it in f32 and a tablet that
|
||||
/// runs it on the Hexagon.
|
||||
pub fn model_ids(self) -> [&'static str; 3] {
|
||||
[self.model_id(), self.model_id_int8(), self.model_id_a16w8()]
|
||||
}
|
||||
|
||||
/// The detector that writes under a pipeline id, if it is one of these.
|
||||
@@ -1769,11 +1782,16 @@ mod tests {
|
||||
#[test]
|
||||
fn a_detectors_two_spellings_share_its_embedder_and_nothing_else() {
|
||||
for d in FaceDetector::ALL {
|
||||
let [f32_id, int8_id] = d.model_ids();
|
||||
let [f32_id, int8_id, a16_id] = d.model_ids();
|
||||
assert_eq!(f32_id, d.model_id());
|
||||
assert_eq!(int8_id, d.model_id_int8());
|
||||
assert_eq!(a16_id, d.model_id_a16w8());
|
||||
assert_ne!(f32_id, int8_id);
|
||||
assert_eq!(f32_id.rsplit('+').next(), int8_id.rsplit('+').next());
|
||||
assert_ne!(f32_id, a16_id);
|
||||
assert_ne!(int8_id, a16_id);
|
||||
for q in [int8_id, a16_id] {
|
||||
assert_eq!(f32_id.rsplit('+').next(), q.rsplit('+').next());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1797,6 +1815,7 @@ mod tests {
|
||||
}
|
||||
assert_eq!(FaceDetector::for_model_id(d.model_id()), Some(d));
|
||||
assert_eq!(FaceDetector::for_model_id(d.model_id_int8()), Some(d));
|
||||
assert_eq!(FaceDetector::for_model_id(d.model_id_a16w8()), Some(d));
|
||||
}
|
||||
assert_eq!(FaceDetector::for_model_id("scrfd_10g+other"), None);
|
||||
}
|
||||
|
||||
@@ -263,10 +263,12 @@ cp "${DEX}" "${OUT}/staging/classes.dex"
|
||||
# native library directory. The build links none of it — the app dlopens
|
||||
# `libonnxruntime.so` at launch and runs on tract if it is not there — so an
|
||||
# APK without these is a slower app, not a broken one, and `RUNTIME_DIR=none`
|
||||
# builds exactly that. 174 MB for the default set; the script says which
|
||||
# Hexagon generations that buys.
|
||||
# builds exactly that. 206 MB for the default set — 32 MB of it the generic
|
||||
# WebGPU build for a phone without a Qualcomm SoC; the script says which
|
||||
# Hexagon generations the rest buys.
|
||||
if [[ "${RUNTIME_DIR}" != "none" ]]; then
|
||||
if [[ ! -f "${RUNTIME_DIR}/lib/libonnxruntime.so" ]]; then
|
||||
if [[ ! -f "${RUNTIME_DIR}/lib/libonnxruntime.so" \
|
||||
|| ! -f "${RUNTIME_DIR}/lib/libonnxruntime_generic.so" ]]; then
|
||||
"${REPO}/tools/fetch-android-runtime.sh" "${RUNTIME_DIR}"
|
||||
fi
|
||||
cp "${RUNTIME_DIR}"/lib/*.so "${OUT}/staging/lib/${ABI}/"
|
||||
@@ -321,6 +323,10 @@ for _dir in face scene inpaint denoise; do
|
||||
for f in "${ASSETS}"/*; do
|
||||
case "$(basename "${f}")" in
|
||||
README.md) continue ;;
|
||||
# The denoisers' any-size exports run whole frames on TensorRT
|
||||
# and CUDA (denoise.md §14); the Hexagon takes fixed shapes, and
|
||||
# 16 MB of graphs it never loads stay out of the APK.
|
||||
mosaic-fast.onnx | mosaic-hq.onnx) continue ;;
|
||||
esac
|
||||
cp "${f}" "${OUT}/staging/assets/models/"
|
||||
_bundled="${_bundled} $(basename "${f}")"
|
||||
|
||||
@@ -31,6 +31,9 @@ ENV DEBIAN_FRONTEND=noninteractive \
|
||||
# ---------------------------------------------------------------------------
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl git git-lfs \
|
||||
# The bundled ONNX Runtime builds arrive as wheels, which are zips
|
||||
# (tools/fetch-bundled-runtimes.sh).
|
||||
unzip \
|
||||
# A *host* C compiler as well as the cross one: build scripts and
|
||||
# proc-macros are compiled for Linux and linked with `cc`, whatever
|
||||
# the target. Without it the very first build script fails with
|
||||
|
||||
@@ -54,10 +54,14 @@ sed 's/$/\r/' "${REPO}/LICENSE" > "${STAGE}/LICENSE"
|
||||
# with its two descriptors, the panorama border filler and the denoiser. The installer
|
||||
# smoke test counts the same directories, so a model added here is expected
|
||||
# there without a number to update.
|
||||
#
|
||||
# Not the quantised siblings (`*.int8.onnx`, `*.a16w8.onnx`, `*.a16w16.onnx`):
|
||||
# they are the Hexagon's forms (docs/dev/inference.md §1.5), and a Windows
|
||||
# machine has no Hexagon to load them.
|
||||
for dir in face scene inpaint denoise; do
|
||||
for f in "${REPO}/models/${dir}"/*; do
|
||||
case "$(basename "${f}")" in
|
||||
README.md) continue ;;
|
||||
README.md | *.int8.onnx | *.a16w8.onnx | *.a16w16.onnx) continue ;;
|
||||
esac
|
||||
if [[ "${f}" == *.onnx && "$(stat -c%s "${f}")" -lt 100000 ]]; then
|
||||
echo "error: $(basename "${f}") is $(stat -c%s "${f}") bytes — an LFS pointer, not a model." >&2
|
||||
@@ -86,6 +90,13 @@ for f in "${REPO}/docs/manual/media"/*; do
|
||||
done
|
||||
echo "==> staged the manual and $(ls "${STAGE}/manual/media" | wc -l) picture(s)"
|
||||
|
||||
# The two ONNX Runtime builds the app chooses between at launch
|
||||
# (docs/dev/inference.md §3.2): Intel's OpenVINO build for an Intel GPU and
|
||||
# the WebGPU build — D3D12 — for any other, both carrying the CPU provider.
|
||||
# Beside the executable under `runtimes\`, where darkroom-desktop looks.
|
||||
"${REPO}/tools/fetch-bundled-runtimes.sh" windows "${STAGE}/runtimes"
|
||||
echo "==> staged $(ls "${STAGE}/runtimes"/* | wc -l) runtime file(s)"
|
||||
|
||||
# One installer in the output directory, the one just built. The directory
|
||||
# is cached between CI runs, so after a version bump a glob over it would find
|
||||
# two and the smoke test would hand Wine both names as one path.
|
||||
|
||||
+207
-1
@@ -331,6 +331,17 @@ If neither holds S's quality within 0.5 dB of fp32 on the real pairs, **v1 is de
|
||||
tablet shows the classical path. The sidecar still records the intent, so a desktop can render the
|
||||
learned result for a photograph edited on the tablet.
|
||||
|
||||
**Measured 2026-10-04 (inference.md §1.5): the second way holds, without the first.** The shipped
|
||||
network, with its Bayer packing re-spelled as `SpaceToDepth` so QNN can hold it (the 6-D reshape
|
||||
it replaces is exact but past the HTP's rank limit), at A16W16 — 16-bit activations and weights —
|
||||
scores within 0.00 dB of f32 at ISO 400–25600 on the tablet's own HTP, and within 0.09 dB with the
|
||||
6D's noise model scaled ×0.5, ×2 and ×4 to stand in for other sensors. A16W8 holds the 6D (worst
|
||||
−0.19 dB at ISO 25600) but not ×4 noise at 25600 (−0.52 dB), so A16W16 is what ships. int8 loses
|
||||
4.7–9.2 dB and fp16 is refused outright. A 1408 tile takes 95 ms on the Hexagon against 1510 ms on
|
||||
the tablet's CPU: about 2.3 s for a 20 MP frame. Calibration ranges come from 96 training-day
|
||||
tiles across every ISO, a third of them with that scaled noise; coverage of other bodies is that
|
||||
synthetic bracket, not their raws.
|
||||
|
||||
## 9. X-Trans
|
||||
|
||||
The requirements tie this stage to FR-RAW-5, and the library has no Fuji raws. What we can do
|
||||
@@ -418,6 +429,201 @@ measured while the GPU sat power-capped at an 810 MHz memory clock; uncapped is
|
||||
about four times faster. The Rust path reproduces the training repository's output to 2.5e-4 at
|
||||
worst; TensorRT fp16 is 75 dB from f32.
|
||||
|
||||
**Not yet:** the result is not cached across sessions (§7.1) — reopening recomputes; the tripod real
|
||||
**Not yet:** ~~the result is not cached across sessions (§7.1) — reopening recomputes;~~ done after
|
||||
0.21.0, §12; the tripod real
|
||||
pairs of §6.1; X-Trans (§9); the hand-written WGSL path, for which `export.py` already writes the
|
||||
weights blob and a manifest a shader can follow.
|
||||
|
||||
## 12. On by default, with a strength, and cached (after 0.21.0)
|
||||
|
||||
The photographer asked for the learned demosaic to be how a raw is developed, not an option found
|
||||
under Detail. So:
|
||||
|
||||
- **On by default, at full strength, on every device.** The switch is `switch_on`, so an untouched
|
||||
photograph writes nothing and is developed from the network everywhere; turning it off is the
|
||||
edit. Which hardware runs it is the inference engine's choice (inference.md), not this setting's:
|
||||
the default does not depend on what a device is believed to manage.
|
||||
- **Strength replaces Keep grain.** 0–100, default 100, and grain = 100 − strength, so it is the
|
||||
same luminance-only blend of §7.2 and moving it is one GPU pass, never a re-run. An edit saved by
|
||||
0.21.0 stored `grain`; it is still read, as its inverse, and never written.
|
||||
- **First in the panel**, above the lens corrections: it decides what every control below is
|
||||
applied to. Its attribute is still Detail, so it also stays where the Detail tab shows it.
|
||||
- **Cached on disk** (§7.1): the network's output for a file, as half floats (about 120 MB for
|
||||
20 MP — no compressor to link on Android), keyed on a SHA-256 of the file's bytes and the model
|
||||
file's name and size, oldest first past a 5 GB budget, beside the inference engine's cache under
|
||||
the data root. The strength is applied afterwards and is not in the key. A reopened photograph,
|
||||
and an export of one already developed, read it back instead of recomputing.
|
||||
|
||||
What it costs: every raw opened runs the network once, with the classical demosaic shown until the
|
||||
result lands, and a first export of an unopened raw runs it too. Every raw renders differently from
|
||||
0.21.0 unless switched off.
|
||||
|
||||
## 13. Three networks and a method (after 0.22.0)
|
||||
|
||||
The photographer asked for a choice between quality and time. `Method` replaces the Apply switch:
|
||||
`Bilinear`, `Fast`, `Medium`, `Best`, by index in that order, default `Best`. An untouched raw
|
||||
writes nothing and develops through `Best`. `apply` is still read and never written: 0 is
|
||||
`Bilinear`, 1 keeps a network already chosen or is the default. A number past the list, from a newer
|
||||
build, reads as the default. A build before this one ignores `method` and develops through its own
|
||||
network, which is the most an older peer can do.
|
||||
|
||||
**The networks** (darkroom-denoise, every one trained on the same data and noise as §11, plus 1,201
|
||||
further frames cropped from the library and 6,000 drawn scenes — polygons, lines of one to four
|
||||
photosites, text, gratings — rendered at 4× through a random affine and smooth displacement, so
|
||||
edges fall off the photosite grid):
|
||||
|
||||
| Method | File | Network | Parameters | GMAC / MP | Halo |
|
||||
|---|---|---|---|---|---|
|
||||
| Best | `mosaic-best-1408.onnx` | two U-Nets of §11's shape (a flat expert from `m2`, an edge expert from the ×100 edge-weighted run) and a 128 k-parameter gate that blends them per photosite | 6.4 M | 110 | 256 |
|
||||
| Medium | `mosaic-medium-1408.onnx` | §11's U-Net, distilled from Best (75 % its output, 25 % the truth) | 3.2 M | 48 | 192 |
|
||||
| Fast | `mosaic-fast-1408.onnx` | widths 16-32-64-128, blocks 1-1-1-2, distilled the same way | 0.93 M | 11 | 192 |
|
||||
|
||||
The gate learned on its own to trust the edge expert at 0.77–0.88 on edges and not at all on flat
|
||||
areas. The mixture's receptive field is the experts' plus the gate's, so it keeps the centre of a
|
||||
1408 tile past a 256 halo, where the single networks keep 1024 past 192. `dr_denoise::Shipped`
|
||||
carries each file's halo, and `TileNet::halo` hands it to the tiler.
|
||||
|
||||
**Quality.** PSNR after the display transform on 1,842 held-out crops, and the width of a hard
|
||||
edge on the drawn chart at ISO 6400 (truth 0.80 photosites; lower is sharper):
|
||||
|
||||
| | ISO 400 | 1600 | 6400 | 25600 | Edge width |
|
||||
|---|---|---|---|---|---|
|
||||
| §11's network | 40.61 | 39.77 | 38.43 | 36.67 | 1.77 |
|
||||
| Best | 40.69 | 39.84 | 38.47 | 36.71 | 0.82 |
|
||||
| Medium | 40.59 | 39.75 | 38.40 | 36.65 | 1.30 |
|
||||
| Fast | 39.90 | 39.06 | 37.54 | 35.33 | 1.84 |
|
||||
| Bilinear | 36.15 | 33.26 | 28.72 | 23.83 | 2.15 |
|
||||
|
||||
On photographs the three are close; on hard edges Best is half as wide as §11's network and
|
||||
Medium most of the way there. Fast costs a dB at high ISO and edges as soft as §11's.
|
||||
|
||||
**Speed**, a whole 20 MP 6D frame, the network alone, TensorRT fp16 on the laptop's RTX 3050
|
||||
(uncapped: memory at 5 GHz), engine already built: Best 2.48 s, Medium 0.79 s, Fast 0.57 s. Decode
|
||||
and the hot-pixel pass add about 0.5 s. The first build of each TensorRT engine takes 80 s (Fast) to
|
||||
190 s (Best), in the background at first launch, cached after.
|
||||
|
||||
**Before the network, two passes changed since §11.**
|
||||
|
||||
- *A noise-aware repair* (`dr_denoise::repair`) after the app's hot-pixel pass: a photosite more
|
||||
than 8σ beyond every same-colour neighbour *and* every adjacent photosite, and more than twice
|
||||
each adjacent one, is clamped to the brightest of its same-colour neighbours; a dead one, to the
|
||||
darkest. The ratio test is what spares a point of light, whose neighbours are lit too. The networks
|
||||
were trained behind the same pass (the Python and Rust agree: 935 repairs on an ISO 25600 frame).
|
||||
- *The tiler feeds the network without waiting*: tiles are gathered on every core by a producer
|
||||
thread one tile ahead, and the output is written back in parallel from the runtime's own buffer.
|
||||
0.14 s of tiler for a frame, which is what keeps Fast under a second.
|
||||
|
||||
**The Hexagon.** Each network has an `.a16w16.onnx` sibling made by `tools/quantise-models.sh
|
||||
--ranges`, the ranges from darkroom-3e's gate (96 training tiles, a third at noise ×2 and ×4). On
|
||||
the 6D gate A16W16 loses 0.00 dB for all three; with the noise scaled ×0.5–×4 at most 0.11 dB.
|
||||
A16W8 holds the gate (≤ 0.27 dB) but loses 0.63 dB on Medium at ×4, so A16W16 stays the form.
|
||||
|
||||
**Cache.** Each network keys its own results (§7.1 keys on the model's file name), and the file is
|
||||
hashed once at open, so changing the method never re-reads it. Choosing `Bilinear` keeps the
|
||||
network's result in memory for the way back; changing to another network drops it, and coming back
|
||||
reads the cache.
|
||||
|
||||
**Packaging.** All six files in the APK (`BUNDLED`, 23 entries, +44.6 MB, ~41 MB compressed); the
|
||||
three f32 networks in the Arch package and the Windows installer, which stage `models/denoise` by
|
||||
directory.
|
||||
|
||||
## 14. A whole frame, not 1408² tiles (after 0.23.0)
|
||||
|
||||
A fixed 1408² tile is exact only past its halo, and Best's halo is 256: of every 1408² it computes
|
||||
it keeps 896², 2.47 photosites of work for each one kept (Medium and Fast keep 1024², 1.89×). On a
|
||||
GPU the network can instead run over the whole frame and its reflected border in one call, which is
|
||||
exact by the same argument (§3.4) and wastes only the border.
|
||||
|
||||
**The networks** are re-exported with any height and width (`mosaic-{best,medium,fast}.onnx` beside
|
||||
the 1408 files; darkroom-denoise `tools/export_whole.py`), from the checkpoints the shipped files
|
||||
came from. The tool refuses unless each matches its 1408 file at 1408² (max |Δ| = 0 for all three),
|
||||
matches torch at 592 × 848, and equals tiled inference over the reflected frame in f64 (≤ 7e-16).
|
||||
The APK leaves them out: the Hexagon takes fixed shapes.
|
||||
|
||||
**Where they run.** `Role::WholeDenoiser` is served by TensorRT and the CUDA provider only, the rungs
|
||||
where a new input size costs nothing at run time; MIGraphX, OpenVINO and CoreML compile per shape,
|
||||
the Hexagon takes fixed shapes, and the CPU would hold gigabytes of f32 activations. Everywhere else
|
||||
`whole_frame_limit()` is `None` and the 1408² tiles run as before. TensorRT gets an optimisation
|
||||
profile up to `WHOLE_FRAME_MAX` (4608 × 3328) — without one a dynamic input compiles a new engine per
|
||||
size at run time — through the runtime's V2 options, since `ort`'s builder has none, and keeps the
|
||||
engine in a directory per model and profile (ONNX Runtime's cache key leaves the shape out).
|
||||
|
||||
**The limit is the card's memory.** TensorRT plans its memory for the profile's largest shape. A
|
||||
profile up to a whole 6D frame with Best's border (4608 × 6656) asked for 4.9–5.9 GB and would not
|
||||
build on the 6 GB RTX 3050. At 15 MP the tiler (`tile::plan`) cuts the frame into the fewest equal
|
||||
tiles under the limit: a 6D frame is two of 4160 × 3248, 27 MP of work for 20 MP kept, against 49 MP
|
||||
in 1408² tiles. If a plan's first call fails, as a GPU out of memory does, its kept centre is halved
|
||||
and the frame planned again.
|
||||
|
||||
**Measured** 2026-10-06 on `_MG_8862` (6D, ISO 8000, 20 MP), RTX 3050 Laptop, TensorRT fp16, P3 /
|
||||
5001 MHz, another session's paused training holding 1.3 GB:
|
||||
|
||||
| Best | Network time | Against the tiles |
|
||||
|---|---|---|
|
||||
| 1408² tiles | 2.60 s | — |
|
||||
| Whole frame, two 4160 × 3248 tiles | **1.37 s** | max \|Δ\| 0.0029, mean 1.1e-5 — fp16's own spread (GPU tiles against CPU tiles: 0.0025) |
|
||||
|
||||
The first build of the whole-frame engine took 28 minutes, in the background at first launch, with
|
||||
the 1408² tiles serving meanwhile — against about 3 minutes for the fixed one; the profile's range
|
||||
is what it tunes across. A cached engine loads in about a second.
|
||||
|
||||
In PyTorch fp16 the same network over the whole 20 MP frame in one call took 3.5× less than in
|
||||
tiles, so a card that holds a whole frame gains more than the 6 GB one does; `WHOLE_FRAME_MAX` is a
|
||||
constant sized for 6 GB until the limit follows the card's memory.
|
||||
|
||||
## 15. Best becomes one network (0.24)
|
||||
|
||||
The photographer's goal for 0.24 was Best's quality in under a second on the laptop. Whole frames
|
||||
(§14) took the mixture from 2.60 s to 1.37 s and no further on a 6 GB card, so the other half was
|
||||
a single network that holds the mixture's quality at a third of its work. Methods are now
|
||||
`Bilinear`, `Fast` and `Best`; Medium and the mixture are retired.
|
||||
|
||||
**The network** is `fb-combo` (darkroom-denoise, 2026-10-07): §11's shape (32-64-128-192, blocks
|
||||
1-1-2-2, 3.2 M parameters, 48 GMAC/MP, halo 192), 20 000 steps from `fb-edges2` ← `student-m`,
|
||||
taught by the mixture at a half share, with 10 % drawn scenes and 25 % crops from the edge-rich
|
||||
cells of the training frames (branch `edge-sampling`). Scored on real photographs — the chart
|
||||
overstated the mixture's lead (a chart-sharp network was softer than Medium on real edges) — on the
|
||||
validation crops in the top quarter for sharp detail:
|
||||
|
||||
| | Edge PSNR, ISO 1600 / 6400 / 25600 | Sharpness kept | Smooth areas | Held-out PSNR, ISO 400 / 1600 / 6400 / 25600 | Chart edge |
|
||||
|---|---|---|---|---|---|
|
||||
| Mixture (Best to 0.23) | 30.71 / 29.93 / 28.55 | 0.899 / 0.868 / 0.782 | 42.61 / 41.71 / 40.12 | 40.69 / 39.84 / 38.47 / 36.71 | 0.82 |
|
||||
| `fb-combo` (Best from 0.24) | 30.67 / 29.87 / 28.49 | 0.902 / 0.874 / 0.792 | 42.54 / 41.57 / 39.85 | 40.64 / 39.78 / 38.38 / 36.54 | 0.89 |
|
||||
| Medium (to 0.23) | 30.43 / 29.68 / 28.38 | 0.896 / 0.862 / 0.776 | 42.59 / 41.69 / 40.08 | 40.59 / 39.75 / 38.40 / 36.65 | 1.30 |
|
||||
|
||||
Edges within 0.04–0.06 dB and more sharpness kept at every ISO; the known shortfall is smooth areas
|
||||
at ISO 25600, 0.27 dB. The photographer took it as it stood at 20 000 of a planned 30 000 steps.
|
||||
Others tried on the way, each short of the mixture on real photographs: `fb-sharp` (drawn scenes,
|
||||
chart-sharp but Medium's real edges), `fb-edges` (half edge-rich crops: edges close, ISO 25600
|
||||
flats −0.24 dB), `fb-edges2` (a quarter: 0.03–0.14 dB short everywhere, chart 1.33–1.47), and a
|
||||
from-scratch 24-48-96-128 between Fast and Medium.
|
||||
|
||||
**Files.** `mosaic-hq-1408.onnx`, `mosaic-hq.onnx` (any size) and `mosaic-hq-1408.a16w16.onnx` for
|
||||
the Hexagon. A new name, not Medium's or Best's: the result cache keys a model by name and size,
|
||||
and this one is byte for byte Medium's size. The tablet form lost 0.00 dB in simulated QDQ at every
|
||||
ISO and at most 0.09 dB across the ×0.5–×4 noise bracket (A16W8 0.08 / 0.26 dB; int8 −10.6 dB);
|
||||
not yet confirmed on the tablet itself.
|
||||
|
||||
**Saved edits** keep their numbers: 2, which was Medium, is now Best; 3, which was Best, is past the
|
||||
end and reads as the default, Best. Both land on the new network with no migration.
|
||||
|
||||
**Measured** 2026-10-07, `_MG_8862`, RTX 3050 Laptop, TensorRT fp16, P3 / 5001 MHz, nothing else on
|
||||
the card:
|
||||
|
||||
| Best | Network time | Peak GPU memory |
|
||||
|---|---|---|
|
||||
| mixture, 1408² tiles (0.23) | 2.60 s | — |
|
||||
| mixture, whole frame (§14) | 1.37 s | — |
|
||||
| `fb-combo`, 1408² tiles | 0.95 s | 0.55 GB |
|
||||
| `fb-combo`, whole frame (two 4160 × 3248) | **0.51–0.54 s** | 1.75 GB |
|
||||
|
||||
Decode and the hot-pixel pass add 0.4–0.5 s, so a photograph is about a second end to end. Whole
|
||||
frame against tiles: max |Δ| 0.0029, 90 dB apart — fp16's spread. The whole-frame engine's first
|
||||
build took 12 minutes (the mixture's 28); from the cache it loads in about a second, so the session
|
||||
keeps the engine's ordinary 30 s idle decay rather than unloading after each photograph: at 1.75 GB
|
||||
it fits beside the develop view on a 6 GB card, and an unload would cost the next photograph a
|
||||
second.
|
||||
|
||||
The manual's close-up for Best is still the mixture's render, which this network matches to within
|
||||
the table above; it is re-recorded with the next pass of `tools/manual/record.sh`.
|
||||
|
||||
|
||||
+152
-10
@@ -127,7 +127,8 @@ Three things the table settles.
|
||||
five of six models. A whole-library face index on the tablet goes from ~100 ms + 39 ms per face
|
||||
to ~1.4 ms + 12 ms per face, and the "Thorough" detector — 3× the cost of "Fast" today — becomes
|
||||
free. Its price is that the models must be **quantised to int8**, which is an accuracy question
|
||||
§5 has to answer before it is believed.
|
||||
§5 has to answer before it is believed. (§1.5 answered it: int8 lost faces, and every model but
|
||||
XFeat ships with 16-bit activations, at about three times these timings.)
|
||||
- **The embedder does not gain from either accelerator.** 112×112 input, per-op overhead
|
||||
dominates; it is 9 ms on the tablet's CPU and 12 ms on its NPU. It stays float, which §7 turns
|
||||
from a performance footnote into a correctness rule.
|
||||
@@ -137,8 +138,101 @@ Three things the table settles.
|
||||
- **On AMD, MIGraphX fp16 is 4–17× the CPU provider** on the detectors and 60× on the
|
||||
inpainter, with the same first-run compile cost as TensorRT and no rung between it and the CPU.
|
||||
|
||||
### 1.5 The Hexagon at every bit width · 2026-10-04
|
||||
|
||||
§1.1's Hexagon column is int8 calibrated on noise: timing only. This is the follow-up — every
|
||||
model, every bit width the HTP offers, calibrated on real photographs and **scored on the tablet
|
||||
itself** (ORT 1.29 + QNN 2.42, `htp_arch` 73), against the f32 model on the same inputs. The
|
||||
"Form shipped" column is the files in `models/`, re-scored on the tablet after
|
||||
`tools/quantise-models.sh` wrote them. The
|
||||
calibration and scoring photographs are 800 from the public COCO val2017 set (CC-BY); the face
|
||||
models' numbers are over the faces in them of at least 32 px. The tools are `tools/quantise-models.sh`
|
||||
and the scratch harness described with it.
|
||||
|
||||
**What the HTP accepts.** fp16: nothing — every fp16 operator fails validation (3110), on QNN
|
||||
2.42 and 2.50, with `htp_arch` and every `soc_model` tried; the fp16 rung stays off the table until
|
||||
someone has Qualcomm's own SDK to say why. 4-bit weights (A8W4, A16W4): load, and wreck accuracy
|
||||
(SCRFD finds 25–35% of f32's faces). What is left: **A8W8 (int8), A16W8 and A16W16**, all running
|
||||
the whole graph. 16-bit activations cost about 3× int8's time, A16W16 about 4×.
|
||||
|
||||
| Model | ORT CPU f32 | Form shipped | Hexagon | On the tablet, against f32 | int8 for comparison |
|
||||
|---|---|---|---|---|---|
|
||||
| scrfd_500m / 2.5g / 10g | 17 / 56 / 198 ms | **A16W8** | 4.2 / 5.1 / 9.0 ms | 100% of faces found in every size band; keypoints 0.3–0.6% of the box | 94–95% of faces at 40–80 px |
|
||||
| 2d106det (landmarks) | 2.8 ms | **A16W8** | 0.5 ms | 0.25 px in the 192 crop (eye points 0.20) | 1.5 px, and 29 partitions at 7.3 ms |
|
||||
| yolo26n-seg | 90 ms | **A16W16**, tail in float | 12.9 ms | 98.2% of objects, mask IoU 0.994 | 74% (simulated) |
|
||||
| yolo26s-sem-ade20k | 151 ms | **A16W16**, attention in float | 15 ms | 98.9% of cells agree on the class, TV 0.009 | 67% |
|
||||
| migan-512 | 488 ms | **A16W16** | 87 ms | 41 dB from f32 in the fill (worst 1%: 30 dB) | 16 dB (simulated) |
|
||||
| xfeat-1024 / 768 | 58 ms | **int8**, rewritten graph | 6.5 ms | panorama alignment 0.45 px from f32's — f32's own refit on 90% of its matches is 0.41 | — |
|
||||
| mosaic-1408 (denoiser) | 1510 ms a tile | **A16W16**, rewritten graph | 95 ms a tile | 0.00 dB at every ISO; ≤ 0.09 dB with the noise scaled ×0.5–×4 | −4.7 to −9.2 dB |
|
||||
| arcface_mbf (embedder) | 8.5 ms | f32, CPU | — | A16W16: cosine 0.9995, p1 0.9967 — misses §7's 0.999 gate | — |
|
||||
| ocec, sgc (eyes) | 1, 1.7 ms | f32, CPU | — | sgc flips 1.45% of views even at A16W16; not worth a millisecond | — |
|
||||
|
||||
Four things the table needed that the f32 graphs did not have, all in `tools/htp_graph.py` and all
|
||||
checked exact against the f32 graph before they are used:
|
||||
|
||||
- **Rank ≤ 5.** QNN's tensors stop at rank 5, and the denoiser packs the mosaic through a 6-D
|
||||
reshape (6007 at compose). For one channel that reshape is `SpaceToDepth(2)`. XFeat's 8×8 unfold
|
||||
is 224 Slices and 6-D Concats; it is `SpaceToDepth(8)` (736 nodes to 60).
|
||||
- **No bilinear Resize at XFeat's sizes** (3110). A half-pixel bilinear resize between fixed sizes
|
||||
is two constant matrices, so it is two MatMuls.
|
||||
- **One scale per tensor.** The segmenter's output rows carry boxes in pixels beside scores in
|
||||
0..1; quantised as one tensor the scores vanish. Everything from the Concat that builds the rows
|
||||
stays float, on the CPU, where the top-300 selection costs nothing.
|
||||
- **Float where the HTP's 16-bit arithmetic drifts.** The scene model's one attention block
|
||||
(two MatMuls and a Softmax over 400 tokens) moved its agreement from 98.7% to 96.7%; it stays
|
||||
float.
|
||||
|
||||
**ORT's CPU simulation of a QDQ graph is not the tablet.** It matched to the hundredth of a dB for
|
||||
the denoiser and to rounding for the detectors, landmarks and XFeat, and it overstated MI-GAN by
|
||||
27 dB and the scene model by three points. Every number above is the device's; a new form is not
|
||||
measured until it has run there.
|
||||
|
||||
**XFeat's int8 loses keypoints and not the panorama.** 83% of f32's keypoints come back within
|
||||
1.5 px; but over the twelve-frame `fixtures/pano/2025-08-05` sweep, the homographies fitted from
|
||||
int8's matches land 0.45 px from f32's in the overlaps — the spread f32 shows against itself
|
||||
(0.41).
|
||||
|
||||
---
|
||||
|
||||
### 1.6 Intel Iris Xe, and the generic rung · 2026-10-04
|
||||
|
||||
The RTX 3050 laptop's other GPU: Raptor Lake-P's Iris Xe (96 EU), Intel's `onnxruntime-openvino`
|
||||
1.24.1 (OpenVINO 2025.4.1) and Microsoft's `onnxruntime-webgpu` 1.27.0, both PyPI wheels, through
|
||||
`ep_probe`. Three warm-ups, the median of 15 runs. Another build shared the CPU during the run, so
|
||||
the CPU columns are a little pessimistic; the GPU columns are not.
|
||||
|
||||
| Model | ORT CPU f32 | OpenVINO CPU | OpenVINO GPU f32 | **OpenVINO GPU fp16** | WebGPU (Iris Xe) |
|
||||
|---|---|---|---|---|---|
|
||||
| scrfd_500m (Fast) | 9.5 | 10.8 | 7.3 | **5.8** | 24.2 |
|
||||
| scrfd_2.5g (Balanced) | 18.5 | 16.2 | 15.5 | **11.1** | 40.8 |
|
||||
| scrfd_10g (Thorough) | 58.5 | 72.4 | 38.7 | **23.0** | 84.5 |
|
||||
| arcface_mbf (per face) | 9.5 | 11.7 | **3.0** | 2.4 | 56.6 |
|
||||
| 2d106det (landmarks) | 10.8 | 2.0 | 2.4 | **1.9** | 42.7 |
|
||||
| yolo26s-sem-ade20k | 57.0 | 47.4 | 26.0 | **16.7** | 53.0 |
|
||||
| xfeat-1024 | 23.5 | 18.5 | 19.9 | **17.5** | 34.2 |
|
||||
| migan-512 (per tile) | 330 | ✗ ¹ | 89.7 | **57.2** | 275 |
|
||||
| mosaic-fast-1408 (per tile) | 159 | 107 | 68.1 | **40.0** | 188 ² |
|
||||
| mosaic-best-1408 (per tile) | 1109 | 1670 | 947 | **604** | 1034 ² |
|
||||
|
||||
¹ OpenVINO's CPU plugin refuses the graph at initialisation. Not shipped (§3.2), so moot.
|
||||
² A later run, after `ep_probe` learned to feed the denoiser's two inputs, under heavier load: the
|
||||
CPU provider took 256 and 1034 ms in that run, so WebGPU beat it by a quarter on mosaic-fast and tied
|
||||
on mosaic-best — the only rows where it is not well behind.
|
||||
|
||||
- **OpenVINO on the Iris Xe beats ONNX Runtime's CPU provider on every model**, 1.3× on XFeat to
|
||||
5.8× on MI-GAN, with a 1–3 s compile per graph and 0.1–0.4 s from its cache. It is the Intel rung.
|
||||
fp16 is worth 1.3–1.7× over f32 here, against 1.1–1.35× on MIGraphX.
|
||||
- **Its "GPU" is OpenCL's first GPU, not Intel's.** Before `intel-compute-runtime` was installed
|
||||
the only OpenCL driver was NVIDIA's, and `device_type=GPU` ran on the RTX 3050 — slower than the
|
||||
CPU, which is the probe's to catch. Read the process's maps for `libigdrcl` before believing a
|
||||
number is the iGPU's.
|
||||
- **WebGPU is slower than the CPU on the Iris Xe** on everything but MI-GAN, as it was on the
|
||||
Adreno (§1.1), and on the RTX 3050 through Vulkan too. It is on the ladder anyway, as the generic
|
||||
rung (§2): for GPUs no vendor rung covers — an AMD card on Windows or without ROCm, a Mali — where
|
||||
it is unmeasured, and the probe's clock decides.
|
||||
- **OpenVINO's CPU plugin is not a better floor.** It wins on some graphs and loses on scrfd_10g,
|
||||
the embedder and mosaic-best, and refuses MI-GAN.
|
||||
|
||||
## 2. The shape of the answer
|
||||
|
||||
A **ladder per platform**, walked at start-up, with the first rung that builds a real session
|
||||
@@ -146,11 +240,12 @@ winning:
|
||||
|
||||
| Platform | 1st | 2nd | 3rd | Floor |
|
||||
|---|---|---|---|---|
|
||||
| Android, Qualcomm with a Hexagon the shipped QNN skel covers (V68–V81) | QNN HTP, int8 model | ORT CPU, f32 model | — | tract |
|
||||
| Android, any other SoC | ORT CPU, f32 | — | — | tract |
|
||||
| Android, Qualcomm with a Hexagon the shipped QNN skel covers (V68–V81) | QNN HTP, each model's quantised form (§1.5) | ORT CPU, f32 model | — | tract |
|
||||
| Android, any other SoC ⁶ | WebGPU (Vulkan), f32 | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, NVIDIA GPU | TensorRT, f32 model, fp16 engine | CUDA provider, f32 | ORT CPU, f32 | tract |
|
||||
| Linux, AMD GPU with ROCm | MIGraphX, f32 model, fp16 program | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, no GPU stack | ORT CPU, f32 | — | — | tract |
|
||||
| Linux / Windows, Intel GPU | OpenVINO, f32 model, fp16 program (§1.6) | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, any other GPU ⁶ | WebGPU (Vulkan / D3D12), f32 | ORT CPU, f32 | — | tract |
|
||||
| macOS ⁵ | CoreML, f32 model, ML Program | ORT CPU, f32 | — | tract |
|
||||
|
||||
⁵ **Unmeasured**, and the one exception to the rule below: nobody here has a Mac. The rung is on
|
||||
@@ -160,11 +255,17 @@ down is refused on the third launch (§4, `attempt`). The embedder stays on the
|
||||
macOS log that shows a probe line is this row's measurement; [macos.md](macos.md) says what to
|
||||
ask for.
|
||||
|
||||
⁶ **The generic rung, unmeasured where it is meant to help.** WebGPU lost to the CPU on every GPU
|
||||
it has been timed on — the Adreno, the Iris Xe, the RTX 3050 (§1.1, §1.6) — none of which it serves
|
||||
here, since each has its own rung. It is on the ladder for the GPUs that have none, on the same
|
||||
terms as CoreML: a WebGPU that is slower than the CPU is rejected by §4's clock, one that errors is
|
||||
recorded as failed. Its first measurement on an AMD card without ROCm, or a Mali, is this row's.
|
||||
|
||||
Deliberately **not** on any ladder, with the measurement that excluded each: NNAPI (no driver),
|
||||
XNNPACK (slower than CPU, aborts on SCRFD), WebGPU (slower than CPU), the Adreno through QNN (works,
|
||||
but never where the Hexagon does not also), CUDA int8 (slower than CUDA f32), the ROCm provider
|
||||
(gone: §1.3). A rung is added to this table by a measurement on this page, not by a provider
|
||||
existing.
|
||||
XNNPACK (slower than CPU, aborts on SCRFD), the Adreno through QNN (works, but never where the
|
||||
Hexagon does not also), CUDA int8 (slower than CUDA f32), the ROCm provider (gone: §1.3),
|
||||
OpenVINO's CPU plugin as a floor (§1.6). A rung is added to this table by a measurement on this
|
||||
page, not by a provider existing — the two footnoted rows are the exceptions, and say so.
|
||||
|
||||
The AMD ladder has no middle rung. TensorRT falls back to the CUDA provider while its engines
|
||||
compile; MIGraphX has no such twin, so its fallback is the CPU provider, and the minute or two of
|
||||
@@ -240,6 +341,40 @@ it.
|
||||
|
||||
Both positions are D13 territory and are recorded there (§12).
|
||||
|
||||
Two more runtimes ship in every desktop package since 0.23 (§3.2), and their parts are all
|
||||
redistributable:
|
||||
|
||||
| Component | Licence | Shipped |
|
||||
|---|---|---|
|
||||
| Intel's `onnxruntime-openvino` build, with OpenVINO 2025.4.1 and oneTBB | MIT; Apache-2.0; Apache-2.0 | Linux and Windows packages, texts beside the libraries |
|
||||
| Microsoft's WebGPU build (Dawn inside); on Windows the DirectX shader compiler | MIT; LLVM / MIT | Linux and Windows packages |
|
||||
| Microsoft's stock `onnxruntime-android` (WebGPU) | MIT | The APK, as `libonnxruntime_generic.so` |
|
||||
|
||||
### 3.2 Several runtimes, one per process
|
||||
|
||||
A runtime carries one vendor's providers: Intel's build has OpenVINO, the `onnxruntime-gpu` wheel
|
||||
CUDA and TensorRT, a ROCm build MIGraphX, Microsoft's WebGPU build the generic rung, the APK's QNN
|
||||
build the Hexagon. No prebuilt carries two vendors, and `set_api` takes one table per process.
|
||||
So a device that may hold several — the package's OpenVINO and WebGPU builds, a CUDA build the user
|
||||
fetched, the distribution's ROCm build — has to choose which to load *before* the probe, and
|
||||
cannot choose by trying.
|
||||
|
||||
`api::install` opens every runtime on the search list, asks each for `GetAvailableProviders`, and
|
||||
loads the one that scores highest against the GPUs `hardware::detect` reads from files: a vendor
|
||||
rung on its own vendor's GPU (NVIDIA driver, `/dev/kfd`, a Qualcomm SoC, macOS) above OpenVINO on
|
||||
an Intel GPU (PCI vendor `0x8086`; on Windows Intel's DCH driver package) above WebGPU above a
|
||||
CPU-only build. Equal scores keep the search order, a perfect fit ends the search — the APK's QNN
|
||||
build is listed first, so on a Qualcomm device the generic build is never opened — and
|
||||
`DARKROOM_ORT_DIR` wins outright. The losers stay mapped: unloading a C++ runtime whose static
|
||||
constructors ran is a crash at exit waiting to happen.
|
||||
|
||||
The desktop packages install the two bundled builds under `runtimes/openvino` and
|
||||
`runtimes/webgpu` beside each place a package installs to, from
|
||||
`tools/fetch-bundled-runtimes.sh` (PyPI wheels pinned by SHA-256, pruned to the native libraries:
|
||||
81 + 31 MB on Linux, 67 + 42 MB on Windows). On Windows the chosen runtime's directory is put on
|
||||
`PATH`, because Intel's build leaves OpenVINO's DLLs for the loader to find there. The Flatpak has
|
||||
no Intel OpenCL driver in its sandbox, so an Intel machine there settles on the CPU.
|
||||
|
||||
---
|
||||
|
||||
## 4. Selection — the probe, its cache, and what it may not do
|
||||
@@ -298,10 +433,10 @@ of which form they load:
|
||||
| Form | Who produces it | When | Needed by |
|
||||
|---|---|---|---|
|
||||
| f32 ONNX, shape-fixed, **opset ≥ 13** | `tools/fix-face-model-shapes.sh`, `tools/export-seg-model.sh` | Release time, once | Every rung except Hexagon |
|
||||
| int8 QDQ ONNX, per-channel, uint8 activations | `tools/quantise-models.sh` (new) | Release time, once, **calibrated on real photographs** | Hexagon |
|
||||
| QDQ ONNX, per-channel — int8, A16W8 or A16W16 per model (§1.5) | `tools/quantise-models.sh` | Release time, once, **calibrated on real photographs**, scored on the tablet | Hexagon |
|
||||
| TensorRT engine (`.engine`, per GPU architecture and TensorRT version) | The app, from the f32 file | First run on that device, in the background | TensorRT rung |
|
||||
| MIGraphX program (`.mxr`, per GPU architecture, MIGraphX version and precision) | The app, from the f32 file | First run on that device, in the background | MIGraphX rung |
|
||||
| QNN context binary | The app, from the int8 file | First run on that device, in the background | Hexagon rung |
|
||||
| QNN context binary | The app, from the quantised file | First run on that device, in the background | Hexagon rung |
|
||||
|
||||
Two rules.
|
||||
|
||||
@@ -380,6 +515,11 @@ already the rule for the detector and because §5 is the gate on whether the int
|
||||
enough to be *offered* at all. f32 on tract, ORT CPU, CUDA and TensorRT-f32 are one identity: the
|
||||
same graph, the same arithmetic, differences at the last bit.
|
||||
|
||||
After §1.5 the Hexagon runs the detectors in **A16W8**, and that is a third spelling:
|
||||
`scrfd_500m_a16+w600k_mbf` and its two siblings. Same rule, same reconciliation; a tablet that
|
||||
indexed under `_i8` keeps those rows, and `FaceDetector::model_ids` answers "has this detector been
|
||||
over this image" for all three forms.
|
||||
|
||||
**The embedder** is where comparability across devices is the whole point, and it is the one
|
||||
model that no accelerator helps (§1.4). So: **the embedder runs in f32 on every rung.** On TensorRT
|
||||
that means the embedder's engine is built without fp16 while the detector's is built with it; on
|
||||
@@ -543,6 +683,8 @@ device are comparable. *Acceptance:* M3.
|
||||
**D13 — updated.** The runtime half is reopened to the extent of §3: the Rust build stays C-free
|
||||
under `alternative-backend`; packages may install a dynamically loaded ONNX Runtime and, per §3.1,
|
||||
the Qualcomm QNN runtime; the NVIDIA libraries are not bundled. The licensing half is unchanged.
|
||||
Since 0.23 every desktop package bundles two runtimes — Intel's OpenVINO build and the WebGPU
|
||||
build — and the APK a second, generic one; the engine loads the one that fits the GPU (§3.2).
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -467,9 +467,11 @@ the tablet. Three ways to make it viable, none built:
|
||||
1. **Fill at a quarter of the resolution and upsample.** Sky and scree
|
||||
tolerate it; twenty-odd tiles, about three minutes on the desktop CPU. A
|
||||
background job with the outbox's patience, not an interactive one.
|
||||
2. **int8 on the tablet's Hexagon through QNN**, where the plain-conv design
|
||||
is the point and the whole graph should run in milliseconds. The setup
|
||||
exists from the eye-state work; MI-GAN is a candidate for the same path.
|
||||
2. **The tablet's Hexagon through QNN**, where the plain-conv design is the
|
||||
point. Measured 2026-10-04 (inference.md §1.5): int8 changes the fill
|
||||
(16 dB from f32's), so it ships with 16-bit activations and weights —
|
||||
87 ms a tile against 488 ms on the tablet's CPU, the whole graph on the
|
||||
NPU, 41 dB from f32 in the hole.
|
||||
3. **A WGSL runtime for those six operators.** A project of its own, and
|
||||
the only route that would make it interactive on the desktop.
|
||||
|
||||
|
||||
@@ -605,6 +605,12 @@ until deleted or renamed. Shipped and imported presets change only the operation
|
||||
name, so a look applied to a corrected photograph keeps the correction; a copy or a saved
|
||||
edit replaces everything in scope.
|
||||
|
||||
The user's presets sync between devices through each library they open, as one file beside the
|
||||
camera profiles. Each name merges on its own against what the last exchange left both sides
|
||||
holding, so presets added on two devices both survive, a deletion on one reaches the other rather
|
||||
than being restored by it, and an edit outlives a deletion made elsewhere. The write is
|
||||
conditional on the server's copy, so two devices exchanging at once cannot save over each other.
|
||||
|
||||
**FR-DEV-7 — Before/after.** Compare current edit state against the unedited original or against
|
||||
a chosen history state.
|
||||
|
||||
@@ -2771,7 +2777,8 @@ every render path would have to remember to call it. They travel with the decode
|
||||
matrix does.
|
||||
|
||||
*What it costs.* Every DNG with an embedded profile renders differently; previews refresh only when
|
||||
rendered again; tablet and desktop release together. The profiles directory does not sync yet.
|
||||
rendered again; tablet and desktop release together. The profiles directory syncs through the
|
||||
library's derived folder (camera-profiles.md §13).
|
||||
|
||||
### D21 — DNG reference tone for raws · **DECIDED 2026-10-03**
|
||||
|
||||
@@ -2794,7 +2801,7 @@ curve a power of 1.5/1.4 about grey (`REFERENCE_CONTRAST` is where the curve is
|
||||
brightness needs nothing: baseline exposure and the curve together land where the earlier exports do. The
|
||||
profile's look strength, vibrance and saturation bought nothing measurable on those exports. The
|
||||
user chose to change every photograph rather than keep edited ones on the old rendering. The
|
||||
fitting tools live outside the repository (`darkroom-lrfit`).
|
||||
fitting tools live outside the repository (`darkroom-lrfit`). *Amended 2026-10-04:* the look strength now defaults to 0. It scored the same at 100, 50 and 0 (held-out MSE 140, 140, 143) and the rendering is 9 % more colourful without it — the table desaturates near-neutral tones, where the default was short of those exports; the user chose more colour.
|
||||
|
||||
*Amended earlier the same day:* the default was **not** decided. The measurement below was against
|
||||
Lightroom previews of photographs carrying the user's Lightroom edits — a house look of HSL
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
# DarkRoom — Sensor health: a dated defect map per body
|
||||
|
||||
**Status:** Spike · 2026-10-04 · not built
|
||||
**Companion to:** [requirements.md](requirements.md) FR-RAW-3, [catalog.md](catalog.md)
|
||||
|
||||
A sensor gains defective photosites as it ages, and a photosite that has gone bad does not recover.
|
||||
This records, per camera body, which photosites are defective and since when, so that the library
|
||||
can show how a sensor has aged and the hot-pixel repair can fix the defects a body is known to have
|
||||
rather than only those that stand out in the frame at hand.
|
||||
|
||||
What exists is the measuring tool: `Demosaicer::find_hot_pixels` (the photosites the repair pass
|
||||
would replace, without replacing them) and `core/dr-gpu/examples/sensor_scan.rs`, which prints them
|
||||
per frame and, with `--probe`, reads a list of coordinates back out of each frame. The rest of this
|
||||
document is what a spike with them on the 6D found, and the design it argues for.
|
||||
|
||||
---
|
||||
|
||||
## 1. What is wanted
|
||||
|
||||
- **Settings → Bodies**, one entry per body, with a graph of the defective share over time in two
|
||||
series: photosites (the raw mosaic) and 2×2 cells holding at least one defective photosite (what
|
||||
reaches a pixel of the output).
|
||||
- **A dated defect map** per body, synced with the library like any other catalog data, and
|
||||
cumulative: an entry is never removed.
|
||||
- **The repair reads the map** whose date is nearest the frame's, and fixes every defect the body had
|
||||
by then, whether or not it stands out in that frame.
|
||||
- One body per model for now: the catalog stores `make model`, not a body serial.
|
||||
|
||||
## 2. What the spike found (6D, 2026-10-03)
|
||||
|
||||
53 CR2s, up to four per quarter at the highest ISO of a day, 2015 to 2026. The library holds almost
|
||||
no 6D raws from 2016–2022, so onsets in that span are dated to the span, not the year. `sensor_scan`
|
||||
ran at 0.8 s per frame, decode included.
|
||||
|
||||
**Persistence separates the sensor from the scene.** 4,179 photosites were flagged at least once;
|
||||
3,554 on one day only (stars, glints, noise). A defect is a photosite that keeps coming back.
|
||||
|
||||
**A frame that does not flag a photosite is not evidence it was clean.** The repair's test is
|
||||
relative to the neighbourhood, so a defect in a lit area does not stand out. Confirmed defects were
|
||||
flagged in a median 20 % of the frames after their first sighting. Only a frame whose neighbourhood
|
||||
at that photosite is dark counts, either way.
|
||||
|
||||
**The 6D hides some defects itself at high ISO.** (2517, 3172) reads 8,000–13,000 over neighbours
|
||||
near 200 at ISO 2000–5000, and does not stand out at all at ISO 6400–12800 (82 over 149 on
|
||||
2025-03-15). The camera appears to map out photosites it knows at those gains. So evidence for this
|
||||
body comes from ISO ≤ 5000; the cut-off must be learned per body, not fixed.
|
||||
|
||||
**Long exposures light everything.** A 9.8 s frame saturated every candidate; it confirms, it does
|
||||
not date.
|
||||
|
||||
**The curve.** Counting a photosite as defective from the first frame where it stands out, provided
|
||||
it stands out in at least 60 % of the observable frames after that (32 defects; 31 with a clean
|
||||
observable frame before onset to bracket it):
|
||||
|
||||
| Year | Defects | Share of photosites |
|
||||
|---|---|---|
|
||||
| 2015 | 2 | 0.1 ppm |
|
||||
| 2016–2021 | 2 | 0.1 ppm |
|
||||
| 2022 | 7 | 0.3 ppm |
|
||||
| 2023 | 26 | 1.3 ppm |
|
||||
| 2026 | 32 | 1.6 ppm |
|
||||
|
||||
(2517, 3172) is the shape every entry should have: clean at ISO 800–1000 in 2015 and at ISO 100–200
|
||||
in 2016, then 338 over 72 at ISO 100 on 2022-08-13 and in every comparable frame since.
|
||||
|
||||
Weak defects exist too, about twice their neighbours ((1814, 3039)); the blind repair misses them in
|
||||
most frames. A known map would catch them.
|
||||
|
||||
## 3. Design it argues for
|
||||
|
||||
- **Evidence per frame, per known photosite**: observable (dark neighbourhood, ISO inside the
|
||||
body's band) and, if so, lit or clean. Not just the frame's flagged list.
|
||||
- **A map entry is a bracket**: last clean observation, first lit observation, strength, kind. Onset
|
||||
lies between the two; the graph plots it at the first, and can show the bracket.
|
||||
- **The repair**: every entry whose first lit date is on or before the frame's capture date; for an
|
||||
entry whose bracket contains the date, probe the photosite in the frame itself.
|
||||
- **Storage and sync**: catalog tables created on first use (as `albums` does), so no schema bump
|
||||
breaks an older peer. They travel in the snapshot by default. Merge is a set union of photosites
|
||||
per body, the earlier first-lit and the later last-clean winning, which makes it commutative and
|
||||
keeps the map cumulative.
|
||||
- **Work**: a sample, not the library. Frames are picked for what they can reveal (dark, mid ISO,
|
||||
long exposures), a few per body per month, and after the first pass only new imports are read.
|
||||
+52
-52
File diff suppressed because one or more lines are too long
+2
-2
@@ -287,7 +287,6 @@ $LOCALAPPDATA\Programs\DarkRoom\
|
||||
models\
|
||||
scrfd_500m_640.onnx scrfd_2.5g_640.onnx scrfd_10g_640.onnx arcface_mbf_b1.onnx
|
||||
2d106det_b1.onnx ocec_s_b1.onnx sgc_l_48_b1.onnx
|
||||
scrfd_500m_640.int8.onnx scrfd_2.5g_640.int8.onnx scrfd_10g_640.int8.onnx
|
||||
yolo26s-sem-ade20k.onnx yolo26s-sem-ade20k.classes.json categories.txt
|
||||
migan-512.onnx
|
||||
manual\
|
||||
@@ -297,7 +296,8 @@ $LOCALAPPDATA\Programs\DarkRoom\
|
||||
```
|
||||
|
||||
Plus a Start Menu shortcut, and nothing on the desktop unless the user ticks it. The models are
|
||||
the same ten files the APK bundles and the PKGBUILD installs; `models\` beside the executable is
|
||||
the f32 files the APK bundles and the PKGBUILD installs — not the APK's quantised siblings, which
|
||||
only a Hexagon runs; `models\` beside the executable is
|
||||
where §3.2's lookup finds them. **No `LICENSE` yet**: the repository has no licence file at its
|
||||
root (the Arch package points at the system's shared GPL text), so the installer has no licence
|
||||
page until one is added — a one-file change, and the `.nsi` says where the page then goes. The face weights carry the research-only grant that
|
||||
|
||||
+33
-15
@@ -207,24 +207,42 @@ the sensor recorded.
|
||||
|
||||
### AI denoise
|
||||
|
||||
For a photograph taken in poor light at a high ISO. `AI Denoise`, in the
|
||||
Detail group, replaces how the camera's raw data is turned into colour: a
|
||||
network trained on this library's own photographs removes the noise and
|
||||
the blotches of colour that come with it, while keeping the fine detail.
|
||||
Look at it at 1:1, where noise lives.
|
||||
How every raw is developed. `AI Denoise`, at the top of the Adjust panel,
|
||||
replaces how the camera's raw data is turned into colour: a network trained
|
||||
on this library's own photographs removes the noise and the blotches of
|
||||
colour that come with it, while keeping the fine detail. Look at it at
|
||||
1:1, where noise lives.
|
||||
|
||||
Switch it on with `Apply`. The photograph keeps showing as it was while
|
||||
the network works, with its progress in the bar at the top, and changes
|
||||
when it is done — a few seconds on a computer with a graphics card, about
|
||||
fifteen on its processor alone, longer on the tablet. `Keep grain` puts
|
||||
back some of what was removed, as grain without colour, for a picture that
|
||||
does not look too smooth.
|
||||
`Method` chooses how:
|
||||
|
||||

|
||||
- `Best`, the default: clean skies and sharp lettering, edges kept as
|
||||
crisp as the camera recorded them.
|
||||
- `Fast`: a smaller network, taught the same way. Visibly noisier at very
|
||||
high ISO than `Best`, but still far cleaner than none, and quicker.
|
||||
- `Bilinear`: the camera's ordinary conversion, noise and all.
|
||||
|
||||
| Before | After |
|
||||
A photograph last edited with `Medium`, which earlier versions offered,
|
||||
opens with `Best`.
|
||||
|
||||
The photograph shows the camera's ordinary conversion while the network
|
||||
works, with its progress in the bar at the top, and changes when it is
|
||||
done — on a laptop's graphics card, about a second for a 20-megapixel
|
||||
photograph with `Best`, reading the file included; longer on a processor
|
||||
alone or on the tablet. After installing, the graphics card spends up to a
|
||||
quarter of an hour preparing each network, once, in the background; the
|
||||
photographs developed meanwhile take a little longer. The result is kept, so a photograph opened again,
|
||||
or exported, does not wait a second time, and switching back to a method
|
||||
already used is quick.
|
||||
`Strength` eases it off: below 100 % it puts back some of what was removed,
|
||||
as grain without colour, for a picture that does not look too smooth.
|
||||
|
||||
The lamp and railing of a night frame at ISO 8000, at 1:1, by each method:
|
||||
|
||||
| Bilinear | Fast |
|
||||
|---|---|
|
||||
|  |  |
|
||||
|  |  |
|
||||
| **Best** | |
|
||||
|  | |
|
||||
|
||||
It works on raw files from any camera with the usual colour pattern of
|
||||
red, green and blue squares — not on JPEGs, and not yet on Fujifilm's
|
||||
@@ -232,7 +250,7 @@ X-Trans. How noisy the camera is at each ISO was measured for the Canon
|
||||
EOS 6D; for other cameras it is read from a DNG's own figures or
|
||||
estimated from the photograph, and the finished job in the activity list
|
||||
says which. An export uses
|
||||
it whenever the photograph has it switched on.
|
||||
the method the photograph has.
|
||||
|
||||
### Moving between photographs
|
||||
|
||||
|
||||
+32
-15
@@ -289,20 +289,37 @@ drawn as hard-edged blocks rather than smoothed, so what you see is what
|
||||
the sensor recorded.</p>
|
||||
<figure><img loading="lazy" src="media/develop-zoom.gif" alt="Zooming to 1:1 with a double-click, panning, then further in with the wheel"><figcaption>Zooming to 1:1 with a double-click, panning, then further in with the wheel</figcaption></figure>
|
||||
<h3 id="ai-denoise">AI denoise</h3>
|
||||
<p>For a photograph taken in poor light at a high ISO. <code>AI Denoise</code>, in the
|
||||
Detail group, replaces how the camera's raw data is turned into colour: a
|
||||
network trained on this library's own photographs removes the noise and
|
||||
the blotches of colour that come with it, while keeping the fine detail.
|
||||
Look at it at 1:1, where noise lives.</p>
|
||||
<p>Switch it on with <code>Apply</code>. The photograph keeps showing as it was while
|
||||
the network works, with its progress in the bar at the top, and changes
|
||||
when it is done — a few seconds on a computer with a graphics card, about
|
||||
fifteen on its processor alone, longer on the tablet. <code>Keep grain</code> puts
|
||||
back some of what was removed, as grain without colour, for a picture that
|
||||
does not look too smooth.</p>
|
||||
<figure><img loading="lazy" src="media/develop-denoise.gif" alt="An ISO 8000 night frame at 1:1, AI Denoise switched on, then some grain kept"><figcaption>An ISO 8000 night frame at 1:1, AI Denoise switched on, then some grain kept</figcaption></figure>
|
||||
<table><thead><tr><th>Before</th><th>After</th></tr></thead><tbody>
|
||||
<tr><td><img src="media/develop-denoise-before.png" alt="The railing and the lamp at ISO 8000, as the camera recorded them" /></td><td><img src="media/develop-denoise-after.png" alt="The same, with AI Denoise" /></td></tr>
|
||||
<p>How every raw is developed. <code>AI Denoise</code>, at the top of the Adjust panel,
|
||||
replaces how the camera's raw data is turned into colour: a network trained
|
||||
on this library's own photographs removes the noise and the blotches of
|
||||
colour that come with it, while keeping the fine detail. Look at it at
|
||||
1:1, where noise lives.</p>
|
||||
<p><code>Method</code> chooses how:</p>
|
||||
<ul>
|
||||
<li><code>Best</code>, the default: clean skies and sharp lettering, edges kept as
|
||||
crisp as the camera recorded them.</li>
|
||||
<li><code>Fast</code>: a smaller network, taught the same way. Visibly noisier at very
|
||||
high ISO than <code>Best</code>, but still far cleaner than none, and quicker.</li>
|
||||
<li><code>Bilinear</code>: the camera's ordinary conversion, noise and all.</li>
|
||||
</ul>
|
||||
<p>A photograph last edited with <code>Medium</code>, which earlier versions offered,
|
||||
opens with <code>Best</code>.</p>
|
||||
<p>The photograph shows the camera's ordinary conversion while the network
|
||||
works, with its progress in the bar at the top, and changes when it is
|
||||
done — on a laptop's graphics card, about a second for a 20-megapixel
|
||||
photograph with <code>Best</code>, reading the file included; longer on a processor
|
||||
alone or on the tablet. After installing, the graphics card spends up to a
|
||||
quarter of an hour preparing each network, once, in the background; the
|
||||
photographs developed meanwhile take a little longer. The result is kept, so a photograph opened again,
|
||||
or exported, does not wait a second time, and switching back to a method
|
||||
already used is quick.
|
||||
<code>Strength</code> eases it off: below 100 % it puts back some of what was removed,
|
||||
as grain without colour, for a picture that does not look too smooth.</p>
|
||||
<p>The lamp and railing of a night frame at ISO 8000, at 1:1, by each method:</p>
|
||||
<table><thead><tr><th>Bilinear</th><th>Fast</th></tr></thead><tbody>
|
||||
<tr><td><img src="media/develop-denoise-bilinear.png" alt="The railing and the lamp at ISO 8000, as the camera recorded them" /></td><td><img src="media/develop-denoise-fast.png" alt="The same, with the Fast network" /></td></tr>
|
||||
<tr><td><strong>Best</strong></td><td></td></tr>
|
||||
<tr><td><img src="media/develop-denoise-best.png" alt="The same, with the Best network" /></td><td></td></tr>
|
||||
</tbody></table>
|
||||
<p>It works on raw files from any camera with the usual colour pattern of
|
||||
red, green and blue squares — not on JPEGs, and not yet on Fujifilm's
|
||||
@@ -310,7 +327,7 @@ X-Trans. How noisy the camera is at each ISO was measured for the Canon
|
||||
EOS 6D; for other cameras it is read from a DNG's own figures or
|
||||
estimated from the photograph, and the finished job in the activity list
|
||||
says which. An export uses
|
||||
it whenever the photograph has it switched on.</p>
|
||||
the method the photograph has.</p>
|
||||
<h3 id="moving-between-photographs">Moving between photographs</h3>
|
||||
<p>The roll along the foot of the canvas holds the photographs the grid was
|
||||
showing; click one to open it. The right arrow, <code>D</code> or space opens the next,
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+16
-4
@@ -14,6 +14,13 @@ Both come from `https://huggingface.co/Ultralytics/YOLO26`. The face weights in
|
||||
The keypoint weights in `keypoints/` and the border filler in `inpaint/` are
|
||||
the other two, and the easiest — see the last two sections.
|
||||
|
||||
Every quantised sibling — `*.int8.onnx`, `*.a16w8.onnx`, `*.a16w16.onnx`, the
|
||||
forms the tablet's Hexagon runs (`tools/quantise-models.sh`) — is the same
|
||||
weights rounded, and carries exactly the grant of the file it was made from.
|
||||
What calibration adds is one minimum and maximum per tensor: from public COCO
|
||||
val2017 photographs (CC-BY 4.0) for the image models, and for the denoiser from
|
||||
the same training tiles its weights were learned from. No image is in the files.
|
||||
|
||||
## The grant
|
||||
|
||||
**Ultralytics releases YOLO under AGPL-3.0**, and the weights carry the same
|
||||
@@ -127,11 +134,16 @@ declined, and the InsightFace grant of D13).
|
||||
|
||||
| File | Source | Trained on | Used by |
|
||||
|---|---|---|---|
|
||||
| `denoise/mosaic-1408.onnx` | trained from scratch in the `darkroom-denoise` repository (2026-10-03, run `m2`, 60 000 steps) | 427 of the maintainer's own base-ISO Canon EOS 6D raws, with the 6D's measured noise added | the learned demosaic and denoise (FR-DEV-3g) |
|
||||
| `denoise/mosaic-hq-1408.onnx` | trained in the `darkroom-denoise` repository (2026-10-07, run `fb-combo`, 20 000 steps, from `fb-edges2` ← `student-m`), taught by the mixture of experts that was Best until 0.24 (run `final`) at a half share, with 10 % drawn scenes and 25 % crops from the edge-rich parts of the training frames | 1,701 of the maintainer's own base-ISO raws and 6,000 synthetic scenes the repository draws itself, with the Canon EOS 6D's measured noise added | the learned demosaic and denoise, Best (FR-DEV-3g) |
|
||||
| `denoise/mosaic-fast-1408.onnx` | distilled from `final` (2026-10-04, run `student-s`, 30 000 steps, from scratch) | the same | Fast |
|
||||
| `denoise/mosaic-{hq,fast}.onnx` | the two networks above with any height and width, by `tools/export_whole.py` in the same repository from the same checkpoints; identical to the 1408 files at 1408² | the same | the same methods, a whole frame at a time on a GPU (denoise.md §14) |
|
||||
|
||||
A U-Net of plain 3×3 convolutions, ReLU, strided and transposed
|
||||
U-Nets of plain 3×3 convolutions, ReLU, strided and transposed
|
||||
convolutions and additive skips — no third-party architecture code or
|
||||
weights — at a fixed `1×1×1408×1408` for `mosaic` and `sigma`, exported by
|
||||
`python -m denoise.export` in `darkroom-denoise`. Trained only on
|
||||
weights — and, for Best, two of them blended per photosite by a small gate
|
||||
network of the same parts. Each at a fixed `1×1×1408×1408` for `mosaic`
|
||||
and `sigma`, exported by `python -m denoise.export` in `darkroom-denoise`;
|
||||
the `.a16w16.onnx` siblings are the same networks quantised for the
|
||||
Hexagon by `tools/quantise-models.sh`. Trained only on
|
||||
photographs the maintainer owns, so the weights carry no grant but the
|
||||
project's own: GPL-3.0-or-later, like the code (denoise.md §10).
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+14
-8
@@ -28,15 +28,21 @@ every reader treats "never read" as unknown, never as closed. They are found in
|
||||
directory as the pair, so a hand-placed pair does not pick up a package's eye models from a
|
||||
directory it otherwise outranks.
|
||||
|
||||
scrfd_500m_640.int8.onnx 0.8 MB the same three, in the form the Hexagon NPU takes
|
||||
scrfd_2.5g_640.int8.onnx 0.9 MB (docs/dev/inference.md §5) — opset 17, per-channel int8
|
||||
scrfd_10g_640.int8.onnx 4.3 MB weights, uint8 activations, calibrated on 96 photographs
|
||||
scrfd_500m_640.a16w8.onnx the same three, in the form the Hexagon NPU takes
|
||||
scrfd_2.5g_640.a16w8.onnx (docs/dev/inference.md §1.5) — 16-bit activations,
|
||||
scrfd_10g_640.a16w8.onnx per-channel 8-bit weights, calibrated on 300 photographs
|
||||
2d106det_b1.a16w8.onnx the landmarks, likewise
|
||||
|
||||
The int8 files are **derived** by `tools/quantise-models.sh` from the f32 ones beside them and
|
||||
travel with them: the engine loads the `.int8.onnx` sibling when the device's backend wants it and
|
||||
the canonical file otherwise, and a library indexed on the int8 form records it as a different
|
||||
detector (`scrfd_500m_i8+w600k_mbf`), because it finds a different set of faces. Every other
|
||||
platform ignores them. The embedder has no int8 form and never will (§7 of the same document).
|
||||
These are **derived** by `tools/quantise-models.sh` from the f32 files beside them and travel with
|
||||
them: the engine loads the `.a16w8.onnx` sibling when the device's backend wants it and the
|
||||
canonical file otherwise, and a library indexed on that form records it as a different detector
|
||||
(`scrfd_500m_a16+w600k_mbf`), because it finds a different set of faces. Every other platform
|
||||
ignores them, and the Windows installer leaves them out. The embedder and the eye classifiers have
|
||||
no quantised form: the embedder's vectors must compare across devices (inference.md §7), and the
|
||||
classifiers cost a millisecond on the CPU.
|
||||
|
||||
The detectors were int8 until §1.5 measured them on the tablet: int8 found 94–95% of f32's faces
|
||||
at 40–80 px, A16W8 all of them. A tablet that indexed under the int8 ids keeps those rows.
|
||||
|
||||
**A clone without git-lfs gets a ~130-byte pointer where each model should be.** Both packagers check
|
||||
for exactly that and refuse, rather than shipping the pointer and failing inside tract on the user's
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+28
-15
@@ -4,7 +4,7 @@
|
||||
# makes `makepkg -si` in this directory install what you are actually working
|
||||
# on. Swap `source` for a tagged tarball when there is something to release.
|
||||
pkgname=darkroom
|
||||
pkgver=0.21.0
|
||||
pkgver=0.24.0
|
||||
# Back to 1 with the version: a new pkgver is a new archive name, so there is
|
||||
# nothing for makepkg to reuse and nothing for a release number to disambiguate.
|
||||
pkgrel=1
|
||||
@@ -15,14 +15,17 @@ license=('GPL-3.0-or-later')
|
||||
# Runtime: Vulkan for wgpu, and a Secret Service implementation for the
|
||||
# Nextcloud credentials (FR-NC-2) — gnome-keyring or kwallet both provide it.
|
||||
depends=('vulkan-icd-loader' 'fontconfig' 'libxkbcommon')
|
||||
makedepends=('cargo' 'git')
|
||||
# ONNX Runtime is loaded from /usr/lib at launch if a package put it there
|
||||
# (docs/inference.md §3): the CPU build is 8–10× the built-in tract, the
|
||||
# ROCm build adds the MIGraphX rung on an AMD GPU. Neither is required.
|
||||
makedepends=('cargo' 'git' 'curl' 'unzip')
|
||||
# Two ONNX Runtime builds ship in /usr/lib/darkroom/runtimes — Intel's
|
||||
# OpenVINO build and the generic WebGPU one, both 8–10× the built-in tract
|
||||
# on the CPU alone — and the app opens every runtime it finds and keeps the
|
||||
# one that fits the GPU (docs/dev/inference.md §3.2). The ROCm build in
|
||||
# /usr/lib adds the MIGraphX rung on an AMD GPU and outranks both there;
|
||||
# Intel's OpenCL driver is what lets OpenVINO reach an Intel GPU.
|
||||
optdepends=('gnome-keyring: store Nextcloud credentials'
|
||||
'kwallet: store Nextcloud credentials'
|
||||
'onnxruntime-cpu: run the neural models on every core'
|
||||
'onnxruntime-rocm: run the neural models on an AMD GPU')
|
||||
'onnxruntime-rocm: run the neural models on an AMD GPU'
|
||||
'intel-compute-runtime: run the neural models on an Intel GPU')
|
||||
options=('!lto') # the workspace sets its own LTO in Cargo.toml
|
||||
|
||||
_repo="$(cd "${startdir}/.." && pwd)"
|
||||
@@ -59,6 +62,10 @@ package() {
|
||||
|
||||
install -Dm644 "README.md" "${pkgdir}/usr/share/doc/${pkgname}/README.md"
|
||||
|
||||
# The bundled runtimes, where darkroom-desktop's search finds them
|
||||
# (`runtimes/` under /usr/lib/darkroom). Pinned wheels, checked by hash.
|
||||
./tools/fetch-bundled-runtimes.sh linux "${pkgdir}/usr/lib/darkroom/runtimes"
|
||||
|
||||
# The manual: the rendered page and its pictures, where the app's Help
|
||||
# opens it (dr_ui::manual, through dr_plat::system_data_dirs). Offline by
|
||||
# design — the help sheet's "See it" links land here, on a machine that
|
||||
@@ -124,12 +131,18 @@ package() {
|
||||
fi
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/migan-512.onnx"
|
||||
|
||||
# The learned demosaic and denoise (the project's own weights, GPL —
|
||||
# models/LICENCE.md). Same pointer check, same directory.
|
||||
_src="models/denoise/mosaic-1408.onnx"
|
||||
if [[ "$(stat -c%s "${_src}")" -lt 100000 ]]; then
|
||||
echo "error: the denoise model is an LFS pointer — run: git lfs pull" >&2
|
||||
return 1
|
||||
fi
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/mosaic-1408.onnx"
|
||||
# The learned demosaic and denoise, one network per method (the project's
|
||||
# own weights, GPL — models/LICENCE.md), each at the fixed 1408 tile and
|
||||
# with any height and width for a whole frame on a GPU (denoise.md §14).
|
||||
# Same pointer check, same directory.
|
||||
for _net in fast hq; do
|
||||
for _file in "mosaic-${_net}-1408.onnx" "mosaic-${_net}.onnx"; do
|
||||
_src="models/denoise/${_file}"
|
||||
if [[ "$(stat -c%s "${_src}")" -lt 100000 ]]; then
|
||||
echo "error: ${_file} is an LFS pointer — run: git lfs pull" >&2
|
||||
return 1
|
||||
fi
|
||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/${_file}"
|
||||
done
|
||||
done
|
||||
}
|
||||
|
||||
@@ -185,6 +185,15 @@ modules:
|
||||
install -Dm644 "models/scene/$m" "/app/share/darkroom/models/$m"
|
||||
done
|
||||
|
||||
# The two runtimes the app chooses between at launch
|
||||
# (docs/dev/inference.md §3.2): Intel's OpenVINO build and the generic
|
||||
# WebGPU one, each with ONNX Runtime's CPU provider — 8–10× the
|
||||
# built-in tract on any machine. Pinned wheels, fetched over the
|
||||
# build's network. In the sandbox the OpenVINO rung finds no Intel
|
||||
# OpenCL driver and the probe settles on the CPU; WebGPU reaches the
|
||||
# GPU through the runtime's Vulkan.
|
||||
- ./tools/fetch-bundled-runtimes.sh linux /app/lib/darkroom/runtimes
|
||||
|
||||
- install -Dm644 README.md /app/share/doc/darkroom/README.md
|
||||
|
||||
sources:
|
||||
|
||||
@@ -84,6 +84,12 @@ Section "DarkRoom" SecMain
|
||||
SetOutPath "$INSTDIR\manual"
|
||||
File /r "${STAGE}\manual\*"
|
||||
|
||||
; ONNX Runtime, twice: Intel's OpenVINO build and the WebGPU build. The
|
||||
; application opens both and keeps the one that fits the GPU
|
||||
; (docs/dev/inference.md §3.2); each also runs the models on every core.
|
||||
SetOutPath "$INSTDIR\runtimes"
|
||||
File /r "${STAGE}\runtimes\*"
|
||||
|
||||
WriteUninstaller "$INSTDIR\uninstall.exe"
|
||||
|
||||
; Add/Remove Programs. HKCU, to match the per-user install.
|
||||
@@ -123,6 +129,7 @@ Section "Uninstall"
|
||||
Delete "$INSTDIR\uninstall.exe"
|
||||
RMDir /r "$INSTDIR\models"
|
||||
RMDir /r "$INSTDIR\manual"
|
||||
RMDir /r "$INSTDIR\runtimes"
|
||||
RMDir "$INSTDIR"
|
||||
|
||||
Delete "$SMPROGRAMS\${NAME}\${NAME}.lnk"
|
||||
|
||||
@@ -3,17 +3,23 @@
|
||||
#
|
||||
# ./tools/fetch-android-runtime.sh [DEST]
|
||||
#
|
||||
# Two Maven artefacts, pinned to each other by ONNX Runtime's own POM:
|
||||
# Three Maven artefacts, the first two pinned to each other by ONNX Runtime's
|
||||
# own POM:
|
||||
#
|
||||
# com.microsoft.onnxruntime:onnxruntime-android-qnn MIT
|
||||
# com.qualcomm.qti:qnn-runtime Qualcomm AI Engine Direct SDK licence
|
||||
# com.microsoft.onnxruntime:onnxruntime-android MIT
|
||||
#
|
||||
# The first is ONNX Runtime built with the CPU, QNN, XNNPACK, NNAPI and WebGPU
|
||||
# providers; the second is Qualcomm's HTP backend — the ARM-side compiler and
|
||||
# the per-generation Hexagon "skel" the DSP loads. Both ship as AARs whose
|
||||
# `jni/arm64-v8a/` is what an APK's `lib/arm64-v8a/` wants, so this script
|
||||
# unpacks exactly that and nothing else, plus the licence texts, which travel
|
||||
# with the libraries (§3.1).
|
||||
# The first is ONNX Runtime built with the CPU and QNN providers; the second
|
||||
# is Qualcomm's HTP backend — the ARM-side compiler and the per-generation
|
||||
# Hexagon "skel" the DSP loads. The third is Microsoft's stock build, which
|
||||
# carries WebGPU (Dawn on Vulkan) and the QNN build does not: the generic
|
||||
# rung for a phone without a Qualcomm SoC (§3.2). It lands as
|
||||
# `libonnxruntime_generic.so` beside the QNN build; the engine opens the QNN
|
||||
# build first, and on a Qualcomm device stops there. All three ship as AARs
|
||||
# whose `jni/arm64-v8a/` is what an APK's `lib/arm64-v8a/` wants, so this
|
||||
# script unpacks exactly that and nothing else, plus the licence texts, which
|
||||
# travel with the libraries (§3.1).
|
||||
#
|
||||
# ## What is and is not taken from the Qualcomm package
|
||||
#
|
||||
@@ -65,6 +71,8 @@ fetch com.microsoft.onnxruntime onnxruntime-android-qnn "${ORT_VERSION}" \
|
||||
"${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar"
|
||||
fetch com.qualcomm.qti qnn-runtime "${QNN_VERSION}" \
|
||||
"${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar"
|
||||
fetch com.microsoft.onnxruntime onnxruntime-android "${ORT_VERSION}" \
|
||||
"${DEST}/aar/onnxruntime-android-${ORT_VERSION}.aar"
|
||||
|
||||
# The ONNX Runtime POM names the QNN version it was built against; a pair
|
||||
# that disagrees loads and then fails at the first graph, which is the kind
|
||||
@@ -82,6 +90,8 @@ rm -rf "${DEST}/lib"
|
||||
mkdir -p "${DEST}/lib"
|
||||
unzip -q -o -j "${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar" \
|
||||
'jni/arm64-v8a/libonnxruntime.so' -d "${DEST}/lib"
|
||||
unzip -q -o -p "${DEST}/aar/onnxruntime-android-${ORT_VERSION}.aar" \
|
||||
'jni/arm64-v8a/libonnxruntime.so' > "${DEST}/lib/libonnxruntime_generic.so"
|
||||
members=(jni/arm64-v8a/libQnnHtp.so jni/arm64-v8a/libQnnHtpPrepare.so jni/arm64-v8a/libQnnSystem.so)
|
||||
for arch in ${QNN_HTP_ARCHS}; do
|
||||
members+=("jni/arm64-v8a/libQnnHtpV${arch}Skel.so" "jni/arm64-v8a/libQnnHtpV${arch}Stub.so")
|
||||
|
||||
Executable
+132
@@ -0,0 +1,132 @@
|
||||
#!/usr/bin/env bash
|
||||
# Fetch the two ONNX Runtime builds a desktop package ships
|
||||
# (docs/dev/inference.md §3.2), each into its own directory:
|
||||
#
|
||||
# DEST/openvino Intel's build: the OpenVINO rung on an Intel GPU
|
||||
# DEST/webgpu Microsoft's WebGPU build: the generic rung on any other
|
||||
#
|
||||
# ./tools/fetch-bundled-runtimes.sh {linux|windows} DEST
|
||||
#
|
||||
# Only one runtime loads per process; the app opens every one it finds and
|
||||
# keeps the one that fits the device's GPU (`dr_inference_engine::api`).
|
||||
# Both carry ONNX Runtime's CPU provider, which is the floor either way.
|
||||
# A CUDA or ROCm runtime is never bundled (§3.1) — the user's own, found
|
||||
# beside these, outranks both on its vendor's GPU.
|
||||
#
|
||||
# The wheels are PyPI's, pinned by SHA-256: the option names the engine
|
||||
# sets were read from these versions' source (CLAUDE.md, "Providers").
|
||||
# Only the native libraries are kept — not the Python bindings, not
|
||||
# OpenVINO's CPU plugin (ONNX Runtime's CPU provider is the floor), not
|
||||
# the duplicate versioned copies a wheel holds as files.
|
||||
#
|
||||
# Licences: ONNX Runtime MIT, OpenVINO Apache-2.0, oneTBB Apache-2.0, the
|
||||
# DirectX shader compiler (Windows WebGPU) LLVM/MIT; the texts go beside
|
||||
# the libraries.
|
||||
set -euo pipefail
|
||||
|
||||
PLATFORM="${1:?usage: fetch-bundled-runtimes.sh linux|windows DEST}"
|
||||
DEST="${2:?usage: fetch-bundled-runtimes.sh linux|windows DEST}"
|
||||
PYPI="https://files.pythonhosted.org/packages"
|
||||
|
||||
case "${PLATFORM}" in
|
||||
linux)
|
||||
ORT_OPENVINO="${PYPI}/08/07/f225999919f56506b603aaa3ff837ad563ab26f86906ed7fa7e5abcd849e/onnxruntime_openvino-1.24.1-cp313-cp313-manylinux_2_28_x86_64.whl"
|
||||
ORT_OPENVINO_SHA=2c3bb73e68ac27f4891af8a595c1faf574ec68b772e6583c90a0b997a1822782
|
||||
ORT_WEBGPU="${PYPI}/7a/4e/782b2457b863e1748866b82d918323e61c2d0108a01906a278e7a86a3a55/onnxruntime_webgpu-1.27.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl"
|
||||
ORT_WEBGPU_SHA=3eb30b487d2b428d2e5c1ac538177ab6336cce9b204197135dc5a16753b0258e
|
||||
# The Linux wheel carries OpenVINO itself, under the names the provider
|
||||
# was linked against.
|
||||
OPENVINO_KEEP=(libonnxruntime.so.1.24.1 libonnxruntime_providers_shared.so
|
||||
libonnxruntime_providers_openvino.so libopenvino.so.2541
|
||||
libopenvino_onnx_frontend.so.2541 libopenvino_intel_gpu_plugin.so
|
||||
libtbb.so.12 libtbbmalloc.so)
|
||||
WEBGPU_KEEP=(libonnxruntime.so.1.27.0 libonnxruntime_providers_shared.so)
|
||||
;;
|
||||
windows)
|
||||
ORT_OPENVINO="${PYPI}/3e/92/46ae2cd565961a89189900f385bb2f13a9fa731ea4674001d23720fbb1e0/onnxruntime_openvino-1.24.1-cp313-cp313-win_amd64.whl"
|
||||
ORT_OPENVINO_SHA=434bf49aa71393c577a456c9d76c98e6d6958a833fa0876793e3d5437b5a511a
|
||||
ORT_WEBGPU="${PYPI}/dd/f3/6294f9617e97035771593d604729a420a848150054cc1c422a47a4915412/onnxruntime_webgpu-1.27.0-cp313-cp313-win_amd64.whl"
|
||||
ORT_WEBGPU_SHA=c45377099fcf23ae87427eb52e2b1d35415cabb4e3419dc645f9ee08f730e9aa
|
||||
# The Windows wheel leaves OpenVINO to the `openvino` wheel; its DLLs
|
||||
# go in the same directory, which the engine puts on the DLL search
|
||||
# path when it chooses this runtime.
|
||||
OPENVINO_LIBS="${PYPI}/3c/e5/da52a86cc5f1c86871002712429cdcca0c0dbff12dfbce730b05db60340b/openvino-2025.4.1-20426-cp313-cp313-win_amd64.whl"
|
||||
OPENVINO_LIBS_SHA=a28eef35e3ed497c3238eb8f3d1ee90647c449707a8b0a7630758cd15555d8dd
|
||||
OPENVINO_KEEP=(onnxruntime.dll onnxruntime_providers_shared.dll
|
||||
onnxruntime_providers_openvino.dll openvino.dll
|
||||
openvino_onnx_frontend.dll openvino_intel_gpu_plugin.dll
|
||||
tbb12.dll tbbmalloc.dll)
|
||||
WEBGPU_KEEP=(onnxruntime.dll onnxruntime_providers_shared.dll dxcompiler.dll dxil.dll)
|
||||
;;
|
||||
*)
|
||||
echo "error: platform is linux or windows, not ${PLATFORM}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
WORK="$(mktemp -d)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
|
||||
# fetch URL SHA256 DIR: download a wheel, check it, unpack it into DIR.
|
||||
fetch() {
|
||||
local url="$1" sha="$2" dir="$3" whl
|
||||
whl="${WORK}/$(basename "${url}")"
|
||||
echo "==> $(basename "${url}")"
|
||||
curl -fsSL -o "${whl}" "${url}"
|
||||
echo "${sha} ${whl}" | sha256sum -c --quiet - || {
|
||||
echo "error: $(basename "${url}") does not match its pinned SHA-256" >&2
|
||||
exit 1
|
||||
}
|
||||
mkdir -p "${dir}"
|
||||
unzip -q -o "${whl}" -d "${dir}"
|
||||
}
|
||||
|
||||
# keep FROM TO NAMES...: copy only NAMES, every one of which must exist.
|
||||
keep() {
|
||||
local from="$1" to="$2" name
|
||||
shift 2
|
||||
mkdir -p "${to}"
|
||||
for name in "$@"; do
|
||||
local src
|
||||
src="$(find "${from}" -name "${name}" -type f | head -1)"
|
||||
[[ -n "${src}" ]] || {
|
||||
echo "error: ${name} is not in the wheel" >&2
|
||||
exit 1
|
||||
}
|
||||
install -m755 "${src}" "${to}/${name}"
|
||||
done
|
||||
}
|
||||
|
||||
fetch "${ORT_OPENVINO}" "${ORT_OPENVINO_SHA}" "${WORK}/openvino"
|
||||
if [[ -n "${OPENVINO_LIBS:-}" ]]; then
|
||||
fetch "${OPENVINO_LIBS}" "${OPENVINO_LIBS_SHA}" "${WORK}/openvino"
|
||||
fi
|
||||
fetch "${ORT_WEBGPU}" "${ORT_WEBGPU_SHA}" "${WORK}/webgpu"
|
||||
|
||||
rm -rf "${DEST}/openvino" "${DEST}/webgpu"
|
||||
keep "${WORK}/openvino" "${DEST}/openvino" "${OPENVINO_KEEP[@]}"
|
||||
keep "${WORK}/webgpu" "${DEST}/webgpu" "${WEBGPU_KEEP[@]}"
|
||||
# The wheels' own licence and notice files — ONNX Runtime keeps its in the
|
||||
# package directory, OpenVINO in `dist-info` — beside what they cover,
|
||||
# each under the name of the directory it came from: two wheels unpacked
|
||||
# into one directory both carry a `LICENSE`.
|
||||
for rt in openvino webgpu; do
|
||||
find "${WORK}/${rt}" -maxdepth 3 -type f \
|
||||
\( -iname 'LICENSE*' -o -iname 'NOTICE*' -o -iname 'ThirdPartyNotices*' \) |
|
||||
while IFS= read -r f; do
|
||||
wheel="$(basename "$(dirname "${f}")" .dist-info)"
|
||||
[[ "${wheel}" == licenses ]] && wheel="$(basename "$(dirname "$(dirname "${f}")")" .dist-info)"
|
||||
install -m644 "${f}" "${DEST}/${rt}/${wheel}.$(basename "${f}")"
|
||||
done
|
||||
done
|
||||
|
||||
# The Linux wheel bundles OpenVINO without its licence; Apache-2.0 asks
|
||||
# for the text beside the binaries. From the tag the libraries were built at.
|
||||
if [[ "${PLATFORM}" == linux ]]; then
|
||||
curl -fsSL -o "${DEST}/openvino/openvino-2025.4.1.LICENSE" \
|
||||
"https://raw.githubusercontent.com/openvinotoolkit/openvino/2025.4.1/LICENSE"
|
||||
echo "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4 ${DEST}/openvino/openvino-2025.4.1.LICENSE" |
|
||||
sha256sum -c --quiet -
|
||||
fi
|
||||
|
||||
du -sh "${DEST}/openvino" "${DEST}/webgpu"
|
||||
@@ -0,0 +1,202 @@
|
||||
"""Exact rewrites that make a graph one QNN's HTP can hold (docs/dev/inference.md §1.5).
|
||||
|
||||
Each rewrite spells the same arithmetic in operators the Hexagon runs, and
|
||||
`rewrite` checks the result against the input graph on ONNX Runtime's CPU
|
||||
before returning it — a rewrite that changes an output by more than float
|
||||
rounding is refused, not shipped.
|
||||
|
||||
- **6-D Bayer pack → SpaceToDepth(2).** The denoiser packs the mosaic with
|
||||
`Reshape(1,C,H/2,2,W/2,2) → Transpose(0,1,3,5,2,4) → Reshape(1,4C,…)`.
|
||||
QNN's tensors stop at rank 5 (error 6007 at compose); for C = 1 that
|
||||
sequence *is* SpaceToDepth.
|
||||
- **Unfold → SpaceToDepth(8).** XFeat spells its 8×8 unfold as 224 Slices,
|
||||
225 Transposes and two 6-D Concats; it is SpaceToDepth(8) of the
|
||||
normalised image, 736 nodes down to 60.
|
||||
- **Computed reshape targets → constants.** `Shape → Slice → Concat` feeding
|
||||
a Reshape, where shape inference proves the answer.
|
||||
- **Bilinear Resize → two MatMuls.** The HTP refuses `ResizeBilinear` at the
|
||||
sizes XFeat uses (3110); a half-pixel bilinear resize between fixed sizes
|
||||
is `X · Rxᵀ` then `Ry ·`, with ONNX's own edge-clamped weights.
|
||||
- **InstanceNormalization → primitives** is not here: it is exact, but the
|
||||
two full-image reductions it becomes cost the HTP more than it saves. XFeat
|
||||
keeps its normalisation in int8, which the HTP takes.
|
||||
"""
|
||||
import numpy as np
|
||||
import onnx
|
||||
import onnxruntime as ort
|
||||
from onnx import helper, numpy_helper, shape_inference
|
||||
|
||||
TOLERANCE = 1e-4 # relative to each output's largest magnitude
|
||||
|
||||
|
||||
def _prune(g):
|
||||
"""Drop nodes nobody reads and initialisers nobody uses."""
|
||||
while True:
|
||||
used = {i for n in g.node for i in n.input} | {o.name for o in g.output}
|
||||
keep = [n for n in g.node if any(o in used for o in n.output)]
|
||||
if len(keep) == len(g.node):
|
||||
break
|
||||
del g.node[:]
|
||||
g.node.extend(keep)
|
||||
used = {i for n in g.node for i in n.input}
|
||||
keep = [i for i in g.initializer if i.name in used]
|
||||
del g.initializer[:]
|
||||
g.initializer.extend(keep)
|
||||
|
||||
|
||||
def bayer_pack(m):
|
||||
g = m.graph
|
||||
prod = {o: n for n in g.node for o in n.output}
|
||||
users = {}
|
||||
for n in g.node:
|
||||
for i in n.input:
|
||||
users.setdefault(i, []).append(n)
|
||||
swaps = {}
|
||||
for n in g.node:
|
||||
if n.op_type != "Transpose":
|
||||
continue
|
||||
perm = next(a.ints for a in n.attribute if a.name == "perm")
|
||||
r1, u = prod.get(n.input[0]), users.get(n.output[0], [])
|
||||
if list(perm) != [0, 1, 3, 5, 2, 4] or not r1 or r1.op_type != "Reshape":
|
||||
continue
|
||||
if len(u) != 1 or u[0].op_type != "Reshape":
|
||||
continue
|
||||
s2d = helper.make_node("SpaceToDepth", [r1.input[0]], [u[0].output[0]], name=n.name + "_s2d", blocksize=2)
|
||||
swaps[id(r1)] = s2d
|
||||
swaps[id(n)] = swaps[id(u[0])] = None
|
||||
nodes = [swaps.get(id(n), n) for n in g.node if swaps.get(id(n), n) is not None]
|
||||
del g.node[:]
|
||||
g.node.extend(nodes)
|
||||
return m
|
||||
|
||||
|
||||
def fold_reshapes(m):
|
||||
m = shape_inference.infer_shapes(m)
|
||||
g = m.graph
|
||||
vi = {v.name: v for v in list(g.value_info) + list(g.output) + list(g.input)}
|
||||
inits = {i.name for i in g.initializer}
|
||||
dims = lambda t: [d.dim_value for d in vi[t].type.tensor_type.shape.dim] if t in vi else []
|
||||
for n in g.node:
|
||||
if n.op_type != "Reshape" or n.input[1] in inits:
|
||||
continue
|
||||
shape = dims(n.output[0])
|
||||
if not shape or 0 in shape:
|
||||
src = dims(n.input[0])
|
||||
if len(src) != 5 or 0 in src: # (1,C,k,H,W) -> (1,C·k,H,W), the denoiser's tile
|
||||
continue
|
||||
shape = [src[0], src[1] * src[2], src[3], src[4]]
|
||||
name = n.output[0] + "_shape"
|
||||
g.initializer.append(numpy_helper.from_array(np.array(shape, np.int64), name))
|
||||
n.input[1] = name
|
||||
_prune(g)
|
||||
del g.value_info[:]
|
||||
return m
|
||||
|
||||
|
||||
def unfold(m, block=8):
|
||||
"""Replace the Slice/Transpose/Concat region ending in the Reshape that
|
||||
produces the (1, block², H/block, W/block) tensor with SpaceToDepth."""
|
||||
g = m.graph
|
||||
prod = {o: n for n in g.node for o in n.output}
|
||||
region_ops = ("Slice", "Transpose", "Concat", "Unsqueeze", "Reshape")
|
||||
inits = {i.name for i in g.initializer}
|
||||
m_inf = shape_inference.infer_shapes(m)
|
||||
vi = {v.name: [d.dim_value for d in v.type.tensor_type.shape.dim] for v in m_inf.graph.value_info}
|
||||
for end in g.node:
|
||||
out = vi.get(end.output[0], [])
|
||||
if end.op_type != "Reshape" or len(out) != 4 or out[1] != block * block:
|
||||
continue
|
||||
seen, stack, leaves = set(), [end.input[0]], set()
|
||||
while stack:
|
||||
t = stack.pop()
|
||||
if t in seen or t in inits:
|
||||
continue
|
||||
seen.add(t)
|
||||
n = prod.get(t)
|
||||
if n is not None and n.op_type in region_ops:
|
||||
stack += list(n.input)
|
||||
else:
|
||||
leaves.add(t)
|
||||
if len(leaves) != 1 or len(seen) < 100: # the hand-written unfold, not an ordinary reshape
|
||||
continue
|
||||
region = {id(end)} | {id(prod[t]) for t in seen if t in prod and prod[t].op_type in region_ops}
|
||||
nodes = []
|
||||
for n in g.node:
|
||||
if id(n) in region:
|
||||
if n is end:
|
||||
nodes.append(helper.make_node("SpaceToDepth", [next(iter(leaves))], [end.output[0]],
|
||||
name=end.name + "_s2d", blocksize=block))
|
||||
continue
|
||||
nodes.append(n)
|
||||
del g.node[:]
|
||||
g.node.extend(nodes)
|
||||
_prune(g)
|
||||
del g.value_info[:]
|
||||
return m
|
||||
return m
|
||||
|
||||
|
||||
def _bilinear(n_in, n_out):
|
||||
r = np.zeros((n_out, n_in), np.float32)
|
||||
for o in range(n_out):
|
||||
x = min(max((o + 0.5) * n_in / n_out - 0.5, 0), n_in - 1)
|
||||
i0 = int(np.floor(x))
|
||||
f = x - i0
|
||||
r[o, i0] += 1 - f
|
||||
r[o, min(i0 + 1, n_in - 1)] += f
|
||||
return r
|
||||
|
||||
|
||||
def resize_matmul(m):
|
||||
"""Every linear, half-pixel Resize between fixed NCHW sizes."""
|
||||
m = shape_inference.infer_shapes(m)
|
||||
g = m.graph
|
||||
vi = {v.name: [d.dim_value for d in v.type.tensor_type.shape.dim] for v in list(g.value_info) + list(g.output)}
|
||||
nodes = []
|
||||
for n in g.node:
|
||||
a = {x.name: helper.get_attribute_value(x) for x in n.attribute}
|
||||
ok = (n.op_type == "Resize" and a.get("mode") == b"linear"
|
||||
and a.get("coordinate_transformation_mode", b"half_pixel") == b"half_pixel"
|
||||
and len(vi.get(n.input[0], [])) == 4 and len(vi.get(n.output[0], [])) == 4)
|
||||
if not ok:
|
||||
nodes.append(n)
|
||||
continue
|
||||
(_, _, h, w), (_, _, h2, w2) = vi[n.input[0]], vi[n.output[0]]
|
||||
t = n.name
|
||||
rx = numpy_helper.from_array(_bilinear(w, w2).T.copy(), t + "_rxT")
|
||||
ry = numpy_helper.from_array(_bilinear(h, h2).T.copy(), t + "_ryT")
|
||||
g.initializer.extend([rx, ry])
|
||||
nodes += [
|
||||
helper.make_node("MatMul", [n.input[0], rx.name], [t + "_w"], name=t + "_mw"),
|
||||
helper.make_node("Transpose", [t + "_w"], [t + "_t"], name=t + "_t1", perm=[0, 1, 3, 2]),
|
||||
helper.make_node("MatMul", [t + "_t", ry.name], [t + "_h"], name=t + "_mh"),
|
||||
helper.make_node("Transpose", [t + "_h"], [n.output[0]], name=t + "_t2", perm=[0, 1, 3, 2]),
|
||||
]
|
||||
del g.node[:]
|
||||
g.node.extend(nodes)
|
||||
_prune(g)
|
||||
del g.value_info[:]
|
||||
return m
|
||||
|
||||
|
||||
REWRITES = {"bayer": [bayer_pack, fold_reshapes], "unfold": [unfold], "resize": [resize_matmul]}
|
||||
|
||||
|
||||
def rewrite(path, names):
|
||||
"""The graph at `path` with the named rewrites applied, checked exact."""
|
||||
m = onnx.load(path)
|
||||
before = len(m.graph.node)
|
||||
for name in names:
|
||||
for step in REWRITES[name]:
|
||||
m = step(m)
|
||||
onnx.checker.check_model(m)
|
||||
a = ort.InferenceSession(path, providers=["CPUExecutionProvider"])
|
||||
b = ort.InferenceSession(m.SerializeToString(), providers=["CPUExecutionProvider"])
|
||||
rng = np.random.default_rng(3)
|
||||
feed = {i.name: rng.random([d if isinstance(d, int) else 1 for d in i.shape], dtype=np.float32) for i in a.get_inputs()}
|
||||
for o, x, y in zip(a.get_outputs(), a.run(None, feed), b.run(None, feed)):
|
||||
err = float(np.abs(x - y).max()) / max(float(np.abs(x).max()), 1e-6)
|
||||
if err > TOLERANCE:
|
||||
raise SystemExit(f"{path}: rewrite {names} moved output {o.name} by {err:.2e} (relative)")
|
||||
print(f" rewrites {'+'.join(names)}: {before} -> {len(m.graph.node)} nodes, exact")
|
||||
return m
|
||||
+35
-26
@@ -842,13 +842,21 @@ def develop_zoom():
|
||||
pause(1.2)
|
||||
|
||||
|
||||
@scene(media=['develop-denoise.gif', 'develop-denoise-before.png', 'develop-denoise-after.png'],
|
||||
# The methods in the order the scene visits them: `Best` is what the
|
||||
# photograph opens with, then each smaller network, then none.
|
||||
DENOISE_METHODS = ['Best', 'Fast', 'Bilinear']
|
||||
DENOISE_CLOSE_UP = 560 # pixels of canvas, square, around the lamp at 1:1
|
||||
|
||||
|
||||
@scene(media=[f'develop-denoise-{m.lower()}.png' for m in DENOISE_METHODS],
|
||||
sources=DEVELOP_SRC + ['ui/dr-ui/src/develop/denoise.rs', 'core/dr-denoise/**',
|
||||
'core/dr-gpu/src/grain.rs', 'models/denoise/**'])
|
||||
'core/dr-pipeline/src/learned_denoise.rs', 'models/denoise/**'])
|
||||
def develop_denoise():
|
||||
"""A night frame at ISO 8000 at 1:1, AI Denoise switched on and landed,
|
||||
then some grain kept. Waits for the network rather than for a fixed
|
||||
time: on the CPU it takes several times what it does on a GPU."""
|
||||
"""The lit lamp and railing of an ISO 8000 night frame at 1:1, once per
|
||||
AI Denoise method, each cut to the same square of the canvas. Waits for
|
||||
each network rather than for a fixed time: on the CPU, which the demo
|
||||
profile uses, Best takes several times what Fast does."""
|
||||
mark = log_size()
|
||||
at_develop(DENOISE)
|
||||
a = dr.photo(0.45, 0.55) # the lit lamp, the railing and the skyline over the water
|
||||
dr.move(*a)
|
||||
@@ -857,32 +865,31 @@ def develop_denoise():
|
||||
pause(1.5)
|
||||
group('Detail')
|
||||
in_column('AI Denoise@Text')
|
||||
shot('develop-denoise-before')
|
||||
rec('develop-denoise')
|
||||
pause(0.8)
|
||||
mark = log_size()
|
||||
dr.click(*denoise_switch())
|
||||
t0 = time.time()
|
||||
while time.time() - t0 < 300 and not denoise_landed(mark):
|
||||
pause(0.5)
|
||||
pause(1.5)
|
||||
shot('develop-denoise-after')
|
||||
slide('Keep grain', 60)
|
||||
pause(2.0)
|
||||
cut()
|
||||
for method in DENOISE_METHODS:
|
||||
if method != 'Best':
|
||||
mark = log_size()
|
||||
dr.click(*in_column(f'{method}@RadioButton'))
|
||||
if method != 'Bilinear':
|
||||
t0 = time.time()
|
||||
while time.time() - t0 < 600 and not denoise_landed(mark):
|
||||
pause(0.5)
|
||||
pause(1.5)
|
||||
close_up(f'develop-denoise-{method.lower()}', a)
|
||||
undo_all()
|
||||
dr.move(*a)
|
||||
dr.x('click', '--repeat', 2, '--delay', 80, 1)
|
||||
pause(1.2)
|
||||
|
||||
|
||||
def denoise_switch():
|
||||
"""The `Apply` box under the AI Denoise heading — the lens profile's
|
||||
switch is also called Apply, so it is found by where it sits."""
|
||||
head = dr.matches('AI Denoise@Text', within=column())[0]
|
||||
below = [e for e in dr.matches('Apply@CheckBox', within=column()) if e['y'] > head['y']]
|
||||
e = min(below, key=lambda e: e['y'])
|
||||
return int(e['x'] + e['w'] / 2), int(e['y'] + e['h'] / 2)
|
||||
def close_up(name, p, size=DENOISE_CLOSE_UP):
|
||||
"""A square of the canvas centred on `p`, kept inside the canvas."""
|
||||
shot(name)
|
||||
x0, y0, x1, y1 = dr.rect('id:canvas-image')
|
||||
half = size // 2
|
||||
cx = min(max(p[0], x0 + half), x1 - half)
|
||||
cy = min(max(p[1], y0 + half), y1 - half)
|
||||
subprocess.run(['mogrify', '-crop', f'{size}x{size}+{cx - half}+{cy - half}', '+repage',
|
||||
f'{OUT}/{name}.png'], check=True)
|
||||
|
||||
|
||||
def log_size():
|
||||
@@ -899,7 +906,9 @@ def denoise_landed(since):
|
||||
try:
|
||||
with open(f'{dr.HOME}/app.log', 'rb') as f:
|
||||
f.seek(since)
|
||||
return b'learned denoise:' in f.read()
|
||||
# The result's line, "learned denoise: W×H on …", not the
|
||||
# repair's, which comes first.
|
||||
return re.search(rb'learned denoise: \d+\xc3\x97', f.read()) is not None
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
+229
-112
@@ -3,143 +3,260 @@
|
||||
import glob
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import onnx
|
||||
import onnxruntime as ort
|
||||
from onnx import version_converter
|
||||
from onnxruntime.quantization import (
|
||||
CalibrationDataReader,
|
||||
CalibrationMethod,
|
||||
QuantFormat,
|
||||
QuantType,
|
||||
quantize_static,
|
||||
)
|
||||
from onnxruntime.quantization.calibrate import create_calibrator
|
||||
from onnxruntime.quantization.calibrate import save_tensors_data
|
||||
from onnxruntime.quantization.shape_inference import quant_pre_process
|
||||
from pathlib import Path
|
||||
from PIL import Image, ImageOps
|
||||
from onnxruntime.quantization import CalibrationDataReader, CalibrationMethod, QuantType, quantize_static
|
||||
from onnxruntime.quantization.calibrate import create_calibrator, save_tensors_data
|
||||
from onnxruntime.quantization.execution_providers.qnn import get_qnn_qdq_config
|
||||
from PIL import Image
|
||||
|
||||
PHOTOS = 64 # enough for a stable range; more only costs time
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import htp_graph # noqa: E402
|
||||
|
||||
MODELS = Path(__file__).resolve().parents[1] / "models"
|
||||
PHOTOS = 300 # calibration photographs; face crops and keypoints come from fewer
|
||||
CHUNK = 4 # inputs whose activations are held at once (scrfd_10g: ~1 GB each)
|
||||
|
||||
Q = QuantType
|
||||
FORMS = {"int8": (Q.QUInt8, Q.QInt8), "a16w8": (Q.QUInt16, Q.QInt8), "a16w16": (Q.QUInt16, Q.QInt16)}
|
||||
|
||||
# Per model: where it lives, the form `Rung::form` gives its role, the exact
|
||||
# rewrites its graph needs, and nodes that stay float on the CPU because one
|
||||
# scale cannot serve the tensor (the segmenter's rows: boxes in pixels beside
|
||||
# scores in 0..1) or because the HTP's 16-bit arithmetic drifts there (the
|
||||
# scene model's attention). Measured, inference.md §1.5.
|
||||
TABLE = {
|
||||
"scrfd_500m_640": dict(dir="face", form="a16w8", feed="scrfd"),
|
||||
"scrfd_2.5g_640": dict(dir="face", form="a16w8", feed="scrfd"),
|
||||
"scrfd_10g_640": dict(dir="face", form="a16w8", feed="scrfd"),
|
||||
"2d106det_b1": dict(dir="face", form="a16w8", feed="landmarks"),
|
||||
"yolo26n-seg": dict(dir="segment", form="a16w16", feed="yolo", float_from="/model.23/Concat_4"),
|
||||
"yolo26s-sem-ade20k": dict(dir="scene", form="a16w16", feed="yolo",
|
||||
float_nodes=["/model.10/m/m.0/attn/MatMul", "/model.10/m/m.0/attn/Softmax",
|
||||
"/model.10/m/m.0/attn/MatMul_1"]),
|
||||
"migan-512": dict(dir="inpaint", form="a16w16", feed="migan"),
|
||||
"xfeat-1024": dict(dir="keypoints", form="int8", feed="xfeat", rewrites=["unfold", "resize"]),
|
||||
"xfeat-768": dict(dir="keypoints", form="int8", feed="xfeat", rewrites=["unfold", "resize"]),
|
||||
"mosaic-fast-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
# Best since 0.24: one network, exported as Fast is, so the same rewrite.
|
||||
"mosaic-hq-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]),
|
||||
}
|
||||
|
||||
|
||||
def letterbox(img, edge, pad, norm):
|
||||
"""The app's Letterbox::sample: fit the long side to `edge`, centre, pad."""
|
||||
img = ImageOps.exif_transpose(img).convert("RGB")
|
||||
w, h = img.size
|
||||
scale = edge / max(w, h)
|
||||
nw, nh = max(1, round(w * scale)), max(1, round(h * scale))
|
||||
img = img.resize((nw, nh), Image.BILINEAR)
|
||||
canvas = Image.new("RGB", (edge, edge), (pad, pad, pad))
|
||||
canvas.paste(img, ((edge - nw) // 2, (edge - nh) // 2))
|
||||
x = np.asarray(canvas, dtype=np.float32) # HWC, 0..255
|
||||
x = norm(x)
|
||||
return np.ascontiguousarray(x.transpose(2, 0, 1))[None] # NCHW
|
||||
# ---- the app's samplers (dr-face Letterbox, dr-segment Letterbox::sample, align.rs) ----
|
||||
|
||||
def load(p):
|
||||
return np.asarray(Image.open(p).convert("RGB"), np.float32) / 255.0
|
||||
|
||||
|
||||
def preprocessing(name, shape):
|
||||
"""Which normalisation this model is fed in the app.
|
||||
|
||||
SCRFD (`dr-face::detect`): `(v - 127.5) / 128`, padded with 114.
|
||||
ArcFace (`dr-face::embed`): the same, on an aligned 112 crop — a
|
||||
letterboxed photograph is the wrong distribution, but the embedder is
|
||||
never quantised (§7), so this is only ever a fallback.
|
||||
YOLO (`dr-segment`): `v / 255`, padded with 0.5.
|
||||
"""
|
||||
edge = shape[-1]
|
||||
if name.startswith("scrfd") or name.startswith("arcface"):
|
||||
return edge, 114, lambda x: (x - 127.5) / 128.0
|
||||
return edge, 128, lambda x: x / 255.0
|
||||
def bilinear(img, sx, sy):
|
||||
h, w = img.shape[:2]
|
||||
x0, y0 = np.floor(sx).astype(np.int64), np.floor(sy).astype(np.int64)
|
||||
fx, fy = (sx - x0)[..., None], (sy - y0)[..., None]
|
||||
c = lambda a, n: np.clip(a, 0, n - 1) # noqa: E731 - neighbours clamp at the edge
|
||||
top = img[c(y0, h), c(x0, w)] * (1 - fx) + img[c(y0, h), c(x0 + 1, w)] * fx
|
||||
bot = img[c(y0 + 1, h), c(x0, w)] * (1 - fx) + img[c(y0 + 1, h), c(x0 + 1, w)] * fx
|
||||
return top * (1 - fy) + bot * fy
|
||||
|
||||
|
||||
class Photos(CalibrationDataReader):
|
||||
"""One chunk of photographs, fed as the app would feed them."""
|
||||
def chw(x):
|
||||
return np.ascontiguousarray(x.transpose(2, 0, 1))[None].astype(np.float32)
|
||||
|
||||
def __init__(self, input_name, paths, edge, pad, norm):
|
||||
self.name = input_name
|
||||
self.items = iter(letterbox(Image.open(p), edge, pad, norm) for p in paths)
|
||||
|
||||
def letterbox(img, edge, yolo):
|
||||
h, w = img.shape[:2]
|
||||
s = min(edge / w, edge / h)
|
||||
px, py = (edge - w * s) / 2, (edge - h * s) / 2
|
||||
ix, iy = np.meshgrid(np.arange(edge) + 0.5, np.arange(edge) + 0.5)
|
||||
if yolo: # semantic.rs: no -0.5, pad 0.5, 0..1
|
||||
sx, sy = (ix - px) / s, (iy - py) / s
|
||||
out = bilinear(img, sx, sy)
|
||||
out[(sx < 0) | (sx >= w) | (sy < 0) | (sy >= h)] = 0.5
|
||||
else: # detect.rs: -0.5, pad 114, (v·255 − 127.5)/128
|
||||
sx, sy = (ix - px) / s - 0.5, (iy - py) / s - 0.5
|
||||
out = (bilinear(img, sx, sy) * 255 - 127.5) / 128
|
||||
out[(sx < -0.5) | (sx > w - 0.5) | (sy < -0.5) | (sy > h - 0.5)] = (114 - 127.5) / 128
|
||||
return chw(out), (s, px, py)
|
||||
|
||||
|
||||
def crop_box(img, x0, y0, bw, bh, ow, oh):
|
||||
"""align.rs crop_box: output (u+.5) → source, −0.5, bilinear, outside black."""
|
||||
u, v = np.meshgrid(np.arange(ow) + 0.5, np.arange(oh) + 0.5)
|
||||
sx, sy = x0 + u * bw / ow - 0.5, y0 + v * bh / oh - 0.5
|
||||
out = bilinear(img, sx, sy)
|
||||
h, w = img.shape[:2]
|
||||
out[(sx < -1) | (sx > w) | (sy < -1) | (sy > h)] = 0
|
||||
return out
|
||||
|
||||
|
||||
def scrfd_boxes(outs, s, px, py):
|
||||
"""detect.rs decode: score ≥ 0.5, greedy NMS at 0.4, min side 24 px."""
|
||||
fmc = len(outs) // 3
|
||||
boxes, scores = [], []
|
||||
for i, st in enumerate([8, 16, 32, 64][:fmc]):
|
||||
sc, bx = outs[i].reshape(-1), outs[fmc + i].reshape(-1, 4)
|
||||
idx = np.nonzero(sc >= 0.5)[0]
|
||||
cell = idx // 2
|
||||
cx, cy = (cell % (640 // st)) * st, (cell // (640 // st)) * st
|
||||
boxes.append(np.stack([cx - bx[idx, 0] * st, cy - bx[idx, 1] * st, cx + bx[idx, 2] * st, cy + bx[idx, 3] * st], 1))
|
||||
scores.append(sc[idx])
|
||||
b, sc = np.concatenate(boxes), np.concatenate(scores)
|
||||
keep = []
|
||||
for i in np.argsort(-sc):
|
||||
x0 = np.maximum(b[i, 0], b[keep, 0]); y0 = np.maximum(b[i, 1], b[keep, 1])
|
||||
x1 = np.minimum(b[i, 2], b[keep, 2]); y1 = np.minimum(b[i, 3], b[keep, 3])
|
||||
inter = np.clip(x1 - x0, 0, None) * np.clip(y1 - y0, 0, None)
|
||||
area = lambda r: (r[..., 2] - r[..., 0]) * (r[..., 3] - r[..., 1]) # noqa: E731
|
||||
if not keep or (inter / (area(b[i]) + area(b[keep]) - inter)).max() <= 0.4:
|
||||
keep.append(i)
|
||||
b = (b[keep] - [px, py, px, py]) / s
|
||||
return b[np.minimum(b[:, 2] - b[:, 0], b[:, 3] - b[:, 1]) >= 32]
|
||||
|
||||
|
||||
# ---- one calibration input per photograph (or per face), as the app makes it ----
|
||||
|
||||
def feeds(kind, photos, model):
|
||||
name = model.get_inputs()[0].name
|
||||
if kind == "scrfd":
|
||||
for p in photos:
|
||||
yield {name: letterbox(load(p), 640, False)[0]}
|
||||
elif kind == "landmarks": # landmarks.rs: 1.5× the box, square, 0..255
|
||||
det = ort.InferenceSession(str(MODELS / "face/scrfd_10g_640.onnx"), providers=["CPUExecutionProvider"])
|
||||
for p in photos:
|
||||
img = load(p)
|
||||
x, ctx = letterbox(img, 640, False)
|
||||
for b in scrfd_boxes(det.run(None, {"input.1": x}), *ctx):
|
||||
cx, cy, side = (b[0] + b[2]) / 2, (b[1] + b[3]) / 2, 1.5 * max(b[2] - b[0], b[3] - b[1])
|
||||
yield {name: chw(crop_box(img, cx - side / 2, cy - side / 2, side, side, 192, 192) * 255)}
|
||||
elif kind == "yolo":
|
||||
for p in photos:
|
||||
yield {name: letterbox(load(p), 640, True)[0]}
|
||||
elif kind == "migan": # migan.rs: ch0 = known − 0.5, ch1–3 = (rgb·2 − 1)·known
|
||||
rng = np.random.default_rng(7)
|
||||
for p in photos:
|
||||
img = load(p)
|
||||
h, w = img.shape[:2]
|
||||
e = min(h, w)
|
||||
sq = Image.fromarray((img[(h - e) // 2:(h + e) // 2, (w - e) // 2:(w + e) // 2] * 255).astype(np.uint8))
|
||||
img = np.asarray(sq.resize((512, 512), Image.BILINEAR), np.float32) / 255
|
||||
known = np.ones((512, 512), np.float32)
|
||||
for _ in range(rng.integers(1, 3)): # a panorama's unknown border: a wedge along one edge
|
||||
side = rng.integers(4)
|
||||
depth = np.linspace(rng.integers(20, 110), rng.integers(20, 110), 512).astype(int)
|
||||
edge = np.arange(512)[:, None] < depth[None, :] # [depth, along]: inside the wedge
|
||||
wedge = edge if side % 2 == 0 else edge[::-1] # top / bottom of a column
|
||||
known[wedge if side < 2 else wedge.T] = 0 # or left / right of a row
|
||||
x = np.concatenate([(known - 0.5)[None], ((img * 2 - 1) * known[..., None]).transpose(2, 0, 1)])[None]
|
||||
yield {name: x.astype(np.float32)}
|
||||
elif kind == "xfeat": # xfeat.rs: grey 0..1, shrink to fit, top-left, zero pad
|
||||
_, _, H, W = [d if isinstance(d, int) else 1 for d in model.get_inputs()[0].shape]
|
||||
for p in photos:
|
||||
g = load(p).mean(2) # a display-rendered photograph is already the app's (R+G+B)/3 ^ 1/2.2
|
||||
h, w = g.shape
|
||||
s = min(W / w, H / h, 1.0)
|
||||
if s < 1:
|
||||
g = np.asarray(Image.fromarray(g).resize((round(w * s), round(h * s)), Image.BOX))
|
||||
pad = np.zeros((H, W), np.float32)
|
||||
pad[:g.shape[0], :g.shape[1]] = g
|
||||
yield {name: pad[None, None]}
|
||||
|
||||
|
||||
class Items(CalibrationDataReader):
|
||||
def __init__(self, items):
|
||||
self.it = iter(items)
|
||||
|
||||
def get_next(self):
|
||||
x = next(self.items, None)
|
||||
return None if x is None else {self.name: x}
|
||||
return next(self.it, None)
|
||||
|
||||
|
||||
# Photographs whose activations are held in memory at once. Every ONNX
|
||||
# Runtime calibrator keeps each image's whole set of activations until it
|
||||
# folds them into a range — a gigabyte an image on the 10g detector at 640²,
|
||||
# and folded once at the end, an OOM kill with no message. Folding every
|
||||
# `CHUNK` images gives ranges identical to folding once (checked on
|
||||
# scrfd_500m, 129 tensors, no difference) at a bounded cost.
|
||||
CHUNK = 4
|
||||
def calibrate(path, items, cache):
|
||||
"""Min/max ranges in chunks — every ORT calibrator holds all activations
|
||||
until it folds them, and the others measurably degrade the result."""
|
||||
cal = create_calibrator(Path(path), None, augmented_model_path=f"{path}.aug.onnx",
|
||||
calibrate_method=CalibrationMethod.MinMax)
|
||||
batch, n = [], 0
|
||||
for item in items:
|
||||
batch.append(item)
|
||||
n += 1
|
||||
if len(batch) == CHUNK:
|
||||
cal.collect_data(Items(batch))
|
||||
batch = []
|
||||
if batch:
|
||||
cal.collect_data(Items(batch))
|
||||
save_tensors_data(cal.compute_data(), cache)
|
||||
os.remove(f"{path}.aug.onnx")
|
||||
return n
|
||||
|
||||
|
||||
def calibrate(pre, name, photos, cache):
|
||||
"""Min/max ranges over `photos`, written to `cache` for quantize_static.
|
||||
|
||||
Plain min/max: the moving average and the strided option of
|
||||
`quantize_static` both measured worse than this on held-out proxies, and
|
||||
the percentile method has no memory bound at all.
|
||||
"""
|
||||
import onnxruntime as ort
|
||||
|
||||
s = ort.InferenceSession(pre, providers=["CPUExecutionProvider"])
|
||||
i = s.get_inputs()[0]
|
||||
shape = [d if isinstance(d, int) else 1 for d in i.shape]
|
||||
edge, pad, norm = preprocessing(name, shape)
|
||||
calibrator = create_calibrator(
|
||||
Path(pre),
|
||||
None,
|
||||
augmented_model_path=f"{pre}.augmented.onnx",
|
||||
calibrate_method=CalibrationMethod.MinMax,
|
||||
)
|
||||
for start in range(0, len(photos), CHUNK):
|
||||
calibrator.collect_data(Photos(i.name, photos[start : start + CHUNK], edge, pad, norm))
|
||||
ranges = calibrator.compute_data()
|
||||
save_tensors_data(ranges, cache)
|
||||
os.remove(f"{pre}.augmented.onnx")
|
||||
def downstream(m, start):
|
||||
names, live = set(), set()
|
||||
for n in m.graph.node:
|
||||
if n.name == start or any(i in live for i in n.input):
|
||||
names.add(n.name)
|
||||
live.update(n.output)
|
||||
return sorted(names)
|
||||
|
||||
|
||||
def main():
|
||||
photo_dir, models = sys.argv[1], sys.argv[2:]
|
||||
photos = sorted(
|
||||
p
|
||||
for ext in ("jpg", "jpeg", "JPG", "JPEG", "png")
|
||||
for p in glob.glob(os.path.join(photo_dir, "**", f"*.{ext}"), recursive=True)
|
||||
)[:PHOTOS]
|
||||
if len(photos) < 20:
|
||||
sys.exit(f"only {len(photos)} photographs under {photo_dir}; calibration wants dozens")
|
||||
print(f"==> calibrating on {len(photos)} photographs")
|
||||
args = sys.argv[1:]
|
||||
ranges = None
|
||||
if args[:1] == ["--ranges"]:
|
||||
ranges, args = args[1], args[2:]
|
||||
photo_dir = None
|
||||
else:
|
||||
photo_dir, args = args[0], args[1:]
|
||||
names = args or [n for n in TABLE if TABLE[n]["feed"]]
|
||||
photos = []
|
||||
if photo_dir:
|
||||
photos = sorted(p for e in ("jpg", "jpeg", "JPG", "JPEG", "png")
|
||||
for p in glob.glob(os.path.join(photo_dir, "**", f"*.{e}"), recursive=True))[:PHOTOS]
|
||||
if len(photos) < 50:
|
||||
sys.exit(f"only {len(photos)} photographs under {photo_dir}; calibration wants hundreds")
|
||||
print(f"==> calibrating on {len(photos)} photographs")
|
||||
|
||||
for src in models:
|
||||
stem, _ = os.path.splitext(src)
|
||||
name = os.path.basename(src)
|
||||
out = f"{stem}.int8.onnx"
|
||||
for stem in names:
|
||||
spec = TABLE[stem]
|
||||
src = MODELS / spec["dir"] / f"{stem}.onnx"
|
||||
out = src.with_name(f"{stem}.{spec['form']}.onnx")
|
||||
work = src.with_name(f"{stem}.quant-work.onnx")
|
||||
print(f" {stem} -> {out.name}")
|
||||
m = onnx.load(src)
|
||||
opset = next((o.version for o in m.opset_import if o.domain in ("", "ai.onnx")), 0)
|
||||
work = f"{stem}.quant-work.onnx"
|
||||
if opset < 13:
|
||||
print(f" {name}: opset {opset} -> 17")
|
||||
m = version_converter.convert_version(m, 17)
|
||||
if next(o.version for o in m.opset_import if o.domain in ("", "ai.onnx")) < 13:
|
||||
m = version_converter.convert_version(m, 17) # per-channel QDQ needs 13
|
||||
m.ir_version = 8
|
||||
onnx.save(m, work)
|
||||
pre = f"{stem}.quant-pre.onnx"
|
||||
quant_pre_process(work, pre)
|
||||
cache = f"{stem}.quant-ranges.json"
|
||||
calibrate(pre, name, photos, cache)
|
||||
quantize_static(
|
||||
pre,
|
||||
out,
|
||||
None,
|
||||
quant_format=QuantFormat.QDQ,
|
||||
per_channel=True,
|
||||
activation_type=QuantType.QUInt8,
|
||||
weight_type=QuantType.QInt8,
|
||||
calibrate_method=CalibrationMethod.MinMax,
|
||||
calibration_cache_path=cache,
|
||||
)
|
||||
for f in (work, pre, cache):
|
||||
os.remove(f)
|
||||
print(f" {out}: {os.path.getsize(out) // 1024} KB")
|
||||
if spec.get("rewrites"):
|
||||
onnx.save(htp_graph.rewrite(str(work), spec["rewrites"]), work)
|
||||
model = ort.InferenceSession(str(work), providers=["CPUExecutionProvider"])
|
||||
cache = str(work) + ".ranges"
|
||||
if spec["feed"] is None:
|
||||
if not ranges:
|
||||
sys.exit(f"{stem} is calibrated on mosaics, not photographs: pass --ranges")
|
||||
cache = ranges
|
||||
else:
|
||||
n = calibrate(str(work), feeds(spec["feed"], photos, model), cache)
|
||||
print(f" {n} calibration inputs")
|
||||
act, wt = FORMS[spec["form"]]
|
||||
# The config needs a reader only to exist; the ranges come from `cache`.
|
||||
zeros = {i.name: np.zeros([d if isinstance(d, int) else 1 for d in i.shape], np.float32)
|
||||
for i in model.get_inputs()}
|
||||
cfg = get_qnn_qdq_config(str(work), Items([zeros]), activation_type=act, weight_type=wt, per_channel=True)
|
||||
exclude = list(cfg.nodes_to_exclude or []) + spec.get("float_nodes", [])
|
||||
if spec.get("float_from"):
|
||||
exclude += downstream(onnx.load(work), spec["float_from"])
|
||||
quantize_static(str(work), str(out), None, quant_format=cfg.quant_format,
|
||||
op_types_to_quantize=cfg.op_types_to_quantize, per_channel=True,
|
||||
activation_type=act, weight_type=wt, nodes_to_exclude=exclude,
|
||||
calibrate_method=CalibrationMethod.MinMax, extra_options=cfg.extra_options,
|
||||
calibration_cache_path=cache)
|
||||
os.remove(work)
|
||||
if cache != ranges:
|
||||
os.remove(cache)
|
||||
print(f" {out.name}: {out.stat().st_size // 1024} KB")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
+22
-18
@@ -1,30 +1,34 @@
|
||||
#!/usr/bin/env bash
|
||||
# Produce the int8 form of a model for the Hexagon (docs/dev/inference.md §5).
|
||||
# Produce the Hexagon's form of each model (docs/dev/inference.md §1.5, §5).
|
||||
#
|
||||
# ./tools/quantise-models.sh PHOTO_DIR MODEL.onnx [MODEL.onnx ...]
|
||||
# ./tools/quantise-models.sh PHOTO_DIR [MODEL ...]
|
||||
# ./tools/quantise-models.sh --ranges RANGES.json mosaic-hq-1408
|
||||
#
|
||||
# Writes `MODEL.int8.onnx` beside each input: a QDQ graph, per-channel int8
|
||||
# weights, uint8 activations — the form QNN's HTP backend takes whole. The
|
||||
# activations' ranges come from running the f32 model over the photographs in
|
||||
# PHOTO_DIR, fed exactly as the app feeds them (letterboxed to the model's
|
||||
# input, the detector's `(x - 127.5) / 128` normalisation), which is why
|
||||
# this is a release-time step and not something the device does: it needs
|
||||
# real photographs and, after it, a person reading §10 M2's numbers.
|
||||
# Writes `<stem>.<form>.onnx` beside each canonical file under models/: a QDQ
|
||||
# graph from QNN's own quantisation config, per-channel weights, in the form
|
||||
# the engine's `Rung::form` names for that role — A16W8, A16W16 or int8, each
|
||||
# the narrowest that held the model's accuracy on the tablet. The activation
|
||||
# ranges come from running the f32 model over the photographs in PHOTO_DIR,
|
||||
# fed exactly as the app feeds them (letterbox maths, pads, normalisation,
|
||||
# face crops through the app's own similarity), which is why this is a
|
||||
# release-time step and not something the device does. With no MODEL, every
|
||||
# model in the table.
|
||||
#
|
||||
# The SCRFD and ArcFace exports are opset 11; per-channel QDQ needs 13, so a
|
||||
# model below 13 is first upgraded to 17. That changes only the graph's
|
||||
# spelling, not a weight — and it is what `tools/fix-face-model-shapes.sh`
|
||||
# will do to the canonical files in the same model release.
|
||||
# The denoiser is calibrated on noisy mosaics, not photographs: its ranges
|
||||
# come from darkroom-denoise's precision gate (`--ranges`), computed on a
|
||||
# smaller tile of the same network — activation ranges do not depend on the
|
||||
# tile's size, and the tensor names match.
|
||||
#
|
||||
# Then measure before shipping: a quantised form is a different network, and
|
||||
# the numbers in inference.md §1.5 are what each one had to hold.
|
||||
#
|
||||
# A venv per run, like fix-face-model-shapes.sh: the tools are not a build
|
||||
# input and nothing in the tree should have them on its path.
|
||||
set -euo pipefail
|
||||
if [ "$#" -lt 2 ]; then
|
||||
sed -n '2,20p' "$0" >&2
|
||||
if [ "$#" -lt 1 ]; then
|
||||
sed -n '2,27p' "$0" >&2
|
||||
exit 2
|
||||
fi
|
||||
PHOTOS="$1"; shift
|
||||
[ -d "${PHOTOS}" ] || { echo "no such directory: ${PHOTOS}" >&2; exit 1; }
|
||||
|
||||
WORK="$(mktemp -d -p /var/tmp quantise-models.XXXXXX)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
@@ -32,4 +36,4 @@ echo "==> venv in ${WORK}"
|
||||
uv venv --python 3.12 "${WORK}/venv" >/dev/null
|
||||
VIRTUAL_ENV="${WORK}/venv" uv pip install --quiet onnx onnxruntime pillow numpy sympy
|
||||
|
||||
exec "${WORK}/venv/bin/python" "$(dirname "$0")/quantise-models.py" "${PHOTOS}" "$@"
|
||||
exec "${WORK}/venv/bin/python" "$(dirname "$0")/quantise-models.py" "$@"
|
||||
|
||||
@@ -42,6 +42,7 @@ dr-ingest.workspace = true
|
||||
# The sameness probe of a catalog duplicate (FR-CAT-11a): SHA-256 over the
|
||||
# ends of each copy, the digest the import already uses for whole files.
|
||||
sha2 = "0.10"
|
||||
half = "2.7"
|
||||
dr-film.workspace = true
|
||||
# The lens profile database, here for the same reason dr-film is: dr-pipeline
|
||||
# knows the maths of lens correction and deliberately has no dependency with
|
||||
|
||||
@@ -96,6 +96,11 @@ pub struct SyncReport {
|
||||
/// (camera-profiles.md §13).
|
||||
pub profiles_uploaded: usize,
|
||||
pub profiles_downloaded: usize,
|
||||
/// TRACES: FR-DEV-6
|
||||
/// The develop presets: whether another device's changes reached this
|
||||
/// one's library, and whether this one's reached the server.
|
||||
pub presets_adopted: bool,
|
||||
pub presets_uploaded: bool,
|
||||
}
|
||||
|
||||
impl SyncReport {
|
||||
@@ -109,6 +114,7 @@ impl SyncReport {
|
||||
|| self.face_shards_downloaded > 0
|
||||
|| self.place_adopted
|
||||
|| self.profiles_downloaded > 0
|
||||
|| self.presets_adopted
|
||||
}
|
||||
}
|
||||
|
||||
@@ -124,12 +130,15 @@ pub enum SyncMessage {
|
||||
///
|
||||
/// Runs on its own thread with its own runtime, like every other network path
|
||||
/// here — the Slint loop must never block (NFR-P9).
|
||||
// One argument per thing the pass touches; see `run` below.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn spawn_sync(
|
||||
conn: Connection,
|
||||
root: String,
|
||||
thumbs_dir: PathBuf,
|
||||
catalog_path: PathBuf,
|
||||
place_path: PathBuf,
|
||||
presets: PresetFiles,
|
||||
scratch: PathBuf,
|
||||
// TRACES: FR-CULL-8
|
||||
// Which face pipeline's shards to export and adopt. From the settings
|
||||
@@ -165,6 +174,7 @@ pub fn spawn_sync(
|
||||
&thumbs_dir,
|
||||
&catalog_path,
|
||||
&place_path,
|
||||
&presets,
|
||||
&scratch,
|
||||
&face_model_id,
|
||||
&tx,
|
||||
@@ -184,7 +194,7 @@ pub fn spawn_sync(
|
||||
rx
|
||||
}
|
||||
|
||||
// Eight, because a sync touches eight distinct things — the same reason
|
||||
// Nine, because a sync touches nine distinct things — the same reason
|
||||
// `repairs::spawn` carries the allow: bundling them into a struct would name
|
||||
// nothing that exists.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
@@ -194,6 +204,7 @@ async fn run(
|
||||
thumbs_dir: &Path,
|
||||
catalog_path: &Path,
|
||||
place_path: &Path,
|
||||
presets: &PresetFiles,
|
||||
scratch: &Path,
|
||||
face_model_id: &str,
|
||||
tx: &std::sync::mpsc::Sender<SyncMessage>,
|
||||
@@ -240,6 +251,12 @@ async fn run(
|
||||
sync_profiles(backend, &base, &dir, &mut report).await;
|
||||
}
|
||||
|
||||
// TRACES: FR-DEV-6
|
||||
// The develop presets, a few kilobytes. Like the profiles, never fails
|
||||
// the pass.
|
||||
let _ = tx.send(SyncMessage::Status("checking presets…".into()));
|
||||
sync_presets(backend, &base, presets, &mut report).await;
|
||||
|
||||
// TRACES: FR-UI-8
|
||||
// Last, and it costs one small GET plus at most one small PUT. Last because
|
||||
// it is the only thing here that is not derived state and so the only thing
|
||||
@@ -1122,6 +1139,156 @@ async fn sync_profiles(
|
||||
}
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-6
|
||||
/// Where this device keeps the preset library, and what the last exchange
|
||||
/// of it with this library left both sides holding.
|
||||
///
|
||||
/// The library is the device's, shared by every library it opens; the base
|
||||
/// is per library, because each library's server holds its own copy and has
|
||||
/// its own history of exchanges with this device.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct PresetFiles {
|
||||
pub library: PathBuf,
|
||||
pub base: PathBuf,
|
||||
}
|
||||
|
||||
/// The preset library on the server, in its own folder so finding it is a
|
||||
/// listing of one file rather than of every shard beside it.
|
||||
const PRESETS_DIR: &str = "presets";
|
||||
const PRESETS_NAME: &str = "library.drpl";
|
||||
|
||||
/// TRACES: FR-DEV-6
|
||||
/// Exchange the develop presets with `<derived>/presets/library.drpl`.
|
||||
///
|
||||
/// Unlike the place, this is merged rather than replaced: a preset saved on
|
||||
/// the tablet and another saved on the desktop between two passes must both
|
||||
/// survive, and a preset deleted on one must not come back from the other.
|
||||
/// [`PresetLibrary::merge`](dr_pipeline::PresetLibrary::merge) decides each
|
||||
/// name against the base the last exchange left, which is what tells a
|
||||
/// deletion here from an addition there.
|
||||
///
|
||||
/// The write is conditional on the server still holding what was read, so
|
||||
/// two devices exchanging at once cannot each save over the other's
|
||||
/// additions; the one that loses the race reads again and merges again.
|
||||
/// Never fails the pass.
|
||||
async fn sync_presets(
|
||||
backend: &dyn RemoteBackend,
|
||||
base: &RemotePath,
|
||||
files: &PresetFiles,
|
||||
report: &mut SyncReport,
|
||||
) {
|
||||
for _ in 0..3 {
|
||||
match exchange_presets(backend, base, files, report).await {
|
||||
Err(RemoteError::PreconditionFailed) => {
|
||||
log::debug!("presets: another device wrote first; reading again");
|
||||
}
|
||||
Err(e) => {
|
||||
log::debug!("not exchanging presets: {e}");
|
||||
return;
|
||||
}
|
||||
Ok(()) => return,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn exchange_presets(
|
||||
backend: &dyn RemoteBackend,
|
||||
base: &RemotePath,
|
||||
files: &PresetFiles,
|
||||
report: &mut SyncReport,
|
||||
) -> Result<(), RemoteError> {
|
||||
use crate::preset_store::PresetStore;
|
||||
use dr_pipeline::PresetLibrary;
|
||||
|
||||
let dir = RemotePath::new(format!("{}/{PRESETS_DIR}", base.as_str()));
|
||||
let target = RemotePath::new(format!("{}/{PRESETS_NAME}", dir.as_str()));
|
||||
|
||||
// A listing that fails reads as "not there". That is safe because the
|
||||
// write below is then `IfAbsent`, which a server holding one refuses.
|
||||
let listed = backend.list(&dir, None).await.ok().and_then(|entries| {
|
||||
entries
|
||||
.into_iter()
|
||||
.find(|e| e.kind == dr_sync::EntryKind::File && e.path.name() == PRESETS_NAME)
|
||||
});
|
||||
|
||||
let mut theirs = match &listed {
|
||||
None => PresetLibrary::default(),
|
||||
Some(_) => {
|
||||
let bytes = read_derived(backend, &target).await?;
|
||||
match std::str::from_utf8(&bytes)
|
||||
.map_err(|e| e.to_string())
|
||||
.and_then(|t| PresetLibrary::parse(t).map_err(|e| e.to_string()))
|
||||
{
|
||||
Ok(library) => library,
|
||||
Err(e) => {
|
||||
// A newer build's format, or damage. Either way not ours
|
||||
// to write over.
|
||||
log::warn!("the preset library on the server will not read ({e}); left alone");
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
let store = PresetStore::open_at(files.library.clone());
|
||||
let read = match store.try_load() {
|
||||
Ok(library) => library.unwrap_or_default(),
|
||||
Err(e) => {
|
||||
// An empty library here would read as every preset deleted.
|
||||
log::warn!("not exchanging presets: {}: {e}", store.path().display());
|
||||
return Ok(());
|
||||
}
|
||||
};
|
||||
let base_store = PresetStore::open_at(files.base.clone());
|
||||
let last = base_store.try_load().ok().flatten().unwrap_or_default();
|
||||
|
||||
// Copies an older build seeded of the shipped presets are not the
|
||||
// photographer's, and are dropped on both sides before anything travels.
|
||||
let mut ours = read.clone();
|
||||
dr_pipeline::bundled::forget_unchanged_copies(&mut ours);
|
||||
dr_pipeline::bundled::forget_unchanged_copies(&mut theirs);
|
||||
|
||||
let merged = PresetLibrary::merge(&last, &ours, &theirs);
|
||||
|
||||
if listed.is_none() && merged.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
if listed.is_none() || merged != theirs {
|
||||
let _ = backend.create_dir(&dir).await;
|
||||
let precondition = match &listed {
|
||||
Some(entry) => dr_sync::Precondition::IfMatch(entry.validator.clone()),
|
||||
None => dr_sync::Precondition::IfAbsent,
|
||||
};
|
||||
backend
|
||||
.put(&target, merged.to_text().into_bytes(), Some(precondition))
|
||||
.await?;
|
||||
report.presets_uploaded = true;
|
||||
}
|
||||
|
||||
if merged != ours {
|
||||
// The develop view saves this file too. A preset it saved since the
|
||||
// read above is merged in rather than written over; it reaches the
|
||||
// server on the next pass, as an addition against the base below.
|
||||
let local = match store.try_load() {
|
||||
Ok(Some(now)) if now != read => PresetLibrary::merge(&read, &merged, &now),
|
||||
_ => merged.clone(),
|
||||
};
|
||||
match store.save(&local) {
|
||||
Ok(()) => report.presets_adopted = true,
|
||||
Err(e) => {
|
||||
log::warn!("saving presets from the library: {e}");
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
}
|
||||
if merged != last {
|
||||
if let Err(e) = base_store.save(&merged) {
|
||||
log::debug!("recording the preset exchange: {e}");
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// TRACES: FR-UI-8
|
||||
/// Fetch just the place, for the handover at launch.
|
||||
///
|
||||
@@ -1934,6 +2101,99 @@ mod derived_guard_tests {
|
||||
let _ = std::fs::remove_dir_all(&root);
|
||||
}
|
||||
|
||||
/// One device's preset files, under `root`.
|
||||
fn preset_device(root: &Path, name: &str) -> PresetFiles {
|
||||
PresetFiles {
|
||||
library: root.join(name).join("presets.drpl"),
|
||||
base: root.join(name).join("presets.base.drpl"),
|
||||
}
|
||||
}
|
||||
|
||||
fn preset_names(files: &PresetFiles) -> Vec<String> {
|
||||
crate::preset_store::PresetStore::open_at(files.library.clone())
|
||||
.load()
|
||||
.names()
|
||||
.map(str::to_string)
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn save_presets(files: &PresetFiles, names: &[&str]) {
|
||||
let mut library = dr_pipeline::PresetLibrary::default();
|
||||
for name in names {
|
||||
library
|
||||
.insert(name, dr_pipeline::Preset::default())
|
||||
.unwrap();
|
||||
}
|
||||
crate::preset_store::PresetStore::open_at(files.library.clone())
|
||||
.save(&library)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn presets_reach_every_device_and_so_do_their_deletions() {
|
||||
// TRACES: FR-DEV-6
|
||||
let root = std::env::temp_dir().join(format!("dr-preset-sync-{}", std::process::id()));
|
||||
let _ = std::fs::remove_dir_all(&root);
|
||||
let server = root.join("server");
|
||||
std::fs::create_dir_all(server.join(".darkroom-derived")).unwrap();
|
||||
let backend = dr_sync_folder::FolderBackend::new(&server).unwrap();
|
||||
let base = RemotePath::new(".darkroom-derived");
|
||||
let (desk, tablet) = (preset_device(&root, "desk"), preset_device(&root, "tablet"));
|
||||
|
||||
save_presets(&desk, &["Warm"]);
|
||||
save_presets(&tablet, &["Mono"]);
|
||||
for files in [&desk, &tablet, &desk] {
|
||||
sync_presets(&backend, &base, files, &mut SyncReport::default()).await;
|
||||
}
|
||||
assert_eq!(preset_names(&desk), vec!["Mono", "Warm"]);
|
||||
assert_eq!(preset_names(&tablet), vec!["Mono", "Warm"]);
|
||||
|
||||
// Deleted on the desk: gone from the tablet, not back on the desk.
|
||||
save_presets(&desk, &["Mono"]);
|
||||
for files in [&desk, &tablet, &desk] {
|
||||
sync_presets(&backend, &base, files, &mut SyncReport::default()).await;
|
||||
}
|
||||
assert_eq!(preset_names(&desk), vec!["Mono"]);
|
||||
assert_eq!(preset_names(&tablet), vec!["Mono"]);
|
||||
|
||||
// Settled: a pass with nothing new writes nothing.
|
||||
let mut quiet = SyncReport::default();
|
||||
sync_presets(&backend, &base, &tablet, &mut quiet).await;
|
||||
assert!(!quiet.presets_uploaded && !quiet.presets_adopted);
|
||||
|
||||
let _ = std::fs::remove_dir_all(&root);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn a_preset_library_the_server_cannot_read_is_left_alone() {
|
||||
// TRACES: FR-DEV-6
|
||||
// A newer build's format reads as unreadable here, and is that build's
|
||||
// presets: neither written over nor taken as an empty library.
|
||||
let root = std::env::temp_dir().join(format!("dr-preset-unread-{}", std::process::id()));
|
||||
let _ = std::fs::remove_dir_all(&root);
|
||||
let server = root.join("server");
|
||||
let held = server.join(".darkroom-derived/presets/library.drpl");
|
||||
std::fs::create_dir_all(held.parent().unwrap()).unwrap();
|
||||
std::fs::write(&held, "drpl 9999\n").unwrap();
|
||||
let backend = dr_sync_folder::FolderBackend::new(&server).unwrap();
|
||||
let desk = preset_device(&root, "desk");
|
||||
save_presets(&desk, &["Warm"]);
|
||||
|
||||
let mut report = SyncReport::default();
|
||||
sync_presets(
|
||||
&backend,
|
||||
&RemotePath::new(".darkroom-derived"),
|
||||
&desk,
|
||||
&mut report,
|
||||
)
|
||||
.await;
|
||||
|
||||
assert!(!report.presets_uploaded);
|
||||
assert_eq!(std::fs::read_to_string(&held).unwrap(), "drpl 9999\n");
|
||||
assert_eq!(preset_names(&desk), vec!["Warm"]);
|
||||
let _ = std::fs::remove_dir_all(&root);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn a_place_that_could_not_be_read_is_never_written_over() {
|
||||
// A dehydrated placeholder, and the record on the server may well be
|
||||
|
||||
+144
-38
@@ -18,6 +18,7 @@ use std::sync::Arc;
|
||||
|
||||
use dr_decode::RawImage;
|
||||
use dr_gpu::{DemosaicedImage, GrainBlend};
|
||||
use dr_pipeline::learned_denoise::Method;
|
||||
|
||||
use super::session::DevelopSession;
|
||||
|
||||
@@ -41,6 +42,12 @@ pub(crate) struct DenoiseState {
|
||||
failed: Option<String>,
|
||||
/// Where the noise figures came from, for the panel.
|
||||
source: Option<dr_denoise::Source>,
|
||||
/// The file's bytes, hashed for the on-disk cache, which keys each
|
||||
/// network's result on them and the model (`denoise_cache::FileHash`).
|
||||
file_hash: Option<super::denoise_cache::FileHash>,
|
||||
/// The method the result, the job and the failure above are for. A
|
||||
/// different one asked for discards them.
|
||||
method: Option<Method>,
|
||||
}
|
||||
|
||||
struct Job {
|
||||
@@ -89,6 +96,11 @@ impl DevelopSession {
|
||||
if self.denoise.mosaic.is_some() {
|
||||
self.denoise.profile = dr_decode::noise_profile(bytes);
|
||||
self.denoise.iso = meta.iso;
|
||||
// TRACES: FR-DEV-3g
|
||||
// The bytes are only here now, so they are hashed now: a
|
||||
// reopened or exported photograph finds its result on disk,
|
||||
// under whichever method it asks for.
|
||||
self.denoise.file_hash = Some(super::denoise_cache::FileHash::of(bytes));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -97,6 +109,7 @@ impl DevelopSession {
|
||||
/// Cheap when nothing changed; the develop view calls it after every
|
||||
/// change to the edit, whatever made it — a slider, undo, a version.
|
||||
pub fn reconcile_denoise(&mut self) {
|
||||
self.forget_other_method();
|
||||
let wanted = self.graph.denoise_applied() && self.denoise.mosaic.is_some();
|
||||
if !wanted {
|
||||
if let Some(job) = self.denoise.job.take() {
|
||||
@@ -112,20 +125,11 @@ impl DevelopSession {
|
||||
{
|
||||
return;
|
||||
}
|
||||
let Some(model) = crate::library::denoise_model() else {
|
||||
self.denoise.failed = Some("the denoise model is not installed".into());
|
||||
let Some(work) = self.work(Arc::new(AtomicBool::new(false))) else {
|
||||
return;
|
||||
};
|
||||
let cancel = work.cancel.clone();
|
||||
let (tx, rx) = mpsc::channel();
|
||||
let cancel = Arc::new(AtomicBool::new(false));
|
||||
let work = Work {
|
||||
ctx: self.ctx.clone(),
|
||||
raw: self.denoise.mosaic.clone().expect("checked above"),
|
||||
profile: self.denoise.profile.clone(),
|
||||
iso: self.denoise.iso,
|
||||
model,
|
||||
cancel: cancel.clone(),
|
||||
};
|
||||
crate::executors::spawn(crate::executors::Executor::Decode, "denoise", move || {
|
||||
let result = work.run(&mut |done, total| {
|
||||
let _ = tx.send(Msg::Progress(done, total));
|
||||
@@ -135,6 +139,51 @@ impl DevelopSession {
|
||||
self.denoise.job = Some(Job { rx, cancel });
|
||||
}
|
||||
|
||||
/// Drop what was computed, or is being computed, for a network the edit
|
||||
/// no longer asks for — a choice in the panel, an undo, a version.
|
||||
/// Each network's result stays in the on-disk cache, so going back to
|
||||
/// one is a read, not a run. The classical demosaic asks for no network
|
||||
/// and drops nothing: the result is kept for the way back.
|
||||
fn forget_other_method(&mut self) {
|
||||
let asked = self.graph.denoise_method();
|
||||
if !asked.learned() || self.denoise.method == Some(asked) {
|
||||
return;
|
||||
}
|
||||
if self.denoise.method.is_none() {
|
||||
// The first network asked for: nothing computed is another's.
|
||||
self.denoise.method = Some(asked);
|
||||
return;
|
||||
}
|
||||
if let Some(job) = self.denoise.job.take() {
|
||||
job.cancel.store(true, Ordering::Relaxed);
|
||||
}
|
||||
self.denoise.result = None;
|
||||
self.denoise.blended = None;
|
||||
self.denoise.failed = None;
|
||||
self.denoise.source = None;
|
||||
self.denoise.method = Some(asked);
|
||||
}
|
||||
|
||||
/// The job for the method the edit asks for, or `None` — with the reason
|
||||
/// kept as the failure — where its network is not installed.
|
||||
fn work(&mut self, cancel: Arc<AtomicBool>) -> Option<Work> {
|
||||
let raw = self.denoise.mosaic.clone()?;
|
||||
let Some((model, net)) = crate::library::denoise_model(self.graph.denoise_method()) else {
|
||||
self.denoise.failed = Some("the denoise model is not installed".into());
|
||||
return None;
|
||||
};
|
||||
Some(Work {
|
||||
ctx: self.ctx.clone(),
|
||||
raw,
|
||||
profile: self.denoise.profile.clone(),
|
||||
iso: self.denoise.iso,
|
||||
cache_key: self.denoise.file_hash.as_ref().map(|h| h.key(&model)),
|
||||
model,
|
||||
net,
|
||||
cancel,
|
||||
})
|
||||
}
|
||||
|
||||
/// Collect what the job sent since the last poll.
|
||||
pub fn poll_denoise(&mut self) -> DenoiseStatus {
|
||||
let Some(job) = &self.denoise.job else {
|
||||
@@ -173,9 +222,10 @@ impl DevelopSession {
|
||||
if !self.graph.denoise_applied() || self.denoise.result.is_some() {
|
||||
return Ok(());
|
||||
}
|
||||
let Some(raw) = self.denoise.mosaic.clone() else {
|
||||
self.forget_other_method();
|
||||
if self.denoise.mosaic.is_none() {
|
||||
return Ok(());
|
||||
};
|
||||
}
|
||||
// Already under way: wait for it rather than start again.
|
||||
if let Some(job) = self.denoise.job.take() {
|
||||
for msg in job.rx.iter() {
|
||||
@@ -184,14 +234,8 @@ impl DevelopSession {
|
||||
}
|
||||
}
|
||||
}
|
||||
let model = crate::library::denoise_model().ok_or("the denoise model is not installed")?;
|
||||
let work = Work {
|
||||
ctx: self.ctx.clone(),
|
||||
raw,
|
||||
profile: self.denoise.profile.clone(),
|
||||
iso: self.denoise.iso,
|
||||
model,
|
||||
cancel: Arc::new(AtomicBool::new(false)),
|
||||
let Some(work) = self.work(Arc::new(AtomicBool::new(false))) else {
|
||||
return Err(self.denoise.failed.clone().unwrap_or_default());
|
||||
};
|
||||
let finished = work.run(&mut |_, _| {})?;
|
||||
self.land(finished)
|
||||
@@ -272,12 +316,33 @@ struct Work {
|
||||
profile: Option<Vec<(f32, f32)>>,
|
||||
iso: Option<u32>,
|
||||
model: std::path::PathBuf,
|
||||
/// The network `model` is: its context and its whole-frame sibling.
|
||||
net: dr_denoise::Shipped,
|
||||
cancel: Arc<AtomicBool>,
|
||||
cache_key: Option<String>,
|
||||
}
|
||||
|
||||
impl Work {
|
||||
fn run(self, progress: &mut dyn FnMut(usize, usize)) -> Result<Finished, String> {
|
||||
let started = std::time::Instant::now();
|
||||
// TRACES: FR-DEV-3g
|
||||
// A result computed before — this photograph opened earlier, or
|
||||
// developed and now exported — is read back rather than recomputed.
|
||||
let cache_dir = super::denoise_cache::dir();
|
||||
if let Some(hit) = self
|
||||
.cache_key
|
||||
.as_deref()
|
||||
.and_then(|key| super::denoise_cache::load(&cache_dir, key))
|
||||
{
|
||||
return Ok(Finished {
|
||||
rgb: hit.rgb,
|
||||
width: hit.width,
|
||||
height: hit.height,
|
||||
source: hit.source,
|
||||
rung: "the cache".into(),
|
||||
seconds: started.elapsed().as_secs_f64(),
|
||||
});
|
||||
}
|
||||
// The app's own hot-pixel pass, on a copy: the classical source was
|
||||
// repaired by the same pass inside `Demosaicer::run`.
|
||||
let mut raw = (*self.raw).clone();
|
||||
@@ -287,7 +352,7 @@ impl Work {
|
||||
let noise = dr_denoise::noise::for_frame_with(&raw, self.profile.as_deref(), self.iso)
|
||||
.ok_or("this photograph gives no way to measure its noise")?;
|
||||
let mut net =
|
||||
dr_denoise::onnx::OnnxNet::from_path(&self.model).map_err(|e| e.to_string())?;
|
||||
dr_denoise::onnx::OnnxNet::open(&self.model, self.net).map_err(|e| e.to_string())?;
|
||||
let rung = net
|
||||
.rung()
|
||||
.map(|r| r.label().to_string())
|
||||
@@ -299,14 +364,24 @@ impl Work {
|
||||
})
|
||||
.map_err(|e| e.to_string())?
|
||||
.ok_or("stopped")?;
|
||||
Ok(Finished {
|
||||
let finished = Finished {
|
||||
rgb,
|
||||
width: raw.crop.width,
|
||||
height: raw.crop.height,
|
||||
source: noise.source,
|
||||
rung,
|
||||
seconds: started.elapsed().as_secs_f64(),
|
||||
})
|
||||
};
|
||||
if let Some(key) = self.cache_key.as_deref() {
|
||||
let entry = super::denoise_cache::Entry {
|
||||
rgb: finished.rgb.clone(),
|
||||
width: finished.width,
|
||||
height: finished.height,
|
||||
source: finished.source,
|
||||
};
|
||||
super::denoise_cache::store(&cache_dir, key, &entry, super::denoise_cache::BUDGET);
|
||||
}
|
||||
Ok(finished)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -385,15 +460,24 @@ mod tests {
|
||||
let result = s.denoise.result.clone().unwrap();
|
||||
let same = |a: &Arc<DemosaicedImage>, b: &Arc<DemosaicedImage>| Arc::ptr_eq(a, b);
|
||||
|
||||
// Off: the classical demosaic, result or no result.
|
||||
assert!(same(&s.developed_source(), &classical));
|
||||
// On, no grain: the network's result as it is.
|
||||
s.graph
|
||||
.set_param(learned_denoise::ID, learned_denoise::APPLY, 1.0);
|
||||
// On by default, at full strength: the network's result as it is.
|
||||
assert!(same(&s.developed_source(), &result));
|
||||
// Grain: a blend, made once per value and reused until it moves.
|
||||
// Bilinear: the classical demosaic, result or no result.
|
||||
s.graph.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
Method::Bilinear.index(),
|
||||
);
|
||||
assert!(same(&s.developed_source(), &classical));
|
||||
s.graph.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
Method::DEFAULT.index(),
|
||||
);
|
||||
// Less strength: a blend, made once per value and reused until it
|
||||
// moves.
|
||||
s.graph
|
||||
.set_param(learned_denoise::ID, learned_denoise::GRAIN, 40.0);
|
||||
.set_param(learned_denoise::ID, learned_denoise::STRENGTH, 60.0);
|
||||
let blended = s.developed_source();
|
||||
assert!(!same(&blended, &result) && !same(&blended, &classical));
|
||||
assert!(
|
||||
@@ -401,7 +485,7 @@ mod tests {
|
||||
"the same grain must not blend again"
|
||||
);
|
||||
s.graph
|
||||
.set_param(learned_denoise::ID, learned_denoise::GRAIN, 60.0);
|
||||
.set_param(learned_denoise::ID, learned_denoise::STRENGTH, 40.0);
|
||||
assert!(!same(&s.developed_source(), &blended));
|
||||
// The sensor's own reading stays the classical one throughout.
|
||||
assert!(same(&s.demosaiced, &classical));
|
||||
@@ -413,19 +497,41 @@ mod tests {
|
||||
let mut s =
|
||||
DevelopSession::open_owned(&ctx, bayer(64, 64), dr_types::Orientation::NORMAL).unwrap();
|
||||
s.denoise.failed = Some("no model".into());
|
||||
s.graph
|
||||
.set_param(learned_denoise::ID, learned_denoise::APPLY, 1.0);
|
||||
s.reconcile_denoise();
|
||||
assert!(
|
||||
s.denoise.job.is_none(),
|
||||
"a failure is not retried while the switch stays on"
|
||||
"a failure is not retried while the method stays"
|
||||
);
|
||||
s.graph.set_param(
|
||||
learned_denoise::ID,
|
||||
learned_denoise::METHOD,
|
||||
Method::Bilinear.index(),
|
||||
);
|
||||
s.graph
|
||||
.set_param(learned_denoise::ID, learned_denoise::APPLY, 0.0);
|
||||
s.reconcile_denoise();
|
||||
assert!(
|
||||
s.denoise.failed.is_none(),
|
||||
"toggling off is how a failure is retried"
|
||||
"choosing bilinear is how a failure is retried"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn another_network_discards_the_result_and_bilinear_keeps_it() {
|
||||
let Some(ctx) = headless() else { return };
|
||||
let mut s =
|
||||
DevelopSession::open_owned(&ctx, bayer(64, 64), dr_types::Orientation::NORMAL).unwrap();
|
||||
s.denoise.method = Some(Method::DEFAULT);
|
||||
landed(&mut s);
|
||||
let to = |s: &mut DevelopSession, m: Method| {
|
||||
s.graph
|
||||
.set_param(learned_denoise::ID, learned_denoise::METHOD, m.index());
|
||||
s.forget_other_method();
|
||||
};
|
||||
to(&mut s, Method::Bilinear);
|
||||
assert!(s.denoise.result.is_some(), "kept for the way back");
|
||||
to(&mut s, Method::DEFAULT);
|
||||
assert!(s.denoise.result.is_some());
|
||||
to(&mut s, Method::Fast);
|
||||
assert!(s.denoise.result.is_none(), "another network's picture");
|
||||
assert_eq!(s.denoise.method, Some(Method::Fast));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,255 @@
|
||||
//! TRACES: FR-DEV-3g
|
||||
//! The learned denoise's results, kept on disk (docs/dev/denoise.md §7.1).
|
||||
//!
|
||||
//! The network takes seconds per photograph and is on by default, so a
|
||||
//! photograph reopened, or exported after it was developed, must not pay
|
||||
//! again. A result is the network's output as it is — linear camera RGB at
|
||||
//! the crop's size — written as half floats: about 120 MB for 20 MP, and no
|
||||
//! compressor to link on Android. The strength slider is applied afterwards
|
||||
//! and is not part of the key, so moving it never invalidates anything.
|
||||
//!
|
||||
//! **Keyed on the file's bytes and the model**: a SHA-256 of what was
|
||||
//! decoded, and the model file's name and size. Anything that changes the
|
||||
//! input or the network changes the key; the edit does not.
|
||||
//!
|
||||
//! **Bounded by a budget**, oldest first: a hit refreshes an entry's time, a
|
||||
//! write evicts what no longer fits. Disposable — a peer of the inference
|
||||
//! engine's cache, never synced — so an entry that cannot be read is simply
|
||||
//! recomputed.
|
||||
|
||||
use std::io::{Read, Write};
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
/// Bytes the cache may hold before the oldest entries go: about forty 20 MP
|
||||
/// photographs.
|
||||
pub const BUDGET: u64 = 5 * 1024 * 1024 * 1024;
|
||||
|
||||
const MAGIC: &[u8; 8] = b"DRDN1\0\0\0";
|
||||
const HEADER: usize = 8 + 4 + 4 + 1;
|
||||
|
||||
/// A cached result: the RGB samples, their size, and where the noise
|
||||
/// figures came from.
|
||||
pub struct Entry {
|
||||
pub rgb: Vec<f32>,
|
||||
pub width: u32,
|
||||
pub height: u32,
|
||||
pub source: dr_denoise::Source,
|
||||
}
|
||||
|
||||
/// The cache's directory: beside the inference engine's, under the data
|
||||
/// root, which is writable on every platform.
|
||||
pub fn dir() -> PathBuf {
|
||||
crate::library::inference_cache_dir()
|
||||
.parent()
|
||||
.map(|p| p.join("denoise-cache"))
|
||||
.unwrap_or_else(|| PathBuf::from("denoise-cache"))
|
||||
}
|
||||
|
||||
/// A file's bytes, hashed once at open: each method's network keys its
|
||||
/// result from this, so changing the method does not read the file again.
|
||||
#[derive(Clone)]
|
||||
pub struct FileHash(Sha256);
|
||||
|
||||
impl FileHash {
|
||||
pub fn of(bytes: &[u8]) -> Self {
|
||||
let mut h = Sha256::new();
|
||||
h.update(bytes);
|
||||
FileHash(h)
|
||||
}
|
||||
|
||||
/// The key for these bytes under a model.
|
||||
pub fn key(&self, model: &Path) -> String {
|
||||
let mut h = self.0.clone();
|
||||
if let Some(name) = model.file_name() {
|
||||
h.update(name.to_string_lossy().as_bytes());
|
||||
}
|
||||
let size = std::fs::metadata(model).map(|m| m.len()).unwrap_or(0);
|
||||
h.update(size.to_le_bytes());
|
||||
let digest = h.finalize();
|
||||
digest.iter().map(|b| format!("{b:02x}")).collect()
|
||||
}
|
||||
}
|
||||
|
||||
fn path_in(dir: &Path, key: &str) -> PathBuf {
|
||||
dir.join(format!("{key}.drdn"))
|
||||
}
|
||||
|
||||
fn source_code(s: dr_denoise::Source) -> u8 {
|
||||
match s {
|
||||
dr_denoise::Source::Table => 0,
|
||||
dr_denoise::Source::DngProfile => 1,
|
||||
dr_denoise::Source::Measured => 2,
|
||||
}
|
||||
}
|
||||
|
||||
fn source_from(code: u8) -> Option<dr_denoise::Source> {
|
||||
Some(match code {
|
||||
0 => dr_denoise::Source::Table,
|
||||
1 => dr_denoise::Source::DngProfile,
|
||||
2 => dr_denoise::Source::Measured,
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
/// The entry for `key`, if one is held and reads back whole. A hit
|
||||
/// refreshes its time, so what is in use outlives what is not.
|
||||
pub fn load(dir: &Path, key: &str) -> Option<Entry> {
|
||||
let path = path_in(dir, key);
|
||||
let mut file = std::fs::File::open(&path).ok()?;
|
||||
let mut head = [0u8; HEADER];
|
||||
file.read_exact(&mut head).ok()?;
|
||||
if &head[..8] != MAGIC {
|
||||
return None;
|
||||
}
|
||||
let width = u32::from_le_bytes(head[8..12].try_into().ok()?);
|
||||
let height = u32::from_le_bytes(head[12..16].try_into().ok()?);
|
||||
let source = source_from(head[16])?;
|
||||
let samples = (width as usize)
|
||||
.checked_mul(height as usize)?
|
||||
.checked_mul(3)?;
|
||||
let mut raw = vec![0u8; samples.checked_mul(2)?];
|
||||
file.read_exact(&mut raw).ok()?;
|
||||
let rgb = raw
|
||||
.chunks_exact(2)
|
||||
.map(|b| half::f16::from_le_bytes([b[0], b[1]]).to_f32())
|
||||
.collect();
|
||||
let _ = file.set_modified(std::time::SystemTime::now());
|
||||
Some(Entry {
|
||||
rgb,
|
||||
width,
|
||||
height,
|
||||
source,
|
||||
})
|
||||
}
|
||||
|
||||
/// Keep `entry` under `key`, then evict to `budget`. Written beside its name
|
||||
/// and renamed, so a reader never sees half a file. Failure only costs a
|
||||
/// recompute next time, so it is logged and swallowed.
|
||||
pub fn store(dir: &Path, key: &str, entry: &Entry, budget: u64) {
|
||||
let result = (|| -> std::io::Result<()> {
|
||||
std::fs::create_dir_all(dir)?;
|
||||
let path = path_in(dir, key);
|
||||
let partial = path.with_extension("part");
|
||||
let mut out = std::io::BufWriter::new(std::fs::File::create(&partial)?);
|
||||
out.write_all(MAGIC)?;
|
||||
out.write_all(&entry.width.to_le_bytes())?;
|
||||
out.write_all(&entry.height.to_le_bytes())?;
|
||||
out.write_all(&[source_code(entry.source)])?;
|
||||
for v in &entry.rgb {
|
||||
out.write_all(&half::f16::from_f32(*v).to_le_bytes())?;
|
||||
}
|
||||
out.into_inner().map_err(|e| e.into_error())?.sync_all()?;
|
||||
std::fs::rename(&partial, &path)
|
||||
})();
|
||||
if let Err(e) = result {
|
||||
log::warn!("denoise cache: not kept ({e})");
|
||||
return;
|
||||
}
|
||||
evict(dir, budget);
|
||||
}
|
||||
|
||||
/// Remove the oldest entries until what is left fits `budget`.
|
||||
pub fn evict(dir: &Path, budget: u64) {
|
||||
let Ok(read) = std::fs::read_dir(dir) else {
|
||||
return;
|
||||
};
|
||||
let mut entries: Vec<(std::time::SystemTime, u64, PathBuf)> = read
|
||||
.flatten()
|
||||
.filter(|e| e.path().extension().is_some_and(|x| x == "drdn"))
|
||||
.filter_map(|e| {
|
||||
let m = e.metadata().ok()?;
|
||||
Some((m.modified().ok()?, m.len(), e.path()))
|
||||
})
|
||||
.collect();
|
||||
let mut total: u64 = entries.iter().map(|(_, len, _)| len).sum();
|
||||
entries.sort_by_key(|(t, _, _)| *t);
|
||||
for (_, len, path) in entries {
|
||||
if total <= budget {
|
||||
break;
|
||||
}
|
||||
if std::fs::remove_file(&path).is_ok() {
|
||||
total -= len;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn scratch(name: &str) -> PathBuf {
|
||||
let d =
|
||||
std::env::temp_dir().join(format!("dr-denoise-cache-{name}-{}", std::process::id()));
|
||||
let _ = std::fs::remove_dir_all(&d);
|
||||
d
|
||||
}
|
||||
|
||||
fn entry(w: u32, h: u32) -> Entry {
|
||||
Entry {
|
||||
rgb: (0..w * h * 3).map(|i| i as f32 / 100.0).collect(),
|
||||
width: w,
|
||||
height: h,
|
||||
source: dr_denoise::Source::DngProfile,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_result_comes_back_as_it_went_in_to_half_precision() {
|
||||
let dir = scratch("roundtrip");
|
||||
let e = entry(4, 3);
|
||||
store(&dir, "k", &e, BUDGET);
|
||||
let back = load(&dir, "k").expect("a hit");
|
||||
assert_eq!((back.width, back.height), (4, 3));
|
||||
assert_eq!(back.source, dr_denoise::Source::DngProfile);
|
||||
for (a, b) in e.rgb.iter().zip(&back.rgb) {
|
||||
assert!((a - b).abs() <= a.abs() * 1e-3 + 1e-4, "{a} {b}");
|
||||
}
|
||||
assert!(load(&dir, "other").is_none());
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_key_follows_the_bytes_and_the_model() {
|
||||
let dir = scratch("key");
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
let model = dir.join("m.onnx");
|
||||
std::fs::write(&model, b"weights").unwrap();
|
||||
let a = FileHash::of(b"photo").key(&model);
|
||||
assert_eq!(a, FileHash::of(b"photo").key(&model));
|
||||
assert_ne!(a, FileHash::of(b"photo2").key(&model));
|
||||
std::fs::write(&model, b"other weights").unwrap();
|
||||
assert_ne!(a, FileHash::of(b"photo").key(&model));
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_oldest_entries_go_first_past_the_budget() {
|
||||
let dir = scratch("evict");
|
||||
let e = entry(10, 10);
|
||||
store(&dir, "old", &e, BUDGET);
|
||||
let one = std::fs::metadata(path_in(&dir, "old")).unwrap().len();
|
||||
std::thread::sleep(std::time::Duration::from_millis(20));
|
||||
store(&dir, "mid", &e, BUDGET);
|
||||
std::thread::sleep(std::time::Duration::from_millis(20));
|
||||
// A hit makes the oldest the most recent.
|
||||
load(&dir, "old").unwrap();
|
||||
std::thread::sleep(std::time::Duration::from_millis(20));
|
||||
store(&dir, "new", &e, 2 * one);
|
||||
assert!(load(&dir, "mid").is_none(), "the least recently used went");
|
||||
assert!(load(&dir, "old").is_some() && load(&dir, "new").is_some());
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_damaged_entry_is_a_miss() {
|
||||
let dir = scratch("damaged");
|
||||
store(&dir, "k", &entry(4, 4), BUDGET);
|
||||
let p = path_in(&dir, "k");
|
||||
let bytes = std::fs::read(&p).unwrap();
|
||||
std::fs::write(&p, &bytes[..bytes.len() / 2]).unwrap();
|
||||
assert!(load(&dir, "k").is_none());
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
}
|
||||
}
|
||||
@@ -18,6 +18,7 @@
|
||||
|
||||
mod curves;
|
||||
mod denoise;
|
||||
mod denoise_cache;
|
||||
mod framing;
|
||||
mod history;
|
||||
mod mask_ops;
|
||||
|
||||
@@ -942,7 +942,8 @@ mod tests {
|
||||
dr_pipeline::ParamId("blue_sat"),
|
||||
)
|
||||
};
|
||||
assert_eq!(blue(), Some(58.0));
|
||||
// Blue is shared across our bands at measured factors; blue takes 0.75.
|
||||
assert_eq!(blue(), Some(58.0 * 0.75));
|
||||
assert!(!s.adopt_earlier_edit(), "taken: a second call does nothing");
|
||||
assert!(!s.graph.is_neutral());
|
||||
s.undo();
|
||||
|
||||
@@ -34,8 +34,12 @@ pub fn init(runtime_dirs: Vec<PathBuf>) {
|
||||
(Role::EyeClassifier, crate::library::EYE_MODEL),
|
||||
(Role::EyeClassifier, crate::library::SUNGLASSES_MODEL),
|
||||
(Role::Inpainter, crate::library::INPAINT_MODEL),
|
||||
(Role::Denoiser, crate::library::DENOISE_MODEL),
|
||||
]);
|
||||
let denoisers = [dr_denoise::FAST, dr_denoise::BEST];
|
||||
wanted.extend(denoisers.map(|n| (Role::Denoiser, n.file)));
|
||||
// Their any-size siblings, which the engine compiles only on a rung
|
||||
// that runs whole frames (TensorRT; denoise.md §14).
|
||||
wanted.extend(denoisers.map(|n| (Role::WholeDenoiser, n.whole)));
|
||||
let models: Vec<(Role, PathBuf)> = wanted
|
||||
.into_iter()
|
||||
.filter_map(|(role, name)| Some((role, crate::library::shared_model(name)?)))
|
||||
@@ -46,12 +50,14 @@ pub fn init(runtime_dirs: Vec<PathBuf>) {
|
||||
cache_dir: crate::library::inference_cache_dir(),
|
||||
models,
|
||||
embedded: {
|
||||
let [landscape, portrait] = dr_pano::xfeat::embedded_model_bytes();
|
||||
vec![
|
||||
(Role::Segmenter, dr_segment::embedded_model_bytes()),
|
||||
(Role::Keypoints, landscape),
|
||||
(Role::Keypoints, portrait),
|
||||
]
|
||||
let [landscape, portrait] = dr_pano::xfeat::embedded_models();
|
||||
let tag = |role| move |(form, bytes)| (role, form, bytes);
|
||||
dr_segment::embedded_models()
|
||||
.into_iter()
|
||||
.map(tag(Role::Segmenter))
|
||||
.chain(landscape.into_iter().map(tag(Role::Keypoints)))
|
||||
.chain(portrait.into_iter().map(tag(Role::Keypoints)))
|
||||
.collect()
|
||||
},
|
||||
ceiling: None,
|
||||
threads: 0,
|
||||
@@ -81,7 +87,7 @@ pub fn user_runtime_dir() -> PathBuf {
|
||||
///
|
||||
/// Reads the shared and system directories only. An account-private model
|
||||
/// directory can override the file `library::face_models` loads, but not
|
||||
/// which form the backend wants, and the int8 sibling is something a
|
||||
/// which form the backend wants, and the quantised sibling is something a
|
||||
/// packager ships, not something a user drops in.
|
||||
pub fn detector_form(detector: FaceDetector) -> Form {
|
||||
let canonical = crate::library::shared_model(detector.file_name())
|
||||
@@ -92,8 +98,11 @@ pub fn detector_form(detector: FaceDetector) -> Form {
|
||||
/// The `faces.model_id` this device indexes under with `detector`.
|
||||
pub fn model_id(detector: FaceDetector) -> &'static str {
|
||||
match detector_form(detector) {
|
||||
Form::F32 => detector.model_id(),
|
||||
Form::Int8 => detector.model_id_int8(),
|
||||
Form::A16W8 => detector.model_id_a16w8(),
|
||||
// No detector is offered in A16W16 (inference.md §1.5); were one, it
|
||||
// would be the network f32 is to the last bit that a person can see.
|
||||
Form::F32 | Form::A16W16 => detector.model_id(),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -223,7 +223,14 @@ fn catalogued(key: &str) -> Option<&'static str> {
|
||||
// panel under its own name — see `rows_filtered`.
|
||||
"param.lens_profile.apply" => "Apply",
|
||||
"param.learned_denoise.apply" => "Apply",
|
||||
"param.learned_denoise.grain" => "Keep grain",
|
||||
// Which demosaic: the classical one, or a network by how long it takes.
|
||||
"param.learned_denoise.method" => "Method",
|
||||
"param.learned_denoise.method.bilinear" => "Bilinear",
|
||||
"param.learned_denoise.method.fast" => "Fast",
|
||||
"param.learned_denoise.method.best" => "Best",
|
||||
// How strongly: 100 % is the network's result, and less puts the
|
||||
// removed noise's brightness back as grain.
|
||||
"param.learned_denoise.strength" => "Strength",
|
||||
"param.camera_profile.apply" => "Use Profile",
|
||||
// The LookTable's strength.
|
||||
"param.camera_profile.look" => "Look Amount",
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user