Merge: master at 0.13.6, with the drag-ghost file and the shared model lookup ported into the split modules
This commit is contained in:
@@ -89,6 +89,30 @@ three `405`s before each `MOVE`. The backend now remembers the collections
|
|||||||
it has confirmed (`known_dirs`) for its lifetime, which is one job. When a
|
it has confirmed (`known_dirs`) for its lifetime, which is one job. When a
|
||||||
per-file operation has a per-batch precondition, satisfy it once.
|
per-file operation has a per-batch precondition, satisfy it once.
|
||||||
|
|
||||||
|
## Providers: read the runtime's source for the version on disk, not the binding
|
||||||
|
|
||||||
|
Two things the MIGraphX rung (2026-09-20) got wrong before it was measured
|
||||||
|
right, both because `ort`'s builder was trusted to mean what its method
|
||||||
|
names say.
|
||||||
|
|
||||||
|
**A binding's option builder may fill a struct the runtime no longer
|
||||||
|
reads.** `ep::MIGraphX::with_save_model` sets fields of the legacy
|
||||||
|
`OrtMIGraphXProviderOptions`; ONNX Runtime 1.29 reads that struct for the
|
||||||
|
precision flags and ignores the rest, so every session compiled for 40 s
|
||||||
|
and the cache directory went nowhere. The option that works
|
||||||
|
(`migraphx_model_cache_dir`) exists only in the generic key/value
|
||||||
|
registration, which `session::migraphx` calls on the API table directly.
|
||||||
|
Before wiring a provider option, fetch the provider's source at the
|
||||||
|
runtime's exact version and find where the option is *read*.
|
||||||
|
|
||||||
|
**A provider's cache key may leave out what you are varying.** MIGraphX
|
||||||
|
keys a compiled program on graph, GPU and its own version — not precision.
|
||||||
|
The first fp16 measurement built in 0.3 s and matched f32 to the tenth of a
|
||||||
|
millisecond, because it had loaded the f32 program. A "from cache" build
|
||||||
|
that is suspiciously fast on the first run of a new configuration is a key
|
||||||
|
collision, not a fast provider; give each precision its own directory (the
|
||||||
|
engine does) and check the cache directory gained a file.
|
||||||
|
|
||||||
## Measuring
|
## Measuring
|
||||||
|
|
||||||
`cargo run --release -p dr-ui --example identity_bench -- CATALOG THUMBS`
|
`cargo run --release -p dr-ui --example identity_bench -- CATALOG THUMBS`
|
||||||
|
|||||||
Generated
+27
-25
@@ -1221,7 +1221,7 @@ checksum = "f27ae1dd37df86211c42e150270f82743308803d90a6f6e6651cd730d5e1732f"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "darkroom-android"
|
name = "darkroom-android"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"android_logger",
|
"android_logger",
|
||||||
"dr-plat",
|
"dr-plat",
|
||||||
@@ -1234,7 +1234,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "darkroom-desktop"
|
name = "darkroom-desktop"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"anyhow",
|
"anyhow",
|
||||||
"dr-plat",
|
"dr-plat",
|
||||||
@@ -1408,7 +1408,7 @@ checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-bench"
|
name = "dr-bench"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"anyhow",
|
"anyhow",
|
||||||
"dr-catalog",
|
"dr-catalog",
|
||||||
@@ -1425,7 +1425,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-catalog"
|
name = "dr-catalog"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-face",
|
"dr-face",
|
||||||
"dr-plat",
|
"dr-plat",
|
||||||
@@ -1440,7 +1440,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-decode"
|
name = "dr-decode"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-types",
|
"dr-types",
|
||||||
"env_logger",
|
"env_logger",
|
||||||
@@ -1454,7 +1454,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-export"
|
name = "dr-export"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-decode",
|
"dr-decode",
|
||||||
"dr-gpu",
|
"dr-gpu",
|
||||||
@@ -1473,7 +1473,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-face"
|
name = "dr-face"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-inference-engine",
|
"dr-inference-engine",
|
||||||
"env_logger",
|
"env_logger",
|
||||||
@@ -1486,7 +1486,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-film"
|
name = "dr-film"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"log",
|
"log",
|
||||||
"serde",
|
"serde",
|
||||||
@@ -1495,7 +1495,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-gpu"
|
name = "dr-gpu"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"bytemuck",
|
"bytemuck",
|
||||||
"dr-decode",
|
"dr-decode",
|
||||||
@@ -1513,8 +1513,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-inference-engine"
|
name = "dr-inference-engine"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
|
"env_logger",
|
||||||
"libloading",
|
"libloading",
|
||||||
"log",
|
"log",
|
||||||
"ort",
|
"ort",
|
||||||
@@ -1527,7 +1528,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-ingest"
|
name = "dr-ingest"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-plat",
|
"dr-plat",
|
||||||
"dr-types",
|
"dr-types",
|
||||||
@@ -1539,7 +1540,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-lens"
|
name = "dr-lens"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"lensfun",
|
"lensfun",
|
||||||
"log",
|
"log",
|
||||||
@@ -1547,7 +1548,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-pano"
|
name = "dr-pano"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-decode",
|
"dr-decode",
|
||||||
"dr-inference-engine",
|
"dr-inference-engine",
|
||||||
@@ -1561,7 +1562,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-pipeline"
|
name = "dr-pipeline"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-types",
|
"dr-types",
|
||||||
"log",
|
"log",
|
||||||
@@ -1570,7 +1571,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-plat"
|
name = "dr-plat"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"android-native-keyring-store",
|
"android-native-keyring-store",
|
||||||
"dr-types",
|
"dr-types",
|
||||||
@@ -1586,7 +1587,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-preset-xmp"
|
name = "dr-preset-xmp"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-pipeline",
|
"dr-pipeline",
|
||||||
"log",
|
"log",
|
||||||
@@ -1596,7 +1597,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-segment"
|
name = "dr-segment"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-inference-engine",
|
"dr-inference-engine",
|
||||||
"env_logger",
|
"env_logger",
|
||||||
@@ -1609,7 +1610,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-sync"
|
name = "dr-sync"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"async-trait",
|
"async-trait",
|
||||||
"dr-plat",
|
"dr-plat",
|
||||||
@@ -1623,7 +1624,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-sync-folder"
|
name = "dr-sync-folder"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"async-trait",
|
"async-trait",
|
||||||
"dr-sync",
|
"dr-sync",
|
||||||
@@ -1635,7 +1636,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-sync-nextcloud"
|
name = "dr-sync-nextcloud"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"async-trait",
|
"async-trait",
|
||||||
"dr-decode",
|
"dr-decode",
|
||||||
@@ -1657,7 +1658,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-thumbs"
|
name = "dr-thumbs"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-types",
|
"dr-types",
|
||||||
"jpeg-encoder",
|
"jpeg-encoder",
|
||||||
@@ -1669,7 +1670,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-types"
|
name = "dr-types"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
@@ -1678,7 +1679,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-ui"
|
name = "dr-ui"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"anyhow",
|
"anyhow",
|
||||||
"async-trait",
|
"async-trait",
|
||||||
@@ -1706,6 +1707,7 @@ dependencies = [
|
|||||||
"jni 0.22.4",
|
"jni 0.22.4",
|
||||||
"log",
|
"log",
|
||||||
"ndk-context",
|
"ndk-context",
|
||||||
|
"png",
|
||||||
"pollster",
|
"pollster",
|
||||||
"reqwest",
|
"reqwest",
|
||||||
"rusqlite",
|
"rusqlite",
|
||||||
@@ -1720,7 +1722,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "dr-xmp"
|
name = "dr-xmp"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"dr-types",
|
"dr-types",
|
||||||
"log",
|
"log",
|
||||||
@@ -7021,7 +7023,7 @@ checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "traceability"
|
name = "traceability"
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"anyhow",
|
"anyhow",
|
||||||
"serde",
|
"serde",
|
||||||
|
|||||||
+1
-1
@@ -29,7 +29,7 @@ members = [
|
|||||||
]
|
]
|
||||||
|
|
||||||
[workspace.package]
|
[workspace.package]
|
||||||
version = "0.13.5"
|
version = "0.13.6"
|
||||||
edition = "2021"
|
edition = "2021"
|
||||||
rust-version = "1.92"
|
rust-version = "1.92"
|
||||||
license = "GPL-3.0-or-later"
|
license = "GPL-3.0-or-later"
|
||||||
|
|||||||
@@ -28,7 +28,10 @@ ort-sys = { version = "2.0.0-rc.13", default-features = false, features = ["disa
|
|||||||
# The NVIDIA rungs exist on the desktop only. These features add `ort`'s
|
# The NVIDIA rungs exist on the desktop only. These features add `ort`'s
|
||||||
# option builders and nothing else — no linking under `alternative-backend` —
|
# option builders and nothing else — no linking under `alternative-backend` —
|
||||||
# but an Android binary has no business carrying even the option names, and
|
# but an Android binary has no business carrying even the option names, and
|
||||||
# the packaging must never be tempted to (§2, §3.1).
|
# the packaging must never be tempted to (§2, §3.1). The AMD rung needs no
|
||||||
|
# feature: MIGraphX is registered through the runtime's generic key/value
|
||||||
|
# entry point (`session::migraphx`), because `ort`'s own builder cannot
|
||||||
|
# name the compiled-program cache.
|
||||||
[target.'cfg(not(target_os = "android"))'.dependencies]
|
[target.'cfg(not(target_os = "android"))'.dependencies]
|
||||||
ort = { workspace = true, features = ["cuda", "tensorrt"] }
|
ort = { workspace = true, features = ["cuda", "tensorrt"] }
|
||||||
|
|
||||||
@@ -42,3 +45,8 @@ default = ["tract"]
|
|||||||
tract = ["dep:ort-tract"]
|
tract = ["dep:ort-tract"]
|
||||||
# Look for `libonnxruntime` on disk and hand its table to `ort`.
|
# Look for `libonnxruntime` on disk and hand its table to `ort`.
|
||||||
native = ["dep:libloading", "dep:ort-sys"]
|
native = ["dep:libloading", "dep:ort-sys"]
|
||||||
|
|
||||||
|
[dev-dependencies]
|
||||||
|
# The `ep_probe` example prints the provider's own diagnostics, which is most
|
||||||
|
# of what a failed rung tells you.
|
||||||
|
env_logger.workspace = true
|
||||||
|
|||||||
@@ -0,0 +1,207 @@
|
|||||||
|
//! Time each execution provider a runtime offers, on the models this
|
||||||
|
//! repository ships — the measurement docs/inference.md §1 requires before a
|
||||||
|
//! rung is added to §2's ladder.
|
||||||
|
//!
|
||||||
|
//! DARKROOM_ORT_DIR=/usr/lib \
|
||||||
|
//! cargo run --release -p dr-inference-engine --features native,tract \
|
||||||
|
//! --example ep_probe -- models/face/scrfd_500m_640.onnx ...
|
||||||
|
//!
|
||||||
|
//! Prints one row per (model, provider): the median of timed runs after
|
||||||
|
//! warm-ups, and the build time, which for a compiling provider is the
|
||||||
|
//! number that decides whether it needs an engine cache. MIGraphX is built
|
||||||
|
//! twice per precision — cold, then again from the cache it just wrote —
|
||||||
|
//! so both numbers are on the page.
|
||||||
|
//!
|
||||||
|
//! The ROCm provider is not in the list: ONNX Runtime removed it in 1.23,
|
||||||
|
//! and 1.29's `onnxruntime-rocm` ships `libonnxruntime_providers_migraphx.so`
|
||||||
|
//! and nothing else for AMD.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
#[derive(Clone, Copy, PartialEq)]
|
||||||
|
enum Ep {
|
||||||
|
Cpu,
|
||||||
|
MiGraphX,
|
||||||
|
MiGraphXFp16,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Ep {
|
||||||
|
fn label(self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
Ep::Cpu => "CPU",
|
||||||
|
Ep::MiGraphX => "MIGraphX f32",
|
||||||
|
Ep::MiGraphXFp16 => "MIGraphX fp16",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build(ep: Ep, bytes: &[u8], threads: usize, cache: &Path) -> ort::Result<ort::session::Session> {
|
||||||
|
let mut b = ort::session::Session::builder()?.with_intra_threads(threads)?;
|
||||||
|
match ep {
|
||||||
|
Ep::Cpu => {}
|
||||||
|
Ep::MiGraphX => migraphx(&mut b, false, &cache.join("f32"))?,
|
||||||
|
Ep::MiGraphXFp16 => migraphx(&mut b, true, &cache.join("fp16"))?,
|
||||||
|
}
|
||||||
|
b.commit_from_memory(bytes)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Register MIGraphX through the generic key/value API. `ort`'s own
|
||||||
|
/// builder fills the legacy `OrtMIGraphXProviderOptions`, which 1.29 reads
|
||||||
|
/// for its precision flags and nothing else: the model cache directory —
|
||||||
|
/// the difference between a 40 s load and a 0.3 s one — only travels this
|
||||||
|
/// way. The cache key is the graph, the GPU and the MIGraphX version, not
|
||||||
|
/// the precision, so each precision gets its own directory.
|
||||||
|
fn migraphx(
|
||||||
|
b: &mut ort::session::builder::SessionBuilder,
|
||||||
|
fp16: bool,
|
||||||
|
cache: &Path,
|
||||||
|
) -> ort::Result<()> {
|
||||||
|
use ort::AsPointer;
|
||||||
|
use std::ffi::CString;
|
||||||
|
std::fs::create_dir_all(cache).map_err(|e| ort::Error::new(e.to_string()))?;
|
||||||
|
let keys = [c"migraphx_fp16_enable", c"migraphx_model_cache_dir"];
|
||||||
|
let values = [
|
||||||
|
CString::new(if fp16 { "1" } else { "0" }).unwrap(),
|
||||||
|
CString::new(cache.to_string_lossy().as_bytes()).unwrap(),
|
||||||
|
];
|
||||||
|
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||||
|
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||||
|
// SAFETY: the documented C call, over arrays that outlive it; the
|
||||||
|
// runtime copies the strings into its own options map.
|
||||||
|
unsafe {
|
||||||
|
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
|
||||||
|
b.ptr_mut(),
|
||||||
|
c"MIGraphX".as_ptr(),
|
||||||
|
key_ptrs.as_ptr(),
|
||||||
|
value_ptrs.as_ptr(),
|
||||||
|
keys.len(),
|
||||||
|
);
|
||||||
|
ort::Error::result_from_status(status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Median of `runs` timed runs over zeros, in milliseconds, after warm-ups.
|
||||||
|
fn time(session: &mut ort::session::Session, warmups: usize, runs: usize) -> Result<f64, String> {
|
||||||
|
let shape: Vec<usize> = session.inputs()[0]
|
||||||
|
.dtype()
|
||||||
|
.tensor_shape()
|
||||||
|
.ok_or("input is not a tensor")?
|
||||||
|
.iter()
|
||||||
|
.map(|&d| if d > 0 { d as usize } else { 1 })
|
||||||
|
.collect();
|
||||||
|
let zeros = vec![0f32; shape.iter().product()];
|
||||||
|
let once = |s: &mut ort::session::Session| -> Result<f64, String> {
|
||||||
|
let input = ort::value::Tensor::from_array((shape.clone(), zeros.clone()))
|
||||||
|
.map_err(|e| e.to_string())?;
|
||||||
|
let t = Instant::now();
|
||||||
|
let out = s.run(ort::inputs![input]).map_err(|e| e.to_string())?;
|
||||||
|
let _ = out[0]
|
||||||
|
.try_extract_tensor::<f32>()
|
||||||
|
.map_err(|e| e.to_string())?;
|
||||||
|
Ok(t.elapsed().as_secs_f64() * 1e3)
|
||||||
|
};
|
||||||
|
for _ in 0..warmups {
|
||||||
|
once(session)?;
|
||||||
|
}
|
||||||
|
let mut times = Vec::with_capacity(runs);
|
||||||
|
for _ in 0..runs {
|
||||||
|
times.push(once(session)?);
|
||||||
|
}
|
||||||
|
times.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
||||||
|
Ok(times[times.len() / 2])
|
||||||
|
}
|
||||||
|
|
||||||
|
fn first_line(s: &str) -> String {
|
||||||
|
s.lines().next().unwrap_or("").chars().take(120).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
|
||||||
|
|
||||||
|
let models: Vec<PathBuf> = std::env::args_os().skip(1).map(PathBuf::from).collect();
|
||||||
|
if models.is_empty() {
|
||||||
|
eprintln!("usage: ep_probe MODEL.onnx [MODEL.onnx ...]");
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
|
||||||
|
dr_inference_engine::ensure_runtime();
|
||||||
|
let runtime = dr_inference_engine::status().runtime;
|
||||||
|
println!("runtime: {}", runtime.label());
|
||||||
|
if !runtime.is_native() {
|
||||||
|
println!("(tract: no provider to compare; set DARKROOM_ORT_DIR)");
|
||||||
|
}
|
||||||
|
|
||||||
|
let threads = std::thread::available_parallelism()
|
||||||
|
.map(|n| n.get().saturating_sub(2).max(1))
|
||||||
|
.unwrap_or(1);
|
||||||
|
println!("intra-op threads: {threads}");
|
||||||
|
let cache = std::env::temp_dir().join("darkroom-ep-probe");
|
||||||
|
let _ = std::fs::remove_dir_all(&cache);
|
||||||
|
println!("compiled-program cache: {}\n", cache.display());
|
||||||
|
|
||||||
|
println!(
|
||||||
|
"{:<28} {:<15} {:>10} {:>10}",
|
||||||
|
"model", "provider", "build s", "median ms"
|
||||||
|
);
|
||||||
|
for model in &models {
|
||||||
|
let bytes = match std::fs::read(model) {
|
||||||
|
Ok(b) => b,
|
||||||
|
Err(e) => {
|
||||||
|
println!("{:<28} read failed: {e}", name(model));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// A compiling provider is built twice: the second build reads the
|
||||||
|
// program the first wrote, and its time is what a launch after the
|
||||||
|
// first costs.
|
||||||
|
let plan = [
|
||||||
|
(Ep::Cpu, false),
|
||||||
|
(Ep::MiGraphX, false),
|
||||||
|
(Ep::MiGraphX, true),
|
||||||
|
(Ep::MiGraphXFp16, false),
|
||||||
|
(Ep::MiGraphXFp16, true),
|
||||||
|
];
|
||||||
|
for (ep, cached) in plan {
|
||||||
|
let started = Instant::now();
|
||||||
|
match build(ep, &bytes, threads, &cache) {
|
||||||
|
Ok(mut session) => {
|
||||||
|
let built = started.elapsed().as_secs_f64();
|
||||||
|
match time(&mut session, 3, 15) {
|
||||||
|
Ok(ms) => println!(
|
||||||
|
"{:<28} {:<15} {:>10.1} {:>10.1}{}",
|
||||||
|
name(model),
|
||||||
|
ep.label(),
|
||||||
|
built,
|
||||||
|
ms,
|
||||||
|
if cached { " (from cache)" } else { "" }
|
||||||
|
),
|
||||||
|
Err(e) => println!(
|
||||||
|
"{:<28} {:<15} {:>10.1} {:>10} {}",
|
||||||
|
name(model),
|
||||||
|
ep.label(),
|
||||||
|
built,
|
||||||
|
"ran ✗",
|
||||||
|
first_line(&e)
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(e) => println!(
|
||||||
|
"{:<28} {:<15} {:>21} {}",
|
||||||
|
name(model),
|
||||||
|
ep.label(),
|
||||||
|
"build ✗",
|
||||||
|
first_line(&e.to_string())
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
println!();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn name(p: &Path) -> String {
|
||||||
|
p.file_name()
|
||||||
|
.unwrap_or(p.as_os_str())
|
||||||
|
.to_string_lossy()
|
||||||
|
.into_owned()
|
||||||
|
}
|
||||||
@@ -0,0 +1,100 @@
|
|||||||
|
//! Walk the ladder as the app does — probe, engines, then a session — and
|
||||||
|
//! say what each step chose. The M5 check of docs/inference.md §6 without
|
||||||
|
//! the app around it.
|
||||||
|
//!
|
||||||
|
//! DARKROOM_ORT_DIR=/usr/lib \
|
||||||
|
//! cargo run --release -p dr-inference-engine --features native,tract \
|
||||||
|
//! --example ladder -- CACHE_DIR models/face/scrfd_500m_640.onnx [MODEL.onnx ...]
|
||||||
|
//!
|
||||||
|
//! Every model named is a `Detector` for the config's purposes, which is
|
||||||
|
//! enough to see the rung taken, the engines compiled and a session land
|
||||||
|
//! on it. Delete `CACHE_DIR` to see the first run again; keep it to see the
|
||||||
|
//! second.
|
||||||
|
|
||||||
|
use std::path::PathBuf;
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init();
|
||||||
|
let mut args = std::env::args_os().skip(1).map(PathBuf::from);
|
||||||
|
let (Some(cache_dir), models) = (args.next(), args.collect::<Vec<_>>()) else {
|
||||||
|
eprintln!("usage: ladder CACHE_DIR MODEL.onnx [MODEL.onnx ...]");
|
||||||
|
std::process::exit(2);
|
||||||
|
};
|
||||||
|
if models.is_empty() {
|
||||||
|
eprintln!("usage: ladder CACHE_DIR MODEL.onnx [MODEL.onnx ...]");
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
|
||||||
|
let runtime_dirs: Vec<PathBuf> = std::env::var_os("DARKROOM_ORT_DIR")
|
||||||
|
.map(PathBuf::from)
|
||||||
|
.into_iter()
|
||||||
|
.collect();
|
||||||
|
let started = Instant::now();
|
||||||
|
dr_inference_engine::init(dr_inference_engine::Config {
|
||||||
|
runtime_dirs,
|
||||||
|
cache_dir: cache_dir.clone(),
|
||||||
|
models: models
|
||||||
|
.iter()
|
||||||
|
.map(|p| (dr_inference_engine::Role::Detector, p.clone()))
|
||||||
|
.collect(),
|
||||||
|
embedded: Vec::new(),
|
||||||
|
ceiling: None,
|
||||||
|
threads: 0,
|
||||||
|
decay: Duration::ZERO,
|
||||||
|
});
|
||||||
|
|
||||||
|
let mut last = String::new();
|
||||||
|
loop {
|
||||||
|
let s = dr_inference_engine::status();
|
||||||
|
let line = format!(
|
||||||
|
"{} · {} · engines {}/{}{}",
|
||||||
|
s.line(),
|
||||||
|
if s.probing {
|
||||||
|
"probing"
|
||||||
|
} else {
|
||||||
|
s.reason.as_str()
|
||||||
|
},
|
||||||
|
s.engines.0,
|
||||||
|
s.engines.1,
|
||||||
|
if s.failed.is_empty() {
|
||||||
|
String::new()
|
||||||
|
} else {
|
||||||
|
format!(
|
||||||
|
" · tried {}",
|
||||||
|
s.failed
|
||||||
|
.iter()
|
||||||
|
.map(|(r, why)| format!("{}: {why}", r.label()))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(" · ")
|
||||||
|
)
|
||||||
|
}
|
||||||
|
);
|
||||||
|
if line != last {
|
||||||
|
println!("{:>6.1} s {line}", started.elapsed().as_secs_f64());
|
||||||
|
last = line;
|
||||||
|
}
|
||||||
|
if !s.probing && s.engines.0 >= s.engines.1 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
std::thread::sleep(Duration::from_millis(500));
|
||||||
|
}
|
||||||
|
|
||||||
|
for path in &models {
|
||||||
|
let bytes = std::fs::read(path).expect("read model");
|
||||||
|
let t = Instant::now();
|
||||||
|
let model = dr_inference_engine::open(
|
||||||
|
dr_inference_engine::Role::Detector,
|
||||||
|
dr_inference_engine::Form::F32,
|
||||||
|
&bytes,
|
||||||
|
)
|
||||||
|
.expect("open model");
|
||||||
|
let acquired = model.acquire().expect("acquire session");
|
||||||
|
println!(
|
||||||
|
"{} on {} in {:.2} s",
|
||||||
|
path.file_name().unwrap().to_string_lossy(),
|
||||||
|
acquired.rung().label(),
|
||||||
|
t.elapsed().as_secs_f64()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -3,8 +3,8 @@
|
|||||||
//!
|
//!
|
||||||
//! Consumers ask for a session by [`Role`] and get `ort`'s `Session` back;
|
//! Consumers ask for a session by [`Role`] and get `ort`'s `Session` back;
|
||||||
//! what built it — tract on one core, ONNX Runtime's CPU pool, a TensorRT
|
//! what built it — tract on one core, ONNX Runtime's CPU pool, a TensorRT
|
||||||
//! engine, the Hexagon — is this crate's business and shows up in
|
//! engine, a MIGraphX program, the Hexagon — is this crate's business and
|
||||||
//! [`status`] for the settings row and nowhere else.
|
//! shows up in [`status`] for the settings row and nowhere else.
|
||||||
//!
|
//!
|
||||||
//! The shape follows §3 of the spec: `ort` links nothing (`alternative-backend`),
|
//! The shape follows §3 of the spec: `ort` links nothing (`alternative-backend`),
|
||||||
//! and the first call hands it an API table from either a `libonnxruntime`
|
//! and the first call hands it an API table from either a `libonnxruntime`
|
||||||
@@ -60,7 +60,9 @@ pub enum Form {
|
|||||||
|
|
||||||
/// A rung of the ladder (§2). Ordered: a user override names the highest rung
|
/// A rung of the ladder (§2). Ordered: a user override names the highest rung
|
||||||
/// the probe may take, and a compiling rung falls back to the one below it
|
/// the probe may take, and a compiling rung falls back to the one below it
|
||||||
/// until its engine exists.
|
/// until its engine exists. The order is within a vendor's ladder — a
|
||||||
|
/// machine has NVIDIA rungs or an AMD rung, never both — so a ceiling is
|
||||||
|
/// read as "no higher than this on whichever ladder the device has".
|
||||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
|
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
|
||||||
pub enum Rung {
|
pub enum Rung {
|
||||||
/// ONNX Runtime's CPU provider, or tract when no runtime file was found.
|
/// ONNX Runtime's CPU provider, or tract when no runtime file was found.
|
||||||
@@ -69,6 +71,11 @@ pub enum Rung {
|
|||||||
Cuda,
|
Cuda,
|
||||||
/// NVIDIA, through a TensorRT engine compiled on this device. Desktop only.
|
/// NVIDIA, through a TensorRT engine compiled on this device. Desktop only.
|
||||||
TensorRt,
|
TensorRt,
|
||||||
|
/// AMD, through a MIGraphX program compiled on this device. Desktop
|
||||||
|
/// only. ONNX Runtime's ROCm provider, the CUDA provider's twin, was
|
||||||
|
/// removed in ONNX Runtime 1.23, so there is no non-compiling AMD rung
|
||||||
|
/// to fall back to: this one falls back to the CPU.
|
||||||
|
MiGraphX,
|
||||||
/// Qualcomm's Hexagon NPU through QNN, int8 models only. Android only.
|
/// Qualcomm's Hexagon NPU through QNN, int8 models only. Android only.
|
||||||
Hexagon,
|
Hexagon,
|
||||||
}
|
}
|
||||||
@@ -79,6 +86,7 @@ impl Rung {
|
|||||||
Rung::Cpu => "CPU",
|
Rung::Cpu => "CPU",
|
||||||
Rung::Cuda => "CUDA",
|
Rung::Cuda => "CUDA",
|
||||||
Rung::TensorRt => "TensorRT",
|
Rung::TensorRt => "TensorRT",
|
||||||
|
Rung::MiGraphX => "MIGraphX",
|
||||||
Rung::Hexagon => "Hexagon NPU",
|
Rung::Hexagon => "Hexagon NPU",
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -88,13 +96,13 @@ impl Rung {
|
|||||||
fn fallback(self) -> Rung {
|
fn fallback(self) -> Rung {
|
||||||
match self {
|
match self {
|
||||||
Rung::TensorRt => Rung::Cuda,
|
Rung::TensorRt => Rung::Cuda,
|
||||||
Rung::Hexagon | Rung::Cuda | Rung::Cpu => Rung::Cpu,
|
Rung::MiGraphX | Rung::Hexagon | Rung::Cuda | Rung::Cpu => Rung::Cpu,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether a session on this rung needs an engine built first.
|
/// Whether a session on this rung needs an engine built first.
|
||||||
fn compiles(self) -> bool {
|
fn compiles(self) -> bool {
|
||||||
matches!(self, Rung::TensorRt | Rung::Hexagon)
|
matches!(self, Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The model form this rung wants for a role.
|
/// The model form this rung wants for a role.
|
||||||
@@ -164,7 +172,7 @@ impl Status {
|
|||||||
pub fn line(&self) -> String {
|
pub fn line(&self) -> String {
|
||||||
let form = match self.rung {
|
let form = match self.rung {
|
||||||
Rung::Hexagon => " · int8",
|
Rung::Hexagon => " · int8",
|
||||||
Rung::TensorRt => " · fp16",
|
Rung::TensorRt | Rung::MiGraphX => " · fp16",
|
||||||
_ => "",
|
_ => "",
|
||||||
};
|
};
|
||||||
format!("{}{} · {}", self.rung.label(), form, self.runtime.label())
|
format!("{}{} · {}", self.rung.label(), form, self.runtime.label())
|
||||||
@@ -269,8 +277,8 @@ fn acquire(role: Role, form: Form, bytes: &Arc<[u8]>, hash: u64) -> Result<Acqui
|
|||||||
return Ok(Acquired { entry });
|
return Ok(Acquired { entry });
|
||||||
}
|
}
|
||||||
|
|
||||||
// Built outside the registry lock: a TensorRT engine load is long enough
|
// Built outside the registry lock: a TensorRT or MIGraphX engine load
|
||||||
// that another role's acquire should not wait on it.
|
// is long enough that another role's acquire should not wait on it.
|
||||||
let session = session::build(rung, role, bytes, &cfg)?;
|
let session = session::build(rung, role, bytes, &cfg)?;
|
||||||
log::debug!("inference: {role:?} loaded on {}", rung.label());
|
log::debug!("inference: {role:?} loaded on {}", rung.label());
|
||||||
let entry = Arc::new(Loaded {
|
let entry = Arc::new(Loaded {
|
||||||
@@ -401,7 +409,16 @@ pub fn status() -> Status {
|
|||||||
runtime: api::runtime(),
|
runtime: api::runtime(),
|
||||||
rung,
|
rung,
|
||||||
reason: s.cache.reason.clone(),
|
reason: s.cache.reason.clone(),
|
||||||
failed: s.cache.failed.clone(),
|
// Only what explains the selection: on an AMD machine the NVIDIA
|
||||||
|
// rungs "not enabled in this build" say nothing about why MIGraphX
|
||||||
|
// was taken. With the floor selected, everything tried is above it.
|
||||||
|
failed: s
|
||||||
|
.cache
|
||||||
|
.failed
|
||||||
|
.iter()
|
||||||
|
.filter(|(r, _)| *r > rung)
|
||||||
|
.cloned()
|
||||||
|
.collect(),
|
||||||
probing: s.probing,
|
probing: s.probing,
|
||||||
engines: if rung.compiles() {
|
engines: if rung.compiles() {
|
||||||
(s.cache.compiled.len(), s.wanted)
|
(s.cache.compiled.len(), s.wanted)
|
||||||
@@ -614,8 +631,33 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_status_reports_only_the_rungs_above_the_selection() {
|
||||||
|
let _serial = serial();
|
||||||
|
let failed = vec![
|
||||||
|
(Rung::TensorRt, "not enabled".to_string()),
|
||||||
|
(Rung::Cuda, "not enabled".to_string()),
|
||||||
|
];
|
||||||
|
let before = state().lock().unwrap().cache.clone();
|
||||||
|
state().lock().unwrap().cache = Cache {
|
||||||
|
rung: Some(Rung::MiGraphX),
|
||||||
|
failed: failed.clone(),
|
||||||
|
..Cache::default()
|
||||||
|
};
|
||||||
|
// An AMD desktop: the NVIDIA rungs below MIGraphX are not the story.
|
||||||
|
assert!(status().failed.is_empty());
|
||||||
|
// An NVIDIA desktop on the CUDA provider: TensorRT's failure is.
|
||||||
|
state().lock().unwrap().cache.rung = Some(Rung::Cuda);
|
||||||
|
assert_eq!(status().failed, vec![failed[0].clone()]);
|
||||||
|
// The floor: everything tried explains it.
|
||||||
|
state().lock().unwrap().cache.rung = Some(Rung::Cpu);
|
||||||
|
assert_eq!(status().failed.len(), 2);
|
||||||
|
state().lock().unwrap().cache = before;
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn the_status_line_reads_as_the_floor_before_init() {
|
fn the_status_line_reads_as_the_floor_before_init() {
|
||||||
|
let _serial = serial();
|
||||||
let s = status();
|
let s = status();
|
||||||
assert_eq!(s.rung, Rung::Cpu);
|
assert_eq!(s.rung, Rung::Cpu);
|
||||||
assert!(s.line().starts_with("CPU"), "{}", s.line());
|
assert!(s.line().starts_with("CPU"), "{}", s.line());
|
||||||
|
|||||||
@@ -16,8 +16,11 @@ use crate::{api::Runtime, state, Cache, Config, Form, Role, Rung};
|
|||||||
fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
|
fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
|
||||||
#[cfg(target_os = "android")]
|
#[cfg(target_os = "android")]
|
||||||
let all = [Rung::Hexagon];
|
let all = [Rung::Hexagon];
|
||||||
|
// A desktop has one vendor's GPU; the other vendor's providers are
|
||||||
|
// "not enabled in this build" or a library that fails to load, and
|
||||||
|
// either answer arrives in milliseconds.
|
||||||
#[cfg(not(target_os = "android"))]
|
#[cfg(not(target_os = "android"))]
|
||||||
let all = [Rung::TensorRt, Rung::Cuda];
|
let all = [Rung::TensorRt, Rung::Cuda, Rung::MiGraphX];
|
||||||
all.into_iter()
|
all.into_iter()
|
||||||
.filter(|r| ceiling.is_none_or(|c| *r <= c))
|
.filter(|r| ceiling.is_none_or(|c| *r <= c))
|
||||||
.collect()
|
.collect()
|
||||||
@@ -206,15 +209,22 @@ fn first_line(s: &str) -> String {
|
|||||||
line[start..].chars().take(200).collect()
|
line[start..].chars().take(200).collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Everything a change of which should re-probe: the runtime and where it
|
/// Everything a change of which should re-probe: the runtime, where it
|
||||||
/// came from, this crate, the platform, the driver or SoC, and the models.
|
/// came from and which providers sit beside it, this crate, the platform,
|
||||||
|
/// the driver or SoC, and the models.
|
||||||
fn fingerprint(runtime: &Runtime, cfg: &Config) -> String {
|
fn fingerprint(runtime: &Runtime, cfg: &Config) -> String {
|
||||||
let mut parts = vec![
|
let mut parts = vec![
|
||||||
format!("engine {}", env!("CARGO_PKG_VERSION")),
|
format!("engine {}", env!("CARGO_PKG_VERSION")),
|
||||||
format!("{} {}", std::env::consts::OS, std::env::consts::ARCH),
|
format!("{} {}", std::env::consts::OS, std::env::consts::ARCH),
|
||||||
match runtime {
|
match runtime {
|
||||||
Runtime::Tract => "tract".to_string(),
|
Runtime::Tract => "tract".to_string(),
|
||||||
Runtime::OnnxRuntime { path, version } => format!("ort {version} {}", path.display()),
|
Runtime::OnnxRuntime { path, version } => {
|
||||||
|
format!(
|
||||||
|
"ort {version} {} [{}]",
|
||||||
|
path.display(),
|
||||||
|
providers_beside(path)
|
||||||
|
)
|
||||||
|
}
|
||||||
},
|
},
|
||||||
device_identity(),
|
device_identity(),
|
||||||
];
|
];
|
||||||
@@ -237,13 +247,41 @@ fn fingerprint(runtime: &Runtime, cfg: &Config) -> String {
|
|||||||
parts.join("\n")
|
parts.join("\n")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The `libonnxruntime_providers_*.so` files in the runtime's directory.
|
||||||
|
/// A distribution's CPU-only and ROCm builds are the same version at the
|
||||||
|
/// same path; the provider libraries beside them are what differs.
|
||||||
|
fn providers_beside(runtime: &Path) -> String {
|
||||||
|
let Some(dir) = runtime.parent() else {
|
||||||
|
return String::new();
|
||||||
|
};
|
||||||
|
let mut names: Vec<String> = std::fs::read_dir(dir)
|
||||||
|
.into_iter()
|
||||||
|
.flatten()
|
||||||
|
.filter_map(|e| e.ok())
|
||||||
|
.filter_map(|e| e.file_name().into_string().ok())
|
||||||
|
.filter(|n| {
|
||||||
|
n.starts_with("libonnxruntime_providers_") || n.starts_with("onnxruntime_providers_")
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
names.sort();
|
||||||
|
names.join(" ")
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(target_os = "linux")]
|
#[cfg(target_os = "linux")]
|
||||||
fn device_identity() -> String {
|
fn device_identity() -> String {
|
||||||
// The NVIDIA driver's version line; absent means no NVIDIA driver.
|
// The NVIDIA driver's version line, or the ROCm release the AMD stack
|
||||||
std::fs::read_to_string("/proc/driver/nvidia/version")
|
// came from (`rocm-core` writes it; the kernel driver has no version
|
||||||
|
// of its own). Absent means neither.
|
||||||
|
if let Some(line) = std::fs::read_to_string("/proc/driver/nvidia/version")
|
||||||
.ok()
|
.ok()
|
||||||
.and_then(|s| s.lines().next().map(str::to_string))
|
.and_then(|s| s.lines().next().map(str::to_string))
|
||||||
.unwrap_or_else(|| "no nvidia driver".into())
|
{
|
||||||
|
return line;
|
||||||
|
}
|
||||||
|
if let Ok(rocm) = std::fs::read_to_string("/opt/rocm/.info/version") {
|
||||||
|
return format!("rocm {}", rocm.trim());
|
||||||
|
}
|
||||||
|
"no nvidia driver, no rocm".into()
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(target_os = "android")]
|
#[cfg(target_os = "android")]
|
||||||
|
|||||||
@@ -83,10 +83,65 @@ fn providers(
|
|||||||
ep::CUDA::default().build(),
|
ep::CUDA::default().build(),
|
||||||
])?)
|
])?)
|
||||||
}
|
}
|
||||||
|
Rung::MiGraphX => {
|
||||||
|
// fp16 on the same terms as TensorRT (§7). MIGraphX compiles a
|
||||||
|
// program per graph — 20–60 s here — and keeps it in the cache
|
||||||
|
// directory, keyed on the graph, the GPU and its own version
|
||||||
|
// but not the precision: hence one directory per precision.
|
||||||
|
// The CPU takes any node it declines.
|
||||||
|
let fp16 = role != Role::Embedder;
|
||||||
|
let cache = cfg
|
||||||
|
.cache_dir
|
||||||
|
.join("migraphx")
|
||||||
|
.join(if fp16 { "fp16" } else { "f32" });
|
||||||
|
let _ = std::fs::create_dir_all(&cache);
|
||||||
|
let mut b = b;
|
||||||
|
migraphx(&mut b, fp16, &cache)?;
|
||||||
|
Ok(b)
|
||||||
|
}
|
||||||
Rung::Hexagon => unreachable!("the Hexagon rung is not on a desktop ladder"),
|
Rung::Hexagon => unreachable!("the Hexagon rung is not on a desktop ladder"),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Register MIGraphX through ONNX Runtime's generic key/value entry point.
|
||||||
|
///
|
||||||
|
/// `ort`'s own builder (`ep::MIGraphX`) fills the legacy
|
||||||
|
/// `OrtMIGraphXProviderOptions`, and 1.29 reads that struct for its
|
||||||
|
/// precision flags and nothing else — the compiled-program cache directory
|
||||||
|
/// is only a key in the generic map (`migraphx_model_cache_dir`), and
|
||||||
|
/// without it every session is a full compile. Registration through the
|
||||||
|
/// generic entry point needs no `ort` feature: it is one call on the API
|
||||||
|
/// table, which is why the crate's `ort` dependency names no AMD feature.
|
||||||
|
#[cfg(not(target_os = "android"))]
|
||||||
|
fn migraphx(
|
||||||
|
b: &mut ort::session::builder::SessionBuilder,
|
||||||
|
fp16: bool,
|
||||||
|
cache: &std::path::Path,
|
||||||
|
) -> ort::Result<()> {
|
||||||
|
use ort::AsPointer;
|
||||||
|
use std::ffi::CString;
|
||||||
|
let keys = [c"migraphx_fp16_enable", c"migraphx_model_cache_dir"];
|
||||||
|
let values = [
|
||||||
|
CString::new(if fp16 { "1" } else { "0" }).unwrap(),
|
||||||
|
CString::new(cache.to_string_lossy().as_bytes())
|
||||||
|
.map_err(|e| ort::Error::new(e.to_string()))?,
|
||||||
|
];
|
||||||
|
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||||
|
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||||
|
// SAFETY: the documented C call over arrays that outlive it; the
|
||||||
|
// runtime copies the strings into its own options map before returning.
|
||||||
|
unsafe {
|
||||||
|
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
|
||||||
|
b.ptr_mut(),
|
||||||
|
c"MIGraphX".as_ptr(),
|
||||||
|
key_ptrs.as_ptr(),
|
||||||
|
value_ptrs.as_ptr(),
|
||||||
|
keys.len(),
|
||||||
|
);
|
||||||
|
ort::Error::result_from_status(status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(target_os = "android")]
|
#[cfg(target_os = "android")]
|
||||||
fn providers(
|
fn providers(
|
||||||
b: ort::session::builder::SessionBuilder,
|
b: ort::session::builder::SessionBuilder,
|
||||||
@@ -120,6 +175,8 @@ fn providers(
|
|||||||
.build()
|
.build()
|
||||||
.error_on_failure()])?)
|
.error_on_failure()])?)
|
||||||
}
|
}
|
||||||
Rung::Cuda | Rung::TensorRt => unreachable!("no NVIDIA rung on Android"),
|
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX => {
|
||||||
|
unreachable!("no desktop GPU rung on Android")
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -303,7 +303,7 @@ fn fill_once(
|
|||||||
}
|
}
|
||||||
|
|
||||||
// The padded canvas with mirrored context, and the hole within it.
|
// The padded canvas with mirrored context, and the hole within it.
|
||||||
let ctx = MirroredContext::build(rgb, width, height, known, mirror_depth);
|
let ctx = MirroredContext::build(rgb, width, height, known, mirror_depth, t);
|
||||||
let (pw, ph) = (ctx.width, ctx.height);
|
let (pw, ph) = (ctx.width, ctx.height);
|
||||||
|
|
||||||
// Tiles that touch the hole, on a grid that reaches both far edges.
|
// Tiles that touch the hole, on a grid that reaches both far edges.
|
||||||
@@ -499,16 +499,36 @@ struct MirroredContext {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl MirroredContext {
|
impl MirroredContext {
|
||||||
fn build(rgb: &[f32], width: usize, height: usize, known: &[bool], depth: usize) -> Self {
|
fn build(
|
||||||
|
rgb: &[f32],
|
||||||
|
width: usize,
|
||||||
|
height: usize,
|
||||||
|
known: &[bool],
|
||||||
|
depth: usize,
|
||||||
|
tile: usize,
|
||||||
|
) -> Self {
|
||||||
if depth == 0 {
|
if depth == 0 {
|
||||||
// Open: the picture as it is, the hole as it is. What the hole
|
// Open: the picture as it is, the hole as it is. What the hole
|
||||||
// holds does not matter — the model masks it out.
|
// holds does not matter — the model masks it out. A picture
|
||||||
|
// smaller than a tile (the merge page's preview) sits at the
|
||||||
|
// origin of a tile-sized canvas whose rest is hole: still the
|
||||||
|
// void as it is, and the only way a tile fits at all.
|
||||||
|
let (pw, ph) = (width.max(tile), height.max(tile));
|
||||||
|
let mut canvas = vec![0.0f32; pw * ph * 3];
|
||||||
|
let mut hole = vec![true; pw * ph];
|
||||||
|
for y in 0..height {
|
||||||
|
canvas[y * pw * 3..(y * pw + width) * 3]
|
||||||
|
.copy_from_slice(&rgb[y * width * 3..(y + 1) * width * 3]);
|
||||||
|
for x in 0..width {
|
||||||
|
hole[y * pw + x] = !known[y * width + x];
|
||||||
|
}
|
||||||
|
}
|
||||||
return MirroredContext {
|
return MirroredContext {
|
||||||
width,
|
width: pw,
|
||||||
height,
|
height: ph,
|
||||||
ring: 0,
|
ring: 0,
|
||||||
rgb: rgb.to_vec(),
|
rgb: canvas,
|
||||||
hole: known.iter().map(|&k| !k).collect(),
|
hole,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
let fold = |d: usize| fold(d, depth);
|
let fold = |d: usize| fold(d, depth);
|
||||||
@@ -715,6 +735,30 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_picture_smaller_than_the_tile_is_still_filled_when_the_void_is_open() {
|
||||||
|
// The merge page's preview is 1600 wide and a few hundred tall —
|
||||||
|
// shorter than a 512 tile. With no ring the canvas is padded to a
|
||||||
|
// tile, the padding hole, and the border is still filled.
|
||||||
|
let (mut rgb, known) = picture(300, 40, 8);
|
||||||
|
let mut model = Flat {
|
||||||
|
tile: 64,
|
||||||
|
seen: Vec::new(),
|
||||||
|
};
|
||||||
|
let tiles = fill_border(&mut rgb, 300, 40, &known, &mut model, test_params(0), &mut |_, _| {})
|
||||||
|
.unwrap();
|
||||||
|
assert!(tiles > 0, "no tile fitted a picture shorter than the tile");
|
||||||
|
for i in 0..300 * 40 {
|
||||||
|
if !known[i] {
|
||||||
|
assert!((rgb[i * 3] - 0.5).abs() < 1e-4, "pixel {i}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// And the model saw the padding as hole, never as black content.
|
||||||
|
for (_, k) in &model.seen {
|
||||||
|
assert_eq!(k.len(), 64 * 64);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn the_fine_passes_run_in_bands_after_the_coarse_one() {
|
fn the_fine_passes_run_in_bands_after_the_coarse_one() {
|
||||||
// A 150-tall hole above and below a picture: the coarse pass sees
|
// A 150-tall hole above and below a picture: the coarse pass sees
|
||||||
@@ -824,7 +868,7 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn the_context_mirrors_the_top_rows_upward() {
|
fn the_context_mirrors_the_top_rows_upward() {
|
||||||
let (rgb, known) = picture(40, 30, 5);
|
let (rgb, known) = picture(40, 30, 5);
|
||||||
let ctx = MirroredContext::build(&rgb, 40, 30, &known, 48);
|
let ctx = MirroredContext::build(&rgb, 40, 30, &known, 48, 64);
|
||||||
let x = RING + 10;
|
let x = RING + 10;
|
||||||
let first = RING + 5;
|
let first = RING + 5;
|
||||||
for k in 1..=4 {
|
for k in 1..=4 {
|
||||||
@@ -846,7 +890,7 @@ mod tests {
|
|||||||
for x in 0..40 {
|
for x in 0..40 {
|
||||||
rgb[(ridge * 40 + x) * 3..(ridge * 40 + x) * 3 + 3].copy_from_slice(&[0.9, 0.1, 0.1]);
|
rgb[(ridge * 40 + x) * 3..(ridge * 40 + x) * 3 + 3].copy_from_slice(&[0.9, 0.1, 0.1]);
|
||||||
}
|
}
|
||||||
let ctx = MirroredContext::build(&rgb, 40, 400, &known, 48);
|
let ctx = MirroredContext::build(&rgb, 40, 400, &known, 48, 64);
|
||||||
let x = RING + 10;
|
let x = RING + 10;
|
||||||
for y in 0..RING + 5 {
|
for y in 0..RING + 5 {
|
||||||
let p = (y * ctx.width + x) * 3;
|
let p = (y * ctx.width + x) * 3;
|
||||||
|
|||||||
+75
-15
@@ -75,7 +75,49 @@ TensorRT's job.
|
|||||||
worst: nearly five minutes), because it is compiling an engine for this exact GPU. The engine caches to disk and the second load is milliseconds. That
|
worst: nearly five minutes), because it is compiling an engine for this exact GPU. The engine caches to disk and the second load is milliseconds. That
|
||||||
number is what §6 is designed around.
|
number is what §6 is designed around.
|
||||||
|
|
||||||
### 1.3 What the numbers say
|
### 1.3 The desktop — Radeon RX 7900 XT, Threadripper 2920X, 24 threads · 2026-09-20
|
||||||
|
|
||||||
|
Arch's `onnxruntime-rocm` 1.29.0 against ROCm 7.2.4 and MIGraphX 7.2.3 (gfx1100), through the
|
||||||
|
same `ep_probe` harness (`core/dr-inference-engine/examples/ep_probe.rs`). Zero input, three
|
||||||
|
warm-ups, the median of 15 runs, on a machine doing nothing else. The build columns are the wall
|
||||||
|
clock of `Session` construction: cold is a MIGraphX compile of the graph for this GPU, cached is the
|
||||||
|
same session loading the program the cold build wrote.
|
||||||
|
|
||||||
|
| Model | ORT CPU f32 | MIGraphX f32 | **MIGraphX fp16** | Compile f32 / fp16 (s) | Cached load (s) |
|
||||||
|
|---|---|---|---|---|---|
|
||||||
|
| scrfd_500m (Fast) | 10.4 | 2.8 | **2.4** | 40 / 58 | 0.3 |
|
||||||
|
| scrfd_2.5g (Balanced) | 20.7 | 3.3 | **2.8** | 37 / 40 | 0.3 |
|
||||||
|
| scrfd_10g (Thorough) | 57.9 | 4.5 | **3.4** | 40 / 48 | 0.4 |
|
||||||
|
| arcface_mbf (per face) | 12.8 | 1.8 | 1.6 | 15 / 21 | 0.4 |
|
||||||
|
| yolo26n-seg | 49.3 | 8.4 | **7.5** | 110 / 136 | 0.9 |
|
||||||
|
| yolo26s-sem-ade20k | 55.8 | 4.8 | **3.8** | 50 / 60 | 0.5 |
|
||||||
|
| 2d106det (landmarks) | 9.9 | 1.2 | 1.0 | 16 / 21 | 0.2 |
|
||||||
|
| ocec_s (eye state) | 2.5 | 0.5 | 0.4 | 17 / 17 | 0.1 |
|
||||||
|
| xfeat-1024 | 26.8 | 10.5 | 9.9 | 37 / 53 | 0.3 |
|
||||||
|
| migan-512 (per tile) | 514 | 12.7 | **8.3** | 102 / 132 | 0.8 |
|
||||||
|
|
||||||
|
Three things the table settles.
|
||||||
|
|
||||||
|
- **The ROCm execution provider does not exist any more.** It was ONNX Runtime's CUDA-provider twin
|
||||||
|
for AMD, removed in the 1.23 release (AMD's builds dropped it from ROCm 7.1); 1.29's ROCm build ships
|
||||||
|
`libonnxruntime_providers_migraphx.so` and nothing else for AMD, and asking for `ROCm` answers
|
||||||
|
"not enabled in this build". So there is no non-compiling AMD rung to sit under MIGraphX the way
|
||||||
|
the CUDA provider sits under TensorRT: the AMD ladder is MIGraphX, then the CPU.
|
||||||
|
- **MIGraphX is a compiling provider, and its cache has to be asked for by name.** 15–135 s per
|
||||||
|
graph cold, under a second from its cache — TensorRT's shape exactly, and §6's design covers it.
|
||||||
|
Two things the provider does that the code has to know: `ort`'s builder fills the legacy options
|
||||||
|
struct, which 1.29 reads for the precision flags only, so the cache directory
|
||||||
|
(`migraphx_model_cache_dir`) reaches it only through the generic key/value registration; and
|
||||||
|
the cache key is the graph, the GPU and the MIGraphX version *without the precision*, so an fp16
|
||||||
|
session pointed at the f32 program's directory silently loads the f32 program (the first fp16
|
||||||
|
row measured here was that, before the directories were split).
|
||||||
|
- **fp16 is worth 10–35% over f32 on this card, not the 3× it is worth on TensorRT**, because
|
||||||
|
MIGraphX f32 is already 3–8× the CPU provider and the small graphs are launch-bound. The
|
||||||
|
detectors at 2.4–3.4 ms sit beside TensorRT fp16's 1.8–3.3 ms on the RTX 3050; the embedder is
|
||||||
|
1.6–1.8 ms on either precision and stays f32 (§7). The whole face pipeline for one image
|
||||||
|
(detector + landmarks + eyes + embedder) is under 6 ms.
|
||||||
|
|
||||||
|
### 1.4 What the numbers say
|
||||||
|
|
||||||
- **`tract` is single-threaded.** The tablet's one X4 core and one Raptor Lake core give the same
|
- **`tract` is single-threaded.** The tablet's one X4 core and one Raptor Lake core give the same
|
||||||
tract numbers. Replacing it with ONNX Runtime's CPU provider, *no accelerator involved*, is 3–6×
|
tract numbers. Replacing it with ONNX Runtime's CPU provider, *no accelerator involved*, is 3–6×
|
||||||
@@ -92,6 +134,8 @@ number is what §6 is designed around.
|
|||||||
- **On NVIDIA, TensorRT fp16 ≈ 3× the CUDA provider**, and the CUDA provider ≈ 2× the
|
- **On NVIDIA, TensorRT fp16 ≈ 3× the CUDA provider**, and the CUDA provider ≈ 2× the
|
||||||
multi-threaded CPU; at fp16 the detectors are 1.8–3.3 ms with no quantisation at all. Both leave the twenty cores free for decoding during a batch index, which the table does not
|
multi-threaded CPU; at fp16 the detectors are 1.8–3.3 ms with no quantisation at all. Both leave the twenty cores free for decoding during a batch index, which the table does not
|
||||||
show and which matters more than the ratio.
|
show and which matters more than the ratio.
|
||||||
|
- **On AMD, MIGraphX fp16 is 4–17× the CPU provider** on the detectors and 60× on the
|
||||||
|
inpainter, with the same first-run compile cost as TensorRT and no rung between it and the CPU.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -105,7 +149,8 @@ winning:
|
|||||||
| Android, Qualcomm with a Hexagon the shipped QNN skel covers (V68–V81) | QNN HTP, int8 model | ORT CPU, f32 model | — | tract |
|
| Android, Qualcomm with a Hexagon the shipped QNN skel covers (V68–V81) | QNN HTP, int8 model | ORT CPU, f32 model | — | tract |
|
||||||
| Android, any other SoC | ORT CPU, f32 | — | — | tract |
|
| Android, any other SoC | ORT CPU, f32 | — | — | tract |
|
||||||
| Linux / Windows, NVIDIA GPU | TensorRT, f32 model, fp16 engine | CUDA provider, f32 | ORT CPU, f32 | tract |
|
| Linux / Windows, NVIDIA GPU | TensorRT, f32 model, fp16 engine | CUDA provider, f32 | ORT CPU, f32 | tract |
|
||||||
| Linux / Windows, no NVIDIA | ORT CPU, f32 | — | — | tract |
|
| Linux, AMD GPU with ROCm | MIGraphX, f32 model, fp16 program | ORT CPU, f32 | — | tract |
|
||||||
|
| Linux / Windows, no GPU stack | ORT CPU, f32 | — | — | tract |
|
||||||
| macOS ⁵ | ORT CPU, f32 | — | — | tract |
|
| macOS ⁵ | ORT CPU, f32 | — | — | tract |
|
||||||
|
|
||||||
⁵ CoreML is the obvious rung and is unmeasured; it is listed so its absence is a gap and not an
|
⁵ CoreML is the obvious rung and is unmeasured; it is listed so its absence is a gap and not an
|
||||||
@@ -113,8 +158,13 @@ oversight.
|
|||||||
|
|
||||||
Deliberately **not** on any ladder, with the measurement that excluded each: NNAPI (no driver),
|
Deliberately **not** on any ladder, with the measurement that excluded each: NNAPI (no driver),
|
||||||
XNNPACK (slower than CPU, aborts on SCRFD), WebGPU (slower than CPU), the Adreno through QNN (works,
|
XNNPACK (slower than CPU, aborts on SCRFD), WebGPU (slower than CPU), the Adreno through QNN (works,
|
||||||
but never where the Hexagon does not also), CUDA int8 (slower than CUDA f32). A rung is added to this
|
but never where the Hexagon does not also), CUDA int8 (slower than CUDA f32), the ROCm provider
|
||||||
table by a measurement on this page, not by a provider existing.
|
(gone: §1.3). A rung is added to this table by a measurement on this page, not by a provider
|
||||||
|
existing.
|
||||||
|
|
||||||
|
The AMD ladder has no middle rung. TensorRT falls back to the CUDA provider while its engines
|
||||||
|
compile; MIGraphX has no such twin, so its fallback is the CPU provider, and the minute or two of
|
||||||
|
compiling on first run (§6) is spent at the floor's speed rather than at half the GPU's.
|
||||||
|
|
||||||
Two things the ladder is *not*: it is not a per-model choice — one backend serves every model on a
|
Two things the ladder is *not*: it is not a per-model choice — one backend serves every model on a
|
||||||
device, because §7's identity rule needs the detector and embedder on the same runtime for the
|
device, because §7's identity rule needs the detector and embedder on the same runtime for the
|
||||||
@@ -174,9 +224,10 @@ is cheaper than discovering it at packaging time.
|
|||||||
| ONNX Runtime | MIT | Yes |
|
| ONNX Runtime | MIT | Yes |
|
||||||
| Qualcomm QNN runtime (`com.qualcomm.qti:qnn-runtime` on Maven) | Qualcomm AI Engine Direct SDK licence — proprietary, redistribution permitted for applications using it | Yes for the APK, with the licence text shipped; not for a source distribution. **To be read in full, not summarised from memory, before the APK gains it.** |
|
| Qualcomm QNN runtime (`com.qualcomm.qti:qnn-runtime` on Maven) | Qualcomm AI Engine Direct SDK licence — proprietary, redistribution permitted for applications using it | Yes for the APK, with the licence text shipped; not for a source distribution. **To be read in full, not summarised from memory, before the APK gains it.** |
|
||||||
| CUDA runtime, cuDNN, TensorRT | NVIDIA EULAs — redistributable with an application, with the licence text, not modifiable | Yes for a package that bundles them. 600 MB. The alternative is to load them from the user's system install if present and skip the rung otherwise — which is what §4's probe does anyway. |
|
| CUDA runtime, cuDNN, TensorRT | NVIDIA EULAs — redistributable with an application, with the licence text, not modifiable | Yes for a package that bundles them. 600 MB. The alternative is to load them from the user's system install if present and skip the rung otherwise — which is what §4's probe does anyway. |
|
||||||
|
| ROCm (HIP, MIOpen, rocBLAS), MIGraphX | MIT | Yes, but the HIP SDK MIGraphX needs is ~15 GB installed. Same answer as NVIDIA: the user's system install, or the rung is skipped. |
|
||||||
|
|
||||||
The position this takes: the **NVIDIA libraries are not bundled**. The desktop package probes for a
|
The position this takes: the **GPU vendors' libraries are not bundled**. The desktop package probes for a
|
||||||
system CUDA/TensorRT install and uses it if it is version-compatible; a desktop without one runs on
|
system CUDA/TensorRT or ROCm/MIGraphX install and uses it if it is version-compatible; a desktop without one runs on
|
||||||
ORT CPU, which is still 8–10× today. Bundling 600 MB for a rung that is 2× again is not a trade
|
ORT CPU, which is still 8–10× today. Bundling 600 MB for a rung that is 2× again is not a trade
|
||||||
worth making unmeasured, and it can be revisited by a measurement on a batch index. The **QNN
|
worth making unmeasured, and it can be revisited by a measurement on a batch index. The **QNN
|
||||||
runtime is bundled** in the APK, because the Hexagon is the difference between a tablet that
|
runtime is bundled** in the APK, because the Hexagon is the difference between a tablet that
|
||||||
@@ -237,6 +288,7 @@ of which form they load:
|
|||||||
| f32 ONNX, shape-fixed, **opset ≥ 13** | `tools/fix-face-model-shapes.sh`, `tools/export-seg-model.sh` | Release time, once | Every rung except Hexagon |
|
| f32 ONNX, shape-fixed, **opset ≥ 13** | `tools/fix-face-model-shapes.sh`, `tools/export-seg-model.sh` | Release time, once | Every rung except Hexagon |
|
||||||
| int8 QDQ ONNX, per-channel, uint8 activations | `tools/quantise-models.sh` (new) | Release time, once, **calibrated on real photographs** | Hexagon |
|
| int8 QDQ ONNX, per-channel, uint8 activations | `tools/quantise-models.sh` (new) | Release time, once, **calibrated on real photographs** | Hexagon |
|
||||||
| TensorRT engine (`.engine`, per GPU architecture and TensorRT version) | The app, from the f32 file | First run on that device, in the background | TensorRT rung |
|
| TensorRT engine (`.engine`, per GPU architecture and TensorRT version) | The app, from the f32 file | First run on that device, in the background | TensorRT rung |
|
||||||
|
| MIGraphX program (`.mxr`, per GPU architecture, MIGraphX version and precision) | The app, from the f32 file | First run on that device, in the background | MIGraphX rung |
|
||||||
| QNN context binary | The app, from the int8 file | First run on that device, in the background | Hexagon rung |
|
| QNN context binary | The app, from the int8 file | First run on that device, in the background | Hexagon rung |
|
||||||
|
|
||||||
Two rules.
|
Two rules.
|
||||||
@@ -256,9 +308,12 @@ prefers. That is a change to the canonical file and so a change to the shipped m
|
|||||||
happens in the same model release as the int8 files.
|
happens in the same model release as the int8 files.
|
||||||
|
|
||||||
**Compilation is a device-time step, and it is cached.** A TensorRT engine is specific to the GPU
|
**Compilation is a device-time step, and it is cached.** A TensorRT engine is specific to the GPU
|
||||||
it was built on and the TensorRT that built it; a QNN context binary is specific to the Hexagon
|
it was built on and the TensorRT that built it; a MIGraphX program to the GPU and the MIGraphX
|
||||||
generation. Neither can ship. Both are built by the app the first time that rung is selected, in
|
that built it; a QNN context binary to the Hexagon generation. None can ship. All are built by
|
||||||
the background (§6), and written beside the probe cache keyed by the same inputs. They are
|
the app the first time that rung is selected, in the background (§6), and written beside the
|
||||||
|
probe cache keyed by the same inputs. MIGraphX's own key leaves out the precision, so the app
|
||||||
|
gives its f32 and fp16 programs separate directories — otherwise the embedder's f32 build would
|
||||||
|
be served the detector's fp16 program, or the reverse. They are
|
||||||
**derived, disposable, and regenerable**: deleting the cache directory costs the next launch a
|
**derived, disposable, and regenerable**: deleting the cache directory costs the next launch a
|
||||||
rebuild and nothing else, and the directory is excluded from anything that syncs (it is a peer of
|
rebuild and nothing else, and the directory is excluded from anything that syncs (it is a peer of
|
||||||
`thumbs`, not of the catalog).
|
`thumbs`, not of the catalog).
|
||||||
@@ -267,17 +322,18 @@ rebuild and nothing else, and the directory is excluded from anything that syncs
|
|||||||
|
|
||||||
## 6. First run — building engines without the user waiting for them
|
## 6. First run — building engines without the user waiting for them
|
||||||
|
|
||||||
The sequence on a device where a compiling rung (TensorRT, Hexagon) is selected:
|
The sequence on a device where a compiling rung (TensorRT, MIGraphX, Hexagon) is selected:
|
||||||
|
|
||||||
1. **Launch.** The runtime loads; the probe (§4) starts in the background; the app serves every
|
1. **Launch.** The runtime loads; the probe (§4) starts in the background; the app serves every
|
||||||
model request from the floor. Face indexing, segmentation and scene grading all work, at
|
model request from the floor. Face indexing, segmentation and scene grading all work, at
|
||||||
today's speed or better (ORT CPU).
|
today's speed or better (ORT CPU).
|
||||||
2. **Probe reports** — say, TensorRT. The compiling rung is now *selected* but has **no engines**.
|
2. **Probe reports** — say, TensorRT. The compiling rung is now *selected* but has **no engines**.
|
||||||
Model requests continue on the fallback rung below it (CUDA provider for TensorRT; ORT CPU for
|
Model requests continue on the fallback rung below it (CUDA provider for TensorRT; ORT CPU for
|
||||||
Hexagon), which needs no compilation and is already faster than the floor.
|
MIGraphX and Hexagon), which needs no compilation and is already faster than the floor.
|
||||||
3. **Engines build**, one model at a time, on a single low-priority background thread, smallest
|
3. **Engines build**, one model at a time, on a single low-priority background thread, smallest
|
||||||
model first so the detector — the one that runs per image — is ready soonest. On the reference
|
model first so the detector — the one that runs per image — is ready soonest. On the reference
|
||||||
desktop that is ~1 minute for the first detector and ~10 minutes for all six at fp16; on the tablet the QNN
|
desktop that is ~1 minute for the first detector and ~10 minutes for all six at fp16; on the AMD
|
||||||
|
desktop 40 s for the first detector and ~8 minutes for the set; on the tablet the QNN
|
||||||
context binaries take 0.8–1.7 s each and the whole set is ready before the user has opened a
|
context binaries take 0.8–1.7 s each and the whole set is ready before the user has opened a
|
||||||
library. Each engine is written to a temporary name and renamed into place, so a request never
|
library. Each engine is written to a temporary name and renamed into place, so a request never
|
||||||
sees a half-written file.
|
sees a half-written file.
|
||||||
@@ -313,7 +369,7 @@ enough to be *offered* at all. f32 on tract, ORT CPU, CUDA and TensorRT-f32 are
|
|||||||
same graph, the same arithmetic, differences at the last bit.
|
same graph, the same arithmetic, differences at the last bit.
|
||||||
|
|
||||||
**The embedder** is where comparability across devices is the whole point, and it is the one
|
**The embedder** is where comparability across devices is the whole point, and it is the one
|
||||||
model that no accelerator helps (§1.3). So: **the embedder runs in f32 on every rung.** On TensorRT
|
model that no accelerator helps (§1.4). So: **the embedder runs in f32 on every rung.** On TensorRT
|
||||||
that means the embedder's engine is built without fp16 while the detector's is built with it; on
|
that means the embedder's engine is built without fp16 while the detector's is built with it; on
|
||||||
the Hexagon it means the embedder is not on the NPU at all — it runs on the ORT CPU rung at 9 ms,
|
the Hexagon it means the embedder is not on the NPU at all — it runs on the ORT CPU rung at 9 ms,
|
||||||
and the ladder's "one backend per device" is, precisely, one backend *per model role*, with the
|
and the ladder's "one backend per device" is, precisely, one backend *per model role*, with the
|
||||||
@@ -344,7 +400,7 @@ core/dr-inference-engine
|
|||||||
src/probe.rs §4 — the ladder per platform, the session-build probe, the cache file
|
src/probe.rs §4 — the ladder per platform, the session-build probe, the cache file
|
||||||
src/engines.rs §6 — background compilation, the cache directory, progress
|
src/engines.rs §6 — background compilation, the cache directory, progress
|
||||||
src/session.rs open(role, bytes) -> ort::Session, applying the rung and the role's precision rule
|
src/session.rs open(role, bytes) -> ort::Session, applying the rung and the role's precision rule
|
||||||
src/api.rs the one unsafe block: dlopen libonnxruntime, fetch OrtApi, ort::set_api — or ort_tract::api()
|
src/api.rs the unsafe block that matters: dlopen libonnxruntime, fetch OrtApi, ort::set_api — or ort_tract::api()
|
||||||
```
|
```
|
||||||
|
|
||||||
- `dr-face` and `dr-segment` **delete** their private `install_backend` and their direct
|
- `dr-face` and `dr-segment` **delete** their private `install_backend` and their direct
|
||||||
@@ -354,7 +410,11 @@ core/dr-inference-engine
|
|||||||
`qnn`: those features add option builders, not linking, under `alternative-backend`. **Verified
|
`qnn`: those features add option builders, not linking, under `alternative-backend`. **Verified
|
||||||
for the QNN, CUDA and TensorRT builders on 2026-09-19** — they go through the API table's generic
|
for the QNN, CUDA and TensorRT builders on 2026-09-19** — they go through the API table's generic
|
||||||
`SessionOptionsAppendExecutionProvider*`. The NNAPI builder resolves a symbol directly and would
|
`SessionOptionsAppendExecutionProvider*`. The NNAPI builder resolves a symbol directly and would
|
||||||
not; it is not needed and is not enabled.
|
not; it is not needed and is not enabled. MIGraphX uses no `ort` feature at all: `ort`'s builder
|
||||||
|
fills the legacy options struct, which ONNX Runtime 1.29 reads for the precision flags and
|
||||||
|
nothing else, and the compiled-program cache directory only travels through the generic
|
||||||
|
key/value entry point (`migraphx_model_cache_dir`). `session::migraphx` makes that one call
|
||||||
|
on the API table itself.
|
||||||
- `dr-ui` owns the settings row, the about-screen line and the progress row; it holds one
|
- `dr-ui` owns the settings row, the about-screen line and the progress row; it holds one
|
||||||
`Sessions` per process, created at launch, and passes it down. `dr_ui::library` gains
|
`Sessions` per process, created at launch, and passes it down. `dr_ui::library` gains
|
||||||
`inference_cache_dir()` beside `shared_face_models_dir()`, on the same account-independent
|
`inference_cache_dir()` beside `shared_face_models_dir()`, on the same account-independent
|
||||||
|
|||||||
+21
-6
@@ -538,9 +538,24 @@ each loss weighting did, and the two runs abandoned (blur under L1 in
|
|||||||
the hole; a brick pattern under a strong adversarial term against a
|
the hole; a brick pattern under a strong adversarial term against a
|
||||||
discriminator that had not learned) — is `runs/` in `darkroom-infill`.
|
discriminator that had not learned) — is `runs/` in `darkroom-infill`.
|
||||||
|
|
||||||
**What is still wrong.** The ground fill is softer than its context —
|
**Second model, the same day.** The morning's fill was soft in the deep
|
||||||
texture, not structure, is what a night on a laptop GPU could not finish.
|
ground bands. The afternoon's run trained the generator against MI-GAN's
|
||||||
The levers, in order: a discriminator that learns (a pretrained one —
|
own pretrained discriminator (non-saturating loss, lazy R1, feature
|
||||||
MI-GAN's own from the unfused checkpoint — instead of a PatchGAN from
|
matching), with fresh noise inputs while training and flip/translation
|
||||||
scratch), feature matching, and more steps at 512. FR-MRG-4's
|
augmentation of the discriminator's input — both needed, or the generator
|
||||||
*experimental* stays.
|
settles into a periodic texture the discriminator cannot see. The shipped
|
||||||
|
weights (step 4 750 of `runs/border-v6`) are level with the stock model on
|
||||||
|
LPIPS (edge 0.125 / corner 0.187 against 0.121 / 0.183) while keeping the
|
||||||
|
PSNR gain (edge 18.0 / corner 15.7 against 16.9 / 14.6). On the fixture
|
||||||
|
the ground bands now carry texture at the right tone; at 1:1 a faint
|
||||||
|
regular hatch is visible in the deepest part.
|
||||||
|
|
||||||
|
**What is still wrong.** The hatch, and any deep textured void the
|
||||||
|
generator must invent. The better answer for those is not generative:
|
||||||
|
seed the void with the picture's own texture in hexagonal cells, let the
|
||||||
|
discriminator rank the candidates, and let the generator heal only the
|
||||||
|
gaps between cells — built and measured in `darkroom-infill`
|
||||||
|
(`infill/hexfill.py`), the most convincing scree corner produced so far,
|
||||||
|
and the next thing to port into `dr_pano::fill` (it needs the
|
||||||
|
discriminator as a second model, ~80 MB fp16). FR-MRG-4's *experimental*
|
||||||
|
stays.
|
||||||
|
|||||||
+12
-12
File diff suppressed because one or more lines are too long
+10
-10
@@ -308,7 +308,7 @@ This replaced a double tap, which had no visible state and could take forty phot
|
|||||||
|
|
||||||
Face indexing reads each face's eyes. The chip drops frames where the chosen people are caught blinking, and leaves sunglasses and eyes it could not read alone.
|
Face indexing reads each face's eyes. The chip drops frames where the chosen people are caught blinking, and leaves sunglasses and eyes it could not read alone.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:2106`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:2113`</sub>
|
||||||
|
|
||||||
### Find photographs with two people in them
|
### Find photographs with two people in them
|
||||||
|
|
||||||
@@ -317,7 +317,7 @@ Face indexing reads each face's eyes. The chip drops frames where the chosen peo
|
|||||||
|
|
||||||
"Any of them" is a union and "all of them" is an intersection. The tray is where both terms and the choice between them live, because a filter belongs on the filter bar.
|
"Any of them" is a union and "all of them" is an intersection. The tray is where both terms and the choice between them live, because a filter belongs on the filter bar.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:2135`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:2142`</sub>
|
||||||
|
|
||||||
### Resize the thumbnails
|
### Resize the thumbnails
|
||||||
|
|
||||||
@@ -326,7 +326,7 @@ Face indexing reads each face's eyes. The chip drops frames where the chosen peo
|
|||||||
|
|
||||||
There is no wheel on a tablet, so without the pinch the cell size could only be changed by a control a finger cannot reach.
|
There is no wheel on a tablet, so without the pinch the cell size could only be changed by a control a finger cannot reach.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:2795`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:2802`</sub>
|
||||||
|
|
||||||
### File photographs in a collection
|
### File photographs in a collection
|
||||||
|
|
||||||
@@ -335,7 +335,7 @@ There is no wheel on a tablet, so without the pinch the cell size could only be
|
|||||||
|
|
||||||
The selection is what the drag carries, which is why selecting several is worth the mode: forty photographs file in one gesture.
|
The selection is what the drag carries, which is why selecting several is worth the mode: forty photographs file in one gesture.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:2992`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:2999`</sub>
|
||||||
|
|
||||||
### Open a photograph
|
### Open a photograph
|
||||||
|
|
||||||
@@ -344,7 +344,7 @@ The selection is what the drag carries, which is why selecting several is worth
|
|||||||
|
|
||||||
A tap opens; a tap that *moved* does not. Travel is what separates a deliberate tap from a hand brushing past, and it is the only thing that does: the two are the same length. An earlier version required the finger to dwell 120 ms instead, and that rejected ordinary taps — a real tap is often quicker than a brush.
|
A tap opens; a tap that *moved* does not. Travel is what separates a deliberate tap from a hand brushing past, and it is the only thing that does: the two are the same length. An earlier version required the finger to dwell 120 ms instead, and that rejected ordinary taps — a real tap is often quicker than a brush.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:3260`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:3267`</sub>
|
||||||
|
|
||||||
### Rate a photograph without opening it
|
### Rate a photograph without opening it
|
||||||
|
|
||||||
@@ -354,7 +354,7 @@ A tap opens; a tap that *moved* does not. Travel is what separates a deliberate
|
|||||||
|
|
||||||
A star has to take the press without it also reaching the cell, or every rating throws the user into develop.
|
A star has to take the press without it also reaching the cell, or every rating throws the user into develop.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:3380`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:3387`</sub>
|
||||||
|
|
||||||
### Choose the frame a folded burst shows
|
### Choose the frame a folded burst shows
|
||||||
|
|
||||||
@@ -363,7 +363,7 @@ A star has to take the press without it also reaching the cell, or every rating
|
|||||||
|
|
||||||
A folded burst draws its earliest frame, which is a fact about the clock and not a judgement about the photograph — nothing in this application ranks a frame (FR-CULL-5). But the point of a burst is that one of the twelve is better than the other eleven, and the photographer is the only one who knows which. So the choice is offered on the frames themselves, while they are open and side by side, which is the one moment the alternatives are on screen to be compared.
|
A folded burst draws its earliest frame, which is a fact about the clock and not a judgement about the photograph — nothing in this application ranks a frame (FR-CULL-5). But the point of a burst is that one of the twelve is better than the other eleven, and the photographer is the only one who knows which. So the choice is offered on the frames themselves, while they are open and side by side, which is the one moment the alternatives are on screen to be compared.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:3511`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:3518`</sub>
|
||||||
|
|
||||||
### Drop the selection but keep selecting
|
### Drop the selection but keep selecting
|
||||||
|
|
||||||
@@ -372,7 +372,7 @@ A folded burst draws its earliest frame, which is a fact about the clock and not
|
|||||||
|
|
||||||
Distinct from Done, which leaves the mode entirely. Clearing keeps it, so the next selection can start straight away.
|
Distinct from Done, which leaves the mode entirely. Clearing keeps it, so the next selection can start straight away.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:4177`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:4184`</sub>
|
||||||
|
|
||||||
### Select everything the grid is showing
|
### Select everything the grid is showing
|
||||||
|
|
||||||
@@ -381,7 +381,7 @@ Distinct from Done, which leaves the mode entirely. Clearing keeps it, so the ne
|
|||||||
|
|
||||||
A scoped grid of two hundred frames is two hundred taps otherwise, and "all of them, except those three" is a far more common shape than the taps it took to say it.
|
A scoped grid of two hundred frames is two hundred taps otherwise, and "all of them, except those three" is a far more common shape than the taps it took to say it.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:4194`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:4201`</sub>
|
||||||
|
|
||||||
### Take photographs out of a collection
|
### Take photographs out of a collection
|
||||||
|
|
||||||
@@ -390,4 +390,4 @@ A scoped grid of two hundred frames is two hundred taps otherwise, and "all of t
|
|||||||
|
|
||||||
The badge on a cell says a photograph is filed in three collections and never which. This is the sheet that names them, and the only way out of one the grid is not currently scoped to.
|
The badge on a cell says a photograph is filed in three collections and never which. This is the sheet that names them, and the only way out of one the grid is not currently scoped to.
|
||||||
|
|
||||||
<sub>`ui/dr-ui/ui/library.slint:4318`</sub>
|
<sub>`ui/dr-ui/ui/library.slint:4325`</sub>
|
||||||
|
|||||||
@@ -74,6 +74,18 @@ photograph can be in several, and the badge on its cell counts them.
|
|||||||
|
|
||||||

|

|
||||||
|
|
||||||
|
Collections nest. Drag one onto another to put it inside; `+` with a
|
||||||
|
collection selected — or `New collection inside` from its menu — makes a
|
||||||
|
child. A parent shows everything its children hold, and its count says so.
|
||||||
|
Right-click a row (hold it, on a tablet) for the menu: rename, nest,
|
||||||
|
move back to the top level, keep it offline, delete.
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
## Developing a photograph
|
## Developing a photograph
|
||||||
|
|
||||||
Click a thumbnail to open it. The column on the right is every adjustment;
|
Click a thumbnail to open it. The column on the right is every adjustment;
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+1
-1
@@ -104,7 +104,7 @@ the dataset licence restricts models trained on it by name.
|
|||||||
|
|
||||||
| File | Source | Trained on | Used by |
|
| File | Source | Trained on | Used by |
|
||||||
|---|---|---|---|
|
|---|---|---|---|
|
||||||
| `inpaint/migan-512.onnx` | `migan_512_places2.pt` from `https://github.com/Picsart-AI-Research/MI-GAN` (Sargsyan et al., ICCV 2023), **fine-tuned** in the `darkroom-infill` repository (2026-09-20) | Places2 by the authors, then ~7 400 of the maintainer's own photographs with border-shaped voids | the panorama border fill (FR-MRG-4) |
|
| `inpaint/migan-512.onnx` | `migan_512_places2.pt` from `https://github.com/Picsart-AI-Research/MI-GAN` (Sargsyan et al., ICCV 2023), **fine-tuned** in the `darkroom-infill` repository (2026-09-20, second model that evening: trained against MI-GAN's own discriminator) | Places2 by the authors, then ~7 400 of the maintainer's own photographs with border-shaped voids | the panorama border fill (FR-MRG-4) |
|
||||||
|
|
||||||
The bare 512 generator at a fixed `1×4×512×512`, six operator types; the
|
The bare 512 generator at a fixed `1×4×512×512`, six operator types; the
|
||||||
tiling, the context and the blend are Rust (`dr_pano::fill`). Since
|
tiling, the context and the blend are Rust (`dr_pano::fill`). Since
|
||||||
|
|||||||
Binary file not shown.
+7
-2
@@ -4,7 +4,7 @@
|
|||||||
# makes `makepkg -si` in this directory install what you are actually working
|
# makes `makepkg -si` in this directory install what you are actually working
|
||||||
# on. Swap `source` for a tagged tarball when there is something to release.
|
# on. Swap `source` for a tagged tarball when there is something to release.
|
||||||
pkgname=darkroom
|
pkgname=darkroom
|
||||||
pkgver=0.13.5
|
pkgver=0.13.6
|
||||||
# Back to 1 with the version: a new pkgver is a new archive name, so there is
|
# Back to 1 with the version: a new pkgver is a new archive name, so there is
|
||||||
# nothing for makepkg to reuse and nothing for a release number to disambiguate.
|
# nothing for makepkg to reuse and nothing for a release number to disambiguate.
|
||||||
pkgrel=1
|
pkgrel=1
|
||||||
@@ -16,8 +16,13 @@ license=('GPL-3.0-or-later')
|
|||||||
# Nextcloud credentials (FR-NC-2) — gnome-keyring or kwallet both provide it.
|
# Nextcloud credentials (FR-NC-2) — gnome-keyring or kwallet both provide it.
|
||||||
depends=('vulkan-icd-loader' 'fontconfig' 'libxkbcommon')
|
depends=('vulkan-icd-loader' 'fontconfig' 'libxkbcommon')
|
||||||
makedepends=('cargo' 'git')
|
makedepends=('cargo' 'git')
|
||||||
|
# ONNX Runtime is loaded from /usr/lib at launch if a package put it there
|
||||||
|
# (docs/inference.md §3): the CPU build is 8–10× the built-in tract, the
|
||||||
|
# ROCm build adds the MIGraphX rung on an AMD GPU. Neither is required.
|
||||||
optdepends=('gnome-keyring: store Nextcloud credentials'
|
optdepends=('gnome-keyring: store Nextcloud credentials'
|
||||||
'kwallet: store Nextcloud credentials')
|
'kwallet: store Nextcloud credentials'
|
||||||
|
'onnxruntime-cpu: run the neural models on every core'
|
||||||
|
'onnxruntime-rocm: run the neural models on an AMD GPU')
|
||||||
options=('!lto') # the workspace sets its own LTO in Cargo.toml
|
options=('!lto') # the workspace sets its own LTO in Cargo.toml
|
||||||
|
|
||||||
_repo="$(cd "${startdir}/.." && pwd)"
|
_repo="$(cd "${startdir}/.." && pwd)"
|
||||||
|
|||||||
@@ -13,6 +13,11 @@
|
|||||||
# which is what this script exists to fix. Nothing NVIDIA is bundled here:
|
# which is what this script exists to fix. Nothing NVIDIA is bundled here:
|
||||||
# the providers load CUDA, cuDNN and TensorRT from the system, and if those
|
# the providers load CUDA, cuDNN and TensorRT from the system, and if those
|
||||||
# are missing the probe says so and the app stays on the CPU.
|
# are missing the probe says so and the app stays on the CPU.
|
||||||
|
#
|
||||||
|
# This is the NVIDIA script. On AMD there is nothing to fetch: the
|
||||||
|
# distribution's ROCm build of ONNX Runtime (Arch's `onnxruntime-rocm`)
|
||||||
|
# carries the MIGraphX provider, and the app finds it in the system library
|
||||||
|
# directory (docs/inference.md §1.3).
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
DEST="${1:-${XDG_DATA_HOME:-${HOME}/.local/share}/darkroom/runtime}"
|
DEST="${1:-${XDG_DATA_HOME:-${HOME}/.local/share}/darkroom/runtime}"
|
||||||
WORK="$(mktemp -d -p /var/tmp fetch-desktop-runtime.XXXXXX)"
|
WORK="$(mktemp -d -p /var/tmp fetch-desktop-runtime.XXXXXX)"
|
||||||
|
|||||||
@@ -122,6 +122,26 @@ def drag(x1, y1, x2, y2, steps=20):
|
|||||||
x('mouseup', 1)
|
x('mouseup', 1)
|
||||||
|
|
||||||
|
|
||||||
|
def drag_path(points, steps=20):
|
||||||
|
"""A drag through several points: out of the grid sideways first, then
|
||||||
|
to the row. A diagonal with much vertical in it is taken by the grid's
|
||||||
|
Flickable as a scroll before the DragArea can claim it."""
|
||||||
|
x('windowfocus', '--sync', win(), check=False)
|
||||||
|
(x1, y1), rest = points[0], points[1:]
|
||||||
|
move(x1, y1)
|
||||||
|
time.sleep(0.15)
|
||||||
|
x('mousedown', 1)
|
||||||
|
time.sleep(0.15)
|
||||||
|
for x2, y2 in rest:
|
||||||
|
for i in range(1, int(steps) + 1):
|
||||||
|
t = i / int(steps)
|
||||||
|
move(int(x1 + (x2 - x1) * t), int(y1 + (y2 - y1) * t))
|
||||||
|
time.sleep(0.03)
|
||||||
|
x1, y1 = x2, y2
|
||||||
|
time.sleep(0.4)
|
||||||
|
x('mouseup', 1)
|
||||||
|
|
||||||
|
|
||||||
def rec_start(out):
|
def rec_start(out):
|
||||||
size = x('getwindowgeometry', win()).split('Geometry: ')[1].strip()
|
size = x('getwindowgeometry', win()).split('Geometry: ')[1].strip()
|
||||||
ox, oy = geometry()
|
ox, oy = geometry()
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ shift || true
|
|||||||
export DR_HOME="${DR_HOME:-/var/tmp/dr-manual}"
|
export DR_HOME="${DR_HOME:-/var/tmp/dr-manual}"
|
||||||
export DR_DISPLAY="${DR_DISPLAY:-:7}"
|
export DR_DISPLAY="${DR_DISPLAY:-:7}"
|
||||||
export DR_BIN="${DR_BIN:-$repo/target/release/darkroom-desktop}"
|
export DR_BIN="${DR_BIN:-$repo/target/release/darkroom-desktop}"
|
||||||
|
export DR_LIBRARY="$library"
|
||||||
media="$repo/docs/manual/media"
|
media="$repo/docs/manual/media"
|
||||||
mkdir -p "$media" "$DR_HOME/xdg/config/darkroom" "$DR_HOME/xdg/data/darkroom"
|
mkdir -p "$media" "$DR_HOME/xdg/config/darkroom" "$DR_HOME/xdg/data/darkroom"
|
||||||
|
|
||||||
|
|||||||
+105
-7
@@ -11,6 +11,8 @@ scene, so a different library needs the numbers looked at again. Panel
|
|||||||
coordinates hold for any library.
|
coordinates hold for any library.
|
||||||
"""
|
"""
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
|
|
||||||
@@ -18,6 +20,7 @@ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|||||||
import drive as dr # noqa: E402
|
import drive as dr # noqa: E402
|
||||||
|
|
||||||
OUT = sys.argv[1] if len(sys.argv) > 1 else '.'
|
OUT = sys.argv[1] if len(sys.argv) > 1 else '.'
|
||||||
|
DEMO_LIBRARY = os.environ.get('DR_LIBRARY', '') # the folder the merge writes its DNG into
|
||||||
|
|
||||||
|
|
||||||
def shot(name):
|
def shot(name):
|
||||||
@@ -130,6 +133,9 @@ def library_thumbsize():
|
|||||||
|
|
||||||
|
|
||||||
def library_selection():
|
def library_selection():
|
||||||
|
# The grid remembers its scroll between launches; the cells below are
|
||||||
|
# the grid at its top.
|
||||||
|
wheel(-40, 900, 500); pause(0.6)
|
||||||
dr.click(*SELECT); pause(0.5)
|
dr.click(*SELECT); pause(0.5)
|
||||||
dr.click(*CELL_PANO_FIRST); pause(0.3)
|
dr.click(*CELL_PANO_FIRST); pause(0.3)
|
||||||
dr.x('keydown', 'shift'); dr.click(*CELL_NY_LAST); dr.x('keyup', 'shift'); pause(0.8)
|
dr.x('keydown', 'shift'); dr.click(*CELL_NY_LAST); dr.x('keyup', 'shift'); pause(0.8)
|
||||||
@@ -151,9 +157,21 @@ def library_keywords():
|
|||||||
dr.click(*SELECT); pause(0.5) # leave selecting
|
dr.click(*SELECT); pause(0.5) # leave selecting
|
||||||
|
|
||||||
|
|
||||||
|
def new_collection(name):
|
||||||
|
# `+` makes a collection inside whichever row is selected, so select the
|
||||||
|
# top first for a top-level one.
|
||||||
|
dr.click(60, 48); pause(0.5)
|
||||||
|
dr.click(212, 22); pause(0.4)
|
||||||
|
dr.x('type', '--delay', 90, name); dr.x('key', 'Return'); pause(1.2)
|
||||||
|
|
||||||
|
|
||||||
|
# The menu a right-click on a collection opens, at 1600×1100.
|
||||||
|
MENU = {'rename': (800, 496), 'inside': (800, 536), 'top': (800, 576),
|
||||||
|
'offline': (800, 616), 'delete': (800, 669), 'cancel': (800, 709)}
|
||||||
|
|
||||||
|
|
||||||
def library_collections():
|
def library_collections():
|
||||||
dr.click(212, 22); pause(0.3) # + in the collections header
|
new_collection('Alps')
|
||||||
dr.x('type', '--delay', 90, 'Alps'); dr.x('key', 'Return'); pause(1.2)
|
|
||||||
rec('library-collections')
|
rec('library-collections')
|
||||||
dr.drag(*CELL_PANO_FIRST, *COLLECTION_ROW, 40); pause(1.5)
|
dr.drag(*CELL_PANO_FIRST, *COLLECTION_ROW, 40); pause(1.5)
|
||||||
dr.click(*SELECT); pause(0.5)
|
dr.click(*SELECT); pause(0.5)
|
||||||
@@ -163,10 +181,34 @@ def library_collections():
|
|||||||
dr.click(*SELECT); pause(0.5)
|
dr.click(*SELECT); pause(0.5)
|
||||||
dr.click(*COLLECTION_ROW); pause(1.5)
|
dr.click(*COLLECTION_ROW); pause(1.5)
|
||||||
cut()
|
cut()
|
||||||
shot('library-collection')
|
|
||||||
dr.click(60, 48); pause(1.2) # All photographs
|
dr.click(60, 48); pause(1.2) # All photographs
|
||||||
|
|
||||||
|
|
||||||
|
def library_nesting():
|
||||||
|
# Continues from library_collections: "Alps" exists with photographs in
|
||||||
|
# it. Make "Trips", drag Alps into it, make "New York" inside Trips from
|
||||||
|
# the menu, file frames there, then open Trips to see it count both.
|
||||||
|
new_collection('Trips')
|
||||||
|
rec('library-nesting')
|
||||||
|
pause(0.5)
|
||||||
|
dr.drag(60, 90, 60, 128, 30); pause(1.5) # Alps onto Trips
|
||||||
|
dr.click(60, 90, 3); pause(1.2) # the menu, on Trips
|
||||||
|
dr.click(*MENU['inside']); pause(0.6)
|
||||||
|
dr.x('type', '--delay', 90, 'New York'); dr.x('key', 'Return'); pause(1.2)
|
||||||
|
dr.click(*SELECT); pause(0.5)
|
||||||
|
dr.click(963, 525); pause(0.3)
|
||||||
|
dr.x('keydown', 'shift'); dr.click(1502, 525); dr.x('keyup', 'shift'); pause(0.8)
|
||||||
|
dr.drag_path([(1143, 525), (200, 525), (90, 146)], 25); pause(1.5) # four frames into New York
|
||||||
|
dr.click(*SELECT); pause(0.5)
|
||||||
|
dr.click(60, 90); pause(1.5) # Trips: both children
|
||||||
|
cut()
|
||||||
|
shot('library-nesting')
|
||||||
|
dr.click(90, 146, 3); pause(1.2) # the menu, on New York
|
||||||
|
shot('library-collection-menu')
|
||||||
|
dr.click(*MENU['cancel']); pause(0.8)
|
||||||
|
dr.click(60, 48); pause(1.2)
|
||||||
|
|
||||||
|
|
||||||
# --- develop ----------------------------------------------------------------
|
# --- develop ----------------------------------------------------------------
|
||||||
def develop():
|
def develop():
|
||||||
open_chinatown()
|
open_chinatown()
|
||||||
@@ -286,14 +328,70 @@ def develop_export():
|
|||||||
|
|
||||||
|
|
||||||
# --- panorama ---------------------------------------------------------------
|
# --- panorama ---------------------------------------------------------------
|
||||||
|
def wait_for_new(folder, suffix, since, timeout):
|
||||||
|
"""Until a file ending in `suffix` newer than `since` appears in
|
||||||
|
`folder`'s tree — the merge's DNG — or `timeout` seconds pass."""
|
||||||
|
t0 = time.time()
|
||||||
|
while time.time() - t0 < timeout:
|
||||||
|
for root, _, files in os.walk(folder):
|
||||||
|
for f in files:
|
||||||
|
if f.lower().endswith(suffix) and os.path.getmtime(os.path.join(root, f)) > since:
|
||||||
|
return True
|
||||||
|
time.sleep(2)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def wait_until_lit(px, py, timeout):
|
||||||
|
"""Until the window pixel at (px, py) is no longer black — the fill
|
||||||
|
preview has reached the corner of the border — or `timeout` passes."""
|
||||||
|
t0 = time.time()
|
||||||
|
while time.time() - t0 < timeout:
|
||||||
|
out = subprocess.run(
|
||||||
|
['import', '-window', dr.win(), '-crop', f'1x1+{px}+{py}', '-depth', '8', 'txt:-'],
|
||||||
|
capture_output=True, text=True).stdout
|
||||||
|
m = re.search(r'\((\d+),(\d+),(\d+)', out)
|
||||||
|
if m and max(int(v) for v in m.groups()) > 24:
|
||||||
|
return True
|
||||||
|
time.sleep(3)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def join(name, parts):
|
||||||
|
"""Concatenate recorded clips into `name`.mp4 and drop the parts."""
|
||||||
|
lst = f'{OUT}/{name}.txt'
|
||||||
|
with open(lst, 'w') as f:
|
||||||
|
for p in parts:
|
||||||
|
f.write(f"file '{OUT}/{p}.mp4'\n")
|
||||||
|
subprocess.run(['ffmpeg', '-hide_banner', '-loglevel', 'error', '-y', '-f', 'concat', '-safe', '0',
|
||||||
|
'-i', lst, '-c', 'copy', f'{OUT}/{name}.mp4'], check=True)
|
||||||
|
os.remove(lst)
|
||||||
|
for p in parts:
|
||||||
|
os.remove(f'{OUT}/{p}.mp4')
|
||||||
|
|
||||||
|
|
||||||
def panorama():
|
def panorama():
|
||||||
to_library()
|
to_library()
|
||||||
library_selection()
|
library_selection()
|
||||||
rec('panorama')
|
rec('panorama-a')
|
||||||
dr.click(1325, 1079); pause(40) # Merge to panorama
|
dr.click(1325, 1079); pause(40) # Merge to panorama
|
||||||
dr.click(184, 637); pause(45) # Fill the border
|
shot('panorama-aligned')
|
||||||
dr.click(1543, 22); pause(60) # Merge
|
dr.click(184, 637); pause(6) # Fill the border
|
||||||
cut()
|
cut()
|
||||||
|
# The preview fills at the working scale — minutes on the CPU rung —
|
||||||
|
# so the film pauses until the border's corner has been painted.
|
||||||
|
wait_until_lit(60, 110, 400)
|
||||||
|
pause(3)
|
||||||
|
shot('panorama-filled')
|
||||||
|
rec('panorama-b')
|
||||||
|
pause(2)
|
||||||
|
started = time.time()
|
||||||
|
dr.click(1543, 22); pause(8) # Merge
|
||||||
|
cut()
|
||||||
|
join('panorama', ['panorama-a', 'panorama-b'])
|
||||||
|
# The merge itself takes minutes too; the last picture is the
|
||||||
|
# composite in the grid.
|
||||||
|
wait_for_new(DEMO_LIBRARY, '.dng', started, 600) if DEMO_LIBRARY else pause(60)
|
||||||
|
pause(5)
|
||||||
shot('panorama-done')
|
shot('panorama-done')
|
||||||
dr.click(*BACK); pause(2)
|
dr.click(*BACK); pause(2)
|
||||||
|
|
||||||
@@ -309,7 +407,7 @@ def settings():
|
|||||||
|
|
||||||
ALL = [
|
ALL = [
|
||||||
'library', 'library_rating', 'library_timeline', 'library_thumbsize',
|
'library', 'library_rating', 'library_timeline', 'library_thumbsize',
|
||||||
'library_selection', 'library_keywords', 'library_collections',
|
'library_selection', 'library_keywords', 'library_collections', 'library_nesting',
|
||||||
'develop', 'develop_groups', 'develop_light', 'develop_zoom', 'develop_wb',
|
'develop', 'develop_groups', 'develop_light', 'develop_zoom', 'develop_wb',
|
||||||
'compose', 'local_segment', 'local_paint', 'local_done', 'repair', 'film',
|
'compose', 'local_segment', 'local_paint', 'local_done', 'repair', 'film',
|
||||||
'presets', 'panorama', 'settings',
|
'presets', 'panorama', 'settings',
|
||||||
|
|||||||
@@ -53,6 +53,10 @@ dr-face = { workspace = true, features = ["inference"] }
|
|||||||
# runtime file; the build stays C-free either way (docs/dev/inference.md §3).
|
# runtime file; the build stays C-free either way (docs/dev/inference.md §3).
|
||||||
dr-inference-engine = { workspace = true, features = ["native"] }
|
dr-inference-engine = { workspace = true, features = ["native"] }
|
||||||
dr-thumbs.workspace = true
|
dr-thumbs.workspace = true
|
||||||
|
# For the drag ghost only: the composite under the cursor has to reach the
|
||||||
|
# renderer through a file, and the fan of thumbnails needs alpha, which the
|
||||||
|
# thumbnail store's JPEG cannot carry. See `collections_ui::drag_image_via_file`.
|
||||||
|
png = "0.18"
|
||||||
# The library module writes scan results straight into the catalog, so it
|
# The library module writes scan results straight into the catalog, so it
|
||||||
# needs the same SQLite types dr-catalog exposes.
|
# needs the same SQLite types dr-catalog exposes.
|
||||||
rusqlite.workspace = true
|
rusqlite.workspace = true
|
||||||
|
|||||||
@@ -86,6 +86,107 @@ pub(super) fn compose_drag_image(thumbs: &[slint::Image]) -> slint::Image {
|
|||||||
slint::Image::from_rgba8_premultiplied(canvas)
|
slint::Image::from_rgba8_premultiplied(canvas)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Hand the composite to the renderer by way of a file.
|
||||||
|
///
|
||||||
|
/// **A workaround for a renderer fault, and it should read as one.** The
|
||||||
|
/// ghost under the cursor is drawn by Slint's own drag overlay, which
|
||||||
|
/// uploads the image as a texture, draws it, and drops the texture in the
|
||||||
|
/// same call. With the wgpu FemtoVG renderer that drop is immediate and
|
||||||
|
/// the draw is deferred to the frame's flush, so by the time the frame is
|
||||||
|
/// rendered the texture is gone and the renderer binds its placeholder
|
||||||
|
/// instead — a solid red rectangle the size of the ghost. An image with a
|
||||||
|
/// cache key is kept in the renderer's texture cache until after the
|
||||||
|
/// flush; an image built from pixels has none, and only a path gives one.
|
||||||
|
/// So the composite goes to disk as a PNG and comes back through
|
||||||
|
/// `load_from_path`. Slint 1.17.1, `draw_image_direct` in the FemtoVG
|
||||||
|
/// item renderer; the GL FemtoVG renderer is not affected.
|
||||||
|
///
|
||||||
|
/// One file per drag, named uniquely: the core caches decoded images by
|
||||||
|
/// path, so reusing a name would show the previous drag's ghost. The file
|
||||||
|
/// is removed when the drag ends, or when the next one begins.
|
||||||
|
///
|
||||||
|
/// If anything on the way fails the composite is handed over as it is,
|
||||||
|
/// which on the affected renderer draws the placeholder — no worse than
|
||||||
|
/// before, and a log line says why.
|
||||||
|
pub(super) fn drag_image_via_file(composite: slint::Image) -> slint::Image {
|
||||||
|
let Some(buffer) = composite.to_rgba8() else {
|
||||||
|
log::warn!("drag ghost: the composite has no pixels to write");
|
||||||
|
return composite;
|
||||||
|
};
|
||||||
|
log::debug!("drag ghost: {}×{}", buffer.width(), buffer.height());
|
||||||
|
if buffer.width() == 0 || buffer.height() == 0 {
|
||||||
|
return composite;
|
||||||
|
}
|
||||||
|
match write_drag_image(&buffer) {
|
||||||
|
Ok(path) => match slint::Image::load_from_path(&path) {
|
||||||
|
Ok(image) => {
|
||||||
|
forget_drag_image_file();
|
||||||
|
*DRAG_IMAGE_FILE.lock().unwrap() = Some(path);
|
||||||
|
image
|
||||||
|
}
|
||||||
|
Err(_) => {
|
||||||
|
log::warn!("drag ghost: {} did not load back", path.display());
|
||||||
|
let _ = std::fs::remove_file(&path);
|
||||||
|
composite
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Err(e) => {
|
||||||
|
log::warn!("drag ghost: {e}");
|
||||||
|
composite
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The file the current drag's ghost is loaded from, if any.
|
||||||
|
static DRAG_IMAGE_FILE: std::sync::Mutex<Option<std::path::PathBuf>> = std::sync::Mutex::new(None);
|
||||||
|
|
||||||
|
fn write_drag_image(
|
||||||
|
buffer: &slint::SharedPixelBuffer<slint::Rgba8Pixel>,
|
||||||
|
) -> std::io::Result<std::path::PathBuf> {
|
||||||
|
use std::sync::atomic::{AtomicU64, Ordering};
|
||||||
|
static SERIAL: AtomicU64 = AtomicU64::new(0);
|
||||||
|
|
||||||
|
let dir = crate::library::scratch_dir();
|
||||||
|
std::fs::create_dir_all(&dir)?;
|
||||||
|
let path = dir.join(format!(
|
||||||
|
"drag-{}-{}.png",
|
||||||
|
std::process::id(),
|
||||||
|
SERIAL.fetch_add(1, Ordering::Relaxed)
|
||||||
|
));
|
||||||
|
let file = std::fs::File::create(&path)?;
|
||||||
|
let mut encoder = png::Encoder::new(
|
||||||
|
std::io::BufWriter::new(file),
|
||||||
|
buffer.width(),
|
||||||
|
buffer.height(),
|
||||||
|
);
|
||||||
|
encoder.set_color(png::ColorType::Rgba);
|
||||||
|
encoder.set_depth(png::BitDepth::Eight);
|
||||||
|
// Fastest: this is a 160px bitmap written once per drag and read once.
|
||||||
|
encoder.set_compression(png::Compression::Fastest);
|
||||||
|
let mut writer = encoder.write_header().map_err(std::io::Error::other)?;
|
||||||
|
writer
|
||||||
|
.write_image_data(buffer.as_bytes())
|
||||||
|
.map_err(std::io::Error::other)?;
|
||||||
|
writer.finish().map_err(std::io::Error::other)?;
|
||||||
|
Ok(path)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn forget_drag_image_file() {
|
||||||
|
if let Some(path) = DRAG_IMAGE_FILE.lock().unwrap().take() {
|
||||||
|
let _ = std::fs::remove_file(path);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Draw one thumbnail into the composite, scaled to `tw`×`th` at `dx`,`dy`.
|
||||||
|
///
|
||||||
|
/// Nearest-neighbour: this is a transient 160px cursor bitmap, and a filtered
|
||||||
|
/// resample would cost more than it could visibly buy. `dim` darkens the layers
|
||||||
|
/// beneath the top one so the stack reads as depth rather than as a smear.
|
||||||
|
///
|
||||||
|
/// The buffer is premultiplied, so the alpha applied here is baked into the
|
||||||
|
/// colour channels as well.
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
|
||||||
/// Draw one thumbnail into the composite, scaled to `tw`×`th` at `dx`,`dy`.
|
/// Draw one thumbnail into the composite, scaled to `tw`×`th` at `dx`,`dy`.
|
||||||
///
|
///
|
||||||
/// Nearest-neighbour: this is a transient 160px cursor bitmap, and a filtered
|
/// Nearest-neighbour: this is a transient 160px cursor bitmap, and a filtered
|
||||||
|
|||||||
@@ -19,7 +19,10 @@ use crate::library;
|
|||||||
use crate::{AppWindow, Collections};
|
use crate::{AppWindow, Collections};
|
||||||
|
|
||||||
use super::controller::{decide_drop, CollectionsController, Drop};
|
use super::controller::{decide_drop, CollectionsController, Drop};
|
||||||
use super::drag::{arm_hold, arm_spring, collapse_spring_opened, compose_drag_image};
|
use super::drag::{
|
||||||
|
arm_hold, arm_spring, collapse_spring_opened, compose_drag_image, drag_image_via_file,
|
||||||
|
forget_drag_image_file,
|
||||||
|
};
|
||||||
use super::press::select_row;
|
use super::press::select_row;
|
||||||
use super::rename_menu::unique_name;
|
use super::rename_menu::unique_name;
|
||||||
use super::trash::{drain_trash, start_restore, start_trash};
|
use super::trash::{drain_trash, start_restore, start_trash};
|
||||||
@@ -603,7 +606,7 @@ fn wire_drag(
|
|||||||
.map(|c| c.thumbnail)
|
.map(|c| c.thumbnail)
|
||||||
.collect()
|
.collect()
|
||||||
};
|
};
|
||||||
w.set_library_drag_image(compose_drag_image(&thumbs));
|
w.set_library_drag_image(drag_image_via_file(compose_drag_image(&thumbs)));
|
||||||
|
|
||||||
*ctl.dragging.borrow_mut() = carried.clone();
|
*ctl.dragging.borrow_mut() = carried.clone();
|
||||||
sync_selection(&w, &ctl, &ids);
|
sync_selection(&w, &ctl, &ids);
|
||||||
@@ -763,6 +766,7 @@ fn wire_drag(
|
|||||||
// released — it holds a copy of every thumbnail it composited.
|
// released — it holds a copy of every thumbnail it composited.
|
||||||
sync_lifted(&w, &[], &visible());
|
sync_lifted(&w, &[], &visible());
|
||||||
w.set_library_drag_image(slint::Image::default());
|
w.set_library_drag_image(slint::Image::default());
|
||||||
|
forget_drag_image_file();
|
||||||
|
|
||||||
if landed.is_some() {
|
if landed.is_some() {
|
||||||
let borrow = catalog.borrow();
|
let borrow = catalog.borrow();
|
||||||
|
|||||||
+24
-20
@@ -19,23 +19,26 @@ use dr_types::FaceDetector;
|
|||||||
/// fingerprints the model files and a probe before they land would be a
|
/// fingerprints the model files and a probe before they land would be a
|
||||||
/// probe of nothing.
|
/// probe of nothing.
|
||||||
pub fn init(runtime_dirs: Vec<PathBuf>) {
|
pub fn init(runtime_dirs: Vec<PathBuf>) {
|
||||||
let dir = crate::library::shared_face_models_dir();
|
// Each file where the app will actually load it from — the user's
|
||||||
let mut models: Vec<(Role, PathBuf)> = FaceDetector::ALL
|
// shared directory, else the package's — so a fresh install with models
|
||||||
|
// only under `/usr/share` probes and compiles for them rather than
|
||||||
|
// finding nothing and settling on the CPU.
|
||||||
|
let mut wanted: Vec<(Role, &str)> = FaceDetector::ALL
|
||||||
.iter()
|
.iter()
|
||||||
.map(|d| (Role::Detector, dir.join(d.file_name())))
|
.map(|d| (Role::Detector, d.file_name()))
|
||||||
|
.collect();
|
||||||
|
wanted.extend([
|
||||||
|
(Role::Embedder, "arcface_mbf_b1.onnx"),
|
||||||
|
(Role::Scene, "yolo26s-sem-ade20k.onnx"),
|
||||||
|
(Role::Landmarks, crate::library::LANDMARK_MODEL),
|
||||||
|
(Role::EyeClassifier, crate::library::EYE_MODEL),
|
||||||
|
(Role::EyeClassifier, crate::library::SUNGLASSES_MODEL),
|
||||||
|
(Role::Inpainter, crate::library::INPAINT_MODEL),
|
||||||
|
]);
|
||||||
|
let models: Vec<(Role, PathBuf)> = wanted
|
||||||
|
.into_iter()
|
||||||
|
.filter_map(|(role, name)| Some((role, crate::library::shared_model(name)?)))
|
||||||
.collect();
|
.collect();
|
||||||
models.push((Role::Embedder, dir.join("arcface_mbf_b1.onnx")));
|
|
||||||
models.push((Role::Scene, dir.join("yolo26s-sem-ade20k.onnx")));
|
|
||||||
models.push((Role::Landmarks, dir.join(crate::library::LANDMARK_MODEL)));
|
|
||||||
models.push((Role::EyeClassifier, dir.join(crate::library::EYE_MODEL)));
|
|
||||||
models.push((
|
|
||||||
Role::EyeClassifier,
|
|
||||||
dir.join(crate::library::SUNGLASSES_MODEL),
|
|
||||||
));
|
|
||||||
if let Some(p) = crate::library::inpaint_model() {
|
|
||||||
models.push((Role::Inpainter, p));
|
|
||||||
}
|
|
||||||
models.retain(|(_, p)| p.is_file());
|
|
||||||
|
|
||||||
dr_inference_engine::init(dr_inference_engine::Config {
|
dr_inference_engine::init(dr_inference_engine::Config {
|
||||||
runtime_dirs,
|
runtime_dirs,
|
||||||
@@ -75,12 +78,13 @@ pub fn user_runtime_dir() -> PathBuf {
|
|||||||
/// Which form the current backend loads `detector` in, given the files on
|
/// Which form the current backend loads `detector` in, given the files on
|
||||||
/// this device — the fact `faces.model_id` has to carry (§7).
|
/// this device — the fact `faces.model_id` has to carry (§7).
|
||||||
///
|
///
|
||||||
/// Reads the shared directory only. An account-private model directory can
|
/// Reads the shared and system directories only. An account-private model
|
||||||
/// override the file `library::face_models` loads, but not which form the
|
/// directory can override the file `library::face_models` loads, but not
|
||||||
/// backend wants, and the int8 sibling is something a packager ships, not
|
/// which form the backend wants, and the int8 sibling is something a
|
||||||
/// something a user drops in.
|
/// packager ships, not something a user drops in.
|
||||||
pub fn detector_form(detector: FaceDetector) -> Form {
|
pub fn detector_form(detector: FaceDetector) -> Form {
|
||||||
let canonical = crate::library::shared_face_models_dir().join(detector.file_name());
|
let canonical = crate::library::shared_model(detector.file_name())
|
||||||
|
.unwrap_or_else(|| crate::library::shared_face_models_dir().join(detector.file_name()));
|
||||||
dr_inference_engine::resolve_model(Role::Detector, &canonical).1
|
dr_inference_engine::resolve_model(Role::Detector, &canonical).1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -227,6 +227,15 @@ pub fn inference_cache_dir() -> PathBuf {
|
|||||||
data_root().join("inference")
|
data_root().join("inference")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Somewhere to put a file that only needs to exist for a moment.
|
||||||
|
///
|
||||||
|
/// Under the data root rather than `std::env::temp_dir()`, which on Android
|
||||||
|
/// names a directory the app cannot write to. Nothing here survives a
|
||||||
|
/// launch on purpose: whoever writes into it deletes what they wrote.
|
||||||
|
pub fn scratch_dir() -> PathBuf {
|
||||||
|
data_root().join("scratch")
|
||||||
|
}
|
||||||
|
|
||||||
/// The detector and embedder files, if both are present — and the eye
|
/// The detector and embedder files, if both are present — and the eye
|
||||||
/// models beside them, if those are.
|
/// models beside them, if those are.
|
||||||
///
|
///
|
||||||
@@ -341,6 +350,18 @@ pub(super) fn system_face_models_dirs() -> Vec<PathBuf> {
|
|||||||
.collect()
|
.collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The file `name` in the shared user models directory, else in the first
|
||||||
|
/// system directory that has it — the search every account-independent
|
||||||
|
/// model lookup makes, and the one the inference engine is told about, so
|
||||||
|
/// that a package's models under `/usr/share` are probed and compiled for
|
||||||
|
/// exactly as a user's own would be.
|
||||||
|
pub fn shared_model(name: &str) -> Option<PathBuf> {
|
||||||
|
std::iter::once(shared_face_models_dir())
|
||||||
|
.chain(system_face_models_dirs())
|
||||||
|
.map(|d| d.join(name))
|
||||||
|
.find(|p| p.is_file())
|
||||||
|
}
|
||||||
|
|
||||||
/// TRACES: FR-MRG-4
|
/// TRACES: FR-MRG-4
|
||||||
/// The panorama border filler, as shipped in `models/inpaint/`.
|
/// The panorama border filler, as shipped in `models/inpaint/`.
|
||||||
pub const INPAINT_MODEL: &str = "migan-512.onnx";
|
pub const INPAINT_MODEL: &str = "migan-512.onnx";
|
||||||
@@ -351,10 +372,7 @@ pub const INPAINT_MODEL: &str = "migan-512.onnx";
|
|||||||
/// the per-account step, because a fill is not identity-bearing and no
|
/// the per-account step, because a fill is not identity-bearing and no
|
||||||
/// library has a reason to pin its own.
|
/// library has a reason to pin its own.
|
||||||
pub fn inpaint_model() -> Option<PathBuf> {
|
pub fn inpaint_model() -> Option<PathBuf> {
|
||||||
std::iter::once(shared_face_models_dir())
|
shared_model(INPAINT_MODEL)
|
||||||
.chain(system_face_models_dirs())
|
|
||||||
.map(|d| d.join(INPAINT_MODEL))
|
|
||||||
.find(|p| p.is_file())
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
|
|||||||
@@ -1771,6 +1771,13 @@ export component LibraryGrid inherits Rectangle {
|
|||||||
// Reported out so Rust can place month headings: a heading belongs on a
|
// Reported out so Rust can place month headings: a heading belongs on a
|
||||||
// cell that begins a row, and only the grid knows how wide a row is.
|
// cell that begins a row, and only the grid knows how wide a row is.
|
||||||
changed columns => { root.columns-changed(root.columns); }
|
changed columns => { root.columns-changed(root.columns); }
|
||||||
|
// And once on creation. `changed` reports a change, and the first
|
||||||
|
// evaluation is not one — so a grid built after the window has already
|
||||||
|
// settled at its size never said how wide it was, and Rust went on
|
||||||
|
// placing headings for the one column it had been told about at
|
||||||
|
// start-up. Every month then began a row and was announced wherever
|
||||||
|
// its first cell fell, mid-row included.
|
||||||
|
init => { root.columns-changed(root.columns); }
|
||||||
property <int> row-count: ceil(root.cells.length / max(1, columns));
|
property <int> row-count: ceil(root.cells.length / max(1, columns));
|
||||||
|
|
||||||
// --- while a pinch is happening, and just after -----------------------
|
// --- while a pinch is happening, and just after -----------------------
|
||||||
|
|||||||
Reference in New Issue
Block a user