Compare commits
12
Commits
v0.22.1
...
7966bf2dd8
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7966bf2dd8 | ||
|
|
4822991bec | ||
|
|
4b33754482 | ||
|
|
b3999bbc0c | ||
|
|
43402bfe5b | ||
|
|
cb97ebe7ac | ||
|
|
2ced6f114f | ||
|
|
85dee4375b | ||
|
|
87c405eb46 | ||
|
|
555ec0efb3 | ||
|
|
0291b80672 | ||
|
|
ef71bb3289 |
@@ -499,6 +499,12 @@ jobs:
|
||||
WANT=$(ls docs/manual/media | wc -l)
|
||||
GOT=$(ls "$INST/manual/media" | wc -l)
|
||||
[ "$GOT" = "$WANT" ] || { echo "FAIL: expected $WANT manual pictures, installed $GOT"; exit 1; }
|
||||
# Both bundled runtimes, each with its provider beside it
|
||||
# (tools/fetch-bundled-runtimes.sh).
|
||||
for f in openvino/onnxruntime.dll openvino/onnxruntime_providers_openvino.dll \
|
||||
openvino/openvino.dll webgpu/onnxruntime.dll webgpu/dxcompiler.dll; do
|
||||
[ -f "$INST/runtimes/$f" ] || { echo "FAIL: runtimes/$f not installed"; exit 1; }
|
||||
done
|
||||
wine reg query 'HKCU\Software\Microsoft\Windows\CurrentVersion\Uninstall\DarkRoom' 2>/dev/null \
|
||||
| grep -q DisplayVersion || { echo "FAIL: no uninstall registry key"; exit 1; }
|
||||
wine "$INST/darkroom.exe" --version 2>/dev/null | grep -q '^darkroom-desktop ' \
|
||||
|
||||
Generated
+26
-26
@@ -1265,7 +1265,7 @@ checksum = "f27ae1dd37df86211c42e150270f82743308803d90a6f6e6651cd730d5e1732f"
|
||||
|
||||
[[package]]
|
||||
name = "darkroom-android"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"android_logger",
|
||||
"dr-plat",
|
||||
@@ -1278,7 +1278,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "darkroom-desktop"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-plat",
|
||||
@@ -1454,7 +1454,7 @@ checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76"
|
||||
|
||||
[[package]]
|
||||
name = "dr-bench"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-catalog",
|
||||
@@ -1471,7 +1471,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-catalog"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-face",
|
||||
"dr-plat",
|
||||
@@ -1486,7 +1486,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-decode"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"env_logger",
|
||||
@@ -1500,7 +1500,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-denoise"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-gpu",
|
||||
@@ -1517,7 +1517,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-export"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-gpu",
|
||||
@@ -1536,7 +1536,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-face"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-inference-engine",
|
||||
"env_logger",
|
||||
@@ -1549,7 +1549,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-film"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"log",
|
||||
"serde",
|
||||
@@ -1558,7 +1558,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-gpu"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
"dr-decode",
|
||||
@@ -1576,7 +1576,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-inference-engine"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"libloading",
|
||||
@@ -1591,7 +1591,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-ingest"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-plat",
|
||||
"dr-types",
|
||||
@@ -1603,7 +1603,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-lens"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"lensfun",
|
||||
"log",
|
||||
@@ -1611,7 +1611,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-pano"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-decode",
|
||||
"dr-inference-engine",
|
||||
@@ -1625,7 +1625,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-pipeline"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"log",
|
||||
@@ -1634,7 +1634,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-plat"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"android-native-keyring-store",
|
||||
"dr-types",
|
||||
@@ -1650,7 +1650,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-preset-xmp"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-pipeline",
|
||||
"log",
|
||||
@@ -1660,7 +1660,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-segment"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-inference-engine",
|
||||
"env_logger",
|
||||
@@ -1673,7 +1673,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-plat",
|
||||
@@ -1687,7 +1687,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync-folder"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-sync",
|
||||
@@ -1699,7 +1699,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-sync-nextcloud"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"dr-decode",
|
||||
@@ -1721,7 +1721,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-thumbs"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"jpeg-encoder",
|
||||
@@ -1733,7 +1733,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-types"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -1742,7 +1742,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-ui"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"async-trait",
|
||||
@@ -1792,7 +1792,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "dr-xmp"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"dr-types",
|
||||
"log",
|
||||
@@ -7126,7 +7126,7 @@ checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3"
|
||||
|
||||
[[package]]
|
||||
name = "traceability"
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"proc-macro2",
|
||||
|
||||
+1
-1
@@ -33,7 +33,7 @@ members = [
|
||||
exclude = ["third_party"]
|
||||
|
||||
[workspace.package]
|
||||
version = "0.22.1"
|
||||
version = "0.23.0"
|
||||
edition = "2021"
|
||||
rust-version = "1.92"
|
||||
license = "GPL-3.0-or-later"
|
||||
|
||||
@@ -201,7 +201,7 @@ controls, its place in the chain and its tests.
|
||||
|
||||
## Where it stands
|
||||
|
||||
**0.22.1**, thirty-seven tagged releases in. 193 numbered requirements in
|
||||
**0.23.0**, thirty-eight tagged releases in. 193 numbered requirements in
|
||||
scope, 85% of them claimed by code and [traced to it](docs/dev/traceability.md);
|
||||
the rest are written down rather than merely absent.
|
||||
|
||||
|
||||
@@ -451,8 +451,13 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
// on a first launch they were not on disk until this line. The runtime
|
||||
// is in the APK's native library directory beside `libdarkroom.so`,
|
||||
// which is also where Qualcomm's DSP loader has to be pointed for the
|
||||
// Hexagon skel (docs/dev/inference.md §3, §8).
|
||||
dr_ui::inference::init(native_library_dir().into_iter().collect());
|
||||
// Hexagon skel (docs/dev/inference.md §3, §8). Two of them: the QNN
|
||||
// build, and the generic WebGPU build by its file name, which the
|
||||
// engine opens only when the first does not fit the SoC (§3.2).
|
||||
let runtimes = native_library_dir()
|
||||
.map(|dir| vec![dir.clone(), dir.join("libonnxruntime_generic.so")])
|
||||
.unwrap_or_default();
|
||||
dr_ui::inference::init(runtimes);
|
||||
}
|
||||
|
||||
/// The directory the system unpacked this APK's native libraries into.
|
||||
|
||||
@@ -106,24 +106,37 @@ fn main() -> anyhow::Result<()> {
|
||||
/// providers, or against the wrong cuDNN — and a system copy whose providers
|
||||
/// do not load is not a problem, only a slower app: the probe builds a real
|
||||
/// session before believing a provider.
|
||||
///
|
||||
/// The order breaks ties only. The engine opens every runtime on this list
|
||||
/// and loads the one whose providers fit the GPU (inference.md §3.2), so a
|
||||
/// package's bundled builds — `runtimes/openvino` and `runtimes/webgpu`
|
||||
/// beside each place a package installs to, from
|
||||
/// `tools/fetch-bundled-runtimes.sh` — sit beside a CUDA or ROCm runtime
|
||||
/// without hiding it.
|
||||
fn runtime_dirs() -> Vec<PathBuf> {
|
||||
// A place a package installs to, and the bundled runtimes under it.
|
||||
fn packaged(dirs: &mut Vec<PathBuf>, base: PathBuf) {
|
||||
dirs.push(base.join("runtimes/openvino"));
|
||||
dirs.push(base.join("runtimes/webgpu"));
|
||||
dirs.push(base);
|
||||
}
|
||||
let mut dirs = Vec::new();
|
||||
if let Some(dir) = std::env::var_os("DARKROOM_ORT_DIR") {
|
||||
dirs.push(PathBuf::from(dir));
|
||||
}
|
||||
if let Ok(exe) = std::env::current_exe() {
|
||||
if let Some(bin) = exe.parent() {
|
||||
dirs.push(bin.to_path_buf());
|
||||
dirs.push(bin.join("../lib/darkroom"));
|
||||
packaged(&mut dirs, bin.to_path_buf());
|
||||
packaged(&mut dirs, bin.join("../lib/darkroom"));
|
||||
}
|
||||
}
|
||||
dirs.push(dr_ui::inference::user_runtime_dir());
|
||||
#[cfg(target_os = "linux")]
|
||||
dirs.extend([
|
||||
PathBuf::from("/app/lib/darkroom"),
|
||||
PathBuf::from("/usr/lib/darkroom"),
|
||||
PathBuf::from("/usr/lib"),
|
||||
]);
|
||||
{
|
||||
packaged(&mut dirs, PathBuf::from("/app/lib/darkroom"));
|
||||
packaged(&mut dirs, PathBuf::from("/usr/lib/darkroom"));
|
||||
dirs.push(PathBuf::from("/usr/lib"));
|
||||
}
|
||||
// An app bundle keeps its libraries in `Contents/Frameworks`, beside
|
||||
// the `Contents/MacOS` the executable is in; then Homebrew's
|
||||
// `onnxruntime`, Apple silicon's prefix before Intel's. Homebrew's build
|
||||
|
||||
@@ -24,6 +24,14 @@ enum Ep {
|
||||
Cpu,
|
||||
MiGraphX,
|
||||
MiGraphXFp16,
|
||||
OpenVinoCpu,
|
||||
OpenVinoGpu,
|
||||
OpenVinoGpuFp16,
|
||||
OpenVinoNpu,
|
||||
/// Dawn's low-power adapter: the integrated GPU on a hybrid machine.
|
||||
WebGpuLow,
|
||||
/// Dawn's high-performance adapter: the discrete one, if there is one.
|
||||
WebGpuHigh,
|
||||
}
|
||||
|
||||
impl Ep {
|
||||
@@ -32,20 +40,113 @@ impl Ep {
|
||||
Ep::Cpu => "CPU",
|
||||
Ep::MiGraphX => "MIGraphX f32",
|
||||
Ep::MiGraphXFp16 => "MIGraphX fp16",
|
||||
Ep::OpenVinoCpu => "OpenVINO CPU",
|
||||
Ep::OpenVinoGpu => "OpenVINO GPU",
|
||||
Ep::OpenVinoGpuFp16 => "OpenVINO GPU16",
|
||||
Ep::OpenVinoNpu => "OpenVINO NPU",
|
||||
Ep::WebGpuLow => "WebGPU low",
|
||||
Ep::WebGpuHigh => "WebGPU high",
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a second build reads what the first one compiled.
|
||||
fn caches(self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Ep::MiGraphX
|
||||
| Ep::MiGraphXFp16
|
||||
| Ep::OpenVinoGpu
|
||||
| Ep::OpenVinoGpuFp16
|
||||
| Ep::OpenVinoNpu
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
fn build(ep: Ep, bytes: &[u8], threads: usize, cache: &Path) -> ort::Result<ort::session::Session> {
|
||||
let mut b = ort::session::Session::builder()?.with_intra_threads(threads)?;
|
||||
let dir = |sub: &str| {
|
||||
let d = cache.join(sub);
|
||||
let _ = std::fs::create_dir_all(&d);
|
||||
d.to_string_lossy().into_owned()
|
||||
};
|
||||
// `GPU` is OpenVINO's first OpenCL GPU, which on a hybrid laptop can be
|
||||
// the discrete NVIDIA one; DARKROOM_OV_GPU=GPU.1 names another.
|
||||
let gpu = std::env::var("DARKROOM_OV_GPU").unwrap_or_else(|_| "GPU".into());
|
||||
match ep {
|
||||
Ep::Cpu => {}
|
||||
Ep::MiGraphX => migraphx(&mut b, false, &cache.join("f32"))?,
|
||||
Ep::MiGraphXFp16 => migraphx(&mut b, true, &cache.join("fp16"))?,
|
||||
// Option names as `openvino_provider_factory.cc` reads them at 1.24.
|
||||
Ep::OpenVinoCpu => append(&mut b, c"OpenVINO", &[("device_type", "CPU".into())])?,
|
||||
Ep::OpenVinoGpu => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[
|
||||
("device_type", gpu.clone()),
|
||||
("precision", "FP32".into()),
|
||||
("cache_dir", dir("ov-gpu-f32")),
|
||||
],
|
||||
)?,
|
||||
Ep::OpenVinoGpuFp16 => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[
|
||||
("device_type", gpu.clone()),
|
||||
("precision", "FP16".into()),
|
||||
("cache_dir", dir("ov-gpu-fp16")),
|
||||
],
|
||||
)?,
|
||||
Ep::OpenVinoNpu => append(
|
||||
&mut b,
|
||||
c"OpenVINO",
|
||||
&[("device_type", "NPU".into()), ("cache_dir", dir("ov-npu"))],
|
||||
)?,
|
||||
// `webgpu_provider_options.h` at 1.27; the runtime prefixes the key.
|
||||
Ep::WebGpuLow => append(
|
||||
&mut b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "low-power".into())],
|
||||
)?,
|
||||
Ep::WebGpuHigh => append(
|
||||
&mut b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "high-performance".into())],
|
||||
)?,
|
||||
}
|
||||
b.commit_from_memory(bytes)
|
||||
}
|
||||
|
||||
/// Any provider through the generic key/value entry point.
|
||||
fn append(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
name: &std::ffi::CStr,
|
||||
options: &[(&str, String)],
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let keys: Vec<CString> = options
|
||||
.iter()
|
||||
.map(|(k, _)| CString::new(*k).unwrap())
|
||||
.collect();
|
||||
let values: Vec<CString> = options
|
||||
.iter()
|
||||
.map(|(_, v)| CString::new(v.as_bytes()).unwrap())
|
||||
.collect();
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
// SAFETY: as `migraphx` below.
|
||||
unsafe {
|
||||
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
|
||||
b.ptr_mut(),
|
||||
name.as_ptr(),
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
);
|
||||
ort::Error::result_from_status(status)
|
||||
}
|
||||
}
|
||||
|
||||
/// Register MIGraphX through the generic key/value API. `ort`'s own
|
||||
/// builder fills the legacy `OrtMIGraphXProviderOptions`, which 1.29 reads
|
||||
/// for its precision flags and nothing else: the model cache directory —
|
||||
@@ -83,19 +184,29 @@ fn migraphx(
|
||||
|
||||
/// Median of `runs` timed runs over zeros, in milliseconds, after warm-ups.
|
||||
fn time(session: &mut ort::session::Session, warmups: usize, runs: usize) -> Result<f64, String> {
|
||||
let shape: Vec<usize> = session.inputs()[0]
|
||||
.dtype()
|
||||
.tensor_shape()
|
||||
.ok_or("input is not a tensor")?
|
||||
.iter()
|
||||
.map(|&d| if d > 0 { d as usize } else { 1 })
|
||||
.collect();
|
||||
let zeros = vec![0f32; shape.iter().product()];
|
||||
// Zeros for every input, not just the first: the denoiser takes
|
||||
// `mosaic` and `sigma`. A dynamic dimension is read as 1.
|
||||
let mut inputs = Vec::new();
|
||||
for input in session.inputs() {
|
||||
let shape: Vec<usize> = input
|
||||
.dtype()
|
||||
.tensor_shape()
|
||||
.ok_or("input is not a tensor")?
|
||||
.iter()
|
||||
.map(|&d| if d > 0 { d as usize } else { 1 })
|
||||
.collect();
|
||||
let zeros = vec![0f32; shape.iter().product()];
|
||||
inputs.push((input.name().to_string(), shape, zeros));
|
||||
}
|
||||
let once = |s: &mut ort::session::Session| -> Result<f64, String> {
|
||||
let input = ort::value::Tensor::from_array((shape.clone(), zeros.clone()))
|
||||
.map_err(|e| e.to_string())?;
|
||||
let mut values = Vec::with_capacity(inputs.len());
|
||||
for (name, shape, zeros) in &inputs {
|
||||
let value = ort::value::Tensor::from_array((shape.clone(), zeros.clone()))
|
||||
.map_err(|e| e.to_string())?;
|
||||
values.push((name.clone(), ort::session::SessionInputValue::from(value)));
|
||||
}
|
||||
let t = Instant::now();
|
||||
let out = s.run(ort::inputs![input]).map_err(|e| e.to_string())?;
|
||||
let out = s.run(values).map_err(|e| e.to_string())?;
|
||||
let _ = out[0]
|
||||
.try_extract_tensor::<f32>()
|
||||
.map_err(|e| e.to_string())?;
|
||||
@@ -155,13 +266,35 @@ fn main() {
|
||||
// A compiling provider is built twice: the second build reads the
|
||||
// program the first wrote, and its time is what a launch after the
|
||||
// first costs.
|
||||
let plan = [
|
||||
(Ep::Cpu, false),
|
||||
(Ep::MiGraphX, false),
|
||||
(Ep::MiGraphX, true),
|
||||
(Ep::MiGraphXFp16, false),
|
||||
(Ep::MiGraphXFp16, true),
|
||||
];
|
||||
// DARKROOM_EPS narrows the list (`cpu,openvino,webgpu,migraphx`);
|
||||
// a runtime without a provider fails its build in a millisecond
|
||||
// anyway, so the default is all of them.
|
||||
let wanted = std::env::var("DARKROOM_EPS").unwrap_or_default();
|
||||
let on = |family: &str| wanted.is_empty() || wanted.split(',').any(|w| w == family);
|
||||
let mut plan = Vec::new();
|
||||
for (family, eps) in [
|
||||
("cpu", &[Ep::Cpu][..]),
|
||||
("migraphx", &[Ep::MiGraphX, Ep::MiGraphXFp16][..]),
|
||||
(
|
||||
"openvino",
|
||||
&[
|
||||
Ep::OpenVinoCpu,
|
||||
Ep::OpenVinoGpu,
|
||||
Ep::OpenVinoGpuFp16,
|
||||
Ep::OpenVinoNpu,
|
||||
][..],
|
||||
),
|
||||
("webgpu", &[Ep::WebGpuLow, Ep::WebGpuHigh][..]),
|
||||
] {
|
||||
if on(family) {
|
||||
for &ep in eps {
|
||||
plan.push((ep, false));
|
||||
if ep.caches() {
|
||||
plan.push((ep, true));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (ep, cached) in plan {
|
||||
let started = Instant::now();
|
||||
match build(ep, &bytes, threads, &cache) {
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
//! cargo run --release -p dr-inference-engine --features native,tract \
|
||||
//! --example ladder -- CACHE_DIR models/face/scrfd_500m_640.onnx [ROLE=MODEL.onnx ...]
|
||||
//!
|
||||
//! `DARKROOM_ORT_DIRS=a:b:c` offers several runtimes, as the app's search
|
||||
//! list does, and shows which the engine chose for this device's GPU.
|
||||
//!
|
||||
//! A bare path is a `Detector`; `denoiser=…`, `scene=…`, `inpainter=…`,
|
||||
//! `landmarks=…` (any `Role`, lower case) says otherwise, so a device can
|
||||
//! show each role taking its own form (inference.md §1.5). Each is opened
|
||||
@@ -31,9 +34,16 @@ fn main() {
|
||||
std::process::exit(2);
|
||||
}
|
||||
|
||||
// DARKROOM_ORT_DIRS lists several, colon-separated, as the app's search
|
||||
// does: the engine loads the one that fits the GPU (§3.2).
|
||||
let runtime_dirs: Vec<PathBuf> = std::env::var_os("DARKROOM_ORT_DIR")
|
||||
.map(PathBuf::from)
|
||||
.into_iter()
|
||||
.chain(
|
||||
std::env::var_os("DARKROOM_ORT_DIRS")
|
||||
.map(|v| std::env::split_paths(&v).collect::<Vec<_>>())
|
||||
.unwrap_or_default(),
|
||||
)
|
||||
.collect();
|
||||
let started = Instant::now();
|
||||
dr_inference_engine::init(dr_inference_engine::Config {
|
||||
|
||||
@@ -51,17 +51,25 @@ pub fn ensure_installed() {
|
||||
}
|
||||
}
|
||||
|
||||
/// Look for `libonnxruntime` in `dirs`, in order, and hand `ort` the first
|
||||
/// table that loads; otherwise tract. Once per process.
|
||||
/// Find every `libonnxruntime` in `dirs`, hand `ort` the table of the one
|
||||
/// that best fits this device's GPUs, and fall to tract if none loads.
|
||||
/// Once per process.
|
||||
///
|
||||
/// Best fit, not first found (§3.2): a device can hold several runtimes —
|
||||
/// the package's OpenVINO build, a CUDA build the user fetched, the
|
||||
/// distribution's ROCm build — and each carries one vendor's providers.
|
||||
/// Between equals, the earlier directory wins, as it always has, and a
|
||||
/// runtime that fits perfectly ends the search: the APK's QNN build on a
|
||||
/// Qualcomm tablet is found first, and the generic build beside it is
|
||||
/// never opened there.
|
||||
/// `DARKROOM_ORT_DIR`, when it loads, wins outright: it is how a person
|
||||
/// says which runtime they mean.
|
||||
pub fn install(dirs: &[PathBuf]) -> Runtime {
|
||||
RUNTIME
|
||||
.get_or_init(|| {
|
||||
#[cfg(feature = "native")]
|
||||
for dir in dirs {
|
||||
match load_native(dir) {
|
||||
Ok(rt) => return rt,
|
||||
Err(e) => log::info!("inference: no runtime in {}: {e}", dir.display()),
|
||||
}
|
||||
if let Some(rt) = install_best(dirs) {
|
||||
return rt;
|
||||
}
|
||||
#[cfg(not(feature = "native"))]
|
||||
let _ = dirs;
|
||||
@@ -70,6 +78,103 @@ pub fn install(dirs: &[PathBuf]) -> Runtime {
|
||||
.clone()
|
||||
}
|
||||
|
||||
/// A runtime opened to read its providers, not yet handed to `ort`.
|
||||
#[cfg(feature = "native")]
|
||||
struct Found {
|
||||
lib: libloading::Library,
|
||||
api: *const ort_sys::OrtApi,
|
||||
path: PathBuf,
|
||||
version: String,
|
||||
providers: Vec<String>,
|
||||
}
|
||||
|
||||
#[cfg(feature = "native")]
|
||||
fn install_best(dirs: &[PathBuf]) -> Option<Runtime> {
|
||||
let named = std::env::var_os("DARKROOM_ORT_DIR").map(PathBuf::from);
|
||||
let gpus = crate::hardware::detect();
|
||||
let mut found: Vec<Found> = Vec::new();
|
||||
let mut seen = std::collections::HashSet::new();
|
||||
for dir in dirs {
|
||||
match open_native(dir) {
|
||||
Ok(f) => {
|
||||
// `bin/../lib/darkroom` and `/usr/lib/darkroom` are one file.
|
||||
if !seen.insert(std::fs::canonicalize(&f.path).unwrap_or(f.path.clone())) {
|
||||
std::mem::forget(f.lib);
|
||||
continue;
|
||||
}
|
||||
log::info!(
|
||||
"inference: ONNX Runtime {} at {} offers {}",
|
||||
f.version,
|
||||
f.path.display(),
|
||||
f.providers.join(", ")
|
||||
);
|
||||
if named.as_deref() == Some(dir.as_path()) {
|
||||
found.clear();
|
||||
found.push(f);
|
||||
break;
|
||||
}
|
||||
let perfect = gpus.score(&f.providers) >= crate::hardware::PERFECT;
|
||||
found.push(f);
|
||||
if perfect {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(e) => log::info!("inference: no runtime in {}: {e}", dir.display()),
|
||||
}
|
||||
}
|
||||
let best = (0..found.len())
|
||||
.max_by_key(|&i| (gpus.score(&found[i].providers), std::cmp::Reverse(i)))?;
|
||||
let chosen = found.swap_remove(best);
|
||||
// The others stay mapped. Unloading a C++ runtime after its static
|
||||
// constructors ran is a crash at exit waiting to happen, and an
|
||||
// unused mapping costs address space, not memory.
|
||||
for other in found {
|
||||
std::mem::forget(other.lib);
|
||||
}
|
||||
log::info!("inference: chose {} for {gpus:?}", chosen.path.display());
|
||||
|
||||
// SAFETY: the table came from this library's `OrtGetApiBase`, and the
|
||||
// library is leaked below, so every pointer in the copy stays valid for
|
||||
// the life of the process.
|
||||
if !ort::set_api(unsafe { (*chosen.api).clone() }) {
|
||||
log::warn!("inference: an API table was already installed");
|
||||
std::mem::forget(chosen.lib);
|
||||
return None;
|
||||
}
|
||||
std::mem::forget(chosen.lib);
|
||||
|
||||
// Qualcomm's DSP loader finds the Hexagon skel through this variable,
|
||||
// and only through it; the runtime's own directory is where the APK
|
||||
// put it. Harmless anywhere else.
|
||||
#[cfg(target_os = "android")]
|
||||
if let Some(dir) = chosen.path.parent().filter(|d| !d.as_os_str().is_empty()) {
|
||||
std::env::set_var("ADSP_LIBRARY_PATH", dir);
|
||||
}
|
||||
|
||||
// Windows looks for a provider's own dependencies — OpenVINO's DLLs,
|
||||
// which Intel's build leaves beside it — on the DLL search path, not in
|
||||
// the provider's directory. Intel's Python shim prepends to `PATH` for
|
||||
// the same reason; so does this, before any provider loads.
|
||||
#[cfg(target_os = "windows")]
|
||||
if let Some(dir) = chosen.path.parent() {
|
||||
let old = std::env::var_os("PATH").unwrap_or_default();
|
||||
let dirs = std::iter::once(dir.to_path_buf()).chain(std::env::split_paths(&old));
|
||||
if let Ok(path) = std::env::join_paths(dirs) {
|
||||
std::env::set_var("PATH", path);
|
||||
}
|
||||
}
|
||||
|
||||
log::info!(
|
||||
"inference: ONNX Runtime {} from {}",
|
||||
chosen.version,
|
||||
chosen.path.display()
|
||||
);
|
||||
Some(Runtime::OnnxRuntime {
|
||||
path: chosen.path,
|
||||
version: chosen.version,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(feature = "tract")]
|
||||
fn install_tract() -> Runtime {
|
||||
let _ = ort::set_api(ort_tract::api());
|
||||
@@ -85,8 +190,12 @@ fn install_tract() -> Runtime {
|
||||
Runtime::Tract
|
||||
}
|
||||
|
||||
/// Open the runtime in `dir` and read what it offers. `dir` may also name
|
||||
/// the library itself — Android has two runtimes and one directory, so the
|
||||
/// second goes by its file name — and an empty path is the bare name
|
||||
/// through the system loader, which on Android is the APK's own copy.
|
||||
#[cfg(feature = "native")]
|
||||
fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
fn open_native(dir: &std::path::Path) -> Result<Found, String> {
|
||||
let name = if cfg!(target_os = "windows") {
|
||||
"onnxruntime.dll"
|
||||
} else if cfg!(any(target_os = "macos", target_os = "ios")) {
|
||||
@@ -94,18 +203,21 @@ fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
} else {
|
||||
"libonnxruntime.so"
|
||||
};
|
||||
// An empty dir means the bare name: the system loader's search, which on
|
||||
// Android includes the APK's own native libraries.
|
||||
let is_library = dir.file_name().and_then(|n| n.to_str()).is_some_and(|n| {
|
||||
n.contains("onnxruntime")
|
||||
&& (n.ends_with(".so") || n.ends_with(".dll") || n.ends_with(".dylib"))
|
||||
});
|
||||
let path = if dir.as_os_str().is_empty() {
|
||||
PathBuf::from(name)
|
||||
} else if is_library {
|
||||
dir.to_path_buf()
|
||||
} else {
|
||||
find_library(dir, name).ok_or("not present")?
|
||||
};
|
||||
|
||||
// SAFETY: the library's initialisers are ONNX Runtime's own; the symbol
|
||||
// is the documented entry point with the documented signature; the table
|
||||
// is copied out and the library handle is leaked, so every pointer in
|
||||
// the copy stays valid for the life of the process.
|
||||
// is the documented entry point with the documented signature. The
|
||||
// table pointer is valid while `lib` is, which the caller keeps.
|
||||
unsafe {
|
||||
let lib = libloading::Library::new(&path).map_err(|e| e.to_string())?;
|
||||
let get_base: libloading::Symbol<
|
||||
@@ -125,24 +237,45 @@ fn load_native(dir: &std::path::Path) -> Result<Runtime, String> {
|
||||
ort_sys::ORT_API_VERSION
|
||||
));
|
||||
}
|
||||
if !ort::set_api((*api).clone()) {
|
||||
return Err("an API table was already installed".into());
|
||||
}
|
||||
std::mem::forget(lib);
|
||||
|
||||
// Qualcomm's DSP loader finds the Hexagon skel through this variable,
|
||||
// and only through it; the runtime's own directory is where the APK
|
||||
// put it. Harmless anywhere else.
|
||||
#[cfg(target_os = "android")]
|
||||
if !dir.as_os_str().is_empty() {
|
||||
std::env::set_var("ADSP_LIBRARY_PATH", dir);
|
||||
}
|
||||
|
||||
log::info!("inference: ONNX Runtime {version} from {}", path.display());
|
||||
Ok(Runtime::OnnxRuntime { path, version })
|
||||
let providers = available_providers(api);
|
||||
Ok(Found {
|
||||
lib,
|
||||
api,
|
||||
path,
|
||||
version,
|
||||
providers,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// The providers compiled into the runtime behind `api` — not the ones this
|
||||
/// device can run, which is the probe's question.
|
||||
///
|
||||
/// # Safety
|
||||
/// `api` must be a live table from `GetApi`.
|
||||
#[cfg(feature = "native")]
|
||||
unsafe fn available_providers(api: *const ort_sys::OrtApi) -> Vec<String> {
|
||||
let mut list: *mut *mut std::ffi::c_char = std::ptr::null_mut();
|
||||
let mut n: std::ffi::c_int = 0;
|
||||
let status = ((*api).GetAvailableProviders)(&mut list, &mut n);
|
||||
if !status.0.is_null() {
|
||||
((*api).ReleaseStatus)(status.0);
|
||||
return Vec::new();
|
||||
}
|
||||
let names = (0..n.max(0) as usize)
|
||||
.map(|i| {
|
||||
std::ffi::CStr::from_ptr(*list.add(i))
|
||||
.to_string_lossy()
|
||||
.into_owned()
|
||||
})
|
||||
.collect();
|
||||
let status = ((*api).ReleaseAvailableProviders)(list, n);
|
||||
if !status.0.is_null() {
|
||||
((*api).ReleaseStatus)(status.0);
|
||||
}
|
||||
names
|
||||
}
|
||||
|
||||
/// `libonnxruntime.so` in `dir`, or a versioned spelling of it —
|
||||
/// `libonnxruntime.so.1.30.0` is what the Python wheel ships, and a package
|
||||
/// that installs only the versioned file is not wrong.
|
||||
|
||||
@@ -48,12 +48,35 @@ pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
/// CoreML's own cache key leaves out the weights of a model loaded from
|
||||
/// memory (`session::coreml`), and one per runtime version, which wrote it.
|
||||
pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||
model_dir(cfg, "coreml", bytes)
|
||||
}
|
||||
|
||||
/// Where OpenVINO compiles `bytes` to, at one precision. OpenVINO hashes
|
||||
/// the model it is given, weights included, but a key that leaves out
|
||||
/// what is being varied has cost a day before (CLAUDE.md, "Providers"),
|
||||
/// and a directory per model and precision costs nothing: the precision
|
||||
/// is a compile option, and the two forms are different programs.
|
||||
pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
|
||||
model_dir(
|
||||
cfg,
|
||||
if fp16 {
|
||||
"openvino/fp16"
|
||||
} else {
|
||||
"openvino/f32"
|
||||
},
|
||||
bytes,
|
||||
)
|
||||
}
|
||||
|
||||
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
|
||||
/// model, and one per runtime version, which wrote it.
|
||||
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
|
||||
let runtime = match crate::api::runtime() {
|
||||
crate::Runtime::OnnxRuntime { version, .. } => version,
|
||||
crate::Runtime::Tract => "tract".into(),
|
||||
};
|
||||
cfg.cache_dir
|
||||
.join("coreml")
|
||||
.join(provider)
|
||||
.join(runtime)
|
||||
.join(format!("{:016x}", hash(bytes)))
|
||||
}
|
||||
|
||||
@@ -0,0 +1,176 @@
|
||||
//! Which GPUs this device has, as far as choosing a runtime needs to know
|
||||
//! (docs/dev/inference.md §3.2).
|
||||
//!
|
||||
//! A runtime carries one vendor's providers — Intel's build has OpenVINO,
|
||||
//! the `onnxruntime-gpu` wheel CUDA and TensorRT, a ROCm build MIGraphX,
|
||||
//! Microsoft's WebGPU build the generic rung — and only one runtime loads
|
||||
//! per process. These checks are what lets `api` load the one that fits
|
||||
//! when a device has several installed. They read files, never a driver:
|
||||
//! a wrong answer costs a slower rung, which the probe still measures, and
|
||||
//! a driver call at start-up could cost the launch.
|
||||
|
||||
/// What a runtime's providers are scored against.
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||
pub struct Gpus {
|
||||
pub nvidia: bool,
|
||||
/// An AMD GPU with the ROCm kernel interface, which MIGraphX needs.
|
||||
pub amd_rocm: bool,
|
||||
pub intel: bool,
|
||||
pub qualcomm: bool,
|
||||
}
|
||||
|
||||
/// The score of a runtime whose vendor rung matches the device's GPU.
|
||||
/// Nothing beats it, so the search stops there.
|
||||
pub const PERFECT: u32 = 3;
|
||||
|
||||
impl Gpus {
|
||||
/// How well a runtime offering `providers` fits this device. The vendor
|
||||
/// rungs score above OpenVINO because a machine with an Intel iGPU and
|
||||
/// an NVIDIA or AMD card wants the card; the generic rung scores above
|
||||
/// a CPU-only build because it carries the same CPU provider and might
|
||||
/// beat it.
|
||||
pub fn score(&self, providers: &[String]) -> u32 {
|
||||
providers
|
||||
.iter()
|
||||
.map(|p| match p.as_str() {
|
||||
"TensorrtExecutionProvider" | "CUDAExecutionProvider" if self.nvidia => PERFECT,
|
||||
"MIGraphXExecutionProvider" if self.amd_rocm => PERFECT,
|
||||
"QNNExecutionProvider" if self.qualcomm => PERFECT,
|
||||
"CoreMLExecutionProvider" => PERFECT,
|
||||
"OpenVINOExecutionProvider" if self.intel => 2,
|
||||
"WebGpuExecutionProvider" => 1,
|
||||
_ => 0,
|
||||
})
|
||||
.max()
|
||||
.unwrap_or(0)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
pub fn detect() -> Gpus {
|
||||
use std::path::Path;
|
||||
// Every DRM card's PCI vendor: an Intel iGPU is `0x8086` whether or
|
||||
// not its compute driver is installed, which the probe finds out.
|
||||
let vendors: Vec<String> = std::fs::read_dir("/sys/class/drm")
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.filter(|e| {
|
||||
let name = e.file_name();
|
||||
let name = name.to_string_lossy();
|
||||
name.starts_with("card") && !name.contains('-')
|
||||
})
|
||||
.filter_map(|e| std::fs::read_to_string(e.path().join("device/vendor")).ok())
|
||||
.map(|v| v.trim().to_string())
|
||||
.collect();
|
||||
Gpus {
|
||||
nvidia: Path::new("/proc/driver/nvidia/version").exists(),
|
||||
amd_rocm: Path::new("/dev/kfd").exists(),
|
||||
intel: vendors.iter().any(|v| v == "0x8086"),
|
||||
qualcomm: false,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "windows")]
|
||||
pub fn detect() -> Gpus {
|
||||
use std::path::PathBuf;
|
||||
let root = std::env::var_os("SystemRoot")
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| PathBuf::from(r"C:\Windows"));
|
||||
let system32 = root.join("System32");
|
||||
// Intel's DCH graphics driver, integrated and Arc alike, installs
|
||||
// from `iigd_dch.inf`; its package directory is the evidence.
|
||||
let intel = std::fs::read_dir(system32.join(r"DriverStore\FileRepository"))
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.any(|e| e.file_name().to_string_lossy().starts_with("iigd_dch"));
|
||||
Gpus {
|
||||
nvidia: system32.join("nvcuda.dll").exists(),
|
||||
amd_rocm: false,
|
||||
intel,
|
||||
qualcomm: false,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
pub fn detect() -> Gpus {
|
||||
// Fail-safe: only a device that names another vendor is not Qualcomm.
|
||||
// `ro.soc.manufacturer` exists from Android 12, and a property or file
|
||||
// the app cannot read reads as nothing; nothing keeps the QNN build
|
||||
// first, as 0.22 had it, where a Qualcomm device mistaken for another
|
||||
// would trade its NPU for the generic rung. Qualcomm's FastRPC library,
|
||||
// which the Hexagon path loads anyway, overrules a name.
|
||||
let soc = crate::probe::system_property("ro.soc.manufacturer");
|
||||
let fastrpc = [
|
||||
"/vendor/lib64/libcdsprpc.so",
|
||||
"/system/vendor/lib64/libcdsprpc.so",
|
||||
]
|
||||
.iter()
|
||||
.any(|p| std::path::Path::new(p).exists());
|
||||
Gpus {
|
||||
qualcomm: qualcomm_soc(&soc) || fastrpc,
|
||||
..Gpus::default()
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `ro.soc.manufacturer` leaves the device Qualcomm's: it says so,
|
||||
/// or it says nothing.
|
||||
#[cfg(any(target_os = "android", test))]
|
||||
fn qualcomm_soc(manufacturer: &str) -> bool {
|
||||
let m = manufacturer.trim();
|
||||
m.is_empty() || m.eq_ignore_ascii_case("QTI") || m.eq_ignore_ascii_case("Qualcomm")
|
||||
}
|
||||
|
||||
#[cfg(not(any(target_os = "linux", target_os = "windows", target_os = "android")))]
|
||||
pub fn detect() -> Gpus {
|
||||
Gpus::default()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn offers(p: &[&str]) -> Vec<String> {
|
||||
p.iter().map(|s| s.to_string()).collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_a_named_other_vendor_is_not_qualcomm() {
|
||||
assert!(qualcomm_soc("QTI"));
|
||||
assert!(qualcomm_soc("Qualcomm"));
|
||||
// Unreadable, or older than Android 12: the QNN build stays first.
|
||||
assert!(qualcomm_soc(""));
|
||||
assert!(!qualcomm_soc("Mediatek"));
|
||||
assert!(!qualcomm_soc("Google"));
|
||||
assert!(!qualcomm_soc("Samsung"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_card_beats_the_integrated_gpu_and_both_beat_the_generic_rung() {
|
||||
let cpu = offers(&["CPUExecutionProvider"]);
|
||||
let nvidia = offers(&[
|
||||
"TensorrtExecutionProvider",
|
||||
"CUDAExecutionProvider",
|
||||
"CPUExecutionProvider",
|
||||
]);
|
||||
let intel = offers(&["OpenVINOExecutionProvider", "CPUExecutionProvider"]);
|
||||
let webgpu = offers(&["WebGpuExecutionProvider", "CPUExecutionProvider"]);
|
||||
let laptop = Gpus {
|
||||
nvidia: true,
|
||||
intel: true,
|
||||
..Gpus::default()
|
||||
};
|
||||
assert!(laptop.score(&nvidia) > laptop.score(&intel));
|
||||
assert!(laptop.score(&intel) > laptop.score(&webgpu));
|
||||
assert!(laptop.score(&webgpu) > laptop.score(&cpu));
|
||||
// No Intel GPU: Intel's build is worth no more than a CPU build to
|
||||
// this device, and the generic rung is worth more.
|
||||
let amd_on_windows = Gpus::default();
|
||||
assert_eq!(amd_on_windows.score(&intel), amd_on_windows.score(&cpu));
|
||||
assert!(amd_on_windows.score(&webgpu) > amd_on_windows.score(&intel));
|
||||
// A ROCm build on a machine without ROCm is a CPU build.
|
||||
let rocm = offers(&["MIGraphXExecutionProvider", "CPUExecutionProvider"]);
|
||||
assert_eq!(amd_on_windows.score(&rocm), 0);
|
||||
}
|
||||
}
|
||||
@@ -21,6 +21,7 @@ use serde::{Deserialize, Serialize};
|
||||
|
||||
mod api;
|
||||
mod engines;
|
||||
mod hardware;
|
||||
mod probe;
|
||||
mod session;
|
||||
|
||||
@@ -110,6 +111,16 @@ pub enum Rung {
|
||||
/// embedder stays on the CPU, as on the Hexagon: the Neural Engine
|
||||
/// computes in fp16 (§7).
|
||||
CoreMl,
|
||||
/// Intel, through OpenVINO on the integrated or Arc GPU. Desktop only.
|
||||
/// Compiles a program per model, as MIGraphX does, so the CPU is its
|
||||
/// fallback; fp16 on the same terms as TensorRT (§7).
|
||||
OpenVino,
|
||||
/// Any other GPU, through ONNX Runtime's WebGPU provider: Dawn on
|
||||
/// Vulkan, D3D12 or Metal. The generic rung, for a GPU no vendor rung
|
||||
/// covers. Measured slower than the CPU on every GPU it has been timed
|
||||
/// on (§1), so it is on the ladder for the GPUs it has not, and the
|
||||
/// probe's clock is what keeps it off the rest.
|
||||
WebGpu,
|
||||
}
|
||||
|
||||
impl Rung {
|
||||
@@ -121,6 +132,8 @@ impl Rung {
|
||||
Rung::MiGraphX => "MIGraphX",
|
||||
Rung::Hexagon => "Hexagon NPU",
|
||||
Rung::CoreMl => "CoreML",
|
||||
Rung::OpenVino => "OpenVINO",
|
||||
Rung::WebGpu => "WebGPU",
|
||||
}
|
||||
}
|
||||
|
||||
@@ -129,7 +142,13 @@ impl Rung {
|
||||
fn fallback(self) -> Rung {
|
||||
match self {
|
||||
Rung::TensorRt => Rung::Cuda,
|
||||
Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::Cuda | Rung::Cpu => Rung::Cpu,
|
||||
Rung::MiGraphX
|
||||
| Rung::Hexagon
|
||||
| Rung::CoreMl
|
||||
| Rung::OpenVino
|
||||
| Rung::WebGpu
|
||||
| Rung::Cuda
|
||||
| Rung::Cpu => Rung::Cpu,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -137,7 +156,7 @@ impl Rung {
|
||||
fn compiles(self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::OpenVino
|
||||
)
|
||||
}
|
||||
|
||||
@@ -230,7 +249,7 @@ impl Status {
|
||||
pub fn line(&self) -> String {
|
||||
let form = match self.rung {
|
||||
Rung::Hexagon => " · quantised",
|
||||
Rung::TensorRt | Rung::MiGraphX => " · fp16",
|
||||
Rung::TensorRt | Rung::MiGraphX | Rung::OpenVino => " · fp16",
|
||||
_ => "",
|
||||
};
|
||||
format!("{}{} · {}", self.rung.label(), form, self.runtime.label())
|
||||
@@ -725,6 +744,8 @@ mod tests {
|
||||
Rung::TensorRt,
|
||||
Rung::MiGraphX,
|
||||
Rung::CoreMl,
|
||||
Rung::OpenVino,
|
||||
Rung::WebGpu,
|
||||
] {
|
||||
assert_eq!(rung.form(Role::Detector), F32);
|
||||
}
|
||||
@@ -765,6 +786,29 @@ mod tests {
|
||||
assert_eq!(on(&s, Role::Embedder), Rung::Cpu);
|
||||
}
|
||||
|
||||
/// OpenVINO compiles a program per model, so a request waits on the CPU
|
||||
/// until the engine thread has built it; WebGPU builds in the session
|
||||
/// and serves at once. Both take every role in f32 graphs.
|
||||
#[test]
|
||||
fn openvino_waits_for_its_program_and_webgpu_does_not() {
|
||||
let hash = engines::hash(b"detector");
|
||||
let mut s = State {
|
||||
config: Config::default(),
|
||||
cache: Cache::default(),
|
||||
probing: false,
|
||||
wanted: 0,
|
||||
};
|
||||
let on = |s: &State, rung| effective_rung(s, rung, Role::Detector, Form::F32, hash);
|
||||
assert_eq!(on(&s, Rung::OpenVino), Rung::Cpu);
|
||||
s.cache
|
||||
.compiled
|
||||
.insert(engines::key_of(Rung::OpenVino, hash));
|
||||
assert_eq!(on(&s, Rung::OpenVino), Rung::OpenVino);
|
||||
assert_eq!(on(&s, Rung::WebGpu), Rung::WebGpu);
|
||||
let embedder = effective_rung(&s, Rung::WebGpu, Role::Embedder, Form::F32, hash);
|
||||
assert_eq!(embedder, Rung::WebGpu);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_status_reports_only_the_rungs_above_the_selection() {
|
||||
let _serial = serial();
|
||||
|
||||
@@ -14,18 +14,27 @@ use crate::{api::Runtime, state, Cache, Config, Form, Role, Rung};
|
||||
|
||||
/// The rungs to try on this platform, best first, under the user's ceiling.
|
||||
fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
|
||||
// WebGPU is the generic rung (§2): it is reached only on a runtime that
|
||||
// carries it, which `api` loads where no vendor's runtime fits the
|
||||
// device, and kept only where it beats the CPU.
|
||||
#[cfg(target_os = "android")]
|
||||
let all = [Rung::Hexagon];
|
||||
let all = [Rung::Hexagon, Rung::WebGpu];
|
||||
// Unmeasured (§2 ⁵): it is on the ladder because the probe's clock and
|
||||
// `attempt` make a wrong guess cost one slow or failed probe, not a
|
||||
// slow or crashing app.
|
||||
#[cfg(target_os = "macos")]
|
||||
let all = [Rung::CoreMl];
|
||||
// A desktop has one vendor's GPU; the other vendor's providers are
|
||||
// "not enabled in this build" or a library that fails to load, and
|
||||
// either answer arrives in milliseconds.
|
||||
// A runtime carries one vendor's providers, chosen for this device's
|
||||
// GPU (`api`); the others are "not enabled in this build", and that
|
||||
// answer arrives in milliseconds.
|
||||
#[cfg(not(any(target_os = "android", target_os = "macos")))]
|
||||
let all = [Rung::TensorRt, Rung::Cuda, Rung::MiGraphX];
|
||||
let all = [
|
||||
Rung::TensorRt,
|
||||
Rung::Cuda,
|
||||
Rung::MiGraphX,
|
||||
Rung::OpenVino,
|
||||
Rung::WebGpu,
|
||||
];
|
||||
all.into_iter()
|
||||
.filter(|r| ceiling.is_none_or(|c| *r <= c))
|
||||
.collect()
|
||||
@@ -157,7 +166,16 @@ pub fn run(runtime: Runtime) {
|
||||
}
|
||||
if cache.rung.is_none() {
|
||||
cache.rung = Some(Rung::Cpu);
|
||||
cache.reason = match cache.failed.first() {
|
||||
// The rung that tried and lost, not the first one the runtime was
|
||||
// never built with: "WebGPU 150 ms, slower than the CPU" says why
|
||||
// this device is on the CPU, "TensorRT not enabled" does not.
|
||||
let tried = cache
|
||||
.failed
|
||||
.iter()
|
||||
.rev()
|
||||
.find(|(_, why)| !why.contains("in this build"))
|
||||
.or(cache.failed.first());
|
||||
cache.reason = match tried {
|
||||
Some((r, why)) => format!("{} {}", r.label(), first_line(why)),
|
||||
None => "the only rung on this platform".into(),
|
||||
};
|
||||
@@ -237,9 +255,14 @@ fn probe_model(cfg: &Config) -> Option<(Role, PathBuf)> {
|
||||
smallest(Some(Role::Detector)).or_else(|| smallest(None))
|
||||
}
|
||||
|
||||
/// Build, run once for the engine, then time three runs; the median in
|
||||
/// milliseconds and, for a compiling rung, the cache key of the engine this
|
||||
/// just built.
|
||||
/// Build, warm up, then time seven runs; the median in milliseconds and,
|
||||
/// for a compiling rung, the cache key of the engine this just built.
|
||||
///
|
||||
/// Three warm-ups, not one: an idle integrated GPU takes a few runs to
|
||||
/// raise its clock. With one, the Iris Xe's OpenVINO lost to the CPU on
|
||||
/// the smallest detector in two probes of three, where warm it is 5.8 ms
|
||||
/// against 9.5 (§1.6). The smallest detector is a GPU's worst case; the
|
||||
/// clock must not also be.
|
||||
fn time_rung(
|
||||
rung: Rung,
|
||||
role: Role,
|
||||
@@ -297,11 +320,15 @@ fn time_rung(
|
||||
.map_err(|e| e.to_string())?;
|
||||
Ok(t.elapsed().as_secs_f64() * 1e3)
|
||||
};
|
||||
run(&mut session)?;
|
||||
let mut times = [run(&mut session)?, run(&mut session)?, run(&mut session)?];
|
||||
for _ in 0..3 {
|
||||
run(&mut session)?;
|
||||
}
|
||||
let mut times = (0..7)
|
||||
.map(|_| run(&mut session))
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
times.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
||||
let key = rung.compiles().then(|| crate::engines::key(rung, &bytes));
|
||||
Ok((times[1], key))
|
||||
Ok((times[times.len() / 2], key))
|
||||
}
|
||||
|
||||
/// The part of a provider's error a person can act on. ONNX Runtime's
|
||||
@@ -382,19 +409,35 @@ fn providers_beside(runtime: &Path) -> String {
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn device_identity() -> String {
|
||||
// The NVIDIA driver's version line, or the ROCm release the AMD stack
|
||||
// came from (`rocm-core` writes it; the kernel driver has no version
|
||||
// of its own). Absent means neither.
|
||||
// The NVIDIA driver's version line, the ROCm release the AMD stack came
|
||||
// from (`rocm-core` writes it; the kernel driver has no version of its
|
||||
// own), and the OpenCL drivers registered — OpenVINO reaches the GPU
|
||||
// through one, and installing Intel's is what makes the Iris Xe a rung.
|
||||
let mut parts = Vec::new();
|
||||
if let Some(line) = std::fs::read_to_string("/proc/driver/nvidia/version")
|
||||
.ok()
|
||||
.and_then(|s| s.lines().next().map(str::to_string))
|
||||
{
|
||||
return line;
|
||||
parts.push(line);
|
||||
}
|
||||
if let Ok(rocm) = std::fs::read_to_string("/opt/rocm/.info/version") {
|
||||
return format!("rocm {}", rocm.trim());
|
||||
parts.push(format!("rocm {}", rocm.trim()));
|
||||
}
|
||||
let mut icds: Vec<String> = std::fs::read_dir("/etc/OpenCL/vendors")
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.file_name().to_string_lossy().into_owned())
|
||||
.collect();
|
||||
icds.sort();
|
||||
if !icds.is_empty() {
|
||||
parts.push(format!("opencl {}", icds.join(" ")));
|
||||
}
|
||||
if parts.is_empty() {
|
||||
"no nvidia driver, no rocm, no opencl".into()
|
||||
} else {
|
||||
parts.join("; ")
|
||||
}
|
||||
"no nvidia driver, no rocm".into()
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
@@ -409,7 +452,7 @@ fn device_identity() -> String {
|
||||
}
|
||||
|
||||
#[cfg(target_os = "android")]
|
||||
fn system_property(name: &str) -> String {
|
||||
pub(crate) fn system_property(name: &str) -> String {
|
||||
extern "C" {
|
||||
fn __system_property_get(
|
||||
name: *const std::ffi::c_char,
|
||||
|
||||
@@ -55,9 +55,10 @@ fn build_with(
|
||||
let context = (rung == Rung::Hexagon).then(|| crate::engines::context_path(cfg, bytes));
|
||||
let ready = context.as_ref().is_some_and(|p| p.is_file());
|
||||
// What the rung keeps for this model: the context the Hexagon is to
|
||||
// write, or the directory CoreML compiles into.
|
||||
// write, or the directory CoreML or OpenVINO compiles into.
|
||||
let per_model = match rung {
|
||||
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
|
||||
Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))),
|
||||
_ if ready => None,
|
||||
_ => context.clone(),
|
||||
};
|
||||
@@ -101,6 +102,13 @@ fn with_runtime_log(
|
||||
.with_log_level(level)?)
|
||||
}
|
||||
|
||||
/// Whether `role` runs in fp16 on a rung that offers it: everything but the
|
||||
/// embedder, whose comparability across devices is worth more than its
|
||||
/// fraction of a millisecond (§7).
|
||||
fn fp16(role: Role) -> bool {
|
||||
role != Role::Embedder
|
||||
}
|
||||
|
||||
/// The intra-op pool: what the config says, else the cores less two for
|
||||
/// the compositor and the decoder (§9). tract ignores it.
|
||||
fn threads(cfg: &Config) -> usize {
|
||||
@@ -137,7 +145,7 @@ fn providers(
|
||||
// (NFR-RES-2). CUDA behind it takes any node TensorRT declines.
|
||||
Ok(b.with_execution_providers([
|
||||
ep::TensorRT::default()
|
||||
.with_fp16(role != Role::Embedder)
|
||||
.with_fp16(fp16(role))
|
||||
.with_engine_cache(true)
|
||||
.with_engine_cache_path(&cache)
|
||||
.with_timing_cache(true)
|
||||
@@ -154,7 +162,7 @@ fn providers(
|
||||
// directory, keyed on the graph, the GPU and its own version
|
||||
// but not the precision: hence one directory per precision.
|
||||
// The CPU takes any node it declines.
|
||||
let fp16 = role != Role::Embedder;
|
||||
let fp16 = fp16(role);
|
||||
let cache = cfg
|
||||
.cache_dir
|
||||
.join("migraphx")
|
||||
@@ -164,6 +172,16 @@ fn providers(
|
||||
migraphx(&mut b, fp16, &cache)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::OpenVino => {
|
||||
let mut b = b;
|
||||
openvino(&mut b, fp16(role), per_model)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::WebGpu => {
|
||||
let mut b = b;
|
||||
webgpu(&mut b)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::Hexagon => unreachable!("the Hexagon rung is not on a desktop ladder"),
|
||||
}
|
||||
}
|
||||
@@ -219,15 +237,79 @@ fn migraphx(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
fp16: bool,
|
||||
cache: &std::path::Path,
|
||||
) -> ort::Result<()> {
|
||||
append(
|
||||
b,
|
||||
c"MIGraphX",
|
||||
&[
|
||||
("migraphx_fp16_enable", if fp16 { "1" } else { "0" }.into()),
|
||||
(
|
||||
"migraphx_model_cache_dir",
|
||||
cache.to_string_lossy().into_owned(),
|
||||
),
|
||||
],
|
||||
)
|
||||
}
|
||||
|
||||
/// OpenVINO on the GPU, compiling into `cache`.
|
||||
///
|
||||
/// The option names are those `openvino_provider_factory.cc` reads at 1.24,
|
||||
/// the version of Intel's `onnxruntime-openvino` build. `GPU` is OpenVINO's
|
||||
/// first OpenCL GPU: the Intel one on a hybrid laptop with both drivers
|
||||
/// installed, but an NVIDIA card through its OpenCL when Intel's is absent
|
||||
/// — slower than the CPU there, and rejected by the probe's clock. The
|
||||
/// precision is always named: the GPU plugin's own default is fp16, and
|
||||
/// the embedder must not get it (§7).
|
||||
#[cfg(not(target_os = "android"))]
|
||||
fn openvino(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
fp16: bool,
|
||||
cache: Option<&std::path::Path>,
|
||||
) -> ort::Result<()> {
|
||||
let mut options = vec![
|
||||
("device_type", "GPU".to_string()),
|
||||
("precision", if fp16 { "FP16" } else { "FP32" }.into()),
|
||||
];
|
||||
if let Some(dir) = cache {
|
||||
let _ = std::fs::create_dir_all(dir);
|
||||
options.push(("cache_dir", dir.to_string_lossy().into_owned()));
|
||||
}
|
||||
append(b, c"OpenVINO", &options)
|
||||
}
|
||||
|
||||
/// WebGPU on the high-performance adapter: the discrete GPU where there is
|
||||
/// one, since the integrated one on a machine with both is the one this
|
||||
/// rung is least likely to beat the CPU on. The key is as
|
||||
/// `webgpu_provider_options.h` spells it, without the `ep.<name>.` prefix
|
||||
/// the runtime adds.
|
||||
fn webgpu(b: &mut ort::session::builder::SessionBuilder) -> ort::Result<()> {
|
||||
append(
|
||||
b,
|
||||
c"WebGPU",
|
||||
&[("powerPreference", "high-performance".to_string())],
|
||||
)
|
||||
}
|
||||
|
||||
/// Register the provider `name` with `options` through the runtime's
|
||||
/// generic key/value entry point, which takes every provider by its short
|
||||
/// name and reads options at the runtime's own version — not at the
|
||||
/// version `ort`'s builders were written against (CLAUDE.md, "Providers").
|
||||
fn append(
|
||||
b: &mut ort::session::builder::SessionBuilder,
|
||||
name: &std::ffi::CStr,
|
||||
options: &[(&str, String)],
|
||||
) -> ort::Result<()> {
|
||||
use ort::AsPointer;
|
||||
use std::ffi::CString;
|
||||
let keys = [c"migraphx_fp16_enable", c"migraphx_model_cache_dir"];
|
||||
let values = [
|
||||
CString::new(if fp16 { "1" } else { "0" }).unwrap(),
|
||||
CString::new(cache.to_string_lossy().as_bytes())
|
||||
.map_err(|e| ort::Error::new(e.to_string()))?,
|
||||
];
|
||||
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
|
||||
let keys = options
|
||||
.iter()
|
||||
.map(|(k, _)| cstr(k))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let values = options
|
||||
.iter()
|
||||
.map(|(_, v)| cstr(v))
|
||||
.collect::<ort::Result<Vec<_>>>()?;
|
||||
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
|
||||
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
|
||||
// SAFETY: the documented C call over arrays that outlive it; the
|
||||
@@ -235,7 +317,7 @@ fn migraphx(
|
||||
unsafe {
|
||||
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
|
||||
b.ptr_mut(),
|
||||
c"MIGraphX".as_ptr(),
|
||||
name.as_ptr(),
|
||||
key_ptrs.as_ptr(),
|
||||
value_ptrs.as_ptr(),
|
||||
keys.len(),
|
||||
@@ -277,7 +359,12 @@ fn providers(
|
||||
.build()
|
||||
.error_on_failure()])?)
|
||||
}
|
||||
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl => {
|
||||
Rung::WebGpu => {
|
||||
let mut b = b;
|
||||
webgpu(&mut b)?;
|
||||
Ok(b)
|
||||
}
|
||||
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl | Rung::OpenVino => {
|
||||
unreachable!("no desktop rung on Android")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -263,10 +263,12 @@ cp "${DEX}" "${OUT}/staging/classes.dex"
|
||||
# native library directory. The build links none of it — the app dlopens
|
||||
# `libonnxruntime.so` at launch and runs on tract if it is not there — so an
|
||||
# APK without these is a slower app, not a broken one, and `RUNTIME_DIR=none`
|
||||
# builds exactly that. 174 MB for the default set; the script says which
|
||||
# Hexagon generations that buys.
|
||||
# builds exactly that. 206 MB for the default set — 32 MB of it the generic
|
||||
# WebGPU build for a phone without a Qualcomm SoC; the script says which
|
||||
# Hexagon generations the rest buys.
|
||||
if [[ "${RUNTIME_DIR}" != "none" ]]; then
|
||||
if [[ ! -f "${RUNTIME_DIR}/lib/libonnxruntime.so" ]]; then
|
||||
if [[ ! -f "${RUNTIME_DIR}/lib/libonnxruntime.so" \
|
||||
|| ! -f "${RUNTIME_DIR}/lib/libonnxruntime_generic.so" ]]; then
|
||||
"${REPO}/tools/fetch-android-runtime.sh" "${RUNTIME_DIR}"
|
||||
fi
|
||||
cp "${RUNTIME_DIR}"/lib/*.so "${OUT}/staging/lib/${ABI}/"
|
||||
|
||||
@@ -31,6 +31,9 @@ ENV DEBIAN_FRONTEND=noninteractive \
|
||||
# ---------------------------------------------------------------------------
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl git git-lfs \
|
||||
# The bundled ONNX Runtime builds arrive as wheels, which are zips
|
||||
# (tools/fetch-bundled-runtimes.sh).
|
||||
unzip \
|
||||
# A *host* C compiler as well as the cross one: build scripts and
|
||||
# proc-macros are compiled for Linux and linked with `cc`, whatever
|
||||
# the target. Without it the very first build script fails with
|
||||
|
||||
@@ -90,6 +90,13 @@ for f in "${REPO}/docs/manual/media"/*; do
|
||||
done
|
||||
echo "==> staged the manual and $(ls "${STAGE}/manual/media" | wc -l) picture(s)"
|
||||
|
||||
# The two ONNX Runtime builds the app chooses between at launch
|
||||
# (docs/dev/inference.md §3.2): Intel's OpenVINO build for an Intel GPU and
|
||||
# the WebGPU build — D3D12 — for any other, both carrying the CPU provider.
|
||||
# Beside the executable under `runtimes\`, where darkroom-desktop looks.
|
||||
"${REPO}/tools/fetch-bundled-runtimes.sh" windows "${STAGE}/runtimes"
|
||||
echo "==> staged $(ls "${STAGE}/runtimes"/* | wc -l) runtime file(s)"
|
||||
|
||||
# One installer in the output directory, the one just built. The directory
|
||||
# is cached between CI runs, so after a version bump a glob over it would find
|
||||
# two and the smoke test would hand Wine both names as one path.
|
||||
|
||||
+88
-6
@@ -194,6 +194,45 @@ int8's matches land 0.45 px from f32's in the overlaps — the spread f32 shows
|
||||
|
||||
---
|
||||
|
||||
### 1.6 Intel Iris Xe, and the generic rung · 2026-10-04
|
||||
|
||||
The RTX 3050 laptop's other GPU: Raptor Lake-P's Iris Xe (96 EU), Intel's `onnxruntime-openvino`
|
||||
1.24.1 (OpenVINO 2025.4.1) and Microsoft's `onnxruntime-webgpu` 1.27.0, both PyPI wheels, through
|
||||
`ep_probe`. Three warm-ups, the median of 15 runs. Another build shared the CPU during the run, so
|
||||
the CPU columns are a little pessimistic; the GPU columns are not.
|
||||
|
||||
| Model | ORT CPU f32 | OpenVINO CPU | OpenVINO GPU f32 | **OpenVINO GPU fp16** | WebGPU (Iris Xe) |
|
||||
|---|---|---|---|---|---|
|
||||
| scrfd_500m (Fast) | 9.5 | 10.8 | 7.3 | **5.8** | 24.2 |
|
||||
| scrfd_2.5g (Balanced) | 18.5 | 16.2 | 15.5 | **11.1** | 40.8 |
|
||||
| scrfd_10g (Thorough) | 58.5 | 72.4 | 38.7 | **23.0** | 84.5 |
|
||||
| arcface_mbf (per face) | 9.5 | 11.7 | **3.0** | 2.4 | 56.6 |
|
||||
| 2d106det (landmarks) | 10.8 | 2.0 | 2.4 | **1.9** | 42.7 |
|
||||
| yolo26s-sem-ade20k | 57.0 | 47.4 | 26.0 | **16.7** | 53.0 |
|
||||
| xfeat-1024 | 23.5 | 18.5 | 19.9 | **17.5** | 34.2 |
|
||||
| migan-512 (per tile) | 330 | ✗ ¹ | 89.7 | **57.2** | 275 |
|
||||
| mosaic-fast-1408 (per tile) | 159 | 107 | 68.1 | **40.0** | 188 ² |
|
||||
| mosaic-best-1408 (per tile) | 1109 | 1670 | 947 | **604** | 1034 ² |
|
||||
|
||||
¹ OpenVINO's CPU plugin refuses the graph at initialisation. Not shipped (§3.2), so moot.
|
||||
² A later run, after `ep_probe` learned to feed the denoiser's two inputs, under heavier load: the
|
||||
CPU provider took 256 and 1034 ms in that run, so WebGPU beat it by a quarter on mosaic-fast and tied
|
||||
on mosaic-best — the only rows where it is not well behind.
|
||||
|
||||
- **OpenVINO on the Iris Xe beats ONNX Runtime's CPU provider on every model**, 1.3× on XFeat to
|
||||
5.8× on MI-GAN, with a 1–3 s compile per graph and 0.1–0.4 s from its cache. It is the Intel rung.
|
||||
fp16 is worth 1.3–1.7× over f32 here, against 1.1–1.35× on MIGraphX.
|
||||
- **Its "GPU" is OpenCL's first GPU, not Intel's.** Before `intel-compute-runtime` was installed
|
||||
the only OpenCL driver was NVIDIA's, and `device_type=GPU` ran on the RTX 3050 — slower than the
|
||||
CPU, which is the probe's to catch. Read the process's maps for `libigdrcl` before believing a
|
||||
number is the iGPU's.
|
||||
- **WebGPU is slower than the CPU on the Iris Xe** on everything but MI-GAN, as it was on the
|
||||
Adreno (§1.1), and on the RTX 3050 through Vulkan too. It is on the ladder anyway, as the generic
|
||||
rung (§2): for GPUs no vendor rung covers — an AMD card on Windows or without ROCm, a Mali — where
|
||||
it is unmeasured, and the probe's clock decides.
|
||||
- **OpenVINO's CPU plugin is not a better floor.** It wins on some graphs and loses on scrfd_10g,
|
||||
the embedder and mosaic-best, and refuses MI-GAN.
|
||||
|
||||
## 2. The shape of the answer
|
||||
|
||||
A **ladder per platform**, walked at start-up, with the first rung that builds a real session
|
||||
@@ -202,10 +241,11 @@ winning:
|
||||
| Platform | 1st | 2nd | 3rd | Floor |
|
||||
|---|---|---|---|---|
|
||||
| Android, Qualcomm with a Hexagon the shipped QNN skel covers (V68–V81) | QNN HTP, each model's quantised form (§1.5) | ORT CPU, f32 model | — | tract |
|
||||
| Android, any other SoC | ORT CPU, f32 | — | — | tract |
|
||||
| Android, any other SoC ⁶ | WebGPU (Vulkan), f32 | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, NVIDIA GPU | TensorRT, f32 model, fp16 engine | CUDA provider, f32 | ORT CPU, f32 | tract |
|
||||
| Linux, AMD GPU with ROCm | MIGraphX, f32 model, fp16 program | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, no GPU stack | ORT CPU, f32 | — | — | tract |
|
||||
| Linux / Windows, Intel GPU | OpenVINO, f32 model, fp16 program (§1.6) | ORT CPU, f32 | — | tract |
|
||||
| Linux / Windows, any other GPU ⁶ | WebGPU (Vulkan / D3D12), f32 | ORT CPU, f32 | — | tract |
|
||||
| macOS ⁵ | CoreML, f32 model, ML Program | ORT CPU, f32 | — | tract |
|
||||
|
||||
⁵ **Unmeasured**, and the one exception to the rule below: nobody here has a Mac. The rung is on
|
||||
@@ -215,11 +255,17 @@ down is refused on the third launch (§4, `attempt`). The embedder stays on the
|
||||
macOS log that shows a probe line is this row's measurement; [macos.md](macos.md) says what to
|
||||
ask for.
|
||||
|
||||
⁶ **The generic rung, unmeasured where it is meant to help.** WebGPU lost to the CPU on every GPU
|
||||
it has been timed on — the Adreno, the Iris Xe, the RTX 3050 (§1.1, §1.6) — none of which it serves
|
||||
here, since each has its own rung. It is on the ladder for the GPUs that have none, on the same
|
||||
terms as CoreML: a WebGPU that is slower than the CPU is rejected by §4's clock, one that errors is
|
||||
recorded as failed. Its first measurement on an AMD card without ROCm, or a Mali, is this row's.
|
||||
|
||||
Deliberately **not** on any ladder, with the measurement that excluded each: NNAPI (no driver),
|
||||
XNNPACK (slower than CPU, aborts on SCRFD), WebGPU (slower than CPU), the Adreno through QNN (works,
|
||||
but never where the Hexagon does not also), CUDA int8 (slower than CUDA f32), the ROCm provider
|
||||
(gone: §1.3). A rung is added to this table by a measurement on this page, not by a provider
|
||||
existing.
|
||||
XNNPACK (slower than CPU, aborts on SCRFD), the Adreno through QNN (works, but never where the
|
||||
Hexagon does not also), CUDA int8 (slower than CUDA f32), the ROCm provider (gone: §1.3),
|
||||
OpenVINO's CPU plugin as a floor (§1.6). A rung is added to this table by a measurement on this
|
||||
page, not by a provider existing — the two footnoted rows are the exceptions, and say so.
|
||||
|
||||
The AMD ladder has no middle rung. TensorRT falls back to the CUDA provider while its engines
|
||||
compile; MIGraphX has no such twin, so its fallback is the CPU provider, and the minute or two of
|
||||
@@ -295,6 +341,40 @@ it.
|
||||
|
||||
Both positions are D13 territory and are recorded there (§12).
|
||||
|
||||
Two more runtimes ship in every desktop package since 0.23 (§3.2), and their parts are all
|
||||
redistributable:
|
||||
|
||||
| Component | Licence | Shipped |
|
||||
|---|---|---|
|
||||
| Intel's `onnxruntime-openvino` build, with OpenVINO 2025.4.1 and oneTBB | MIT; Apache-2.0; Apache-2.0 | Linux and Windows packages, texts beside the libraries |
|
||||
| Microsoft's WebGPU build (Dawn inside); on Windows the DirectX shader compiler | MIT; LLVM / MIT | Linux and Windows packages |
|
||||
| Microsoft's stock `onnxruntime-android` (WebGPU) | MIT | The APK, as `libonnxruntime_generic.so` |
|
||||
|
||||
### 3.2 Several runtimes, one per process
|
||||
|
||||
A runtime carries one vendor's providers: Intel's build has OpenVINO, the `onnxruntime-gpu` wheel
|
||||
CUDA and TensorRT, a ROCm build MIGraphX, Microsoft's WebGPU build the generic rung, the APK's QNN
|
||||
build the Hexagon. No prebuilt carries two vendors, and `set_api` takes one table per process.
|
||||
So a device that may hold several — the package's OpenVINO and WebGPU builds, a CUDA build the user
|
||||
fetched, the distribution's ROCm build — has to choose which to load *before* the probe, and
|
||||
cannot choose by trying.
|
||||
|
||||
`api::install` opens every runtime on the search list, asks each for `GetAvailableProviders`, and
|
||||
loads the one that scores highest against the GPUs `hardware::detect` reads from files: a vendor
|
||||
rung on its own vendor's GPU (NVIDIA driver, `/dev/kfd`, a Qualcomm SoC, macOS) above OpenVINO on
|
||||
an Intel GPU (PCI vendor `0x8086`; on Windows Intel's DCH driver package) above WebGPU above a
|
||||
CPU-only build. Equal scores keep the search order, a perfect fit ends the search — the APK's QNN
|
||||
build is listed first, so on a Qualcomm device the generic build is never opened — and
|
||||
`DARKROOM_ORT_DIR` wins outright. The losers stay mapped: unloading a C++ runtime whose static
|
||||
constructors ran is a crash at exit waiting to happen.
|
||||
|
||||
The desktop packages install the two bundled builds under `runtimes/openvino` and
|
||||
`runtimes/webgpu` beside each place a package installs to, from
|
||||
`tools/fetch-bundled-runtimes.sh` (PyPI wheels pinned by SHA-256, pruned to the native libraries:
|
||||
81 + 31 MB on Linux, 67 + 42 MB on Windows). On Windows the chosen runtime's directory is put on
|
||||
`PATH`, because Intel's build leaves OpenVINO's DLLs for the loader to find there. The Flatpak has
|
||||
no Intel OpenCL driver in its sandbox, so an Intel machine there settles on the CPU.
|
||||
|
||||
---
|
||||
|
||||
## 4. Selection — the probe, its cache, and what it may not do
|
||||
@@ -603,6 +683,8 @@ device are comparable. *Acceptance:* M3.
|
||||
**D13 — updated.** The runtime half is reopened to the extent of §3: the Rust build stays C-free
|
||||
under `alternative-backend`; packages may install a dynamically loaded ONNX Runtime and, per §3.1,
|
||||
the Qualcomm QNN runtime; the NVIDIA libraries are not bundled. The licensing half is unchanged.
|
||||
Since 0.23 every desktop package bundles two runtimes — Intel's OpenVINO build and the WebGPU
|
||||
build — and the APK a second, generic one; the engine loads the one that fits the GPU (§3.2).
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ Denominators are parsed from [`requirements.md`](requirements.md) at run time, n
|
||||
|
||||
| Metric | Value |
|
||||
|---|---|
|
||||
| Source files scanned | 502 |
|
||||
| Source files scanned | 503 |
|
||||
| TRACES tags found | 2112 |
|
||||
| Requirements defined | 193 |
|
||||
| Requirements deferred (post-v1) | 24 |
|
||||
@@ -141,7 +141,7 @@ _None._
|
||||
| FR-PLAT-AND-3 | [`core/dr-catalog/src/jobs.rs:1`](../../core/dr-catalog/src/jobs.rs#L1), [`core/dr-catalog/src/runner.rs:1`](../../core/dr-catalog/src/runner.rs#L1), [`core/dr-catalog/tests/every_queued_kind_has_a_consumer.rs:1`](../../core/dr-catalog/tests/every_queued_kind_has_a_consumer.rs#L1) |
|
||||
| FR-PLAT-AND-4 | [`core/dr-catalog/src/runner.rs:1`](../../core/dr-catalog/src/runner.rs#L1) |
|
||||
| FR-PLAT-AND-5 | [`apps/darkroom-android/src/lib.rs:166`](../../apps/darkroom-android/src/lib.rs#L166), [`core/dr-gpu/src/adjust.rs:1472`](../../core/dr-gpu/src/adjust.rs#L1472), [`core/dr-gpu/src/detail.rs:161`](../../core/dr-gpu/src/detail.rs#L161), [`core/dr-gpu/src/detail.rs:533`](../../core/dr-gpu/src/detail.rs#L533), [`core/dr-gpu/tests/detail_stage.rs:415`](../../core/dr-gpu/tests/detail_stage.rs#L415), [`ui/dr-ui/src/develop/render.rs:654`](../../ui/dr-ui/src/develop/render.rs#L654), [`ui/dr-ui/src/identity_ui.rs:125`](../../ui/dr-ui/src/identity_ui.rs#L125), [`ui/dr-ui/src/lib.rs:1309`](../../ui/dr-ui/src/lib.rs#L1309), [`ui/dr-ui/src/lib.rs:1745`](../../ui/dr-ui/src/lib.rs#L1745), [`ui/dr-ui/src/memory.rs:163`](../../ui/dr-ui/src/memory.rs#L163), [`ui/dr-ui/src/memory.rs:1`](../../ui/dr-ui/src/memory.rs#L1) |
|
||||
| FR-PLAT-AND-6 | [`apps/darkroom-android/src/lib.rs:472`](../../apps/darkroom-android/src/lib.rs#L472) |
|
||||
| FR-PLAT-AND-6 | [`apps/darkroom-android/src/lib.rs:477`](../../apps/darkroom-android/src/lib.rs#L477) |
|
||||
| FR-PLAT-LIN-1 | [`core/dr-types/src/settings.rs:1`](../../core/dr-types/src/settings.rs#L1), [`platform/dr-plat/src/dirs.rs:1`](../../platform/dr-plat/src/dirs.rs#L1), [`platform/dr-plat/src/storage.rs:344`](../../platform/dr-plat/src/storage.rs#L344), [`ui/dr-ui/src/lib.rs:1486`](../../ui/dr-ui/src/lib.rs#L1486), [`ui/dr-ui/src/preset_store.rs:1`](../../ui/dr-ui/src/preset_store.rs#L1), [`ui/dr-ui/src/settings_store.rs:1`](../../ui/dr-ui/src/settings_store.rs#L1) |
|
||||
| FR-PLAT-LIN-2 | [`platform/dr-plat/src/display.rs:1`](../../platform/dr-plat/src/display.rs#L1), [`platform/dr-plat/src/display/wayland.rs:1`](../../platform/dr-plat/src/display/wayland.rs#L1), [`platform/dr-plat/src/display/x11.rs:1`](../../platform/dr-plat/src/display/x11.rs#L1) |
|
||||
| FR-PLAT-WIN-1 | [`platform/dr-plat/src/dirs.rs:1`](../../platform/dr-plat/src/dirs.rs#L1) |
|
||||
|
||||
+14
-7
@@ -4,7 +4,7 @@
|
||||
# makes `makepkg -si` in this directory install what you are actually working
|
||||
# on. Swap `source` for a tagged tarball when there is something to release.
|
||||
pkgname=darkroom
|
||||
pkgver=0.22.1
|
||||
pkgver=0.23.0
|
||||
# Back to 1 with the version: a new pkgver is a new archive name, so there is
|
||||
# nothing for makepkg to reuse and nothing for a release number to disambiguate.
|
||||
pkgrel=1
|
||||
@@ -15,14 +15,17 @@ license=('GPL-3.0-or-later')
|
||||
# Runtime: Vulkan for wgpu, and a Secret Service implementation for the
|
||||
# Nextcloud credentials (FR-NC-2) — gnome-keyring or kwallet both provide it.
|
||||
depends=('vulkan-icd-loader' 'fontconfig' 'libxkbcommon')
|
||||
makedepends=('cargo' 'git')
|
||||
# ONNX Runtime is loaded from /usr/lib at launch if a package put it there
|
||||
# (docs/inference.md §3): the CPU build is 8–10× the built-in tract, the
|
||||
# ROCm build adds the MIGraphX rung on an AMD GPU. Neither is required.
|
||||
makedepends=('cargo' 'git' 'curl' 'unzip')
|
||||
# Two ONNX Runtime builds ship in /usr/lib/darkroom/runtimes — Intel's
|
||||
# OpenVINO build and the generic WebGPU one, both 8–10× the built-in tract
|
||||
# on the CPU alone — and the app opens every runtime it finds and keeps the
|
||||
# one that fits the GPU (docs/dev/inference.md §3.2). The ROCm build in
|
||||
# /usr/lib adds the MIGraphX rung on an AMD GPU and outranks both there;
|
||||
# Intel's OpenCL driver is what lets OpenVINO reach an Intel GPU.
|
||||
optdepends=('gnome-keyring: store Nextcloud credentials'
|
||||
'kwallet: store Nextcloud credentials'
|
||||
'onnxruntime-cpu: run the neural models on every core'
|
||||
'onnxruntime-rocm: run the neural models on an AMD GPU')
|
||||
'onnxruntime-rocm: run the neural models on an AMD GPU'
|
||||
'intel-compute-runtime: run the neural models on an Intel GPU')
|
||||
options=('!lto') # the workspace sets its own LTO in Cargo.toml
|
||||
|
||||
_repo="$(cd "${startdir}/.." && pwd)"
|
||||
@@ -59,6 +62,10 @@ package() {
|
||||
|
||||
install -Dm644 "README.md" "${pkgdir}/usr/share/doc/${pkgname}/README.md"
|
||||
|
||||
# The bundled runtimes, where darkroom-desktop's search finds them
|
||||
# (`runtimes/` under /usr/lib/darkroom). Pinned wheels, checked by hash.
|
||||
./tools/fetch-bundled-runtimes.sh linux "${pkgdir}/usr/lib/darkroom/runtimes"
|
||||
|
||||
# The manual: the rendered page and its pictures, where the app's Help
|
||||
# opens it (dr_ui::manual, through dr_plat::system_data_dirs). Offline by
|
||||
# design — the help sheet's "See it" links land here, on a machine that
|
||||
|
||||
@@ -185,6 +185,15 @@ modules:
|
||||
install -Dm644 "models/scene/$m" "/app/share/darkroom/models/$m"
|
||||
done
|
||||
|
||||
# The two runtimes the app chooses between at launch
|
||||
# (docs/dev/inference.md §3.2): Intel's OpenVINO build and the generic
|
||||
# WebGPU one, each with ONNX Runtime's CPU provider — 8–10× the
|
||||
# built-in tract on any machine. Pinned wheels, fetched over the
|
||||
# build's network. In the sandbox the OpenVINO rung finds no Intel
|
||||
# OpenCL driver and the probe settles on the CPU; WebGPU reaches the
|
||||
# GPU through the runtime's Vulkan.
|
||||
- ./tools/fetch-bundled-runtimes.sh linux /app/lib/darkroom/runtimes
|
||||
|
||||
- install -Dm644 README.md /app/share/doc/darkroom/README.md
|
||||
|
||||
sources:
|
||||
|
||||
@@ -84,6 +84,12 @@ Section "DarkRoom" SecMain
|
||||
SetOutPath "$INSTDIR\manual"
|
||||
File /r "${STAGE}\manual\*"
|
||||
|
||||
; ONNX Runtime, twice: Intel's OpenVINO build and the WebGPU build. The
|
||||
; application opens both and keeps the one that fits the GPU
|
||||
; (docs/dev/inference.md §3.2); each also runs the models on every core.
|
||||
SetOutPath "$INSTDIR\runtimes"
|
||||
File /r "${STAGE}\runtimes\*"
|
||||
|
||||
WriteUninstaller "$INSTDIR\uninstall.exe"
|
||||
|
||||
; Add/Remove Programs. HKCU, to match the per-user install.
|
||||
@@ -123,6 +129,7 @@ Section "Uninstall"
|
||||
Delete "$INSTDIR\uninstall.exe"
|
||||
RMDir /r "$INSTDIR\models"
|
||||
RMDir /r "$INSTDIR\manual"
|
||||
RMDir /r "$INSTDIR\runtimes"
|
||||
RMDir "$INSTDIR"
|
||||
|
||||
Delete "$SMPROGRAMS\${NAME}\${NAME}.lnk"
|
||||
|
||||
@@ -3,17 +3,23 @@
|
||||
#
|
||||
# ./tools/fetch-android-runtime.sh [DEST]
|
||||
#
|
||||
# Two Maven artefacts, pinned to each other by ONNX Runtime's own POM:
|
||||
# Three Maven artefacts, the first two pinned to each other by ONNX Runtime's
|
||||
# own POM:
|
||||
#
|
||||
# com.microsoft.onnxruntime:onnxruntime-android-qnn MIT
|
||||
# com.qualcomm.qti:qnn-runtime Qualcomm AI Engine Direct SDK licence
|
||||
# com.microsoft.onnxruntime:onnxruntime-android MIT
|
||||
#
|
||||
# The first is ONNX Runtime built with the CPU, QNN, XNNPACK, NNAPI and WebGPU
|
||||
# providers; the second is Qualcomm's HTP backend — the ARM-side compiler and
|
||||
# the per-generation Hexagon "skel" the DSP loads. Both ship as AARs whose
|
||||
# `jni/arm64-v8a/` is what an APK's `lib/arm64-v8a/` wants, so this script
|
||||
# unpacks exactly that and nothing else, plus the licence texts, which travel
|
||||
# with the libraries (§3.1).
|
||||
# The first is ONNX Runtime built with the CPU and QNN providers; the second
|
||||
# is Qualcomm's HTP backend — the ARM-side compiler and the per-generation
|
||||
# Hexagon "skel" the DSP loads. The third is Microsoft's stock build, which
|
||||
# carries WebGPU (Dawn on Vulkan) and the QNN build does not: the generic
|
||||
# rung for a phone without a Qualcomm SoC (§3.2). It lands as
|
||||
# `libonnxruntime_generic.so` beside the QNN build; the engine opens the QNN
|
||||
# build first, and on a Qualcomm device stops there. All three ship as AARs
|
||||
# whose `jni/arm64-v8a/` is what an APK's `lib/arm64-v8a/` wants, so this
|
||||
# script unpacks exactly that and nothing else, plus the licence texts, which
|
||||
# travel with the libraries (§3.1).
|
||||
#
|
||||
# ## What is and is not taken from the Qualcomm package
|
||||
#
|
||||
@@ -65,6 +71,8 @@ fetch com.microsoft.onnxruntime onnxruntime-android-qnn "${ORT_VERSION}" \
|
||||
"${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar"
|
||||
fetch com.qualcomm.qti qnn-runtime "${QNN_VERSION}" \
|
||||
"${DEST}/aar/qnn-runtime-${QNN_VERSION}.aar"
|
||||
fetch com.microsoft.onnxruntime onnxruntime-android "${ORT_VERSION}" \
|
||||
"${DEST}/aar/onnxruntime-android-${ORT_VERSION}.aar"
|
||||
|
||||
# The ONNX Runtime POM names the QNN version it was built against; a pair
|
||||
# that disagrees loads and then fails at the first graph, which is the kind
|
||||
@@ -82,6 +90,8 @@ rm -rf "${DEST}/lib"
|
||||
mkdir -p "${DEST}/lib"
|
||||
unzip -q -o -j "${DEST}/aar/onnxruntime-android-qnn-${ORT_VERSION}.aar" \
|
||||
'jni/arm64-v8a/libonnxruntime.so' -d "${DEST}/lib"
|
||||
unzip -q -o -p "${DEST}/aar/onnxruntime-android-${ORT_VERSION}.aar" \
|
||||
'jni/arm64-v8a/libonnxruntime.so' > "${DEST}/lib/libonnxruntime_generic.so"
|
||||
members=(jni/arm64-v8a/libQnnHtp.so jni/arm64-v8a/libQnnHtpPrepare.so jni/arm64-v8a/libQnnSystem.so)
|
||||
for arch in ${QNN_HTP_ARCHS}; do
|
||||
members+=("jni/arm64-v8a/libQnnHtpV${arch}Skel.so" "jni/arm64-v8a/libQnnHtpV${arch}Stub.so")
|
||||
|
||||
Executable
+132
@@ -0,0 +1,132 @@
|
||||
#!/usr/bin/env bash
|
||||
# Fetch the two ONNX Runtime builds a desktop package ships
|
||||
# (docs/dev/inference.md §3.2), each into its own directory:
|
||||
#
|
||||
# DEST/openvino Intel's build: the OpenVINO rung on an Intel GPU
|
||||
# DEST/webgpu Microsoft's WebGPU build: the generic rung on any other
|
||||
#
|
||||
# ./tools/fetch-bundled-runtimes.sh {linux|windows} DEST
|
||||
#
|
||||
# Only one runtime loads per process; the app opens every one it finds and
|
||||
# keeps the one that fits the device's GPU (`dr_inference_engine::api`).
|
||||
# Both carry ONNX Runtime's CPU provider, which is the floor either way.
|
||||
# A CUDA or ROCm runtime is never bundled (§3.1) — the user's own, found
|
||||
# beside these, outranks both on its vendor's GPU.
|
||||
#
|
||||
# The wheels are PyPI's, pinned by SHA-256: the option names the engine
|
||||
# sets were read from these versions' source (CLAUDE.md, "Providers").
|
||||
# Only the native libraries are kept — not the Python bindings, not
|
||||
# OpenVINO's CPU plugin (ONNX Runtime's CPU provider is the floor), not
|
||||
# the duplicate versioned copies a wheel holds as files.
|
||||
#
|
||||
# Licences: ONNX Runtime MIT, OpenVINO Apache-2.0, oneTBB Apache-2.0, the
|
||||
# DirectX shader compiler (Windows WebGPU) LLVM/MIT; the texts go beside
|
||||
# the libraries.
|
||||
set -euo pipefail
|
||||
|
||||
PLATFORM="${1:?usage: fetch-bundled-runtimes.sh linux|windows DEST}"
|
||||
DEST="${2:?usage: fetch-bundled-runtimes.sh linux|windows DEST}"
|
||||
PYPI="https://files.pythonhosted.org/packages"
|
||||
|
||||
case "${PLATFORM}" in
|
||||
linux)
|
||||
ORT_OPENVINO="${PYPI}/08/07/f225999919f56506b603aaa3ff837ad563ab26f86906ed7fa7e5abcd849e/onnxruntime_openvino-1.24.1-cp313-cp313-manylinux_2_28_x86_64.whl"
|
||||
ORT_OPENVINO_SHA=2c3bb73e68ac27f4891af8a595c1faf574ec68b772e6583c90a0b997a1822782
|
||||
ORT_WEBGPU="${PYPI}/7a/4e/782b2457b863e1748866b82d918323e61c2d0108a01906a278e7a86a3a55/onnxruntime_webgpu-1.27.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl"
|
||||
ORT_WEBGPU_SHA=3eb30b487d2b428d2e5c1ac538177ab6336cce9b204197135dc5a16753b0258e
|
||||
# The Linux wheel carries OpenVINO itself, under the names the provider
|
||||
# was linked against.
|
||||
OPENVINO_KEEP=(libonnxruntime.so.1.24.1 libonnxruntime_providers_shared.so
|
||||
libonnxruntime_providers_openvino.so libopenvino.so.2541
|
||||
libopenvino_onnx_frontend.so.2541 libopenvino_intel_gpu_plugin.so
|
||||
libtbb.so.12 libtbbmalloc.so)
|
||||
WEBGPU_KEEP=(libonnxruntime.so.1.27.0 libonnxruntime_providers_shared.so)
|
||||
;;
|
||||
windows)
|
||||
ORT_OPENVINO="${PYPI}/3e/92/46ae2cd565961a89189900f385bb2f13a9fa731ea4674001d23720fbb1e0/onnxruntime_openvino-1.24.1-cp313-cp313-win_amd64.whl"
|
||||
ORT_OPENVINO_SHA=434bf49aa71393c577a456c9d76c98e6d6958a833fa0876793e3d5437b5a511a
|
||||
ORT_WEBGPU="${PYPI}/dd/f3/6294f9617e97035771593d604729a420a848150054cc1c422a47a4915412/onnxruntime_webgpu-1.27.0-cp313-cp313-win_amd64.whl"
|
||||
ORT_WEBGPU_SHA=c45377099fcf23ae87427eb52e2b1d35415cabb4e3419dc645f9ee08f730e9aa
|
||||
# The Windows wheel leaves OpenVINO to the `openvino` wheel; its DLLs
|
||||
# go in the same directory, which the engine puts on the DLL search
|
||||
# path when it chooses this runtime.
|
||||
OPENVINO_LIBS="${PYPI}/3c/e5/da52a86cc5f1c86871002712429cdcca0c0dbff12dfbce730b05db60340b/openvino-2025.4.1-20426-cp313-cp313-win_amd64.whl"
|
||||
OPENVINO_LIBS_SHA=a28eef35e3ed497c3238eb8f3d1ee90647c449707a8b0a7630758cd15555d8dd
|
||||
OPENVINO_KEEP=(onnxruntime.dll onnxruntime_providers_shared.dll
|
||||
onnxruntime_providers_openvino.dll openvino.dll
|
||||
openvino_onnx_frontend.dll openvino_intel_gpu_plugin.dll
|
||||
tbb12.dll tbbmalloc.dll)
|
||||
WEBGPU_KEEP=(onnxruntime.dll onnxruntime_providers_shared.dll dxcompiler.dll dxil.dll)
|
||||
;;
|
||||
*)
|
||||
echo "error: platform is linux or windows, not ${PLATFORM}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
WORK="$(mktemp -d)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
|
||||
# fetch URL SHA256 DIR: download a wheel, check it, unpack it into DIR.
|
||||
fetch() {
|
||||
local url="$1" sha="$2" dir="$3" whl
|
||||
whl="${WORK}/$(basename "${url}")"
|
||||
echo "==> $(basename "${url}")"
|
||||
curl -fsSL -o "${whl}" "${url}"
|
||||
echo "${sha} ${whl}" | sha256sum -c --quiet - || {
|
||||
echo "error: $(basename "${url}") does not match its pinned SHA-256" >&2
|
||||
exit 1
|
||||
}
|
||||
mkdir -p "${dir}"
|
||||
unzip -q -o "${whl}" -d "${dir}"
|
||||
}
|
||||
|
||||
# keep FROM TO NAMES...: copy only NAMES, every one of which must exist.
|
||||
keep() {
|
||||
local from="$1" to="$2" name
|
||||
shift 2
|
||||
mkdir -p "${to}"
|
||||
for name in "$@"; do
|
||||
local src
|
||||
src="$(find "${from}" -name "${name}" -type f | head -1)"
|
||||
[[ -n "${src}" ]] || {
|
||||
echo "error: ${name} is not in the wheel" >&2
|
||||
exit 1
|
||||
}
|
||||
install -m755 "${src}" "${to}/${name}"
|
||||
done
|
||||
}
|
||||
|
||||
fetch "${ORT_OPENVINO}" "${ORT_OPENVINO_SHA}" "${WORK}/openvino"
|
||||
if [[ -n "${OPENVINO_LIBS:-}" ]]; then
|
||||
fetch "${OPENVINO_LIBS}" "${OPENVINO_LIBS_SHA}" "${WORK}/openvino"
|
||||
fi
|
||||
fetch "${ORT_WEBGPU}" "${ORT_WEBGPU_SHA}" "${WORK}/webgpu"
|
||||
|
||||
rm -rf "${DEST}/openvino" "${DEST}/webgpu"
|
||||
keep "${WORK}/openvino" "${DEST}/openvino" "${OPENVINO_KEEP[@]}"
|
||||
keep "${WORK}/webgpu" "${DEST}/webgpu" "${WEBGPU_KEEP[@]}"
|
||||
# The wheels' own licence and notice files — ONNX Runtime keeps its in the
|
||||
# package directory, OpenVINO in `dist-info` — beside what they cover,
|
||||
# each under the name of the directory it came from: two wheels unpacked
|
||||
# into one directory both carry a `LICENSE`.
|
||||
for rt in openvino webgpu; do
|
||||
find "${WORK}/${rt}" -maxdepth 3 -type f \
|
||||
\( -iname 'LICENSE*' -o -iname 'NOTICE*' -o -iname 'ThirdPartyNotices*' \) |
|
||||
while IFS= read -r f; do
|
||||
wheel="$(basename "$(dirname "${f}")" .dist-info)"
|
||||
[[ "${wheel}" == licenses ]] && wheel="$(basename "$(dirname "$(dirname "${f}")")" .dist-info)"
|
||||
install -m644 "${f}" "${DEST}/${rt}/${wheel}.$(basename "${f}")"
|
||||
done
|
||||
done
|
||||
|
||||
# The Linux wheel bundles OpenVINO without its licence; Apache-2.0 asks
|
||||
# for the text beside the binaries. From the tag the libraries were built at.
|
||||
if [[ "${PLATFORM}" == linux ]]; then
|
||||
curl -fsSL -o "${DEST}/openvino/openvino-2025.4.1.LICENSE" \
|
||||
"https://raw.githubusercontent.com/openvinotoolkit/openvino/2025.4.1/LICENSE"
|
||||
echo "c71d239df91726fc519c6eb72d318ec65820627232b2f796219e87dcf35d0ab4 ${DEST}/openvino/openvino-2025.4.1.LICENSE" |
|
||||
sha256sum -c --quiet -
|
||||
fi
|
||||
|
||||
du -sh "${DEST}/openvino" "${DEST}/webgpu"
|
||||
Reference in New Issue
Block a user