From ecb648818bcb7345300f75fdd9adf1feea003015 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 19 Sep 2026 21:15:55 +0200 Subject: [PATCH] Search the user's own runtime directory before the system library The reference desktop's only system ONNX Runtime is Arch's onnxruntime-opt-cuda: 1.29, built without TensorRT and against cuDNN 8 on a cuDNN 9 machine. The probe rejects both providers correctly and the app runs on the CPU provider, which is right and not what anyone wants. runtime/ beside the models is now searched ahead of /usr/lib, tools/fetch-desktop-runtime.sh fills it with the four libraries from the current onnxruntime-gpu wheel (cuDNN 9, TensorRT 10), and the About caption lists every rung that lost and why, not only the first. Verified: the app selects TensorRT from that directory with no environment variable set. --- apps/darkroom-desktop/src/main.rs | 13 ++++++++---- core/dr-inference-engine/src/lib.rs | 3 +++ tools/fetch-desktop-runtime.sh | 32 +++++++++++++++++++++++++++++ ui/dr-ui/src/inference.rs | 24 +++++++++++++++++++++- 4 files changed, 67 insertions(+), 5 deletions(-) create mode 100755 tools/fetch-desktop-runtime.sh diff --git a/apps/darkroom-desktop/src/main.rs b/apps/darkroom-desktop/src/main.rs index 1b9933b..806ad28 100644 --- a/apps/darkroom-desktop/src/main.rs +++ b/apps/darkroom-desktop/src/main.rs @@ -83,10 +83,14 @@ fn main() -> anyhow::Result<()> { /// `DARKROOM_ORT_DIR` is for a developer pointing at a runtime that is not /// installed — the wheel's `capi` directory, say. Then beside the executable /// and in the package's private library directory, for a package that -/// bundles its own; then the Flatpak prefix; then the system library -/// directory, for a distribution that ships ONNX Runtime as a package of its -/// own. A system copy whose GPU providers do not load is not a problem: the -/// probe builds a real session before believing a provider. +/// bundles its own; then the user's own `runtime/` beside the models, where +/// `tools/fetch-desktop-runtime.sh` puts one; then the Flatpak prefix; then +/// the system library directory, for a distribution that ships ONNX Runtime +/// as a package of its own. The user's copy outranks the system's because +/// the system's is the one most likely to be built without the GPU +/// providers, or against the wrong cuDNN — and a system copy whose providers +/// do not load is not a problem, only a slower app: the probe builds a real +/// session before believing a provider. fn runtime_dirs() -> Vec { let mut dirs = Vec::new(); if let Some(dir) = std::env::var_os("DARKROOM_ORT_DIR") { @@ -98,6 +102,7 @@ fn runtime_dirs() -> Vec { dirs.push(bin.join("../lib/darkroom")); } } + dirs.push(dr_ui::inference::user_runtime_dir()); #[cfg(target_os = "linux")] dirs.extend([ PathBuf::from("/app/lib/darkroom"), diff --git a/core/dr-inference-engine/src/lib.rs b/core/dr-inference-engine/src/lib.rs index 6a20911..22848d2 100644 --- a/core/dr-inference-engine/src/lib.rs +++ b/core/dr-inference-engine/src/lib.rs @@ -155,6 +155,8 @@ pub struct Status { /// Engines compiled and engines wanted, for a compiling rung; `(0, 0)` /// otherwise. pub engines: (usize, usize), + /// Every rung above the selected one that was tried, and why it lost. + pub failed: Vec<(Rung, String)>, } impl Status { @@ -399,6 +401,7 @@ pub fn status() -> Status { runtime: api::runtime(), rung, reason: s.cache.reason.clone(), + failed: s.cache.failed.clone(), probing: s.probing, engines: if rung.compiles() { (s.cache.compiled.len(), s.wanted) diff --git a/tools/fetch-desktop-runtime.sh b/tools/fetch-desktop-runtime.sh new file mode 100755 index 0000000..2cd4875 --- /dev/null +++ b/tools/fetch-desktop-runtime.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +# Put a GPU-capable ONNX Runtime where the desktop app looks for one +# (docs/inference.md §3): `runtime/` beside the models in the user data +# directory, ahead of the system library. +# +# ./tools/fetch-desktop-runtime.sh [DEST] +# +# The source is the `onnxruntime-gpu` wheel: the one build that carries the +# CUDA *and* TensorRT providers against the cuDNN and TensorRT majors current +# on this machine. Distribution packages tend to have neither — Arch's +# `onnxruntime-opt-cuda` is built without TensorRT and against cuDNN 8 — and +# the probe rejects them correctly and leaves the app on the CPU provider, +# which is what this script exists to fix. Nothing NVIDIA is bundled here: +# the providers load CUDA, cuDNN and TensorRT from the system, and if those +# are missing the probe says so and the app stays on the CPU. +set -euo pipefail +DEST="${1:-${XDG_DATA_HOME:-${HOME}/.local/share}/darkroom/runtime}" +WORK="$(mktemp -d -p /var/tmp fetch-desktop-runtime.XXXXXX)" +trap 'rm -rf "${WORK}"' EXIT + +echo "==> downloading the onnxruntime-gpu wheel" +uv venv --python 3.12 "${WORK}/venv" >/dev/null +VIRTUAL_ENV="${WORK}/venv" uv pip install --quiet onnxruntime-gpu +CAPI="$(find "${WORK}/venv" -type d -path '*/onnxruntime/capi' | head -1)" +[[ -n "${CAPI}" ]] || { echo "error: no capi directory in the wheel" >&2; exit 1; } + +mkdir -p "${DEST}" +# The runtime and its provider libraries; not the Python binding. +cp "${CAPI}"/libonnxruntime.so* "${CAPI}"/libonnxruntime_providers_*.so "${DEST}/" +echo "==> runtime in ${DEST}:" +ls -1 "${DEST}" | sed 's/^/ /' +echo " (the app finds it on its next launch; Settings › About › Inference says what it chose)" diff --git a/ui/dr-ui/src/inference.rs b/ui/dr-ui/src/inference.rs index aa2a06e..38b3ae3 100644 --- a/ui/dr-ui/src/inference.rs +++ b/ui/dr-ui/src/inference.rs @@ -60,6 +60,18 @@ pub fn init(runtime_dirs: Vec) { crate::memory::evict_at(crate::memory::Tier::Gpu, dr_inference_engine::release_all); } +/// Where a person can put a runtime by hand: `runtime/` beside the models, +/// searched before any system library. The system copy on the reference +/// desktop is built without TensorRT and against the wrong cuDNN, and a +/// working one is four files from the `onnxruntime-gpu` wheel; this is +/// where they go, and `tools/fetch-desktop-runtime.sh` puts them there. +pub fn user_runtime_dir() -> PathBuf { + crate::library::shared_face_models_dir() + .parent() + .map(|p| p.join("runtime")) + .unwrap_or_else(|| PathBuf::from("runtime")) +} + /// Which form the current backend loads `detector` in, given the files on /// this device — the fact `faces.model_id` has to carry (§7). /// @@ -94,8 +106,18 @@ pub fn about_lines() -> (String, String) { status.engines.0, status.engines.1 ) - } else { + } else if status.failed.is_empty() { status.reason + } else { + // Every rung that was tried and why it lost, not only the first: + // "TensorRT: not enabled in this build" says nothing about why CUDA + // was not taken instead. + status + .failed + .iter() + .map(|(rung, why)| format!("{}: {why}", rung.label())) + .collect::>() + .join(" · ") }; (line, detail) }