Search the user's own runtime directory before the system library
The reference desktop's only system ONNX Runtime is Arch's onnxruntime-opt-cuda: 1.29, built without TensorRT and against cuDNN 8 on a cuDNN 9 machine. The probe rejects both providers correctly and the app runs on the CPU provider, which is right and not what anyone wants. runtime/ beside the models is now searched ahead of /usr/lib, tools/fetch-desktop-runtime.sh fills it with the four libraries from the current onnxruntime-gpu wheel (cuDNN 9, TensorRT 10), and the About caption lists every rung that lost and why, not only the first. Verified: the app selects TensorRT from that directory with no environment variable set.
This commit is contained in:
@@ -83,10 +83,14 @@ fn main() -> anyhow::Result<()> {
|
||||
/// `DARKROOM_ORT_DIR` is for a developer pointing at a runtime that is not
|
||||
/// installed — the wheel's `capi` directory, say. Then beside the executable
|
||||
/// and in the package's private library directory, for a package that
|
||||
/// bundles its own; then the Flatpak prefix; then the system library
|
||||
/// directory, for a distribution that ships ONNX Runtime as a package of its
|
||||
/// own. A system copy whose GPU providers do not load is not a problem: the
|
||||
/// probe builds a real session before believing a provider.
|
||||
/// bundles its own; then the user's own `runtime/` beside the models, where
|
||||
/// `tools/fetch-desktop-runtime.sh` puts one; then the Flatpak prefix; then
|
||||
/// the system library directory, for a distribution that ships ONNX Runtime
|
||||
/// as a package of its own. The user's copy outranks the system's because
|
||||
/// the system's is the one most likely to be built without the GPU
|
||||
/// providers, or against the wrong cuDNN — and a system copy whose providers
|
||||
/// do not load is not a problem, only a slower app: the probe builds a real
|
||||
/// session before believing a provider.
|
||||
fn runtime_dirs() -> Vec<PathBuf> {
|
||||
let mut dirs = Vec::new();
|
||||
if let Some(dir) = std::env::var_os("DARKROOM_ORT_DIR") {
|
||||
@@ -98,6 +102,7 @@ fn runtime_dirs() -> Vec<PathBuf> {
|
||||
dirs.push(bin.join("../lib/darkroom"));
|
||||
}
|
||||
}
|
||||
dirs.push(dr_ui::inference::user_runtime_dir());
|
||||
#[cfg(target_os = "linux")]
|
||||
dirs.extend([
|
||||
PathBuf::from("/app/lib/darkroom"),
|
||||
|
||||
@@ -155,6 +155,8 @@ pub struct Status {
|
||||
/// Engines compiled and engines wanted, for a compiling rung; `(0, 0)`
|
||||
/// otherwise.
|
||||
pub engines: (usize, usize),
|
||||
/// Every rung above the selected one that was tried, and why it lost.
|
||||
pub failed: Vec<(Rung, String)>,
|
||||
}
|
||||
|
||||
impl Status {
|
||||
@@ -399,6 +401,7 @@ pub fn status() -> Status {
|
||||
runtime: api::runtime(),
|
||||
rung,
|
||||
reason: s.cache.reason.clone(),
|
||||
failed: s.cache.failed.clone(),
|
||||
probing: s.probing,
|
||||
engines: if rung.compiles() {
|
||||
(s.cache.compiled.len(), s.wanted)
|
||||
|
||||
Executable
+32
@@ -0,0 +1,32 @@
|
||||
#!/usr/bin/env bash
|
||||
# Put a GPU-capable ONNX Runtime where the desktop app looks for one
|
||||
# (docs/inference.md §3): `runtime/` beside the models in the user data
|
||||
# directory, ahead of the system library.
|
||||
#
|
||||
# ./tools/fetch-desktop-runtime.sh [DEST]
|
||||
#
|
||||
# The source is the `onnxruntime-gpu` wheel: the one build that carries the
|
||||
# CUDA *and* TensorRT providers against the cuDNN and TensorRT majors current
|
||||
# on this machine. Distribution packages tend to have neither — Arch's
|
||||
# `onnxruntime-opt-cuda` is built without TensorRT and against cuDNN 8 — and
|
||||
# the probe rejects them correctly and leaves the app on the CPU provider,
|
||||
# which is what this script exists to fix. Nothing NVIDIA is bundled here:
|
||||
# the providers load CUDA, cuDNN and TensorRT from the system, and if those
|
||||
# are missing the probe says so and the app stays on the CPU.
|
||||
set -euo pipefail
|
||||
DEST="${1:-${XDG_DATA_HOME:-${HOME}/.local/share}/darkroom/runtime}"
|
||||
WORK="$(mktemp -d -p /var/tmp fetch-desktop-runtime.XXXXXX)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
|
||||
echo "==> downloading the onnxruntime-gpu wheel"
|
||||
uv venv --python 3.12 "${WORK}/venv" >/dev/null
|
||||
VIRTUAL_ENV="${WORK}/venv" uv pip install --quiet onnxruntime-gpu
|
||||
CAPI="$(find "${WORK}/venv" -type d -path '*/onnxruntime/capi' | head -1)"
|
||||
[[ -n "${CAPI}" ]] || { echo "error: no capi directory in the wheel" >&2; exit 1; }
|
||||
|
||||
mkdir -p "${DEST}"
|
||||
# The runtime and its provider libraries; not the Python binding.
|
||||
cp "${CAPI}"/libonnxruntime.so* "${CAPI}"/libonnxruntime_providers_*.so "${DEST}/"
|
||||
echo "==> runtime in ${DEST}:"
|
||||
ls -1 "${DEST}" | sed 's/^/ /'
|
||||
echo " (the app finds it on its next launch; Settings › About › Inference says what it chose)"
|
||||
@@ -60,6 +60,18 @@ pub fn init(runtime_dirs: Vec<PathBuf>) {
|
||||
crate::memory::evict_at(crate::memory::Tier::Gpu, dr_inference_engine::release_all);
|
||||
}
|
||||
|
||||
/// Where a person can put a runtime by hand: `runtime/` beside the models,
|
||||
/// searched before any system library. The system copy on the reference
|
||||
/// desktop is built without TensorRT and against the wrong cuDNN, and a
|
||||
/// working one is four files from the `onnxruntime-gpu` wheel; this is
|
||||
/// where they go, and `tools/fetch-desktop-runtime.sh` puts them there.
|
||||
pub fn user_runtime_dir() -> PathBuf {
|
||||
crate::library::shared_face_models_dir()
|
||||
.parent()
|
||||
.map(|p| p.join("runtime"))
|
||||
.unwrap_or_else(|| PathBuf::from("runtime"))
|
||||
}
|
||||
|
||||
/// Which form the current backend loads `detector` in, given the files on
|
||||
/// this device — the fact `faces.model_id` has to carry (§7).
|
||||
///
|
||||
@@ -94,8 +106,18 @@ pub fn about_lines() -> (String, String) {
|
||||
status.engines.0,
|
||||
status.engines.1
|
||||
)
|
||||
} else {
|
||||
} else if status.failed.is_empty() {
|
||||
status.reason
|
||||
} else {
|
||||
// Every rung that was tried and why it lost, not only the first:
|
||||
// "TensorRT: not enabled in this build" says nothing about why CUDA
|
||||
// was not taken instead.
|
||||
status
|
||||
.failed
|
||||
.iter()
|
||||
.map(|(rung, why)| format!("{}: {why}", rung.label()))
|
||||
.collect::<Vec<_>>()
|
||||
.join(" · ")
|
||||
};
|
||||
(line, detail)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user