From cbbe67fbd7d16af66be503e86c7b3920a7a4e4ce Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sat, 19 Sep 2026 16:13:19 +0200 Subject: [PATCH] Let the probe's clock be its proof, not disable_cpu_ep_fallback The strict flag refused the Hexagon over the ten quantise/dequantise nodes at the graph's edges that QNN declines by policy, which cost microseconds. A provider that hands real work to the CPU is slower than the CPU floor and the timing already rejects it; the tablet measured 2.3 ms on the NPU against a 29.7 ms floor. --- core/dr-inference-engine/src/engines.rs | 2 +- core/dr-inference-engine/src/lib.rs | 2 +- core/dr-inference-engine/src/probe.rs | 18 +++++++++--------- core/dr-inference-engine/src/session.rs | 20 +++++++------------- docs/inference.md | 13 ++++++++----- 5 files changed, 26 insertions(+), 29 deletions(-) diff --git a/core/dr-inference-engine/src/engines.rs b/core/dr-inference-engine/src/engines.rs index 12df295..2502255 100644 --- a/core/dr-inference-engine/src/engines.rs +++ b/core/dr-inference-engine/src/engines.rs @@ -89,7 +89,7 @@ pub fn run() { } log::info!("inference: compiling {name} for {}", rung.label()); let started = std::time::Instant::now(); - match crate::session::build(rung, role, &bytes, &cfg, false) { + match crate::session::build(rung, role, &bytes, &cfg) { Ok(session) => { drop(session); let mut s = state().lock().unwrap(); diff --git a/core/dr-inference-engine/src/lib.rs b/core/dr-inference-engine/src/lib.rs index c9f3c4a..e44fe1d 100644 --- a/core/dr-inference-engine/src/lib.rs +++ b/core/dr-inference-engine/src/lib.rs @@ -253,7 +253,7 @@ fn acquire(role: Role, form: Form, bytes: &Arc<[u8]>) -> Result // Built outside the registry lock: a TensorRT engine load is long enough // that another role's acquire should not wait on it. - let session = session::build(rung, role, bytes, &cfg, false)?; + let session = session::build(rung, role, bytes, &cfg)?; log::debug!("inference: {role:?} loaded on {}", rung.label()); let entry = Arc::new(Loaded { rung, diff --git a/core/dr-inference-engine/src/probe.rs b/core/dr-inference-engine/src/probe.rs index 22cd714..a9fd877 100644 --- a/core/dr-inference-engine/src/probe.rs +++ b/core/dr-inference-engine/src/probe.rs @@ -1,9 +1,9 @@ //! Walk the ladder, once, by building real sessions (docs/inference.md §4). //! -//! A rung is taken when a strict session builds on it, runs, and is faster -//! than the floor. Both halves matter: a provider can register and then fail -//! at partition time, and a provider can take a graph and run it slower than -//! the CPU would have. The outcome is cached against a fingerprint of the +//! A rung is taken when a session builds on it, runs, and is faster than +//! the floor. Both halves matter: a provider can register and then fail at +//! partition time, and a provider can take a graph — or quietly hand most +//! of it back to the CPU — and run it slower than the CPU would have. The outcome is cached against a fingerprint of the //! runtime, the driver, the hardware and the models, and trusted until any //! of those changes. @@ -135,9 +135,9 @@ fn probe_model(cfg: &Config) -> Option<(Role, PathBuf)> { smallest(Some(Role::Detector)).or_else(|| smallest(None)) } -/// Build strictly, run once for the engine, then time three runs; the -/// median in milliseconds and, for a compiling rung, the cache key of the -/// engine this just built. +/// Build, run once for the engine, then time three runs; the median in +/// milliseconds and, for a compiling rung, the cache key of the engine this +/// just built. fn time_rung( rung: Rung, role: Role, @@ -157,8 +157,8 @@ fn time_rung( }; let bytes = std::fs::read(&path).map_err(|e| e.to_string())?; let started = Instant::now(); - let mut session = crate::session::build(rung, role, &bytes, cfg, true) - .map_err(|e| first_line(&e.to_string()))?; + let mut session = + crate::session::build(rung, role, &bytes, cfg).map_err(|e| first_line(&e.to_string()))?; log::info!( "inference: {} session built in {:.1} s", rung.label(), diff --git a/core/dr-inference-engine/src/session.rs b/core/dr-inference-engine/src/session.rs index 485ae82..6b70373 100644 --- a/core/dr-inference-engine/src/session.rs +++ b/core/dr-inference-engine/src/session.rs @@ -7,22 +7,16 @@ use crate::{Config, Role, Rung}; /// Build a session for `bytes` on `rung`. /// -/// `strict` is the probe's flag: with it, a provider that would hand any -/// node to the CPU fails the build instead, so "the session built" means -/// "the provider took the graph" and not "the provider registered" (§4). -pub fn build( - rung: Rung, - role: Role, - bytes: &[u8], - cfg: &Config, - strict: bool, -) -> ort::Result { +/// Not strict about the CPU: `session.disable_cpu_ep_fallback` was tried as +/// the probe's proof that a provider took the graph, and it refuses the +/// Hexagon over the ten quantise/dequantise nodes at the graph's edges that +/// QNN declines by policy and that cost microseconds. The probe's proof is +/// its clock instead (§4): a provider that hands real work to the CPU is +/// slower than the CPU floor and rejected by the same measurement. +pub fn build(rung: Rung, role: Role, bytes: &[u8], cfg: &Config) -> ort::Result { let mut b = Session::builder()? .with_optimization_level(GraphOptimizationLevel::Level3)? .with_intra_threads(threads(cfg))?; - if strict && rung != Rung::Cpu { - b = b.with_config_entry("session.disable_cpu_ep_fallback", "1")?; - } // A Hexagon session loads the compiled context when there is one and // compiles it from the model when there is not; the engine thread is // what makes the second case rare (§6). diff --git a/docs/inference.md b/docs/inference.md index 709fb7a..55a72c3 100644 --- a/docs/inference.md +++ b/docs/inference.md @@ -195,11 +195,14 @@ state where the provider registered and the session then failed, and a provider took the graph, and rejected every node at partition time. The probe therefore: 1. Loads the runtime library (§3), or falls to tract and stops. -2. For each rung in this platform's ladder, in order: builds a session for the **smallest model - in the set** (`scrfd_500m`) on that provider with `error_on_failure`, runs it once on a fixed - input, and reads back the provider assignment from the session — the rung is taken only if the - provider ran **at least 95% of the graph's nodes**. A provider that silently hands the graph to - the CPU is the CPU rung with extra overhead, and the app should say "CPU". +2. Times the **smallest detector** on the CPU provider first — the floor. Then, for each rung + in this platform's ladder, in order: builds a session for the same model on that provider + with `error_on_failure`, runs it once on a fixed input, and times three more runs. **The rung + is taken only if its median beats the floor.** That one measurement is the proof the provider + took the graph: one that silently hands the work to the CPU is the CPU rung with extra + overhead, slower than the floor, and rejected. (ONNX Runtime's + `session.disable_cpu_ep_fallback` was the first draft of this proof and refuses the Hexagon + over the ten quantise/dequantise nodes at the graph's edges that QNN declines by policy.) 3. Records the outcome — rung, runtime version, provider version, device identity (GPU name and compute capability; SoC model and Hexagon arch), and the models' content hashes — to a small file beside `shared_face_models_dir`. The next start-up trusts the file **unless** any of those