Let the probe's clock be its proof, not disable_cpu_ep_fallback
The strict flag refused the Hexagon over the ten quantise/dequantise nodes at the graph's edges that QNN declines by policy, which cost microseconds. A provider that hands real work to the CPU is slower than the CPU floor and the timing already rejects it; the tablet measured 2.3 ms on the NPU against a 29.7 ms floor.
This commit is contained in:
@@ -89,7 +89,7 @@ pub fn run() {
|
||||
}
|
||||
log::info!("inference: compiling {name} for {}", rung.label());
|
||||
let started = std::time::Instant::now();
|
||||
match crate::session::build(rung, role, &bytes, &cfg, false) {
|
||||
match crate::session::build(rung, role, &bytes, &cfg) {
|
||||
Ok(session) => {
|
||||
drop(session);
|
||||
let mut s = state().lock().unwrap();
|
||||
|
||||
@@ -253,7 +253,7 @@ fn acquire(role: Role, form: Form, bytes: &Arc<[u8]>) -> Result<Acquired, Error>
|
||||
|
||||
// Built outside the registry lock: a TensorRT engine load is long enough
|
||||
// that another role's acquire should not wait on it.
|
||||
let session = session::build(rung, role, bytes, &cfg, false)?;
|
||||
let session = session::build(rung, role, bytes, &cfg)?;
|
||||
log::debug!("inference: {role:?} loaded on {}", rung.label());
|
||||
let entry = Arc::new(Loaded {
|
||||
rung,
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
//! Walk the ladder, once, by building real sessions (docs/inference.md §4).
|
||||
//!
|
||||
//! A rung is taken when a strict session builds on it, runs, and is faster
|
||||
//! than the floor. Both halves matter: a provider can register and then fail
|
||||
//! at partition time, and a provider can take a graph and run it slower than
|
||||
//! the CPU would have. The outcome is cached against a fingerprint of the
|
||||
//! A rung is taken when a session builds on it, runs, and is faster than
|
||||
//! the floor. Both halves matter: a provider can register and then fail at
|
||||
//! partition time, and a provider can take a graph — or quietly hand most
|
||||
//! of it back to the CPU — and run it slower than the CPU would have. The outcome is cached against a fingerprint of the
|
||||
//! runtime, the driver, the hardware and the models, and trusted until any
|
||||
//! of those changes.
|
||||
|
||||
@@ -135,9 +135,9 @@ fn probe_model(cfg: &Config) -> Option<(Role, PathBuf)> {
|
||||
smallest(Some(Role::Detector)).or_else(|| smallest(None))
|
||||
}
|
||||
|
||||
/// Build strictly, run once for the engine, then time three runs; the
|
||||
/// median in milliseconds and, for a compiling rung, the cache key of the
|
||||
/// engine this just built.
|
||||
/// Build, run once for the engine, then time three runs; the median in
|
||||
/// milliseconds and, for a compiling rung, the cache key of the engine this
|
||||
/// just built.
|
||||
fn time_rung(
|
||||
rung: Rung,
|
||||
role: Role,
|
||||
@@ -157,8 +157,8 @@ fn time_rung(
|
||||
};
|
||||
let bytes = std::fs::read(&path).map_err(|e| e.to_string())?;
|
||||
let started = Instant::now();
|
||||
let mut session = crate::session::build(rung, role, &bytes, cfg, true)
|
||||
.map_err(|e| first_line(&e.to_string()))?;
|
||||
let mut session =
|
||||
crate::session::build(rung, role, &bytes, cfg).map_err(|e| first_line(&e.to_string()))?;
|
||||
log::info!(
|
||||
"inference: {} session built in {:.1} s",
|
||||
rung.label(),
|
||||
|
||||
@@ -7,22 +7,16 @@ use crate::{Config, Role, Rung};
|
||||
|
||||
/// Build a session for `bytes` on `rung`.
|
||||
///
|
||||
/// `strict` is the probe's flag: with it, a provider that would hand any
|
||||
/// node to the CPU fails the build instead, so "the session built" means
|
||||
/// "the provider took the graph" and not "the provider registered" (§4).
|
||||
pub fn build(
|
||||
rung: Rung,
|
||||
role: Role,
|
||||
bytes: &[u8],
|
||||
cfg: &Config,
|
||||
strict: bool,
|
||||
) -> ort::Result<Session> {
|
||||
/// Not strict about the CPU: `session.disable_cpu_ep_fallback` was tried as
|
||||
/// the probe's proof that a provider took the graph, and it refuses the
|
||||
/// Hexagon over the ten quantise/dequantise nodes at the graph's edges that
|
||||
/// QNN declines by policy and that cost microseconds. The probe's proof is
|
||||
/// its clock instead (§4): a provider that hands real work to the CPU is
|
||||
/// slower than the CPU floor and rejected by the same measurement.
|
||||
pub fn build(rung: Rung, role: Role, bytes: &[u8], cfg: &Config) -> ort::Result<Session> {
|
||||
let mut b = Session::builder()?
|
||||
.with_optimization_level(GraphOptimizationLevel::Level3)?
|
||||
.with_intra_threads(threads(cfg))?;
|
||||
if strict && rung != Rung::Cpu {
|
||||
b = b.with_config_entry("session.disable_cpu_ep_fallback", "1")?;
|
||||
}
|
||||
// A Hexagon session loads the compiled context when there is one and
|
||||
// compiles it from the model when there is not; the engine thread is
|
||||
// what makes the second case rare (§6).
|
||||
|
||||
Reference in New Issue
Block a user