Probe the Hexagon strictly, and probe again after falling back to the CPU
0.22.0's first launch on the tablet: QNN could not create its device (QNN_DEVICE_ERROR_INVALID_CONFIG), the session built anyway with every node on the CPU behind the provider, and the probe timed that - 28.5 ms against the CPU's own 19.4 - and rejected the Hexagon. The verdict was cached under the fingerprint, so every model stayed on the CPU on every later launch: AI denoise took 30-131 s a photograph instead of seconds. The same A16W8 detector with the APK's own libraries runs on the HTP in 4.4 ms. The probe's Hexagon session now sets session.disable_cpu_ep_fallback, so a device that cannot take the graph fails the probe instead of being timed as the CPU. Only the probe: shipped graphs may keep nodes on the CPU on purpose. And a selection that fell back to the CPU after an accelerator failed or lost is probed again on the next launches, up to three probes per fingerprint; a cache written by 0.22.0 reads as never retried, so the tablet probes again once this is installed.
This commit is contained in:
@@ -31,6 +31,38 @@ fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Probes under one fingerprint that may end on the CPU after an
|
||||
/// accelerator failed or lost, before that answer is kept.
|
||||
const RETRIES: u32 = 3;
|
||||
|
||||
/// What a cached probe result is good for.
|
||||
#[derive(Debug, PartialEq)]
|
||||
enum Reuse {
|
||||
/// Use it as it is.
|
||||
Keep,
|
||||
/// Probe again: it fell back to the CPU after this many probes.
|
||||
Again(u32),
|
||||
/// Another device, runtime or model set: probe from the start.
|
||||
Fresh,
|
||||
}
|
||||
|
||||
/// The CPU because an accelerator failed or lost is asked again on the next
|
||||
/// launches, a few times: a failure can be a moment's (QNN could not create
|
||||
/// its device on 0.22.0's first launch after the update), and keeping it for
|
||||
/// good left the tablet's every model on the CPU. Bounded, so a wedged
|
||||
/// driver costs a few launches, not all.
|
||||
fn reuse(cached: &Cache, fingerprint: &str) -> Reuse {
|
||||
if cached.fingerprint != fingerprint || cached.rung.is_none() {
|
||||
return Reuse::Fresh;
|
||||
}
|
||||
let fell_back = cached.rung == Some(Rung::Cpu) && !cached.failed.is_empty();
|
||||
if fell_back && cached.attempts < RETRIES {
|
||||
Reuse::Again(cached.attempts)
|
||||
} else {
|
||||
Reuse::Keep
|
||||
}
|
||||
}
|
||||
|
||||
/// The probe body. Sets the cache and clears `probing` when done; never
|
||||
/// panics out, because a failed probe is a result (the floor) and not an
|
||||
/// error.
|
||||
@@ -38,20 +70,33 @@ pub fn run(runtime: Runtime) {
|
||||
let cfg = state().lock().unwrap().config.clone();
|
||||
let fingerprint = fingerprint(&runtime, &cfg);
|
||||
|
||||
let mut attempts = 0;
|
||||
if let Some(cached) = read_cache(&cfg) {
|
||||
if cached.fingerprint == fingerprint && cached.rung.is_some() {
|
||||
log::info!(
|
||||
"inference: cached selection {} ({})",
|
||||
cached.rung.unwrap().label(),
|
||||
cached.reason
|
||||
);
|
||||
finish(cached);
|
||||
return;
|
||||
match reuse(&cached, &fingerprint) {
|
||||
Reuse::Keep => {
|
||||
log::info!(
|
||||
"inference: cached selection {} ({})",
|
||||
cached.rung.map_or("?", |r| r.label()),
|
||||
cached.reason
|
||||
);
|
||||
finish(cached);
|
||||
return;
|
||||
}
|
||||
Reuse::Again(n) => {
|
||||
attempts = n;
|
||||
log::info!(
|
||||
"inference: probing again after falling back to the CPU ({}), attempt {} of {RETRIES}",
|
||||
cached.reason,
|
||||
n + 1
|
||||
);
|
||||
}
|
||||
Reuse::Fresh => {}
|
||||
}
|
||||
}
|
||||
|
||||
let mut cache = Cache {
|
||||
fingerprint,
|
||||
attempts: attempts + 1,
|
||||
..Cache::default()
|
||||
};
|
||||
|
||||
@@ -212,8 +257,8 @@ fn time_rung(
|
||||
}
|
||||
let bytes = std::fs::read(&path).map_err(|e| e.to_string())?;
|
||||
let started = Instant::now();
|
||||
let mut session =
|
||||
crate::session::build(rung, role, &bytes, cfg).map_err(|e| first_line(&e.to_string()))?;
|
||||
let mut session = crate::session::build_probe(rung, role, &bytes, cfg)
|
||||
.map_err(|e| first_line(&e.to_string()))?;
|
||||
log::info!(
|
||||
"inference: {} session built in {:.1} s",
|
||||
rung.label(),
|
||||
@@ -494,4 +539,34 @@ mod tests {
|
||||
died_inside(&cfg, "probe TensorRT", 2);
|
||||
assert_eq!(attempt(&cfg, "probe CUDA", || 7), Ok(7));
|
||||
}
|
||||
|
||||
/// The tablet's cache after 0.22.0's first launch, as 0.22.0 wrote it:
|
||||
/// no `attempts`, the Hexagon "rejected", the CPU selected.
|
||||
const TABLET: &str = r#"{"fingerprint":"f","rung":"Cpu","reason":"Hexagon NPU 28.5 ms, slower than the CPU's 19.4 ms","compiled":[],"failed":[["Hexagon","28.5 ms, slower than the CPU's 19.4 ms"]]}"#;
|
||||
|
||||
#[test]
|
||||
fn a_fall_back_to_the_cpu_is_probed_again_a_few_times() {
|
||||
let mut cache: Cache = serde_json::from_str(TABLET).unwrap();
|
||||
assert_eq!(cache.attempts, 0, "a 0.22.0 cache reads as never retried");
|
||||
assert_eq!(reuse(&cache, "f"), Reuse::Again(0));
|
||||
cache.attempts = RETRIES - 1;
|
||||
assert_eq!(reuse(&cache, "f"), Reuse::Again(RETRIES - 1));
|
||||
cache.attempts = RETRIES;
|
||||
assert_eq!(reuse(&cache, "f"), Reuse::Keep, "then it is kept");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_accelerator_chosen_or_a_cpu_only_device_is_kept() {
|
||||
let mut cache: Cache = serde_json::from_str(TABLET).unwrap();
|
||||
cache.rung = Some(Rung::Hexagon);
|
||||
assert_eq!(reuse(&cache, "f"), Reuse::Keep);
|
||||
cache.rung = Some(Rung::Cpu);
|
||||
cache.failed.clear();
|
||||
assert_eq!(
|
||||
reuse(&cache, "f"),
|
||||
Reuse::Keep,
|
||||
"nothing failed: the only rung"
|
||||
);
|
||||
assert_eq!(reuse(&cache, "other"), Reuse::Fresh);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user