diff --git a/core/dr-inference-engine/src/engines.rs b/core/dr-inference-engine/src/engines.rs index 31525a3..6539298 100644 --- a/core/dr-inference-engine/src/engines.rs +++ b/core/dr-inference-engine/src/engines.rs @@ -48,12 +48,35 @@ pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf { /// CoreML's own cache key leaves out the weights of a model loaded from /// memory (`session::coreml`), and one per runtime version, which wrote it. pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf { + model_dir(cfg, "coreml", bytes) +} + +/// Where OpenVINO compiles `bytes` to, at one precision. OpenVINO hashes +/// the model it is given, weights included, but a key that leaves out +/// what is being varied has cost a day before (CLAUDE.md, "Providers"), +/// and a directory per model and precision costs nothing: the precision +/// is a compile option, and the two forms are different programs. +pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf { + model_dir( + cfg, + if fp16 { + "openvino/fp16" + } else { + "openvino/f32" + }, + bytes, + ) +} + +/// `///`: one per +/// model, and one per runtime version, which wrote it. +fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf { let runtime = match crate::api::runtime() { crate::Runtime::OnnxRuntime { version, .. } => version, crate::Runtime::Tract => "tract".into(), }; cfg.cache_dir - .join("coreml") + .join(provider) .join(runtime) .join(format!("{:016x}", hash(bytes))) } diff --git a/core/dr-inference-engine/src/lib.rs b/core/dr-inference-engine/src/lib.rs index 721bb28..876d893 100644 --- a/core/dr-inference-engine/src/lib.rs +++ b/core/dr-inference-engine/src/lib.rs @@ -110,6 +110,16 @@ pub enum Rung { /// embedder stays on the CPU, as on the Hexagon: the Neural Engine /// computes in fp16 (§7). CoreMl, + /// Intel, through OpenVINO on the integrated or Arc GPU. Desktop only. + /// Compiles a program per model, as MIGraphX does, so the CPU is its + /// fallback; fp16 on the same terms as TensorRT (§7). + OpenVino, + /// Any other GPU, through ONNX Runtime's WebGPU provider: Dawn on + /// Vulkan, D3D12 or Metal. The generic rung, for a GPU no vendor rung + /// covers. Measured slower than the CPU on every GPU it has been timed + /// on (§1), so it is on the ladder for the GPUs it has not, and the + /// probe's clock is what keeps it off the rest. + WebGpu, } impl Rung { @@ -121,6 +131,8 @@ impl Rung { Rung::MiGraphX => "MIGraphX", Rung::Hexagon => "Hexagon NPU", Rung::CoreMl => "CoreML", + Rung::OpenVino => "OpenVINO", + Rung::WebGpu => "WebGPU", } } @@ -129,7 +141,13 @@ impl Rung { fn fallback(self) -> Rung { match self { Rung::TensorRt => Rung::Cuda, - Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::Cuda | Rung::Cpu => Rung::Cpu, + Rung::MiGraphX + | Rung::Hexagon + | Rung::CoreMl + | Rung::OpenVino + | Rung::WebGpu + | Rung::Cuda + | Rung::Cpu => Rung::Cpu, } } @@ -137,7 +155,7 @@ impl Rung { fn compiles(self) -> bool { matches!( self, - Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl + Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::OpenVino ) } @@ -230,7 +248,7 @@ impl Status { pub fn line(&self) -> String { let form = match self.rung { Rung::Hexagon => " · quantised", - Rung::TensorRt | Rung::MiGraphX => " · fp16", + Rung::TensorRt | Rung::MiGraphX | Rung::OpenVino => " · fp16", _ => "", }; format!("{}{} · {}", self.rung.label(), form, self.runtime.label()) @@ -725,6 +743,8 @@ mod tests { Rung::TensorRt, Rung::MiGraphX, Rung::CoreMl, + Rung::OpenVino, + Rung::WebGpu, ] { assert_eq!(rung.form(Role::Detector), F32); } @@ -765,6 +785,29 @@ mod tests { assert_eq!(on(&s, Role::Embedder), Rung::Cpu); } + /// OpenVINO compiles a program per model, so a request waits on the CPU + /// until the engine thread has built it; WebGPU builds in the session + /// and serves at once. Both take every role in f32 graphs. + #[test] + fn openvino_waits_for_its_program_and_webgpu_does_not() { + let hash = engines::hash(b"detector"); + let mut s = State { + config: Config::default(), + cache: Cache::default(), + probing: false, + wanted: 0, + }; + let on = |s: &State, rung| effective_rung(s, rung, Role::Detector, Form::F32, hash); + assert_eq!(on(&s, Rung::OpenVino), Rung::Cpu); + s.cache + .compiled + .insert(engines::key_of(Rung::OpenVino, hash)); + assert_eq!(on(&s, Rung::OpenVino), Rung::OpenVino); + assert_eq!(on(&s, Rung::WebGpu), Rung::WebGpu); + let embedder = effective_rung(&s, Rung::WebGpu, Role::Embedder, Form::F32, hash); + assert_eq!(embedder, Rung::WebGpu); + } + #[test] fn the_status_reports_only_the_rungs_above_the_selection() { let _serial = serial(); diff --git a/core/dr-inference-engine/src/probe.rs b/core/dr-inference-engine/src/probe.rs index be8f895..119105b 100644 --- a/core/dr-inference-engine/src/probe.rs +++ b/core/dr-inference-engine/src/probe.rs @@ -14,18 +14,27 @@ use crate::{api::Runtime, state, Cache, Config, Form, Role, Rung}; /// The rungs to try on this platform, best first, under the user's ceiling. fn ladder(ceiling: Option) -> Vec { + // WebGPU is the generic rung (§2): it is reached only on a runtime that + // carries it, which `api` loads where no vendor's runtime fits the + // device, and kept only where it beats the CPU. #[cfg(target_os = "android")] - let all = [Rung::Hexagon]; + let all = [Rung::Hexagon, Rung::WebGpu]; // Unmeasured (§2 ⁵): it is on the ladder because the probe's clock and // `attempt` make a wrong guess cost one slow or failed probe, not a // slow or crashing app. #[cfg(target_os = "macos")] let all = [Rung::CoreMl]; - // A desktop has one vendor's GPU; the other vendor's providers are - // "not enabled in this build" or a library that fails to load, and - // either answer arrives in milliseconds. + // A runtime carries one vendor's providers, chosen for this device's + // GPU (`api`); the others are "not enabled in this build", and that + // answer arrives in milliseconds. #[cfg(not(any(target_os = "android", target_os = "macos")))] - let all = [Rung::TensorRt, Rung::Cuda, Rung::MiGraphX]; + let all = [ + Rung::TensorRt, + Rung::Cuda, + Rung::MiGraphX, + Rung::OpenVino, + Rung::WebGpu, + ]; all.into_iter() .filter(|r| ceiling.is_none_or(|c| *r <= c)) .collect() diff --git a/core/dr-inference-engine/src/session.rs b/core/dr-inference-engine/src/session.rs index b60dbef..1b09930 100644 --- a/core/dr-inference-engine/src/session.rs +++ b/core/dr-inference-engine/src/session.rs @@ -55,9 +55,10 @@ fn build_with( let context = (rung == Rung::Hexagon).then(|| crate::engines::context_path(cfg, bytes)); let ready = context.as_ref().is_some_and(|p| p.is_file()); // What the rung keeps for this model: the context the Hexagon is to - // write, or the directory CoreML compiles into. + // write, or the directory CoreML or OpenVINO compiles into. let per_model = match rung { Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)), + Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))), _ if ready => None, _ => context.clone(), }; @@ -101,6 +102,13 @@ fn with_runtime_log( .with_log_level(level)?) } +/// Whether `role` runs in fp16 on a rung that offers it: everything but the +/// embedder, whose comparability across devices is worth more than its +/// fraction of a millisecond (§7). +fn fp16(role: Role) -> bool { + role != Role::Embedder +} + /// The intra-op pool: what the config says, else the cores less two for /// the compositor and the decoder (§9). tract ignores it. fn threads(cfg: &Config) -> usize { @@ -137,7 +145,7 @@ fn providers( // (NFR-RES-2). CUDA behind it takes any node TensorRT declines. Ok(b.with_execution_providers([ ep::TensorRT::default() - .with_fp16(role != Role::Embedder) + .with_fp16(fp16(role)) .with_engine_cache(true) .with_engine_cache_path(&cache) .with_timing_cache(true) @@ -154,7 +162,7 @@ fn providers( // directory, keyed on the graph, the GPU and its own version // but not the precision: hence one directory per precision. // The CPU takes any node it declines. - let fp16 = role != Role::Embedder; + let fp16 = fp16(role); let cache = cfg .cache_dir .join("migraphx") @@ -164,6 +172,16 @@ fn providers( migraphx(&mut b, fp16, &cache)?; Ok(b) } + Rung::OpenVino => { + let mut b = b; + openvino(&mut b, fp16(role), per_model)?; + Ok(b) + } + Rung::WebGpu => { + let mut b = b; + webgpu(&mut b)?; + Ok(b) + } Rung::Hexagon => unreachable!("the Hexagon rung is not on a desktop ladder"), } } @@ -219,15 +237,79 @@ fn migraphx( b: &mut ort::session::builder::SessionBuilder, fp16: bool, cache: &std::path::Path, +) -> ort::Result<()> { + append( + b, + c"MIGraphX", + &[ + ("migraphx_fp16_enable", if fp16 { "1" } else { "0" }.into()), + ( + "migraphx_model_cache_dir", + cache.to_string_lossy().into_owned(), + ), + ], + ) +} + +/// OpenVINO on the GPU, compiling into `cache`. +/// +/// The option names are those `openvino_provider_factory.cc` reads at 1.24, +/// the version of Intel's `onnxruntime-openvino` build. `GPU` is OpenVINO's +/// first OpenCL GPU: the Intel one on a hybrid laptop with both drivers +/// installed, but an NVIDIA card through its OpenCL when Intel's is absent +/// — slower than the CPU there, and rejected by the probe's clock. The +/// precision is always named: the GPU plugin's own default is fp16, and +/// the embedder must not get it (§7). +#[cfg(not(target_os = "android"))] +fn openvino( + b: &mut ort::session::builder::SessionBuilder, + fp16: bool, + cache: Option<&std::path::Path>, +) -> ort::Result<()> { + let mut options = vec![ + ("device_type", "GPU".to_string()), + ("precision", if fp16 { "FP16" } else { "FP32" }.into()), + ]; + if let Some(dir) = cache { + let _ = std::fs::create_dir_all(dir); + options.push(("cache_dir", dir.to_string_lossy().into_owned())); + } + append(b, c"OpenVINO", &options) +} + +/// WebGPU on the high-performance adapter: the discrete GPU where there is +/// one, since the integrated one on a machine with both is the one this +/// rung is least likely to beat the CPU on. The key is as +/// `webgpu_provider_options.h` spells it, without the `ep..` prefix +/// the runtime adds. +fn webgpu(b: &mut ort::session::builder::SessionBuilder) -> ort::Result<()> { + append( + b, + c"WebGPU", + &[("powerPreference", "high-performance".to_string())], + ) +} + +/// Register the provider `name` with `options` through the runtime's +/// generic key/value entry point, which takes every provider by its short +/// name and reads options at the runtime's own version — not at the +/// version `ort`'s builders were written against (CLAUDE.md, "Providers"). +fn append( + b: &mut ort::session::builder::SessionBuilder, + name: &std::ffi::CStr, + options: &[(&str, String)], ) -> ort::Result<()> { use ort::AsPointer; use std::ffi::CString; - let keys = [c"migraphx_fp16_enable", c"migraphx_model_cache_dir"]; - let values = [ - CString::new(if fp16 { "1" } else { "0" }).unwrap(), - CString::new(cache.to_string_lossy().as_bytes()) - .map_err(|e| ort::Error::new(e.to_string()))?, - ]; + let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string())); + let keys = options + .iter() + .map(|(k, _)| cstr(k)) + .collect::>>()?; + let values = options + .iter() + .map(|(_, v)| cstr(v)) + .collect::>>()?; let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect(); let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect(); // SAFETY: the documented C call over arrays that outlive it; the @@ -235,7 +317,7 @@ fn migraphx( unsafe { let status = (ort::api().SessionOptionsAppendExecutionProvider)( b.ptr_mut(), - c"MIGraphX".as_ptr(), + name.as_ptr(), key_ptrs.as_ptr(), value_ptrs.as_ptr(), keys.len(), @@ -277,7 +359,12 @@ fn providers( .build() .error_on_failure()])?) } - Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl => { + Rung::WebGpu => { + let mut b = b; + webgpu(&mut b)?; + Ok(b) + } + Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl | Rung::OpenVino => { unreachable!("no desktop rung on Android") } }