Add OpenVINO and WebGPU rungs to the inference ladder

OpenVINO is the Intel rung: the integrated or Arc GPU, fp16 for every
role but the embedder, a compiled program per model kept in a directory
per model, precision and runtime version. On the Iris Xe it beats ONNX
Runtime's CPU provider on every shipped model — scrfd_10g 23 ms against
58, the scene model 17 against 57, MI-GAN 57 against 330, a denoise tile
40 against 158.

WebGPU is the generic rung for a GPU no vendor rung covers. It was slower
than the CPU on the Iris Xe, the RTX 3050 and the Adreno, so it is on the
ladder for the GPUs it has not been timed on, behind the probe's clock.

MIGraphX's registration becomes one generic key/value helper that all
three share, with option names read from each runtime's own source.
This commit is contained in:
2026-10-04 21:00:15 -04:00
parent 555ec0efb3
commit 87c405eb46
4 changed files with 182 additions and 20 deletions
+24 -1
View File
@@ -48,12 +48,35 @@ pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf {
/// CoreML's own cache key leaves out the weights of a model loaded from
/// memory (`session::coreml`), and one per runtime version, which wrote it.
pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
model_dir(cfg, "coreml", bytes)
}
/// Where OpenVINO compiles `bytes` to, at one precision. OpenVINO hashes
/// the model it is given, weights included, but a key that leaves out
/// what is being varied has cost a day before (CLAUDE.md, "Providers"),
/// and a directory per model and precision costs nothing: the precision
/// is a compile option, and the two forms are different programs.
pub fn openvino_dir(cfg: &Config, bytes: &[u8], fp16: bool) -> PathBuf {
model_dir(
cfg,
if fp16 {
"openvino/fp16"
} else {
"openvino/f32"
},
bytes,
)
}
/// `<cache>/<provider>/<runtime version>/<hash of the bytes>`: one per
/// model, and one per runtime version, which wrote it.
fn model_dir(cfg: &Config, provider: &str, bytes: &[u8]) -> PathBuf {
let runtime = match crate::api::runtime() {
crate::Runtime::OnnxRuntime { version, .. } => version,
crate::Runtime::Tract => "tract".into(),
};
cfg.cache_dir
.join("coreml")
.join(provider)
.join(runtime)
.join(format!("{:016x}", hash(bytes)))
}
+46 -3
View File
@@ -110,6 +110,16 @@ pub enum Rung {
/// embedder stays on the CPU, as on the Hexagon: the Neural Engine
/// computes in fp16 (§7).
CoreMl,
/// Intel, through OpenVINO on the integrated or Arc GPU. Desktop only.
/// Compiles a program per model, as MIGraphX does, so the CPU is its
/// fallback; fp16 on the same terms as TensorRT (§7).
OpenVino,
/// Any other GPU, through ONNX Runtime's WebGPU provider: Dawn on
/// Vulkan, D3D12 or Metal. The generic rung, for a GPU no vendor rung
/// covers. Measured slower than the CPU on every GPU it has been timed
/// on (§1), so it is on the ladder for the GPUs it has not, and the
/// probe's clock is what keeps it off the rest.
WebGpu,
}
impl Rung {
@@ -121,6 +131,8 @@ impl Rung {
Rung::MiGraphX => "MIGraphX",
Rung::Hexagon => "Hexagon NPU",
Rung::CoreMl => "CoreML",
Rung::OpenVino => "OpenVINO",
Rung::WebGpu => "WebGPU",
}
}
@@ -129,7 +141,13 @@ impl Rung {
fn fallback(self) -> Rung {
match self {
Rung::TensorRt => Rung::Cuda,
Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::Cuda | Rung::Cpu => Rung::Cpu,
Rung::MiGraphX
| Rung::Hexagon
| Rung::CoreMl
| Rung::OpenVino
| Rung::WebGpu
| Rung::Cuda
| Rung::Cpu => Rung::Cpu,
}
}
@@ -137,7 +155,7 @@ impl Rung {
fn compiles(self) -> bool {
matches!(
self,
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::OpenVino
)
}
@@ -230,7 +248,7 @@ impl Status {
pub fn line(&self) -> String {
let form = match self.rung {
Rung::Hexagon => " · quantised",
Rung::TensorRt | Rung::MiGraphX => " · fp16",
Rung::TensorRt | Rung::MiGraphX | Rung::OpenVino => " · fp16",
_ => "",
};
format!("{}{} · {}", self.rung.label(), form, self.runtime.label())
@@ -725,6 +743,8 @@ mod tests {
Rung::TensorRt,
Rung::MiGraphX,
Rung::CoreMl,
Rung::OpenVino,
Rung::WebGpu,
] {
assert_eq!(rung.form(Role::Detector), F32);
}
@@ -765,6 +785,29 @@ mod tests {
assert_eq!(on(&s, Role::Embedder), Rung::Cpu);
}
/// OpenVINO compiles a program per model, so a request waits on the CPU
/// until the engine thread has built it; WebGPU builds in the session
/// and serves at once. Both take every role in f32 graphs.
#[test]
fn openvino_waits_for_its_program_and_webgpu_does_not() {
let hash = engines::hash(b"detector");
let mut s = State {
config: Config::default(),
cache: Cache::default(),
probing: false,
wanted: 0,
};
let on = |s: &State, rung| effective_rung(s, rung, Role::Detector, Form::F32, hash);
assert_eq!(on(&s, Rung::OpenVino), Rung::Cpu);
s.cache
.compiled
.insert(engines::key_of(Rung::OpenVino, hash));
assert_eq!(on(&s, Rung::OpenVino), Rung::OpenVino);
assert_eq!(on(&s, Rung::WebGpu), Rung::WebGpu);
let embedder = effective_rung(&s, Rung::WebGpu, Role::Embedder, Form::F32, hash);
assert_eq!(embedder, Rung::WebGpu);
}
#[test]
fn the_status_reports_only_the_rungs_above_the_selection() {
let _serial = serial();
+14 -5
View File
@@ -14,18 +14,27 @@ use crate::{api::Runtime, state, Cache, Config, Form, Role, Rung};
/// The rungs to try on this platform, best first, under the user's ceiling.
fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
// WebGPU is the generic rung (§2): it is reached only on a runtime that
// carries it, which `api` loads where no vendor's runtime fits the
// device, and kept only where it beats the CPU.
#[cfg(target_os = "android")]
let all = [Rung::Hexagon];
let all = [Rung::Hexagon, Rung::WebGpu];
// Unmeasured (§2 ⁵): it is on the ladder because the probe's clock and
// `attempt` make a wrong guess cost one slow or failed probe, not a
// slow or crashing app.
#[cfg(target_os = "macos")]
let all = [Rung::CoreMl];
// A desktop has one vendor's GPU; the other vendor's providers are
// "not enabled in this build" or a library that fails to load, and
// either answer arrives in milliseconds.
// A runtime carries one vendor's providers, chosen for this device's
// GPU (`api`); the others are "not enabled in this build", and that
// answer arrives in milliseconds.
#[cfg(not(any(target_os = "android", target_os = "macos")))]
let all = [Rung::TensorRt, Rung::Cuda, Rung::MiGraphX];
let all = [
Rung::TensorRt,
Rung::Cuda,
Rung::MiGraphX,
Rung::OpenVino,
Rung::WebGpu,
];
all.into_iter()
.filter(|r| ceiling.is_none_or(|c| *r <= c))
.collect()
+98 -11
View File
@@ -55,9 +55,10 @@ fn build_with(
let context = (rung == Rung::Hexagon).then(|| crate::engines::context_path(cfg, bytes));
let ready = context.as_ref().is_some_and(|p| p.is_file());
// What the rung keeps for this model: the context the Hexagon is to
// write, or the directory CoreML compiles into.
// write, or the directory CoreML or OpenVINO compiles into.
let per_model = match rung {
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
Rung::OpenVino => Some(crate::engines::openvino_dir(cfg, bytes, fp16(role))),
_ if ready => None,
_ => context.clone(),
};
@@ -101,6 +102,13 @@ fn with_runtime_log(
.with_log_level(level)?)
}
/// Whether `role` runs in fp16 on a rung that offers it: everything but the
/// embedder, whose comparability across devices is worth more than its
/// fraction of a millisecond (§7).
fn fp16(role: Role) -> bool {
role != Role::Embedder
}
/// The intra-op pool: what the config says, else the cores less two for
/// the compositor and the decoder (§9). tract ignores it.
fn threads(cfg: &Config) -> usize {
@@ -137,7 +145,7 @@ fn providers(
// (NFR-RES-2). CUDA behind it takes any node TensorRT declines.
Ok(b.with_execution_providers([
ep::TensorRT::default()
.with_fp16(role != Role::Embedder)
.with_fp16(fp16(role))
.with_engine_cache(true)
.with_engine_cache_path(&cache)
.with_timing_cache(true)
@@ -154,7 +162,7 @@ fn providers(
// directory, keyed on the graph, the GPU and its own version
// but not the precision: hence one directory per precision.
// The CPU takes any node it declines.
let fp16 = role != Role::Embedder;
let fp16 = fp16(role);
let cache = cfg
.cache_dir
.join("migraphx")
@@ -164,6 +172,16 @@ fn providers(
migraphx(&mut b, fp16, &cache)?;
Ok(b)
}
Rung::OpenVino => {
let mut b = b;
openvino(&mut b, fp16(role), per_model)?;
Ok(b)
}
Rung::WebGpu => {
let mut b = b;
webgpu(&mut b)?;
Ok(b)
}
Rung::Hexagon => unreachable!("the Hexagon rung is not on a desktop ladder"),
}
}
@@ -219,15 +237,79 @@ fn migraphx(
b: &mut ort::session::builder::SessionBuilder,
fp16: bool,
cache: &std::path::Path,
) -> ort::Result<()> {
append(
b,
c"MIGraphX",
&[
("migraphx_fp16_enable", if fp16 { "1" } else { "0" }.into()),
(
"migraphx_model_cache_dir",
cache.to_string_lossy().into_owned(),
),
],
)
}
/// OpenVINO on the GPU, compiling into `cache`.
///
/// The option names are those `openvino_provider_factory.cc` reads at 1.24,
/// the version of Intel's `onnxruntime-openvino` build. `GPU` is OpenVINO's
/// first OpenCL GPU: the Intel one on a hybrid laptop with both drivers
/// installed, but an NVIDIA card through its OpenCL when Intel's is absent
/// — slower than the CPU there, and rejected by the probe's clock. The
/// precision is always named: the GPU plugin's own default is fp16, and
/// the embedder must not get it (§7).
#[cfg(not(target_os = "android"))]
fn openvino(
b: &mut ort::session::builder::SessionBuilder,
fp16: bool,
cache: Option<&std::path::Path>,
) -> ort::Result<()> {
let mut options = vec![
("device_type", "GPU".to_string()),
("precision", if fp16 { "FP16" } else { "FP32" }.into()),
];
if let Some(dir) = cache {
let _ = std::fs::create_dir_all(dir);
options.push(("cache_dir", dir.to_string_lossy().into_owned()));
}
append(b, c"OpenVINO", &options)
}
/// WebGPU on the high-performance adapter: the discrete GPU where there is
/// one, since the integrated one on a machine with both is the one this
/// rung is least likely to beat the CPU on. The key is as
/// `webgpu_provider_options.h` spells it, without the `ep.<name>.` prefix
/// the runtime adds.
fn webgpu(b: &mut ort::session::builder::SessionBuilder) -> ort::Result<()> {
append(
b,
c"WebGPU",
&[("powerPreference", "high-performance".to_string())],
)
}
/// Register the provider `name` with `options` through the runtime's
/// generic key/value entry point, which takes every provider by its short
/// name and reads options at the runtime's own version — not at the
/// version `ort`'s builders were written against (CLAUDE.md, "Providers").
fn append(
b: &mut ort::session::builder::SessionBuilder,
name: &std::ffi::CStr,
options: &[(&str, String)],
) -> ort::Result<()> {
use ort::AsPointer;
use std::ffi::CString;
let keys = [c"migraphx_fp16_enable", c"migraphx_model_cache_dir"];
let values = [
CString::new(if fp16 { "1" } else { "0" }).unwrap(),
CString::new(cache.to_string_lossy().as_bytes())
.map_err(|e| ort::Error::new(e.to_string()))?,
];
let cstr = |s: &str| CString::new(s).map_err(|e| ort::Error::new(e.to_string()));
let keys = options
.iter()
.map(|(k, _)| cstr(k))
.collect::<ort::Result<Vec<_>>>()?;
let values = options
.iter()
.map(|(_, v)| cstr(v))
.collect::<ort::Result<Vec<_>>>()?;
let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect();
let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect();
// SAFETY: the documented C call over arrays that outlive it; the
@@ -235,7 +317,7 @@ fn migraphx(
unsafe {
let status = (ort::api().SessionOptionsAppendExecutionProvider)(
b.ptr_mut(),
c"MIGraphX".as_ptr(),
name.as_ptr(),
key_ptrs.as_ptr(),
value_ptrs.as_ptr(),
keys.len(),
@@ -277,7 +359,12 @@ fn providers(
.build()
.error_on_failure()])?)
}
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl => {
Rung::WebGpu => {
let mut b = b;
webgpu(&mut b)?;
Ok(b)
}
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl | Rung::OpenVino => {
unreachable!("no desktop rung on Android")
}
}