//! Time each execution provider a runtime offers, on the models this //! repository ships — the measurement docs/inference.md §1 requires before a //! rung is added to §2's ladder. //! //! DARKROOM_ORT_DIR=/usr/lib \ //! cargo run --release -p dr-inference-engine --features native,tract \ //! --example ep_probe -- models/face/scrfd_500m_640.onnx ... //! //! Prints one row per (model, provider): the median of timed runs after //! warm-ups, and the build time, which for a compiling provider is the //! number that decides whether it needs an engine cache. MIGraphX is built //! twice per precision — cold, then again from the cache it just wrote — //! so both numbers are on the page. //! //! The ROCm provider is not in the list: ONNX Runtime removed it in 1.23, //! and 1.29's `onnxruntime-rocm` ships `libonnxruntime_providers_migraphx.so` //! and nothing else for AMD. use std::path::{Path, PathBuf}; use std::time::Instant; #[derive(Clone, Copy, PartialEq)] enum Ep { Cpu, MiGraphX, MiGraphXFp16, } impl Ep { fn label(self) -> &'static str { match self { Ep::Cpu => "CPU", Ep::MiGraphX => "MIGraphX f32", Ep::MiGraphXFp16 => "MIGraphX fp16", } } } fn build(ep: Ep, bytes: &[u8], threads: usize, cache: &Path) -> ort::Result { let mut b = ort::session::Session::builder()?.with_intra_threads(threads)?; match ep { Ep::Cpu => {} Ep::MiGraphX => migraphx(&mut b, false, &cache.join("f32"))?, Ep::MiGraphXFp16 => migraphx(&mut b, true, &cache.join("fp16"))?, } b.commit_from_memory(bytes) } /// Register MIGraphX through the generic key/value API. `ort`'s own /// builder fills the legacy `OrtMIGraphXProviderOptions`, which 1.29 reads /// for its precision flags and nothing else: the model cache directory — /// the difference between a 40 s load and a 0.3 s one — only travels this /// way. The cache key is the graph, the GPU and the MIGraphX version, not /// the precision, so each precision gets its own directory. fn migraphx( b: &mut ort::session::builder::SessionBuilder, fp16: bool, cache: &Path, ) -> ort::Result<()> { use ort::AsPointer; use std::ffi::CString; std::fs::create_dir_all(cache).map_err(|e| ort::Error::new(e.to_string()))?; let keys = [c"migraphx_fp16_enable", c"migraphx_model_cache_dir"]; let values = [ CString::new(if fp16 { "1" } else { "0" }).unwrap(), CString::new(cache.to_string_lossy().as_bytes()).unwrap(), ]; let key_ptrs: Vec<_> = keys.iter().map(|k| k.as_ptr()).collect(); let value_ptrs: Vec<_> = values.iter().map(|v| v.as_ptr()).collect(); // SAFETY: the documented C call, over arrays that outlive it; the // runtime copies the strings into its own options map. unsafe { let status = (ort::api().SessionOptionsAppendExecutionProvider)( b.ptr_mut(), c"MIGraphX".as_ptr(), key_ptrs.as_ptr(), value_ptrs.as_ptr(), keys.len(), ); ort::Error::result_from_status(status) } } /// Median of `runs` timed runs over zeros, in milliseconds, after warm-ups. fn time(session: &mut ort::session::Session, warmups: usize, runs: usize) -> Result { let shape: Vec = session.inputs()[0] .dtype() .tensor_shape() .ok_or("input is not a tensor")? .iter() .map(|&d| if d > 0 { d as usize } else { 1 }) .collect(); let zeros = vec![0f32; shape.iter().product()]; let once = |s: &mut ort::session::Session| -> Result { let input = ort::value::Tensor::from_array((shape.clone(), zeros.clone())) .map_err(|e| e.to_string())?; let t = Instant::now(); let out = s.run(ort::inputs![input]).map_err(|e| e.to_string())?; let _ = out[0] .try_extract_tensor::() .map_err(|e| e.to_string())?; Ok(t.elapsed().as_secs_f64() * 1e3) }; for _ in 0..warmups { once(session)?; } let mut times = Vec::with_capacity(runs); for _ in 0..runs { times.push(once(session)?); } times.sort_by(|a, b| a.partial_cmp(b).unwrap()); Ok(times[times.len() / 2]) } fn first_line(s: &str) -> String { s.lines().next().unwrap_or("").chars().take(120).collect() } fn main() { env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info")).init(); let models: Vec = std::env::args_os().skip(1).map(PathBuf::from).collect(); if models.is_empty() { eprintln!("usage: ep_probe MODEL.onnx [MODEL.onnx ...]"); std::process::exit(2); } dr_inference_engine::ensure_runtime(); let runtime = dr_inference_engine::status().runtime; println!("runtime: {}", runtime.label()); if !runtime.is_native() { println!("(tract: no provider to compare; set DARKROOM_ORT_DIR)"); } let threads = std::thread::available_parallelism() .map(|n| n.get().saturating_sub(2).max(1)) .unwrap_or(1); println!("intra-op threads: {threads}"); let cache = std::env::temp_dir().join("darkroom-ep-probe"); let _ = std::fs::remove_dir_all(&cache); println!("compiled-program cache: {}\n", cache.display()); println!( "{:<28} {:<15} {:>10} {:>10}", "model", "provider", "build s", "median ms" ); for model in &models { let bytes = match std::fs::read(model) { Ok(b) => b, Err(e) => { println!("{:<28} read failed: {e}", name(model)); continue; } }; // A compiling provider is built twice: the second build reads the // program the first wrote, and its time is what a launch after the // first costs. let plan = [ (Ep::Cpu, false), (Ep::MiGraphX, false), (Ep::MiGraphX, true), (Ep::MiGraphXFp16, false), (Ep::MiGraphXFp16, true), ]; for (ep, cached) in plan { let started = Instant::now(); match build(ep, &bytes, threads, &cache) { Ok(mut session) => { let built = started.elapsed().as_secs_f64(); match time(&mut session, 3, 15) { Ok(ms) => println!( "{:<28} {:<15} {:>10.1} {:>10.1}{}", name(model), ep.label(), built, ms, if cached { " (from cache)" } else { "" } ), Err(e) => println!( "{:<28} {:<15} {:>10.1} {:>10} {}", name(model), ep.label(), built, "ran ✗", first_line(&e) ), } } Err(e) => println!( "{:<28} {:<15} {:>21} {}", name(model), ep.label(), "build ✗", first_line(&e.to_string()) ), } } println!(); } } fn name(p: &Path) -> String { p.file_name() .unwrap_or(p.as_os_str()) .to_string_lossy() .into_owned() }