Add a CoreML rung on macOS
The macOS ladder was the CPU provider alone, with CoreML listed as a gap. It is now CoreML, then the CPU, then tract — unmeasured, since nobody here has a Mac, and safe to ship unmeasured because the probe's clock rejects a CoreML slower than the CPU and `attempt` refuses one that crashes. - `Rung::CoreMl`, a compiling rung like TensorRT: an ML Program with every compute unit allowed, falling back to the CPU until each model's program is built. The embedder stays on the CPU, as on the Hexagon (§7). - The cache is one directory per model and runtime version. CoreML keys a model committed from memory on its input and node names, not its weights (ONNX Runtime 1.29, coreml_execution_provider.cc), so two exports of one architecture would otherwise share a program. - The fingerprint on macOS is the chip and the OS release, which ships CoreML. - The desktop looks for the runtime in the bundle's Contents/Frameworks and Homebrew's prefixes; fetch-desktop-runtime.sh on a Mac downloads ONNX Runtime 1.29.0 for Apple silicon, which carries CoreML. docs/dev/macos.md says what exists, how to build it, and which log lines to ask a Mac user for.
This commit is contained in:
@@ -124,5 +124,21 @@ fn runtime_dirs() -> Vec<PathBuf> {
|
|||||||
PathBuf::from("/usr/lib/darkroom"),
|
PathBuf::from("/usr/lib/darkroom"),
|
||||||
PathBuf::from("/usr/lib"),
|
PathBuf::from("/usr/lib"),
|
||||||
]);
|
]);
|
||||||
|
// An app bundle keeps its libraries in `Contents/Frameworks`, beside
|
||||||
|
// the `Contents/MacOS` the executable is in; then Homebrew's
|
||||||
|
// `onnxruntime`, Apple silicon's prefix before Intel's. Homebrew's build
|
||||||
|
// may lack CoreML, which the probe finds out for itself.
|
||||||
|
#[cfg(target_os = "macos")]
|
||||||
|
{
|
||||||
|
if let Ok(exe) = std::env::current_exe() {
|
||||||
|
if let Some(bin) = exe.parent() {
|
||||||
|
dirs.push(bin.join("../Frameworks"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
dirs.extend([
|
||||||
|
PathBuf::from("/opt/homebrew/lib"),
|
||||||
|
PathBuf::from("/usr/local/lib"),
|
||||||
|
]);
|
||||||
|
}
|
||||||
dirs
|
dirs
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -38,6 +38,11 @@ ort = { workspace = true, features = ["cuda", "tensorrt"] }
|
|||||||
[target.'cfg(target_os = "android")'.dependencies]
|
[target.'cfg(target_os = "android")'.dependencies]
|
||||||
ort = { workspace = true, features = ["qnn"] }
|
ort = { workspace = true, features = ["qnn"] }
|
||||||
|
|
||||||
|
# The Apple rung: CoreML's option builder, which fills the runtime's generic
|
||||||
|
# key/value map. `ort-sys`'s `coreml` feature is empty; nothing links.
|
||||||
|
[target.'cfg(target_os = "macos")'.dependencies]
|
||||||
|
ort = { workspace = true, features = ["coreml"] }
|
||||||
|
|
||||||
[features]
|
[features]
|
||||||
# The floor: `tract` supplies the API table when no runtime file is found, or
|
# The floor: `tract` supplies the API table when no runtime file is found, or
|
||||||
# always, in a build without `native`. Tests want this and nothing else.
|
# always, in a build without `native`. Tests want this and nothing else.
|
||||||
|
|||||||
@@ -44,6 +44,20 @@ pub fn context_path(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
|||||||
.join(format!("{:016x}_ctx.onnx", hash(bytes)))
|
.join(format!("{:016x}_ctx.onnx", hash(bytes)))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Where CoreML compiles `bytes` to: one directory per model, because
|
||||||
|
/// CoreML's own cache key leaves out the weights of a model loaded from
|
||||||
|
/// memory (`session::coreml`), and one per runtime version, which wrote it.
|
||||||
|
pub fn coreml_dir(cfg: &Config, bytes: &[u8]) -> PathBuf {
|
||||||
|
let runtime = match crate::api::runtime() {
|
||||||
|
crate::Runtime::OnnxRuntime { version, .. } => version,
|
||||||
|
crate::Runtime::Tract => "tract".into(),
|
||||||
|
};
|
||||||
|
cfg.cache_dir
|
||||||
|
.join("coreml")
|
||||||
|
.join(runtime)
|
||||||
|
.join(format!("{:016x}", hash(bytes)))
|
||||||
|
}
|
||||||
|
|
||||||
/// After the probe: compile every configured model the selected rung can
|
/// After the probe: compile every configured model the selected rung can
|
||||||
/// take, smallest first, recording each as it lands.
|
/// take, smallest first, recording each as it lands.
|
||||||
pub fn run() {
|
pub fn run() {
|
||||||
|
|||||||
@@ -78,6 +78,12 @@ pub enum Rung {
|
|||||||
MiGraphX,
|
MiGraphX,
|
||||||
/// Qualcomm's Hexagon NPU through QNN, int8 models only. Android only.
|
/// Qualcomm's Hexagon NPU through QNN, int8 models only. Android only.
|
||||||
Hexagon,
|
Hexagon,
|
||||||
|
/// Apple, through CoreML: the Neural Engine, the GPU or the CPU, as
|
||||||
|
/// CoreML schedules it. macOS only. Compiles an ML Program per model on
|
||||||
|
/// first use, so it is a compiling rung with the CPU below it. The
|
||||||
|
/// embedder stays on the CPU, as on the Hexagon: the Neural Engine
|
||||||
|
/// computes in fp16 (§7).
|
||||||
|
CoreMl,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Rung {
|
impl Rung {
|
||||||
@@ -88,6 +94,7 @@ impl Rung {
|
|||||||
Rung::TensorRt => "TensorRT",
|
Rung::TensorRt => "TensorRT",
|
||||||
Rung::MiGraphX => "MIGraphX",
|
Rung::MiGraphX => "MIGraphX",
|
||||||
Rung::Hexagon => "Hexagon NPU",
|
Rung::Hexagon => "Hexagon NPU",
|
||||||
|
Rung::CoreMl => "CoreML",
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -96,13 +103,16 @@ impl Rung {
|
|||||||
fn fallback(self) -> Rung {
|
fn fallback(self) -> Rung {
|
||||||
match self {
|
match self {
|
||||||
Rung::TensorRt => Rung::Cuda,
|
Rung::TensorRt => Rung::Cuda,
|
||||||
Rung::MiGraphX | Rung::Hexagon | Rung::Cuda | Rung::Cpu => Rung::Cpu,
|
Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl | Rung::Cuda | Rung::Cpu => Rung::Cpu,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether a session on this rung needs an engine built first.
|
/// Whether a session on this rung needs an engine built first.
|
||||||
fn compiles(self) -> bool {
|
fn compiles(self) -> bool {
|
||||||
matches!(self, Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon)
|
matches!(
|
||||||
|
self,
|
||||||
|
Rung::TensorRt | Rung::MiGraphX | Rung::Hexagon | Rung::CoreMl
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The model form this rung wants for a role.
|
/// The model form this rung wants for a role.
|
||||||
@@ -116,9 +126,11 @@ impl Rung {
|
|||||||
/// Whether this rung runs `role` at all. The Hexagon takes int8 graphs
|
/// Whether this rung runs `role` at all. The Hexagon takes int8 graphs
|
||||||
/// only, and the embedder is never int8 (§7) — it runs on the CPU
|
/// only, and the embedder is never int8 (§7) — it runs on the CPU
|
||||||
/// beside a detector on the NPU, so its vectors compare across devices.
|
/// beside a detector on the NPU, so its vectors compare across devices.
|
||||||
|
/// CoreML is kept off the embedder for the same reason: the Neural
|
||||||
|
/// Engine is fp16, and which unit runs a graph is CoreML's choice.
|
||||||
fn serves(self, role: Role) -> bool {
|
fn serves(self, role: Role) -> bool {
|
||||||
match self {
|
match self {
|
||||||
Rung::Hexagon => role != Role::Embedder,
|
Rung::Hexagon | Rung::CoreMl => role != Role::Embedder,
|
||||||
_ => true,
|
_ => true,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -637,6 +649,27 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn coreml_takes_a_compiled_detector_and_never_the_embedder() {
|
||||||
|
let hash = engines::hash(b"detector");
|
||||||
|
let mut s = State {
|
||||||
|
config: Config::default(),
|
||||||
|
cache: Cache {
|
||||||
|
rung: Some(Rung::CoreMl),
|
||||||
|
..Cache::default()
|
||||||
|
},
|
||||||
|
probing: false,
|
||||||
|
wanted: 0,
|
||||||
|
};
|
||||||
|
let on = |s: &State, role| effective_rung(s, Rung::CoreMl, role, Form::F32, hash);
|
||||||
|
// Before its program is compiled the detector waits on the CPU.
|
||||||
|
assert_eq!(on(&s, Role::Detector), Rung::Cpu);
|
||||||
|
s.cache.compiled.insert(engines::key_of(Rung::CoreMl, hash));
|
||||||
|
assert_eq!(on(&s, Role::Detector), Rung::CoreMl);
|
||||||
|
// The embedder does not move, compiled or not (§7).
|
||||||
|
assert_eq!(on(&s, Role::Embedder), Rung::Cpu);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn the_status_reports_only_the_rungs_above_the_selection() {
|
fn the_status_reports_only_the_rungs_above_the_selection() {
|
||||||
let _serial = serial();
|
let _serial = serial();
|
||||||
|
|||||||
@@ -16,10 +16,15 @@ use crate::{api::Runtime, state, Cache, Config, Form, Role, Rung};
|
|||||||
fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
|
fn ladder(ceiling: Option<Rung>) -> Vec<Rung> {
|
||||||
#[cfg(target_os = "android")]
|
#[cfg(target_os = "android")]
|
||||||
let all = [Rung::Hexagon];
|
let all = [Rung::Hexagon];
|
||||||
|
// Unmeasured (§2 ⁵): it is on the ladder because the probe's clock and
|
||||||
|
// `attempt` make a wrong guess cost one slow or failed probe, not a
|
||||||
|
// slow or crashing app.
|
||||||
|
#[cfg(target_os = "macos")]
|
||||||
|
let all = [Rung::CoreMl];
|
||||||
// A desktop has one vendor's GPU; the other vendor's providers are
|
// A desktop has one vendor's GPU; the other vendor's providers are
|
||||||
// "not enabled in this build" or a library that fails to load, and
|
// "not enabled in this build" or a library that fails to load, and
|
||||||
// either answer arrives in milliseconds.
|
// either answer arrives in milliseconds.
|
||||||
#[cfg(not(target_os = "android"))]
|
#[cfg(not(any(target_os = "android", target_os = "macos")))]
|
||||||
let all = [Rung::TensorRt, Rung::Cuda, Rung::MiGraphX];
|
let all = [Rung::TensorRt, Rung::Cuda, Rung::MiGraphX];
|
||||||
all.into_iter()
|
all.into_iter()
|
||||||
.filter(|r| ceiling.is_none_or(|c| *r <= c))
|
.filter(|r| ceiling.is_none_or(|c| *r <= c))
|
||||||
@@ -360,7 +365,50 @@ fn system_property(name: &str) -> String {
|
|||||||
String::from_utf8_lossy(&buf[..n.max(0) as usize]).into_owned()
|
String::from_utf8_lossy(&buf[..n.max(0) as usize]).into_owned()
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(any(target_os = "linux", target_os = "android")))]
|
#[cfg(target_os = "macos")]
|
||||||
|
fn device_identity() -> String {
|
||||||
|
// The chip, and the OS release: CoreML ships with the OS, so a macOS
|
||||||
|
// update is a new provider as surely as a new driver is on Linux.
|
||||||
|
format!(
|
||||||
|
"{} macOS {}",
|
||||||
|
sysctl("machdep.cpu.brand_string"),
|
||||||
|
sysctl("kern.osproductversion")
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(target_os = "macos")]
|
||||||
|
fn sysctl(name: &str) -> String {
|
||||||
|
extern "C" {
|
||||||
|
fn sysctlbyname(
|
||||||
|
name: *const std::ffi::c_char,
|
||||||
|
oldp: *mut std::ffi::c_void,
|
||||||
|
oldlenp: *mut usize,
|
||||||
|
newp: *mut std::ffi::c_void,
|
||||||
|
newlen: usize,
|
||||||
|
) -> i32;
|
||||||
|
}
|
||||||
|
let name = std::ffi::CString::new(name).unwrap();
|
||||||
|
let mut buf = [0u8; 256];
|
||||||
|
let mut len = buf.len();
|
||||||
|
// SAFETY: libSystem's documented call; `len` is the buffer's size in and
|
||||||
|
// the string's length, with its terminator, out.
|
||||||
|
let rc = unsafe {
|
||||||
|
sysctlbyname(
|
||||||
|
name.as_ptr(),
|
||||||
|
buf.as_mut_ptr().cast(),
|
||||||
|
&mut len,
|
||||||
|
std::ptr::null_mut(),
|
||||||
|
0,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
if rc != 0 {
|
||||||
|
return String::new();
|
||||||
|
}
|
||||||
|
let s = &buf[..len.min(buf.len())];
|
||||||
|
String::from_utf8_lossy(s.strip_suffix(&[0]).unwrap_or(s)).into_owned()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(not(any(target_os = "linux", target_os = "android", target_os = "macos")))]
|
||||||
fn device_identity() -> String {
|
fn device_identity() -> String {
|
||||||
String::new()
|
String::new()
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -27,13 +27,14 @@ pub fn build(rung: Rung, role: Role, bytes: &[u8], cfg: &Config) -> ort::Result<
|
|||||||
// what makes the second case rare (§6).
|
// what makes the second case rare (§6).
|
||||||
let context = (rung == Rung::Hexagon).then(|| crate::engines::context_path(cfg, bytes));
|
let context = (rung == Rung::Hexagon).then(|| crate::engines::context_path(cfg, bytes));
|
||||||
let ready = context.as_ref().is_some_and(|p| p.is_file());
|
let ready = context.as_ref().is_some_and(|p| p.is_file());
|
||||||
b = providers(
|
// What the rung keeps for this model: the context the Hexagon is to
|
||||||
b,
|
// write, or the directory CoreML compiles into.
|
||||||
rung,
|
let per_model = match rung {
|
||||||
role,
|
Rung::CoreMl => Some(crate::engines::coreml_dir(cfg, bytes)),
|
||||||
cfg,
|
_ if ready => None,
|
||||||
if ready { None } else { context.as_deref() },
|
_ => context.clone(),
|
||||||
)?;
|
};
|
||||||
|
b = providers(b, rung, role, cfg, per_model.as_deref())?;
|
||||||
match (ready, context) {
|
match (ready, context) {
|
||||||
(true, Some(path)) => b.commit_from_file(path),
|
(true, Some(path)) => b.commit_from_file(path),
|
||||||
_ => b.commit_from_memory(bytes),
|
_ => b.commit_from_memory(bytes),
|
||||||
@@ -90,11 +91,12 @@ fn providers(
|
|||||||
rung: Rung,
|
rung: Rung,
|
||||||
role: Role,
|
role: Role,
|
||||||
cfg: &Config,
|
cfg: &Config,
|
||||||
_generate_context: Option<&std::path::Path>,
|
per_model: Option<&std::path::Path>,
|
||||||
) -> ort::Result<ort::session::builder::SessionBuilder> {
|
) -> ort::Result<ort::session::builder::SessionBuilder> {
|
||||||
use ort::ep;
|
use ort::ep;
|
||||||
match rung {
|
match rung {
|
||||||
Rung::Cpu => Ok(b),
|
Rung::Cpu => Ok(b),
|
||||||
|
Rung::CoreMl => coreml(b, per_model),
|
||||||
Rung::Cuda => {
|
Rung::Cuda => {
|
||||||
Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?)
|
Ok(b.with_execution_providers([ep::CUDA::default().build().error_on_failure()])?)
|
||||||
}
|
}
|
||||||
@@ -139,6 +141,43 @@ fn providers(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// CoreML, compiling an ML Program — the format with the operators these
|
||||||
|
/// graphs use and the one that reaches the Neural Engine — into `cache`.
|
||||||
|
///
|
||||||
|
/// The option names are those ONNX Runtime 1.29 reads from the generic
|
||||||
|
/// key/value map (`coreml_options.cc`), which is what `ort`'s builder
|
||||||
|
/// fills. The cache is per model because of how CoreML keys it: a model
|
||||||
|
/// committed from memory, as every session here is, has no path, and the
|
||||||
|
/// key falls back to a hash of the graph's input and node names — not its
|
||||||
|
/// weights. Two exports of one architecture would share a program. The
|
||||||
|
/// directory `engines::coreml_dir` names is the hash of the bytes.
|
||||||
|
///
|
||||||
|
/// Every compute unit is allowed, so CoreML may place a graph on the
|
||||||
|
/// Neural Engine, the GPU or the CPU; the probe's clock judges the result.
|
||||||
|
#[cfg(target_os = "macos")]
|
||||||
|
fn coreml(
|
||||||
|
b: ort::session::builder::SessionBuilder,
|
||||||
|
cache: Option<&std::path::Path>,
|
||||||
|
) -> ort::Result<ort::session::builder::SessionBuilder> {
|
||||||
|
use ort::ep::{self, coreml};
|
||||||
|
let mut ep = ep::CoreML::default()
|
||||||
|
.with_model_format(coreml::ModelFormat::MLProgram)
|
||||||
|
.with_compute_units(coreml::ComputeUnits::All);
|
||||||
|
if let Some(dir) = cache {
|
||||||
|
let _ = std::fs::create_dir_all(dir);
|
||||||
|
ep = ep.with_model_cache_dir(dir.to_string_lossy());
|
||||||
|
}
|
||||||
|
Ok(b.with_execution_providers([ep.build().error_on_failure()])?)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(not(any(target_os = "android", target_os = "macos")))]
|
||||||
|
fn coreml(
|
||||||
|
_b: ort::session::builder::SessionBuilder,
|
||||||
|
_cache: Option<&std::path::Path>,
|
||||||
|
) -> ort::Result<ort::session::builder::SessionBuilder> {
|
||||||
|
unreachable!("the CoreML rung is on the macOS ladder only")
|
||||||
|
}
|
||||||
|
|
||||||
/// Register MIGraphX through ONNX Runtime's generic key/value entry point.
|
/// Register MIGraphX through ONNX Runtime's generic key/value entry point.
|
||||||
///
|
///
|
||||||
/// `ort`'s own builder (`ep::MIGraphX`) fills the legacy
|
/// `ort`'s own builder (`ep::MIGraphX`) fills the legacy
|
||||||
@@ -211,8 +250,8 @@ fn providers(
|
|||||||
.build()
|
.build()
|
||||||
.error_on_failure()])?)
|
.error_on_failure()])?)
|
||||||
}
|
}
|
||||||
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX => {
|
Rung::Cuda | Rung::TensorRt | Rung::MiGraphX | Rung::CoreMl => {
|
||||||
unreachable!("no desktop GPU rung on Android")
|
unreachable!("no desktop rung on Android")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -151,10 +151,14 @@ winning:
|
|||||||
| Linux / Windows, NVIDIA GPU | TensorRT, f32 model, fp16 engine | CUDA provider, f32 | ORT CPU, f32 | tract |
|
| Linux / Windows, NVIDIA GPU | TensorRT, f32 model, fp16 engine | CUDA provider, f32 | ORT CPU, f32 | tract |
|
||||||
| Linux, AMD GPU with ROCm | MIGraphX, f32 model, fp16 program | ORT CPU, f32 | — | tract |
|
| Linux, AMD GPU with ROCm | MIGraphX, f32 model, fp16 program | ORT CPU, f32 | — | tract |
|
||||||
| Linux / Windows, no GPU stack | ORT CPU, f32 | — | — | tract |
|
| Linux / Windows, no GPU stack | ORT CPU, f32 | — | — | tract |
|
||||||
| macOS ⁵ | ORT CPU, f32 | — | — | tract |
|
| macOS ⁵ | CoreML, f32 model, ML Program | ORT CPU, f32 | — | tract |
|
||||||
|
|
||||||
⁵ CoreML is the obvious rung and is unmeasured; it is listed so its absence is a gap and not an
|
⁵ **Unmeasured**, and the one exception to the rule below: nobody here has a Mac. The rung is on
|
||||||
oversight.
|
the ladder because the probe makes a wrong guess cheap — a CoreML that is slower than the CPU is
|
||||||
|
rejected by §4's clock, one that errors is recorded as failed, and one that takes the process
|
||||||
|
down is refused on the third launch (§4, `attempt`). The embedder stays on the CPU (§7). The first
|
||||||
|
macOS log that shows a probe line is this row's measurement; [macos.md](macos.md) says what to
|
||||||
|
ask for.
|
||||||
|
|
||||||
Deliberately **not** on any ladder, with the measurement that excluded each: NNAPI (no driver),
|
Deliberately **not** on any ladder, with the measurement that excluded each: NNAPI (no driver),
|
||||||
XNNPACK (slower than CPU, aborts on SCRFD), WebGPU (slower than CPU), the Adreno through QNN (works,
|
XNNPACK (slower than CPU, aborts on SCRFD), WebGPU (slower than CPU), the Adreno through QNN (works,
|
||||||
|
|||||||
@@ -0,0 +1,78 @@
|
|||||||
|
# macOS
|
||||||
|
|
||||||
|
macOS is out of scope for v1 ([requirements.md](requirements.md)), and nobody working on
|
||||||
|
DarkRoom has a Mac. This page records what exists anyway, and how a macOS build is set up so
|
||||||
|
that someone who does have one can send back enough to fix what they hit.
|
||||||
|
|
||||||
|
## 1. What exists
|
||||||
|
|
||||||
|
- **Inference** ([inference.md §2](inference.md)). The macOS ladder is CoreML, then ONNX
|
||||||
|
Runtime's CPU provider, then tract. CoreML is unmeasured. The probe decides whether it is used,
|
||||||
|
and the crash guard (§4, `attempt`) covers the case where the provider takes the process down.
|
||||||
|
The device fingerprint is the chip (`machdep.cpu.brand_string`) and the OS release, because
|
||||||
|
CoreML ships with the OS.
|
||||||
|
- **Where files go** ([`dr_plat::dirs`](../../platform/dr-plat/src/dirs.rs)). The Unix rules,
|
||||||
|
except the state directory (the log and crash records), which is `~/Library/Logs/darkroom`.
|
||||||
|
- **A diagnostic build**, described in §3.
|
||||||
|
|
||||||
|
The rest is not built, packaged or run on macOS by anyone here. This covers the window,
|
||||||
|
Metal through wgpu, the display profile (FR-DSP-8 asks X11 and Wayland), the keyring, the
|
||||||
|
bundle, and signing. `dr-plat` sends every non-Android Unix to the X11/Wayland dependencies.
|
||||||
|
|
||||||
|
## 2. Building
|
||||||
|
|
||||||
|
The Rust side compile-checks from Linux:
|
||||||
|
|
||||||
|
rustup target add aarch64-apple-darwin
|
||||||
|
cargo check --target aarch64-apple-darwin -p dr-inference-engine --features native
|
||||||
|
|
||||||
|
Linking needs Apple's SDK, which means a Mac. On one:
|
||||||
|
|
||||||
|
cargo build --profile diagnostic -p darkroom-desktop
|
||||||
|
./tools/fetch-desktop-runtime.sh # ONNX Runtime 1.29.0 with CoreML, Apple silicon only
|
||||||
|
|
||||||
|
The fetch script puts `libonnxruntime.dylib` in the user's `runtime/` directory, next to the
|
||||||
|
models. The app also looks in `Contents/Frameworks` of its own bundle, and in Homebrew's
|
||||||
|
`/opt/homebrew/lib` and `/usr/local/lib`. Homebrew's build may not include CoreML; the probe
|
||||||
|
reports that as a failed rung and uses the CPU.
|
||||||
|
|
||||||
|
**For whoever packages it.** A notarised app runs with the hardened runtime, whose library
|
||||||
|
validation refuses to `dlopen` a library signed by another team. A bundled
|
||||||
|
`Contents/Frameworks/libonnxruntime.dylib` must be signed with the app. A runtime the user
|
||||||
|
fetched needs the `com.apple.security.cs.disable-library-validation` entitlement, or it will not
|
||||||
|
load, and the app will be the tract build without saying why beyond one log line.
|
||||||
|
|
||||||
|
## 3. The diagnostic build
|
||||||
|
|
||||||
|
Every macOS build is in the hands of someone who can send a log but cannot attach a debugger,
|
||||||
|
so it is set up to log like a debug build while running at release speed.
|
||||||
|
|
||||||
|
- **The log says more.** With no `RUST_LOG`, the desktop's default filter is `debug` for every
|
||||||
|
`dr_*` crate, for `darkroom_desktop`, and for `onnxruntime`. That last one is ONNX Runtime's
|
||||||
|
own session log, which the engine forwards into `log` on every platform (`session.rs`,
|
||||||
|
`with_runtime_log`). At `debug` it includes how many nodes each provider took. At `trace`
|
||||||
|
(`RUST_LOG=onnxruntime=trace`) it lists every node's placement, which is long. The log cap is
|
||||||
|
the same as everywhere (two files of 4 MiB).
|
||||||
|
- **Backtraces have line numbers.** `--profile diagnostic` is release plus line tables. On
|
||||||
|
macOS the tables go into a `.dSYM` beside the executable, and the backtrace in a crash record
|
||||||
|
finds them only if the `.dSYM` stays next to the binary. Keep it in the bundle.
|
||||||
|
|
||||||
|
## 4. What to ask a Mac user for
|
||||||
|
|
||||||
|
`~/Library/Logs/darkroom/darkroom.log`, plus `darkroom.log.1` if present, after the first launch
|
||||||
|
and after the first scan with faces. Console.app lists it under *Log Reports*. The Settings
|
||||||
|
diagnostics bundle collects the same files. The lines that answer the open questions are:
|
||||||
|
|
||||||
|
| Line | What it tells us |
|
||||||
|
|---|---|
|
||||||
|
| `inference: ONNX Runtime … from …` / `inference: runtime tract` | Whether a runtime was found, and which one |
|
||||||
|
| `inference: floor … ms on the CPU provider` | The CPU number for §2's table |
|
||||||
|
| `inference: CoreML session built in … s` | CoreML's first compile of the probe model |
|
||||||
|
| `inference: CoreML rejected: …` / `failed: …` | Why the CPU was kept |
|
||||||
|
| `onnxruntime` lines naming `CoreMLExecutionProvider::GetCapability` | How much of the graph CoreML took |
|
||||||
|
| `inference: the app died during …` | The crash guard fired, and on what |
|
||||||
|
| `inference: compiling … for CoreML` / `ready on CoreML in … s` | Each model's compile, and any that CoreML refused |
|
||||||
|
|
||||||
|
Also ask for the settings row (*Settings › About › Inference*), which is one line and says the
|
||||||
|
same in short. When one of these logs comes back with CoreML numbers, they go into
|
||||||
|
[inference.md §1–2](inference.md), and footnote ⁵ becomes a measurement.
|
||||||
@@ -20,6 +20,32 @@
|
|||||||
# directory (docs/inference.md §1.3).
|
# directory (docs/inference.md §1.3).
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
DEST="${1:-${XDG_DATA_HOME:-${HOME}/.local/share}/darkroom/runtime}"
|
DEST="${1:-${XDG_DATA_HOME:-${HOME}/.local/share}/darkroom/runtime}"
|
||||||
|
|
||||||
|
# macOS: Microsoft's release archive, which carries the CoreML provider in the
|
||||||
|
# one library. Pinned, because the CoreML options the engine sets were read
|
||||||
|
# from this version's source (docs/dev/macos.md, CLAUDE.md "Providers").
|
||||||
|
# Apple silicon only: no Intel archive is published since 1.29; an Intel Mac
|
||||||
|
# takes Homebrew's `onnxruntime` or stays on tract.
|
||||||
|
if [[ "$(uname -s)" == Darwin ]]; then
|
||||||
|
ORT_VERSION=1.29.0
|
||||||
|
[[ "$(uname -m)" == arm64 ]] || {
|
||||||
|
echo "error: no ONNX Runtime ${ORT_VERSION} archive for $(uname -m); try: brew install onnxruntime" >&2
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
NAME="onnxruntime-osx-arm64-${ORT_VERSION}"
|
||||||
|
WORK="$(mktemp -d)"
|
||||||
|
trap 'rm -rf "${WORK}"' EXIT
|
||||||
|
echo "==> downloading ${NAME}"
|
||||||
|
curl -fsSL "https://github.com/microsoft/onnxruntime/releases/download/v${ORT_VERSION}/${NAME}.tgz" \
|
||||||
|
| tar xz -C "${WORK}"
|
||||||
|
mkdir -p "${DEST}"
|
||||||
|
cp "${WORK}/${NAME}/lib/libonnxruntime.dylib" "${WORK}/${NAME}/LICENSE" "${DEST}/"
|
||||||
|
echo "==> runtime in ${DEST}:"
|
||||||
|
ls -1 "${DEST}" | sed 's/^/ /'
|
||||||
|
echo " (the app finds it on its next launch; Settings › About › Inference says what it chose)"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
WORK="$(mktemp -d -p /var/tmp fetch-desktop-runtime.XXXXXX)"
|
WORK="$(mktemp -d -p /var/tmp fetch-desktop-runtime.XXXXXX)"
|
||||||
trap 'rm -rf "${WORK}"' EXIT
|
trap 'rm -rf "${WORK}"' EXIT
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user