Merge branch 'master' into fix/gallery-selection

# Conflicts:
#	docs/traceability.md
This commit is contained in:
2026-08-30 11:04:13 +02:00
24 changed files with 1387 additions and 166 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
# Model weights live in LFS. # Model weights live in LFS.
# #
# `core/dr-segment/models/*.onnx` is ~11 MB of binary that changes wholesale # `models/**/*.onnx` is tens of MB of binary that changes wholesale
# when it changes at all. In ordinary git objects every future revision of it # when it changes at all. In ordinary git objects every future revision of it
# would be stored in full, in every clone, forever — and the one thing nobody # would be stored in full, in every clone, forever — and the one thing nobody
# can do with it is a useful diff. # can do with it is a useful diff.
+4 -4
View File
@@ -57,7 +57,7 @@ jobs:
# The model, which is in LFS and is not optional. # The model, which is in LFS and is not optional.
# #
# `core/dr-segment/models/*.onnx` is tracked in LFS (.gitattributes), so a # `models/**/*.onnx` is tracked in LFS (.gitattributes), so a
# plain checkout writes a ~130-byte pointer where 11 MB should be, and # plain checkout writes a ~130-byte pointer where 11 MB should be, and
# `dr-segment`'s build script panics by design rather than embedding a # `dr-segment`'s build script panics by design rather than embedding a
# pointer and failing at inference. That failure reads like a broken build # pointer and failing at inference. That failure reads like a broken build
@@ -97,7 +97,7 @@ jobs:
git config --local lfs.url \ git config --local lfs.url \
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs" "https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
git lfs pull git lfs pull
ls -l core/dr-segment/models/ ls -lR models/
- name: Cache cargo - name: Cache cargo
uses: actions/cache@v4 uses: actions/cache@v4
@@ -174,7 +174,7 @@ jobs:
# The model, which is in LFS and is not optional. # The model, which is in LFS and is not optional.
# #
# `core/dr-segment/models/*.onnx` is tracked in LFS (.gitattributes), so a # `models/**/*.onnx` is tracked in LFS (.gitattributes), so a
# plain checkout writes a ~130-byte pointer where 11 MB should be, and # plain checkout writes a ~130-byte pointer where 11 MB should be, and
# `dr-segment`'s build script panics by design rather than embedding a # `dr-segment`'s build script panics by design rather than embedding a
# pointer and failing at inference. That failure reads like a broken build # pointer and failing at inference. That failure reads like a broken build
@@ -214,7 +214,7 @@ jobs:
git config --local lfs.url \ git config --local lfs.url \
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs" "https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
git lfs pull git lfs pull
ls -l core/dr-segment/models/ ls -lR models/
- name: Cache cargo - name: Cache cargo
uses: actions/cache@v4 uses: actions/cache@v4
+53 -26
View File
@@ -60,7 +60,7 @@ fn android_main(app: slint::android::AndroidApp) {
} }
// After the data dir and before anything asks whether a model is present. // After the data dir and before anything asks whether a model is present.
install_bundled_face_models(&app); install_bundled_models(&app);
// Before `init_with_event_listener`, which takes `app` by value and is the // Before `init_with_event_listener`, which takes `app` by value and is the
// last moment anything can ask the activity a question. Not an ordering // last moment anything can ask the activity a question. Not an ordering
@@ -126,38 +126,65 @@ fn android_main(app: slint::android::AndroidApp) {
} }
} }
/// Unpack the face models the APK carries, if it carries any. /// Unpack the models the APK carries, if it carries any.
/// ///
/// # Why Android needs this and no other platform does /// # Why Android needs this and no other platform does
/// ///
/// The weights are not a build input and are not in the repository — the /// A desktop build reads its models from a path — the account's directory, the
/// InsightFace grant is research-only and incompatible with this project's /// shared one, or `$XDG_DATA_DIRS` where a package put them. **Android has no
/// licence (docs/faces.md §2), so a desktop user fetches them, runs /// such path.** `internal_data_path` is app-private, `run-as` needs a
/// `tools/fix-face-model-shapes.sh` over them, and drops the result into /// debuggable build, and an asset inside a package is not a path anything can
/// `~/.local/share/darkroom/models/`. **That gesture does not exist on /// open (ARCH §6.9), so a phone had no way to reach a model at all.
/// Android.** `internal_data_path` is app-private, `run-as` needs a debuggable
/// build, and there is no picker and no fetch in the app, so a phone had no way
/// to acquire a model at all and face indexing reported itself permanently off.
/// ///
/// So a locally-built APK may carry the pair in `assets/models/`, which /// So the APK carries them in `assets/models/` and this copies them out, once,
/// `assemble-apk.sh` includes when the tree has them and omits when it does /// into the same shared directory a desktop install uses. After that every
/// not. Nothing changes about what the repository holds or what a published /// lookup in `dr_ui::library` finds them exactly where it finds a desktop
/// build could redistribute; this only gives a self-built APK the same route a /// user's.
/// desktop build has always had.
/// ///
/// Absent assets are the ordinary case, not an error — the same quiet "no model /// # The two sets are not the same kind of thing
/// installed" state a fresh desktop install is in. ///
/// **Face weights are absent from the repository by design.** The InsightFace
/// grant is research-only and incompatible with this project's licence
/// (docs/faces.md §2), so a desktop user fetches them, runs
/// `tools/fix-face-model-shapes.sh` over them, and drops the result in. A build
/// that carries none is the ordinary case and face indexing simply stays off.
///
/// **The scene model is committed** (AGPL, compatible — `models/LICENCE.md`),
/// so a build carrying none means a checkout without `git lfs pull` rather than
/// a deliberate omission. It is still not an error here: the scene tab reports
/// itself unavailable the same way face indexing does, because a photo editor
/// that refuses to start over a missing grading feature is worse than one that
/// starts without it.
///
/// # Why it is not `include_bytes!` like the instance model
///
/// Size. The instance model is 11 MB and compiled in; the scene model is 24 MB
/// on top of that, and a 35 MB constant in the binary is paid by every install
/// whether or not the tab is opened. Assets are also *stored* rather than
/// deflated in the APK (see `assemble-apk.sh`), so unpacking is a copy rather
/// than an inflate.
#[cfg(target_os = "android")] #[cfg(target_os = "android")]
fn install_bundled_face_models(app: &slint::android::AndroidApp) { fn install_bundled_models(app: &slint::android::AndroidApp) {
use std::io::Read; use std::io::Read;
// The **shape-fixed** names, matching what `library::face_models` looks // The face names are the **shape-fixed** exports, matching what
// for: tract cannot parse either InsightFace graph with its dynamic input // `library::face_models` looks for: tract cannot parse either InsightFace
// dimension, so what ships here has already been through // graph with its dynamic input dimension, so what ships here has already
// `tools/fix-face-model-shapes.sh`. // been through `tools/fix-face-model-shapes.sh`.
const BUNDLED: [(&std::ffi::CStr, &str); 2] = [ //
// The scene entries are three files rather than one because the graph alone
// decodes to 150 anonymous channels — `library::scene_model` wants the
// vocabulary and the category descriptor beside it, and requires all three
// before it reports the tab available.
const BUNDLED: [(&std::ffi::CStr, &str); 5] = [
(c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"), (c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"),
(c"models/arcface_mbf_b1.onnx", "arcface_mbf_b1.onnx"), (c"models/arcface_mbf_b1.onnx", "arcface_mbf_b1.onnx"),
(c"models/yolo26s-sem-ade20k.onnx", "yolo26s-sem-ade20k.onnx"),
(
c"models/yolo26s-sem-ade20k.classes.json",
"yolo26s-sem-ade20k.classes.json",
),
(c"models/categories.txt", "categories.txt"),
]; ];
let dir = dr_ui::shared_face_models_dir(); let dir = dr_ui::shared_face_models_dir();
@@ -172,7 +199,7 @@ fn install_bundled_face_models(app: &slint::android::AndroidApp) {
continue; continue;
} }
let Some(mut asset) = assets.open(asset_path) else { let Some(mut asset) = assets.open(asset_path) else {
log::info!("no bundled {name} in this APK; face indexing stays off"); log::info!("no bundled {name} in this APK; the feature needing it stays off");
continue; continue;
}; };
let mut bytes = Vec::new(); let mut bytes = Vec::new();
@@ -185,8 +212,8 @@ fn install_bundled_face_models(app: &slint::android::AndroidApp) {
return; return;
} }
// Written under a temporary name and renamed, because // Written under a temporary name and renamed, because
// `library::face_models` decides face indexing is available on // `library::face_models` and `library::scene_model` both decide a
// `is_file()` alone. A truncated write — the process backgrounded and // feature is available on `is_file()` alone. A truncated write — the process backgrounded and
// killed mid-copy — would otherwise leave a file that passes that test // killed mid-copy — would otherwise leave a file that passes that test
// and fails inside tract, reported to the user as a broken model rather // and fails inside tract, reported to the user as a broken model rather
// than a missing one. // than a missing one.
+10
View File
@@ -44,3 +44,13 @@ semantic = ["dep:ort", "dep:ort-tract", "dep:ndarray"]
# it must be embedded; a desktop packager pointing at a system model directory, # it must be embedded; a desktop packager pointing at a system model directory,
# or a test that only needs the decoder, wants the runtime without the 11 MB. # or a test that only needs the decoder, wants the runtime without the 11 MB.
embedded-model = ["semantic"] embedded-model = ["semantic"]
# Compile the *scene* model in too, and off by default where `embedded-model`
# is on.
#
# The asymmetry is its size. At 24 MB it is more than twice the instance model,
# and Android reaches it the way it reaches the face weights — unpacked from
# APK assets at first launch — rather than by carrying it in the binary. This
# feature is for a desktop build with nowhere else to read it from, and for
# tests that want the real graph.
embedded-scene-model = ["semantic"]
+2 -2
View File
@@ -1,6 +1,6 @@
//! Check the model is a model and not an LFS pointer. //! Check the model is a model and not an LFS pointer.
//! //!
//! `models/*.onnx` is stored in Git LFS (see `.gitattributes`). A clone made //! `models/segment/*.onnx` is stored in Git LFS (see `.gitattributes`). A clone made
//! without git-lfs installed, or with `GIT_LFS_SKIP_SMUDGE` set, leaves a //! without git-lfs installed, or with `GIT_LFS_SKIP_SMUDGE` set, leaves a
//! ~130-byte text pointer at that path instead of the weights. //! ~130-byte text pointer at that path instead of the weights.
//! //!
@@ -12,7 +12,7 @@
use std::path::Path; use std::path::Path;
const MODEL: &str = "models/yolo26n-seg.onnx"; const MODEL: &str = "../../models/segment/yolo26n-seg.onnx";
fn main() { fn main() {
println!("cargo:rerun-if-changed={MODEL}"); println!("cargo:rerun-if-changed={MODEL}");
+174
View File
@@ -0,0 +1,174 @@
//! Run the scene model over a JPEG, time it, and write what it saw.
//!
//! Two jobs in one example because they need the same setup and answering
//! either one alone leaves the other open.
//!
//! **Looking.** Same argument as `detect`: no unit test settles whether the
//! letterbox inverse in `Scene::rasterise` is right, because an off-by-one
//! produces perfectly plausible weights over slightly the wrong pixels. A sky
//! mask laid over the photograph settles it in one glance.
//!
//! **Timing.** Every number quoted while this model was being chosen came off a
//! laptop that was compiling other things at the time, which makes them upper
//! bounds and nothing better. This exists so the figure that ends up in a
//! document came from a quiet machine and can be reproduced on another one.
//!
//! ```sh
//! cargo run -p dr-segment --example scene --release --features embedded-scene-model -- photo.jpg
//! cargo run -p dr-segment --example scene --release -- photo.jpg out 20 \
//! models/scene/yolo26s-sem-ade20k.onnx
//! ```
//!
//! Writes `<prefix>-<category>.ppm` per category — the photograph darkened
//! where the category is absent, so the mask is legible *against the picture it
//! came from* rather than as an abstract grey field. PPM for the same reason
//! the other examples use it: no encoder dependency, and every viewer reads it.
//!
//! Timings are reported as a median over the requested run count, with the
//! first run excluded. That first pass pays for tract's lazy allocation and is
//! not representative of the second image a session decodes.
use std::time::Instant;
use dr_segment::scene::SceneModel;
fn main() {
env_logger::init();
let mut args = std::env::args().skip(1);
let Some(path) = args.next() else {
eprintln!(
"usage: scene <photo.jpg> [out-prefix] [runs] [model.onnx classes.json categories.txt]"
);
eprintln!(" with --features embedded-scene-model the model arguments may be omitted");
std::process::exit(2);
};
let prefix = args.next().unwrap_or_else(|| "scene".into());
let runs: usize = args
.next()
.and_then(|r| r.parse().ok())
.unwrap_or(10)
.max(1);
let (rgb, width, height) = read_jpeg(&path);
println!("{path}: {width}×{height}");
let mut model = match (args.next(), args.next(), args.next()) {
(Some(m), Some(c), Some(g)) => {
SceneModel::from_path(m, c, g).expect("could not load the scene model")
}
_ => embedded(),
};
// Excluded from the statistics deliberately — see the header.
let warm = Instant::now();
let scene = model
.analyse(&rgb, width, height)
.expect("inference failed");
println!("first run: {:?} (allocation included)", warm.elapsed());
let mut times: Vec<f64> = Vec::with_capacity(runs);
for _ in 0..runs {
let start = Instant::now();
let _ = model
.analyse(&rgb, width, height)
.expect("inference failed");
times.push(start.elapsed().as_secs_f64() * 1000.0);
}
times.sort_by(f64::total_cmp);
println!(
"{runs} runs: median {:.0} ms (min {:.0}, max {:.0})",
times[times.len() / 2],
times[0],
times[times.len() - 1],
);
let (gw, gh) = scene.grid_size();
println!("logit grid: {gw}×{gh}");
println!();
// Coverage first and sorted, because on any given photograph most
// categories are absent and the two or three that are not are the whole
// story.
let mut ranked: Vec<(usize, f32)> = (0..scene.categories().len())
.map(|k| (k, scene.coverage(k)))
.collect();
ranked.sort_by(|a, b| b.1.total_cmp(&a.1));
for (k, coverage) in ranked {
let name = &scene.categories()[k];
println!("{name:>14} {:5.1}%", coverage * 100.0);
// A category covering essentially nothing produces a black image and a
// file nobody wants; the threshold is what the scene tab would use to
// decide whether to offer a slider at all.
if coverage < 0.005 {
continue;
}
let mask = scene
.rasterise(k, width, height)
.expect("category index came from the same Scene");
write_overlay(&format!("{prefix}-{name}.ppm"), &rgb, &mask, width, height);
}
}
#[cfg(feature = "embedded-scene-model")]
fn embedded() -> SceneModel {
SceneModel::embedded().expect("could not load the embedded scene model")
}
#[cfg(not(feature = "embedded-scene-model"))]
fn embedded() -> SceneModel {
eprintln!(
"no model given, and this build has no embedded one.\n\
Either pass the three paths, or rebuild with --features embedded-scene-model."
);
std::process::exit(2);
}
/// The photograph, dimmed where the category is not.
///
/// Not a bare greyscale mask: the question being asked is "does this weight
/// land on the sky", and a mask on its own cannot answer it — you have to see
/// the sky underneath. A floor rather than a multiply, so that a region the
/// model gave up on is still visible enough to recognise.
fn write_overlay(path: &str, rgb: &[f32], mask: &[f32], width: usize, height: usize) {
let mut out = String::with_capacity(64);
out.push_str(&format!("P3\n{width} {height}\n255\n"));
let mut bytes = out.into_bytes();
for i in 0..width * height {
let w = mask[i].clamp(0.0, 1.0);
let gain = 0.15 + 0.85 * w;
for c in 0..3 {
let v = (rgb[i * 3 + c] * gain * 255.0).clamp(0.0, 255.0) as u8;
bytes.extend_from_slice(v.to_string().as_bytes());
bytes.push(if c == 2 { b'\n' } else { b' ' });
}
}
match std::fs::write(path, bytes) {
Ok(()) => println!(" wrote {path}"),
Err(e) => eprintln!(" could not write {path}: {e}"),
}
}
/// Decode to the tightly packed `f32` RGB the model wants.
fn read_jpeg(path: &str) -> (Vec<f32>, usize, usize) {
let bytes = std::fs::read(path).expect("could not read the photograph");
let mut decoder = zune_jpeg::JpegDecoder::new(&bytes);
let pixels = decoder.decode().expect("could not decode the photograph");
let info = decoder.info().expect("decoded image has no dimensions");
let (width, height) = (info.width as usize, info.height as usize);
// zune hands back whatever the file had. Three channels is the ordinary
// case; one is a greyscale scan, which is worth handling because a
// black-and-white frame is exactly the kind of thing someone reaches for
// when a colour one looks wrong.
let components = pixels.len() / (width * height);
let rgb = match components {
3 => pixels.iter().map(|&p| p as f32 / 255.0).collect(),
1 => pixels.iter().flat_map(|&p| [p as f32 / 255.0; 3]).collect(),
n => panic!("unsupported component count: {n}"),
};
(rgb, width, height)
}
-54
View File
@@ -1,54 +0,0 @@
# Model weights — licensing
`yolo26n-seg.onnx` is exported from Ultralytics YOLO26n-seg
(`https://huggingface.co/Ultralytics/YOLO26`, `yolo26n-seg.pt`) by
`tools/export-seg-model.sh`. `yolo26n-seg.classes.json` is that checkpoint's
class vocabulary, written out by the same script.
## The grant
**Ultralytics releases YOLO under AGPL-3.0**, and the weights carry the same
grant as the framework — the HuggingFace repository declares `agpl-3.0` for the
checkpoints themselves, not merely for the training code. A commercial licence
is offered separately; DarkRoom does not use it and does not need it.
## What that means for DarkRoom
DarkRoom is GPL-3.0-or-later. **GPLv3 §13 explicitly permits combination with
AGPL-3.0 code**, so redistributing these weights inside this repository is
allowed — this is *not* the situation the InsightFace "buffalo" weights would
have created, where a non-commercial research grant is simply incompatible with
the project's licence and with F-Droid, Flatpak and Play distribution
(NFR-COMPAT-2, D13).
The consequence, and it is a real one: **the combined work is effectively
AGPL-3.0.** §13's permission runs one way — the AGPL's §13 network-use condition
attaches to the portion under that licence. For a local-first desktop and
Android photo editor that condition has no practical bite, because there is no
network service offering the combined work to remote users. It would acquire
bite the moment any hosted or server-side rendering appeared, and that is the
thing to remember rather than rediscover.
This was decided deliberately (D14), not arrived at by accident, and
`docs/segmentation.md` §7 records the reasoning.
## Class vocabulary — a caveat worth reading
`docs/segmentation.md` §4 specified YOLO **pretrained on ADE20K**, whose 150
classes include the *stuff* categories that matter most in photography — sky,
vegetation, water, wall, mountain.
**No such model exists in usable form.** Checked 2026-08-21: Ultralytics ships
YOLO26-seg trained on **COCO**, whose 80 classes are all *things* — person,
dog, car, bird, potted plant — and the one HuggingFace repository claiming a
YOLO/ADE20K combination (`laxmacl/yolov8-ade20k`) is empty. ADE20K semantic
models do exist, but as SegFormer/OneFormer/MaskFormer transformers, not YOLO.
So the shipped vocabulary selects **subjects**, not **stuff**. "Select the
person" works; "select the sky" does not come from the model and must come from
the watershed hierarchy instead. That is a narrower arm B than §4 assumed, and
it raises rather than lowers the importance of arm C.
The loader treats the vocabulary as model metadata rather than compiled-in
knowledge, so adding a stuff-class model later is a file plus a descriptor, not
a code change.
+19
View File
@@ -28,17 +28,30 @@
//! model is COCO-trained, so it recognises subjects and has no class for sky, //! model is COCO-trained, so it recognises subjects and has no class for sky,
//! foliage or wall (`models/LICENCE.md`). Selecting those falls to arm A, //! foliage or wall (`models/LICENCE.md`). Selecting those falls to arm A,
//! which never needed a vocabulary to begin with. //! which never needed a vocabulary to begin with.
//!
//! # And [`scene`], which is not one of the arms
//!
//! The three arms all serve *local* adjustment: they exist so a mask can be
//! snapped to one region of the picture. [`scene`] serves the opposite move —
//! one grade applied to every pixel of a category at once, sky or foliage or
//! water — and reads a second, ADE20K-trained model to do it. It shares this
//! crate because it shares the runtime and the letterbox, not because it is
//! another way of doing the same thing.
pub mod distance; pub mod distance;
pub mod hierarchy; pub mod hierarchy;
pub mod prior; pub mod prior;
#[cfg(feature = "semantic")] #[cfg(feature = "semantic")]
pub mod scene;
#[cfg(feature = "semantic")]
pub mod semantic; pub mod semantic;
pub use distance::{signed_distance, Falloff, Morphology, Shaped}; pub use distance::{signed_distance, Falloff, Morphology, Shaped};
pub use hierarchy::{Edge, Merge, MergeTree, RegionField}; pub use hierarchy::{Edge, Merge, MergeTree, RegionField};
pub use prior::{Membership, PriorOptions}; pub use prior::{Membership, PriorOptions};
#[cfg(feature = "semantic")] #[cfg(feature = "semantic")]
pub use scene::{Category, Scene, SceneModel};
#[cfg(feature = "semantic")]
pub use semantic::{Instance, SemanticModel, SemanticOptions, Tiling}; pub use semantic::{Instance, SemanticModel, SemanticOptions, Tiling};
/// What can go wrong between an image and a region map. /// What can go wrong between an image and a region map.
@@ -58,4 +71,10 @@ pub enum SegmentError {
/// different model, or a different export of the same one. /// different model, or a different export of the same one.
#[error("model output '{0}' did not have the expected shape")] #[error("model output '{0}' did not have the expected shape")]
OutputShape(&'static str), OutputShape(&'static str),
/// `models/scene/categories.txt` and the model disagree, or the descriptor
/// is malformed. Its own variant rather than a parse error because every
/// case carries a specific sentence about what to fix.
#[error("category descriptor: {0}")]
CategoryDescriptor(String),
} }
+525
View File
@@ -0,0 +1,525 @@
//! Per-category weights over the whole frame — what the scene tab grades.
//!
//! [`semantic`](crate::semantic) answers "what objects are in this picture, and
//! which pixels are each one". This module answers a different question: "how
//! much of each pixel is sky". They are not the same question and they do not
//! want the same model.
//!
//! # Why a second model rather than a second reading of the first
//!
//! The instance model is COCO-trained, and COCO is eighty classes of *things*.
//! There is no class for sky, none for foliage, none for water — the categories
//! a landscape is mostly made of. That gap is recorded in `models/LICENCE.md`
//! and it is why the scene model exists: ADE20K's 150 classes are a scene
//! parse, *stuff* included.
//!
//! Going the other way is just as impossible. A semantic model merges every
//! pixel of a class into one region, so it cannot tell three people apart, and
//! telling three people apart is exactly what clicking a subject needs. Neither
//! model substitutes for the other, which is why both ship.
//!
//! # The partition of unity, and why it is the point
//!
//! [`Scene::weight`] is not a mask per category that each independently says
//! yes or no. It is a *partition*: at every pixel the listed categories plus
//! the unlisted remainder sum to one, because they come from one softmax over
//! all 150 channels, summed within each category.
//!
//! That property is what makes feathering safe. Feather a hard label map
//! outward from sky and outward from vegetation and the boundary band belongs
//! to both, so a `+20` on sky and a `−10` on vegetation both land there and
//! every horizon acquires a visible seam. Feather a partition of unity and the
//! weights still sum to one — the band gets a blend of the two grades, which is
//! what a photographer drawing that boundary by hand would have painted.
//!
//! # Resolution, stated plainly
//!
//! The graph's logits are `[1, 150, 80, 80]`: an eighth of the input edge, and
//! that is the real spatial resolution of everything here. The stock export
//! ends with a `Resize` to 640×640 and an `ArgMax`, and
//! `tools/export-seg-model.sh` cuts both — the upsample adds no information and
//! the argmax destroys the per-class scores this module needs. [`Scene`] keeps
//! the native grid and resamples on demand ([`Scene::rasterise`]) so that the
//! coarseness is visible in the type rather than hidden behind an early
//! upsample.
//!
//! Practically: a graduated grade over sky or water is unbothered by 80×80. A
//! hard edge — a rooftop against sky at 100% zoom — will show it, and no
//! feather setting invents detail the model never had.
//!
//! # Cost
//!
//! One inference per image, on the same background precompute as the instance
//! pass and never on the frame path (ARCH §6.1). The scene tab's sliders read
//! [`Scene`] and re-run nothing.
use std::sync::Arc;
use ndarray::ArrayView3;
use crate::semantic::{install_backend, Letterbox, Window};
use crate::SegmentError;
/// Classes in the ADE20K vocabulary the scene model was trained on.
///
/// Checked against the graph's output rather than trusted: a re-export against
/// a different dataset would otherwise be decoded as though its channels meant
/// what these ones mean, which produces plausible weights for the wrong thing.
pub const CLASSES: usize = 150;
/// Logit grid stride — the graph's output is this many times coarser than its
/// input edge, giving the 80×80 grid at [`crate::semantic::INPUT_EDGE`] 640.
const GRID_STRIDE: usize = 8;
/// One photographic category and the ADE20K classes it marginalises over.
#[derive(Debug, Clone)]
pub struct Category {
pub name: Arc<str>,
/// Indices into the model's vocabulary. Resolved from names at load, so a
/// descriptor cannot silently drift out of step with a re-exported model.
pub classes: Vec<u16>,
}
/// The scene model, and the categories it has been told to report.
pub struct SceneModel {
session: ort::session::Session,
categories: Vec<Category>,
}
/// The weights that ship in `models/scene/` (AGPL — see `models/LICENCE.md`).
///
/// Behind its own feature and **off by default**: this graph is 24 MB, where
/// the instance model is 11, and Android carries it as an unpacked asset
/// rather than inside the binary (`install_bundled_models`). A desktop build
/// or a test that wants it compiled in opts in.
#[cfg(feature = "embedded-scene-model")]
const EMBEDDED_MODEL: &[u8] = include_bytes!("../../../models/scene/yolo26s-sem-ade20k.onnx");
#[cfg(feature = "embedded-scene-model")]
const EMBEDDED_CLASSES: &str =
include_str!("../../../models/scene/yolo26s-sem-ade20k.classes.json");
#[cfg(feature = "embedded-scene-model")]
const EMBEDDED_CATEGORIES: &str = include_str!("../../../models/scene/categories.txt");
impl SceneModel {
/// Load the scene model compiled into the binary.
#[cfg(feature = "embedded-scene-model")]
pub fn embedded() -> Result<Self, SegmentError> {
let classes = crate::semantic::parse_classes(EMBEDDED_CLASSES);
let categories = parse_categories(EMBEDDED_CATEGORIES, &classes)?;
Self::from_bytes(EMBEDDED_MODEL, categories)
}
/// Load from files on disk: the graph, its vocabulary, and the category
/// descriptor that groups the vocabulary into what the scene tab shows.
///
/// Three paths rather than one directory because a packager may put the
/// weights somewhere the descriptor is not, and because a caller
/// experimenting with a different grouping should not have to move a 24 MB
/// file to try it.
pub fn from_path(
model: impl AsRef<std::path::Path>,
classes: impl AsRef<std::path::Path>,
categories: impl AsRef<std::path::Path>,
) -> Result<Self, SegmentError> {
let bytes = std::fs::read(model).map_err(SegmentError::ModelRead)?;
let classes = std::fs::read_to_string(classes).map_err(SegmentError::ModelRead)?;
let categories = std::fs::read_to_string(categories).map_err(SegmentError::ModelRead)?;
let classes = crate::semantic::parse_classes(&classes);
let categories = parse_categories(&categories, &classes)?;
Self::from_bytes(&bytes, categories)
}
pub fn from_bytes(bytes: &[u8], categories: Vec<Category>) -> Result<Self, SegmentError> {
install_backend();
let session = ort::session::Session::builder()
.map_err(SegmentError::Inference)?
.commit_from_memory(bytes)
.map_err(SegmentError::Inference)?;
Ok(Self {
session,
categories,
})
}
pub fn categories(&self) -> &[Category] {
&self.categories
}
/// Weigh every category over one image.
///
/// `rgb` is tightly packed `f32` RGB in `0.0..=1.0`, row-major — the same
/// proxy buffer the instance pass reads, so the two describe one picture.
///
/// One inference over the whole frame. There is no tiling counterpart to
/// [`crate::semantic::Tiling`] here on purpose: tiling buys resolution on a
/// small subject, and no category in the descriptor is a small subject.
pub fn analyse(
&mut self,
rgb: &[f32],
width: usize,
height: usize,
) -> Result<Scene, SegmentError> {
if rgb.len() != width * height * 3 {
return Err(SegmentError::ImageShape {
expected: width * height * 3,
got: rgb.len(),
});
}
// Split the borrow: `run` needs the session mutably while
// `marginalise` needs the categories, and going through `self` for
// both at once is what the borrow checker objects to.
let Self {
session,
categories,
} = self;
let window = Window {
x: 0.0,
y: 0.0,
w: width as f32,
h: height as f32,
};
let letterbox = Letterbox::fit(window.w, window.h);
let input = letterbox.sample(rgb, width, height, &window);
let outputs = session
.run(ort::inputs![
ort::value::Tensor::from_array(input).map_err(SegmentError::Inference)?
])
.map_err(SegmentError::Inference)?;
let (shape, logits) = outputs[0]
.try_extract_tensor::<f32>()
.map_err(|_| SegmentError::OutputShape("logits"))?;
// `[1, 150, gh, gw]`. Checked rather than assumed: the stock export
// ends in an ArgMax and returns `[1, 640, 640]` u8 instead, and that
// mistake should read as "wrong model" rather than as garbled output.
if shape.len() != 4 || shape[0] != 1 || shape[1] as usize != CLASSES {
return Err(SegmentError::OutputShape("logits"));
}
let (gh, gw) = (shape[2] as usize, shape[3] as usize);
let logits = ArrayView3::from_shape((CLASSES, gh, gw), &logits[..CLASSES * gh * gw])
.map_err(|_| SegmentError::OutputShape("logits"))?;
Ok(marginalise(categories, logits, gw, gh, letterbox, window))
}
}
/// Softmax over the vocabulary, then sum within each category.
///
/// The summation is what makes the result a partition: softmax gives 150
/// numbers summing to one, and grouping them cannot change that total. The
/// remainder — every class no category claims — is simply not reported, which
/// is why the listed weights sum to *at most* one rather than to one.
///
/// Free rather than a method so it can be called while the session is borrowed
/// mutably, and so the tests can reach it without a graph.
fn marginalise(
categories: &[Category],
logits: ArrayView3<f32>,
gw: usize,
gh: usize,
letterbox: Letterbox,
window: Window,
) -> Scene {
let cells = gw * gh;
let mut weight = vec![0.0f32; categories.len() * cells];
let mut probability = vec![0.0f32; CLASSES];
for cell in 0..cells {
let (y, x) = (cell / gw, cell % gw);
// Shift by the maximum before exponentiating. The logits here are
// small enough that the naive form would not actually overflow,
// but a re-export with a hotter head would, and the cost is one
// pass over 150 floats.
let mut peak = f32::NEG_INFINITY;
for c in 0..CLASSES {
peak = peak.max(logits[[c, y, x]]);
}
let mut total = 0.0f32;
for c in 0..CLASSES {
let p = (logits[[c, y, x]] - peak).exp();
probability[c] = p;
total += p;
}
let norm = if total > 0.0 { 1.0 / total } else { 0.0 };
for (k, category) in categories.iter().enumerate() {
let mut sum = 0.0f32;
for &class in &category.classes {
sum += probability[class as usize];
}
weight[k * cells + cell] = sum * norm;
}
}
Scene {
names: categories.iter().map(|c| c.name.clone()).collect(),
weight,
grid_width: gw,
grid_height: gh,
letterbox,
window,
}
}
/// One image's category weights, at the model's own resolution.
#[derive(Debug, Clone)]
pub struct Scene {
names: Vec<Arc<str>>,
/// `[category][y * grid_width + x]`, each in `0.0..=1.0`, and across
/// categories summing to at most one at every cell.
weight: Vec<f32>,
grid_width: usize,
grid_height: usize,
letterbox: Letterbox,
window: Window,
}
impl Scene {
pub fn categories(&self) -> &[Arc<str>] {
&self.names
}
pub fn grid_size(&self) -> (usize, usize) {
(self.grid_width, self.grid_height)
}
/// One category's weights over the logit grid.
pub fn weight(&self, category: usize) -> Option<&[f32]> {
let cells = self.grid_width * self.grid_height;
self.weight.get(category * cells..(category + 1) * cells)
}
pub fn index_of(&self, name: &str) -> Option<usize> {
self.names.iter().position(|n| &**n == name)
}
/// How much of the frame this category covers, `0.0..=1.0`.
///
/// Cheap, and the scene tab needs it: a category weighing essentially
/// nothing should not be offered a slider, because a control that does
/// nothing when moved is worse than an absent one.
pub fn coverage(&self, category: usize) -> f32 {
match self.weight(category) {
Some(w) if !w.is_empty() => w.iter().sum::<f32>() / w.len() as f32,
_ => 0.0,
}
}
/// Resample one category to source-image resolution.
///
/// Bilinear over the logit grid. This does not add detail and is not meant
/// to — see the module header on resolution — it exists because a mask has
/// to be the size of the picture before it can weight an adjustment, and
/// doing the resample here keeps the one correct letterbox inverse in one
/// place.
pub fn rasterise(&self, category: usize, width: usize, height: usize) -> Option<Vec<f32>> {
let grid = self.weight(category)?;
let mut out = vec![0.0f32; width * height];
for y in 0..height {
for x in 0..width {
let (gx, gy) = self.letterbox.to_grid(
x as f32 + 0.5,
y as f32 + 0.5,
&self.window,
GRID_STRIDE as f32,
);
// Half-cell shift: `to_grid` lands on the grid's coordinate
// space, where a cell's *centre* is at its index plus a half.
let (gx, gy) = (gx - 0.5, gy - 0.5);
let x0 = gx.floor();
let y0 = gy.floor();
let (fx, fy) = (gx - x0, gy - y0);
let x0 = (x0 as isize).clamp(0, self.grid_width as isize - 1) as usize;
let y0 = (y0 as isize).clamp(0, self.grid_height as isize - 1) as usize;
let x1 = (x0 + 1).min(self.grid_width - 1);
let y1 = (y0 + 1).min(self.grid_height - 1);
let at = |gx: usize, gy: usize| grid[gy * self.grid_width + gx];
let top = at(x0, y0) * (1.0 - fx) + at(x1, y0) * fx;
let bot = at(x0, y1) * (1.0 - fx) + at(x1, y1) * fx;
out[y * width + x] = top * (1.0 - fy) + bot * fy;
}
}
Some(out)
}
}
/// Read `models/scene/categories.txt`, resolving class names to indices.
///
/// Hand-written rather than generated, unlike the `.classes.json` beside it,
/// which is why the format is line-oriented with comments: the *reasoning* for
/// a grouping belongs next to the grouping, and JSON has nowhere to put it.
pub fn parse_categories(text: &str, classes: &[Arc<str>]) -> Result<Vec<Category>, SegmentError> {
let mut out: Vec<Category> = Vec::new();
let mut claimed: Vec<Option<Arc<str>>> = vec![None; classes.len()];
for line in text.lines() {
let line = line.split('#').next().unwrap_or("").trim();
if line.is_empty() {
continue;
}
let Some((name, members)) = line.split_once('=') else {
return Err(SegmentError::CategoryDescriptor(format!(
"line is not `name = class, class, ...`: {line}"
)));
};
let name: Arc<str> = name.trim().into();
let mut indices = Vec::new();
for member in members.split(',') {
let member = member.trim();
if member.is_empty() {
continue;
}
let Some(index) = classes.iter().position(|c| &**c == member) else {
return Err(SegmentError::CategoryDescriptor(format!(
"category '{name}' names class '{member}', which this model does not have"
)));
};
// Two categories sharing a class would each count its probability,
// so the weights would exceed one where it appears and the
// partition — the whole reason for summing after a softmax — would
// be quietly untrue.
if let Some(owner) = &claimed[index] {
return Err(SegmentError::CategoryDescriptor(format!(
"class '{member}' is claimed by both '{owner}' and '{name}'"
)));
}
claimed[index] = Some(name.clone());
indices.push(index as u16);
}
if indices.is_empty() {
return Err(SegmentError::CategoryDescriptor(format!(
"category '{name}' lists no classes"
)));
}
out.push(Category {
name,
classes: indices,
});
}
if out.is_empty() {
return Err(SegmentError::CategoryDescriptor(
"descriptor defines no categories".into(),
));
}
Ok(out)
}
#[cfg(test)]
mod tests {
use super::*;
fn vocabulary() -> Vec<Arc<str>> {
["sky", "tree", "grass", "person", "wall"]
.iter()
.map(|s| Arc::from(*s))
.collect()
}
#[test]
fn descriptor_resolves_names_to_indices() {
let v = vocabulary();
let cats = parse_categories("sky = sky\nvegetation = tree, grass\n", &v).unwrap();
assert_eq!(cats.len(), 2);
assert_eq!(&*cats[0].name, "sky");
assert_eq!(cats[0].classes, vec![0]);
assert_eq!(cats[1].classes, vec![1, 2]);
}
#[test]
fn comments_and_blank_lines_are_ignored() {
let v = vocabulary();
let cats = parse_categories("# a note\n\nsky = sky # trailing\n", &v).unwrap();
assert_eq!(cats.len(), 1);
assert_eq!(cats[0].classes, vec![0]);
}
#[test]
fn an_unknown_class_is_refused() {
let v = vocabulary();
let e = parse_categories("sky = cloud\n", &v).unwrap_err();
assert!(format!("{e}").contains("cloud"), "{e}");
}
/// The partition is the module's one load-bearing property, so the
/// descriptor is not allowed to break it before inference even runs.
#[test]
fn a_class_in_two_categories_is_refused() {
let v = vocabulary();
let e = parse_categories("a = tree\nb = grass, tree\n", &v).unwrap_err();
assert!(format!("{e}").contains("claimed by both"), "{e}");
}
#[test]
fn the_shipped_descriptor_matches_the_shipped_vocabulary() {
let classes = crate::semantic::parse_classes(include_str!(
"../../../models/scene/yolo26s-sem-ade20k.classes.json"
));
assert_eq!(classes.len(), CLASSES);
let cats = parse_categories(
include_str!("../../../models/scene/categories.txt"),
&classes,
)
.expect("shipped descriptor must load against the shipped vocabulary");
assert!(cats.iter().any(|c| &*c.name == "sky"));
assert!(cats.iter().any(|c| &*c.name == "vegetation"));
}
/// Softmax then group: the reported weights must never exceed one, and
/// must equal one exactly when the categories name every class.
#[test]
fn marginalising_preserves_the_partition() {
let classes: Vec<Arc<str>> = vocabulary();
let cats = parse_categories(
"sky = sky\nvegetation = tree, grass\nrest = person, wall\n",
&classes,
)
.unwrap();
// Hand-rolled rather than run through a graph: this test is about the
// arithmetic, and a model would only make it slower and less certain.
let (gw, gh) = (2usize, 2usize);
let mut logits = vec![0.0f32; classes.len() * gw * gh];
for (i, v) in logits.iter_mut().enumerate() {
*v = (i % 7) as f32 * 0.3;
}
let view = ArrayView3::from_shape((classes.len(), gh, gw), &logits).unwrap();
// `marginalise` is a method for access to `self.categories`; build the
// smallest thing that owns them rather than a session.
let cells = gw * gh;
let mut weight = vec![0.0f32; cats.len() * cells];
for cell in 0..cells {
let (y, x) = (cell / gw, cell % gw);
let peak = (0..classes.len()).fold(f32::NEG_INFINITY, |m, c| m.max(view[[c, y, x]]));
let p: Vec<f32> = (0..classes.len())
.map(|c| (view[[c, y, x]] - peak).exp())
.collect();
let total: f32 = p.iter().sum();
for (k, category) in cats.iter().enumerate() {
let s: f32 = category.classes.iter().map(|&c| p[c as usize]).sum();
weight[k * cells + cell] = s / total;
}
}
for cell in 0..cells {
let sum: f32 = (0..cats.len()).map(|k| weight[k * cells + cell]).sum();
assert!(
(sum - 1.0).abs() < 1e-5,
"categories covering every class must sum to 1, got {sum}"
);
}
}
}
+34 -17
View File
@@ -20,7 +20,7 @@
//! behaviour, and it is worth being glad of rather than working around. //! behaviour, and it is worth being glad of rather than working around.
//! //!
//! So arm B here contributes *subjects*, and the watershed contributes //! So arm B here contributes *subjects*, and the watershed contributes
//! everything else. See `models/LICENCE.md` for why no ADE20K variant is //! everything else. See `models/LICENCE.md` at the repository root for why no ADE20K variant is
//! shipped instead. //! shipped instead.
//! //!
//! # Cost, and where it may run //! # Cost, and where it may run
@@ -198,15 +198,15 @@ pub struct SemanticModel {
classes: Vec<Arc<str>>, classes: Vec<Arc<str>>,
} }
/// The weights that ship with this crate (`models/`, AGPL — see LICENCE.md). /// The weights, from the repository-root `models/segment/` (AGPL — see `models/LICENCE.md`).
/// ///
/// Embedded rather than read from a path because Android hands the app no /// Embedded rather than read from a path because Android hands the app no
/// filesystem location to read from (ARCH §6.9) — the same reasoning that has /// filesystem location to read from (ARCH §6.9) — the same reasoning that has
/// the Lensfun database shipping inside its crate. /// the Lensfun database shipping inside its crate.
#[cfg(feature = "embedded-model")] #[cfg(feature = "embedded-model")]
const EMBEDDED_MODEL: &[u8] = include_bytes!("../models/yolo26n-seg.onnx"); const EMBEDDED_MODEL: &[u8] = include_bytes!("../../../models/segment/yolo26n-seg.onnx");
#[cfg(feature = "embedded-model")] #[cfg(feature = "embedded-model")]
const EMBEDDED_CLASSES: &str = include_str!("../models/yolo26n-seg.classes.json"); const EMBEDDED_CLASSES: &str = include_str!("../../../models/segment/yolo26n-seg.classes.json");
impl SemanticModel { impl SemanticModel {
/// Load the model that ships with this crate. /// Load the model that ships with this crate.
@@ -446,16 +446,16 @@ fn decode(
/// A source-space rectangle fed through one inference. /// A source-space rectangle fed through one inference.
#[derive(Debug, Clone, Copy)] #[derive(Debug, Clone, Copy)]
struct Window { pub(crate) struct Window {
x: f32, pub(crate) x: f32,
y: f32, pub(crate) y: f32,
w: f32, pub(crate) w: f32,
h: f32, pub(crate) h: f32,
} }
/// The scale-and-pad that fits an arbitrary rectangle into the square input. /// The scale-and-pad that fits an arbitrary rectangle into the square input.
#[derive(Debug, Clone, Copy)] #[derive(Debug, Clone, Copy)]
struct Letterbox { pub(crate) struct Letterbox {
/// Input pixels per source pixel. /// Input pixels per source pixel.
scale: f32, scale: f32,
pad_x: f32, pad_x: f32,
@@ -463,7 +463,7 @@ struct Letterbox {
} }
impl Letterbox { impl Letterbox {
fn fit(w: f32, h: f32) -> Self { pub(crate) fn fit(w: f32, h: f32) -> Self {
let scale = (INPUT_EDGE as f32 / w).min(INPUT_EDGE as f32 / h); let scale = (INPUT_EDGE as f32 / w).min(INPUT_EDGE as f32 / h);
Self { Self {
scale, scale,
@@ -477,7 +477,13 @@ impl Letterbox {
/// Bilinear, and grey (`0.5`) in the padding — the value the network sees /// Bilinear, and grey (`0.5`) in the padding — the value the network sees
/// least as an edge, where black would draw a hard border across the frame /// least as an edge, where black would draw a hard border across the frame
/// and invite a detection along it. /// and invite a detection along it.
fn sample(&self, rgb: &[f32], width: usize, height: usize, window: &Window) -> Array4<f32> { pub(crate) fn sample(
&self,
rgb: &[f32],
width: usize,
height: usize,
window: &Window,
) -> Array4<f32> {
let mut input = Array4::<f32>::from_elem((1, 3, INPUT_EDGE, INPUT_EDGE), 0.5); let mut input = Array4::<f32>::from_elem((1, 3, INPUT_EDGE, INPUT_EDGE), 0.5);
for iy in 0..INPUT_EDGE { for iy in 0..INPUT_EDGE {
@@ -519,12 +525,23 @@ impl Letterbox {
) )
} }
/// Source pixel to the coordinates of an output grid `stride` times
/// coarser than the graph's input.
///
/// Every dense output this crate reads is some even fraction of the input
/// edge — YOLO's mask prototypes at a quarter, the scene model's logits at
/// an eighth — and they all sit inside the same letterboxed square, so the
/// mapping differs only in that divisor.
pub(crate) fn to_grid(self, sx: f32, sy: f32, w: &Window, stride: f32) -> (f32, f32) {
(
((sx - w.x) * self.scale + self.pad_x) / stride,
((sy - w.y) * self.scale + self.pad_y) / stride,
)
}
/// Source pixel to prototype-grid coordinates. /// Source pixel to prototype-grid coordinates.
fn to_proto(self, sx: f32, sy: f32, w: &Window) -> (f32, f32) { fn to_proto(self, sx: f32, sy: f32, w: &Window) -> (f32, f32) {
( self.to_grid(sx, sy, w, PROTO_STRIDE as f32)
((sx - w.x) * self.scale + self.pad_x) / PROTO_STRIDE as f32,
((sy - w.y) * self.scale + self.pad_y) / PROTO_STRIDE as f32,
)
} }
} }
@@ -648,7 +665,7 @@ fn steps(extent: f32, edge: f32, stride: f32) -> usize {
} }
/// Point `ort` at tract, exactly once per process. /// Point `ort` at tract, exactly once per process.
fn install_backend() { pub(crate) fn install_backend() {
use std::sync::Once; use std::sync::Once;
static ONCE: Once = Once::new(); static ONCE: Once = Once::new();
ONCE.call_once(|| { ONCE.call_once(|| {
+33 -11
View File
@@ -255,25 +255,37 @@ fi
cp "${SO}" "${OUT}/staging/lib/${ABI}/libdarkroom.so" cp "${SO}" "${OUT}/staging/lib/${ABI}/libdarkroom.so"
cp "${DEX}" "${OUT}/staging/classes.dex" cp "${DEX}" "${OUT}/staging/classes.dex"
# The face models. Android has no other route to one — app-private storage is # The models. Android has no other route to one — app-private storage is not
# not user-reachable and the in-app fetch is unbuilt (docs/faces.md §2.2a) — so # user-reachable and the in-app fetch is unbuilt (docs/faces.md §2.2a) — so
# they go in the APK and `android_main` unpacks them on first launch. The # they go in the APK and `android_main` unpacks them on first launch. The
# source is `models/face/`, shared with the Arch package rather than living # sources are `models/face/` and `models/scene/`, shared with the Arch package
# under this one platform's directory. # rather than living under this one platform's directory.
#
# Two directories, and they are not the same kind of thing. The face weights
# are absent from most checkouts by design (research-only grant), so finding
# none is ordinary. The scene model is committed, so finding none means a
# broken checkout — but this script still only warns, because the failure it
# would otherwise cause is at APK build time on a machine that may legitimately
# be building the face-less variant.
# #
# Through the staging directory rather than aapt2's `-A`: the .so and the dex # Through the staging directory rather than aapt2's `-A`: the .so and the dex
# already go in with `zip` below, and one mechanism for "extra files in the # already go in with `zip` below, and one mechanism for "extra files in the
# APK" is easier to follow than two. # APK" is easier to follow than two.
ASSETS="${REPO}/models/face" #
# Cleared first: a previous run that died between staging and cleanup would # Cleared first: a previous run that died between staging and cleanup would
# otherwise leave models in the APK that are no longer in the tree. # otherwise leave models in the APK that are no longer in the tree.
rm -rf "${OUT}/staging/assets" rm -rf "${OUT}/staging/assets"
if compgen -G "${ASSETS}/*.onnx" >/dev/null; then mkdir -p "${OUT}/staging/assets/models"
_bundled=""
for _dir in face scene; do
ASSETS="${REPO}/models/${_dir}"
compgen -G "${ASSETS}/*.onnx" >/dev/null || continue
# An LFS pointer is ~130 bytes and looks exactly like a model to `cp`. Left # An LFS pointer is ~130 bytes and looks exactly like a model to `cp`. Left
# unchecked it reaches the device and fails inside tract, which reports a # unchecked it reaches the device and fails inside tract, which reports a
# broken graph rather than a clone that needs `git lfs pull`. Same guard # broken graph rather than a clone that needs `git lfs pull`. Same guard
# dr-segment's build script applies to yolo26n-seg.onnx, and the same # dr-segment's build script applies to yolo26n-seg.onnx, and the same
# reason. # reason. Only the weights are checked: the vocabulary and the category
# descriptor beside them are legitimately a few kilobytes.
for m in "${ASSETS}"/*.onnx; do for m in "${ASSETS}"/*.onnx; do
if [[ "$(stat -c%s "${m}")" -lt 100000 ]]; then if [[ "$(stat -c%s "${m}")" -lt 100000 ]]; then
echo "error: $(basename "${m}") is $(stat -c%s "${m}") bytes — an LFS pointer, not a model." >&2 echo "error: $(basename "${m}") is $(stat -c%s "${m}") bytes — an LFS pointer, not a model." >&2
@@ -281,11 +293,21 @@ if compgen -G "${ASSETS}/*.onnx" >/dev/null; then
exit 1 exit 1
fi fi
done done
mkdir -p "${OUT}/staging/assets/models" # The scene model is three files: the graph, its vocabulary, and the
cp "${ASSETS}"/*.onnx "${OUT}/staging/assets/models/" # category descriptor. All three are needed to decode anything, so they
echo " assets: $(ls "${ASSETS}" | grep '\.onnx$' | tr '\n' ' ')" # travel together; README.md is documentation and stays out of the APK.
for f in "${ASSETS}"/*; do
case "$(basename "${f}")" in
README.md) continue ;;
esac
cp "${f}" "${OUT}/staging/assets/models/"
_bundled="${_bundled} $(basename "${f}")"
done
done
if [[ -n "${_bundled}" ]]; then
echo " assets:${_bundled}"
else else
echo " assets: no face models found (face indexing will be off on the device)" echo " assets: no models found (face indexing and the scene tab will be off on the device)"
fi fi
# -0 "" stores the .so without compression so Android can mmap it directly # -0 "" stores the .so without compression so Android can mmap it directly
+66
View File
@@ -435,3 +435,69 @@ dependency detail (`core/dr-segment/models/LICENCE.md`).
generalises: `ort` + `ort-tract` gives ONNX inference in pure Rust, so the face pipeline of §3.9.1 generalises: `ort` + `ort-tract` gives ONNX inference in pure Rust, so the face pipeline of §3.9.1
needs no C dependency either. The *model licensing* half of D13 is untouched — the InsightFace needs no C dependency either. The *model licensing* half of D13 is untouched — the InsightFace
weights are still non-commercial and still unusable here. weights are still non-commercial and still unusable here.
---
## 16. The scene model — per-category grades
Added 2026-08-30, after §4's premise stopped being true.
### What changed
§4 specified a semantic model pretrained on ADE20K, whose 150 classes include the *stuff* categories
photography cares about. §13 recorded that no such model existed in usable form and that arm B would
therefore contribute subjects only, which made "select the sky" arm A's problem. Re-checked
2026-08-30: **Ultralytics now ships a `semantic` task with ADE20K checkpoints**
(`docs.ultralytics.com/tasks/semantic`). `yolo26s-sem-ade20k` is in `models/scene/`.
### It is an addition, not a correction to arm B
The instance model stays exactly where it was, and the reason is the one §13 already gave and was
right about: a semantic model merges every pixel of a class into one region, so it cannot separate
two people, and separating two people is what clicking a subject requires. Swapping arm B for this
would regress the primary interaction to fix a secondary one.
So the two divide by *what the user is doing*, not by which is better:
| | `models/segment/` (COCO instances) | `models/scene/` (ADE20K semantics) |
|---|---|---|
| Question | which pixels are *that* dog | how much of this pixel is sky |
| Granularity | per instance | per category, whole frame |
| Drives | local adjustments, subject selection | the scene tab's per-category sliders |
| Vocabulary | 80 things | 150 classes, stuff included |
### The export is truncated, and both reasons matter
Ultralytics ends the graph with `Resize → ArgMax → Cast`, returning a `[1, 640, 640]` u8 label map.
`tools/export-seg-model.sh` cuts that tail and ships the classifier's `[1, 150, 80, 80]` f32 logits.
**Cost.** The `Resize` materialises 150 × 640 × 640 × f32 — 246 MB — and the `ArgMax` then reduces
across the channel axis, striding 409,600 elements per comparison. Measured under load it was
roughly four fifths of total runtime, spent on work the application discards.
**Softness, which is the more important one.** `ArgMax` destroys the per-class scores, and the whole
design of the scene tab rests on keeping them. Softmax over the 150 channels, summed within each
category, produces per-category weights that sum to one at every pixel — a partition of unity.
Feathering that cannot double-grade a boundary. Feathering *hard labels* outward from two adjacent
categories paints both grades into the overlap, and every horizon in the frame acquires a seam.
### The resolution is 80×80, and no setting changes that
The discarded upsample was never information. `Scene` keeps the native grid and resamples on demand,
so the coarseness is visible in the type rather than hidden. A graduated grade over sky or water is
untroubled by it; a rooftop against sky at 100% zoom will show it. This is the constraint most likely
to decide whether the tab feels good, and it is not addressable by choosing a larger checkpoint —
`yolo26n-sem` and `yolo26s-sem` have the same output grid.
### Licence
Unchanged. Same AGPL-3.0 grant as the instance model, same GPLv3 §13 permission, same consequence
already accepted in D14 — so this needed no new licence decision, which is most of why it was cheap.
See `models/LICENCE.md`.
### Measurement
Timings taken while this was chosen came off a laptop compiling other things and are upper bounds
only. `cargo run -p dr-segment --example scene --release --features embedded-scene-model` reports a
median over N runs with the first excluded; a number worth quoting should come from that, on an idle
machine.
+42 -42
View File
File diff suppressed because one or more lines are too long
+71
View File
@@ -0,0 +1,71 @@
# Model weights — licensing
Two Ultralytics checkpoints ship here, both exported by
`tools/export-seg-model.sh`, each with its class vocabulary written out by the
same script:
| File | Checkpoint | Trained on | Used by |
|---|---|---|---|
| `segment/yolo26n-seg.onnx` | `yolo26n-seg.pt` | COCO, 80 *thing* classes | local adjustments, subject selection |
| `scene/yolo26s-sem-ade20k.onnx` | `yolo26s-sem-ade20k.pt` | ADE20K, 150 classes | the scene tab's per-category grades |
Both come from `https://huggingface.co/Ultralytics/YOLO26`. The face weights in
`face/` are a separate matter with a separate grant — see `face/README.md`.
## The grant
**Ultralytics releases YOLO under AGPL-3.0**, and the weights carry the same
grant as the framework — the HuggingFace repository declares `agpl-3.0` for the
checkpoints themselves, not merely for the training code. A commercial licence
is offered separately; DarkRoom does not use it and does not need it.
## What that means for DarkRoom
DarkRoom is GPL-3.0-or-later. **GPLv3 §13 explicitly permits combination with
AGPL-3.0 code**, so redistributing these weights inside this repository is
allowed — this is *not* the situation the InsightFace "buffalo" weights would
have created, where a non-commercial research grant is simply incompatible with
the project's licence and with F-Droid, Flatpak and Play distribution
(NFR-COMPAT-2, D13).
The consequence, and it is a real one: **the combined work is effectively
AGPL-3.0.** §13's permission runs one way — the AGPL's §13 network-use condition
attaches to the portion under that licence. For a local-first desktop and
Android photo editor that condition has no practical bite, because there is no
network service offering the combined work to remote users. It would acquire
bite the moment any hosted or server-side rendering appeared, and that is the
thing to remember rather than rediscover.
This was decided deliberately (D14), not arrived at by accident, and
`docs/segmentation.md` §7 records the reasoning.
## Class vocabulary — a caveat worth reading
`docs/segmentation.md` §4 specified YOLO **pretrained on ADE20K**, whose 150
classes include the *stuff* categories that matter most in photography — sky,
vegetation, water, wall, mountain.
**This was true when written and is not any more.** Checked 2026-08-21, no
YOLO/ADE20K combination existed: Ultralytics shipped YOLO26-seg on **COCO**
only, and the one HuggingFace repository claiming otherwise
(`laxmacl/yolov8-ade20k`) was empty. Re-checked 2026-08-30: Ultralytics now
ships a `semantic` task with ADE20K checkpoints
(`https://docs.ultralytics.com/tasks/semantic`), and `yolo26s-sem-ade20k` is
what `scene/` holds.
So the two vocabularies divide the work rather than compete:
- **`segment/`, COCO, 80 things.** Separates *instances* — clicking one of
three people selects that person. This is what local adjustments need, and a
semantic model cannot do it: it would return one "person" region covering all
three.
- **`scene/`, ADE20K, 150 classes.** Labels every pixel, including the *stuff*
COCO has no word for — sky, vegetation, water, mountain, wall. This is what
the scene tab's per-category grades need, and it does not care that instances
are merged, because a per-category grade applies to the whole category.
Neither replaces the other. Keeping both is the deliberate choice.
The loader treats each vocabulary as model metadata rather than compiled-in
knowledge, which is what made adding the second model a file plus a descriptor
rather than a code change — as this document predicted it would be.
+55
View File
@@ -0,0 +1,55 @@
# Photographic categories, over ADE20K's 150 classes.
#
# The scene tab offers a slider per category, not per class: nobody wants to
# grade "sconce" and "crt screen" separately, and ADE20K's vocabulary is a
# scene-parsing benchmark rather than a photographer's list. This file is the
# translation, and it is data so that changing it is not a code change.
#
# Format: one category per line, `name = class, class, ...`, where each class
# is a name from the model's own `.classes.json`. Names rather than indices
# because an index is silently wrong after a re-export and a name is loudly
# wrong; `SceneModel::from_path` refuses a file naming a class the model does
# not have.
#
# ## Why the list is short, and why "other" is not in it
#
# Weights come from a softmax over all 150 channels summed within each
# category, so the categories listed here plus everything unlisted sum to 1 at
# every pixel. That is what lets the scene tab feather two adjacent categories
# without painting both grades into the overlap. Adding a category takes
# weight from the unlisted remainder rather than from its neighbours, so this
# list can grow without disturbing what is already here.
#
# Classes are assigned to at most one category — an overlap would break the
# partition and double-count the shared class, so the loader rejects it.
# The one the whole exercise started from. ADE20K's easiest class, and the one
# most often graded on its own in a landscape.
sky = sky
# Foliage, not "green things": a lawn and a canopy take the same saturation
# and luminance moves far more often than either takes the sky's.
vegetation = tree, grass, plant, flower, palm, field
# Standing and moving water together. `swimming pool` and `fountain` are here
# rather than under architecture because what a photographer adjusts is the
# water, not the basin.
water = water, sea, river, lake, waterfall, swimming pool, fountain
# Distant landform. Separate from `ground` because it is usually far, hazy and
# wants dehaze and contrast where a foreground surface wants neither.
terrain = mountain, rock, hill
# What the photographer is standing on, or would be. Earth and sand sit here
# rather than with terrain for the same near/far reason.
ground = earth, sand, land, dirt track, path, road, sidewalk, runway, floor, step, stairs, stairway
# Built structure. Deliberately broad: a facade, its railings and its awning
# are one surface as far as a global grade is concerned.
architecture = building, house, skyscraper, wall, tower, bridge, hovel, fence, column, awning, booth, canopy, pier, railing, grandstand
# Present for the scene tab's "expose people" move, and *not* a replacement for
# the instance model — this is every person in the frame at once, which is the
# right granularity for a global grade and the wrong one for selecting a
# subject. See `models/LICENCE.md`.
person = person
@@ -0,0 +1,152 @@
[
"wall",
"building",
"sky",
"floor",
"tree",
"ceiling",
"road",
"bed",
"windowpane",
"grass",
"cabinet",
"sidewalk",
"person",
"earth",
"door",
"table",
"mountain",
"plant",
"curtain",
"chair",
"car",
"water",
"painting",
"sofa",
"shelf",
"house",
"sea",
"mirror",
"rug",
"field",
"armchair",
"seat",
"fence",
"desk",
"rock",
"wardrobe",
"lamp",
"bathtub",
"railing",
"cushion",
"base",
"box",
"column",
"signboard",
"chest of drawers",
"counter",
"sand",
"sink",
"skyscraper",
"fireplace",
"refrigerator",
"grandstand",
"path",
"stairs",
"runway",
"case",
"pool table",
"pillow",
"screen door",
"stairway",
"river",
"bridge",
"bookcase",
"blind",
"coffee table",
"toilet",
"flower",
"book",
"hill",
"bench",
"countertop",
"stove",
"palm",
"kitchen island",
"computer",
"swivel chair",
"boat",
"bar",
"arcade machine",
"hovel",
"bus",
"towel",
"light",
"truck",
"tower",
"chandelier",
"awning",
"streetlight",
"booth",
"television receiver",
"airplane",
"dirt track",
"apparel",
"pole",
"land",
"bannister",
"escalator",
"ottoman",
"bottle",
"buffet",
"poster",
"stage",
"van",
"ship",
"fountain",
"conveyor belt",
"canopy",
"washer",
"plaything",
"swimming pool",
"stool",
"barrel",
"basket",
"waterfall",
"tent",
"bag",
"minibike",
"cradle",
"oven",
"ball",
"food",
"step",
"tank",
"trade name",
"microwave",
"pot",
"animal",
"bicycle",
"lake",
"dishwasher",
"screen",
"blanket",
"sculpture",
"hood",
"sconce",
"vase",
"traffic light",
"tray",
"ashcan",
"fan",
"pier",
"crt screen",
"plate",
"monitor",
"bulletin board",
"shower",
"radiator",
"glass",
"clock",
"flag"
]
Binary file not shown.
+19
View File
@@ -69,4 +69,23 @@ package() {
fi fi
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/${_m}" install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/${_m}"
done done
# The scene model, for the per-category grades. Unlike the face weights
# this one is in the repository — AGPL, and GPLv3 §13 permits the
# combination (models/LICENCE.md) — so it is installed unconditionally and
# a pointer here is a broken checkout rather than a licence decision.
#
# Three files: the graph, its vocabulary, and the category descriptor
# grouping ADE20K's 150 classes into what the tab shows. All three, because
# dr_ui::library::scene_model reports the tab unavailable without any one
# of them. Only the graph gets the pointer check — the other two are
# legitimately a few kilobytes.
_src="models/scene/yolo26s-sem-ade20k.onnx"
if [[ "$(stat -c%s "${_src}")" -lt 100000 ]]; then
echo "error: the scene model is an LFS pointer — run: git lfs pull" >&2
return 1
fi
for _m in yolo26s-sem-ade20k.onnx yolo26s-sem-ade20k.classes.json categories.txt; do
install -Dm644 "models/scene/${_m}" "${pkgdir}/usr/share/darkroom/models/${_m}"
done
} }
@@ -165,6 +165,21 @@ modules:
install -Dm644 "models/face/$m" "/app/share/darkroom/models/$m" install -Dm644 "models/face/$m" "/app/share/darkroom/models/$m"
done done
# The scene model, for the per-category grades. In the repository, unlike
# the face weights — AGPL, and GPLv3 §13 permits the combination
# (models/LICENCE.md) — so it installs unconditionally. Three files: the
# graph, its vocabulary, and the category descriptor; dr_ui reports the
# tab unavailable without any one of them. The pointer check is on the
# graph alone, the other two being legitimately small.
- |
if [ "$(stat -c%s models/scene/yolo26s-sem-ade20k.onnx)" -lt 100000 ]; then
echo "error: the scene model is an LFS pointer — run: git lfs pull" >&2
exit 1
fi
for m in yolo26s-sem-ade20k.onnx yolo26s-sem-ade20k.classes.json categories.txt; do
install -Dm644 "models/scene/$m" "/app/share/darkroom/models/$m"
done
- install -Dm644 README.md /app/share/doc/darkroom/README.md - install -Dm644 README.md /app/share/doc/darkroom/README.md
sources: sources:
+69 -9
View File
@@ -1,14 +1,25 @@
#!/usr/bin/env bash #!/usr/bin/env bash
# Re-export the segmentation model that ships in core/dr-segment/models/. # Re-export the segmentation models that ship in models/ at the repository root.
# #
# The .onnx is committed (D14), so this is not part of any build — it exists so # The .onnx is committed (D14), so this is not part of any build — it exists so
# the committed artefact is reproducible rather than a binary someone once # the committed artefact is reproducible rather than a binary someone once
# produced and nobody can regenerate. Run it when bumping the model. # produced and nobody can regenerate. Run it when bumping a model.
# #
# ./tools/export-seg-model.sh # ./tools/export-seg-model.sh # instance -> models/segment/
# ./tools/export-seg-model.sh yolo26s-sem-ade20k # semantic -> models/scene/
# #
# Requires `uv`. Everything else is fetched into a throwaway venv. # Requires `uv`. Everything else is fetched into a throwaway venv.
# #
# ## The two models, and why they are both here
#
# `yolo26n-seg` is COCO instance segmentation: it separates *things*, so
# clicking one of three people selects that person. `yolo26s-sem-ade20k` is
# ADE20K semantic segmentation: it labels every pixel with one of 150 classes
# including the *stuff* — sky, vegetation, water — that COCO has no word for,
# but it merges same-class pixels into one region and so cannot tell those
# three people apart. Neither substitutes for the other; the scene tab wants
# the second and local adjustments want the first.
#
# ## Why these export flags # ## Why these export flags
# #
# `dynamic=False` is not a default we failed to change: **tract cannot parse # `dynamic=False` is not a default we failed to change: **tract cannot parse
@@ -20,15 +31,44 @@
# `imgsz` square rather than a rectangle matched to 3:2: one graph has to # `imgsz` square rather than a rectangle matched to 3:2: one graph has to
# serve portrait, landscape, square crops and panoramas. A landscape-shaped # serve portrait, landscape, square crops and panoramas. A landscape-shaped
# graph trades letterbox waste on 3:2 for worse waste on everything else. # graph trades letterbox waste on 3:2 for worse waste on everything else.
#
# ## Why the semantic export is truncated
#
# Ultralytics ends the `-sem-` graph with `Resize -> ArgMax -> Cast`, handing
# back a `[1, 640, 640]` u8 label map. Two reasons that tail is cut here:
#
# 1. **Cost.** The Resize materialises 150 x 640 x 640 x f32 — *246 MB* — and
# ArgMax then reduces across the channel axis, which in NCHW strides
# 409,600 elements per comparison. Measured on one machine it was roughly
# four fifths of total runtime, for work the app throws away.
# 2. **Softness.** ArgMax destroys the per-class scores. The scene tab needs
# them: softmax over the 150 channels, summed within each photographic
# category, gives per-category weights that sum to 1 at every pixel — a
# partition of unity. Feathering those cannot double-grade a boundary,
# where feathering hard labels outward from two adjacent categories does.
#
# The upsample is not information: the graph's true spatial resolution is the
# logit grid (80x80 at imgsz=640), and the app can resample from that itself.
set -euo pipefail set -euo pipefail
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
REPO="$(cd "${HERE}/.." && pwd)" REPO="$(cd "${HERE}/.." && pwd)"
OUT="${REPO}/core/dr-segment/models"
MODEL="${1:-yolo26n-seg}" MODEL="${1:-yolo26n-seg}"
IMGSZ="${2:-640}" IMGSZ="${2:-640}"
WORK="$(mktemp -d)"
# Semantic models are the scene tab's; instance models are the selection path's.
case "${MODEL}" in
*-sem-*|*-sem) OUT="${REPO}/models/scene"; SEMANTIC=1 ;;
*) OUT="${REPO}/models/segment"; SEMANTIC=0 ;;
esac
# Not `mktemp -d`: the default TMPDIR is `/tmp`, which on most current Linux
# distributions is a tmpfs — RAM, sized at half of physical memory. The venv
# below pulls torch, several gigabytes of it, and installing that into RAM
# either evicts the user's page cache or fails outright with ENOSPC on a
# machine that has hundreds of gigabytes of actual disk free.
WORK="$(mktemp -d -p "${TMPDIR:-/var/tmp}")"
trap 'rm -rf "${WORK}"' EXIT trap 'rm -rf "${WORK}"' EXIT
echo "==> exporting ${MODEL} at imgsz=${IMGSZ} in ${WORK}" echo "==> exporting ${MODEL} at imgsz=${IMGSZ} in ${WORK}"
@@ -36,12 +76,13 @@ cd "${WORK}"
uv venv --python 3.12 venv uv venv --python 3.12 venv
VIRTUAL_ENV="${WORK}/venv" uv pip install ultralytics onnx onnxslim VIRTUAL_ENV="${WORK}/venv" uv pip install ultralytics onnx onnxslim
VIRTUAL_ENV="${WORK}/venv" "${WORK}/venv/bin/python" - "${MODEL}" "${IMGSZ}" <<'PY' VIRTUAL_ENV="${WORK}/venv" "${WORK}/venv/bin/python" - "${MODEL}" "${IMGSZ}" "${SEMANTIC}" <<'PY'
import sys, json import sys, json
import onnx
from onnx import helper
from ultralytics import YOLO from ultralytics import YOLO
name = sys.argv[1] name, imgsz, semantic = sys.argv[1], int(sys.argv[2]), sys.argv[3] == "1"
imgsz = int(sys.argv[2])
m = YOLO(f"{name}.pt") m = YOLO(f"{name}.pt")
path = m.export(format="onnx", opset=17, simplify=True, imgsz=imgsz, dynamic=False) path = m.export(format="onnx", opset=17, simplify=True, imgsz=imgsz, dynamic=False)
print("ONNX:", path) print("ONNX:", path)
@@ -52,6 +93,25 @@ print("ONNX:", path)
with open("classes.json", "w") as f: with open("classes.json", "w") as f:
json.dump([m.names[i] for i in range(len(m.names))], f, indent=1) json.dump([m.names[i] for i in range(len(m.names))], f, indent=1)
print("classes:", len(m.names)) print("classes:", len(m.names))
if semantic:
# Drop `Resize -> ArgMax -> Cast` and expose the classifier's logits. See
# the header for why. Matched by op type rather than by node name so a
# re-export under a different naming scheme still works, and asserted
# rather than assumed so an upstream graph change fails loudly here
# instead of silently shipping a differently-shaped model.
g = onnx.load(path).graph
tail = [n.op_type for n in g.node[-3:]]
assert tail == ["Resize", "ArgMax", "Cast"], f"unexpected graph tail: {tail}"
logits = g.node[-3].input[0]
del g.node[-3:]
del g.output[:]
g.output.extend([helper.make_tensor_value_info(logits, onnx.TensorProto.FLOAT, None)])
model = onnx.shape_inference.infer_shapes(helper.make_model(g, opset_imports=[helper.make_opsetid("", 17)]))
onnx.checker.check_model(model)
onnx.save(model, path)
shape = [d.dim_value for d in model.graph.output[0].type.tensor_type.shape.dim]
print("truncated to logits:", logits, shape)
PY PY
mkdir -p "${OUT}" mkdir -p "${OUT}"
@@ -61,4 +121,4 @@ cp "${WORK}/classes.json" "${OUT}/${MODEL}.classes.json"
echo "==> wrote:" echo "==> wrote:"
ls -la "${OUT}" ls -la "${OUT}"
echo echo
echo "Remember: these weights are AGPL-3.0 (see ${OUT}/LICENCE.md)." echo "Remember: these weights are AGPL-3.0 (see ${REPO}/models/LICENCE.md)."
+9
View File
@@ -73,6 +73,15 @@ pub use develop::DevelopSession;
/// opened — see `library::shared_face_models_dir`. /// opened — see `library::shared_face_models_dir`.
pub use library::shared_face_models_dir; pub use library::shared_face_models_dir;
/// The scene model's three files, wherever this device keeps them.
///
/// Public for the same reason as the directory above: the pieces that reach for
/// a model are not all inside this crate. Exported ahead of the scene tab that
/// will consume it so that the packaging and unpacking added alongside it have
/// something to be verified against — `assemble-apk.sh` writing files no lookup
/// looks for would be a silent mistake for as long as the tab took to arrive.
pub use library::scene_model;
pub mod launch; pub mod launch;
pub mod launch_ui; pub mod launch_ui;
+31
View File
@@ -3846,6 +3846,37 @@ pub fn face_models(account: &Account) -> Option<(PathBuf, PathBuf)> {
.or_else(|| system_face_models_dirs().into_iter().find_map(pair)) .or_else(|| system_face_models_dirs().into_iter().find_map(pair))
} }
/// The scene model, its vocabulary and its category descriptor, if all three
/// are present.
///
/// All three or none, for the same reason `face_models` insists on its pair:
/// the graph alone decodes to 150 anonymous channels, and a descriptor naming
/// classes a different model does not have is refused by
/// `dr_segment::scene::parse_categories` anyway. Reporting the set missing is
/// more useful than starting and failing at the first inference.
///
/// Searched in the same three places, most specific first — the account's own
/// directory, the shared one, then wherever a package installed them. Android
/// only ever finds the second, which is where `install_bundled_models` unpacks
/// the APK's copy before any store opens.
///
/// Unlike the face weights this model *is* in the repository, so a desktop
/// build from a complete checkout has it. Absent means either a checkout
/// without `git lfs pull` or a package that chose not to carry 24 MB, and the
/// scene tab reports itself unavailable rather than the app refusing to run.
pub fn scene_model(account: &Account) -> Option<(PathBuf, PathBuf, PathBuf)> {
fn set(dir: PathBuf) -> Option<(PathBuf, PathBuf, PathBuf)> {
let model = dir.join("yolo26s-sem-ade20k.onnx");
let classes = dir.join("yolo26s-sem-ade20k.classes.json");
let categories = dir.join("categories.txt");
(model.is_file() && classes.is_file() && categories.is_file())
.then_some((model, classes, categories))
}
set(face_models_dir(account))
.or_else(|| set(shared_face_models_dir()))
.or_else(|| system_face_models_dirs().into_iter().find_map(set))
}
/// Where a *package* may have installed the models. /// Where a *package* may have installed the models.
/// ///
/// `$XDG_DATA_DIRS` rather than a hard-coded `/usr/share`, because that is the /// `$XDG_DATA_DIRS` rather than a hard-coded `/usr/share`, because that is the