Merge branch 'master' into fix/gallery-selection
# Conflicts: # docs/traceability.md
This commit is contained in:
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
# Model weights live in LFS.
|
# Model weights live in LFS.
|
||||||
#
|
#
|
||||||
# `core/dr-segment/models/*.onnx` is ~11 MB of binary that changes wholesale
|
# `models/**/*.onnx` is tens of MB of binary that changes wholesale
|
||||||
# when it changes at all. In ordinary git objects every future revision of it
|
# when it changes at all. In ordinary git objects every future revision of it
|
||||||
# would be stored in full, in every clone, forever — and the one thing nobody
|
# would be stored in full, in every clone, forever — and the one thing nobody
|
||||||
# can do with it is a useful diff.
|
# can do with it is a useful diff.
|
||||||
|
|||||||
@@ -57,7 +57,7 @@ jobs:
|
|||||||
|
|
||||||
# The model, which is in LFS and is not optional.
|
# The model, which is in LFS and is not optional.
|
||||||
#
|
#
|
||||||
# `core/dr-segment/models/*.onnx` is tracked in LFS (.gitattributes), so a
|
# `models/**/*.onnx` is tracked in LFS (.gitattributes), so a
|
||||||
# plain checkout writes a ~130-byte pointer where 11 MB should be, and
|
# plain checkout writes a ~130-byte pointer where 11 MB should be, and
|
||||||
# `dr-segment`'s build script panics by design rather than embedding a
|
# `dr-segment`'s build script panics by design rather than embedding a
|
||||||
# pointer and failing at inference. That failure reads like a broken build
|
# pointer and failing at inference. That failure reads like a broken build
|
||||||
@@ -97,7 +97,7 @@ jobs:
|
|||||||
git config --local lfs.url \
|
git config --local lfs.url \
|
||||||
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
|
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
|
||||||
git lfs pull
|
git lfs pull
|
||||||
ls -l core/dr-segment/models/
|
ls -lR models/
|
||||||
|
|
||||||
- name: Cache cargo
|
- name: Cache cargo
|
||||||
uses: actions/cache@v4
|
uses: actions/cache@v4
|
||||||
@@ -174,7 +174,7 @@ jobs:
|
|||||||
|
|
||||||
# The model, which is in LFS and is not optional.
|
# The model, which is in LFS and is not optional.
|
||||||
#
|
#
|
||||||
# `core/dr-segment/models/*.onnx` is tracked in LFS (.gitattributes), so a
|
# `models/**/*.onnx` is tracked in LFS (.gitattributes), so a
|
||||||
# plain checkout writes a ~130-byte pointer where 11 MB should be, and
|
# plain checkout writes a ~130-byte pointer where 11 MB should be, and
|
||||||
# `dr-segment`'s build script panics by design rather than embedding a
|
# `dr-segment`'s build script panics by design rather than embedding a
|
||||||
# pointer and failing at inference. That failure reads like a broken build
|
# pointer and failing at inference. That failure reads like a broken build
|
||||||
@@ -214,7 +214,7 @@ jobs:
|
|||||||
git config --local lfs.url \
|
git config --local lfs.url \
|
||||||
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
|
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
|
||||||
git lfs pull
|
git lfs pull
|
||||||
ls -l core/dr-segment/models/
|
ls -lR models/
|
||||||
|
|
||||||
- name: Cache cargo
|
- name: Cache cargo
|
||||||
uses: actions/cache@v4
|
uses: actions/cache@v4
|
||||||
|
|||||||
@@ -60,7 +60,7 @@ fn android_main(app: slint::android::AndroidApp) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// After the data dir and before anything asks whether a model is present.
|
// After the data dir and before anything asks whether a model is present.
|
||||||
install_bundled_face_models(&app);
|
install_bundled_models(&app);
|
||||||
|
|
||||||
// Before `init_with_event_listener`, which takes `app` by value and is the
|
// Before `init_with_event_listener`, which takes `app` by value and is the
|
||||||
// last moment anything can ask the activity a question. Not an ordering
|
// last moment anything can ask the activity a question. Not an ordering
|
||||||
@@ -126,38 +126,65 @@ fn android_main(app: slint::android::AndroidApp) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Unpack the face models the APK carries, if it carries any.
|
/// Unpack the models the APK carries, if it carries any.
|
||||||
///
|
///
|
||||||
/// # Why Android needs this and no other platform does
|
/// # Why Android needs this and no other platform does
|
||||||
///
|
///
|
||||||
/// The weights are not a build input and are not in the repository — the
|
/// A desktop build reads its models from a path — the account's directory, the
|
||||||
/// InsightFace grant is research-only and incompatible with this project's
|
/// shared one, or `$XDG_DATA_DIRS` where a package put them. **Android has no
|
||||||
/// licence (docs/faces.md §2), so a desktop user fetches them, runs
|
/// such path.** `internal_data_path` is app-private, `run-as` needs a
|
||||||
/// `tools/fix-face-model-shapes.sh` over them, and drops the result into
|
/// debuggable build, and an asset inside a package is not a path anything can
|
||||||
/// `~/.local/share/darkroom/models/`. **That gesture does not exist on
|
/// open (ARCH §6.9), so a phone had no way to reach a model at all.
|
||||||
/// Android.** `internal_data_path` is app-private, `run-as` needs a debuggable
|
|
||||||
/// build, and there is no picker and no fetch in the app, so a phone had no way
|
|
||||||
/// to acquire a model at all and face indexing reported itself permanently off.
|
|
||||||
///
|
///
|
||||||
/// So a locally-built APK may carry the pair in `assets/models/`, which
|
/// So the APK carries them in `assets/models/` and this copies them out, once,
|
||||||
/// `assemble-apk.sh` includes when the tree has them and omits when it does
|
/// into the same shared directory a desktop install uses. After that every
|
||||||
/// not. Nothing changes about what the repository holds or what a published
|
/// lookup in `dr_ui::library` finds them exactly where it finds a desktop
|
||||||
/// build could redistribute; this only gives a self-built APK the same route a
|
/// user's.
|
||||||
/// desktop build has always had.
|
|
||||||
///
|
///
|
||||||
/// Absent assets are the ordinary case, not an error — the same quiet "no model
|
/// # The two sets are not the same kind of thing
|
||||||
/// installed" state a fresh desktop install is in.
|
///
|
||||||
|
/// **Face weights are absent from the repository by design.** The InsightFace
|
||||||
|
/// grant is research-only and incompatible with this project's licence
|
||||||
|
/// (docs/faces.md §2), so a desktop user fetches them, runs
|
||||||
|
/// `tools/fix-face-model-shapes.sh` over them, and drops the result in. A build
|
||||||
|
/// that carries none is the ordinary case and face indexing simply stays off.
|
||||||
|
///
|
||||||
|
/// **The scene model is committed** (AGPL, compatible — `models/LICENCE.md`),
|
||||||
|
/// so a build carrying none means a checkout without `git lfs pull` rather than
|
||||||
|
/// a deliberate omission. It is still not an error here: the scene tab reports
|
||||||
|
/// itself unavailable the same way face indexing does, because a photo editor
|
||||||
|
/// that refuses to start over a missing grading feature is worse than one that
|
||||||
|
/// starts without it.
|
||||||
|
///
|
||||||
|
/// # Why it is not `include_bytes!` like the instance model
|
||||||
|
///
|
||||||
|
/// Size. The instance model is 11 MB and compiled in; the scene model is 24 MB
|
||||||
|
/// on top of that, and a 35 MB constant in the binary is paid by every install
|
||||||
|
/// whether or not the tab is opened. Assets are also *stored* rather than
|
||||||
|
/// deflated in the APK (see `assemble-apk.sh`), so unpacking is a copy rather
|
||||||
|
/// than an inflate.
|
||||||
#[cfg(target_os = "android")]
|
#[cfg(target_os = "android")]
|
||||||
fn install_bundled_face_models(app: &slint::android::AndroidApp) {
|
fn install_bundled_models(app: &slint::android::AndroidApp) {
|
||||||
use std::io::Read;
|
use std::io::Read;
|
||||||
|
|
||||||
// The **shape-fixed** names, matching what `library::face_models` looks
|
// The face names are the **shape-fixed** exports, matching what
|
||||||
// for: tract cannot parse either InsightFace graph with its dynamic input
|
// `library::face_models` looks for: tract cannot parse either InsightFace
|
||||||
// dimension, so what ships here has already been through
|
// graph with its dynamic input dimension, so what ships here has already
|
||||||
// `tools/fix-face-model-shapes.sh`.
|
// been through `tools/fix-face-model-shapes.sh`.
|
||||||
const BUNDLED: [(&std::ffi::CStr, &str); 2] = [
|
//
|
||||||
|
// The scene entries are three files rather than one because the graph alone
|
||||||
|
// decodes to 150 anonymous channels — `library::scene_model` wants the
|
||||||
|
// vocabulary and the category descriptor beside it, and requires all three
|
||||||
|
// before it reports the tab available.
|
||||||
|
const BUNDLED: [(&std::ffi::CStr, &str); 5] = [
|
||||||
(c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"),
|
(c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"),
|
||||||
(c"models/arcface_mbf_b1.onnx", "arcface_mbf_b1.onnx"),
|
(c"models/arcface_mbf_b1.onnx", "arcface_mbf_b1.onnx"),
|
||||||
|
(c"models/yolo26s-sem-ade20k.onnx", "yolo26s-sem-ade20k.onnx"),
|
||||||
|
(
|
||||||
|
c"models/yolo26s-sem-ade20k.classes.json",
|
||||||
|
"yolo26s-sem-ade20k.classes.json",
|
||||||
|
),
|
||||||
|
(c"models/categories.txt", "categories.txt"),
|
||||||
];
|
];
|
||||||
|
|
||||||
let dir = dr_ui::shared_face_models_dir();
|
let dir = dr_ui::shared_face_models_dir();
|
||||||
@@ -172,7 +199,7 @@ fn install_bundled_face_models(app: &slint::android::AndroidApp) {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let Some(mut asset) = assets.open(asset_path) else {
|
let Some(mut asset) = assets.open(asset_path) else {
|
||||||
log::info!("no bundled {name} in this APK; face indexing stays off");
|
log::info!("no bundled {name} in this APK; the feature needing it stays off");
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
let mut bytes = Vec::new();
|
let mut bytes = Vec::new();
|
||||||
@@ -185,8 +212,8 @@ fn install_bundled_face_models(app: &slint::android::AndroidApp) {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
// Written under a temporary name and renamed, because
|
// Written under a temporary name and renamed, because
|
||||||
// `library::face_models` decides face indexing is available on
|
// `library::face_models` and `library::scene_model` both decide a
|
||||||
// `is_file()` alone. A truncated write — the process backgrounded and
|
// feature is available on `is_file()` alone. A truncated write — the process backgrounded and
|
||||||
// killed mid-copy — would otherwise leave a file that passes that test
|
// killed mid-copy — would otherwise leave a file that passes that test
|
||||||
// and fails inside tract, reported to the user as a broken model rather
|
// and fails inside tract, reported to the user as a broken model rather
|
||||||
// than a missing one.
|
// than a missing one.
|
||||||
|
|||||||
@@ -44,3 +44,13 @@ semantic = ["dep:ort", "dep:ort-tract", "dep:ndarray"]
|
|||||||
# it must be embedded; a desktop packager pointing at a system model directory,
|
# it must be embedded; a desktop packager pointing at a system model directory,
|
||||||
# or a test that only needs the decoder, wants the runtime without the 11 MB.
|
# or a test that only needs the decoder, wants the runtime without the 11 MB.
|
||||||
embedded-model = ["semantic"]
|
embedded-model = ["semantic"]
|
||||||
|
|
||||||
|
# Compile the *scene* model in too, and off by default where `embedded-model`
|
||||||
|
# is on.
|
||||||
|
#
|
||||||
|
# The asymmetry is its size. At 24 MB it is more than twice the instance model,
|
||||||
|
# and Android reaches it the way it reaches the face weights — unpacked from
|
||||||
|
# APK assets at first launch — rather than by carrying it in the binary. This
|
||||||
|
# feature is for a desktop build with nowhere else to read it from, and for
|
||||||
|
# tests that want the real graph.
|
||||||
|
embedded-scene-model = ["semantic"]
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
//! Check the model is a model and not an LFS pointer.
|
//! Check the model is a model and not an LFS pointer.
|
||||||
//!
|
//!
|
||||||
//! `models/*.onnx` is stored in Git LFS (see `.gitattributes`). A clone made
|
//! `models/segment/*.onnx` is stored in Git LFS (see `.gitattributes`). A clone made
|
||||||
//! without git-lfs installed, or with `GIT_LFS_SKIP_SMUDGE` set, leaves a
|
//! without git-lfs installed, or with `GIT_LFS_SKIP_SMUDGE` set, leaves a
|
||||||
//! ~130-byte text pointer at that path instead of the weights.
|
//! ~130-byte text pointer at that path instead of the weights.
|
||||||
//!
|
//!
|
||||||
@@ -12,7 +12,7 @@
|
|||||||
|
|
||||||
use std::path::Path;
|
use std::path::Path;
|
||||||
|
|
||||||
const MODEL: &str = "models/yolo26n-seg.onnx";
|
const MODEL: &str = "../../models/segment/yolo26n-seg.onnx";
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
println!("cargo:rerun-if-changed={MODEL}");
|
println!("cargo:rerun-if-changed={MODEL}");
|
||||||
|
|||||||
@@ -0,0 +1,174 @@
|
|||||||
|
//! Run the scene model over a JPEG, time it, and write what it saw.
|
||||||
|
//!
|
||||||
|
//! Two jobs in one example because they need the same setup and answering
|
||||||
|
//! either one alone leaves the other open.
|
||||||
|
//!
|
||||||
|
//! **Looking.** Same argument as `detect`: no unit test settles whether the
|
||||||
|
//! letterbox inverse in `Scene::rasterise` is right, because an off-by-one
|
||||||
|
//! produces perfectly plausible weights over slightly the wrong pixels. A sky
|
||||||
|
//! mask laid over the photograph settles it in one glance.
|
||||||
|
//!
|
||||||
|
//! **Timing.** Every number quoted while this model was being chosen came off a
|
||||||
|
//! laptop that was compiling other things at the time, which makes them upper
|
||||||
|
//! bounds and nothing better. This exists so the figure that ends up in a
|
||||||
|
//! document came from a quiet machine and can be reproduced on another one.
|
||||||
|
//!
|
||||||
|
//! ```sh
|
||||||
|
//! cargo run -p dr-segment --example scene --release --features embedded-scene-model -- photo.jpg
|
||||||
|
//! cargo run -p dr-segment --example scene --release -- photo.jpg out 20 \
|
||||||
|
//! models/scene/yolo26s-sem-ade20k.onnx
|
||||||
|
//! ```
|
||||||
|
//!
|
||||||
|
//! Writes `<prefix>-<category>.ppm` per category — the photograph darkened
|
||||||
|
//! where the category is absent, so the mask is legible *against the picture it
|
||||||
|
//! came from* rather than as an abstract grey field. PPM for the same reason
|
||||||
|
//! the other examples use it: no encoder dependency, and every viewer reads it.
|
||||||
|
//!
|
||||||
|
//! Timings are reported as a median over the requested run count, with the
|
||||||
|
//! first run excluded. That first pass pays for tract's lazy allocation and is
|
||||||
|
//! not representative of the second image a session decodes.
|
||||||
|
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
use dr_segment::scene::SceneModel;
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
env_logger::init();
|
||||||
|
|
||||||
|
let mut args = std::env::args().skip(1);
|
||||||
|
let Some(path) = args.next() else {
|
||||||
|
eprintln!(
|
||||||
|
"usage: scene <photo.jpg> [out-prefix] [runs] [model.onnx classes.json categories.txt]"
|
||||||
|
);
|
||||||
|
eprintln!(" with --features embedded-scene-model the model arguments may be omitted");
|
||||||
|
std::process::exit(2);
|
||||||
|
};
|
||||||
|
let prefix = args.next().unwrap_or_else(|| "scene".into());
|
||||||
|
let runs: usize = args
|
||||||
|
.next()
|
||||||
|
.and_then(|r| r.parse().ok())
|
||||||
|
.unwrap_or(10)
|
||||||
|
.max(1);
|
||||||
|
|
||||||
|
let (rgb, width, height) = read_jpeg(&path);
|
||||||
|
println!("{path}: {width}×{height}");
|
||||||
|
|
||||||
|
let mut model = match (args.next(), args.next(), args.next()) {
|
||||||
|
(Some(m), Some(c), Some(g)) => {
|
||||||
|
SceneModel::from_path(m, c, g).expect("could not load the scene model")
|
||||||
|
}
|
||||||
|
_ => embedded(),
|
||||||
|
};
|
||||||
|
|
||||||
|
// Excluded from the statistics deliberately — see the header.
|
||||||
|
let warm = Instant::now();
|
||||||
|
let scene = model
|
||||||
|
.analyse(&rgb, width, height)
|
||||||
|
.expect("inference failed");
|
||||||
|
println!("first run: {:?} (allocation included)", warm.elapsed());
|
||||||
|
|
||||||
|
let mut times: Vec<f64> = Vec::with_capacity(runs);
|
||||||
|
for _ in 0..runs {
|
||||||
|
let start = Instant::now();
|
||||||
|
let _ = model
|
||||||
|
.analyse(&rgb, width, height)
|
||||||
|
.expect("inference failed");
|
||||||
|
times.push(start.elapsed().as_secs_f64() * 1000.0);
|
||||||
|
}
|
||||||
|
times.sort_by(f64::total_cmp);
|
||||||
|
println!(
|
||||||
|
"{runs} runs: median {:.0} ms (min {:.0}, max {:.0})",
|
||||||
|
times[times.len() / 2],
|
||||||
|
times[0],
|
||||||
|
times[times.len() - 1],
|
||||||
|
);
|
||||||
|
|
||||||
|
let (gw, gh) = scene.grid_size();
|
||||||
|
println!("logit grid: {gw}×{gh}");
|
||||||
|
println!();
|
||||||
|
|
||||||
|
// Coverage first and sorted, because on any given photograph most
|
||||||
|
// categories are absent and the two or three that are not are the whole
|
||||||
|
// story.
|
||||||
|
let mut ranked: Vec<(usize, f32)> = (0..scene.categories().len())
|
||||||
|
.map(|k| (k, scene.coverage(k)))
|
||||||
|
.collect();
|
||||||
|
ranked.sort_by(|a, b| b.1.total_cmp(&a.1));
|
||||||
|
|
||||||
|
for (k, coverage) in ranked {
|
||||||
|
let name = &scene.categories()[k];
|
||||||
|
println!("{name:>14} {:5.1}%", coverage * 100.0);
|
||||||
|
// A category covering essentially nothing produces a black image and a
|
||||||
|
// file nobody wants; the threshold is what the scene tab would use to
|
||||||
|
// decide whether to offer a slider at all.
|
||||||
|
if coverage < 0.005 {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mask = scene
|
||||||
|
.rasterise(k, width, height)
|
||||||
|
.expect("category index came from the same Scene");
|
||||||
|
write_overlay(&format!("{prefix}-{name}.ppm"), &rgb, &mask, width, height);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "embedded-scene-model")]
|
||||||
|
fn embedded() -> SceneModel {
|
||||||
|
SceneModel::embedded().expect("could not load the embedded scene model")
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(not(feature = "embedded-scene-model"))]
|
||||||
|
fn embedded() -> SceneModel {
|
||||||
|
eprintln!(
|
||||||
|
"no model given, and this build has no embedded one.\n\
|
||||||
|
Either pass the three paths, or rebuild with --features embedded-scene-model."
|
||||||
|
);
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The photograph, dimmed where the category is not.
|
||||||
|
///
|
||||||
|
/// Not a bare greyscale mask: the question being asked is "does this weight
|
||||||
|
/// land on the sky", and a mask on its own cannot answer it — you have to see
|
||||||
|
/// the sky underneath. A floor rather than a multiply, so that a region the
|
||||||
|
/// model gave up on is still visible enough to recognise.
|
||||||
|
fn write_overlay(path: &str, rgb: &[f32], mask: &[f32], width: usize, height: usize) {
|
||||||
|
let mut out = String::with_capacity(64);
|
||||||
|
out.push_str(&format!("P3\n{width} {height}\n255\n"));
|
||||||
|
let mut bytes = out.into_bytes();
|
||||||
|
|
||||||
|
for i in 0..width * height {
|
||||||
|
let w = mask[i].clamp(0.0, 1.0);
|
||||||
|
let gain = 0.15 + 0.85 * w;
|
||||||
|
for c in 0..3 {
|
||||||
|
let v = (rgb[i * 3 + c] * gain * 255.0).clamp(0.0, 255.0) as u8;
|
||||||
|
bytes.extend_from_slice(v.to_string().as_bytes());
|
||||||
|
bytes.push(if c == 2 { b'\n' } else { b' ' });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
match std::fs::write(path, bytes) {
|
||||||
|
Ok(()) => println!(" wrote {path}"),
|
||||||
|
Err(e) => eprintln!(" could not write {path}: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode to the tightly packed `f32` RGB the model wants.
|
||||||
|
fn read_jpeg(path: &str) -> (Vec<f32>, usize, usize) {
|
||||||
|
let bytes = std::fs::read(path).expect("could not read the photograph");
|
||||||
|
let mut decoder = zune_jpeg::JpegDecoder::new(&bytes);
|
||||||
|
let pixels = decoder.decode().expect("could not decode the photograph");
|
||||||
|
let info = decoder.info().expect("decoded image has no dimensions");
|
||||||
|
let (width, height) = (info.width as usize, info.height as usize);
|
||||||
|
|
||||||
|
// zune hands back whatever the file had. Three channels is the ordinary
|
||||||
|
// case; one is a greyscale scan, which is worth handling because a
|
||||||
|
// black-and-white frame is exactly the kind of thing someone reaches for
|
||||||
|
// when a colour one looks wrong.
|
||||||
|
let components = pixels.len() / (width * height);
|
||||||
|
let rgb = match components {
|
||||||
|
3 => pixels.iter().map(|&p| p as f32 / 255.0).collect(),
|
||||||
|
1 => pixels.iter().flat_map(|&p| [p as f32 / 255.0; 3]).collect(),
|
||||||
|
n => panic!("unsupported component count: {n}"),
|
||||||
|
};
|
||||||
|
(rgb, width, height)
|
||||||
|
}
|
||||||
@@ -1,54 +0,0 @@
|
|||||||
# Model weights — licensing
|
|
||||||
|
|
||||||
`yolo26n-seg.onnx` is exported from Ultralytics YOLO26n-seg
|
|
||||||
(`https://huggingface.co/Ultralytics/YOLO26`, `yolo26n-seg.pt`) by
|
|
||||||
`tools/export-seg-model.sh`. `yolo26n-seg.classes.json` is that checkpoint's
|
|
||||||
class vocabulary, written out by the same script.
|
|
||||||
|
|
||||||
## The grant
|
|
||||||
|
|
||||||
**Ultralytics releases YOLO under AGPL-3.0**, and the weights carry the same
|
|
||||||
grant as the framework — the HuggingFace repository declares `agpl-3.0` for the
|
|
||||||
checkpoints themselves, not merely for the training code. A commercial licence
|
|
||||||
is offered separately; DarkRoom does not use it and does not need it.
|
|
||||||
|
|
||||||
## What that means for DarkRoom
|
|
||||||
|
|
||||||
DarkRoom is GPL-3.0-or-later. **GPLv3 §13 explicitly permits combination with
|
|
||||||
AGPL-3.0 code**, so redistributing these weights inside this repository is
|
|
||||||
allowed — this is *not* the situation the InsightFace "buffalo" weights would
|
|
||||||
have created, where a non-commercial research grant is simply incompatible with
|
|
||||||
the project's licence and with F-Droid, Flatpak and Play distribution
|
|
||||||
(NFR-COMPAT-2, D13).
|
|
||||||
|
|
||||||
The consequence, and it is a real one: **the combined work is effectively
|
|
||||||
AGPL-3.0.** §13's permission runs one way — the AGPL's §13 network-use condition
|
|
||||||
attaches to the portion under that licence. For a local-first desktop and
|
|
||||||
Android photo editor that condition has no practical bite, because there is no
|
|
||||||
network service offering the combined work to remote users. It would acquire
|
|
||||||
bite the moment any hosted or server-side rendering appeared, and that is the
|
|
||||||
thing to remember rather than rediscover.
|
|
||||||
|
|
||||||
This was decided deliberately (D14), not arrived at by accident, and
|
|
||||||
`docs/segmentation.md` §7 records the reasoning.
|
|
||||||
|
|
||||||
## Class vocabulary — a caveat worth reading
|
|
||||||
|
|
||||||
`docs/segmentation.md` §4 specified YOLO **pretrained on ADE20K**, whose 150
|
|
||||||
classes include the *stuff* categories that matter most in photography — sky,
|
|
||||||
vegetation, water, wall, mountain.
|
|
||||||
|
|
||||||
**No such model exists in usable form.** Checked 2026-08-21: Ultralytics ships
|
|
||||||
YOLO26-seg trained on **COCO**, whose 80 classes are all *things* — person,
|
|
||||||
dog, car, bird, potted plant — and the one HuggingFace repository claiming a
|
|
||||||
YOLO/ADE20K combination (`laxmacl/yolov8-ade20k`) is empty. ADE20K semantic
|
|
||||||
models do exist, but as SegFormer/OneFormer/MaskFormer transformers, not YOLO.
|
|
||||||
|
|
||||||
So the shipped vocabulary selects **subjects**, not **stuff**. "Select the
|
|
||||||
person" works; "select the sky" does not come from the model and must come from
|
|
||||||
the watershed hierarchy instead. That is a narrower arm B than §4 assumed, and
|
|
||||||
it raises rather than lowers the importance of arm C.
|
|
||||||
|
|
||||||
The loader treats the vocabulary as model metadata rather than compiled-in
|
|
||||||
knowledge, so adding a stuff-class model later is a file plus a descriptor, not
|
|
||||||
a code change.
|
|
||||||
@@ -28,17 +28,30 @@
|
|||||||
//! model is COCO-trained, so it recognises subjects and has no class for sky,
|
//! model is COCO-trained, so it recognises subjects and has no class for sky,
|
||||||
//! foliage or wall (`models/LICENCE.md`). Selecting those falls to arm A,
|
//! foliage or wall (`models/LICENCE.md`). Selecting those falls to arm A,
|
||||||
//! which never needed a vocabulary to begin with.
|
//! which never needed a vocabulary to begin with.
|
||||||
|
//!
|
||||||
|
//! # And [`scene`], which is not one of the arms
|
||||||
|
//!
|
||||||
|
//! The three arms all serve *local* adjustment: they exist so a mask can be
|
||||||
|
//! snapped to one region of the picture. [`scene`] serves the opposite move —
|
||||||
|
//! one grade applied to every pixel of a category at once, sky or foliage or
|
||||||
|
//! water — and reads a second, ADE20K-trained model to do it. It shares this
|
||||||
|
//! crate because it shares the runtime and the letterbox, not because it is
|
||||||
|
//! another way of doing the same thing.
|
||||||
|
|
||||||
pub mod distance;
|
pub mod distance;
|
||||||
pub mod hierarchy;
|
pub mod hierarchy;
|
||||||
pub mod prior;
|
pub mod prior;
|
||||||
#[cfg(feature = "semantic")]
|
#[cfg(feature = "semantic")]
|
||||||
|
pub mod scene;
|
||||||
|
#[cfg(feature = "semantic")]
|
||||||
pub mod semantic;
|
pub mod semantic;
|
||||||
|
|
||||||
pub use distance::{signed_distance, Falloff, Morphology, Shaped};
|
pub use distance::{signed_distance, Falloff, Morphology, Shaped};
|
||||||
pub use hierarchy::{Edge, Merge, MergeTree, RegionField};
|
pub use hierarchy::{Edge, Merge, MergeTree, RegionField};
|
||||||
pub use prior::{Membership, PriorOptions};
|
pub use prior::{Membership, PriorOptions};
|
||||||
#[cfg(feature = "semantic")]
|
#[cfg(feature = "semantic")]
|
||||||
|
pub use scene::{Category, Scene, SceneModel};
|
||||||
|
#[cfg(feature = "semantic")]
|
||||||
pub use semantic::{Instance, SemanticModel, SemanticOptions, Tiling};
|
pub use semantic::{Instance, SemanticModel, SemanticOptions, Tiling};
|
||||||
|
|
||||||
/// What can go wrong between an image and a region map.
|
/// What can go wrong between an image and a region map.
|
||||||
@@ -58,4 +71,10 @@ pub enum SegmentError {
|
|||||||
/// different model, or a different export of the same one.
|
/// different model, or a different export of the same one.
|
||||||
#[error("model output '{0}' did not have the expected shape")]
|
#[error("model output '{0}' did not have the expected shape")]
|
||||||
OutputShape(&'static str),
|
OutputShape(&'static str),
|
||||||
|
|
||||||
|
/// `models/scene/categories.txt` and the model disagree, or the descriptor
|
||||||
|
/// is malformed. Its own variant rather than a parse error because every
|
||||||
|
/// case carries a specific sentence about what to fix.
|
||||||
|
#[error("category descriptor: {0}")]
|
||||||
|
CategoryDescriptor(String),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,525 @@
|
|||||||
|
//! Per-category weights over the whole frame — what the scene tab grades.
|
||||||
|
//!
|
||||||
|
//! [`semantic`](crate::semantic) answers "what objects are in this picture, and
|
||||||
|
//! which pixels are each one". This module answers a different question: "how
|
||||||
|
//! much of each pixel is sky". They are not the same question and they do not
|
||||||
|
//! want the same model.
|
||||||
|
//!
|
||||||
|
//! # Why a second model rather than a second reading of the first
|
||||||
|
//!
|
||||||
|
//! The instance model is COCO-trained, and COCO is eighty classes of *things*.
|
||||||
|
//! There is no class for sky, none for foliage, none for water — the categories
|
||||||
|
//! a landscape is mostly made of. That gap is recorded in `models/LICENCE.md`
|
||||||
|
//! and it is why the scene model exists: ADE20K's 150 classes are a scene
|
||||||
|
//! parse, *stuff* included.
|
||||||
|
//!
|
||||||
|
//! Going the other way is just as impossible. A semantic model merges every
|
||||||
|
//! pixel of a class into one region, so it cannot tell three people apart, and
|
||||||
|
//! telling three people apart is exactly what clicking a subject needs. Neither
|
||||||
|
//! model substitutes for the other, which is why both ship.
|
||||||
|
//!
|
||||||
|
//! # The partition of unity, and why it is the point
|
||||||
|
//!
|
||||||
|
//! [`Scene::weight`] is not a mask per category that each independently says
|
||||||
|
//! yes or no. It is a *partition*: at every pixel the listed categories plus
|
||||||
|
//! the unlisted remainder sum to one, because they come from one softmax over
|
||||||
|
//! all 150 channels, summed within each category.
|
||||||
|
//!
|
||||||
|
//! That property is what makes feathering safe. Feather a hard label map
|
||||||
|
//! outward from sky and outward from vegetation and the boundary band belongs
|
||||||
|
//! to both, so a `+20` on sky and a `−10` on vegetation both land there and
|
||||||
|
//! every horizon acquires a visible seam. Feather a partition of unity and the
|
||||||
|
//! weights still sum to one — the band gets a blend of the two grades, which is
|
||||||
|
//! what a photographer drawing that boundary by hand would have painted.
|
||||||
|
//!
|
||||||
|
//! # Resolution, stated plainly
|
||||||
|
//!
|
||||||
|
//! The graph's logits are `[1, 150, 80, 80]`: an eighth of the input edge, and
|
||||||
|
//! that is the real spatial resolution of everything here. The stock export
|
||||||
|
//! ends with a `Resize` to 640×640 and an `ArgMax`, and
|
||||||
|
//! `tools/export-seg-model.sh` cuts both — the upsample adds no information and
|
||||||
|
//! the argmax destroys the per-class scores this module needs. [`Scene`] keeps
|
||||||
|
//! the native grid and resamples on demand ([`Scene::rasterise`]) so that the
|
||||||
|
//! coarseness is visible in the type rather than hidden behind an early
|
||||||
|
//! upsample.
|
||||||
|
//!
|
||||||
|
//! Practically: a graduated grade over sky or water is unbothered by 80×80. A
|
||||||
|
//! hard edge — a rooftop against sky at 100% zoom — will show it, and no
|
||||||
|
//! feather setting invents detail the model never had.
|
||||||
|
//!
|
||||||
|
//! # Cost
|
||||||
|
//!
|
||||||
|
//! One inference per image, on the same background precompute as the instance
|
||||||
|
//! pass and never on the frame path (ARCH §6.1). The scene tab's sliders read
|
||||||
|
//! [`Scene`] and re-run nothing.
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use ndarray::ArrayView3;
|
||||||
|
|
||||||
|
use crate::semantic::{install_backend, Letterbox, Window};
|
||||||
|
use crate::SegmentError;
|
||||||
|
|
||||||
|
/// Classes in the ADE20K vocabulary the scene model was trained on.
|
||||||
|
///
|
||||||
|
/// Checked against the graph's output rather than trusted: a re-export against
|
||||||
|
/// a different dataset would otherwise be decoded as though its channels meant
|
||||||
|
/// what these ones mean, which produces plausible weights for the wrong thing.
|
||||||
|
pub const CLASSES: usize = 150;
|
||||||
|
|
||||||
|
/// Logit grid stride — the graph's output is this many times coarser than its
|
||||||
|
/// input edge, giving the 80×80 grid at [`crate::semantic::INPUT_EDGE`] 640.
|
||||||
|
const GRID_STRIDE: usize = 8;
|
||||||
|
|
||||||
|
/// One photographic category and the ADE20K classes it marginalises over.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct Category {
|
||||||
|
pub name: Arc<str>,
|
||||||
|
/// Indices into the model's vocabulary. Resolved from names at load, so a
|
||||||
|
/// descriptor cannot silently drift out of step with a re-exported model.
|
||||||
|
pub classes: Vec<u16>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The scene model, and the categories it has been told to report.
|
||||||
|
pub struct SceneModel {
|
||||||
|
session: ort::session::Session,
|
||||||
|
categories: Vec<Category>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The weights that ship in `models/scene/` (AGPL — see `models/LICENCE.md`).
|
||||||
|
///
|
||||||
|
/// Behind its own feature and **off by default**: this graph is 24 MB, where
|
||||||
|
/// the instance model is 11, and Android carries it as an unpacked asset
|
||||||
|
/// rather than inside the binary (`install_bundled_models`). A desktop build
|
||||||
|
/// or a test that wants it compiled in opts in.
|
||||||
|
#[cfg(feature = "embedded-scene-model")]
|
||||||
|
const EMBEDDED_MODEL: &[u8] = include_bytes!("../../../models/scene/yolo26s-sem-ade20k.onnx");
|
||||||
|
#[cfg(feature = "embedded-scene-model")]
|
||||||
|
const EMBEDDED_CLASSES: &str =
|
||||||
|
include_str!("../../../models/scene/yolo26s-sem-ade20k.classes.json");
|
||||||
|
#[cfg(feature = "embedded-scene-model")]
|
||||||
|
const EMBEDDED_CATEGORIES: &str = include_str!("../../../models/scene/categories.txt");
|
||||||
|
|
||||||
|
impl SceneModel {
|
||||||
|
/// Load the scene model compiled into the binary.
|
||||||
|
#[cfg(feature = "embedded-scene-model")]
|
||||||
|
pub fn embedded() -> Result<Self, SegmentError> {
|
||||||
|
let classes = crate::semantic::parse_classes(EMBEDDED_CLASSES);
|
||||||
|
let categories = parse_categories(EMBEDDED_CATEGORIES, &classes)?;
|
||||||
|
Self::from_bytes(EMBEDDED_MODEL, categories)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Load from files on disk: the graph, its vocabulary, and the category
|
||||||
|
/// descriptor that groups the vocabulary into what the scene tab shows.
|
||||||
|
///
|
||||||
|
/// Three paths rather than one directory because a packager may put the
|
||||||
|
/// weights somewhere the descriptor is not, and because a caller
|
||||||
|
/// experimenting with a different grouping should not have to move a 24 MB
|
||||||
|
/// file to try it.
|
||||||
|
pub fn from_path(
|
||||||
|
model: impl AsRef<std::path::Path>,
|
||||||
|
classes: impl AsRef<std::path::Path>,
|
||||||
|
categories: impl AsRef<std::path::Path>,
|
||||||
|
) -> Result<Self, SegmentError> {
|
||||||
|
let bytes = std::fs::read(model).map_err(SegmentError::ModelRead)?;
|
||||||
|
let classes = std::fs::read_to_string(classes).map_err(SegmentError::ModelRead)?;
|
||||||
|
let categories = std::fs::read_to_string(categories).map_err(SegmentError::ModelRead)?;
|
||||||
|
let classes = crate::semantic::parse_classes(&classes);
|
||||||
|
let categories = parse_categories(&categories, &classes)?;
|
||||||
|
Self::from_bytes(&bytes, categories)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn from_bytes(bytes: &[u8], categories: Vec<Category>) -> Result<Self, SegmentError> {
|
||||||
|
install_backend();
|
||||||
|
|
||||||
|
let session = ort::session::Session::builder()
|
||||||
|
.map_err(SegmentError::Inference)?
|
||||||
|
.commit_from_memory(bytes)
|
||||||
|
.map_err(SegmentError::Inference)?;
|
||||||
|
|
||||||
|
Ok(Self {
|
||||||
|
session,
|
||||||
|
categories,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn categories(&self) -> &[Category] {
|
||||||
|
&self.categories
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Weigh every category over one image.
|
||||||
|
///
|
||||||
|
/// `rgb` is tightly packed `f32` RGB in `0.0..=1.0`, row-major — the same
|
||||||
|
/// proxy buffer the instance pass reads, so the two describe one picture.
|
||||||
|
///
|
||||||
|
/// One inference over the whole frame. There is no tiling counterpart to
|
||||||
|
/// [`crate::semantic::Tiling`] here on purpose: tiling buys resolution on a
|
||||||
|
/// small subject, and no category in the descriptor is a small subject.
|
||||||
|
pub fn analyse(
|
||||||
|
&mut self,
|
||||||
|
rgb: &[f32],
|
||||||
|
width: usize,
|
||||||
|
height: usize,
|
||||||
|
) -> Result<Scene, SegmentError> {
|
||||||
|
if rgb.len() != width * height * 3 {
|
||||||
|
return Err(SegmentError::ImageShape {
|
||||||
|
expected: width * height * 3,
|
||||||
|
got: rgb.len(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// Split the borrow: `run` needs the session mutably while
|
||||||
|
// `marginalise` needs the categories, and going through `self` for
|
||||||
|
// both at once is what the borrow checker objects to.
|
||||||
|
let Self {
|
||||||
|
session,
|
||||||
|
categories,
|
||||||
|
} = self;
|
||||||
|
|
||||||
|
let window = Window {
|
||||||
|
x: 0.0,
|
||||||
|
y: 0.0,
|
||||||
|
w: width as f32,
|
||||||
|
h: height as f32,
|
||||||
|
};
|
||||||
|
let letterbox = Letterbox::fit(window.w, window.h);
|
||||||
|
let input = letterbox.sample(rgb, width, height, &window);
|
||||||
|
|
||||||
|
let outputs = session
|
||||||
|
.run(ort::inputs![
|
||||||
|
ort::value::Tensor::from_array(input).map_err(SegmentError::Inference)?
|
||||||
|
])
|
||||||
|
.map_err(SegmentError::Inference)?;
|
||||||
|
|
||||||
|
let (shape, logits) = outputs[0]
|
||||||
|
.try_extract_tensor::<f32>()
|
||||||
|
.map_err(|_| SegmentError::OutputShape("logits"))?;
|
||||||
|
|
||||||
|
// `[1, 150, gh, gw]`. Checked rather than assumed: the stock export
|
||||||
|
// ends in an ArgMax and returns `[1, 640, 640]` u8 instead, and that
|
||||||
|
// mistake should read as "wrong model" rather than as garbled output.
|
||||||
|
if shape.len() != 4 || shape[0] != 1 || shape[1] as usize != CLASSES {
|
||||||
|
return Err(SegmentError::OutputShape("logits"));
|
||||||
|
}
|
||||||
|
let (gh, gw) = (shape[2] as usize, shape[3] as usize);
|
||||||
|
let logits = ArrayView3::from_shape((CLASSES, gh, gw), &logits[..CLASSES * gh * gw])
|
||||||
|
.map_err(|_| SegmentError::OutputShape("logits"))?;
|
||||||
|
|
||||||
|
Ok(marginalise(categories, logits, gw, gh, letterbox, window))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Softmax over the vocabulary, then sum within each category.
|
||||||
|
///
|
||||||
|
/// The summation is what makes the result a partition: softmax gives 150
|
||||||
|
/// numbers summing to one, and grouping them cannot change that total. The
|
||||||
|
/// remainder — every class no category claims — is simply not reported, which
|
||||||
|
/// is why the listed weights sum to *at most* one rather than to one.
|
||||||
|
///
|
||||||
|
/// Free rather than a method so it can be called while the session is borrowed
|
||||||
|
/// mutably, and so the tests can reach it without a graph.
|
||||||
|
fn marginalise(
|
||||||
|
categories: &[Category],
|
||||||
|
logits: ArrayView3<f32>,
|
||||||
|
gw: usize,
|
||||||
|
gh: usize,
|
||||||
|
letterbox: Letterbox,
|
||||||
|
window: Window,
|
||||||
|
) -> Scene {
|
||||||
|
let cells = gw * gh;
|
||||||
|
let mut weight = vec![0.0f32; categories.len() * cells];
|
||||||
|
let mut probability = vec![0.0f32; CLASSES];
|
||||||
|
|
||||||
|
for cell in 0..cells {
|
||||||
|
let (y, x) = (cell / gw, cell % gw);
|
||||||
|
|
||||||
|
// Shift by the maximum before exponentiating. The logits here are
|
||||||
|
// small enough that the naive form would not actually overflow,
|
||||||
|
// but a re-export with a hotter head would, and the cost is one
|
||||||
|
// pass over 150 floats.
|
||||||
|
let mut peak = f32::NEG_INFINITY;
|
||||||
|
for c in 0..CLASSES {
|
||||||
|
peak = peak.max(logits[[c, y, x]]);
|
||||||
|
}
|
||||||
|
let mut total = 0.0f32;
|
||||||
|
for c in 0..CLASSES {
|
||||||
|
let p = (logits[[c, y, x]] - peak).exp();
|
||||||
|
probability[c] = p;
|
||||||
|
total += p;
|
||||||
|
}
|
||||||
|
let norm = if total > 0.0 { 1.0 / total } else { 0.0 };
|
||||||
|
|
||||||
|
for (k, category) in categories.iter().enumerate() {
|
||||||
|
let mut sum = 0.0f32;
|
||||||
|
for &class in &category.classes {
|
||||||
|
sum += probability[class as usize];
|
||||||
|
}
|
||||||
|
weight[k * cells + cell] = sum * norm;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Scene {
|
||||||
|
names: categories.iter().map(|c| c.name.clone()).collect(),
|
||||||
|
weight,
|
||||||
|
grid_width: gw,
|
||||||
|
grid_height: gh,
|
||||||
|
letterbox,
|
||||||
|
window,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One image's category weights, at the model's own resolution.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct Scene {
|
||||||
|
names: Vec<Arc<str>>,
|
||||||
|
/// `[category][y * grid_width + x]`, each in `0.0..=1.0`, and across
|
||||||
|
/// categories summing to at most one at every cell.
|
||||||
|
weight: Vec<f32>,
|
||||||
|
grid_width: usize,
|
||||||
|
grid_height: usize,
|
||||||
|
letterbox: Letterbox,
|
||||||
|
window: Window,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Scene {
|
||||||
|
pub fn categories(&self) -> &[Arc<str>] {
|
||||||
|
&self.names
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn grid_size(&self) -> (usize, usize) {
|
||||||
|
(self.grid_width, self.grid_height)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One category's weights over the logit grid.
|
||||||
|
pub fn weight(&self, category: usize) -> Option<&[f32]> {
|
||||||
|
let cells = self.grid_width * self.grid_height;
|
||||||
|
self.weight.get(category * cells..(category + 1) * cells)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn index_of(&self, name: &str) -> Option<usize> {
|
||||||
|
self.names.iter().position(|n| &**n == name)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How much of the frame this category covers, `0.0..=1.0`.
|
||||||
|
///
|
||||||
|
/// Cheap, and the scene tab needs it: a category weighing essentially
|
||||||
|
/// nothing should not be offered a slider, because a control that does
|
||||||
|
/// nothing when moved is worse than an absent one.
|
||||||
|
pub fn coverage(&self, category: usize) -> f32 {
|
||||||
|
match self.weight(category) {
|
||||||
|
Some(w) if !w.is_empty() => w.iter().sum::<f32>() / w.len() as f32,
|
||||||
|
_ => 0.0,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resample one category to source-image resolution.
|
||||||
|
///
|
||||||
|
/// Bilinear over the logit grid. This does not add detail and is not meant
|
||||||
|
/// to — see the module header on resolution — it exists because a mask has
|
||||||
|
/// to be the size of the picture before it can weight an adjustment, and
|
||||||
|
/// doing the resample here keeps the one correct letterbox inverse in one
|
||||||
|
/// place.
|
||||||
|
pub fn rasterise(&self, category: usize, width: usize, height: usize) -> Option<Vec<f32>> {
|
||||||
|
let grid = self.weight(category)?;
|
||||||
|
let mut out = vec![0.0f32; width * height];
|
||||||
|
|
||||||
|
for y in 0..height {
|
||||||
|
for x in 0..width {
|
||||||
|
let (gx, gy) = self.letterbox.to_grid(
|
||||||
|
x as f32 + 0.5,
|
||||||
|
y as f32 + 0.5,
|
||||||
|
&self.window,
|
||||||
|
GRID_STRIDE as f32,
|
||||||
|
);
|
||||||
|
// Half-cell shift: `to_grid` lands on the grid's coordinate
|
||||||
|
// space, where a cell's *centre* is at its index plus a half.
|
||||||
|
let (gx, gy) = (gx - 0.5, gy - 0.5);
|
||||||
|
let x0 = gx.floor();
|
||||||
|
let y0 = gy.floor();
|
||||||
|
let (fx, fy) = (gx - x0, gy - y0);
|
||||||
|
let x0 = (x0 as isize).clamp(0, self.grid_width as isize - 1) as usize;
|
||||||
|
let y0 = (y0 as isize).clamp(0, self.grid_height as isize - 1) as usize;
|
||||||
|
let x1 = (x0 + 1).min(self.grid_width - 1);
|
||||||
|
let y1 = (y0 + 1).min(self.grid_height - 1);
|
||||||
|
|
||||||
|
let at = |gx: usize, gy: usize| grid[gy * self.grid_width + gx];
|
||||||
|
let top = at(x0, y0) * (1.0 - fx) + at(x1, y0) * fx;
|
||||||
|
let bot = at(x0, y1) * (1.0 - fx) + at(x1, y1) * fx;
|
||||||
|
out[y * width + x] = top * (1.0 - fy) + bot * fy;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Some(out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read `models/scene/categories.txt`, resolving class names to indices.
|
||||||
|
///
|
||||||
|
/// Hand-written rather than generated, unlike the `.classes.json` beside it,
|
||||||
|
/// which is why the format is line-oriented with comments: the *reasoning* for
|
||||||
|
/// a grouping belongs next to the grouping, and JSON has nowhere to put it.
|
||||||
|
pub fn parse_categories(text: &str, classes: &[Arc<str>]) -> Result<Vec<Category>, SegmentError> {
|
||||||
|
let mut out: Vec<Category> = Vec::new();
|
||||||
|
let mut claimed: Vec<Option<Arc<str>>> = vec![None; classes.len()];
|
||||||
|
|
||||||
|
for line in text.lines() {
|
||||||
|
let line = line.split('#').next().unwrap_or("").trim();
|
||||||
|
if line.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let Some((name, members)) = line.split_once('=') else {
|
||||||
|
return Err(SegmentError::CategoryDescriptor(format!(
|
||||||
|
"line is not `name = class, class, ...`: {line}"
|
||||||
|
)));
|
||||||
|
};
|
||||||
|
let name: Arc<str> = name.trim().into();
|
||||||
|
|
||||||
|
let mut indices = Vec::new();
|
||||||
|
for member in members.split(',') {
|
||||||
|
let member = member.trim();
|
||||||
|
if member.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let Some(index) = classes.iter().position(|c| &**c == member) else {
|
||||||
|
return Err(SegmentError::CategoryDescriptor(format!(
|
||||||
|
"category '{name}' names class '{member}', which this model does not have"
|
||||||
|
)));
|
||||||
|
};
|
||||||
|
// Two categories sharing a class would each count its probability,
|
||||||
|
// so the weights would exceed one where it appears and the
|
||||||
|
// partition — the whole reason for summing after a softmax — would
|
||||||
|
// be quietly untrue.
|
||||||
|
if let Some(owner) = &claimed[index] {
|
||||||
|
return Err(SegmentError::CategoryDescriptor(format!(
|
||||||
|
"class '{member}' is claimed by both '{owner}' and '{name}'"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
claimed[index] = Some(name.clone());
|
||||||
|
indices.push(index as u16);
|
||||||
|
}
|
||||||
|
|
||||||
|
if indices.is_empty() {
|
||||||
|
return Err(SegmentError::CategoryDescriptor(format!(
|
||||||
|
"category '{name}' lists no classes"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
out.push(Category {
|
||||||
|
name,
|
||||||
|
classes: indices,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
if out.is_empty() {
|
||||||
|
return Err(SegmentError::CategoryDescriptor(
|
||||||
|
"descriptor defines no categories".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn vocabulary() -> Vec<Arc<str>> {
|
||||||
|
["sky", "tree", "grass", "person", "wall"]
|
||||||
|
.iter()
|
||||||
|
.map(|s| Arc::from(*s))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn descriptor_resolves_names_to_indices() {
|
||||||
|
let v = vocabulary();
|
||||||
|
let cats = parse_categories("sky = sky\nvegetation = tree, grass\n", &v).unwrap();
|
||||||
|
assert_eq!(cats.len(), 2);
|
||||||
|
assert_eq!(&*cats[0].name, "sky");
|
||||||
|
assert_eq!(cats[0].classes, vec![0]);
|
||||||
|
assert_eq!(cats[1].classes, vec![1, 2]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn comments_and_blank_lines_are_ignored() {
|
||||||
|
let v = vocabulary();
|
||||||
|
let cats = parse_categories("# a note\n\nsky = sky # trailing\n", &v).unwrap();
|
||||||
|
assert_eq!(cats.len(), 1);
|
||||||
|
assert_eq!(cats[0].classes, vec![0]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_unknown_class_is_refused() {
|
||||||
|
let v = vocabulary();
|
||||||
|
let e = parse_categories("sky = cloud\n", &v).unwrap_err();
|
||||||
|
assert!(format!("{e}").contains("cloud"), "{e}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The partition is the module's one load-bearing property, so the
|
||||||
|
/// descriptor is not allowed to break it before inference even runs.
|
||||||
|
#[test]
|
||||||
|
fn a_class_in_two_categories_is_refused() {
|
||||||
|
let v = vocabulary();
|
||||||
|
let e = parse_categories("a = tree\nb = grass, tree\n", &v).unwrap_err();
|
||||||
|
assert!(format!("{e}").contains("claimed by both"), "{e}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_shipped_descriptor_matches_the_shipped_vocabulary() {
|
||||||
|
let classes = crate::semantic::parse_classes(include_str!(
|
||||||
|
"../../../models/scene/yolo26s-sem-ade20k.classes.json"
|
||||||
|
));
|
||||||
|
assert_eq!(classes.len(), CLASSES);
|
||||||
|
let cats = parse_categories(
|
||||||
|
include_str!("../../../models/scene/categories.txt"),
|
||||||
|
&classes,
|
||||||
|
)
|
||||||
|
.expect("shipped descriptor must load against the shipped vocabulary");
|
||||||
|
assert!(cats.iter().any(|c| &*c.name == "sky"));
|
||||||
|
assert!(cats.iter().any(|c| &*c.name == "vegetation"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Softmax then group: the reported weights must never exceed one, and
|
||||||
|
/// must equal one exactly when the categories name every class.
|
||||||
|
#[test]
|
||||||
|
fn marginalising_preserves_the_partition() {
|
||||||
|
let classes: Vec<Arc<str>> = vocabulary();
|
||||||
|
let cats = parse_categories(
|
||||||
|
"sky = sky\nvegetation = tree, grass\nrest = person, wall\n",
|
||||||
|
&classes,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
// Hand-rolled rather than run through a graph: this test is about the
|
||||||
|
// arithmetic, and a model would only make it slower and less certain.
|
||||||
|
let (gw, gh) = (2usize, 2usize);
|
||||||
|
let mut logits = vec![0.0f32; classes.len() * gw * gh];
|
||||||
|
for (i, v) in logits.iter_mut().enumerate() {
|
||||||
|
*v = (i % 7) as f32 * 0.3;
|
||||||
|
}
|
||||||
|
let view = ArrayView3::from_shape((classes.len(), gh, gw), &logits).unwrap();
|
||||||
|
|
||||||
|
// `marginalise` is a method for access to `self.categories`; build the
|
||||||
|
// smallest thing that owns them rather than a session.
|
||||||
|
let cells = gw * gh;
|
||||||
|
let mut weight = vec![0.0f32; cats.len() * cells];
|
||||||
|
for cell in 0..cells {
|
||||||
|
let (y, x) = (cell / gw, cell % gw);
|
||||||
|
let peak = (0..classes.len()).fold(f32::NEG_INFINITY, |m, c| m.max(view[[c, y, x]]));
|
||||||
|
let p: Vec<f32> = (0..classes.len())
|
||||||
|
.map(|c| (view[[c, y, x]] - peak).exp())
|
||||||
|
.collect();
|
||||||
|
let total: f32 = p.iter().sum();
|
||||||
|
for (k, category) in cats.iter().enumerate() {
|
||||||
|
let s: f32 = category.classes.iter().map(|&c| p[c as usize]).sum();
|
||||||
|
weight[k * cells + cell] = s / total;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for cell in 0..cells {
|
||||||
|
let sum: f32 = (0..cats.len()).map(|k| weight[k * cells + cell]).sum();
|
||||||
|
assert!(
|
||||||
|
(sum - 1.0).abs() < 1e-5,
|
||||||
|
"categories covering every class must sum to 1, got {sum}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -20,7 +20,7 @@
|
|||||||
//! behaviour, and it is worth being glad of rather than working around.
|
//! behaviour, and it is worth being glad of rather than working around.
|
||||||
//!
|
//!
|
||||||
//! So arm B here contributes *subjects*, and the watershed contributes
|
//! So arm B here contributes *subjects*, and the watershed contributes
|
||||||
//! everything else. See `models/LICENCE.md` for why no ADE20K variant is
|
//! everything else. See `models/LICENCE.md` at the repository root for why no ADE20K variant is
|
||||||
//! shipped instead.
|
//! shipped instead.
|
||||||
//!
|
//!
|
||||||
//! # Cost, and where it may run
|
//! # Cost, and where it may run
|
||||||
@@ -198,15 +198,15 @@ pub struct SemanticModel {
|
|||||||
classes: Vec<Arc<str>>,
|
classes: Vec<Arc<str>>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The weights that ship with this crate (`models/`, AGPL — see LICENCE.md).
|
/// The weights, from the repository-root `models/segment/` (AGPL — see `models/LICENCE.md`).
|
||||||
///
|
///
|
||||||
/// Embedded rather than read from a path because Android hands the app no
|
/// Embedded rather than read from a path because Android hands the app no
|
||||||
/// filesystem location to read from (ARCH §6.9) — the same reasoning that has
|
/// filesystem location to read from (ARCH §6.9) — the same reasoning that has
|
||||||
/// the Lensfun database shipping inside its crate.
|
/// the Lensfun database shipping inside its crate.
|
||||||
#[cfg(feature = "embedded-model")]
|
#[cfg(feature = "embedded-model")]
|
||||||
const EMBEDDED_MODEL: &[u8] = include_bytes!("../models/yolo26n-seg.onnx");
|
const EMBEDDED_MODEL: &[u8] = include_bytes!("../../../models/segment/yolo26n-seg.onnx");
|
||||||
#[cfg(feature = "embedded-model")]
|
#[cfg(feature = "embedded-model")]
|
||||||
const EMBEDDED_CLASSES: &str = include_str!("../models/yolo26n-seg.classes.json");
|
const EMBEDDED_CLASSES: &str = include_str!("../../../models/segment/yolo26n-seg.classes.json");
|
||||||
|
|
||||||
impl SemanticModel {
|
impl SemanticModel {
|
||||||
/// Load the model that ships with this crate.
|
/// Load the model that ships with this crate.
|
||||||
@@ -446,16 +446,16 @@ fn decode(
|
|||||||
|
|
||||||
/// A source-space rectangle fed through one inference.
|
/// A source-space rectangle fed through one inference.
|
||||||
#[derive(Debug, Clone, Copy)]
|
#[derive(Debug, Clone, Copy)]
|
||||||
struct Window {
|
pub(crate) struct Window {
|
||||||
x: f32,
|
pub(crate) x: f32,
|
||||||
y: f32,
|
pub(crate) y: f32,
|
||||||
w: f32,
|
pub(crate) w: f32,
|
||||||
h: f32,
|
pub(crate) h: f32,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The scale-and-pad that fits an arbitrary rectangle into the square input.
|
/// The scale-and-pad that fits an arbitrary rectangle into the square input.
|
||||||
#[derive(Debug, Clone, Copy)]
|
#[derive(Debug, Clone, Copy)]
|
||||||
struct Letterbox {
|
pub(crate) struct Letterbox {
|
||||||
/// Input pixels per source pixel.
|
/// Input pixels per source pixel.
|
||||||
scale: f32,
|
scale: f32,
|
||||||
pad_x: f32,
|
pad_x: f32,
|
||||||
@@ -463,7 +463,7 @@ struct Letterbox {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl Letterbox {
|
impl Letterbox {
|
||||||
fn fit(w: f32, h: f32) -> Self {
|
pub(crate) fn fit(w: f32, h: f32) -> Self {
|
||||||
let scale = (INPUT_EDGE as f32 / w).min(INPUT_EDGE as f32 / h);
|
let scale = (INPUT_EDGE as f32 / w).min(INPUT_EDGE as f32 / h);
|
||||||
Self {
|
Self {
|
||||||
scale,
|
scale,
|
||||||
@@ -477,7 +477,13 @@ impl Letterbox {
|
|||||||
/// Bilinear, and grey (`0.5`) in the padding — the value the network sees
|
/// Bilinear, and grey (`0.5`) in the padding — the value the network sees
|
||||||
/// least as an edge, where black would draw a hard border across the frame
|
/// least as an edge, where black would draw a hard border across the frame
|
||||||
/// and invite a detection along it.
|
/// and invite a detection along it.
|
||||||
fn sample(&self, rgb: &[f32], width: usize, height: usize, window: &Window) -> Array4<f32> {
|
pub(crate) fn sample(
|
||||||
|
&self,
|
||||||
|
rgb: &[f32],
|
||||||
|
width: usize,
|
||||||
|
height: usize,
|
||||||
|
window: &Window,
|
||||||
|
) -> Array4<f32> {
|
||||||
let mut input = Array4::<f32>::from_elem((1, 3, INPUT_EDGE, INPUT_EDGE), 0.5);
|
let mut input = Array4::<f32>::from_elem((1, 3, INPUT_EDGE, INPUT_EDGE), 0.5);
|
||||||
|
|
||||||
for iy in 0..INPUT_EDGE {
|
for iy in 0..INPUT_EDGE {
|
||||||
@@ -519,12 +525,23 @@ impl Letterbox {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Source pixel to the coordinates of an output grid `stride` times
|
||||||
|
/// coarser than the graph's input.
|
||||||
|
///
|
||||||
|
/// Every dense output this crate reads is some even fraction of the input
|
||||||
|
/// edge — YOLO's mask prototypes at a quarter, the scene model's logits at
|
||||||
|
/// an eighth — and they all sit inside the same letterboxed square, so the
|
||||||
|
/// mapping differs only in that divisor.
|
||||||
|
pub(crate) fn to_grid(self, sx: f32, sy: f32, w: &Window, stride: f32) -> (f32, f32) {
|
||||||
|
(
|
||||||
|
((sx - w.x) * self.scale + self.pad_x) / stride,
|
||||||
|
((sy - w.y) * self.scale + self.pad_y) / stride,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
/// Source pixel to prototype-grid coordinates.
|
/// Source pixel to prototype-grid coordinates.
|
||||||
fn to_proto(self, sx: f32, sy: f32, w: &Window) -> (f32, f32) {
|
fn to_proto(self, sx: f32, sy: f32, w: &Window) -> (f32, f32) {
|
||||||
(
|
self.to_grid(sx, sy, w, PROTO_STRIDE as f32)
|
||||||
((sx - w.x) * self.scale + self.pad_x) / PROTO_STRIDE as f32,
|
|
||||||
((sy - w.y) * self.scale + self.pad_y) / PROTO_STRIDE as f32,
|
|
||||||
)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -648,7 +665,7 @@ fn steps(extent: f32, edge: f32, stride: f32) -> usize {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Point `ort` at tract, exactly once per process.
|
/// Point `ort` at tract, exactly once per process.
|
||||||
fn install_backend() {
|
pub(crate) fn install_backend() {
|
||||||
use std::sync::Once;
|
use std::sync::Once;
|
||||||
static ONCE: Once = Once::new();
|
static ONCE: Once = Once::new();
|
||||||
ONCE.call_once(|| {
|
ONCE.call_once(|| {
|
||||||
|
|||||||
@@ -255,25 +255,37 @@ fi
|
|||||||
cp "${SO}" "${OUT}/staging/lib/${ABI}/libdarkroom.so"
|
cp "${SO}" "${OUT}/staging/lib/${ABI}/libdarkroom.so"
|
||||||
cp "${DEX}" "${OUT}/staging/classes.dex"
|
cp "${DEX}" "${OUT}/staging/classes.dex"
|
||||||
|
|
||||||
# The face models. Android has no other route to one — app-private storage is
|
# The models. Android has no other route to one — app-private storage is not
|
||||||
# not user-reachable and the in-app fetch is unbuilt (docs/faces.md §2.2a) — so
|
# user-reachable and the in-app fetch is unbuilt (docs/faces.md §2.2a) — so
|
||||||
# they go in the APK and `android_main` unpacks them on first launch. The
|
# they go in the APK and `android_main` unpacks them on first launch. The
|
||||||
# source is `models/face/`, shared with the Arch package rather than living
|
# sources are `models/face/` and `models/scene/`, shared with the Arch package
|
||||||
# under this one platform's directory.
|
# rather than living under this one platform's directory.
|
||||||
|
#
|
||||||
|
# Two directories, and they are not the same kind of thing. The face weights
|
||||||
|
# are absent from most checkouts by design (research-only grant), so finding
|
||||||
|
# none is ordinary. The scene model is committed, so finding none means a
|
||||||
|
# broken checkout — but this script still only warns, because the failure it
|
||||||
|
# would otherwise cause is at APK build time on a machine that may legitimately
|
||||||
|
# be building the face-less variant.
|
||||||
#
|
#
|
||||||
# Through the staging directory rather than aapt2's `-A`: the .so and the dex
|
# Through the staging directory rather than aapt2's `-A`: the .so and the dex
|
||||||
# already go in with `zip` below, and one mechanism for "extra files in the
|
# already go in with `zip` below, and one mechanism for "extra files in the
|
||||||
# APK" is easier to follow than two.
|
# APK" is easier to follow than two.
|
||||||
ASSETS="${REPO}/models/face"
|
#
|
||||||
# Cleared first: a previous run that died between staging and cleanup would
|
# Cleared first: a previous run that died between staging and cleanup would
|
||||||
# otherwise leave models in the APK that are no longer in the tree.
|
# otherwise leave models in the APK that are no longer in the tree.
|
||||||
rm -rf "${OUT}/staging/assets"
|
rm -rf "${OUT}/staging/assets"
|
||||||
if compgen -G "${ASSETS}/*.onnx" >/dev/null; then
|
mkdir -p "${OUT}/staging/assets/models"
|
||||||
|
_bundled=""
|
||||||
|
for _dir in face scene; do
|
||||||
|
ASSETS="${REPO}/models/${_dir}"
|
||||||
|
compgen -G "${ASSETS}/*.onnx" >/dev/null || continue
|
||||||
# An LFS pointer is ~130 bytes and looks exactly like a model to `cp`. Left
|
# An LFS pointer is ~130 bytes and looks exactly like a model to `cp`. Left
|
||||||
# unchecked it reaches the device and fails inside tract, which reports a
|
# unchecked it reaches the device and fails inside tract, which reports a
|
||||||
# broken graph rather than a clone that needs `git lfs pull`. Same guard
|
# broken graph rather than a clone that needs `git lfs pull`. Same guard
|
||||||
# dr-segment's build script applies to yolo26n-seg.onnx, and the same
|
# dr-segment's build script applies to yolo26n-seg.onnx, and the same
|
||||||
# reason.
|
# reason. Only the weights are checked: the vocabulary and the category
|
||||||
|
# descriptor beside them are legitimately a few kilobytes.
|
||||||
for m in "${ASSETS}"/*.onnx; do
|
for m in "${ASSETS}"/*.onnx; do
|
||||||
if [[ "$(stat -c%s "${m}")" -lt 100000 ]]; then
|
if [[ "$(stat -c%s "${m}")" -lt 100000 ]]; then
|
||||||
echo "error: $(basename "${m}") is $(stat -c%s "${m}") bytes — an LFS pointer, not a model." >&2
|
echo "error: $(basename "${m}") is $(stat -c%s "${m}") bytes — an LFS pointer, not a model." >&2
|
||||||
@@ -281,11 +293,21 @@ if compgen -G "${ASSETS}/*.onnx" >/dev/null; then
|
|||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
done
|
done
|
||||||
mkdir -p "${OUT}/staging/assets/models"
|
# The scene model is three files: the graph, its vocabulary, and the
|
||||||
cp "${ASSETS}"/*.onnx "${OUT}/staging/assets/models/"
|
# category descriptor. All three are needed to decode anything, so they
|
||||||
echo " assets: $(ls "${ASSETS}" | grep '\.onnx$' | tr '\n' ' ')"
|
# travel together; README.md is documentation and stays out of the APK.
|
||||||
|
for f in "${ASSETS}"/*; do
|
||||||
|
case "$(basename "${f}")" in
|
||||||
|
README.md) continue ;;
|
||||||
|
esac
|
||||||
|
cp "${f}" "${OUT}/staging/assets/models/"
|
||||||
|
_bundled="${_bundled} $(basename "${f}")"
|
||||||
|
done
|
||||||
|
done
|
||||||
|
if [[ -n "${_bundled}" ]]; then
|
||||||
|
echo " assets:${_bundled}"
|
||||||
else
|
else
|
||||||
echo " assets: no face models found (face indexing will be off on the device)"
|
echo " assets: no models found (face indexing and the scene tab will be off on the device)"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# -0 "" stores the .so without compression so Android can mmap it directly
|
# -0 "" stores the .so without compression so Android can mmap it directly
|
||||||
|
|||||||
@@ -435,3 +435,69 @@ dependency detail (`core/dr-segment/models/LICENCE.md`).
|
|||||||
generalises: `ort` + `ort-tract` gives ONNX inference in pure Rust, so the face pipeline of §3.9.1
|
generalises: `ort` + `ort-tract` gives ONNX inference in pure Rust, so the face pipeline of §3.9.1
|
||||||
needs no C dependency either. The *model licensing* half of D13 is untouched — the InsightFace
|
needs no C dependency either. The *model licensing* half of D13 is untouched — the InsightFace
|
||||||
weights are still non-commercial and still unusable here.
|
weights are still non-commercial and still unusable here.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 16. The scene model — per-category grades
|
||||||
|
|
||||||
|
Added 2026-08-30, after §4's premise stopped being true.
|
||||||
|
|
||||||
|
### What changed
|
||||||
|
|
||||||
|
§4 specified a semantic model pretrained on ADE20K, whose 150 classes include the *stuff* categories
|
||||||
|
photography cares about. §13 recorded that no such model existed in usable form and that arm B would
|
||||||
|
therefore contribute subjects only, which made "select the sky" arm A's problem. Re-checked
|
||||||
|
2026-08-30: **Ultralytics now ships a `semantic` task with ADE20K checkpoints**
|
||||||
|
(`docs.ultralytics.com/tasks/semantic`). `yolo26s-sem-ade20k` is in `models/scene/`.
|
||||||
|
|
||||||
|
### It is an addition, not a correction to arm B
|
||||||
|
|
||||||
|
The instance model stays exactly where it was, and the reason is the one §13 already gave and was
|
||||||
|
right about: a semantic model merges every pixel of a class into one region, so it cannot separate
|
||||||
|
two people, and separating two people is what clicking a subject requires. Swapping arm B for this
|
||||||
|
would regress the primary interaction to fix a secondary one.
|
||||||
|
|
||||||
|
So the two divide by *what the user is doing*, not by which is better:
|
||||||
|
|
||||||
|
| | `models/segment/` (COCO instances) | `models/scene/` (ADE20K semantics) |
|
||||||
|
|---|---|---|
|
||||||
|
| Question | which pixels are *that* dog | how much of this pixel is sky |
|
||||||
|
| Granularity | per instance | per category, whole frame |
|
||||||
|
| Drives | local adjustments, subject selection | the scene tab's per-category sliders |
|
||||||
|
| Vocabulary | 80 things | 150 classes, stuff included |
|
||||||
|
|
||||||
|
### The export is truncated, and both reasons matter
|
||||||
|
|
||||||
|
Ultralytics ends the graph with `Resize → ArgMax → Cast`, returning a `[1, 640, 640]` u8 label map.
|
||||||
|
`tools/export-seg-model.sh` cuts that tail and ships the classifier's `[1, 150, 80, 80]` f32 logits.
|
||||||
|
|
||||||
|
**Cost.** The `Resize` materialises 150 × 640 × 640 × f32 — 246 MB — and the `ArgMax` then reduces
|
||||||
|
across the channel axis, striding 409,600 elements per comparison. Measured under load it was
|
||||||
|
roughly four fifths of total runtime, spent on work the application discards.
|
||||||
|
|
||||||
|
**Softness, which is the more important one.** `ArgMax` destroys the per-class scores, and the whole
|
||||||
|
design of the scene tab rests on keeping them. Softmax over the 150 channels, summed within each
|
||||||
|
category, produces per-category weights that sum to one at every pixel — a partition of unity.
|
||||||
|
Feathering that cannot double-grade a boundary. Feathering *hard labels* outward from two adjacent
|
||||||
|
categories paints both grades into the overlap, and every horizon in the frame acquires a seam.
|
||||||
|
|
||||||
|
### The resolution is 80×80, and no setting changes that
|
||||||
|
|
||||||
|
The discarded upsample was never information. `Scene` keeps the native grid and resamples on demand,
|
||||||
|
so the coarseness is visible in the type rather than hidden. A graduated grade over sky or water is
|
||||||
|
untroubled by it; a rooftop against sky at 100% zoom will show it. This is the constraint most likely
|
||||||
|
to decide whether the tab feels good, and it is not addressable by choosing a larger checkpoint —
|
||||||
|
`yolo26n-sem` and `yolo26s-sem` have the same output grid.
|
||||||
|
|
||||||
|
### Licence
|
||||||
|
|
||||||
|
Unchanged. Same AGPL-3.0 grant as the instance model, same GPLv3 §13 permission, same consequence
|
||||||
|
already accepted in D14 — so this needed no new licence decision, which is most of why it was cheap.
|
||||||
|
See `models/LICENCE.md`.
|
||||||
|
|
||||||
|
### Measurement
|
||||||
|
|
||||||
|
Timings taken while this was chosen came off a laptop compiling other things and are upper bounds
|
||||||
|
only. `cargo run -p dr-segment --example scene --release --features embedded-scene-model` reports a
|
||||||
|
median over N runs with the first excluded; a number worth quoting should come from that, on an idle
|
||||||
|
machine.
|
||||||
|
|||||||
+42
-42
File diff suppressed because one or more lines are too long
@@ -0,0 +1,71 @@
|
|||||||
|
# Model weights — licensing
|
||||||
|
|
||||||
|
Two Ultralytics checkpoints ship here, both exported by
|
||||||
|
`tools/export-seg-model.sh`, each with its class vocabulary written out by the
|
||||||
|
same script:
|
||||||
|
|
||||||
|
| File | Checkpoint | Trained on | Used by |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `segment/yolo26n-seg.onnx` | `yolo26n-seg.pt` | COCO, 80 *thing* classes | local adjustments, subject selection |
|
||||||
|
| `scene/yolo26s-sem-ade20k.onnx` | `yolo26s-sem-ade20k.pt` | ADE20K, 150 classes | the scene tab's per-category grades |
|
||||||
|
|
||||||
|
Both come from `https://huggingface.co/Ultralytics/YOLO26`. The face weights in
|
||||||
|
`face/` are a separate matter with a separate grant — see `face/README.md`.
|
||||||
|
|
||||||
|
## The grant
|
||||||
|
|
||||||
|
**Ultralytics releases YOLO under AGPL-3.0**, and the weights carry the same
|
||||||
|
grant as the framework — the HuggingFace repository declares `agpl-3.0` for the
|
||||||
|
checkpoints themselves, not merely for the training code. A commercial licence
|
||||||
|
is offered separately; DarkRoom does not use it and does not need it.
|
||||||
|
|
||||||
|
## What that means for DarkRoom
|
||||||
|
|
||||||
|
DarkRoom is GPL-3.0-or-later. **GPLv3 §13 explicitly permits combination with
|
||||||
|
AGPL-3.0 code**, so redistributing these weights inside this repository is
|
||||||
|
allowed — this is *not* the situation the InsightFace "buffalo" weights would
|
||||||
|
have created, where a non-commercial research grant is simply incompatible with
|
||||||
|
the project's licence and with F-Droid, Flatpak and Play distribution
|
||||||
|
(NFR-COMPAT-2, D13).
|
||||||
|
|
||||||
|
The consequence, and it is a real one: **the combined work is effectively
|
||||||
|
AGPL-3.0.** §13's permission runs one way — the AGPL's §13 network-use condition
|
||||||
|
attaches to the portion under that licence. For a local-first desktop and
|
||||||
|
Android photo editor that condition has no practical bite, because there is no
|
||||||
|
network service offering the combined work to remote users. It would acquire
|
||||||
|
bite the moment any hosted or server-side rendering appeared, and that is the
|
||||||
|
thing to remember rather than rediscover.
|
||||||
|
|
||||||
|
This was decided deliberately (D14), not arrived at by accident, and
|
||||||
|
`docs/segmentation.md` §7 records the reasoning.
|
||||||
|
|
||||||
|
## Class vocabulary — a caveat worth reading
|
||||||
|
|
||||||
|
`docs/segmentation.md` §4 specified YOLO **pretrained on ADE20K**, whose 150
|
||||||
|
classes include the *stuff* categories that matter most in photography — sky,
|
||||||
|
vegetation, water, wall, mountain.
|
||||||
|
|
||||||
|
**This was true when written and is not any more.** Checked 2026-08-21, no
|
||||||
|
YOLO/ADE20K combination existed: Ultralytics shipped YOLO26-seg on **COCO**
|
||||||
|
only, and the one HuggingFace repository claiming otherwise
|
||||||
|
(`laxmacl/yolov8-ade20k`) was empty. Re-checked 2026-08-30: Ultralytics now
|
||||||
|
ships a `semantic` task with ADE20K checkpoints
|
||||||
|
(`https://docs.ultralytics.com/tasks/semantic`), and `yolo26s-sem-ade20k` is
|
||||||
|
what `scene/` holds.
|
||||||
|
|
||||||
|
So the two vocabularies divide the work rather than compete:
|
||||||
|
|
||||||
|
- **`segment/`, COCO, 80 things.** Separates *instances* — clicking one of
|
||||||
|
three people selects that person. This is what local adjustments need, and a
|
||||||
|
semantic model cannot do it: it would return one "person" region covering all
|
||||||
|
three.
|
||||||
|
- **`scene/`, ADE20K, 150 classes.** Labels every pixel, including the *stuff*
|
||||||
|
COCO has no word for — sky, vegetation, water, mountain, wall. This is what
|
||||||
|
the scene tab's per-category grades need, and it does not care that instances
|
||||||
|
are merged, because a per-category grade applies to the whole category.
|
||||||
|
|
||||||
|
Neither replaces the other. Keeping both is the deliberate choice.
|
||||||
|
|
||||||
|
The loader treats each vocabulary as model metadata rather than compiled-in
|
||||||
|
knowledge, which is what made adding the second model a file plus a descriptor
|
||||||
|
rather than a code change — as this document predicted it would be.
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
# Photographic categories, over ADE20K's 150 classes.
|
||||||
|
#
|
||||||
|
# The scene tab offers a slider per category, not per class: nobody wants to
|
||||||
|
# grade "sconce" and "crt screen" separately, and ADE20K's vocabulary is a
|
||||||
|
# scene-parsing benchmark rather than a photographer's list. This file is the
|
||||||
|
# translation, and it is data so that changing it is not a code change.
|
||||||
|
#
|
||||||
|
# Format: one category per line, `name = class, class, ...`, where each class
|
||||||
|
# is a name from the model's own `.classes.json`. Names rather than indices
|
||||||
|
# because an index is silently wrong after a re-export and a name is loudly
|
||||||
|
# wrong; `SceneModel::from_path` refuses a file naming a class the model does
|
||||||
|
# not have.
|
||||||
|
#
|
||||||
|
# ## Why the list is short, and why "other" is not in it
|
||||||
|
#
|
||||||
|
# Weights come from a softmax over all 150 channels summed within each
|
||||||
|
# category, so the categories listed here plus everything unlisted sum to 1 at
|
||||||
|
# every pixel. That is what lets the scene tab feather two adjacent categories
|
||||||
|
# without painting both grades into the overlap. Adding a category takes
|
||||||
|
# weight from the unlisted remainder rather than from its neighbours, so this
|
||||||
|
# list can grow without disturbing what is already here.
|
||||||
|
#
|
||||||
|
# Classes are assigned to at most one category — an overlap would break the
|
||||||
|
# partition and double-count the shared class, so the loader rejects it.
|
||||||
|
|
||||||
|
# The one the whole exercise started from. ADE20K's easiest class, and the one
|
||||||
|
# most often graded on its own in a landscape.
|
||||||
|
sky = sky
|
||||||
|
|
||||||
|
# Foliage, not "green things": a lawn and a canopy take the same saturation
|
||||||
|
# and luminance moves far more often than either takes the sky's.
|
||||||
|
vegetation = tree, grass, plant, flower, palm, field
|
||||||
|
|
||||||
|
# Standing and moving water together. `swimming pool` and `fountain` are here
|
||||||
|
# rather than under architecture because what a photographer adjusts is the
|
||||||
|
# water, not the basin.
|
||||||
|
water = water, sea, river, lake, waterfall, swimming pool, fountain
|
||||||
|
|
||||||
|
# Distant landform. Separate from `ground` because it is usually far, hazy and
|
||||||
|
# wants dehaze and contrast where a foreground surface wants neither.
|
||||||
|
terrain = mountain, rock, hill
|
||||||
|
|
||||||
|
# What the photographer is standing on, or would be. Earth and sand sit here
|
||||||
|
# rather than with terrain for the same near/far reason.
|
||||||
|
ground = earth, sand, land, dirt track, path, road, sidewalk, runway, floor, step, stairs, stairway
|
||||||
|
|
||||||
|
# Built structure. Deliberately broad: a facade, its railings and its awning
|
||||||
|
# are one surface as far as a global grade is concerned.
|
||||||
|
architecture = building, house, skyscraper, wall, tower, bridge, hovel, fence, column, awning, booth, canopy, pier, railing, grandstand
|
||||||
|
|
||||||
|
# Present for the scene tab's "expose people" move, and *not* a replacement for
|
||||||
|
# the instance model — this is every person in the frame at once, which is the
|
||||||
|
# right granularity for a global grade and the wrong one for selecting a
|
||||||
|
# subject. See `models/LICENCE.md`.
|
||||||
|
person = person
|
||||||
@@ -0,0 +1,152 @@
|
|||||||
|
[
|
||||||
|
"wall",
|
||||||
|
"building",
|
||||||
|
"sky",
|
||||||
|
"floor",
|
||||||
|
"tree",
|
||||||
|
"ceiling",
|
||||||
|
"road",
|
||||||
|
"bed",
|
||||||
|
"windowpane",
|
||||||
|
"grass",
|
||||||
|
"cabinet",
|
||||||
|
"sidewalk",
|
||||||
|
"person",
|
||||||
|
"earth",
|
||||||
|
"door",
|
||||||
|
"table",
|
||||||
|
"mountain",
|
||||||
|
"plant",
|
||||||
|
"curtain",
|
||||||
|
"chair",
|
||||||
|
"car",
|
||||||
|
"water",
|
||||||
|
"painting",
|
||||||
|
"sofa",
|
||||||
|
"shelf",
|
||||||
|
"house",
|
||||||
|
"sea",
|
||||||
|
"mirror",
|
||||||
|
"rug",
|
||||||
|
"field",
|
||||||
|
"armchair",
|
||||||
|
"seat",
|
||||||
|
"fence",
|
||||||
|
"desk",
|
||||||
|
"rock",
|
||||||
|
"wardrobe",
|
||||||
|
"lamp",
|
||||||
|
"bathtub",
|
||||||
|
"railing",
|
||||||
|
"cushion",
|
||||||
|
"base",
|
||||||
|
"box",
|
||||||
|
"column",
|
||||||
|
"signboard",
|
||||||
|
"chest of drawers",
|
||||||
|
"counter",
|
||||||
|
"sand",
|
||||||
|
"sink",
|
||||||
|
"skyscraper",
|
||||||
|
"fireplace",
|
||||||
|
"refrigerator",
|
||||||
|
"grandstand",
|
||||||
|
"path",
|
||||||
|
"stairs",
|
||||||
|
"runway",
|
||||||
|
"case",
|
||||||
|
"pool table",
|
||||||
|
"pillow",
|
||||||
|
"screen door",
|
||||||
|
"stairway",
|
||||||
|
"river",
|
||||||
|
"bridge",
|
||||||
|
"bookcase",
|
||||||
|
"blind",
|
||||||
|
"coffee table",
|
||||||
|
"toilet",
|
||||||
|
"flower",
|
||||||
|
"book",
|
||||||
|
"hill",
|
||||||
|
"bench",
|
||||||
|
"countertop",
|
||||||
|
"stove",
|
||||||
|
"palm",
|
||||||
|
"kitchen island",
|
||||||
|
"computer",
|
||||||
|
"swivel chair",
|
||||||
|
"boat",
|
||||||
|
"bar",
|
||||||
|
"arcade machine",
|
||||||
|
"hovel",
|
||||||
|
"bus",
|
||||||
|
"towel",
|
||||||
|
"light",
|
||||||
|
"truck",
|
||||||
|
"tower",
|
||||||
|
"chandelier",
|
||||||
|
"awning",
|
||||||
|
"streetlight",
|
||||||
|
"booth",
|
||||||
|
"television receiver",
|
||||||
|
"airplane",
|
||||||
|
"dirt track",
|
||||||
|
"apparel",
|
||||||
|
"pole",
|
||||||
|
"land",
|
||||||
|
"bannister",
|
||||||
|
"escalator",
|
||||||
|
"ottoman",
|
||||||
|
"bottle",
|
||||||
|
"buffet",
|
||||||
|
"poster",
|
||||||
|
"stage",
|
||||||
|
"van",
|
||||||
|
"ship",
|
||||||
|
"fountain",
|
||||||
|
"conveyor belt",
|
||||||
|
"canopy",
|
||||||
|
"washer",
|
||||||
|
"plaything",
|
||||||
|
"swimming pool",
|
||||||
|
"stool",
|
||||||
|
"barrel",
|
||||||
|
"basket",
|
||||||
|
"waterfall",
|
||||||
|
"tent",
|
||||||
|
"bag",
|
||||||
|
"minibike",
|
||||||
|
"cradle",
|
||||||
|
"oven",
|
||||||
|
"ball",
|
||||||
|
"food",
|
||||||
|
"step",
|
||||||
|
"tank",
|
||||||
|
"trade name",
|
||||||
|
"microwave",
|
||||||
|
"pot",
|
||||||
|
"animal",
|
||||||
|
"bicycle",
|
||||||
|
"lake",
|
||||||
|
"dishwasher",
|
||||||
|
"screen",
|
||||||
|
"blanket",
|
||||||
|
"sculpture",
|
||||||
|
"hood",
|
||||||
|
"sconce",
|
||||||
|
"vase",
|
||||||
|
"traffic light",
|
||||||
|
"tray",
|
||||||
|
"ashcan",
|
||||||
|
"fan",
|
||||||
|
"pier",
|
||||||
|
"crt screen",
|
||||||
|
"plate",
|
||||||
|
"monitor",
|
||||||
|
"bulletin board",
|
||||||
|
"shower",
|
||||||
|
"radiator",
|
||||||
|
"glass",
|
||||||
|
"clock",
|
||||||
|
"flag"
|
||||||
|
]
|
||||||
Binary file not shown.
@@ -69,4 +69,23 @@ package() {
|
|||||||
fi
|
fi
|
||||||
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/${_m}"
|
install -Dm644 "${_src}" "${pkgdir}/usr/share/darkroom/models/${_m}"
|
||||||
done
|
done
|
||||||
|
|
||||||
|
# The scene model, for the per-category grades. Unlike the face weights
|
||||||
|
# this one is in the repository — AGPL, and GPLv3 §13 permits the
|
||||||
|
# combination (models/LICENCE.md) — so it is installed unconditionally and
|
||||||
|
# a pointer here is a broken checkout rather than a licence decision.
|
||||||
|
#
|
||||||
|
# Three files: the graph, its vocabulary, and the category descriptor
|
||||||
|
# grouping ADE20K's 150 classes into what the tab shows. All three, because
|
||||||
|
# dr_ui::library::scene_model reports the tab unavailable without any one
|
||||||
|
# of them. Only the graph gets the pointer check — the other two are
|
||||||
|
# legitimately a few kilobytes.
|
||||||
|
_src="models/scene/yolo26s-sem-ade20k.onnx"
|
||||||
|
if [[ "$(stat -c%s "${_src}")" -lt 100000 ]]; then
|
||||||
|
echo "error: the scene model is an LFS pointer — run: git lfs pull" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
for _m in yolo26s-sem-ade20k.onnx yolo26s-sem-ade20k.classes.json categories.txt; do
|
||||||
|
install -Dm644 "models/scene/${_m}" "${pkgdir}/usr/share/darkroom/models/${_m}"
|
||||||
|
done
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -165,6 +165,21 @@ modules:
|
|||||||
install -Dm644 "models/face/$m" "/app/share/darkroom/models/$m"
|
install -Dm644 "models/face/$m" "/app/share/darkroom/models/$m"
|
||||||
done
|
done
|
||||||
|
|
||||||
|
# The scene model, for the per-category grades. In the repository, unlike
|
||||||
|
# the face weights — AGPL, and GPLv3 §13 permits the combination
|
||||||
|
# (models/LICENCE.md) — so it installs unconditionally. Three files: the
|
||||||
|
# graph, its vocabulary, and the category descriptor; dr_ui reports the
|
||||||
|
# tab unavailable without any one of them. The pointer check is on the
|
||||||
|
# graph alone, the other two being legitimately small.
|
||||||
|
- |
|
||||||
|
if [ "$(stat -c%s models/scene/yolo26s-sem-ade20k.onnx)" -lt 100000 ]; then
|
||||||
|
echo "error: the scene model is an LFS pointer — run: git lfs pull" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
for m in yolo26s-sem-ade20k.onnx yolo26s-sem-ade20k.classes.json categories.txt; do
|
||||||
|
install -Dm644 "models/scene/$m" "/app/share/darkroom/models/$m"
|
||||||
|
done
|
||||||
|
|
||||||
- install -Dm644 README.md /app/share/doc/darkroom/README.md
|
- install -Dm644 README.md /app/share/doc/darkroom/README.md
|
||||||
|
|
||||||
sources:
|
sources:
|
||||||
|
|||||||
@@ -1,14 +1,25 @@
|
|||||||
#!/usr/bin/env bash
|
#!/usr/bin/env bash
|
||||||
# Re-export the segmentation model that ships in core/dr-segment/models/.
|
# Re-export the segmentation models that ship in models/ at the repository root.
|
||||||
#
|
#
|
||||||
# The .onnx is committed (D14), so this is not part of any build — it exists so
|
# The .onnx is committed (D14), so this is not part of any build — it exists so
|
||||||
# the committed artefact is reproducible rather than a binary someone once
|
# the committed artefact is reproducible rather than a binary someone once
|
||||||
# produced and nobody can regenerate. Run it when bumping the model.
|
# produced and nobody can regenerate. Run it when bumping a model.
|
||||||
#
|
#
|
||||||
# ./tools/export-seg-model.sh
|
# ./tools/export-seg-model.sh # instance -> models/segment/
|
||||||
|
# ./tools/export-seg-model.sh yolo26s-sem-ade20k # semantic -> models/scene/
|
||||||
#
|
#
|
||||||
# Requires `uv`. Everything else is fetched into a throwaway venv.
|
# Requires `uv`. Everything else is fetched into a throwaway venv.
|
||||||
#
|
#
|
||||||
|
# ## The two models, and why they are both here
|
||||||
|
#
|
||||||
|
# `yolo26n-seg` is COCO instance segmentation: it separates *things*, so
|
||||||
|
# clicking one of three people selects that person. `yolo26s-sem-ade20k` is
|
||||||
|
# ADE20K semantic segmentation: it labels every pixel with one of 150 classes
|
||||||
|
# including the *stuff* — sky, vegetation, water — that COCO has no word for,
|
||||||
|
# but it merges same-class pixels into one region and so cannot tell those
|
||||||
|
# three people apart. Neither substitutes for the other; the scene tab wants
|
||||||
|
# the second and local adjustments want the first.
|
||||||
|
#
|
||||||
# ## Why these export flags
|
# ## Why these export flags
|
||||||
#
|
#
|
||||||
# `dynamic=False` is not a default we failed to change: **tract cannot parse
|
# `dynamic=False` is not a default we failed to change: **tract cannot parse
|
||||||
@@ -20,15 +31,44 @@
|
|||||||
# `imgsz` square rather than a rectangle matched to 3:2: one graph has to
|
# `imgsz` square rather than a rectangle matched to 3:2: one graph has to
|
||||||
# serve portrait, landscape, square crops and panoramas. A landscape-shaped
|
# serve portrait, landscape, square crops and panoramas. A landscape-shaped
|
||||||
# graph trades letterbox waste on 3:2 for worse waste on everything else.
|
# graph trades letterbox waste on 3:2 for worse waste on everything else.
|
||||||
|
#
|
||||||
|
# ## Why the semantic export is truncated
|
||||||
|
#
|
||||||
|
# Ultralytics ends the `-sem-` graph with `Resize -> ArgMax -> Cast`, handing
|
||||||
|
# back a `[1, 640, 640]` u8 label map. Two reasons that tail is cut here:
|
||||||
|
#
|
||||||
|
# 1. **Cost.** The Resize materialises 150 x 640 x 640 x f32 — *246 MB* — and
|
||||||
|
# ArgMax then reduces across the channel axis, which in NCHW strides
|
||||||
|
# 409,600 elements per comparison. Measured on one machine it was roughly
|
||||||
|
# four fifths of total runtime, for work the app throws away.
|
||||||
|
# 2. **Softness.** ArgMax destroys the per-class scores. The scene tab needs
|
||||||
|
# them: softmax over the 150 channels, summed within each photographic
|
||||||
|
# category, gives per-category weights that sum to 1 at every pixel — a
|
||||||
|
# partition of unity. Feathering those cannot double-grade a boundary,
|
||||||
|
# where feathering hard labels outward from two adjacent categories does.
|
||||||
|
#
|
||||||
|
# The upsample is not information: the graph's true spatial resolution is the
|
||||||
|
# logit grid (80x80 at imgsz=640), and the app can resample from that itself.
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
REPO="$(cd "${HERE}/.." && pwd)"
|
REPO="$(cd "${HERE}/.." && pwd)"
|
||||||
OUT="${REPO}/core/dr-segment/models"
|
|
||||||
|
|
||||||
MODEL="${1:-yolo26n-seg}"
|
MODEL="${1:-yolo26n-seg}"
|
||||||
IMGSZ="${2:-640}"
|
IMGSZ="${2:-640}"
|
||||||
WORK="$(mktemp -d)"
|
|
||||||
|
# Semantic models are the scene tab's; instance models are the selection path's.
|
||||||
|
case "${MODEL}" in
|
||||||
|
*-sem-*|*-sem) OUT="${REPO}/models/scene"; SEMANTIC=1 ;;
|
||||||
|
*) OUT="${REPO}/models/segment"; SEMANTIC=0 ;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
# Not `mktemp -d`: the default TMPDIR is `/tmp`, which on most current Linux
|
||||||
|
# distributions is a tmpfs — RAM, sized at half of physical memory. The venv
|
||||||
|
# below pulls torch, several gigabytes of it, and installing that into RAM
|
||||||
|
# either evicts the user's page cache or fails outright with ENOSPC on a
|
||||||
|
# machine that has hundreds of gigabytes of actual disk free.
|
||||||
|
WORK="$(mktemp -d -p "${TMPDIR:-/var/tmp}")"
|
||||||
trap 'rm -rf "${WORK}"' EXIT
|
trap 'rm -rf "${WORK}"' EXIT
|
||||||
|
|
||||||
echo "==> exporting ${MODEL} at imgsz=${IMGSZ} in ${WORK}"
|
echo "==> exporting ${MODEL} at imgsz=${IMGSZ} in ${WORK}"
|
||||||
@@ -36,12 +76,13 @@ cd "${WORK}"
|
|||||||
uv venv --python 3.12 venv
|
uv venv --python 3.12 venv
|
||||||
VIRTUAL_ENV="${WORK}/venv" uv pip install ultralytics onnx onnxslim
|
VIRTUAL_ENV="${WORK}/venv" uv pip install ultralytics onnx onnxslim
|
||||||
|
|
||||||
VIRTUAL_ENV="${WORK}/venv" "${WORK}/venv/bin/python" - "${MODEL}" "${IMGSZ}" <<'PY'
|
VIRTUAL_ENV="${WORK}/venv" "${WORK}/venv/bin/python" - "${MODEL}" "${IMGSZ}" "${SEMANTIC}" <<'PY'
|
||||||
import sys, json
|
import sys, json
|
||||||
|
import onnx
|
||||||
|
from onnx import helper
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
|
|
||||||
name = sys.argv[1]
|
name, imgsz, semantic = sys.argv[1], int(sys.argv[2]), sys.argv[3] == "1"
|
||||||
imgsz = int(sys.argv[2])
|
|
||||||
m = YOLO(f"{name}.pt")
|
m = YOLO(f"{name}.pt")
|
||||||
path = m.export(format="onnx", opset=17, simplify=True, imgsz=imgsz, dynamic=False)
|
path = m.export(format="onnx", opset=17, simplify=True, imgsz=imgsz, dynamic=False)
|
||||||
print("ONNX:", path)
|
print("ONNX:", path)
|
||||||
@@ -52,6 +93,25 @@ print("ONNX:", path)
|
|||||||
with open("classes.json", "w") as f:
|
with open("classes.json", "w") as f:
|
||||||
json.dump([m.names[i] for i in range(len(m.names))], f, indent=1)
|
json.dump([m.names[i] for i in range(len(m.names))], f, indent=1)
|
||||||
print("classes:", len(m.names))
|
print("classes:", len(m.names))
|
||||||
|
|
||||||
|
if semantic:
|
||||||
|
# Drop `Resize -> ArgMax -> Cast` and expose the classifier's logits. See
|
||||||
|
# the header for why. Matched by op type rather than by node name so a
|
||||||
|
# re-export under a different naming scheme still works, and asserted
|
||||||
|
# rather than assumed so an upstream graph change fails loudly here
|
||||||
|
# instead of silently shipping a differently-shaped model.
|
||||||
|
g = onnx.load(path).graph
|
||||||
|
tail = [n.op_type for n in g.node[-3:]]
|
||||||
|
assert tail == ["Resize", "ArgMax", "Cast"], f"unexpected graph tail: {tail}"
|
||||||
|
logits = g.node[-3].input[0]
|
||||||
|
del g.node[-3:]
|
||||||
|
del g.output[:]
|
||||||
|
g.output.extend([helper.make_tensor_value_info(logits, onnx.TensorProto.FLOAT, None)])
|
||||||
|
model = onnx.shape_inference.infer_shapes(helper.make_model(g, opset_imports=[helper.make_opsetid("", 17)]))
|
||||||
|
onnx.checker.check_model(model)
|
||||||
|
onnx.save(model, path)
|
||||||
|
shape = [d.dim_value for d in model.graph.output[0].type.tensor_type.shape.dim]
|
||||||
|
print("truncated to logits:", logits, shape)
|
||||||
PY
|
PY
|
||||||
|
|
||||||
mkdir -p "${OUT}"
|
mkdir -p "${OUT}"
|
||||||
@@ -61,4 +121,4 @@ cp "${WORK}/classes.json" "${OUT}/${MODEL}.classes.json"
|
|||||||
echo "==> wrote:"
|
echo "==> wrote:"
|
||||||
ls -la "${OUT}"
|
ls -la "${OUT}"
|
||||||
echo
|
echo
|
||||||
echo "Remember: these weights are AGPL-3.0 (see ${OUT}/LICENCE.md)."
|
echo "Remember: these weights are AGPL-3.0 (see ${REPO}/models/LICENCE.md)."
|
||||||
|
|||||||
@@ -73,6 +73,15 @@ pub use develop::DevelopSession;
|
|||||||
/// opened — see `library::shared_face_models_dir`.
|
/// opened — see `library::shared_face_models_dir`.
|
||||||
pub use library::shared_face_models_dir;
|
pub use library::shared_face_models_dir;
|
||||||
|
|
||||||
|
/// The scene model's three files, wherever this device keeps them.
|
||||||
|
///
|
||||||
|
/// Public for the same reason as the directory above: the pieces that reach for
|
||||||
|
/// a model are not all inside this crate. Exported ahead of the scene tab that
|
||||||
|
/// will consume it so that the packaging and unpacking added alongside it have
|
||||||
|
/// something to be verified against — `assemble-apk.sh` writing files no lookup
|
||||||
|
/// looks for would be a silent mistake for as long as the tab took to arrive.
|
||||||
|
pub use library::scene_model;
|
||||||
|
|
||||||
pub mod launch;
|
pub mod launch;
|
||||||
pub mod launch_ui;
|
pub mod launch_ui;
|
||||||
|
|
||||||
|
|||||||
@@ -3846,6 +3846,37 @@ pub fn face_models(account: &Account) -> Option<(PathBuf, PathBuf)> {
|
|||||||
.or_else(|| system_face_models_dirs().into_iter().find_map(pair))
|
.or_else(|| system_face_models_dirs().into_iter().find_map(pair))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The scene model, its vocabulary and its category descriptor, if all three
|
||||||
|
/// are present.
|
||||||
|
///
|
||||||
|
/// All three or none, for the same reason `face_models` insists on its pair:
|
||||||
|
/// the graph alone decodes to 150 anonymous channels, and a descriptor naming
|
||||||
|
/// classes a different model does not have is refused by
|
||||||
|
/// `dr_segment::scene::parse_categories` anyway. Reporting the set missing is
|
||||||
|
/// more useful than starting and failing at the first inference.
|
||||||
|
///
|
||||||
|
/// Searched in the same three places, most specific first — the account's own
|
||||||
|
/// directory, the shared one, then wherever a package installed them. Android
|
||||||
|
/// only ever finds the second, which is where `install_bundled_models` unpacks
|
||||||
|
/// the APK's copy before any store opens.
|
||||||
|
///
|
||||||
|
/// Unlike the face weights this model *is* in the repository, so a desktop
|
||||||
|
/// build from a complete checkout has it. Absent means either a checkout
|
||||||
|
/// without `git lfs pull` or a package that chose not to carry 24 MB, and the
|
||||||
|
/// scene tab reports itself unavailable rather than the app refusing to run.
|
||||||
|
pub fn scene_model(account: &Account) -> Option<(PathBuf, PathBuf, PathBuf)> {
|
||||||
|
fn set(dir: PathBuf) -> Option<(PathBuf, PathBuf, PathBuf)> {
|
||||||
|
let model = dir.join("yolo26s-sem-ade20k.onnx");
|
||||||
|
let classes = dir.join("yolo26s-sem-ade20k.classes.json");
|
||||||
|
let categories = dir.join("categories.txt");
|
||||||
|
(model.is_file() && classes.is_file() && categories.is_file())
|
||||||
|
.then_some((model, classes, categories))
|
||||||
|
}
|
||||||
|
set(face_models_dir(account))
|
||||||
|
.or_else(|| set(shared_face_models_dir()))
|
||||||
|
.or_else(|| system_face_models_dirs().into_iter().find_map(set))
|
||||||
|
}
|
||||||
|
|
||||||
/// Where a *package* may have installed the models.
|
/// Where a *package* may have installed the models.
|
||||||
///
|
///
|
||||||
/// `$XDG_DATA_DIRS` rather than a hard-coded `/usr/share`, because that is the
|
/// `$XDG_DATA_DIRS` rather than a hard-coded `/usr/share`, because that is the
|
||||||
|
|||||||
Reference in New Issue
Block a user