Merge branch 'master' into worktree-faces-scrfd-mbf
Build and test / Desktop (Linux) (push) Failing after 25s
Build and test / Layer separation (push) Successful in 22s
Traceability / Requirement traces (push) Successful in 58s
🐳 Android image / Build and push (push) Successful in 1s
Build and test / android-image (push) Successful in 1s
Build and test / Android (aarch64) (push) Failing after 33m26s
Build and test / Desktop (Linux) (push) Failing after 25s
Build and test / Layer separation (push) Successful in 22s
Traceability / Requirement traces (push) Successful in 58s
🐳 Android image / Build and push (push) Successful in 1s
Build and test / android-image (push) Successful in 1s
Build and test / Android (aarch64) (push) Failing after 33m26s
# Conflicts: # docs/traceability.md # ui/dr-ui/src/develop.rs # ui/dr-ui/src/segmentation.rs
This commit is contained in:
+237
-11
@@ -34,6 +34,7 @@ use std::sync::Arc;
|
||||
|
||||
use dr_gpu::GpuContext;
|
||||
use dr_pipeline::mask::segmentation_signature;
|
||||
use dr_types::Orientation;
|
||||
|
||||
/// One recognised object.
|
||||
#[derive(Debug, Clone)]
|
||||
@@ -243,27 +244,41 @@ impl Default for Options {
|
||||
|
||||
/// Find what can be selected in this photograph.
|
||||
///
|
||||
/// `rgb` is the proxy the model reads — tightly packed RGB floats at
|
||||
/// `(width, height)`. Passed in rather than derived here because the caller
|
||||
/// already has the rendered proxy, and re-deriving it would mean a second
|
||||
/// readback of something the CPU is holding.
|
||||
/// `rgb` is the rendered proxy — tightly packed RGB floats at
|
||||
/// `(width, height)`, in the **sensor's** own orientation. Passed in rather
|
||||
/// than derived here because the caller already has it, and re-deriving it
|
||||
/// would mean a second readback of something the CPU is holding.
|
||||
///
|
||||
/// `orientation` is the file's EXIF tag composed with whatever turns the
|
||||
/// photographer has since applied — `Framing::effective_orientation`, one
|
||||
/// permutation covering both. The model is shown the picture through it and
|
||||
/// its answers come back without it, so everything this returns is in sensor
|
||||
/// space exactly as it was before the detector was taught to read.
|
||||
pub fn compute(
|
||||
_ctx: &GpuContext,
|
||||
rgb: &[f32],
|
||||
width: usize,
|
||||
height: usize,
|
||||
orientation: Orientation,
|
||||
options: &Options,
|
||||
) -> Result<Segmentation, String> {
|
||||
let found = detect(rgb, width, height, options.fine)?;
|
||||
// The model reads the photograph; everything else here speaks sensor.
|
||||
let (stood_up, uw, uh) = upright(rgb, width, height, orientation);
|
||||
let found = detect(&stood_up, uw, uh, options.fine)?;
|
||||
|
||||
let instances: Vec<InstanceSummary> = found
|
||||
.iter()
|
||||
.filter(|i| i.score >= options.confidence)
|
||||
.map(|i| InstanceSummary {
|
||||
class_name: i.class_name.clone(),
|
||||
score: i.score,
|
||||
mask: quantise(&i.mask),
|
||||
bbox: i.bbox,
|
||||
.map(|i| {
|
||||
let (mask, bbox) = lay_down(&i.mask, i.bbox, uw, uh, orientation);
|
||||
InstanceSummary {
|
||||
class_name: i.class_name.clone(),
|
||||
score: i.score,
|
||||
// Quantised after the permutation, so the byte stored is a
|
||||
// rounding of the model's own coverage and not of a copy.
|
||||
mask: quantise(&mask),
|
||||
bbox,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
@@ -278,21 +293,122 @@ pub fn compute(
|
||||
// a signature, a layer built against the coarse pass would be silently
|
||||
// reinterpreted against the fine one. That is a *wrong* mask, which is
|
||||
// far worse than a stale one, because nothing announces it.
|
||||
//
|
||||
// The orientation is in for the same reason and it is not hypothetical:
|
||||
// turning the photograph changes what the model recognises, so a run
|
||||
// before a quarter turn and a run after it are different instance
|
||||
// lists. Two lists that happened to come out the same length would
|
||||
// otherwise share a signature, and a layer built against the first
|
||||
// would be silently re-indexed into the second.
|
||||
options.confidence.to_bits() as u64
|
||||
^ if options.fine {
|
||||
0x9E37_79B9_7F4A_7C15
|
||||
} else {
|
||||
0
|
||||
},
|
||||
}
|
||||
^ orientation_key(orientation),
|
||||
);
|
||||
|
||||
Ok(Segmentation {
|
||||
instances,
|
||||
signature,
|
||||
// **Sensor space, not the model's.** `lay_down` put every mask back,
|
||||
// so the grid a stored layer indexes into is the one it always was —
|
||||
// see `upright` for why the model saw a different one.
|
||||
proxy: (width, height),
|
||||
})
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-3 | FR-DEV-3h
|
||||
/// Turn the proxy the way the photographer is looking at it.
|
||||
///
|
||||
/// **Why this exists at all.** A camera held sideways writes its sensor rows
|
||||
/// the way it always does, and the render puts them right by way of
|
||||
/// `Framing`. The proxy the model reads is deliberately rendered through a
|
||||
/// *neutral* graph — the detection has to survive an exposure change, or
|
||||
/// every slider would invalidate the masks built on it — and neutral took the
|
||||
/// orientation with it. So the detector was handed a portrait frame lying on
|
||||
/// its side, and a model trained on upright photographs is very bad at those.
|
||||
/// Measured end to end on one 22 MP frame of two people and a dog: `person
|
||||
/// 0.36` and nothing else, against `dog 0.82, person 0.61, person 0.49` for
|
||||
/// the same pixels stood up.
|
||||
///
|
||||
/// One line, because the permutation belongs to
|
||||
/// [`dr_types::Orientation`] and every other consumer goes through the same
|
||||
/// one — the grid's thumbnails included, which is what makes "upright" mean
|
||||
/// one thing across the application rather than one thing per caller.
|
||||
pub(crate) fn upright(
|
||||
rgb: &[f32],
|
||||
width: usize,
|
||||
height: usize,
|
||||
orientation: Orientation,
|
||||
) -> (Vec<f32>, usize, usize) {
|
||||
let (out, w, h) = orientation.into_shown(rgb, width as u32, height as u32, 3);
|
||||
(out, w as usize, h as usize)
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-3
|
||||
/// Put what the model answered back onto the sensor's grid.
|
||||
///
|
||||
/// The counterpart of [`upright`], and the two are always used as a pair: a
|
||||
/// mask is only ever in the model's frame between those two calls. Returned
|
||||
/// together rather than as two functions a caller composes, because calling
|
||||
/// one and forgetting the other is silent — the mask lands a quarter turn off
|
||||
/// the subject, which reads as a bad detection rather than as a bug.
|
||||
///
|
||||
/// `dw`/`dh` are the *shown* dimensions, as [`upright`] returned them.
|
||||
pub(crate) fn lay_down(
|
||||
mask: &[f32],
|
||||
bbox: (f32, f32, f32, f32),
|
||||
dw: usize,
|
||||
dh: usize,
|
||||
orientation: Orientation,
|
||||
) -> (Vec<f32>, (f32, f32, f32, f32)) {
|
||||
let (out, sw, sh) = orientation.into_stored(mask, dw as u32, dh as u32, 1);
|
||||
(out, lay_down_bbox(bbox, dw, dh, sw, sh, orientation))
|
||||
}
|
||||
|
||||
/// [`lay_down`] for a box.
|
||||
///
|
||||
/// Normalised on the way in and scaled on the way out, so the turn itself is
|
||||
/// `Orientation::into_stored_rect` rather than a fourth copy of the corner
|
||||
/// arithmetic. A box is the one place a permutation can be *nearly* right —
|
||||
/// the corners land correctly and `x0 > x1` — so the shared map takes the
|
||||
/// extremes and this only has to say what space it is in.
|
||||
fn lay_down_bbox(
|
||||
bbox: (f32, f32, f32, f32),
|
||||
dw: usize,
|
||||
dh: usize,
|
||||
sw: u32,
|
||||
sh: u32,
|
||||
orientation: Orientation,
|
||||
) -> (f32, f32, f32, f32) {
|
||||
if dw == 0 || dh == 0 {
|
||||
return bbox;
|
||||
}
|
||||
let (fw, fh) = (dw as f32, dh as f32);
|
||||
let shown = dr_types::ShownRect {
|
||||
x: bbox.0 / fw,
|
||||
y: bbox.1 / fh,
|
||||
width: (bbox.2 - bbox.0) / fw,
|
||||
height: (bbox.3 - bbox.1) / fh,
|
||||
};
|
||||
let stored = orientation.into_stored_rect(shown);
|
||||
(
|
||||
stored.x * sw as f32,
|
||||
stored.y * sh as f32,
|
||||
(stored.x + stored.width) * sw as f32,
|
||||
(stored.y + stored.height) * sh as f32,
|
||||
)
|
||||
}
|
||||
|
||||
/// One of eight transforms, as bits a signature can carry.
|
||||
fn orientation_key(orientation: Orientation) -> u64 {
|
||||
u64::from(orientation.quarter_turns)
|
||||
| (u64::from(orientation.flip_h) << 2)
|
||||
| (u64::from(orientation.flip_v) << 3)
|
||||
}
|
||||
|
||||
/// Classes a recognised face may put a name on.
|
||||
///
|
||||
/// Only these. A face inside a `tv` or a `laptop` is a photograph of someone on
|
||||
@@ -499,6 +615,116 @@ mod tests {
|
||||
assert_eq!(quantise(&[-1.0, 2.0]), vec![0, 255]);
|
||||
}
|
||||
|
||||
/// A non-square, wholly asymmetric grid: every pixel is its own index, so
|
||||
/// any permutation that is not the intended one shows up as a mismatch
|
||||
/// rather than being hidden by a symmetry.
|
||||
fn ramp(w: usize, h: usize) -> Vec<f32> {
|
||||
(0..w * h).flat_map(|i| [i as f32, 0.0, 0.0]).collect()
|
||||
}
|
||||
|
||||
fn red(rgb: &[f32]) -> Vec<f32> {
|
||||
rgb.chunks_exact(3).map(|p| p[0]).collect()
|
||||
}
|
||||
|
||||
/// The property the whole fix rests on: what the model is shown and what
|
||||
/// comes back are the same permutation, run in opposite directions. If
|
||||
/// they ever disagree, every subject mask lands somewhere other than its
|
||||
/// subject — and looks like a mask while doing it.
|
||||
#[test]
|
||||
fn standing_a_frame_up_and_laying_it_down_is_the_identity() {
|
||||
const W: usize = 5;
|
||||
const H: usize = 3;
|
||||
let source = ramp(W, H);
|
||||
|
||||
for tag in 1..=8u16 {
|
||||
let o = Orientation::from_exif(tag);
|
||||
let (up, uw, uh) = upright(&source, W, H, o);
|
||||
let (ow, oh) = o.oriented_size(W as u32, H as u32);
|
||||
assert_eq!(
|
||||
(uw, uh),
|
||||
(ow as usize, oh as usize),
|
||||
"tag {tag}: the upright size is the oriented one"
|
||||
);
|
||||
|
||||
let (back, _) = lay_down(&red(&up), (0.0, 0.0, 1.0, 1.0), uw, uh, o);
|
||||
assert_eq!(back, red(&source), "tag {tag} did not come back");
|
||||
}
|
||||
}
|
||||
|
||||
/// The colour channels must travel together. Reading a pixel three times
|
||||
/// with one index arithmetic mistake gives a plausible image with its
|
||||
/// channels sheared, which the model would still detect *something* in.
|
||||
#[test]
|
||||
fn a_turn_carries_whole_pixels() {
|
||||
let rgb: Vec<f32> = (0..2 * 3)
|
||||
.flat_map(|i| [i as f32, i as f32 + 100.0, i as f32 + 200.0])
|
||||
.collect();
|
||||
let o = Orientation::from_exif(6);
|
||||
let (up, uw, uh) = upright(&rgb, 2, 3, o);
|
||||
|
||||
assert_eq!((uw, uh), (3, 2));
|
||||
for p in up.chunks_exact(3) {
|
||||
assert_eq!(p[1], p[0] + 100.0, "green left its pixel");
|
||||
assert_eq!(p[2], p[0] + 200.0, "blue left its pixel");
|
||||
}
|
||||
}
|
||||
|
||||
/// A portrait frame is the case this exists for: the sensor is landscape,
|
||||
/// the photograph is not, and the model has to be given the photograph.
|
||||
#[test]
|
||||
fn a_sideways_frame_reaches_the_model_upright() {
|
||||
let o = Orientation::from_exif(6);
|
||||
assert!(!o.is_normal());
|
||||
|
||||
let (_, uw, uh) = upright(&ramp(1600, 1066), 1600, 1066, o);
|
||||
assert_eq!((uw, uh), (1066, 1600), "the model still got a landscape");
|
||||
}
|
||||
|
||||
/// Where the model's box ends up, worked out by hand for the one turn a
|
||||
/// portrait phone or a sideways body actually writes.
|
||||
#[test]
|
||||
fn a_box_comes_back_in_sensor_pixels() {
|
||||
let o = Orientation::from_exif(6);
|
||||
// Shown 4x6; the sensor it came from is 6x4.
|
||||
let bbox = lay_down_bbox((0.0, 0.0, 2.0, 3.0), 4, 6, 6, 4, o);
|
||||
assert_eq!(bbox, (0.0, 2.0, 3.0, 4.0));
|
||||
}
|
||||
|
||||
/// Whatever a turn does to a box, it must still read low-to-high — a
|
||||
/// permutation exchanges which corner is which, and the rest of the mask
|
||||
/// pipeline measures `(x1 - x0)` without checking the sign.
|
||||
#[test]
|
||||
fn a_restored_box_keeps_its_corners_in_order() {
|
||||
for tag in 1..=8u16 {
|
||||
let o = Orientation::from_exif(tag);
|
||||
let (sw, sh) = o.oriented_size(9, 6);
|
||||
let (x0, y0, x1, y1) = lay_down_bbox((1.0, 2.0, 7.0, 5.0), 9, 6, sw, sh, o);
|
||||
assert!(x0 <= x1, "tag {tag}: x runs backwards");
|
||||
assert!(y0 <= y1, "tag {tag}: y runs backwards");
|
||||
// A permutation moves a box; it does not resize one.
|
||||
let (sw, sh) = o.oriented_size(9, 6);
|
||||
assert!(x1 <= sw as f32 && y1 <= sh as f32, "tag {tag}: box escaped");
|
||||
assert!((((x1 - x0) * (y1 - y0)) - 18.0).abs() < 1e-3, "tag {tag}");
|
||||
}
|
||||
}
|
||||
|
||||
/// Turning the photograph changes what the model recognises, so the two
|
||||
/// runs are different instance lists. If they could share a signature, a
|
||||
/// layer built against one would be silently re-indexed into the other —
|
||||
/// the same failure the tiling flag is in the signature to prevent.
|
||||
#[test]
|
||||
fn turning_the_photograph_changes_the_signature() {
|
||||
let mut seen = std::collections::HashSet::new();
|
||||
for tag in 1..=8u16 {
|
||||
let o = Orientation::from_exif(tag);
|
||||
assert!(
|
||||
seen.insert(orientation_key(o)),
|
||||
"tag {tag} shares a key with an earlier one"
|
||||
);
|
||||
}
|
||||
assert_eq!(seen.len(), 8, "eight tags, but some collapsed");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn confidence_changes_the_signature() {
|
||||
// A different threshold is a different instance list, so the indices a
|
||||
|
||||
Reference in New Issue
Block a user