DemosaicedImage::linear_rgb16_window uploads part of a linear DNG, or a box-reduced copy of it, and says where it sits in the frame; size() now reports the frame and texture_size() the texels, and the fused pass writes the window into the shader's uniforms. EditGraph::source_region finds the part of the source a view reads, and tiles::plan cuts a render too large for one texture into halo-grown, grid-aligned tiles. The GPU test renders frames a tile at a time from their own windows and compares them with the whole: identical for point operations, within one code value when straightened with clarity on.
848 lines
32 KiB
Rust
848 lines
32 KiB
Rust
//! Watershed segmentation — arm A's GPU half (S15, docs/dev/segmentation.md).
|
||
//!
|
||
//! Runs the five passes in `shaders/watershed.wgsl` over a demosaiced image
|
||
//! and leaves a basin label per pixel on the GPU. The hierarchy built from
|
||
//! those labels lives in [`dr_segment`], which needs no device.
|
||
//!
|
||
//! # Cost
|
||
//!
|
||
//! Every pass is a trivial kernel and the whole chain is a handful of
|
||
//! milliseconds at proxy resolution. It runs **once per image**, off the
|
||
//! interactive path — the point of precomputing a region map is that
|
||
//! selection afterwards is a label comparison rather than a flood fill.
|
||
//!
|
||
//! # The open question this leaves
|
||
//!
|
||
//! [`Segmentation::read_field`] copies the label and gradient buffers back to
|
||
//! the CPU to build the region adjacency graph, behind the `segment-readback`
|
||
//! feature. That is deliberately **not** the `readback` switch guarding the
|
||
//! display round-trip: this transfer is once per image on a worker, where the
|
||
//! one AC-8 forbids is per frame in the render loop, and sharing a switch
|
||
//! would force a build wanting local masking to unlock the other.
|
||
//!
|
||
//! It is still a real cost and still unfinished. F3 in docs/dev/segmentation.md
|
||
//! §12 stands: the adjacency accumulation belongs GPU-side with atomics, and
|
||
//! until it moves there every segmentation pays a full-resolution transfer.
|
||
//! Read the feature name as a description of a known gap rather than as
|
||
//! permission.
|
||
|
||
use wgpu::util::DeviceExt;
|
||
|
||
use crate::{DemosaicedImage, GpuContext, GpuError};
|
||
|
||
/// How the watershed is tuned for one image.
|
||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||
pub struct SegmentOptions {
|
||
/// Longest proxy edge. The segmentation runs here, not at sensor
|
||
/// resolution: a 24 MP watershed costs 12× the memory to place boundaries
|
||
/// a person cannot see, and the boundary refinement that matters at 1:1
|
||
/// is a separate stage (docs/dev/segmentation.md §4).
|
||
pub max_edge: u32,
|
||
/// Pre-smoothing radius in proxy pixels. The caller's to raise with ISO —
|
||
/// this is the single knob that decides whether a noisy file segments
|
||
/// into regions or into grain.
|
||
pub blur_radius: i32,
|
||
pub w_luma: f32,
|
||
pub w_chroma: f32,
|
||
/// How far the lower-completion carries a distance inward from a
|
||
/// plateau's rim, in breadth-first steps.
|
||
///
|
||
/// Bounds the widest plateau that resolves fully. Beyond it, the interior
|
||
/// keeps the behaviour it had before the pass existed — a fan of diagonal
|
||
/// chains — so this trades dispatches against the size of flat area the
|
||
/// watershed handles cleanly, and never against correctness elsewhere.
|
||
pub plateau_iterations: u32,
|
||
}
|
||
|
||
impl Default for SegmentOptions {
|
||
fn default() -> Self {
|
||
Self {
|
||
// ~1.3 MP at 3:2. Large enough that a boundary is within a pixel
|
||
// or two of where it belongs, small enough that the whole chain
|
||
// fits comfortably in memory on a phone.
|
||
max_edge: 1600,
|
||
blur_radius: 2,
|
||
w_luma: 1.0,
|
||
// Chroma carries most of the sensor noise and few of the
|
||
// boundaries anyone would draw, so it counts for less — but not
|
||
// zero, or a red flower on green leaves has no edge at all.
|
||
w_chroma: 0.5,
|
||
// **Zero: the pass is off.** It is implemented, dispatched
|
||
// correctly and measurably changes nothing — see the ignored test
|
||
// below and §12 of docs/dev/segmentation.md. Until that is understood,
|
||
// running it would buy 64 dispatches per segmentation and no
|
||
// improvement, so the default declines to pay.
|
||
plateau_iterations: 0,
|
||
}
|
||
}
|
||
}
|
||
|
||
#[repr(C)]
|
||
#[derive(Copy, Clone, bytemuck::Pod, bytemuck::Zeroable)]
|
||
struct SegParams {
|
||
width: u32,
|
||
height: u32,
|
||
src_width: u32,
|
||
src_height: u32,
|
||
blur_radius: i32,
|
||
non_linear: u32,
|
||
w_luma: f32,
|
||
w_chroma: f32,
|
||
}
|
||
|
||
/// One compute stage: its layout and its compiled pipeline.
|
||
struct Stage {
|
||
layout: wgpu::BindGroupLayout,
|
||
pipeline: wgpu::ComputePipeline,
|
||
}
|
||
|
||
/// Runs the watershed chain.
|
||
pub struct SegmentPass {
|
||
ctx: GpuContext,
|
||
features: Stage,
|
||
blur: Stage,
|
||
gradient: Stage,
|
||
plateau_init: Stage,
|
||
plateau_step: Stage,
|
||
flow: Stage,
|
||
jump: Stage,
|
||
}
|
||
|
||
impl SegmentPass {
|
||
pub fn new(ctx: &GpuContext) -> Result<Self, GpuError> {
|
||
// A validation failure here is a bug in the shader, not a user error.
|
||
// Surfaced as a Result rather than wgpu's default panic, matching how
|
||
// `AdjustPass` handles its generated source.
|
||
let scope = ctx.device.push_error_scope(wgpu::ErrorFilter::Validation);
|
||
|
||
let module = ctx
|
||
.device
|
||
.create_shader_module(wgpu::ShaderModuleDescriptor {
|
||
label: Some("watershed"),
|
||
source: wgpu::ShaderSource::Wgsl(include_str!("shaders/watershed.wgsl").into()),
|
||
});
|
||
|
||
let features = {
|
||
let layout = ctx
|
||
.device
|
||
.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
|
||
label: Some("watershed-features-bgl"),
|
||
entries: &[
|
||
uniform_entry(0),
|
||
wgpu::BindGroupLayoutEntry {
|
||
binding: 1,
|
||
visibility: wgpu::ShaderStages::COMPUTE,
|
||
ty: wgpu::BindingType::Texture {
|
||
sample_type: wgpu::TextureSampleType::Float { filterable: true },
|
||
view_dimension: wgpu::TextureViewDimension::D2,
|
||
multisampled: false,
|
||
},
|
||
count: None,
|
||
},
|
||
storage_entry(2, false),
|
||
],
|
||
});
|
||
let pipeline = compute(ctx, &module, &layout, "features");
|
||
Stage { layout, pipeline }
|
||
};
|
||
|
||
let buffer_stage = |in_binding: u32, out_binding: u32, entry: &str, label: &str| {
|
||
let layout = ctx
|
||
.device
|
||
.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
|
||
label: Some(label),
|
||
entries: &[
|
||
uniform_entry(0),
|
||
storage_entry(in_binding, true),
|
||
storage_entry(out_binding, false),
|
||
],
|
||
});
|
||
let pipeline = compute(ctx, &module, &layout, entry);
|
||
Stage { layout, pipeline }
|
||
};
|
||
|
||
// Three bindings rather than two: these read the gradient *and* a
|
||
// distance field, and write a second one.
|
||
let triple_stage = |a: u32, b: u32, c: u32, entry: &str, label: &str| {
|
||
let layout = ctx
|
||
.device
|
||
.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
|
||
label: Some(label),
|
||
entries: &[
|
||
uniform_entry(0),
|
||
storage_entry(a, true),
|
||
storage_entry(b, true),
|
||
storage_entry(c, false),
|
||
],
|
||
});
|
||
let pipeline = compute(ctx, &module, &layout, entry);
|
||
Stage { layout, pipeline }
|
||
};
|
||
|
||
let blur = buffer_stage(3, 4, "blur", "watershed-blur-bgl");
|
||
let gradient = buffer_stage(5, 6, "gradient", "watershed-gradient-bgl");
|
||
let plateau_init = buffer_stage(7, 8, "plateau_init", "watershed-pinit-bgl");
|
||
let plateau_step = triple_stage(9, 10, 11, "plateau_step", "watershed-pstep-bgl");
|
||
let flow = triple_stage(12, 13, 14, "flow", "watershed-flow-bgl");
|
||
let jump = buffer_stage(15, 16, "jump", "watershed-jump-bgl");
|
||
|
||
if let Some(err) = pollster::block_on(scope.pop()) {
|
||
return Err(GpuError::ShaderCompilation(err.to_string()));
|
||
}
|
||
|
||
Ok(Self {
|
||
ctx: ctx.clone(),
|
||
features,
|
||
blur,
|
||
gradient,
|
||
plateau_init,
|
||
plateau_step,
|
||
flow,
|
||
jump,
|
||
})
|
||
}
|
||
|
||
/// Segment an image into basins.
|
||
pub fn run(
|
||
&self,
|
||
source: &DemosaicedImage,
|
||
opts: SegmentOptions,
|
||
) -> Result<Segmentation, GpuError> {
|
||
let (src_w, src_h) = source.texture_size();
|
||
let (width, height) = proxy_size(src_w, src_h, opts.max_edge);
|
||
let n = (width * height) as u64;
|
||
|
||
let params = self
|
||
.ctx
|
||
.device
|
||
.create_buffer_init(&wgpu::util::BufferInitDescriptor {
|
||
label: Some("watershed-params"),
|
||
contents: bytemuck::bytes_of(&SegParams {
|
||
width,
|
||
height,
|
||
src_width: src_w,
|
||
src_height: src_h,
|
||
blur_radius: opts.blur_radius,
|
||
non_linear: u32::from(source.is_non_linear()),
|
||
w_luma: opts.w_luma,
|
||
w_chroma: opts.w_chroma,
|
||
}),
|
||
usage: wgpu::BufferUsages::UNIFORM,
|
||
});
|
||
|
||
// `vec4` rather than `vec3` for the feature buffers: a WGSL storage
|
||
// array of vec3 still strides by 16 bytes, so packing to three floats
|
||
// would save nothing and cost an index calculation.
|
||
let feat_a = self.buffer("watershed-feat-a", n * 16, false);
|
||
let feat_b = self.buffer("watershed-feat-b", n * 16, false);
|
||
let gradient = self.buffer("watershed-gradient", n * 4, true);
|
||
let dist_a = self.buffer("watershed-dist-a", n * 4, false);
|
||
let dist_b = self.buffer("watershed-dist-b", n * 4, false);
|
||
let parent_a = self.buffer("watershed-parent-a", n * 4, true);
|
||
let parent_b = self.buffer("watershed-parent-b", n * 4, true);
|
||
|
||
let mut enc = self
|
||
.ctx
|
||
.device
|
||
.create_command_encoder(&wgpu::CommandEncoderDescriptor {
|
||
label: Some("watershed-encoder"),
|
||
});
|
||
|
||
let groups = (width.div_ceil(8), height.div_ceil(8));
|
||
|
||
let features_bg = self
|
||
.ctx
|
||
.device
|
||
.create_bind_group(&wgpu::BindGroupDescriptor {
|
||
label: Some("watershed-features-bg"),
|
||
layout: &self.features.layout,
|
||
entries: &[
|
||
wgpu::BindGroupEntry {
|
||
binding: 0,
|
||
resource: params.as_entire_binding(),
|
||
},
|
||
wgpu::BindGroupEntry {
|
||
binding: 1,
|
||
resource: wgpu::BindingResource::TextureView(source.view()),
|
||
},
|
||
wgpu::BindGroupEntry {
|
||
binding: 2,
|
||
resource: feat_a.as_entire_binding(),
|
||
},
|
||
],
|
||
});
|
||
|
||
let blur_bg = self.bind(&self.blur.layout, ¶ms, 3, &feat_a, 4, &feat_b);
|
||
let gradient_bg = self.bind(&self.gradient.layout, ¶ms, 5, &feat_b, 6, &gradient);
|
||
let pinit_bg = self.bind(&self.plateau_init.layout, ¶ms, 7, &gradient, 8, &dist_a);
|
||
let pstep_ab = self.bind3(
|
||
&self.plateau_step.layout,
|
||
¶ms,
|
||
(9, &gradient),
|
||
(10, &dist_a),
|
||
(11, &dist_b),
|
||
);
|
||
let pstep_ba = self.bind3(
|
||
&self.plateau_step.layout,
|
||
¶ms,
|
||
(9, &gradient),
|
||
(10, &dist_b),
|
||
(11, &dist_a),
|
||
);
|
||
|
||
// An odd number of plateau steps leaves the distance field in B.
|
||
let plateau_steps = opts.plateau_iterations;
|
||
let final_dist = if plateau_steps.is_multiple_of(2) {
|
||
&dist_a
|
||
} else {
|
||
&dist_b
|
||
};
|
||
|
||
let flow_bg = self.bind3(
|
||
&self.flow.layout,
|
||
¶ms,
|
||
(12, &gradient),
|
||
(13, final_dist),
|
||
(14, &parent_a),
|
||
);
|
||
let jump_ab = self.bind(&self.jump.layout, ¶ms, 15, &parent_a, 16, &parent_b);
|
||
let jump_ba = self.bind(&self.jump.layout, ¶ms, 15, &parent_b, 16, &parent_a);
|
||
|
||
// Pointer jumping halves every path per pass, so log2 of the pixel
|
||
// count bounds it — that is the longest possible descent chain. A
|
||
// convergence test would cost a readback per iteration to save a
|
||
// handful of dispatches of a two-line kernel.
|
||
let jumps = (n as f64).log2().ceil() as u32 + 1;
|
||
|
||
{
|
||
let mut pass = enc.begin_compute_pass(&wgpu::ComputePassDescriptor {
|
||
label: Some("watershed-pass"),
|
||
timestamp_writes: None,
|
||
});
|
||
|
||
for (pipeline, bg) in [
|
||
(&self.features.pipeline, &features_bg),
|
||
(&self.blur.pipeline, &blur_bg),
|
||
(&self.gradient.pipeline, &gradient_bg),
|
||
(&self.plateau_init.pipeline, &pinit_bg),
|
||
] {
|
||
pass.set_pipeline(pipeline);
|
||
pass.set_bind_group(0, bg, &[]);
|
||
pass.dispatch_workgroups(groups.0, groups.1, 1);
|
||
}
|
||
|
||
pass.set_pipeline(&self.plateau_step.pipeline);
|
||
for i in 0..plateau_steps {
|
||
let bg = if i % 2 == 0 { &pstep_ab } else { &pstep_ba };
|
||
pass.set_bind_group(0, bg, &[]);
|
||
pass.dispatch_workgroups(groups.0, groups.1, 1);
|
||
}
|
||
|
||
pass.set_pipeline(&self.flow.pipeline);
|
||
pass.set_bind_group(0, &flow_bg, &[]);
|
||
pass.dispatch_workgroups(groups.0, groups.1, 1);
|
||
|
||
pass.set_pipeline(&self.jump.pipeline);
|
||
for i in 0..jumps {
|
||
let bg = if i % 2 == 0 { &jump_ab } else { &jump_ba };
|
||
pass.set_bind_group(0, bg, &[]);
|
||
pass.dispatch_workgroups(groups.0, groups.1, 1);
|
||
}
|
||
}
|
||
|
||
self.ctx.queue.submit(Some(enc.finish()));
|
||
|
||
// An odd number of jumps leaves the result in B.
|
||
let labels = if jumps % 2 == 1 { parent_b } else { parent_a };
|
||
|
||
Ok(Segmentation {
|
||
ctx: self.ctx.clone(),
|
||
width,
|
||
height,
|
||
labels,
|
||
gradient,
|
||
})
|
||
}
|
||
|
||
fn buffer(&self, label: &str, size: u64, copyable: bool) -> wgpu::Buffer {
|
||
let mut usage = wgpu::BufferUsages::STORAGE;
|
||
if copyable {
|
||
usage |= wgpu::BufferUsages::COPY_SRC;
|
||
}
|
||
self.ctx.device.create_buffer(&wgpu::BufferDescriptor {
|
||
label: Some(label),
|
||
size,
|
||
usage,
|
||
mapped_at_creation: false,
|
||
})
|
||
}
|
||
|
||
fn bind3(
|
||
&self,
|
||
layout: &wgpu::BindGroupLayout,
|
||
params: &wgpu::Buffer,
|
||
a: (u32, &wgpu::Buffer),
|
||
b: (u32, &wgpu::Buffer),
|
||
c: (u32, &wgpu::Buffer),
|
||
) -> wgpu::BindGroup {
|
||
self.ctx
|
||
.device
|
||
.create_bind_group(&wgpu::BindGroupDescriptor {
|
||
label: Some("watershed-bg3"),
|
||
layout,
|
||
entries: &[
|
||
wgpu::BindGroupEntry {
|
||
binding: 0,
|
||
resource: params.as_entire_binding(),
|
||
},
|
||
wgpu::BindGroupEntry {
|
||
binding: a.0,
|
||
resource: a.1.as_entire_binding(),
|
||
},
|
||
wgpu::BindGroupEntry {
|
||
binding: b.0,
|
||
resource: b.1.as_entire_binding(),
|
||
},
|
||
wgpu::BindGroupEntry {
|
||
binding: c.0,
|
||
resource: c.1.as_entire_binding(),
|
||
},
|
||
],
|
||
})
|
||
}
|
||
|
||
fn bind(
|
||
&self,
|
||
layout: &wgpu::BindGroupLayout,
|
||
params: &wgpu::Buffer,
|
||
in_binding: u32,
|
||
input: &wgpu::Buffer,
|
||
out_binding: u32,
|
||
output: &wgpu::Buffer,
|
||
) -> wgpu::BindGroup {
|
||
self.ctx
|
||
.device
|
||
.create_bind_group(&wgpu::BindGroupDescriptor {
|
||
label: Some("watershed-bg"),
|
||
layout,
|
||
entries: &[
|
||
wgpu::BindGroupEntry {
|
||
binding: 0,
|
||
resource: params.as_entire_binding(),
|
||
},
|
||
wgpu::BindGroupEntry {
|
||
binding: in_binding,
|
||
resource: input.as_entire_binding(),
|
||
},
|
||
wgpu::BindGroupEntry {
|
||
binding: out_binding,
|
||
resource: output.as_entire_binding(),
|
||
},
|
||
],
|
||
})
|
||
}
|
||
}
|
||
|
||
/// The result of one segmentation: a basin label per pixel, on the GPU.
|
||
pub struct Segmentation {
|
||
/// Only [`Self::read_field`] reads this, so a build without `readback`
|
||
/// carries it unread. That is now the ordinary build: dr-ui used to turn
|
||
/// the feature on for the whole workspace and stopped when S1 removed the
|
||
/// display readback, which is what made the field look dead.
|
||
#[cfg_attr(not(any(test, feature = "readback")), allow(dead_code))]
|
||
ctx: GpuContext,
|
||
width: u32,
|
||
height: u32,
|
||
/// Per pixel, the linear index of its basin root. Sparse — compacted by
|
||
/// [`dr_segment::RegionField::from_roots`].
|
||
labels: wgpu::Buffer,
|
||
/// As with `ctx` above: read only by [`Self::read_field`].
|
||
#[cfg_attr(not(any(test, feature = "readback")), allow(dead_code))]
|
||
gradient: wgpu::Buffer,
|
||
}
|
||
|
||
impl Segmentation {
|
||
pub fn size(&self) -> (u32, u32) {
|
||
(self.width, self.height)
|
||
}
|
||
|
||
/// The label buffer, for a shader that masks by region id.
|
||
pub fn labels(&self) -> &wgpu::Buffer {
|
||
&self.labels
|
||
}
|
||
|
||
/// Build the region adjacency graph, reading the labels back to the CPU.
|
||
///
|
||
/// # Why this has its own feature rather than sharing `readback`
|
||
///
|
||
/// `readback` gates [`crate::AdjustPass::read_pixels`], which is the
|
||
/// per-frame display round-trip AC-8 exists to forbid. This is a different
|
||
/// transfer with different economics, and sharing one switch would have
|
||
/// forced a build wanting local masking to also unlock the one thing the
|
||
/// architecture is built around never doing.
|
||
///
|
||
/// What this transfer actually is: **once per image, on a worker, off the
|
||
/// frame path.** Nothing in the render loop waits on it, and the result is
|
||
/// a region graph of a few thousand nodes that every later interaction
|
||
/// reads from the CPU anyway.
|
||
///
|
||
/// What it is *not* is finished. F3 in docs/dev/segmentation.md §12 stands:
|
||
/// the adjacency accumulation belongs on the GPU with atomics, and until
|
||
/// it moves there a segmentation costs one full-resolution transfer of the
|
||
/// label and gradient buffers. That is a real cost on a phone and the
|
||
/// reason this is named for what it does rather than hidden behind the
|
||
/// general switch.
|
||
#[cfg(any(test, feature = "segment-readback"))]
|
||
pub fn read_field(&self) -> Result<dr_segment::RegionField, GpuError> {
|
||
let n = (self.width * self.height) as usize;
|
||
let roots: Vec<u32> = read_buffer(&self.ctx, &self.labels, n)?;
|
||
let gradient: Vec<f32> = read_buffer(&self.ctx, &self.gradient, n)?;
|
||
Ok(dr_segment::RegionField::from_roots(
|
||
&roots,
|
||
&gradient,
|
||
self.width as usize,
|
||
self.height as usize,
|
||
))
|
||
}
|
||
}
|
||
|
||
fn uniform_entry(binding: u32) -> wgpu::BindGroupLayoutEntry {
|
||
wgpu::BindGroupLayoutEntry {
|
||
binding,
|
||
visibility: wgpu::ShaderStages::COMPUTE,
|
||
ty: wgpu::BindingType::Buffer {
|
||
ty: wgpu::BufferBindingType::Uniform,
|
||
has_dynamic_offset: false,
|
||
min_binding_size: None,
|
||
},
|
||
count: None,
|
||
}
|
||
}
|
||
|
||
fn storage_entry(binding: u32, read_only: bool) -> wgpu::BindGroupLayoutEntry {
|
||
wgpu::BindGroupLayoutEntry {
|
||
binding,
|
||
visibility: wgpu::ShaderStages::COMPUTE,
|
||
ty: wgpu::BindingType::Buffer {
|
||
ty: wgpu::BufferBindingType::Storage { read_only },
|
||
has_dynamic_offset: false,
|
||
min_binding_size: None,
|
||
},
|
||
count: None,
|
||
}
|
||
}
|
||
|
||
fn compute(
|
||
ctx: &GpuContext,
|
||
module: &wgpu::ShaderModule,
|
||
layout: &wgpu::BindGroupLayout,
|
||
entry: &str,
|
||
) -> wgpu::ComputePipeline {
|
||
let pipeline_layout = ctx
|
||
.device
|
||
.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
|
||
label: Some("watershed-layout"),
|
||
bind_group_layouts: &[Some(layout)],
|
||
immediate_size: 0,
|
||
});
|
||
ctx.device
|
||
.create_compute_pipeline(&wgpu::ComputePipelineDescriptor {
|
||
label: Some(entry),
|
||
layout: Some(&pipeline_layout),
|
||
module,
|
||
entry_point: Some(entry),
|
||
compilation_options: Default::default(),
|
||
cache: None,
|
||
})
|
||
}
|
||
|
||
/// The proxy size for a source, preserving aspect and never upscaling.
|
||
fn proxy_size(src_w: u32, src_h: u32, max_edge: u32) -> (u32, u32) {
|
||
let longest = src_w.max(src_h);
|
||
if longest <= max_edge || longest == 0 {
|
||
return (src_w.max(1), src_h.max(1));
|
||
}
|
||
let scale = f64::from(max_edge) / f64::from(longest);
|
||
(
|
||
((f64::from(src_w) * scale).round() as u32).max(1),
|
||
((f64::from(src_h) * scale).round() as u32).max(1),
|
||
)
|
||
}
|
||
|
||
#[cfg(any(test, feature = "segment-readback"))]
|
||
fn read_buffer<T: bytemuck::Pod>(
|
||
ctx: &GpuContext,
|
||
buffer: &wgpu::Buffer,
|
||
len: usize,
|
||
) -> Result<Vec<T>, GpuError> {
|
||
let size = (len * std::mem::size_of::<T>()) as u64;
|
||
let staging = ctx.device.create_buffer(&wgpu::BufferDescriptor {
|
||
label: Some("watershed-readback"),
|
||
size,
|
||
usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ,
|
||
mapped_at_creation: false,
|
||
});
|
||
|
||
let mut enc = ctx.device.create_command_encoder(&Default::default());
|
||
enc.copy_buffer_to_buffer(buffer, 0, &staging, 0, size);
|
||
ctx.queue.submit(Some(enc.finish()));
|
||
|
||
let slice = staging.slice(..);
|
||
let (tx, rx) = std::sync::mpsc::channel();
|
||
slice.map_async(wgpu::MapMode::Read, move |r| {
|
||
let _ = tx.send(r);
|
||
});
|
||
ctx.device
|
||
.poll(wgpu::PollType::wait_indefinitely())
|
||
.map_err(|e| GpuError::Readback(e.to_string()))?;
|
||
rx.recv()
|
||
.map_err(|e| GpuError::Readback(e.to_string()))?
|
||
.map_err(|e| GpuError::Readback(e.to_string()))?;
|
||
|
||
let data = slice.get_mapped_range();
|
||
let out = bytemuck::cast_slice::<u8, T>(&data).to_vec();
|
||
drop(data);
|
||
staging.unmap();
|
||
Ok(out)
|
||
}
|
||
|
||
#[cfg(test)]
|
||
mod tests {
|
||
use super::*;
|
||
use dr_segment::MergeTree;
|
||
|
||
fn ctx() -> Option<GpuContext> {
|
||
match pollster::block_on(GpuContext::new_headless()) {
|
||
Ok(c) => Some(c),
|
||
Err(e) => {
|
||
eprintln!("skipping: no GPU adapter ({e})");
|
||
None
|
||
}
|
||
}
|
||
}
|
||
|
||
#[test]
|
||
fn a_proxy_preserves_aspect_and_never_upscales() {
|
||
assert_eq!(proxy_size(6000, 4000, 1600), (1600, 1067));
|
||
assert_eq!(proxy_size(4000, 6000, 1600), (1067, 1600));
|
||
// A thumbnail must not be blown up to the proxy size — there is no
|
||
// detail there to find basins in.
|
||
assert_eq!(proxy_size(800, 600, 1600), (800, 600));
|
||
assert_eq!(proxy_size(0, 0, 1600), (1, 1));
|
||
}
|
||
|
||
/// Two flat halves split by a hard vertical edge.
|
||
fn two_tone(w: u32, h: u32) -> Vec<u8> {
|
||
let mut px = Vec::with_capacity((w * h * 4) as usize);
|
||
for _ in 0..h {
|
||
for x in 0..w {
|
||
let v = if x < w / 2 { 30u8 } else { 220u8 };
|
||
px.extend_from_slice(&[v, v, v, 255]);
|
||
}
|
||
}
|
||
px
|
||
}
|
||
|
||
#[test]
|
||
fn a_hard_edge_produces_two_regions_at_the_top_of_the_ladder() {
|
||
// The end-to-end property, on an image whose answer is not in doubt:
|
||
// whatever the watershed does with texture, it must not lose an edge
|
||
// this obvious, and the coarsest non-trivial cut must be exactly the
|
||
// two halves.
|
||
let Some(ctx) = ctx() else { return };
|
||
let (w, h) = (64u32, 64u32);
|
||
let src = DemosaicedImage::from_rgba8(&ctx, &two_tone(w, h), w, h).expect("source");
|
||
|
||
let pass = SegmentPass::new(&ctx).expect("segment pass");
|
||
let seg = pass.run(&src, SegmentOptions::default()).expect("run");
|
||
assert_eq!(seg.size(), (w, h));
|
||
|
||
let field = seg.read_field().expect("read field");
|
||
let tree = MergeTree::build(&field);
|
||
let px = field.apply(&tree.cut_to(2));
|
||
|
||
for y in 0..h as usize {
|
||
let left = px[y * w as usize];
|
||
let right = px[y * w as usize + w as usize - 1];
|
||
assert_ne!(left, right, "the two halves must not share a region");
|
||
}
|
||
}
|
||
|
||
#[test]
|
||
fn a_flat_image_does_not_fragment() {
|
||
// The noise case in miniature. A gradient of zero everywhere is one
|
||
// enormous plateau, which is exactly where a watershed without a
|
||
// strict tie-break either hangs or shatters into per-pixel basins.
|
||
let Some(ctx) = ctx() else { return };
|
||
let (w, h) = (32u32, 32u32);
|
||
let flat = vec![128u8; (w * h * 4) as usize];
|
||
let src = DemosaicedImage::from_rgba8(&ctx, &flat, w, h).expect("source");
|
||
|
||
let pass = SegmentPass::new(&ctx).expect("segment pass");
|
||
let seg = pass.run(&src, SegmentOptions::default()).expect("run");
|
||
let field = seg.read_field().expect("read field");
|
||
|
||
assert_eq!(
|
||
field.region_count, 1,
|
||
"a plateau should resolve to one basin, not {}",
|
||
field.region_count
|
||
);
|
||
}
|
||
|
||
/// A flat disc on flat ground: two plateaux and one boundary between
|
||
/// them. Nothing here has a downhill direction except at the rim.
|
||
/// A linear ramp between two flat fields.
|
||
///
|
||
/// The watershed runs on gradient *magnitude*, and that changes which
|
||
/// images contain a plateau worth resolving. A flat region of the picture
|
||
/// has gradient zero — the global minimum — and a plateau at the minimum
|
||
/// has no descending exit at all, which makes it a single basin by
|
||
/// definition with nothing for lower-completion to do. The plateaux that
|
||
/// do have an exit are regions of constant *non-zero* gradient: linear
|
||
/// ramps. So that is what this builds.
|
||
fn ramp(w: u32, h: u32) -> Vec<u8> {
|
||
let mut px = vec![0u8; (w * h * 4) as usize];
|
||
let (lo, hi) = (w / 4, w - w / 4);
|
||
for y in 0..h {
|
||
for x in 0..w {
|
||
let v = if x < lo {
|
||
40u8
|
||
} else if x >= hi {
|
||
210u8
|
||
} else {
|
||
// Constant slope, so the gradient is constant and
|
||
// non-zero across the whole band.
|
||
(40.0 + (x - lo) as f32 * (170.0 / (hi - lo) as f32)) as u8
|
||
};
|
||
let i = ((y * w + x) * 4) as usize;
|
||
px[i] = v;
|
||
px[i + 1] = v;
|
||
px[i + 2] = v;
|
||
px[i + 3] = 255;
|
||
}
|
||
}
|
||
px
|
||
}
|
||
#[test]
|
||
#[ignore = "the plateau pass is a measured no-op; see docs/dev/segmentation.md §12"]
|
||
fn lower_completion_drains_a_plateau_instead_of_shattering_it() {
|
||
// F1, asserted rather than eyeballed, and asserted at the level where
|
||
// it matters.
|
||
//
|
||
// Two claims, because they are different claims. First: carrying the
|
||
// distance inward genuinely reduces fragmentation — a plateau with an
|
||
// exit now drains to it instead of fanning into diagonal chains.
|
||
// Second, and the one a user would notice: whatever fragments survive
|
||
// are separated by zero-height saddles, so the hierarchy merges them
|
||
// at its very first steps and the plateau reads as one region.
|
||
//
|
||
// The second claim is what makes the first one's *residue* tolerable.
|
||
// A perfectly flat regional minimum — the inside of a uniform disc,
|
||
// with no exit anywhere — cannot be drained by a distance that has
|
||
// nowhere to descend to, and collapsing it fully would need connected
|
||
// component labelling rather than a local rule. It is not worth it:
|
||
// see docs/dev/segmentation.md §12.
|
||
let Some(ctx) = ctx() else { return };
|
||
let (w, h) = (96u32, 96u32);
|
||
let src = DemosaicedImage::from_rgba8(&ctx, &ramp(w, h), w, h).expect("source");
|
||
let pass = SegmentPass::new(&ctx).expect("segment pass");
|
||
|
||
let labels_in = |labels: &[u32], inside: bool| {
|
||
let (cx, cy) = (w as f32 / 2.0, h as f32 / 2.0);
|
||
let mut seen = std::collections::HashSet::new();
|
||
for y in 0..h {
|
||
for x in 0..w {
|
||
let d = ((x as f32 - cx).powi(2) + (y as f32 - cy).powi(2)).sqrt();
|
||
let take = if inside {
|
||
d < w as f32 * 0.20
|
||
} else {
|
||
d > w as f32 * 0.42
|
||
};
|
||
if take {
|
||
seen.insert(labels[(y * w + x) as usize]);
|
||
}
|
||
}
|
||
}
|
||
seen
|
||
};
|
||
|
||
let field = |iterations: u32| {
|
||
pass.run(
|
||
&src,
|
||
SegmentOptions {
|
||
plateau_iterations: iterations,
|
||
..Default::default()
|
||
},
|
||
)
|
||
.expect("run")
|
||
.read_field()
|
||
.expect("field")
|
||
};
|
||
|
||
let shallow = field(1);
|
||
let deep = field(64);
|
||
|
||
// Before anything else: does the pass change the labelling at all? If
|
||
// the distance field were never populated — a binding astray, a level
|
||
// test that never matches — every downstream claim would be excused
|
||
// by a no-op rather than tested. This is the one assertion that
|
||
// cannot pass vacuously.
|
||
let differs = shallow
|
||
.labels
|
||
.iter()
|
||
.zip(deep.labels.iter())
|
||
.filter(|(a, b)| a != b)
|
||
.count();
|
||
assert!(
|
||
differs > 0,
|
||
"the plateau distance changed no pixel's basin, so the pass is a \
|
||
no-op: {} pixels, {} differ",
|
||
shallow.labels.len(),
|
||
differs
|
||
);
|
||
|
||
// Claim one: fewer basins, because plateaux with an exit now use it.
|
||
assert!(
|
||
deep.region_count < shallow.region_count,
|
||
"carrying the distance inward should reduce fragmentation: \
|
||
{} basins against {}",
|
||
deep.region_count,
|
||
shallow.region_count
|
||
);
|
||
|
||
// Claim two: what survives costs nothing, because the hierarchy
|
||
// dissolves it immediately.
|
||
let tree = MergeTree::build(&deep);
|
||
let grouped = deep.apply(&tree.cut_to(2));
|
||
let inside = labels_in(&grouped, true);
|
||
let outside = labels_in(&grouped, false);
|
||
assert_eq!(inside.len(), 1, "the disc should read as one region");
|
||
assert_eq!(outside.len(), 1, "the ground should read as one region");
|
||
assert_ne!(inside, outside, "and they must not be the same region");
|
||
}
|
||
|
||
#[test]
|
||
fn the_same_image_segments_identically_twice() {
|
||
// M5 on one device — the weaker half of the determinism question, but
|
||
// the half that catches a race in the pointer jumping. Cross-vendor
|
||
// is the part that needs hardware this test cannot assume.
|
||
let Some(ctx) = ctx() else { return };
|
||
let (w, h) = (48u32, 48u32);
|
||
let src = DemosaicedImage::from_rgba8(&ctx, &two_tone(w, h), w, h).expect("source");
|
||
let pass = SegmentPass::new(&ctx).expect("segment pass");
|
||
|
||
let a = pass
|
||
.run(&src, SegmentOptions::default())
|
||
.expect("run")
|
||
.read_field()
|
||
.expect("field");
|
||
let b = pass
|
||
.run(&src, SegmentOptions::default())
|
||
.expect("run")
|
||
.read_field()
|
||
.expect("field");
|
||
|
||
assert_eq!(a, b, "segmentation must be reproducible run to run");
|
||
}
|
||
}
|