Merge branch 'worktree-agent-a75dc051d9bf691de' into integration

# Conflicts:
#	docs/traceability.md
This commit is contained in:
2026-08-22 15:34:29 +02:00
13 changed files with 3060 additions and 93 deletions
+6
View File
@@ -25,6 +25,12 @@ pollster.workspace = true
[dev-dependencies]
env_logger.workspace = true
# The detail stage's test consumer — a box blur that is not a develop operation
# and never reaches the panel. An abstraction with no consumers is a guess, and
# this is the one that proves the neighbourhood passes compile, ping-pong,
# encode once, and scale between a proxy and an export. A dev-dependency, so a
# shipping `dr-gpu` does not carry it.
dr-pipeline = { workspace = true, features = ["detail-probe"] }
# The local-adjustment example needs the model, which the library half of this
# crate deliberately does not: `dr-gpu` holds the shaders, and the inference
# runtime belongs to whoever is asking a question about the picture.
+424 -74
View File
@@ -16,9 +16,11 @@
use std::collections::HashMap;
use dr_pipeline::ComposedShader;
use dr_pipeline::detail::ComposedDetail;
use dr_pipeline::{ComposedShader, OutputMode};
use wgpu::util::DeviceExt;
use crate::detail::DetailRunner;
use crate::readback::await_mapping;
use crate::{DemosaicedImage, GpuContext, GpuError};
@@ -62,6 +64,44 @@ pub struct AdjustPass {
current: usize,
/// Bound at `@binding(3)` when the edit carries no mask layers.
empty_masks: wgpu::TextureView,
/// TRACES: FR-DEV-3 | FR-DEV-3d
/// The neighbourhood stage — sharpening, noise reduction, clarity and the
/// rest of FR-DEV-3's detail set, which cannot be fused into the shader
/// above because they read pixels they are not writing.
///
/// It lives here rather than beside this pass because the two are one
/// render: when a detail chain is present the fused pass writes a linear
/// intermediate the runner owns, and the runner's last pass writes
/// [`Self::targets`]. Kept as separate objects, a caller could hold a
/// stale intermediate against a fresh colour result with nothing to tell
/// it apart.
detail: DetailRunner,
/// The bind group layout for a fused pass writing a linear intermediate.
///
/// A second layout rather than a second pass: the only difference is the
/// storage texture's format, which is part of the layout and cannot be
/// varied per bind group. Built once here, so a detail operation being
/// switched on does not build a pipeline layout mid-frame.
linear_bind_group_layout: wgpu::BindGroupLayout,
linear_pipeline_layout: wgpu::PipelineLayout,
/// TRACES: FR-DEV-3d
/// What the linear intermediate currently holds, and at what size.
///
/// **This is where `Affects::Detail` stops being bookkeeping.** The key is
/// everything the fused dispatch depends on — the caller's
/// `Invalidation::through(Affects::Colour)`, the compiled structure, the
/// uniform values and the output size. When it matches, the colour pass is
/// skipped and only the detail passes run, so dragging a sharpening slider
/// costs a convolution and not a re-render of the whole chain (FR-DEV-3d).
///
/// Cleared by any render that does not write it, so a stale intermediate
/// cannot survive a change of image and be handed to a later detail chain.
colour_key: Option<(u64, u32, u32)>,
/// Fused dispatches actually encoded. Exposed so a test can see the reuse
/// above happening rather than take it on trust.
colour_dispatches: usize,
/// Detail dispatches encoded.
detail_dispatches: usize,
}
struct Target {
@@ -75,58 +115,7 @@ impl AdjustPass {
pub const FORMAT: wgpu::TextureFormat = wgpu::TextureFormat::Rgba8Unorm;
pub fn new(ctx: &GpuContext) -> Self {
let bind_group_layout =
ctx.device
.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
label: Some("adjust-bgl"),
entries: &[
// The demosaiced source.
wgpu::BindGroupLayoutEntry {
binding: 0,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::Texture {
sample_type: wgpu::TextureSampleType::Float { filterable: true },
view_dimension: wgpu::TextureViewDimension::D2,
multisampled: false,
},
count: None,
},
wgpu::BindGroupLayoutEntry {
binding: 1,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::Buffer {
ty: wgpu::BufferBindingType::Uniform,
has_dynamic_offset: false,
min_binding_size: None,
},
count: None,
},
wgpu::BindGroupLayoutEntry {
binding: 2,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::StorageTexture {
access: wgpu::StorageTextureAccess::WriteOnly,
format: Self::FORMAT,
view_dimension: wgpu::TextureViewDimension::D2,
},
count: None,
},
// The local-adjustment masks. Present in every layout
// whether or not the edit has any, because the layout
// is built once here and the generated shader declares
// the binding unconditionally for exactly that reason.
wgpu::BindGroupLayoutEntry {
binding: 3,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::Texture {
sample_type: wgpu::TextureSampleType::Float { filterable: true },
view_dimension: wgpu::TextureViewDimension::D2Array,
multisampled: false,
},
count: None,
},
],
});
let bind_group_layout = Self::layout_writing(ctx, Self::FORMAT, "adjust-bgl");
let pipeline_layout = ctx
.device
@@ -136,6 +125,23 @@ impl AdjustPass {
immediate_size: 0,
});
// The same layout with an `Rgba16Float` storage texture, for the fused
// pass when a detail stage follows it and it hands on linear working
// values instead of encoding (see `dr_pipeline::OutputMode`). The
// format is part of a bind group layout and cannot be varied per bind
// group, so this is a second layout rather than a second binding —
// built here, once, so that switching sharpening on does not construct
// a pipeline layout in the middle of a frame.
let linear_bind_group_layout =
Self::layout_writing(ctx, crate::detail::INTERMEDIATE_FORMAT, "adjust-linear-bgl");
let linear_pipeline_layout =
ctx.device
.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
label: Some("adjust-linear-layout"),
bind_group_layouts: &[Some(&linear_bind_group_layout)],
immediate_size: 0,
});
// A 1x1 single-layer mask, bound when the edit has no local
// adjustments. The generated shader never samples it — no layer block
// is emitted — but a bind group must still satisfy the layout.
@@ -167,9 +173,81 @@ impl AdjustPass {
targets: [None, None],
current: 0,
empty_masks,
detail: DetailRunner::new(ctx),
linear_bind_group_layout,
linear_pipeline_layout,
colour_key: None,
colour_dispatches: 0,
detail_dispatches: 0,
}
}
/// The fused pass's bind group layout, for a given storage format.
///
/// Two of these exist — one writing `Rgba8Unorm` and one writing
/// `Rgba16Float` — and they differ in exactly one field. Written once and
/// parameterised rather than copied, because two copies of a four-entry
/// layout is how the mask binding comes to be present in one and absent
/// from the other, and a bind group that satisfies neither is a validation
/// error a long way from its cause.
fn layout_writing(
ctx: &GpuContext,
format: wgpu::TextureFormat,
label: &str,
) -> wgpu::BindGroupLayout {
ctx.device
.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
label: Some(label),
entries: &[
// The demosaiced source.
wgpu::BindGroupLayoutEntry {
binding: 0,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::Texture {
sample_type: wgpu::TextureSampleType::Float { filterable: true },
view_dimension: wgpu::TextureViewDimension::D2,
multisampled: false,
},
count: None,
},
wgpu::BindGroupLayoutEntry {
binding: 1,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::Buffer {
ty: wgpu::BufferBindingType::Uniform,
has_dynamic_offset: false,
min_binding_size: None,
},
count: None,
},
wgpu::BindGroupLayoutEntry {
binding: 2,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::StorageTexture {
access: wgpu::StorageTextureAccess::WriteOnly,
format,
view_dimension: wgpu::TextureViewDimension::D2,
},
count: None,
},
// The local-adjustment masks. Present in every layout
// whether or not the edit has any, because the layout is
// built once here and the generated shader declares the
// binding unconditionally for exactly that reason.
wgpu::BindGroupLayoutEntry {
binding: 3,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::Texture {
sample_type: wgpu::TextureSampleType::Float { filterable: true },
view_dimension: wgpu::TextureViewDimension::D2Array,
multisampled: false,
},
count: None,
},
],
})
}
/// Compile a composed shader, or return the cached pipeline.
///
/// Compilation errors carry the generated source, since a stray line
@@ -197,12 +275,22 @@ impl AdjustPass {
source: wgpu::ShaderSource::Wgsl(shader.source.as_str().into()),
});
// The layout matching what this shader was composed to write. The
// structure hash covers the generated source and the source
// carries the storage format, so the two can never disagree — a
// cached pipeline is always paired with the layout it was built
// against.
let layout = match shader.output_mode {
OutputMode::Encoded => &self.pipeline_layout,
OutputMode::LinearWorking => &self.linear_pipeline_layout,
};
let pipeline =
self.ctx
.device
.create_compute_pipeline(&wgpu::ComputePipelineDescriptor {
label: Some("adjust-pipeline"),
layout: Some(&self.pipeline_layout),
layout: Some(layout),
module: &module,
entry_point: Some("main"),
compilation_options: Default::default(),
@@ -310,28 +398,31 @@ impl AdjustPass {
height: u32,
masks: Option<&crate::MaskArray>,
) -> Result<&wgpu::Texture, GpuError> {
if shader.output_mode != OutputMode::Encoded {
// Composed for a detail stage and dispatched without one. The
// shader writes `rgba16float` and this path binds an `rgba8unorm`
// storage texture, which wgpu rejects — but well after the point
// where the mistake is legible. Saying so here names the actual
// error: the edit has a neighbourhood operation and needs
// `render_detailed`.
return Err(GpuError::ShaderCompilation(
"this shader was composed with a detail stage and writes linear \
working values; render it with `render_detailed` and the \
matching chain from `EditGraph::compose_detail`"
.into(),
));
}
// Any render that does not write the linear intermediate leaves
// whatever is in it belonging to some other edit — or some other
// photograph. Forgetting this is how a detail chain comes to be run
// over a stale colour result, so the key is dropped rather than
// reasoned about.
self.colour_key = None;
let (width, height) = (width.max(1), height.max(1));
self.ensure_target(width, height);
// Base uniforms: the camera matrix and as-shot white balance, which
// every generated shader reads regardless of which operations are
// active. Framing's slots follow them and are filled by the composer,
// which is why only the first sixteen are written here.
let mut uniforms = shader.uniforms.clone();
if uniforms.len() < RESERVED_FIELDS {
uniforms.resize(RESERVED_FIELDS, 0.0);
}
let m = source.color_matrix();
let wb = source.as_shot_wb();
// Rows padded to vec4 for std140 alignment.
uniforms[0..4].copy_from_slice(&[m[0], m[1], m[2], 0.0]);
uniforms[4..8].copy_from_slice(&[m[3], m[4], m[5], 0.0]);
uniforms[8..12].copy_from_slice(&[m[6], m[7], m[8], 0.0]);
// The fourth slot is the non-linear flag, not padding: it tells the
// shader whether to linearise the sampled texel before any operation
// runs. See `DemosaicedImage::is_non_linear`.
let non_linear = if source.is_non_linear() { 1.0 } else { 0.0 };
uniforms[12..16].copy_from_slice(&[wb[0], wb[1], wb[2], non_linear]);
let uniforms = Self::fused_uniforms(source, shader);
let params_buf = self
.ctx
@@ -394,6 +485,7 @@ impl AdjustPass {
pass.dispatch_workgroups(width.div_ceil(8), height.div_ceil(8), 1);
}
self.ctx.queue.submit(Some(enc.finish()));
self.colour_dispatches += 1;
Ok(&self.targets[self.current]
.as_ref()
@@ -401,12 +493,270 @@ impl AdjustPass {
.texture)
}
/// TRACES: FR-DEV-3 | FR-DEV-3d | FR-DEV-4 | FR-DSP-1
/// Render one frame with a neighbourhood stage.
///
/// `shader` and `detail` must be the two halves of **one** composition —
/// `EditGraph::compose_for` and `EditGraph::compose_detail_for` on the same
/// graph, at the same output space. The fused pass stops at linear working
/// values when a detail stage exists and the last detail pass performs the
/// output transform, so a mismatched pair either encodes twice or not at
/// all.
///
/// An empty `detail` falls through to [`Self::render_masked`], which is
/// the honest thing to do rather than an optimisation: an edit with no
/// active sharpening *is* an ordinary edit, and it should cost exactly
/// what one costs.
///
/// # `colour_key`, and why the caller supplies it
///
/// It is `Invalidation::through(Affects::Colour)` for this edit, mixed
/// with whatever names the photograph — a `VersionId`, typically. When it
/// is unchanged, and the size and the composed shader and its uniforms are
/// unchanged with it, the fused dispatch is **skipped** and the linear
/// intermediate from the previous frame is convolved again. Dragging a
/// sharpening slider then costs the detail passes alone, which is the
/// reuse FR-DEV-3d asks for and the operational meaning of
/// `Affects::Detail`.
///
/// The caller supplies it rather than this pass deriving it because only
/// the caller knows which *image* is on screen. Everything else that goes
/// into the fused dispatch — the shader's structure, its uniform values,
/// the output size — is mixed in here, so a caller cannot make the reuse
/// unsound by supplying a key that is merely coarse. It can only do so by
/// supplying one that fails to distinguish two photographs, which is why
/// the identity of the image is spelled out as its job.
// Eight arguments, and every one of them is a distinct thing the render
// depends on: the image, both halves of the composition, the size, the
// masks and the cache key. Bundling them into a struct would move the
// problem rather than solve it — the caller would fill in the same eight
// fields — and would hide that composing the two halves apart is the one
// mistake this signature exists to make visible.
#[allow(clippy::too_many_arguments)]
pub fn render_detailed(
&mut self,
source: &DemosaicedImage,
shader: &ComposedShader,
width: u32,
height: u32,
masks: Option<&crate::MaskArray>,
detail: &ComposedDetail,
colour_key: u64,
) -> Result<&wgpu::Texture, GpuError> {
if detail.is_empty() {
return self.render_masked(source, shader, width, height, masks);
}
if shader.output_mode != OutputMode::LinearWorking {
return Err(GpuError::ShaderCompilation(
"this detail chain expects a fused pass composed to hand on \
linear working values, but the shader given encodes its own \
output; compose both halves from the same graph"
.into(),
));
}
let (width, height) = (width.max(1), height.max(1));
self.ensure_target(width, height);
let uniforms = Self::fused_uniforms(source, shader);
let key = Self::colour_signature(colour_key, shader, &uniforms, masks);
let reuse = self.colour_key == Some((key, width, height));
// Compile before borrowing anything: `pipeline` and `colour_target`
// both want `&mut self`, and the second holds its borrow across the
// encode below.
self.pipeline(shader)?;
let colour_view = self
.detail
.colour_target(detail.len(), width, height)
.clone();
let mut enc = self
.ctx
.device
.create_command_encoder(&wgpu::CommandEncoderDescriptor {
label: Some("adjust-detail-encoder"),
});
if !reuse {
let params_buf = self
.ctx
.device
.create_buffer_init(&wgpu::util::BufferInitDescriptor {
label: Some("adjust-params"),
contents: bytemuck::cast_slice(&uniforms),
usage: wgpu::BufferUsages::UNIFORM,
});
let bind_group = self
.ctx
.device
.create_bind_group(&wgpu::BindGroupDescriptor {
label: Some("adjust-linear-bg"),
layout: &self.linear_bind_group_layout,
entries: &[
wgpu::BindGroupEntry {
binding: 0,
resource: wgpu::BindingResource::TextureView(source.view()),
},
wgpu::BindGroupEntry {
binding: 1,
resource: params_buf.as_entire_binding(),
},
wgpu::BindGroupEntry {
binding: 2,
resource: wgpu::BindingResource::TextureView(&colour_view),
},
wgpu::BindGroupEntry {
binding: 3,
resource: wgpu::BindingResource::TextureView(
masks.map_or(&self.empty_masks, |m| m.view()),
),
},
],
});
let pipeline = self
.cache
.get(&shader.structure_hash)
.expect("compiled above");
let mut pass = enc.begin_compute_pass(&wgpu::ComputePassDescriptor {
label: Some("adjust-pass"),
timestamp_writes: None,
});
pass.set_pipeline(pipeline);
pass.set_bind_group(0, &bind_group, &[]);
pass.dispatch_workgroups(width.div_ceil(8), height.div_ceil(8), 1);
drop(pass);
self.colour_dispatches += 1;
}
// One encoder for the colour pass and every detail pass, submitted
// once — the shape `MaskPass::render` established. Submission order is
// the whole of the synchronisation: each pass reads what the previous
// one wrote, through the same queue.
let target_view = self.targets[self.current]
.as_ref()
.expect("ensured above")
.view
.clone();
let ran = self
.detail
.encode(&mut enc, detail, &target_view, width, height)?;
self.ctx.queue.submit(Some(enc.finish()));
self.detail_dispatches += ran;
self.colour_key = Some((key, width, height));
Ok(&self.targets[self.current]
.as_ref()
.expect("ensured above")
.texture)
}
/// The fused pass's uniform block, with the source's own values written in.
///
/// Split out because both render paths need exactly this and a second copy
/// would eventually disagree about where the camera matrix goes — which is
/// silent, and corrupts every operation's uniforms downstream of it.
fn fused_uniforms(source: &DemosaicedImage, shader: &ComposedShader) -> Vec<f32> {
// Base uniforms: the camera matrix and as-shot white balance, which
// every generated shader reads regardless of which operations are
// active. Framing's slots follow them and are filled by the composer,
// which is why only the first sixteen are written here.
let mut uniforms = shader.uniforms.clone();
if uniforms.len() < RESERVED_FIELDS {
uniforms.resize(RESERVED_FIELDS, 0.0);
}
let m = source.color_matrix();
let wb = source.as_shot_wb();
// Rows padded to vec4 for std140 alignment.
uniforms[0..4].copy_from_slice(&[m[0], m[1], m[2], 0.0]);
uniforms[4..8].copy_from_slice(&[m[3], m[4], m[5], 0.0]);
uniforms[8..12].copy_from_slice(&[m[6], m[7], m[8], 0.0]);
// The fourth slot is the non-linear flag, not padding: it tells the
// shader whether to linearise the sampled texel before any operation
// runs. See `DemosaicedImage::is_non_linear`.
let non_linear = if source.is_non_linear() { 1.0 } else { 0.0 };
uniforms[12..16].copy_from_slice(&[wb[0], wb[1], wb[2], non_linear]);
uniforms
}
/// TRACES: FR-DEV-3d
/// Everything the fused dispatch depends on, in one integer.
///
/// The caller's edit key, plus the three things the caller does not know
/// about: which pipeline was compiled, what was uploaded to it, and which
/// mask array was bound. Hashing the uniforms rather than trusting the
/// caller's key to cover them is what makes the reuse safe against a
/// caller whose key is coarser than it should be — and the uniforms are
/// parameters and matrix coefficients from the CPU, never rendered floats,
/// so hashing their bit patterns satisfies ARCH §6.13.
fn colour_signature(
caller: u64,
shader: &ComposedShader,
uniforms: &[f32],
masks: Option<&crate::MaskArray>,
) -> u64 {
let mut h: u64 = 0xcbf2_9ce4_8422_2325;
let mut mix = |v: u64| {
for byte in v.to_le_bytes() {
h ^= u64::from(byte);
h = h.wrapping_mul(0x100_0000_01b3);
}
};
mix(caller);
mix(shader.structure_hash);
for v in uniforms {
// Negative zero folded onto zero: the two render identically, and
// a slider that reached zero from below must not miss the cache.
mix(u64::from(if *v == 0.0 { 0 } else { v.to_bits() }));
}
match masks {
None => mix(0),
Some(m) => {
let (w, h) = m.size();
mix(1);
mix(u64::from(w));
mix(u64::from(h));
mix(u64::from(m.layers()));
}
}
h
}
/// How many distinct pipelines are compiled. Exposed for tests asserting
/// that slider movement does not recompile.
pub fn cached_pipelines(&self) -> usize {
self.cache.len()
}
/// How many detail-pass pipelines are compiled. As above, for the stage
/// that runs after this one.
pub fn cached_detail_pipelines(&self) -> usize {
self.detail.cached_pipelines()
}
/// TRACES: FR-DEV-3d
/// Fused colour dispatches encoded since this pass was created.
///
/// Exists to be asserted on. The saving `Affects::Detail` buys — a
/// sharpening slider that does not re-run the colour chain — is invisible
/// in the output by construction, since the picture is meant to be
/// identical either way. A counter is the only thing that can see it.
pub fn colour_dispatches(&self) -> usize {
self.colour_dispatches
}
/// Detail dispatches encoded since this pass was created.
pub fn detail_dispatches(&self) -> usize {
self.detail_dispatches
}
/// How many linear intermediates have been allocated. For tests: see
/// [`crate::MaskPass::allocations`] for the regression this catches.
pub fn detail_allocations(&self) -> usize {
self.detail.allocations()
}
/// The texture the last render wrote, if there has been one.
pub fn output(&self) -> Option<&wgpu::Texture> {
self.targets[self.current].as_ref().map(|t| &t.texture)
@@ -501,7 +851,7 @@ impl AdjustPass {
}
/// Number the lines of generated source, so a compiler error can be located.
fn numbered(src: &str) -> String {
pub(crate) fn numbered(src: &str) -> String {
src.lines()
.enumerate()
.map(|(i, l)| format!("{:>4} | {l}", i + 1))
+396
View File
@@ -0,0 +1,396 @@
//! The detail stage — running `dr-pipeline`'s neighbourhood passes.
//!
//! Where [`crate::AdjustPass`] fuses every point operation into one dispatch,
//! this runs the operations that cannot be fused because they read pixels they
//! are not writing: sharpening, noise reduction, clarity, texture, dehaze,
//! spot removal (FR-DEV-3, FR-DEV-8). `dr_pipeline::detail` decides *what* they
//! are and generates their WGSL; this compiles it, finds it somewhere to
//! write, and dispatches it.
//!
//! # Nothing round-trips
//!
//! Every intermediate here is a `wgpu::Texture` and none of them is ever
//! mapped. The chain is `demosaiced -> fused -> f16 -> f16 -> ... -> rgba8`,
//! all of it on the device, and the last write lands in the same texture the
//! compositor was already being handed. ARCH §6.1 and FR-DEV-4 are satisfied
//! by there being no code here that could violate them, which is the only
//! guarantee worth having.
//!
//! # Following the mask pass rather than inventing a second pattern
//!
//! `mask.rs` established how multi-target work is done in this crate, and this
//! copies it deliberately:
//!
//! - **One encoder for the whole chain.** The mask pass rasterises every layer
//! into one command buffer and submits once; this does the same for every
//! pass. Submission order is the only synchronisation either needs, because
//! both write and then read through the same queue.
//! - **Textures reallocated on size change, never per frame.** `ensure_array`
//! there, [`Intermediates::ensure`] here. Steady-state rendering at one
//! viewport size allocates nothing.
//! - **An allocation counter that exists to be asserted on.** Reallocating per
//! frame instead of per resize costs a great deal of bandwidth and shows up
//! nowhere in the output, which is exactly the kind of regression that needs
//! a test that can see it.
//! - **Pipelines cached by structure hash**, as `AdjustPass` caches its own.
//! Moving a slider re-uploads a uniform buffer; it does not recompile.
//!
//! # The ping-pong, and why there are at most three textures
//!
//! Slot 0 holds what the fused colour pass wrote. It is kept **across frames**,
//! which is what makes [`dr_pipeline::Affects::Detail`] mean something: when
//! only a detail parameter has moved, the colour key is unchanged, the fused
//! dispatch is skipped, and dragging a sharpening slider costs the detail
//! passes alone (FR-DEV-3d).
//!
//! The remaining passes alternate between slots 1 and 2, and the last one
//! writes the display texture directly rather than an intermediate — so a
//! chain of *N* passes costs *N* dispatches and not *N* + 1, and there is no
//! resolve pass to pay for. That leaves the allocation at `1 + min(N-1, 2)`
//! textures: one for a single-pass operation, two for a separable blur, three
//! however long the chain gets after that.
use std::collections::HashMap;
use dr_pipeline::detail::{ComposedDetail, ComposedDetailPass};
use wgpu::util::DeviceExt as _;
use crate::{GpuContext, GpuError};
/// The format every intermediate carries.
///
/// The same `Rgba16Float` the demosaicer produces and the same one ARCH §5.2
/// names as the working precision (FR-DEV-2). It is not a free choice: the
/// stage exists between the colour pass and the output transform precisely so
/// that a kernel runs on linear values at full internal precision, and an
/// 8-bit intermediate would quantise twice and convolve display-encoded
/// numbers — which is how sharpening comes to band a clear sky.
pub const INTERMEDIATE_FORMAT: wgpu::TextureFormat = wgpu::TextureFormat::Rgba16Float;
/// One linear working texture.
struct Slot {
#[allow(dead_code)]
texture: wgpu::Texture,
view: wgpu::TextureView,
}
/// The pool of linear intermediates, sized to the chain and the viewport.
struct Intermediates {
slots: Vec<Slot>,
width: u32,
height: u32,
allocations: usize,
}
impl Intermediates {
fn new() -> Self {
Self {
slots: Vec::new(),
width: 0,
height: 0,
allocations: 0,
}
}
/// Make sure `count` textures of this size exist.
///
/// Grows but never shrinks within a size: an edit that briefly had a
/// three-pass chain and then a one-pass one keeps the spare texture rather
/// than freeing and reallocating it the next time the user turns the
/// operation back on. A size change drops the lot, because none of them
/// fits any more.
fn ensure(&mut self, ctx: &GpuContext, count: usize, width: u32, height: u32) {
if self.width != width || self.height != height {
self.slots.clear();
self.width = width;
self.height = height;
}
while self.slots.len() < count {
let texture = ctx.device.create_texture(&wgpu::TextureDescriptor {
label: Some("detail-intermediate"),
size: wgpu::Extent3d {
width,
height,
depth_or_array_layers: 1,
},
mip_level_count: 1,
sample_count: 1,
dimension: wgpu::TextureDimension::D2,
format: INTERMEDIATE_FORMAT,
// STORAGE_BINDING to be written by a compute pass and
// TEXTURE_BINDING to be read by the next one. Nothing else:
// no RENDER_ATTACHMENT, because unlike the adjust pass's
// output these are never handed to a compositor, and no
// COPY_SRC, because nothing reads them back — that is the
// point (ARCH §6.1).
usage: wgpu::TextureUsages::STORAGE_BINDING
| wgpu::TextureUsages::TEXTURE_BINDING,
view_formats: &[],
});
let view = texture.create_view(&Default::default());
self.slots.push(Slot { texture, view });
self.allocations += 1;
}
}
}
/// Runs the detail stage.
///
/// Owned by [`crate::AdjustPass`] rather than standing alone, because the two
/// halves are one render: the fused pass writes slot 0, this reads it, and the
/// last pass writes the adjust pass's own output texture. Splitting them into
/// two objects with two lifetimes would mean a caller could hold a stale
/// intermediate against a fresh colour result and never be told.
pub(crate) struct DetailRunner {
ctx: GpuContext,
/// Layout for a pass writing another linear intermediate.
to_linear: Layout,
/// Layout for the last pass, which writes the display texture.
to_output: Layout,
/// Compiled pipelines by pass structure hash.
cache: HashMap<u64, wgpu::ComputePipeline>,
pool: Intermediates,
}
struct Layout {
bind_group: wgpu::BindGroupLayout,
pipeline: wgpu::PipelineLayout,
}
impl DetailRunner {
pub(crate) fn new(ctx: &GpuContext) -> Self {
Self {
ctx: ctx.clone(),
to_linear: Layout::new(ctx, INTERMEDIATE_FORMAT, "detail-linear"),
to_output: Layout::new(ctx, crate::AdjustPass::FORMAT, "detail-output"),
cache: HashMap::new(),
pool: Intermediates::new(),
}
}
/// The view the fused colour pass should write, given a chain of `passes`.
///
/// Slot 0, always — it is the one that survives between frames so that a
/// detail-only change can skip the colour dispatch entirely.
pub(crate) fn colour_target(
&mut self,
passes: usize,
width: u32,
height: u32,
) -> &wgpu::TextureView {
// One for the colour pass's result, then one per hand-off between
// detail passes, capped at two because a ping-pong needs no more: the
// last pass writes the display texture rather than an intermediate.
let needed = 1 + passes.saturating_sub(1).min(2);
self.pool.ensure(&self.ctx, needed, width, height);
&self.pool.slots[0].view
}
/// Encode every pass of `chain`, the last one writing `output`.
///
/// The caller must already have run the fused colour pass into
/// [`Self::colour_target`] — or established that a previous frame's is
/// still valid, which is the whole point of keeping slot 0.
pub(crate) fn encode(
&mut self,
encoder: &mut wgpu::CommandEncoder,
chain: &ComposedDetail,
output: &wgpu::TextureView,
width: u32,
height: u32,
) -> Result<usize, GpuError> {
for pass in &chain.passes {
self.compile(pass)?;
}
for (index, pass) in chain.passes.iter().enumerate() {
// Read what the previous pass wrote; write the next slot, or the
// display texture if this is the last one. `index % 2` alternates
// between slots 1 and 2, so a pass never reads the texture it is
// writing — which on a compute pass is not an error the driver
// reports, merely a picture that depends on scheduling.
let source_slot = if index == 0 { 0 } else { 2 - (index % 2) };
let source = &self.pool.slots[source_slot].view;
let destination = if pass.writes_output {
output
} else {
&self.pool.slots[1 + (index % 2)].view
};
let layout = if pass.writes_output {
&self.to_output
} else {
&self.to_linear
};
let params = self
.ctx
.device
.create_buffer_init(&wgpu::util::BufferInitDescriptor {
label: Some("detail-params"),
contents: bytemuck::cast_slice(&pass.uniforms),
usage: wgpu::BufferUsages::UNIFORM,
});
let bind_group = self
.ctx
.device
.create_bind_group(&wgpu::BindGroupDescriptor {
label: Some("detail-bg"),
layout: &layout.bind_group,
entries: &[
wgpu::BindGroupEntry {
binding: 0,
resource: wgpu::BindingResource::TextureView(source),
},
wgpu::BindGroupEntry {
binding: 1,
resource: params.as_entire_binding(),
},
wgpu::BindGroupEntry {
binding: 2,
resource: wgpu::BindingResource::TextureView(destination),
},
],
});
let pipeline = self
.cache
.get(&pass.structure_hash)
.expect("compiled above");
let mut compute = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor {
label: Some(pass.label.as_str()),
timestamp_writes: None,
});
compute.set_pipeline(pipeline);
compute.set_bind_group(0, &bind_group, &[]);
compute.dispatch_workgroups(width.div_ceil(8), height.div_ceil(8), 1);
}
Ok(chain.passes.len())
}
/// Compile one pass, or leave the cached pipeline in place.
///
/// A validation error here is a codegen bug rather than anything the user
/// did, so it is caught in an error scope and returned with the generated
/// source and the pass's label attached — a line number against code
/// nobody wrote, from one of several passes, is otherwise close to
/// unactionable.
fn compile(&mut self, pass: &ComposedDetailPass) -> Result<(), GpuError> {
if self.cache.contains_key(&pass.structure_hash) {
return Ok(());
}
let scope = self
.ctx
.device
.push_error_scope(wgpu::ErrorFilter::Validation);
let module = self
.ctx
.device
.create_shader_module(wgpu::ShaderModuleDescriptor {
label: Some(pass.label.as_str()),
source: wgpu::ShaderSource::Wgsl(pass.source.as_str().into()),
});
let layout = if pass.writes_output {
&self.to_output
} else {
&self.to_linear
};
let pipeline = self
.ctx
.device
.create_compute_pipeline(&wgpu::ComputePipelineDescriptor {
label: Some(pass.label.as_str()),
layout: Some(&layout.pipeline),
module: &module,
entry_point: Some("main"),
compilation_options: Default::default(),
cache: None,
});
if let Some(err) = pollster::block_on(scope.pop()) {
return Err(GpuError::ShaderCompilation(format!(
"detail pass {}: {err}\n\n--- generated source ---\n{}",
pass.label,
crate::adjust::numbered(&pass.source)
)));
}
self.cache.insert(pass.structure_hash, pipeline);
Ok(())
}
/// How many distinct detail pipelines are compiled. For tests asserting
/// that slider movement does not recompile.
pub(crate) fn cached_pipelines(&self) -> usize {
self.cache.len()
}
/// How many intermediate textures have been allocated since this pass was
/// created. For tests — see [`crate::MaskPass::allocations`] for the
/// regression this shape of counter exists to catch.
pub(crate) fn allocations(&self) -> usize {
self.pool.allocations
}
}
impl Layout {
fn new(ctx: &GpuContext, format: wgpu::TextureFormat, label: &str) -> Self {
let bind_group = ctx
.device
.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
label: Some(label),
entries: &[
// The previous stage's result.
wgpu::BindGroupLayoutEntry {
binding: 0,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::Texture {
sample_type: wgpu::TextureSampleType::Float { filterable: true },
view_dimension: wgpu::TextureViewDimension::D2,
multisampled: false,
},
count: None,
},
wgpu::BindGroupLayoutEntry {
binding: 1,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::Buffer {
ty: wgpu::BufferBindingType::Uniform,
has_dynamic_offset: false,
min_binding_size: None,
},
count: None,
},
wgpu::BindGroupLayoutEntry {
binding: 2,
visibility: wgpu::ShaderStages::COMPUTE,
ty: wgpu::BindingType::StorageTexture {
access: wgpu::StorageTextureAccess::WriteOnly,
format,
view_dimension: wgpu::TextureViewDimension::D2,
},
count: None,
},
],
});
let pipeline = ctx
.device
.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
label: Some(label),
bind_group_layouts: &[Some(&bind_group)],
immediate_size: 0,
});
Self {
bind_group,
pipeline,
}
}
}
+6
View File
@@ -19,12 +19,18 @@ use wgpu::util::DeviceExt;
mod adjust;
mod demosaic;
mod detail;
mod error;
mod histogram;
mod mask;
mod readback;
mod segment;
pub use adjust::AdjustPass;
// The format the neighbourhood stage works in. Public because it is a promise
// rather than an implementation detail: a detail pass is guaranteed linear,
// unclipped, full internal precision (FR-DEV-2), and anyone reasoning about
// VRAM at 24 MP needs to know what an intermediate costs.
pub use detail::INTERMEDIATE_FORMAT as DETAIL_INTERMEDIATE_FORMAT;
pub use demosaic::{DemosaicedImage, Demosaicer};
pub use error::GpuError;
// Renamed on the way out: `BINS` says enough inside `histogram`, and nothing
+408
View File
@@ -0,0 +1,408 @@
//! The neighbourhood stage, end to end on a real device.
//!
//! `dr-pipeline`'s own tests assert what the composer *generates*; nothing
//! there can tell whether the WGSL compiles, whether pass two is handed what
//! pass one wrote, or whether the output transform happens exactly once. Those
//! are questions only a GPU answers, and they are the ones that decide whether
//! a future sharpening operation works or draws nonsense.
//!
//! The consumer is `detail_probe`, a separable box blur that is not a develop
//! operation (see `dr_pipeline::detail::probe`). A box blur is used because its
//! answer is known in closed form: over a step edge it produces a ramp exactly
//! `2r + 1` pixels wide with a computable value at every step, so these tests
//! assert **pixels** rather than "something changed".
//!
//! # Reading the expected values
//!
//! The source is uploaded through `DemosaicedImage::from_rgba8`, which flags it
//! non-linear, so the generated shader decodes sRGB before any operation runs.
//! A black/white step therefore reaches the detail stage as linear 0.0 and 1.0
//! exactly. The blur averages those, and the last detail pass re-encodes. So
//! the expected byte at a column is `srgb_encode(white_taps / (2r + 1))`, with
//! taps clamped at the border — which is exactly what `expected_profile`
//! computes.
use dr_gpu::{AdjustPass, DemosaicedImage, GpuContext};
use dr_pipeline::descriptor::{OpId, ParamId};
use dr_pipeline::detail::probe::BoxBlur;
use dr_pipeline::{Affects, EditGraph, OutputMode};
use dr_types::ColourSpace;
const PROBE: OpId = OpId("detail_probe");
const RADIUS: ParamId = ParamId("radius");
fn ctx() -> Option<GpuContext> {
// CI runners and headless machines may have no usable adapter. Skip rather
// than fail, exactly as the rest of this crate's device tests do.
match pollster::block_on(GpuContext::new_headless()) {
Ok(c) => Some(c),
Err(e) => {
eprintln!("skipping: no GPU adapter ({e})");
None
}
}
}
/// A vertical step edge: black to the left of `size / 2`, white to the right.
///
/// The one image whose blur is worth checking by hand. A gradient would
/// average to itself and hide a kernel that is off by one; a step does not.
fn step_edge(ctx: &GpuContext, size: u32) -> DemosaicedImage {
let data: Vec<u8> = (0..size * size)
.flat_map(|i| {
let x = i % size;
let v = if x < size / 2 { 0u8 } else { 255 };
[v, v, v, 255]
})
.collect();
DemosaicedImage::from_rgba8(ctx, &data, size, size).expect("upload")
}
/// One row of the rendered image, red channel, as bytes.
fn row(pixels: &[u8], size: u32, y: u32) -> Vec<u8> {
(0..size)
.map(|x| pixels[((y * size + x) * 4) as usize])
.collect()
}
fn srgb_encode(v: f32) -> u8 {
let e = if v <= 0.003_130_8 {
v * 12.92
} else {
1.055 * v.powf(1.0 / 2.4) - 0.055
};
(e.clamp(0.0, 1.0) * 255.0).round() as u8
}
/// What a separable box blur of radius `r` must produce over the step edge.
fn expected_profile(size: u32, r: i32) -> Vec<u8> {
let last = size as i32 - 1;
let edge = (size / 2) as i32;
(0..size as i32)
.map(|x| {
let white = (-r..=r)
.filter(|i| (x + i).clamp(0, last) >= edge)
.count();
srgb_encode(white as f32 / (2 * r + 1) as f32)
})
.collect()
}
/// Render one graph, with its detail stage, and read the pixels back.
///
/// This is the whole calling convention a frontend has to adopt, in five
/// lines: compose both halves from one graph at one output space, ask the
/// graph for the scale, and pass the invalidation key through.
fn render(
ctx: &GpuContext,
pass: &mut AdjustPass,
graph: &EditGraph,
source: &DemosaicedImage,
out: u32,
) -> Vec<u8> {
let _ = ctx;
let shader = graph.compose_for(ColourSpace::Srgb);
let scale = graph.render_scale(source.size(), (out, out));
let detail = graph.compose_detail_for(scale, ColourSpace::Srgb);
let key = graph.invalidation().through(Affects::Colour);
pass.render_detailed(source, &shader, out, out, None, &detail, key)
.expect("render");
pass.export_pixels().expect("readback").0
}
#[test]
fn a_neighbourhood_pass_produces_the_pixels_it_should() {
// The whole seam, proved once: an operation that reads its neighbours runs
// on the GPU, and the values it writes are the ones a box blur is defined
// to write. Not "the edge got softer" — every byte of the ramp.
let Some(ctx) = ctx() else { return };
const SIZE: u32 = 64;
let mut graph = EditGraph::with_detail_probe();
graph.set_param(PROBE, RADIUS, 0.0625); // 4 px on a 64 px edge
let source = step_edge(&ctx, SIZE);
let mut pass = AdjustPass::new(&ctx);
let pixels = render(&ctx, &mut pass, &graph, &source, SIZE);
let got = row(&pixels, SIZE, SIZE / 2);
let r = BoxBlur::with_radius(0.0625).kernel(graph.render_scale((SIZE, SIZE), (SIZE, SIZE)));
assert_eq!(r, 4, "5/64 of the shorter edge, rounded");
let want = expected_profile(SIZE, r as i32);
for (x, (a, b)) in got.iter().zip(&want).enumerate() {
assert!(
a.abs_diff(*b) <= 2,
"column {x}: got {a}, expected {b}\ngot: {got:?}\nwant: {want:?}"
);
}
}
#[test]
fn the_second_pass_reads_what_the_first_one_wrote() {
// The ping-pong, stated as a property of the picture rather than of the
// plumbing. A separable blur is symmetric: applied to a *horizontal* step
// it must also soften a horizontal edge in the other direction. Wire the
// second pass to read the original again and the vertical smear vanishes,
// which is exactly what this sees.
let Some(ctx) = ctx() else { return };
const SIZE: u32 = 64;
// A quadrant image: the vertical pass has something to do only if it is
// reading the horizontal pass's output rather than the source.
let data: Vec<u8> = (0..SIZE * SIZE)
.flat_map(|i| {
let (x, y) = (i % SIZE, i / SIZE);
let v = if (x < SIZE / 2) == (y < SIZE / 2) {
0u8
} else {
255
};
[v, v, v, 255]
})
.collect();
let source = DemosaicedImage::from_rgba8(&ctx, &data, SIZE, SIZE).expect("upload");
let mut graph = EditGraph::with_detail_probe();
graph.set_param(PROBE, RADIUS, 0.0625);
let mut pass = AdjustPass::new(&ctx);
let pixels = render(&ctx, &mut pass, &graph, &source, SIZE);
// Two separable passes compose into a true two-dimensional box average —
// but only if the second reads the first's output. Computed in closed form
// over the same window the shader uses, so this is an assertion about
// values rather than about direction.
let r = 4i32;
let last = SIZE as i32 - 1;
let half = (SIZE / 2) as i32;
let quadrant_is_black = |x: i32, y: i32| (x < half) == (y < half);
let want: Vec<u8> = (0..SIZE as i32)
.map(|x| {
let y = half;
let mut white = 0usize;
for dy in -r..=r {
for dx in -r..=r {
let (sx, sy) = ((x + dx).clamp(0, last), (y + dy).clamp(0, last));
if !quadrant_is_black(sx, sy) {
white += 1;
}
}
}
srgb_encode(white as f32 / ((2 * r + 1) * (2 * r + 1)) as f32)
})
.collect();
let got = row(&pixels, SIZE, SIZE / 2);
for (x, (a, b)) in got.iter().zip(&want).enumerate() {
// A second pass reading the *source* instead would leave column 20 at
// 255 where a real 2D average puts it near 196 — so the failure this
// catches is loud, not marginal.
assert!(
a.abs_diff(*b) <= 2,
"column {x}: got {a}, expected {b}\ngot: {got:?}\nwant: {want:?}"
);
}
}
#[test]
fn an_inactive_detail_operation_costs_exactly_nothing() {
// The rule the whole pipeline rests on, carried into this stage. A
// photograph with no sharpening must render through the single fused
// dispatch it always did, allocate no intermediate, and — the part worth
// checking — produce byte-identical pixels to a graph that has no
// neighbourhood operation in it at all.
let Some(ctx) = ctx() else { return };
const SIZE: u32 = 32;
let source = step_edge(&ctx, SIZE);
let probe = EditGraph::with_detail_probe();
assert_eq!(
probe.compose_for(ColourSpace::Srgb).output_mode,
OutputMode::Encoded,
"a neutral detail operation must not change how the fused pass ends"
);
let mut with_probe = AdjustPass::new(&ctx);
let a = render(&ctx, &mut with_probe, &probe, &source, SIZE);
assert_eq!(with_probe.colour_dispatches(), 1);
assert_eq!(with_probe.detail_dispatches(), 0);
assert_eq!(with_probe.detail_allocations(), 0, "nothing was allocated");
let plain = EditGraph::default_chain();
let mut without = AdjustPass::new(&ctx);
let b = render(&ctx, &mut without, &plain, &source, SIZE);
assert_eq!(a, b, "an operation at its defaults must not touch the image");
}
#[test]
fn moving_a_detail_parameter_does_not_re_run_the_colour_pass() {
// TRACES: FR-DEV-3d, and the operational point of `Affects::Detail`.
//
// Invisible in the output by construction — the picture is meant to be
// whatever the sharpening says whichever way it was computed — so a
// dispatch counter is the only thing that can see it. Without this, the
// whole invalidation story is a comment.
let Some(ctx) = ctx() else { return };
const SIZE: u32 = 64;
let source = step_edge(&ctx, SIZE);
let mut pass = AdjustPass::new(&ctx);
let mut graph = EditGraph::with_detail_probe();
graph.set_param(PROBE, RADIUS, 0.0625);
render(&ctx, &mut pass, &graph, &source, SIZE);
assert_eq!(pass.colour_dispatches(), 1);
assert_eq!(pass.detail_dispatches(), 2, "a separable blur is two passes");
// Drag the sharpening slider. The colour chain is untouched, so the linear
// intermediate it wrote is still exactly right.
graph.set_param(PROBE, RADIUS, 0.09);
render(&ctx, &mut pass, &graph, &source, SIZE);
assert_eq!(
pass.colour_dispatches(),
1,
"the fused colour pass re-ran for a change it does not depend on"
);
assert_eq!(pass.detail_dispatches(), 4);
// Now move exposure. The detail stage reads what the colour pass wrote, so
// this one genuinely does have to re-run both — anything else would show a
// sharpened version of the previous exposure.
graph.set_param(
dr_pipeline::ops::exposure::ID,
dr_pipeline::ops::exposure::EXPOSURE,
1.0,
);
render(&ctx, &mut pass, &graph, &source, SIZE);
assert_eq!(pass.colour_dispatches(), 2);
assert_eq!(pass.detail_dispatches(), 6);
}
#[test]
fn dragging_a_slider_recompiles_nothing_and_reallocates_nothing() {
// The two costs that are ruinous per frame and invisible in the output.
// Both are the same rule the rest of the crate follows: values ride in a
// uniform buffer, and textures are reallocated on resize rather than on
// change.
let Some(ctx) = ctx() else { return };
const SIZE: u32 = 48;
let source = step_edge(&ctx, SIZE);
let mut pass = AdjustPass::new(&ctx);
let mut graph = EditGraph::with_detail_probe();
graph.set_param(PROBE, RADIUS, 0.05);
render(&ctx, &mut pass, &graph, &source, SIZE);
let pipelines = pass.cached_detail_pipelines();
let allocations = pass.detail_allocations();
assert_eq!(pipelines, 2, "one per pass of the separable blur");
assert_eq!(allocations, 2, "the colour result, and one hand-off");
for radius in [0.06, 0.07, 0.08, 0.09] {
graph.set_param(PROBE, RADIUS, radius);
render(&ctx, &mut pass, &graph, &source, SIZE);
}
assert_eq!(
pass.cached_detail_pipelines(),
pipelines,
"a radius is a uniform, not a shader"
);
assert_eq!(
pass.detail_allocations(),
allocations,
"a steady viewport must allocate nothing"
);
// A resize is the one thing that legitimately reallocates.
render(&ctx, &mut pass, &graph, &source, SIZE / 2);
assert!(pass.detail_allocations() > allocations);
}
#[test]
fn a_proxy_and_an_export_agree_about_where_the_effect_lands() {
// TRACES: FR-DSP-1 — the subtle one, and the reason `RenderScale` exists.
//
// The same edit, rendered at two resolutions. A radius stored as a
// fraction of the shorter edge must produce a transition covering the same
// *proportion* of the frame at both, or a sharpening tuned on screen is a
// different sharpening in the exported file.
//
// The tolerance is a pixel's worth at the smaller size, because the kernel
// is an integer count and 6.25% of 64 pixels is not 6.25% of 128. That
// rounding is the whole of the error, and it is bounded by half a render
// pixel by construction.
let Some(ctx) = ctx() else { return };
const SOURCE: u32 = 128;
let source = step_edge(&ctx, SOURCE);
let mut graph = EditGraph::with_detail_probe();
graph.set_param(PROBE, RADIUS, 0.0625);
let spread = |out: u32| -> f32 {
let mut pass = AdjustPass::new(&ctx);
let pixels = render(&ctx, &mut pass, &graph, &source, out);
let line = row(&pixels, out, out / 2);
// Where the ramp starts and ends, in fractions of the frame.
let first = line.iter().position(|&v| v > 4).expect("a ramp") as f32;
let last = line.iter().rposition(|&v| v < 251).expect("a ramp") as f32;
(last - first) / out as f32
};
let proxy = spread(SOURCE / 2);
let export = spread(SOURCE);
assert!(
(proxy - export).abs() < 0.03,
"the effect covers {proxy:.3} of the proxy and {export:.3} of the \
export; a radius tuned on screen must land in the file"
);
// And it is a real transition in both, not two flat images agreeing.
assert!(proxy > 0.08 && export > 0.08, "{proxy:.3} / {export:.3}");
}
#[test]
fn the_two_halves_of_one_composition_must_be_dispatched_together() {
// The failure this guards is a bad one to debug: a shader composed to hand
// on linear working values, bound to an rgba8 storage texture. wgpu
// rejects it, but the message is about a bind group, a long way from the
// caller that composed one half of an edit and rendered the other.
let Some(ctx) = ctx() else { return };
const SIZE: u32 = 32;
let source = step_edge(&ctx, SIZE);
let mut pass = AdjustPass::new(&ctx);
let mut graph = EditGraph::with_detail_probe();
graph.set_param(PROBE, RADIUS, 0.0625);
let shader = graph.compose_for(ColourSpace::Srgb);
assert_eq!(shader.output_mode, OutputMode::LinearWorking);
let err = pass
.render_masked(&source, &shader, SIZE, SIZE, None)
.expect_err("a linear-working shader has no business in the plain path");
assert!(
format!("{err}").contains("render_detailed"),
"the error should name the way out: {err}"
);
}
#[test]
fn an_empty_chain_falls_through_to_the_ordinary_render() {
// A caller that always goes through `render_detailed` — which is what a
// frontend will do, since it does not want to branch on whether the user
// has sharpening on — must pay exactly nothing for the edits that have
// none.
let Some(ctx) = ctx() else { return };
const SIZE: u32 = 32;
let source = step_edge(&ctx, SIZE);
let graph = EditGraph::default_chain();
let mut pass = AdjustPass::new(&ctx);
let shader = graph.compose_for(ColourSpace::Srgb);
let scale = graph.render_scale((SIZE, SIZE), (SIZE, SIZE));
let detail = graph.compose_detail_for(scale, ColourSpace::Srgb);
assert!(detail.is_empty());
pass.render_detailed(&source, &shader, SIZE, SIZE, None, &detail, 0)
.expect("render");
assert_eq!(pass.colour_dispatches(), 1);
assert_eq!(pass.detail_dispatches(), 0);
assert_eq!(pass.detail_allocations(), 0);
}