Develop a linear DNG from windows and reduced copies of it

DemosaicedImage::linear_rgb16_window uploads part of a linear DNG, or a
box-reduced copy of it, and says where it sits in the frame; size() now
reports the frame and texture_size() the texels, and the fused pass
writes the window into the shader's uniforms. EditGraph::source_region
finds the part of the source a view reads, and tiles::plan cuts a render
too large for one texture into halo-grown, grid-aligned tiles.

The GPU test renders frames a tile at a time from their own windows and
compares them with the whole: identical for point operations, within one
code value when straightened with clarity on.
This commit is contained in:
2026-09-27 17:35:16 -04:00
parent 1eab723c90
commit 0007fa459f
9 changed files with 600 additions and 38 deletions
+7
View File
@@ -1350,6 +1350,13 @@ impl AdjustPass {
// runs. See `DemosaicedImage::is_non_linear`.
let non_linear = if source.is_non_linear() { 1.0 } else { 0.0 };
uniforms[12..16].copy_from_slice(&[wb[0], wb[1], wb[2], non_linear]);
// TRACES: FR-DSP-2 | NFR-RES-2
// Which part of the photograph the texture holds. The whole of it for
// every source that fits in one texture, which writes back exactly
// what the composer put there.
let w = dr_pipeline::SOURCE_WINDOW_UNIFORM_OFFSET;
uniforms[w..w + dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS]
.copy_from_slice(&source.window_uniforms());
uniforms
}
+145 -11
View File
@@ -114,8 +114,20 @@ pub struct DemosaicedImage {
/// Which upload this is, unique for the life of the process. See
/// [`Self::id`].
id: u64,
/// TRACES: FR-DSP-2 | NFR-RES-2
/// The whole frame's size in pixels — what [`Self::size`] reports.
/// The texture's own size when it holds the whole frame at full
/// resolution, which is every photograph that fits in one.
frame: (u32, u32),
/// Which part of the frame the texture holds, as origin and extent in
/// normalised frame coordinates. `[0, 0, 1, 1]` for the whole frame,
/// reduced or not. See [`Self::window_uniforms`].
window: [f32; 4],
}
/// The window of a texture that holds the whole frame.
const WHOLE_FRAME: [f32; 4] = [0.0, 0.0, 1.0, 1.0];
/// The next [`DemosaicedImage::id`].
fn next_image_id() -> u64 {
static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1);
@@ -133,10 +145,50 @@ impl DemosaicedImage {
&self.view
}
/// The size of the photograph this stands for, in its own pixels.
///
/// **Not necessarily the texture's.** For a photograph larger than one
/// texture this is a reduced copy of it or a window cut from it, and
/// everything that sizes a render, a crop or a kernel has to go on
/// measuring the photograph. What indexes the texture's texels asks
/// [`Self::texture_size`] instead.
pub fn size(&self) -> (u32, u32) {
self.frame
}
/// The texture's own size in texels.
pub fn texture_size(&self) -> (u32, u32) {
(self.width, self.height)
}
/// TRACES: FR-DSP-2 | NFR-RES-2
/// The source window uniforms the fused shader reads, in the order
/// `dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS` declares them.
///
/// The second `vec4` is zero for a texture that holds the whole frame at
/// full resolution, so the shader measures the texture itself exactly as
/// it did before windows existed.
pub fn window_uniforms(&self) -> [f32; dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS] {
let [x, y, w, h] = self.window;
let (fw, fh) = if self.is_whole() {
(0.0, 0.0)
} else {
(self.frame.0 as f32, self.frame.1 as f32)
};
[x, y, w, h, fw, fh, 0.0, 0.0]
}
/// Whether the texture is the whole frame at full resolution.
pub fn is_whole(&self) -> bool {
self.window == WHOLE_FRAME && self.frame == (self.width, self.height)
}
/// The window this texture holds, as origin and extent in normalised
/// frame coordinates.
pub fn window(&self) -> [f32; 4] {
self.window
}
/// Which texture this is, as a number that is never reused.
///
/// For a cache that has to know it is still looking at the same pixels
@@ -274,6 +326,8 @@ impl DemosaicedImage {
// the highlights of an image that was already finished.
non_linear: true,
id: next_image_id(),
frame: (width, height),
window: WHOLE_FRAME,
})
}
}
@@ -289,6 +343,41 @@ impl DemosaicedImage {
/// body that took its sources.
pub fn from_linear_rgb16(ctx: &GpuContext, raw: &RawImage) -> Result<Self, GpuError> {
let (width, height) = (raw.crop.width.max(1), raw.crop.height.max(1));
Self::linear_rgb16_window(ctx, raw, [0, 0, width, height], 1)
}
/// TRACES: FR-DSP-2 | NFR-RES-2
/// Part of a linear DNG, or a reduced copy of it, for a photograph too
/// large to hold in one texture.
///
/// `region` is `[x, y, width, height]` in pixels of the frame (the
/// file's crop), clamped to it. `reduce` averages `reduce × reduce`
/// blocks into one texel — a box filter, which is what a reduced copy
/// that is only ever displayed smaller than itself needs, and which keeps
/// the samples in scene-linear light where an average means something.
///
/// The texture then knows where it sits ([`Self::window`]) and how large
/// the photograph is ([`Self::size`]), and the fused shader maps each
/// output pixel's position in the *photograph* into it. So a crop, a
/// rotation or a mask drawn on the reduced copy lands on the same pixels
/// of a full-resolution window, and an export in tiles is the same
/// picture as one that fitted.
///
/// Refused only if the result itself does not fit the device.
pub fn linear_rgb16_window(
ctx: &GpuContext,
raw: &RawImage,
region: [u32; 4],
reduce: u32,
) -> Result<Self, GpuError> {
let frame = (raw.crop.width.max(1), raw.crop.height.max(1));
let k = reduce.max(1);
let x0 = region[0].min(frame.0 - 1);
let y0 = region[1].min(frame.1 - 1);
let rw = region[2].clamp(1, frame.0 - x0);
let rh = region[3].clamp(1, frame.1 - y0);
let (width, height) = (rw.div_ceil(k), rh.div_ceil(k));
let limits = ctx.device.limits();
if width > limits.max_texture_dimension_2d || height > limits.max_texture_dimension_2d {
return Err(GpuError::TooLarge(format!(
@@ -308,19 +397,49 @@ impl DemosaicedImage {
}
let black = black_per_cell(raw);
let inv = inv_range_per_cell(raw);
// Per channel rather than per CFA cell: R, G, B are the first three.
let mut half: Vec<u16> = Vec::with_capacity((width * height * 4) as usize);
for y in 0..height as usize {
let row = (raw.crop.y as usize + y) * stride + raw.crop.x as usize * 3;
for x in 0..width as usize {
let p = &raw.data[row + x * 3..row + x * 3 + 3];
for c in 0..3 {
let v = (f32::from(p[c]) - black[c]) * inv[c];
half.push(f32_to_f16_bits_unclamped(v));
// One output row per task: a 200-megapixel reduction is a second of
// one core, and the rows are independent.
let row_texels = width as usize * 4;
let mut half = vec![0u16; row_texels * height as usize];
let fill_row = |ty: usize, out: &mut [u16]| {
let sy0 = y0 as usize + ty * k as usize;
let sy1 = (sy0 + k as usize).min((y0 + rh) as usize);
for tx in 0..width as usize {
let sx0 = x0 as usize + tx * k as usize;
let sx1 = (sx0 + k as usize).min((x0 + rw) as usize);
let mut acc = [0f32; 3];
for sy in sy0..sy1 {
let row = (raw.crop.y as usize + sy) * stride + raw.crop.x as usize * 3;
for sx in sx0..sx1 {
let p = &raw.data[row + sx * 3..row + sx * 3 + 3];
for c in 0..3 {
acc[c] += f32::from(p[c]);
}
}
}
half.push(f32_to_f16_bits(1.0));
let n = ((sy1 - sy0) * (sx1 - sx0)).max(1) as f32;
let texel = &mut out[tx * 4..tx * 4 + 4];
for c in 0..3 {
let v = (acc[c] / n - black[c]) * inv[c];
texel[c] = f32_to_f16_bits_unclamped(v);
}
texel[3] = f32_to_f16_bits(1.0);
}
}
};
let threads = std::thread::available_parallelism().map_or(1, |n| n.get());
let rows_per = (height as usize).div_ceil(threads).max(1);
std::thread::scope(|scope| {
for (chunk, rows) in half.chunks_mut(rows_per * row_texels).enumerate() {
let fill_row = &fill_row;
scope.spawn(move || {
for (i, out) in rows.chunks_mut(row_texels).enumerate() {
fill_row(chunk * rows_per + i, out);
}
});
}
});
let texture = ctx.device.create_texture_with_data(
&ctx.queue,
&wgpu::TextureDescriptor {
@@ -341,6 +460,17 @@ impl DemosaicedImage {
bytemuck::cast_slice(&half),
);
let view = texture.create_view(&Default::default());
// The extent is the texels' own, `width × k`, not the region's: the
// last block of a reduction may run past the frame's edge, and
// stretching it to fit would put every texel slightly off the
// pixels it averaged. The shader's bounds test is on the frame, so
// nothing past the edge is ever read.
let window = [
x0 as f32 / frame.0 as f32,
y0 as f32 / frame.1 as f32,
(width * k) as f32 / frame.0 as f32,
(height * k) as f32 / frame.1 as f32,
];
Ok(Self {
texture,
view,
@@ -350,6 +480,8 @@ impl DemosaicedImage {
as_shot_wb: [raw.wb_coeffs[0], raw.wb_coeffs[1], raw.wb_coeffs[2]],
non_linear: false,
id: next_image_id(),
frame,
window,
})
}
}
@@ -768,6 +900,8 @@ impl Demosaicer {
// transfer function.
non_linear: false,
id: next_image_id(),
frame: (width, height),
window: WHOLE_FRAME,
})
}
}
+1 -1
View File
@@ -928,7 +928,7 @@ impl MaskPass {
// the only readers and they are skipped in that case.
let source_step = match source {
Some(image) => {
let (sw, sh) = image.size();
let (sw, sh) = image.texture_size();
[
sw as f32 / width.max(1) as f32,
sh as f32 / height.max(1) as f32,
+1 -1
View File
@@ -208,7 +208,7 @@ impl SegmentPass {
source: &DemosaicedImage,
opts: SegmentOptions,
) -> Result<Segmentation, GpuError> {
let (src_w, src_h) = source.size();
let (src_w, src_h) = source.texture_size();
let (width, height) = proxy_size(src_w, src_h, opts.max_edge);
let n = (width * height) as u64;