Develop a linear DNG from windows and reduced copies of it
DemosaicedImage::linear_rgb16_window uploads part of a linear DNG, or a box-reduced copy of it, and says where it sits in the frame; size() now reports the frame and texture_size() the texels, and the fused pass writes the window into the shader's uniforms. EditGraph::source_region finds the part of the source a view reads, and tiles::plan cuts a render too large for one texture into halo-grown, grid-aligned tiles. The GPU test renders frames a tile at a time from their own windows and compares them with the whole: identical for point operations, within one code value when straightened with clarity on.
This commit is contained in:
@@ -1350,6 +1350,13 @@ impl AdjustPass {
|
||||
// runs. See `DemosaicedImage::is_non_linear`.
|
||||
let non_linear = if source.is_non_linear() { 1.0 } else { 0.0 };
|
||||
uniforms[12..16].copy_from_slice(&[wb[0], wb[1], wb[2], non_linear]);
|
||||
// TRACES: FR-DSP-2 | NFR-RES-2
|
||||
// Which part of the photograph the texture holds. The whole of it for
|
||||
// every source that fits in one texture, which writes back exactly
|
||||
// what the composer put there.
|
||||
let w = dr_pipeline::SOURCE_WINDOW_UNIFORM_OFFSET;
|
||||
uniforms[w..w + dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS]
|
||||
.copy_from_slice(&source.window_uniforms());
|
||||
uniforms
|
||||
}
|
||||
|
||||
|
||||
+145
-11
@@ -114,8 +114,20 @@ pub struct DemosaicedImage {
|
||||
/// Which upload this is, unique for the life of the process. See
|
||||
/// [`Self::id`].
|
||||
id: u64,
|
||||
/// TRACES: FR-DSP-2 | NFR-RES-2
|
||||
/// The whole frame's size in pixels — what [`Self::size`] reports.
|
||||
/// The texture's own size when it holds the whole frame at full
|
||||
/// resolution, which is every photograph that fits in one.
|
||||
frame: (u32, u32),
|
||||
/// Which part of the frame the texture holds, as origin and extent in
|
||||
/// normalised frame coordinates. `[0, 0, 1, 1]` for the whole frame,
|
||||
/// reduced or not. See [`Self::window_uniforms`].
|
||||
window: [f32; 4],
|
||||
}
|
||||
|
||||
/// The window of a texture that holds the whole frame.
|
||||
const WHOLE_FRAME: [f32; 4] = [0.0, 0.0, 1.0, 1.0];
|
||||
|
||||
/// The next [`DemosaicedImage::id`].
|
||||
fn next_image_id() -> u64 {
|
||||
static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1);
|
||||
@@ -133,10 +145,50 @@ impl DemosaicedImage {
|
||||
&self.view
|
||||
}
|
||||
|
||||
/// The size of the photograph this stands for, in its own pixels.
|
||||
///
|
||||
/// **Not necessarily the texture's.** For a photograph larger than one
|
||||
/// texture this is a reduced copy of it or a window cut from it, and
|
||||
/// everything that sizes a render, a crop or a kernel has to go on
|
||||
/// measuring the photograph. What indexes the texture's texels asks
|
||||
/// [`Self::texture_size`] instead.
|
||||
pub fn size(&self) -> (u32, u32) {
|
||||
self.frame
|
||||
}
|
||||
|
||||
/// The texture's own size in texels.
|
||||
pub fn texture_size(&self) -> (u32, u32) {
|
||||
(self.width, self.height)
|
||||
}
|
||||
|
||||
/// TRACES: FR-DSP-2 | NFR-RES-2
|
||||
/// The source window uniforms the fused shader reads, in the order
|
||||
/// `dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS` declares them.
|
||||
///
|
||||
/// The second `vec4` is zero for a texture that holds the whole frame at
|
||||
/// full resolution, so the shader measures the texture itself exactly as
|
||||
/// it did before windows existed.
|
||||
pub fn window_uniforms(&self) -> [f32; dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS] {
|
||||
let [x, y, w, h] = self.window;
|
||||
let (fw, fh) = if self.is_whole() {
|
||||
(0.0, 0.0)
|
||||
} else {
|
||||
(self.frame.0 as f32, self.frame.1 as f32)
|
||||
};
|
||||
[x, y, w, h, fw, fh, 0.0, 0.0]
|
||||
}
|
||||
|
||||
/// Whether the texture is the whole frame at full resolution.
|
||||
pub fn is_whole(&self) -> bool {
|
||||
self.window == WHOLE_FRAME && self.frame == (self.width, self.height)
|
||||
}
|
||||
|
||||
/// The window this texture holds, as origin and extent in normalised
|
||||
/// frame coordinates.
|
||||
pub fn window(&self) -> [f32; 4] {
|
||||
self.window
|
||||
}
|
||||
|
||||
/// Which texture this is, as a number that is never reused.
|
||||
///
|
||||
/// For a cache that has to know it is still looking at the same pixels
|
||||
@@ -274,6 +326,8 @@ impl DemosaicedImage {
|
||||
// the highlights of an image that was already finished.
|
||||
non_linear: true,
|
||||
id: next_image_id(),
|
||||
frame: (width, height),
|
||||
window: WHOLE_FRAME,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -289,6 +343,41 @@ impl DemosaicedImage {
|
||||
/// body that took its sources.
|
||||
pub fn from_linear_rgb16(ctx: &GpuContext, raw: &RawImage) -> Result<Self, GpuError> {
|
||||
let (width, height) = (raw.crop.width.max(1), raw.crop.height.max(1));
|
||||
Self::linear_rgb16_window(ctx, raw, [0, 0, width, height], 1)
|
||||
}
|
||||
|
||||
/// TRACES: FR-DSP-2 | NFR-RES-2
|
||||
/// Part of a linear DNG, or a reduced copy of it, for a photograph too
|
||||
/// large to hold in one texture.
|
||||
///
|
||||
/// `region` is `[x, y, width, height]` in pixels of the frame (the
|
||||
/// file's crop), clamped to it. `reduce` averages `reduce × reduce`
|
||||
/// blocks into one texel — a box filter, which is what a reduced copy
|
||||
/// that is only ever displayed smaller than itself needs, and which keeps
|
||||
/// the samples in scene-linear light where an average means something.
|
||||
///
|
||||
/// The texture then knows where it sits ([`Self::window`]) and how large
|
||||
/// the photograph is ([`Self::size`]), and the fused shader maps each
|
||||
/// output pixel's position in the *photograph* into it. So a crop, a
|
||||
/// rotation or a mask drawn on the reduced copy lands on the same pixels
|
||||
/// of a full-resolution window, and an export in tiles is the same
|
||||
/// picture as one that fitted.
|
||||
///
|
||||
/// Refused only if the result itself does not fit the device.
|
||||
pub fn linear_rgb16_window(
|
||||
ctx: &GpuContext,
|
||||
raw: &RawImage,
|
||||
region: [u32; 4],
|
||||
reduce: u32,
|
||||
) -> Result<Self, GpuError> {
|
||||
let frame = (raw.crop.width.max(1), raw.crop.height.max(1));
|
||||
let k = reduce.max(1);
|
||||
let x0 = region[0].min(frame.0 - 1);
|
||||
let y0 = region[1].min(frame.1 - 1);
|
||||
let rw = region[2].clamp(1, frame.0 - x0);
|
||||
let rh = region[3].clamp(1, frame.1 - y0);
|
||||
let (width, height) = (rw.div_ceil(k), rh.div_ceil(k));
|
||||
|
||||
let limits = ctx.device.limits();
|
||||
if width > limits.max_texture_dimension_2d || height > limits.max_texture_dimension_2d {
|
||||
return Err(GpuError::TooLarge(format!(
|
||||
@@ -308,19 +397,49 @@ impl DemosaicedImage {
|
||||
}
|
||||
let black = black_per_cell(raw);
|
||||
let inv = inv_range_per_cell(raw);
|
||||
// Per channel rather than per CFA cell: R, G, B are the first three.
|
||||
let mut half: Vec<u16> = Vec::with_capacity((width * height * 4) as usize);
|
||||
for y in 0..height as usize {
|
||||
let row = (raw.crop.y as usize + y) * stride + raw.crop.x as usize * 3;
|
||||
for x in 0..width as usize {
|
||||
let p = &raw.data[row + x * 3..row + x * 3 + 3];
|
||||
for c in 0..3 {
|
||||
let v = (f32::from(p[c]) - black[c]) * inv[c];
|
||||
half.push(f32_to_f16_bits_unclamped(v));
|
||||
|
||||
// One output row per task: a 200-megapixel reduction is a second of
|
||||
// one core, and the rows are independent.
|
||||
let row_texels = width as usize * 4;
|
||||
let mut half = vec![0u16; row_texels * height as usize];
|
||||
let fill_row = |ty: usize, out: &mut [u16]| {
|
||||
let sy0 = y0 as usize + ty * k as usize;
|
||||
let sy1 = (sy0 + k as usize).min((y0 + rh) as usize);
|
||||
for tx in 0..width as usize {
|
||||
let sx0 = x0 as usize + tx * k as usize;
|
||||
let sx1 = (sx0 + k as usize).min((x0 + rw) as usize);
|
||||
let mut acc = [0f32; 3];
|
||||
for sy in sy0..sy1 {
|
||||
let row = (raw.crop.y as usize + sy) * stride + raw.crop.x as usize * 3;
|
||||
for sx in sx0..sx1 {
|
||||
let p = &raw.data[row + sx * 3..row + sx * 3 + 3];
|
||||
for c in 0..3 {
|
||||
acc[c] += f32::from(p[c]);
|
||||
}
|
||||
}
|
||||
}
|
||||
half.push(f32_to_f16_bits(1.0));
|
||||
let n = ((sy1 - sy0) * (sx1 - sx0)).max(1) as f32;
|
||||
let texel = &mut out[tx * 4..tx * 4 + 4];
|
||||
for c in 0..3 {
|
||||
let v = (acc[c] / n - black[c]) * inv[c];
|
||||
texel[c] = f32_to_f16_bits_unclamped(v);
|
||||
}
|
||||
texel[3] = f32_to_f16_bits(1.0);
|
||||
}
|
||||
}
|
||||
};
|
||||
let threads = std::thread::available_parallelism().map_or(1, |n| n.get());
|
||||
let rows_per = (height as usize).div_ceil(threads).max(1);
|
||||
std::thread::scope(|scope| {
|
||||
for (chunk, rows) in half.chunks_mut(rows_per * row_texels).enumerate() {
|
||||
let fill_row = &fill_row;
|
||||
scope.spawn(move || {
|
||||
for (i, out) in rows.chunks_mut(row_texels).enumerate() {
|
||||
fill_row(chunk * rows_per + i, out);
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
let texture = ctx.device.create_texture_with_data(
|
||||
&ctx.queue,
|
||||
&wgpu::TextureDescriptor {
|
||||
@@ -341,6 +460,17 @@ impl DemosaicedImage {
|
||||
bytemuck::cast_slice(&half),
|
||||
);
|
||||
let view = texture.create_view(&Default::default());
|
||||
// The extent is the texels' own, `width × k`, not the region's: the
|
||||
// last block of a reduction may run past the frame's edge, and
|
||||
// stretching it to fit would put every texel slightly off the
|
||||
// pixels it averaged. The shader's bounds test is on the frame, so
|
||||
// nothing past the edge is ever read.
|
||||
let window = [
|
||||
x0 as f32 / frame.0 as f32,
|
||||
y0 as f32 / frame.1 as f32,
|
||||
(width * k) as f32 / frame.0 as f32,
|
||||
(height * k) as f32 / frame.1 as f32,
|
||||
];
|
||||
Ok(Self {
|
||||
texture,
|
||||
view,
|
||||
@@ -350,6 +480,8 @@ impl DemosaicedImage {
|
||||
as_shot_wb: [raw.wb_coeffs[0], raw.wb_coeffs[1], raw.wb_coeffs[2]],
|
||||
non_linear: false,
|
||||
id: next_image_id(),
|
||||
frame,
|
||||
window,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -768,6 +900,8 @@ impl Demosaicer {
|
||||
// transfer function.
|
||||
non_linear: false,
|
||||
id: next_image_id(),
|
||||
frame: (width, height),
|
||||
window: WHOLE_FRAME,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -928,7 +928,7 @@ impl MaskPass {
|
||||
// the only readers and they are skipped in that case.
|
||||
let source_step = match source {
|
||||
Some(image) => {
|
||||
let (sw, sh) = image.size();
|
||||
let (sw, sh) = image.texture_size();
|
||||
[
|
||||
sw as f32 / width.max(1) as f32,
|
||||
sh as f32 / height.max(1) as f32,
|
||||
|
||||
@@ -208,7 +208,7 @@ impl SegmentPass {
|
||||
source: &DemosaicedImage,
|
||||
opts: SegmentOptions,
|
||||
) -> Result<Segmentation, GpuError> {
|
||||
let (src_w, src_h) = source.size();
|
||||
let (src_w, src_h) = source.texture_size();
|
||||
let (width, height) = proxy_size(src_w, src_h, opts.max_edge);
|
||||
let n = (width * height) as u64;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user