Develop a linear DNG from windows and reduced copies of it
DemosaicedImage::linear_rgb16_window uploads part of a linear DNG, or a box-reduced copy of it, and says where it sits in the frame; size() now reports the frame and texture_size() the texels, and the fused pass writes the window into the shader's uniforms. EditGraph::source_region finds the part of the source a view reads, and tiles::plan cuts a render too large for one texture into halo-grown, grid-aligned tiles. The GPU test renders frames a tile at a time from their own windows and compares them with the whole: identical for point operations, within one code value when straightened with clarity on.
This commit is contained in:
@@ -1350,6 +1350,13 @@ impl AdjustPass {
|
||||
// runs. See `DemosaicedImage::is_non_linear`.
|
||||
let non_linear = if source.is_non_linear() { 1.0 } else { 0.0 };
|
||||
uniforms[12..16].copy_from_slice(&[wb[0], wb[1], wb[2], non_linear]);
|
||||
// TRACES: FR-DSP-2 | NFR-RES-2
|
||||
// Which part of the photograph the texture holds. The whole of it for
|
||||
// every source that fits in one texture, which writes back exactly
|
||||
// what the composer put there.
|
||||
let w = dr_pipeline::SOURCE_WINDOW_UNIFORM_OFFSET;
|
||||
uniforms[w..w + dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS]
|
||||
.copy_from_slice(&source.window_uniforms());
|
||||
uniforms
|
||||
}
|
||||
|
||||
|
||||
+145
-11
@@ -114,8 +114,20 @@ pub struct DemosaicedImage {
|
||||
/// Which upload this is, unique for the life of the process. See
|
||||
/// [`Self::id`].
|
||||
id: u64,
|
||||
/// TRACES: FR-DSP-2 | NFR-RES-2
|
||||
/// The whole frame's size in pixels — what [`Self::size`] reports.
|
||||
/// The texture's own size when it holds the whole frame at full
|
||||
/// resolution, which is every photograph that fits in one.
|
||||
frame: (u32, u32),
|
||||
/// Which part of the frame the texture holds, as origin and extent in
|
||||
/// normalised frame coordinates. `[0, 0, 1, 1]` for the whole frame,
|
||||
/// reduced or not. See [`Self::window_uniforms`].
|
||||
window: [f32; 4],
|
||||
}
|
||||
|
||||
/// The window of a texture that holds the whole frame.
|
||||
const WHOLE_FRAME: [f32; 4] = [0.0, 0.0, 1.0, 1.0];
|
||||
|
||||
/// The next [`DemosaicedImage::id`].
|
||||
fn next_image_id() -> u64 {
|
||||
static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1);
|
||||
@@ -133,10 +145,50 @@ impl DemosaicedImage {
|
||||
&self.view
|
||||
}
|
||||
|
||||
/// The size of the photograph this stands for, in its own pixels.
|
||||
///
|
||||
/// **Not necessarily the texture's.** For a photograph larger than one
|
||||
/// texture this is a reduced copy of it or a window cut from it, and
|
||||
/// everything that sizes a render, a crop or a kernel has to go on
|
||||
/// measuring the photograph. What indexes the texture's texels asks
|
||||
/// [`Self::texture_size`] instead.
|
||||
pub fn size(&self) -> (u32, u32) {
|
||||
self.frame
|
||||
}
|
||||
|
||||
/// The texture's own size in texels.
|
||||
pub fn texture_size(&self) -> (u32, u32) {
|
||||
(self.width, self.height)
|
||||
}
|
||||
|
||||
/// TRACES: FR-DSP-2 | NFR-RES-2
|
||||
/// The source window uniforms the fused shader reads, in the order
|
||||
/// `dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS` declares them.
|
||||
///
|
||||
/// The second `vec4` is zero for a texture that holds the whole frame at
|
||||
/// full resolution, so the shader measures the texture itself exactly as
|
||||
/// it did before windows existed.
|
||||
pub fn window_uniforms(&self) -> [f32; dr_pipeline::SOURCE_WINDOW_UNIFORM_FIELDS] {
|
||||
let [x, y, w, h] = self.window;
|
||||
let (fw, fh) = if self.is_whole() {
|
||||
(0.0, 0.0)
|
||||
} else {
|
||||
(self.frame.0 as f32, self.frame.1 as f32)
|
||||
};
|
||||
[x, y, w, h, fw, fh, 0.0, 0.0]
|
||||
}
|
||||
|
||||
/// Whether the texture is the whole frame at full resolution.
|
||||
pub fn is_whole(&self) -> bool {
|
||||
self.window == WHOLE_FRAME && self.frame == (self.width, self.height)
|
||||
}
|
||||
|
||||
/// The window this texture holds, as origin and extent in normalised
|
||||
/// frame coordinates.
|
||||
pub fn window(&self) -> [f32; 4] {
|
||||
self.window
|
||||
}
|
||||
|
||||
/// Which texture this is, as a number that is never reused.
|
||||
///
|
||||
/// For a cache that has to know it is still looking at the same pixels
|
||||
@@ -274,6 +326,8 @@ impl DemosaicedImage {
|
||||
// the highlights of an image that was already finished.
|
||||
non_linear: true,
|
||||
id: next_image_id(),
|
||||
frame: (width, height),
|
||||
window: WHOLE_FRAME,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -289,6 +343,41 @@ impl DemosaicedImage {
|
||||
/// body that took its sources.
|
||||
pub fn from_linear_rgb16(ctx: &GpuContext, raw: &RawImage) -> Result<Self, GpuError> {
|
||||
let (width, height) = (raw.crop.width.max(1), raw.crop.height.max(1));
|
||||
Self::linear_rgb16_window(ctx, raw, [0, 0, width, height], 1)
|
||||
}
|
||||
|
||||
/// TRACES: FR-DSP-2 | NFR-RES-2
|
||||
/// Part of a linear DNG, or a reduced copy of it, for a photograph too
|
||||
/// large to hold in one texture.
|
||||
///
|
||||
/// `region` is `[x, y, width, height]` in pixels of the frame (the
|
||||
/// file's crop), clamped to it. `reduce` averages `reduce × reduce`
|
||||
/// blocks into one texel — a box filter, which is what a reduced copy
|
||||
/// that is only ever displayed smaller than itself needs, and which keeps
|
||||
/// the samples in scene-linear light where an average means something.
|
||||
///
|
||||
/// The texture then knows where it sits ([`Self::window`]) and how large
|
||||
/// the photograph is ([`Self::size`]), and the fused shader maps each
|
||||
/// output pixel's position in the *photograph* into it. So a crop, a
|
||||
/// rotation or a mask drawn on the reduced copy lands on the same pixels
|
||||
/// of a full-resolution window, and an export in tiles is the same
|
||||
/// picture as one that fitted.
|
||||
///
|
||||
/// Refused only if the result itself does not fit the device.
|
||||
pub fn linear_rgb16_window(
|
||||
ctx: &GpuContext,
|
||||
raw: &RawImage,
|
||||
region: [u32; 4],
|
||||
reduce: u32,
|
||||
) -> Result<Self, GpuError> {
|
||||
let frame = (raw.crop.width.max(1), raw.crop.height.max(1));
|
||||
let k = reduce.max(1);
|
||||
let x0 = region[0].min(frame.0 - 1);
|
||||
let y0 = region[1].min(frame.1 - 1);
|
||||
let rw = region[2].clamp(1, frame.0 - x0);
|
||||
let rh = region[3].clamp(1, frame.1 - y0);
|
||||
let (width, height) = (rw.div_ceil(k), rh.div_ceil(k));
|
||||
|
||||
let limits = ctx.device.limits();
|
||||
if width > limits.max_texture_dimension_2d || height > limits.max_texture_dimension_2d {
|
||||
return Err(GpuError::TooLarge(format!(
|
||||
@@ -308,19 +397,49 @@ impl DemosaicedImage {
|
||||
}
|
||||
let black = black_per_cell(raw);
|
||||
let inv = inv_range_per_cell(raw);
|
||||
// Per channel rather than per CFA cell: R, G, B are the first three.
|
||||
let mut half: Vec<u16> = Vec::with_capacity((width * height * 4) as usize);
|
||||
for y in 0..height as usize {
|
||||
let row = (raw.crop.y as usize + y) * stride + raw.crop.x as usize * 3;
|
||||
for x in 0..width as usize {
|
||||
let p = &raw.data[row + x * 3..row + x * 3 + 3];
|
||||
for c in 0..3 {
|
||||
let v = (f32::from(p[c]) - black[c]) * inv[c];
|
||||
half.push(f32_to_f16_bits_unclamped(v));
|
||||
|
||||
// One output row per task: a 200-megapixel reduction is a second of
|
||||
// one core, and the rows are independent.
|
||||
let row_texels = width as usize * 4;
|
||||
let mut half = vec![0u16; row_texels * height as usize];
|
||||
let fill_row = |ty: usize, out: &mut [u16]| {
|
||||
let sy0 = y0 as usize + ty * k as usize;
|
||||
let sy1 = (sy0 + k as usize).min((y0 + rh) as usize);
|
||||
for tx in 0..width as usize {
|
||||
let sx0 = x0 as usize + tx * k as usize;
|
||||
let sx1 = (sx0 + k as usize).min((x0 + rw) as usize);
|
||||
let mut acc = [0f32; 3];
|
||||
for sy in sy0..sy1 {
|
||||
let row = (raw.crop.y as usize + sy) * stride + raw.crop.x as usize * 3;
|
||||
for sx in sx0..sx1 {
|
||||
let p = &raw.data[row + sx * 3..row + sx * 3 + 3];
|
||||
for c in 0..3 {
|
||||
acc[c] += f32::from(p[c]);
|
||||
}
|
||||
}
|
||||
}
|
||||
half.push(f32_to_f16_bits(1.0));
|
||||
let n = ((sy1 - sy0) * (sx1 - sx0)).max(1) as f32;
|
||||
let texel = &mut out[tx * 4..tx * 4 + 4];
|
||||
for c in 0..3 {
|
||||
let v = (acc[c] / n - black[c]) * inv[c];
|
||||
texel[c] = f32_to_f16_bits_unclamped(v);
|
||||
}
|
||||
texel[3] = f32_to_f16_bits(1.0);
|
||||
}
|
||||
}
|
||||
};
|
||||
let threads = std::thread::available_parallelism().map_or(1, |n| n.get());
|
||||
let rows_per = (height as usize).div_ceil(threads).max(1);
|
||||
std::thread::scope(|scope| {
|
||||
for (chunk, rows) in half.chunks_mut(rows_per * row_texels).enumerate() {
|
||||
let fill_row = &fill_row;
|
||||
scope.spawn(move || {
|
||||
for (i, out) in rows.chunks_mut(row_texels).enumerate() {
|
||||
fill_row(chunk * rows_per + i, out);
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
let texture = ctx.device.create_texture_with_data(
|
||||
&ctx.queue,
|
||||
&wgpu::TextureDescriptor {
|
||||
@@ -341,6 +460,17 @@ impl DemosaicedImage {
|
||||
bytemuck::cast_slice(&half),
|
||||
);
|
||||
let view = texture.create_view(&Default::default());
|
||||
// The extent is the texels' own, `width × k`, not the region's: the
|
||||
// last block of a reduction may run past the frame's edge, and
|
||||
// stretching it to fit would put every texel slightly off the
|
||||
// pixels it averaged. The shader's bounds test is on the frame, so
|
||||
// nothing past the edge is ever read.
|
||||
let window = [
|
||||
x0 as f32 / frame.0 as f32,
|
||||
y0 as f32 / frame.1 as f32,
|
||||
(width * k) as f32 / frame.0 as f32,
|
||||
(height * k) as f32 / frame.1 as f32,
|
||||
];
|
||||
Ok(Self {
|
||||
texture,
|
||||
view,
|
||||
@@ -350,6 +480,8 @@ impl DemosaicedImage {
|
||||
as_shot_wb: [raw.wb_coeffs[0], raw.wb_coeffs[1], raw.wb_coeffs[2]],
|
||||
non_linear: false,
|
||||
id: next_image_id(),
|
||||
frame,
|
||||
window,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -768,6 +900,8 @@ impl Demosaicer {
|
||||
// transfer function.
|
||||
non_linear: false,
|
||||
id: next_image_id(),
|
||||
frame: (width, height),
|
||||
window: WHOLE_FRAME,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -928,7 +928,7 @@ impl MaskPass {
|
||||
// the only readers and they are skipped in that case.
|
||||
let source_step = match source {
|
||||
Some(image) => {
|
||||
let (sw, sh) = image.size();
|
||||
let (sw, sh) = image.texture_size();
|
||||
[
|
||||
sw as f32 / width.max(1) as f32,
|
||||
sh as f32 / height.max(1) as f32,
|
||||
|
||||
@@ -208,7 +208,7 @@ impl SegmentPass {
|
||||
source: &DemosaicedImage,
|
||||
opts: SegmentOptions,
|
||||
) -> Result<Segmentation, GpuError> {
|
||||
let (src_w, src_h) = source.size();
|
||||
let (src_w, src_h) = source.texture_size();
|
||||
let (width, height) = proxy_size(src_w, src_h, opts.max_edge);
|
||||
let n = (width * height) as u64;
|
||||
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
//! TRACES: FR-DSP-2 | NFR-RES-2
|
||||
//! A photograph larger than one texture, developed from windows of it.
|
||||
//!
|
||||
//! The claim under test is that the window is invisible: a frame rendered a
|
||||
//! tile at a time, each tile from only the part of the source it reads, is the
|
||||
//! frame rendered whole. `dr-pipeline` can check the plan — the tiles cover
|
||||
//! the frame once, each is grown by the reach — but not that the shader's
|
||||
//! mapping into a window lands on the texel the whole texture would have
|
||||
//! given, which only a device answers.
|
||||
//!
|
||||
//! The frames here are small and the "device limit" is a number passed in,
|
||||
//! so the tiling is exercised on any adapter, including one whose real limit
|
||||
//! a test image could never approach.
|
||||
|
||||
use dr_decode::{CfaPattern, CropRect, RawImage};
|
||||
use dr_gpu::{AdjustPass, DemosaicedImage, GpuContext};
|
||||
use dr_pipeline::descriptor::{OpId, ParamId};
|
||||
use dr_pipeline::framing::ANGLE;
|
||||
use dr_pipeline::{tiles, Affects, EditGraph};
|
||||
use dr_types::ColourSpace;
|
||||
|
||||
fn ctx() -> Option<GpuContext> {
|
||||
match pollster::block_on(GpuContext::new_headless()) {
|
||||
Ok(c) => Some(c),
|
||||
Err(e) => {
|
||||
eprintln!("skipping: no GPU adapter ({e})");
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A linear RGB frame with detail at every scale: a slow gradient for the
|
||||
/// tone controls and a hash for the kernels, so a tile that read one pixel
|
||||
/// off would show.
|
||||
fn linear_frame(w: u32, h: u32, noise: bool) -> RawImage {
|
||||
let mut data = Vec::with_capacity((w * h * 3) as usize);
|
||||
for y in 0..h {
|
||||
for x in 0..w {
|
||||
let base = 4000.0 + 30000.0 * (x as f32 / w as f32) + 12000.0 * (y as f32 / h as f32);
|
||||
let hash = if noise {
|
||||
((x.wrapping_mul(73_856_093) ^ y.wrapping_mul(19_349_663)) % 8000) as f32
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
for c in 0..3 {
|
||||
data.push((base * (0.7 + 0.15 * c as f32) + hash) as u16);
|
||||
}
|
||||
}
|
||||
}
|
||||
RawImage {
|
||||
width: w,
|
||||
height: h,
|
||||
data,
|
||||
cfa_pattern: CfaPattern::Unknown,
|
||||
black_level: [512; 4],
|
||||
white_level: 65535,
|
||||
wb_coeffs: [2.0, 1.0, 1.5, 1.0],
|
||||
color_matrix: Some([1.6, -0.5, -0.1, -0.2, 1.4, -0.2, 0.0, -0.4, 1.4]),
|
||||
samples_per_pixel: 3,
|
||||
profile: None,
|
||||
make: String::new(),
|
||||
model: String::new(),
|
||||
crop: CropRect {
|
||||
x: 0,
|
||||
y: 0,
|
||||
width: w,
|
||||
height: h,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// Render `graph` over `source` at `size` and read it back.
|
||||
fn render(
|
||||
pass: &mut AdjustPass,
|
||||
graph: &EditGraph,
|
||||
source: &DemosaicedImage,
|
||||
size: (u32, u32),
|
||||
) -> Vec<u8> {
|
||||
let shader = graph.compose_for(ColourSpace::Srgb);
|
||||
let detail = graph.compose_detail(source.size(), size);
|
||||
let key = graph.invalidation().through(Affects::Colour);
|
||||
pass.render_detailed(source, &shader, size.0, size.1, None, &detail, key)
|
||||
.expect("render");
|
||||
pass.export_pixels().expect("readback").0
|
||||
}
|
||||
|
||||
/// The frame at full resolution, a tile at a time, each from its own window.
|
||||
fn render_tiled(
|
||||
ctx: &GpuContext,
|
||||
pass: &mut AdjustPass,
|
||||
graph: &mut EditGraph,
|
||||
raw: &RawImage,
|
||||
max_edge: u32,
|
||||
) -> (Vec<u8>, usize) {
|
||||
let frame = (raw.crop.width, raw.crop.height);
|
||||
let out = graph.output_size(frame.0, frame.1);
|
||||
let reach = graph.compose_detail(frame, out).reach();
|
||||
let plan = tiles::plan(out, max_edge, reach).expect("a plan");
|
||||
let mut pixels = vec![0u8; (out.0 * out.1 * 4) as usize];
|
||||
for t in &plan {
|
||||
graph.framing_mut().set_view(t.view(out));
|
||||
let r = graph.source_region(frame, 0);
|
||||
let x0 = (r.x * frame.0 as f32).floor() as u32;
|
||||
let y0 = (r.y * frame.1 as f32).floor() as u32;
|
||||
let x1 = ((r.x + r.width) * frame.0 as f32).ceil() as u32;
|
||||
let y1 = ((r.y + r.height) * frame.1 as f32).ceil() as u32;
|
||||
let window = DemosaicedImage::linear_rgb16_window(ctx, raw, [x0, y0, x1 - x0, y1 - y0], 1)
|
||||
.expect("window");
|
||||
assert_eq!(window.size(), frame, "a window measures the frame");
|
||||
let tile = render(pass, graph, &window, (t.grown[2], t.grown[3]));
|
||||
let (ox, oy) = t.keep_offset();
|
||||
for row in 0..t.keep[3] {
|
||||
let src = (((oy + row) * t.grown[2] + ox) * 4) as usize;
|
||||
let dst = (((t.keep[1] + row) * out.0 + t.keep[0]) * 4) as usize;
|
||||
let n = (t.keep[2] * 4) as usize;
|
||||
pixels[dst..dst + n].copy_from_slice(&tile[src..src + n]);
|
||||
}
|
||||
}
|
||||
graph
|
||||
.framing_mut()
|
||||
.set_view(dr_pipeline::CropRect::default());
|
||||
(pixels, plan.len())
|
||||
}
|
||||
|
||||
fn largest_difference(a: &[u8], b: &[u8]) -> u8 {
|
||||
a.iter()
|
||||
.zip(b)
|
||||
.map(|(x, y)| x.abs_diff(*y))
|
||||
.max()
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiles_of_windows_are_the_whole_frame() {
|
||||
// Point operations only, unrotated: every output pixel is an exact load
|
||||
// of one source texel, so the tiled frame has to be the whole one to
|
||||
// the bit.
|
||||
let Some(ctx) = ctx() else { return };
|
||||
let raw = linear_frame(200, 120, true);
|
||||
let mut graph = EditGraph::default_chain();
|
||||
graph.set_param(OpId("exposure"), ParamId("exposure"), 0.7);
|
||||
let mut pass = AdjustPass::new(&ctx);
|
||||
|
||||
let whole = DemosaicedImage::from_linear_rgb16(&ctx, &raw).unwrap();
|
||||
assert!(whole.is_whole());
|
||||
let reference = render(&mut pass, &graph, &whole, (200, 120));
|
||||
let (tiled, n) = render_tiled(&ctx, &mut pass, &mut graph, &raw, 64);
|
||||
assert!(n > 4, "the frame should have been cut, got {n} tile(s)");
|
||||
assert_eq!(largest_difference(&reference, &tiled), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_straightened_frame_with_clarity_tiles_without_seams() {
|
||||
// The hard case: a free angle samples between texels, and clarity reads
|
||||
// a wide neighbourhood on a reduced grid. The halo and the grid
|
||||
// alignment are what keep the tiles' edges out of the picture; a code
|
||||
// value of rounding is all that may differ.
|
||||
let Some(ctx) = ctx() else { return };
|
||||
let raw = linear_frame(320, 208, true);
|
||||
let mut graph = EditGraph::default_chain();
|
||||
graph.set_param(OpId("clarity"), ParamId("amount"), 60.0);
|
||||
graph.framing_mut().set_param(ANGLE, 3.0);
|
||||
let mut pass = AdjustPass::new(&ctx);
|
||||
|
||||
let whole = DemosaicedImage::from_linear_rgb16(&ctx, &raw).unwrap();
|
||||
let out = graph.output_size(320, 208);
|
||||
let reference = render(&mut pass, &graph, &whole, out);
|
||||
let (tiled, n) = render_tiled(&ctx, &mut pass, &mut graph, &raw, 160);
|
||||
assert!(n > 1, "the frame should have been cut, got {n} tile(s)");
|
||||
let worst = largest_difference(&reference, &tiled);
|
||||
assert!(
|
||||
worst <= 1,
|
||||
"tiles differ from the whole frame by {worst} code values"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_reduced_copy_stands_for_the_whole_frame() {
|
||||
// The canvas at fit renders from a copy reduced to fit the device. It
|
||||
// must measure the photograph, not itself, or a crop drawn on it lands
|
||||
// somewhere else in the export; and rendered small it must look like the
|
||||
// full frame rendered small.
|
||||
let Some(ctx) = ctx() else { return };
|
||||
let raw = linear_frame(400, 240, false);
|
||||
let mut graph = EditGraph::default_chain();
|
||||
graph.set_crop(dr_pipeline::CropRect {
|
||||
x: 0.25,
|
||||
y: 0.1,
|
||||
width: 0.5,
|
||||
height: 0.6,
|
||||
});
|
||||
let mut pass = AdjustPass::new(&ctx);
|
||||
|
||||
let whole = DemosaicedImage::from_linear_rgb16(&ctx, &raw).unwrap();
|
||||
let reduced = DemosaicedImage::linear_rgb16_window(&ctx, &raw, [0, 0, 400, 240], 3).unwrap();
|
||||
assert_eq!(reduced.size(), (400, 240));
|
||||
assert_eq!(reduced.texture_size(), (134, 80));
|
||||
assert!(!reduced.is_whole());
|
||||
|
||||
let size = (50, 36);
|
||||
let a = render(&mut pass, &graph, &whole, size);
|
||||
let b = render(&mut pass, &graph, &reduced, size);
|
||||
let worst = largest_difference(&a, &b);
|
||||
assert!(
|
||||
worst <= 3,
|
||||
"the reduced copy renders {worst} code values away"
|
||||
);
|
||||
}
|
||||
Reference in New Issue
Block a user