//! TRACES: FR-DSP-7 //! Counting the display frame, on the device that drew it. //! //! # Why a compute reduction and not a CPU pass over the readback //! //! There is, today, a whole frame already sitting in CPU memory every time the //! canvas updates — `AdjustPass::read_output`, the temporary bridge that spike //! S1 removes. Walking it to build a histogram would have been perhaps thirty //! lines and no shader at all, and it was the obvious thing to reach for. //! //! It was rejected for two reasons, in this order. //! //! FR-DSP-7 states the mechanism, not just the feature: "these derive from a //! GPU-side reduction into a small buffer. Per-frame CPU readback of image data //! is prohibited." A histogram built on the bridge would be correct today and //! *deleted* by S1 — it would be new code whose only foundation is the one //! thing the architecture is committed to removing, and the histogram would //! then be the reason the bridge could not go. //! //! And the cost does not scale the way the shortcut implies. Counting 2 MP on //! one CPU thread is several milliseconds of the settle frame; the reduction //! below is a fraction of one, and what crosses the bus is 4104 bytes //! regardless of the image. The simpler-looking option is simpler only while //! the frame happens to be lying there. //! //! # What it costs and when it runs //! //! One dispatch plus a 4 KB buffer copy and a mapping — a device sync point. //! FR-DSP-7 requires that this not extend the FR-DSP-3 frame budget, so the //! interface runs it on the *settled* frame only, never on the draft frames a //! drag produces. The histogram of an image being dragged past is not read //! anyway; the one that arrives when the slider stops is. use wgpu::util::DeviceExt; use crate::readback::await_mapping; use crate::{GpuContext, GpuError}; /// Levels per channel. 256, so a bin *is* an output code value and no /// re-bucketing stands between the count and what the display shows. pub const BINS: usize = 256; /// Four channel histograms plus the two clip counters, as the shader lays them /// out. Kept next to the shader's own constants because the two must agree. const CHANNELS: usize = 4; const CLIPPED_HIGH: usize = CHANNELS * BINS; const CLIPPED_LOW: usize = CHANNELS * BINS + 1; const SLOTS: usize = CHANNELS * BINS + 2; /// TRACES: FR-DSP-7 /// A counted frame: how many pixels sit at each output level. /// /// Counts, not proportions. Turning these into something drawable — folding /// 256 bins into the columns a 280px panel can show, choosing a peak to scale /// against — is presentation, and belongs to whoever is drawing (ARCH §4.3a). /// What this crate owes is the numbers. #[derive(Clone, Debug, PartialEq, Eq)] pub struct Histogram { red: [u32; BINS], green: [u32; BINS], blue: [u32; BINS], luma: [u32; BINS], clipped_highlights: u32, clipped_shadows: u32, pixels: u32, } impl Histogram { /// Counts per output level, darkest first. pub fn red(&self) -> &[u32; BINS] { &self.red } pub fn green(&self) -> &[u32; BINS] { &self.green } pub fn blue(&self) -> &[u32; BINS] { &self.blue } /// Rec.709 luma of the encoded values — the axis a photographer reads /// exposure off. See the shader for why it is weighted in fixed point. pub fn luma(&self) -> &[u32; BINS] { &self.luma } /// Pixels with **any** channel at 255, and with any channel at 0. /// /// Any rather than all, because a single blown channel is detail that is /// already gone: a red that has hit the ceiling has no gradation left in it /// however much green and blue still hold. pub fn clipped_highlights(&self) -> u32 { self.clipped_highlights } pub fn clipped_shadows(&self) -> u32 { self.clipped_shadows } /// Pixels counted. The denominator for the two figures above. pub fn pixels(&self) -> u32 { self.pixels } /// Rebuild from the flat slot array the shader writes. /// /// `pixels` is summed from the red channel rather than taken from the image /// dimensions: every pixel lands in exactly one red bin, so the sum *is* /// the count, and deriving it that way makes a dropped or double-counted /// texel show up as a wrong denominator instead of hiding. fn from_slots(slots: &[u32]) -> Result { if slots.len() < SLOTS { return Err(GpuError::Readback(format!( "histogram readback was {} slots, expected {SLOTS}", slots.len() ))); } let channel = |i: usize| -> [u32; BINS] { let mut out = [0u32; BINS]; out.copy_from_slice(&slots[i * BINS..(i + 1) * BINS]); out }; let red = channel(0); Ok(Self { pixels: red.iter().sum(), red, green: channel(1), blue: channel(2), luma: channel(3), clipped_highlights: slots[CLIPPED_HIGH], clipped_shadows: slots[CLIPPED_LOW], }) } } /// The dispatch's view of the frame. Padded to 16 bytes for std140. #[repr(C)] #[derive(Copy, Clone, bytemuck::Pod, bytemuck::Zeroable)] struct Dims { width: u32, height: u32, pad_0: u32, pad_1: u32, } /// TRACES: FR-DSP-7 /// Counts a rendered frame into [`Histogram`]. /// /// Holds its buffers for the life of the session. They are a fixed 4104 bytes /// whatever the image size — the one property that makes this affordable — so /// there is nothing to reallocate when the viewport changes, unlike the display /// target beside it. pub struct HistogramPass { ctx: GpuContext, pipeline: wgpu::ComputePipeline, bind_group_layout: wgpu::BindGroupLayout, /// Where the shader accumulates. Cleared before each dispatch. bins: wgpu::Buffer, /// Mappable destination; a storage buffer cannot also be `MAP_READ`. staging: wgpu::Buffer, dims: wgpu::Buffer, } impl HistogramPass { pub fn new(ctx: &GpuContext) -> Result { // A validation error here is a bug in the shader beside this file, not // anything a user did — surfaced as a `Result` rather than left to // wgpu's default handler, which panics. let scope = ctx.device.push_error_scope(wgpu::ErrorFilter::Validation); let module = ctx .device .create_shader_module(wgpu::ShaderModuleDescriptor { label: Some("histogram"), source: wgpu::ShaderSource::Wgsl(include_str!("shaders/histogram.wgsl").into()), }); let bind_group_layout = ctx.device .create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { label: Some("histogram-bgl"), entries: &[ // The rendered frame, sampled with `textureLoad` — the // same texture the compositor shows, so what is counted // is what is on screen. wgpu::BindGroupLayoutEntry { binding: 0, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Texture { sample_type: wgpu::TextureSampleType::Float { filterable: true }, view_dimension: wgpu::TextureViewDimension::D2, multisampled: false, }, count: None, }, wgpu::BindGroupLayoutEntry { binding: 1, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Storage { read_only: false }, has_dynamic_offset: false, min_binding_size: None, }, count: None, }, wgpu::BindGroupLayoutEntry { binding: 2, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Uniform, has_dynamic_offset: false, min_binding_size: None, }, count: None, }, ], }); let layout = ctx .device .create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { label: Some("histogram-layout"), bind_group_layouts: &[Some(&bind_group_layout)], immediate_size: 0, }); let pipeline = ctx .device .create_compute_pipeline(&wgpu::ComputePipelineDescriptor { label: Some("histogram-pipeline"), layout: Some(&layout), module: &module, entry_point: Some("main"), compilation_options: Default::default(), cache: None, }); if let Some(err) = pollster::block_on(scope.pop()) { return Err(GpuError::ShaderCompilation(err.to_string())); } let bytes = (SLOTS * std::mem::size_of::()) as u64; let bins = ctx.device.create_buffer(&wgpu::BufferDescriptor { label: Some("histogram-bins"), size: bytes, usage: wgpu::BufferUsages::STORAGE | wgpu::BufferUsages::COPY_SRC | wgpu::BufferUsages::COPY_DST, mapped_at_creation: false, }); let staging = ctx.device.create_buffer(&wgpu::BufferDescriptor { label: Some("histogram-staging"), size: bytes, usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ, mapped_at_creation: false, }); let dims = ctx .device .create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("histogram-dims"), contents: bytemuck::bytes_of(&Dims { width: 0, height: 0, pad_0: 0, pad_1: 0, }), usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST, }); Ok(Self { ctx: ctx.clone(), pipeline, bind_group_layout, bins, staging, dims, }) } /// TRACES: FR-DSP-7 /// Count one rendered frame. /// /// `texture` must carry `TEXTURE_BINDING`, which `AdjustPass`'s output /// does because the compositor samples it. pub fn compute(&self, texture: &wgpu::Texture) -> Result { let (width, height) = (texture.width(), texture.height()); self.ctx.queue.write_buffer( &self.dims, 0, bytemuck::bytes_of(&Dims { width, height, pad_0: 0, pad_1: 0, }), ); let view = texture.create_view(&Default::default()); let bind_group = self .ctx .device .create_bind_group(&wgpu::BindGroupDescriptor { label: Some("histogram-bg"), layout: &self.bind_group_layout, entries: &[ wgpu::BindGroupEntry { binding: 0, resource: wgpu::BindingResource::TextureView(&view), }, wgpu::BindGroupEntry { binding: 1, resource: self.bins.as_entire_binding(), }, wgpu::BindGroupEntry { binding: 2, resource: self.dims.as_entire_binding(), }, ], }); let mut enc = self .ctx .device .create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some("histogram-encoder"), }); // The accumulator is reused between frames, so it carries the previous // frame's counts until this line. Forgetting it does not fail — it // quietly integrates every frame since the image opened, which looks // like a histogram that will not respond to the exposure slider. enc.clear_buffer(&self.bins, 0, None); { let mut pass = enc.begin_compute_pass(&wgpu::ComputePassDescriptor { label: Some("histogram-pass"), timestamp_writes: None, }); pass.set_pipeline(&self.pipeline); pass.set_bind_group(0, &bind_group, &[]); pass.dispatch_workgroups(width.div_ceil(16), height.div_ceil(16), 1); } enc.copy_buffer_to_buffer(&self.bins, 0, &self.staging, 0, self.staging.size()); self.ctx.queue.submit(Some(enc.finish())); let slice = self.staging.slice(..); let (tx, rx) = std::sync::mpsc::channel(); slice.map_async(wgpu::MapMode::Read, move |r| { let _ = tx.send(r); }); await_mapping(&self.ctx, &rx)?; let data = slice.get_mapped_range(); let slots: Vec = bytemuck::cast_slice::(&data).to_vec(); drop(data); self.staging.unmap(); Histogram::from_slots(&slots) } } #[cfg(test)] mod tests { use super::*; fn ctx() -> Option { match pollster::block_on(GpuContext::new_headless()) { Ok(c) => Some(c), Err(e) => { eprintln!("skipping: no GPU adapter ({e})"); None } } } /// Upload `rgba` as a texture the pass can read, the way `AdjustPass` /// hands its output over. fn texture(ctx: &GpuContext, rgba: &[u8], width: u32, height: u32) -> wgpu::Texture { let tex = ctx.device.create_texture(&wgpu::TextureDescriptor { label: Some("histogram-test-source"), size: wgpu::Extent3d { width, height, depth_or_array_layers: 1, }, mip_level_count: 1, sample_count: 1, dimension: wgpu::TextureDimension::D2, format: wgpu::TextureFormat::Rgba8Unorm, usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST, view_formats: &[], }); ctx.queue.write_texture( wgpu::TexelCopyTextureInfo { texture: &tex, mip_level: 0, origin: wgpu::Origin3d::ZERO, aspect: wgpu::TextureAspect::All, }, rgba, wgpu::TexelCopyBufferLayout { offset: 0, bytes_per_row: Some(width * 4), rows_per_image: Some(height), }, wgpu::Extent3d { width, height, depth_or_array_layers: 1, }, ); ctx.queue.submit(std::iter::empty()); tex } /// The reference the shader is checked against: the same bucketing, written /// the obvious way on the CPU. /// /// Deliberately a *second* implementation rather than shared code. The bugs /// this is here to catch — a workgroup tile that is never merged, an edge /// tile counted twice, a luma weighting that carries past 255 — are all /// bugs a shared implementation would commit identically on both sides and /// so could not detect. fn expected(rgba: &[u8]) -> Histogram { let mut slots = vec![0u32; SLOTS]; for px in rgba.chunks_exact(4) { let (r, g, b) = (px[0] as usize, px[1] as usize, px[2] as usize); let y = (54 * r + 183 * g + 19 * b) >> 8; slots[r] += 1; slots[BINS + g] += 1; slots[2 * BINS + b] += 1; slots[3 * BINS + y] += 1; if px[0] == 255 || px[1] == 255 || px[2] == 255 { slots[CLIPPED_HIGH] += 1; } if px[0] == 0 || px[1] == 0 || px[2] == 0 { slots[CLIPPED_LOW] += 1; } } Histogram::from_slots(&slots).expect("slot count") } #[test] fn a_flat_frame_puts_every_pixel_in_one_bin() { // The arithmetic at its most checkable: 64x64 pixels of one value must // produce exactly 4096 in exactly one bin and nothing anywhere else. // A tile that failed to merge, or merged twice, changes this number — // and a histogram that is merely "roughly right" is a histogram nobody // can set a black point from. let Some(ctx) = ctx() else { return }; let pass = HistogramPass::new(&ctx).expect("pass"); let rgba: Vec = std::iter::repeat_n([90u8, 140, 200, 255], 64 * 64) .flatten() .collect(); let tex = texture(&ctx, &rgba, 64, 64); let hist = pass.compute(&tex).expect("compute"); assert_eq!(hist.pixels(), 4096); assert_eq!(hist.red()[90], 4096); assert_eq!(hist.green()[140], 4096); assert_eq!(hist.blue()[200], 4096); assert_eq!( hist.red().iter().filter(|c| **c > 0).count(), 1, "one value can only occupy one bin" ); // (54*90 + 183*140 + 19*200) >> 8 = 34280 >> 8 = 133. assert_eq!(hist.luma()[133], 4096); assert_eq!(hist.clipped_highlights(), 0); assert_eq!(hist.clipped_shadows(), 0); } #[test] fn every_level_is_reachable_and_lands_where_it_belongs() { // A ramp covering all 256 codes, four pixels each. This is the test // that would catch an off-by-one in the quantisation — a `floor` where // a rounding was needed shifts the whole ramp down one bin and leaves // 255 empty, which on a real photograph looks like nothing at all. let Some(ctx) = ctx() else { return }; let pass = HistogramPass::new(&ctx).expect("pass"); let (w, h) = (256u32, 4u32); let mut rgba = Vec::with_capacity((w * h * 4) as usize); for _ in 0..h { for x in 0..w { let v = x as u8; rgba.extend_from_slice(&[v, v, v, 255]); } } let tex = texture(&ctx, &rgba, w, h); let hist = pass.compute(&tex).expect("compute"); assert_eq!(hist.pixels(), w * h); for level in 0..BINS { assert_eq!( hist.red()[level], h, "level {level} should hold exactly {h} pixels" ); // Neutral, so luma must land on the same bin as the channels do. assert_eq!(hist.luma()[level], h, "luma drifted at level {level}"); } } #[test] fn the_shader_agrees_with_a_cpu_count_of_the_same_frame() { // The cross-check, on a frame with no structure for a wrong dispatch to // hide behind: a size that is not a multiple of the 16x16 workgroup, so // the edge tiles run off the image, and pseudo-random content so every // bin is occupied unevenly. Exact equality — the reduction is integer // throughout precisely so this can be an `assert_eq`, not a tolerance. let Some(ctx) = ctx() else { return }; let pass = HistogramPass::new(&ctx).expect("pass"); let (w, h) = (101u32, 37u32); let mut rgba = Vec::with_capacity((w * h * 4) as usize); let mut state = 0x2545_F491_4F6C_DD1Du64; for _ in 0..w * h { for _ in 0..3 { // xorshift64*, so the frame is identical on every machine and a // failure can be reproduced rather than merely observed. state ^= state >> 12; state ^= state << 25; state ^= state >> 27; rgba.push((state.wrapping_mul(0x2545_F491_4F6C_DD1D) >> 56) as u8); } rgba.push(255); } let tex = texture(&ctx, &rgba, w, h); let hist = pass.compute(&tex).expect("compute"); assert_eq!(hist.pixels(), w * h, "an edge tile was dropped or doubled"); assert_eq!(hist, expected(&rgba)); } #[test] fn clipping_is_counted_per_pixel_and_not_per_channel() { // The distinction the indicator rests on. A pixel with two channels at // the ceiling is *one* clipped pixel; counting channels would report // 200% of a frame clipped, and a percentage that can exceed 100 is a // readout nobody will trust again. let Some(ctx) = ctx() else { return }; let pass = HistogramPass::new(&ctx).expect("pass"); let rgba: Vec = [ // Two channels blown, one pixel clipped. [255u8, 255, 10, 255], // One channel blown — still clipped, which is the point of "any". [255, 10, 10, 255], // Clean. [10, 10, 10, 255], // Black in one channel only: a clipped shadow. [0, 10, 10, 255], ] .concat(); let tex = texture(&ctx, &rgba, 4, 1); let hist = pass.compute(&tex).expect("compute"); assert_eq!(hist.pixels(), 4); assert_eq!(hist.clipped_highlights(), 2); assert_eq!(hist.clipped_shadows(), 1); } #[test] fn a_second_frame_replaces_the_first_rather_than_adding_to_it() { // The accumulator is reused, so a missing clear integrates every frame // since the session opened. The symptom is subtle and awful: the // histogram keeps its shape and simply stops responding to the sliders, // because each frame's contribution shrinks against the running total. let Some(ctx) = ctx() else { return }; let pass = HistogramPass::new(&ctx).expect("pass"); let dark: Vec = std::iter::repeat_n([40u8, 40, 40, 255], 16 * 16) .flatten() .collect(); let bright: Vec = std::iter::repeat_n([210u8, 210, 210, 255], 16 * 16) .flatten() .collect(); let first = pass.compute(&texture(&ctx, &dark, 16, 16)).expect("first"); assert_eq!(first.red()[40], 256); let second = pass .compute(&texture(&ctx, &bright, 16, 16)) .expect("second"); assert_eq!(second.pixels(), 256, "the previous frame was still counted"); assert_eq!(second.red()[40], 0); assert_eq!(second.red()[210], 256); } }