//! The detail stage — running `dr-pipeline`'s neighbourhood passes. //! //! Where [`crate::AdjustPass`] fuses every point operation into one dispatch, //! this runs the operations that cannot be fused because they read pixels they //! are not writing: sharpening, noise reduction, clarity, texture, dehaze, //! spot removal (FR-DEV-3, FR-DEV-8). `dr_pipeline::detail` decides *what* they //! are and generates their WGSL; this compiles it, finds it somewhere to //! write, and dispatches it. //! //! # Nothing round-trips //! //! Every intermediate here is a `wgpu::Texture` and none of them is ever //! mapped. The chain is `demosaiced -> fused -> f16 -> f16 -> ... -> rgba8`, //! all of it on the device, and the last write lands in the same texture the //! compositor was already being handed. ARCH §6.1 and FR-DEV-4 are satisfied //! by there being no code here that could violate them, which is the only //! guarantee worth having. //! //! # Following the mask pass rather than inventing a second pattern //! //! `mask.rs` established how multi-target work is done in this crate, and this //! copies it deliberately: //! //! - **One encoder for the whole chain.** The mask pass rasterises every layer //! into one command buffer and submits once; this does the same for every //! pass. Submission order is the only synchronisation either needs, because //! both write and then read through the same queue. //! - **Textures reallocated on size change, never per frame.** `ensure_array` //! there, [`Intermediates::ensure`] here. Steady-state rendering at one //! viewport size allocates nothing. //! - **An allocation counter that exists to be asserted on.** Reallocating per //! frame instead of per resize costs a great deal of bandwidth and shows up //! nowhere in the output, which is exactly the kind of regression that needs //! a test that can see it. //! - **Pipelines cached by structure hash**, as `AdjustPass` caches its own. //! Moving a slider re-uploads a uniform buffer; it does not recompile. //! //! # The ping-pong, and why there are at most three textures //! //! Slot 0 holds what the fused colour pass wrote. It is kept **across frames**, //! which is what makes [`dr_pipeline::Affects::Detail`] mean something: when //! only a detail parameter has moved, the colour key is unchanged, the fused //! dispatch is skipped, and dragging a sharpening slider costs the detail //! passes alone (FR-DEV-3d). //! //! The remaining passes alternate between slots 1 and 2, and the last one //! writes the display texture directly rather than an intermediate — so a //! chain of *N* passes costs *N* dispatches and not *N* + 1, and there is no //! resolve pass to pay for. That leaves the allocation at `1 + min(N-1, 2)` //! textures: one for a single-pass operation, two for a separable blur, three //! however long the chain gets after that. use std::collections::HashMap; use dr_pipeline::detail::{ComposedDetail, ComposedDetailPass}; use wgpu::util::DeviceExt as _; use crate::{GpuContext, GpuError}; /// The format every intermediate carries. /// /// The same `Rgba16Float` the demosaicer produces and the same one ARCH §5.2 /// names as the working precision (FR-DEV-2). It is not a free choice: the /// stage exists between the colour pass and the output transform precisely so /// that a kernel runs on linear values at full internal precision, and an /// 8-bit intermediate would quantise twice and convolve display-encoded /// numbers — which is how sharpening comes to band a clear sky. pub const INTERMEDIATE_FORMAT: wgpu::TextureFormat = wgpu::TextureFormat::Rgba16Float; /// One linear working texture. struct Slot { #[allow(dead_code)] texture: wgpu::Texture, view: wgpu::TextureView, } /// The pool of linear intermediates, sized to the chain and the viewport. struct Intermediates { slots: Vec, width: u32, height: u32, allocations: usize, } impl Intermediates { fn new() -> Self { Self { slots: Vec::new(), width: 0, height: 0, allocations: 0, } } /// Make sure `count` textures of this size exist. /// /// Grows but never shrinks within a size: an edit that briefly had a /// three-pass chain and then a one-pass one keeps the spare texture rather /// than freeing and reallocating it the next time the user turns the /// operation back on. A size change drops the lot, because none of them /// fits any more. fn ensure(&mut self, ctx: &GpuContext, count: usize, width: u32, height: u32) { if self.width != width || self.height != height { self.slots.clear(); self.width = width; self.height = height; } while self.slots.len() < count { let texture = ctx.device.create_texture(&wgpu::TextureDescriptor { label: Some("detail-intermediate"), size: wgpu::Extent3d { width, height, depth_or_array_layers: 1, }, mip_level_count: 1, sample_count: 1, dimension: wgpu::TextureDimension::D2, format: INTERMEDIATE_FORMAT, // STORAGE_BINDING to be written by a compute pass and // TEXTURE_BINDING to be read by the next one. Nothing else: // no RENDER_ATTACHMENT, because unlike the adjust pass's // output these are never handed to a compositor, and no // COPY_SRC, because nothing reads them back — that is the // point (ARCH §6.1). usage: wgpu::TextureUsages::STORAGE_BINDING | wgpu::TextureUsages::TEXTURE_BINDING, view_formats: &[], }); let view = texture.create_view(&Default::default()); self.slots.push(Slot { texture, view }); self.allocations += 1; } } } /// Runs the detail stage. /// /// Owned by [`crate::AdjustPass`] rather than standing alone, because the two /// halves are one render: the fused pass writes slot 0, this reads it, and the /// last pass writes the adjust pass's own output texture. Splitting them into /// two objects with two lifetimes would mean a caller could hold a stale /// intermediate against a fresh colour result and never be told. pub(crate) struct DetailRunner { ctx: GpuContext, /// Layout for a pass writing another linear intermediate. to_linear: Layout, /// Layout for the last pass, which writes the display texture. to_output: Layout, /// Compiled pipelines by pass structure hash. cache: HashMap, pool: Intermediates, /// See [`placeholder_instances`]. no_instances: wgpu::Buffer, } struct Layout { bind_group: wgpu::BindGroupLayout, pipeline: wgpu::PipelineLayout, } /// What binding 3 holds for a pass that declared no instance list. /// /// One zeroed element, allocated once. Zero-length storage buffers cannot be /// bound, and the passes that read this binding are exactly the ones that /// uploaded something of their own, so nothing ever reads the placeholder's /// contents — it exists to keep one bind group layout serving both kinds of /// pass. fn placeholder_instances(ctx: &GpuContext) -> wgpu::Buffer { ctx.device .create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("detail-instances-placeholder"), contents: bytemuck::cast_slice(&[[0.0f32; 4]]), usage: wgpu::BufferUsages::STORAGE, }) } impl DetailRunner { pub(crate) fn new(ctx: &GpuContext) -> Self { Self { ctx: ctx.clone(), to_linear: Layout::new(ctx, INTERMEDIATE_FORMAT, "detail-linear"), to_output: Layout::new(ctx, crate::AdjustPass::FORMAT, "detail-output"), cache: HashMap::new(), pool: Intermediates::new(), no_instances: placeholder_instances(ctx), } } /// The view the fused colour pass should write, given a chain of `passes`. /// /// Slot 0, always — it is the one that survives between frames so that a /// detail-only change can skip the colour dispatch entirely. pub(crate) fn colour_target( &mut self, passes: usize, width: u32, height: u32, ) -> &wgpu::TextureView { // One for the colour pass's result, then one per hand-off between // detail passes, capped at two because a ping-pong needs no more: the // last pass writes the display texture rather than an intermediate. let needed = 1 + passes.saturating_sub(1).min(2); self.pool.ensure(&self.ctx, needed, width, height); &self.pool.slots[0].view } /// Encode every pass of `chain`, the last one writing `output`. /// /// The caller must already have run the fused colour pass into /// [`Self::colour_target`] — or established that a previous frame's is /// still valid, which is the whole point of keeping slot 0. pub(crate) fn encode( &mut self, encoder: &mut wgpu::CommandEncoder, chain: &ComposedDetail, output: &wgpu::TextureView, width: u32, height: u32, ) -> Result { for pass in &chain.passes { self.compile(pass)?; } for (index, pass) in chain.passes.iter().enumerate() { // Read what the previous pass wrote; write the next slot, or the // display texture if this is the last one. `index % 2` alternates // between slots 1 and 2, so a pass never reads the texture it is // writing — which on a compute pass is not an error the driver // reports, merely a picture that depends on scheduling. let source_slot = if index == 0 { 0 } else { 2 - (index % 2) }; let source = &self.pool.slots[source_slot].view; let destination = if pass.writes_output { output } else { &self.pool.slots[1 + (index % 2)].view }; let layout = if pass.writes_output { &self.to_output } else { &self.to_linear }; let params = self .ctx .device .create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("detail-params"), contents: bytemuck::cast_slice(&pass.uniforms), usage: wgpu::BufferUsages::UNIFORM, }); // TRACES: FR-DEV-8 // The instance list, uploaded only by the passes that have one. A // kernel pass — which is every pass that is a convolution — is // handed the placeholder allocated once in `new`, because a storage // buffer of length zero is not bindable and allocating a fresh // sixteen bytes per pass per frame is the per-frame allocation this // module's documentation exists to refuse. let instances = (!pass.storage.is_empty()).then(|| { self.ctx .device .create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("detail-instances"), contents: bytemuck::cast_slice(pass.storage.as_slice()), usage: wgpu::BufferUsages::STORAGE, }) }); let instances = instances.as_ref().unwrap_or(&self.no_instances); let bind_group = self .ctx .device .create_bind_group(&wgpu::BindGroupDescriptor { label: Some("detail-bg"), layout: &layout.bind_group, entries: &[ wgpu::BindGroupEntry { binding: 0, resource: wgpu::BindingResource::TextureView(source), }, wgpu::BindGroupEntry { binding: 1, resource: params.as_entire_binding(), }, wgpu::BindGroupEntry { binding: 2, resource: wgpu::BindingResource::TextureView(destination), }, wgpu::BindGroupEntry { binding: 3, resource: instances.as_entire_binding(), }, ], }); let pipeline = self .cache .get(&pass.structure_hash) .expect("compiled above"); let mut compute = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor { label: Some(pass.label.as_str()), timestamp_writes: None, }); compute.set_pipeline(pipeline); compute.set_bind_group(0, &bind_group, &[]); compute.dispatch_workgroups(width.div_ceil(8), height.div_ceil(8), 1); } Ok(chain.passes.len()) } /// Compile one pass, or leave the cached pipeline in place. /// /// A validation error here is a codegen bug rather than anything the user /// did, so it is caught in an error scope and returned with the generated /// source and the pass's label attached — a line number against code /// nobody wrote, from one of several passes, is otherwise close to /// unactionable. fn compile(&mut self, pass: &ComposedDetailPass) -> Result<(), GpuError> { if self.cache.contains_key(&pass.structure_hash) { return Ok(()); } let scope = self .ctx .device .push_error_scope(wgpu::ErrorFilter::Validation); let module = self .ctx .device .create_shader_module(wgpu::ShaderModuleDescriptor { label: Some(pass.label.as_str()), source: wgpu::ShaderSource::Wgsl(pass.source.as_str().into()), }); let layout = if pass.writes_output { &self.to_output } else { &self.to_linear }; let pipeline = self .ctx .device .create_compute_pipeline(&wgpu::ComputePipelineDescriptor { label: Some(pass.label.as_str()), layout: Some(&layout.pipeline), module: &module, entry_point: Some("main"), compilation_options: Default::default(), cache: None, }); if let Some(err) = pollster::block_on(scope.pop()) { return Err(GpuError::ShaderCompilation(format!( "detail pass {}: {err}\n\n--- generated source ---\n{}", pass.label, crate::adjust::numbered(&pass.source) ))); } self.cache.insert(pass.structure_hash, pipeline); Ok(()) } /// How many distinct detail pipelines are compiled. For tests asserting /// that slider movement does not recompile. pub(crate) fn cached_pipelines(&self) -> usize { self.cache.len() } /// How many intermediate textures have been allocated since this pass was /// created. For tests — see [`crate::MaskPass::allocations`] for the /// regression this shape of counter exists to catch. pub(crate) fn allocations(&self) -> usize { self.pool.allocations } } /// A read-only storage buffer entry, as `mask.rs` declares its strokes. fn storage_entry(binding: u32) -> wgpu::BindGroupLayoutEntry { wgpu::BindGroupLayoutEntry { binding, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Storage { read_only: true }, has_dynamic_offset: false, min_binding_size: None, }, count: None, } } impl Layout { fn new(ctx: &GpuContext, format: wgpu::TextureFormat, label: &str) -> Self { let bind_group = ctx .device .create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { label: Some(label), entries: &[ // The previous stage's result. wgpu::BindGroupLayoutEntry { binding: 0, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Texture { sample_type: wgpu::TextureSampleType::Float { filterable: true }, view_dimension: wgpu::TextureViewDimension::D2, multisampled: false, }, count: None, }, wgpu::BindGroupLayoutEntry { binding: 1, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Uniform, has_dynamic_offset: false, min_binding_size: None, }, count: None, }, wgpu::BindGroupLayoutEntry { binding: 2, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::StorageTexture { access: wgpu::StorageTextureAccess::WriteOnly, format, view_dimension: wgpu::TextureViewDimension::D2, }, count: None, }, // TRACES: FR-DEV-8 // The instance list, for a pass whose work is a list rather // than a kernel (`DetailPass::storage`). Every other pass // gets `Intermediates`' placeholder here — one entry on both // layouts rather than two more layouts, since a convolution // that never reads the buffer costs nothing for it being // bound. storage_entry(3), ], }); let pipeline = ctx .device .create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { label: Some(label), bind_group_layouts: &[Some(&bind_group)], immediate_size: 0, }); Self { bind_group, pipeline, } } }