//! The detail stage — running `dr-pipeline`'s neighbourhood passes. //! //! Where [`crate::AdjustPass`] fuses every point operation into one dispatch, //! this runs the operations that cannot be fused because they read pixels they //! are not writing: sharpening, noise reduction, clarity, texture, dehaze, //! spot removal (FR-DEV-3, FR-DEV-8). `dr_pipeline::detail` decides *what* they //! are and generates their WGSL; this compiles it, finds it somewhere to //! write, and dispatches it. //! //! # Nothing round-trips //! //! Every intermediate here is a `wgpu::Texture` and none of them is ever //! mapped. The chain is `demosaiced -> fused -> f16 -> f16 -> ... -> rgba8`, //! all of it on the device, and the last write lands in the same texture the //! compositor was already being handed. ARCH §6.1 and FR-DEV-4 are satisfied //! by there being no code here that could violate them, which is the only //! guarantee worth having. //! //! # Following the mask pass rather than inventing a second pattern //! //! `mask.rs` established how multi-target work is done in this crate, and this //! copies it deliberately: //! //! - **One encoder for the whole chain.** The mask pass rasterises every layer //! into one command buffer and submits once; this does the same for every //! pass. Submission order is the only synchronisation either needs, because //! both write and then read through the same queue. //! - **Textures reallocated on size change, never per frame.** `ensure_array` //! there, [`Intermediates::ensure`] here. Steady-state rendering at one //! viewport size allocates nothing. //! - **An allocation counter that exists to be asserted on.** Reallocating per //! frame instead of per resize costs a great deal of bandwidth and shows up //! nowhere in the output, which is exactly the kind of regression that needs //! a test that can see it. //! - **Pipelines cached by structure hash**, as `AdjustPass` caches its own. //! Moving a slider re-uploads a uniform buffer; it does not recompile. //! //! # The ping-pong, and why there are at most three textures //! //! Slot 0 holds what the fused colour pass wrote. It is kept **across frames**, //! which is what makes [`dr_pipeline::Affects::Detail`] mean something: when //! only a detail parameter has moved, the colour key is unchanged, the fused //! dispatch is skipped, and dragging a sharpening slider costs the detail //! passes alone (FR-DEV-3d). //! //! The remaining passes alternate between slots 1 and 2, and the last one //! writes the display texture directly rather than an intermediate — so a //! chain of *N* passes costs *N* dispatches and not *N* + 1, and there is no //! resolve pass to pay for. That leaves the allocation at `1 + min(N-1, 2)` //! textures: one for a single-pass operation, two for a separable blur, three //! however long the chain gets after that. //! //! # The reduced chain, and why a second one was needed //! //! A pass may declare [`dr_pipeline::detail::DetailPass::output_scale`] and //! write a target a fraction of the render size — clarity's base does, which //! is what TD-4 bought back. Such a pass cannot be part of the ping-pong //! above, and the reason is the shape of an unsharp mask rather than anything //! about textures: the pass that *combines* needs the blur **and** the //! full-resolution colour, and a colour that has been through a quarter-scale //! target is no longer full resolution. If the scaled passes wrote into the //! main chain they would destroy the very thing the last pass is going to //! subtract from. //! //! So there are two chains. The full-resolution one carries the colour and is //! untouched by a scaled pass; the reduced one carries the base. A scaled pass //! reads the reduced chain if anything has been written to it and the //! full-resolution chain otherwise — which is exactly "read what the pass //! before you wrote", the same rule as before. A full-resolution pass always //! reads the full-resolution chain, and sees the reduced one through binding 4 //! as `reduced_at()`. //! //! One reduced buffer, not one per operation. Two operations both wanting a //! reduced base in the same frame would need more, and nothing does: clarity //! is the only caller and texture's band is a decade finer, so it must stay at //! full resolution. A `debug_assert` in [`DetailRunner::encode`] holds that //! claim rather than leaving it as a comment. use std::collections::HashMap; use dr_pipeline::detail::{ComposedDetail, ComposedDetailPass}; use wgpu::util::DeviceExt as _; use crate::{GpuContext, GpuError}; /// The format every intermediate carries. /// /// The same `Rgba16Float` the demosaicer produces and the same one ARCH §5.2 /// names as the working precision (FR-DEV-2). It is not a free choice: the /// stage exists between the colour pass and the output transform precisely so /// that a kernel runs on linear values at full internal precision, and an /// 8-bit intermediate would quantise twice and convolve display-encoded /// numbers — which is how sharpening comes to band a clear sky. pub const INTERMEDIATE_FORMAT: wgpu::TextureFormat = wgpu::TextureFormat::Rgba16Float; /// One linear working texture. struct Slot { #[allow(dead_code)] texture: wgpu::Texture, view: wgpu::TextureView, } /// The pool of linear intermediates, sized to the chain and the viewport. struct Intermediates { slots: Vec, width: u32, height: u32, allocations: usize, } impl Intermediates { fn new() -> Self { Self { slots: Vec::new(), width: 0, height: 0, allocations: 0, } } /// Make sure `count` textures of this size exist. /// /// Grows but never shrinks within a size: an edit that briefly had a /// three-pass chain and then a one-pass one keeps the spare texture rather /// than freeing and reallocating it the next time the user turns the /// operation back on. A size change drops the lot, because none of them /// fits any more. fn ensure(&mut self, ctx: &GpuContext, count: usize, width: u32, height: u32) { if self.width != width || self.height != height { self.slots.clear(); self.width = width; self.height = height; } while self.slots.len() < count { let texture = ctx.device.create_texture(&wgpu::TextureDescriptor { label: Some("detail-intermediate"), size: wgpu::Extent3d { width, height, depth_or_array_layers: 1, }, mip_level_count: 1, sample_count: 1, dimension: wgpu::TextureDimension::D2, format: INTERMEDIATE_FORMAT, // STORAGE_BINDING to be written by a compute pass and // TEXTURE_BINDING to be read by the next one. Nothing else: // no RENDER_ATTACHMENT, because unlike the adjust pass's // output these are never handed to a compositor, and no // COPY_SRC, because nothing reads them back — that is the // point (ARCH §6.1). usage: wgpu::TextureUsages::STORAGE_BINDING | wgpu::TextureUsages::TEXTURE_BINDING, view_formats: &[], }); let view = texture.create_view(&Default::default()); self.slots.push(Slot { texture, view }); self.allocations += 1; } } /// TRACES: FR-PLAT-AND-5 /// Drop the pool, leaving it as [`Intermediates::new`] left it. /// /// The size is reset along with the slots, not merely because it is tidy: /// [`Self::ensure`] only refills when the count is short *or* the size /// differs, so a pool cleared while still claiming its old dimensions is /// indistinguishable from one that never held anything — which is fine /// here, and would stop being fine the moment `ensure` grew a fast path /// that trusted the stored size. `allocations` deliberately keeps /// counting: it exists so a test can see textures being made, and a /// counter reset on eviction would hide a reallocation storm rather than /// report one. fn release(&mut self) { self.slots.clear(); self.width = 0; self.height = 0; } } /// Runs the detail stage. /// /// Owned by [`crate::AdjustPass`] rather than standing alone, because the two /// halves are one render: the fused pass writes slot 0, this reads it, and the /// last pass writes the adjust pass's own output texture. Splitting them into /// two objects with two lifetimes would mean a caller could hold a stale /// intermediate against a fresh colour result and never be told. pub(crate) struct DetailRunner { ctx: GpuContext, /// Layout for a pass writing another linear intermediate. to_linear: Layout, /// Layout for the last pass, which writes the display texture. to_output: Layout, /// Compiled pipelines by pass structure hash. cache: HashMap, pool: Intermediates, /// The reduced chain — see the module documentation. Its own pool rather /// than more slots in `pool`, because its textures are a different size /// and [`Intermediates::ensure`] drops the lot when the size changes. reduced: Intermediates, /// See [`placeholder_instances`]. no_instances: wgpu::Buffer, /// See [`placeholder_reduced`]. no_reduced: wgpu::TextureView, } struct Layout { bind_group: wgpu::BindGroupLayout, pipeline: wgpu::PipelineLayout, } /// What binding 3 holds for a pass that declared no instance list. /// /// One zeroed element, allocated once. Zero-length storage buffers cannot be /// bound, and the passes that read this binding are exactly the ones that /// uploaded something of their own, so nothing ever reads the placeholder's /// contents — it exists to keep one bind group layout serving both kinds of /// pass. fn placeholder_instances(ctx: &GpuContext) -> wgpu::Buffer { ctx.device .create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("detail-instances-placeholder"), contents: bytemuck::cast_slice(&[[0.0f32; 4]]), usage: wgpu::BufferUsages::STORAGE, }) } /// What binding 4 holds for a pass that never calls `reduced_at`. /// /// The same trick as [`placeholder_instances`], for the same reason: one bind /// group layout has to serve a pass that reads the reduced chain and a pass /// that has never heard of it, and a binding cannot be left unbound. 1x1 and /// allocated once, so the cost of the arrangement is four bytes for the life /// of the runner. fn placeholder_reduced(ctx: &GpuContext) -> wgpu::TextureView { ctx.device .create_texture(&wgpu::TextureDescriptor { label: Some("detail-reduced-placeholder"), size: wgpu::Extent3d { width: 1, height: 1, depth_or_array_layers: 1, }, mip_level_count: 1, sample_count: 1, dimension: wgpu::TextureDimension::D2, format: INTERMEDIATE_FORMAT, usage: wgpu::TextureUsages::TEXTURE_BINDING, view_formats: &[], }) .create_view(&Default::default()) } impl DetailRunner { pub(crate) fn new(ctx: &GpuContext) -> Self { Self { ctx: ctx.clone(), to_linear: Layout::new(ctx, INTERMEDIATE_FORMAT, "detail-linear"), to_output: Layout::new(ctx, crate::AdjustPass::FORMAT, "detail-output"), cache: HashMap::new(), pool: Intermediates::new(), reduced: Intermediates::new(), no_instances: placeholder_instances(ctx), no_reduced: placeholder_reduced(ctx), } } /// The view the fused colour pass should write, given a chain of `passes`. /// /// Slot 0, always — it is the one that survives between frames so that a /// detail-only change can skip the colour dispatch entirely. pub(crate) fn colour_target( &mut self, passes: usize, width: u32, height: u32, ) -> &wgpu::TextureView { // One for the colour pass's result, then one per hand-off between // detail passes, capped at two because a ping-pong needs no more: the // last pass writes the display texture rather than an intermediate. let needed = 1 + passes.saturating_sub(1).min(2); self.pool.ensure(&self.ctx, needed, width, height); &self.pool.slots[0].view } /// Encode every pass of `chain`, the last one writing `output`. /// /// The caller must already have run the fused colour pass into /// [`Self::colour_target`] — or established that a previous frame's is /// still valid, which is the whole point of keeping slot 0. pub(crate) fn encode( &mut self, encoder: &mut wgpu::CommandEncoder, chain: &ComposedDetail, output: &wgpu::TextureView, width: u32, height: u32, ) -> Result { for pass in &chain.passes { self.compile(pass)?; } // The reduced chain's size, allocated once for the whole chain. // // One scale per chain, so the first scaled pass names the only scale // there is — see the module documentation for why one buffer is // enough, and the assertion for what would have to change. if let Some(scale) = chain .passes .iter() .map(|p| p.output_scale) .find(|&scale| scale > 1) { debug_assert!( chain .passes .iter() .all(|p| p.output_scale == 1 || p.output_scale == scale), "two reduced scales in one chain, and the runner holds one \ reduced buffer" ); // Two, for the ping-pong the reduce and the two blur halves need. // A separable blur cannot read the texture it is writing. self.reduced.ensure( &self.ctx, 2, width.div_ceil(scale).max(1), height.div_ceil(scale).max(1), ); } // Where each chain last wrote. `full` starts at slot 0 — what the // fused colour pass left there — and `carried` starts empty, which is // what makes the first scaled pass read the colour rather than an // uninitialised base. let mut full = 0usize; let mut full_writes = 0usize; let mut carried: Option = None; let mut reduced_writes = 0usize; for pass in chain.passes.iter() { let scaled = pass.output_scale > 1; // The last pass carries the output transform into the display // texture, which is the render size by definition. A scaled pass // there would bind a shader dispatching over a quarter-size grid // to a full-size target and write a quarter of the picture — a // wrong image rather than a validation failure, so it is caught // here and named. if scaled && pass.writes_output { return Err(GpuError::ShaderCompilation(format!( "detail pass {} declares output_scale {} and is last in \ the chain; the output transform is written at the render \ size", pass.label, pass.output_scale ))); } let (dispatch_w, dispatch_h) = if scaled { ( width.div_ceil(pass.output_scale).max(1), height.div_ceil(pass.output_scale).max(1), ) } else { (width, height) }; // Read what the previous pass in *this pass's own chain* wrote; // write the next slot of it, or the display texture if this is the // last pass. Alternating slots is what stops a pass reading the // texture it is writing — on a compute pass that is not an error // the driver reports, merely a picture that depends on scheduling. let source = match (scaled, carried) { (true, Some(slot)) => &self.reduced.slots[slot].view, _ => &self.pool.slots[full].view, }; let destination = if pass.writes_output { output } else if scaled { &self.reduced.slots[reduced_writes % 2].view } else { &self.pool.slots[1 + (full_writes % 2)].view }; // Binding 4. Present for every pass, because one bind group layout // serves both kinds; a pass that never calls `reduced_at` gets the // 1x1 placeholder and never reads it. let reduced_source = match carried { Some(slot) => &self.reduced.slots[slot].view, None => &self.no_reduced, }; let layout = if pass.writes_output { &self.to_output } else { &self.to_linear }; let params = self .ctx .device .create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("detail-params"), contents: bytemuck::cast_slice(&pass.uniforms), usage: wgpu::BufferUsages::UNIFORM, }); // TRACES: FR-DEV-8 // The instance list, uploaded only by the passes that have one. A // kernel pass — which is every pass that is a convolution — is // handed the placeholder allocated once in `new`, because a storage // buffer of length zero is not bindable and allocating a fresh // sixteen bytes per pass per frame is the per-frame allocation this // module's documentation exists to refuse. let instances = (!pass.storage.is_empty()).then(|| { self.ctx .device .create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("detail-instances"), contents: bytemuck::cast_slice(pass.storage.as_slice()), usage: wgpu::BufferUsages::STORAGE, }) }); let instances = instances.as_ref().unwrap_or(&self.no_instances); let bind_group = self .ctx .device .create_bind_group(&wgpu::BindGroupDescriptor { label: Some("detail-bg"), layout: &layout.bind_group, entries: &[ wgpu::BindGroupEntry { binding: 0, resource: wgpu::BindingResource::TextureView(source), }, wgpu::BindGroupEntry { binding: 1, resource: params.as_entire_binding(), }, wgpu::BindGroupEntry { binding: 2, resource: wgpu::BindingResource::TextureView(destination), }, wgpu::BindGroupEntry { binding: 3, resource: instances.as_entire_binding(), }, wgpu::BindGroupEntry { binding: 4, resource: wgpu::BindingResource::TextureView(reduced_source), }, ], }); let pipeline = self .cache .get(&pass.structure_hash) .expect("compiled above"); let mut compute = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor { label: Some(pass.label.as_str()), timestamp_writes: None, }); compute.set_pipeline(pipeline); compute.set_bind_group(0, &bind_group, &[]); compute.dispatch_workgroups(dispatch_w.div_ceil(8), dispatch_h.div_ceil(8), 1); drop(compute); if pass.writes_output { // Nothing downstream to hand anything to. } else if scaled { carried = Some(reduced_writes % 2); reduced_writes += 1; } else { full = 1 + (full_writes % 2); full_writes += 1; // A full-resolution pass consumes the reduced chain. It is the // combine — the base has been subtracted and now lives in the // colour — so a later operation must not be handed a base // belonging to this one. carried = None; } } Ok(chain.passes.len()) } /// Compile one pass, or leave the cached pipeline in place. /// /// A validation error here is a codegen bug rather than anything the user /// did, so it is caught in an error scope and returned with the generated /// source and the pass's label attached — a line number against code /// nobody wrote, from one of several passes, is otherwise close to /// unactionable. fn compile(&mut self, pass: &ComposedDetailPass) -> Result<(), GpuError> { if self.cache.contains_key(&pass.structure_hash) { return Ok(()); } let scope = self .ctx .device .push_error_scope(wgpu::ErrorFilter::Validation); let module = self .ctx .device .create_shader_module(wgpu::ShaderModuleDescriptor { label: Some(pass.label.as_str()), source: wgpu::ShaderSource::Wgsl(pass.source.as_str().into()), }); let layout = if pass.writes_output { &self.to_output } else { &self.to_linear }; let pipeline = self .ctx .device .create_compute_pipeline(&wgpu::ComputePipelineDescriptor { label: Some(pass.label.as_str()), layout: Some(&layout.pipeline), module: &module, entry_point: Some("main"), compilation_options: Default::default(), cache: None, }); if let Some(err) = pollster::block_on(scope.pop()) { return Err(GpuError::ShaderCompilation(format!( "detail pass {}: {err}\n\n--- generated source ---\n{}", pass.label, crate::adjust::numbered(&pass.source) ))); } self.cache.insert(pass.structure_hash, pipeline); Ok(()) } /// How many distinct detail pipelines are compiled. For tests asserting /// that slider movement does not recompile. pub(crate) fn cached_pipelines(&self) -> usize { self.cache.len() } /// TRACES: FR-PLAT-AND-5 /// Give back everything this stage is only holding to be fast. /// /// Both pools and the pipeline cache. Nothing here is state: a pool slot /// is re-created by the next [`Intermediates::ensure`] and a pipeline by /// the next compile-on-miss, so the only cost of this call is the work of /// doing both again. pub(crate) fn release_caches(&mut self) { self.cache.clear(); self.pool.release(); self.reduced.release(); } /// How many intermediate textures have been allocated since this pass was /// created. For tests — see [`crate::MaskPass::allocations`] for the /// regression this shape of counter exists to catch. pub(crate) fn allocations(&self) -> usize { // Both pools. A reduced buffer reallocated every frame is exactly the // regression this counter exists to catch, and counting only the // full-resolution one would hide it. self.pool.allocations + self.reduced.allocations } } /// A read-only storage buffer entry, as `mask.rs` declares its strokes. fn storage_entry(binding: u32) -> wgpu::BindGroupLayoutEntry { wgpu::BindGroupLayoutEntry { binding, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Storage { read_only: true }, has_dynamic_offset: false, min_binding_size: None, }, count: None, } } impl Layout { fn new(ctx: &GpuContext, format: wgpu::TextureFormat, label: &str) -> Self { let bind_group = ctx .device .create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { label: Some(label), entries: &[ // The previous stage's result. wgpu::BindGroupLayoutEntry { binding: 0, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Texture { sample_type: wgpu::TextureSampleType::Float { filterable: true }, view_dimension: wgpu::TextureViewDimension::D2, multisampled: false, }, count: None, }, wgpu::BindGroupLayoutEntry { binding: 1, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Uniform, has_dynamic_offset: false, min_binding_size: None, }, count: None, }, wgpu::BindGroupLayoutEntry { binding: 2, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::StorageTexture { access: wgpu::StorageTextureAccess::WriteOnly, format, view_dimension: wgpu::TextureViewDimension::D2, }, count: None, }, // TRACES: FR-DEV-8 // The instance list, for a pass whose work is a list rather // than a kernel (`DetailPass::storage`). Every other pass // gets `Intermediates`' placeholder here — one entry on both // layouts rather than two more layouts, since a convolution // that never reads the buffer costs nothing for it being // bound. storage_entry(3), // The reduced chain, for a pass that calls `reduced_at`. // Bound on every layout for the same reason binding 3 is: // a pass that never reads it costs nothing for it being // there, and two more layouts would cost a great deal more // than that. wgpu::BindGroupLayoutEntry { binding: 4, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Texture { sample_type: wgpu::TextureSampleType::Float { filterable: true }, view_dimension: wgpu::TextureViewDimension::D2, multisampled: false, }, count: None, }, ], }); let pipeline = ctx .device .create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { label: Some(label), bind_group_layouts: &[Some(&bind_group)], immediate_size: 0, }); Self { bind_group, pipeline, } } }