//! GPU device and compute for DarkRoom. //! //! In v0.1 this exists to prove one thing: a compute shader can write a //! texture that reaches the screen without a CPU round-trip (ARCH §6.1). It //! holds no pipeline, no tiling, and no masks — those arrive in v0.2. //! //! Deliberately free of UI dependencies (ARCH §6.5a). The texture is handed //! out as a `wgpu::Texture`; who composites it is not this crate's concern. //! //! That independence is why [`GpuContext::new_shared`] hands back the raw //! instance and adapter rather than talking to a compositor itself: the //! compositor will only sample a texture that came from the device *it* draws //! with, so somebody has to make one device for both — but it does not have to //! be this crate, and this crate must not know who it is. use std::sync::Arc; use wgpu::util::DeviceExt; mod adjust; mod demosaic; mod error; pub mod hierarchy; mod histogram; mod readback; mod segment; pub use adjust::AdjustPass; pub use demosaic::{DemosaicedImage, Demosaicer}; pub use error::GpuError; // Renamed on the way out: `BINS` says enough inside `histogram`, and nothing // at all at a crate root shared with demosaic and segmentation. pub use histogram::{Histogram, HistogramPass, BINS as HISTOGRAM_BINS}; pub use segment::{SegmentOptions, SegmentPass, Segmentation}; /// Owns the wgpu device and queue. /// /// One device is shared by the compute pipeline and the UI, which is what /// allows compositing with no interop layer. Cloning is cheap and shares the /// same underlying device. #[derive(Clone)] pub struct GpuContext { pub device: Arc, pub queue: Arc, adapter_info: wgpu::AdapterInfo, } /// TRACES: FR-DSP-1 | AC-8 /// One device, opened so that a compositor can be made to share it. /// /// The texture the adjust pass writes only reaches the screen without a copy /// if the compositor is drawing with the *same* `wgpu::Device` — two devices /// are two address spaces, and a texture from one is not a texture the other /// can sample. So the device cannot be an implementation detail of either /// side; it has to be made once and handed to both. /// /// [`Self::ctx`] is what the compute passes want. The instance and adapter are /// what a compositor wants in order to adopt the same setup — Slint's /// `WGPUConfiguration::Manual` asks for all four pieces — and they are handed /// out raw rather than wrapped, because naming Slint here would put a UI /// dependency in the one crate that must not have one (ARCH §6.5a). pub struct SharedGpu { /// The context every compute pass in this crate runs on. pub ctx: GpuContext, /// The instance the compositor will create its window surface from. pub instance: wgpu::Instance, /// The adapter [`Self::ctx`]'s device came from. pub adapter: wgpu::Adapter, } impl GpuContext { /// Create a headless context — no surface, no window. /// /// Used by tests, by the examples, and by anything that only needs to /// compute. A context opened this way cannot be shared with a compositor: /// see [`Self::new_shared`] for that, and for why the difference matters. pub async fn new_headless() -> Result { // GL is allowed alongside Vulkan here and nowhere else: a machine with // no Vulkan loader should still run the tests, and a headless context // never has to produce a window surface — which is precisely the thing // the GL backend cannot do from an instance opened without a display // handle. Self::open(wgpu::Backends::VULKAN | wgpu::Backends::GL) .await .map(|shared| shared.ctx) } /// TRACES: FR-DSP-1 | AC-8 /// Open a device intended to be shared with the compositor. /// /// Vulkan only, unlike [`Self::new_headless`]. The caller will hand the /// instance to a compositor that has to create a *window surface* from it, /// and wgpu's GL backend reaches its display through EGL at instance /// creation — an instance opened without a display handle, which is the /// only kind available before a window exists, cannot then produce a GL /// surface. Vulkan takes the window handle at surface creation instead, so /// it is the only backend this order of operations permits. /// /// A machine with no Vulkan therefore gets no shared device, and the /// caller is expected to carry on without the develop path rather than /// refuse to start. pub async fn new_shared() -> Result { // Vulkan on both targets (D1), and here it is not merely the // preference — see above. Self::open(wgpu::Backends::VULKAN).await } async fn open(backends: wgpu::Backends) -> Result { // `new_without_display_handle` rather than a struct literal: the // descriptor carries a boxed display handle and so has no `Default`, // and there is no window yet to take one from in either case. let mut descriptor = wgpu::InstanceDescriptor::new_without_display_handle(); descriptor.backends = backends; let instance = wgpu::Instance::new(descriptor); let adapter = instance .request_adapter(&wgpu::RequestAdapterOptions { power_preference: wgpu::PowerPreference::HighPerformance, compatible_surface: None, force_fallback_adapter: false, }) .await // A `Result` since wgpu 24, where it was an `Option`. The error // says which backends were tried, which is worth more than the // bare "no adapter" this used to report. .map_err(|_| GpuError::NoAdapter)?; let adapter_info = adapter.get_info(); log::info!( "gpu: {} ({:?}, {:?})", adapter_info.name, adapter_info.device_type, adapter_info.backend ); let (device, queue) = adapter .request_device(&wgpu::DeviceDescriptor { label: Some("darkroom-device"), required_features: wgpu::Features::empty(), // Defaults, not `downlevel_defaults`: storage textures // in compute shaders are required, and the downlevel tier // does not guarantee them. This is effectively our GPU // floor (NFR-COMPAT-1). // // `using_resolution` raises only the texture-dimension limits, // to whatever this adapter actually offers. That matters once // a compositor shares this device: the default ceiling is // 8192, and a swapchain image for a large or scaled display // can exceed it — a limit we chose for our own compute passes // would otherwise silently cap somebody else's window. required_limits: wgpu::Limits::default().using_resolution(adapter.limits()), memory_hints: wgpu::MemoryHints::Performance, // Nothing behind a feature flag wgpu itself calls unstable — // the pipeline is ordinary compute and storage textures. experimental_features: wgpu::ExperimentalFeatures::disabled(), // The API trace, absorbed into the descriptor in wgpu 25 from // the second argument this call used to take. trace: wgpu::Trace::Off, }) .await .map_err(|e| GpuError::DeviceRequest(e.to_string()))?; Ok(SharedGpu { ctx: Self { device: Arc::new(device), queue: Arc::new(queue), adapter_info, }, instance, adapter, }) } /// Build a context from a device and queue owned by someone else — the /// path used when Slint has already created them. pub fn from_parts( device: Arc, queue: Arc, adapter_info: wgpu::AdapterInfo, ) -> Self { Self { device, queue, adapter_info, } } pub fn adapter_name(&self) -> &str { &self.adapter_info.name } pub fn backend(&self) -> wgpu::Backend { self.adapter_info.backend } } #[repr(C)] #[derive(Copy, Clone, Debug, bytemuck::Pod, bytemuck::Zeroable)] struct Params { width: u32, height: u32, phase: f32, _pad: f32, } /// A compute pass writing into a storage texture. /// /// Stands in for the develop pipeline in v0.1. What matters is the shape: /// compute writes a texture, the texture is handed to the compositor, and /// pixels never travel back through the CPU. /// TRACES: FR-DEV-4 | R4 pub struct RenderTarget { ctx: GpuContext, texture: wgpu::Texture, view: wgpu::TextureView, pipeline: wgpu::ComputePipeline, bind_group_layout: wgpu::BindGroupLayout, bind_group: wgpu::BindGroup, params_buf: wgpu::Buffer, width: u32, height: u32, /// Reused staging buffer for the temporary readback path. Allocating one /// per frame is a significant cost at large window sizes. #[cfg(any(test, feature = "readback"))] readback_buf: std::cell::RefCell>, } impl RenderTarget { pub const FORMAT: wgpu::TextureFormat = wgpu::TextureFormat::Rgba8Unorm; pub fn new(ctx: &GpuContext, width: u32, height: u32) -> Result { let (width, height) = (width.max(1), height.max(1)); let shader = ctx .device .create_shader_module(wgpu::ShaderModuleDescriptor { label: Some("gradient"), source: wgpu::ShaderSource::Wgsl(include_str!("shaders/gradient.wgsl").into()), }); let bind_group_layout = ctx.device .create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { label: Some("render-target-bgl"), entries: &[ wgpu::BindGroupLayoutEntry { binding: 0, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::StorageTexture { access: wgpu::StorageTextureAccess::WriteOnly, format: Self::FORMAT, view_dimension: wgpu::TextureViewDimension::D2, }, count: None, }, wgpu::BindGroupLayoutEntry { binding: 1, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Uniform, has_dynamic_offset: false, min_binding_size: None, }, count: None, }, ], }); let layout = ctx .device .create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { label: Some("render-target-layout"), bind_group_layouts: &[Some(&bind_group_layout)], immediate_size: 0, }); let pipeline = ctx .device .create_compute_pipeline(&wgpu::ComputePipelineDescriptor { label: Some("gradient-pipeline"), layout: Some(&layout), module: &shader, entry_point: Some("main"), compilation_options: Default::default(), cache: None, }); let params_buf = ctx .device .create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("params"), contents: bytemuck::bytes_of(&Params { width, height, phase: 0.0, _pad: 0.0, }), usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST, }); let (texture, view) = Self::create_texture(ctx, width, height); let bind_group = Self::create_bind_group(ctx, &bind_group_layout, &view, ¶ms_buf); Ok(Self { ctx: ctx.clone(), texture, view, pipeline, bind_group_layout, bind_group, params_buf, width, height, #[cfg(any(test, feature = "readback"))] readback_buf: std::cell::RefCell::new(None), }) } fn create_texture( ctx: &GpuContext, width: u32, height: u32, ) -> (wgpu::Texture, wgpu::TextureView) { let texture = ctx.device.create_texture(&wgpu::TextureDescriptor { label: Some("render-target"), size: wgpu::Extent3d { width, height, depth_or_array_layers: 1, }, mip_level_count: 1, sample_count: 1, dimension: wgpu::TextureDimension::D2, format: Self::FORMAT, // STORAGE_BINDING to write from compute; TEXTURE_BINDING so the // compositor can sample it. COPY_SRC exists only for tests — // production never reads this back (ARCH §6.1). // // RENDER_ATTACHMENT is not something this pass ever uses. It is // there because Slint refuses to import a texture without it // (`TextureImportError::InvalidUsage`), the compositor having to // assume it may need to draw into what it was given. Declaring an // unused capability costs an allocation flag and buys the whole // zero-copy path, so it is a cheap price for AC-8. usage: wgpu::TextureUsages::STORAGE_BINDING | wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::RENDER_ATTACHMENT | wgpu::TextureUsages::COPY_SRC, view_formats: &[], }); let view = texture.create_view(&Default::default()); (texture, view) } fn create_bind_group( ctx: &GpuContext, layout: &wgpu::BindGroupLayout, view: &wgpu::TextureView, params: &wgpu::Buffer, ) -> wgpu::BindGroup { ctx.device.create_bind_group(&wgpu::BindGroupDescriptor { label: Some("render-target-bg"), layout, entries: &[ wgpu::BindGroupEntry { binding: 0, resource: wgpu::BindingResource::TextureView(view), }, wgpu::BindGroupEntry { binding: 1, resource: params.as_entire_binding(), }, ], }) } /// Resize, reallocating the texture. No-op when unchanged. pub fn resize(&mut self, width: u32, height: u32) { let (width, height) = (width.max(1), height.max(1)); if width == self.width && height == self.height { return; } let (texture, view) = Self::create_texture(&self.ctx, width, height); self.bind_group = Self::create_bind_group(&self.ctx, &self.bind_group_layout, &view, &self.params_buf); self.texture = texture; self.view = view; self.width = width; self.height = height; #[cfg(any(test, feature = "readback"))] { // Size changed, so the staging buffer no longer fits. *self.readback_buf.borrow_mut() = None; } } /// Run the compute pass. Results stay on the GPU. pub fn render(&self, phase: f32) { self.ctx.queue.write_buffer( &self.params_buf, 0, bytemuck::bytes_of(&Params { width: self.width, height: self.height, phase, _pad: 0.0, }), ); let mut enc = self .ctx .device .create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some("render-encoder"), }); { let mut pass = enc.begin_compute_pass(&wgpu::ComputePassDescriptor { label: Some("gradient-pass"), timestamp_writes: None, }); pass.set_pipeline(&self.pipeline); pass.set_bind_group(0, &self.bind_group, &[]); // 8x8 workgroups, rounded up so edge pixels are covered. pass.dispatch_workgroups(self.width.div_ceil(8), self.height.div_ceil(8), 1); } self.ctx.queue.submit(Some(enc.finish())); } pub fn texture(&self) -> &wgpu::Texture { &self.texture } pub fn view(&self) -> &wgpu::TextureView { &self.view } pub fn size(&self) -> (u32, u32) { (self.width, self.height) } /// Read pixels back to the CPU. /// /// **Tests only.** Production code must never call this — it is exactly /// the round-trip ARCH §6.1 forbids, and AC-8 asserts it does not happen. #[cfg(any(test, feature = "readback"))] pub async fn read_pixels(&self) -> Result, GpuError> { // Buffer rows must be aligned to COPY_BYTES_PER_ROW_ALIGNMENT (256). let unpadded = self.width * 4; let align = wgpu::COPY_BYTES_PER_ROW_ALIGNMENT; let padded = unpadded.div_ceil(align) * align; let needed = (padded * self.height) as u64; let mut slot = self.readback_buf.borrow_mut(); if slot.as_ref().map(|(_, p)| *p) != Some(padded) { *slot = Some(( self.ctx.device.create_buffer(&wgpu::BufferDescriptor { label: Some("readback"), size: needed, usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ, mapped_at_creation: false, }), padded, )); } let buf = &slot.as_ref().unwrap().0; let mut enc = self.ctx.device.create_command_encoder(&Default::default()); enc.copy_texture_to_buffer( wgpu::TexelCopyTextureInfo { texture: &self.texture, mip_level: 0, origin: wgpu::Origin3d::ZERO, aspect: wgpu::TextureAspect::All, }, wgpu::TexelCopyBufferInfo { buffer: buf, layout: wgpu::TexelCopyBufferLayout { offset: 0, bytes_per_row: Some(padded), rows_per_image: Some(self.height), }, }, wgpu::Extent3d { width: self.width, height: self.height, depth_or_array_layers: 1, }, ); self.ctx.queue.submit(Some(enc.finish())); let slice = buf.slice(..); let (tx, rx) = std::sync::mpsc::channel(); slice.map_async(wgpu::MapMode::Read, move |r| { let _ = tx.send(r); }); // Fallible since wgpu 26, and worth propagating rather than ignoring: // the failure it reports is a lost device (NFR-R7), and without this // the map callback below simply never arrives and the error surfaces // as a timeout somewhere less informative. self.ctx .device .poll(wgpu::PollType::wait_indefinitely()) .map_err(|e| GpuError::Readback(e.to_string()))?; rx.recv() .map_err(|e| GpuError::Readback(e.to_string()))? .map_err(|e| GpuError::Readback(e.to_string()))?; // Strip row padding. let data = slice.get_mapped_range(); let mut out = Vec::with_capacity((unpadded * self.height) as usize); for row in 0..self.height { let start = (row * padded) as usize; out.extend_from_slice(&data[start..start + unpadded as usize]); } drop(data); buf.unmap(); Ok(out) } } #[cfg(test)] mod tests { use super::*; fn ctx() -> Option { // CI runners and headless machines may have no usable adapter. Skip // rather than fail — the device-dependent assertions still run // wherever a GPU exists. match pollster::block_on(GpuContext::new_headless()) { Ok(c) => Some(c), Err(e) => { eprintln!("skipping: no GPU adapter ({e})"); None } } } #[test] fn compute_writes_the_texture() { let Some(ctx) = ctx() else { return }; let rt = RenderTarget::new(&ctx, 64, 64).expect("render target"); rt.render(0.0); let px = pollster::block_on(rt.read_pixels()).expect("readback"); assert_eq!(px.len(), 64 * 64 * 4); // The shader writes opaque pixels everywhere; an all-zero buffer would // mean the dispatch silently did nothing. assert!( px.chunks_exact(4).all(|p| p[3] == 255), "every pixel should be opaque" ); assert!( px.iter().any(|&b| b != 0), "texture should not be uniformly zero" ); } #[test] fn phase_changes_output() { let Some(ctx) = ctx() else { return }; let rt = RenderTarget::new(&ctx, 32, 32).expect("render target"); rt.render(0.0); let a = pollster::block_on(rt.read_pixels()).expect("readback"); rt.render(std::f32::consts::PI); let b = pollster::block_on(rt.read_pixels()).expect("readback"); assert_ne!(a, b, "moving the highlight should change the image"); } #[test] fn resize_reallocates() { let Some(ctx) = ctx() else { return }; let mut rt = RenderTarget::new(&ctx, 16, 16).expect("render target"); assert_eq!(rt.size(), (16, 16)); rt.resize(48, 24); assert_eq!(rt.size(), (48, 24)); rt.render(0.0); let px = pollster::block_on(rt.read_pixels()).expect("readback"); assert_eq!(px.len(), 48 * 24 * 4); } #[test] fn zero_size_is_clamped() { let Some(ctx) = ctx() else { return }; // A minimised window reports zero; texture creation would panic. let rt = RenderTarget::new(&ctx, 0, 0).expect("render target"); assert_eq!(rt.size(), (1, 1)); } }