Initial workspace: GPU context, compute pass, adaptive Slint shell
Establishes the v0.1 foundations on both platforms: - dr-types: SourceRef (never a filesystem path — Android SAF has none), Format, Availability, Validator with ETag quote normalisation - dr-gpu: wgpu device, compute pass writing a storage texture, resize - dr-ui: Slint shell with FR-UI-1 adaptive layout, computed in Rust to avoid a binding loop - docker/android: pinned toolchain, verified producing API 28 ARM binaries Measured the cost of the temporary CPU readback path (dr-gpu bench): compute is 0.06-0.28ms across sizes while readback is 0.63-7.43ms, so readback is 90-96% of frame time and scales with area. Recorded in ARCH §6.1 — this is why spike S1 is the priority. Mitigations pending S1: reuse the staging buffer, apply at most one resize per frame, and cap render resolution at 2048 on the long edge. 10 tests passing; core crates cross-compile for aarch64-linux-android.
This commit is contained in:
@@ -0,0 +1,494 @@
|
||||
//! GPU device and compute for DarkRoom.
|
||||
//!
|
||||
//! In v0.1 this exists to prove one thing: a compute shader can write a
|
||||
//! texture that reaches the screen without a CPU round-trip (ARCH §6.1). It
|
||||
//! holds no pipeline, no tiling, and no masks — those arrive in v0.2.
|
||||
//!
|
||||
//! Deliberately free of UI dependencies (ARCH §6.5a). The texture is handed
|
||||
//! out as a `wgpu::Texture`; who composites it is not this crate's concern.
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use wgpu::util::DeviceExt;
|
||||
|
||||
mod error;
|
||||
pub use error::GpuError;
|
||||
|
||||
/// Owns the wgpu device and queue.
|
||||
///
|
||||
/// One device is shared by the compute pipeline and the UI, which is what
|
||||
/// allows compositing with no interop layer. Cloning is cheap and shares the
|
||||
/// same underlying device.
|
||||
#[derive(Clone)]
|
||||
pub struct GpuContext {
|
||||
pub device: Arc<wgpu::Device>,
|
||||
pub queue: Arc<wgpu::Queue>,
|
||||
adapter_info: wgpu::AdapterInfo,
|
||||
}
|
||||
|
||||
impl GpuContext {
|
||||
/// Create a headless context — no surface, no window.
|
||||
///
|
||||
/// Used by tests and by the Slint path, which supplies its own surface.
|
||||
pub async fn new_headless() -> Result<Self, GpuError> {
|
||||
let instance = wgpu::Instance::new(wgpu::InstanceDescriptor {
|
||||
// Vulkan on both targets (D1). GL is allowed as a fallback so a
|
||||
// machine without a Vulkan loader still runs the tests.
|
||||
backends: wgpu::Backends::VULKAN | wgpu::Backends::GL,
|
||||
..Default::default()
|
||||
});
|
||||
|
||||
let adapter = instance
|
||||
.request_adapter(&wgpu::RequestAdapterOptions {
|
||||
power_preference: wgpu::PowerPreference::HighPerformance,
|
||||
compatible_surface: None,
|
||||
force_fallback_adapter: false,
|
||||
})
|
||||
.await
|
||||
.ok_or(GpuError::NoAdapter)?;
|
||||
|
||||
let adapter_info = adapter.get_info();
|
||||
log::info!(
|
||||
"gpu: {} ({:?}, {:?})",
|
||||
adapter_info.name,
|
||||
adapter_info.device_type,
|
||||
adapter_info.backend
|
||||
);
|
||||
|
||||
let (device, queue) = adapter
|
||||
.request_device(
|
||||
&wgpu::DeviceDescriptor {
|
||||
label: Some("darkroom-device"),
|
||||
required_features: wgpu::Features::empty(),
|
||||
// Defaults, not `downlevel_defaults`: storage textures
|
||||
// in compute shaders are required, and the downlevel tier
|
||||
// does not guarantee them. This is effectively our GPU
|
||||
// floor (NFR-COMPAT-1).
|
||||
required_limits: wgpu::Limits::default(),
|
||||
memory_hints: wgpu::MemoryHints::Performance,
|
||||
},
|
||||
None,
|
||||
)
|
||||
.await
|
||||
.map_err(|e| GpuError::DeviceRequest(e.to_string()))?;
|
||||
|
||||
Ok(Self {
|
||||
device: Arc::new(device),
|
||||
queue: Arc::new(queue),
|
||||
adapter_info,
|
||||
})
|
||||
}
|
||||
|
||||
/// Build a context from a device and queue owned by someone else — the
|
||||
/// path used when Slint has already created them.
|
||||
pub fn from_parts(
|
||||
device: Arc<wgpu::Device>,
|
||||
queue: Arc<wgpu::Queue>,
|
||||
adapter_info: wgpu::AdapterInfo,
|
||||
) -> Self {
|
||||
Self {
|
||||
device,
|
||||
queue,
|
||||
adapter_info,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn adapter_name(&self) -> &str {
|
||||
&self.adapter_info.name
|
||||
}
|
||||
|
||||
pub fn backend(&self) -> wgpu::Backend {
|
||||
self.adapter_info.backend
|
||||
}
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Copy, Clone, Debug, bytemuck::Pod, bytemuck::Zeroable)]
|
||||
struct Params {
|
||||
width: u32,
|
||||
height: u32,
|
||||
phase: f32,
|
||||
_pad: f32,
|
||||
}
|
||||
|
||||
/// A compute pass writing into a storage texture.
|
||||
///
|
||||
/// Stands in for the develop pipeline in v0.1. What matters is the shape:
|
||||
/// compute writes a texture, the texture is handed to the compositor, and
|
||||
/// pixels never travel back through the CPU.
|
||||
pub struct RenderTarget {
|
||||
ctx: GpuContext,
|
||||
texture: wgpu::Texture,
|
||||
view: wgpu::TextureView,
|
||||
pipeline: wgpu::ComputePipeline,
|
||||
bind_group_layout: wgpu::BindGroupLayout,
|
||||
bind_group: wgpu::BindGroup,
|
||||
params_buf: wgpu::Buffer,
|
||||
width: u32,
|
||||
height: u32,
|
||||
/// Reused staging buffer for the temporary readback path. Allocating one
|
||||
/// per frame is a significant cost at large window sizes.
|
||||
#[cfg(any(test, feature = "readback"))]
|
||||
readback_buf: std::cell::RefCell<Option<(wgpu::Buffer, u32)>>,
|
||||
}
|
||||
|
||||
impl RenderTarget {
|
||||
pub const FORMAT: wgpu::TextureFormat = wgpu::TextureFormat::Rgba8Unorm;
|
||||
|
||||
pub fn new(ctx: &GpuContext, width: u32, height: u32) -> Result<Self, GpuError> {
|
||||
let (width, height) = (width.max(1), height.max(1));
|
||||
|
||||
let shader = ctx
|
||||
.device
|
||||
.create_shader_module(wgpu::ShaderModuleDescriptor {
|
||||
label: Some("gradient"),
|
||||
source: wgpu::ShaderSource::Wgsl(
|
||||
include_str!("shaders/gradient.wgsl").into(),
|
||||
),
|
||||
});
|
||||
|
||||
let bind_group_layout =
|
||||
ctx.device
|
||||
.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
|
||||
label: Some("render-target-bgl"),
|
||||
entries: &[
|
||||
wgpu::BindGroupLayoutEntry {
|
||||
binding: 0,
|
||||
visibility: wgpu::ShaderStages::COMPUTE,
|
||||
ty: wgpu::BindingType::StorageTexture {
|
||||
access: wgpu::StorageTextureAccess::WriteOnly,
|
||||
format: Self::FORMAT,
|
||||
view_dimension: wgpu::TextureViewDimension::D2,
|
||||
},
|
||||
count: None,
|
||||
},
|
||||
wgpu::BindGroupLayoutEntry {
|
||||
binding: 1,
|
||||
visibility: wgpu::ShaderStages::COMPUTE,
|
||||
ty: wgpu::BindingType::Buffer {
|
||||
ty: wgpu::BufferBindingType::Uniform,
|
||||
has_dynamic_offset: false,
|
||||
min_binding_size: None,
|
||||
},
|
||||
count: None,
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
let layout = ctx
|
||||
.device
|
||||
.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
|
||||
label: Some("render-target-layout"),
|
||||
bind_group_layouts: &[&bind_group_layout],
|
||||
push_constant_ranges: &[],
|
||||
});
|
||||
|
||||
let pipeline = ctx
|
||||
.device
|
||||
.create_compute_pipeline(&wgpu::ComputePipelineDescriptor {
|
||||
label: Some("gradient-pipeline"),
|
||||
layout: Some(&layout),
|
||||
module: &shader,
|
||||
entry_point: Some("main"),
|
||||
compilation_options: Default::default(),
|
||||
cache: None,
|
||||
});
|
||||
|
||||
let params_buf = ctx
|
||||
.device
|
||||
.create_buffer_init(&wgpu::util::BufferInitDescriptor {
|
||||
label: Some("params"),
|
||||
contents: bytemuck::bytes_of(&Params {
|
||||
width,
|
||||
height,
|
||||
phase: 0.0,
|
||||
_pad: 0.0,
|
||||
}),
|
||||
usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
|
||||
});
|
||||
|
||||
let (texture, view) = Self::create_texture(ctx, width, height);
|
||||
let bind_group = Self::create_bind_group(ctx, &bind_group_layout, &view, ¶ms_buf);
|
||||
|
||||
Ok(Self {
|
||||
ctx: ctx.clone(),
|
||||
texture,
|
||||
view,
|
||||
pipeline,
|
||||
bind_group_layout,
|
||||
bind_group,
|
||||
params_buf,
|
||||
width,
|
||||
height,
|
||||
#[cfg(any(test, feature = "readback"))]
|
||||
readback_buf: std::cell::RefCell::new(None),
|
||||
})
|
||||
}
|
||||
|
||||
fn create_texture(
|
||||
ctx: &GpuContext,
|
||||
width: u32,
|
||||
height: u32,
|
||||
) -> (wgpu::Texture, wgpu::TextureView) {
|
||||
let texture = ctx.device.create_texture(&wgpu::TextureDescriptor {
|
||||
label: Some("render-target"),
|
||||
size: wgpu::Extent3d {
|
||||
width,
|
||||
height,
|
||||
depth_or_array_layers: 1,
|
||||
},
|
||||
mip_level_count: 1,
|
||||
sample_count: 1,
|
||||
dimension: wgpu::TextureDimension::D2,
|
||||
format: Self::FORMAT,
|
||||
// STORAGE_BINDING to write from compute; TEXTURE_BINDING so the
|
||||
// compositor can sample it. COPY_SRC exists only for tests —
|
||||
// production never reads this back (ARCH §6.1).
|
||||
usage: wgpu::TextureUsages::STORAGE_BINDING
|
||||
| wgpu::TextureUsages::TEXTURE_BINDING
|
||||
| wgpu::TextureUsages::COPY_SRC,
|
||||
view_formats: &[],
|
||||
});
|
||||
let view = texture.create_view(&Default::default());
|
||||
(texture, view)
|
||||
}
|
||||
|
||||
fn create_bind_group(
|
||||
ctx: &GpuContext,
|
||||
layout: &wgpu::BindGroupLayout,
|
||||
view: &wgpu::TextureView,
|
||||
params: &wgpu::Buffer,
|
||||
) -> wgpu::BindGroup {
|
||||
ctx.device.create_bind_group(&wgpu::BindGroupDescriptor {
|
||||
label: Some("render-target-bg"),
|
||||
layout,
|
||||
entries: &[
|
||||
wgpu::BindGroupEntry {
|
||||
binding: 0,
|
||||
resource: wgpu::BindingResource::TextureView(view),
|
||||
},
|
||||
wgpu::BindGroupEntry {
|
||||
binding: 1,
|
||||
resource: params.as_entire_binding(),
|
||||
},
|
||||
],
|
||||
})
|
||||
}
|
||||
|
||||
/// Resize, reallocating the texture. No-op when unchanged.
|
||||
pub fn resize(&mut self, width: u32, height: u32) {
|
||||
let (width, height) = (width.max(1), height.max(1));
|
||||
if width == self.width && height == self.height {
|
||||
return;
|
||||
}
|
||||
let (texture, view) = Self::create_texture(&self.ctx, width, height);
|
||||
self.bind_group = Self::create_bind_group(
|
||||
&self.ctx,
|
||||
&self.bind_group_layout,
|
||||
&view,
|
||||
&self.params_buf,
|
||||
);
|
||||
self.texture = texture;
|
||||
self.view = view;
|
||||
self.width = width;
|
||||
self.height = height;
|
||||
#[cfg(any(test, feature = "readback"))]
|
||||
{
|
||||
// Size changed, so the staging buffer no longer fits.
|
||||
*self.readback_buf.borrow_mut() = None;
|
||||
}
|
||||
}
|
||||
|
||||
/// Run the compute pass. Results stay on the GPU.
|
||||
pub fn render(&self, phase: f32) {
|
||||
self.ctx.queue.write_buffer(
|
||||
&self.params_buf,
|
||||
0,
|
||||
bytemuck::bytes_of(&Params {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
phase,
|
||||
_pad: 0.0,
|
||||
}),
|
||||
);
|
||||
|
||||
let mut enc = self
|
||||
.ctx
|
||||
.device
|
||||
.create_command_encoder(&wgpu::CommandEncoderDescriptor {
|
||||
label: Some("render-encoder"),
|
||||
});
|
||||
{
|
||||
let mut pass = enc.begin_compute_pass(&wgpu::ComputePassDescriptor {
|
||||
label: Some("gradient-pass"),
|
||||
timestamp_writes: None,
|
||||
});
|
||||
pass.set_pipeline(&self.pipeline);
|
||||
pass.set_bind_group(0, &self.bind_group, &[]);
|
||||
// 8x8 workgroups, rounded up so edge pixels are covered.
|
||||
pass.dispatch_workgroups(self.width.div_ceil(8), self.height.div_ceil(8), 1);
|
||||
}
|
||||
self.ctx.queue.submit(Some(enc.finish()));
|
||||
}
|
||||
|
||||
pub fn texture(&self) -> &wgpu::Texture {
|
||||
&self.texture
|
||||
}
|
||||
|
||||
pub fn view(&self) -> &wgpu::TextureView {
|
||||
&self.view
|
||||
}
|
||||
|
||||
pub fn size(&self) -> (u32, u32) {
|
||||
(self.width, self.height)
|
||||
}
|
||||
|
||||
/// Read pixels back to the CPU.
|
||||
///
|
||||
/// **Tests only.** Production code must never call this — it is exactly
|
||||
/// the round-trip ARCH §6.1 forbids, and AC-8 asserts it does not happen.
|
||||
#[cfg(any(test, feature = "readback"))]
|
||||
pub async fn read_pixels(&self) -> Result<Vec<u8>, GpuError> {
|
||||
// Buffer rows must be aligned to COPY_BYTES_PER_ROW_ALIGNMENT (256).
|
||||
let unpadded = self.width * 4;
|
||||
let align = wgpu::COPY_BYTES_PER_ROW_ALIGNMENT;
|
||||
let padded = unpadded.div_ceil(align) * align;
|
||||
|
||||
let needed = (padded * self.height) as u64;
|
||||
let mut slot = self.readback_buf.borrow_mut();
|
||||
if slot.as_ref().map(|(_, p)| *p) != Some(padded) {
|
||||
*slot = Some((
|
||||
self.ctx.device.create_buffer(&wgpu::BufferDescriptor {
|
||||
label: Some("readback"),
|
||||
size: needed,
|
||||
usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ,
|
||||
mapped_at_creation: false,
|
||||
}),
|
||||
padded,
|
||||
));
|
||||
}
|
||||
let buf = &slot.as_ref().unwrap().0;
|
||||
|
||||
let mut enc = self
|
||||
.ctx
|
||||
.device
|
||||
.create_command_encoder(&Default::default());
|
||||
enc.copy_texture_to_buffer(
|
||||
wgpu::ImageCopyTexture {
|
||||
texture: &self.texture,
|
||||
mip_level: 0,
|
||||
origin: wgpu::Origin3d::ZERO,
|
||||
aspect: wgpu::TextureAspect::All,
|
||||
},
|
||||
wgpu::ImageCopyBuffer {
|
||||
buffer: buf,
|
||||
layout: wgpu::ImageDataLayout {
|
||||
offset: 0,
|
||||
bytes_per_row: Some(padded),
|
||||
rows_per_image: Some(self.height),
|
||||
},
|
||||
},
|
||||
wgpu::Extent3d {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
depth_or_array_layers: 1,
|
||||
},
|
||||
);
|
||||
self.ctx.queue.submit(Some(enc.finish()));
|
||||
|
||||
let slice = buf.slice(..);
|
||||
let (tx, rx) = std::sync::mpsc::channel();
|
||||
slice.map_async(wgpu::MapMode::Read, move |r| {
|
||||
let _ = tx.send(r);
|
||||
});
|
||||
self.ctx.device.poll(wgpu::Maintain::Wait);
|
||||
rx.recv()
|
||||
.map_err(|e| GpuError::Readback(e.to_string()))?
|
||||
.map_err(|e| GpuError::Readback(e.to_string()))?;
|
||||
|
||||
// Strip row padding.
|
||||
let data = slice.get_mapped_range();
|
||||
let mut out = Vec::with_capacity((unpadded * self.height) as usize);
|
||||
for row in 0..self.height {
|
||||
let start = (row * padded) as usize;
|
||||
out.extend_from_slice(&data[start..start + unpadded as usize]);
|
||||
}
|
||||
drop(data);
|
||||
buf.unmap();
|
||||
Ok(out)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn ctx() -> Option<GpuContext> {
|
||||
// CI runners and headless machines may have no usable adapter. Skip
|
||||
// rather than fail — the device-dependent assertions still run
|
||||
// wherever a GPU exists.
|
||||
match pollster::block_on(GpuContext::new_headless()) {
|
||||
Ok(c) => Some(c),
|
||||
Err(e) => {
|
||||
eprintln!("skipping: no GPU adapter ({e})");
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compute_writes_the_texture() {
|
||||
let Some(ctx) = ctx() else { return };
|
||||
let rt = RenderTarget::new(&ctx, 64, 64).expect("render target");
|
||||
rt.render(0.0);
|
||||
|
||||
let px = pollster::block_on(rt.read_pixels()).expect("readback");
|
||||
assert_eq!(px.len(), 64 * 64 * 4);
|
||||
|
||||
// The shader writes opaque pixels everywhere; an all-zero buffer would
|
||||
// mean the dispatch silently did nothing.
|
||||
assert!(
|
||||
px.chunks_exact(4).all(|p| p[3] == 255),
|
||||
"every pixel should be opaque"
|
||||
);
|
||||
assert!(
|
||||
px.iter().any(|&b| b != 0),
|
||||
"texture should not be uniformly zero"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn phase_changes_output() {
|
||||
let Some(ctx) = ctx() else { return };
|
||||
let rt = RenderTarget::new(&ctx, 32, 32).expect("render target");
|
||||
|
||||
rt.render(0.0);
|
||||
let a = pollster::block_on(rt.read_pixels()).expect("readback");
|
||||
rt.render(std::f32::consts::PI);
|
||||
let b = pollster::block_on(rt.read_pixels()).expect("readback");
|
||||
|
||||
assert_ne!(a, b, "moving the highlight should change the image");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_reallocates() {
|
||||
let Some(ctx) = ctx() else { return };
|
||||
let mut rt = RenderTarget::new(&ctx, 16, 16).expect("render target");
|
||||
assert_eq!(rt.size(), (16, 16));
|
||||
|
||||
rt.resize(48, 24);
|
||||
assert_eq!(rt.size(), (48, 24));
|
||||
|
||||
rt.render(0.0);
|
||||
let px = pollster::block_on(rt.read_pixels()).expect("readback");
|
||||
assert_eq!(px.len(), 48 * 24 * 4);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn zero_size_is_clamped() {
|
||||
let Some(ctx) = ctx() else { return };
|
||||
// A minimised window reports zero; texture creation would panic.
|
||||
let rt = RenderTarget::new(&ctx, 0, 0).expect("render target");
|
||||
assert_eq!(rt.size(), (1, 1));
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user