Merge branch 'zero-copy-display'

# Conflicts:
#	core/dr-gpu/src/adjust.rs
#	ui/dr-ui/src/develop.rs
This commit is contained in:
2026-08-17 10:04:14 +02:00
11 changed files with 564 additions and 139 deletions
+3 -1
View File
@@ -26,7 +26,9 @@ required-features = ["readback"]
[[example]]
name = "develop"
required-features = ["readback"]
# No `readback` needed since S1: this writes a file, so it goes through
# `export_pixels`, which is ungated precisely because an export is not the
# round-trip AC-8 forbids.
[features]
default = []
+5 -2
View File
@@ -5,7 +5,7 @@
//! isolation; this proves they compose into an image a person would accept.
//!
//! ```sh
//! cargo run -p dr-gpu --example develop --features readback -- IMG.CR2 out.ppm
//! cargo run -p dr-gpu --example develop -- IMG.CR2 out.ppm
//! ```
//!
//! PPM because it needs no encoder dependency and every image viewer reads
@@ -145,7 +145,10 @@ fn main() {
adjust.cached_pipelines()
);
let (pixels, pw, ph) = adjust.read_output().expect("readback");
// `export_pixels`, because that is honestly what this is: the frame is
// going into a PPM, not onto a screen. See the note on that method for
// why the two readbacks were never the same thing (AC-8).
let (pixels, pw, ph) = adjust.export_pixels().expect("readback");
// Sanity: an all-black or all-white result means something upstream
// failed silently, and it is far easier to see here than in a viewer.
+143 -41
View File
@@ -32,6 +32,18 @@ use crate::{DemosaicedImage, GpuContext, GpuError};
/// reads them.
const RESERVED_FIELDS: usize = dr_pipeline::RESERVED_UNIFORM_FIELDS;
/// How many non-blocking polls a readback gets before it is called failed.
///
/// A bound rather than a spin forever: if the device is lost the map callback
/// never arrives, and an unbounded loop would hang the interface rather than
/// surfacing the error. Set far above any plausible completion — the copy this
/// waits on is milliseconds — so it is reached only when something is wrong.
///
/// Ungated along with [`AdjustPass::export_pixels`], the one readback that
/// survives S1: an export reads pixels back in a shipping build, and a lost
/// device mid-export must surface as an error rather than a hung interface.
const READBACK_POLL_LIMIT: u32 = 100_000;
/// Runs composed operation chains against demosaiced images.
pub struct AdjustPass {
ctx: GpuContext,
@@ -39,8 +51,27 @@ pub struct AdjustPass {
pipeline_layout: wgpu::PipelineLayout,
/// Compiled pipelines by structure hash (ARCH §5.6).
cache: HashMap<u64, wgpu::ComputePipeline>,
/// Output texture, reallocated only when the size changes.
target: Option<Target>,
/// TRACES: FR-DSP-1 | AC-8
/// Output textures, written alternately, each reallocated only when the
/// size changes.
///
/// **Two, and the second one is not an optimisation — it is what makes the
/// zero-copy path visible.** Since S1 the compositor is handed this
/// texture rather than a copy of its pixels, and Slint decides whether to
/// repaint by comparing the image property against its previous value. Two
/// images wrapping the *same* `wgpu::Texture` compare equal, so a pass
/// that always wrote one texture would recompute every frame on the GPU
/// and never once be asked to show it. Alternating makes each frame a
/// genuinely different value, which is the only thing that makes it a
/// different picture as far as the property system is concerned.
///
/// It also settles the question of whether the compositor is still
/// sampling last frame while this frame's dispatch overwrites it. Both go
/// through one queue, so submission order already answers that — but not
/// having to rely on it is worth a texture.
targets: [Option<Target>; 2],
/// Which of [`Self::targets`] the last render wrote.
current: usize,
}
struct Target {
@@ -106,7 +137,8 @@ impl AdjustPass {
bind_group_layout,
pipeline_layout,
cache: HashMap::new(),
target: None,
targets: [None, None],
current: 0,
}
}
@@ -165,13 +197,20 @@ impl AdjustPass {
.expect("just inserted"))
}
/// Ensure the output texture matches the requested size.
/// Move to the other output texture and make sure it is the right size.
///
/// The rotation is unconditional; the reallocation is not. Steady-state
/// rendering at one viewport size therefore allocates nothing and simply
/// ping-pongs between two textures — see [`Self::targets`] for why there
/// are two. A resize reallocates whichever one comes up next, so the two
/// converge on the new size over two frames rather than in one lump.
fn ensure_target(&mut self, width: u32, height: u32) {
let matches = self
.target
self.current ^= 1;
let slot = &mut self.targets[self.current];
if slot
.as_ref()
.is_some_and(|t| t.width == width && t.height == height);
if matches {
.is_some_and(|t| t.width == width && t.height == height)
{
return;
}
@@ -186,13 +225,24 @@ impl AdjustPass {
sample_count: 1,
dimension: wgpu::TextureDimension::D2,
format: Self::FORMAT,
// STORAGE_BINDING to write from compute, TEXTURE_BINDING so the
// compositor can sample it, COPY_SRC for `export_pixels`.
//
// RENDER_ATTACHMENT is never used by this pass and is required
// anyway: Slint rejects an imported texture that lacks it
// (`TextureImportError::InvalidUsage`), because a compositor
// handed a texture has to assume it may need to draw into it. The
// format is likewise not a free choice — `Rgba8Unorm` and
// `Rgba8UnormSrgb` are the only two the import accepts, which is
// why `FORMAT` is what it is.
usage: wgpu::TextureUsages::STORAGE_BINDING
| wgpu::TextureUsages::TEXTURE_BINDING
| wgpu::TextureUsages::RENDER_ATTACHMENT
| wgpu::TextureUsages::COPY_SRC,
view_formats: &[],
});
let view = texture.create_view(&Default::default());
self.target = Some(Target {
self.targets[self.current] = Some(Target {
texture,
view,
width,
@@ -251,7 +301,7 @@ impl AdjustPass {
.cache
.get(&shader.structure_hash)
.expect("compiled above");
let target = self.target.as_ref().expect("ensured above");
let target = self.targets[self.current].as_ref().expect("ensured above");
let bind_group = self
.ctx
@@ -292,7 +342,10 @@ impl AdjustPass {
}
self.ctx.queue.submit(Some(enc.finish()));
Ok(&self.target.as_ref().expect("ensured above").texture)
Ok(&self.targets[self.current]
.as_ref()
.expect("ensured above")
.texture)
}
/// How many distinct pipelines are compiled. Exposed for tests asserting
@@ -301,48 +354,37 @@ impl AdjustPass {
self.cache.len()
}
/// The texture the last render wrote, if there has been one.
pub fn output(&self) -> Option<&wgpu::Texture> {
self.target.as_ref().map(|t| &t.texture)
self.targets[self.current].as_ref().map(|t| &t.texture)
}
/// Copy the output to the CPU as tightly packed RGBA8.
///
/// **A temporary bridge, not the display path.** ARCH §6.1 forbids this
/// round-trip in production and AC-8 asserts it does not happen; it
/// exists only because Slint's texture-import path is unwired until
/// spike S1. Measured cost at 4K is ~7 ms against a 0.28 ms compute pass
/// — 96% of the frame — so this must go, and the `readback` feature gate
/// keeps it out of a shipping build.
#[cfg(any(test, feature = "readback"))]
pub fn read_output(&self) -> Result<(Vec<u8>, u32, u32), GpuError> {
self.copy_output()
}
/// TRACES: FR-EXP-9
/// TRACES: FR-EXP-9 | AC-8
/// Copy the output to the CPU **for export**.
///
/// The same transfer as [`Self::read_output`] and deliberately not the
/// same method, because the two are opposites in intent and only one of
/// them is a defect.
/// This method had a twin, `read_output`, which performed exactly the same
/// transfer for the display path. Spike S1 deleted the twin and left this
/// one, and the difference between them is worth writing down because it
/// is the whole of AC-8.
///
/// Reading pixels back to display them is what ARCH §6.1 forbids and AC-8
/// asserts against: the compositor could have sampled that texture where
/// it stood, and the round-trip costs 96% of the frame at 4K. Reading them
/// back to *encode a JPEG* is not a shortcut around anything — a file is
/// made of bytes on the CPU, and there is no path to one that does not
/// pass through here.
/// Reading pixels back to *display* them is what ARCH §6.1 forbids: the
/// compositor could have sampled that texture where it stood, and the
/// round-trip cost 96% of the frame at 4K — ~7 ms against a 0.28 ms
/// compute pass. There is now no method that does it, which is a stronger
/// guarantee than a feature gate: the display readback cannot be called
/// back into existence by turning something on.
///
/// So this is ungated where `read_output` is behind a feature: an export
/// must work in a shipping build, and the gate exists to keep the display
/// bridge out of one. Keeping them separate also means the instrumentation
/// AC-8 calls for can count display readbacks without counting exports.
/// Reading them back to *encode a file* is not a shortcut around anything.
/// A JPEG is made of bytes on the CPU and there is no path to one that
/// does not pass through here, so this is ungated and belongs in a
/// shipping build.
pub fn export_pixels(&self) -> Result<(Vec<u8>, u32, u32), GpuError> {
self.copy_output()
}
/// The transfer itself, shared by both readers above.
/// The transfer itself.
fn copy_output(&self) -> Result<(Vec<u8>, u32, u32), GpuError> {
let Some(target) = self.target.as_ref() else {
let Some(target) = self.targets[self.current].as_ref() else {
return Err(GpuError::Readback("nothing rendered yet".into()));
};
let (w, h) = (target.width, target.height);
@@ -1102,6 +1144,66 @@ mod tests {
assert_eq!(read_centre(&ctx, t)[3], 255);
}
/// TRACES: FR-DSP-1 | AC-8
#[test]
fn the_output_is_importable_by_a_compositor() {
// Every condition Slint checks before it will adopt a texture
// (`slint::wgpu_29`: `TextureImportError`). They are asserted here,
// in the crate that owns the descriptor, because failing them does not
// fail a build or a shader — it fails at runtime, on the frame the
// image is handed over, and only where there is a screen to hand it
// to. Nothing else in the test suite would notice.
let Some(ctx) = ctx() else { return };
let mut pass = AdjustPass::new(&ctx);
let img = grey_image(&ctx, 4000);
let shader = EditGraph::default_chain().compose();
let t = pass.render(&img, &shader, 16, 16).expect("render");
assert!(
matches!(
t.format(),
wgpu::TextureFormat::Rgba8Unorm | wgpu::TextureFormat::Rgba8UnormSrgb
),
"import accepts only the two 8-bit RGBA formats, not {:?}",
t.format()
);
assert!(
t.usage().contains(wgpu::TextureUsages::TEXTURE_BINDING),
"the compositor has to sample it"
);
assert!(
t.usage().contains(wgpu::TextureUsages::RENDER_ATTACHMENT),
"Slint requires this even though the adjust pass never uses it"
);
}
/// TRACES: FR-DSP-1 | AC-8
#[test]
fn consecutive_frames_are_different_textures() {
// Not a detail: the compositor is handed this texture rather than a
// copy of its pixels, and Slint repaints only when the image property
// *changes*. Two images over one texture compare equal, so writing the
// same texture every frame would leave a slider moving the pixels on
// the GPU and nothing at all on screen — the frame would be correct
// and invisible, which is the worst kind of wrong.
//
// No display is needed to catch it, because the equality Slint tests
// is the equality asserted here.
let Some(ctx) = ctx() else { return };
let mut pass = AdjustPass::new(&ctx);
let img = grey_image(&ctx, 4000);
let shader = EditGraph::default_chain().compose();
let first = pass.render(&img, &shader, 16, 16).expect("render").clone();
let second = pass.render(&img, &shader, 16, 16).expect("render").clone();
assert_ne!(first, second, "the compositor cannot tell these two apart");
// And back again, so the alternation is a rotation between two rather
// than an allocation per frame — which at 4K would be 33 MB a frame.
let third = pass.render(&img, &shader, 16, 16).expect("render").clone();
assert_eq!(first, third, "a third texture was allocated");
}
#[test]
fn the_output_resizes_with_the_viewport() {
let Some(ctx) = ctx() else { return };
+89 -11
View File
@@ -6,6 +6,12 @@
//!
//! Deliberately free of UI dependencies (ARCH §6.5a). The texture is handed
//! out as a `wgpu::Texture`; who composites it is not this crate's concern.
//!
//! That independence is why [`GpuContext::new_shared`] hands back the raw
//! instance and adapter rather than talking to a compositor itself: the
//! compositor will only sample a texture that came from the device *it* draws
//! with, so somebody has to make one device for both — but it does not have to
//! be this crate, and this crate must not know who it is.
use std::sync::Arc;
@@ -38,19 +44,72 @@ pub struct GpuContext {
adapter_info: wgpu::AdapterInfo,
}
/// TRACES: FR-DSP-1 | AC-8
/// One device, opened so that a compositor can be made to share it.
///
/// The texture the adjust pass writes only reaches the screen without a copy
/// if the compositor is drawing with the *same* `wgpu::Device` — two devices
/// are two address spaces, and a texture from one is not a texture the other
/// can sample. So the device cannot be an implementation detail of either
/// side; it has to be made once and handed to both.
///
/// [`Self::ctx`] is what the compute passes want. The instance and adapter are
/// what a compositor wants in order to adopt the same setup — Slint's
/// `WGPUConfiguration::Manual` asks for all four pieces — and they are handed
/// out raw rather than wrapped, because naming Slint here would put a UI
/// dependency in the one crate that must not have one (ARCH §6.5a).
pub struct SharedGpu {
/// The context every compute pass in this crate runs on.
pub ctx: GpuContext,
/// The instance the compositor will create its window surface from.
pub instance: wgpu::Instance,
/// The adapter [`Self::ctx`]'s device came from.
pub adapter: wgpu::Adapter,
}
impl GpuContext {
/// Create a headless context — no surface, no window.
///
/// Used by tests and by the Slint path, which supplies its own surface.
/// Used by tests, by the examples, and by anything that only needs to
/// compute. A context opened this way cannot be shared with a compositor:
/// see [`Self::new_shared`] for that, and for why the difference matters.
pub async fn new_headless() -> Result<Self, GpuError> {
// GL is allowed alongside Vulkan here and nowhere else: a machine with
// no Vulkan loader should still run the tests, and a headless context
// never has to produce a window surface — which is precisely the thing
// the GL backend cannot do from an instance opened without a display
// handle.
Self::open(wgpu::Backends::VULKAN | wgpu::Backends::GL)
.await
.map(|shared| shared.ctx)
}
/// TRACES: FR-DSP-1 | AC-8
/// Open a device intended to be shared with the compositor.
///
/// Vulkan only, unlike [`Self::new_headless`]. The caller will hand the
/// instance to a compositor that has to create a *window surface* from it,
/// and wgpu's GL backend reaches its display through EGL at instance
/// creation — an instance opened without a display handle, which is the
/// only kind available before a window exists, cannot then produce a GL
/// surface. Vulkan takes the window handle at surface creation instead, so
/// it is the only backend this order of operations permits.
///
/// A machine with no Vulkan therefore gets no shared device, and the
/// caller is expected to carry on without the develop path rather than
/// refuse to start.
pub async fn new_shared() -> Result<SharedGpu, GpuError> {
// Vulkan on both targets (D1), and here it is not merely the
// preference — see above.
Self::open(wgpu::Backends::VULKAN).await
}
async fn open(backends: wgpu::Backends) -> Result<SharedGpu, GpuError> {
// `new_without_display_handle` rather than a struct literal: the
// descriptor carries a boxed display handle and so has no `Default`,
// and a headless context is precisely the case with no display to
// hand it.
// and there is no window yet to take one from in either case.
let mut descriptor = wgpu::InstanceDescriptor::new_without_display_handle();
// Vulkan on both targets (D1). GL is allowed as a fallback so a
// machine without a Vulkan loader still runs the tests.
descriptor.backends = wgpu::Backends::VULKAN | wgpu::Backends::GL;
descriptor.backends = backends;
let instance = wgpu::Instance::new(descriptor);
let adapter = instance
@@ -81,7 +140,14 @@ impl GpuContext {
// in compute shaders are required, and the downlevel tier
// does not guarantee them. This is effectively our GPU
// floor (NFR-COMPAT-1).
required_limits: wgpu::Limits::default(),
//
// `using_resolution` raises only the texture-dimension limits,
// to whatever this adapter actually offers. That matters once
// a compositor shares this device: the default ceiling is
// 8192, and a swapchain image for a large or scaled display
// can exceed it — a limit we chose for our own compute passes
// would otherwise silently cap somebody else's window.
required_limits: wgpu::Limits::default().using_resolution(adapter.limits()),
memory_hints: wgpu::MemoryHints::Performance,
// Nothing behind a feature flag wgpu itself calls unstable —
// the pipeline is ordinary compute and storage textures.
@@ -93,10 +159,14 @@ impl GpuContext {
.await
.map_err(|e| GpuError::DeviceRequest(e.to_string()))?;
Ok(Self {
device: Arc::new(device),
queue: Arc::new(queue),
adapter_info,
Ok(SharedGpu {
ctx: Self {
device: Arc::new(device),
queue: Arc::new(queue),
adapter_info,
},
instance,
adapter,
})
}
@@ -264,8 +334,16 @@ impl RenderTarget {
// STORAGE_BINDING to write from compute; TEXTURE_BINDING so the
// compositor can sample it. COPY_SRC exists only for tests —
// production never reads this back (ARCH §6.1).
//
// RENDER_ATTACHMENT is not something this pass ever uses. It is
// there because Slint refuses to import a texture without it
// (`TextureImportError::InvalidUsage`), the compositor having to
// assume it may need to draw into what it was given. Declaring an
// unused capability costs an allocation flag and buys the whole
// zero-copy path, so it is a cheap price for AC-8.
usage: wgpu::TextureUsages::STORAGE_BINDING
| wgpu::TextureUsages::TEXTURE_BINDING
| wgpu::TextureUsages::RENDER_ATTACHMENT
| wgpu::TextureUsages::COPY_SRC,
view_formats: &[],
});
+7
View File
@@ -329,12 +329,19 @@ impl SegmentPass {
/// The result of one segmentation: a basin label per pixel, on the GPU.
pub struct Segmentation {
/// Only [`Self::read_field`] reads this, so a build without `readback`
/// carries it unread. That is now the ordinary build: dr-ui used to turn
/// the feature on for the whole workspace and stopped when S1 removed the
/// display readback, which is what made the field look dead.
#[cfg_attr(not(any(test, feature = "readback")), allow(dead_code))]
ctx: GpuContext,
width: u32,
height: u32,
/// Per pixel, the linear index of its basin root. Sparse — compacted by
/// [`crate::hierarchy::RegionField::from_roots`].
labels: wgpu::Buffer,
/// As with `ctx` above: read only by [`Self::read_field`].
#[cfg_attr(not(any(test, feature = "readback")), allow(dead_code))]
gradient: wgpu::Buffer,
}