390 lines
16 KiB
WebGPU Shading Language
390 lines
16 KiB
WebGPU Shading Language
// Rasterise one local-adjustment mask into a layer of the mask array.
|
||
//
|
||
// ARCH §5.4: every mask becomes pixels here and never in CPU memory. One draw
|
||
// per layer, each targeting its own array slice, run only when a mask's
|
||
// *shape* changes — moving a slider on a masked layer re-runs the adjust
|
||
// shader and not this one.
|
||
//
|
||
// # Why this is a render pass and not a compute one
|
||
//
|
||
// The natural shape for this is a compute shader writing a storage texture,
|
||
// and the format is what rules that out: **R8Unorm is not a core storage
|
||
// format**, so a compute path has to widen the mask to R32Float or RGBA8 —
|
||
// four bytes per pixel per layer. At eight layers over a 24 MP export that is
|
||
// 768 MB of masks, against 192 MB at one byte. A colour attachment takes
|
||
// R8Unorm happily, so the mask stays one byte and the pass becomes a
|
||
// full-screen triangle.
|
||
//
|
||
// The array slice is chosen by the *view* the caller attaches, so there is no
|
||
// slot uniform here — one less thing that can disagree with the shader.
|
||
|
||
struct MaskParams {
|
||
// Output size, which is the render size rather than the segmentation's.
|
||
width: u32,
|
||
height: u32,
|
||
// Label field size. Different from the above: the watershed runs at a
|
||
// proxy resolution, and the mask is drawn at whatever the display or the
|
||
// export asked for.
|
||
label_width: u32,
|
||
label_height: u32,
|
||
|
||
// 0 = regions, 1 = linear, 2 = radial, 3 = subject, 4 = brush.
|
||
//
|
||
// A brush does not read this — it has its own entry points, because it is
|
||
// the one mask that is not a function of the whole frame — but it is set
|
||
// anyway so a captured frame says which kind of mask a pass was drawing.
|
||
mode: u32,
|
||
// How many regions the label field holds, so an out-of-range label is
|
||
// caught rather than read past the end of `selected`.
|
||
region_count: u32,
|
||
// Softening applied to a region mask, in output pixels.
|
||
feather: f32,
|
||
// 0 hard, 1 linear, 2 smooth, 3 gaussian, 4 exponential. Kept in step with
|
||
// `falloff_code` on the Rust side.
|
||
falloff: u32,
|
||
|
||
// Geometry. Meaning depends on `mode`. The centre is in normalised 0..1
|
||
// coordinates; every distance below it is in the isotropic frame units
|
||
// `frame_delta` establishes.
|
||
centre: vec2<f32>,
|
||
// Linear: (cos, sin) of the ramp direction. Radial: semi-axes.
|
||
axis: vec2<f32>,
|
||
// Linear: ramp width. Radial: edge falloff as a fraction of the radius.
|
||
softness: f32,
|
||
// Radial only: rotation of the ellipse.
|
||
angle: f32,
|
||
_pad1: vec2<f32>,
|
||
}
|
||
|
||
@group(0) @binding(0) var<uniform> p: MaskParams;
|
||
// Compacted region id per pixel of the label field. Compacted rather than the
|
||
// watershed's raw basin roots: the roots are sparse indices into pixel space,
|
||
// so indexing a per-region array by one would need a table as large as the
|
||
// image. The compaction happens once, when the segmentation is built.
|
||
@group(0) @binding(1) var<storage, read> labels: array<u32>;
|
||
// One entry per region: non-zero if the region is in this mask. Small — a few
|
||
// thousand bytes — which is what makes changing a selection cheap.
|
||
@group(0) @binding(2) var<storage, read> selected: array<u32>;
|
||
// The **signed distance** from one subject's boundary, in proxy pixels:
|
||
// positive inside, negative outside. A 1x1 placeholder when the layer is not a
|
||
// subject — the binding is fixed, and a second pipeline differing only in what
|
||
// it ignores would be worse than a wasted texel.
|
||
//
|
||
// A distance field rather than a finished alpha is what makes growing,
|
||
// shrinking and feathering free: each is arithmetic on this, so a slider moves
|
||
// a uniform instead of rebuilding a mask.
|
||
@group(0) @binding(3) var subject: texture_2d<f32>;
|
||
|
||
// A full-screen triangle rather than a quad: three vertices instead of six,
|
||
// no shared edge for the rasteriser to crack along, and no vertex buffer.
|
||
@vertex
|
||
fn vs(@builtin(vertex_index) i: u32) -> @builtin(position) vec4<f32> {
|
||
let x = f32(i32(i) / 2) * 4.0 - 1.0;
|
||
let y = f32(i32(i) & 1) * 4.0 - 1.0;
|
||
return vec4<f32>(x, y, 0.0, 1.0);
|
||
}
|
||
|
||
fn region_at(px: vec2<i32>) -> u32 {
|
||
// Nearest-neighbour from output space into the label field. Deliberately
|
||
// not bilinear: region ids are *names*, and the average of region 4 and
|
||
// region 9 is not region 6.
|
||
let fx = (f32(px.x) + 0.5) / f32(p.width);
|
||
let fy = (f32(px.y) + 0.5) / f32(p.height);
|
||
let lx = clamp(i32(fx * f32(p.label_width)), 0, i32(p.label_width) - 1);
|
||
let ly = clamp(i32(fy * f32(p.label_height)), 0, i32(p.label_height) - 1);
|
||
return labels[u32(ly) * p.label_width + u32(lx)];
|
||
}
|
||
|
||
fn in_selection(px: vec2<i32>) -> f32 {
|
||
let r = region_at(px);
|
||
if (r >= p.region_count) {
|
||
return 0.0;
|
||
}
|
||
return select(0.0, 1.0, selected[r] != 0u);
|
||
}
|
||
|
||
fn region_mask(px: vec2<i32>) -> f32 {
|
||
let hard = in_selection(px);
|
||
if (p.feather <= 0.0) {
|
||
return hard;
|
||
}
|
||
|
||
// Box-average the binary selection over the feather radius. Cheap, and it
|
||
// is the whole reason a region mask does not look cut out with scissors:
|
||
// the watershed boundary is pixel-exact, which is correct and also harsher
|
||
// than any edit wants at a subject's edge.
|
||
let r = i32(ceil(p.feather));
|
||
var total = 0.0;
|
||
var n = 0.0;
|
||
for (var dy = -r; dy <= r; dy = dy + 1) {
|
||
for (var dx = -r; dx <= r; dx = dx + 1) {
|
||
let q = clamp(
|
||
px + vec2<i32>(dx, dy),
|
||
vec2<i32>(0, 0),
|
||
vec2<i32>(i32(p.width) - 1, i32(p.height) - 1),
|
||
);
|
||
total = total + in_selection(q);
|
||
n = n + 1.0;
|
||
}
|
||
}
|
||
return total / n;
|
||
}
|
||
|
||
// Offset from a gradient's centre, in the frame's own **isotropic** units:
|
||
// y spans 0..1 and x spans 0..aspect, so a step of the same length means the
|
||
// same distance whichever way it points.
|
||
//
|
||
// Without this the geometry lives in raw 0..1, where one axis is compressed
|
||
// against the other by the aspect ratio — so a 45° ramp is not at 45° on
|
||
// anything but a square frame, and a radial with equal radii draws an ellipse.
|
||
// Both faults are invisible in the stored numbers and obvious the moment a
|
||
// handle is dragged on a photograph, which is what this exists for.
|
||
fn frame_delta(uv: vec2<f32>) -> vec2<f32> {
|
||
let aspect = vec2<f32>(f32(p.width) / f32(max(p.height, 1u)), 1.0);
|
||
return (uv - p.centre) * aspect;
|
||
}
|
||
|
||
fn linear_mask(uv: vec2<f32>) -> f32 {
|
||
// Signed distance along the ramp direction, from the centre.
|
||
let d = dot(frame_delta(uv), p.axis);
|
||
if (p.softness <= 0.0) {
|
||
return select(0.0, 1.0, d >= 0.0);
|
||
}
|
||
return smoothstep(-p.softness * 0.5, p.softness * 0.5, d);
|
||
}
|
||
|
||
fn radial_mask(uv: vec2<f32>) -> f32 {
|
||
let ca = cos(-p.angle);
|
||
let sa = sin(-p.angle);
|
||
let d = frame_delta(uv);
|
||
// Into the ellipse's own frame, then normalised by its semi-axes so the
|
||
// problem becomes a unit circle.
|
||
let local = vec2<f32>(d.x * ca - d.y * sa, d.x * sa + d.y * ca);
|
||
let r = length(local / max(p.axis, vec2<f32>(1e-6)));
|
||
|
||
let edge = clamp(p.softness, 0.0, 1.0);
|
||
if (edge <= 0.0) {
|
||
return select(0.0, 1.0, r <= 1.0);
|
||
}
|
||
return 1.0 - smoothstep(1.0 - edge, 1.0, r);
|
||
}
|
||
|
||
// Coverage for one object, from its distance field.
|
||
//
|
||
// Bilinear on the *distance*, which is the reason this is a distance field at
|
||
// all: distance varies smoothly across the boundary where coverage does not,
|
||
// so interpolating it gives a clean sub-pixel edge even though the model's
|
||
// own mask was quarter-resolution.
|
||
fn subject_mask(uv: vec2<f32>) -> f32 {
|
||
let dims = vec2<f32>(textureDimensions(subject));
|
||
let last = vec2<i32>(dims) - vec2<i32>(1);
|
||
|
||
let t = uv * dims - vec2<f32>(0.5);
|
||
let base = vec2<i32>(floor(t));
|
||
let f = fract(t);
|
||
|
||
let p0 = clamp(base, vec2<i32>(0), last);
|
||
let p1 = clamp(base + vec2<i32>(1), vec2<i32>(0), last);
|
||
|
||
let a = textureLoad(subject, vec2<i32>(p0.x, p0.y), 0).r;
|
||
let b = textureLoad(subject, vec2<i32>(p1.x, p0.y), 0).r;
|
||
let c = textureLoad(subject, vec2<i32>(p0.x, p1.y), 0).r;
|
||
let d = textureLoad(subject, vec2<i32>(p1.x, p1.y), 0).r;
|
||
|
||
// `angle` carries the morphology offset in pixels: positive grows the
|
||
// mask, negative shrinks it. Adding it before the falloff is what makes
|
||
// dilation move the boundary rather than merely brighten the edge.
|
||
let dist = mix(mix(a, b, f.x), mix(c, d, f.x), f.y) + p.angle;
|
||
|
||
// `softness` is the feather half-width, also in pixels.
|
||
if (p.softness <= 0.0) {
|
||
return select(0.0, 1.0, dist >= 0.0);
|
||
}
|
||
let t_norm = dist / p.softness;
|
||
|
||
// Every curve is 0.5 at the boundary, so changing the falloff changes how
|
||
// the transition looks and never where it sits.
|
||
switch p.falloff {
|
||
case 0u: { return select(0.0, 1.0, dist >= 0.0); }
|
||
case 1u: { return clamp(t_norm * 0.5 + 0.5, 0.0, 1.0); }
|
||
case 3u: { return 1.0 / (1.0 + exp(-3.0 * t_norm)); }
|
||
case 4u: {
|
||
if (t_norm >= 0.0) {
|
||
return 1.0 - 0.5 * exp(-3.0 * t_norm);
|
||
}
|
||
return 0.5 * exp(3.0 * t_norm);
|
||
}
|
||
default: {
|
||
let x = clamp(t_norm * 0.5 + 0.5, 0.0, 1.0);
|
||
return x * x * (3.0 - 2.0 * x);
|
||
}
|
||
}
|
||
}
|
||
|
||
@fragment
|
||
fn fs(@builtin(position) pos: vec4<f32>) -> @location(0) vec4<f32> {
|
||
let px = vec2<i32>(i32(pos.x), i32(pos.y));
|
||
|
||
// Normalised, so a gradient's geometry survives a crop or an export at
|
||
// another size — the mask is defined on the frame, not on a pixel count.
|
||
let uv = vec2<f32>(pos.x / f32(p.width), pos.y / f32(p.height));
|
||
|
||
var m = 0.0;
|
||
switch p.mode {
|
||
case 0u: { m = region_mask(px); }
|
||
case 1u: { m = linear_mask(uv); }
|
||
case 2u: { m = radial_mask(uv); }
|
||
case 3u: { m = subject_mask(uv); }
|
||
default: { m = 0.0; }
|
||
}
|
||
|
||
return vec4<f32>(clamp(m, 0.0, 1.0), 0.0, 0.0, 1.0);
|
||
}
|
||
|
||
// ---------------------------------------------------------------------------
|
||
// Brush strokes (ARCH §5.4)
|
||
// ---------------------------------------------------------------------------
|
||
//
|
||
// The mask the architecture was written for. What arrives is a list of
|
||
// positions, a radius, a hardness and a flow; what leaves is pixels. Nothing
|
||
// between the two ever exists in CPU memory, which is the whole difference from
|
||
// darktable, where the same strokes are rasterised on the CPU and the lag makes
|
||
// painting unusable.
|
||
//
|
||
// # Why the strokes are not drawn by the full-screen triangle above
|
||
//
|
||
// Cost. A swept disc is the minimum distance to any segment of its polyline, so
|
||
// evaluating one stroke costs a distance per segment *per pixel*. Over the
|
||
// whole frame that is `pixels × segments`, and a stroke that wandered across
|
||
// the photograph has both terms large at once.
|
||
//
|
||
// So each stroke is drawn over its own bounding box instead, expanded by the
|
||
// radius. The rasteriser then never invokes the fragment shader for a pixel the
|
||
// stroke cannot reach, and the cost becomes `area(box) × segments` — for the
|
||
// ordinary case, a dab or a swipe, a small fraction of the frame. The model
|
||
// splits a long gesture into strokes of bounded length for the same reason:
|
||
// both terms of that product grow with how far one stroke travelled.
|
||
//
|
||
// # Why the strokes composite with fixed-function blending
|
||
//
|
||
// Add is `dst + a(1 - dst)` and erase is `dst(1 - a)`, which are exactly a
|
||
// source-over and a one-minus-source blend. Expressing them as blend state
|
||
// rather than as arithmetic in the shader is what allows one draw per stroke:
|
||
// the accumulating mask is the attachment, and no pass ever has to read the
|
||
// slice it is writing.
|
||
|
||
struct StrokeHeader {
|
||
// Bounding box in normalised coordinates, already grown by the radius and
|
||
// a texel — the vertex shader trusts it and draws nothing outside it.
|
||
lo: vec2<f32>,
|
||
hi: vec2<f32>,
|
||
// Radius in units of the frame's shorter edge, so a dab is round on a frame
|
||
// that is not square.
|
||
radius: f32,
|
||
// Fraction of the radius that is fully covered.
|
||
hardness: f32,
|
||
// Coverage deposited where the stroke is solid.
|
||
flow: f32,
|
||
// Window into `stroke_points`.
|
||
first: u32,
|
||
count: u32,
|
||
_pad: u32,
|
||
}
|
||
|
||
@group(0) @binding(4) var<storage, read> strokes: array<StrokeHeader>;
|
||
@group(0) @binding(5) var<storage, read> stroke_points: array<vec2<f32>>;
|
||
|
||
struct BrushVertex {
|
||
@builtin(position) pos: vec4<f32>,
|
||
// Flat: a stroke index interpolated across its own quad would name a
|
||
// different stroke in the middle of it.
|
||
@location(0) @interpolate(flat) stroke: u32,
|
||
}
|
||
|
||
// Six vertices per stroke, non-instanced.
|
||
//
|
||
// Deliberately not one instance per stroke: `@builtin(instance_index)` with a
|
||
// non-zero first instance needs base-instance support, which the GL backend
|
||
// this has to run on under Android cannot promise. Dividing the vertex index
|
||
// costs one integer operation and works everywhere.
|
||
@vertex
|
||
fn vs_brush(@builtin(vertex_index) v: u32) -> BrushVertex {
|
||
var quad = array<vec2<f32>, 6>(
|
||
vec2<f32>(0.0, 0.0), vec2<f32>(1.0, 0.0), vec2<f32>(0.0, 1.0),
|
||
vec2<f32>(0.0, 1.0), vec2<f32>(1.0, 0.0), vec2<f32>(1.0, 1.0),
|
||
);
|
||
|
||
let i = v / 6u;
|
||
let s = strokes[i];
|
||
let uv = mix(s.lo, s.hi, quad[v % 6u]);
|
||
|
||
var out: BrushVertex;
|
||
// y is flipped because normalised mask coordinates run downwards, the way
|
||
// the fragment shader above reads them, and clip space runs upwards. A
|
||
// stroke drawn without this lands mirrored about the horizon, which is
|
||
// plausible enough on a symmetric test image to survive a careless check.
|
||
out.pos = vec4<f32>(uv.x * 2.0 - 1.0, 1.0 - uv.y * 2.0, 0.0, 1.0);
|
||
out.stroke = i;
|
||
return out;
|
||
}
|
||
|
||
// Into units of the frame's shorter edge.
|
||
//
|
||
// Without this the brush would be a circle in normalised coordinates, which on
|
||
// a 3:2 frame is an ellipse half again as wide as it is tall. A brush whose dab
|
||
// is not round is not a brush.
|
||
fn to_square(uv: vec2<f32>) -> vec2<f32> {
|
||
let dims = vec2<f32>(f32(p.width), f32(p.height));
|
||
return uv * dims / min(dims.x, dims.y);
|
||
}
|
||
|
||
fn segment_distance(q: vec2<f32>, a: vec2<f32>, b: vec2<f32>) -> f32 {
|
||
let ab = b - a;
|
||
let len2 = dot(ab, ab);
|
||
// A finger that stopped and went back leaves a zero-length segment, and
|
||
// dividing by its length is a NaN — which propagates through the min()
|
||
// below and takes the whole stroke with it.
|
||
if (len2 <= 1e-12) {
|
||
return length(q - a);
|
||
}
|
||
let t = clamp(dot(q - a, ab) / len2, 0.0, 1.0);
|
||
return length(q - (a + ab * t));
|
||
}
|
||
|
||
@fragment
|
||
fn fs_brush(in: BrushVertex) -> @location(0) vec4<f32> {
|
||
let s = strokes[in.stroke];
|
||
let q = to_square(vec2<f32>(in.pos.x / f32(p.width), in.pos.y / f32(p.height)));
|
||
|
||
// The *minimum* over the segments, which is the maximum of their coverage.
|
||
// Accumulating the segments instead would make a stroke that crosses itself
|
||
// — every circle, every scribble — build up a bright patch where it did,
|
||
// and a soft brush would go blotchy along any curve tight enough for
|
||
// consecutive dabs to overlap, which is all of them.
|
||
var d = 1e30;
|
||
if (s.count == 1u) {
|
||
// A tap. One point is a legitimate stroke, and it paints one dab.
|
||
d = length(q - to_square(stroke_points[s.first]));
|
||
} else {
|
||
for (var k = 0u; k + 1u < s.count; k = k + 1u) {
|
||
d = min(
|
||
d,
|
||
segment_distance(
|
||
q,
|
||
to_square(stroke_points[s.first + k]),
|
||
to_square(stroke_points[s.first + k + 1u]),
|
||
),
|
||
);
|
||
}
|
||
}
|
||
|
||
// Even at full hardness the edge keeps a one-pixel ramp. A true step would
|
||
// alias into a staircase, and the mask is sampled bilinearly at whatever
|
||
// zoom the user is inspecting it at — which is where an edge is judged.
|
||
let texel = 1.0 / f32(min(p.width, p.height));
|
||
let inner = min(s.radius * clamp(s.hardness, 0.0, 1.0), max(s.radius - texel, 0.0));
|
||
let coverage = 1.0 - smoothstep(inner, s.radius, d);
|
||
|
||
return vec4<f32>(clamp(coverage * s.flow, 0.0, 1.0), 0.0, 0.0, 1.0);
|
||
}
|