The merge weighted every overlap pixel by its distance from each frame's edge, a 200 px linear cross-fade. Anything the frames disagreed on — parallax in the near foreground, grass in the wind, a walker — came out twice at half strength: a soft double edge at 1:1. dr_pano::seam picks, per output texel at proxy resolution, which frame a pixel comes from. Where a new frame overlaps the composite the cost is the gain-corrected difference plus local detail plus nearness to either frame's edge, taken as the worst over a small window, and the cut is a dynamic-programming path across the overlap. merge.wgsl weights each frame by its tent-filtered share of that map, a 64 px blend that follows the seam, with the edge feather kept as the fallback. The page's preview uses the same map, and examples/merge.rs takes --feather-only for comparison.
264 lines
10 KiB
WebGPU Shading Language
264 lines
10 KiB
WebGPU Shading Language
// TRACES: FR-MRG-10 | FR-MRG-11
|
|
// The merge: one source tile warped into one output chunk, accumulated.
|
|
//
|
|
// Two entry points. `warp` runs once per (chunk, frame): for every chunk
|
|
// pixel it asks which direction that pixel looks along, turns the
|
|
// direction into the frame's camera, projects it to a source pixel, and
|
|
// if that pixel is inside the tile that was rendered for this chunk,
|
|
// samples it and adds it — weighted by the frame's share of the seam map
|
|
// there, or by its distance from the frame's edge where there is no map —
|
|
// into the accumulator. `resolve` runs once per chunk after every frame
|
|
// has been added: divides the sums by the weights and packs the result as
|
|
// sixteen-bit samples at the sensor's scale (FR-MRG-3).
|
|
//
|
|
// The accumulator is a buffer and not a storage texture, because WebGPU
|
|
// allows a read-write storage texture only in the 32-bit single-channel
|
|
// formats, and this wants four channels. The tile is sampled by hand from
|
|
// four `textureLoad`s rather than through a sampler, because `rgba32float`
|
|
// is not filterable without an optional feature, and the tile is
|
|
// `rgba32float` on purpose (panorama.md §5.1).
|
|
//
|
|
// The projection maths is `dr_pano::projection` verbatim; the two must
|
|
// agree, and a golden test compares them.
|
|
|
|
struct Params {
|
|
// Where the chunk's pixel (0, 0) sits in centred output coordinates,
|
|
// and the chunk's size.
|
|
chunk_origin: vec2<f32>,
|
|
chunk_size: vec2<u32>,
|
|
// 0 perspective, 1 cylindrical, 2 spherical; and the projection's
|
|
// scale (the cylinder's radius, the sphere's, the plane's distance) in
|
|
// output pixels.
|
|
projection: u32,
|
|
proj_scale: f32,
|
|
// The frame's focal length in source pixels, and the gain the frame's
|
|
// exposure is corrected by.
|
|
focal: f32,
|
|
gain: f32,
|
|
// World → this frame's camera: the transpose of its rotation, one row
|
|
// per vec4 (padded).
|
|
r0: vec4<f32>,
|
|
r1: vec4<f32>,
|
|
r2: vec4<f32>,
|
|
// The full frame's size in source pixels (for the edge weight), the
|
|
// tile's origin within the frame, and the tile's size.
|
|
frame_size: vec2<f32>,
|
|
tile_origin: vec2<f32>,
|
|
tile_size: vec2<u32>,
|
|
// Pixels over which the weight ramps from the edge to full.
|
|
feather: f32,
|
|
// Where a sample starts to count as blown (`CLIP_ONSET`), and the
|
|
// white balance the composite will be developed with.
|
|
clip_onset: f32,
|
|
balance: vec4<f32>,
|
|
// The seam map (`dr_pano::seam`): where its texel (0, 0)'s corner sits
|
|
// in this output's centred coordinates, its size, output pixels per
|
|
// texel, the blend's radius in texels, which frame this dispatch is,
|
|
// and whether there is a map at all.
|
|
seam_origin: vec2<f32>,
|
|
seam_size: vec2<u32>,
|
|
seam_px: f32,
|
|
seam_radius: f32,
|
|
frame_index: u32,
|
|
seam_on: u32,
|
|
};
|
|
|
|
@group(0) @binding(0) var<uniform> p: Params;
|
|
@group(0) @binding(1) var tile: texture_2d<f32>;
|
|
// rgb·w summed, then w: four floats per chunk pixel.
|
|
@group(0) @binding(2) var<storage, read_write> acc: array<vec4<f32>>;
|
|
// One frame index per texel, 255 for none.
|
|
@group(0) @binding(3) var seams: texture_2d<u32>;
|
|
|
|
const NO_FRAME: u32 = 255u;
|
|
|
|
fn label(i: i32, j: i32) -> u32 {
|
|
if (i < 0 || j < 0 || i >= i32(p.seam_size.x) || j >= i32(p.seam_size.y)) {
|
|
return NO_FRAME;
|
|
}
|
|
return textureLoad(seams, vec2<i32>(i, j), 0).r;
|
|
}
|
|
|
|
// This frame's share of the seam map about output point (u, v): the
|
|
// tent-weighted fraction of the texels within the radius that it owns, and
|
|
// the weight of the texels owned by anyone (zero where the map has nothing
|
|
// to say). `SeamMap::share` verbatim.
|
|
fn seam_share(u: f32, v: f32) -> vec2<f32> {
|
|
let x = (u - p.seam_origin.x) / p.seam_px - 0.5;
|
|
let y = (v - p.seam_origin.y) / p.seam_px - 0.5;
|
|
let r = max(p.seam_radius, 1.0);
|
|
let x0 = i32(ceil(x - r));
|
|
let x1 = i32(floor(x + r));
|
|
let y0 = i32(ceil(y - r));
|
|
let y1 = i32(floor(y + r));
|
|
// Most pixels are nowhere near a seam: if the window's corners, edge
|
|
// midpoints and centre agree, so does the window. A seam crossing it
|
|
// has to cross its border, between two of those.
|
|
let xm = i32(round(x));
|
|
let ym = i32(round(y));
|
|
let c = label(xm, ym);
|
|
if (label(x0, y0) == c && label(x1, y0) == c && label(x0, y1) == c && label(x1, y1) == c
|
|
&& label(xm, y0) == c && label(xm, y1) == c && label(x0, ym) == c && label(x1, ym) == c) {
|
|
if (c == NO_FRAME) {
|
|
return vec2<f32>(0.0, 0.0);
|
|
}
|
|
return vec2<f32>(select(0.0, 1.0, c == p.frame_index), 1.0);
|
|
}
|
|
var mine = 0.0;
|
|
var owned = 0.0;
|
|
for (var j = y0; j <= y1; j = j + 1) {
|
|
let wy = 1.0 - abs(y - f32(j)) / r;
|
|
if (wy <= 0.0) {
|
|
continue;
|
|
}
|
|
for (var i = x0; i <= x1; i = i + 1) {
|
|
let wx = 1.0 - abs(x - f32(i)) / r;
|
|
let l = label(i, j);
|
|
if (wx <= 0.0 || l == NO_FRAME) {
|
|
continue;
|
|
}
|
|
owned = owned + wx * wy;
|
|
if (l == p.frame_index) {
|
|
mine = mine + wx * wy;
|
|
}
|
|
}
|
|
}
|
|
if (owned <= 0.0) {
|
|
return vec2<f32>(0.0, 0.0);
|
|
}
|
|
return vec2<f32>(mine / owned, 1.0);
|
|
}
|
|
|
|
fn to_direction(u: f32, v: f32) -> vec3<f32> {
|
|
let s = p.proj_scale;
|
|
if (p.projection == 0u) {
|
|
return normalize(vec3<f32>(u, v, s));
|
|
}
|
|
if (p.projection == 1u) {
|
|
let theta = u / s;
|
|
return normalize(vec3<f32>(sin(theta), v / s, cos(theta)));
|
|
}
|
|
let theta = u / s;
|
|
let phi = v / s;
|
|
return vec3<f32>(sin(theta) * cos(phi), sin(phi), cos(theta) * cos(phi));
|
|
}
|
|
|
|
fn load(x: i32, y: i32) -> vec4<f32> {
|
|
return textureLoad(tile, vec2<i32>(x, y), 0);
|
|
}
|
|
|
|
@compute @workgroup_size(8, 8, 1)
|
|
fn warp(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
if (gid.x >= p.chunk_size.x || gid.y >= p.chunk_size.y) {
|
|
return;
|
|
}
|
|
let u = p.chunk_origin.x + f32(gid.x) + 0.5;
|
|
let v = p.chunk_origin.y + f32(gid.y) + 0.5;
|
|
let d = to_direction(u, v);
|
|
let c = vec3<f32>(dot(p.r0.xyz, d), dot(p.r1.xyz, d), dot(p.r2.xyz, d));
|
|
if (c.z <= 1e-6) {
|
|
return;
|
|
}
|
|
// Source pixel, in the full frame, with the principal point at its
|
|
// centre. `- 0.5` puts pixel centres on integer coordinates for the
|
|
// bilinear fetch below.
|
|
let sx = p.focal * c.x / c.z + p.frame_size.x * 0.5 - 0.5;
|
|
let sy = p.focal * c.y / c.z + p.frame_size.y * 0.5 - 0.5;
|
|
// Weight: distance to the nearest frame edge, in pixels, over the
|
|
// feather. Zero outside the frame.
|
|
let edge = min(min(sx, p.frame_size.x - 1.0 - sx), min(sy, p.frame_size.y - 1.0 - sy));
|
|
if (edge <= 0.0) {
|
|
return;
|
|
}
|
|
var w = clamp(edge / max(p.feather, 1.0), 0.0, 1.0);
|
|
// With seams, the share of the map scales it. The small floor keeps
|
|
// the feather underneath as the answer wherever no frame that reaches
|
|
// this pixel owns it — the map is coarser than the output, so at the
|
|
// frames' outer edges it can name a frame that falls just short.
|
|
if (p.seam_on != 0u) {
|
|
let s = seam_share(u, v);
|
|
if (s.y > 0.0) {
|
|
w = w * (s.x + 1e-4);
|
|
}
|
|
}
|
|
// Into the tile.
|
|
let tx = sx - p.tile_origin.x;
|
|
let ty = sy - p.tile_origin.y;
|
|
let tw = f32(p.tile_size.x);
|
|
let th = f32(p.tile_size.y);
|
|
if (tx < 0.0 || ty < 0.0 || tx > tw - 1.0 || ty > th - 1.0) {
|
|
return;
|
|
}
|
|
let x0 = i32(floor(tx));
|
|
let y0 = i32(floor(ty));
|
|
let x1 = min(x0 + 1, i32(p.tile_size.x) - 1);
|
|
let y1 = min(y0 + 1, i32(p.tile_size.y) - 1);
|
|
let fx = tx - f32(x0);
|
|
let fy = ty - f32(y0);
|
|
// The four texels, with their alpha: the tap writes alpha 0 where the
|
|
// lens correction found no source pixel, and a sample that touches one
|
|
// of those is a partial pixel — down-weighted by exactly how much of
|
|
// it is missing, and dropped when all of it is.
|
|
let s00 = load(x0, y0);
|
|
let s10 = load(x1, y0);
|
|
let s01 = load(x0, y1);
|
|
let s11 = load(x1, y1);
|
|
let top = mix(s00, s10, fx);
|
|
let bot = mix(s01, s11, fx);
|
|
let s = mix(top, bot, fy);
|
|
if (s.a <= 0.001) {
|
|
return;
|
|
}
|
|
// Colour is the alpha-weighted mean of the texels that exist.
|
|
let cam = s.rgb / s.a;
|
|
// **A blown sample is written as grey, before the gain.** A clipped
|
|
// photosite arrives as (1, 1, 1), which is not a colour: balanced, it
|
|
// is magenta, and the develop's highlight desaturation only rescues it
|
|
// while it is still at the white level. A gain below one moved it off
|
|
// that level, and a feather mixed it into a neighbour's real sky, so
|
|
// the composite's blown clouds came out pink. Written instead as the
|
|
// camera value the balance maps to grey — the develop pipeline's own
|
|
// neutral, the brightest balanced channel — it survives both.
|
|
let clipped = smoothstep(p.clip_onset, 1.0, max(cam.r, max(cam.g, cam.b)));
|
|
let balanced = cam * p.balance.rgb;
|
|
let grey = vec3<f32>(max(balanced.r, max(balanced.g, balanced.b))) / p.balance.rgb;
|
|
let rgb = mix(cam, grey, clipped) * p.gain;
|
|
let wa = w * s.a;
|
|
let i = gid.y * p.chunk_size.x + gid.x;
|
|
acc[i] = acc[i] + vec4<f32>(rgb * wa, wa);
|
|
}
|
|
|
|
// Resolve: the accumulated chunk to sixteen-bit samples.
|
|
struct ResolveParams {
|
|
chunk_size: vec2<u32>,
|
|
// Multiplies a normalised value (1.0 = the sensor's white) back to the
|
|
// sensor's scale: the source's white minus its black (FR-MRG-3).
|
|
scale: f32,
|
|
_pad: f32,
|
|
};
|
|
|
|
@group(0) @binding(0) var<uniform> rp: ResolveParams;
|
|
@group(0) @binding(1) var<storage, read> racc: array<vec4<f32>>;
|
|
// Two u32 per pixel: (r | g << 16), (b | coverage << 16). Coverage is
|
|
// 65535 where any frame reached the pixel and 0 where none did, so the
|
|
// CPU can tell an empty pixel from a black one.
|
|
@group(0) @binding(2) var<storage, read_write> out: array<vec2<u32>>;
|
|
|
|
@compute @workgroup_size(8, 8, 1)
|
|
fn resolve(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
if (gid.x >= rp.chunk_size.x || gid.y >= rp.chunk_size.y) {
|
|
return;
|
|
}
|
|
let i = gid.y * rp.chunk_size.x + gid.x;
|
|
let a = racc[i];
|
|
if (a.w <= 0.0) {
|
|
out[i] = vec2<u32>(0u, 0u);
|
|
return;
|
|
}
|
|
let rgb = clamp(a.rgb / a.w * rp.scale, vec3<f32>(0.0), vec3<f32>(65535.0));
|
|
let r = u32(round(rgb.r));
|
|
let g = u32(round(rgb.g));
|
|
let b = u32(round(rgb.b));
|
|
out[i] = vec2<u32>(r | (g << 16u), b | (65535u << 16u));
|
|
}
|