dr-gpu: the merge pass — warp, accumulate, resolve, chunk by chunk
merge.wgsl warps one camera-space tile into one output chunk — output pixel to direction (the projection maths of dr_pano::projection, verbatim), direction to the frame's camera, camera to source pixel, bilinear by hand from four textureLoads because rgba32float is not filterable — and adds it into a storage-buffer accumulator weighted by its distance from the frame's edge. A resolve pass divides by the weights and packs sixteen-bit samples at the sensor's scale with a coverage bit. MergePass::merge drives it: bands of rows, chunks across a band, and for each chunk only the frames whose footprint meets it, each rendered as the source rectangle the chunk needs and nothing more. The working set is one chunk, one tile and one band (FR-MRG-11); the frame textures are the caller's to cache. Feathered, not seamed; gain a scalar per frame — the blend quality is panorama.md §10's step 5, after the path writes a file.
This commit is contained in:
@@ -0,0 +1,152 @@
|
||||
// TRACES: FR-MRG-10 | FR-MRG-11
|
||||
// The merge: one source tile warped into one output chunk, accumulated.
|
||||
//
|
||||
// Two entry points. `warp` runs once per (chunk, frame): for every chunk
|
||||
// pixel it asks which direction that pixel looks along, turns the
|
||||
// direction into the frame's camera, projects it to a source pixel, and
|
||||
// if that pixel is inside the tile that was rendered for this chunk,
|
||||
// samples it and adds it — weighted by its distance from the frame's edge
|
||||
// — into the accumulator. `resolve` runs once per chunk after every frame
|
||||
// has been added: divides the sums by the weights and packs the result as
|
||||
// sixteen-bit samples at the sensor's scale (FR-MRG-3).
|
||||
//
|
||||
// The accumulator is a buffer and not a storage texture, because WebGPU
|
||||
// allows a read-write storage texture only in the 32-bit single-channel
|
||||
// formats, and this wants four channels. The tile is sampled by hand from
|
||||
// four `textureLoad`s rather than through a sampler, because `rgba32float`
|
||||
// is not filterable without an optional feature, and the tile is
|
||||
// `rgba32float` on purpose (panorama.md §5.1).
|
||||
//
|
||||
// The projection maths is `dr_pano::projection` verbatim; the two must
|
||||
// agree, and a golden test compares them.
|
||||
|
||||
struct Params {
|
||||
// Where the chunk's pixel (0, 0) sits in centred output coordinates,
|
||||
// and the chunk's size.
|
||||
chunk_origin: vec2<f32>,
|
||||
chunk_size: vec2<u32>,
|
||||
// 0 perspective, 1 cylindrical, 2 spherical; and the projection's
|
||||
// scale (the cylinder's radius, the sphere's, the plane's distance) in
|
||||
// output pixels.
|
||||
projection: u32,
|
||||
proj_scale: f32,
|
||||
// The frame's focal length in source pixels, and the gain the frame's
|
||||
// exposure is corrected by.
|
||||
focal: f32,
|
||||
gain: f32,
|
||||
// World → this frame's camera: the transpose of its rotation, one row
|
||||
// per vec4 (padded).
|
||||
r0: vec4<f32>,
|
||||
r1: vec4<f32>,
|
||||
r2: vec4<f32>,
|
||||
// The full frame's size in source pixels (for the edge weight), the
|
||||
// tile's origin within the frame, and the tile's size.
|
||||
frame_size: vec2<f32>,
|
||||
tile_origin: vec2<f32>,
|
||||
tile_size: vec2<u32>,
|
||||
// Pixels over which the weight ramps from the edge to full.
|
||||
feather: f32,
|
||||
_pad: f32,
|
||||
};
|
||||
|
||||
@group(0) @binding(0) var<uniform> p: Params;
|
||||
@group(0) @binding(1) var tile: texture_2d<f32>;
|
||||
// rgb·w summed, then w: four floats per chunk pixel.
|
||||
@group(0) @binding(2) var<storage, read_write> acc: array<vec4<f32>>;
|
||||
|
||||
fn to_direction(u: f32, v: f32) -> vec3<f32> {
|
||||
let s = p.proj_scale;
|
||||
if (p.projection == 0u) {
|
||||
return normalize(vec3<f32>(u, v, s));
|
||||
}
|
||||
if (p.projection == 1u) {
|
||||
let theta = u / s;
|
||||
return normalize(vec3<f32>(sin(theta), v / s, cos(theta)));
|
||||
}
|
||||
let theta = u / s;
|
||||
let phi = v / s;
|
||||
return vec3<f32>(sin(theta) * cos(phi), sin(phi), cos(theta) * cos(phi));
|
||||
}
|
||||
|
||||
fn load(x: i32, y: i32) -> vec3<f32> {
|
||||
return textureLoad(tile, vec2<i32>(x, y), 0).rgb;
|
||||
}
|
||||
|
||||
@compute @workgroup_size(8, 8, 1)
|
||||
fn warp(@builtin(global_invocation_id) gid: vec3<u32>) {
|
||||
if (gid.x >= p.chunk_size.x || gid.y >= p.chunk_size.y) {
|
||||
return;
|
||||
}
|
||||
let u = p.chunk_origin.x + f32(gid.x) + 0.5;
|
||||
let v = p.chunk_origin.y + f32(gid.y) + 0.5;
|
||||
let d = to_direction(u, v);
|
||||
let c = vec3<f32>(dot(p.r0.xyz, d), dot(p.r1.xyz, d), dot(p.r2.xyz, d));
|
||||
if (c.z <= 1e-6) {
|
||||
return;
|
||||
}
|
||||
// Source pixel, in the full frame, with the principal point at its
|
||||
// centre. `- 0.5` puts pixel centres on integer coordinates for the
|
||||
// bilinear fetch below.
|
||||
let sx = p.focal * c.x / c.z + p.frame_size.x * 0.5 - 0.5;
|
||||
let sy = p.focal * c.y / c.z + p.frame_size.y * 0.5 - 0.5;
|
||||
// Weight: distance to the nearest frame edge, in pixels, over the
|
||||
// feather. Zero outside the frame.
|
||||
let edge = min(min(sx, p.frame_size.x - 1.0 - sx), min(sy, p.frame_size.y - 1.0 - sy));
|
||||
if (edge <= 0.0) {
|
||||
return;
|
||||
}
|
||||
let w = clamp(edge / max(p.feather, 1.0), 0.0, 1.0);
|
||||
// Into the tile.
|
||||
let tx = sx - p.tile_origin.x;
|
||||
let ty = sy - p.tile_origin.y;
|
||||
let tw = f32(p.tile_size.x);
|
||||
let th = f32(p.tile_size.y);
|
||||
if (tx < 0.0 || ty < 0.0 || tx > tw - 1.0 || ty > th - 1.0) {
|
||||
return;
|
||||
}
|
||||
let x0 = i32(floor(tx));
|
||||
let y0 = i32(floor(ty));
|
||||
let x1 = min(x0 + 1, i32(p.tile_size.x) - 1);
|
||||
let y1 = min(y0 + 1, i32(p.tile_size.y) - 1);
|
||||
let fx = tx - f32(x0);
|
||||
let fy = ty - f32(y0);
|
||||
let top = mix(load(x0, y0), load(x1, y0), fx);
|
||||
let bot = mix(load(x0, y1), load(x1, y1), fx);
|
||||
let rgb = mix(top, bot, fy) * p.gain;
|
||||
let i = gid.y * p.chunk_size.x + gid.x;
|
||||
acc[i] = acc[i] + vec4<f32>(rgb * w, w);
|
||||
}
|
||||
|
||||
// Resolve: the accumulated chunk to sixteen-bit samples.
|
||||
struct ResolveParams {
|
||||
chunk_size: vec2<u32>,
|
||||
// Multiplies a normalised value (1.0 = the sensor's white) back to the
|
||||
// sensor's scale: the source's white minus its black (FR-MRG-3).
|
||||
scale: f32,
|
||||
_pad: f32,
|
||||
};
|
||||
|
||||
@group(0) @binding(0) var<uniform> rp: ResolveParams;
|
||||
@group(0) @binding(1) var<storage, read> racc: array<vec4<f32>>;
|
||||
// Two u32 per pixel: (r | g << 16), (b | coverage << 16). Coverage is
|
||||
// 65535 where any frame reached the pixel and 0 where none did, so the
|
||||
// CPU can tell an empty pixel from a black one.
|
||||
@group(0) @binding(2) var<storage, read_write> out: array<vec2<u32>>;
|
||||
|
||||
@compute @workgroup_size(8, 8, 1)
|
||||
fn resolve(@builtin(global_invocation_id) gid: vec3<u32>) {
|
||||
if (gid.x >= rp.chunk_size.x || gid.y >= rp.chunk_size.y) {
|
||||
return;
|
||||
}
|
||||
let i = gid.y * rp.chunk_size.x + gid.x;
|
||||
let a = racc[i];
|
||||
if (a.w <= 0.0) {
|
||||
out[i] = vec2<u32>(0u, 0u);
|
||||
return;
|
||||
}
|
||||
let rgb = clamp(a.rgb / a.w * rp.scale, vec3<f32>(0.0), vec3<f32>(65535.0));
|
||||
let r = u32(round(rgb.r));
|
||||
let g = u32(round(rgb.g));
|
||||
let b = u32(round(rgb.b));
|
||||
out[i] = vec2<u32>(r | (g << 16u), b | (65535u << 16u));
|
||||
}
|
||||
Reference in New Issue
Block a user