A spot set now composes detail passes of its own, one per round, and they go ahead of every operation's kernel. That placement is the decision worth recording: a sharpening pass reads a neighbourhood, so sharpening a dust mark before removing it smears its edge into pixels the repair's disc does not cover, and what survives is a faint over-sharpened ring around an otherwise perfect patch. It also disagrees with ARCH §5.2, which draws spot removal after clarity — docs/spot-removal.md §5.1 is where that is argued out. Every length reaching the shader is in render pixels, converted here where the framing is in scope. Both the centre and the source go through `Framing::output_at` — the same map the fused pass applies to every pixel — so a rotated photograph rotates the offset with no trigonometry, and the radius is found by mapping a point one radius above the centre and measuring, rather than by multiplying by a ratio this function has no business knowing about. The tests turn and crop the frame and expect the mark to stay gone, which is the property that arrangement buys. compose_full now takes the spot set, because a photograph with a repair and no sharpening still has a detail stage: a fused pass that encoded its own output there would quantise twice and bind to a texture of the wrong format. compose_detail_for takes the source size for the same kind of reason — a RenderScale describes the region on screen, and a spot is stored against the photograph. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
359 lines
12 KiB
Rust
359 lines
12 KiB
Rust
//! Local adjustments end to end, on a real photograph.
|
||
//!
|
||
//! Two edits a photographer actually makes, both driven by the model finding
|
||
//! the subject rather than by anyone drawing a shape:
|
||
//!
|
||
//! - **The subject in colour, everything else monochrome.** One layer, the
|
||
//! subject's mask inverted, saturation at −100.
|
||
//! - **The subject lifted out of its background.** Two layers over the same
|
||
//! mask: the subject brightened, the background pulled down.
|
||
//!
|
||
//! ```sh
|
||
//! cargo run -p dr-gpu --example local --release \
|
||
//! --features segment-readback -- photo.CR2 out
|
||
//! ```
|
||
//!
|
||
//! Writes `<prefix>-original.ppm`, `<prefix>-colour-pop.ppm`,
|
||
//! `<prefix>-subject-lift.ppm` and `<prefix>-mask.ppm`. PPM for the reason
|
||
//! every other example here uses it: no encoder dependency, and every viewer
|
||
//! reads it.
|
||
//!
|
||
//! # What this is really testing
|
||
//!
|
||
//! That the whole chain agrees with itself. The mask is rasterised in *source*
|
||
//! space at proxy resolution and sampled by the composed shader after the
|
||
//! framing map, so a fault anywhere in that handoff — a transposed axis, a
|
||
//! mask pinned to the viewport, a slice read from the wrong layer — shows up
|
||
//! here as an adjustment in the wrong place, and nowhere else.
|
||
|
||
use dr_gpu::{AdjustPass, DemosaicedImage, Demosaicer, GpuContext, MaskPass, SubjectMasks};
|
||
use dr_pipeline::descriptor::ParamId;
|
||
use dr_pipeline::mask::{MaskLayer, MaskSource, MaskStack, Morphology};
|
||
use dr_pipeline::operation::compose_full;
|
||
use dr_pipeline::spot::SpotSet;
|
||
use dr_pipeline::{ops, EditGraph, Framing};
|
||
use dr_segment::{SemanticModel, SemanticOptions, Shaped};
|
||
use dr_types::ColourSpace;
|
||
|
||
/// Longest edge the mask and the model work at.
|
||
const PROXY: u32 = 1600;
|
||
/// Longest edge of the written frames.
|
||
const OUT: u32 = 1400;
|
||
|
||
fn main() {
|
||
env_logger::init();
|
||
|
||
let mut args = std::env::args().skip(1);
|
||
let Some(path) = args.next() else {
|
||
eprintln!("usage: local <photo.CR2|photo.RAF> [out-prefix]");
|
||
std::process::exit(2);
|
||
};
|
||
let prefix = args.next().unwrap_or_else(|| "local".into());
|
||
|
||
let ctx = pollster::block_on(GpuContext::new_headless()).expect("gpu context");
|
||
println!("gpu {}", ctx.adapter_name());
|
||
|
||
// ---- the photograph ---------------------------------------------------
|
||
let bytes = std::fs::read(&path).expect("read file");
|
||
let raw = dr_decode::decode(&bytes).expect("decode");
|
||
println!("source {} × {}", raw.crop.width, raw.crop.height);
|
||
let source = Demosaicer::new(&ctx)
|
||
.expect("demosaicer")
|
||
.run(&raw)
|
||
.expect("demosaic");
|
||
|
||
// ---- what the model sees ----------------------------------------------
|
||
//
|
||
// The *unedited* image, so the detection does not shift when the edit
|
||
// does. Through `export_pixels`, which is ungated: an export is not the
|
||
// display round-trip AC-8 forbids, and neither is this.
|
||
let (sw, sh) = source.size();
|
||
let scale = (PROXY as f32 / sw.max(sh) as f32).min(1.0);
|
||
let (pw, ph) = (
|
||
((sw as f32 * scale) as u32).max(1),
|
||
((sh as f32 * scale) as u32).max(1),
|
||
);
|
||
|
||
let neutral = EditGraph::default_chain();
|
||
let mut proxy_pass = AdjustPass::new(&ctx);
|
||
proxy_pass
|
||
.render(&source, &neutral.compose(), pw, ph)
|
||
.expect("proxy render");
|
||
let (rgba, pw, ph) = proxy_pass.export_pixels().expect("proxy readback");
|
||
println!("proxy {pw} × {ph}");
|
||
|
||
let rgb: Vec<f32> = rgba
|
||
.chunks_exact(4)
|
||
.flat_map(|p| {
|
||
[
|
||
p[0] as f32 / 255.0,
|
||
p[1] as f32 / 255.0,
|
||
p[2] as f32 / 255.0,
|
||
]
|
||
})
|
||
.collect();
|
||
|
||
// ---- find the subject -------------------------------------------------
|
||
let t = std::time::Instant::now();
|
||
let mut model = SemanticModel::embedded().expect("model");
|
||
let instances = model
|
||
.detect(&rgb, pw as usize, ph as usize, &SemanticOptions::default())
|
||
.expect("detect");
|
||
println!(
|
||
"detect {} found in {:.0} ms",
|
||
instances.len(),
|
||
t.elapsed().as_secs_f32() * 1000.0
|
||
);
|
||
for (i, inst) in instances.iter().enumerate() {
|
||
println!(" [{i}] {:<14} {:.2}", inst.class_name, inst.score);
|
||
}
|
||
|
||
let Some((index, subject)) = pick_subject(&instances) else {
|
||
eprintln!("\nNothing recognised in this frame — nothing to adjust locally.");
|
||
eprintln!("The model knows COCO's 80 classes; a landscape with no person,");
|
||
eprintln!("animal or vehicle in it has no subject for it to find.");
|
||
std::process::exit(1);
|
||
};
|
||
println!(
|
||
"subject [{index}] {} at {:.2}",
|
||
subject.class_name, subject.score
|
||
);
|
||
|
||
// Quantised exactly as the develop session does, so this example exercises
|
||
// the shipping path rather than a shortcut around it.
|
||
let alpha: Vec<u8> = subject
|
||
.mask
|
||
.iter()
|
||
.map(|&v| (v.clamp(0.0, 1.0) * 255.0).round() as u8)
|
||
.collect();
|
||
|
||
let (ow, oh) = fit(sw, sh, OUT);
|
||
let mut masks = MaskPass::new(&ctx).expect("mask pass");
|
||
let mut adjust = AdjustPass::new(&ctx);
|
||
|
||
// ---- the original, for comparison -------------------------------------
|
||
adjust
|
||
.render(&source, &neutral.compose(), ow, oh)
|
||
.expect("render");
|
||
write(&format!("{prefix}-original.ppm"), &adjust);
|
||
|
||
// ---- 1. the subject in colour, the rest monochrome --------------------
|
||
//
|
||
// One layer, inverted. Inverting rather than making a second mask for the
|
||
// background is the whole point of having one: there is exactly one
|
||
// boundary, so there is exactly one thing to get right.
|
||
let mut pop = MaskStack::new();
|
||
let mut drain = subject_layer("m1", index, subject);
|
||
drain.invert = true;
|
||
drain.set_param("saturation", ParamId("saturation"), -100.0);
|
||
// A touch of feather, or the colour stops dead on the model's outline and
|
||
// the eye goes straight to the edge instead of to the subject.
|
||
drain.feather = 0.02;
|
||
pop.push(drain);
|
||
|
||
render_stack(
|
||
&ctx,
|
||
&source,
|
||
&mut masks,
|
||
&mut adjust,
|
||
&pop,
|
||
&alpha,
|
||
pw,
|
||
ph,
|
||
ow,
|
||
oh,
|
||
);
|
||
write(&format!("{prefix}-colour-pop.ppm"), &adjust);
|
||
|
||
// ---- 2. lift the subject out of its background ------------------------
|
||
let mut lift = MaskStack::new();
|
||
|
||
let mut brighter = subject_layer("m1", index, subject);
|
||
brighter.set_param("exposure", ParamId("exposure"), 0.45);
|
||
brighter.feather = 0.015;
|
||
lift.push(brighter);
|
||
|
||
let mut darker = subject_layer("m2", index, subject);
|
||
darker.invert = true;
|
||
darker.set_param("exposure", ParamId("exposure"), -0.55);
|
||
darker.set_param("saturation", ParamId("saturation"), -25.0);
|
||
darker.feather = 0.03;
|
||
lift.push(darker);
|
||
|
||
render_stack(
|
||
&ctx,
|
||
&source,
|
||
&mut masks,
|
||
&mut adjust,
|
||
&lift,
|
||
&alpha,
|
||
pw,
|
||
ph,
|
||
ow,
|
||
oh,
|
||
);
|
||
write(&format!("{prefix}-subject-lift.ppm"), &adjust);
|
||
|
||
// ---- 3. the same edit, grown and shrunk -------------------------------
|
||
//
|
||
// The model's outline is approximately right and slightly soft, so the
|
||
// everyday correction is to move it: grow to catch a halo the detector
|
||
// stopped short of, shrink to pull off one it caught. Both are a threshold
|
||
// of the distance field, which is why they cost a uniform.
|
||
for (name, morphology, radius) in [
|
||
("grown", Morphology::Dilate, 0.012),
|
||
("shrunk", Morphology::Erode, 0.012),
|
||
] {
|
||
let mut stack = MaskStack::new();
|
||
let mut layer = subject_layer("m1", index, subject);
|
||
layer.invert = true;
|
||
layer.set_param("saturation", ParamId("saturation"), -100.0);
|
||
layer.feather = 0.004;
|
||
layer.morphology = morphology;
|
||
layer.morph_radius = radius;
|
||
stack.push(layer);
|
||
|
||
render_stack(
|
||
&ctx,
|
||
&source,
|
||
&mut masks,
|
||
&mut adjust,
|
||
&stack,
|
||
&alpha,
|
||
pw,
|
||
ph,
|
||
ow,
|
||
oh,
|
||
);
|
||
write(&format!("{prefix}-{name}.ppm"), &adjust);
|
||
}
|
||
|
||
// ---- the mask itself, to check the outline ----------------------------
|
||
write_mask(&format!("{prefix}-mask.ppm"), &alpha, pw, ph);
|
||
|
||
println!("\nwrote {prefix}-original.ppm");
|
||
println!(" {prefix}-colour-pop.ppm");
|
||
println!(" {prefix}-subject-lift.ppm");
|
||
println!(" {prefix}-grown.ppm, {prefix}-shrunk.ppm");
|
||
println!(" {prefix}-mask.ppm");
|
||
}
|
||
|
||
/// A layer masked to one detected object.
|
||
fn subject_layer(id: &str, index: usize, subject: &dr_segment::Instance) -> MaskLayer {
|
||
let mut layer = MaskLayer::new(
|
||
id,
|
||
MaskSource::Subject {
|
||
// One segmentation in this process, so any signature agrees with
|
||
// itself; the session computes a real one.
|
||
signature: 0,
|
||
index: index as u32,
|
||
class: subject.class_name.to_string(),
|
||
score: subject.score,
|
||
},
|
||
);
|
||
layer.name = subject.class_name.to_string();
|
||
layer
|
||
}
|
||
|
||
/// The most promising thing to adjust.
|
||
///
|
||
/// Prefers a person, then falls back to the strongest detection of anything.
|
||
/// Not because people are special to the pipeline, but because they are what a
|
||
/// local adjustment is usually *for*, and an example that picks the parked car
|
||
/// behind the subject demonstrates the mechanism while missing the point.
|
||
fn pick_subject(instances: &[dr_segment::Instance]) -> Option<(usize, &dr_segment::Instance)> {
|
||
instances
|
||
.iter()
|
||
.enumerate()
|
||
.find(|(_, i)| &*i.class_name == "person")
|
||
.or_else(|| instances.iter().enumerate().next())
|
||
}
|
||
|
||
#[allow(clippy::too_many_arguments)]
|
||
fn render_stack(
|
||
ctx: &GpuContext,
|
||
source: &DemosaicedImage,
|
||
masks: &mut MaskPass,
|
||
adjust: &mut AdjustPass,
|
||
stack: &MaskStack,
|
||
coverage: &[u8],
|
||
pw: u32,
|
||
ph: u32,
|
||
ow: u32,
|
||
oh: u32,
|
||
) {
|
||
// One signed distance field per active layer, in that order — the order
|
||
// the rasteriser indexes them by. Built here rather than once up front
|
||
// because a compound morphology rebuilds the field, so it belongs to the
|
||
// layer that shaped it rather than to the object.
|
||
let fields: Vec<Vec<f32>> = stack
|
||
.active()
|
||
.map(|layer| {
|
||
Shaped::build(
|
||
coverage,
|
||
pw as usize,
|
||
ph as usize,
|
||
128,
|
||
match layer.morphology {
|
||
Morphology::None => dr_segment::Morphology::None,
|
||
Morphology::Dilate => dr_segment::Morphology::Dilate,
|
||
Morphology::Erode => dr_segment::Morphology::Erode,
|
||
Morphology::Close => dr_segment::Morphology::Close,
|
||
Morphology::Open => dr_segment::Morphology::Open,
|
||
},
|
||
layer.morph_radius * pw.min(ph) as f32,
|
||
)
|
||
.distance
|
||
})
|
||
.collect();
|
||
let refs: Vec<&[f32]> = fields.iter().map(|f| f.as_slice()).collect();
|
||
let subjects = SubjectMasks::upload(ctx, &refs, pw, ph).expect("upload fields");
|
||
|
||
// Rasterised at *proxy* size in source space, then sampled by the shader
|
||
// after the framing map — which is what makes one mask correct at every
|
||
// output size, zoom and crop.
|
||
let array = masks
|
||
.render(stack, None, Some(&subjects), pw, ph)
|
||
.expect("rasterise masks");
|
||
|
||
let shader = compose_full(
|
||
&ops::chain(),
|
||
&Framing::new(),
|
||
ColourSpace::Srgb,
|
||
stack,
|
||
&SpotSet::new(),
|
||
);
|
||
adjust
|
||
.render_masked(source, &shader, ow, oh, Some(array))
|
||
.expect("render");
|
||
}
|
||
|
||
fn fit(w: u32, h: u32, longest: u32) -> (u32, u32) {
|
||
let s = (longest as f32 / w.max(h) as f32).min(1.0);
|
||
(
|
||
((w as f32 * s) as u32).max(1),
|
||
((h as f32 * s) as u32).max(1),
|
||
)
|
||
}
|
||
|
||
fn write(path: &str, adjust: &AdjustPass) {
|
||
let (rgba, w, h) = adjust.export_pixels().expect("readback");
|
||
let rgb: Vec<u8> = rgba
|
||
.chunks_exact(4)
|
||
.flat_map(|p| [p[0], p[1], p[2]])
|
||
.collect();
|
||
write_ppm(path, &rgb, w, h);
|
||
}
|
||
|
||
fn write_mask(path: &str, alpha: &[u8], w: u32, h: u32) {
|
||
let rgb: Vec<u8> = alpha.iter().flat_map(|&a| [a, a, a]).collect();
|
||
write_ppm(path, &rgb, w, h);
|
||
}
|
||
|
||
fn write_ppm(path: &str, rgb: &[u8], w: u32, h: u32) {
|
||
use std::io::Write as _;
|
||
let mut f = std::io::BufWriter::new(std::fs::File::create(path).expect("create"));
|
||
write!(f, "P6\n{w} {h}\n255\n").expect("header");
|
||
f.write_all(rgb).expect("body");
|
||
}
|