//! What a frame actually costs — the measurement FR-DSP-2 is waiting on. //! //! `docs/dev/display-and-extension.md` §2 argues that tiled computation predates //! the fused-shader design and may not need to exist: the composer folds every //! active operation into **one dispatch over a viewport-sized target**, so the //! problem tiles were invented to solve may already be solved. That argument //! is only worth as much as the numbers behind it, and the decision rule was //! fixed in advance — if the 99th percentile sits inside 16 ms, FR-DSP-2 is //! rewritten as a scheduling concern for export rather than implemented on the //! interactive path. //! //! This is the instrument that decides it. Three measurements: //! //! - **M1** — the fused pass at proxy resolution, at three viewport sizes and //! three chain lengths. //! - **M2** — the same with the view zoomed to 1:1 on a 60 MP source, which is //! the case FR-DSP-5 names. //! - **M3** — the neighbourhood stage on its own. It is the only part of the //! chain that is not one read and one write, so it is the plausible //! budget-breaker and deserves to be measured apart from the fused pass //! rather than hidden inside its total. //! //! ```sh //! cargo run --release -p dr-gpu --example frame_budget //! ``` //! //! **Release, always.** A debug build measures rustc's shadow, not the GPU's: //! the per-frame CPU half is dominated by shader-source assembly, which is //! string formatting and is several times slower unoptimised. //! //! The committed numbers live in `docs/dev/frame-budget.md`. Rerun this and diff //! that file; a regression should be a diff rather than somebody's memory. //! //! # Why the 99th percentile and not the mean //! //! A slider drag is judged by its worst frame. A chain averaging 4 ms with one //! frame in fifty at 30 ms reads as a stutter, and the mean says nothing about //! it. Percentiles are nearest-rank over the samples with the warm-up already //! discarded — see [`Percentiles`]. //! //! # What is being timed //! //! Each frame is `compose` (CPU: assemble the WGSL and its uniforms) followed //! by `render_detailed` and a `poll` that waits for the device to go idle. //! Waiting serialises the GPU work into the frame it belongs to, which is //! pessimistic — a real presentation pipeline overlaps a frame's tail with the //! next frame's head — and pessimistic is the right direction for a budget. //! //! The compose half is reported separately because it is *not* GPU work and //! would otherwise be invisible: it happens on the UI thread on every slider //! event, and if it were the expensive half then tiling could not help at all. use std::time::Instant; use dr_film::bake::{bake, Recipe}; use dr_gpu::{AdjustPass, DemosaicedImage, GpuContext}; use dr_pipeline::descriptor::ParamKind; use dr_pipeline::ops::{ blacks_whites, contrast, exposure, highlights_shadows, local_contrast, noise_reduction, vibrance, FilmTables, }; use dr_pipeline::{Attribute, CropRect, EditGraph, OpId, ParamId}; /// The synthetic source: 9504 × 6336 is 60.2 MP, which is a Sony A7R V or a /// Fujifilm GFX 100 with a little taken off. §2's M2 names 60 MP, so this is /// that number rather than a round one. const SOURCE: (u32, u32) = (9504, 6336); /// Frames measured per row, after the warm-up. Enough that the 99th percentile /// means something (nearest-rank picks the second-worst of 100) without the /// whole matrix taking minutes. const FRAMES: usize = 100; /// Frames discarded before measurement begins. /// /// The first frame at a new size allocates a render target, the first frame /// with a new chain compiles a pipeline, and the first frame of a detail stage /// allocates its ping-pong pair. None of those recur during a drag, so /// including them would measure the wrong thing — but they are worth knowing /// about, so [`Run::first_frame_ms`] reports the very first one separately. const WARMUP: usize = 12; /// The viewport sizes from §2's M1: a laptop panel, a 16:10 desktop, and 4K. const SIZES: [(u32, u32); 3] = [(1920, 1200), (2560, 1600), (3840, 2160)]; /// The frame budget FR-DSP-3 is asserting against, in milliseconds. 60 Hz. const BUDGET_MS: f64 = 16.0; fn main() { env_logger::init(); let ctx = match pollster::block_on(GpuContext::new_headless()) { Ok(c) => c, Err(e) => { eprintln!("no GPU adapter: {e}"); std::process::exit(1); } }; println!("adapter {} ({:?})", ctx.adapter_name(), ctx.backend()); let limit = DemosaicedImage::max_dimension(&ctx); if SOURCE.0 > limit { eprintln!( "this adapter caps textures at {limit} px; {} × {} will not fit", SOURCE.0, SOURCE.1 ); std::process::exit(1); } let t = Instant::now(); let source = synthetic_source(&ctx); println!( "source {} × {} ({:.1} MP, {:.0} MB as rgba16f), built in {:.1} s", SOURCE.0, SOURCE.1, (SOURCE.0 as f64 * SOURCE.1 as f64) / 1e6, (SOURCE.0 as f64 * SOURCE.1 as f64 * 8.0) / 1e6, t.elapsed().as_secs_f64() ); println!("budget {BUDGET_MS:.0} ms (FR-DSP-3)\n"); let mut adjust = AdjustPass::new(&ctx); m1_and_m2(&ctx, &mut adjust, &source); m3(&ctx, &mut adjust, &source); } // --------------------------------------------------------------------------- // M1 and M2 — the fused pass, fit and at 1:1 // --------------------------------------------------------------------------- /// The chain lengths §2 asks for. /// /// "Every operation" means every operation in [`EditGraph::default_chain`], /// which is what `ops/*.yaml` generates. The lens corrections are not in it — /// they are constructed from a matched profile rather than being part of the /// default chain — and neither are masks or spot repairs, which are stacks of /// their own. So this is an upper bound on the *declared* chain and a lower /// bound on the worst edit a photograph can carry; §7 of the spec asks for /// honesty about exactly this sort of thing. /// /// [`Chain::AllPoint`] is not in §2's list and is the row that decides the /// question. §2 asks whether *one dispatch over a viewport-sized target* is /// fast enough; "every operation" mixes that dispatch together with the /// neighbourhood stage, which is not one dispatch and is measured separately in /// M3. Splitting the two apart is what lets M1 answer the question it was /// written to answer instead of a different, harder one. #[derive(Clone, Copy, PartialEq)] enum Chain { One, Five, /// Every operation that contributes a fragment to the fused shader — the /// full point-operation chain, and nothing with a kernel. AllPoint, Everything, } impl Chain { fn label(self) -> &'static str { match self { Chain::One => "one", Chain::Five => "five", Chain::AllPoint => "point", Chain::Everything => "all", } } } fn m1_and_m2(ctx: &GpuContext, adjust: &mut AdjustPass, source: &DemosaicedImage) { for (title, zoomed) in [ ( "M1 — fused frame at proxy resolution (fit; the develop view)", false, ), ( "M2 — the same view zoomed to 1:1 on the 60 MP source (FR-DSP-5)", true, ), ] { println!("{title}"); header(); for size in SIZES { for chain in [Chain::One, Chain::Five, Chain::AllPoint, Chain::Everything] { let mut graph = build_chain(chain); if zoomed { graph.framing_mut().set_view(one_to_one(size)); } if matches!(chain, Chain::AllPoint | Chain::Everything) { adjust.set_film(Some(&film_tables())); } // The drag: exposure, moved by a hundredth of a stop per frame. // A real slider does exactly this, and it is what stops the // fused dispatch being skipped by `render_detailed`'s colour // reuse — which would turn M1 into a measurement of the detail // stage by accident. let run = measure(ctx, adjust, source, &mut graph, size, |g, i| { g.set_param(exposure::ID, exposure::EXPOSURE, 0.30 + i as f32 * 0.01); }); adjust.set_film(None); row(size, chain.label(), &run); } } println!(); } println!(" `shader` is `EditGraph::compose` alone; `cpu` adds the detail chain and the"); println!(" invalidation hash, which is everything the develop view does per frame before"); println!(" it dispatches. `gpu` is submit plus wait-for-idle. `TOTAL` ranks cpu + gpu"); println!(" summed within each frame — the column the {BUDGET_MS:.0} ms budget is judged on."); println!(); println!(" `point` is every operation that contributes a fragment to the fused shader;"); println!(" `all` is that plus the four neighbourhood operations, which is why the two"); println!(" differ by roughly what M3 charges for the detail stage on its own."); println!(); } // --------------------------------------------------------------------------- // M3 — the detail stage alone // --------------------------------------------------------------------------- /// Neighbourhood operations, timed with the fused dispatch deliberately reused. /// /// `render_detailed` skips the fused pass when the colour key, the shader, its /// uniforms and the size are all unchanged — that is FR-DEV-3d, and it is also /// the lever that isolates M3: move only a detail operation's amount and the /// timing covers the convolutions and nothing else. The `colour` column proves /// the isolation held rather than asserting it: it counts fused dispatches over /// the measured frames, and a zero there is what makes the number mean "detail /// alone". fn m3(ctx: &GpuContext, adjust: &mut AdjustPass, source: &DemosaicedImage) { println!("M3 — the neighbourhood stage alone (fused dispatch reused; FR-DEV-3d)"); println!( " {:>11} {:>8} {:>5} {:>4} {:>6} {:>6} {:>8} {:>7} {:>7} {:>7}", "size", "stage", "view", "pass", "radius", "colour", "cpu p99", "p50", "p99", "max" ); for size in SIZES { for zoomed in [false, true] { for (label, build) in DETAIL_CHAINS { let mut graph = build(); if zoomed { graph.framing_mut().set_view(one_to_one(size)); } let composed = graph.compose_detail(source.size(), size); let (passes, radius) = (composed.len(), composed.radius()); // Only the detail parameter moves, so the colour half of the // chain is bit-identical frame to frame and gets reused. let run = measure(ctx, adjust, source, &mut graph, size, |g, i| { let drift = (i % 20) as f32; g.set_param( local_contrast::CLARITY, local_contrast::AMOUNT, 60.0 + drift, ); g.set_param( local_contrast::TEXTURE, local_contrast::AMOUNT, 60.0 + drift, ); g.set_param( noise_reduction::ID, noise_reduction::LUMINANCE, 60.0 + drift, ); g.set_param(noise_reduction::ID, noise_reduction::CHROMA, 60.0 + drift); }); println!( " {:>5}x{:<5} {:>8} {:>5} {:>4} {:>6} {:>6} {:>6.2}ms {:>5.2}ms {:>5.2}ms {:>5.2}ms", size.0, size.1, label, if zoomed { "1:1" } else { "fit" }, passes, radius, run.colour_dispatches, run.cpu.p99, run.frame.p50, run.frame.p99, run.frame.max ); } } } println!(); println!(" `pass` is dispatches in the detail chain and `radius` the widest halo any of"); println!(" them reads, in render pixels. `colour` is fused dispatches over the {FRAMES}"); println!(" measured frames, and must be 0 for the row to mean what it says."); println!(); println!(" Clarity's kernel is a fraction of the *frame*, so it grows with the viewport"); println!(" and not with the zoom. Noise reduction's is a fraction of the *sensor*, so it"); println!(" grows with the zoom and not with the viewport. That is why both are here."); } /// The detail chains M3 walks: the widest kernel alone, then all four /// neighbourhood operations at once. /// /// Clarity is first because §2 names it — "a separable blur at a large radius /// is the plausible budget-breaker" — and its σ is 1.2% of the shorter edge, /// which at 4K is a 52-pixel radius and the widest kernel anywhere in the /// pipeline. /// A named detail chain: a label for the table, and the graph it builds. type DetailChain = (&'static str, fn() -> EditGraph); const DETAIL_CHAINS: [DetailChain; 2] = [ ("clarity", || { let mut g = EditGraph::default_chain(); g.set_param(local_contrast::CLARITY, local_contrast::AMOUNT, 60.0); g }), ("all four", || { let mut g = EditGraph::default_chain(); g.set_param(local_contrast::CLARITY, local_contrast::AMOUNT, 60.0); g.set_param(local_contrast::TEXTURE, local_contrast::AMOUNT, 60.0); g.set_param(noise_reduction::ID, noise_reduction::LUMINANCE, 60.0); g.set_param(noise_reduction::ID, noise_reduction::CHROMA, 60.0); g.set_param( dr_pipeline::ops::capture_sharpen::ID, dr_pipeline::ops::capture_sharpen::AMOUNT, 60.0, ); g }), ]; // --------------------------------------------------------------------------- // The measurement itself // --------------------------------------------------------------------------- /// Nearest-rank percentiles over a set of frame times, in milliseconds. /// /// Nearest-rank rather than an interpolating definition because the samples /// *are* the population — there is no distribution being estimated, only a /// hundred frames that either happened inside the budget or did not. At /// `FRAMES = 100` the 99th percentile is the second-worst frame, which is the /// honest reading of "one stutter in a hundred is one too many" without /// letting a single scheduler hiccup on an unrelated process decide the /// verdict. struct Percentiles { p50: f64, p99: f64, max: f64, } impl Percentiles { fn of(mut samples: Vec) -> Self { samples.sort_by(f64::total_cmp); let rank = |p: f64| { let n = samples.len(); let i = ((p * n as f64).ceil() as usize).clamp(1, n) - 1; samples[i] }; Self { p50: rank(0.50), p99: rank(0.99), max: samples[samples.len() - 1], } } } struct Run { /// CPU: assembling the fused shader alone. shader: Percentiles, /// CPU: everything `DevelopSession::render` does before it dispatches — /// the fused shader, the detail chain, and the invalidation hash the /// colour key comes from. Split out from [`Self::shader`] because if the /// expensive half of a frame turns out to be string formatting on the UI /// thread, no amount of tiling helps and the conclusion is a different /// one. cpu: Percentiles, /// Submit plus wait for the device to go idle. frame: Percentiles, /// CPU and GPU summed **per frame**, then ranked. /// /// Not the sum of the two percentiles above, which would be a number no /// frame ever took: the CPU's worst frame and the GPU's worst frame are /// not generally the same frame, and adding them invents a stutter that /// did not happen. This is the column the budget is judged on. total: Percentiles, /// The very first frame of all, warm-up included — pipeline compilation /// and target allocation. Reported because it is real (it is what a /// photograph opening costs) and because it must not be inside the /// percentiles. first_frame_ms: f64, /// Fused dispatches over the measured frames. `FRAMES` when the colour /// chain is moving, 0 when only a detail parameter is. colour_dispatches: usize, } /// Render `FRAMES` frames, moving a parameter between each, and time them. /// /// `drag` receives the frame index and is expected to move whatever this row /// is measuring the drag of. It is called for the warm-up frames too, so that /// nothing measured is the first of its kind. fn measure( ctx: &GpuContext, adjust: &mut AdjustPass, source: &DemosaicedImage, graph: &mut EditGraph, size: (u32, u32), drag: impl Fn(&mut EditGraph, usize), ) -> Run { // A constant, because every row here renders the same photograph. In the // app this is the `VersionId` mixed with the colour invalidation — see // `render_detailed`'s note on why the caller owns it. const COLOUR_KEY: u64 = 0x0dd_ba11; let src = source.size(); let mut first_frame_ms = f64::NAN; for i in 0..WARMUP { drag(graph, i); let t = Instant::now(); frame(ctx, adjust, source, graph, src, size, COLOUR_KEY); if i == 0 { first_frame_ms = t.elapsed().as_secs_f64() * 1e3; } } let dispatches_before = adjust.colour_dispatches(); let mut shader_ms = Vec::with_capacity(FRAMES); let mut cpu_ms = Vec::with_capacity(FRAMES); let mut gpu_ms = Vec::with_capacity(FRAMES); let mut total_ms = Vec::with_capacity(FRAMES); for i in 0..FRAMES { drag(graph, WARMUP + i); // The same three calls `DevelopSession::render` makes, in the same // order, so that this is the develop view's frame and not an // idealisation of it. let t0 = Instant::now(); let shader = graph.compose(); let after_shader = t0.elapsed(); let detail = graph.compose_detail(src, size); let colour_key = graph.invalidation().through(dr_pipeline::Affects::Colour); let cpu = t0.elapsed(); let t1 = Instant::now(); adjust .render_detailed( source, &shader, size.0, size.1, None, &detail, colour_key ^ COLOUR_KEY, ) .expect("render"); ctx.device .poll(wgpu::PollType::wait_indefinitely()) .expect("poll"); let gpu = t1.elapsed(); shader_ms.push(after_shader.as_secs_f64() * 1e3); cpu_ms.push(cpu.as_secs_f64() * 1e3); gpu_ms.push(gpu.as_secs_f64() * 1e3); total_ms.push((cpu + gpu).as_secs_f64() * 1e3); } Run { shader: Percentiles::of(shader_ms), cpu: Percentiles::of(cpu_ms), frame: Percentiles::of(gpu_ms), total: Percentiles::of(total_ms), first_frame_ms, colour_dispatches: adjust.colour_dispatches() - dispatches_before, } } /// One frame, composition included, with nothing timed. The warm-up path. fn frame( ctx: &GpuContext, adjust: &mut AdjustPass, source: &DemosaicedImage, graph: &EditGraph, src: (u32, u32), size: (u32, u32), colour_key: u64, ) { let shader = graph.compose(); let detail = graph.compose_detail(src, size); let key = graph.invalidation().through(dr_pipeline::Affects::Colour) ^ colour_key; adjust .render_detailed(source, &shader, size.0, size.1, None, &detail, key) .expect("render"); ctx.device .poll(wgpu::PollType::wait_indefinitely()) .expect("poll"); } fn header() { println!( " {:>11} {:>5} {:>8} {:>8} {:>8} {:>8} {:>8} {:>7}", "size", "chain", "shader", "cpu p99", "gpu p50", "gpu p99", "TOTAL", "first" ); } fn row(size: (u32, u32), chain: &str, run: &Run) { println!( " {:>5}x{:<5} {:>5} {:>6.2}ms {:>6.2}ms {:>6.2}ms {:>6.2}ms {:>6.2}ms {:>5.0}ms{}", size.0, size.1, chain, run.shader.p99, run.cpu.p99, run.frame.p50, run.frame.p99, run.total.p99, run.first_frame_ms, if run.total.p99 > BUDGET_MS { " OVER" } else { "" } ); } // --------------------------------------------------------------------------- // Fixtures // --------------------------------------------------------------------------- /// The view rect that puts one render pixel on one source pixel. /// /// This is FR-DSP-5's mechanism stated as arithmetic: the render target keeps /// its size while the sampled region shrinks, so a view covering exactly /// `render / source` of the frame samples one-for-one. Nothing anywhere /// switches to a "full resolution path"; the zoom *is* the full-resolution /// path. fn one_to_one(render: (u32, u32)) -> CropRect { let w = render.0 as f32 / SOURCE.0 as f32; let h = render.1 as f32 / SOURCE.1 as f32; CropRect { // Centred, so the sampled region is somewhere a person would actually // look and not a corner the caches treat differently. x: (1.0 - w) * 0.5, y: (1.0 - h) * 0.5, width: w, height: h, } } /// A 60 MP source with detail at every scale. /// /// Not flat and not noise. A flat frame lets the memory system serve every /// sample from one cache line, which flatters the wide kernels of M3 by an /// amount that has nothing to do with photographs; pure noise does the /// opposite. This is a coarse gradient with a fine dither on top, which is /// closer to a real frame's spectrum than either and costs one multiply per /// pixel to generate. fn synthetic_source(ctx: &GpuContext) -> DemosaicedImage { let (w, h) = SOURCE; let mut rgba = vec![0u8; (w as usize) * (h as usize) * 4]; for y in 0..h as usize { let row = y * (w as usize) * 4; for x in 0..w as usize { // A cheap integer hash for the fine structure, so neighbouring // pixels differ and a bilateral filter has something to reject. let n = (x.wrapping_mul(2_654_435_761) ^ y.wrapping_mul(1_640_531_527)) >> 13; let dither = (n & 0x1f) as u32; let gx = (x * 200 / w as usize) as u32; let gy = (y * 55 / h as usize) as u32; let px = &mut rgba[row + x * 4..row + x * 4 + 4]; px[0] = (30 + gx + dither).min(255) as u8; px[1] = (40 + gy + dither).min(255) as u8; px[2] = (60 + gx / 2 + gy + dither).min(255) as u8; px[3] = 255; } } DemosaicedImage::from_rgba8(ctx, &rgba, w, h).expect("upload source") } fn build_chain(chain: Chain) -> EditGraph { let mut graph = EditGraph::default_chain(); match chain { Chain::One => { graph.set_param(exposure::ID, exposure::EXPOSURE, 0.3); } Chain::Five => { graph.set_param(exposure::ID, exposure::EXPOSURE, 0.3); graph.set_param(contrast::ID, contrast::CONTRAST, 25.0); graph.set_param( highlights_shadows::ID, highlights_shadows::HIGHLIGHTS, -40.0, ); graph.set_param(blacks_whites::ID, blacks_whites::BLACKS, -20.0); graph.set_param(vibrance::ID, vibrance::VIBRANCE, 35.0); } Chain::AllPoint => { activate_everything(&mut graph, false); } Chain::Everything => { activate_everything(&mut graph, true); } } graph } /// Move every parameter of every operation off its default. /// /// Written against [`EditGraph::capabilities`] rather than as a list of /// operations, for the same reason the panel is: this bench must not need /// editing when an operation is added, or it will quietly stop measuring "all /// of them" on the first day somebody declares a new node. /// /// A quarter of the way from the default towards the maximum, which is a /// setting a photographer might plausibly reach and — far more importantly — /// is *active*, since an operation sitting at its neutral contributes no /// fragment at all and would make this row a shorter chain wearing a longer /// chain's label. /// /// `neighbourhood` selects whether the operations declaring /// [`Attribute::Detail`] are moved as well. Left off, what remains is exactly /// the set that contributes a fragment to the fused shader — see [`Chain`] for /// why that distinction is the point of this bench. fn activate_everything(graph: &mut EditGraph, neighbourhood: bool) { let moves: Vec<(OpId, ParamId, f32)> = graph .capabilities() .iter() .filter(|op| neighbourhood || !op.attributes.contains(&Attribute::Detail)) .flat_map(|op| { op.params.iter().map(move |p| { let value = match p.kind { ParamKind::Scalar { min, max, .. } => { // Towards whichever end is further away, so a // parameter defaulting to its maximum still moves. let far = if (max - p.default).abs() >= (p.default - min).abs() { max } else { min }; p.default + (far - p.default) * 0.25 } ParamKind::Bool => 1.0, // The second variant when there is one. The first is the // default by construction — see `ParamDescriptor::choice`. // Bound by reference: a descriptor's variants became an // owned `Vec` when descriptors stopped being `&'static`, // and this arm only ever reads the length. ParamKind::Enum { ref variants } => { if variants.len() > 1 { 1.0 } else { 0.0 } } }; (op.id, p.id, value) }) }) .collect(); for (op, param, value) in moves { graph.set_param(op, param, value); } // The film stock is not a slider and so is not reachable through the loop // above: `FilmSim::is_active` is true when it has tables and false // otherwise (see `graph::Film`). Without this the "all" row would be the // whole chain minus its single most expensive node, which is a 3D lookup // and a pair of curve textures. graph.set_film(Some(dr_pipeline::graph::Film { stock: STOCK.into(), print: None, tables: film_tables(), })); } /// The stock the "every operation" chain renders through. A colour negative /// viewed directly, which is the more expensive of the two paths: the LUT is /// consulted either way, and skipping the print step is one fewer thing that /// could be mistaken for the measurement. const STOCK: &str = "kodak_portra_400"; /// The stock the "all operations" row renders through, in the layout the pass /// binds. Baked once per row; baking is milliseconds and happens off the frame /// path in the app too. fn film_tables() -> FilmTables { let film = dr_film::find(STOCK).expect("stock"); let baked = bake(&Recipe::new(film, None)); FilmTables { exposure_matrix: baked.exposure_matrix, curves: baked.curves.clone(), curve_log_min: baked.curve_log_min, curve_log_max: baked.curve_log_max, lut: baked.lut.clone(), density_max: baked.density_max, lut_size: baked.lut_size, // Grain off. It is a per-pixel hash and would be measured; it is also // not part of every edit, and the chain being measured here is "every // operation active", not "every option of every operation". grain_particles: [0.0; 3], grain_density_max: [baked.density_max; 3], grain_uniformity: 0.97, } }