//! The benchmark suite `docs/dev/requirements.md` §8 has been promising. //! //! §8 says performance is verified by *"an automated benchmark suite against a //! synthetic 50k catalog, run per-commit … A regression beyond stated //! tolerance fails the build."* Until this crate there was none: no `benches/`, //! no `[[bench]]`, no criterion, no fixture. Ten performance requirements could //! therefore be neither passed nor failed, and five of them carried a //! requirement tag anyway. //! //! ```sh //! cargo run --release -p dr-bench -- check # measure and gate //! cargo run --release -p dr-bench -- check --reference # …on the reference desktop //! cargo run --release -p dr-bench -- record --reference # rewrite the baseline //! ``` //! //! **Release, always.** The workspace builds its own crates at `opt-level = 0` //! in dev (see the root `Cargo.toml`), and every number here is dominated by //! this workspace's own code — the JPEG decode, the box filter, the resample, //! the sharpen. A debug run measures rustc's shadow, exactly as //! `core/dr-gpu/examples/frame_budget.rs` warns for the same reason. The //! harness says so at the top of every report rather than trusting anyone to //! remember. //! //! # What this covers, and what it deliberately does not //! //! Covered, with a gate: //! //! - **NFR-P1** and R2's catalog clause — opening a 50k catalog and painting //! the first grid ([`catalog_open`]). //! - **NFR-P3** — thumbnail throughput on the embedded preview path //! ([`thumbnails`]). //! //! Measured, reported, and honestly *not* tagged, because only part of the //! requirement is in reach without a GPU or a running UI: //! //! - **NFR-P7** — the encode half of a 24 MP export ([`exporting`]). A //! one-sided gate: it can fail the requirement, it cannot pass it. //! - **NFR-P8** — the catalog layer's share of idle RSS ([`memory`]), together //! with the answer to the question §4.1 asks about GPU memory. //! //! Out of scope entirely, and left to say so rather than faked: NFR-P2, P4, //! P5, P6, P9, P10, P11, P12, P13, P14 and P15. Every one of them needs a //! frame-timing probe inside a running Slint application, a GPU adapter, or //! both. The GPU half of the suite that *does* exist is //! `core/dr-gpu/tests/frame_budget.rs`, which asserts FR-DSP-3 and skips itself //! where there is no adapter; `.gitea/workflows/benchmark.yml` runs it as its //! own job for exactly that reason. //! //! # Exit codes //! //! `0` everything inside its budget and its tolerance; `1` a gate failed; `2` //! the harness itself could not run. Distinguished because a CI log that says //! "failed" should not leave anyone guessing whether the code got slower or the //! fixture would not build. mod baseline; mod catalog_open; mod exporting; mod fixture; mod memory; mod stats; mod thumbnails; use std::collections::BTreeMap; use std::path::PathBuf; use anyhow::Result; use baseline::{Baseline, Judging}; /// The seed the committed baseline describes. /// /// Changing it invalidates every recorded figure, because it changes the /// library being measured. That is why it is a constant here and a field in /// the stamp rather than something a flag quietly varies. const SEED: u64 = 20_260_829; /// Rows in the synthetic library. §8 says 50k; this is that. const IMAGES: usize = 50_000; /// Thumbnails produced for the throughput row. /// /// Twelve hundred rather than fifty thousand. At the target rate the whole /// library is eight minutes of CI, and a rate measured over 1,200 images is /// the same rate — the sweep has no state that changes after the first chunk, /// which the per-image percentile alongside it is there to demonstrate. const THUMBNAILS: usize = 1_200; /// Windows fetched for the scroll row. /// /// A hundred, so nearest-rank puts the 99th percentile on the second-worst — /// the same reading `core/dr-gpu/examples/frame_budget.rs` takes of a hundred /// frames, and the reason it takes it: one bad one in a hundred is one too /// many, and a single scheduler hiccup on an unrelated process should not /// decide the verdict on its own. const WINDOWS: usize = 100; fn main() { env_logger::init(); match run() { Ok(true) => {} Ok(false) => std::process::exit(1), Err(e) => { eprintln!("dr-bench: {e:#}"); std::process::exit(2); } } } /// Returns whether every gate passed. fn run() -> Result { let mut args = std::env::args().skip(1); let command = args.next().unwrap_or_else(|| "help".to_string()); // The probe takes a path and nothing else — see `memory` for why it is a // separate process rather than a function call. if command == "memory-probe" { let path = args .next() .ok_or_else(|| anyhow::anyhow!("memory-probe needs a catalog path"))?; memory::probe(&PathBuf::from(path))?; return Ok(true); } let mut reference = false; let mut fixture_dir = fixture::default_dir(); let mut baseline_path = Baseline::default_path()?; let mut lanes = thumbnails::default_lanes(); let mut thumbnail_count = THUMBNAILS; while let Some(flag) = args.next() { match flag.as_str() { "--reference" => reference = true, "--fixture" => fixture_dir = PathBuf::from(expect_value(&mut args, "--fixture")?), "--baseline" => baseline_path = PathBuf::from(expect_value(&mut args, "--baseline")?), "--lanes" => lanes = expect_value(&mut args, "--lanes")?.parse()?, "--thumbnails" => thumbnail_count = expect_value(&mut args, "--thumbnails")?.parse()?, other => anyhow::bail!("unknown flag {other}; try `dr-bench help`"), } } match command.as_str() { "run" => measure_and_report( &fixture_dir, &baseline_path, reference, lanes, thumbnail_count, Mode::Report, ), "check" => measure_and_report( &fixture_dir, &baseline_path, reference, lanes, thumbnail_count, Mode::Gate, ), "record" => measure_and_report( &fixture_dir, &baseline_path, reference, lanes, thumbnail_count, Mode::Record, ), "help" | "--help" | "-h" => { print_help(); Ok(true) } other => anyhow::bail!("unknown command {other}; try `dr-bench help`"), } } fn expect_value(args: &mut impl Iterator, flag: &str) -> Result { args.next() .ok_or_else(|| anyhow::anyhow!("{flag} needs a value")) } /// What a run does with what it measured. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum Mode { /// Print, judge nothing. Report, /// Print and fail the build on a violated budget or a regression. Gate, /// Print and rewrite the committed baseline from what was measured. Record, } fn print_help() { println!( "\ dr-bench — DarkRoom's performance suite (docs/dev/requirements.md §8) run measure and print, judging nothing check measure, print, and exit 1 on a violated budget or a regression record measure and rewrite docs/dev/bench-baseline.json from the result Flags: --reference this machine is the reference desktop, so machine-sensitive budgets are asserted rather than reported --fixture where the synthetic 50k catalog lives (or $DR_BENCH_DIR) --baseline the committed numbers (default docs/dev/bench-baseline.json) --lanes sweep lanes for the thumbnail row (default: CPU threads) --thumbnails images in the thumbnail row (default 1200) Build it in release. A debug build measures the compiler, not the pipeline — see the module documentation and core/dr-gpu/examples/frame_budget.rs." ); } // --------------------------------------------------------------------------- // The run // --------------------------------------------------------------------------- fn measure_and_report( fixture_dir: &std::path::Path, baseline_path: &std::path::Path, reference: bool, lanes: usize, thumbnail_count: usize, mode: Mode, ) -> Result { let machine = baseline::machine_id(); println!("DarkRoom benchmark suite — the half that needs no GPU"); println!("machine {machine}"); if cfg!(debug_assertions) { println!( "profile DEBUG — every figure below is several times worse than the \ product's. Rerun with --release." ); } else { println!("profile release"); } let (fx, built) = fixture::build(fixture_dir, SEED, IMAGES)?; println!( "fixture {} images, {} sources at {}x{}, seed {}, catalog {:.1} MB{}", fx.stamp.images, fx.stamp.sources, fx.stamp.preview_width, fx.stamp.preview_height, fx.stamp.seed, fx.catalog_bytes as f64 / 1e6, if built { " (built just now)" } else { "" } ); println!(" {}", fx.dir.display()); println!(); // Memory first, and in its own process. See `memory` for why: measuring // RSS after the thumbnail sweep would report the sweep's high-water mark // wearing the catalog's name. let rss = match memory::in_a_fresh_process(&fx.catalog) { Ok(rss) => Some(rss), Err(e) => { println!("memory unavailable: {e}"); None } }; let open = catalog_open::measure(&fx.catalog, WINDOWS)?; let thumbs = thumbnails::measure(&fx.sources, &fx.thumbs, thumbnail_count, lanes)?; let exports = exporting::measure()?; print_details(&open, &thumbs, &exports, rss); let values = collect(&open, &thumbs, &exports, rss); let mut base = Baseline::load(baseline_path)?; if mode == Mode::Record { if !reference { println!( "note recording from a machine that has not declared itself the \ reference desktop.\n §8 records the reference desktop's numbers; \ this overwrites them." ); } for (key, value) in &values { if let Some(metric) = base.metrics.get_mut(key) { metric.recorded = Some(*value); } } base.recorded_on = Some(machine); base.recorded_at_unix = Some(baseline::now_unix()); base.fixture = Some(fx.stamp.clone()); base.save(baseline_path)?; println!("\nrecorded {}", baseline_path.display()); return Ok(true); } let same_machine = base.recorded_on.as_deref() == Some(machine.as_str()); let comparable = base.fixture.as_ref().is_some_and(|f| *f == fx.stamp); let cx = Judging { reference, same_machine: same_machine && comparable, tolerance: base.tolerance, }; let failures = print_verdict(&base, &values, cx, &machine, comparable); if failures.is_empty() { return Ok(true); } println!(); for line in &failures { println!("FAIL {line}"); } println!(); println!( " {} gate(s) failed. docs/dev/benchmarks.md says what each metric measures and\n \ docs/dev/bench-baseline.json holds the numbers these used to be.", failures.len() ); Ok(mode != Mode::Gate) } /// The metric keys, and the measurement each one reads. /// /// One place, so the report, the gate and the recorded file cannot disagree /// about what `catalog_open_ms` means. fn collect( open: &catalog_open::Open, thumbs: &thumbnails::Throughput, exports: &[exporting::EncodeRun], rss: Option, ) -> BTreeMap { let mut v = BTreeMap::new(); v.insert("catalog_open_ms".to_string(), open.cold_ms); v.insert("catalog_open_warm_ms".to_string(), open.warm_ms); v.insert("catalog_window_p99_ms".to_string(), open.window_ms.p99); v.insert("catalog_filtered_ms".to_string(), open.filtered_ms); let ips = thumbs.images_per_second; v.insert("thumbnail_throughput_ips".to_string(), ips); v.insert( "thumbnail_per_image_p99_ms".to_string(), thumbs.per_image.p99, ); for row in exports { v.insert(row.key.to_string(), row.times.p99); } if let Some(rss) = rss { v.insert("catalog_idle_rss_mb".to_string(), rss.now_mb()); } v } // --------------------------------------------------------------------------- // The report // --------------------------------------------------------------------------- fn print_details( open: &catalog_open::Open, thumbs: &thumbnails::Throughput, exports: &[exporting::EncodeRun], rss: Option, ) { println!("Opening the catalog (NFR-P1, and R2's second sentence)"); println!( " {:>10.1} ms open, count, first window and timeline — cold, first \ connection of the process", open.cold_ms ); println!( " {:>10.1} ms the same four calls on a second connection", open.warm_ms ); println!( " {:>10.1} ms Catalog::open alone (connect, migrate, backfill)", open.open_only_ms ); println!( " {:>10.1} ms count and first window under a rating filter", open.filtered_ms ); println!( " {:>10.2} ms one 400-row window at a random offset, p99 of {WINDOWS} \ (p50 {:.2}, max {:.2})", open.window_ms.p99, open.window_ms.p50, open.window_ms.max ); println!( " {} images, {} monthly timeline buckets", open.images, open.buckets ); println!(); println!("Thumbnails on the embedded preview path (NFR-P3: >= 100 img/s)"); println!( " {:>10.1} img/s over {} images on {} lanes, {:.2} s of wall clock", thumbs.images_per_second, thumbs.images, thumbs.lanes, thumbs.wall_ms / 1e3 ); println!( " {:>10.2} ms per image on its lane, p99 (p50 {:.2}, max {:.2})", thumbs.per_image.p99, thumbs.per_image.p50, thumbs.per_image.max ); println!( " {} stored, {} failed", thumbs.stored, thumbs.failed ); println!(); println!(" The bytes are in memory before the clock starts, so this is CPU"); println!(" throughput with the fetch already paid for. That is what the target's"); println!(" \"embedded preview path\" names; it is not a claim about a remote library."); println!(); println!("Exporting 24 MP — the encode half only (NFR-P7 is the whole chain)"); println!( " {:>14} {:>11} {:>9} {:>9} {:>9}", "sizing", "output", "p50", "p99", "file" ); for row in exports { println!( " {:>14} {:>5}x{:<5} {:>7.1}ms {:>7.1}ms {:>7.1}MB", row.label, row.width, row.height, row.times.p50, row.times.p99, row.bytes as f64 / 1e6 ); } println!(); println!(" No GPU render is in these figures, so they cannot pass NFR-P7 — only fail"); println!(" it. See tools/bench/src/exporting.rs for why that is still worth gating."); println!(); println!("Idle memory with the catalog open (NFR-P8's catalog share)"); match rss { Some(rss) => { println!( " {:>10.1} MB resident after opening 50k and scrolling 10k rows", rss.now_mb() ); println!(" {:>10.1} MB peak for that process", rss.peak_mb()); println!(); println!(" RSS is exclusive of device-local GPU memory and this process has no"); println!(" toolkit, no adapter and no decode cache — so it is the catalog layer's"); println!(" share of NFR-P8's 500 MB, not NFR-P8. tools/bench/src/memory.rs holds"); println!(" the answer to the question §4.1 asks, and the change it recommends."); } None => println!(" not measured on this platform"), } println!(); } /// Print the gate table and return the failures, one line each. fn print_verdict( base: &Baseline, values: &BTreeMap, cx: Judging, machine: &str, comparable_fixture: bool, ) -> Vec { println!("Against docs/dev/bench-baseline.json"); match (&base.recorded_on, cx.same_machine) { (None, _) => println!( " No baseline has been recorded yet. Budgets are still gated; drift is not.\n \ Run `dr-bench record --reference` on the reference desktop and commit the diff." ), (Some(on), true) => println!(" Recorded on {on} — drift below is a verdict."), (Some(on), false) if !comparable_fixture => println!( " Recorded on {on} against a different fixture; drift below is information only." ), (Some(on), false) => println!( " Recorded on {on}, and this is {machine}. Drift below is information, not a verdict." ), } if !cx.reference { println!( " Not the reference desktop (--reference), so machine-sensitive budgets are\n \ reported rather than asserted — §8 names the reference desktop, not CI." ); } println!(); println!( " {:<30} {:>12} {:>13} {:>12} {:>8} verdict", "metric", "measured", "budget", "baseline", "drift" ); let mut failures = Vec::new(); for (key, measured) in values { let Some(metric) = base.metrics.get(key) else { println!( " {key:<30} {measured:>12.2} {:>13} {:>12} {:>8} not in the baseline", "—", "—", "—" ); continue; }; let j = metric.judge(*measured, cx); let budget = match (metric.budget, metric.direction) { (None, _) => "—".to_string(), (Some(b), baseline::Direction::LowerIsBetter) => format!("< {b:.1}"), (Some(b), baseline::Direction::HigherIsBetter) => format!("> {b:.1}"), }; let recorded = match metric.recorded { Some(r) => format!("{r:.2}"), None => "—".to_string(), }; let drift = match j.drift { Some(d) => format!("{:+.1}%", d * 100.0), None => "—".to_string(), }; let mut verdict = String::new(); if j.over_budget { verdict.push_str("OVER BUDGET "); failures.push(format!( "{key} is {measured:.2} {}, past {budget} ({})", metric.unit, metric.requirement )); } if j.regressed { verdict.push_str("REGRESSED "); failures.push(format!( "{key} drifted {drift} against the baseline, past the {:.0}% tolerance", cx.tolerance * 100.0 )); } if verdict.is_empty() { verdict.push_str(if j.budget_deferred { "ok (budget deferred)" } else { "ok" }); } println!(" {key:<30} {measured:>12.2} {budget:>13} {recorded:>12} {drift:>8} {verdict}"); } // A metric in the file that nothing measured is a harness that has drifted // from its own record, and is worth saying out loud rather than leaving as // a row that quietly stopped appearing. for key in base.metrics.keys() { if !values.contains_key(key) { println!(" {key:<30} {:>12} measured nothing this run", "—"); } } failures }