diff --git a/.gitea/workflows/benchmark.yml b/.gitea/workflows/benchmark.yml new file mode 100644 index 0000000..782a927 --- /dev/null +++ b/.gitea/workflows/benchmark.yml @@ -0,0 +1,193 @@ +name: Benchmarks + +# The suite docs/requirements.md §8 has been promising since it was written: +# "an automated benchmark suite against a synthetic 50k catalog, run per-commit +# … A regression beyond stated tolerance fails the build." +# +# Its own workflow rather than a step inside build-and-test.yml, and the reason +# is what a failure here means. A red `Build and test` says the code is wrong; a +# red `Benchmarks` says the code is slower than it was, which is a different +# conversation, is read by different people, and must not be reachable by +# retrying a flaky compile. +# +# # Why this is split in two +# +# §8 names "the reference desktop", not CI, and it is right to. So: +# +# cpu — runs on every push. It needs no adapter and no display, and the +# budgets it asserts (a 50k catalog opening inside two seconds) have +# two orders of magnitude of headroom, so a modest runner can be held +# to them honestly. Machine-sensitive budgets — throughput targets +# written for a 24-thread desktop — are reported here rather than +# asserted; `dr-bench` decides that per metric and says so in its +# report. Asserting them on a two-core container would produce exactly +# what core/dr-gpu/tests/frame_budget.rs refused to produce: a red gate +# everybody learns to ignore. +# +# gpu — the frame budget, which already exists and already skips itself where +# there is no adapter. Not on push: it would build wgpu and naga on +# every commit to establish, every time, that this runner has no GPU. It +# runs on demand (Actions → Run workflow) so that a runner that *does* +# have one can be pointed at it, and the numbers it produces belong in +# docs/frame-budget.md by hand, as they already are. + +on: + push: + branches: [main, master, develop] + pull_request: + branches: [main, master, develop] + workflow_dispatch: + +jobs: + cpu: + runs-on: linux/amd64 + name: CPU and I/O (per commit) + # Node for actions/checkout and actions/cache, which the bare runner image + # cannot execute. Rust is installed below. + container: + image: catthehacker/ubuntu:act-latest + + env: + # Same reasoning as the desktop job in build-and-test.yml: incremental + # state exists to make the *second* build in a working tree fast, which is + # not a thing a fresh checkout has, and it fills the runner's disk. + CARGO_INCREMENTAL: 0 + # The fixture, out of the workspace so actions/cache never picks it up. + # A 14 MB synthetic catalog is two seconds to regenerate and would + # otherwise be uploaded and downloaded on every push to save them. + DR_BENCH_DIR: /tmp/darkroom-bench + + steps: + - name: Checkout + uses: actions/checkout@v4 + + # No `git lfs pull` here, deliberately. `dr-bench` depends on the catalog, + # the decoder, the thumbnail store and the encoder, and on nothing that + # reaches `dr-segment` — so the model this repository keeps in LFS is not + # part of this job's dependency graph and fetching it would be a minute + # spent on a file nothing opens. + + - name: Cache cargo + uses: actions/cache@v4 + with: + path: | + ~/.cargo/registry + ~/.cargo/git + target + key: bench-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }} + + # Pinned to the workspace rust-version, as every other job here is: a + # floating toolchain turns an unrelated push into a mystery failure, and + # for a benchmark it would turn one into a mystery *regression*. + # + # rust-analyzer is named for the reason build-and-test.yml gives: rustup + # reconciles rust-toolchain.toml on the first cargo call whatever this + # step asks for, so naming it keeps the download inside the step that says + # it is installing things. + - name: Install Rust 1.92.0 + run: | + set -e + curl -fsSL https://sh.rustup.rs | sh -s -- \ + -y --no-modify-path --profile minimal --default-toolchain 1.92.0 \ + --component rust-analyzer + echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" + + # `-p dr-bench`, not `--workspace`. The whole point of that crate having + # no GPU and no UI dependency is that this job resolves the catalog, the + # decoder and the encoders and stops there — a few minutes rather than the + # release build of Slint and wgpu the desktop job pays for. + # + # Release, and it is not optional: the workspace builds its own crates at + # opt-level = 0 in dev, and every figure this produces is dominated by + # this workspace's own code. A debug run would measure rustc. + - name: Build the suite + run: cargo build --release -p dr-bench + + # Exit 1 is a violated budget or a regression past tolerance; exit 2 is + # the harness failing to run at all. Both fail the job, and the report + # above the failure says which. + - name: Measure, and gate + run: cargo run --release -p dr-bench -- check + + - name: Disk after + if: always() + run: df -h /workspace 2>/dev/null || df -h . + + gpu: + # On demand only — see the header. A runner with a Vulkan device can be + # pointed at this; one without will skip the measurement and say so, which + # is the same posture the rest of this repository's device tests take. + if: github.event_name == 'workflow_dispatch' + runs-on: linux/amd64 + name: Frame budget (on demand) + container: + image: catthehacker/ubuntu:act-latest + + env: + CARGO_INCREMENTAL: 0 + + steps: + - name: Checkout + uses: actions/checkout@v4 + + # `dr-gpu` depends on `dr-segment` for the watershed's pixel passes. Its + # default features are off, so no weights are compiled in — but the fetch + # is cheap insurance and its failure is not fatal. The header of the same + # step in build-and-test.yml explains why the extraheader is stripped + # rather than reused: two Authorization headers is a 400 from Gitea, one + # step after the batch call that had just succeeded. + - name: Fetch the segmentation model + continue-on-error: true + env: + LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }} + run: | + set -e + git lfs install --local + git config --local --get-regexp '^http\..*extraheader$' \ + | cut -d' ' -f1 | sort -u \ + | while read -r key; do git config --local --unset-all "$key"; done || true + git config --local lfs.url \ + "https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs" + git lfs pull + + - name: Cache cargo + uses: actions/cache@v4 + with: + path: | + ~/.cargo/registry + ~/.cargo/git + target + key: bench-gpu-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }} + + - name: Build dependencies + run: | + apt-get update -qq + apt-get install -y -qq pkg-config libfontconfig1-dev libxkbcommon-dev + + - name: Install Rust 1.92.0 + run: | + set -e + curl -fsSL https://sh.rustup.rs | sh -s -- \ + -y --no-modify-path --profile minimal --default-toolchain 1.92.0 \ + --component rust-analyzer + echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" + + # The guard, in release. Its own module documentation is explicit that a + # release run checks strictly more than a dev one: the CPU half of a frame + # is shader-string assembly, which is several times slower unoptimised, so + # it is folded into the assertion only when debug_assertions is off. + # + # With no adapter this prints "skipping: no GPU adapter" and passes. A + # test that cannot run is not evidence either way, and turning that into a + # failure would make the job useless on the runner it usually lands on. + - name: Frame budget (FR-DSP-3) + run: cargo test --release -p dr-gpu --test frame_budget -- --nocapture + + # The instrument behind docs/frame-budget.md. It exits non-zero with no + # adapter, which is right for a tool a person runs deliberately and wrong + # for a job that usually has none — hence continue-on-error. Its table is + # in the log for whoever asked for this run; the committed numbers are + # still updated by hand, as that file says. + - name: Frame budget table + continue-on-error: true + run: cargo run --release -p dr-gpu --example frame_budget diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 7c554e9..c730269 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -83,6 +83,21 @@ GPU tests skip themselves where there is no adapter rather than failing — a test that cannot run is not evidence either way — so a green run on a machine without a GPU is expected, and does not mean the GPU paths were exercised. +There is a fifth check, and it is not in that list because you are unlikely to +break it by accident: + +```bash +cargo run --release -p dr-bench -- check +``` + +That is the benchmark suite (`docs/requirements.md` §8), which builds a +synthetic 50,000-image catalog and fails the build if a performance target is +missed or a measurement has drifted past its tolerance. It runs on every push in +its own workflow. [`docs/benchmarks.md`](docs/benchmarks.md) says what it +measures, what it deliberately does not, and how to read a failure. If you have +touched the catalog, the decoder, the thumbnail store or the exporter, run it +before you send. + ## Requirements and traceability [`requirements.md`](docs/requirements.md) is the register of record. @@ -150,6 +165,7 @@ One commit per change. If you fixed two things, that is two commits. | [`core/dr-pipeline/ops/README.md`](core/dr-pipeline/ops/README.md) | Adding or changing a develop operation — start here regardless | | [`docs/architecture.md`](docs/architecture.md) | Anything touching the render path, catalog or sync | | [`docs/code-health.md`](docs/code-health.md) | Deciding what to work on; grades each seam by what it costs | +| [`docs/benchmarks.md`](docs/benchmarks.md) | A change that could plausibly cost time or memory | | [`docs/technical-debt.md`](docs/technical-debt.md) | Something looks wrong — check it was not chosen | | [`docs/distribution.md`](docs/distribution.md) | Packaging a build, or adding a permission to one | | [`docs/requirements.md`](docs/requirements.md) | Reference, not reading | diff --git a/Cargo.lock b/Cargo.lock index 57645f1..afcc877 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1403,6 +1403,23 @@ version = "0.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76" +[[package]] +name = "dr-bench" +version = "0.9.0" +dependencies = [ + "anyhow", + "dr-catalog", + "dr-decode", + "dr-export", + "dr-thumbs", + "dr-types", + "env_logger", + "log", + "rusqlite", + "serde", + "serde_json", +] + [[package]] name = "dr-catalog" version = "0.9.0" diff --git a/Cargo.toml b/Cargo.toml index 1a57353..875ca97 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -21,6 +21,7 @@ members = [ "ui/dr-ui", "apps/darkroom-desktop", "apps/darkroom-android", + "tools/bench", "tools/traceability", ] diff --git a/docs/bench-baseline.json b/docs/bench-baseline.json new file mode 100644 index 0000000..579873f --- /dev/null +++ b/docs/bench-baseline.json @@ -0,0 +1,113 @@ +{ + "_readme": [ + "The committed numbers for DarkRoom's benchmark suite (docs/requirements.md §8).", + "Produced and checked by `cargo run --release -p dr-bench`; docs/benchmarks.md explains each metric.", + "", + "Two gates, and they are not the same gate. `budget` is the requirement's own threshold and never moves.", + "`recorded` is what the reference desktop last measured, and a run that drifts past `tolerance` beyond it", + "fails the build even while still inside the budget — which is how most performance rot actually arrives.", + "", + "`recorded` is null on every metric because nobody has run the suite yet. That is deliberate: writing", + "plausible-looking figures here would make every later comparison a comparison against a guess. Run", + "`cargo run --release -p dr-bench -- record --reference` on the reference desktop and commit the diff.", + "Until then the budget gate works and the regression gate says, in the report, that it cannot.", + "", + "`machine_sensitive` says whether a budget is a statement about a machine as much as about the code.", + "Those budgets are asserted only under --reference: §8 names the reference desktop, and a two-core CI", + "container cannot speak to a target written for twenty-four threads. Asserting one there would produce a", + "red gate everybody learns to ignore, which is the trap core/dr-gpu/tests/frame_budget.rs already avoids.", + "", + "Several metrics carry a requirement ID with a qualifier. Read those literally. NFR-P7's budget here is", + "checked against the encode half of an export only — no GPU render is in the figure — so it can fail the", + "requirement and cannot pass it, and no TRACES tag claims otherwise. NFR-P8 has no budget at all yet,", + "because nobody has decided how much of its 500 MB belongs to the catalog layer; this records the number", + "that decision needs." + ], + "tolerance": 0.15, + "recorded_on": null, + "recorded_at_unix": null, + "fixture": null, + "metrics": { + "catalog_filtered_ms": { + "requirement": "FR-CAT-6", + "what": "Count plus first window under a rating filter, which compiles to a correlated subquery over versions.", + "unit": "ms", + "direction": "lower_is_better", + "machine_sensitive": true, + "budget": null, + "recorded": null + }, + "catalog_idle_rss_mb": { + "requirement": "NFR-P8 (the catalog layer's share only — no toolkit, no adapter, no decode cache)", + "what": "Resident memory of a process that opened the 50k catalog and scrolled ten thousand rows.", + "unit": "MB", + "direction": "lower_is_better", + "machine_sensitive": false, + "budget": null, + "recorded": null + }, + "catalog_open_ms": { + "requirement": "NFR-P1", + "what": "Catalog::open plus the count, first window and timeline the grid cannot paint without.", + "unit": "ms", + "direction": "lower_is_better", + "machine_sensitive": false, + "budget": 2000.0, + "recorded": null + }, + "catalog_open_warm_ms": { + "requirement": "NFR-P1", + "what": "The same four calls on a second connection, with SQLite's page cache already warm.", + "unit": "ms", + "direction": "lower_is_better", + "machine_sensitive": false, + "budget": 2000.0, + "recorded": null + }, + "catalog_window_p99_ms": { + "requirement": "FR-CAT-4", + "what": "One 400-row grid window at a random offset, p99 of one hundred.", + "unit": "ms", + "direction": "lower_is_better", + "machine_sensitive": true, + "budget": null, + "recorded": null + }, + "export_24mp_long_edge_2048_ms": { + "requirement": "FR-EXP-3", + "what": "The web export: resample a 24 MP frame to a 2048 px long edge, sharpen, encode. p99 of five.", + "unit": "ms", + "direction": "lower_is_better", + "machine_sensitive": true, + "budget": null, + "recorded": null + }, + "export_24mp_original_ms": { + "requirement": "NFR-P7 (the encode half only — the GPU render is not in this figure)", + "what": "Resample, output-sharpen and JPEG-encode a 24 MP frame at source size. p99 of five.", + "unit": "ms", + "direction": "lower_is_better", + "machine_sensitive": true, + "budget": 2000.0, + "recorded": null + }, + "thumbnail_per_image_p99_ms": { + "requirement": "NFR-P3", + "what": "One thumbnail on its own lane: decode the preview, downscale, orient, encode. p99.", + "unit": "ms", + "direction": "lower_is_better", + "machine_sensitive": true, + "budget": null, + "recorded": null + }, + "thumbnail_throughput_ips": { + "requirement": "NFR-P3", + "what": "Whole-sweep throughput: 1200 thumbnails through the sweep's chunk-and-lane shape, wall clock.", + "unit": "img/s", + "direction": "higher_is_better", + "machine_sensitive": true, + "budget": 100.0, + "recorded": null + } + } +} diff --git a/docs/benchmarks.md b/docs/benchmarks.md new file mode 100644 index 0000000..5947862 --- /dev/null +++ b/docs/benchmarks.md @@ -0,0 +1,243 @@ +# The benchmark suite + +**Status:** Built, not yet recorded · 2026-08-30 +**Companion to:** [requirements.md](requirements.md) §4.1 (performance targets) · §8 (verification) +**Instrument:** [`tools/bench`](../tools/bench) — `cargo run --release -p dr-bench -- check` +**Committed numbers:** [`bench-baseline.json`](bench-baseline.json) +**GPU half:** [`core/dr-gpu/tests/frame_budget.rs`](../core/dr-gpu/tests/frame_budget.rs) · +[frame-budget.md](frame-budget.md) + +§8 has said since it was written that performance is verified by *"an automated +benchmark suite against a synthetic 50k catalog, run per-commit … A regression +beyond stated tolerance fails the build."* Until this suite there was none. No +`benches/`, no `[[bench]]`, no criterion, no fixture — and ten performance +requirements that could therefore be neither passed nor failed, five of them +carrying a `TRACES:` tag regardless. + +This file is what the suite covers, what it deliberately does not, and how to +read a failure. + +--- + +## The state of it, first + +**No numbers have been recorded yet.** Every `recorded` field in +[`bench-baseline.json`](bench-baseline.json) is `null`, on purpose: writing +plausible-looking figures into a baseline would make every later comparison a +comparison against a guess, and the first real regression would be invisible. + +To record them, on the reference desktop: + +```sh +cargo run --release -p dr-bench -- record --reference +``` + +and commit the diff. Until that happens the **budget** gate works — a catalog +that takes three seconds to open fails the build today — and the **regression** +gate reports that it has nothing to compare against, rather than pretending. + +--- + +## What it measures + +| Metric | Requirement | Gated? | +|---|---|---| +| `catalog_open_ms` | **NFR-P1**, and R2's second sentence | Yes, everywhere — budget 2000 ms | +| `catalog_open_warm_ms` | NFR-P1, page cache warm | Yes, everywhere — budget 2000 ms | +| `catalog_window_p99_ms` | FR-CAT-4 | Regression only | +| `catalog_filtered_ms` | FR-CAT-6 | Regression only | +| `thumbnail_throughput_ips` | **NFR-P3** | Budget 100 img/s, on the reference desktop | +| `thumbnail_per_image_p99_ms` | NFR-P3 | Regression only | +| `export_24mp_original_ms` | NFR-P7, **encode half only** | One-sided: can fail it, cannot pass it | +| `export_24mp_long_edge_2048_ms` | FR-EXP-3 | Regression only | +| `catalog_idle_rss_mb` | NFR-P8, **catalog layer only** | Regression only — see below | + +Two of those rows carry a qualifier, and the qualifiers are the point. + +### Requirements this can now pass *or* fail + +**NFR-P1 — catalog open under 2 s.** The measured span is the four things the +library view cannot paint without: `Catalog::open` (which connects, migrates and +**backfills**, and the backfill is three passes over the images table on every +open), `count`, the first 400-row `window`, and the monthly `timeline`. Tagged +`TRACES: NFR-P1` in [`tools/bench/src/catalog_open.rs`](../tools/bench/src/catalog_open.rs), +because a build that breaks it fails this gate. + +**NFR-P3 — ≥ 100 images per second on the embedded preview path.** The +per-image work is exactly what `spawn_thumbnail_sweep` does — `decode_jpeg`, +`Preview::downscale_to`, `Preview::apply_orientation`, `encode_rgba`, +`ThumbStore::put` — arranged in the same shape: chunks of 96, lanes owning +disjoint slices, and the single thread that owns the store writing the finished +chunk. Tagged `TRACES: NFR-P3` in +[`tools/bench/src/thumbnails.rs`](../tools/bench/src/thumbnails.rs). + +### Requirements this can only half-answer, and is not tagged for + +**NFR-P7 — 24 MP export under 2 s, full chain.** The full chain is decode, +demosaic, a full-resolution GPU render, a read-back, then resize, sharpen and +encode. Only the last three run without an adapter. So the figure here is a +**lower bound** on the requirement: exceeding 2 s in the encode alone violates +NFR-P7 no matter how fast the render is, and coming in under it proves nothing. +The budget is gated on that basis and there is no `TRACES: NFR-P7` anywhere in +`tools/bench`. + +**NFR-P8 — idle memory under 500 MB.** The probe is a fresh process holding the +catalog and nothing else: no Slint, no wgpu device, no font stack, no decode +cache. Its RSS is the catalog layer's *share* of that 500 MB, not the figure the +requirement is about. It carries no budget for a reason given below. + +### Requirements out of scope, listed so their absence reads as a decision + +NFR-P2 (grid scroll at 60 fps), P4 (open in develop), P5 (slider to visible), +P6 (pan/zoom), P9 (UI-executor blocking), P10 (touch response), P11 (layout +transition), P12 (warm shader setup), P13 (next image in culling), P14 (focus +peaking), P15 (drawn mask stroke). Every one of them needs a frame-timing probe +inside a running Slint application, a GPU adapter, or both. None is faked here. + +The GPU half of the story that *does* exist is +[frame-budget.md](frame-budget.md) and its guard test, which asserts FR-DSP-3 +and skips itself where there is no adapter. `.gitea/workflows/benchmark.yml` +runs it as its own job for exactly that reason. + +--- + +## The fixture + +Fifty thousand rows over a pool of twelve real image files. Rows are cheap and +pixels are not: everything the catalog half touches is rows and is therefore +exact at full scale, and everything the pixel half touches is one file at a time +and does not care how many rows point at it. The result is ~14 MB on disk +instead of ~2 TB, and neither half is flattered by that. + +| | | +|---|---| +| Rows | 50,000 images, 50,000 default versions, 400 folders, one root | +| Capture times | Twelve years from a fixed epoch, so the timeline has ~144 monthly buckets | +| Sources | 12 synthesised JPEGs at 1620 × 1080 — the size `dr-decode` records a CR2 carrying in IFD2 | +| Seed | 20260829, in [`tools/bench/src/main.rs`](../tools/bench/src/main.rs) | +| Location | `$DR_BENCH_DIR`, else the system temporary directory | + +It is reproducible from the seed, and a `stamp.json` beside it records what it +was built from — seed, row count, source count, preview size, and `dr-catalog`'s +schema version. A mismatch rebuilds rather than silently measuring a different +workload than the baseline describes. + +Two honest limits on it: + +- **The page cache is warm.** The fixture was written by this suite or by an + earlier run of it, so neither the catalog open nor the thumbnail sweep pays + for a cold disk. On the reference desktop's NVMe a genuinely cold read of a + 14 MB catalog is tens of milliseconds; on spinning rust it is not. +- **The sources are synthetic.** A coarse gradient with a fine dither, which is + what `frame_budget.rs` synthesises for the same reason — a flat frame lets the + memory system serve every sample from one cache line and flatters a box + filter, and pure noise defeats the entropy coder in the other direction. + +--- + +## Two gates, and how to read a failure + +**Budget.** The requirement's own threshold. It does not move. Failing it means +a requirement is violated. + +**Regression.** More than 15% worse than the last recorded figure *on the same +machine, against the same fixture*. Failing it means the code got slower while +still inside the requirement — which is how most performance rot actually +arrives, never over the line, always a little worse, until one day the line is +crossed by a change that was not the cause. + +A metric declares whether its budget is `machine_sensitive`. Those are asserted +only under `--reference`, and reported everywhere else. §8 names *"the reference +desktop"*, not CI, and it is right to: a container with two cores cannot speak +to a throughput target written for twenty-four threads, and asserting one there +would produce exactly what `core/dr-gpu/tests/frame_budget.rs` refused to +produce — *"a red suite that everyone learns to ignore"*. Catalog open is not +machine-sensitive: 2 s against an expected figure two orders of magnitude +smaller is a threshold any machine can be held to. + +Exit codes: `0` everything passed, `1` a gate failed, `2` the harness itself +could not run. Distinguished so a CI log that says "failed" does not leave +anyone guessing whether the code got slower or the fixture would not build. + +**Release, always.** The workspace builds its own crates at `opt-level = 0` in +dev, and every figure here is dominated by this workspace's own code — the JPEG +decode, the box filter, the resample, the sharpen. A debug run measures rustc's +shadow. The report says which profile it was built in on its second line. + +--- + +## NFR-P8, and the question §4.1 asks + +§4.1 says NFR-P8 *"must state whether it measures RSS inclusive or exclusive of +GPU allocations, and whether it holds after SQLite's page cache warms on a 50k +catalog."* Both halves have an answer. + +**On the page cache: warm.** The probe runs the count, the timeline and +twenty-five windows before it reads its counters, so SQLite's cache holds the +b-tree pages a scroll touches. That is the right side to err on — a figure taken +before the cache warms would understate a steady-state library. + +**On GPU memory: RSS is exclusive of device-local allocations, and cannot be +made otherwise.** A Vulkan allocation in a device-local heap never enters the +process's address space, so nothing under `/proc/self/status` can see it. What +*does* land in RSS is the host-visible side — staging buffers, mapped upload +rings, the read-back `AdjustPass` performs on export — plus the driver's own +resident pages. + +So "idle memory < 500 MB" is two questions wearing one number, and a build +holding 400 MB of RSS and 3 GB of textures would pass it. + +**Recommendation: NFR-P8 should be restated as two figures** — host RSS +exclusive of device-local memory, and a separate VRAM ceiling read from the +adapter — because the second is the one that decides whether the application +survives beside a browser on an 8 GB card, and nothing in this repository +measures it today. + +**And a decision is outstanding.** `catalog_idle_rss_mb` carries no budget +because nobody has decided how much of the 500 MB belongs to the catalog layer +and how much to everything above it. The suite records the number so that +decision can be taken against a measurement rather than an estimate. When it is +taken, put the figure in `budget` and the metric becomes a gate. + +--- + +## What is not measured, and would be worth adding + +- **The UI's own open.** `ui/dr-ui/src/library.rs` does not call + `Catalog::count` or `Catalog::window`; it issues its own SQL against the same + tables, with a `VISIBLE` predicate and a burst-folding clause. `dr-bench` + cannot see those without depending on `dr-ui`, which would drag Slint into a + job that has no display. **Falsifiable end:** when the grid's queries move + down into `dr-catalog` — which is where SQL over catalog tables belongs — + `catalog_open_ms` becomes the whole of the application's open and this caveat + can be deleted rather than argued about. +- **The remote sweep.** `spawn_thumbnail_sweep`'s wall clock against a real + server is latency, not CPU, and is what FR-NC-3's design is judged by. It + needs a server and belongs in a different kind of test. +- **A cold disk.** See the fixture's limits above. +- **Android.** §4.1 states a second column of targets and §8 asks for + "periodically on the named reference Android devices". Nothing here runs on a + device. Spike S10 is the piece of work that would start it. +- **Everything with a frame in it.** See the out-of-scope list above. + +--- + +## Running it + +```sh +# Measure and print. Judges nothing. +cargo run --release -p dr-bench -- run + +# Measure and gate. What CI runs. +cargo run --release -p dr-bench -- check + +# The same, with machine-sensitive budgets asserted too. +cargo run --release -p dr-bench -- check --reference + +# Rewrite bench-baseline.json from this run, and commit the diff. +cargo run --release -p dr-bench -- record --reference +``` + +Useful flags: `--fixture ` (or `$DR_BENCH_DIR`) to put the synthetic +catalog somewhere specific, `--lanes ` to pin the sweep's parallelism, and +`--thumbnails ` to lengthen or shorten the throughput row. diff --git a/docs/outstanding.md b/docs/outstanding.md index 1bc0473..171ca18 100644 --- a/docs/outstanding.md +++ b/docs/outstanding.md @@ -285,31 +285,43 @@ load-bearing for interoperating with the editors FR-CAT-14 imports from. --- -## 8. The performance targets are unverified, not unmet +## 8. The performance targets are half-verified, and the half that is left is the hard one -Eleven of the fifteen §4.1 targets carry no tag: NFR-P2, -P3, -P4, -P6, -P7, -P8, -P10, -P11, -P12, --P14, -P15. That is the uninteresting part of this section. +§8 and §4.1 both require the same thing in the same words: an automated benchmark suite against a +synthetic 50k catalog, run per commit, where **"a regression beyond a stated tolerance is a build +failure, not a notification."** For most of this project's life it did not exist — no `benches/`, no +criterion, no synthetic catalog, and three CI workflows that between them measured nothing. -The interesting part is that §8 and §4.1 both require the same thing, in the same words, and it does -not exist: an automated benchmark suite against a synthetic 50k catalog, run per commit, where **"a -regression beyond a stated tolerance is a build failure, not a notification."** There is no -`benches/` directory in the workspace, no criterion dependency, and no synthetic catalog. The three -CI workflows run `cargo fmt --check`, clippy, `cargo test --workspace`, a release build, an Android -cross-build and a layering check. None of them measures anything, so there is no baseline to -regress against and no tolerance to exceed. +**It exists now, for everything that does not need a frame.** [`tools/bench`](../tools/bench) builds +a deterministic 50,000-row catalog over a pool of a dozen real files, measures against it, and fails +the build on a violated budget or a drift past tolerance; +[`.gitea/workflows/benchmark.yml`](../.gitea/workflows/benchmark.yml) runs it on every push, and +[benchmarks.md](benchmarks.md) is the account of what it does and does not cover. **NFR-P1** and +**NFR-P3** are now genuinely gated, and R2's "catalog opens in under 2s" clause with them. -What does exist is narrower and genuinely good: `dr-gpu/examples/frame_budget` is a real instrument, -its results are committed in [frame-budget.md](frame-budget.md) with the machine and profile named, -and TD-4's before-and-after was measured with it. But it is run by hand — frame-budget.md's own -instruction is "rerun and diff this file" — and the guard version that does live in CI skips itself -where there is no GPU adapter, which the workflow notes is the normal case on a runner, while -asserting its CPU half only when `debug_assertions` is off, which a dev-profile `cargo test` is not. -In CI it therefore asserts approximately nothing. +Three qualifications, all of them stated in the harness itself rather than only here: -**The claim to take from this is precise.** Nothing here says the performance targets are missed. -Several are plausibly met. It says that if one were broken tomorrow, nobody would find out — which -is the failure mode §8 was written to prevent, and the reason it belongs in this document rather -than in a backlog. +- **The numbers have not been recorded yet.** Every `recorded` field in + [bench-baseline.json](bench-baseline.json) is `null`, deliberately: a fabricated baseline is worse + than none. Until `dr-bench record --reference` is run on the reference desktop and committed, the + budget gate works and the regression gate does not. +- **NFR-P7 and NFR-P8 are half-measured and are not tagged.** The export row covers the encode half + of the chain and no GPU render, so it can fail the requirement and cannot pass it. The memory row + covers a process holding the catalog and nothing else — no toolkit, no adapter — so it is the + catalog layer's share of the 500 MB rather than the figure NFR-P8 is about. Neither carries a + `TRACES:` tag, which is the point. +- **NFR-P8 needs a decision, not more code.** How much of its 500 MB belongs below the UI is + unstated, and until somebody says, the metric can record but not judge. [benchmarks.md](benchmarks.md) + also answers the question §4.1 raises about GPU memory — RSS cannot see device-local allocations + at all — and recommends restating the requirement as two figures. + +**What is left is the frame-timing half, and it is the hard one.** NFR-P2, -P4, -P5, -P6, -P9, -P10, +-P11, -P12, -P13, -P14 and -P15 all need a probe inside a running Slint application, a GPU adapter, +or both. `dr-gpu/examples/frame_budget` is a real instrument for the GPU part and its results are +committed in [frame-budget.md](frame-budget.md) with the machine and profile named — but it is run +by hand, and the guard version in CI skips itself where there is no adapter, which is the normal +case on a runner. So the claim to take from this section is now narrower than it was, and still +true: **a scroll that dropped to 30 fps tomorrow would reach a user before it reached CI.** --- diff --git a/tools/bench/Cargo.toml b/tools/bench/Cargo.toml new file mode 100644 index 0000000..2ed24d8 --- /dev/null +++ b/tools/bench/Cargo.toml @@ -0,0 +1,28 @@ +[package] +name = "dr-bench" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true + +# The suite deliberately depends on no GPU crate and no UI toolkit. +# +# That is not tidiness, it is what makes the CI job affordable: `cargo build +# --release -p dr-bench` resolves the catalog, the decoder, the thumbnail store +# and the encoder, and stops there. Adding `dr-gpu` would pull wgpu and naga +# into a job that has no adapter to use them with, and adding `dr-ui` would +# pull Slint. The GPU half of the suite is `dr-gpu`'s own frame-budget test, +# which already exists and already skips itself where there is no device — see +# `docs/benchmarks.md`. +[dependencies] +dr-types.workspace = true +dr-catalog.workspace = true +dr-thumbs.workspace = true +dr-decode.workspace = true +dr-export.workspace = true +rusqlite.workspace = true +serde.workspace = true +serde_json.workspace = true +anyhow.workspace = true +log.workspace = true +env_logger.workspace = true diff --git a/tools/bench/src/baseline.rs b/tools/bench/src/baseline.rs new file mode 100644 index 0000000..d8321b1 --- /dev/null +++ b/tools/bench/src/baseline.rs @@ -0,0 +1,300 @@ +//! The committed numbers, and what counts as a regression against them. +//! +//! `docs/frame-budget.md` commits its measurements by hand and says why: *"a +//! regression should be a diff rather than somebody's memory."* This is the +//! same idea in a form a program can read, because §8 asks for more than a +//! record — *"a regression beyond stated tolerance fails the build"*. +//! +//! # Two gates, and they are not the same gate +//! +//! **The budget** is the requirement's own number: 2 s to open a catalog, 100 +//! images a second through the preview path. It does not move. A build that +//! violates it has violated a requirement, and no amount of "but it was always +//! like that" changes it. +//! +//! **The baseline** is what this machine last measured. It moves — deliberately +//! and by hand, through `dr-bench record` — and its job is to catch the change +//! that is still inside the budget but has halved the headroom. Most real +//! performance rot arrives that way: never over the line, always a little +//! worse, until one day the line is crossed by a change that was not the cause. +//! +//! # Why a machine-sensitive metric skips its budget off the reference desktop +//! +//! §8 names *"the reference desktop"*, not CI, and it is right to. A container +//! with two cores cannot speak to a throughput target written for a +//! twenty-four-thread machine, and asserting it there would produce exactly +//! what `core/dr-gpu/tests/frame_budget.rs` refused to produce: *"a red suite +//! that everyone learns to ignore"*. So a metric declares whether its budget +//! is machine-sensitive. Those budgets are asserted under `--reference` and +//! reported everywhere else; the ones with orders of magnitude of headroom — +//! catalog open against two seconds — are asserted everywhere, because a +//! failure there is a real failure on any machine. +//! +//! # Why the committed file starts with no numbers in it +//! +//! Because nobody had run it yet. Writing plausible-looking figures into a +//! baseline is the one thing that would make the whole suite worthless: every +//! later comparison would be against a guess, and the first genuine regression +//! would be invisible or, worse, a fabricated improvement. `recorded` is +//! therefore `null` until somebody runs `dr-bench record --reference` on the +//! reference desktop and commits the diff. Until then the budget gate works +//! and the regression gate says so rather than pretending. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +use anyhow::{Context, Result}; +use serde::{Deserialize, Serialize}; + +use crate::fixture::Stamp; + +/// Which way is better. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum Direction { + LowerIsBetter, + HigherIsBetter, +} + +/// One measured quantity: what it is, what it must be, and what it was. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct Metric { + /// The requirement ID this speaks to, or a note saying it speaks to only + /// part of one. Free text on purpose: several of these describe a fraction + /// of a requirement, and a bare ID here would read as the whole of it. + pub requirement: String, + /// One sentence a reader of the JSON can understand without the code. + pub what: String, + pub unit: String, + pub direction: Direction, + /// Whether the budget below is a statement about a machine as much as + /// about the code. See this module's header. + pub machine_sensitive: bool, + /// The requirement's own threshold, where it has one this can check. + pub budget: Option, + /// What the last `record` measured. `null` until one has been taken. + pub recorded: Option, +} + +/// The committed file. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct Baseline { + /// Prose for whoever opens the JSON first. Kept in the file rather than + /// only in `docs/benchmarks.md`, because the person who finds this in a + /// failing CI log is not reading the docs directory at that moment. + #[serde(rename = "_readme")] + pub readme: Vec, + /// How much worse than [`Metric::recorded`] a metric may get before the + /// build fails, as a fraction. 0.15 is fifteen per cent. + pub tolerance: f64, + /// The machine the recorded figures came from, as [`machine_id`] spells + /// it. Compared before a regression is judged: drift against a different + /// machine's numbers is not a regression, it is a different machine. + pub recorded_on: Option, + /// Unix seconds. An integer rather than a formatted date because this + /// workspace has no date library and adding one for a comment would be a + /// poor trade. + pub recorded_at_unix: Option, + /// The fixture the recorded figures describe. A number measured against a + /// different workload is not comparable, and this is what says so. + pub fixture: Option, + pub metrics: BTreeMap, +} + +/// How a measurement compares. +#[derive(Debug, Clone, Copy)] +pub struct Judgement { + /// Past the requirement's own threshold. A build failure. + pub over_budget: bool, + /// Worse than the recorded baseline by more than the tolerance. A build + /// failure. + pub regressed: bool, + /// Fractional change against the baseline, positive meaning worse. + pub drift: Option, + /// A budget exists but was not asserted, because it is machine-sensitive + /// and this is not the reference desktop. + pub budget_deferred: bool, +} + +/// What a judgement is made in the light of. +#[derive(Debug, Clone, Copy)] +pub struct Judging { + /// This run declares itself the reference desktop. + pub reference: bool, + /// This run is on the machine the baseline was recorded on. + pub same_machine: bool, + pub tolerance: f64, +} + +impl Metric { + /// Judge `measured` against the budget and the baseline. + pub fn judge(&self, measured: f64, cx: Judging) -> Judgement { + let violates = |threshold: f64| match self.direction { + Direction::LowerIsBetter => measured > threshold, + Direction::HigherIsBetter => measured < threshold, + }; + let assert_budget = self.budget.is_some() && (cx.reference || !self.machine_sensitive); + + let drift = match self.recorded { + // A recorded zero would divide by nothing, and a recorded figure + // of zero is a broken record rather than a very fast one. + Some(was) if was > 0.0 => Some(match self.direction { + Direction::LowerIsBetter => (measured - was) / was, + Direction::HigherIsBetter => (was - measured) / was, + }), + _ => None, + }; + + Judgement { + over_budget: assert_budget && self.budget.is_some_and(violates), + regressed: cx.same_machine && drift.is_some_and(|d| d > cx.tolerance), + drift, + budget_deferred: self.budget.is_some() && !assert_budget, + } + } +} + +impl Baseline { + pub fn load(path: &Path) -> Result { + let text = std::fs::read_to_string(path) + .with_context(|| format!("reading the baseline at {}", path.display()))?; + serde_json::from_str(&text) + .with_context(|| format!("parsing the baseline at {}", path.display())) + } + + pub fn save(&self, path: &Path) -> Result<()> { + let mut text = serde_json::to_string_pretty(self)?; + // A trailing newline, so the file is a well-behaved text file and a + // `record` that changed nothing produces an empty diff. + text.push('\n'); + std::fs::write(path, text) + .with_context(|| format!("writing the baseline to {}", path.display())) + } + + /// Where the committed baseline lives, found the way `tools/traceability` + /// finds the repo root: by walking up from this crate's manifest until + /// `docs/requirements.md` appears. + pub fn default_path() -> Result { + let mut dir = PathBuf::from(env!("CARGO_MANIFEST_DIR")); + while !dir.join("docs/requirements.md").exists() { + if !dir.pop() { + anyhow::bail!("could not locate the repo root above this crate"); + } + } + Ok(dir.join("docs/bench-baseline.json")) + } +} + +/// How this machine is named in the baseline. +/// +/// Host name plus thread count. Not a hardware inventory — it exists to answer +/// one question, "are these numbers from here?", and to answer it the same way +/// twice on the same box. A container whose hostname changes per run therefore +/// never matches, which is the correct answer for CI: its drift is information, +/// not a verdict. +pub fn machine_id() -> String { + let host = std::fs::read_to_string("/proc/sys/kernel/hostname") + .map(|s| s.trim().to_string()) + .unwrap_or_else(|_| "unknown-host".to_string()); + let threads = std::thread::available_parallelism() + .map(|n| n.get()) + .unwrap_or(0); + format!("{host} ({threads} threads)") +} + +/// Unix seconds now, or 0 if the clock is before 1970, which it is not. +pub fn now_unix() -> i64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_secs() as i64) + .unwrap_or(0) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn metric(direction: Direction, budget: Option, recorded: Option) -> Metric { + Metric { + requirement: "NFR-TEST".into(), + what: "a test metric".into(), + unit: "ms".into(), + direction, + machine_sensitive: false, + budget, + recorded, + } + } + + fn cx(same_machine: bool) -> Judging { + Judging { + reference: true, + same_machine, + tolerance: 0.15, + } + } + + #[test] + fn a_budget_is_directional() { + // The bug this exists to prevent: judging a throughput target the way + // a latency target is judged, so a suite that got twice as slow passes + // and one that got twice as fast fails. + let latency = metric(Direction::LowerIsBetter, Some(100.0), None); + assert!(latency.judge(101.0, cx(true)).over_budget); + assert!(!latency.judge(99.0, cx(true)).over_budget); + + let throughput = metric(Direction::HigherIsBetter, Some(100.0), None); + assert!(throughput.judge(99.0, cx(true)).over_budget); + assert!(!throughput.judge(101.0, cx(true)).over_budget); + } + + #[test] + fn drift_is_positive_when_things_got_worse_whichever_way_that_is() { + let latency = metric(Direction::LowerIsBetter, None, Some(100.0)); + assert!(latency.judge(120.0, cx(true)).drift.unwrap() > 0.0); + let throughput = metric(Direction::HigherIsBetter, None, Some(100.0)); + assert!(throughput.judge(80.0, cx(true)).drift.unwrap() > 0.0); + } + + #[test] + fn drift_against_another_machine_is_reported_but_never_a_failure() { + // CI is not the reference desktop. Its numbers are worth printing and + // are not a verdict on anybody's commit. + let m = metric(Direction::LowerIsBetter, None, Some(100.0)); + let elsewhere = m.judge(400.0, cx(false)); + assert!(elsewhere.drift.unwrap() > 0.15); + assert!(!elsewhere.regressed); + assert!(m.judge(400.0, cx(true)).regressed); + } + + #[test] + fn an_unrecorded_metric_cannot_regress() { + // The state the committed file ships in. It must gate on the budget + // and stay silent about drift, rather than treating null as zero and + // declaring an infinite regression. + let m = metric(Direction::LowerIsBetter, Some(100.0), None); + let j = m.judge(50.0, cx(true)); + assert!(j.drift.is_none()); + assert!(!j.regressed); + assert!(!j.over_budget); + } + + #[test] + fn a_machine_sensitive_budget_defers_off_the_reference_desktop() { + let mut m = metric(Direction::HigherIsBetter, Some(100.0), None); + m.machine_sensitive = true; + let on_ci = Judging { + reference: false, + same_machine: false, + tolerance: 0.15, + }; + let j = m.judge(10.0, on_ci); + // A two-core runner must not fail a target written for twenty-four. + assert!(!j.over_budget); + // And the report has to say the budget was not applied, rather than + // letting a deferred budget read as a passed one. + assert!(j.budget_deferred); + // On the reference desktop the same figure is judged. + assert!(m.judge(10.0, cx(false)).over_budget); + } +} diff --git a/tools/bench/src/catalog_open.rs b/tools/bench/src/catalog_open.rs new file mode 100644 index 0000000..791cc53 --- /dev/null +++ b/tools/bench/src/catalog_open.rs @@ -0,0 +1,178 @@ +//! TRACES: NFR-P1 +//! Opening a fifty-thousand-image catalog, and what that actually involves. +//! +//! NFR-P1 says under two seconds on the reference desktop. Until this file +//! existed the number had never been measured, which made it a wish — and +//! `dr-catalog`'s `lib.rs` has carried an `NFR-P1` tag the whole time on code +//! that describes the target rather than checking it. This is the check. +//! +//! # What counts as "open" +//! +//! Not `Catalog::open` alone. That call returns before anything is on screen, +//! and a user's "the catalog opened" is the moment the grid has cells in it. +//! So the measured span is the four things the library view cannot paint +//! without: +//! +//! 1. [`Catalog::open`] — connect, migrate if needed, and **backfill**. The +//! backfill is the interesting one: `schema::backfill` runs on every open +//! and is three passes over the images table, so it is O(library) work on a +//! path whose budget is stated in absolute seconds. +//! 2. [`Catalog::count`] — the total, which is what sizes the scrollbar. +//! 3. [`Catalog::window`] — the first screenful of rows. +//! 4. [`Catalog::timeline`] — the scrubber's buckets, drawn beside the grid +//! from the first frame. +//! +//! `open` alone is reported separately anyway, because if the two ever diverge +//! sharply the fix is in a different place. +//! +//! # What this does *not* measure, said out loud +//! +//! `ui/dr-ui/src/library.rs` does not call [`Catalog::count`] or +//! [`Catalog::window`]. It issues its own SQL — `total_images_scoped`, +//! `total_images_filtered` and friends — against the same tables, with a +//! `VISIBLE` predicate and a burst-folding clause that this crate cannot see +//! without depending on the UI, which would drag Slint into a benchmark job +//! that has no display. So the number here is the **catalog crate's** open +//! path, and the application's is that plus whatever those queries cost. +//! +//! That gap is a real limit on what this file can certify, and it has a +//! falsifiable end: when the grid's queries move down into `dr-catalog` — +//! which is where SQL over catalog tables belongs — this measurement becomes +//! the whole of the application's open, and the caveat can be deleted rather +//! than argued about. +//! +//! # Cold and warm +//! +//! Both are reported. The first open in a process pays for SQLite's page cache +//! being empty and for the schema being read; the second pays for neither, and +//! is what a user gets when they close and reopen a library in the same +//! session. §4.1 asks the question directly for NFR-P8 and it is worth having +//! the answer here too. Neither figure is a genuinely cold *disk*: the fixture +//! was written by this same suite or by an earlier run of it, so the file is +//! in the OS page cache. On the reference desktop's NVMe a truly cold read of +//! a ~14 MB file is a few tens of milliseconds; on a spinning disk it is not. + +use std::path::Path; +use std::time::Instant; + +use anyhow::Result; +use dr_catalog::{Catalog, Granularity, Query}; +use dr_types::Selector; + +use crate::stats::{ms, Percentiles, Rng}; + +/// Rows fetched for the first screenful. +/// +/// A dense grid on a 4K display is around three hundred cells; four hundred is +/// that plus the prefetch margin R2 asks for. Not the whole library, because +/// FR-CAT-4 is explicit that memory must not scale with it — a benchmark that +/// asked for 50,000 rows would be measuring the requirement's violation. +const WINDOW: usize = 400; + +/// The clock the query compiler is handed. +/// +/// Fixed rather than read from the system, so a rolling date filter would +/// compile to the same SQL on every run. The unfiltered query does not consult +/// it at all; this is here so that adding a dated row later does not silently +/// make the suite time-dependent. +const NOW: i64 = 2_000_000_000; + +/// What one measured open produced. +pub struct Open { + /// Open, count, first window, timeline — the whole span, first time. + pub cold_ms: f64, + /// The same four calls on a second connection in the same process. + pub warm_ms: f64, + /// [`Catalog::open`] on its own, out of the cold span. + pub open_only_ms: f64, + /// How many images the count found. Reported so a fixture that failed to + /// populate cannot masquerade as a very fast open. + pub images: usize, + /// Timeline buckets at monthly granularity. + pub buckets: usize, + /// One window fetched at a random offset — the scroll, minus the drawing. + pub window_ms: Percentiles, + /// Count plus first window under a rating filter, which compiles to a + /// correlated subquery over `versions` (see `query::default_version_scalar`). + pub filtered_ms: f64, +} + +/// Measure an open of the catalog at `path`, then `windows` random windows. +pub fn measure(path: &Path, windows: usize) -> Result { + let q = Query::default(); + + let started = Instant::now(); + let catalog = open(path)?; + let open_only_ms = ms(started.elapsed()); + let images = catalog.count(&q, NOW)?; + let rows = catalog.window(&q, 0..WINDOW, NOW)?; + let buckets = catalog.timeline(&q, Granularity::Month, NOW)?.len(); + let cold_ms = ms(started.elapsed()); + + // A catalog that returned nothing would post an excellent time. Checked + // rather than trusted, because the failure mode is a *fast* wrong answer. + anyhow::ensure!( + !rows.is_empty() && images > 0 && buckets > 0, + "the fixture catalog answered with {images} images, {} rows and {buckets} buckets — \ + the measurement below would be meaningless", + rows.len() + ); + drop(catalog); + + let started = Instant::now(); + let catalog = open(path)?; + let _ = catalog.count(&q, NOW)?; + let _ = catalog.window(&q, 0..WINDOW, NOW)?; + let _ = catalog.timeline(&q, Granularity::Month, NOW)?; + let warm_ms = ms(started.elapsed()); + + // The scroll. Offsets are drawn from a fixed seed rather than swept in + // order, because a sequential sweep would be answered increasingly out of + // SQLite's own cache and would flatter the deep end of the library — which + // is exactly the end a person reaches by dragging the scrollbar. + let mut rng = Rng::new(0x5C_20_11); + let span = images.saturating_sub(WINDOW).max(1) as u64; + // Discarded: the first window of a new connection compiles the statement + // and faults in the b-tree's upper levels, and neither recurs while + // scrolling. + for _ in 0..4 { + let start = rng.below(span) as usize; + let _ = catalog.window(&q, start..start + WINDOW, NOW)?; + } + let mut samples = Vec::with_capacity(windows); + for _ in 0..windows { + let start = rng.below(span) as usize; + let t = Instant::now(); + let rows = catalog.window(&q, start..start + WINDOW, NOW)?; + samples.push(ms(t.elapsed())); + debug_assert!(!rows.is_empty()); + } + + // A filter that has to reach the default version for every candidate row. + // Cheap to add and the one query shape in the grid that is not a scan of + // `images` alone, so a regression in it would otherwise show up first as a + // user complaint. + let rated = Query { + filter: Selector::Rating { min: 2 }, + ..Query::default() + }; + let t = Instant::now(); + let _ = catalog.count(&rated, NOW)?; + let _ = catalog.window(&rated, 0..WINDOW, NOW)?; + let filtered_ms = ms(t.elapsed()); + + Ok(Open { + cold_ms, + warm_ms, + open_only_ms, + images, + buckets, + window_ms: Percentiles::of(samples), + filtered_ms, + }) +} + +fn open(path: &Path) -> Result { + Catalog::open(path) + .map_err(|e| anyhow::anyhow!("opening the fixture catalog at {}: {e}", path.display())) +} diff --git a/tools/bench/src/exporting.rs b/tools/bench/src/exporting.rs new file mode 100644 index 0000000..16dcc6e --- /dev/null +++ b/tools/bench/src/exporting.rs @@ -0,0 +1,139 @@ +//! The half of a 24 MP export that needs no GPU. +//! +//! # This cannot certify NFR-P7, and is not tagged as though it could +//! +//! NFR-P7 is "full-resolution export (24 MP, **full chain**) < 2 s". The full +//! chain is decode, demosaic, a GPU render at full resolution, a read-back, +//! and then everything `dr-export` does — resize, output sharpening, encode. +//! Only the last three of those run without an adapter, and the CI runner has +//! none. So what is measured here is the encode half, and no requirement tag +//! anywhere in this crate names NFR-P7. +//! +//! (Written without the tag's own spelling on purpose. `tools/traceability` +//! matches the marker anywhere on a line and parses the identifier after it, so +//! a sentence saying "there is no tag for NFR-P7" would *be* a tag for NFR-P7 — +//! a disclaimer that made itself false.) +//! +//! That is a deliberate refusal rather than an oversight. `CONTRIBUTING.md` +//! asks that a requirement be closed by a test that would fail if the +//! behaviour were removed, and `docs/code-health.md` CH-4 records what the +//! coverage figure looks like when tags are hung on plumbing instead. A tag +//! here would say the export budget is checked; the GPU half of it would still +//! be unchecked. +//! +//! # What the number is still good for +//! +//! It is a **one-sided** gate, and that is worth having. The encode half is a +//! lower bound on the whole: if resizing, sharpening and encoding 24 MP alone +//! take longer than two seconds, NFR-P7 is violated no matter how fast the +//! render is. So the budget in `docs/bench-baseline.json` is the requirement's +//! own 2000 ms, and exceeding it fails the build honestly. Coming in under it +//! proves nothing about the requirement, and the report says so rather than +//! printing a tick. +//! +//! # Two rows +//! +//! `Original` is the one the budget is judged on: it is the archival export, +//! the largest encode, and the case FR-EXP-9 is about. `LongEdge(2048)` is the +//! ordinary web export, where the resample does real work and the encode does +//! very little — it is reported because a regression in `size::resample` would +//! be invisible in the first row, where source and target dimensions are equal. + +use std::time::Instant; + +use anyhow::Result; +use dr_export::{export, Frame}; +use dr_types::{ExportFormat, ExportSettings, OutputSharpening, SizingMode}; + +use crate::fixture::plausible_frame; +use crate::stats::{ms, Percentiles}; + +/// The frame every row exports. +/// +/// 6000 × 4000 is 24.0 MP — a full-frame body, and the exact figure NFR-P7 +/// names. As RGBA8 it is 96 MB, and the export path holds a resized copy and a +/// sharpened copy alongside it, so a run needs roughly 300 MB of headroom. +/// Worth knowing before a small runner reports this as a mysterious kill. +pub const SOURCE: (u32, u32) = (6000, 4000); + +/// Measured exports per row. Not a hundred: one 24 MP encode is most of a +/// second, and a hundred of them would be a two-minute CI step to establish +/// what five establish. Nearest-rank p99 of five is the worst of the five, +/// which for a row this expensive is the honest reading anyway. +const RUNS: usize = 5; + +/// A row of the export table. +pub struct EncodeRun { + /// The name this row carries in `docs/bench-baseline.json`. + pub key: &'static str, + pub label: &'static str, + pub width: u32, + pub height: u32, + /// Encoded file size, so a row that silently stopped compressing is + /// visible as well as a row that got slow. + pub bytes: usize, + pub times: Percentiles, +} + +/// Export the same 24 MP frame at each sizing, timing `dr_export::export`. +pub fn measure() -> Result> { + let (w, h) = SOURCE; + // Detail at every scale, for the same reason the thumbnail fixture has it: + // a flat frame compresses to almost nothing and would make the encoder + // look several times faster than any photograph makes it. + let frame = Frame::new(w, h, plausible_frame(w, h, 0)) + .map_err(|e| anyhow::anyhow!("building the 24 MP bench frame: {e}"))?; + + let sizings: [(&'static str, &'static str, SizingMode); 2] = [ + ("export_24mp_original_ms", "original", SizingMode::Original), + ( + "export_24mp_long_edge_2048_ms", + "long edge 2048", + SizingMode::LongEdge(2048), + ), + ]; + + let mut rows = Vec::with_capacity(sizings.len()); + for (key, label, sizing) in sizings { + let settings = ExportSettings { + format: ExportFormat::Jpeg, + // 90 is the default and what a photographer would not need to + // change; quality moves encode time, so it belongs in the record. + quality: 90, + sizing, + sharpening: OutputSharpening::Screen, + ..Default::default() + }; + + // Discarded. The first export of a process grows the allocator to hold + // three 24 MP buffers, which is a cost paid once and not per file in + // the batch export FR-EXP-7 describes. + let warm = run_once(&frame, &settings)?; + + let mut samples = Vec::with_capacity(RUNS); + let mut last = warm; + for _ in 0..RUNS { + let started = Instant::now(); + last = run_once(&frame, &settings)?; + samples.push(ms(started.elapsed())); + } + + rows.push(EncodeRun { + key, + label, + width: last.0, + height: last.1, + bytes: last.2, + times: Percentiles::of(samples), + }); + } + + Ok(rows) +} + +/// One export, returning what it produced rather than the pixels. +fn run_once(frame: &Frame, settings: &ExportSettings) -> Result<(u32, u32, usize)> { + let encoded = export(frame, settings, "bench.jpg".to_string(), None) + .map_err(|e| anyhow::anyhow!("exporting the bench frame: {e}"))?; + Ok((encoded.width, encoded.height, encoded.bytes.len())) +} diff --git a/tools/bench/src/fixture.rs b/tools/bench/src/fixture.rs new file mode 100644 index 0000000..fa292d3 --- /dev/null +++ b/tools/bench/src/fixture.rs @@ -0,0 +1,448 @@ +//! The synthetic 50k catalog, and the handful of real files it points at. +//! +//! `docs/requirements.md` §8 asks for "an automated benchmark suite against a +//! synthetic 50k catalog". The hard part of that sentence is *50k*: a real +//! library of that size is several terabytes and cannot live in a repository, +//! in a CI cache, or on a laptop that also has to compile the thing. +//! +//! # The trick, and what it costs +//! +//! Rows are cheap and pixels are not. So this builds **fifty thousand catalog +//! rows** over a **pool of a dozen real image files**, each referenced by +//! several thousand of them. Everything the catalog half of the suite measures +//! — opening, counting, windowing, bucketing a timeline — touches only rows, +//! and is therefore exact. Everything the pixel half measures — decode, +//! downscale, orient, encode — touches one file at a time and does not care +//! how many rows point at it. The fixture is ~14 MB on disk instead of ~2 TB +//! and neither half is flattered by that. +//! +//! What it *does* cost is stated rather than hidden: the file pool is small +//! enough to sit in the OS page cache, so [`crate::thumbnails`] measures CPU +//! throughput with the read already paid for. That is the right thing to +//! measure for NFR-P3 — the target is written about the embedded preview path, +//! not about a disk — but it is not a claim about a cold library on spinning +//! rust, and the harness does not make one. +//! +//! # Reproducible from a seed +//! +//! Every value comes from [`Rng`], seeded once. Two machines running the same +//! seed build byte-comparable catalogs, which is the property that lets a +//! number measured on the reference desktop be compared with a number measured +//! anywhere else. [`Stamp`] records what a directory was built from, so a +//! fixture is reused when it matches and rebuilt when it does not — including +//! when `dr-catalog`'s schema version moves, since a catalog built by an older +//! build would otherwise be measured through a migration that a user's would +//! not run. +//! +//! # The sources are generated, not committed +//! +//! No photograph in this repository is licensed for redistribution, and a +//! dozen camera previews would be megabytes of binary in git for ever. So the +//! pool is synthesised: a coarse gradient with a fine dither on top, which is +//! the same shape `core/dr-gpu/examples/frame_budget.rs` synthesises its source +//! from and for the same reason. A flat frame lets the memory system serve +//! every sample from one cache line, which flatters a box filter by an amount +//! that has nothing to do with photographs; pure noise defeats the JPEG +//! encoder's entropy coder in the other direction and would make the encode +//! half of a thumbnail look worse than any real image ever does. + +use std::path::{Path, PathBuf}; + +use anyhow::{Context, Result}; +use dr_catalog::Catalog; +use serde::{Deserialize, Serialize}; + +use crate::stats::Rng; + +/// How many distinct image files the pool holds. +/// +/// Twelve rather than one, so that a decode measured over a batch is not one +/// file's quirks repeated — a single frame that happened to compress unusually +/// well would set the whole number — and rather than fifty thousand, so the +/// fixture stays a directory a person can look at. +pub const SOURCE_POOL: usize = 12; + +/// The size of one pooled file, in pixels. +/// +/// 1620×1080 is not a round number: it is what `core/dr-decode/src/preview.rs` +/// records a Canon CR2 carrying in IFD2, and the embedded preview is what +/// NFR-P3 names. A camera JPEG is 24 MP and a camera *preview* is about this, +/// so measuring the preview path against a 24 MP file would measure something +/// the sweep never does. +pub const PREVIEW: (u32, u32) = (1620, 1080); + +/// Folders the rows are spread across. +/// +/// Spread evenly and without regard to capture date, because what a folder +/// count decides is the cost of the folder filter's `IN (SELECT …)` and the +/// size of the `folders` table — not which image is in which. +const FOLDERS: usize = 400; + +/// The earliest capture time in the fixture: 13 December 2015, UTC. +/// +/// Fixed rather than relative to the clock. A library whose dates moved with +/// the calendar would make `timeline` bucket differently from one month to the +/// next, and a benchmark that measures a different query each time it runs is +/// not measuring a regression. +const EPOCH: i64 = 1_450_000_000; + +/// The span capture times are drawn from: twelve years. +/// +/// Long enough that the timeline query has real structure to bucket — at +/// monthly granularity that is ~144 buckets, which is the shape the scrubber +/// actually draws — and not so long that a year holds too few frames to look +/// like a library. +const SPAN: i64 = 12 * 365 * 86_400; + +/// Bodies and lenses, for the columns the camera and lens filters read. +const CAMERAS: [&str; 6] = [ + "Canon EOS R5", + "Nikon Z 7II", + "Sony ILCE-7RM5", + "Fujifilm X-T5", + "Panasonic DC-S5M2", + "OM SYSTEM OM-1", +]; + +const LENSES: [&str; 6] = [ + "RF24-70mm F2.8 L IS USM", + "NIKKOR Z 50mm f/1.8 S", + "FE 85mm F1.4 GM", + "XF16-55mmF2.8 R LM WR", + "LUMIX S 20-60mm F3.5-5.6", + "M.Zuiko Digital ED 12-40mm F2.8", +]; + +/// What a fixture directory was built from. +/// +/// Written beside the catalog and compared on every run. A mismatch rebuilds: +/// silently reusing a fixture built from a different seed, a different row +/// count or an older schema would compare two numbers that describe two +/// different workloads, which is worse than having no number at all. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct Stamp { + /// Bumped by hand whenever anything in this file changes what gets built. + /// The seed cannot carry that: the same seed through different generation + /// code produces a different library. + pub generator: u32, + pub seed: u64, + pub images: usize, + pub sources: usize, + pub folders: usize, + pub preview_width: u32, + pub preview_height: u32, + /// `dr-catalog`'s schema version at build time. + pub schema_version: i64, +} + +/// Bump on any change to what [`build`] writes. +const GENERATOR: u32 = 1; + +/// A built fixture on disk. +pub struct Fixture { + pub dir: PathBuf, + pub catalog: PathBuf, + pub thumbs: PathBuf, + pub sources: Vec, + pub stamp: Stamp, + /// The catalog file's size, reported because it is the thing an open has + /// to read and because it is the honest denominator for "is 2 s a lot". + pub catalog_bytes: u64, +} + +/// Where a fixture lives by default. +/// +/// The temporary directory rather than `target/`, for two reasons. It survives +/// `cargo clean`, so a fixture is built once per machine rather than once per +/// clean; and it is not inside anything CI caches, so a 14 MB catalog is not +/// uploaded and downloaded on every push to save the two seconds it takes to +/// generate. `DR_BENCH_DIR` overrides it. +pub fn default_dir() -> PathBuf { + match std::env::var_os("DR_BENCH_DIR") { + Some(dir) => PathBuf::from(dir), + None => std::env::temp_dir().join("darkroom-bench"), + } +} + +/// Build the fixture under `dir`, or confirm the one already there. +/// +/// Returns whether it had to be built, so the caller can say so: a run that +/// includes fixture generation has a warm page cache for the catalog file it +/// is about to open, and a reader comparing two numbers deserves to know which +/// of them was measured that way. +pub fn build(dir: &Path, seed: u64, images: usize) -> Result<(Fixture, bool)> { + let stamp = Stamp { + generator: GENERATOR, + seed, + images, + sources: SOURCE_POOL, + folders: FOLDERS, + preview_width: PREVIEW.0, + preview_height: PREVIEW.1, + schema_version: dr_catalog::schema::SCHEMA_VERSION, + }; + + let catalog = dir.join("catalog.sqlite"); + let stamp_path = dir.join("stamp.json"); + let sources: Vec = (0..SOURCE_POOL) + .map(|i| dir.join("sources").join(format!("preview-{i:02}.jpg"))) + .collect(); + + let usable = matches_stamp(&stamp_path, &stamp) + && catalog.is_file() + && sources.iter().all(|p| p.is_file()); + + if !usable { + std::fs::create_dir_all(dir.join("sources")) + .with_context(|| format!("creating the fixture directory {}", dir.display()))?; + // The stamp goes last. A build interrupted halfway leaves no stamp, so + // the next run rebuilds rather than measuring a truncated catalog. + let _ = std::fs::remove_file(&stamp_path); + write_sources(&sources, seed)?; + write_catalog(&catalog, seed, images)?; + std::fs::write(&stamp_path, serde_json::to_vec_pretty(&stamp)?) + .with_context(|| format!("writing {}", stamp_path.display()))?; + } + + let catalog_bytes = std::fs::metadata(&catalog) + .with_context(|| format!("stat {}", catalog.display()))? + .len(); + + Ok(( + Fixture { + dir: dir.to_path_buf(), + catalog, + thumbs: dir.join("thumbs"), + sources, + stamp, + catalog_bytes, + }, + !usable, + )) +} + +fn matches_stamp(path: &Path, want: &Stamp) -> bool { + let Ok(text) = std::fs::read_to_string(path) else { + return false; + }; + matches!(serde_json::from_str::(&text), Ok(have) if have == *want) +} + +// --------------------------------------------------------------------------- +// The file pool +// --------------------------------------------------------------------------- + +/// Write the pool of JPEGs the pixel half decodes. +/// +/// Encoded through [`dr_thumbs::encode_rgba`] rather than a second encoder +/// call of this crate's own. That is the quality the store already uses (82), +/// which is a little below what a camera writes its previews at, and it is one +/// fewer place for an encoder setting to drift. Stated because it is visible +/// in the result: a slightly softer source decodes marginally faster than a +/// camera's own preview would. +fn write_sources(paths: &[PathBuf], seed: u64) -> Result<()> { + let (w, h) = PREVIEW; + for (i, path) in paths.iter().enumerate() { + let rgba = plausible_frame(w, h, seed ^ (i as u64)); + let jpeg = dr_thumbs::encode_rgba(w, h, &rgba) + .map_err(|e| anyhow::anyhow!("encoding the fixture source {}: {e}", path.display()))?; + std::fs::write(path, jpeg).with_context(|| format!("writing {}", path.display()))?; + } + Ok(()) +} + +/// RGBA with detail at every scale: a coarse gradient plus a fine dither. +/// +/// See this module's header for why neither a flat frame nor pure noise would +/// do. `salt` moves the gradient and the dither together so the twelve files +/// differ from one another rather than being twelve copies with different +/// names — a JPEG encoder that saw the same image twelve times would have the +/// same cache behaviour every time, which a library does not. +pub fn plausible_frame(w: u32, h: u32, salt: u64) -> Vec { + let mut rgba = vec![0u8; (w as usize) * (h as usize) * 4]; + let bias = (salt % 97) as u32; + for y in 0..h as usize { + let row = y * (w as usize) * 4; + for x in 0..w as usize { + // A cheap integer hash, so neighbouring pixels differ and the + // encoder has real high-frequency content to spend bits on. + let n = (x.wrapping_mul(2_654_435_761) ^ y.wrapping_mul(1_640_531_527)) >> 13; + let dither = (n & 0x1f) as u32; + let gx = (x * 200 / (w as usize).max(1)) as u32; + let gy = (y * 55 / (h as usize).max(1)) as u32; + let px = &mut rgba[row + x * 4..row + x * 4 + 4]; + px[0] = (30 + bias + gx + dither).min(255) as u8; + px[1] = (40 + gy + dither).min(255) as u8; + px[2] = (60 + gx / 2 + gy + dither).min(255) as u8; + px[3] = 255; + } + } + rgba +} + +// --------------------------------------------------------------------------- +// The catalog +// --------------------------------------------------------------------------- + +/// Write a catalog holding `images` rows, plus a default version for each. +/// +/// The default versions are not decoration. `Catalog::open` backfills them for +/// any image that lacks one (see `schema::backfill`), so a fixture without +/// them would charge every measured open for fifty thousand inserts once and +/// nothing thereafter — a first number that bore no relation to the second, +/// and a benchmark whose result depended on whether it had been run before. +fn write_catalog(path: &Path, seed: u64, images: usize) -> Result<()> { + for suffix in ["", "-wal", "-shm"] { + let mut p = path.as_os_str().to_os_string(); + p.push(suffix); + let _ = std::fs::remove_file(PathBuf::from(p)); + } + + let catalog = Catalog::open(path) + .map_err(|e| anyhow::anyhow!("creating the fixture catalog at {}: {e}", path.display()))?; + let conn = catalog.connection(); + let mut rng = Rng::new(seed); + + conn.execute_batch("BEGIN")?; + + conn.execute( + "INSERT INTO roots(id, kind, label, last_seen, scan_generation) + VALUES (1, 'local', '/library', ?1, 1)", + rusqlite::params![EPOCH], + )?; + + { + let mut folder = conn.prepare( + "INSERT INTO folders(id, root_id, parent_id, path, mtime, entry_count, + scanned_generation) + VALUES (?1, 1, NULL, ?2, ?3, ?4, 1)", + )?; + for f in 0..FOLDERS { + let id = f as i64 + 1; + let folder_path = format!("/library/{:04}/{:02}", 2016 + f / 12, f % 12 + 1); + let mtime = EPOCH + f as i64 * 86_400; + let entries = (images / FOLDERS.max(1)) as i64; + folder.execute(rusqlite::params![id, folder_path, mtime, entries])?; + } + } + + { + let mut image = conn.prepare( + "INSERT INTO images(id, root_id, folder_id, source_ref, format, w, h, + captured_at, captured_offset, camera, lens, iso, + aperture, shutter, availability, file_size, + file_mtime, metadata_state, added_at) + VALUES (?1, 1, ?2, ?3, 'CR3', ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11, ?12, + ?13, ?14, ?15, ?16, ?17)", + )?; + let mut version = conn.prepare( + "INSERT INTO versions(id, image_id, uuid, name, is_default, rating, + label, flag) + VALUES (?1, ?1, ?2, 'Original', 1, ?3, ?4, ?5)", + )?; + + // Every value is bound to a local before it reaches `params!`. Not + // style: the macro takes a reference to each argument, and an + // expression like `TABLE[rng.below(n) as usize]` inside it borrows an + // element of a temporary array while `rng` is also being borrowed + // mutably. Locals make the evaluation order and the lifetimes obvious. + const OFFSETS: [i64; 5] = [0, 60, 120, -300, 540]; + const ISOS: [i64; 7] = [100, 200, 400, 800, 1600, 3200, 6400]; + const APERTURES: [f64; 6] = [1.4, 1.8, 2.8, 4.0, 5.6, 8.0]; + const SHUTTERS: [f64; 6] = [0.004, 0.008, 0.0167, 0.005, 0.002, 0.5]; + // Mostly metadata-only, as a large library on a laptop is: some + // previewed, a few with the original present. + const AVAILABILITY: [i64; 6] = [0, 0, 0, 1, 1, 2]; + + for i in 0..images { + let id = i as i64 + 1; + let bucket = i % FOLDERS; + let folder = bucket as i64 + 1; + let source_ref = format!( + "/library/{:04}/{:02}/IMG_{id:05}.CR3", + 2016 + bucket / 12, + bucket % 12 + 1 + ); + let captured = EPOCH + rng.below(SPAN as u64) as i64; + // A quarter of the library shot in portrait, which is what makes + // the thumbnail path's orientation permutation a real cost rather + // than a branch that is never taken. + let (w, h) = if i % 4 == 3 { + (4000i64, 6000i64) + } else { + (6000i64, 4000i64) + }; + // Two per cent still awaiting full EXIF — a library is never + // entirely finished being read, and the grid has to render that + // state (`metadata_state` 1). + let state: i64 = if rng.below(50) == 0 { 1 } else { 2 }; + let camera = CAMERAS[rng.below(CAMERAS.len() as u64) as usize]; + let lens = LENSES[rng.below(LENSES.len() as u64) as usize]; + // Minutes east of UTC: a library shot in a handful of places. + let offset = OFFSETS[rng.below(OFFSETS.len() as u64) as usize]; + let iso = ISOS[rng.below(ISOS.len() as u64) as usize]; + let aperture = APERTURES[rng.below(APERTURES.len() as u64) as usize]; + let shutter = SHUTTERS[rng.below(SHUTTERS.len() as u64) as usize]; + let availability = AVAILABILITY[rng.below(AVAILABILITY.len() as u64) as usize]; + let file_size = 20_000_000i64 + rng.below(30_000_000) as i64; + let file_mtime = captured + 60; + let added_at = captured + 3600; + + image.execute(rusqlite::params![ + id, + folder, + source_ref, + w, + h, + captured, + offset, + camera, + lens, + iso, + aperture, + shutter, + availability, + file_size, + file_mtime, + state, + added_at, + ])?; + + // Ratings skewed the way a culled library is: most unrated, a few + // picks, fewer still at five stars. + let rating: i64 = match rng.below(100) { + 0..=69 => 0, + 70..=84 => 1, + 85..=93 => 2, + 94..=97 => 3, + 98 => 4, + _ => 5, + }; + let flag: i64 = match rng.below(100) { + 0..=79 => 0, + 80..=94 => 1, + _ => 2, + }; + let label: Option = match rng.below(100) { + 0..=89 => None, + n => Some((n % 5) as i64 + 1), + }; + let a = rng.next_u64(); + let b = rng.next_u64(); + let uuid = format!("{a:016x}{b:016x}"); + version.execute(rusqlite::params![id, uuid, rating, label, flag])?; + } + } + + conn.execute_batch("COMMIT")?; + + // Deliberately no `ANALYZE`. The application never runs one, so a fixture + // that did would be measuring a query plan no user's catalog gets — and a + // plan chosen from statistics is exactly the sort of thing that would make + // the benchmark faster than the product. + + // Dropping the connection checkpoints the WAL, so the file the next open + // reads is the whole catalog rather than a stub plus a journal. + drop(catalog); + Ok(()) +} diff --git a/tools/bench/src/main.rs b/tools/bench/src/main.rs new file mode 100644 index 0000000..ef4e14a --- /dev/null +++ b/tools/bench/src/main.rs @@ -0,0 +1,539 @@ +//! The benchmark suite `docs/requirements.md` §8 has been promising. +//! +//! §8 says performance is verified by *"an automated benchmark suite against a +//! synthetic 50k catalog, run per-commit … A regression beyond stated +//! tolerance fails the build."* Until this crate there was none: no `benches/`, +//! no `[[bench]]`, no criterion, no fixture. Ten performance requirements could +//! therefore be neither passed nor failed, and five of them carried a +//! requirement tag anyway. +//! +//! ```sh +//! cargo run --release -p dr-bench -- check # measure and gate +//! cargo run --release -p dr-bench -- check --reference # …on the reference desktop +//! cargo run --release -p dr-bench -- record --reference # rewrite the baseline +//! ``` +//! +//! **Release, always.** The workspace builds its own crates at `opt-level = 0` +//! in dev (see the root `Cargo.toml`), and every number here is dominated by +//! this workspace's own code — the JPEG decode, the box filter, the resample, +//! the sharpen. A debug run measures rustc's shadow, exactly as +//! `core/dr-gpu/examples/frame_budget.rs` warns for the same reason. The +//! harness says so at the top of every report rather than trusting anyone to +//! remember. +//! +//! # What this covers, and what it deliberately does not +//! +//! Covered, with a gate: +//! +//! - **NFR-P1** and R2's catalog clause — opening a 50k catalog and painting +//! the first grid ([`catalog_open`]). +//! - **NFR-P3** — thumbnail throughput on the embedded preview path +//! ([`thumbnails`]). +//! +//! Measured, reported, and honestly *not* tagged, because only part of the +//! requirement is in reach without a GPU or a running UI: +//! +//! - **NFR-P7** — the encode half of a 24 MP export ([`exporting`]). A +//! one-sided gate: it can fail the requirement, it cannot pass it. +//! - **NFR-P8** — the catalog layer's share of idle RSS ([`memory`]), together +//! with the answer to the question §4.1 asks about GPU memory. +//! +//! Out of scope entirely, and left to say so rather than faked: NFR-P2, P4, +//! P5, P6, P9, P10, P11, P12, P13, P14 and P15. Every one of them needs a +//! frame-timing probe inside a running Slint application, a GPU adapter, or +//! both. The GPU half of the suite that *does* exist is +//! `core/dr-gpu/tests/frame_budget.rs`, which asserts FR-DSP-3 and skips itself +//! where there is no adapter; `.gitea/workflows/benchmark.yml` runs it as its +//! own job for exactly that reason. +//! +//! # Exit codes +//! +//! `0` everything inside its budget and its tolerance; `1` a gate failed; `2` +//! the harness itself could not run. Distinguished because a CI log that says +//! "failed" should not leave anyone guessing whether the code got slower or the +//! fixture would not build. + +mod baseline; +mod catalog_open; +mod exporting; +mod fixture; +mod memory; +mod stats; +mod thumbnails; + +use std::collections::BTreeMap; +use std::path::PathBuf; + +use anyhow::Result; + +use baseline::{Baseline, Judging}; + +/// The seed the committed baseline describes. +/// +/// Changing it invalidates every recorded figure, because it changes the +/// library being measured. That is why it is a constant here and a field in +/// the stamp rather than something a flag quietly varies. +const SEED: u64 = 20_260_829; + +/// Rows in the synthetic library. §8 says 50k; this is that. +const IMAGES: usize = 50_000; + +/// Thumbnails produced for the throughput row. +/// +/// Twelve hundred rather than fifty thousand. At the target rate the whole +/// library is eight minutes of CI, and a rate measured over 1,200 images is +/// the same rate — the sweep has no state that changes after the first chunk, +/// which the per-image percentile alongside it is there to demonstrate. +const THUMBNAILS: usize = 1_200; + +/// Windows fetched for the scroll row. +/// +/// A hundred, so nearest-rank puts the 99th percentile on the second-worst — +/// the same reading `core/dr-gpu/examples/frame_budget.rs` takes of a hundred +/// frames, and the reason it takes it: one bad one in a hundred is one too +/// many, and a single scheduler hiccup on an unrelated process should not +/// decide the verdict on its own. +const WINDOWS: usize = 100; + +fn main() { + env_logger::init(); + match run() { + Ok(true) => {} + Ok(false) => std::process::exit(1), + Err(e) => { + eprintln!("dr-bench: {e:#}"); + std::process::exit(2); + } + } +} + +/// Returns whether every gate passed. +fn run() -> Result { + let mut args = std::env::args().skip(1); + let command = args.next().unwrap_or_else(|| "help".to_string()); + + // The probe takes a path and nothing else — see `memory` for why it is a + // separate process rather than a function call. + if command == "memory-probe" { + let path = args + .next() + .ok_or_else(|| anyhow::anyhow!("memory-probe needs a catalog path"))?; + memory::probe(&PathBuf::from(path))?; + return Ok(true); + } + + let mut reference = false; + let mut fixture_dir = fixture::default_dir(); + let mut baseline_path = Baseline::default_path()?; + let mut lanes = thumbnails::default_lanes(); + let mut thumbnail_count = THUMBNAILS; + + while let Some(flag) = args.next() { + match flag.as_str() { + "--reference" => reference = true, + "--fixture" => fixture_dir = PathBuf::from(expect_value(&mut args, "--fixture")?), + "--baseline" => baseline_path = PathBuf::from(expect_value(&mut args, "--baseline")?), + "--lanes" => lanes = expect_value(&mut args, "--lanes")?.parse()?, + "--thumbnails" => thumbnail_count = expect_value(&mut args, "--thumbnails")?.parse()?, + other => anyhow::bail!("unknown flag {other}; try `dr-bench help`"), + } + } + + match command.as_str() { + "run" => measure_and_report( + &fixture_dir, + &baseline_path, + reference, + lanes, + thumbnail_count, + Mode::Report, + ), + "check" => measure_and_report( + &fixture_dir, + &baseline_path, + reference, + lanes, + thumbnail_count, + Mode::Gate, + ), + "record" => measure_and_report( + &fixture_dir, + &baseline_path, + reference, + lanes, + thumbnail_count, + Mode::Record, + ), + "help" | "--help" | "-h" => { + print_help(); + Ok(true) + } + other => anyhow::bail!("unknown command {other}; try `dr-bench help`"), + } +} + +fn expect_value(args: &mut impl Iterator, flag: &str) -> Result { + args.next() + .ok_or_else(|| anyhow::anyhow!("{flag} needs a value")) +} + +/// What a run does with what it measured. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Mode { + /// Print, judge nothing. + Report, + /// Print and fail the build on a violated budget or a regression. + Gate, + /// Print and rewrite the committed baseline from what was measured. + Record, +} + +fn print_help() { + println!( + "\ +dr-bench — DarkRoom's performance suite (docs/requirements.md §8) + + run measure and print, judging nothing + check measure, print, and exit 1 on a violated budget or a regression + record measure and rewrite docs/bench-baseline.json from the result + +Flags: + --reference this machine is the reference desktop, so machine-sensitive + budgets are asserted rather than reported + --fixture where the synthetic 50k catalog lives (or $DR_BENCH_DIR) + --baseline the committed numbers (default docs/bench-baseline.json) + --lanes sweep lanes for the thumbnail row (default: CPU threads) + --thumbnails images in the thumbnail row (default 1200) + +Build it in release. A debug build measures the compiler, not the pipeline — +see the module documentation and core/dr-gpu/examples/frame_budget.rs." + ); +} + +// --------------------------------------------------------------------------- +// The run +// --------------------------------------------------------------------------- + +fn measure_and_report( + fixture_dir: &std::path::Path, + baseline_path: &std::path::Path, + reference: bool, + lanes: usize, + thumbnail_count: usize, + mode: Mode, +) -> Result { + let machine = baseline::machine_id(); + println!("DarkRoom benchmark suite — the half that needs no GPU"); + println!("machine {machine}"); + if cfg!(debug_assertions) { + println!( + "profile DEBUG — every figure below is several times worse than the \ + product's. Rerun with --release." + ); + } else { + println!("profile release"); + } + + let (fx, built) = fixture::build(fixture_dir, SEED, IMAGES)?; + println!( + "fixture {} images, {} sources at {}x{}, seed {}, catalog {:.1} MB{}", + fx.stamp.images, + fx.stamp.sources, + fx.stamp.preview_width, + fx.stamp.preview_height, + fx.stamp.seed, + fx.catalog_bytes as f64 / 1e6, + if built { " (built just now)" } else { "" } + ); + println!(" {}", fx.dir.display()); + println!(); + + // Memory first, and in its own process. See `memory` for why: measuring + // RSS after the thumbnail sweep would report the sweep's high-water mark + // wearing the catalog's name. + let rss = match memory::in_a_fresh_process(&fx.catalog) { + Ok(rss) => Some(rss), + Err(e) => { + println!("memory unavailable: {e}"); + None + } + }; + + let open = catalog_open::measure(&fx.catalog, WINDOWS)?; + let thumbs = thumbnails::measure(&fx.sources, &fx.thumbs, thumbnail_count, lanes)?; + let exports = exporting::measure()?; + + print_details(&open, &thumbs, &exports, rss); + + let values = collect(&open, &thumbs, &exports, rss); + let mut base = Baseline::load(baseline_path)?; + + if mode == Mode::Record { + if !reference { + println!( + "note recording from a machine that has not declared itself the \ + reference desktop.\n §8 records the reference desktop's numbers; \ + this overwrites them." + ); + } + for (key, value) in &values { + if let Some(metric) = base.metrics.get_mut(key) { + metric.recorded = Some(*value); + } + } + base.recorded_on = Some(machine); + base.recorded_at_unix = Some(baseline::now_unix()); + base.fixture = Some(fx.stamp.clone()); + base.save(baseline_path)?; + println!("\nrecorded {}", baseline_path.display()); + return Ok(true); + } + + let same_machine = base.recorded_on.as_deref() == Some(machine.as_str()); + let comparable = base.fixture.as_ref().is_some_and(|f| *f == fx.stamp); + let cx = Judging { + reference, + same_machine: same_machine && comparable, + tolerance: base.tolerance, + }; + let failures = print_verdict(&base, &values, cx, &machine, comparable); + + if failures.is_empty() { + return Ok(true); + } + println!(); + for line in &failures { + println!("FAIL {line}"); + } + println!(); + println!( + " {} gate(s) failed. docs/benchmarks.md says what each metric measures and\n \ + docs/bench-baseline.json holds the numbers these used to be.", + failures.len() + ); + Ok(mode != Mode::Gate) +} + +/// The metric keys, and the measurement each one reads. +/// +/// One place, so the report, the gate and the recorded file cannot disagree +/// about what `catalog_open_ms` means. +fn collect( + open: &catalog_open::Open, + thumbs: &thumbnails::Throughput, + exports: &[exporting::EncodeRun], + rss: Option, +) -> BTreeMap { + let mut v = BTreeMap::new(); + v.insert("catalog_open_ms".to_string(), open.cold_ms); + v.insert("catalog_open_warm_ms".to_string(), open.warm_ms); + v.insert("catalog_window_p99_ms".to_string(), open.window_ms.p99); + v.insert("catalog_filtered_ms".to_string(), open.filtered_ms); + let ips = thumbs.images_per_second; + v.insert("thumbnail_throughput_ips".to_string(), ips); + v.insert("thumbnail_per_image_p99_ms".to_string(), thumbs.per_image.p99); + for row in exports { + v.insert(row.key.to_string(), row.times.p99); + } + if let Some(rss) = rss { + v.insert("catalog_idle_rss_mb".to_string(), rss.now_mb()); + } + v +} + +// --------------------------------------------------------------------------- +// The report +// --------------------------------------------------------------------------- + +fn print_details( + open: &catalog_open::Open, + thumbs: &thumbnails::Throughput, + exports: &[exporting::EncodeRun], + rss: Option, +) { + println!("Opening the catalog (NFR-P1, and R2's second sentence)"); + println!( + " {:>10.1} ms open, count, first window and timeline — cold, first \ + connection of the process", + open.cold_ms + ); + println!( + " {:>10.1} ms the same four calls on a second connection", + open.warm_ms + ); + println!( + " {:>10.1} ms Catalog::open alone (connect, migrate, backfill)", + open.open_only_ms + ); + println!( + " {:>10.1} ms count and first window under a rating filter", + open.filtered_ms + ); + println!( + " {:>10.2} ms one 400-row window at a random offset, p99 of {WINDOWS} \ + (p50 {:.2}, max {:.2})", + open.window_ms.p99, open.window_ms.p50, open.window_ms.max + ); + println!( + " {} images, {} monthly timeline buckets", + open.images, open.buckets + ); + println!(); + + println!("Thumbnails on the embedded preview path (NFR-P3: >= 100 img/s)"); + println!( + " {:>10.1} img/s over {} images on {} lanes, {:.2} s of wall clock", + thumbs.images_per_second, + thumbs.images, + thumbs.lanes, + thumbs.wall_ms / 1e3 + ); + println!( + " {:>10.2} ms per image on its lane, p99 (p50 {:.2}, max {:.2})", + thumbs.per_image.p99, thumbs.per_image.p50, thumbs.per_image.max + ); + println!( + " {} stored, {} failed", + thumbs.stored, thumbs.failed + ); + println!(); + println!(" The bytes are in memory before the clock starts, so this is CPU"); + println!(" throughput with the fetch already paid for. That is what the target's"); + println!(" \"embedded preview path\" names; it is not a claim about a remote library."); + println!(); + + println!("Exporting 24 MP — the encode half only (NFR-P7 is the whole chain)"); + println!( + " {:>14} {:>11} {:>9} {:>9} {:>9}", + "sizing", "output", "p50", "p99", "file" + ); + for row in exports { + println!( + " {:>14} {:>5}x{:<5} {:>7.1}ms {:>7.1}ms {:>7.1}MB", + row.label, + row.width, + row.height, + row.times.p50, + row.times.p99, + row.bytes as f64 / 1e6 + ); + } + println!(); + println!(" No GPU render is in these figures, so they cannot pass NFR-P7 — only fail"); + println!(" it. See tools/bench/src/exporting.rs for why that is still worth gating."); + println!(); + + println!("Idle memory with the catalog open (NFR-P8's catalog share)"); + match rss { + Some(rss) => { + println!( + " {:>10.1} MB resident after opening 50k and scrolling 10k rows", + rss.now_mb() + ); + println!(" {:>10.1} MB peak for that process", rss.peak_mb()); + println!(); + println!(" RSS is exclusive of device-local GPU memory and this process has no"); + println!(" toolkit, no adapter and no decode cache — so it is the catalog layer's"); + println!(" share of NFR-P8's 500 MB, not NFR-P8. tools/bench/src/memory.rs holds"); + println!(" the answer to the question §4.1 asks, and the change it recommends."); + } + None => println!(" not measured on this platform"), + } + println!(); +} + +/// Print the gate table and return the failures, one line each. +fn print_verdict( + base: &Baseline, + values: &BTreeMap, + cx: Judging, + machine: &str, + comparable_fixture: bool, +) -> Vec { + println!("Against docs/bench-baseline.json"); + match (&base.recorded_on, cx.same_machine) { + (None, _) => println!( + " No baseline has been recorded yet. Budgets are still gated; drift is not.\n \ + Run `dr-bench record --reference` on the reference desktop and commit the diff." + ), + (Some(on), true) => println!(" Recorded on {on} — drift below is a verdict."), + (Some(on), false) if !comparable_fixture => println!( + " Recorded on {on} against a different fixture; drift below is information only." + ), + (Some(on), false) => println!( + " Recorded on {on}, and this is {machine}. Drift below is information, not a verdict." + ), + } + if !cx.reference { + println!( + " Not the reference desktop (--reference), so machine-sensitive budgets are\n \ + reported rather than asserted — §8 names the reference desktop, not CI." + ); + } + println!(); + println!( + " {:<30} {:>12} {:>13} {:>12} {:>8} {}", + "metric", "measured", "budget", "baseline", "drift", "verdict" + ); + + let mut failures = Vec::new(); + for (key, measured) in values { + let Some(metric) = base.metrics.get(key) else { + println!( + " {key:<30} {measured:>12.2} {:>13} {:>12} {:>8} not in the baseline", + "—", "—", "—" + ); + continue; + }; + let j = metric.judge(*measured, cx); + let budget = match (metric.budget, metric.direction) { + (None, _) => "—".to_string(), + (Some(b), baseline::Direction::LowerIsBetter) => format!("< {b:.1}"), + (Some(b), baseline::Direction::HigherIsBetter) => format!("> {b:.1}"), + }; + let recorded = match metric.recorded { + Some(r) => format!("{r:.2}"), + None => "—".to_string(), + }; + let drift = match j.drift { + Some(d) => format!("{:+.1}%", d * 100.0), + None => "—".to_string(), + }; + let mut verdict = String::new(); + if j.over_budget { + verdict.push_str("OVER BUDGET "); + failures.push(format!( + "{key} is {measured:.2} {}, past {budget} ({})", + metric.unit, metric.requirement + )); + } + if j.regressed { + verdict.push_str("REGRESSED "); + failures.push(format!( + "{key} drifted {drift} against the baseline, past the {:.0}% tolerance", + cx.tolerance * 100.0 + )); + } + if verdict.is_empty() { + verdict.push_str(if j.budget_deferred { + "ok (budget deferred)" + } else { + "ok" + }); + } + println!( + " {key:<30} {measured:>12.2} {budget:>13} {recorded:>12} {drift:>8} {verdict}" + ); + } + + // A metric in the file that nothing measured is a harness that has drifted + // from its own record, and is worth saying out loud rather than leaving as + // a row that quietly stopped appearing. + for key in base.metrics.keys() { + if !values.contains_key(key) { + println!(" {key:<30} {:>12} measured nothing this run", "—"); + } + } + + failures +} diff --git a/tools/bench/src/memory.rs b/tools/bench/src/memory.rs new file mode 100644 index 0000000..da4067b --- /dev/null +++ b/tools/bench/src/memory.rs @@ -0,0 +1,179 @@ +//! Idle memory with a 50k catalog open — and the question NFR-P8 leaves open. +//! +//! # The question §4.1 asks, answered +//! +//! §4.1 says of NFR-P8: *"must state whether it measures RSS inclusive or +//! exclusive of GPU allocations, and whether it holds after SQLite's page cache +//! warms on a 50k catalog."* Both halves have an answer, and neither is +//! flattering. +//! +//! **On GPU memory: what this reports is RSS, and RSS is exclusive of +//! device-local GPU allocations.** A Vulkan allocation in device-local heap +//! never enters the process's address space, so no counter under +//! `/proc/self/status` can see it; what *does* land in RSS is the host-visible +//! side — staging buffers, mapped upload rings, the read-back `AdjustPass` +//! performs on export — and the driver's own resident pages. So "RSS < 500 MB" +//! is not one budget, it is two questions wearing one number, and a build that +//! kept RSS at 400 MB while holding 3 GB of textures would pass it. +//! +//! The recommendation this measurement exists to support: **NFR-P8 should be +//! restated as two figures** — host RSS exclusive of device-local memory, and +//! a separate VRAM ceiling read from the adapter — because the second is the +//! one that decides whether the application survives beside a browser on an +//! 8 GB card, and nothing in this repository currently measures it. +//! +//! **On the page cache: warm.** The probe runs the queries before it reads the +//! counter, so SQLite's page cache holds the b-tree pages a grid scroll +//! touches. That is the right side to err on — a figure taken before the cache +//! warms would understate a steady-state library — and it is why the probe +//! scrolls rather than opening and stopping. +//! +//! # Why this is a subprocess +//! +//! RSS is a high-water-influenced property of a *process*, not of a function. +//! Building a 50k fixture allocates hundreds of megabytes; decoding thumbnails +//! allocates more; the allocator returns some of it to the OS and keeps the +//! rest. Measuring after any of that would report the harness's history rather +//! than the catalog's cost. So the probe is a fresh process that opens the +//! catalog, does the grid's work, reads its own counters and exits. +//! +//! # What this cannot certify, said plainly +//! +//! Not NFR-P8. The requirement is about the *application* at idle — Slint, the +//! wgpu device, the font stack, the decode cache and the catalog together — and +//! this process contains only the last of those. No requirement tag in this +//! crate names NFR-P8, for that reason — and see `exporting.rs` for why that +//! sentence avoids spelling the tag out. +//! +//! What it is, is the catalog layer's share, measured rather than guessed. The +//! decision NFR-P8 actually needs — how much of the 500 MB belongs to the +//! catalog and how much to everything above it — is a decision somebody has to +//! take, and taking it against a recorded number is better than taking it +//! against an estimate. That is what this records. Until it is taken, the +//! metric carries no budget and gates only against its own baseline. + +use std::path::Path; + +use anyhow::Result; +use dr_catalog::{Catalog, Granularity, Query}; + +/// Rows fetched per window while the probe scrolls. The same 400 +/// [`crate::catalog_open`] uses, for the same reason. +const WINDOW: usize = 400; + +/// Windows the probe pages through before reading the counter. +/// +/// Twenty-five is ten thousand rows: enough that SQLite's page cache holds a +/// realistic working set and that any per-window leak would be visible, and +/// far short of the whole library, which FR-CAT-4 forbids holding anyway. +const WINDOWS: usize = 25; + +/// A fixed clock, for the reason `catalog_open`'s `NOW` gives: nothing here +/// should depend on the day it runs. +const NOW: i64 = 2_000_000_000; + +/// Resident memory, in kilobytes, as Linux reports it. +#[derive(Debug, Clone, Copy)] +pub struct Rss { + /// `VmRSS`: resident now. + pub now_kb: u64, + /// `VmHWM`: the peak this process reached. Reported alongside because a + /// process that touched 900 MB and gave it back is not idling at 200 MB in + /// any sense a user would recognise — the pages came from somewhere. + pub peak_kb: u64, +} + +impl Rss { + pub fn now_mb(&self) -> f64 { + self.now_kb as f64 / 1024.0 + } + + pub fn peak_mb(&self) -> f64 { + self.peak_kb as f64 / 1024.0 + } +} + +/// Read this process's own counters. +/// +/// `None` anywhere without a Linux-shaped `/proc` — including Android, where +/// the file exists but a benchmark does not run, and macOS, where it does not. +/// Returning `None` rather than zero is deliberate: a memory figure of zero +/// would be reported as an excellent result. +pub fn of_this_process() -> Option { + let status = std::fs::read_to_string("/proc/self/status").ok()?; + let mut now = None; + let mut peak = None; + for line in status.lines() { + if let Some(rest) = line.strip_prefix("VmRSS:") { + now = rest.split_whitespace().next()?.parse::().ok(); + } else if let Some(rest) = line.strip_prefix("VmHWM:") { + peak = rest.split_whitespace().next()?.parse::().ok(); + } + } + Some(Rss { + now_kb: now?, + peak_kb: peak?, + }) +} + +/// The probe: open the catalog, do what the grid does, print the counters. +/// +/// Stdout is one line of `key=value` pairs rather than JSON, because the only +/// reader is [`in_a_fresh_process`] and a format a human can read in a log is +/// worth more here than one a parser prefers. +pub fn probe(catalog_path: &Path) -> Result<()> { + let catalog = Catalog::open(catalog_path) + .map_err(|e| anyhow::anyhow!("opening {} : {e}", catalog_path.display()))?; + let q = Query::default(); + + let images = catalog.count(&q, NOW)?; + let buckets = catalog.timeline(&q, Granularity::Month, NOW)?.len(); + + // Scroll, keeping only the window in hand — which is what the grid does, + // and what FR-CAT-4 requires it to do. If this ever starts costing memory + // proportional to how far the user scrolled, that is the bug this figure + // exists to catch. + let mut rows = 0usize; + let span = images.saturating_sub(WINDOW).max(1); + for i in 0..WINDOWS { + let start = (i * span) / WINDOWS.max(1); + rows = catalog.window(&q, start..start + WINDOW, NOW)?.len(); + } + + let Some(rss) = of_this_process() else { + anyhow::bail!("no /proc/self/status on this platform; RSS cannot be read"); + }; + println!( + "rss_kb={} peak_kb={} images={images} buckets={buckets} last_window={rows}", + rss.now_kb, rss.peak_kb + ); + Ok(()) +} + +/// Run [`probe`] in a fresh copy of this executable and read back its counters. +pub fn in_a_fresh_process(catalog_path: &Path) -> Result { + let exe = std::env::current_exe()?; + let output = std::process::Command::new(&exe) + .arg("memory-probe") + .arg(catalog_path) + .output()?; + + if !output.status.success() { + anyhow::bail!( + "the memory probe exited with {}: {}", + output.status, + String::from_utf8_lossy(&output.stderr).trim() + ); + } + + let text = String::from_utf8_lossy(&output.stdout); + let field = |key: &str| -> Option { + text.split_whitespace() + .find_map(|pair| pair.strip_prefix(key)) + .and_then(|v| v.parse::().ok()) + }; + let (Some(now_kb), Some(peak_kb)) = (field("rss_kb="), field("peak_kb=")) else { + anyhow::bail!("the memory probe printed something unreadable: {}", text.trim()); + }; + Ok(Rss { now_kb, peak_kb }) +} diff --git a/tools/bench/src/stats.rs b/tools/bench/src/stats.rs new file mode 100644 index 0000000..8dad9f5 --- /dev/null +++ b/tools/bench/src/stats.rs @@ -0,0 +1,122 @@ +//! Ranking a set of samples, the way `dr-gpu`'s frame budget ranks them. +//! +//! Copied in spirit rather than shared, because the two live in different +//! dependency worlds — `core/dr-gpu/examples/frame_budget.rs` is an example +//! inside a crate this one deliberately does not depend on (see `Cargo.toml`). +//! The arithmetic is identical on purpose: two percentile definitions in one +//! repository is how two benchmarks come to disagree about the same machine. +//! +//! # Nearest-rank, not an interpolating definition +//! +//! The samples *are* the population. There is no distribution being estimated +//! here, only a set of catalog opens or thumbnail encodes that either happened +//! inside the target or did not. At 100 samples the 99th percentile is the +//! second-worst, which is the honest reading of "one bad one in a hundred is +//! one too many" without letting a single scheduler hiccup on an unrelated +//! process decide the verdict. + +use std::time::Duration; + +/// Nearest-rank percentiles over a set of samples, in the caller's unit. +#[derive(Debug, Clone, Copy)] +pub struct Percentiles { + pub p50: f64, + pub p99: f64, + pub max: f64, +} + +impl Percentiles { + /// Rank `samples`. Panics on an empty set, which is a harness bug rather + /// than a measurement: a row with nothing in it must not print a zero that + /// reads like a very fast result. + pub fn of(mut samples: Vec) -> Self { + assert!( + !samples.is_empty(), + "percentiles of an empty sample set — the measurement produced nothing" + ); + samples.sort_by(f64::total_cmp); + let rank = |p: f64| { + let n = samples.len(); + let i = ((p * n as f64).ceil() as usize).clamp(1, n) - 1; + samples[i] + }; + Percentiles { + p50: rank(0.50), + p99: rank(0.99), + max: samples[samples.len() - 1], + } + } +} + +/// A duration in milliseconds, which is the unit every timing here is stated +/// in. One spelling, so no row is accidentally in seconds. +pub fn ms(d: Duration) -> f64 { + d.as_secs_f64() * 1e3 +} + +/// A deterministic generator, so a fixture is reproducible from its seed. +/// +/// SplitMix64. Chosen because it is eight lines, has no dependency, and passes +/// the only test that matters here — that the same seed produces the same +/// catalog on the reference desktop and on the CI runner, so a number measured +/// in one place describes the same workload as a number measured in the other. +/// Nothing cryptographic depends on it. +pub struct Rng(u64); + +impl Rng { + pub fn new(seed: u64) -> Self { + Rng(seed) + } + + pub fn next_u64(&mut self) -> u64 { + self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15); + let mut z = self.0; + z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); + z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); + z ^ (z >> 31) + } + + /// A value in `0..n`. Modulo-biased, which does not matter for a fixture: + /// nothing here is a statistical test, only a spread of plausible values. + pub fn below(&mut self, n: u64) -> u64 { + self.next_u64() % n.max(1) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_ninety_ninth_of_a_hundred_is_the_second_worst() { + // The property the whole suite's verdict rests on. Off by one here and + // every threshold is judged against the worst sample instead. + let samples: Vec = (1..=100).map(|n| n as f64).collect(); + let p = Percentiles::of(samples); + assert_eq!(p.p99, 99.0); + assert_eq!(p.max, 100.0); + assert_eq!(p.p50, 50.0); + } + + #[test] + fn a_single_sample_ranks_as_itself() { + // A row measured once — a cold catalog open — must not divide by zero + // or index off the end. + let p = Percentiles::of(vec![7.5]); + assert_eq!((p.p50, p.p99, p.max), (7.5, 7.5, 7.5)); + } + + #[test] + fn the_same_seed_gives_the_same_sequence() { + // Reproducibility from a seed is what makes a committed baseline mean + // anything: two runs must describe the same catalog. + let mut a = Rng::new(20_260_829); + let mut b = Rng::new(20_260_829); + let mut c = Rng::new(20_260_830); + let first: Vec = (0..8).map(|_| a.next_u64()).collect(); + let same: Vec = (0..8).map(|_| b.next_u64()).collect(); + let other: Vec = (0..8).map(|_| c.next_u64()).collect(); + assert_eq!(first, same); + assert_ne!(first, other); + } +} diff --git a/tools/bench/src/thumbnails.rs b/tools/bench/src/thumbnails.rs new file mode 100644 index 0000000..0768eec --- /dev/null +++ b/tools/bench/src/thumbnails.rs @@ -0,0 +1,272 @@ +//! TRACES: NFR-P3 +//! Thumbnail throughput on the embedded preview path. +//! +//! NFR-P3 asks for **≥ 100 images per second** on the reference desktop +//! through the embedded preview path, and ≥ 25 on a mid-range Android device. +//! Nothing had ever counted. +//! +//! # What is timed, and why it is not the sweep itself +//! +//! `ui/dr-ui/src/library.rs`'s [`spawn_thumbnail_sweep`] is the whole-library +//! pass, and it is the machinery this mirrors: a chunk at a time, lanes owning +//! disjoint slices, every lane decoding and encoding on its own, and the +//! single thread that owns the store writing the finished chunk. That shape is +//! reproduced here because it is the shape that decides the number — where the +//! parallelism is, and where the one lock is. +//! +//! It is *mirrored* rather than called, for a reason worth stating plainly: +//! that function takes a `RemoteBackend` and spends most of its wall clock in +//! WebDAV round trips. Calling it from a benchmark would need a Nextcloud +//! server, and what it would then measure is somebody's network. The per-image +//! work is identical either way — `dr_decode::decode_jpeg`, +//! `Preview::downscale_to`, `Preview::apply_orientation`, +//! `dr_thumbs::encode_rgba`, `ThumbStore::put` — and that work is what a target +//! written in images per second is about. +//! +//! So: **the bytes are already in memory when the clock starts.** This is CPU +//! throughput for the preview path with the fetch paid for, which is what a +//! local library gives you and what the target's "embedded preview path" +//! names. It is not a claim about a remote library, whose ceiling is latency +//! and which [`spawn_thumbnail_sweep`] exists to hide rather than to beat. +//! +//! # The plain-JPEG branch, deliberately +//! +//! `fetch_preview` has two arms: a plain JPEG is its own preview, and anything +//! else is located inside the container first. The fixture's files take the +//! first arm, so `locate_preview` is not in the measured span. That is honest +//! for two reasons — a library of camera JPEGs is a real library and takes +//! exactly this path, and for a RAW the located preview *is* a JPEG of about +//! this size, so what changes is a header walk of a few microseconds against a +//! decode of several milliseconds. What is genuinely not measured is the +//! container parse of an exotic format, and nothing here pretends otherwise. +//! +//! [`spawn_thumbnail_sweep`]: ../../dr_ui/library/fn.spawn_thumbnail_sweep.html + +use std::path::{Path, PathBuf}; +use std::time::Instant; + +use anyhow::{Context, Result}; +use dr_thumbs::{ThumbSize, ThumbStore, Thumbnail}; +use dr_types::Orientation; + +use crate::stats::{ms, Percentiles}; + +/// Images per chunk handed back to the thread that owns the store. +/// +/// 96, which is `SWEEP_CHUNK` in `library.rs`. Copied rather than chosen: the +/// point of this row is to describe the sweep's behaviour, and a different +/// chunk size would move the ratio of lane work to store work. +const CHUNK: usize = 96; + +/// The size class the whole-library pass fills. +/// +/// Grid only, which is `SWEEP_THUMB_SIZE`. The large class is four times the +/// transfer for a detail only a zoomed cell asks for, so the sweep does not +/// produce it and neither does this. +const SIZE: ThumbSize = ThumbSize::Grid; + +/// What a measured sweep produced. +pub struct Throughput { + pub images: usize, + pub lanes: usize, + pub wall_ms: f64, + pub images_per_second: f64, + /// Per image, on the lane that produced it: decode, downscale, orient, + /// encode. Not the store write, which happens elsewhere by design. + pub per_image: Percentiles, + /// Thumbnails that reached the store. + pub stored: usize, + /// Images whose preview would not decode. Any non-zero is a broken + /// fixture, not a slow one. + pub failed: usize, +} + +/// One lane's output: what it made, what each one cost, and what it dropped. +struct Lane { + made: Vec<(u64, Thumbnail)>, + times: Vec, + failed: usize, +} + +/// Sweep `images` thumbnails from `sources`, `lanes` at a time. +/// +/// `store_dir` is emptied first. Re-storing an id that is already present is +/// an `UPDATE` in place rather than an insert (see `ThumbStore::put`), and a +/// run that measured updates would not be measuring the pass this describes — +/// the sweep's work list is by construction what the store does *not* have. +pub fn measure( + sources: &[PathBuf], + store_dir: &Path, + images: usize, + lanes: usize, +) -> Result { + anyhow::ensure!(!sources.is_empty(), "no fixture sources to sweep"); + anyhow::ensure!(lanes > 0, "a sweep needs at least one lane"); + + let bytes: Vec> = sources + .iter() + .map(|p| std::fs::read(p).with_context(|| format!("reading {}", p.display()))) + .collect::>()?; + + let _ = std::fs::remove_dir_all(store_dir); + let mut store = ThumbStore::open(store_dir) + .map_err(|e| anyhow::anyhow!("opening the bench thumbnail store: {e}"))?; + + // Warm-up: every source decoded once, discarded. The first decode of a + // file faults in its Huffman tables and grows the allocator's arenas to + // the size a 1620x1080 RGBA buffer needs, and neither recurs across a + // sweep of thousands. + let warm_failures = Lane::run(&bytes, (0..bytes.len()).collect()).failed; + anyhow::ensure!( + warm_failures == 0, + "{warm_failures} of the {} fixture sources would not decode", + bytes.len() + ); + + let mut samples: Vec = Vec::with_capacity(images); + let mut stored = 0usize; + let mut failed = 0usize; + + let started = Instant::now(); + for chunk_start in (0..images).step_by(CHUNK) { + let chunk_end = (chunk_start + CHUNK).min(images); + // Each lane takes every `lanes`-th image of the chunk, which is how + // `spawn_thumbnail_sweep` splits one: disjoint slices, nothing shared, + // no lock. + let work: Vec> = (0..lanes) + .map(|lane| (chunk_start..chunk_end).skip(lane).step_by(lanes).collect()) + .collect(); + + let source_slice: &[Vec] = &bytes; + let produced: Vec = std::thread::scope(|scope| { + let handles: Vec<_> = work + .into_iter() + .map(|lane| scope.spawn(move || Lane::run(source_slice, lane))) + .collect(); + handles + .into_iter() + .map(|h| h.join().expect("a sweep lane panicked")) + .collect() + }); + + // The store is `&mut` and single-writer, so the chunk is written here + // and not on the lanes. This is the sweep's own discipline and it is + // part of the number: if the store were the bottleneck, no amount of + // lane parallelism would help and the row would say so. + for lane in produced { + samples.extend(lane.times); + failed += lane.failed; + for (file_id, thumb) in lane.made { + match store.put(file_id, SIZE, &thumb) { + Ok(_) => stored += 1, + Err(e) => { + log::warn!("storing bench thumbnail {file_id}: {e}"); + failed += 1; + } + } + } + } + } + let wall = started.elapsed(); + + anyhow::ensure!( + failed == 0, + "{failed} of {images} thumbnails failed; the throughput below would be \ + measuring how fast this gives up" + ); + + let wall_ms = ms(wall); + Ok(Throughput { + images, + lanes, + wall_ms, + images_per_second: images as f64 / wall.as_secs_f64().max(f64::MIN_POSITIVE), + per_image: Percentiles::of(samples), + stored, + failed, + }) +} + +impl Lane { + /// Turn each of `work`'s previews into a stored thumbnail's worth of bytes. + /// + /// The four calls, in the order `fetch_preview` and `encode_preview` make + /// them. `is_complete_jpeg` is included because it is in the real path and + /// because leaving it out would be the sort of small omission that turns a + /// measurement into an estimate: a truncated JPEG decodes "successfully" + /// into a partial frame, so the check is not optional and its cost is not + /// somebody else's. + fn run(sources: &[Vec], work: Vec) -> Lane { + let mut made = Vec::with_capacity(work.len()); + let mut times = Vec::with_capacity(work.len()); + let mut failed = 0usize; + + for index in work { + let bytes = &sources[index % sources.len()]; + // The fixture makes every fourth image portrait, so a quarter of + // these pay for the permutation `apply_orientation` performs. In a + // real library that fraction is whatever the photographer shot. + let orientation = if index % 4 == 3 { + Orientation::from_exif(6) + } else { + Orientation::default() + }; + + let started = Instant::now(); + if !dr_decode::is_complete_jpeg(bytes) { + failed += 1; + continue; + } + let mut preview = match dr_decode::decode_jpeg(bytes) { + Ok(p) => p, + Err(e) => { + log::debug!("bench preview {index}: {e}"); + failed += 1; + continue; + } + }; + preview.downscale_to(SIZE.edge()); + preview.apply_orientation(orientation); + let rgba = &preview.rgba; + let encoded = match dr_thumbs::encode_rgba(preview.width, preview.height, rgba) { + Ok(b) => b, + Err(e) => { + log::debug!("bench thumbnail {index}: {e}"); + failed += 1; + continue; + } + }; + times.push(ms(started.elapsed())); + + made.push(( + index as u64 + 1, + Thumbnail { + width: preview.width, + height: preview.height, + bytes: encoded, + }, + )); + } + + Lane { + made, + times, + failed, + } + } +} + +/// How many lanes to sweep with by default. +/// +/// The machine's threads, not `SWEEP_LANES`. The sweep's six are sized for +/// *latency* — each image is ~0.6 s of WebDAV round trip and almost no CPU, so +/// six in flight is a queue depth rather than a core count. With the bytes +/// already in memory the work is purely CPU, and six lanes on a 24-thread +/// desktop would report a quarter of the throughput the machine has. Stated +/// here rather than buried, because it is the one place this deviates from the +/// shape it otherwise copies. +pub fn default_lanes() -> usize { + std::thread::available_parallelism() + .map(|n| n.get()) + .unwrap_or(1) +}