diff --git a/.gitea/workflows/benchmark.yml b/.gitea/workflows/benchmark.yml
new file mode 100644
index 0000000..782a927
--- /dev/null
+++ b/.gitea/workflows/benchmark.yml
@@ -0,0 +1,193 @@
+name: Benchmarks
+
+# The suite docs/requirements.md §8 has been promising since it was written:
+# "an automated benchmark suite against a synthetic 50k catalog, run per-commit
+# … A regression beyond stated tolerance fails the build."
+#
+# Its own workflow rather than a step inside build-and-test.yml, and the reason
+# is what a failure here means. A red `Build and test` says the code is wrong; a
+# red `Benchmarks` says the code is slower than it was, which is a different
+# conversation, is read by different people, and must not be reachable by
+# retrying a flaky compile.
+#
+# # Why this is split in two
+#
+# §8 names "the reference desktop", not CI, and it is right to. So:
+#
+# cpu — runs on every push. It needs no adapter and no display, and the
+# budgets it asserts (a 50k catalog opening inside two seconds) have
+# two orders of magnitude of headroom, so a modest runner can be held
+# to them honestly. Machine-sensitive budgets — throughput targets
+# written for a 24-thread desktop — are reported here rather than
+# asserted; `dr-bench` decides that per metric and says so in its
+# report. Asserting them on a two-core container would produce exactly
+# what core/dr-gpu/tests/frame_budget.rs refused to produce: a red gate
+# everybody learns to ignore.
+#
+# gpu — the frame budget, which already exists and already skips itself where
+# there is no adapter. Not on push: it would build wgpu and naga on
+# every commit to establish, every time, that this runner has no GPU. It
+# runs on demand (Actions → Run workflow) so that a runner that *does*
+# have one can be pointed at it, and the numbers it produces belong in
+# docs/frame-budget.md by hand, as they already are.
+
+on:
+ push:
+ branches: [main, master, develop]
+ pull_request:
+ branches: [main, master, develop]
+ workflow_dispatch:
+
+jobs:
+ cpu:
+ runs-on: linux/amd64
+ name: CPU and I/O (per commit)
+ # Node for actions/checkout and actions/cache, which the bare runner image
+ # cannot execute. Rust is installed below.
+ container:
+ image: catthehacker/ubuntu:act-latest
+
+ env:
+ # Same reasoning as the desktop job in build-and-test.yml: incremental
+ # state exists to make the *second* build in a working tree fast, which is
+ # not a thing a fresh checkout has, and it fills the runner's disk.
+ CARGO_INCREMENTAL: 0
+ # The fixture, out of the workspace so actions/cache never picks it up.
+ # A 14 MB synthetic catalog is two seconds to regenerate and would
+ # otherwise be uploaded and downloaded on every push to save them.
+ DR_BENCH_DIR: /tmp/darkroom-bench
+
+ steps:
+ - name: Checkout
+ uses: actions/checkout@v4
+
+ # No `git lfs pull` here, deliberately. `dr-bench` depends on the catalog,
+ # the decoder, the thumbnail store and the encoder, and on nothing that
+ # reaches `dr-segment` — so the model this repository keeps in LFS is not
+ # part of this job's dependency graph and fetching it would be a minute
+ # spent on a file nothing opens.
+
+ - name: Cache cargo
+ uses: actions/cache@v4
+ with:
+ path: |
+ ~/.cargo/registry
+ ~/.cargo/git
+ target
+ key: bench-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
+
+ # Pinned to the workspace rust-version, as every other job here is: a
+ # floating toolchain turns an unrelated push into a mystery failure, and
+ # for a benchmark it would turn one into a mystery *regression*.
+ #
+ # rust-analyzer is named for the reason build-and-test.yml gives: rustup
+ # reconciles rust-toolchain.toml on the first cargo call whatever this
+ # step asks for, so naming it keeps the download inside the step that says
+ # it is installing things.
+ - name: Install Rust 1.92.0
+ run: |
+ set -e
+ curl -fsSL https://sh.rustup.rs | sh -s -- \
+ -y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
+ --component rust-analyzer
+ echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
+
+ # `-p dr-bench`, not `--workspace`. The whole point of that crate having
+ # no GPU and no UI dependency is that this job resolves the catalog, the
+ # decoder and the encoders and stops there — a few minutes rather than the
+ # release build of Slint and wgpu the desktop job pays for.
+ #
+ # Release, and it is not optional: the workspace builds its own crates at
+ # opt-level = 0 in dev, and every figure this produces is dominated by
+ # this workspace's own code. A debug run would measure rustc.
+ - name: Build the suite
+ run: cargo build --release -p dr-bench
+
+ # Exit 1 is a violated budget or a regression past tolerance; exit 2 is
+ # the harness failing to run at all. Both fail the job, and the report
+ # above the failure says which.
+ - name: Measure, and gate
+ run: cargo run --release -p dr-bench -- check
+
+ - name: Disk after
+ if: always()
+ run: df -h /workspace 2>/dev/null || df -h .
+
+ gpu:
+ # On demand only — see the header. A runner with a Vulkan device can be
+ # pointed at this; one without will skip the measurement and say so, which
+ # is the same posture the rest of this repository's device tests take.
+ if: github.event_name == 'workflow_dispatch'
+ runs-on: linux/amd64
+ name: Frame budget (on demand)
+ container:
+ image: catthehacker/ubuntu:act-latest
+
+ env:
+ CARGO_INCREMENTAL: 0
+
+ steps:
+ - name: Checkout
+ uses: actions/checkout@v4
+
+ # `dr-gpu` depends on `dr-segment` for the watershed's pixel passes. Its
+ # default features are off, so no weights are compiled in — but the fetch
+ # is cheap insurance and its failure is not fatal. The header of the same
+ # step in build-and-test.yml explains why the extraheader is stripped
+ # rather than reused: two Authorization headers is a 400 from Gitea, one
+ # step after the batch call that had just succeeded.
+ - name: Fetch the segmentation model
+ continue-on-error: true
+ env:
+ LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
+ run: |
+ set -e
+ git lfs install --local
+ git config --local --get-regexp '^http\..*extraheader$' \
+ | cut -d' ' -f1 | sort -u \
+ | while read -r key; do git config --local --unset-all "$key"; done || true
+ git config --local lfs.url \
+ "https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
+ git lfs pull
+
+ - name: Cache cargo
+ uses: actions/cache@v4
+ with:
+ path: |
+ ~/.cargo/registry
+ ~/.cargo/git
+ target
+ key: bench-gpu-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
+
+ - name: Build dependencies
+ run: |
+ apt-get update -qq
+ apt-get install -y -qq pkg-config libfontconfig1-dev libxkbcommon-dev
+
+ - name: Install Rust 1.92.0
+ run: |
+ set -e
+ curl -fsSL https://sh.rustup.rs | sh -s -- \
+ -y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
+ --component rust-analyzer
+ echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
+
+ # The guard, in release. Its own module documentation is explicit that a
+ # release run checks strictly more than a dev one: the CPU half of a frame
+ # is shader-string assembly, which is several times slower unoptimised, so
+ # it is folded into the assertion only when debug_assertions is off.
+ #
+ # With no adapter this prints "skipping: no GPU adapter" and passes. A
+ # test that cannot run is not evidence either way, and turning that into a
+ # failure would make the job useless on the runner it usually lands on.
+ - name: Frame budget (FR-DSP-3)
+ run: cargo test --release -p dr-gpu --test frame_budget -- --nocapture
+
+ # The instrument behind docs/frame-budget.md. It exits non-zero with no
+ # adapter, which is right for a tool a person runs deliberately and wrong
+ # for a job that usually has none — hence continue-on-error. Its table is
+ # in the log for whoever asked for this run; the committed numbers are
+ # still updated by hand, as that file says.
+ - name: Frame budget table
+ continue-on-error: true
+ run: cargo run --release -p dr-gpu --example frame_budget
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index 7c554e9..c730269 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -83,6 +83,21 @@ GPU tests skip themselves where there is no adapter rather than failing — a
test that cannot run is not evidence either way — so a green run on a machine
without a GPU is expected, and does not mean the GPU paths were exercised.
+There is a fifth check, and it is not in that list because you are unlikely to
+break it by accident:
+
+```bash
+cargo run --release -p dr-bench -- check
+```
+
+That is the benchmark suite (`docs/requirements.md` §8), which builds a
+synthetic 50,000-image catalog and fails the build if a performance target is
+missed or a measurement has drifted past its tolerance. It runs on every push in
+its own workflow. [`docs/benchmarks.md`](docs/benchmarks.md) says what it
+measures, what it deliberately does not, and how to read a failure. If you have
+touched the catalog, the decoder, the thumbnail store or the exporter, run it
+before you send.
+
## Requirements and traceability
[`requirements.md`](docs/requirements.md) is the register of record.
@@ -150,6 +165,7 @@ One commit per change. If you fixed two things, that is two commits.
| [`core/dr-pipeline/ops/README.md`](core/dr-pipeline/ops/README.md) | Adding or changing a develop operation — start here regardless |
| [`docs/architecture.md`](docs/architecture.md) | Anything touching the render path, catalog or sync |
| [`docs/code-health.md`](docs/code-health.md) | Deciding what to work on; grades each seam by what it costs |
+| [`docs/benchmarks.md`](docs/benchmarks.md) | A change that could plausibly cost time or memory |
| [`docs/technical-debt.md`](docs/technical-debt.md) | Something looks wrong — check it was not chosen |
| [`docs/distribution.md`](docs/distribution.md) | Packaging a build, or adding a permission to one |
| [`docs/requirements.md`](docs/requirements.md) | Reference, not reading |
diff --git a/Cargo.lock b/Cargo.lock
index 57645f1..afcc877 100644
--- a/Cargo.lock
+++ b/Cargo.lock
@@ -1403,6 +1403,23 @@ version = "0.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76"
+[[package]]
+name = "dr-bench"
+version = "0.9.0"
+dependencies = [
+ "anyhow",
+ "dr-catalog",
+ "dr-decode",
+ "dr-export",
+ "dr-thumbs",
+ "dr-types",
+ "env_logger",
+ "log",
+ "rusqlite",
+ "serde",
+ "serde_json",
+]
+
[[package]]
name = "dr-catalog"
version = "0.9.0"
diff --git a/Cargo.toml b/Cargo.toml
index 1a57353..875ca97 100644
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -21,6 +21,7 @@ members = [
"ui/dr-ui",
"apps/darkroom-desktop",
"apps/darkroom-android",
+ "tools/bench",
"tools/traceability",
]
diff --git a/docs/bench-baseline.json b/docs/bench-baseline.json
new file mode 100644
index 0000000..579873f
--- /dev/null
+++ b/docs/bench-baseline.json
@@ -0,0 +1,113 @@
+{
+ "_readme": [
+ "The committed numbers for DarkRoom's benchmark suite (docs/requirements.md §8).",
+ "Produced and checked by `cargo run --release -p dr-bench`; docs/benchmarks.md explains each metric.",
+ "",
+ "Two gates, and they are not the same gate. `budget` is the requirement's own threshold and never moves.",
+ "`recorded` is what the reference desktop last measured, and a run that drifts past `tolerance` beyond it",
+ "fails the build even while still inside the budget — which is how most performance rot actually arrives.",
+ "",
+ "`recorded` is null on every metric because nobody has run the suite yet. That is deliberate: writing",
+ "plausible-looking figures here would make every later comparison a comparison against a guess. Run",
+ "`cargo run --release -p dr-bench -- record --reference` on the reference desktop and commit the diff.",
+ "Until then the budget gate works and the regression gate says, in the report, that it cannot.",
+ "",
+ "`machine_sensitive` says whether a budget is a statement about a machine as much as about the code.",
+ "Those budgets are asserted only under --reference: §8 names the reference desktop, and a two-core CI",
+ "container cannot speak to a target written for twenty-four threads. Asserting one there would produce a",
+ "red gate everybody learns to ignore, which is the trap core/dr-gpu/tests/frame_budget.rs already avoids.",
+ "",
+ "Several metrics carry a requirement ID with a qualifier. Read those literally. NFR-P7's budget here is",
+ "checked against the encode half of an export only — no GPU render is in the figure — so it can fail the",
+ "requirement and cannot pass it, and no TRACES tag claims otherwise. NFR-P8 has no budget at all yet,",
+ "because nobody has decided how much of its 500 MB belongs to the catalog layer; this records the number",
+ "that decision needs."
+ ],
+ "tolerance": 0.15,
+ "recorded_on": null,
+ "recorded_at_unix": null,
+ "fixture": null,
+ "metrics": {
+ "catalog_filtered_ms": {
+ "requirement": "FR-CAT-6",
+ "what": "Count plus first window under a rating filter, which compiles to a correlated subquery over versions.",
+ "unit": "ms",
+ "direction": "lower_is_better",
+ "machine_sensitive": true,
+ "budget": null,
+ "recorded": null
+ },
+ "catalog_idle_rss_mb": {
+ "requirement": "NFR-P8 (the catalog layer's share only — no toolkit, no adapter, no decode cache)",
+ "what": "Resident memory of a process that opened the 50k catalog and scrolled ten thousand rows.",
+ "unit": "MB",
+ "direction": "lower_is_better",
+ "machine_sensitive": false,
+ "budget": null,
+ "recorded": null
+ },
+ "catalog_open_ms": {
+ "requirement": "NFR-P1",
+ "what": "Catalog::open plus the count, first window and timeline the grid cannot paint without.",
+ "unit": "ms",
+ "direction": "lower_is_better",
+ "machine_sensitive": false,
+ "budget": 2000.0,
+ "recorded": null
+ },
+ "catalog_open_warm_ms": {
+ "requirement": "NFR-P1",
+ "what": "The same four calls on a second connection, with SQLite's page cache already warm.",
+ "unit": "ms",
+ "direction": "lower_is_better",
+ "machine_sensitive": false,
+ "budget": 2000.0,
+ "recorded": null
+ },
+ "catalog_window_p99_ms": {
+ "requirement": "FR-CAT-4",
+ "what": "One 400-row grid window at a random offset, p99 of one hundred.",
+ "unit": "ms",
+ "direction": "lower_is_better",
+ "machine_sensitive": true,
+ "budget": null,
+ "recorded": null
+ },
+ "export_24mp_long_edge_2048_ms": {
+ "requirement": "FR-EXP-3",
+ "what": "The web export: resample a 24 MP frame to a 2048 px long edge, sharpen, encode. p99 of five.",
+ "unit": "ms",
+ "direction": "lower_is_better",
+ "machine_sensitive": true,
+ "budget": null,
+ "recorded": null
+ },
+ "export_24mp_original_ms": {
+ "requirement": "NFR-P7 (the encode half only — the GPU render is not in this figure)",
+ "what": "Resample, output-sharpen and JPEG-encode a 24 MP frame at source size. p99 of five.",
+ "unit": "ms",
+ "direction": "lower_is_better",
+ "machine_sensitive": true,
+ "budget": 2000.0,
+ "recorded": null
+ },
+ "thumbnail_per_image_p99_ms": {
+ "requirement": "NFR-P3",
+ "what": "One thumbnail on its own lane: decode the preview, downscale, orient, encode. p99.",
+ "unit": "ms",
+ "direction": "lower_is_better",
+ "machine_sensitive": true,
+ "budget": null,
+ "recorded": null
+ },
+ "thumbnail_throughput_ips": {
+ "requirement": "NFR-P3",
+ "what": "Whole-sweep throughput: 1200 thumbnails through the sweep's chunk-and-lane shape, wall clock.",
+ "unit": "img/s",
+ "direction": "higher_is_better",
+ "machine_sensitive": true,
+ "budget": 100.0,
+ "recorded": null
+ }
+ }
+}
diff --git a/docs/benchmarks.md b/docs/benchmarks.md
new file mode 100644
index 0000000..5947862
--- /dev/null
+++ b/docs/benchmarks.md
@@ -0,0 +1,243 @@
+# The benchmark suite
+
+**Status:** Built, not yet recorded · 2026-08-30
+**Companion to:** [requirements.md](requirements.md) §4.1 (performance targets) · §8 (verification)
+**Instrument:** [`tools/bench`](../tools/bench) — `cargo run --release -p dr-bench -- check`
+**Committed numbers:** [`bench-baseline.json`](bench-baseline.json)
+**GPU half:** [`core/dr-gpu/tests/frame_budget.rs`](../core/dr-gpu/tests/frame_budget.rs) ·
+[frame-budget.md](frame-budget.md)
+
+§8 has said since it was written that performance is verified by *"an automated
+benchmark suite against a synthetic 50k catalog, run per-commit … A regression
+beyond stated tolerance fails the build."* Until this suite there was none. No
+`benches/`, no `[[bench]]`, no criterion, no fixture — and ten performance
+requirements that could therefore be neither passed nor failed, five of them
+carrying a `TRACES:` tag regardless.
+
+This file is what the suite covers, what it deliberately does not, and how to
+read a failure.
+
+---
+
+## The state of it, first
+
+**No numbers have been recorded yet.** Every `recorded` field in
+[`bench-baseline.json`](bench-baseline.json) is `null`, on purpose: writing
+plausible-looking figures into a baseline would make every later comparison a
+comparison against a guess, and the first real regression would be invisible.
+
+To record them, on the reference desktop:
+
+```sh
+cargo run --release -p dr-bench -- record --reference
+```
+
+and commit the diff. Until that happens the **budget** gate works — a catalog
+that takes three seconds to open fails the build today — and the **regression**
+gate reports that it has nothing to compare against, rather than pretending.
+
+---
+
+## What it measures
+
+| Metric | Requirement | Gated? |
+|---|---|---|
+| `catalog_open_ms` | **NFR-P1**, and R2's second sentence | Yes, everywhere — budget 2000 ms |
+| `catalog_open_warm_ms` | NFR-P1, page cache warm | Yes, everywhere — budget 2000 ms |
+| `catalog_window_p99_ms` | FR-CAT-4 | Regression only |
+| `catalog_filtered_ms` | FR-CAT-6 | Regression only |
+| `thumbnail_throughput_ips` | **NFR-P3** | Budget 100 img/s, on the reference desktop |
+| `thumbnail_per_image_p99_ms` | NFR-P3 | Regression only |
+| `export_24mp_original_ms` | NFR-P7, **encode half only** | One-sided: can fail it, cannot pass it |
+| `export_24mp_long_edge_2048_ms` | FR-EXP-3 | Regression only |
+| `catalog_idle_rss_mb` | NFR-P8, **catalog layer only** | Regression only — see below |
+
+Two of those rows carry a qualifier, and the qualifiers are the point.
+
+### Requirements this can now pass *or* fail
+
+**NFR-P1 — catalog open under 2 s.** The measured span is the four things the
+library view cannot paint without: `Catalog::open` (which connects, migrates and
+**backfills**, and the backfill is three passes over the images table on every
+open), `count`, the first 400-row `window`, and the monthly `timeline`. Tagged
+`TRACES: NFR-P1` in [`tools/bench/src/catalog_open.rs`](../tools/bench/src/catalog_open.rs),
+because a build that breaks it fails this gate.
+
+**NFR-P3 — ≥ 100 images per second on the embedded preview path.** The
+per-image work is exactly what `spawn_thumbnail_sweep` does — `decode_jpeg`,
+`Preview::downscale_to`, `Preview::apply_orientation`, `encode_rgba`,
+`ThumbStore::put` — arranged in the same shape: chunks of 96, lanes owning
+disjoint slices, and the single thread that owns the store writing the finished
+chunk. Tagged `TRACES: NFR-P3` in
+[`tools/bench/src/thumbnails.rs`](../tools/bench/src/thumbnails.rs).
+
+### Requirements this can only half-answer, and is not tagged for
+
+**NFR-P7 — 24 MP export under 2 s, full chain.** The full chain is decode,
+demosaic, a full-resolution GPU render, a read-back, then resize, sharpen and
+encode. Only the last three run without an adapter. So the figure here is a
+**lower bound** on the requirement: exceeding 2 s in the encode alone violates
+NFR-P7 no matter how fast the render is, and coming in under it proves nothing.
+The budget is gated on that basis and there is no `TRACES: NFR-P7` anywhere in
+`tools/bench`.
+
+**NFR-P8 — idle memory under 500 MB.** The probe is a fresh process holding the
+catalog and nothing else: no Slint, no wgpu device, no font stack, no decode
+cache. Its RSS is the catalog layer's *share* of that 500 MB, not the figure the
+requirement is about. It carries no budget for a reason given below.
+
+### Requirements out of scope, listed so their absence reads as a decision
+
+NFR-P2 (grid scroll at 60 fps), P4 (open in develop), P5 (slider to visible),
+P6 (pan/zoom), P9 (UI-executor blocking), P10 (touch response), P11 (layout
+transition), P12 (warm shader setup), P13 (next image in culling), P14 (focus
+peaking), P15 (drawn mask stroke). Every one of them needs a frame-timing probe
+inside a running Slint application, a GPU adapter, or both. None is faked here.
+
+The GPU half of the story that *does* exist is
+[frame-budget.md](frame-budget.md) and its guard test, which asserts FR-DSP-3
+and skips itself where there is no adapter. `.gitea/workflows/benchmark.yml`
+runs it as its own job for exactly that reason.
+
+---
+
+## The fixture
+
+Fifty thousand rows over a pool of twelve real image files. Rows are cheap and
+pixels are not: everything the catalog half touches is rows and is therefore
+exact at full scale, and everything the pixel half touches is one file at a time
+and does not care how many rows point at it. The result is ~14 MB on disk
+instead of ~2 TB, and neither half is flattered by that.
+
+| | |
+|---|---|
+| Rows | 50,000 images, 50,000 default versions, 400 folders, one root |
+| Capture times | Twelve years from a fixed epoch, so the timeline has ~144 monthly buckets |
+| Sources | 12 synthesised JPEGs at 1620 × 1080 — the size `dr-decode` records a CR2 carrying in IFD2 |
+| Seed | 20260829, in [`tools/bench/src/main.rs`](../tools/bench/src/main.rs) |
+| Location | `$DR_BENCH_DIR`, else the system temporary directory |
+
+It is reproducible from the seed, and a `stamp.json` beside it records what it
+was built from — seed, row count, source count, preview size, and `dr-catalog`'s
+schema version. A mismatch rebuilds rather than silently measuring a different
+workload than the baseline describes.
+
+Two honest limits on it:
+
+- **The page cache is warm.** The fixture was written by this suite or by an
+ earlier run of it, so neither the catalog open nor the thumbnail sweep pays
+ for a cold disk. On the reference desktop's NVMe a genuinely cold read of a
+ 14 MB catalog is tens of milliseconds; on spinning rust it is not.
+- **The sources are synthetic.** A coarse gradient with a fine dither, which is
+ what `frame_budget.rs` synthesises for the same reason — a flat frame lets the
+ memory system serve every sample from one cache line and flatters a box
+ filter, and pure noise defeats the entropy coder in the other direction.
+
+---
+
+## Two gates, and how to read a failure
+
+**Budget.** The requirement's own threshold. It does not move. Failing it means
+a requirement is violated.
+
+**Regression.** More than 15% worse than the last recorded figure *on the same
+machine, against the same fixture*. Failing it means the code got slower while
+still inside the requirement — which is how most performance rot actually
+arrives, never over the line, always a little worse, until one day the line is
+crossed by a change that was not the cause.
+
+A metric declares whether its budget is `machine_sensitive`. Those are asserted
+only under `--reference`, and reported everywhere else. §8 names *"the reference
+desktop"*, not CI, and it is right to: a container with two cores cannot speak
+to a throughput target written for twenty-four threads, and asserting one there
+would produce exactly what `core/dr-gpu/tests/frame_budget.rs` refused to
+produce — *"a red suite that everyone learns to ignore"*. Catalog open is not
+machine-sensitive: 2 s against an expected figure two orders of magnitude
+smaller is a threshold any machine can be held to.
+
+Exit codes: `0` everything passed, `1` a gate failed, `2` the harness itself
+could not run. Distinguished so a CI log that says "failed" does not leave
+anyone guessing whether the code got slower or the fixture would not build.
+
+**Release, always.** The workspace builds its own crates at `opt-level = 0` in
+dev, and every figure here is dominated by this workspace's own code — the JPEG
+decode, the box filter, the resample, the sharpen. A debug run measures rustc's
+shadow. The report says which profile it was built in on its second line.
+
+---
+
+## NFR-P8, and the question §4.1 asks
+
+§4.1 says NFR-P8 *"must state whether it measures RSS inclusive or exclusive of
+GPU allocations, and whether it holds after SQLite's page cache warms on a 50k
+catalog."* Both halves have an answer.
+
+**On the page cache: warm.** The probe runs the count, the timeline and
+twenty-five windows before it reads its counters, so SQLite's cache holds the
+b-tree pages a scroll touches. That is the right side to err on — a figure taken
+before the cache warms would understate a steady-state library.
+
+**On GPU memory: RSS is exclusive of device-local allocations, and cannot be
+made otherwise.** A Vulkan allocation in a device-local heap never enters the
+process's address space, so nothing under `/proc/self/status` can see it. What
+*does* land in RSS is the host-visible side — staging buffers, mapped upload
+rings, the read-back `AdjustPass` performs on export — plus the driver's own
+resident pages.
+
+So "idle memory < 500 MB" is two questions wearing one number, and a build
+holding 400 MB of RSS and 3 GB of textures would pass it.
+
+**Recommendation: NFR-P8 should be restated as two figures** — host RSS
+exclusive of device-local memory, and a separate VRAM ceiling read from the
+adapter — because the second is the one that decides whether the application
+survives beside a browser on an 8 GB card, and nothing in this repository
+measures it today.
+
+**And a decision is outstanding.** `catalog_idle_rss_mb` carries no budget
+because nobody has decided how much of the 500 MB belongs to the catalog layer
+and how much to everything above it. The suite records the number so that
+decision can be taken against a measurement rather than an estimate. When it is
+taken, put the figure in `budget` and the metric becomes a gate.
+
+---
+
+## What is not measured, and would be worth adding
+
+- **The UI's own open.** `ui/dr-ui/src/library.rs` does not call
+ `Catalog::count` or `Catalog::window`; it issues its own SQL against the same
+ tables, with a `VISIBLE` predicate and a burst-folding clause. `dr-bench`
+ cannot see those without depending on `dr-ui`, which would drag Slint into a
+ job that has no display. **Falsifiable end:** when the grid's queries move
+ down into `dr-catalog` — which is where SQL over catalog tables belongs —
+ `catalog_open_ms` becomes the whole of the application's open and this caveat
+ can be deleted rather than argued about.
+- **The remote sweep.** `spawn_thumbnail_sweep`'s wall clock against a real
+ server is latency, not CPU, and is what FR-NC-3's design is judged by. It
+ needs a server and belongs in a different kind of test.
+- **A cold disk.** See the fixture's limits above.
+- **Android.** §4.1 states a second column of targets and §8 asks for
+ "periodically on the named reference Android devices". Nothing here runs on a
+ device. Spike S10 is the piece of work that would start it.
+- **Everything with a frame in it.** See the out-of-scope list above.
+
+---
+
+## Running it
+
+```sh
+# Measure and print. Judges nothing.
+cargo run --release -p dr-bench -- run
+
+# Measure and gate. What CI runs.
+cargo run --release -p dr-bench -- check
+
+# The same, with machine-sensitive budgets asserted too.
+cargo run --release -p dr-bench -- check --reference
+
+# Rewrite bench-baseline.json from this run, and commit the diff.
+cargo run --release -p dr-bench -- record --reference
+```
+
+Useful flags: `--fixture
` (or `$DR_BENCH_DIR`) to put the synthetic
+catalog somewhere specific, `--lanes ` to pin the sweep's parallelism, and
+`--thumbnails ` to lengthen or shorten the throughput row.
diff --git a/docs/outstanding.md b/docs/outstanding.md
index 1bc0473..171ca18 100644
--- a/docs/outstanding.md
+++ b/docs/outstanding.md
@@ -285,31 +285,43 @@ load-bearing for interoperating with the editors FR-CAT-14 imports from.
---
-## 8. The performance targets are unverified, not unmet
+## 8. The performance targets are half-verified, and the half that is left is the hard one
-Eleven of the fifteen §4.1 targets carry no tag: NFR-P2, -P3, -P4, -P6, -P7, -P8, -P10, -P11, -P12,
--P14, -P15. That is the uninteresting part of this section.
+§8 and §4.1 both require the same thing in the same words: an automated benchmark suite against a
+synthetic 50k catalog, run per commit, where **"a regression beyond a stated tolerance is a build
+failure, not a notification."** For most of this project's life it did not exist — no `benches/`, no
+criterion, no synthetic catalog, and three CI workflows that between them measured nothing.
-The interesting part is that §8 and §4.1 both require the same thing, in the same words, and it does
-not exist: an automated benchmark suite against a synthetic 50k catalog, run per commit, where **"a
-regression beyond a stated tolerance is a build failure, not a notification."** There is no
-`benches/` directory in the workspace, no criterion dependency, and no synthetic catalog. The three
-CI workflows run `cargo fmt --check`, clippy, `cargo test --workspace`, a release build, an Android
-cross-build and a layering check. None of them measures anything, so there is no baseline to
-regress against and no tolerance to exceed.
+**It exists now, for everything that does not need a frame.** [`tools/bench`](../tools/bench) builds
+a deterministic 50,000-row catalog over a pool of a dozen real files, measures against it, and fails
+the build on a violated budget or a drift past tolerance;
+[`.gitea/workflows/benchmark.yml`](../.gitea/workflows/benchmark.yml) runs it on every push, and
+[benchmarks.md](benchmarks.md) is the account of what it does and does not cover. **NFR-P1** and
+**NFR-P3** are now genuinely gated, and R2's "catalog opens in under 2s" clause with them.
-What does exist is narrower and genuinely good: `dr-gpu/examples/frame_budget` is a real instrument,
-its results are committed in [frame-budget.md](frame-budget.md) with the machine and profile named,
-and TD-4's before-and-after was measured with it. But it is run by hand — frame-budget.md's own
-instruction is "rerun and diff this file" — and the guard version that does live in CI skips itself
-where there is no GPU adapter, which the workflow notes is the normal case on a runner, while
-asserting its CPU half only when `debug_assertions` is off, which a dev-profile `cargo test` is not.
-In CI it therefore asserts approximately nothing.
+Three qualifications, all of them stated in the harness itself rather than only here:
-**The claim to take from this is precise.** Nothing here says the performance targets are missed.
-Several are plausibly met. It says that if one were broken tomorrow, nobody would find out — which
-is the failure mode §8 was written to prevent, and the reason it belongs in this document rather
-than in a backlog.
+- **The numbers have not been recorded yet.** Every `recorded` field in
+ [bench-baseline.json](bench-baseline.json) is `null`, deliberately: a fabricated baseline is worse
+ than none. Until `dr-bench record --reference` is run on the reference desktop and committed, the
+ budget gate works and the regression gate does not.
+- **NFR-P7 and NFR-P8 are half-measured and are not tagged.** The export row covers the encode half
+ of the chain and no GPU render, so it can fail the requirement and cannot pass it. The memory row
+ covers a process holding the catalog and nothing else — no toolkit, no adapter — so it is the
+ catalog layer's share of the 500 MB rather than the figure NFR-P8 is about. Neither carries a
+ `TRACES:` tag, which is the point.
+- **NFR-P8 needs a decision, not more code.** How much of its 500 MB belongs below the UI is
+ unstated, and until somebody says, the metric can record but not judge. [benchmarks.md](benchmarks.md)
+ also answers the question §4.1 raises about GPU memory — RSS cannot see device-local allocations
+ at all — and recommends restating the requirement as two figures.
+
+**What is left is the frame-timing half, and it is the hard one.** NFR-P2, -P4, -P5, -P6, -P9, -P10,
+-P11, -P12, -P13, -P14 and -P15 all need a probe inside a running Slint application, a GPU adapter,
+or both. `dr-gpu/examples/frame_budget` is a real instrument for the GPU part and its results are
+committed in [frame-budget.md](frame-budget.md) with the machine and profile named — but it is run
+by hand, and the guard version in CI skips itself where there is no adapter, which is the normal
+case on a runner. So the claim to take from this section is now narrower than it was, and still
+true: **a scroll that dropped to 30 fps tomorrow would reach a user before it reached CI.**
---
diff --git a/tools/bench/Cargo.toml b/tools/bench/Cargo.toml
new file mode 100644
index 0000000..2ed24d8
--- /dev/null
+++ b/tools/bench/Cargo.toml
@@ -0,0 +1,28 @@
+[package]
+name = "dr-bench"
+version.workspace = true
+edition.workspace = true
+rust-version.workspace = true
+license.workspace = true
+
+# The suite deliberately depends on no GPU crate and no UI toolkit.
+#
+# That is not tidiness, it is what makes the CI job affordable: `cargo build
+# --release -p dr-bench` resolves the catalog, the decoder, the thumbnail store
+# and the encoder, and stops there. Adding `dr-gpu` would pull wgpu and naga
+# into a job that has no adapter to use them with, and adding `dr-ui` would
+# pull Slint. The GPU half of the suite is `dr-gpu`'s own frame-budget test,
+# which already exists and already skips itself where there is no device — see
+# `docs/benchmarks.md`.
+[dependencies]
+dr-types.workspace = true
+dr-catalog.workspace = true
+dr-thumbs.workspace = true
+dr-decode.workspace = true
+dr-export.workspace = true
+rusqlite.workspace = true
+serde.workspace = true
+serde_json.workspace = true
+anyhow.workspace = true
+log.workspace = true
+env_logger.workspace = true
diff --git a/tools/bench/src/baseline.rs b/tools/bench/src/baseline.rs
new file mode 100644
index 0000000..d8321b1
--- /dev/null
+++ b/tools/bench/src/baseline.rs
@@ -0,0 +1,300 @@
+//! The committed numbers, and what counts as a regression against them.
+//!
+//! `docs/frame-budget.md` commits its measurements by hand and says why: *"a
+//! regression should be a diff rather than somebody's memory."* This is the
+//! same idea in a form a program can read, because §8 asks for more than a
+//! record — *"a regression beyond stated tolerance fails the build"*.
+//!
+//! # Two gates, and they are not the same gate
+//!
+//! **The budget** is the requirement's own number: 2 s to open a catalog, 100
+//! images a second through the preview path. It does not move. A build that
+//! violates it has violated a requirement, and no amount of "but it was always
+//! like that" changes it.
+//!
+//! **The baseline** is what this machine last measured. It moves — deliberately
+//! and by hand, through `dr-bench record` — and its job is to catch the change
+//! that is still inside the budget but has halved the headroom. Most real
+//! performance rot arrives that way: never over the line, always a little
+//! worse, until one day the line is crossed by a change that was not the cause.
+//!
+//! # Why a machine-sensitive metric skips its budget off the reference desktop
+//!
+//! §8 names *"the reference desktop"*, not CI, and it is right to. A container
+//! with two cores cannot speak to a throughput target written for a
+//! twenty-four-thread machine, and asserting it there would produce exactly
+//! what `core/dr-gpu/tests/frame_budget.rs` refused to produce: *"a red suite
+//! that everyone learns to ignore"*. So a metric declares whether its budget
+//! is machine-sensitive. Those budgets are asserted under `--reference` and
+//! reported everywhere else; the ones with orders of magnitude of headroom —
+//! catalog open against two seconds — are asserted everywhere, because a
+//! failure there is a real failure on any machine.
+//!
+//! # Why the committed file starts with no numbers in it
+//!
+//! Because nobody had run it yet. Writing plausible-looking figures into a
+//! baseline is the one thing that would make the whole suite worthless: every
+//! later comparison would be against a guess, and the first genuine regression
+//! would be invisible or, worse, a fabricated improvement. `recorded` is
+//! therefore `null` until somebody runs `dr-bench record --reference` on the
+//! reference desktop and commits the diff. Until then the budget gate works
+//! and the regression gate says so rather than pretending.
+
+use std::collections::BTreeMap;
+use std::path::{Path, PathBuf};
+
+use anyhow::{Context, Result};
+use serde::{Deserialize, Serialize};
+
+use crate::fixture::Stamp;
+
+/// Which way is better.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
+#[serde(rename_all = "snake_case")]
+pub enum Direction {
+ LowerIsBetter,
+ HigherIsBetter,
+}
+
+/// One measured quantity: what it is, what it must be, and what it was.
+#[derive(Debug, Clone, Serialize, Deserialize)]
+pub struct Metric {
+ /// The requirement ID this speaks to, or a note saying it speaks to only
+ /// part of one. Free text on purpose: several of these describe a fraction
+ /// of a requirement, and a bare ID here would read as the whole of it.
+ pub requirement: String,
+ /// One sentence a reader of the JSON can understand without the code.
+ pub what: String,
+ pub unit: String,
+ pub direction: Direction,
+ /// Whether the budget below is a statement about a machine as much as
+ /// about the code. See this module's header.
+ pub machine_sensitive: bool,
+ /// The requirement's own threshold, where it has one this can check.
+ pub budget: Option,
+ /// What the last `record` measured. `null` until one has been taken.
+ pub recorded: Option,
+}
+
+/// The committed file.
+#[derive(Debug, Clone, Serialize, Deserialize)]
+pub struct Baseline {
+ /// Prose for whoever opens the JSON first. Kept in the file rather than
+ /// only in `docs/benchmarks.md`, because the person who finds this in a
+ /// failing CI log is not reading the docs directory at that moment.
+ #[serde(rename = "_readme")]
+ pub readme: Vec,
+ /// How much worse than [`Metric::recorded`] a metric may get before the
+ /// build fails, as a fraction. 0.15 is fifteen per cent.
+ pub tolerance: f64,
+ /// The machine the recorded figures came from, as [`machine_id`] spells
+ /// it. Compared before a regression is judged: drift against a different
+ /// machine's numbers is not a regression, it is a different machine.
+ pub recorded_on: Option,
+ /// Unix seconds. An integer rather than a formatted date because this
+ /// workspace has no date library and adding one for a comment would be a
+ /// poor trade.
+ pub recorded_at_unix: Option,
+ /// The fixture the recorded figures describe. A number measured against a
+ /// different workload is not comparable, and this is what says so.
+ pub fixture: Option,
+ pub metrics: BTreeMap,
+}
+
+/// How a measurement compares.
+#[derive(Debug, Clone, Copy)]
+pub struct Judgement {
+ /// Past the requirement's own threshold. A build failure.
+ pub over_budget: bool,
+ /// Worse than the recorded baseline by more than the tolerance. A build
+ /// failure.
+ pub regressed: bool,
+ /// Fractional change against the baseline, positive meaning worse.
+ pub drift: Option,
+ /// A budget exists but was not asserted, because it is machine-sensitive
+ /// and this is not the reference desktop.
+ pub budget_deferred: bool,
+}
+
+/// What a judgement is made in the light of.
+#[derive(Debug, Clone, Copy)]
+pub struct Judging {
+ /// This run declares itself the reference desktop.
+ pub reference: bool,
+ /// This run is on the machine the baseline was recorded on.
+ pub same_machine: bool,
+ pub tolerance: f64,
+}
+
+impl Metric {
+ /// Judge `measured` against the budget and the baseline.
+ pub fn judge(&self, measured: f64, cx: Judging) -> Judgement {
+ let violates = |threshold: f64| match self.direction {
+ Direction::LowerIsBetter => measured > threshold,
+ Direction::HigherIsBetter => measured < threshold,
+ };
+ let assert_budget = self.budget.is_some() && (cx.reference || !self.machine_sensitive);
+
+ let drift = match self.recorded {
+ // A recorded zero would divide by nothing, and a recorded figure
+ // of zero is a broken record rather than a very fast one.
+ Some(was) if was > 0.0 => Some(match self.direction {
+ Direction::LowerIsBetter => (measured - was) / was,
+ Direction::HigherIsBetter => (was - measured) / was,
+ }),
+ _ => None,
+ };
+
+ Judgement {
+ over_budget: assert_budget && self.budget.is_some_and(violates),
+ regressed: cx.same_machine && drift.is_some_and(|d| d > cx.tolerance),
+ drift,
+ budget_deferred: self.budget.is_some() && !assert_budget,
+ }
+ }
+}
+
+impl Baseline {
+ pub fn load(path: &Path) -> Result {
+ let text = std::fs::read_to_string(path)
+ .with_context(|| format!("reading the baseline at {}", path.display()))?;
+ serde_json::from_str(&text)
+ .with_context(|| format!("parsing the baseline at {}", path.display()))
+ }
+
+ pub fn save(&self, path: &Path) -> Result<()> {
+ let mut text = serde_json::to_string_pretty(self)?;
+ // A trailing newline, so the file is a well-behaved text file and a
+ // `record` that changed nothing produces an empty diff.
+ text.push('\n');
+ std::fs::write(path, text)
+ .with_context(|| format!("writing the baseline to {}", path.display()))
+ }
+
+ /// Where the committed baseline lives, found the way `tools/traceability`
+ /// finds the repo root: by walking up from this crate's manifest until
+ /// `docs/requirements.md` appears.
+ pub fn default_path() -> Result {
+ let mut dir = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
+ while !dir.join("docs/requirements.md").exists() {
+ if !dir.pop() {
+ anyhow::bail!("could not locate the repo root above this crate");
+ }
+ }
+ Ok(dir.join("docs/bench-baseline.json"))
+ }
+}
+
+/// How this machine is named in the baseline.
+///
+/// Host name plus thread count. Not a hardware inventory — it exists to answer
+/// one question, "are these numbers from here?", and to answer it the same way
+/// twice on the same box. A container whose hostname changes per run therefore
+/// never matches, which is the correct answer for CI: its drift is information,
+/// not a verdict.
+pub fn machine_id() -> String {
+ let host = std::fs::read_to_string("/proc/sys/kernel/hostname")
+ .map(|s| s.trim().to_string())
+ .unwrap_or_else(|_| "unknown-host".to_string());
+ let threads = std::thread::available_parallelism()
+ .map(|n| n.get())
+ .unwrap_or(0);
+ format!("{host} ({threads} threads)")
+}
+
+/// Unix seconds now, or 0 if the clock is before 1970, which it is not.
+pub fn now_unix() -> i64 {
+ std::time::SystemTime::now()
+ .duration_since(std::time::UNIX_EPOCH)
+ .map(|d| d.as_secs() as i64)
+ .unwrap_or(0)
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+
+ fn metric(direction: Direction, budget: Option, recorded: Option) -> Metric {
+ Metric {
+ requirement: "NFR-TEST".into(),
+ what: "a test metric".into(),
+ unit: "ms".into(),
+ direction,
+ machine_sensitive: false,
+ budget,
+ recorded,
+ }
+ }
+
+ fn cx(same_machine: bool) -> Judging {
+ Judging {
+ reference: true,
+ same_machine,
+ tolerance: 0.15,
+ }
+ }
+
+ #[test]
+ fn a_budget_is_directional() {
+ // The bug this exists to prevent: judging a throughput target the way
+ // a latency target is judged, so a suite that got twice as slow passes
+ // and one that got twice as fast fails.
+ let latency = metric(Direction::LowerIsBetter, Some(100.0), None);
+ assert!(latency.judge(101.0, cx(true)).over_budget);
+ assert!(!latency.judge(99.0, cx(true)).over_budget);
+
+ let throughput = metric(Direction::HigherIsBetter, Some(100.0), None);
+ assert!(throughput.judge(99.0, cx(true)).over_budget);
+ assert!(!throughput.judge(101.0, cx(true)).over_budget);
+ }
+
+ #[test]
+ fn drift_is_positive_when_things_got_worse_whichever_way_that_is() {
+ let latency = metric(Direction::LowerIsBetter, None, Some(100.0));
+ assert!(latency.judge(120.0, cx(true)).drift.unwrap() > 0.0);
+ let throughput = metric(Direction::HigherIsBetter, None, Some(100.0));
+ assert!(throughput.judge(80.0, cx(true)).drift.unwrap() > 0.0);
+ }
+
+ #[test]
+ fn drift_against_another_machine_is_reported_but_never_a_failure() {
+ // CI is not the reference desktop. Its numbers are worth printing and
+ // are not a verdict on anybody's commit.
+ let m = metric(Direction::LowerIsBetter, None, Some(100.0));
+ let elsewhere = m.judge(400.0, cx(false));
+ assert!(elsewhere.drift.unwrap() > 0.15);
+ assert!(!elsewhere.regressed);
+ assert!(m.judge(400.0, cx(true)).regressed);
+ }
+
+ #[test]
+ fn an_unrecorded_metric_cannot_regress() {
+ // The state the committed file ships in. It must gate on the budget
+ // and stay silent about drift, rather than treating null as zero and
+ // declaring an infinite regression.
+ let m = metric(Direction::LowerIsBetter, Some(100.0), None);
+ let j = m.judge(50.0, cx(true));
+ assert!(j.drift.is_none());
+ assert!(!j.regressed);
+ assert!(!j.over_budget);
+ }
+
+ #[test]
+ fn a_machine_sensitive_budget_defers_off_the_reference_desktop() {
+ let mut m = metric(Direction::HigherIsBetter, Some(100.0), None);
+ m.machine_sensitive = true;
+ let on_ci = Judging {
+ reference: false,
+ same_machine: false,
+ tolerance: 0.15,
+ };
+ let j = m.judge(10.0, on_ci);
+ // A two-core runner must not fail a target written for twenty-four.
+ assert!(!j.over_budget);
+ // And the report has to say the budget was not applied, rather than
+ // letting a deferred budget read as a passed one.
+ assert!(j.budget_deferred);
+ // On the reference desktop the same figure is judged.
+ assert!(m.judge(10.0, cx(false)).over_budget);
+ }
+}
diff --git a/tools/bench/src/catalog_open.rs b/tools/bench/src/catalog_open.rs
new file mode 100644
index 0000000..791cc53
--- /dev/null
+++ b/tools/bench/src/catalog_open.rs
@@ -0,0 +1,178 @@
+//! TRACES: NFR-P1
+//! Opening a fifty-thousand-image catalog, and what that actually involves.
+//!
+//! NFR-P1 says under two seconds on the reference desktop. Until this file
+//! existed the number had never been measured, which made it a wish — and
+//! `dr-catalog`'s `lib.rs` has carried an `NFR-P1` tag the whole time on code
+//! that describes the target rather than checking it. This is the check.
+//!
+//! # What counts as "open"
+//!
+//! Not `Catalog::open` alone. That call returns before anything is on screen,
+//! and a user's "the catalog opened" is the moment the grid has cells in it.
+//! So the measured span is the four things the library view cannot paint
+//! without:
+//!
+//! 1. [`Catalog::open`] — connect, migrate if needed, and **backfill**. The
+//! backfill is the interesting one: `schema::backfill` runs on every open
+//! and is three passes over the images table, so it is O(library) work on a
+//! path whose budget is stated in absolute seconds.
+//! 2. [`Catalog::count`] — the total, which is what sizes the scrollbar.
+//! 3. [`Catalog::window`] — the first screenful of rows.
+//! 4. [`Catalog::timeline`] — the scrubber's buckets, drawn beside the grid
+//! from the first frame.
+//!
+//! `open` alone is reported separately anyway, because if the two ever diverge
+//! sharply the fix is in a different place.
+//!
+//! # What this does *not* measure, said out loud
+//!
+//! `ui/dr-ui/src/library.rs` does not call [`Catalog::count`] or
+//! [`Catalog::window`]. It issues its own SQL — `total_images_scoped`,
+//! `total_images_filtered` and friends — against the same tables, with a
+//! `VISIBLE` predicate and a burst-folding clause that this crate cannot see
+//! without depending on the UI, which would drag Slint into a benchmark job
+//! that has no display. So the number here is the **catalog crate's** open
+//! path, and the application's is that plus whatever those queries cost.
+//!
+//! That gap is a real limit on what this file can certify, and it has a
+//! falsifiable end: when the grid's queries move down into `dr-catalog` —
+//! which is where SQL over catalog tables belongs — this measurement becomes
+//! the whole of the application's open, and the caveat can be deleted rather
+//! than argued about.
+//!
+//! # Cold and warm
+//!
+//! Both are reported. The first open in a process pays for SQLite's page cache
+//! being empty and for the schema being read; the second pays for neither, and
+//! is what a user gets when they close and reopen a library in the same
+//! session. §4.1 asks the question directly for NFR-P8 and it is worth having
+//! the answer here too. Neither figure is a genuinely cold *disk*: the fixture
+//! was written by this same suite or by an earlier run of it, so the file is
+//! in the OS page cache. On the reference desktop's NVMe a truly cold read of
+//! a ~14 MB file is a few tens of milliseconds; on a spinning disk it is not.
+
+use std::path::Path;
+use std::time::Instant;
+
+use anyhow::Result;
+use dr_catalog::{Catalog, Granularity, Query};
+use dr_types::Selector;
+
+use crate::stats::{ms, Percentiles, Rng};
+
+/// Rows fetched for the first screenful.
+///
+/// A dense grid on a 4K display is around three hundred cells; four hundred is
+/// that plus the prefetch margin R2 asks for. Not the whole library, because
+/// FR-CAT-4 is explicit that memory must not scale with it — a benchmark that
+/// asked for 50,000 rows would be measuring the requirement's violation.
+const WINDOW: usize = 400;
+
+/// The clock the query compiler is handed.
+///
+/// Fixed rather than read from the system, so a rolling date filter would
+/// compile to the same SQL on every run. The unfiltered query does not consult
+/// it at all; this is here so that adding a dated row later does not silently
+/// make the suite time-dependent.
+const NOW: i64 = 2_000_000_000;
+
+/// What one measured open produced.
+pub struct Open {
+ /// Open, count, first window, timeline — the whole span, first time.
+ pub cold_ms: f64,
+ /// The same four calls on a second connection in the same process.
+ pub warm_ms: f64,
+ /// [`Catalog::open`] on its own, out of the cold span.
+ pub open_only_ms: f64,
+ /// How many images the count found. Reported so a fixture that failed to
+ /// populate cannot masquerade as a very fast open.
+ pub images: usize,
+ /// Timeline buckets at monthly granularity.
+ pub buckets: usize,
+ /// One window fetched at a random offset — the scroll, minus the drawing.
+ pub window_ms: Percentiles,
+ /// Count plus first window under a rating filter, which compiles to a
+ /// correlated subquery over `versions` (see `query::default_version_scalar`).
+ pub filtered_ms: f64,
+}
+
+/// Measure an open of the catalog at `path`, then `windows` random windows.
+pub fn measure(path: &Path, windows: usize) -> Result {
+ let q = Query::default();
+
+ let started = Instant::now();
+ let catalog = open(path)?;
+ let open_only_ms = ms(started.elapsed());
+ let images = catalog.count(&q, NOW)?;
+ let rows = catalog.window(&q, 0..WINDOW, NOW)?;
+ let buckets = catalog.timeline(&q, Granularity::Month, NOW)?.len();
+ let cold_ms = ms(started.elapsed());
+
+ // A catalog that returned nothing would post an excellent time. Checked
+ // rather than trusted, because the failure mode is a *fast* wrong answer.
+ anyhow::ensure!(
+ !rows.is_empty() && images > 0 && buckets > 0,
+ "the fixture catalog answered with {images} images, {} rows and {buckets} buckets — \
+ the measurement below would be meaningless",
+ rows.len()
+ );
+ drop(catalog);
+
+ let started = Instant::now();
+ let catalog = open(path)?;
+ let _ = catalog.count(&q, NOW)?;
+ let _ = catalog.window(&q, 0..WINDOW, NOW)?;
+ let _ = catalog.timeline(&q, Granularity::Month, NOW)?;
+ let warm_ms = ms(started.elapsed());
+
+ // The scroll. Offsets are drawn from a fixed seed rather than swept in
+ // order, because a sequential sweep would be answered increasingly out of
+ // SQLite's own cache and would flatter the deep end of the library — which
+ // is exactly the end a person reaches by dragging the scrollbar.
+ let mut rng = Rng::new(0x5C_20_11);
+ let span = images.saturating_sub(WINDOW).max(1) as u64;
+ // Discarded: the first window of a new connection compiles the statement
+ // and faults in the b-tree's upper levels, and neither recurs while
+ // scrolling.
+ for _ in 0..4 {
+ let start = rng.below(span) as usize;
+ let _ = catalog.window(&q, start..start + WINDOW, NOW)?;
+ }
+ let mut samples = Vec::with_capacity(windows);
+ for _ in 0..windows {
+ let start = rng.below(span) as usize;
+ let t = Instant::now();
+ let rows = catalog.window(&q, start..start + WINDOW, NOW)?;
+ samples.push(ms(t.elapsed()));
+ debug_assert!(!rows.is_empty());
+ }
+
+ // A filter that has to reach the default version for every candidate row.
+ // Cheap to add and the one query shape in the grid that is not a scan of
+ // `images` alone, so a regression in it would otherwise show up first as a
+ // user complaint.
+ let rated = Query {
+ filter: Selector::Rating { min: 2 },
+ ..Query::default()
+ };
+ let t = Instant::now();
+ let _ = catalog.count(&rated, NOW)?;
+ let _ = catalog.window(&rated, 0..WINDOW, NOW)?;
+ let filtered_ms = ms(t.elapsed());
+
+ Ok(Open {
+ cold_ms,
+ warm_ms,
+ open_only_ms,
+ images,
+ buckets,
+ window_ms: Percentiles::of(samples),
+ filtered_ms,
+ })
+}
+
+fn open(path: &Path) -> Result {
+ Catalog::open(path)
+ .map_err(|e| anyhow::anyhow!("opening the fixture catalog at {}: {e}", path.display()))
+}
diff --git a/tools/bench/src/exporting.rs b/tools/bench/src/exporting.rs
new file mode 100644
index 0000000..16dcc6e
--- /dev/null
+++ b/tools/bench/src/exporting.rs
@@ -0,0 +1,139 @@
+//! The half of a 24 MP export that needs no GPU.
+//!
+//! # This cannot certify NFR-P7, and is not tagged as though it could
+//!
+//! NFR-P7 is "full-resolution export (24 MP, **full chain**) < 2 s". The full
+//! chain is decode, demosaic, a GPU render at full resolution, a read-back,
+//! and then everything `dr-export` does — resize, output sharpening, encode.
+//! Only the last three of those run without an adapter, and the CI runner has
+//! none. So what is measured here is the encode half, and no requirement tag
+//! anywhere in this crate names NFR-P7.
+//!
+//! (Written without the tag's own spelling on purpose. `tools/traceability`
+//! matches the marker anywhere on a line and parses the identifier after it, so
+//! a sentence saying "there is no tag for NFR-P7" would *be* a tag for NFR-P7 —
+//! a disclaimer that made itself false.)
+//!
+//! That is a deliberate refusal rather than an oversight. `CONTRIBUTING.md`
+//! asks that a requirement be closed by a test that would fail if the
+//! behaviour were removed, and `docs/code-health.md` CH-4 records what the
+//! coverage figure looks like when tags are hung on plumbing instead. A tag
+//! here would say the export budget is checked; the GPU half of it would still
+//! be unchecked.
+//!
+//! # What the number is still good for
+//!
+//! It is a **one-sided** gate, and that is worth having. The encode half is a
+//! lower bound on the whole: if resizing, sharpening and encoding 24 MP alone
+//! take longer than two seconds, NFR-P7 is violated no matter how fast the
+//! render is. So the budget in `docs/bench-baseline.json` is the requirement's
+//! own 2000 ms, and exceeding it fails the build honestly. Coming in under it
+//! proves nothing about the requirement, and the report says so rather than
+//! printing a tick.
+//!
+//! # Two rows
+//!
+//! `Original` is the one the budget is judged on: it is the archival export,
+//! the largest encode, and the case FR-EXP-9 is about. `LongEdge(2048)` is the
+//! ordinary web export, where the resample does real work and the encode does
+//! very little — it is reported because a regression in `size::resample` would
+//! be invisible in the first row, where source and target dimensions are equal.
+
+use std::time::Instant;
+
+use anyhow::Result;
+use dr_export::{export, Frame};
+use dr_types::{ExportFormat, ExportSettings, OutputSharpening, SizingMode};
+
+use crate::fixture::plausible_frame;
+use crate::stats::{ms, Percentiles};
+
+/// The frame every row exports.
+///
+/// 6000 × 4000 is 24.0 MP — a full-frame body, and the exact figure NFR-P7
+/// names. As RGBA8 it is 96 MB, and the export path holds a resized copy and a
+/// sharpened copy alongside it, so a run needs roughly 300 MB of headroom.
+/// Worth knowing before a small runner reports this as a mysterious kill.
+pub const SOURCE: (u32, u32) = (6000, 4000);
+
+/// Measured exports per row. Not a hundred: one 24 MP encode is most of a
+/// second, and a hundred of them would be a two-minute CI step to establish
+/// what five establish. Nearest-rank p99 of five is the worst of the five,
+/// which for a row this expensive is the honest reading anyway.
+const RUNS: usize = 5;
+
+/// A row of the export table.
+pub struct EncodeRun {
+ /// The name this row carries in `docs/bench-baseline.json`.
+ pub key: &'static str,
+ pub label: &'static str,
+ pub width: u32,
+ pub height: u32,
+ /// Encoded file size, so a row that silently stopped compressing is
+ /// visible as well as a row that got slow.
+ pub bytes: usize,
+ pub times: Percentiles,
+}
+
+/// Export the same 24 MP frame at each sizing, timing `dr_export::export`.
+pub fn measure() -> Result> {
+ let (w, h) = SOURCE;
+ // Detail at every scale, for the same reason the thumbnail fixture has it:
+ // a flat frame compresses to almost nothing and would make the encoder
+ // look several times faster than any photograph makes it.
+ let frame = Frame::new(w, h, plausible_frame(w, h, 0))
+ .map_err(|e| anyhow::anyhow!("building the 24 MP bench frame: {e}"))?;
+
+ let sizings: [(&'static str, &'static str, SizingMode); 2] = [
+ ("export_24mp_original_ms", "original", SizingMode::Original),
+ (
+ "export_24mp_long_edge_2048_ms",
+ "long edge 2048",
+ SizingMode::LongEdge(2048),
+ ),
+ ];
+
+ let mut rows = Vec::with_capacity(sizings.len());
+ for (key, label, sizing) in sizings {
+ let settings = ExportSettings {
+ format: ExportFormat::Jpeg,
+ // 90 is the default and what a photographer would not need to
+ // change; quality moves encode time, so it belongs in the record.
+ quality: 90,
+ sizing,
+ sharpening: OutputSharpening::Screen,
+ ..Default::default()
+ };
+
+ // Discarded. The first export of a process grows the allocator to hold
+ // three 24 MP buffers, which is a cost paid once and not per file in
+ // the batch export FR-EXP-7 describes.
+ let warm = run_once(&frame, &settings)?;
+
+ let mut samples = Vec::with_capacity(RUNS);
+ let mut last = warm;
+ for _ in 0..RUNS {
+ let started = Instant::now();
+ last = run_once(&frame, &settings)?;
+ samples.push(ms(started.elapsed()));
+ }
+
+ rows.push(EncodeRun {
+ key,
+ label,
+ width: last.0,
+ height: last.1,
+ bytes: last.2,
+ times: Percentiles::of(samples),
+ });
+ }
+
+ Ok(rows)
+}
+
+/// One export, returning what it produced rather than the pixels.
+fn run_once(frame: &Frame, settings: &ExportSettings) -> Result<(u32, u32, usize)> {
+ let encoded = export(frame, settings, "bench.jpg".to_string(), None)
+ .map_err(|e| anyhow::anyhow!("exporting the bench frame: {e}"))?;
+ Ok((encoded.width, encoded.height, encoded.bytes.len()))
+}
diff --git a/tools/bench/src/fixture.rs b/tools/bench/src/fixture.rs
new file mode 100644
index 0000000..fa292d3
--- /dev/null
+++ b/tools/bench/src/fixture.rs
@@ -0,0 +1,448 @@
+//! The synthetic 50k catalog, and the handful of real files it points at.
+//!
+//! `docs/requirements.md` §8 asks for "an automated benchmark suite against a
+//! synthetic 50k catalog". The hard part of that sentence is *50k*: a real
+//! library of that size is several terabytes and cannot live in a repository,
+//! in a CI cache, or on a laptop that also has to compile the thing.
+//!
+//! # The trick, and what it costs
+//!
+//! Rows are cheap and pixels are not. So this builds **fifty thousand catalog
+//! rows** over a **pool of a dozen real image files**, each referenced by
+//! several thousand of them. Everything the catalog half of the suite measures
+//! — opening, counting, windowing, bucketing a timeline — touches only rows,
+//! and is therefore exact. Everything the pixel half measures — decode,
+//! downscale, orient, encode — touches one file at a time and does not care
+//! how many rows point at it. The fixture is ~14 MB on disk instead of ~2 TB
+//! and neither half is flattered by that.
+//!
+//! What it *does* cost is stated rather than hidden: the file pool is small
+//! enough to sit in the OS page cache, so [`crate::thumbnails`] measures CPU
+//! throughput with the read already paid for. That is the right thing to
+//! measure for NFR-P3 — the target is written about the embedded preview path,
+//! not about a disk — but it is not a claim about a cold library on spinning
+//! rust, and the harness does not make one.
+//!
+//! # Reproducible from a seed
+//!
+//! Every value comes from [`Rng`], seeded once. Two machines running the same
+//! seed build byte-comparable catalogs, which is the property that lets a
+//! number measured on the reference desktop be compared with a number measured
+//! anywhere else. [`Stamp`] records what a directory was built from, so a
+//! fixture is reused when it matches and rebuilt when it does not — including
+//! when `dr-catalog`'s schema version moves, since a catalog built by an older
+//! build would otherwise be measured through a migration that a user's would
+//! not run.
+//!
+//! # The sources are generated, not committed
+//!
+//! No photograph in this repository is licensed for redistribution, and a
+//! dozen camera previews would be megabytes of binary in git for ever. So the
+//! pool is synthesised: a coarse gradient with a fine dither on top, which is
+//! the same shape `core/dr-gpu/examples/frame_budget.rs` synthesises its source
+//! from and for the same reason. A flat frame lets the memory system serve
+//! every sample from one cache line, which flatters a box filter by an amount
+//! that has nothing to do with photographs; pure noise defeats the JPEG
+//! encoder's entropy coder in the other direction and would make the encode
+//! half of a thumbnail look worse than any real image ever does.
+
+use std::path::{Path, PathBuf};
+
+use anyhow::{Context, Result};
+use dr_catalog::Catalog;
+use serde::{Deserialize, Serialize};
+
+use crate::stats::Rng;
+
+/// How many distinct image files the pool holds.
+///
+/// Twelve rather than one, so that a decode measured over a batch is not one
+/// file's quirks repeated — a single frame that happened to compress unusually
+/// well would set the whole number — and rather than fifty thousand, so the
+/// fixture stays a directory a person can look at.
+pub const SOURCE_POOL: usize = 12;
+
+/// The size of one pooled file, in pixels.
+///
+/// 1620×1080 is not a round number: it is what `core/dr-decode/src/preview.rs`
+/// records a Canon CR2 carrying in IFD2, and the embedded preview is what
+/// NFR-P3 names. A camera JPEG is 24 MP and a camera *preview* is about this,
+/// so measuring the preview path against a 24 MP file would measure something
+/// the sweep never does.
+pub const PREVIEW: (u32, u32) = (1620, 1080);
+
+/// Folders the rows are spread across.
+///
+/// Spread evenly and without regard to capture date, because what a folder
+/// count decides is the cost of the folder filter's `IN (SELECT …)` and the
+/// size of the `folders` table — not which image is in which.
+const FOLDERS: usize = 400;
+
+/// The earliest capture time in the fixture: 13 December 2015, UTC.
+///
+/// Fixed rather than relative to the clock. A library whose dates moved with
+/// the calendar would make `timeline` bucket differently from one month to the
+/// next, and a benchmark that measures a different query each time it runs is
+/// not measuring a regression.
+const EPOCH: i64 = 1_450_000_000;
+
+/// The span capture times are drawn from: twelve years.
+///
+/// Long enough that the timeline query has real structure to bucket — at
+/// monthly granularity that is ~144 buckets, which is the shape the scrubber
+/// actually draws — and not so long that a year holds too few frames to look
+/// like a library.
+const SPAN: i64 = 12 * 365 * 86_400;
+
+/// Bodies and lenses, for the columns the camera and lens filters read.
+const CAMERAS: [&str; 6] = [
+ "Canon EOS R5",
+ "Nikon Z 7II",
+ "Sony ILCE-7RM5",
+ "Fujifilm X-T5",
+ "Panasonic DC-S5M2",
+ "OM SYSTEM OM-1",
+];
+
+const LENSES: [&str; 6] = [
+ "RF24-70mm F2.8 L IS USM",
+ "NIKKOR Z 50mm f/1.8 S",
+ "FE 85mm F1.4 GM",
+ "XF16-55mmF2.8 R LM WR",
+ "LUMIX S 20-60mm F3.5-5.6",
+ "M.Zuiko Digital ED 12-40mm F2.8",
+];
+
+/// What a fixture directory was built from.
+///
+/// Written beside the catalog and compared on every run. A mismatch rebuilds:
+/// silently reusing a fixture built from a different seed, a different row
+/// count or an older schema would compare two numbers that describe two
+/// different workloads, which is worse than having no number at all.
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+pub struct Stamp {
+ /// Bumped by hand whenever anything in this file changes what gets built.
+ /// The seed cannot carry that: the same seed through different generation
+ /// code produces a different library.
+ pub generator: u32,
+ pub seed: u64,
+ pub images: usize,
+ pub sources: usize,
+ pub folders: usize,
+ pub preview_width: u32,
+ pub preview_height: u32,
+ /// `dr-catalog`'s schema version at build time.
+ pub schema_version: i64,
+}
+
+/// Bump on any change to what [`build`] writes.
+const GENERATOR: u32 = 1;
+
+/// A built fixture on disk.
+pub struct Fixture {
+ pub dir: PathBuf,
+ pub catalog: PathBuf,
+ pub thumbs: PathBuf,
+ pub sources: Vec,
+ pub stamp: Stamp,
+ /// The catalog file's size, reported because it is the thing an open has
+ /// to read and because it is the honest denominator for "is 2 s a lot".
+ pub catalog_bytes: u64,
+}
+
+/// Where a fixture lives by default.
+///
+/// The temporary directory rather than `target/`, for two reasons. It survives
+/// `cargo clean`, so a fixture is built once per machine rather than once per
+/// clean; and it is not inside anything CI caches, so a 14 MB catalog is not
+/// uploaded and downloaded on every push to save the two seconds it takes to
+/// generate. `DR_BENCH_DIR` overrides it.
+pub fn default_dir() -> PathBuf {
+ match std::env::var_os("DR_BENCH_DIR") {
+ Some(dir) => PathBuf::from(dir),
+ None => std::env::temp_dir().join("darkroom-bench"),
+ }
+}
+
+/// Build the fixture under `dir`, or confirm the one already there.
+///
+/// Returns whether it had to be built, so the caller can say so: a run that
+/// includes fixture generation has a warm page cache for the catalog file it
+/// is about to open, and a reader comparing two numbers deserves to know which
+/// of them was measured that way.
+pub fn build(dir: &Path, seed: u64, images: usize) -> Result<(Fixture, bool)> {
+ let stamp = Stamp {
+ generator: GENERATOR,
+ seed,
+ images,
+ sources: SOURCE_POOL,
+ folders: FOLDERS,
+ preview_width: PREVIEW.0,
+ preview_height: PREVIEW.1,
+ schema_version: dr_catalog::schema::SCHEMA_VERSION,
+ };
+
+ let catalog = dir.join("catalog.sqlite");
+ let stamp_path = dir.join("stamp.json");
+ let sources: Vec = (0..SOURCE_POOL)
+ .map(|i| dir.join("sources").join(format!("preview-{i:02}.jpg")))
+ .collect();
+
+ let usable = matches_stamp(&stamp_path, &stamp)
+ && catalog.is_file()
+ && sources.iter().all(|p| p.is_file());
+
+ if !usable {
+ std::fs::create_dir_all(dir.join("sources"))
+ .with_context(|| format!("creating the fixture directory {}", dir.display()))?;
+ // The stamp goes last. A build interrupted halfway leaves no stamp, so
+ // the next run rebuilds rather than measuring a truncated catalog.
+ let _ = std::fs::remove_file(&stamp_path);
+ write_sources(&sources, seed)?;
+ write_catalog(&catalog, seed, images)?;
+ std::fs::write(&stamp_path, serde_json::to_vec_pretty(&stamp)?)
+ .with_context(|| format!("writing {}", stamp_path.display()))?;
+ }
+
+ let catalog_bytes = std::fs::metadata(&catalog)
+ .with_context(|| format!("stat {}", catalog.display()))?
+ .len();
+
+ Ok((
+ Fixture {
+ dir: dir.to_path_buf(),
+ catalog,
+ thumbs: dir.join("thumbs"),
+ sources,
+ stamp,
+ catalog_bytes,
+ },
+ !usable,
+ ))
+}
+
+fn matches_stamp(path: &Path, want: &Stamp) -> bool {
+ let Ok(text) = std::fs::read_to_string(path) else {
+ return false;
+ };
+ matches!(serde_json::from_str::(&text), Ok(have) if have == *want)
+}
+
+// ---------------------------------------------------------------------------
+// The file pool
+// ---------------------------------------------------------------------------
+
+/// Write the pool of JPEGs the pixel half decodes.
+///
+/// Encoded through [`dr_thumbs::encode_rgba`] rather than a second encoder
+/// call of this crate's own. That is the quality the store already uses (82),
+/// which is a little below what a camera writes its previews at, and it is one
+/// fewer place for an encoder setting to drift. Stated because it is visible
+/// in the result: a slightly softer source decodes marginally faster than a
+/// camera's own preview would.
+fn write_sources(paths: &[PathBuf], seed: u64) -> Result<()> {
+ let (w, h) = PREVIEW;
+ for (i, path) in paths.iter().enumerate() {
+ let rgba = plausible_frame(w, h, seed ^ (i as u64));
+ let jpeg = dr_thumbs::encode_rgba(w, h, &rgba)
+ .map_err(|e| anyhow::anyhow!("encoding the fixture source {}: {e}", path.display()))?;
+ std::fs::write(path, jpeg).with_context(|| format!("writing {}", path.display()))?;
+ }
+ Ok(())
+}
+
+/// RGBA with detail at every scale: a coarse gradient plus a fine dither.
+///
+/// See this module's header for why neither a flat frame nor pure noise would
+/// do. `salt` moves the gradient and the dither together so the twelve files
+/// differ from one another rather than being twelve copies with different
+/// names — a JPEG encoder that saw the same image twelve times would have the
+/// same cache behaviour every time, which a library does not.
+pub fn plausible_frame(w: u32, h: u32, salt: u64) -> Vec {
+ let mut rgba = vec![0u8; (w as usize) * (h as usize) * 4];
+ let bias = (salt % 97) as u32;
+ for y in 0..h as usize {
+ let row = y * (w as usize) * 4;
+ for x in 0..w as usize {
+ // A cheap integer hash, so neighbouring pixels differ and the
+ // encoder has real high-frequency content to spend bits on.
+ let n = (x.wrapping_mul(2_654_435_761) ^ y.wrapping_mul(1_640_531_527)) >> 13;
+ let dither = (n & 0x1f) as u32;
+ let gx = (x * 200 / (w as usize).max(1)) as u32;
+ let gy = (y * 55 / (h as usize).max(1)) as u32;
+ let px = &mut rgba[row + x * 4..row + x * 4 + 4];
+ px[0] = (30 + bias + gx + dither).min(255) as u8;
+ px[1] = (40 + gy + dither).min(255) as u8;
+ px[2] = (60 + gx / 2 + gy + dither).min(255) as u8;
+ px[3] = 255;
+ }
+ }
+ rgba
+}
+
+// ---------------------------------------------------------------------------
+// The catalog
+// ---------------------------------------------------------------------------
+
+/// Write a catalog holding `images` rows, plus a default version for each.
+///
+/// The default versions are not decoration. `Catalog::open` backfills them for
+/// any image that lacks one (see `schema::backfill`), so a fixture without
+/// them would charge every measured open for fifty thousand inserts once and
+/// nothing thereafter — a first number that bore no relation to the second,
+/// and a benchmark whose result depended on whether it had been run before.
+fn write_catalog(path: &Path, seed: u64, images: usize) -> Result<()> {
+ for suffix in ["", "-wal", "-shm"] {
+ let mut p = path.as_os_str().to_os_string();
+ p.push(suffix);
+ let _ = std::fs::remove_file(PathBuf::from(p));
+ }
+
+ let catalog = Catalog::open(path)
+ .map_err(|e| anyhow::anyhow!("creating the fixture catalog at {}: {e}", path.display()))?;
+ let conn = catalog.connection();
+ let mut rng = Rng::new(seed);
+
+ conn.execute_batch("BEGIN")?;
+
+ conn.execute(
+ "INSERT INTO roots(id, kind, label, last_seen, scan_generation)
+ VALUES (1, 'local', '/library', ?1, 1)",
+ rusqlite::params![EPOCH],
+ )?;
+
+ {
+ let mut folder = conn.prepare(
+ "INSERT INTO folders(id, root_id, parent_id, path, mtime, entry_count,
+ scanned_generation)
+ VALUES (?1, 1, NULL, ?2, ?3, ?4, 1)",
+ )?;
+ for f in 0..FOLDERS {
+ let id = f as i64 + 1;
+ let folder_path = format!("/library/{:04}/{:02}", 2016 + f / 12, f % 12 + 1);
+ let mtime = EPOCH + f as i64 * 86_400;
+ let entries = (images / FOLDERS.max(1)) as i64;
+ folder.execute(rusqlite::params![id, folder_path, mtime, entries])?;
+ }
+ }
+
+ {
+ let mut image = conn.prepare(
+ "INSERT INTO images(id, root_id, folder_id, source_ref, format, w, h,
+ captured_at, captured_offset, camera, lens, iso,
+ aperture, shutter, availability, file_size,
+ file_mtime, metadata_state, added_at)
+ VALUES (?1, 1, ?2, ?3, 'CR3', ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11, ?12,
+ ?13, ?14, ?15, ?16, ?17)",
+ )?;
+ let mut version = conn.prepare(
+ "INSERT INTO versions(id, image_id, uuid, name, is_default, rating,
+ label, flag)
+ VALUES (?1, ?1, ?2, 'Original', 1, ?3, ?4, ?5)",
+ )?;
+
+ // Every value is bound to a local before it reaches `params!`. Not
+ // style: the macro takes a reference to each argument, and an
+ // expression like `TABLE[rng.below(n) as usize]` inside it borrows an
+ // element of a temporary array while `rng` is also being borrowed
+ // mutably. Locals make the evaluation order and the lifetimes obvious.
+ const OFFSETS: [i64; 5] = [0, 60, 120, -300, 540];
+ const ISOS: [i64; 7] = [100, 200, 400, 800, 1600, 3200, 6400];
+ const APERTURES: [f64; 6] = [1.4, 1.8, 2.8, 4.0, 5.6, 8.0];
+ const SHUTTERS: [f64; 6] = [0.004, 0.008, 0.0167, 0.005, 0.002, 0.5];
+ // Mostly metadata-only, as a large library on a laptop is: some
+ // previewed, a few with the original present.
+ const AVAILABILITY: [i64; 6] = [0, 0, 0, 1, 1, 2];
+
+ for i in 0..images {
+ let id = i as i64 + 1;
+ let bucket = i % FOLDERS;
+ let folder = bucket as i64 + 1;
+ let source_ref = format!(
+ "/library/{:04}/{:02}/IMG_{id:05}.CR3",
+ 2016 + bucket / 12,
+ bucket % 12 + 1
+ );
+ let captured = EPOCH + rng.below(SPAN as u64) as i64;
+ // A quarter of the library shot in portrait, which is what makes
+ // the thumbnail path's orientation permutation a real cost rather
+ // than a branch that is never taken.
+ let (w, h) = if i % 4 == 3 {
+ (4000i64, 6000i64)
+ } else {
+ (6000i64, 4000i64)
+ };
+ // Two per cent still awaiting full EXIF — a library is never
+ // entirely finished being read, and the grid has to render that
+ // state (`metadata_state` 1).
+ let state: i64 = if rng.below(50) == 0 { 1 } else { 2 };
+ let camera = CAMERAS[rng.below(CAMERAS.len() as u64) as usize];
+ let lens = LENSES[rng.below(LENSES.len() as u64) as usize];
+ // Minutes east of UTC: a library shot in a handful of places.
+ let offset = OFFSETS[rng.below(OFFSETS.len() as u64) as usize];
+ let iso = ISOS[rng.below(ISOS.len() as u64) as usize];
+ let aperture = APERTURES[rng.below(APERTURES.len() as u64) as usize];
+ let shutter = SHUTTERS[rng.below(SHUTTERS.len() as u64) as usize];
+ let availability = AVAILABILITY[rng.below(AVAILABILITY.len() as u64) as usize];
+ let file_size = 20_000_000i64 + rng.below(30_000_000) as i64;
+ let file_mtime = captured + 60;
+ let added_at = captured + 3600;
+
+ image.execute(rusqlite::params![
+ id,
+ folder,
+ source_ref,
+ w,
+ h,
+ captured,
+ offset,
+ camera,
+ lens,
+ iso,
+ aperture,
+ shutter,
+ availability,
+ file_size,
+ file_mtime,
+ state,
+ added_at,
+ ])?;
+
+ // Ratings skewed the way a culled library is: most unrated, a few
+ // picks, fewer still at five stars.
+ let rating: i64 = match rng.below(100) {
+ 0..=69 => 0,
+ 70..=84 => 1,
+ 85..=93 => 2,
+ 94..=97 => 3,
+ 98 => 4,
+ _ => 5,
+ };
+ let flag: i64 = match rng.below(100) {
+ 0..=79 => 0,
+ 80..=94 => 1,
+ _ => 2,
+ };
+ let label: Option = match rng.below(100) {
+ 0..=89 => None,
+ n => Some((n % 5) as i64 + 1),
+ };
+ let a = rng.next_u64();
+ let b = rng.next_u64();
+ let uuid = format!("{a:016x}{b:016x}");
+ version.execute(rusqlite::params![id, uuid, rating, label, flag])?;
+ }
+ }
+
+ conn.execute_batch("COMMIT")?;
+
+ // Deliberately no `ANALYZE`. The application never runs one, so a fixture
+ // that did would be measuring a query plan no user's catalog gets — and a
+ // plan chosen from statistics is exactly the sort of thing that would make
+ // the benchmark faster than the product.
+
+ // Dropping the connection checkpoints the WAL, so the file the next open
+ // reads is the whole catalog rather than a stub plus a journal.
+ drop(catalog);
+ Ok(())
+}
diff --git a/tools/bench/src/main.rs b/tools/bench/src/main.rs
new file mode 100644
index 0000000..ef4e14a
--- /dev/null
+++ b/tools/bench/src/main.rs
@@ -0,0 +1,539 @@
+//! The benchmark suite `docs/requirements.md` §8 has been promising.
+//!
+//! §8 says performance is verified by *"an automated benchmark suite against a
+//! synthetic 50k catalog, run per-commit … A regression beyond stated
+//! tolerance fails the build."* Until this crate there was none: no `benches/`,
+//! no `[[bench]]`, no criterion, no fixture. Ten performance requirements could
+//! therefore be neither passed nor failed, and five of them carried a
+//! requirement tag anyway.
+//!
+//! ```sh
+//! cargo run --release -p dr-bench -- check # measure and gate
+//! cargo run --release -p dr-bench -- check --reference # …on the reference desktop
+//! cargo run --release -p dr-bench -- record --reference # rewrite the baseline
+//! ```
+//!
+//! **Release, always.** The workspace builds its own crates at `opt-level = 0`
+//! in dev (see the root `Cargo.toml`), and every number here is dominated by
+//! this workspace's own code — the JPEG decode, the box filter, the resample,
+//! the sharpen. A debug run measures rustc's shadow, exactly as
+//! `core/dr-gpu/examples/frame_budget.rs` warns for the same reason. The
+//! harness says so at the top of every report rather than trusting anyone to
+//! remember.
+//!
+//! # What this covers, and what it deliberately does not
+//!
+//! Covered, with a gate:
+//!
+//! - **NFR-P1** and R2's catalog clause — opening a 50k catalog and painting
+//! the first grid ([`catalog_open`]).
+//! - **NFR-P3** — thumbnail throughput on the embedded preview path
+//! ([`thumbnails`]).
+//!
+//! Measured, reported, and honestly *not* tagged, because only part of the
+//! requirement is in reach without a GPU or a running UI:
+//!
+//! - **NFR-P7** — the encode half of a 24 MP export ([`exporting`]). A
+//! one-sided gate: it can fail the requirement, it cannot pass it.
+//! - **NFR-P8** — the catalog layer's share of idle RSS ([`memory`]), together
+//! with the answer to the question §4.1 asks about GPU memory.
+//!
+//! Out of scope entirely, and left to say so rather than faked: NFR-P2, P4,
+//! P5, P6, P9, P10, P11, P12, P13, P14 and P15. Every one of them needs a
+//! frame-timing probe inside a running Slint application, a GPU adapter, or
+//! both. The GPU half of the suite that *does* exist is
+//! `core/dr-gpu/tests/frame_budget.rs`, which asserts FR-DSP-3 and skips itself
+//! where there is no adapter; `.gitea/workflows/benchmark.yml` runs it as its
+//! own job for exactly that reason.
+//!
+//! # Exit codes
+//!
+//! `0` everything inside its budget and its tolerance; `1` a gate failed; `2`
+//! the harness itself could not run. Distinguished because a CI log that says
+//! "failed" should not leave anyone guessing whether the code got slower or the
+//! fixture would not build.
+
+mod baseline;
+mod catalog_open;
+mod exporting;
+mod fixture;
+mod memory;
+mod stats;
+mod thumbnails;
+
+use std::collections::BTreeMap;
+use std::path::PathBuf;
+
+use anyhow::Result;
+
+use baseline::{Baseline, Judging};
+
+/// The seed the committed baseline describes.
+///
+/// Changing it invalidates every recorded figure, because it changes the
+/// library being measured. That is why it is a constant here and a field in
+/// the stamp rather than something a flag quietly varies.
+const SEED: u64 = 20_260_829;
+
+/// Rows in the synthetic library. §8 says 50k; this is that.
+const IMAGES: usize = 50_000;
+
+/// Thumbnails produced for the throughput row.
+///
+/// Twelve hundred rather than fifty thousand. At the target rate the whole
+/// library is eight minutes of CI, and a rate measured over 1,200 images is
+/// the same rate — the sweep has no state that changes after the first chunk,
+/// which the per-image percentile alongside it is there to demonstrate.
+const THUMBNAILS: usize = 1_200;
+
+/// Windows fetched for the scroll row.
+///
+/// A hundred, so nearest-rank puts the 99th percentile on the second-worst —
+/// the same reading `core/dr-gpu/examples/frame_budget.rs` takes of a hundred
+/// frames, and the reason it takes it: one bad one in a hundred is one too
+/// many, and a single scheduler hiccup on an unrelated process should not
+/// decide the verdict on its own.
+const WINDOWS: usize = 100;
+
+fn main() {
+ env_logger::init();
+ match run() {
+ Ok(true) => {}
+ Ok(false) => std::process::exit(1),
+ Err(e) => {
+ eprintln!("dr-bench: {e:#}");
+ std::process::exit(2);
+ }
+ }
+}
+
+/// Returns whether every gate passed.
+fn run() -> Result {
+ let mut args = std::env::args().skip(1);
+ let command = args.next().unwrap_or_else(|| "help".to_string());
+
+ // The probe takes a path and nothing else — see `memory` for why it is a
+ // separate process rather than a function call.
+ if command == "memory-probe" {
+ let path = args
+ .next()
+ .ok_or_else(|| anyhow::anyhow!("memory-probe needs a catalog path"))?;
+ memory::probe(&PathBuf::from(path))?;
+ return Ok(true);
+ }
+
+ let mut reference = false;
+ let mut fixture_dir = fixture::default_dir();
+ let mut baseline_path = Baseline::default_path()?;
+ let mut lanes = thumbnails::default_lanes();
+ let mut thumbnail_count = THUMBNAILS;
+
+ while let Some(flag) = args.next() {
+ match flag.as_str() {
+ "--reference" => reference = true,
+ "--fixture" => fixture_dir = PathBuf::from(expect_value(&mut args, "--fixture")?),
+ "--baseline" => baseline_path = PathBuf::from(expect_value(&mut args, "--baseline")?),
+ "--lanes" => lanes = expect_value(&mut args, "--lanes")?.parse()?,
+ "--thumbnails" => thumbnail_count = expect_value(&mut args, "--thumbnails")?.parse()?,
+ other => anyhow::bail!("unknown flag {other}; try `dr-bench help`"),
+ }
+ }
+
+ match command.as_str() {
+ "run" => measure_and_report(
+ &fixture_dir,
+ &baseline_path,
+ reference,
+ lanes,
+ thumbnail_count,
+ Mode::Report,
+ ),
+ "check" => measure_and_report(
+ &fixture_dir,
+ &baseline_path,
+ reference,
+ lanes,
+ thumbnail_count,
+ Mode::Gate,
+ ),
+ "record" => measure_and_report(
+ &fixture_dir,
+ &baseline_path,
+ reference,
+ lanes,
+ thumbnail_count,
+ Mode::Record,
+ ),
+ "help" | "--help" | "-h" => {
+ print_help();
+ Ok(true)
+ }
+ other => anyhow::bail!("unknown command {other}; try `dr-bench help`"),
+ }
+}
+
+fn expect_value(args: &mut impl Iterator- , flag: &str) -> Result {
+ args.next()
+ .ok_or_else(|| anyhow::anyhow!("{flag} needs a value"))
+}
+
+/// What a run does with what it measured.
+#[derive(Debug, Clone, Copy, PartialEq, Eq)]
+enum Mode {
+ /// Print, judge nothing.
+ Report,
+ /// Print and fail the build on a violated budget or a regression.
+ Gate,
+ /// Print and rewrite the committed baseline from what was measured.
+ Record,
+}
+
+fn print_help() {
+ println!(
+ "\
+dr-bench — DarkRoom's performance suite (docs/requirements.md §8)
+
+ run measure and print, judging nothing
+ check measure, print, and exit 1 on a violated budget or a regression
+ record measure and rewrite docs/bench-baseline.json from the result
+
+Flags:
+ --reference this machine is the reference desktop, so machine-sensitive
+ budgets are asserted rather than reported
+ --fixture where the synthetic 50k catalog lives (or $DR_BENCH_DIR)
+ --baseline the committed numbers (default docs/bench-baseline.json)
+ --lanes sweep lanes for the thumbnail row (default: CPU threads)
+ --thumbnails images in the thumbnail row (default 1200)
+
+Build it in release. A debug build measures the compiler, not the pipeline —
+see the module documentation and core/dr-gpu/examples/frame_budget.rs."
+ );
+}
+
+// ---------------------------------------------------------------------------
+// The run
+// ---------------------------------------------------------------------------
+
+fn measure_and_report(
+ fixture_dir: &std::path::Path,
+ baseline_path: &std::path::Path,
+ reference: bool,
+ lanes: usize,
+ thumbnail_count: usize,
+ mode: Mode,
+) -> Result {
+ let machine = baseline::machine_id();
+ println!("DarkRoom benchmark suite — the half that needs no GPU");
+ println!("machine {machine}");
+ if cfg!(debug_assertions) {
+ println!(
+ "profile DEBUG — every figure below is several times worse than the \
+ product's. Rerun with --release."
+ );
+ } else {
+ println!("profile release");
+ }
+
+ let (fx, built) = fixture::build(fixture_dir, SEED, IMAGES)?;
+ println!(
+ "fixture {} images, {} sources at {}x{}, seed {}, catalog {:.1} MB{}",
+ fx.stamp.images,
+ fx.stamp.sources,
+ fx.stamp.preview_width,
+ fx.stamp.preview_height,
+ fx.stamp.seed,
+ fx.catalog_bytes as f64 / 1e6,
+ if built { " (built just now)" } else { "" }
+ );
+ println!(" {}", fx.dir.display());
+ println!();
+
+ // Memory first, and in its own process. See `memory` for why: measuring
+ // RSS after the thumbnail sweep would report the sweep's high-water mark
+ // wearing the catalog's name.
+ let rss = match memory::in_a_fresh_process(&fx.catalog) {
+ Ok(rss) => Some(rss),
+ Err(e) => {
+ println!("memory unavailable: {e}");
+ None
+ }
+ };
+
+ let open = catalog_open::measure(&fx.catalog, WINDOWS)?;
+ let thumbs = thumbnails::measure(&fx.sources, &fx.thumbs, thumbnail_count, lanes)?;
+ let exports = exporting::measure()?;
+
+ print_details(&open, &thumbs, &exports, rss);
+
+ let values = collect(&open, &thumbs, &exports, rss);
+ let mut base = Baseline::load(baseline_path)?;
+
+ if mode == Mode::Record {
+ if !reference {
+ println!(
+ "note recording from a machine that has not declared itself the \
+ reference desktop.\n §8 records the reference desktop's numbers; \
+ this overwrites them."
+ );
+ }
+ for (key, value) in &values {
+ if let Some(metric) = base.metrics.get_mut(key) {
+ metric.recorded = Some(*value);
+ }
+ }
+ base.recorded_on = Some(machine);
+ base.recorded_at_unix = Some(baseline::now_unix());
+ base.fixture = Some(fx.stamp.clone());
+ base.save(baseline_path)?;
+ println!("\nrecorded {}", baseline_path.display());
+ return Ok(true);
+ }
+
+ let same_machine = base.recorded_on.as_deref() == Some(machine.as_str());
+ let comparable = base.fixture.as_ref().is_some_and(|f| *f == fx.stamp);
+ let cx = Judging {
+ reference,
+ same_machine: same_machine && comparable,
+ tolerance: base.tolerance,
+ };
+ let failures = print_verdict(&base, &values, cx, &machine, comparable);
+
+ if failures.is_empty() {
+ return Ok(true);
+ }
+ println!();
+ for line in &failures {
+ println!("FAIL {line}");
+ }
+ println!();
+ println!(
+ " {} gate(s) failed. docs/benchmarks.md says what each metric measures and\n \
+ docs/bench-baseline.json holds the numbers these used to be.",
+ failures.len()
+ );
+ Ok(mode != Mode::Gate)
+}
+
+/// The metric keys, and the measurement each one reads.
+///
+/// One place, so the report, the gate and the recorded file cannot disagree
+/// about what `catalog_open_ms` means.
+fn collect(
+ open: &catalog_open::Open,
+ thumbs: &thumbnails::Throughput,
+ exports: &[exporting::EncodeRun],
+ rss: Option,
+) -> BTreeMap {
+ let mut v = BTreeMap::new();
+ v.insert("catalog_open_ms".to_string(), open.cold_ms);
+ v.insert("catalog_open_warm_ms".to_string(), open.warm_ms);
+ v.insert("catalog_window_p99_ms".to_string(), open.window_ms.p99);
+ v.insert("catalog_filtered_ms".to_string(), open.filtered_ms);
+ let ips = thumbs.images_per_second;
+ v.insert("thumbnail_throughput_ips".to_string(), ips);
+ v.insert("thumbnail_per_image_p99_ms".to_string(), thumbs.per_image.p99);
+ for row in exports {
+ v.insert(row.key.to_string(), row.times.p99);
+ }
+ if let Some(rss) = rss {
+ v.insert("catalog_idle_rss_mb".to_string(), rss.now_mb());
+ }
+ v
+}
+
+// ---------------------------------------------------------------------------
+// The report
+// ---------------------------------------------------------------------------
+
+fn print_details(
+ open: &catalog_open::Open,
+ thumbs: &thumbnails::Throughput,
+ exports: &[exporting::EncodeRun],
+ rss: Option,
+) {
+ println!("Opening the catalog (NFR-P1, and R2's second sentence)");
+ println!(
+ " {:>10.1} ms open, count, first window and timeline — cold, first \
+ connection of the process",
+ open.cold_ms
+ );
+ println!(
+ " {:>10.1} ms the same four calls on a second connection",
+ open.warm_ms
+ );
+ println!(
+ " {:>10.1} ms Catalog::open alone (connect, migrate, backfill)",
+ open.open_only_ms
+ );
+ println!(
+ " {:>10.1} ms count and first window under a rating filter",
+ open.filtered_ms
+ );
+ println!(
+ " {:>10.2} ms one 400-row window at a random offset, p99 of {WINDOWS} \
+ (p50 {:.2}, max {:.2})",
+ open.window_ms.p99, open.window_ms.p50, open.window_ms.max
+ );
+ println!(
+ " {} images, {} monthly timeline buckets",
+ open.images, open.buckets
+ );
+ println!();
+
+ println!("Thumbnails on the embedded preview path (NFR-P3: >= 100 img/s)");
+ println!(
+ " {:>10.1} img/s over {} images on {} lanes, {:.2} s of wall clock",
+ thumbs.images_per_second,
+ thumbs.images,
+ thumbs.lanes,
+ thumbs.wall_ms / 1e3
+ );
+ println!(
+ " {:>10.2} ms per image on its lane, p99 (p50 {:.2}, max {:.2})",
+ thumbs.per_image.p99, thumbs.per_image.p50, thumbs.per_image.max
+ );
+ println!(
+ " {} stored, {} failed",
+ thumbs.stored, thumbs.failed
+ );
+ println!();
+ println!(" The bytes are in memory before the clock starts, so this is CPU");
+ println!(" throughput with the fetch already paid for. That is what the target's");
+ println!(" \"embedded preview path\" names; it is not a claim about a remote library.");
+ println!();
+
+ println!("Exporting 24 MP — the encode half only (NFR-P7 is the whole chain)");
+ println!(
+ " {:>14} {:>11} {:>9} {:>9} {:>9}",
+ "sizing", "output", "p50", "p99", "file"
+ );
+ for row in exports {
+ println!(
+ " {:>14} {:>5}x{:<5} {:>7.1}ms {:>7.1}ms {:>7.1}MB",
+ row.label,
+ row.width,
+ row.height,
+ row.times.p50,
+ row.times.p99,
+ row.bytes as f64 / 1e6
+ );
+ }
+ println!();
+ println!(" No GPU render is in these figures, so they cannot pass NFR-P7 — only fail");
+ println!(" it. See tools/bench/src/exporting.rs for why that is still worth gating.");
+ println!();
+
+ println!("Idle memory with the catalog open (NFR-P8's catalog share)");
+ match rss {
+ Some(rss) => {
+ println!(
+ " {:>10.1} MB resident after opening 50k and scrolling 10k rows",
+ rss.now_mb()
+ );
+ println!(" {:>10.1} MB peak for that process", rss.peak_mb());
+ println!();
+ println!(" RSS is exclusive of device-local GPU memory and this process has no");
+ println!(" toolkit, no adapter and no decode cache — so it is the catalog layer's");
+ println!(" share of NFR-P8's 500 MB, not NFR-P8. tools/bench/src/memory.rs holds");
+ println!(" the answer to the question §4.1 asks, and the change it recommends.");
+ }
+ None => println!(" not measured on this platform"),
+ }
+ println!();
+}
+
+/// Print the gate table and return the failures, one line each.
+fn print_verdict(
+ base: &Baseline,
+ values: &BTreeMap,
+ cx: Judging,
+ machine: &str,
+ comparable_fixture: bool,
+) -> Vec {
+ println!("Against docs/bench-baseline.json");
+ match (&base.recorded_on, cx.same_machine) {
+ (None, _) => println!(
+ " No baseline has been recorded yet. Budgets are still gated; drift is not.\n \
+ Run `dr-bench record --reference` on the reference desktop and commit the diff."
+ ),
+ (Some(on), true) => println!(" Recorded on {on} — drift below is a verdict."),
+ (Some(on), false) if !comparable_fixture => println!(
+ " Recorded on {on} against a different fixture; drift below is information only."
+ ),
+ (Some(on), false) => println!(
+ " Recorded on {on}, and this is {machine}. Drift below is information, not a verdict."
+ ),
+ }
+ if !cx.reference {
+ println!(
+ " Not the reference desktop (--reference), so machine-sensitive budgets are\n \
+ reported rather than asserted — §8 names the reference desktop, not CI."
+ );
+ }
+ println!();
+ println!(
+ " {:<30} {:>12} {:>13} {:>12} {:>8} {}",
+ "metric", "measured", "budget", "baseline", "drift", "verdict"
+ );
+
+ let mut failures = Vec::new();
+ for (key, measured) in values {
+ let Some(metric) = base.metrics.get(key) else {
+ println!(
+ " {key:<30} {measured:>12.2} {:>13} {:>12} {:>8} not in the baseline",
+ "—", "—", "—"
+ );
+ continue;
+ };
+ let j = metric.judge(*measured, cx);
+ let budget = match (metric.budget, metric.direction) {
+ (None, _) => "—".to_string(),
+ (Some(b), baseline::Direction::LowerIsBetter) => format!("< {b:.1}"),
+ (Some(b), baseline::Direction::HigherIsBetter) => format!("> {b:.1}"),
+ };
+ let recorded = match metric.recorded {
+ Some(r) => format!("{r:.2}"),
+ None => "—".to_string(),
+ };
+ let drift = match j.drift {
+ Some(d) => format!("{:+.1}%", d * 100.0),
+ None => "—".to_string(),
+ };
+ let mut verdict = String::new();
+ if j.over_budget {
+ verdict.push_str("OVER BUDGET ");
+ failures.push(format!(
+ "{key} is {measured:.2} {}, past {budget} ({})",
+ metric.unit, metric.requirement
+ ));
+ }
+ if j.regressed {
+ verdict.push_str("REGRESSED ");
+ failures.push(format!(
+ "{key} drifted {drift} against the baseline, past the {:.0}% tolerance",
+ cx.tolerance * 100.0
+ ));
+ }
+ if verdict.is_empty() {
+ verdict.push_str(if j.budget_deferred {
+ "ok (budget deferred)"
+ } else {
+ "ok"
+ });
+ }
+ println!(
+ " {key:<30} {measured:>12.2} {budget:>13} {recorded:>12} {drift:>8} {verdict}"
+ );
+ }
+
+ // A metric in the file that nothing measured is a harness that has drifted
+ // from its own record, and is worth saying out loud rather than leaving as
+ // a row that quietly stopped appearing.
+ for key in base.metrics.keys() {
+ if !values.contains_key(key) {
+ println!(" {key:<30} {:>12} measured nothing this run", "—");
+ }
+ }
+
+ failures
+}
diff --git a/tools/bench/src/memory.rs b/tools/bench/src/memory.rs
new file mode 100644
index 0000000..da4067b
--- /dev/null
+++ b/tools/bench/src/memory.rs
@@ -0,0 +1,179 @@
+//! Idle memory with a 50k catalog open — and the question NFR-P8 leaves open.
+//!
+//! # The question §4.1 asks, answered
+//!
+//! §4.1 says of NFR-P8: *"must state whether it measures RSS inclusive or
+//! exclusive of GPU allocations, and whether it holds after SQLite's page cache
+//! warms on a 50k catalog."* Both halves have an answer, and neither is
+//! flattering.
+//!
+//! **On GPU memory: what this reports is RSS, and RSS is exclusive of
+//! device-local GPU allocations.** A Vulkan allocation in device-local heap
+//! never enters the process's address space, so no counter under
+//! `/proc/self/status` can see it; what *does* land in RSS is the host-visible
+//! side — staging buffers, mapped upload rings, the read-back `AdjustPass`
+//! performs on export — and the driver's own resident pages. So "RSS < 500 MB"
+//! is not one budget, it is two questions wearing one number, and a build that
+//! kept RSS at 400 MB while holding 3 GB of textures would pass it.
+//!
+//! The recommendation this measurement exists to support: **NFR-P8 should be
+//! restated as two figures** — host RSS exclusive of device-local memory, and
+//! a separate VRAM ceiling read from the adapter — because the second is the
+//! one that decides whether the application survives beside a browser on an
+//! 8 GB card, and nothing in this repository currently measures it.
+//!
+//! **On the page cache: warm.** The probe runs the queries before it reads the
+//! counter, so SQLite's page cache holds the b-tree pages a grid scroll
+//! touches. That is the right side to err on — a figure taken before the cache
+//! warms would understate a steady-state library — and it is why the probe
+//! scrolls rather than opening and stopping.
+//!
+//! # Why this is a subprocess
+//!
+//! RSS is a high-water-influenced property of a *process*, not of a function.
+//! Building a 50k fixture allocates hundreds of megabytes; decoding thumbnails
+//! allocates more; the allocator returns some of it to the OS and keeps the
+//! rest. Measuring after any of that would report the harness's history rather
+//! than the catalog's cost. So the probe is a fresh process that opens the
+//! catalog, does the grid's work, reads its own counters and exits.
+//!
+//! # What this cannot certify, said plainly
+//!
+//! Not NFR-P8. The requirement is about the *application* at idle — Slint, the
+//! wgpu device, the font stack, the decode cache and the catalog together — and
+//! this process contains only the last of those. No requirement tag in this
+//! crate names NFR-P8, for that reason — and see `exporting.rs` for why that
+//! sentence avoids spelling the tag out.
+//!
+//! What it is, is the catalog layer's share, measured rather than guessed. The
+//! decision NFR-P8 actually needs — how much of the 500 MB belongs to the
+//! catalog and how much to everything above it — is a decision somebody has to
+//! take, and taking it against a recorded number is better than taking it
+//! against an estimate. That is what this records. Until it is taken, the
+//! metric carries no budget and gates only against its own baseline.
+
+use std::path::Path;
+
+use anyhow::Result;
+use dr_catalog::{Catalog, Granularity, Query};
+
+/// Rows fetched per window while the probe scrolls. The same 400
+/// [`crate::catalog_open`] uses, for the same reason.
+const WINDOW: usize = 400;
+
+/// Windows the probe pages through before reading the counter.
+///
+/// Twenty-five is ten thousand rows: enough that SQLite's page cache holds a
+/// realistic working set and that any per-window leak would be visible, and
+/// far short of the whole library, which FR-CAT-4 forbids holding anyway.
+const WINDOWS: usize = 25;
+
+/// A fixed clock, for the reason `catalog_open`'s `NOW` gives: nothing here
+/// should depend on the day it runs.
+const NOW: i64 = 2_000_000_000;
+
+/// Resident memory, in kilobytes, as Linux reports it.
+#[derive(Debug, Clone, Copy)]
+pub struct Rss {
+ /// `VmRSS`: resident now.
+ pub now_kb: u64,
+ /// `VmHWM`: the peak this process reached. Reported alongside because a
+ /// process that touched 900 MB and gave it back is not idling at 200 MB in
+ /// any sense a user would recognise — the pages came from somewhere.
+ pub peak_kb: u64,
+}
+
+impl Rss {
+ pub fn now_mb(&self) -> f64 {
+ self.now_kb as f64 / 1024.0
+ }
+
+ pub fn peak_mb(&self) -> f64 {
+ self.peak_kb as f64 / 1024.0
+ }
+}
+
+/// Read this process's own counters.
+///
+/// `None` anywhere without a Linux-shaped `/proc` — including Android, where
+/// the file exists but a benchmark does not run, and macOS, where it does not.
+/// Returning `None` rather than zero is deliberate: a memory figure of zero
+/// would be reported as an excellent result.
+pub fn of_this_process() -> Option {
+ let status = std::fs::read_to_string("/proc/self/status").ok()?;
+ let mut now = None;
+ let mut peak = None;
+ for line in status.lines() {
+ if let Some(rest) = line.strip_prefix("VmRSS:") {
+ now = rest.split_whitespace().next()?.parse::().ok();
+ } else if let Some(rest) = line.strip_prefix("VmHWM:") {
+ peak = rest.split_whitespace().next()?.parse::().ok();
+ }
+ }
+ Some(Rss {
+ now_kb: now?,
+ peak_kb: peak?,
+ })
+}
+
+/// The probe: open the catalog, do what the grid does, print the counters.
+///
+/// Stdout is one line of `key=value` pairs rather than JSON, because the only
+/// reader is [`in_a_fresh_process`] and a format a human can read in a log is
+/// worth more here than one a parser prefers.
+pub fn probe(catalog_path: &Path) -> Result<()> {
+ let catalog = Catalog::open(catalog_path)
+ .map_err(|e| anyhow::anyhow!("opening {} : {e}", catalog_path.display()))?;
+ let q = Query::default();
+
+ let images = catalog.count(&q, NOW)?;
+ let buckets = catalog.timeline(&q, Granularity::Month, NOW)?.len();
+
+ // Scroll, keeping only the window in hand — which is what the grid does,
+ // and what FR-CAT-4 requires it to do. If this ever starts costing memory
+ // proportional to how far the user scrolled, that is the bug this figure
+ // exists to catch.
+ let mut rows = 0usize;
+ let span = images.saturating_sub(WINDOW).max(1);
+ for i in 0..WINDOWS {
+ let start = (i * span) / WINDOWS.max(1);
+ rows = catalog.window(&q, start..start + WINDOW, NOW)?.len();
+ }
+
+ let Some(rss) = of_this_process() else {
+ anyhow::bail!("no /proc/self/status on this platform; RSS cannot be read");
+ };
+ println!(
+ "rss_kb={} peak_kb={} images={images} buckets={buckets} last_window={rows}",
+ rss.now_kb, rss.peak_kb
+ );
+ Ok(())
+}
+
+/// Run [`probe`] in a fresh copy of this executable and read back its counters.
+pub fn in_a_fresh_process(catalog_path: &Path) -> Result {
+ let exe = std::env::current_exe()?;
+ let output = std::process::Command::new(&exe)
+ .arg("memory-probe")
+ .arg(catalog_path)
+ .output()?;
+
+ if !output.status.success() {
+ anyhow::bail!(
+ "the memory probe exited with {}: {}",
+ output.status,
+ String::from_utf8_lossy(&output.stderr).trim()
+ );
+ }
+
+ let text = String::from_utf8_lossy(&output.stdout);
+ let field = |key: &str| -> Option {
+ text.split_whitespace()
+ .find_map(|pair| pair.strip_prefix(key))
+ .and_then(|v| v.parse::().ok())
+ };
+ let (Some(now_kb), Some(peak_kb)) = (field("rss_kb="), field("peak_kb=")) else {
+ anyhow::bail!("the memory probe printed something unreadable: {}", text.trim());
+ };
+ Ok(Rss { now_kb, peak_kb })
+}
diff --git a/tools/bench/src/stats.rs b/tools/bench/src/stats.rs
new file mode 100644
index 0000000..8dad9f5
--- /dev/null
+++ b/tools/bench/src/stats.rs
@@ -0,0 +1,122 @@
+//! Ranking a set of samples, the way `dr-gpu`'s frame budget ranks them.
+//!
+//! Copied in spirit rather than shared, because the two live in different
+//! dependency worlds — `core/dr-gpu/examples/frame_budget.rs` is an example
+//! inside a crate this one deliberately does not depend on (see `Cargo.toml`).
+//! The arithmetic is identical on purpose: two percentile definitions in one
+//! repository is how two benchmarks come to disagree about the same machine.
+//!
+//! # Nearest-rank, not an interpolating definition
+//!
+//! The samples *are* the population. There is no distribution being estimated
+//! here, only a set of catalog opens or thumbnail encodes that either happened
+//! inside the target or did not. At 100 samples the 99th percentile is the
+//! second-worst, which is the honest reading of "one bad one in a hundred is
+//! one too many" without letting a single scheduler hiccup on an unrelated
+//! process decide the verdict.
+
+use std::time::Duration;
+
+/// Nearest-rank percentiles over a set of samples, in the caller's unit.
+#[derive(Debug, Clone, Copy)]
+pub struct Percentiles {
+ pub p50: f64,
+ pub p99: f64,
+ pub max: f64,
+}
+
+impl Percentiles {
+ /// Rank `samples`. Panics on an empty set, which is a harness bug rather
+ /// than a measurement: a row with nothing in it must not print a zero that
+ /// reads like a very fast result.
+ pub fn of(mut samples: Vec) -> Self {
+ assert!(
+ !samples.is_empty(),
+ "percentiles of an empty sample set — the measurement produced nothing"
+ );
+ samples.sort_by(f64::total_cmp);
+ let rank = |p: f64| {
+ let n = samples.len();
+ let i = ((p * n as f64).ceil() as usize).clamp(1, n) - 1;
+ samples[i]
+ };
+ Percentiles {
+ p50: rank(0.50),
+ p99: rank(0.99),
+ max: samples[samples.len() - 1],
+ }
+ }
+}
+
+/// A duration in milliseconds, which is the unit every timing here is stated
+/// in. One spelling, so no row is accidentally in seconds.
+pub fn ms(d: Duration) -> f64 {
+ d.as_secs_f64() * 1e3
+}
+
+/// A deterministic generator, so a fixture is reproducible from its seed.
+///
+/// SplitMix64. Chosen because it is eight lines, has no dependency, and passes
+/// the only test that matters here — that the same seed produces the same
+/// catalog on the reference desktop and on the CI runner, so a number measured
+/// in one place describes the same workload as a number measured in the other.
+/// Nothing cryptographic depends on it.
+pub struct Rng(u64);
+
+impl Rng {
+ pub fn new(seed: u64) -> Self {
+ Rng(seed)
+ }
+
+ pub fn next_u64(&mut self) -> u64 {
+ self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15);
+ let mut z = self.0;
+ z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
+ z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
+ z ^ (z >> 31)
+ }
+
+ /// A value in `0..n`. Modulo-biased, which does not matter for a fixture:
+ /// nothing here is a statistical test, only a spread of plausible values.
+ pub fn below(&mut self, n: u64) -> u64 {
+ self.next_u64() % n.max(1)
+ }
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+
+ #[test]
+ fn the_ninety_ninth_of_a_hundred_is_the_second_worst() {
+ // The property the whole suite's verdict rests on. Off by one here and
+ // every threshold is judged against the worst sample instead.
+ let samples: Vec = (1..=100).map(|n| n as f64).collect();
+ let p = Percentiles::of(samples);
+ assert_eq!(p.p99, 99.0);
+ assert_eq!(p.max, 100.0);
+ assert_eq!(p.p50, 50.0);
+ }
+
+ #[test]
+ fn a_single_sample_ranks_as_itself() {
+ // A row measured once — a cold catalog open — must not divide by zero
+ // or index off the end.
+ let p = Percentiles::of(vec![7.5]);
+ assert_eq!((p.p50, p.p99, p.max), (7.5, 7.5, 7.5));
+ }
+
+ #[test]
+ fn the_same_seed_gives_the_same_sequence() {
+ // Reproducibility from a seed is what makes a committed baseline mean
+ // anything: two runs must describe the same catalog.
+ let mut a = Rng::new(20_260_829);
+ let mut b = Rng::new(20_260_829);
+ let mut c = Rng::new(20_260_830);
+ let first: Vec = (0..8).map(|_| a.next_u64()).collect();
+ let same: Vec = (0..8).map(|_| b.next_u64()).collect();
+ let other: Vec = (0..8).map(|_| c.next_u64()).collect();
+ assert_eq!(first, same);
+ assert_ne!(first, other);
+ }
+}
diff --git a/tools/bench/src/thumbnails.rs b/tools/bench/src/thumbnails.rs
new file mode 100644
index 0000000..0768eec
--- /dev/null
+++ b/tools/bench/src/thumbnails.rs
@@ -0,0 +1,272 @@
+//! TRACES: NFR-P3
+//! Thumbnail throughput on the embedded preview path.
+//!
+//! NFR-P3 asks for **≥ 100 images per second** on the reference desktop
+//! through the embedded preview path, and ≥ 25 on a mid-range Android device.
+//! Nothing had ever counted.
+//!
+//! # What is timed, and why it is not the sweep itself
+//!
+//! `ui/dr-ui/src/library.rs`'s [`spawn_thumbnail_sweep`] is the whole-library
+//! pass, and it is the machinery this mirrors: a chunk at a time, lanes owning
+//! disjoint slices, every lane decoding and encoding on its own, and the
+//! single thread that owns the store writing the finished chunk. That shape is
+//! reproduced here because it is the shape that decides the number — where the
+//! parallelism is, and where the one lock is.
+//!
+//! It is *mirrored* rather than called, for a reason worth stating plainly:
+//! that function takes a `RemoteBackend` and spends most of its wall clock in
+//! WebDAV round trips. Calling it from a benchmark would need a Nextcloud
+//! server, and what it would then measure is somebody's network. The per-image
+//! work is identical either way — `dr_decode::decode_jpeg`,
+//! `Preview::downscale_to`, `Preview::apply_orientation`,
+//! `dr_thumbs::encode_rgba`, `ThumbStore::put` — and that work is what a target
+//! written in images per second is about.
+//!
+//! So: **the bytes are already in memory when the clock starts.** This is CPU
+//! throughput for the preview path with the fetch paid for, which is what a
+//! local library gives you and what the target's "embedded preview path"
+//! names. It is not a claim about a remote library, whose ceiling is latency
+//! and which [`spawn_thumbnail_sweep`] exists to hide rather than to beat.
+//!
+//! # The plain-JPEG branch, deliberately
+//!
+//! `fetch_preview` has two arms: a plain JPEG is its own preview, and anything
+//! else is located inside the container first. The fixture's files take the
+//! first arm, so `locate_preview` is not in the measured span. That is honest
+//! for two reasons — a library of camera JPEGs is a real library and takes
+//! exactly this path, and for a RAW the located preview *is* a JPEG of about
+//! this size, so what changes is a header walk of a few microseconds against a
+//! decode of several milliseconds. What is genuinely not measured is the
+//! container parse of an exotic format, and nothing here pretends otherwise.
+//!
+//! [`spawn_thumbnail_sweep`]: ../../dr_ui/library/fn.spawn_thumbnail_sweep.html
+
+use std::path::{Path, PathBuf};
+use std::time::Instant;
+
+use anyhow::{Context, Result};
+use dr_thumbs::{ThumbSize, ThumbStore, Thumbnail};
+use dr_types::Orientation;
+
+use crate::stats::{ms, Percentiles};
+
+/// Images per chunk handed back to the thread that owns the store.
+///
+/// 96, which is `SWEEP_CHUNK` in `library.rs`. Copied rather than chosen: the
+/// point of this row is to describe the sweep's behaviour, and a different
+/// chunk size would move the ratio of lane work to store work.
+const CHUNK: usize = 96;
+
+/// The size class the whole-library pass fills.
+///
+/// Grid only, which is `SWEEP_THUMB_SIZE`. The large class is four times the
+/// transfer for a detail only a zoomed cell asks for, so the sweep does not
+/// produce it and neither does this.
+const SIZE: ThumbSize = ThumbSize::Grid;
+
+/// What a measured sweep produced.
+pub struct Throughput {
+ pub images: usize,
+ pub lanes: usize,
+ pub wall_ms: f64,
+ pub images_per_second: f64,
+ /// Per image, on the lane that produced it: decode, downscale, orient,
+ /// encode. Not the store write, which happens elsewhere by design.
+ pub per_image: Percentiles,
+ /// Thumbnails that reached the store.
+ pub stored: usize,
+ /// Images whose preview would not decode. Any non-zero is a broken
+ /// fixture, not a slow one.
+ pub failed: usize,
+}
+
+/// One lane's output: what it made, what each one cost, and what it dropped.
+struct Lane {
+ made: Vec<(u64, Thumbnail)>,
+ times: Vec,
+ failed: usize,
+}
+
+/// Sweep `images` thumbnails from `sources`, `lanes` at a time.
+///
+/// `store_dir` is emptied first. Re-storing an id that is already present is
+/// an `UPDATE` in place rather than an insert (see `ThumbStore::put`), and a
+/// run that measured updates would not be measuring the pass this describes —
+/// the sweep's work list is by construction what the store does *not* have.
+pub fn measure(
+ sources: &[PathBuf],
+ store_dir: &Path,
+ images: usize,
+ lanes: usize,
+) -> Result {
+ anyhow::ensure!(!sources.is_empty(), "no fixture sources to sweep");
+ anyhow::ensure!(lanes > 0, "a sweep needs at least one lane");
+
+ let bytes: Vec> = sources
+ .iter()
+ .map(|p| std::fs::read(p).with_context(|| format!("reading {}", p.display())))
+ .collect::>()?;
+
+ let _ = std::fs::remove_dir_all(store_dir);
+ let mut store = ThumbStore::open(store_dir)
+ .map_err(|e| anyhow::anyhow!("opening the bench thumbnail store: {e}"))?;
+
+ // Warm-up: every source decoded once, discarded. The first decode of a
+ // file faults in its Huffman tables and grows the allocator's arenas to
+ // the size a 1620x1080 RGBA buffer needs, and neither recurs across a
+ // sweep of thousands.
+ let warm_failures = Lane::run(&bytes, (0..bytes.len()).collect()).failed;
+ anyhow::ensure!(
+ warm_failures == 0,
+ "{warm_failures} of the {} fixture sources would not decode",
+ bytes.len()
+ );
+
+ let mut samples: Vec = Vec::with_capacity(images);
+ let mut stored = 0usize;
+ let mut failed = 0usize;
+
+ let started = Instant::now();
+ for chunk_start in (0..images).step_by(CHUNK) {
+ let chunk_end = (chunk_start + CHUNK).min(images);
+ // Each lane takes every `lanes`-th image of the chunk, which is how
+ // `spawn_thumbnail_sweep` splits one: disjoint slices, nothing shared,
+ // no lock.
+ let work: Vec> = (0..lanes)
+ .map(|lane| (chunk_start..chunk_end).skip(lane).step_by(lanes).collect())
+ .collect();
+
+ let source_slice: &[Vec] = &bytes;
+ let produced: Vec = std::thread::scope(|scope| {
+ let handles: Vec<_> = work
+ .into_iter()
+ .map(|lane| scope.spawn(move || Lane::run(source_slice, lane)))
+ .collect();
+ handles
+ .into_iter()
+ .map(|h| h.join().expect("a sweep lane panicked"))
+ .collect()
+ });
+
+ // The store is `&mut` and single-writer, so the chunk is written here
+ // and not on the lanes. This is the sweep's own discipline and it is
+ // part of the number: if the store were the bottleneck, no amount of
+ // lane parallelism would help and the row would say so.
+ for lane in produced {
+ samples.extend(lane.times);
+ failed += lane.failed;
+ for (file_id, thumb) in lane.made {
+ match store.put(file_id, SIZE, &thumb) {
+ Ok(_) => stored += 1,
+ Err(e) => {
+ log::warn!("storing bench thumbnail {file_id}: {e}");
+ failed += 1;
+ }
+ }
+ }
+ }
+ }
+ let wall = started.elapsed();
+
+ anyhow::ensure!(
+ failed == 0,
+ "{failed} of {images} thumbnails failed; the throughput below would be \
+ measuring how fast this gives up"
+ );
+
+ let wall_ms = ms(wall);
+ Ok(Throughput {
+ images,
+ lanes,
+ wall_ms,
+ images_per_second: images as f64 / wall.as_secs_f64().max(f64::MIN_POSITIVE),
+ per_image: Percentiles::of(samples),
+ stored,
+ failed,
+ })
+}
+
+impl Lane {
+ /// Turn each of `work`'s previews into a stored thumbnail's worth of bytes.
+ ///
+ /// The four calls, in the order `fetch_preview` and `encode_preview` make
+ /// them. `is_complete_jpeg` is included because it is in the real path and
+ /// because leaving it out would be the sort of small omission that turns a
+ /// measurement into an estimate: a truncated JPEG decodes "successfully"
+ /// into a partial frame, so the check is not optional and its cost is not
+ /// somebody else's.
+ fn run(sources: &[Vec], work: Vec) -> Lane {
+ let mut made = Vec::with_capacity(work.len());
+ let mut times = Vec::with_capacity(work.len());
+ let mut failed = 0usize;
+
+ for index in work {
+ let bytes = &sources[index % sources.len()];
+ // The fixture makes every fourth image portrait, so a quarter of
+ // these pay for the permutation `apply_orientation` performs. In a
+ // real library that fraction is whatever the photographer shot.
+ let orientation = if index % 4 == 3 {
+ Orientation::from_exif(6)
+ } else {
+ Orientation::default()
+ };
+
+ let started = Instant::now();
+ if !dr_decode::is_complete_jpeg(bytes) {
+ failed += 1;
+ continue;
+ }
+ let mut preview = match dr_decode::decode_jpeg(bytes) {
+ Ok(p) => p,
+ Err(e) => {
+ log::debug!("bench preview {index}: {e}");
+ failed += 1;
+ continue;
+ }
+ };
+ preview.downscale_to(SIZE.edge());
+ preview.apply_orientation(orientation);
+ let rgba = &preview.rgba;
+ let encoded = match dr_thumbs::encode_rgba(preview.width, preview.height, rgba) {
+ Ok(b) => b,
+ Err(e) => {
+ log::debug!("bench thumbnail {index}: {e}");
+ failed += 1;
+ continue;
+ }
+ };
+ times.push(ms(started.elapsed()));
+
+ made.push((
+ index as u64 + 1,
+ Thumbnail {
+ width: preview.width,
+ height: preview.height,
+ bytes: encoded,
+ },
+ ));
+ }
+
+ Lane {
+ made,
+ times,
+ failed,
+ }
+ }
+}
+
+/// How many lanes to sweep with by default.
+///
+/// The machine's threads, not `SWEEP_LANES`. The sweep's six are sized for
+/// *latency* — each image is ~0.6 s of WebDAV round trip and almost no CPU, so
+/// six in flight is a queue depth rather than a core count. With the bytes
+/// already in memory the work is purely CPU, and six lanes on a 24-thread
+/// desktop would report a quarter of the throughput the machine has. Stated
+/// here rather than buried, because it is the one place this deviates from the
+/// shape it otherwise copies.
+pub fn default_lanes() -> usize {
+ std::thread::available_parallelism()
+ .map(|n| n.get())
+ .unwrap_or(1)
+}