Merge: measure the catalog, so §8's promise stops being a promise
There was no benchmark harness of any kind -- no benches/, no criterion, no synthetic fixture -- while §8 promised a suite run per commit that fails the build on regression. Ten performance requirements could be neither passed nor failed. tools/bench builds a deterministic 50,000-row catalog over a pool of twelve generated JPEGs, about 14 MB, reproducible from a seed, with a stamp so it rebuilds rather than silently comparing against a different workload. It depends on nothing GPU or UI, which is what makes the CI job affordable. NFR-P1 and NFR-P3 are gated and tagged. NFR-P7, NFR-P8 and R2 are measured but deliberately untagged: the export gate is one-sided, the memory figure is the catalog layer's share rather than the whole, and R2's first sentence is a 60 fps scroll a catalog benchmark cannot claim. Every recorded value in the baseline is null. Nobody has run this on the reference desktop, and a fabricated figure would make every later comparison a comparison against a guess. First run on this machine: catalog opens in 70 ms against a 2 s budget, and thumbnail throughput measures 37 img/s against a target of 100 -- reported rather than asserted here, and the first evidence that the target may not hold. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,193 @@
|
||||
name: Benchmarks
|
||||
|
||||
# The suite docs/requirements.md §8 has been promising since it was written:
|
||||
# "an automated benchmark suite against a synthetic 50k catalog, run per-commit
|
||||
# … A regression beyond stated tolerance fails the build."
|
||||
#
|
||||
# Its own workflow rather than a step inside build-and-test.yml, and the reason
|
||||
# is what a failure here means. A red `Build and test` says the code is wrong; a
|
||||
# red `Benchmarks` says the code is slower than it was, which is a different
|
||||
# conversation, is read by different people, and must not be reachable by
|
||||
# retrying a flaky compile.
|
||||
#
|
||||
# # Why this is split in two
|
||||
#
|
||||
# §8 names "the reference desktop", not CI, and it is right to. So:
|
||||
#
|
||||
# cpu — runs on every push. It needs no adapter and no display, and the
|
||||
# budgets it asserts (a 50k catalog opening inside two seconds) have
|
||||
# two orders of magnitude of headroom, so a modest runner can be held
|
||||
# to them honestly. Machine-sensitive budgets — throughput targets
|
||||
# written for a 24-thread desktop — are reported here rather than
|
||||
# asserted; `dr-bench` decides that per metric and says so in its
|
||||
# report. Asserting them on a two-core container would produce exactly
|
||||
# what core/dr-gpu/tests/frame_budget.rs refused to produce: a red gate
|
||||
# everybody learns to ignore.
|
||||
#
|
||||
# gpu — the frame budget, which already exists and already skips itself where
|
||||
# there is no adapter. Not on push: it would build wgpu and naga on
|
||||
# every commit to establish, every time, that this runner has no GPU. It
|
||||
# runs on demand (Actions → Run workflow) so that a runner that *does*
|
||||
# have one can be pointed at it, and the numbers it produces belong in
|
||||
# docs/frame-budget.md by hand, as they already are.
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main, master, develop]
|
||||
pull_request:
|
||||
branches: [main, master, develop]
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
cpu:
|
||||
runs-on: linux/amd64
|
||||
name: CPU and I/O (per commit)
|
||||
# Node for actions/checkout and actions/cache, which the bare runner image
|
||||
# cannot execute. Rust is installed below.
|
||||
container:
|
||||
image: catthehacker/ubuntu:act-latest
|
||||
|
||||
env:
|
||||
# Same reasoning as the desktop job in build-and-test.yml: incremental
|
||||
# state exists to make the *second* build in a working tree fast, which is
|
||||
# not a thing a fresh checkout has, and it fills the runner's disk.
|
||||
CARGO_INCREMENTAL: 0
|
||||
# The fixture, out of the workspace so actions/cache never picks it up.
|
||||
# A 14 MB synthetic catalog is two seconds to regenerate and would
|
||||
# otherwise be uploaded and downloaded on every push to save them.
|
||||
DR_BENCH_DIR: /tmp/darkroom-bench
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# No `git lfs pull` here, deliberately. `dr-bench` depends on the catalog,
|
||||
# the decoder, the thumbnail store and the encoder, and on nothing that
|
||||
# reaches `dr-segment` — so the model this repository keeps in LFS is not
|
||||
# part of this job's dependency graph and fetching it would be a minute
|
||||
# spent on a file nothing opens.
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
target
|
||||
key: bench-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
|
||||
|
||||
# Pinned to the workspace rust-version, as every other job here is: a
|
||||
# floating toolchain turns an unrelated push into a mystery failure, and
|
||||
# for a benchmark it would turn one into a mystery *regression*.
|
||||
#
|
||||
# rust-analyzer is named for the reason build-and-test.yml gives: rustup
|
||||
# reconciles rust-toolchain.toml on the first cargo call whatever this
|
||||
# step asks for, so naming it keeps the download inside the step that says
|
||||
# it is installing things.
|
||||
- name: Install Rust 1.92.0
|
||||
run: |
|
||||
set -e
|
||||
curl -fsSL https://sh.rustup.rs | sh -s -- \
|
||||
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
|
||||
--component rust-analyzer
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
|
||||
# `-p dr-bench`, not `--workspace`. The whole point of that crate having
|
||||
# no GPU and no UI dependency is that this job resolves the catalog, the
|
||||
# decoder and the encoders and stops there — a few minutes rather than the
|
||||
# release build of Slint and wgpu the desktop job pays for.
|
||||
#
|
||||
# Release, and it is not optional: the workspace builds its own crates at
|
||||
# opt-level = 0 in dev, and every figure this produces is dominated by
|
||||
# this workspace's own code. A debug run would measure rustc.
|
||||
- name: Build the suite
|
||||
run: cargo build --release -p dr-bench
|
||||
|
||||
# Exit 1 is a violated budget or a regression past tolerance; exit 2 is
|
||||
# the harness failing to run at all. Both fail the job, and the report
|
||||
# above the failure says which.
|
||||
- name: Measure, and gate
|
||||
run: cargo run --release -p dr-bench -- check
|
||||
|
||||
- name: Disk after
|
||||
if: always()
|
||||
run: df -h /workspace 2>/dev/null || df -h .
|
||||
|
||||
gpu:
|
||||
# On demand only — see the header. A runner with a Vulkan device can be
|
||||
# pointed at this; one without will skip the measurement and say so, which
|
||||
# is the same posture the rest of this repository's device tests take.
|
||||
if: github.event_name == 'workflow_dispatch'
|
||||
runs-on: linux/amd64
|
||||
name: Frame budget (on demand)
|
||||
container:
|
||||
image: catthehacker/ubuntu:act-latest
|
||||
|
||||
env:
|
||||
CARGO_INCREMENTAL: 0
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# `dr-gpu` depends on `dr-segment` for the watershed's pixel passes. Its
|
||||
# default features are off, so no weights are compiled in — but the fetch
|
||||
# is cheap insurance and its failure is not fatal. The header of the same
|
||||
# step in build-and-test.yml explains why the extraheader is stripped
|
||||
# rather than reused: two Authorization headers is a 400 from Gitea, one
|
||||
# step after the batch call that had just succeeded.
|
||||
- name: Fetch the segmentation model
|
||||
continue-on-error: true
|
||||
env:
|
||||
LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
|
||||
run: |
|
||||
set -e
|
||||
git lfs install --local
|
||||
git config --local --get-regexp '^http\..*extraheader$' \
|
||||
| cut -d' ' -f1 | sort -u \
|
||||
| while read -r key; do git config --local --unset-all "$key"; done || true
|
||||
git config --local lfs.url \
|
||||
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
|
||||
git lfs pull
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
target
|
||||
key: bench-gpu-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
|
||||
|
||||
- name: Build dependencies
|
||||
run: |
|
||||
apt-get update -qq
|
||||
apt-get install -y -qq pkg-config libfontconfig1-dev libxkbcommon-dev
|
||||
|
||||
- name: Install Rust 1.92.0
|
||||
run: |
|
||||
set -e
|
||||
curl -fsSL https://sh.rustup.rs | sh -s -- \
|
||||
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
|
||||
--component rust-analyzer
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
|
||||
# The guard, in release. Its own module documentation is explicit that a
|
||||
# release run checks strictly more than a dev one: the CPU half of a frame
|
||||
# is shader-string assembly, which is several times slower unoptimised, so
|
||||
# it is folded into the assertion only when debug_assertions is off.
|
||||
#
|
||||
# With no adapter this prints "skipping: no GPU adapter" and passes. A
|
||||
# test that cannot run is not evidence either way, and turning that into a
|
||||
# failure would make the job useless on the runner it usually lands on.
|
||||
- name: Frame budget (FR-DSP-3)
|
||||
run: cargo test --release -p dr-gpu --test frame_budget -- --nocapture
|
||||
|
||||
# The instrument behind docs/frame-budget.md. It exits non-zero with no
|
||||
# adapter, which is right for a tool a person runs deliberately and wrong
|
||||
# for a job that usually has none — hence continue-on-error. Its table is
|
||||
# in the log for whoever asked for this run; the committed numbers are
|
||||
# still updated by hand, as that file says.
|
||||
- name: Frame budget table
|
||||
continue-on-error: true
|
||||
run: cargo run --release -p dr-gpu --example frame_budget
|
||||
@@ -83,6 +83,21 @@ GPU tests skip themselves where there is no adapter rather than failing — a
|
||||
test that cannot run is not evidence either way — so a green run on a machine
|
||||
without a GPU is expected, and does not mean the GPU paths were exercised.
|
||||
|
||||
There is a fifth check, and it is not in that list because you are unlikely to
|
||||
break it by accident:
|
||||
|
||||
```bash
|
||||
cargo run --release -p dr-bench -- check
|
||||
```
|
||||
|
||||
That is the benchmark suite (`docs/requirements.md` §8), which builds a
|
||||
synthetic 50,000-image catalog and fails the build if a performance target is
|
||||
missed or a measurement has drifted past its tolerance. It runs on every push in
|
||||
its own workflow. [`docs/benchmarks.md`](docs/benchmarks.md) says what it
|
||||
measures, what it deliberately does not, and how to read a failure. If you have
|
||||
touched the catalog, the decoder, the thumbnail store or the exporter, run it
|
||||
before you send.
|
||||
|
||||
## Requirements and traceability
|
||||
|
||||
[`requirements.md`](docs/requirements.md) is the register of record.
|
||||
@@ -150,6 +165,7 @@ One commit per change. If you fixed two things, that is two commits.
|
||||
| [`core/dr-pipeline/ops/README.md`](core/dr-pipeline/ops/README.md) | Adding or changing a develop operation — start here regardless |
|
||||
| [`docs/architecture.md`](docs/architecture.md) | Anything touching the render path, catalog or sync |
|
||||
| [`docs/code-health.md`](docs/code-health.md) | Deciding what to work on; grades each seam by what it costs |
|
||||
| [`docs/benchmarks.md`](docs/benchmarks.md) | A change that could plausibly cost time or memory |
|
||||
| [`docs/technical-debt.md`](docs/technical-debt.md) | Something looks wrong — check it was not chosen |
|
||||
| [`docs/distribution.md`](docs/distribution.md) | Packaging a build, or adding a permission to one |
|
||||
| [`docs/requirements.md`](docs/requirements.md) | Reference, not reading |
|
||||
|
||||
Generated
+17
@@ -1403,6 +1403,23 @@ version = "0.1.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d8b14ccef22fc6f5a8f4d7d768562a182c04ce9a3b3157b91390b52ddfdf1a76"
|
||||
|
||||
[[package]]
|
||||
name = "dr-bench"
|
||||
version = "0.9.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"dr-catalog",
|
||||
"dr-decode",
|
||||
"dr-export",
|
||||
"dr-thumbs",
|
||||
"dr-types",
|
||||
"env_logger",
|
||||
"log",
|
||||
"rusqlite",
|
||||
"serde",
|
||||
"serde_json",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "dr-catalog"
|
||||
version = "0.9.0"
|
||||
|
||||
@@ -21,6 +21,7 @@ members = [
|
||||
"ui/dr-ui",
|
||||
"apps/darkroom-desktop",
|
||||
"apps/darkroom-android",
|
||||
"tools/bench",
|
||||
"tools/traceability",
|
||||
]
|
||||
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
{
|
||||
"_readme": [
|
||||
"The committed numbers for DarkRoom's benchmark suite (docs/requirements.md §8).",
|
||||
"Produced and checked by `cargo run --release -p dr-bench`; docs/benchmarks.md explains each metric.",
|
||||
"",
|
||||
"Two gates, and they are not the same gate. `budget` is the requirement's own threshold and never moves.",
|
||||
"`recorded` is what the reference desktop last measured, and a run that drifts past `tolerance` beyond it",
|
||||
"fails the build even while still inside the budget — which is how most performance rot actually arrives.",
|
||||
"",
|
||||
"`recorded` is null on every metric because nobody has run the suite yet. That is deliberate: writing",
|
||||
"plausible-looking figures here would make every later comparison a comparison against a guess. Run",
|
||||
"`cargo run --release -p dr-bench -- record --reference` on the reference desktop and commit the diff.",
|
||||
"Until then the budget gate works and the regression gate says, in the report, that it cannot.",
|
||||
"",
|
||||
"`machine_sensitive` says whether a budget is a statement about a machine as much as about the code.",
|
||||
"Those budgets are asserted only under --reference: §8 names the reference desktop, and a two-core CI",
|
||||
"container cannot speak to a target written for twenty-four threads. Asserting one there would produce a",
|
||||
"red gate everybody learns to ignore, which is the trap core/dr-gpu/tests/frame_budget.rs already avoids.",
|
||||
"",
|
||||
"Several metrics carry a requirement ID with a qualifier. Read those literally. NFR-P7's budget here is",
|
||||
"checked against the encode half of an export only — no GPU render is in the figure — so it can fail the",
|
||||
"requirement and cannot pass it, and no TRACES tag claims otherwise. NFR-P8 has no budget at all yet,",
|
||||
"because nobody has decided how much of its 500 MB belongs to the catalog layer; this records the number",
|
||||
"that decision needs."
|
||||
],
|
||||
"tolerance": 0.15,
|
||||
"recorded_on": null,
|
||||
"recorded_at_unix": null,
|
||||
"fixture": null,
|
||||
"metrics": {
|
||||
"catalog_filtered_ms": {
|
||||
"requirement": "FR-CAT-6",
|
||||
"what": "Count plus first window under a rating filter, which compiles to a correlated subquery over versions.",
|
||||
"unit": "ms",
|
||||
"direction": "lower_is_better",
|
||||
"machine_sensitive": true,
|
||||
"budget": null,
|
||||
"recorded": null
|
||||
},
|
||||
"catalog_idle_rss_mb": {
|
||||
"requirement": "NFR-P8 (the catalog layer's share only — no toolkit, no adapter, no decode cache)",
|
||||
"what": "Resident memory of a process that opened the 50k catalog and scrolled ten thousand rows.",
|
||||
"unit": "MB",
|
||||
"direction": "lower_is_better",
|
||||
"machine_sensitive": false,
|
||||
"budget": null,
|
||||
"recorded": null
|
||||
},
|
||||
"catalog_open_ms": {
|
||||
"requirement": "NFR-P1",
|
||||
"what": "Catalog::open plus the count, first window and timeline the grid cannot paint without.",
|
||||
"unit": "ms",
|
||||
"direction": "lower_is_better",
|
||||
"machine_sensitive": false,
|
||||
"budget": 2000.0,
|
||||
"recorded": null
|
||||
},
|
||||
"catalog_open_warm_ms": {
|
||||
"requirement": "NFR-P1",
|
||||
"what": "The same four calls on a second connection, with SQLite's page cache already warm.",
|
||||
"unit": "ms",
|
||||
"direction": "lower_is_better",
|
||||
"machine_sensitive": false,
|
||||
"budget": 2000.0,
|
||||
"recorded": null
|
||||
},
|
||||
"catalog_window_p99_ms": {
|
||||
"requirement": "FR-CAT-4",
|
||||
"what": "One 400-row grid window at a random offset, p99 of one hundred.",
|
||||
"unit": "ms",
|
||||
"direction": "lower_is_better",
|
||||
"machine_sensitive": true,
|
||||
"budget": null,
|
||||
"recorded": null
|
||||
},
|
||||
"export_24mp_long_edge_2048_ms": {
|
||||
"requirement": "FR-EXP-3",
|
||||
"what": "The web export: resample a 24 MP frame to a 2048 px long edge, sharpen, encode. p99 of five.",
|
||||
"unit": "ms",
|
||||
"direction": "lower_is_better",
|
||||
"machine_sensitive": true,
|
||||
"budget": null,
|
||||
"recorded": null
|
||||
},
|
||||
"export_24mp_original_ms": {
|
||||
"requirement": "NFR-P7 (the encode half only — the GPU render is not in this figure)",
|
||||
"what": "Resample, output-sharpen and JPEG-encode a 24 MP frame at source size. p99 of five.",
|
||||
"unit": "ms",
|
||||
"direction": "lower_is_better",
|
||||
"machine_sensitive": true,
|
||||
"budget": 2000.0,
|
||||
"recorded": null
|
||||
},
|
||||
"thumbnail_per_image_p99_ms": {
|
||||
"requirement": "NFR-P3",
|
||||
"what": "One thumbnail on its own lane: decode the preview, downscale, orient, encode. p99.",
|
||||
"unit": "ms",
|
||||
"direction": "lower_is_better",
|
||||
"machine_sensitive": true,
|
||||
"budget": null,
|
||||
"recorded": null
|
||||
},
|
||||
"thumbnail_throughput_ips": {
|
||||
"requirement": "NFR-P3",
|
||||
"what": "Whole-sweep throughput: 1200 thumbnails through the sweep's chunk-and-lane shape, wall clock.",
|
||||
"unit": "img/s",
|
||||
"direction": "higher_is_better",
|
||||
"machine_sensitive": true,
|
||||
"budget": 100.0,
|
||||
"recorded": null
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,243 @@
|
||||
# The benchmark suite
|
||||
|
||||
**Status:** Built, not yet recorded · 2026-08-30
|
||||
**Companion to:** [requirements.md](requirements.md) §4.1 (performance targets) · §8 (verification)
|
||||
**Instrument:** [`tools/bench`](../tools/bench) — `cargo run --release -p dr-bench -- check`
|
||||
**Committed numbers:** [`bench-baseline.json`](bench-baseline.json)
|
||||
**GPU half:** [`core/dr-gpu/tests/frame_budget.rs`](../core/dr-gpu/tests/frame_budget.rs) ·
|
||||
[frame-budget.md](frame-budget.md)
|
||||
|
||||
§8 has said since it was written that performance is verified by *"an automated
|
||||
benchmark suite against a synthetic 50k catalog, run per-commit … A regression
|
||||
beyond stated tolerance fails the build."* Until this suite there was none. No
|
||||
`benches/`, no `[[bench]]`, no criterion, no fixture — and ten performance
|
||||
requirements that could therefore be neither passed nor failed, five of them
|
||||
carrying a `TRACES:` tag regardless.
|
||||
|
||||
This file is what the suite covers, what it deliberately does not, and how to
|
||||
read a failure.
|
||||
|
||||
---
|
||||
|
||||
## The state of it, first
|
||||
|
||||
**No numbers have been recorded yet.** Every `recorded` field in
|
||||
[`bench-baseline.json`](bench-baseline.json) is `null`, on purpose: writing
|
||||
plausible-looking figures into a baseline would make every later comparison a
|
||||
comparison against a guess, and the first real regression would be invisible.
|
||||
|
||||
To record them, on the reference desktop:
|
||||
|
||||
```sh
|
||||
cargo run --release -p dr-bench -- record --reference
|
||||
```
|
||||
|
||||
and commit the diff. Until that happens the **budget** gate works — a catalog
|
||||
that takes three seconds to open fails the build today — and the **regression**
|
||||
gate reports that it has nothing to compare against, rather than pretending.
|
||||
|
||||
---
|
||||
|
||||
## What it measures
|
||||
|
||||
| Metric | Requirement | Gated? |
|
||||
|---|---|---|
|
||||
| `catalog_open_ms` | **NFR-P1**, and R2's second sentence | Yes, everywhere — budget 2000 ms |
|
||||
| `catalog_open_warm_ms` | NFR-P1, page cache warm | Yes, everywhere — budget 2000 ms |
|
||||
| `catalog_window_p99_ms` | FR-CAT-4 | Regression only |
|
||||
| `catalog_filtered_ms` | FR-CAT-6 | Regression only |
|
||||
| `thumbnail_throughput_ips` | **NFR-P3** | Budget 100 img/s, on the reference desktop |
|
||||
| `thumbnail_per_image_p99_ms` | NFR-P3 | Regression only |
|
||||
| `export_24mp_original_ms` | NFR-P7, **encode half only** | One-sided: can fail it, cannot pass it |
|
||||
| `export_24mp_long_edge_2048_ms` | FR-EXP-3 | Regression only |
|
||||
| `catalog_idle_rss_mb` | NFR-P8, **catalog layer only** | Regression only — see below |
|
||||
|
||||
Two of those rows carry a qualifier, and the qualifiers are the point.
|
||||
|
||||
### Requirements this can now pass *or* fail
|
||||
|
||||
**NFR-P1 — catalog open under 2 s.** The measured span is the four things the
|
||||
library view cannot paint without: `Catalog::open` (which connects, migrates and
|
||||
**backfills**, and the backfill is three passes over the images table on every
|
||||
open), `count`, the first 400-row `window`, and the monthly `timeline`. Tagged
|
||||
`TRACES: NFR-P1` in [`tools/bench/src/catalog_open.rs`](../tools/bench/src/catalog_open.rs),
|
||||
because a build that breaks it fails this gate.
|
||||
|
||||
**NFR-P3 — ≥ 100 images per second on the embedded preview path.** The
|
||||
per-image work is exactly what `spawn_thumbnail_sweep` does — `decode_jpeg`,
|
||||
`Preview::downscale_to`, `Preview::apply_orientation`, `encode_rgba`,
|
||||
`ThumbStore::put` — arranged in the same shape: chunks of 96, lanes owning
|
||||
disjoint slices, and the single thread that owns the store writing the finished
|
||||
chunk. Tagged `TRACES: NFR-P3` in
|
||||
[`tools/bench/src/thumbnails.rs`](../tools/bench/src/thumbnails.rs).
|
||||
|
||||
### Requirements this can only half-answer, and is not tagged for
|
||||
|
||||
**NFR-P7 — 24 MP export under 2 s, full chain.** The full chain is decode,
|
||||
demosaic, a full-resolution GPU render, a read-back, then resize, sharpen and
|
||||
encode. Only the last three run without an adapter. So the figure here is a
|
||||
**lower bound** on the requirement: exceeding 2 s in the encode alone violates
|
||||
NFR-P7 no matter how fast the render is, and coming in under it proves nothing.
|
||||
The budget is gated on that basis and there is no `TRACES: NFR-P7` anywhere in
|
||||
`tools/bench`.
|
||||
|
||||
**NFR-P8 — idle memory under 500 MB.** The probe is a fresh process holding the
|
||||
catalog and nothing else: no Slint, no wgpu device, no font stack, no decode
|
||||
cache. Its RSS is the catalog layer's *share* of that 500 MB, not the figure the
|
||||
requirement is about. It carries no budget for a reason given below.
|
||||
|
||||
### Requirements out of scope, listed so their absence reads as a decision
|
||||
|
||||
NFR-P2 (grid scroll at 60 fps), P4 (open in develop), P5 (slider to visible),
|
||||
P6 (pan/zoom), P9 (UI-executor blocking), P10 (touch response), P11 (layout
|
||||
transition), P12 (warm shader setup), P13 (next image in culling), P14 (focus
|
||||
peaking), P15 (drawn mask stroke). Every one of them needs a frame-timing probe
|
||||
inside a running Slint application, a GPU adapter, or both. None is faked here.
|
||||
|
||||
The GPU half of the story that *does* exist is
|
||||
[frame-budget.md](frame-budget.md) and its guard test, which asserts FR-DSP-3
|
||||
and skips itself where there is no adapter. `.gitea/workflows/benchmark.yml`
|
||||
runs it as its own job for exactly that reason.
|
||||
|
||||
---
|
||||
|
||||
## The fixture
|
||||
|
||||
Fifty thousand rows over a pool of twelve real image files. Rows are cheap and
|
||||
pixels are not: everything the catalog half touches is rows and is therefore
|
||||
exact at full scale, and everything the pixel half touches is one file at a time
|
||||
and does not care how many rows point at it. The result is ~14 MB on disk
|
||||
instead of ~2 TB, and neither half is flattered by that.
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Rows | 50,000 images, 50,000 default versions, 400 folders, one root |
|
||||
| Capture times | Twelve years from a fixed epoch, so the timeline has ~144 monthly buckets |
|
||||
| Sources | 12 synthesised JPEGs at 1620 × 1080 — the size `dr-decode` records a CR2 carrying in IFD2 |
|
||||
| Seed | 20260829, in [`tools/bench/src/main.rs`](../tools/bench/src/main.rs) |
|
||||
| Location | `$DR_BENCH_DIR`, else the system temporary directory |
|
||||
|
||||
It is reproducible from the seed, and a `stamp.json` beside it records what it
|
||||
was built from — seed, row count, source count, preview size, and `dr-catalog`'s
|
||||
schema version. A mismatch rebuilds rather than silently measuring a different
|
||||
workload than the baseline describes.
|
||||
|
||||
Two honest limits on it:
|
||||
|
||||
- **The page cache is warm.** The fixture was written by this suite or by an
|
||||
earlier run of it, so neither the catalog open nor the thumbnail sweep pays
|
||||
for a cold disk. On the reference desktop's NVMe a genuinely cold read of a
|
||||
14 MB catalog is tens of milliseconds; on spinning rust it is not.
|
||||
- **The sources are synthetic.** A coarse gradient with a fine dither, which is
|
||||
what `frame_budget.rs` synthesises for the same reason — a flat frame lets the
|
||||
memory system serve every sample from one cache line and flatters a box
|
||||
filter, and pure noise defeats the entropy coder in the other direction.
|
||||
|
||||
---
|
||||
|
||||
## Two gates, and how to read a failure
|
||||
|
||||
**Budget.** The requirement's own threshold. It does not move. Failing it means
|
||||
a requirement is violated.
|
||||
|
||||
**Regression.** More than 15% worse than the last recorded figure *on the same
|
||||
machine, against the same fixture*. Failing it means the code got slower while
|
||||
still inside the requirement — which is how most performance rot actually
|
||||
arrives, never over the line, always a little worse, until one day the line is
|
||||
crossed by a change that was not the cause.
|
||||
|
||||
A metric declares whether its budget is `machine_sensitive`. Those are asserted
|
||||
only under `--reference`, and reported everywhere else. §8 names *"the reference
|
||||
desktop"*, not CI, and it is right to: a container with two cores cannot speak
|
||||
to a throughput target written for twenty-four threads, and asserting one there
|
||||
would produce exactly what `core/dr-gpu/tests/frame_budget.rs` refused to
|
||||
produce — *"a red suite that everyone learns to ignore"*. Catalog open is not
|
||||
machine-sensitive: 2 s against an expected figure two orders of magnitude
|
||||
smaller is a threshold any machine can be held to.
|
||||
|
||||
Exit codes: `0` everything passed, `1` a gate failed, `2` the harness itself
|
||||
could not run. Distinguished so a CI log that says "failed" does not leave
|
||||
anyone guessing whether the code got slower or the fixture would not build.
|
||||
|
||||
**Release, always.** The workspace builds its own crates at `opt-level = 0` in
|
||||
dev, and every figure here is dominated by this workspace's own code — the JPEG
|
||||
decode, the box filter, the resample, the sharpen. A debug run measures rustc's
|
||||
shadow. The report says which profile it was built in on its second line.
|
||||
|
||||
---
|
||||
|
||||
## NFR-P8, and the question §4.1 asks
|
||||
|
||||
§4.1 says NFR-P8 *"must state whether it measures RSS inclusive or exclusive of
|
||||
GPU allocations, and whether it holds after SQLite's page cache warms on a 50k
|
||||
catalog."* Both halves have an answer.
|
||||
|
||||
**On the page cache: warm.** The probe runs the count, the timeline and
|
||||
twenty-five windows before it reads its counters, so SQLite's cache holds the
|
||||
b-tree pages a scroll touches. That is the right side to err on — a figure taken
|
||||
before the cache warms would understate a steady-state library.
|
||||
|
||||
**On GPU memory: RSS is exclusive of device-local allocations, and cannot be
|
||||
made otherwise.** A Vulkan allocation in a device-local heap never enters the
|
||||
process's address space, so nothing under `/proc/self/status` can see it. What
|
||||
*does* land in RSS is the host-visible side — staging buffers, mapped upload
|
||||
rings, the read-back `AdjustPass` performs on export — plus the driver's own
|
||||
resident pages.
|
||||
|
||||
So "idle memory < 500 MB" is two questions wearing one number, and a build
|
||||
holding 400 MB of RSS and 3 GB of textures would pass it.
|
||||
|
||||
**Recommendation: NFR-P8 should be restated as two figures** — host RSS
|
||||
exclusive of device-local memory, and a separate VRAM ceiling read from the
|
||||
adapter — because the second is the one that decides whether the application
|
||||
survives beside a browser on an 8 GB card, and nothing in this repository
|
||||
measures it today.
|
||||
|
||||
**And a decision is outstanding.** `catalog_idle_rss_mb` carries no budget
|
||||
because nobody has decided how much of the 500 MB belongs to the catalog layer
|
||||
and how much to everything above it. The suite records the number so that
|
||||
decision can be taken against a measurement rather than an estimate. When it is
|
||||
taken, put the figure in `budget` and the metric becomes a gate.
|
||||
|
||||
---
|
||||
|
||||
## What is not measured, and would be worth adding
|
||||
|
||||
- **The UI's own open.** `ui/dr-ui/src/library.rs` does not call
|
||||
`Catalog::count` or `Catalog::window`; it issues its own SQL against the same
|
||||
tables, with a `VISIBLE` predicate and a burst-folding clause. `dr-bench`
|
||||
cannot see those without depending on `dr-ui`, which would drag Slint into a
|
||||
job that has no display. **Falsifiable end:** when the grid's queries move
|
||||
down into `dr-catalog` — which is where SQL over catalog tables belongs —
|
||||
`catalog_open_ms` becomes the whole of the application's open and this caveat
|
||||
can be deleted rather than argued about.
|
||||
- **The remote sweep.** `spawn_thumbnail_sweep`'s wall clock against a real
|
||||
server is latency, not CPU, and is what FR-NC-3's design is judged by. It
|
||||
needs a server and belongs in a different kind of test.
|
||||
- **A cold disk.** See the fixture's limits above.
|
||||
- **Android.** §4.1 states a second column of targets and §8 asks for
|
||||
"periodically on the named reference Android devices". Nothing here runs on a
|
||||
device. Spike S10 is the piece of work that would start it.
|
||||
- **Everything with a frame in it.** See the out-of-scope list above.
|
||||
|
||||
---
|
||||
|
||||
## Running it
|
||||
|
||||
```sh
|
||||
# Measure and print. Judges nothing.
|
||||
cargo run --release -p dr-bench -- run
|
||||
|
||||
# Measure and gate. What CI runs.
|
||||
cargo run --release -p dr-bench -- check
|
||||
|
||||
# The same, with machine-sensitive budgets asserted too.
|
||||
cargo run --release -p dr-bench -- check --reference
|
||||
|
||||
# Rewrite bench-baseline.json from this run, and commit the diff.
|
||||
cargo run --release -p dr-bench -- record --reference
|
||||
```
|
||||
|
||||
Useful flags: `--fixture <dir>` (or `$DR_BENCH_DIR`) to put the synthetic
|
||||
catalog somewhere specific, `--lanes <n>` to pin the sweep's parallelism, and
|
||||
`--thumbnails <n>` to lengthen or shorten the throughput row.
|
||||
+33
-21
@@ -316,31 +316,43 @@ for interoperating with the editors FR-CAT-14 imports from.
|
||||
|
||||
---
|
||||
|
||||
## 8. The performance targets are unverified, not unmet
|
||||
## 8. The performance targets are half-verified, and the half that is left is the hard one
|
||||
|
||||
Eleven of the fifteen §4.1 targets carry no tag: NFR-P2, -P3, -P4, -P6, -P7, -P8, -P10, -P11, -P12,
|
||||
-P14, -P15. That is the uninteresting part of this section.
|
||||
§8 and §4.1 both require the same thing in the same words: an automated benchmark suite against a
|
||||
synthetic 50k catalog, run per commit, where **"a regression beyond a stated tolerance is a build
|
||||
failure, not a notification."** For most of this project's life it did not exist — no `benches/`, no
|
||||
criterion, no synthetic catalog, and three CI workflows that between them measured nothing.
|
||||
|
||||
The interesting part is that §8 and §4.1 both require the same thing, in the same words, and it does
|
||||
not exist: an automated benchmark suite against a synthetic 50k catalog, run per commit, where **"a
|
||||
regression beyond a stated tolerance is a build failure, not a notification."** There is no
|
||||
`benches/` directory in the workspace, no criterion dependency, and no synthetic catalog. The three
|
||||
CI workflows run `cargo fmt --check`, clippy, `cargo test --workspace`, a release build, an Android
|
||||
cross-build and a layering check. None of them measures anything, so there is no baseline to
|
||||
regress against and no tolerance to exceed.
|
||||
**It exists now, for everything that does not need a frame.** [`tools/bench`](../tools/bench) builds
|
||||
a deterministic 50,000-row catalog over a pool of a dozen real files, measures against it, and fails
|
||||
the build on a violated budget or a drift past tolerance;
|
||||
[`.gitea/workflows/benchmark.yml`](../.gitea/workflows/benchmark.yml) runs it on every push, and
|
||||
[benchmarks.md](benchmarks.md) is the account of what it does and does not cover. **NFR-P1** and
|
||||
**NFR-P3** are now genuinely gated, and R2's "catalog opens in under 2s" clause with them.
|
||||
|
||||
What does exist is narrower and genuinely good: `dr-gpu/examples/frame_budget` is a real instrument,
|
||||
its results are committed in [frame-budget.md](frame-budget.md) with the machine and profile named,
|
||||
and TD-4's before-and-after was measured with it. But it is run by hand — frame-budget.md's own
|
||||
instruction is "rerun and diff this file" — and the guard version that does live in CI skips itself
|
||||
where there is no GPU adapter, which the workflow notes is the normal case on a runner, while
|
||||
asserting its CPU half only when `debug_assertions` is off, which a dev-profile `cargo test` is not.
|
||||
In CI it therefore asserts approximately nothing.
|
||||
Three qualifications, all of them stated in the harness itself rather than only here:
|
||||
|
||||
**The claim to take from this is precise.** Nothing here says the performance targets are missed.
|
||||
Several are plausibly met. It says that if one were broken tomorrow, nobody would find out — which
|
||||
is the failure mode §8 was written to prevent, and the reason it belongs in this document rather
|
||||
than in a backlog.
|
||||
- **The numbers have not been recorded yet.** Every `recorded` field in
|
||||
[bench-baseline.json](bench-baseline.json) is `null`, deliberately: a fabricated baseline is worse
|
||||
than none. Until `dr-bench record --reference` is run on the reference desktop and committed, the
|
||||
budget gate works and the regression gate does not.
|
||||
- **NFR-P7 and NFR-P8 are half-measured and are not tagged.** The export row covers the encode half
|
||||
of the chain and no GPU render, so it can fail the requirement and cannot pass it. The memory row
|
||||
covers a process holding the catalog and nothing else — no toolkit, no adapter — so it is the
|
||||
catalog layer's share of the 500 MB rather than the figure NFR-P8 is about. Neither carries a
|
||||
`TRACES:` tag, which is the point.
|
||||
- **NFR-P8 needs a decision, not more code.** How much of its 500 MB belongs below the UI is
|
||||
unstated, and until somebody says, the metric can record but not judge. [benchmarks.md](benchmarks.md)
|
||||
also answers the question §4.1 raises about GPU memory — RSS cannot see device-local allocations
|
||||
at all — and recommends restating the requirement as two figures.
|
||||
|
||||
**What is left is the frame-timing half, and it is the hard one.** NFR-P2, -P4, -P5, -P6, -P9, -P10,
|
||||
-P11, -P12, -P13, -P14 and -P15 all need a probe inside a running Slint application, a GPU adapter,
|
||||
or both. `dr-gpu/examples/frame_budget` is a real instrument for the GPU part and its results are
|
||||
committed in [frame-budget.md](frame-budget.md) with the machine and profile named — but it is run
|
||||
by hand, and the guard version in CI skips itself where there is no adapter, which is the normal
|
||||
case on a runner. So the claim to take from this section is now narrower than it was, and still
|
||||
true: **a scroll that dropped to 30 fps tomorrow would reach a user before it reached CI.**
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
[package]
|
||||
name = "dr-bench"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
license.workspace = true
|
||||
|
||||
# The suite deliberately depends on no GPU crate and no UI toolkit.
|
||||
#
|
||||
# That is not tidiness, it is what makes the CI job affordable: `cargo build
|
||||
# --release -p dr-bench` resolves the catalog, the decoder, the thumbnail store
|
||||
# and the encoder, and stops there. Adding `dr-gpu` would pull wgpu and naga
|
||||
# into a job that has no adapter to use them with, and adding `dr-ui` would
|
||||
# pull Slint. The GPU half of the suite is `dr-gpu`'s own frame-budget test,
|
||||
# which already exists and already skips itself where there is no device — see
|
||||
# `docs/benchmarks.md`.
|
||||
[dependencies]
|
||||
dr-types.workspace = true
|
||||
dr-catalog.workspace = true
|
||||
dr-thumbs.workspace = true
|
||||
dr-decode.workspace = true
|
||||
dr-export.workspace = true
|
||||
rusqlite.workspace = true
|
||||
serde.workspace = true
|
||||
serde_json.workspace = true
|
||||
anyhow.workspace = true
|
||||
log.workspace = true
|
||||
env_logger.workspace = true
|
||||
@@ -0,0 +1,300 @@
|
||||
//! The committed numbers, and what counts as a regression against them.
|
||||
//!
|
||||
//! `docs/frame-budget.md` commits its measurements by hand and says why: *"a
|
||||
//! regression should be a diff rather than somebody's memory."* This is the
|
||||
//! same idea in a form a program can read, because §8 asks for more than a
|
||||
//! record — *"a regression beyond stated tolerance fails the build"*.
|
||||
//!
|
||||
//! # Two gates, and they are not the same gate
|
||||
//!
|
||||
//! **The budget** is the requirement's own number: 2 s to open a catalog, 100
|
||||
//! images a second through the preview path. It does not move. A build that
|
||||
//! violates it has violated a requirement, and no amount of "but it was always
|
||||
//! like that" changes it.
|
||||
//!
|
||||
//! **The baseline** is what this machine last measured. It moves — deliberately
|
||||
//! and by hand, through `dr-bench record` — and its job is to catch the change
|
||||
//! that is still inside the budget but has halved the headroom. Most real
|
||||
//! performance rot arrives that way: never over the line, always a little
|
||||
//! worse, until one day the line is crossed by a change that was not the cause.
|
||||
//!
|
||||
//! # Why a machine-sensitive metric skips its budget off the reference desktop
|
||||
//!
|
||||
//! §8 names *"the reference desktop"*, not CI, and it is right to. A container
|
||||
//! with two cores cannot speak to a throughput target written for a
|
||||
//! twenty-four-thread machine, and asserting it there would produce exactly
|
||||
//! what `core/dr-gpu/tests/frame_budget.rs` refused to produce: *"a red suite
|
||||
//! that everyone learns to ignore"*. So a metric declares whether its budget
|
||||
//! is machine-sensitive. Those budgets are asserted under `--reference` and
|
||||
//! reported everywhere else; the ones with orders of magnitude of headroom —
|
||||
//! catalog open against two seconds — are asserted everywhere, because a
|
||||
//! failure there is a real failure on any machine.
|
||||
//!
|
||||
//! # Why the committed file starts with no numbers in it
|
||||
//!
|
||||
//! Because nobody had run it yet. Writing plausible-looking figures into a
|
||||
//! baseline is the one thing that would make the whole suite worthless: every
|
||||
//! later comparison would be against a guess, and the first genuine regression
|
||||
//! would be invisible or, worse, a fabricated improvement. `recorded` is
|
||||
//! therefore `null` until somebody runs `dr-bench record --reference` on the
|
||||
//! reference desktop and commits the diff. Until then the budget gate works
|
||||
//! and the regression gate says so rather than pretending.
|
||||
|
||||
use std::collections::BTreeMap;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::fixture::Stamp;
|
||||
|
||||
/// Which way is better.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(rename_all = "snake_case")]
|
||||
pub enum Direction {
|
||||
LowerIsBetter,
|
||||
HigherIsBetter,
|
||||
}
|
||||
|
||||
/// One measured quantity: what it is, what it must be, and what it was.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct Metric {
|
||||
/// The requirement ID this speaks to, or a note saying it speaks to only
|
||||
/// part of one. Free text on purpose: several of these describe a fraction
|
||||
/// of a requirement, and a bare ID here would read as the whole of it.
|
||||
pub requirement: String,
|
||||
/// One sentence a reader of the JSON can understand without the code.
|
||||
pub what: String,
|
||||
pub unit: String,
|
||||
pub direction: Direction,
|
||||
/// Whether the budget below is a statement about a machine as much as
|
||||
/// about the code. See this module's header.
|
||||
pub machine_sensitive: bool,
|
||||
/// The requirement's own threshold, where it has one this can check.
|
||||
pub budget: Option<f64>,
|
||||
/// What the last `record` measured. `null` until one has been taken.
|
||||
pub recorded: Option<f64>,
|
||||
}
|
||||
|
||||
/// The committed file.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct Baseline {
|
||||
/// Prose for whoever opens the JSON first. Kept in the file rather than
|
||||
/// only in `docs/benchmarks.md`, because the person who finds this in a
|
||||
/// failing CI log is not reading the docs directory at that moment.
|
||||
#[serde(rename = "_readme")]
|
||||
pub readme: Vec<String>,
|
||||
/// How much worse than [`Metric::recorded`] a metric may get before the
|
||||
/// build fails, as a fraction. 0.15 is fifteen per cent.
|
||||
pub tolerance: f64,
|
||||
/// The machine the recorded figures came from, as [`machine_id`] spells
|
||||
/// it. Compared before a regression is judged: drift against a different
|
||||
/// machine's numbers is not a regression, it is a different machine.
|
||||
pub recorded_on: Option<String>,
|
||||
/// Unix seconds. An integer rather than a formatted date because this
|
||||
/// workspace has no date library and adding one for a comment would be a
|
||||
/// poor trade.
|
||||
pub recorded_at_unix: Option<i64>,
|
||||
/// The fixture the recorded figures describe. A number measured against a
|
||||
/// different workload is not comparable, and this is what says so.
|
||||
pub fixture: Option<Stamp>,
|
||||
pub metrics: BTreeMap<String, Metric>,
|
||||
}
|
||||
|
||||
/// How a measurement compares.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct Judgement {
|
||||
/// Past the requirement's own threshold. A build failure.
|
||||
pub over_budget: bool,
|
||||
/// Worse than the recorded baseline by more than the tolerance. A build
|
||||
/// failure.
|
||||
pub regressed: bool,
|
||||
/// Fractional change against the baseline, positive meaning worse.
|
||||
pub drift: Option<f64>,
|
||||
/// A budget exists but was not asserted, because it is machine-sensitive
|
||||
/// and this is not the reference desktop.
|
||||
pub budget_deferred: bool,
|
||||
}
|
||||
|
||||
/// What a judgement is made in the light of.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct Judging {
|
||||
/// This run declares itself the reference desktop.
|
||||
pub reference: bool,
|
||||
/// This run is on the machine the baseline was recorded on.
|
||||
pub same_machine: bool,
|
||||
pub tolerance: f64,
|
||||
}
|
||||
|
||||
impl Metric {
|
||||
/// Judge `measured` against the budget and the baseline.
|
||||
pub fn judge(&self, measured: f64, cx: Judging) -> Judgement {
|
||||
let violates = |threshold: f64| match self.direction {
|
||||
Direction::LowerIsBetter => measured > threshold,
|
||||
Direction::HigherIsBetter => measured < threshold,
|
||||
};
|
||||
let assert_budget = self.budget.is_some() && (cx.reference || !self.machine_sensitive);
|
||||
|
||||
let drift = match self.recorded {
|
||||
// A recorded zero would divide by nothing, and a recorded figure
|
||||
// of zero is a broken record rather than a very fast one.
|
||||
Some(was) if was > 0.0 => Some(match self.direction {
|
||||
Direction::LowerIsBetter => (measured - was) / was,
|
||||
Direction::HigherIsBetter => (was - measured) / was,
|
||||
}),
|
||||
_ => None,
|
||||
};
|
||||
|
||||
Judgement {
|
||||
over_budget: assert_budget && self.budget.is_some_and(violates),
|
||||
regressed: cx.same_machine && drift.is_some_and(|d| d > cx.tolerance),
|
||||
drift,
|
||||
budget_deferred: self.budget.is_some() && !assert_budget,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Baseline {
|
||||
pub fn load(path: &Path) -> Result<Baseline> {
|
||||
let text = std::fs::read_to_string(path)
|
||||
.with_context(|| format!("reading the baseline at {}", path.display()))?;
|
||||
serde_json::from_str(&text)
|
||||
.with_context(|| format!("parsing the baseline at {}", path.display()))
|
||||
}
|
||||
|
||||
pub fn save(&self, path: &Path) -> Result<()> {
|
||||
let mut text = serde_json::to_string_pretty(self)?;
|
||||
// A trailing newline, so the file is a well-behaved text file and a
|
||||
// `record` that changed nothing produces an empty diff.
|
||||
text.push('\n');
|
||||
std::fs::write(path, text)
|
||||
.with_context(|| format!("writing the baseline to {}", path.display()))
|
||||
}
|
||||
|
||||
/// Where the committed baseline lives, found the way `tools/traceability`
|
||||
/// finds the repo root: by walking up from this crate's manifest until
|
||||
/// `docs/requirements.md` appears.
|
||||
pub fn default_path() -> Result<PathBuf> {
|
||||
let mut dir = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
|
||||
while !dir.join("docs/requirements.md").exists() {
|
||||
if !dir.pop() {
|
||||
anyhow::bail!("could not locate the repo root above this crate");
|
||||
}
|
||||
}
|
||||
Ok(dir.join("docs/bench-baseline.json"))
|
||||
}
|
||||
}
|
||||
|
||||
/// How this machine is named in the baseline.
|
||||
///
|
||||
/// Host name plus thread count. Not a hardware inventory — it exists to answer
|
||||
/// one question, "are these numbers from here?", and to answer it the same way
|
||||
/// twice on the same box. A container whose hostname changes per run therefore
|
||||
/// never matches, which is the correct answer for CI: its drift is information,
|
||||
/// not a verdict.
|
||||
pub fn machine_id() -> String {
|
||||
let host = std::fs::read_to_string("/proc/sys/kernel/hostname")
|
||||
.map(|s| s.trim().to_string())
|
||||
.unwrap_or_else(|_| "unknown-host".to_string());
|
||||
let threads = std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(0);
|
||||
format!("{host} ({threads} threads)")
|
||||
}
|
||||
|
||||
/// Unix seconds now, or 0 if the clock is before 1970, which it is not.
|
||||
pub fn now_unix() -> i64 {
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_secs() as i64)
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn metric(direction: Direction, budget: Option<f64>, recorded: Option<f64>) -> Metric {
|
||||
Metric {
|
||||
requirement: "NFR-TEST".into(),
|
||||
what: "a test metric".into(),
|
||||
unit: "ms".into(),
|
||||
direction,
|
||||
machine_sensitive: false,
|
||||
budget,
|
||||
recorded,
|
||||
}
|
||||
}
|
||||
|
||||
fn cx(same_machine: bool) -> Judging {
|
||||
Judging {
|
||||
reference: true,
|
||||
same_machine,
|
||||
tolerance: 0.15,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_budget_is_directional() {
|
||||
// The bug this exists to prevent: judging a throughput target the way
|
||||
// a latency target is judged, so a suite that got twice as slow passes
|
||||
// and one that got twice as fast fails.
|
||||
let latency = metric(Direction::LowerIsBetter, Some(100.0), None);
|
||||
assert!(latency.judge(101.0, cx(true)).over_budget);
|
||||
assert!(!latency.judge(99.0, cx(true)).over_budget);
|
||||
|
||||
let throughput = metric(Direction::HigherIsBetter, Some(100.0), None);
|
||||
assert!(throughput.judge(99.0, cx(true)).over_budget);
|
||||
assert!(!throughput.judge(101.0, cx(true)).over_budget);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn drift_is_positive_when_things_got_worse_whichever_way_that_is() {
|
||||
let latency = metric(Direction::LowerIsBetter, None, Some(100.0));
|
||||
assert!(latency.judge(120.0, cx(true)).drift.unwrap() > 0.0);
|
||||
let throughput = metric(Direction::HigherIsBetter, None, Some(100.0));
|
||||
assert!(throughput.judge(80.0, cx(true)).drift.unwrap() > 0.0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn drift_against_another_machine_is_reported_but_never_a_failure() {
|
||||
// CI is not the reference desktop. Its numbers are worth printing and
|
||||
// are not a verdict on anybody's commit.
|
||||
let m = metric(Direction::LowerIsBetter, None, Some(100.0));
|
||||
let elsewhere = m.judge(400.0, cx(false));
|
||||
assert!(elsewhere.drift.unwrap() > 0.15);
|
||||
assert!(!elsewhere.regressed);
|
||||
assert!(m.judge(400.0, cx(true)).regressed);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unrecorded_metric_cannot_regress() {
|
||||
// The state the committed file ships in. It must gate on the budget
|
||||
// and stay silent about drift, rather than treating null as zero and
|
||||
// declaring an infinite regression.
|
||||
let m = metric(Direction::LowerIsBetter, Some(100.0), None);
|
||||
let j = m.judge(50.0, cx(true));
|
||||
assert!(j.drift.is_none());
|
||||
assert!(!j.regressed);
|
||||
assert!(!j.over_budget);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_machine_sensitive_budget_defers_off_the_reference_desktop() {
|
||||
let mut m = metric(Direction::HigherIsBetter, Some(100.0), None);
|
||||
m.machine_sensitive = true;
|
||||
let on_ci = Judging {
|
||||
reference: false,
|
||||
same_machine: false,
|
||||
tolerance: 0.15,
|
||||
};
|
||||
let j = m.judge(10.0, on_ci);
|
||||
// A two-core runner must not fail a target written for twenty-four.
|
||||
assert!(!j.over_budget);
|
||||
// And the report has to say the budget was not applied, rather than
|
||||
// letting a deferred budget read as a passed one.
|
||||
assert!(j.budget_deferred);
|
||||
// On the reference desktop the same figure is judged.
|
||||
assert!(m.judge(10.0, cx(false)).over_budget);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,178 @@
|
||||
//! TRACES: NFR-P1
|
||||
//! Opening a fifty-thousand-image catalog, and what that actually involves.
|
||||
//!
|
||||
//! NFR-P1 says under two seconds on the reference desktop. Until this file
|
||||
//! existed the number had never been measured, which made it a wish — and
|
||||
//! `dr-catalog`'s `lib.rs` has carried an `NFR-P1` tag the whole time on code
|
||||
//! that describes the target rather than checking it. This is the check.
|
||||
//!
|
||||
//! # What counts as "open"
|
||||
//!
|
||||
//! Not `Catalog::open` alone. That call returns before anything is on screen,
|
||||
//! and a user's "the catalog opened" is the moment the grid has cells in it.
|
||||
//! So the measured span is the four things the library view cannot paint
|
||||
//! without:
|
||||
//!
|
||||
//! 1. [`Catalog::open`] — connect, migrate if needed, and **backfill**. The
|
||||
//! backfill is the interesting one: `schema::backfill` runs on every open
|
||||
//! and is three passes over the images table, so it is O(library) work on a
|
||||
//! path whose budget is stated in absolute seconds.
|
||||
//! 2. [`Catalog::count`] — the total, which is what sizes the scrollbar.
|
||||
//! 3. [`Catalog::window`] — the first screenful of rows.
|
||||
//! 4. [`Catalog::timeline`] — the scrubber's buckets, drawn beside the grid
|
||||
//! from the first frame.
|
||||
//!
|
||||
//! `open` alone is reported separately anyway, because if the two ever diverge
|
||||
//! sharply the fix is in a different place.
|
||||
//!
|
||||
//! # What this does *not* measure, said out loud
|
||||
//!
|
||||
//! `ui/dr-ui/src/library.rs` does not call [`Catalog::count`] or
|
||||
//! [`Catalog::window`]. It issues its own SQL — `total_images_scoped`,
|
||||
//! `total_images_filtered` and friends — against the same tables, with a
|
||||
//! `VISIBLE` predicate and a burst-folding clause that this crate cannot see
|
||||
//! without depending on the UI, which would drag Slint into a benchmark job
|
||||
//! that has no display. So the number here is the **catalog crate's** open
|
||||
//! path, and the application's is that plus whatever those queries cost.
|
||||
//!
|
||||
//! That gap is a real limit on what this file can certify, and it has a
|
||||
//! falsifiable end: when the grid's queries move down into `dr-catalog` —
|
||||
//! which is where SQL over catalog tables belongs — this measurement becomes
|
||||
//! the whole of the application's open, and the caveat can be deleted rather
|
||||
//! than argued about.
|
||||
//!
|
||||
//! # Cold and warm
|
||||
//!
|
||||
//! Both are reported. The first open in a process pays for SQLite's page cache
|
||||
//! being empty and for the schema being read; the second pays for neither, and
|
||||
//! is what a user gets when they close and reopen a library in the same
|
||||
//! session. §4.1 asks the question directly for NFR-P8 and it is worth having
|
||||
//! the answer here too. Neither figure is a genuinely cold *disk*: the fixture
|
||||
//! was written by this same suite or by an earlier run of it, so the file is
|
||||
//! in the OS page cache. On the reference desktop's NVMe a truly cold read of
|
||||
//! a ~14 MB file is a few tens of milliseconds; on a spinning disk it is not.
|
||||
|
||||
use std::path::Path;
|
||||
use std::time::Instant;
|
||||
|
||||
use anyhow::Result;
|
||||
use dr_catalog::{Catalog, Granularity, Query};
|
||||
use dr_types::Selector;
|
||||
|
||||
use crate::stats::{ms, Percentiles, Rng};
|
||||
|
||||
/// Rows fetched for the first screenful.
|
||||
///
|
||||
/// A dense grid on a 4K display is around three hundred cells; four hundred is
|
||||
/// that plus the prefetch margin R2 asks for. Not the whole library, because
|
||||
/// FR-CAT-4 is explicit that memory must not scale with it — a benchmark that
|
||||
/// asked for 50,000 rows would be measuring the requirement's violation.
|
||||
const WINDOW: usize = 400;
|
||||
|
||||
/// The clock the query compiler is handed.
|
||||
///
|
||||
/// Fixed rather than read from the system, so a rolling date filter would
|
||||
/// compile to the same SQL on every run. The unfiltered query does not consult
|
||||
/// it at all; this is here so that adding a dated row later does not silently
|
||||
/// make the suite time-dependent.
|
||||
const NOW: i64 = 2_000_000_000;
|
||||
|
||||
/// What one measured open produced.
|
||||
pub struct Open {
|
||||
/// Open, count, first window, timeline — the whole span, first time.
|
||||
pub cold_ms: f64,
|
||||
/// The same four calls on a second connection in the same process.
|
||||
pub warm_ms: f64,
|
||||
/// [`Catalog::open`] on its own, out of the cold span.
|
||||
pub open_only_ms: f64,
|
||||
/// How many images the count found. Reported so a fixture that failed to
|
||||
/// populate cannot masquerade as a very fast open.
|
||||
pub images: usize,
|
||||
/// Timeline buckets at monthly granularity.
|
||||
pub buckets: usize,
|
||||
/// One window fetched at a random offset — the scroll, minus the drawing.
|
||||
pub window_ms: Percentiles,
|
||||
/// Count plus first window under a rating filter, which compiles to a
|
||||
/// correlated subquery over `versions` (see `query::default_version_scalar`).
|
||||
pub filtered_ms: f64,
|
||||
}
|
||||
|
||||
/// Measure an open of the catalog at `path`, then `windows` random windows.
|
||||
pub fn measure(path: &Path, windows: usize) -> Result<Open> {
|
||||
let q = Query::default();
|
||||
|
||||
let started = Instant::now();
|
||||
let catalog = open(path)?;
|
||||
let open_only_ms = ms(started.elapsed());
|
||||
let images = catalog.count(&q, NOW)?;
|
||||
let rows = catalog.window(&q, 0..WINDOW, NOW)?;
|
||||
let buckets = catalog.timeline(&q, Granularity::Month, NOW)?.len();
|
||||
let cold_ms = ms(started.elapsed());
|
||||
|
||||
// A catalog that returned nothing would post an excellent time. Checked
|
||||
// rather than trusted, because the failure mode is a *fast* wrong answer.
|
||||
anyhow::ensure!(
|
||||
!rows.is_empty() && images > 0 && buckets > 0,
|
||||
"the fixture catalog answered with {images} images, {} rows and {buckets} buckets — \
|
||||
the measurement below would be meaningless",
|
||||
rows.len()
|
||||
);
|
||||
drop(catalog);
|
||||
|
||||
let started = Instant::now();
|
||||
let catalog = open(path)?;
|
||||
let _ = catalog.count(&q, NOW)?;
|
||||
let _ = catalog.window(&q, 0..WINDOW, NOW)?;
|
||||
let _ = catalog.timeline(&q, Granularity::Month, NOW)?;
|
||||
let warm_ms = ms(started.elapsed());
|
||||
|
||||
// The scroll. Offsets are drawn from a fixed seed rather than swept in
|
||||
// order, because a sequential sweep would be answered increasingly out of
|
||||
// SQLite's own cache and would flatter the deep end of the library — which
|
||||
// is exactly the end a person reaches by dragging the scrollbar.
|
||||
let mut rng = Rng::new(0x5C_20_11);
|
||||
let span = images.saturating_sub(WINDOW).max(1) as u64;
|
||||
// Discarded: the first window of a new connection compiles the statement
|
||||
// and faults in the b-tree's upper levels, and neither recurs while
|
||||
// scrolling.
|
||||
for _ in 0..4 {
|
||||
let start = rng.below(span) as usize;
|
||||
let _ = catalog.window(&q, start..start + WINDOW, NOW)?;
|
||||
}
|
||||
let mut samples = Vec::with_capacity(windows);
|
||||
for _ in 0..windows {
|
||||
let start = rng.below(span) as usize;
|
||||
let t = Instant::now();
|
||||
let rows = catalog.window(&q, start..start + WINDOW, NOW)?;
|
||||
samples.push(ms(t.elapsed()));
|
||||
debug_assert!(!rows.is_empty());
|
||||
}
|
||||
|
||||
// A filter that has to reach the default version for every candidate row.
|
||||
// Cheap to add and the one query shape in the grid that is not a scan of
|
||||
// `images` alone, so a regression in it would otherwise show up first as a
|
||||
// user complaint.
|
||||
let rated = Query {
|
||||
filter: Selector::Rating { min: 2 },
|
||||
..Query::default()
|
||||
};
|
||||
let t = Instant::now();
|
||||
let _ = catalog.count(&rated, NOW)?;
|
||||
let _ = catalog.window(&rated, 0..WINDOW, NOW)?;
|
||||
let filtered_ms = ms(t.elapsed());
|
||||
|
||||
Ok(Open {
|
||||
cold_ms,
|
||||
warm_ms,
|
||||
open_only_ms,
|
||||
images,
|
||||
buckets,
|
||||
window_ms: Percentiles::of(samples),
|
||||
filtered_ms,
|
||||
})
|
||||
}
|
||||
|
||||
fn open(path: &Path) -> Result<Catalog> {
|
||||
Catalog::open(path)
|
||||
.map_err(|e| anyhow::anyhow!("opening the fixture catalog at {}: {e}", path.display()))
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
//! The half of a 24 MP export that needs no GPU.
|
||||
//!
|
||||
//! # This cannot certify NFR-P7, and is not tagged as though it could
|
||||
//!
|
||||
//! NFR-P7 is "full-resolution export (24 MP, **full chain**) < 2 s". The full
|
||||
//! chain is decode, demosaic, a GPU render at full resolution, a read-back,
|
||||
//! and then everything `dr-export` does — resize, output sharpening, encode.
|
||||
//! Only the last three of those run without an adapter, and the CI runner has
|
||||
//! none. So what is measured here is the encode half, and no requirement tag
|
||||
//! anywhere in this crate names NFR-P7.
|
||||
//!
|
||||
//! (Written without the tag's own spelling on purpose. `tools/traceability`
|
||||
//! matches the marker anywhere on a line and parses the identifier after it, so
|
||||
//! a sentence saying "there is no tag for NFR-P7" would *be* a tag for NFR-P7 —
|
||||
//! a disclaimer that made itself false.)
|
||||
//!
|
||||
//! That is a deliberate refusal rather than an oversight. `CONTRIBUTING.md`
|
||||
//! asks that a requirement be closed by a test that would fail if the
|
||||
//! behaviour were removed, and `docs/code-health.md` CH-4 records what the
|
||||
//! coverage figure looks like when tags are hung on plumbing instead. A tag
|
||||
//! here would say the export budget is checked; the GPU half of it would still
|
||||
//! be unchecked.
|
||||
//!
|
||||
//! # What the number is still good for
|
||||
//!
|
||||
//! It is a **one-sided** gate, and that is worth having. The encode half is a
|
||||
//! lower bound on the whole: if resizing, sharpening and encoding 24 MP alone
|
||||
//! take longer than two seconds, NFR-P7 is violated no matter how fast the
|
||||
//! render is. So the budget in `docs/bench-baseline.json` is the requirement's
|
||||
//! own 2000 ms, and exceeding it fails the build honestly. Coming in under it
|
||||
//! proves nothing about the requirement, and the report says so rather than
|
||||
//! printing a tick.
|
||||
//!
|
||||
//! # Two rows
|
||||
//!
|
||||
//! `Original` is the one the budget is judged on: it is the archival export,
|
||||
//! the largest encode, and the case FR-EXP-9 is about. `LongEdge(2048)` is the
|
||||
//! ordinary web export, where the resample does real work and the encode does
|
||||
//! very little — it is reported because a regression in `size::resample` would
|
||||
//! be invisible in the first row, where source and target dimensions are equal.
|
||||
|
||||
use std::time::Instant;
|
||||
|
||||
use anyhow::Result;
|
||||
use dr_export::{export, Frame};
|
||||
use dr_types::{ExportFormat, ExportSettings, OutputSharpening, SizingMode};
|
||||
|
||||
use crate::fixture::plausible_frame;
|
||||
use crate::stats::{ms, Percentiles};
|
||||
|
||||
/// The frame every row exports.
|
||||
///
|
||||
/// 6000 × 4000 is 24.0 MP — a full-frame body, and the exact figure NFR-P7
|
||||
/// names. As RGBA8 it is 96 MB, and the export path holds a resized copy and a
|
||||
/// sharpened copy alongside it, so a run needs roughly 300 MB of headroom.
|
||||
/// Worth knowing before a small runner reports this as a mysterious kill.
|
||||
pub const SOURCE: (u32, u32) = (6000, 4000);
|
||||
|
||||
/// Measured exports per row. Not a hundred: one 24 MP encode is most of a
|
||||
/// second, and a hundred of them would be a two-minute CI step to establish
|
||||
/// what five establish. Nearest-rank p99 of five is the worst of the five,
|
||||
/// which for a row this expensive is the honest reading anyway.
|
||||
const RUNS: usize = 5;
|
||||
|
||||
/// A row of the export table.
|
||||
pub struct EncodeRun {
|
||||
/// The name this row carries in `docs/bench-baseline.json`.
|
||||
pub key: &'static str,
|
||||
pub label: &'static str,
|
||||
pub width: u32,
|
||||
pub height: u32,
|
||||
/// Encoded file size, so a row that silently stopped compressing is
|
||||
/// visible as well as a row that got slow.
|
||||
pub bytes: usize,
|
||||
pub times: Percentiles,
|
||||
}
|
||||
|
||||
/// Export the same 24 MP frame at each sizing, timing `dr_export::export`.
|
||||
pub fn measure() -> Result<Vec<EncodeRun>> {
|
||||
let (w, h) = SOURCE;
|
||||
// Detail at every scale, for the same reason the thumbnail fixture has it:
|
||||
// a flat frame compresses to almost nothing and would make the encoder
|
||||
// look several times faster than any photograph makes it.
|
||||
let frame = Frame::new(w, h, plausible_frame(w, h, 0))
|
||||
.map_err(|e| anyhow::anyhow!("building the 24 MP bench frame: {e}"))?;
|
||||
|
||||
let sizings: [(&'static str, &'static str, SizingMode); 2] = [
|
||||
("export_24mp_original_ms", "original", SizingMode::Original),
|
||||
(
|
||||
"export_24mp_long_edge_2048_ms",
|
||||
"long edge 2048",
|
||||
SizingMode::LongEdge(2048),
|
||||
),
|
||||
];
|
||||
|
||||
let mut rows = Vec::with_capacity(sizings.len());
|
||||
for (key, label, sizing) in sizings {
|
||||
let settings = ExportSettings {
|
||||
format: ExportFormat::Jpeg,
|
||||
// 90 is the default and what a photographer would not need to
|
||||
// change; quality moves encode time, so it belongs in the record.
|
||||
quality: 90,
|
||||
sizing,
|
||||
sharpening: OutputSharpening::Screen,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
// Discarded. The first export of a process grows the allocator to hold
|
||||
// three 24 MP buffers, which is a cost paid once and not per file in
|
||||
// the batch export FR-EXP-7 describes.
|
||||
let warm = run_once(&frame, &settings)?;
|
||||
|
||||
let mut samples = Vec::with_capacity(RUNS);
|
||||
let mut last = warm;
|
||||
for _ in 0..RUNS {
|
||||
let started = Instant::now();
|
||||
last = run_once(&frame, &settings)?;
|
||||
samples.push(ms(started.elapsed()));
|
||||
}
|
||||
|
||||
rows.push(EncodeRun {
|
||||
key,
|
||||
label,
|
||||
width: last.0,
|
||||
height: last.1,
|
||||
bytes: last.2,
|
||||
times: Percentiles::of(samples),
|
||||
});
|
||||
}
|
||||
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
/// One export, returning what it produced rather than the pixels.
|
||||
fn run_once(frame: &Frame, settings: &ExportSettings) -> Result<(u32, u32, usize)> {
|
||||
let encoded = export(frame, settings, "bench.jpg".to_string(), None)
|
||||
.map_err(|e| anyhow::anyhow!("exporting the bench frame: {e}"))?;
|
||||
Ok((encoded.width, encoded.height, encoded.bytes.len()))
|
||||
}
|
||||
@@ -0,0 +1,448 @@
|
||||
//! The synthetic 50k catalog, and the handful of real files it points at.
|
||||
//!
|
||||
//! `docs/requirements.md` §8 asks for "an automated benchmark suite against a
|
||||
//! synthetic 50k catalog". The hard part of that sentence is *50k*: a real
|
||||
//! library of that size is several terabytes and cannot live in a repository,
|
||||
//! in a CI cache, or on a laptop that also has to compile the thing.
|
||||
//!
|
||||
//! # The trick, and what it costs
|
||||
//!
|
||||
//! Rows are cheap and pixels are not. So this builds **fifty thousand catalog
|
||||
//! rows** over a **pool of a dozen real image files**, each referenced by
|
||||
//! several thousand of them. Everything the catalog half of the suite measures
|
||||
//! — opening, counting, windowing, bucketing a timeline — touches only rows,
|
||||
//! and is therefore exact. Everything the pixel half measures — decode,
|
||||
//! downscale, orient, encode — touches one file at a time and does not care
|
||||
//! how many rows point at it. The fixture is ~14 MB on disk instead of ~2 TB
|
||||
//! and neither half is flattered by that.
|
||||
//!
|
||||
//! What it *does* cost is stated rather than hidden: the file pool is small
|
||||
//! enough to sit in the OS page cache, so [`crate::thumbnails`] measures CPU
|
||||
//! throughput with the read already paid for. That is the right thing to
|
||||
//! measure for NFR-P3 — the target is written about the embedded preview path,
|
||||
//! not about a disk — but it is not a claim about a cold library on spinning
|
||||
//! rust, and the harness does not make one.
|
||||
//!
|
||||
//! # Reproducible from a seed
|
||||
//!
|
||||
//! Every value comes from [`Rng`], seeded once. Two machines running the same
|
||||
//! seed build byte-comparable catalogs, which is the property that lets a
|
||||
//! number measured on the reference desktop be compared with a number measured
|
||||
//! anywhere else. [`Stamp`] records what a directory was built from, so a
|
||||
//! fixture is reused when it matches and rebuilt when it does not — including
|
||||
//! when `dr-catalog`'s schema version moves, since a catalog built by an older
|
||||
//! build would otherwise be measured through a migration that a user's would
|
||||
//! not run.
|
||||
//!
|
||||
//! # The sources are generated, not committed
|
||||
//!
|
||||
//! No photograph in this repository is licensed for redistribution, and a
|
||||
//! dozen camera previews would be megabytes of binary in git for ever. So the
|
||||
//! pool is synthesised: a coarse gradient with a fine dither on top, which is
|
||||
//! the same shape `core/dr-gpu/examples/frame_budget.rs` synthesises its source
|
||||
//! from and for the same reason. A flat frame lets the memory system serve
|
||||
//! every sample from one cache line, which flatters a box filter by an amount
|
||||
//! that has nothing to do with photographs; pure noise defeats the JPEG
|
||||
//! encoder's entropy coder in the other direction and would make the encode
|
||||
//! half of a thumbnail look worse than any real image ever does.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use dr_catalog::Catalog;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::stats::Rng;
|
||||
|
||||
/// How many distinct image files the pool holds.
|
||||
///
|
||||
/// Twelve rather than one, so that a decode measured over a batch is not one
|
||||
/// file's quirks repeated — a single frame that happened to compress unusually
|
||||
/// well would set the whole number — and rather than fifty thousand, so the
|
||||
/// fixture stays a directory a person can look at.
|
||||
pub const SOURCE_POOL: usize = 12;
|
||||
|
||||
/// The size of one pooled file, in pixels.
|
||||
///
|
||||
/// 1620×1080 is not a round number: it is what `core/dr-decode/src/preview.rs`
|
||||
/// records a Canon CR2 carrying in IFD2, and the embedded preview is what
|
||||
/// NFR-P3 names. A camera JPEG is 24 MP and a camera *preview* is about this,
|
||||
/// so measuring the preview path against a 24 MP file would measure something
|
||||
/// the sweep never does.
|
||||
pub const PREVIEW: (u32, u32) = (1620, 1080);
|
||||
|
||||
/// Folders the rows are spread across.
|
||||
///
|
||||
/// Spread evenly and without regard to capture date, because what a folder
|
||||
/// count decides is the cost of the folder filter's `IN (SELECT …)` and the
|
||||
/// size of the `folders` table — not which image is in which.
|
||||
const FOLDERS: usize = 400;
|
||||
|
||||
/// The earliest capture time in the fixture: 13 December 2015, UTC.
|
||||
///
|
||||
/// Fixed rather than relative to the clock. A library whose dates moved with
|
||||
/// the calendar would make `timeline` bucket differently from one month to the
|
||||
/// next, and a benchmark that measures a different query each time it runs is
|
||||
/// not measuring a regression.
|
||||
const EPOCH: i64 = 1_450_000_000;
|
||||
|
||||
/// The span capture times are drawn from: twelve years.
|
||||
///
|
||||
/// Long enough that the timeline query has real structure to bucket — at
|
||||
/// monthly granularity that is ~144 buckets, which is the shape the scrubber
|
||||
/// actually draws — and not so long that a year holds too few frames to look
|
||||
/// like a library.
|
||||
const SPAN: i64 = 12 * 365 * 86_400;
|
||||
|
||||
/// Bodies and lenses, for the columns the camera and lens filters read.
|
||||
const CAMERAS: [&str; 6] = [
|
||||
"Canon EOS R5",
|
||||
"Nikon Z 7II",
|
||||
"Sony ILCE-7RM5",
|
||||
"Fujifilm X-T5",
|
||||
"Panasonic DC-S5M2",
|
||||
"OM SYSTEM OM-1",
|
||||
];
|
||||
|
||||
const LENSES: [&str; 6] = [
|
||||
"RF24-70mm F2.8 L IS USM",
|
||||
"NIKKOR Z 50mm f/1.8 S",
|
||||
"FE 85mm F1.4 GM",
|
||||
"XF16-55mmF2.8 R LM WR",
|
||||
"LUMIX S 20-60mm F3.5-5.6",
|
||||
"M.Zuiko Digital ED 12-40mm F2.8",
|
||||
];
|
||||
|
||||
/// What a fixture directory was built from.
|
||||
///
|
||||
/// Written beside the catalog and compared on every run. A mismatch rebuilds:
|
||||
/// silently reusing a fixture built from a different seed, a different row
|
||||
/// count or an older schema would compare two numbers that describe two
|
||||
/// different workloads, which is worse than having no number at all.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
pub struct Stamp {
|
||||
/// Bumped by hand whenever anything in this file changes what gets built.
|
||||
/// The seed cannot carry that: the same seed through different generation
|
||||
/// code produces a different library.
|
||||
pub generator: u32,
|
||||
pub seed: u64,
|
||||
pub images: usize,
|
||||
pub sources: usize,
|
||||
pub folders: usize,
|
||||
pub preview_width: u32,
|
||||
pub preview_height: u32,
|
||||
/// `dr-catalog`'s schema version at build time.
|
||||
pub schema_version: i64,
|
||||
}
|
||||
|
||||
/// Bump on any change to what [`build`] writes.
|
||||
const GENERATOR: u32 = 1;
|
||||
|
||||
/// A built fixture on disk.
|
||||
pub struct Fixture {
|
||||
pub dir: PathBuf,
|
||||
pub catalog: PathBuf,
|
||||
pub thumbs: PathBuf,
|
||||
pub sources: Vec<PathBuf>,
|
||||
pub stamp: Stamp,
|
||||
/// The catalog file's size, reported because it is the thing an open has
|
||||
/// to read and because it is the honest denominator for "is 2 s a lot".
|
||||
pub catalog_bytes: u64,
|
||||
}
|
||||
|
||||
/// Where a fixture lives by default.
|
||||
///
|
||||
/// The temporary directory rather than `target/`, for two reasons. It survives
|
||||
/// `cargo clean`, so a fixture is built once per machine rather than once per
|
||||
/// clean; and it is not inside anything CI caches, so a 14 MB catalog is not
|
||||
/// uploaded and downloaded on every push to save the two seconds it takes to
|
||||
/// generate. `DR_BENCH_DIR` overrides it.
|
||||
pub fn default_dir() -> PathBuf {
|
||||
match std::env::var_os("DR_BENCH_DIR") {
|
||||
Some(dir) => PathBuf::from(dir),
|
||||
None => std::env::temp_dir().join("darkroom-bench"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Build the fixture under `dir`, or confirm the one already there.
|
||||
///
|
||||
/// Returns whether it had to be built, so the caller can say so: a run that
|
||||
/// includes fixture generation has a warm page cache for the catalog file it
|
||||
/// is about to open, and a reader comparing two numbers deserves to know which
|
||||
/// of them was measured that way.
|
||||
pub fn build(dir: &Path, seed: u64, images: usize) -> Result<(Fixture, bool)> {
|
||||
let stamp = Stamp {
|
||||
generator: GENERATOR,
|
||||
seed,
|
||||
images,
|
||||
sources: SOURCE_POOL,
|
||||
folders: FOLDERS,
|
||||
preview_width: PREVIEW.0,
|
||||
preview_height: PREVIEW.1,
|
||||
schema_version: dr_catalog::schema::SCHEMA_VERSION,
|
||||
};
|
||||
|
||||
let catalog = dir.join("catalog.sqlite");
|
||||
let stamp_path = dir.join("stamp.json");
|
||||
let sources: Vec<PathBuf> = (0..SOURCE_POOL)
|
||||
.map(|i| dir.join("sources").join(format!("preview-{i:02}.jpg")))
|
||||
.collect();
|
||||
|
||||
let usable = matches_stamp(&stamp_path, &stamp)
|
||||
&& catalog.is_file()
|
||||
&& sources.iter().all(|p| p.is_file());
|
||||
|
||||
if !usable {
|
||||
std::fs::create_dir_all(dir.join("sources"))
|
||||
.with_context(|| format!("creating the fixture directory {}", dir.display()))?;
|
||||
// The stamp goes last. A build interrupted halfway leaves no stamp, so
|
||||
// the next run rebuilds rather than measuring a truncated catalog.
|
||||
let _ = std::fs::remove_file(&stamp_path);
|
||||
write_sources(&sources, seed)?;
|
||||
write_catalog(&catalog, seed, images)?;
|
||||
std::fs::write(&stamp_path, serde_json::to_vec_pretty(&stamp)?)
|
||||
.with_context(|| format!("writing {}", stamp_path.display()))?;
|
||||
}
|
||||
|
||||
let catalog_bytes = std::fs::metadata(&catalog)
|
||||
.with_context(|| format!("stat {}", catalog.display()))?
|
||||
.len();
|
||||
|
||||
Ok((
|
||||
Fixture {
|
||||
dir: dir.to_path_buf(),
|
||||
catalog,
|
||||
thumbs: dir.join("thumbs"),
|
||||
sources,
|
||||
stamp,
|
||||
catalog_bytes,
|
||||
},
|
||||
!usable,
|
||||
))
|
||||
}
|
||||
|
||||
fn matches_stamp(path: &Path, want: &Stamp) -> bool {
|
||||
let Ok(text) = std::fs::read_to_string(path) else {
|
||||
return false;
|
||||
};
|
||||
matches!(serde_json::from_str::<Stamp>(&text), Ok(have) if have == *want)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The file pool
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Write the pool of JPEGs the pixel half decodes.
|
||||
///
|
||||
/// Encoded through [`dr_thumbs::encode_rgba`] rather than a second encoder
|
||||
/// call of this crate's own. That is the quality the store already uses (82),
|
||||
/// which is a little below what a camera writes its previews at, and it is one
|
||||
/// fewer place for an encoder setting to drift. Stated because it is visible
|
||||
/// in the result: a slightly softer source decodes marginally faster than a
|
||||
/// camera's own preview would.
|
||||
fn write_sources(paths: &[PathBuf], seed: u64) -> Result<()> {
|
||||
let (w, h) = PREVIEW;
|
||||
for (i, path) in paths.iter().enumerate() {
|
||||
let rgba = plausible_frame(w, h, seed ^ (i as u64));
|
||||
let jpeg = dr_thumbs::encode_rgba(w, h, &rgba)
|
||||
.map_err(|e| anyhow::anyhow!("encoding the fixture source {}: {e}", path.display()))?;
|
||||
std::fs::write(path, jpeg).with_context(|| format!("writing {}", path.display()))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// RGBA with detail at every scale: a coarse gradient plus a fine dither.
|
||||
///
|
||||
/// See this module's header for why neither a flat frame nor pure noise would
|
||||
/// do. `salt` moves the gradient and the dither together so the twelve files
|
||||
/// differ from one another rather than being twelve copies with different
|
||||
/// names — a JPEG encoder that saw the same image twelve times would have the
|
||||
/// same cache behaviour every time, which a library does not.
|
||||
pub fn plausible_frame(w: u32, h: u32, salt: u64) -> Vec<u8> {
|
||||
let mut rgba = vec![0u8; (w as usize) * (h as usize) * 4];
|
||||
let bias = (salt % 97) as u32;
|
||||
for y in 0..h as usize {
|
||||
let row = y * (w as usize) * 4;
|
||||
for x in 0..w as usize {
|
||||
// A cheap integer hash, so neighbouring pixels differ and the
|
||||
// encoder has real high-frequency content to spend bits on.
|
||||
let n = (x.wrapping_mul(2_654_435_761) ^ y.wrapping_mul(1_640_531_527)) >> 13;
|
||||
let dither = (n & 0x1f) as u32;
|
||||
let gx = (x * 200 / (w as usize).max(1)) as u32;
|
||||
let gy = (y * 55 / (h as usize).max(1)) as u32;
|
||||
let px = &mut rgba[row + x * 4..row + x * 4 + 4];
|
||||
px[0] = (30 + bias + gx + dither).min(255) as u8;
|
||||
px[1] = (40 + gy + dither).min(255) as u8;
|
||||
px[2] = (60 + gx / 2 + gy + dither).min(255) as u8;
|
||||
px[3] = 255;
|
||||
}
|
||||
}
|
||||
rgba
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The catalog
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Write a catalog holding `images` rows, plus a default version for each.
|
||||
///
|
||||
/// The default versions are not decoration. `Catalog::open` backfills them for
|
||||
/// any image that lacks one (see `schema::backfill`), so a fixture without
|
||||
/// them would charge every measured open for fifty thousand inserts once and
|
||||
/// nothing thereafter — a first number that bore no relation to the second,
|
||||
/// and a benchmark whose result depended on whether it had been run before.
|
||||
fn write_catalog(path: &Path, seed: u64, images: usize) -> Result<()> {
|
||||
for suffix in ["", "-wal", "-shm"] {
|
||||
let mut p = path.as_os_str().to_os_string();
|
||||
p.push(suffix);
|
||||
let _ = std::fs::remove_file(PathBuf::from(p));
|
||||
}
|
||||
|
||||
let catalog = Catalog::open(path)
|
||||
.map_err(|e| anyhow::anyhow!("creating the fixture catalog at {}: {e}", path.display()))?;
|
||||
let conn = catalog.connection();
|
||||
let mut rng = Rng::new(seed);
|
||||
|
||||
conn.execute_batch("BEGIN")?;
|
||||
|
||||
conn.execute(
|
||||
"INSERT INTO roots(id, kind, label, last_seen, scan_generation)
|
||||
VALUES (1, 'local', '/library', ?1, 1)",
|
||||
rusqlite::params![EPOCH],
|
||||
)?;
|
||||
|
||||
{
|
||||
let mut folder = conn.prepare(
|
||||
"INSERT INTO folders(id, root_id, parent_id, path, mtime, entry_count,
|
||||
scanned_generation)
|
||||
VALUES (?1, 1, NULL, ?2, ?3, ?4, 1)",
|
||||
)?;
|
||||
for f in 0..FOLDERS {
|
||||
let id = f as i64 + 1;
|
||||
let folder_path = format!("/library/{:04}/{:02}", 2016 + f / 12, f % 12 + 1);
|
||||
let mtime = EPOCH + f as i64 * 86_400;
|
||||
let entries = (images / FOLDERS.max(1)) as i64;
|
||||
folder.execute(rusqlite::params![id, folder_path, mtime, entries])?;
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
let mut image = conn.prepare(
|
||||
"INSERT INTO images(id, root_id, folder_id, source_ref, format, w, h,
|
||||
captured_at, captured_offset, camera, lens, iso,
|
||||
aperture, shutter, availability, file_size,
|
||||
file_mtime, metadata_state, added_at)
|
||||
VALUES (?1, 1, ?2, ?3, 'CR3', ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11, ?12,
|
||||
?13, ?14, ?15, ?16, ?17)",
|
||||
)?;
|
||||
let mut version = conn.prepare(
|
||||
"INSERT INTO versions(id, image_id, uuid, name, is_default, rating,
|
||||
label, flag)
|
||||
VALUES (?1, ?1, ?2, 'Original', 1, ?3, ?4, ?5)",
|
||||
)?;
|
||||
|
||||
// Every value is bound to a local before it reaches `params!`. Not
|
||||
// style: the macro takes a reference to each argument, and an
|
||||
// expression like `TABLE[rng.below(n) as usize]` inside it borrows an
|
||||
// element of a temporary array while `rng` is also being borrowed
|
||||
// mutably. Locals make the evaluation order and the lifetimes obvious.
|
||||
const OFFSETS: [i64; 5] = [0, 60, 120, -300, 540];
|
||||
const ISOS: [i64; 7] = [100, 200, 400, 800, 1600, 3200, 6400];
|
||||
const APERTURES: [f64; 6] = [1.4, 1.8, 2.8, 4.0, 5.6, 8.0];
|
||||
const SHUTTERS: [f64; 6] = [0.004, 0.008, 0.0167, 0.005, 0.002, 0.5];
|
||||
// Mostly metadata-only, as a large library on a laptop is: some
|
||||
// previewed, a few with the original present.
|
||||
const AVAILABILITY: [i64; 6] = [0, 0, 0, 1, 1, 2];
|
||||
|
||||
for i in 0..images {
|
||||
let id = i as i64 + 1;
|
||||
let bucket = i % FOLDERS;
|
||||
let folder = bucket as i64 + 1;
|
||||
let source_ref = format!(
|
||||
"/library/{:04}/{:02}/IMG_{id:05}.CR3",
|
||||
2016 + bucket / 12,
|
||||
bucket % 12 + 1
|
||||
);
|
||||
let captured = EPOCH + rng.below(SPAN as u64) as i64;
|
||||
// A quarter of the library shot in portrait, which is what makes
|
||||
// the thumbnail path's orientation permutation a real cost rather
|
||||
// than a branch that is never taken.
|
||||
let (w, h) = if i % 4 == 3 {
|
||||
(4000i64, 6000i64)
|
||||
} else {
|
||||
(6000i64, 4000i64)
|
||||
};
|
||||
// Two per cent still awaiting full EXIF — a library is never
|
||||
// entirely finished being read, and the grid has to render that
|
||||
// state (`metadata_state` 1).
|
||||
let state: i64 = if rng.below(50) == 0 { 1 } else { 2 };
|
||||
let camera = CAMERAS[rng.below(CAMERAS.len() as u64) as usize];
|
||||
let lens = LENSES[rng.below(LENSES.len() as u64) as usize];
|
||||
// Minutes east of UTC: a library shot in a handful of places.
|
||||
let offset = OFFSETS[rng.below(OFFSETS.len() as u64) as usize];
|
||||
let iso = ISOS[rng.below(ISOS.len() as u64) as usize];
|
||||
let aperture = APERTURES[rng.below(APERTURES.len() as u64) as usize];
|
||||
let shutter = SHUTTERS[rng.below(SHUTTERS.len() as u64) as usize];
|
||||
let availability = AVAILABILITY[rng.below(AVAILABILITY.len() as u64) as usize];
|
||||
let file_size = 20_000_000i64 + rng.below(30_000_000) as i64;
|
||||
let file_mtime = captured + 60;
|
||||
let added_at = captured + 3600;
|
||||
|
||||
image.execute(rusqlite::params![
|
||||
id,
|
||||
folder,
|
||||
source_ref,
|
||||
w,
|
||||
h,
|
||||
captured,
|
||||
offset,
|
||||
camera,
|
||||
lens,
|
||||
iso,
|
||||
aperture,
|
||||
shutter,
|
||||
availability,
|
||||
file_size,
|
||||
file_mtime,
|
||||
state,
|
||||
added_at,
|
||||
])?;
|
||||
|
||||
// Ratings skewed the way a culled library is: most unrated, a few
|
||||
// picks, fewer still at five stars.
|
||||
let rating: i64 = match rng.below(100) {
|
||||
0..=69 => 0,
|
||||
70..=84 => 1,
|
||||
85..=93 => 2,
|
||||
94..=97 => 3,
|
||||
98 => 4,
|
||||
_ => 5,
|
||||
};
|
||||
let flag: i64 = match rng.below(100) {
|
||||
0..=79 => 0,
|
||||
80..=94 => 1,
|
||||
_ => 2,
|
||||
};
|
||||
let label: Option<i64> = match rng.below(100) {
|
||||
0..=89 => None,
|
||||
n => Some((n % 5) as i64 + 1),
|
||||
};
|
||||
let a = rng.next_u64();
|
||||
let b = rng.next_u64();
|
||||
let uuid = format!("{a:016x}{b:016x}");
|
||||
version.execute(rusqlite::params![id, uuid, rating, label, flag])?;
|
||||
}
|
||||
}
|
||||
|
||||
conn.execute_batch("COMMIT")?;
|
||||
|
||||
// Deliberately no `ANALYZE`. The application never runs one, so a fixture
|
||||
// that did would be measuring a query plan no user's catalog gets — and a
|
||||
// plan chosen from statistics is exactly the sort of thing that would make
|
||||
// the benchmark faster than the product.
|
||||
|
||||
// Dropping the connection checkpoints the WAL, so the file the next open
|
||||
// reads is the whole catalog rather than a stub plus a journal.
|
||||
drop(catalog);
|
||||
Ok(())
|
||||
}
|
||||
@@ -0,0 +1,540 @@
|
||||
//! The benchmark suite `docs/requirements.md` §8 has been promising.
|
||||
//!
|
||||
//! §8 says performance is verified by *"an automated benchmark suite against a
|
||||
//! synthetic 50k catalog, run per-commit … A regression beyond stated
|
||||
//! tolerance fails the build."* Until this crate there was none: no `benches/`,
|
||||
//! no `[[bench]]`, no criterion, no fixture. Ten performance requirements could
|
||||
//! therefore be neither passed nor failed, and five of them carried a
|
||||
//! requirement tag anyway.
|
||||
//!
|
||||
//! ```sh
|
||||
//! cargo run --release -p dr-bench -- check # measure and gate
|
||||
//! cargo run --release -p dr-bench -- check --reference # …on the reference desktop
|
||||
//! cargo run --release -p dr-bench -- record --reference # rewrite the baseline
|
||||
//! ```
|
||||
//!
|
||||
//! **Release, always.** The workspace builds its own crates at `opt-level = 0`
|
||||
//! in dev (see the root `Cargo.toml`), and every number here is dominated by
|
||||
//! this workspace's own code — the JPEG decode, the box filter, the resample,
|
||||
//! the sharpen. A debug run measures rustc's shadow, exactly as
|
||||
//! `core/dr-gpu/examples/frame_budget.rs` warns for the same reason. The
|
||||
//! harness says so at the top of every report rather than trusting anyone to
|
||||
//! remember.
|
||||
//!
|
||||
//! # What this covers, and what it deliberately does not
|
||||
//!
|
||||
//! Covered, with a gate:
|
||||
//!
|
||||
//! - **NFR-P1** and R2's catalog clause — opening a 50k catalog and painting
|
||||
//! the first grid ([`catalog_open`]).
|
||||
//! - **NFR-P3** — thumbnail throughput on the embedded preview path
|
||||
//! ([`thumbnails`]).
|
||||
//!
|
||||
//! Measured, reported, and honestly *not* tagged, because only part of the
|
||||
//! requirement is in reach without a GPU or a running UI:
|
||||
//!
|
||||
//! - **NFR-P7** — the encode half of a 24 MP export ([`exporting`]). A
|
||||
//! one-sided gate: it can fail the requirement, it cannot pass it.
|
||||
//! - **NFR-P8** — the catalog layer's share of idle RSS ([`memory`]), together
|
||||
//! with the answer to the question §4.1 asks about GPU memory.
|
||||
//!
|
||||
//! Out of scope entirely, and left to say so rather than faked: NFR-P2, P4,
|
||||
//! P5, P6, P9, P10, P11, P12, P13, P14 and P15. Every one of them needs a
|
||||
//! frame-timing probe inside a running Slint application, a GPU adapter, or
|
||||
//! both. The GPU half of the suite that *does* exist is
|
||||
//! `core/dr-gpu/tests/frame_budget.rs`, which asserts FR-DSP-3 and skips itself
|
||||
//! where there is no adapter; `.gitea/workflows/benchmark.yml` runs it as its
|
||||
//! own job for exactly that reason.
|
||||
//!
|
||||
//! # Exit codes
|
||||
//!
|
||||
//! `0` everything inside its budget and its tolerance; `1` a gate failed; `2`
|
||||
//! the harness itself could not run. Distinguished because a CI log that says
|
||||
//! "failed" should not leave anyone guessing whether the code got slower or the
|
||||
//! fixture would not build.
|
||||
|
||||
mod baseline;
|
||||
mod catalog_open;
|
||||
mod exporting;
|
||||
mod fixture;
|
||||
mod memory;
|
||||
mod stats;
|
||||
mod thumbnails;
|
||||
|
||||
use std::collections::BTreeMap;
|
||||
use std::path::PathBuf;
|
||||
|
||||
use anyhow::Result;
|
||||
|
||||
use baseline::{Baseline, Judging};
|
||||
|
||||
/// The seed the committed baseline describes.
|
||||
///
|
||||
/// Changing it invalidates every recorded figure, because it changes the
|
||||
/// library being measured. That is why it is a constant here and a field in
|
||||
/// the stamp rather than something a flag quietly varies.
|
||||
const SEED: u64 = 20_260_829;
|
||||
|
||||
/// Rows in the synthetic library. §8 says 50k; this is that.
|
||||
const IMAGES: usize = 50_000;
|
||||
|
||||
/// Thumbnails produced for the throughput row.
|
||||
///
|
||||
/// Twelve hundred rather than fifty thousand. At the target rate the whole
|
||||
/// library is eight minutes of CI, and a rate measured over 1,200 images is
|
||||
/// the same rate — the sweep has no state that changes after the first chunk,
|
||||
/// which the per-image percentile alongside it is there to demonstrate.
|
||||
const THUMBNAILS: usize = 1_200;
|
||||
|
||||
/// Windows fetched for the scroll row.
|
||||
///
|
||||
/// A hundred, so nearest-rank puts the 99th percentile on the second-worst —
|
||||
/// the same reading `core/dr-gpu/examples/frame_budget.rs` takes of a hundred
|
||||
/// frames, and the reason it takes it: one bad one in a hundred is one too
|
||||
/// many, and a single scheduler hiccup on an unrelated process should not
|
||||
/// decide the verdict on its own.
|
||||
const WINDOWS: usize = 100;
|
||||
|
||||
fn main() {
|
||||
env_logger::init();
|
||||
match run() {
|
||||
Ok(true) => {}
|
||||
Ok(false) => std::process::exit(1),
|
||||
Err(e) => {
|
||||
eprintln!("dr-bench: {e:#}");
|
||||
std::process::exit(2);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns whether every gate passed.
|
||||
fn run() -> Result<bool> {
|
||||
let mut args = std::env::args().skip(1);
|
||||
let command = args.next().unwrap_or_else(|| "help".to_string());
|
||||
|
||||
// The probe takes a path and nothing else — see `memory` for why it is a
|
||||
// separate process rather than a function call.
|
||||
if command == "memory-probe" {
|
||||
let path = args
|
||||
.next()
|
||||
.ok_or_else(|| anyhow::anyhow!("memory-probe needs a catalog path"))?;
|
||||
memory::probe(&PathBuf::from(path))?;
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
let mut reference = false;
|
||||
let mut fixture_dir = fixture::default_dir();
|
||||
let mut baseline_path = Baseline::default_path()?;
|
||||
let mut lanes = thumbnails::default_lanes();
|
||||
let mut thumbnail_count = THUMBNAILS;
|
||||
|
||||
while let Some(flag) = args.next() {
|
||||
match flag.as_str() {
|
||||
"--reference" => reference = true,
|
||||
"--fixture" => fixture_dir = PathBuf::from(expect_value(&mut args, "--fixture")?),
|
||||
"--baseline" => baseline_path = PathBuf::from(expect_value(&mut args, "--baseline")?),
|
||||
"--lanes" => lanes = expect_value(&mut args, "--lanes")?.parse()?,
|
||||
"--thumbnails" => thumbnail_count = expect_value(&mut args, "--thumbnails")?.parse()?,
|
||||
other => anyhow::bail!("unknown flag {other}; try `dr-bench help`"),
|
||||
}
|
||||
}
|
||||
|
||||
match command.as_str() {
|
||||
"run" => measure_and_report(
|
||||
&fixture_dir,
|
||||
&baseline_path,
|
||||
reference,
|
||||
lanes,
|
||||
thumbnail_count,
|
||||
Mode::Report,
|
||||
),
|
||||
"check" => measure_and_report(
|
||||
&fixture_dir,
|
||||
&baseline_path,
|
||||
reference,
|
||||
lanes,
|
||||
thumbnail_count,
|
||||
Mode::Gate,
|
||||
),
|
||||
"record" => measure_and_report(
|
||||
&fixture_dir,
|
||||
&baseline_path,
|
||||
reference,
|
||||
lanes,
|
||||
thumbnail_count,
|
||||
Mode::Record,
|
||||
),
|
||||
"help" | "--help" | "-h" => {
|
||||
print_help();
|
||||
Ok(true)
|
||||
}
|
||||
other => anyhow::bail!("unknown command {other}; try `dr-bench help`"),
|
||||
}
|
||||
}
|
||||
|
||||
fn expect_value(args: &mut impl Iterator<Item = String>, flag: &str) -> Result<String> {
|
||||
args.next()
|
||||
.ok_or_else(|| anyhow::anyhow!("{flag} needs a value"))
|
||||
}
|
||||
|
||||
/// What a run does with what it measured.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
enum Mode {
|
||||
/// Print, judge nothing.
|
||||
Report,
|
||||
/// Print and fail the build on a violated budget or a regression.
|
||||
Gate,
|
||||
/// Print and rewrite the committed baseline from what was measured.
|
||||
Record,
|
||||
}
|
||||
|
||||
fn print_help() {
|
||||
println!(
|
||||
"\
|
||||
dr-bench — DarkRoom's performance suite (docs/requirements.md §8)
|
||||
|
||||
run measure and print, judging nothing
|
||||
check measure, print, and exit 1 on a violated budget or a regression
|
||||
record measure and rewrite docs/bench-baseline.json from the result
|
||||
|
||||
Flags:
|
||||
--reference this machine is the reference desktop, so machine-sensitive
|
||||
budgets are asserted rather than reported
|
||||
--fixture <dir> where the synthetic 50k catalog lives (or $DR_BENCH_DIR)
|
||||
--baseline <file> the committed numbers (default docs/bench-baseline.json)
|
||||
--lanes <n> sweep lanes for the thumbnail row (default: CPU threads)
|
||||
--thumbnails <n> images in the thumbnail row (default 1200)
|
||||
|
||||
Build it in release. A debug build measures the compiler, not the pipeline —
|
||||
see the module documentation and core/dr-gpu/examples/frame_budget.rs."
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The run
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn measure_and_report(
|
||||
fixture_dir: &std::path::Path,
|
||||
baseline_path: &std::path::Path,
|
||||
reference: bool,
|
||||
lanes: usize,
|
||||
thumbnail_count: usize,
|
||||
mode: Mode,
|
||||
) -> Result<bool> {
|
||||
let machine = baseline::machine_id();
|
||||
println!("DarkRoom benchmark suite — the half that needs no GPU");
|
||||
println!("machine {machine}");
|
||||
if cfg!(debug_assertions) {
|
||||
println!(
|
||||
"profile DEBUG — every figure below is several times worse than the \
|
||||
product's. Rerun with --release."
|
||||
);
|
||||
} else {
|
||||
println!("profile release");
|
||||
}
|
||||
|
||||
let (fx, built) = fixture::build(fixture_dir, SEED, IMAGES)?;
|
||||
println!(
|
||||
"fixture {} images, {} sources at {}x{}, seed {}, catalog {:.1} MB{}",
|
||||
fx.stamp.images,
|
||||
fx.stamp.sources,
|
||||
fx.stamp.preview_width,
|
||||
fx.stamp.preview_height,
|
||||
fx.stamp.seed,
|
||||
fx.catalog_bytes as f64 / 1e6,
|
||||
if built { " (built just now)" } else { "" }
|
||||
);
|
||||
println!(" {}", fx.dir.display());
|
||||
println!();
|
||||
|
||||
// Memory first, and in its own process. See `memory` for why: measuring
|
||||
// RSS after the thumbnail sweep would report the sweep's high-water mark
|
||||
// wearing the catalog's name.
|
||||
let rss = match memory::in_a_fresh_process(&fx.catalog) {
|
||||
Ok(rss) => Some(rss),
|
||||
Err(e) => {
|
||||
println!("memory unavailable: {e}");
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
let open = catalog_open::measure(&fx.catalog, WINDOWS)?;
|
||||
let thumbs = thumbnails::measure(&fx.sources, &fx.thumbs, thumbnail_count, lanes)?;
|
||||
let exports = exporting::measure()?;
|
||||
|
||||
print_details(&open, &thumbs, &exports, rss);
|
||||
|
||||
let values = collect(&open, &thumbs, &exports, rss);
|
||||
let mut base = Baseline::load(baseline_path)?;
|
||||
|
||||
if mode == Mode::Record {
|
||||
if !reference {
|
||||
println!(
|
||||
"note recording from a machine that has not declared itself the \
|
||||
reference desktop.\n §8 records the reference desktop's numbers; \
|
||||
this overwrites them."
|
||||
);
|
||||
}
|
||||
for (key, value) in &values {
|
||||
if let Some(metric) = base.metrics.get_mut(key) {
|
||||
metric.recorded = Some(*value);
|
||||
}
|
||||
}
|
||||
base.recorded_on = Some(machine);
|
||||
base.recorded_at_unix = Some(baseline::now_unix());
|
||||
base.fixture = Some(fx.stamp.clone());
|
||||
base.save(baseline_path)?;
|
||||
println!("\nrecorded {}", baseline_path.display());
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
let same_machine = base.recorded_on.as_deref() == Some(machine.as_str());
|
||||
let comparable = base.fixture.as_ref().is_some_and(|f| *f == fx.stamp);
|
||||
let cx = Judging {
|
||||
reference,
|
||||
same_machine: same_machine && comparable,
|
||||
tolerance: base.tolerance,
|
||||
};
|
||||
let failures = print_verdict(&base, &values, cx, &machine, comparable);
|
||||
|
||||
if failures.is_empty() {
|
||||
return Ok(true);
|
||||
}
|
||||
println!();
|
||||
for line in &failures {
|
||||
println!("FAIL {line}");
|
||||
}
|
||||
println!();
|
||||
println!(
|
||||
" {} gate(s) failed. docs/benchmarks.md says what each metric measures and\n \
|
||||
docs/bench-baseline.json holds the numbers these used to be.",
|
||||
failures.len()
|
||||
);
|
||||
Ok(mode != Mode::Gate)
|
||||
}
|
||||
|
||||
/// The metric keys, and the measurement each one reads.
|
||||
///
|
||||
/// One place, so the report, the gate and the recorded file cannot disagree
|
||||
/// about what `catalog_open_ms` means.
|
||||
fn collect(
|
||||
open: &catalog_open::Open,
|
||||
thumbs: &thumbnails::Throughput,
|
||||
exports: &[exporting::EncodeRun],
|
||||
rss: Option<memory::Rss>,
|
||||
) -> BTreeMap<String, f64> {
|
||||
let mut v = BTreeMap::new();
|
||||
v.insert("catalog_open_ms".to_string(), open.cold_ms);
|
||||
v.insert("catalog_open_warm_ms".to_string(), open.warm_ms);
|
||||
v.insert("catalog_window_p99_ms".to_string(), open.window_ms.p99);
|
||||
v.insert("catalog_filtered_ms".to_string(), open.filtered_ms);
|
||||
let ips = thumbs.images_per_second;
|
||||
v.insert("thumbnail_throughput_ips".to_string(), ips);
|
||||
v.insert(
|
||||
"thumbnail_per_image_p99_ms".to_string(),
|
||||
thumbs.per_image.p99,
|
||||
);
|
||||
for row in exports {
|
||||
v.insert(row.key.to_string(), row.times.p99);
|
||||
}
|
||||
if let Some(rss) = rss {
|
||||
v.insert("catalog_idle_rss_mb".to_string(), rss.now_mb());
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The report
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn print_details(
|
||||
open: &catalog_open::Open,
|
||||
thumbs: &thumbnails::Throughput,
|
||||
exports: &[exporting::EncodeRun],
|
||||
rss: Option<memory::Rss>,
|
||||
) {
|
||||
println!("Opening the catalog (NFR-P1, and R2's second sentence)");
|
||||
println!(
|
||||
" {:>10.1} ms open, count, first window and timeline — cold, first \
|
||||
connection of the process",
|
||||
open.cold_ms
|
||||
);
|
||||
println!(
|
||||
" {:>10.1} ms the same four calls on a second connection",
|
||||
open.warm_ms
|
||||
);
|
||||
println!(
|
||||
" {:>10.1} ms Catalog::open alone (connect, migrate, backfill)",
|
||||
open.open_only_ms
|
||||
);
|
||||
println!(
|
||||
" {:>10.1} ms count and first window under a rating filter",
|
||||
open.filtered_ms
|
||||
);
|
||||
println!(
|
||||
" {:>10.2} ms one 400-row window at a random offset, p99 of {WINDOWS} \
|
||||
(p50 {:.2}, max {:.2})",
|
||||
open.window_ms.p99, open.window_ms.p50, open.window_ms.max
|
||||
);
|
||||
println!(
|
||||
" {} images, {} monthly timeline buckets",
|
||||
open.images, open.buckets
|
||||
);
|
||||
println!();
|
||||
|
||||
println!("Thumbnails on the embedded preview path (NFR-P3: >= 100 img/s)");
|
||||
println!(
|
||||
" {:>10.1} img/s over {} images on {} lanes, {:.2} s of wall clock",
|
||||
thumbs.images_per_second,
|
||||
thumbs.images,
|
||||
thumbs.lanes,
|
||||
thumbs.wall_ms / 1e3
|
||||
);
|
||||
println!(
|
||||
" {:>10.2} ms per image on its lane, p99 (p50 {:.2}, max {:.2})",
|
||||
thumbs.per_image.p99, thumbs.per_image.p50, thumbs.per_image.max
|
||||
);
|
||||
println!(
|
||||
" {} stored, {} failed",
|
||||
thumbs.stored, thumbs.failed
|
||||
);
|
||||
println!();
|
||||
println!(" The bytes are in memory before the clock starts, so this is CPU");
|
||||
println!(" throughput with the fetch already paid for. That is what the target's");
|
||||
println!(" \"embedded preview path\" names; it is not a claim about a remote library.");
|
||||
println!();
|
||||
|
||||
println!("Exporting 24 MP — the encode half only (NFR-P7 is the whole chain)");
|
||||
println!(
|
||||
" {:>14} {:>11} {:>9} {:>9} {:>9}",
|
||||
"sizing", "output", "p50", "p99", "file"
|
||||
);
|
||||
for row in exports {
|
||||
println!(
|
||||
" {:>14} {:>5}x{:<5} {:>7.1}ms {:>7.1}ms {:>7.1}MB",
|
||||
row.label,
|
||||
row.width,
|
||||
row.height,
|
||||
row.times.p50,
|
||||
row.times.p99,
|
||||
row.bytes as f64 / 1e6
|
||||
);
|
||||
}
|
||||
println!();
|
||||
println!(" No GPU render is in these figures, so they cannot pass NFR-P7 — only fail");
|
||||
println!(" it. See tools/bench/src/exporting.rs for why that is still worth gating.");
|
||||
println!();
|
||||
|
||||
println!("Idle memory with the catalog open (NFR-P8's catalog share)");
|
||||
match rss {
|
||||
Some(rss) => {
|
||||
println!(
|
||||
" {:>10.1} MB resident after opening 50k and scrolling 10k rows",
|
||||
rss.now_mb()
|
||||
);
|
||||
println!(" {:>10.1} MB peak for that process", rss.peak_mb());
|
||||
println!();
|
||||
println!(" RSS is exclusive of device-local GPU memory and this process has no");
|
||||
println!(" toolkit, no adapter and no decode cache — so it is the catalog layer's");
|
||||
println!(" share of NFR-P8's 500 MB, not NFR-P8. tools/bench/src/memory.rs holds");
|
||||
println!(" the answer to the question §4.1 asks, and the change it recommends.");
|
||||
}
|
||||
None => println!(" not measured on this platform"),
|
||||
}
|
||||
println!();
|
||||
}
|
||||
|
||||
/// Print the gate table and return the failures, one line each.
|
||||
fn print_verdict(
|
||||
base: &Baseline,
|
||||
values: &BTreeMap<String, f64>,
|
||||
cx: Judging,
|
||||
machine: &str,
|
||||
comparable_fixture: bool,
|
||||
) -> Vec<String> {
|
||||
println!("Against docs/bench-baseline.json");
|
||||
match (&base.recorded_on, cx.same_machine) {
|
||||
(None, _) => println!(
|
||||
" No baseline has been recorded yet. Budgets are still gated; drift is not.\n \
|
||||
Run `dr-bench record --reference` on the reference desktop and commit the diff."
|
||||
),
|
||||
(Some(on), true) => println!(" Recorded on {on} — drift below is a verdict."),
|
||||
(Some(on), false) if !comparable_fixture => println!(
|
||||
" Recorded on {on} against a different fixture; drift below is information only."
|
||||
),
|
||||
(Some(on), false) => println!(
|
||||
" Recorded on {on}, and this is {machine}. Drift below is information, not a verdict."
|
||||
),
|
||||
}
|
||||
if !cx.reference {
|
||||
println!(
|
||||
" Not the reference desktop (--reference), so machine-sensitive budgets are\n \
|
||||
reported rather than asserted — §8 names the reference desktop, not CI."
|
||||
);
|
||||
}
|
||||
println!();
|
||||
println!(
|
||||
" {:<30} {:>12} {:>13} {:>12} {:>8} verdict",
|
||||
"metric", "measured", "budget", "baseline", "drift"
|
||||
);
|
||||
|
||||
let mut failures = Vec::new();
|
||||
for (key, measured) in values {
|
||||
let Some(metric) = base.metrics.get(key) else {
|
||||
println!(
|
||||
" {key:<30} {measured:>12.2} {:>13} {:>12} {:>8} not in the baseline",
|
||||
"—", "—", "—"
|
||||
);
|
||||
continue;
|
||||
};
|
||||
let j = metric.judge(*measured, cx);
|
||||
let budget = match (metric.budget, metric.direction) {
|
||||
(None, _) => "—".to_string(),
|
||||
(Some(b), baseline::Direction::LowerIsBetter) => format!("< {b:.1}"),
|
||||
(Some(b), baseline::Direction::HigherIsBetter) => format!("> {b:.1}"),
|
||||
};
|
||||
let recorded = match metric.recorded {
|
||||
Some(r) => format!("{r:.2}"),
|
||||
None => "—".to_string(),
|
||||
};
|
||||
let drift = match j.drift {
|
||||
Some(d) => format!("{:+.1}%", d * 100.0),
|
||||
None => "—".to_string(),
|
||||
};
|
||||
let mut verdict = String::new();
|
||||
if j.over_budget {
|
||||
verdict.push_str("OVER BUDGET ");
|
||||
failures.push(format!(
|
||||
"{key} is {measured:.2} {}, past {budget} ({})",
|
||||
metric.unit, metric.requirement
|
||||
));
|
||||
}
|
||||
if j.regressed {
|
||||
verdict.push_str("REGRESSED ");
|
||||
failures.push(format!(
|
||||
"{key} drifted {drift} against the baseline, past the {:.0}% tolerance",
|
||||
cx.tolerance * 100.0
|
||||
));
|
||||
}
|
||||
if verdict.is_empty() {
|
||||
verdict.push_str(if j.budget_deferred {
|
||||
"ok (budget deferred)"
|
||||
} else {
|
||||
"ok"
|
||||
});
|
||||
}
|
||||
println!(" {key:<30} {measured:>12.2} {budget:>13} {recorded:>12} {drift:>8} {verdict}");
|
||||
}
|
||||
|
||||
// A metric in the file that nothing measured is a harness that has drifted
|
||||
// from its own record, and is worth saying out loud rather than leaving as
|
||||
// a row that quietly stopped appearing.
|
||||
for key in base.metrics.keys() {
|
||||
if !values.contains_key(key) {
|
||||
println!(" {key:<30} {:>12} measured nothing this run", "—");
|
||||
}
|
||||
}
|
||||
|
||||
failures
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
//! Idle memory with a 50k catalog open — and the question NFR-P8 leaves open.
|
||||
//!
|
||||
//! # The question §4.1 asks, answered
|
||||
//!
|
||||
//! §4.1 says of NFR-P8: *"must state whether it measures RSS inclusive or
|
||||
//! exclusive of GPU allocations, and whether it holds after SQLite's page cache
|
||||
//! warms on a 50k catalog."* Both halves have an answer, and neither is
|
||||
//! flattering.
|
||||
//!
|
||||
//! **On GPU memory: what this reports is RSS, and RSS is exclusive of
|
||||
//! device-local GPU allocations.** A Vulkan allocation in device-local heap
|
||||
//! never enters the process's address space, so no counter under
|
||||
//! `/proc/self/status` can see it; what *does* land in RSS is the host-visible
|
||||
//! side — staging buffers, mapped upload rings, the read-back `AdjustPass`
|
||||
//! performs on export — and the driver's own resident pages. So "RSS < 500 MB"
|
||||
//! is not one budget, it is two questions wearing one number, and a build that
|
||||
//! kept RSS at 400 MB while holding 3 GB of textures would pass it.
|
||||
//!
|
||||
//! The recommendation this measurement exists to support: **NFR-P8 should be
|
||||
//! restated as two figures** — host RSS exclusive of device-local memory, and
|
||||
//! a separate VRAM ceiling read from the adapter — because the second is the
|
||||
//! one that decides whether the application survives beside a browser on an
|
||||
//! 8 GB card, and nothing in this repository currently measures it.
|
||||
//!
|
||||
//! **On the page cache: warm.** The probe runs the queries before it reads the
|
||||
//! counter, so SQLite's page cache holds the b-tree pages a grid scroll
|
||||
//! touches. That is the right side to err on — a figure taken before the cache
|
||||
//! warms would understate a steady-state library — and it is why the probe
|
||||
//! scrolls rather than opening and stopping.
|
||||
//!
|
||||
//! # Why this is a subprocess
|
||||
//!
|
||||
//! RSS is a high-water-influenced property of a *process*, not of a function.
|
||||
//! Building a 50k fixture allocates hundreds of megabytes; decoding thumbnails
|
||||
//! allocates more; the allocator returns some of it to the OS and keeps the
|
||||
//! rest. Measuring after any of that would report the harness's history rather
|
||||
//! than the catalog's cost. So the probe is a fresh process that opens the
|
||||
//! catalog, does the grid's work, reads its own counters and exits.
|
||||
//!
|
||||
//! # What this cannot certify, said plainly
|
||||
//!
|
||||
//! Not NFR-P8. The requirement is about the *application* at idle — Slint, the
|
||||
//! wgpu device, the font stack, the decode cache and the catalog together — and
|
||||
//! this process contains only the last of those. No requirement tag in this
|
||||
//! crate names NFR-P8, for that reason — and see `exporting.rs` for why that
|
||||
//! sentence avoids spelling the tag out.
|
||||
//!
|
||||
//! What it is, is the catalog layer's share, measured rather than guessed. The
|
||||
//! decision NFR-P8 actually needs — how much of the 500 MB belongs to the
|
||||
//! catalog and how much to everything above it — is a decision somebody has to
|
||||
//! take, and taking it against a recorded number is better than taking it
|
||||
//! against an estimate. That is what this records. Until it is taken, the
|
||||
//! metric carries no budget and gates only against its own baseline.
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use anyhow::Result;
|
||||
use dr_catalog::{Catalog, Granularity, Query};
|
||||
|
||||
/// Rows fetched per window while the probe scrolls. The same 400
|
||||
/// [`crate::catalog_open`] uses, for the same reason.
|
||||
const WINDOW: usize = 400;
|
||||
|
||||
/// Windows the probe pages through before reading the counter.
|
||||
///
|
||||
/// Twenty-five is ten thousand rows: enough that SQLite's page cache holds a
|
||||
/// realistic working set and that any per-window leak would be visible, and
|
||||
/// far short of the whole library, which FR-CAT-4 forbids holding anyway.
|
||||
const WINDOWS: usize = 25;
|
||||
|
||||
/// A fixed clock, for the reason `catalog_open`'s `NOW` gives: nothing here
|
||||
/// should depend on the day it runs.
|
||||
const NOW: i64 = 2_000_000_000;
|
||||
|
||||
/// Resident memory, in kilobytes, as Linux reports it.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct Rss {
|
||||
/// `VmRSS`: resident now.
|
||||
pub now_kb: u64,
|
||||
/// `VmHWM`: the peak this process reached. Reported alongside because a
|
||||
/// process that touched 900 MB and gave it back is not idling at 200 MB in
|
||||
/// any sense a user would recognise — the pages came from somewhere.
|
||||
pub peak_kb: u64,
|
||||
}
|
||||
|
||||
impl Rss {
|
||||
pub fn now_mb(&self) -> f64 {
|
||||
self.now_kb as f64 / 1024.0
|
||||
}
|
||||
|
||||
pub fn peak_mb(&self) -> f64 {
|
||||
self.peak_kb as f64 / 1024.0
|
||||
}
|
||||
}
|
||||
|
||||
/// Read this process's own counters.
|
||||
///
|
||||
/// `None` anywhere without a Linux-shaped `/proc` — including Android, where
|
||||
/// the file exists but a benchmark does not run, and macOS, where it does not.
|
||||
/// Returning `None` rather than zero is deliberate: a memory figure of zero
|
||||
/// would be reported as an excellent result.
|
||||
pub fn of_this_process() -> Option<Rss> {
|
||||
let status = std::fs::read_to_string("/proc/self/status").ok()?;
|
||||
let mut now = None;
|
||||
let mut peak = None;
|
||||
for line in status.lines() {
|
||||
if let Some(rest) = line.strip_prefix("VmRSS:") {
|
||||
now = rest.split_whitespace().next()?.parse::<u64>().ok();
|
||||
} else if let Some(rest) = line.strip_prefix("VmHWM:") {
|
||||
peak = rest.split_whitespace().next()?.parse::<u64>().ok();
|
||||
}
|
||||
}
|
||||
Some(Rss {
|
||||
now_kb: now?,
|
||||
peak_kb: peak?,
|
||||
})
|
||||
}
|
||||
|
||||
/// The probe: open the catalog, do what the grid does, print the counters.
|
||||
///
|
||||
/// Stdout is one line of `key=value` pairs rather than JSON, because the only
|
||||
/// reader is [`in_a_fresh_process`] and a format a human can read in a log is
|
||||
/// worth more here than one a parser prefers.
|
||||
pub fn probe(catalog_path: &Path) -> Result<()> {
|
||||
let catalog = Catalog::open(catalog_path)
|
||||
.map_err(|e| anyhow::anyhow!("opening {} : {e}", catalog_path.display()))?;
|
||||
let q = Query::default();
|
||||
|
||||
let images = catalog.count(&q, NOW)?;
|
||||
let buckets = catalog.timeline(&q, Granularity::Month, NOW)?.len();
|
||||
|
||||
// Scroll, keeping only the window in hand — which is what the grid does,
|
||||
// and what FR-CAT-4 requires it to do. If this ever starts costing memory
|
||||
// proportional to how far the user scrolled, that is the bug this figure
|
||||
// exists to catch.
|
||||
let mut rows = 0usize;
|
||||
let span = images.saturating_sub(WINDOW).max(1);
|
||||
for i in 0..WINDOWS {
|
||||
let start = (i * span) / WINDOWS.max(1);
|
||||
rows = catalog.window(&q, start..start + WINDOW, NOW)?.len();
|
||||
}
|
||||
|
||||
let Some(rss) = of_this_process() else {
|
||||
anyhow::bail!("no /proc/self/status on this platform; RSS cannot be read");
|
||||
};
|
||||
println!(
|
||||
"rss_kb={} peak_kb={} images={images} buckets={buckets} last_window={rows}",
|
||||
rss.now_kb, rss.peak_kb
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Run [`probe`] in a fresh copy of this executable and read back its counters.
|
||||
pub fn in_a_fresh_process(catalog_path: &Path) -> Result<Rss> {
|
||||
let exe = std::env::current_exe()?;
|
||||
let output = std::process::Command::new(&exe)
|
||||
.arg("memory-probe")
|
||||
.arg(catalog_path)
|
||||
.output()?;
|
||||
|
||||
if !output.status.success() {
|
||||
anyhow::bail!(
|
||||
"the memory probe exited with {}: {}",
|
||||
output.status,
|
||||
String::from_utf8_lossy(&output.stderr).trim()
|
||||
);
|
||||
}
|
||||
|
||||
let text = String::from_utf8_lossy(&output.stdout);
|
||||
let field = |key: &str| -> Option<u64> {
|
||||
text.split_whitespace()
|
||||
.find_map(|pair| pair.strip_prefix(key))
|
||||
.and_then(|v| v.parse::<u64>().ok())
|
||||
};
|
||||
let (Some(now_kb), Some(peak_kb)) = (field("rss_kb="), field("peak_kb=")) else {
|
||||
anyhow::bail!(
|
||||
"the memory probe printed something unreadable: {}",
|
||||
text.trim()
|
||||
);
|
||||
};
|
||||
Ok(Rss { now_kb, peak_kb })
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
//! Ranking a set of samples, the way `dr-gpu`'s frame budget ranks them.
|
||||
//!
|
||||
//! Copied in spirit rather than shared, because the two live in different
|
||||
//! dependency worlds — `core/dr-gpu/examples/frame_budget.rs` is an example
|
||||
//! inside a crate this one deliberately does not depend on (see `Cargo.toml`).
|
||||
//! The arithmetic is identical on purpose: two percentile definitions in one
|
||||
//! repository is how two benchmarks come to disagree about the same machine.
|
||||
//!
|
||||
//! # Nearest-rank, not an interpolating definition
|
||||
//!
|
||||
//! The samples *are* the population. There is no distribution being estimated
|
||||
//! here, only a set of catalog opens or thumbnail encodes that either happened
|
||||
//! inside the target or did not. At 100 samples the 99th percentile is the
|
||||
//! second-worst, which is the honest reading of "one bad one in a hundred is
|
||||
//! one too many" without letting a single scheduler hiccup on an unrelated
|
||||
//! process decide the verdict.
|
||||
|
||||
use std::time::Duration;
|
||||
|
||||
/// Nearest-rank percentiles over a set of samples, in the caller's unit.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct Percentiles {
|
||||
pub p50: f64,
|
||||
pub p99: f64,
|
||||
pub max: f64,
|
||||
}
|
||||
|
||||
impl Percentiles {
|
||||
/// Rank `samples`. Panics on an empty set, which is a harness bug rather
|
||||
/// than a measurement: a row with nothing in it must not print a zero that
|
||||
/// reads like a very fast result.
|
||||
pub fn of(mut samples: Vec<f64>) -> Self {
|
||||
assert!(
|
||||
!samples.is_empty(),
|
||||
"percentiles of an empty sample set — the measurement produced nothing"
|
||||
);
|
||||
samples.sort_by(f64::total_cmp);
|
||||
let rank = |p: f64| {
|
||||
let n = samples.len();
|
||||
let i = ((p * n as f64).ceil() as usize).clamp(1, n) - 1;
|
||||
samples[i]
|
||||
};
|
||||
Percentiles {
|
||||
p50: rank(0.50),
|
||||
p99: rank(0.99),
|
||||
max: samples[samples.len() - 1],
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A duration in milliseconds, which is the unit every timing here is stated
|
||||
/// in. One spelling, so no row is accidentally in seconds.
|
||||
pub fn ms(d: Duration) -> f64 {
|
||||
d.as_secs_f64() * 1e3
|
||||
}
|
||||
|
||||
/// A deterministic generator, so a fixture is reproducible from its seed.
|
||||
///
|
||||
/// SplitMix64. Chosen because it is eight lines, has no dependency, and passes
|
||||
/// the only test that matters here — that the same seed produces the same
|
||||
/// catalog on the reference desktop and on the CI runner, so a number measured
|
||||
/// in one place describes the same workload as a number measured in the other.
|
||||
/// Nothing cryptographic depends on it.
|
||||
pub struct Rng(u64);
|
||||
|
||||
impl Rng {
|
||||
pub fn new(seed: u64) -> Self {
|
||||
Rng(seed)
|
||||
}
|
||||
|
||||
pub fn next_u64(&mut self) -> u64 {
|
||||
self.0 = self.0.wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||
let mut z = self.0;
|
||||
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||
z ^ (z >> 31)
|
||||
}
|
||||
|
||||
/// A value in `0..n`. Modulo-biased, which does not matter for a fixture:
|
||||
/// nothing here is a statistical test, only a spread of plausible values.
|
||||
pub fn below(&mut self, n: u64) -> u64 {
|
||||
self.next_u64() % n.max(1)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn the_ninety_ninth_of_a_hundred_is_the_second_worst() {
|
||||
// The property the whole suite's verdict rests on. Off by one here and
|
||||
// every threshold is judged against the worst sample instead.
|
||||
let samples: Vec<f64> = (1..=100).map(|n| n as f64).collect();
|
||||
let p = Percentiles::of(samples);
|
||||
assert_eq!(p.p99, 99.0);
|
||||
assert_eq!(p.max, 100.0);
|
||||
assert_eq!(p.p50, 50.0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_single_sample_ranks_as_itself() {
|
||||
// A row measured once — a cold catalog open — must not divide by zero
|
||||
// or index off the end.
|
||||
let p = Percentiles::of(vec![7.5]);
|
||||
assert_eq!((p.p50, p.p99, p.max), (7.5, 7.5, 7.5));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_same_seed_gives_the_same_sequence() {
|
||||
// Reproducibility from a seed is what makes a committed baseline mean
|
||||
// anything: two runs must describe the same catalog.
|
||||
let mut a = Rng::new(20_260_829);
|
||||
let mut b = Rng::new(20_260_829);
|
||||
let mut c = Rng::new(20_260_830);
|
||||
let first: Vec<u64> = (0..8).map(|_| a.next_u64()).collect();
|
||||
let same: Vec<u64> = (0..8).map(|_| b.next_u64()).collect();
|
||||
let other: Vec<u64> = (0..8).map(|_| c.next_u64()).collect();
|
||||
assert_eq!(first, same);
|
||||
assert_ne!(first, other);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,272 @@
|
||||
//! TRACES: NFR-P3
|
||||
//! Thumbnail throughput on the embedded preview path.
|
||||
//!
|
||||
//! NFR-P3 asks for **≥ 100 images per second** on the reference desktop
|
||||
//! through the embedded preview path, and ≥ 25 on a mid-range Android device.
|
||||
//! Nothing had ever counted.
|
||||
//!
|
||||
//! # What is timed, and why it is not the sweep itself
|
||||
//!
|
||||
//! `ui/dr-ui/src/library.rs`'s [`spawn_thumbnail_sweep`] is the whole-library
|
||||
//! pass, and it is the machinery this mirrors: a chunk at a time, lanes owning
|
||||
//! disjoint slices, every lane decoding and encoding on its own, and the
|
||||
//! single thread that owns the store writing the finished chunk. That shape is
|
||||
//! reproduced here because it is the shape that decides the number — where the
|
||||
//! parallelism is, and where the one lock is.
|
||||
//!
|
||||
//! It is *mirrored* rather than called, for a reason worth stating plainly:
|
||||
//! that function takes a `RemoteBackend` and spends most of its wall clock in
|
||||
//! WebDAV round trips. Calling it from a benchmark would need a Nextcloud
|
||||
//! server, and what it would then measure is somebody's network. The per-image
|
||||
//! work is identical either way — `dr_decode::decode_jpeg`,
|
||||
//! `Preview::downscale_to`, `Preview::apply_orientation`,
|
||||
//! `dr_thumbs::encode_rgba`, `ThumbStore::put` — and that work is what a target
|
||||
//! written in images per second is about.
|
||||
//!
|
||||
//! So: **the bytes are already in memory when the clock starts.** This is CPU
|
||||
//! throughput for the preview path with the fetch paid for, which is what a
|
||||
//! local library gives you and what the target's "embedded preview path"
|
||||
//! names. It is not a claim about a remote library, whose ceiling is latency
|
||||
//! and which [`spawn_thumbnail_sweep`] exists to hide rather than to beat.
|
||||
//!
|
||||
//! # The plain-JPEG branch, deliberately
|
||||
//!
|
||||
//! `fetch_preview` has two arms: a plain JPEG is its own preview, and anything
|
||||
//! else is located inside the container first. The fixture's files take the
|
||||
//! first arm, so `locate_preview` is not in the measured span. That is honest
|
||||
//! for two reasons — a library of camera JPEGs is a real library and takes
|
||||
//! exactly this path, and for a RAW the located preview *is* a JPEG of about
|
||||
//! this size, so what changes is a header walk of a few microseconds against a
|
||||
//! decode of several milliseconds. What is genuinely not measured is the
|
||||
//! container parse of an exotic format, and nothing here pretends otherwise.
|
||||
//!
|
||||
//! [`spawn_thumbnail_sweep`]: ../../dr_ui/library/fn.spawn_thumbnail_sweep.html
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::time::Instant;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use dr_thumbs::{ThumbSize, ThumbStore, Thumbnail};
|
||||
use dr_types::Orientation;
|
||||
|
||||
use crate::stats::{ms, Percentiles};
|
||||
|
||||
/// Images per chunk handed back to the thread that owns the store.
|
||||
///
|
||||
/// 96, which is `SWEEP_CHUNK` in `library.rs`. Copied rather than chosen: the
|
||||
/// point of this row is to describe the sweep's behaviour, and a different
|
||||
/// chunk size would move the ratio of lane work to store work.
|
||||
const CHUNK: usize = 96;
|
||||
|
||||
/// The size class the whole-library pass fills.
|
||||
///
|
||||
/// Grid only, which is `SWEEP_THUMB_SIZE`. The large class is four times the
|
||||
/// transfer for a detail only a zoomed cell asks for, so the sweep does not
|
||||
/// produce it and neither does this.
|
||||
const SIZE: ThumbSize = ThumbSize::Grid;
|
||||
|
||||
/// What a measured sweep produced.
|
||||
pub struct Throughput {
|
||||
pub images: usize,
|
||||
pub lanes: usize,
|
||||
pub wall_ms: f64,
|
||||
pub images_per_second: f64,
|
||||
/// Per image, on the lane that produced it: decode, downscale, orient,
|
||||
/// encode. Not the store write, which happens elsewhere by design.
|
||||
pub per_image: Percentiles,
|
||||
/// Thumbnails that reached the store.
|
||||
pub stored: usize,
|
||||
/// Images whose preview would not decode. Any non-zero is a broken
|
||||
/// fixture, not a slow one.
|
||||
pub failed: usize,
|
||||
}
|
||||
|
||||
/// One lane's output: what it made, what each one cost, and what it dropped.
|
||||
struct Lane {
|
||||
made: Vec<(u64, Thumbnail)>,
|
||||
times: Vec<f64>,
|
||||
failed: usize,
|
||||
}
|
||||
|
||||
/// Sweep `images` thumbnails from `sources`, `lanes` at a time.
|
||||
///
|
||||
/// `store_dir` is emptied first. Re-storing an id that is already present is
|
||||
/// an `UPDATE` in place rather than an insert (see `ThumbStore::put`), and a
|
||||
/// run that measured updates would not be measuring the pass this describes —
|
||||
/// the sweep's work list is by construction what the store does *not* have.
|
||||
pub fn measure(
|
||||
sources: &[PathBuf],
|
||||
store_dir: &Path,
|
||||
images: usize,
|
||||
lanes: usize,
|
||||
) -> Result<Throughput> {
|
||||
anyhow::ensure!(!sources.is_empty(), "no fixture sources to sweep");
|
||||
anyhow::ensure!(lanes > 0, "a sweep needs at least one lane");
|
||||
|
||||
let bytes: Vec<Vec<u8>> = sources
|
||||
.iter()
|
||||
.map(|p| std::fs::read(p).with_context(|| format!("reading {}", p.display())))
|
||||
.collect::<Result<_>>()?;
|
||||
|
||||
let _ = std::fs::remove_dir_all(store_dir);
|
||||
let mut store = ThumbStore::open(store_dir)
|
||||
.map_err(|e| anyhow::anyhow!("opening the bench thumbnail store: {e}"))?;
|
||||
|
||||
// Warm-up: every source decoded once, discarded. The first decode of a
|
||||
// file faults in its Huffman tables and grows the allocator's arenas to
|
||||
// the size a 1620x1080 RGBA buffer needs, and neither recurs across a
|
||||
// sweep of thousands.
|
||||
let warm_failures = Lane::run(&bytes, (0..bytes.len()).collect()).failed;
|
||||
anyhow::ensure!(
|
||||
warm_failures == 0,
|
||||
"{warm_failures} of the {} fixture sources would not decode",
|
||||
bytes.len()
|
||||
);
|
||||
|
||||
let mut samples: Vec<f64> = Vec::with_capacity(images);
|
||||
let mut stored = 0usize;
|
||||
let mut failed = 0usize;
|
||||
|
||||
let started = Instant::now();
|
||||
for chunk_start in (0..images).step_by(CHUNK) {
|
||||
let chunk_end = (chunk_start + CHUNK).min(images);
|
||||
// Each lane takes every `lanes`-th image of the chunk, which is how
|
||||
// `spawn_thumbnail_sweep` splits one: disjoint slices, nothing shared,
|
||||
// no lock.
|
||||
let work: Vec<Vec<usize>> = (0..lanes)
|
||||
.map(|lane| (chunk_start..chunk_end).skip(lane).step_by(lanes).collect())
|
||||
.collect();
|
||||
|
||||
let source_slice: &[Vec<u8>] = &bytes;
|
||||
let produced: Vec<Lane> = std::thread::scope(|scope| {
|
||||
let handles: Vec<_> = work
|
||||
.into_iter()
|
||||
.map(|lane| scope.spawn(move || Lane::run(source_slice, lane)))
|
||||
.collect();
|
||||
handles
|
||||
.into_iter()
|
||||
.map(|h| h.join().expect("a sweep lane panicked"))
|
||||
.collect()
|
||||
});
|
||||
|
||||
// The store is `&mut` and single-writer, so the chunk is written here
|
||||
// and not on the lanes. This is the sweep's own discipline and it is
|
||||
// part of the number: if the store were the bottleneck, no amount of
|
||||
// lane parallelism would help and the row would say so.
|
||||
for lane in produced {
|
||||
samples.extend(lane.times);
|
||||
failed += lane.failed;
|
||||
for (file_id, thumb) in lane.made {
|
||||
match store.put(file_id, SIZE, &thumb) {
|
||||
Ok(_) => stored += 1,
|
||||
Err(e) => {
|
||||
log::warn!("storing bench thumbnail {file_id}: {e}");
|
||||
failed += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
let wall = started.elapsed();
|
||||
|
||||
anyhow::ensure!(
|
||||
failed == 0,
|
||||
"{failed} of {images} thumbnails failed; the throughput below would be \
|
||||
measuring how fast this gives up"
|
||||
);
|
||||
|
||||
let wall_ms = ms(wall);
|
||||
Ok(Throughput {
|
||||
images,
|
||||
lanes,
|
||||
wall_ms,
|
||||
images_per_second: images as f64 / wall.as_secs_f64().max(f64::MIN_POSITIVE),
|
||||
per_image: Percentiles::of(samples),
|
||||
stored,
|
||||
failed,
|
||||
})
|
||||
}
|
||||
|
||||
impl Lane {
|
||||
/// Turn each of `work`'s previews into a stored thumbnail's worth of bytes.
|
||||
///
|
||||
/// The four calls, in the order `fetch_preview` and `encode_preview` make
|
||||
/// them. `is_complete_jpeg` is included because it is in the real path and
|
||||
/// because leaving it out would be the sort of small omission that turns a
|
||||
/// measurement into an estimate: a truncated JPEG decodes "successfully"
|
||||
/// into a partial frame, so the check is not optional and its cost is not
|
||||
/// somebody else's.
|
||||
fn run(sources: &[Vec<u8>], work: Vec<usize>) -> Lane {
|
||||
let mut made = Vec::with_capacity(work.len());
|
||||
let mut times = Vec::with_capacity(work.len());
|
||||
let mut failed = 0usize;
|
||||
|
||||
for index in work {
|
||||
let bytes = &sources[index % sources.len()];
|
||||
// The fixture makes every fourth image portrait, so a quarter of
|
||||
// these pay for the permutation `apply_orientation` performs. In a
|
||||
// real library that fraction is whatever the photographer shot.
|
||||
let orientation = if index % 4 == 3 {
|
||||
Orientation::from_exif(6)
|
||||
} else {
|
||||
Orientation::default()
|
||||
};
|
||||
|
||||
let started = Instant::now();
|
||||
if !dr_decode::is_complete_jpeg(bytes) {
|
||||
failed += 1;
|
||||
continue;
|
||||
}
|
||||
let mut preview = match dr_decode::decode_jpeg(bytes) {
|
||||
Ok(p) => p,
|
||||
Err(e) => {
|
||||
log::debug!("bench preview {index}: {e}");
|
||||
failed += 1;
|
||||
continue;
|
||||
}
|
||||
};
|
||||
preview.downscale_to(SIZE.edge());
|
||||
preview.apply_orientation(orientation);
|
||||
let rgba = &preview.rgba;
|
||||
let encoded = match dr_thumbs::encode_rgba(preview.width, preview.height, rgba) {
|
||||
Ok(b) => b,
|
||||
Err(e) => {
|
||||
log::debug!("bench thumbnail {index}: {e}");
|
||||
failed += 1;
|
||||
continue;
|
||||
}
|
||||
};
|
||||
times.push(ms(started.elapsed()));
|
||||
|
||||
made.push((
|
||||
index as u64 + 1,
|
||||
Thumbnail {
|
||||
width: preview.width,
|
||||
height: preview.height,
|
||||
bytes: encoded,
|
||||
},
|
||||
));
|
||||
}
|
||||
|
||||
Lane {
|
||||
made,
|
||||
times,
|
||||
failed,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// How many lanes to sweep with by default.
|
||||
///
|
||||
/// The machine's threads, not `SWEEP_LANES`. The sweep's six are sized for
|
||||
/// *latency* — each image is ~0.6 s of WebDAV round trip and almost no CPU, so
|
||||
/// six in flight is a queue depth rather than a core count. With the bytes
|
||||
/// already in memory the work is purely CPU, and six lanes on a 24-thread
|
||||
/// desktop would report a quarter of the throughput the machine has. Stated
|
||||
/// here rather than buried, because it is the one place this deviates from the
|
||||
/// shape it otherwise copies.
|
||||
pub fn default_lanes() -> usize {
|
||||
std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1)
|
||||
}
|
||||
Reference in New Issue
Block a user