//! TRACES: NFR-P3 //! Thumbnail throughput on the embedded preview path. //! //! NFR-P3 asks for **≥ 100 images per second** on the reference desktop //! through the embedded preview path, and ≥ 25 on a mid-range Android device. //! Nothing had ever counted. //! //! # What is timed, and why it is not the sweep itself //! //! `ui/dr-ui/src/library.rs`'s [`spawn_thumbnail_sweep`] is the whole-library //! pass, and it is the machinery this mirrors: a chunk at a time, lanes owning //! disjoint slices, every lane decoding and encoding on its own, and the //! single thread that owns the store writing the finished chunk. That shape is //! reproduced here because it is the shape that decides the number — where the //! parallelism is, and where the one lock is. //! //! It is *mirrored* rather than called, for a reason worth stating plainly: //! that function takes a `RemoteBackend` and spends most of its wall clock in //! WebDAV round trips. Calling it from a benchmark would need a Nextcloud //! server, and what it would then measure is somebody's network. The per-image //! work is identical either way — `dr_decode::decode_jpeg`, //! `Preview::downscale_to`, `Preview::apply_orientation`, //! `dr_thumbs::encode_rgba`, `ThumbStore::put` — and that work is what a target //! written in images per second is about. //! //! So: **the bytes are already in memory when the clock starts.** This is CPU //! throughput for the preview path with the fetch paid for, which is what a //! local library gives you and what the target's "embedded preview path" //! names. It is not a claim about a remote library, whose ceiling is latency //! and which [`spawn_thumbnail_sweep`] exists to hide rather than to beat. //! //! # The plain-JPEG branch, deliberately //! //! `fetch_preview` has two arms: a plain JPEG is its own preview, and anything //! else is located inside the container first. The fixture's files take the //! first arm, so `locate_preview` is not in the measured span. That is honest //! for two reasons — a library of camera JPEGs is a real library and takes //! exactly this path, and for a RAW the located preview *is* a JPEG of about //! this size, so what changes is a header walk of a few microseconds against a //! decode of several milliseconds. What is genuinely not measured is the //! container parse of an exotic format, and nothing here pretends otherwise. //! //! [`spawn_thumbnail_sweep`]: ../../dr_ui/library/fn.spawn_thumbnail_sweep.html use std::path::{Path, PathBuf}; use std::time::Instant; use anyhow::{Context, Result}; use dr_thumbs::{ThumbSize, ThumbStore, Thumbnail}; use dr_types::Orientation; use crate::stats::{ms, Percentiles}; /// Images per chunk handed back to the thread that owns the store. /// /// 96, which is `SWEEP_CHUNK` in `library.rs`. Copied rather than chosen: the /// point of this row is to describe the sweep's behaviour, and a different /// chunk size would move the ratio of lane work to store work. const CHUNK: usize = 96; /// The size class the whole-library pass fills. /// /// Grid only, which is `SWEEP_THUMB_SIZE`. The large class is four times the /// transfer for a detail only a zoomed cell asks for, so the sweep does not /// produce it and neither does this. const SIZE: ThumbSize = ThumbSize::Grid; /// What a measured sweep produced. pub struct Throughput { pub images: usize, pub lanes: usize, pub wall_ms: f64, pub images_per_second: f64, /// Per image, on the lane that produced it: decode, downscale, orient, /// encode. Not the store write, which happens elsewhere by design. pub per_image: Percentiles, /// Thumbnails that reached the store. pub stored: usize, /// Images whose preview would not decode. Any non-zero is a broken /// fixture, not a slow one. pub failed: usize, } /// One lane's output: what it made, what each one cost, and what it dropped. struct Lane { made: Vec<(u64, Thumbnail)>, times: Vec, failed: usize, } /// Sweep `images` thumbnails from `sources`, `lanes` at a time. /// /// `store_dir` is emptied first. Re-storing an id that is already present is /// an `UPDATE` in place rather than an insert (see `ThumbStore::put`), and a /// run that measured updates would not be measuring the pass this describes — /// the sweep's work list is by construction what the store does *not* have. pub fn measure( sources: &[PathBuf], store_dir: &Path, images: usize, lanes: usize, ) -> Result { anyhow::ensure!(!sources.is_empty(), "no fixture sources to sweep"); anyhow::ensure!(lanes > 0, "a sweep needs at least one lane"); let bytes: Vec> = sources .iter() .map(|p| std::fs::read(p).with_context(|| format!("reading {}", p.display()))) .collect::>()?; let _ = std::fs::remove_dir_all(store_dir); let mut store = ThumbStore::open(store_dir) .map_err(|e| anyhow::anyhow!("opening the bench thumbnail store: {e}"))?; // Warm-up: every source decoded once, discarded. The first decode of a // file faults in its Huffman tables and grows the allocator's arenas to // the size a 1620x1080 RGBA buffer needs, and neither recurs across a // sweep of thousands. let warm_failures = Lane::run(&bytes, (0..bytes.len()).collect()).failed; anyhow::ensure!( warm_failures == 0, "{warm_failures} of the {} fixture sources would not decode", bytes.len() ); let mut samples: Vec = Vec::with_capacity(images); let mut stored = 0usize; let mut failed = 0usize; let started = Instant::now(); for chunk_start in (0..images).step_by(CHUNK) { let chunk_end = (chunk_start + CHUNK).min(images); // Each lane takes every `lanes`-th image of the chunk, which is how // `spawn_thumbnail_sweep` splits one: disjoint slices, nothing shared, // no lock. let work: Vec> = (0..lanes) .map(|lane| (chunk_start..chunk_end).skip(lane).step_by(lanes).collect()) .collect(); let source_slice: &[Vec] = &bytes; let produced: Vec = std::thread::scope(|scope| { let handles: Vec<_> = work .into_iter() .map(|lane| scope.spawn(move || Lane::run(source_slice, lane))) .collect(); handles .into_iter() .map(|h| h.join().expect("a sweep lane panicked")) .collect() }); // The store is `&mut` and single-writer, so the chunk is written here // and not on the lanes. This is the sweep's own discipline and it is // part of the number: if the store were the bottleneck, no amount of // lane parallelism would help and the row would say so. for lane in produced { samples.extend(lane.times); failed += lane.failed; for (file_id, thumb) in lane.made { match store.put(file_id, SIZE, &thumb) { Ok(_) => stored += 1, Err(e) => { log::warn!("storing bench thumbnail {file_id}: {e}"); failed += 1; } } } } } let wall = started.elapsed(); anyhow::ensure!( failed == 0, "{failed} of {images} thumbnails failed; the throughput below would be \ measuring how fast this gives up" ); let wall_ms = ms(wall); Ok(Throughput { images, lanes, wall_ms, images_per_second: images as f64 / wall.as_secs_f64().max(f64::MIN_POSITIVE), per_image: Percentiles::of(samples), stored, failed, }) } impl Lane { /// Turn each of `work`'s previews into a stored thumbnail's worth of bytes. /// /// The four calls, in the order `fetch_preview` and `encode_preview` make /// them. `is_complete_jpeg` is included because it is in the real path and /// because leaving it out would be the sort of small omission that turns a /// measurement into an estimate: a truncated JPEG decodes "successfully" /// into a partial frame, so the check is not optional and its cost is not /// somebody else's. fn run(sources: &[Vec], work: Vec) -> Lane { let mut made = Vec::with_capacity(work.len()); let mut times = Vec::with_capacity(work.len()); let mut failed = 0usize; for index in work { let bytes = &sources[index % sources.len()]; // The fixture makes every fourth image portrait, so a quarter of // these pay for the permutation `apply_orientation` performs. In a // real library that fraction is whatever the photographer shot. let orientation = if index % 4 == 3 { Orientation::from_exif(6) } else { Orientation::default() }; let started = Instant::now(); if !dr_decode::is_complete_jpeg(bytes) { failed += 1; continue; } let mut preview = match dr_decode::decode_jpeg(bytes) { Ok(p) => p, Err(e) => { log::debug!("bench preview {index}: {e}"); failed += 1; continue; } }; preview.downscale_to(SIZE.edge()); preview.apply_orientation(orientation); let rgba = &preview.rgba; let encoded = match dr_thumbs::encode_rgba(preview.width, preview.height, rgba) { Ok(b) => b, Err(e) => { log::debug!("bench thumbnail {index}: {e}"); failed += 1; continue; } }; times.push(ms(started.elapsed())); made.push(( index as u64 + 1, Thumbnail { width: preview.width, height: preview.height, bytes: encoded, }, )); } Lane { made, times, failed, } } } /// How many lanes to sweep with by default. /// /// The machine's threads, not `SWEEP_LANES`. The sweep's six are sized for /// *latency* — each image is ~0.6 s of WebDAV round trip and almost no CPU, so /// six in flight is a queue depth rather than a core count. With the bytes /// already in memory the work is purely CPU, and six lanes on a 24-thread /// desktop would report a quarter of the throughput the machine has. Stated /// here rather than buried, because it is the one place this deviates from the /// shape it otherwise copies. pub fn default_lanes() -> usize { std::thread::available_parallelism() .map(|n| n.get()) .unwrap_or(1) }