//! TRACES: FR-CAT-1 | FR-CAT-4 | FR-NC-3 | NFR-P9 //! Opening a remote library: scan → catalog → grid. //! //! This is the wire between three pieces that already worked separately — //! `dr_sync::scan` walks the tree, `dr_catalog` indexes it, and //! `dr_decode::preview` turns bytes into pixels. Until now the "Open library" //! button logged its intent and stopped. //! //! # Threading //! //! Slint's event loop is single-threaded and must never block (NFR-P9), so //! every network and decode operation runs on a worker thread and results //! return through an mpsc channel drained by a Slint timer. That is the same //! shape the login flow uses; it is repeated rather than shared because the //! message types differ and a generic version would obscure both. //! //! # Why thumbnails are fetched, not derived from the scan //! //! A scan yields paths and sizes, nothing visual. Each thumbnail costs its own //! range request, so they are fetched **only for cells the grid actually //! wants** — never for the whole library up front. On the reference library //! that is the difference between a few MB and ~370 GB (ARCH §6.7). use std::path::PathBuf; use std::sync::mpsc::{Receiver, Sender}; use dr_catalog::{Catalog, JobKind, Priority}; use dr_sync::{Account, Connection, RemoteBackend, RemoteError, RemoteId, RemotePath}; use dr_thumbs::ThumbStore; use crate::sidecar_cache::SidecarCache; use dr_types::FormatFilter; /// Largest preview worth fetching whole. /// /// A located preview above this is skipped rather than transferred: past a few /// MB the saving over the full file stops justifying the wait, and a 256px /// thumbnail needs nothing like that much detail. const MAX_PREVIEW_BYTES: u64 = 8 * 1024 * 1024; /// Longest edge face indexing works at. /// /// Not the full preview: `index_proxy` needs packed `f32` RGB, which is 12 /// bytes a pixel, so a 24 MP frame would be ~288 MB and the fetch lanes hold /// one each. 3072 costs ~75 MB at the same moment and still puts a face 2% /// across the frame at ~61 source pixels, against 5 on a grid thumbnail. /// /// Raise it if the embedder is ever given a larger input than 112: it is the /// resolution the *crop* is sampled from, so it bounds face quality directly. const FACE_SOURCE_EDGE: u32 = 3072; /// Progress and results from the scan worker. #[derive(Debug)] pub enum ScanMessage { /// Directories walked so far, and images found. Progress { directories: usize, pruned: usize, images: usize, }, /// The scan finished and the catalog is populated. /// /// `found` counts what this scan *listed*, which on an incremental rescan /// is only what changed — pruned directories contribute nothing. `total` /// is what the catalog actually holds, which is what the grid shows. /// Conflating them made a successful no-op rescan report "0 images" and /// blank the library. Done { found: usize, total: usize, pruned: usize, elapsed_ms: u64, /// TRACES: FR-CAT-8 | FR-NC-9 /// Judgements this scan took *in* from other devices' sidecars. /// /// Counted and reported rather than left to the log because it is the /// only visible sign that a cull made elsewhere has arrived. A grid /// that silently gains three hundred stars is indistinguishable from /// one that has gone wrong. judgements: usize, }, /// The scan could not finish. /// /// `offline` distinguishes "the server could not be reached" from "the /// server refused", and it is carried here rather than re-derived because /// the classification is only possible on the worker side: crossing the /// channel flattens a [`dr_sync::RemoteError`] into a message, and no /// amount of string matching on the far side can reliably recover it. /// Without the flag a dead connection and a bad password produce the same /// banner, which sends the user to re-enter a credential that was fine. /// /// `lost_root` is the same idea one step further out, and it is carried /// separately from `offline` rather than folded into it because the two /// end differently. An offline library comes back when the network does, /// with nothing asked of anyone; a library whose root cannot be opened /// comes back only when someone restores access to it — a share put back /// on the server, a drive plugged in, and in time a document tree granted /// again once one can be (FR-PLAT-AND-2). Both show the same grid of what /// is stored locally, and they must not offer the same explanation. Failed { message: String, offline: bool, lost_root: bool, }, } /// One decoded thumbnail, ready for the grid. #[derive(Debug)] pub struct ThumbnailReady { /// Index into the grid model this belongs to. pub row: usize, pub width: u32, pub height: u32, pub rgba: Vec, /// Whether these pixels came off local disk rather than the server. /// /// The grid paints both identically, so this exists solely for /// reachability: a store hit is evidence about the *cache*, not the /// network, and treating one as proof of connectivity clears offline mode /// before a single request has been attempted. pub from_cache: bool, } /// Capture metadata read from the same header the thumbnail needed. /// /// Free: the header fetch happens either way, so parsing EXIF out of it costs /// no extra transfer. That is what fills the timeline as the user browses, /// rather than a separate 6 GB sweep over the library. #[derive(Debug, Clone)] pub struct MetadataFound { pub image_id: i64, pub captured_at: Option, pub captured_offset: Option, pub camera: Option, pub lens: Option, pub iso: Option, } /// Messages from the thumbnail worker. #[derive(Debug)] pub enum ThumbnailMessage { Ready(Box), /// No preview could be extracted. The cell stays a placeholder rather than /// silently retrying forever. Unavailable { row: usize, reason: String, }, /// How the batch split between the store and the network. /// /// Sent once, before any fetch. Without it there is no way to tell a /// working cache from a broken one — both fill the grid, one just costs /// nothing. Plan { cached: usize, fetching: usize, dating: usize, }, /// One header-only date read is starting. /// /// Reported separately from thumbnail progress: this work produces no /// visible cell, so without it the window looks idle while it runs. DateProgress, /// Capture dates were written to the catalog. /// /// The timeline is rebuilt on this rather than per image — a histogram /// that redrew 120 times during a batch would flicker for no benefit. DatesRecorded(usize), /// TRACES: FR-CAT-9 /// The server could not be reached while filling this batch. /// /// Distinct from a run of [`Unavailable`](Self::Unavailable): those are /// per-image verdicts ("this file has no extractable preview") and leave /// the rest of the library alone, where this is a statement about the /// connection. Sent at most once per batch, because a dropped connection /// produces one of these per *cell* otherwise and the banner would be /// rewritten sixty times. Offline { reason: String, }, } /// TRACES: FR-CAT-15 | FR-CAT-11 /// What it means for an image to be visible in the library. /// /// Two exclusions, for two different reasons, and both must appear in *every* /// query that counts or lists cells — the grid, the timeline, the metadata /// sweep. A predicate present in four of five places is worse than absent: the /// counts disagree with the cells and neither looks wrong on its own. /// /// - `shadowed_by IS NULL` — a JPEG the camera wrote beside its RAW is that /// same frame, not a second photograph. /// - `trashed_at IS NULL` — a soft-deleted image has been moved to the trash /// folder and is listed only by the trash view. const VISIBLE: &str = "i.shadowed_by IS NULL AND i.trashed_at IS NULL"; /// [`VISIBLE`] for queries that do not alias `images`. const VISIBLE_UNALIASED: &str = "shadowed_by IS NULL AND trashed_at IS NULL"; /// TRACES: FR-CAT-15 /// What the *trash view* lists: exactly what [`VISIBLE`] excludes on the second /// clause, and still excludes on the first. /// /// The inversion is deliberate and only correct on `trashed_at`. A shadowed JPEG /// is not a separate photograph in the trash any more than it is in the library /// — trashing a RAW takes its sibling with it, and listing both would offer to /// restore the same frame twice. const TRASHED: &str = "i.shadowed_by IS NULL AND i.trashed_at IS NOT NULL"; /// TRACES: FR-CULL-5 /// The clause that hides the frames a collapsed burst is standing in for. /// /// Subject to exactly the discipline [`VISIBLE`] is under, and for the same /// reason: the header's count, the scrollbar's size, the run a shift-click /// resolves and the ordinal a scrub lands on are four answers about one list. /// A burst folded away in the cells but still counted in the total would leave /// the grid ending in rows that draw nothing, with no clue why. /// /// The predicate itself is `dr_catalog::bursts`'s, not this file's, so the /// interface and the pass that writes the table cannot come to disagree about /// what collapsed means. /// /// A function rather than a constant because it has to name the image table, /// and the grid aliases it as `i` where the timeline's queries do not. `image` /// is a table name from this file and never anything a user supplied. fn uncollapsed(image: &str) -> String { format!(" AND {}", dr_catalog::bursts::not_collapsed_away(image)) } /// TRACES: FR-CAT-4 /// The order the grid lists photographs in: when they were taken. /// /// The file name breaks ties and nothing more — two frames of one burst, or a /// RAW beside the JPEG the camera wrote with it. What a photographer looks for /// is the afternoon, not what the camera called the file, and a grid ordered by /// name interleaves every camera and every card that ever wrote into the same /// folder. /// /// Shared rather than spelled out per query, for the same reason [`VISIBLE`] is: /// the window, the count and the run a shift-click resolves are three answers /// about one list, and an ordering that drifted between them would select /// photographs the user never saw without any of it looking wrong. /// /// Undated images sort last in either direction. EXIF is read as thumbnails /// load, so a freshly scanned library would otherwise open on the images it /// knows least about. const GRID_ORDER: &str = "ORDER BY i.captured_at IS NULL, i.captured_at ASC, i.source_ref ASC"; /// TRACES: FR-CAT-15 /// [`GRID_ORDER`] for the trash, which is ordered by when a thing was deleted — /// see [`read_trashed_cells`] for why that view answers a different question. const TRASH_ORDER: &str = "ORDER BY i.trashed_at DESC, i.source_ref ASC"; /// TRACES: FR-CAT-6 | FR-CULL-4 /// What the grid is narrowed to by the rating filter bar. /// /// Applied in **SQL**, not by filtering the rows after reading them. On a /// remote library a drawn-then-hidden cell has already cost a thumbnail /// fetch, which is the transfer FR-NC-3 exists to avoid — and the count in the /// header has to agree with the cells, which it cannot if the two are computed /// at different stages. /// Not `Copy`: [`RatingFilter::people`] is a `Vec`. Every query path already /// takes this by reference, so the only casualties were two `..*self` struct /// updates, which clone instead. #[derive(Debug, Clone, PartialEq, Eq, Default)] pub struct RatingFilter { /// Minimum stars. 0 means no star constraint. pub min_rating: u8, /// Only images nothing has judged yet — neither starred nor flagged. /// This is what lets a culling session resume where it stopped. pub unjudged: bool, /// `None` for no flag constraint, otherwise exactly that flag. pub flag: Option, /// TRACES: FR-CAT-9 /// Only images whose original is stored on this device. /// /// Carried here, beside the rating terms, because every query path already /// threads this one struct: adding a parallel parameter to /// `read_cells_scoped`, `read_cells_all` and both counts would give four /// call sites the chance to disagree about what the grid is showing, and /// the count disagreeing with the cells is the specific bug this type's /// "filter in SQL" rule exists to prevent. pub local_only: bool, /// Show only photographs captured within this range, as UTC seconds. /// /// Half-open ends are meaningful: a `from` with no `to` reads as /// "everything since". Undated images are excluded whenever either end is /// set — they cannot be placed on the axis the user is narrowing, and /// showing them anyway makes the range look broken. /// /// Here rather than a parallel parameter for the reason `local_only` gives /// above: the count and the cells must be narrowed by the same thing. pub captured_from: Option, pub captured_to: Option, /// TRACES: FR-CULL-11 /// Only photographs these people appear in. /// /// The way back from a face to the pictures it came from, which is the /// question the People screen leaves the user holding: they have just /// identified someone, and what they want next is *everything with them in /// it*. Without this the identification is a dead end. /// /// Here rather than a grid scope of its own, for the reason `local_only` /// gives above — the count and the cells must be narrowed by the same /// thing, and this struct is the one narrowing every query path already /// threads. It composes with the rest for free: three-star photographs of /// Anna from last summer is this term ANDed with two others. /// /// Suggested faces count, not only confirmed ones. A user who has just /// grouped someone and not yet confirmed a single face would otherwise get /// an empty grid, which reads as "no photographs of this person" rather /// than "you have not ticked anything yet". /// /// A set rather than one id, because the two questions a photographer /// actually asks are "every picture of Anna *or* Bob" and "the pictures /// they are *both* in", and the second is not reachable by any sequence of /// single-person filters. [`RatingFilter::people_mode`] picks between them. pub people: Vec, /// Whether [`RatingFilter::people`] is a union or an intersection. pub people_mode: PeopleMode, } /// How several people combine when the grid is narrowed by identity. /// /// TRACES: FR-CULL-11 #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub enum PeopleMode { /// Photographs holding **any** of them — the union. /// /// The default, and the right one for one person, where the two modes are /// identical. It is also the forgiving direction: adding a second person to /// a union can only ever show more, so a user who has not noticed the /// toggle never ends up staring at an empty grid wondering what they broke. #[default] Any, /// Photographs holding **all** of them — the intersection. /// /// "Pictures of the two of them together", which is the one worth having a /// mode for: it is how you find the photograph you remember rather than /// scrolling everything either of them appears in. All, } impl RatingFilter { /// Whether this narrows anything, so the caller can skip the join. pub fn is_unfiltered(&self) -> bool { self.min_rating == 0 && !self.unjudged && self.flag.is_none() && !self.local_only && self.captured_from.is_none() && self.captured_to.is_none() && self.people.is_empty() } /// Whether a date range is narrowing the grid. pub fn has_date_range(&self) -> bool { self.captured_from.is_some() || self.captured_to.is_some() } /// The same filter with the date range lifted. /// /// The timeline uses this: the histogram is how the range is *chosen*, so /// drawing it through the range would collapse the axis onto the current /// selection and leave nowhere to widen it back out from. pub fn without_date_range(&self) -> Self { Self { captured_from: None, captured_to: None, ..self.clone() } } /// The SQL predicate, against an `images` aliased as `i`. /// /// Returns a `String` of conditions ANDed together, or an empty string /// where nothing is constrained. Every branch is built from integers this /// code owns — no caller text reaches the SQL, so there is nothing to /// escape. /// /// A correlated subquery per term rather than a join to `versions`: an /// image with no version row must still be *findable* as unrated, and an /// inner join would silently drop exactly those images — the ones a /// library scanned before ratings existed consists entirely of. fn sql(&self) -> String { let mut terms = Vec::new(); if self.min_rating > 0 { terms.push(format!( "coalesce((SELECT dv.rating FROM versions dv WHERE dv.image_id = i.id AND dv.is_default = 1 LIMIT 1), 0) >= {}", self.min_rating )); } // Integers this code owns, formatted straight in like the rating terms // above — no caller text reaches the SQL. if let Some(from) = self.captured_from { terms.push(format!("i.captured_at >= {from}")); } if let Some(to) = self.captured_to { terms.push(format!("i.captured_at <= {to}")); } if self.has_date_range() { // An undated image cannot be inside or outside a range. Excluding // it is the honest answer; the comparisons above would drop it // anyway, and saying so keeps that from looking accidental. terms.push("i.captured_at IS NOT NULL".to_string()); } if self.unjudged { // Both axes: a frame that was picked but never starred has been // judged, and re-presenting it would undo the user's decision to // move past it. terms.push( "coalesce((SELECT dv.rating FROM versions dv WHERE dv.image_id = i.id AND dv.is_default = 1 LIMIT 1), 0) = 0 AND coalesce((SELECT dv.flag FROM versions dv WHERE dv.image_id = i.id AND dv.is_default = 1 LIMIT 1), 0) = 0" .to_string(), ); } if !self.people.is_empty() { // Integers this code owns, like every other term here — the ids // come from the catalog, never from typed text, so there is // nothing to escape. let ids = self .people .iter() .map(|p| p.to_string()) .collect::>() .join(","); terms.push(match self.people_mode { // `EXISTS` rather than a join, so a photograph holding three // faces of the same person appears once — the grid shows // pictures, not faces. PeopleMode::Any => format!( "EXISTS (SELECT 1 FROM faces f JOIN face_person fp ON fp.face_id = f.id WHERE f.image_id = i.id AND fp.person_id IN ({ids}))" ), // Counting *distinct* people rather than ANDing one EXISTS per // person: same result, one subquery instead of n, and it does // not grow the statement with the selection. `DISTINCT` is // what makes it correct — three faces of Anna in one frame // must not satisfy a filter asking for Anna and Bob. PeopleMode::All => format!( "(SELECT COUNT(DISTINCT fp.person_id) FROM faces f JOIN face_person fp ON fp.face_id = f.id WHERE f.image_id = i.id AND fp.person_id IN ({ids})) = {}", self.people.len() ), }); } if let Some(flag) = self.flag { terms.push(format!( "coalesce((SELECT dv.flag FROM versions dv WHERE dv.image_id = i.id AND dv.is_default = 1 LIMIT 1), 0) = {}", flag_code(flag) )); } if self.local_only { // `tier_actual`, not `tier_desired`: the question is what is // *here*, not what a pin has promised will be. An image queued for // download is exactly the one that cannot be opened yet, so // showing it under "on this device" would be the wrong answer to // the only question this filter is asked. terms.push(format!( "EXISTS (SELECT 1 FROM image_cache ic WHERE ic.image_id = i.id AND ic.tier_actual >= {})", dr_types::Tier::Original.stored() )); } if terms.is_empty() { String::new() } else { format!(" AND ({})", terms.join(") AND (")) } } } /// The stored integer for a flag, matching `dr_catalog::rating`'s encoding. fn flag_code(f: dr_types::FlagState) -> i64 { match f { dr_types::FlagState::Unflagged => 0, dr_types::FlagState::Pick => 1, dr_types::FlagState::Reject => 2, } } /// TRACES: FR-CAT-8 | FR-NC-8 | FR-CULL-4 | FR-DEV-6 /// One amendment to one image's sidecar, on its way to the server. #[derive(Debug, Clone)] pub struct SidecarWrite { /// Remote path of the *image*. The sidecar sits beside it, with the /// extension replaced — that adjacency is what makes a sidecar findable /// without an index (ARCH §6.12). pub image_path: String, pub version_uuid: String, pub amendment: Amendment, } /// What a write changes about the version it names. /// /// An enum rather than a struct of optional fields because the two are written /// by different actions with different failure costs, and because a write must /// never carry a *stale* copy of what it is not changing. A settings write that /// also carried a rating would have to have read one from somewhere, and the /// obvious somewhere — the catalog, moments earlier — is exactly how a cull /// made between the read and the write gets silently reverted. /// /// Everything not named by the variant is left as the file had it, which is /// what makes the read-modify-write in [`write_one_sidecar`] a genuine /// amendment rather than a replacement. #[derive(Debug, Clone)] pub enum Amendment { /// A star rating and a pick/reject flag — the cull. Judgement { rating: u8, flag: u8 }, /// TRACES: FR-DEV-6 /// Copied develop settings, applied within `scope`. /// /// Carries the [`Scope`] rather than a pre-filtered preset so the target's /// own framing can be spared *at the file*: excluding framing means /// leaving the keys already in the sidecar untouched, which cannot be /// expressed by the parameter list alone. Settings { preset: dr_pipeline::Preset, scope: dr_pipeline::Scope, /// TRACES: FR-DEV-3f /// The film stock, when this is an image's own edit being written back. /// /// Two levels of `Option`, and both are load-bearing. The outer says /// whether this write concerns the film at all — a paste does not, /// exactly as it carries no masks. The inner is the choice itself, and /// `Some(None)` is a real edit: "develop this normally again". Without /// the distinction, clearing a film could never be saved. film: Option>, /// TRACES: FR-DEV-5 /// The named snapshots, when this is an image's own edit being /// written back: the ones the session holds, and the ids it deleted. /// A paste carries none — a snapshot is a state of one photograph. snapshots: Option<(Vec, Vec)>, /// TRACES: FR-DEV-3 | FR-CAT-8 /// The local adjustments, when this is an image's own edit being /// written back rather than a paste onto someone else's. /// /// A `Preset` is a parameter map, and a mask is not a parameter — it /// is a rule about *where*, with a chain of its own. So a save that /// carried only the preset wrote the sliders and silently dropped /// every local adjustment: the sidecar format has stored masks since /// they were added and `Version::apply` restores them, but nothing /// ever put any there. The mask survived until the session ended and /// then did not exist. /// /// `None` for a paste, which must not carry the source image's masks /// onto the target: a mask is drawn against one photograph and means /// nothing on another, and `Scope` cannot express that because it /// filters parameters. masks: Option, }, } /// Where an image's sidecar lives. /// /// The image's own path with the extension replaced, not appended: `a.CR2` /// becomes `a.drsc`, so a RAW and the JPEG beside it share one sidecar and /// therefore one judgement. That is the intended behaviour — they are the same /// photograph (FR-CAT-11), and the pairing logic in `dr_catalog::schema` /// already treats them so. pub fn sidecar_path(image_path: &str) -> String { let stem = match image_path.rsplit_once('.') { // Only an extension in the final segment counts; a dot in a directory // name must not truncate the path. Some((stem, ext)) if !ext.contains('/') => stem, _ => image_path, }; format!("{stem}.{}", dr_pipeline::sidecar::EXTENSION) } /// TRACES: FR-CAT-8 | FR-CAT-9 | FR-NC-10 /// Persist amendments to sidecars beside their images. /// /// # Why this reads before it writes /// /// A sidecar is the authoritative store and may already hold an edit made on /// this or another device. Writing a fresh document containing only a rating /// would delete that edit — the exact silent data loss the format's /// unknown-key preservation exists to prevent. So each file is fetched, /// parsed, amended, and written back; a fetch that 404s simply means there is /// no sidecar yet and a new one is created. /// /// # Why the local write is the commit point /// /// FR-CAT-9 requires that edits made offline *queue and apply when the source /// returns*. So every amendment is written to the local cache first and the /// upload is best-effort: an entry stays marked pending until the server has /// actually taken it, and [`spawn_outbox_drain`] retries the marked ones later. /// /// This is what makes `offline` a parameter rather than a reason to skip. It /// was one: a cull or a paste made with no connection used to be dropped /// entirely, which for a pasted edit meant it survived nowhere at all — the /// catalog holds no parameters. Now the two cases differ only in whether the /// upload is attempted. /// /// # Why failure here is logged rather than surfaced /// /// The write has already succeeded locally by the time the network is touched, /// so nothing is lost by a failure and there is nothing for the user to do /// about it. Interrupting a cull with an error dialog per frame would be far /// worse than the risk. The counts are reported once, at the end. pub fn spawn_sidecar_writes( conn: Connection, writes: Vec, cache_dir: PathBuf, offline: bool, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let cache = SidecarCache::open(cache_dir); // The runtime and the backend are only needed to *upload*. Offline, // neither is built — and a failure to build either is not a failure to // record the edit, it just means every write is queued instead. let rt = if offline { None } else { match crate::net_runtime::build() { Ok(rt) => Some(rt), Err(e) => { log::debug!("no runtime for sidecar upload ({e}); queueing"); None } } }; // Every write, recorded locally and queued. The offline path, and the // fallback whenever a backend could not be built. let queue_all = || { let mut report = SidecarReport::default(); for w in &writes { match write_one_sidecar(&cache, w) { Ok(Outcome::Uploaded) => report.written += 1, Ok(Outcome::Queued) => report.queued += 1, Err(e) => { // Warn, not debug. This is unsynced user work — a rating or an // edit that exists only on this device — and the path is // the only thing that says *which* photograph and *where* // the server refused it. Filtered out at the default // level, a 403 on one file is indistinguishable from a // whole library failing. log::warn!("sidecar for {}: {e}", w.image_path); report.last_error = Some(e); report.failed += 1; } } } report }; let report = match rt { None => queue_all(), Some(rt) => rt.block_on(async { match crate::remote::connect(&conn) { Ok(b) => { let mut report = SidecarReport::default(); for w in &writes { match write_one_sidecar_online(&*b, &cache, w).await { Ok(Outcome::Uploaded) => report.written += 1, Ok(Outcome::Queued) => report.queued += 1, Err(e) => { // Warn, not debug. This is unsynced user work — a rating or an // edit that exists only on this device — and the path is // the only thing that says *which* photograph and *where* // the server refused it. Filtered out at the default // level, a 403 on one file is indistinguishable from a // whole library failing. log::warn!("sidecar for {}: {e}", w.image_path); report.last_error = Some(e); report.failed += 1; } } } report } // No backend: the edits are still recorded locally and // will go up with the next drain. Err(e) => { log::debug!("no backend for sidecar upload ({e}); queueing"); queue_all() } } }), }; let _ = tx.send(SidecarMessage::Finished { written: report.written, queued: report.queued, failed: report.failed, last_error: report.last_error, }); }); rx } /// What one write ended up doing. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum Outcome { /// Recorded locally and accepted by the server. Uploaded, /// Recorded locally, still in the outbox. Queued, } /// Running totals for a batch, so the loop bodies stay readable. #[derive(Debug, Default)] struct SidecarReport { written: usize, queued: usize, failed: usize, last_error: Option, } /// The outcome of a batch of sidecar writes. #[derive(Debug)] pub enum SidecarMessage { Finished { written: usize, /// Recorded locally but not yet on the server — offline, or an upload /// that failed. These are retried by [`spawn_outbox_drain`], so this /// is a count of *deferred* work rather than of losses. queued: usize, failed: usize, /// Reported once rather than per file: a network that is down fails /// every write with the same message, and forty identical lines in the /// status bar say nothing forty times. last_error: Option, }, } /// Apply an amendment to a document, returning the new one. /// /// Split out from both write paths so that online and offline produce /// *identical* documents: the only thing that differs between them is which /// base was read and whether an upload follows. A second copy of this for the /// offline case is how the two would come to disagree about what a paste means. fn amend(base: dr_pipeline::Sidecar, w: &SidecarWrite) -> dr_pipeline::Sidecar { let mut sidecar = base; // TRACES: FR-NC-8 | FR-NC-9 // Close any split this file already carries, *before* looking for our own // version, and onto the uuid this write is about to use. // // Every device used to mint its own uuid for the same photograph, so a // frame edited on two of them holds two `default = 1` blocks and the // lookup below misses both — adding a third rather than amending either. // Fusing first folds them into one under `w.version_uuid`, which turns the // miss into a hit and makes this an amendment of the other device's work // instead of a rival to it. // // Idempotent: a file with one default and the right uuid is returned // byte-identical, so this costs nothing on the ordinary write. sidecar.fuse_default_versions(Some(&w.version_uuid)); // Amend the version this write belongs to, creating it if the file did // not have one — a photograph nobody has edited anywhere. let mut version = sidecar .versions .get(&w.version_uuid) .cloned() .unwrap_or_else(|| dr_pipeline::sidecar::Version { uuid: w.version_uuid.clone(), name: "Default".to_string(), is_default: true, revision: 0, ..Default::default() }); // Only what the amendment names. Everything else in the version — the // rating a settings write must not touch, the crop an adjustments-only // paste must spare, the unknown keys of an operation this build lacks — // survives because it was read from the file and is written back. match &w.amendment { Amendment::Judgement { rating, flag } => { version.rating = *rating; version.flag = *flag; } Amendment::Settings { preset, scope, masks, film, .. } => { preset.amend(&mut version.params, *scope); // TRACES: FR-DEV-3f // Wholesale, like the masks below and for the same reason: this is // the whole of the image's own choice as it stands, so clearing a // film has to leave the sidecar too. if let Some(film) = film { version.film = film.clone(); } // Replaced wholesale rather than merged: this is the whole of the // image's local adjustment stack as it stands, so a layer the user // deleted has to leave the sidecar too. Cross-device merging of // two stacks is `Sidecar::merge`'s job and happens on sync, not // here (FR-NC-9). if let Some(masks) = masks { version.masks = masks.clone(); } } } // A judgement is an edit as far as the merge is concerned, and so is a // paste: without the bump, a device that touched the same frame earlier // would win on revision and this write would be discarded at the next sync // (FR-NC-9). version.revision = version.revision.saturating_add(1); version.modified = now_secs(); sidecar.put(version); // TRACES: FR-DEV-5 | FR-NC-9 // After the version, and against the uuid the fuse settled on: a // snapshot points at its edit by uuid, and the edit may have just been // renamed onto the canonical one. if let Amendment::Settings { snapshots: Some((kept, removed)), .. } = &w.amendment { sidecar.replace_snapshots(&w.version_uuid, kept.clone(), removed); } sidecar } /// TRACES: FR-CAT-9 /// Record an amendment with no server to send it to. /// /// The base is whatever the cache holds, which is either what the server last /// had or what earlier offline writes have already built on top of it. Either /// way the result is queued, and the drain reconciles it with the server's own /// copy when the connection returns — that reconciliation is a *merge* /// (FR-NC-9), not an overwrite, so building on a possibly-stale base here does /// not cost another device's work. fn write_one_sidecar(cache: &SidecarCache, w: &SidecarWrite) -> Result { let path = sidecar_path(&w.image_path); let base = cache.load(&path).unwrap_or_default(); cache.store(&path, &amend(base, w), true)?; Ok(Outcome::Queued) } /// TRACES: FR-CAT-8 | FR-CAT-9 /// Read-modify-write one sidecar, with a server to read from and send to. async fn write_one_sidecar_online( backend: &dyn RemoteBackend, cache: &SidecarCache, w: &SidecarWrite, ) -> Result { let path_str = sidecar_path(&w.image_path); let path = RemotePath::new(path_str.clone()); let id = RemoteId::Path(path.clone()); // An existing sidecar may hold an edit. Absent is the normal case on a // library that has never been edited, and is not an error. let existing = backend.get(&id, None).await.ok(); // A corrupt sidecar is *not* overwritten: that would destroy an edit this // build merely failed to understand. Refused before anything is written, // locally or remotely, so the cache cannot end up holding a document that // silently discarded the file's real contents. if let Some(bytes) = existing.as_deref() { if !bytes.is_empty() { let text = String::from_utf8_lossy(bytes); if dr_pipeline::Sidecar::parse(&text).is_err() { return Err(format!("sidecar at {path_str} is unreadable")); } } } let base = existing .as_deref() .map(|bytes| String::from_utf8_lossy(bytes).into_owned()) .and_then(|text| dr_pipeline::Sidecar::parse(&text).ok()) // No sidecar on the server. The cache may still hold queued offline // work for this image, and taking `default()` here would drop it. .or_else(|| cache.load(&path_str)) .unwrap_or_default(); let sidecar = amend(base, w); // Locally first: this is the commit point, and an upload that fails after // it leaves the edit queued rather than lost. cache.store(&path_str, &sidecar, true)?; backend .put(&path, sidecar.to_text().into_bytes(), None) .await .map_err(|e| e.to_string())?; // Accepted by the server, so it leaves the outbox. The document stays // cached, which is what lets the next offline open still show the edit. cache.store(&path_str, &sidecar, false)?; Ok(Outcome::Uploaded) } /// TRACES: FR-CAT-9 | FR-NC-9 | FR-NC-10 /// Upload everything the outbox is still holding. /// /// # Why this merges rather than uploads /// /// A queued edit was built on whatever this device last saw. While it sat in /// the outbox another device may have edited the same photograph, and simply /// PUTting the local document would discard that work — the precise failure /// FR-NC-9's node-level merge exists to prevent. So each entry is reconciled /// against the server's current copy before it goes up, and disjoint edits /// (a crop made here, an exposure change made there) both survive. /// /// # Why an entry stays queued on failure /// /// The marker is cleared only after the server has taken the bytes. A drain /// interrupted halfway leaves the rest of the outbox exactly as it was, so /// nothing depends on this running to completion. pub fn spawn_outbox_drain(conn: Connection, cache_dir: PathBuf) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let cache = SidecarCache::open(cache_dir); let queued = cache.pending(); if queued.is_empty() { let _ = tx.send(SidecarMessage::Finished { written: 0, queued: 0, failed: 0, last_error: None, }); return; } log::info!("draining {} queued sidecar(s)", queued.len()); let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { let _ = tx.send(SidecarMessage::Finished { written: 0, queued: queued.len(), failed: 0, last_error: Some(e.to_string()), }); return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { let _ = tx.send(SidecarMessage::Finished { written: 0, queued: queued.len(), failed: 0, last_error: Some(e.to_string()), }); return; } }; let mut report = SidecarReport::default(); for path_str in &queued { match drain_one(&*backend, &cache, path_str).await { Ok(()) => report.written += 1, Err(e) => { // Warn, for the reason the write path does: this is // unsynced work and the path is what makes the failure // actionable. log::warn!("draining {path_str}: {e}"); report.last_error = Some(e); report.failed += 1; // Still queued — the marker was never cleared. report.queued += 1; } } } let _ = tx.send(SidecarMessage::Finished { written: report.written, queued: report.queued, failed: report.failed, last_error: report.last_error, }); }); }); rx } /// Reconcile one queued sidecar with the server and upload it. async fn drain_one( backend: &dyn RemoteBackend, cache: &SidecarCache, path_str: &str, ) -> Result<(), String> { let Some(mut local) = cache.load(path_str) else { // The document went while the drain was running. Nothing to send. return Ok(()); }; let path = RemotePath::new(path_str.to_string()); let id = RemoteId::Path(path.clone()); // TRACES: FR-NC-6c // A miss and a placeholder are not the same answer, and conflating them // destroys work. This read decides whether the sidecar already on the // remote is merged in; treating "the content is not on this device" as // "there is no sidecar" writes a fresh document over an existing one and // discards every edit another device put there — the exact loss the // format's unknown-key preservation exists to prevent. // // A sidecar is a few kilobytes, so the right response to a placeholder is // to fetch it, not to give up. Where that is impossible — no client // running — the entry stays queued, which is what the outbox is for. let remote = match backend.get(&id, None).await { Ok(bytes) => Some(bytes), Err(RemoteError::NotFound(_)) => None, Err(RemoteError::NotMaterialised(_)) => { backend .materialise(&id) .await .map_err(|e| format!("sidecar is not on this device ({e})"))?; match backend.get(&id, None).await { Ok(bytes) => Some(bytes), Err(e) => return Err(format!("sidecar could not be read ({e})")), } } // Anything else — a refused read, a dead connection — leaves the entry // queued rather than resolved by overwriting. Err(e) => return Err(format!("sidecar could not be read ({e})")), }; if let Some(bytes) = remote.as_deref() { if !bytes.is_empty() { let text = String::from_utf8_lossy(bytes); match dr_pipeline::Sidecar::parse(&text) { Ok(remote) => merge_into(&mut local, &remote), // Unreadable on the server. Uploading over it would destroy an // edit this build failed to understand, so the entry stays // queued rather than being resolved destructively. Err(e) => return Err(format!("remote sidecar is unreadable ({e})")), } } } // TRACES: FR-NC-8 | FR-NC-9 // `merge_into` reconciles version by version *by uuid*, so two devices' // independently minted defaults pass straight through it and both land in // what is about to be uploaded. Fusing here is what stops the outbox from // publishing the split rather than resolving it. // // No canonical uuid: this entry may have been queued by a build that had // not derived one yet, and the smallest uuid is device-independent, which // is all convergence needs. The next write from either device moves it // onto the derived identity. local.fuse_default_versions(None); backend .put(&path, local.to_text().into_bytes(), None) .await .map_err(|e| e.to_string())?; cache.store(path_str, &local, false) } /// TRACES: FR-NC-9 /// Merge the server's copy into ours, version by version. /// /// No common ancestor is available — the outbox stores the result, not the /// base it was built from — so the merge runs with `None`, which treats every /// key either side holds as changed. Disjoint keys therefore still both /// survive, and a key both sides set resolves by revision exactly as it would /// with a base. What is lost without one is the ability to see a *deletion*: /// a parameter reset to default on the other device reads as absent rather /// than as removed, so our value stands. That is the same direction of caution /// the judgement merge takes — an edit is preserved rather than erased. fn merge_into(local: &mut dr_pipeline::Sidecar, remote: &dr_pipeline::Sidecar) { for (uuid, their_version) in &remote.versions { match local.versions.get(uuid).cloned() { Some(mut ours) => { ours.merge(their_version, None); local.put(ours); } // A version only the server has — another device's virtual copy // (FR-CAT-12). Keeping it is what stops one device's upload from // deleting another's work. None => local.put(their_version.clone()), } } } /// Where the catalog for an account lives. /// /// Keyed by [`Account::namespace`] so two accounts do not share an index — /// two servers, two logins on one server, or two folders on one disk. Under /// the XDG data directory, not cache: the catalog is rebuildable but /// rebuilding it costs a full rescan, so it is not something to discard on a /// cache sweep. /// /// The namespace is the account's to compute, not this function's, because it /// is also frozen: it names the directory an existing install's catalog, /// thumbnail shards and un-uploaded sidecars are already in. pub fn catalog_path(account: &Account) -> PathBuf { data_root().join(account.namespace()).join("catalog.sqlite") } /// TRACES: FR-UI-8 /// Where this library's last position is remembered. /// /// Beside the catalog, under the same account namespace, for the reason /// `catalog_path` gives: a place belongs to one library, and two folders on one /// disk are two libraries with two positions. /// /// **The name matches the file that travels.** The copy on the server is /// `place.json` under `.darkroom-derived/`, and the exchange between them is a /// straight newest-wins swap of the same bytes — so calling the local one /// anything else would be one more thing to keep in step for no gain. /// /// In the data directory rather than the cache one. The consequence is milder /// here than for the sidecars `data_root` was moved for — losing a place costs /// a scroll, not a day of culling — but a file the system is free to delete is /// one that would rarely survive long enough to be read. pub fn place_path(account: &Account) -> PathBuf { data_root().join(account.namespace()).join("place.json") } /// The directory every account's data hangs off. /// /// **Not the cache directory, and on Android that distinction is the whole /// point.** Neither `XDG_DATA_HOME` nor `HOME` is set there, so this used /// to fall through to `temp_dir()` — which Android resolves to the app's /// *cache*, a directory the system deletes without asking under storage /// pressure. /// /// What sits beside a catalog is not disposable. `sidecars/` is the /// commit point for every rating and edit made offline (see /// `sidecar_cache`), and `outbox/` holds exports the user has been told /// succeeded. A day of culling on a train, evicted by the OS before it ever /// reached the server, is the worst failure this application can have, and /// it would be silent. /// /// `dr_sync::account::declared_data_dir` is the persistent per-app directory /// the Android entry point establishes before anything opens a store. A /// desktop declares nothing and takes the platform's data directory from /// `dr_plat::dirs` — XDG on Linux, `%LOCALAPPDATA%` on Windows — which keeps /// the established location on Linux rather than moving anyone's catalog. fn data_root() -> PathBuf { match dr_sync::account::declared_data_dir() { Some(declared) => declared.join("darkroom"), None => dr_plat::base_dir(dr_plat::Base::Data), } } /// TRACES: FR-NC-10 | NFR-R1 /// Move an account's data out of the cache directory it used to live in. /// /// Called once at startup, before anything opens a store. The durable /// location changed when `catalog_path` stopped falling through to /// `temp_dir()` on Android, and without this the app would find no catalog, /// rescan a library of tens of thousands of images over the network, and /// re-fetch every thumbnail — while the old copy sat in a directory the /// system was free to delete. /// /// Worse than the cost: `sidecars/` and `outbox/` hold work that exists /// nowhere else. Abandoning them would discard offline ratings and edits that /// had not yet synced, silently, as an upgrade. /// /// A rename, not a copy: both directories are inside the app's own data on /// one filesystem, so it is atomic and cannot half-finish. If the destination /// already exists this does nothing — the migration has run, or this is a /// fresh install, and in neither case may it overwrite live data. pub fn migrate_legacy_cache_data(account: &Account) { // Only meaningful where the old fallback and the new one differ, which is // exactly the platform that had the problem. On a desktop with XDG set, // both resolve to the same place and this returns immediately. let legacy_base = std::env::temp_dir(); let Some(current) = catalog_path(account).parent().map(|p| p.to_path_buf()) else { return; }; let Some(account) = current.file_name() else { return; }; let legacy = legacy_base.join("darkroom").join(account); move_account_dir(&legacy, ¤t); } /// The move itself, separated so it can be tested against ordinary /// directories rather than the platform's idea of a cache. fn move_account_dir(legacy: &std::path::Path, current: &std::path::Path) { if legacy == current || !legacy.is_dir() || current.exists() { return; } if let Some(parent) = current.parent() { if let Err(e) = std::fs::create_dir_all(parent) { log::warn!("preparing {}: {e}", parent.display()); return; } } match std::fs::rename(legacy, current) { Ok(()) => log::info!( "moved library data out of the cache: {} -> {}", legacy.display(), current.display() ), // Reported rather than fatal: a failed move leaves the old copy where // it was and costs a rescan, which is recoverable. Stopping the app // over it would not be. Err(e) => log::warn!( "could not move {} to {}: {e}", legacy.display(), current.display() ), } } /// Run a scan on a worker thread, writing results into the catalog. /// /// Returns the receiver the UI drains. The worker owns its own tokio runtime /// and backend; nothing here touches the Slint event loop. pub fn spawn_scan( conn: Connection, root: String, filter: FormatFilter, catalog_path: PathBuf, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let started = std::time::Instant::now(); if let Err(e) = run_scan(&tx, conn, root, filter, catalog_path, started) { let _ = tx.send(ScanMessage::Failed { message: e.message, offline: e.offline, lost_root: e.lost_root, }); } }); rx } /// A scan failure that still knows whether it was a connectivity failure. /// /// The scan crosses a thread boundary, so the typed error cannot travel with /// it; this carries the one bit that must survive. struct ScanFailure { message: String, offline: bool, lost_root: bool, } impl ScanFailure { /// A failure that is nothing to do with reachability — local I/O, a /// runtime that would not start, a catalog that would not open. fn local(message: impl std::fmt::Display) -> Self { Self { message: message.to_string(), offline: false, lost_root: false, } } } impl From for ScanFailure { fn from(e: dr_sync::RemoteError) -> Self { Self { offline: e.indicates_offline(), lost_root: e.indicates_lost_root(), message: e.to_string(), } } } /// TRACES: FR-PLAT-AND-2 | FR-CAT-9 /// Record that a library can no longer be opened, without losing it. /// /// Called on the worker, before the failure crosses the channel, because this /// is where the catalog handle is — and because the marking must be durable /// whether or not anyone is left to draw a banner. A process killed between /// the failure and the next launch must still come back knowing what it could /// not reach. /// /// Nothing is deleted. Every rating, every edit and every row stays exactly /// where it was; what changes is that the images now say they are offline, so /// the grid can show them as held-not-here rather than as ordinary /// photographs whose thumbnails happen to be failing one at a time. /// /// A root with no row yet is the first scan of a library that has never /// succeeded, and there is nothing to mark — the failure alone is the whole /// story, and the launch screen is where it is told. fn mark_library_offline(catalog: &Catalog, root: &str) { let conn = catalog.connection(); let root_id: Option = conn .query_row( "SELECT id FROM roots WHERE label = ?1 AND kind = 'remote'", [root], |r| r.get(0), ) .ok(); let Some(root_id) = root_id else { log::info!("library {root} has no catalog root yet; nothing to mark offline"); return; }; match dr_catalog::mark_root_offline(conn, dr_types::RootId(root_id as u64)) { Ok(()) => log::warn!("library {root} is unreachable; its images are marked offline"), Err(e) => log::error!("could not mark {root} offline: {e}"), } } fn run_scan( tx: &Sender, conn: Connection, root: String, filter: FormatFilter, catalog_path: PathBuf, started: std::time::Instant, ) -> Result<(), ScanFailure> { if let Some(dir) = catalog_path.parent() { std::fs::create_dir_all(dir) .map_err(|e| ScanFailure::local(format!("creating {}: {e}", dir.display())))?; } let catalog = Catalog::open(&catalog_path).map_err(ScanFailure::local)?; let rt = crate::net_runtime::build().map_err(ScanFailure::local)?; rt.block_on(async { // TRACES: FR-PLAT-AND-2 | FR-CAT-9 // Classified rather than flattened to a local failure, because the // removed-card case never gets as far as a request: the folder // connector checks its root when it is constructed, so a library on an // ejected card fails here and not in the walk. Reported as an ordinary // error it left the grid showing a healthy library of images that // could no longer be opened, one silent thumbnail failure at a time. let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { if e.indicates_lost_root() { mark_library_offline(&catalog, &root); } return Err(e.into()); } }; // Stored folder ETags, so an unchanged subtree is skipped whole. On a // first run this is empty and the walk is complete; on every run after // it is what keeps cost proportional to what changed (ARCH §8.4). let known = load_folder_etags(&catalog, &root); let scanned = dr_sync::scan(&*backend, &RemotePath::new(&root), &filter, &known, |p| { let _ = tx.send(ScanMessage::Progress { directories: p.directories_listed, pruned: p.directories_pruned, images: p.images_found, }); }) .await; // TRACES: FR-PLAT-AND-2 | FR-CAT-9 // Written before the failure is reported, not after: the banner is a // consequence of the catalog state and not the other way round, and a // process that dies between the two must come back knowing. let result = match scanned { Ok(r) => r, Err(e) => { if e.indicates_lost_root() { mark_library_offline(&catalog, &root); } return Err(e.into()); } }; persist(&catalog, &root, &result).map_err(ScanFailure::local)?; // TRACES: FR-CAT-8 | FR-NC-9 // Take in what other devices have judged. After `persist`, because a // judgement lands on an image's version row and the image has to be in // the catalog first — a sidecar seen in the same listing as a // photograph this scan has only just discovered is the ordinary case // on a library another device imported. // // Deliberately not fatal. A pull that fails leaves the ratings this // device already had exactly where they were and the sidecar's ETag // unrecorded, so the next scan tries again; failing the whole scan over // it would blank a grid that was working. let judgements = match pull_sidecars(&*backend, &catalog, &root, &result.sidecars).await { Ok(n) => n, Err(e) => { log::warn!("reading sidecars from the library: {e}"); 0 } }; // Report what the catalog holds, not what this pass listed. An // incremental rescan lists only what changed, so its own count is // near zero on a healthy library. let total = total_images(&catalog).unwrap_or(result.images.len()); let _ = tx.send(ScanMessage::Done { found: result.images.len(), total, pruned: result.progress.directories_pruned, elapsed_ms: started.elapsed().as_millis() as u64, judgements, }); Ok(()) }) } /// TRACES: FR-CAT-8 | FR-CAT-9 | FR-NC-9 /// Take other devices' judgements out of the library's sidecars. /// /// # Why this had to exist /// /// A rating is written to the catalog and to the photograph's sidecar, and the /// sidecar is the authoritative one (ARCH §6.12). Nothing ever read one back. /// The scan indexed files, `derived_sync` exchanged thumbnails, collections /// and keywords, `dr_catalog::merge` reconciled everything in the catalog /// *except* `versions.rating` and `versions.flag`, and the single sidecar /// reader ran when one photograph was opened in develop and handed its result /// to the develop graph. `JobKind::ReadSidecar` had been declared for this /// since the queue was written and was never enqueued or handled. /// /// So judgements travelled outward only. The grid draws `versions.rating`, and /// a cull done on another device could not reach it by any path the app had. /// /// # What it costs /// /// Nothing on a library nobody has edited. The sidecars come from listings the /// walk was making anyway, an unchanged directory is pruned before it is /// listed at all, and a sidecar whose ETag matches what this device last read /// is skipped without a request. What is left is one GET per sidecar that /// genuinely changed — which is the number of photographs somebody edited. /// /// # Why a judgement can only be added, never withdrawn /// /// `Version::merge`'s rule, applied here: zero is *unjudged*, and a device that /// has never rated a frame is indistinguishable from one that deliberately /// cleared it. Taking a remote zero over a local star would let a device that /// was never involved erase an afternoon's culling. So a remote zero is /// ignored, and the cost is that clearing a rating does not propagate. /// /// Returns how many images gained a judgement. async fn pull_sidecars( backend: &dyn RemoteBackend, catalog: &Catalog, root: &str, seen: &[dr_sync::RemoteEntry], ) -> Result { if seen.is_empty() { return Ok(0); } let conn = catalog.connection(); let root_id: i64 = conn .query_row( "SELECT id FROM roots WHERE label = ?1 AND kind = 'remote'", [root], |r| r.get(0), ) .map_err(|e| e.to_string())?; let known = load_sidecar_etags(catalog, root_id); let mut applied = 0usize; for entry in seen { let path = entry.path.as_str(); if known.get(path).is_some_and(|e| *e == entry.validator) { continue; } let bytes = match backend.get(&RemoteId::Path(entry.path.clone()), None).await { Ok(b) => b, // Gone between the listing and the fetch, or not on this device and // not worth materialising a whole library for. Neither is an error, // and neither records an ETag — so the next scan tries again. Err(e) => { log::debug!("reading sidecar {path}: {e}"); continue; } }; let text = String::from_utf8_lossy(&bytes); // TRACES: FR-CAT-13 // A standard XMP beside the photograph — Lightroom's, darktable's, // anybody's — takes the other branch: reconciled field by field with // the catalog winning, and a disagreement recorded for the reload // the requirement asks to be offered. See `xmp_sync`. if crate::xmp_sync::is_xmp(path) { match crate::xmp_sync::take_in(conn, root_id, path, &text, now_secs()) { Ok(taken) => { applied += taken.changed; if !taken.conflicts.is_empty() { log::info!( "xmp sidecar {path} disagrees with the catalog on {:?}; \ a reload is offered in Settings", taken.conflicts ); } // Recorded only once it reached a photograph: a sidecar // that arrived before its image is read again next time. if taken.described > 0 { record_sidecar_read(conn, root_id, path, &entry.validator); } } Err(e) => log::warn!("xmp sidecar at {path} is unreadable ({e})"), } continue; } let mut sidecar = match dr_pipeline::Sidecar::parse(&text) { Ok(s) => s, // Unreadable is not empty. Recording the ETag would mean never // looking at it again, and a build that understands it may be // along; leaving it unrecorded costs one GET per scan and keeps // the door open. Err(e) => { log::warn!("sidecar at {path} is unreadable ({e})"); continue; } }; sidecar.fuse_default_versions(None); let Some(version) = sidecar.default_version() else { // A sidecar with no default version — someone else's virtual copy // and nothing more. Nothing to take, but it *was* read, so its // ETag is recorded and it is not fetched again. record_sidecar_read(conn, root_id, path, &entry.validator); continue; }; match apply_judgement(conn, root_id, path, version.rating, version.flag) { Ok(n) => { applied += n; record_sidecar_read(conn, root_id, path, &entry.validator); } Err(e) => log::debug!("applying {path}: {e}"), } } if applied > 0 { log::info!("{applied} judgement(s) arrived from other devices"); } Ok(applied) } /// TRACES: FR-CAT-13 | NFR-R4 /// One image's ratings, label and keywords, on their way to the `.xmp` /// beside it. /// /// Distinct from [`SidecarWrite`], which amends DarkRoom's own document /// through the cache and the outbox. This is best-effort in the other /// direction: the catalog and the `.drsc` are authoritative, the `.xmp` is a /// courtesy to whatever else reads the folder, and a write that cannot /// happen now is written again — from the catalog, whole — the next time /// anything about the photograph is judged. So nothing is queued. #[derive(Debug, Clone)] pub struct XmpWrite { pub image_path: String, pub record: dr_xmp::Xmp, } /// TRACES: FR-CAT-13 | NFR-R4 /// Write each record into the XMP sidecar beside its image. /// /// An existing sidecar of either spelling is rewritten in place, which is the /// whole point of `dr_xmp::rewrite`: only the properties DarkRoom owns move, /// and another application's settings, comments and namespaces come through /// byte for byte. A photograph with neither gets a new file under Lightroom's /// name. Reported once at the end, as the judgement writes are. pub fn spawn_xmp_writes(conn: Connection, writes: Vec) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let mut written = 0usize; let mut failed = 0usize; let mut last_error = None; let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { let _ = tx.send(XmpMessage::Finished { written: 0, failed: writes.len(), last_error: Some(e.to_string()), }); return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { failed = writes.len(); last_error = Some(e.to_string()); return; } }; for w in &writes { match write_one_xmp(&*backend, w).await { Ok(()) => written += 1, Err(e) => { log::warn!("xmp sidecar for {}: {e}", w.image_path); last_error = Some(e); failed += 1; } } } }); let _ = tx.send(XmpMessage::Finished { written, failed, last_error, }); }); rx } /// The outcome of a batch of XMP writes. #[derive(Debug)] pub enum XmpMessage { Finished { written: usize, failed: usize, last_error: Option, }, } /// Read-modify-write one image's XMP sidecar on the server. async fn write_one_xmp(backend: &dyn RemoteBackend, w: &XmpWrite) -> Result<(), String> { let [darktable, lightroom] = crate::xmp_sync::candidate_paths(&w.image_path); // Whichever exists is the one rewritten; neither existing means the // Lightroom spelling is created. An existing file this build cannot // parse is left alone rather than replaced — it is somebody else's // document, and a refusal is recoverable where an overwrite is not. let mut target = lightroom.clone(); let mut existing: Option = None; for path in [&darktable, &lightroom] { if let Ok(bytes) = backend .get(&RemoteId::Path(RemotePath::new(path.clone())), None) .await { if !bytes.is_empty() { target = path.clone(); existing = Some(String::from_utf8_lossy(&bytes).into_owned()); break; } } } let text = match existing { Some(text) => { // The file's caption, copyright and hierarchy come through: the // catalog has nowhere to keep them, and a rewrite that said // nothing about them would remove them. let theirs = dr_xmp::Xmp::parse(&text) .map_err(|e| format!("{target} is not a sidecar this build can read: {e}"))?; let mut record = w.record.clone(); crate::xmp_sync::carry_through(&mut record, &theirs); record .rewrite(&text) .map_err(|e| format!("{target} is not a sidecar this build can rewrite: {e}"))? } None => w.record.to_text(), }; backend .put(&RemotePath::new(target), text.into_bytes(), None) .await .map(|_| ()) .map_err(|e| e.to_string()) } /// TRACES: FR-CAT-13 /// The offered reload: re-read the sidecars a person chose to trust, with /// the sidecar winning. One fetch per path, the catalog opened on this /// thread as the scan opens it. pub fn spawn_xmp_reload( conn: Connection, root: String, catalog_path: PathBuf, paths: Vec, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let mut written = 0usize; let mut failed = 0usize; let mut last_error = None; let outcome: Result<(), String> = (|| { let catalog = Catalog::open(&catalog_path).map_err(|e| e.to_string())?; let root_id: i64 = catalog .connection() .query_row( "SELECT id FROM roots WHERE label = ?1 AND kind = 'remote'", [&root], |r| r.get(0), ) .map_err(|e| e.to_string())?; let rt = crate::net_runtime::build().map_err(|e| e.to_string())?; rt.block_on(async { let backend = crate::remote::connect(&conn).map_err(|e| e.to_string())?; for path in &paths { let fetched = backend .get(&RemoteId::Path(RemotePath::new(path.clone())), None) .await .map_err(|e| e.to_string()) .and_then(|bytes| { let text = String::from_utf8_lossy(&bytes); crate::xmp_sync::reload(catalog.connection(), root_id, path, &text) }); match fetched { Ok(taken) => written += taken.changed, Err(e) => { log::warn!("reloading {path}: {e}"); last_error = Some(e); failed += 1; } } } Ok(()) }) })(); if let Err(e) = outcome { failed = paths.len(); last_error = Some(e); } let _ = tx.send(XmpMessage::Finished { written, failed, last_error, }); }); rx } /// What this device has already read, so a pull fetches only what changed. fn load_sidecar_etags( catalog: &Catalog, root_id: i64, ) -> std::collections::HashMap { let mut out = std::collections::HashMap::new(); let Ok(mut stmt) = catalog .connection() .prepare("SELECT path, etag FROM sidecars WHERE root_id = ?1 AND etag IS NOT NULL") else { return out; }; if let Ok(rows) = stmt.query_map([root_id], |r| { Ok((r.get::<_, String>(0)?, r.get::<_, String>(1)?)) }) { for (path, etag) in rows.flatten() { out.insert(path, dr_sync::Validator::new(etag)); } } out } /// Remember that this sidecar has been taken in at this ETag. /// /// Written only after the judgement has landed, for the same reason /// `dr_sync::scan` records a directory's ETag only after listing it: recording /// it first would let a failure look like work already done, and the edit would /// never be read again. fn record_sidecar_read( conn: &rusqlite::Connection, root_id: i64, path: &str, etag: &dr_sync::Validator, ) { let done = conn.execute( "INSERT INTO sidecars(root_id, path, etag, read_at) VALUES (?1, ?2, ?3, ?4) ON CONFLICT(root_id, path) DO UPDATE SET etag = excluded.etag, read_at = excluded.read_at", rusqlite::params![root_id, path, etag.as_str(), now_secs()], ); if let Err(e) = done { log::debug!("recording sidecar {path}: {e}"); } } /// Apply one sidecar's judgement to the image or images it describes. /// /// # Why more than one image /// /// A sidecar is named for the stem it shares with its photograph, so a RAW and /// the JPEG the camera wrote beside it — one photograph under FR-CAT-11 — share /// a document, and both rows have to carry the judgement or the grid disagrees /// with itself depending on which of the pair it is showing. /// /// # Why the match is verified in Rust /// /// The `LIKE` narrows the search to rows sharing the stem, and it is only a /// filter: `library::sidecar_path` is what actually decides, applied to each /// candidate. A path holding a `%`, a `_` or a bracket would otherwise match /// more than it should, and a judgement landing on the wrong photograph is a /// silent, permanent wrong. /// /// Returns how many images gained a judgement they did not have. fn apply_judgement( conn: &rusqlite::Connection, root_id: i64, sidecar: &str, rating: u8, flag: u8, ) -> Result { let stem = sidecar .rsplit_once('.') .map(|(s, _)| s) .unwrap_or(sidecar) .to_string(); // The escape is the point: `%` and `_` are wildcards, and a photographer's // folder is entitled to contain both. let prefix = stem .replace('\\', "\\\\") .replace('%', "\\%") .replace('_', "\\_"); let candidates: Vec<(i64, String)> = { let mut stmt = conn .prepare( "SELECT id, source_ref FROM images WHERE root_id = ?1 AND source_ref LIKE ?2 ESCAPE '\\'", ) .map_err(|e| e.to_string())?; let rows = stmt .query_map(rusqlite::params![root_id, format!("{prefix}.%")], |r| { Ok((r.get(0)?, r.get(1)?)) }) .map_err(|e| e.to_string())?; rows.filter_map(Result::ok) .filter(|(_, source): &(i64, String)| sidecar_path(source) == sidecar) .collect() }; let mut applied = 0usize; for (image, _) in candidates { let id = dr_types::ImageId(image as u64); let version = dr_catalog::rating::default_version_id(conn, id).map_err(|e| e.to_string())?; // The sidecar's value is taken, not the larger of the two. It is the // authoritative store and the fuse above has already resolved any // contest between devices on `revision` — so lowering a rating from // four to one on the tablet has to lower it here, and taking a maximum // would quietly refuse every demotion the photographer ever made. // // A zero is the one thing not taken: it means *never judged* rather // than "judged zero", so a sidecar that carries none cannot erase a // star this device holds. The same asymmetry, and the same direction // of caution, as `dr_pipeline::sidecar::merge_judgement`. The cost is // the one that rule always carries — clearing a rating does not // propagate. let changed = conn .execute( "UPDATE versions SET rating = CASE WHEN ?2 > 0 THEN ?2 ELSE rating END, flag = CASE WHEN ?3 > 0 THEN ?3 ELSE flag END WHERE id = ?1 AND ((?2 > 0 AND rating <> ?2) OR (?3 > 0 AND flag <> ?3))", rusqlite::params![version, rating.min(5) as i64, flag.min(2) as i64], ) .map_err(|e| e.to_string())?; applied += changed; } Ok(applied) } /// Read back the folder ETags stored by a previous scan. /// /// A failure here is not fatal — an empty map simply means no pruning, which /// is correct but slower. Refusing to scan because the last scan's bookkeeping /// is unreadable would be the worse outcome. fn load_folder_etags( catalog: &Catalog, root: &str, ) -> std::collections::HashMap { let mut out = std::collections::HashMap::new(); let sql = "SELECT f.path, f.etag FROM folders f JOIN roots r ON r.id = f.root_id WHERE r.label = ?1 AND f.etag IS NOT NULL"; let Ok(mut stmt) = catalog.connection().prepare(sql) else { return out; }; let rows = stmt.query_map([root], |r| { Ok((r.get::<_, String>(0)?, r.get::<_, String>(1)?)) }); if let Ok(rows) = rows { for (path, etag) in rows.flatten() { out.insert(RemotePath::new(path), dr_sync::Validator::new(etag)); } } out } /// Write a scan's findings into the catalog. /// /// Images insert at `metadata_state = 1` (stat-only): the scan knows name and /// size but has read no EXIF, and pretending otherwise would make a date /// filter silently wrong. A `Thumbnail` job is enqueued per image, coalescing /// with anything already pending. fn persist( catalog: &Catalog, root: &str, result: &dr_sync::ScanResult, ) -> Result<(), dr_catalog::CatalogError> { let conn = catalog.connection(); let tx = conn.unchecked_transaction()?; // One root row per library folder, reused across scans. tx.execute( "INSERT INTO roots(kind, label, last_seen) VALUES ('remote', ?1, ?2) ON CONFLICT DO NOTHING", rusqlite::params![root, now_secs()], )?; let root_id: i64 = tx.query_row( "SELECT id FROM roots WHERE label = ?1 AND kind = 'remote'", [root], |r| r.get(0), )?; // Folder ETags first — without these persisted, the next scan prunes // nothing and walks the whole tree again (ARCH §6.6). for (path, validator) in &result.directories { tx.execute( "INSERT INTO folders(root_id, path, etag) VALUES (?1, ?2, ?3) ON CONFLICT(root_id, path) DO UPDATE SET etag = excluded.etag", rusqlite::params![root_id, path.as_str(), validator.as_str()], )?; } for entry in &result.images { let folder_id: Option = entry.path.parent().and_then(|p| { tx.query_row( "SELECT id FROM folders WHERE root_id = ?1 AND path = ?2", rusqlite::params![root_id, p.as_str()], |r| r.get(0), ) .ok() }); // TRACES: FR-PLAT-AND-2 | FR-CAT-9 // The `availability` arm is what ends an offline library, and it does // it one photograph at a time. 3 is `Availability::Offline` and 0 is // `MetadataOnly`, the same code this statement inserts new rows with — // so a row that was marked offline when the root became unreachable is // returned to exactly the state a fresh scan would have given it, and // a row that was never marked is not touched at all. // // Conditional rather than a blanket reset for the same reason // `dr_catalog::walk` restores per file rather than per root: the only // thing that may clear "I could not reach this" is having reached it, // and this statement runs precisely once per file the scan listed. tx.execute( "INSERT INTO images(root_id, folder_id, source_ref, format, file_size, availability, metadata_state, added_at) VALUES (?1, ?2, ?3, ?4, ?5, 0, 1, ?6) ON CONFLICT(root_id, source_ref) DO UPDATE SET file_size = excluded.file_size, folder_id = excluded.folder_id, availability = CASE WHEN images.availability = 3 THEN 0 ELSE images.availability END", rusqlite::params![ root_id, folder_id, entry.path.as_str(), entry .path .name() .rsplit_once('.') .map(|(_, e)| e.to_ascii_lowercase()), entry.size as i64, now_secs(), ], )?; let image_id: i64 = tx.query_row( "SELECT id FROM images WHERE root_id = ?1 AND source_ref = ?2", rusqlite::params![root_id, entry.path.as_str()], |r| r.get(0), )?; // Remote identity, keyed on oc:fileid so a server-side move is a move // rather than a re-download (FR-NC-5). if let RemoteId::Stable(file_id) = entry.id { tx.execute( "INSERT INTO remote(image_id, file_id, etag, remote_path) VALUES (?1, ?2, ?3, ?4) ON CONFLICT(image_id) DO UPDATE SET etag = excluded.etag, remote_path = excluded.remote_path", rusqlite::params![ image_id, file_id as i64, entry.validator.as_str(), entry.path.as_str() ], )?; } } tx.commit()?; // Thumbnail jobs after the commit, so a failure mid-insert does not leave // jobs pointing at rows that never landed. for entry in &result.images { if let Ok(image_id) = conn.query_row( "SELECT id FROM images WHERE root_id = ?1 AND source_ref = ?2", rusqlite::params![root_id, entry.path.as_str()], |r| r.get::<_, i64>(0), ) { let _ = dr_catalog::jobs::enqueue( conn, JobKind::Thumbnail, Some(image_id), Priority::Background, None, ); } } Ok(()) } /// What the grid wants a thumbnail for. /// /// Carries the `oc:fileid` as well as the path, because that is what the /// shared store keys on — stable across a server-side move, and the same id /// every other client sees (FR-NC-5). #[derive(Debug, Clone)] pub struct ThumbnailRequest { pub row: usize, pub path: String, /// `None` where the scan found no stable id; such an image is fetched but /// not stored, since there is no durable key to store it under. pub file_id: Option, /// File length, needed to reject a preview range that points past the end /// of the file (NFR-SEC-1). pub size: u64, /// Catalog row, so EXIF read from the header can be written back. pub image_id: i64, /// Which resolution this cell needs, from how large it is drawn. A zoomed /// grid asks for the large class; a wall of small cells does not. /// /// Named apart from `size`, which is the file's length in bytes — the two /// are unrelated and confusing them would fetch the wrong thing. pub thumb_size: dr_thumbs::ThumbSize, /// Whether this image still needs its EXIF read. Where false the header is /// still fetched — the preview needs it — but nothing is parsed or written. pub needs_metadata: bool, /// Keep the preview at the resolution it was decoded at, ignoring /// `thumb_size`. /// /// For face indexing, which wants the pixels a thumbnail throws away: a /// face 2% across the frame is 5 px on a grid thumbnail and 120 px on the /// embedded preview, and 112 is what the embedder samples. Capped by /// [`FACE_SOURCE_EDGE`] rather than truly unbounded, because a 24 MP buffer /// converted to `f32` RGB is ~288 MB and several lanes hold one at once. pub full_resolution: bool, } /// Why a full fetch failed, keeping the one bit the UI cannot re-derive. /// /// The same reasoning as [`ScanFailure`]: the typed error cannot cross the /// channel, and "offline" versus "refused" decides whether develop shows /// "you are offline — this image is not stored locally" or a real error. #[derive(Debug)] pub struct FetchFailure { pub message: String, pub offline: bool, } impl FetchFailure { fn local(message: impl std::fmt::Display) -> Self { Self { message: message.to_string(), offline: false, } } } impl From for FetchFailure { fn from(e: dr_sync::RemoteError) -> Self { Self { offline: e.indicates_offline(), message: e.to_string(), } } } impl std::fmt::Display for FetchFailure { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { f.write_str(&self.message) } } /// TRACES: FR-NC-6a /// Progress from the pin worker. #[derive(Debug)] pub enum PinMessage { /// How many originals the pin still needs. Sent once, before any transfer. Planned { total: usize, }, /// One original landed. Stored { done: usize, }, /// The pin is fully downloaded. Done { stored: usize, bytes: u64, }, Failed { message: String, offline: bool, }, } /// TRACES: FR-NC-6a /// Download every original a pin has asked for. /// /// Whole files, deliberately: a pin exists so the photographs can be *edited* /// away from the server, and develop needs every photosite. This is the one /// place in the app that fetches originals in bulk, which is why FR-NC-6 /// makes it opt-in rather than something sync does on its own. /// /// Sequential rather than parallel. The lanes that make the thumbnail sweep /// fast are wrong here: these are tens of megabytes each, so concurrency buys /// little against a single connection's bandwidth and costs a great deal of /// memory — and it is the same contention that produced 423 Locked in the /// sweep. pub fn spawn_pin_fetch( conn: Connection, catalog_path: PathBuf, cache_dir: PathBuf, budget: dr_catalog::Budget, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let (store, catalog) = match ( dr_catalog::Cache::open(&cache_dir, budget), Catalog::open(&catalog_path), ) { (Ok(s), Ok(c)) => (s, c), (Err(e), _) | (_, Err(e)) => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: false, }); return; } }; let pending = match store.pending_pins(catalog.connection()) { Ok(p) => p, Err(e) => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: false, }); return; } }; if tx .send(PinMessage::Planned { total: pending.len(), }) .is_err() { return; } if pending.is_empty() { let _ = tx.send(PinMessage::Done { stored: 0, bytes: 0, }); return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: false, }); return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: false, }); return; } }; let mut stored = 0usize; let mut bytes_total = 0u64; for image in pending { let Some(source_ref) = source_ref_of(&catalog, image) else { // Catalogued and then removed while the pin was pending. continue; }; let id = RemoteId::Path(RemotePath::new(&source_ref)); // TRACES: FR-NC-6c // On a placeholder library "pin" means *keep it downloaded*, // not "make a second copy". The original materialises in the // library folder itself, so copying it under `originals/` // would hold every pinned photograph twice — and the copy // would be the half the budget could evict while the real disk // cost stayed. Only the bookkeeping is recorded, with no path, // so nothing here can ever delete a file inside a synced tree // (see `Cache::record_in_place`). if backend.capabilities().materialisation.can_materialise() { match backend.materialise(&id).await { Ok(_) => { let bytes = size_of(&catalog, image).unwrap_or(0); if let Err(e) = store.record_in_place( catalog.connection(), image, bytes, true, now_secs(), ) { log::warn!("recording pinned {source_ref}: {e}"); continue; } stored += 1; bytes_total += bytes; if tx.send(PinMessage::Stored { done: stored }).is_err() { return; } } Err(e) if e.indicates_offline() => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: true, }); return; } Err(e) => log::warn!("pinning {source_ref}: {e}"), } continue; } match backend.get(&id, None).await { Ok(bytes) => { // `pinned: true` — this is the population the budget // must never evict, which is the entire promise the // user made when they pinned the collection. if let Err(e) = store.store( catalog.connection(), image, &source_ref, &bytes, true, now_secs(), ) { log::warn!("storing pinned {source_ref}: {e}"); continue; } stored += 1; bytes_total += bytes.len() as u64; if tx.send(PinMessage::Stored { done: stored }).is_err() { return; } } Err(e) if e.indicates_offline() => { // Stop rather than failing each remaining file against // a dead connection. What was downloaded stays // downloaded, and `pending_pins` resumes from there. let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: true, }); return; } Err(e) => { // One unreadable file must not abandon the whole pin. log::warn!("pinning {source_ref}: {e}"); } } } let _ = tx.send(PinMessage::Done { stored, bytes: bytes_total, }); }); }); rx } /// TRACES: FR-NC-6c /// Hand a set of photographs back to the sync client, freeing their disk. /// /// The other half of pinning on a placeholder library. `Cache::release` drops /// the bookkeeping and — correctly — deletes nothing, because the rows it /// holds for a library like this name no file of ours (`record_in_place`). /// The bytes are in the library folder, and only the client may take them /// back. /// /// **This is a dehydration, not a deletion, and the distinction is the whole /// safety of the feature.** Removing a materialised file inside a synced tree /// propagates to the server and deletes the photograph everywhere. /// /// Best effort per image: a file the client refuses to release simply stays, /// which costs disk and loses nothing. pub fn spawn_dehydrate( conn: Connection, catalog_path: PathBuf, images: Vec, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let Ok(catalog) = Catalog::open(&catalog_path) else { return; }; let Ok(rt) = crate::net_runtime::build() else { return; }; rt.block_on(async { let Ok(backend) = crate::remote::connect(&conn) else { return; }; // Nothing to do where content is not a thing that can be given // back — a server library, or a plain folder. if !backend.capabilities().materialisation.can_materialise() { return; } let mut released = 0usize; for image in images { let Some(source_ref) = source_ref_of(&catalog, image) else { continue; }; let id = RemoteId::Path(RemotePath::new(&source_ref)); match backend.dematerialise(&id).await { Ok(()) => released += 1, Err(e) => log::debug!("releasing {source_ref}: {e}"), } } log::info!("released {released} photograph(s) back to the sync client"); let _ = tx.send(released); }); }); rx } /// What an image occupies, as the catalog recorded it. /// /// Zero where the scan could not tell — a placeholder reports no size, because /// a one-byte stub says nothing about what it stands for (ARCH §9.0a). A pin /// that cannot state its cost is better than one that states a wrong one. fn size_of(catalog: &Catalog, image: dr_types::ImageId) -> Option { catalog .connection() .query_row( "SELECT file_size FROM images WHERE id = ?1", rusqlite::params![image.0 as i64], |r| r.get::<_, Option>(0), ) .ok() .flatten() .map(|v| v.max(0) as u64) } /// The remote path for a catalogued image. fn source_ref_of(catalog: &Catalog, image: dr_types::ImageId) -> Option { catalog .connection() .query_row( "SELECT source_ref FROM images WHERE id = ?1", rusqlite::params![image.0 as i64], |r| r.get(0), ) .ok() } /// TRACES: FR-NC-6a | FR-CAT-9 /// Where a cached original is kept and how much may be kept. /// /// Passed in rather than derived here so the caller owns the policy: the /// budget is a user setting, and this function is on a worker thread with no /// access to one. pub struct CacheContext { pub dir: PathBuf, pub catalog_path: PathBuf, pub image: dr_types::ImageId, pub budget: dr_catalog::Budget, /// Whether a downloaded original is kept. /// /// Only the write. A cache is always *read*, because bytes already on disk /// cost nothing to use and declining them would re-download an image that /// is present — including every pinned one, which would leave a pinned /// collection unopenable offline the moment this was switched off. pub store: bool, } /// TRACES: FR-CAT-8 | FR-DEV-6 /// Fetch and parse the sidecar beside one image. /// /// # Why the edit is read from the file rather than the catalog /// /// The catalog carries a `graph_hash` and no parameters, and it is /// *disposable* (ARCH §6.12) — a rebuild would silently return every /// photograph to neutral. The sidecar is the authoritative store, so it is /// what an open reads, and that is also what makes an edit pasted on the /// desktop appear when the same frame is opened on the phone. /// /// # Why absence and failure are the same answer here /// /// `None` means "open this image at its defaults", which is right for a /// photograph that has never been edited — the overwhelmingly common case on a /// fresh library — and equally right when the network is down. The alternative, /// refusing to open the image because its sidecar could not be read, would make /// an unreachable server also mean an unviewable library. /// /// The one case that is *not* harmless is a sidecar that exists but does not /// parse. That still opens at defaults, but the write path /// ([`write_one_sidecar`]) independently refuses to overwrite a file it could /// not read, so an edit this build failed to understand is never destroyed by /// having been opened. pub fn spawn_sidecar_fetch( conn: Connection, image_path: String, cache_dir: PathBuf, offline: bool, ) -> Receiver> { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let cache = SidecarCache::open(cache_dir); let path_str = sidecar_path(&image_path); // TRACES: FR-CAT-9 | FR-NC-10 // The cache wins outright when it is holding work the server has not // seen. Fetching in that state would answer with a document *older* // than the edit sitting in the outbox, and opening the photograph // would silently show it without the change the user just made — // which the next save would then write back over the top of. if cache.is_pending(&path_str) { log::debug!("{path_str} has queued local edits; opening from the cache"); let _ = tx.send(cache.load(&path_str)); return; } // Offline there is nothing to ask, and the cache is the whole answer. let rt = if offline { None } else { match crate::net_runtime::build() { Ok(e) => Some(e), Err(e) => { log::debug!("sidecar fetch runtime: {e}"); None } } }; let Some(rt) = rt else { let _ = tx.send(cache.load(&path_str)); return; }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { log::debug!("sidecar fetch backend: {e}"); let _ = tx.send(cache.load(&path_str)); return; } }; let path = RemotePath::new(path_str.clone()); let id = RemoteId::Path(path.clone()); // A 404 is the normal case on a library that has never been // edited, so this is `ok()` rather than an error path. let Ok(bytes) = backend.get(&id, None).await else { // Unreachable, or no such file. The cache cannot tell those // apart and does not need to: either way it holds the best // answer this device has. let _ = tx.send(cache.load(&path_str)); return; }; let text = String::from_utf8_lossy(&bytes).into_owned(); let parsed = match dr_pipeline::Sidecar::parse(&text) { Ok(mut s) => { // TRACES: FR-NC-8 | FR-NC-9 // Opening a photograph must show everything that has been // done to it, not whichever of two split default versions // happens to win. Fused in memory with no canonical uuid // to impose — this is a read, and the write path is where // the identity is decided. // // The fused document is what gets cached below, so the // next offline open sees the union too. s.fuse_default_versions(None); Some(s) } Err(e) => { log::warn!("sidecar at {} is unreadable ({e})", path.as_str()); None } }; // Populate the cache from what the server said, so the *next* // open of this photograph works with no connection. Clean rather // than pending: this content came from the server, so there is // nothing to send back. if let Some(sidecar) = parsed.as_ref() { if let Err(e) = cache.store(&path_str, sidecar, false) { log::debug!("caching {path_str}: {e}"); } } let _ = tx.send(parsed); }); }); rx } /// Fetch one file in full, for opening it in develop. /// /// Deliberately *not* the preview path. Browsing fetches a range and decodes /// an embedded JPEG (FR-NC-3); develop needs every byte, because demosaic /// needs every photosite. On a RAW file that is tens of megabytes, which is /// why this is a click-triggered download and not something the grid does. /// /// # Read-through /// /// With a `cache`, this checks disk before the network and stores what it /// downloads. That is what makes opening the same photograph twice cost one /// transfer, and what leaves a working session's images openable offline /// without anyone having pinned anything. /// /// A cache miss is not an error and a cache failure is not fatal: both fall /// through to the network, which is exactly the behaviour that existed before /// the cache did. /// /// Returns the bytes on a channel rather than blocking: the download runs on /// its own thread and the UI stays live, exactly as thumbnail fetching does. /// The work itself is [`fetch_original`], which is also what the /// [`Prefetcher`] runs — one transfer path, so a photograph fetched ahead is /// stored exactly as one fetched on a click. pub fn spawn_full_fetch( conn: Connection, path: String, cache: Option, ) -> Receiver, FetchFailure>> { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let _ = tx.send(fetch_original(conn, &path, cache.as_ref())); }); rx } /// TRACES: FR-NC-6a /// Every original transfer that is under way right now, by remote path. /// /// **One file, one transfer.** The [`Prefetcher`] fetches the photographs /// beside the open one before they are asked for, and the whole point is that /// the user then asks for one of them — often while it is still coming down. /// Without this the click would miss the cache, start a second download of /// the same file, and the two would halve each other's bandwidth for the rest /// of the transfer. With it, the click finds the path claimed, waits for the /// prefetch to store its bytes, and reads them from disk. /// /// Process-wide rather than passed in, because the property it enforces is /// process-wide: there is no caller for whom two concurrent downloads of one /// file is the right answer. Keyed on the path rather than the image id /// because that is the one name every caller has. static IN_FLIGHT: std::sync::LazyLock = std::sync::LazyLock::new(InFlight::default); #[derive(Default)] struct InFlight { busy: std::sync::Mutex>, freed: std::sync::Condvar, } impl InFlight { /// Take `path` for this thread, or wait for whoever holds it. /// /// `Some` is a claim, released when the guard drops. `None` means another /// thread held the path and has now let it go — so the caller's cache /// check is worth repeating, because that thread has very probably just /// stored what the caller was about to download. fn claim(&self, path: &str) -> Option> { let mut busy = self.busy.lock().unwrap_or_else(|e| e.into_inner()); if busy.insert(path.to_string()) { return Some(InFlightGuard { of: self, path: path.to_string(), }); } while busy.contains(path) { busy = self.freed.wait(busy).unwrap_or_else(|e| e.into_inner()); } None } fn release(&self, path: &str) { self.busy .lock() .unwrap_or_else(|e| e.into_inner()) .remove(path); self.freed.notify_all(); } } /// A claim on a path, dropped on every exit from the fetch — a failed /// download must free the path too, or the click waiting on it never wakes. struct InFlightGuard<'a> { of: &'a InFlight, path: String, } impl Drop for InFlightGuard<'_> { fn drop(&mut self) { self.of.release(&self.path); } } /// The blocking body of [`spawn_full_fetch`]: cache, then in-flight registry, /// then network, storing what it downloads when the cache says to. fn fetch_original( conn: Connection, path: &str, cache: Option<&CacheContext>, ) -> Result, FetchFailure> { // Opened on this thread: `rusqlite::Connection` is not `Send`, and the // UI thread's handle cannot be borrowed across the spawn. let cached = cache.and_then(|c| { let store = dr_catalog::Cache::open(&c.dir, c.budget).ok()?; let conn = Catalog::open(&c.catalog_path).ok()?; Some((store, conn)) }); let from_cache = || { let (c, (store, conn)) = (cache?, cached.as_ref()?); match store.load(conn.connection(), c.image, now_secs()) { Ok(Some(bytes)) => { log::info!("{path}: {} bytes from the local cache", bytes.len()); Some(bytes) } Ok(None) => None, // A cache that cannot be read is a cache miss, not a failure // to open the photograph. Err(e) => { log::debug!("cache lookup for {path}: {e}"); None } } }; // Miss, claim, and if the claim had to wait, look again: the thread that // held the path has finished with it, and what it fetched is on disk. let _claim = loop { if let Some(bytes) = from_cache() { return Ok(bytes); } if let Some(claim) = IN_FLIGHT.claim(path) { break claim; } }; let rt = crate::net_runtime::build().map_err(FetchFailure::local)?; rt.block_on(async { let backend = crate::remote::connect(&conn).map_err(FetchFailure::local)?; let id = RemoteId::Path(RemotePath::new(path)); let bytes = backend.get(&id, None).await?; // Store before returning, so the bytes are on disk by the time the // image is on screen. Doing it after would leave a window where // closing the app immediately lost the download. if let (Some(c), Some((store, conn))) = (cache.filter(|c| c.store), cached.as_ref()) { // `pinned: false` — this is the passive population. A pin is // something the user asks for explicitly; opening an image is // not that, and treating it as one would make the pinned set // grow silently and never be evicted. if let Err(e) = store.store(conn.connection(), c.image, path, &bytes, false, now_secs()) { log::debug!("caching {path}: {e}"); } else if let Err(e) = store.enforce(conn.connection()) { log::debug!("enforcing the cache budget: {e}"); } } Ok(bytes) }) } /// TRACES: FR-NC-6a | FR-UI-4 /// One original to fetch ahead of its being asked for. pub struct PrefetchJob { pub path: String, pub cache: CacheContext, } /// What the [`Prefetcher`] is doing, for the activity list. pub enum PrefetchEvent { /// A transfer has started for this path. Started(String), /// And has ended — stored, or not; either way the row can go. Ended(String), } /// TRACES: FR-NC-6a | FR-UI-4 /// Fetches the photographs beside the open one into the cache, ahead of the /// step that asks for them. /// /// **Why this exists.** Walking the photo roll is one click per frame, and /// without this every click is a download of tens of megabytes with a /// "Downloading…" line over an empty canvas. A photographer moving between a /// pair of near-identical frames does that a dozen times. Fetching the two /// neighbours while the current photograph is being looked at turns the next /// step into a disk read, which is what makes stepping feel like stepping. /// /// **One worker, one wish.** A single thread serves the *latest* request and /// nothing older. Each open replaces the previous wish outright, so a fast /// walk along the roll does not leave a trail of stale downloads competing /// with the one the user is actually waiting on; a job already under way is /// finished rather than abandoned, because the bytes are mostly here. Jobs /// run in the order given — next before previous, since that is the way a /// roll is mostly walked — and one at a time, so two neighbours never halve /// each other's bandwidth. /// /// **What it never does.** It never fetches into a cache that would not keep /// the bytes: the caller only hands it jobs whose cache stores, because a /// prefetch that is discarded on arrival is pure transfer for nothing — and /// "keep opened originals" being off is the user saying this device is /// metered or small (FR-NC-6). It never starts while offline, for the same /// reason. And it fetches only the immediate neighbours: originals are /// "explicit pin or on-demand open only", and ±1 is as far as "on demand" /// honestly stretches. pub struct Prefetcher { shared: std::sync::Arc, events: Receiver, } #[derive(Default)] struct PrefetchShared { wanted: std::sync::Mutex, changed: std::sync::Condvar, } /// The latest wish, and a generation so the worker can tell it has been /// replaced mid-list. #[derive(Default)] struct Wanted { generation: u64, conn: Option, jobs: Vec, } impl Wanted { /// Replace whatever was wanted with `jobs`. fn replace(&mut self, conn: Connection, jobs: Vec) { self.generation += 1; self.conn = Some(conn); self.jobs = jobs; } /// Whether a wish taken at `generation` is still the current one. fn is_current(&self, generation: u64) -> bool { self.generation == generation } } impl Default for Prefetcher { fn default() -> Self { Self::new() } } impl Prefetcher { /// Start the worker. It sleeps until the first [`Self::want`]. pub fn new() -> Self { let shared = std::sync::Arc::new(PrefetchShared::default()); let (tx, events) = std::sync::mpsc::channel(); let worker = shared.clone(); std::thread::Builder::new() .name("prefetch".into()) .spawn(move || serve_prefetches(&worker, &tx)) .expect("spawning the prefetch worker"); Self { shared, events } } /// Fetch these, in this order, instead of whatever was asked for before. /// /// An empty list is a valid wish: it cancels the rest of the previous /// one, and is what a photograph with no neighbours in the window asks. pub fn want(&self, conn: Connection, jobs: Vec) { self.shared .wanted .lock() .unwrap_or_else(|e| e.into_inner()) .replace(conn, jobs); self.shared.changed.notify_one(); } /// Everything the worker has reported since the last poll. pub fn poll(&self) -> Vec { std::iter::from_fn(|| self.events.try_recv().ok()).collect() } } /// The worker: take the current wish, serve it job by job, stop the moment it /// is superseded, sleep until the next one. fn serve_prefetches(shared: &PrefetchShared, events: &Sender) { loop { let (generation, conn, jobs) = { let mut wanted = shared.wanted.lock().unwrap_or_else(|e| e.into_inner()); while wanted.jobs.is_empty() { wanted = shared .changed .wait(wanted) .unwrap_or_else(|e| e.into_inner()); } let jobs = std::mem::take(&mut wanted.jobs); let Some(conn) = wanted.conn.clone() else { continue; }; (wanted.generation, conn, jobs) }; for job in jobs { let current = shared .wanted .lock() .unwrap_or_else(|e| e.into_inner()) .is_current(generation); if !current { break; } if holds_original(&job.cache) { continue; } // A closed channel means the window is gone: nothing to fetch // for any more. if events .send(PrefetchEvent::Started(job.path.clone())) .is_err() { return; } match fetch_original(conn.clone(), &job.path, Some(&job.cache)) { Ok(bytes) => log::info!("{}: {} bytes fetched ahead", job.path, bytes.len()), // Not a failure anyone needs to hear about now: the click // that wants this photograph will try again and say so. Err(e) => log::debug!("fetching {} ahead: {}", job.path, e.message), } if events.send(PrefetchEvent::Ended(job.path)).is_err() { return; } } } } /// Whether the cache already has this original — a row check, not a read, /// so asking costs nothing and touches no `last_used`. fn holds_original(cache: &CacheContext) -> bool { let Ok(store) = dr_catalog::Cache::open(&cache.dir, cache.budget) else { return false; }; let Ok(catalog) = Catalog::open(&cache.catalog_path) else { return false; }; store.holds_original(catalog.connection(), cache.image) } /// Serve thumbnails for a set of rows: store first, network second. /// /// The store is consulted before any request goes out, so a second launch — /// or a second device that synced the shards — fills the grid with no transfer /// at all. Only genuine misses reach the network. pub fn spawn_thumbnails( conn: Connection, wanted: Vec, store_dir: PathBuf, catalog_path: PathBuf, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let mut store = match ThumbStore::open(&store_dir) { Ok(s) => Some(s), Err(e) => { // A broken store costs speed, never correctness — every // thumbnail can still be fetched. log::warn!("thumbnail store unavailable, fetching everything: {e}"); None } }; // Split the batch before delivering anything, so the plan can be // reported first and the UI knows the shape of the work up front. // Decoding happens here rather than in the split, because a corrupt // blob turns a hit into a miss. let mut hits = Vec::new(); let mut to_fetch = Vec::new(); // Images whose thumbnail is cached but whose date is still unknown. // // These need a header read even though no pixels are wanted. Without // this pass an image is dated *only* on the one visit that produced // its thumbnail — so a library browsed once before the EXIF code // existed, or synced from another device's shards, stays permanently // undated and never appears on the timeline. let mut metadata_only = Vec::new(); for req in wanted { let stored = req .file_id .zip(store.as_ref()) .and_then(|(id, s)| s.get(id, req.thumb_size).ok().flatten()); match stored.map(|t| dr_thumbs::decode_rgba(&t.bytes)) { Some(Ok((width, height, rgba))) => { if req.needs_metadata { metadata_only.push(req.clone()); } hits.push(ThumbnailReady { row: req.row, width, height, rgba, from_cache: true, }); } // A corrupt stored blob is a miss, not a failure. Some(Err(e)) => { log::debug!("stored thumbnail unreadable, refetching: {e}"); to_fetch.push(req); } None => to_fetch.push(req), } } log::info!( "thumbnails: {} from store, {} to fetch{}", hits.len(), to_fetch.len(), if metadata_only.is_empty() { String::new() } else { format!(" · {} dates to read", metadata_only.len()) } ); if tx .send(ThumbnailMessage::Plan { cached: hits.len(), fetching: to_fetch.len(), dating: metadata_only.len(), }) .is_err() { return; } for hit in hits { if tx.send(ThumbnailMessage::Ready(Box::new(hit))).is_err() { return; } } if to_fetch.is_empty() && metadata_only.is_empty() { return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { for req in &to_fetch { let _ = tx.send(ThumbnailMessage::Unavailable { row: req.row, reason: e.to_string(), }); } return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { for req in &to_fetch { let _ = tx.send(ThumbnailMessage::Unavailable { row: req.row, reason: e.to_string(), }); } return; } }; // Batched rather than written per image: one transaction per // batch instead of 120, and the grid does not need each date the // instant it is read. let mut found = Vec::new(); // Set when the server proves unreachable, which abandons the rest // of the batch. The remaining cells would each take a full timeout // to reach the same conclusion — on a 120-cell window, minutes of // the grid appearing to load against a server that is not there. let mut offline = false; for req in to_fetch { let msg = fetch_one(&*backend, store.as_mut(), &req, &mut found).await; offline = matches!(msg, ThumbnailMessage::Offline { .. }); // A closed channel means the window went away mid-fetch. if tx.send(msg).is_err() || offline { break; } } // Dates for images whose pixels were already cached. Header only — // no preview range, no decode. // // Flushed in chunks rather than once at the end: 119 sequential // header fetches take tens of seconds, and a single write at the // finish loses every one of them if the window closes first. It // also lets the timeline appear while the rest are still arriving. // // Skipped entirely when the connection has already failed: these // are network reads too, and there is nothing left to read from. const FLUSH_EVERY: usize = 16; if !offline { log::info!("reading dates for {} image(s)", metadata_only.len()); for req in metadata_only { if tx.send(ThumbnailMessage::DateProgress).is_err() { break; } read_metadata_only(&*backend, &req, &mut found).await; if found.len() >= FLUSH_EVERY { flush_metadata(&catalog_path, &mut found, &tx); } } } // Always flushed, even when the batch was abandoned: whatever was // read before the connection died is still true, and discarding it // would mean re-fetching those headers next time. flush_metadata(&catalog_path, &mut found, &tx); }); }); rx } async fn fetch_one( backend: &dyn RemoteBackend, store: Option<&mut ThumbStore>, req: &ThumbnailRequest, found_metadata: &mut Vec, ) -> ThumbnailMessage { let preview = match fetch_preview(backend, req, found_metadata).await { PreviewOutcome::Ready(p) => p, PreviewOutcome::Unavailable(reason) => { return ThumbnailMessage::Unavailable { row: req.row, reason, } } PreviewOutcome::Offline(reason) => return ThumbnailMessage::Offline { reason }, }; // Persist for next time, and for every other client that syncs the shard. // A store failure is logged and dropped: the pixels are already in hand, // and refusing to display them because they could not be cached would be // the wrong trade. if let (Some(store), Some(file_id)) = (store, req.file_id) { if let Some(thumb) = encode_preview(file_id, &preview) { store_thumbnail(store, file_id, req.thumb_size, &thumb); } } ThumbnailMessage::Ready(Box::new(ThumbnailReady { row: req.row, width: preview.width, height: preview.height, rgba: preview.rgba, from_cache: false, })) } /// What one fetch produced. /// /// Separate from [`ThumbnailMessage`] because not every caller has a grid row /// to report against or a store to write through. The whole-library pass /// ([`spawn_thumbnail_sweep`]) fetches on several lanes at once and stores the /// results on the one thread that owns the store, so it needs the pixels /// *before* anything is written or addressed to a cell. enum PreviewOutcome { Ready(dr_decode::Preview), /// This image has no usable preview. The batch continues past it. Unavailable(String), /// The server is unreachable, so nothing after this would succeed either. Offline(String), } /// Fetch a preview in two stages: header, then the exact preview range. /// /// This is what FR-NC-3 specifies, and the single-stage version it replaces /// was wrong in a way that looked like corruption: fetching a fixed prefix cut /// the embedded JPEG partway through, and decoders render a truncated JPEG as /// the top fraction of the frame rather than reporting an error. async fn fetch_preview( backend: &dyn RemoteBackend, req: &ThumbnailRequest, found_metadata: &mut Vec, ) -> PreviewOutcome { let id = RemoteId::Path(RemotePath::new(&req.path)); // A connection failure is not this image's verdict. Reported as such so // the caller can stop the batch rather than marking sixty cells // individually unpreviewable over one dropped connection — a state the // grid would then keep until something forced a reload. let classify = |e: dr_sync::RemoteError| { if e.indicates_offline() { PreviewOutcome::Offline(e.to_string()) } else { PreviewOutcome::Unavailable(e.to_string()) } }; // Stage one: the header, enough to parse the container's IFDs. let header = match backend.get(&id, Some(0..dr_decode::HEADER_BYTES)).await { Ok(b) => b, Err(e) => return classify(e), }; // The same bytes carry EXIF. Reading it here is free — the alternative is // a second 256 KB fetch per image over the whole library. if req.needs_metadata { collect_metadata(&header, req, found_metadata); } // Read unconditionally, unlike the rest of the EXIF above: `needs_metadata` // is false once an image has been catalogued, but a thumbnail can still be // regenerated long after that — a cleared cache, a new size — and a // thumbnail that came out upright the first time must come out upright // every time. This is a header walk, not a decode; see `dr_decode::orientation`. let orientation = dr_decode::orientation(&header).unwrap_or_default(); // A plain JPEG is its own preview; anything else needs locating. let bytes = if header.starts_with(&[0xFF, 0xD8, 0xFF]) { match backend.get(&id, None).await { Ok(b) => b, Err(e) => return classify(e), } } else { let Some(loc) = dr_decode::locate_preview(&header, req.size) else { // No locatable preview. Declining beats fetching the whole file: // that is the 370 GB path FR-NC-3 exists to avoid. return PreviewOutcome::Unavailable("no locatable embedded preview".into()); }; if loc.len() > MAX_PREVIEW_BYTES { return PreviewOutcome::Unavailable(format!( "preview is {} bytes, too large", loc.len() )); } // Stage two: exactly the preview's bytes. match backend.get(&id, Some(loc.range.clone())).await { Ok(b) => b, Err(e) => return classify(e), } }; // Verify before decoding. A truncated JPEG decodes "successfully" into a // partial frame, so without this the broken result reaches the cache and // the screen looking like a corrupt file. if !dr_decode::is_complete_jpeg(&bytes) { return PreviewOutcome::Unavailable("preview bytes are incomplete".into()); } // Decode on the worker, never the UI thread. let mut preview = match dr_decode::decode_jpeg(&bytes) { Ok(p) => p, Err(e) => return PreviewOutcome::Unavailable(e.to_string()), }; // Face indexing keeps the detail; every other caller is filling a cell of a // known size and the full preview is waste from here on. preview.downscale_to(if req.full_resolution { FACE_SOURCE_EDGE } else { req.thumb_size.edge() }); // Turn it the right way up before it is measured, cached or shown. An // embedded preview is written in the sensor's orientation, so without this // every frame shot in portrait lies on its side in the grid — and, because // the cache is keyed by file and size alone, stays that way. // // After the downscale, so the permutation moves thumbnail-sized bytes // rather than the full preview's. // // **Doing it here is what keeps face geometry honest.** Detection runs on // whatever this returns, so returning the photograph rather than the sensor // means every box and landmark is already in the space the catalog stores // and the develop overlay draws — no second mapping to get backwards, which // is the one orientation bug this codebase keeps having. It costs a // permutation of a larger buffer for the face path; that is ~15 ms against // a decode of ~150 ms, and it buys the whole class of bug. preview.apply_orientation(orientation); PreviewOutcome::Ready(preview) } /// Compress a decoded preview to what the store holds. /// /// Split from the write so the whole-library pass can do it on the lane that /// fetched the image: encoding is the one part of storing a thumbnail that /// costs CPU rather than the store's lock, and it turns a 256 KB RGBA buffer /// into ~20 KB before the chunk is handed to the single thread that owns the /// store. fn encode_preview(file_id: u64, preview: &dr_decode::Preview) -> Option { match dr_thumbs::encode_rgba(preview.width, preview.height, &preview.rgba) { Ok(bytes) => Some(dr_thumbs::Thumbnail { width: preview.width, height: preview.height, bytes, }), Err(e) => { log::debug!("encoding thumbnail {file_id}: {e}"); None } } } /// Put a thumbnail in the store, logging rather than failing. /// /// A store failure costs a re-fetch next time and nothing else — the pixels /// are already in hand, and the caller has something to show or count either /// way (ARCH §6.12: the store is derived, never authoritative). fn store_thumbnail( store: &mut ThumbStore, file_id: u64, size: dr_thumbs::ThumbSize, thumb: &dr_thumbs::Thumbnail, ) -> bool { match store.put(file_id, size, thumb) { Ok(_) => true, Err(e) => { log::debug!("storing thumbnail {file_id}: {e}"); false } } } /// Parse EXIF out of a header and record it. /// /// Shared by both paths — the thumbnail fetch, which gets the header anyway, /// and the header-only pass for images whose pixels were already cached. fn collect_metadata(header: &[u8], req: &ThumbnailRequest, out: &mut Vec) { let Ok(md) = dr_decode::metadata(header) else { return; }; out.push(MetadataFound { image_id: req.image_id, captured_at: md.captured_at, captured_offset: md.captured_offset, camera: camera_label(md.make.as_deref(), md.model.as_deref()), lens: md.lens.map(|l| l.trim().to_string()), iso: md.iso, }); } /// TRACES: FR-CAT-11 /// The camera string the catalog stores, from an EXIF make and model. /// /// One definition rather than one per caller, because an import's duplicate /// check compares against what a scan wrote (`dr_catalog::dedup`). Two /// spellings of the same body would not fail loudly — they would silently /// disable the cheap tier, and every re-inserted card would transfer in full /// before the digest caught it. pub fn camera_label(make: Option<&str>, model: Option<&str>) -> Option { match (make, model) { // Bodies repeat the make inside the model ("Canon EOS 6D"), so // joining unconditionally yields "Canon Canon EOS 6D". (Some(make), Some(model)) if model.starts_with(make) => Some(model.trim().to_string()), (Some(make), Some(model)) => Some(format!("{} {}", make.trim(), model.trim())), (None, Some(model)) => Some(model.trim().to_string()), _ => None, } } /// Write a batch of dates and tell the UI, draining `found`. /// /// Separate from the loop so the same path serves both the periodic flush and /// the final one, and so a write failure is reported once rather than being /// silently swallowed by the caller. fn flush_metadata( catalog_path: &std::path::Path, found: &mut Vec, tx: &Sender, ) { if found.is_empty() { return; } match Catalog::open(catalog_path) { Ok(cat) => match write_metadata(&cat, found) { Ok(n) => { log::info!("recorded capture dates for {n} of {} image(s)", found.len()); // Tell the UI so the timeline can appear. Without this the // histogram only shows up on the next window load, which on a // fully cached library may be never. let _ = tx.send(ThumbnailMessage::DatesRecorded(n)); } Err(e) => log::warn!("writing metadata: {e}"), }, Err(e) => log::warn!("opening catalog to write metadata: {e}"), } found.clear(); } /// Read only the date for an image whose thumbnail is already cached. /// /// One 256 KB header request, no preview range and no decode. This is what /// gets a library dated when its thumbnails came from the store — including /// shards synced from another device, which carry pixels but no metadata. /// Read a header for its date. /// /// Returns whether the file was **reached**, which the caller needs and cannot /// otherwise tell: a header that carried no EXIF and a fetch that never /// happened both leave `found` untouched, and recording the second as "this /// image has no date" would let one lock mark it dateless for good. async fn read_metadata_only( backend: &dyn RemoteBackend, req: &ThumbnailRequest, found: &mut Vec, ) -> bool { let id = RemoteId::Path(RemotePath::new(&req.path)); // Retried, because one failure here is usually a lock rather than a // verdict. Nextcloud's file locking answers a plain *read* with 423 under // concurrency, and the identical range succeeds moments later — measured // against a real server while twelve lanes were running. Without a retry // those images sit out the whole pass over a lock that lasted a moment. // // Bounded and short: a genuinely missing or forbidden file must not cost // three round trips before the sweep moves on. const ATTEMPTS: usize = 3; for attempt in 1..=ATTEMPTS { match backend.get(&id, Some(0..dr_decode::HEADER_BYTES)).await { Ok(header) => { collect_metadata(&header, req, found); return true; } Err(e) if e.is_transient() && attempt < ATTEMPTS => { // Backing off at all matters more than the exact interval: the // contention that produced the lock is our own lanes, so any // pause lets the holder finish. tokio::time::sleep(std::time::Duration::from_millis(200 * attempt as u64)).await; } Err(e) => { // Not surfaced: a missing date leaves the image off the // timeline rather than breaking anything, and the next sweep // retries it regardless. log::debug!("reading date for {} ({attempt} attempts): {e}", req.path); return false; } } } false } /// Write capture metadata read during the thumbnail pass. /// /// Promotes each row from `metadata_state = 1` (stat-only) to 2 (full EXIF), /// which is what makes it eligible for the timeline. A row whose EXIF was /// unreadable stays at 1 rather than being marked done with empty fields, so a /// later attempt can retry it. /// /// Returns how many rows were promoted. pub fn write_metadata( catalog: &Catalog, found: &[MetadataFound], ) -> Result { let conn = catalog.connection(); let tx = conn.unchecked_transaction()?; let mut promoted = 0; for m in found { // Only a real timestamp counts as fully read. Camera and lens without // a date leave the image unplaceable on a timeline, which is exactly // the state the grid needs to distinguish. let state = if m.captured_at.is_some() { 2 } else { 1 }; tx.execute( "UPDATE images SET captured_at = coalesce(?2, captured_at), captured_offset = coalesce(?3, captured_offset), camera = coalesce(?4, camera), lens = coalesce(?5, lens), iso = coalesce(?6, iso), metadata_state = max(metadata_state, ?7) WHERE id = ?1", rusqlite::params![ m.image_id, m.captured_at, m.captured_offset, m.camera, m.lens, m.iso, state, ], )?; if state == 2 { promoted += 1; } } tx.commit()?; Ok(promoted) } /// Progress from the whole-library sweep. #[derive(Debug)] pub enum SweepMessage { /// How many images still need work, counted once at the start. Total(usize), /// Another chunk finished. Carries cumulative counts. Progress { done: usize, dated: usize, }, Finished { dated: usize, }, } /// Await every future concurrently, returning results in order. /// /// A hand-rolled `join_all` rather than a `futures` dependency for one /// function. Polling a `Vec` of futures in a loop is exactly what the crate's /// version does; the ordering guarantee is what lets the caller pair results /// back to their inputs. async fn futures_join_all(futures: impl IntoIterator) -> Vec where F: std::future::Future, { use std::pin::Pin; use std::task::Poll; // Boxed so each future has a stable address while it is polled in place. let mut pending: Vec>>> = futures.into_iter().map(|f| Some(Box::pin(f))).collect(); let mut done: Vec> = (0..pending.len()).map(|_| None).collect(); std::future::poll_fn(move |cx| { let mut all_ready = true; for (slot, out) in pending.iter_mut().zip(done.iter_mut()) { let Some(fut) = slot else { continue }; match fut.as_mut().poll(cx) { Poll::Ready(v) => { *out = Some(v); // Dropped as soon as it completes, so a long-running lane // does not hold a finished one's resources. *slot = None; } Poll::Pending => all_ready = false, } } if all_ready { Poll::Ready(done.iter_mut().filter_map(Option::take).collect()) } else { Poll::Pending } }) .await } /// How many images one sweep chunk handles before committing. /// /// Small enough that a kill loses little, large enough that the catalog is not /// reopened per image. A multiple of [`SWEEP_LANES`] so every lane gets equal /// work and no chunk ends with most lanes idle. const SWEEP_CHUNK: usize = 96; /// How many fetches the sweep keeps in flight. /// /// Each is ~0.6 s of round-trip latency and almost no bandwidth — a 256 KB /// header — so the sequential version spent essentially all its time waiting. /// Twelve lanes turn ~3 hours into ~15 minutes on the reference library. /// /// Deliberately bounded rather than unlimited: the grid's own interactive /// fetches share this server, and a sweep that saturated the connection would /// make browsing feel broken while it ran. /// /// **Lowered from twelve after measuring.** Twelve produced 423 Locked on a /// real server — Nextcloud's file locking answering a plain read under /// contention we were creating ourselves. Six keeps most of the speedup /// without provoking it; the retry above covers what still slips through. const SWEEP_LANES: usize = 6; /// TRACES: FR-CULL-8 | NFR-RES-2 /// The largest original the face sweep will fetch, in bytes. /// /// A budget, not a correctness rule: nothing about a byte count says whether /// a file decodes. It exists because the sweep fetches the whole original /// before it can learn anything about it, and the one file in the reference /// library above this line is a 521 MB stitched panorama the decoder refuses /// on sight — so every pass on the tablet spent half a gigabyte of Wi-Fi to /// find that out again. Below the line: every camera RAW this library holds, /// the largest a 60 MB medium-format file; above it, four files, all /// panoramas. /// /// A file over budget is marked examined with nothing found and a zero /// edge, the same mark a file the decoder cannot open gets, so the count in /// the sweep's report says it was skipped and a later pass can select it. /// That later pass is the real answer for a panorama — read it in tiles, /// detect in each, and stitch the boxes back — and this constant is the /// placeholder for it, not a decision that panoramas hold no faces. const SWEEP_MAX_ORIGINAL_BYTES: u64 = 256 * 1024 * 1024; /// Date **every** image in the library, not just the ones on screen. /// /// Thumbnails are deliberately *not* fetched here. A thumbnail needs the /// mutable store, which cannot be shared across the parallel lanes below, and /// it costs 1–3 MB against a date's 256 KB. Dating the whole library is what /// the timeline needs; thumbnails arrive as cells are actually browsed, which /// is the FR-NC-3 posture anyway. /// /// The grid's own fetches cover what is on screen; this covers the rest, so the /// timeline describes the whole library rather than the part that happened to /// be scrolled past. It is resumable by construction — each pass queries for /// what is still missing, so a kill mid-sweep costs only the current chunk. /// /// Runs at the back of the queue by design: it holds no lock the grid needs, /// and its chunked commits keep write transactions short. pub fn spawn_sweep(conn: Connection, catalog_path: PathBuf) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let catalog = match Catalog::open(&catalog_path) { Ok(c) => c, Err(e) => { // Silent failure here left the sweep looking like it had run // and found nothing: no progress, no error, 17,397 images // still unindexed. log::warn!( "sweep: cannot open catalog at {}: {e}", catalog_path.display() ); let _ = tx.send(SweepMessage::Finished { dated: 0 }); return; } }; let outstanding = count_outstanding(&catalog).unwrap_or(0); if outstanding == 0 { let _ = tx.send(SweepMessage::Finished { dated: 0 }); return; } log::info!("sweep: {outstanding} image(s) need a date or a thumbnail"); if tx.send(SweepMessage::Total(outstanding)).is_err() { return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { log::warn!("sweep: no runtime: {e}"); return; } }; rt.block_on(async { let Ok(backend) = crate::remote::connect(&conn) else { return; }; let (mut done, mut dated) = (0usize, 0usize); loop { // Re-queried each pass rather than held as one long list: the // grid is dating images at the same time, and a stale list // would refetch what it already covered. let chunk = match next_outstanding(&catalog, SWEEP_CHUNK) { Ok(c) if c.is_empty() => break, Ok(c) => c, Err(e) => { log::warn!("sweep: {e}"); break; } }; let chunk_started = std::time::Instant::now(); log::debug!( "sweep: chunk of {} starting at image {}", chunk.len(), chunk[0].image_id ); // Twelve lanes over the chunk. Each lane owns a disjoint slice // and its own `found` vector, so nothing is shared and no lock // is needed; the results are concatenated after the join. // // The thumbnail store is the exception — it is `&mut` and // cannot be shared — so lanes only *read* metadata and any // missing thumbnail is left to the interactive path. Dating the // library is what the sweep is for; thumbnails arrive as cells // are browsed. let lanes: Vec> = (0..SWEEP_LANES) .map(|lane| chunk.iter().skip(lane).step_by(SWEEP_LANES).collect()) .collect(); let results = futures_join_all(lanes.into_iter().map(|lane| { let backend: &dyn RemoteBackend = &*backend; async move { let mut found = Vec::new(); let mut reached = Vec::new(); for req in lane { if read_metadata_only(backend, req, &mut found).await { reached.push(req.image_id); } } (found, reached) } })) .await; let mut found = Vec::new(); let mut reached = std::collections::HashSet::new(); for (lane_found, lane_reached) in results { found.extend(lane_found); reached.extend(lane_reached); } done += chunk.len(); // An image whose header carried no EXIF at all yields nothing // to `found`, so nothing marks it examined and the next sweep // fetches it again — for ever. Darktable exports strip // metadata by default, and 2,188 of them in the reference // library meant 2,188 pointless round trips per run. // // Recorded as examined with no date: the file was read and // genuinely has none, which is a different state from "not // looked at yet" and must not be confused with it. let answered: std::collections::HashSet = found.iter().map(|m| m.image_id).collect(); // Only files actually read. One that could not be fetched is // left alone so the next pass retries it, rather than being // written off over a lock or a dropped connection. found.extend( chunk .iter() .filter(|r| { reached.contains(&r.image_id) && !answered.contains(&r.image_id) }) .map(|r| MetadataFound { image_id: r.image_id, captured_at: None, captured_offset: None, camera: None, lens: None, iso: None, }), ); dated += found.iter().filter(|m| m.captured_at.is_some()).count(); let read = answered.len(); flush_sweep(&catalog, &mut found); log::info!( "sweep: {done} done, {dated} dated ({read} read in {:.1}s)", chunk_started.elapsed().as_secs_f64() ); if tx.send(SweepMessage::Progress { done, dated }).is_err() { return; } } log::info!("sweep complete: {dated} date(s) recorded over {done} image(s)"); let _ = tx.send(SweepMessage::Finished { dated }); }); }); rx } /// How many images still lack a date or a thumbnail. fn count_outstanding(catalog: &Catalog) -> Result { let n: i64 = catalog.connection().query_row( &format!( "SELECT count(*) FROM images WHERE metadata_state < 2 AND {VISIBLE_UNALIASED}" ), [], |r| r.get(0), )?; Ok(n as usize) } /// The next images needing work. /// /// Ordered by id so the sweep advances deterministically and a resumed run /// picks up where it left off rather than revisiting. fn next_outstanding( catalog: &Catalog, limit: usize, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT i.id, i.source_ref, r.file_id, i.file_size FROM images i LEFT JOIN remote r ON r.image_id = i.id WHERE i.metadata_state < 2 AND {VISIBLE} ORDER BY i.id LIMIT ?1" ))?; let rows = stmt .query_map([limit as i64], |r| { Ok(ThumbnailRequest { // The sweep indexes dates, and reads headers only — the size // never reaches a fetch, but it must name something. thumb_size: dr_thumbs::ThumbSize::Grid, // Row index is meaningless here — the sweep touches no grid // cell, so nothing consumes it. row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id: r.get::<_, Option>(2)?.map(|v| v as u64), size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, needs_metadata: true, full_resolution: false, }) })? .collect::, _>>()?; Ok(rows) } /// Commit a sweep chunk. /// /// An image whose header yielded no date is still marked done, or the sweep /// would revisit it forever. `write_metadata` records `metadata_state = 1` for /// those, so this promotes them explicitly. fn flush_sweep(catalog: &Catalog, found: &mut Vec) { if found.is_empty() { return; } if let Err(e) = write_metadata(catalog, found) { log::warn!("sweep: writing metadata: {e}"); found.clear(); return; } // Mark the dateless as examined. Without this they stay at state 1 and the // sweep loops over them on every pass, never terminating. let ids: Vec = found .iter() .filter(|m| m.captured_at.is_none()) .map(|m| m.image_id) .collect(); for id in ids { let _ = catalog .connection() .execute("UPDATE images SET metadata_state = 2 WHERE id = ?1", [id]); } found.clear(); } /// TRACES: FR-CAT-3 | FR-NC-3 | NFR-RES-4 /// TRACES: FR-CULL-8 /// Every visible image this model has not been run over. /// /// **No thumbnail-store filter.** The pass this feeds fetches its own pixels, /// so an image with no proxy is work to be done rather than work to be skipped /// — which is the whole difference between indexing a library and indexing the /// fraction of it that has been browsed. /// /// **An image whose faces are merely unmeasured is not here.** Schema V14 /// forgot the run marker of every such image so that *something* would look /// at it again; [`faces_unmeasured`] is that something, and it does the cheap /// thing — embed the faces already found — where this list would have the /// whole detection run again. Both lists are drawn from the same catalog in /// the same sweep, so the exclusion is in the query rather than left to the /// caller to remember. fn faces_unindexed( catalog: &Catalog, model_id: &str, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT i.id, i.source_ref, r.file_id, i.file_size FROM images i JOIN remote r ON r.image_id = i.id WHERE r.file_id IS NOT NULL AND {VISIBLE} AND NOT EXISTS ( SELECT 1 FROM face_index fi WHERE fi.image_id = i.id AND {fi_embedder} = ?1 ) AND NOT EXISTS ( SELECT 1 FROM faces f WHERE f.image_id = i.id AND {f_embedder} = ?1 AND f.quality IS NULL ) ORDER BY i.id", fi_embedder = dr_catalog::faces::embedder_sql("fi.model_id"), f_embedder = dr_catalog::faces::embedder_sql("f.model_id"), ))?; let rows = stmt .query_map([dr_catalog::faces::embedder_of(model_id)], |r| { Ok(ThumbnailRequest { // Face indexing wants the detail a thumbnail discards. full_resolution: true, // Named because the field must say something; ignored, because // `full_resolution` overrides it. thumb_size: dr_thumbs::ThumbSize::Large, // No grid cell waits on this. row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id: r.get::<_, Option>(2)?.map(|v| v as u64), size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, // The thumbnail sweep owns dating. Reading EXIF here would // write the same rows from a second pass for no gain. needs_metadata: false, }) })? .collect::, _>>()?; Ok(rows) } /// Images whose faces have nothing left to be cut out of. /// /// A face is stored normalised and drawn by cropping the proxy it was found on /// (`identity::decode_proxy`). Where that proxy is gone the People screen draws /// "no preview" for every cell and cannot repair itself, because the image /// already has its `face_index` row and so is not outstanding work. /// /// Two ways in: an indexing pass that fetched a preview and did not keep it, /// and an ordinary cache eviction. Both are the same state, and re-running /// detection over the image fixes it — `record_detections` replaces rather than /// appends, and carries the user's confirmations across the replacement, so /// this costs a fetch and loses nothing. fn faces_without_proxy( catalog: &Catalog, store: &ThumbStore, model_id: &str, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT DISTINCT i.id, i.source_ref, r.file_id, i.file_size FROM images i JOIN remote r ON r.image_id = i.id JOIN faces f ON f.image_id = i.id WHERE r.file_id IS NOT NULL AND {VISIBLE} AND {embedder} = ?1 ORDER BY i.id", embedder = dr_catalog::faces::embedder_sql("f.model_id"), ))?; let rows = stmt .query_map([dr_catalog::faces::embedder_of(model_id)], |r| { Ok(ThumbnailRequest { full_resolution: true, thumb_size: dr_thumbs::ThumbSize::Large, row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id: r.get::<_, Option>(2)?.map(|v| v as u64), size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, needs_metadata: false, }) })? .filter_map(Result::ok) .filter(|req| { req.file_id .is_some_and(|id| !store.contains(id, dr_thumbs::ThumbSize::Large)) }) .collect(); Ok(rows) } /// TRACES: FR-CULL-9 /// Images holding a face this model found before its quality was kept. /// /// The work list of the sweep's measuring pass: every stored face whose vector /// is a unit one (schema V14) is embedded again from the native render, with /// the landmarks it already has, and the raw vector and its length written /// over it. Nothing is re-detected and no face changes identity — see /// `dr_catalog::faces::record_measurements`. /// /// Costs what the indexing pass costs per image, an original fetched and /// rendered, because the length exists only at the moment of embedding and /// there is no embedding without the pixels. What it saves is the detector, /// and — the part that matters — every suggestion and confirmation on those /// faces, which a re-detection would rebuild from box overlap. fn faces_unmeasured( catalog: &Catalog, model_id: &str, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT DISTINCT i.id, i.source_ref, r.file_id, i.file_size FROM images i JOIN remote r ON r.image_id = i.id JOIN faces f ON f.image_id = i.id WHERE r.file_id IS NOT NULL AND {VISIBLE} AND {embedder} = ?1 AND f.quality IS NULL ORDER BY i.id", embedder = dr_catalog::faces::embedder_sql("f.model_id"), ))?; let rows = stmt .query_map([dr_catalog::faces::embedder_of(model_id)], |r| { Ok(ThumbnailRequest { full_resolution: true, thumb_size: dr_thumbs::ThumbSize::Large, row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id: r.get::<_, Option>(2)?.map(|v| v as u64), size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, needs_metadata: false, }) })? .collect::, _>>()?; Ok(rows) } /// Images indexed by a detector the chosen one outranks. /// /// The tail of the sweep's work list: behind everything nothing has looked /// at, because a photograph with no faces recorded is worth more than one /// whose faces a weaker detector may have under-counted. `weaker` is /// `FaceDetector::supersedes` for the current choice, and it is a list of /// exact ids rather than "anything else sharing the embedder" so the /// re-detection only ever runs upwards — a device set to the fast detector /// leaves a peer's thorough pass alone. /// /// The user's confirmations survive the re-detection by box overlap /// (`faces::record_detections`), which is what makes this safe to run over a /// library that is already named. fn faces_superseded( catalog: &Catalog, weaker: &[&str], ) -> Result, dr_catalog::CatalogError> { if weaker.is_empty() { return Ok(Vec::new()); } let placeholders = weaker .iter() .enumerate() .map(|(i, _)| format!("?{}", i + 1)) .collect::>() .join(", "); let mut stmt = catalog.connection().prepare(&format!( "SELECT i.id, i.source_ref, r.file_id, i.file_size FROM images i JOIN remote r ON r.image_id = i.id WHERE r.file_id IS NOT NULL AND {VISIBLE} AND EXISTS ( SELECT 1 FROM face_index fi WHERE fi.image_id = i.id AND fi.model_id IN ({placeholders}) ) ORDER BY i.id" ))?; let rows = stmt .query_map(rusqlite::params_from_iter(weaker.iter()), |r| { Ok(ThumbnailRequest { full_resolution: true, thumb_size: dr_thumbs::ThumbSize::Large, row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id: r.get::<_, Option>(2)?.map(|v| v as u64), size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, needs_metadata: false, }) })? .collect::, _>>()?; Ok(rows) } /// Take the originals over [`SWEEP_MAX_ORIGINAL_BYTES`] out of the work list, /// marking each as examined so no later pass fetches it either. /// /// Before the fetch loop and on the byte count the catalog already holds, /// which is the whole point: the decision costs nothing, where fetching the /// file to decide would cost the file. Returns how many were set aside, which /// the sweep reports as failed — not indexed, and said so. fn set_aside_oversized( catalog: &Catalog, model_id: &str, wanted: &mut Vec<(ThumbnailRequest, SweepWork)>, ) -> usize { let mut skipped = 0usize; wanted.retain(|(req, _)| { if req.size <= SWEEP_MAX_ORIGINAL_BYTES { return true; } log::warn!( "face sweep: {} is {} MB, over the {} MB budget for a background fetch; \ marked as examined and skipped", req.path, req.size >> 20, SWEEP_MAX_ORIGINAL_BYTES >> 20 ); if let Err(e) = dr_catalog::faces::record_detections( catalog.connection(), dr_types::ImageId(req.image_id as u64), model_id, 0, &[], ) { log::warn!("face sweep: marking {} as skipped: {e}", req.image_id); } skipped += 1; false }); skipped } /// What the sweep does with one fetched original. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum SweepWork { /// Detect and embed from scratch: an image never indexed, or one whose /// faces have nothing left to be cut from. Detect, /// Embed the faces already found, again — see [`faces_unmeasured`]. Measure, } /// TRACES: FR-CULL-8 | FR-EXP-9 | NFR-ARCH-2 | NFR-RES-2 /// Index faces across the **whole** library, at native resolution. /// /// # The two resolutions, and why they are not one /// /// FR-CULL-8 asks for a native render, a *reduction* for the detector, and the /// crop taken back out of the native buffer. That is not three sizes for the /// sake of it. The detector letterboxes whatever it is handed into a fixed /// 640×640, so above that its input resolution decides nothing and paying for /// it is waste; `align::warp` produces the fixed 112×112 ArcFace sees, so /// *its* input resolution decides everything and economising there is a /// silent loss. The two stages want opposite things, and a single buffer /// serving both is how this pass previously came to store 47% of the /// reference library's faces upsampled (faces.md §7b). /// /// # Why this replaced a pass that read an embedded preview /// /// The previous version range-fetched the JPEG preview embedded in each RAW /// (FR-NC-3) and used it for both stages. It was cheap and it was the tier /// FR-CULL-8 named at the time. On the reference library that preview tops out /// at 3072 px against a ~6000 px sensor, which is what put those 8,505 faces /// below the embedder's 112 px with no way to tell from the catalog that /// anything was wrong. /// /// Before that it filtered its work list to images with a `ThumbSize::Large` /// proxy already on disk, which nothing filled for a whole library, so it /// reached 220 images out of 23,529. /// /// # Cost, stated plainly /// /// **One whole original per un-indexed image, and one full render.** On the /// reference library that is 412 GB and roughly a hundred minutes of decode — /// a different order of thing from the byte ranges this used to pay, which is /// why FR-CULL-8 makes a whole-library pass a transfer under FR-NC-6 rather /// than something that may start on its own. /// /// Nothing is kept that was not already wanted: the original is borrowed and /// given back (ARCH §9.0a), and the only thing written per image is the /// 1024 px proxy the People screen crops from, and only where a face was /// found. Resumable by construction — the work list is what the catalog has no /// `face_index` row for, so a kill costs the images in flight and nothing else. #[allow(clippy::too_many_arguments)] pub fn spawn_face_sweep( conn: Connection, catalog_path: PathBuf, store_dir: PathBuf, detector_model: PathBuf, embedder_model: PathBuf, model_id: String, supersedes: Vec, options: dr_face::DetectOptions, gpu: dr_gpu::GpuContext, ) -> Receiver { use crate::faces::FaceSweepMessage; let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let finish_empty = |tx: &Sender| { let _ = tx.send(FaceSweepMessage::Finished { images: 0, faces: 0, failed: 0, }); }; let catalog = match Catalog::open(&catalog_path) { Ok(c) => c, Err(e) => { log::warn!("face sweep: cannot open catalog: {e}"); finish_empty(&tx); return; } }; // The proxy an indexed face is later *cut from*. // // `identity::decode_proxy` reads `FACE_TIER` out of this store to draw // the People screen, so a face found on a preview that was fetched and // dropped has nowhere to come from and the grid shows "no preview" for // every cell. Fetching the pixels and not keeping them was the whole // bug: they cost a round trip and a decode, and the crop needs them // again the moment the user looks at the person. let mut store = match ThumbStore::open(&store_dir) { Ok(s) => s, Err(e) => { log::warn!("face sweep: cannot open the thumbnail store: {e}"); finish_empty(&tx); return; } }; // Models before the work list: they are the expensive failure, and // listing twenty thousand images before discovering the weights are // missing helps nobody. A library with no model installed takes this // path, so it is a quiet return rather than an error. let mut detector = match dr_face::Detector::from_path(&detector_model) { Ok(d) => d, Err(e) => { log::warn!("face sweep: cannot load the detector: {e}"); finish_empty(&tx); return; } }; let mut embedder = match dr_face::Embedder::from_path( &embedder_model, dr_face::ModelId::new(model_id.clone()), ) { Ok(e) => e, Err(e) => { log::warn!("face sweep: cannot load the embedder: {e}"); finish_empty(&tx); return; } }; // **Repairs first, and the order is the whole point.** These are the // images the People screen is drawing *right now* and failing to, and // there are a few hundred of them against tens of thousands of // un-indexed ones. Appended instead, they sit two hours of fetching // down the queue and the screen stays empty for the whole session — // which is indistinguishable from the repair not existing. let mut wanted: Vec<(ThumbnailRequest, SweepWork)> = match faces_without_proxy(&catalog, &store, &model_id) { Ok(repair) => { if !repair.is_empty() { log::info!( "face sweep: repairing {} image(s) whose faces have no proxy to crop from", repair.len() ); } repair.into_iter().map(|r| (r, SweepWork::Detect)).collect() } Err(e) => { log::warn!("face sweep: looking for orphaned faces: {e}"); Vec::new() } }; // Then the faces to measure again. Behind the proxy repair and ahead // of the rest for the same reason that one leads: these are faces the // People screen is showing and grouping *now*, on vectors the gallery // rule cannot act on. An image already queued for a full re-detection // gets its quality from that, so it is not queued twice. let mut queued: std::collections::HashSet = wanted.iter().map(|(r, _)| r.image_id).collect(); match faces_unmeasured(&catalog, &model_id) { Ok(measure) => { let fresh: Vec<_> = measure .into_iter() .filter(|r| queued.insert(r.image_id)) .collect(); if !fresh.is_empty() { log::info!( "face sweep: measuring the faces on {} image(s) stored before their quality was kept", fresh.len() ); } wanted.extend(fresh.into_iter().map(|r| (r, SweepWork::Measure))); } Err(e) => log::warn!("face sweep: looking for unmeasured faces: {e}"), } // Disjoint from both of the above by construction: an image with faces // recorded is not an image with no `face_index` row, and one whose // faces are unmeasured is excluded by the query. match faces_unindexed(&catalog, &model_id) { Ok(fresh) => wanted.extend(fresh.into_iter().map(|r| (r, SweepWork::Detect))), Err(e) => { log::warn!("face sweep: {e}"); if wanted.is_empty() { finish_empty(&tx); return; } } } // Last: what a weaker detector already indexed. Everything above is // a photograph the People screen cannot show at all; these it shows // already, and the re-detection only adds the faces that detector // missed. See `faces_superseded`. let weaker: Vec<&str> = supersedes.iter().map(String::as_str).collect(); let mut upgrades = 0usize; match faces_superseded(&catalog, &weaker) { Ok(older) => { let fresh: Vec<_> = older .into_iter() .filter(|r| queued.insert(r.image_id)) .collect(); upgrades = fresh.len(); wanted.extend(fresh.into_iter().map(|r| (r, SweepWork::Detect))); } Err(e) => log::warn!("face sweep: looking for images under a weaker detector: {e}"), } let skipped = set_aside_oversized(&catalog, &model_id, &mut wanted); let total = wanted.len(); if total == 0 { log::info!("face sweep: every image has been through this model"); let _ = tx.send(FaceSweepMessage::Finished { images: 0, faces: 0, failed: skipped, }); return; } if upgrades > 0 { log::info!( "face sweep: {total} image(s) to index, {upgrades} of them indexed by a weaker detector" ); } else { log::info!("face sweep: {total} image(s) to index"); } if tx.send(FaceSweepMessage::Total(total)).is_err() { return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { log::warn!("face sweep: no runtime: {e}"); finish_empty(&tx); return; } }; rt.block_on(async { // Through `remote::connect`, which is the only place in the // interface that knows whose backend this is. let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { log::warn!("face sweep: {e}"); finish_empty(&tx); return; } }; // `failed` starts at what the size budget already set aside, so // the report counts them the way it counts a file that would not // decode: not indexed, and said so. let (mut images, mut found, mut failed) = (0usize, 0usize, skipped); let mut offline = false; // TRACES: FR-NC-6c // The same borrow the thumbnail sweep makes, and deliberately its // own pool: the two passes run at different times, so sharing one // would keep every file the earlier pass touched hydrated until // the later one finished. Each gives its own back (ARCH §9.0a). let pool = dr_sync_folder::BorrowPool::new(); // TRACES: NFR-RES-2 // **Fetch wide, render narrow.** The chunk is one original per // lane and not the 96 the preview sweep used, because the two // passes hold different things: that one kept an 8 MB preview per // image, this one keeps whole RAWs. Six at ~23 MB is a working set // a phone can carry; ninety-six is not. // // The render is then sequential, and that is not a limitation to // be optimised away later. There is one GPU, so concurrent renders // would queue on it anyway, and each one materialises a native // frame -- 96 MB for a 24 MP photograph. Overlapping them would // multiply the one allocation that actually threatens the budget // while buying no parallelism that exists. for chunk in wanted.chunks(SWEEP_LANES) { let fetched = futures_join_all(chunk.iter().map(|(req, work)| { let backend = &*backend; let pool = &pool; async move { let held = match pool.borrow(backend, &RemotePath::new(&req.path)).await { Ok(h) => h, Err(e) if e.indicates_offline() => { log::info!("face sweep: {e}"); return (req, *work, Err(FetchOutcome::Offline)); } Err(e) => { log::debug!("face sweep: {}: {e}", req.path); return (req, *work, Err(FetchOutcome::Failed)); } }; // TRACES: FR-CULL-8 // The whole file, not FR-NC-3's byte range. The // requirement now asks for a native render and there // is no native render without the original -- which // is why the pass is a transfer under FR-NC-6 and says // so before it starts. let id = RemoteId::Path(RemotePath::new(&req.path)); let got = match backend.get(&id, None).await { Ok(b) => Ok(b), Err(e) if e.indicates_offline() => { log::info!("face sweep: server unreachable: {e}"); Err(FetchOutcome::Offline) } Err(e) => { log::debug!("face sweep: {}: {e}", req.path); Err(FetchOutcome::Failed) } }; drop(held); (req, *work, got) } })) .await; let mut lane_failed = 0usize; for (req, work, got) in fetched { let bytes = match got { Ok(b) => b, Err(FetchOutcome::Offline) => { offline = true; continue; } Err(FetchOutcome::Failed) => { lane_failed += 1; continue; } }; if work == SweepWork::Measure { let image = dr_types::ImageId(req.image_id as u64); match measure_one_native( &gpu, &mut embedder, &catalog, image, &model_id, &bytes, ) { Ok(n) => { images += 1; found += n; if tx .send(FaceSweepMessage::Indexed { image, faces: n }) .is_err() { log::info!("face sweep: cancelled after {images} image(s)"); pool.release_all(&*backend).await; return; } } Err(e) => { log::debug!("face sweep: measuring {}: {e}", req.path); lane_failed += 1; } } continue; } match index_one_native(&gpu, &mut detector, &mut embedder, &bytes, &options) { Ok((faces, edge, proxy)) => { // Before the detections, so a kill between the two // leaves a proxy with no faces recorded -- which // the next pass simply re-indexes -- rather than // faces with no proxy, which is the state that // draws an empty grid and cannot repair itself. if let (Some(file_id), Some(thumb)) = (req.file_id, proxy) { store_thumbnail( &mut store, file_id, dr_thumbs::ThumbSize::Large, &thumb, ); } match dr_catalog::faces::record_detections( catalog.connection(), dr_types::ImageId(req.image_id as u64), &model_id, edge, &faces, ) { Ok(_) => { images += 1; found += faces.len(); if tx .send(FaceSweepMessage::Indexed { image: dr_types::ImageId(req.image_id as u64), faces: faces.len(), }) .is_err() { // Receiver dropped: the screen closed, // or the user pressed Stop. Everything // written so far stays written -- and // everything borrowed is given back. log::info!("face sweep: cancelled after {images} image(s)"); pool.release_all(&*backend).await; return; } } Err(e) => { log::warn!( "face sweep: storing faces for {}: {e}", req.image_id ); lane_failed += 1; } } } Err(e) => { // A file the decoder cannot open fails the same // way on every pass, and every pass fetched it // first: a 521 MB panorama the decoder refuses // was downloaded once per sweep, on a tablet, // and reported at debug level where nobody saw // it. It is marked examined with nothing found, // so the next pass does not fetch it again; the // zero edge is what says why, and is what a // later "try again with a better decoder" pass // would select on. The count still says it // failed, because it did. log::warn!("face sweep: {}: {e}", req.path); if let Err(e) = dr_catalog::faces::record_detections( catalog.connection(), dr_types::ImageId(req.image_id as u64), &model_id, 0, &[], ) { log::warn!( "face sweep: marking {} as unreadable: {e}", req.image_id ); } lane_failed += 1; } } } failed += lane_failed; if lane_failed > 0 && tx .send(FaceSweepMessage::Failed { images: lane_failed, }) .is_err() { log::info!("face sweep: cancelled after {images} image(s)"); pool.release_all(&*backend).await; return; } if offline { break; } } let returned = pool.release_all(&*backend).await; if returned.released > 0 { log::info!( "face sweep: released {} borrowed file(s)", returned.released ); } log::info!( "face sweep: {found} face(s) across {images} image(s), {failed} failed{}", if offline { ", server went away" } else { "" } ); let _ = tx.send(FaceSweepMessage::Finished { images, faces: found, failed, }); }); }); rx } /// TRACES: FR-CULL-8 | FR-EXP-9 /// Open one original for a native render, orientation applied and nothing else. /// /// The half of [`index_one_native`] that has nothing to do with faces, exposed /// because measuring what this pass is worth means rendering the same file two /// ways and comparing the crops — see `examples/face_native.rs`. A tool that /// had to reimplement the render would be measuring its own reimplementation. pub fn render_native(gpu: &dr_gpu::GpuContext, bytes: &[u8]) -> Result { open_native(gpu, bytes)?.render_for_export(dr_types::ColourSpace::Srgb) } /// The session behind [`render_native`], kept private because `DevelopSession` /// is. The indexing path needs the session itself rather than just its frame: /// it renders the People screen's proxy from the same open session rather than /// opening the file twice. fn open_native( gpu: &dr_gpu::GpuContext, bytes: &[u8], ) -> Result { // Whatever the header says, or an empty one for a file that has none: the // session takes its orientation from it, and remembers the rest for // anything that later exports from this session (FR-EXP-8). let meta = dr_decode::metadata(bytes).unwrap_or_default(); crate::open_session(gpu, bytes, &meta) } /// Why one original did not arrive. /// /// Named rather than a bool because the two mean opposite things to the loop: /// one image failing is one image, and the server going away means nothing /// after it would have worked either. enum FetchOutcome { Failed, Offline, } /// TRACES: FR-CULL-8 | FR-EXP-9 /// Render one original at native resolution and index the faces in it. /// /// The FR-CULL-8 pipeline end to end, for one photograph: decode, render /// through the same path export uses, detect on a reduction, crop from the /// native frame. Returns the faces, the native long edge that went into the /// run marker, and the 1024px proxy the People screen later cuts thumbnails /// from. /// /// # Why the stored edit is not applied /// /// `export::render_from_library` fetches the sidecar and applies it, because /// an export is of the photograph the user has made. This is not: the face /// geometry stored in the catalog is normalised to the frame, so applying a /// crop would record faces against a frame that changes whenever the user /// changes their mind, and every stored box would silently become wrong. What /// is applied is the orientation, which is a fact about the file rather than /// an edit. fn index_one_native( gpu: &dr_gpu::GpuContext, detector: &mut dr_face::Detector, embedder: &mut dr_face::Embedder, bytes: &[u8], options: &dr_face::DetectOptions, ) -> Result< ( Vec, u32, Option, ), String, > { let mut session = open_native(gpu, bytes)?; let frame = session.render_for_export(dr_types::ColourSpace::Srgb)?; let edge = frame.width.max(frame.height); let faces = crate::faces::index_native( detector, embedder, &frame.rgba, frame.width as usize, frame.height as usize, options, ) .map_err(|e| e.to_string())?; // Only where there is a face to cut out of it. Two thirds of a personal // library is landscapes and documents, and those never need a crop -- so // this fills the large class for the images the People screen will // actually ask about and leaves the rest alone. // // Rendered rather than downscaled from the frame in hand: the session is // still open and `render_thumbnail` is the path the grid's own thumbnails // take, so the proxy this writes is the one the store would have had // anyway. let proxy = if faces.is_empty() { None } else { match session.render_thumbnail(dr_thumbs::ThumbSize::Large.edge()) { Ok((w, h, rgba)) => match dr_thumbs::encode_rgba(w, h, &rgba) { Ok(bytes) => Some(dr_thumbs::Thumbnail { width: w, height: h, bytes, }), Err(e) => { log::debug!("encoding a face proxy: {e}"); None } }, Err(e) => { log::debug!("rendering a face proxy: {e}"); None } } }; Ok((faces, edge, proxy)) } /// TRACES: FR-CULL-8 | FR-CULL-9 /// Render one original at native resolution and embed its stored faces again. /// /// [`index_one_native`]'s sibling for the measuring pass: the same render, no /// detection, and the result written *over* the faces rather than in place of /// them. Returns how many faces were measured. /// /// The proxy is left alone. The faces here were found on a pass that stored /// one, or [`faces_without_proxy`] would have claimed the image first. fn measure_one_native( gpu: &dr_gpu::GpuContext, embedder: &mut dr_face::Embedder, catalog: &Catalog, image: dr_types::ImageId, model_id: &str, bytes: &[u8], ) -> Result { let faces = dr_catalog::faces::unmeasured_on_image(catalog.connection(), image, model_id) .map_err(|e| e.to_string())?; if faces.is_empty() { return Ok(0); } let frame = render_native(gpu, bytes)?; let edge = frame.width.max(frame.height); let measured = crate::faces::measure_native( embedder, &frame.rgba, frame.width as usize, frame.height as usize, &faces, ) .map_err(|e| e.to_string())?; let n = measured.measured.len(); dr_catalog::faces::record_measurements( catalog.connection(), image, model_id, edge, &measured.measured, &measured.dropped, ) .map_err(|e| e.to_string())?; Ok(n) } /// The class the whole-library pass fills. /// /// Grid only, deliberately. The large class is four times the transfer for a /// detail only a zoomed cell or the loupe asks for — on the reference library /// that is ~200 MB of shards against ~860 MB, paid by *every* device that /// syncs them (see [`dr_thumbs::ThumbSize`]). A photograph actually looked at /// closely still gets its large thumbnail from the interactive path. const SWEEP_THUMB_SIZE: dr_thumbs::ThumbSize = dr_thumbs::ThumbSize::Grid; /// Progress from the whole-library thumbnail pass. #[derive(Debug)] pub enum ThumbSweepMessage { /// How many images still lack a thumbnail, counted once at the start. Total(usize), /// Another chunk finished. Carries cumulative counts. Progress { done: usize, stored: usize }, Finished { stored: usize, failed: usize, /// Stopped early because the server stopped answering. The pass is /// resumable, so this is "come back later", not a failure. offline: bool, }, } /// TRACES: FR-CAT-3 | FR-NC-3 | FR-NC-7 /// Thumbnail **every** image in the library, not just the ones browsed. /// /// # Why this exists next to the grid's own fetching /// /// The interactive path fills cells as they are scrolled past, which is the /// right posture for a remote library (FR-NC-3) and the wrong one for handing /// the result to a second device: a tablet that syncs the shards inherits only /// the fraction of the library its sibling happened to look at. This is the /// deliberate, user-launched version of the same work — an hour of range /// fetches paid once, on the machine that can afford it, so every other client /// gets a full grid for the cost of a few hundred MB (`derived_sync`). /// /// # Shape, and why it borrows the metadata sweep's /// /// Chunked and lane-parallel exactly as [`spawn_sweep`] is, for the same /// reason: each image is ~0.6 s of round-trip latency and almost no /// bandwidth, so the sequential version spends its life waiting. What differs /// is the store — it is `&mut` and cannot be shared across lanes, which is why /// the metadata sweep skips thumbnails entirely. Here the lanes fetch, decode /// and *encode*, and only the ~20 KB result crosses back to this thread, which /// owns the store and writes the chunk in one go. So the parallelism is real /// and the single-writer rule is never bent. /// /// Dates arrive free: the header a preview needs is the header EXIF lives in, /// so an image this pass reaches is dated on the same fetch rather than /// costing a second one. /// /// Resumable by construction — the work list is what the store does not have, /// so a kill costs the chunk in flight and nothing more. An image with no /// locatable preview is retried on a later run; it is one header fetch, and /// the alternative is a second piece of state that has to be invalidated when /// a file is replaced. pub fn spawn_thumbnail_sweep( conn: Connection, catalog_path: PathBuf, store_dir: PathBuf, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let finish_empty = |tx: &Sender| { let _ = tx.send(ThumbSweepMessage::Finished { stored: 0, failed: 0, offline: false, }); }; let catalog = match Catalog::open(&catalog_path) { Ok(c) => c, Err(e) => { log::warn!( "thumbnail sweep: cannot open catalog at {}: {e}", catalog_path.display() ); finish_empty(&tx); return; } }; // Unlike the grid's fetch, which carries on without a store and simply // shows what it downloaded, a store that will not open ends this: the // pass exists to fill it, and running an hour of transfers with // nowhere to put them would be worse than not starting. let mut store = match ThumbStore::open(&store_dir) { Ok(s) => s, Err(e) => { log::warn!( "thumbnail sweep: cannot open the thumbnail store at {}: {e}", store_dir.display() ); finish_empty(&tx); return; } }; let wanted = match thumbnails_outstanding(&catalog, &store) { Ok(w) => w, Err(e) => { log::warn!("thumbnail sweep: {e}"); finish_empty(&tx); return; } }; let total = wanted.len(); if total == 0 { log::info!("thumbnail sweep: every image already has a thumbnail"); finish_empty(&tx); return; } log::info!("thumbnail sweep: {total} image(s) need a thumbnail"); if tx.send(ThumbSweepMessage::Total(total)).is_err() { return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { log::warn!("thumbnail sweep: no runtime: {e}"); finish_empty(&tx); return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { log::warn!("thumbnail sweep: {e}"); finish_empty(&tx); return; } }; let (mut done, mut stored, mut failed) = (0usize, 0usize, 0usize); let mut offline = false; let mut found = Vec::new(); // TRACES: FR-NC-6c // On a placeholder library the bytes may not be here at all, and // this is a pass the user asked for — so it may fetch them, which // browsing may not (ARCH §9.0a). Every file is *borrowed*: what // this pass downloads it gives back, and what the user already had // it leaves alone. Against a server or a plain folder every borrow // is a no-op, so there is one code path rather than two. let pool = dr_sync_folder::BorrowPool::new(); for chunk in wanted.chunks(SWEEP_CHUNK) { // Each lane owns a disjoint slice and its own output, so // nothing is shared and no lock is needed. The store is not // touched here — see the note on the function. let lanes: Vec> = (0..SWEEP_LANES) .map(|lane| chunk.iter().skip(lane).step_by(SWEEP_LANES).collect()) .collect(); let results = futures_join_all(lanes.into_iter().map(|lane| { let backend: &dyn RemoteBackend = &*backend; let pool = &pool; async move { let mut made: Vec<(u64, dr_thumbs::Thumbnail)> = Vec::new(); let mut found = Vec::new(); let mut attempted = 0usize; let mut failed = 0usize; let mut offline = false; for req in lane { // Enforced by the query, which joins `remote`: an // image with no file id has nothing to key the // store on and is not a candidate. let Some(file_id) = req.file_id else { continue }; attempted += 1; // Held for this image only. A failure to fetch is // this image's verdict, not the batch's: a client // that cannot reach the server reports it as // offline through the usual path below. let _held = match pool.borrow(backend, &RemotePath::new(&req.path)).await { Ok(h) => h, Err(e) if e.indicates_offline() => { log::info!("thumbnail sweep: {e}"); attempted -= 1; offline = true; break; } Err(e) => { log::debug!("thumbnail sweep: {}: {e}", req.path); failed += 1; continue; } }; match fetch_preview(backend, req, &mut found).await { PreviewOutcome::Ready(preview) => { match encode_preview(file_id, &preview) { Some(thumb) => made.push((file_id, thumb)), None => failed += 1, } } PreviewOutcome::Unavailable(reason) => { log::debug!("thumbnail sweep: {}: {reason}", req.path); failed += 1; } // Nothing after this would reach the server // either, so the lane stops rather than // spending a timeout per remaining image. PreviewOutcome::Offline(reason) => { log::info!("thumbnail sweep: server unreachable: {reason}"); attempted -= 1; offline = true; break; } } } (made, found, attempted, failed, offline) } })) .await; for (made, lane_found, attempted, lane_failed, lane_offline) in results { done += attempted; failed += lane_failed; offline |= lane_offline; found.extend(lane_found); for (file_id, thumb) in made { if store_thumbnail(&mut store, file_id, SWEEP_THUMB_SIZE, &thumb) { stored += 1; } else { failed += 1; } } } // Committed per chunk rather than at the end, so a kill keeps // every date read so far — the same bargain the metadata sweep // makes, and for the same reason. flush_sweep(&catalog, &mut found); if tx .send(ThumbSweepMessage::Progress { done, stored }) .is_err() { // Cancelled. Hand back what was borrowed before leaving, // or a stopped pass costs the disk of everything it had // reached and delivers nothing for it. pool.release_all(&*backend).await; return; } if offline { break; } } flush_sweep(&catalog, &mut found); // Give back everything this pass fetched, before reporting done — // a user watching the disk should see it return, and a pass that // reported success while still holding the library would be // lying about what it cost. let returned = pool.release_all(&*backend).await; if returned.released > 0 { log::info!( "thumbnail sweep: released {} borrowed file(s)", returned.released ); } log::info!("thumbnail sweep: {stored} stored, {failed} without a usable preview"); let _ = tx.send(ThumbSweepMessage::Finished { stored, failed, offline, }); }); }); rx } /// Every visible image on the server that the store has no grid thumbnail for. /// /// Joined against `remote` rather than left-joined: the store is keyed on /// Nextcloud's `oc:fileid` (FR-NC-5), so an image the scan recorded without /// one cannot be stored and is not work this pass can do. /// /// The whole list is built up front rather than re-queried per chunk, unlike /// the metadata sweep: "does the store have this" is answered by the store's /// index, which this thread is also the one writing, so a stale list is not a /// risk the way a concurrently-dating grid made it one there. fn thumbnails_outstanding( catalog: &Catalog, store: &ThumbStore, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT i.id, i.source_ref, r.file_id, i.file_size, i.metadata_state FROM images i JOIN remote r ON r.image_id = i.id WHERE r.file_id IS NOT NULL AND {VISIBLE} ORDER BY i.id" ))?; let rows = stmt .query_map([], |r| { let file_id = r.get::<_, Option>(2)?.map(|v| v as u64); Ok(ThumbnailRequest { thumb_size: SWEEP_THUMB_SIZE, // No grid cell is waiting on this, so nothing consumes the row. row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id, size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, // The header this fetch reads is the one EXIF lives in, so an // undated image is dated on the way past for nothing. needs_metadata: r.get::<_, i64>(4)? < 2, full_resolution: false, }) })? .filter_map(Result::ok) .filter(|req| { req.file_id .is_some_and(|id| !store.contains(id, SWEEP_THUMB_SIZE)) }) .collect(); Ok(rows) } /// Where an account's thumbnail shards live. /// /// Beside the catalog rather than in the cache directory: these sync to the /// server and are shared with other clients, so discarding them on a cache /// sweep would cost a re-download for everyone. pub fn thumbs_dir(account: &Account) -> PathBuf { catalog_path(account) .parent() .map(|p| p.join("thumbs")) .unwrap_or_else(|| std::env::temp_dir().join("darkroom-thumbs")) } /// Where the face models live, beside the catalog and the thumbnails. /// /// **Not shipped with the application** and not a build input: the InsightFace /// weights carry a non-commercial research grant incompatible with this /// project's licence, so the user obtains them and the app loads them from here /// (docs/faces.md §2). An absent directory is the ordinary state of a fresh /// install, not an error. pub fn face_models_dir(account: &Account) -> PathBuf { catalog_path(account) .parent() .map(|p| p.join("models")) .unwrap_or_else(|| std::env::temp_dir().join("darkroom-models")) } /// Where face models live for *every* account on this device. /// /// Account-independent, unlike the catalog: a model is identified by /// `faces.model_id` (catalog.md §10.1) and not by who is signed in, so two /// accounts have no reason to hold two 15 MB copies of the same weights. This /// is also the only directory an Android build can populate for itself — the /// entry point extracts the APK's bundled copy here, and no session exists at /// that point to key a per-account path off. /// /// **That extraction races the first seconds of a launch and is meant to.** It /// is 41 MB of copying and it used to happen before the first frame, which on a /// tablet is an ANR (`darkroom-android`'s `install_bundled_models`). So a /// lookup here can answer "absent" for a model that is on its way; each file is /// renamed into place, so what a lookup never sees is a half-written one. pub fn shared_face_models_dir() -> PathBuf { data_root().join("models") } /// The detector and embedder files, if both are present. /// /// Both or neither: an embedder with no detector has nothing to embed, and a /// detector with no embedder finds faces it cannot tell apart. Reporting the /// pair missing is more useful than half-starting. /// /// The names are the **shape-fixed** exports, not what InsightFace ships: /// `tools/fix-face-model-shapes.sh` has to run over the originals first, /// because tract cannot parse either graph with a dynamic input. /// /// Searched in three places, most specific first: /// /// 1. **The account's own directory.** A library can be pinned to its own /// weights — a model swap is a `model_id` change and a re-index, and someone /// mid-migration needs one account's pair to stay put without holding the /// other back. /// 2. **The shared user directory.** Where a user drops a pair by hand, and /// where the Android entry point unpacks the copy the APK carries. /// 3. **The system directories.** Where a package installs them — the Arch /// package puts the pair in `/usr/share/darkroom/models`. Last, so anything /// the user placed themselves outranks what the package shipped. pub fn face_models( account: &Account, detector: dr_types::FaceDetector, ) -> Option<(PathBuf, PathBuf)> { let pair = |dir: &PathBuf| { let detector = dir.join(detector.file_name()); let embedder = dir.join("arcface_mbf_b1.onnx"); (detector.is_file() && embedder.is_file()).then_some((detector, embedder)) }; let mut searched = vec![face_models_dir(account), shared_face_models_dir()]; searched.extend(system_face_models_dirs()); let found = searched.iter().find_map(pair); if found.is_none() { // The settings page can only say "not installed". This is the line // that says where it looked, which is the whole of what a user with // the files in the wrong place needs — and the first thing to read // when a freshly installed package reports no model. log::warn!( "face models: no directory holds both {} and arcface_mbf_b1.onnx; searched {}", detector.file_name(), searched .iter() .map(|d| d.display().to_string()) .collect::>() .join(", ") ); } found } /// The scene model, its vocabulary and its category descriptor, if all three /// are present. /// /// All three or none, for the same reason `face_models` insists on its pair: /// the graph alone decodes to 150 anonymous channels, and a descriptor naming /// classes a different model does not have is refused by /// `dr_segment::scene::parse_categories` anyway. Reporting the set missing is /// more useful than starting and failing at the first inference. /// /// Searched in the same three places, most specific first — the account's own /// directory, the shared one, then wherever a package installed them. Android /// only ever finds the second, which is where `install_bundled_models` unpacks /// the APK's copy before any store opens. /// /// Unlike the face weights this model *is* in the repository, so a desktop /// build from a complete checkout has it. Absent means either a checkout /// without `git lfs pull` or a package that chose not to carry 24 MB, and the /// scene tab reports itself unavailable rather than the app refusing to run. pub fn scene_model(account: &Account) -> Option<(PathBuf, PathBuf, PathBuf)> { fn set(dir: PathBuf) -> Option<(PathBuf, PathBuf, PathBuf)> { let model = dir.join("yolo26s-sem-ade20k.onnx"); let classes = dir.join("yolo26s-sem-ade20k.classes.json"); let categories = dir.join("categories.txt"); (model.is_file() && classes.is_file() && categories.is_file()) .then_some((model, classes, categories)) } set(face_models_dir(account)) .or_else(|| set(shared_face_models_dir())) .or_else(|| system_face_models_dirs().into_iter().find_map(set)) } /// Where a *package* may have installed the models. /// /// `dr_plat::system_data_dirs` has the rule per platform: `$XDG_DATA_DIRS` on /// Linux, the executable's own directory on Windows, nothing on Android — /// there the APK's copy is unpacked into the shared user directory instead, /// because an asset inside a package is not a path anything can read from /// (ARCH §6.9). Last in the search order on every platform, so a pair the /// user placed by hand outranks the installed one. fn system_face_models_dirs() -> Vec { dr_plat::system_data_dirs() .into_iter() .map(|d| d.join("models")) .collect() } /// One grid cell's data, read from the catalog. #[derive(Debug, Clone, PartialEq, Eq)] pub struct LibraryCell { pub image_id: i64, pub name: String, pub remote_path: String, /// `oc:fileid`, the key the shared thumbnail store uses. `None` for an /// image the scan found without a stable id. pub file_id: Option, /// File length, for bounds-checking a located preview range. pub size: u64, /// 0 = nothing, 1 = stat-only, 2 = full EXIF. pub metadata_state: u8, /// UTC seconds, once EXIF has been read. pub captured_at: Option, } /// Read a window of cells out of the catalog. /// /// Windowed rather than wholesale: a 17k-image library must not become 17k /// rows in a Slint model (FR-CAT-4). /// The unscoped form, kept as the name the tests and any future caller reach /// for. The UI goes through [`read_cells_scoped`], because a collection may be /// selected. #[cfg(test)] pub fn read_cells( catalog: &Catalog, offset: usize, limit: usize, ) -> Result, dr_catalog::CatalogError> { read_cells_scoped(catalog, None, &RatingFilter::default(), offset, limit) } /// Read a window of cells, optionally narrowed to one collection. /// /// A collection *set* shows its descendants' images too — a parent whose /// children hold everything would otherwise read as empty, which makes nesting /// look broken. The id list comes from /// [`dr_catalog::collections::descendants`], which is depth-guarded. /// /// Ordering comes from [`grid_order_for`]: manual position for a single manual /// collection, capture time for a set — because position is only meaningful /// inside one collection, and this query also serves sets, where two children's /// positions are unrelated integers. pub fn read_cells_scoped( catalog: &Catalog, scope: Option, filter: &RatingFilter, offset: usize, limit: usize, ) -> Result, dr_catalog::CatalogError> { let Some(scope) = scope else { return read_cells_all(catalog, filter, offset, limit); }; let ids = dr_catalog::collections::descendants(catalog.connection(), scope)?; // Placeholders are generated from the *count* of ids, never from user text. let placeholders = std::iter::repeat_n("?", ids.len()) .collect::>() .join(","); let rated = filter.sql(); let folded = uncollapsed("i"); let (order, order_params) = grid_order_for(catalog, Some(scope)); let sql = format!( "SELECT {CELL_COLUMNS} FROM images i WHERE {VISIBLE}{rated}{folded} AND i.id IN (SELECT image_id FROM collection_members WHERE collection_id IN ({placeholders})) {order} LIMIT ? OFFSET ?" ); // Bound in the order the `?`s appear: the scope's ids in the WHERE, then // whatever the ORDER BY needs, then the window. let mut params: Vec = ids .iter() .map(|c| rusqlite::types::Value::Integer(c.0 as i64)) .collect(); params.extend(order_params); params.push(rusqlite::types::Value::Integer(limit as i64)); params.push(rusqlite::types::Value::Integer(offset as i64)); let mut rows = { let mut stmt = catalog.connection().prepare(&sql)?; let read = stmt .query_map(rusqlite::params_from_iter(params.iter()), row_to_cell)? .collect::, _>>()?; read }; attach_file_ids(catalog, &mut rows); Ok(rows) } fn read_cells_all( catalog: &Catalog, filter: &RatingFilter, offset: usize, limit: usize, ) -> Result, dr_catalog::CatalogError> { let rated = filter.sql(); let folded = uncollapsed("i"); let mut rows = { let mut stmt = catalog.connection().prepare(&format!( "SELECT {CELL_COLUMNS} FROM images i WHERE {VISIBLE}{rated}{folded} {GRID_ORDER} LIMIT ?1 OFFSET ?2" ))?; let read = stmt .query_map(rusqlite::params![limit as i64, offset as i64], row_to_cell)? .collect::, _>>()?; read }; attach_file_ids(catalog, &mut rows); Ok(rows) } /// TRACES: FR-CAT-15 /// Read a window of *trashed* cells, newest deletion first. /// /// Ordered by when it was trashed rather than by capture time, which is what /// every other view sorts by. The question in the trash is "what did I just /// delete?", not "when was this taken" — a mistaken delete is corrected within /// seconds, and burying it among photographs from the same afternoon would make /// the one row the user is looking for the hardest one to find. /// /// The rating filter is deliberately not applied. It narrows a *culling* pass, /// and a trash that hid rows because of a filter set elsewhere would look like /// it had lost them. pub fn read_trashed_cells( catalog: &Catalog, offset: usize, limit: usize, ) -> Result, dr_catalog::CatalogError> { let mut rows = { let mut stmt = catalog.connection().prepare(&format!( "SELECT {CELL_COLUMNS} FROM images i WHERE {TRASHED} {TRASH_ORDER} LIMIT ?1 OFFSET ?2" ))?; let read = stmt .query_map(rusqlite::params![limit as i64, offset as i64], row_to_cell)? .collect::, _>>()?; read }; attach_file_ids(catalog, &mut rows); Ok(rows) } /// TRACES: FR-CAT-15 /// How many images the trash view would list. /// /// Counts exactly what [`read_trashed_cells`] lists — same predicate, no filter /// — so the scrollbar and the header cannot disagree with the cells. pub fn total_trashed(catalog: &Catalog) -> Result { let n: i64 = catalog.connection().query_row( &format!("SELECT count(*) FROM images i WHERE {TRASHED}"), [], |r| r.get(0), )?; Ok(n as usize) } /// TRACES: FR-CAT-5 /// The ids of grid rows `first..=last`, in the order the grid lists them. /// /// What a shift-click actually means. The gesture names two *ordinals* and the /// photographs between them are mostly not loaded — the grid is a window of a /// hundred or so over a library of twenty thousand — so a range answered from /// the window selected the handful on screen and silently dropped the rest. /// The catalog knows the whole run, and with the ordering indexed it is one /// seek rather than a scan. /// /// Bounded by the same predicates, the same filter and the same [`GRID_ORDER`] /// the window itself is read with. An ordinal only names a photograph relative /// to an ordering, so a range taken through any other one is a range through a /// different library. /// /// `trash` picks the trash view's list, which is the other thing the grid can /// be showing and is ordered by deletion time rather than capture time. /// /// An empty result means the run is empty or the query failed; callers treat /// the two the same, because both leave the selection where it was. pub fn read_ids_span( catalog: &Catalog, scope: Option, filter: &RatingFilter, trash: bool, first: usize, last: usize, ) -> Result, dr_catalog::CatalogError> { let Some(count) = (last + 1).checked_sub(first) else { return Ok(Vec::new()); }; let (sql, mut params) = if trash { ( format!("SELECT i.id FROM images i WHERE {TRASHED} {TRASH_ORDER} LIMIT ? OFFSET ?"), Vec::new(), ) } else { let (clause, mut params) = scope_clause(catalog, scope)?; let rated = filter.sql(); let folded = uncollapsed("i"); // The same ordering the cells were drawn with, from the same place. // A range is a pair of ordinals, and an ordinal read through a // different ORDER BY names a different photograph. let (order, order_params) = grid_order_for(catalog, scope); params.extend(order_params); ( format!( "SELECT i.id FROM images i WHERE {VISIBLE}{rated}{folded}{clause} {order} LIMIT ? OFFSET ?" ), params, ) }; params.push(rusqlite::types::Value::Integer(count as i64)); params.push(rusqlite::types::Value::Integer(first as i64)); let mut stmt = catalog.connection().prepare(&sql)?; let ids = stmt .query_map(rusqlite::params_from_iter(params.iter()), |r| { Ok(dr_types::ImageId(r.get::<_, i64>(0)? as u64)) })? .collect::, _>>()?; Ok(ids) } /// TRACES: FR-UI-8 /// Where one photograph sits in the grid, by its remote path. /// /// The inverse of [`read_ids_span`], and it exists for the same reason that one /// does: an ordinal only names a photograph relative to an ordering, so it has /// to be *computed* through the ordering the cells are drawn with rather than /// guessed at. A restored place that landed on a row read through a different /// `ORDER BY` would open the library at a photograph the user has never seen, /// which looks exactly like the position having been forgotten. /// /// # Why a window function rather than a count /// /// The obvious implementation is "count the rows that sort before this one", /// and it would need this file to spell the ordering out a second time — as an /// inequality, with its own handling of the `captured_at IS NULL` term and its /// own tie-break. [`grid_order_for`] already warns what a second spelling /// costs, and a manually ordered collection makes it worse: that ordering is a /// correlated subquery, and an inequality over it is not something anyone /// should have to read. /// /// `row_number() OVER ({order})` takes the ordering *verbatim* from the same /// function the window read uses, so the two cannot disagree by construction. /// It is a full pass over the scope rather than an index seek, which is the /// cost of that guarantee — and it is paid once, at launch, against a query the /// grid runs several times per screenful of scrolling. /// /// `Ok(None)` means the path is not in this grid: deleted, trashed, filtered /// out, or in a collection the place did not name. The caller falls back to /// when the photograph was taken, which is what makes a place survive its /// subject. /// /// # Binding order /// /// The window's parameters come first, unlike in [`read_ids_span`]. SQLite /// binds anonymous `?` by their position **in the SQL text**, and here the /// `OVER (...)` clause is in the select list — ahead of the `WHERE` the scope /// narrows. Swapping the two silently looks up a collection by an image id. pub fn ordinal_of_path( catalog: &Catalog, scope: Option, filter: &RatingFilter, trash: bool, path: &str, ) -> Result, dr_catalog::CatalogError> { let (inner, mut params) = if trash { ( format!( "SELECT i.source_ref AS sref, row_number() OVER ({TRASH_ORDER}) - 1 AS ord FROM images i WHERE {TRASHED}" ), Vec::new(), ) } else { let (clause, scope_params) = scope_clause(catalog, scope)?; let rated = filter.sql(); let folded = uncollapsed("i"); let (order, mut params) = grid_order_for(catalog, scope); // The window first, then the scope — see the note above. params.extend(scope_params); ( format!( "SELECT i.source_ref AS sref, row_number() OVER ({order}) - 1 AS ord FROM images i WHERE {VISIBLE}{rated}{folded}{clause}" ), params, ) }; params.push(rusqlite::types::Value::Text(path.to_string())); let mut stmt = catalog .connection() .prepare(&format!("SELECT ord FROM ({inner}) WHERE sref = ?"))?; let mut rows = stmt.query(rusqlite::params_from_iter(params.iter()))?; match rows.next()? { Some(r) => Ok(Some(r.get::<_, i64>(0)?.max(0) as usize)), None => Ok(None), } } /// The columns every windowed read selects, in the order [`row_to_cell`] reads /// them. /// /// Named rather than repeated so the three readers cannot drift — and so that /// the one column that is *not* here stays conspicuous. See /// [`attach_file_ids`] for why the server's file id is fetched separately. const CELL_COLUMNS: &str = "i.id, i.source_ref, i.file_size, i.metadata_state, i.captured_at"; /// Shared row mapping, so the scoped and unscoped queries cannot drift. /// /// `file_id` is left empty here and filled by [`attach_file_ids`]. fn row_to_cell(r: &rusqlite::Row) -> rusqlite::Result { let path: String = r.get(1)?; Ok(LibraryCell { image_id: r.get(0)?, name: path.rsplit(['/', ':']).next().unwrap_or(&path).to_string(), remote_path: path, file_id: None, size: r.get::<_, Option>(2)?.unwrap_or(0) as u64, metadata_state: r.get::<_, i64>(3)? as u8, captured_at: r.get(4)?, }) } /// TRACES: NFR-P5 /// Fill in each cell's server file id, in one query for the whole window. /// /// # Why this is not a `LEFT JOIN` any more /// /// It was, and it was the single most expensive thing the grid did while a /// finger was on it. A window is `ORDER BY ... LIMIT n OFFSET k`, and SQLite /// answers a join like that by joining *first* and paging after — so reading /// 280 cells at offset 20,000 meant an index seek into `remote` for all 24,000 /// rows, 23,720 of which were then discarded. Measured at 15.2 ms, inside the /// scroll handler, against a 16.7 ms frame. /// /// Paging over `images` alone is 0.36 ms with `images_grid_order` (schema V7), /// and this fetches the ids for the 280 rows that survived. The same shape the /// badge and rating reads already use: one query for the window, never one per /// cell. /// /// Silent on failure, and cells keep `file_id: None`: that is the same state a /// photograph the scan has not reached the server for is in, and the callers /// already treat it as "no cached thumbnail to key on" rather than an error. fn attach_file_ids(catalog: &Catalog, cells: &mut [LibraryCell]) { if cells.is_empty() { return; } let placeholders = std::iter::repeat_n("?", cells.len()) .collect::>() .join(","); let sql = format!("SELECT image_id, file_id FROM remote WHERE image_id IN ({placeholders})"); let params: Vec = cells .iter() .map(|c| rusqlite::types::Value::Integer(c.image_id)) .collect(); let mut stmt = match catalog.connection().prepare(&sql) { Ok(s) => s, Err(e) => { log::debug!("reading file ids for the window: {e}"); return; } }; let rows = stmt.query_map(rusqlite::params_from_iter(params.iter()), |r| { Ok((r.get::<_, i64>(0)?, r.get::<_, Option>(1)?)) }); let found: std::collections::HashMap> = match rows { Ok(rows) => rows.flatten().collect(), Err(e) => { log::debug!("reading file ids for the window: {e}"); return; } }; for cell in cells { cell.file_id = found .get(&cell.image_id) .copied() .flatten() .map(|v| v as u64); } } /// Total images in the catalog, or in one collection and its descendants. /// /// Counts exactly what [`read_cells_scoped`] would list, filter included. The /// two must agree: the header says "412 images" and the grid's scrollbar is /// sized from the same number, so a count that ignored the filter would leave /// the user scrolling through empty rows. pub fn total_images_scoped( catalog: &Catalog, scope: Option, filter: &RatingFilter, ) -> Result { let Some(scope) = scope else { return total_images_filtered(catalog, filter); }; let ids = dr_catalog::collections::descendants(catalog.connection(), scope)?; let placeholders = std::iter::repeat_n("?", ids.len()) .collect::>() .join(","); let rated = filter.sql(); // DISTINCT: an image in both a parent and a child is one photograph, and a // count that disagrees with the number of cells drawn is worse than either // number alone. // // Counted through `images` rather than over `collection_members` alone, so // `VISIBLE` applies — a trashed photograph is still a member row, and // counting it made the header claim images the grid would not draw. let folded = uncollapsed("i"); let sql = format!( "SELECT count(DISTINCT i.id) FROM images i WHERE {VISIBLE}{rated}{folded} AND i.id IN (SELECT image_id FROM collection_members WHERE collection_id IN ({placeholders}))" ); let params: Vec = ids .iter() .map(|c| rusqlite::types::Value::Integer(c.0 as i64)) .collect(); let n: i64 = catalog .connection() .query_row(&sql, rusqlite::params_from_iter(params.iter()), |r| { r.get(0) })?; Ok(n as usize) } /// TRACES: FR-CAT-7 /// Every member of `scope`, in the order its positions put them. /// /// The *whole* membership, not the window and not the filtered view. A reorder /// rewrites positions, and [`dr_catalog::collections::set_order`] only touches /// the rows it is given — so writing back a filtered subset would leave the /// images the filter is hiding at their old positions, interleaved with the new /// ones arbitrarily. The user reorders what they can see; the rows they cannot /// keep their place relative to it. pub fn read_member_order( catalog: &Catalog, scope: dr_types::CollectionId, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare( "SELECT image_id FROM collection_members WHERE collection_id = ?1 ORDER BY position ASC, image_id ASC", )?; let ids = stmt .query_map([scope.0 as i64], |r| { Ok(dr_types::ImageId(r.get::<_, i64>(0)? as u64)) })? .collect::, _>>()?; Ok(ids) } /// TRACES: FR-CAT-7 /// `current` with `moving` lifted out and set down beside `target`. /// /// `target` names a *photograph*, not an index, and that is the point: the grid /// may be filtered, so the cell the user dropped on sits at one position in /// what they can see and another in the membership being rewritten. An id /// survives both. `after` puts the run on the far side of it, which is the only /// way to name the last place in a collection — there is no cell beyond the /// last one to drop in front of. /// /// The run keeps the order `current` has it in rather than the order the /// selection was built in: the user is looking at the grid, and a selection /// gathered by tapping the last frame first should not reverse itself on being /// moved. /// /// A `target` that is itself being moved leaves the run at the end. There is no /// gap between a run and itself to land in, so the caller refuses that drop /// before it gets here; this is what the function does rather than panicking if /// one ever arrives. /// /// Pure, so the awkward half of a drag can be tested without a window. pub fn reordered( current: &[dr_types::ImageId], moving: &[dr_types::ImageId], target: dr_types::ImageId, after: bool, ) -> Vec { let lifting: std::collections::BTreeSet<_> = moving.iter().copied().collect(); let rest: Vec<_> = current .iter() .copied() .filter(|id| !lifting.contains(id)) .collect(); let run: Vec<_> = current .iter() .copied() .filter(|id| lifting.contains(id)) .collect(); // Resolved against `rest`, not against `current`: the run has already been // lifted, so an index into the original list would be off by however many // of it sat ahead of the target. let at = match rest.iter().position(|id| *id == target) { Some(at) if after => at + 1, Some(at) => at, None => rest.len(), }; let mut out = Vec::with_capacity(current.len()); out.extend_from_slice(&rest[..at]); out.extend(run); out.extend_from_slice(&rest[at..]); out } /// TRACES: FR-CAT-7 /// The ORDER BY the grid reads `scope` with, and the parameters it binds. /// /// Manual position where the grid is scoped to a single manual collection with /// no children; [`GRID_ORDER`] — capture time, then filename — everywhere else. /// /// **Why the narrowing.** `position` is a column of `collection_members`, so it /// only exists relative to one collection. A collection *set* shows its /// descendants' images too, and two children's positions are unrelated integers /// that would interleave arbitrarily; a smart collection has no member rows to /// carry a position at all. Outside those cases there is no manual order to /// read, and falling back is the only honest answer. /// /// **Why every reader must agree.** An ordinal only names a photograph relative /// to an ordering. The window read and the span read are two halves of one /// grid: a shift-click resolved through a different ORDER BY than the cells /// were drawn with selects a different run than the one on screen, and the user /// finds out when the export runs. That is the same invariant /// [`read_ids_span`] already states about `GRID_ORDER`, widened to cover the /// case where the ordering depends on the scope. /// /// A correlated subquery rather than a join, so the FROM and WHERE the two /// readers already share are untouched: position is looked up per row through /// `collection_members`' primary key, which is `(collection_id, image_id)`. fn grid_order_for( catalog: &Catalog, scope: Option, ) -> (String, Vec) { let Some(id) = scope else { return (GRID_ORDER.to_string(), Vec::new()); }; // A set orders by capture time. `descendants` includes the collection // itself, so one entry means it has no children. let alone = dr_catalog::collections::descendants(catalog.connection(), id) .map(|d| d.len() == 1) .unwrap_or(false); let manual = matches!( dr_catalog::collections::kind(catalog.connection(), id), Ok(Some(dr_catalog::collections::CollectionKind::Manual)) ); if !alone || !manual { return (GRID_ORDER.to_string(), Vec::new()); } // `i.id` breaks the tie. Positions are dense after a `set_order`, but a // collection that has never been reordered by hand has whatever // `add_images` assigned, and two rows can share a position if a merge from // another device brought one in — an ordering that is not total is an // ordering the window read and the span read can disagree about. ( "ORDER BY (SELECT cm.position FROM collection_members cm WHERE cm.collection_id = ? AND cm.image_id = i.id) ASC, i.id ASC" .to_string(), vec![rusqlite::types::Value::Integer(id.0 as i64)], ) } /// The SQL restricting a query to `scope` and its descendants, with the bound /// parameters to go with it. /// /// Shared by the span and the histogram so the two cannot drift: an axis drawn /// over one set of images and bars counted over another puts the bars in the /// wrong place. fn scope_clause( catalog: &Catalog, scope: Option, ) -> Result<(String, Vec), dr_catalog::CatalogError> { let Some(scope) = scope else { return Ok((String::new(), Vec::new())); }; let ids = dr_catalog::collections::descendants(catalog.connection(), scope)?; let placeholders = std::iter::repeat_n("?", ids.len()) .collect::>() .join(","); Ok(( format!( " AND i.id IN (SELECT image_id FROM collection_members WHERE collection_id IN ({placeholders}))" ), ids.iter() .map(|c| rusqlite::types::Value::Integer(c.0 as i64)) .collect(), )) } /// Earliest and latest capture time within `scope`, honouring the filter. /// /// The timeline's extent. Taken over the same images the histogram counts, so /// opening a collection shows that collection's years rather than the whole /// library's — the axis was previously spanning everything, which left a /// collection's bars crushed into a sliver of it. pub fn span_scoped( catalog: &Catalog, scope: Option, filter: &RatingFilter, ) -> Option<(i64, i64)> { let (clause, params) = scope_clause(catalog, scope).ok()?; // Full extent, not the chosen range — see `without_date_range`. let rated = filter.without_date_range().sql(); let sql = format!( "SELECT min(i.captured_at), max(i.captured_at) FROM images i WHERE {VISIBLE}{rated} AND i.captured_at IS NOT NULL{clause}" ); catalog .connection() .query_row(&sql, rusqlite::params_from_iter(params.iter()), |r| { Ok((r.get::<_, Option>(0)?, r.get::<_, Option>(1)?)) }) .ok() .and_then(|(lo, hi)| Some((lo?, hi?))) } /// TRACES: FR-CAT-6 /// The capture-time histogram as a fixed number of equal bins across /// `from..=to`, empty ones included. /// /// # Why not calendar buckets /// /// [`timeline_scoped`] groups by year, month, day or hour, which has two /// consequences the axis cannot live with. /// /// It emits **only the buckets that hold photographs**, and the widget gives /// every bar an equal slot — so a library with a gap in it drew a February /// that was six months wide. The position marker, the range band and a click /// are all linear in time, so on a sparse library they pointed at bars that /// were somewhere else. Equal bins including the empty ones make a bar's /// position on the track and the date under it the same quantity. /// /// And the **count is free to jump by a factor of twelve** between one unit /// and the next, so each zoom step halved the number of bars until a /// threshold was crossed: zooming in made the picture coarser, twice out of /// every three steps. A fixed count re-bins on every zoom instead, which is /// what makes each step show finer structure rather than the same structure /// drawn wider. /// /// The date range is lifted from the filter, like the bars' other terms are /// not: this histogram is *how a range is chosen*, and drawing through the /// range would empty every bin outside it and leave nothing to widen into. /// /// `bins` is clamped to at least one — a zero would be a division by zero in /// SQL, and the caller's number comes from a hand-editable settings file. pub fn timeline_uniform( catalog: &Catalog, scope: Option, filter: &RatingFilter, from: i64, to: i64, bins: u32, ) -> Result, dr_catalog::CatalogError> { let bins = bins.max(1) as i64; // At least one second, or every photograph lands in bin zero. let span = (to - from).max(1); let (clause, params) = scope_clause(catalog, scope)?; let rated = filter.without_date_range().sql(); // The bin index is arithmetic on the stored UTC instant, not `strftime` on // a local one. A bin is not a calendar unit — it has no local midnight to // respect — and the axis it is drawn on is labelled from the same UTC // instants, so bucketing the two differently is the one way they could // disagree about which bar a photograph belongs to. // // `min` caps the last edge: an image captured at exactly `to` divides to // `bins`, which would be a bin past the end of the axis. // // Integers this code owns, formatted straight in — the same rule the // rating terms follow. They cannot be bound parameters here without // ordering them against the scope clause's, which appears later in the // text but is bound first. let sql = format!( "SELECT min({bins} - 1, (i.captured_at - {from}) * {bins} / {span}) AS b, count(*) AS n FROM images i WHERE {VISIBLE}{rated} AND i.captured_at IS NOT NULL{clause} AND i.captured_at >= {from} AND i.captured_at <= {to} GROUP BY b ORDER BY b ASC" ); let conn = catalog.connection(); let mut stmt = conn.prepare(&sql)?; let counted = stmt .query_map(rusqlite::params_from_iter(params.iter()), |r| { Ok((r.get::<_, i64>(0)?, r.get::<_, i64>(1)? as u32)) })? .collect::, _>>()?; // Every bin, in order, whether or not the query returned one for it. The // start is the bin's own left edge rather than the earliest photograph in // it: an empty bin has no photograph to take one from, and a bar drawn at // its contents' position rather than its bin's would put the axis back // where the calendar buckets left it. let mut bars: Vec = (0..bins) .map(|i| dr_catalog::TimeBucket { start: from + (i * span) / bins, count: 0, }) .collect(); for (i, n) in counted { if let Some(bar) = bars.get_mut(i.clamp(0, bins - 1) as usize) { bar.count = n; } } Ok(bars) } /// Total images in the catalog, honouring the rating filter. fn total_images_filtered( catalog: &Catalog, filter: &RatingFilter, ) -> Result { let rated = filter.sql(); let folded = uncollapsed("i"); let n: i64 = catalog.connection().query_row( &format!("SELECT count(*) FROM images i WHERE {VISIBLE}{rated}{folded}"), [], |r| r.get(0), )?; Ok(n as usize) } /// TRACES: FR-CAT-9 /// How many visible images have their original stored on this device. /// /// Whole-library, like the star counts beside it: the chip says what narrowing /// to it would show, so counting only the current window would make it /// describe the view it exists to change. pub fn local_original_count(catalog: &Catalog) -> Result { let n: i64 = catalog.connection().query_row( &format!( "SELECT count(*) FROM images i WHERE {VISIBLE} AND EXISTS (SELECT 1 FROM image_cache ic WHERE ic.image_id = i.id AND ic.tier_actual >= {})", dr_types::Tier::Original.stored() ), [], |r| r.get(0), )?; Ok(n as usize) } /// Total images in the catalog, unfiltered. /// /// What the scan reports and what the sidebar's "all images" row shows — the /// size of the library itself, not of the current view. pub fn total_images(catalog: &Catalog) -> Result { let n: i64 = catalog.connection().query_row( &format!("SELECT count(*) FROM images WHERE {VISIBLE_UNALIASED}"), [], |r| r.get(0), )?; Ok(n as usize) } pub fn now_secs() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map(|d| d.as_secs() as i64) .unwrap_or(0) } #[cfg(test)] mod tests { use super::*; use dr_sync::RemoteEntry; #[test] fn join_all_preserves_order_regardless_of_completion() { // The ordering guarantee is what lets a caller pair results back to // their inputs; without it a lane's dates could be attributed to the // wrong images. let rt = crate::net_runtime::build().unwrap(); let out = rt.block_on(async { futures_join_all(vec![ Box::pin(async { 1 }) as std::pin::Pin>>, Box::pin(async { tokio::task::yield_now().await; tokio::task::yield_now().await; 2 }), Box::pin(async { tokio::task::yield_now().await; 3 }), ]) .await }); assert_eq!(out, vec![1, 2, 3]); } #[test] fn join_all_of_nothing_completes() { let rt = crate::net_runtime::build().unwrap(); let out: Vec = rt.block_on(async { futures_join_all(Vec::>::new()).await }); assert!(out.is_empty()); } #[test] fn sweep_lanes_divide_a_chunk_without_loss() { // Every image in a chunk must land in exactly one lane: a striding // split that dropped or duplicated one would silently under- or // double-index the library. let chunk: Vec = (0..SWEEP_CHUNK).collect(); let lanes: Vec> = (0..SWEEP_LANES) .map(|l| chunk.iter().skip(l).step_by(SWEEP_LANES).copied().collect()) .collect(); let mut seen: Vec = lanes.iter().flatten().copied().collect(); seen.sort_unstable(); assert_eq!(seen, chunk); // Evenly divided, so no lane sits idle while another finishes. assert!(lanes.iter().all(|l| l.len() == SWEEP_CHUNK / SWEEP_LANES)); } #[test] fn a_short_chunk_still_covers_every_image() { // The last chunk of a library is rarely a full multiple of the lanes. let chunk: Vec = (0..5).collect(); let lanes: Vec> = (0..SWEEP_LANES) .map(|l| chunk.iter().skip(l).step_by(SWEEP_LANES).copied().collect()) .collect(); let mut seen: Vec = lanes.iter().flatten().copied().collect(); seen.sort_unstable(); assert_eq!(seen, chunk); } #[test] fn a_scrub_ordinal_matches_the_grid_position() { // The scrub's count and the grid's window must use *identical* // predicates and ordering, or the view lands somewhere else. Counting // only dated images against a grid that also shows undated ones put a // click near the end of the axis near the top of the library. let catalog = Catalog::in_memory().unwrap(); let c = catalog.connection(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); // A mix: dated, undated, and one shadowed by a RAW sibling. for (id, name, captured, shadow) in [ (1i64, "a.CR2", Some(100i64), None), (2, "b.CR2", Some(200), None), (3, "b.JPG", Some(200), Some(2i64)), (4, "c.CR2", Some(300), None), (5, "d.CR2", None, None), ] { c.execute( "INSERT INTO images(id, root_id, source_ref, captured_at, shadowed_by, added_at) VALUES (?1, 1, ?2, ?3, ?4, 0)", rusqlite::params![id, name, captured, shadow], ) .unwrap(); } // The grid's own window, in its own order. let cells = read_cells(&catalog, 0, 100).unwrap(); let names: Vec<&str> = cells.iter().map(|c| c.name.as_str()).collect(); assert_eq!( names, vec!["a.CR2", "b.CR2", "c.CR2", "d.CR2"], "shadowed hidden, undated last" ); // Scrubbing to each image's instant must give its index in that list. for (when, expected) in [(100i64, 0usize), (200, 1), (300, 2)] { let ordinal: i64 = c .query_row( "SELECT count(*) FROM images WHERE shadowed_by IS NULL AND captured_at IS NOT NULL AND captured_at < ?1", [when], |r| r.get(0), ) .unwrap(); assert_eq!( ordinal as usize, expected, "scrubbing to {when} must land on grid row {expected}" ); } } /// The fixture the ordinal tests share: a mix of dated, undated, shadowed /// and trashed rows, which is what makes the grid's ordering non-obvious. fn a_small_library() -> Catalog { let catalog = Catalog::in_memory().unwrap(); let c = catalog.connection(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); for (id, name, captured, shadow, trashed) in [ (1i64, "2019/a.CR2", Some(100i64), None, None), (2, "2019/b.CR2", Some(200), None, None), // Shadowed by its RAW sibling: never a row of the grid. (3, "2019/b.JPG", Some(200), Some(2i64), None), (4, "2020/c.CR2", Some(300), None, None), // Undated sorts last, whatever its name. (5, "2018/d.CR2", None, None, None), // Trashed: out of the library, and the only row of the trash. (6, "2020/e.CR2", Some(400), None, Some(9i64)), ] { c.execute( "INSERT INTO images(id, root_id, source_ref, captured_at, shadowed_by, trashed_at, added_at) VALUES (?1, 1, ?2, ?3, ?4, ?5, 0)", rusqlite::params![id, name, captured, shadow, trashed], ) .unwrap(); } catalog } #[test] fn every_cell_reports_the_ordinal_it_is_drawn_at() { // TRACES: FR-UI-8 // The invariant the whole restore rests on, stated the strongest way // there is: walk the window the grid actually draws and ask for each // cell's ordinal by path. Anything less — a spot check, or a hand-built // expectation — would pass while `ordinal_of_path` and `read_cells` // disagreed about undated rows, shadowed rows or the tie-break, which // is precisely where an ordering drifts. let catalog = a_small_library(); let filter = RatingFilter::default(); let cells = read_cells(&catalog, 0, 100).unwrap(); assert_eq!(cells.len(), 4, "shadowed and trashed are not rows"); for (i, cell) in cells.iter().enumerate() { assert_eq!( ordinal_of_path(&catalog, None, &filter, false, &cell.remote_path).unwrap(), Some(i), "{} is drawn at row {i}", cell.remote_path ); } } #[test] fn a_photograph_that_is_not_in_the_grid_reports_nothing() { // Deleted, never scanned, or hidden behind its RAW. All three are the // same answer, and the caller falls back to the capture time — which is // only reachable if this says `None` rather than guessing. let catalog = a_small_library(); let filter = RatingFilter::default(); assert_eq!( ordinal_of_path(&catalog, None, &filter, false, "2019/gone.CR2").unwrap(), None, "never heard of it" ); assert_eq!( ordinal_of_path(&catalog, None, &filter, false, "2019/b.JPG").unwrap(), None, "shadowed by its RAW" ); assert_eq!( ordinal_of_path(&catalog, None, &filter, false, "2020/e.CR2").unwrap(), None, "trashed" ); } #[test] fn the_trash_is_ordinalled_through_its_own_list() { // The trash orders by deletion time and lists exactly what the library // excludes, so an ordinal taken through the library's ordering would // name a different photograph — or, here, nothing at all. let catalog = a_small_library(); let filter = RatingFilter::default(); assert_eq!( ordinal_of_path(&catalog, None, &filter, true, "2020/e.CR2").unwrap(), Some(0) ); assert_eq!( ordinal_of_path(&catalog, None, &filter, true, "2019/a.CR2").unwrap(), None, "a live photograph is not in the trash" ); } #[test] fn a_filter_moves_the_ordinal_with_the_grid() { // TRACES: FR-UI-8 // A place carries the filter it was recorded under *and* the position, // and the second is only meaningful under the first. Restoring them in // the wrong order — position, then filter — would land on a row of a // list that no longer exists, which is the bug this pairing exists to // rule out. let catalog = a_small_library(); let c = catalog.connection(); // Three stars on the third photograph only. c.execute( "INSERT INTO versions(id, image_id, uuid, name, is_default, rating) VALUES (1, 4, 'u4', 'default', 1, 3)", [], ) .unwrap(); let unfiltered = RatingFilter::default(); assert_eq!( ordinal_of_path(&catalog, None, &unfiltered, false, "2020/c.CR2").unwrap(), Some(2) ); let starred = RatingFilter { min_rating: 3, ..RatingFilter::default() }; assert_eq!( ordinal_of_path(&catalog, None, &starred, false, "2020/c.CR2").unwrap(), Some(0), "the only survivor of the filter is the first row of it" ); assert_eq!( ordinal_of_path(&catalog, None, &starred, false, "2019/a.CR2").unwrap(), None, "filtered out, so it has no position in this grid" ); } #[test] fn a_collection_is_ordinalled_through_its_own_scope() { // The window's parameters are bound ahead of the scope's — see the note // on `ordinal_of_path`. Getting that order wrong looks up a collection // by an image id, which fails silently as "not in this grid". let catalog = a_small_library(); let c = catalog.connection(); let coll = dr_catalog::collections::create(c, "Trip", None, dr_catalog::CollectionKind::Manual) .unwrap(); // The second and third photographs, out of order, so position matters. dr_catalog::collections::add_images(c, coll, &[dr_types::ImageId(4), dr_types::ImageId(2)]) .unwrap(); let filter = RatingFilter::default(); let cells = read_cells_scoped(&catalog, Some(coll), &filter, 0, 100).unwrap(); assert_eq!(cells.len(), 2); for (i, cell) in cells.iter().enumerate() { assert_eq!( ordinal_of_path(&catalog, Some(coll), &filter, false, &cell.remote_path).unwrap(), Some(i), "{} is drawn at row {i} of the collection", cell.remote_path ); } assert_eq!( ordinal_of_path(&catalog, Some(coll), &filter, false, "2019/a.CR2").unwrap(), None, "not a member" ); } #[test] fn the_thumbnail_pass_asks_only_for_what_is_missing() { // The work list is the whole point of the pass being resumable and of // it being safe to press twice: it is derived from what the store // lacks, not from a flag in the catalog. Three things it must respect // — a thumbnail already stored, a trashed or shadowed image, and an // image with no `oc:fileid`, which the store cannot key on at all. let catalog = Catalog::in_memory().unwrap(); let c = catalog.connection(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); // (id, name, file_id, shadowed_by, trashed_at) for (id, name, file_id, shadow, trashed) in [ (1i64, "a.CR2", Some(11i64), None, None), (2, "b.CR2", Some(22), None, None), (3, "b.JPG", Some(33), Some(2i64), None), (4, "c.CR2", Some(44), None, Some(1000i64)), // Scanned without a file id: nothing to key the store on. (5, "d.CR2", None, None, None), ] { c.execute( "INSERT INTO images(id, root_id, source_ref, shadowed_by, trashed_at, added_at) VALUES (?1, 1, ?2, ?3, ?4, 0)", rusqlite::params![id, name, shadow, trashed], ) .unwrap(); if let Some(file_id) = file_id { c.execute( "INSERT INTO remote(image_id, file_id) VALUES (?1, ?2)", rusqlite::params![id, file_id], ) .unwrap(); } } let dir = std::env::temp_dir().join(format!("dr-thumb-sweep-{}", std::process::id())); let _ = std::fs::remove_dir_all(&dir); std::fs::create_dir_all(&dir).unwrap(); let mut store = ThumbStore::open(&dir).unwrap(); let all = thumbnails_outstanding(&catalog, &store).unwrap(); let names: Vec<&str> = all.iter().map(|r| r.path.as_str()).collect(); assert_eq!( names, vec!["a.CR2", "b.CR2"], "shadowed, trashed and file-id-less images are not work this pass can do" ); // Store one, and it drops out — this is what stops a second run // re-fetching an hour of previews. store .put( 11, SWEEP_THUMB_SIZE, &dr_thumbs::Thumbnail { width: 4, height: 4, bytes: vec![0xFF, 0xD8, 0xFF, 0xD9], }, ) .unwrap(); let rest = thumbnails_outstanding(&catalog, &store).unwrap(); let names: Vec<&str> = rest.iter().map(|r| r.path.as_str()).collect(); assert_eq!(names, vec!["b.CR2"]); // The large class is a different key, so filling the grid class does // not make the pass think the library is done at another size. assert!(!store.contains(11, dr_thumbs::ThumbSize::Large)); let _ = std::fs::remove_dir_all(&dir); } /// An account for the path tests, defaulting to the connector every /// existing install uses. fn account(endpoint: &str, user: &str) -> Account { Account::new("nextcloud", endpoint).with_login(user, user) } #[test] fn catalog_paths_separate_accounts() { // Two accounts on one machine must not share an index, or one // library's images appear in the other. let a = catalog_path(&account("https://cloud.example", "duncan")); let b = catalog_path(&account("https://cloud.example", "someone")); let c = catalog_path(&account("https://other.example", "duncan")); assert_ne!(a, b); assert_ne!(a, c); } #[test] fn a_folder_library_gets_its_own_catalog() { // The same rule across backends: a folder library on this machine // must not land in the directory a server account is already using. let server = catalog_path(&account("https://cloud.example", "duncan")); let folder = catalog_path(&Account::new("folder", "/mnt/photos")); assert_ne!(server, folder); assert_ne!( folder, catalog_path(&Account::new("folder", "/mnt/other-photos")) ); } #[test] fn a_legacy_cache_directory_is_moved_rather_than_abandoned() { // The upgrade hazard: `sidecars/` and `outbox/` hold work that exists // nowhere else, so leaving them behind in a directory the system may // empty would discard unsynced ratings and edits as a side effect of // installing a new build. let root = std::env::temp_dir().join(format!("dr-migrate-{}", std::process::id())); let _ = std::fs::remove_dir_all(&root); let legacy = root.join("darkroom").join("cloud-example-duncan"); std::fs::create_dir_all(legacy.join("sidecars")).unwrap(); std::fs::write(legacy.join("catalog.sqlite"), b"catalog").unwrap(); std::fs::write(legacy.join("sidecars").join("a.drsc"), b"an unsynced edit").unwrap(); let current = root.join("new").join("cloud-example-duncan"); move_account_dir(&legacy, ¤t); assert!(!legacy.exists(), "the old copy must not be left behind"); assert_eq!( std::fs::read(current.join("catalog.sqlite")).unwrap(), b"catalog" ); assert_eq!( std::fs::read(current.join("sidecars").join("a.drsc")).unwrap(), b"an unsynced edit", "an unsynced edit must survive the move" ); let _ = std::fs::remove_dir_all(&root); } #[test] fn a_migration_never_overwrites_live_data() { // Running twice, or a fresh install that already has a catalog. The // destination wins: it is the one the application is using. let root = std::env::temp_dir().join(format!("dr-migrate2-{}", std::process::id())); let _ = std::fs::remove_dir_all(&root); let legacy = root.join("old").join("acct"); let current = root.join("new").join("acct"); std::fs::create_dir_all(&legacy).unwrap(); std::fs::create_dir_all(¤t).unwrap(); std::fs::write(legacy.join("catalog.sqlite"), b"stale").unwrap(); std::fs::write(current.join("catalog.sqlite"), b"live").unwrap(); move_account_dir(&legacy, ¤t); assert_eq!( std::fs::read(current.join("catalog.sqlite")).unwrap(), b"live" ); let _ = std::fs::remove_dir_all(&root); } #[test] fn durable_data_never_lands_in_a_cache_directory() { // The fault this guards against is silent and total: on Android the // fallback used to be `temp_dir()`, which resolves to the app's cache // — a directory the system empties under storage pressure. Beside this // catalog sit `sidecars/`, the commit point for every offline rating // and edit, and `outbox/`, holding exports the user was told had // succeeded. Losing a day of culling to an OS housekeeping pass, with // no error and no trace, is the worst outcome this application has. let path = catalog_path(&account("https://cloud.example", "duncan")); let text = path.to_string_lossy().to_lowercase(); assert!( !text.contains("/cache/") && !text.contains("/tmp/"), "the catalog and everything beside it must be durable, got {}", path.display() ); } #[test] fn the_outbox_and_sidecars_sit_beside_the_catalog() { // Stated as a test because three separate call sites derive their // location by taking this path's parent, and a change here moves all // of them at once — including the two holding unsynced user work. let catalog = catalog_path(&account("https://cloud.example", "duncan")); let parent = catalog.parent().expect("a parent"); assert_eq!( crate::export::outbox_dir(&account("https://cloud.example", "duncan")), parent.join("outbox") ); } #[test] fn catalog_path_is_filesystem_safe() { let p = catalog_path(&account("https://cloud.example.com:8443/nc", "duncan")); let s = p.to_string_lossy(); assert!(!s.contains("://")); assert!(!s.contains(':') || cfg!(windows)); } #[test] fn persist_inserts_images_and_folder_etags() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 1001, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); assert_eq!(total_images(&catalog).unwrap(), 1); // Folder ETags must persist or the next scan prunes nothing. let etag: String = catalog .connection() .query_row( "SELECT etag FROM folders WHERE path = 'PhotosRaw'", [], |r| r.get(0), ) .unwrap(); assert_eq!(etag, "e1"); } #[test] fn images_land_as_stat_only_not_full_metadata() { // The scan read no EXIF. Claiming otherwise would make a date filter // silently wrong on a freshly scanned library. let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 1001, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let state: i64 = catalog .connection() .query_row("SELECT metadata_state FROM images", [], |r| r.get(0)) .unwrap(); assert_eq!(state, 1); } #[test] fn rescanning_updates_rather_than_duplicating() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 1001, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); persist(&catalog, "PhotosRaw", &result).unwrap(); assert_eq!(total_images(&catalog).unwrap(), 1, "no duplicate rows"); let roots: i64 = catalog .connection() .query_row("SELECT count(*) FROM roots", [], |r| r.get(0)) .unwrap(); assert_eq!(roots, 1, "no duplicate roots"); } #[test] fn stable_file_ids_are_recorded() { // FR-NC-5: a server-side move must be a move, not a re-download. let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 4242, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let file_id: i64 = catalog .connection() .query_row("SELECT file_id FROM remote", [], |r| r.get(0)) .unwrap(); assert_eq!(file_id, 4242); } #[test] fn a_thumbnail_job_is_queued_per_image() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![ entry("PhotosRaw/a.CR2", 1, 30_000_000), entry("PhotosRaw/b.CR2", 2, 30_000_000), ], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let jobs: i64 = catalog .connection() .query_row("SELECT count(*) FROM jobs WHERE kind = 2", [], |r| r.get(0)) .unwrap(); assert_eq!(jobs, 2); } #[test] fn a_second_scan_does_not_multiply_jobs() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 1, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); persist(&catalog, "PhotosRaw", &result).unwrap(); let jobs: i64 = catalog .connection() .query_row("SELECT count(*) FROM jobs WHERE kind = 2", [], |r| r.get(0)) .unwrap(); assert_eq!(jobs, 1, "coalesced, not queued twice"); } #[test] fn folder_etags_round_trip_for_the_next_scan() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![], directories: vec![ (RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1")), ( RemotePath::new("PhotosRaw/2026"), dr_sync::Validator::new("e2"), ), ], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let known = load_folder_etags(&catalog, "PhotosRaw"); assert_eq!(known.len(), 2); assert_eq!( known .get(&RemotePath::new("PhotosRaw/2026")) .map(|v| v.as_str()), Some("e2") ); } #[test] fn cells_are_windowed_not_wholesale() { let catalog = Catalog::in_memory().unwrap(); let images: Vec = (0..50) .map(|i| entry(&format!("PhotosRaw/img{i:03}.CR2"), i as u64, 1000)) .collect(); let result = dr_sync::ScanResult { images, directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let page = read_cells(&catalog, 10, 5).unwrap(); assert_eq!(page.len(), 5); assert_eq!(page[0].name, "img010.CR2"); } /// TRACES: NFR-P5 /// The grid's window read must be answered by walking `images_grid_order`, /// never by sorting the library into a temp b-tree. /// /// This asserts on the *query plan* rather than on a duration, because the /// failure has no other symptom: a `GRID_ORDER` edited out of step with the /// index in schema V7, or a column added back into the paging query that /// drags `remote` in again, both still return the right cells. They just /// return them after sorting 24,000 rows, inside the scroll handler — which /// is the jitter this pair was introduced to remove, and it would come back /// silently. #[test] fn the_window_read_walks_the_ordering_index() { let catalog = with_images(20); // Including the burst clause, because the grid includes it: a // predicate that quietly cost the ordering index would put the sort // back and this is the only place that would notice. let folded = uncollapsed("i"); let plan: Vec = catalog .connection() .prepare(&format!( "EXPLAIN QUERY PLAN SELECT {CELL_COLUMNS} FROM images i WHERE {VISIBLE}{folded} {GRID_ORDER} LIMIT 10 OFFSET 5" )) .unwrap() .query_map([], |r| r.get::<_, String>(3)) .unwrap() .flatten() .collect(); let plan = plan.join(" | "); assert!( plan.contains("images_grid_order"), "the window read is not using the ordering index: {plan}" ); assert!( !plan.contains("TEMP B-TREE"), "the window read is still sorting the whole library: {plan}" ); assert!( !plan.to_lowercase().contains("remote"), "the window read is joining `remote` again, which pages the whole \ library before it discards it: {plan}" ); } /// The file ids still arrive, now that they come from a second query. #[test] fn a_window_still_carries_the_server_file_ids() { let catalog = with_images(20); let page = read_cells(&catalog, 5, 4).unwrap(); assert_eq!(page.len(), 4); assert!( page.iter().all(|c| c.file_id.is_some()), "a cell lost its file id when the join was split out" ); } /// A catalog with `n` images, ready to file into collections. fn with_images(n: usize) -> Catalog { let catalog = Catalog::in_memory().unwrap(); let images: Vec = (0..n) .map(|i| entry(&format!("PhotosRaw/img{i:03}.CR2"), i as u64, 1000)) .collect(); let result = dr_sync::ScanResult { images, directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); catalog } /// TRACES: FR-CULL-5 /// A folded burst takes rows out of the cells, the count and the range a /// shift-click resolves — all three, together. /// /// This is the test that would fail if the clause were added to four of /// the five queries that need it. That failure has no other symptom: the /// header claims images the grid will not draw, the scrollbar sizes itself /// for rows that are not there, and neither number looks wrong on its own. #[test] fn folding_a_burst_takes_the_same_rows_out_of_every_answer() { use dr_catalog::bursts::{self, Rules, Signature}; let catalog = with_images(4); // Three of the four are one burst: a second apart, one signature. let ids = image_ids(&catalog); for (n, id) in ids.iter().enumerate() { let hash = if n < 3 { 0xFF00 } else { 0x00FF }; catalog .connection() .execute( "UPDATE images SET captured_at = ?2, camera = 'Canon EOS R5', perceptual_hash = ?3 WHERE id = ?1", rusqlite::params![id.0 as i64, 1_000 + n as i64, Signature(hash).to_stored()], ) .unwrap(); } bursts::regroup(catalog.connection(), Rules::default()).unwrap(); let filter = RatingFilter::default(); // Open, as a new burst is: nothing has been taken away yet. assert_eq!(read_cells(&catalog, 0, 50).unwrap().len(), 4); assert_eq!(total_images_filtered(&catalog, &filter).unwrap(), 4); bursts::set_expanded(catalog.connection(), ids[0], false).unwrap(); let cells = read_cells(&catalog, 0, 50).unwrap(); assert_eq!(cells.len(), 2, "the folded frames are still in the cells"); assert_eq!( total_images_filtered(&catalog, &filter).unwrap(), cells.len(), "the header's count and the cells disagree" ); assert_eq!( read_ids_span(&catalog, None, &filter, false, 0, 49) .unwrap() .len(), cells.len(), "a shift-click over the whole grid would select frames it cannot show" ); } fn image_ids(catalog: &Catalog) -> Vec { let mut stmt = catalog .connection() .prepare("SELECT id FROM images ORDER BY source_ref") .unwrap(); stmt.query_map([], |r| Ok(dr_types::ImageId(r.get::<_, i64>(0)? as u64))) .unwrap() .map(Result::unwrap) .collect() } /// A face with no proxy left draws "no preview" and cannot repair itself: /// the image has its `face_index` row, so it is not outstanding work. The /// sweep has to pick it up by a second route. #[test] fn a_face_whose_proxy_is_gone_is_work_again() { let catalog = with_images(3); let ids = image_ids(&catalog); // An empty store, which is the state the bug lives in: the face is // recorded and there is nothing on disk to cut it out of. let store_dir = std::env::temp_dir().join(format!("dr-face-proxy-test-{}", std::process::id())); let _ = std::fs::remove_dir_all(&store_dir); std::fs::create_dir_all(&store_dir).unwrap(); let store = ThumbStore::open(&store_dir).unwrap(); let face = dr_catalog::faces::DetectedFace { x: 0.1, y: 0.1, w: 0.2, h: 0.2, landmarks: [(0.0, 0.0); 5], confidence: 0.9, embedding: vec![0u8; 1024], crop_px: 120.0, quality: None, crop: Vec::new(), model_id: "w600k_mbf".into(), }; dr_catalog::faces::record_detections( catalog.connection(), ids[0], "w600k_mbf", 1024, std::slice::from_ref(&face), ) .unwrap(); // Indexed, so not outstanding — but with nothing to crop from. assert!(!faces_unindexed(&catalog, "w600k_mbf") .unwrap() .iter() .any(|r| r.image_id == ids[0].0 as i64)); let repair = faces_without_proxy(&catalog, &store, "w600k_mbf").unwrap(); assert_eq!(repair.len(), 1, "the orphaned face was not picked up"); assert_eq!(repair[0].image_id, ids[0].0 as i64); let _ = std::fs::remove_dir_all(&store_dir); } /// Ordering is load-bearing. A repair queued behind every un-indexed image /// in the library is a repair that does not happen inside a session, and /// the screen it was meant to fix stays empty. #[test] fn repairs_are_reached_before_the_rest_of_the_library() { let catalog = with_images(50); let ids = image_ids(&catalog); let store_dir = std::env::temp_dir().join(format!( "dr-face-order-test-{}-{:?}", std::process::id(), std::thread::current().id() )); let _ = std::fs::remove_dir_all(&store_dir); std::fs::create_dir_all(&store_dir).unwrap(); let store = ThumbStore::open(&store_dir).unwrap(); // One image late in the library has a face and no proxy. let face = dr_catalog::faces::DetectedFace { x: 0.1, y: 0.1, w: 0.2, h: 0.2, landmarks: [(0.0, 0.0); 5], confidence: 0.9, embedding: vec![0u8; 1024], crop_px: 120.0, quality: None, crop: Vec::new(), model_id: "w600k_mbf".into(), }; let orphan = ids[40]; dr_catalog::faces::record_detections( catalog.connection(), orphan, "w600k_mbf", 1024, std::slice::from_ref(&face), ) .unwrap(); let mut wanted = faces_without_proxy(&catalog, &store, "w600k_mbf").unwrap(); wanted.extend(faces_unindexed(&catalog, "w600k_mbf").unwrap()); assert_eq!( wanted.first().map(|r| r.image_id), Some(orphan.0 as i64), "the image the screen cannot draw must be fetched first" ); assert_eq!( wanted.len(), 50, "49 un-indexed plus the one being repaired" ); let _ = std::fs::remove_dir_all(&store_dir); } /// TRACES: FR-CULL-8 | NFR-RES-2 /// A panorama over the fetch budget leaves the work list before anything /// is fetched, and does not come back on the next pass. #[test] fn an_original_over_the_size_budget_is_set_aside_once() { let catalog = Catalog::in_memory().unwrap(); let images = vec![ entry("PhotosRaw/IMG_4181.dng", 1, 23_230_634), entry("PhotosRaw/IMG_4181-Pano.dng", 2, 521_218_956), entry("PhotosRaw/IMG_4182.dng", 3, 22_841_218), ]; let result = dr_sync::ScanResult { images, directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let mut wanted: Vec<_> = faces_unindexed(&catalog, "w600k_mbf") .unwrap() .into_iter() .map(|r| (r, SweepWork::Detect)) .collect(); assert_eq!(wanted.len(), 3); let skipped = set_aside_oversized(&catalog, "w600k_mbf", &mut wanted); assert_eq!(skipped, 1); assert_eq!(wanted.len(), 2, "the two camera files are still work"); assert!( wanted.iter().all(|(r, _)| !r.path.contains("Pano")), "the panorama must not be fetched" ); // Marked, so the next pass does not offer it again: the point is to // spend the half gigabyte never, not once per sweep. let again = faces_unindexed(&catalog, "w600k_mbf").unwrap(); assert_eq!(again.len(), 2); assert!(again.iter().all(|r| !r.path.contains("Pano"))); } /// The regression this whole pass exists for. /// /// The previous work list intersected with the thumbnail store, so a /// library nobody had zoomed into produced an empty one and "index the /// whole library" indexed nothing. Nothing here puts a proxy on disk. #[test] fn every_unindexed_image_is_work_even_with_no_proxy_anywhere() { let catalog = with_images(10); let wanted = faces_unindexed(&catalog, "w600k_mbf").unwrap(); assert_eq!( wanted.len(), 10, "a library with no thumbnails is still work" ); assert!( wanted.iter().all(|r| r.full_resolution), "the pass must keep the detail a thumbnail would throw away" ); } /// `face_index` records that detection *ran*, so an image with no face in /// it must not come back on the next pass — otherwise a personal library, /// which is mostly landscapes and documents, never finishes. #[test] fn an_image_already_run_over_is_not_work_again() { let catalog = with_images(3); let ids = image_ids(&catalog); dr_catalog::faces::record_detections(catalog.connection(), ids[0], "w600k_mbf", 1024, &[]) .unwrap(); let wanted = faces_unindexed(&catalog, "w600k_mbf").unwrap(); assert_eq!(wanted.len(), 2); assert!(!wanted.iter().any(|r| r.image_id == ids[0].0 as i64)); // A different model has seen none of them. assert_eq!(faces_unindexed(&catalog, "other").unwrap().len(), 3); } /// A detector change keeps the embedder, so it is not a new library: the /// images the old detector ran over are not "unindexed" for the new one. /// They are an *upgrade*, listed separately and only when the chosen /// detector outranks the one that indexed them — never the other way, /// or a tablet on the fast detector would undo the desktop's thorough /// pass. #[test] fn a_stronger_detector_upgrades_rather_than_re_indexes() { let catalog = with_images(3); let ids = image_ids(&catalog); let conn = catalog.connection(); dr_catalog::faces::record_detections(conn, ids[0], "w600k_mbf", 1024, &[]).unwrap(); dr_catalog::faces::record_detections(conn, ids[1], "scrfd_10g+w600k_mbf", 1024, &[]) .unwrap(); let fresh = faces_unindexed(&catalog, "scrfd_10g+w600k_mbf").unwrap(); assert_eq!( fresh.len(), 1, "the earlier detector's images were re-queued" ); assert_eq!(fresh[0].image_id, ids[2].0 as i64); // Thorough re-runs what Fast did; Fast leaves what Thorough did alone. let up = faces_superseded(&catalog, &["w600k_mbf", "scrfd_2.5g+w600k_mbf"]).unwrap(); assert_eq!(up.len(), 1); assert_eq!(up[0].image_id, ids[0].0 as i64); assert!(faces_superseded(&catalog, &[]).unwrap().is_empty()); } /// The state schema V14 leaves: a face with no quality and an image with /// no marker. It is the measuring pass's work, and *only* that pass's — a /// full re-detection of the same image would throw away every suggestion /// on it for nothing. #[test] fn an_unmeasured_face_is_measured_rather_than_re_detected() { let catalog = with_images(3); let ids = image_ids(&catalog); let stored = |quality: Option| dr_catalog::faces::DetectedFace { x: 0.1, y: 0.1, w: 0.2, h: 0.2, landmarks: [(0.0, 0.0); 5], confidence: 0.9, embedding: vec![0u8; 1024], crop_px: 120.0, quality, crop: Vec::new(), model_id: "w600k_mbf".into(), }; let conn = catalog.connection(); dr_catalog::faces::record_detections(conn, ids[0], "w600k_mbf", 4000, &[stored(None)]) .unwrap(); dr_catalog::faces::record_detections( conn, ids[1], "w600k_mbf", 4000, &[stored(Some(18.0))], ) .unwrap(); // What V14 does to the first: the marker goes, the face stays. dr_catalog::faces::clear_index_marker(conn, ids[0], "w600k_mbf").unwrap(); let measure = faces_unmeasured(&catalog, "w600k_mbf").unwrap(); assert_eq!(measure.len(), 1); assert_eq!(measure[0].image_id, ids[0].0 as i64); assert!(measure[0].full_resolution); // Not re-detected, marker or no marker; the third image, never seen, // still is. let detect = faces_unindexed(&catalog, "w600k_mbf").unwrap(); assert_eq!(detect.len(), 1); assert_eq!(detect[0].image_id, ids[2].0 as i64); } #[test] fn a_scoped_grid_shows_only_that_collections_images() { use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(10); let ids = image_ids(&catalog); let c = coll::create( catalog.connection(), "Selects", None, CollectionKind::Manual, ) .unwrap(); coll::add_images(catalog.connection(), c, &ids[2..5]).unwrap(); let cells = read_cells_scoped(&catalog, Some(c), &RatingFilter::default(), 0, 120).unwrap(); assert_eq!(cells.len(), 3); assert_eq!( total_images_scoped(&catalog, Some(c), &RatingFilter::default()).unwrap(), 3 ); // Unscoped is still the whole library. assert_eq!( total_images_scoped(&catalog, None, &RatingFilter::default()).unwrap(), 10 ); } #[test] fn a_collection_set_shows_its_childrens_images() { // A parent whose children hold everything must not read as empty — // that is what makes nesting look broken. use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(10); let ids = image_ids(&catalog); let trips = coll::create(catalog.connection(), "Trips", None, CollectionKind::Manual).unwrap(); let iceland = coll::create( catalog.connection(), "Iceland", Some(trips), CollectionKind::Manual, ) .unwrap(); coll::add_images(catalog.connection(), iceland, &ids[0..4]).unwrap(); // The parent itself has no direct members at all. let cells = read_cells_scoped(&catalog, Some(trips), &RatingFilter::default(), 0, 120).unwrap(); assert_eq!(cells.len(), 4, "the set shows what its children hold"); assert_eq!( total_images_scoped(&catalog, Some(trips), &RatingFilter::default()).unwrap(), 4 ); } #[test] fn an_image_in_both_a_parent_and_a_child_is_shown_once() { // The count and the number of cells drawn must agree, or neither is // believable. use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(10); let ids = image_ids(&catalog); let trips = coll::create(catalog.connection(), "Trips", None, CollectionKind::Manual).unwrap(); let iceland = coll::create( catalog.connection(), "Iceland", Some(trips), CollectionKind::Manual, ) .unwrap(); coll::add_images(catalog.connection(), trips, &ids[0..2]).unwrap(); coll::add_images(catalog.connection(), iceland, &ids[0..3]).unwrap(); let cells = read_cells_scoped(&catalog, Some(trips), &RatingFilter::default(), 0, 120).unwrap(); assert_eq!(cells.len(), 3, "images 0..3, each once"); assert_eq!( total_images_scoped(&catalog, Some(trips), &RatingFilter::default()).unwrap(), 3 ); } #[test] fn a_scoped_window_still_pages() { // FR-CAT-4 applies inside a collection too: a 5,000-image collection // must not become 5,000 rows. use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(30); let ids = image_ids(&catalog); let c = coll::create(catalog.connection(), "Big", None, CollectionKind::Manual).unwrap(); coll::add_images(catalog.connection(), c, &ids).unwrap(); let page = read_cells_scoped(&catalog, Some(c), &RatingFilter::default(), 10, 5).unwrap(); assert_eq!(page.len(), 5); assert_eq!(page[0].name, "img010.CR2"); } #[test] fn an_empty_collection_reads_as_empty_rather_than_as_the_whole_library() { // The failure that would make scoping useless: an empty IN-list // matching everything. use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(10); let c = coll::create(catalog.connection(), "Empty", None, CollectionKind::Manual).unwrap(); assert!( read_cells_scoped(&catalog, Some(c), &RatingFilter::default(), 0, 120) .unwrap() .is_empty() ); assert_eq!( total_images_scoped(&catalog, Some(c), &RatingFilter::default()).unwrap(), 0 ); } #[test] fn cells_carry_the_file_id_the_thumbnail_store_keys_on() { // Without this the store can never be hit: every launch would refetch // every thumbnail over the network. let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 7777, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e"))], progress: Default::default(), sidecars: Vec::new(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let cells = read_cells(&catalog, 0, 10).unwrap(); assert_eq!(cells[0].file_id, Some(7777)); } #[test] fn a_stored_thumbnail_survives_a_restart() { // The end-to-end property the store exists for: encode, persist, // reopen, decode. A second launch must not re-fetch. let dir = std::env::temp_dir().join(format!("dr-ui-thumbs-{}", std::process::id())); let _ = std::fs::remove_dir_all(&dir); let rgba: Vec = std::iter::repeat_n([90u8, 140, 200, 255], 32 * 32) .flatten() .collect(); { let mut store = ThumbStore::open(&dir).unwrap(); let bytes = dr_thumbs::encode_rgba(32, 32, &rgba).unwrap(); store .put( 4242, dr_thumbs::ThumbSize::Grid, &dr_thumbs::Thumbnail { width: 32, height: 32, bytes, }, ) .unwrap(); } let store = ThumbStore::open(&dir).unwrap(); let stored = store .get(4242, dr_thumbs::ThumbSize::Grid) .unwrap() .expect("persisted"); let (w, h, out) = dr_thumbs::decode_rgba(&stored.bytes).unwrap(); assert_eq!((w, h), (32, 32)); // Lossy, so compare approximately — a blue-ish pixel must stay blue. assert!(out[2] > out[0], "channel order survived the round trip"); } #[test] fn a_store_hit_is_not_evidence_the_server_is_reachable() { // The regression this guards: store hits were delivered as the same // `Ready` the network path sends, and the UI took any `Ready` as proof // of connectivity. A mostly-cached window then declared "back online" // against a server that was down — clearing the banner and kicking off // a sweep that immediately failed, on every scroll. let dir = std::env::temp_dir().join(format!("dr-ui-provenance-{}", std::process::id())); let _ = std::fs::remove_dir_all(&dir); let rgba: Vec = std::iter::repeat_n([10u8, 20, 30, 255], 8 * 8) .flatten() .collect(); let mut store = ThumbStore::open(&dir).unwrap(); let bytes = dr_thumbs::encode_rgba(8, 8, &rgba).unwrap(); store .put( 99, dr_thumbs::ThumbSize::Grid, &dr_thumbs::Thumbnail { width: 8, height: 8, bytes, }, ) .unwrap(); // The split in `spawn_thumbnails`: a hit decodes off local disk and is // reported with `from_cache` set, which is what the reachability gate // keys on. let stored = store .get(99, dr_thumbs::ThumbSize::Grid) .unwrap() .expect("stored"); let (width, height, rgba) = dr_thumbs::decode_rgba(&stored.bytes).unwrap(); let hit = ThumbnailReady { row: 0, width, height, rgba, from_cache: true, }; assert!( hit.from_cache, "a thumbnail read from the store must not be mistaken for a fetch" ); // And the mechanism it feeds: an offline tracker must survive it. let mut reach = dr_sync::Reachability::new(); let now = std::time::Instant::now(); reach.mark_unreachable("network error".into(), now); if !hit.from_cache { reach.mark_reachable(now); } assert!( reach.is_offline(), "replaying cached thumbnails must leave offline mode intact" ); } #[test] fn thumbs_live_beside_the_catalog_not_in_the_cache() { // They sync to the server and are shared with other clients, so a // cache sweep must not discard them. let cat = catalog_path(&account("https://cloud.example", "duncan")); let thumbs = thumbs_dir(&account("https://cloud.example", "duncan")); assert_eq!(thumbs.parent(), cat.parent()); } // --- the trash view (FR-CAT-15) --------------------------------------- /// A catalog with `n` scanned images, none trashed. fn scanned(n: u64) -> Catalog { let catalog = Catalog::in_memory().unwrap(); let images = (1..=n) .map(|i| entry(&format!("PhotosRaw/IMG_{i:04}.CR2"), 1000 + i, 30_000_000)) .collect(); persist( &catalog, "PhotosRaw", &dr_sync::ScanResult { images, directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), sidecars: Vec::new(), }, ) .unwrap(); catalog } /// Mark one image trashed, at a given instant. fn trash_at(catalog: &Catalog, path: &str, when: i64) { let n = catalog .connection() .execute( "UPDATE images SET trashed_at = ?1, trashed_from = source_ref WHERE source_ref = ?2", rusqlite::params![when, path], ) .unwrap(); assert_eq!(n, 1, "fixture should have trashed exactly {path}"); } #[test] fn the_trash_lists_what_the_library_hides() { // The whole point of the view: these rows exist and no other query in // the application will show them. let catalog = scanned(3); trash_at(&catalog, "PhotosRaw/IMG_0002.CR2", 100); let trashed = read_trashed_cells(&catalog, 0, 50).unwrap(); assert_eq!(trashed.len(), 1); assert_eq!(trashed[0].remote_path, "PhotosRaw/IMG_0002.CR2"); // And it has left the library in the same move. let live = read_cells_all(&catalog, &RatingFilter::default(), 0, 50).unwrap(); assert_eq!(live.len(), 2); assert!(!live.iter().any(|c| c.remote_path.contains("IMG_0002"))); } #[test] fn an_untrashed_library_has_an_empty_trash() { let catalog = scanned(3); assert!(read_trashed_cells(&catalog, 0, 50).unwrap().is_empty()); assert_eq!(total_trashed(&catalog).unwrap(), 0); } #[test] fn the_trash_count_agrees_with_the_cells_it_lists() { // The header and the scrollbar are sized from the count while the grid // draws the cells. Two predicates that drift leave the user scrolling // through rows that are not there. let catalog = scanned(5); for (i, when) in [(1, 100), (3, 200), (5, 300)] { trash_at(&catalog, &format!("PhotosRaw/IMG_{i:04}.CR2"), when); } assert_eq!(total_trashed(&catalog).unwrap(), 3); assert_eq!(read_trashed_cells(&catalog, 0, 50).unwrap().len(), 3); } // --- manual order within a collection (FR-CAT-7) ------------------------ fn ids(n: &[u64]) -> Vec { n.iter().copied().map(dr_types::ImageId).collect() } #[test] fn a_run_moved_forward_lands_before_the_photograph_it_was_dropped_on() { let current = ids(&[1, 2, 3, 4, 5]); assert_eq!( reordered(¤t, &ids(&[4]), dr_types::ImageId(2), false), ids(&[1, 4, 2, 3, 5]) ); } #[test] fn a_run_moved_backward_lands_before_it_too() { // The direction of travel must not change what "before this one" means, // or the same drop would land in two different places depending on // where the photograph came from. let current = ids(&[1, 2, 3, 4, 5]); assert_eq!( reordered(¤t, &ids(&[2]), dr_types::ImageId(5), false), ids(&[1, 3, 4, 2, 5]) ); } #[test] fn the_trailing_half_of_the_last_cell_is_how_the_end_is_reached() { // There is no cell beyond the last one to drop in front of, so without // `after` the final position is unreachable — which is exactly the // place a "put this at the end" drag is aiming for. let current = ids(&[1, 2, 3]); assert_eq!( reordered(¤t, &ids(&[1]), dr_types::ImageId(3), true), ids(&[2, 3, 1]) ); } #[test] fn a_moved_run_keeps_the_order_the_grid_shows_it_in() { // Not the order the selection was built in. A user who tapped the last // frame first has said nothing about how the run should be arranged — // only about where it should go. let current = ids(&[1, 2, 3, 4, 5]); assert_eq!( reordered(¤t, &ids(&[5, 1]), dr_types::ImageId(3), false), ids(&[2, 1, 5, 3, 4]) ); } #[test] fn a_run_dropped_on_one_of_its_own_members_stays_together() { // The caller refuses this drop, so it is only reachable if that guard // is ever lost. It must not lose photographs when it is. let current = ids(&[1, 2, 3, 4]); let moved = reordered(¤t, &ids(&[2, 3]), dr_types::ImageId(3), false); assert_eq!(moved.len(), current.len(), "nothing was dropped"); let mut sorted = moved.clone(); sorted.sort(); assert_eq!(sorted, ids(&[1, 2, 3, 4]), "and nothing was invented"); } #[test] fn a_reorder_never_loses_or_duplicates_a_member() { // The property that matters most: this writes the whole membership // back, so a run that dropped one image would delete it from the // collection. let current = ids(&[1, 2, 3, 4, 5, 6]); for target in [1u64, 2, 3, 4, 5, 6] { for after in [false, true] { let moved = reordered(¤t, &ids(&[2, 5]), dr_types::ImageId(target), after); let mut sorted = moved.clone(); sorted.sort(); assert_eq!( sorted, ids(&[1, 2, 3, 4, 5, 6]), "target {target}, after {after}" ); } } } /// The scoped grid and the range a shift-click resolves are two halves of /// one ordering. This is the assertion that keeps them one: an ordinal read /// through a different ORDER BY names a different photograph, and the user /// finds out when the export runs. #[test] fn a_manual_collection_is_read_and_spanned_in_the_order_it_was_given() { let catalog = with_images(5); let all = image_ids(&catalog); let id = dr_catalog::collections::create( catalog.connection(), "Trip", None, dr_catalog::collections::CollectionKind::Manual, ) .unwrap(); dr_catalog::collections::add_images(catalog.connection(), id, &all).unwrap(); // Reversed, so position and capture time disagree about everything. let wanted: Vec<_> = all.iter().rev().copied().collect(); dr_catalog::collections::set_order(catalog.connection(), id, &wanted).unwrap(); let cells = read_cells_scoped(&catalog, Some(id), &RatingFilter::default(), 0, 50).unwrap(); let drawn: Vec<_> = cells .iter() .map(|c| dr_types::ImageId(c.image_id as u64)) .collect(); assert_eq!(drawn, wanted, "the grid draws the order that was written"); let spanned = read_ids_span(&catalog, Some(id), &RatingFilter::default(), false, 0, 4).unwrap(); assert_eq!(spanned, wanted, "and a range resolves through the same one"); assert_eq!( read_member_order(&catalog, id).unwrap(), wanted, "and so does the membership a reorder rewrites" ); } #[test] fn a_collection_with_children_falls_back_to_capture_time() { // A set draws its descendants' images too, and two children's positions // are unrelated integers. Ordering by them interleaves the two // arbitrarily, which is worse than an order that at least means // something. let catalog = with_images(4); let all = image_ids(&catalog); let parent = dr_catalog::collections::create( catalog.connection(), "Iceland", None, dr_catalog::collections::CollectionKind::Manual, ) .unwrap(); dr_catalog::collections::create( catalog.connection(), "Day one", Some(parent), dr_catalog::collections::CollectionKind::Manual, ) .unwrap(); dr_catalog::collections::add_images(catalog.connection(), parent, &all).unwrap(); let reversed: Vec<_> = all.iter().rev().copied().collect(); dr_catalog::collections::set_order(catalog.connection(), parent, &reversed).unwrap(); let cells = read_cells_scoped(&catalog, Some(parent), &RatingFilter::default(), 0, 50).unwrap(); let drawn: Vec<_> = cells .iter() .map(|c| dr_types::ImageId(c.image_id as u64)) .collect(); assert_eq!(drawn, all, "capture time, not the positions that were set"); } #[test] fn a_smart_collection_has_no_manual_order_to_read() { // No member rows at all, so `position` is not a column any of its // images have. Falling back is the only thing there is to do. let catalog = with_images(3); let id = dr_catalog::collections::create( catalog.connection(), "Picks", None, dr_catalog::collections::CollectionKind::Smart, ) .unwrap(); let (order, params) = grid_order_for(&catalog, Some(id)); assert_eq!(order, GRID_ORDER); assert!(params.is_empty()); } #[test] fn the_most_recently_trashed_image_is_listed_first() { // A mistaken delete is corrected within seconds, so the row the user // wants is the one they just made — not the oldest photograph. let catalog = scanned(3); trash_at(&catalog, "PhotosRaw/IMG_0001.CR2", 100); trash_at(&catalog, "PhotosRaw/IMG_0002.CR2", 300); trash_at(&catalog, "PhotosRaw/IMG_0003.CR2", 200); let order: Vec = read_trashed_cells(&catalog, 0, 50) .unwrap() .into_iter() .map(|c| c.remote_path) .collect(); assert_eq!( order, vec![ "PhotosRaw/IMG_0002.CR2".to_string(), "PhotosRaw/IMG_0003.CR2".to_string(), "PhotosRaw/IMG_0001.CR2".to_string(), ] ); } #[test] fn a_shadowed_jpeg_is_not_listed_beside_the_raw_it_belongs_to() { // The inversion applies to `trashed_at` only. Trashing a RAW takes its // sibling JPEG with it, and listing both would offer to restore the // same frame twice. let catalog = scanned(2); let c = catalog.connection(); let raw: i64 = c .query_row( "SELECT id FROM images WHERE source_ref = 'PhotosRaw/IMG_0001.CR2'", [], |r| r.get(0), ) .unwrap(); c.execute( "UPDATE images SET shadowed_by = ?1 WHERE source_ref = 'PhotosRaw/IMG_0002.CR2'", [raw], ) .unwrap(); trash_at(&catalog, "PhotosRaw/IMG_0001.CR2", 100); trash_at(&catalog, "PhotosRaw/IMG_0002.CR2", 100); let trashed = read_trashed_cells(&catalog, 0, 50).unwrap(); assert_eq!(trashed.len(), 1, "one photograph, not two"); assert_eq!(trashed[0].remote_path, "PhotosRaw/IMG_0001.CR2"); assert_eq!(total_trashed(&catalog).unwrap(), 1, "and the count agrees"); } #[test] fn the_trash_window_pages_like_the_grid_does() { // The trash uses the same windowed read as the library, so a large one // must not try to draw itself in a single query. let catalog = scanned(6); for i in 1..=6 { trash_at( &catalog, &format!("PhotosRaw/IMG_{i:04}.CR2"), 100 + i as i64, ); } let first = read_trashed_cells(&catalog, 0, 2).unwrap(); let second = read_trashed_cells(&catalog, 2, 2).unwrap(); assert_eq!(first.len(), 2); assert_eq!(second.len(), 2); assert!( first .iter() .all(|a| !second.iter().any(|b| b.image_id == a.image_id)), "pages must not overlap" ); } #[test] fn a_rating_filter_does_not_hide_anything_in_the_trash() { // The filter narrows a culling pass. A trash that dropped rows because // of a filter set elsewhere would look like it had lost them. let catalog = scanned(3); trash_at(&catalog, "PhotosRaw/IMG_0001.CR2", 100); // Nothing here is rated, so a four-star library filter would empty any // view that honoured it. let strict = RatingFilter { min_rating: 4, ..Default::default() }; assert!(read_cells_all(&catalog, &strict, 0, 50).unwrap().is_empty()); assert_eq!(read_trashed_cells(&catalog, 0, 50).unwrap().len(), 1); } #[test] fn a_date_range_narrows_the_grid_and_the_count_together() { // The whole reason the range lives on `RatingFilter`: every query path // threads that one struct, so the header cannot claim a total the grid // does not draw. let catalog = scanned(3); let conn = catalog.connection(); for (n, at) in [(1, 1_000), (2, 5_000), (3, 9_000)] { conn.execute( "UPDATE images SET captured_at = ?2 WHERE source_ref LIKE ?1", rusqlite::params![format!("%IMG_000{n}%"), at], ) .unwrap(); } let ranged = RatingFilter { captured_from: Some(4_000), captured_to: Some(6_000), ..Default::default() }; assert_eq!(read_cells_all(&catalog, &ranged, 0, 50).unwrap().len(), 1); assert_eq!(total_images_filtered(&catalog, &ranged).unwrap(), 1); } #[test] fn an_undated_image_is_not_shown_inside_a_date_range() { // It cannot be in or out of a span. Drawing it anyway makes a range the // user just chose look as though it had not applied. let catalog = scanned(2); catalog .connection() .execute("UPDATE images SET captured_at = NULL", []) .unwrap(); let ranged = RatingFilter { captured_from: Some(0), captured_to: Some(i64::MAX), ..Default::default() }; assert!(read_cells_all(&catalog, &ranged, 0, 50).unwrap().is_empty()); } #[test] fn a_span_reads_the_whole_run_whether_or_not_it_is_loaded() { // The shift-click this exists for. The grid holds a window of five and // the user names a run of twelve, so seven of them have no cell and no // id anywhere in the UI — but they are still what was asked for, and // the catalog is what knows them. let catalog = scanned(12); let filter = RatingFilter::default(); let loaded = read_cells_all(&catalog, &filter, 0, 5).unwrap(); assert_eq!(loaded.len(), 5, "the window is smaller than the run"); let whole: Vec<_> = read_cells_all(&catalog, &filter, 0, 50) .unwrap() .iter() .map(|c| dr_types::ImageId(c.image_id as u64)) .collect(); let span = read_ids_span(&catalog, None, &filter, false, 0, 11).unwrap(); assert_eq!(span.len(), 12); assert_eq!(span, whole, "the run is the grid's own list, in its order"); } #[test] fn a_span_starts_and_ends_where_it_was_asked_to() { // Ordinals index the grid's list, so a run has to be exactly the slice // of it the two ends name — one off at either end selects a // photograph the user did not point at. let catalog = scanned(12); let filter = RatingFilter::default(); let whole: Vec<_> = read_cells_all(&catalog, &filter, 0, 50) .unwrap() .iter() .map(|c| dr_types::ImageId(c.image_id as u64)) .collect(); let span = read_ids_span(&catalog, None, &filter, false, 4, 6).unwrap(); assert_eq!(span, whole[4..=6], "ordinals 4..=6, inclusive at both ends"); } #[test] fn a_span_is_ordered_by_capture_time_rather_than_by_name() { // A card written by two cameras interleaves names that have nothing to // do with each other. What a photographer means by "everything between // these two" is a stretch of an afternoon, so the run has to be taken // through capture time — the ordering the grid draws them in. let catalog = scanned(3); let conn = catalog.connection(); for (n, at) in [(1, 9_000), (2, 5_000), (3, 1_000)] { conn.execute( "UPDATE images SET captured_at = ?2 WHERE source_ref LIKE ?1", rusqlite::params![format!("%IMG_000{n}%"), at], ) .unwrap(); } let span = read_ids_span(&catalog, None, &RatingFilter::default(), false, 0, 2).unwrap(); let names: Vec = span .iter() .map(|id| { conn.query_row( "SELECT source_ref FROM images WHERE id = ?1", [id.0 as i64], |r| r.get::<_, String>(0), ) .unwrap() }) .collect(); assert!( names[0].ends_with("IMG_0003.CR2") && names[1].ends_with("IMG_0002.CR2") && names[2].ends_with("IMG_0001.CR2"), "earliest first, which here is the reverse of the file names: {names:?}" ); } #[test] fn the_histogram_ignores_the_range_it_is_used_to_choose() { // Drawing the axis through the chosen range would collapse it onto the // selection, leaving nowhere to widen back out from. let catalog = dated_three(); let ranged = RatingFilter { captured_from: Some(4_000), captured_to: Some(6_000), ..Default::default() }; assert_eq!( span_scoped(&catalog, None, &ranged), Some((1_000, 9_000)), "the axis must keep describing the whole extent" ); } /// Three photographs at 1000, 5000 and 9000 seconds. fn dated_three() -> Catalog { let catalog = scanned(3); let conn = catalog.connection(); for (n, at) in [(1, 1_000), (2, 5_000), (3, 9_000)] { conn.execute( "UPDATE images SET captured_at = ?2 WHERE source_ref LIKE ?1", rusqlite::params![format!("%IMG_000{n}%"), at], ) .unwrap(); } catalog } #[test] fn the_histogram_has_the_number_of_bins_it_was_asked_for() { // Fixed, whatever the span holds. The axis draws one bar per bin and // positions it by index, so a query that returned only the occupied // ones would put the bars at the wrong dates. let catalog = dated_three(); let filter = RatingFilter::default(); for bins in [1_u32, 8, 32, 64] { let bars = timeline_uniform(&catalog, None, &filter, 1_000, 9_000, bins).unwrap(); assert_eq!(bars.len() as u32, bins); assert_eq!( bars.iter().map(|b| b.count).sum::(), 3, "every photograph is counted exactly once" ); } } #[test] fn a_bin_starts_where_the_axis_says_it_does() { // The bar's start is its bin's left edge, not the earliest photograph // in it. It is what the position marker and the range band are drawn // against, and an empty bin has no photograph to borrow a date from. let catalog = dated_three(); let bars = timeline_uniform(&catalog, None, &RatingFilter::default(), 0, 8_000, 8).unwrap(); for (i, bar) in bars.iter().enumerate() { assert_eq!(bar.start, i as i64 * 1_000); } // 1000 and 5000 land in their own bins, 9000 is past the end. assert_eq!(bars[1].count, 1); assert_eq!(bars[5].count, 1); assert_eq!(bars.iter().map(|b| b.count).sum::(), 2); } #[test] fn the_last_bin_holds_a_photograph_taken_at_the_very_end() { // The division puts an image captured at exactly `to` one bin past the // axis. Uncapped it would be dropped from the histogram — and it is // precisely the image that defines the extent, so it would go missing // on every unzoomed library. let catalog = dated_three(); let bars = timeline_uniform(&catalog, None, &RatingFilter::default(), 1_000, 9_000, 4).unwrap(); assert_eq!(bars.len(), 4); assert_eq!(bars[3].count, 1, "the image at 9000 is in the last bin"); assert_eq!(bars[0].count, 1); } #[test] fn the_bins_ignore_the_range_they_are_used_to_choose() { // Same rule as the extent: the bars outside the band are what the // range is widened back into, so counting through the range would // leave every one of them empty. let catalog = dated_three(); let ranged = RatingFilter { captured_from: Some(4_000), captured_to: Some(6_000), ..Default::default() }; let bars = timeline_uniform(&catalog, None, &ranged, 1_000, 9_000, 4).unwrap(); assert_eq!( bars.iter().map(|b| b.count).sum::(), 3, "all three, not just the one inside the range" ); } #[test] fn the_bins_still_honour_every_other_filter() { // A histogram of the five-star frames is a fair question, and the bars // have to agree with the grid beneath them. let catalog = dated_three(); let strict = RatingFilter { min_rating: 4, ..Default::default() }; let bars = timeline_uniform(&catalog, None, &strict, 1_000, 9_000, 4).unwrap(); assert_eq!(bars.len(), 4, "the axis keeps its shape"); assert_eq!( bars.iter().map(|b| b.count).sum::(), 0, "nothing here is rated" ); } fn entry(path: &str, file_id: u64, size: u64) -> RemoteEntry { RemoteEntry { id: RemoteId::Stable(file_id), path: RemotePath::new(path), kind: dr_sync::EntryKind::File, validator: dr_sync::Validator::new("v"), size, modified: None, has_preview: false, materialised: true, } } // --- the local-only filter (FR-CAT-9) --------------------------------- /// Record that an image's original is held locally at `tier`. fn cache_at(catalog: &Catalog, id: dr_types::ImageId, tier: dr_types::Tier) { catalog .connection() .execute( "INSERT INTO image_cache (image_id, tier_actual, bytes) VALUES (?1, ?2, 0)", rusqlite::params![id.0 as i64, tier.stored()], ) .unwrap(); } #[test] fn local_only_shows_just_the_images_held_here() { let catalog = with_images(10); let ids = image_ids(&catalog); for id in &ids[0..3] { cache_at(&catalog, *id, dr_types::Tier::Original); } let filter = RatingFilter { local_only: true, ..Default::default() }; let cells = read_cells_all(&catalog, &filter, 0, 120).unwrap(); assert_eq!(cells.len(), 3); // The count the header shows must agree with the cells drawn, which is // the whole reason the predicate lives in SQL rather than in a // post-filter over the rows. assert_eq!(total_images_filtered(&catalog, &filter).unwrap(), 3); assert_eq!(local_original_count(&catalog).unwrap(), 3); } #[test] fn a_cached_preview_is_not_a_local_original() { // The filter answers "can I open this in develop right now", and a // preview cannot. Counting it would put images in the offline set that // fail the moment they are clicked. let catalog = with_images(5); let ids = image_ids(&catalog); cache_at(&catalog, ids[0], dr_types::Tier::Preview); cache_at(&catalog, ids[1], dr_types::Tier::Original); let filter = RatingFilter { local_only: true, ..Default::default() }; assert_eq!(read_cells_all(&catalog, &filter, 0, 120).unwrap().len(), 1); assert_eq!(local_original_count(&catalog).unwrap(), 1); } #[test] fn local_only_composes_with_the_rating_filter() { // "Five-star frames I can actually edit on this train" is one filter, // not a mode that replaces the others. let catalog = with_images(6); let ids = image_ids(&catalog); for id in &ids[0..4] { cache_at(&catalog, *id, dr_types::Tier::Original); } // Rate two of the cached ones, and one that is not cached. for id in [ids[0], ids[1], ids[5]] { dr_catalog::rating::set_rating(catalog.connection(), id, 5).unwrap(); } let filter = RatingFilter { min_rating: 5, local_only: true, ..Default::default() }; let cells = read_cells_all(&catalog, &filter, 0, 120).unwrap(); assert_eq!(cells.len(), 2, "five-starred AND held locally"); assert_eq!(total_images_filtered(&catalog, &filter).unwrap(), 2); } #[test] fn an_empty_cache_is_not_an_empty_library() { // The unfiltered grid must not depend on the cache table having rows — // a library nothing has been downloaded from is still a full library. let catalog = with_images(4); assert_eq!(local_original_count(&catalog).unwrap(), 0); assert_eq!( read_cells_all(&catalog, &RatingFilter::default(), 0, 120) .unwrap() .len(), 4 ); } #[test] fn local_only_counts_as_a_narrowing_filter() { // `is_unfiltered` gates the "filtered" indicator. Reporting this one as // unfiltered would leave a narrowed grid looking like the whole // library, which is the state the indicator exists to prevent. assert!(RatingFilter::default().is_unfiltered()); assert!(!RatingFilter { local_only: true, ..Default::default() } .is_unfiltered()); } // ── narrowing the grid by identity ──────────────────────────────────── /// Put `person` on the given images, as a suggestion. fn assign( catalog: &Catalog, person: dr_catalog::faces::PersonId, images: &[dr_types::ImageId], ) { for img in images { let face = dr_catalog::faces::DetectedFace { x: 0.1, y: 0.1, w: 0.2, h: 0.2, landmarks: [(0.0, 0.0); 5], confidence: 0.9, embedding: vec![0u8; 1024], crop_px: 120.0, quality: None, crop: Vec::new(), model_id: "w600k_mbf".into(), }; // Appends rather than replaces across calls for *different* // people, because `record_detections` clears the image first — // so the second person's face is added by hand. let existing: Vec = { let mut q = catalog .connection() .prepare("SELECT id FROM faces WHERE image_id = ?1") .unwrap(); q.query_map([img.0 as i64], |r| r.get(0)) .unwrap() .map(Result::unwrap) .collect() }; let id = if existing.is_empty() { dr_catalog::faces::record_detections( catalog.connection(), *img, "w600k_mbf", 1024, std::slice::from_ref(&face), ) .unwrap()[0] } else { catalog .connection() .execute( "INSERT INTO faces (image_id, x, y, w, h, landmarks, detector_confidence, embedding, crop_px, model_id, detected_at) VALUES (?1, 0.5, 0.5, 0.2, 0.2, X'00', 0.9, X'00', 120.0, 'w600k_mbf', 0)", [img.0 as i64], ) .unwrap(); dr_catalog::faces::FaceId(catalog.connection().last_insert_rowid() as u64) }; dr_catalog::faces::suggest(catalog.connection(), id, person, 0.9).unwrap(); } } /// The two questions a photographer actually asks, and the reason the /// filter holds a set rather than one id: the intersection is not reachable /// by any sequence of single-person filters. #[test] fn people_narrow_the_grid_as_a_union_or_an_intersection() { let catalog = with_images(4); let ids = image_ids(&catalog); let anna = dr_catalog::faces::create_person(catalog.connection(), "Anna").unwrap(); let bob = dr_catalog::faces::create_person(catalog.connection(), "Bob").unwrap(); // 0: Anna. 1: both. 2: Bob. 3: neither. assign(&catalog, anna, &[ids[0], ids[1]]); assign(&catalog, bob, &[ids[1], ids[2]]); let count = |people: Vec, mode: PeopleMode| { let f = RatingFilter { people, people_mode: mode, ..Default::default() }; total_images_scoped(&catalog, None, &f).unwrap() }; assert_eq!(count(vec![anna.0], PeopleMode::Any), 2, "Anna alone"); assert_eq!(count(vec![bob.0], PeopleMode::Any), 2, "Bob alone"); assert_eq!( count(vec![anna.0, bob.0], PeopleMode::Any), 3, "the union should hold every picture either is in" ); assert_eq!( count(vec![anna.0, bob.0], PeopleMode::All), 1, "the intersection should hold only the picture they share" ); } /// One person is the same filter either way, and the UI leans on that to /// hide the toggle until there are two. #[test] fn one_person_reads_the_same_in_both_modes() { let catalog = with_images(3); let ids = image_ids(&catalog); let anna = dr_catalog::faces::create_person(catalog.connection(), "Anna").unwrap(); assign(&catalog, anna, &[ids[0], ids[1]]); for mode in [PeopleMode::Any, PeopleMode::All] { let f = RatingFilter { people: vec![anna.0], people_mode: mode, ..Default::default() }; assert_eq!(total_images_scoped(&catalog, None, &f).unwrap(), 2); } } /// Three faces of one person in a frame must not satisfy "Anna and Bob". /// This is what `COUNT(DISTINCT ...)` is for, and it is the intersection's /// one real trap. #[test] fn repeated_faces_of_one_person_do_not_satisfy_an_intersection() { let catalog = with_images(2); let ids = image_ids(&catalog); let anna = dr_catalog::faces::create_person(catalog.connection(), "Anna").unwrap(); let bob = dr_catalog::faces::create_person(catalog.connection(), "Bob").unwrap(); // Two separate faces, both Anna, in the same photograph. assign(&catalog, anna, &[ids[0]]); assign(&catalog, anna, &[ids[0]]); let f = RatingFilter { people: vec![anna.0, bob.0], people_mode: PeopleMode::All, ..Default::default() }; assert_eq!(total_images_scoped(&catalog, None, &f).unwrap(), 0); } #[test] fn no_people_narrows_nothing() { let catalog = with_images(3); let f = RatingFilter::default(); assert!(f.is_unfiltered()); assert_eq!(total_images_scoped(&catalog, None, &f).unwrap(), 3); } } /// TRACES: FR-CAT-8 | FR-NC-9 /// Taking another device's judgement out of a sidecar and into the grid. #[cfg(test)] mod reading_judgements_back { use super::*; /// A remote library holding the given files, indexed as a scan leaves it. fn library(files: &[&str]) -> Catalog { let cat = Catalog::in_memory().unwrap(); let c = cat.connection(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); for (i, f) in files.iter().enumerate() { c.execute( "INSERT INTO images(root_id, source_ref, added_at) VALUES (1, ?1, 0)", [f], ) .unwrap(); let image = c.last_insert_rowid(); c.execute( "INSERT INTO remote(image_id, file_id) VALUES (?1, ?2)", rusqlite::params![image, 1000 + i as i64], ) .unwrap(); } dr_catalog::rating::ensure_default_versions(c).unwrap(); cat } fn judgements(cat: &Catalog) -> Vec<(String, i64, i64)> { let c = cat.connection(); let mut stmt = c .prepare( "SELECT i.source_ref, v.rating, v.flag FROM images i JOIN versions v ON v.image_id = i.id AND v.is_default = 1 ORDER BY i.source_ref", ) .unwrap(); let v = stmt .query_map([], |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?))) .unwrap() .map(Result::unwrap) .collect(); v } /// The regression, in one line: a rating in a sidecar reaches the grid. /// Before this there was no path by which it could. #[test] fn a_rating_from_another_device_reaches_the_catalog() { let cat = library(&["2026/a.CR2"]); let n = apply_judgement(cat.connection(), 1, "2026/a.drsc", 4, 1).unwrap(); assert_eq!(n, 1); assert_eq!(judgements(&cat), vec![("2026/a.CR2".to_string(), 4, 1)]); } /// A RAW and the JPEG beside it are one photograph (FR-CAT-11) sharing one /// sidecar, so both rows have to carry the judgement — otherwise the grid /// disagrees with itself depending which of the pair it draws. #[test] fn both_halves_of_a_raw_and_jpeg_pair_are_judged() { let cat = library(&["2026/a.CR2", "2026/a.jpg"]); let n = apply_judgement(cat.connection(), 1, "2026/a.drsc", 3, 0).unwrap(); assert_eq!(n, 2); assert_eq!( judgements(&cat), vec![ ("2026/a.CR2".to_string(), 3, 0), ("2026/a.jpg".to_string(), 3, 0) ] ); } /// A demotion has to travel. Taking the larger of the two would refuse /// every rating the photographer ever lowered — and lowering one is most /// of what a second pass over a shoot does. #[test] fn a_lowered_rating_travels() { let cat = library(&["2026/a.CR2"]); apply_judgement(cat.connection(), 1, "2026/a.drsc", 4, 0).unwrap(); apply_judgement(cat.connection(), 1, "2026/a.drsc", 1, 0).unwrap(); assert_eq!(judgements(&cat)[0].1, 1); } /// But a zero is *unjudged*, not "judged zero". A device that never culled /// the frame must not erase the stars of one that did. #[test] fn an_unjudged_sidecar_does_not_erase_a_local_rating() { let cat = library(&["2026/a.CR2"]); apply_judgement(cat.connection(), 1, "2026/a.drsc", 5, 2).unwrap(); assert_eq!( apply_judgement(cat.connection(), 1, "2026/a.drsc", 0, 0).unwrap(), 0 ); assert_eq!(judgements(&cat), vec![("2026/a.CR2".to_string(), 5, 2)]); } /// Idempotent: a scan runs repeatedly, and a sidecar whose judgement is /// already in the catalog must report no change or the status line claims /// work that did not happen. #[test] fn applying_the_same_judgement_twice_changes_nothing() { let cat = library(&["2026/a.CR2"]); assert_eq!( apply_judgement(cat.connection(), 1, "2026/a.drsc", 4, 1).unwrap(), 1 ); assert_eq!( apply_judgement(cat.connection(), 1, "2026/a.drsc", 4, 1).unwrap(), 0 ); } /// The `LIKE` is a filter and not the decision. A stem holding a wildcard /// would otherwise reach photographs it has nothing to do with, and a /// rating landing on the wrong frame is silent and permanent. #[test] fn a_wildcard_in_a_path_does_not_reach_another_photograph() { // `_` is LIKE's single-character wildcard, so an unescaped `a_b` stem // would also match `axb`. let cat = library(&["2026/a_b.CR2", "2026/axb.CR2"]); apply_judgement(cat.connection(), 1, "2026/a_b.drsc", 5, 0).unwrap(); assert_eq!( judgements(&cat), vec![ ("2026/a_b.CR2".to_string(), 5, 0), ("2026/axb.CR2".to_string(), 0, 0) ] ); } /// A stem that is a prefix of another must not spill onto it: `a.drsc` /// describes `a.CR2`, never `ab.CR2`. #[test] fn a_shared_prefix_is_not_a_shared_sidecar() { let cat = library(&["2026/a.CR2", "2026/ab.CR2"]); apply_judgement(cat.connection(), 1, "2026/a.drsc", 5, 0).unwrap(); assert_eq!( judgements(&cat), vec![ ("2026/a.CR2".to_string(), 5, 0), ("2026/ab.CR2".to_string(), 0, 0) ] ); } /// A sidecar for a photograph this device has not indexed is not an error: /// the scan may have pruned the folder, or the file may be a format this /// device does not accept. #[test] fn a_sidecar_with_no_photograph_here_is_not_a_failure() { let cat = library(&["2026/a.CR2"]); assert_eq!( apply_judgement(cat.connection(), 1, "2026/elsewhere.drsc", 5, 0).unwrap(), 0 ); } } /// TRACES: FR-NC-8 | FR-NC-9 /// What a write does to a sidecar another device has already edited. #[cfg(test)] mod amending_across_devices { use super::*; fn judgement(uuid: &str, rating: u8) -> SidecarWrite { SidecarWrite { image_path: "PhotosRaw/incoming/a.CR2".to_string(), version_uuid: uuid.to_string(), amendment: Amendment::Judgement { rating, flag: 0 }, } } fn version(uuid: &str, revision: u64, modified: i64, rating: u8) -> dr_pipeline::Version { dr_pipeline::sidecar::Version { uuid: uuid.to_string(), name: "Default".to_string(), is_default: true, revision, modified, rating, ..Default::default() } } /// The regression. A sidecar already holding two independently minted /// defaults used to gain a *third* on the next write, because the lookup /// is by uuid and neither of the two was ours. #[test] fn a_write_onto_a_split_sidecar_does_not_add_a_third_version() { let mut base = dr_pipeline::Sidecar::new(); base.put(version("tablet-uuid", 1, 100, 4)); base.put(version("laptop-uuid", 2, 200, 1)); let out = amend(base, &judgement("derived-uuid", 5)); assert_eq!(out.versions.len(), 1, "the split must be closed, not grown"); assert!(out.versions.contains_key("derived-uuid")); assert_eq!(out.versions["derived-uuid"].rating, 5); } /// The other device's work has to survive the fold, or closing the split /// would be the same data loss by a different route. #[test] fn the_other_devices_edit_survives_the_write() { let mut theirs = version("tablet-uuid", 3, 300, 4); theirs .params .insert(("exposure".into(), "exposure".into()), 0.75); let mut base = dr_pipeline::Sidecar::new(); base.put(theirs); // Ours is a rating, which touches no parameter at all. let out = amend(base, &judgement("derived-uuid", 2)); let v = &out.versions["derived-uuid"]; assert_eq!( v.params.get(&("exposure".into(), "exposure".into())), Some(&0.75), "the tablet's exposure was dropped by our rating" ); assert_eq!(v.rating, 2, "and our own judgement did not land"); } /// A sidecar this device has already written must not be disturbed: the /// ordinary case is one default under the right uuid, and fusing it has to /// be a no-op beyond the amendment itself. #[test] fn the_ordinary_write_is_unaffected() { let mut base = dr_pipeline::Sidecar::new(); base.put(version("derived-uuid", 7, 700, 3)); let out = amend(base, &judgement("derived-uuid", 5)); assert_eq!(out.versions.len(), 1); assert_eq!(out.versions["derived-uuid"].rating, 5); assert_eq!( out.versions["derived-uuid"].revision, 8, "one bump for one edit" ); } /// A virtual copy is not a rival default and must be left where it is. #[test] fn a_named_version_is_not_folded_into_the_default() { let mut copy = version("for-print", 4, 400, 5); copy.is_default = false; copy.name = "For print".to_string(); let mut base = dr_pipeline::Sidecar::new(); base.put(version("tablet-uuid", 1, 100, 4)); base.put(copy); let out = amend(base, &judgement("derived-uuid", 2)); assert_eq!(out.versions.len(), 2); assert_eq!(out.versions["for-print"].name, "For print"); assert_eq!(out.versions["for-print"].rating, 5); } } #[cfg(test)] mod fetching_ahead { use super::*; /// The whole reason the registry exists: a click on a photograph that is /// being fetched ahead waits for that transfer rather than starting its /// own, and is told so — `None` — so it looks in the cache again. #[test] fn a_second_claim_waits_for_the_first_to_be_released() { let registry = std::sync::Arc::new(InFlight::default()); let first = registry.claim("shoot/one.CR2"); assert!(first.is_some(), "an unclaimed path is claimed outright"); let (tx, rx) = std::sync::mpsc::channel(); let waiter = { let registry = registry.clone(); std::thread::spawn(move || { tx.send(()).unwrap(); registry.claim("shoot/one.CR2").is_some() }) }; rx.recv().unwrap(); // The waiter is blocked on the first claim. Not provable without a // sleep, but a release that reaches it proves the wait ended there. assert!( !waiter.is_finished(), "the second claim must not return while the first is held" ); drop(first); let claimed = waiter.join().unwrap(); assert!( !claimed, "after waiting, the caller is told to recheck the cache" ); assert!( registry.claim("shoot/one.CR2").is_some(), "and once nobody holds the path it can be claimed again" ); } /// Different photographs never wait on each other. #[test] fn distinct_paths_are_claimed_independently() { let registry = InFlight::default(); let _a = registry.claim("a.CR2"); assert!(registry.claim("b.CR2").is_some()); } /// A newer wish supersedes an older one mid-list; the worker checks this /// between jobs, and it is what keeps a fast walk along the roll from /// queueing every neighbour it passed. #[test] fn a_new_wish_supersedes_the_one_being_served() { let mut wanted = Wanted::default(); let conn = test_connection(); wanted.replace(conn.clone(), Vec::new()); let taken = wanted.generation; assert!(wanted.is_current(taken)); wanted.replace(conn, Vec::new()); assert!( !wanted.is_current(taken), "the list taken before the replacement is stale" ); } fn test_connection() -> Connection { Connection::new( Account::new("nextcloud", "https://cloud.example").with_login("d", "d"), None, ) } }