//! TRACES: FR-CAT-1 | FR-CAT-4 | FR-NC-3 | NFR-P9 //! Opening a remote library: scan → catalog → grid. //! //! This is the wire between three pieces that already worked separately — //! `dr_sync::scan` walks the tree, `dr_catalog` indexes it, and //! `dr_decode::preview` turns bytes into pixels. Until now the "Open library" //! button logged its intent and stopped. //! //! # Threading //! //! Slint's event loop is single-threaded and must never block (NFR-P9), so //! every network and decode operation runs on a worker thread and results //! return through an mpsc channel drained by a Slint timer. That is the same //! shape the login flow uses; it is repeated rather than shared because the //! message types differ and a generic version would obscure both. //! //! # Why thumbnails are fetched, not derived from the scan //! //! A scan yields paths and sizes, nothing visual. Each thumbnail costs its own //! range request, so they are fetched **only for cells the grid actually //! wants** — never for the whole library up front. On the reference library //! that is the difference between a few MB and ~370 GB (ARCH §6.7). use std::path::PathBuf; use std::sync::mpsc::{Receiver, Sender}; use dr_catalog::{Catalog, JobKind, Priority}; use dr_sync::{Account, Connection, RemoteBackend, RemoteError, RemoteId, RemotePath}; use dr_thumbs::ThumbStore; use crate::sidecar_cache::SidecarCache; use dr_types::FormatFilter; /// Largest preview worth fetching whole. /// /// A located preview above this is skipped rather than transferred: past a few /// MB the saving over the full file stops justifying the wait, and a 256px /// thumbnail needs nothing like that much detail. const MAX_PREVIEW_BYTES: u64 = 8 * 1024 * 1024; /// Longest edge face indexing works at. /// /// Not the full preview: `index_proxy` needs packed `f32` RGB, which is 12 /// bytes a pixel, so a 24 MP frame would be ~288 MB and the fetch lanes hold /// one each. 3072 costs ~75 MB at the same moment and still puts a face 2% /// across the frame at ~61 source pixels, against 5 on a grid thumbnail. /// /// Raise it if the embedder is ever given a larger input than 112: it is the /// resolution the *crop* is sampled from, so it bounds face quality directly. const FACE_SOURCE_EDGE: u32 = 3072; /// Progress and results from the scan worker. #[derive(Debug)] pub enum ScanMessage { /// Directories walked so far, and images found. Progress { directories: usize, pruned: usize, images: usize, }, /// The scan finished and the catalog is populated. /// /// `found` counts what this scan *listed*, which on an incremental rescan /// is only what changed — pruned directories contribute nothing. `total` /// is what the catalog actually holds, which is what the grid shows. /// Conflating them made a successful no-op rescan report "0 images" and /// blank the library. Done { found: usize, total: usize, pruned: usize, elapsed_ms: u64, }, /// The scan could not finish. /// /// `offline` distinguishes "the server could not be reached" from "the /// server refused", and it is carried here rather than re-derived because /// the classification is only possible on the worker side: crossing the /// channel flattens a [`dr_sync::RemoteError`] into a message, and no /// amount of string matching on the far side can reliably recover it. /// Without the flag a dead connection and a bad password produce the same /// banner, which sends the user to re-enter a credential that was fine. Failed { message: String, offline: bool }, } /// One decoded thumbnail, ready for the grid. #[derive(Debug)] pub struct ThumbnailReady { /// Index into the grid model this belongs to. pub row: usize, pub width: u32, pub height: u32, pub rgba: Vec, /// Whether these pixels came off local disk rather than the server. /// /// The grid paints both identically, so this exists solely for /// reachability: a store hit is evidence about the *cache*, not the /// network, and treating one as proof of connectivity clears offline mode /// before a single request has been attempted. pub from_cache: bool, } /// Capture metadata read from the same header the thumbnail needed. /// /// Free: the header fetch happens either way, so parsing EXIF out of it costs /// no extra transfer. That is what fills the timeline as the user browses, /// rather than a separate 6 GB sweep over the library. #[derive(Debug, Clone)] pub struct MetadataFound { pub image_id: i64, pub captured_at: Option, pub captured_offset: Option, pub camera: Option, pub lens: Option, pub iso: Option, } /// Messages from the thumbnail worker. #[derive(Debug)] pub enum ThumbnailMessage { Ready(Box), /// No preview could be extracted. The cell stays a placeholder rather than /// silently retrying forever. Unavailable { row: usize, reason: String, }, /// How the batch split between the store and the network. /// /// Sent once, before any fetch. Without it there is no way to tell a /// working cache from a broken one — both fill the grid, one just costs /// nothing. Plan { cached: usize, fetching: usize, dating: usize, }, /// One header-only date read is starting. /// /// Reported separately from thumbnail progress: this work produces no /// visible cell, so without it the window looks idle while it runs. DateProgress, /// Capture dates were written to the catalog. /// /// The timeline is rebuilt on this rather than per image — a histogram /// that redrew 120 times during a batch would flicker for no benefit. DatesRecorded(usize), /// TRACES: FR-CAT-9 /// The server could not be reached while filling this batch. /// /// Distinct from a run of [`Unavailable`](Self::Unavailable): those are /// per-image verdicts ("this file has no extractable preview") and leave /// the rest of the library alone, where this is a statement about the /// connection. Sent at most once per batch, because a dropped connection /// produces one of these per *cell* otherwise and the banner would be /// rewritten sixty times. Offline { reason: String, }, } /// TRACES: FR-CAT-15 | FR-CAT-11 /// What it means for an image to be visible in the library. /// /// Two exclusions, for two different reasons, and both must appear in *every* /// query that counts or lists cells — the grid, the timeline, the metadata /// sweep. A predicate present in four of five places is worse than absent: the /// counts disagree with the cells and neither looks wrong on its own. /// /// - `shadowed_by IS NULL` — a JPEG the camera wrote beside its RAW is that /// same frame, not a second photograph. /// - `trashed_at IS NULL` — a soft-deleted image has been moved to the trash /// folder and is listed only by the trash view. const VISIBLE: &str = "i.shadowed_by IS NULL AND i.trashed_at IS NULL"; /// [`VISIBLE`] for queries that do not alias `images`. const VISIBLE_UNALIASED: &str = "shadowed_by IS NULL AND trashed_at IS NULL"; /// TRACES: FR-CAT-15 /// What the *trash view* lists: exactly what [`VISIBLE`] excludes on the second /// clause, and still excludes on the first. /// /// The inversion is deliberate and only correct on `trashed_at`. A shadowed JPEG /// is not a separate photograph in the trash any more than it is in the library /// — trashing a RAW takes its sibling with it, and listing both would offer to /// restore the same frame twice. const TRASHED: &str = "i.shadowed_by IS NULL AND i.trashed_at IS NOT NULL"; /// TRACES: FR-CAT-4 /// The order the grid lists photographs in: when they were taken. /// /// The file name breaks ties and nothing more — two frames of one burst, or a /// RAW beside the JPEG the camera wrote with it. What a photographer looks for /// is the afternoon, not what the camera called the file, and a grid ordered by /// name interleaves every camera and every card that ever wrote into the same /// folder. /// /// Shared rather than spelled out per query, for the same reason [`VISIBLE`] is: /// the window, the count and the run a shift-click resolves are three answers /// about one list, and an ordering that drifted between them would select /// photographs the user never saw without any of it looking wrong. /// /// Undated images sort last in either direction. EXIF is read as thumbnails /// load, so a freshly scanned library would otherwise open on the images it /// knows least about. const GRID_ORDER: &str = "ORDER BY i.captured_at IS NULL, i.captured_at ASC, i.source_ref ASC"; /// TRACES: FR-CAT-15 /// [`GRID_ORDER`] for the trash, which is ordered by when a thing was deleted — /// see [`read_trashed_cells`] for why that view answers a different question. const TRASH_ORDER: &str = "ORDER BY i.trashed_at DESC, i.source_ref ASC"; /// TRACES: FR-CAT-6 | FR-CULL-4 /// What the grid is narrowed to by the rating filter bar. /// /// Applied in **SQL**, not by filtering the rows after reading them. On a /// remote library a drawn-then-hidden cell has already cost a thumbnail /// fetch, which is the transfer FR-NC-3 exists to avoid — and the count in the /// header has to agree with the cells, which it cannot if the two are computed /// at different stages. /// Not `Copy`: [`RatingFilter::people`] is a `Vec`. Every query path already /// takes this by reference, so the only casualties were two `..*self` struct /// updates, which clone instead. #[derive(Debug, Clone, PartialEq, Eq, Default)] pub struct RatingFilter { /// Minimum stars. 0 means no star constraint. pub min_rating: u8, /// Only images nothing has judged yet — neither starred nor flagged. /// This is what lets a culling session resume where it stopped. pub unjudged: bool, /// `None` for no flag constraint, otherwise exactly that flag. pub flag: Option, /// TRACES: FR-CAT-9 /// Only images whose original is stored on this device. /// /// Carried here, beside the rating terms, because every query path already /// threads this one struct: adding a parallel parameter to /// `read_cells_scoped`, `read_cells_all` and both counts would give four /// call sites the chance to disagree about what the grid is showing, and /// the count disagreeing with the cells is the specific bug this type's /// "filter in SQL" rule exists to prevent. pub local_only: bool, /// Show only photographs captured within this range, as UTC seconds. /// /// Half-open ends are meaningful: a `from` with no `to` reads as /// "everything since". Undated images are excluded whenever either end is /// set — they cannot be placed on the axis the user is narrowing, and /// showing them anyway makes the range look broken. /// /// Here rather than a parallel parameter for the reason `local_only` gives /// above: the count and the cells must be narrowed by the same thing. pub captured_from: Option, pub captured_to: Option, /// TRACES: FR-CULL-11 /// Only photographs these people appear in. /// /// The way back from a face to the pictures it came from, which is the /// question the People screen leaves the user holding: they have just /// identified someone, and what they want next is *everything with them in /// it*. Without this the identification is a dead end. /// /// Here rather than a grid scope of its own, for the reason `local_only` /// gives above — the count and the cells must be narrowed by the same /// thing, and this struct is the one narrowing every query path already /// threads. It composes with the rest for free: three-star photographs of /// Anna from last summer is this term ANDed with two others. /// /// Suggested faces count, not only confirmed ones. A user who has just /// grouped someone and not yet confirmed a single face would otherwise get /// an empty grid, which reads as "no photographs of this person" rather /// than "you have not ticked anything yet". /// /// A set rather than one id, because the two questions a photographer /// actually asks are "every picture of Anna *or* Bob" and "the pictures /// they are *both* in", and the second is not reachable by any sequence of /// single-person filters. [`RatingFilter::people_mode`] picks between them. pub people: Vec, /// Whether [`RatingFilter::people`] is a union or an intersection. pub people_mode: PeopleMode, } /// How several people combine when the grid is narrowed by identity. /// /// TRACES: FR-CULL-11 #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub enum PeopleMode { /// Photographs holding **any** of them — the union. /// /// The default, and the right one for one person, where the two modes are /// identical. It is also the forgiving direction: adding a second person to /// a union can only ever show more, so a user who has not noticed the /// toggle never ends up staring at an empty grid wondering what they broke. #[default] Any, /// Photographs holding **all** of them — the intersection. /// /// "Pictures of the two of them together", which is the one worth having a /// mode for: it is how you find the photograph you remember rather than /// scrolling everything either of them appears in. All, } impl RatingFilter { /// Whether this narrows anything, so the caller can skip the join. pub fn is_unfiltered(&self) -> bool { self.min_rating == 0 && !self.unjudged && self.flag.is_none() && !self.local_only && self.captured_from.is_none() && self.captured_to.is_none() && self.people.is_empty() } /// Whether a date range is narrowing the grid. pub fn has_date_range(&self) -> bool { self.captured_from.is_some() || self.captured_to.is_some() } /// The same filter with the date range lifted. /// /// The timeline uses this: the histogram is how the range is *chosen*, so /// drawing it through the range would collapse the axis onto the current /// selection and leave nowhere to widen it back out from. pub fn without_date_range(&self) -> Self { Self { captured_from: None, captured_to: None, ..self.clone() } } /// The SQL predicate, against an `images` aliased as `i`. /// /// Returns a `String` of conditions ANDed together, or an empty string /// where nothing is constrained. Every branch is built from integers this /// code owns — no caller text reaches the SQL, so there is nothing to /// escape. /// /// A correlated subquery per term rather than a join to `versions`: an /// image with no version row must still be *findable* as unrated, and an /// inner join would silently drop exactly those images — the ones a /// library scanned before ratings existed consists entirely of. fn sql(&self) -> String { let mut terms = Vec::new(); if self.min_rating > 0 { terms.push(format!( "coalesce((SELECT dv.rating FROM versions dv WHERE dv.image_id = i.id AND dv.is_default = 1 LIMIT 1), 0) >= {}", self.min_rating )); } // Integers this code owns, formatted straight in like the rating terms // above — no caller text reaches the SQL. if let Some(from) = self.captured_from { terms.push(format!("i.captured_at >= {from}")); } if let Some(to) = self.captured_to { terms.push(format!("i.captured_at <= {to}")); } if self.has_date_range() { // An undated image cannot be inside or outside a range. Excluding // it is the honest answer; the comparisons above would drop it // anyway, and saying so keeps that from looking accidental. terms.push("i.captured_at IS NOT NULL".to_string()); } if self.unjudged { // Both axes: a frame that was picked but never starred has been // judged, and re-presenting it would undo the user's decision to // move past it. terms.push( "coalesce((SELECT dv.rating FROM versions dv WHERE dv.image_id = i.id AND dv.is_default = 1 LIMIT 1), 0) = 0 AND coalesce((SELECT dv.flag FROM versions dv WHERE dv.image_id = i.id AND dv.is_default = 1 LIMIT 1), 0) = 0" .to_string(), ); } if !self.people.is_empty() { // Integers this code owns, like every other term here — the ids // come from the catalog, never from typed text, so there is // nothing to escape. let ids = self .people .iter() .map(|p| p.to_string()) .collect::>() .join(","); terms.push(match self.people_mode { // `EXISTS` rather than a join, so a photograph holding three // faces of the same person appears once — the grid shows // pictures, not faces. PeopleMode::Any => format!( "EXISTS (SELECT 1 FROM faces f JOIN face_person fp ON fp.face_id = f.id WHERE f.image_id = i.id AND fp.person_id IN ({ids}))" ), // Counting *distinct* people rather than ANDing one EXISTS per // person: same result, one subquery instead of n, and it does // not grow the statement with the selection. `DISTINCT` is // what makes it correct — three faces of Anna in one frame // must not satisfy a filter asking for Anna and Bob. PeopleMode::All => format!( "(SELECT COUNT(DISTINCT fp.person_id) FROM faces f JOIN face_person fp ON fp.face_id = f.id WHERE f.image_id = i.id AND fp.person_id IN ({ids})) = {}", self.people.len() ), }); } if let Some(flag) = self.flag { terms.push(format!( "coalesce((SELECT dv.flag FROM versions dv WHERE dv.image_id = i.id AND dv.is_default = 1 LIMIT 1), 0) = {}", flag_code(flag) )); } if self.local_only { // `tier_actual`, not `tier_desired`: the question is what is // *here*, not what a pin has promised will be. An image queued for // download is exactly the one that cannot be opened yet, so // showing it under "on this device" would be the wrong answer to // the only question this filter is asked. terms.push(format!( "EXISTS (SELECT 1 FROM image_cache ic WHERE ic.image_id = i.id AND ic.tier_actual >= {})", dr_types::Tier::Original.stored() )); } if terms.is_empty() { String::new() } else { format!(" AND ({})", terms.join(") AND (")) } } } /// The stored integer for a flag, matching `dr_catalog::rating`'s encoding. fn flag_code(f: dr_types::FlagState) -> i64 { match f { dr_types::FlagState::Unflagged => 0, dr_types::FlagState::Pick => 1, dr_types::FlagState::Reject => 2, } } /// TRACES: FR-CAT-8 | FR-NC-8 | FR-CULL-4 | FR-DEV-6 /// One amendment to one image's sidecar, on its way to the server. #[derive(Debug, Clone)] pub struct SidecarWrite { /// Remote path of the *image*. The sidecar sits beside it, with the /// extension replaced — that adjacency is what makes a sidecar findable /// without an index (ARCH §6.12). pub image_path: String, pub version_uuid: String, pub amendment: Amendment, } /// What a write changes about the version it names. /// /// An enum rather than a struct of optional fields because the two are written /// by different actions with different failure costs, and because a write must /// never carry a *stale* copy of what it is not changing. A settings write that /// also carried a rating would have to have read one from somewhere, and the /// obvious somewhere — the catalog, moments earlier — is exactly how a cull /// made between the read and the write gets silently reverted. /// /// Everything not named by the variant is left as the file had it, which is /// what makes the read-modify-write in [`write_one_sidecar`] a genuine /// amendment rather than a replacement. #[derive(Debug, Clone)] pub enum Amendment { /// A star rating and a pick/reject flag — the cull. Judgement { rating: u8, flag: u8 }, /// TRACES: FR-DEV-6 /// Copied develop settings, applied within `scope`. /// /// Carries the [`Scope`] rather than a pre-filtered preset so the target's /// own framing can be spared *at the file*: excluding framing means /// leaving the keys already in the sidecar untouched, which cannot be /// expressed by the parameter list alone. Settings { preset: dr_pipeline::Preset, scope: dr_pipeline::Scope, /// TRACES: FR-DEV-3f /// The film stock, when this is an image's own edit being written back. /// /// Two levels of `Option`, and both are load-bearing. The outer says /// whether this write concerns the film at all — a paste does not, /// exactly as it carries no masks. The inner is the choice itself, and /// `Some(None)` is a real edit: "develop this normally again". Without /// the distinction, clearing a film could never be saved. film: Option>, /// TRACES: FR-DEV-3 | FR-CAT-8 /// The local adjustments, when this is an image's own edit being /// written back rather than a paste onto someone else's. /// /// A `Preset` is a parameter map, and a mask is not a parameter — it /// is a rule about *where*, with a chain of its own. So a save that /// carried only the preset wrote the sliders and silently dropped /// every local adjustment: the sidecar format has stored masks since /// they were added and `Version::apply` restores them, but nothing /// ever put any there. The mask survived until the session ended and /// then did not exist. /// /// `None` for a paste, which must not carry the source image's masks /// onto the target: a mask is drawn against one photograph and means /// nothing on another, and `Scope` cannot express that because it /// filters parameters. masks: Option, }, } /// Where an image's sidecar lives. /// /// The image's own path with the extension replaced, not appended: `a.CR2` /// becomes `a.drsc`, so a RAW and the JPEG beside it share one sidecar and /// therefore one judgement. That is the intended behaviour — they are the same /// photograph (FR-CAT-11), and the pairing logic in `dr_catalog::schema` /// already treats them so. pub fn sidecar_path(image_path: &str) -> String { let stem = match image_path.rsplit_once('.') { // Only an extension in the final segment counts; a dot in a directory // name must not truncate the path. Some((stem, ext)) if !ext.contains('/') => stem, _ => image_path, }; format!("{stem}.{}", dr_pipeline::sidecar::EXTENSION) } /// TRACES: FR-CAT-8 | FR-CAT-9 | FR-NC-10 /// Persist amendments to sidecars beside their images. /// /// # Why this reads before it writes /// /// A sidecar is the authoritative store and may already hold an edit made on /// this or another device. Writing a fresh document containing only a rating /// would delete that edit — the exact silent data loss the format's /// unknown-key preservation exists to prevent. So each file is fetched, /// parsed, amended, and written back; a fetch that 404s simply means there is /// no sidecar yet and a new one is created. /// /// # Why the local write is the commit point /// /// FR-CAT-9 requires that edits made offline *queue and apply when the source /// returns*. So every amendment is written to the local cache first and the /// upload is best-effort: an entry stays marked pending until the server has /// actually taken it, and [`spawn_outbox_drain`] retries the marked ones later. /// /// This is what makes `offline` a parameter rather than a reason to skip. It /// was one: a cull or a paste made with no connection used to be dropped /// entirely, which for a pasted edit meant it survived nowhere at all — the /// catalog holds no parameters. Now the two cases differ only in whether the /// upload is attempted. /// /// # Why failure here is logged rather than surfaced /// /// The write has already succeeded locally by the time the network is touched, /// so nothing is lost by a failure and there is nothing for the user to do /// about it. Interrupting a cull with an error dialog per frame would be far /// worse than the risk. The counts are reported once, at the end. pub fn spawn_sidecar_writes( conn: Connection, writes: Vec, cache_dir: PathBuf, offline: bool, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let cache = SidecarCache::open(cache_dir); // The runtime and the backend are only needed to *upload*. Offline, // neither is built — and a failure to build either is not a failure to // record the edit, it just means every write is queued instead. let rt = if offline { None } else { match crate::net_runtime::build() { Ok(rt) => Some(rt), Err(e) => { log::debug!("no runtime for sidecar upload ({e}); queueing"); None } } }; // Every write, recorded locally and queued. The offline path, and the // fallback whenever a backend could not be built. let queue_all = || { let mut report = SidecarReport::default(); for w in &writes { match write_one_sidecar(&cache, w) { Ok(Outcome::Uploaded) => report.written += 1, Ok(Outcome::Queued) => report.queued += 1, Err(e) => { // Warn, not debug. This is unsynced user work — a rating or an // edit that exists only on this device — and the path is // the only thing that says *which* photograph and *where* // the server refused it. Filtered out at the default // level, a 403 on one file is indistinguishable from a // whole library failing. log::warn!("sidecar for {}: {e}", w.image_path); report.last_error = Some(e); report.failed += 1; } } } report }; let report = match rt { None => queue_all(), Some(rt) => rt.block_on(async { match crate::remote::connect(&conn) { Ok(b) => { let mut report = SidecarReport::default(); for w in &writes { match write_one_sidecar_online(&*b, &cache, w).await { Ok(Outcome::Uploaded) => report.written += 1, Ok(Outcome::Queued) => report.queued += 1, Err(e) => { // Warn, not debug. This is unsynced user work — a rating or an // edit that exists only on this device — and the path is // the only thing that says *which* photograph and *where* // the server refused it. Filtered out at the default // level, a 403 on one file is indistinguishable from a // whole library failing. log::warn!("sidecar for {}: {e}", w.image_path); report.last_error = Some(e); report.failed += 1; } } } report } // No backend: the edits are still recorded locally and // will go up with the next drain. Err(e) => { log::debug!("no backend for sidecar upload ({e}); queueing"); queue_all() } } }), }; let _ = tx.send(SidecarMessage::Finished { written: report.written, queued: report.queued, failed: report.failed, last_error: report.last_error, }); }); rx } /// What one write ended up doing. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum Outcome { /// Recorded locally and accepted by the server. Uploaded, /// Recorded locally, still in the outbox. Queued, } /// Running totals for a batch, so the loop bodies stay readable. #[derive(Debug, Default)] struct SidecarReport { written: usize, queued: usize, failed: usize, last_error: Option, } /// The outcome of a batch of sidecar writes. #[derive(Debug)] pub enum SidecarMessage { Finished { written: usize, /// Recorded locally but not yet on the server — offline, or an upload /// that failed. These are retried by [`spawn_outbox_drain`], so this /// is a count of *deferred* work rather than of losses. queued: usize, failed: usize, /// Reported once rather than per file: a network that is down fails /// every write with the same message, and forty identical lines in the /// status bar say nothing forty times. last_error: Option, }, } /// Apply an amendment to a document, returning the new one. /// /// Split out from both write paths so that online and offline produce /// *identical* documents: the only thing that differs between them is which /// base was read and whether an upload follows. A second copy of this for the /// offline case is how the two would come to disagree about what a paste means. fn amend(base: dr_pipeline::Sidecar, w: &SidecarWrite) -> dr_pipeline::Sidecar { let mut sidecar = base; // Amend the version this write belongs to, creating it if the file did // not have one. The uuid comes from the catalog, so the same photograph // keeps one identity across devices (FR-NC-8). let mut version = sidecar .versions .get(&w.version_uuid) .cloned() .unwrap_or_else(|| dr_pipeline::sidecar::Version { uuid: w.version_uuid.clone(), name: "Default".to_string(), is_default: true, revision: 0, ..Default::default() }); // Only what the amendment names. Everything else in the version — the // rating a settings write must not touch, the crop an adjustments-only // paste must spare, the unknown keys of an operation this build lacks — // survives because it was read from the file and is written back. match &w.amendment { Amendment::Judgement { rating, flag } => { version.rating = *rating; version.flag = *flag; } Amendment::Settings { preset, scope, masks, film, } => { preset.amend(&mut version.params, *scope); // TRACES: FR-DEV-3f // Wholesale, like the masks below and for the same reason: this is // the whole of the image's own choice as it stands, so clearing a // film has to leave the sidecar too. if let Some(film) = film { version.film = film.clone(); } // Replaced wholesale rather than merged: this is the whole of the // image's local adjustment stack as it stands, so a layer the user // deleted has to leave the sidecar too. Cross-device merging of // two stacks is `Sidecar::merge`'s job and happens on sync, not // here (FR-NC-9). if let Some(masks) = masks { version.masks = masks.clone(); } } } // A judgement is an edit as far as the merge is concerned, and so is a // paste: without the bump, a device that touched the same frame earlier // would win on revision and this write would be discarded at the next sync // (FR-NC-9). version.revision = version.revision.saturating_add(1); version.modified = now_secs(); sidecar.put(version); sidecar } /// TRACES: FR-CAT-9 /// Record an amendment with no server to send it to. /// /// The base is whatever the cache holds, which is either what the server last /// had or what earlier offline writes have already built on top of it. Either /// way the result is queued, and the drain reconciles it with the server's own /// copy when the connection returns — that reconciliation is a *merge* /// (FR-NC-9), not an overwrite, so building on a possibly-stale base here does /// not cost another device's work. fn write_one_sidecar(cache: &SidecarCache, w: &SidecarWrite) -> Result { let path = sidecar_path(&w.image_path); let base = cache.load(&path).unwrap_or_default(); cache.store(&path, &amend(base, w), true)?; Ok(Outcome::Queued) } /// TRACES: FR-CAT-8 | FR-CAT-9 /// Read-modify-write one sidecar, with a server to read from and send to. async fn write_one_sidecar_online( backend: &dyn RemoteBackend, cache: &SidecarCache, w: &SidecarWrite, ) -> Result { let path_str = sidecar_path(&w.image_path); let path = RemotePath::new(path_str.clone()); let id = RemoteId::Path(path.clone()); // An existing sidecar may hold an edit. Absent is the normal case on a // library that has never been edited, and is not an error. let existing = backend.get(&id, None).await.ok(); // A corrupt sidecar is *not* overwritten: that would destroy an edit this // build merely failed to understand. Refused before anything is written, // locally or remotely, so the cache cannot end up holding a document that // silently discarded the file's real contents. if let Some(bytes) = existing.as_deref() { if !bytes.is_empty() { let text = String::from_utf8_lossy(bytes); if dr_pipeline::Sidecar::parse(&text).is_err() { return Err(format!("sidecar at {path_str} is unreadable")); } } } let base = existing .as_deref() .map(|bytes| String::from_utf8_lossy(bytes).into_owned()) .and_then(|text| dr_pipeline::Sidecar::parse(&text).ok()) // No sidecar on the server. The cache may still hold queued offline // work for this image, and taking `default()` here would drop it. .or_else(|| cache.load(&path_str)) .unwrap_or_default(); let sidecar = amend(base, w); // Locally first: this is the commit point, and an upload that fails after // it leaves the edit queued rather than lost. cache.store(&path_str, &sidecar, true)?; backend .put(&path, sidecar.to_text().into_bytes(), None) .await .map_err(|e| e.to_string())?; // Accepted by the server, so it leaves the outbox. The document stays // cached, which is what lets the next offline open still show the edit. cache.store(&path_str, &sidecar, false)?; Ok(Outcome::Uploaded) } /// TRACES: FR-CAT-9 | FR-NC-9 | FR-NC-10 /// Upload everything the outbox is still holding. /// /// # Why this merges rather than uploads /// /// A queued edit was built on whatever this device last saw. While it sat in /// the outbox another device may have edited the same photograph, and simply /// PUTting the local document would discard that work — the precise failure /// FR-NC-9's node-level merge exists to prevent. So each entry is reconciled /// against the server's current copy before it goes up, and disjoint edits /// (a crop made here, an exposure change made there) both survive. /// /// # Why an entry stays queued on failure /// /// The marker is cleared only after the server has taken the bytes. A drain /// interrupted halfway leaves the rest of the outbox exactly as it was, so /// nothing depends on this running to completion. pub fn spawn_outbox_drain(conn: Connection, cache_dir: PathBuf) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let cache = SidecarCache::open(cache_dir); let queued = cache.pending(); if queued.is_empty() { let _ = tx.send(SidecarMessage::Finished { written: 0, queued: 0, failed: 0, last_error: None, }); return; } log::info!("draining {} queued sidecar(s)", queued.len()); let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { let _ = tx.send(SidecarMessage::Finished { written: 0, queued: queued.len(), failed: 0, last_error: Some(e.to_string()), }); return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { let _ = tx.send(SidecarMessage::Finished { written: 0, queued: queued.len(), failed: 0, last_error: Some(e.to_string()), }); return; } }; let mut report = SidecarReport::default(); for path_str in &queued { match drain_one(&*backend, &cache, path_str).await { Ok(()) => report.written += 1, Err(e) => { // Warn, for the reason the write path does: this is // unsynced work and the path is what makes the failure // actionable. log::warn!("draining {path_str}: {e}"); report.last_error = Some(e); report.failed += 1; // Still queued — the marker was never cleared. report.queued += 1; } } } let _ = tx.send(SidecarMessage::Finished { written: report.written, queued: report.queued, failed: report.failed, last_error: report.last_error, }); }); }); rx } /// Reconcile one queued sidecar with the server and upload it. async fn drain_one( backend: &dyn RemoteBackend, cache: &SidecarCache, path_str: &str, ) -> Result<(), String> { let Some(mut local) = cache.load(path_str) else { // The document went while the drain was running. Nothing to send. return Ok(()); }; let path = RemotePath::new(path_str.to_string()); let id = RemoteId::Path(path.clone()); // TRACES: FR-NC-6c // A miss and a placeholder are not the same answer, and conflating them // destroys work. This read decides whether the sidecar already on the // remote is merged in; treating "the content is not on this device" as // "there is no sidecar" writes a fresh document over an existing one and // discards every edit another device put there — the exact loss the // format's unknown-key preservation exists to prevent. // // A sidecar is a few kilobytes, so the right response to a placeholder is // to fetch it, not to give up. Where that is impossible — no client // running — the entry stays queued, which is what the outbox is for. let remote = match backend.get(&id, None).await { Ok(bytes) => Some(bytes), Err(RemoteError::NotFound(_)) => None, Err(RemoteError::NotMaterialised(_)) => { backend .materialise(&id) .await .map_err(|e| format!("sidecar is not on this device ({e})"))?; match backend.get(&id, None).await { Ok(bytes) => Some(bytes), Err(e) => return Err(format!("sidecar could not be read ({e})")), } } // Anything else — a refused read, a dead connection — leaves the entry // queued rather than resolved by overwriting. Err(e) => return Err(format!("sidecar could not be read ({e})")), }; if let Some(bytes) = remote.as_deref() { if !bytes.is_empty() { let text = String::from_utf8_lossy(bytes); match dr_pipeline::Sidecar::parse(&text) { Ok(remote) => merge_into(&mut local, &remote), // Unreadable on the server. Uploading over it would destroy an // edit this build failed to understand, so the entry stays // queued rather than being resolved destructively. Err(e) => return Err(format!("remote sidecar is unreadable ({e})")), } } } backend .put(&path, local.to_text().into_bytes(), None) .await .map_err(|e| e.to_string())?; cache.store(path_str, &local, false) } /// TRACES: FR-NC-9 /// Merge the server's copy into ours, version by version. /// /// No common ancestor is available — the outbox stores the result, not the /// base it was built from — so the merge runs with `None`, which treats every /// key either side holds as changed. Disjoint keys therefore still both /// survive, and a key both sides set resolves by revision exactly as it would /// with a base. What is lost without one is the ability to see a *deletion*: /// a parameter reset to default on the other device reads as absent rather /// than as removed, so our value stands. That is the same direction of caution /// the judgement merge takes — an edit is preserved rather than erased. fn merge_into(local: &mut dr_pipeline::Sidecar, remote: &dr_pipeline::Sidecar) { for (uuid, their_version) in &remote.versions { match local.versions.get(uuid).cloned() { Some(mut ours) => { ours.merge(their_version, None); local.put(ours); } // A version only the server has — another device's virtual copy // (FR-CAT-12). Keeping it is what stops one device's upload from // deleting another's work. None => local.put(their_version.clone()), } } } /// Where the catalog for an account lives. /// /// Keyed by [`Account::namespace`] so two accounts do not share an index — /// two servers, two logins on one server, or two folders on one disk. Under /// the XDG data directory, not cache: the catalog is rebuildable but /// rebuilding it costs a full rescan, so it is not something to discard on a /// cache sweep. /// /// The namespace is the account's to compute, not this function's, because it /// is also frozen: it names the directory an existing install's catalog, /// thumbnail shards and un-uploaded sidecars are already in. pub fn catalog_path(account: &Account) -> PathBuf { data_root().join(account.namespace()).join("catalog.sqlite") } /// The directory every account's data hangs off. /// /// **Not the cache directory, and on Android that distinction is the whole /// point.** Neither `XDG_DATA_HOME` nor `HOME` is set there, so this used /// to fall through to `temp_dir()` — which Android resolves to the app's /// *cache*, a directory the system deletes without asking under storage /// pressure. /// /// What sits beside a catalog is not disposable. `sidecars/` is the /// commit point for every rating and edit made offline (see /// `sidecar_cache`), and `outbox/` holds exports the user has been told /// succeeded. A day of culling on a train, evicted by the OS before it ever /// reached the server, is the worst failure this application can have, and /// it would be silent. /// /// `AccountStore::data_dir()` is the persistent per-app directory the /// Android entry point establishes before anything opens a store. On a /// desktop it is the XDG config directory, and the two lines below keep the /// established XDG *data* location there rather than moving anyone's /// catalog. fn data_root() -> PathBuf { let base = std::env::var_os("XDG_DATA_HOME") .map(PathBuf::from) .or_else(|| std::env::var_os("HOME").map(|h| PathBuf::from(h).join(".local/share"))) .unwrap_or_else(dr_sync::AccountStore::data_dir); base.join("darkroom") } /// TRACES: FR-NC-10 | NFR-R1 /// Move an account's data out of the cache directory it used to live in. /// /// Called once at startup, before anything opens a store. The durable /// location changed when `catalog_path` stopped falling through to /// `temp_dir()` on Android, and without this the app would find no catalog, /// rescan a library of tens of thousands of images over the network, and /// re-fetch every thumbnail — while the old copy sat in a directory the /// system was free to delete. /// /// Worse than the cost: `sidecars/` and `outbox/` hold work that exists /// nowhere else. Abandoning them would discard offline ratings and edits that /// had not yet synced, silently, as an upgrade. /// /// A rename, not a copy: both directories are inside the app's own data on /// one filesystem, so it is atomic and cannot half-finish. If the destination /// already exists this does nothing — the migration has run, or this is a /// fresh install, and in neither case may it overwrite live data. pub fn migrate_legacy_cache_data(account: &Account) { // Only meaningful where the old fallback and the new one differ, which is // exactly the platform that had the problem. On a desktop with XDG set, // both resolve to the same place and this returns immediately. let legacy_base = std::env::temp_dir(); let Some(current) = catalog_path(account).parent().map(|p| p.to_path_buf()) else { return; }; let Some(account) = current.file_name() else { return; }; let legacy = legacy_base.join("darkroom").join(account); move_account_dir(&legacy, ¤t); } /// The move itself, separated so it can be tested against ordinary /// directories rather than the platform's idea of a cache. fn move_account_dir(legacy: &std::path::Path, current: &std::path::Path) { if legacy == current || !legacy.is_dir() || current.exists() { return; } if let Some(parent) = current.parent() { if let Err(e) = std::fs::create_dir_all(parent) { log::warn!("preparing {}: {e}", parent.display()); return; } } match std::fs::rename(legacy, current) { Ok(()) => log::info!( "moved library data out of the cache: {} -> {}", legacy.display(), current.display() ), // Reported rather than fatal: a failed move leaves the old copy where // it was and costs a rescan, which is recoverable. Stopping the app // over it would not be. Err(e) => log::warn!( "could not move {} to {}: {e}", legacy.display(), current.display() ), } } /// Run a scan on a worker thread, writing results into the catalog. /// /// Returns the receiver the UI drains. The worker owns its own tokio runtime /// and backend; nothing here touches the Slint event loop. pub fn spawn_scan( conn: Connection, root: String, filter: FormatFilter, catalog_path: PathBuf, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let started = std::time::Instant::now(); if let Err(e) = run_scan(&tx, conn, root, filter, catalog_path, started) { let _ = tx.send(ScanMessage::Failed { message: e.message, offline: e.offline, }); } }); rx } /// A scan failure that still knows whether it was a connectivity failure. /// /// The scan crosses a thread boundary, so the typed error cannot travel with /// it; this carries the one bit that must survive. struct ScanFailure { message: String, offline: bool, } impl ScanFailure { /// A failure that is nothing to do with reachability — local I/O, a /// runtime that would not start, a catalog that would not open. fn local(message: impl std::fmt::Display) -> Self { Self { message: message.to_string(), offline: false, } } } impl From for ScanFailure { fn from(e: dr_sync::RemoteError) -> Self { Self { offline: e.indicates_offline(), message: e.to_string(), } } } fn run_scan( tx: &Sender, conn: Connection, root: String, filter: FormatFilter, catalog_path: PathBuf, started: std::time::Instant, ) -> Result<(), ScanFailure> { if let Some(dir) = catalog_path.parent() { std::fs::create_dir_all(dir) .map_err(|e| ScanFailure::local(format!("creating {}: {e}", dir.display())))?; } let catalog = Catalog::open(&catalog_path).map_err(ScanFailure::local)?; let rt = crate::net_runtime::build().map_err(ScanFailure::local)?; rt.block_on(async { let backend = crate::remote::connect(&conn).map_err(ScanFailure::local)?; // Stored folder ETags, so an unchanged subtree is skipped whole. On a // first run this is empty and the walk is complete; on every run after // it is what keeps cost proportional to what changed (ARCH §8.4). let known = load_folder_etags(&catalog, &root); let result = dr_sync::scan(&*backend, &RemotePath::new(&root), &filter, &known, |p| { let _ = tx.send(ScanMessage::Progress { directories: p.directories_listed, pruned: p.directories_pruned, images: p.images_found, }); }) .await?; persist(&catalog, &root, &result).map_err(ScanFailure::local)?; // Report what the catalog holds, not what this pass listed. An // incremental rescan lists only what changed, so its own count is // near zero on a healthy library. let total = total_images(&catalog).unwrap_or(result.images.len()); let _ = tx.send(ScanMessage::Done { found: result.images.len(), total, pruned: result.progress.directories_pruned, elapsed_ms: started.elapsed().as_millis() as u64, }); Ok(()) }) } /// Read back the folder ETags stored by a previous scan. /// /// A failure here is not fatal — an empty map simply means no pruning, which /// is correct but slower. Refusing to scan because the last scan's bookkeeping /// is unreadable would be the worse outcome. fn load_folder_etags( catalog: &Catalog, root: &str, ) -> std::collections::HashMap { let mut out = std::collections::HashMap::new(); let sql = "SELECT f.path, f.etag FROM folders f JOIN roots r ON r.id = f.root_id WHERE r.label = ?1 AND f.etag IS NOT NULL"; let Ok(mut stmt) = catalog.connection().prepare(sql) else { return out; }; let rows = stmt.query_map([root], |r| { Ok((r.get::<_, String>(0)?, r.get::<_, String>(1)?)) }); if let Ok(rows) = rows { for (path, etag) in rows.flatten() { out.insert(RemotePath::new(path), dr_sync::Validator::new(etag)); } } out } /// Write a scan's findings into the catalog. /// /// Images insert at `metadata_state = 1` (stat-only): the scan knows name and /// size but has read no EXIF, and pretending otherwise would make a date /// filter silently wrong. A `Thumbnail` job is enqueued per image, coalescing /// with anything already pending. fn persist( catalog: &Catalog, root: &str, result: &dr_sync::ScanResult, ) -> Result<(), dr_catalog::CatalogError> { let conn = catalog.connection(); let tx = conn.unchecked_transaction()?; // One root row per library folder, reused across scans. tx.execute( "INSERT INTO roots(kind, label, last_seen) VALUES ('remote', ?1, ?2) ON CONFLICT DO NOTHING", rusqlite::params![root, now_secs()], )?; let root_id: i64 = tx.query_row( "SELECT id FROM roots WHERE label = ?1 AND kind = 'remote'", [root], |r| r.get(0), )?; // Folder ETags first — without these persisted, the next scan prunes // nothing and walks the whole tree again (ARCH §6.6). for (path, validator) in &result.directories { tx.execute( "INSERT INTO folders(root_id, path, etag) VALUES (?1, ?2, ?3) ON CONFLICT(root_id, path) DO UPDATE SET etag = excluded.etag", rusqlite::params![root_id, path.as_str(), validator.as_str()], )?; } for entry in &result.images { let folder_id: Option = entry.path.parent().and_then(|p| { tx.query_row( "SELECT id FROM folders WHERE root_id = ?1 AND path = ?2", rusqlite::params![root_id, p.as_str()], |r| r.get(0), ) .ok() }); tx.execute( "INSERT INTO images(root_id, folder_id, source_ref, format, file_size, availability, metadata_state, added_at) VALUES (?1, ?2, ?3, ?4, ?5, 0, 1, ?6) ON CONFLICT(root_id, source_ref) DO UPDATE SET file_size = excluded.file_size, folder_id = excluded.folder_id", rusqlite::params![ root_id, folder_id, entry.path.as_str(), entry .path .name() .rsplit_once('.') .map(|(_, e)| e.to_ascii_lowercase()), entry.size as i64, now_secs(), ], )?; let image_id: i64 = tx.query_row( "SELECT id FROM images WHERE root_id = ?1 AND source_ref = ?2", rusqlite::params![root_id, entry.path.as_str()], |r| r.get(0), )?; // Remote identity, keyed on oc:fileid so a server-side move is a move // rather than a re-download (FR-NC-5). if let RemoteId::Stable(file_id) = entry.id { tx.execute( "INSERT INTO remote(image_id, file_id, etag, remote_path) VALUES (?1, ?2, ?3, ?4) ON CONFLICT(image_id) DO UPDATE SET etag = excluded.etag, remote_path = excluded.remote_path", rusqlite::params![ image_id, file_id as i64, entry.validator.as_str(), entry.path.as_str() ], )?; } } tx.commit()?; // Thumbnail jobs after the commit, so a failure mid-insert does not leave // jobs pointing at rows that never landed. for entry in &result.images { if let Ok(image_id) = conn.query_row( "SELECT id FROM images WHERE root_id = ?1 AND source_ref = ?2", rusqlite::params![root_id, entry.path.as_str()], |r| r.get::<_, i64>(0), ) { let _ = dr_catalog::jobs::enqueue( conn, JobKind::Thumbnail, Some(image_id), Priority::Background, None, ); } } Ok(()) } /// What the grid wants a thumbnail for. /// /// Carries the `oc:fileid` as well as the path, because that is what the /// shared store keys on — stable across a server-side move, and the same id /// every other client sees (FR-NC-5). #[derive(Debug, Clone)] pub struct ThumbnailRequest { pub row: usize, pub path: String, /// `None` where the scan found no stable id; such an image is fetched but /// not stored, since there is no durable key to store it under. pub file_id: Option, /// File length, needed to reject a preview range that points past the end /// of the file (NFR-SEC-1). pub size: u64, /// Catalog row, so EXIF read from the header can be written back. pub image_id: i64, /// Which resolution this cell needs, from how large it is drawn. A zoomed /// grid asks for the large class; a wall of small cells does not. /// /// Named apart from `size`, which is the file's length in bytes — the two /// are unrelated and confusing them would fetch the wrong thing. pub thumb_size: dr_thumbs::ThumbSize, /// Whether this image still needs its EXIF read. Where false the header is /// still fetched — the preview needs it — but nothing is parsed or written. pub needs_metadata: bool, /// Keep the preview at the resolution it was decoded at, ignoring /// `thumb_size`. /// /// For face indexing, which wants the pixels a thumbnail throws away: a /// face 2% across the frame is 5 px on a grid thumbnail and 120 px on the /// embedded preview, and 112 is what the embedder samples. Capped by /// [`FACE_SOURCE_EDGE`] rather than truly unbounded, because a 24 MP buffer /// converted to `f32` RGB is ~288 MB and several lanes hold one at once. pub full_resolution: bool, } /// Why a full fetch failed, keeping the one bit the UI cannot re-derive. /// /// The same reasoning as [`ScanFailure`]: the typed error cannot cross the /// channel, and "offline" versus "refused" decides whether develop shows /// "you are offline — this image is not stored locally" or a real error. #[derive(Debug)] pub struct FetchFailure { pub message: String, pub offline: bool, } impl FetchFailure { fn local(message: impl std::fmt::Display) -> Self { Self { message: message.to_string(), offline: false, } } } impl From for FetchFailure { fn from(e: dr_sync::RemoteError) -> Self { Self { offline: e.indicates_offline(), message: e.to_string(), } } } impl std::fmt::Display for FetchFailure { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { f.write_str(&self.message) } } /// TRACES: FR-NC-6a /// Progress from the pin worker. #[derive(Debug)] pub enum PinMessage { /// How many originals the pin still needs. Sent once, before any transfer. Planned { total: usize, }, /// One original landed. Stored { done: usize, }, /// The pin is fully downloaded. Done { stored: usize, bytes: u64, }, Failed { message: String, offline: bool, }, } /// TRACES: FR-NC-6a /// Download every original a pin has asked for. /// /// Whole files, deliberately: a pin exists so the photographs can be *edited* /// away from the server, and develop needs every photosite. This is the one /// place in the app that fetches originals in bulk, which is why FR-NC-6 /// makes it opt-in rather than something sync does on its own. /// /// Sequential rather than parallel. The lanes that make the thumbnail sweep /// fast are wrong here: these are tens of megabytes each, so concurrency buys /// little against a single connection's bandwidth and costs a great deal of /// memory — and it is the same contention that produced 423 Locked in the /// sweep. pub fn spawn_pin_fetch( conn: Connection, catalog_path: PathBuf, cache_dir: PathBuf, budget: dr_catalog::Budget, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let (store, catalog) = match ( dr_catalog::Cache::open(&cache_dir, budget), Catalog::open(&catalog_path), ) { (Ok(s), Ok(c)) => (s, c), (Err(e), _) | (_, Err(e)) => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: false, }); return; } }; let pending = match store.pending_pins(catalog.connection()) { Ok(p) => p, Err(e) => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: false, }); return; } }; if tx .send(PinMessage::Planned { total: pending.len(), }) .is_err() { return; } if pending.is_empty() { let _ = tx.send(PinMessage::Done { stored: 0, bytes: 0, }); return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: false, }); return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: false, }); return; } }; let mut stored = 0usize; let mut bytes_total = 0u64; for image in pending { let Some(source_ref) = source_ref_of(&catalog, image) else { // Catalogued and then removed while the pin was pending. continue; }; let id = RemoteId::Path(RemotePath::new(&source_ref)); // TRACES: FR-NC-6c // On a placeholder library "pin" means *keep it downloaded*, // not "make a second copy". The original materialises in the // library folder itself, so copying it under `originals/` // would hold every pinned photograph twice — and the copy // would be the half the budget could evict while the real disk // cost stayed. Only the bookkeeping is recorded, with no path, // so nothing here can ever delete a file inside a synced tree // (see `Cache::record_in_place`). if backend.capabilities().materialisation.can_materialise() { match backend.materialise(&id).await { Ok(_) => { let bytes = size_of(&catalog, image).unwrap_or(0); if let Err(e) = store.record_in_place( catalog.connection(), image, bytes, true, now_secs(), ) { log::warn!("recording pinned {source_ref}: {e}"); continue; } stored += 1; bytes_total += bytes; if tx.send(PinMessage::Stored { done: stored }).is_err() { return; } } Err(e) if e.indicates_offline() => { let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: true, }); return; } Err(e) => log::warn!("pinning {source_ref}: {e}"), } continue; } match backend.get(&id, None).await { Ok(bytes) => { // `pinned: true` — this is the population the budget // must never evict, which is the entire promise the // user made when they pinned the collection. if let Err(e) = store.store( catalog.connection(), image, &source_ref, &bytes, true, now_secs(), ) { log::warn!("storing pinned {source_ref}: {e}"); continue; } stored += 1; bytes_total += bytes.len() as u64; if tx.send(PinMessage::Stored { done: stored }).is_err() { return; } } Err(e) if e.indicates_offline() => { // Stop rather than failing each remaining file against // a dead connection. What was downloaded stays // downloaded, and `pending_pins` resumes from there. let _ = tx.send(PinMessage::Failed { message: e.to_string(), offline: true, }); return; } Err(e) => { // One unreadable file must not abandon the whole pin. log::warn!("pinning {source_ref}: {e}"); } } } let _ = tx.send(PinMessage::Done { stored, bytes: bytes_total, }); }); }); rx } /// TRACES: FR-NC-6c /// Hand a set of photographs back to the sync client, freeing their disk. /// /// The other half of pinning on a placeholder library. `Cache::release` drops /// the bookkeeping and — correctly — deletes nothing, because the rows it /// holds for a library like this name no file of ours (`record_in_place`). /// The bytes are in the library folder, and only the client may take them /// back. /// /// **This is a dehydration, not a deletion, and the distinction is the whole /// safety of the feature.** Removing a materialised file inside a synced tree /// propagates to the server and deletes the photograph everywhere. /// /// Best effort per image: a file the client refuses to release simply stays, /// which costs disk and loses nothing. pub fn spawn_dehydrate( conn: Connection, catalog_path: PathBuf, images: Vec, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let Ok(catalog) = Catalog::open(&catalog_path) else { return; }; let Ok(rt) = crate::net_runtime::build() else { return; }; rt.block_on(async { let Ok(backend) = crate::remote::connect(&conn) else { return; }; // Nothing to do where content is not a thing that can be given // back — a server library, or a plain folder. if !backend.capabilities().materialisation.can_materialise() { return; } let mut released = 0usize; for image in images { let Some(source_ref) = source_ref_of(&catalog, image) else { continue; }; let id = RemoteId::Path(RemotePath::new(&source_ref)); match backend.dematerialise(&id).await { Ok(()) => released += 1, Err(e) => log::debug!("releasing {source_ref}: {e}"), } } log::info!("released {released} photograph(s) back to the sync client"); let _ = tx.send(released); }); }); rx } /// What an image occupies, as the catalog recorded it. /// /// Zero where the scan could not tell — a placeholder reports no size, because /// a one-byte stub says nothing about what it stands for (ARCH §9.0a). A pin /// that cannot state its cost is better than one that states a wrong one. fn size_of(catalog: &Catalog, image: dr_types::ImageId) -> Option { catalog .connection() .query_row( "SELECT file_size FROM images WHERE id = ?1", rusqlite::params![image.0 as i64], |r| r.get::<_, Option>(0), ) .ok() .flatten() .map(|v| v.max(0) as u64) } /// The remote path for a catalogued image. fn source_ref_of(catalog: &Catalog, image: dr_types::ImageId) -> Option { catalog .connection() .query_row( "SELECT source_ref FROM images WHERE id = ?1", rusqlite::params![image.0 as i64], |r| r.get(0), ) .ok() } /// TRACES: FR-NC-6a | FR-CAT-9 /// Where a cached original is kept and how much may be kept. /// /// Passed in rather than derived here so the caller owns the policy: the /// budget is a user setting, and this function is on a worker thread with no /// access to one. pub struct CacheContext { pub dir: PathBuf, pub catalog_path: PathBuf, pub image: dr_types::ImageId, pub budget: dr_catalog::Budget, /// Whether a downloaded original is kept. /// /// Only the write. A cache is always *read*, because bytes already on disk /// cost nothing to use and declining them would re-download an image that /// is present — including every pinned one, which would leave a pinned /// collection unopenable offline the moment this was switched off. pub store: bool, } /// Fetch one file in full, for opening it in develop. /// /// Deliberately *not* the preview path. Browsing fetches a range and decodes /// an embedded JPEG (FR-NC-3); develop needs every byte, because demosaic /// needs every photosite. On a RAW file that is tens of megabytes, which is /// why this is a click-triggered download and not something the grid does. /// /// # Read-through /// /// With a `cache`, this checks disk before the network and stores what it /// downloads. That is what makes opening the same photograph twice cost one /// transfer, and what leaves a working session's images openable offline /// without anyone having pinned anything. /// /// A cache miss is not an error and a cache failure is not fatal: both fall /// through to the network, which is exactly the behaviour that existed before /// the cache did. /// /// Returns the bytes on a channel rather than blocking: the download runs on /// its own thread and the UI stays live, exactly as thumbnail fetching does. /// TRACES: FR-CAT-8 | FR-DEV-6 /// Fetch and parse the sidecar beside one image. /// /// # Why the edit is read from the file rather than the catalog /// /// The catalog carries a `graph_hash` and no parameters, and it is /// *disposable* (ARCH §6.12) — a rebuild would silently return every /// photograph to neutral. The sidecar is the authoritative store, so it is /// what an open reads, and that is also what makes an edit pasted on the /// desktop appear when the same frame is opened on the phone. /// /// # Why absence and failure are the same answer here /// /// `None` means "open this image at its defaults", which is right for a /// photograph that has never been edited — the overwhelmingly common case on a /// fresh library — and equally right when the network is down. The alternative, /// refusing to open the image because its sidecar could not be read, would make /// an unreachable server also mean an unviewable library. /// /// The one case that is *not* harmless is a sidecar that exists but does not /// parse. That still opens at defaults, but the write path /// ([`write_one_sidecar`]) independently refuses to overwrite a file it could /// not read, so an edit this build failed to understand is never destroyed by /// having been opened. pub fn spawn_sidecar_fetch( conn: Connection, image_path: String, cache_dir: PathBuf, offline: bool, ) -> Receiver> { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let cache = SidecarCache::open(cache_dir); let path_str = sidecar_path(&image_path); // TRACES: FR-CAT-9 | FR-NC-10 // The cache wins outright when it is holding work the server has not // seen. Fetching in that state would answer with a document *older* // than the edit sitting in the outbox, and opening the photograph // would silently show it without the change the user just made — // which the next save would then write back over the top of. if cache.is_pending(&path_str) { log::debug!("{path_str} has queued local edits; opening from the cache"); let _ = tx.send(cache.load(&path_str)); return; } // Offline there is nothing to ask, and the cache is the whole answer. let rt = if offline { None } else { match crate::net_runtime::build() { Ok(e) => Some(e), Err(e) => { log::debug!("sidecar fetch runtime: {e}"); None } } }; let Some(rt) = rt else { let _ = tx.send(cache.load(&path_str)); return; }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { log::debug!("sidecar fetch backend: {e}"); let _ = tx.send(cache.load(&path_str)); return; } }; let path = RemotePath::new(path_str.clone()); let id = RemoteId::Path(path.clone()); // A 404 is the normal case on a library that has never been // edited, so this is `ok()` rather than an error path. let Ok(bytes) = backend.get(&id, None).await else { // Unreachable, or no such file. The cache cannot tell those // apart and does not need to: either way it holds the best // answer this device has. let _ = tx.send(cache.load(&path_str)); return; }; let text = String::from_utf8_lossy(&bytes).into_owned(); let parsed = match dr_pipeline::Sidecar::parse(&text) { Ok(s) => Some(s), Err(e) => { log::warn!("sidecar at {} is unreadable ({e})", path.as_str()); None } }; // Populate the cache from what the server said, so the *next* // open of this photograph works with no connection. Clean rather // than pending: this content came from the server, so there is // nothing to send back. if let Some(sidecar) = parsed.as_ref() { if let Err(e) = cache.store(&path_str, sidecar, false) { log::debug!("caching {path_str}: {e}"); } } let _ = tx.send(parsed); }); }); rx } pub fn spawn_full_fetch( conn: Connection, path: String, cache: Option, ) -> Receiver, FetchFailure>> { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { // Opened on this thread: `rusqlite::Connection` is not `Send`, and the // UI thread's handle cannot be borrowed across the spawn. let cached = cache.as_ref().and_then(|c| { let store = dr_catalog::Cache::open(&c.dir, c.budget).ok()?; let conn = Catalog::open(&c.catalog_path).ok()?; Some((store, conn)) }); if let (Some(c), Some((store, conn))) = (cache.as_ref(), cached.as_ref()) { match store.load(conn.connection(), c.image, now_secs()) { Ok(Some(bytes)) => { log::info!("{path}: {} bytes from the local cache", bytes.len()); let _ = tx.send(Ok(bytes)); return; } Ok(None) => {} // A cache that cannot be read is a cache miss, not a failure // to open the photograph. Err(e) => log::debug!("cache lookup for {path}: {e}"), } } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { let _ = tx.send(Err(FetchFailure::local(e))); return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { let _ = tx.send(Err(FetchFailure::local(e))); return; } }; let id = RemoteId::Path(RemotePath::new(&path)); let got = backend.get(&id, None).await.map_err(FetchFailure::from); // Store before sending, so the bytes are on disk by the time the // image is on screen. Doing it after would leave a window where // closing the app immediately lost the download. if let (Ok(bytes), Some(c), Some((store, conn))) = (&got, cache.as_ref().filter(|c| c.store), cached.as_ref()) { // `pinned: false` — this is the passive population. A pin is // something the user asks for explicitly; opening an image is // not that, and treating it as one would make the pinned set // grow silently and never be evicted. if let Err(e) = store.store(conn.connection(), c.image, &path, bytes, false, now_secs()) { log::debug!("caching {path}: {e}"); } else if let Err(e) = store.enforce(conn.connection()) { log::debug!("enforcing the cache budget: {e}"); } } let _ = tx.send(got); }); }); rx } /// Serve thumbnails for a set of rows: store first, network second. /// /// The store is consulted before any request goes out, so a second launch — /// or a second device that synced the shards — fills the grid with no transfer /// at all. Only genuine misses reach the network. pub fn spawn_thumbnails( conn: Connection, wanted: Vec, store_dir: PathBuf, catalog_path: PathBuf, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let mut store = match ThumbStore::open(&store_dir) { Ok(s) => Some(s), Err(e) => { // A broken store costs speed, never correctness — every // thumbnail can still be fetched. log::warn!("thumbnail store unavailable, fetching everything: {e}"); None } }; // Split the batch before delivering anything, so the plan can be // reported first and the UI knows the shape of the work up front. // Decoding happens here rather than in the split, because a corrupt // blob turns a hit into a miss. let mut hits = Vec::new(); let mut to_fetch = Vec::new(); // Images whose thumbnail is cached but whose date is still unknown. // // These need a header read even though no pixels are wanted. Without // this pass an image is dated *only* on the one visit that produced // its thumbnail — so a library browsed once before the EXIF code // existed, or synced from another device's shards, stays permanently // undated and never appears on the timeline. let mut metadata_only = Vec::new(); for req in wanted { let stored = req .file_id .zip(store.as_ref()) .and_then(|(id, s)| s.get(id, req.thumb_size).ok().flatten()); match stored.map(|t| dr_thumbs::decode_rgba(&t.bytes)) { Some(Ok((width, height, rgba))) => { if req.needs_metadata { metadata_only.push(req.clone()); } hits.push(ThumbnailReady { row: req.row, width, height, rgba, from_cache: true, }); } // A corrupt stored blob is a miss, not a failure. Some(Err(e)) => { log::debug!("stored thumbnail unreadable, refetching: {e}"); to_fetch.push(req); } None => to_fetch.push(req), } } log::info!( "thumbnails: {} from store, {} to fetch{}", hits.len(), to_fetch.len(), if metadata_only.is_empty() { String::new() } else { format!(" · {} dates to read", metadata_only.len()) } ); if tx .send(ThumbnailMessage::Plan { cached: hits.len(), fetching: to_fetch.len(), dating: metadata_only.len(), }) .is_err() { return; } for hit in hits { if tx.send(ThumbnailMessage::Ready(Box::new(hit))).is_err() { return; } } if to_fetch.is_empty() && metadata_only.is_empty() { return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { for req in &to_fetch { let _ = tx.send(ThumbnailMessage::Unavailable { row: req.row, reason: e.to_string(), }); } return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { for req in &to_fetch { let _ = tx.send(ThumbnailMessage::Unavailable { row: req.row, reason: e.to_string(), }); } return; } }; // Batched rather than written per image: one transaction per // batch instead of 120, and the grid does not need each date the // instant it is read. let mut found = Vec::new(); // Set when the server proves unreachable, which abandons the rest // of the batch. The remaining cells would each take a full timeout // to reach the same conclusion — on a 120-cell window, minutes of // the grid appearing to load against a server that is not there. let mut offline = false; for req in to_fetch { let msg = fetch_one(&*backend, store.as_mut(), &req, &mut found).await; offline = matches!(msg, ThumbnailMessage::Offline { .. }); // A closed channel means the window went away mid-fetch. if tx.send(msg).is_err() || offline { break; } } // Dates for images whose pixels were already cached. Header only — // no preview range, no decode. // // Flushed in chunks rather than once at the end: 119 sequential // header fetches take tens of seconds, and a single write at the // finish loses every one of them if the window closes first. It // also lets the timeline appear while the rest are still arriving. // // Skipped entirely when the connection has already failed: these // are network reads too, and there is nothing left to read from. const FLUSH_EVERY: usize = 16; if !offline { log::info!("reading dates for {} image(s)", metadata_only.len()); for req in metadata_only { if tx.send(ThumbnailMessage::DateProgress).is_err() { break; } read_metadata_only(&*backend, &req, &mut found).await; if found.len() >= FLUSH_EVERY { flush_metadata(&catalog_path, &mut found, &tx); } } } // Always flushed, even when the batch was abandoned: whatever was // read before the connection died is still true, and discarding it // would mean re-fetching those headers next time. flush_metadata(&catalog_path, &mut found, &tx); }); }); rx } async fn fetch_one( backend: &dyn RemoteBackend, store: Option<&mut ThumbStore>, req: &ThumbnailRequest, found_metadata: &mut Vec, ) -> ThumbnailMessage { let preview = match fetch_preview(backend, req, found_metadata).await { PreviewOutcome::Ready(p) => p, PreviewOutcome::Unavailable(reason) => { return ThumbnailMessage::Unavailable { row: req.row, reason, } } PreviewOutcome::Offline(reason) => return ThumbnailMessage::Offline { reason }, }; // Persist for next time, and for every other client that syncs the shard. // A store failure is logged and dropped: the pixels are already in hand, // and refusing to display them because they could not be cached would be // the wrong trade. if let (Some(store), Some(file_id)) = (store, req.file_id) { if let Some(thumb) = encode_preview(file_id, &preview) { store_thumbnail(store, file_id, req.thumb_size, &thumb); } } ThumbnailMessage::Ready(Box::new(ThumbnailReady { row: req.row, width: preview.width, height: preview.height, rgba: preview.rgba, from_cache: false, })) } /// What one fetch produced. /// /// Separate from [`ThumbnailMessage`] because not every caller has a grid row /// to report against or a store to write through. The whole-library pass /// ([`spawn_thumbnail_sweep`]) fetches on several lanes at once and stores the /// results on the one thread that owns the store, so it needs the pixels /// *before* anything is written or addressed to a cell. enum PreviewOutcome { Ready(dr_decode::Preview), /// This image has no usable preview. The batch continues past it. Unavailable(String), /// The server is unreachable, so nothing after this would succeed either. Offline(String), } /// Fetch a preview in two stages: header, then the exact preview range. /// /// This is what FR-NC-3 specifies, and the single-stage version it replaces /// was wrong in a way that looked like corruption: fetching a fixed prefix cut /// the embedded JPEG partway through, and decoders render a truncated JPEG as /// the top fraction of the frame rather than reporting an error. async fn fetch_preview( backend: &dyn RemoteBackend, req: &ThumbnailRequest, found_metadata: &mut Vec, ) -> PreviewOutcome { let id = RemoteId::Path(RemotePath::new(&req.path)); // A connection failure is not this image's verdict. Reported as such so // the caller can stop the batch rather than marking sixty cells // individually unpreviewable over one dropped connection — a state the // grid would then keep until something forced a reload. let classify = |e: dr_sync::RemoteError| { if e.indicates_offline() { PreviewOutcome::Offline(e.to_string()) } else { PreviewOutcome::Unavailable(e.to_string()) } }; // Stage one: the header, enough to parse the container's IFDs. let header = match backend.get(&id, Some(0..dr_decode::HEADER_BYTES)).await { Ok(b) => b, Err(e) => return classify(e), }; // The same bytes carry EXIF. Reading it here is free — the alternative is // a second 256 KB fetch per image over the whole library. if req.needs_metadata { collect_metadata(&header, req, found_metadata); } // Read unconditionally, unlike the rest of the EXIF above: `needs_metadata` // is false once an image has been catalogued, but a thumbnail can still be // regenerated long after that — a cleared cache, a new size — and a // thumbnail that came out upright the first time must come out upright // every time. This is a header walk, not a decode; see `dr_decode::orientation`. let orientation = dr_decode::orientation(&header).unwrap_or_default(); // A plain JPEG is its own preview; anything else needs locating. let bytes = if header.starts_with(&[0xFF, 0xD8, 0xFF]) { match backend.get(&id, None).await { Ok(b) => b, Err(e) => return classify(e), } } else { let Some(loc) = dr_decode::locate_preview(&header, req.size) else { // No locatable preview. Declining beats fetching the whole file: // that is the 370 GB path FR-NC-3 exists to avoid. return PreviewOutcome::Unavailable("no locatable embedded preview".into()); }; if loc.len() > MAX_PREVIEW_BYTES { return PreviewOutcome::Unavailable(format!( "preview is {} bytes, too large", loc.len() )); } // Stage two: exactly the preview's bytes. match backend.get(&id, Some(loc.range.clone())).await { Ok(b) => b, Err(e) => return classify(e), } }; // Verify before decoding. A truncated JPEG decodes "successfully" into a // partial frame, so without this the broken result reaches the cache and // the screen looking like a corrupt file. if !dr_decode::is_complete_jpeg(&bytes) { return PreviewOutcome::Unavailable("preview bytes are incomplete".into()); } // Decode on the worker, never the UI thread. let mut preview = match dr_decode::decode_jpeg(&bytes) { Ok(p) => p, Err(e) => return PreviewOutcome::Unavailable(e.to_string()), }; // Face indexing keeps the detail; every other caller is filling a cell of a // known size and the full preview is waste from here on. preview.downscale_to(if req.full_resolution { FACE_SOURCE_EDGE } else { req.thumb_size.edge() }); // Turn it the right way up before it is measured, cached or shown. An // embedded preview is written in the sensor's orientation, so without this // every frame shot in portrait lies on its side in the grid — and, because // the cache is keyed by file and size alone, stays that way. // // After the downscale, so the permutation moves thumbnail-sized bytes // rather than the full preview's. // // **Doing it here is what keeps face geometry honest.** Detection runs on // whatever this returns, so returning the photograph rather than the sensor // means every box and landmark is already in the space the catalog stores // and the develop overlay draws — no second mapping to get backwards, which // is the one orientation bug this codebase keeps having. It costs a // permutation of a larger buffer for the face path; that is ~15 ms against // a decode of ~150 ms, and it buys the whole class of bug. preview.apply_orientation(orientation); PreviewOutcome::Ready(preview) } /// Compress a decoded preview to what the store holds. /// /// Split from the write so the whole-library pass can do it on the lane that /// fetched the image: encoding is the one part of storing a thumbnail that /// costs CPU rather than the store's lock, and it turns a 256 KB RGBA buffer /// into ~20 KB before the chunk is handed to the single thread that owns the /// store. fn encode_preview(file_id: u64, preview: &dr_decode::Preview) -> Option { match dr_thumbs::encode_rgba(preview.width, preview.height, &preview.rgba) { Ok(bytes) => Some(dr_thumbs::Thumbnail { width: preview.width, height: preview.height, bytes, }), Err(e) => { log::debug!("encoding thumbnail {file_id}: {e}"); None } } } /// Put a thumbnail in the store, logging rather than failing. /// /// A store failure costs a re-fetch next time and nothing else — the pixels /// are already in hand, and the caller has something to show or count either /// way (ARCH §6.12: the store is derived, never authoritative). fn store_thumbnail( store: &mut ThumbStore, file_id: u64, size: dr_thumbs::ThumbSize, thumb: &dr_thumbs::Thumbnail, ) -> bool { match store.put(file_id, size, thumb) { Ok(_) => true, Err(e) => { log::debug!("storing thumbnail {file_id}: {e}"); false } } } /// Parse EXIF out of a header and record it. /// /// Shared by both paths — the thumbnail fetch, which gets the header anyway, /// and the header-only pass for images whose pixels were already cached. fn collect_metadata(header: &[u8], req: &ThumbnailRequest, out: &mut Vec) { let Ok(md) = dr_decode::metadata(header) else { return; }; out.push(MetadataFound { image_id: req.image_id, captured_at: md.captured_at, captured_offset: md.captured_offset, camera: camera_label(md.make.as_deref(), md.model.as_deref()), lens: md.lens.map(|l| l.trim().to_string()), iso: md.iso, }); } /// TRACES: FR-CAT-11 /// The camera string the catalog stores, from an EXIF make and model. /// /// One definition rather than one per caller, because an import's duplicate /// check compares against what a scan wrote (`dr_catalog::dedup`). Two /// spellings of the same body would not fail loudly — they would silently /// disable the cheap tier, and every re-inserted card would transfer in full /// before the digest caught it. pub fn camera_label(make: Option<&str>, model: Option<&str>) -> Option { match (make, model) { // Bodies repeat the make inside the model ("Canon EOS 6D"), so // joining unconditionally yields "Canon Canon EOS 6D". (Some(make), Some(model)) if model.starts_with(make) => Some(model.trim().to_string()), (Some(make), Some(model)) => Some(format!("{} {}", make.trim(), model.trim())), (None, Some(model)) => Some(model.trim().to_string()), _ => None, } } /// Write a batch of dates and tell the UI, draining `found`. /// /// Separate from the loop so the same path serves both the periodic flush and /// the final one, and so a write failure is reported once rather than being /// silently swallowed by the caller. fn flush_metadata( catalog_path: &std::path::Path, found: &mut Vec, tx: &Sender, ) { if found.is_empty() { return; } match Catalog::open(catalog_path) { Ok(cat) => match write_metadata(&cat, found) { Ok(n) => { log::info!("recorded capture dates for {n} of {} image(s)", found.len()); // Tell the UI so the timeline can appear. Without this the // histogram only shows up on the next window load, which on a // fully cached library may be never. let _ = tx.send(ThumbnailMessage::DatesRecorded(n)); } Err(e) => log::warn!("writing metadata: {e}"), }, Err(e) => log::warn!("opening catalog to write metadata: {e}"), } found.clear(); } /// Read only the date for an image whose thumbnail is already cached. /// /// One 256 KB header request, no preview range and no decode. This is what /// gets a library dated when its thumbnails came from the store — including /// shards synced from another device, which carry pixels but no metadata. /// Read a header for its date. /// /// Returns whether the file was **reached**, which the caller needs and cannot /// otherwise tell: a header that carried no EXIF and a fetch that never /// happened both leave `found` untouched, and recording the second as "this /// image has no date" would let one lock mark it dateless for good. async fn read_metadata_only( backend: &dyn RemoteBackend, req: &ThumbnailRequest, found: &mut Vec, ) -> bool { let id = RemoteId::Path(RemotePath::new(&req.path)); // Retried, because one failure here is usually a lock rather than a // verdict. Nextcloud's file locking answers a plain *read* with 423 under // concurrency, and the identical range succeeds moments later — measured // against a real server while twelve lanes were running. Without a retry // those images sit out the whole pass over a lock that lasted a moment. // // Bounded and short: a genuinely missing or forbidden file must not cost // three round trips before the sweep moves on. const ATTEMPTS: usize = 3; for attempt in 1..=ATTEMPTS { match backend.get(&id, Some(0..dr_decode::HEADER_BYTES)).await { Ok(header) => { collect_metadata(&header, req, found); return true; } Err(e) if e.is_transient() && attempt < ATTEMPTS => { // Backing off at all matters more than the exact interval: the // contention that produced the lock is our own lanes, so any // pause lets the holder finish. tokio::time::sleep(std::time::Duration::from_millis(200 * attempt as u64)).await; } Err(e) => { // Not surfaced: a missing date leaves the image off the // timeline rather than breaking anything, and the next sweep // retries it regardless. log::debug!("reading date for {} ({attempt} attempts): {e}", req.path); return false; } } } false } /// Write capture metadata read during the thumbnail pass. /// /// Promotes each row from `metadata_state = 1` (stat-only) to 2 (full EXIF), /// which is what makes it eligible for the timeline. A row whose EXIF was /// unreadable stays at 1 rather than being marked done with empty fields, so a /// later attempt can retry it. /// /// Returns how many rows were promoted. pub fn write_metadata( catalog: &Catalog, found: &[MetadataFound], ) -> Result { let conn = catalog.connection(); let tx = conn.unchecked_transaction()?; let mut promoted = 0; for m in found { // Only a real timestamp counts as fully read. Camera and lens without // a date leave the image unplaceable on a timeline, which is exactly // the state the grid needs to distinguish. let state = if m.captured_at.is_some() { 2 } else { 1 }; tx.execute( "UPDATE images SET captured_at = coalesce(?2, captured_at), captured_offset = coalesce(?3, captured_offset), camera = coalesce(?4, camera), lens = coalesce(?5, lens), iso = coalesce(?6, iso), metadata_state = max(metadata_state, ?7) WHERE id = ?1", rusqlite::params![ m.image_id, m.captured_at, m.captured_offset, m.camera, m.lens, m.iso, state, ], )?; if state == 2 { promoted += 1; } } tx.commit()?; Ok(promoted) } /// Progress from the whole-library sweep. #[derive(Debug)] pub enum SweepMessage { /// How many images still need work, counted once at the start. Total(usize), /// Another chunk finished. Carries cumulative counts. Progress { done: usize, dated: usize, }, Finished { dated: usize, }, } /// Await every future concurrently, returning results in order. /// /// A hand-rolled `join_all` rather than a `futures` dependency for one /// function. Polling a `Vec` of futures in a loop is exactly what the crate's /// version does; the ordering guarantee is what lets the caller pair results /// back to their inputs. async fn futures_join_all(futures: impl IntoIterator) -> Vec where F: std::future::Future, { use std::pin::Pin; use std::task::Poll; // Boxed so each future has a stable address while it is polled in place. let mut pending: Vec>>> = futures.into_iter().map(|f| Some(Box::pin(f))).collect(); let mut done: Vec> = (0..pending.len()).map(|_| None).collect(); std::future::poll_fn(move |cx| { let mut all_ready = true; for (slot, out) in pending.iter_mut().zip(done.iter_mut()) { let Some(fut) = slot else { continue }; match fut.as_mut().poll(cx) { Poll::Ready(v) => { *out = Some(v); // Dropped as soon as it completes, so a long-running lane // does not hold a finished one's resources. *slot = None; } Poll::Pending => all_ready = false, } } if all_ready { Poll::Ready(done.iter_mut().filter_map(Option::take).collect()) } else { Poll::Pending } }) .await } /// How many images one sweep chunk handles before committing. /// /// Small enough that a kill loses little, large enough that the catalog is not /// reopened per image. A multiple of [`SWEEP_LANES`] so every lane gets equal /// work and no chunk ends with most lanes idle. const SWEEP_CHUNK: usize = 96; /// How many fetches the sweep keeps in flight. /// /// Each is ~0.6 s of round-trip latency and almost no bandwidth — a 256 KB /// header — so the sequential version spent essentially all its time waiting. /// Twelve lanes turn ~3 hours into ~15 minutes on the reference library. /// /// Deliberately bounded rather than unlimited: the grid's own interactive /// fetches share this server, and a sweep that saturated the connection would /// make browsing feel broken while it ran. /// /// **Lowered from twelve after measuring.** Twelve produced 423 Locked on a /// real server — Nextcloud's file locking answering a plain read under /// contention we were creating ourselves. Six keeps most of the speedup /// without provoking it; the retry above covers what still slips through. const SWEEP_LANES: usize = 6; /// Date **every** image in the library, not just the ones on screen. /// /// Thumbnails are deliberately *not* fetched here. A thumbnail needs the /// mutable store, which cannot be shared across the parallel lanes below, and /// it costs 1–3 MB against a date's 256 KB. Dating the whole library is what /// the timeline needs; thumbnails arrive as cells are actually browsed, which /// is the FR-NC-3 posture anyway. /// /// The grid's own fetches cover what is on screen; this covers the rest, so the /// timeline describes the whole library rather than the part that happened to /// be scrolled past. It is resumable by construction — each pass queries for /// what is still missing, so a kill mid-sweep costs only the current chunk. /// /// Runs at the back of the queue by design: it holds no lock the grid needs, /// and its chunked commits keep write transactions short. pub fn spawn_sweep(conn: Connection, catalog_path: PathBuf) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let catalog = match Catalog::open(&catalog_path) { Ok(c) => c, Err(e) => { // Silent failure here left the sweep looking like it had run // and found nothing: no progress, no error, 17,397 images // still unindexed. log::warn!( "sweep: cannot open catalog at {}: {e}", catalog_path.display() ); let _ = tx.send(SweepMessage::Finished { dated: 0 }); return; } }; let outstanding = count_outstanding(&catalog).unwrap_or(0); if outstanding == 0 { let _ = tx.send(SweepMessage::Finished { dated: 0 }); return; } log::info!("sweep: {outstanding} image(s) need a date or a thumbnail"); if tx.send(SweepMessage::Total(outstanding)).is_err() { return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { log::warn!("sweep: no runtime: {e}"); return; } }; rt.block_on(async { let Ok(backend) = crate::remote::connect(&conn) else { return; }; let (mut done, mut dated) = (0usize, 0usize); loop { // Re-queried each pass rather than held as one long list: the // grid is dating images at the same time, and a stale list // would refetch what it already covered. let chunk = match next_outstanding(&catalog, SWEEP_CHUNK) { Ok(c) if c.is_empty() => break, Ok(c) => c, Err(e) => { log::warn!("sweep: {e}"); break; } }; let chunk_started = std::time::Instant::now(); log::debug!( "sweep: chunk of {} starting at image {}", chunk.len(), chunk[0].image_id ); // Twelve lanes over the chunk. Each lane owns a disjoint slice // and its own `found` vector, so nothing is shared and no lock // is needed; the results are concatenated after the join. // // The thumbnail store is the exception — it is `&mut` and // cannot be shared — so lanes only *read* metadata and any // missing thumbnail is left to the interactive path. Dating the // library is what the sweep is for; thumbnails arrive as cells // are browsed. let lanes: Vec> = (0..SWEEP_LANES) .map(|lane| chunk.iter().skip(lane).step_by(SWEEP_LANES).collect()) .collect(); let results = futures_join_all(lanes.into_iter().map(|lane| { let backend: &dyn RemoteBackend = &*backend; async move { let mut found = Vec::new(); let mut reached = Vec::new(); for req in lane { if read_metadata_only(backend, req, &mut found).await { reached.push(req.image_id); } } (found, reached) } })) .await; let mut found = Vec::new(); let mut reached = std::collections::HashSet::new(); for (lane_found, lane_reached) in results { found.extend(lane_found); reached.extend(lane_reached); } done += chunk.len(); // An image whose header carried no EXIF at all yields nothing // to `found`, so nothing marks it examined and the next sweep // fetches it again — for ever. Darktable exports strip // metadata by default, and 2,188 of them in the reference // library meant 2,188 pointless round trips per run. // // Recorded as examined with no date: the file was read and // genuinely has none, which is a different state from "not // looked at yet" and must not be confused with it. let answered: std::collections::HashSet = found.iter().map(|m| m.image_id).collect(); // Only files actually read. One that could not be fetched is // left alone so the next pass retries it, rather than being // written off over a lock or a dropped connection. found.extend( chunk .iter() .filter(|r| { reached.contains(&r.image_id) && !answered.contains(&r.image_id) }) .map(|r| MetadataFound { image_id: r.image_id, captured_at: None, captured_offset: None, camera: None, lens: None, iso: None, }), ); dated += found.iter().filter(|m| m.captured_at.is_some()).count(); let read = answered.len(); flush_sweep(&catalog, &mut found); log::info!( "sweep: {done} done, {dated} dated ({read} read in {:.1}s)", chunk_started.elapsed().as_secs_f64() ); if tx.send(SweepMessage::Progress { done, dated }).is_err() { return; } } log::info!("sweep complete: {dated} date(s) recorded over {done} image(s)"); let _ = tx.send(SweepMessage::Finished { dated }); }); }); rx } /// How many images still lack a date or a thumbnail. fn count_outstanding(catalog: &Catalog) -> Result { let n: i64 = catalog.connection().query_row( &format!( "SELECT count(*) FROM images WHERE metadata_state < 2 AND {VISIBLE_UNALIASED}" ), [], |r| r.get(0), )?; Ok(n as usize) } /// The next images needing work. /// /// Ordered by id so the sweep advances deterministically and a resumed run /// picks up where it left off rather than revisiting. fn next_outstanding( catalog: &Catalog, limit: usize, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT i.id, i.source_ref, r.file_id, i.file_size FROM images i LEFT JOIN remote r ON r.image_id = i.id WHERE i.metadata_state < 2 AND {VISIBLE} ORDER BY i.id LIMIT ?1" ))?; let rows = stmt .query_map([limit as i64], |r| { Ok(ThumbnailRequest { // The sweep indexes dates, and reads headers only — the size // never reaches a fetch, but it must name something. thumb_size: dr_thumbs::ThumbSize::Grid, // Row index is meaningless here — the sweep touches no grid // cell, so nothing consumes it. row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id: r.get::<_, Option>(2)?.map(|v| v as u64), size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, needs_metadata: true, full_resolution: false, }) })? .collect::, _>>()?; Ok(rows) } /// Commit a sweep chunk. /// /// An image whose header yielded no date is still marked done, or the sweep /// would revisit it forever. `write_metadata` records `metadata_state = 1` for /// those, so this promotes them explicitly. fn flush_sweep(catalog: &Catalog, found: &mut Vec) { if found.is_empty() { return; } if let Err(e) = write_metadata(catalog, found) { log::warn!("sweep: writing metadata: {e}"); found.clear(); return; } // Mark the dateless as examined. Without this they stay at state 1 and the // sweep loops over them on every pass, never terminating. let ids: Vec = found .iter() .filter(|m| m.captured_at.is_none()) .map(|m| m.image_id) .collect(); for id in ids { let _ = catalog .connection() .execute("UPDATE images SET metadata_state = 2 WHERE id = ?1", [id]); } found.clear(); } /// TRACES: FR-CAT-3 | FR-NC-3 | NFR-RES-4 /// TRACES: FR-CULL-8 /// Every visible image this model has not been run over. /// /// **No thumbnail-store filter.** The pass this feeds fetches its own pixels, /// so an image with no proxy is work to be done rather than work to be skipped /// — which is the whole difference between indexing a library and indexing the /// fraction of it that has been browsed. fn faces_unindexed( catalog: &Catalog, model_id: &str, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT i.id, i.source_ref, r.file_id, i.file_size FROM images i JOIN remote r ON r.image_id = i.id WHERE r.file_id IS NOT NULL AND {VISIBLE} AND NOT EXISTS ( SELECT 1 FROM face_index fi WHERE fi.image_id = i.id AND fi.model_id = ?1 ) ORDER BY i.id" ))?; let rows = stmt .query_map([model_id], |r| { Ok(ThumbnailRequest { // Face indexing wants the detail a thumbnail discards. full_resolution: true, // Named because the field must say something; ignored, because // `full_resolution` overrides it. thumb_size: dr_thumbs::ThumbSize::Large, // No grid cell waits on this. row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id: r.get::<_, Option>(2)?.map(|v| v as u64), size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, // The thumbnail sweep owns dating. Reading EXIF here would // write the same rows from a second pass for no gain. needs_metadata: false, }) })? .collect::, _>>()?; Ok(rows) } /// One image the face sweep got through, on its way back from a fetch lane. /// /// The catalog id, the faces found, the long edge they were normalised /// against, and the proxy to keep — `None` where the image held no face, since /// nothing will ever ask to crop one out of it. type IndexedImage = ( i64, Vec, u32, Option<(u64, dr_thumbs::Thumbnail)>, ); /// Images whose faces have nothing left to be cut out of. /// /// A face is stored normalised and drawn by cropping the proxy it was found on /// (`identity::decode_proxy`). Where that proxy is gone the People screen draws /// "no preview" for every cell and cannot repair itself, because the image /// already has its `face_index` row and so is not outstanding work. /// /// Two ways in: an indexing pass that fetched a preview and did not keep it, /// and an ordinary cache eviction. Both are the same state, and re-running /// detection over the image fixes it — `record_detections` replaces rather than /// appends, and carries the user's confirmations across the replacement, so /// this costs a fetch and loses nothing. fn faces_without_proxy( catalog: &Catalog, store: &ThumbStore, model_id: &str, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT DISTINCT i.id, i.source_ref, r.file_id, i.file_size FROM images i JOIN remote r ON r.image_id = i.id JOIN faces f ON f.image_id = i.id WHERE r.file_id IS NOT NULL AND {VISIBLE} AND f.model_id = ?1 ORDER BY i.id" ))?; let rows = stmt .query_map([model_id], |r| { Ok(ThumbnailRequest { full_resolution: true, thumb_size: dr_thumbs::ThumbSize::Large, row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id: r.get::<_, Option>(2)?.map(|v| v as u64), size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, needs_metadata: false, }) })? .filter_map(Result::ok) .filter(|req| { req.file_id .is_some_and(|id| !store.contains(id, dr_thumbs::ThumbSize::Large)) }) .collect(); Ok(rows) } /// TRACES: FR-CULL-8 | NFR-ARCH-2 /// Index faces across the **whole** library, fetching what it needs. /// /// # Why this replaced a pass that read the thumbnail store /// /// The previous sweep filtered its work list down to images that already had a /// `ThumbSize::Large` proxy on disk, on the reading that FR-CULL-8 keeps face /// indexing off the network. It does not: it keeps indexing off the *full /// decode*, and says plainly that "where no proxy exists, the job requests one /// at background priority". Nothing filled the large class for a whole library /// — `SWEEP_THUMB_SIZE` is deliberately `Grid` — so the pass could only ever /// reach photographs the user had personally zoomed into. Measured on the /// reference library: 220 images indexed out of 23,529. /// /// So this fetches, by exactly the two-stage route the thumbnail sweep uses: /// the header, then the located preview's own byte range (FR-NC-3). No full /// file is pulled and no RAW is decoded — an embedded preview is a JPEG. /// /// # Why it does not simply ride the thumbnail sweep /// /// It could, and it would be free: that pass already fetches the largest /// preview, decodes it, and downscales it to 256. Riding it is the right shape /// for *new* images and is the obvious next step. It cannot serve this /// operation, though, because the thumbnail sweep's work list is what the store /// does not have — so every image already thumbnailed, which for an established /// library is most of them, would never come back past the detector. /// /// # Cost, stated plainly /// /// One preview fetch per un-indexed image, capped at [`MAX_PREVIEW_BYTES`]. /// That is the same transfer the thumbnail sweep pays per image, paid a second /// time because the first one kept only 256 px. Resumable by construction: the /// work list is what the catalog has no `face_index` row for, so a kill costs /// the images in flight and nothing else. #[allow(clippy::too_many_arguments)] pub fn spawn_face_sweep( conn: Connection, catalog_path: PathBuf, store_dir: PathBuf, detector_model: PathBuf, embedder_model: PathBuf, model_id: String, options: dr_face::DetectOptions, ) -> Receiver { use crate::faces::FaceSweepMessage; let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let finish_empty = |tx: &Sender| { let _ = tx.send(FaceSweepMessage::Finished { images: 0, faces: 0, failed: 0, }); }; let catalog = match Catalog::open(&catalog_path) { Ok(c) => c, Err(e) => { log::warn!("face sweep: cannot open catalog: {e}"); finish_empty(&tx); return; } }; // The proxy an indexed face is later *cut from*. // // `identity::decode_proxy` reads `FACE_TIER` out of this store to draw // the People screen, so a face found on a preview that was fetched and // dropped has nowhere to come from and the grid shows "no preview" for // every cell. Fetching the pixels and not keeping them was the whole // bug: they cost a round trip and a decode, and the crop needs them // again the moment the user looks at the person. let mut store = match ThumbStore::open(&store_dir) { Ok(s) => s, Err(e) => { log::warn!("face sweep: cannot open the thumbnail store: {e}"); finish_empty(&tx); return; } }; // Models before the work list: they are the expensive failure, and // listing twenty thousand images before discovering the weights are // missing helps nobody. A library with no model installed takes this // path, so it is a quiet return rather than an error. let mut detector = match dr_face::Detector::from_path(&detector_model) { Ok(d) => d, Err(e) => { log::warn!("face sweep: cannot load the detector: {e}"); finish_empty(&tx); return; } }; let mut embedder = match dr_face::Embedder::from_path( &embedder_model, dr_face::ModelId::new(model_id.clone()), ) { Ok(e) => e, Err(e) => { log::warn!("face sweep: cannot load the embedder: {e}"); finish_empty(&tx); return; } }; // **Repairs first, and the order is the whole point.** These are the // images the People screen is drawing *right now* and failing to, and // there are a few hundred of them against tens of thousands of // un-indexed ones. Appended instead, they sit two hours of fetching // down the queue and the screen stays empty for the whole session — // which is indistinguishable from the repair not existing. let mut wanted = match faces_without_proxy(&catalog, &store, &model_id) { Ok(repair) => { if !repair.is_empty() { log::info!( "face sweep: repairing {} image(s) whose faces have no proxy to crop from", repair.len() ); } repair } Err(e) => { log::warn!("face sweep: looking for orphaned faces: {e}"); Vec::new() } }; // Disjoint from the above by construction: an image with faces recorded // is not an image with no `face_index` row. match faces_unindexed(&catalog, &model_id) { Ok(fresh) => wanted.extend(fresh), Err(e) => { log::warn!("face sweep: {e}"); if wanted.is_empty() { finish_empty(&tx); return; } } } let total = wanted.len(); if total == 0 { log::info!("face sweep: every image has been through this model"); finish_empty(&tx); return; } log::info!("face sweep: {total} image(s) to index"); if tx.send(FaceSweepMessage::Total(total)).is_err() { return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { log::warn!("face sweep: no runtime: {e}"); finish_empty(&tx); return; } }; rt.block_on(async { // Through `remote::connect`, which is the only place in the // interface that knows whose backend this is. let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { log::warn!("face sweep: {e}"); finish_empty(&tx); return; } }; // One detector and one embedder for every lane. // // The lanes are concurrent futures on a single thread, not threads, // so they interleave only at await points — and inference contains // none. A `RefCell` borrow therefore never overlaps another, and // the alternative, a pair per lane, would be ~16 MB of weights // duplicated for no parallelism at all. let models = std::cell::RefCell::new((&mut detector, &mut embedder)); let (mut done, mut images, mut found, mut failed) = (0usize, 0usize, 0usize, 0usize); let mut offline = false; // TRACES: FR-NC-6c // The same borrow the thumbnail sweep makes, and deliberately its // own pool: the two passes run at different times, so sharing one // would keep every file the earlier pass touched hydrated until // the later one finished. Each gives its own back (ARCH §9.0a). let pool = dr_sync_folder::BorrowPool::new(); for chunk in wanted.chunks(SWEEP_CHUNK) { let lanes: Vec> = (0..SWEEP_LANES) .map(|lane| chunk.iter().skip(lane).step_by(SWEEP_LANES).collect()) .collect(); let results = futures_join_all(lanes.into_iter().map(|lane| { let backend = &*backend; let models = ⊧ let options = &options; let pool = &pool; async move { let mut indexed: Vec = Vec::new(); let mut discard = Vec::new(); let mut attempted = 0usize; let mut failed = 0usize; let mut offline = false; for req in lane { attempted += 1; let _held = match pool.borrow(backend, &RemotePath::new(&req.path)).await { Ok(h) => h, Err(e) if e.indicates_offline() => { log::info!("face sweep: {e}"); attempted -= 1; offline = true; break; } Err(e) => { log::debug!("face sweep: {}: {e}", req.path); failed += 1; continue; } }; match fetch_preview(backend, req, &mut discard).await { PreviewOutcome::Ready(mut preview) => { // No await inside this borrow — see the // note where `models` is built. let found = { let mut m = models.borrow_mut(); let (det, emb) = &mut *m; crate::faces::index_preview(det, emb, &preview, options) }; match found { Ok((faces, edge)) => { // Keep the proxy only where there // is a face to cut out of it. Two // thirds of a personal library is // landscapes and documents // (docs/faces.md §7a), and those // never need a crop — so this fills // the large class for the images // the People screen will actually // ask about and leaves the rest // alone, rather than paying the // whole-library cost // `SWEEP_THUMB_SIZE` avoids. // // Downscaled only now: detection // needed the full buffer, and this // is the last use of it. let keep = match (faces.is_empty(), req.file_id) { (false, Some(file_id)) => { preview.downscale_to( dr_thumbs::ThumbSize::Large.edge(), ); encode_preview(file_id, &preview) .map(|t| (file_id, t)) } _ => None, }; indexed.push((req.image_id, faces, edge, keep)); } Err(e) => { log::debug!("face sweep: {}: {e}", req.path); failed += 1; } } } PreviewOutcome::Unavailable(reason) => { log::debug!("face sweep: {}: {reason}", req.path); failed += 1; } PreviewOutcome::Offline(reason) => { log::info!("face sweep: server unreachable: {reason}"); attempted -= 1; offline = true; break; } } } (indexed, attempted, failed, offline) } })) .await; for (indexed, attempted, lane_failed, lane_offline) in results { done += attempted; failed += lane_failed; offline |= lane_offline; for (image_id, faces, edge, keep) in indexed { // Before the detections, so a kill between the two // leaves a proxy with no faces recorded — which the // next pass simply re-indexes — rather than faces with // no proxy, which is the state that draws an empty // grid and cannot repair itself. if let Some((file_id, thumb)) = keep { store_thumbnail( &mut store, file_id, dr_thumbs::ThumbSize::Large, &thumb, ); } // Written per image, including the ones with no face in // them: `face_index` records that detection *ran*, and // zero is its most valuable value — without the row, // every landscape and document scan returns on the next // pass, for ever (docs/faces.md §7a). match dr_catalog::faces::record_detections( catalog.connection(), dr_types::ImageId(image_id as u64), &model_id, edge, &faces, ) { Ok(_) => { images += 1; found += faces.len(); if tx .send(FaceSweepMessage::Indexed { image: dr_types::ImageId(image_id as u64), faces: faces.len(), }) .is_err() { // Receiver dropped: the screen closed, or // the user pressed Stop. Everything written // so far stays written — and everything // borrowed is given back. A cancelled pass // that kept the library hydrated would be // the worst of both: the disk spent and // the work abandoned. log::info!("face sweep: cancelled after {images} image(s)"); pool.release_all(&*backend).await; return; } } Err(e) => { log::warn!("face sweep: storing faces for {image_id}: {e}"); failed += 1; } } } } let _ = done; if offline { break; } } let returned = pool.release_all(&*backend).await; if returned.released > 0 { log::info!( "face sweep: released {} borrowed file(s)", returned.released ); } log::info!( "face sweep: {found} face(s) across {images} image(s), {failed} failed{}", if offline { ", server went away" } else { "" } ); let _ = tx.send(FaceSweepMessage::Finished { images, faces: found, failed, }); }); }); rx } /// The class the whole-library pass fills. /// /// Grid only, deliberately. The large class is four times the transfer for a /// detail only a zoomed cell or the loupe asks for — on the reference library /// that is ~200 MB of shards against ~860 MB, paid by *every* device that /// syncs them (see [`dr_thumbs::ThumbSize`]). A photograph actually looked at /// closely still gets its large thumbnail from the interactive path. const SWEEP_THUMB_SIZE: dr_thumbs::ThumbSize = dr_thumbs::ThumbSize::Grid; /// Progress from the whole-library thumbnail pass. #[derive(Debug)] pub enum ThumbSweepMessage { /// How many images still lack a thumbnail, counted once at the start. Total(usize), /// Another chunk finished. Carries cumulative counts. Progress { done: usize, stored: usize }, Finished { stored: usize, failed: usize, /// Stopped early because the server stopped answering. The pass is /// resumable, so this is "come back later", not a failure. offline: bool, }, } /// TRACES: FR-CAT-3 | FR-NC-3 | FR-NC-7 /// Thumbnail **every** image in the library, not just the ones browsed. /// /// # Why this exists next to the grid's own fetching /// /// The interactive path fills cells as they are scrolled past, which is the /// right posture for a remote library (FR-NC-3) and the wrong one for handing /// the result to a second device: a tablet that syncs the shards inherits only /// the fraction of the library its sibling happened to look at. This is the /// deliberate, user-launched version of the same work — an hour of range /// fetches paid once, on the machine that can afford it, so every other client /// gets a full grid for the cost of a few hundred MB (`derived_sync`). /// /// # Shape, and why it borrows the metadata sweep's /// /// Chunked and lane-parallel exactly as [`spawn_sweep`] is, for the same /// reason: each image is ~0.6 s of round-trip latency and almost no /// bandwidth, so the sequential version spends its life waiting. What differs /// is the store — it is `&mut` and cannot be shared across lanes, which is why /// the metadata sweep skips thumbnails entirely. Here the lanes fetch, decode /// and *encode*, and only the ~20 KB result crosses back to this thread, which /// owns the store and writes the chunk in one go. So the parallelism is real /// and the single-writer rule is never bent. /// /// Dates arrive free: the header a preview needs is the header EXIF lives in, /// so an image this pass reaches is dated on the same fetch rather than /// costing a second one. /// /// Resumable by construction — the work list is what the store does not have, /// so a kill costs the chunk in flight and nothing more. An image with no /// locatable preview is retried on a later run; it is one header fetch, and /// the alternative is a second piece of state that has to be invalidated when /// a file is replaced. pub fn spawn_thumbnail_sweep( conn: Connection, catalog_path: PathBuf, store_dir: PathBuf, ) -> Receiver { let (tx, rx) = std::sync::mpsc::channel(); std::thread::spawn(move || { let finish_empty = |tx: &Sender| { let _ = tx.send(ThumbSweepMessage::Finished { stored: 0, failed: 0, offline: false, }); }; let catalog = match Catalog::open(&catalog_path) { Ok(c) => c, Err(e) => { log::warn!( "thumbnail sweep: cannot open catalog at {}: {e}", catalog_path.display() ); finish_empty(&tx); return; } }; // Unlike the grid's fetch, which carries on without a store and simply // shows what it downloaded, a store that will not open ends this: the // pass exists to fill it, and running an hour of transfers with // nowhere to put them would be worse than not starting. let mut store = match ThumbStore::open(&store_dir) { Ok(s) => s, Err(e) => { log::warn!( "thumbnail sweep: cannot open the thumbnail store at {}: {e}", store_dir.display() ); finish_empty(&tx); return; } }; let wanted = match thumbnails_outstanding(&catalog, &store) { Ok(w) => w, Err(e) => { log::warn!("thumbnail sweep: {e}"); finish_empty(&tx); return; } }; let total = wanted.len(); if total == 0 { log::info!("thumbnail sweep: every image already has a thumbnail"); finish_empty(&tx); return; } log::info!("thumbnail sweep: {total} image(s) need a thumbnail"); if tx.send(ThumbSweepMessage::Total(total)).is_err() { return; } let rt = match crate::net_runtime::build() { Ok(rt) => rt, Err(e) => { log::warn!("thumbnail sweep: no runtime: {e}"); finish_empty(&tx); return; } }; rt.block_on(async { let backend = match crate::remote::connect(&conn) { Ok(b) => b, Err(e) => { log::warn!("thumbnail sweep: {e}"); finish_empty(&tx); return; } }; let (mut done, mut stored, mut failed) = (0usize, 0usize, 0usize); let mut offline = false; let mut found = Vec::new(); // TRACES: FR-NC-6c // On a placeholder library the bytes may not be here at all, and // this is a pass the user asked for — so it may fetch them, which // browsing may not (ARCH §9.0a). Every file is *borrowed*: what // this pass downloads it gives back, and what the user already had // it leaves alone. Against a server or a plain folder every borrow // is a no-op, so there is one code path rather than two. let pool = dr_sync_folder::BorrowPool::new(); for chunk in wanted.chunks(SWEEP_CHUNK) { // Each lane owns a disjoint slice and its own output, so // nothing is shared and no lock is needed. The store is not // touched here — see the note on the function. let lanes: Vec> = (0..SWEEP_LANES) .map(|lane| chunk.iter().skip(lane).step_by(SWEEP_LANES).collect()) .collect(); let results = futures_join_all(lanes.into_iter().map(|lane| { let backend: &dyn RemoteBackend = &*backend; let pool = &pool; async move { let mut made: Vec<(u64, dr_thumbs::Thumbnail)> = Vec::new(); let mut found = Vec::new(); let mut attempted = 0usize; let mut failed = 0usize; let mut offline = false; for req in lane { // Enforced by the query, which joins `remote`: an // image with no file id has nothing to key the // store on and is not a candidate. let Some(file_id) = req.file_id else { continue }; attempted += 1; // Held for this image only. A failure to fetch is // this image's verdict, not the batch's: a client // that cannot reach the server reports it as // offline through the usual path below. let _held = match pool.borrow(backend, &RemotePath::new(&req.path)).await { Ok(h) => h, Err(e) if e.indicates_offline() => { log::info!("thumbnail sweep: {e}"); attempted -= 1; offline = true; break; } Err(e) => { log::debug!("thumbnail sweep: {}: {e}", req.path); failed += 1; continue; } }; match fetch_preview(backend, req, &mut found).await { PreviewOutcome::Ready(preview) => { match encode_preview(file_id, &preview) { Some(thumb) => made.push((file_id, thumb)), None => failed += 1, } } PreviewOutcome::Unavailable(reason) => { log::debug!("thumbnail sweep: {}: {reason}", req.path); failed += 1; } // Nothing after this would reach the server // either, so the lane stops rather than // spending a timeout per remaining image. PreviewOutcome::Offline(reason) => { log::info!("thumbnail sweep: server unreachable: {reason}"); attempted -= 1; offline = true; break; } } } (made, found, attempted, failed, offline) } })) .await; for (made, lane_found, attempted, lane_failed, lane_offline) in results { done += attempted; failed += lane_failed; offline |= lane_offline; found.extend(lane_found); for (file_id, thumb) in made { if store_thumbnail(&mut store, file_id, SWEEP_THUMB_SIZE, &thumb) { stored += 1; } else { failed += 1; } } } // Committed per chunk rather than at the end, so a kill keeps // every date read so far — the same bargain the metadata sweep // makes, and for the same reason. flush_sweep(&catalog, &mut found); if tx .send(ThumbSweepMessage::Progress { done, stored }) .is_err() { // Cancelled. Hand back what was borrowed before leaving, // or a stopped pass costs the disk of everything it had // reached and delivers nothing for it. pool.release_all(&*backend).await; return; } if offline { break; } } flush_sweep(&catalog, &mut found); // Give back everything this pass fetched, before reporting done — // a user watching the disk should see it return, and a pass that // reported success while still holding the library would be // lying about what it cost. let returned = pool.release_all(&*backend).await; if returned.released > 0 { log::info!( "thumbnail sweep: released {} borrowed file(s)", returned.released ); } log::info!("thumbnail sweep: {stored} stored, {failed} without a usable preview"); let _ = tx.send(ThumbSweepMessage::Finished { stored, failed, offline, }); }); }); rx } /// Every visible image on the server that the store has no grid thumbnail for. /// /// Joined against `remote` rather than left-joined: the store is keyed on /// Nextcloud's `oc:fileid` (FR-NC-5), so an image the scan recorded without /// one cannot be stored and is not work this pass can do. /// /// The whole list is built up front rather than re-queried per chunk, unlike /// the metadata sweep: "does the store have this" is answered by the store's /// index, which this thread is also the one writing, so a stale list is not a /// risk the way a concurrently-dating grid made it one there. fn thumbnails_outstanding( catalog: &Catalog, store: &ThumbStore, ) -> Result, dr_catalog::CatalogError> { let mut stmt = catalog.connection().prepare(&format!( "SELECT i.id, i.source_ref, r.file_id, i.file_size, i.metadata_state FROM images i JOIN remote r ON r.image_id = i.id WHERE r.file_id IS NOT NULL AND {VISIBLE} ORDER BY i.id" ))?; let rows = stmt .query_map([], |r| { let file_id = r.get::<_, Option>(2)?.map(|v| v as u64); Ok(ThumbnailRequest { thumb_size: SWEEP_THUMB_SIZE, // No grid cell is waiting on this, so nothing consumes the row. row: 0, image_id: r.get(0)?, path: r.get(1)?, file_id, size: r.get::<_, Option>(3)?.unwrap_or(0) as u64, // The header this fetch reads is the one EXIF lives in, so an // undated image is dated on the way past for nothing. needs_metadata: r.get::<_, i64>(4)? < 2, full_resolution: false, }) })? .filter_map(Result::ok) .filter(|req| { req.file_id .is_some_and(|id| !store.contains(id, SWEEP_THUMB_SIZE)) }) .collect(); Ok(rows) } /// Where an account's thumbnail shards live. /// /// Beside the catalog rather than in the cache directory: these sync to the /// server and are shared with other clients, so discarding them on a cache /// sweep would cost a re-download for everyone. pub fn thumbs_dir(account: &Account) -> PathBuf { catalog_path(account) .parent() .map(|p| p.join("thumbs")) .unwrap_or_else(|| std::env::temp_dir().join("darkroom-thumbs")) } /// Where the face models live, beside the catalog and the thumbnails. /// /// **Not shipped with the application** and not a build input: the InsightFace /// weights carry a non-commercial research grant incompatible with this /// project's licence, so the user obtains them and the app loads them from here /// (docs/faces.md §2). An absent directory is the ordinary state of a fresh /// install, not an error. pub fn face_models_dir(account: &Account) -> PathBuf { catalog_path(account) .parent() .map(|p| p.join("models")) .unwrap_or_else(|| std::env::temp_dir().join("darkroom-models")) } /// Where face models live for *every* account on this device. /// /// Account-independent, unlike the catalog: a model is identified by /// `faces.model_id` (catalog.md §10.1) and not by who is signed in, so two /// accounts have no reason to hold two 15 MB copies of the same weights. This /// is also the only directory an Android build can populate for itself — the /// entry point extracts the APK's bundled copy here before any store opens, /// and at that moment no session exists to key a per-account path off. pub fn shared_face_models_dir() -> PathBuf { data_root().join("models") } /// The detector and embedder files, if both are present. /// /// Both or neither: an embedder with no detector has nothing to embed, and a /// detector with no embedder finds faces it cannot tell apart. Reporting the /// pair missing is more useful than half-starting. /// /// The names are the **shape-fixed** exports, not what InsightFace ships: /// `tools/fix-face-model-shapes.sh` has to run over the originals first, /// because tract cannot parse either graph with a dynamic input. /// /// Searched in three places, most specific first: /// /// 1. **The account's own directory.** A library can be pinned to its own /// weights — a model swap is a `model_id` change and a re-index, and someone /// mid-migration needs one account's pair to stay put without holding the /// other back. /// 2. **The shared user directory.** Where a user drops a pair by hand, and /// where the Android entry point unpacks the copy the APK carries. /// 3. **The system directories.** Where a package installs them — the Arch /// package puts the pair in `/usr/share/darkroom/models`. Last, so anything /// the user placed themselves outranks what the package shipped. pub fn face_models(account: &Account) -> Option<(PathBuf, PathBuf)> { fn pair(dir: PathBuf) -> Option<(PathBuf, PathBuf)> { let detector = dir.join("scrfd_500m_640.onnx"); let embedder = dir.join("arcface_mbf_b1.onnx"); (detector.is_file() && embedder.is_file()).then_some((detector, embedder)) } pair(face_models_dir(account)) .or_else(|| pair(shared_face_models_dir())) .or_else(|| system_face_models_dirs().into_iter().find_map(pair)) } /// Where a *package* may have installed the models. /// /// `$XDG_DATA_DIRS` rather than a hard-coded `/usr/share`, because that is the /// variable a distribution, a prefix install, or a Nix-style store sets to say /// where its data went, and the default it falls back to is exactly the pair of /// paths that would otherwise have been hard-coded. /// /// Empty on Android, which has no such directories: there the APK's copy is /// unpacked into the shared user directory instead, because an asset inside a /// package is not a path anything can read from (ARCH §6.9). fn system_face_models_dirs() -> Vec { if cfg!(target_os = "android") { return Vec::new(); } let dirs = std::env::var("XDG_DATA_DIRS") .ok() .filter(|v| !v.is_empty()) .unwrap_or_else(|| "/usr/local/share:/usr/share".into()); dirs.split(':') .filter(|d| !d.is_empty()) .map(|d| PathBuf::from(d).join("darkroom").join("models")) .collect() } /// One grid cell's data, read from the catalog. #[derive(Debug, Clone, PartialEq, Eq)] pub struct LibraryCell { pub image_id: i64, pub name: String, pub remote_path: String, /// `oc:fileid`, the key the shared thumbnail store uses. `None` for an /// image the scan found without a stable id. pub file_id: Option, /// File length, for bounds-checking a located preview range. pub size: u64, /// 0 = nothing, 1 = stat-only, 2 = full EXIF. pub metadata_state: u8, /// UTC seconds, once EXIF has been read. pub captured_at: Option, } /// Read a window of cells out of the catalog. /// /// Windowed rather than wholesale: a 17k-image library must not become 17k /// rows in a Slint model (FR-CAT-4). /// The unscoped form, kept as the name the tests and any future caller reach /// for. The UI goes through [`read_cells_scoped`], because a collection may be /// selected. #[cfg(test)] pub fn read_cells( catalog: &Catalog, offset: usize, limit: usize, ) -> Result, dr_catalog::CatalogError> { read_cells_scoped(catalog, None, &RatingFilter::default(), offset, limit) } /// Read a window of cells, optionally narrowed to one collection. /// /// A collection *set* shows its descendants' images too — a parent whose /// children hold everything would otherwise read as empty, which makes nesting /// look broken. The id list comes from /// [`dr_catalog::collections::descendants`], which is depth-guarded. /// /// Ordering matches the unscoped grid (capture time, then name) rather than /// manual position: position is only meaningful inside one collection and this /// query also serves sets, where two children's positions are unrelated. pub fn read_cells_scoped( catalog: &Catalog, scope: Option, filter: &RatingFilter, offset: usize, limit: usize, ) -> Result, dr_catalog::CatalogError> { let Some(scope) = scope else { return read_cells_all(catalog, filter, offset, limit); }; let ids = dr_catalog::collections::descendants(catalog.connection(), scope)?; // Placeholders are generated from the *count* of ids, never from user text. let placeholders = std::iter::repeat_n("?", ids.len()) .collect::>() .join(","); let rated = filter.sql(); let sql = format!( "SELECT {CELL_COLUMNS} FROM images i WHERE {VISIBLE}{rated} AND i.id IN (SELECT image_id FROM collection_members WHERE collection_id IN ({placeholders})) {GRID_ORDER} LIMIT ? OFFSET ?" ); let mut params: Vec = ids .iter() .map(|c| rusqlite::types::Value::Integer(c.0 as i64)) .collect(); params.push(rusqlite::types::Value::Integer(limit as i64)); params.push(rusqlite::types::Value::Integer(offset as i64)); let mut rows = { let mut stmt = catalog.connection().prepare(&sql)?; let read = stmt .query_map(rusqlite::params_from_iter(params.iter()), row_to_cell)? .collect::, _>>()?; read }; attach_file_ids(catalog, &mut rows); Ok(rows) } fn read_cells_all( catalog: &Catalog, filter: &RatingFilter, offset: usize, limit: usize, ) -> Result, dr_catalog::CatalogError> { let rated = filter.sql(); let mut rows = { let mut stmt = catalog.connection().prepare(&format!( "SELECT {CELL_COLUMNS} FROM images i WHERE {VISIBLE}{rated} {GRID_ORDER} LIMIT ?1 OFFSET ?2" ))?; let read = stmt .query_map(rusqlite::params![limit as i64, offset as i64], row_to_cell)? .collect::, _>>()?; read }; attach_file_ids(catalog, &mut rows); Ok(rows) } /// TRACES: FR-CAT-15 /// Read a window of *trashed* cells, newest deletion first. /// /// Ordered by when it was trashed rather than by capture time, which is what /// every other view sorts by. The question in the trash is "what did I just /// delete?", not "when was this taken" — a mistaken delete is corrected within /// seconds, and burying it among photographs from the same afternoon would make /// the one row the user is looking for the hardest one to find. /// /// The rating filter is deliberately not applied. It narrows a *culling* pass, /// and a trash that hid rows because of a filter set elsewhere would look like /// it had lost them. pub fn read_trashed_cells( catalog: &Catalog, offset: usize, limit: usize, ) -> Result, dr_catalog::CatalogError> { let mut rows = { let mut stmt = catalog.connection().prepare(&format!( "SELECT {CELL_COLUMNS} FROM images i WHERE {TRASHED} {TRASH_ORDER} LIMIT ?1 OFFSET ?2" ))?; let read = stmt .query_map(rusqlite::params![limit as i64, offset as i64], row_to_cell)? .collect::, _>>()?; read }; attach_file_ids(catalog, &mut rows); Ok(rows) } /// TRACES: FR-CAT-15 /// How many images the trash view would list. /// /// Counts exactly what [`read_trashed_cells`] lists — same predicate, no filter /// — so the scrollbar and the header cannot disagree with the cells. pub fn total_trashed(catalog: &Catalog) -> Result { let n: i64 = catalog.connection().query_row( &format!("SELECT count(*) FROM images i WHERE {TRASHED}"), [], |r| r.get(0), )?; Ok(n as usize) } /// TRACES: FR-CAT-5 /// The ids of grid rows `first..=last`, in the order the grid lists them. /// /// What a shift-click actually means. The gesture names two *ordinals* and the /// photographs between them are mostly not loaded — the grid is a window of a /// hundred or so over a library of twenty thousand — so a range answered from /// the window selected the handful on screen and silently dropped the rest. /// The catalog knows the whole run, and with the ordering indexed it is one /// seek rather than a scan. /// /// Bounded by the same predicates, the same filter and the same [`GRID_ORDER`] /// the window itself is read with. An ordinal only names a photograph relative /// to an ordering, so a range taken through any other one is a range through a /// different library. /// /// `trash` picks the trash view's list, which is the other thing the grid can /// be showing and is ordered by deletion time rather than capture time. /// /// An empty result means the run is empty or the query failed; callers treat /// the two the same, because both leave the selection where it was. pub fn read_ids_span( catalog: &Catalog, scope: Option, filter: &RatingFilter, trash: bool, first: usize, last: usize, ) -> Result, dr_catalog::CatalogError> { let Some(count) = (last + 1).checked_sub(first) else { return Ok(Vec::new()); }; let (sql, mut params) = if trash { ( format!("SELECT i.id FROM images i WHERE {TRASHED} {TRASH_ORDER} LIMIT ? OFFSET ?"), Vec::new(), ) } else { let (clause, params) = scope_clause(catalog, scope)?; let rated = filter.sql(); ( format!( "SELECT i.id FROM images i WHERE {VISIBLE}{rated}{clause} {GRID_ORDER} LIMIT ? OFFSET ?" ), params, ) }; params.push(rusqlite::types::Value::Integer(count as i64)); params.push(rusqlite::types::Value::Integer(first as i64)); let mut stmt = catalog.connection().prepare(&sql)?; let ids = stmt .query_map(rusqlite::params_from_iter(params.iter()), |r| { Ok(dr_types::ImageId(r.get::<_, i64>(0)? as u64)) })? .collect::, _>>()?; Ok(ids) } /// The columns every windowed read selects, in the order [`row_to_cell`] reads /// them. /// /// Named rather than repeated so the three readers cannot drift — and so that /// the one column that is *not* here stays conspicuous. See /// [`attach_file_ids`] for why the server's file id is fetched separately. const CELL_COLUMNS: &str = "i.id, i.source_ref, i.file_size, i.metadata_state, i.captured_at"; /// Shared row mapping, so the scoped and unscoped queries cannot drift. /// /// `file_id` is left empty here and filled by [`attach_file_ids`]. fn row_to_cell(r: &rusqlite::Row) -> rusqlite::Result { let path: String = r.get(1)?; Ok(LibraryCell { image_id: r.get(0)?, name: path.rsplit(['/', ':']).next().unwrap_or(&path).to_string(), remote_path: path, file_id: None, size: r.get::<_, Option>(2)?.unwrap_or(0) as u64, metadata_state: r.get::<_, i64>(3)? as u8, captured_at: r.get(4)?, }) } /// TRACES: NFR-P5 /// Fill in each cell's server file id, in one query for the whole window. /// /// # Why this is not a `LEFT JOIN` any more /// /// It was, and it was the single most expensive thing the grid did while a /// finger was on it. A window is `ORDER BY ... LIMIT n OFFSET k`, and SQLite /// answers a join like that by joining *first* and paging after — so reading /// 280 cells at offset 20,000 meant an index seek into `remote` for all 24,000 /// rows, 23,720 of which were then discarded. Measured at 15.2 ms, inside the /// scroll handler, against a 16.7 ms frame. /// /// Paging over `images` alone is 0.36 ms with `images_grid_order` (schema V7), /// and this fetches the ids for the 280 rows that survived. The same shape the /// badge and rating reads already use: one query for the window, never one per /// cell. /// /// Silent on failure, and cells keep `file_id: None`: that is the same state a /// photograph the scan has not reached the server for is in, and the callers /// already treat it as "no cached thumbnail to key on" rather than an error. fn attach_file_ids(catalog: &Catalog, cells: &mut [LibraryCell]) { if cells.is_empty() { return; } let placeholders = std::iter::repeat_n("?", cells.len()) .collect::>() .join(","); let sql = format!("SELECT image_id, file_id FROM remote WHERE image_id IN ({placeholders})"); let params: Vec = cells .iter() .map(|c| rusqlite::types::Value::Integer(c.image_id)) .collect(); let mut stmt = match catalog.connection().prepare(&sql) { Ok(s) => s, Err(e) => { log::debug!("reading file ids for the window: {e}"); return; } }; let rows = stmt.query_map(rusqlite::params_from_iter(params.iter()), |r| { Ok((r.get::<_, i64>(0)?, r.get::<_, Option>(1)?)) }); let found: std::collections::HashMap> = match rows { Ok(rows) => rows.flatten().collect(), Err(e) => { log::debug!("reading file ids for the window: {e}"); return; } }; for cell in cells { cell.file_id = found .get(&cell.image_id) .copied() .flatten() .map(|v| v as u64); } } /// Total images in the catalog, or in one collection and its descendants. /// /// Counts exactly what [`read_cells_scoped`] would list, filter included. The /// two must agree: the header says "412 images" and the grid's scrollbar is /// sized from the same number, so a count that ignored the filter would leave /// the user scrolling through empty rows. pub fn total_images_scoped( catalog: &Catalog, scope: Option, filter: &RatingFilter, ) -> Result { let Some(scope) = scope else { return total_images_filtered(catalog, filter); }; let ids = dr_catalog::collections::descendants(catalog.connection(), scope)?; let placeholders = std::iter::repeat_n("?", ids.len()) .collect::>() .join(","); let rated = filter.sql(); // DISTINCT: an image in both a parent and a child is one photograph, and a // count that disagrees with the number of cells drawn is worse than either // number alone. // // Counted through `images` rather than over `collection_members` alone, so // `VISIBLE` applies — a trashed photograph is still a member row, and // counting it made the header claim images the grid would not draw. let sql = format!( "SELECT count(DISTINCT i.id) FROM images i WHERE {VISIBLE}{rated} AND i.id IN (SELECT image_id FROM collection_members WHERE collection_id IN ({placeholders}))" ); let params: Vec = ids .iter() .map(|c| rusqlite::types::Value::Integer(c.0 as i64)) .collect(); let n: i64 = catalog .connection() .query_row(&sql, rusqlite::params_from_iter(params.iter()), |r| { r.get(0) })?; Ok(n as usize) } /// The SQL restricting a query to `scope` and its descendants, with the bound /// parameters to go with it. /// /// Shared by the span and the histogram so the two cannot drift: an axis drawn /// over one set of images and bars counted over another puts the bars in the /// wrong place. fn scope_clause( catalog: &Catalog, scope: Option, ) -> Result<(String, Vec), dr_catalog::CatalogError> { let Some(scope) = scope else { return Ok((String::new(), Vec::new())); }; let ids = dr_catalog::collections::descendants(catalog.connection(), scope)?; let placeholders = std::iter::repeat_n("?", ids.len()) .collect::>() .join(","); Ok(( format!( " AND i.id IN (SELECT image_id FROM collection_members WHERE collection_id IN ({placeholders}))" ), ids.iter() .map(|c| rusqlite::types::Value::Integer(c.0 as i64)) .collect(), )) } /// Earliest and latest capture time within `scope`, honouring the filter. /// /// The timeline's extent. Taken over the same images the histogram counts, so /// opening a collection shows that collection's years rather than the whole /// library's — the axis was previously spanning everything, which left a /// collection's bars crushed into a sliver of it. pub fn span_scoped( catalog: &Catalog, scope: Option, filter: &RatingFilter, ) -> Option<(i64, i64)> { let (clause, params) = scope_clause(catalog, scope).ok()?; // Full extent, not the chosen range — see `without_date_range`. let rated = filter.without_date_range().sql(); let sql = format!( "SELECT min(i.captured_at), max(i.captured_at) FROM images i WHERE {VISIBLE}{rated} AND i.captured_at IS NOT NULL{clause}" ); catalog .connection() .query_row(&sql, rusqlite::params_from_iter(params.iter()), |r| { Ok((r.get::<_, Option>(0)?, r.get::<_, Option>(1)?)) }) .ok() .and_then(|(lo, hi)| Some((lo?, hi?))) } /// TRACES: FR-CAT-6 /// The capture-time histogram as a fixed number of equal bins across /// `from..=to`, empty ones included. /// /// # Why not calendar buckets /// /// [`timeline_scoped`] groups by year, month, day or hour, which has two /// consequences the axis cannot live with. /// /// It emits **only the buckets that hold photographs**, and the widget gives /// every bar an equal slot — so a library with a gap in it drew a February /// that was six months wide. The position marker, the range band and a click /// are all linear in time, so on a sparse library they pointed at bars that /// were somewhere else. Equal bins including the empty ones make a bar's /// position on the track and the date under it the same quantity. /// /// And the **count is free to jump by a factor of twelve** between one unit /// and the next, so each zoom step halved the number of bars until a /// threshold was crossed: zooming in made the picture coarser, twice out of /// every three steps. A fixed count re-bins on every zoom instead, which is /// what makes each step show finer structure rather than the same structure /// drawn wider. /// /// The date range is lifted from the filter, like the bars' other terms are /// not: this histogram is *how a range is chosen*, and drawing through the /// range would empty every bin outside it and leave nothing to widen into. /// /// `bins` is clamped to at least one — a zero would be a division by zero in /// SQL, and the caller's number comes from a hand-editable settings file. pub fn timeline_uniform( catalog: &Catalog, scope: Option, filter: &RatingFilter, from: i64, to: i64, bins: u32, ) -> Result, dr_catalog::CatalogError> { let bins = bins.max(1) as i64; // At least one second, or every photograph lands in bin zero. let span = (to - from).max(1); let (clause, params) = scope_clause(catalog, scope)?; let rated = filter.without_date_range().sql(); // The bin index is arithmetic on the stored UTC instant, not `strftime` on // a local one. A bin is not a calendar unit — it has no local midnight to // respect — and the axis it is drawn on is labelled from the same UTC // instants, so bucketing the two differently is the one way they could // disagree about which bar a photograph belongs to. // // `min` caps the last edge: an image captured at exactly `to` divides to // `bins`, which would be a bin past the end of the axis. // // Integers this code owns, formatted straight in — the same rule the // rating terms follow. They cannot be bound parameters here without // ordering them against the scope clause's, which appears later in the // text but is bound first. let sql = format!( "SELECT min({bins} - 1, (i.captured_at - {from}) * {bins} / {span}) AS b, count(*) AS n FROM images i WHERE {VISIBLE}{rated} AND i.captured_at IS NOT NULL{clause} AND i.captured_at >= {from} AND i.captured_at <= {to} GROUP BY b ORDER BY b ASC" ); let conn = catalog.connection(); let mut stmt = conn.prepare(&sql)?; let counted = stmt .query_map(rusqlite::params_from_iter(params.iter()), |r| { Ok((r.get::<_, i64>(0)?, r.get::<_, i64>(1)? as u32)) })? .collect::, _>>()?; // Every bin, in order, whether or not the query returned one for it. The // start is the bin's own left edge rather than the earliest photograph in // it: an empty bin has no photograph to take one from, and a bar drawn at // its contents' position rather than its bin's would put the axis back // where the calendar buckets left it. let mut bars: Vec = (0..bins) .map(|i| dr_catalog::TimeBucket { start: from + (i * span) / bins, count: 0, }) .collect(); for (i, n) in counted { if let Some(bar) = bars.get_mut(i.clamp(0, bins - 1) as usize) { bar.count = n; } } Ok(bars) } /// Total images in the catalog, honouring the rating filter. fn total_images_filtered( catalog: &Catalog, filter: &RatingFilter, ) -> Result { let rated = filter.sql(); let n: i64 = catalog.connection().query_row( &format!("SELECT count(*) FROM images i WHERE {VISIBLE}{rated}"), [], |r| r.get(0), )?; Ok(n as usize) } /// TRACES: FR-CAT-9 /// How many visible images have their original stored on this device. /// /// Whole-library, like the star counts beside it: the chip says what narrowing /// to it would show, so counting only the current window would make it /// describe the view it exists to change. pub fn local_original_count(catalog: &Catalog) -> Result { let n: i64 = catalog.connection().query_row( &format!( "SELECT count(*) FROM images i WHERE {VISIBLE} AND EXISTS (SELECT 1 FROM image_cache ic WHERE ic.image_id = i.id AND ic.tier_actual >= {})", dr_types::Tier::Original.stored() ), [], |r| r.get(0), )?; Ok(n as usize) } /// Total images in the catalog, unfiltered. /// /// What the scan reports and what the sidebar's "all images" row shows — the /// size of the library itself, not of the current view. pub fn total_images(catalog: &Catalog) -> Result { let n: i64 = catalog.connection().query_row( &format!("SELECT count(*) FROM images WHERE {VISIBLE_UNALIASED}"), [], |r| r.get(0), )?; Ok(n as usize) } pub fn now_secs() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map(|d| d.as_secs() as i64) .unwrap_or(0) } #[cfg(test)] mod tests { use super::*; use dr_sync::RemoteEntry; #[test] fn join_all_preserves_order_regardless_of_completion() { // The ordering guarantee is what lets a caller pair results back to // their inputs; without it a lane's dates could be attributed to the // wrong images. let rt = crate::net_runtime::build().unwrap(); let out = rt.block_on(async { futures_join_all(vec![ Box::pin(async { 1 }) as std::pin::Pin>>, Box::pin(async { tokio::task::yield_now().await; tokio::task::yield_now().await; 2 }), Box::pin(async { tokio::task::yield_now().await; 3 }), ]) .await }); assert_eq!(out, vec![1, 2, 3]); } #[test] fn join_all_of_nothing_completes() { let rt = crate::net_runtime::build().unwrap(); let out: Vec = rt.block_on(async { futures_join_all(Vec::>::new()).await }); assert!(out.is_empty()); } #[test] fn sweep_lanes_divide_a_chunk_without_loss() { // Every image in a chunk must land in exactly one lane: a striding // split that dropped or duplicated one would silently under- or // double-index the library. let chunk: Vec = (0..SWEEP_CHUNK).collect(); let lanes: Vec> = (0..SWEEP_LANES) .map(|l| chunk.iter().skip(l).step_by(SWEEP_LANES).copied().collect()) .collect(); let mut seen: Vec = lanes.iter().flatten().copied().collect(); seen.sort_unstable(); assert_eq!(seen, chunk); // Evenly divided, so no lane sits idle while another finishes. assert!(lanes.iter().all(|l| l.len() == SWEEP_CHUNK / SWEEP_LANES)); } #[test] fn a_short_chunk_still_covers_every_image() { // The last chunk of a library is rarely a full multiple of the lanes. let chunk: Vec = (0..5).collect(); let lanes: Vec> = (0..SWEEP_LANES) .map(|l| chunk.iter().skip(l).step_by(SWEEP_LANES).copied().collect()) .collect(); let mut seen: Vec = lanes.iter().flatten().copied().collect(); seen.sort_unstable(); assert_eq!(seen, chunk); } #[test] fn a_scrub_ordinal_matches_the_grid_position() { // The scrub's count and the grid's window must use *identical* // predicates and ordering, or the view lands somewhere else. Counting // only dated images against a grid that also shows undated ones put a // click near the end of the axis near the top of the library. let catalog = Catalog::in_memory().unwrap(); let c = catalog.connection(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); // A mix: dated, undated, and one shadowed by a RAW sibling. for (id, name, captured, shadow) in [ (1i64, "a.CR2", Some(100i64), None), (2, "b.CR2", Some(200), None), (3, "b.JPG", Some(200), Some(2i64)), (4, "c.CR2", Some(300), None), (5, "d.CR2", None, None), ] { c.execute( "INSERT INTO images(id, root_id, source_ref, captured_at, shadowed_by, added_at) VALUES (?1, 1, ?2, ?3, ?4, 0)", rusqlite::params![id, name, captured, shadow], ) .unwrap(); } // The grid's own window, in its own order. let cells = read_cells(&catalog, 0, 100).unwrap(); let names: Vec<&str> = cells.iter().map(|c| c.name.as_str()).collect(); assert_eq!( names, vec!["a.CR2", "b.CR2", "c.CR2", "d.CR2"], "shadowed hidden, undated last" ); // Scrubbing to each image's instant must give its index in that list. for (when, expected) in [(100i64, 0usize), (200, 1), (300, 2)] { let ordinal: i64 = c .query_row( "SELECT count(*) FROM images WHERE shadowed_by IS NULL AND captured_at IS NOT NULL AND captured_at < ?1", [when], |r| r.get(0), ) .unwrap(); assert_eq!( ordinal as usize, expected, "scrubbing to {when} must land on grid row {expected}" ); } } #[test] fn the_thumbnail_pass_asks_only_for_what_is_missing() { // The work list is the whole point of the pass being resumable and of // it being safe to press twice: it is derived from what the store // lacks, not from a flag in the catalog. Three things it must respect // — a thumbnail already stored, a trashed or shadowed image, and an // image with no `oc:fileid`, which the store cannot key on at all. let catalog = Catalog::in_memory().unwrap(); let c = catalog.connection(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); // (id, name, file_id, shadowed_by, trashed_at) for (id, name, file_id, shadow, trashed) in [ (1i64, "a.CR2", Some(11i64), None, None), (2, "b.CR2", Some(22), None, None), (3, "b.JPG", Some(33), Some(2i64), None), (4, "c.CR2", Some(44), None, Some(1000i64)), // Scanned without a file id: nothing to key the store on. (5, "d.CR2", None, None, None), ] { c.execute( "INSERT INTO images(id, root_id, source_ref, shadowed_by, trashed_at, added_at) VALUES (?1, 1, ?2, ?3, ?4, 0)", rusqlite::params![id, name, shadow, trashed], ) .unwrap(); if let Some(file_id) = file_id { c.execute( "INSERT INTO remote(image_id, file_id) VALUES (?1, ?2)", rusqlite::params![id, file_id], ) .unwrap(); } } let dir = std::env::temp_dir().join(format!("dr-thumb-sweep-{}", std::process::id())); let _ = std::fs::remove_dir_all(&dir); std::fs::create_dir_all(&dir).unwrap(); let mut store = ThumbStore::open(&dir).unwrap(); let all = thumbnails_outstanding(&catalog, &store).unwrap(); let names: Vec<&str> = all.iter().map(|r| r.path.as_str()).collect(); assert_eq!( names, vec!["a.CR2", "b.CR2"], "shadowed, trashed and file-id-less images are not work this pass can do" ); // Store one, and it drops out — this is what stops a second run // re-fetching an hour of previews. store .put( 11, SWEEP_THUMB_SIZE, &dr_thumbs::Thumbnail { width: 4, height: 4, bytes: vec![0xFF, 0xD8, 0xFF, 0xD9], }, ) .unwrap(); let rest = thumbnails_outstanding(&catalog, &store).unwrap(); let names: Vec<&str> = rest.iter().map(|r| r.path.as_str()).collect(); assert_eq!(names, vec!["b.CR2"]); // The large class is a different key, so filling the grid class does // not make the pass think the library is done at another size. assert!(!store.contains(11, dr_thumbs::ThumbSize::Large)); let _ = std::fs::remove_dir_all(&dir); } /// An account for the path tests, defaulting to the connector every /// existing install uses. fn account(endpoint: &str, user: &str) -> Account { Account::new("nextcloud", endpoint).with_login(user, user) } #[test] fn catalog_paths_separate_accounts() { // Two accounts on one machine must not share an index, or one // library's images appear in the other. let a = catalog_path(&account("https://cloud.example", "duncan")); let b = catalog_path(&account("https://cloud.example", "someone")); let c = catalog_path(&account("https://other.example", "duncan")); assert_ne!(a, b); assert_ne!(a, c); } #[test] fn a_folder_library_gets_its_own_catalog() { // The same rule across backends: a folder library on this machine // must not land in the directory a server account is already using. let server = catalog_path(&account("https://cloud.example", "duncan")); let folder = catalog_path(&Account::new("folder", "/mnt/photos")); assert_ne!(server, folder); assert_ne!( folder, catalog_path(&Account::new("folder", "/mnt/other-photos")) ); } #[test] fn a_legacy_cache_directory_is_moved_rather_than_abandoned() { // The upgrade hazard: `sidecars/` and `outbox/` hold work that exists // nowhere else, so leaving them behind in a directory the system may // empty would discard unsynced ratings and edits as a side effect of // installing a new build. let root = std::env::temp_dir().join(format!("dr-migrate-{}", std::process::id())); let _ = std::fs::remove_dir_all(&root); let legacy = root.join("darkroom").join("cloud-example-duncan"); std::fs::create_dir_all(legacy.join("sidecars")).unwrap(); std::fs::write(legacy.join("catalog.sqlite"), b"catalog").unwrap(); std::fs::write(legacy.join("sidecars").join("a.drsc"), b"an unsynced edit").unwrap(); let current = root.join("new").join("cloud-example-duncan"); move_account_dir(&legacy, ¤t); assert!(!legacy.exists(), "the old copy must not be left behind"); assert_eq!( std::fs::read(current.join("catalog.sqlite")).unwrap(), b"catalog" ); assert_eq!( std::fs::read(current.join("sidecars").join("a.drsc")).unwrap(), b"an unsynced edit", "an unsynced edit must survive the move" ); let _ = std::fs::remove_dir_all(&root); } #[test] fn a_migration_never_overwrites_live_data() { // Running twice, or a fresh install that already has a catalog. The // destination wins: it is the one the application is using. let root = std::env::temp_dir().join(format!("dr-migrate2-{}", std::process::id())); let _ = std::fs::remove_dir_all(&root); let legacy = root.join("old").join("acct"); let current = root.join("new").join("acct"); std::fs::create_dir_all(&legacy).unwrap(); std::fs::create_dir_all(¤t).unwrap(); std::fs::write(legacy.join("catalog.sqlite"), b"stale").unwrap(); std::fs::write(current.join("catalog.sqlite"), b"live").unwrap(); move_account_dir(&legacy, ¤t); assert_eq!( std::fs::read(current.join("catalog.sqlite")).unwrap(), b"live" ); let _ = std::fs::remove_dir_all(&root); } #[test] fn durable_data_never_lands_in_a_cache_directory() { // The fault this guards against is silent and total: on Android the // fallback used to be `temp_dir()`, which resolves to the app's cache // — a directory the system empties under storage pressure. Beside this // catalog sit `sidecars/`, the commit point for every offline rating // and edit, and `outbox/`, holding exports the user was told had // succeeded. Losing a day of culling to an OS housekeeping pass, with // no error and no trace, is the worst outcome this application has. let path = catalog_path(&account("https://cloud.example", "duncan")); let text = path.to_string_lossy().to_lowercase(); assert!( !text.contains("/cache/") && !text.contains("/tmp/"), "the catalog and everything beside it must be durable, got {}", path.display() ); } #[test] fn the_outbox_and_sidecars_sit_beside_the_catalog() { // Stated as a test because three separate call sites derive their // location by taking this path's parent, and a change here moves all // of them at once — including the two holding unsynced user work. let catalog = catalog_path(&account("https://cloud.example", "duncan")); let parent = catalog.parent().expect("a parent"); assert_eq!( crate::export::outbox_dir(&account("https://cloud.example", "duncan")), parent.join("outbox") ); } #[test] fn catalog_path_is_filesystem_safe() { let p = catalog_path(&account("https://cloud.example.com:8443/nc", "duncan")); let s = p.to_string_lossy(); assert!(!s.contains("://")); assert!(!s.contains(':') || cfg!(windows)); } #[test] fn persist_inserts_images_and_folder_etags() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 1001, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); assert_eq!(total_images(&catalog).unwrap(), 1); // Folder ETags must persist or the next scan prunes nothing. let etag: String = catalog .connection() .query_row( "SELECT etag FROM folders WHERE path = 'PhotosRaw'", [], |r| r.get(0), ) .unwrap(); assert_eq!(etag, "e1"); } #[test] fn images_land_as_stat_only_not_full_metadata() { // The scan read no EXIF. Claiming otherwise would make a date filter // silently wrong on a freshly scanned library. let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 1001, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let state: i64 = catalog .connection() .query_row("SELECT metadata_state FROM images", [], |r| r.get(0)) .unwrap(); assert_eq!(state, 1); } #[test] fn rescanning_updates_rather_than_duplicating() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 1001, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); persist(&catalog, "PhotosRaw", &result).unwrap(); assert_eq!(total_images(&catalog).unwrap(), 1, "no duplicate rows"); let roots: i64 = catalog .connection() .query_row("SELECT count(*) FROM roots", [], |r| r.get(0)) .unwrap(); assert_eq!(roots, 1, "no duplicate roots"); } #[test] fn stable_file_ids_are_recorded() { // FR-NC-5: a server-side move must be a move, not a re-download. let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 4242, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let file_id: i64 = catalog .connection() .query_row("SELECT file_id FROM remote", [], |r| r.get(0)) .unwrap(); assert_eq!(file_id, 4242); } #[test] fn a_thumbnail_job_is_queued_per_image() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![ entry("PhotosRaw/a.CR2", 1, 30_000_000), entry("PhotosRaw/b.CR2", 2, 30_000_000), ], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let jobs: i64 = catalog .connection() .query_row("SELECT count(*) FROM jobs WHERE kind = 2", [], |r| r.get(0)) .unwrap(); assert_eq!(jobs, 2); } #[test] fn a_second_scan_does_not_multiply_jobs() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 1, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); persist(&catalog, "PhotosRaw", &result).unwrap(); let jobs: i64 = catalog .connection() .query_row("SELECT count(*) FROM jobs WHERE kind = 2", [], |r| r.get(0)) .unwrap(); assert_eq!(jobs, 1, "coalesced, not queued twice"); } #[test] fn folder_etags_round_trip_for_the_next_scan() { let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![], directories: vec![ (RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1")), ( RemotePath::new("PhotosRaw/2026"), dr_sync::Validator::new("e2"), ), ], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let known = load_folder_etags(&catalog, "PhotosRaw"); assert_eq!(known.len(), 2); assert_eq!( known .get(&RemotePath::new("PhotosRaw/2026")) .map(|v| v.as_str()), Some("e2") ); } #[test] fn cells_are_windowed_not_wholesale() { let catalog = Catalog::in_memory().unwrap(); let images: Vec = (0..50) .map(|i| entry(&format!("PhotosRaw/img{i:03}.CR2"), i as u64, 1000)) .collect(); let result = dr_sync::ScanResult { images, directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let page = read_cells(&catalog, 10, 5).unwrap(); assert_eq!(page.len(), 5); assert_eq!(page[0].name, "img010.CR2"); } /// TRACES: NFR-P5 /// The grid's window read must be answered by walking `images_grid_order`, /// never by sorting the library into a temp b-tree. /// /// This asserts on the *query plan* rather than on a duration, because the /// failure has no other symptom: a `GRID_ORDER` edited out of step with the /// index in schema V7, or a column added back into the paging query that /// drags `remote` in again, both still return the right cells. They just /// return them after sorting 24,000 rows, inside the scroll handler — which /// is the jitter this pair was introduced to remove, and it would come back /// silently. #[test] fn the_window_read_walks_the_ordering_index() { let catalog = with_images(20); let plan: Vec = catalog .connection() .prepare(&format!( "EXPLAIN QUERY PLAN SELECT {CELL_COLUMNS} FROM images i WHERE {VISIBLE} {GRID_ORDER} LIMIT 10 OFFSET 5" )) .unwrap() .query_map([], |r| r.get::<_, String>(3)) .unwrap() .flatten() .collect(); let plan = plan.join(" | "); assert!( plan.contains("images_grid_order"), "the window read is not using the ordering index: {plan}" ); assert!( !plan.contains("TEMP B-TREE"), "the window read is still sorting the whole library: {plan}" ); assert!( !plan.to_lowercase().contains("remote"), "the window read is joining `remote` again, which pages the whole \ library before it discards it: {plan}" ); } /// The file ids still arrive, now that they come from a second query. #[test] fn a_window_still_carries_the_server_file_ids() { let catalog = with_images(20); let page = read_cells(&catalog, 5, 4).unwrap(); assert_eq!(page.len(), 4); assert!( page.iter().all(|c| c.file_id.is_some()), "a cell lost its file id when the join was split out" ); } /// A catalog with `n` images, ready to file into collections. fn with_images(n: usize) -> Catalog { let catalog = Catalog::in_memory().unwrap(); let images: Vec = (0..n) .map(|i| entry(&format!("PhotosRaw/img{i:03}.CR2"), i as u64, 1000)) .collect(); let result = dr_sync::ScanResult { images, directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); catalog } fn image_ids(catalog: &Catalog) -> Vec { let mut stmt = catalog .connection() .prepare("SELECT id FROM images ORDER BY source_ref") .unwrap(); stmt.query_map([], |r| Ok(dr_types::ImageId(r.get::<_, i64>(0)? as u64))) .unwrap() .map(Result::unwrap) .collect() } /// A face with no proxy left draws "no preview" and cannot repair itself: /// the image has its `face_index` row, so it is not outstanding work. The /// sweep has to pick it up by a second route. #[test] fn a_face_whose_proxy_is_gone_is_work_again() { let catalog = with_images(3); let ids = image_ids(&catalog); // An empty store, which is the state the bug lives in: the face is // recorded and there is nothing on disk to cut it out of. let store_dir = std::env::temp_dir().join(format!("dr-face-proxy-test-{}", std::process::id())); let _ = std::fs::remove_dir_all(&store_dir); std::fs::create_dir_all(&store_dir).unwrap(); let store = ThumbStore::open(&store_dir).unwrap(); let face = dr_catalog::faces::DetectedFace { x: 0.1, y: 0.1, w: 0.2, h: 0.2, landmarks: [(0.0, 0.0); 5], confidence: 0.9, embedding: vec![0u8; 1024], crop_px: 120.0, crop: Vec::new(), model_id: "w600k_mbf".into(), }; dr_catalog::faces::record_detections( catalog.connection(), ids[0], "w600k_mbf", 1024, std::slice::from_ref(&face), ) .unwrap(); // Indexed, so not outstanding — but with nothing to crop from. assert!(!faces_unindexed(&catalog, "w600k_mbf") .unwrap() .iter() .any(|r| r.image_id == ids[0].0 as i64)); let repair = faces_without_proxy(&catalog, &store, "w600k_mbf").unwrap(); assert_eq!(repair.len(), 1, "the orphaned face was not picked up"); assert_eq!(repair[0].image_id, ids[0].0 as i64); let _ = std::fs::remove_dir_all(&store_dir); } /// Ordering is load-bearing. A repair queued behind every un-indexed image /// in the library is a repair that does not happen inside a session, and /// the screen it was meant to fix stays empty. #[test] fn repairs_are_reached_before_the_rest_of_the_library() { let catalog = with_images(50); let ids = image_ids(&catalog); let store_dir = std::env::temp_dir().join(format!( "dr-face-order-test-{}-{:?}", std::process::id(), std::thread::current().id() )); let _ = std::fs::remove_dir_all(&store_dir); std::fs::create_dir_all(&store_dir).unwrap(); let store = ThumbStore::open(&store_dir).unwrap(); // One image late in the library has a face and no proxy. let face = dr_catalog::faces::DetectedFace { x: 0.1, y: 0.1, w: 0.2, h: 0.2, landmarks: [(0.0, 0.0); 5], confidence: 0.9, embedding: vec![0u8; 1024], crop_px: 120.0, crop: Vec::new(), model_id: "w600k_mbf".into(), }; let orphan = ids[40]; dr_catalog::faces::record_detections( catalog.connection(), orphan, "w600k_mbf", 1024, std::slice::from_ref(&face), ) .unwrap(); let mut wanted = faces_without_proxy(&catalog, &store, "w600k_mbf").unwrap(); wanted.extend(faces_unindexed(&catalog, "w600k_mbf").unwrap()); assert_eq!( wanted.first().map(|r| r.image_id), Some(orphan.0 as i64), "the image the screen cannot draw must be fetched first" ); assert_eq!( wanted.len(), 50, "49 un-indexed plus the one being repaired" ); let _ = std::fs::remove_dir_all(&store_dir); } /// The regression this whole pass exists for. /// /// The previous work list intersected with the thumbnail store, so a /// library nobody had zoomed into produced an empty one and "index the /// whole library" indexed nothing. Nothing here puts a proxy on disk. #[test] fn every_unindexed_image_is_work_even_with_no_proxy_anywhere() { let catalog = with_images(10); let wanted = faces_unindexed(&catalog, "w600k_mbf").unwrap(); assert_eq!( wanted.len(), 10, "a library with no thumbnails is still work" ); assert!( wanted.iter().all(|r| r.full_resolution), "the pass must keep the detail a thumbnail would throw away" ); } /// `face_index` records that detection *ran*, so an image with no face in /// it must not come back on the next pass — otherwise a personal library, /// which is mostly landscapes and documents, never finishes. #[test] fn an_image_already_run_over_is_not_work_again() { let catalog = with_images(3); let ids = image_ids(&catalog); dr_catalog::faces::record_detections(catalog.connection(), ids[0], "w600k_mbf", 1024, &[]) .unwrap(); let wanted = faces_unindexed(&catalog, "w600k_mbf").unwrap(); assert_eq!(wanted.len(), 2); assert!(!wanted.iter().any(|r| r.image_id == ids[0].0 as i64)); // A different model has seen none of them. assert_eq!(faces_unindexed(&catalog, "other").unwrap().len(), 3); } #[test] fn a_scoped_grid_shows_only_that_collections_images() { use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(10); let ids = image_ids(&catalog); let c = coll::create( catalog.connection(), "Selects", None, CollectionKind::Manual, ) .unwrap(); coll::add_images(catalog.connection(), c, &ids[2..5]).unwrap(); let cells = read_cells_scoped(&catalog, Some(c), &RatingFilter::default(), 0, 120).unwrap(); assert_eq!(cells.len(), 3); assert_eq!( total_images_scoped(&catalog, Some(c), &RatingFilter::default()).unwrap(), 3 ); // Unscoped is still the whole library. assert_eq!( total_images_scoped(&catalog, None, &RatingFilter::default()).unwrap(), 10 ); } #[test] fn a_collection_set_shows_its_childrens_images() { // A parent whose children hold everything must not read as empty — // that is what makes nesting look broken. use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(10); let ids = image_ids(&catalog); let trips = coll::create(catalog.connection(), "Trips", None, CollectionKind::Manual).unwrap(); let iceland = coll::create( catalog.connection(), "Iceland", Some(trips), CollectionKind::Manual, ) .unwrap(); coll::add_images(catalog.connection(), iceland, &ids[0..4]).unwrap(); // The parent itself has no direct members at all. let cells = read_cells_scoped(&catalog, Some(trips), &RatingFilter::default(), 0, 120).unwrap(); assert_eq!(cells.len(), 4, "the set shows what its children hold"); assert_eq!( total_images_scoped(&catalog, Some(trips), &RatingFilter::default()).unwrap(), 4 ); } #[test] fn an_image_in_both_a_parent_and_a_child_is_shown_once() { // The count and the number of cells drawn must agree, or neither is // believable. use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(10); let ids = image_ids(&catalog); let trips = coll::create(catalog.connection(), "Trips", None, CollectionKind::Manual).unwrap(); let iceland = coll::create( catalog.connection(), "Iceland", Some(trips), CollectionKind::Manual, ) .unwrap(); coll::add_images(catalog.connection(), trips, &ids[0..2]).unwrap(); coll::add_images(catalog.connection(), iceland, &ids[0..3]).unwrap(); let cells = read_cells_scoped(&catalog, Some(trips), &RatingFilter::default(), 0, 120).unwrap(); assert_eq!(cells.len(), 3, "images 0..3, each once"); assert_eq!( total_images_scoped(&catalog, Some(trips), &RatingFilter::default()).unwrap(), 3 ); } #[test] fn a_scoped_window_still_pages() { // FR-CAT-4 applies inside a collection too: a 5,000-image collection // must not become 5,000 rows. use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(30); let ids = image_ids(&catalog); let c = coll::create(catalog.connection(), "Big", None, CollectionKind::Manual).unwrap(); coll::add_images(catalog.connection(), c, &ids).unwrap(); let page = read_cells_scoped(&catalog, Some(c), &RatingFilter::default(), 10, 5).unwrap(); assert_eq!(page.len(), 5); assert_eq!(page[0].name, "img010.CR2"); } #[test] fn an_empty_collection_reads_as_empty_rather_than_as_the_whole_library() { // The failure that would make scoping useless: an empty IN-list // matching everything. use dr_catalog::collections::{self as coll, CollectionKind}; let catalog = with_images(10); let c = coll::create(catalog.connection(), "Empty", None, CollectionKind::Manual).unwrap(); assert!( read_cells_scoped(&catalog, Some(c), &RatingFilter::default(), 0, 120) .unwrap() .is_empty() ); assert_eq!( total_images_scoped(&catalog, Some(c), &RatingFilter::default()).unwrap(), 0 ); } #[test] fn cells_carry_the_file_id_the_thumbnail_store_keys_on() { // Without this the store can never be hit: every launch would refetch // every thumbnail over the network. let catalog = Catalog::in_memory().unwrap(); let result = dr_sync::ScanResult { images: vec![entry("PhotosRaw/a.CR2", 7777, 30_000_000)], directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e"))], progress: Default::default(), }; persist(&catalog, "PhotosRaw", &result).unwrap(); let cells = read_cells(&catalog, 0, 10).unwrap(); assert_eq!(cells[0].file_id, Some(7777)); } #[test] fn a_stored_thumbnail_survives_a_restart() { // The end-to-end property the store exists for: encode, persist, // reopen, decode. A second launch must not re-fetch. let dir = std::env::temp_dir().join(format!("dr-ui-thumbs-{}", std::process::id())); let _ = std::fs::remove_dir_all(&dir); let rgba: Vec = std::iter::repeat_n([90u8, 140, 200, 255], 32 * 32) .flatten() .collect(); { let mut store = ThumbStore::open(&dir).unwrap(); let bytes = dr_thumbs::encode_rgba(32, 32, &rgba).unwrap(); store .put( 4242, dr_thumbs::ThumbSize::Grid, &dr_thumbs::Thumbnail { width: 32, height: 32, bytes, }, ) .unwrap(); } let store = ThumbStore::open(&dir).unwrap(); let stored = store .get(4242, dr_thumbs::ThumbSize::Grid) .unwrap() .expect("persisted"); let (w, h, out) = dr_thumbs::decode_rgba(&stored.bytes).unwrap(); assert_eq!((w, h), (32, 32)); // Lossy, so compare approximately — a blue-ish pixel must stay blue. assert!(out[2] > out[0], "channel order survived the round trip"); } #[test] fn a_store_hit_is_not_evidence_the_server_is_reachable() { // The regression this guards: store hits were delivered as the same // `Ready` the network path sends, and the UI took any `Ready` as proof // of connectivity. A mostly-cached window then declared "back online" // against a server that was down — clearing the banner and kicking off // a sweep that immediately failed, on every scroll. let dir = std::env::temp_dir().join(format!("dr-ui-provenance-{}", std::process::id())); let _ = std::fs::remove_dir_all(&dir); let rgba: Vec = std::iter::repeat_n([10u8, 20, 30, 255], 8 * 8) .flatten() .collect(); let mut store = ThumbStore::open(&dir).unwrap(); let bytes = dr_thumbs::encode_rgba(8, 8, &rgba).unwrap(); store .put( 99, dr_thumbs::ThumbSize::Grid, &dr_thumbs::Thumbnail { width: 8, height: 8, bytes, }, ) .unwrap(); // The split in `spawn_thumbnails`: a hit decodes off local disk and is // reported with `from_cache` set, which is what the reachability gate // keys on. let stored = store .get(99, dr_thumbs::ThumbSize::Grid) .unwrap() .expect("stored"); let (width, height, rgba) = dr_thumbs::decode_rgba(&stored.bytes).unwrap(); let hit = ThumbnailReady { row: 0, width, height, rgba, from_cache: true, }; assert!( hit.from_cache, "a thumbnail read from the store must not be mistaken for a fetch" ); // And the mechanism it feeds: an offline tracker must survive it. let mut reach = dr_sync::Reachability::new(); let now = std::time::Instant::now(); reach.mark_unreachable("network error".into(), now); if !hit.from_cache { reach.mark_reachable(now); } assert!( reach.is_offline(), "replaying cached thumbnails must leave offline mode intact" ); } #[test] fn thumbs_live_beside_the_catalog_not_in_the_cache() { // They sync to the server and are shared with other clients, so a // cache sweep must not discard them. let cat = catalog_path(&account("https://cloud.example", "duncan")); let thumbs = thumbs_dir(&account("https://cloud.example", "duncan")); assert_eq!(thumbs.parent(), cat.parent()); } // --- the trash view (FR-CAT-15) --------------------------------------- /// A catalog with `n` scanned images, none trashed. fn scanned(n: u64) -> Catalog { let catalog = Catalog::in_memory().unwrap(); let images = (1..=n) .map(|i| entry(&format!("PhotosRaw/IMG_{i:04}.CR2"), 1000 + i, 30_000_000)) .collect(); persist( &catalog, "PhotosRaw", &dr_sync::ScanResult { images, directories: vec![(RemotePath::new("PhotosRaw"), dr_sync::Validator::new("e1"))], progress: Default::default(), }, ) .unwrap(); catalog } /// Mark one image trashed, at a given instant. fn trash_at(catalog: &Catalog, path: &str, when: i64) { let n = catalog .connection() .execute( "UPDATE images SET trashed_at = ?1, trashed_from = source_ref WHERE source_ref = ?2", rusqlite::params![when, path], ) .unwrap(); assert_eq!(n, 1, "fixture should have trashed exactly {path}"); } #[test] fn the_trash_lists_what_the_library_hides() { // The whole point of the view: these rows exist and no other query in // the application will show them. let catalog = scanned(3); trash_at(&catalog, "PhotosRaw/IMG_0002.CR2", 100); let trashed = read_trashed_cells(&catalog, 0, 50).unwrap(); assert_eq!(trashed.len(), 1); assert_eq!(trashed[0].remote_path, "PhotosRaw/IMG_0002.CR2"); // And it has left the library in the same move. let live = read_cells_all(&catalog, &RatingFilter::default(), 0, 50).unwrap(); assert_eq!(live.len(), 2); assert!(!live.iter().any(|c| c.remote_path.contains("IMG_0002"))); } #[test] fn an_untrashed_library_has_an_empty_trash() { let catalog = scanned(3); assert!(read_trashed_cells(&catalog, 0, 50).unwrap().is_empty()); assert_eq!(total_trashed(&catalog).unwrap(), 0); } #[test] fn the_trash_count_agrees_with_the_cells_it_lists() { // The header and the scrollbar are sized from the count while the grid // draws the cells. Two predicates that drift leave the user scrolling // through rows that are not there. let catalog = scanned(5); for (i, when) in [(1, 100), (3, 200), (5, 300)] { trash_at(&catalog, &format!("PhotosRaw/IMG_{i:04}.CR2"), when); } assert_eq!(total_trashed(&catalog).unwrap(), 3); assert_eq!(read_trashed_cells(&catalog, 0, 50).unwrap().len(), 3); } #[test] fn the_most_recently_trashed_image_is_listed_first() { // A mistaken delete is corrected within seconds, so the row the user // wants is the one they just made — not the oldest photograph. let catalog = scanned(3); trash_at(&catalog, "PhotosRaw/IMG_0001.CR2", 100); trash_at(&catalog, "PhotosRaw/IMG_0002.CR2", 300); trash_at(&catalog, "PhotosRaw/IMG_0003.CR2", 200); let order: Vec = read_trashed_cells(&catalog, 0, 50) .unwrap() .into_iter() .map(|c| c.remote_path) .collect(); assert_eq!( order, vec![ "PhotosRaw/IMG_0002.CR2".to_string(), "PhotosRaw/IMG_0003.CR2".to_string(), "PhotosRaw/IMG_0001.CR2".to_string(), ] ); } #[test] fn a_shadowed_jpeg_is_not_listed_beside_the_raw_it_belongs_to() { // The inversion applies to `trashed_at` only. Trashing a RAW takes its // sibling JPEG with it, and listing both would offer to restore the // same frame twice. let catalog = scanned(2); let c = catalog.connection(); let raw: i64 = c .query_row( "SELECT id FROM images WHERE source_ref = 'PhotosRaw/IMG_0001.CR2'", [], |r| r.get(0), ) .unwrap(); c.execute( "UPDATE images SET shadowed_by = ?1 WHERE source_ref = 'PhotosRaw/IMG_0002.CR2'", [raw], ) .unwrap(); trash_at(&catalog, "PhotosRaw/IMG_0001.CR2", 100); trash_at(&catalog, "PhotosRaw/IMG_0002.CR2", 100); let trashed = read_trashed_cells(&catalog, 0, 50).unwrap(); assert_eq!(trashed.len(), 1, "one photograph, not two"); assert_eq!(trashed[0].remote_path, "PhotosRaw/IMG_0001.CR2"); assert_eq!(total_trashed(&catalog).unwrap(), 1, "and the count agrees"); } #[test] fn the_trash_window_pages_like_the_grid_does() { // The trash uses the same windowed read as the library, so a large one // must not try to draw itself in a single query. let catalog = scanned(6); for i in 1..=6 { trash_at( &catalog, &format!("PhotosRaw/IMG_{i:04}.CR2"), 100 + i as i64, ); } let first = read_trashed_cells(&catalog, 0, 2).unwrap(); let second = read_trashed_cells(&catalog, 2, 2).unwrap(); assert_eq!(first.len(), 2); assert_eq!(second.len(), 2); assert!( first .iter() .all(|a| !second.iter().any(|b| b.image_id == a.image_id)), "pages must not overlap" ); } #[test] fn a_rating_filter_does_not_hide_anything_in_the_trash() { // The filter narrows a culling pass. A trash that dropped rows because // of a filter set elsewhere would look like it had lost them. let catalog = scanned(3); trash_at(&catalog, "PhotosRaw/IMG_0001.CR2", 100); // Nothing here is rated, so a four-star library filter would empty any // view that honoured it. let strict = RatingFilter { min_rating: 4, ..Default::default() }; assert!(read_cells_all(&catalog, &strict, 0, 50).unwrap().is_empty()); assert_eq!(read_trashed_cells(&catalog, 0, 50).unwrap().len(), 1); } #[test] fn a_date_range_narrows_the_grid_and_the_count_together() { // The whole reason the range lives on `RatingFilter`: every query path // threads that one struct, so the header cannot claim a total the grid // does not draw. let catalog = scanned(3); let conn = catalog.connection(); for (n, at) in [(1, 1_000), (2, 5_000), (3, 9_000)] { conn.execute( "UPDATE images SET captured_at = ?2 WHERE source_ref LIKE ?1", rusqlite::params![format!("%IMG_000{n}%"), at], ) .unwrap(); } let ranged = RatingFilter { captured_from: Some(4_000), captured_to: Some(6_000), ..Default::default() }; assert_eq!(read_cells_all(&catalog, &ranged, 0, 50).unwrap().len(), 1); assert_eq!(total_images_filtered(&catalog, &ranged).unwrap(), 1); } #[test] fn an_undated_image_is_not_shown_inside_a_date_range() { // It cannot be in or out of a span. Drawing it anyway makes a range the // user just chose look as though it had not applied. let catalog = scanned(2); catalog .connection() .execute("UPDATE images SET captured_at = NULL", []) .unwrap(); let ranged = RatingFilter { captured_from: Some(0), captured_to: Some(i64::MAX), ..Default::default() }; assert!(read_cells_all(&catalog, &ranged, 0, 50).unwrap().is_empty()); } #[test] fn a_span_reads_the_whole_run_whether_or_not_it_is_loaded() { // The shift-click this exists for. The grid holds a window of five and // the user names a run of twelve, so seven of them have no cell and no // id anywhere in the UI — but they are still what was asked for, and // the catalog is what knows them. let catalog = scanned(12); let filter = RatingFilter::default(); let loaded = read_cells_all(&catalog, &filter, 0, 5).unwrap(); assert_eq!(loaded.len(), 5, "the window is smaller than the run"); let whole: Vec<_> = read_cells_all(&catalog, &filter, 0, 50) .unwrap() .iter() .map(|c| dr_types::ImageId(c.image_id as u64)) .collect(); let span = read_ids_span(&catalog, None, &filter, false, 0, 11).unwrap(); assert_eq!(span.len(), 12); assert_eq!(span, whole, "the run is the grid's own list, in its order"); } #[test] fn a_span_starts_and_ends_where_it_was_asked_to() { // Ordinals index the grid's list, so a run has to be exactly the slice // of it the two ends name — one off at either end selects a // photograph the user did not point at. let catalog = scanned(12); let filter = RatingFilter::default(); let whole: Vec<_> = read_cells_all(&catalog, &filter, 0, 50) .unwrap() .iter() .map(|c| dr_types::ImageId(c.image_id as u64)) .collect(); let span = read_ids_span(&catalog, None, &filter, false, 4, 6).unwrap(); assert_eq!(span, whole[4..=6], "ordinals 4..=6, inclusive at both ends"); } #[test] fn a_span_is_ordered_by_capture_time_rather_than_by_name() { // A card written by two cameras interleaves names that have nothing to // do with each other. What a photographer means by "everything between // these two" is a stretch of an afternoon, so the run has to be taken // through capture time — the ordering the grid draws them in. let catalog = scanned(3); let conn = catalog.connection(); for (n, at) in [(1, 9_000), (2, 5_000), (3, 1_000)] { conn.execute( "UPDATE images SET captured_at = ?2 WHERE source_ref LIKE ?1", rusqlite::params![format!("%IMG_000{n}%"), at], ) .unwrap(); } let span = read_ids_span(&catalog, None, &RatingFilter::default(), false, 0, 2).unwrap(); let names: Vec = span .iter() .map(|id| { conn.query_row( "SELECT source_ref FROM images WHERE id = ?1", [id.0 as i64], |r| r.get::<_, String>(0), ) .unwrap() }) .collect(); assert!( names[0].ends_with("IMG_0003.CR2") && names[1].ends_with("IMG_0002.CR2") && names[2].ends_with("IMG_0001.CR2"), "earliest first, which here is the reverse of the file names: {names:?}" ); } #[test] fn the_histogram_ignores_the_range_it_is_used_to_choose() { // Drawing the axis through the chosen range would collapse it onto the // selection, leaving nowhere to widen back out from. let catalog = dated_three(); let ranged = RatingFilter { captured_from: Some(4_000), captured_to: Some(6_000), ..Default::default() }; assert_eq!( span_scoped(&catalog, None, &ranged), Some((1_000, 9_000)), "the axis must keep describing the whole extent" ); } /// Three photographs at 1000, 5000 and 9000 seconds. fn dated_three() -> Catalog { let catalog = scanned(3); let conn = catalog.connection(); for (n, at) in [(1, 1_000), (2, 5_000), (3, 9_000)] { conn.execute( "UPDATE images SET captured_at = ?2 WHERE source_ref LIKE ?1", rusqlite::params![format!("%IMG_000{n}%"), at], ) .unwrap(); } catalog } #[test] fn the_histogram_has_the_number_of_bins_it_was_asked_for() { // Fixed, whatever the span holds. The axis draws one bar per bin and // positions it by index, so a query that returned only the occupied // ones would put the bars at the wrong dates. let catalog = dated_three(); let filter = RatingFilter::default(); for bins in [1_u32, 8, 32, 64] { let bars = timeline_uniform(&catalog, None, &filter, 1_000, 9_000, bins).unwrap(); assert_eq!(bars.len() as u32, bins); assert_eq!( bars.iter().map(|b| b.count).sum::(), 3, "every photograph is counted exactly once" ); } } #[test] fn a_bin_starts_where_the_axis_says_it_does() { // The bar's start is its bin's left edge, not the earliest photograph // in it. It is what the position marker and the range band are drawn // against, and an empty bin has no photograph to borrow a date from. let catalog = dated_three(); let bars = timeline_uniform(&catalog, None, &RatingFilter::default(), 0, 8_000, 8).unwrap(); for (i, bar) in bars.iter().enumerate() { assert_eq!(bar.start, i as i64 * 1_000); } // 1000 and 5000 land in their own bins, 9000 is past the end. assert_eq!(bars[1].count, 1); assert_eq!(bars[5].count, 1); assert_eq!(bars.iter().map(|b| b.count).sum::(), 2); } #[test] fn the_last_bin_holds_a_photograph_taken_at_the_very_end() { // The division puts an image captured at exactly `to` one bin past the // axis. Uncapped it would be dropped from the histogram — and it is // precisely the image that defines the extent, so it would go missing // on every unzoomed library. let catalog = dated_three(); let bars = timeline_uniform(&catalog, None, &RatingFilter::default(), 1_000, 9_000, 4).unwrap(); assert_eq!(bars.len(), 4); assert_eq!(bars[3].count, 1, "the image at 9000 is in the last bin"); assert_eq!(bars[0].count, 1); } #[test] fn the_bins_ignore_the_range_they_are_used_to_choose() { // Same rule as the extent: the bars outside the band are what the // range is widened back into, so counting through the range would // leave every one of them empty. let catalog = dated_three(); let ranged = RatingFilter { captured_from: Some(4_000), captured_to: Some(6_000), ..Default::default() }; let bars = timeline_uniform(&catalog, None, &ranged, 1_000, 9_000, 4).unwrap(); assert_eq!( bars.iter().map(|b| b.count).sum::(), 3, "all three, not just the one inside the range" ); } #[test] fn the_bins_still_honour_every_other_filter() { // A histogram of the five-star frames is a fair question, and the bars // have to agree with the grid beneath them. let catalog = dated_three(); let strict = RatingFilter { min_rating: 4, ..Default::default() }; let bars = timeline_uniform(&catalog, None, &strict, 1_000, 9_000, 4).unwrap(); assert_eq!(bars.len(), 4, "the axis keeps its shape"); assert_eq!( bars.iter().map(|b| b.count).sum::(), 0, "nothing here is rated" ); } fn entry(path: &str, file_id: u64, size: u64) -> RemoteEntry { RemoteEntry { id: RemoteId::Stable(file_id), path: RemotePath::new(path), kind: dr_sync::EntryKind::File, validator: dr_sync::Validator::new("v"), size, modified: None, has_preview: false, materialised: true, } } // --- the local-only filter (FR-CAT-9) --------------------------------- /// Record that an image's original is held locally at `tier`. fn cache_at(catalog: &Catalog, id: dr_types::ImageId, tier: dr_types::Tier) { catalog .connection() .execute( "INSERT INTO image_cache (image_id, tier_actual, bytes) VALUES (?1, ?2, 0)", rusqlite::params![id.0 as i64, tier.stored()], ) .unwrap(); } #[test] fn local_only_shows_just_the_images_held_here() { let catalog = with_images(10); let ids = image_ids(&catalog); for id in &ids[0..3] { cache_at(&catalog, *id, dr_types::Tier::Original); } let filter = RatingFilter { local_only: true, ..Default::default() }; let cells = read_cells_all(&catalog, &filter, 0, 120).unwrap(); assert_eq!(cells.len(), 3); // The count the header shows must agree with the cells drawn, which is // the whole reason the predicate lives in SQL rather than in a // post-filter over the rows. assert_eq!(total_images_filtered(&catalog, &filter).unwrap(), 3); assert_eq!(local_original_count(&catalog).unwrap(), 3); } #[test] fn a_cached_preview_is_not_a_local_original() { // The filter answers "can I open this in develop right now", and a // preview cannot. Counting it would put images in the offline set that // fail the moment they are clicked. let catalog = with_images(5); let ids = image_ids(&catalog); cache_at(&catalog, ids[0], dr_types::Tier::Preview); cache_at(&catalog, ids[1], dr_types::Tier::Original); let filter = RatingFilter { local_only: true, ..Default::default() }; assert_eq!(read_cells_all(&catalog, &filter, 0, 120).unwrap().len(), 1); assert_eq!(local_original_count(&catalog).unwrap(), 1); } #[test] fn local_only_composes_with_the_rating_filter() { // "Five-star frames I can actually edit on this train" is one filter, // not a mode that replaces the others. let catalog = with_images(6); let ids = image_ids(&catalog); for id in &ids[0..4] { cache_at(&catalog, *id, dr_types::Tier::Original); } // Rate two of the cached ones, and one that is not cached. for id in [ids[0], ids[1], ids[5]] { dr_catalog::rating::set_rating(catalog.connection(), id, 5).unwrap(); } let filter = RatingFilter { min_rating: 5, local_only: true, ..Default::default() }; let cells = read_cells_all(&catalog, &filter, 0, 120).unwrap(); assert_eq!(cells.len(), 2, "five-starred AND held locally"); assert_eq!(total_images_filtered(&catalog, &filter).unwrap(), 2); } #[test] fn an_empty_cache_is_not_an_empty_library() { // The unfiltered grid must not depend on the cache table having rows — // a library nothing has been downloaded from is still a full library. let catalog = with_images(4); assert_eq!(local_original_count(&catalog).unwrap(), 0); assert_eq!( read_cells_all(&catalog, &RatingFilter::default(), 0, 120) .unwrap() .len(), 4 ); } #[test] fn local_only_counts_as_a_narrowing_filter() { // `is_unfiltered` gates the "filtered" indicator. Reporting this one as // unfiltered would leave a narrowed grid looking like the whole // library, which is the state the indicator exists to prevent. assert!(RatingFilter::default().is_unfiltered()); assert!(!RatingFilter { local_only: true, ..Default::default() } .is_unfiltered()); } // ── narrowing the grid by identity ──────────────────────────────────── /// Put `person` on the given images, as a suggestion. fn assign( catalog: &Catalog, person: dr_catalog::faces::PersonId, images: &[dr_types::ImageId], ) { for img in images { let face = dr_catalog::faces::DetectedFace { x: 0.1, y: 0.1, w: 0.2, h: 0.2, landmarks: [(0.0, 0.0); 5], confidence: 0.9, embedding: vec![0u8; 1024], crop_px: 120.0, crop: Vec::new(), model_id: "w600k_mbf".into(), }; // Appends rather than replaces across calls for *different* // people, because `record_detections` clears the image first — // so the second person's face is added by hand. let existing: Vec = { let mut q = catalog .connection() .prepare("SELECT id FROM faces WHERE image_id = ?1") .unwrap(); q.query_map([img.0 as i64], |r| r.get(0)) .unwrap() .map(Result::unwrap) .collect() }; let id = if existing.is_empty() { dr_catalog::faces::record_detections( catalog.connection(), *img, "w600k_mbf", 1024, std::slice::from_ref(&face), ) .unwrap()[0] } else { catalog .connection() .execute( "INSERT INTO faces (image_id, x, y, w, h, landmarks, detector_confidence, embedding, crop_px, model_id, detected_at) VALUES (?1, 0.5, 0.5, 0.2, 0.2, X'00', 0.9, X'00', 120.0, 'w600k_mbf', 0)", [img.0 as i64], ) .unwrap(); dr_catalog::faces::FaceId(catalog.connection().last_insert_rowid() as u64) }; dr_catalog::faces::suggest(catalog.connection(), id, person, 0.9).unwrap(); } } /// The two questions a photographer actually asks, and the reason the /// filter holds a set rather than one id: the intersection is not reachable /// by any sequence of single-person filters. #[test] fn people_narrow_the_grid_as_a_union_or_an_intersection() { let catalog = with_images(4); let ids = image_ids(&catalog); let anna = dr_catalog::faces::create_person(catalog.connection(), "Anna").unwrap(); let bob = dr_catalog::faces::create_person(catalog.connection(), "Bob").unwrap(); // 0: Anna. 1: both. 2: Bob. 3: neither. assign(&catalog, anna, &[ids[0], ids[1]]); assign(&catalog, bob, &[ids[1], ids[2]]); let count = |people: Vec, mode: PeopleMode| { let f = RatingFilter { people, people_mode: mode, ..Default::default() }; total_images_scoped(&catalog, None, &f).unwrap() }; assert_eq!(count(vec![anna.0], PeopleMode::Any), 2, "Anna alone"); assert_eq!(count(vec![bob.0], PeopleMode::Any), 2, "Bob alone"); assert_eq!( count(vec![anna.0, bob.0], PeopleMode::Any), 3, "the union should hold every picture either is in" ); assert_eq!( count(vec![anna.0, bob.0], PeopleMode::All), 1, "the intersection should hold only the picture they share" ); } /// One person is the same filter either way, and the UI leans on that to /// hide the toggle until there are two. #[test] fn one_person_reads_the_same_in_both_modes() { let catalog = with_images(3); let ids = image_ids(&catalog); let anna = dr_catalog::faces::create_person(catalog.connection(), "Anna").unwrap(); assign(&catalog, anna, &[ids[0], ids[1]]); for mode in [PeopleMode::Any, PeopleMode::All] { let f = RatingFilter { people: vec![anna.0], people_mode: mode, ..Default::default() }; assert_eq!(total_images_scoped(&catalog, None, &f).unwrap(), 2); } } /// Three faces of one person in a frame must not satisfy "Anna and Bob". /// This is what `COUNT(DISTINCT ...)` is for, and it is the intersection's /// one real trap. #[test] fn repeated_faces_of_one_person_do_not_satisfy_an_intersection() { let catalog = with_images(2); let ids = image_ids(&catalog); let anna = dr_catalog::faces::create_person(catalog.connection(), "Anna").unwrap(); let bob = dr_catalog::faces::create_person(catalog.connection(), "Bob").unwrap(); // Two separate faces, both Anna, in the same photograph. assign(&catalog, anna, &[ids[0]]); assign(&catalog, anna, &[ids[0]]); let f = RatingFilter { people: vec![anna.0, bob.0], people_mode: PeopleMode::All, ..Default::default() }; assert_eq!(total_images_scoped(&catalog, None, &f).unwrap(), 0); } #[test] fn no_people_narrows_nothing() { let catalog = with_images(3); let f = RatingFilter::default(); assert!(f.is_unfiltered()); assert_eq!(total_images_scoped(&catalog, None, &f).unwrap(), 3); } }