//! TRACES: FR-CAT-3 | FR-NC-3 | NFR-RES-4 //! A shared, syncable thumbnail store. //! //! # Why thumbnails sync when the catalog mostly does not //! //! A thumbnail is the one derived artefact worth sending over the wire. It is //! expensive to produce — a range fetch plus a decode, per image — and //! identical for every client looking at the same file. A second device that //! can download a shard gets a full grid without re-fetching a byte of RAW. //! On the reference library that is the difference between a usable tablet and //! one that spends an hour filling in cells it could have been handed. //! //! This does not make thumbnails authoritative. Losing the store costs //! regeneration, nothing more; sidecars remain the only thing in the trust //! path (ARCH §6.12). //! //! # Shape //! //! ```text //! thumbs/ //! index.sqlite fileid → shard, plus size accounting //! shard-0000.sqlite ≤ 25 MB of blobs //! shard-0001.sqlite //! ``` //! //! **Sharded, and sized small on purpose.** The cap is not about SQLite's //! limits — it is about sync granularity. A single growing database means //! every client re-downloads the whole thing whenever one thumbnail is added. //! With sequential fill only the newest shard is ever dirty, so a client that //! is up to date transfers one small file. Sealed shards never change again, //! which also makes them safe to cache indefinitely. //! //! # Identity //! //! Keyed on Nextcloud's `oc:fileid`, which is stable across server-side rename //! and move (FR-NC-5) and which every client already holds from PROPFIND. The //! consequence, accepted deliberately: shards are account-scoped, so the same //! photograph on two different servers is thumbnailed twice. use std::path::{Path, PathBuf}; use rusqlite::Connection; pub mod codec; pub mod error; pub use codec::{decode_rgba, encode_rgba}; pub use error::ThumbError; /// Maximum bytes per shard before a new one is started. /// /// 25 MB, chosen for *sync* cost rather than storage: it bounds what a client /// re-downloads when the active shard changes, and keeps a stalled transfer /// cheap to retry. At ~20 KB per 256px JPEG that is roughly 1,200 thumbnails /// per shard, so the 17k-image reference library lands in ~14 shards. pub const SHARD_MAX_BYTES: u64 = 25 * 1024 * 1024; /// Which resolution a stored thumbnail is. /// /// Two classes rather than one, because the grid zooms: 256px is right for a /// wall of small cells and soft on a large one, while storing everything large /// would take the reference library from ~200 MB to ~860 MB — and the shards /// **sync**, so that is transfer cost on every device, not just disk. /// /// The discriminant is part of the store key, so both classes coexist and a /// library thumbnailed at one size is not invalidated by the other appearing. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)] #[repr(i64)] pub enum ThumbSize { /// Grid cells at their usual size. ~10 KB each. Grid = 0, /// Zoomed cells, the loupe, and the filmstrip. ~45 KB each, fetched only /// where something actually asks for that detail. Large = 1, } impl ThumbSize { /// Long edge in pixels. pub fn edge(self) -> u32 { match self { ThumbSize::Grid => 256, ThumbSize::Large => 1024, } } /// The smallest class that can fill a cell of this size without visibly /// softening. /// /// Compared against the *drawn* size, so a high-DPI display asking for /// 300 logical pixels at 2x gets the large class, as it should. pub fn for_cell(pixels: u32) -> Self { if pixels > ThumbSize::Grid.edge() { ThumbSize::Large } else { ThumbSize::Grid } } fn from_i64(v: i64) -> Self { match v { 1 => ThumbSize::Large, _ => ThumbSize::Grid, } } } /// Long edge of a grid thumbnail. /// /// Kept as the name callers already use; prefer [`ThumbSize::edge`]. pub const THUMBNAIL_EDGE: u32 = 256; /// A thumbnail's bytes and dimensions. #[derive(Debug, Clone, PartialEq, Eq)] pub struct Thumbnail { pub width: u32, pub height: u32, /// Encoded image bytes — JPEG. Stored encoded, not raw RGBA: a 256×256 /// RGBA buffer is 256 KB where its JPEG is ~20 KB, and a store that shipped /// raw pixels would be an order of magnitude more sync traffic. pub bytes: Vec, } /// The sharded store. pub struct ThumbStore { dir: PathBuf, index: Connection, client: String, } impl ThumbStore { /// Open or create a store under `dir`. pub fn open(dir: &Path) -> Result { std::fs::create_dir_all(dir).map_err(|e| ThumbError::Io(e.to_string()))?; let index = Connection::open(dir.join("index.sqlite"))?; index.pragma_update(None, "journal_mode", "WAL")?; index.pragma_update(None, "synchronous", "NORMAL")?; index.execute_batch(INDEX_SCHEMA)?; // A store written before the size class existed holds grid-sized // entries under a bare `file_id` key; bring it forward rather than // discarding every thumbnail already fetched. migrate_size_column(&index, "entries")?; let client = mint_client_id(&index)?; Ok(Self { dir: dir.to_path_buf(), index, client, }) } /// This store's identity among the clients sharing a library. /// /// Shard ids are per-store — every client fills its own numbering from 0 — /// so a name that travels needs this to say *whose* shard 3 it is. pub fn client_id(&self) -> &str { &self.client } /// The size a remote shard had when it was last merged, if it ever was. /// /// A size rather than a bare flag: a peer's sealed shard never changes /// again, but its open one grows, and re-merging the grown copy is how the /// thumbnails it gained since arrive. pub fn adopted(&self, name: &str) -> Option { self.index .query_row("SELECT size FROM adopted WHERE name = ?1", [name], |r| { r.get::<_, i64>(0) }) .ok() .map(|bytes| bytes as u64) } /// Record that a remote shard of this name and size has been merged. pub fn record_adopted(&self, name: &str, size: u64) -> Result<(), ThumbError> { self.index.execute( "INSERT INTO adopted(name, size) VALUES (?1, ?2) ON CONFLICT(name) DO UPDATE SET size = excluded.size", rusqlite::params![name, size as i64], )?; Ok(()) } /// Fetch a thumbnail by file id. pub fn get(&self, file_id: u64, size: ThumbSize) -> Result, ThumbError> { let shard: Option = self .index .query_row( "SELECT shard FROM entries WHERE file_id = ?1 AND size = ?2", [file_id as i64, size as i64], |r| r.get(0), ) .ok(); let Some(shard) = shard else { return Ok(None); }; let conn = self.open_shard(shard as u32, false)?; let got = conn .query_row( "SELECT width, height, bytes FROM thumbs WHERE file_id = ?1 AND size = ?2", [file_id as i64, size as i64], |r| { Ok(Thumbnail { width: r.get::<_, i64>(0)? as u32, height: r.get::<_, i64>(1)? as u32, bytes: r.get(2)?, }) }, ) .ok(); Ok(got) } /// Whether a thumbnail is already stored. /// /// Cheaper than [`get`](Self::get) — the index alone answers it, with no /// shard opened and no blob read. pub fn contains(&self, file_id: u64, size: ThumbSize) -> bool { self.index .query_row( "SELECT 1 FROM entries WHERE file_id = ?1 AND size = ?2", [file_id as i64, size as i64], |_| Ok(()), ) .is_ok() } /// Which of these file ids are missing, preserving order. /// /// The grid's real question — "what must I fetch for these cells" — asked /// in one pass rather than one query per cell. pub fn missing(&self, file_ids: &[u64], size: ThumbSize) -> Vec { file_ids .iter() .copied() .filter(|id| !self.contains(*id, size)) .collect() } /// Every file id stored at one size, read from the index in one query. /// /// For a pass that asks about thousands of images at once — the face /// audit, the repair lists. Each [`contains`](Self::contains) is a /// prepared statement and a b-tree probe; asked ten thousand times over a /// scan it costs more than the scan does, where one walk of the index is /// a few milliseconds and answers every row. pub fn held(&self, size: ThumbSize) -> Result, ThumbError> { let mut stmt = self .index .prepare("SELECT file_id FROM entries WHERE size = ?1")?; let ids = stmt .query_map([size as i64], |r| r.get::<_, i64>(0))? .collect::, _>>()?; Ok(ids.into_iter().map(|id| id as u64).collect()) } /// Store a thumbnail, opening a new shard if the active one is full. /// /// Re-storing an existing id overwrites in place rather than migrating it /// to the active shard: a sealed shard must stay byte-identical for other /// clients, and rewriting one would force them all to re-sync it. pub fn put( &mut self, file_id: u64, size: ThumbSize, thumb: &Thumbnail, ) -> Result { if let Some(existing) = self.shard_of(file_id, size) { let conn = self.open_shard(existing, true)?; conn.execute( "UPDATE thumbs SET width = ?3, height = ?4, bytes = ?5 WHERE file_id = ?1 AND size = ?2", rusqlite::params![ file_id as i64, size as i64, thumb.width as i64, thumb.height as i64, thumb.bytes ], )?; return Ok(existing); } let shard = self.active_shard(thumb.bytes.len() as u64)?; let conn = self.open_shard(shard, true)?; conn.execute( "INSERT INTO thumbs(file_id, size, width, height, bytes) VALUES (?1, ?2, ?3, ?4, ?5) ON CONFLICT(file_id, size) DO UPDATE SET width = excluded.width, height = excluded.height, bytes = excluded.bytes", rusqlite::params![ file_id as i64, size as i64, thumb.width as i64, thumb.height as i64, thumb.bytes ], )?; self.index.execute( "INSERT INTO entries(file_id, size, shard, bytes) VALUES (?1, ?2, ?3, ?4) ON CONFLICT(file_id, size) DO UPDATE SET shard = excluded.shard, bytes = excluded.bytes", rusqlite::params![ file_id as i64, size as i64, shard as i64, thumb.bytes.len() as i64 ], )?; self.index.execute( "INSERT INTO shards(id, bytes, sealed) VALUES (?1, ?2, 0) ON CONFLICT(id) DO UPDATE SET bytes = shards.bytes + ?2", rusqlite::params![shard as i64, thumb.bytes.len() as i64], )?; Ok(shard) } fn shard_of(&self, file_id: u64, size: ThumbSize) -> Option { self.index .query_row( "SELECT shard FROM entries WHERE file_id = ?1 AND size = ?2", [file_id as i64, size as i64], |r| r.get::<_, i64>(0), ) .ok() .map(|v| v as u32) } /// The shard that should receive `incoming` bytes. /// /// Seals the current shard and starts a new one when it would exceed the /// cap. Sealing is recorded rather than inferred from file size, so a /// shard stays sealed even if a later SQLite vacuum shrinks it. fn active_shard(&self, incoming: u64) -> Result { let current: Option<(i64, i64)> = self .index .query_row( "SELECT id, bytes FROM shards WHERE sealed = 0 ORDER BY id DESC LIMIT 1", [], |r| Ok((r.get(0)?, r.get(1)?)), ) .ok(); match current { Some((id, bytes)) if (bytes as u64) + incoming <= SHARD_MAX_BYTES => Ok(id as u32), Some((id, _)) => { self.index .execute("UPDATE shards SET sealed = 1 WHERE id = ?1", [id])?; let next = (id + 1) as u32; self.index.execute( "INSERT INTO shards(id, bytes, sealed) VALUES (?1, 0, 0) ON CONFLICT(id) DO NOTHING", [next as i64], )?; Ok(next) } None => { self.index.execute( "INSERT INTO shards(id, bytes, sealed) VALUES (0, 0, 0) ON CONFLICT(id) DO NOTHING", [], )?; Ok(0) } } } fn open_shard(&self, shard: u32, create: bool) -> Result { let path = self.shard_path(shard); if !create && !path.exists() { return Err(ThumbError::MissingShard(shard)); } let conn = Connection::open(&path)?; conn.pragma_update(None, "journal_mode", "WAL")?; conn.pragma_update(None, "synchronous", "NORMAL")?; if create { conn.execute_batch(SHARD_SCHEMA)?; } // Every shard carries its own `thumbs` table, so migrating the index // alone is not enough — and `CREATE TABLE IF NOT EXISTS` leaves an // existing one untouched, so a shard written before the size class // keeps the old shape and every write to it fails with "no column // named size". migrate_size_column(&conn, "thumbs")?; Ok(conn) } /// TRACES: FR-CAT-15 | NFR-RES-4 /// Drop thumbnails for images that no longer exist. /// /// The permanent-delete half of the trash: an image purged from the library /// must not leave a preview behind. The shards **sync**, so a stale entry is /// not merely local clutter — every other client keeps showing a thumbnail /// for a photograph that is gone, and keeps downloading it. /// /// # What this deliberately does not do /// /// It does not rewrite the shard's blob, and it does not `VACUUM`. A sealed /// shard must stay **byte-identical** or every client re-downloads all 25 MB /// of it to reclaim 20 KB — the store's whole sync design (see the module /// preamble) rests on sealed shards never changing. So the blob is deleted /// from the *unsealed* shard only, and in a sealed one the row is left in /// place and merely unlinked from the index. /// /// The consequence, accepted: a sealed shard can carry dead weight. It is /// bounded — 25 MB per shard, only for images actually purged — and it is /// invisible, because [`get`](Self::get) resolves through the index and an /// unindexed blob is unreachable. Reclaiming it belongs to a compaction pass /// that rewrites a shard under a new id, which is a separate concern with its /// own sync cost. /// /// Returns how many entries were forgotten. pub fn forget(&mut self, file_ids: &[u64]) -> Result { if file_ids.is_empty() { return Ok(0); } // Which shard each id lives in, and whether that shard is still open for // writing. Read before deleting anything, so the index is consulted once. let mut in_unsealed: Vec<(u64, u32)> = Vec::new(); let mut forgotten = 0usize; let tx = self.index.unchecked_transaction()?; for &file_id in file_ids { // Every size class, not just one: an image browsed at two sizes has // two entries, and reading a single row would leave the other's // bytes on the shard's tally forever — sealing it early on space // nothing occupies. let rows: Vec<(i64, i64, i64, i64)> = { let mut stmt = tx.prepare( "SELECT e.size, e.shard, e.bytes, s.sealed FROM entries e JOIN shards s ON s.id = e.shard WHERE e.file_id = ?1", )?; let mapped = stmt.query_map([file_id as i64], |r| { Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)) })?; mapped.collect::, _>>()? }; // Never stored, or already forgotten. Not an error: a purge runs // over whatever the catalog knew about, and a thumbnail that was // never fetched is the normal case. if rows.is_empty() { continue; } for (size, shard, bytes, sealed) in rows { tx.execute( "DELETE FROM entries WHERE file_id = ?1 AND size = ?2", [file_id as i64, size], )?; forgotten += 1; if sealed != 0 { continue; } // The active shard is not yet anyone's cached copy, so its // bytes can genuinely be reclaimed and the accounting // corrected — which also means the shard does not seal // prematurely on space that is no longer used. in_unsealed.push((file_id, shard as u32)); tx.execute( "UPDATE shards SET bytes = max(0, bytes - ?2) WHERE id = ?1", rusqlite::params![shard, bytes], )?; } } tx.commit()?; // Blobs after the index commits. The other order would leave the index // pointing at a blob that is gone if this failed partway — a `get` would // then find an entry and no thumbnail, which reads as corruption. This // way a failure here leaves an unreachable blob, which is the same // harmless state a sealed shard is in by design. for (file_id, shard) in in_unsealed { match self.open_shard(shard, false) { Ok(conn) => { if let Err(e) = conn.execute("DELETE FROM thumbs WHERE file_id = ?1", [file_id as i64]) { log::debug!("forgetting thumbnail {file_id} in shard {shard}: {e}"); } } Err(e) => log::debug!("opening shard {shard} to forget {file_id}: {e}"), } } Ok(forgotten) } pub fn shard_path(&self, shard: u32) -> PathBuf { self.dir.join(format!("shard-{shard:04}.sqlite")) } /// Write a coherent copy of one shard to `dest`, ready to upload. /// /// Not a file copy. Every shard is in WAL mode and every `put` opens its /// own connection, so while thumbnails are being generated on several /// threads at once — which is exactly when the first sync pass runs — /// there is nearly always a connection open and the log is never /// checkpointed. The main file then holds whatever the *last* quiet /// moment left in it, which for a shard created seconds ago is nothing: /// zero bytes, the schema still in the log. Reading it uploaded an empty /// file, and every other device merging it failed with "no such table: /// thumbs". The backup API serialises against writers and copies the /// database as it is, log included. pub fn snapshot_shard(&self, shard: u32, dest: &Path) -> Result<(), ThumbError> { let source = self.open_shard(shard, false)?; let _ = std::fs::remove_file(dest); let mut out = Connection::open(dest)?; let backup = rusqlite::backup::Backup::new(&source, &mut out)?; // rusqlite asserts a positive page count where SQLite would take -1 // for "everything"; a shard is capped well under this many pages. backup.run_to_completion(i32::MAX, std::time::Duration::ZERO, None)?; drop(backup); Ok(()) } pub fn index_path(&self) -> PathBuf { self.dir.join("index.sqlite") } /// Every shard, with whether it is sealed. /// /// The sync layer's input: sealed shards never change again, so once a /// client has one it need never ask for it. Only the unsealed shard and /// the index are worth re-checking. pub fn shards(&self) -> Result, ThumbError> { let mut stmt = self .index .prepare("SELECT id, bytes, sealed FROM shards ORDER BY id")?; let rows = stmt .query_map([], |r| { Ok(ShardInfo { id: r.get::<_, i64>(0)? as u32, bytes: r.get::<_, i64>(1)? as u64, sealed: r.get::<_, i64>(2)? != 0, }) })? .collect::, _>>()?; Ok(rows) } /// Total thumbnails stored. pub fn len(&self) -> usize { self.index .query_row("SELECT count(*) FROM entries", [], |r| r.get::<_, i64>(0)) .unwrap_or(0) as usize } pub fn is_empty(&self) -> bool { self.len() == 0 } /// Merge a shard downloaded from another client. /// /// Insert-only: a thumbnail we already hold is kept. They are derived from /// the same bytes by the same code, so neither copy is better, and /// preferring ours avoids rewriting a shard other clients have already /// synced. /// /// Returns how many were adopted. pub fn merge_shard(&mut self, downloaded: &Path) -> Result { let src = Connection::open_with_flags( downloaded, rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY | rusqlite::OpenFlags::SQLITE_OPEN_NO_MUTEX, )?; // A shard from a client that predates the size class has no `size` // column; its thumbnails are all grid-sized, which is what the // fallback below assumes. let has_size = src .prepare("SELECT * FROM thumbs LIMIT 0") .map(|stmt| stmt.column_names().contains(&"size")) .unwrap_or(false); let sql = if has_size { "SELECT file_id, size, width, height, bytes FROM thumbs" } else { "SELECT file_id, 0 AS size, width, height, bytes FROM thumbs" }; let mut stmt = src.prepare(sql)?; let incoming = stmt .query_map([], |r| { Ok(( r.get::<_, i64>(0)? as u64, ThumbSize::from_i64(r.get::<_, i64>(1)?), Thumbnail { width: r.get::<_, i64>(2)? as u32, height: r.get::<_, i64>(3)? as u32, bytes: r.get(4)?, }, )) })? .collect::, _>>()?; let mut adopted = 0; for (file_id, size, thumb) in incoming { if self.contains(file_id, size) { continue; } self.put(file_id, size, &thumb)?; adopted += 1; } Ok(adopted) } } /// One shard's sync-relevant state. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct ShardInfo { pub id: u32, pub bytes: u64, /// A sealed shard is immutable. Once downloaded it never needs re-checking. pub sealed: bool, } const INDEX_SCHEMA: &str = r#" CREATE TABLE IF NOT EXISTS entries ( file_id INTEGER NOT NULL, -- oc:fileid, stable across rename/move -- Which resolution. Part of the key so both classes coexist: adding the -- large class must not invalidate a library already thumbnailed small. size INTEGER NOT NULL DEFAULT 0, shard INTEGER NOT NULL, bytes INTEGER NOT NULL, PRIMARY KEY(file_id, size) ); CREATE INDEX IF NOT EXISTS entries_shard ON entries(shard); CREATE TABLE IF NOT EXISTS shards ( id INTEGER PRIMARY KEY, bytes INTEGER NOT NULL DEFAULT 0, -- Recorded, not inferred from file size: a vacuum could shrink a sealed -- shard below the cap and it must still stay closed, or clients that -- already hold it would see it change. sealed INTEGER NOT NULL DEFAULT 0 ); -- Facts about this store rather than about any thumbnail. The index never -- leaves the device, so what lives here is safe to be device-specific. CREATE TABLE IF NOT EXISTS meta ( key TEXT PRIMARY KEY, value TEXT NOT NULL ); -- Remote shards already merged, under the name and size they had on the -- server. Nothing in a merged thumbnail records where it came from, so without -- this ledger a client either re-downloads every peer's shard on every sync or -- guesses from its own numbering — and its own numbering says nothing about -- anyone else's. CREATE TABLE IF NOT EXISTS adopted ( name TEXT PRIMARY KEY, size INTEGER NOT NULL ); "#; /// Read this store's client id, minting one on first open. /// /// It lives in the index because that is where the shard numbering it /// qualifies lives: a store deleted and rebuilt starts again at shard 0, and /// must not claim the remote names its predecessor wrote. SQLite's own /// randomness keeps it dependency-free, and six bytes separate far more /// devices than one account ever has. fn mint_client_id(conn: &Connection) -> Result { conn.execute( "INSERT INTO meta(key, value) VALUES('client_id', lower(hex(randomblob(6)))) ON CONFLICT(key) DO NOTHING", [], )?; Ok( conn.query_row("SELECT value FROM meta WHERE key = 'client_id'", [], |r| { r.get(0) })?, ) } const SHARD_SCHEMA: &str = r#" CREATE TABLE IF NOT EXISTS thumbs ( file_id INTEGER NOT NULL, size INTEGER NOT NULL DEFAULT 0, width INTEGER NOT NULL, height INTEGER NOT NULL, bytes BLOB NOT NULL, PRIMARY KEY(file_id, size) ); "#; /// Bring a store written before the size class existed up to date. /// /// Those entries are all grid-sized, which is what the column defaults to, so /// the migration is purely structural — no thumbnail is discarded and nothing /// is re-fetched. A store that has never been opened by an older build runs /// this as a no-op. /// /// Sealed shards *are* rewritten here, which normally the design forbids /// (their immutability is what makes syncing cheap). It is acceptable exactly /// once: every client migrates the same way, and the alternative is discarding /// every thumbnail already fetched. fn migrate_size_column(conn: &Connection, table: &str) -> Result<(), ThumbError> { let has_size: bool = conn .prepare(&format!("SELECT * FROM {table} LIMIT 0")) .map(|stmt| stmt.column_names().contains(&"size")) .unwrap_or(true); if has_size { return Ok(()); } // SQLite cannot add a column to a primary key, so the table is rebuilt. let tx = conn.unchecked_transaction()?; tx.execute_batch(&format!( "ALTER TABLE {table} RENAME TO {table}_old; {} INSERT INTO {table} SELECT file_id, 0, {cols} FROM {table}_old; DROP TABLE {table}_old;", if table == "entries" { INDEX_ENTRIES_TABLE } else { SHARD_SCHEMA }, cols = if table == "entries" { "shard, bytes" } else { "width, height, bytes" } ))?; tx.commit()?; Ok(()) } /// The `entries` table alone, for the rebuild above. const INDEX_ENTRIES_TABLE: &str = r#" CREATE TABLE entries ( file_id INTEGER NOT NULL, size INTEGER NOT NULL DEFAULT 0, shard INTEGER NOT NULL, bytes INTEGER NOT NULL, PRIMARY KEY(file_id, size) ); "#; #[cfg(test)] mod tests { use super::*; fn store() -> (ThumbStore, PathBuf) { let dir = std::env::temp_dir().join(format!( "dr-thumbs-{}-{:?}", std::process::id(), std::thread::current().id() )); let _ = std::fs::remove_dir_all(&dir); (ThumbStore::open(&dir).unwrap(), dir) } fn thumb(size: usize) -> Thumbnail { Thumbnail { width: 256, height: 170, bytes: vec![0xAB; size], } } #[test] fn a_snapshot_carries_what_the_shard_file_does_not_yet() { // A thumbnail worker holding the shard open keeps the log from being // checkpointed; the file on disk is then not the database. The // upload used to read that file. let (mut s, _d) = store(); s.put(1, ThumbSize::Grid, &thumb(1024)).unwrap(); let path = s.shard_path(0); let worker = Connection::open(&path).unwrap(); worker.pragma_update(None, "journal_mode", "WAL").unwrap(); s.put(2, ThumbSize::Grid, &thumb(2048)).unwrap(); s.put(3, ThumbSize::Grid, &thumb(2048)).unwrap(); // What a byte-for-byte reader sees is at most what the last // checkpoint left; the log is the part a copy misses. let copy = _d.join("copied.sqlite"); std::fs::copy(&path, ©).unwrap(); let file_rows = Connection::open(©) .ok() .and_then(|c| { c.query_row("SELECT count(*) FROM thumbs", [], |r| r.get::<_, i64>(0)) .ok() }) .unwrap_or(0); let snap = _d.join("upload.sqlite"); s.snapshot_shard(0, &snap).unwrap(); let snapshot = Connection::open(&snap).unwrap(); let rows: i64 = snapshot .query_row("SELECT count(*) FROM thumbs", [], |r| r.get(0)) .unwrap(); assert_eq!(rows, 3, "the snapshot is the whole shard"); assert!( file_rows < 3, "the file alone lagged ({file_rows} rows), which is what the snapshot exists for" ); drop(worker); } #[test] fn a_forgotten_thumbnail_is_no_longer_served() { // The point of the whole method: a purged photograph must not keep a // preview, on this device or any that syncs the shards. let (mut s, _d) = store(); s.put(1, ThumbSize::Grid, &thumb(1024)).unwrap(); s.put(2, ThumbSize::Grid, &thumb(1024)).unwrap(); assert_eq!(s.forget(&[1]).unwrap(), 1); assert!(s.get(1, ThumbSize::Grid).unwrap().is_none()); assert!(!s.contains(1, ThumbSize::Grid)); // And its neighbour is untouched. assert!(s.get(2, ThumbSize::Grid).unwrap().is_some()); assert_eq!(s.len(), 1); } #[test] fn forgetting_reclaims_bytes_in_the_open_shard() { // Otherwise the shard seals on space nothing is using, and the store // grows a shard per purge. let (mut s, _d) = store(); s.put(1, ThumbSize::Grid, &thumb(4096)).unwrap(); let before = s.shards().unwrap()[0].bytes; s.forget(&[1]).unwrap(); let after = s.shards().unwrap()[0].bytes; assert!(after < before, "{after} should be below {before}"); } #[test] fn forgetting_from_a_sealed_shard_leaves_it_byte_identical() { // The invariant the store's sync design rests on: a sealed shard never // changes, or every client re-downloads 25 MB to reclaim 20 KB. let (mut s, _d) = store(); let big = (SHARD_MAX_BYTES / 4) as usize; for id in 0..5 { s.put(id, ThumbSize::Grid, &thumb(big)).unwrap(); } let shards = s.shards().unwrap(); assert!(shards[0].sealed, "precondition: shard 0 is sealed"); let sealed_path = s.shard_path(0); let before = std::fs::metadata(&sealed_path).unwrap().len(); let before_bytes = shards[0].bytes; // Image 0 is in the sealed shard. assert_eq!(s.forget(&[0]).unwrap(), 1); // Unreachable through the index, which is what matters to a reader... assert!(s.get(0, ThumbSize::Grid).unwrap().is_none()); // ...but the file itself is untouched. assert_eq!( std::fs::metadata(&sealed_path).unwrap().len(), before, "a sealed shard must not be rewritten" ); // And its accounting is left alone, so it cannot un-seal. let after = s.shards().unwrap(); assert_eq!(after[0].bytes, before_bytes); assert!(after[0].sealed); } #[test] fn forgetting_something_never_stored_is_not_an_error() { // A purge runs over whatever the catalog knew; most images never had a // thumbnail fetched. let (mut s, _d) = store(); s.put(1, ThumbSize::Grid, &thumb(512)).unwrap(); assert_eq!(s.forget(&[42, 43]).unwrap(), 0); assert_eq!(s.forget(&[]).unwrap(), 0); assert_eq!(s.len(), 1, "nothing else went"); } #[test] fn forgetting_twice_is_idempotent() { // An empty-trash retried after a partial failure runs over the same ids. let (mut s, _d) = store(); s.put(1, ThumbSize::Grid, &thumb(512)).unwrap(); assert_eq!(s.forget(&[1]).unwrap(), 1); assert_eq!(s.forget(&[1]).unwrap(), 0); } #[test] fn a_forgotten_id_can_be_stored_again() { // Restoring from the server's own trashbin, or re-adding the same file: // the id is stable, so the store must accept it back. let (mut s, _d) = store(); s.put(1, ThumbSize::Grid, &thumb(512)).unwrap(); s.forget(&[1]).unwrap(); s.put(1, ThumbSize::Grid, &thumb(512)).unwrap(); assert!(s.get(1, ThumbSize::Grid).unwrap().is_some()); } #[test] fn the_two_size_classes_coexist() { // Adding the large class must not evict or shadow the grid one: the // same photograph is legitimately stored at both. let (mut s, _d) = store(); s.put(7, ThumbSize::Grid, &thumb(100)).unwrap(); s.put(7, ThumbSize::Large, &thumb(900)).unwrap(); assert_eq!(s.get(7, ThumbSize::Grid).unwrap().unwrap().bytes.len(), 100); assert_eq!( s.get(7, ThumbSize::Large).unwrap().unwrap().bytes.len(), 900 ); assert_eq!(s.len(), 2, "counted separately"); } #[test] fn a_missing_large_is_not_satisfied_by_the_grid_one() { // Otherwise a zoomed cell would silently show a 256px thumbnail // upscaled, which is the softness the large class exists to avoid. let (mut s, _d) = store(); s.put(7, ThumbSize::Grid, &thumb(100)).unwrap(); assert!(s.contains(7, ThumbSize::Grid)); assert!(!s.contains(7, ThumbSize::Large)); assert!(s.get(7, ThumbSize::Large).unwrap().is_none()); assert_eq!(s.missing(&[7], ThumbSize::Large), vec![7]); } #[test] fn forgetting_an_image_drops_every_size() { // A purged photograph must leave no preview at any size — the shards // sync, so a survivor keeps appearing on every other client. let (mut s, _d) = store(); s.put(7, ThumbSize::Grid, &thumb(100)).unwrap(); s.put(7, ThumbSize::Large, &thumb(900)).unwrap(); assert_eq!(s.forget(&[7]).unwrap(), 2, "both entries counted"); assert!(!s.contains(7, ThumbSize::Grid)); assert!(!s.contains(7, ThumbSize::Large)); assert_eq!(s.len(), 0); } #[test] fn the_cell_size_picks_the_class() { assert_eq!(ThumbSize::for_cell(180), ThumbSize::Grid); assert_eq!(ThumbSize::for_cell(256), ThumbSize::Grid); // Past the grid class's own edge, upscaling would show. assert_eq!(ThumbSize::for_cell(257), ThumbSize::Large); assert_eq!(ThumbSize::for_cell(400), ThumbSize::Large); } #[test] fn a_shard_written_before_the_size_class_accepts_new_thumbnails() { // The index is not the only table with a `size` column: every shard // carries its own `thumbs`. Migrating the index alone left existing // shards in the old shape, and `CREATE TABLE IF NOT EXISTS` will not // fix one — so every write failed with "no column named size" and the // store silently stopped accepting thumbnails. let dir = std::env::temp_dir().join(format!("dr-thumbs-shardmig-{}", std::process::id())); let _ = std::fs::remove_dir_all(&dir); std::fs::create_dir_all(&dir).unwrap(); // A shard in the old shape, holding one thumbnail. { let c = Connection::open(dir.join("shard-0000.sqlite")).unwrap(); c.execute_batch( "CREATE TABLE thumbs ( file_id INTEGER PRIMARY KEY, width INTEGER NOT NULL, height INTEGER NOT NULL, bytes BLOB NOT NULL); INSERT INTO thumbs VALUES (42, 8, 8, x'FFD8FFD9');", ) .unwrap(); } { let c = Connection::open(dir.join("index.sqlite")).unwrap(); c.execute_batch( "CREATE TABLE entries ( file_id INTEGER PRIMARY KEY, shard INTEGER NOT NULL, bytes INTEGER NOT NULL); CREATE TABLE shards ( id INTEGER PRIMARY KEY, bytes INTEGER NOT NULL DEFAULT 0, sealed INTEGER NOT NULL DEFAULT 0); INSERT INTO shards(id, bytes, sealed) VALUES (0, 4, 0); INSERT INTO entries(file_id, shard, bytes) VALUES (42, 0, 4);", ) .unwrap(); } let mut s = ThumbStore::open(&dir).unwrap(); // The old thumbnail survives as grid-sized... assert!(s.get(42, ThumbSize::Grid).unwrap().is_some()); // ...and the shard now accepts new writes rather than rejecting them. s.put(43, ThumbSize::Grid, &thumb(64)) .expect("a migrated shard must accept writes"); assert!(s.contains(43, ThumbSize::Grid)); } #[test] fn a_store_written_before_the_size_class_keeps_its_thumbnails() { // The migration case: an existing library must not lose the thumbnails // it already paid to fetch, and its entries are all grid-sized. let dir = std::env::temp_dir().join(format!("dr-thumbs-migrate-{}", std::process::id())); let _ = std::fs::remove_dir_all(&dir); std::fs::create_dir_all(&dir).unwrap(); // An index in the old shape: no `size`, `file_id` alone as the key. { let c = Connection::open(dir.join("index.sqlite")).unwrap(); c.execute_batch( "CREATE TABLE entries ( file_id INTEGER PRIMARY KEY, shard INTEGER NOT NULL, bytes INTEGER NOT NULL); CREATE TABLE shards ( id INTEGER PRIMARY KEY, bytes INTEGER NOT NULL DEFAULT 0, sealed INTEGER NOT NULL DEFAULT 0); INSERT INTO shards(id, bytes, sealed) VALUES (0, 64, 0); INSERT INTO entries(file_id, shard, bytes) VALUES (42, 0, 64);", ) .unwrap(); } let s = ThumbStore::open(&dir).unwrap(); assert!( s.contains(42, ThumbSize::Grid), "an existing entry survives as grid-sized" ); assert_eq!(s.len(), 1); } #[test] fn a_stored_thumbnail_round_trips() { let (mut s, _d) = store(); s.put(1001, ThumbSize::Grid, &thumb(1024)).unwrap(); let got = s.get(1001, ThumbSize::Grid).unwrap().expect("stored"); assert_eq!(got.width, 256); assert_eq!(got.bytes.len(), 1024); } #[test] fn an_unknown_id_is_none_not_an_error() { let (s, _d) = store(); assert!(s.get(9999, ThumbSize::Grid).unwrap().is_none()); } #[test] fn contains_avoids_reading_the_blob() { let (mut s, _d) = store(); s.put(1, ThumbSize::Grid, &thumb(512)).unwrap(); assert!(s.contains(1, ThumbSize::Grid)); assert!(!s.contains(2, ThumbSize::Grid)); } #[test] fn missing_reports_only_what_is_absent() { let (mut s, _d) = store(); s.put(1, ThumbSize::Grid, &thumb(64)).unwrap(); s.put(3, ThumbSize::Grid, &thumb(64)).unwrap(); assert_eq!(s.missing(&[1, 2, 3, 4], ThumbSize::Grid), vec![2, 4]); } #[test] fn everything_small_lands_in_one_shard() { let (mut s, _d) = store(); for id in 0..20 { s.put(id, ThumbSize::Grid, &thumb(1024)).unwrap(); } assert_eq!(s.shards().unwrap().len(), 1); assert_eq!(s.len(), 20); } #[test] fn a_full_shard_seals_and_the_next_one_opens() { let (mut s, _d) = store(); // Quarter-cap blobs: the fifth cannot fit alongside four others. let big = (SHARD_MAX_BYTES / 4) as usize; for id in 0..5 { s.put(id, ThumbSize::Grid, &thumb(big)).unwrap(); } let shards = s.shards().unwrap(); assert_eq!(shards.len(), 2, "should have rolled over"); assert!(shards[0].sealed, "the full shard is sealed"); assert!(!shards[1].sealed, "the new one is open"); } #[test] fn a_sealed_shard_never_reopens() { // The property sync depends on: once a client has a sealed shard it // never needs to ask again. Reopening one would force every client to // re-download it. let (mut s, _d) = store(); let big = (SHARD_MAX_BYTES / 4) as usize; for id in 0..5 { s.put(id, ThumbSize::Grid, &thumb(big)).unwrap(); } // Small enough to fit in the sealed shard, but it must not go there. s.put(100, ThumbSize::Grid, &thumb(16)).unwrap(); let shards = s.shards().unwrap(); assert!(shards[0].sealed); assert_eq!( s.shard_of(100, ThumbSize::Grid), Some(1), "new writes go to the open shard, not back into a sealed one" ); } #[test] fn rewriting_updates_in_place_rather_than_migrating() { // Moving an entry to the active shard would rewrite a sealed shard, // which every other client has already synced. let (mut s, _d) = store(); let big = (SHARD_MAX_BYTES / 4) as usize; for id in 0..5 { s.put(id, ThumbSize::Grid, &thumb(big)).unwrap(); } let original = s.shard_of(0, ThumbSize::Grid).unwrap(); s.put(0, ThumbSize::Grid, &thumb(32)).unwrap(); assert_eq!( s.shard_of(0, ThumbSize::Grid), Some(original), "must stay put" ); assert_eq!( s.get(0, ThumbSize::Grid).unwrap().unwrap().bytes.len(), 32, "but update" ); } #[test] fn shard_ids_are_zero_padded_for_stable_ordering() { let (s, _d) = store(); let p = s.shard_path(7); assert!(p.to_string_lossy().ends_with("shard-0007.sqlite")); } #[test] fn every_store_gets_its_own_stable_client_id() { let (mine, dir) = store(); let id = mine.client_id().to_string(); assert!(!id.is_empty()); drop(mine); // Stable across reopen, or a client would orphan its own uploads and // re-download them as if they were a peer's. assert_eq!(ThumbStore::open(&dir).unwrap().client_id(), id); // Distinct per store, which is the property the remote naming rests // on: two devices both filling shard 0 must not name one file. let (theirs, _d) = store(); assert_ne!(theirs.client_id(), id); } #[test] fn the_adoption_ledger_remembers_a_merged_shard_by_size() { let (mine, dir) = store(); assert_eq!(mine.adopted("shard-abc-0000.sqlite"), None); mine.record_adopted("shard-abc-0000.sqlite", 1234).unwrap(); assert_eq!(mine.adopted("shard-abc-0000.sqlite"), Some(1234)); drop(mine); // Survives reopen: the point is to not re-download across sessions. let reopened = ThumbStore::open(&dir).unwrap(); assert_eq!(reopened.adopted("shard-abc-0000.sqlite"), Some(1234)); // A peer's open shard grows, and the new size is what says so. reopened .record_adopted("shard-abc-0000.sqlite", 5678) .unwrap(); assert_eq!(reopened.adopted("shard-abc-0000.sqlite"), Some(5678)); } #[test] fn reopening_finds_what_was_stored() { let (mut s, dir) = store(); s.put(42, ThumbSize::Grid, &thumb(256)).unwrap(); drop(s); let reopened = ThumbStore::open(&dir).unwrap(); assert!(reopened.contains(42, ThumbSize::Grid)); assert_eq!(reopened.len(), 1); } #[test] fn merging_another_clients_shard_adopts_only_what_is_new() { let (mut mine, _d1) = store(); mine.put(1, ThumbSize::Grid, &thumb(64)).unwrap(); // A second store standing in for another device's downloaded shard. let other_dir = std::env::temp_dir().join(format!("dr-thumbs-other-{}", std::process::id())); let _ = std::fs::remove_dir_all(&other_dir); let mut theirs = ThumbStore::open(&other_dir).unwrap(); theirs.put(1, ThumbSize::Grid, &thumb(999)).unwrap(); // we already have this one theirs.put(2, ThumbSize::Grid, &thumb(64)).unwrap(); theirs.put(3, ThumbSize::Grid, &thumb(64)).unwrap(); let adopted = mine.merge_shard(&theirs.shard_path(0)).unwrap(); assert_eq!(adopted, 2, "only the two we lacked"); assert_eq!( mine.get(1, ThumbSize::Grid).unwrap().unwrap().bytes.len(), 64, "ours is kept, not overwritten" ); assert!(mine.contains(2, ThumbSize::Grid) && mine.contains(3, ThumbSize::Grid)); } #[test] fn merging_is_idempotent() { let (mut mine, _d1) = store(); let other_dir = std::env::temp_dir().join(format!("dr-thumbs-idem-{}", std::process::id())); let _ = std::fs::remove_dir_all(&other_dir); let mut theirs = ThumbStore::open(&other_dir).unwrap(); theirs.put(10, ThumbSize::Grid, &thumb(64)).unwrap(); assert_eq!(mine.merge_shard(&theirs.shard_path(0)).unwrap(), 1); assert_eq!( mine.merge_shard(&theirs.shard_path(0)).unwrap(), 0, "a repeated merge adopts nothing" ); } #[test] fn a_missing_shard_errors_rather_than_panicking() { let (s, _d) = store(); assert!(matches!( s.open_shard(99, false), Err(ThumbError::MissingShard(99)) )); } }