//! TRACES: FR-CAT-2 | NFR-R5 //! Schema definition and forward-only migrations. //! //! The catalog is an *index*, not a source of truth (ARCH §6.12) — it is //! deletable and rebuildable from sources plus sidecars. That is what makes //! migration failure survivable, and why the recovery path is the normal //! mechanism rather than a last resort. //! //! Migrations are forward-only, transactional, and idempotent on retry //! (NFR-R5). The app refuses to open a catalog newer than it understands //! rather than corrupting it. use rusqlite::Connection; use crate::error::CatalogError; /// Schema version this build writes and understands. pub const SCHEMA_VERSION: i64 = 20; /// Apply migrations up to [`SCHEMA_VERSION`]. /// /// Returns the version migrated from, so callers can log or back up before a /// real migration (NFR-R2 requires a backup before schema change). pub fn migrate(conn: &Connection) -> Result { let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?; if from > SCHEMA_VERSION { return Err(CatalogError::SchemaTooNew { found: from, supported: SCHEMA_VERSION, }); } if from == SCHEMA_VERSION { return Ok(from); } // Each step runs in its own transaction so a failure leaves the catalog // at a coherent version rather than half-migrated. if from < 1 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V1)?; tx.pragma_update(None, "user_version", 1)?; tx.commit()?; } if from < 2 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V2)?; tx.pragma_update(None, "user_version", 2)?; tx.commit()?; } if from < 3 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V3)?; tx.pragma_update(None, "user_version", 3)?; tx.commit()?; } if from < 4 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V4)?; tx.pragma_update(None, "user_version", 4)?; tx.commit()?; } if from < 5 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V5)?; tx.pragma_update(None, "user_version", 5)?; tx.commit()?; } if from < 6 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V6)?; tx.pragma_update(None, "user_version", 6)?; tx.commit()?; } if from < 7 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V7)?; tx.pragma_update(None, "user_version", 7)?; tx.commit()?; } if from < 8 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V8)?; tx.pragma_update(None, "user_version", 8)?; tx.commit()?; } if from < 9 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V9)?; tx.pragma_update(None, "user_version", 9)?; tx.commit()?; } if from < 10 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V10)?; tx.pragma_update(None, "user_version", 10)?; tx.commit()?; } if from < 11 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V11)?; tx.pragma_update(None, "user_version", 11)?; tx.commit()?; } if from < 12 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V12)?; tx.pragma_update(None, "user_version", 12)?; tx.commit()?; } if from < 13 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V13)?; tx.pragma_update(None, "user_version", 13)?; tx.commit()?; } if from < 14 { let tx = conn.unchecked_transaction()?; // `ALTER TABLE ... ADD COLUMN` has no `IF NOT EXISTS`, and NFR-R5 // wants this re-enterable: a catalog whose `user_version` was rewound // by a rollback already has the column, and would otherwise fail its // next open on it. let has_quality: bool = tx .prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = 'quality'")? .exists([])?; if !has_quality { tx.execute_batch("ALTER TABLE faces ADD COLUMN quality REAL;")?; } tx.execute_batch(V14)?; tx.pragma_update(None, "user_version", 14)?; tx.commit()?; } if from < 15 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V15)?; tx.pragma_update(None, "user_version", 15)?; tx.commit()?; } if from < 16 { let tx = conn.unchecked_transaction()?; // Guarded like V14's column, and for the same reason: `ALTER TABLE // ... ADD COLUMN` has no `IF NOT EXISTS`, and this step must be // re-enterable (NFR-R5). for column in EYE_COLUMNS { let present: bool = tx .prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = ?1")? .exists([column])?; if !present { tx.execute_batch(&format!("ALTER TABLE faces ADD COLUMN {column} REAL;"))?; } } tx.pragma_update(None, "user_version", 16)?; tx.commit()?; } if from < 17 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V17)?; tx.pragma_update(None, "user_version", 17)?; tx.commit()?; } if from < 18 { let tx = conn.unchecked_transaction()?; // Guarded like V14's and V16's columns: ALTER has no IF NOT EXISTS // and the step must be re-enterable (NFR-R5). let present: bool = tx .prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = 'landmarks_dense'")? .exists([])?; if !present { tx.execute_batch("ALTER TABLE faces ADD COLUMN landmarks_dense BLOB;")?; } tx.pragma_update(None, "user_version", 18)?; tx.commit()?; } if from < 19 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V19)?; tx.pragma_update(None, "user_version", 19)?; tx.commit()?; } if from < 20 { let tx = conn.unchecked_transaction()?; v20_markers_name_the_detector_that_found_the_faces(&tx)?; tx.pragma_update(None, "user_version", 20)?; tx.commit()?; } Ok(from) } // V20 -- TRACES: FR-CAT-7 // // Run markers that named the wrong detector, put right. // // `faces::record_updates` -- the write behind the quality, eye and crop // passes -- re-marked an image under the pipeline the pass ran as, while // the faces it had updated kept the id of the detector that found them. // A marker of `scrfd_10g+w600k_mbf` over faces spelled `w600k_mbf` reads, // to every consumer, as the thorough detector having examined the image: // the upgrade repair skips it, and `face_shard::export_to_shards` selects // its faces by the marker's id, finds none, and tells every other device // that the thorough detector found nothing there. The desktop's shard index // held 54 such entries over photographs with named faces, and the tablet's // eye pass over faces it had adopted from the desktop had made 430 more. // // The write is fixed to keep the marker under the faces' own id. This puts // the markers already written right, with a fresh time so the export sends // each image again under an entry newer than the empty one -- which is what // `held_model` orders by. Where the right marker is still there beside the // wrong one (the old write inserted rather than replaced), the wrong one // goes and the right one is refreshed for the same reason: its entry in // the shards is older than the empty one, and a device that has neither // would take the empty one. An image V14 left with faces and no marker at // all is not touched: that state is the quality pass's cue, and the fixed // write marks it correctly when the pass reaches it. // // Restated in Rust rather than SQL because the embedder half of a pipeline // id is `faces::embedder_sql`, which this must agree with. fn v20_markers_name_the_detector_that_found_the_faces(tx: &Connection) -> Result<(), CatalogError> { let fi = crate::faces::embedder_sql("face_index.model_id"); let f = crate::faces::embedder_sql("f.model_id"); // A marker is wrong when the image holds faces of its embedder under // another id. First the wrong ones that sit beside a right one -- the // update below would collide with it -- then the rest are renamed. let wrong = format!( "EXISTS (SELECT 1 FROM faces f WHERE f.image_id = face_index.image_id AND {f} = {fi} AND f.model_id != face_index.model_id)" ); let found_by = format!( "(SELECT MIN(f.model_id) FROM faces f WHERE f.image_id = face_index.image_id AND {f} = {fi})" ); let now = crate::faces::now_secs(); tx.execute( &format!( "UPDATE face_index SET indexed_at = ?1 WHERE model_id = {found_by} AND EXISTS (SELECT 1 FROM face_index w WHERE w.image_id = face_index.image_id AND w.model_id != face_index.model_id AND {} = {fi})", crate::faces::embedder_sql("w.model_id") ), [now], )?; tx.execute( &format!( "DELETE FROM face_index WHERE {wrong} AND EXISTS (SELECT 1 FROM face_index o WHERE o.image_id = face_index.image_id AND o.model_id = {found_by})" ), [], )?; tx.execute( &format!( "UPDATE face_index SET model_id = {found_by}, faces_found = (SELECT COUNT(*) FROM faces f WHERE f.image_id = face_index.image_id AND {f} = {fi}), indexed_at = ?1 WHERE {wrong}" ), [now], )?; Ok(()) } /// The seven columns V16 adds to `faces`, in the order the readers name them. /// /// Named once because three places have to agree on them: this migration, /// [`for_attached`], and the face shard's own catch-up (`face_shard`). pub const EYE_COLUMNS: [&str; 7] = [ "eye_right", "eye_right_px", "eye_right_sharp", "eye_left", "eye_left_px", "eye_left_sharp", "sunglasses", ]; /// Recompute columns a migration added, for rows that predate it. /// /// A migration adds a column with a default; it cannot know what the value /// *should* be for the rows already present. Without a backfill those rows are /// silently partial — present, queryable, and wrong — which is worse than /// missing, because nothing signals that they need attention. /// /// Cheap enough to run on every open: each pass is one indexed UPDATE, and /// re-running it is a no-op once the values are already right. /// /// Returns how many rows each backfill touched, for logging. pub fn backfill(conn: &Connection) -> Result, CatalogError> { let mut out = Vec::new(); // v2: `shadowed_by`. A JPEG sitting beside a RAW of the same name is the // camera's own rendering of that frame, not a second photograph, so it is // hidden from the grid, the timeline and the sweep. let n = pair_raw_and_jpeg(conn)?; if n > 0 { out.push(("shadowed_by", n)); } // v3: every image needs a default version to carry its rating and flag. // Libraries scanned before ratings existed have images and no versions at // all, so there was nowhere for a judgement to go — see [`crate::rating`]. let n = crate::rating::ensure_default_versions(conn)?; if n > 0 { out.push(("default_versions", n)); } // TRACES: FR-NC-8 | FR-NC-9 // The uuid on those rows is the cross-device merge identity, and it used // to be generated rather than derived. This comment said so, and said it // as though generating it were the point — it was the bug. Two devices // minted different uuids for one photograph, so the sidecar they shared // grew a `default = 1` block each and neither ever saw the other's work. // // Runs after the pass above so a row created a moment ago is already // derived and matches nothing here. Ordering the other way would be // correct too, just wasteful. let n = crate::rating::align_default_version_uuids(conn)?; if n > 0 { out.push(("derived_version_uuids", n)); } // v6: a vocabulary row for every word some image already carries. // // Three ways a catalog arrives holding assignments with no term behind // them, and all three are normal rather than exceptional: a library // keyworded by a build that predates this table, a catalog rebuilt from // sidecars (which carry the word and not the identity), and an import from // Lightroom or darktable (FR-CAT-14). Without this the words are // searchable but absent from the vocabulary list, which reads as the // keywords having been lost. let n = crate::keywords::adopt_orphan_terms(conn)?; if n > 0 { out.push(("keyword_terms", n)); } Ok(out) } /// How long a connection waits for a writer to finish before giving up. /// /// TRACES: NFR-R1 /// SQLite's default is **zero** — the loser of a race gets `SQLITE_BUSY` at /// once rather than a turn — and WAL does not change that for two writers. One /// writer and many readers is the case WAL makes free; this is the other one, /// and this application has it constantly: the face sweep commits a batch while /// reclustering reads, the derived sync imports shards while the sweep writes. /// /// Without a timeout that contention was *lost work*, not a retry. A face /// sweep that had already paid for the detection and the embedding — the /// expensive part, seconds per image — threw the result away on /// `storing faces for 214: database is locked` and moved on, and both the /// desktop and the tablet logged runs of those on consecutive images. /// /// Ten seconds, matching the figure the job runner's tests already use for the /// same reason. It is far longer than any transaction here (a sweep batch is /// sub-second; the slowest is a WAL checkpoint of a 130 MB catalog), so in /// practice it is a bound on pathology rather than a wait anyone sits through. /// The tension with NFR-P9 is real but one-sided: a query on the UI thread /// would rather wait for its turn than fail, because the failure is what the /// user sees as "cannot open catalog". const BUSY_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10); /// Connection setup applied on every open, migration or not. /// /// WAL is required by NFR-R1: it survives power loss without corruption, and /// it lets a background job write while the grid reads. pub fn configure(conn: &Connection) -> Result<(), CatalogError> { // Before the pragmas, so that a connection racing a migration waits for it // rather than failing on the first statement it tries. conn.busy_timeout(BUSY_TIMEOUT)?; conn.pragma_update(None, "journal_mode", "WAL")?; // NORMAL rather than FULL: with WAL this is durable across process death // (which is what FR-PLAT-AND-3 cares about) and only risks the last // transaction on power loss. The catalog is rebuildable; the sidecars are // not, and they are written separately with their own fsync discipline. conn.pragma_update(None, "synchronous", "NORMAL")?; conn.pragma_update(None, "foreign_keys", true)?; // A scan touching thousands of rows is transient; let SQLite spill to // memory rather than materialising temp b-trees on disk. conn.pragma_update(None, "temp_store", "MEMORY")?; Ok(()) } /// The v1 schema rewritten to target an attached database. /// /// Needed because a downloaded remote catalog is `ATTACH`ed under its own /// schema name before merging, and tests build one from scratch. SQLite has no /// "create these tables over there" form, so the names are rewritten. /// /// The rewrite is textual and therefore only as good as the naming discipline /// in [`V1`]: every `CREATE TABLE`/`CREATE INDEX` must name its object /// unqualified, which they do. pub fn v1_for_attached(schema_name: &str) -> String { rewrite_for_attached(V1, schema_name) // REFERENCES within an attached schema resolve to that schema already, // so foreign keys need no rewriting — but the ON clause of an index // does, and `CREATE INDEX x.name ON table` is the correct form. } /// Every table this build knows about, rewritten to target an attached /// database. /// /// [`v1_for_attached`] is kept alongside this rather than replaced by it: a /// remote catalog written by an older build genuinely has only the v1 tables, /// and the merge has to keep working against one (see /// [`crate::merge::merge_keywords`]). Building that case in a test needs a way /// to say "v1 and no more". /// /// Only the migrations that *create* objects appear here. V2 through V5 are /// `ALTER TABLE ... ADD COLUMN`, and the columns they add are local index /// state — shadowing, trashing, cache pinning — that a merge never reads /// across the attachment. /// /// V7 creates an object and is still excluded, which is the one exception to /// that rule and not an oversight: it indexes `shadowed_by` and `trashed_at`, /// the very columns V2 through V5 add and this function leaves out, so /// creating it over there would fail on columns that are not there. Nothing is /// lost by its absence — it exists to make the *grid* page quickly, and the /// grid never reads across an attachment. /// /// V11 is excluded on the same grounds and for the plainer reason that a merge /// has nothing to do with it: burst grouping is rebuilt locally from local /// signatures, and no code reads a remote catalog's `burst_*` tables. Its /// `ALTER TABLE` would fail here anyway, being unqualifiable by the rewrite. pub fn for_attached(schema_name: &str) -> String { // V10 is `ALTER TABLE`, which the textual rewrite cannot qualify, so its // columns are spelled out. A remote genuinely older than V10 is a real // case and `merge::merge_people_within` probes for them; this is the // *current* shape, which is what the tests want. format!( "{}\n{}\n{}\n\ ALTER TABLE {schema_name}.people ADD COLUMN ignored INTEGER NOT NULL DEFAULT 0;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN crop BLOB;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN quality REAL;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN eye_right REAL;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN eye_right_px REAL;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN eye_right_sharp REAL;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN eye_left REAL;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN eye_left_px REAL;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN eye_left_sharp REAL;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN sunglasses REAL;\n\ ALTER TABLE {schema_name}.faces ADD COLUMN landmarks_dense BLOB;", rewrite_for_attached(V1, schema_name), rewrite_for_attached(V6, schema_name), rewrite_for_attached(V8, schema_name), ) } /// Qualify every object a `CREATE` statement names with `schema_name`. /// /// The rewrite is textual and therefore only as good as the naming discipline /// in the batches it is given: every `CREATE TABLE`/`CREATE INDEX` must name /// its object unqualified, which they do. fn rewrite_for_attached(sql: &str, schema_name: &str) -> String { sql.replace("CREATE TABLE ", &format!("CREATE TABLE {schema_name}.")) .replace("CREATE INDEX ", &format!("CREATE INDEX {schema_name}.")) .replace( "CREATE UNIQUE INDEX ", &format!("CREATE UNIQUE INDEX {schema_name}."), ) } /// Mark each JPEG that sits beside a RAW of the same name. /// /// Matched on folder plus stem, case-insensitively. Same-folder is what makes /// this safe: cameras write the pair side by side, and matching across folders /// would risk pairing unrelated frames, since camera filenames wrap at /// IMG_9999 (FR-CAT-11). /// /// Done in Rust rather than SQL because the comparison needs a filename stem, /// and SQLite has no such function without enabling `rusqlite/functions` — /// a dependency feature for one string operation, whose SQL spelling would be /// unreadable and would mishandle names with no extension. fn pair_raw_and_jpeg(conn: &Connection) -> Result { use std::collections::HashMap; // (folder, lowercase stem) -> RAW id, built in one pass over the RAWs. let mut raws: HashMap<(Option, String), i64> = HashMap::new(); { let mut stmt = conn.prepare( "SELECT id, folder_id, source_ref FROM images WHERE lower(format) IN ('cr2','cr3','nef','arw','raf','rw2','orf','dng')", )?; let rows = stmt.query_map([], |r| { Ok(( r.get::<_, i64>(0)?, r.get::<_, Option>(1)?, r.get::<_, String>(2)?, )) })?; for row in rows { let (id, folder, path) = row?; raws.insert((folder, stem_of(&path).to_ascii_lowercase()), id); } } if raws.is_empty() { return Ok(0); } let pairs: Vec<(i64, i64)> = { let mut stmt = conn.prepare( "SELECT id, folder_id, source_ref FROM images WHERE lower(format) IN ('jpg','jpeg') AND shadowed_by IS NULL", )?; let rows = stmt.query_map([], |r| { Ok(( r.get::<_, i64>(0)?, r.get::<_, Option>(1)?, r.get::<_, String>(2)?, )) })?; rows.filter_map(|row| { let (id, folder, path) = row.ok()?; let raw = raws.get(&(folder, stem_of(&path).to_ascii_lowercase()))?; Some((id, *raw)) }) .collect() }; let tx = conn.unchecked_transaction()?; for (jpeg, raw) in &pairs { tx.execute( "UPDATE images SET shadowed_by = ?2 WHERE id = ?1", [jpeg, raw], )?; } tx.commit()?; Ok(pairs.len()) } /// A filename without its extension. /// /// Only the final path component, and only its last dot — a directory /// containing a dot must not truncate the name. fn stem_of(path: &str) -> &str { let name = path.rsplit(['/', ':']).next().unwrap_or(path); match name.rsplit_once('.') { Some((stem, _)) if !stem.is_empty() => stem, _ => name, } } /// TRACES: FR-CAT-4 | NFR-P5 /// The order the grid reads in, as an index. /// /// # What this is for /// /// Every window the grid loads is `ORDER BY ... LIMIT n OFFSET k`, and without /// an index that matches the ordering SQLite answers it by sorting the whole /// library into a temp b-tree and then discarding the first `k` rows. Measured /// on 24,000 images at offset 20,000, one window read cost 15 ms — a frame /// budget of 16.7 ms, spent inside the scroll handler, several times per /// screenful. That is the jitter. /// /// With this index the same read is a walk along it: 0.36 ms. /// /// # Why the shape is what it is /// /// `captured_at IS NULL` leads, because [`crate::library`]'s `GRID_ORDER` does /// — undated images sort last, and an ordinary index on `captured_at` cannot /// answer that, since the expression is not a column. SQLite indexes /// expressions, so it is spelled out here exactly as the query spells it; the /// two must stay identical or the planner silently falls back to sorting and /// the cost comes back with no other symptom. /// /// `source_ref` is included because it breaks ties in the same ordering, and an /// index that stopped at `captured_at` would leave a sort for the ties. /// /// # Partial, on the same predicate the grid filters by /// /// The grid never lists shadowed or trashed rows, so an index carrying them /// would be larger than the question ever asks about, and — more to the point — /// a partial index is only usable when its `WHERE` is implied by the query's, /// which is what makes this one apply to the grid's reads and to nothing else. /// /// # It does not cover the rating filter or a collection scope /// /// Both narrow the walk rather than reorder it, so the index still supplies the /// ordering and SQLite tests the extra predicate per row. That is the cheap /// direction: the expensive part was never the filtering, it was the sort. const V10: &str = r#" -- TRACES: FR-CULL-10 | FR-CULL-12 -- Two columns the People screen turned out to need, and neither is derivable. -- A person the user does not want to identify. -- -- Most of a real library's clusters are strangers: people in the background of -- a street, guests at somebody else's party, a face on a poster. They are -- correctly detected and correctly grouped, and the user will never name any of -- them -- but they crowd out the handful of groups that matter, and there is no -- way to tell "not yet looked at" from "looked at, don't care" without -- recording the second. -- -- **User data**, and the reason this is a column rather than a deletion: a -- deleted cluster comes straight back on the next Regroup, because the faces -- are still there and still similar. Nothing short of remembering the judgement -- survives re-clustering, which is the same argument `face_person_rejected` -- makes one level down (FR-CULL-12). ALTER TABLE people ADD COLUMN ignored INTEGER NOT NULL DEFAULT 0; -- The face, cut out and kept. -- -- A face used to be drawn by decoding the 1024px proxy it was found on and -- cutting the box out again, every time the screen opened. That made the People -- screen a *derivative of the thumbnail cache*: evict a proxy -- which the -- cache is entitled to do at any moment -- and the cell goes blank, with no way -- back short of re-fetching the original over the network and re-detecting it. -- It also cost one full JPEG decode per image per visit to show a 96px cell. -- -- So the crop is cut once, when the pixels are already in hand at detection -- time, and kept. Small: a 160px JPEG is a few KB, against ~250 KB for the -- proxy it replaces reading. -- -- Nullable, because a face indexed before this column existed has no crop and -- must still work -- the reader falls back to the old proxy path, and the next -- indexing pass fills it in. -- -- **Stripped from the sync snapshot.** The catalog is uploaded whole, so this -- would otherwise put tens of MB of JPEG on every sync; crops travel in the -- face shards instead, which is where the bulk per-face data already goes -- (`face_shard`). See `sync::snapshot_for_upload`. ALTER TABLE faces ADD COLUMN crop BLOB; "#; /// TRACES: FR-CULL-5 /// Burst grouping: which frames are one moment, and which one stands for it. /// /// The reasoning behind the grouping itself is in [`crate::bursts`]; what /// belongs here is why it is stored in three pieces rather than one. /// /// **`images.perceptual_hash` is a column, not a table**, for the same reason /// `content_hash` is: it is one number per image, NULL until something has had /// the pixels in hand, and every query that wants it is already reading the /// image row. It is local derived state — a rebuilt catalog recomputes it from /// thumbnails — which is also why it is absent from [`for_attached`], alongside /// the shadowing and trashing columns V2 through V5 add. /// /// **`burst_members` is rewritten whole by every pass.** No id of its own: the /// group is named by the image id of its earliest frame, so a burst that has not /// changed keeps its name across a regroup and the interface can remember that /// this one is open. There is no `bursts` table to go with it because a group /// has no properties beyond its members — inventing a row for it would create an /// identity that survives the grouping being rebuilt, which is precisely what /// must not happen. /// /// **`burst_pick` is the one thing here that is not derived**, and it is a /// separate table so that rewriting the grouping cannot erase it. A /// representative stored on `burst_members` would be forgotten every time a /// frame arrived; the user would be asked the same question after every scan. /// The same argument `people.ignored` makes in V10, one subsystem over. /// /// **`burst_expanded` is view state in the catalog**, which is unusual enough to /// justify. The grid is a window over an ordered query — `LIMIT n OFFSET k` — /// so what a collapsed burst hides has to be decided by the query, or the row /// count stops agreeing with the scrollbar and the ordinals a scrub resolves to. /// Once SQL has to see it, this is where it lives. Nothing else reads it, and it /// is emptied of stale groups by every pass. const V11: &str = r#" -- A 64-bit perceptual signature. Local derived state: NULL until something has -- decoded the image, recomputed from thumbnails if the catalog is rebuilt, and -- comparable only to signatures produced by the same build (`bursts`). ALTER TABLE images ADD COLUMN perceptual_hash INTEGER; CREATE TABLE burst_members ( -- One burst at most per image: a frame belongs to the moment it was taken -- in, and nothing else. image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE, -- The image id of the burst's earliest frame. Not a foreign key by -- accident: the leader is itself a member, so this genuinely references -- images(id), and cascading its deletion is right. burst_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE, -- The frame the group collapses to. Exactly one per burst. representative INTEGER NOT NULL DEFAULT 0 ); -- Counting a burst's frames and listing them are what the grid asks for, once -- per window; without this both are a scan of every grouped frame in the -- library. CREATE INDEX burst_members_burst ON burst_members(burst_id); -- The user's own choice of representative. User data, never rewritten by a -- grouping pass -- see the module doc above. CREATE TABLE burst_pick ( image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE ); -- Bursts the grid is currently showing in full. CREATE TABLE burst_expanded ( burst_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE ); "#; const V12: &str = r#" -- TRACES: FR-CULL-8 -- Forget the runs that were made against a proxy too small to find a face on. -- -- Detection used to accept any proxy, and one of the two sweeps detected on -- the stored 1024px tier. On the reference library that produced 0.078 faces -- per image against 1.82 for the same photographs at 2048 or better -- and -- every one of those runs left a `face_index` row behind saying the image had -- been examined. That row is what makes the damage permanent: the work list is -- "images with no row", so a photograph examined badly is indistinguishable -- from one examined well, and is never offered to a later pass. -- -- Deleting the marker is the whole repair, and it is deliberately not a -- deletion of anything else. The `faces` rows those runs found stay exactly -- where they are and keep drawing the People screen until a better pass -- replaces them, and `record_detections` carries the user's confirmed names -- across that replacement by box overlap. So this costs a re-fetch of the -- affected images and loses no work the user has done. -- -- The threshold is written out rather than taken from `dr_face::MIN_CROP_EDGE` -- on purpose. A migration has to keep meaning what it meant on the day it ran; -- binding it to a constant someone may raise later would silently change what -- an old catalog gets migrated to. DELETE FROM face_index WHERE source_edge <= 1024; "#; const V13: &str = r#" -- TRACES: FR-CAT-8 | FR-NC-9 -- Which sidecars this device has read, and at what ETag. -- -- The sidecar is the authoritative store for a rating and an edit, and until -- this table existed nothing ever read one back into the catalog: judgements -- travelled outward only. A cull done on a tablet reached the server and -- stopped there, because the scan indexes photographs, the derived sync moves -- thumbnails and collections, and the one reader that existed ran when a single -- photograph was opened in develop and fed only the develop graph. The grid -- draws `versions.rating`, so another device's afternoon of culling was -- invisible on this one -- permanently, by every path the app had. -- -- What this holds is the ETag, not the content. It is the record of what has -- already been taken in, so a pull fetches only what changed: `dr_sync::scan` -- reports every sidecar it saw in listings it was making anyway, and this -- decides which of them are worth a GET. -- -- Keyed on the sidecar's own remote path rather than on an image id. One -- sidecar can describe two images -- a RAW and the JPEG beside it are one -- photograph (FR-CAT-11) and share a document -- and a path is what the scan -- reports and what a fetch addresses, so keying on anything else would mean -- deriving one from the other in two places. -- -- Rebuildable like the rest of the catalog: losing this table costs one pass -- that re-reads every sidecar and reaches exactly the same state. -- -- `IF NOT EXISTS` because NFR-R5 asks for migrations that are idempotent on -- retry, and this one can genuinely be re-entered: a catalog whose -- `user_version` was rewound -- by a rollback to an older build, or by a -- recovery -- would otherwise fail its next open on a table it already has. CREATE TABLE IF NOT EXISTS sidecars ( root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE, path TEXT NOT NULL, etag TEXT, -- Unix seconds, for diagnosing a pull that is not making progress. read_at INTEGER NOT NULL DEFAULT 0, PRIMARY KEY(root_id, path) ); "#; const V14: &str = r#" -- TRACES: FR-CULL-9 | FR-CULL-10 -- How recognisable the model found each face, and a second look at the faces -- it was never asked about. -- -- The embedder's raw output has a length, and the length is a quality -- reading: it grows with how much of a face the model could make out, and a -- blur, an occlusion or a hard profile comes out short (dr_face::embedding, -- `MIN_GALLERY_QUALITY`). Normalising threw it away. A short vector sits -- near the middle of the sphere and matches a little of everyone, which is -- how one bad crop bridges two people in a grouping pass -- so a face below -- the floor is compared against the others and never compared *against*. -- -- Nullable, and NULL means "never measured": every face indexed before this -- version stored the unit vector, whose length is one whatever the crop was. -- A face with no reading is admitted to the gallery, because a rule that -- cannot be checked should admit rather than exclude -- but it is also a -- face this rule is not yet protecting anyone from, and the only way to -- measure it is to embed it again. -- -- The `face-quality` repair is what does that (`dr_ui::repairs`, once the -- sweep's measuring pass): it lists every face with no reading, and each is -- embedded again from the native render with the landmarks it already has, -- the raw vector written over the old one (`record_updates`) and nothing -- else touched -- not the id, not the box, not who the user said it was. -- The faces keep drawing the People screen throughout. -- -- The run markers of those images are forgotten too, exactly as V12 forgot -- the runs made against too small a proxy. The build this shipped in had no -- measuring pass yet, and a marker is the one thing that stops a face ever -- being looked at again; with the repair in place, detection leaves an -- image holding this embedder's faces to it rather than detecting from -- scratch, so the deletion costs nothing -- and an image that was examined -- and found empty keeps its marker, since there is nothing on it to measure. -- -- The cost is a re-fetch of every image with a face on it, on the next pass -- the user starts. That is a whole-library transfer (FR-NC-6), and it starts -- when they say so, not here. -- -- From this version the `embedding` blob is the **raw** model output rather -- than the unit vector V8 describes -- the length is the quality, and a store -- that kept only the direction had thrown it away. Readers re-normalise on -- load, so a unit blob from before and a raw blob from now compare alike; -- `quality` is that length kept beside the blob for the readers that never -- load the vector, and NULL rather than 1.0 for the old rows, because a unit -- vector reads as a length of one and one is not "unmeasured". -- -- The column itself is added in `migrate`, guarded, because ALTER has no -- IF NOT EXISTS and this step has to be re-enterable (NFR-R5). DELETE FROM face_index WHERE EXISTS (SELECT 1 FROM faces f WHERE f.image_id = face_index.image_id AND f.model_id = face_index.model_id); "#; const V15: &str = r#" -- TRACES: FR-CAT-13 -- Where a standard XMP sidecar and the catalog disagree. -- -- An `.xmp` beside a photograph is read on the same pull as DarkRoom's own -- sidecar, and reconciled field by field (`dr_xmp::reconcile`): keywords -- union, and a rating, label or caption is taken only where the catalog holds -- none. That rule is the safe one and it is not always the right one -- a -- rating changed in Lightroom after it was changed here is a genuine -- disagreement, and a standard XMP carries no revision to settle it by. So -- the disagreement is written here instead of being resolved, and the -- requirement's "a metadata reload offered" is a row in this table with a -- button in front of it: the reload re-reads the file with the sidecar -- winning, and deletes the row. -- -- Keyed on the sidecar's path like `sidecars` is, and for the same reason: a -- path is what the scan reports, what a fetch addresses, and what the ETag -- that noticed the change belongs to. `fields` is the disagreeing fields as -- `dr_xmp` names them, space-separated, for the line the settings page shows. -- -- Rebuildable: the next pull that sees a changed ETag writes the row again. CREATE TABLE IF NOT EXISTS xmp_conflicts ( root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE, path TEXT NOT NULL, fields TEXT NOT NULL, seen_at INTEGER NOT NULL DEFAULT 0, PRIMARY KEY(root_id, path) ); "#; // V19 -- TRACES: NFR-P9 // // The indexes the repair counts are served from, and V17's lesson applied // to the rest of the face columns. // // "How many images still owe a quality reading" was answered per image: a // correlated EXISTS over `faces` that had to open each face's row to look // at one nullable column -- the row being eight kilobytes of embedding and // crop. Six such counts run every time the Identity screen opens and every // time a sweep ends, 160 ms of them on the reference library. Three // partial indexes hold only the faces still owing each pass, keyed by the // image and carrying the model id the predicate also reads, so the count // walks a few thousand index entries and touches no row at all -- and each // index shrinks to nothing as its pass completes. The planner takes them // when the count is driven from `faces` (`repairs::count`) and ignores // them inside the per-image EXISTS, which is why that function has two // spellings of the same predicate. // // `faces_image_model` replaces `faces_image`: the same key with the model // id beside it, so "does this image hold this embedder's faces" -- asked in // the audit, the proxy repair and the outstanding-detection count -- is an // index-only probe where it used to read the row for the model id. Every // lookup that used `faces_image` is served by its prefix. // // Not applied to attached catalogs, like V7 and V17: an index is a local // concern, and a merge never runs these queries across an attachment. const V19: &str = r#" CREATE INDEX IF NOT EXISTS faces_image_model ON faces(image_id, model_id); DROP INDEX IF EXISTS faces_image; CREATE INDEX IF NOT EXISTS faces_owed_quality ON faces(image_id, model_id) WHERE quality IS NULL; CREATE INDEX IF NOT EXISTS faces_owed_crop ON faces(image_id, model_id) WHERE crop IS NULL; CREATE INDEX IF NOT EXISTS faces_owed_eyes ON faces(image_id, model_id) WHERE eye_right IS NULL OR landmarks_dense IS NULL; "#; // V18 -- TRACES: FR-CULL-8a | FR-CULL-12 // // The 106 dense landmarks the eye pass reads its eye boxes from, kept beside // the reading as `dr_face::Landmarks::to_packed_bytes`: 106 x (x, y) as // 16-bit fixed point over the frame, 424 bytes a face, a seventh of a // pixel on a 6000-pixel frame. Derived data under FR-CULL-12 -- rebuilt by // re-reading, never in a sidecar -- and stored for the same reason the // embedding is: it cost a fetch of the original and a model run, and the // next per-face pass (head pose, expression) should not have to pay either // again. NULL where the face was never read. // // Added in `migrate`, guarded, like every ALTER here (NFR-R5). const V17: &str = r#" -- TRACES: FR-CULL-8a | FR-CULL-13 | NFR-P9 -- The eyes-open filter's index, and a lesson about where a column lands. -- -- The people filter is a correlated EXISTS over `faces` per image, and it -- was fast because `faces_image` *covers* it: the subquery never touched a -- row. Reading V16's seven eye columns in the same subquery did touch the -- row -- and `ALTER TABLE ADD COLUMN` puts a column at the end of the -- record, after the 1 KB embedding and the ~5 KB crop, so every check -- dragged six kilobytes off disk to reach seven floats. Measured on the -- reference library: 24 seconds for one count, thirteen of them system -- time. With this index the same count takes five milliseconds, because -- the subquery is served from the index again and never reads a row. -- -- The columns are listed in EYE_COLUMNS' order behind `image_id`, which is -- the key the subquery searches on. Nothing else changed in V17; a catalog -- already at V16 needs only this. CREATE INDEX IF NOT EXISTS faces_eyes ON faces( image_id, eye_right, eye_right_px, eye_right_sharp, eye_left, eye_left_px, eye_left_sharp, sunglasses ); "#; // V16 -- TRACES: FR-CULL-8a // // What each face's eyes are doing: for each eye P(open), the source pixels // across its box and the sharpness of the patch the classifier saw; and // P(sunglasses) for the head. Seven numbers rather than a verdict, because // the verdict is a rule with thresholds in it (dr_face::eyes::EyeReading:: // state) and a rule belongs in code that can be changed, not in rows that // would have to be re-measured. // // The pixels and the sharpness are what stop a smear reading as a blink: an // eye too small or too soft to read is not asked, and a face with no // readable eye is "unclear", which no filter drops. Sunglasses are a column // of their own for the same kind of reason — the eye classifier answers // confidently over dark glass, and its answer means nothing there. A filter // for "eyes open" reads all seven. // // NULL means "never measured" -- a face indexed before this version, or on a // device without the eye models -- and a NULL is left alone by every filter // that reads these, so an old library does not empty its grid the moment the // chip is pressed. The sweep's measuring pass fills them in, from the native // render, with the landmarks already stored: the same pass V14 built for the // embedding's length, extended to ask the eye models too. No run marker is // forgotten here, for the reason V14's note gives -- the measuring pass // finds its own work by the NULL, and deleting markers would only put the // detector back over images it has finished with. // // The columns are added in `migrate`, guarded, because ALTER has no IF NOT // EXISTS and the step has to be re-enterable (NFR-R5). Their names are // `EYE_COLUMNS`. const V9: &str = r#" -- TRACES: FR-CULL-8 -- A record that face detection has *run* on an image, distinct from what it -- found. -- -- # Why the faces table cannot answer this -- -- Without this, "has this image been indexed" is asked as "does it have any -- faces", and those are not the same question. **A photograph with no faces in -- it is indistinguishable from one that has never been looked at**, so every -- indexing pass re-examines every landscape, every still life and every -- document scan in the library, for ever. In a typical personal library that is -- most of it: the pass never converges, and the cost is paid again on every -- run rather than once. -- -- It also makes a coverage figure possible, which is the thing a user actually -- wants to see — "4,812 of 5,000 images indexed" — where counting face rows -- can only ever report how many faces exist. -- -- # Why it is keyed on the model -- -- Embeddings from different models are not comparable, so a model change has -- to re-index. Keying the marker on `(image_id, model_id)` makes that -- automatic: new model, no marker, image comes back into the queue. The id -- names the whole pipeline -- detector and embedder together -- because -- changing either changes what is found. CREATE TABLE face_index ( image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE, model_id TEXT NOT NULL, indexed_at INTEGER NOT NULL, -- Zero is a real and common answer, and recording it is the entire point. faces_found INTEGER NOT NULL, -- Long edge of the proxy this ran against. A face too small to detect at -- 1024 may be findable at 2048, so a library whose proxies grow can -- re-index the images that stand to gain instead of all of them. source_edge INTEGER NOT NULL, PRIMARY KEY (image_id, model_id) ); CREATE INDEX face_index_model ON face_index(model_id); "#; const V8: &str = r#" -- TRACES: FR-CULL-8 | FR-CULL-9 | FR-CULL-10 | FR-CULL-11 | FR-CULL-12 | NFR-SEC-5 -- People and faces (docs/faces.md, docs/catalog.md §10). -- -- Everything here is **derived data** except one column. Faces, landmarks, -- embeddings, cluster assignments and suggestions are all reproducible by -- re-indexing and are never written to a sidecar (FR-CULL-12); a person's -- *name*, once the user has confirmed it, is a human judgement of the same -- class as a rating and travels with the photograph. -- -- That asymmetry is the whole design: deleting the catalog costs an afternoon -- of re-indexing and loses nothing the user typed (ARCH §6.12). CREATE TABLE people ( id INTEGER PRIMARY KEY, -- Merge identity, not the name. Two devices that independently name the -- same cluster produce two people; merging them keys on this, exactly as -- collections do (FR-CAT-7, ARCH §6.3). uuid TEXT NOT NULL UNIQUE, name TEXT NOT NULL, -- Tombstone-by-redirect. A merged person must outlive its merge or a -- device that still holds it resurrects it on the next sync -- the same -- hazard collections have, solved the same way. merged_into INTEGER REFERENCES people(id) ON DELETE SET NULL, created INTEGER NOT NULL, revision INTEGER NOT NULL DEFAULT 1, modified INTEGER NOT NULL ); CREATE TABLE faces ( id INTEGER PRIMARY KEY, image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE, -- Normalised to the image's long edge, so a face survives the proxy it was -- found on being evicted and regenerated at another resolution. Storing -- pixels would bind a face to a resolution the cache is entitled to change. x REAL NOT NULL, y REAL NOT NULL, w REAL NOT NULL, h REAL NOT NULL, landmarks BLOB NOT NULL, -- 5 x (x, y) f32, normalised likewise detector_confidence REAL NOT NULL, embedding BLOB NOT NULL, -- 512 x f16; unit length until V14, raw since -- Source pixels across the aligned 112x112 crop (docs/faces.md §7). -- -- Not cosmetic: it is the honest quality signal for the UI, a feature in -- the §8 calibration -- FR-CULL-9 names face size as an axis along which an -- uncalibrated similarity misbehaves -- and the selector a later -- higher-resolution re-embedding pass would run on. crop_px REAL NOT NULL, -- Which model produced this embedding. -- -- The one mistake in this subsystem that yields plausible-looking garbage -- rather than an error: embeddings from different models are not -- comparable. Storing the model with the vector makes a model change -- detectable and re-indexable instead of quietly poisoning every -- similarity in the library. model_id TEXT NOT NULL, detected_at INTEGER NOT NULL ); CREATE INDEX faces_image ON faces(image_id); CREATE INDEX faces_model ON faces(model_id); CREATE TABLE face_person ( face_id INTEGER PRIMARY KEY REFERENCES faces(id) ON DELETE CASCADE, person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE, -- Calibrated P(this face is this person), never a raw cosine (FR-CULL-9). probability REAL NOT NULL, -- The user said so. Never overwritten by a later inference pass. -- -- A column rather than a probability of 1.0, because a confirmation is a -- different kind of fact from a confident guess and collapsing them loses -- the ability to recompute suggestions without touching user data. confirmed INTEGER NOT NULL DEFAULT 0 ); CREATE INDEX face_person_person ON face_person(person_id, confirmed); -- Faces the user has explicitly said are NOT a given person. -- -- Needed because rejection is not the absence of an assignment: without it, -- the next clustering pass re-suggests exactly the face the user just pushed -- away, and the tool feels broken. Same reasoning as `confirmed` -- a -- judgement is user data (FR-CULL-12) whichever direction it points. CREATE TABLE face_person_rejected ( face_id INTEGER NOT NULL REFERENCES faces(id) ON DELETE CASCADE, person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE, PRIMARY KEY (face_id, person_id) ); -- The FR-CULL-9 calibration, fitted from this library's own faces. -- -- One row per model, because the fit is a property of the embedding space and -- a library indexed across a model change holds two. `face_set_hash` is what -- makes a stale fit detectable: a materially changed library recomputes rather -- than trusting numbers derived from a set that no longer exists. CREATE TABLE face_calibration ( model_id TEXT PRIMARY KEY, -- P(same) = sigmoid(a*cos + b + w_size*log2(min(crop_px)) + log_prior_odds) a REAL NOT NULL, b REAL NOT NULL, w_size REAL NOT NULL DEFAULT 0.0, -- Whether the fit is usable at all. When it is not, the UI says the -- confidence is unavailable; it does not present an untuned default as -- though it were measured (FR-CULL-9). valid INTEGER NOT NULL DEFAULT 0, positive_pairs INTEGER NOT NULL DEFAULT 0, negative_pairs INTEGER NOT NULL DEFAULT 0, face_set_hash TEXT NOT NULL, fitted_at INTEGER NOT NULL ); "#; const V7: &str = r#" CREATE INDEX images_grid_order ON images(captured_at IS NULL, captured_at, source_ref) WHERE shadowed_by IS NULL AND trashed_at IS NULL; "#; const V6: &str = r#" -- TRACES: FR-CAT-5 | FR-CAT-6 | FR-NC-9 -- Keywords gain an identity, so that renaming and deleting one can cross -- between devices. -- -- The v1 `keywords` table is the *assignment*: one row per (version, word), -- and the word is stored as text. That stays exactly as it is, and this -- migration adds nothing to it, for a reason that is easy to get backwards. -- -- # Why assignments keep the text rather than pointing at a row here -- -- The catalog is a rebuildable index (ARCH §6.12). What an image is keyworded -- with is authoritative in the sidecar and in XMP `dc:subject` (FR-CAT-13), -- and both of those carry a *string*. Rewriting the join to reference -- `keyword_terms(id)` would mean a catalog rebuilt from sidecars had to invent -- term rows before it could record a single assignment, and an integer that -- means nothing on the other device would sit where the durable fact belongs. -- It would also break `crate::query`, which matches `kw.keyword` directly and -- must keep hitting `keywords_term` on a 50k library (FR-CAT-6). -- -- So the text is the fact and this table is the *identity*: it exists to give -- a rename and a deletion something a merge can key on, and to let a keyword -- exist in the vocabulary before any photograph carries it. CREATE TABLE keyword_terms ( id INTEGER PRIMARY KEY, -- Device-independent identity, as `collections.uuid` is. The integer id is -- local and collides across devices. uuid TEXT NOT NULL UNIQUE, -- The word itself, and the value written into every assignment row. name TEXT NOT NULL, created INTEGER NOT NULL, -- Monotonic, bumped on every local edit. `crate::merge` compares these -- rather than timestamps, so a clock-skewed device cannot silently win. revision INTEGER NOT NULL DEFAULT 1, modified INTEGER NOT NULL, -- Tombstone, so a merge against a device that still holds the keyword does -- not resurrect it. deleted INTEGER NOT NULL DEFAULT 0 ); -- Deliberately **not** UNIQUE. -- -- Two devices that each type "Iceland" create two rows with two uuids, and -- both are correct until they meet. A unique constraint would abort the merge -- transaction at exactly that moment — the ordinary case, not a corner one. -- Uniqueness is instead reached by convergence: `crate::keywords::create` -- resolves an existing name locally, and `crate::keywords::fuse_duplicates` -- collapses a cross-device pair onto the lexicographically smaller uuid, which -- both devices compute identically without talking to each other. -- -- Partial on `deleted = 0` because every lookup here is a live one: the -- vocabulary list, the resolve-by-name in `create`, and the fuse pass all -- exclude tombstones, and including them would grow the index with every -- keyword the library has ever had rather than with the ones it has. CREATE INDEX keyword_terms_name ON keyword_terms(name) WHERE deleted = 0; "#; const V5: &str = r#" -- TRACES: FR-NC-6a | FR-CAT-9 | NFR-RES-4 -- Offline availability: what is kept, why it is kept, and where it lives. -- -- `pinned` separates a promise from a convenience, and the distinction has to -- be a *column* rather than something inferred from `pinned_by_rule`. A pin is -- the user saying "this collection comes with me"; a passively cached original -- is the app noticing they opened something. Only the second is evictable, so -- the eviction query has to be able to ask the question directly — and it has -- to keep answering correctly for an image whose pinning rule was since -- deleted, which `pinned_by_rule` alone cannot do because it is -- ON DELETE SET NULL. ALTER TABLE image_cache ADD COLUMN pinned INTEGER NOT NULL DEFAULT 0; -- Where the cached original actually is, relative to the cache directory. -- Relative rather than absolute: the library moves between machines and -- between an app sandbox and a user directory, and an absolute path baked in -- at download time would break on every one of those. ALTER TABLE image_cache ADD COLUMN path TEXT; -- Eviction reads exactly this: unpinned rows, oldest use first. Partial on -- `pinned = 0` because pinned rows are never candidates and including them -- would make the index proportional to the whole library rather than to the -- passive cache. CREATE INDEX image_cache_evictable ON image_cache(last_used) WHERE pinned = 0; "#; const V4: &str = r#" -- TRACES: FR-CAT-15 -- Soft delete. A trashed image is a real file that has been *moved* to a trash -- folder under the library root, not a row hidden by a flag: the catalog is a -- rebuildable index (ARCH §6.12), so a flag alone would evaporate the moment -- the catalog was deleted and every trashed photograph would return. -- -- `source_ref` follows the file to its new path, because that is where the bytes -- now are and every fetch resolves through it. `trashed_from` remembers where it -- came from, which is the only way a restore can put it back — the trash is flat -- and the original folder structure is not recoverable from the trashed path. ALTER TABLE images ADD COLUMN trashed_at INTEGER; ALTER TABLE images ADD COLUMN trashed_from TEXT; -- Partial: almost no rows are trashed, and the grid's "not trashed" predicate is -- answered by the absence of an entry rather than by scanning every image. CREATE INDEX images_trashed ON images(trashed_at) WHERE trashed_at IS NOT NULL; "#; const V3: &str = r#" -- Ratings and flags are read per grid window and counted for the filter bar's -- histogram, both of which key on the *default* version. Without this the -- histogram is a full scan of `versions` on every judgement. -- -- Partial on `is_default`: a virtual copy's rating is real but is never what -- these two queries ask for, and excluding them keeps the index roughly one -- entry per image rather than one per version. CREATE INDEX versions_judgement ON versions(image_id, rating, flag) WHERE is_default = 1; "#; const V2: &str = r#" -- A JPEG the camera wrote alongside a RAW of the same name is that RAW's own -- rendering, not a second photograph. Recording *which* RAW shadows it, rather -- than a bare flag, keeps the relationship usable: the JPEG is a ready-made -- preview for its RAW, and the pairing can be undone without a rescan. ALTER TABLE images ADD COLUMN shadowed_by INTEGER REFERENCES images(id) ON DELETE SET NULL; CREATE INDEX images_shadowed ON images(shadowed_by) WHERE shadowed_by IS NOT NULL; "#; const V1: &str = r#" -- Roots ------------------------------------------------------------------- CREATE TABLE roots ( id INTEGER PRIMARY KEY, kind TEXT NOT NULL, -- 'local' | 'saf' | 'remote' grant_blob BLOB, -- SAF persisted permission; NULL on Linux label TEXT NOT NULL, last_seen INTEGER, -- Bumped once per completed scan. Folders record the generation they were -- reached in; anything older was not reached and no longer exists. scan_generation INTEGER NOT NULL DEFAULT 0, -- One row per granted location. Without this, a rescan inserts a second -- root for the same folder and the library silently fragments across -- them — images split between roots, and pruning compares against the -- wrong generation. UNIQUE(kind, label) ); -- Folders: the unit of change detection, local and remote alike ----------- CREATE TABLE folders ( id INTEGER PRIMARY KEY, root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE, parent_id INTEGER REFERENCES folders(id) ON DELETE CASCADE, path TEXT NOT NULL, -- Remote: the propagating ETag that makes a no-op sync one request. etag TEXT, -- Local: directory mtime plus direct-entry count. mtime alone misses a -- paired create+delete inside one timestamp tick; the count narrows that. mtime INTEGER, entry_count INTEGER, scanned_generation INTEGER NOT NULL DEFAULT 0, UNIQUE(root_id, path) ); CREATE INDEX folders_parent ON folders(parent_id); -- Images ------------------------------------------------------------------ CREATE TABLE images ( id INTEGER PRIMARY KEY, root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE, folder_id INTEGER REFERENCES folders(id) ON DELETE CASCADE, source_ref TEXT NOT NULL, -- Expensive: requires reading the whole file. Computed only when -- something needs it (import dedup, reconnect-by-hash), never in a scan. content_hash TEXT, format TEXT, w INTEGER, h INTEGER, -- UTC seconds. NULL until EXIF is read, or if the file carries none. captured_at INTEGER, -- Minutes east of UTC. A photograph's timestamp is local to where it was -- taken; storing UTC alone makes a Tokyo shoot span two days in Paris. captured_offset INTEGER, camera TEXT, lens TEXT, iso INTEGER, aperture REAL, shutter REAL, availability INTEGER NOT NULL DEFAULT 0, file_size INTEGER, file_mtime INTEGER, -- 0 = nothing, 1 = stat-only, 2 = full EXIF. The grid is usable at 1. metadata_state INTEGER NOT NULL DEFAULT 0, sidecar_mtime INTEGER, added_at INTEGER NOT NULL, UNIQUE(root_id, source_ref) ); CREATE INDEX images_captured ON images(captured_at); CREATE INDEX images_folder ON images(folder_id); -- Partial: content_hash is NULL for most rows most of the time, and the -- non-NULL subset is exactly what reconnect and dedup query. CREATE INDEX images_hash ON images(content_hash) WHERE content_hash IS NOT NULL; -- Versions ---------------------------------------------------------------- CREATE TABLE versions ( id INTEGER PRIMARY KEY, image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE, uuid TEXT NOT NULL UNIQUE, name TEXT NOT NULL, is_default INTEGER NOT NULL DEFAULT 0, graph_hash TEXT, rating INTEGER NOT NULL DEFAULT 0, label INTEGER, flag INTEGER NOT NULL DEFAULT 0 ); CREATE INDEX versions_image ON versions(image_id); CREATE TABLE keywords ( version_id INTEGER NOT NULL REFERENCES versions(id) ON DELETE CASCADE, keyword TEXT NOT NULL, PRIMARY KEY(version_id, keyword) ); CREATE INDEX keywords_term ON keywords(keyword); -- Remote mapping ---------------------------------------------------------- CREATE TABLE remote ( image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE, -- oc:fileid — stable across server-side rename and move, so a move is not -- a re-download of 80 MB. file_id INTEGER NOT NULL, etag TEXT, sync_state INTEGER NOT NULL DEFAULT 0, remote_path TEXT ); CREATE UNIQUE INDEX remote_file ON remote(file_id); -- Collections ------------------------------------------------------------- CREATE TABLE collections ( id INTEGER PRIMARY KEY, -- Device-independent identity. The integer id is local and collides -- across devices; the UUID is what a cross-device merge keys on. uuid TEXT NOT NULL UNIQUE, name TEXT NOT NULL, parent_id INTEGER REFERENCES collections(id) ON DELETE CASCADE, kind INTEGER NOT NULL, -- 0 = manual, 1 = smart selector_json TEXT, -- smart only created INTEGER NOT NULL, -- Monotonic per collection, bumped on every local edit. Merge compares -- these rather than file mtimes, so a clock-skewed device cannot silently -- win. revision INTEGER NOT NULL DEFAULT 1, modified INTEGER NOT NULL, -- Tombstone. A deleted collection must outlive its deletion, or a merge -- with a device that still has it would resurrect it. deleted INTEGER NOT NULL DEFAULT 0 ); CREATE TABLE collection_members ( collection_id INTEGER NOT NULL REFERENCES collections(id) ON DELETE CASCADE, image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE, position INTEGER, -- manual ordering; NULL = by capture time added INTEGER NOT NULL, PRIMARY KEY(collection_id, image_id) ); CREATE INDEX members_image ON collection_members(image_id); -- Cache ------------------------------------------------------------------- CREATE TABLE cache ( id INTEGER PRIMARY KEY, version_id INTEGER REFERENCES versions(id) ON DELETE CASCADE, image_id INTEGER REFERENCES images(id) ON DELETE CASCADE, kind INTEGER NOT NULL, -- thumbnail | proxy | original resolution INTEGER, graph_hash TEXT, path TEXT NOT NULL, bytes INTEGER NOT NULL, last_used INTEGER NOT NULL ); CREATE INDEX cache_lru ON cache(last_used); CREATE TABLE cache_rules ( id INTEGER PRIMARY KEY, selector_json TEXT NOT NULL, tier INTEGER NOT NULL, priority INTEGER NOT NULL DEFAULT 0, enabled INTEGER NOT NULL DEFAULT 1 ); CREATE TABLE image_cache ( image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE, tier_actual INTEGER NOT NULL DEFAULT 0, -- Materialised rather than recomputed, so the grid can draw availability -- badges without evaluating every rule for every visible cell. tier_desired INTEGER NOT NULL DEFAULT 0, bytes INTEGER NOT NULL DEFAULT 0, last_used INTEGER, pinned_by_rule INTEGER REFERENCES cache_rules(id) ON DELETE SET NULL ); -- Jobs -------------------------------------------------------------------- CREATE TABLE jobs ( id INTEGER PRIMARY KEY, kind INTEGER NOT NULL, subject_id INTEGER, priority INTEGER NOT NULL DEFAULT 0, state INTEGER NOT NULL DEFAULT 0, -- 0=pending 1=running 2=failed attempts INTEGER NOT NULL DEFAULT 0, not_before INTEGER NOT NULL DEFAULT 0, payload TEXT, last_error TEXT, -- Coalescing. Enqueueing the same work twice updates one row rather than -- queueing it twice, which is what makes "enqueue on any change" safe to -- call liberally. UNIQUE(kind, subject_id) ); CREATE INDEX jobs_ready ON jobs(state, priority DESC, not_before); "#; #[cfg(test)] mod tests { #[test] fn a_writer_waits_for_its_turn_rather_than_losing_its_work() { // The failure this exists for: a face sweep that had already paid for // the detection and the embedding threw the result away on // "database is locked" and moved on. WAL does not help here — it makes // one writer and many readers free, and this is two writers. let dir = std::env::temp_dir().join(format!( "dr-busy-{}-{:?}", std::process::id(), std::thread::current().id() )); let _ = std::fs::remove_dir_all(&dir); std::fs::create_dir_all(&dir).unwrap(); let path = dir.join("catalog.sqlite"); let held = rusqlite::Connection::open(&path).unwrap(); configure(&held).unwrap(); migrate(&held).unwrap(); let other = rusqlite::Connection::open(&path).unwrap(); configure(&other).unwrap(); // Every connection carries the timeout, which is what makes the wait // below a wait rather than an immediate error. let timeout: i64 = other .query_row("PRAGMA busy_timeout", [], |r| r.get(0)) .unwrap(); assert_eq!(timeout, BUSY_TIMEOUT.as_millis() as i64); // A writer holds the database; the other one must still get its turn // once the first commits, rather than failing at the moment it asks. let writing = held.unchecked_transaction().unwrap(); held.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')", [], ) .unwrap(); let handle = std::thread::spawn(move || { other.execute( "INSERT INTO roots(id, kind, label) VALUES (2, 'local', 'two')", [], ) }); std::thread::sleep(std::time::Duration::from_millis(150)); writing.commit().unwrap(); assert!( handle.join().unwrap().is_ok(), "the second writer waited and then wrote, rather than erroring" ); let _ = std::fs::remove_dir_all(&dir); } use super::*; fn mem() -> Connection { let c = Connection::open_in_memory().unwrap(); configure(&c).unwrap(); c } /// How many rows a named backfill touched, ignoring the others. /// /// Asserting on the whole vector would couple every test to which other /// backfills happen to exist. fn backfilled(c: &Connection, what: &str) -> usize { backfill(c) .unwrap() .into_iter() .find(|(name, _)| *name == what) .map(|(_, n)| n) .unwrap_or(0) } /// Insert an image and return its id. fn image(c: &Connection, folder: Option, name: &str, format: &str) -> i64 { c.execute( "INSERT INTO images(root_id, folder_id, source_ref, format, added_at) VALUES (1, ?1, ?2, ?3, 0)", rusqlite::params![folder, name, format], ) .unwrap(); c.last_insert_rowid() } fn with_root() -> Connection { let c = mem(); migrate(&c).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); c.execute( "INSERT INTO folders(id, root_id, path) VALUES (1, 1, 'a'), (2, 1, 'b')", [], ) .unwrap(); c } #[test] fn a_jpeg_beside_its_raw_is_shadowed() { // The camera's own rendering of a frame, not a second photograph. let c = with_root(); let raw = image(&c, Some(1), "a/IMG_1234.CR2", "cr2"); let jpeg = image(&c, Some(1), "a/IMG_1234.JPG", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 1); let got: Option = c .query_row( "SELECT shadowed_by FROM images WHERE id = ?1", [jpeg], |r| r.get(0), ) .unwrap(); assert_eq!(got, Some(raw)); } #[test] fn extension_case_does_not_matter() { let c = with_root(); image(&c, Some(1), "a/IMG_1.cr2", "cr2"); image(&c, Some(1), "a/img_1.JPG", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 1); } #[test] fn a_standalone_jpeg_is_untouched() { // Scanned film has no RAW sibling and must stay visible — 2,656 of // them in the reference library. let c = with_root(); image(&c, Some(1), "a/SCAN_0001.jpg", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 0); } #[test] fn a_jpeg_in_a_different_folder_is_not_shadowed() { // Camera filenames wrap at IMG_9999, so the same stem recurs across // shoots (FR-CAT-11). Only a same-folder pair is safe to collapse. let c = with_root(); image(&c, Some(1), "a/IMG_1234.CR2", "cr2"); image(&c, Some(2), "b/IMG_1234.JPG", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 0); } #[test] fn a_raw_is_never_shadowed_by_a_jpeg() { // The relationship is one-way: the RAW is the photograph. let c = with_root(); let raw = image(&c, Some(1), "a/IMG_1.CR2", "cr2"); image(&c, Some(1), "a/IMG_1.JPG", "jpg"); backfill(&c).unwrap(); let got: Option = c .query_row("SELECT shadowed_by FROM images WHERE id = ?1", [raw], |r| { r.get(0) }) .unwrap(); assert_eq!(got, None); } #[test] fn backfill_is_idempotent() { // It runs on every open, so a second pass must find nothing to do. let c = with_root(); image(&c, Some(1), "a/IMG_1.CR2", "cr2"); image(&c, Some(1), "a/IMG_1.JPG", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 1); assert_eq!(backfilled(&c, "shadowed_by"), 0, "second pass is a no-op"); } #[test] fn a_v1_catalog_gains_the_column_and_is_backfilled() { // The migration case that motivated this: rows already present when a // column is added are silently partial until something backfills them. let c = mem(); c.execute_batch(V1).unwrap(); c.pragma_update(None, "user_version", 1).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); c.execute( "INSERT INTO images(root_id, source_ref, format, added_at) VALUES (1, 'IMG_9.CR2', 'cr2', 0), (1, 'IMG_9.JPG', 'jpg', 0)", [], ) .unwrap(); assert_eq!(migrate(&c).unwrap(), 1, "migrated from v1"); assert_eq!(backfilled(&c, "shadowed_by"), 1); } #[test] fn a_v4_catalog_gains_the_pinning_columns() { // TRACES: FR-NC-6a // An existing library must not have to be rescanned to gain offline // pinning. The rows are already there; only the columns are new. let c = mem(); c.execute_batch(V1).unwrap(); c.execute_batch(V2).unwrap(); c.execute_batch(V3).unwrap(); c.execute_batch(V4).unwrap(); c.pragma_update(None, "user_version", 4).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (7, 1, 'IMG_7.CR2', 0)", [], ) .unwrap(); // A cache row written before pinning existed. c.execute( "INSERT INTO image_cache(image_id, tier_actual, bytes) VALUES (7, 2, 100)", [], ) .unwrap(); assert_eq!(migrate(&c).unwrap(), 4, "migrated from v4"); // The pre-existing row survives, and defaults to unpinned — the safe // direction, since claiming a pin nobody made would exempt it from // eviction for ever. let (pinned, bytes): (i64, i64) = c .query_row( "SELECT pinned, bytes FROM image_cache WHERE image_id = 7", [], |r| Ok((r.get(0)?, r.get(1)?)), ) .unwrap(); assert_eq!(pinned, 0); assert_eq!(bytes, 100, "the existing row is untouched"); } #[test] fn a_v5_catalog_keeps_its_keywords_and_gains_their_identities() { // TRACES: FR-CAT-5 // The migration case that matters here: a library keyworded by an // import or an older build already has assignment rows, and they must // survive into the vocabulary rather than being left searchable but // invisible. let c = mem(); for step in [V1, V2, V3, V4, V5] { c.execute_batch(step).unwrap(); } c.pragma_update(None, "user_version", 5).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (7, 1, 'IMG_7.CR3', 0)", [], ) .unwrap(); c.execute( "INSERT INTO versions(id, image_id, uuid, name, is_default) VALUES (1, 7, 'v-7', 'Default', 1)", [], ) .unwrap(); c.execute( "INSERT INTO keywords(version_id, keyword) VALUES (1, 'puffin')", [], ) .unwrap(); assert_eq!(migrate(&c).unwrap(), 5, "migrated from v5"); assert_eq!(backfilled(&c, "keyword_terms"), 1); let name: String = c .query_row("SELECT name FROM keyword_terms", [], |r| r.get(0)) .unwrap(); assert_eq!(name, "puffin"); // The assignment is untouched — it is the durable fact, and the term // row is only its identity. let n: i64 = c .query_row("SELECT count(*) FROM keywords", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 1); // It runs on every open, so a second pass must find nothing to do. assert_eq!(backfilled(&c, "keyword_terms"), 0); } #[test] fn two_devices_may_both_hold_a_term_of_the_same_name() { // Deliberately not a unique index. Two devices each typing "Iceland" // is the ordinary case, and a constraint would abort the merge // transaction at exactly the moment they first sync. let c = mem(); migrate(&c).unwrap(); c.execute( "INSERT INTO keyword_terms(uuid, name, created, revision, modified) VALUES ('a', 'Iceland', 0, 1, 1), ('b', 'Iceland', 0, 1, 1)", [], ) .unwrap(); let n: i64 = c .query_row("SELECT count(*) FROM keyword_terms", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 2); } #[test] fn stems_ignore_directories_containing_dots() { assert_eq!(stem_of("2026.08/IMG_1.CR2"), "IMG_1"); assert_eq!(stem_of("IMG_1.CR2"), "IMG_1"); assert_eq!(stem_of("noextension"), "noextension"); // A dotfile is all stem, not an empty name with an extension. assert_eq!(stem_of(".hidden"), ".hidden"); } #[test] fn migrate_creates_schema_at_current_version() { let c = mem(); assert_eq!(migrate(&c).unwrap(), 0); let v: i64 = c .query_row("PRAGMA user_version", [], |r| r.get(0)) .unwrap(); assert_eq!(v, SCHEMA_VERSION); } #[test] fn migrate_is_idempotent() { let c = mem(); migrate(&c).unwrap(); // Re-running must not error or duplicate anything — NFR-R5 requires // idempotency on retry, since a migration can be interrupted. assert_eq!(migrate(&c).unwrap(), SCHEMA_VERSION); } /// V16 adds its columns guarded, so a catalog whose version was rewound /// after the columns landed — the rollback NFR-R5 contemplates — migrates /// again rather than failing on "duplicate column". #[test] fn the_eye_columns_survive_a_rewound_version() { let c = mem(); migrate(&c).unwrap(); for column in EYE_COLUMNS { let present: bool = c .prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = ?1") .unwrap() .exists([column]) .unwrap(); assert!(present, "{column} missing after migration"); } c.pragma_update(None, "user_version", 15).unwrap(); assert_eq!(migrate(&c).unwrap(), 15); let indexed: bool = c .prepare("SELECT 1 FROM sqlite_master WHERE type = 'index' AND name = 'faces_eyes'") .unwrap() .exists([]) .unwrap(); assert!(indexed, "V17's covering index is there"); let v: i64 = c .query_row("PRAGMA user_version", [], |r| r.get(0)) .unwrap(); assert_eq!(v, SCHEMA_VERSION); } #[test] fn refuses_a_catalog_from_a_newer_build() { let c = mem(); migrate(&c).unwrap(); c.pragma_update(None, "user_version", SCHEMA_VERSION + 1) .unwrap(); // Opening it read-write would corrupt data this build cannot // represent. Refusing is the specified behaviour (NFR-R5). assert!(matches!( migrate(&c), Err(CatalogError::SchemaTooNew { .. }) )); } #[test] fn foreign_keys_cascade_from_root_to_image() { let c = mem(); migrate(&c).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)", [], ) .unwrap(); c.execute("DELETE FROM roots WHERE id = 1", []).unwrap(); let n: i64 = c .query_row("SELECT count(*) FROM images", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 0, "images must not outlive their root"); } #[test] fn v12_forgets_runs_made_on_a_proxy_too_small_to_see_a_face() { let c = mem(); // Migrate to 11, then seed the state V12 exists to repair: markers // written at the 1024 store tier beside ones written on a real // preview. c.pragma_update(None, "user_version", 0).unwrap(); migrate(&c).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0),(4,1,'d',0)", [], ) .unwrap(); for (image, edge) in [(1, 896), (2, 1024), (3, 1025), (4, 2560)] { c.execute( "INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge) VALUES (?1, 'm', 0, 0, ?2)", rusqlite::params![image, edge], ) .unwrap(); } c.pragma_update(None, "user_version", 11).unwrap(); migrate(&c).unwrap(); let kept: Vec = c .prepare("SELECT image_id FROM face_index ORDER BY image_id") .unwrap() .query_map([], |r| r.get(0)) .unwrap() .map(Result::unwrap) .collect(); // 1024 goes: it is exactly ThumbSize::Large, the tier that produced // the bad runs. 1025 stays, or the floor and the repair disagree // about the same boundary. assert_eq!(kept, vec![3, 4]); } #[test] fn v14_forgets_runs_that_found_faces_but_never_measured_them() { let c = mem(); c.pragma_update(None, "user_version", 0).unwrap(); migrate(&c).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0)", [], ) .unwrap(); // Image 1 was examined and holds a face; 2 was examined and found // empty; 3 holds a face found by a different model. for (image, model) in [(1, "m"), (2, "m"), (3, "m")] { c.execute( "INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge) VALUES (?1, ?2, 0, 0, 2560)", rusqlite::params![image, model], ) .unwrap(); } for (image, model) in [(1, "m"), (3, "other")] { c.execute( "INSERT INTO faces (image_id, x, y, w, h, landmarks, detector_confidence, embedding, crop_px, model_id, detected_at) VALUES (?1, 0.1, 0.1, 0.2, 0.2, X'00', 0.9, X'00', 180.0, ?2, 0)", rusqlite::params![image, model], ) .unwrap(); } c.pragma_update(None, "user_version", 13).unwrap(); migrate(&c).unwrap(); let kept: Vec = c .prepare("SELECT image_id FROM face_index ORDER BY image_id") .unwrap() .query_map([], |r| r.get(0)) .unwrap() .map(Result::unwrap) .collect(); // 1 goes: it has a face with no quality. 2 stays: nothing on it to // measure. 3 stays: its face belongs to a run this marker does not // describe. assert_eq!(kept, vec![2, 3]); // And the faces themselves are untouched. let faces: i64 = c .query_row("SELECT count(*) FROM faces", [], |r| r.get(0)) .unwrap(); assert_eq!(faces, 2); } #[test] fn v20_renames_markers_to_the_detector_that_found_the_faces() { let c = mem(); c.pragma_update(None, "user_version", 0).unwrap(); migrate(&c).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0),(4,1,'d',0),(5,1,'e',0)", [], ) .unwrap(); // 1: the desktop's case -- old faces, re-marked as thorough. // 2: the tablet's case -- adopted thorough faces, re-marked int8, // and the right marker still beside it (refreshed, so it is // exported again over the empty entry). // 3: right already. 4: examined and empty. 5: V14's state, faces // and no marker. for (image, model) in [ (1, "scrfd_10g+w600k_mbf"), (2, "scrfd_10g_i8+w600k_mbf"), (2, "scrfd_10g+w600k_mbf"), (3, "scrfd_10g+w600k_mbf"), (4, "scrfd_10g+w600k_mbf"), ] { c.execute( "INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge) VALUES (?1, ?2, 100, 0, 6000)", rusqlite::params![image, model], ) .unwrap(); } for (image, model) in [ (1, "w600k_mbf"), (1, "w600k_mbf"), (2, "scrfd_10g+w600k_mbf"), (3, "scrfd_10g+w600k_mbf"), (5, "w600k_mbf"), ] { c.execute( "INSERT INTO faces (image_id, x, y, w, h, landmarks, detector_confidence, embedding, crop_px, model_id, detected_at) VALUES (?1, 0.1, 0.1, 0.2, 0.2, X'00', 0.9, X'00', 180.0, ?2, 0)", rusqlite::params![image, model], ) .unwrap(); } c.pragma_update(None, "user_version", 19).unwrap(); migrate(&c).unwrap(); let markers: Vec<(i64, String, i64, bool)> = c .prepare( "SELECT image_id, model_id, faces_found, indexed_at > 100 FROM face_index ORDER BY image_id, model_id", ) .unwrap() .query_map([], |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?))) .unwrap() .map(Result::unwrap) .collect(); assert_eq!( markers, vec![ (1, "w600k_mbf".to_string(), 2, true), (2, "scrfd_10g+w600k_mbf".to_string(), 0, true), (3, "scrfd_10g+w600k_mbf".to_string(), 0, false), (4, "scrfd_10g+w600k_mbf".to_string(), 0, false), ] ); // Re-enterable: nothing left to rename. c.pragma_update(None, "user_version", 19).unwrap(); migrate(&c).unwrap(); let n: i64 = c .query_row("SELECT count(*) FROM face_index", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 4); } #[test] fn job_uniqueness_coalesces_rather_than_duplicating() { let c = mem(); migrate(&c).unwrap(); for _ in 0..5 { c.execute( "INSERT INTO jobs(kind, subject_id, priority) VALUES (1, 42, 0) ON CONFLICT(kind, subject_id) DO UPDATE SET priority = max(priority, excluded.priority)", [], ) .unwrap(); } let n: i64 = c .query_row("SELECT count(*) FROM jobs", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 1, "five enqueues of the same work is one job"); } }