//! TRACES: FR-CAT-2 | NFR-R5 //! Schema definition and forward-only migrations. //! //! The catalog is an *index*, not a source of truth (ARCH §6.12) — it is //! deletable and rebuildable from sources plus sidecars. That is what makes //! migration failure survivable, and why the recovery path is the normal //! mechanism rather than a last resort. //! //! Migrations are forward-only, transactional, and idempotent on retry //! (NFR-R5). The app refuses to open a catalog newer than it understands //! rather than corrupting it. use rusqlite::Connection; use crate::error::CatalogError; /// Schema version this build writes and understands. pub const SCHEMA_VERSION: i64 = 8; /// Apply migrations up to [`SCHEMA_VERSION`]. /// /// Returns the version migrated from, so callers can log or back up before a /// real migration (NFR-R2 requires a backup before schema change). pub fn migrate(conn: &Connection) -> Result { let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?; if from > SCHEMA_VERSION { return Err(CatalogError::SchemaTooNew { found: from, supported: SCHEMA_VERSION, }); } if from == SCHEMA_VERSION { return Ok(from); } // Each step runs in its own transaction so a failure leaves the catalog // at a coherent version rather than half-migrated. if from < 1 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V1)?; tx.pragma_update(None, "user_version", 1)?; tx.commit()?; } if from < 2 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V2)?; tx.pragma_update(None, "user_version", 2)?; tx.commit()?; } if from < 3 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V3)?; tx.pragma_update(None, "user_version", 3)?; tx.commit()?; } if from < 4 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V4)?; tx.pragma_update(None, "user_version", 4)?; tx.commit()?; } if from < 5 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V5)?; tx.pragma_update(None, "user_version", 5)?; tx.commit()?; } if from < 6 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V6)?; tx.pragma_update(None, "user_version", 6)?; tx.commit()?; } if from < 7 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V7)?; tx.pragma_update(None, "user_version", 7)?; tx.commit()?; } if from < 8 { let tx = conn.unchecked_transaction()?; tx.execute_batch(V8)?; tx.pragma_update(None, "user_version", 8)?; tx.commit()?; } Ok(from) } /// Recompute columns a migration added, for rows that predate it. /// /// A migration adds a column with a default; it cannot know what the value /// *should* be for the rows already present. Without a backfill those rows are /// silently partial — present, queryable, and wrong — which is worse than /// missing, because nothing signals that they need attention. /// /// Cheap enough to run on every open: each pass is one indexed UPDATE, and /// re-running it is a no-op once the values are already right. /// /// Returns how many rows each backfill touched, for logging. pub fn backfill(conn: &Connection) -> Result, CatalogError> { let mut out = Vec::new(); // v2: `shadowed_by`. A JPEG sitting beside a RAW of the same name is the // camera's own rendering of that frame, not a second photograph, so it is // hidden from the grid, the timeline and the sweep. let n = pair_raw_and_jpeg(conn)?; if n > 0 { out.push(("shadowed_by", n)); } // v3: every image needs a default version to carry its rating and flag. // Libraries scanned before ratings existed have images and no versions at // all, so there was nowhere for a judgement to go — see // [`crate::rating`]. Backfilled rather than migrated in SQL because the // UUID per row is the cross-device merge identity and must be generated, // not derived. let n = crate::rating::ensure_default_versions(conn)?; if n > 0 { out.push(("default_versions", n)); } // v6: a vocabulary row for every word some image already carries. // // Three ways a catalog arrives holding assignments with no term behind // them, and all three are normal rather than exceptional: a library // keyworded by a build that predates this table, a catalog rebuilt from // sidecars (which carry the word and not the identity), and an import from // Lightroom or darktable (FR-CAT-14). Without this the words are // searchable but absent from the vocabulary list, which reads as the // keywords having been lost. let n = crate::keywords::adopt_orphan_terms(conn)?; if n > 0 { out.push(("keyword_terms", n)); } Ok(out) } /// Connection setup applied on every open, migration or not. /// /// WAL is required by NFR-R1: it survives power loss without corruption, and /// it lets a background job write while the grid reads. pub fn configure(conn: &Connection) -> Result<(), CatalogError> { conn.pragma_update(None, "journal_mode", "WAL")?; // NORMAL rather than FULL: with WAL this is durable across process death // (which is what FR-PLAT-AND-3 cares about) and only risks the last // transaction on power loss. The catalog is rebuildable; the sidecars are // not, and they are written separately with their own fsync discipline. conn.pragma_update(None, "synchronous", "NORMAL")?; conn.pragma_update(None, "foreign_keys", true)?; // A scan touching thousands of rows is transient; let SQLite spill to // memory rather than materialising temp b-trees on disk. conn.pragma_update(None, "temp_store", "MEMORY")?; Ok(()) } /// The v1 schema rewritten to target an attached database. /// /// Needed because a downloaded remote catalog is `ATTACH`ed under its own /// schema name before merging, and tests build one from scratch. SQLite has no /// "create these tables over there" form, so the names are rewritten. /// /// The rewrite is textual and therefore only as good as the naming discipline /// in [`V1`]: every `CREATE TABLE`/`CREATE INDEX` must name its object /// unqualified, which they do. pub fn v1_for_attached(schema_name: &str) -> String { rewrite_for_attached(V1, schema_name) // REFERENCES within an attached schema resolve to that schema already, // so foreign keys need no rewriting — but the ON clause of an index // does, and `CREATE INDEX x.name ON table` is the correct form. } /// Every table this build knows about, rewritten to target an attached /// database. /// /// [`v1_for_attached`] is kept alongside this rather than replaced by it: a /// remote catalog written by an older build genuinely has only the v1 tables, /// and the merge has to keep working against one (see /// [`crate::merge::merge_keywords`]). Building that case in a test needs a way /// to say "v1 and no more". /// /// Only the migrations that *create* objects appear here. V2 through V5 are /// `ALTER TABLE ... ADD COLUMN`, and the columns they add are local index /// state — shadowing, trashing, cache pinning — that a merge never reads /// across the attachment. /// /// V7 creates an object and is still excluded, which is the one exception to /// that rule and not an oversight: it indexes `shadowed_by` and `trashed_at`, /// the very columns V2 through V5 add and this function leaves out, so /// creating it over there would fail on columns that are not there. Nothing is /// lost by its absence — it exists to make the *grid* page quickly, and the /// grid never reads across an attachment. pub fn for_attached(schema_name: &str) -> String { format!( "{}\n{}", rewrite_for_attached(V1, schema_name), rewrite_for_attached(V6, schema_name) ) } /// Qualify every object a `CREATE` statement names with `schema_name`. /// /// The rewrite is textual and therefore only as good as the naming discipline /// in the batches it is given: every `CREATE TABLE`/`CREATE INDEX` must name /// its object unqualified, which they do. fn rewrite_for_attached(sql: &str, schema_name: &str) -> String { sql.replace("CREATE TABLE ", &format!("CREATE TABLE {schema_name}.")) .replace("CREATE INDEX ", &format!("CREATE INDEX {schema_name}.")) .replace( "CREATE UNIQUE INDEX ", &format!("CREATE UNIQUE INDEX {schema_name}."), ) } /// Mark each JPEG that sits beside a RAW of the same name. /// /// Matched on folder plus stem, case-insensitively. Same-folder is what makes /// this safe: cameras write the pair side by side, and matching across folders /// would risk pairing unrelated frames, since camera filenames wrap at /// IMG_9999 (FR-CAT-11). /// /// Done in Rust rather than SQL because the comparison needs a filename stem, /// and SQLite has no such function without enabling `rusqlite/functions` — /// a dependency feature for one string operation, whose SQL spelling would be /// unreadable and would mishandle names with no extension. fn pair_raw_and_jpeg(conn: &Connection) -> Result { use std::collections::HashMap; // (folder, lowercase stem) -> RAW id, built in one pass over the RAWs. let mut raws: HashMap<(Option, String), i64> = HashMap::new(); { let mut stmt = conn.prepare( "SELECT id, folder_id, source_ref FROM images WHERE lower(format) IN ('cr2','cr3','nef','arw','raf','rw2','orf','dng')", )?; let rows = stmt.query_map([], |r| { Ok(( r.get::<_, i64>(0)?, r.get::<_, Option>(1)?, r.get::<_, String>(2)?, )) })?; for row in rows { let (id, folder, path) = row?; raws.insert((folder, stem_of(&path).to_ascii_lowercase()), id); } } if raws.is_empty() { return Ok(0); } let pairs: Vec<(i64, i64)> = { let mut stmt = conn.prepare( "SELECT id, folder_id, source_ref FROM images WHERE lower(format) IN ('jpg','jpeg') AND shadowed_by IS NULL", )?; let rows = stmt.query_map([], |r| { Ok(( r.get::<_, i64>(0)?, r.get::<_, Option>(1)?, r.get::<_, String>(2)?, )) })?; rows.filter_map(|row| { let (id, folder, path) = row.ok()?; let raw = raws.get(&(folder, stem_of(&path).to_ascii_lowercase()))?; Some((id, *raw)) }) .collect() }; let tx = conn.unchecked_transaction()?; for (jpeg, raw) in &pairs { tx.execute( "UPDATE images SET shadowed_by = ?2 WHERE id = ?1", [jpeg, raw], )?; } tx.commit()?; Ok(pairs.len()) } /// A filename without its extension. /// /// Only the final path component, and only its last dot — a directory /// containing a dot must not truncate the name. fn stem_of(path: &str) -> &str { let name = path.rsplit(['/', ':']).next().unwrap_or(path); match name.rsplit_once('.') { Some((stem, _)) if !stem.is_empty() => stem, _ => name, } } /// TRACES: FR-CAT-4 | NFR-P5 /// The order the grid reads in, as an index. /// /// # What this is for /// /// Every window the grid loads is `ORDER BY ... LIMIT n OFFSET k`, and without /// an index that matches the ordering SQLite answers it by sorting the whole /// library into a temp b-tree and then discarding the first `k` rows. Measured /// on 24,000 images at offset 20,000, one window read cost 15 ms — a frame /// budget of 16.7 ms, spent inside the scroll handler, several times per /// screenful. That is the jitter. /// /// With this index the same read is a walk along it: 0.36 ms. /// /// # Why the shape is what it is /// /// `captured_at IS NULL` leads, because [`crate::library`]'s `GRID_ORDER` does /// — undated images sort last, and an ordinary index on `captured_at` cannot /// answer that, since the expression is not a column. SQLite indexes /// expressions, so it is spelled out here exactly as the query spells it; the /// two must stay identical or the planner silently falls back to sorting and /// the cost comes back with no other symptom. /// /// `source_ref` is included because it breaks ties in the same ordering, and an /// index that stopped at `captured_at` would leave a sort for the ties. /// /// # Partial, on the same predicate the grid filters by /// /// The grid never lists shadowed or trashed rows, so an index carrying them /// would be larger than the question ever asks about, and — more to the point — /// a partial index is only usable when its `WHERE` is implied by the query's, /// which is what makes this one apply to the grid's reads and to nothing else. /// /// # It does not cover the rating filter or a collection scope /// /// Both narrow the walk rather than reorder it, so the index still supplies the /// ordering and SQLite tests the extra predicate per row. That is the cheap /// direction: the expensive part was never the filtering, it was the sort. const V8: &str = r#" -- TRACES: FR-CULL-8 | FR-CULL-9 | FR-CULL-10 | FR-CULL-11 | FR-CULL-12 | NFR-SEC-5 -- People and faces (docs/faces.md, docs/catalog.md §10). -- -- Everything here is **derived data** except one column. Faces, landmarks, -- embeddings, cluster assignments and suggestions are all reproducible by -- re-indexing and are never written to a sidecar (FR-CULL-12); a person's -- *name*, once the user has confirmed it, is a human judgement of the same -- class as a rating and travels with the photograph. -- -- That asymmetry is the whole design: deleting the catalog costs an afternoon -- of re-indexing and loses nothing the user typed (ARCH §6.12). CREATE TABLE people ( id INTEGER PRIMARY KEY, -- Merge identity, not the name. Two devices that independently name the -- same cluster produce two people; merging them keys on this, exactly as -- collections do (FR-CAT-7, ARCH §6.3). uuid TEXT NOT NULL UNIQUE, name TEXT NOT NULL, -- Tombstone-by-redirect. A merged person must outlive its merge or a -- device that still holds it resurrects it on the next sync -- the same -- hazard collections have, solved the same way. merged_into INTEGER REFERENCES people(id) ON DELETE SET NULL, created INTEGER NOT NULL, revision INTEGER NOT NULL DEFAULT 1, modified INTEGER NOT NULL ); CREATE TABLE faces ( id INTEGER PRIMARY KEY, image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE, -- Normalised to the image's long edge, so a face survives the proxy it was -- found on being evicted and regenerated at another resolution. Storing -- pixels would bind a face to a resolution the cache is entitled to change. x REAL NOT NULL, y REAL NOT NULL, w REAL NOT NULL, h REAL NOT NULL, landmarks BLOB NOT NULL, -- 5 x (x, y) f32, normalised likewise detector_confidence REAL NOT NULL, embedding BLOB NOT NULL, -- 512 x f16, L2-normalised -- Source pixels across the aligned 112x112 crop (docs/faces.md §7). -- -- Not cosmetic: it is the honest quality signal for the UI, a feature in -- the §8 calibration -- FR-CULL-9 names face size as an axis along which an -- uncalibrated similarity misbehaves -- and the selector a later -- higher-resolution re-embedding pass would run on. crop_px REAL NOT NULL, -- Which model produced this embedding. -- -- The one mistake in this subsystem that yields plausible-looking garbage -- rather than an error: embeddings from different models are not -- comparable. Storing the model with the vector makes a model change -- detectable and re-indexable instead of quietly poisoning every -- similarity in the library. model_id TEXT NOT NULL, detected_at INTEGER NOT NULL ); CREATE INDEX faces_image ON faces(image_id); CREATE INDEX faces_model ON faces(model_id); CREATE TABLE face_person ( face_id INTEGER PRIMARY KEY REFERENCES faces(id) ON DELETE CASCADE, person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE, -- Calibrated P(this face is this person), never a raw cosine (FR-CULL-9). probability REAL NOT NULL, -- The user said so. Never overwritten by a later inference pass. -- -- A column rather than a probability of 1.0, because a confirmation is a -- different kind of fact from a confident guess and collapsing them loses -- the ability to recompute suggestions without touching user data. confirmed INTEGER NOT NULL DEFAULT 0 ); CREATE INDEX face_person_person ON face_person(person_id, confirmed); -- Faces the user has explicitly said are NOT a given person. -- -- Needed because rejection is not the absence of an assignment: without it, -- the next clustering pass re-suggests exactly the face the user just pushed -- away, and the tool feels broken. Same reasoning as `confirmed` -- a -- judgement is user data (FR-CULL-12) whichever direction it points. CREATE TABLE face_person_rejected ( face_id INTEGER NOT NULL REFERENCES faces(id) ON DELETE CASCADE, person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE, PRIMARY KEY (face_id, person_id) ); -- The FR-CULL-9 calibration, fitted from this library's own faces. -- -- One row per model, because the fit is a property of the embedding space and -- a library indexed across a model change holds two. `face_set_hash` is what -- makes a stale fit detectable: a materially changed library recomputes rather -- than trusting numbers derived from a set that no longer exists. CREATE TABLE face_calibration ( model_id TEXT PRIMARY KEY, -- P(same) = sigmoid(a*cos + b + w_size*log2(min(crop_px)) + log_prior_odds) a REAL NOT NULL, b REAL NOT NULL, w_size REAL NOT NULL DEFAULT 0.0, -- Whether the fit is usable at all. When it is not, the UI says the -- confidence is unavailable; it does not present an untuned default as -- though it were measured (FR-CULL-9). valid INTEGER NOT NULL DEFAULT 0, positive_pairs INTEGER NOT NULL DEFAULT 0, negative_pairs INTEGER NOT NULL DEFAULT 0, face_set_hash TEXT NOT NULL, fitted_at INTEGER NOT NULL ); "#; const V7: &str = r#" CREATE INDEX images_grid_order ON images(captured_at IS NULL, captured_at, source_ref) WHERE shadowed_by IS NULL AND trashed_at IS NULL; "#; const V6: &str = r#" -- TRACES: FR-CAT-5 | FR-CAT-6 | FR-NC-9 -- Keywords gain an identity, so that renaming and deleting one can cross -- between devices. -- -- The v1 `keywords` table is the *assignment*: one row per (version, word), -- and the word is stored as text. That stays exactly as it is, and this -- migration adds nothing to it, for a reason that is easy to get backwards. -- -- # Why assignments keep the text rather than pointing at a row here -- -- The catalog is a rebuildable index (ARCH §6.12). What an image is keyworded -- with is authoritative in the sidecar and in XMP `dc:subject` (FR-CAT-13), -- and both of those carry a *string*. Rewriting the join to reference -- `keyword_terms(id)` would mean a catalog rebuilt from sidecars had to invent -- term rows before it could record a single assignment, and an integer that -- means nothing on the other device would sit where the durable fact belongs. -- It would also break `crate::query`, which matches `kw.keyword` directly and -- must keep hitting `keywords_term` on a 50k library (FR-CAT-6). -- -- So the text is the fact and this table is the *identity*: it exists to give -- a rename and a deletion something a merge can key on, and to let a keyword -- exist in the vocabulary before any photograph carries it. CREATE TABLE keyword_terms ( id INTEGER PRIMARY KEY, -- Device-independent identity, as `collections.uuid` is. The integer id is -- local and collides across devices. uuid TEXT NOT NULL UNIQUE, -- The word itself, and the value written into every assignment row. name TEXT NOT NULL, created INTEGER NOT NULL, -- Monotonic, bumped on every local edit. `crate::merge` compares these -- rather than timestamps, so a clock-skewed device cannot silently win. revision INTEGER NOT NULL DEFAULT 1, modified INTEGER NOT NULL, -- Tombstone, so a merge against a device that still holds the keyword does -- not resurrect it. deleted INTEGER NOT NULL DEFAULT 0 ); -- Deliberately **not** UNIQUE. -- -- Two devices that each type "Iceland" create two rows with two uuids, and -- both are correct until they meet. A unique constraint would abort the merge -- transaction at exactly that moment — the ordinary case, not a corner one. -- Uniqueness is instead reached by convergence: `crate::keywords::create` -- resolves an existing name locally, and `crate::keywords::fuse_duplicates` -- collapses a cross-device pair onto the lexicographically smaller uuid, which -- both devices compute identically without talking to each other. -- -- Partial on `deleted = 0` because every lookup here is a live one: the -- vocabulary list, the resolve-by-name in `create`, and the fuse pass all -- exclude tombstones, and including them would grow the index with every -- keyword the library has ever had rather than with the ones it has. CREATE INDEX keyword_terms_name ON keyword_terms(name) WHERE deleted = 0; "#; const V5: &str = r#" -- TRACES: FR-NC-6a | FR-CAT-9 | NFR-RES-4 -- Offline availability: what is kept, why it is kept, and where it lives. -- -- `pinned` separates a promise from a convenience, and the distinction has to -- be a *column* rather than something inferred from `pinned_by_rule`. A pin is -- the user saying "this collection comes with me"; a passively cached original -- is the app noticing they opened something. Only the second is evictable, so -- the eviction query has to be able to ask the question directly — and it has -- to keep answering correctly for an image whose pinning rule was since -- deleted, which `pinned_by_rule` alone cannot do because it is -- ON DELETE SET NULL. ALTER TABLE image_cache ADD COLUMN pinned INTEGER NOT NULL DEFAULT 0; -- Where the cached original actually is, relative to the cache directory. -- Relative rather than absolute: the library moves between machines and -- between an app sandbox and a user directory, and an absolute path baked in -- at download time would break on every one of those. ALTER TABLE image_cache ADD COLUMN path TEXT; -- Eviction reads exactly this: unpinned rows, oldest use first. Partial on -- `pinned = 0` because pinned rows are never candidates and including them -- would make the index proportional to the whole library rather than to the -- passive cache. CREATE INDEX image_cache_evictable ON image_cache(last_used) WHERE pinned = 0; "#; const V4: &str = r#" -- TRACES: FR-CAT-15 -- Soft delete. A trashed image is a real file that has been *moved* to a trash -- folder under the library root, not a row hidden by a flag: the catalog is a -- rebuildable index (ARCH §6.12), so a flag alone would evaporate the moment -- the catalog was deleted and every trashed photograph would return. -- -- `source_ref` follows the file to its new path, because that is where the bytes -- now are and every fetch resolves through it. `trashed_from` remembers where it -- came from, which is the only way a restore can put it back — the trash is flat -- and the original folder structure is not recoverable from the trashed path. ALTER TABLE images ADD COLUMN trashed_at INTEGER; ALTER TABLE images ADD COLUMN trashed_from TEXT; -- Partial: almost no rows are trashed, and the grid's "not trashed" predicate is -- answered by the absence of an entry rather than by scanning every image. CREATE INDEX images_trashed ON images(trashed_at) WHERE trashed_at IS NOT NULL; "#; const V3: &str = r#" -- Ratings and flags are read per grid window and counted for the filter bar's -- histogram, both of which key on the *default* version. Without this the -- histogram is a full scan of `versions` on every judgement. -- -- Partial on `is_default`: a virtual copy's rating is real but is never what -- these two queries ask for, and excluding them keeps the index roughly one -- entry per image rather than one per version. CREATE INDEX versions_judgement ON versions(image_id, rating, flag) WHERE is_default = 1; "#; const V2: &str = r#" -- A JPEG the camera wrote alongside a RAW of the same name is that RAW's own -- rendering, not a second photograph. Recording *which* RAW shadows it, rather -- than a bare flag, keeps the relationship usable: the JPEG is a ready-made -- preview for its RAW, and the pairing can be undone without a rescan. ALTER TABLE images ADD COLUMN shadowed_by INTEGER REFERENCES images(id) ON DELETE SET NULL; CREATE INDEX images_shadowed ON images(shadowed_by) WHERE shadowed_by IS NOT NULL; "#; const V1: &str = r#" -- Roots ------------------------------------------------------------------- CREATE TABLE roots ( id INTEGER PRIMARY KEY, kind TEXT NOT NULL, -- 'local' | 'saf' | 'remote' grant_blob BLOB, -- SAF persisted permission; NULL on Linux label TEXT NOT NULL, last_seen INTEGER, -- Bumped once per completed scan. Folders record the generation they were -- reached in; anything older was not reached and no longer exists. scan_generation INTEGER NOT NULL DEFAULT 0, -- One row per granted location. Without this, a rescan inserts a second -- root for the same folder and the library silently fragments across -- them — images split between roots, and pruning compares against the -- wrong generation. UNIQUE(kind, label) ); -- Folders: the unit of change detection, local and remote alike ----------- CREATE TABLE folders ( id INTEGER PRIMARY KEY, root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE, parent_id INTEGER REFERENCES folders(id) ON DELETE CASCADE, path TEXT NOT NULL, -- Remote: the propagating ETag that makes a no-op sync one request. etag TEXT, -- Local: directory mtime plus direct-entry count. mtime alone misses a -- paired create+delete inside one timestamp tick; the count narrows that. mtime INTEGER, entry_count INTEGER, scanned_generation INTEGER NOT NULL DEFAULT 0, UNIQUE(root_id, path) ); CREATE INDEX folders_parent ON folders(parent_id); -- Images ------------------------------------------------------------------ CREATE TABLE images ( id INTEGER PRIMARY KEY, root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE, folder_id INTEGER REFERENCES folders(id) ON DELETE CASCADE, source_ref TEXT NOT NULL, -- Expensive: requires reading the whole file. Computed only when -- something needs it (import dedup, reconnect-by-hash), never in a scan. content_hash TEXT, format TEXT, w INTEGER, h INTEGER, -- UTC seconds. NULL until EXIF is read, or if the file carries none. captured_at INTEGER, -- Minutes east of UTC. A photograph's timestamp is local to where it was -- taken; storing UTC alone makes a Tokyo shoot span two days in Paris. captured_offset INTEGER, camera TEXT, lens TEXT, iso INTEGER, aperture REAL, shutter REAL, availability INTEGER NOT NULL DEFAULT 0, file_size INTEGER, file_mtime INTEGER, -- 0 = nothing, 1 = stat-only, 2 = full EXIF. The grid is usable at 1. metadata_state INTEGER NOT NULL DEFAULT 0, sidecar_mtime INTEGER, added_at INTEGER NOT NULL, UNIQUE(root_id, source_ref) ); CREATE INDEX images_captured ON images(captured_at); CREATE INDEX images_folder ON images(folder_id); -- Partial: content_hash is NULL for most rows most of the time, and the -- non-NULL subset is exactly what reconnect and dedup query. CREATE INDEX images_hash ON images(content_hash) WHERE content_hash IS NOT NULL; -- Versions ---------------------------------------------------------------- CREATE TABLE versions ( id INTEGER PRIMARY KEY, image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE, uuid TEXT NOT NULL UNIQUE, name TEXT NOT NULL, is_default INTEGER NOT NULL DEFAULT 0, graph_hash TEXT, rating INTEGER NOT NULL DEFAULT 0, label INTEGER, flag INTEGER NOT NULL DEFAULT 0 ); CREATE INDEX versions_image ON versions(image_id); CREATE TABLE keywords ( version_id INTEGER NOT NULL REFERENCES versions(id) ON DELETE CASCADE, keyword TEXT NOT NULL, PRIMARY KEY(version_id, keyword) ); CREATE INDEX keywords_term ON keywords(keyword); -- Remote mapping ---------------------------------------------------------- CREATE TABLE remote ( image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE, -- oc:fileid — stable across server-side rename and move, so a move is not -- a re-download of 80 MB. file_id INTEGER NOT NULL, etag TEXT, sync_state INTEGER NOT NULL DEFAULT 0, remote_path TEXT ); CREATE UNIQUE INDEX remote_file ON remote(file_id); -- Collections ------------------------------------------------------------- CREATE TABLE collections ( id INTEGER PRIMARY KEY, -- Device-independent identity. The integer id is local and collides -- across devices; the UUID is what a cross-device merge keys on. uuid TEXT NOT NULL UNIQUE, name TEXT NOT NULL, parent_id INTEGER REFERENCES collections(id) ON DELETE CASCADE, kind INTEGER NOT NULL, -- 0 = manual, 1 = smart selector_json TEXT, -- smart only created INTEGER NOT NULL, -- Monotonic per collection, bumped on every local edit. Merge compares -- these rather than file mtimes, so a clock-skewed device cannot silently -- win. revision INTEGER NOT NULL DEFAULT 1, modified INTEGER NOT NULL, -- Tombstone. A deleted collection must outlive its deletion, or a merge -- with a device that still has it would resurrect it. deleted INTEGER NOT NULL DEFAULT 0 ); CREATE TABLE collection_members ( collection_id INTEGER NOT NULL REFERENCES collections(id) ON DELETE CASCADE, image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE, position INTEGER, -- manual ordering; NULL = by capture time added INTEGER NOT NULL, PRIMARY KEY(collection_id, image_id) ); CREATE INDEX members_image ON collection_members(image_id); -- Cache ------------------------------------------------------------------- CREATE TABLE cache ( id INTEGER PRIMARY KEY, version_id INTEGER REFERENCES versions(id) ON DELETE CASCADE, image_id INTEGER REFERENCES images(id) ON DELETE CASCADE, kind INTEGER NOT NULL, -- thumbnail | proxy | original resolution INTEGER, graph_hash TEXT, path TEXT NOT NULL, bytes INTEGER NOT NULL, last_used INTEGER NOT NULL ); CREATE INDEX cache_lru ON cache(last_used); CREATE TABLE cache_rules ( id INTEGER PRIMARY KEY, selector_json TEXT NOT NULL, tier INTEGER NOT NULL, priority INTEGER NOT NULL DEFAULT 0, enabled INTEGER NOT NULL DEFAULT 1 ); CREATE TABLE image_cache ( image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE, tier_actual INTEGER NOT NULL DEFAULT 0, -- Materialised rather than recomputed, so the grid can draw availability -- badges without evaluating every rule for every visible cell. tier_desired INTEGER NOT NULL DEFAULT 0, bytes INTEGER NOT NULL DEFAULT 0, last_used INTEGER, pinned_by_rule INTEGER REFERENCES cache_rules(id) ON DELETE SET NULL ); -- Jobs -------------------------------------------------------------------- CREATE TABLE jobs ( id INTEGER PRIMARY KEY, kind INTEGER NOT NULL, subject_id INTEGER, priority INTEGER NOT NULL DEFAULT 0, state INTEGER NOT NULL DEFAULT 0, -- 0=pending 1=running 2=failed attempts INTEGER NOT NULL DEFAULT 0, not_before INTEGER NOT NULL DEFAULT 0, payload TEXT, last_error TEXT, -- Coalescing. Enqueueing the same work twice updates one row rather than -- queueing it twice, which is what makes "enqueue on any change" safe to -- call liberally. UNIQUE(kind, subject_id) ); CREATE INDEX jobs_ready ON jobs(state, priority DESC, not_before); "#; #[cfg(test)] mod tests { use super::*; fn mem() -> Connection { let c = Connection::open_in_memory().unwrap(); configure(&c).unwrap(); c } /// How many rows a named backfill touched, ignoring the others. /// /// Asserting on the whole vector would couple every test to which other /// backfills happen to exist. fn backfilled(c: &Connection, what: &str) -> usize { backfill(c) .unwrap() .into_iter() .find(|(name, _)| *name == what) .map(|(_, n)| n) .unwrap_or(0) } /// Insert an image and return its id. fn image(c: &Connection, folder: Option, name: &str, format: &str) -> i64 { c.execute( "INSERT INTO images(root_id, folder_id, source_ref, format, added_at) VALUES (1, ?1, ?2, ?3, 0)", rusqlite::params![folder, name, format], ) .unwrap(); c.last_insert_rowid() } fn with_root() -> Connection { let c = mem(); migrate(&c).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); c.execute( "INSERT INTO folders(id, root_id, path) VALUES (1, 1, 'a'), (2, 1, 'b')", [], ) .unwrap(); c } #[test] fn a_jpeg_beside_its_raw_is_shadowed() { // The camera's own rendering of a frame, not a second photograph. let c = with_root(); let raw = image(&c, Some(1), "a/IMG_1234.CR2", "cr2"); let jpeg = image(&c, Some(1), "a/IMG_1234.JPG", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 1); let got: Option = c .query_row( "SELECT shadowed_by FROM images WHERE id = ?1", [jpeg], |r| r.get(0), ) .unwrap(); assert_eq!(got, Some(raw)); } #[test] fn extension_case_does_not_matter() { let c = with_root(); image(&c, Some(1), "a/IMG_1.cr2", "cr2"); image(&c, Some(1), "a/img_1.JPG", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 1); } #[test] fn a_standalone_jpeg_is_untouched() { // Scanned film has no RAW sibling and must stay visible — 2,656 of // them in the reference library. let c = with_root(); image(&c, Some(1), "a/SCAN_0001.jpg", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 0); } #[test] fn a_jpeg_in_a_different_folder_is_not_shadowed() { // Camera filenames wrap at IMG_9999, so the same stem recurs across // shoots (FR-CAT-11). Only a same-folder pair is safe to collapse. let c = with_root(); image(&c, Some(1), "a/IMG_1234.CR2", "cr2"); image(&c, Some(2), "b/IMG_1234.JPG", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 0); } #[test] fn a_raw_is_never_shadowed_by_a_jpeg() { // The relationship is one-way: the RAW is the photograph. let c = with_root(); let raw = image(&c, Some(1), "a/IMG_1.CR2", "cr2"); image(&c, Some(1), "a/IMG_1.JPG", "jpg"); backfill(&c).unwrap(); let got: Option = c .query_row("SELECT shadowed_by FROM images WHERE id = ?1", [raw], |r| { r.get(0) }) .unwrap(); assert_eq!(got, None); } #[test] fn backfill_is_idempotent() { // It runs on every open, so a second pass must find nothing to do. let c = with_root(); image(&c, Some(1), "a/IMG_1.CR2", "cr2"); image(&c, Some(1), "a/IMG_1.JPG", "jpg"); assert_eq!(backfilled(&c, "shadowed_by"), 1); assert_eq!(backfilled(&c, "shadowed_by"), 0, "second pass is a no-op"); } #[test] fn a_v1_catalog_gains_the_column_and_is_backfilled() { // The migration case that motivated this: rows already present when a // column is added are silently partial until something backfills them. let c = mem(); c.execute_batch(V1).unwrap(); c.pragma_update(None, "user_version", 1).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); c.execute( "INSERT INTO images(root_id, source_ref, format, added_at) VALUES (1, 'IMG_9.CR2', 'cr2', 0), (1, 'IMG_9.JPG', 'jpg', 0)", [], ) .unwrap(); assert_eq!(migrate(&c).unwrap(), 1, "migrated from v1"); assert_eq!(backfilled(&c, "shadowed_by"), 1); } #[test] fn a_v4_catalog_gains_the_pinning_columns() { // TRACES: FR-NC-6a // An existing library must not have to be rescanned to gain offline // pinning. The rows are already there; only the columns are new. let c = mem(); c.execute_batch(V1).unwrap(); c.execute_batch(V2).unwrap(); c.execute_batch(V3).unwrap(); c.execute_batch(V4).unwrap(); c.pragma_update(None, "user_version", 4).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (7, 1, 'IMG_7.CR2', 0)", [], ) .unwrap(); // A cache row written before pinning existed. c.execute( "INSERT INTO image_cache(image_id, tier_actual, bytes) VALUES (7, 2, 100)", [], ) .unwrap(); assert_eq!(migrate(&c).unwrap(), 4, "migrated from v4"); // The pre-existing row survives, and defaults to unpinned — the safe // direction, since claiming a pin nobody made would exempt it from // eviction for ever. let (pinned, bytes): (i64, i64) = c .query_row( "SELECT pinned, bytes FROM image_cache WHERE image_id = 7", [], |r| Ok((r.get(0)?, r.get(1)?)), ) .unwrap(); assert_eq!(pinned, 0); assert_eq!(bytes, 100, "the existing row is untouched"); } #[test] fn a_v5_catalog_keeps_its_keywords_and_gains_their_identities() { // TRACES: FR-CAT-5 // The migration case that matters here: a library keyworded by an // import or an older build already has assignment rows, and they must // survive into the vocabulary rather than being left searchable but // invisible. let c = mem(); for step in [V1, V2, V3, V4, V5] { c.execute_batch(step).unwrap(); } c.pragma_update(None, "user_version", 5).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (7, 1, 'IMG_7.CR3', 0)", [], ) .unwrap(); c.execute( "INSERT INTO versions(id, image_id, uuid, name, is_default) VALUES (1, 7, 'v-7', 'Default', 1)", [], ) .unwrap(); c.execute( "INSERT INTO keywords(version_id, keyword) VALUES (1, 'puffin')", [], ) .unwrap(); assert_eq!(migrate(&c).unwrap(), 5, "migrated from v5"); assert_eq!(backfilled(&c, "keyword_terms"), 1); let name: String = c .query_row("SELECT name FROM keyword_terms", [], |r| r.get(0)) .unwrap(); assert_eq!(name, "puffin"); // The assignment is untouched — it is the durable fact, and the term // row is only its identity. let n: i64 = c .query_row("SELECT count(*) FROM keywords", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 1); // It runs on every open, so a second pass must find nothing to do. assert_eq!(backfilled(&c, "keyword_terms"), 0); } #[test] fn two_devices_may_both_hold_a_term_of_the_same_name() { // Deliberately not a unique index. Two devices each typing "Iceland" // is the ordinary case, and a constraint would abort the merge // transaction at exactly the moment they first sync. let c = mem(); migrate(&c).unwrap(); c.execute( "INSERT INTO keyword_terms(uuid, name, created, revision, modified) VALUES ('a', 'Iceland', 0, 1, 1), ('b', 'Iceland', 0, 1, 1)", [], ) .unwrap(); let n: i64 = c .query_row("SELECT count(*) FROM keyword_terms", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 2); } #[test] fn stems_ignore_directories_containing_dots() { assert_eq!(stem_of("2026.08/IMG_1.CR2"), "IMG_1"); assert_eq!(stem_of("IMG_1.CR2"), "IMG_1"); assert_eq!(stem_of("noextension"), "noextension"); // A dotfile is all stem, not an empty name with an extension. assert_eq!(stem_of(".hidden"), ".hidden"); } #[test] fn migrate_creates_schema_at_current_version() { let c = mem(); assert_eq!(migrate(&c).unwrap(), 0); let v: i64 = c .query_row("PRAGMA user_version", [], |r| r.get(0)) .unwrap(); assert_eq!(v, SCHEMA_VERSION); } #[test] fn migrate_is_idempotent() { let c = mem(); migrate(&c).unwrap(); // Re-running must not error or duplicate anything — NFR-R5 requires // idempotency on retry, since a migration can be interrupted. assert_eq!(migrate(&c).unwrap(), SCHEMA_VERSION); } #[test] fn refuses_a_catalog_from_a_newer_build() { let c = mem(); migrate(&c).unwrap(); c.pragma_update(None, "user_version", SCHEMA_VERSION + 1) .unwrap(); // Opening it read-write would corrupt data this build cannot // represent. Refusing is the specified behaviour (NFR-R5). assert!(matches!( migrate(&c), Err(CatalogError::SchemaTooNew { .. }) )); } #[test] fn foreign_keys_cascade_from_root_to_image() { let c = mem(); migrate(&c).unwrap(); c.execute( "INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')", [], ) .unwrap(); c.execute( "INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)", [], ) .unwrap(); c.execute("DELETE FROM roots WHERE id = 1", []).unwrap(); let n: i64 = c .query_row("SELECT count(*) FROM images", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 0, "images must not outlive their root"); } #[test] fn job_uniqueness_coalesces_rather_than_duplicating() { let c = mem(); migrate(&c).unwrap(); for _ in 0..5 { c.execute( "INSERT INTO jobs(kind, subject_id, priority) VALUES (1, 42, 0) ON CONFLICT(kind, subject_id) DO UPDATE SET priority = max(priority, excluded.priority)", [], ) .unwrap(); } let n: i64 = c .query_row("SELECT count(*) FROM jobs", [], |r| r.get(0)) .unwrap(); assert_eq!(n, 1, "five enqueues of the same work is one job"); } }