Files
DarkRoom/core/dr-catalog/src/schema.rs
T
dtourolleandClaude Opus 5 26a1eb7e28 Record that face detection has run, not just what it found
An image with no faces in it was indistinguishable from one that had
never been looked at, so every landscape, still life and document scan in
the library was re-detected on every pass, for ever. In a real library
that is most of it: on the 23,527-image test library, 64 of the first 110
images indexed contain no face at all.

Schema v9 adds face_index, a run marker per (image, model) carrying the
face count and the proxy edge it read. Keyed on the model, so a model
change puts every image back in the queue by itself.

That makes a coverage figure possible, which is the thing a user actually
wants to see. The audit also splits the outstanding set by whether a
proxy exists, because 23,417 awaiting a proxy and 110 ready to index are
different problems, and telling the user to run indexing again would not
fix the first.

The Identity screen gains Index faces, Stop, and the coverage line.
examples/face_index.rs is the same check and sweep without a window,
which is the right shape for an overnight pass.

Measured on the real library in release: 3.5 images/second, 110 images
and 125 faces in 30 seconds, and a second run correctly finds nothing
left to do.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-26 22:13:41 +02:00

1153 lines
46 KiB
Rust

//! TRACES: FR-CAT-2 | NFR-R5
//! Schema definition and forward-only migrations.
//!
//! The catalog is an *index*, not a source of truth (ARCH §6.12) — it is
//! deletable and rebuildable from sources plus sidecars. That is what makes
//! migration failure survivable, and why the recovery path is the normal
//! mechanism rather than a last resort.
//!
//! Migrations are forward-only, transactional, and idempotent on retry
//! (NFR-R5). The app refuses to open a catalog newer than it understands
//! rather than corrupting it.
use rusqlite::Connection;
use crate::error::CatalogError;
/// Schema version this build writes and understands.
pub const SCHEMA_VERSION: i64 = 9;
/// Apply migrations up to [`SCHEMA_VERSION`].
///
/// Returns the version migrated from, so callers can log or back up before a
/// real migration (NFR-R2 requires a backup before schema change).
pub fn migrate(conn: &Connection) -> Result<i64, CatalogError> {
let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
if from > SCHEMA_VERSION {
return Err(CatalogError::SchemaTooNew {
found: from,
supported: SCHEMA_VERSION,
});
}
if from == SCHEMA_VERSION {
return Ok(from);
}
// Each step runs in its own transaction so a failure leaves the catalog
// at a coherent version rather than half-migrated.
if from < 1 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V1)?;
tx.pragma_update(None, "user_version", 1)?;
tx.commit()?;
}
if from < 2 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V2)?;
tx.pragma_update(None, "user_version", 2)?;
tx.commit()?;
}
if from < 3 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V3)?;
tx.pragma_update(None, "user_version", 3)?;
tx.commit()?;
}
if from < 4 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V4)?;
tx.pragma_update(None, "user_version", 4)?;
tx.commit()?;
}
if from < 5 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V5)?;
tx.pragma_update(None, "user_version", 5)?;
tx.commit()?;
}
if from < 6 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V6)?;
tx.pragma_update(None, "user_version", 6)?;
tx.commit()?;
}
if from < 7 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V7)?;
tx.pragma_update(None, "user_version", 7)?;
tx.commit()?;
}
if from < 8 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V8)?;
tx.pragma_update(None, "user_version", 8)?;
tx.commit()?;
}
if from < 9 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V9)?;
tx.pragma_update(None, "user_version", 9)?;
tx.commit()?;
}
Ok(from)
}
/// Recompute columns a migration added, for rows that predate it.
///
/// A migration adds a column with a default; it cannot know what the value
/// *should* be for the rows already present. Without a backfill those rows are
/// silently partial — present, queryable, and wrong — which is worse than
/// missing, because nothing signals that they need attention.
///
/// Cheap enough to run on every open: each pass is one indexed UPDATE, and
/// re-running it is a no-op once the values are already right.
///
/// Returns how many rows each backfill touched, for logging.
pub fn backfill(conn: &Connection) -> Result<Vec<(&'static str, usize)>, CatalogError> {
let mut out = Vec::new();
// v2: `shadowed_by`. A JPEG sitting beside a RAW of the same name is the
// camera's own rendering of that frame, not a second photograph, so it is
// hidden from the grid, the timeline and the sweep.
let n = pair_raw_and_jpeg(conn)?;
if n > 0 {
out.push(("shadowed_by", n));
}
// v3: every image needs a default version to carry its rating and flag.
// Libraries scanned before ratings existed have images and no versions at
// all, so there was nowhere for a judgement to go — see
// [`crate::rating`]. Backfilled rather than migrated in SQL because the
// UUID per row is the cross-device merge identity and must be generated,
// not derived.
let n = crate::rating::ensure_default_versions(conn)?;
if n > 0 {
out.push(("default_versions", n));
}
// v6: a vocabulary row for every word some image already carries.
//
// Three ways a catalog arrives holding assignments with no term behind
// them, and all three are normal rather than exceptional: a library
// keyworded by a build that predates this table, a catalog rebuilt from
// sidecars (which carry the word and not the identity), and an import from
// Lightroom or darktable (FR-CAT-14). Without this the words are
// searchable but absent from the vocabulary list, which reads as the
// keywords having been lost.
let n = crate::keywords::adopt_orphan_terms(conn)?;
if n > 0 {
out.push(("keyword_terms", n));
}
Ok(out)
}
/// Connection setup applied on every open, migration or not.
///
/// WAL is required by NFR-R1: it survives power loss without corruption, and
/// it lets a background job write while the grid reads.
pub fn configure(conn: &Connection) -> Result<(), CatalogError> {
conn.pragma_update(None, "journal_mode", "WAL")?;
// NORMAL rather than FULL: with WAL this is durable across process death
// (which is what FR-PLAT-AND-3 cares about) and only risks the last
// transaction on power loss. The catalog is rebuildable; the sidecars are
// not, and they are written separately with their own fsync discipline.
conn.pragma_update(None, "synchronous", "NORMAL")?;
conn.pragma_update(None, "foreign_keys", true)?;
// A scan touching thousands of rows is transient; let SQLite spill to
// memory rather than materialising temp b-trees on disk.
conn.pragma_update(None, "temp_store", "MEMORY")?;
Ok(())
}
/// The v1 schema rewritten to target an attached database.
///
/// Needed because a downloaded remote catalog is `ATTACH`ed under its own
/// schema name before merging, and tests build one from scratch. SQLite has no
/// "create these tables over there" form, so the names are rewritten.
///
/// The rewrite is textual and therefore only as good as the naming discipline
/// in [`V1`]: every `CREATE TABLE`/`CREATE INDEX` must name its object
/// unqualified, which they do.
pub fn v1_for_attached(schema_name: &str) -> String {
rewrite_for_attached(V1, schema_name)
// REFERENCES within an attached schema resolve to that schema already,
// so foreign keys need no rewriting — but the ON clause of an index
// does, and `CREATE INDEX x.name ON table` is the correct form.
}
/// Every table this build knows about, rewritten to target an attached
/// database.
///
/// [`v1_for_attached`] is kept alongside this rather than replaced by it: a
/// remote catalog written by an older build genuinely has only the v1 tables,
/// and the merge has to keep working against one (see
/// [`crate::merge::merge_keywords`]). Building that case in a test needs a way
/// to say "v1 and no more".
///
/// Only the migrations that *create* objects appear here. V2 through V5 are
/// `ALTER TABLE ... ADD COLUMN`, and the columns they add are local index
/// state — shadowing, trashing, cache pinning — that a merge never reads
/// across the attachment.
///
/// V7 creates an object and is still excluded, which is the one exception to
/// that rule and not an oversight: it indexes `shadowed_by` and `trashed_at`,
/// the very columns V2 through V5 add and this function leaves out, so
/// creating it over there would fail on columns that are not there. Nothing is
/// lost by its absence — it exists to make the *grid* page quickly, and the
/// grid never reads across an attachment.
pub fn for_attached(schema_name: &str) -> String {
format!(
"{}\n{}",
rewrite_for_attached(V1, schema_name),
rewrite_for_attached(V6, schema_name)
)
}
/// Qualify every object a `CREATE` statement names with `schema_name`.
///
/// The rewrite is textual and therefore only as good as the naming discipline
/// in the batches it is given: every `CREATE TABLE`/`CREATE INDEX` must name
/// its object unqualified, which they do.
fn rewrite_for_attached(sql: &str, schema_name: &str) -> String {
sql.replace("CREATE TABLE ", &format!("CREATE TABLE {schema_name}."))
.replace("CREATE INDEX ", &format!("CREATE INDEX {schema_name}."))
.replace(
"CREATE UNIQUE INDEX ",
&format!("CREATE UNIQUE INDEX {schema_name}."),
)
}
/// Mark each JPEG that sits beside a RAW of the same name.
///
/// Matched on folder plus stem, case-insensitively. Same-folder is what makes
/// this safe: cameras write the pair side by side, and matching across folders
/// would risk pairing unrelated frames, since camera filenames wrap at
/// IMG_9999 (FR-CAT-11).
///
/// Done in Rust rather than SQL because the comparison needs a filename stem,
/// and SQLite has no such function without enabling `rusqlite/functions` —
/// a dependency feature for one string operation, whose SQL spelling would be
/// unreadable and would mishandle names with no extension.
fn pair_raw_and_jpeg(conn: &Connection) -> Result<usize, CatalogError> {
use std::collections::HashMap;
// (folder, lowercase stem) -> RAW id, built in one pass over the RAWs.
let mut raws: HashMap<(Option<i64>, String), i64> = HashMap::new();
{
let mut stmt = conn.prepare(
"SELECT id, folder_id, source_ref FROM images
WHERE lower(format) IN
('cr2','cr3','nef','arw','raf','rw2','orf','dng')",
)?;
let rows = stmt.query_map([], |r| {
Ok((
r.get::<_, i64>(0)?,
r.get::<_, Option<i64>>(1)?,
r.get::<_, String>(2)?,
))
})?;
for row in rows {
let (id, folder, path) = row?;
raws.insert((folder, stem_of(&path).to_ascii_lowercase()), id);
}
}
if raws.is_empty() {
return Ok(0);
}
let pairs: Vec<(i64, i64)> = {
let mut stmt = conn.prepare(
"SELECT id, folder_id, source_ref FROM images
WHERE lower(format) IN ('jpg','jpeg') AND shadowed_by IS NULL",
)?;
let rows = stmt.query_map([], |r| {
Ok((
r.get::<_, i64>(0)?,
r.get::<_, Option<i64>>(1)?,
r.get::<_, String>(2)?,
))
})?;
rows.filter_map(|row| {
let (id, folder, path) = row.ok()?;
let raw = raws.get(&(folder, stem_of(&path).to_ascii_lowercase()))?;
Some((id, *raw))
})
.collect()
};
let tx = conn.unchecked_transaction()?;
for (jpeg, raw) in &pairs {
tx.execute(
"UPDATE images SET shadowed_by = ?2 WHERE id = ?1",
[jpeg, raw],
)?;
}
tx.commit()?;
Ok(pairs.len())
}
/// A filename without its extension.
///
/// Only the final path component, and only its last dot — a directory
/// containing a dot must not truncate the name.
fn stem_of(path: &str) -> &str {
let name = path.rsplit(['/', ':']).next().unwrap_or(path);
match name.rsplit_once('.') {
Some((stem, _)) if !stem.is_empty() => stem,
_ => name,
}
}
/// TRACES: FR-CAT-4 | NFR-P5
/// The order the grid reads in, as an index.
///
/// # What this is for
///
/// Every window the grid loads is `ORDER BY ... LIMIT n OFFSET k`, and without
/// an index that matches the ordering SQLite answers it by sorting the whole
/// library into a temp b-tree and then discarding the first `k` rows. Measured
/// on 24,000 images at offset 20,000, one window read cost 15 ms — a frame
/// budget of 16.7 ms, spent inside the scroll handler, several times per
/// screenful. That is the jitter.
///
/// With this index the same read is a walk along it: 0.36 ms.
///
/// # Why the shape is what it is
///
/// `captured_at IS NULL` leads, because [`crate::library`]'s `GRID_ORDER` does
/// — undated images sort last, and an ordinary index on `captured_at` cannot
/// answer that, since the expression is not a column. SQLite indexes
/// expressions, so it is spelled out here exactly as the query spells it; the
/// two must stay identical or the planner silently falls back to sorting and
/// the cost comes back with no other symptom.
///
/// `source_ref` is included because it breaks ties in the same ordering, and an
/// index that stopped at `captured_at` would leave a sort for the ties.
///
/// # Partial, on the same predicate the grid filters by
///
/// The grid never lists shadowed or trashed rows, so an index carrying them
/// would be larger than the question ever asks about, and — more to the point —
/// a partial index is only usable when its `WHERE` is implied by the query's,
/// which is what makes this one apply to the grid's reads and to nothing else.
///
/// # It does not cover the rating filter or a collection scope
///
/// Both narrow the walk rather than reorder it, so the index still supplies the
/// ordering and SQLite tests the extra predicate per row. That is the cheap
/// direction: the expensive part was never the filtering, it was the sort.
const V9: &str = r#"
-- TRACES: FR-CULL-8
-- A record that face detection has *run* on an image, distinct from what it
-- found.
--
-- # Why the faces table cannot answer this
--
-- Without this, "has this image been indexed" is asked as "does it have any
-- faces", and those are not the same question. **A photograph with no faces in
-- it is indistinguishable from one that has never been looked at**, so every
-- indexing pass re-examines every landscape, every still life and every
-- document scan in the library, for ever. In a typical personal library that is
-- most of it: the pass never converges, and the cost is paid again on every
-- run rather than once.
--
-- It also makes a coverage figure possible, which is the thing a user actually
-- wants to see — "4,812 of 5,000 images indexed" — where counting face rows
-- can only ever report how many faces exist.
--
-- # Why it is keyed on the model
--
-- Embeddings from different models are not comparable, so a model change has
-- to re-index. Keying the marker on `(image_id, model_id)` makes that
-- automatic: new model, no marker, image comes back into the queue. The id
-- names the whole pipeline -- detector and embedder together -- because
-- changing either changes what is found.
CREATE TABLE face_index (
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
model_id TEXT NOT NULL,
indexed_at INTEGER NOT NULL,
-- Zero is a real and common answer, and recording it is the entire point.
faces_found INTEGER NOT NULL,
-- Long edge of the proxy this ran against. A face too small to detect at
-- 1024 may be findable at 2048, so a library whose proxies grow can
-- re-index the images that stand to gain instead of all of them.
source_edge INTEGER NOT NULL,
PRIMARY KEY (image_id, model_id)
);
CREATE INDEX face_index_model ON face_index(model_id);
"#;
const V8: &str = r#"
-- TRACES: FR-CULL-8 | FR-CULL-9 | FR-CULL-10 | FR-CULL-11 | FR-CULL-12 | NFR-SEC-5
-- People and faces (docs/faces.md, docs/catalog.md §10).
--
-- Everything here is **derived data** except one column. Faces, landmarks,
-- embeddings, cluster assignments and suggestions are all reproducible by
-- re-indexing and are never written to a sidecar (FR-CULL-12); a person's
-- *name*, once the user has confirmed it, is a human judgement of the same
-- class as a rating and travels with the photograph.
--
-- That asymmetry is the whole design: deleting the catalog costs an afternoon
-- of re-indexing and loses nothing the user typed (ARCH §6.12).
CREATE TABLE people (
id INTEGER PRIMARY KEY,
-- Merge identity, not the name. Two devices that independently name the
-- same cluster produce two people; merging them keys on this, exactly as
-- collections do (FR-CAT-7, ARCH §6.3).
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
-- Tombstone-by-redirect. A merged person must outlive its merge or a
-- device that still holds it resurrects it on the next sync -- the same
-- hazard collections have, solved the same way.
merged_into INTEGER REFERENCES people(id) ON DELETE SET NULL,
created INTEGER NOT NULL,
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL
);
CREATE TABLE faces (
id INTEGER PRIMARY KEY,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
-- Normalised to the image's long edge, so a face survives the proxy it was
-- found on being evicted and regenerated at another resolution. Storing
-- pixels would bind a face to a resolution the cache is entitled to change.
x REAL NOT NULL, y REAL NOT NULL, w REAL NOT NULL, h REAL NOT NULL,
landmarks BLOB NOT NULL, -- 5 x (x, y) f32, normalised likewise
detector_confidence REAL NOT NULL,
embedding BLOB NOT NULL, -- 512 x f16, L2-normalised
-- Source pixels across the aligned 112x112 crop (docs/faces.md §7).
--
-- Not cosmetic: it is the honest quality signal for the UI, a feature in
-- the §8 calibration -- FR-CULL-9 names face size as an axis along which an
-- uncalibrated similarity misbehaves -- and the selector a later
-- higher-resolution re-embedding pass would run on.
crop_px REAL NOT NULL,
-- Which model produced this embedding.
--
-- The one mistake in this subsystem that yields plausible-looking garbage
-- rather than an error: embeddings from different models are not
-- comparable. Storing the model with the vector makes a model change
-- detectable and re-indexable instead of quietly poisoning every
-- similarity in the library.
model_id TEXT NOT NULL,
detected_at INTEGER NOT NULL
);
CREATE INDEX faces_image ON faces(image_id);
CREATE INDEX faces_model ON faces(model_id);
CREATE TABLE face_person (
face_id INTEGER PRIMARY KEY REFERENCES faces(id) ON DELETE CASCADE,
person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE,
-- Calibrated P(this face is this person), never a raw cosine (FR-CULL-9).
probability REAL NOT NULL,
-- The user said so. Never overwritten by a later inference pass.
--
-- A column rather than a probability of 1.0, because a confirmation is a
-- different kind of fact from a confident guess and collapsing them loses
-- the ability to recompute suggestions without touching user data.
confirmed INTEGER NOT NULL DEFAULT 0
);
CREATE INDEX face_person_person ON face_person(person_id, confirmed);
-- Faces the user has explicitly said are NOT a given person.
--
-- Needed because rejection is not the absence of an assignment: without it,
-- the next clustering pass re-suggests exactly the face the user just pushed
-- away, and the tool feels broken. Same reasoning as `confirmed` -- a
-- judgement is user data (FR-CULL-12) whichever direction it points.
CREATE TABLE face_person_rejected (
face_id INTEGER NOT NULL REFERENCES faces(id) ON DELETE CASCADE,
person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE,
PRIMARY KEY (face_id, person_id)
);
-- The FR-CULL-9 calibration, fitted from this library's own faces.
--
-- One row per model, because the fit is a property of the embedding space and
-- a library indexed across a model change holds two. `face_set_hash` is what
-- makes a stale fit detectable: a materially changed library recomputes rather
-- than trusting numbers derived from a set that no longer exists.
CREATE TABLE face_calibration (
model_id TEXT PRIMARY KEY,
-- P(same) = sigmoid(a*cos + b + w_size*log2(min(crop_px)) + log_prior_odds)
a REAL NOT NULL,
b REAL NOT NULL,
w_size REAL NOT NULL DEFAULT 0.0,
-- Whether the fit is usable at all. When it is not, the UI says the
-- confidence is unavailable; it does not present an untuned default as
-- though it were measured (FR-CULL-9).
valid INTEGER NOT NULL DEFAULT 0,
positive_pairs INTEGER NOT NULL DEFAULT 0,
negative_pairs INTEGER NOT NULL DEFAULT 0,
face_set_hash TEXT NOT NULL,
fitted_at INTEGER NOT NULL
);
"#;
const V7: &str = r#"
CREATE INDEX images_grid_order
ON images(captured_at IS NULL, captured_at, source_ref)
WHERE shadowed_by IS NULL AND trashed_at IS NULL;
"#;
const V6: &str = r#"
-- TRACES: FR-CAT-5 | FR-CAT-6 | FR-NC-9
-- Keywords gain an identity, so that renaming and deleting one can cross
-- between devices.
--
-- The v1 `keywords` table is the *assignment*: one row per (version, word),
-- and the word is stored as text. That stays exactly as it is, and this
-- migration adds nothing to it, for a reason that is easy to get backwards.
--
-- # Why assignments keep the text rather than pointing at a row here
--
-- The catalog is a rebuildable index (ARCH §6.12). What an image is keyworded
-- with is authoritative in the sidecar and in XMP `dc:subject` (FR-CAT-13),
-- and both of those carry a *string*. Rewriting the join to reference
-- `keyword_terms(id)` would mean a catalog rebuilt from sidecars had to invent
-- term rows before it could record a single assignment, and an integer that
-- means nothing on the other device would sit where the durable fact belongs.
-- It would also break `crate::query`, which matches `kw.keyword` directly and
-- must keep hitting `keywords_term` on a 50k library (FR-CAT-6).
--
-- So the text is the fact and this table is the *identity*: it exists to give
-- a rename and a deletion something a merge can key on, and to let a keyword
-- exist in the vocabulary before any photograph carries it.
CREATE TABLE keyword_terms (
id INTEGER PRIMARY KEY,
-- Device-independent identity, as `collections.uuid` is. The integer id is
-- local and collides across devices.
uuid TEXT NOT NULL UNIQUE,
-- The word itself, and the value written into every assignment row.
name TEXT NOT NULL,
created INTEGER NOT NULL,
-- Monotonic, bumped on every local edit. `crate::merge` compares these
-- rather than timestamps, so a clock-skewed device cannot silently win.
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL,
-- Tombstone, so a merge against a device that still holds the keyword does
-- not resurrect it.
deleted INTEGER NOT NULL DEFAULT 0
);
-- Deliberately **not** UNIQUE.
--
-- Two devices that each type "Iceland" create two rows with two uuids, and
-- both are correct until they meet. A unique constraint would abort the merge
-- transaction at exactly that moment — the ordinary case, not a corner one.
-- Uniqueness is instead reached by convergence: `crate::keywords::create`
-- resolves an existing name locally, and `crate::keywords::fuse_duplicates`
-- collapses a cross-device pair onto the lexicographically smaller uuid, which
-- both devices compute identically without talking to each other.
--
-- Partial on `deleted = 0` because every lookup here is a live one: the
-- vocabulary list, the resolve-by-name in `create`, and the fuse pass all
-- exclude tombstones, and including them would grow the index with every
-- keyword the library has ever had rather than with the ones it has.
CREATE INDEX keyword_terms_name ON keyword_terms(name) WHERE deleted = 0;
"#;
const V5: &str = r#"
-- TRACES: FR-NC-6a | FR-CAT-9 | NFR-RES-4
-- Offline availability: what is kept, why it is kept, and where it lives.
--
-- `pinned` separates a promise from a convenience, and the distinction has to
-- be a *column* rather than something inferred from `pinned_by_rule`. A pin is
-- the user saying "this collection comes with me"; a passively cached original
-- is the app noticing they opened something. Only the second is evictable, so
-- the eviction query has to be able to ask the question directly — and it has
-- to keep answering correctly for an image whose pinning rule was since
-- deleted, which `pinned_by_rule` alone cannot do because it is
-- ON DELETE SET NULL.
ALTER TABLE image_cache ADD COLUMN pinned INTEGER NOT NULL DEFAULT 0;
-- Where the cached original actually is, relative to the cache directory.
-- Relative rather than absolute: the library moves between machines and
-- between an app sandbox and a user directory, and an absolute path baked in
-- at download time would break on every one of those.
ALTER TABLE image_cache ADD COLUMN path TEXT;
-- Eviction reads exactly this: unpinned rows, oldest use first. Partial on
-- `pinned = 0` because pinned rows are never candidates and including them
-- would make the index proportional to the whole library rather than to the
-- passive cache.
CREATE INDEX image_cache_evictable ON image_cache(last_used)
WHERE pinned = 0;
"#;
const V4: &str = r#"
-- TRACES: FR-CAT-15
-- Soft delete. A trashed image is a real file that has been *moved* to a trash
-- folder under the library root, not a row hidden by a flag: the catalog is a
-- rebuildable index (ARCH §6.12), so a flag alone would evaporate the moment
-- the catalog was deleted and every trashed photograph would return.
--
-- `source_ref` follows the file to its new path, because that is where the bytes
-- now are and every fetch resolves through it. `trashed_from` remembers where it
-- came from, which is the only way a restore can put it back — the trash is flat
-- and the original folder structure is not recoverable from the trashed path.
ALTER TABLE images ADD COLUMN trashed_at INTEGER;
ALTER TABLE images ADD COLUMN trashed_from TEXT;
-- Partial: almost no rows are trashed, and the grid's "not trashed" predicate is
-- answered by the absence of an entry rather than by scanning every image.
CREATE INDEX images_trashed ON images(trashed_at) WHERE trashed_at IS NOT NULL;
"#;
const V3: &str = r#"
-- Ratings and flags are read per grid window and counted for the filter bar's
-- histogram, both of which key on the *default* version. Without this the
-- histogram is a full scan of `versions` on every judgement.
--
-- Partial on `is_default`: a virtual copy's rating is real but is never what
-- these two queries ask for, and excluding them keeps the index roughly one
-- entry per image rather than one per version.
CREATE INDEX versions_judgement ON versions(image_id, rating, flag)
WHERE is_default = 1;
"#;
const V2: &str = r#"
-- A JPEG the camera wrote alongside a RAW of the same name is that RAW's own
-- rendering, not a second photograph. Recording *which* RAW shadows it, rather
-- than a bare flag, keeps the relationship usable: the JPEG is a ready-made
-- preview for its RAW, and the pairing can be undone without a rescan.
ALTER TABLE images ADD COLUMN shadowed_by INTEGER REFERENCES images(id) ON DELETE SET NULL;
CREATE INDEX images_shadowed ON images(shadowed_by) WHERE shadowed_by IS NOT NULL;
"#;
const V1: &str = r#"
-- Roots -------------------------------------------------------------------
CREATE TABLE roots (
id INTEGER PRIMARY KEY,
kind TEXT NOT NULL, -- 'local' | 'saf' | 'remote'
grant_blob BLOB, -- SAF persisted permission; NULL on Linux
label TEXT NOT NULL,
last_seen INTEGER,
-- Bumped once per completed scan. Folders record the generation they were
-- reached in; anything older was not reached and no longer exists.
scan_generation INTEGER NOT NULL DEFAULT 0,
-- One row per granted location. Without this, a rescan inserts a second
-- root for the same folder and the library silently fragments across
-- them — images split between roots, and pruning compares against the
-- wrong generation.
UNIQUE(kind, label)
);
-- Folders: the unit of change detection, local and remote alike -----------
CREATE TABLE folders (
id INTEGER PRIMARY KEY,
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
parent_id INTEGER REFERENCES folders(id) ON DELETE CASCADE,
path TEXT NOT NULL,
-- Remote: the propagating ETag that makes a no-op sync one request.
etag TEXT,
-- Local: directory mtime plus direct-entry count. mtime alone misses a
-- paired create+delete inside one timestamp tick; the count narrows that.
mtime INTEGER,
entry_count INTEGER,
scanned_generation INTEGER NOT NULL DEFAULT 0,
UNIQUE(root_id, path)
);
CREATE INDEX folders_parent ON folders(parent_id);
-- Images ------------------------------------------------------------------
CREATE TABLE images (
id INTEGER PRIMARY KEY,
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
folder_id INTEGER REFERENCES folders(id) ON DELETE CASCADE,
source_ref TEXT NOT NULL,
-- Expensive: requires reading the whole file. Computed only when
-- something needs it (import dedup, reconnect-by-hash), never in a scan.
content_hash TEXT,
format TEXT,
w INTEGER,
h INTEGER,
-- UTC seconds. NULL until EXIF is read, or if the file carries none.
captured_at INTEGER,
-- Minutes east of UTC. A photograph's timestamp is local to where it was
-- taken; storing UTC alone makes a Tokyo shoot span two days in Paris.
captured_offset INTEGER,
camera TEXT,
lens TEXT,
iso INTEGER,
aperture REAL,
shutter REAL,
availability INTEGER NOT NULL DEFAULT 0,
file_size INTEGER,
file_mtime INTEGER,
-- 0 = nothing, 1 = stat-only, 2 = full EXIF. The grid is usable at 1.
metadata_state INTEGER NOT NULL DEFAULT 0,
sidecar_mtime INTEGER,
added_at INTEGER NOT NULL,
UNIQUE(root_id, source_ref)
);
CREATE INDEX images_captured ON images(captured_at);
CREATE INDEX images_folder ON images(folder_id);
-- Partial: content_hash is NULL for most rows most of the time, and the
-- non-NULL subset is exactly what reconnect and dedup query.
CREATE INDEX images_hash ON images(content_hash) WHERE content_hash IS NOT NULL;
-- Versions ----------------------------------------------------------------
CREATE TABLE versions (
id INTEGER PRIMARY KEY,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
is_default INTEGER NOT NULL DEFAULT 0,
graph_hash TEXT,
rating INTEGER NOT NULL DEFAULT 0,
label INTEGER,
flag INTEGER NOT NULL DEFAULT 0
);
CREATE INDEX versions_image ON versions(image_id);
CREATE TABLE keywords (
version_id INTEGER NOT NULL REFERENCES versions(id) ON DELETE CASCADE,
keyword TEXT NOT NULL,
PRIMARY KEY(version_id, keyword)
);
CREATE INDEX keywords_term ON keywords(keyword);
-- Remote mapping ----------------------------------------------------------
CREATE TABLE remote (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
-- oc:fileid — stable across server-side rename and move, so a move is not
-- a re-download of 80 MB.
file_id INTEGER NOT NULL,
etag TEXT,
sync_state INTEGER NOT NULL DEFAULT 0,
remote_path TEXT
);
CREATE UNIQUE INDEX remote_file ON remote(file_id);
-- Collections -------------------------------------------------------------
CREATE TABLE collections (
id INTEGER PRIMARY KEY,
-- Device-independent identity. The integer id is local and collides
-- across devices; the UUID is what a cross-device merge keys on.
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
parent_id INTEGER REFERENCES collections(id) ON DELETE CASCADE,
kind INTEGER NOT NULL, -- 0 = manual, 1 = smart
selector_json TEXT, -- smart only
created INTEGER NOT NULL,
-- Monotonic per collection, bumped on every local edit. Merge compares
-- these rather than file mtimes, so a clock-skewed device cannot silently
-- win.
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL,
-- Tombstone. A deleted collection must outlive its deletion, or a merge
-- with a device that still has it would resurrect it.
deleted INTEGER NOT NULL DEFAULT 0
);
CREATE TABLE collection_members (
collection_id INTEGER NOT NULL REFERENCES collections(id) ON DELETE CASCADE,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
position INTEGER, -- manual ordering; NULL = by capture time
added INTEGER NOT NULL,
PRIMARY KEY(collection_id, image_id)
);
CREATE INDEX members_image ON collection_members(image_id);
-- Cache -------------------------------------------------------------------
CREATE TABLE cache (
id INTEGER PRIMARY KEY,
version_id INTEGER REFERENCES versions(id) ON DELETE CASCADE,
image_id INTEGER REFERENCES images(id) ON DELETE CASCADE,
kind INTEGER NOT NULL, -- thumbnail | proxy | original
resolution INTEGER,
graph_hash TEXT,
path TEXT NOT NULL,
bytes INTEGER NOT NULL,
last_used INTEGER NOT NULL
);
CREATE INDEX cache_lru ON cache(last_used);
CREATE TABLE cache_rules (
id INTEGER PRIMARY KEY,
selector_json TEXT NOT NULL,
tier INTEGER NOT NULL,
priority INTEGER NOT NULL DEFAULT 0,
enabled INTEGER NOT NULL DEFAULT 1
);
CREATE TABLE image_cache (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
tier_actual INTEGER NOT NULL DEFAULT 0,
-- Materialised rather than recomputed, so the grid can draw availability
-- badges without evaluating every rule for every visible cell.
tier_desired INTEGER NOT NULL DEFAULT 0,
bytes INTEGER NOT NULL DEFAULT 0,
last_used INTEGER,
pinned_by_rule INTEGER REFERENCES cache_rules(id) ON DELETE SET NULL
);
-- Jobs --------------------------------------------------------------------
CREATE TABLE jobs (
id INTEGER PRIMARY KEY,
kind INTEGER NOT NULL,
subject_id INTEGER,
priority INTEGER NOT NULL DEFAULT 0,
state INTEGER NOT NULL DEFAULT 0, -- 0=pending 1=running 2=failed
attempts INTEGER NOT NULL DEFAULT 0,
not_before INTEGER NOT NULL DEFAULT 0,
payload TEXT,
last_error TEXT,
-- Coalescing. Enqueueing the same work twice updates one row rather than
-- queueing it twice, which is what makes "enqueue on any change" safe to
-- call liberally.
UNIQUE(kind, subject_id)
);
CREATE INDEX jobs_ready ON jobs(state, priority DESC, not_before);
"#;
#[cfg(test)]
mod tests {
use super::*;
fn mem() -> Connection {
let c = Connection::open_in_memory().unwrap();
configure(&c).unwrap();
c
}
/// How many rows a named backfill touched, ignoring the others.
///
/// Asserting on the whole vector would couple every test to which other
/// backfills happen to exist.
fn backfilled(c: &Connection, what: &str) -> usize {
backfill(c)
.unwrap()
.into_iter()
.find(|(name, _)| *name == what)
.map(|(_, n)| n)
.unwrap_or(0)
}
/// Insert an image and return its id.
fn image(c: &Connection, folder: Option<i64>, name: &str, format: &str) -> i64 {
c.execute(
"INSERT INTO images(root_id, folder_id, source_ref, format, added_at)
VALUES (1, ?1, ?2, ?3, 0)",
rusqlite::params![folder, name, format],
)
.unwrap();
c.last_insert_rowid()
}
fn with_root() -> Connection {
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO folders(id, root_id, path) VALUES (1, 1, 'a'), (2, 1, 'b')",
[],
)
.unwrap();
c
}
#[test]
fn a_jpeg_beside_its_raw_is_shadowed() {
// The camera's own rendering of a frame, not a second photograph.
let c = with_root();
let raw = image(&c, Some(1), "a/IMG_1234.CR2", "cr2");
let jpeg = image(&c, Some(1), "a/IMG_1234.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
let got: Option<i64> = c
.query_row(
"SELECT shadowed_by FROM images WHERE id = ?1",
[jpeg],
|r| r.get(0),
)
.unwrap();
assert_eq!(got, Some(raw));
}
#[test]
fn extension_case_does_not_matter() {
let c = with_root();
image(&c, Some(1), "a/IMG_1.cr2", "cr2");
image(&c, Some(1), "a/img_1.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
}
#[test]
fn a_standalone_jpeg_is_untouched() {
// Scanned film has no RAW sibling and must stay visible — 2,656 of
// them in the reference library.
let c = with_root();
image(&c, Some(1), "a/SCAN_0001.jpg", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 0);
}
#[test]
fn a_jpeg_in_a_different_folder_is_not_shadowed() {
// Camera filenames wrap at IMG_9999, so the same stem recurs across
// shoots (FR-CAT-11). Only a same-folder pair is safe to collapse.
let c = with_root();
image(&c, Some(1), "a/IMG_1234.CR2", "cr2");
image(&c, Some(2), "b/IMG_1234.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 0);
}
#[test]
fn a_raw_is_never_shadowed_by_a_jpeg() {
// The relationship is one-way: the RAW is the photograph.
let c = with_root();
let raw = image(&c, Some(1), "a/IMG_1.CR2", "cr2");
image(&c, Some(1), "a/IMG_1.JPG", "jpg");
backfill(&c).unwrap();
let got: Option<i64> = c
.query_row("SELECT shadowed_by FROM images WHERE id = ?1", [raw], |r| {
r.get(0)
})
.unwrap();
assert_eq!(got, None);
}
#[test]
fn backfill_is_idempotent() {
// It runs on every open, so a second pass must find nothing to do.
let c = with_root();
image(&c, Some(1), "a/IMG_1.CR2", "cr2");
image(&c, Some(1), "a/IMG_1.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
assert_eq!(backfilled(&c, "shadowed_by"), 0, "second pass is a no-op");
}
#[test]
fn a_v1_catalog_gains_the_column_and_is_backfilled() {
// The migration case that motivated this: rows already present when a
// column is added are silently partial until something backfills them.
let c = mem();
c.execute_batch(V1).unwrap();
c.pragma_update(None, "user_version", 1).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, format, added_at)
VALUES (1, 'IMG_9.CR2', 'cr2', 0), (1, 'IMG_9.JPG', 'jpg', 0)",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 1, "migrated from v1");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
}
#[test]
fn a_v4_catalog_gains_the_pinning_columns() {
// TRACES: FR-NC-6a
// An existing library must not have to be rescanned to gain offline
// pinning. The rows are already there; only the columns are new.
let c = mem();
c.execute_batch(V1).unwrap();
c.execute_batch(V2).unwrap();
c.execute_batch(V3).unwrap();
c.execute_batch(V4).unwrap();
c.pragma_update(None, "user_version", 4).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (7, 1, 'IMG_7.CR2', 0)",
[],
)
.unwrap();
// A cache row written before pinning existed.
c.execute(
"INSERT INTO image_cache(image_id, tier_actual, bytes) VALUES (7, 2, 100)",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 4, "migrated from v4");
// The pre-existing row survives, and defaults to unpinned — the safe
// direction, since claiming a pin nobody made would exempt it from
// eviction for ever.
let (pinned, bytes): (i64, i64) = c
.query_row(
"SELECT pinned, bytes FROM image_cache WHERE image_id = 7",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(pinned, 0);
assert_eq!(bytes, 100, "the existing row is untouched");
}
#[test]
fn a_v5_catalog_keeps_its_keywords_and_gains_their_identities() {
// TRACES: FR-CAT-5
// The migration case that matters here: a library keyworded by an
// import or an older build already has assignment rows, and they must
// survive into the vocabulary rather than being left searchable but
// invisible.
let c = mem();
for step in [V1, V2, V3, V4, V5] {
c.execute_batch(step).unwrap();
}
c.pragma_update(None, "user_version", 5).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (7, 1, 'IMG_7.CR3', 0)",
[],
)
.unwrap();
c.execute(
"INSERT INTO versions(id, image_id, uuid, name, is_default)
VALUES (1, 7, 'v-7', 'Default', 1)",
[],
)
.unwrap();
c.execute(
"INSERT INTO keywords(version_id, keyword) VALUES (1, 'puffin')",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 5, "migrated from v5");
assert_eq!(backfilled(&c, "keyword_terms"), 1);
let name: String = c
.query_row("SELECT name FROM keyword_terms", [], |r| r.get(0))
.unwrap();
assert_eq!(name, "puffin");
// The assignment is untouched — it is the durable fact, and the term
// row is only its identity.
let n: i64 = c
.query_row("SELECT count(*) FROM keywords", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 1);
// It runs on every open, so a second pass must find nothing to do.
assert_eq!(backfilled(&c, "keyword_terms"), 0);
}
#[test]
fn two_devices_may_both_hold_a_term_of_the_same_name() {
// Deliberately not a unique index. Two devices each typing "Iceland"
// is the ordinary case, and a constraint would abort the merge
// transaction at exactly the moment they first sync.
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO keyword_terms(uuid, name, created, revision, modified)
VALUES ('a', 'Iceland', 0, 1, 1), ('b', 'Iceland', 0, 1, 1)",
[],
)
.unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM keyword_terms", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 2);
}
#[test]
fn stems_ignore_directories_containing_dots() {
assert_eq!(stem_of("2026.08/IMG_1.CR2"), "IMG_1");
assert_eq!(stem_of("IMG_1.CR2"), "IMG_1");
assert_eq!(stem_of("noextension"), "noextension");
// A dotfile is all stem, not an empty name with an extension.
assert_eq!(stem_of(".hidden"), ".hidden");
}
#[test]
fn migrate_creates_schema_at_current_version() {
let c = mem();
assert_eq!(migrate(&c).unwrap(), 0);
let v: i64 = c
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, SCHEMA_VERSION);
}
#[test]
fn migrate_is_idempotent() {
let c = mem();
migrate(&c).unwrap();
// Re-running must not error or duplicate anything — NFR-R5 requires
// idempotency on retry, since a migration can be interrupted.
assert_eq!(migrate(&c).unwrap(), SCHEMA_VERSION);
}
#[test]
fn refuses_a_catalog_from_a_newer_build() {
let c = mem();
migrate(&c).unwrap();
c.pragma_update(None, "user_version", SCHEMA_VERSION + 1)
.unwrap();
// Opening it read-write would corrupt data this build cannot
// represent. Refusing is the specified behaviour (NFR-R5).
assert!(matches!(
migrate(&c),
Err(CatalogError::SchemaTooNew { .. })
));
}
#[test]
fn foreign_keys_cascade_from_root_to_image() {
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)",
[],
)
.unwrap();
c.execute("DELETE FROM roots WHERE id = 1", []).unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 0, "images must not outlive their root");
}
#[test]
fn job_uniqueness_coalesces_rather_than_duplicating() {
let c = mem();
migrate(&c).unwrap();
for _ in 0..5 {
c.execute(
"INSERT INTO jobs(kind, subject_id, priority) VALUES (1, 42, 0)
ON CONFLICT(kind, subject_id)
DO UPDATE SET priority = max(priority, excluded.priority)",
[],
)
.unwrap();
}
let n: i64 = c
.query_row("SELECT count(*) FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 1, "five enqueues of the same work is one job");
}
}