Files
DarkRoom/core/dr-catalog/src/schema.rs
T
dtourolle 6b1aac477d Put the developer docs under docs/dev and index the folder for users first
docs/ had 26 developer documents flat beside the manual, and the two
audiences are very differently sized: most readers want the manual and
the gesture reference, a few want the register, the designs and the
measurements. The manual and gestures.md stay at the top; everything for
someone changing the code moves to docs/dev/, and the two documents that
name their own successors — the v0.1 milestone and the UI-refinement plan
— go to docs/dev/archive/ rather than being deleted, since both are still
cited. docs/README.md is the index, users first.

Every reference follows: code comments, Cargo manifests, the workflows,
the pre-commit hook, the bench and traceability tools (which locate the
repo root by docs/dev/requirements.md now), packaging, the Docker READMEs,
CLAUDE.md, CONTRIBUTING.md and the README. The matrix links one level
deeper and is regenerated. Links out of the moved documents into the tree
gain a level; a link checker over every Markdown file finds none broken.
2026-09-20 16:20:15 +02:00

2049 lines
86 KiB
Rust

//! TRACES: FR-CAT-2 | NFR-R5
//! Schema definition and forward-only migrations.
//!
//! The catalog is an *index*, not a source of truth (ARCH §6.12) — it is
//! deletable and rebuildable from sources plus sidecars. That is what makes
//! migration failure survivable, and why the recovery path is the normal
//! mechanism rather than a last resort.
//!
//! Migrations are forward-only, transactional, and idempotent on retry
//! (NFR-R5). The app refuses to open a catalog newer than it understands
//! rather than corrupting it.
use rusqlite::Connection;
use crate::error::CatalogError;
/// Schema version this build writes and understands.
pub const SCHEMA_VERSION: i64 = 20;
/// Apply migrations up to [`SCHEMA_VERSION`].
///
/// Returns the version migrated from, so callers can log or back up before a
/// real migration (NFR-R2 requires a backup before schema change).
pub fn migrate(conn: &Connection) -> Result<i64, CatalogError> {
let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
if from > SCHEMA_VERSION {
return Err(CatalogError::SchemaTooNew {
found: from,
supported: SCHEMA_VERSION,
});
}
if from == SCHEMA_VERSION {
return Ok(from);
}
// Each step runs in its own transaction so a failure leaves the catalog
// at a coherent version rather than half-migrated.
if from < 1 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V1)?;
tx.pragma_update(None, "user_version", 1)?;
tx.commit()?;
}
if from < 2 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V2)?;
tx.pragma_update(None, "user_version", 2)?;
tx.commit()?;
}
if from < 3 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V3)?;
tx.pragma_update(None, "user_version", 3)?;
tx.commit()?;
}
if from < 4 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V4)?;
tx.pragma_update(None, "user_version", 4)?;
tx.commit()?;
}
if from < 5 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V5)?;
tx.pragma_update(None, "user_version", 5)?;
tx.commit()?;
}
if from < 6 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V6)?;
tx.pragma_update(None, "user_version", 6)?;
tx.commit()?;
}
if from < 7 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V7)?;
tx.pragma_update(None, "user_version", 7)?;
tx.commit()?;
}
if from < 8 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V8)?;
tx.pragma_update(None, "user_version", 8)?;
tx.commit()?;
}
if from < 9 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V9)?;
tx.pragma_update(None, "user_version", 9)?;
tx.commit()?;
}
if from < 10 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V10)?;
tx.pragma_update(None, "user_version", 10)?;
tx.commit()?;
}
if from < 11 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V11)?;
tx.pragma_update(None, "user_version", 11)?;
tx.commit()?;
}
if from < 12 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V12)?;
tx.pragma_update(None, "user_version", 12)?;
tx.commit()?;
}
if from < 13 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V13)?;
tx.pragma_update(None, "user_version", 13)?;
tx.commit()?;
}
if from < 14 {
let tx = conn.unchecked_transaction()?;
// `ALTER TABLE ... ADD COLUMN` has no `IF NOT EXISTS`, and NFR-R5
// wants this re-enterable: a catalog whose `user_version` was rewound
// by a rollback already has the column, and would otherwise fail its
// next open on it.
let has_quality: bool = tx
.prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = 'quality'")?
.exists([])?;
if !has_quality {
tx.execute_batch("ALTER TABLE faces ADD COLUMN quality REAL;")?;
}
tx.execute_batch(V14)?;
tx.pragma_update(None, "user_version", 14)?;
tx.commit()?;
}
if from < 15 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V15)?;
tx.pragma_update(None, "user_version", 15)?;
tx.commit()?;
}
if from < 16 {
let tx = conn.unchecked_transaction()?;
// Guarded like V14's column, and for the same reason: `ALTER TABLE
// ... ADD COLUMN` has no `IF NOT EXISTS`, and this step must be
// re-enterable (NFR-R5).
for column in EYE_COLUMNS {
let present: bool = tx
.prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = ?1")?
.exists([column])?;
if !present {
tx.execute_batch(&format!("ALTER TABLE faces ADD COLUMN {column} REAL;"))?;
}
}
tx.pragma_update(None, "user_version", 16)?;
tx.commit()?;
}
if from < 17 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V17)?;
tx.pragma_update(None, "user_version", 17)?;
tx.commit()?;
}
if from < 18 {
let tx = conn.unchecked_transaction()?;
// Guarded like V14's and V16's columns: ALTER has no IF NOT EXISTS
// and the step must be re-enterable (NFR-R5).
let present: bool = tx
.prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = 'landmarks_dense'")?
.exists([])?;
if !present {
tx.execute_batch("ALTER TABLE faces ADD COLUMN landmarks_dense BLOB;")?;
}
tx.pragma_update(None, "user_version", 18)?;
tx.commit()?;
}
if from < 19 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V19)?;
tx.pragma_update(None, "user_version", 19)?;
tx.commit()?;
}
if from < 20 {
let tx = conn.unchecked_transaction()?;
v20_markers_name_the_detector_that_found_the_faces(&tx)?;
tx.pragma_update(None, "user_version", 20)?;
tx.commit()?;
}
Ok(from)
}
// V20 -- TRACES: FR-CAT-7
//
// Run markers that named the wrong detector, put right.
//
// `faces::record_updates` -- the write behind the quality, eye and crop
// passes -- re-marked an image under the pipeline the pass ran as, while
// the faces it had updated kept the id of the detector that found them.
// A marker of `scrfd_10g+w600k_mbf` over faces spelled `w600k_mbf` reads,
// to every consumer, as the thorough detector having examined the image:
// the upgrade repair skips it, and `face_shard::export_to_shards` selects
// its faces by the marker's id, finds none, and tells every other device
// that the thorough detector found nothing there. The desktop's shard index
// held 54 such entries over photographs with named faces, and the tablet's
// eye pass over faces it had adopted from the desktop had made 430 more.
//
// The write is fixed to keep the marker under the faces' own id. This puts
// the markers already written right, with a fresh time so the export sends
// each image again under an entry newer than the empty one -- which is what
// `held_model` orders by. Where the right marker is still there beside the
// wrong one (the old write inserted rather than replaced), the wrong one
// goes and the right one is refreshed for the same reason: its entry in
// the shards is older than the empty one, and a device that has neither
// would take the empty one. An image V14 left with faces and no marker at
// all is not touched: that state is the quality pass's cue, and the fixed
// write marks it correctly when the pass reaches it.
//
// Restated in Rust rather than SQL because the embedder half of a pipeline
// id is `faces::embedder_sql`, which this must agree with.
fn v20_markers_name_the_detector_that_found_the_faces(tx: &Connection) -> Result<(), CatalogError> {
let fi = crate::faces::embedder_sql("face_index.model_id");
let f = crate::faces::embedder_sql("f.model_id");
// A marker is wrong when the image holds faces of its embedder under
// another id. First the wrong ones that sit beside a right one -- the
// update below would collide with it -- then the rest are renamed.
let wrong = format!(
"EXISTS (SELECT 1 FROM faces f
WHERE f.image_id = face_index.image_id
AND {f} = {fi}
AND f.model_id != face_index.model_id)"
);
let found_by = format!(
"(SELECT MIN(f.model_id) FROM faces f
WHERE f.image_id = face_index.image_id AND {f} = {fi})"
);
let now = crate::faces::now_secs();
tx.execute(
&format!(
"UPDATE face_index
SET indexed_at = ?1
WHERE model_id = {found_by}
AND EXISTS (SELECT 1 FROM face_index w
WHERE w.image_id = face_index.image_id
AND w.model_id != face_index.model_id
AND {} = {fi})",
crate::faces::embedder_sql("w.model_id")
),
[now],
)?;
tx.execute(
&format!(
"DELETE FROM face_index
WHERE {wrong}
AND EXISTS (SELECT 1 FROM face_index o
WHERE o.image_id = face_index.image_id
AND o.model_id = {found_by})"
),
[],
)?;
tx.execute(
&format!(
"UPDATE face_index
SET model_id = {found_by},
faces_found = (SELECT COUNT(*) FROM faces f
WHERE f.image_id = face_index.image_id AND {f} = {fi}),
indexed_at = ?1
WHERE {wrong}"
),
[now],
)?;
Ok(())
}
/// The seven columns V16 adds to `faces`, in the order the readers name them.
///
/// Named once because three places have to agree on them: this migration,
/// [`for_attached`], and the face shard's own catch-up (`face_shard`).
pub const EYE_COLUMNS: [&str; 7] = [
"eye_right",
"eye_right_px",
"eye_right_sharp",
"eye_left",
"eye_left_px",
"eye_left_sharp",
"sunglasses",
];
/// Recompute columns a migration added, for rows that predate it.
///
/// A migration adds a column with a default; it cannot know what the value
/// *should* be for the rows already present. Without a backfill those rows are
/// silently partial — present, queryable, and wrong — which is worse than
/// missing, because nothing signals that they need attention.
///
/// Cheap enough to run on every open: each pass is one indexed UPDATE, and
/// re-running it is a no-op once the values are already right.
///
/// Returns how many rows each backfill touched, for logging.
pub fn backfill(conn: &Connection) -> Result<Vec<(&'static str, usize)>, CatalogError> {
let mut out = Vec::new();
// v2: `shadowed_by`. A JPEG sitting beside a RAW of the same name is the
// camera's own rendering of that frame, not a second photograph, so it is
// hidden from the grid, the timeline and the sweep.
let n = pair_raw_and_jpeg(conn)?;
if n > 0 {
out.push(("shadowed_by", n));
}
// v3: every image needs a default version to carry its rating and flag.
// Libraries scanned before ratings existed have images and no versions at
// all, so there was nowhere for a judgement to go — see [`crate::rating`].
let n = crate::rating::ensure_default_versions(conn)?;
if n > 0 {
out.push(("default_versions", n));
}
// TRACES: FR-NC-8 | FR-NC-9
// The uuid on those rows is the cross-device merge identity, and it used
// to be generated rather than derived. This comment said so, and said it
// as though generating it were the point — it was the bug. Two devices
// minted different uuids for one photograph, so the sidecar they shared
// grew a `default = 1` block each and neither ever saw the other's work.
//
// Runs after the pass above so a row created a moment ago is already
// derived and matches nothing here. Ordering the other way would be
// correct too, just wasteful.
let n = crate::rating::align_default_version_uuids(conn)?;
if n > 0 {
out.push(("derived_version_uuids", n));
}
// v6: a vocabulary row for every word some image already carries.
//
// Three ways a catalog arrives holding assignments with no term behind
// them, and all three are normal rather than exceptional: a library
// keyworded by a build that predates this table, a catalog rebuilt from
// sidecars (which carry the word and not the identity), and an import from
// Lightroom or darktable (FR-CAT-14). Without this the words are
// searchable but absent from the vocabulary list, which reads as the
// keywords having been lost.
let n = crate::keywords::adopt_orphan_terms(conn)?;
if n > 0 {
out.push(("keyword_terms", n));
}
Ok(out)
}
/// How long a connection waits for a writer to finish before giving up.
///
/// TRACES: NFR-R1
/// SQLite's default is **zero** — the loser of a race gets `SQLITE_BUSY` at
/// once rather than a turn — and WAL does not change that for two writers. One
/// writer and many readers is the case WAL makes free; this is the other one,
/// and this application has it constantly: the face sweep commits a batch while
/// reclustering reads, the derived sync imports shards while the sweep writes.
///
/// Without a timeout that contention was *lost work*, not a retry. A face
/// sweep that had already paid for the detection and the embedding — the
/// expensive part, seconds per image — threw the result away on
/// `storing faces for 214: database is locked` and moved on, and both the
/// desktop and the tablet logged runs of those on consecutive images.
///
/// Ten seconds, matching the figure the job runner's tests already use for the
/// same reason. It is far longer than any transaction here (a sweep batch is
/// sub-second; the slowest is a WAL checkpoint of a 130 MB catalog), so in
/// practice it is a bound on pathology rather than a wait anyone sits through.
/// The tension with NFR-P9 is real but one-sided: a query on the UI thread
/// would rather wait for its turn than fail, because the failure is what the
/// user sees as "cannot open catalog".
const BUSY_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10);
/// Connection setup applied on every open, migration or not.
///
/// WAL is required by NFR-R1: it survives power loss without corruption, and
/// it lets a background job write while the grid reads.
pub fn configure(conn: &Connection) -> Result<(), CatalogError> {
// Before the pragmas, so that a connection racing a migration waits for it
// rather than failing on the first statement it tries.
conn.busy_timeout(BUSY_TIMEOUT)?;
conn.pragma_update(None, "journal_mode", "WAL")?;
// NORMAL rather than FULL: with WAL this is durable across process death
// (which is what FR-PLAT-AND-3 cares about) and only risks the last
// transaction on power loss. The catalog is rebuildable; the sidecars are
// not, and they are written separately with their own fsync discipline.
conn.pragma_update(None, "synchronous", "NORMAL")?;
conn.pragma_update(None, "foreign_keys", true)?;
// A scan touching thousands of rows is transient; let SQLite spill to
// memory rather than materialising temp b-trees on disk.
conn.pragma_update(None, "temp_store", "MEMORY")?;
Ok(())
}
/// The v1 schema rewritten to target an attached database.
///
/// Needed because a downloaded remote catalog is `ATTACH`ed under its own
/// schema name before merging, and tests build one from scratch. SQLite has no
/// "create these tables over there" form, so the names are rewritten.
///
/// The rewrite is textual and therefore only as good as the naming discipline
/// in [`V1`]: every `CREATE TABLE`/`CREATE INDEX` must name its object
/// unqualified, which they do.
pub fn v1_for_attached(schema_name: &str) -> String {
rewrite_for_attached(V1, schema_name)
// REFERENCES within an attached schema resolve to that schema already,
// so foreign keys need no rewriting — but the ON clause of an index
// does, and `CREATE INDEX x.name ON table` is the correct form.
}
/// Every table this build knows about, rewritten to target an attached
/// database.
///
/// [`v1_for_attached`] is kept alongside this rather than replaced by it: a
/// remote catalog written by an older build genuinely has only the v1 tables,
/// and the merge has to keep working against one (see
/// [`crate::merge::merge_keywords`]). Building that case in a test needs a way
/// to say "v1 and no more".
///
/// Only the migrations that *create* objects appear here. V2 through V5 are
/// `ALTER TABLE ... ADD COLUMN`, and the columns they add are local index
/// state — shadowing, trashing, cache pinning — that a merge never reads
/// across the attachment.
///
/// V7 creates an object and is still excluded, which is the one exception to
/// that rule and not an oversight: it indexes `shadowed_by` and `trashed_at`,
/// the very columns V2 through V5 add and this function leaves out, so
/// creating it over there would fail on columns that are not there. Nothing is
/// lost by its absence — it exists to make the *grid* page quickly, and the
/// grid never reads across an attachment.
///
/// V11 is excluded on the same grounds and for the plainer reason that a merge
/// has nothing to do with it: burst grouping is rebuilt locally from local
/// signatures, and no code reads a remote catalog's `burst_*` tables. Its
/// `ALTER TABLE` would fail here anyway, being unqualifiable by the rewrite.
pub fn for_attached(schema_name: &str) -> String {
// V10 is `ALTER TABLE`, which the textual rewrite cannot qualify, so its
// columns are spelled out. A remote genuinely older than V10 is a real
// case and `merge::merge_people_within` probes for them; this is the
// *current* shape, which is what the tests want.
format!(
"{}\n{}\n{}\n\
ALTER TABLE {schema_name}.people ADD COLUMN ignored INTEGER NOT NULL DEFAULT 0;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN crop BLOB;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN quality REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_right REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_right_px REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_right_sharp REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_left REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_left_px REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_left_sharp REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN sunglasses REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN landmarks_dense BLOB;",
rewrite_for_attached(V1, schema_name),
rewrite_for_attached(V6, schema_name),
rewrite_for_attached(V8, schema_name),
)
}
/// Qualify every object a `CREATE` statement names with `schema_name`.
///
/// The rewrite is textual and therefore only as good as the naming discipline
/// in the batches it is given: every `CREATE TABLE`/`CREATE INDEX` must name
/// its object unqualified, which they do.
fn rewrite_for_attached(sql: &str, schema_name: &str) -> String {
sql.replace("CREATE TABLE ", &format!("CREATE TABLE {schema_name}."))
.replace("CREATE INDEX ", &format!("CREATE INDEX {schema_name}."))
.replace(
"CREATE UNIQUE INDEX ",
&format!("CREATE UNIQUE INDEX {schema_name}."),
)
}
/// Mark each JPEG that sits beside a RAW of the same name.
///
/// Matched on folder plus stem, case-insensitively. Same-folder is what makes
/// this safe: cameras write the pair side by side, and matching across folders
/// would risk pairing unrelated frames, since camera filenames wrap at
/// IMG_9999 (FR-CAT-11).
///
/// Done in Rust rather than SQL because the comparison needs a filename stem,
/// and SQLite has no such function without enabling `rusqlite/functions` —
/// a dependency feature for one string operation, whose SQL spelling would be
/// unreadable and would mishandle names with no extension.
fn pair_raw_and_jpeg(conn: &Connection) -> Result<usize, CatalogError> {
use std::collections::HashMap;
// (folder, lowercase stem) -> RAW id, built in one pass over the RAWs.
let mut raws: HashMap<(Option<i64>, String), i64> = HashMap::new();
{
let mut stmt = conn.prepare(
"SELECT id, folder_id, source_ref FROM images
WHERE lower(format) IN
('cr2','cr3','nef','arw','raf','rw2','orf','dng')",
)?;
let rows = stmt.query_map([], |r| {
Ok((
r.get::<_, i64>(0)?,
r.get::<_, Option<i64>>(1)?,
r.get::<_, String>(2)?,
))
})?;
for row in rows {
let (id, folder, path) = row?;
raws.insert((folder, stem_of(&path).to_ascii_lowercase()), id);
}
}
if raws.is_empty() {
return Ok(0);
}
let pairs: Vec<(i64, i64)> = {
let mut stmt = conn.prepare(
"SELECT id, folder_id, source_ref FROM images
WHERE lower(format) IN ('jpg','jpeg') AND shadowed_by IS NULL",
)?;
let rows = stmt.query_map([], |r| {
Ok((
r.get::<_, i64>(0)?,
r.get::<_, Option<i64>>(1)?,
r.get::<_, String>(2)?,
))
})?;
rows.filter_map(|row| {
let (id, folder, path) = row.ok()?;
let raw = raws.get(&(folder, stem_of(&path).to_ascii_lowercase()))?;
Some((id, *raw))
})
.collect()
};
let tx = conn.unchecked_transaction()?;
for (jpeg, raw) in &pairs {
tx.execute(
"UPDATE images SET shadowed_by = ?2 WHERE id = ?1",
[jpeg, raw],
)?;
}
tx.commit()?;
Ok(pairs.len())
}
/// A filename without its extension.
///
/// Only the final path component, and only its last dot — a directory
/// containing a dot must not truncate the name.
fn stem_of(path: &str) -> &str {
let name = path.rsplit(['/', ':']).next().unwrap_or(path);
match name.rsplit_once('.') {
Some((stem, _)) if !stem.is_empty() => stem,
_ => name,
}
}
/// TRACES: FR-CAT-4 | NFR-P5
/// The order the grid reads in, as an index.
///
/// # What this is for
///
/// Every window the grid loads is `ORDER BY ... LIMIT n OFFSET k`, and without
/// an index that matches the ordering SQLite answers it by sorting the whole
/// library into a temp b-tree and then discarding the first `k` rows. Measured
/// on 24,000 images at offset 20,000, one window read cost 15 ms — a frame
/// budget of 16.7 ms, spent inside the scroll handler, several times per
/// screenful. That is the jitter.
///
/// With this index the same read is a walk along it: 0.36 ms.
///
/// # Why the shape is what it is
///
/// `captured_at IS NULL` leads, because [`crate::library`]'s `GRID_ORDER` does
/// — undated images sort last, and an ordinary index on `captured_at` cannot
/// answer that, since the expression is not a column. SQLite indexes
/// expressions, so it is spelled out here exactly as the query spells it; the
/// two must stay identical or the planner silently falls back to sorting and
/// the cost comes back with no other symptom.
///
/// `source_ref` is included because it breaks ties in the same ordering, and an
/// index that stopped at `captured_at` would leave a sort for the ties.
///
/// # Partial, on the same predicate the grid filters by
///
/// The grid never lists shadowed or trashed rows, so an index carrying them
/// would be larger than the question ever asks about, and — more to the point —
/// a partial index is only usable when its `WHERE` is implied by the query's,
/// which is what makes this one apply to the grid's reads and to nothing else.
///
/// # It does not cover the rating filter or a collection scope
///
/// Both narrow the walk rather than reorder it, so the index still supplies the
/// ordering and SQLite tests the extra predicate per row. That is the cheap
/// direction: the expensive part was never the filtering, it was the sort.
const V10: &str = r#"
-- TRACES: FR-CULL-10 | FR-CULL-12
-- Two columns the People screen turned out to need, and neither is derivable.
-- A person the user does not want to identify.
--
-- Most of a real library's clusters are strangers: people in the background of
-- a street, guests at somebody else's party, a face on a poster. They are
-- correctly detected and correctly grouped, and the user will never name any of
-- them -- but they crowd out the handful of groups that matter, and there is no
-- way to tell "not yet looked at" from "looked at, don't care" without
-- recording the second.
--
-- **User data**, and the reason this is a column rather than a deletion: a
-- deleted cluster comes straight back on the next Regroup, because the faces
-- are still there and still similar. Nothing short of remembering the judgement
-- survives re-clustering, which is the same argument `face_person_rejected`
-- makes one level down (FR-CULL-12).
ALTER TABLE people ADD COLUMN ignored INTEGER NOT NULL DEFAULT 0;
-- The face, cut out and kept.
--
-- A face used to be drawn by decoding the 1024px proxy it was found on and
-- cutting the box out again, every time the screen opened. That made the People
-- screen a *derivative of the thumbnail cache*: evict a proxy -- which the
-- cache is entitled to do at any moment -- and the cell goes blank, with no way
-- back short of re-fetching the original over the network and re-detecting it.
-- It also cost one full JPEG decode per image per visit to show a 96px cell.
--
-- So the crop is cut once, when the pixels are already in hand at detection
-- time, and kept. Small: a 160px JPEG is a few KB, against ~250 KB for the
-- proxy it replaces reading.
--
-- Nullable, because a face indexed before this column existed has no crop and
-- must still work -- the reader falls back to the old proxy path, and the next
-- indexing pass fills it in.
--
-- **Stripped from the sync snapshot.** The catalog is uploaded whole, so this
-- would otherwise put tens of MB of JPEG on every sync; crops travel in the
-- face shards instead, which is where the bulk per-face data already goes
-- (`face_shard`). See `sync::snapshot_for_upload`.
ALTER TABLE faces ADD COLUMN crop BLOB;
"#;
/// TRACES: FR-CULL-5
/// Burst grouping: which frames are one moment, and which one stands for it.
///
/// The reasoning behind the grouping itself is in [`crate::bursts`]; what
/// belongs here is why it is stored in three pieces rather than one.
///
/// **`images.perceptual_hash` is a column, not a table**, for the same reason
/// `content_hash` is: it is one number per image, NULL until something has had
/// the pixels in hand, and every query that wants it is already reading the
/// image row. It is local derived state — a rebuilt catalog recomputes it from
/// thumbnails — which is also why it is absent from [`for_attached`], alongside
/// the shadowing and trashing columns V2 through V5 add.
///
/// **`burst_members` is rewritten whole by every pass.** No id of its own: the
/// group is named by the image id of its earliest frame, so a burst that has not
/// changed keeps its name across a regroup and the interface can remember that
/// this one is open. There is no `bursts` table to go with it because a group
/// has no properties beyond its members — inventing a row for it would create an
/// identity that survives the grouping being rebuilt, which is precisely what
/// must not happen.
///
/// **`burst_pick` is the one thing here that is not derived**, and it is a
/// separate table so that rewriting the grouping cannot erase it. A
/// representative stored on `burst_members` would be forgotten every time a
/// frame arrived; the user would be asked the same question after every scan.
/// The same argument `people.ignored` makes in V10, one subsystem over.
///
/// **`burst_expanded` is view state in the catalog**, which is unusual enough to
/// justify. The grid is a window over an ordered query — `LIMIT n OFFSET k` —
/// so what a collapsed burst hides has to be decided by the query, or the row
/// count stops agreeing with the scrollbar and the ordinals a scrub resolves to.
/// Once SQL has to see it, this is where it lives. Nothing else reads it, and it
/// is emptied of stale groups by every pass.
const V11: &str = r#"
-- A 64-bit perceptual signature. Local derived state: NULL until something has
-- decoded the image, recomputed from thumbnails if the catalog is rebuilt, and
-- comparable only to signatures produced by the same build (`bursts`).
ALTER TABLE images ADD COLUMN perceptual_hash INTEGER;
CREATE TABLE burst_members (
-- One burst at most per image: a frame belongs to the moment it was taken
-- in, and nothing else.
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
-- The image id of the burst's earliest frame. Not a foreign key by
-- accident: the leader is itself a member, so this genuinely references
-- images(id), and cascading its deletion is right.
burst_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
-- The frame the group collapses to. Exactly one per burst.
representative INTEGER NOT NULL DEFAULT 0
);
-- Counting a burst's frames and listing them are what the grid asks for, once
-- per window; without this both are a scan of every grouped frame in the
-- library.
CREATE INDEX burst_members_burst ON burst_members(burst_id);
-- The user's own choice of representative. User data, never rewritten by a
-- grouping pass -- see the module doc above.
CREATE TABLE burst_pick (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE
);
-- Bursts the grid is currently showing in full.
CREATE TABLE burst_expanded (
burst_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE
);
"#;
const V12: &str = r#"
-- TRACES: FR-CULL-8
-- Forget the runs that were made against a proxy too small to find a face on.
--
-- Detection used to accept any proxy, and one of the two sweeps detected on
-- the stored 1024px tier. On the reference library that produced 0.078 faces
-- per image against 1.82 for the same photographs at 2048 or better -- and
-- every one of those runs left a `face_index` row behind saying the image had
-- been examined. That row is what makes the damage permanent: the work list is
-- "images with no row", so a photograph examined badly is indistinguishable
-- from one examined well, and is never offered to a later pass.
--
-- Deleting the marker is the whole repair, and it is deliberately not a
-- deletion of anything else. The `faces` rows those runs found stay exactly
-- where they are and keep drawing the People screen until a better pass
-- replaces them, and `record_detections` carries the user's confirmed names
-- across that replacement by box overlap. So this costs a re-fetch of the
-- affected images and loses no work the user has done.
--
-- The threshold is written out rather than taken from `dr_face::MIN_CROP_EDGE`
-- on purpose. A migration has to keep meaning what it meant on the day it ran;
-- binding it to a constant someone may raise later would silently change what
-- an old catalog gets migrated to.
DELETE FROM face_index WHERE source_edge <= 1024;
"#;
const V13: &str = r#"
-- TRACES: FR-CAT-8 | FR-NC-9
-- Which sidecars this device has read, and at what ETag.
--
-- The sidecar is the authoritative store for a rating and an edit, and until
-- this table existed nothing ever read one back into the catalog: judgements
-- travelled outward only. A cull done on a tablet reached the server and
-- stopped there, because the scan indexes photographs, the derived sync moves
-- thumbnails and collections, and the one reader that existed ran when a single
-- photograph was opened in develop and fed only the develop graph. The grid
-- draws `versions.rating`, so another device's afternoon of culling was
-- invisible on this one -- permanently, by every path the app had.
--
-- What this holds is the ETag, not the content. It is the record of what has
-- already been taken in, so a pull fetches only what changed: `dr_sync::scan`
-- reports every sidecar it saw in listings it was making anyway, and this
-- decides which of them are worth a GET.
--
-- Keyed on the sidecar's own remote path rather than on an image id. One
-- sidecar can describe two images -- a RAW and the JPEG beside it are one
-- photograph (FR-CAT-11) and share a document -- and a path is what the scan
-- reports and what a fetch addresses, so keying on anything else would mean
-- deriving one from the other in two places.
--
-- Rebuildable like the rest of the catalog: losing this table costs one pass
-- that re-reads every sidecar and reaches exactly the same state.
--
-- `IF NOT EXISTS` because NFR-R5 asks for migrations that are idempotent on
-- retry, and this one can genuinely be re-entered: a catalog whose
-- `user_version` was rewound -- by a rollback to an older build, or by a
-- recovery -- would otherwise fail its next open on a table it already has.
CREATE TABLE IF NOT EXISTS sidecars (
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
path TEXT NOT NULL,
etag TEXT,
-- Unix seconds, for diagnosing a pull that is not making progress.
read_at INTEGER NOT NULL DEFAULT 0,
PRIMARY KEY(root_id, path)
);
"#;
const V14: &str = r#"
-- TRACES: FR-CULL-9 | FR-CULL-10
-- How recognisable the model found each face, and a second look at the faces
-- it was never asked about.
--
-- The embedder's raw output has a length, and the length is a quality
-- reading: it grows with how much of a face the model could make out, and a
-- blur, an occlusion or a hard profile comes out short (dr_face::embedding,
-- `MIN_GALLERY_QUALITY`). Normalising threw it away. A short vector sits
-- near the middle of the sphere and matches a little of everyone, which is
-- how one bad crop bridges two people in a grouping pass -- so a face below
-- the floor is compared against the others and never compared *against*.
--
-- Nullable, and NULL means "never measured": every face indexed before this
-- version stored the unit vector, whose length is one whatever the crop was.
-- A face with no reading is admitted to the gallery, because a rule that
-- cannot be checked should admit rather than exclude -- but it is also a
-- face this rule is not yet protecting anyone from, and the only way to
-- measure it is to embed it again.
--
-- The `face-quality` repair is what does that (`dr_ui::repairs`, once the
-- sweep's measuring pass): it lists every face with no reading, and each is
-- embedded again from the native render with the landmarks it already has,
-- the raw vector written over the old one (`record_updates`) and nothing
-- else touched -- not the id, not the box, not who the user said it was.
-- The faces keep drawing the People screen throughout.
--
-- The run markers of those images are forgotten too, exactly as V12 forgot
-- the runs made against too small a proxy. The build this shipped in had no
-- measuring pass yet, and a marker is the one thing that stops a face ever
-- being looked at again; with the repair in place, detection leaves an
-- image holding this embedder's faces to it rather than detecting from
-- scratch, so the deletion costs nothing -- and an image that was examined
-- and found empty keeps its marker, since there is nothing on it to measure.
--
-- The cost is a re-fetch of every image with a face on it, on the next pass
-- the user starts. That is a whole-library transfer (FR-NC-6), and it starts
-- when they say so, not here.
--
-- From this version the `embedding` blob is the **raw** model output rather
-- than the unit vector V8 describes -- the length is the quality, and a store
-- that kept only the direction had thrown it away. Readers re-normalise on
-- load, so a unit blob from before and a raw blob from now compare alike;
-- `quality` is that length kept beside the blob for the readers that never
-- load the vector, and NULL rather than 1.0 for the old rows, because a unit
-- vector reads as a length of one and one is not "unmeasured".
--
-- The column itself is added in `migrate`, guarded, because ALTER has no
-- IF NOT EXISTS and this step has to be re-enterable (NFR-R5).
DELETE FROM face_index
WHERE EXISTS (SELECT 1 FROM faces f
WHERE f.image_id = face_index.image_id
AND f.model_id = face_index.model_id);
"#;
const V15: &str = r#"
-- TRACES: FR-CAT-13
-- Where a standard XMP sidecar and the catalog disagree.
--
-- An `.xmp` beside a photograph is read on the same pull as DarkRoom's own
-- sidecar, and reconciled field by field (`dr_xmp::reconcile`): keywords
-- union, and a rating, label or caption is taken only where the catalog holds
-- none. That rule is the safe one and it is not always the right one -- a
-- rating changed in Lightroom after it was changed here is a genuine
-- disagreement, and a standard XMP carries no revision to settle it by. So
-- the disagreement is written here instead of being resolved, and the
-- requirement's "a metadata reload offered" is a row in this table with a
-- button in front of it: the reload re-reads the file with the sidecar
-- winning, and deletes the row.
--
-- Keyed on the sidecar's path like `sidecars` is, and for the same reason: a
-- path is what the scan reports, what a fetch addresses, and what the ETag
-- that noticed the change belongs to. `fields` is the disagreeing fields as
-- `dr_xmp` names them, space-separated, for the line the settings page shows.
--
-- Rebuildable: the next pull that sees a changed ETag writes the row again.
CREATE TABLE IF NOT EXISTS xmp_conflicts (
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
path TEXT NOT NULL,
fields TEXT NOT NULL,
seen_at INTEGER NOT NULL DEFAULT 0,
PRIMARY KEY(root_id, path)
);
"#;
// V19 -- TRACES: NFR-P9
//
// The indexes the repair counts are served from, and V17's lesson applied
// to the rest of the face columns.
//
// "How many images still owe a quality reading" was answered per image: a
// correlated EXISTS over `faces` that had to open each face's row to look
// at one nullable column -- the row being eight kilobytes of embedding and
// crop. Six such counts run every time the Identity screen opens and every
// time a sweep ends, 160 ms of them on the reference library. Three
// partial indexes hold only the faces still owing each pass, keyed by the
// image and carrying the model id the predicate also reads, so the count
// walks a few thousand index entries and touches no row at all -- and each
// index shrinks to nothing as its pass completes. The planner takes them
// when the count is driven from `faces` (`repairs::count`) and ignores
// them inside the per-image EXISTS, which is why that function has two
// spellings of the same predicate.
//
// `faces_image_model` replaces `faces_image`: the same key with the model
// id beside it, so "does this image hold this embedder's faces" -- asked in
// the audit, the proxy repair and the outstanding-detection count -- is an
// index-only probe where it used to read the row for the model id. Every
// lookup that used `faces_image` is served by its prefix.
//
// Not applied to attached catalogs, like V7 and V17: an index is a local
// concern, and a merge never runs these queries across an attachment.
const V19: &str = r#"
CREATE INDEX IF NOT EXISTS faces_image_model ON faces(image_id, model_id);
DROP INDEX IF EXISTS faces_image;
CREATE INDEX IF NOT EXISTS faces_owed_quality ON faces(image_id, model_id)
WHERE quality IS NULL;
CREATE INDEX IF NOT EXISTS faces_owed_crop ON faces(image_id, model_id)
WHERE crop IS NULL;
CREATE INDEX IF NOT EXISTS faces_owed_eyes ON faces(image_id, model_id)
WHERE eye_right IS NULL OR landmarks_dense IS NULL;
"#;
// V18 -- TRACES: FR-CULL-8a | FR-CULL-12
//
// The 106 dense landmarks the eye pass reads its eye boxes from, kept beside
// the reading as `dr_face::Landmarks::to_packed_bytes`: 106 x (x, y) as
// 16-bit fixed point over the frame, 424 bytes a face, a seventh of a
// pixel on a 6000-pixel frame. Derived data under FR-CULL-12 -- rebuilt by
// re-reading, never in a sidecar -- and stored for the same reason the
// embedding is: it cost a fetch of the original and a model run, and the
// next per-face pass (head pose, expression) should not have to pay either
// again. NULL where the face was never read.
//
// Added in `migrate`, guarded, like every ALTER here (NFR-R5).
const V17: &str = r#"
-- TRACES: FR-CULL-8a | FR-CULL-13 | NFR-P9
-- The eyes-open filter's index, and a lesson about where a column lands.
--
-- The people filter is a correlated EXISTS over `faces` per image, and it
-- was fast because `faces_image` *covers* it: the subquery never touched a
-- row. Reading V16's seven eye columns in the same subquery did touch the
-- row -- and `ALTER TABLE ADD COLUMN` puts a column at the end of the
-- record, after the 1 KB embedding and the ~5 KB crop, so every check
-- dragged six kilobytes off disk to reach seven floats. Measured on the
-- reference library: 24 seconds for one count, thirteen of them system
-- time. With this index the same count takes five milliseconds, because
-- the subquery is served from the index again and never reads a row.
--
-- The columns are listed in EYE_COLUMNS' order behind `image_id`, which is
-- the key the subquery searches on. Nothing else changed in V17; a catalog
-- already at V16 needs only this.
CREATE INDEX IF NOT EXISTS faces_eyes ON faces(
image_id, eye_right, eye_right_px, eye_right_sharp,
eye_left, eye_left_px, eye_left_sharp, sunglasses
);
"#;
// V16 -- TRACES: FR-CULL-8a
//
// What each face's eyes are doing: for each eye P(open), the source pixels
// across its box and the sharpness of the patch the classifier saw; and
// P(sunglasses) for the head. Seven numbers rather than a verdict, because
// the verdict is a rule with thresholds in it (dr_face::eyes::EyeReading::
// state) and a rule belongs in code that can be changed, not in rows that
// would have to be re-measured.
//
// The pixels and the sharpness are what stop a smear reading as a blink: an
// eye too small or too soft to read is not asked, and a face with no
// readable eye is "unclear", which no filter drops. Sunglasses are a column
// of their own for the same kind of reason — the eye classifier answers
// confidently over dark glass, and its answer means nothing there. A filter
// for "eyes open" reads all seven.
//
// NULL means "never measured" -- a face indexed before this version, or on a
// device without the eye models -- and a NULL is left alone by every filter
// that reads these, so an old library does not empty its grid the moment the
// chip is pressed. The sweep's measuring pass fills them in, from the native
// render, with the landmarks already stored: the same pass V14 built for the
// embedding's length, extended to ask the eye models too. No run marker is
// forgotten here, for the reason V14's note gives -- the measuring pass
// finds its own work by the NULL, and deleting markers would only put the
// detector back over images it has finished with.
//
// The columns are added in `migrate`, guarded, because ALTER has no IF NOT
// EXISTS and the step has to be re-enterable (NFR-R5). Their names are
// `EYE_COLUMNS`.
const V9: &str = r#"
-- TRACES: FR-CULL-8
-- A record that face detection has *run* on an image, distinct from what it
-- found.
--
-- # Why the faces table cannot answer this
--
-- Without this, "has this image been indexed" is asked as "does it have any
-- faces", and those are not the same question. **A photograph with no faces in
-- it is indistinguishable from one that has never been looked at**, so every
-- indexing pass re-examines every landscape, every still life and every
-- document scan in the library, for ever. In a typical personal library that is
-- most of it: the pass never converges, and the cost is paid again on every
-- run rather than once.
--
-- It also makes a coverage figure possible, which is the thing a user actually
-- wants to see — "4,812 of 5,000 images indexed" — where counting face rows
-- can only ever report how many faces exist.
--
-- # Why it is keyed on the model
--
-- Embeddings from different models are not comparable, so a model change has
-- to re-index. Keying the marker on `(image_id, model_id)` makes that
-- automatic: new model, no marker, image comes back into the queue. The id
-- names the whole pipeline -- detector and embedder together -- because
-- changing either changes what is found.
CREATE TABLE face_index (
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
model_id TEXT NOT NULL,
indexed_at INTEGER NOT NULL,
-- Zero is a real and common answer, and recording it is the entire point.
faces_found INTEGER NOT NULL,
-- Long edge of the proxy this ran against. A face too small to detect at
-- 1024 may be findable at 2048, so a library whose proxies grow can
-- re-index the images that stand to gain instead of all of them.
source_edge INTEGER NOT NULL,
PRIMARY KEY (image_id, model_id)
);
CREATE INDEX face_index_model ON face_index(model_id);
"#;
const V8: &str = r#"
-- TRACES: FR-CULL-8 | FR-CULL-9 | FR-CULL-10 | FR-CULL-11 | FR-CULL-12 | NFR-SEC-5
-- People and faces (docs/dev/faces.md, docs/dev/catalog.md §10).
--
-- Everything here is **derived data** except one column. Faces, landmarks,
-- embeddings, cluster assignments and suggestions are all reproducible by
-- re-indexing and are never written to a sidecar (FR-CULL-12); a person's
-- *name*, once the user has confirmed it, is a human judgement of the same
-- class as a rating and travels with the photograph.
--
-- That asymmetry is the whole design: deleting the catalog costs an afternoon
-- of re-indexing and loses nothing the user typed (ARCH §6.12).
CREATE TABLE people (
id INTEGER PRIMARY KEY,
-- Merge identity, not the name. Two devices that independently name the
-- same cluster produce two people; merging them keys on this, exactly as
-- collections do (FR-CAT-7, ARCH §6.3).
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
-- Tombstone-by-redirect. A merged person must outlive its merge or a
-- device that still holds it resurrects it on the next sync -- the same
-- hazard collections have, solved the same way.
merged_into INTEGER REFERENCES people(id) ON DELETE SET NULL,
created INTEGER NOT NULL,
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL
);
CREATE TABLE faces (
id INTEGER PRIMARY KEY,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
-- Normalised to the image's long edge, so a face survives the proxy it was
-- found on being evicted and regenerated at another resolution. Storing
-- pixels would bind a face to a resolution the cache is entitled to change.
x REAL NOT NULL, y REAL NOT NULL, w REAL NOT NULL, h REAL NOT NULL,
landmarks BLOB NOT NULL, -- 5 x (x, y) f32, normalised likewise
detector_confidence REAL NOT NULL,
embedding BLOB NOT NULL, -- 512 x f16; unit length until V14, raw since
-- Source pixels across the aligned 112x112 crop (docs/dev/faces.md §7).
--
-- Not cosmetic: it is the honest quality signal for the UI, a feature in
-- the §8 calibration -- FR-CULL-9 names face size as an axis along which an
-- uncalibrated similarity misbehaves -- and the selector a later
-- higher-resolution re-embedding pass would run on.
crop_px REAL NOT NULL,
-- Which model produced this embedding.
--
-- The one mistake in this subsystem that yields plausible-looking garbage
-- rather than an error: embeddings from different models are not
-- comparable. Storing the model with the vector makes a model change
-- detectable and re-indexable instead of quietly poisoning every
-- similarity in the library.
model_id TEXT NOT NULL,
detected_at INTEGER NOT NULL
);
CREATE INDEX faces_image ON faces(image_id);
CREATE INDEX faces_model ON faces(model_id);
CREATE TABLE face_person (
face_id INTEGER PRIMARY KEY REFERENCES faces(id) ON DELETE CASCADE,
person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE,
-- Calibrated P(this face is this person), never a raw cosine (FR-CULL-9).
probability REAL NOT NULL,
-- The user said so. Never overwritten by a later inference pass.
--
-- A column rather than a probability of 1.0, because a confirmation is a
-- different kind of fact from a confident guess and collapsing them loses
-- the ability to recompute suggestions without touching user data.
confirmed INTEGER NOT NULL DEFAULT 0
);
CREATE INDEX face_person_person ON face_person(person_id, confirmed);
-- Faces the user has explicitly said are NOT a given person.
--
-- Needed because rejection is not the absence of an assignment: without it,
-- the next clustering pass re-suggests exactly the face the user just pushed
-- away, and the tool feels broken. Same reasoning as `confirmed` -- a
-- judgement is user data (FR-CULL-12) whichever direction it points.
CREATE TABLE face_person_rejected (
face_id INTEGER NOT NULL REFERENCES faces(id) ON DELETE CASCADE,
person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE,
PRIMARY KEY (face_id, person_id)
);
-- The FR-CULL-9 calibration, fitted from this library's own faces.
--
-- One row per model, because the fit is a property of the embedding space and
-- a library indexed across a model change holds two. `face_set_hash` is what
-- makes a stale fit detectable: a materially changed library recomputes rather
-- than trusting numbers derived from a set that no longer exists.
CREATE TABLE face_calibration (
model_id TEXT PRIMARY KEY,
-- P(same) = sigmoid(a*cos + b + w_size*log2(min(crop_px)) + log_prior_odds)
a REAL NOT NULL,
b REAL NOT NULL,
w_size REAL NOT NULL DEFAULT 0.0,
-- Whether the fit is usable at all. When it is not, the UI says the
-- confidence is unavailable; it does not present an untuned default as
-- though it were measured (FR-CULL-9).
valid INTEGER NOT NULL DEFAULT 0,
positive_pairs INTEGER NOT NULL DEFAULT 0,
negative_pairs INTEGER NOT NULL DEFAULT 0,
face_set_hash TEXT NOT NULL,
fitted_at INTEGER NOT NULL
);
"#;
const V7: &str = r#"
CREATE INDEX images_grid_order
ON images(captured_at IS NULL, captured_at, source_ref)
WHERE shadowed_by IS NULL AND trashed_at IS NULL;
"#;
const V6: &str = r#"
-- TRACES: FR-CAT-5 | FR-CAT-6 | FR-NC-9
-- Keywords gain an identity, so that renaming and deleting one can cross
-- between devices.
--
-- The v1 `keywords` table is the *assignment*: one row per (version, word),
-- and the word is stored as text. That stays exactly as it is, and this
-- migration adds nothing to it, for a reason that is easy to get backwards.
--
-- # Why assignments keep the text rather than pointing at a row here
--
-- The catalog is a rebuildable index (ARCH §6.12). What an image is keyworded
-- with is authoritative in the sidecar and in XMP `dc:subject` (FR-CAT-13),
-- and both of those carry a *string*. Rewriting the join to reference
-- `keyword_terms(id)` would mean a catalog rebuilt from sidecars had to invent
-- term rows before it could record a single assignment, and an integer that
-- means nothing on the other device would sit where the durable fact belongs.
-- It would also break `crate::query`, which matches `kw.keyword` directly and
-- must keep hitting `keywords_term` on a 50k library (FR-CAT-6).
--
-- So the text is the fact and this table is the *identity*: it exists to give
-- a rename and a deletion something a merge can key on, and to let a keyword
-- exist in the vocabulary before any photograph carries it.
CREATE TABLE keyword_terms (
id INTEGER PRIMARY KEY,
-- Device-independent identity, as `collections.uuid` is. The integer id is
-- local and collides across devices.
uuid TEXT NOT NULL UNIQUE,
-- The word itself, and the value written into every assignment row.
name TEXT NOT NULL,
created INTEGER NOT NULL,
-- Monotonic, bumped on every local edit. `crate::merge` compares these
-- rather than timestamps, so a clock-skewed device cannot silently win.
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL,
-- Tombstone, so a merge against a device that still holds the keyword does
-- not resurrect it.
deleted INTEGER NOT NULL DEFAULT 0
);
-- Deliberately **not** UNIQUE.
--
-- Two devices that each type "Iceland" create two rows with two uuids, and
-- both are correct until they meet. A unique constraint would abort the merge
-- transaction at exactly that moment — the ordinary case, not a corner one.
-- Uniqueness is instead reached by convergence: `crate::keywords::create`
-- resolves an existing name locally, and `crate::keywords::fuse_duplicates`
-- collapses a cross-device pair onto the lexicographically smaller uuid, which
-- both devices compute identically without talking to each other.
--
-- Partial on `deleted = 0` because every lookup here is a live one: the
-- vocabulary list, the resolve-by-name in `create`, and the fuse pass all
-- exclude tombstones, and including them would grow the index with every
-- keyword the library has ever had rather than with the ones it has.
CREATE INDEX keyword_terms_name ON keyword_terms(name) WHERE deleted = 0;
"#;
const V5: &str = r#"
-- TRACES: FR-NC-6a | FR-CAT-9 | NFR-RES-4
-- Offline availability: what is kept, why it is kept, and where it lives.
--
-- `pinned` separates a promise from a convenience, and the distinction has to
-- be a *column* rather than something inferred from `pinned_by_rule`. A pin is
-- the user saying "this collection comes with me"; a passively cached original
-- is the app noticing they opened something. Only the second is evictable, so
-- the eviction query has to be able to ask the question directly — and it has
-- to keep answering correctly for an image whose pinning rule was since
-- deleted, which `pinned_by_rule` alone cannot do because it is
-- ON DELETE SET NULL.
ALTER TABLE image_cache ADD COLUMN pinned INTEGER NOT NULL DEFAULT 0;
-- Where the cached original actually is, relative to the cache directory.
-- Relative rather than absolute: the library moves between machines and
-- between an app sandbox and a user directory, and an absolute path baked in
-- at download time would break on every one of those.
ALTER TABLE image_cache ADD COLUMN path TEXT;
-- Eviction reads exactly this: unpinned rows, oldest use first. Partial on
-- `pinned = 0` because pinned rows are never candidates and including them
-- would make the index proportional to the whole library rather than to the
-- passive cache.
CREATE INDEX image_cache_evictable ON image_cache(last_used)
WHERE pinned = 0;
"#;
const V4: &str = r#"
-- TRACES: FR-CAT-15
-- Soft delete. A trashed image is a real file that has been *moved* to a trash
-- folder under the library root, not a row hidden by a flag: the catalog is a
-- rebuildable index (ARCH §6.12), so a flag alone would evaporate the moment
-- the catalog was deleted and every trashed photograph would return.
--
-- `source_ref` follows the file to its new path, because that is where the bytes
-- now are and every fetch resolves through it. `trashed_from` remembers where it
-- came from, which is the only way a restore can put it back — the trash is flat
-- and the original folder structure is not recoverable from the trashed path.
ALTER TABLE images ADD COLUMN trashed_at INTEGER;
ALTER TABLE images ADD COLUMN trashed_from TEXT;
-- Partial: almost no rows are trashed, and the grid's "not trashed" predicate is
-- answered by the absence of an entry rather than by scanning every image.
CREATE INDEX images_trashed ON images(trashed_at) WHERE trashed_at IS NOT NULL;
"#;
const V3: &str = r#"
-- Ratings and flags are read per grid window and counted for the filter bar's
-- histogram, both of which key on the *default* version. Without this the
-- histogram is a full scan of `versions` on every judgement.
--
-- Partial on `is_default`: a virtual copy's rating is real but is never what
-- these two queries ask for, and excluding them keeps the index roughly one
-- entry per image rather than one per version.
CREATE INDEX versions_judgement ON versions(image_id, rating, flag)
WHERE is_default = 1;
"#;
const V2: &str = r#"
-- A JPEG the camera wrote alongside a RAW of the same name is that RAW's own
-- rendering, not a second photograph. Recording *which* RAW shadows it, rather
-- than a bare flag, keeps the relationship usable: the JPEG is a ready-made
-- preview for its RAW, and the pairing can be undone without a rescan.
ALTER TABLE images ADD COLUMN shadowed_by INTEGER REFERENCES images(id) ON DELETE SET NULL;
CREATE INDEX images_shadowed ON images(shadowed_by) WHERE shadowed_by IS NOT NULL;
"#;
const V1: &str = r#"
-- Roots -------------------------------------------------------------------
CREATE TABLE roots (
id INTEGER PRIMARY KEY,
kind TEXT NOT NULL, -- 'local' | 'saf' | 'remote'
grant_blob BLOB, -- SAF persisted permission; NULL on Linux
label TEXT NOT NULL,
last_seen INTEGER,
-- Bumped once per completed scan. Folders record the generation they were
-- reached in; anything older was not reached and no longer exists.
scan_generation INTEGER NOT NULL DEFAULT 0,
-- One row per granted location. Without this, a rescan inserts a second
-- root for the same folder and the library silently fragments across
-- them — images split between roots, and pruning compares against the
-- wrong generation.
UNIQUE(kind, label)
);
-- Folders: the unit of change detection, local and remote alike -----------
CREATE TABLE folders (
id INTEGER PRIMARY KEY,
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
parent_id INTEGER REFERENCES folders(id) ON DELETE CASCADE,
path TEXT NOT NULL,
-- Remote: the propagating ETag that makes a no-op sync one request.
etag TEXT,
-- Local: directory mtime plus direct-entry count. mtime alone misses a
-- paired create+delete inside one timestamp tick; the count narrows that.
mtime INTEGER,
entry_count INTEGER,
scanned_generation INTEGER NOT NULL DEFAULT 0,
UNIQUE(root_id, path)
);
CREATE INDEX folders_parent ON folders(parent_id);
-- Images ------------------------------------------------------------------
CREATE TABLE images (
id INTEGER PRIMARY KEY,
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
folder_id INTEGER REFERENCES folders(id) ON DELETE CASCADE,
source_ref TEXT NOT NULL,
-- Expensive: requires reading the whole file. Computed only when
-- something needs it (import dedup, reconnect-by-hash), never in a scan.
content_hash TEXT,
format TEXT,
w INTEGER,
h INTEGER,
-- UTC seconds. NULL until EXIF is read, or if the file carries none.
captured_at INTEGER,
-- Minutes east of UTC. A photograph's timestamp is local to where it was
-- taken; storing UTC alone makes a Tokyo shoot span two days in Paris.
captured_offset INTEGER,
camera TEXT,
lens TEXT,
iso INTEGER,
aperture REAL,
shutter REAL,
availability INTEGER NOT NULL DEFAULT 0,
file_size INTEGER,
file_mtime INTEGER,
-- 0 = nothing, 1 = stat-only, 2 = full EXIF. The grid is usable at 1.
metadata_state INTEGER NOT NULL DEFAULT 0,
sidecar_mtime INTEGER,
added_at INTEGER NOT NULL,
UNIQUE(root_id, source_ref)
);
CREATE INDEX images_captured ON images(captured_at);
CREATE INDEX images_folder ON images(folder_id);
-- Partial: content_hash is NULL for most rows most of the time, and the
-- non-NULL subset is exactly what reconnect and dedup query.
CREATE INDEX images_hash ON images(content_hash) WHERE content_hash IS NOT NULL;
-- Versions ----------------------------------------------------------------
CREATE TABLE versions (
id INTEGER PRIMARY KEY,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
is_default INTEGER NOT NULL DEFAULT 0,
graph_hash TEXT,
rating INTEGER NOT NULL DEFAULT 0,
label INTEGER,
flag INTEGER NOT NULL DEFAULT 0
);
CREATE INDEX versions_image ON versions(image_id);
CREATE TABLE keywords (
version_id INTEGER NOT NULL REFERENCES versions(id) ON DELETE CASCADE,
keyword TEXT NOT NULL,
PRIMARY KEY(version_id, keyword)
);
CREATE INDEX keywords_term ON keywords(keyword);
-- Remote mapping ----------------------------------------------------------
CREATE TABLE remote (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
-- oc:fileid — stable across server-side rename and move, so a move is not
-- a re-download of 80 MB.
file_id INTEGER NOT NULL,
etag TEXT,
sync_state INTEGER NOT NULL DEFAULT 0,
remote_path TEXT
);
CREATE UNIQUE INDEX remote_file ON remote(file_id);
-- Collections -------------------------------------------------------------
CREATE TABLE collections (
id INTEGER PRIMARY KEY,
-- Device-independent identity. The integer id is local and collides
-- across devices; the UUID is what a cross-device merge keys on.
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
parent_id INTEGER REFERENCES collections(id) ON DELETE CASCADE,
kind INTEGER NOT NULL, -- 0 = manual, 1 = smart
selector_json TEXT, -- smart only
created INTEGER NOT NULL,
-- Monotonic per collection, bumped on every local edit. Merge compares
-- these rather than file mtimes, so a clock-skewed device cannot silently
-- win.
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL,
-- Tombstone. A deleted collection must outlive its deletion, or a merge
-- with a device that still has it would resurrect it.
deleted INTEGER NOT NULL DEFAULT 0
);
CREATE TABLE collection_members (
collection_id INTEGER NOT NULL REFERENCES collections(id) ON DELETE CASCADE,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
position INTEGER, -- manual ordering; NULL = by capture time
added INTEGER NOT NULL,
PRIMARY KEY(collection_id, image_id)
);
CREATE INDEX members_image ON collection_members(image_id);
-- Cache -------------------------------------------------------------------
CREATE TABLE cache (
id INTEGER PRIMARY KEY,
version_id INTEGER REFERENCES versions(id) ON DELETE CASCADE,
image_id INTEGER REFERENCES images(id) ON DELETE CASCADE,
kind INTEGER NOT NULL, -- thumbnail | proxy | original
resolution INTEGER,
graph_hash TEXT,
path TEXT NOT NULL,
bytes INTEGER NOT NULL,
last_used INTEGER NOT NULL
);
CREATE INDEX cache_lru ON cache(last_used);
CREATE TABLE cache_rules (
id INTEGER PRIMARY KEY,
selector_json TEXT NOT NULL,
tier INTEGER NOT NULL,
priority INTEGER NOT NULL DEFAULT 0,
enabled INTEGER NOT NULL DEFAULT 1
);
CREATE TABLE image_cache (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
tier_actual INTEGER NOT NULL DEFAULT 0,
-- Materialised rather than recomputed, so the grid can draw availability
-- badges without evaluating every rule for every visible cell.
tier_desired INTEGER NOT NULL DEFAULT 0,
bytes INTEGER NOT NULL DEFAULT 0,
last_used INTEGER,
pinned_by_rule INTEGER REFERENCES cache_rules(id) ON DELETE SET NULL
);
-- Jobs --------------------------------------------------------------------
CREATE TABLE jobs (
id INTEGER PRIMARY KEY,
kind INTEGER NOT NULL,
subject_id INTEGER,
priority INTEGER NOT NULL DEFAULT 0,
state INTEGER NOT NULL DEFAULT 0, -- 0=pending 1=running 2=failed
attempts INTEGER NOT NULL DEFAULT 0,
not_before INTEGER NOT NULL DEFAULT 0,
payload TEXT,
last_error TEXT,
-- Coalescing. Enqueueing the same work twice updates one row rather than
-- queueing it twice, which is what makes "enqueue on any change" safe to
-- call liberally.
UNIQUE(kind, subject_id)
);
CREATE INDEX jobs_ready ON jobs(state, priority DESC, not_before);
"#;
#[cfg(test)]
mod tests {
#[test]
fn a_writer_waits_for_its_turn_rather_than_losing_its_work() {
// The failure this exists for: a face sweep that had already paid for
// the detection and the embedding threw the result away on
// "database is locked" and moved on. WAL does not help here — it makes
// one writer and many readers free, and this is two writers.
let dir = std::env::temp_dir().join(format!(
"dr-busy-{}-{:?}",
std::process::id(),
std::thread::current().id()
));
let _ = std::fs::remove_dir_all(&dir);
std::fs::create_dir_all(&dir).unwrap();
let path = dir.join("catalog.sqlite");
let held = rusqlite::Connection::open(&path).unwrap();
configure(&held).unwrap();
migrate(&held).unwrap();
let other = rusqlite::Connection::open(&path).unwrap();
configure(&other).unwrap();
// Every connection carries the timeout, which is what makes the wait
// below a wait rather than an immediate error.
let timeout: i64 = other
.query_row("PRAGMA busy_timeout", [], |r| r.get(0))
.unwrap();
assert_eq!(timeout, BUSY_TIMEOUT.as_millis() as i64);
// A writer holds the database; the other one must still get its turn
// once the first commits, rather than failing at the moment it asks.
let writing = held.unchecked_transaction().unwrap();
held.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
let handle = std::thread::spawn(move || {
other.execute(
"INSERT INTO roots(id, kind, label) VALUES (2, 'local', 'two')",
[],
)
});
std::thread::sleep(std::time::Duration::from_millis(150));
writing.commit().unwrap();
assert!(
handle.join().unwrap().is_ok(),
"the second writer waited and then wrote, rather than erroring"
);
let _ = std::fs::remove_dir_all(&dir);
}
use super::*;
fn mem() -> Connection {
let c = Connection::open_in_memory().unwrap();
configure(&c).unwrap();
c
}
/// How many rows a named backfill touched, ignoring the others.
///
/// Asserting on the whole vector would couple every test to which other
/// backfills happen to exist.
fn backfilled(c: &Connection, what: &str) -> usize {
backfill(c)
.unwrap()
.into_iter()
.find(|(name, _)| *name == what)
.map(|(_, n)| n)
.unwrap_or(0)
}
/// Insert an image and return its id.
fn image(c: &Connection, folder: Option<i64>, name: &str, format: &str) -> i64 {
c.execute(
"INSERT INTO images(root_id, folder_id, source_ref, format, added_at)
VALUES (1, ?1, ?2, ?3, 0)",
rusqlite::params![folder, name, format],
)
.unwrap();
c.last_insert_rowid()
}
fn with_root() -> Connection {
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO folders(id, root_id, path) VALUES (1, 1, 'a'), (2, 1, 'b')",
[],
)
.unwrap();
c
}
#[test]
fn a_jpeg_beside_its_raw_is_shadowed() {
// The camera's own rendering of a frame, not a second photograph.
let c = with_root();
let raw = image(&c, Some(1), "a/IMG_1234.CR2", "cr2");
let jpeg = image(&c, Some(1), "a/IMG_1234.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
let got: Option<i64> = c
.query_row(
"SELECT shadowed_by FROM images WHERE id = ?1",
[jpeg],
|r| r.get(0),
)
.unwrap();
assert_eq!(got, Some(raw));
}
#[test]
fn extension_case_does_not_matter() {
let c = with_root();
image(&c, Some(1), "a/IMG_1.cr2", "cr2");
image(&c, Some(1), "a/img_1.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
}
#[test]
fn a_standalone_jpeg_is_untouched() {
// Scanned film has no RAW sibling and must stay visible — 2,656 of
// them in the reference library.
let c = with_root();
image(&c, Some(1), "a/SCAN_0001.jpg", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 0);
}
#[test]
fn a_jpeg_in_a_different_folder_is_not_shadowed() {
// Camera filenames wrap at IMG_9999, so the same stem recurs across
// shoots (FR-CAT-11). Only a same-folder pair is safe to collapse.
let c = with_root();
image(&c, Some(1), "a/IMG_1234.CR2", "cr2");
image(&c, Some(2), "b/IMG_1234.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 0);
}
#[test]
fn a_raw_is_never_shadowed_by_a_jpeg() {
// The relationship is one-way: the RAW is the photograph.
let c = with_root();
let raw = image(&c, Some(1), "a/IMG_1.CR2", "cr2");
image(&c, Some(1), "a/IMG_1.JPG", "jpg");
backfill(&c).unwrap();
let got: Option<i64> = c
.query_row("SELECT shadowed_by FROM images WHERE id = ?1", [raw], |r| {
r.get(0)
})
.unwrap();
assert_eq!(got, None);
}
#[test]
fn backfill_is_idempotent() {
// It runs on every open, so a second pass must find nothing to do.
let c = with_root();
image(&c, Some(1), "a/IMG_1.CR2", "cr2");
image(&c, Some(1), "a/IMG_1.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
assert_eq!(backfilled(&c, "shadowed_by"), 0, "second pass is a no-op");
}
#[test]
fn a_v1_catalog_gains_the_column_and_is_backfilled() {
// The migration case that motivated this: rows already present when a
// column is added are silently partial until something backfills them.
let c = mem();
c.execute_batch(V1).unwrap();
c.pragma_update(None, "user_version", 1).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, format, added_at)
VALUES (1, 'IMG_9.CR2', 'cr2', 0), (1, 'IMG_9.JPG', 'jpg', 0)",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 1, "migrated from v1");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
}
#[test]
fn a_v4_catalog_gains_the_pinning_columns() {
// TRACES: FR-NC-6a
// An existing library must not have to be rescanned to gain offline
// pinning. The rows are already there; only the columns are new.
let c = mem();
c.execute_batch(V1).unwrap();
c.execute_batch(V2).unwrap();
c.execute_batch(V3).unwrap();
c.execute_batch(V4).unwrap();
c.pragma_update(None, "user_version", 4).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (7, 1, 'IMG_7.CR2', 0)",
[],
)
.unwrap();
// A cache row written before pinning existed.
c.execute(
"INSERT INTO image_cache(image_id, tier_actual, bytes) VALUES (7, 2, 100)",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 4, "migrated from v4");
// The pre-existing row survives, and defaults to unpinned — the safe
// direction, since claiming a pin nobody made would exempt it from
// eviction for ever.
let (pinned, bytes): (i64, i64) = c
.query_row(
"SELECT pinned, bytes FROM image_cache WHERE image_id = 7",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(pinned, 0);
assert_eq!(bytes, 100, "the existing row is untouched");
}
#[test]
fn a_v5_catalog_keeps_its_keywords_and_gains_their_identities() {
// TRACES: FR-CAT-5
// The migration case that matters here: a library keyworded by an
// import or an older build already has assignment rows, and they must
// survive into the vocabulary rather than being left searchable but
// invisible.
let c = mem();
for step in [V1, V2, V3, V4, V5] {
c.execute_batch(step).unwrap();
}
c.pragma_update(None, "user_version", 5).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (7, 1, 'IMG_7.CR3', 0)",
[],
)
.unwrap();
c.execute(
"INSERT INTO versions(id, image_id, uuid, name, is_default)
VALUES (1, 7, 'v-7', 'Default', 1)",
[],
)
.unwrap();
c.execute(
"INSERT INTO keywords(version_id, keyword) VALUES (1, 'puffin')",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 5, "migrated from v5");
assert_eq!(backfilled(&c, "keyword_terms"), 1);
let name: String = c
.query_row("SELECT name FROM keyword_terms", [], |r| r.get(0))
.unwrap();
assert_eq!(name, "puffin");
// The assignment is untouched — it is the durable fact, and the term
// row is only its identity.
let n: i64 = c
.query_row("SELECT count(*) FROM keywords", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 1);
// It runs on every open, so a second pass must find nothing to do.
assert_eq!(backfilled(&c, "keyword_terms"), 0);
}
#[test]
fn two_devices_may_both_hold_a_term_of_the_same_name() {
// Deliberately not a unique index. Two devices each typing "Iceland"
// is the ordinary case, and a constraint would abort the merge
// transaction at exactly the moment they first sync.
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO keyword_terms(uuid, name, created, revision, modified)
VALUES ('a', 'Iceland', 0, 1, 1), ('b', 'Iceland', 0, 1, 1)",
[],
)
.unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM keyword_terms", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 2);
}
#[test]
fn stems_ignore_directories_containing_dots() {
assert_eq!(stem_of("2026.08/IMG_1.CR2"), "IMG_1");
assert_eq!(stem_of("IMG_1.CR2"), "IMG_1");
assert_eq!(stem_of("noextension"), "noextension");
// A dotfile is all stem, not an empty name with an extension.
assert_eq!(stem_of(".hidden"), ".hidden");
}
#[test]
fn migrate_creates_schema_at_current_version() {
let c = mem();
assert_eq!(migrate(&c).unwrap(), 0);
let v: i64 = c
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, SCHEMA_VERSION);
}
#[test]
fn migrate_is_idempotent() {
let c = mem();
migrate(&c).unwrap();
// Re-running must not error or duplicate anything — NFR-R5 requires
// idempotency on retry, since a migration can be interrupted.
assert_eq!(migrate(&c).unwrap(), SCHEMA_VERSION);
}
/// V16 adds its columns guarded, so a catalog whose version was rewound
/// after the columns landed — the rollback NFR-R5 contemplates — migrates
/// again rather than failing on "duplicate column".
#[test]
fn the_eye_columns_survive_a_rewound_version() {
let c = mem();
migrate(&c).unwrap();
for column in EYE_COLUMNS {
let present: bool = c
.prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = ?1")
.unwrap()
.exists([column])
.unwrap();
assert!(present, "{column} missing after migration");
}
c.pragma_update(None, "user_version", 15).unwrap();
assert_eq!(migrate(&c).unwrap(), 15);
let indexed: bool = c
.prepare("SELECT 1 FROM sqlite_master WHERE type = 'index' AND name = 'faces_eyes'")
.unwrap()
.exists([])
.unwrap();
assert!(indexed, "V17's covering index is there");
let v: i64 = c
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, SCHEMA_VERSION);
}
#[test]
fn refuses_a_catalog_from_a_newer_build() {
let c = mem();
migrate(&c).unwrap();
c.pragma_update(None, "user_version", SCHEMA_VERSION + 1)
.unwrap();
// Opening it read-write would corrupt data this build cannot
// represent. Refusing is the specified behaviour (NFR-R5).
assert!(matches!(
migrate(&c),
Err(CatalogError::SchemaTooNew { .. })
));
}
#[test]
fn foreign_keys_cascade_from_root_to_image() {
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)",
[],
)
.unwrap();
c.execute("DELETE FROM roots WHERE id = 1", []).unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 0, "images must not outlive their root");
}
#[test]
fn v12_forgets_runs_made_on_a_proxy_too_small_to_see_a_face() {
let c = mem();
// Migrate to 11, then seed the state V12 exists to repair: markers
// written at the 1024 store tier beside ones written on a real
// preview.
c.pragma_update(None, "user_version", 0).unwrap();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0),(4,1,'d',0)",
[],
)
.unwrap();
for (image, edge) in [(1, 896), (2, 1024), (3, 1025), (4, 2560)] {
c.execute(
"INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge)
VALUES (?1, 'm', 0, 0, ?2)",
rusqlite::params![image, edge],
)
.unwrap();
}
c.pragma_update(None, "user_version", 11).unwrap();
migrate(&c).unwrap();
let kept: Vec<i64> = c
.prepare("SELECT image_id FROM face_index ORDER BY image_id")
.unwrap()
.query_map([], |r| r.get(0))
.unwrap()
.map(Result::unwrap)
.collect();
// 1024 goes: it is exactly ThumbSize::Large, the tier that produced
// the bad runs. 1025 stays, or the floor and the repair disagree
// about the same boundary.
assert_eq!(kept, vec![3, 4]);
}
#[test]
fn v14_forgets_runs_that_found_faces_but_never_measured_them() {
let c = mem();
c.pragma_update(None, "user_version", 0).unwrap();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0)",
[],
)
.unwrap();
// Image 1 was examined and holds a face; 2 was examined and found
// empty; 3 holds a face found by a different model.
for (image, model) in [(1, "m"), (2, "m"), (3, "m")] {
c.execute(
"INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge)
VALUES (?1, ?2, 0, 0, 2560)",
rusqlite::params![image, model],
)
.unwrap();
}
for (image, model) in [(1, "m"), (3, "other")] {
c.execute(
"INSERT INTO faces
(image_id, x, y, w, h, landmarks, detector_confidence, embedding,
crop_px, model_id, detected_at)
VALUES (?1, 0.1, 0.1, 0.2, 0.2, X'00', 0.9, X'00', 180.0, ?2, 0)",
rusqlite::params![image, model],
)
.unwrap();
}
c.pragma_update(None, "user_version", 13).unwrap();
migrate(&c).unwrap();
let kept: Vec<i64> = c
.prepare("SELECT image_id FROM face_index ORDER BY image_id")
.unwrap()
.query_map([], |r| r.get(0))
.unwrap()
.map(Result::unwrap)
.collect();
// 1 goes: it has a face with no quality. 2 stays: nothing on it to
// measure. 3 stays: its face belongs to a run this marker does not
// describe.
assert_eq!(kept, vec![2, 3]);
// And the faces themselves are untouched.
let faces: i64 = c
.query_row("SELECT count(*) FROM faces", [], |r| r.get(0))
.unwrap();
assert_eq!(faces, 2);
}
#[test]
fn v20_renames_markers_to_the_detector_that_found_the_faces() {
let c = mem();
c.pragma_update(None, "user_version", 0).unwrap();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0),(4,1,'d',0),(5,1,'e',0)",
[],
)
.unwrap();
// 1: the desktop's case -- old faces, re-marked as thorough.
// 2: the tablet's case -- adopted thorough faces, re-marked int8,
// and the right marker still beside it (refreshed, so it is
// exported again over the empty entry).
// 3: right already. 4: examined and empty. 5: V14's state, faces
// and no marker.
for (image, model) in [
(1, "scrfd_10g+w600k_mbf"),
(2, "scrfd_10g_i8+w600k_mbf"),
(2, "scrfd_10g+w600k_mbf"),
(3, "scrfd_10g+w600k_mbf"),
(4, "scrfd_10g+w600k_mbf"),
] {
c.execute(
"INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge)
VALUES (?1, ?2, 100, 0, 6000)",
rusqlite::params![image, model],
)
.unwrap();
}
for (image, model) in [
(1, "w600k_mbf"),
(1, "w600k_mbf"),
(2, "scrfd_10g+w600k_mbf"),
(3, "scrfd_10g+w600k_mbf"),
(5, "w600k_mbf"),
] {
c.execute(
"INSERT INTO faces
(image_id, x, y, w, h, landmarks, detector_confidence, embedding,
crop_px, model_id, detected_at)
VALUES (?1, 0.1, 0.1, 0.2, 0.2, X'00', 0.9, X'00', 180.0, ?2, 0)",
rusqlite::params![image, model],
)
.unwrap();
}
c.pragma_update(None, "user_version", 19).unwrap();
migrate(&c).unwrap();
let markers: Vec<(i64, String, i64, bool)> = c
.prepare(
"SELECT image_id, model_id, faces_found, indexed_at > 100
FROM face_index ORDER BY image_id, model_id",
)
.unwrap()
.query_map([], |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)))
.unwrap()
.map(Result::unwrap)
.collect();
assert_eq!(
markers,
vec![
(1, "w600k_mbf".to_string(), 2, true),
(2, "scrfd_10g+w600k_mbf".to_string(), 0, true),
(3, "scrfd_10g+w600k_mbf".to_string(), 0, false),
(4, "scrfd_10g+w600k_mbf".to_string(), 0, false),
]
);
// Re-enterable: nothing left to rename.
c.pragma_update(None, "user_version", 19).unwrap();
migrate(&c).unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM face_index", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 4);
}
#[test]
fn job_uniqueness_coalesces_rather_than_duplicating() {
let c = mem();
migrate(&c).unwrap();
for _ in 0..5 {
c.execute(
"INSERT INTO jobs(kind, subject_id, priority) VALUES (1, 42, 0)
ON CONFLICT(kind, subject_id)
DO UPDATE SET priority = max(priority, excluded.priority)",
[],
)
.unwrap();
}
let n: i64 = c
.query_row("SELECT count(*) FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 1, "five enqueues of the same work is one job");
}
}