Files
DarkRoom/core/dr-catalog/src/schema.rs
T
dtourolle 1f26e1e627 Pair RAW and JPEG from the unpaired JPEGs, not from every RAW
`Catalog::open` runs the backfill every time, and every worker thread
opens its own catalog: the develop view does it to fetch each original and
again for each neighbour it prefetches, and the sync, sweep, burst and
thumbnail workers each do it too. On the reference library (24k images)
an open cost 26 ms of CPU, and most of it was `pair_raw_and_jpeg` reading
all 17,000 RAWs into a map of lowercased stems to find partners for the
1,900 JPEGs that have none -- the same 1,900 on every open.

It now starts from the small side. The unpaired JPEGs are read first, and
it stops there if there are none; otherwise it reads the RAWs in the
folders those JPEGs sit in (plus the unfiled ones when an unfiled JPEG is
waiting), which is 142 on the reference library. A pair is same-folder by
definition, so no pairing is lost; the RAWs are read in id order, so where
two share a stem the later one still wins as it did in the table scan; and
a pass with nothing to pair no longer opens and commits an empty write
transaction.

catalog_bench, best of 20, CPU: `Catalog::open` 26 ms -> 12 ms together
with the next commit (the backfill 24 ms -> 11 ms; this step is ~10 ms of
that). A test covers pairs found among other folders and unfiled images.
2026-09-25 22:06:58 -04:00

2106 lines
88 KiB
Rust

//! TRACES: FR-CAT-2 | NFR-R5
//! Schema definition and forward-only migrations.
//!
//! The catalog is an *index*, not a source of truth (ARCH §6.12) — it is
//! deletable and rebuildable from sources plus sidecars. That is what makes
//! migration failure survivable, and why the recovery path is the normal
//! mechanism rather than a last resort.
//!
//! Migrations are forward-only, transactional, and idempotent on retry
//! (NFR-R5). The app refuses to open a catalog newer than it understands
//! rather than corrupting it.
use rusqlite::Connection;
use crate::error::CatalogError;
/// Schema version this build writes and understands.
pub const SCHEMA_VERSION: i64 = 20;
/// Apply migrations up to [`SCHEMA_VERSION`].
///
/// Returns the version migrated from, so callers can log or back up before a
/// real migration (NFR-R2 requires a backup before schema change).
pub fn migrate(conn: &Connection) -> Result<i64, CatalogError> {
let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
if from > SCHEMA_VERSION {
return Err(CatalogError::SchemaTooNew {
found: from,
supported: SCHEMA_VERSION,
});
}
if from == SCHEMA_VERSION {
return Ok(from);
}
// Each step runs in its own transaction so a failure leaves the catalog
// at a coherent version rather than half-migrated.
if from < 1 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V1)?;
tx.pragma_update(None, "user_version", 1)?;
tx.commit()?;
}
if from < 2 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V2)?;
tx.pragma_update(None, "user_version", 2)?;
tx.commit()?;
}
if from < 3 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V3)?;
tx.pragma_update(None, "user_version", 3)?;
tx.commit()?;
}
if from < 4 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V4)?;
tx.pragma_update(None, "user_version", 4)?;
tx.commit()?;
}
if from < 5 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V5)?;
tx.pragma_update(None, "user_version", 5)?;
tx.commit()?;
}
if from < 6 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V6)?;
tx.pragma_update(None, "user_version", 6)?;
tx.commit()?;
}
if from < 7 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V7)?;
tx.pragma_update(None, "user_version", 7)?;
tx.commit()?;
}
if from < 8 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V8)?;
tx.pragma_update(None, "user_version", 8)?;
tx.commit()?;
}
if from < 9 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V9)?;
tx.pragma_update(None, "user_version", 9)?;
tx.commit()?;
}
if from < 10 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V10)?;
tx.pragma_update(None, "user_version", 10)?;
tx.commit()?;
}
if from < 11 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V11)?;
tx.pragma_update(None, "user_version", 11)?;
tx.commit()?;
}
if from < 12 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V12)?;
tx.pragma_update(None, "user_version", 12)?;
tx.commit()?;
}
if from < 13 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V13)?;
tx.pragma_update(None, "user_version", 13)?;
tx.commit()?;
}
if from < 14 {
let tx = conn.unchecked_transaction()?;
// `ALTER TABLE ... ADD COLUMN` has no `IF NOT EXISTS`, and NFR-R5
// wants this re-enterable: a catalog whose `user_version` was rewound
// by a rollback already has the column, and would otherwise fail its
// next open on it.
let has_quality: bool = tx
.prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = 'quality'")?
.exists([])?;
if !has_quality {
tx.execute_batch("ALTER TABLE faces ADD COLUMN quality REAL;")?;
}
tx.execute_batch(V14)?;
tx.pragma_update(None, "user_version", 14)?;
tx.commit()?;
}
if from < 15 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V15)?;
tx.pragma_update(None, "user_version", 15)?;
tx.commit()?;
}
if from < 16 {
let tx = conn.unchecked_transaction()?;
// Guarded like V14's column, and for the same reason: `ALTER TABLE
// ... ADD COLUMN` has no `IF NOT EXISTS`, and this step must be
// re-enterable (NFR-R5).
for column in EYE_COLUMNS {
let present: bool = tx
.prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = ?1")?
.exists([column])?;
if !present {
tx.execute_batch(&format!("ALTER TABLE faces ADD COLUMN {column} REAL;"))?;
}
}
tx.pragma_update(None, "user_version", 16)?;
tx.commit()?;
}
if from < 17 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V17)?;
tx.pragma_update(None, "user_version", 17)?;
tx.commit()?;
}
if from < 18 {
let tx = conn.unchecked_transaction()?;
// Guarded like V14's and V16's columns: ALTER has no IF NOT EXISTS
// and the step must be re-enterable (NFR-R5).
let present: bool = tx
.prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = 'landmarks_dense'")?
.exists([])?;
if !present {
tx.execute_batch("ALTER TABLE faces ADD COLUMN landmarks_dense BLOB;")?;
}
tx.pragma_update(None, "user_version", 18)?;
tx.commit()?;
}
if from < 19 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V19)?;
tx.pragma_update(None, "user_version", 19)?;
tx.commit()?;
}
if from < 20 {
let tx = conn.unchecked_transaction()?;
v20_markers_name_the_detector_that_found_the_faces(&tx)?;
tx.pragma_update(None, "user_version", 20)?;
tx.commit()?;
}
Ok(from)
}
// V20 -- TRACES: FR-CAT-7
//
// Run markers that named the wrong detector, put right.
//
// `faces::record_updates` -- the write behind the quality, eye and crop
// passes -- re-marked an image under the pipeline the pass ran as, while
// the faces it had updated kept the id of the detector that found them.
// A marker of `scrfd_10g+w600k_mbf` over faces spelled `w600k_mbf` reads,
// to every consumer, as the thorough detector having examined the image:
// the upgrade repair skips it, and `face_shard::export_to_shards` selects
// its faces by the marker's id, finds none, and tells every other device
// that the thorough detector found nothing there. The desktop's shard index
// held 54 such entries over photographs with named faces, and the tablet's
// eye pass over faces it had adopted from the desktop had made 430 more.
//
// The write is fixed to keep the marker under the faces' own id. This puts
// the markers already written right, with a fresh time so the export sends
// each image again under an entry newer than the empty one -- which is what
// `held_model` orders by. Where the right marker is still there beside the
// wrong one (the old write inserted rather than replaced), the wrong one
// goes and the right one is refreshed for the same reason: its entry in
// the shards is older than the empty one, and a device that has neither
// would take the empty one. An image V14 left with faces and no marker at
// all is not touched: that state is the quality pass's cue, and the fixed
// write marks it correctly when the pass reaches it.
//
// Restated in Rust rather than SQL because the embedder half of a pipeline
// id is `faces::embedder_sql`, which this must agree with.
fn v20_markers_name_the_detector_that_found_the_faces(tx: &Connection) -> Result<(), CatalogError> {
let fi = crate::faces::embedder_sql("face_index.model_id");
let f = crate::faces::embedder_sql("f.model_id");
// A marker is wrong when the image holds faces of its embedder under
// another id. First the wrong ones that sit beside a right one -- the
// update below would collide with it -- then the rest are renamed.
let wrong = format!(
"EXISTS (SELECT 1 FROM faces f
WHERE f.image_id = face_index.image_id
AND {f} = {fi}
AND f.model_id != face_index.model_id)"
);
let found_by = format!(
"(SELECT MIN(f.model_id) FROM faces f
WHERE f.image_id = face_index.image_id AND {f} = {fi})"
);
let now = crate::faces::now_secs();
tx.execute(
&format!(
"UPDATE face_index
SET indexed_at = ?1
WHERE model_id = {found_by}
AND EXISTS (SELECT 1 FROM face_index w
WHERE w.image_id = face_index.image_id
AND w.model_id != face_index.model_id
AND {} = {fi})",
crate::faces::embedder_sql("w.model_id")
),
[now],
)?;
tx.execute(
&format!(
"DELETE FROM face_index
WHERE {wrong}
AND EXISTS (SELECT 1 FROM face_index o
WHERE o.image_id = face_index.image_id
AND o.model_id = {found_by})"
),
[],
)?;
tx.execute(
&format!(
"UPDATE face_index
SET model_id = {found_by},
faces_found = (SELECT COUNT(*) FROM faces f
WHERE f.image_id = face_index.image_id AND {f} = {fi}),
indexed_at = ?1
WHERE {wrong}"
),
[now],
)?;
Ok(())
}
/// The seven columns V16 adds to `faces`, in the order the readers name them.
///
/// Named once because three places have to agree on them: this migration,
/// [`for_attached`], and the face shard's own catch-up (`face_shard`).
pub const EYE_COLUMNS: [&str; 7] = [
"eye_right",
"eye_right_px",
"eye_right_sharp",
"eye_left",
"eye_left_px",
"eye_left_sharp",
"sunglasses",
];
/// Recompute columns a migration added, for rows that predate it.
///
/// A migration adds a column with a default; it cannot know what the value
/// *should* be for the rows already present. Without a backfill those rows are
/// silently partial — present, queryable, and wrong — which is worse than
/// missing, because nothing signals that they need attention.
///
/// Cheap enough to run on every open: each pass is one indexed UPDATE, and
/// re-running it is a no-op once the values are already right.
///
/// Returns how many rows each backfill touched, for logging.
pub fn backfill(conn: &Connection) -> Result<Vec<(&'static str, usize)>, CatalogError> {
let mut out = Vec::new();
// v2: `shadowed_by`. A JPEG sitting beside a RAW of the same name is the
// camera's own rendering of that frame, not a second photograph, so it is
// hidden from the grid, the timeline and the sweep.
let n = pair_raw_and_jpeg(conn)?;
if n > 0 {
out.push(("shadowed_by", n));
}
// v3: every image needs a default version to carry its rating and flag.
// Libraries scanned before ratings existed have images and no versions at
// all, so there was nowhere for a judgement to go — see [`crate::rating`].
let n = crate::rating::ensure_default_versions(conn)?;
if n > 0 {
out.push(("default_versions", n));
}
// TRACES: FR-NC-8 | FR-NC-9
// The uuid on those rows is the cross-device merge identity, and it used
// to be generated rather than derived. This comment said so, and said it
// as though generating it were the point — it was the bug. Two devices
// minted different uuids for one photograph, so the sidecar they shared
// grew a `default = 1` block each and neither ever saw the other's work.
//
// Runs after the pass above so a row created a moment ago is already
// derived and matches nothing here. Ordering the other way would be
// correct too, just wasteful.
let n = crate::rating::align_default_version_uuids(conn)?;
if n > 0 {
out.push(("derived_version_uuids", n));
}
// v6: a vocabulary row for every word some image already carries.
//
// Three ways a catalog arrives holding assignments with no term behind
// them, and all three are normal rather than exceptional: a library
// keyworded by a build that predates this table, a catalog rebuilt from
// sidecars (which carry the word and not the identity), and an import from
// Lightroom or darktable (FR-CAT-14). Without this the words are
// searchable but absent from the vocabulary list, which reads as the
// keywords having been lost.
let n = crate::keywords::adopt_orphan_terms(conn)?;
if n > 0 {
out.push(("keyword_terms", n));
}
Ok(out)
}
/// How long a connection waits for a writer to finish before giving up.
///
/// TRACES: NFR-R1
/// SQLite's default is **zero** — the loser of a race gets `SQLITE_BUSY` at
/// once rather than a turn — and WAL does not change that for two writers. One
/// writer and many readers is the case WAL makes free; this is the other one,
/// and this application has it constantly: the face sweep commits a batch while
/// reclustering reads, the derived sync imports shards while the sweep writes.
///
/// Without a timeout that contention was *lost work*, not a retry. A face
/// sweep that had already paid for the detection and the embedding — the
/// expensive part, seconds per image — threw the result away on
/// `storing faces for 214: database is locked` and moved on, and both the
/// desktop and the tablet logged runs of those on consecutive images.
///
/// Ten seconds, matching the figure the job runner's tests already use for the
/// same reason. It is far longer than any transaction here (a sweep batch is
/// sub-second; the slowest is a WAL checkpoint of a 130 MB catalog), so in
/// practice it is a bound on pathology rather than a wait anyone sits through.
/// The tension with NFR-P9 is real but one-sided: a query on the UI thread
/// would rather wait for its turn than fail, because the failure is what the
/// user sees as "cannot open catalog".
const BUSY_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10);
/// Connection setup applied on every open, migration or not.
///
/// WAL is required by NFR-R1: it survives power loss without corruption, and
/// it lets a background job write while the grid reads.
pub fn configure(conn: &Connection) -> Result<(), CatalogError> {
// Before the pragmas, so that a connection racing a migration waits for it
// rather than failing on the first statement it tries.
conn.busy_timeout(BUSY_TIMEOUT)?;
conn.pragma_update(None, "journal_mode", "WAL")?;
// NORMAL rather than FULL: with WAL this is durable across process death
// (which is what FR-PLAT-AND-3 cares about) and only risks the last
// transaction on power loss. The catalog is rebuildable; the sidecars are
// not, and they are written separately with their own fsync discipline.
conn.pragma_update(None, "synchronous", "NORMAL")?;
conn.pragma_update(None, "foreign_keys", true)?;
// A scan touching thousands of rows is transient; let SQLite spill to
// memory rather than materialising temp b-trees on disk.
conn.pragma_update(None, "temp_store", "MEMORY")?;
Ok(())
}
/// The v1 schema rewritten to target an attached database.
///
/// Needed because a downloaded remote catalog is `ATTACH`ed under its own
/// schema name before merging, and tests build one from scratch. SQLite has no
/// "create these tables over there" form, so the names are rewritten.
///
/// The rewrite is textual and therefore only as good as the naming discipline
/// in [`V1`]: every `CREATE TABLE`/`CREATE INDEX` must name its object
/// unqualified, which they do.
pub fn v1_for_attached(schema_name: &str) -> String {
rewrite_for_attached(V1, schema_name)
// REFERENCES within an attached schema resolve to that schema already,
// so foreign keys need no rewriting — but the ON clause of an index
// does, and `CREATE INDEX x.name ON table` is the correct form.
}
/// Every table this build knows about, rewritten to target an attached
/// database.
///
/// [`v1_for_attached`] is kept alongside this rather than replaced by it: a
/// remote catalog written by an older build genuinely has only the v1 tables,
/// and the merge has to keep working against one (see
/// [`crate::merge::merge_keywords`]). Building that case in a test needs a way
/// to say "v1 and no more".
///
/// Only the migrations that *create* objects appear here. V2 through V5 are
/// `ALTER TABLE ... ADD COLUMN`, and the columns they add are local index
/// state — shadowing, trashing, cache pinning — that a merge never reads
/// across the attachment.
///
/// V7 creates an object and is still excluded, which is the one exception to
/// that rule and not an oversight: it indexes `shadowed_by` and `trashed_at`,
/// the very columns V2 through V5 add and this function leaves out, so
/// creating it over there would fail on columns that are not there. Nothing is
/// lost by its absence — it exists to make the *grid* page quickly, and the
/// grid never reads across an attachment.
///
/// V11 is excluded on the same grounds and for the plainer reason that a merge
/// has nothing to do with it: burst grouping is rebuilt locally from local
/// signatures, and no code reads a remote catalog's `burst_*` tables. Its
/// `ALTER TABLE` would fail here anyway, being unqualifiable by the rewrite.
pub fn for_attached(schema_name: &str) -> String {
// V10 is `ALTER TABLE`, which the textual rewrite cannot qualify, so its
// columns are spelled out. A remote genuinely older than V10 is a real
// case and `merge::merge_people_within` probes for them; this is the
// *current* shape, which is what the tests want.
format!(
"{}\n{}\n{}\n\
ALTER TABLE {schema_name}.people ADD COLUMN ignored INTEGER NOT NULL DEFAULT 0;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN crop BLOB;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN quality REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_right REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_right_px REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_right_sharp REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_left REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_left_px REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN eye_left_sharp REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN sunglasses REAL;\n\
ALTER TABLE {schema_name}.faces ADD COLUMN landmarks_dense BLOB;",
rewrite_for_attached(V1, schema_name),
rewrite_for_attached(V6, schema_name),
rewrite_for_attached(V8, schema_name),
)
}
/// Qualify every object a `CREATE` statement names with `schema_name`.
///
/// The rewrite is textual and therefore only as good as the naming discipline
/// in the batches it is given: every `CREATE TABLE`/`CREATE INDEX` must name
/// its object unqualified, which they do.
fn rewrite_for_attached(sql: &str, schema_name: &str) -> String {
sql.replace("CREATE TABLE ", &format!("CREATE TABLE {schema_name}."))
.replace("CREATE INDEX ", &format!("CREATE INDEX {schema_name}."))
.replace(
"CREATE UNIQUE INDEX ",
&format!("CREATE UNIQUE INDEX {schema_name}."),
)
}
/// Mark each JPEG that sits beside a RAW of the same name.
///
/// Matched on folder plus stem, case-insensitively. Same-folder is what makes
/// this safe: cameras write the pair side by side, and matching across folders
/// would risk pairing unrelated frames, since camera filenames wrap at
/// IMG_9999 (FR-CAT-11).
///
/// Done in Rust rather than SQL because the comparison needs a filename stem,
/// and SQLite has no such function without enabling `rusqlite/functions` —
/// a dependency feature for one string operation, whose SQL spelling would be
/// unreadable and would mishandle names with no extension.
fn pair_raw_and_jpeg(conn: &Connection) -> Result<usize, CatalogError> {
use std::collections::HashMap;
// The small side first: the JPEGs not yet paired. On a settled library
// these are the ones with no RAW beside them -- 1,900 of 24,000 on the
// reference library -- and this runs on every open, including the ones
// the develop view makes for each photograph it fetches. Reading every
// RAW to find the handful that share a folder with one of them was most
// of what opening the catalog cost.
let jpegs: Vec<(i64, Option<i64>, String)> = {
let mut stmt = conn.prepare(
"SELECT id, folder_id, source_ref FROM images
WHERE lower(format) IN ('jpg','jpeg') AND shadowed_by IS NULL",
)?;
let rows = stmt.query_map([], |r| {
Ok((
r.get::<_, i64>(0)?,
r.get::<_, Option<i64>>(1)?,
r.get::<_, String>(2)?,
))
})?;
rows.filter_map(Result::ok).collect()
};
if jpegs.is_empty() {
return Ok(0);
}
// (folder, lowercase stem) -> RAW id, over the folders those JPEGs are in
// and no others: a pair is same-folder by definition. Ordered by id so
// that where two RAWs share a stem the later one wins, as it did when this
// read every RAW in table order.
let mut folders: Vec<i64> = jpegs.iter().filter_map(|(_, f, _)| *f).collect();
folders.sort_unstable();
folders.dedup();
let unfiled = jpegs.iter().any(|(_, f, _)| f.is_none());
let folders = serde_json::to_string(&folders).unwrap_or_else(|_| "[]".to_string());
let mut raws: HashMap<(Option<i64>, String), i64> = HashMap::new();
{
let mut stmt = conn.prepare(
"SELECT id, folder_id, source_ref FROM images
WHERE lower(format) IN
('cr2','cr3','nef','arw','raf','rw2','orf','dng')
AND (folder_id IN (SELECT value FROM json_each(?1))
OR (?2 AND folder_id IS NULL))
ORDER BY id",
)?;
let rows = stmt.query_map(rusqlite::params![folders, unfiled], |r| {
Ok((
r.get::<_, i64>(0)?,
r.get::<_, Option<i64>>(1)?,
r.get::<_, String>(2)?,
))
})?;
for row in rows {
let (id, folder, path) = row?;
raws.insert((folder, stem_of(&path).to_ascii_lowercase()), id);
}
}
if raws.is_empty() {
return Ok(0);
}
let pairs: Vec<(i64, i64)> = jpegs
.iter()
.filter_map(|(id, folder, path)| {
let raw = raws.get(&(*folder, stem_of(path).to_ascii_lowercase()))?;
Some((*id, *raw))
})
.collect();
if pairs.is_empty() {
return Ok(0);
}
let tx = conn.unchecked_transaction()?;
for (jpeg, raw) in &pairs {
tx.execute(
"UPDATE images SET shadowed_by = ?2 WHERE id = ?1",
[jpeg, raw],
)?;
}
tx.commit()?;
Ok(pairs.len())
}
/// A filename without its extension.
///
/// Only the final path component, and only its last dot — a directory
/// containing a dot must not truncate the name.
fn stem_of(path: &str) -> &str {
let name = path.rsplit(['/', ':']).next().unwrap_or(path);
match name.rsplit_once('.') {
Some((stem, _)) if !stem.is_empty() => stem,
_ => name,
}
}
/// TRACES: FR-CAT-4 | NFR-P5
/// The order the grid reads in, as an index.
///
/// # What this is for
///
/// Every window the grid loads is `ORDER BY ... LIMIT n OFFSET k`, and without
/// an index that matches the ordering SQLite answers it by sorting the whole
/// library into a temp b-tree and then discarding the first `k` rows. Measured
/// on 24,000 images at offset 20,000, one window read cost 15 ms — a frame
/// budget of 16.7 ms, spent inside the scroll handler, several times per
/// screenful. That is the jitter.
///
/// With this index the same read is a walk along it: 0.36 ms.
///
/// # Why the shape is what it is
///
/// `captured_at IS NULL` leads, because [`crate::library`]'s `GRID_ORDER` does
/// — undated images sort last, and an ordinary index on `captured_at` cannot
/// answer that, since the expression is not a column. SQLite indexes
/// expressions, so it is spelled out here exactly as the query spells it; the
/// two must stay identical or the planner silently falls back to sorting and
/// the cost comes back with no other symptom.
///
/// `source_ref` is included because it breaks ties in the same ordering, and an
/// index that stopped at `captured_at` would leave a sort for the ties.
///
/// # Partial, on the same predicate the grid filters by
///
/// The grid never lists shadowed or trashed rows, so an index carrying them
/// would be larger than the question ever asks about, and — more to the point —
/// a partial index is only usable when its `WHERE` is implied by the query's,
/// which is what makes this one apply to the grid's reads and to nothing else.
///
/// # It does not cover the rating filter or a collection scope
///
/// Both narrow the walk rather than reorder it, so the index still supplies the
/// ordering and SQLite tests the extra predicate per row. That is the cheap
/// direction: the expensive part was never the filtering, it was the sort.
const V10: &str = r#"
-- TRACES: FR-CULL-10 | FR-CULL-12
-- Two columns the People screen turned out to need, and neither is derivable.
-- A person the user does not want to identify.
--
-- Most of a real library's clusters are strangers: people in the background of
-- a street, guests at somebody else's party, a face on a poster. They are
-- correctly detected and correctly grouped, and the user will never name any of
-- them -- but they crowd out the handful of groups that matter, and there is no
-- way to tell "not yet looked at" from "looked at, don't care" without
-- recording the second.
--
-- **User data**, and the reason this is a column rather than a deletion: a
-- deleted cluster comes straight back on the next Regroup, because the faces
-- are still there and still similar. Nothing short of remembering the judgement
-- survives re-clustering, which is the same argument `face_person_rejected`
-- makes one level down (FR-CULL-12).
ALTER TABLE people ADD COLUMN ignored INTEGER NOT NULL DEFAULT 0;
-- The face, cut out and kept.
--
-- A face used to be drawn by decoding the 1024px proxy it was found on and
-- cutting the box out again, every time the screen opened. That made the People
-- screen a *derivative of the thumbnail cache*: evict a proxy -- which the
-- cache is entitled to do at any moment -- and the cell goes blank, with no way
-- back short of re-fetching the original over the network and re-detecting it.
-- It also cost one full JPEG decode per image per visit to show a 96px cell.
--
-- So the crop is cut once, when the pixels are already in hand at detection
-- time, and kept. Small: a 160px JPEG is a few KB, against ~250 KB for the
-- proxy it replaces reading.
--
-- Nullable, because a face indexed before this column existed has no crop and
-- must still work -- the reader falls back to the old proxy path, and the next
-- indexing pass fills it in.
--
-- **Stripped from the sync snapshot.** The catalog is uploaded whole, so this
-- would otherwise put tens of MB of JPEG on every sync; crops travel in the
-- face shards instead, which is where the bulk per-face data already goes
-- (`face_shard`). See `sync::snapshot_for_upload`.
ALTER TABLE faces ADD COLUMN crop BLOB;
"#;
/// TRACES: FR-CULL-5
/// Burst grouping: which frames are one moment, and which one stands for it.
///
/// The reasoning behind the grouping itself is in [`crate::bursts`]; what
/// belongs here is why it is stored in three pieces rather than one.
///
/// **`images.perceptual_hash` is a column, not a table**, for the same reason
/// `content_hash` is: it is one number per image, NULL until something has had
/// the pixels in hand, and every query that wants it is already reading the
/// image row. It is local derived state — a rebuilt catalog recomputes it from
/// thumbnails — which is also why it is absent from [`for_attached`], alongside
/// the shadowing and trashing columns V2 through V5 add.
///
/// **`burst_members` is rewritten whole by every pass.** No id of its own: the
/// group is named by the image id of its earliest frame, so a burst that has not
/// changed keeps its name across a regroup and the interface can remember that
/// this one is open. There is no `bursts` table to go with it because a group
/// has no properties beyond its members — inventing a row for it would create an
/// identity that survives the grouping being rebuilt, which is precisely what
/// must not happen.
///
/// **`burst_pick` is the one thing here that is not derived**, and it is a
/// separate table so that rewriting the grouping cannot erase it. A
/// representative stored on `burst_members` would be forgotten every time a
/// frame arrived; the user would be asked the same question after every scan.
/// The same argument `people.ignored` makes in V10, one subsystem over.
///
/// **`burst_expanded` is view state in the catalog**, which is unusual enough to
/// justify. The grid is a window over an ordered query — `LIMIT n OFFSET k` —
/// so what a collapsed burst hides has to be decided by the query, or the row
/// count stops agreeing with the scrollbar and the ordinals a scrub resolves to.
/// Once SQL has to see it, this is where it lives. Nothing else reads it, and it
/// is emptied of stale groups by every pass.
const V11: &str = r#"
-- A 64-bit perceptual signature. Local derived state: NULL until something has
-- decoded the image, recomputed from thumbnails if the catalog is rebuilt, and
-- comparable only to signatures produced by the same build (`bursts`).
ALTER TABLE images ADD COLUMN perceptual_hash INTEGER;
CREATE TABLE burst_members (
-- One burst at most per image: a frame belongs to the moment it was taken
-- in, and nothing else.
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
-- The image id of the burst's earliest frame. Not a foreign key by
-- accident: the leader is itself a member, so this genuinely references
-- images(id), and cascading its deletion is right.
burst_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
-- The frame the group collapses to. Exactly one per burst.
representative INTEGER NOT NULL DEFAULT 0
);
-- Counting a burst's frames and listing them are what the grid asks for, once
-- per window; without this both are a scan of every grouped frame in the
-- library.
CREATE INDEX burst_members_burst ON burst_members(burst_id);
-- The user's own choice of representative. User data, never rewritten by a
-- grouping pass -- see the module doc above.
CREATE TABLE burst_pick (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE
);
-- Bursts the grid is currently showing in full.
CREATE TABLE burst_expanded (
burst_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE
);
"#;
const V12: &str = r#"
-- TRACES: FR-CULL-8
-- Forget the runs that were made against a proxy too small to find a face on.
--
-- Detection used to accept any proxy, and one of the two sweeps detected on
-- the stored 1024px tier. On the reference library that produced 0.078 faces
-- per image against 1.82 for the same photographs at 2048 or better -- and
-- every one of those runs left a `face_index` row behind saying the image had
-- been examined. That row is what makes the damage permanent: the work list is
-- "images with no row", so a photograph examined badly is indistinguishable
-- from one examined well, and is never offered to a later pass.
--
-- Deleting the marker is the whole repair, and it is deliberately not a
-- deletion of anything else. The `faces` rows those runs found stay exactly
-- where they are and keep drawing the People screen until a better pass
-- replaces them, and `record_detections` carries the user's confirmed names
-- across that replacement by box overlap. So this costs a re-fetch of the
-- affected images and loses no work the user has done.
--
-- The threshold is written out rather than taken from `dr_face::MIN_CROP_EDGE`
-- on purpose. A migration has to keep meaning what it meant on the day it ran;
-- binding it to a constant someone may raise later would silently change what
-- an old catalog gets migrated to.
DELETE FROM face_index WHERE source_edge <= 1024;
"#;
const V13: &str = r#"
-- TRACES: FR-CAT-8 | FR-NC-9
-- Which sidecars this device has read, and at what ETag.
--
-- The sidecar is the authoritative store for a rating and an edit, and until
-- this table existed nothing ever read one back into the catalog: judgements
-- travelled outward only. A cull done on a tablet reached the server and
-- stopped there, because the scan indexes photographs, the derived sync moves
-- thumbnails and collections, and the one reader that existed ran when a single
-- photograph was opened in develop and fed only the develop graph. The grid
-- draws `versions.rating`, so another device's afternoon of culling was
-- invisible on this one -- permanently, by every path the app had.
--
-- What this holds is the ETag, not the content. It is the record of what has
-- already been taken in, so a pull fetches only what changed: `dr_sync::scan`
-- reports every sidecar it saw in listings it was making anyway, and this
-- decides which of them are worth a GET.
--
-- Keyed on the sidecar's own remote path rather than on an image id. One
-- sidecar can describe two images -- a RAW and the JPEG beside it are one
-- photograph (FR-CAT-11) and share a document -- and a path is what the scan
-- reports and what a fetch addresses, so keying on anything else would mean
-- deriving one from the other in two places.
--
-- Rebuildable like the rest of the catalog: losing this table costs one pass
-- that re-reads every sidecar and reaches exactly the same state.
--
-- `IF NOT EXISTS` because NFR-R5 asks for migrations that are idempotent on
-- retry, and this one can genuinely be re-entered: a catalog whose
-- `user_version` was rewound -- by a rollback to an older build, or by a
-- recovery -- would otherwise fail its next open on a table it already has.
CREATE TABLE IF NOT EXISTS sidecars (
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
path TEXT NOT NULL,
etag TEXT,
-- Unix seconds, for diagnosing a pull that is not making progress.
read_at INTEGER NOT NULL DEFAULT 0,
PRIMARY KEY(root_id, path)
);
"#;
const V14: &str = r#"
-- TRACES: FR-CULL-9 | FR-CULL-10
-- How recognisable the model found each face, and a second look at the faces
-- it was never asked about.
--
-- The embedder's raw output has a length, and the length is a quality
-- reading: it grows with how much of a face the model could make out, and a
-- blur, an occlusion or a hard profile comes out short (dr_face::embedding,
-- `MIN_GALLERY_QUALITY`). Normalising threw it away. A short vector sits
-- near the middle of the sphere and matches a little of everyone, which is
-- how one bad crop bridges two people in a grouping pass -- so a face below
-- the floor is compared against the others and never compared *against*.
--
-- Nullable, and NULL means "never measured": every face indexed before this
-- version stored the unit vector, whose length is one whatever the crop was.
-- A face with no reading is admitted to the gallery, because a rule that
-- cannot be checked should admit rather than exclude -- but it is also a
-- face this rule is not yet protecting anyone from, and the only way to
-- measure it is to embed it again.
--
-- The `face-quality` repair is what does that (`dr_ui::repairs`, once the
-- sweep's measuring pass): it lists every face with no reading, and each is
-- embedded again from the native render with the landmarks it already has,
-- the raw vector written over the old one (`record_updates`) and nothing
-- else touched -- not the id, not the box, not who the user said it was.
-- The faces keep drawing the People screen throughout.
--
-- The run markers of those images are forgotten too, exactly as V12 forgot
-- the runs made against too small a proxy. The build this shipped in had no
-- measuring pass yet, and a marker is the one thing that stops a face ever
-- being looked at again; with the repair in place, detection leaves an
-- image holding this embedder's faces to it rather than detecting from
-- scratch, so the deletion costs nothing -- and an image that was examined
-- and found empty keeps its marker, since there is nothing on it to measure.
--
-- The cost is a re-fetch of every image with a face on it, on the next pass
-- the user starts. That is a whole-library transfer (FR-NC-6), and it starts
-- when they say so, not here.
--
-- From this version the `embedding` blob is the **raw** model output rather
-- than the unit vector V8 describes -- the length is the quality, and a store
-- that kept only the direction had thrown it away. Readers re-normalise on
-- load, so a unit blob from before and a raw blob from now compare alike;
-- `quality` is that length kept beside the blob for the readers that never
-- load the vector, and NULL rather than 1.0 for the old rows, because a unit
-- vector reads as a length of one and one is not "unmeasured".
--
-- The column itself is added in `migrate`, guarded, because ALTER has no
-- IF NOT EXISTS and this step has to be re-enterable (NFR-R5).
DELETE FROM face_index
WHERE EXISTS (SELECT 1 FROM faces f
WHERE f.image_id = face_index.image_id
AND f.model_id = face_index.model_id);
"#;
const V15: &str = r#"
-- TRACES: FR-CAT-13
-- Where a standard XMP sidecar and the catalog disagree.
--
-- An `.xmp` beside a photograph is read on the same pull as DarkRoom's own
-- sidecar, and reconciled field by field (`dr_xmp::reconcile`): keywords
-- union, and a rating, label or caption is taken only where the catalog holds
-- none. That rule is the safe one and it is not always the right one -- a
-- rating changed in Lightroom after it was changed here is a genuine
-- disagreement, and a standard XMP carries no revision to settle it by. So
-- the disagreement is written here instead of being resolved, and the
-- requirement's "a metadata reload offered" is a row in this table with a
-- button in front of it: the reload re-reads the file with the sidecar
-- winning, and deletes the row.
--
-- Keyed on the sidecar's path like `sidecars` is, and for the same reason: a
-- path is what the scan reports, what a fetch addresses, and what the ETag
-- that noticed the change belongs to. `fields` is the disagreeing fields as
-- `dr_xmp` names them, space-separated, for the line the settings page shows.
--
-- Rebuildable: the next pull that sees a changed ETag writes the row again.
CREATE TABLE IF NOT EXISTS xmp_conflicts (
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
path TEXT NOT NULL,
fields TEXT NOT NULL,
seen_at INTEGER NOT NULL DEFAULT 0,
PRIMARY KEY(root_id, path)
);
"#;
// V19 -- TRACES: NFR-P9
//
// The indexes the repair counts are served from, and V17's lesson applied
// to the rest of the face columns.
//
// "How many images still owe a quality reading" was answered per image: a
// correlated EXISTS over `faces` that had to open each face's row to look
// at one nullable column -- the row being eight kilobytes of embedding and
// crop. Six such counts run every time the Identity screen opens and every
// time a sweep ends, 160 ms of them on the reference library. Three
// partial indexes hold only the faces still owing each pass, keyed by the
// image and carrying the model id the predicate also reads, so the count
// walks a few thousand index entries and touches no row at all -- and each
// index shrinks to nothing as its pass completes. The planner takes them
// when the count is driven from `faces` (`repairs::count`) and ignores
// them inside the per-image EXISTS, which is why that function has two
// spellings of the same predicate.
//
// `faces_image_model` replaces `faces_image`: the same key with the model
// id beside it, so "does this image hold this embedder's faces" -- asked in
// the audit, the proxy repair and the outstanding-detection count -- is an
// index-only probe where it used to read the row for the model id. Every
// lookup that used `faces_image` is served by its prefix.
//
// Not applied to attached catalogs, like V7 and V17: an index is a local
// concern, and a merge never runs these queries across an attachment.
const V19: &str = r#"
CREATE INDEX IF NOT EXISTS faces_image_model ON faces(image_id, model_id);
DROP INDEX IF EXISTS faces_image;
CREATE INDEX IF NOT EXISTS faces_owed_quality ON faces(image_id, model_id)
WHERE quality IS NULL;
CREATE INDEX IF NOT EXISTS faces_owed_crop ON faces(image_id, model_id)
WHERE crop IS NULL;
CREATE INDEX IF NOT EXISTS faces_owed_eyes ON faces(image_id, model_id)
WHERE eye_right IS NULL OR landmarks_dense IS NULL;
"#;
// V18 -- TRACES: FR-CULL-8a | FR-CULL-12
//
// The 106 dense landmarks the eye pass reads its eye boxes from, kept beside
// the reading as `dr_face::Landmarks::to_packed_bytes`: 106 x (x, y) as
// 16-bit fixed point over the frame, 424 bytes a face, a seventh of a
// pixel on a 6000-pixel frame. Derived data under FR-CULL-12 -- rebuilt by
// re-reading, never in a sidecar -- and stored for the same reason the
// embedding is: it cost a fetch of the original and a model run, and the
// next per-face pass (head pose, expression) should not have to pay either
// again. NULL where the face was never read.
//
// Added in `migrate`, guarded, like every ALTER here (NFR-R5).
const V17: &str = r#"
-- TRACES: FR-CULL-8a | FR-CULL-13 | NFR-P9
-- The eyes-open filter's index, and a lesson about where a column lands.
--
-- The people filter is a correlated EXISTS over `faces` per image, and it
-- was fast because `faces_image` *covers* it: the subquery never touched a
-- row. Reading V16's seven eye columns in the same subquery did touch the
-- row -- and `ALTER TABLE ADD COLUMN` puts a column at the end of the
-- record, after the 1 KB embedding and the ~5 KB crop, so every check
-- dragged six kilobytes off disk to reach seven floats. Measured on the
-- reference library: 24 seconds for one count, thirteen of them system
-- time. With this index the same count takes five milliseconds, because
-- the subquery is served from the index again and never reads a row.
--
-- The columns are listed in EYE_COLUMNS' order behind `image_id`, which is
-- the key the subquery searches on. Nothing else changed in V17; a catalog
-- already at V16 needs only this.
CREATE INDEX IF NOT EXISTS faces_eyes ON faces(
image_id, eye_right, eye_right_px, eye_right_sharp,
eye_left, eye_left_px, eye_left_sharp, sunglasses
);
"#;
// V16 -- TRACES: FR-CULL-8a
//
// What each face's eyes are doing: for each eye P(open), the source pixels
// across its box and the sharpness of the patch the classifier saw; and
// P(sunglasses) for the head. Seven numbers rather than a verdict, because
// the verdict is a rule with thresholds in it (dr_face::eyes::EyeReading::
// state) and a rule belongs in code that can be changed, not in rows that
// would have to be re-measured.
//
// The pixels and the sharpness are what stop a smear reading as a blink: an
// eye too small or too soft to read is not asked, and a face with no
// readable eye is "unclear", which no filter drops. Sunglasses are a column
// of their own for the same kind of reason — the eye classifier answers
// confidently over dark glass, and its answer means nothing there. A filter
// for "eyes open" reads all seven.
//
// NULL means "never measured" -- a face indexed before this version, or on a
// device without the eye models -- and a NULL is left alone by every filter
// that reads these, so an old library does not empty its grid the moment the
// chip is pressed. The sweep's measuring pass fills them in, from the native
// render, with the landmarks already stored: the same pass V14 built for the
// embedding's length, extended to ask the eye models too. No run marker is
// forgotten here, for the reason V14's note gives -- the measuring pass
// finds its own work by the NULL, and deleting markers would only put the
// detector back over images it has finished with.
//
// The columns are added in `migrate`, guarded, because ALTER has no IF NOT
// EXISTS and the step has to be re-enterable (NFR-R5). Their names are
// `EYE_COLUMNS`.
const V9: &str = r#"
-- TRACES: FR-CULL-8
-- A record that face detection has *run* on an image, distinct from what it
-- found.
--
-- # Why the faces table cannot answer this
--
-- Without this, "has this image been indexed" is asked as "does it have any
-- faces", and those are not the same question. **A photograph with no faces in
-- it is indistinguishable from one that has never been looked at**, so every
-- indexing pass re-examines every landscape, every still life and every
-- document scan in the library, for ever. In a typical personal library that is
-- most of it: the pass never converges, and the cost is paid again on every
-- run rather than once.
--
-- It also makes a coverage figure possible, which is the thing a user actually
-- wants to see — "4,812 of 5,000 images indexed" — where counting face rows
-- can only ever report how many faces exist.
--
-- # Why it is keyed on the model
--
-- Embeddings from different models are not comparable, so a model change has
-- to re-index. Keying the marker on `(image_id, model_id)` makes that
-- automatic: new model, no marker, image comes back into the queue. The id
-- names the whole pipeline -- detector and embedder together -- because
-- changing either changes what is found.
CREATE TABLE face_index (
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
model_id TEXT NOT NULL,
indexed_at INTEGER NOT NULL,
-- Zero is a real and common answer, and recording it is the entire point.
faces_found INTEGER NOT NULL,
-- Long edge of the proxy this ran against. A face too small to detect at
-- 1024 may be findable at 2048, so a library whose proxies grow can
-- re-index the images that stand to gain instead of all of them.
source_edge INTEGER NOT NULL,
PRIMARY KEY (image_id, model_id)
);
CREATE INDEX face_index_model ON face_index(model_id);
"#;
const V8: &str = r#"
-- TRACES: FR-CULL-8 | FR-CULL-9 | FR-CULL-10 | FR-CULL-11 | FR-CULL-12 | NFR-SEC-5
-- People and faces (docs/dev/faces.md, docs/dev/catalog.md §10).
--
-- Everything here is **derived data** except one column. Faces, landmarks,
-- embeddings, cluster assignments and suggestions are all reproducible by
-- re-indexing and are never written to a sidecar (FR-CULL-12); a person's
-- *name*, once the user has confirmed it, is a human judgement of the same
-- class as a rating and travels with the photograph.
--
-- That asymmetry is the whole design: deleting the catalog costs an afternoon
-- of re-indexing and loses nothing the user typed (ARCH §6.12).
CREATE TABLE people (
id INTEGER PRIMARY KEY,
-- Merge identity, not the name. Two devices that independently name the
-- same cluster produce two people; merging them keys on this, exactly as
-- collections do (FR-CAT-7, ARCH §6.3).
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
-- Tombstone-by-redirect. A merged person must outlive its merge or a
-- device that still holds it resurrects it on the next sync -- the same
-- hazard collections have, solved the same way.
merged_into INTEGER REFERENCES people(id) ON DELETE SET NULL,
created INTEGER NOT NULL,
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL
);
CREATE TABLE faces (
id INTEGER PRIMARY KEY,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
-- Normalised to the image's long edge, so a face survives the proxy it was
-- found on being evicted and regenerated at another resolution. Storing
-- pixels would bind a face to a resolution the cache is entitled to change.
x REAL NOT NULL, y REAL NOT NULL, w REAL NOT NULL, h REAL NOT NULL,
landmarks BLOB NOT NULL, -- 5 x (x, y) f32, normalised likewise
detector_confidence REAL NOT NULL,
embedding BLOB NOT NULL, -- 512 x f16; unit length until V14, raw since
-- Source pixels across the aligned 112x112 crop (docs/dev/faces.md §7).
--
-- Not cosmetic: it is the honest quality signal for the UI, a feature in
-- the §8 calibration -- FR-CULL-9 names face size as an axis along which an
-- uncalibrated similarity misbehaves -- and the selector a later
-- higher-resolution re-embedding pass would run on.
crop_px REAL NOT NULL,
-- Which model produced this embedding.
--
-- The one mistake in this subsystem that yields plausible-looking garbage
-- rather than an error: embeddings from different models are not
-- comparable. Storing the model with the vector makes a model change
-- detectable and re-indexable instead of quietly poisoning every
-- similarity in the library.
model_id TEXT NOT NULL,
detected_at INTEGER NOT NULL
);
CREATE INDEX faces_image ON faces(image_id);
CREATE INDEX faces_model ON faces(model_id);
CREATE TABLE face_person (
face_id INTEGER PRIMARY KEY REFERENCES faces(id) ON DELETE CASCADE,
person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE,
-- Calibrated P(this face is this person), never a raw cosine (FR-CULL-9).
probability REAL NOT NULL,
-- The user said so. Never overwritten by a later inference pass.
--
-- A column rather than a probability of 1.0, because a confirmation is a
-- different kind of fact from a confident guess and collapsing them loses
-- the ability to recompute suggestions without touching user data.
confirmed INTEGER NOT NULL DEFAULT 0
);
CREATE INDEX face_person_person ON face_person(person_id, confirmed);
-- Faces the user has explicitly said are NOT a given person.
--
-- Needed because rejection is not the absence of an assignment: without it,
-- the next clustering pass re-suggests exactly the face the user just pushed
-- away, and the tool feels broken. Same reasoning as `confirmed` -- a
-- judgement is user data (FR-CULL-12) whichever direction it points.
CREATE TABLE face_person_rejected (
face_id INTEGER NOT NULL REFERENCES faces(id) ON DELETE CASCADE,
person_id INTEGER NOT NULL REFERENCES people(id) ON DELETE CASCADE,
PRIMARY KEY (face_id, person_id)
);
-- The FR-CULL-9 calibration, fitted from this library's own faces.
--
-- One row per model, because the fit is a property of the embedding space and
-- a library indexed across a model change holds two. `face_set_hash` is what
-- makes a stale fit detectable: a materially changed library recomputes rather
-- than trusting numbers derived from a set that no longer exists.
CREATE TABLE face_calibration (
model_id TEXT PRIMARY KEY,
-- P(same) = sigmoid(a*cos + b + w_size*log2(min(crop_px)) + log_prior_odds)
a REAL NOT NULL,
b REAL NOT NULL,
w_size REAL NOT NULL DEFAULT 0.0,
-- Whether the fit is usable at all. When it is not, the UI says the
-- confidence is unavailable; it does not present an untuned default as
-- though it were measured (FR-CULL-9).
valid INTEGER NOT NULL DEFAULT 0,
positive_pairs INTEGER NOT NULL DEFAULT 0,
negative_pairs INTEGER NOT NULL DEFAULT 0,
face_set_hash TEXT NOT NULL,
fitted_at INTEGER NOT NULL
);
"#;
const V7: &str = r#"
CREATE INDEX images_grid_order
ON images(captured_at IS NULL, captured_at, source_ref)
WHERE shadowed_by IS NULL AND trashed_at IS NULL;
"#;
const V6: &str = r#"
-- TRACES: FR-CAT-5 | FR-CAT-6 | FR-NC-9
-- Keywords gain an identity, so that renaming and deleting one can cross
-- between devices.
--
-- The v1 `keywords` table is the *assignment*: one row per (version, word),
-- and the word is stored as text. That stays exactly as it is, and this
-- migration adds nothing to it, for a reason that is easy to get backwards.
--
-- # Why assignments keep the text rather than pointing at a row here
--
-- The catalog is a rebuildable index (ARCH §6.12). What an image is keyworded
-- with is authoritative in the sidecar and in XMP `dc:subject` (FR-CAT-13),
-- and both of those carry a *string*. Rewriting the join to reference
-- `keyword_terms(id)` would mean a catalog rebuilt from sidecars had to invent
-- term rows before it could record a single assignment, and an integer that
-- means nothing on the other device would sit where the durable fact belongs.
-- It would also break `crate::query`, which matches `kw.keyword` directly and
-- must keep hitting `keywords_term` on a 50k library (FR-CAT-6).
--
-- So the text is the fact and this table is the *identity*: it exists to give
-- a rename and a deletion something a merge can key on, and to let a keyword
-- exist in the vocabulary before any photograph carries it.
CREATE TABLE keyword_terms (
id INTEGER PRIMARY KEY,
-- Device-independent identity, as `collections.uuid` is. The integer id is
-- local and collides across devices.
uuid TEXT NOT NULL UNIQUE,
-- The word itself, and the value written into every assignment row.
name TEXT NOT NULL,
created INTEGER NOT NULL,
-- Monotonic, bumped on every local edit. `crate::merge` compares these
-- rather than timestamps, so a clock-skewed device cannot silently win.
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL,
-- Tombstone, so a merge against a device that still holds the keyword does
-- not resurrect it.
deleted INTEGER NOT NULL DEFAULT 0
);
-- Deliberately **not** UNIQUE.
--
-- Two devices that each type "Iceland" create two rows with two uuids, and
-- both are correct until they meet. A unique constraint would abort the merge
-- transaction at exactly that moment — the ordinary case, not a corner one.
-- Uniqueness is instead reached by convergence: `crate::keywords::create`
-- resolves an existing name locally, and `crate::keywords::fuse_duplicates`
-- collapses a cross-device pair onto the lexicographically smaller uuid, which
-- both devices compute identically without talking to each other.
--
-- Partial on `deleted = 0` because every lookup here is a live one: the
-- vocabulary list, the resolve-by-name in `create`, and the fuse pass all
-- exclude tombstones, and including them would grow the index with every
-- keyword the library has ever had rather than with the ones it has.
CREATE INDEX keyword_terms_name ON keyword_terms(name) WHERE deleted = 0;
"#;
const V5: &str = r#"
-- TRACES: FR-NC-6a | FR-CAT-9 | NFR-RES-4
-- Offline availability: what is kept, why it is kept, and where it lives.
--
-- `pinned` separates a promise from a convenience, and the distinction has to
-- be a *column* rather than something inferred from `pinned_by_rule`. A pin is
-- the user saying "this collection comes with me"; a passively cached original
-- is the app noticing they opened something. Only the second is evictable, so
-- the eviction query has to be able to ask the question directly — and it has
-- to keep answering correctly for an image whose pinning rule was since
-- deleted, which `pinned_by_rule` alone cannot do because it is
-- ON DELETE SET NULL.
ALTER TABLE image_cache ADD COLUMN pinned INTEGER NOT NULL DEFAULT 0;
-- Where the cached original actually is, relative to the cache directory.
-- Relative rather than absolute: the library moves between machines and
-- between an app sandbox and a user directory, and an absolute path baked in
-- at download time would break on every one of those.
ALTER TABLE image_cache ADD COLUMN path TEXT;
-- Eviction reads exactly this: unpinned rows, oldest use first. Partial on
-- `pinned = 0` because pinned rows are never candidates and including them
-- would make the index proportional to the whole library rather than to the
-- passive cache.
CREATE INDEX image_cache_evictable ON image_cache(last_used)
WHERE pinned = 0;
"#;
const V4: &str = r#"
-- TRACES: FR-CAT-15
-- Soft delete. A trashed image is a real file that has been *moved* to a trash
-- folder under the library root, not a row hidden by a flag: the catalog is a
-- rebuildable index (ARCH §6.12), so a flag alone would evaporate the moment
-- the catalog was deleted and every trashed photograph would return.
--
-- `source_ref` follows the file to its new path, because that is where the bytes
-- now are and every fetch resolves through it. `trashed_from` remembers where it
-- came from, which is the only way a restore can put it back — the trash is flat
-- and the original folder structure is not recoverable from the trashed path.
ALTER TABLE images ADD COLUMN trashed_at INTEGER;
ALTER TABLE images ADD COLUMN trashed_from TEXT;
-- Partial: almost no rows are trashed, and the grid's "not trashed" predicate is
-- answered by the absence of an entry rather than by scanning every image.
CREATE INDEX images_trashed ON images(trashed_at) WHERE trashed_at IS NOT NULL;
"#;
const V3: &str = r#"
-- Ratings and flags are read per grid window and counted for the filter bar's
-- histogram, both of which key on the *default* version. Without this the
-- histogram is a full scan of `versions` on every judgement.
--
-- Partial on `is_default`: a virtual copy's rating is real but is never what
-- these two queries ask for, and excluding them keeps the index roughly one
-- entry per image rather than one per version.
CREATE INDEX versions_judgement ON versions(image_id, rating, flag)
WHERE is_default = 1;
"#;
const V2: &str = r#"
-- A JPEG the camera wrote alongside a RAW of the same name is that RAW's own
-- rendering, not a second photograph. Recording *which* RAW shadows it, rather
-- than a bare flag, keeps the relationship usable: the JPEG is a ready-made
-- preview for its RAW, and the pairing can be undone without a rescan.
ALTER TABLE images ADD COLUMN shadowed_by INTEGER REFERENCES images(id) ON DELETE SET NULL;
CREATE INDEX images_shadowed ON images(shadowed_by) WHERE shadowed_by IS NOT NULL;
"#;
const V1: &str = r#"
-- Roots -------------------------------------------------------------------
CREATE TABLE roots (
id INTEGER PRIMARY KEY,
kind TEXT NOT NULL, -- 'local' | 'saf' | 'remote'
grant_blob BLOB, -- SAF persisted permission; NULL on Linux
label TEXT NOT NULL,
last_seen INTEGER,
-- Bumped once per completed scan. Folders record the generation they were
-- reached in; anything older was not reached and no longer exists.
scan_generation INTEGER NOT NULL DEFAULT 0,
-- One row per granted location. Without this, a rescan inserts a second
-- root for the same folder and the library silently fragments across
-- them — images split between roots, and pruning compares against the
-- wrong generation.
UNIQUE(kind, label)
);
-- Folders: the unit of change detection, local and remote alike -----------
CREATE TABLE folders (
id INTEGER PRIMARY KEY,
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
parent_id INTEGER REFERENCES folders(id) ON DELETE CASCADE,
path TEXT NOT NULL,
-- Remote: the propagating ETag that makes a no-op sync one request.
etag TEXT,
-- Local: directory mtime plus direct-entry count. mtime alone misses a
-- paired create+delete inside one timestamp tick; the count narrows that.
mtime INTEGER,
entry_count INTEGER,
scanned_generation INTEGER NOT NULL DEFAULT 0,
UNIQUE(root_id, path)
);
CREATE INDEX folders_parent ON folders(parent_id);
-- Images ------------------------------------------------------------------
CREATE TABLE images (
id INTEGER PRIMARY KEY,
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
folder_id INTEGER REFERENCES folders(id) ON DELETE CASCADE,
source_ref TEXT NOT NULL,
-- Expensive: requires reading the whole file. Computed only when
-- something needs it (import dedup, reconnect-by-hash), never in a scan.
content_hash TEXT,
format TEXT,
w INTEGER,
h INTEGER,
-- UTC seconds. NULL until EXIF is read, or if the file carries none.
captured_at INTEGER,
-- Minutes east of UTC. A photograph's timestamp is local to where it was
-- taken; storing UTC alone makes a Tokyo shoot span two days in Paris.
captured_offset INTEGER,
camera TEXT,
lens TEXT,
iso INTEGER,
aperture REAL,
shutter REAL,
availability INTEGER NOT NULL DEFAULT 0,
file_size INTEGER,
file_mtime INTEGER,
-- 0 = nothing, 1 = stat-only, 2 = full EXIF. The grid is usable at 1.
metadata_state INTEGER NOT NULL DEFAULT 0,
sidecar_mtime INTEGER,
added_at INTEGER NOT NULL,
UNIQUE(root_id, source_ref)
);
CREATE INDEX images_captured ON images(captured_at);
CREATE INDEX images_folder ON images(folder_id);
-- Partial: content_hash is NULL for most rows most of the time, and the
-- non-NULL subset is exactly what reconnect and dedup query.
CREATE INDEX images_hash ON images(content_hash) WHERE content_hash IS NOT NULL;
-- Versions ----------------------------------------------------------------
CREATE TABLE versions (
id INTEGER PRIMARY KEY,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
is_default INTEGER NOT NULL DEFAULT 0,
graph_hash TEXT,
rating INTEGER NOT NULL DEFAULT 0,
label INTEGER,
flag INTEGER NOT NULL DEFAULT 0
);
CREATE INDEX versions_image ON versions(image_id);
CREATE TABLE keywords (
version_id INTEGER NOT NULL REFERENCES versions(id) ON DELETE CASCADE,
keyword TEXT NOT NULL,
PRIMARY KEY(version_id, keyword)
);
CREATE INDEX keywords_term ON keywords(keyword);
-- Remote mapping ----------------------------------------------------------
CREATE TABLE remote (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
-- oc:fileid — stable across server-side rename and move, so a move is not
-- a re-download of 80 MB.
file_id INTEGER NOT NULL,
etag TEXT,
sync_state INTEGER NOT NULL DEFAULT 0,
remote_path TEXT
);
CREATE UNIQUE INDEX remote_file ON remote(file_id);
-- Collections -------------------------------------------------------------
CREATE TABLE collections (
id INTEGER PRIMARY KEY,
-- Device-independent identity. The integer id is local and collides
-- across devices; the UUID is what a cross-device merge keys on.
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
parent_id INTEGER REFERENCES collections(id) ON DELETE CASCADE,
kind INTEGER NOT NULL, -- 0 = manual, 1 = smart
selector_json TEXT, -- smart only
created INTEGER NOT NULL,
-- Monotonic per collection, bumped on every local edit. Merge compares
-- these rather than file mtimes, so a clock-skewed device cannot silently
-- win.
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL,
-- Tombstone. A deleted collection must outlive its deletion, or a merge
-- with a device that still has it would resurrect it.
deleted INTEGER NOT NULL DEFAULT 0
);
CREATE TABLE collection_members (
collection_id INTEGER NOT NULL REFERENCES collections(id) ON DELETE CASCADE,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
position INTEGER, -- manual ordering; NULL = by capture time
added INTEGER NOT NULL,
PRIMARY KEY(collection_id, image_id)
);
CREATE INDEX members_image ON collection_members(image_id);
-- Cache -------------------------------------------------------------------
CREATE TABLE cache (
id INTEGER PRIMARY KEY,
version_id INTEGER REFERENCES versions(id) ON DELETE CASCADE,
image_id INTEGER REFERENCES images(id) ON DELETE CASCADE,
kind INTEGER NOT NULL, -- thumbnail | proxy | original
resolution INTEGER,
graph_hash TEXT,
path TEXT NOT NULL,
bytes INTEGER NOT NULL,
last_used INTEGER NOT NULL
);
CREATE INDEX cache_lru ON cache(last_used);
CREATE TABLE cache_rules (
id INTEGER PRIMARY KEY,
selector_json TEXT NOT NULL,
tier INTEGER NOT NULL,
priority INTEGER NOT NULL DEFAULT 0,
enabled INTEGER NOT NULL DEFAULT 1
);
CREATE TABLE image_cache (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
tier_actual INTEGER NOT NULL DEFAULT 0,
-- Materialised rather than recomputed, so the grid can draw availability
-- badges without evaluating every rule for every visible cell.
tier_desired INTEGER NOT NULL DEFAULT 0,
bytes INTEGER NOT NULL DEFAULT 0,
last_used INTEGER,
pinned_by_rule INTEGER REFERENCES cache_rules(id) ON DELETE SET NULL
);
-- Jobs --------------------------------------------------------------------
CREATE TABLE jobs (
id INTEGER PRIMARY KEY,
kind INTEGER NOT NULL,
subject_id INTEGER,
priority INTEGER NOT NULL DEFAULT 0,
state INTEGER NOT NULL DEFAULT 0, -- 0=pending 1=running 2=failed
attempts INTEGER NOT NULL DEFAULT 0,
not_before INTEGER NOT NULL DEFAULT 0,
payload TEXT,
last_error TEXT,
-- Coalescing. Enqueueing the same work twice updates one row rather than
-- queueing it twice, which is what makes "enqueue on any change" safe to
-- call liberally.
UNIQUE(kind, subject_id)
);
CREATE INDEX jobs_ready ON jobs(state, priority DESC, not_before);
"#;
#[cfg(test)]
mod tests {
#[test]
fn a_writer_waits_for_its_turn_rather_than_losing_its_work() {
// The failure this exists for: a face sweep that had already paid for
// the detection and the embedding threw the result away on
// "database is locked" and moved on. WAL does not help here — it makes
// one writer and many readers free, and this is two writers.
let dir = std::env::temp_dir().join(format!(
"dr-busy-{}-{:?}",
std::process::id(),
std::thread::current().id()
));
let _ = std::fs::remove_dir_all(&dir);
std::fs::create_dir_all(&dir).unwrap();
let path = dir.join("catalog.sqlite");
let held = rusqlite::Connection::open(&path).unwrap();
configure(&held).unwrap();
migrate(&held).unwrap();
let other = rusqlite::Connection::open(&path).unwrap();
configure(&other).unwrap();
// Every connection carries the timeout, which is what makes the wait
// below a wait rather than an immediate error.
let timeout: i64 = other
.query_row("PRAGMA busy_timeout", [], |r| r.get(0))
.unwrap();
assert_eq!(timeout, BUSY_TIMEOUT.as_millis() as i64);
// A writer holds the database; the other one must still get its turn
// once the first commits, rather than failing at the moment it asks.
let writing = held.unchecked_transaction().unwrap();
held.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
let handle = std::thread::spawn(move || {
other.execute(
"INSERT INTO roots(id, kind, label) VALUES (2, 'local', 'two')",
[],
)
});
std::thread::sleep(std::time::Duration::from_millis(150));
writing.commit().unwrap();
assert!(
handle.join().unwrap().is_ok(),
"the second writer waited and then wrote, rather than erroring"
);
let _ = std::fs::remove_dir_all(&dir);
}
use super::*;
fn mem() -> Connection {
let c = Connection::open_in_memory().unwrap();
configure(&c).unwrap();
c
}
/// How many rows a named backfill touched, ignoring the others.
///
/// Asserting on the whole vector would couple every test to which other
/// backfills happen to exist.
fn backfilled(c: &Connection, what: &str) -> usize {
backfill(c)
.unwrap()
.into_iter()
.find(|(name, _)| *name == what)
.map(|(_, n)| n)
.unwrap_or(0)
}
/// Insert an image and return its id.
fn image(c: &Connection, folder: Option<i64>, name: &str, format: &str) -> i64 {
c.execute(
"INSERT INTO images(root_id, folder_id, source_ref, format, added_at)
VALUES (1, ?1, ?2, ?3, 0)",
rusqlite::params![folder, name, format],
)
.unwrap();
c.last_insert_rowid()
}
fn with_root() -> Connection {
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO folders(id, root_id, path) VALUES (1, 1, 'a'), (2, 1, 'b')",
[],
)
.unwrap();
c
}
#[test]
fn a_jpeg_beside_its_raw_is_shadowed() {
// The camera's own rendering of a frame, not a second photograph.
let c = with_root();
let raw = image(&c, Some(1), "a/IMG_1234.CR2", "cr2");
let jpeg = image(&c, Some(1), "a/IMG_1234.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
let got: Option<i64> = c
.query_row(
"SELECT shadowed_by FROM images WHERE id = ?1",
[jpeg],
|r| r.get(0),
)
.unwrap();
assert_eq!(got, Some(raw));
}
#[test]
fn extension_case_does_not_matter() {
let c = with_root();
image(&c, Some(1), "a/IMG_1.cr2", "cr2");
image(&c, Some(1), "a/img_1.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
}
#[test]
fn a_standalone_jpeg_is_untouched() {
// Scanned film has no RAW sibling and must stay visible — 2,656 of
// them in the reference library.
let c = with_root();
image(&c, Some(1), "a/SCAN_0001.jpg", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 0);
}
#[test]
fn a_jpeg_in_a_different_folder_is_not_shadowed() {
// Camera filenames wrap at IMG_9999, so the same stem recurs across
// shoots (FR-CAT-11). Only a same-folder pair is safe to collapse.
let c = with_root();
image(&c, Some(1), "a/IMG_1234.CR2", "cr2");
image(&c, Some(2), "b/IMG_1234.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 0);
}
#[test]
fn a_pair_is_found_among_other_folders_and_unfiled_images() {
// The RAWs are read only from the folders an unpaired JPEG is in,
// plus the unfiled ones when an unfiled JPEG is waiting: each JPEG
// must still find its own sibling, and only its own.
let c = with_root();
let raw_a = image(&c, Some(1), "a/IMG_7.CR2", "cr2");
image(&c, Some(2), "b/IMG_7.CR2", "cr2");
image(&c, Some(2), "b/IMG_8.CR2", "cr2");
let raw_unfiled = image(&c, None, "IMG_9.DNG", "dng");
let jpeg_a = image(&c, Some(1), "a/IMG_7.JPG", "jpg");
let jpeg_unfiled = image(&c, None, "IMG_9.jpg", "jpg");
image(&c, Some(1), "a/IMG_9.jpg", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 2);
let of = |id: i64| -> Option<i64> {
c.query_row("SELECT shadowed_by FROM images WHERE id = ?1", [id], |r| {
r.get(0)
})
.unwrap()
};
assert_eq!(of(jpeg_a), Some(raw_a));
assert_eq!(of(jpeg_unfiled), Some(raw_unfiled));
assert_eq!(
backfilled(&c, "shadowed_by"),
0,
"settled on the second pass"
);
}
#[test]
fn a_raw_is_never_shadowed_by_a_jpeg() {
// The relationship is one-way: the RAW is the photograph.
let c = with_root();
let raw = image(&c, Some(1), "a/IMG_1.CR2", "cr2");
image(&c, Some(1), "a/IMG_1.JPG", "jpg");
backfill(&c).unwrap();
let got: Option<i64> = c
.query_row("SELECT shadowed_by FROM images WHERE id = ?1", [raw], |r| {
r.get(0)
})
.unwrap();
assert_eq!(got, None);
}
#[test]
fn backfill_is_idempotent() {
// It runs on every open, so a second pass must find nothing to do.
let c = with_root();
image(&c, Some(1), "a/IMG_1.CR2", "cr2");
image(&c, Some(1), "a/IMG_1.JPG", "jpg");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
assert_eq!(backfilled(&c, "shadowed_by"), 0, "second pass is a no-op");
}
#[test]
fn a_v1_catalog_gains_the_column_and_is_backfilled() {
// The migration case that motivated this: rows already present when a
// column is added are silently partial until something backfills them.
let c = mem();
c.execute_batch(V1).unwrap();
c.pragma_update(None, "user_version", 1).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, format, added_at)
VALUES (1, 'IMG_9.CR2', 'cr2', 0), (1, 'IMG_9.JPG', 'jpg', 0)",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 1, "migrated from v1");
assert_eq!(backfilled(&c, "shadowed_by"), 1);
}
#[test]
fn a_v4_catalog_gains_the_pinning_columns() {
// TRACES: FR-NC-6a
// An existing library must not have to be rescanned to gain offline
// pinning. The rows are already there; only the columns are new.
let c = mem();
c.execute_batch(V1).unwrap();
c.execute_batch(V2).unwrap();
c.execute_batch(V3).unwrap();
c.execute_batch(V4).unwrap();
c.pragma_update(None, "user_version", 4).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (7, 1, 'IMG_7.CR2', 0)",
[],
)
.unwrap();
// A cache row written before pinning existed.
c.execute(
"INSERT INTO image_cache(image_id, tier_actual, bytes) VALUES (7, 2, 100)",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 4, "migrated from v4");
// The pre-existing row survives, and defaults to unpinned — the safe
// direction, since claiming a pin nobody made would exempt it from
// eviction for ever.
let (pinned, bytes): (i64, i64) = c
.query_row(
"SELECT pinned, bytes FROM image_cache WHERE image_id = 7",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(pinned, 0);
assert_eq!(bytes, 100, "the existing row is untouched");
}
#[test]
fn a_v5_catalog_keeps_its_keywords_and_gains_their_identities() {
// TRACES: FR-CAT-5
// The migration case that matters here: a library keyworded by an
// import or an older build already has assignment rows, and they must
// survive into the vocabulary rather than being left searchable but
// invisible.
let c = mem();
for step in [V1, V2, V3, V4, V5] {
c.execute_batch(step).unwrap();
}
c.pragma_update(None, "user_version", 5).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (7, 1, 'IMG_7.CR3', 0)",
[],
)
.unwrap();
c.execute(
"INSERT INTO versions(id, image_id, uuid, name, is_default)
VALUES (1, 7, 'v-7', 'Default', 1)",
[],
)
.unwrap();
c.execute(
"INSERT INTO keywords(version_id, keyword) VALUES (1, 'puffin')",
[],
)
.unwrap();
assert_eq!(migrate(&c).unwrap(), 5, "migrated from v5");
assert_eq!(backfilled(&c, "keyword_terms"), 1);
let name: String = c
.query_row("SELECT name FROM keyword_terms", [], |r| r.get(0))
.unwrap();
assert_eq!(name, "puffin");
// The assignment is untouched — it is the durable fact, and the term
// row is only its identity.
let n: i64 = c
.query_row("SELECT count(*) FROM keywords", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 1);
// It runs on every open, so a second pass must find nothing to do.
assert_eq!(backfilled(&c, "keyword_terms"), 0);
}
#[test]
fn two_devices_may_both_hold_a_term_of_the_same_name() {
// Deliberately not a unique index. Two devices each typing "Iceland"
// is the ordinary case, and a constraint would abort the merge
// transaction at exactly the moment they first sync.
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO keyword_terms(uuid, name, created, revision, modified)
VALUES ('a', 'Iceland', 0, 1, 1), ('b', 'Iceland', 0, 1, 1)",
[],
)
.unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM keyword_terms", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 2);
}
#[test]
fn stems_ignore_directories_containing_dots() {
assert_eq!(stem_of("2026.08/IMG_1.CR2"), "IMG_1");
assert_eq!(stem_of("IMG_1.CR2"), "IMG_1");
assert_eq!(stem_of("noextension"), "noextension");
// A dotfile is all stem, not an empty name with an extension.
assert_eq!(stem_of(".hidden"), ".hidden");
}
#[test]
fn migrate_creates_schema_at_current_version() {
let c = mem();
assert_eq!(migrate(&c).unwrap(), 0);
let v: i64 = c
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, SCHEMA_VERSION);
}
#[test]
fn migrate_is_idempotent() {
let c = mem();
migrate(&c).unwrap();
// Re-running must not error or duplicate anything — NFR-R5 requires
// idempotency on retry, since a migration can be interrupted.
assert_eq!(migrate(&c).unwrap(), SCHEMA_VERSION);
}
/// V16 adds its columns guarded, so a catalog whose version was rewound
/// after the columns landed — the rollback NFR-R5 contemplates — migrates
/// again rather than failing on "duplicate column".
#[test]
fn the_eye_columns_survive_a_rewound_version() {
let c = mem();
migrate(&c).unwrap();
for column in EYE_COLUMNS {
let present: bool = c
.prepare("SELECT 1 FROM pragma_table_info('faces') WHERE name = ?1")
.unwrap()
.exists([column])
.unwrap();
assert!(present, "{column} missing after migration");
}
c.pragma_update(None, "user_version", 15).unwrap();
assert_eq!(migrate(&c).unwrap(), 15);
let indexed: bool = c
.prepare("SELECT 1 FROM sqlite_master WHERE type = 'index' AND name = 'faces_eyes'")
.unwrap()
.exists([])
.unwrap();
assert!(indexed, "V17's covering index is there");
let v: i64 = c
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, SCHEMA_VERSION);
}
#[test]
fn refuses_a_catalog_from_a_newer_build() {
let c = mem();
migrate(&c).unwrap();
c.pragma_update(None, "user_version", SCHEMA_VERSION + 1)
.unwrap();
// Opening it read-write would corrupt data this build cannot
// represent. Refusing is the specified behaviour (NFR-R5).
assert!(matches!(
migrate(&c),
Err(CatalogError::SchemaTooNew { .. })
));
}
#[test]
fn foreign_keys_cascade_from_root_to_image() {
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)",
[],
)
.unwrap();
c.execute("DELETE FROM roots WHERE id = 1", []).unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 0, "images must not outlive their root");
}
#[test]
fn v12_forgets_runs_made_on_a_proxy_too_small_to_see_a_face() {
let c = mem();
// Migrate to 11, then seed the state V12 exists to repair: markers
// written at the 1024 store tier beside ones written on a real
// preview.
c.pragma_update(None, "user_version", 0).unwrap();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0),(4,1,'d',0)",
[],
)
.unwrap();
for (image, edge) in [(1, 896), (2, 1024), (3, 1025), (4, 2560)] {
c.execute(
"INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge)
VALUES (?1, 'm', 0, 0, ?2)",
rusqlite::params![image, edge],
)
.unwrap();
}
c.pragma_update(None, "user_version", 11).unwrap();
migrate(&c).unwrap();
let kept: Vec<i64> = c
.prepare("SELECT image_id FROM face_index ORDER BY image_id")
.unwrap()
.query_map([], |r| r.get(0))
.unwrap()
.map(Result::unwrap)
.collect();
// 1024 goes: it is exactly ThumbSize::Large, the tier that produced
// the bad runs. 1025 stays, or the floor and the repair disagree
// about the same boundary.
assert_eq!(kept, vec![3, 4]);
}
#[test]
fn v14_forgets_runs_that_found_faces_but_never_measured_them() {
let c = mem();
c.pragma_update(None, "user_version", 0).unwrap();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0)",
[],
)
.unwrap();
// Image 1 was examined and holds a face; 2 was examined and found
// empty; 3 holds a face found by a different model.
for (image, model) in [(1, "m"), (2, "m"), (3, "m")] {
c.execute(
"INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge)
VALUES (?1, ?2, 0, 0, 2560)",
rusqlite::params![image, model],
)
.unwrap();
}
for (image, model) in [(1, "m"), (3, "other")] {
c.execute(
"INSERT INTO faces
(image_id, x, y, w, h, landmarks, detector_confidence, embedding,
crop_px, model_id, detected_at)
VALUES (?1, 0.1, 0.1, 0.2, 0.2, X'00', 0.9, X'00', 180.0, ?2, 0)",
rusqlite::params![image, model],
)
.unwrap();
}
c.pragma_update(None, "user_version", 13).unwrap();
migrate(&c).unwrap();
let kept: Vec<i64> = c
.prepare("SELECT image_id FROM face_index ORDER BY image_id")
.unwrap()
.query_map([], |r| r.get(0))
.unwrap()
.map(Result::unwrap)
.collect();
// 1 goes: it has a face with no quality. 2 stays: nothing on it to
// measure. 3 stays: its face belongs to a run this marker does not
// describe.
assert_eq!(kept, vec![2, 3]);
// And the faces themselves are untouched.
let faces: i64 = c
.query_row("SELECT count(*) FROM faces", [], |r| r.get(0))
.unwrap();
assert_eq!(faces, 2);
}
#[test]
fn v20_renames_markers_to_the_detector_that_found_the_faces() {
let c = mem();
c.pragma_update(None, "user_version", 0).unwrap();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1,1,'a',0),(2,1,'b',0),(3,1,'c',0),(4,1,'d',0),(5,1,'e',0)",
[],
)
.unwrap();
// 1: the desktop's case -- old faces, re-marked as thorough.
// 2: the tablet's case -- adopted thorough faces, re-marked int8,
// and the right marker still beside it (refreshed, so it is
// exported again over the empty entry).
// 3: right already. 4: examined and empty. 5: V14's state, faces
// and no marker.
for (image, model) in [
(1, "scrfd_10g+w600k_mbf"),
(2, "scrfd_10g_i8+w600k_mbf"),
(2, "scrfd_10g+w600k_mbf"),
(3, "scrfd_10g+w600k_mbf"),
(4, "scrfd_10g+w600k_mbf"),
] {
c.execute(
"INSERT INTO face_index(image_id, model_id, indexed_at, faces_found, source_edge)
VALUES (?1, ?2, 100, 0, 6000)",
rusqlite::params![image, model],
)
.unwrap();
}
for (image, model) in [
(1, "w600k_mbf"),
(1, "w600k_mbf"),
(2, "scrfd_10g+w600k_mbf"),
(3, "scrfd_10g+w600k_mbf"),
(5, "w600k_mbf"),
] {
c.execute(
"INSERT INTO faces
(image_id, x, y, w, h, landmarks, detector_confidence, embedding,
crop_px, model_id, detected_at)
VALUES (?1, 0.1, 0.1, 0.2, 0.2, X'00', 0.9, X'00', 180.0, ?2, 0)",
rusqlite::params![image, model],
)
.unwrap();
}
c.pragma_update(None, "user_version", 19).unwrap();
migrate(&c).unwrap();
let markers: Vec<(i64, String, i64, bool)> = c
.prepare(
"SELECT image_id, model_id, faces_found, indexed_at > 100
FROM face_index ORDER BY image_id, model_id",
)
.unwrap()
.query_map([], |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)))
.unwrap()
.map(Result::unwrap)
.collect();
assert_eq!(
markers,
vec![
(1, "w600k_mbf".to_string(), 2, true),
(2, "scrfd_10g+w600k_mbf".to_string(), 0, true),
(3, "scrfd_10g+w600k_mbf".to_string(), 0, false),
(4, "scrfd_10g+w600k_mbf".to_string(), 0, false),
]
);
// Re-enterable: nothing left to rename.
c.pragma_update(None, "user_version", 19).unwrap();
migrate(&c).unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM face_index", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 4);
}
#[test]
fn job_uniqueness_coalesces_rather_than_duplicating() {
let c = mem();
migrate(&c).unwrap();
for _ in 0..5 {
c.execute(
"INSERT INTO jobs(kind, subject_id, priority) VALUES (1, 42, 0)
ON CONFLICT(kind, subject_id)
DO UPDATE SET priority = max(priority, excluded.priority)",
[],
)
.unwrap();
}
let n: i64 = c
.query_row("SELECT count(*) FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 1, "five enqueues of the same work is one job");
}
}