Add folder scan with format selection; validate A3 on a real library

Library setup as the user described it: pick a folder, choose which RAW
types to look for, scan recursively.

  dr-types::FormatFilter  the tick-box selection, seeing through VFS
                          placeholder suffixes so a dehydrated CR2 still
                          matches as a CR2
  dr-sync::scan           recursive walk, Depth:1 per directory, pruning
                          unchanged subtrees where the backend propagates
                          directory ETags

Verified against nextcloud.tourolle.paris (34.0.2) on a real library:

  browse root      32 entries, 98ms
  scan PhotosRaw   17,185 RAW files in 334 directories, 34.1s
                   (7,836 CR2 + 9,349 DNG)
  range read       262KB of a 21.5MB DNG in 119ms — 1.22% of the file,
                   and enough to read "Canon EOS 6D | ISO 100"

That last line is assumption A3 validated on real data. Cataloguing this
library by whole-file fetch would move roughly 370GB; the range path
moves a few MB.

Pruning is capability-gated rather than assumed: with per-entry ETags a
probe costs a request and proves nothing about children, so it is skipped
entirely. A test asserts zero probes in that case.

Still unresolved: /core/preview returns 400 for every parameter
combination tried, including on a JPEG the server reports as having a
preview. Not a request-shape bug — it fails identically bare. Recorded
rather than worked around; ARCH §6.7 already treats server previews as
opportunistic, so nothing depends on it.
This commit is contained in:
2026-08-09 12:22:31 +02:00
parent fbadf9afc8
commit c8bb08e661
29 changed files with 7193 additions and 234 deletions
+353
View File
@@ -0,0 +1,353 @@
//! TRACES: FR-CAT-2 | NFR-R5
//! Schema definition and forward-only migrations.
//!
//! The catalog is an *index*, not a source of truth (ARCH §6.12) — it is
//! deletable and rebuildable from sources plus sidecars. That is what makes
//! migration failure survivable, and why the recovery path is the normal
//! mechanism rather than a last resort.
//!
//! Migrations are forward-only, transactional, and idempotent on retry
//! (NFR-R5). The app refuses to open a catalog newer than it understands
//! rather than corrupting it.
use rusqlite::Connection;
use crate::error::CatalogError;
/// Schema version this build writes and understands.
pub const SCHEMA_VERSION: i64 = 1;
/// Apply migrations up to [`SCHEMA_VERSION`].
///
/// Returns the version migrated from, so callers can log or back up before a
/// real migration (NFR-R2 requires a backup before schema change).
pub fn migrate(conn: &Connection) -> Result<i64, CatalogError> {
let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
if from > SCHEMA_VERSION {
return Err(CatalogError::SchemaTooNew {
found: from,
supported: SCHEMA_VERSION,
});
}
if from == SCHEMA_VERSION {
return Ok(from);
}
// Each step runs in its own transaction so a failure leaves the catalog
// at a coherent version rather than half-migrated.
if from < 1 {
let tx = conn.unchecked_transaction()?;
tx.execute_batch(V1)?;
tx.pragma_update(None, "user_version", 1)?;
tx.commit()?;
}
Ok(from)
}
/// Connection setup applied on every open, migration or not.
///
/// WAL is required by NFR-R1: it survives power loss without corruption, and
/// it lets a background job write while the grid reads.
pub fn configure(conn: &Connection) -> Result<(), CatalogError> {
conn.pragma_update(None, "journal_mode", "WAL")?;
// NORMAL rather than FULL: with WAL this is durable across process death
// (which is what FR-PLAT-AND-3 cares about) and only risks the last
// transaction on power loss. The catalog is rebuildable; the sidecars are
// not, and they are written separately with their own fsync discipline.
conn.pragma_update(None, "synchronous", "NORMAL")?;
conn.pragma_update(None, "foreign_keys", true)?;
// A scan touching thousands of rows is transient; let SQLite spill to
// memory rather than materialising temp b-trees on disk.
conn.pragma_update(None, "temp_store", "MEMORY")?;
Ok(())
}
/// The v1 schema rewritten to target an attached database.
///
/// Needed because a downloaded remote catalog is `ATTACH`ed under its own
/// schema name before merging, and tests build one from scratch. SQLite has no
/// "create these tables over there" form, so the names are rewritten.
///
/// The rewrite is textual and therefore only as good as the naming discipline
/// in [`V1`]: every `CREATE TABLE`/`CREATE INDEX` must name its object
/// unqualified, which they do.
pub fn v1_for_attached(schema_name: &str) -> String {
V1.replace("CREATE TABLE ", &format!("CREATE TABLE {schema_name}."))
.replace("CREATE INDEX ", &format!("CREATE INDEX {schema_name}."))
.replace(
"CREATE UNIQUE INDEX ",
&format!("CREATE UNIQUE INDEX {schema_name}."),
)
// REFERENCES within an attached schema resolve to that schema already,
// so foreign keys need no rewriting — but the ON clause of an index
// does, and `CREATE INDEX x.name ON table` is the correct form.
}
const V1: &str = r#"
-- Roots -------------------------------------------------------------------
CREATE TABLE roots (
id INTEGER PRIMARY KEY,
kind TEXT NOT NULL, -- 'local' | 'saf' | 'remote'
grant_blob BLOB, -- SAF persisted permission; NULL on Linux
label TEXT NOT NULL,
last_seen INTEGER,
-- Bumped once per completed scan. Folders record the generation they were
-- reached in; anything older was not reached and no longer exists.
scan_generation INTEGER NOT NULL DEFAULT 0
);
-- Folders: the unit of change detection, local and remote alike -----------
CREATE TABLE folders (
id INTEGER PRIMARY KEY,
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
parent_id INTEGER REFERENCES folders(id) ON DELETE CASCADE,
path TEXT NOT NULL,
-- Remote: the propagating ETag that makes a no-op sync one request.
etag TEXT,
-- Local: directory mtime plus direct-entry count. mtime alone misses a
-- paired create+delete inside one timestamp tick; the count narrows that.
mtime INTEGER,
entry_count INTEGER,
scanned_generation INTEGER NOT NULL DEFAULT 0,
UNIQUE(root_id, path)
);
CREATE INDEX folders_parent ON folders(parent_id);
-- Images ------------------------------------------------------------------
CREATE TABLE images (
id INTEGER PRIMARY KEY,
root_id INTEGER NOT NULL REFERENCES roots(id) ON DELETE CASCADE,
folder_id INTEGER REFERENCES folders(id) ON DELETE CASCADE,
source_ref TEXT NOT NULL,
-- Expensive: requires reading the whole file. Computed only when
-- something needs it (import dedup, reconnect-by-hash), never in a scan.
content_hash TEXT,
format TEXT,
w INTEGER,
h INTEGER,
-- UTC seconds. NULL until EXIF is read, or if the file carries none.
captured_at INTEGER,
-- Minutes east of UTC. A photograph's timestamp is local to where it was
-- taken; storing UTC alone makes a Tokyo shoot span two days in Paris.
captured_offset INTEGER,
camera TEXT,
lens TEXT,
iso INTEGER,
aperture REAL,
shutter REAL,
availability INTEGER NOT NULL DEFAULT 0,
file_size INTEGER,
file_mtime INTEGER,
-- 0 = nothing, 1 = stat-only, 2 = full EXIF. The grid is usable at 1.
metadata_state INTEGER NOT NULL DEFAULT 0,
sidecar_mtime INTEGER,
added_at INTEGER NOT NULL,
UNIQUE(root_id, source_ref)
);
CREATE INDEX images_captured ON images(captured_at);
CREATE INDEX images_folder ON images(folder_id);
-- Partial: content_hash is NULL for most rows most of the time, and the
-- non-NULL subset is exactly what reconnect and dedup query.
CREATE INDEX images_hash ON images(content_hash) WHERE content_hash IS NOT NULL;
-- Versions ----------------------------------------------------------------
CREATE TABLE versions (
id INTEGER PRIMARY KEY,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
is_default INTEGER NOT NULL DEFAULT 0,
graph_hash TEXT,
rating INTEGER NOT NULL DEFAULT 0,
label INTEGER,
flag INTEGER NOT NULL DEFAULT 0
);
CREATE INDEX versions_image ON versions(image_id);
CREATE TABLE keywords (
version_id INTEGER NOT NULL REFERENCES versions(id) ON DELETE CASCADE,
keyword TEXT NOT NULL,
PRIMARY KEY(version_id, keyword)
);
CREATE INDEX keywords_term ON keywords(keyword);
-- Remote mapping ----------------------------------------------------------
CREATE TABLE remote (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
-- oc:fileid — stable across server-side rename and move, so a move is not
-- a re-download of 80 MB.
file_id INTEGER NOT NULL,
etag TEXT,
sync_state INTEGER NOT NULL DEFAULT 0,
remote_path TEXT
);
CREATE UNIQUE INDEX remote_file ON remote(file_id);
-- Collections -------------------------------------------------------------
CREATE TABLE collections (
id INTEGER PRIMARY KEY,
-- Device-independent identity. The integer id is local and collides
-- across devices; the UUID is what a cross-device merge keys on.
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
parent_id INTEGER REFERENCES collections(id) ON DELETE CASCADE,
kind INTEGER NOT NULL, -- 0 = manual, 1 = smart
selector_json TEXT, -- smart only
created INTEGER NOT NULL,
-- Monotonic per collection, bumped on every local edit. Merge compares
-- these rather than file mtimes, so a clock-skewed device cannot silently
-- win.
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL,
-- Tombstone. A deleted collection must outlive its deletion, or a merge
-- with a device that still has it would resurrect it.
deleted INTEGER NOT NULL DEFAULT 0
);
CREATE TABLE collection_members (
collection_id INTEGER NOT NULL REFERENCES collections(id) ON DELETE CASCADE,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
position INTEGER, -- manual ordering; NULL = by capture time
added INTEGER NOT NULL,
PRIMARY KEY(collection_id, image_id)
);
CREATE INDEX members_image ON collection_members(image_id);
-- Cache -------------------------------------------------------------------
CREATE TABLE cache (
id INTEGER PRIMARY KEY,
version_id INTEGER REFERENCES versions(id) ON DELETE CASCADE,
image_id INTEGER REFERENCES images(id) ON DELETE CASCADE,
kind INTEGER NOT NULL, -- thumbnail | proxy | original
resolution INTEGER,
graph_hash TEXT,
path TEXT NOT NULL,
bytes INTEGER NOT NULL,
last_used INTEGER NOT NULL
);
CREATE INDEX cache_lru ON cache(last_used);
CREATE TABLE cache_rules (
id INTEGER PRIMARY KEY,
selector_json TEXT NOT NULL,
tier INTEGER NOT NULL,
priority INTEGER NOT NULL DEFAULT 0,
enabled INTEGER NOT NULL DEFAULT 1
);
CREATE TABLE image_cache (
image_id INTEGER PRIMARY KEY REFERENCES images(id) ON DELETE CASCADE,
tier_actual INTEGER NOT NULL DEFAULT 0,
-- Materialised rather than recomputed, so the grid can draw availability
-- badges without evaluating every rule for every visible cell.
tier_desired INTEGER NOT NULL DEFAULT 0,
bytes INTEGER NOT NULL DEFAULT 0,
last_used INTEGER,
pinned_by_rule INTEGER REFERENCES cache_rules(id) ON DELETE SET NULL
);
-- Jobs --------------------------------------------------------------------
CREATE TABLE jobs (
id INTEGER PRIMARY KEY,
kind INTEGER NOT NULL,
subject_id INTEGER,
priority INTEGER NOT NULL DEFAULT 0,
state INTEGER NOT NULL DEFAULT 0, -- 0=pending 1=running 2=failed
attempts INTEGER NOT NULL DEFAULT 0,
not_before INTEGER NOT NULL DEFAULT 0,
payload TEXT,
last_error TEXT,
-- Coalescing. Enqueueing the same work twice updates one row rather than
-- queueing it twice, which is what makes "enqueue on any change" safe to
-- call liberally.
UNIQUE(kind, subject_id)
);
CREATE INDEX jobs_ready ON jobs(state, priority DESC, not_before);
"#;
#[cfg(test)]
mod tests {
use super::*;
fn mem() -> Connection {
let c = Connection::open_in_memory().unwrap();
configure(&c).unwrap();
c
}
#[test]
fn migrate_creates_schema_at_current_version() {
let c = mem();
assert_eq!(migrate(&c).unwrap(), 0);
let v: i64 = c
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, SCHEMA_VERSION);
}
#[test]
fn migrate_is_idempotent() {
let c = mem();
migrate(&c).unwrap();
// Re-running must not error or duplicate anything — NFR-R5 requires
// idempotency on retry, since a migration can be interrupted.
assert_eq!(migrate(&c).unwrap(), SCHEMA_VERSION);
}
#[test]
fn refuses_a_catalog_from_a_newer_build() {
let c = mem();
migrate(&c).unwrap();
c.pragma_update(None, "user_version", SCHEMA_VERSION + 1)
.unwrap();
// Opening it read-write would corrupt data this build cannot
// represent. Refusing is the specified behaviour (NFR-R5).
assert!(matches!(
migrate(&c),
Err(CatalogError::SchemaTooNew { .. })
));
}
#[test]
fn foreign_keys_cascade_from_root_to_image() {
let c = mem();
migrate(&c).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'test')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)",
[],
)
.unwrap();
c.execute("DELETE FROM roots WHERE id = 1", []).unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 0, "images must not outlive their root");
}
#[test]
fn job_uniqueness_coalesces_rather_than_duplicating() {
let c = mem();
migrate(&c).unwrap();
for _ in 0..5 {
c.execute(
"INSERT INTO jobs(kind, subject_id, priority) VALUES (1, 42, 0)
ON CONFLICT(kind, subject_id)
DO UPDATE SET priority = max(priority, excluded.priority)",
[],
)
.unwrap();
}
let n: i64 = c
.query_row("SELECT count(*) FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 1, "five enqueues of the same work is one job");
}
}