Files
DarkRoom/core/dr-catalog/src/lib.rs
T
dtourolle d5b1f6bff5 Add collections, ratings, and soft delete to the catalog
Three features over a shared schema migration.

Collections: a tree of manual collections plus smart collections whose
membership *is* their stored selector. Dropping images onto a smart
collection is refused rather than silently discarded, so the UI can say why
the drop did nothing — member rows there would be a second source of truth
that nothing reads.

Ratings: the star and pick/reject axes, kept independent.

Trash: soft delete to a folder, then permanent delete.

Catalog::open now backfills after migrating. A migration adds a column but
cannot know what the value should be for rows that already existed;
backfilling on open is what stops those rows being silently partial.

Timeline queries exclude shadowed JPEGs, which would otherwise double every
paired shot in the histogram, and gain a range-bounded variant so zooming in
returns finer buckets rather than the same coarse ones with the ends cropped.

Assisted-by: LLM
2026-08-09 20:42:13 +02:00

437 lines
15 KiB
Rust

//! TRACES: FR-CAT-2 | FR-CAT-4 | FR-CAT-6 | NFR-P1
//! The catalog: a rebuildable index over the library.
//!
//! Not a source of truth. Sidecars next to the images hold the authoritative
//! edit state (ARCH §6.12), and this file is deletable at any time — rebuilt
//! by rescanning sources and reading sidecars. That inversion is deliberate:
//! darktable maintains both a database and sidecars while achieving the
//! reliability of neither.
//!
//! # What lives here
//!
//! - [`schema`] — tables and forward-only migrations
//! - [`scan`] — incremental discovery that prunes unchanged directories
//! - [`query`] — selectors compiled to indexed SQL, windowed for the grid
//! - [`collections`] — the collection tree and membership the UI edits
//! - [`jobs`] — the durable background work queue
//! - [`trash`] — soft delete to a folder, then permanent delete
//! - [`merge`] / [`sync`] — cross-device collection merging
//!
//! # The one thing everything is designed around
//!
//! **Work is proportional to what changed, or to what the user is looking at —
//! never to library size.** A 50k-image library that has not changed costs one
//! metadata probe per folder to verify (§scan), no thumbnails to regenerate
//! (§jobs coalescing), and no rule evaluation per grid cell (materialised
//! `tier_desired`).
use std::path::Path;
use dr_types::{Availability, ImageId};
use rusqlite::Connection;
pub mod collections;
pub mod error;
pub mod jobs;
pub mod merge;
pub mod query;
pub mod rating;
pub mod scan;
pub mod schema;
pub mod sync;
pub mod trash;
pub use collections::{Collection, CollectionKind, TreeRow};
pub use error::CatalogError;
pub use jobs::{Job, JobKind, Priority};
pub use merge::MergeReport;
pub use query::{Query, Sort};
pub use rating::{Judgement, MAX_RATING};
pub use scan::{DirAction, DirState, EntryAction, ScanOutcome};
pub use trash::{TrashedImage, TRASH_DIR};
/// One row of the library grid.
///
/// Exactly what a cell draws and nothing more — no join per cell, and
/// availability reads a materialised column rather than evaluating cache rules
/// (ARCH §9.5).
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct GridRow {
pub id: ImageId,
pub name: String,
pub availability: Availability,
/// UTC seconds. `None` until EXIF has been read.
pub captured_at: Option<i64>,
/// Minutes east of UTC, for rendering the photographer's local time.
pub captured_offset: Option<i32>,
/// 0 = nothing, 1 = stat-only, 2 = full EXIF.
pub metadata_state: u8,
}
/// A count of images in one time bucket, for the timeline scrubber.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct TimeBucket {
/// UTC seconds at the bucket's start.
pub start: i64,
pub count: u32,
}
/// Time bucket size.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Granularity {
Year,
Month,
Day,
Hour,
}
impl Granularity {
/// SQLite `strftime` format that collapses a timestamp to this bucket.
///
/// Applied to **local** time, not UTC: "everything from 3 August" means
/// the photographer's 3 August, which is why `captured_offset` is stored
/// alongside the UTC timestamp.
fn strftime(self) -> &'static str {
match self {
Granularity::Year => "%Y",
Granularity::Month => "%Y-%m",
Granularity::Day => "%Y-%m-%d",
Granularity::Hour => "%Y-%m-%dT%H",
}
}
/// A sensible bucket size for a span of seconds, so the UI need not guess.
pub fn for_span(seconds: i64) -> Self {
const DAY: i64 = 86_400;
match seconds {
s if s > 5 * 365 * DAY => Granularity::Year,
s if s > 90 * DAY => Granularity::Month,
s if s > 2 * DAY => Granularity::Day,
_ => Granularity::Hour,
}
}
}
/// A connection to the catalog.
pub struct Catalog {
conn: Connection,
}
impl Catalog {
/// Open or create a catalog, migrating it forward if needed.
pub fn open(path: &Path) -> Result<Self, CatalogError> {
let conn = Connection::open(path)?;
schema::configure(&conn)?;
let from = schema::migrate(&conn)?;
// A migration adds a column; it cannot know what the value should be
// for rows that already existed. Backfilling on open is what stops
// those rows being silently partial.
for (what, n) in schema::backfill(&conn)? {
log::info!("backfilled {what} for {n} row(s) (schema was v{from})");
}
Ok(Catalog { conn })
}
/// An in-memory catalog, for tests and for a throwaway import preview.
pub fn in_memory() -> Result<Self, CatalogError> {
let conn = Connection::open_in_memory()?;
schema::configure(&conn)?;
schema::migrate(&conn)?;
schema::backfill(&conn)?;
Ok(Catalog { conn })
}
/// Escape hatch for modules that need raw access. Not part of the UI-facing
/// surface.
pub fn connection(&self) -> &Connection {
&self.conn
}
/// How many images match.
///
/// Returned alongside the first window so the grid can size its scrollbar
/// and paint in one round trip.
pub fn count(&self, q: &Query, now: i64) -> Result<usize, CatalogError> {
let c = query::compile(&q.filter, now);
let sql = query::count_sql(&c);
let n: i64 =
self.conn
.query_row(&sql, rusqlite::params_from_iter(c.params.iter()), |r| {
r.get(0)
})?;
Ok(n as usize)
}
/// Fetch one window of results.
///
/// Never returns the whole catalog: FR-CAT-4 requires memory bounded
/// independently of library size.
pub fn window(
&self,
q: &Query,
range: std::ops::Range<usize>,
now: i64,
) -> Result<Vec<GridRow>, CatalogError> {
let c = query::compile(&q.filter, now);
let sql = query::window_sql(q, &c);
let mut params = c.params.clone();
params.push(rusqlite::types::Value::Integer(range.len() as i64));
params.push(rusqlite::types::Value::Integer(range.start as i64));
let mut stmt = self.conn.prepare(&sql)?;
let rows = stmt
.query_map(rusqlite::params_from_iter(params.iter()), |r| {
let source_ref: String = r.get(1)?;
let avail: i64 = r.get(2)?;
Ok(GridRow {
id: ImageId(r.get::<_, i64>(0)? as u64),
name: source_ref
.rsplit(['/', ':'])
.next()
.unwrap_or(&source_ref)
.to_string(),
availability: decode_availability(avail),
captured_at: r.get(3)?,
captured_offset: r.get::<_, Option<i64>>(4)?.map(|v| v as i32),
metadata_state: r.get::<_, i64>(5)? as u8,
})
})?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// Counts per time bucket, for the timeline scrubber.
///
/// One grouped aggregate over the `images_captured` index — not 50k rows
/// handed to the UI to bucket itself.
pub fn timeline(
&self,
q: &Query,
g: Granularity,
now: i64,
) -> Result<Vec<TimeBucket>, CatalogError> {
let c = query::compile(&q.filter, now);
// Bucketed in local time: captured_offset is minutes east of UTC, and
// NULL falls back to UTC rather than dropping the row.
let sql = format!(
"SELECT min(captured_at) AS start,
count(*) AS n
FROM images
-- A shadowed JPEG is the same frame as its RAW; counting both
-- would double every paired shot in the histogram.
WHERE {} AND captured_at IS NOT NULL AND shadowed_by IS NULL
GROUP BY strftime('{}', captured_at + coalesce(captured_offset, 0) * 60,
'unixepoch')
ORDER BY start ASC",
c.where_sql,
g.strftime()
);
let mut stmt = self.conn.prepare(&sql)?;
let rows = stmt
.query_map(rusqlite::params_from_iter(c.params.iter()), |r| {
Ok(TimeBucket {
start: r.get(0)?,
count: r.get::<_, i64>(1)? as u32,
})
})?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// Counts per time bucket, bounded to a date range.
///
/// What a zoomed timeline needs: [`timeline`](Self::timeline) always spans
/// the whole library, so zooming in would return the same coarse buckets
/// with the ends cropped rather than finer detail over a narrower span.
pub fn timeline_range(
&self,
q: &Query,
g: Granularity,
from: i64,
to: i64,
now: i64,
) -> Result<Vec<TimeBucket>, CatalogError> {
let c = query::compile(&q.filter, now);
let sql = format!(
"SELECT min(captured_at) AS start,
count(*) AS n
FROM images
WHERE {} AND captured_at IS NOT NULL AND shadowed_by IS NULL
AND captured_at >= ?{} AND captured_at <= ?{}
GROUP BY strftime('{}', captured_at + coalesce(captured_offset, 0) * 60,
'unixepoch')
ORDER BY start ASC",
c.where_sql,
c.params.len() + 1,
c.params.len() + 2,
g.strftime()
);
let mut params = c.params.clone();
params.push(rusqlite::types::Value::Integer(from));
params.push(rusqlite::types::Value::Integer(to));
let mut stmt = self.conn.prepare(&sql)?;
let rows = stmt
.query_map(rusqlite::params_from_iter(params.iter()), |r| {
Ok(TimeBucket {
start: r.get(0)?,
count: r.get::<_, i64>(1)? as u32,
})
})?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// Merge a downloaded remote catalog's collections into this one.
///
/// See [`sync`] for why only collections cross over.
pub fn merge_remote_catalog(&self, remote: &Path) -> Result<MergeReport, CatalogError> {
sync::merge_remote(&self.conn, remote)
}
/// Write a consistent snapshot ready to upload.
pub fn snapshot_for_upload(&self, dest: &Path) -> Result<(), CatalogError> {
sync::snapshot_for_upload(&self.conn, dest)
}
}
fn decode_availability(v: i64) -> Availability {
match v {
1 => Availability::Preview,
2 => Availability::Original,
3 => Availability::Offline,
_ => Availability::MetadataOnly,
}
}
#[cfg(test)]
mod tests {
use super::*;
use dr_types::Selector;
fn seeded() -> Catalog {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
// Three images across two days, one with no EXIF read yet.
for (id, name, captured, state) in [
(1i64, "a.CR3", Some(1_000_000i64), 2i64),
(2, "b.CR3", Some(1_100_000), 2),
(3, "c.CR3", None, 1),
] {
c.execute(
"INSERT INTO images(id, root_id, source_ref, captured_at, metadata_state, added_at)
VALUES (?1, 1, ?2, ?3, ?4, 0)",
rusqlite::params![id, name, captured, state],
)
.unwrap();
}
cat
}
#[test]
fn count_and_window_agree() {
let cat = seeded();
let q = Query::default();
assert_eq!(cat.count(&q, 0).unwrap(), 3);
assert_eq!(cat.window(&q, 0..10, 0).unwrap().len(), 3);
}
#[test]
fn window_is_bounded_by_the_requested_range() {
// FR-CAT-4: memory independent of catalog size.
let cat = seeded();
let rows = cat.window(&Query::default(), 0..2, 0).unwrap();
assert_eq!(rows.len(), 2);
}
#[test]
fn paging_covers_every_row_exactly_once() {
let cat = seeded();
let q = Query::default();
let mut seen = Vec::new();
for start in (0..3).step_by(2) {
seen.extend(cat.window(&q, start..start + 2, 0).unwrap());
}
let mut ids: Vec<u64> = seen.iter().map(|r| r.id.0).collect();
ids.sort_unstable();
assert_eq!(ids, vec![1, 2, 3]);
}
#[test]
fn an_image_without_capture_time_sorts_last_not_first() {
// Otherwise a freshly scanned library leads with whatever has not been
// read yet, which looks like corruption to the user.
let cat = seeded();
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
assert_eq!(rows.last().unwrap().id, ImageId(3));
}
#[test]
fn metadata_state_reaches_the_grid() {
// The grid needs it to distinguish "no photos on this date" from
// "EXIF not read yet" (FR-NC-6c's honesty principle).
let cat = seeded();
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
let pending = rows.iter().find(|r| r.id == ImageId(3)).unwrap();
assert_eq!(pending.metadata_state, 1);
}
#[test]
fn a_filter_narrows_the_count() {
let cat = seeded();
let q = Query {
filter: Selector::Text("a.CR3".into()),
..Default::default()
};
assert_eq!(cat.count(&q, 0).unwrap(), 1);
}
#[test]
fn timeline_buckets_and_skips_unread_images() {
let cat = seeded();
let buckets = cat
.timeline(&Query::default(), Granularity::Day, 0)
.unwrap();
// Two images with timestamps, one day apart in UTC; the third has no
// capture time and cannot be placed on a timeline at all.
let total: u32 = buckets.iter().map(|b| b.count).sum();
assert_eq!(total, 2);
}
#[test]
fn timeline_granularity_follows_the_span() {
const DAY: i64 = 86_400;
assert_eq!(Granularity::for_span(10 * 365 * DAY), Granularity::Year);
assert_eq!(Granularity::for_span(120 * DAY), Granularity::Month);
assert_eq!(Granularity::for_span(10 * DAY), Granularity::Day);
assert_eq!(Granularity::for_span(3600), Granularity::Hour);
}
#[test]
fn names_are_derived_for_both_paths_and_saf_ids() {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'saf', 'tree')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1, 1, 'primary:DCIM/Camera/IMG_1.CR3', 0)",
[],
)
.unwrap();
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
assert_eq!(rows[0].name, "IMG_1.CR3");
}
}