//! TRACES: FR-CAT-7 | FR-NC-9 //! Merging a remote catalog's collections into the local one. //! //! # Why this is a merge and not a copy //! //! The catalog file syncs to Nextcloud, and a device that finds a newer remote //! copy must not simply replace its own — whichever device synced second would //! lose everything the first did not have. So the remote file is downloaded to //! a side path, `ATTACH`ed, and merged table by table. //! //! Row-level merging needs identities that are stable across devices, and //! `collections.id INTEGER PRIMARY KEY` is not: two devices independently //! allocate id 1 for different collections. Hence `collections.uuid`, which is //! what everything here keys on. The integer id stays local and is never //! compared across catalogs. //! //! # Conflict rule //! //! Per collection, by `revision` — a monotonic counter bumped on every local //! edit — with `modified` timestamp only as a tiebreak. Comparing revisions //! rather than mtimes means a device with a skewed clock cannot silently win //! (the failure mode FR-NC-9 avoids for sidecars, applied here). //! //! Membership merges as a **set union**, not last-writer-wins: two devices //! each adding different images to the same collection keep both sets. That //! is almost always what the user meant, and the exception — a removal racing //! an addition — resolves in favour of the addition, which is recoverable by //! removing it again. Silently losing an addition is not. //! //! # Deletion //! //! A deleted collection leaves a tombstone (`deleted = 1`), because a merge //! against a device that still holds it would otherwise resurrect it. The //! tombstone carries a revision like any other edit, so deletion competes on //! the same footing as a rename. use rusqlite::Connection; use crate::error::CatalogError; /// How a collection differed between the two catalogs. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum MergeVerdict { /// Present only remotely — insert it. InsertedFromRemote, /// Remote revision is higher — take its fields. UpdatedFromRemote, /// Local revision is at least as high — keep ours. KeptLocal, /// Remote says deleted, and wins on revision. DeletedByRemote, } /// What a merge did, for logging and for telling the user. #[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] pub struct MergeReport { pub inserted: usize, pub updated: usize, pub kept_local: usize, pub deleted: usize, pub members_added: usize, } impl MergeReport { /// Whether the local catalog changed, and so needs re-uploading. pub fn local_changed(&self) -> bool { self.inserted > 0 || self.updated > 0 || self.deleted > 0 || self.members_added > 0 } /// Whether the local catalog holds anything the remote did not, and so /// must be uploaded even if nothing was taken from the remote. pub fn should_upload(&self) -> bool { self.kept_local > 0 || self.local_changed() } } /// Decide one collection, given both sides' revisions. /// /// Split out from the SQL so the rule is testable on its own — it is the part /// that decides whether a user loses a collection. pub fn verdict( local: Option<(i64, i64)>, // (revision, modified) remote: (i64, i64), remote_deleted: bool, ) -> MergeVerdict { let (r_rev, r_mod) = remote; match local { None if remote_deleted => { // A tombstone for something we never had. Recording it still // matters: without it, a third device could reintroduce the // collection through us. MergeVerdict::DeletedByRemote } None => MergeVerdict::InsertedFromRemote, Some((l_rev, l_mod)) => { // Revision first; timestamp only to break an exact tie. Equal // revisions with equal timestamps keep local, so a merge that // changes nothing is stable and repeatable. let remote_wins = r_rev > l_rev || (r_rev == l_rev && r_mod > l_mod); if !remote_wins { MergeVerdict::KeptLocal } else if remote_deleted { MergeVerdict::DeletedByRemote } else { MergeVerdict::UpdatedFromRemote } } } } /// Merge collections and membership from an attached catalog. /// /// The remote catalog must already be attached under the schema name /// `remote_cat`; [`crate::Catalog::merge_attached_collections`] handles that. /// /// Runs in one transaction: a merge either lands whole or not at all. pub fn merge_collections(conn: &Connection) -> Result { let tx = conn.unchecked_transaction()?; let mut report = MergeReport::default(); // ---- collections ------------------------------------------------------ { let mut stmt = tx.prepare( // The parent arrives as a *uuid*, not `r.parent_id`: row ids are // local to a catalog, so the remote's integer means nothing here. "SELECT r.uuid, r.name, rp.uuid, r.kind, r.selector_json, r.created, r.revision, r.modified, r.deleted, l.revision, l.modified FROM remote_cat.collections r LEFT JOIN main.collections l ON l.uuid = r.uuid LEFT JOIN remote_cat.collections rp ON rp.id = r.parent_id", )?; struct Incoming { uuid: String, name: String, kind: i64, /// The parent's uuid, resolved to a local row id once every /// incoming collection exists — a child can arrive before its /// parent, so this cannot be applied inline. parent_uuid: Option, selector_json: Option, created: i64, revision: i64, modified: i64, // No `deleted` field: the verdict already encodes it, and keeping // both invites the two disagreeing. verdict: MergeVerdict, } let rows: Vec = stmt .query_map([], |r| { let deleted: i64 = r.get(8)?; let local_rev: Option = r.get(9)?; let local_mod: Option = r.get(10)?; let revision: i64 = r.get(6)?; let modified: i64 = r.get(7)?; Ok(Incoming { uuid: r.get(0)?, name: r.get(1)?, kind: r.get(3)?, parent_uuid: r.get(2)?, selector_json: r.get(4)?, created: r.get(5)?, revision, modified, verdict: verdict(local_rev.zip(local_mod), (revision, modified), deleted != 0), }) })? .collect::>()?; // Parentage is applied after the loop: a child can arrive before its // parent, so resolving the uuid inline would find nothing and silently // flatten the tree. let mut reparent: Vec<(String, Option)> = Vec::new(); for row in rows { match row.verdict { MergeVerdict::KeptLocal => { report.kept_local += 1; } MergeVerdict::InsertedFromRemote => { tx.execute( "INSERT INTO main.collections (uuid, name, parent_id, kind, selector_json, created, revision, modified, deleted) VALUES (?1, ?2, NULL, ?3, ?4, ?5, ?6, ?7, 0)", rusqlite::params![ row.uuid, row.name, row.kind, row.selector_json, row.created, row.revision, row.modified, ], )?; reparent.push((row.uuid.clone(), row.parent_uuid.clone())); report.inserted += 1; } MergeVerdict::UpdatedFromRemote => { tx.execute( "UPDATE main.collections SET name = ?2, kind = ?3, selector_json = ?4, revision = ?5, modified = ?6, deleted = 0 WHERE uuid = ?1", rusqlite::params![ row.uuid, row.name, row.kind, row.selector_json, row.revision, row.modified, ], )?; reparent.push((row.uuid.clone(), row.parent_uuid.clone())); report.updated += 1; } MergeVerdict::DeletedByRemote => { // Tombstone rather than DELETE: the row must outlive the // deletion or a third device reintroduces it. tx.execute( "INSERT INTO main.collections (uuid, name, kind, created, revision, modified, deleted) VALUES (?1, ?2, ?3, ?4, ?5, ?6, 1) ON CONFLICT(uuid) DO UPDATE SET deleted = 1, revision = ?5, modified = ?6", rusqlite::params![ row.uuid, row.name, row.kind, row.created, row.revision, row.modified, ], )?; tx.execute( "DELETE FROM main.collection_members WHERE collection_id = (SELECT id FROM main.collections WHERE uuid = ?1)", [&row.uuid], )?; report.deleted += 1; } } } // Second pass: every incoming collection now exists locally, so a // parent uuid can be resolved to a row id. A parent we have never seen // resolves to NULL, which leaves the collection at the top level — // wrong, but visible and recoverable, where a dangling id would not be. // // Without this the tree flattened on every sync: `parent_id` is a local // row id and was written as NULL rather than translated, so a nested // collection came back from a round trip at the top level. for (uuid, parent_uuid) in reparent { // Remote's tree is acyclic and so is ours, but the union of the two // need not be: if we hold A above B and the remote holds B above A, // applying only the winning half closes a loop, and `descendants` // would then spin. Walk up from the proposed parent first; reaching // the collection itself means this edge would close a cycle, so the // safe move is to leave it where it is. if let Some(ref parent) = parent_uuid { let closes_cycle: bool = tx.query_row( "WITH RECURSIVE up(id) AS ( SELECT id FROM main.collections WHERE uuid = ?2 UNION SELECT c.parent_id FROM main.collections c JOIN up ON c.id = up.id WHERE c.parent_id IS NOT NULL ) SELECT EXISTS( SELECT 1 FROM up WHERE id = (SELECT id FROM main.collections WHERE uuid = ?1))", rusqlite::params![uuid, parent], |r| r.get(0), )?; if closes_cycle { continue; } } tx.execute( "UPDATE main.collections SET parent_id = (SELECT id FROM main.collections WHERE uuid = ?2) WHERE uuid = ?1", rusqlite::params![uuid, parent_uuid], )?; } } // ---- membership ------------------------------------------------------- // // Set union, keyed on (collection uuid, image content hash). The hash // rather than the image id, for the same reason collections use a uuid: // image ids are local. An image the remote has and we do not is skipped — // it will join when a scan or sync catalogues it, and the next merge picks // it up. // // Tombstoned collections are excluded, or a merge would repopulate a // collection it had just deleted. let added = tx.execute( "INSERT OR IGNORE INTO main.collection_members(collection_id, image_id, position, added) SELECT lc.id, li.id, rm.position, rm.added FROM remote_cat.collection_members rm JOIN remote_cat.collections rc ON rc.id = rm.collection_id JOIN main.collections lc ON lc.uuid = rc.uuid AND lc.deleted = 0 JOIN remote_cat.images ri ON ri.id = rm.image_id JOIN main.images li ON li.content_hash = ri.content_hash WHERE ri.content_hash IS NOT NULL", [], )?; report.members_added = added; tx.commit()?; Ok(report) } #[cfg(test)] mod tests { use super::*; use crate::schema; #[test] fn a_collection_we_lack_is_taken_from_remote() { assert_eq!( verdict(None, (1, 100), false), MergeVerdict::InsertedFromRemote ); } #[test] fn higher_remote_revision_wins() { assert_eq!( verdict(Some((3, 100)), (4, 50), false), MergeVerdict::UpdatedFromRemote ); } #[test] fn a_skewed_clock_cannot_beat_a_higher_local_revision() { // The remote's timestamp is far in the future, but it has seen fewer // edits. Revision decides, so the skewed device does not silently // overwrite real work. assert_eq!( verdict(Some((9, 100)), (2, 999_999), false), MergeVerdict::KeptLocal ); } #[test] fn equal_revisions_break_on_timestamp() { assert_eq!( verdict(Some((3, 100)), (3, 200), false), MergeVerdict::UpdatedFromRemote ); assert_eq!( verdict(Some((3, 200)), (3, 100), false), MergeVerdict::KeptLocal ); } #[test] fn an_identical_collection_is_stable() { // Merging twice must not oscillate or report spurious changes. assert_eq!( verdict(Some((3, 100)), (3, 100), false), MergeVerdict::KeptLocal ); } #[test] fn deletion_competes_on_revision_like_any_other_edit() { // Remote deleted it at revision 5; we renamed it at revision 4. The // deletion is newer, so it wins. assert_eq!( verdict(Some((4, 100)), (5, 100), true), MergeVerdict::DeletedByRemote ); // But a stale deletion does not undo a newer local edit. assert_eq!( verdict(Some((6, 100)), (5, 100), true), MergeVerdict::KeptLocal ); } #[test] fn a_tombstone_for_something_we_never_had_is_recorded() { // Otherwise this device could reintroduce the collection to a third. assert_eq!(verdict(None, (2, 100), true), MergeVerdict::DeletedByRemote); } // ---- integration over two real catalogs ------------------------------ fn two_catalogs() -> Connection { let c = Connection::open_in_memory().unwrap(); schema::configure(&c).unwrap(); schema::migrate(&c).unwrap(); // A second in-memory database standing in for the downloaded remote. c.execute_batch("ATTACH ':memory:' AS remote_cat").unwrap(); let remote_schema = super::super::schema::v1_for_attached("remote_cat"); c.execute_batch(&remote_schema).unwrap(); c } fn add_image(c: &Connection, db: &str, id: i64, hash: &str) { c.execute( &format!( "INSERT INTO {db}.roots(id, kind, label) VALUES (1, 'local', 'r') ON CONFLICT(id) DO NOTHING" ), [], ) .unwrap(); c.execute( &format!( "INSERT INTO {db}.images(id, root_id, source_ref, content_hash, added_at) VALUES (?1, 1, ?2, ?3, 0)" ), rusqlite::params![id, format!("img{id}.CR3"), hash], ) .unwrap(); } fn add_collection(c: &Connection, db: &str, id: i64, uuid: &str, name: &str, rev: i64) { c.execute( &format!( "INSERT INTO {db}.collections(id, uuid, name, kind, created, revision, modified) VALUES (?1, ?2, ?3, 0, 0, ?4, ?4)" ), rusqlite::params![id, uuid, name, rev], ) .unwrap(); } /// The parent's uuid for `uuid`, or None if it sits at the top level. fn parent_of(c: &Connection, uuid: &str) -> Option { c.query_row( "SELECT p.uuid FROM main.collections ch JOIN main.collections p ON p.id = ch.parent_id WHERE ch.uuid = ?1", [uuid], |r| r.get(0), ) .ok() } #[test] fn a_nested_collection_arrives_still_nested() { // The tree used to flatten on every sync: `parent_id` is a local row id, // and the merge wrote NULL rather than translating it through the uuid. let c = two_catalogs(); add_collection(&c, "remote_cat", 1, "uuid-holidays", "holidays", 1); add_collection(&c, "remote_cat", 2, "uuid-arosa", "arosa", 1); c.execute( "UPDATE remote_cat.collections SET parent_id = 1 WHERE id = 2", [], ) .unwrap(); merge_collections(&c).unwrap(); assert_eq!( parent_of(&c, "uuid-arosa").as_deref(), Some("uuid-holidays"), "a sub-collection must not be promoted to the top level by a merge" ); } #[test] fn a_parent_arriving_after_its_child_still_adopts_it() { // Row order is whatever the query returns, so the child can be inserted // first. Parentage is applied in a second pass for exactly this reason. let c = two_catalogs(); // Child has the lower id, so it is seen before the parent exists. add_collection(&c, "remote_cat", 1, "uuid-child", "arosa", 1); add_collection(&c, "remote_cat", 2, "uuid-parent", "holidays", 1); c.execute( "UPDATE remote_cat.collections SET parent_id = 2 WHERE id = 1", [], ) .unwrap(); merge_collections(&c).unwrap(); assert_eq!( parent_of(&c, "uuid-child").as_deref(), Some("uuid-parent"), "resolution must not depend on the order rows happen to arrive in" ); } #[test] fn disagreeing_devices_cannot_close_a_parent_cycle() { // Local holds A above B; the remote holds B above A and wins on // revision. Applying that blindly makes each the other's parent, and // every tree walk then spins. let c = two_catalogs(); add_collection(&c, "main", 1, "uuid-a", "A", 1); add_collection(&c, "main", 2, "uuid-b", "B", 1); c.execute("UPDATE main.collections SET parent_id = 1 WHERE id = 2", []) .unwrap(); // Remote: A under B, at a higher revision so it is the winner. add_collection(&c, "remote_cat", 1, "uuid-a", "A", 9); add_collection(&c, "remote_cat", 2, "uuid-b", "B", 9); c.execute( "UPDATE remote_cat.collections SET parent_id = 2 WHERE id = 1", [], ) .unwrap(); merge_collections(&c).unwrap(); let a = parent_of(&c, "uuid-a"); let b = parent_of(&c, "uuid-b"); assert!( !(a.as_deref() == Some("uuid-b") && b.as_deref() == Some("uuid-a")), "merge closed a cycle: A parented under B and B under A" ); } #[test] fn disjoint_collections_from_two_devices_both_survive() { // The property the whole design exists for: neither device loses work. let c = two_catalogs(); add_collection(&c, "main", 1, "uuid-local", "Iceland", 1); add_collection(&c, "remote_cat", 1, "uuid-remote", "Portugal", 1); let report = merge_collections(&c).unwrap(); assert_eq!(report.inserted, 1); let names: Vec = c .prepare("SELECT name FROM main.collections ORDER BY name") .unwrap() .query_map([], |r| r.get(0)) .unwrap() .collect::>() .unwrap(); assert_eq!(names, vec!["Iceland", "Portugal"]); } #[test] fn membership_unions_rather_than_replacing() { // Two devices each added a different image to the same collection. let c = two_catalogs(); add_collection(&c, "main", 1, "shared", "Trip", 1); add_collection(&c, "remote_cat", 1, "shared", "Trip", 1); add_image(&c, "main", 1, "hash-a"); add_image(&c, "main", 2, "hash-b"); add_image(&c, "remote_cat", 1, "hash-b"); c.execute( "INSERT INTO main.collection_members(collection_id, image_id, added) VALUES (1, 1, 0)", [], ) .unwrap(); c.execute( "INSERT INTO remote_cat.collection_members(collection_id, image_id, added) VALUES (1, 1, 0)", [], ) .unwrap(); merge_collections(&c).unwrap(); let n: i64 = c .query_row("SELECT count(*) FROM main.collection_members", [], |r| { r.get(0) }) .unwrap(); assert_eq!(n, 2, "both devices' additions survive"); } #[test] fn membership_maps_across_devices_by_content_hash() { // The same photograph carries different integer ids on each device. // Keying on the id would attach the wrong image. let c = two_catalogs(); add_collection(&c, "main", 1, "shared", "Trip", 1); add_collection(&c, "remote_cat", 1, "shared", "Trip", 1); add_image(&c, "main", 77, "same-photo"); add_image(&c, "remote_cat", 3, "same-photo"); c.execute( "INSERT INTO remote_cat.collection_members(collection_id, image_id, added) VALUES (1, 3, 0)", [], ) .unwrap(); merge_collections(&c).unwrap(); let img: i64 = c .query_row("SELECT image_id FROM main.collection_members", [], |r| { r.get(0) }) .unwrap(); assert_eq!(img, 77, "resolved to the local id for the same photo"); } #[test] fn an_image_we_do_not_have_yet_is_skipped_not_errored() { let c = two_catalogs(); add_collection(&c, "main", 1, "shared", "Trip", 1); add_collection(&c, "remote_cat", 1, "shared", "Trip", 1); add_image(&c, "remote_cat", 1, "not-here-yet"); c.execute( "INSERT INTO remote_cat.collection_members(collection_id, image_id, added) VALUES (1, 1, 0)", [], ) .unwrap(); let report = merge_collections(&c).unwrap(); assert_eq!(report.members_added, 0); // It joins on a later merge, once a scan has catalogued the file. } #[test] fn a_remote_deletion_does_not_resurrect_via_membership() { let c = two_catalogs(); add_collection(&c, "main", 1, "doomed", "Old", 1); add_image(&c, "main", 1, "hash-a"); add_image(&c, "remote_cat", 1, "hash-a"); c.execute( "INSERT INTO remote_cat.collections(id, uuid, name, kind, created, revision, modified, deleted) VALUES (1, 'doomed', 'Old', 0, 0, 5, 5, 1)", [], ) .unwrap(); c.execute( "INSERT INTO remote_cat.collection_members(collection_id, image_id, added) VALUES (1, 1, 0)", [], ) .unwrap(); let report = merge_collections(&c).unwrap(); assert_eq!(report.deleted, 1); let n: i64 = c .query_row("SELECT count(*) FROM main.collection_members", [], |r| { r.get(0) }) .unwrap(); assert_eq!(n, 0, "membership must not repopulate a deleted collection"); } #[test] fn merging_twice_changes_nothing_the_second_time() { let c = two_catalogs(); add_collection(&c, "remote_cat", 1, "uuid-r", "Portugal", 1); let first = merge_collections(&c).unwrap(); assert!(first.local_changed()); let second = merge_collections(&c).unwrap(); assert!(!second.local_changed(), "merge must be idempotent"); } #[test] fn keeping_local_still_marks_the_catalog_for_upload() { // We hold something the remote does not, so the remote is stale even // though we took nothing from it. let c = two_catalogs(); add_collection(&c, "main", 1, "shared", "Renamed here", 5); add_collection(&c, "remote_cat", 1, "shared", "Old name", 2); let report = merge_collections(&c).unwrap(); assert_eq!(report.kept_local, 1); assert!(report.should_upload()); } }