2342 lines
95 KiB
Rust
2342 lines
95 KiB
Rust
//! TRACES: FR-CAT-7 | FR-CAT-5 | FR-NC-9
|
|
//! Merging a remote catalog's collections and keywords into the local one.
|
|
//!
|
|
//! # Why this is a merge and not a copy
|
|
//!
|
|
//! The catalog file syncs to Nextcloud, and a device that finds a newer remote
|
|
//! copy must not simply replace its own — whichever device synced second would
|
|
//! lose everything the first did not have. So the remote file is downloaded to
|
|
//! a side path, `ATTACH`ed, and merged table by table.
|
|
//!
|
|
//! Row-level merging needs identities that are stable across devices, and
|
|
//! `collections.id INTEGER PRIMARY KEY` is not: two devices independently
|
|
//! allocate id 1 for different collections. Hence `collections.uuid`, which is
|
|
//! what everything here keys on. The integer id stays local and is never
|
|
//! compared across catalogs.
|
|
//!
|
|
//! # Conflict rule
|
|
//!
|
|
//! Per collection, by `revision` — a monotonic counter bumped on every local
|
|
//! edit — with `modified` timestamp only as a tiebreak. Comparing revisions
|
|
//! rather than mtimes means a device with a skewed clock cannot silently win
|
|
//! (the failure mode FR-NC-9 avoids for sidecars, applied here).
|
|
//!
|
|
//! Membership merges as a **set union**, not last-writer-wins: two devices
|
|
//! each adding different images to the same collection keep both sets. That
|
|
//! is almost always what the user meant, and the exception — a removal racing
|
|
//! an addition — resolves in favour of the addition, which is recoverable by
|
|
//! removing it again. Silently losing an addition is not.
|
|
//!
|
|
//! # Deletion
|
|
//!
|
|
//! A deleted collection leaves a tombstone (`deleted = 1`), because a merge
|
|
//! against a device that still holds it would otherwise resurrect it. The
|
|
//! tombstone carries a revision like any other edit, so deletion competes on
|
|
//! the same footing as a rename.
|
|
//!
|
|
//! # Keywords merge on the same three rules
|
|
//!
|
|
//! [`merge_keywords`] reuses all of the above rather than inventing a second
|
|
//! set of rules, because a keyword is the same shape of problem as a
|
|
//! collection: a named thing with a device-independent identity, and a
|
|
//! many-to-many join to images.
|
|
//!
|
|
//! - The **vocabulary** (`keyword_terms`) is decided per row by [`verdict`],
|
|
//! exactly as collections are.
|
|
//! - The **assignments** (`keywords`) are a set union, exactly as membership
|
|
//! is: two devices each keywording different photographs "puffin" keep both
|
|
//! sets, and two devices each keywording the *same* photograph converge on
|
|
//! one row rather than one of them winning.
|
|
//! - **Deletion** tombstones, and takes the assignments with it.
|
|
//!
|
|
//! Two things are genuinely different, and both are consequences of assignments
|
|
//! storing the *word* rather than a row id:
|
|
//!
|
|
//! 1. A tombstone deletes assignments **by name**, so a deletion still lands on
|
|
//! a device that had minted its own identity for the same word. The union
|
|
//! then refuses to readmit a word a winning tombstone has just removed —
|
|
//! without that filter, the other device's live assignments would resurrect
|
|
//! it on the very same pass.
|
|
//! 2. Two devices that independently typed the same word arrive with two uuids
|
|
//! for one keyword. [`crate::keywords::fuse_duplicates`] collapses them onto
|
|
//! the lexicographically smaller one, which both devices compute identically.
|
|
//! A unique index on the name would instead abort the merge transaction at
|
|
//! that moment, which is the ordinary case rather than a corner one.
|
|
|
|
use rusqlite::{Connection, OptionalExtension};
|
|
|
|
use crate::error::CatalogError;
|
|
|
|
/// How a collection differed between the two catalogs.
|
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
|
pub enum MergeVerdict {
|
|
/// Present only remotely — insert it.
|
|
InsertedFromRemote,
|
|
/// Remote revision is higher — take its fields.
|
|
UpdatedFromRemote,
|
|
/// Local revision is at least as high — keep ours.
|
|
KeptLocal,
|
|
/// Remote says deleted, and wins on revision.
|
|
DeletedByRemote,
|
|
}
|
|
|
|
/// What a merge did, for logging and for telling the user.
|
|
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
|
pub struct MergeReport {
|
|
pub inserted: usize,
|
|
pub updated: usize,
|
|
pub kept_local: usize,
|
|
pub deleted: usize,
|
|
pub members_added: usize,
|
|
|
|
// Keywords are counted separately from collections rather than summed into
|
|
// the same fields. The report is shown to the user — "3 collections, 11
|
|
// keywords" is a sentence; "14 things" is not — and a merge that went wrong
|
|
// is far easier to place when the counts say which half it went wrong in.
|
|
/// Keywords the remote had and this device did not.
|
|
pub keywords_inserted: usize,
|
|
/// Keywords the remote had renamed, or brought back from a tombstone.
|
|
pub keywords_updated: usize,
|
|
/// Keywords the remote deleted, and this device has now deleted too.
|
|
pub keywords_deleted: usize,
|
|
/// Keywords where this device's revision was at least as high.
|
|
pub keywords_kept_local: usize,
|
|
/// People the remote had and this device did not.
|
|
pub people_inserted: usize,
|
|
/// People the remote had renamed, set aside, or brought back.
|
|
pub people_updated: usize,
|
|
/// People where this device's revision was at least as high.
|
|
pub people_kept_local: usize,
|
|
/// Faces this device now agrees belong to somebody.
|
|
pub faces_assigned: usize,
|
|
/// Faces whose local confirmation outranked the remote's.
|
|
pub faces_kept_local: usize,
|
|
/// "Not this person" judgements taken from the remote.
|
|
pub faces_rejected: usize,
|
|
|
|
/// Redundant identities for one word, retired by
|
|
/// [`crate::keywords::fuse_duplicates`].
|
|
pub keywords_fused: usize,
|
|
/// Keyword assignments taken from the remote.
|
|
pub keywords_assigned: usize,
|
|
/// Images whose capture metadata was taken from the remote.
|
|
pub metadata_adopted: usize,
|
|
}
|
|
|
|
impl MergeReport {
|
|
/// Whether the local catalog changed, and so needs re-uploading.
|
|
pub fn local_changed(&self) -> bool {
|
|
self.inserted > 0
|
|
|| self.updated > 0
|
|
|| self.deleted > 0
|
|
|| self.members_added > 0
|
|
|| self.keywords_inserted > 0
|
|
|| self.keywords_updated > 0
|
|
|| self.keywords_deleted > 0
|
|
|| self.keywords_fused > 0
|
|
|| self.keywords_assigned > 0
|
|
|| self.metadata_adopted > 0
|
|
}
|
|
|
|
/// Whether the local catalog holds anything the remote did not, and so
|
|
/// must be uploaded even if nothing was taken from the remote.
|
|
pub fn should_upload(&self) -> bool {
|
|
self.kept_local > 0 || self.keywords_kept_local > 0 || self.local_changed()
|
|
}
|
|
}
|
|
|
|
/// Decide one collection, given both sides' revisions.
|
|
///
|
|
/// Split out from the SQL so the rule is testable on its own — it is the part
|
|
/// that decides whether a user loses a collection.
|
|
pub fn verdict(
|
|
local: Option<(i64, i64)>, // (revision, modified)
|
|
remote: (i64, i64),
|
|
remote_deleted: bool,
|
|
) -> MergeVerdict {
|
|
let (r_rev, r_mod) = remote;
|
|
match local {
|
|
None if remote_deleted => {
|
|
// A tombstone for something we never had. Recording it still
|
|
// matters: without it, a third device could reintroduce the
|
|
// collection through us.
|
|
MergeVerdict::DeletedByRemote
|
|
}
|
|
None => MergeVerdict::InsertedFromRemote,
|
|
Some((l_rev, l_mod)) => {
|
|
// Revision first; timestamp only to break an exact tie. Equal
|
|
// revisions with equal timestamps keep local, so a merge that
|
|
// changes nothing is stable and repeatable.
|
|
let remote_wins = r_rev > l_rev || (r_rev == l_rev && r_mod > l_mod);
|
|
if !remote_wins {
|
|
MergeVerdict::KeptLocal
|
|
} else if remote_deleted {
|
|
MergeVerdict::DeletedByRemote
|
|
} else {
|
|
MergeVerdict::UpdatedFromRemote
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Merge everything that syncs, from an attached catalog.
|
|
///
|
|
/// The remote catalog must already be attached under the schema name
|
|
/// `remote_cat`; [`crate::sync::merge_remote`] handles that.
|
|
///
|
|
/// **One transaction over both halves.** Keywords and collections are
|
|
/// independent as data, but a merge that landed the collections and then failed
|
|
/// on the keywords would leave a catalog that has already taken the remote's
|
|
/// revisions for half of itself — and the next attempt, seeing those revisions,
|
|
/// would decline to take them again. Half a merge is not a state that can be
|
|
/// resumed, so it is not a state that can be reached.
|
|
pub fn merge_all(conn: &Connection) -> Result<MergeReport, CatalogError> {
|
|
let tx = conn.unchecked_transaction()?;
|
|
let mut report = MergeReport::default();
|
|
merge_collections_within(&tx, &mut report)?;
|
|
merge_keywords_within(&tx, &mut report)?;
|
|
merge_people_within(&tx, &mut report)?;
|
|
merge_metadata_within(&tx, &mut report)?;
|
|
tx.commit()?;
|
|
Ok(report)
|
|
}
|
|
|
|
/// Adopt capture metadata from an attached catalog, on its own.
|
|
pub fn merge_metadata(conn: &Connection) -> Result<MergeReport, CatalogError> {
|
|
let tx = conn.unchecked_transaction()?;
|
|
let mut report = MergeReport::default();
|
|
merge_metadata_within(&tx, &mut report)?;
|
|
tx.commit()?;
|
|
Ok(report)
|
|
}
|
|
|
|
/// Capture metadata a peer's sweep already read, for images this device has
|
|
/// not dated yet.
|
|
///
|
|
/// The `images` table is local state and the merge leaves it alone — except
|
|
/// for these columns, which are not: a capture time, an offset, a camera, a
|
|
/// lens and an ISO are facts about the file's bytes, identical on every
|
|
/// device, and read by fetching a header per image across the whole library
|
|
/// (`dr_ui::library::spawn_sweep`). A fresh device inherits its peers'
|
|
/// thumbnails and faces from the shards and then spent hours re-reading
|
|
/// every header for the timeline; the snapshot it had just merged held
|
|
/// every one of those dates.
|
|
///
|
|
/// Matched by `oc:fileid`, as collection membership is. Only rows still at
|
|
/// `metadata_state < 2` take anything, and only from a remote row at 2: a
|
|
/// date this device read for itself is never overwritten, and a peer that
|
|
/// has not read one has nothing to give. The sweep's own query
|
|
/// (`metadata_state < 2`) then finds nothing left to do for them.
|
|
const METADATA_BY_FILE_ID: &str = "
|
|
UPDATE main.images
|
|
SET captured_at = r.captured_at,
|
|
captured_offset = coalesce(main.images.captured_offset, r.captured_offset),
|
|
camera = coalesce(main.images.camera, r.camera),
|
|
lens = coalesce(main.images.lens, r.lens),
|
|
iso = coalesce(main.images.iso, r.iso),
|
|
metadata_state = 2
|
|
FROM (SELECT lr.image_id, ri.captured_at, ri.captured_offset,
|
|
ri.camera, ri.lens, ri.iso
|
|
FROM remote_cat.images ri
|
|
JOIN remote_cat.remote rr ON rr.image_id = ri.id
|
|
JOIN main.remote lr ON lr.file_id = rr.file_id
|
|
WHERE ri.metadata_state >= 2 AND ri.captured_at IS NOT NULL) AS r
|
|
WHERE main.images.id = r.image_id
|
|
AND main.images.metadata_state < 2";
|
|
|
|
fn merge_metadata_within(tx: &Connection, report: &mut MergeReport) -> Result<(), CatalogError> {
|
|
// A snapshot from before these columns, or from a library with no server
|
|
// behind it, has nothing to join on.
|
|
if !remote_has(tx, "remote")?
|
|
|| !remote_has_column(tx, "images", "metadata_state")?
|
|
|| !remote_has_column(tx, "images", "captured_offset")?
|
|
{
|
|
return Ok(());
|
|
}
|
|
report.metadata_adopted = tx.execute(METADATA_BY_FILE_ID, [])?;
|
|
Ok(())
|
|
}
|
|
|
|
/// Merge people and identity judgements from an attached catalog.
|
|
///
|
|
/// The people half of [`merge_all`], on its own, for the same reason the other
|
|
/// two have one: the rules are independent and worth exercising alone.
|
|
pub fn merge_people(conn: &Connection) -> Result<MergeReport, CatalogError> {
|
|
let tx = conn.unchecked_transaction()?;
|
|
let mut report = MergeReport::default();
|
|
merge_people_within(&tx, &mut report)?;
|
|
tx.commit()?;
|
|
Ok(report)
|
|
}
|
|
|
|
/// Merge collections and membership from an attached catalog.
|
|
///
|
|
/// The collections half of [`merge_all`], on its own. Kept as a public entry
|
|
/// point because the two halves are genuinely independent, and because the
|
|
/// rules for this one are worth being able to exercise without a keyword in
|
|
/// sight.
|
|
///
|
|
/// Runs in one transaction: a merge either lands whole or not at all.
|
|
pub fn merge_collections(conn: &Connection) -> Result<MergeReport, CatalogError> {
|
|
let tx = conn.unchecked_transaction()?;
|
|
let mut report = MergeReport::default();
|
|
merge_collections_within(&tx, &mut report)?;
|
|
tx.commit()?;
|
|
Ok(report)
|
|
}
|
|
|
|
/// Merge the keyword vocabulary and its assignments from an attached catalog.
|
|
///
|
|
/// The keywords half of [`merge_all`], on its own. See the module header for
|
|
/// the three rules and the two places keywords differ from collections.
|
|
pub fn merge_keywords(conn: &Connection) -> Result<MergeReport, CatalogError> {
|
|
let tx = conn.unchecked_transaction()?;
|
|
let mut report = MergeReport::default();
|
|
merge_keywords_within(&tx, &mut report)?;
|
|
tx.commit()?;
|
|
Ok(report)
|
|
}
|
|
|
|
fn merge_collections_within(tx: &Connection, report: &mut MergeReport) -> Result<(), CatalogError> {
|
|
// ---- collections ------------------------------------------------------
|
|
{
|
|
let mut stmt = tx.prepare(
|
|
// The parent arrives as a *uuid*, not `r.parent_id`: row ids are
|
|
// local to a catalog, so the remote's integer means nothing here.
|
|
"SELECT r.uuid, r.name, rp.uuid, r.kind, r.selector_json,
|
|
r.created, r.revision, r.modified, r.deleted,
|
|
l.revision, l.modified
|
|
FROM remote_cat.collections r
|
|
LEFT JOIN main.collections l ON l.uuid = r.uuid
|
|
LEFT JOIN remote_cat.collections rp ON rp.id = r.parent_id",
|
|
)?;
|
|
|
|
struct Incoming {
|
|
uuid: String,
|
|
name: String,
|
|
kind: i64,
|
|
/// The parent's uuid, resolved to a local row id once every
|
|
/// incoming collection exists — a child can arrive before its
|
|
/// parent, so this cannot be applied inline.
|
|
parent_uuid: Option<String>,
|
|
selector_json: Option<String>,
|
|
created: i64,
|
|
revision: i64,
|
|
modified: i64,
|
|
// No `deleted` field: the verdict already encodes it, and keeping
|
|
// both invites the two disagreeing.
|
|
verdict: MergeVerdict,
|
|
}
|
|
|
|
let rows: Vec<Incoming> = stmt
|
|
.query_map([], |r| {
|
|
let deleted: i64 = r.get(8)?;
|
|
let local_rev: Option<i64> = r.get(9)?;
|
|
let local_mod: Option<i64> = r.get(10)?;
|
|
let revision: i64 = r.get(6)?;
|
|
let modified: i64 = r.get(7)?;
|
|
Ok(Incoming {
|
|
uuid: r.get(0)?,
|
|
name: r.get(1)?,
|
|
kind: r.get(3)?,
|
|
parent_uuid: r.get(2)?,
|
|
selector_json: r.get(4)?,
|
|
created: r.get(5)?,
|
|
revision,
|
|
modified,
|
|
verdict: verdict(local_rev.zip(local_mod), (revision, modified), deleted != 0),
|
|
})
|
|
})?
|
|
.collect::<Result<_, _>>()?;
|
|
|
|
// Parentage is applied after the loop: a child can arrive before its
|
|
// parent, so resolving the uuid inline would find nothing and silently
|
|
// flatten the tree.
|
|
let mut reparent: Vec<(String, Option<String>)> = Vec::new();
|
|
|
|
for row in rows {
|
|
match row.verdict {
|
|
MergeVerdict::KeptLocal => {
|
|
report.kept_local += 1;
|
|
}
|
|
MergeVerdict::InsertedFromRemote => {
|
|
tx.execute(
|
|
"INSERT INTO main.collections
|
|
(uuid, name, parent_id, kind, selector_json,
|
|
created, revision, modified, deleted)
|
|
VALUES (?1, ?2, NULL, ?3, ?4, ?5, ?6, ?7, 0)",
|
|
rusqlite::params![
|
|
row.uuid,
|
|
row.name,
|
|
row.kind,
|
|
row.selector_json,
|
|
row.created,
|
|
row.revision,
|
|
row.modified,
|
|
],
|
|
)?;
|
|
reparent.push((row.uuid.clone(), row.parent_uuid.clone()));
|
|
report.inserted += 1;
|
|
}
|
|
MergeVerdict::UpdatedFromRemote => {
|
|
tx.execute(
|
|
"UPDATE main.collections
|
|
SET name = ?2, kind = ?3, selector_json = ?4,
|
|
revision = ?5, modified = ?6, deleted = 0
|
|
WHERE uuid = ?1",
|
|
rusqlite::params![
|
|
row.uuid,
|
|
row.name,
|
|
row.kind,
|
|
row.selector_json,
|
|
row.revision,
|
|
row.modified,
|
|
],
|
|
)?;
|
|
reparent.push((row.uuid.clone(), row.parent_uuid.clone()));
|
|
report.updated += 1;
|
|
}
|
|
MergeVerdict::DeletedByRemote => {
|
|
// Tombstone rather than DELETE: the row must outlive the
|
|
// deletion or a third device reintroduces it.
|
|
tx.execute(
|
|
"INSERT INTO main.collections
|
|
(uuid, name, kind, created, revision, modified, deleted)
|
|
VALUES (?1, ?2, ?3, ?4, ?5, ?6, 1)
|
|
ON CONFLICT(uuid) DO UPDATE SET
|
|
deleted = 1, revision = ?5, modified = ?6",
|
|
rusqlite::params![
|
|
row.uuid,
|
|
row.name,
|
|
row.kind,
|
|
row.created,
|
|
row.revision,
|
|
row.modified,
|
|
],
|
|
)?;
|
|
tx.execute(
|
|
"DELETE FROM main.collection_members
|
|
WHERE collection_id = (SELECT id FROM main.collections WHERE uuid = ?1)",
|
|
[&row.uuid],
|
|
)?;
|
|
report.deleted += 1;
|
|
}
|
|
}
|
|
}
|
|
|
|
// Second pass: every incoming collection now exists locally, so a
|
|
// parent uuid can be resolved to a row id. A parent we have never seen
|
|
// resolves to NULL, which leaves the collection at the top level —
|
|
// wrong, but visible and recoverable, where a dangling id would not be.
|
|
//
|
|
// Without this the tree flattened on every sync: `parent_id` is a local
|
|
// row id and was written as NULL rather than translated, so a nested
|
|
// collection came back from a round trip at the top level.
|
|
for (uuid, parent_uuid) in reparent {
|
|
// Remote's tree is acyclic and so is ours, but the union of the two
|
|
// need not be: if we hold A above B and the remote holds B above A,
|
|
// applying only the winning half closes a loop, and `descendants`
|
|
// would then spin. Walk up from the proposed parent first; reaching
|
|
// the collection itself means this edge would close a cycle, so the
|
|
// safe move is to leave it where it is.
|
|
if let Some(ref parent) = parent_uuid {
|
|
let closes_cycle: bool = tx.query_row(
|
|
"WITH RECURSIVE up(id) AS (
|
|
SELECT id FROM main.collections WHERE uuid = ?2
|
|
UNION
|
|
SELECT c.parent_id FROM main.collections c
|
|
JOIN up ON c.id = up.id
|
|
WHERE c.parent_id IS NOT NULL
|
|
)
|
|
SELECT EXISTS(
|
|
SELECT 1 FROM up
|
|
WHERE id = (SELECT id FROM main.collections WHERE uuid = ?1))",
|
|
rusqlite::params![uuid, parent],
|
|
|r| r.get(0),
|
|
)?;
|
|
if closes_cycle {
|
|
continue;
|
|
}
|
|
}
|
|
tx.execute(
|
|
"UPDATE main.collections
|
|
SET parent_id = (SELECT id FROM main.collections WHERE uuid = ?2)
|
|
WHERE uuid = ?1",
|
|
rusqlite::params![uuid, parent_uuid],
|
|
)?;
|
|
}
|
|
}
|
|
|
|
// ---- membership -------------------------------------------------------
|
|
//
|
|
// Set union, keyed on (collection uuid, image identity). Not the image id,
|
|
// for the same reason collections use a uuid: image ids are local. An image
|
|
// the remote has and we do not is skipped — it will join when a scan or
|
|
// sync catalogues it, and the next merge picks it up.
|
|
//
|
|
// Both identities are tried, and the file id first. See
|
|
// [`MEMBERS_BY_FILE_ID`] for why keying on the content hash alone made this
|
|
// whole union a no-op on every ordinary library.
|
|
//
|
|
// Tombstoned collections are excluded, or a merge would repopulate a
|
|
// collection it had just deleted.
|
|
let added = tx.execute(MEMBERS_BY_FILE_ID, [])? + tx.execute(MEMBERS_BY_CONTENT_HASH, [])?;
|
|
report.members_added = added;
|
|
|
|
Ok(())
|
|
}
|
|
|
|
/// Take membership for images both devices know by the server's file id.
|
|
///
|
|
/// **This is the identity that exists.** Membership was keyed on
|
|
/// `images.content_hash` alone, and the schema is explicit that the column is
|
|
/// "computed only when something needs it (import dedup, reconnect-by-hash),
|
|
/// never in a scan" — so on an ordinary library it is NULL for every row, the
|
|
/// join matched nothing, and `WHERE ri.content_hash IS NOT NULL` discarded what
|
|
/// little was left. Collections synced their names, because those are keyed on
|
|
/// a uuid, and arrived empty on every device. A 23,000-image library had a
|
|
/// content hash for none of them and an `oc:fileid` for all of them.
|
|
///
|
|
/// `oc:fileid` is recorded for every image the moment a remote scan sees it, is
|
|
/// stable across server-side rename and move (FR-NC-5), and is the same integer
|
|
/// on every device pointed at the same Nextcloud — which is exactly the
|
|
/// situation where two devices share collections. It is already what the
|
|
/// thumbnail shards are keyed by, and what [`ASSIGN_BY_FILE_ID`] uses for
|
|
/// keywords; membership was the one thing left behind.
|
|
const MEMBERS_BY_FILE_ID: &str = "
|
|
INSERT OR IGNORE INTO main.collection_members(collection_id, image_id, position, added)
|
|
SELECT lc.id, li.id, rm.position, rm.added
|
|
FROM remote_cat.collection_members rm
|
|
JOIN remote_cat.collections rc ON rc.id = rm.collection_id
|
|
JOIN main.collections lc ON lc.uuid = rc.uuid AND lc.deleted = 0
|
|
JOIN remote_cat.remote rr ON rr.image_id = rm.image_id
|
|
JOIN main.remote lr ON lr.file_id = rr.file_id
|
|
JOIN main.images li ON li.id = lr.image_id";
|
|
|
|
/// The same union for a library with no server behind it.
|
|
///
|
|
/// A local-only library has no `remote` rows at all, so [`MEMBERS_BY_FILE_ID`]
|
|
/// matches nothing and the content hash is the only identity available. Kept
|
|
/// rather than replaced: where a hash *has* been computed — an imported card,
|
|
/// a reconnect — it is a true identity, and one that survives a library moving
|
|
/// between servers.
|
|
///
|
|
/// Both statements run. `INSERT OR IGNORE` against the
|
|
/// `(collection_id, image_id)` primary key makes the overlap free.
|
|
const MEMBERS_BY_CONTENT_HASH: &str = "
|
|
INSERT OR IGNORE INTO main.collection_members(collection_id, image_id, position, added)
|
|
SELECT lc.id, li.id, rm.position, rm.added
|
|
FROM remote_cat.collection_members rm
|
|
JOIN remote_cat.collections rc ON rc.id = rm.collection_id
|
|
JOIN main.collections lc ON lc.uuid = rc.uuid AND lc.deleted = 0
|
|
JOIN remote_cat.images ri ON ri.id = rm.image_id
|
|
JOIN main.images li ON li.content_hash = ri.content_hash
|
|
WHERE ri.content_hash IS NOT NULL";
|
|
|
|
/// Schema name the downloaded remote catalog is attached under.
|
|
///
|
|
/// Repeated from [`crate::sync`] rather than shared, because the SQL below
|
|
/// spells it inline and a constant that only half the file used would be worse
|
|
/// than no constant at all.
|
|
const REMOTE: &str = "remote_cat";
|
|
|
|
/// The keyword half. See the module header.
|
|
fn merge_keywords_within(tx: &Connection, report: &mut MergeReport) -> Result<(), CatalogError> {
|
|
// ---- the vocabulary ---------------------------------------------------
|
|
//
|
|
// A remote written before schema v6 has no `keyword_terms` at all, and
|
|
// `remote_is_mergeable` deliberately admits it: the check is that the
|
|
// remote is not *newer* than us. So the table's absence is a normal state
|
|
// and not an error. Its assignments still merge below — those have been in
|
|
// the schema since v1 — and its words gain identities on that device the
|
|
// next time it opens the catalog and backfills.
|
|
if attached_has_table(tx, REMOTE, "keyword_terms")? {
|
|
struct Incoming {
|
|
uuid: String,
|
|
name: String,
|
|
/// What this device currently calls the same identity, if it has
|
|
/// it. A rename is applied to the assignment rows by rewriting this
|
|
/// text, so it has to be read before the term row is overwritten.
|
|
local_name: Option<String>,
|
|
created: i64,
|
|
revision: i64,
|
|
modified: i64,
|
|
verdict: MergeVerdict,
|
|
}
|
|
|
|
let rows: Vec<Incoming> = {
|
|
let mut stmt = tx.prepare(
|
|
"SELECT r.uuid, r.name, r.created, r.revision, r.modified, r.deleted,
|
|
l.name, l.revision, l.modified
|
|
FROM remote_cat.keyword_terms r
|
|
LEFT JOIN main.keyword_terms l ON l.uuid = r.uuid",
|
|
)?;
|
|
let found = stmt
|
|
.query_map([], |r| {
|
|
let deleted: i64 = r.get(5)?;
|
|
let local_rev: Option<i64> = r.get(7)?;
|
|
let local_mod: Option<i64> = r.get(8)?;
|
|
let revision: i64 = r.get(3)?;
|
|
let modified: i64 = r.get(4)?;
|
|
Ok(Incoming {
|
|
uuid: r.get(0)?,
|
|
name: r.get(1)?,
|
|
local_name: r.get(6)?,
|
|
created: r.get(2)?,
|
|
revision,
|
|
modified,
|
|
verdict: verdict(
|
|
local_rev.zip(local_mod),
|
|
(revision, modified),
|
|
deleted != 0,
|
|
),
|
|
})
|
|
})?
|
|
.collect::<Result<Vec<_>, _>>()?;
|
|
found
|
|
};
|
|
|
|
for row in rows {
|
|
match row.verdict {
|
|
MergeVerdict::KeptLocal => {
|
|
report.keywords_kept_local += 1;
|
|
}
|
|
MergeVerdict::InsertedFromRemote => {
|
|
tx.execute(
|
|
"INSERT INTO main.keyword_terms
|
|
(uuid, name, created, revision, modified, deleted)
|
|
VALUES (?1, ?2, ?3, ?4, ?5, 0)",
|
|
rusqlite::params![
|
|
row.uuid,
|
|
row.name,
|
|
row.created,
|
|
row.revision,
|
|
row.modified,
|
|
],
|
|
)?;
|
|
report.keywords_inserted += 1;
|
|
}
|
|
MergeVerdict::UpdatedFromRemote => {
|
|
// The assignments carry the *word*, so taking a new name
|
|
// for an identity we already hold means rewriting every row
|
|
// spelt the old way. Without this the vocabulary would show
|
|
// the new spelling and the search would only find the old.
|
|
if let Some(old) = row.local_name.filter(|n| *n != row.name) {
|
|
tx.execute(
|
|
"INSERT OR IGNORE INTO main.keywords(version_id, keyword)
|
|
SELECT version_id, ?2 FROM main.keywords WHERE keyword = ?1",
|
|
rusqlite::params![old, row.name],
|
|
)?;
|
|
tx.execute("DELETE FROM main.keywords WHERE keyword = ?1", [&old])?;
|
|
}
|
|
tx.execute(
|
|
"UPDATE main.keyword_terms
|
|
SET name = ?2, revision = ?3, modified = ?4, deleted = 0
|
|
WHERE uuid = ?1",
|
|
rusqlite::params![row.uuid, row.name, row.revision, row.modified],
|
|
)?;
|
|
report.keywords_updated += 1;
|
|
}
|
|
MergeVerdict::DeletedByRemote => {
|
|
// Tombstone rather than DELETE, or a third device
|
|
// reintroduces the keyword through us.
|
|
tx.execute(
|
|
"INSERT INTO main.keyword_terms
|
|
(uuid, name, created, revision, modified, deleted)
|
|
VALUES (?1, ?2, ?3, ?4, ?5, 1)
|
|
ON CONFLICT(uuid) DO UPDATE SET
|
|
deleted = 1, revision = ?4, modified = ?5",
|
|
rusqlite::params![
|
|
row.uuid,
|
|
row.name,
|
|
row.created,
|
|
row.revision,
|
|
row.modified,
|
|
],
|
|
)?;
|
|
// **By name, not by identity.** This device may well have
|
|
// minted its own uuid for the same word before the two ever
|
|
// synced, in which case deleting by uuid would tombstone a
|
|
// row that nothing is assigned to and leave every
|
|
// photograph still carrying the word.
|
|
tx.execute("DELETE FROM main.keywords WHERE keyword = ?1", [&row.name])?;
|
|
report.keywords_deleted += 1;
|
|
}
|
|
}
|
|
}
|
|
|
|
// Two devices that each typed "Iceland" now hold two identities for one
|
|
// word. Collapse them before the assignments arrive, so the vocabulary
|
|
// the user sees after a sync has one row per word.
|
|
report.keywords_fused = crate::keywords::fuse_duplicates(tx)?;
|
|
}
|
|
|
|
// ---- assignments ------------------------------------------------------
|
|
//
|
|
// Set union, and the union is the whole point: FR-NC-9's principle applied
|
|
// to metadata rather than to edit nodes. Two devices that keyworded
|
|
// different frames "puffin" both keep their work, and neither loses it to
|
|
// whichever synced second.
|
|
//
|
|
// A removal therefore does not propagate — the remote's assignment simply
|
|
// reappears. That is the same trade-off collection membership makes above,
|
|
// and for the same reason: an unwanted keyword is removed again in a
|
|
// second, and a silently lost afternoon of keywording is not recoverable at
|
|
// all. Making removal propagate needs a tombstone per assignment, which is
|
|
// a schema change and a merge rule of its own.
|
|
//
|
|
// An incoming keyword lands on the local default version, so an image that
|
|
// has not got one yet would silently drop it. That is not a rare state: the
|
|
// invariant is maintained by a backfill on open, and a scan that ran since
|
|
// has added rows it has not covered. Losing a word the user typed on
|
|
// another device, for a bookkeeping reason, would be the wrong answer —
|
|
// this is idempotent and writes nothing once the invariant holds.
|
|
crate::rating::ensure_default_versions_within(tx)?;
|
|
|
|
// Two passes rather than one statement with an `OR`, because they resolve
|
|
// *different identities* for the same photograph and each wants its own
|
|
// index. See [`ASSIGN_BY_FILE_ID`] for why there are two at all.
|
|
for sql in [ASSIGN_BY_FILE_ID, ASSIGN_BY_CONTENT_HASH] {
|
|
report.keywords_assigned += tx.execute(sql, [])?;
|
|
}
|
|
|
|
Ok(())
|
|
}
|
|
|
|
/// Take assignments for images both devices know by the server's file id.
|
|
///
|
|
/// **Preferred over the content hash**, and the reason is that `content_hash`
|
|
/// is expensive — the schema says so, and it is computed only when import
|
|
/// dedup or a reconnect asks for it, which for most libraries is never. Keying
|
|
/// keywords on it alone would mean the union quietly did nothing for the
|
|
/// ordinary image, which is the exact failure this merge exists to prevent.
|
|
///
|
|
/// `oc:fileid` is the opposite: it is recorded for every image the moment a
|
|
/// remote scan sees it, it is stable across server-side renames and moves, and
|
|
/// it is the same integer on every device pointed at the same Nextcloud — which
|
|
/// is precisely the situation where two devices are keywording one library.
|
|
///
|
|
/// The keyword lands on the local image's **default version**, not on the
|
|
/// version it came from. Version uuids do not reconcile across devices in the
|
|
/// catalog: [`crate::rating::ensure_default_versions`] mints a fresh one per
|
|
/// device, so the same photograph's default versions have different uuids on
|
|
/// two machines and a uuid-keyed join would union nothing at all. Version
|
|
/// identity is reconciled in the *sidecar* (FR-NC-8), and until a merged
|
|
/// version arrives through there, the default version is both where
|
|
/// [`crate::keywords::assign`] writes and where the panel reads — so it is the
|
|
/// one place the word can land and be seen.
|
|
const ASSIGN_BY_FILE_ID: &str = "
|
|
INSERT OR IGNORE INTO main.keywords(version_id, keyword)
|
|
SELECT lv.id, rk.keyword
|
|
FROM remote_cat.keywords rk
|
|
JOIN remote_cat.versions rv ON rv.id = rk.version_id
|
|
JOIN remote_cat.remote rr ON rr.image_id = rv.image_id
|
|
JOIN main.remote lr ON lr.file_id = rr.file_id
|
|
JOIN main.versions lv ON lv.image_id = lr.image_id AND lv.is_default = 1
|
|
WHERE NOT EXISTS (SELECT 1 FROM main.keyword_terms t
|
|
WHERE t.name = rk.keyword AND t.deleted = 1)
|
|
OR EXISTS (SELECT 1 FROM main.keyword_terms t
|
|
WHERE t.name = rk.keyword AND t.deleted = 0)";
|
|
|
|
/// The same union for a library with no server behind it.
|
|
///
|
|
/// A local-only library has no `remote` rows at all, so [`ASSIGN_BY_FILE_ID`]
|
|
/// matches nothing and this is the only identity available — the same pair
|
|
/// collection membership uses, so a library where membership merges has
|
|
/// keywords that merge too.
|
|
const ASSIGN_BY_CONTENT_HASH: &str = "
|
|
INSERT OR IGNORE INTO main.keywords(version_id, keyword)
|
|
SELECT lv.id, rk.keyword
|
|
FROM remote_cat.keywords rk
|
|
JOIN remote_cat.versions rv ON rv.id = rk.version_id
|
|
JOIN remote_cat.images ri ON ri.id = rv.image_id
|
|
JOIN main.images li ON li.content_hash = ri.content_hash
|
|
JOIN main.versions lv ON lv.image_id = li.id AND lv.is_default = 1
|
|
WHERE ri.content_hash IS NOT NULL
|
|
AND (NOT EXISTS (SELECT 1 FROM main.keyword_terms t
|
|
WHERE t.name = rk.keyword AND t.deleted = 1)
|
|
OR EXISTS (SELECT 1 FROM main.keyword_terms t
|
|
WHERE t.name = rk.keyword AND t.deleted = 0))";
|
|
|
|
/// Whether an attached database holds a table of this name.
|
|
///
|
|
/// The schema name and the table name are both literals from this file, never
|
|
/// user text — but they are still bound rather than formatted where SQLite
|
|
/// allows it, because the habit is what keeps the one that eventually is user
|
|
/// text from being formatted by accident.
|
|
fn attached_has_table(conn: &Connection, schema: &str, table: &str) -> Result<bool, CatalogError> {
|
|
let sql =
|
|
format!("SELECT count(*) FROM {schema}.sqlite_master WHERE type = 'table' AND name = ?1");
|
|
let n: i64 = conn.query_row(&sql, [table], |r| r.get(0))?;
|
|
Ok(n > 0)
|
|
}
|
|
|
|
// ── people, and who the user said they are ────────────────────────────────
|
|
|
|
/// Merge people and the user's identity judgements from an attached catalog.
|
|
///
|
|
/// # Why this is here at all
|
|
///
|
|
/// Face *data* syncs as sealed shards ([`crate::face_shard`]) — boxes,
|
|
/// landmarks, embeddings, the run marker. What the shards deliberately do not
|
|
/// carry is who anybody **is**: the person rows, their names, and the
|
|
/// assignments joining the two. Those were supposed to travel in the catalog
|
|
/// snapshot, which is a whole-file copy and therefore does contain them — but
|
|
/// the snapshot is *merged*, not adopted, and this merge only ever looked at
|
|
/// collections and keywords. So a second device received every face and no
|
|
/// people at all, and drew an empty People screen over a full catalog.
|
|
///
|
|
/// # What travels, and what is recomputed
|
|
///
|
|
/// The rule this module already follows for the rest of the catalog: user
|
|
/// judgements travel, inference is rebuilt. Concretely (docs/dev/faces.md, and the
|
|
/// asymmetry `crate::faces` opens with):
|
|
///
|
|
/// - **People** — uuid, name, and whether the user set them aside. Merged by
|
|
/// uuid on `revision`, exactly as a collection is.
|
|
/// - **Confirmations** — the user said this face is this person.
|
|
/// - **Rejections** — the user said it is *not*, which is equally a fact and
|
|
/// is why re-clustering does not put it back.
|
|
/// - **The assignments inside an ignored group** — carried even though they are
|
|
/// only suggestions, because they are what anchors the ignore. Without them
|
|
/// a group set aside on one device reappears on the other, which is the same
|
|
/// fault that made "Not interested" not stick locally.
|
|
///
|
|
/// Ordinary suggestions are *not* carried. They are this pass's own output,
|
|
/// clustering is deterministic, and both devices hold the same embeddings — so
|
|
/// each recomputes them and arrives at the same answer. Shipping them would
|
|
/// double the merge for no new information.
|
|
///
|
|
/// # Faces have no cross-device identity, so one is derived
|
|
///
|
|
/// `faces.id` is a local row id and means nothing in another catalog; there is
|
|
/// no uuid to fall back on. What both devices *do* agree on is `oc:fileid` and
|
|
/// the box, so a remote face is matched to the local face on the same
|
|
/// photograph whose box overlaps it most, above a floor of 0.5 IoU.
|
|
///
|
|
/// That is not a new rule: it is the one
|
|
/// [`crate::faces::record_detections`] already uses to carry a confirmation
|
|
/// across a re-index, and it is loose on purpose — the question is "is this the
|
|
/// same face in the frame", not "is this the same rectangle", and a device
|
|
/// running a newer detector is entitled to have moved the box a little.
|
|
fn merge_people_within(tx: &Connection, report: &mut MergeReport) -> Result<(), CatalogError> {
|
|
// A remote written before faces existed has none of these tables, and one
|
|
// written before V10 has no `ignored`. Both are ordinary — `remote_is_
|
|
// mergeable` admits any catalog at or below this schema version — so they
|
|
// are probed for rather than assumed, and an absent one skips this half
|
|
// instead of aborting a merge that would otherwise have succeeded.
|
|
if !remote_has(tx, "people")? || !remote_has(tx, "faces")? {
|
|
return Ok(());
|
|
}
|
|
let ignored_col = remote_has_column(tx, "people", "ignored")?;
|
|
// Spelled into the SQL rather than branched around it: the two queries
|
|
// below would otherwise each need a second copy.
|
|
let ignored_sel = if ignored_col { "r.ignored" } else { "0" };
|
|
|
|
// ---- the people themselves -------------------------------------------
|
|
struct Incoming {
|
|
uuid: String,
|
|
name: String,
|
|
ignored: bool,
|
|
created: i64,
|
|
revision: i64,
|
|
modified: i64,
|
|
/// Resolved after every person exists, since a merge target can arrive
|
|
/// after the person redirecting to it.
|
|
merged_into_uuid: Option<String>,
|
|
verdict: MergeVerdict,
|
|
}
|
|
|
|
let rows: Vec<Incoming> = {
|
|
let mut stmt = tx.prepare(&format!(
|
|
"SELECT r.uuid, r.name, {ignored_sel}, r.created, r.revision, r.modified,
|
|
rm.uuid, l.revision, l.modified
|
|
FROM remote_cat.people r
|
|
LEFT JOIN main.people l ON l.uuid = r.uuid
|
|
LEFT JOIN remote_cat.people rm ON rm.id = r.merged_into"
|
|
))?;
|
|
let mapped = stmt.query_map([], |r| {
|
|
let revision: i64 = r.get(4)?;
|
|
let modified: i64 = r.get(5)?;
|
|
let local_rev: Option<i64> = r.get(7)?;
|
|
let local_mod: Option<i64> = r.get(8)?;
|
|
Ok(Incoming {
|
|
uuid: r.get(0)?,
|
|
name: r.get(1)?,
|
|
ignored: r.get(2)?,
|
|
created: r.get(3)?,
|
|
revision,
|
|
modified,
|
|
merged_into_uuid: r.get(6)?,
|
|
// People have no tombstone: a person is merged away rather
|
|
// than deleted, and `merged_into` is that redirect.
|
|
verdict: verdict(local_rev.zip(local_mod), (revision, modified), false),
|
|
})
|
|
})?;
|
|
mapped.collect::<Result<_, _>>()?
|
|
};
|
|
|
|
for p in &rows {
|
|
match p.verdict {
|
|
MergeVerdict::InsertedFromRemote => {
|
|
tx.execute(
|
|
"INSERT INTO people (uuid, name, ignored, created, revision, modified)
|
|
VALUES (?1, ?2, ?3, ?4, ?5, ?6)",
|
|
rusqlite::params![p.uuid, p.name, p.ignored, p.created, p.revision, p.modified],
|
|
)?;
|
|
report.people_inserted += 1;
|
|
}
|
|
MergeVerdict::UpdatedFromRemote | MergeVerdict::DeletedByRemote => {
|
|
tx.execute(
|
|
"UPDATE people
|
|
SET name = ?2, ignored = ?3, revision = ?4, modified = ?5
|
|
WHERE uuid = ?1",
|
|
rusqlite::params![p.uuid, p.name, p.ignored, p.revision, p.modified],
|
|
)?;
|
|
report.people_updated += 1;
|
|
}
|
|
MergeVerdict::KeptLocal => report.people_kept_local += 1,
|
|
}
|
|
}
|
|
|
|
// Redirects, once every person on both sides exists locally.
|
|
for p in rows.iter().filter(|p| p.merged_into_uuid.is_some()) {
|
|
if p.verdict == MergeVerdict::KeptLocal {
|
|
continue;
|
|
}
|
|
tx.execute(
|
|
"UPDATE people
|
|
SET merged_into = (SELECT id FROM people WHERE uuid = ?2)
|
|
WHERE uuid = ?1",
|
|
rusqlite::params![p.uuid, p.merged_into_uuid],
|
|
)?;
|
|
}
|
|
|
|
// ---- match the remote's faces onto this device's ----------------------
|
|
let face_map = match_faces(tx)?;
|
|
if face_map.is_empty() {
|
|
return Ok(());
|
|
}
|
|
|
|
// ---- confirmations, and the anchors under an ignored group -----------
|
|
{
|
|
let mut stmt = tx.prepare(&format!(
|
|
"SELECT fp.face_id, p.uuid, fp.probability, fp.confirmed
|
|
FROM remote_cat.face_person fp
|
|
JOIN remote_cat.people p ON p.id = fp.person_id
|
|
WHERE fp.confirmed = 1 OR {} = 1",
|
|
if ignored_col { "p.ignored" } else { "0" }
|
|
))?;
|
|
let incoming: Vec<(i64, String, f64, bool)> = stmt
|
|
.query_map([], |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)))?
|
|
.collect::<Result<_, _>>()?;
|
|
|
|
for (remote_face, uuid, probability, confirmed) in incoming {
|
|
let Some(&local_face) = face_map.get(&remote_face) else {
|
|
continue;
|
|
};
|
|
let person: Option<i64> = tx
|
|
.query_row("SELECT id FROM people WHERE uuid = ?1", [&uuid], |r| {
|
|
r.get(0)
|
|
})
|
|
.optional()?;
|
|
let Some(person) = person else { continue };
|
|
|
|
// A local confirmation is never overwritten, in either direction.
|
|
// Two devices confirming the same face as different people is a
|
|
// genuine disagreement and there is no revision on an assignment to
|
|
// settle it with; silently taking the remote's answer would let a
|
|
// sync undo something the user did here. It stays as it is, and the
|
|
// user can change it on the device they are looking at.
|
|
let locally_confirmed: bool = tx.query_row(
|
|
"SELECT EXISTS(SELECT 1 FROM face_person
|
|
WHERE face_id = ?1 AND confirmed = 1)",
|
|
[local_face],
|
|
|r| r.get(0),
|
|
)?;
|
|
if locally_confirmed {
|
|
report.faces_kept_local += 1;
|
|
continue;
|
|
}
|
|
|
|
// A rejection here outranks an assignment from elsewhere: it is
|
|
// this user's judgement about this pair, and re-suggesting what
|
|
// they pushed away is the behaviour that makes the feature feel
|
|
// broken.
|
|
let rejected: bool = tx.query_row(
|
|
"SELECT EXISTS(SELECT 1 FROM face_person_rejected
|
|
WHERE face_id = ?1 AND person_id = ?2)",
|
|
[local_face, person],
|
|
|r| r.get(0),
|
|
)?;
|
|
if rejected {
|
|
continue;
|
|
}
|
|
|
|
tx.execute(
|
|
"INSERT INTO face_person (face_id, person_id, probability, confirmed)
|
|
VALUES (?1, ?2, ?3, ?4)
|
|
ON CONFLICT(face_id) DO UPDATE SET
|
|
person_id = excluded.person_id,
|
|
probability = excluded.probability,
|
|
confirmed = excluded.confirmed",
|
|
rusqlite::params![local_face, person, probability, confirmed],
|
|
)?;
|
|
report.faces_assigned += 1;
|
|
}
|
|
}
|
|
|
|
// ---- rejections, as a set union --------------------------------------
|
|
{
|
|
let mut stmt = tx.prepare(
|
|
"SELECT fr.face_id, p.uuid
|
|
FROM remote_cat.face_person_rejected fr
|
|
JOIN remote_cat.people p ON p.id = fr.person_id",
|
|
)?;
|
|
let incoming: Vec<(i64, String)> = stmt
|
|
.query_map([], |r| Ok((r.get(0)?, r.get(1)?)))?
|
|
.collect::<Result<_, _>>()?;
|
|
|
|
for (remote_face, uuid) in incoming {
|
|
let Some(&local_face) = face_map.get(&remote_face) else {
|
|
continue;
|
|
};
|
|
let person: Option<i64> = tx
|
|
.query_row("SELECT id FROM people WHERE uuid = ?1", [&uuid], |r| {
|
|
r.get(0)
|
|
})
|
|
.optional()?;
|
|
let Some(person) = person else { continue };
|
|
|
|
let n = tx.execute(
|
|
"INSERT OR IGNORE INTO face_person_rejected (face_id, person_id)
|
|
VALUES (?1, ?2)",
|
|
[local_face, person],
|
|
)?;
|
|
report.faces_rejected += n;
|
|
|
|
// A rejection that lands on a face currently *suggested* to be
|
|
// that person has to take the suggestion with it, or the screen
|
|
// keeps offering exactly what the other device just refused.
|
|
tx.execute(
|
|
"DELETE FROM face_person
|
|
WHERE face_id = ?1 AND person_id = ?2 AND confirmed = 0",
|
|
[local_face, person],
|
|
)?;
|
|
}
|
|
}
|
|
|
|
Ok(())
|
|
}
|
|
|
|
/// Whether the attached remote holds a table.
|
|
fn remote_has(tx: &Connection, table: &str) -> Result<bool, CatalogError> {
|
|
Ok(tx.query_row(
|
|
"SELECT EXISTS(SELECT 1 FROM remote_cat.sqlite_master
|
|
WHERE type = 'table' AND name = ?1)",
|
|
[table],
|
|
|r| r.get::<_, bool>(0),
|
|
)?)
|
|
}
|
|
|
|
/// Whether a table in the attached remote holds a column.
|
|
fn remote_has_column(tx: &Connection, table: &str, column: &str) -> Result<bool, CatalogError> {
|
|
// `pragma_table_info` takes the schema as a second argument, which is the
|
|
// only way to ask about an attached database rather than the main one.
|
|
let mut stmt =
|
|
tx.prepare("SELECT 1 FROM pragma_table_info(?1, 'remote_cat') WHERE name = ?2")?;
|
|
Ok(stmt.exists(rusqlite::params![table, column])?)
|
|
}
|
|
|
|
/// Remote face row id to local face row id, by photograph and box overlap.
|
|
///
|
|
/// See [`merge_people_within`] for why a face has no shared identity and this
|
|
/// has to be derived. Faces are compared within an *embedder*
|
|
/// (`faces::embedder_of`), not within an exact pipeline id: two detectors in
|
|
/// front of the same embedder draw boxes around the same faces, and a
|
|
/// confirmation made on one device's box is about the face, not the
|
|
/// rectangle — the same judgement `faces::record_detections` makes when it
|
|
/// carries a confirmation across a re-detection. Keying on the exact id was
|
|
/// what let a detector change strand every name on the device that made it.
|
|
fn match_faces(tx: &Connection) -> Result<std::collections::HashMap<i64, i64>, CatalogError> {
|
|
/// Loose on purpose — "the same face in the frame", not "the same
|
|
/// rectangle". The figure `record_detections` uses for the same job.
|
|
const MIN_IOU: f32 = 0.5;
|
|
|
|
type Boxed = (i64, f32, f32, f32, f32);
|
|
|
|
// Local faces, grouped by the photograph's cross-device id.
|
|
let mut local: std::collections::HashMap<(i64, String), Vec<Boxed>> =
|
|
std::collections::HashMap::new();
|
|
{
|
|
let mut stmt = tx.prepare(
|
|
"SELECT f.id, r.file_id, f.model_id, f.x, f.y, f.w, f.h
|
|
FROM main.faces f
|
|
JOIN main.remote r ON r.image_id = f.image_id
|
|
WHERE r.file_id IS NOT NULL",
|
|
)?;
|
|
let rows = stmt.query_map([], |r| {
|
|
Ok((
|
|
r.get::<_, i64>(1)?,
|
|
r.get::<_, String>(2)?,
|
|
(
|
|
r.get::<_, i64>(0)?,
|
|
r.get::<_, f64>(3)? as f32,
|
|
r.get::<_, f64>(4)? as f32,
|
|
r.get::<_, f64>(5)? as f32,
|
|
r.get::<_, f64>(6)? as f32,
|
|
),
|
|
))
|
|
})?;
|
|
for row in rows {
|
|
let (file_id, model, boxed) = row?;
|
|
let embedder = crate::faces::embedder_of(&model).to_string();
|
|
local.entry((file_id, embedder)).or_default().push(boxed);
|
|
}
|
|
}
|
|
if local.is_empty() {
|
|
return Ok(Default::default());
|
|
}
|
|
|
|
let mut map = std::collections::HashMap::new();
|
|
let mut stmt = tx.prepare(
|
|
"SELECT f.id, r.file_id, f.model_id, f.x, f.y, f.w, f.h
|
|
FROM remote_cat.faces f
|
|
JOIN remote_cat.remote r ON r.image_id = f.image_id
|
|
WHERE r.file_id IS NOT NULL",
|
|
)?;
|
|
let rows = stmt.query_map([], |r| {
|
|
Ok((
|
|
r.get::<_, i64>(0)?,
|
|
r.get::<_, i64>(1)?,
|
|
r.get::<_, String>(2)?,
|
|
(
|
|
r.get::<_, f64>(3)? as f32,
|
|
r.get::<_, f64>(4)? as f32,
|
|
r.get::<_, f64>(5)? as f32,
|
|
r.get::<_, f64>(6)? as f32,
|
|
),
|
|
))
|
|
})?;
|
|
|
|
for row in rows {
|
|
let (remote_id, file_id, model, rbox) = row?;
|
|
let embedder = crate::faces::embedder_of(&model).to_string();
|
|
let Some(candidates) = local.get(&(file_id, embedder)) else {
|
|
continue;
|
|
};
|
|
let best = candidates
|
|
.iter()
|
|
.map(|&(id, x, y, w, h)| (id, iou(rbox, (x, y, w, h))))
|
|
.filter(|&(_, score)| score >= MIN_IOU)
|
|
.max_by(|a, b| a.1.total_cmp(&b.1));
|
|
if let Some((local_id, _)) = best {
|
|
map.insert(remote_id, local_id);
|
|
}
|
|
}
|
|
Ok(map)
|
|
}
|
|
|
|
/// Intersection over union of two `(x, y, w, h)` boxes.
|
|
fn iou(a: (f32, f32, f32, f32), b: (f32, f32, f32, f32)) -> f32 {
|
|
let x0 = a.0.max(b.0);
|
|
let y0 = a.1.max(b.1);
|
|
let x1 = (a.0 + a.2).min(b.0 + b.2);
|
|
let y1 = (a.1 + a.3).min(b.1 + b.3);
|
|
let inter = (x1 - x0).max(0.0) * (y1 - y0).max(0.0);
|
|
let union = a.2 * a.3 + b.2 * b.3 - inter;
|
|
if union <= 0.0 {
|
|
0.0
|
|
} else {
|
|
inter / union
|
|
}
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
use crate::schema;
|
|
|
|
#[test]
|
|
fn a_collection_we_lack_is_taken_from_remote() {
|
|
assert_eq!(
|
|
verdict(None, (1, 100), false),
|
|
MergeVerdict::InsertedFromRemote
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn higher_remote_revision_wins() {
|
|
assert_eq!(
|
|
verdict(Some((3, 100)), (4, 50), false),
|
|
MergeVerdict::UpdatedFromRemote
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_skewed_clock_cannot_beat_a_higher_local_revision() {
|
|
// The remote's timestamp is far in the future, but it has seen fewer
|
|
// edits. Revision decides, so the skewed device does not silently
|
|
// overwrite real work.
|
|
assert_eq!(
|
|
verdict(Some((9, 100)), (2, 999_999), false),
|
|
MergeVerdict::KeptLocal
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn equal_revisions_break_on_timestamp() {
|
|
assert_eq!(
|
|
verdict(Some((3, 100)), (3, 200), false),
|
|
MergeVerdict::UpdatedFromRemote
|
|
);
|
|
assert_eq!(
|
|
verdict(Some((3, 200)), (3, 100), false),
|
|
MergeVerdict::KeptLocal
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn an_identical_collection_is_stable() {
|
|
// Merging twice must not oscillate or report spurious changes.
|
|
assert_eq!(
|
|
verdict(Some((3, 100)), (3, 100), false),
|
|
MergeVerdict::KeptLocal
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn deletion_competes_on_revision_like_any_other_edit() {
|
|
// Remote deleted it at revision 5; we renamed it at revision 4. The
|
|
// deletion is newer, so it wins.
|
|
assert_eq!(
|
|
verdict(Some((4, 100)), (5, 100), true),
|
|
MergeVerdict::DeletedByRemote
|
|
);
|
|
// But a stale deletion does not undo a newer local edit.
|
|
assert_eq!(
|
|
verdict(Some((6, 100)), (5, 100), true),
|
|
MergeVerdict::KeptLocal
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_tombstone_for_something_we_never_had_is_recorded() {
|
|
// Otherwise this device could reintroduce the collection to a third.
|
|
assert_eq!(verdict(None, (2, 100), true), MergeVerdict::DeletedByRemote);
|
|
}
|
|
|
|
// ---- integration over two real catalogs ------------------------------
|
|
|
|
/// A fresh device takes the capture dates a peer's sweep read, matched by
|
|
/// `oc:fileid`, and never overwrites a date it read for itself.
|
|
#[test]
|
|
fn capture_metadata_arrives_for_undated_images_only() {
|
|
let c = two_catalogs();
|
|
// Three photographs on both devices: 1 undated here and dated there;
|
|
// 2 dated on both, differently; 3 undated on both.
|
|
for id in 1..=3 {
|
|
add_image_without_hash(&c, "main", id);
|
|
add_image_without_hash(&c, "remote_cat", id + 10);
|
|
add_remote_id(&c, "main", id, 100 + id);
|
|
add_remote_id(&c, "remote_cat", id + 10, 100 + id);
|
|
}
|
|
c.execute(
|
|
"UPDATE remote_cat.images
|
|
SET captured_at = 1000, captured_offset = 60, camera = 'X', metadata_state = 2
|
|
WHERE id = 11",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
c.execute(
|
|
"UPDATE remote_cat.images SET captured_at = 2000, metadata_state = 2 WHERE id = 12",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
c.execute(
|
|
"UPDATE main.images SET captured_at = 2222, metadata_state = 2 WHERE id = 2",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
let report = merge_metadata(&c).unwrap();
|
|
assert_eq!(report.metadata_adopted, 1);
|
|
|
|
let row = |id: i64| -> (Option<i64>, Option<i64>, Option<String>, i64) {
|
|
c.query_row(
|
|
"SELECT captured_at, captured_offset, camera, metadata_state
|
|
FROM main.images WHERE id = ?1",
|
|
[id],
|
|
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)),
|
|
)
|
|
.unwrap()
|
|
};
|
|
assert_eq!(row(1), (Some(1000), Some(60), Some("X".into()), 2));
|
|
assert_eq!(row(2), (Some(2222), None, None, 2));
|
|
assert_eq!(row(3), (None, None, None, 0));
|
|
|
|
// Idempotent: a second pass finds nothing left to take.
|
|
assert_eq!(merge_metadata(&c).unwrap().metadata_adopted, 0);
|
|
}
|
|
|
|
fn two_catalogs() -> Connection {
|
|
attached_remote(schema::for_attached("remote_cat"))
|
|
}
|
|
|
|
/// The same pair, but with the remote stopped at v1 — a device running a
|
|
/// build from before keywords had identities.
|
|
fn two_catalogs_with_a_v1_remote() -> Connection {
|
|
attached_remote(schema::v1_for_attached("remote_cat"))
|
|
}
|
|
|
|
fn attached_remote(remote_schema: String) -> Connection {
|
|
let c = Connection::open_in_memory().unwrap();
|
|
schema::configure(&c).unwrap();
|
|
schema::migrate(&c).unwrap();
|
|
// A second in-memory database standing in for the downloaded remote.
|
|
c.execute_batch("ATTACH ':memory:' AS remote_cat").unwrap();
|
|
c.execute_batch(&remote_schema).unwrap();
|
|
c
|
|
}
|
|
|
|
/// An image with no content hash — which is every image in a library that
|
|
/// has only ever been scanned.
|
|
fn add_image_without_hash(c: &Connection, db: &str, id: i64) {
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.roots(id, kind, label) VALUES (1, 'remote', 'lib')
|
|
ON CONFLICT(id) DO NOTHING"
|
|
),
|
|
[],
|
|
)
|
|
.unwrap();
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.images(id, root_id, source_ref, added_at)
|
|
VALUES (?1, 1, ?2, 0)"
|
|
),
|
|
rusqlite::params![id, format!("img{id}.CR3")],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
fn count(c: &Connection, sql: &str) -> i64 {
|
|
c.query_row(sql, [], |r| r.get(0)).unwrap()
|
|
}
|
|
|
|
/// Give an image the `oc:fileid` a remote scan records for it.
|
|
///
|
|
/// What every image in a real synced library has, and what none of them
|
|
/// has a content hash for.
|
|
fn add_remote_id(c: &Connection, db: &str, image_id: i64, file_id: i64) {
|
|
c.execute(
|
|
&format!("INSERT INTO {db}.remote(image_id, file_id) VALUES (?1, ?2)"),
|
|
rusqlite::params![image_id, file_id],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
fn add_image(c: &Connection, db: &str, id: i64, hash: &str) {
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.roots(id, kind, label) VALUES (1, 'local', 'r')
|
|
ON CONFLICT(id) DO NOTHING"
|
|
),
|
|
[],
|
|
)
|
|
.unwrap();
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.images(id, root_id, source_ref, content_hash, added_at)
|
|
VALUES (?1, 1, ?2, ?3, 0)"
|
|
),
|
|
rusqlite::params![id, format!("img{id}.CR3"), hash],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
fn add_collection(c: &Connection, db: &str, id: i64, uuid: &str, name: &str, rev: i64) {
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.collections(id, uuid, name, kind, created, revision, modified)
|
|
VALUES (?1, ?2, ?3, 0, 0, ?4, ?4)"
|
|
),
|
|
rusqlite::params![id, uuid, name, rev],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
/// The parent's uuid for `uuid`, or None if it sits at the top level.
|
|
fn parent_of(c: &Connection, uuid: &str) -> Option<String> {
|
|
c.query_row(
|
|
"SELECT p.uuid FROM main.collections ch
|
|
JOIN main.collections p ON p.id = ch.parent_id
|
|
WHERE ch.uuid = ?1",
|
|
[uuid],
|
|
|r| r.get(0),
|
|
)
|
|
.ok()
|
|
}
|
|
|
|
#[test]
|
|
fn a_nested_collection_arrives_still_nested() {
|
|
// The tree used to flatten on every sync: `parent_id` is a local row id,
|
|
// and the merge wrote NULL rather than translating it through the uuid.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "remote_cat", 1, "uuid-holidays", "holidays", 1);
|
|
add_collection(&c, "remote_cat", 2, "uuid-arosa", "arosa", 1);
|
|
c.execute(
|
|
"UPDATE remote_cat.collections SET parent_id = 1 WHERE id = 2",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
merge_collections(&c).unwrap();
|
|
|
|
assert_eq!(
|
|
parent_of(&c, "uuid-arosa").as_deref(),
|
|
Some("uuid-holidays"),
|
|
"a sub-collection must not be promoted to the top level by a merge"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_parent_arriving_after_its_child_still_adopts_it() {
|
|
// Row order is whatever the query returns, so the child can be inserted
|
|
// first. Parentage is applied in a second pass for exactly this reason.
|
|
let c = two_catalogs();
|
|
// Child has the lower id, so it is seen before the parent exists.
|
|
add_collection(&c, "remote_cat", 1, "uuid-child", "arosa", 1);
|
|
add_collection(&c, "remote_cat", 2, "uuid-parent", "holidays", 1);
|
|
c.execute(
|
|
"UPDATE remote_cat.collections SET parent_id = 2 WHERE id = 1",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
merge_collections(&c).unwrap();
|
|
|
|
assert_eq!(
|
|
parent_of(&c, "uuid-child").as_deref(),
|
|
Some("uuid-parent"),
|
|
"resolution must not depend on the order rows happen to arrive in"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn disagreeing_devices_cannot_close_a_parent_cycle() {
|
|
// Local holds A above B; the remote holds B above A and wins on
|
|
// revision. Applying that blindly makes each the other's parent, and
|
|
// every tree walk then spins.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "uuid-a", "A", 1);
|
|
add_collection(&c, "main", 2, "uuid-b", "B", 1);
|
|
c.execute("UPDATE main.collections SET parent_id = 1 WHERE id = 2", [])
|
|
.unwrap();
|
|
|
|
// Remote: A under B, at a higher revision so it is the winner.
|
|
add_collection(&c, "remote_cat", 1, "uuid-a", "A", 9);
|
|
add_collection(&c, "remote_cat", 2, "uuid-b", "B", 9);
|
|
c.execute(
|
|
"UPDATE remote_cat.collections SET parent_id = 2 WHERE id = 1",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
merge_collections(&c).unwrap();
|
|
|
|
let a = parent_of(&c, "uuid-a");
|
|
let b = parent_of(&c, "uuid-b");
|
|
assert!(
|
|
!(a.as_deref() == Some("uuid-b") && b.as_deref() == Some("uuid-a")),
|
|
"merge closed a cycle: A parented under B and B under A"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn disjoint_collections_from_two_devices_both_survive() {
|
|
// The property the whole design exists for: neither device loses work.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "uuid-local", "Iceland", 1);
|
|
add_collection(&c, "remote_cat", 1, "uuid-remote", "Portugal", 1);
|
|
|
|
let report = merge_collections(&c).unwrap();
|
|
assert_eq!(report.inserted, 1);
|
|
|
|
let names: Vec<String> = c
|
|
.prepare("SELECT name FROM main.collections ORDER BY name")
|
|
.unwrap()
|
|
.query_map([], |r| r.get(0))
|
|
.unwrap()
|
|
.collect::<Result<_, _>>()
|
|
.unwrap();
|
|
assert_eq!(names, vec!["Iceland", "Portugal"]);
|
|
}
|
|
|
|
#[test]
|
|
fn membership_unions_rather_than_replacing() {
|
|
// Two devices each added a different image to the same collection.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "shared", "Trip", 1);
|
|
add_collection(&c, "remote_cat", 1, "shared", "Trip", 1);
|
|
add_image(&c, "main", 1, "hash-a");
|
|
add_image(&c, "main", 2, "hash-b");
|
|
add_image(&c, "remote_cat", 1, "hash-b");
|
|
|
|
c.execute(
|
|
"INSERT INTO main.collection_members(collection_id, image_id, added) VALUES (1, 1, 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
c.execute(
|
|
"INSERT INTO remote_cat.collection_members(collection_id, image_id, added)
|
|
VALUES (1, 1, 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
merge_collections(&c).unwrap();
|
|
|
|
let n: i64 = c
|
|
.query_row("SELECT count(*) FROM main.collection_members", [], |r| {
|
|
r.get(0)
|
|
})
|
|
.unwrap();
|
|
assert_eq!(n, 2, "both devices' additions survive");
|
|
}
|
|
|
|
#[test]
|
|
fn a_library_that_never_computed_a_hash_still_merges_membership() {
|
|
// The bug this replaces. `content_hash` is computed only by import
|
|
// dedup or a reconnect — never by a scan — so a synced library has one
|
|
// for no image at all. Keying membership on it alone meant collections
|
|
// arrived with their names and none of their contents, on every device,
|
|
// for every user.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "shared", "Trip", 1);
|
|
add_collection(&c, "remote_cat", 1, "shared", "Trip", 1);
|
|
|
|
// No hashes anywhere, and a file id for both — a real library.
|
|
add_image_without_hash(&c, "main", 77);
|
|
add_image_without_hash(&c, "remote_cat", 3);
|
|
add_remote_id(&c, "main", 77, 90210);
|
|
add_remote_id(&c, "remote_cat", 3, 90210);
|
|
|
|
c.execute(
|
|
"INSERT INTO remote_cat.collection_members(collection_id, image_id, added)
|
|
VALUES (1, 3, 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
let report = merge_collections(&c).unwrap();
|
|
assert_eq!(report.members_added, 1);
|
|
|
|
let img: i64 = c
|
|
.query_row("SELECT image_id FROM main.collection_members", [], |r| {
|
|
r.get(0)
|
|
})
|
|
.unwrap();
|
|
assert_eq!(img, 77, "resolved through the server's file id");
|
|
}
|
|
|
|
#[test]
|
|
fn a_different_photograph_on_the_server_is_not_adopted() {
|
|
// The file id is an identity, so two different ids must not join.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "shared", "Trip", 1);
|
|
add_collection(&c, "remote_cat", 1, "shared", "Trip", 1);
|
|
add_image_without_hash(&c, "main", 77);
|
|
add_image_without_hash(&c, "remote_cat", 3);
|
|
add_remote_id(&c, "main", 77, 111);
|
|
add_remote_id(&c, "remote_cat", 3, 222);
|
|
|
|
c.execute(
|
|
"INSERT INTO remote_cat.collection_members(collection_id, image_id, added)
|
|
VALUES (1, 3, 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
let report = merge_collections(&c).unwrap();
|
|
assert_eq!(report.members_added, 0);
|
|
assert_eq!(count(&c, "SELECT COUNT(*) FROM main.collection_members"), 0);
|
|
}
|
|
|
|
#[test]
|
|
fn an_image_identified_both_ways_is_added_once() {
|
|
// Both statements run, and their overlap must be free rather than a
|
|
// constraint violation or a double count.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "shared", "Trip", 1);
|
|
add_collection(&c, "remote_cat", 1, "shared", "Trip", 1);
|
|
add_image(&c, "main", 77, "same-photo");
|
|
add_image(&c, "remote_cat", 3, "same-photo");
|
|
add_remote_id(&c, "main", 77, 90210);
|
|
add_remote_id(&c, "remote_cat", 3, 90210);
|
|
|
|
c.execute(
|
|
"INSERT INTO remote_cat.collection_members(collection_id, image_id, added)
|
|
VALUES (1, 3, 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
merge_collections(&c).unwrap();
|
|
assert_eq!(count(&c, "SELECT COUNT(*) FROM main.collection_members"), 1);
|
|
}
|
|
|
|
#[test]
|
|
fn membership_maps_across_devices_by_content_hash() {
|
|
// The same photograph carries different integer ids on each device.
|
|
// Keying on the id would attach the wrong image.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "shared", "Trip", 1);
|
|
add_collection(&c, "remote_cat", 1, "shared", "Trip", 1);
|
|
add_image(&c, "main", 77, "same-photo");
|
|
add_image(&c, "remote_cat", 3, "same-photo");
|
|
|
|
c.execute(
|
|
"INSERT INTO remote_cat.collection_members(collection_id, image_id, added)
|
|
VALUES (1, 3, 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
merge_collections(&c).unwrap();
|
|
|
|
let img: i64 = c
|
|
.query_row("SELECT image_id FROM main.collection_members", [], |r| {
|
|
r.get(0)
|
|
})
|
|
.unwrap();
|
|
assert_eq!(img, 77, "resolved to the local id for the same photo");
|
|
}
|
|
|
|
#[test]
|
|
fn an_image_we_do_not_have_yet_is_skipped_not_errored() {
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "shared", "Trip", 1);
|
|
add_collection(&c, "remote_cat", 1, "shared", "Trip", 1);
|
|
add_image(&c, "remote_cat", 1, "not-here-yet");
|
|
c.execute(
|
|
"INSERT INTO remote_cat.collection_members(collection_id, image_id, added)
|
|
VALUES (1, 1, 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
let report = merge_collections(&c).unwrap();
|
|
assert_eq!(report.members_added, 0);
|
|
// It joins on a later merge, once a scan has catalogued the file.
|
|
}
|
|
|
|
#[test]
|
|
fn a_remote_deletion_does_not_resurrect_via_membership() {
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "doomed", "Old", 1);
|
|
add_image(&c, "main", 1, "hash-a");
|
|
add_image(&c, "remote_cat", 1, "hash-a");
|
|
c.execute(
|
|
"INSERT INTO remote_cat.collections(id, uuid, name, kind, created, revision, modified, deleted)
|
|
VALUES (1, 'doomed', 'Old', 0, 0, 5, 5, 1)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
c.execute(
|
|
"INSERT INTO remote_cat.collection_members(collection_id, image_id, added)
|
|
VALUES (1, 1, 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
let report = merge_collections(&c).unwrap();
|
|
assert_eq!(report.deleted, 1);
|
|
let n: i64 = c
|
|
.query_row("SELECT count(*) FROM main.collection_members", [], |r| {
|
|
r.get(0)
|
|
})
|
|
.unwrap();
|
|
assert_eq!(n, 0, "membership must not repopulate a deleted collection");
|
|
}
|
|
|
|
#[test]
|
|
fn merging_twice_changes_nothing_the_second_time() {
|
|
let c = two_catalogs();
|
|
add_collection(&c, "remote_cat", 1, "uuid-r", "Portugal", 1);
|
|
|
|
let first = merge_collections(&c).unwrap();
|
|
assert!(first.local_changed());
|
|
|
|
let second = merge_collections(&c).unwrap();
|
|
assert!(!second.local_changed(), "merge must be idempotent");
|
|
}
|
|
|
|
// ---- keywords --------------------------------------------------------
|
|
|
|
/// Give an image a default version, as every write path assumes it has.
|
|
fn add_version(c: &Connection, db: &str, image: i64, uuid: &str) -> i64 {
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.versions(image_id, uuid, name, is_default)
|
|
VALUES (?1, ?2, 'Default', 1)"
|
|
),
|
|
rusqlite::params![image, uuid],
|
|
)
|
|
.unwrap();
|
|
c.last_insert_rowid()
|
|
}
|
|
|
|
/// Put a word on an image's default version, creating the version.
|
|
///
|
|
/// The version uuid is derived from the database *and* the image, so the
|
|
/// two catalogs never accidentally agree on one — which is the real
|
|
/// situation, and the reason the assignment union cannot key on it.
|
|
fn keyword(c: &Connection, db: &str, image: i64, word: &str) {
|
|
let existing: Option<i64> = c
|
|
.query_row(
|
|
&format!("SELECT id FROM {db}.versions WHERE image_id = ?1 AND is_default = 1"),
|
|
[image],
|
|
|r| r.get(0),
|
|
)
|
|
.ok();
|
|
let version =
|
|
existing.unwrap_or_else(|| add_version(c, db, image, &format!("v-{db}-{image}")));
|
|
c.execute(
|
|
&format!("INSERT OR IGNORE INTO {db}.keywords(version_id, keyword) VALUES (?1, ?2)"),
|
|
rusqlite::params![version, word],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
fn add_term(c: &Connection, db: &str, uuid: &str, name: &str, rev: i64, deleted: i64) {
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.keyword_terms(uuid, name, created, revision, modified, deleted)
|
|
VALUES (?1, ?2, 0, ?3, ?3, ?4)"
|
|
),
|
|
rusqlite::params![uuid, name, rev, deleted],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
/// Map an image to a server file id, as a remote scan does.
|
|
fn add_file_id(c: &Connection, db: &str, image: i64, file_id: i64) {
|
|
c.execute(
|
|
&format!("INSERT INTO {db}.remote(image_id, file_id) VALUES (?1, ?2)"),
|
|
rusqlite::params![image, file_id],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
/// Every word on an image locally, sorted.
|
|
fn words_on(c: &Connection, image: i64) -> Vec<String> {
|
|
let mut stmt = c
|
|
.prepare(
|
|
"SELECT DISTINCT k.keyword FROM main.keywords k
|
|
JOIN main.versions v ON v.id = k.version_id
|
|
WHERE v.image_id = ?1 ORDER BY k.keyword",
|
|
)
|
|
.unwrap();
|
|
let rows = stmt.query_map([image], |r| r.get(0)).unwrap();
|
|
rows.collect::<Result<Vec<_>, _>>().unwrap()
|
|
}
|
|
|
|
fn live_terms(c: &Connection) -> Vec<String> {
|
|
let mut stmt = c
|
|
.prepare("SELECT name FROM main.keyword_terms WHERE deleted = 0 ORDER BY name")
|
|
.unwrap();
|
|
let rows = stmt.query_map([], |r| r.get(0)).unwrap();
|
|
rows.collect::<Result<Vec<_>, _>>().unwrap()
|
|
}
|
|
|
|
#[test]
|
|
fn two_devices_keywording_different_photographs_both_survive() {
|
|
// FR-NC-9's principle applied to metadata: disjoint work merges to the
|
|
// union, and neither device loses an afternoon to whoever synced last.
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_image(&c, db, 1, "hash-a");
|
|
add_image(&c, db, 2, "hash-b");
|
|
}
|
|
keyword(&c, "main", 1, "puffin");
|
|
keyword(&c, "remote_cat", 2, "gannet");
|
|
|
|
merge_keywords(&c).unwrap();
|
|
|
|
assert_eq!(words_on(&c, 1), ["puffin"]);
|
|
assert_eq!(words_on(&c, 2), ["gannet"]);
|
|
}
|
|
|
|
#[test]
|
|
fn two_devices_keywording_one_photograph_keep_both_words() {
|
|
// The case the union is really for: the same frame, two different
|
|
// words, and last-writer-wins would silently drop one of them.
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_image(&c, db, 1, "hash-a");
|
|
}
|
|
keyword(&c, "main", 1, "puffin");
|
|
keyword(&c, "remote_cat", 1, "Iceland");
|
|
|
|
let report = merge_keywords(&c).unwrap();
|
|
assert_eq!(report.keywords_assigned, 1);
|
|
assert_eq!(words_on(&c, 1), ["Iceland", "puffin"]);
|
|
}
|
|
|
|
#[test]
|
|
fn a_word_both_devices_already_had_is_not_duplicated() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_image(&c, db, 1, "hash-a");
|
|
keyword(&c, db, 1, "puffin");
|
|
}
|
|
|
|
let report = merge_keywords(&c).unwrap();
|
|
assert_eq!(report.keywords_assigned, 0);
|
|
assert_eq!(words_on(&c, 1), ["puffin"]);
|
|
}
|
|
|
|
#[test]
|
|
fn keywords_reach_an_image_the_server_names_but_no_one_has_hashed() {
|
|
// `content_hash` is computed only when import dedup or a reconnect asks
|
|
// for it, so for most images it is NULL — and a union keyed on it alone
|
|
// would quietly do nothing for the ordinary photograph. The file id is
|
|
// recorded by every remote scan, which is exactly the situation where
|
|
// two devices are keywording one library.
|
|
let c = two_catalogs();
|
|
c.execute(
|
|
"INSERT INTO main.roots(id, kind, label) VALUES (1, 'remote', 'r')",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
c.execute(
|
|
"INSERT INTO remote_cat.roots(id, kind, label) VALUES (1, 'remote', 'r')",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
// Different row ids for one photograph, and no hash on either side.
|
|
c.execute(
|
|
"INSERT INTO main.images(id, root_id, source_ref, added_at)
|
|
VALUES (77, 1, 'IMG_1.CR3', 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
c.execute(
|
|
"INSERT INTO remote_cat.images(id, root_id, source_ref, added_at)
|
|
VALUES (3, 1, 'IMG_1.CR3', 0)",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
add_file_id(&c, "main", 77, 9001);
|
|
add_file_id(&c, "remote_cat", 3, 9001);
|
|
add_version(&c, "main", 77, "v-main");
|
|
keyword(&c, "remote_cat", 3, "puffin");
|
|
|
|
merge_keywords(&c).unwrap();
|
|
assert_eq!(words_on(&c, 77), ["puffin"]);
|
|
}
|
|
|
|
#[test]
|
|
fn keywords_map_across_devices_by_content_hash_where_there_is_no_server() {
|
|
// A local-only library has no `remote` rows at all, so the hash is the
|
|
// only identity available — and it is the one membership already uses.
|
|
let c = two_catalogs();
|
|
add_image(&c, "main", 77, "same-photo");
|
|
add_image(&c, "remote_cat", 3, "same-photo");
|
|
add_version(&c, "main", 77, "v-main");
|
|
keyword(&c, "remote_cat", 3, "puffin");
|
|
|
|
merge_keywords(&c).unwrap();
|
|
assert_eq!(words_on(&c, 77), ["puffin"]);
|
|
}
|
|
|
|
#[test]
|
|
fn the_vocabulary_merges_by_uuid_and_a_skewed_clock_cannot_win() {
|
|
let c = two_catalogs();
|
|
add_term(&c, "main", "u-1", "Iceland", 9, 0);
|
|
add_term(&c, "remote_cat", "u-1", "iceland", 2, 0);
|
|
add_term(&c, "remote_cat", "u-2", "puffin", 1, 0);
|
|
|
|
let report = merge_keywords(&c).unwrap();
|
|
assert_eq!(report.keywords_kept_local, 1);
|
|
assert_eq!(report.keywords_inserted, 1);
|
|
assert_eq!(live_terms(&c), ["Iceland", "puffin"]);
|
|
}
|
|
|
|
#[test]
|
|
fn a_remote_rename_moves_this_device_s_assignments_too() {
|
|
// The failure this exists to stop: the vocabulary shows the corrected
|
|
// spelling and the search still only finds the old one.
|
|
let c = two_catalogs();
|
|
add_image(&c, "main", 1, "hash-a");
|
|
add_term(&c, "main", "u-1", "Icland", 1, 0);
|
|
keyword(&c, "main", 1, "Icland");
|
|
add_term(&c, "remote_cat", "u-1", "Iceland", 4, 0);
|
|
|
|
let report = merge_keywords(&c).unwrap();
|
|
assert_eq!(report.keywords_updated, 1);
|
|
assert_eq!(live_terms(&c), ["Iceland"]);
|
|
assert_eq!(words_on(&c, 1), ["Iceland"]);
|
|
}
|
|
|
|
#[test]
|
|
fn a_remote_deletion_takes_the_word_off_every_photograph() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_image(&c, db, 1, "hash-a");
|
|
}
|
|
add_term(&c, "main", "u-1", "blurry", 1, 0);
|
|
keyword(&c, "main", 1, "blurry");
|
|
add_term(&c, "remote_cat", "u-1", "blurry", 5, 1);
|
|
|
|
let report = merge_keywords(&c).unwrap();
|
|
assert_eq!(report.keywords_deleted, 1);
|
|
assert!(live_terms(&c).is_empty());
|
|
assert!(words_on(&c, 1).is_empty());
|
|
}
|
|
|
|
#[test]
|
|
fn a_deletion_is_not_undone_by_the_union_on_the_same_pass() {
|
|
// The remote deleted the word *and* still carries assignments for it —
|
|
// it has not yet had the chance to sweep them, or a third device put
|
|
// them there. Without the tombstone filter the union would put the word
|
|
// straight back on the photograph the deletion had just cleared.
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_image(&c, db, 1, "hash-a");
|
|
}
|
|
add_term(&c, "main", "u-1", "blurry", 1, 0);
|
|
keyword(&c, "main", 1, "blurry");
|
|
add_term(&c, "remote_cat", "u-1", "blurry", 5, 1);
|
|
keyword(&c, "remote_cat", 1, "blurry");
|
|
|
|
merge_keywords(&c).unwrap();
|
|
assert!(words_on(&c, 1).is_empty(), "a deleted keyword came back");
|
|
}
|
|
|
|
#[test]
|
|
fn a_deletion_lands_even_when_the_two_devices_minted_different_uuids() {
|
|
// Both typed "blurry" before they ever synced, so this device's row has
|
|
// a uuid the remote has never heard of. Deleting by identity would
|
|
// tombstone nothing and leave every photograph still carrying the word.
|
|
let c = two_catalogs();
|
|
add_image(&c, "main", 1, "hash-a");
|
|
add_term(&c, "main", "mine", "blurry", 1, 0);
|
|
keyword(&c, "main", 1, "blurry");
|
|
add_term(&c, "remote_cat", "theirs", "blurry", 5, 1);
|
|
|
|
merge_keywords(&c).unwrap();
|
|
assert!(words_on(&c, 1).is_empty());
|
|
}
|
|
|
|
#[test]
|
|
fn two_devices_that_typed_one_word_end_up_with_one_keyword() {
|
|
// Neither is wrong until they meet, which is why the name carries no
|
|
// unique index — a constraint would abort the merge at this moment.
|
|
let c = two_catalogs();
|
|
add_term(&c, "main", "zzzz", "Iceland", 3, 0);
|
|
add_term(&c, "remote_cat", "aaaa", "Iceland", 1, 0);
|
|
|
|
let report = merge_keywords(&c).unwrap();
|
|
assert_eq!(report.keywords_fused, 1);
|
|
assert_eq!(live_terms(&c), ["Iceland"]);
|
|
|
|
let survivor: String = c
|
|
.query_row("SELECT uuid FROM main.keyword_terms", [], |r| r.get(0))
|
|
.unwrap();
|
|
assert_eq!(
|
|
survivor, "aaaa",
|
|
"both devices must pick the same survivor without asking each other"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn merging_keywords_twice_changes_nothing_the_second_time() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_image(&c, db, 1, "hash-a");
|
|
}
|
|
add_term(&c, "remote_cat", "u-1", "puffin", 1, 0);
|
|
keyword(&c, "remote_cat", 1, "puffin");
|
|
|
|
let first = merge_keywords(&c).unwrap();
|
|
assert!(first.local_changed());
|
|
let second = merge_keywords(&c).unwrap();
|
|
assert!(!second.local_changed(), "merge must be idempotent");
|
|
}
|
|
|
|
#[test]
|
|
fn a_remote_from_before_keyword_identities_still_contributes_its_words() {
|
|
// `remote_is_mergeable` admits an older remote on purpose — the check
|
|
// is that it is not *newer* than us. A missing table is therefore a
|
|
// normal state and must not fail the merge.
|
|
let c = two_catalogs_with_a_v1_remote();
|
|
for db in ["main", "remote_cat"] {
|
|
add_image(&c, db, 1, "hash-a");
|
|
}
|
|
keyword(&c, "remote_cat", 1, "puffin");
|
|
|
|
let report = merge_keywords(&c).unwrap();
|
|
assert_eq!(report.keywords_assigned, 1);
|
|
assert_eq!(words_on(&c, 1), ["puffin"]);
|
|
}
|
|
|
|
#[test]
|
|
fn merge_all_lands_both_halves() {
|
|
let c = two_catalogs();
|
|
add_collection(&c, "remote_cat", 1, "u-coll", "Portugal", 1);
|
|
for db in ["main", "remote_cat"] {
|
|
add_image(&c, db, 1, "hash-a");
|
|
}
|
|
keyword(&c, "remote_cat", 1, "puffin");
|
|
|
|
let report = merge_all(&c).unwrap();
|
|
assert_eq!(report.inserted, 1);
|
|
assert_eq!(report.keywords_assigned, 1);
|
|
}
|
|
|
|
#[test]
|
|
fn keeping_local_still_marks_the_catalog_for_upload() {
|
|
// We hold something the remote does not, so the remote is stale even
|
|
// though we took nothing from it.
|
|
let c = two_catalogs();
|
|
add_collection(&c, "main", 1, "shared", "Renamed here", 5);
|
|
add_collection(&c, "remote_cat", 1, "shared", "Old name", 2);
|
|
|
|
let report = merge_collections(&c).unwrap();
|
|
assert_eq!(report.kept_local, 1);
|
|
assert!(report.should_upload());
|
|
}
|
|
|
|
// ── people, and who the user said they are ────────────────────────────
|
|
|
|
/// An image present in `db` and carrying the cross-device file id both
|
|
/// catalogs agree on.
|
|
fn add_synced_image(c: &Connection, db: &str, id: i64, file_id: i64) {
|
|
add_image_without_hash(c, db, id);
|
|
c.execute(
|
|
&format!("INSERT INTO {db}.remote(image_id, file_id) VALUES (?1, ?2)"),
|
|
rusqlite::params![id, file_id],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
/// A face on `image`, at a box the caller can nudge to test the matching.
|
|
fn add_face(c: &Connection, db: &str, id: i64, image: i64, x: f64) -> i64 {
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.faces
|
|
(id, image_id, x, y, w, h, landmarks, detector_confidence,
|
|
embedding, crop_px, model_id, detected_at)
|
|
VALUES (?1, ?2, ?3, 0.2, 0.2, 0.2, X'00', 0.9, X'00', 150.0,
|
|
'w600k_mbf', 0)"
|
|
),
|
|
rusqlite::params![id, image, x],
|
|
)
|
|
.unwrap();
|
|
id
|
|
}
|
|
|
|
fn add_person(c: &Connection, db: &str, id: i64, uuid: &str, name: &str, ignored: bool) {
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.people(id, uuid, name, ignored, created, revision, modified)
|
|
VALUES (?1, ?2, ?3, ?4, 0, 1, 1)"
|
|
),
|
|
rusqlite::params![id, uuid, name, ignored],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
fn assign(c: &Connection, db: &str, face: i64, person: i64, confirmed: bool) {
|
|
c.execute(
|
|
&format!(
|
|
"INSERT INTO {db}.face_person(face_id, person_id, probability, confirmed)
|
|
VALUES (?1, ?2, 0.9, ?3)"
|
|
),
|
|
rusqlite::params![face, person, confirmed],
|
|
)
|
|
.unwrap();
|
|
}
|
|
|
|
fn person_of(c: &Connection, face: i64) -> Option<(String, bool)> {
|
|
c.query_row(
|
|
"SELECT p.name, fp.confirmed
|
|
FROM main.face_person fp JOIN main.people p ON p.id = fp.person_id
|
|
WHERE fp.face_id = ?1",
|
|
[face],
|
|
|r| Ok((r.get(0)?, r.get(1)?)),
|
|
)
|
|
.optional()
|
|
.unwrap()
|
|
}
|
|
|
|
/// The bug: a second device received every face through the shards and no
|
|
/// people at all, because this merge only ever looked at collections and
|
|
/// keywords. It drew an empty People screen over a full catalog.
|
|
#[test]
|
|
fn a_named_person_and_their_confirmed_face_cross_over() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
// The same face, found independently on each device, so the row ids
|
|
// differ — which is the whole difficulty.
|
|
let local = add_face(&c, "main", 7, 1, 0.30);
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.31);
|
|
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
|
assign(&c, "remote_cat", remote, 3, true);
|
|
|
|
let report = merge_all(&c).unwrap();
|
|
assert_eq!(report.people_inserted, 1);
|
|
assert_eq!(report.faces_assigned, 1);
|
|
assert_eq!(person_of(&c, local), Some(("Anna".to_string(), true)));
|
|
}
|
|
|
|
/// The bug this rule exists for: the desktop switched to a stronger
|
|
/// detector and confirmed 3,500 faces under the old pipeline id; the
|
|
/// tablet held the same faces under the new one, and not one name
|
|
/// crossed, because the match demanded the exact id. Same photograph,
|
|
/// same box, same embedder — that is the same face.
|
|
#[test]
|
|
fn a_confirmation_crosses_a_detector_change() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
let local = add_face(&c, "main", 7, 1, 0.30);
|
|
c.execute(
|
|
"UPDATE main.faces SET model_id = 'scrfd_10g+w600k_mbf' WHERE id = ?1",
|
|
[local],
|
|
)
|
|
.unwrap();
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.31);
|
|
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
|
assign(&c, "remote_cat", remote, 3, true);
|
|
|
|
let report = merge_all(&c).unwrap();
|
|
assert_eq!(report.faces_assigned, 1);
|
|
assert_eq!(person_of(&c, local), Some(("Anna".to_string(), true)));
|
|
}
|
|
|
|
/// A different embedder is a different space, and a box there is a face
|
|
/// nobody here has a vector for.
|
|
#[test]
|
|
fn a_confirmation_does_not_cross_an_embedder_change() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
let local = add_face(&c, "main", 7, 1, 0.30);
|
|
c.execute(
|
|
"UPDATE main.faces SET model_id = 'scrfd_10g+other_embedder' WHERE id = ?1",
|
|
[local],
|
|
)
|
|
.unwrap();
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.31);
|
|
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
|
assign(&c, "remote_cat", remote, 3, true);
|
|
|
|
merge_all(&c).unwrap();
|
|
assert_eq!(person_of(&c, local), None, "matched across embedders");
|
|
}
|
|
|
|
/// Boxes from two devices are close but not identical. Matching has to be
|
|
/// by overlap, not equality, or nothing ever lines up.
|
|
#[test]
|
|
fn a_face_in_a_different_photograph_is_not_matched() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
add_synced_image(&c, db, 2, 6000);
|
|
}
|
|
let elsewhere = add_face(&c, "main", 7, 2, 0.30);
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.30);
|
|
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
|
assign(&c, "remote_cat", remote, 3, true);
|
|
|
|
merge_all(&c).unwrap();
|
|
assert_eq!(person_of(&c, elsewhere), None, "matched across photographs");
|
|
}
|
|
|
|
/// Two faces in one frame, and the judgement must land on the right one.
|
|
#[test]
|
|
fn the_overlapping_face_is_the_one_that_gets_the_name() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
let left = add_face(&c, "main", 7, 1, 0.10);
|
|
let right = add_face(&c, "main", 8, 1, 0.70);
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.71);
|
|
add_person(&c, "remote_cat", 3, "u-bob", "Bob", false);
|
|
assign(&c, "remote_cat", remote, 3, true);
|
|
|
|
merge_all(&c).unwrap();
|
|
assert_eq!(person_of(&c, right), Some(("Bob".to_string(), true)));
|
|
assert_eq!(person_of(&c, left), None);
|
|
}
|
|
|
|
/// A group set aside on one device stays set aside on the other — which
|
|
/// needs its *suggestions* to travel, since that is what anchors it.
|
|
#[test]
|
|
fn a_group_set_aside_stays_set_aside_on_the_other_device() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
let local = add_face(&c, "main", 7, 1, 0.30);
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.30);
|
|
add_person(&c, "remote_cat", 3, "u-stranger", "", true);
|
|
// Only ever a suggestion, which is exactly why it needs carrying.
|
|
assign(&c, "remote_cat", remote, 3, false);
|
|
|
|
merge_all(&c).unwrap();
|
|
let ignored: bool = c
|
|
.query_row(
|
|
"SELECT ignored FROM main.people WHERE uuid = 'u-stranger'",
|
|
[],
|
|
|r| r.get(0),
|
|
)
|
|
.unwrap();
|
|
assert!(ignored, "the set-aside flag did not travel");
|
|
assert_eq!(person_of(&c, local), Some((String::new(), false)));
|
|
}
|
|
|
|
/// An ordinary suggestion is this pass's own output. Both devices hold the
|
|
/// same embeddings and clustering is deterministic, so each recomputes it —
|
|
/// shipping it would double the merge for no new information.
|
|
#[test]
|
|
fn an_ordinary_suggestion_does_not_travel() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
let local = add_face(&c, "main", 7, 1, 0.30);
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.30);
|
|
add_person(&c, "remote_cat", 3, "u-guess", "", false);
|
|
assign(&c, "remote_cat", remote, 3, false);
|
|
|
|
merge_all(&c).unwrap();
|
|
assert_eq!(person_of(&c, local), None);
|
|
}
|
|
|
|
/// A sync must not undo what the user did on the device they are holding.
|
|
#[test]
|
|
fn a_local_confirmation_outranks_the_remotes() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
let local = add_face(&c, "main", 7, 1, 0.30);
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.30);
|
|
add_person(&c, "main", 1, "u-anna", "Anna", false);
|
|
add_person(&c, "remote_cat", 3, "u-bob", "Bob", false);
|
|
assign(&c, "main", local, 1, true);
|
|
assign(&c, "remote_cat", remote, 3, true);
|
|
|
|
let report = merge_all(&c).unwrap();
|
|
assert_eq!(report.faces_kept_local, 1);
|
|
assert_eq!(person_of(&c, local), Some(("Anna".to_string(), true)));
|
|
}
|
|
|
|
/// "Not this person" is a judgement too, and it has to outrank an
|
|
/// assignment arriving from elsewhere.
|
|
#[test]
|
|
fn a_rejection_travels_and_removes_the_suggestion_it_contradicts() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
let local = add_face(&c, "main", 7, 1, 0.30);
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.30);
|
|
add_person(&c, "main", 1, "u-anna", "Anna", false);
|
|
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
|
// Locally suggested; the other device says it is not her.
|
|
assign(&c, "main", local, 1, false);
|
|
c.execute(
|
|
"INSERT INTO remote_cat.face_person_rejected(face_id, person_id)
|
|
VALUES (?1, 3)",
|
|
[remote],
|
|
)
|
|
.unwrap();
|
|
|
|
merge_all(&c).unwrap();
|
|
assert_eq!(person_of(&c, local), None, "the refused suggestion stayed");
|
|
let rejected: bool = c
|
|
.query_row(
|
|
"SELECT EXISTS(SELECT 1 FROM main.face_person_rejected WHERE face_id = ?1)",
|
|
[local],
|
|
|r| r.get(0),
|
|
)
|
|
.unwrap();
|
|
assert!(rejected);
|
|
}
|
|
|
|
/// A rename on the other device wins on revision, like everything else.
|
|
#[test]
|
|
fn a_higher_revision_renames_a_person() {
|
|
let c = two_catalogs();
|
|
add_person(&c, "main", 1, "u-anna", "Ana", false);
|
|
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
|
c.execute(
|
|
"UPDATE remote_cat.people SET revision = 5, modified = 5 WHERE uuid = 'u-anna'",
|
|
[],
|
|
)
|
|
.unwrap();
|
|
|
|
let report = merge_all(&c).unwrap();
|
|
assert_eq!(report.people_updated, 1);
|
|
let name: String = c
|
|
.query_row(
|
|
"SELECT name FROM main.people WHERE uuid = 'u-anna'",
|
|
[],
|
|
|r| r.get(0),
|
|
)
|
|
.unwrap();
|
|
assert_eq!(name, "Anna");
|
|
}
|
|
|
|
#[test]
|
|
fn a_lower_revision_does_not_rename_a_person() {
|
|
let c = two_catalogs();
|
|
add_person(&c, "main", 1, "u-anna", "Anna", false);
|
|
add_person(&c, "remote_cat", 3, "u-anna", "Ana", false);
|
|
c.execute("UPDATE main.people SET revision = 9, modified = 9", [])
|
|
.unwrap();
|
|
|
|
let report = merge_all(&c).unwrap();
|
|
assert_eq!(report.people_kept_local, 1);
|
|
let name: String = c
|
|
.query_row(
|
|
"SELECT name FROM main.people WHERE uuid = 'u-anna'",
|
|
[],
|
|
|r| r.get(0),
|
|
)
|
|
.unwrap();
|
|
assert_eq!(name, "Anna");
|
|
}
|
|
|
|
/// Merging twice must not double anything — a sync runs on every pass.
|
|
#[test]
|
|
fn merging_people_twice_changes_nothing_the_second_time() {
|
|
let c = two_catalogs();
|
|
for db in ["main", "remote_cat"] {
|
|
add_synced_image(&c, db, 1, 5000);
|
|
}
|
|
add_face(&c, "main", 7, 1, 0.30);
|
|
let remote = add_face(&c, "remote_cat", 42, 1, 0.30);
|
|
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
|
assign(&c, "remote_cat", remote, 3, true);
|
|
|
|
merge_all(&c).unwrap();
|
|
let second = merge_all(&c).unwrap();
|
|
assert_eq!(second.people_inserted, 0);
|
|
let people: i64 = c
|
|
.query_row("SELECT COUNT(*) FROM main.people", [], |r| r.get(0))
|
|
.unwrap();
|
|
assert_eq!(people, 1);
|
|
}
|
|
}
|