Sync face data as sealed shards, so a second device does not re-index

Indexing 23,500 images is about two hours of CPU, and the result is
byte-identical on every device: the same model over the same proxy
produces the same embedding. Paying for it once per account rather than
once per device is the point.

Shards rather than the catalog snapshot, because the snapshot goes up
whole on every sync and a fully indexed library carries roughly 30 MB of
embeddings. That is exactly the cost the thumbnail store's 25 MB cap
exists to bound, so face shards use the same cap -- imported from
dr_thumbs rather than restated, since the number is a statement about
sync cost and the two must not drift apart.

The split follows the one already there: bulk immutable data in sealed
shards, small mutable data in the catalog snapshot. Faces, landmarks,
embeddings and run markers shard; people, names and assignments ride the
catalog and merge by uuid.

Keyed on oc:fileid throughout, never on image_id, because a row id means
nothing on another device.

The run marker travels with the faces it describes. Without it a
receiving device cannot tell an image with no faces from one never
examined, and would re-detect every landscape it had just adopted.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-08-26 22:41:50 +02:00
co-authored by Claude Opus 5
parent 26a1eb7e28
commit 96d07da15f
7 changed files with 1179 additions and 6 deletions
+46 -1
View File
@@ -4,6 +4,7 @@
//! indexes whatever is missing one.
//!
//! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR [--run DET.onnx EMB.onnx]
//! cargo run -p dr-ui --example face_index -- CATALOG.db THUMBS_DIR --cluster
//!
//! # Why this exists beside the button in the Identity screen
//!
@@ -81,6 +82,23 @@ fn main() {
println!(" ready to index {}", audit.ready);
println!(" awaiting proxy {}", audit.awaiting_proxy);
// Grouping is a separate step from indexing on purpose: it is a
// whole-library operation over the embeddings detection produced, and it is
// worth running *after* a sweep rather than during one (catalog.md §10.2).
if args.iter().any(|a| a == "--cluster") {
match dr_ui::faces::recluster(&catalog, MODEL_ID, dr_face::DEFAULT_MERGE_PROBABILITY) {
Ok((suggested, created)) => {
println!("\nclustering: {suggested} suggestion(s), {created} new group(s)");
report_people(&catalog);
}
Err(e) => {
eprintln!("clustering failed: {e}");
std::process::exit(1);
}
}
return;
}
let run = args.iter().position(|a| a == "--run");
let Some(i) = run else {
if audit.coverage.is_complete() {
@@ -129,7 +147,7 @@ fn main() {
seen += 1;
// One line per image would be thousands of lines; one per
// twenty-five is enough to see it moving and to estimate.
if seen % 25 == 0 || faces > 0 {
if seen.is_multiple_of(25) || faces > 0 {
let rate = seen as f64 / start.elapsed().as_secs_f64().max(1e-6);
println!(
" {seen}/{total} {:.2} img/s ~{:.0} min left",
@@ -154,4 +172,31 @@ fn main() {
if let Ok(a) = faces::audit(&catalog, &store, MODEL_ID) {
println!("{}", a.summary());
}
println!("\nrun again with --cluster to group these faces into people.");
}
/// What the clustering proposed, largest group first.
fn report_people(catalog: &Catalog) {
let Ok(people) = dr_catalog::faces::people(catalog.connection()) else {
return;
};
if people.is_empty() {
println!("no groups — too few faces, or none similar enough to group.");
return;
}
println!("\n{} group(s):", people.len());
for p in people.iter().take(30) {
let name = if p.name.is_empty() {
"(unnamed)".to_string()
} else {
p.name.clone()
};
println!(
" {name:<24} {} confirmed, {} suggested",
p.confirmed_faces, p.suggested_faces
);
}
if people.len() > 30 {
println!(" … and {} more", people.len() - 30);
}
}
+163
View File
@@ -60,6 +60,16 @@ pub struct SyncReport {
/// devices already had gains *no* collection, and reporting only the
/// former left the sidebar showing no count beside a full collection.
pub members_gained: usize,
// Face data is counted apart from thumbnails for the same reason keywords
// are counted apart from collections: "adopted 4,812 faces" is a sentence
// the user can act on, and folding it into the thumbnail count would hide
// the one number that says whether this device still has hours of indexing
// ahead of it.
pub face_shards_uploaded: usize,
pub face_shards_downloaded: usize,
/// Images whose faces this device took from a peer instead of detecting.
pub faces_adopted: usize,
}
impl SyncReport {
@@ -68,6 +78,8 @@ impl SyncReport {
|| self.shards_downloaded > 0
|| self.catalog_uploaded
|| self.catalog_merged
|| self.face_shards_uploaded > 0
|| self.face_shards_downloaded > 0
}
}
@@ -145,6 +157,9 @@ async fn run(
let _ = tx.send(SyncMessage::Status("checking thumbnails…".into()));
sync_shards(backend, &base, thumbs_dir, scratch, &mut report).await?;
let _ = tx.send(SyncMessage::Status("checking faces…".into()));
sync_face_shards(backend, &base, catalog_path, scratch, &mut report).await?;
let _ = tx.send(SyncMessage::Status("checking collections…".into()));
sync_catalog(backend, &base, catalog_path, scratch, &mut report).await?;
@@ -296,6 +311,154 @@ async fn sync_shards(
Ok(())
}
/// Push and pull face shards, so a second device does not re-index the library.
///
/// Mirrors [`sync_shards`] deliberately, down to the sealed-shard skip and the
/// adopted ledger: face data has exactly the properties that made that design
/// right for thumbnails. It is bulk, it is immutable once written, and it is
/// byte-identical on every device, because the same model over the same proxy
/// is deterministic.
///
/// The catalog is the source and the destination; the shards are only the
/// carrier. So this exports the catalog's new faces into the local shard store
/// first, syncs the shards, and imports whatever arrived back into the catalog.
async fn sync_face_shards(
backend: &NextcloudBackend,
base: &RemotePath,
catalog_path: &Path,
scratch: &Path,
report: &mut SyncReport,
) -> Result<(), String> {
use dr_catalog::face_shard::{self, FaceShardStore};
// Beside the catalog, next to the thumbnails, and under the same folder the
// scanner already excludes.
let dir = match catalog_path.parent() {
Some(p) => p.join("faces"),
None => return Ok(()),
};
let mut store = match FaceShardStore::open(&dir) {
Ok(s) => s,
Err(e) => {
// Not a sync failure: the next pass retries once it opens.
log::debug!("face shard store unavailable: {e}");
return Ok(());
}
};
let client = store.client_id().to_string();
let model = crate::identity_ui::MODEL_ID;
// ---- everything this device has detected, into the shards ------------
if let Ok(catalog) = dr_catalog::Catalog::open(catalog_path) {
match face_shard::export_to_shards(catalog.connection(), &mut store, model) {
Ok(0) => {}
Ok(n) => log::info!("face sync: {n} newly indexed image(s) ready to upload"),
Err(e) => log::warn!("face sync: exporting to shards: {e}"),
}
}
// Its own folder under the derived directory, so a client that does not
// care about faces lists thumbnails without paging past them.
let face_base = RemotePath::new(format!("{}/faces", base.as_str()));
let _ = backend.create_dir(&face_base).await;
let remote: std::collections::HashMap<String, u64> = backend
.list(&face_base, None)
.await
.map(|entries| {
entries
.into_iter()
.filter(|e| e.kind == dr_sync::EntryKind::File)
.map(|e| (e.path.name().to_string(), e.size))
.collect()
})
.unwrap_or_default();
// ---- upload ----------------------------------------------------------
let local = store.shards().map_err(|e| e.to_string())?;
for shard in &local {
let path = store.shard_path(shard.id);
let Ok(bytes) = std::fs::read(&path) else {
continue;
};
let name = shard_name(&client, shard.id);
// Sealed and present means byte-identical, and the client is in the
// name so nobody else could have written it. The open shard goes up
// again whenever its size differs, which is the only way it changes.
let skip = match remote.get(&name) {
Some(_) if shard.sealed => true,
Some(size) => *size == bytes.len() as u64,
None => false,
};
if skip {
continue;
}
let target = RemotePath::new(format!("{}/{name}", face_base.as_str()));
match backend.put(&target, bytes, None).await {
Ok(_) => report.face_shards_uploaded += 1,
Err(e) => log::warn!("uploading face shard {name}: {e}"),
}
}
// ---- download --------------------------------------------------------
for (name, size) in &remote {
let Some((owner, _)) = parse_shard(name) else {
continue;
};
if owner == client || store.has_adopted(name, *size) {
continue;
}
let source = RemotePath::new(format!("{}/{name}", face_base.as_str()));
let bytes = match backend.get(&RemoteId::Path(source), None).await {
Ok(b) => b,
Err(e) => {
log::warn!("downloading face shard {name}: {e}");
continue;
}
};
// Into scratch and merged, never dropped into the store directory: a
// downloaded shard's id is the *other* device's numbering, and two
// devices independently fill shard 0.
let tmp = scratch.join(name);
if std::fs::write(&tmp, &bytes).is_err() {
continue;
}
match store.merge_shard(&tmp) {
Ok(_) => {
report.face_shards_downloaded += 1;
// Recorded only on success, so a failed merge is retried next
// pass rather than written off.
let _ = store.mark_adopted(name, *size);
}
Err(e) => log::warn!("merging face shard {name}: {e}"),
}
let _ = std::fs::remove_file(&tmp);
}
// ---- and back into the catalog ---------------------------------------
//
// Last, and unconditionally rather than only when something downloaded: a
// previous pass may have merged shards into the store and then failed
// before importing, and this is what recovers from that.
if let Ok(catalog) = dr_catalog::Catalog::open(catalog_path) {
match face_shard::import_from_shards(catalog.connection(), &store, model) {
Ok(0) => {}
Ok(n) => {
report.faces_adopted = n;
log::info!("face sync: adopted {n} image(s) already indexed elsewhere");
}
Err(e) => log::warn!("face sync: importing from shards: {e}"),
}
}
Ok(())
}
/// Whether a flat-named remote shard is this client's own earlier upload.
///
/// Before the name carried a client every client wrote `shard-NNNN.sqlite`, so