Merge: one face population per embedder, whichever detector found them
This commit is contained in:
@@ -149,6 +149,29 @@ impl FaceShardStore {
|
||||
.flatten()
|
||||
}
|
||||
|
||||
/// The pipeline this store holds an image under, among those sharing
|
||||
/// `model_id`'s embedder — the most recently indexed where a peer has
|
||||
/// sent more than one.
|
||||
///
|
||||
/// What the import asks: not "has anyone run *this* detector over it" but
|
||||
/// "does anyone hold comparable faces for it". See `faces::embedder_of`.
|
||||
pub fn held_model(&self, file_id: u64, model_id: &str) -> Option<String> {
|
||||
self.index
|
||||
.query_row(
|
||||
&format!(
|
||||
"SELECT model_id FROM entries
|
||||
WHERE file_id = ?1 AND {} = ?2
|
||||
ORDER BY indexed_at DESC NULLS LAST, model_id",
|
||||
crate::faces::embedder_sql("model_id")
|
||||
),
|
||||
rusqlite::params![file_id as i64, crate::faces::embedder_of(model_id)],
|
||||
|r| r.get::<_, String>(0),
|
||||
)
|
||||
.optional()
|
||||
.ok()
|
||||
.flatten()
|
||||
}
|
||||
|
||||
pub fn contains(&self, file_id: u64, model_id: &str) -> bool {
|
||||
self.index
|
||||
.query_row(
|
||||
@@ -608,16 +631,21 @@ pub fn export_to_shards_reporting(
|
||||
model_id: &str,
|
||||
progress: &mut dyn FnMut(usize, usize),
|
||||
) -> Result<usize, CatalogError> {
|
||||
let mut q = conn.prepare(
|
||||
"SELECT r.file_id, fi.image_id, fi.source_edge, fi.indexed_at
|
||||
// Every pipeline sharing this one's embedder, each image under the id
|
||||
// that actually indexed it. A device that switched detectors still holds
|
||||
// most of its library under the previous id, and those faces are exactly
|
||||
// as comparable — and as wanted by a peer — as the new ones.
|
||||
let mut q = conn.prepare(&format!(
|
||||
"SELECT r.file_id, fi.image_id, fi.source_edge, fi.indexed_at, fi.model_id
|
||||
FROM face_index fi
|
||||
JOIN remote r ON r.image_id = fi.image_id
|
||||
WHERE fi.model_id = ?1
|
||||
WHERE {} = ?1
|
||||
ORDER BY fi.image_id",
|
||||
)?;
|
||||
let rows: Vec<(i64, i64, i64, i64)> = q
|
||||
.query_map([model_id], |r| {
|
||||
Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?))
|
||||
crate::faces::embedder_sql("fi.model_id")
|
||||
))?;
|
||||
let rows: Vec<(i64, i64, i64, i64, String)> = q
|
||||
.query_map([crate::faces::embedder_of(model_id)], |r| {
|
||||
Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?, r.get(4)?))
|
||||
})?
|
||||
.collect::<Result<_, _>>()?;
|
||||
|
||||
@@ -627,7 +655,8 @@ pub fn export_to_shards_reporting(
|
||||
|
||||
let total = rows.len();
|
||||
let mut exported = 0;
|
||||
for (seen, (file_id, image_id, edge, indexed_at)) in rows.into_iter().enumerate() {
|
||||
for (seen, (file_id, image_id, edge, indexed_at, model_id)) in rows.into_iter().enumerate() {
|
||||
let model_id = model_id.as_str();
|
||||
if seen.is_multiple_of(REPORT_EVERY) {
|
||||
progress(seen, total);
|
||||
}
|
||||
@@ -689,10 +718,20 @@ pub fn export_to_shards_reporting(
|
||||
/// adopted rather than re-detected, which is the difference between a new
|
||||
/// device being useful in a minute and in two hours.
|
||||
///
|
||||
/// Skips any image this device has already indexed itself. Local work is not
|
||||
/// second-guessed by a peer's — the two should agree, since the same model over
|
||||
/// the same proxy is deterministic, but where they do not, the copy this device
|
||||
/// computed is the one it can vouch for.
|
||||
/// Skips any image this device has already indexed itself under this
|
||||
/// pipeline or any sharing its embedder — unless the peer ran a detector that
|
||||
/// outranks the one that indexed it here. Local work is not second-guessed
|
||||
/// by a peer's equal: the two should agree, since the same model over the
|
||||
/// same proxy is deterministic, and where they do not, the copy this device
|
||||
/// computed is the one it can vouch for. A peer's *stronger* pass is another
|
||||
/// matter: it is the re-detection this device's own sweep would queue
|
||||
/// (`FaceDetector::supersedes`), already done, and taking it is what spares
|
||||
/// a tablet the fetch. Names survive the replacement by box overlap, as they
|
||||
/// do a local re-detection.
|
||||
///
|
||||
/// A peer's faces are taken under whichever compatible detector found them:
|
||||
/// a tablet set to the fast detector adopts the desktop's thorough pass
|
||||
/// rather than re-detecting it worse.
|
||||
///
|
||||
/// Returns how many images were adopted.
|
||||
pub fn import_from_shards(
|
||||
@@ -700,26 +739,49 @@ pub fn import_from_shards(
|
||||
store: &FaceShardStore,
|
||||
model_id: &str,
|
||||
) -> Result<usize, CatalogError> {
|
||||
use dr_types::FaceDetector;
|
||||
|
||||
// Only images this device actually has. A shard covers the whole account,
|
||||
// and a device holding a subset of the library should take only its own
|
||||
// part rather than accumulating faces for photographs it cannot show.
|
||||
let mut q = conn.prepare(
|
||||
"SELECT r.file_id, r.image_id
|
||||
//
|
||||
// With the pipeline that indexed each one here, or NULL: the marker is
|
||||
// what decides whether a peer's copy is a gap filled or an upgrade.
|
||||
let mut q = conn.prepare(&format!(
|
||||
"SELECT r.file_id, r.image_id,
|
||||
(SELECT fi.model_id FROM face_index fi
|
||||
WHERE fi.image_id = r.image_id AND {} = ?1)
|
||||
FROM remote r
|
||||
JOIN images i ON i.id = r.image_id
|
||||
WHERE i.trashed_at IS NULL
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM face_index fi
|
||||
WHERE fi.image_id = r.image_id AND fi.model_id = ?1
|
||||
)",
|
||||
)?;
|
||||
let candidates: Vec<(i64, i64)> = q
|
||||
.query_map([model_id], |r| Ok((r.get(0)?, r.get(1)?)))?
|
||||
WHERE i.trashed_at IS NULL",
|
||||
crate::faces::embedder_sql("fi.model_id")
|
||||
))?;
|
||||
let candidates: Vec<(i64, i64, Option<String>)> = q
|
||||
.query_map([crate::faces::embedder_of(model_id)], |r| {
|
||||
Ok((r.get(0)?, r.get(1)?, r.get(2)?))
|
||||
})?
|
||||
.collect::<Result<_, _>>()?;
|
||||
|
||||
let mut adopted = 0;
|
||||
for (file_id, image_id) in candidates {
|
||||
let Some((faces, edge)) = store.get_image(file_id as u64, model_id)? else {
|
||||
for (file_id, image_id, local) in candidates {
|
||||
let Some(held) = store.held_model(file_id as u64, model_id) else {
|
||||
continue;
|
||||
};
|
||||
if let Some(local) = local {
|
||||
// An unknown detector on either side cannot be ranked, and an
|
||||
// unranked peer is treated as an equal: kept out.
|
||||
let upgrade = match (
|
||||
FaceDetector::for_model_id(&held),
|
||||
FaceDetector::for_model_id(&local),
|
||||
) {
|
||||
(Some(theirs), Some(ours)) => theirs.outranks(ours),
|
||||
_ => false,
|
||||
};
|
||||
if !upgrade {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let Some((faces, edge)) = store.get_image(file_id as u64, &held)? else {
|
||||
continue;
|
||||
};
|
||||
// A peer that embedded before the quality was kept has done work this
|
||||
@@ -753,7 +815,7 @@ pub fn import_from_shards(
|
||||
crate::faces::record_detections(
|
||||
conn,
|
||||
dr_types::ImageId(image_id as u64),
|
||||
model_id,
|
||||
&held,
|
||||
edge,
|
||||
&local,
|
||||
)?;
|
||||
@@ -1197,6 +1259,49 @@ mod catalog_round_trip {
|
||||
assert!(emb.iter().any(|e| e.embedding[0] == 1));
|
||||
}
|
||||
|
||||
/// The desktop switched to a stronger detector part-way through the
|
||||
/// library, so its faces sit under two pipeline ids. A tablet on the
|
||||
/// original detector must receive *all* of them — each under the id that
|
||||
/// found it — and not re-detect the thorough half worse.
|
||||
#[test]
|
||||
fn every_generation_sharing_an_embedder_travels_and_is_adopted() {
|
||||
let a = device(&[(1, 5001), (2, 5002)]);
|
||||
let b = device(&[(90, 5001), (91, 5002)]);
|
||||
|
||||
faces::record_detections(&a, dr_types::ImageId(1), "w600k_mbf", 1024, &[detected(1)])
|
||||
.unwrap();
|
||||
let mut thorough = detected(2);
|
||||
thorough.model_id = "scrfd_10g+w600k_mbf".into();
|
||||
faces::record_detections(
|
||||
&a,
|
||||
dr_types::ImageId(2),
|
||||
"scrfd_10g+w600k_mbf",
|
||||
1024,
|
||||
&[thorough],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let mut store_a = FaceShardStore::open(&tempdir("a")).unwrap();
|
||||
assert_eq!(
|
||||
export_to_shards(&a, &mut store_a, "scrfd_10g+w600k_mbf").unwrap(),
|
||||
2,
|
||||
"the export left the earlier detector's images behind"
|
||||
);
|
||||
|
||||
let mut store_b = FaceShardStore::open(&tempdir("b")).unwrap();
|
||||
store_b.merge_shard(&store_a.shard_path(0)).unwrap();
|
||||
assert_eq!(import_from_shards(&b, &store_b, "w600k_mbf").unwrap(), 2);
|
||||
|
||||
assert_eq!(faces::coverage(&b, "w600k_mbf").unwrap().outstanding(), 0);
|
||||
let old = faces::for_image(&b, dr_types::ImageId(90)).unwrap();
|
||||
let new = faces::for_image(&b, dr_types::ImageId(91)).unwrap();
|
||||
assert_eq!(old[0].model_id, "w600k_mbf");
|
||||
assert_eq!(
|
||||
new[0].model_id, "scrfd_10g+w600k_mbf",
|
||||
"adopted under the wrong id"
|
||||
);
|
||||
}
|
||||
|
||||
/// A face a peer embedded without measuring it is work this device
|
||||
/// cannot finish, and adopting it would write the marker that stops it
|
||||
/// ever being measured. The image stays outstanding instead.
|
||||
@@ -1257,6 +1362,59 @@ mod catalog_round_trip {
|
||||
assert_eq!(emb[0].embedding[0], 9, "B's own embedding was overwritten");
|
||||
}
|
||||
|
||||
/// A peer's stronger detector is the re-detection this device would
|
||||
/// otherwise queue for itself. Taking it saves the fetch; the name the
|
||||
/// user confirmed here rides across on the box, as it would locally.
|
||||
#[test]
|
||||
fn a_peers_stronger_pass_replaces_a_weaker_local_one_and_keeps_the_name() {
|
||||
let a = device(&[(1, 5001)]);
|
||||
let b = device(&[(50, 5001)]);
|
||||
|
||||
let ids =
|
||||
faces::record_detections(&b, dr_types::ImageId(50), "w600k_mbf", 1024, &[detected(9)])
|
||||
.unwrap();
|
||||
let anna = faces::create_person(&b, "Anna").unwrap();
|
||||
faces::confirm(&b, ids[0], anna).unwrap();
|
||||
|
||||
let mut thorough = detected(7);
|
||||
thorough.model_id = "scrfd_10g+w600k_mbf".into();
|
||||
let mut second = detected(8);
|
||||
second.model_id = "scrfd_10g+w600k_mbf".into();
|
||||
second.x = 0.6;
|
||||
faces::record_detections(
|
||||
&a,
|
||||
dr_types::ImageId(1),
|
||||
"scrfd_10g+w600k_mbf",
|
||||
1024,
|
||||
&[thorough, second],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let mut store = FaceShardStore::open(&tempdir("upgrade")).unwrap();
|
||||
export_to_shards(&a, &mut store, "scrfd_10g+w600k_mbf").unwrap();
|
||||
assert_eq!(import_from_shards(&b, &store, "w600k_mbf").unwrap(), 1);
|
||||
|
||||
let got = faces::for_image(&b, dr_types::ImageId(50)).unwrap();
|
||||
assert_eq!(got.len(), 2, "the stronger pass was not adopted");
|
||||
let named = got
|
||||
.iter()
|
||||
.find(|f| f.person == Some(anna))
|
||||
.expect("the name was lost");
|
||||
assert!(named.confirmed);
|
||||
assert_eq!(named.model_id, "scrfd_10g+w600k_mbf");
|
||||
|
||||
// And never downwards: A on the fast detector keeps B's thorough faces.
|
||||
let mut store_b = FaceShardStore::open(&tempdir("downgrade")).unwrap();
|
||||
faces::record_detections(&b, dr_types::ImageId(50), "w600k_mbf", 1024, &[detected(9)])
|
||||
.unwrap();
|
||||
export_to_shards(&b, &mut store_b, "w600k_mbf").unwrap();
|
||||
assert_eq!(
|
||||
import_from_shards(&a, &store_b, "scrfd_10g+w600k_mbf").unwrap(),
|
||||
0
|
||||
);
|
||||
assert_eq!(faces::for_image(&a, dr_types::ImageId(1)).unwrap().len(), 2);
|
||||
}
|
||||
|
||||
/// A device holding a subset of the library takes only its own part.
|
||||
#[test]
|
||||
fn a_device_ignores_faces_for_photographs_it_does_not_have() {
|
||||
|
||||
+186
-38
@@ -26,6 +26,25 @@
|
||||
//! disposable index; the name is irreplaceable and goes to the sidecar
|
||||
//! (FR-CULL-12), which is not this module's job.
|
||||
//!
|
||||
//! # One population per embedder, not one per detector
|
||||
//!
|
||||
//! `faces.model_id` names a pipeline, `detector+embedder`, and a detector
|
||||
//! change writes new rows under a new id. What it must *not* do is split the
|
||||
//! library in two: the People screen, the clustering pass, the sync merge and
|
||||
//! the shard import all read "the faces" — and the faces are every row whose
|
||||
//! embedder half matches, because the embedder is what makes two vectors
|
||||
//! comparable and the detector only decides where the boxes are. Before this
|
||||
//! rule each of those read the exact id, and choosing a better detector
|
||||
//! emptied every screen until a 400 GB re-index had run on every device,
|
||||
//! while a name confirmed on one device could not reach the other because the
|
||||
//! two held the same face under different ids.
|
||||
//!
|
||||
//! So reads key on [`embedder_of`] the id, via [`embedder_sql`]; writes keep
|
||||
//! the exact id, so which detector drew a box stays recorded; and
|
||||
//! [`record_detections`] is where the generations meet — an image holds one
|
||||
//! pipeline's faces at a time, and a re-detection carries the user's
|
||||
//! confirmations onto the boxes that replace them.
|
||||
//!
|
||||
//! # Privacy is structural here, not policy
|
||||
//!
|
||||
//! NFR-SEC-5 puts face data under a stricter rule than the rest of the catalog:
|
||||
@@ -40,6 +59,30 @@ use dr_types::ImageId;
|
||||
|
||||
use crate::error::CatalogError;
|
||||
|
||||
/// The embedder half of a model id: what makes two faces comparable.
|
||||
///
|
||||
/// `scrfd_10g+w600k_mbf` and `w600k_mbf` are the same embedder behind two
|
||||
/// detectors, and their vectors live in one space. A bare id is its own
|
||||
/// embedder — the first pipeline was written without a detector prefix, and
|
||||
/// every library indexed before the choice existed is under that spelling.
|
||||
pub fn embedder_of(model_id: &str) -> &str {
|
||||
model_id.rsplit('+').next().unwrap_or(model_id)
|
||||
}
|
||||
|
||||
/// The SQL for [`embedder_of`] over a column, for a `WHERE` that means "the
|
||||
/// same population as this pipeline" rather than "this exact pipeline".
|
||||
///
|
||||
/// Callers pass `embedder_of(model_id)` as the bound value. A `CASE` rather
|
||||
/// than `LIKE`, so the bare and the qualified spelling compare as the same
|
||||
/// thing without a wildcard that `w600k_mbf_v2` would also match.
|
||||
pub fn embedder_sql(column: &str) -> String {
|
||||
format!(
|
||||
"CASE WHEN instr({column}, '+') > 0
|
||||
THEN substr({column}, instr({column}, '+') + 1)
|
||||
ELSE {column} END"
|
||||
)
|
||||
}
|
||||
|
||||
/// A face's row id. Local to this catalog, like [`crate::keywords::KeywordId`].
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
|
||||
pub struct FaceId(pub u64);
|
||||
@@ -330,8 +373,11 @@ pub fn record_measurements(
|
||||
tx.execute("DELETE FROM faces WHERE id = ?1", [f.0 as i64])?;
|
||||
}
|
||||
let remaining: i64 = tx.query_row(
|
||||
"SELECT COUNT(*) FROM faces WHERE image_id = ?1 AND model_id = ?2",
|
||||
rusqlite::params![image_id.0 as i64, model_id],
|
||||
&format!(
|
||||
"SELECT COUNT(*) FROM faces WHERE image_id = ?1 AND {} = ?2",
|
||||
embedder_sql("model_id")
|
||||
),
|
||||
rusqlite::params![image_id.0 as i64, embedder_of(model_id)],
|
||||
|r| r.get(0),
|
||||
)?;
|
||||
tx.execute(
|
||||
@@ -365,7 +411,7 @@ pub fn unmeasured_on_image(
|
||||
) -> Result<Vec<Face>, CatalogError> {
|
||||
Ok(for_image(conn, image_id)?
|
||||
.into_iter()
|
||||
.filter(|f| f.model_id == model_id && f.quality.is_none())
|
||||
.filter(|f| embedder_of(&f.model_id) == embedder_of(model_id) && f.quality.is_none())
|
||||
.collect())
|
||||
}
|
||||
|
||||
@@ -374,8 +420,11 @@ pub fn unmeasured_on_image(
|
||||
/// What the measuring pass has left to do, for a screen that wants to say so.
|
||||
pub fn faces_unmeasured(conn: &Connection, model_id: &str) -> Result<u64, CatalogError> {
|
||||
conn.query_row(
|
||||
"SELECT COUNT(*) FROM faces WHERE model_id = ?1 AND quality IS NULL",
|
||||
[model_id],
|
||||
&format!(
|
||||
"SELECT COUNT(*) FROM faces WHERE {} = ?1 AND quality IS NULL",
|
||||
embedder_sql("model_id")
|
||||
),
|
||||
[embedder_of(model_id)],
|
||||
|r| r.get::<_, i64>(0),
|
||||
)
|
||||
.map(|n| n as u64)
|
||||
@@ -443,13 +492,19 @@ pub fn coverage(conn: &Connection, model_id: &str) -> Result<Coverage, CatalogEr
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)?;
|
||||
// One row per image, whichever compatible pipeline wrote it: an image
|
||||
// holds one pipeline's faces at a time (`record_detections`), so the
|
||||
// markers of one embedder never double-count a photograph.
|
||||
let (indexed, without, faces): (i64, i64, i64) = conn.query_row(
|
||||
"SELECT COUNT(*), COALESCE(SUM(faces_found = 0), 0), COALESCE(SUM(faces_found), 0)
|
||||
FROM face_index fi
|
||||
JOIN images i ON i.id = fi.image_id
|
||||
WHERE fi.model_id = ?1
|
||||
AND i.trashed_at IS NULL AND i.shadowed_by IS NULL",
|
||||
[model_id],
|
||||
&format!(
|
||||
"SELECT COUNT(*), COALESCE(SUM(faces_found = 0), 0), COALESCE(SUM(faces_found), 0)
|
||||
FROM face_index fi
|
||||
JOIN images i ON i.id = fi.image_id
|
||||
WHERE {} = ?1
|
||||
AND i.trashed_at IS NULL AND i.shadowed_by IS NULL",
|
||||
embedder_sql("fi.model_id")
|
||||
),
|
||||
[embedder_of(model_id)],
|
||||
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)),
|
||||
)?;
|
||||
|
||||
@@ -461,15 +516,19 @@ pub fn coverage(conn: &Connection, model_id: &str) -> Result<Coverage, CatalogEr
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether one image has been through this model.
|
||||
/// Whether one image has been through this model, or one it shares an
|
||||
/// embedder with.
|
||||
pub fn is_indexed(
|
||||
conn: &Connection,
|
||||
image_id: ImageId,
|
||||
model_id: &str,
|
||||
) -> Result<bool, CatalogError> {
|
||||
conn.query_row(
|
||||
"SELECT EXISTS(SELECT 1 FROM face_index WHERE image_id = ?1 AND model_id = ?2)",
|
||||
rusqlite::params![image_id.0 as i64, model_id],
|
||||
&format!(
|
||||
"SELECT EXISTS(SELECT 1 FROM face_index WHERE image_id = ?1 AND {} = ?2)",
|
||||
embedder_sql("model_id")
|
||||
),
|
||||
rusqlite::params![image_id.0 as i64, embedder_of(model_id)],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.map_err(Into::into)
|
||||
@@ -486,8 +545,11 @@ pub fn clear_index_marker(
|
||||
model_id: &str,
|
||||
) -> Result<(), CatalogError> {
|
||||
conn.execute(
|
||||
"DELETE FROM face_index WHERE image_id = ?1 AND model_id = ?2",
|
||||
rusqlite::params![image_id.0 as i64, model_id],
|
||||
&format!(
|
||||
"DELETE FROM face_index WHERE image_id = ?1 AND {} = ?2",
|
||||
embedder_sql("model_id")
|
||||
),
|
||||
rusqlite::params![image_id.0 as i64, embedder_of(model_id)],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
@@ -513,12 +575,15 @@ pub fn for_image(conn: &Connection, image_id: ImageId) -> Result<Vec<Face>, Cata
|
||||
/// unassigned and still belongs in the next clustering round, just not in that
|
||||
/// person's cluster.
|
||||
pub fn unassigned(conn: &Connection, model_id: &str) -> Result<Vec<FaceId>, CatalogError> {
|
||||
let mut q = conn.prepare(
|
||||
let mut q = conn.prepare(&format!(
|
||||
"SELECT f.id FROM faces f
|
||||
LEFT JOIN face_person fp ON fp.face_id = f.id
|
||||
WHERE fp.face_id IS NULL AND f.model_id = ?1",
|
||||
)?;
|
||||
let rows = q.query_map([model_id], |r| Ok(FaceId(r.get::<_, i64>(0)? as u64)))?;
|
||||
WHERE fp.face_id IS NULL AND {} = ?1",
|
||||
embedder_sql("f.model_id")
|
||||
))?;
|
||||
let rows = q.query_map([embedder_of(model_id)], |r| {
|
||||
Ok(FaceId(r.get::<_, i64>(0)? as u64))
|
||||
})?;
|
||||
rows.collect::<Result<_, _>>().map_err(Into::into)
|
||||
}
|
||||
|
||||
@@ -539,15 +604,18 @@ pub struct StoredEmbedding {
|
||||
|
||||
/// Embeddings for clustering, oldest first so the pass is deterministic.
|
||||
///
|
||||
/// Returned as raw f16 blobs rather than decoded vectors: the caller is
|
||||
/// `dr-face`, which owns the decoding, and a catalog that widened them here
|
||||
/// would double the memory of the one operation that holds them all at once.
|
||||
/// Every face this model's embedder produced, whichever detector found it —
|
||||
/// see the module note. Returned as raw f16 blobs rather than decoded vectors:
|
||||
/// the caller is `dr-face`, which owns the decoding, and a catalog that
|
||||
/// widened them here would double the memory of the one operation that holds
|
||||
/// them all at once.
|
||||
pub fn embeddings(conn: &Connection, model_id: &str) -> Result<Vec<StoredEmbedding>, CatalogError> {
|
||||
let mut q = conn.prepare(
|
||||
let mut q = conn.prepare(&format!(
|
||||
"SELECT id, image_id, embedding, crop_px, quality FROM faces
|
||||
WHERE model_id = ?1 ORDER BY id",
|
||||
)?;
|
||||
let rows = q.query_map([model_id], |r| {
|
||||
WHERE {} = ?1 ORDER BY id",
|
||||
embedder_sql("model_id")
|
||||
))?;
|
||||
let rows = q.query_map([embedder_of(model_id)], |r| {
|
||||
Ok(StoredEmbedding {
|
||||
face: FaceId(r.get::<_, i64>(0)? as u64),
|
||||
image: ImageId(r.get::<_, i64>(1)? as u64),
|
||||
@@ -823,6 +891,11 @@ pub fn images_for_person(
|
||||
// ── calibration ───────────────────────────────────────────────────────────
|
||||
|
||||
/// Store a fitted calibration, replacing any previous fit for the model.
|
||||
///
|
||||
/// Keyed on the embedder: a calibration is a fit over pairwise similarities,
|
||||
/// and those are the same space for every detector in front of one embedder.
|
||||
/// Keying it on the full pipeline would discard the fit — and the library's
|
||||
/// sharpened confidences with it — every time the detector was changed.
|
||||
pub fn put_calibration(
|
||||
conn: &Connection,
|
||||
model_id: &str,
|
||||
@@ -842,7 +915,7 @@ pub fn put_calibration(
|
||||
face_set_hash = excluded.face_set_hash,
|
||||
fitted_at = excluded.fitted_at",
|
||||
rusqlite::params![
|
||||
model_id,
|
||||
embedder_of(model_id),
|
||||
cal.a as f64,
|
||||
cal.b as f64,
|
||||
cal.w_size as f64,
|
||||
@@ -868,7 +941,7 @@ pub fn calibration(
|
||||
conn.query_row(
|
||||
"SELECT a, b, w_size, valid, positive_pairs, negative_pairs, face_set_hash
|
||||
FROM face_calibration WHERE model_id = ?1",
|
||||
[model_id],
|
||||
[embedder_of(model_id)],
|
||||
|r| {
|
||||
Ok((
|
||||
Calibration {
|
||||
@@ -958,8 +1031,11 @@ pub fn crop(conn: &Connection, face: FaceId) -> Result<Option<Vec<u8>>, CatalogE
|
||||
/// and what a "re-index to fill these in" prompt would be counting.
|
||||
pub fn faces_without_crop(conn: &Connection, model_id: &str) -> Result<u64, CatalogError> {
|
||||
conn.query_row(
|
||||
"SELECT COUNT(*) FROM faces WHERE model_id = ?1 AND crop IS NULL",
|
||||
[model_id],
|
||||
&format!(
|
||||
"SELECT COUNT(*) FROM faces WHERE {} = ?1 AND crop IS NULL",
|
||||
embedder_sql("model_id")
|
||||
),
|
||||
[embedder_of(model_id)],
|
||||
|r| r.get::<_, i64>(0),
|
||||
)
|
||||
.map(|n| n as u64)
|
||||
@@ -1271,6 +1347,64 @@ mod tests {
|
||||
assert_eq!(for_image(&c, img).unwrap().len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_embedder_is_the_half_after_the_plus() {
|
||||
assert_eq!(embedder_of("w600k_mbf"), "w600k_mbf");
|
||||
assert_eq!(embedder_of("scrfd_10g+w600k_mbf"), "w600k_mbf");
|
||||
assert_eq!(embedder_of("scrfd_2.5g+w600k_mbf"), "w600k_mbf");
|
||||
assert_eq!(embedder_of("scrfd_10g+other"), "other");
|
||||
}
|
||||
|
||||
/// A detector change must not split the library: the clustering pass,
|
||||
/// the coverage figure and the "is this indexed" question all see every
|
||||
/// generation that shares an embedder, and none of a different one.
|
||||
#[test]
|
||||
fn every_detector_in_front_of_one_embedder_is_one_population() {
|
||||
let c = db();
|
||||
let old = image(&c, 1);
|
||||
let new = image(&c, 2);
|
||||
let foreign = image(&c, 3);
|
||||
record_detections(&c, old, "w600k_mbf", 1024, &[face(1)]).unwrap();
|
||||
let mut f = face(2);
|
||||
f.model_id = "scrfd_10g+w600k_mbf".into();
|
||||
record_detections(&c, new, "scrfd_10g+w600k_mbf", 1024, &[f]).unwrap();
|
||||
let mut g = face(3);
|
||||
g.model_id = "scrfd_10g+other".into();
|
||||
record_detections(&c, foreign, "scrfd_10g+other", 1024, &[g]).unwrap();
|
||||
|
||||
for id in ["w600k_mbf", "scrfd_10g+w600k_mbf", "scrfd_2.5g+w600k_mbf"] {
|
||||
assert_eq!(embeddings(&c, id).unwrap().len(), 2, "{id}");
|
||||
assert_eq!(unassigned(&c, id).unwrap().len(), 2, "{id}");
|
||||
assert_eq!(coverage(&c, id).unwrap().indexed, 2, "{id}");
|
||||
assert!(is_indexed(&c, old, id).unwrap(), "{id}");
|
||||
assert!(is_indexed(&c, new, id).unwrap(), "{id}");
|
||||
assert!(!is_indexed(&c, foreign, id).unwrap(), "{id}");
|
||||
}
|
||||
assert_eq!(embeddings(&c, "scrfd_10g+other").unwrap().len(), 1);
|
||||
assert_eq!(coverage(&c, "scrfd_10g+other").unwrap().indexed, 1);
|
||||
}
|
||||
|
||||
/// The fit is over the embedder's similarity space, so it is the same fit
|
||||
/// whichever detector is chosen — and a detector change must not discard
|
||||
/// it.
|
||||
#[test]
|
||||
fn a_calibration_outlives_a_detector_change() {
|
||||
let c = db();
|
||||
let cal = Calibration {
|
||||
a: 1.5,
|
||||
b: -0.5,
|
||||
w_size: 0.1,
|
||||
valid: true,
|
||||
positive_pairs: 40,
|
||||
negative_pairs: 400,
|
||||
};
|
||||
put_calibration(&c, "w600k_mbf", &cal, "abc").unwrap();
|
||||
let (got, hash) = calibration(&c, "scrfd_10g+w600k_mbf").unwrap().unwrap();
|
||||
assert_eq!(got, cal);
|
||||
assert_eq!(hash, "abc");
|
||||
assert!(calibration(&c, "scrfd_10g+other").unwrap().is_none());
|
||||
}
|
||||
|
||||
/// The invariant FR-CULL-10 turns on: a re-index with a better model must
|
||||
/// not throw away the user's own labelling.
|
||||
#[test]
|
||||
@@ -1622,9 +1756,12 @@ mod tests {
|
||||
assert_eq!(edge, 2048, "the marker should record the newer proxy");
|
||||
}
|
||||
|
||||
/// A user who switches detector and back must not find the images the
|
||||
/// second pipeline visited reported as done under the first with no
|
||||
/// faces behind the marker.
|
||||
/// An image holds one pipeline's faces at a time, so the first pipeline's
|
||||
/// marker goes with its faces — but the image stays indexed for every
|
||||
/// pipeline sharing the embedder, because the faces behind the marker
|
||||
/// that remains are the same population. Switching detector and back
|
||||
/// therefore neither re-queues the image nor leaves a marker with nothing
|
||||
/// behind it.
|
||||
#[test]
|
||||
fn re_indexing_under_another_model_drops_the_first_models_marker() {
|
||||
let c = db();
|
||||
@@ -1632,13 +1769,24 @@ mod tests {
|
||||
record_detections(&c, img, "w600k_mbf", 2048, &[face(1)]).unwrap();
|
||||
record_detections(&c, img, "scrfd_2.5g+w600k_mbf", 2048, &[face(2), face(3)]).unwrap();
|
||||
|
||||
assert!(is_indexed(&c, img, "scrfd_2.5g+w600k_mbf").unwrap());
|
||||
assert!(
|
||||
!is_indexed(&c, img, "w600k_mbf").unwrap(),
|
||||
let markers: Vec<String> = {
|
||||
let mut q = c
|
||||
.prepare("SELECT model_id FROM face_index WHERE image_id = ?1")
|
||||
.unwrap();
|
||||
q.query_map([img.0 as i64], |r| r.get(0))
|
||||
.unwrap()
|
||||
.collect::<Result<_, _>>()
|
||||
.unwrap()
|
||||
};
|
||||
assert_eq!(
|
||||
markers,
|
||||
vec!["scrfd_2.5g+w600k_mbf".to_string()],
|
||||
"the first pipeline's marker outlived its faces"
|
||||
);
|
||||
assert!(is_indexed(&c, img, "scrfd_2.5g+w600k_mbf").unwrap());
|
||||
assert!(is_indexed(&c, img, "w600k_mbf").unwrap());
|
||||
assert_eq!(for_image(&c, img).unwrap().len(), 2);
|
||||
assert_eq!(coverage(&c, "w600k_mbf").unwrap().outstanding(), 1);
|
||||
assert_eq!(coverage(&c, "w600k_mbf").unwrap().outstanding(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -991,9 +991,13 @@ fn remote_has_column(tx: &Connection, table: &str, column: &str) -> Result<bool,
|
||||
/// Remote face row id to local face row id, by photograph and box overlap.
|
||||
///
|
||||
/// See [`merge_people_within`] for why a face has no shared identity and this
|
||||
/// has to be derived. Only faces from the same model are compared: boxes from
|
||||
/// two different detectors are not the same measurement, and matching across
|
||||
/// them would attach a judgement to a face nobody looked at.
|
||||
/// has to be derived. Faces are compared within an *embedder*
|
||||
/// (`faces::embedder_of`), not within an exact pipeline id: two detectors in
|
||||
/// front of the same embedder draw boxes around the same faces, and a
|
||||
/// confirmation made on one device's box is about the face, not the
|
||||
/// rectangle — the same judgement `faces::record_detections` makes when it
|
||||
/// carries a confirmation across a re-detection. Keying on the exact id was
|
||||
/// what let a detector change strand every name on the device that made it.
|
||||
fn match_faces(tx: &Connection) -> Result<std::collections::HashMap<i64, i64>, CatalogError> {
|
||||
/// Loose on purpose — "the same face in the frame", not "the same
|
||||
/// rectangle". The figure `record_detections` uses for the same job.
|
||||
@@ -1026,7 +1030,8 @@ fn match_faces(tx: &Connection) -> Result<std::collections::HashMap<i64, i64>, C
|
||||
})?;
|
||||
for row in rows {
|
||||
let (file_id, model, boxed) = row?;
|
||||
local.entry((file_id, model)).or_default().push(boxed);
|
||||
let embedder = crate::faces::embedder_of(&model).to_string();
|
||||
local.entry((file_id, embedder)).or_default().push(boxed);
|
||||
}
|
||||
}
|
||||
if local.is_empty() {
|
||||
@@ -1056,7 +1061,8 @@ fn match_faces(tx: &Connection) -> Result<std::collections::HashMap<i64, i64>, C
|
||||
|
||||
for row in rows {
|
||||
let (remote_id, file_id, model, rbox) = row?;
|
||||
let Some(candidates) = local.get(&(file_id, model)) else {
|
||||
let embedder = crate::faces::embedder_of(&model).to_string();
|
||||
let Some(candidates) = local.get(&(file_id, embedder)) else {
|
||||
continue;
|
||||
};
|
||||
let best = candidates
|
||||
@@ -1977,6 +1983,54 @@ mod tests {
|
||||
assert_eq!(person_of(&c, local), Some(("Anna".to_string(), true)));
|
||||
}
|
||||
|
||||
/// The bug this rule exists for: the desktop switched to a stronger
|
||||
/// detector and confirmed 3,500 faces under the old pipeline id; the
|
||||
/// tablet held the same faces under the new one, and not one name
|
||||
/// crossed, because the match demanded the exact id. Same photograph,
|
||||
/// same box, same embedder — that is the same face.
|
||||
#[test]
|
||||
fn a_confirmation_crosses_a_detector_change() {
|
||||
let c = two_catalogs();
|
||||
for db in ["main", "remote_cat"] {
|
||||
add_synced_image(&c, db, 1, 5000);
|
||||
}
|
||||
let local = add_face(&c, "main", 7, 1, 0.30);
|
||||
c.execute(
|
||||
"UPDATE main.faces SET model_id = 'scrfd_10g+w600k_mbf' WHERE id = ?1",
|
||||
[local],
|
||||
)
|
||||
.unwrap();
|
||||
let remote = add_face(&c, "remote_cat", 42, 1, 0.31);
|
||||
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
||||
assign(&c, "remote_cat", remote, 3, true);
|
||||
|
||||
let report = merge_all(&c).unwrap();
|
||||
assert_eq!(report.faces_assigned, 1);
|
||||
assert_eq!(person_of(&c, local), Some(("Anna".to_string(), true)));
|
||||
}
|
||||
|
||||
/// A different embedder is a different space, and a box there is a face
|
||||
/// nobody here has a vector for.
|
||||
#[test]
|
||||
fn a_confirmation_does_not_cross_an_embedder_change() {
|
||||
let c = two_catalogs();
|
||||
for db in ["main", "remote_cat"] {
|
||||
add_synced_image(&c, db, 1, 5000);
|
||||
}
|
||||
let local = add_face(&c, "main", 7, 1, 0.30);
|
||||
c.execute(
|
||||
"UPDATE main.faces SET model_id = 'scrfd_10g+other_embedder' WHERE id = ?1",
|
||||
[local],
|
||||
)
|
||||
.unwrap();
|
||||
let remote = add_face(&c, "remote_cat", 42, 1, 0.31);
|
||||
add_person(&c, "remote_cat", 3, "u-anna", "Anna", false);
|
||||
assign(&c, "remote_cat", remote, 3, true);
|
||||
|
||||
merge_all(&c).unwrap();
|
||||
assert_eq!(person_of(&c, local), None, "matched across embedders");
|
||||
}
|
||||
|
||||
/// Boxes from two devices are close but not identical. Matching has to be
|
||||
/// by overlap, not equality, or nothing ever lines up.
|
||||
#[test]
|
||||
|
||||
@@ -186,17 +186,26 @@ pub struct FaceSettings {
|
||||
/// and a tablet on a battery want different answers — so it is a setting,
|
||||
/// per device, like the rest of this file.
|
||||
///
|
||||
/// # A detector is half of a model id
|
||||
/// # A detector is half of a model id, and the half that does not split
|
||||
/// # the library
|
||||
///
|
||||
/// Every face row, run marker, sync shard and calibration is keyed by
|
||||
/// `faces.model_id` (catalog.md §10.1), and the schema's whole reason for
|
||||
/// carrying that column is that *a model change is a new id and a re-index*
|
||||
/// rather than a silent change under existing data. A detector change is a
|
||||
/// model change: it decides which faces exist and where the landmarks that
|
||||
/// align them land. So each variant names its own pipeline, and choosing
|
||||
/// another one puts every image back in the queue and shows the People screen
|
||||
/// for the new pipeline — empty until the sweep has run, with confirmed names
|
||||
/// carried across by `record_detections`' box overlap.
|
||||
/// Every face row, run marker, sync shard and calibration carries
|
||||
/// `faces.model_id` (catalog.md §10.1), spelled `detector+embedder`, so which
|
||||
/// pipeline produced a face is always on record. But every reader of "the
|
||||
/// faces" keys on the *embedder* half (`dr_catalog::faces::embedder_of`):
|
||||
/// the embedder is what makes two vectors comparable, and the detector only
|
||||
/// decides where the boxes are. So the three variants here are one
|
||||
/// population, and choosing another one empties nothing — the People screen,
|
||||
/// the clustering and the sync all go on over every face already found.
|
||||
/// What a stronger choice does is queue the images a weaker one indexed for
|
||||
/// re-detection, after the ones nothing has indexed ([`Self::supersedes`]),
|
||||
/// with confirmed names carried across by `record_detections`' box overlap.
|
||||
///
|
||||
/// The first version of this setting keyed everything on the full id, and
|
||||
/// choosing Thorough emptied both devices' People screens until a whole-
|
||||
/// library re-index had run on each — and, worse, stranded every name
|
||||
/// confirmed on one device, because the other held the same faces under a
|
||||
/// different id and the merge would not match them.
|
||||
///
|
||||
/// The first variant's id is the bare embedder name, because that is the id
|
||||
/// every library indexed before this setting existed was written under;
|
||||
@@ -251,6 +260,41 @@ impl FaceDetector {
|
||||
}
|
||||
}
|
||||
|
||||
/// The detector that writes under a pipeline id, if it is one of these.
|
||||
///
|
||||
/// The inverse of [`Self::model_id`]. `None` for an id from another
|
||||
/// embedder or a build this one does not know, which a caller treats as
|
||||
/// "cannot rank" rather than as weaker than anything.
|
||||
pub fn for_model_id(model_id: &str) -> Option<FaceDetector> {
|
||||
FaceDetector::ALL
|
||||
.into_iter()
|
||||
.find(|d| d.model_id() == model_id)
|
||||
}
|
||||
|
||||
/// Whether this detector finds more than `other` does — the measured
|
||||
/// order of §12.3, which is also the order of [`Self::ALL`].
|
||||
pub fn outranks(self, other: FaceDetector) -> bool {
|
||||
let rank = |d: FaceDetector| FaceDetector::ALL.iter().position(|x| *x == d);
|
||||
rank(self) > rank(other)
|
||||
}
|
||||
|
||||
/// The pipeline ids this detector is worth re-running over.
|
||||
///
|
||||
/// Every variant shares one embedder, so a library indexed under any of
|
||||
/// them is one population (`dr_catalog::faces::embedder_of`) and a
|
||||
/// detector change empties nothing. What a stronger detector *adds* is
|
||||
/// the faces a weaker one missed, so a sweep re-detects the images a
|
||||
/// weaker one indexed — after the ones nothing has indexed — and never
|
||||
/// the other way round: a tablet set to Fast keeps the desktop's
|
||||
/// Thorough faces rather than replacing them with fewer.
|
||||
pub fn supersedes(self) -> &'static [&'static str] {
|
||||
match self {
|
||||
FaceDetector::Scrfd500m => &[],
|
||||
FaceDetector::Scrfd2_5g => &["w600k_mbf"],
|
||||
FaceDetector::Scrfd10g => &["w600k_mbf", "scrfd_2.5g+w600k_mbf"],
|
||||
}
|
||||
}
|
||||
|
||||
/// What the picker calls it.
|
||||
pub fn label(self) -> &'static str {
|
||||
match self {
|
||||
@@ -1682,6 +1726,26 @@ mod tests {
|
||||
/// The detector every existing library was indexed with must keep the id
|
||||
/// those libraries were written under, or an upgrade would report every
|
||||
/// one of them un-indexed.
|
||||
#[test]
|
||||
fn detectors_rank_in_the_measured_order_and_only_supersede_downwards() {
|
||||
use FaceDetector::*;
|
||||
assert!(Scrfd10g.outranks(Scrfd2_5g));
|
||||
assert!(Scrfd2_5g.outranks(Scrfd500m));
|
||||
assert!(!Scrfd500m.outranks(Scrfd10g));
|
||||
assert!(!Scrfd10g.outranks(Scrfd10g));
|
||||
for d in FaceDetector::ALL {
|
||||
for weaker in d.supersedes() {
|
||||
let w = FaceDetector::for_model_id(weaker).expect(weaker);
|
||||
assert!(
|
||||
d.outranks(w),
|
||||
"{d:?} lists {weaker} but does not outrank it"
|
||||
);
|
||||
}
|
||||
assert_eq!(FaceDetector::for_model_id(d.model_id()), Some(d));
|
||||
}
|
||||
assert_eq!(FaceDetector::for_model_id("scrfd_10g+other"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_default_detector_keeps_the_legacy_model_id() {
|
||||
assert_eq!(FaceDetector::default(), FaceDetector::Scrfd500m);
|
||||
|
||||
+29
-10
@@ -540,8 +540,8 @@ images indexed, **64 contain no face at all**.
|
||||
|
||||
So `face_index` records the *run*: one row per `(image, model)` carrying the timestamp, the number of
|
||||
faces found — zero is the interesting value — and the long edge of the proxy it read. Keyed on the
|
||||
model, so a model change puts every image back in the queue without anyone having to remember to
|
||||
clear anything.
|
||||
model, so an *embedder* change puts every image back in the queue without anyone having to remember
|
||||
to clear anything; a detector change in front of the same embedder does not (§12.3).
|
||||
|
||||
Three things fall out of it that were not otherwise available:
|
||||
|
||||
@@ -1100,14 +1100,33 @@ is not a desktop-only feature (NFR-RES-2). It loads in tract with the same fix a
|
||||
decodes through the same nine-output path unchanged. Same licence, same `buffalo_m` release page.
|
||||
|
||||
**Which detector runs is a setting** — `FaceSettings::detector`, per device, on the settings page
|
||||
beside the indexing button as Fast / Balanced / Thorough. All three files ship. Each detector is
|
||||
its own `faces.model_id` (`w600k_mbf` for `500M`, unchanged, so nothing already indexed is
|
||||
disturbed; `scrfd_2.5g+w600k_mbf` and `scrfd_10g+w600k_mbf` for the others), which is the
|
||||
mechanism §2.1 always intended for a model change: the coverage figure restarts at zero under the
|
||||
new id, the sweep re-detects, `record_detections` carries confirmed names across by box overlap and
|
||||
drops the previous pipeline's marker for each image it revisits, and the sync shards are keyed by
|
||||
the same id so a peer on another setting neither adopts nor pollutes them. The default stays `500M`
|
||||
so that an upgrade changes nothing until the user chooses; the recommendation is `2.5G`.
|
||||
beside the indexing button as Fast / Balanced / Thorough. All three files ship. Each detector
|
||||
writes its own `faces.model_id` (`w600k_mbf` for `500M`, unchanged; `scrfd_2.5g+w600k_mbf` and
|
||||
`scrfd_10g+w600k_mbf` for the others), so which pipeline drew a box is always on record. The
|
||||
default stays `500M` so that an upgrade changes nothing until the user chooses; the recommendation
|
||||
is `2.5G`.
|
||||
|
||||
**One population per embedder, not one per detector · 2026-09-19.** The first cut of the setting
|
||||
keyed every reader on the full id — the clustering pass, the coverage figure, the sweep's work
|
||||
list, the shard export and import, and the sync merge's face matching — on the theory that a
|
||||
detector change is a model change. Measured on the reference library it was a disaster: choosing
|
||||
Thorough on both devices restarted coverage at 1,834 of 19,140, the People screen showed only the
|
||||
faces the new pipeline had reached, the desktop's 3,583 confirmations under the old id could not
|
||||
reach the tablet because the merge demanded the same id on both sides, and each device faced a
|
||||
~400 GB re-fetch before the library looked whole again. The embedder is `w600k_mbf` in every
|
||||
variant; its vectors are one space, and the detector only decides where the boxes are.
|
||||
|
||||
So every reader now keys on the embedder half of the id (`faces::embedder_of`, and
|
||||
`embedder_sql` for the queries): all three detectors are one population, and changing between
|
||||
them empties nothing. `record_detections` is unchanged — an image holds one pipeline's faces at a
|
||||
time, and a re-detection carries confirmations across by box overlap — and it is where the
|
||||
generations meet. The merge's `match_faces` matches within an embedder for the same reason. The
|
||||
shards travel every generation, each under its own id, and a peer adopts whichever it is sent.
|
||||
What a stronger choice still does is queue the images a weaker detector indexed for re-detection
|
||||
(`FaceDetector::supersedes`), after the ones nothing has indexed and never downwards, so a tablet
|
||||
on Fast keeps the desktop's Thorough faces rather than replacing them with fewer. The calibration
|
||||
(§8) is keyed on the embedder too: it is a fit over the similarity space, and that space did not
|
||||
change.
|
||||
|
||||
---
|
||||
|
||||
|
||||
+28
-28
File diff suppressed because one or more lines are too long
+13
-10
@@ -91,14 +91,15 @@ pub struct FaceRequest {
|
||||
/// examined, so every landscape in the library would be re-detected on every
|
||||
/// run, for ever. See the V9 migration.
|
||||
///
|
||||
/// Keyed on the model, so a model upgrade re-indexes rather than leaving the
|
||||
/// library half-described by weights that are no longer comparable.
|
||||
/// Keyed on the embedder, so an embedder upgrade re-indexes rather than
|
||||
/// leaving the library half-described by weights that are no longer
|
||||
/// comparable — and a detector change, which keeps the embedder, does not.
|
||||
pub fn faces_outstanding(
|
||||
catalog: &Catalog,
|
||||
store: &ThumbStore,
|
||||
model_id: &str,
|
||||
) -> Result<Vec<FaceRequest>, dr_catalog::CatalogError> {
|
||||
let mut stmt = catalog.connection().prepare(
|
||||
let mut stmt = catalog.connection().prepare(&format!(
|
||||
"SELECT i.id, r.file_id
|
||||
FROM images i
|
||||
JOIN remote r ON r.image_id = i.id
|
||||
@@ -107,12 +108,13 @@ pub fn faces_outstanding(
|
||||
AND i.shadowed_by IS NULL
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM face_index fi
|
||||
WHERE fi.image_id = i.id AND fi.model_id = ?1
|
||||
WHERE fi.image_id = i.id AND {} = ?1
|
||||
)
|
||||
ORDER BY i.id",
|
||||
)?;
|
||||
faces::embedder_sql("fi.model_id")
|
||||
))?;
|
||||
let rows = stmt
|
||||
.query_map([model_id], |r| {
|
||||
.query_map([faces::embedder_of(model_id)], |r| {
|
||||
Ok(FaceRequest {
|
||||
image_id: ImageId(r.get::<_, i64>(0)? as u64),
|
||||
file_id: r.get::<_, i64>(1)? as u64,
|
||||
@@ -209,7 +211,7 @@ pub fn audit(
|
||||
// for the same reason `faces::coverage` excludes them: they are the JPEG
|
||||
// half of a pair, no sweep will ever index one, and counting them makes
|
||||
// the outstanding figure a number that cannot reach zero.
|
||||
let mut stmt = conn.prepare(
|
||||
let mut stmt = conn.prepare(&format!(
|
||||
"SELECT r.file_id
|
||||
FROM images i
|
||||
JOIN remote r ON r.image_id = i.id
|
||||
@@ -218,12 +220,13 @@ pub fn audit(
|
||||
AND i.shadowed_by IS NULL
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM face_index fi
|
||||
WHERE fi.image_id = i.id AND fi.model_id = ?1
|
||||
WHERE fi.image_id = i.id AND {} = ?1
|
||||
)",
|
||||
)?;
|
||||
faces::embedder_sql("fi.model_id")
|
||||
))?;
|
||||
let (mut ready, mut awaiting) = (0u64, 0u64);
|
||||
for file_id in stmt
|
||||
.query_map([model_id], |r| r.get::<_, i64>(0))?
|
||||
.query_map([faces::embedder_of(model_id)], |r| r.get::<_, i64>(0))?
|
||||
.filter_map(Result::ok)
|
||||
{
|
||||
if store.contains(file_id as u64, FACE_TIER) {
|
||||
|
||||
@@ -1089,6 +1089,14 @@ pub fn wire<S, M, P>(
|
||||
detector,
|
||||
embedder,
|
||||
model_id(&settings_for_sweep),
|
||||
settings_for_sweep
|
||||
.snapshot()
|
||||
.faces
|
||||
.detector
|
||||
.supersedes()
|
||||
.iter()
|
||||
.map(|m| m.to_string())
|
||||
.collect(),
|
||||
dr_face::DetectOptions::default(),
|
||||
gpu,
|
||||
));
|
||||
|
||||
+123
-10
@@ -3800,16 +3800,18 @@ fn faces_unindexed(
|
||||
WHERE r.file_id IS NOT NULL AND {VISIBLE}
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM face_index fi
|
||||
WHERE fi.image_id = i.id AND fi.model_id = ?1
|
||||
WHERE fi.image_id = i.id AND {fi_embedder} = ?1
|
||||
)
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM faces f
|
||||
WHERE f.image_id = i.id AND f.model_id = ?1 AND f.quality IS NULL
|
||||
WHERE f.image_id = i.id AND {f_embedder} = ?1 AND f.quality IS NULL
|
||||
)
|
||||
ORDER BY i.id"
|
||||
ORDER BY i.id",
|
||||
fi_embedder = dr_catalog::faces::embedder_sql("fi.model_id"),
|
||||
f_embedder = dr_catalog::faces::embedder_sql("f.model_id"),
|
||||
))?;
|
||||
let rows = stmt
|
||||
.query_map([model_id], |r| {
|
||||
.query_map([dr_catalog::faces::embedder_of(model_id)], |r| {
|
||||
Ok(ThumbnailRequest {
|
||||
// Face indexing wants the detail a thumbnail discards.
|
||||
full_resolution: true,
|
||||
@@ -3853,11 +3855,12 @@ fn faces_without_proxy(
|
||||
FROM images i
|
||||
JOIN remote r ON r.image_id = i.id
|
||||
JOIN faces f ON f.image_id = i.id
|
||||
WHERE r.file_id IS NOT NULL AND {VISIBLE} AND f.model_id = ?1
|
||||
ORDER BY i.id"
|
||||
WHERE r.file_id IS NOT NULL AND {VISIBLE} AND {embedder} = ?1
|
||||
ORDER BY i.id",
|
||||
embedder = dr_catalog::faces::embedder_sql("f.model_id"),
|
||||
))?;
|
||||
let rows = stmt
|
||||
.query_map([model_id], |r| {
|
||||
.query_map([dr_catalog::faces::embedder_of(model_id)], |r| {
|
||||
Ok(ThumbnailRequest {
|
||||
full_resolution: true,
|
||||
thumb_size: dr_thumbs::ThumbSize::Large,
|
||||
@@ -3902,11 +3905,66 @@ fn faces_unmeasured(
|
||||
JOIN remote r ON r.image_id = i.id
|
||||
JOIN faces f ON f.image_id = i.id
|
||||
WHERE r.file_id IS NOT NULL AND {VISIBLE}
|
||||
AND f.model_id = ?1 AND f.quality IS NULL
|
||||
AND {embedder} = ?1 AND f.quality IS NULL
|
||||
ORDER BY i.id",
|
||||
embedder = dr_catalog::faces::embedder_sql("f.model_id"),
|
||||
))?;
|
||||
let rows = stmt
|
||||
.query_map([dr_catalog::faces::embedder_of(model_id)], |r| {
|
||||
Ok(ThumbnailRequest {
|
||||
full_resolution: true,
|
||||
thumb_size: dr_thumbs::ThumbSize::Large,
|
||||
row: 0,
|
||||
image_id: r.get(0)?,
|
||||
path: r.get(1)?,
|
||||
file_id: r.get::<_, Option<i64>>(2)?.map(|v| v as u64),
|
||||
size: r.get::<_, Option<i64>>(3)?.unwrap_or(0) as u64,
|
||||
needs_metadata: false,
|
||||
})
|
||||
})?
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
/// Images indexed by a detector the chosen one outranks.
|
||||
///
|
||||
/// The tail of the sweep's work list: behind everything nothing has looked
|
||||
/// at, because a photograph with no faces recorded is worth more than one
|
||||
/// whose faces a weaker detector may have under-counted. `weaker` is
|
||||
/// `FaceDetector::supersedes` for the current choice, and it is a list of
|
||||
/// exact ids rather than "anything else sharing the embedder" so the
|
||||
/// re-detection only ever runs upwards — a device set to the fast detector
|
||||
/// leaves a peer's thorough pass alone.
|
||||
///
|
||||
/// The user's confirmations survive the re-detection by box overlap
|
||||
/// (`faces::record_detections`), which is what makes this safe to run over a
|
||||
/// library that is already named.
|
||||
fn faces_superseded(
|
||||
catalog: &Catalog,
|
||||
weaker: &[&str],
|
||||
) -> Result<Vec<ThumbnailRequest>, dr_catalog::CatalogError> {
|
||||
if weaker.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let placeholders = weaker
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, _)| format!("?{}", i + 1))
|
||||
.collect::<Vec<_>>()
|
||||
.join(", ");
|
||||
let mut stmt = catalog.connection().prepare(&format!(
|
||||
"SELECT i.id, i.source_ref, r.file_id, i.file_size
|
||||
FROM images i
|
||||
JOIN remote r ON r.image_id = i.id
|
||||
WHERE r.file_id IS NOT NULL AND {VISIBLE}
|
||||
AND EXISTS (
|
||||
SELECT 1 FROM face_index fi
|
||||
WHERE fi.image_id = i.id AND fi.model_id IN ({placeholders})
|
||||
)
|
||||
ORDER BY i.id"
|
||||
))?;
|
||||
let rows = stmt
|
||||
.query_map([model_id], |r| {
|
||||
.query_map(rusqlite::params_from_iter(weaker.iter()), |r| {
|
||||
Ok(ThumbnailRequest {
|
||||
full_resolution: true,
|
||||
thumb_size: dr_thumbs::ThumbSize::Large,
|
||||
@@ -3981,6 +4039,7 @@ pub fn spawn_face_sweep(
|
||||
detector_model: PathBuf,
|
||||
embedder_model: PathBuf,
|
||||
model_id: String,
|
||||
supersedes: Vec<String>,
|
||||
options: dr_face::DetectOptions,
|
||||
gpu: dr_gpu::GpuContext,
|
||||
) -> Receiver<crate::faces::FaceSweepMessage> {
|
||||
@@ -4107,13 +4166,37 @@ pub fn spawn_face_sweep(
|
||||
}
|
||||
}
|
||||
|
||||
// Last: what a weaker detector already indexed. Everything above is
|
||||
// a photograph the People screen cannot show at all; these it shows
|
||||
// already, and the re-detection only adds the faces that detector
|
||||
// missed. See `faces_superseded`.
|
||||
let weaker: Vec<&str> = supersedes.iter().map(String::as_str).collect();
|
||||
let mut upgrades = 0usize;
|
||||
match faces_superseded(&catalog, &weaker) {
|
||||
Ok(older) => {
|
||||
let fresh: Vec<_> = older
|
||||
.into_iter()
|
||||
.filter(|r| queued.insert(r.image_id))
|
||||
.collect();
|
||||
upgrades = fresh.len();
|
||||
wanted.extend(fresh.into_iter().map(|r| (r, SweepWork::Detect)));
|
||||
}
|
||||
Err(e) => log::warn!("face sweep: looking for images under a weaker detector: {e}"),
|
||||
}
|
||||
|
||||
let total = wanted.len();
|
||||
if total == 0 {
|
||||
log::info!("face sweep: every image has been through this model");
|
||||
finish_empty(&tx);
|
||||
return;
|
||||
}
|
||||
log::info!("face sweep: {total} image(s) to index");
|
||||
if upgrades > 0 {
|
||||
log::info!(
|
||||
"face sweep: {total} image(s) to index, {upgrades} of them indexed by a weaker detector"
|
||||
);
|
||||
} else {
|
||||
log::info!("face sweep: {total} image(s) to index");
|
||||
}
|
||||
if tx.send(FaceSweepMessage::Total(total)).is_err() {
|
||||
return;
|
||||
}
|
||||
@@ -6739,6 +6822,36 @@ mod tests {
|
||||
assert_eq!(faces_unindexed(&catalog, "other").unwrap().len(), 3);
|
||||
}
|
||||
|
||||
/// A detector change keeps the embedder, so it is not a new library: the
|
||||
/// images the old detector ran over are not "unindexed" for the new one.
|
||||
/// They are an *upgrade*, listed separately and only when the chosen
|
||||
/// detector outranks the one that indexed them — never the other way,
|
||||
/// or a tablet on the fast detector would undo the desktop's thorough
|
||||
/// pass.
|
||||
#[test]
|
||||
fn a_stronger_detector_upgrades_rather_than_re_indexes() {
|
||||
let catalog = with_images(3);
|
||||
let ids = image_ids(&catalog);
|
||||
let conn = catalog.connection();
|
||||
dr_catalog::faces::record_detections(conn, ids[0], "w600k_mbf", 1024, &[]).unwrap();
|
||||
dr_catalog::faces::record_detections(conn, ids[1], "scrfd_10g+w600k_mbf", 1024, &[])
|
||||
.unwrap();
|
||||
|
||||
let fresh = faces_unindexed(&catalog, "scrfd_10g+w600k_mbf").unwrap();
|
||||
assert_eq!(
|
||||
fresh.len(),
|
||||
1,
|
||||
"the earlier detector's images were re-queued"
|
||||
);
|
||||
assert_eq!(fresh[0].image_id, ids[2].0 as i64);
|
||||
|
||||
// Thorough re-runs what Fast did; Fast leaves what Thorough did alone.
|
||||
let up = faces_superseded(&catalog, &["w600k_mbf", "scrfd_2.5g+w600k_mbf"]).unwrap();
|
||||
assert_eq!(up.len(), 1);
|
||||
assert_eq!(up[0].image_id, ids[0].0 as i64);
|
||||
assert!(faces_superseded(&catalog, &[]).unwrap().is_empty());
|
||||
}
|
||||
|
||||
/// The state schema V14 leaves: a face with no quality and an image with
|
||||
/// no marker. It is the measuring pass's work, and *only* that pass's — a
|
||||
/// full re-detection of the same image would throw away every suggestion
|
||||
|
||||
Reference in New Issue
Block a user