Count the outstanding repairs from the faces, on partial indexes

"How many images still owe a quality reading" was a correlated EXISTS per
image over `faces`, and the face row is 8 KB of embedding and crop before
the column it looks at, so each count opened every row. Six such counts
run on every open of the Identity screen and at the end of every sweep:
160 ms on the reference library.

V19 adds three partial indexes holding only the faces still owing each
pass, keyed on the image and carrying the model id the predicate reads,
and replaces `faces_image` with `(image_id, model_id)` so "does this image
hold this embedder's faces" is answered from the index too. The planner
takes a partial index when the count is driven from `faces` and ignores it
inside the EXISTS, so `Needs::Face` carries the per-face fragment and
`repairs::count` spells the query from the faces' side; the list and the
per-image check keep the EXISTS. A test holds the two spellings to the
same answer for every repair.
This commit is contained in:
2026-09-20 10:56:37 +02:00
parent 73845d8a77
commit 388bda6af3
3 changed files with 199 additions and 51 deletions
+127 -24
View File
@@ -82,10 +82,28 @@ pub enum Needs {
/// image still owes it. Evaluated for the list, for the count, and again
/// per image before the handler runs.
Sql(String),
/// SQL over `faces f`, true where the face still owes it; the image owes
/// the repair if any of its faces does.
///
/// Kept as the per-face fragment rather than folded into an image
/// predicate, because the two questions asked of it want opposite
/// shapes. The list and the per-image check want `EXISTS (... WHERE
/// f.image_id = i.id AND fragment)`, one probe per image. The count
/// wants to start from the faces, where the partial indexes V19 keeps
/// for exactly these fragments make it a walk over the few thousand
/// still owing rather than a probe into eight-kilobyte rows for every
/// image in the library -- and the planner will not use those indexes
/// from inside the EXISTS.
Face(String),
/// Evaluated once, at the start of the job.
Set(SetFn),
}
/// [`Needs::Face`] as an image predicate: the image holds a face owing it.
fn any_face(fragment: &str) -> String {
format!("EXISTS (SELECT 1 FROM faces f WHERE f.image_id = i.id AND {fragment})")
}
/// A handler: fill one image, given what was fetched for it.
pub type ApplyFn = fn(&mut Toolkit, &Catalog, &Target, &mut Fetched) -> Result<usize, Failure>;
@@ -290,11 +308,10 @@ pub fn registry(
.join(", ");
format!("EXISTS (SELECT 1 FROM face_index fi WHERE fi.image_id = i.id AND fi.model_id IN ({list}))")
};
let face_needing = |pred: &str| {
format!(
"EXISTS (SELECT 1 FROM faces f WHERE f.image_id = i.id AND {f_embedder} = '{embedder}' AND ({pred}))"
)
};
// The per-face fragment `Needs::Face` carries: this embedder's face,
// still owing the pass. Spelled as the partial indexes' WHERE clauses
// are (V19), which is what lets the count be served from them.
let face_needing = |pred: &str| format!("{f_embedder} = '{embedder}' AND ({pred})");
let can_detect = caps.gpu && caps.face_models;
@@ -340,7 +357,7 @@ pub fn registry(
out.push(Repair {
name: "face-quality",
label: "images with faces to read for quality",
needs: Needs::Sql(face_needing(NEEDS_QUALITY)),
needs: Needs::Face(face_needing(NEEDS_QUALITY)),
input: Input::NativeRender,
apply: quality,
give_up: None,
@@ -350,7 +367,7 @@ pub fn registry(
out.push(Repair {
name: "face-eyes",
label: "images with faces to read for eye state",
needs: Needs::Sql(face_needing(NEEDS_EYES)),
needs: Needs::Face(face_needing(NEEDS_EYES)),
input: Input::NativeRender,
apply: eyes,
give_up: None,
@@ -360,7 +377,7 @@ pub fn registry(
out.push(Repair {
name: "face-crop",
label: "images with faces without a crop",
needs: Needs::Sql(face_needing(NEEDS_CROP)),
needs: Needs::Face(face_needing(NEEDS_CROP)),
input: Input::NativeRender,
apply: crop,
give_up: None,
@@ -760,6 +777,7 @@ fn listed(
) -> Result<Vec<Target>, dr_catalog::CatalogError> {
let (predicate, set) = match &repair.needs {
Needs::Sql(sql) => (sql.clone(), None),
Needs::Face(fragment) => (any_face(fragment), None),
Needs::Set(f) => ("1".to_string(), Some(f(catalog, store)?)),
};
let mut stmt = catalog.connection().prepare(&format!(
@@ -792,21 +810,23 @@ fn still_owed(
set: Option<&HashSet<i64>>,
image: ImageId,
) -> bool {
match (&repair.needs, set) {
(Needs::Set(_), Some(s)) => s.contains(&(image.0 as i64)),
(Needs::Set(_), None) => false,
(Needs::Sql(sql), _) => catalog
.connection()
.query_row(
&format!(
"SELECT EXISTS (SELECT 1 FROM images i JOIN remote r ON r.image_id = i.id
WHERE i.id = ?1 AND ({sql}))"
),
[image.0 as i64],
|r| r.get::<_, bool>(0),
)
.unwrap_or(false),
}
let sql = match (&repair.needs, set) {
(Needs::Set(_), Some(s)) => return s.contains(&(image.0 as i64)),
(Needs::Set(_), None) => return false,
(Needs::Sql(sql), _) => sql.clone(),
(Needs::Face(fragment), _) => any_face(fragment),
};
catalog
.connection()
.query_row(
&format!(
"SELECT EXISTS (SELECT 1 FROM images i JOIN remote r ON r.image_id = i.id
WHERE i.id = ?1 AND ({sql}))"
),
[image.0 as i64],
|r| r.get::<_, bool>(0),
)
.unwrap_or(false)
}
/// How many images each repair still lists, for the settings line and the
@@ -849,6 +869,21 @@ fn count(
)?;
Ok(n as u64)
}
// From the faces, not the images: see `Needs::Face`.
Needs::Face(fragment) => {
let n: i64 = catalog.connection().query_row(
&format!(
"SELECT COUNT(DISTINCT f.image_id)
FROM faces f
JOIN images i ON i.id = f.image_id
JOIN remote r ON r.image_id = i.id
WHERE {fragment} AND r.file_id IS NOT NULL AND {VISIBLE}"
),
[],
|r| r.get(0),
)?;
Ok(n as u64)
}
// The set is built from its own query and may name images `listed`
// would not visit, so it is intersected with the same base rather
// than trusted for its size.
@@ -898,7 +933,7 @@ fn plan(
for repair in repairs {
let set = match &repair.needs {
Needs::Set(f) => Some(f(catalog, store)?),
Needs::Sql(_) => None,
Needs::Sql(_) | Needs::Face(_) => None,
};
let listed = listed(catalog, store, repair)?;
if !listed.is_empty() {
@@ -1456,6 +1491,74 @@ mod tests {
let _ = std::fs::remove_dir_all(dir);
}
/// `counts` answers from the faces, `listed` from the images (see
/// `Needs::Face`), and the two spellings of each predicate have to
/// agree -- for every repair, on a library where each has something to
/// do and something already done.
#[test]
fn counts_are_the_sizes_of_the_lists() {
let catalog = with_images(5);
let ids = image_ids(&catalog);
let (store, dir) = store();
let conn = catalog.connection();
// 0: two faces, one measured, neither read for eyes, one without a
// crop -- the per-face repairs disagree about it face by face.
faces::record_detections(
conn,
ids[0],
"w600k_mbf",
4000,
&[
face("w600k_mbf", None, vec![1]),
face("w600k_mbf", Some(18.0), Vec::new()),
],
)
.unwrap();
// 1: done, under the chosen detector.
faces::record_detections(
conn,
ids[1],
"scrfd_10g+w600k_mbf",
4000,
&[complete("scrfd_10g+w600k_mbf")],
)
.unwrap();
// 2: examined by a weaker detector, nothing found.
faces::record_detections(conn, ids[2], "w600k_mbf", 4000, &[]).unwrap();
// 3, 4: never examined.
let repairs = registry(
Scope::Outstanding,
"scrfd_10g+w600k_mbf",
FaceDetector::Scrfd10g,
ALL,
);
let counted = counts(&catalog, &store, &repairs).unwrap();
for (repair, (label, n)) in repairs.iter().zip(counted) {
assert_eq!(label, repair.label);
let list = listed(&catalog, &store, repair).unwrap();
assert_eq!(n as usize, list.len(), "{}", repair.name);
}
// And the fixture exercised what it claims to.
let names: Vec<&str> = repairs.iter().map(|r| r.name).collect();
for name in [
"face-quality",
"face-eyes",
"face-crop",
"face-detection",
"face-upgrade",
] {
assert!(names.contains(&name), "{name} missing from the registry");
assert!(
!listed(&catalog, &store, by_name(&repairs, name))
.unwrap()
.is_empty(),
"{name} has nothing to do"
);
}
let _ = std::fs::remove_dir_all(dir);
}
/// The state schema V14 leaves: a face with no quality and an image
/// with no marker. It is the quality repair's work, and *only* that
/// repair's -- a full re-detection of the same image would throw away