From 9e786532a1796db12284eebe9e3b3a64d918c802 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Fri, 25 Sep 2026 22:04:42 -0400 Subject: [PATCH] Look for orphaned keywords once per word, not once per assignment `adopt_orphan_terms` runs in the backfill on every catalog open. Its check -- is there a vocabulary row for this word, tombstones included -- cannot use `keyword_terms_name`, which is partial on `deleted = 0`, so the correlated subquery scanned the vocabulary once for each of the 10,800 assignment rows before `DISTINCT` threw the repeats away: 3.5 ms per open on the reference library. The distinct words are taken first and the check runs once per word -- a few dozen scans of a few dozen rows. Same rows out, since `DISTINCT` over the assignments is exactly the set of words. catalog_bench, best of 20: 3.45 ms -> 0.30 ms. --- core/dr-catalog/src/keywords.rs | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/core/dr-catalog/src/keywords.rs b/core/dr-catalog/src/keywords.rs index 4326234..84e1b8f 100644 --- a/core/dr-catalog/src/keywords.rs +++ b/core/dr-catalog/src/keywords.rs @@ -491,7 +491,12 @@ pub fn adopt_orphan_terms(conn: &Connection) -> Result { // quietly readmitted to the vocabulary; it stays visible as an // orphan in [`for_images`] instead, which is a state someone can // see and act on rather than one that silently undoes a deletion. - "SELECT DISTINCT k.keyword FROM keywords k + // + // The distinct words first, then the check: the vocabulary has no + // index a tombstone-inclusive lookup can use, so checking once per + // assignment scanned it 10,000 times on every open. Once per word + // is a few dozen scans of a few dozen rows. + "SELECT k.keyword FROM (SELECT DISTINCT keyword FROM keywords) k WHERE NOT EXISTS (SELECT 1 FROM keyword_terms t WHERE t.name = k.keyword)", )?;