Write a scan's findings with prepared statements, and its jobs in the same commit
`persist` runs after every scan, for every photograph the scan listed. On a settled library that is the folders whose ETag changed -- a sidecar written there by a rating is enough -- so one relisted folder of 1,600 images is an ordinary pass, and a first scan is all 24,000. Per photograph it prepared four statements from their SQL (a folder lookup, the image upsert, the id read-back, the remote upsert) and then, after the commit, found the image again by path and enqueued its thumbnail job as an autocommitting statement of its own -- a commit per photograph, for rows that were almost all already queued. Now the statements are prepared once per pass, a folder's id is looked up once per folder rather than once per photograph in it, and the job is enqueued inside the transaction with the id already in hand. That also makes the job atomic with the row it points at, which is what the old ordering after the commit was trying to guarantee. `jobs::enqueue` uses a cached statement for the same reason. persist_bench on a copy of the reference catalog, CPU, best of runs: largest folder (1,589 images) 102-118 ms -> 10-13 ms whole library (23,582 images) 1.55-2.19 s -> 188-192 ms The fingerprint of images, remote, jobs and folders after the run is the same for both builds.
This commit is contained in:
@@ -603,103 +603,120 @@ pub(super) fn persist(
|
||||
|
||||
// Folder ETags first — without these persisted, the next scan prunes
|
||||
// nothing and walks the whole tree again (ARCH §6.6).
|
||||
for (path, validator) in &result.directories {
|
||||
tx.execute(
|
||||
{
|
||||
let mut folder = tx.prepare_cached(
|
||||
"INSERT INTO folders(root_id, path, etag) VALUES (?1, ?2, ?3)
|
||||
ON CONFLICT(root_id, path) DO UPDATE SET etag = excluded.etag",
|
||||
rusqlite::params![root_id, path.as_str(), validator.as_str()],
|
||||
)?;
|
||||
for (path, validator) in &result.directories {
|
||||
folder.execute(rusqlite::params![
|
||||
root_id,
|
||||
path.as_str(),
|
||||
validator.as_str()
|
||||
])?;
|
||||
}
|
||||
}
|
||||
|
||||
// Every statement below runs once per photograph listed -- 1,600 for one
|
||||
// folder whose sidecar changed, 24,000 for a first scan -- so each is
|
||||
// prepared once, and a folder's id is looked up once per folder rather
|
||||
// than once per photograph in it.
|
||||
let mut folder_of =
|
||||
tx.prepare_cached("SELECT id FROM folders WHERE root_id = ?1 AND path = ?2")?;
|
||||
let mut folder_ids: std::collections::HashMap<String, Option<i64>> =
|
||||
std::collections::HashMap::new();
|
||||
// TRACES: FR-PLAT-AND-2 | FR-CAT-9
|
||||
// The `availability` arm is what ends an offline library, and it does
|
||||
// it one photograph at a time. 3 is `Availability::Offline` and 0 is
|
||||
// `MetadataOnly`, the same code this statement inserts new rows with —
|
||||
// so a row that was marked offline when the root became unreachable is
|
||||
// returned to exactly the state a fresh scan would have given it, and
|
||||
// a row that was never marked is not touched at all.
|
||||
//
|
||||
// Conditional rather than a blanket reset for the same reason
|
||||
// `dr_catalog::walk` restores per file rather than per root: the only
|
||||
// thing that may clear "I could not reach this" is having reached it,
|
||||
// and this statement runs precisely once per file the scan listed.
|
||||
let mut upsert = tx.prepare_cached(
|
||||
"INSERT INTO images(root_id, folder_id, source_ref, format, file_size,
|
||||
availability, metadata_state, added_at)
|
||||
VALUES (?1, ?2, ?3, ?4, ?5, 0, 1, ?6)
|
||||
ON CONFLICT(root_id, source_ref) DO UPDATE SET
|
||||
file_size = excluded.file_size,
|
||||
folder_id = excluded.folder_id,
|
||||
availability = CASE WHEN images.availability = 3
|
||||
THEN 0 ELSE images.availability END",
|
||||
)?;
|
||||
let mut image_of =
|
||||
tx.prepare_cached("SELECT id FROM images WHERE root_id = ?1 AND source_ref = ?2")?;
|
||||
let mut remote = tx.prepare_cached(
|
||||
"INSERT INTO remote(image_id, file_id, etag, remote_path)
|
||||
VALUES (?1, ?2, ?3, ?4)
|
||||
ON CONFLICT(image_id) DO UPDATE SET
|
||||
etag = excluded.etag, remote_path = excluded.remote_path",
|
||||
)?;
|
||||
|
||||
for entry in &result.images {
|
||||
let folder_id: Option<i64> = entry.path.parent().and_then(|p| {
|
||||
tx.query_row(
|
||||
"SELECT id FROM folders WHERE root_id = ?1 AND path = ?2",
|
||||
rusqlite::params![root_id, p.as_str()],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.ok()
|
||||
});
|
||||
let folder_id: Option<i64> = match entry.path.parent() {
|
||||
None => None,
|
||||
Some(p) => match folder_ids.get(p.as_str()) {
|
||||
Some(id) => *id,
|
||||
None => {
|
||||
let id = folder_of
|
||||
.query_row(rusqlite::params![root_id, p.as_str()], |r| r.get(0))
|
||||
.ok();
|
||||
folder_ids.insert(p.as_str().to_string(), id);
|
||||
id
|
||||
}
|
||||
},
|
||||
};
|
||||
|
||||
// TRACES: FR-PLAT-AND-2 | FR-CAT-9
|
||||
// The `availability` arm is what ends an offline library, and it does
|
||||
// it one photograph at a time. 3 is `Availability::Offline` and 0 is
|
||||
// `MetadataOnly`, the same code this statement inserts new rows with —
|
||||
// so a row that was marked offline when the root became unreachable is
|
||||
// returned to exactly the state a fresh scan would have given it, and
|
||||
// a row that was never marked is not touched at all.
|
||||
//
|
||||
// Conditional rather than a blanket reset for the same reason
|
||||
// `dr_catalog::walk` restores per file rather than per root: the only
|
||||
// thing that may clear "I could not reach this" is having reached it,
|
||||
// and this statement runs precisely once per file the scan listed.
|
||||
tx.execute(
|
||||
"INSERT INTO images(root_id, folder_id, source_ref, format, file_size,
|
||||
availability, metadata_state, added_at)
|
||||
VALUES (?1, ?2, ?3, ?4, ?5, 0, 1, ?6)
|
||||
ON CONFLICT(root_id, source_ref) DO UPDATE SET
|
||||
file_size = excluded.file_size,
|
||||
folder_id = excluded.folder_id,
|
||||
availability = CASE WHEN images.availability = 3
|
||||
THEN 0 ELSE images.availability END",
|
||||
rusqlite::params![
|
||||
root_id,
|
||||
folder_id,
|
||||
entry.path.as_str(),
|
||||
entry
|
||||
.path
|
||||
.name()
|
||||
.rsplit_once('.')
|
||||
.map(|(_, e)| e.to_ascii_lowercase()),
|
||||
entry.size as i64,
|
||||
now_secs(),
|
||||
],
|
||||
)?;
|
||||
upsert.execute(rusqlite::params![
|
||||
root_id,
|
||||
folder_id,
|
||||
entry.path.as_str(),
|
||||
entry
|
||||
.path
|
||||
.name()
|
||||
.rsplit_once('.')
|
||||
.map(|(_, e)| e.to_ascii_lowercase()),
|
||||
entry.size as i64,
|
||||
now_secs(),
|
||||
])?;
|
||||
|
||||
let image_id: i64 = tx.query_row(
|
||||
"SELECT id FROM images WHERE root_id = ?1 AND source_ref = ?2",
|
||||
rusqlite::params![root_id, entry.path.as_str()],
|
||||
|r| r.get(0),
|
||||
)?;
|
||||
let image_id: i64 = image_of
|
||||
.query_row(rusqlite::params![root_id, entry.path.as_str()], |r| {
|
||||
r.get(0)
|
||||
})?;
|
||||
|
||||
// Remote identity, keyed on oc:fileid so a server-side move is a move
|
||||
// rather than a re-download (FR-NC-5).
|
||||
if let RemoteId::Stable(file_id) = entry.id {
|
||||
tx.execute(
|
||||
"INSERT INTO remote(image_id, file_id, etag, remote_path)
|
||||
VALUES (?1, ?2, ?3, ?4)
|
||||
ON CONFLICT(image_id) DO UPDATE SET
|
||||
etag = excluded.etag, remote_path = excluded.remote_path",
|
||||
rusqlite::params![
|
||||
image_id,
|
||||
file_id as i64,
|
||||
entry.validator.as_str(),
|
||||
entry.path.as_str()
|
||||
],
|
||||
)?;
|
||||
remote.execute(rusqlite::params![
|
||||
image_id,
|
||||
file_id as i64,
|
||||
entry.validator.as_str(),
|
||||
entry.path.as_str()
|
||||
])?;
|
||||
}
|
||||
|
||||
// Its thumbnail job, in the same transaction as the row it points at,
|
||||
// so a failure mid-insert cannot leave one pointing at a row that
|
||||
// never landed. It used to follow the commit, one autocommitting
|
||||
// statement per photograph, each finding its image again by path: a
|
||||
// relisted folder of 1,600 paid 1,600 commits for rows that are
|
||||
// almost all already queued.
|
||||
let _ = dr_catalog::jobs::enqueue(
|
||||
&tx,
|
||||
JobKind::Thumbnail,
|
||||
Some(image_id),
|
||||
Priority::Background,
|
||||
None,
|
||||
);
|
||||
}
|
||||
|
||||
drop((folder_of, upsert, image_of, remote));
|
||||
tx.commit()?;
|
||||
|
||||
// Thumbnail jobs after the commit, so a failure mid-insert does not leave
|
||||
// jobs pointing at rows that never landed.
|
||||
for entry in &result.images {
|
||||
if let Ok(image_id) = conn.query_row(
|
||||
"SELECT id FROM images WHERE root_id = ?1 AND source_ref = ?2",
|
||||
rusqlite::params![root_id, entry.path.as_str()],
|
||||
|r| r.get::<_, i64>(0),
|
||||
) {
|
||||
let _ = dr_catalog::jobs::enqueue(
|
||||
conn,
|
||||
JobKind::Thumbnail,
|
||||
Some(image_id),
|
||||
Priority::Background,
|
||||
None,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user