wip: ingest

This commit is contained in:
2026-08-22 14:12:40 +02:00
parent 98e0ad0537
commit 735683b849
10 changed files with 1267 additions and 11 deletions
+308
View File
@@ -0,0 +1,308 @@
//! TRACES: FR-CAT-11
//! Has this photograph been imported before?
//!
//! Two tiers, because neither alone is enough and they cost very different
//! amounts. The metadata tier — capture time, camera, size, the name the
//! camera gave it — is answerable from the catalog before a byte leaves the
//! card, which is what makes re-inserting an already-imported card cost a
//! metadata read per file rather than a full transfer. The content tier
//! catches what the first misses: the same frame arriving under a different
//! name, from a second card, or after somebody renamed it.
//!
//! # Why the filename is compared here rather than in SQL
//!
//! `images.source_ref` holds the whole opaque key — a relative path on Linux,
//! a document id on SAF — and the camera's filename is only its last
//! component. Matching that in SQL means `LIKE '%/IMG_0001.CR3'`, which cannot
//! use an index, scans the whole table, and is wrong on SAF where the
//! separator is not `/`. So the query narrows on the indexed columns and the
//! handful of rows that survive are compared in Rust, the same way the grid
//! already derives a display name.
//!
//! # Filename alone is never sufficient
//!
//! Camera filenames wrap at `IMG_9999` and start again, so a library of any
//! age holds several unrelated `IMG_0001.CR3`. That is why the cheap tier
//! carries capture time and camera as well, and why the expensive tier exists
//! at all.
use rusqlite::Connection;
use crate::CatalogError;
/// The last component of a stored source reference.
///
/// Splits on both separators for the same reason `Catalog::window` does: the
/// key's shape belongs to the storage that produced it, and a SAF document id
/// is delimited with `:`.
fn file_name(source_ref: &str) -> &str {
source_ref.rsplit(['/', ':']).next().unwrap_or(source_ref)
}
/// Whether the catalog already holds this photograph, on metadata alone.
///
/// `camera` is the joined make-and-model string the scan stores, not the raw
/// EXIF pair — the caller composes it the same way, or the comparison is
/// always false.
///
/// A `captured_at` of `None` makes this answer `false` rather than matching
/// every undated image in the library: without a capture time the key is
/// filename plus size, which two frames from the same body collide on
/// routinely. An undated file falls through to the content tier, which is
/// slower and right.
pub fn seen_by_metadata(
conn: &Connection,
captured_at: Option<i64>,
camera: Option<&str>,
size: u64,
original_name: &str,
) -> Result<bool, CatalogError> {
let Some(captured_at) = captured_at else {
return Ok(false);
};
// `images_captured` indexes the capture time, so this reads a few rows
// even in a library of fifty thousand: one instant to the second holds
// one frame, or a handful on a body shooting a burst.
let mut stmt = conn.prepare(
"SELECT source_ref FROM images
WHERE captured_at = ?1
AND (?2 IS NULL OR camera IS ?2)
AND (file_size IS NULL OR file_size = ?3)",
)?;
let mut rows = stmt.query(rusqlite::params![captured_at, camera, size as i64])?;
while let Some(row) = rows.next()? {
let source_ref: String = row.get(0)?;
if file_name(&source_ref).eq_ignore_ascii_case(original_name) {
return Ok(true);
}
}
Ok(false)
}
/// Whether these exact bytes are already in the library.
///
/// The tier that costs a read of the file. Cheap here — `images_hash` is a
/// partial index over the rows that have one — and expensive for the caller,
/// which had to hash something to ask.
pub fn seen_by_content(conn: &Connection, digest: &str) -> Result<bool, CatalogError> {
let n: i64 = conn.query_row(
"SELECT COUNT(*) FROM images WHERE content_hash = ?1",
[digest],
|r| r.get(0),
)?;
Ok(n > 0)
}
/// Record the digest of a file the import computed.
///
/// An import reads every byte anyway, so the hash is free at that moment and
/// costs a full read of an 80 MB file at any other. Storing it is what lets
/// the *next* import answer [`seen_by_content`] without reading anything.
///
/// Matched on `source_ref` within a root, which is how the scan that just
/// catalogued the imported file identifies it. Returns how many rows were
/// updated: zero means the scan has not reached the file yet, which is a
/// normal race and not an error.
pub fn set_content_hash(
conn: &Connection,
root_id: u64,
source_ref: &str,
digest: &str,
) -> Result<usize, CatalogError> {
Ok(conn.execute(
"UPDATE images SET content_hash = ?3
WHERE root_id = ?1 AND source_ref = ?2",
rusqlite::params![root_id as i64, source_ref, digest],
)?)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::Catalog;
/// A catalog holding one photograph, as a scan plus a metadata pass would
/// leave it.
fn with_one() -> Catalog {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, captured_at, camera, file_size,
content_hash, added_at)
VALUES (1, '2026/2026-08-22/IMG_0001.CR3', 1787407200, 'Canon EOS R5',
9, 'deadbeef', 0)",
[],
)
.unwrap();
cat
}
#[test]
fn re_inserting_the_same_card_is_recognised_before_a_transfer() {
let cat = with_one();
assert!(seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
9,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn a_different_frame_at_the_same_instant_is_not_a_duplicate() {
// Two bodies firing together, or a burst. The name separates them.
let cat = with_one();
assert!(!seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
9,
"IMG_0002.CR3"
)
.unwrap());
}
#[test]
fn the_same_name_from_a_different_camera_is_not_a_duplicate() {
// IMG_0001.CR3 exists on every card ever formatted.
let cat = with_one();
assert!(!seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("NIKON Z 9"),
9,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn the_same_name_at_a_different_time_is_not_a_duplicate() {
// The IMG_9999 wrap: the library holds an unrelated IMG_0001.CR3 from
// four years ago, and matching on name alone would refuse to import
// today's.
let cat = with_one();
assert!(!seen_by_metadata(
cat.connection(),
Some(1_600_000_000),
Some("Canon EOS R5"),
9,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn an_undated_file_falls_through_to_the_content_tier() {
// Not "matches everything undated" — that would silently refuse to
// import a whole card of scanned film.
let cat = with_one();
assert!(!seen_by_metadata(cat.connection(), None, None, 9, "IMG_0001.CR3").unwrap());
}
#[test]
fn a_file_that_grew_is_not_the_one_already_held() {
// A truncated earlier import, or a different rendition of the same
// frame. Same instant, same camera, same name, different bytes.
let cat = with_one();
assert!(!seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
1234,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn a_row_with_no_recorded_size_still_matches() {
// The scan stores a size, but a row merged from another device may
// not have one, and refusing to match it would re-import the library.
let cat = with_one();
cat.connection()
.execute("UPDATE images SET file_size = NULL", [])
.unwrap();
assert!(seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
9,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn the_same_frame_renamed_is_caught_by_its_bytes() {
let cat = with_one();
// The metadata tier misses it...
assert!(!seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
9,
"holiday-42.CR3"
)
.unwrap());
// ...and the content tier does not.
assert!(seen_by_content(cat.connection(), "deadbeef").unwrap());
assert!(!seen_by_content(cat.connection(), "cafe").unwrap());
}
#[test]
fn a_digest_recorded_now_answers_the_next_import() {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, added_at)
VALUES (1, '2026/2026-08-22/IMG_0001.CR3', 0)",
[],
)
.unwrap();
assert!(!seen_by_content(c, "abc123").unwrap());
let n = set_content_hash(c, 1, "2026/2026-08-22/IMG_0001.CR3", "abc123").unwrap();
assert_eq!(n, 1);
assert!(seen_by_content(c, "abc123").unwrap());
}
#[test]
fn recording_a_digest_before_the_scan_arrives_is_not_an_error() {
// The import writes the file and the scan catalogues it; between those
// two moments there is no row to update, and that is a race rather
// than a failure.
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
assert_eq!(
set_content_hash(c, 1, "not/scanned/yet.CR3", "abc").unwrap(),
0
);
}
#[test]
fn a_name_is_the_last_component_of_either_kind_of_key() {
assert_eq!(file_name("2026/2026-08-22/IMG_0001.CR3"), "IMG_0001.CR3");
// A SAF document id delimits with a colon.
assert_eq!(file_name("primary:DCIM/Camera/IMG_1.CR3"), "IMG_1.CR3");
assert_eq!(file_name("IMG_0001.CR3"), "IMG_0001.CR3");
}
}
+2
View File
@@ -33,6 +33,7 @@ use rusqlite::Connection;
pub mod cache;
pub mod collections;
pub mod dedup;
pub mod error;
pub mod jobs;
pub mod merge;
@@ -46,6 +47,7 @@ pub mod walk;
pub use cache::{Budget, Cache, DEFAULT_BUDGET_BYTES};
pub use collections::{Collection, CollectionKind, TreeRow};
pub use dedup::{seen_by_content, seen_by_metadata, set_content_hash};
pub use error::CatalogError;
pub use jobs::{Job, JobKind, Priority};
pub use merge::MergeReport;