Merge: recover a damaged catalog, and capture a crash locally

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-08-30 13:45:18 +02:00
co-authored by Claude Opus 5
15 changed files with 1954 additions and 14 deletions
+60 -2
View File
@@ -1,14 +1,37 @@
//! TRACES: NFR-ARCH-4 | NFR-R5
//! TRACES: NFR-ARCH-4 | NFR-R5 | NFR-R6
//! Catalog errors.
//!
//! Typed and attached to the affected subject rather than panicking — a
//! corrupt row or a failed job marks one image and lets the batch continue.
//!
//! # Why `From<rusqlite::Error>` is written by hand
//!
//! One class of SQLite failure is not about the statement that hit it: when
//! the file itself is damaged, *every* query fails, and which one the user
//! happened to trigger first says nothing. Before this, corruption reached the
//! interface as whatever `Sqlite(...)` the first failing query produced —
//! "database disk image is malformed" attached to a thumbnail refresh — and
//! there was nowhere to hang a recovery offer.
//!
//! So the conversion classifies rather than wraps: `SQLITE_CORRUPT` and
//! `SQLITE_NOTADB` become [`CatalogError::Corrupt`] wherever they arise, which
//! means a background job that trips over the damage reports the same thing
//! the startup check does (see [`crate::recovery`]).
/// Something went wrong talking to the catalog.
#[derive(Debug, thiserror::Error)]
pub enum CatalogError {
#[error("sqlite: {0}")]
Sqlite(#[from] rusqlite::Error),
Sqlite(#[source] rusqlite::Error),
/// The catalog file is damaged.
///
/// Its own variant because it is the one error with a *user-facing
/// remedy*: restore the NFR-R2 backup, or discard the index and rebuild it
/// from sources plus sidecars (NFR-R6, invariant §5.2.4). Every other
/// variant here is either a caller's mistake or a fact about one row.
#[error("the catalog file is damaged: {detail}")]
Corrupt { detail: String },
/// The catalog was written by a newer build.
///
@@ -74,3 +97,38 @@ pub enum CatalogError {
#[error("io: {0}")]
Io(String),
}
impl From<rusqlite::Error> for CatalogError {
fn from(e: rusqlite::Error) -> Self {
if is_corruption(&e) {
// `to_string` rather than keeping the error: the detail is going
// into a dialog and into a log line, and the recovery path has no
// use for the rusqlite type once it knows the file is damaged.
CatalogError::Corrupt {
detail: e.to_string(),
}
} else {
CatalogError::Sqlite(e)
}
}
}
/// Whether a SQLite failure means the *file* is damaged rather than the
/// statement wrong.
///
/// `SQLITE_NOTADB` is included because it is what a truncated or overwritten
/// catalog produces — SQLite cannot read the header, so it declines to call it
/// a database at all. To a user those are the same accident, and the same two
/// offers answer both.
///
/// Deliberately *not* included: `SQLITE_CANTOPEN` (a missing file, which
/// `Connection::open` fixes by creating one), `SQLITE_BUSY`, and
/// `SQLITE_IOERR` — a failing disk or a dropped network mount is a different
/// problem, and telling the user to rebuild their index would be a wrong
/// answer delivered confidently.
fn is_corruption(e: &rusqlite::Error) -> bool {
matches!(
e.sqlite_error_code(),
Some(rusqlite::ErrorCode::DatabaseCorrupt) | Some(rusqlite::ErrorCode::NotADatabase)
)
}
+37
View File
@@ -21,6 +21,7 @@
//! - [`runner`] — the thing that drains it, driven by whoever owns the thread
//! - [`trash`] — soft delete to a folder, then permanent delete
//! - [`merge`] / [`sync`] — cross-device merging of collections and keywords
//! - [`recovery`] — backups, and the two offers made when this file is damaged
//!
//! # The one thing everything is designed around
//!
@@ -47,6 +48,7 @@ pub mod keywords;
pub mod merge;
pub mod query;
pub mod rating;
pub mod recovery;
pub mod runner;
pub mod scan;
pub mod schema;
@@ -65,6 +67,7 @@ pub use keywords::{Coverage, Keyword, KeywordId, SelectionKeyword};
pub use merge::MergeReport;
pub use query::{Query, Sort};
pub use rating::{Judgement, MAX_RATING};
pub use recovery::Backup;
// Not `runner::Budget`: `cache::Budget` already owns that name here and
// means something else entirely (bytes on disk, not jobs in a slot).
// Callers spell the work budget `runner::Budget`, where it is unambiguous.
@@ -222,9 +225,23 @@ pub struct Catalog {
impl Catalog {
/// Open or create a catalog, migrating it forward if needed.
///
/// Does **not** verify the file — see [`Self::open_verified`], and
/// [`recovery`] for why the check is bound to startup rather than to every
/// open. Damage this trips over on the way past is still reported as
/// [`CatalogError::Corrupt`] rather than as a stray SQLite error.
pub fn open(path: &Path) -> Result<Self, CatalogError> {
let conn = Connection::open(path)?;
schema::configure(&conn)?;
// NFR-R2, and the reason it is *here*: a migration is the one routine
// operation that rewrites table structure, so it is the likeliest way
// this file becomes unreadable — and afterwards there is no
// pre-migration state left to copy. A failure to take the copy is
// logged rather than raised: a full disk must not be the thing that
// makes a library unopenable.
if let Err(e) = recovery::backup_before_migration(&conn, path) {
log::warn!("could not back up before migrating: {e}");
}
let from = schema::migrate(&conn)?;
// A migration adds a column; it cannot know what the value should be
// for rows that already existed. Backfilling on open is what stops
@@ -235,6 +252,26 @@ impl Catalog {
Ok(Catalog { conn })
}
/// TRACES: NFR-R6
/// Open a catalog, checking the file first.
///
/// What startup calls. On [`CatalogError::Corrupt`] the caller has a user
/// in front of it and must make the two offers [`recovery`] describes,
/// rather than reporting a SQLite message on a banner and carrying on into
/// a scan that would write into the damage.
///
/// Checked *before* opening rather than after, because opening runs
/// migrations: a damaged catalog that happens to have an intact header
/// would otherwise be migrated — rewriting structure on top of structure
/// that is already wrong — before anybody asked whether it was sound.
pub fn open_verified(path: &Path) -> Result<Self, CatalogError> {
// A catalog that is not there yet is not damaged; `open` creates it.
if path.is_file() {
recovery::check_file(path)?;
}
Self::open(path)
}
/// An in-memory catalog, for tests and for a throwaway import preview.
pub fn in_memory() -> Result<Self, CatalogError> {
let conn = Connection::open_in_memory()?;
+662
View File
@@ -0,0 +1,662 @@
//! TRACES: NFR-R2 | NFR-R6
//! What to do once the index is already damaged.
//!
//! # Why this can be a small module
//!
//! Because of a property the rest of the catalog was built to keep: the
//! catalog is an *index*, not a source of truth (ARCH §6.12, invariant
//! §5.2.4). Sidecars beside the images hold the authoritative ratings,
//! keywords and edit graphs, for every catalogued image and whether or not a
//! remote account exists (FR-CAT-8). So the worst outcome available here is a
//! rescan — expensive, but not a loss.
//!
//! That is the second offer. The first is cheaper and loses nothing at all: a
//! backup, restored.
//!
//! # The one thing a rebuild does not recover
//!
//! **Collections.** A manual collection is a set of images the user assembled
//! by hand and nothing in the filesystem records it (`docs/catalog.md` §8.1) —
//! which is the whole reason the catalog file itself syncs. So the two offers
//! are not interchangeable, and the interface must not present them as if they
//! were: a restore keeps the user's collections, a rebuild does not.
//!
//! # When the check runs, and when it does not
//!
//! [`integrity_check`] reads every page of the database. That is affordable
//! once, at startup, where a failure has a user in front of it who can answer
//! a question — and it is *not* affordable on every [`Catalog::open`], which
//! this application does per background task, dozens of times a session. So
//! the check is bound to [`Catalog::open_verified`] rather than to `open`,
//! and the cheap half of the story — classifying `SQLITE_CORRUPT` and
//! `SQLITE_NOTADB` as [`CatalogError::Corrupt`] — happens for free on every
//! query through [`crate::error`]'s conversion. A background job that trips
//! over the damage first therefore reports the same thing the startup check
//! would have.
//!
//! [`Catalog::open`]: crate::Catalog::open
//! [`Catalog::open_verified`]: crate::Catalog::open_verified
use std::path::{Path, PathBuf};
use std::time::{SystemTime, UNIX_EPOCH};
use rusqlite::Connection;
use crate::error::CatalogError;
use crate::schema;
/// Directory backups live in, relative to the catalog file.
///
/// Beside the catalog rather than in the cache directory, and that is the
/// point of the choice: this is the copy the user falls back on, and a cache
/// is a place the operating system is entitled to empty without asking
/// (see `library::data_root` for the same reasoning about sidecars).
const BACKUP_DIR: &str = "backups";
/// How many backups are kept.
///
/// Small on purpose. A backup is a full copy of a catalog that is tens of
/// megabytes at 50k images, and the value of the third-oldest one is close to
/// zero: corruption is noticed at the next launch, not months later. What the
/// depth buys is protection against backing *up* the damage — if a corrupt
/// catalog is copied before anyone notices, the generation behind it is still
/// clean.
pub const KEEP_BACKUPS: usize = 3;
/// Suffix given to a catalog that has been set aside as damaged.
///
/// Kept rather than deleted. It costs disk this application would rather not
/// spend, and it is still the right call: `.sqlite` files have been recovered
/// by hand before, the user has not consented to a deletion, and NFR-R4's
/// instinct — never destroy what the user did not ask you to destroy — does
/// not stop applying at the catalog's edge.
const DAMAGED_SUFFIX: &str = "damaged";
/// One kept backup.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Backup {
pub path: PathBuf,
/// UTC seconds at which it was taken, read from the filename rather than
/// from the filesystem: a copy, a restore or a sync can rewrite an mtime,
/// and then the newest backup is not the one that looks newest.
pub taken_at: i64,
pub bytes: u64,
}
/// Where backups for `catalog` are kept.
pub fn backup_dir(catalog: &Path) -> PathBuf {
catalog
.parent()
.unwrap_or_else(|| Path::new("."))
.join(BACKUP_DIR)
}
/// Check the database this connection is attached to.
///
/// `quick_check` rather than `integrity_check`: the difference is that
/// `quick_check` skips verifying that every index agrees with its table, which
/// is the expensive half and the half this application least needs — every
/// index here is derivable, and `REINDEX` fixes one without anybody being
/// asked a question. What is left still reads every page, and catches the
/// damage that matters: torn b-trees, a bad freelist, a truncated file.
///
/// Returns [`CatalogError::Corrupt`] carrying what SQLite said, so the message
/// the user sees is the diagnosis rather than a paraphrase of it.
pub fn integrity_check(conn: &Connection) -> Result<(), CatalogError> {
// The argument caps how many problems are reported. One is enough: the
// answer is the same whether the file has one damaged page or nine
// hundred, and an unbounded check on a badly damaged file can run for a
// very long time producing a list nobody will read.
let mut stmt = conn.prepare("PRAGMA quick_check(1)")?;
let rows: Vec<String> = stmt
.query_map([], |r| r.get(0))?
.collect::<Result<Vec<_>, _>>()?;
// A healthy database answers with the single row "ok".
if rows.len() == 1 && rows[0] == "ok" {
return Ok(());
}
Err(CatalogError::Corrupt {
detail: rows.join("; "),
})
}
/// Check a catalog file that is not currently open.
///
/// Used before a restore: a backup is only worth swapping in if it is sound,
/// and swapping in a second damaged file — leaving the user with no catalog
/// and no offer left — is the failure this exists to prevent.
pub fn check_file(path: &Path) -> Result<(), CatalogError> {
if !path.is_file() {
return Err(CatalogError::Io(format!("{} is missing", path.display())));
}
// Read-write rather than read-only, which reads oddly for a check. A
// backup carries the WAL journal mode in its header because it was copied
// page-for-page from a WAL database, and SQLite cannot open one read-only
// without a shared-memory file it is then not allowed to create. Nothing
// here writes; the connection is opened, read, and dropped.
let conn = Connection::open(path)?;
integrity_check(&conn)
}
/// Take a backup of the open catalog.
///
/// Returns the file written. Older generations beyond [`KEEP_BACKUPS`] are
/// pruned, newest kept.
///
/// Goes through `crate::sync::copy_to` — SQLite's own backup API after a
/// TRUNCATE checkpoint — rather than copying the file. A WAL database is not
/// one file, and `fs::copy` of the main file alone would silently back up a
/// state that is older than the catalog and possibly torn, which is the one
/// failure mode a backup cannot afford.
pub fn backup(conn: &Connection, catalog: &Path) -> Result<PathBuf, CatalogError> {
let dir = backup_dir(catalog);
std::fs::create_dir_all(&dir)
.map_err(|e| CatalogError::Io(format!("creating {}: {e}", dir.display())))?;
let dest = dir.join(format!("catalog-{}.sqlite", now()));
// A second backup within the same second would otherwise land on the first
// one's name. Rare, and only reachable from tests and a retry, but the
// result would be a half-overwritten backup rather than two.
if dest.exists() {
std::fs::remove_file(&dest)
.map_err(|e| CatalogError::Io(format!("replacing {}: {e}", dest.display())))?;
}
// Dropped immediately: the copy is complete when `copy_to` returns, and
// holding the connection open would leave a `-wal` beside a file whose
// whole purpose is to be a single self-contained artefact.
drop(crate::sync::copy_to(conn, &dest)?);
prune(catalog);
Ok(dest)
}
/// Back up before a migration, if there is anything to back up.
///
/// Called from [`Catalog::open`](crate::Catalog::open) between `configure` and
/// `migrate`. NFR-R2 asks for this and the reasoning is narrower than "backups
/// are prudent": a migration is the one routine operation that rewrites table
/// structure, so it is the likeliest way a catalog becomes unreadable, and it
/// is the one moment where the pre-change state is still on disk to be copied.
/// Afterwards there is nothing left to take a copy *of*.
///
/// A no-op in the two cases where it would cost without buying anything: a
/// catalog already at [`schema::SCHEMA_VERSION`], and a brand-new file at
/// version 0 with no tables in it yet.
pub fn backup_before_migration(conn: &Connection, catalog: &Path) -> Result<(), CatalogError> {
let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
if from == 0 || from >= schema::SCHEMA_VERSION {
return Ok(());
}
let path = backup(conn, catalog)?;
log::info!(
"backed up catalog at v{from} to {} before migrating to v{}",
path.display(),
schema::SCHEMA_VERSION
);
Ok(())
}
/// The backups available for `catalog`, newest first.
///
/// Never fails: an unreadable or absent backup directory means there are no
/// backups, which is a fact about the offer to make rather than an error to
/// report on top of the corruption the user is already looking at.
pub fn backups(catalog: &Path) -> Vec<Backup> {
let dir = backup_dir(catalog);
let Ok(entries) = std::fs::read_dir(&dir) else {
return Vec::new();
};
let mut out: Vec<Backup> = entries
.flatten()
.filter_map(|e| {
let path = e.path();
let taken_at = timestamp_of(&path)?;
let bytes = e.metadata().ok()?.len();
Some(Backup {
path,
taken_at,
bytes,
})
})
.collect();
out.sort_by_key(|b| std::cmp::Reverse(b.taken_at));
out
}
/// Put a backup back in place of the damaged catalog.
///
/// **Every connection to `catalog` must be closed first.** This replaces the
/// file underneath anything still holding it open, which on a live connection
/// is how a *second* corrupt catalog gets made.
///
/// The order is deliberate:
///
/// 1. The backup is checked. A restore that installs a second damaged file
/// leaves the user with nothing to try next.
/// 2. The damaged catalog is renamed aside, and its `-wal` and `-shm` are
/// **deleted**. This is the step that is easy to leave out and fatal to
/// leave out: a journal belonging to the old file, sitting beside the new
/// one under the same name, is replayed into it on the next open. That is
/// not a restore, it is a fresh corruption with the evidence gone.
/// 3. The backup is *copied* into place, not moved, so a failure here can be
/// retried against the same backup.
pub fn restore(catalog: &Path, backup: &Path) -> Result<(), CatalogError> {
check_file(backup)?;
set_aside(catalog)?;
std::fs::copy(backup, catalog).map_err(|e| {
CatalogError::Io(format!(
"restoring {} from {}: {e}",
catalog.display(),
backup.display()
))
})?;
log::info!(
"restored {} from backup {}",
catalog.display(),
backup.display()
);
Ok(())
}
/// Move a damaged catalog out of the way so the next open builds a fresh one.
///
/// This is the rebuild path (NFR-R6's second offer) and also the first half of
/// a [`restore`]. Nothing else is needed to rebuild: the next
/// [`Catalog::open`](crate::Catalog::open) creates an empty catalog at the
/// current schema, and the ordinary scan repopulates it from sources and
/// sidecars — which is precisely invariant §5.2.4 being spent rather than
/// merely asserted.
///
/// Returns where the damaged file was put, or `None` if there was no catalog
/// to move — a caller may be recovering from a file SQLite could not open
/// because it was never created.
pub fn set_aside(catalog: &Path) -> Result<Option<PathBuf>, CatalogError> {
let moved = if catalog.exists() {
let dest = with_suffix(catalog, DAMAGED_SUFFIX);
// An earlier damaged copy is replaced rather than accumulating: two of
// these is two full-size catalogs on the user's disk, and the older
// one has already been superseded by a recovery the user completed.
let _ = std::fs::remove_file(&dest);
// The rename first, so that a failure here leaves the journals with
// the file they belong to rather than orphaned beside a catalog that
// is still in use.
std::fs::rename(catalog, &dest).map_err(|e| {
CatalogError::Io(format!(
"setting aside {} as {}: {e}",
catalog.display(),
dest.display()
))
})?;
log::warn!(
"catalog {} was damaged; kept as {}",
catalog.display(),
dest.display()
);
Some(dest)
} else {
None
};
// Then the journals, whether or not there was a catalog to move: a `-wal`
// orphaned beside a missing database is replayed into whatever takes that
// name next, which would not be a restore but a fresh corruption with the
// evidence gone.
for sidecar in journals(catalog) {
if let Err(e) = std::fs::remove_file(&sidecar) {
if e.kind() != std::io::ErrorKind::NotFound {
return Err(CatalogError::Io(format!(
"removing stale journal {}: {e}",
sidecar.display()
)));
}
}
}
Ok(moved)
}
/// Delete backups beyond [`KEEP_BACKUPS`].
///
/// Best-effort and silent about individual failures: failing to delete an old
/// backup is not a reason to fail the new one, which is already written.
fn prune(catalog: &Path) {
for old in backups(catalog).into_iter().skip(KEEP_BACKUPS) {
if let Err(e) = std::fs::remove_file(&old.path) {
log::warn!("could not prune backup {}: {e}", old.path.display());
}
}
}
/// The WAL and shared-memory files SQLite keeps beside a database.
fn journals(catalog: &Path) -> [PathBuf; 2] {
[with_suffix(catalog, "wal"), with_suffix(catalog, "shm")]
}
/// `catalog.sqlite` plus `-suffix`, the way SQLite names its own sidecars.
///
/// Appended to the whole filename rather than replacing the extension, so
/// `catalog.sqlite-wal` is what SQLite would look for and `catalog.sqlite-
/// damaged` sorts next to the catalog it came from.
fn with_suffix(catalog: &Path, suffix: &str) -> PathBuf {
let mut s = catalog.as_os_str().to_os_string();
s.push("-");
s.push(suffix);
PathBuf::from(s)
}
/// Read the timestamp out of a backup's filename, or `None` if this is not one.
///
/// Doubles as the filter that keeps [`backups`] from offering the user
/// something that is not a catalog — a stray file in the directory, or a `-wal`
/// left by a crash mid-backup.
fn timestamp_of(path: &Path) -> Option<i64> {
let name = path.file_name()?.to_str()?;
name.strip_prefix("catalog-")?
.strip_suffix(".sqlite")?
.parse()
.ok()
}
/// Seconds since the epoch, or 0 if the clock is before it.
fn now() -> i64 {
SystemTime::now()
.duration_since(UNIX_EPOCH)
.map(|d| d.as_secs() as i64)
.unwrap_or(0)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::Catalog;
use std::io::{Seek, SeekFrom, Write};
/// A scratch directory that cleans up with the test.
fn tempdir(tag: &str) -> PathBuf {
let base = std::env::temp_dir().join(format!(
"dr-recovery-{tag}-{}-{:?}",
std::process::id(),
std::thread::current().id()
));
let _ = std::fs::remove_dir_all(&base);
std::fs::create_dir_all(&base).unwrap();
base
}
/// A catalog on disk with enough rows to span several pages, closed.
///
/// Closed matters: WAL means the rows are in `catalog.sqlite-wal` until
/// something checkpoints, and a test that corrupted the main file while
/// the data was still in the journal would be corrupting empty space.
fn fixture(path: &Path, images: i64) {
let cat = Catalog::open(path).unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
VALUES ('u1', 'Iceland', 0, 0, 1, 1)",
[],
)
.unwrap();
for i in 1..=images {
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (?1, 1, ?2, 0)",
rusqlite::params![i, format!("DCIM/IMG_{i:05}.CR3")],
)
.unwrap();
}
crate::sync::checkpoint(c).unwrap();
drop(cat);
}
/// Scribble over everything past the first two pages.
///
/// Past them rather than over them so that page 1 — the header and the
/// schema — survives: this produces a file SQLite is willing to open and
/// then finds damaged, which is the case `quick_check` exists for. Wiping
/// the header instead produces `SQLITE_NOTADB` at the first pragma, which
/// is a different branch and has its own test.
fn corrupt(path: &Path) {
let mut f = std::fs::OpenOptions::new().write(true).open(path).unwrap();
let len = f.metadata().unwrap().len();
assert!(
len > 8192,
"fixture is only {len} bytes; corrupting past page 2 would be a no-op"
);
let junk = vec![0x5a_u8; (len - 8192) as usize];
f.seek(SeekFrom::Start(8192)).unwrap();
f.write_all(&junk).unwrap();
f.sync_all().unwrap();
}
#[test]
fn a_healthy_catalog_passes() {
let cat = Catalog::in_memory().unwrap();
integrity_check(cat.connection()).unwrap();
}
#[test]
fn a_corrupt_catalog_is_reported_as_corrupt_not_as_sqlite() {
// The whole point of the variant: this used to arrive as whatever
// rusqlite error the first failing query produced, with nowhere to
// hang a recovery offer.
let dir = tempdir("detect");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
corrupt(&path);
assert!(matches!(
Catalog::open_verified(&path),
Err(CatalogError::Corrupt { .. })
));
}
#[test]
fn a_file_that_is_not_a_database_is_also_corrupt() {
// A truncated or overwritten catalog never reaches `quick_check`: the
// first pragma fails with SQLITE_NOTADB. Same accident to the user,
// same two offers, so it must classify the same way.
let dir = tempdir("notadb");
let path = dir.join("catalog.sqlite");
std::fs::write(&path, b"this is not a catalog, it is a text file\n").unwrap();
assert!(matches!(
Catalog::open(&path),
Err(CatalogError::Corrupt { .. })
));
}
#[test]
fn restoring_a_backup_recovers_the_collections_a_rebuild_would_lose() {
// The first NFR-R6 branch, asserted on the thing that distinguishes it
// from the second: a collection exists nowhere but the catalog, so it
// is the evidence that the *contents* came back and not merely a
// readable file (docs/catalog.md §8.1).
let dir = tempdir("restore");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
{
let cat = Catalog::open(&path).unwrap();
backup(cat.connection(), &path).unwrap();
}
corrupt(&path);
assert!(matches!(
Catalog::open_verified(&path),
Err(CatalogError::Corrupt { .. })
));
let newest = backups(&path).into_iter().next().expect("a backup exists");
restore(&path, &newest.path).unwrap();
let cat = Catalog::open_verified(&path).unwrap();
let name: String = cat
.connection()
.query_row("SELECT name FROM collections", [], |r| r.get(0))
.unwrap();
assert_eq!(name, "Iceland");
let images: i64 = cat
.connection()
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(images, 500);
}
#[test]
fn a_damaged_backup_is_refused_rather_than_installed() {
let dir = tempdir("badbackup");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
{
let cat = Catalog::open(&path).unwrap();
backup(cat.connection(), &path).unwrap();
}
let newest = backups(&path).into_iter().next().unwrap();
corrupt(&newest.path);
corrupt(&path);
assert!(matches!(
restore(&path, &newest.path),
Err(CatalogError::Corrupt { .. })
));
// And the damaged catalog is still where it was, so the second offer
// is still available.
assert!(path.exists());
}
#[test]
fn setting_aside_leaves_a_fresh_catalog_to_rebuild_into() {
// The second NFR-R6 branch. What makes it a rebuild rather than a data
// loss is invariant §5.2.4, which lives outside this crate — what is
// testable here is that the damaged file is out of the way, kept, and
// that the next open succeeds on an empty catalog at the current
// schema, which is what a scan then fills.
let dir = tempdir("rebuild");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
corrupt(&path);
let kept = set_aside(&path).unwrap().expect("the catalog was there");
assert!(kept.exists(), "the damaged catalog was deleted, not kept");
assert!(!path.exists());
let cat = Catalog::open_verified(&path).unwrap();
let images: i64 = cat
.connection()
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(images, 0);
let v: i64 = cat
.connection()
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, schema::SCHEMA_VERSION);
}
#[test]
fn a_stale_journal_does_not_follow_the_catalog_into_recovery() {
// The step that is easy to omit: a `-wal` belonging to the damaged
// file is replayed into whatever takes its name next.
let dir = tempdir("journal");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
std::fs::write(with_suffix(&path, "wal"), b"stale").unwrap();
set_aside(&path).unwrap();
assert!(!with_suffix(&path, "wal").exists());
}
#[test]
fn a_migration_is_backed_up_before_it_runs() {
// NFR-R2's second clause, against a real v1 catalog rather than a
// faked version number: the point is not that *a* file appears but
// that it holds the state from before the migration, which is the only
// state that is any use if the migration is what breaks it.
let dir = tempdir("premigrate");
let path = dir.join("catalog.sqlite");
{
let c = Connection::open(&path).unwrap();
schema::configure(&c).unwrap();
// `v1_for_attached` names the schema it targets, and "main" is a
// schema like any other — so this is the real v1, without needing
// `V1` itself to become visible outside its module.
c.execute_batch(&schema::v1_for_attached("main")).unwrap();
c.pragma_update(None, "user_version", 1).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
crate::sync::checkpoint(&c).unwrap();
}
assert!(backups(&path).is_empty());
Catalog::open(&path).unwrap();
let taken = backups(&path);
assert_eq!(taken.len(), 1, "no backup was taken before the migration");
check_file(&taken[0].path).unwrap();
let kept = Connection::open(&taken[0].path).unwrap();
let v: i64 = kept
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, 1, "the backup was taken after the migration, not before");
}
#[test]
fn opening_an_up_to_date_catalog_takes_no_backup() {
// Or every background task that opens the catalog would copy it.
let dir = tempdir("nobackup");
let path = dir.join("catalog.sqlite");
fixture(&path, 10);
Catalog::open(&path).unwrap();
assert!(backups(&path).is_empty());
}
#[test]
fn only_the_newest_generations_are_kept() {
let dir = tempdir("prune");
let path = dir.join("catalog.sqlite");
fixture(&path, 10);
let cat = Catalog::open(&path).unwrap();
// Written by hand rather than by calling `backup` in a loop: the
// filename carries whole seconds, so real calls would collide.
std::fs::create_dir_all(backup_dir(&path)).unwrap();
for t in 1..=KEEP_BACKUPS as i64 + 2 {
drop(
crate::sync::copy_to(
cat.connection(),
&backup_dir(&path).join(format!("catalog-{t}.sqlite")),
)
.unwrap(),
);
}
prune(&path);
let kept = backups(&path);
assert_eq!(kept.len(), KEEP_BACKUPS);
// Newest first, and the newest is the highest timestamp.
assert_eq!(kept[0].taken_at, KEEP_BACKUPS as i64 + 2);
}
#[test]
fn a_stray_file_in_the_backup_directory_is_not_offered_as_one() {
let dir = tempdir("stray");
let path = dir.join("catalog.sqlite");
fixture(&path, 10);
std::fs::create_dir_all(backup_dir(&path)).unwrap();
std::fs::write(backup_dir(&path).join("notes.txt"), b"hello").unwrap();
std::fs::write(backup_dir(&path).join("catalog-7.sqlite-wal"), b"x").unwrap();
assert!(backups(&path).is_empty());
}
}
+16 -2
View File
@@ -52,6 +52,21 @@ pub fn checkpoint(conn: &Connection) -> Result<(), CatalogError> {
/// coherent even with writers active. Callers should still prefer a quiet
/// moment — this competes with background jobs for the write lock.
pub fn snapshot_for_upload(conn: &Connection, dest: &Path) -> Result<(), CatalogError> {
let out = copy_to(conn, dest)?;
strip_face_crops(&out)?;
Ok(())
}
/// Checkpoint, then copy the whole database to `dest`, and hand back the
/// connection to the copy.
///
/// Split out from [`snapshot_for_upload`] because [`crate::recovery`] wants
/// exactly this and none of what follows it there: an NFR-R2 backup is the
/// file the user may have to *live on*, so it keeps the face crops that an
/// upload strips. Sharing the copy rather than reimplementing it is what keeps
/// the WAL discipline in one place — a backup taken with `fs::copy` would be
/// the torn snapshot this module's header exists to warn about.
pub(crate) fn copy_to(conn: &Connection, dest: &Path) -> Result<Connection, CatalogError> {
checkpoint(conn)?;
let mut out = Connection::open(dest)?;
@@ -63,8 +78,7 @@ pub fn snapshot_for_upload(conn: &Connection, dest: &Path) -> Result<(), Catalog
backup.run_to_completion(i32::MAX, std::time::Duration::ZERO, None)?;
drop(backup);
strip_face_crops(&out)?;
Ok(())
Ok(out)
}
/// Drop the stored face crops from a snapshot before it is uploaded.