|
|
|
@@ -0,0 +1,662 @@
|
|
|
|
|
//! TRACES: NFR-R2 | NFR-R6
|
|
|
|
|
//! What to do once the index is already damaged.
|
|
|
|
|
//!
|
|
|
|
|
//! # Why this can be a small module
|
|
|
|
|
//!
|
|
|
|
|
//! Because of a property the rest of the catalog was built to keep: the
|
|
|
|
|
//! catalog is an *index*, not a source of truth (ARCH §6.12, invariant
|
|
|
|
|
//! §5.2.4). Sidecars beside the images hold the authoritative ratings,
|
|
|
|
|
//! keywords and edit graphs, for every catalogued image and whether or not a
|
|
|
|
|
//! remote account exists (FR-CAT-8). So the worst outcome available here is a
|
|
|
|
|
//! rescan — expensive, but not a loss.
|
|
|
|
|
//!
|
|
|
|
|
//! That is the second offer. The first is cheaper and loses nothing at all: a
|
|
|
|
|
//! backup, restored.
|
|
|
|
|
//!
|
|
|
|
|
//! # The one thing a rebuild does not recover
|
|
|
|
|
//!
|
|
|
|
|
//! **Collections.** A manual collection is a set of images the user assembled
|
|
|
|
|
//! by hand and nothing in the filesystem records it (`docs/catalog.md` §8.1) —
|
|
|
|
|
//! which is the whole reason the catalog file itself syncs. So the two offers
|
|
|
|
|
//! are not interchangeable, and the interface must not present them as if they
|
|
|
|
|
//! were: a restore keeps the user's collections, a rebuild does not.
|
|
|
|
|
//!
|
|
|
|
|
//! # When the check runs, and when it does not
|
|
|
|
|
//!
|
|
|
|
|
//! [`integrity_check`] reads every page of the database. That is affordable
|
|
|
|
|
//! once, at startup, where a failure has a user in front of it who can answer
|
|
|
|
|
//! a question — and it is *not* affordable on every [`Catalog::open`], which
|
|
|
|
|
//! this application does per background task, dozens of times a session. So
|
|
|
|
|
//! the check is bound to [`Catalog::open_verified`] rather than to `open`,
|
|
|
|
|
//! and the cheap half of the story — classifying `SQLITE_CORRUPT` and
|
|
|
|
|
//! `SQLITE_NOTADB` as [`CatalogError::Corrupt`] — happens for free on every
|
|
|
|
|
//! query through [`crate::error`]'s conversion. A background job that trips
|
|
|
|
|
//! over the damage first therefore reports the same thing the startup check
|
|
|
|
|
//! would have.
|
|
|
|
|
//!
|
|
|
|
|
//! [`Catalog::open`]: crate::Catalog::open
|
|
|
|
|
//! [`Catalog::open_verified`]: crate::Catalog::open_verified
|
|
|
|
|
|
|
|
|
|
use std::path::{Path, PathBuf};
|
|
|
|
|
use std::time::{SystemTime, UNIX_EPOCH};
|
|
|
|
|
|
|
|
|
|
use rusqlite::Connection;
|
|
|
|
|
|
|
|
|
|
use crate::error::CatalogError;
|
|
|
|
|
use crate::schema;
|
|
|
|
|
|
|
|
|
|
/// Directory backups live in, relative to the catalog file.
|
|
|
|
|
///
|
|
|
|
|
/// Beside the catalog rather than in the cache directory, and that is the
|
|
|
|
|
/// point of the choice: this is the copy the user falls back on, and a cache
|
|
|
|
|
/// is a place the operating system is entitled to empty without asking
|
|
|
|
|
/// (see `library::data_root` for the same reasoning about sidecars).
|
|
|
|
|
const BACKUP_DIR: &str = "backups";
|
|
|
|
|
|
|
|
|
|
/// How many backups are kept.
|
|
|
|
|
///
|
|
|
|
|
/// Small on purpose. A backup is a full copy of a catalog that is tens of
|
|
|
|
|
/// megabytes at 50k images, and the value of the third-oldest one is close to
|
|
|
|
|
/// zero: corruption is noticed at the next launch, not months later. What the
|
|
|
|
|
/// depth buys is protection against backing *up* the damage — if a corrupt
|
|
|
|
|
/// catalog is copied before anyone notices, the generation behind it is still
|
|
|
|
|
/// clean.
|
|
|
|
|
pub const KEEP_BACKUPS: usize = 3;
|
|
|
|
|
|
|
|
|
|
/// Suffix given to a catalog that has been set aside as damaged.
|
|
|
|
|
///
|
|
|
|
|
/// Kept rather than deleted. It costs disk this application would rather not
|
|
|
|
|
/// spend, and it is still the right call: `.sqlite` files have been recovered
|
|
|
|
|
/// by hand before, the user has not consented to a deletion, and NFR-R4's
|
|
|
|
|
/// instinct — never destroy what the user did not ask you to destroy — does
|
|
|
|
|
/// not stop applying at the catalog's edge.
|
|
|
|
|
const DAMAGED_SUFFIX: &str = "damaged";
|
|
|
|
|
|
|
|
|
|
/// One kept backup.
|
|
|
|
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
|
|
|
|
pub struct Backup {
|
|
|
|
|
pub path: PathBuf,
|
|
|
|
|
/// UTC seconds at which it was taken, read from the filename rather than
|
|
|
|
|
/// from the filesystem: a copy, a restore or a sync can rewrite an mtime,
|
|
|
|
|
/// and then the newest backup is not the one that looks newest.
|
|
|
|
|
pub taken_at: i64,
|
|
|
|
|
pub bytes: u64,
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Where backups for `catalog` are kept.
|
|
|
|
|
pub fn backup_dir(catalog: &Path) -> PathBuf {
|
|
|
|
|
catalog
|
|
|
|
|
.parent()
|
|
|
|
|
.unwrap_or_else(|| Path::new("."))
|
|
|
|
|
.join(BACKUP_DIR)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Check the database this connection is attached to.
|
|
|
|
|
///
|
|
|
|
|
/// `quick_check` rather than `integrity_check`: the difference is that
|
|
|
|
|
/// `quick_check` skips verifying that every index agrees with its table, which
|
|
|
|
|
/// is the expensive half and the half this application least needs — every
|
|
|
|
|
/// index here is derivable, and `REINDEX` fixes one without anybody being
|
|
|
|
|
/// asked a question. What is left still reads every page, and catches the
|
|
|
|
|
/// damage that matters: torn b-trees, a bad freelist, a truncated file.
|
|
|
|
|
///
|
|
|
|
|
/// Returns [`CatalogError::Corrupt`] carrying what SQLite said, so the message
|
|
|
|
|
/// the user sees is the diagnosis rather than a paraphrase of it.
|
|
|
|
|
pub fn integrity_check(conn: &Connection) -> Result<(), CatalogError> {
|
|
|
|
|
// The argument caps how many problems are reported. One is enough: the
|
|
|
|
|
// answer is the same whether the file has one damaged page or nine
|
|
|
|
|
// hundred, and an unbounded check on a badly damaged file can run for a
|
|
|
|
|
// very long time producing a list nobody will read.
|
|
|
|
|
let mut stmt = conn.prepare("PRAGMA quick_check(1)")?;
|
|
|
|
|
let rows: Vec<String> = stmt
|
|
|
|
|
.query_map([], |r| r.get(0))?
|
|
|
|
|
.collect::<Result<Vec<_>, _>>()?;
|
|
|
|
|
|
|
|
|
|
// A healthy database answers with the single row "ok".
|
|
|
|
|
if rows.len() == 1 && rows[0] == "ok" {
|
|
|
|
|
return Ok(());
|
|
|
|
|
}
|
|
|
|
|
Err(CatalogError::Corrupt {
|
|
|
|
|
detail: rows.join("; "),
|
|
|
|
|
})
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Check a catalog file that is not currently open.
|
|
|
|
|
///
|
|
|
|
|
/// Used before a restore: a backup is only worth swapping in if it is sound,
|
|
|
|
|
/// and swapping in a second damaged file — leaving the user with no catalog
|
|
|
|
|
/// and no offer left — is the failure this exists to prevent.
|
|
|
|
|
pub fn check_file(path: &Path) -> Result<(), CatalogError> {
|
|
|
|
|
if !path.is_file() {
|
|
|
|
|
return Err(CatalogError::Io(format!("{} is missing", path.display())));
|
|
|
|
|
}
|
|
|
|
|
// Read-write rather than read-only, which reads oddly for a check. A
|
|
|
|
|
// backup carries the WAL journal mode in its header because it was copied
|
|
|
|
|
// page-for-page from a WAL database, and SQLite cannot open one read-only
|
|
|
|
|
// without a shared-memory file it is then not allowed to create. Nothing
|
|
|
|
|
// here writes; the connection is opened, read, and dropped.
|
|
|
|
|
let conn = Connection::open(path)?;
|
|
|
|
|
integrity_check(&conn)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Take a backup of the open catalog.
|
|
|
|
|
///
|
|
|
|
|
/// Returns the file written. Older generations beyond [`KEEP_BACKUPS`] are
|
|
|
|
|
/// pruned, newest kept.
|
|
|
|
|
///
|
|
|
|
|
/// Goes through `crate::sync::copy_to` — SQLite's own backup API after a
|
|
|
|
|
/// TRUNCATE checkpoint — rather than copying the file. A WAL database is not
|
|
|
|
|
/// one file, and `fs::copy` of the main file alone would silently back up a
|
|
|
|
|
/// state that is older than the catalog and possibly torn, which is the one
|
|
|
|
|
/// failure mode a backup cannot afford.
|
|
|
|
|
pub fn backup(conn: &Connection, catalog: &Path) -> Result<PathBuf, CatalogError> {
|
|
|
|
|
let dir = backup_dir(catalog);
|
|
|
|
|
std::fs::create_dir_all(&dir)
|
|
|
|
|
.map_err(|e| CatalogError::Io(format!("creating {}: {e}", dir.display())))?;
|
|
|
|
|
|
|
|
|
|
let dest = dir.join(format!("catalog-{}.sqlite", now()));
|
|
|
|
|
// A second backup within the same second would otherwise land on the first
|
|
|
|
|
// one's name. Rare, and only reachable from tests and a retry, but the
|
|
|
|
|
// result would be a half-overwritten backup rather than two.
|
|
|
|
|
if dest.exists() {
|
|
|
|
|
std::fs::remove_file(&dest)
|
|
|
|
|
.map_err(|e| CatalogError::Io(format!("replacing {}: {e}", dest.display())))?;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Dropped immediately: the copy is complete when `copy_to` returns, and
|
|
|
|
|
// holding the connection open would leave a `-wal` beside a file whose
|
|
|
|
|
// whole purpose is to be a single self-contained artefact.
|
|
|
|
|
drop(crate::sync::copy_to(conn, &dest)?);
|
|
|
|
|
|
|
|
|
|
prune(catalog);
|
|
|
|
|
Ok(dest)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Back up before a migration, if there is anything to back up.
|
|
|
|
|
///
|
|
|
|
|
/// Called from [`Catalog::open`](crate::Catalog::open) between `configure` and
|
|
|
|
|
/// `migrate`. NFR-R2 asks for this and the reasoning is narrower than "backups
|
|
|
|
|
/// are prudent": a migration is the one routine operation that rewrites table
|
|
|
|
|
/// structure, so it is the likeliest way a catalog becomes unreadable, and it
|
|
|
|
|
/// is the one moment where the pre-change state is still on disk to be copied.
|
|
|
|
|
/// Afterwards there is nothing left to take a copy *of*.
|
|
|
|
|
///
|
|
|
|
|
/// A no-op in the two cases where it would cost without buying anything: a
|
|
|
|
|
/// catalog already at [`schema::SCHEMA_VERSION`], and a brand-new file at
|
|
|
|
|
/// version 0 with no tables in it yet.
|
|
|
|
|
pub fn backup_before_migration(conn: &Connection, catalog: &Path) -> Result<(), CatalogError> {
|
|
|
|
|
let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
|
|
|
|
|
if from == 0 || from >= schema::SCHEMA_VERSION {
|
|
|
|
|
return Ok(());
|
|
|
|
|
}
|
|
|
|
|
let path = backup(conn, catalog)?;
|
|
|
|
|
log::info!(
|
|
|
|
|
"backed up catalog at v{from} to {} before migrating to v{}",
|
|
|
|
|
path.display(),
|
|
|
|
|
schema::SCHEMA_VERSION
|
|
|
|
|
);
|
|
|
|
|
Ok(())
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// The backups available for `catalog`, newest first.
|
|
|
|
|
///
|
|
|
|
|
/// Never fails: an unreadable or absent backup directory means there are no
|
|
|
|
|
/// backups, which is a fact about the offer to make rather than an error to
|
|
|
|
|
/// report on top of the corruption the user is already looking at.
|
|
|
|
|
pub fn backups(catalog: &Path) -> Vec<Backup> {
|
|
|
|
|
let dir = backup_dir(catalog);
|
|
|
|
|
let Ok(entries) = std::fs::read_dir(&dir) else {
|
|
|
|
|
return Vec::new();
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
let mut out: Vec<Backup> = entries
|
|
|
|
|
.flatten()
|
|
|
|
|
.filter_map(|e| {
|
|
|
|
|
let path = e.path();
|
|
|
|
|
let taken_at = timestamp_of(&path)?;
|
|
|
|
|
let bytes = e.metadata().ok()?.len();
|
|
|
|
|
Some(Backup {
|
|
|
|
|
path,
|
|
|
|
|
taken_at,
|
|
|
|
|
bytes,
|
|
|
|
|
})
|
|
|
|
|
})
|
|
|
|
|
.collect();
|
|
|
|
|
out.sort_by_key(|b| std::cmp::Reverse(b.taken_at));
|
|
|
|
|
out
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Put a backup back in place of the damaged catalog.
|
|
|
|
|
///
|
|
|
|
|
/// **Every connection to `catalog` must be closed first.** This replaces the
|
|
|
|
|
/// file underneath anything still holding it open, which on a live connection
|
|
|
|
|
/// is how a *second* corrupt catalog gets made.
|
|
|
|
|
///
|
|
|
|
|
/// The order is deliberate:
|
|
|
|
|
///
|
|
|
|
|
/// 1. The backup is checked. A restore that installs a second damaged file
|
|
|
|
|
/// leaves the user with nothing to try next.
|
|
|
|
|
/// 2. The damaged catalog is renamed aside, and its `-wal` and `-shm` are
|
|
|
|
|
/// **deleted**. This is the step that is easy to leave out and fatal to
|
|
|
|
|
/// leave out: a journal belonging to the old file, sitting beside the new
|
|
|
|
|
/// one under the same name, is replayed into it on the next open. That is
|
|
|
|
|
/// not a restore, it is a fresh corruption with the evidence gone.
|
|
|
|
|
/// 3. The backup is *copied* into place, not moved, so a failure here can be
|
|
|
|
|
/// retried against the same backup.
|
|
|
|
|
pub fn restore(catalog: &Path, backup: &Path) -> Result<(), CatalogError> {
|
|
|
|
|
check_file(backup)?;
|
|
|
|
|
set_aside(catalog)?;
|
|
|
|
|
std::fs::copy(backup, catalog).map_err(|e| {
|
|
|
|
|
CatalogError::Io(format!(
|
|
|
|
|
"restoring {} from {}: {e}",
|
|
|
|
|
catalog.display(),
|
|
|
|
|
backup.display()
|
|
|
|
|
))
|
|
|
|
|
})?;
|
|
|
|
|
log::info!(
|
|
|
|
|
"restored {} from backup {}",
|
|
|
|
|
catalog.display(),
|
|
|
|
|
backup.display()
|
|
|
|
|
);
|
|
|
|
|
Ok(())
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Move a damaged catalog out of the way so the next open builds a fresh one.
|
|
|
|
|
///
|
|
|
|
|
/// This is the rebuild path (NFR-R6's second offer) and also the first half of
|
|
|
|
|
/// a [`restore`]. Nothing else is needed to rebuild: the next
|
|
|
|
|
/// [`Catalog::open`](crate::Catalog::open) creates an empty catalog at the
|
|
|
|
|
/// current schema, and the ordinary scan repopulates it from sources and
|
|
|
|
|
/// sidecars — which is precisely invariant §5.2.4 being spent rather than
|
|
|
|
|
/// merely asserted.
|
|
|
|
|
///
|
|
|
|
|
/// Returns where the damaged file was put, or `None` if there was no catalog
|
|
|
|
|
/// to move — a caller may be recovering from a file SQLite could not open
|
|
|
|
|
/// because it was never created.
|
|
|
|
|
pub fn set_aside(catalog: &Path) -> Result<Option<PathBuf>, CatalogError> {
|
|
|
|
|
let moved = if catalog.exists() {
|
|
|
|
|
let dest = with_suffix(catalog, DAMAGED_SUFFIX);
|
|
|
|
|
// An earlier damaged copy is replaced rather than accumulating: two of
|
|
|
|
|
// these is two full-size catalogs on the user's disk, and the older
|
|
|
|
|
// one has already been superseded by a recovery the user completed.
|
|
|
|
|
let _ = std::fs::remove_file(&dest);
|
|
|
|
|
// The rename first, so that a failure here leaves the journals with
|
|
|
|
|
// the file they belong to rather than orphaned beside a catalog that
|
|
|
|
|
// is still in use.
|
|
|
|
|
std::fs::rename(catalog, &dest).map_err(|e| {
|
|
|
|
|
CatalogError::Io(format!(
|
|
|
|
|
"setting aside {} as {}: {e}",
|
|
|
|
|
catalog.display(),
|
|
|
|
|
dest.display()
|
|
|
|
|
))
|
|
|
|
|
})?;
|
|
|
|
|
log::warn!(
|
|
|
|
|
"catalog {} was damaged; kept as {}",
|
|
|
|
|
catalog.display(),
|
|
|
|
|
dest.display()
|
|
|
|
|
);
|
|
|
|
|
Some(dest)
|
|
|
|
|
} else {
|
|
|
|
|
None
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
// Then the journals, whether or not there was a catalog to move: a `-wal`
|
|
|
|
|
// orphaned beside a missing database is replayed into whatever takes that
|
|
|
|
|
// name next, which would not be a restore but a fresh corruption with the
|
|
|
|
|
// evidence gone.
|
|
|
|
|
for sidecar in journals(catalog) {
|
|
|
|
|
if let Err(e) = std::fs::remove_file(&sidecar) {
|
|
|
|
|
if e.kind() != std::io::ErrorKind::NotFound {
|
|
|
|
|
return Err(CatalogError::Io(format!(
|
|
|
|
|
"removing stale journal {}: {e}",
|
|
|
|
|
sidecar.display()
|
|
|
|
|
)));
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
Ok(moved)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Delete backups beyond [`KEEP_BACKUPS`].
|
|
|
|
|
///
|
|
|
|
|
/// Best-effort and silent about individual failures: failing to delete an old
|
|
|
|
|
/// backup is not a reason to fail the new one, which is already written.
|
|
|
|
|
fn prune(catalog: &Path) {
|
|
|
|
|
for old in backups(catalog).into_iter().skip(KEEP_BACKUPS) {
|
|
|
|
|
if let Err(e) = std::fs::remove_file(&old.path) {
|
|
|
|
|
log::warn!("could not prune backup {}: {e}", old.path.display());
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// The WAL and shared-memory files SQLite keeps beside a database.
|
|
|
|
|
fn journals(catalog: &Path) -> [PathBuf; 2] {
|
|
|
|
|
[with_suffix(catalog, "wal"), with_suffix(catalog, "shm")]
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// `catalog.sqlite` plus `-suffix`, the way SQLite names its own sidecars.
|
|
|
|
|
///
|
|
|
|
|
/// Appended to the whole filename rather than replacing the extension, so
|
|
|
|
|
/// `catalog.sqlite-wal` is what SQLite would look for and `catalog.sqlite-
|
|
|
|
|
/// damaged` sorts next to the catalog it came from.
|
|
|
|
|
fn with_suffix(catalog: &Path, suffix: &str) -> PathBuf {
|
|
|
|
|
let mut s = catalog.as_os_str().to_os_string();
|
|
|
|
|
s.push("-");
|
|
|
|
|
s.push(suffix);
|
|
|
|
|
PathBuf::from(s)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Read the timestamp out of a backup's filename, or `None` if this is not one.
|
|
|
|
|
///
|
|
|
|
|
/// Doubles as the filter that keeps [`backups`] from offering the user
|
|
|
|
|
/// something that is not a catalog — a stray file in the directory, or a `-wal`
|
|
|
|
|
/// left by a crash mid-backup.
|
|
|
|
|
fn timestamp_of(path: &Path) -> Option<i64> {
|
|
|
|
|
let name = path.file_name()?.to_str()?;
|
|
|
|
|
name.strip_prefix("catalog-")?
|
|
|
|
|
.strip_suffix(".sqlite")?
|
|
|
|
|
.parse()
|
|
|
|
|
.ok()
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Seconds since the epoch, or 0 if the clock is before it.
|
|
|
|
|
fn now() -> i64 {
|
|
|
|
|
SystemTime::now()
|
|
|
|
|
.duration_since(UNIX_EPOCH)
|
|
|
|
|
.map(|d| d.as_secs() as i64)
|
|
|
|
|
.unwrap_or(0)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[cfg(test)]
|
|
|
|
|
mod tests {
|
|
|
|
|
use super::*;
|
|
|
|
|
use crate::Catalog;
|
|
|
|
|
use std::io::{Seek, SeekFrom, Write};
|
|
|
|
|
|
|
|
|
|
/// A scratch directory that cleans up with the test.
|
|
|
|
|
fn tempdir(tag: &str) -> PathBuf {
|
|
|
|
|
let base = std::env::temp_dir().join(format!(
|
|
|
|
|
"dr-recovery-{tag}-{}-{:?}",
|
|
|
|
|
std::process::id(),
|
|
|
|
|
std::thread::current().id()
|
|
|
|
|
));
|
|
|
|
|
let _ = std::fs::remove_dir_all(&base);
|
|
|
|
|
std::fs::create_dir_all(&base).unwrap();
|
|
|
|
|
base
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// A catalog on disk with enough rows to span several pages, closed.
|
|
|
|
|
///
|
|
|
|
|
/// Closed matters: WAL means the rows are in `catalog.sqlite-wal` until
|
|
|
|
|
/// something checkpoints, and a test that corrupted the main file while
|
|
|
|
|
/// the data was still in the journal would be corrupting empty space.
|
|
|
|
|
fn fixture(path: &Path, images: i64) {
|
|
|
|
|
let cat = Catalog::open(path).unwrap();
|
|
|
|
|
let c = cat.connection();
|
|
|
|
|
c.execute(
|
|
|
|
|
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
|
|
|
|
[],
|
|
|
|
|
)
|
|
|
|
|
.unwrap();
|
|
|
|
|
c.execute(
|
|
|
|
|
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
|
|
|
|
|
VALUES ('u1', 'Iceland', 0, 0, 1, 1)",
|
|
|
|
|
[],
|
|
|
|
|
)
|
|
|
|
|
.unwrap();
|
|
|
|
|
for i in 1..=images {
|
|
|
|
|
c.execute(
|
|
|
|
|
"INSERT INTO images(id, root_id, source_ref, added_at)
|
|
|
|
|
VALUES (?1, 1, ?2, 0)",
|
|
|
|
|
rusqlite::params![i, format!("DCIM/IMG_{i:05}.CR3")],
|
|
|
|
|
)
|
|
|
|
|
.unwrap();
|
|
|
|
|
}
|
|
|
|
|
crate::sync::checkpoint(c).unwrap();
|
|
|
|
|
drop(cat);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/// Scribble over everything past the first two pages.
|
|
|
|
|
///
|
|
|
|
|
/// Past them rather than over them so that page 1 — the header and the
|
|
|
|
|
/// schema — survives: this produces a file SQLite is willing to open and
|
|
|
|
|
/// then finds damaged, which is the case `quick_check` exists for. Wiping
|
|
|
|
|
/// the header instead produces `SQLITE_NOTADB` at the first pragma, which
|
|
|
|
|
/// is a different branch and has its own test.
|
|
|
|
|
fn corrupt(path: &Path) {
|
|
|
|
|
let mut f = std::fs::OpenOptions::new().write(true).open(path).unwrap();
|
|
|
|
|
let len = f.metadata().unwrap().len();
|
|
|
|
|
assert!(
|
|
|
|
|
len > 8192,
|
|
|
|
|
"fixture is only {len} bytes; corrupting past page 2 would be a no-op"
|
|
|
|
|
);
|
|
|
|
|
let junk = vec![0x5a_u8; (len - 8192) as usize];
|
|
|
|
|
f.seek(SeekFrom::Start(8192)).unwrap();
|
|
|
|
|
f.write_all(&junk).unwrap();
|
|
|
|
|
f.sync_all().unwrap();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn a_healthy_catalog_passes() {
|
|
|
|
|
let cat = Catalog::in_memory().unwrap();
|
|
|
|
|
integrity_check(cat.connection()).unwrap();
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn a_corrupt_catalog_is_reported_as_corrupt_not_as_sqlite() {
|
|
|
|
|
// The whole point of the variant: this used to arrive as whatever
|
|
|
|
|
// rusqlite error the first failing query produced, with nowhere to
|
|
|
|
|
// hang a recovery offer.
|
|
|
|
|
let dir = tempdir("detect");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
fixture(&path, 500);
|
|
|
|
|
corrupt(&path);
|
|
|
|
|
|
|
|
|
|
assert!(matches!(
|
|
|
|
|
Catalog::open_verified(&path),
|
|
|
|
|
Err(CatalogError::Corrupt { .. })
|
|
|
|
|
));
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn a_file_that_is_not_a_database_is_also_corrupt() {
|
|
|
|
|
// A truncated or overwritten catalog never reaches `quick_check`: the
|
|
|
|
|
// first pragma fails with SQLITE_NOTADB. Same accident to the user,
|
|
|
|
|
// same two offers, so it must classify the same way.
|
|
|
|
|
let dir = tempdir("notadb");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
std::fs::write(&path, b"this is not a catalog, it is a text file\n").unwrap();
|
|
|
|
|
|
|
|
|
|
assert!(matches!(
|
|
|
|
|
Catalog::open(&path),
|
|
|
|
|
Err(CatalogError::Corrupt { .. })
|
|
|
|
|
));
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn restoring_a_backup_recovers_the_collections_a_rebuild_would_lose() {
|
|
|
|
|
// The first NFR-R6 branch, asserted on the thing that distinguishes it
|
|
|
|
|
// from the second: a collection exists nowhere but the catalog, so it
|
|
|
|
|
// is the evidence that the *contents* came back and not merely a
|
|
|
|
|
// readable file (docs/catalog.md §8.1).
|
|
|
|
|
let dir = tempdir("restore");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
fixture(&path, 500);
|
|
|
|
|
|
|
|
|
|
{
|
|
|
|
|
let cat = Catalog::open(&path).unwrap();
|
|
|
|
|
backup(cat.connection(), &path).unwrap();
|
|
|
|
|
}
|
|
|
|
|
corrupt(&path);
|
|
|
|
|
assert!(matches!(
|
|
|
|
|
Catalog::open_verified(&path),
|
|
|
|
|
Err(CatalogError::Corrupt { .. })
|
|
|
|
|
));
|
|
|
|
|
|
|
|
|
|
let newest = backups(&path).into_iter().next().expect("a backup exists");
|
|
|
|
|
restore(&path, &newest.path).unwrap();
|
|
|
|
|
|
|
|
|
|
let cat = Catalog::open_verified(&path).unwrap();
|
|
|
|
|
let name: String = cat
|
|
|
|
|
.connection()
|
|
|
|
|
.query_row("SELECT name FROM collections", [], |r| r.get(0))
|
|
|
|
|
.unwrap();
|
|
|
|
|
assert_eq!(name, "Iceland");
|
|
|
|
|
let images: i64 = cat
|
|
|
|
|
.connection()
|
|
|
|
|
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
|
|
|
|
|
.unwrap();
|
|
|
|
|
assert_eq!(images, 500);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn a_damaged_backup_is_refused_rather_than_installed() {
|
|
|
|
|
let dir = tempdir("badbackup");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
fixture(&path, 500);
|
|
|
|
|
{
|
|
|
|
|
let cat = Catalog::open(&path).unwrap();
|
|
|
|
|
backup(cat.connection(), &path).unwrap();
|
|
|
|
|
}
|
|
|
|
|
let newest = backups(&path).into_iter().next().unwrap();
|
|
|
|
|
corrupt(&newest.path);
|
|
|
|
|
corrupt(&path);
|
|
|
|
|
|
|
|
|
|
assert!(matches!(
|
|
|
|
|
restore(&path, &newest.path),
|
|
|
|
|
Err(CatalogError::Corrupt { .. })
|
|
|
|
|
));
|
|
|
|
|
// And the damaged catalog is still where it was, so the second offer
|
|
|
|
|
// is still available.
|
|
|
|
|
assert!(path.exists());
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn setting_aside_leaves_a_fresh_catalog_to_rebuild_into() {
|
|
|
|
|
// The second NFR-R6 branch. What makes it a rebuild rather than a data
|
|
|
|
|
// loss is invariant §5.2.4, which lives outside this crate — what is
|
|
|
|
|
// testable here is that the damaged file is out of the way, kept, and
|
|
|
|
|
// that the next open succeeds on an empty catalog at the current
|
|
|
|
|
// schema, which is what a scan then fills.
|
|
|
|
|
let dir = tempdir("rebuild");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
fixture(&path, 500);
|
|
|
|
|
corrupt(&path);
|
|
|
|
|
|
|
|
|
|
let kept = set_aside(&path).unwrap().expect("the catalog was there");
|
|
|
|
|
assert!(kept.exists(), "the damaged catalog was deleted, not kept");
|
|
|
|
|
assert!(!path.exists());
|
|
|
|
|
|
|
|
|
|
let cat = Catalog::open_verified(&path).unwrap();
|
|
|
|
|
let images: i64 = cat
|
|
|
|
|
.connection()
|
|
|
|
|
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
|
|
|
|
|
.unwrap();
|
|
|
|
|
assert_eq!(images, 0);
|
|
|
|
|
let v: i64 = cat
|
|
|
|
|
.connection()
|
|
|
|
|
.query_row("PRAGMA user_version", [], |r| r.get(0))
|
|
|
|
|
.unwrap();
|
|
|
|
|
assert_eq!(v, schema::SCHEMA_VERSION);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn a_stale_journal_does_not_follow_the_catalog_into_recovery() {
|
|
|
|
|
// The step that is easy to omit: a `-wal` belonging to the damaged
|
|
|
|
|
// file is replayed into whatever takes its name next.
|
|
|
|
|
let dir = tempdir("journal");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
fixture(&path, 500);
|
|
|
|
|
std::fs::write(with_suffix(&path, "wal"), b"stale").unwrap();
|
|
|
|
|
|
|
|
|
|
set_aside(&path).unwrap();
|
|
|
|
|
assert!(!with_suffix(&path, "wal").exists());
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn a_migration_is_backed_up_before_it_runs() {
|
|
|
|
|
// NFR-R2's second clause, against a real v1 catalog rather than a
|
|
|
|
|
// faked version number: the point is not that *a* file appears but
|
|
|
|
|
// that it holds the state from before the migration, which is the only
|
|
|
|
|
// state that is any use if the migration is what breaks it.
|
|
|
|
|
let dir = tempdir("premigrate");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
{
|
|
|
|
|
let c = Connection::open(&path).unwrap();
|
|
|
|
|
schema::configure(&c).unwrap();
|
|
|
|
|
// `v1_for_attached` names the schema it targets, and "main" is a
|
|
|
|
|
// schema like any other — so this is the real v1, without needing
|
|
|
|
|
// `V1` itself to become visible outside its module.
|
|
|
|
|
c.execute_batch(&schema::v1_for_attached("main")).unwrap();
|
|
|
|
|
c.pragma_update(None, "user_version", 1).unwrap();
|
|
|
|
|
c.execute(
|
|
|
|
|
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
|
|
|
|
[],
|
|
|
|
|
)
|
|
|
|
|
.unwrap();
|
|
|
|
|
crate::sync::checkpoint(&c).unwrap();
|
|
|
|
|
}
|
|
|
|
|
assert!(backups(&path).is_empty());
|
|
|
|
|
|
|
|
|
|
Catalog::open(&path).unwrap();
|
|
|
|
|
|
|
|
|
|
let taken = backups(&path);
|
|
|
|
|
assert_eq!(taken.len(), 1, "no backup was taken before the migration");
|
|
|
|
|
check_file(&taken[0].path).unwrap();
|
|
|
|
|
let kept = Connection::open(&taken[0].path).unwrap();
|
|
|
|
|
let v: i64 = kept
|
|
|
|
|
.query_row("PRAGMA user_version", [], |r| r.get(0))
|
|
|
|
|
.unwrap();
|
|
|
|
|
assert_eq!(v, 1, "the backup was taken after the migration, not before");
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn opening_an_up_to_date_catalog_takes_no_backup() {
|
|
|
|
|
// Or every background task that opens the catalog would copy it.
|
|
|
|
|
let dir = tempdir("nobackup");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
fixture(&path, 10);
|
|
|
|
|
Catalog::open(&path).unwrap();
|
|
|
|
|
assert!(backups(&path).is_empty());
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn only_the_newest_generations_are_kept() {
|
|
|
|
|
let dir = tempdir("prune");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
fixture(&path, 10);
|
|
|
|
|
let cat = Catalog::open(&path).unwrap();
|
|
|
|
|
|
|
|
|
|
// Written by hand rather than by calling `backup` in a loop: the
|
|
|
|
|
// filename carries whole seconds, so real calls would collide.
|
|
|
|
|
std::fs::create_dir_all(backup_dir(&path)).unwrap();
|
|
|
|
|
for t in 1..=KEEP_BACKUPS as i64 + 2 {
|
|
|
|
|
drop(
|
|
|
|
|
crate::sync::copy_to(
|
|
|
|
|
cat.connection(),
|
|
|
|
|
&backup_dir(&path).join(format!("catalog-{t}.sqlite")),
|
|
|
|
|
)
|
|
|
|
|
.unwrap(),
|
|
|
|
|
);
|
|
|
|
|
}
|
|
|
|
|
prune(&path);
|
|
|
|
|
|
|
|
|
|
let kept = backups(&path);
|
|
|
|
|
assert_eq!(kept.len(), KEEP_BACKUPS);
|
|
|
|
|
// Newest first, and the newest is the highest timestamp.
|
|
|
|
|
assert_eq!(kept[0].taken_at, KEEP_BACKUPS as i64 + 2);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#[test]
|
|
|
|
|
fn a_stray_file_in_the_backup_directory_is_not_offered_as_one() {
|
|
|
|
|
let dir = tempdir("stray");
|
|
|
|
|
let path = dir.join("catalog.sqlite");
|
|
|
|
|
fixture(&path, 10);
|
|
|
|
|
std::fs::create_dir_all(backup_dir(&path)).unwrap();
|
|
|
|
|
std::fs::write(backup_dir(&path).join("notes.txt"), b"hello").unwrap();
|
|
|
|
|
std::fs::write(backup_dir(&path).join("catalog-7.sqlite-wal"), b"x").unwrap();
|
|
|
|
|
|
|
|
|
|
assert!(backups(&path).is_empty());
|
|
|
|
|
}
|
|
|
|
|
}
|