Compare commits

..
1 Commits
Author SHA1 Message Date
Gitea Actions 3d3083cc32 manual: deploy from 880915e061 2026-10-09 15:31:44 +00:00
850 changed files with 151 additions and 374005 deletions
-41
View File
@@ -1,41 +0,0 @@
{
"permissions": {
"allow": [
"Bash(curl -s \"https://api.github.com/search/code?q=WrapTexture+org:Noesis\" -H \"Accept: application/vnd.github+json\")",
"Bash(curl -s \"https://api.github.com/orgs/Noesis/repos?per_page=100\")",
"WebFetch(domain:wiki.wxwidgets.org)",
"Bash(curl -sS \"https://gitlab.com/sciter-engine/sciter-js-sdk/-/raw/main/samples.gpu/hello-es-triangle.htm\")",
"Bash(curl -sS \"https://gitlab.com/api/v4/projects/sciter-engine%2Fsciter-js-sdk/repository/tree?path=include&recursive=true&per_page=100&ref=main\")",
"Bash(python3 -c ' *)",
"Bash(curl -sL --max-time 40 \"https://api.github.com/repos/wxWidgets/wxWidgets/contents/src?ref=master\")",
"Bash(curl -sL --max-time 40 \"https://api.github.com/repos/wxWidgets/wxWidgets/contents/include/wx/android?ref=master\")",
"Bash(curl -sL --max-time 40 -H \"Accept: application/vnd.github.text-match+json\" \"https://api.github.com/search/code?q=vulkan+repo:wxWidgets/wxWidgets\")",
"Bash(curl -sS \"https://gitlab.com/sciter-engine/sciter-js-sdk/-/raw/main/include/sciter-x-video-api.h\")",
"WebFetch(domain:docs.wxwidgets.org)",
"Bash(curl -sS \"https://gitlab.com/sciter-engine/sciter-js-sdk/-/raw/main/CHANGELOG.md\")",
"Bash(curl -sL --max-time 40 \"https://raw.githubusercontent.com/wxWidgets/wxWidgets/master/docs/readme.txt\")",
"Bash(curl -sL --max-time 40 \"https://raw.githubusercontent.com/wxWidgets/wxWidgets/master/docs/licence.txt\")",
"Bash(curl -sL --max-time 40 \"https://raw.githubusercontent.com/wxWidgets/wxWidgets/master/docs/licendu.txt\")",
"Bash(curl -sS \"https://gitlab.com/api/v4/projects/sciter-engine%2Fsciter-js-sdk/repository/tree?path=build&recursive=true&per_page=100&ref=main\")",
"Bash(curl -sS \"https://gitlab.com/sciter-engine/sciter-js-sdk/-/raw/main/premake5.lua\")",
"WebFetch(domain:slack-chats.kotlinlang.org)",
"Bash(curl -sS -L \"https://sciter.com/\")",
"Bash(curl -sL --max-time 45 \"https://api.github.com/orgs/ultralight-ux/repos?per_page=100&sort=pushed\")",
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/GNOME/gtk/-/raw/main/gdk/gdkdmabuftexturebuilder.h\")",
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/GNOME/gtk/-/raw/main/gdk/gdkgltexturebuilder.h\")",
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/api/v4/projects/GNOME%2Fgtk/repository/tree?path=gdk&ref=main&per_page=100\")",
"WebFetch(domain:docs.slint.dev)",
"WebFetch(domain:releases.slint.dev)",
"WebFetch(domain:flutter.dev)",
"Bash(curl -sS -L \"https://sciter.com/forums/topic/status-of-quark-sciter-lite-sciterjs-android-ios/\")",
"Bash(curl -sL --max-time 45 \"https://api.github.com/repos/ultralight-ux/AppCore/git/trees/master?recursive=1\")",
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/GNOME/gtk/-/raw/main/gdk/meson.build\")",
"WebFetch(domain:www.jetbrains.com)",
"Bash(curl -sS \"https://gitlab.com/api/v4/projects/sciter-engine%2Fsciter-js-sdk/repository/tree?path=demos.lite&recursive=true&per_page=100&ref=main\")",
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/api/v4/projects/GNOME%2Fgtk/repository/commits?path=gdk/android/gdkandroidglcontext.c&ref_name=main&per_page=20\")",
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/GNOME/gtk/-/raw/main/gdk/android/meson.build\")",
"WebFetch(domain:docs.sciter.com)",
"Bash(curl -sS -L \"https://sciter.com/support-of-displayflex-and-displaygrid-in-sciter/\")"
]
}
}
View File
-31
View File
@@ -1,31 +0,0 @@
# Model weights live in LFS.
#
# `models/**/*.onnx` is tens of MB of binary that changes wholesale
# when it changes at all. In ordinary git objects every future revision of it
# would be stored in full, in every clone, forever — and the one thing nobody
# can do with it is a useful diff.
#
# Consequence worth knowing before it bites: a clone without git-lfs gets a
# ~130-byte pointer file where the model should be. `dr-segment`'s build script
# detects exactly that and fails with an instruction rather than embedding the
# pointer and failing at inference time.
*.onnx filter=lfs diff=lfs merge=lfs -text
# Test photographs live in LFS too, and are fetched only by the tests that
# need them.
#
# `fixtures/**` holds real camera files — a twelve-frame panorama set is
# 325 MB — and CI's `git lfs pull` excludes the directory, so a checkout
# carries pointers there until a merge test asks for the frames. Same
# reasoning as the models, with the opposite default: the model is not
# optional and the fixtures are.
fixtures/** filter=lfs diff=lfs merge=lfs -text
# The manual's pictures live in LFS for the same reason the models do: a
# screenshot or a GIF changes wholesale when the interface it shows changes,
# and every re-recording would otherwise stay in every clone for good. The
# desktop and benchmark legs exclude the directory, since nothing they build
# or test reads it; the Android and Windows legs fetch it, because the APK
# and the installer carry the manual (docs/manual/index.html) with its
# pictures, and their packagers refuse a pointer.
docs/manual/media/** filter=lfs diff=lfs merge=lfs -text
-170
View File
@@ -1,170 +0,0 @@
name: '🐳 Android image'
# Builds and pushes gitea.tourolle.paris/dtourolle/darkroom-android, the job
# container for the Android leg of build-and-test.yml.
#
# It exists because that image previously lived only on a developer's laptop:
# the workflow referenced a tag that had never been pushed, and every Android
# job died at `docker pull` with "manifest unknown" before running a step. The
# image is now reproducible from the repo rather than from one machine.
#
# Called by build-and-test.yml on every push, and runnable by hand via
# workflow_dispatch. It is cheap when nothing changed — see the guard below.
on:
workflow_call:
inputs:
force:
description: 'Rebuild even if the registry already has this image ("true"/"false")'
type: string
default: 'false'
workflow_dispatch:
inputs:
force:
description: 'Rebuild even if the registry already has this image ("true"/"false")'
type: string
default: 'false'
# Gitea's act_runner mangles boolean workflow inputs passed through an
# expression — they arrive as false regardless of what was sent. Every input
# here is a string compared with == 'true', as in KPN's docker.yaml.
env:
IMAGE: gitea.tourolle.paris/dtourolle/darkroom-android
jobs:
build:
runs-on: linux/amd64
name: Build and push
# Deliberately NOT in a container: this job needs the host Docker daemon to
# build an image, and the host's cached ~/.docker/config.json to push it.
# That is also why there is no `docker login` step — the runner host was
# authenticated to the registry during setup.
steps:
# The host has no Node, so the JS-based actions/checkout cannot run here.
# A minimal shallow fetch with plain git gets the same tree.
- name: Checkout
run: |
set -e
git init -q .
git remote add origin "${{ github.server_url }}/${{ github.repository }}.git"
git -c http.extraheader="AUTHORIZATION: basic $(printf '%s' '${{ github.actor }}:${{ github.token }}' | base64 -w0)" \
fetch --depth 1 origin "${{ github.sha }}"
git checkout -q FETCH_HEAD
# The image is tagged by the content of docker/android, not by the commit
# that happened to touch it. `git rev-parse HEAD:<dir>` is the tree object
# id — it changes when and only when a file in that directory changes, so
# an unrelated push reuses the existing image and a Dockerfile edit can
# never silently keep serving a stale `latest`.
#
# Using the commit sha instead would rebuild 7 GB on every push; using a
# paths-filter action would need a container that has Node, and the only
# one this repo would reach for is the very image being built.
- name: Resolve image tag
id: tag
run: |
set -e
TREE=$(git rev-parse HEAD:docker/android)
echo "tree=$TREE" >> "$GITHUB_OUTPUT"
echo "docker/android tree: $TREE"
# Skip the build when the registry already holds this exact content. This
# is what keeps the job a few seconds long on a normal push, and what
# makes it self-healing: if the tag is missing for any reason, including
# the image having never been pushed at all, it gets built here.
#
# The probe is curl against the registry API, NOT `docker manifest
# inspect`. The latter exits 1 on this registry even for tags that are
# demonstrably present — jellytau-builder:latest answers HTTP 200 to the
# API while `docker manifest inspect` reports "manifest unknown" for it.
# Trusting that would have rebuilt 7 GB on every single push.
#
# A HEAD request also gives the digest for free, which is how the repoint
# decision below is made without pulling any layers.
- name: Query registry
id: check
env:
# The runner's own credentials, so this does not depend on how the
# host's ~/.docker/config.json happens to be set up.
REG_USER: ${{ github.actor }}
REG_PASS: ${{ github.token }}
TREE: ${{ steps.tag.outputs.tree }}
run: |
set -eu
ACCEPT='application/vnd.oci.image.index.v1+json,application/vnd.docker.distribution.manifest.v2+json,application/vnd.oci.image.manifest.v1+json,application/vnd.docker.distribution.manifest.list.v2+json'
API="https://gitea.tourolle.paris/v2/dtourolle/darkroom-android/manifests"
# Prints "<http-status> <digest-or-empty>" for a tag.
probe() {
curl -sI -u "$REG_USER:$REG_PASS" -H "Accept: $ACCEPT" "$API/$1" \
| tr -d '\r' \
| awk 'BEGIN{s="000";d=""} /^HTTP/{s=$2} tolower($1)=="docker-content-digest:"{d=$2} END{print s, d}'
}
read -r TREE_STATUS TREE_DIGEST <<EOF
$(probe "$TREE")
EOF
read -r LATEST_STATUS LATEST_DIGEST <<EOF
$(probe latest)
EOF
echo "tag $TREE -> HTTP $TREE_STATUS ${TREE_DIGEST:-(no digest)}"
echo "tag latest -> HTTP $LATEST_STATUS ${LATEST_DIGEST:-(no digest)}"
# Build unless the registry definitively confirms this content is
# already there. An auth failure or an unreachable registry lands
# here too, and rebuilding needlessly is the safe direction to fail —
# skipping a build that was needed is what breaks the Android job.
if [ "${{ inputs.force }}" = "true" ]; then
echo "forced rebuild requested"
echo "build=true" >> "$GITHUB_OUTPUT"
echo "repoint=false" >> "$GITHUB_OUTPUT"
elif [ "$TREE_STATUS" != "200" ]; then
echo "registry does not have this content — building"
echo "build=true" >> "$GITHUB_OUTPUT"
echo "repoint=false" >> "$GITHUB_OUTPUT"
elif [ -n "$TREE_DIGEST" ] && [ "$TREE_DIGEST" = "$LATEST_DIGEST" ]; then
echo "registry is already correct — nothing to do"
echo "build=false" >> "$GITHUB_OUTPUT"
echo "repoint=false" >> "$GITHUB_OUTPUT"
else
echo "content is present but latest points elsewhere — repointing"
echo "build=false" >> "$GITHUB_OUTPUT"
echo "repoint=true" >> "$GITHUB_OUTPUT"
fi
# Context is docker/android, matching the README's build command. The
# Dockerfile COPYs nothing from the repo, so it needs no wider context —
# and a narrow context keeps the daemon from tarring up the whole tree,
# target/ included.
- name: Build
if: ${{ steps.check.outputs.build == 'true' }}
run: |
set -e
docker build \
-t "$IMAGE:${{ steps.tag.outputs.tree }}" \
-t "$IMAGE:latest" \
docker/android
# Both tags are pushed: the tree tag is what the guard above looks for on
# the next run, and `latest` is what build-and-test.yml pulls.
- name: Push
if: ${{ steps.check.outputs.build == 'true' }}
run: |
set -e
docker push "$IMAGE:${{ steps.tag.outputs.tree }}"
docker push "$IMAGE:latest"
# A cache hit on the tree tag says nothing about where `latest` points — a
# reverted Dockerfile or a build from another branch can leave it on
# different content. This runs only when the digests above actually
# disagree, so the common case costs nothing; the layers are already in
# the registry, so the push that follows uploads a manifest, not 7 GB.
- name: Repoint latest
if: ${{ steps.check.outputs.repoint == 'true' }}
run: |
set -e
docker pull "$IMAGE:${{ steps.tag.outputs.tree }}"
docker tag "$IMAGE:${{ steps.tag.outputs.tree }}" "$IMAGE:latest"
docker push "$IMAGE:latest"
-193
View File
@@ -1,193 +0,0 @@
name: Benchmarks
# The suite docs/dev/requirements.md §8 has been promising since it was written:
# "an automated benchmark suite against a synthetic 50k catalog, run per-commit
# … A regression beyond stated tolerance fails the build."
#
# Its own workflow rather than a step inside build-and-test.yml, and the reason
# is what a failure here means. A red `Build and test` says the code is wrong; a
# red `Benchmarks` says the code is slower than it was, which is a different
# conversation, is read by different people, and must not be reachable by
# retrying a flaky compile.
#
# # Why this is split in two
#
# §8 names "the reference desktop", not CI, and it is right to. So:
#
# cpu — runs on every push. It needs no adapter and no display, and the
# budgets it asserts (a 50k catalog opening inside two seconds) have
# two orders of magnitude of headroom, so a modest runner can be held
# to them honestly. Machine-sensitive budgets — throughput targets
# written for a 24-thread desktop — are reported here rather than
# asserted; `dr-bench` decides that per metric and says so in its
# report. Asserting them on a two-core container would produce exactly
# what core/dr-gpu/tests/frame_budget.rs refused to produce: a red gate
# everybody learns to ignore.
#
# gpu — the frame budget, which already exists and already skips itself where
# there is no adapter. Not on push: it would build wgpu and naga on
# every commit to establish, every time, that this runner has no GPU. It
# runs on demand (Actions → Run workflow) so that a runner that *does*
# have one can be pointed at it, and the numbers it produces belong in
# docs/dev/frame-budget.md by hand, as they already are.
on:
push:
branches: [main, master, develop]
pull_request:
branches: [main, master, develop]
workflow_dispatch:
jobs:
cpu:
runs-on: linux/amd64
name: CPU and I/O (per commit)
# Node for actions/checkout and actions/cache, which the bare runner image
# cannot execute. Rust is installed below.
container:
image: catthehacker/ubuntu:act-latest
env:
# Same reasoning as the desktop job in build-and-test.yml: incremental
# state exists to make the *second* build in a working tree fast, which is
# not a thing a fresh checkout has, and it fills the runner's disk.
CARGO_INCREMENTAL: 0
# The fixture, out of the workspace so actions/cache never picks it up.
# A 14 MB synthetic catalog is two seconds to regenerate and would
# otherwise be uploaded and downloaded on every push to save them.
DR_BENCH_DIR: /tmp/darkroom-bench
steps:
- name: Checkout
uses: actions/checkout@v4
# No `git lfs pull` here, deliberately. `dr-bench` depends on the catalog,
# the decoder, the thumbnail store and the encoder, and on nothing that
# reaches `dr-segment` — so the model this repository keeps in LFS is not
# part of this job's dependency graph and fetching it would be a minute
# spent on a file nothing opens.
- name: Cache cargo
uses: actions/cache@v4
with:
path: |
~/.cargo/registry
~/.cargo/git
target
key: bench-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
# Pinned to the workspace rust-version, as every other job here is: a
# floating toolchain turns an unrelated push into a mystery failure, and
# for a benchmark it would turn one into a mystery *regression*.
#
# rust-analyzer is named for the reason build-and-test.yml gives: rustup
# reconciles rust-toolchain.toml on the first cargo call whatever this
# step asks for, so naming it keeps the download inside the step that says
# it is installing things.
- name: Install Rust 1.92.0
run: |
set -e
curl -fsSL https://sh.rustup.rs | sh -s -- \
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
--component rust-analyzer
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
# `-p dr-bench`, not `--workspace`. The whole point of that crate having
# no GPU and no UI dependency is that this job resolves the catalog, the
# decoder and the encoders and stops there — a few minutes rather than the
# release build of Slint and wgpu the desktop job pays for.
#
# Release, and it is not optional: the workspace builds its own crates at
# opt-level = 0 in dev, and every figure this produces is dominated by
# this workspace's own code. A debug run would measure rustc.
- name: Build the suite
run: cargo build --release -p dr-bench
# Exit 1 is a violated budget or a regression past tolerance; exit 2 is
# the harness failing to run at all. Both fail the job, and the report
# above the failure says which.
- name: Measure, and gate
run: cargo run --release -p dr-bench -- check
- name: Disk after
if: always()
run: df -h /workspace 2>/dev/null || df -h .
gpu:
# On demand only — see the header. A runner with a Vulkan device can be
# pointed at this; one without will skip the measurement and say so, which
# is the same posture the rest of this repository's device tests take.
if: github.event_name == 'workflow_dispatch'
runs-on: linux/amd64
name: Frame budget (on demand)
container:
image: catthehacker/ubuntu:act-latest
env:
CARGO_INCREMENTAL: 0
steps:
- name: Checkout
uses: actions/checkout@v4
# `dr-gpu` depends on `dr-segment` for the watershed's pixel passes. Its
# default features are off, so no weights are compiled in — but the fetch
# is cheap insurance and its failure is not fatal. The header of the same
# step in build-and-test.yml explains why the extraheader is stripped
# rather than reused: two Authorization headers is a 400 from Gitea, one
# step after the batch call that had just succeeded.
- name: Fetch the segmentation model
continue-on-error: true
env:
LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
run: |
set -e
git lfs install --local
git config --local --get-regexp '^http\..*extraheader$' \
| cut -d' ' -f1 | sort -u \
| while read -r key; do git config --local --unset-all "$key"; done || true
git config --local lfs.url \
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
git lfs pull --exclude="fixtures/**,docs/manual/media/**"
- name: Cache cargo
uses: actions/cache@v4
with:
path: |
~/.cargo/registry
~/.cargo/git
target
key: bench-gpu-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
- name: Build dependencies
run: |
apt-get update -qq
apt-get install -y -qq pkg-config libfontconfig1-dev libxkbcommon-dev
- name: Install Rust 1.92.0
run: |
set -e
curl -fsSL https://sh.rustup.rs | sh -s -- \
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
--component rust-analyzer
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
# The guard, in release. Its own module documentation is explicit that a
# release run checks strictly more than a dev one: the CPU half of a frame
# is shader-string assembly, which is several times slower unoptimised, so
# it is folded into the assertion only when debug_assertions is off.
#
# With no adapter this prints "skipping: no GPU adapter" and passes. A
# test that cannot run is not evidence either way, and turning that into a
# failure would make the job useless on the runner it usually lands on.
- name: Frame budget (FR-DSP-3)
run: cargo test --release -p dr-gpu --test frame_budget -- --nocapture
# The instrument behind docs/dev/frame-budget.md. It exits non-zero with no
# adapter, which is right for a tool a person runs deliberately and wrong
# for a job that usually has none — hence continue-on-error. Its table is
# in the log for whoever asked for this run; the committed numbers are
# still updated by hand, as that file says.
- name: Frame budget table
continue-on-error: true
run: cargo run --release -p dr-gpu --example frame_budget
-590
View File
@@ -1,590 +0,0 @@
name: Build and test
# Desktop and Android are built on every push, per the v0.1 decision to carry
# both platforms from the first commit. An Android break is then caught the day
# it lands rather than at a porting milestone.
on:
push:
branches: [main, master, develop]
# A release tag builds again and publishes what it built (the `release`
# job at the end). The master push of the same commit has usually filled
# the caches, so the second run is the warm one.
tags: ['v*']
pull_request:
branches: [main, master, develop]
jobs:
# The Android job runs inside an image that this repo builds. Ensure it is in
# the registry before anything tries to pull it — see android-image.yml for
# why this is a job rather than a documented manual step. It is a no-op of a
# few seconds unless docker/android actually changed.
android-image:
uses: ./.gitea/workflows/android-image.yml
desktop:
runs-on: linux/amd64
name: Desktop (Linux)
# actions/checkout and actions/cache are JavaScript actions: the runner
# executes them with Node from inside this container. The bare runner image
# has none, so the job failed at checkout before reaching any build step.
container:
image: catthehacker/ubuntu:act-latest
# This job filled the runner's disk and died mid-link with "No space left
# on device" — LLVM reporting an IO failure on its output stream, which
# reads like a compiler crash and is not one.
#
# `target/debug` was 24 GB against `target/release`'s 2.6 GB: 15 GB of it
# debug info in `debug/deps`, 3.6 GB incremental state. Neither earns its
# space here. Nothing attaches a debugger to a CI run, and incremental
# compilation exists to make the *second* build in a working tree fast,
# which is not a thing a fresh checkout has. Turning both off is the
# standard CI setting rather than a trick.
#
# Measured on this workspace: the same `cargo test --workspace --no-run`
# tree goes from 24 GB to 3.3 GB, `debug/deps` from 15 GB to 2.8 GB.
#
# Backtraces still name functions without debug info; they lose file and
# line numbers. If a test failure ever needs those, drop DEBUG to 1
# (line-tables-only) rather than back to 2.
#
# This is a mitigation, not a fix. If the runner is full of anything other
# than this job's own output, it will still be full afterwards.
env:
CARGO_INCREMENTAL: 0
CARGO_PROFILE_DEV_DEBUG: 0
steps:
- name: Checkout
uses: actions/checkout@v4
# The model, which is in LFS and is not optional.
#
# `models/**/*.onnx` is tracked in LFS (.gitattributes), so a
# plain checkout writes a ~130-byte pointer where 11 MB should be, and
# `dr-segment`'s build script panics by design rather than embedding a
# pointer and failing at inference. That failure reads like a broken build
# instead of a missing fetch, which is how it went unnoticed.
#
# Not `lfs: true` on the checkout above, and no `Authorization` header
# here either. Both install a blanket header for every request to this
# host, and the object download is the one request that already carries
# one: `git lfs pull` asks `/info/lfs/objects/batch` first, and Gitea
# answers with a short-lived `Bearer` JWT scoped to that object. git-lfs
# then sends the JWT *and* the configured header, and two `Authorization`
# headers is a 400 from Gitea — reported as
# LFS: Client error: .../info/lfs/objects/<oid>
# one step after the batch call that had just succeeded, which reads like
# a rejected credential rather than a duplicated one. A lone token header
# is understood fine; it is only the collision that fails.
#
# So: strip the headers and hand the token to git-lfs as an ordinary
# credential instead. It authenticates the batch call and leaves the
# per-object JWT untouched. Gitea authenticates on the password, so the
# username is a placeholder. Nothing later in this job talks to the
# remote, so dropping checkout's header costs us nothing.
#
# `continue-on-error` deliberately: if this cannot authenticate, the build
# below still runs and fails with the build script's own message, which
# names the real problem. A checkout that dies here says nothing.
- name: Fetch the segmentation model
continue-on-error: true
env:
LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
run: |
set -e
git lfs install --local
git config --local --get-regexp '^http\..*extraheader$' \
| cut -d' ' -f1 | sort -u \
| while read -r key; do git config --local --unset-all "$key"; done || true
git config --local lfs.url \
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
# The manual's pictures too: the APK carries the manual, and
# assemble-apk.sh refuses a pointer where a picture should be.
git lfs pull --exclude="fixtures/**"
ls -lR models/
- name: Cache cargo
uses: actions/cache@v4
with:
path: |
~/.cargo/registry
~/.cargo/git
target
key: desktop-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
# Slint and winit need these at build time; the runner image is minimal.
- name: Build dependencies
run: |
apt-get update -qq
apt-get install -y -qq pkg-config libfontconfig1-dev libxkbcommon-dev
# The act image ships Node but no Rust. Pinned to the workspace
# rust-version so CI, the Android image, and local builds agree — a
# floating toolchain turns an unrelated push into a mystery failure.
#
# The component list mirrors rust-toolchain.toml's, rust-analyzer
# included, even though nothing in this job runs it. rustup reconciles
# that file against the installed toolchain on the first cargo call in
# the work tree and fetches whatever is missing — so leaving it out does
# not save the download, it only moves it into the middle of a build
# step where it is nobody's line item. Naming it here keeps every fetch
# inside the step whose name says it is installing things.
- name: Install Rust 1.92.0
run: |
set -e
curl -fsSL https://sh.rustup.rs | sh -s -- \
-y --no-modify-path --profile minimal \
--default-toolchain 1.92.0 --component rustfmt,clippy,rust-analyzer
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
# Free space before and after the expensive steps, so a repeat of the
# disk exhaustion above is one line to diagnose instead of a puzzling
# LLVM error.
- name: Disk before
run: df -h /workspace 2>/dev/null || df -h .
- name: Format
run: cargo fmt --all -- --check
- name: Clippy
run: cargo clippy --workspace --all-targets -- -D warnings
# GPU tests skip themselves where no adapter is present rather than
# failing — CI runners generally have none, and a test that cannot run is
# not evidence either way.
- name: Test
run: cargo test --workspace
- name: Build
run: cargo build --workspace --release
# Only on a release tag: the binary is 150 MB and nothing but the
# release job wants it.
- name: Upload the desktop binary
if: startsWith(github.ref, 'refs/tags/v')
uses: actions/upload-artifact@v3
with:
name: darkroom-desktop-x86_64-linux
path: target/release/darkroom-desktop
if-no-files-found: error
- name: Disk after
if: always()
run: df -h /workspace 2>/dev/null || df -h .
android:
runs-on: linux/amd64
name: Android (aarch64)
# Waits for the image build. Without this the pull races the push and the
# job dies with "manifest unknown" before its first step, which is the
# failure mode this ordering exists to remove.
needs: android-image
container:
image: gitea.tourolle.paris/dtourolle/darkroom-android:latest
steps:
- name: Checkout
uses: actions/checkout@v4
# The model, which is in LFS and is not optional.
#
# `models/**/*.onnx` is tracked in LFS (.gitattributes), so a
# plain checkout writes a ~130-byte pointer where 11 MB should be, and
# `dr-segment`'s build script panics by design rather than embedding a
# pointer and failing at inference. That failure reads like a broken build
# instead of a missing fetch, which is how it went unnoticed.
#
# Not `lfs: true` on the checkout above, and no `Authorization` header
# here either. Both install a blanket header for every request to this
# host, and the object download is the one request that already carries
# one: `git lfs pull` asks `/info/lfs/objects/batch` first, and Gitea
# answers with a short-lived `Bearer` JWT scoped to that object. git-lfs
# then sends the JWT *and* the configured header, and two `Authorization`
# headers is a 400 from Gitea — reported as
# LFS: Client error: .../info/lfs/objects/<oid>
# one step after the batch call that had just succeeded, which reads like
# a rejected credential rather than a duplicated one. A lone token header
# is understood fine; it is only the collision that fails.
#
# So: strip the headers and hand the token to git-lfs as an ordinary
# credential instead. It authenticates the batch call and leaves the
# per-object JWT untouched. Gitea authenticates on the password, so the
# username is a placeholder. Nothing later in this job talks to the
# remote, so dropping checkout's header costs us nothing.
#
# `continue-on-error` deliberately: if this cannot authenticate, the build
# below still runs and fails with the build script's own message, which
# names the real problem. A checkout that dies here says nothing.
- name: Fetch the segmentation model
continue-on-error: true
env:
LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
run: |
set -e
git lfs install --local
git config --local --get-regexp '^http\..*extraheader$' \
| cut -d' ' -f1 | sort -u \
| while read -r key; do git config --local --unset-all "$key"; done || true
git config --local lfs.url \
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
# The manual's pictures too: the APK carries the manual, and
# assemble-apk.sh refuses a pointer where a picture should be.
git lfs pull --exclude="fixtures/**"
ls -lR models/
- name: Cache cargo
uses: actions/cache@v4
with:
path: |
/opt/cargo/registry
target-android
key: android-${{ hashFiles('**/Cargo.lock') }}
# A fast gate on the crates most likely to break the cross-compile, run
# before the expensive part. It is `cargo check`, so it type-checks
# without linking and returns in a fraction of the time the step below
# takes.
#
# Not a statement that only these crates cross-compile — `darkroom-android`
# and the whole UI stack beneath it build for aarch64 too, which is what
# the API-level step below does. This one exists to fail fast and name a
# smaller suspect when it does.
- name: Cross-compile core
env:
CARGO_TARGET_DIR: target-android
run: cargo check -p dr-types -p dr-gpu -p dr-sync --target aarch64-linux-android
# The linker targets MIN_API, not the compile SDK. cargo-ndk otherwise
# defaults to API 21, far below the Vulkan floor this app needs — and the
# mismatch is invisible until a device refuses to install.
#
# Look under the target triple, and fail on a mismatch. Searching the
# whole target dir for the first `*.so` found the host proc-macro
# libraries in target-android/debug/deps instead — x86-64 objects built
# by the runner's gcc, whose .comment section says nothing about Android
# and can never contradict the expected API. The step passed regardless
# of what the linker actually did, which is the one thing it exists to
# rule out.
- name: Verify minimum API level
env:
CARGO_TARGET_DIR: target-android
run: |
set -e
# `darkroom-android`, not a core crate: this step reads the API level
# out of a *linked* object, and only that crate produces one. It is
# the workspace's single `crate-type = ["cdylib"]`; a library crate
# builds an rlib, which is an archive of object files that no linker
# has yet touched and that `file` therefore has nothing to say about.
# Asking for `-p dr-gpu` here could only ever reach the "no aarch64
# .so was produced" branch below, whatever the linker did.
#
# It is also the honest artefact to check: the .so this names is the
# one that ships in the APK, so the API level verified here is the
# API level a device will refuse to install against.
cargo ndk -t arm64-v8a -o target-android/jniLibs \
build -p darkroom-android --release
MIN_API=$(sed -n 's/^ARG MIN_API=\([0-9]*\).*/\1/p' docker/android/Dockerfile)
# Empty on both sides would compare equal and pass, so neither side
# is allowed to be the result of a failed parse.
if [ -z "$MIN_API" ]; then
echo "no ARG MIN_API= in docker/android/Dockerfile"
exit 1
fi
SO=$(find target-android/aarch64-linux-android/release -maxdepth 1 -name '*.so' | head -1)
if [ -z "$SO" ]; then
echo "no aarch64 .so was produced"
exit 1
fi
echo "checking $SO"
# `file` is kept for the log — it names the NDK that built this — but
# the check no longer depends on it.
file "$SO" || true
# The API level is the first word of the `.note.android.ident` ELF
# note, little-endian. Read the note rather than asking `file` for it:
# `file` only prints "for Android 28" when its magic database is new
# enough to decode that note, and this image's is not. The parse then
# produced nothing, `${API:-unknown}` reported "unknown", and every
# push failed here for weeks on a .so that was linked perfectly
# correctly. A note read straight out of the ELF cannot go stale that
# way.
readelf -n "$SO" | sed -n '/android.ident/,+3p'
HEX=$(readelf -n "$SO" 2>/dev/null \
| awk '/description data:/ { print $6 $5 $4 $3; exit }')
if [ -z "$HEX" ]; then
echo "FAIL: no .note.android.ident in $SO — nothing states an API level"
exit 1
fi
API=$(( 0x$HEX ))
if [ "$API" != "$MIN_API" ]; then
echo "FAIL: linked for Android $API, expected $MIN_API"
exit 1
fi
echo "OK: linked for Android $API"
# The APK itself, so a run leaves something installable behind rather
# than only the knowledge that it would have linked. The assembly is
# `docker/android/assemble-apk.sh`, shared with `package.sh` so the file
# a device gets from `package.sh --install` and the file published here
# are built by the same code — see that script's header.
#
# `KEYSTORE` deliberately points at a throwaway directory instead of its
# default under `target-android`: that directory is what `actions/cache`
# restores and saves, and a signing key has no business in a build cache
# or in anything this job uploads. A fresh debug key per run is the right
# trade for an artefact whose purpose is getting the app onto a test
# device; nothing upgrades in place over it, which is the one thing a
# stable key would buy.
- name: Package the APK
env:
CARGO_TARGET_DIR: target-android
# Absent secrets mean a debug signature, which is what a fork or a
# branch build should get. Set all three (see docs/dev/android-signing.md)
# and the same job produces a release-signed APK instead.
ANDROID_KEYSTORE_BASE64: ${{ secrets.ANDROID_KEYSTORE_BASE64 }}
KEYSTORE_PASS: ${{ secrets.ANDROID_KEYSTORE_PASSWORD }}
KEY_PASS: ${{ secrets.ANDROID_KEY_PASSWORD }}
KEY_ALIAS: ${{ secrets.ANDROID_KEY_ALIAS }}
run: |
set -e
KEYDIR="$(mktemp -d)"
chmod 700 "$KEYDIR"
trap 'rm -rf "$KEYDIR"' EXIT
if [ -n "$ANDROID_KEYSTORE_BASE64" ]; then
# The keystore reaches the runner base64-encoded because a secret
# is a string. It is written under a 0700 mktemp directory, never
# into the workspace: `target-android` is what actions/cache saves,
# and the upload step globs the workspace.
printf '%s' "$ANDROID_KEYSTORE_BASE64" | base64 -d > "$KEYDIR/release.keystore"
export KEYSTORE="$KEYDIR/release.keystore"
else
# Not an error. Unset the rest so assemble-apk.sh takes its debug
# path cleanly rather than seeing a half-configured release one.
export KEYSTORE="$KEYDIR/debug.keystore"
unset KEYSTORE_PASS KEY_PASS KEY_ALIAS
fi
REPO="$PWD" TARGET_DIR="$PWD/target-android" \
bash docker/android/assemble-apk.sh
# v3, not v4. v4 is untested against this Gitea and its runner; v3 is
# what JellyTau uploads its APK with on this same runner, so it is the
# version known to work here rather than the version that ought to.
#
# `if-no-files-found: error` because the failure this guards against is
# a green run with an empty artefact list, which reads as success until
# somebody goes looking for the file.
- name: Upload the APK
uses: actions/upload-artifact@v3
with:
name: darkroom-arm64-v8a-apk
path: target-android/apk/darkroom.apk
if-no-files-found: error
windows-image:
uses: ./.gitea/workflows/windows-image.yml
# TRACES: FR-PLAT-WIN-3
# The Windows executable and its installer, cross-built from Linux
# (docs/dev/windows.md §7). No Windows machine anywhere in this job: what it
# can prove is that the binary links, is a Windows executable with no
# MinGW runtime imports, starts under Wine, and that the installer installs
# and uninstalls under Wine. What it cannot prove — a Vulkan device, a
# render, the secret store — is a release step on a real machine (§6).
windows:
runs-on: linux/amd64
name: Windows (x86_64, cross)
needs: windows-image
container:
image: gitea.tourolle.paris/dtourolle/darkroom-windows:latest
env:
CARGO_INCREMENTAL: 0
CARGO_PROFILE_DEV_DEBUG: 0
CARGO_TARGET_DIR: target-windows
# Wine keeps its prefix under $HOME, which the image points at a
# directory that does not exist in a fresh container.
HOME: /tmp/home
steps:
- name: Checkout
uses: actions/checkout@v4
# Same step as the desktop leg: the models are LFS objects and the
# packager refuses pointers.
- name: Fetch the models
env:
LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
run: |
set -e
git lfs install --local
git config --local --get-regexp '^http\..*extraheader$' \
| cut -d' ' -f1 | sort -u \
| while read -r key; do git config --local --unset-all "$key"; done || true
git config --local lfs.url \
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
# The manual's pictures too: the installer carries the manual, and
# package.sh refuses a pointer where a picture should be.
git lfs pull --exclude="fixtures/**"
ls -l models/face models/scene
- name: Cache cargo
uses: actions/cache@v4
with:
path: |
/opt/cargo/registry
target-windows
key: windows-${{ hashFiles('**/Cargo.lock') }}
# The cfg(windows) branches are linted here and nowhere else: the
# desktop leg's clippy never compiles them.
- name: Clippy for the target
run: cargo clippy --release --target x86_64-pc-windows-gnu -p darkroom-desktop -- -D warnings
- name: Build
run: cargo build --release --target x86_64-pc-windows-gnu -p darkroom-desktop
- name: Smoke-test the executable
run: |
set -e
mkdir -p "$HOME"
EXE=target-windows/x86_64-pc-windows-gnu/release/darkroom-desktop.exe
file "$EXE"
file "$EXE" | grep -q 'PE32+' || { echo "FAIL: not a PE32+ executable"; exit 1; }
file "$EXE" | grep -q '(GUI)' || { echo "FAIL: not a GUI-subsystem executable"; exit 1; }
if x86_64-w64-mingw32-objdump -p "$EXE" | grep -iE 'libwinpthread|libgcc|libstdc'; then
echo "FAIL: the executable imports a MinGW runtime DLL"
exit 1
fi
x86_64-w64-mingw32-objdump -p "$EXE" | grep 'DLL Name' | sort -u
wineboot --init >/dev/null 2>&1 || true
OUT=$(wine "$EXE" --version 2>/dev/null)
echo "wine: $OUT"
echo "$OUT" | grep -q '^darkroom-desktop ' || { echo "FAIL: --version did not answer under Wine"; exit 1; }
- name: Package the installer
run: bash docker/windows/package.sh
- name: Smoke-test the installer
run: |
set -e
SETUP=$(ls target-windows/installer/DarkRoom-*-x86_64-setup.exe)
file "$SETUP" | grep -q 'PE32+' || { echo "FAIL: the installer is not 64-bit"; exit 1; }
wine "$SETUP" /S 2>/dev/null
INST=$(echo "$HOME"/.wine/drive_c/users/*/AppData/Local/Programs/DarkRoom)
ls "$INST"
# As many files as package.sh stages: everything but the READMEs in
# the directories it copies. A literal here went stale the first
# time a model was added.
WANT=$(find models/face models/scene models/inpaint -maxdepth 1 -type f ! -name README.md | wc -l)
GOT=$(ls "$INST/models" | wc -l)
[ "$GOT" = "$WANT" ] || { echo "FAIL: expected $WANT model files, installed $GOT"; exit 1; }
# The manual, and every picture it shows, counted the same way.
[ -f "$INST/manual/index.html" ] || { echo "FAIL: no manual installed"; exit 1; }
WANT=$(ls docs/manual/media | wc -l)
GOT=$(ls "$INST/manual/media" | wc -l)
[ "$GOT" = "$WANT" ] || { echo "FAIL: expected $WANT manual pictures, installed $GOT"; exit 1; }
wine reg query 'HKCU\Software\Microsoft\Windows\CurrentVersion\Uninstall\DarkRoom' 2>/dev/null \
| grep -q DisplayVersion || { echo "FAIL: no uninstall registry key"; exit 1; }
wine "$INST/darkroom.exe" --version 2>/dev/null | grep -q '^darkroom-desktop ' \
|| { echo "FAIL: the installed executable does not run"; exit 1; }
wine "$INST/uninstall.exe" /S 2>/dev/null
sleep 3
[ ! -e "$INST" ] || { echo "FAIL: uninstall left $INST behind"; ls -R "$INST"; exit 1; }
echo "OK: installed and uninstalled under Wine"
- name: Upload the installer
uses: actions/upload-artifact@v3
with:
name: darkroom-windows-x86_64-setup
path: target-windows/installer/DarkRoom-*-x86_64-setup.exe
if-no-files-found: error
layering:
runs-on: linux/amd64
name: Layer separation
# Node for the JS actions, as above. cargo comes from rustup below.
container:
image: catthehacker/ubuntu:act-latest
steps:
- name: Checkout
uses: actions/checkout@v4
# `cargo tree` resolves the dependency graph, so it needs the registry
# index but no system libraries — this job builds nothing.
#
# rust-analyzer is named for the reason given in the desktop job: rustup
# installs rust-toolchain.toml's components on the first cargo call
# whether or not this step asks for them, and an unasked-for download is
# the one nobody can find in the log.
- name: Install Rust 1.92.0
run: |
set -e
curl -fsSL https://sh.rustup.rs | sh -s -- \
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
--component rust-analyzer
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
# ARCH §6.5a: no core/ crate may depend on the UI toolkit. One stray
# `use slint::` costs headless golden-image testing and the
# one-operation-two-presentations property together, and nothing else
# would notice.
- name: Core crates must not depend on the UI
run: |
set -e
FAILED=0
for crate in dr-types dr-gpu dr-sync; do
if cargo tree -p "$crate" -e normal 2>/dev/null | grep -qE '\bslint\b|\bi-slint'; then
echo "FAIL: $crate depends on Slint (ARCH §6.5a)"
FAILED=1
else
echo "ok: $crate"
fi
done
exit $FAILED
# A v* tag becomes a Gitea Release carrying the three builds and their
# SHA256SUMS, titled and described by the tag's message. Until this job
# existed every release was made by hand, and most tags never got one.
#
# It needs all three platform jobs, so a tag whose tests fail publishes
# nothing; re-run the failed job and this one follows. The work is
# tools/publish-release.sh, which is also how a release is finished by hand.
release:
if: startsWith(github.ref, 'refs/tags/v')
needs: [desktop, android, windows]
runs-on: linux/amd64
name: Publish the release
container:
image: catthehacker/ubuntu:act-latest
permissions:
contents: write
steps:
- name: Checkout
uses: actions/checkout@v4
- name: Fetch the builds
uses: actions/download-artifact@v3
with:
path: dist
# Named for the download page, with the version in each name the way
# the hand-made releases had them. The installer already carries its
# version from package.sh.
- name: Publish
env:
GITEA_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
TAG: ${{ github.ref_name }}
run: |
set -e
V="${TAG#v}"
ls -lR dist
mkdir -p out
cp dist/darkroom-arm64-v8a-apk/darkroom.apk "out/darkroom-${V}-arm64-v8a.apk"
cp dist/darkroom-desktop-x86_64-linux/darkroom-desktop "out/darkroom-desktop-${V}-x86_64-linux"
chmod +x "out/darkroom-desktop-${V}-x86_64-linux"
cp dist/darkroom-windows-x86_64-setup/DarkRoom-${V}-x86_64-setup.exe out/
bash tools/publish-release.sh "$TAG" out/*
-145
View File
@@ -1,145 +0,0 @@
name: Traceability
# Mirrors JellyTau's traceability gate, including the reason it exists.
#
# That gate divided a traced count by frozen literal denominators while the
# requirements file grew past them, reported 158% coverage, and so could never
# fail its own threshold. Two rules follow, and the extractor's own tests
# enforce both:
#
# 1. Denominators are parsed from docs/dev/requirements.md at run time.
# 2. Coverage is |traced ∩ defined| / |defined|, never a raw traced count.
#
# This job is static analysis of source comments plus markdown parsing, so it
# needs no GPU and no Android SDK — only the Rust toolchain.
on:
push:
branches: [main, master, develop]
pull_request:
branches: [main, master, develop]
jobs:
traceability:
runs-on: linux/amd64
name: Requirement traces
# Node for actions/checkout and actions/cache, which the bare runner image
# cannot execute. Rust is installed below.
container:
image: catthehacker/ubuntu:act-latest
steps:
- name: Checkout
uses: actions/checkout@v4
with:
fetch-depth: 0
- name: Cache cargo
uses: actions/cache@v4
with:
path: |
~/.cargo/registry
~/.cargo/git
target
key: traces-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
# Source-comment and markdown parsing only, so the minimal profile is
# enough — no system libraries and nothing this job itself needs beyond
# cargo. rust-analyzer is here anyway because rust-toolchain.toml lists
# it: rustup installs that file's components on the first cargo call in
# the work tree regardless, and a download named in the install step
# beats the same download appearing unannounced inside the gate.
- name: Install Rust 1.92.0
run: |
set -e
curl -fsSL https://sh.rustup.rs | sh -s -- \
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
--component rust-analyzer
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
# The gate's own arithmetic is the thing being trusted, so its tests run
# before it does. Untested gate logic is exactly how JellyTau's 158% went
# unnoticed for months.
- name: Test the extractor
run: cargo test -p traceability
# Structural failures are unconditional and do not depend on the coverage
# threshold: zero requirements parsed, zero files scanned, a ratio above
# 100%, or any orphan tag all fail the build. A misconfigured run must not
# report a plausible-looking 0%.
# Every picture the manual shows is made by a scene in
# tools/manual/scenes.py, and every picture a scene makes is shown.
# Two files read; no app, no display.
- name: Manual pictures have scenes
run: tools/manual/record.sh --check
- name: Traceability gate
run: cargo run -q -p traceability -- check
- name: Regenerate matrix and check it is committed
run: |
set -e
cargo run -q -p traceability -- report
if ! git diff --quiet docs/dev/traceability.md; then
echo ""
echo "docs/dev/traceability.md is out of date."
echo "Run: cargo run -p traceability -- report"
git diff --stat docs/dev/traceability.md
exit 1
fi
# The gesture vocabulary, from the same scanner and under the same rule.
#
# Blocking, and for a sharper reason than the matrix: these two artefacts
# are not only read, one of them is *shown to the user*. A stale
# `gesture_book.rs` is a help sheet in the application telling somebody to
# perform a gesture that was removed — worse than no help sheet, because
# they will conclude the application is broken rather than the page.
#
# This also fails on a malformed tag, so a typo costs a gesture its
# desktop half loudly rather than silently — and on a key a Slint
# handler binds that no tag names, or a key a tag names that no handler
# binds (tools/traceability/src/keymap.rs).
- name: Regenerate the gesture vocabulary and check it is committed
run: cargo run -q -p traceability -- gestures-check
# The manual's page, which the packages carry and the help sheet links
# into. Blocking for the gesture book's reason: it is shown to the user,
# and a page that disagrees with the README is a manual describing an
# application that no longer exists.
- name: Regenerate the manual page and check it is committed
run: cargo run -q -p traceability -- manual-check
# Advisory, not blocking: not every file implements a requirement, and a
# tag on every function is noise that rots faster than it helps. Tag the
# unit that decides.
- name: Check changed files for tags
if: github.event_name == 'pull_request'
run: |
set -e
CHANGED=$(git diff --name-only "origin/${{ github.base_ref }}...HEAD" \
| grep -E '\.(rs|slint|wgsl)$' || true)
[ -z "$CHANGED" ] && { echo "No source files changed."; exit 0; }
MISSING=0
for file in $CHANGED; do
case "$file" in
*/tests/*|*/test_*|tools/*) continue ;;
esac
[ -f "$file" ] || continue
if ! grep -q 'TRACES:' "$file"; then
echo " no TRACES tag: $file"
MISSING=$((MISSING + 1))
fi
done
if [ "$MISSING" -gt 0 ]; then
echo ""
echo "$MISSING changed file(s) carry no requirement tag."
echo "Format: /// TRACES: FR-CAT-1, FR-CAT-2 | NFR-P1"
echo " (comma separates IDs, pipe groups types)"
fi
- name: Summary
if: always()
run: head -30 docs/dev/traceability.md || true
-170
View File
@@ -1,170 +0,0 @@
name: '🐳 Windows image'
# Builds and pushes gitea.tourolle.paris/dtourolle/darkroom-windows, the job
# container for the Windows leg of build-and-test.yml.
#
# The same shape as android-image.yml, for the same reason that one exists:
# an image that lives only on a developer's laptop is a job that dies at
# `docker pull`. Built from docker/windows, tagged by that directory's tree
# id, skipped when the registry already has it.
#
# Called by build-and-test.yml on every push, and runnable by hand via
# workflow_dispatch. It is cheap when nothing changed — see the guard below.
on:
workflow_call:
inputs:
force:
description: 'Rebuild even if the registry already has this image ("true"/"false")'
type: string
default: 'false'
workflow_dispatch:
inputs:
force:
description: 'Rebuild even if the registry already has this image ("true"/"false")'
type: string
default: 'false'
# Gitea's act_runner mangles boolean workflow inputs passed through an
# expression — they arrive as false regardless of what was sent. Every input
# here is a string compared with == 'true', as in KPN's docker.yaml.
env:
IMAGE: gitea.tourolle.paris/dtourolle/darkroom-windows
jobs:
build:
runs-on: linux/amd64
name: Build and push
# Deliberately NOT in a container: this job needs the host Docker daemon to
# build an image, and the host's cached ~/.docker/config.json to push it.
# That is also why there is no `docker login` step — the runner host was
# authenticated to the registry during setup.
steps:
# The host has no Node, so the JS-based actions/checkout cannot run here.
# A minimal shallow fetch with plain git gets the same tree.
- name: Checkout
run: |
set -e
git init -q .
git remote add origin "${{ github.server_url }}/${{ github.repository }}.git"
git -c http.extraheader="AUTHORIZATION: basic $(printf '%s' '${{ github.actor }}:${{ github.token }}' | base64 -w0)" \
fetch --depth 1 origin "${{ github.sha }}"
git checkout -q FETCH_HEAD
# The image is tagged by the content of docker/windows, not by the commit
# that happened to touch it. `git rev-parse HEAD:<dir>` is the tree object
# id — it changes when and only when a file in that directory changes, so
# an unrelated push reuses the existing image and a Dockerfile edit can
# never silently keep serving a stale `latest`.
#
# Using the commit sha instead would rebuild 2.5 GB on every push; using a
# paths-filter action would need a container that has Node, and the only
# one this repo would reach for is the very image being built.
- name: Resolve image tag
id: tag
run: |
set -e
TREE=$(git rev-parse HEAD:docker/windows)
echo "tree=$TREE" >> "$GITHUB_OUTPUT"
echo "docker/windows tree: $TREE"
# Skip the build when the registry already holds this exact content. This
# is what keeps the job a few seconds long on a normal push, and what
# makes it self-healing: if the tag is missing for any reason, including
# the image having never been pushed at all, it gets built here.
#
# The probe is curl against the registry API, NOT `docker manifest
# inspect`. The latter exits 1 on this registry even for tags that are
# demonstrably present — jellytau-builder:latest answers HTTP 200 to the
# API while `docker manifest inspect` reports "manifest unknown" for it.
# Trusting that would have rebuilt 7 GB on every single push.
#
# A HEAD request also gives the digest for free, which is how the repoint
# decision below is made without pulling any layers.
- name: Query registry
id: check
env:
# The runner's own credentials, so this does not depend on how the
# host's ~/.docker/config.json happens to be set up.
REG_USER: ${{ github.actor }}
REG_PASS: ${{ github.token }}
TREE: ${{ steps.tag.outputs.tree }}
run: |
set -eu
ACCEPT='application/vnd.oci.image.index.v1+json,application/vnd.docker.distribution.manifest.v2+json,application/vnd.oci.image.manifest.v1+json,application/vnd.docker.distribution.manifest.list.v2+json'
API="https://gitea.tourolle.paris/v2/dtourolle/darkroom-windows/manifests"
# Prints "<http-status> <digest-or-empty>" for a tag.
probe() {
curl -sI -u "$REG_USER:$REG_PASS" -H "Accept: $ACCEPT" "$API/$1" \
| tr -d '\r' \
| awk 'BEGIN{s="000";d=""} /^HTTP/{s=$2} tolower($1)=="docker-content-digest:"{d=$2} END{print s, d}'
}
read -r TREE_STATUS TREE_DIGEST <<EOF
$(probe "$TREE")
EOF
read -r LATEST_STATUS LATEST_DIGEST <<EOF
$(probe latest)
EOF
echo "tag $TREE -> HTTP $TREE_STATUS ${TREE_DIGEST:-(no digest)}"
echo "tag latest -> HTTP $LATEST_STATUS ${LATEST_DIGEST:-(no digest)}"
# Build unless the registry definitively confirms this content is
# already there. An auth failure or an unreachable registry lands
# here too, and rebuilding needlessly is the safe direction to fail —
# skipping a build that was needed is what breaks the Windows job.
if [ "${{ inputs.force }}" = "true" ]; then
echo "forced rebuild requested"
echo "build=true" >> "$GITHUB_OUTPUT"
echo "repoint=false" >> "$GITHUB_OUTPUT"
elif [ "$TREE_STATUS" != "200" ]; then
echo "registry does not have this content — building"
echo "build=true" >> "$GITHUB_OUTPUT"
echo "repoint=false" >> "$GITHUB_OUTPUT"
elif [ -n "$TREE_DIGEST" ] && [ "$TREE_DIGEST" = "$LATEST_DIGEST" ]; then
echo "registry is already correct — nothing to do"
echo "build=false" >> "$GITHUB_OUTPUT"
echo "repoint=false" >> "$GITHUB_OUTPUT"
else
echo "content is present but latest points elsewhere — repointing"
echo "build=false" >> "$GITHUB_OUTPUT"
echo "repoint=true" >> "$GITHUB_OUTPUT"
fi
# Context is docker/windows, matching the README's build command. The
# Dockerfile COPYs nothing from the repo, so it needs no wider context —
# and a narrow context keeps the daemon from tarring up the whole tree,
# target/ included.
- name: Build
if: ${{ steps.check.outputs.build == 'true' }}
run: |
set -e
docker build \
-t "$IMAGE:${{ steps.tag.outputs.tree }}" \
-t "$IMAGE:latest" \
docker/windows
# Both tags are pushed: the tree tag is what the guard above looks for on
# the next run, and `latest` is what build-and-test.yml pulls.
- name: Push
if: ${{ steps.check.outputs.build == 'true' }}
run: |
set -e
docker push "$IMAGE:${{ steps.tag.outputs.tree }}"
docker push "$IMAGE:latest"
# A cache hit on the tree tag says nothing about where `latest` points — a
# reverted Dockerfile or a build from another branch can leave it on
# different content. This runs only when the digests above actually
# disagree, so the common case costs nothing; the layers are already in
# the registry, so the push that follows uploads a manifest, not 2.5 GB.
- name: Repoint latest
if: ${{ steps.check.outputs.repoint == 'true' }}
run: |
set -e
docker pull "$IMAGE:${{ steps.tag.outputs.tree }}"
docker tag "$IMAGE:${{ steps.tag.outputs.tree }}" "$IMAGE:latest"
docker push "$IMAGE:latest"
-3
View File
@@ -1,3 +0,0 @@
#!/bin/sh
command -v git-lfs >/dev/null 2>&1 || { printf >&2 "\n%s\n\n" "This repository is configured for Git LFS but 'git-lfs' was not found on your path. If you no longer wish to use Git LFS, remove this hook by deleting the 'post-checkout' file in the hooks directory (set by 'core.hookspath'; usually '.git/hooks')."; exit 2; }
git lfs post-checkout "$@"
-3
View File
@@ -1,3 +0,0 @@
#!/bin/sh
command -v git-lfs >/dev/null 2>&1 || { printf >&2 "\n%s\n\n" "This repository is configured for Git LFS but 'git-lfs' was not found on your path. If you no longer wish to use Git LFS, remove this hook by deleting the 'post-commit' file in the hooks directory (set by 'core.hookspath'; usually '.git/hooks')."; exit 2; }
git lfs post-commit "$@"
-3
View File
@@ -1,3 +0,0 @@
#!/bin/sh
command -v git-lfs >/dev/null 2>&1 || { printf >&2 "\n%s\n\n" "This repository is configured for Git LFS but 'git-lfs' was not found on your path. If you no longer wish to use Git LFS, remove this hook by deleting the 'post-merge' file in the hooks directory (set by 'core.hookspath'; usually '.git/hooks')."; exit 2; }
git lfs post-merge "$@"
-81
View File
@@ -1,81 +0,0 @@
#!/usr/bin/env bash
# Keep the generated artefacts in step with the tags in the tree.
#
# Two of them now, from the same scanner: the requirements matrix and the
# gesture vocabulary. Both are generated *from* the tree and cite line numbers
# in it, so both go stale on any commit that moves a line — a `cargo fmt` sweep
# above all, but equally a commit that merely adds a paragraph above a tag.
#
# The gate regenerates the matrix in CI and fails if the result differs from
# what is committed. That is the right check — a matrix that disagrees with the
# tree is worse than none, because it is read as current — but it fails *after*
# a push, on a commit that is otherwise fine, and it has now done so on six
# commits in a row because adding a `TRACES:` tag and regenerating the matrix
# are two actions and only the first is on anyone's mind.
#
# So it happens here instead, where the tags are being changed.
#
# Only when something that can carry a tag is staged: a commit touching
# workflows, packaging or the matrix itself pays nothing.
set -euo pipefail
staged="$(git diff --cached --name-only --diff-filter=ACMR)"
if ! grep -qE '\.(rs|slint|yaml|md)$' <<< "${staged}"; then
exit 0
fi
# The artefacts are generated from the tree, so regenerating them because one
# was itself edited would be circular.
case "$(tr -d '[:space:]' <<< "${staged}")" in
docs/dev/traceability.md | docs/gestures.md | ui/dr-ui/src/gesture_book.rs | docs/manual/index.html)
exit 0
;;
esac
repo="$(git rev-parse --show-toplevel)"
cd "${repo}"
# Quiet unless it has something to say. A hook that prints on every commit is
# a hook people start passing --no-verify to.
if ! cargo run -q -p traceability -- report >/dev/null 2>&1; then
echo "pre-commit: could not run the traceability report; leaving the matrix alone" >&2
exit 0
fi
if ! git diff --quiet -- docs/dev/traceability.md; then
git add docs/dev/traceability.md
echo "pre-commit: regenerated docs/dev/traceability.md and staged it"
fi
# The gesture vocabulary, same discipline.
#
# **Failure here is reported and not swallowed**, unlike the matrix above. A
# matrix that will not build leaves the previous one in place, which is merely
# stale; a malformed `GESTURE:` block means a gesture the user is about to be
# told about in the wrong words, or not at all. The gate would catch it in CI
# either way — this is only about catching it a push earlier.
if ! out="$(cargo run -q -p traceability -- gestures 2>&1)"; then
echo "pre-commit: the gesture scan failed — the tags below need fixing" >&2
echo "${out}" >&2
exit 1
fi
for f in docs/gestures.md ui/dr-ui/src/gesture_book.rs; do
if ! git diff --quiet -- "${f}"; then
git add "${f}"
echo "pre-commit: regenerated ${f} and staged it"
fi
done
# The manual's page, when its source is part of the commit. Rendered from
# nothing but the README, so there is no reason to pay for it otherwise.
if grep -qx 'docs/manual/README.md' <<< "${staged}"; then
if ! out="$(cargo run -q -p traceability -- manual 2>&1)"; then
echo "pre-commit: the manual would not render" >&2
echo "${out}" >&2
exit 1
fi
if ! git diff --quiet -- docs/manual/index.html; then
git add docs/manual/index.html
echo "pre-commit: regenerated docs/manual/index.html and staged it"
fi
fi
-3
View File
@@ -1,3 +0,0 @@
#!/bin/sh
command -v git-lfs >/dev/null 2>&1 || { printf >&2 "\n%s\n\n" "This repository is configured for Git LFS but 'git-lfs' was not found on your path. If you no longer wish to use Git LFS, remove this hook by deleting the 'pre-push' file in the hooks directory (set by 'core.hookspath'; usually '.git/hooks')."; exit 2; }
git lfs pre-push "$@"
-26
View File
@@ -1,26 +0,0 @@
/target
/target-android
Cargo.lock.bak
*.log
# makepkg build products. `packaging/PKGBUILD` and the .desktop entry are
# sources and belong in the tree; everything makepkg derives from them does
# not — `pkg/` and `src/` are staging directories it recreates on every run,
# and the package itself is 33 MB of compiled output.
/packaging/pkg/
/packaging/src/
/packaging/*.pkg.tar.*
/packaging/*.log
# Cached upstream film profiles, re-fetchable with
# tools/film-profiles/convert.py --fetch. Not source: the converted
# profiles in core/dr-film/profiles are.
tools/film-profiles/upstream/
# flatpak-builder's cache and its output tree. `packaging/flatpak/` holds the
# manifest, which is source; everything a build derives from it is not — and
# `.flatpak-builder/` in particular caches an unpacked copy of the whole
# checkout, so it is larger than the repository it sits in.
/.flatpak-builder/
/build/
__pycache__/
-165
View File
@@ -1,165 +0,0 @@
# Working in this repository
Notes for anyone — person or agent — changing this code. They record what
went wrong once and what the fix looked like, so the same shape is not
written again. Requirements live in `docs/dev/requirements.md`; this file is
about habits, not features.
## Catalog reads: work is proportional to what changed, never to library size
`docs/dev/catalog.md §1` states the rule. These are the ways it was broken on
the Identity screen, found when every confirm click cost half a second on a
24k-image library (2026-09-19), and what each fix looked like.
**A redraw must know what changed.** A click handler that calls "refresh
everything" pays for everything. `identity_ui::refresh` takes a `Changed`:
a confirm re-reads the rail and the grid and *not* the coverage line,
because moving a face between people cannot alter how many images are
indexed. Before adding a read to a shared refresh, ask which events can
change its answer, and gate it on those.
**Count with `COUNT(*)`, never with `.len()` on a list you then drop.**
`repairs::counts` used to build every repair's work list — a `Target` with
its path per row, sorted into visiting order — to report its length. Six
repairs, 350 ms, nothing kept. If the caller wants a number, the query
returns a number.
**One query, not one per row.** `ThumbStore::contains` in a filter over
5,000 rows is 5,000 prepared statements; `ThumbStore::held(size)` reads the
index once into a set. The same applies to any `query_row` inside a loop
over a result set — including `deep_count` per sidebar row, which is fine
at sidebar scale and would not be at grid scale. Aggregate in one
statement and look up in memory.
**SQL text in a loop is a prepare in a loop.** rusqlite's `execute` and
`query_row` compile their statement on every call. A loop that calls them
per row pays a prepare per row even when each query is a primary-key seek:
the merge of a synced catalog prepared four statements for each of 13,000
incoming faces (450 ms of a pass that changed nothing), `persist` four per
photograph a scan listed (1.5 s for a first scan), the shard sync one per
image each way. Hoist the statement, use `prepare_cached`, or — better, when
the loop asks the same table about every row — read that table once into a
map. And do not rewrite a row with what it already holds: an upsert of
identical values still dirties the page.
**A `LIKE` is case-insensitive, and no index here serves that.**
`source_ref LIKE 'stem.%'` read every name of the root per sidecar a pull
took in. When the check that decides is exact, spell the prefix as a range
(`>= 'stem.' AND < 'stem/'`, `/` being the byte after `.`), which the
`(root_id, source_ref)` key answers with a seek.
**`Catalog::open` is not free, and every worker thread calls it.** The
backfill runs on every open, and the develop view opens a catalog to fetch
each original and again for each neighbour it prefetches. Keep each
backfill step's no-op case to a read of the small side — the unpaired
JPEGs, not every RAW; the distinct keywords, not every assignment — and
measure an open with `catalog_bench` after adding one.
**Filter and aggregate in SQL, and aggregate the small side first.**
`faces::people` read 19,000 rows, grouped, sorted them by name, and the
screen threw 17,000 away (empty unnamed groups). `people_in_use` filters in
the `WHERE`, and joins `people` to a pre-aggregated `face_person` (2,000
groups) rather than grouping after a `LEFT JOIN` over every person. The
sort then sees only the rows that will be drawn.
**Wide rows make "just check one column" a table scan.** A `faces` row is
~8 KB (a 1 KB embedding and a ~5 KB crop, then the columns added later).
Any predicate that reads `quality`, `crop` or an eye column for every face
reads every row. V17 learned this for the eye filter; V19 applies it to the
repair counts with partial indexes (`faces_owed_*`) that hold only the rows
still owing, keyed on what the predicate joins on and carrying `model_id`
because the predicate reads it. Two things to know about them:
- **Drive the count from the small side.** SQLite uses a partial index
when the query starts from `faces` (`repairs::count`, `Needs::Face`) and
ignores it inside a correlated `EXISTS (... WHERE f.image_id = i.id ...)`.
That is why `Needs::Face` carries the per-face fragment and spells it two
ways.
- **Spell the predicate as the index's `WHERE` is spelled.** `NEEDS_EYES`
is `(f.eye_right IS NULL OR f.landmarks_dense IS NULL)` because
`faces_owed_eyes` is `WHERE eye_right IS NULL OR landmarks_dense IS NULL`.
Change one, change both, and `counts_are_the_sizes_of_the_lists` will
tell you if they drift.
Check a query's plan with `EXPLAIN QUERY PLAN` against a copy of a real
catalog before trusting an index exists for it: "SEARCH ... USING COVERING
INDEX" is the answer you want, "SEARCH f USING INDEX faces_image" on a wide
table means every probe opens a row.
## Catalog writes: one transaction per user action
`faces::confirm` opens a transaction. Calling it in a loop over a group is
a commit per face; `faces::confirm_all` is two statements and one commit,
`faces::reassign` one transaction for a whole split. When a UI action
touches N rows, give the catalog a function that takes the N, not a loop
that calls the one-row function N times — `unchecked_transaction` cannot
nest, so this has to be designed in at the catalog layer, not wrapped
from above.
## Screens: keep what is already decoded
`identity::load_faces` takes the crops the grid is currently showing and
hands them back into the new cells. Before that, a click re-read 4 MB of
crop blobs and decoded 700 JPEGs to produce the pixels already on screen.
When a redraw replaces a model, the expensive parts of the old model — a
decoded image, a cut portrait — are the first thing to reuse; only the row
that changed needs new work. Drain the old cells rather than cloning them.
## Remote calls: one round trip per file, not one per ancestor
`NextcloudBackend::move_to` guaranteed its destination's parent with a
`MKCOL` per ancestor from the account root, on every file of a batch —
three `405`s before each `MOVE`. The backend now remembers the collections
it has confirmed (`known_dirs`) for its lifetime, which is one job. When a
per-file operation has a per-batch precondition, satisfy it once.
## Providers: read the runtime's source for the version on disk, not the binding
Two things the MIGraphX rung (2026-09-20) got wrong before it was measured
right, both because `ort`'s builder was trusted to mean what its method
names say.
**A binding's option builder may fill a struct the runtime no longer
reads.** `ep::MIGraphX::with_save_model` sets fields of the legacy
`OrtMIGraphXProviderOptions`; ONNX Runtime 1.29 reads that struct for the
precision flags and ignores the rest, so every session compiled for 40 s
and the cache directory went nowhere. The option that works
(`migraphx_model_cache_dir`) exists only in the generic key/value
registration, which `session::migraphx` calls on the API table directly.
Before wiring a provider option, fetch the provider's source at the
runtime's exact version and find where the option is *read*.
**A provider's cache key may leave out what you are varying.** MIGraphX
keys a compiled program on graph, GPU and its own version — not precision.
The first fp16 measurement built in 0.3 s and matched f32 to the tenth of a
millisecond, because it had loaded the f32 program. A "from cache" build
that is suspiciously fast on the first run of a new configuration is a key
collision, not a fast provider; give each precision its own directory (the
engine does) and check the cache directory gained a file.
## Measuring
`cargo run --release -p dr-ui --example identity_bench -- CATALOG THUMBS`
times what one click on the Identity screen reads and what the batch
operations write. Run it against a **copy** of a real catalog (it writes),
never the library's own file; `sqlite3 catalog.sqlite ".backup copy.sqlite"`
takes a consistent one while the app runs. Compare the `cpu` column when
other builds are running on the machine — the wall clock doubles under
load, the CPU figure does not. Keep the binary from before the change and
run both back to back rather than trusting numbers taken an hour apart.
`cargo run --release -p dr-catalog --example catalog_bench -- CATALOG
[FACES_DIR]` does the same for opening the catalog (the backfill step by
step), the upload snapshot, a merge, and the face shard export and import;
`persist_bench`, an ignored test in `dr-ui`'s scan module, replays a scan's
`persist` and a sidecar pull (`DR_BENCH_CATALOG=copy.sqlite cargo test
--release -p dr-ui --lib persist_bench -- --ignored --nocapture
--test-threads=1`). Both take copies; hand `catalog_bench` a copy of the
face store directory too.
Reference figures from the 2026-09-19 fixes, largest person (754 faces),
24k images, 19k faces, before → after. What one click read: `load_people`
22 ms → 12 ms, `load_faces` 316 ms → 2.4 ms, `audit` 190 ms → not run
(66 ms when it is, on open and at the end of a sweep). What one click
wrote: `confirm_all` 16 ms → 2 ms, `split_off` 23 ms → 4.5 ms. A click on
the face grid went from ~530 ms of catalog work to ~15 ms.
-198
View File
@@ -1,198 +0,0 @@
# Contributing to DarkRoom
There is a lot of documentation here — twenty-odd documents and 192 numbered
requirements — and almost all of it is written for someone who has already
decided to work on this. This file is the other thing: how to get a first
change landed without reading any of it.
## The shortest useful contribution
**A develop operation is one file.** Not one file plus a registration, plus a
shader edit, plus a control in the UI — one file:
```
core/dr-pipeline/ops/split_toning.yaml
```
`build.rs` finds it with `read_dir`, compiles it into Rust implementing
`Operation`, and from there it is indistinguishable from a hand-written node.
It arrives with controls built from its declared parameter kinds, a place in
the chain from `order:`, a place in the panel from `attributes:`, sidecar
persistence, and its own tests — which are declared in the same file and run
under `cargo test`.
Read [`core/dr-pipeline/ops/README.md`](core/dr-pipeline/ops/README.md) and
copy [`exposure.yaml`](core/dr-pipeline/ops/exposure.yaml). Split toning,
colour zones, selective colour and channel-mixer variants are all pure point
operations, which means all of them are declarations rather than code.
If you want to understand one thing about the architecture before starting,
make it this: **the core describes its capabilities and the interface composes
them.** No code in `ui/` names an operation, and a test enforces that
(`ui/dr-ui/tests/ui_names_no_operation.rs`). It is why your node needs no UI
change.
## Getting it to build
**Git LFS is required.** Model weights are stored in LFS, and a clone made
without it leaves a ~130-byte text pointer where an 11 MB model should be:
```bash
git lfs install && git lfs pull
```
Forget this and `dr-segment`'s build script stops with an instruction rather
than embedding the pointer and failing at inference time — but it is easier to
run the two commands now.
**The toolchain pins itself.** `rust-toolchain.toml` selects 1.92.0 and rustup
fetches it on first use. Do not override it; `cargo fmt` and `clippy` are both
version-sensitive and CI runs exactly this version.
**System packages.** Slint and winit need these at build time. On Debian or
Ubuntu:
```bash
sudo apt-get install pkg-config libfontconfig1-dev libxkbcommon-dev
```
**Then:**
```bash
cargo run -p darkroom-desktop
```
The first build resolves some 850 crates and takes a while — on a laptop, long
enough to look like a hang. It is not one.
Android is a containerised toolchain and is not needed for most work; see
[`docker/android/README.md`](docker/android/README.md) if you get there.
## What CI will check
All four of these run on every push, so run them before you send anything:
```bash
cargo fmt --all -- --check
cargo clippy --workspace --all-targets -- -D warnings
cargo test --workspace
cargo build --workspace --release
```
GPU tests skip themselves where there is no adapter rather than failing — a
test that cannot run is not evidence either way — so a green run on a machine
without a GPU is expected, and does not mean the GPU paths were exercised.
There is a fifth check, and it is not in that list because you are unlikely to
break it by accident:
```bash
cargo run --release -p dr-bench -- check
```
That is the benchmark suite (`docs/dev/requirements.md` §8), which builds a
synthetic 50,000-image catalog and fails the build if a performance target is
missed or a measurement has drifted past its tolerance. It runs on every push in
its own workflow. [`docs/dev/benchmarks.md`](docs/dev/benchmarks.md) says what it
measures, what it deliberately does not, and how to read a failure. If you have
touched the catalog, the decoder, the thumbnail store or the exporter, run it
before you send.
## Requirements and traceability
[`requirements.md`](docs/dev/requirements.md) is the register of record.
[`traceability.md`](docs/dev/traceability.md) is generated from `TRACES:` tags in
the source and must never be hand-edited:
```rust
// TRACES: FR-DEV-3a | FR-DEV-3c
```
Tags are read from `.rs`, `.slint`, `.wgsl` and `.yaml` — the last so a
declared operation can record the requirement it satisfies, since the Rust it
generates lands in `OUT_DIR` and is not scanned.
A pre-commit hook regenerates the matrix and stages it whenever you touch
something that can carry a tag, so you should not have to think about it. If
you do need to run it by hand:
```bash
cargo run -p traceability -- report
```
Note that it tracks line numbers, so a change that only moves code still moves
the matrix. Never regenerate it with a stale prebuilt binary.
**One convention that the tooling cannot enforce.** A tag proves that a tag
exists, not that the code under it does the thing — `docs/dev/code-health.md`
CH-4 has the details, and two requirements currently read as covered on the
strength of plumbing a future feature would use. So: **close a requirement
with a test that would fail if the behaviour were removed.** Coverage that
moves slowly and means something beats coverage that moves quickly.
**Keys and gestures are held the same way.** A key handler in Slint compares
one canonical chord, `Keys.chord(event) == "Ctrl+Z"`, under a `// KEYMAP:`
comment naming its section of the gesture book, and every key it binds must be
named by a `GESTURE:` block beside it. `cargo run -p traceability -- gestures`
regenerates [`docs/gestures.md`](docs/gestures.md) and the in-app help sheet
from those blocks; `-- gestures-check` fails when a handler binds a key no tag
names, or a tag names a key no handler binds. A `manual:` field in a block
links the gesture to a section of the manual, and a heading that is not there
fails the scan.
**The manual is checked too.** `docs/manual/index.html` is rendered from
`docs/manual/README.md` by `-- manual` and `-- manual-check` fails when they
differ; `tools/manual/record.sh --check` fails when the manual shows a picture
no scene in `tools/manual/scenes.py` makes. If you change what a pictured
screen looks like, [`tools/manual`](tools/manual/README.md) says how to record
it again. The pre-commit hook regenerates the matrix, the gesture book and the
page; CI runs all three checks.
## Two invariants the build defends
Worth knowing before you trip one, because both failures name a requirement
rather than a line:
- **No operation may be named in `ui/`** (FR-DEV-3a). Special-casing one
operation in the panel to fix a layout problem is how a generated interface
stops being generated. If a node needs presentation the panel cannot give it,
the answer is a `presentation:` hint in the declaration and a `WidgetKind`,
not a branch in `develop.rs`.
- **The operation schema rejects ambiguity at build time**: a duplicate
`order:`, a filename disagreeing with its `id:`, a default outside its own
range, an expression naming something that is not a parameter. Each error
names the key you got wrong and exits rather than panicking.
## Commit messages
Imperative subject describing the change from the reader's side — "Offer the
merge when two people turn out to share a name", not "fix: merge dialog". No
conventional-commits prefixes.
The body is where the reasoning goes, and it is expected to be substantial when
the change is. This codebase records *why* far more than most, in commits and
in comments alike, and that is the single habit most worth adopting: the
constraint you worked around is invisible to whoever reads the diff next.
One commit per change. If you fixed two things, that is two commits.
## Where to read next, in order
| Document | Read it when |
|---|---|
| [`core/dr-pipeline/ops/README.md`](core/dr-pipeline/ops/README.md) | Adding or changing a develop operation — start here regardless |
| [`docs/dev/architecture.md`](docs/dev/architecture.md) | Anything touching the render path, catalog or sync |
| [`docs/dev/code-health.md`](docs/dev/code-health.md) | Deciding what to work on; grades each seam by what it costs |
| [`docs/dev/benchmarks.md`](docs/dev/benchmarks.md) | A change that could plausibly cost time or memory |
| [`docs/dev/technical-debt.md`](docs/dev/technical-debt.md) | Something looks wrong — check it was not chosen |
| [`docs/dev/distribution.md`](docs/dev/distribution.md) | Packaging a build, or adding a permission to one |
| [`docs/dev/requirements.md`](docs/dev/requirements.md) | Reference, not reading |
`technical-debt.md` is the one to check before "fixing" anything surprising.
It records compromises that were deliberate, each with the reasoning and a
falsifiable condition for when it stops being one — the point being that you
can tell a constraint from an accident without asking.
## Licence
GPL-3.0-or-later. By contributing you agree your work is licensed the same way.
Generated
-9017
View File
File diff suppressed because it is too large Load Diff
-280
View File
@@ -1,280 +0,0 @@
[workspace]
resolver = "2"
members = [
"core/dr-types",
"core/dr-catalog",
"core/dr-thumbs",
"core/dr-decode",
"core/dr-export",
"core/dr-face",
"core/dr-film",
"core/dr-inference-engine",
"core/dr-ingest",
"core/dr-gpu",
"core/dr-lens",
"core/dr-pano",
"core/dr-pipeline",
"core/dr-preset-xmp",
"core/dr-segment",
"core/dr-sync",
"core/dr-sync-folder",
"core/dr-sync-nextcloud",
"core/dr-xmp",
"platform/dr-plat",
"ui/dr-ui",
"apps/darkroom-desktop",
"apps/darkroom-android",
"tools/bench",
"tools/traceability",
]
# Patched copies of upstream crates, not our code: see third_party/README.md.
# Excluded so `--workspace` does not test, lint or format them as ours.
exclude = ["third_party"]
[workspace.package]
version = "0.18.0"
edition = "2021"
rust-version = "1.92"
license = "GPL-3.0-or-later"
repository = "https://github.com/dtourolle/DarkRoom"
[workspace.dependencies]
# Internal
dr-types = { path = "core/dr-types" }
dr-catalog = { path = "core/dr-catalog" }
dr-thumbs = { path = "core/dr-thumbs" }
dr-decode = { path = "core/dr-decode" }
dr-export = { path = "core/dr-export" }
# Stated explicitly for the same reason as `dr-segment` below: no dependant
# should drag in an ONNX runtime by accident. Members opt in with
# `features = ["inference"]`.
dr-face = { path = "core/dr-face", default-features = false }
dr-film = { path = "core/dr-film" }
# `tract` on by default so a test binary can open a session with nothing
# installed; the apps add `native` to look for a runtime file (docs/dev/inference.md §3).
dr-inference-engine = { path = "core/dr-inference-engine" }
dr-ingest = { path = "core/dr-ingest" }
dr-gpu = { path = "core/dr-gpu" }
dr-lens = { path = "core/dr-lens" }
# Optional runtime, like `dr-segment`: the geometry never needs a model.
dr-pano = { path = "core/dr-pano", default-features = false }
dr-pipeline = { path = "core/dr-pipeline" }
dr-preset-xmp = { path = "core/dr-preset-xmp" }
# `default-features = false` belongs *here*, not on each dependant: a member
# inheriting a workspace dependency cannot turn its default features off, so
# writing it below would silently do nothing and every crate touching
# `dr-segment` would drag in tract and 11 MB of weights. Members opt in with
# `features = ["semantic", "embedded-model"]` instead.
dr-segment = { path = "core/dr-segment", default-features = false }
dr-plat = { path = "platform/dr-plat" }
dr-sync = { path = "core/dr-sync" }
dr-sync-folder = { path = "core/dr-sync-folder" }
dr-sync-nextcloud = { path = "core/dr-sync-nextcloud" }
dr-xmp = { path = "core/dr-xmp" }
dr-ui = { path = "ui/dr-ui" }
# GPU + UI
#
# The wgpu version is not a free choice: it is dictated by Slint. Importing a
# texture into the scene (ARCH §6.1, spike S1) requires it to come from the
# *same* `wgpu::Device` Slint renders with, and Slint will only hand out a
# device of the version it was compiled against. Slint 1.17 offers
# `unstable-wgpu-28` and `unstable-wgpu-29` and nothing older, so 29 it is —
# pinned to the same `29.0.4` floor Slint itself requires, because two
# semver-compatible-but-different wgpu crates in one tree are two *types*, and
# the device would not typecheck across them.
#
# Consequently: bumping Slint may force a wgpu bump, and wgpu cannot be bumped
# on its own. They move together or not at all.
wgpu = "29.0.4"
slint = { version = "1.17", default-features = false }
slint-build = "1.17"
# UI token codegen (S2): style.yaml -> theme.slint. serde_yaml was deprecated
# by its maintainer in 2024 and serde_yml, the first fork, has since been
# deprecated too; serde_norway is the fork still receiving releases. Its
# mappings preserve insertion order, which is what lets the generated Slint
# keep the token ordering the YAML author chose.
serde_norway = "0.9"
# Foundations
anyhow = "1"
thiserror = "2"
log = "0.4"
env_logger = "0.11"
pollster = "0.4"
# Networking — no mature Nextcloud crate exists; the connector is hand-rolled
# over reqwest (D7). reqwest_dav was evaluated and is too thin to build on.
# `rustls-no-provider` rather than `rustls`: the latter defaults to the
# aws-lc-rs crypto provider, whose aws-lc-sys crate is C and fails to
# cross-compile for Android — precisely the NDK pain D1 chose Rust to avoid.
# ring is pure Rust apart from a small asm core that does build under the NDK.
#
# `rustls-tls-webpki-roots-no-provider` rather than plain `rustls-no-provider`:
# the latter verifies against rustls-platform-verifier, which reaches the
# Android trust store over JNI and panics mid-handshake unless initialised from
# Java first — the crash D7 predicted and spike S3 exists to resolve properly.
# The panic surfaces inside tokio, which catches task panics itself, so it
# reaches the UI as a worker that stopped rather than as an error.
#
# webpki-roots is the escape hatch D7 records: a root store compiled into the
# binary, no JNI, identical on both platforms. The trade is real and belongs in
# S3's scope — user-installed and enterprise CAs are not consulted, and the
# roots go stale with the release rather than with the OS.
reqwest = { version = "0.13", default-features = false, features = ["rustls-no-provider", "webpki-roots", "stream", "json"] }
rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] }
quick-xml = "0.41"
tokio = { version = "1", features = ["rt-multi-thread", "macros", "sync", "time"] }
url = "2.5"
async-trait = "0.1"
serde = { version = "1", features = ["derive"] }
serde_json = "1"
# The manual's HTML rendering (tools/traceability). Already in the tree as
# Slint's Markdown parser, so this adds a dependency edge and no crate; only
# the HTML writer is needed, not the command-line front end.
pulldown-cmark = { version = "0.13", default-features = false, features = ["html"] }
base64 = "0.23"
# Display-server clients, for FR-DSP-8's per-display profile acquisition.
#
# Neither is a new cost: winit already builds both, so the versions are the
# ones Slint's backend has resolved to and pinning anything else here would
# compile a second copy. Both are pure Rust — x11rb speaks the X11 wire
# protocol itself rather than binding libxcb, and wayland-client binds
# libwayland only under a feature that is off — which keeps the Android
# cross-compile a plain Rust dependency graph, the same criterion as the TLS
# and SQLite choices above. They are declared under a target predicate that
# excludes Android, where neither display server exists.
#
# `staging` on wayland-protocols is what carries `wp_color_manager_v1`: the
# colour-management extension is still staging upstream, which is the
# protocol-level statement of the thing FR-DSP-8 anticipates when it says
# Wayland's colour management "is not universally available".
x11rb = { version = "0.13", features = ["randr"] }
wayland-client = "0.31"
wayland-protocols = { version = "0.32", features = ["client", "staging"] }
# Platform secure storage: Secret Service on Linux, Keystore on Android
# (FR-NC-2). Credentials never touch the catalog or a plain file.
# keyring 4 restructured its features: `v1` is the default set and brings
# the zbus Secret Service backend, which is what GNOME Keyring and KWallet
# (via ksecretd) both speak.
keyring = { version = "4", features = ["v1"] }
# The Android half of the same project: a keyring-core CredentialStore backed
# by AndroidKeyStore AES-GCM over SharedPreferences (FR-PLAT-AND-1). It reads
# the JavaVM and Context from ndk-context, which android-activity populates
# before `android_main` runs, so no Kotlin shim of our own is needed.
#
# This is the keyring-core API, not the v1 `Entry` API the Linux path uses;
# the two impls are deliberately separate rather than sharing a code path.
android-native-keyring-store = "1.0.0"
keyring-core = "1"
# Decode. rawler is the pure-Rust decoder (D2); zune-jpeg decodes the
# embedded previews rawler extracts.
# Catalog. `bundled` compiles SQLite from source rather than linking the
# system library — the same cross-compilation reasoning as the TLS choice
# above: no system dependency to satisfy under the Android NDK.
#
# `backup` is not optional in practice: it is what takes a consistent snapshot
# of a live WAL database for upload. A filesystem copy of `catalog.sqlite`
# while a `-wal` exists beside it uploads a torn file.
rusqlite = { version = "0.40", features = ["bundled", "backup"] }
rawler = "0.7"
zune-jpeg = "0.4.21"
# Thumbnails are stored encoded, not as raw RGBA: a 256px RGBA buffer is
# ~256 KB against ~20 KB as JPEG, and the store syncs to Nextcloud where that
# 13× is transfer cost on every client. Pure Rust, no C dependency — the same
# criterion behind the TLS and SQLite choices above.
jpeg-encoder = "0.7"
bytemuck = { version = "1", features = ["derive"] }
# Lens correction profiles. A pure-Rust port of Lensfun rather than a binding
# to the C library, for the same cross-compilation reason as the TLS and
# SQLite choices above: liblensfun would be a third C dependency to satisfy
# under the Android NDK.
#
# The database ships *inside* the crate — 56 XML files, gzipped at build time
# and decompressed on first lookup. That matters beyond convenience: Android
# gives us no filesystem path (ARCH §6.9), so a database loaded from a
# system directory would have nowhere to live there.
#
# Licence: LGPL-3.0-or-later, which upgrades cleanly into our GPLv3 (D8).
# The upstream Lensfun *database* is CC-BY-SA and is redistributed by the
# crate; attribution belongs in the about screen.
#
# Caveat worth remembering: this is a third-party port at 0.7.0, not upstream
# Lensfun. Verified working against the bundled database (interpolation
# between calibration points, and an unknown lens returning empty rather than
# panicking), but the pipeline talks to it through its own profile types so
# swapping it out is not a pipeline change.
lensfun = "0.7"
# Neural inference for semantic segmentation (S15 arm B, D14).
#
# D13 framed this as a choice between `ort` (fast, best operator coverage, and
# a C++ dependency to cross-compile under the NDK) and a pure-Rust runtime
# (policy-compliant, unproven coverage). That framing turned out to be a false
# choice: `ort` 2.0's `alternative-backend` feature *disables the linking
# entirely* and lets a different engine supply the `OrtApi`, and `ort-tract` —
# same authors, MIT/Apache — supplies it from `tract`, which is pure Rust.
#
# So we get `ort`'s API with no C at all. `download-binaries` and `tls-native`
# are off with `default-features = false`, which is the point: nothing is
# fetched at build time and nothing is linked, so the Android cross-compile
# sees an ordinary Rust dependency graph. That is the same reasoning as rustls
# over aws-lc-rs and bundled SQLite, applied to inference — D13's largest
# tolerated exception turns out not to be needed.
#
# The trade is real and belongs on the record: tract is slower than the C++
# runtime and covers fewer operators. Both were measured rather than assumed
# before this landed — yolo26n-seg loads with **zero unsupported operators**
# and runs 640x640 in ~470 ms on the reference desktop's CPU. That is fine for
# a once-per-image precompute off the frame path (ARCH §6.1) and would not be
# fine for anything per-frame, which is a constraint on what may be built on
# top rather than on this choice.
#
# Pinned to an rc: `ort` 2.0 has been in rc for a long while and `ort-tract`
# exists only against it. Worth revisiting at 2.0 final.
ort = { version = "2.0.0-rc.13", default-features = false, features = ["alternative-backend", "ndarray", "std"] }
ort-tract = "0.4"
# Not a free choice: it is the version `ort` exposes its tensors through, so
# two semver-incompatible ndarrays would not typecheck across the boundary —
# the same coupling wgpu has with Slint above.
ndarray = "0.17"
[profile.dev]
# Dev builds are tuned for how fast they *compile*, not for how fast they run.
# Optimisation is a release concern; `[profile.release]` below is where it
# belongs.
#
# This deliberately reverses an earlier choice. Dependencies used to be built
# at `opt-level = 2` here, because wgpu and image decoding are slow without it.
# That is still true, and it is the price: a debug run of the app, and the
# decode- and GPU-heavy tests, are slower than they were. What it buys is that
# nothing has to be optimised before it can be compiled — which is the cost
# paid on every edit, by every worktree, rather than only when something is
# actually run.
#
# If a particular crate turns out to be the one that makes a test unbearable,
# raise it alone rather than restoring the blanket rule:
#
# [profile.dev.package.zune-jpeg]
# opt-level = 2
opt-level = 0
[profile.release]
lto = "thin"
codegen-units = 1
# Two upstream crates carry a local patch so that the Android build can draw
# with wgpu on a rotated display (technical-debt.md TD-1). Both are exact
# copies of the version the lockfile already resolves, plus that patch;
# third_party/README.md says what was changed and how to carry it forward
# when Slint or wgpu moves.
[patch.crates-io]
wgpu-hal = { path = "third_party/wgpu-hal-29.0.4" }
i-slint-renderer-skia = { path = "third_party/i-slint-renderer-skia-1.17.1" }
-232
View File
@@ -1,232 +0,0 @@
GNU GENERAL PUBLIC LICENSE
Version 3, 29 June 2007
Copyright © 2007 Free Software Foundation, Inc. <https://fsf.org/>
Everyone is permitted to copy and distribute verbatim copies of this license document, but changing it is not allowed.
Preamble
The GNU General Public License is a free, copyleft license for software and other kinds of works.
The licenses for most software and other practical works are designed to take away your freedom to share and change the works. By contrast, the GNU General Public License is intended to guarantee your freedom to share and change all versions of a program--to make sure it remains free software for all its users. We, the Free Software Foundation, use the GNU General Public License for most of our software; it applies also to any other work released this way by its authors. You can apply it to your programs, too.
When we speak of free software, we are referring to freedom, not price. Our General Public Licenses are designed to make sure that you have the freedom to distribute copies of free software (and charge for them if you wish), that you receive source code or can get it if you want it, that you can change the software or use pieces of it in new free programs, and that you know you can do these things.
To protect your rights, we need to prevent others from denying you these rights or asking you to surrender the rights. Therefore, you have certain responsibilities if you distribute copies of the software, or if you modify it: responsibilities to respect the freedom of others.
For example, if you distribute copies of such a program, whether gratis or for a fee, you must pass on to the recipients the same freedoms that you received. You must make sure that they, too, receive or can get the source code. And you must show them these terms so they know their rights.
Developers that use the GNU GPL protect your rights with two steps: (1) assert copyright on the software, and (2) offer you this License giving you legal permission to copy, distribute and/or modify it.
For the developers' and authors' protection, the GPL clearly explains that there is no warranty for this free software. For both users' and authors' sake, the GPL requires that modified versions be marked as changed, so that their problems will not be attributed erroneously to authors of previous versions.
Some devices are designed to deny users access to install or run modified versions of the software inside them, although the manufacturer can do so. This is fundamentally incompatible with the aim of protecting users' freedom to change the software. The systematic pattern of such abuse occurs in the area of products for individuals to use, which is precisely where it is most unacceptable. Therefore, we have designed this version of the GPL to prohibit the practice for those products. If such problems arise substantially in other domains, we stand ready to extend this provision to those domains in future versions of the GPL, as needed to protect the freedom of users.
Finally, every program is threatened constantly by software patents. States should not allow patents to restrict development and use of software on general-purpose computers, but in those that do, we wish to avoid the special danger that patents applied to a free program could make it effectively proprietary. To prevent this, the GPL assures that patents cannot be used to render the program non-free.
The precise terms and conditions for copying, distribution and modification follow.
TERMS AND CONDITIONS
0. Definitions.
“This License” refers to version 3 of the GNU General Public License.
“Copyright” also means copyright-like laws that apply to other kinds of works, such as semiconductor masks.
“The Program” refers to any copyrightable work licensed under this License. Each licensee is addressed as “you”. “Licensees” and “recipients” may be individuals or organizations.
To “modify” a work means to copy from or adapt all or part of the work in a fashion requiring copyright permission, other than the making of an exact copy. The resulting work is called a “modified version” of the earlier work or a work “based on” the earlier work.
A “covered work” means either the unmodified Program or a work based on the Program.
To “propagate” a work means to do anything with it that, without permission, would make you directly or secondarily liable for infringement under applicable copyright law, except executing it on a computer or modifying a private copy. Propagation includes copying, distribution (with or without modification), making available to the public, and in some countries other activities as well.
To “convey” a work means any kind of propagation that enables other parties to make or receive copies. Mere interaction with a user through a computer network, with no transfer of a copy, is not conveying.
An interactive user interface displays “Appropriate Legal Notices” to the extent that it includes a convenient and prominently visible feature that (1) displays an appropriate copyright notice, and (2) tells the user that there is no warranty for the work (except to the extent that warranties are provided), that licensees may convey the work under this License, and how to view a copy of this License. If the interface presents a list of user commands or options, such as a menu, a prominent item in the list meets this criterion.
1. Source Code.
The “source code” for a work means the preferred form of the work for making modifications to it. “Object code” means any non-source form of a work.
A “Standard Interface” means an interface that either is an official standard defined by a recognized standards body, or, in the case of interfaces specified for a particular programming language, one that is widely used among developers working in that language.
The “System Libraries” of an executable work include anything, other than the work as a whole, that (a) is included in the normal form of packaging a Major Component, but which is not part of that Major Component, and (b) serves only to enable use of the work with that Major Component, or to implement a Standard Interface for which an implementation is available to the public in source code form. A “Major Component”, in this context, means a major essential component (kernel, window system, and so on) of the specific operating system (if any) on which the executable work runs, or a compiler used to produce the work, or an object code interpreter used to run it.
The “Corresponding Source” for a work in object code form means all the source code needed to generate, install, and (for an executable work) run the object code and to modify the work, including scripts to control those activities. However, it does not include the work's System Libraries, or general-purpose tools or generally available free programs which are used unmodified in performing those activities but which are not part of the work. For example, Corresponding Source includes interface definition files associated with source files for the work, and the source code for shared libraries and dynamically linked subprograms that the work is specifically designed to require, such as by intimate data communication or control flow between those subprograms and other parts of the work.
The Corresponding Source need not include anything that users can regenerate automatically from other parts of the Corresponding Source.
The Corresponding Source for a work in source code form is that same work.
2. Basic Permissions.
All rights granted under this License are granted for the term of copyright on the Program, and are irrevocable provided the stated conditions are met. This License explicitly affirms your unlimited permission to run the unmodified Program. The output from running a covered work is covered by this License only if the output, given its content, constitutes a covered work. This License acknowledges your rights of fair use or other equivalent, as provided by copyright law.
You may make, run and propagate covered works that you do not convey, without conditions so long as your license otherwise remains in force. You may convey covered works to others for the sole purpose of having them make modifications exclusively for you, or provide you with facilities for running those works, provided that you comply with the terms of this License in conveying all material for which you do not control copyright. Those thus making or running the covered works for you must do so exclusively on your behalf, under your direction and control, on terms that prohibit them from making any copies of your copyrighted material outside their relationship with you.
Conveying under any other circumstances is permitted solely under the conditions stated below. Sublicensing is not allowed; section 10 makes it unnecessary.
3. Protecting Users' Legal Rights From Anti-Circumvention Law.
No covered work shall be deemed part of an effective technological measure under any applicable law fulfilling obligations under article 11 of the WIPO copyright treaty adopted on 20 December 1996, or similar laws prohibiting or restricting circumvention of such measures.
When you convey a covered work, you waive any legal power to forbid circumvention of technological measures to the extent such circumvention is effected by exercising rights under this License with respect to the covered work, and you disclaim any intention to limit operation or modification of the work as a means of enforcing, against the work's users, your or third parties' legal rights to forbid circumvention of technological measures.
4. Conveying Verbatim Copies.
You may convey verbatim copies of the Program's source code as you receive it, in any medium, provided that you conspicuously and appropriately publish on each copy an appropriate copyright notice; keep intact all notices stating that this License and any non-permissive terms added in accord with section 7 apply to the code; keep intact all notices of the absence of any warranty; and give all recipients a copy of this License along with the Program.
You may charge any price or no price for each copy that you convey, and you may offer support or warranty protection for a fee.
5. Conveying Modified Source Versions.
You may convey a work based on the Program, or the modifications to produce it from the Program, in the form of source code under the terms of section 4, provided that you also meet all of these conditions:
a) The work must carry prominent notices stating that you modified it, and giving a relevant date.
b) The work must carry prominent notices stating that it is released under this License and any conditions added under section 7. This requirement modifies the requirement in section 4 to “keep intact all notices”.
c) You must license the entire work, as a whole, under this License to anyone who comes into possession of a copy. This License will therefore apply, along with any applicable section 7 additional terms, to the whole of the work, and all its parts, regardless of how they are packaged. This License gives no permission to license the work in any other way, but it does not invalidate such permission if you have separately received it.
d) If the work has interactive user interfaces, each must display Appropriate Legal Notices; however, if the Program has interactive interfaces that do not display Appropriate Legal Notices, your work need not make them do so.
A compilation of a covered work with other separate and independent works, which are not by their nature extensions of the covered work, and which are not combined with it such as to form a larger program, in or on a volume of a storage or distribution medium, is called an “aggregate” if the compilation and its resulting copyright are not used to limit the access or legal rights of the compilation's users beyond what the individual works permit. Inclusion of a covered work in an aggregate does not cause this License to apply to the other parts of the aggregate.
6. Conveying Non-Source Forms.
You may convey a covered work in object code form under the terms of sections 4 and 5, provided that you also convey the machine-readable Corresponding Source under the terms of this License, in one of these ways:
a) Convey the object code in, or embodied in, a physical product (including a physical distribution medium), accompanied by the Corresponding Source fixed on a durable physical medium customarily used for software interchange.
b) Convey the object code in, or embodied in, a physical product (including a physical distribution medium), accompanied by a written offer, valid for at least three years and valid for as long as you offer spare parts or customer support for that product model, to give anyone who possesses the object code either (1) a copy of the Corresponding Source for all the software in the product that is covered by this License, on a durable physical medium customarily used for software interchange, for a price no more than your reasonable cost of physically performing this conveying of source, or (2) access to copy the Corresponding Source from a network server at no charge.
c) Convey individual copies of the object code with a copy of the written offer to provide the Corresponding Source. This alternative is allowed only occasionally and noncommercially, and only if you received the object code with such an offer, in accord with subsection 6b.
d) Convey the object code by offering access from a designated place (gratis or for a charge), and offer equivalent access to the Corresponding Source in the same way through the same place at no further charge. You need not require recipients to copy the Corresponding Source along with the object code. If the place to copy the object code is a network server, the Corresponding Source may be on a different server (operated by you or a third party) that supports equivalent copying facilities, provided you maintain clear directions next to the object code saying where to find the Corresponding Source. Regardless of what server hosts the Corresponding Source, you remain obligated to ensure that it is available for as long as needed to satisfy these requirements.
e) Convey the object code using peer-to-peer transmission, provided you inform other peers where the object code and Corresponding Source of the work are being offered to the general public at no charge under subsection 6d.
A separable portion of the object code, whose source code is excluded from the Corresponding Source as a System Library, need not be included in conveying the object code work.
A “User Product” is either (1) a “consumer product”, which means any tangible personal property which is normally used for personal, family, or household purposes, or (2) anything designed or sold for incorporation into a dwelling. In determining whether a product is a consumer product, doubtful cases shall be resolved in favor of coverage. For a particular product received by a particular user, “normally used” refers to a typical or common use of that class of product, regardless of the status of the particular user or of the way in which the particular user actually uses, or expects or is expected to use, the product. A product is a consumer product regardless of whether the product has substantial commercial, industrial or non-consumer uses, unless such uses represent the only significant mode of use of the product.
“Installation Information” for a User Product means any methods, procedures, authorization keys, or other information required to install and execute modified versions of a covered work in that User Product from a modified version of its Corresponding Source. The information must suffice to ensure that the continued functioning of the modified object code is in no case prevented or interfered with solely because modification has been made.
If you convey an object code work under this section in, or with, or specifically for use in, a User Product, and the conveying occurs as part of a transaction in which the right of possession and use of the User Product is transferred to the recipient in perpetuity or for a fixed term (regardless of how the transaction is characterized), the Corresponding Source conveyed under this section must be accompanied by the Installation Information. But this requirement does not apply if neither you nor any third party retains the ability to install modified object code on the User Product (for example, the work has been installed in ROM).
The requirement to provide Installation Information does not include a requirement to continue to provide support service, warranty, or updates for a work that has been modified or installed by the recipient, or for the User Product in which it has been modified or installed. Access to a network may be denied when the modification itself materially and adversely affects the operation of the network or violates the rules and protocols for communication across the network.
Corresponding Source conveyed, and Installation Information provided, in accord with this section must be in a format that is publicly documented (and with an implementation available to the public in source code form), and must require no special password or key for unpacking, reading or copying.
7. Additional Terms.
“Additional permissions” are terms that supplement the terms of this License by making exceptions from one or more of its conditions. Additional permissions that are applicable to the entire Program shall be treated as though they were included in this License, to the extent that they are valid under applicable law. If additional permissions apply only to part of the Program, that part may be used separately under those permissions, but the entire Program remains governed by this License without regard to the additional permissions.
When you convey a copy of a covered work, you may at your option remove any additional permissions from that copy, or from any part of it. (Additional permissions may be written to require their own removal in certain cases when you modify the work.) You may place additional permissions on material, added by you to a covered work, for which you have or can give appropriate copyright permission.
Notwithstanding any other provision of this License, for material you add to a covered work, you may (if authorized by the copyright holders of that material) supplement the terms of this License with terms:
a) Disclaiming warranty or limiting liability differently from the terms of sections 15 and 16 of this License; or
b) Requiring preservation of specified reasonable legal notices or author attributions in that material or in the Appropriate Legal Notices displayed by works containing it; or
c) Prohibiting misrepresentation of the origin of that material, or requiring that modified versions of such material be marked in reasonable ways as different from the original version; or
d) Limiting the use for publicity purposes of names of licensors or authors of the material; or
e) Declining to grant rights under trademark law for use of some trade names, trademarks, or service marks; or
f) Requiring indemnification of licensors and authors of that material by anyone who conveys the material (or modified versions of it) with contractual assumptions of liability to the recipient, for any liability that these contractual assumptions directly impose on those licensors and authors.
All other non-permissive additional terms are considered “further restrictions” within the meaning of section 10. If the Program as you received it, or any part of it, contains a notice stating that it is governed by this License along with a term that is a further restriction, you may remove that term. If a license document contains a further restriction but permits relicensing or conveying under this License, you may add to a covered work material governed by the terms of that license document, provided that the further restriction does not survive such relicensing or conveying.
If you add terms to a covered work in accord with this section, you must place, in the relevant source files, a statement of the additional terms that apply to those files, or a notice indicating where to find the applicable terms.
Additional terms, permissive or non-permissive, may be stated in the form of a separately written license, or stated as exceptions; the above requirements apply either way.
8. Termination.
You may not propagate or modify a covered work except as expressly provided under this License. Any attempt otherwise to propagate or modify it is void, and will automatically terminate your rights under this License (including any patent licenses granted under the third paragraph of section 11).
However, if you cease all violation of this License, then your license from a particular copyright holder is reinstated (a) provisionally, unless and until the copyright holder explicitly and finally terminates your license, and (b) permanently, if the copyright holder fails to notify you of the violation by some reasonable means prior to 60 days after the cessation.
Moreover, your license from a particular copyright holder is reinstated permanently if the copyright holder notifies you of the violation by some reasonable means, this is the first time you have received notice of violation of this License (for any work) from that copyright holder, and you cure the violation prior to 30 days after your receipt of the notice.
Termination of your rights under this section does not terminate the licenses of parties who have received copies or rights from you under this License. If your rights have been terminated and not permanently reinstated, you do not qualify to receive new licenses for the same material under section 10.
9. Acceptance Not Required for Having Copies.
You are not required to accept this License in order to receive or run a copy of the Program. Ancillary propagation of a covered work occurring solely as a consequence of using peer-to-peer transmission to receive a copy likewise does not require acceptance. However, nothing other than this License grants you permission to propagate or modify any covered work. These actions infringe copyright if you do not accept this License. Therefore, by modifying or propagating a covered work, you indicate your acceptance of this License to do so.
10. Automatic Licensing of Downstream Recipients.
Each time you convey a covered work, the recipient automatically receives a license from the original licensors, to run, modify and propagate that work, subject to this License. You are not responsible for enforcing compliance by third parties with this License.
An “entity transaction” is a transaction transferring control of an organization, or substantially all assets of one, or subdividing an organization, or merging organizations. If propagation of a covered work results from an entity transaction, each party to that transaction who receives a copy of the work also receives whatever licenses to the work the party's predecessor in interest had or could give under the previous paragraph, plus a right to possession of the Corresponding Source of the work from the predecessor in interest, if the predecessor has it or can get it with reasonable efforts.
You may not impose any further restrictions on the exercise of the rights granted or affirmed under this License. For example, you may not impose a license fee, royalty, or other charge for exercise of rights granted under this License, and you may not initiate litigation (including a cross-claim or counterclaim in a lawsuit) alleging that any patent claim is infringed by making, using, selling, offering for sale, or importing the Program or any portion of it.
11. Patents.
A “contributor” is a copyright holder who authorizes use under this License of the Program or a work on which the Program is based. The work thus licensed is called the contributor's “contributor version”.
A contributor's “essential patent claims” are all patent claims owned or controlled by the contributor, whether already acquired or hereafter acquired, that would be infringed by some manner, permitted by this License, of making, using, or selling its contributor version, but do not include claims that would be infringed only as a consequence of further modification of the contributor version. For purposes of this definition, “control” includes the right to grant patent sublicenses in a manner consistent with the requirements of this License.
Each contributor grants you a non-exclusive, worldwide, royalty-free patent license under the contributor's essential patent claims, to make, use, sell, offer for sale, import and otherwise run, modify and propagate the contents of its contributor version.
In the following three paragraphs, a “patent license” is any express agreement or commitment, however denominated, not to enforce a patent (such as an express permission to practice a patent or covenant not to sue for patent infringement). To “grant” such a patent license to a party means to make such an agreement or commitment not to enforce a patent against the party.
If you convey a covered work, knowingly relying on a patent license, and the Corresponding Source of the work is not available for anyone to copy, free of charge and under the terms of this License, through a publicly available network server or other readily accessible means, then you must either (1) cause the Corresponding Source to be so available, or (2) arrange to deprive yourself of the benefit of the patent license for this particular work, or (3) arrange, in a manner consistent with the requirements of this License, to extend the patent license to downstream recipients. “Knowingly relying” means you have actual knowledge that, but for the patent license, your conveying the covered work in a country, or your recipient's use of the covered work in a country, would infringe one or more identifiable patents in that country that you have reason to believe are valid.
If, pursuant to or in connection with a single transaction or arrangement, you convey, or propagate by procuring conveyance of, a covered work, and grant a patent license to some of the parties receiving the covered work authorizing them to use, propagate, modify or convey a specific copy of the covered work, then the patent license you grant is automatically extended to all recipients of the covered work and works based on it.
A patent license is “discriminatory” if it does not include within the scope of its coverage, prohibits the exercise of, or is conditioned on the non-exercise of one or more of the rights that are specifically granted under this License. You may not convey a covered work if you are a party to an arrangement with a third party that is in the business of distributing software, under which you make payment to the third party based on the extent of your activity of conveying the work, and under which the third party grants, to any of the parties who would receive the covered work from you, a discriminatory patent license (a) in connection with copies of the covered work conveyed by you (or copies made from those copies), or (b) primarily for and in connection with specific products or compilations that contain the covered work, unless you entered into that arrangement, or that patent license was granted, prior to 28 March 2007.
Nothing in this License shall be construed as excluding or limiting any implied license or other defenses to infringement that may otherwise be available to you under applicable patent law.
12. No Surrender of Others' Freedom.
If conditions are imposed on you (whether by court order, agreement or otherwise) that contradict the conditions of this License, they do not excuse you from the conditions of this License. If you cannot convey a covered work so as to satisfy simultaneously your obligations under this License and any other pertinent obligations, then as a consequence you may not convey it at all. For example, if you agree to terms that obligate you to collect a royalty for further conveying from those to whom you convey the Program, the only way you could satisfy both those terms and this License would be to refrain entirely from conveying the Program.
13. Use with the GNU Affero General Public License.
Notwithstanding any other provision of this License, you have permission to link or combine any covered work with a work licensed under version 3 of the GNU Affero General Public License into a single combined work, and to convey the resulting work. The terms of this License will continue to apply to the part which is the covered work, but the special requirements of the GNU Affero General Public License, section 13, concerning interaction through a network will apply to the combination as such.
14. Revised Versions of this License.
The Free Software Foundation may publish revised and/or new versions of the GNU General Public License from time to time. Such new versions will be similar in spirit to the present version, but may differ in detail to address new problems or concerns.
Each version is given a distinguishing version number. If the Program specifies that a certain numbered version of the GNU General Public License “or any later version” applies to it, you have the option of following the terms and conditions either of that numbered version or of any later version published by the Free Software Foundation. If the Program does not specify a version number of the GNU General Public License, you may choose any version ever published by the Free Software Foundation.
If the Program specifies that a proxy can decide which future versions of the GNU General Public License can be used, that proxy's public statement of acceptance of a version permanently authorizes you to choose that version for the Program.
Later license versions may give you additional or different permissions. However, no additional obligations are imposed on any author or copyright holder as a result of your choosing to follow a later version.
15. Disclaimer of Warranty.
THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM “AS IS” WITHOUT WARRANTY OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF ALL NECESSARY SERVICING, REPAIR OR CORRECTION.
16. Limitation of Liability.
IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS), EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH DAMAGES.
17. Interpretation of Sections 15 and 16.
If the disclaimer of warranty and limitation of liability provided above cannot be given local legal effect according to their terms, reviewing courts shall apply local law that most closely approximates an absolute waiver of all civil liability in connection with the Program, unless a warranty or assumption of liability accompanies a copy of the Program in return for a fee.
END OF TERMS AND CONDITIONS
How to Apply These Terms to Your New Programs
If you develop a new program, and you want it to be of the greatest possible use to the public, the best way to achieve this is to make it free software which everyone can redistribute and change under these terms.
To do so, attach the following notices to the program. It is safest to attach them to the start of each source file to most effectively state the exclusion of warranty; and each file should have at least the “copyright” line and a pointer to where the full notice is found.
<one line to give the program's name and a brief idea of what it does.>
Copyright (C) <year> <name of author>
This program is free software: you can redistribute it and/or modify it under the terms of the GNU General Public License as published by the Free Software Foundation, either version 3 of the License, or (at your option) any later version.
This program is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for more details.
You should have received a copy of the GNU General Public License along with this program. If not, see <https://www.gnu.org/licenses/>.
Also add information on how to contact you by electronic and paper mail.
If the program does terminal interaction, make it output a short notice like this when it starts in an interactive mode:
<program> Copyright (C) <year> <name of author>
This program comes with ABSOLUTELY NO WARRANTY; for details type `show w'.
This is free software, and you are welcome to redistribute it under certain conditions; type `show c' for details.
The hypothetical commands `show w' and `show c' should show the appropriate parts of the General Public License. Of course, your program's commands might be different; for a GUI interface, you would use an “about box”.
You should also get your employer (if you work as a programmer) or school, if any, to sign a “copyright disclaimer” for the program, if necessary. For more information on this, and how to apply and follow the GNU GPL, see <https://www.gnu.org/licenses/>.
The GNU General Public License does not permit incorporating your program into proprietary programs. If your program is a subroutine library, you may consider it more useful to permit linking proprietary applications with the library. If this is what you want to do, use the GNU Lesser General Public License instead of this License. But first, please read <https://www.gnu.org/philosophy/why-not-lgpl.html>.
-146
View File
@@ -1,146 +0,0 @@
# DarkRoom
A non-destructive RAW photo editor and library for Linux and Android, with a
GPU develop pipeline, a catalog that syncs between devices, and no account,
no telemetry and no cloud of its own.
[![The library: seventy frames, the timeline beside them, the filter bar above](docs/manual/media/library.png)](docs/manual/README.md)
**[The manual](docs/manual/README.md)** shows every feature, pictured from
the application itself. This page says what it is, how to get it, and what
is still missing.
## What it does
**A library.** Point it at a folder — on this machine, on a network mount,
or one a Nextcloud client keeps in virtual-files mode, where a placeholder
is treated as the photograph rather than as a one-byte file — or at a
Nextcloud account directly; a photograph that is only on the server opens on
its thumbnail with the download's progress over it. The grid is virtualised,
ordered by capture time with a timeline beside it, and filtered by rating,
flag, colour label, person and whether the file is here. Ratings, colour
labels, keywords, collections and a trash that survives a crash
mid-operation. Card ingest. Bursts fold. The same RAW catalogued twice — a
dated folder and a backup beside it — is found, proved the same, and folded
onto one copy with the spares in the trash. Face detection and identity,
with the index syncing between devices.
**Developing.** Eighteen declared operations fused into one compute
dispatch, plus the neighbourhood work that cannot be: clarity, texture,
capture sharpening, noise reduction, lens correction, spectral film
simulation. Crop, straighten and correct converging verticals, spot repair,
and local adjustments over masks the model draws — click a subject or a
category, then paint, subtract a gradient or keep only where two selections
agree, grow or shrink the edge. Focus peaking and a raw histogram for judging
what is recoverable. Presets, with a collection shipped in the application —
everyday corrections, and a look for each measured colour, cinema and
black-and-white stock — and Lightroom presets imported as looks that leave a
photograph's own corrections alone. XMP sidecars other editors read.
[![Segmenting an urban scene and choosing the sky as a mask](docs/manual/media/local-segment.png)](docs/manual/README.md#local-adjustments)
**Panoramas.** Select the frames, align, choose a projection, fill the
ragged border rather than crop it, and the composite lands beside its
sources as a DNG, with a sidecar recording what it was merged from.
[![Twelve hand-held frames aligned on a cylinder](docs/manual/media/panorama-aligned.png)](docs/manual/README.md#merging-a-panorama)
**From the keyboard, and with its manual.** Rating, flagging and labelling
have keys in the grid and in develop, as do zoom, undo and stepping through a
shoot in develop, and none of them is keyboard-only. The help sheet (`F1`, or
`?` in develop) lists every key and gesture, generated from the code that
binds it, and links them to the sections of the manual that show them — the
manual ships with the application and opens offline.
**Export.** JPEG, PNG, AVIF, JPEG XL, 8- and 16-bit TIFF, with resize, output
sharpening, a naming template and a colour space — into albums: named export
folders on this machine or on the server, never inside the library, which
remember the photograph behind each file and sync between devices as
collections do.
**On both platforms.** The same core runs on a desktop and a 12-inch
tablet; the interface is one layout, tuned for a wide viewport with touch
targets throughout. On both, the develop view draws the compute pass's
texture directly — no readback between the GPU and the screen.
## Getting it
| Platform | How | State |
|---|---|---|
| Arch Linux | [`packaging/PKGBUILD`](packaging/PKGBUILD) — `makepkg -si` | Built from every release |
| Android | The APK from each CI run, or `./docker/android/package.sh --install` | Runs on a tablet; F-Droid not yet submitted |
| Windows | `DarkRoom-<version>-x86_64-setup.exe`, cross-built by CI ([windows.md](docs/dev/windows.md)) | Verified under Wine only; unsigned |
| Flatpak | [`packaging/flatpak/`](packaging/flatpak/) | Manifest in tree; folders are chosen through the portal, but no Flatpak has been built to prove it |
Or build it. Git LFS is required for the model weights, and the toolchain
pins itself to 1.92.0:
```bash
git lfs install && git lfs pull
cargo run --release -p darkroom-desktop
```
Android, through the containerised toolchain ([docker/android](docker/android/README.md)):
```bash
./docker/android/build.sh cargo ndk -t arm64-v8a build --release
```
[CONTRIBUTING.md](CONTRIBUTING.md) has the system packages, the four
commands CI runs against what you send, and the shortest useful
contribution — a develop operation is one YAML file, and it arrives with its
controls, its place in the chain and its tests.
## Where it stands
**0.18.0**, twenty-six tagged releases in. 192 numbered requirements in
scope, 84% of them claimed by code and [traced to it](docs/dev/traceability.md);
the rest are written down rather than merely absent.
**Not built:** plugins (post-v1, [D12](docs/dev/requirements.md)), compare and
survey culling, AI denoise, tiled rendering, HDR merge and
focus stacking, importing a Lightroom or darktable catalog, translations
beyond the launch screen, most of the Android platform integration beyond
running, and a Flatpak actually built and run in its sandbox. The
performance targets are half verified: the per-commit benchmark suite §8
requires exists for everything that does not need a frame — the catalog,
the scan, the thumbnails — and not yet for the render path, so a regression
there fails nothing.
[outstanding.md](docs/dev/outstanding.md) is the list, with the reasoning for
each.
## Documentation
[docs/README.md](docs/README.md) is the index. The short version, for someone using it:
| | |
|---|---|
| [manual](docs/manual/README.md) | Every feature, pictured |
| [gestures.md](docs/gestures.md) | How it is driven — generated from the code, so it cannot describe a gesture that does not exist |
For someone changing it:
| | |
|---|---|
| [CONTRIBUTING.md](CONTRIBUTING.md) | How to land a first change without reading the rest |
| [requirements.md](docs/dev/requirements.md) | What the software must do — the numbered register, and the decisions |
| [architecture.md](docs/dev/architecture.md) | How it is built — crates, the GPU pipeline, the data model, sync |
| [technical-debt.md](docs/dev/technical-debt.md) | Compromises taken deliberately, each with the condition that retires it |
| [outstanding.md](docs/dev/outstanding.md) | What is not built, and whether that is a decision or a gap |
| [code-health.md](docs/dev/code-health.md) | What a contribution costs, per seam, measured |
| [traceability.md](docs/dev/traceability.md) | Generated: which requirement is claimed by which file |
Designs, one per subsystem:
[segmentation](docs/dev/segmentation.md) and [mask editing](docs/dev/mask-editing.md) ·
[spot removal](docs/dev/spot-removal.md) · [panorama](docs/dev/panorama.md) ·
[faces](docs/dev/faces.md) · [inference](docs/dev/inference.md) ·
[storage and sync](docs/dev/storage.md) · [catalog](docs/dev/catalog.md) ·
[display and extension](docs/dev/display-and-extension.md) ·
[navigation](docs/dev/ui-navigation.md) · [distribution](docs/dev/distribution.md) ·
[windows](docs/dev/windows.md) · [benchmarks](docs/dev/benchmarks.md).
## Licence
GPL-3.0-or-later. The photographs in the manual and the test fixtures are
the author's and are there to show and test this project, nothing else.
The model weights carry their own licences — [models/LICENCE.md](models/LICENCE.md).
-54
View File
@@ -1,54 +0,0 @@
[package]
name = "darkroom-android"
version.workspace = true
edition.workspace = true
rust-version.workspace = true
license.workspace = true
# A cdylib, not a bin: Android loads the app as a shared library and calls
# `android_main` through android-activity's glue. Nothing execs a binary, so
# there is no `main` to provide.
[lib]
name = "darkroom"
crate-type = ["cdylib"]
[dependencies]
# No backend feature to select: dr-ui picks its Slint backend from the target,
# so building for aarch64-linux-android gets android-activity automatically.
dr-ui.workspace = true
# For `account::set_data_dir`: only the platform entry point knows where Android
# lets this app keep files, and it must be set before any store is opened.
dr-sync.workspace = true
# For the panic hook, for `state::set_state_dir` and for
# `diagnostics::install`. Android has no XDG directories, so the entry point is
# the only place that knows where a crash record or a log file may be written,
# and both have to be in place before anything can fail.
dr-plat.workspace = true
# Directly, not just through dr-ui: `android_main` takes an `AndroidApp` and
# calls `slint::android::init`, both of which come from this crate. The backend
# feature comes from dr-ui's target-specific dependency.
slint.workspace = true
log.workspace = true
android_logger = "0.15"
# The launch Intent and the share sheet are Java-only surfaces — see `intents`
# — and JNI is the only way to reach them.
#
# Target-gated because the crate still has to compile on the host: it is a
# workspace member, `cargo test --workspace` builds it, and the manifest tests
# in `lib.rs` are the one part of it that runs there.
#
# 0.21 rather than the 0.22 that android-activity 0.6 uses. Both are already in
# the lock — Slint's Android backend depends on two major versions of
# android-activity and pulls both — so this adds nothing to the build either
# way, and every object here comes from a raw pointer rather than from a type
# android-activity handed over, so the two never have to agree.
[target.'cfg(target_os = "android")'.dependencies]
jni = "0.21"
[features]
# Mirrors darkroom-desktop: the CPU readback path is gone since S1 landed
# zero-copy. It mattered more here than on desktop — the same wrong path with
# far less memory bandwidth to absorb it (ARCH §6.1) — but it is untested on a
# device, since S1 was verified on desktop only.
default = []
@@ -1,200 +0,0 @@
<?xml version="1.0" encoding="utf-8"?>
<!--
DarkRoom Android manifest.
Deliberately minimal: this packages the viewer for on-device testing (spike
S2 needs Adreno and Mali hardware, which no emulator represents). Nothing
here is a distribution manifest yet. Only network access is declared: file
access needs no manifest permission because the library grid reads through
SAF, which grants per-tree at runtime (ARCH §6.9).
Minimal is not the same as empty, and the entries below that are not the
activity are the difference. A manifest is the only place a component can be
declared: an intent filter is how the system learns this app is worth
offering for a photograph, and a provider is how it learns the class exists
at all. Neither can be moved into code (FR-PLAT-AND-6).
-->
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
package="paris.tourolle.darkroom">
<!-- Everything the app does with a server needs this: Login Flow v2, the
WebDAV listing, thumbnail and image fetches. Without it Android refuses
socket creation outright, and the failure is invisible — no panic to
catch, no log line, just a worker thread that stops. Storage is the
separate case that genuinely needs no permission here, because SAF
grants per-tree at runtime (ARCH §6.9). -->
<uses-permission android:name="android.permission.INTERNET" />
<!-- Read before deciding whether a sync may run: FR-NC-6 gates background
work on unmetered-and-charging, which means knowing the network type. -->
<uses-permission android:name="android.permission.ACCESS_NETWORK_STATE" />
<!-- Vulkan 1.1 is what wgpu needs; the API 28 floor is where support is
dependable (NFR-COMPAT-1). Marked required so an unsupported device
fails at install rather than at first frame. -->
<uses-feature
android:name="android.hardware.vulkan.version"
android:version="0x00401000"
android:required="true" />
<!-- One name covers both icon generations, which is the point of the
`anydpi-v26` qualifier: @mipmap/ic_launcher resolves to the adaptive
icon at res/mipmap-anydpi-v26/ic_launcher.xml on API 26 and up, and to
the density-matched ic_launcher.png below that. Since minSdk is 28 the
PNGs are only ever reached by tooling, but they cost little and aapt2
wants a real drawable behind the name. `roundIcon` is deliberately
absent: it predates adaptive icons and a launcher that reads it would
also be one that ignores the XML, which no device here is.
The adaptive icon has three layers rather than two. The third,
monochrome, is what lets Android 13's themed-icon setting recolour it
instead of dropping the app out of the themed set. -->
<application
android:label="DarkRoom"
android:icon="@mipmap/ic_launcher"
android:hasCode="true"
android:allowBackup="false"
android:supportsRtl="true">
<!-- NativeActivity rather than a Kotlin Activity: android-activity's
glue loads libdarkroom.so and calls android_main. `android.app.lib_name`
is how it learns which library to load, and must match [lib].name.
`singleTask` because a second instance of this activity is not
survivable. The intent filters below mean another app can now
launch it while it is already running, and under the default
launch mode that starts a *second* NativeActivity — in the
caller's task, in this same process, calling android_main again.
Two Slint backends and two wgpu devices in one process is not a
degraded experience, it is a failed second launch on top of a
working first one.
What it costs, stated plainly: a share that arrives while DarkRoom
is already running brings it forward without opening the image.
The Intent goes to `onNewIntent`, and android-activity's event
stream has no variant for it (MainEvent in 0.6 stops at Destroy),
so nothing native ever sees it. Reading it would mean a Java
Activity subclass forwarding it across JNI — the same shape of
change FR-PLAT-AND-5 declined for onTrimMemory, and for the same
reason. Launched from cold, which is the ordinary case for "open
this photograph", the Intent is on getIntent() and is read. -->
<activity
android:name="android.app.NativeActivity"
android:exported="true"
android:launchMode="singleTask"
android:configChanges="orientation|keyboardHidden|screenSize|screenLayout|density|uiMode"
android:windowSoftInputMode="adjustResize">
<meta-data
android:name="android.app.lib_name"
android:value="darkroom" />
<intent-filter>
<action android:name="android.intent.action.MAIN" />
<category android:name="android.intent.category.LAUNCHER" />
</intent-filter>
<!-- FR-PLAT-AND-6, inbound. The traceability tool reads .rs,
.slint, .wgsl and .yaml, so this is a reference and not a
tag; the tag that counts is on the test in lib.rs that
asserts these declarations are still here.
Opening a photograph from a gallery, a file manager or a
download. `android_main` reads the launch Intent through
`Intents.receive` and the named images become the browsing
list, exactly as paths on the desktop command line do.
`image/*` and not a wider match, even though it misses raws:
a provider that does not recognise `.CR3` reports it as
`application/octet-stream`, and claiming that type would put
DarkRoom in the chooser for every unidentified binary on the
device — an APK, a database, a partial download. Being absent
from one gallery's menu is a smaller failure than being
present in all of them. DNG, which providers do know as
`image/x-adobe-dng`, matches here already.
BROWSABLE is what lets a browser's finished download and a
link hand the file over; without it those routes silently do
not list the app. -->
<intent-filter>
<action android:name="android.intent.action.VIEW" />
<category android:name="android.intent.category.DEFAULT" />
<category android:name="android.intent.category.BROWSABLE" />
<data android:mimeType="image/*" />
</intent-filter>
<!-- The share sheet, one photograph or a selection of them.
SEND_MULTIPLE is declared because the sheet offers this app
for a multi-selection only if it says it accepts one, and a
culling tool that can be sent a single frame and not a burst
is the wrong way round.
ACTION_EDIT is deliberately not here. It is a promise to write
the result back to the URI it was handed, and nothing in this
app does: an edit lands in a sidecar beside the original
(FR-CAT-8). Registering for it would put DarkRoom in the "edit
with" menu and lose the user's work every time. -->
<intent-filter>
<action android:name="android.intent.action.SEND" />
<action android:name="android.intent.action.SEND_MULTIPLE" />
<category android:name="android.intent.category.DEFAULT" />
<data android:mimeType="image/*" />
</intent-filter>
</activity>
<!-- The manual (dr_ui::manual): a WebView over the copy the APK
carries in assets/manual. See ManualActivity.java for why it is
not the browser.
Not exported: nothing outside this app has a reason to start it,
and dr_ui starts it by class name, which needs no intent filter.
Its own task entry is not wanted either — it is a page over the
app, and Back returns to the photograph it was opened from.
configChanges so a rotation reflows the page rather than
reloading it at the top. -->
<activity
android:name="paris.tourolle.darkroom.ManualActivity"
android:exported="false"
android:label="DarkRoom manual"
android:theme="@style/ManualTheme"
android:configChanges="orientation|keyboardHidden|screenSize|screenLayout|uiMode" />
<!-- FR-EXP-10: the system's folder picker, for an album's folder on
this device. NativeActivity's onActivityResult is not ours, so
this activity exists only to ask and hand the answer back (see
FolderPicker.java). Translucent and without a title so nothing
of it shows but the system chooser; not exported, and started by
class name from dr_ui::saf. -->
<activity
android:name="paris.tourolle.darkroom.FolderPicker"
android:exported="false"
android:theme="@android:style/Theme.Translucent.NoTitleBar"
android:configChanges="orientation|keyboardHidden|screenSize|screenLayout|uiMode" />
<!-- FR-PLAT-AND-6, outbound. Android has refused file:// URIs
between apps since API 24 — handing one out raises
FileUriExposedException in *this* process — so an exported JPEG
reaches the share sheet as a content:// URI or not at all.
Not AndroidX's FileProvider: that is a Maven artefact, and this
build has no Gradle and no dependency resolver (docker/android/
README.md). ExportProvider does the same hundred lines against one
fixed root.
`exported="false"` with `grantUriPermissions="true"` is the whole
security model, and the two halves are not redundant. Exported
false means no app may address the provider on its own account;
the grant flag means a URI this app puts in an Intent carries a
read permission for that one file, for the lifetime of the
receiving task. Without the grant flag the share sheet opens and
every target fails with SecurityException; with `exported="true"`
instead, every app on the device could read the app's private
directory. The authority must equal ExportProvider.AUTHORITY — a
mismatch is a SecurityException in somebody else's app, so a test
in lib.rs compares the two strings. -->
<provider
android:name="paris.tourolle.darkroom.ExportProvider"
android:authorities="paris.tourolle.darkroom.exports"
android:exported="false"
android:grantUriPermissions="true" />
</application>
</manifest>
@@ -1,260 +0,0 @@
package paris.tourolle.darkroom;
import android.content.ContentProvider;
import android.content.ContentValues;
import android.content.Context;
import android.database.Cursor;
import android.database.MatrixCursor;
import android.net.Uri;
import android.os.ParcelFileDescriptor;
import android.provider.OpenableColumns;
import android.util.Log;
import android.webkit.MimeTypeMap;
import java.io.File;
import java.io.FileNotFoundException;
import java.io.IOException;
import java.util.List;
import java.util.Locale;
/**
* Hands an exported file to another app, and hands out nothing else.
*
* <p>FR-PLAT-AND-6's outbound half. Android has refused {@code file://} URIs
* between apps since API 24 — passing one raises {@code FileUriExposedException}
* in the *sending* process — so the only way to give a photo to the share sheet
* is a {@code content://} URI backed by a provider, plus a per-Intent read
* grant that expires with the task that received it.
*
* <h2>Why this is not AndroidX's FileProvider</h2>
*
* <p>Because AndroidX is a Maven artefact and this build has no Gradle and no
* dependency resolver (see docker/android/README.md). Pulling in the one class
* would mean adopting the whole mechanism that fetches it. What
* {@code FileProvider} does is a hundred lines — map a request path onto a
* directory, refuse anything outside it, answer the two columns the share sheet
* reads — and those lines are below. The configuration it takes as an XML
* {@code <meta-data>} resource is a constant here instead, because there is
* exactly one directory worth serving and a second place to state it is a
* second place for it to be wrong.
*
* <h2>The one directory</h2>
*
* <p>{@code getFilesDir()}, which is the same directory the Rust side calls
* {@code internal_data_path} and passes to {@code dr_sync::account::set_data_dir}
* — {@code ANativeActivity.internalDataPath} and {@code Context.getFilesDir()}
* are the same path. Everything the app writes for itself, the export outbox
* included, is under it. Nothing else is reachable: a request is resolved
* against the real filesystem with {@link File#getCanonicalFile()} and then
* checked to be *inside* that root, so {@code ../} and a symlink planted in the
* outbox are refused by the same test. Serving a path the caller composed,
* unchecked, would turn a share button into a reader for every file this app
* can see, which on Android includes credentials and the whole catalog.
*
* <p>{@code android:exported="false"} in the manifest is the outer half of the
* same rule: no app can address this provider at all except through a URI this
* app handed it with a read grant attached.
*/
public final class ExportProvider extends ContentProvider {
private static final String TAG = "DarkRoom";
/**
* Must equal {@code android:authorities} in AndroidManifest.xml.
*
* <p>A mismatch is not a build error and not a runtime error here: it is a
* {@code SecurityException} in whichever app opened the share sheet, naming
* an authority that does not exist. A test in {@code lib.rs} asserts the
* two strings are the same for that reason.
*/
public static final String AUTHORITY = "paris.tourolle.darkroom.exports";
/** Nothing to set up; the root is resolved per request against the context. */
@Override
public boolean onCreate() {
return true;
}
/**
* The {@code content://} URI for a file, or null if it is not one this
* provider may serve.
*
* <p>Returning null rather than an unusable URI keeps the refusal at the
* point where the path is known. A URI for a file outside the root would be
* rejected later by {@link #openFile}, in the *receiving* app's stack trace,
* where nothing says which of our files was asked for.
*/
public static Uri uriFor(Context context, File file) {
try {
File root = root(context);
File target = file.getCanonicalFile();
String relative = within(root, target);
if (relative == null) {
Log.w(TAG, "not shareable, outside " + root + ": " + target);
return null;
}
// Built segment by segment rather than with a composed path
// string: appendPath percent-encodes, and getPathSegments below
// decodes symmetrically. A file called "Rue d'Alésia.jpg" survives
// the round trip only because both halves agree.
Uri.Builder builder = new Uri.Builder().scheme("content").authority(AUTHORITY);
for (String segment : relative.split("/")) {
if (!segment.isEmpty()) {
builder.appendPath(segment);
}
}
return builder.build();
} catch (IOException e) {
Log.w(TAG, "cannot resolve " + file + " for sharing: " + e);
return null;
}
}
/**
* The two columns a share target actually reads.
*
* <p>Without {@code _display_name} the receiving app shows the URI's last
* segment, and without {@code _size} a mail client cannot tell whether the
* attachment fits before it starts reading. Both are optional in the sense
* that the transfer still works; both are the difference between "DSC_4471
* final.jpg, 8.2 MB" and an unnamed blob.
*/
@Override
public Cursor query(Uri uri, String[] projection, String selection,
String[] selectionArgs, String sortOrder) {
File file = resolve(uri);
if (file == null) {
return null;
}
String[] columns = projection != null
? projection
: new String[] {OpenableColumns.DISPLAY_NAME, OpenableColumns.SIZE};
MatrixCursor cursor = new MatrixCursor(columns, 1);
MatrixCursor.RowBuilder row = cursor.newRow();
for (String column : columns) {
if (OpenableColumns.DISPLAY_NAME.equals(column)) {
row.add(file.getName());
} else if (OpenableColumns.SIZE.equals(column)) {
row.add(file.length());
} else {
// A column we do not have. Null rather than omitted: a cursor
// whose row is shorter than its projection throws in the
// caller, which is a crash in someone else's app.
row.add(null);
}
}
return cursor;
}
/**
* From the extension, because that is all there is.
*
* <p>The type decides which apps the chooser offers, so guessing wrong
* narrows the sheet rather than breaking the transfer. Exports are JPEG,
* PNG or TIFF and {@code MimeTypeMap} knows all three.
*/
@Override
public String getType(Uri uri) {
File file = resolve(uri);
if (file == null) {
return null;
}
String name = file.getName();
int dot = name.lastIndexOf('.');
if (dot >= 0 && dot < name.length() - 1) {
String extension = name.substring(dot + 1).toLowerCase(Locale.ROOT);
String type = MimeTypeMap.getSingleton().getMimeTypeFromExtension(extension);
if (type != null) {
return type;
}
}
return "application/octet-stream";
}
/**
* Read-only, always.
*
* <p>A write mode is refused rather than quietly downgraded: a caller that
* asked for "rw" intends to save something back, and letting it open the
* file read-only would fail at its first write with an error about a
* descriptor rather than about permission. Nothing this app shares is meant
* to be edited in place by the app it was shared with.
*/
@Override
public ParcelFileDescriptor openFile(Uri uri, String mode) throws FileNotFoundException {
if (!"r".equals(mode)) {
throw new SecurityException("this provider is read-only, asked for '" + mode + "'");
}
File file = resolve(uri);
if (file == null) {
throw new FileNotFoundException("no such export: " + uri);
}
return ParcelFileDescriptor.open(file, ParcelFileDescriptor.MODE_READ_ONLY);
}
@Override
public Uri insert(Uri uri, ContentValues values) {
throw new UnsupportedOperationException("exports are written by the app, not through it");
}
@Override
public int update(Uri uri, ContentValues values, String selection, String[] selectionArgs) {
throw new UnsupportedOperationException("exports are written by the app, not through it");
}
@Override
public int delete(Uri uri, String selection, String[] selectionArgs) {
throw new UnsupportedOperationException("exports are deleted by the app, not through it");
}
/** The served root, resolved through the filesystem so the check below is real. */
private static File root(Context context) throws IOException {
return context.getFilesDir().getCanonicalFile();
}
/** The file a request names, or null if it names anything else. */
private File resolve(Uri uri) {
Context context = getContext();
if (context == null) {
return null;
}
List<String> segments = uri.getPathSegments();
if (segments.isEmpty()) {
return null;
}
try {
File root = root(context);
File candidate = root;
for (String segment : segments) {
candidate = new File(candidate, segment);
}
candidate = candidate.getCanonicalFile();
if (within(root, candidate) == null || !candidate.isFile()) {
Log.w(TAG, "refused " + uri);
return null;
}
return candidate;
} catch (IOException e) {
Log.w(TAG, "refused " + uri + ": " + e);
return null;
}
}
/**
* {@code target}'s path relative to {@code root}, or null if it is not
* under it.
*
* <p>Both sides are canonical by the time they get here, which is what
* makes one string comparison enough for {@code ../} and for a symlink
* alike. The trailing separator matters: without it a sibling directory
* whose name merely starts with the root's — {@code /data/.../files.old} —
* passes.
*/
private static String within(File root, File target) {
String rootPath = root.getPath() + File.separator;
String targetPath = target.getPath();
if (!targetPath.startsWith(rootPath)) {
return null;
}
return targetPath.substring(rootPath.length());
}
}
@@ -1,117 +0,0 @@
package paris.tourolle.darkroom;
import android.app.Activity;
import android.content.ActivityNotFoundException;
import android.content.Context;
import android.content.Intent;
import android.net.Uri;
import android.os.Bundle;
import android.util.Log;
/**
* The system's folder picker, for an album's folder on this device (FR-EXP-10).
*
* <h2>Why an activity of its own</h2>
*
* <p>{@code ACTION_OPEN_DOCUMENT_TREE} answers through
* {@code onActivityResult}, and the main activity is {@code NativeActivity},
* whose result callback is not ours to override. So this one exists only to
* ask: it starts the picker, takes the answer, and finishes — no layout, a
* translucent theme, nothing on screen but the system's own chooser, which has
* its own "New folder".
*
* <p>The answer is left in a static for Rust to poll ({@link #poll}), rather
* than called back into native code: a callback would need a registered
* native method and a thread to deliver on, and a poll from the Slint timer
* that is already running is one static call.
*
* <h2>The grant</h2>
*
* <p>A tree URI is usable only while its permission is held, and a plain
* result grants it until the process dies. {@code takePersistableUriPermission}
* keeps it across restarts — an album's folder is chosen once and exported to
* for months.
*/
public final class FolderPicker extends Activity {
private static final String TAG = "DarkRoom";
private static final int REQUEST = 0x5AF;
/** The last answer: a tree URI, "" for a cancel, null while none has come. */
private static volatile String answer = null;
/**
* Start asking. Clears any answer left from before.
*
* <p>Takes a {@code Context} rather than an {@code Activity}, because what
* native code holds (ndk_context's handle) is the application context,
* and starting an activity from one that is not an activity needs
* {@code FLAG_ACTIVITY_NEW_TASK} — without it the call throws. The picker
* shares the app's task affinity, so it still opens over the app and Back
* still returns to it.
*/
public static void start(Context from) {
answer = null;
Intent intent = new Intent(from, FolderPicker.class);
if (!(from instanceof Activity)) {
intent.addFlags(Intent.FLAG_ACTIVITY_NEW_TASK);
}
from.startActivity(intent);
}
/**
* The answer, once: a tree URI, "" if the user backed out, or null while
* the picker is still open. Reading it clears it, so a second poll after a
* cancel does not see the cancel again.
*/
public static String poll() {
String a = answer;
if (a != null) {
answer = null;
}
return a;
}
@Override
protected void onCreate(Bundle state) {
super.onCreate(state);
// Recreated after a rotation with the picker already up: asking again
// would stack a second chooser over the first.
if (state != null) {
return;
}
Intent pick = new Intent(Intent.ACTION_OPEN_DOCUMENT_TREE);
pick.addFlags(Intent.FLAG_GRANT_READ_URI_PERMISSION
| Intent.FLAG_GRANT_WRITE_URI_PERMISSION
| Intent.FLAG_GRANT_PERSISTABLE_URI_PERMISSION);
try {
startActivityForResult(pick, REQUEST);
} catch (ActivityNotFoundException e) {
Log.w(TAG, "no folder picker on this device", e);
answer = "";
finish();
}
}
@Override
protected void onActivityResult(int request, int result, Intent data) {
if (request != REQUEST) {
return;
}
Uri tree = (result == RESULT_OK && data != null) ? data.getData() : null;
if (tree == null) {
answer = "";
} else {
try {
getContentResolver().takePersistableUriPermission(tree,
Intent.FLAG_GRANT_READ_URI_PERMISSION
| Intent.FLAG_GRANT_WRITE_URI_PERMISSION);
} catch (SecurityException e) {
// Still usable this session; said in the log so a folder that
// stops working after a restart has an explanation.
Log.w(TAG, "the folder grant could not be kept: " + tree, e);
}
answer = tree.toString();
}
finish();
}
}
@@ -1,320 +0,0 @@
package paris.tourolle.darkroom;
import android.app.Activity;
import android.content.ActivityNotFoundException;
import android.content.ContentResolver;
import android.content.Context;
import android.content.Intent;
import android.database.Cursor;
import android.net.Uri;
import android.provider.OpenableColumns;
import android.util.Log;
import java.io.File;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.OutputStream;
import java.util.ArrayList;
import java.util.List;
/**
* The two directions of FR-PLAT-AND-6: what the app was opened *with*, and
* handing a finished export to somebody else.
*
* <h2>Why this is Java and not JNI in lib.rs</h2>
*
* <p>Every call below is reachable over JNI, and doing it that way would be
* roughly forty {@code call_method} invocations with their signatures written
* out as strings — each one a name Java checks at run time and nothing checks
* at build time. The Rust side would then hold the exact logic that is here,
* expressed less clearly, and a typo in {@code "()Landroid/content/Intent;"}
* would surface on a device as a {@code NoSuchMethodError} rather than at the
* compiler. So the platform work stays on the platform's side and the JNI
* surface is two calls, both taking and returning strings.
*
* <p>The class is only reachable because the APK now compiles Java at all; see
* docker/android/assemble-apk.sh.
*/
public final class Intents {
private static final String TAG = "DarkRoom";
/**
* Where incoming images are copied, under {@code getCacheDir()}.
*
* <p>The cache and not the data directory, deliberately: these are copies
* of somebody else's file, the app has no claim on them once the session
* ends, and the cache is the one place Android may reclaim under storage
* pressure without the user being asked. Putting them in the data
* directory would grow the app's footprint by a RAW file per share, for
* ever, with nothing that ever deletes them.
*/
private static final String INBOX = "incoming";
private Intents() {
}
/**
* The images this launch was asked to open, as paths the decoder can read.
*
* <p>Empty for an ordinary launch from the launcher, which is the common
* case and not a failure.
*
* <h3>Why the bytes are copied</h3>
*
* <p>A share arrives as a {@code content://} URI, which is a handle into
* another app's provider and not a path — there is no filename behind it to
* open, and the grant that makes it readable belongs to this task and dies
* with it. DarkRoom's decoders take paths (ARCH §6.9 is the note that
* Android has no paths to give), so the choice is to copy or to teach the
* whole read path about URIs, and the second is FR-PLAT-AND-1's SAF
* connector, which is not built.
*
* <p>So it is a copy, and the cost is honest: a 60 MB raw file is written
* once, to the cache, before the viewer opens. It is bounded by the share
* being a deliberate act — a person picked these files — rather than by
* anything this code does.
*
* <p>The inbox is emptied first. Without that, every share ever received
* accumulates until the platform decides the cache is too large, and the
* files are indistinguishable from each other by then.
*/
public static String[] receive(Activity activity) {
List<Uri> uris = incoming(activity.getIntent());
if (uris.isEmpty()) {
return new String[0];
}
File inbox = new File(activity.getCacheDir(), INBOX);
empty(inbox);
if (!inbox.mkdirs() && !inbox.isDirectory()) {
Log.e(TAG, "cannot create " + inbox + "; the launch intent is dropped");
return new String[0];
}
List<String> paths = new ArrayList<String>();
for (Uri uri : uris) {
String path = localise(activity, uri, inbox, paths.size());
if (path != null) {
paths.add(path);
}
}
Log.i(TAG, "launch intent carried " + paths.size() + " of " + uris.size() + " image(s)");
return paths.toArray(new String[0]);
}
/**
* Offer a file this app produced to whatever else is installed.
*
* <p>Returns false when there is nothing to offer it to, or when the file
* is not one {@link ExportProvider} may serve — both of which the caller
* has to be able to say out loud, because from the user's side a share
* button that does nothing is indistinguishable from one that failed.
*
* <p>{@code FLAG_GRANT_READ_URI_PERMISSION} is the whole security model:
* the provider is not exported, so the receiving app can reach this one
* file, for as long as its task lives, and nothing else ever.
*/
public static boolean share(Activity activity, String path, String mimeType) {
Uri uri = ExportProvider.uriFor(activity, new File(path));
if (uri == null) {
return false;
}
Intent send = new Intent(Intent.ACTION_SEND);
send.setType(mimeType != null && !mimeType.isEmpty() ? mimeType : "image/*");
send.putExtra(Intent.EXTRA_STREAM, uri);
send.addFlags(Intent.FLAG_GRANT_READ_URI_PERMISSION);
// Always a chooser, never a direct start. Android's "remembered
// default" for ACTION_SEND is a per-user setting this app has no
// business consuming: the app a photograph should go to differs every
// time, and the one time it does not, the sheet is one extra tap.
Intent chooser = Intent.createChooser(send, null);
try {
activity.startActivity(chooser);
return true;
} catch (ActivityNotFoundException e) {
Log.w(TAG, "nothing installed accepts " + mimeType + ": " + e);
return false;
}
}
/**
* The URIs an Intent carries, by the action that carried them.
*
* <p>Only the actions the manifest registers for. An action we did not
* declare cannot arrive, so handling one here would be code that reads as
* support for something the launcher will never offer.
*/
@SuppressWarnings("deprecation")
private static List<Uri> incoming(Intent intent) {
List<Uri> uris = new ArrayList<Uri>();
if (intent == null) {
return uris;
}
String action = intent.getAction();
if (Intent.ACTION_VIEW.equals(action)) {
add(uris, intent.getData());
} else if (Intent.ACTION_SEND.equals(action)) {
// The typed getParcelableExtra(String, Class) overload is API 33,
// and minSdk is 28. The deprecated form is the only one that exists
// on every device this APK installs on.
add(uris, (Uri) intent.getParcelableExtra(Intent.EXTRA_STREAM));
} else if (Intent.ACTION_SEND_MULTIPLE.equals(action)) {
ArrayList<Uri> many = intent.getParcelableArrayListExtra(Intent.EXTRA_STREAM);
if (many != null) {
for (Uri uri : many) {
add(uris, uri);
}
}
}
return uris;
}
private static void add(List<Uri> uris, Uri uri) {
if (uri != null) {
uris.add(uri);
}
}
/** A URI as a readable path, copying it into the inbox if it is not one already. */
private static String localise(Context context, Uri uri, File inbox, int index) {
// A file:// URI is already a path, and copying it would double a raw
// file on disk to no end. Rare — the platform has refused file:// URIs
// between apps since API 24 — but it is what a shell `am start -d
// file:///sdcard/…` produces, which is how this path gets tested
// without a second app installed.
if (ContentResolver.SCHEME_FILE.equals(uri.getScheme())) {
String path = uri.getPath();
if (path != null && new File(path).canRead()) {
return path;
}
Log.w(TAG, "cannot read " + uri);
return null;
}
File dest = new File(inbox, unique(inbox, displayName(context, uri), index));
InputStream in = null;
OutputStream out = null;
try {
in = context.getContentResolver().openInputStream(uri);
if (in == null) {
Log.w(TAG, "no stream behind " + uri);
return null;
}
out = new FileOutputStream(dest);
byte[] buffer = new byte[64 * 1024];
int read;
while ((read = in.read(buffer)) > 0) {
out.write(buffer, 0, read);
}
out.flush();
return dest.getAbsolutePath();
} catch (IOException e) {
Log.w(TAG, "cannot copy " + uri + ": " + e);
// The partial copy is removed rather than left: it has the name and
// the extension of a photograph and none of the bytes, and the
// decoder would report it as a corrupt file rather than a failed
// transfer.
dest.delete();
return null;
} catch (SecurityException e) {
// The grant on a shared URI dies with the task that received it.
// A process resumed from a saved state can find itself holding a
// URI it may no longer read (FR-PLAT-AND-3), and that is a lost
// permission rather than a broken file.
Log.w(TAG, "no longer permitted to read " + uri + ": " + e);
dest.delete();
return null;
} finally {
close(in);
close(out);
}
}
/**
* What the sending app calls the file, reduced to something safe to write.
*
* <p>The name is chosen by another application and lands in a path this one
* composes, so it is filtered rather than trusted: a name containing a
* separator would place the copy outside the inbox, and one beginning with
* a dot would hide it from everything that lists the directory. What
* survives is the part a photographer recognises — {@code DSC_4471.NEF} —
* which is the only reason to use the sender's name at all.
*/
private static String displayName(Context context, Uri uri) {
String name = null;
Cursor cursor = null;
try {
cursor = context.getContentResolver().query(
uri, new String[] {OpenableColumns.DISPLAY_NAME}, null, null, null);
if (cursor != null && cursor.moveToFirst() && !cursor.isNull(0)) {
name = cursor.getString(0);
}
} catch (Exception e) {
// Providers are other people's code and any of them may throw.
// A name is a convenience; failing the whole open over it is not.
Log.d(TAG, "no display name for " + uri + ": " + e);
} finally {
if (cursor != null) {
cursor.close();
}
}
if (name == null) {
name = uri.getLastPathSegment();
}
if (name == null) {
return "shared";
}
StringBuilder safe = new StringBuilder(name.length());
for (int i = 0; i < name.length(); i++) {
char c = name.charAt(i);
boolean ok = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z')
|| (c >= '0' && c <= '9') || c == '.' || c == '-' || c == '_';
safe.append(ok ? c : '_');
}
while (safe.length() > 0 && safe.charAt(0) == '.') {
safe.deleteCharAt(0);
}
return safe.length() > 0 ? safe.toString() : "shared";
}
/**
* A name nothing in the inbox has yet.
*
* <p>A multi-image share of a burst arrives as several files a camera named
* the same thing in different folders, and the second one silently
* overwriting the first would show the user one photograph where they
* picked four.
*/
private static String unique(File inbox, String name, int index) {
if (!new File(inbox, name).exists()) {
return name;
}
return index + "-" + name;
}
private static void close(java.io.Closeable stream) {
if (stream != null) {
try {
stream.close();
} catch (IOException e) {
Log.d(TAG, "close failed: " + e);
}
}
}
/** Delete the inbox's contents, one level deep, which is all it ever has. */
private static void empty(File inbox) {
File[] stale = inbox.listFiles();
if (stale == null) {
return;
}
for (File file : stale) {
if (!file.delete()) {
Log.d(TAG, "could not remove stale " + file);
}
}
}
}
@@ -1,105 +0,0 @@
package paris.tourolle.darkroom;
import android.app.Activity;
import android.content.ActivityNotFoundException;
import android.content.Intent;
import android.net.Uri;
import android.os.Bundle;
import android.webkit.WebResourceRequest;
import android.webkit.WebSettings;
import android.webkit.WebView;
import android.webkit.WebViewClient;
/**
* The manual that ships in the APK, shown in a WebView.
*
* <h2>Why an activity of our own rather than the browser</h2>
*
* <p>The desktop hands the manual to the system browser. Android leaves no
* way to do the same: the page is an asset inside the APK, which is not a
* file; an unpacked copy in app-private storage is a file no browser may
* read; a {@code file:} URI handed to another app is refused since API 24;
* and a {@code content:} URI serves the page but leaves the browser to fetch
* every picture by a relative URL against the provider, which browsers do not
* reliably do. A WebView reads {@code file:///android_asset/} straight from
* the APK, pictures and section anchor included, and nothing is unpacked.
*
* <h2>What it is not</h2>
*
* <p>A browser. JavaScript stays off (the page has none), and a link that
* leaves the manual — the design documents are on the forge — goes to the
* user's browser rather than opening inside this view, so the only thing ever
* shown here is the page the APK carries.
*
* <p>Started by {@code dr_ui::manual} with {@code Intent.setClassName}, so the
* name here and there must agree; a test in lib.rs checks the manifest
* declares it.
*/
public final class ManualActivity extends Activity {
/** The section to open at, a heading's anchor. Absent opens the top. */
public static final String EXTRA_ANCHOR = "anchor";
private static final String PAGE = "file:///android_asset/manual/index.html";
private WebView web;
@Override
protected void onCreate(Bundle saved) {
super.onCreate(saved);
setTitle("DarkRoom manual");
web = new WebView(this);
WebSettings settings = web.getSettings();
settings.setJavaScriptEnabled(false);
// Pinch to zoom into a screenshot, which is 1600 pixels wide and drawn
// at the width of a phone.
settings.setBuiltInZoomControls(true);
settings.setDisplayZoomControls(false);
web.setWebViewClient(new WebViewClient() {
@Override
public boolean shouldOverrideUrlLoading(WebView view, WebResourceRequest request) {
Uri uri = request.getUrl();
if ("file".equals(uri.getScheme())) {
return false;
}
try {
startActivity(new Intent(Intent.ACTION_VIEW, uri));
} catch (ActivityNotFoundException e) {
// No browser on the device: the link does nothing, which
// is all it could do.
}
return true;
}
});
setContentView(web);
if (saved != null) {
web.restoreState(saved);
} else {
String anchor = getIntent().getStringExtra(EXTRA_ANCHOR);
web.loadUrl(anchor == null || anchor.isEmpty() ? PAGE : PAGE + "#" + anchor);
}
}
@Override
protected void onSaveInstanceState(Bundle out) {
super.onSaveInstanceState(out);
web.saveState(out);
}
/** Back walks back through the sections visited, then leaves. */
@Override
public void onBackPressed() {
if (web.canGoBack()) {
web.goBack();
} else {
super.onBackPressed();
}
}
@Override
protected void onDestroy() {
web.destroy();
super.onDestroy();
}
}
@@ -1,114 +0,0 @@
package paris.tourolle.darkroom;
import android.content.ContentResolver;
import android.content.Context;
import android.database.Cursor;
import android.net.Uri;
import android.provider.DocumentsContract;
import android.util.Log;
import java.io.IOException;
import java.io.OutputStream;
/**
* Writing an export into a folder the user granted through
* {@link FolderPicker} — the Storage Access Framework, which is the only way
* this app reaches a folder on the device (FR-PLAT-AND-1).
*
* <p>A tree URI is not a path: a child is found by listing the folder and
* matching its display name, and created through the provider, which may
* rename it on a collision. So the name that was actually written is handed
* back, and the album records that one.
*
* <p>Two static calls, strings and a byte array in, a string out, for the
* reason {@link Intents} gives: every call here would be a signature typed as
* a string on the Rust side, and the fewer of those the better.
*/
public final class Saf {
private static final String TAG = "DarkRoom";
private Saf() {
}
/** Whether {@code name} already exists in the folder. False on any error. */
public static boolean exists(Context context, String tree, String name) {
try {
return find(context.getContentResolver(), Uri.parse(tree), name) != null;
} catch (RuntimeException e) {
Log.w(TAG, "checking " + name + " in " + tree, e);
return false;
}
}
/**
* Write {@code bytes} as {@code name} in the folder, replacing a file of
* that name when {@code replace} is set.
*
* @return the name the file has in the folder — the provider may have
* added " (1)" — or null on failure, with the reason in the log.
*/
public static String write(Context context, String tree, String name, String mime,
byte[] bytes, boolean replace) {
ContentResolver resolver = context.getContentResolver();
Uri treeUri = Uri.parse(tree);
try {
Uri target = replace ? find(resolver, treeUri, name) : null;
if (target == null) {
Uri folder = DocumentsContract.buildDocumentUriUsingTree(treeUri,
DocumentsContract.getTreeDocumentId(treeUri));
target = DocumentsContract.createDocument(resolver, folder, mime, name);
}
if (target == null) {
Log.w(TAG, "the folder refused to create " + name + " in " + tree);
return null;
}
// "wt": truncate. A replacement shorter than what it replaces
// must not keep the old file's tail.
try (OutputStream out = resolver.openOutputStream(target, "wt")) {
if (out == null) {
Log.w(TAG, "no stream for " + target);
return null;
}
out.write(bytes);
}
String written = displayName(resolver, target);
return written != null ? written : name;
} catch (IOException | RuntimeException e) {
Log.w(TAG, "writing " + name + " to " + tree, e);
return null;
}
}
/** The document for {@code name} directly in the tree's folder, or null. */
private static Uri find(ContentResolver resolver, Uri tree, String name) {
String folderId = DocumentsContract.getTreeDocumentId(tree);
Uri children = DocumentsContract.buildChildDocumentsUriUsingTree(tree, folderId);
String[] columns = {
DocumentsContract.Document.COLUMN_DOCUMENT_ID,
DocumentsContract.Document.COLUMN_DISPLAY_NAME,
};
try (Cursor c = resolver.query(children, columns, null, null, null)) {
if (c == null) {
return null;
}
while (c.moveToNext()) {
if (name.equals(c.getString(1))) {
return DocumentsContract.buildDocumentUriUsingTree(tree, c.getString(0));
}
}
}
return null;
}
private static String displayName(ContentResolver resolver, Uri document) {
String[] columns = {DocumentsContract.Document.COLUMN_DISPLAY_NAME};
try (Cursor c = resolver.query(document, columns, null, null, null)) {
if (c != null && c.moveToFirst()) {
return c.getString(0);
}
} catch (RuntimeException e) {
Log.w(TAG, "reading the name of " + document, e);
}
return null;
}
}
@@ -1,6 +0,0 @@
<?xml version="1.0" encoding="utf-8"?>
<adaptive-icon xmlns:android="http://schemas.android.com/apk/res/android">
<background android:drawable="@mipmap/ic_launcher_background"/>
<foreground android:drawable="@mipmap/ic_launcher_foreground"/>
<monochrome android:drawable="@mipmap/ic_launcher_monochrome"/>
</adaptive-icon>
Binary file not shown.

Before

Width:  |  Height:  |  Size: 9.5 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 518 B

Binary file not shown.

Before

Width:  |  Height:  |  Size: 25 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 25 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 5.0 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 343 B

Binary file not shown.

Before

Width:  |  Height:  |  Size: 13 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 13 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 15 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 680 B

Binary file not shown.

Before

Width:  |  Height:  |  Size: 43 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 43 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 30 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 1005 B

Binary file not shown.

Before

Width:  |  Height:  |  Size: 87 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 87 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 51 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.6 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 151 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 151 KiB

@@ -1,5 +0,0 @@
<?xml version="1.0" encoding="utf-8"?>
<!-- Day or night as the system is; see values/themes.xml. -->
<resources>
<style name="ManualTheme" parent="@android:style/Theme.DeviceDefault.DayNight" />
</resources>
@@ -1,10 +0,0 @@
<?xml version="1.0" encoding="utf-8"?>
<!--
The manual's theme (ManualActivity). Light below API 29, which has no
day-night theme in the platform; values-v29 follows the system from there.
The WebView takes prefers-color-scheme from whether this theme is light, and
the manual's stylesheet takes its colours from that.
-->
<resources>
<style name="ManualTheme" parent="@android:style/Theme.DeviceDefault.Light" />
</resources>
-192
View File
@@ -1,192 +0,0 @@
//! What the app was launched with, and handing a finished export back out.
//!
//! FR-PLAT-AND-6's Rust side, which is deliberately the thin side. Both
//! directions are implemented in `android/java/paris/tourolle/darkroom/` and
//! everything here is the two calls that reach them; `Intents.java` carries the
//! reasoning for the split. The short version is that a JNI method signature is
//! a string Java resolves at run time and nothing checks at build time, so
//! forty of them is forty ways for a rename to become a `NoSuchMethodError` on
//! somebody's tablet. Two is two.
//!
//! # Nothing here fails loudly
//!
//! A class the loader cannot see, a pending Java exception, a shared URI whose
//! grant died with the task that received it: each ends as a log line and an
//! empty result. This runs on the way to [`dr_ui::run`], before a window
//! exists, and the alternative to opening with an empty browsing list is not
//! opening at all.
use std::path::{Path, PathBuf};
use jni::errors::Result as JniResult;
use jni::objects::{JClass, JObject, JObjectArray, JString, JValue};
use jni::{JNIEnv, JavaVM};
/// The class both directions live in, named the way `loadClass` wants it —
/// dots, not slashes. `find_class` takes the other form, and this code calls
/// neither by accident; see [`load_class`].
const INTENTS: &str = "paris.tourolle.darkroom.Intents";
/// The images this launch was asked to open, already local and readable.
///
/// Empty for an ordinary launch from the launcher, which is the common case
/// and not a failure. What comes back is passed to `dr_ui::run` exactly as
/// command-line paths are on the desktop, so a shared photograph becomes the
/// browsing list and `startup_action` shows it rather than the launch screen.
pub fn launch_images(app: &slint::android::AndroidApp) -> Vec<PathBuf> {
with_activity(app, "reading the launch intent", |env, activity| {
let class = load_class(env, activity, INTENTS)?;
let returned = env
.call_static_method(
&class,
"receive",
"(Landroid/app/Activity;)[Ljava/lang/String;",
&[JValue::Object(activity)],
)?
.l()?;
let array = JObjectArray::from(returned);
let count = env.get_array_length(&array)?;
let mut paths = Vec::with_capacity(count as usize);
for i in 0..count {
let element = env.get_object_array_element(&array, i)?;
let text: String = env.get_string(&JString::from(element))?.into();
paths.push(PathBuf::from(text));
}
Ok(paths)
})
.unwrap_or_default()
}
/// Offer a file this app produced to whatever else is installed.
///
/// `false` means the sheet did not open — the file is not under the directory
/// [`ExportProvider`] serves, or nothing installed accepts the type. Both are
/// answers a caller has to be able to give the user, because a share control
/// that silently does nothing is indistinguishable from one that failed.
///
/// **This half has no caller yet, and that is the honest state of it.** The
/// provider, the URI grant and the chooser are all here and are what
/// FR-PLAT-AND-6 asks for; what is missing is a share control in the interface,
/// which lives in `ui/dr-ui` and needs one thing this signature shows: an
/// `AndroidApp` to call through. Wiring it means keeping a clone of the app —
/// it is `Clone` and cheap — somewhere `ui/` can reach, which is a change to
/// how the platform entry point talks to the interface rather than a change
/// here. Until that exists this function is reachable and untested, and it is
/// deliberately not tagged as covering the requirement.
///
/// `mime` decides which applications the chooser offers; the empty string
/// falls back to `image/*` on the Java side.
pub fn share(app: &slint::android::AndroidApp, file: &Path, mime: &str) -> bool {
with_activity(app, "opening the share sheet", |env, activity| {
let class = load_class(env, activity, INTENTS)?;
let path = env.new_string(file.to_string_lossy().as_ref())?;
let mime = env.new_string(mime)?;
env.call_static_method(
&class,
"share",
"(Landroid/app/Activity;Ljava/lang/String;Ljava/lang/String;)Z",
&[
JValue::Object(activity),
JValue::Object(&path),
JValue::Object(&mime),
],
)?
.z()
})
.unwrap_or(false)
}
/// Attach to the JVM, borrow the activity, and run `body` against both.
///
/// Shared by the two entry points because the three steps before the
/// interesting one are identical and each has its own way of failing. `body`
/// returning `Err` is reported here, once, in the one place that can also clear
/// a pending Java exception — see [`report`].
fn with_activity<T>(
app: &slint::android::AndroidApp,
doing: &str,
body: impl FnOnce(&mut JNIEnv, &JObject) -> JniResult<T>,
) -> Option<T> {
let vm = match unsafe { JavaVM::from_raw(app.vm_as_ptr().cast()) } {
Ok(vm) => vm,
Err(e) => {
log::error!("no JVM handle, so {doing} is skipped: {e}");
return None;
}
};
// Cheap when the thread is already attached, which it is: the glue
// attached it before it called `android_main`. The guard exists for the
// case where it is not, and costs a lookup where it is.
let mut env = match vm.attach_current_thread() {
Ok(env) => env,
Err(e) => {
log::error!("cannot attach to the JVM, so {doing} is skipped: {e}");
return None;
}
};
// SAFETY: `activity_as_ptr` documents this as an unowned JNI *global*
// reference to the Activity, valid for as long as the `AndroidApp` it came
// from. `JObject` in jni 0.21 is a plain wrapper with no `Drop`, so
// borrowing it here cannot delete a reference this code does not own — the
// one way to get this wrong is `AutoLocal` or a `GlobalRef`, both of which
// would free it out from under android-activity.
let activity = unsafe { JObject::from_raw(app.activity_as_ptr().cast()) };
match body(&mut env, &activity) {
Ok(value) => Some(value),
Err(e) => {
report(&mut env, doing, &e);
None
}
}
}
/// Look an app class up through the *activity's* class loader.
///
/// `find_class` is the obvious call and the wrong one. JNI resolves a class
/// against the loader belonging to the Java frame beneath the call, and on this
/// thread there is no such frame: `android_main` runs on a thread the native
/// glue created and attached itself, so the loader in scope is the system one.
/// It knows every class in the platform and nothing at all from this APK, and
/// says so as a `ClassNotFoundException` naming a class that is plainly in the
/// dex — which reads as a broken build rather than as the wrong loader.
///
/// The activity is a Java object, so its loader is the app's.
fn load_class<'local>(
env: &mut JNIEnv<'local>,
activity: &JObject,
name: &str,
) -> JniResult<JClass<'local>> {
let loader = env
.call_method(activity, "getClassLoader", "()Ljava/lang/ClassLoader;", &[])?
.l()?;
let name = env.new_string(name)?;
let class = env
.call_method(
&loader,
"loadClass",
"(Ljava/lang/String;)Ljava/lang/Class;",
&[JValue::Object(&name)],
)?
.l()?;
Ok(JClass::from(class))
}
/// Log a JNI failure, and clear the exception behind it if there is one.
///
/// The clearing is not tidiness. A Java exception raised through JNI stays
/// *pending* on the thread, and the next JNI call made while one is pending
/// aborts the process — so a swallowed exception here would come back as a
/// crash somewhere unrelated, most likely inside Slint. `exception_describe`
/// first, because the trace it prints to logcat is the only place the Java
/// class and line survive; `jni::errors::Error::JavaException` on its own says
/// neither.
fn report(env: &mut JNIEnv, doing: &str, e: &jni::errors::Error) {
log::error!("{doing} failed: {e}");
if let Ok(true) = env.exception_check() {
let _ = env.exception_describe();
let _ = env.exception_clear();
}
}
-633
View File
@@ -1,633 +0,0 @@
//! DarkRoom Android entry point.
//!
//! The counterpart to `darkroom-desktop`'s `main`, with two differences that
//! come from the platform rather than from choice:
//!
//! * There are no command-line paths. Android's SAF hands out document URIs,
//! not filesystem paths (ARCH §6.9), so the viewer opens with an empty
//! browsing list and the library grid is the only way in.
//! * Logging goes to logcat *and* to a file. `env_logger` writes to stderr,
//! which Android discards; logcat replaces it, and a rotating file beside it
//! replaces the thing logcat cannot be — a record that outlives the session
//! and can be sent to somebody (NFR-OPS-1, [`dr_plat::diagnostics`]).
//!
//! The first of those has one exception, and it is the launch `Intent`: a
//! gallery, a file manager or the share sheet can name images to open, and
//! those arrive as URIs on an `Intent` rather than as words on a command line.
//! [`intents`] turns them into paths, and from there they are the same list
//! the desktop builds from `argv` (FR-PLAT-AND-6).
// The whole module is JNI against classes that exist only in the APK, so it
// is gated with everything else that cannot compile off-device.
#[cfg(target_os = "android")]
mod intents;
// `slint::android` exists only when compiling for Android, so the whole entry
// point is gated on the target rather than on a feature. Without this the
// crate is still a workspace member on the host, and `cargo test --workspace`
// fails to compile it — a build break that only ever appears off-device.
#[cfg(target_os = "android")]
/// TRACES: M-13 | M-14
/// Android application entry point, called by android-activity's glue.
#[no_mangle]
fn android_main(app: slint::android::AndroidApp) {
// **The first statement in the process, and it has to be.** Everything
// between here and `install` returning runs with no logger installed at
// all: asking the activity for its external directory, `create_dir_all`
// and an `open` on a FUSE-backed volume the system may still be mounting.
// A failure or a stall in any of it is invisible on every surface there
// is — no file yet, and nothing in logcat either — which is precisely the
// kind of launch logcat exists to debug.
//
// `AndroidLogger` rather than `init_once`, so logcat can be *teed* rather
// than replaced: `init_once` installs itself as the global logger and
// there is only one of those. Everything that reached logcat before the
// file existed still reaches it, at the same level and under the same tag;
// the file is strictly additional.
let console = android_logger::AndroidLogger::new(
android_logger::Config::default()
.with_max_level(log::LevelFilter::Info)
.with_tag("DarkRoom"),
);
// Handed to the logger directly, and **not** written as `log::info!`,
// which here would compile and emit nothing: the facade's maximum level is
// `Off` until `diagnostics::install` sets it, and the macro tests that
// before it reaches any logger at all. This call skips the facade and
// reaches `__android_log_write` with nothing in between.
//
// That independence is the second reason for it. When the log is silent,
// this line is what says which half is at fault: present here and absent
// below means the `log` wiring, absent in both means liblog is not
// delivering this process's records — a question about the device, which
// no amount of reading this file can answer.
log::Log::log(
&console,
&log::Record::builder()
.level(log::Level::Info)
.target(module_path!())
.module_path(Some(module_path!()))
.args(format_args!(
"DarkRoom v{} starting; logcat only until the log file opens",
env!("CARGO_PKG_VERSION")
))
.build(),
);
// Before the file logger, because it needs somewhere to write.
//
// **The external directory, not the internal one, and the difference is
// the entire point of the file.** Both are app-private and both survive
// backgrounding — the volatile one is the *cache* directory, which is not
// in play here. What separates them is retrieval:
// `/data/data/<pkg>/files` needs `run-as` against a debuggable build or
// root to read, and `/sdcard/Android/data/<pkg>/files` is a plain
// `adb pull` from any build, needing no permission since API 19. A log
// nobody can get off the device does not do the job NFR-OPS-1 describes.
//
// The consequence is that anyone holding the tablet can read it, which is
// why `dr_plat::diagnostics` redacts at the sink and why configuration —
// the account list, and the credential reference beside it — stays on
// `internal_data_path` below rather than moving here (NFR-SEC-2).
let external = app.external_data_path();
if let Some(dir) = external.clone().or_else(|| app.internal_data_path()) {
dr_plat::set_state_dir(dir);
}
let logging = dr_plat::diagnostics::install(Box::new(console), log::LevelFilter::Info);
// Panics go to stderr, and Android discards stderr. Without this hook a
// worker thread that panics is invisible: the process survives, the
// channel it was writing to closes, and the UI reports only that
// something "failed unexpectedly" with no way to find out what.
//
// This used to be one `log::error!` of the raw panic, which had two
// problems: logcat is a ring buffer that is gone by the time a user
// reports anything, and the raw message can carry a document URI naming
// their library or a credential a library interpolated into an error
// (NFR-SEC-2). `dr_plat::crash` writes a redacted record to disk and logs
// the redacted form. Nothing uploads it.
//
// Before `set_state_dir` on purpose: the hook resolves the directory when
// it fires, so installing it first covers the startup below rather than
// leaving it uncovered.
dr_plat::crash::install(env!("CARGO_PKG_VERSION"));
log::info!("DarkRoom v{}", env!("CARGO_PKG_VERSION"));
// Said in logcat as well as in the file, because the first thing anybody
// asked for a log needs is where it is — and on a device that is a path
// nobody can guess and a command nobody remembers.
match &logging {
dr_plat::Installed::ToFile(path) => {
log::info!("logging to {}", path.display());
if external.is_none() {
log::warn!(
"no external storage; the log is app-private and needs \
`adb shell run-as paris.tourolle.darkroom cat files/darkroom.log` \
on a debuggable build"
);
}
}
dr_plat::Installed::ConsoleOnly(why) => {
log::warn!("no log file this session, only logcat: {why}");
}
}
// Before anything opens a store: Android has no $HOME and no XDG
// directories, so the default guess resolves to a path the app cannot
// write. Nothing failed loudly — the session list went to a doomed path, so
// the account survived only as long as the process and backgrounding the app
// lost the sign-in. `internal_data_path` is the app's private directory
// (ARCH §6.9).
match app.internal_data_path() {
Some(dir) => {
log::info!("data dir: {}", dir.display());
// Crash records go beside the account data rather than under it:
// both are app-private and neither is a cache, which is the whole
// distinction that matters here (see `dr_plat::crash::state_dir`).
dr_plat::crash::set_state_dir(dir.join("state"));
dr_sync::account::set_data_dir(dir);
}
None => log::error!("no internal data path; settings will not persist"),
}
// After the data dir, because it writes beside the catalog. **Not** before
// the first frame any more — it starts a worker and returns; see the
// function for what it used to cost the launch.
install_bundled_models(app.clone());
// Before `init_with_event_listener`, which takes `app` by value and is the
// last moment anything can ask the activity a question. Not an ordering
// preference — after that line there is no `app` left to read the Intent
// through.
let opened_with = intents::launch_images(&app);
// TRACES: FR-PLAT-AND-5
// The listener is the whole reason this is not the one-line
// `slint::android::init(app)`. Slint owns the event loop on Android, so
// the platform's lifecycle and memory events reach the application only if
// it asks for them here — and it must ask *before* the loop starts, which
// is why this sits between the data directory and `dr_ui::run`.
//
// The listener runs inside `poll_events`, on the same thread the event
// loop and every interface cache live on, which is what lets
// `dr_ui::memory` be a thread-local registry of plain `Fn()` rather than a
// cross-thread channel (see its module documentation).
//
// # Why two events and not eight
//
// FR-PLAT-AND-5 names `onTrimMemory`, whose `TRIM_MEMORY_*` levels grade
// how badly the system wants the memory back. Those levels do not exist
// here: `ComponentCallbacks2` is a Java interface implemented by an
// `Activity` or `Application`, and this app has neither — it is a bare
// `NativeActivity`, whose native callback table offers only the ungraded
// `onLowMemory`. android-activity surfaces exactly that as `LowMemory`.
// Reading the grades would mean shipping a Java subclass to forward them,
// which is a distribution-manifest change and not this one.
//
// `Stop` recovers the one grade that matters most anyway, and for free.
// It is the moment the activity stops being visible — `TRIM_MEMORY_UI_HIDDEN`
// in all but name — and it is the cheapest possible time to give memory
// back, because nothing that is freed has to be drawn again before anyone
// sees it. `Pause` deliberately does not qualify: a permission dialog or
// the share sheet pauses an activity that is still on screen behind it,
// and throwing away its render pipeline would make every such interruption
// cost a full re-render.
use slint::android::android_activity::{MainEvent, PollEvent};
if let Err(e) = slint::android::init_with_event_listener(app, |event| match event {
PollEvent::Main(MainEvent::LowMemory) => {
dr_ui::memory::relieve(dr_ui::memory::Level::Critical);
}
PollEvent::Main(MainEvent::Stop) => {
dr_ui::memory::relieve(dr_ui::memory::Level::UiHidden);
}
_ => {}
}) {
log::error!("Slint Android backend failed to initialise: {e}");
return;
}
// The launch Intent's images, where there were any, standing in for the
// desktop's argv — `launch::startup_action` treats a non-empty list as
// "the user asked for these specifically", which is exactly what a share
// or a tap in a gallery is. Empty for an ordinary launch, and the library
// opens as before.
//
// Returning from `android_main` ends the process, so a failure here is
// logged rather than propagated — there is no shell to show `Err` to.
if let Err(e) = dr_ui::run(opened_with) {
log::error!("DarkRoom exited with error: {e:#}");
}
}
/// Unpack the models the APK carries, if it carries any.
///
/// # Why Android needs this and no other platform does
///
/// A desktop build reads its models from a path — the account's directory, the
/// shared one, or `$XDG_DATA_DIRS` where a package put them. **Android has no
/// such path.** `internal_data_path` is app-private, `run-as` needs a
/// debuggable build, and an asset inside a package is not a path anything can
/// open (ARCH §6.9), so a phone had no way to reach a model at all.
///
/// So the APK carries them in `assets/models/` and this copies them out, once,
/// into the same shared directory a desktop install uses. After that every
/// lookup in `dr_ui::library` finds them exactly where it finds a desktop
/// user's.
///
/// # The two sets are not the same kind of thing
///
/// **Face weights are absent from the repository by design.** The InsightFace
/// grant is research-only and incompatible with this project's licence
/// (docs/dev/faces.md §2), so a desktop user fetches them, runs
/// `tools/fix-face-model-shapes.sh` over them, and drops the result in. A build
/// that carries none is the ordinary case and face indexing simply stays off.
///
/// **The scene model is committed** (AGPL, compatible — `models/LICENCE.md`),
/// so a build carrying none means a checkout without `git lfs pull` rather than
/// a deliberate omission. It is still not an error here: the scene tab reports
/// itself unavailable the same way face indexing does, because a photo editor
/// that refuses to start over a missing grading feature is worse than one that
/// starts without it.
///
/// # Why it is not `include_bytes!` like the instance model
///
/// Size. The instance model is 11 MB and compiled in; the scene model is 24 MB
/// on top of that, and a 35 MB constant in the binary is paid by every install
/// whether or not the tab is opened. Assets are also *stored* rather than
/// deflated in the APK (see `assemble-apk.sh`), so unpacking is a copy rather
/// than an inflate.
///
/// # Why it returns before it has done anything
///
/// **`android_main` runs with the input channel unserviced.** Nothing drains
/// it until Slint reaches `poll_events`, and Slint does not reach `poll_events`
/// until `dr_ui::run` calls `window.run()`, which is the last line of it. So
/// every millisecond spent between the top of `android_main` and that line is a
/// millisecond in which Android's input dispatcher gets no answer, and five
/// thousand of them is an ANR by definition — the system puts "DarkRoom isn't
/// responding" over a window that has never painted, and offers to kill it.
///
/// This copied **41 MB** on the first launch after an install: 24.9 MB of scene
/// model, 13.6 MB of embedder, 2.5 MB of detector, each read whole out of the
/// APK and written to `/data`. v0.10.0 added the scene model, which was 60% of
/// that total; v0.10.0 is the release the ANR appeared in, and the 8,010 minor
/// faults in its report are what 41 MB of freshly touched pages looks like.
/// The two further detectors the settings page offers since have made it
/// 61 MB, which is the same argument with a larger number.
///
/// So it runs on a worker (NFR-ARCH-1: nothing blocking on the UI executor) and
/// this function returns as soon as the thread is running. Nothing on the
/// launch path waits for it, and no other startup step needs its result.
///
/// # The window in which a model looks absent, and why that is honest enough
///
/// Until the copy finishes, `library::face_models` and `library::scene_model`
/// answer `is_file()` about files that are not written yet, so both report
/// their feature unavailable — the same answer they give a build carrying no
/// weights at all, which is the ordinary case this whole path was written
/// around. It is briefly pessimistic rather than wrong, it lasts about as long
/// as it takes to read one screenful of the grid, and the temporary name
/// [`unpack_bundled_models`] writes under is what stops it being worse than
/// pessimistic: a lookup never sees a half-written file, only an absent one.
#[cfg(target_os = "android")]
fn install_bundled_models(app: slint::android::AndroidApp) {
// Detached rather than joined: there is no later moment on the launch path
// that wants the answer, and a handle nobody joins is a handle nobody can
// forget to. `AndroidApp` is documented `Send` and `Sync` and is an `Arc`
// internally, so the clone costs a refcount; `asset_manager` is asked for
// on the worker because `AAssetManager` is thread-safe by contract and
// reading the pointer takes only the app's read lock, which `poll_events`
// also only ever holds shared.
std::thread::spawn(move || unpack_bundled_models(&app));
}
/// The copy itself, on the worker [`install_bundled_models`] starts.
#[cfg(target_os = "android")]
fn unpack_bundled_models(app: &slint::android::AndroidApp) {
use std::io::Read;
let started = std::time::Instant::now();
// The face names are the **shape-fixed** exports, matching what
// `library::face_models` looks for: tract cannot parse either InsightFace
// graph with its dynamic input dimension, so what ships here has already
// been through `tools/fix-face-model-shapes.sh`.
//
// The scene entries are three files rather than one because the graph alone
// decodes to 150 anonymous channels — `library::scene_model` wants the
// vocabulary and the category descriptor beside it, and requires all three
// before it reports the tab available.
//
// Three detectors, because which one runs is a setting
// (`FaceDetector`, docs/dev/faces.md §12.3) and a tablet has no other way to
// obtain the one it was not shipped with. Twenty megabytes of APK for
// the choice; the embedder is the same for all three.
//
// Then the three eye-state models (docs/dev/faces.md §17): landmarks, open
// or closed, sunglasses. The app indexes without them; with them the
// eyes-open filter has something to read, and a tablet has no other way
// to get them either.
//
// The int8 forms beside the three detectors are what the Hexagon runs
// (docs/dev/inference.md §5); the engine loads the sibling when the probe
// chose that rung and ignores it otherwise.
const BUNDLED: [(&std::ffi::CStr, &str); 14] = [
(c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"),
(
c"models/scrfd_500m_640.int8.onnx",
"scrfd_500m_640.int8.onnx",
),
(c"models/scrfd_2.5g_640.onnx", "scrfd_2.5g_640.onnx"),
(
c"models/scrfd_2.5g_640.int8.onnx",
"scrfd_2.5g_640.int8.onnx",
),
(c"models/scrfd_10g_640.onnx", "scrfd_10g_640.onnx"),
(c"models/scrfd_10g_640.int8.onnx", "scrfd_10g_640.int8.onnx"),
(c"models/arcface_mbf_b1.onnx", "arcface_mbf_b1.onnx"),
(c"models/2d106det_b1.onnx", "2d106det_b1.onnx"),
(c"models/ocec_s_b1.onnx", "ocec_s_b1.onnx"),
(c"models/sgc_l_48_b1.onnx", "sgc_l_48_b1.onnx"),
(c"models/yolo26s-sem-ade20k.onnx", "yolo26s-sem-ade20k.onnx"),
(
c"models/yolo26s-sem-ade20k.classes.json",
"yolo26s-sem-ade20k.classes.json",
),
(c"models/categories.txt", "categories.txt"),
// The panorama border filler (FR-MRG-4); MIT, 28 MB.
(c"models/migan-512.onnx", "migan-512.onnx"),
];
let dir = dr_ui::shared_face_models_dir();
let assets = app.asset_manager();
let mut copied = 0u64;
for (asset_path, name) in BUNDLED {
let dest = dir.join(name);
// Already unpacked. Not re-read on every launch: this is 73 MB of
// copying across the ten entries, and the file does not change without
// the APK changing, at which point the install wiped it anyway. It
// matters more now than it did — a launch that skips every entry here
// costs nothing at all, which is what makes the second launch after an
// install cheap even though the first one is not.
if dest.is_file() {
continue;
}
let Some(mut asset) = assets.open(asset_path) else {
log::info!("no bundled {name} in this APK; the feature needing it stays off");
continue;
};
let mut bytes = Vec::new();
if let Err(e) = asset.read_to_end(&mut bytes) {
log::error!("bundled {name} could not be read: {e}");
continue;
}
if let Err(e) = std::fs::create_dir_all(&dir) {
log::error!("cannot create {}: {e}", dir.display());
return;
}
// Written under a temporary name and renamed, because
// `library::face_models` and `library::scene_model` both decide a
// feature is available on `is_file()` alone. A truncated write — the process backgrounded and
// killed mid-copy — would otherwise leave a file that passes that test
// and fails inside tract, reported to the user as a broken model rather
// than a missing one.
let part = dir.join(format!("{name}.part"));
match std::fs::write(&part, &bytes).and_then(|()| std::fs::rename(&part, &dest)) {
Ok(()) => {
copied += bytes.len() as u64;
log::info!("installed bundled {name} ({} bytes)", bytes.len());
}
Err(e) => {
log::error!("cannot install {name}: {e}");
let _ = std::fs::remove_file(&part);
}
}
}
// The figure this whole function is about. Said even when it is zero, so a
// launch that ANRs anyway can be told apart from one that spent its six
// seconds here — on a second launch there is nothing left to copy and the
// line reads `0 bytes`.
log::info!(
"bundled models ready: {copied} bytes copied in {} ms",
started.elapsed().as_millis()
);
// Now, and not at launch: the probe fingerprints the model files, and
// on a first launch they were not on disk until this line. The runtime
// is in the APK's native library directory beside `libdarkroom.so`,
// which is also where Qualcomm's DSP loader has to be pointed for the
// Hexagon skel (docs/dev/inference.md §3, §8).
dr_ui::inference::init(native_library_dir().into_iter().collect());
}
/// The directory the system unpacked this APK's native libraries into.
///
/// Read from where the loader put *this* library rather than asked of the
/// activity: `android-activity` does not expose `nativeLibraryDir`, and the
/// answer is in `/proc/self/maps` for free.
#[cfg(target_os = "android")]
fn native_library_dir() -> Option<std::path::PathBuf> {
let maps = std::fs::read_to_string("/proc/self/maps").ok()?;
maps.lines()
.filter_map(|l| l.split_whitespace().nth(5))
.find(|p| p.ends_with("/libdarkroom.so"))
.and_then(|p| std::path::Path::new(p).parent().map(Into::into))
}
/// TRACES: FR-PLAT-AND-6
/// The declarations that make this app a receiver, held to on the host.
///
/// Everything FR-PLAT-AND-6 does on a device is unreachable from `cargo test`:
/// there is no `Intent` off-device and no `ContentProvider` to instantiate. But
/// the requirement is not only behaviour — half of it is *declaration*, and a
/// declaration can be wrong in ways that compile perfectly and fail silently.
/// An intent filter that is deleted takes the app out of every gallery's "open
/// with" menu with nothing to notice; an authority that stops matching the
/// class it names raises a `SecurityException` in whichever other app opened
/// the share sheet, which is the last place anybody would look for it.
///
/// The manifest is read by aapt2 and the Java by javac, so a Rust build sees
/// neither. `include_str!` is what puts them where a test can reach them, and
/// this is the only place in the workspace that does.
#[cfg(test)]
mod tests {
/// The manifest with its comments removed and its whitespace flattened, so
/// a match is about the declaration and not about how it is indented.
fn manifest() -> String {
let xml = include_str!("../android/AndroidManifest.xml");
let mut out = String::with_capacity(xml.len());
let mut rest = xml;
// Comments first, and not by regex over the whole file: several of them
// quote the very attribute names the assertions below look for, so a
// test that read them would pass on the strength of the prose
// explaining an entry that had been deleted.
while let Some(start) = rest.find("<!--") {
out.push_str(&rest[..start]);
match rest[start..].find("-->") {
Some(end) => rest = &rest[start + end + 3..],
None => {
rest = "";
break;
}
}
}
out.push_str(rest);
out.split_whitespace().collect::<Vec<_>>().join(" ")
}
/// The body of each `<intent-filter>`, so an action and a MIME type are
/// checked to be in the *same* filter. Two filters, one naming the action
/// and one naming the type, register for neither.
fn intent_filters(manifest: &str) -> Vec<&str> {
manifest
.split("<intent-filter>")
.skip(1)
.filter_map(|filter| filter.split("</intent-filter>").next())
.collect()
}
/// The single `<provider>` element, attributes and all.
fn provider(manifest: &str) -> String {
let start = manifest
.find("<provider")
.expect("no <provider> in the manifest");
let rest = &manifest[start..];
let end = rest.find("/>").expect("unterminated <provider> element");
rest[..end + 2].to_string()
}
fn attribute(element: &str, name: &str) -> Option<String> {
let key = format!("{name}=\"");
let start = element.find(&key)? + key.len();
let value = element[start..].split('"').next()?;
Some(value.to_string())
}
#[test]
fn a_gallery_can_open_a_photograph_in_this_app() {
let manifest = manifest();
let registered = intent_filters(&manifest).iter().any(|filter| {
filter.contains("android.intent.action.VIEW")
&& filter.contains("android.intent.category.DEFAULT")
&& filter.contains(r#"android:mimeType="image/*""#)
});
assert!(
registered,
"no VIEW filter for image/*: nothing will offer DarkRoom for a photograph"
);
}
#[test]
fn the_share_sheet_can_send_one_image_or_several() {
let manifest = manifest();
let registered = intent_filters(&manifest).iter().any(|filter| {
// The closing quote matters: SEND is a prefix of SEND_MULTIPLE, so
// a bare substring test passes on a filter that declares only the
// second and would not be offered for a single photograph.
filter.contains(r#"android.intent.action.SEND""#)
&& filter.contains(r#"android.intent.action.SEND_MULTIPLE""#)
&& filter.contains("android.intent.category.DEFAULT")
&& filter.contains(r#"android:mimeType="image/*""#)
});
assert!(
registered,
"no SEND/SEND_MULTIPLE filter for image/*, so the share sheet will not list DarkRoom"
);
}
#[test]
fn one_activity_ever_so_a_second_launch_cannot_start_a_second_one() {
// Not style. Another app can now launch this activity while it is
// already running, and the default launch mode answers that by
// creating a second NativeActivity in this process — a second
// android_main, a second Slint backend, a second wgpu device.
assert!(
manifest().contains(r#"android:launchMode="singleTask""#),
"the activity must be singleTask; see the manifest comment"
);
}
#[test]
fn the_provider_authority_is_the_one_the_class_answers_to() {
let manifest = manifest();
let element = provider(&manifest);
let declared =
attribute(&element, "android:authorities").expect("the provider declares no authority");
let java = include_str!("../android/java/paris/tourolle/darkroom/ExportProvider.java");
let constant = java
.split("AUTHORITY = \"")
.nth(1)
.and_then(|rest| rest.split('"').next())
.expect("ExportProvider declares no AUTHORITY constant");
assert_eq!(
declared, constant,
"the manifest and ExportProvider disagree about the authority; \
a share would fail as a SecurityException inside the receiving app"
);
let class = attribute(&element, "android:name").expect("the provider declares no class");
let (package, _) = class
.rsplit_once('.')
.expect("the provider class is unqualified");
assert!(
java.contains(&format!("package {package};")),
"the manifest names {class}, which is not the class in ExportProvider.java"
);
}
/// `dr_ui::manual` starts the manual by class name. A name the manifest
/// does not declare is an `ActivityNotFoundException` on the device and a
/// Manual button that does nothing, so the three spellings — dr_ui's, the
/// manifest's and the Java file's — are checked to be one.
#[test]
fn the_manual_activity_dr_ui_starts_is_declared() {
let manifest = manifest();
let wanted = dr_ui::manual::ANDROID_ACTIVITY;
let element = manifest
.split("<activity")
.skip(1)
.find(|a| attribute(a, "android:name").as_deref() == Some(wanted))
.unwrap_or_else(|| panic!("the manifest declares no activity {wanted}"));
assert_eq!(
attribute(element, "android:exported").as_deref(),
Some("false"),
"the manual activity has no reason to be startable by another app"
);
let java = include_str!("../android/java/paris/tourolle/darkroom/ManualActivity.java");
let (package, class) = wanted.rsplit_once('.').expect("unqualified class name");
assert!(java.contains(&format!("package {package};")));
assert!(java.contains(&format!("class {class} ")));
assert!(
java.contains(&format!(
"EXTRA_ANCHOR = \"{}\"",
dr_ui::manual::ANDROID_EXTRA_ANCHOR
)),
"ManualActivity reads the section from a different extra than dr_ui writes"
);
}
#[test]
fn the_provider_hands_out_one_file_at_a_time_and_nothing_by_itself() {
let manifest = manifest();
let element = provider(&manifest);
// The two halves are not redundant. Without the grant, every share
// target fails; exported, every app on the device could read this
// app's private directory.
assert_eq!(
attribute(&element, "android:exported").as_deref(),
Some("false"),
"an exported provider would serve the app's private directory to anything installed"
);
assert_eq!(
attribute(&element, "android:grantUriPermissions").as_deref(),
Some("true"),
"without URI grants the share sheet opens and every target fails to read the file"
);
}
}
-30
View File
@@ -1,30 +0,0 @@
[package]
name = "darkroom-desktop"
version.workspace = true
edition.workspace = true
rust-version.workspace = true
license.workspace = true
[dependencies]
dr-ui = { workspace = true, features = ["scene-model"] }
# For the panic hook and the log sink, directly rather than through dr-ui:
# both have to be installed before `dr_ui::run`, because a panic during startup
# is exactly the one they exist to catch and record (NFR-OPS-1, NFR-OPS-2).
dr-plat.workspace = true
anyhow.workspace = true
env_logger.workspace = true
log.workspace = true
# The Windows resource block — icon and version — compiled in by build.rs.
# Unconditional rather than under `[target.'cfg(windows)']`, because a cfg on
# a build-dependency is evaluated against the *host* — the machine running
# the build script — and this is built for Windows from Linux. The script
# itself returns before touching the crate on every other target.
[build-dependencies]
winresource = "0.1"
[features]
default = []
# The manual's recording hook (dr-ui's `automation`); tools/manual/record.sh
# builds with it, nothing else does.
automation = ["dr-ui/automation"]
-58
View File
@@ -1,58 +0,0 @@
//! TRACES: FR-PLAT-WIN-2
//! The Windows resource block: icon and version, compiled into the executable.
//!
//! Windows takes an application's icon and its "Details" tab from a resource
//! inside the `.exe`, not from a `.desktop` file, so without this the installed
//! program shows the generic executable icon in Explorer, the Start Menu and
//! the taskbar, and reports no version. Nothing here runs for any other
//! target: the whole body is behind the target-OS check, and the crate that
//! does the work is a build-dependency only.
//!
//! The icon is the same PNG every other platform uses, wrapped into an `.ico`
//! in `OUT_DIR` rather than committed: an ICO entry may *be* a PNG (Vista and
//! later read them directly), so the wrapper is a 22-byte header and the
//! file's bytes, and a generated binary stays out of the tree.
use std::io::Write as _;
use std::path::PathBuf;
fn main() {
println!("cargo:rerun-if-changed=build.rs");
if std::env::var("CARGO_CFG_TARGET_OS").as_deref() != Ok("windows") {
return;
}
let png = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../ui/dr-ui/ui/app-icon.png");
println!("cargo:rerun-if-changed={}", png.display());
let bytes = std::fs::read(&png).expect("read app-icon.png");
let ico = PathBuf::from(std::env::var("OUT_DIR").unwrap()).join("darkroom.ico");
write_png_ico(&ico, &bytes, 256).expect("write darkroom.ico");
let mut res = winresource::WindowsResource::new();
res.set_icon(ico.to_str().unwrap());
res.set("ProductName", "DarkRoom");
res.set("FileDescription", "DarkRoom");
res.set("LegalCopyright", "GPL-3.0-or-later");
// Cross-compiling: `winresource` looks for a `windres` for the target and
// the Windows image names it explicitly, for the same reason the Android
// image names its linkers.
if let Ok(windres) = std::env::var("WINDRES") {
res.set_windres_path(&windres);
}
res.compile().expect("compile the Windows resource block");
}
/// One PNG image as an `.ico`. `edge` is the PNG's width and height; 256 is
/// written as 0 per the format.
fn write_png_ico(path: &std::path::Path, png: &[u8], edge: u32) -> std::io::Result<()> {
let mut f = std::fs::File::create(path)?;
let dim = if edge >= 256 { 0u8 } else { edge as u8 };
// ICONDIR: reserved, type 1 (icon), one image.
f.write_all(&[0, 0, 1, 0, 1, 0])?;
// ICONDIRENTRY: width, height, palette 0, reserved, planes 1, bpp 32,
// byte length, offset (6 + 16).
f.write_all(&[dim, dim, 0, 0, 1, 0, 32, 0])?;
f.write_all(&(png.len() as u32).to_le_bytes())?;
f.write_all(&22u32.to_le_bytes())?;
f.write_all(png)
}
-113
View File
@@ -1,113 +0,0 @@
//! DarkRoom desktop entry point.
//!
//! darkroom-desktop <file-or-directory>...
//! darkroom-desktop --version
// TRACES: FR-PLAT-WIN-2
// A GUI-subsystem executable, or Windows opens a console window behind the
// application for the life of the process. Release only: the console is where
// the log goes when there is no file, and a debug build is run from one.
// `--version` still prints under this — stdout is simply not attached when
// launched from Explorer, which is not where anyone asks for a version.
#![cfg_attr(all(windows, not(debug_assertions)), windows_subsystem = "windows")]
use std::path::PathBuf;
use dr_plat::diagnostics::Installed;
fn main() -> anyhow::Result<()> {
// TRACES: FR-PLAT-WIN-3
// Before the logger, the crash hook and everything else: this exists so a
// build made on a machine that cannot run the application — the Linux CI
// producing the Windows binary, checked under Wine — has an exit that
// proves the executable starts without opening a window or touching the
// user's directories (docs/dev/windows.md §6).
if std::env::args().nth(1).as_deref() == Some("--version") {
println!("darkroom-desktop {}", env!("CARGO_PKG_VERSION"));
return Ok(());
}
// Built rather than `init`ed, so the same logger can be handed to the
// diagnostics tee: `env_logger` keeps writing to stderr exactly as before,
// and every record it accepts is also appended to the on-disk log
// (NFR-OPS-1). `filter()` is asked afterwards because the environment may
// have overridden the default below, and the file must not be quieter than
// the terminal.
let console = env_logger::Builder::from_env(env_logger::Env::default().default_filter_or(
"info,wgpu_core=warn,wgpu_hal=warn,zbus=warn,tracing=warn,calloop=warn,rawler=warn",
))
.build();
let level = console.filter();
let logging = dr_plat::diagnostics::install(Box::new(console), level);
// Immediately after the logger and before anything that could fail. Until
// now a panic on desktop went to stderr and died with the terminal, which
// means every panic a user has ever hit was unreportable: the process
// survives (the panicking worker does not), a control goes dead, and there
// is nothing on disk to say why. The record is local and stays local —
// there is no upload path, by design; see `dr_plat::crash`.
dr_plat::crash::install(env!("CARGO_PKG_VERSION"));
log::info!("DarkRoom v{}", env!("CARGO_PKG_VERSION"));
// First thing in the file, so a user asked for "the log" can find it
// without being told a path over the phone.
match &logging {
Installed::ToFile(path) => log::info!("logging to {}", path.display()),
Installed::ConsoleOnly(why) => log::warn!("no log file this session: {why}"),
}
let paths: Vec<PathBuf> = std::env::args().skip(1).map(PathBuf::from).collect();
if paths.is_empty() {
eprintln!("usage: darkroom-desktop <file-or-directory>...");
}
// Before the window: the probe runs on its own thread and the first
// frame does not wait for it, but the models a background job asks for
// should already know where the runtime is (docs/dev/inference.md §4).
dr_ui::inference::init(runtime_dirs());
dr_ui::run(paths)?;
// Skip Rust's normal static/thread-local teardown on the way out: a
// background zbus/keyring connection opened by dr_ui::launch_ui can
// still be alive here, and unwinding through it races its async-io
// reactor thread, panicking with "thread local ... during or after
// destruction" when the window is closed.
std::process::exit(0);
}
/// Where a desktop package may have put `libonnxruntime`, most specific
/// first. None of these existing is the tract build, which is a complete
/// application and not an error (docs/dev/inference.md §3).
///
/// `DARKROOM_ORT_DIR` is for a developer pointing at a runtime that is not
/// installed — the wheel's `capi` directory, say. Then beside the executable
/// and in the package's private library directory, for a package that
/// bundles its own; then the user's own `runtime/` beside the models, where
/// `tools/fetch-desktop-runtime.sh` puts one; then the Flatpak prefix; then
/// the system library directory, for a distribution that ships ONNX Runtime
/// as a package of its own. The user's copy outranks the system's because
/// the system's is the one most likely to be built without the GPU
/// providers, or against the wrong cuDNN — and a system copy whose providers
/// do not load is not a problem, only a slower app: the probe builds a real
/// session before believing a provider.
fn runtime_dirs() -> Vec<PathBuf> {
let mut dirs = Vec::new();
if let Some(dir) = std::env::var_os("DARKROOM_ORT_DIR") {
dirs.push(PathBuf::from(dir));
}
if let Ok(exe) = std::env::current_exe() {
if let Some(bin) = exe.parent() {
dirs.push(bin.to_path_buf());
dirs.push(bin.join("../lib/darkroom"));
}
}
dirs.push(dr_ui::inference::user_runtime_dir());
#[cfg(target_os = "linux")]
dirs.extend([
PathBuf::from("/app/lib/darkroom"),
PathBuf::from("/usr/lib/darkroom"),
PathBuf::from("/usr/lib"),
]);
dirs
}
-34
View File
@@ -1,34 +0,0 @@
[package]
name = "dr-catalog"
version.workspace = true
edition.workspace = true
rust-version.workspace = true
license.workspace = true
[dependencies]
dr-types.workspace = true
# The face subsystem's arithmetic — `Calibration` in particular, so the sigmoid
# that turns a cosine into a probability has exactly one definition. Default
# features are off, so this brings in no ONNX runtime and no weights: only the
# model-free half compiles here.
dr-face.workspace = true
# For `SHARD_MAX_BYTES` alone. The face shards are capped at the same 25 MB the
# thumbnail shards are, and sharing the constant is what keeps them from
# drifting apart — the cap is a statement about sync cost, not about thumbnails.
dr-thumbs.workspace = true
# The `Storage` trait, and nothing else from it. A scan has to read a real
# directory, and this is how `core/` reaches the platform without a
# `#[cfg(target_os)]` of its own (ARCH §4.1: calls go downward).
dr-plat.workspace = true
rusqlite.workspace = true
thiserror.workspace = true
log.workspace = true
# `collections.selector_json` — the stored form of a smart collection's
# selector. The column predates this dependency; nothing else here is JSON.
serde_json.workspace = true
# For the `scan_local` example only, which is a diagnostic tool: what it is
# diagnosing is often a folder the scan warned about and skipped, and those
# warnings go to `log`.
[dev-dependencies]
env_logger.workspace = true
-294
View File
@@ -1,294 +0,0 @@
//! What the catalog's routine reads cost on a real library, off the GUI.
//!
//! cargo run --release -p dr-catalog --example catalog_bench -- CATALOG.sqlite [FACES_DIR] [--remote PEER.sqlite]
//!
//! Times `Catalog::open` — which every worker thread pays, including the
//! develop view's fetch of each original and each neighbour it prefetches —
//! and the backfill that runs inside it, step by step. Run it against a
//! *copy* of a real catalog: opening migrates and backfills, which write.
//!
//! Then what the library screen reads on every keystroke and scroll: the
//! filter chips' counts, the keyword panel, and the grid's total. Two of
//! those live in `dr-ui` (`library::local_original_count` and the grid
//! count), whose `library` module is private; their SQL is spelled here as
//! it is spelled there, and has to be kept in step by hand.
//!
//! `--remote` also times a merge with another device's catalog — the
//! server snapshot — which is the pass where the two disagree: faces one
//! side found and the other did not, boxes that moved. The merge with a copy
//! of itself matches every face by its box and never reaches that work. The
//! first of its runs writes what the peer brought; the rest are the steady
//! state, so compare two builds from two fresh copies of one catalog.
//!
//! The figures are for reading side by side before and after a change; they
//! are not a gate. Compare the `cpu` column when the machine is busy. The
//! answers are printed too, so two builds can be checked for agreeing.
use std::path::PathBuf;
use std::time::{Duration, Instant};
use dr_catalog::{keywords, rating, schema, Catalog};
fn main() {
let mut args: Vec<String> = std::env::args().skip(1).collect();
let peer = args.iter().position(|a| a == "--remote").map(|at| {
let path = args.get(at + 1).map(PathBuf::from).unwrap_or_else(|| {
eprintln!("--remote needs a catalog");
std::process::exit(2);
});
args.drain(at..at + 2);
path
});
let Some(path) = args.first().map(PathBuf::from) else {
eprintln!("usage: catalog_bench CATALOG.sqlite");
std::process::exit(2);
};
// Once untimed, so a migration or a first backfill is not in the figures.
drop(Catalog::open(&path).expect("catalog"));
time("Catalog::open", 20, || {
drop(Catalog::open(&path).unwrap());
});
// One develop landing, in the two shapes the app has had. Five opens is
// what `fetch_original` and a `holds_original` per prefetched neighbour
// cost when each asked on a connection of its own; two is the fetch plus
// one connection the prefetch worker keeps for its batch's row checks.
// Five images from the library, against an empty cache: the question is
// asked the same way whatever the answer.
let images: Vec<dr_types::ImageId> = {
let c = Catalog::open(&path).unwrap();
let mut stmt = c
.connection()
.prepare("SELECT id FROM images ORDER BY id LIMIT 5 OFFSET 1000")
.unwrap();
let ids = stmt
.query_map([], |r| r.get::<_, i64>(0))
.unwrap()
.map(|id| dr_types::ImageId(id.unwrap() as u64))
.collect();
ids
};
let cache_dir = path.with_extension("bench-cache");
let budget = dr_catalog::Budget::default();
time("landing: 5 opens (fetch + 4 row checks)", 20, || {
let store = dr_catalog::Cache::open(&cache_dir, budget).unwrap();
let c = Catalog::open(&path).unwrap();
let _ = store.load(c.connection(), images[0], 0).unwrap();
for &image in &images[1..] {
let store = dr_catalog::Cache::open(&cache_dir, budget).unwrap();
let c = Catalog::open(&path).unwrap();
let _ = store.holds_original(c.connection(), image);
}
});
time("landing: 2 opens (fetch + held row checks)", 20, || {
let store = dr_catalog::Cache::open(&cache_dir, budget).unwrap();
let c = Catalog::open(&path).unwrap();
let _ = store.load(c.connection(), images[0], 0).unwrap();
let held = Catalog::open(&path).unwrap();
for &image in &images[1..] {
let store = dr_catalog::Cache::open(&cache_dir, budget).unwrap();
let _ = store.holds_original(held.connection(), image);
}
});
let _ = std::fs::remove_dir_all(&cache_dir);
let catalog = Catalog::open(&path).unwrap();
let conn = catalog.connection();
time("schema::backfill (all steps)", 20, || {
schema::backfill(conn).unwrap();
});
time(" rating::ensure_default_versions", 20, || {
rating::ensure_default_versions(conn).unwrap();
});
time(" rating::align_default_version_uuids", 20, || {
rating::align_default_version_uuids(conn).unwrap();
});
time(" keywords::adopt_orphan_terms", 20, || {
keywords::adopt_orphan_terms(conn).unwrap();
});
interactive(conn);
// A sync pass: the upload snapshot, then a merge of the catalog with a
// copy of itself — every row a match, which is the steady state.
let scratch = path.with_extension("bench-snapshot");
time("snapshot_for_upload", 3, || {
let _ = std::fs::remove_file(&scratch);
catalog.snapshot_for_upload(&scratch).unwrap();
});
println!(
" snapshot size {:.1} MB",
std::fs::metadata(&scratch).map(|m| m.len()).unwrap_or(0) as f64 / 1e6
);
let remote = path.with_extension("bench-remote");
let _ = std::fs::remove_file(&remote);
conn.execute("VACUUM INTO ?1", [remote.to_string_lossy().as_ref()])
.unwrap();
time("merge_remote_catalog (self)", 5, || {
catalog.merge_remote_catalog(&remote).unwrap();
});
let _ = std::fs::remove_file(&scratch);
let _ = std::fs::remove_file(&remote);
if let Some(peer) = &peer {
// A copy, so nothing the merge does to its input reaches the file
// the caller named.
std::fs::copy(peer, &remote).unwrap();
let mut first = None;
time("merge_remote_catalog (--remote)", 5, || {
let report = catalog.merge_remote_catalog(&remote).unwrap();
first.get_or_insert(report);
});
println!(" first pass: {first:?}");
let _ = std::fs::remove_file(&remote);
}
// The face half of a sync pass, against a copy of the face store: both
// directions in the steady state, where nothing is new either way.
if let Some(faces) = args.get(1).map(PathBuf::from) {
let model = "scrfd_10g+w600k_mbf";
let mut store = dr_catalog::FaceShardStore::open(&faces).unwrap();
println!(
" first export sent {}, first import adopted {}",
dr_catalog::face_shard::export_to_shards(conn, &mut store, model).unwrap(),
dr_catalog::face_shard::import_from_shards(conn, &store, model).unwrap()
);
time("face_shard::export_to_shards (steady)", 5, || {
dr_catalog::face_shard::export_to_shards(conn, &mut store, model).unwrap();
});
time("face_shard::import_from_shards (steady)", 5, || {
dr_catalog::face_shard::import_from_shards(conn, &store, model).unwrap();
});
}
}
/// What one click in the library reads: a rating or label keystroke
/// refreshes the chips, a selection change redraws the keyword panel, and
/// every scroll reload counts the grid.
fn interactive(conn: &rusqlite::Connection) {
println!(
" rating_histogram {:?}, label_histogram {:?}, local originals {}, grid {} / rated {}",
rating::rating_histogram(conn).unwrap(),
rating::label_histogram(conn).unwrap(),
local_original_count(conn),
grid_count(conn, ""),
grid_count(conn, RATED_AT_LEAST_ONE),
);
let words = keywords::list(conn).unwrap();
println!(
" keywords::list {} terms, digest {:016x}",
words.len(),
digest(&format!("{words:?}"))
);
// A selection the size of a grid window, from the start of the library.
let selection: Vec<dr_types::ImageId> = conn
.prepare("SELECT id FROM images ORDER BY id LIMIT 120")
.unwrap()
.query_map([], |r| Ok(dr_types::ImageId(r.get::<_, i64>(0)? as u64)))
.unwrap()
.collect::<Result<_, _>>()
.unwrap();
println!(
" keywords::for_images digest {:016x}",
digest(&format!(
"{:?}",
keywords::for_images(conn, &selection).unwrap()
))
);
time("rating::rating_histogram", 50, || {
rating::rating_histogram(conn).unwrap();
});
time("library::local_original_count", 50, || {
local_original_count(conn);
});
time("rating::label_histogram", 50, || {
rating::label_histogram(conn).unwrap();
});
time("keywords::list", 50, || {
keywords::list(conn).unwrap();
});
time("keywords::for_images (120)", 50, || {
keywords::for_images(conn, &selection).unwrap();
});
time("grid count", 50, || {
grid_count(conn, "");
});
time("grid count, rated >= 1", 50, || {
grid_count(conn, RATED_AT_LEAST_ONE);
});
}
/// `dr_ui::library::local_original_count`, spelled as it is there.
fn local_original_count(conn: &rusqlite::Connection) -> i64 {
conn.query_row(
"SELECT count(*) FROM images i
WHERE i.shadowed_by IS NULL AND i.trashed_at IS NULL
AND i.id IN (SELECT ic.image_id FROM image_cache ic
WHERE ic.tier_actual >= 2)",
[],
|r| r.get(0),
)
.unwrap()
}
/// `RatingFilter::sql` for one star and up.
const RATED_AT_LEAST_ONE: &str = " AND coalesce((SELECT dv.rating FROM versions dv
WHERE dv.image_id = i.id AND dv.is_default = 1
LIMIT 1), 0) >= 1";
/// `dr_ui::library::total_images_filtered`, spelled as it is there.
fn grid_count(conn: &rusqlite::Connection, rated: &str) -> i64 {
let visible = "i.shadowed_by IS NULL AND i.trashed_at IS NULL";
let hidden = dr_catalog::bursts::collapsed_away_frames("i");
conn.query_row(
&format!(
"SELECT (SELECT count(*) FROM images i WHERE {visible}{rated})
- (SELECT count(*) FROM {hidden} AND {visible}{rated})"
),
[],
|r| r.get(0),
)
.unwrap()
}
/// FNV-1a, to print a long answer as something two runs can compare.
fn digest(s: &str) -> u64 {
s.bytes().fold(0xcbf29ce484222325, |h, b| {
(h ^ u64::from(b)).wrapping_mul(0x100000001b3)
})
}
/// Run `f` a few times and print the best wall-clock, the median, and the
/// best CPU time — the figure to compare across runs on a busy machine.
fn time(label: &str, runs: usize, mut f: impl FnMut()) {
let mut wall: Vec<Duration> = Vec::with_capacity(runs);
let mut cpu: Vec<Duration> = Vec::with_capacity(runs);
for _ in 0..runs {
let c = cpu_now();
let t = Instant::now();
f();
wall.push(t.elapsed());
cpu.push(cpu_now().saturating_sub(c));
}
wall.sort();
cpu.sort();
println!(
"{label:42} best {:8.2} ms median {:8.2} ms cpu {:8.2} ms",
wall[0].as_secs_f64() * 1e3,
wall[runs / 2].as_secs_f64() * 1e3,
cpu[0].as_secs_f64() * 1e3
);
}
/// This thread's time on a CPU so far, from `/proc/self/schedstat`; zero where
/// the file is missing, which only makes the CPU column useless.
fn cpu_now() -> Duration {
std::fs::read_to_string("/proc/self/schedstat")
.ok()
.and_then(|s| s.split_whitespace().next()?.parse::<u64>().ok())
.map(Duration::from_nanos)
.unwrap_or_default()
}
-141
View File
@@ -1,141 +0,0 @@
//! Run the people and face deduplication (#78) on a copy of a real catalog.
//!
//! cargo run --release -p dr-catalog --example dedup_people -- COPY.sqlite [--peer PEER_COPY.sqlite]
//!
//! It writes: run it against a *copy* (`sqlite3 catalog.sqlite ".backup
//! copy.sqlite"`), never the library's own file. Prints the live people and
//! faces before and after, what the first run merged and kept apart, and
//! how long the first and a second run took -- the second is the cost the
//! job adds to every sync once a catalog is clean.
//!
//! `--peer` then plays a sync round trip with another device's catalog (a
//! copy of the server snapshot, which it also writes): the peer merges this
//! one as the previous release would, with no job after it, then this one
//! merges the peer back through `sync::merge_remote`, twice. The named
//! people each side lists are printed after each step; they should agree.
use std::path::PathBuf;
use std::time::Instant;
use dr_catalog::{dedup_people, merge, schema, sync};
use rusqlite::Connection;
fn open(path: &std::path::Path) -> Connection {
let conn = Connection::open(path).expect("open the catalog copy");
schema::configure(&conn).expect("configure");
schema::migrate(&conn).expect("migrate");
conn
}
/// The named people a device lists, as `name (uuid prefix)`, sorted.
fn named(conn: &Connection) -> Vec<String> {
let mut v: Vec<String> = conn
.prepare(
"SELECT name, substr(uuid, 1, 8) FROM people
WHERE merged_into IS NULL AND trim(name) <> ''",
)
.unwrap()
.query_map([], |r| {
Ok(format!(
"{} ({})",
r.get::<_, String>(0)?,
r.get::<_, String>(1)?
))
})
.unwrap()
.collect::<Result<_, _>>()
.unwrap();
v.sort();
v
}
fn main() {
let mut args: Vec<String> = std::env::args().skip(1).collect();
let peer = args.iter().position(|a| a == "--peer").map(|at| {
let p = PathBuf::from(&args[at + 1]);
args.drain(at..at + 2);
p
});
let Some(path) = args.first().map(PathBuf::from) else {
eprintln!("usage: dedup_people COPY.sqlite [--peer PEER_COPY.sqlite]");
std::process::exit(2);
};
let conn = open(&path);
let counts = |label: &str| {
let q = |sql: &str| -> i64 { conn.query_row(sql, [], |r| r.get(0)).unwrap() };
println!(
"{label}: {} people listed ({} named), {} redirects, {} faces, {} confirmed",
q("SELECT COUNT(*) FROM people WHERE merged_into IS NULL"),
q("SELECT COUNT(*) FROM people WHERE merged_into IS NULL AND trim(name) <> ''"),
q("SELECT COUNT(*) FROM people WHERE merged_into IS NOT NULL"),
q("SELECT COUNT(*) FROM faces"),
q("SELECT COUNT(*) FROM face_person WHERE confirmed = 1"),
);
};
counts("before");
for pass in ["first", "second", "third"] {
let started = Instant::now();
let report = dedup_people::run(&conn).expect("dedup");
let took = started.elapsed();
println!("{pass} run: {took:?}, changed: {}", report.changed());
if pass == "first" {
println!(" merged: {:?}", report.merged);
for k in &report.kept_apart {
println!(
" kept apart: {:?} ({}) from {}: {:?}",
k.name, k.uuid, k.survivor, k.why
);
}
println!(
" redirects followed {}, cycles broken {}, faces fused {}, faces confirmed apart {}",
report.redirects_followed,
report.cycles_broken,
report.faces_fused,
report.faces_confirmed_apart
);
}
}
counts("after");
let Some(peer_path) = peer else { return };
let peer = open(&peer_path);
let show = |step: &str| {
let (ours, theirs) = (named(&conn), named(&peer));
println!(
"{step}: this device lists {} named, the peer {}; {}",
ours.len(),
theirs.len(),
if ours == theirs {
"the same".to_string()
} else {
format!("differ:\n here {ours:?}\n peer {theirs:?}")
}
);
};
show("before the round trip");
for round in 1..=2 {
peer.execute(
"ATTACH DATABASE ?1 AS remote_cat",
[path.to_string_lossy().as_ref()],
)
.unwrap();
let theirs = merge::merge_all(&peer).expect("the peer's merge");
peer.execute("DETACH DATABASE remote_cat", []).unwrap();
println!(
"round {round}: the peer took {} people updated, {} inserted",
theirs.people_updated, theirs.people_inserted
);
show(&format!("round {round}, after the peer's merge"));
let started = Instant::now();
let ours = sync::merge_remote(&conn, &peer_path).expect("our merge");
println!(
"round {round}: merge_remote with the job took {:?}; {} people updated, {} inserted",
started.elapsed(),
ours.people_updated,
ours.people_inserted
);
show(&format!("round {round}, after ours"));
}
}
-420
View File
@@ -1,420 +0,0 @@
//! What the suggestion confidence would say about a real library.
//!
//! cargo run --release -p dr-catalog --example face_confidence -- CATALOG.sqlite [--full]
//!
//! Read-only: it writes nothing to the catalog, so it can be pointed at a copy
//! of a live library and re-run at will.
//!
//! # What it measures
//!
//! The user's own confirmations are the only ground truth a library has, so
//! the evaluation is leave-one-out over them: hide one confirmed face, ask the
//! scorer which of the confirmed identities it belongs to, and compare with
//! what the user said. Faces from the same photograph are excluded exactly as
//! the clusterer excludes them, so nothing is scored against a co-occurrence
//! that would never have been allowed to merge.
//!
//! Two numbers are compared on that task: the share (`dr_face::assign`) and the
//! mean-within-group figure it replaced. Accuracy says which one picks the
//! right person; the reliability table says whether the percentage the user is
//! shown means what it claims — which is the question FR-CULL-9 exists for.
use std::collections::HashMap;
use dr_catalog::faces::{self, PersonId};
use dr_catalog::Catalog;
const MODEL_ID: &str = "w600k_mbf";
const TOP: usize = 10;
struct Known {
image: u64,
person: PersonId,
embedding: Vec<f32>,
crop_px: f32,
}
/// What the catalog holds per face, decoded: photograph, vector, size,
/// quality.
type Decoded = (u64, Vec<f32>, f32, Option<f32>);
fn main() {
let args: Vec<String> = std::env::args().skip(1).collect();
let Some(path) = args.first() else {
eprintln!("usage: face_confidence CATALOG.sqlite [--full]");
std::process::exit(2);
};
let catalog = Catalog::open(std::path::Path::new(path)).expect("open catalog");
let conn = catalog.connection();
let cal = match faces::calibration(conn, MODEL_ID) {
Ok(Some((c, _))) => c,
_ => dr_face::Calibration::default(),
};
println!(
"calibration: a={:.2} b={:.2} w_size={:.3} valid={} (P=0.5 at cosine {:.3})",
cal.a,
cal.b,
cal.w_size,
cal.valid,
cal.boundary_at(0.5, 150.0, 0.0)
);
let model = dr_face::ModelId::new(MODEL_ID.to_string());
let stored = faces::embeddings(conn, MODEL_ID).expect("embeddings");
let mut embedding_of: HashMap<faces::FaceId, Decoded> = HashMap::new();
for f in stored {
if let Some(e) = dr_face::Embedding::from_f16_bytes(model.clone(), &f.embedding) {
embedding_of.insert(f.face, (f.image.0, e.v.to_vec(), f.crop_px, f.quality));
}
}
println!("faces with embeddings: {}", embedding_of.len());
// The ground truth: every confirmed face, under the person the user put it
// on. Identities with a single confirmation are dropped — leaving one out
// leaves that identity with no evidence at all, so they measure nothing.
let people = faces::people(conn).expect("people");
let mut known: Vec<Known> = Vec::new();
let mut identities = 0usize;
for p in &people {
if p.confirmed_faces < 2 {
continue;
}
let mut mine = Vec::new();
for f in faces::for_person(conn, p.id, false).expect("faces") {
if !f.confirmed {
continue;
}
if let Some((image, embedding, crop_px, _)) = embedding_of.get(&f.id) {
mine.push(Known {
image: *image,
person: p.id,
embedding: embedding.clone(),
crop_px: *crop_px,
});
}
}
if mine.len() >= 2 {
identities += 1;
known.extend(mine);
}
}
println!(
"ground truth: {} confirmed faces across {identities} identities\n",
known.len()
);
if known.len() < 2 {
println!("not enough confirmations to evaluate.");
return;
}
let mut share_right = 0usize;
let mut mean_right = 0usize;
// (share of the winner, was the winner correct)
let mut reliability: Vec<(f32, bool)> = Vec::with_capacity(known.len());
// What each scorer would have *displayed* for the correct answer.
let mut shown_share = Vec::with_capacity(known.len());
let mut shown_mean = Vec::with_capacity(known.len());
for (i, me) in known.iter().enumerate() {
let mut per_person: HashMap<PersonId, Vec<f32>> = HashMap::new();
// The mean baseline is the old code's, which had no floor: it averaged
// over every member of the group.
let mut all_person: HashMap<PersonId, Vec<f32>> = HashMap::new();
for (j, them) in known.iter().enumerate() {
if i == j || me.image == them.image {
continue;
}
let cos: f32 = me
.embedding
.iter()
.zip(&them.embedding)
.map(|(a, b)| a * b)
.sum();
// The floor the real scorer sees: `cluster_scored` scans at
// `RIVAL_FLOOR` and `identity_shares` never learns about a pair
// below it. Summing the near-orthogonal ones here instead of
// dropping them is not a stricter test, it is a different
// function — fifty identities contributing their *upper tail* of
// noise outweigh one contributing a real match.
let probability = cal.probability(cos, me.crop_px.min(them.crop_px), 0.0);
if probability >= dr_face::RIVAL_FLOOR {
per_person.entry(them.person).or_default().push(probability);
}
all_person.entry(them.person).or_default().push(probability);
}
// The share: sum of the best TOP matches per identity, normalised.
// Evidence, and the coherence that goes with it: the sum of the best
// TOP matches, and their mean. dr_face::assign shows the product of
// that mean and the identity's share of the total.
let mut evidence: Vec<(PersonId, f32, f32)> = per_person
.iter()
.map(|(&p, probabilities)| {
let mut v = probabilities.clone();
v.sort_by(|a, b| b.total_cmp(a));
let counted = v.len().min(TOP);
let sum = v.iter().take(TOP).sum::<f32>();
(p, sum, sum / counted as f32)
})
.collect();
let total: f32 = evidence.iter().map(|(_, s, _)| *s).sum();
evidence.sort_by(|a, b| b.1.total_cmp(&a.1));
// The number it replaced: the mean over every member of the identity.
let mut means: Vec<(PersonId, f32)> = all_person
.iter()
.map(|(&p, probabilities)| {
(
p,
probabilities.iter().sum::<f32>() / probabilities.len() as f32,
)
})
.collect();
means.sort_by(|a, b| b.1.total_cmp(&a.1));
if let (Some(&(winner, score, coherence)), true) = (evidence.first(), total > 0.0) {
let correct = winner == me.person;
share_right += correct as usize;
// What the screen would say about the identity it picked.
reliability.push((coherence * score / total, correct));
let ours = evidence
.iter()
.find(|(p, _, _)| *p == me.person)
.map(|(_, s, c)| c * s / total)
.unwrap_or(0.0);
shown_share.push(ours);
}
if let Some(&(winner, _)) = means.first() {
mean_right += (winner == me.person) as usize;
shown_mean.push(
means
.iter()
.find(|(p, _)| *p == me.person)
.map(|(_, s)| *s)
.unwrap_or(0.0),
);
}
}
let n = known.len() as f64;
println!("which identity does this face belong to? (leave-one-out, top-1)");
println!(
" share of evidence {:>6.2}% ({share_right}/{})",
100.0 * share_right as f64 / n,
known.len()
);
println!(
" mean within group {:>6.2}% ({mean_right}/{})\n",
100.0 * mean_right as f64 / n,
known.len()
);
println!("what the screen would show for the answer the user gave:");
band(" share ", &shown_share);
band(" mean ", &shown_mean);
println!("\nreliability of the share — is a stated {{n}}% right {{n}}% of the time?");
println!(
" {:>12} {:>7} {:>9} {:>8}",
"stated", "faces", "correct", "gap"
);
for (lo, hi) in [
(0.0, 0.5),
(0.5, 0.6),
(0.6, 0.7),
(0.7, 0.8),
(0.8, 0.9),
(0.9, 0.95),
(0.95, 1.001),
] {
let bucket: Vec<bool> = reliability
.iter()
.filter(|(s, _)| *s >= lo && *s < hi)
.map(|(_, c)| *c)
.collect();
if bucket.is_empty() {
continue;
}
let observed = bucket.iter().filter(|c| **c).count() as f64 / bucket.len() as f64;
let stated = reliability
.iter()
.filter(|(s, _)| *s >= lo && *s < hi)
.map(|(s, _)| *s as f64)
.sum::<f64>()
/ bucket.len() as f64;
println!(
" {:>5.0}–{:>3.0}% {:>9} {:>8.1}% {:>+7.1}",
lo * 100.0,
hi.min(1.0) * 100.0,
bucket.len(),
100.0 * observed,
100.0 * (observed - stated)
);
}
if args.iter().any(|a| a == "--full") {
// The confirmations go in as anchors, exactly as `recluster` sends
// them: they are what makes a group a named identity, and therefore
// what makes it a rival.
let mut confirmed = HashMap::new();
for p in &people {
for f in faces::for_person(conn, p.id, false).expect("faces") {
if f.confirmed {
confirmed.insert(f.id, p.id.0);
}
}
}
full_library(&embedding_of, &confirmed, &cal);
}
}
/// Where a set of confidences actually falls.
fn band(label: &str, v: &[f32]) {
if v.is_empty() {
return;
}
let mut s = v.to_vec();
s.sort_by(|a, b| a.total_cmp(b));
let pct = |q: f64| s[((s.len() - 1) as f64 * q) as usize];
let mean = s.iter().sum::<f32>() / s.len() as f32;
println!(
"{label} median {:>5.1}% mean {:>5.1}% p10 {:>5.1}% p90 {:>5.1}% under 50%: {:>5.1}%",
100.0 * pct(0.5),
100.0 * mean,
100.0 * pct(0.10),
100.0 * pct(0.90),
100.0 * s.iter().filter(|x| **x < 0.5).count() as f32 / s.len() as f32
);
}
/// The whole library through the real clusterer, for the numbers it would
/// actually write.
fn full_library(
embedding_of: &HashMap<faces::FaceId, Decoded>,
confirmed: &HashMap<faces::FaceId, u64>,
cal: &dr_face::Calibration,
) {
let mut candidates: Vec<dr_face::Candidate> = embedding_of
.iter()
.map(
|(id, (image, embedding, crop_px, quality))| dr_face::Candidate {
face: id.0,
image: *image,
embedding: embedding.clone(),
crop_px: *crop_px,
quality: *quality,
confirmed_person: confirmed.get(id).copied(),
},
)
.collect();
candidates.sort_by_key(|c| c.face);
println!("\nthe whole library, at the default merge probability:");
// The three phases, separately, because "a regroup takes n seconds" does
// not tell anyone which half to optimise — and the answer differs between
// a desktop and a tablet (docs/dev/faces.md §9).
{
let dim = candidates.first().map(|c| c.embedding.len()).unwrap_or(0);
let flat: Vec<f32> = candidates
.iter()
.flat_map(|c| c.embedding.clone())
.collect();
let crop_px: Vec<f32> = candidates.iter().map(|c| c.crop_px).collect();
let images: Vec<u64> = candidates.iter().map(|c| c.image).collect();
let gallery: Vec<bool> = candidates.iter().map(|c| c.in_gallery()).collect();
let view = dr_face::neighbours::Faces {
embeddings: &flat,
dim,
crop_px: &crop_px,
images: &images,
gallery: &gallery,
};
let t = std::time::Instant::now();
let evidence = dr_face::neighbours::above_threshold(&view, cal, dr_face::RIVAL_FLOOR);
let scan = t.elapsed().as_secs_f64();
// `cluster` runs its own scan at the merge threshold, so the
// agglomeration is what is left after taking one scan off the total.
let t = std::time::Instant::now();
let clusters = dr_face::cluster(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY);
let agglomerate = t.elapsed().as_secs_f64() - scan;
let t = std::time::Instant::now();
let _ = dr_face::identity_shares(&gallery, &clusters, &evidence, dr_face::TOP_MATCHES);
println!(
" scan {scan:.2}s ({} evidence pairs) · agglomerate {agglomerate:.2}s · score {:.2}s",
evidence.len(),
t.elapsed().as_secs_f64()
);
}
let start = std::time::Instant::now();
let grouping = dr_face::cluster_scored(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY);
let real: Vec<_> = grouping
.clusters
.iter()
.filter(|c| c.members.len() >= 2)
.collect();
let grouped: usize = real.iter().map(|c| c.members.len()).sum();
println!(
" {} face(s) → {} group(s) of two or more, holding {grouped} faces ({:.0}%), in {:.1}s",
candidates.len(),
real.len(),
100.0 * grouped as f64 / candidates.len() as f64,
start.elapsed().as_secs_f64()
);
let named: Vec<_> = real.iter().filter(|c| c.person.is_some()).collect();
println!(
" {} of those group(s) carry a confirmation, holding {} faces",
named.len(),
named.iter().map(|c| c.members.len()).sum::<usize>()
);
let shown: Vec<f32> = real
.iter()
.flat_map(|c| c.members.iter().map(|&m| grouping.confidence[m]))
.collect();
band(" new, all groups ", &shown);
let onto_people: Vec<f32> = named
.iter()
.flat_map(|c| c.members.iter().map(|&m| grouping.confidence[m]))
.collect();
band(" new, onto a person", &onto_people);
// The number the old code would have written for the same grouping.
let means: Vec<f32> = real
.iter()
.flat_map(|c| {
c.members.iter().map(|&m| {
let me = &candidates[m];
let mut sum = 0.0;
let mut n = 0.0;
for &other in &c.members {
if other == m {
continue;
}
let them = &candidates[other];
let cos: f32 = me
.embedding
.iter()
.zip(&them.embedding)
.map(|(a, b)| a * b)
.sum();
sum += cal.probability(cos, me.crop_px.min(them.crop_px), 0.0);
n += 1.0;
}
if n == 0.0 {
1.0
} else {
sum / n
}
})
})
.collect();
band(" old, all groups ", &means);
}
-111
View File
@@ -1,111 +0,0 @@
//! Scan a real folder on this machine into a catalog, and say what it cost.
//!
//! cargo run -p dr-catalog --example scan_local -- ~/Pictures [catalog.sqlite]
//!
//! **Run it twice.** The first run is a full walk; the second is the one worth
//! watching, because on an unchanged library it should list no directories at
//! all and take a fraction of the time. That difference is NFR-P1, and a
//! synthetic test cannot show it at the scale a real library does — 121,785
//! files in a synced folder is a different question from twenty in a temporary
//! directory.
//!
//! Writes only to the catalog file, which defaults to a fixed path in the
//! system temporary directory so a second run has something to compare
//! against. Nothing in the scanned folder is touched.
use std::path::PathBuf;
use dr_catalog::walk::{ensure_root, scan_root, RootKind};
use dr_catalog::Catalog;
use dr_plat::LocalStorage;
use dr_types::FormatFilter;
fn main() {
env_logger::init();
let mut args = std::env::args().skip(1);
let Some(dir) = args.next().map(PathBuf::from) else {
eprintln!("usage: scan_local <directory> [catalog.sqlite]");
std::process::exit(2);
};
let catalog_path = args
.next()
.map(PathBuf::from)
.unwrap_or_else(|| std::env::temp_dir().join("darkroom-scan-local.sqlite"));
let catalog = match Catalog::open(&catalog_path) {
Ok(c) => c,
Err(e) => {
eprintln!("cannot open {}: {e}", catalog_path.display());
std::process::exit(1);
}
};
println!("catalog: {}", catalog_path.display());
// The label is how the grant is spelled, and the only place a path is
// written down. Everything after this line addresses files by `RootId`.
let label = dir.display().to_string();
let root = match ensure_root(catalog.connection(), RootKind::Local, &label) {
Ok(r) => r,
Err(e) => {
eprintln!("cannot record the root: {e}");
std::process::exit(1);
}
};
let storage = LocalStorage::with_root(root, &dir);
let now = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs() as i64)
.unwrap_or(0);
let started = std::time::Instant::now();
let report = match scan_root(
catalog.connection(),
&storage,
root,
&FormatFilter::all(),
now,
|| false,
|p| {
// One line per hundred directories: enough to show it is alive on a
// large library, not enough to be the thing that slows it down.
let visited = p.directories_listed + p.directories_pruned;
if visited % 100 == 0 {
println!(
" … {visited} directories ({} pruned), {} images",
p.directories_pruned, p.images_found
);
}
},
) {
Ok(r) => r,
Err(e) => {
eprintln!("scan failed: {e}");
std::process::exit(1);
}
};
let elapsed = started.elapsed();
let total: i64 = catalog
.connection()
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap_or(-1);
println!("\noutcome: {:?}", report.outcome);
println!(
"directories: {} listed, {} pruned",
report.progress.directories_listed, report.progress.directories_pruned
);
println!(
"images: {} new, {} changed, {} unchanged, {} removed",
report.inserted, report.updated, report.unchanged, report.images_removed
);
println!("folders: {} removed", report.folders_removed);
println!("catalogued: {total} in total");
println!("took: {:.2?}", elapsed);
if report.progress.directories_listed == 0 && report.progress.directories_pruned > 0 {
println!("\nnothing had changed: every folder was proven unchanged by one probe");
}
}
-111
View File
@@ -1,111 +0,0 @@
//! What a second device ends up with after adopting this library.
//!
//! Stands up an empty catalog, gives it the images the real one has, adopts the
//! face shards into it exactly as a sync would, merges the real catalog in as a
//! remote — and then counts. The point is to answer "why does the tablet show
//! fewer faces for this person" without needing the tablet.
//!
//! cargo run -p dr-catalog --example sync_probe -- CATALOG.sqlite FACES_DIR
use std::path::PathBuf;
use dr_catalog::face_shard::{self, FaceShardStore};
use dr_catalog::Catalog;
const MODEL: &str = "w600k_mbf";
fn main() {
let args: Vec<String> = std::env::args().skip(1).collect();
if args.len() < 2 {
eprintln!("usage: sync_probe CATALOG.sqlite FACES_DIR");
std::process::exit(2);
}
let source = PathBuf::from(&args[0]);
let faces_dir = PathBuf::from(&args[1]);
let dir = std::env::temp_dir().join(format!("dr-sync-probe-{}", std::process::id()));
let _ = std::fs::remove_dir_all(&dir);
std::fs::create_dir_all(&dir).unwrap();
let dest = dir.join("catalog.sqlite");
let far = Catalog::open(&dest).expect("fresh catalog");
let conn = far.connection();
// The images a scan would have found. Nothing else: no faces, no people.
conn.execute(
"ATTACH DATABASE ?1 AS src",
[source.to_string_lossy().as_ref()],
)
.unwrap();
// Foreign keys off for the copy: `images` carries self-references
// (`shadowed_by`) that are only consistent once every row is in, and this
// is a bulk clone rather than an edit.
conn.execute_batch(
"PRAGMA foreign_keys = OFF;
INSERT INTO roots SELECT * FROM src.roots;
INSERT INTO images SELECT * FROM src.images;
INSERT INTO remote SELECT * FROM src.remote;
PRAGMA foreign_keys = ON;",
)
.unwrap();
let images: i64 = conn
.query_row("SELECT COUNT(*) FROM images", [], |r| r.get(0))
.unwrap();
conn.execute_batch("DETACH DATABASE src").unwrap();
println!("second device starts with {images} image(s), no faces");
// Adopt every shard, which is what a completed face sync leaves behind.
let store = FaceShardStore::open(&faces_dir).expect("shard store");
let adopted = face_shard::import_from_shards(conn, &store, MODEL).expect("import");
let faces: i64 = conn
.query_row("SELECT COUNT(*) FROM faces", [], |r| r.get(0))
.unwrap();
println!("adopted {adopted} image(s) from the shards -> {faces} face(s)");
// Then the catalog merge, which is where people and their judgements come.
let report = dr_catalog::sync::merge_remote(conn, &source).expect("merge");
println!(
"merge: {} people in, {} updated, {} kept local, {} face(s) assigned, \
{} kept local, {} rejection(s)",
report.people_inserted,
report.people_updated,
report.people_kept_local,
report.faces_assigned,
report.faces_kept_local,
report.faces_rejected,
);
// Per person, against what the source holds.
conn.execute(
"ATTACH DATABASE ?1 AS src",
[source.to_string_lossy().as_ref()],
)
.unwrap();
let mut q = conn
.prepare(
"SELECT p.name,
(SELECT COUNT(*) FROM src.face_person sfp
JOIN src.people sp ON sp.id = sfp.person_id
WHERE sp.uuid = p.uuid) AS there,
(SELECT COUNT(*) FROM face_person fp WHERE fp.person_id = p.id) AS here
FROM people p
WHERE p.name != ''
ORDER BY there DESC LIMIT 12",
)
.unwrap();
println!("\n{:<24} {:>8} {:>8}", "person", "source", "here");
let rows = q
.query_map([], |r| {
Ok((
r.get::<_, String>(0)?,
r.get::<_, i64>(1)?,
r.get::<_, i64>(2)?,
))
})
.unwrap();
for row in rows.flatten() {
println!("{:<24} {:>8} {:>8}", row.0, row.1, row.2);
}
println!("\nprobe catalog left at {}", dest.display());
}
-516
View File
@@ -1,516 +0,0 @@
//! TRACES: FR-EXP-10 | FR-EXP-6 | FR-CAT-7
//! Albums: named export folders, and which photographs went into each.
//!
//! An album is where finished pictures go — a folder of JPEGs somebody else
//! looks at — as opposed to a collection, which is a set of originals the
//! photographer works on. The folder holds only the exported files. What the
//! catalog adds is the link back: each export is recorded against the image
//! it was rendered from, so opening an album in the library shows the RAWs
//! behind its JPEGs, and re-exporting after an edit is one selection away.
//!
//! # Where the folder is, and why that is two tables
//!
//! An album's folder is either on the library's server or on this device.
//!
//! A **server folder** is one path on the account, the same from every device
//! signed in to it, so it lives on the album row and syncs with it.
//!
//! A **local folder** — a filesystem path on a desktop, a Storage Access
//! Framework tree on Android — means nothing on any other device. It lives in
//! `album_folders`, which the merge never reads and the upload snapshot drops
//! ([`crate::sync::snapshot_for_upload`]). An album made on the desktop with a
//! local folder therefore reaches the tablet as an album with no folder there
//! yet, which is true, and which the tablet can fix by choosing one.
//!
//! # Created on first use, not by a migration
//!
//! A new schema version makes every older build refuse this catalog's
//! snapshot at sync (`crate::sync::remote_is_mergeable`), so the tablet would
//! stop merging collections, keywords and people until it was updated — for
//! a feature it does not have. The tables are created by [`ensure_tables`]
//! instead, the way `dedup_probes` is; an older build that meets them ignores
//! them, and its merge keeps working.
//!
//! # Sync
//!
//! Albums merge by uuid and revision with tombstones, and their exports as a
//! set union keyed on the image's server file id — the rules
//! [`crate::merge`] applies to collections, for the same reasons.
use rusqlite::{Connection, OptionalExtension};
use dr_types::ImageId;
use crate::error::CatalogError;
/// Identifies an album within one catalog. Local, like every integer id here;
/// the uuid is what crosses devices.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
pub struct AlbumId(pub u64);
/// Where an album's files go.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Place {
/// A folder on the library's server, relative to the account root, with no
/// leading slash. The same on every device.
Server(String),
/// A folder on this device: a filesystem path, or on Android a SAF tree
/// URI. Never synced.
Local(String),
}
/// One album, as the sidebar and the export sheet show it.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Album {
pub id: AlbumId,
pub uuid: String,
pub name: String,
/// Where exports go from this device, or `None` for an album whose folder
/// is local to another device and has not been chosen here.
pub place: Option<Place>,
/// Distinct photographs exported into it — what the grid shows when the
/// album is opened.
pub sources: usize,
}
/// Create the album tables if this catalog does not have them yet.
///
/// Cheap when they exist: `IF NOT EXISTS` is answered from the schema, and
/// every function below calls this first so no caller has to remember to.
pub fn ensure_tables(conn: &Connection) -> Result<(), CatalogError> {
conn.execute_batch(
"CREATE TABLE IF NOT EXISTS albums (
id INTEGER PRIMARY KEY,
-- The merge identity; the integer id is local.
uuid TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
-- A folder on the server, relative to the account root. NULL for
-- an album whose folder is local to some device.
server_path TEXT,
created INTEGER NOT NULL,
revision INTEGER NOT NULL DEFAULT 1,
modified INTEGER NOT NULL,
deleted INTEGER NOT NULL DEFAULT 0
);
-- One row per file written into an album. Keyed on the file, not the
-- image: a photograph exported twice — two crops, or once before an
-- edit and once after — is two files in the folder and two rows here.
CREATE TABLE IF NOT EXISTS album_exports (
album_id INTEGER NOT NULL REFERENCES albums(id) ON DELETE CASCADE,
file_name TEXT NOT NULL,
image_id INTEGER NOT NULL REFERENCES images(id) ON DELETE CASCADE,
exported_at INTEGER NOT NULL,
PRIMARY KEY (album_id, file_name)
);
CREATE INDEX IF NOT EXISTS album_exports_image ON album_exports(image_id);
-- This device's folder for an album. Never merged, never uploaded.
CREATE TABLE IF NOT EXISTS album_folders (
album_id INTEGER PRIMARY KEY REFERENCES albums(id) ON DELETE CASCADE,
folder TEXT NOT NULL
);",
)?;
Ok(())
}
/// Make an album.
///
/// The name is trimmed and must not be empty; two albums may share one, as
/// two collections may, because the uuid is the identity and refusing a
/// duplicate name here would refuse it on one device and not another.
pub fn create(conn: &Connection, name: &str, place: &Place) -> Result<AlbumId, CatalogError> {
ensure_tables(conn)?;
let name = name.trim();
if name.is_empty() {
return Err(CatalogError::EmptyName);
}
let now = now_secs();
let tx = conn.unchecked_transaction()?;
tx.execute(
"INSERT INTO albums(uuid, name, server_path, created, revision, modified)
VALUES (?1, ?2, ?3, ?4, 1, ?4)",
rusqlite::params![
crate::collections::new_uuid(),
name,
server_path(place),
now
],
)?;
let id = AlbumId(tx.last_insert_rowid() as u64);
if let Place::Local(folder) = place {
tx.execute(
"INSERT INTO album_folders(album_id, folder) VALUES (?1, ?2)",
rusqlite::params![id.0 as i64, folder],
)?;
}
tx.commit()?;
Ok(id)
}
/// Rename an album. The folder keeps its name: the album is what the
/// photographer calls it, the folder is what is already out there.
pub fn rename(conn: &Connection, id: AlbumId, name: &str) -> Result<(), CatalogError> {
ensure_tables(conn)?;
let name = name.trim();
if name.is_empty() {
return Err(CatalogError::EmptyName);
}
let n = conn.execute(
"UPDATE albums SET name = ?2, revision = revision + 1, modified = ?3
WHERE id = ?1 AND deleted = 0",
rusqlite::params![id.0 as i64, name, now_secs()],
)?;
if n == 0 {
return Err(CatalogError::NoSuchAlbum(id.0));
}
Ok(())
}
/// Point an album at a different folder, from this device.
///
/// A server folder replaces the synced path, and bumps the revision so the
/// move reaches every device. A local folder is recorded for this device
/// only; it also clears a server path, because an album goes to one place and
/// the photographer has just said which.
pub fn set_place(conn: &Connection, id: AlbumId, place: &Place) -> Result<(), CatalogError> {
ensure_tables(conn)?;
let tx = conn.unchecked_transaction()?;
let n = tx.execute(
"UPDATE albums SET server_path = ?2, revision = revision + 1, modified = ?3
WHERE id = ?1 AND deleted = 0",
rusqlite::params![id.0 as i64, server_path(place), now_secs()],
)?;
if n == 0 {
return Err(CatalogError::NoSuchAlbum(id.0));
}
match place {
Place::Local(folder) => tx.execute(
"INSERT INTO album_folders(album_id, folder) VALUES (?1, ?2)
ON CONFLICT(album_id) DO UPDATE SET folder = excluded.folder",
rusqlite::params![id.0 as i64, folder],
)?,
Place::Server(_) => tx.execute(
"DELETE FROM album_folders WHERE album_id = ?1",
[id.0 as i64],
)?,
};
tx.commit()?;
Ok(())
}
/// Delete an album, leaving a tombstone. The files in its folder are not
/// touched: they are finished work somebody may already have been sent a
/// link to, and the album was only ever this catalog's note of them.
pub fn delete(conn: &Connection, id: AlbumId) -> Result<(), CatalogError> {
ensure_tables(conn)?;
let tx = conn.unchecked_transaction()?;
let n = tx.execute(
"UPDATE albums SET deleted = 1, revision = revision + 1, modified = ?2
WHERE id = ?1 AND deleted = 0",
rusqlite::params![id.0 as i64, now_secs()],
)?;
if n == 0 {
return Err(CatalogError::NoSuchAlbum(id.0));
}
tx.execute(
"DELETE FROM album_exports WHERE album_id = ?1",
[id.0 as i64],
)?;
tx.execute(
"DELETE FROM album_folders WHERE album_id = ?1",
[id.0 as i64],
)?;
tx.commit()?;
Ok(())
}
/// Every live album, by name, with how many photographs each holds.
///
/// One statement: the counts are aggregated from `album_exports` first and
/// joined to the (few) albums, not counted per row.
pub fn list(conn: &Connection) -> Result<Vec<Album>, CatalogError> {
ensure_tables(conn)?;
let mut stmt = conn.prepare(
"SELECT a.id, a.uuid, a.name, a.server_path, f.folder, coalesce(e.n, 0)
FROM albums a
LEFT JOIN album_folders f ON f.album_id = a.id
LEFT JOIN (SELECT album_id, count(DISTINCT image_id) AS n
FROM album_exports GROUP BY album_id) e
ON e.album_id = a.id
WHERE a.deleted = 0
ORDER BY a.name COLLATE NOCASE, a.id",
)?;
let rows = stmt
.query_map([], album_from_row)?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// One album, or `None` if it is gone.
pub fn get(conn: &Connection, id: AlbumId) -> Result<Option<Album>, CatalogError> {
ensure_tables(conn)?;
Ok(conn
.query_row(
"SELECT a.id, a.uuid, a.name, a.server_path, f.folder,
(SELECT count(DISTINCT image_id) FROM album_exports WHERE album_id = a.id)
FROM albums a
LEFT JOIN album_folders f ON f.album_id = a.id
WHERE a.id = ?1 AND a.deleted = 0",
[id.0 as i64],
album_from_row,
)
.optional()?)
}
/// The album with this uuid, if this catalog holds it live.
pub fn id_for_uuid(conn: &Connection, uuid: &str) -> Result<Option<AlbumId>, CatalogError> {
ensure_tables(conn)?;
Ok(conn
.query_row(
"SELECT id FROM albums WHERE uuid = ?1 AND deleted = 0",
[uuid],
|r| r.get::<_, i64>(0),
)
.optional()?
.map(|id| AlbumId(id as u64)))
}
/// Record the files one export wrote into an album, and which image each
/// came from. One transaction for the batch, however many files it placed.
///
/// A file name already recorded is re-pointed at the image that wrote it
/// last: an export that overwrote `IMG_0001.jpg` replaced the picture in the
/// folder, and the link must say what is there now.
pub fn record_exports(
conn: &Connection,
id: AlbumId,
files: &[(ImageId, String)],
) -> Result<(), CatalogError> {
ensure_tables(conn)?;
if files.is_empty() {
return Ok(());
}
let tx = conn.unchecked_transaction()?;
let now = now_secs();
{
let mut insert = tx.prepare(
"INSERT INTO album_exports(album_id, file_name, image_id, exported_at)
VALUES (?1, ?2, ?3, ?4)
ON CONFLICT(album_id, file_name) DO UPDATE SET
image_id = excluded.image_id, exported_at = excluded.exported_at",
)?;
for (image, name) in files {
insert.execute(rusqlite::params![id.0 as i64, name, image.0 as i64, now])?;
}
}
tx.commit()?;
Ok(())
}
/// The photographs behind an album's files, most recently exported first —
/// what the grid shows when the album is opened.
pub fn sources(conn: &Connection, id: AlbumId) -> Result<Vec<ImageId>, CatalogError> {
ensure_tables(conn)?;
let mut stmt = conn.prepare(
"SELECT image_id FROM album_exports
WHERE album_id = ?1
GROUP BY image_id
ORDER BY max(exported_at) DESC, image_id",
)?;
let rows = stmt
.query_map([id.0 as i64], |r| r.get::<_, i64>(0))?
.map(|r| r.map(|i| ImageId(i as u64)))
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// The names of the files an image left in an album — the "which JPEG is
/// this" half of the link.
pub fn files_of(
conn: &Connection,
id: AlbumId,
image: ImageId,
) -> Result<Vec<String>, CatalogError> {
ensure_tables(conn)?;
let mut stmt = conn.prepare(
"SELECT file_name FROM album_exports
WHERE album_id = ?1 AND image_id = ?2
ORDER BY exported_at DESC, file_name",
)?;
let rows = stmt
.query_map(rusqlite::params![id.0 as i64, image.0 as i64], |r| r.get(0))?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
fn album_from_row(r: &rusqlite::Row<'_>) -> rusqlite::Result<Album> {
let server: Option<String> = r.get(3)?;
let local: Option<String> = r.get(4)?;
Ok(Album {
id: AlbumId(r.get::<_, i64>(0)? as u64),
uuid: r.get(1)?,
name: r.get(2)?,
// A server path wins: `set_place` clears the local folder when it
// sets one, so both being present means a merge brought a server
// path in over a local choice — and the newer revision decided that.
place: server.map(Place::Server).or(local.map(Place::Local)),
sources: r.get::<_, i64>(5)? as usize,
})
}
fn server_path(place: &Place) -> Option<&str> {
match place {
Place::Server(p) => Some(p.trim_matches('/')),
Place::Local(_) => None,
}
}
fn now_secs() -> i64 {
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_secs() as i64)
.unwrap_or(0)
}
#[cfg(test)]
mod tests {
use super::*;
/// A catalog, and the connection to it. The `Catalog` has to outlive the
/// connection it hands out, so tests hold both.
fn catalog() -> crate::Catalog {
let cat = crate::Catalog::in_memory().unwrap();
cat.connection()
.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
cat
}
fn image(conn: &Connection, path: &str) -> ImageId {
conn.execute(
"INSERT INTO images(root_id, source_ref, added_at) VALUES (1, ?1, 0)",
[path],
)
.unwrap();
ImageId(conn.last_insert_rowid() as u64)
}
#[test]
fn an_album_lists_with_its_place_and_no_photographs() {
let cat = catalog();
let conn = cat.connection();
let web = create(conn, " Web ", &Place::Server("Shared/Web/".into())).unwrap();
let print = create(conn, "Print", &Place::Local("/mnt/print".into())).unwrap();
let all = list(conn).unwrap();
assert_eq!(all.len(), 2);
assert_eq!(all[0].id, print, "sorted by name");
assert_eq!(all[0].place, Some(Place::Local("/mnt/print".into())));
assert_eq!(all[1].id, web);
assert_eq!(all[1].name, "Web", "trimmed");
assert_eq!(all[1].place, Some(Place::Server("Shared/Web".into())));
assert_eq!(all[1].sources, 0);
}
#[test]
fn an_empty_name_is_refused() {
let cat = catalog();
let conn = cat.connection();
assert!(matches!(
create(conn, " ", &Place::Local("/x".into())),
Err(CatalogError::EmptyName)
));
}
#[test]
fn exports_link_files_back_to_their_images() {
let cat = catalog();
let conn = cat.connection();
let album = create(conn, "Web", &Place::Local("/out".into())).unwrap();
let a = image(conn, "a.cr3");
let b = image(conn, "b.cr3");
record_exports(
conn,
album,
&[
(a, "a.jpg".into()),
(a, "a (1).jpg".into()),
(b, "b.jpg".into()),
],
)
.unwrap();
let got = get(conn, album).unwrap().unwrap();
assert_eq!(got.sources, 2, "two photographs, three files");
let mut s = sources(conn, album).unwrap();
s.sort();
assert_eq!(s, vec![a, b]);
assert_eq!(files_of(conn, album, a).unwrap().len(), 2);
}
#[test]
fn an_overwritten_file_points_at_what_wrote_it_last() {
let cat = catalog();
let conn = cat.connection();
let album = create(conn, "Web", &Place::Local("/out".into())).unwrap();
let a = image(conn, "a.cr3");
let b = image(conn, "b.cr3");
record_exports(conn, album, &[(a, "x.jpg".into())]).unwrap();
record_exports(conn, album, &[(b, "x.jpg".into())]).unwrap();
assert_eq!(sources(conn, album).unwrap(), vec![b]);
}
#[test]
fn moving_to_the_server_forgets_the_local_folder() {
let cat = catalog();
let conn = cat.connection();
let album = create(conn, "Web", &Place::Local("/out".into())).unwrap();
set_place(conn, album, &Place::Server("Web".into())).unwrap();
assert_eq!(
get(conn, album).unwrap().unwrap().place,
Some(Place::Server("Web".into()))
);
set_place(conn, album, &Place::Local("/again".into())).unwrap();
assert_eq!(
get(conn, album).unwrap().unwrap().place,
Some(Place::Local("/again".into()))
);
}
#[test]
fn a_deleted_album_is_gone_and_its_uuid_no_longer_resolves() {
let cat = catalog();
let conn = cat.connection();
let album = create(conn, "Web", &Place::Local("/out".into())).unwrap();
let uuid = get(conn, album).unwrap().unwrap().uuid;
let a = image(conn, "a.cr3");
record_exports(conn, album, &[(a, "a.jpg".into())]).unwrap();
delete(conn, album).unwrap();
assert!(list(conn).unwrap().is_empty());
assert_eq!(id_for_uuid(conn, &uuid).unwrap(), None);
assert!(matches!(
rename(conn, album, "Again"),
Err(CatalogError::NoSuchAlbum(_))
));
}
#[test]
fn a_rename_bumps_the_revision_the_merge_compares() {
let cat = catalog();
let conn = cat.connection();
let album = create(conn, "Web", &Place::Local("/out".into())).unwrap();
rename(conn, album, "Website").unwrap();
let rev: i64 = conn
.query_row(
"SELECT revision FROM albums WHERE id = ?1",
[album.0 as i64],
|r| r.get(0),
)
.unwrap();
assert_eq!(rev, 2);
}
}
-417
View File
@@ -1,417 +0,0 @@
//! TRACES: NFR-P9
//! Which catalog files this process has already backfilled, and as of what.
//!
//! # Why this exists
//!
//! [`crate::schema::backfill`] used to run inside every [`crate::Catalog::open`],
//! and every worker thread opens its own connection. Landing on a photograph
//! in develop opened the catalog five times — the fetch of the original and a
//! cache check per prefetched neighbour — and each open paid the whole
//! backfill: an anti-join of every image against its versions, a pass over
//! every default version's uuid, the unpaired JPEGs and the keyword
//! vocabulary. On the reference library that was ~12 ms an open and ~60 ms of
//! CPU a landing, spent confirming that nothing had changed since the open
//! before.
//!
//! # What makes skipping it safe
//!
//! Everything the backfill repairs is a row some write *added*: an image
//! inserted by a scan or an import has no default version and may be the RAW
//! beside an unpaired JPEG; a version merged or restored from an older build
//! may carry a minted uuid; a keyword assignment merged from a remote may name
//! a word with no term. So the question "is any work owed?" is answered by
//! whether those tables have gained rows since the last backfill, and that is
//! a read of each table's last row — the last page of its b-tree — rather than
//! a scan.
//!
//! # Why the last row, and not only its id
//!
//! None of these tables is `AUTOINCREMENT`, so SQLite hands out the largest
//! rowid plus one, and an id freed by deleting the newest row is handed out
//! again. That is an ordinary sequence, not a contrived one: emptying the
//! trash of the newest photograph and then scanning a new one, or a local
//! folder's walk removing a renamed file's row and inserting the new name in
//! the same pass. `max(id)` does not move, and neither does `count(*)`. And
//! the row that took the id is exactly one that needs the backfill, because
//! neither scan creates default versions — `persist` and the walk insert the
//! image and leave the version, the pairing and the keyword terms to the next
//! open. Skipped, it would go without them until the app restarted: a rating
//! or a keyword with nowhere to land, a JPEG beside its RAW shown twice.
//!
//! So the stamp carries the last row's content as well as its id: the newest
//! image's path, when it was added, and **whether it has a version**; the
//! newest version's image; the newest assignment's word and version. Whether
//! the newest image has a version is the part that cannot be fooled: once the
//! backfill has run, every image has one, and a row that has just taken a
//! freed id has none, so the two stamps differ whatever the path and the time
//! say. The others make the newest version or assignment a different row
//! whenever a different one took its id; one that is the same content at the
//! same id is the same row as far as the backfill is concerned.
//!
//! The [`Stamp`] is those, the schema version, and the file's identity.
//! An open whose stamp matches the one recorded at the last backfill of the
//! same path skips it; anything else runs it. That covers the cases that must
//! run it:
//!
//! - **The first open in a process.** Nothing is recorded yet.
//! - **A migration.** `user_version` is in the stamp, and [`crate::Catalog::open`]
//! also runs the backfill unconditionally whenever `migrate` moved the
//! schema, because that is what the backfill was written for.
//! - **A pulled catalog.** The merge inserts assignments, which moves the
//! stamp; and [`crate::sync::merge_remote`] [`forget`]s the path as well, so
//! the next open backfills even when every incoming row collided.
//! - **A file replaced underneath the path** — a restore from backup, a
//! rebuild, a catalog copied in. On unix the device and inode are in the
//! stamp, and a replacement is a new inode; [`crate::recovery::set_aside`],
//! the first step of both a restore and a rebuild, forgets the path too.
//! - **Another process writing.** The stamp is read from the file, not from
//! anything this process did, so a scan in a second instance moves it just
//! the same.
//!
//! # Why the stamp is taken before the backfill
//!
//! The backfill adds versions and terms itself, so a stamp read afterwards
//! would describe its own writes. Read afterwards it could also describe an
//! image another connection inserted between the backfill's read and the
//! stamp's — and record that image as covered when it was not. Read before,
//! the worst case is the reverse: the backfill's own inserts move the stamp,
//! and the next open runs one more backfill that finds nothing. That costs one
//! redundant pass after a backfill that did real work, and never misses a row.
//!
//! # What it does not see
//!
//! An `UPDATE` that creates work without adding a row. None of this build's
//! writers does: a scan's move of a file is a new `source_ref` and so a new
//! image, and uuids are only rewritten by the backfill itself. Should one
//! appear, the cost is that its repair waits for the next insert or the next
//! start of the app — which is exactly where the backfill ran before it ran on
//! every open.
//!
//! Kept in memory rather than in the catalog on purpose: a row in the file
//! would travel in the sync snapshot and would need a table an older build
//! does not have, and a flag that another device's catalog carried in would
//! say nothing about this one.
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::{Mutex, OnceLock};
use rusqlite::Connection;
use crate::error::CatalogError;
/// What a catalog looked like, as far as the backfill cares.
#[derive(Debug, Clone, PartialEq, Eq)]
pub(crate) struct Stamp {
/// Device and inode, so a file swapped in under the same name is a new
/// catalog. `None` where the platform has no such thing.
file: Option<(u64, u64)>,
user_version: i64,
/// The newest image: id, path, when added, and whether it has a version.
last_image: Option<String>,
/// The newest version: id and the image it belongs to.
last_version: Option<String>,
/// The newest keyword assignment: rowid, version and word.
last_keyword: Option<String>,
}
/// The stamp recorded at the last backfill, per catalog file.
fn done() -> &'static Mutex<HashMap<PathBuf, Stamp>> {
static DONE: OnceLock<Mutex<HashMap<PathBuf, Stamp>>> = OnceLock::new();
DONE.get_or_init(Default::default)
}
/// One name per file, whichever spelling of its path the caller used.
fn key(path: &Path) -> PathBuf {
std::fs::canonicalize(path).unwrap_or_else(|_| path.to_path_buf())
}
/// Read the stamp of the catalog behind `conn`, which was opened from `path`.
///
/// One statement: the last row of each of three tables, each found by
/// descending its rowid b-tree to the last page, plus one probe of
/// `versions_image` for the newest image — and a `stat` of the file.
pub(crate) fn stamp(conn: &Connection, path: &Path) -> Result<Stamp, CatalogError> {
let (user_version, last_image, last_version, last_keyword) = conn.query_row(
"SELECT (SELECT user_version FROM pragma_user_version),
(SELECT printf('%d|%d|%d|%s', i.id, i.added_at,
EXISTS (SELECT 1 FROM versions v WHERE v.image_id = i.id),
i.source_ref)
FROM images i ORDER BY i.id DESC LIMIT 1),
(SELECT printf('%d|%d', id, image_id)
FROM versions ORDER BY id DESC LIMIT 1),
(SELECT printf('%d|%d|%s', rowid, version_id, keyword)
FROM keywords ORDER BY rowid DESC LIMIT 1)",
[],
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)),
)?;
Ok(Stamp {
file: file_identity(path),
user_version,
last_image,
last_version,
last_keyword,
})
}
#[cfg(unix)]
fn file_identity(path: &Path) -> Option<(u64, u64)> {
use std::os::unix::fs::MetadataExt;
std::fs::metadata(path).ok().map(|m| (m.dev(), m.ino()))
}
#[cfg(not(unix))]
fn file_identity(_path: &Path) -> Option<(u64, u64)> {
None
}
/// Whether the catalog at `path` was last backfilled at exactly `stamp`.
pub(crate) fn is_current(path: &Path, stamp: &Stamp) -> bool {
done()
.lock()
.unwrap_or_else(|e| e.into_inner())
.get(&key(path))
== Some(stamp)
}
/// Record that the catalog at `path` has been backfilled as of `stamp`.
pub(crate) fn record(path: &Path, stamp: Stamp) {
done()
.lock()
.unwrap_or_else(|e| e.into_inner())
.insert(key(path), stamp);
}
/// Make the next open of `path` backfill, whatever its stamp says.
///
/// For the writers that know they have changed the catalog wholesale — a
/// merge of a pulled catalog, a restore from backup — so their correctness
/// does not rest on the stamp happening to move.
pub(crate) fn forget(path: &Path) {
done()
.lock()
.unwrap_or_else(|e| e.into_inner())
.remove(&key(path));
}
#[cfg(test)]
mod tests {
use crate::rating::derived_version_uuid;
use crate::Catalog;
use std::path::PathBuf;
/// A catalog file of its own, holding one image the server has named,
/// backfilled and settled.
///
/// Opened three times on the way: to create it; after the image went in,
/// which gives the image its default version; and once more, because that
/// version moved the stamp and the next open runs the one redundant pass
/// the module header describes. After that the stamp stands still.
fn catalog(tag: &str) -> PathBuf {
let dir = std::env::temp_dir().join(format!(
"dr-backfilled-{tag}-{}-{:?}",
std::process::id(),
std::thread::current().id()
));
let _ = std::fs::remove_dir_all(&dir);
std::fs::create_dir_all(&dir).unwrap();
let path = dir.join("catalog.sqlite");
{
let cat = Catalog::open(&path).unwrap();
let c = cat.connection();
c.execute_batch(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'Photos');
INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1, 1, 'Photos/a.CR3', 0);
INSERT INTO remote(image_id, file_id) VALUES (1, 77);",
)
.unwrap();
}
assert_eq!(uuid(&path, 1), Some(derived_version_uuid(77)));
assert_eq!(uuid(&path, 1), Some(derived_version_uuid(77)));
path
}
/// The default version's uuid for `image`, read through an ordinary open.
fn uuid(path: &std::path::Path, image: i64) -> Option<String> {
let cat = Catalog::open(path).unwrap();
cat.connection()
.query_row(
"SELECT uuid FROM versions WHERE image_id = ?1 AND is_default = 1",
[image],
|r| r.get(0),
)
.ok()
}
/// Put the one row back into the state the backfill repairs, with an
/// `UPDATE` — which moves none of the stamp's maxima, so only the stamp's
/// other parts or an explicit `forget` can bring the backfill back.
fn unalign(path: &std::path::Path) {
rusqlite::Connection::open(path)
.unwrap()
.execute("UPDATE versions SET uuid = 'minted' WHERE image_id = 1", [])
.unwrap();
}
#[test]
fn an_unchanged_catalog_is_not_backfilled_again() {
let path = catalog("unchanged");
unalign(&path);
assert_eq!(
uuid(&path, 1).as_deref(),
Some("minted"),
"nothing was added since the last backfill, so the open skipped it"
);
}
#[test]
fn an_image_a_scan_added_is_backfilled_on_the_next_open() {
let path = catalog("scanned");
rusqlite::Connection::open(&path)
.unwrap()
.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (2, 1, 'Photos/b.CR3', 0)",
[],
)
.unwrap();
assert!(uuid(&path, 2).is_some(), "the new image got its version");
}
/// The id of a deleted newest row is handed out again, so `max(id)` is
/// the same before and after — the sequence emptying the trash and then
/// scanning makes. The image that took the id still needs its version.
///
/// A virtual copy on the older image holds the newest version id, so
/// the deletion does not move `max(versions.id)` either: nothing the old
/// stamp read changes, which is the case that went unrepaired.
#[test]
fn an_image_that_reuses_a_deleted_id_is_backfilled_on_the_next_open() {
let path = catalog("reused");
rusqlite::Connection::open(&path)
.unwrap()
.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (2, 1, 'Photos/b.CR3', 0)",
[],
)
.unwrap();
assert!(uuid(&path, 2).is_some());
rusqlite::Connection::open(&path)
.unwrap()
.execute(
"INSERT INTO versions(image_id, uuid, name, is_default)
VALUES (1, 'copy', 'Crop', 0)",
[],
)
.unwrap();
// Settle: one open backfills after the new version, one more runs
// the redundant pass and records the stamp that stands.
assert!(uuid(&path, 2).is_some());
assert!(uuid(&path, 2).is_some());
let c = rusqlite::Connection::open(&path).unwrap();
let before: (i64, i64) = c
.query_row(
"SELECT (SELECT max(id) FROM images), (SELECT max(id) FROM versions)",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
c.execute("DELETE FROM images WHERE id = 2", []).unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, added_at)
VALUES (1, 'Photos/c.CR3', 0)",
[],
)
.unwrap();
let after: (i64, i64) = c
.query_row(
"SELECT (SELECT max(id) FROM images), (SELECT max(id) FROM versions)",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(before, after, "SQLite handed the freed id out again");
drop(c);
assert!(uuid(&path, 2).is_some(), "the new image got its version");
}
/// The same, with the same file coming back at the same id in the same
/// second: path and time match, and only the missing version tells.
#[test]
fn the_same_file_back_at_the_same_id_is_backfilled_on_the_next_open() {
let path = catalog("returned");
let c = rusqlite::Connection::open(&path).unwrap();
// As above: a newer version on another image keeps the deletion
// from moving `max(versions.id)`.
c.execute_batch(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (0, 1, 'Photos/0.CR3', 0);
INSERT INTO versions(image_id, uuid, name, is_default)
VALUES (0, 'copy', 'Crop', 0);",
)
.unwrap();
drop(c);
assert!(uuid(&path, 1).is_some());
assert!(uuid(&path, 1).is_some());
let c = rusqlite::Connection::open(&path).unwrap();
c.execute_batch(
"DELETE FROM images WHERE id = 1;
INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1, 1, 'Photos/a.CR3', 0);
INSERT INTO remote(image_id, file_id) VALUES (1, 77);",
)
.unwrap();
drop(c);
assert_eq!(uuid(&path, 1), Some(derived_version_uuid(77)));
}
#[test]
fn the_first_open_after_a_migration_backfills() {
let path = catalog("migrated");
unalign(&path);
rusqlite::Connection::open(&path)
.unwrap()
.pragma_update(None, "user_version", crate::schema::SCHEMA_VERSION - 1)
.unwrap();
assert_eq!(uuid(&path, 1), Some(derived_version_uuid(77)));
}
#[test]
fn the_first_open_after_a_pulled_catalog_backfills() {
let path = catalog("pulled");
// The remote is this catalog as it stands, so every row the merge
// offers collides and nothing in the stamp moves: only the merge
// saying so can make the next open backfill.
let remote = path.with_file_name("remote.sqlite");
rusqlite::Connection::open(&path)
.unwrap()
.execute("VACUUM INTO ?1", [remote.to_string_lossy().as_ref()])
.unwrap();
unalign(&path);
assert_eq!(uuid(&path, 1).as_deref(), Some("minted"));
Catalog::open(&path)
.unwrap()
.merge_remote_catalog(&remote)
.unwrap();
assert_eq!(uuid(&path, 1), Some(derived_version_uuid(77)));
}
#[cfg(unix)]
#[test]
fn a_catalog_replaced_under_the_same_name_backfills() {
let path = catalog("replaced");
unalign(&path);
assert_eq!(uuid(&path, 1).as_deref(), Some("minted"));
// Every connection is closed, so the WAL is folded in and the main
// file is the whole catalog. A copy renamed over it is the same rows
// in a new file — which is what a restore or a copied-in catalog is.
let copy = path.with_file_name("copy.sqlite");
std::fs::copy(&path, &copy).unwrap();
std::fs::rename(&copy, &path).unwrap();
assert_eq!(uuid(&path, 1), Some(derived_version_uuid(77)));
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-308
View File
@@ -1,308 +0,0 @@
//! TRACES: FR-CAT-11
//! Has this photograph been imported before?
//!
//! Two tiers, because neither alone is enough and they cost very different
//! amounts. The metadata tier — capture time, camera, size, the name the
//! camera gave it — is answerable from the catalog before a byte leaves the
//! card, which is what makes re-inserting an already-imported card cost a
//! metadata read per file rather than a full transfer. The content tier
//! catches what the first misses: the same frame arriving under a different
//! name, from a second card, or after somebody renamed it.
//!
//! # Why the filename is compared here rather than in SQL
//!
//! `images.source_ref` holds the whole opaque key — a relative path on Linux,
//! a document id on SAF — and the camera's filename is only its last
//! component. Matching that in SQL means `LIKE '%/IMG_0001.CR3'`, which cannot
//! use an index, scans the whole table, and is wrong on SAF where the
//! separator is not `/`. So the query narrows on the indexed columns and the
//! handful of rows that survive are compared in Rust, the same way the grid
//! already derives a display name.
//!
//! # Filename alone is never sufficient
//!
//! Camera filenames wrap at `IMG_9999` and start again, so a library of any
//! age holds several unrelated `IMG_0001.CR3`. That is why the cheap tier
//! carries capture time and camera as well, and why the expensive tier exists
//! at all.
use rusqlite::Connection;
use crate::CatalogError;
/// The last component of a stored source reference.
///
/// Splits on both separators for the same reason `Catalog::window` does: the
/// key's shape belongs to the storage that produced it, and a SAF document id
/// is delimited with `:`.
fn file_name(source_ref: &str) -> &str {
source_ref.rsplit(['/', ':']).next().unwrap_or(source_ref)
}
/// Whether the catalog already holds this photograph, on metadata alone.
///
/// `camera` is the joined make-and-model string the scan stores, not the raw
/// EXIF pair — the caller composes it the same way, or the comparison is
/// always false.
///
/// A `captured_at` of `None` makes this answer `false` rather than matching
/// every undated image in the library: without a capture time the key is
/// filename plus size, which two frames from the same body collide on
/// routinely. An undated file falls through to the content tier, which is
/// slower and right.
pub fn seen_by_metadata(
conn: &Connection,
captured_at: Option<i64>,
camera: Option<&str>,
size: u64,
original_name: &str,
) -> Result<bool, CatalogError> {
let Some(captured_at) = captured_at else {
return Ok(false);
};
// `images_captured` indexes the capture time, so this reads a few rows
// even in a library of fifty thousand: one instant to the second holds
// one frame, or a handful on a body shooting a burst.
let mut stmt = conn.prepare(
"SELECT source_ref FROM images
WHERE captured_at = ?1
AND (?2 IS NULL OR camera IS ?2)
AND (file_size IS NULL OR file_size = ?3)",
)?;
let mut rows = stmt.query(rusqlite::params![captured_at, camera, size as i64])?;
while let Some(row) = rows.next()? {
let source_ref: String = row.get(0)?;
if file_name(&source_ref).eq_ignore_ascii_case(original_name) {
return Ok(true);
}
}
Ok(false)
}
/// Whether these exact bytes are already in the library.
///
/// The tier that costs a read of the file. Cheap here — `images_hash` is a
/// partial index over the rows that have one — and expensive for the caller,
/// which had to hash something to ask.
pub fn seen_by_content(conn: &Connection, digest: &str) -> Result<bool, CatalogError> {
let n: i64 = conn.query_row(
"SELECT COUNT(*) FROM images WHERE content_hash = ?1",
[digest],
|r| r.get(0),
)?;
Ok(n > 0)
}
/// Record the digest of a file the import computed.
///
/// An import reads every byte anyway, so the hash is free at that moment and
/// costs a full read of an 80 MB file at any other. Storing it is what lets
/// the *next* import answer [`seen_by_content`] without reading anything.
///
/// Matched on `source_ref` within a root, which is how the scan that just
/// catalogued the imported file identifies it. Returns how many rows were
/// updated: zero means the scan has not reached the file yet, which is a
/// normal race and not an error.
pub fn set_content_hash(
conn: &Connection,
root_id: u64,
source_ref: &str,
digest: &str,
) -> Result<usize, CatalogError> {
Ok(conn.execute(
"UPDATE images SET content_hash = ?3
WHERE root_id = ?1 AND source_ref = ?2",
rusqlite::params![root_id as i64, source_ref, digest],
)?)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::Catalog;
/// A catalog holding one photograph, as a scan plus a metadata pass would
/// leave it.
fn with_one() -> Catalog {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, captured_at, camera, file_size,
content_hash, added_at)
VALUES (1, '2026/2026-08-22/IMG_0001.CR3', 1787407200, 'Canon EOS R5',
9, 'deadbeef', 0)",
[],
)
.unwrap();
cat
}
#[test]
fn re_inserting_the_same_card_is_recognised_before_a_transfer() {
let cat = with_one();
assert!(seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
9,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn a_different_frame_at_the_same_instant_is_not_a_duplicate() {
// Two bodies firing together, or a burst. The name separates them.
let cat = with_one();
assert!(!seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
9,
"IMG_0002.CR3"
)
.unwrap());
}
#[test]
fn the_same_name_from_a_different_camera_is_not_a_duplicate() {
// IMG_0001.CR3 exists on every card ever formatted.
let cat = with_one();
assert!(!seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("NIKON Z 9"),
9,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn the_same_name_at_a_different_time_is_not_a_duplicate() {
// The IMG_9999 wrap: the library holds an unrelated IMG_0001.CR3 from
// four years ago, and matching on name alone would refuse to import
// today's.
let cat = with_one();
assert!(!seen_by_metadata(
cat.connection(),
Some(1_600_000_000),
Some("Canon EOS R5"),
9,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn an_undated_file_falls_through_to_the_content_tier() {
// Not "matches everything undated" — that would silently refuse to
// import a whole card of scanned film.
let cat = with_one();
assert!(!seen_by_metadata(cat.connection(), None, None, 9, "IMG_0001.CR3").unwrap());
}
#[test]
fn a_file_that_grew_is_not_the_one_already_held() {
// A truncated earlier import, or a different rendition of the same
// frame. Same instant, same camera, same name, different bytes.
let cat = with_one();
assert!(!seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
1234,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn a_row_with_no_recorded_size_still_matches() {
// The scan stores a size, but a row merged from another device may
// not have one, and refusing to match it would re-import the library.
let cat = with_one();
cat.connection()
.execute("UPDATE images SET file_size = NULL", [])
.unwrap();
assert!(seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
9,
"IMG_0001.CR3"
)
.unwrap());
}
#[test]
fn the_same_frame_renamed_is_caught_by_its_bytes() {
let cat = with_one();
// The metadata tier misses it...
assert!(!seen_by_metadata(
cat.connection(),
Some(1_787_407_200),
Some("Canon EOS R5"),
9,
"holiday-42.CR3"
)
.unwrap());
// ...and the content tier does not.
assert!(seen_by_content(cat.connection(), "deadbeef").unwrap());
assert!(!seen_by_content(cat.connection(), "cafe").unwrap());
}
#[test]
fn a_digest_recorded_now_answers_the_next_import() {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, added_at)
VALUES (1, '2026/2026-08-22/IMG_0001.CR3', 0)",
[],
)
.unwrap();
assert!(!seen_by_content(c, "abc123").unwrap());
let n = set_content_hash(c, 1, "2026/2026-08-22/IMG_0001.CR3", "abc123").unwrap();
assert_eq!(n, 1);
assert!(seen_by_content(c, "abc123").unwrap());
}
#[test]
fn recording_a_digest_before_the_scan_arrives_is_not_an_error() {
// The import writes the file and the scan catalogues it; between those
// two moments there is no row to update, and that is a race rather
// than a failure.
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
assert_eq!(
set_content_hash(c, 1, "not/scanned/yet.CR3", "abc").unwrap(),
0
);
}
#[test]
fn a_name_is_the_last_component_of_either_kind_of_key() {
assert_eq!(file_name("2026/2026-08-22/IMG_0001.CR3"), "IMG_0001.CR3");
// A SAF document id delimits with a colon.
assert_eq!(file_name("primary:DCIM/Camera/IMG_1.CR3"), "IMG_1.CR3");
assert_eq!(file_name("IMG_0001.CR3"), "IMG_0001.CR3");
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-151
View File
@@ -1,151 +0,0 @@
//! TRACES: NFR-ARCH-4 | NFR-R5 | NFR-R6
//! Catalog errors.
//!
//! Typed and attached to the affected subject rather than panicking — a
//! corrupt row or a failed job marks one image and lets the batch continue.
//!
//! # Why `From<rusqlite::Error>` is written by hand
//!
//! One class of SQLite failure is not about the statement that hit it: when
//! the file itself is damaged, *every* query fails, and which one the user
//! happened to trigger first says nothing. Before this, corruption reached the
//! interface as whatever `Sqlite(...)` the first failing query produced —
//! "database disk image is malformed" attached to a thumbnail refresh — and
//! there was nowhere to hang a recovery offer.
//!
//! So the conversion classifies rather than wraps: `SQLITE_CORRUPT` and
//! `SQLITE_NOTADB` become [`CatalogError::Corrupt`] wherever they arise, which
//! means a background job that trips over the damage reports the same thing
//! the startup check does (see [`crate::recovery`]).
/// Something went wrong talking to the catalog.
#[derive(Debug, thiserror::Error)]
pub enum CatalogError {
#[error("sqlite: {0}")]
Sqlite(#[source] rusqlite::Error),
/// The catalog file is damaged.
///
/// Its own variant because it is the one error with a *user-facing
/// remedy*: restore the NFR-R2 backup, or discard the index and rebuild it
/// from sources plus sidecars (NFR-R6, invariant §5.2.4). Every other
/// variant here is either a caller's mistake or a fact about one row.
#[error("the catalog file is damaged: {detail}")]
Corrupt { detail: String },
/// The catalog was written by a newer build.
///
/// Opening it read-write would corrupt state this build cannot represent,
/// so the app refuses and says so (NFR-R5).
#[error("catalog schema v{found} is newer than this build supports (v{supported})")]
SchemaTooNew { found: i64, supported: i64 },
/// A scan could not reach a root at all.
///
/// Distinct from "files are missing": this aborts the scan *before* the
/// deletion sweep, because every folder would look unreached and the sweep
/// would delete the whole library (FR-CAT-9).
#[error("root {0} is unreachable; scan aborted without pruning")]
RootUnreachable(u64),
/// A scan was asked for a root the catalog has no row for.
///
/// A caller's mistake rather than a user's: the row is created when the
/// grant is obtained, because the label — the path, the tree URI — is known
/// only there. Inventing one here would file the library under a name
/// nothing else would look it up by.
#[error("no such root: {0}")]
NoSuchRoot(u64),
/// A smart collection whose selector references itself, directly or via
/// another collection.
#[error("collection {0} would form a cycle")]
CollectionCycle(u64),
#[error("no such collection: {0}")]
NoSuchCollection(u64),
/// An album the caller named is gone — deleted here, or by a merge while
/// its id sat in a UI model.
#[error("no such album: {0}")]
NoSuchAlbum(u64),
/// A name that is empty once trimmed. Refused rather than stored, because
/// a row with no name is one the sidebar cannot draw and nobody can pick.
#[error("a name is required")]
EmptyName,
/// A keyword the caller named is gone — deleted, or fused into another by a
/// merge while its id sat in a UI model.
///
/// Its own variant rather than a silent no-op because the two are different
/// answers to the user: a rename that quietly did nothing looks exactly like
/// a rename that did not take.
#[error("no such keyword: {0}")]
NoSuchKeyword(u64),
/// Images were dropped onto a smart collection.
///
/// A smart collection's membership *is* its selector, so member rows would
/// be a second source of truth that nothing reads. Refused rather than
/// silently discarded, so the UI can say why the drop did nothing.
#[error("collection {0} is a saved filter; its contents cannot be edited by hand")]
SmartCollectionNotEditable(u64),
#[error("malformed stored selector: {0}")]
BadSelector(String),
/// A name the user typed that cannot be stored — blank, or one a sibling
/// already holds.
///
/// Its own variant rather than a reused `BadSelector`, because this one is
/// shown to the user verbatim: it has to read as a sentence about their
/// collection, not as a diagnostic about a stored selector.
#[error("{0}")]
BadName(String),
#[error("io: {0}")]
Io(String),
/// TRACES: FR-CAT-11a
/// A duplicate group planned earlier no longer holds: a copy was trashed,
/// rescanned or changed since the review was drawn. The group is left
/// untouched rather than consolidated on a stale plan.
#[error("no longer a duplicate: {0}")]
StaleDuplicate(String),
}
impl From<rusqlite::Error> for CatalogError {
fn from(e: rusqlite::Error) -> Self {
if is_corruption(&e) {
// `to_string` rather than keeping the error: the detail is going
// into a dialog and into a log line, and the recovery path has no
// use for the rusqlite type once it knows the file is damaged.
CatalogError::Corrupt {
detail: e.to_string(),
}
} else {
CatalogError::Sqlite(e)
}
}
}
/// Whether a SQLite failure means the *file* is damaged rather than the
/// statement wrong.
///
/// `SQLITE_NOTADB` is included because it is what a truncated or overwritten
/// catalog produces — SQLite cannot read the header, so it declines to call it
/// a database at all. To a user those are the same accident, and the same two
/// offers answer both.
///
/// Deliberately *not* included: `SQLITE_CANTOPEN` (a missing file, which
/// `Connection::open` fixes by creating one), `SQLITE_BUSY`, and
/// `SQLITE_IOERR` — a failing disk or a dropped network mount is a different
/// problem, and telling the user to rebuild their index would be a wrong
/// answer delivered confidently.
fn is_corruption(e: &rusqlite::Error) -> bool {
matches!(
e.sqlite_error_code(),
Some(rusqlite::ErrorCode::DatabaseCorrupt) | Some(rusqlite::ErrorCode::NotADatabase)
)
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-919
View File
@@ -1,919 +0,0 @@
//! TRACES: FR-CAT-3 | NFR-ARCH-2 | FR-PLAT-AND-3
//! The background work queue.
//!
//! Jobs live in the catalog, so they survive process death — routine on
//! Android rather than exceptional (FR-PLAT-AND-3). Two properties carry the
//! design:
//!
//! - **Coalescing.** `UNIQUE(kind, subject_id)` makes enqueueing idempotent,
//! so every code path that notices a change can just enqueue and let the
//! table absorb the redundancy.
//! - **Priority shared with the GPU scheduler** (ARCH §5.3), so one notion of
//! urgency governs the whole app and visible work always preempts bulk work.
//!
//! Nothing in this file runs a job. [`crate::runner`] is the other half — the
//! one that claims from this table, does the work through a handler, and
//! reports back. Worth knowing because for a long time it did not exist: every
//! producer called [`enqueue`] and nothing ever called [`claim_next`], so the
//! table only ever grew.
use rusqlite::{Connection, OptionalExtension};
use crate::error::CatalogError;
/// What a job does.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[repr(i64)]
pub enum JobKind {
/// Recursive incremental scan from a folder (§scan).
ScanFolder = 0,
/// Promote an image from stat-only to full EXIF.
ExtractMetadata = 1,
/// Build or rebuild a thumbnail. **Retired** — see [`JobKind::RETIRED`].
///
/// Kept so the number stays taken: a catalog written by 0.16.0 or earlier
/// holds rows of kind 2, and reusing it would hand them to whatever took
/// its place.
Thumbnail = 2,
/// A sidecar on disk is newer than what the catalog read.
ReadSidecar = 3,
/// Flush a local edit to its sidecar. Debounced, never per slider tick.
WriteSidecar = 4,
/// Whole-file hash. On demand only — import dedup, reconnect-by-hash.
ContentHash = 5,
/// Range-extract an embedded preview from a remote file (FR-NC-3).
FetchPreview = 6,
/// Fetch a full original: pinned by rule, or explicitly asked for.
FetchOriginal = 7,
/// Detect and embed the faces in one image (FR-CULL-8).
///
/// One job does both, rather than splitting them: the proxy is already
/// decoded and in memory, and the natural unit of resumable work is one
/// photograph. Splitting would double the queue's row count for nothing.
///
/// Runs against the proxy tier, never a full decode — a library that has
/// been browsed has already paid for its proxies, so face indexing adds no
/// RAW decodes that were not already happening.
DetectFaces = 8,
}
impl JobKind {
/// Every kind, so code that has to enumerate them cannot quietly miss one
/// that was added later. A `match` would catch that; a hand-written array
/// at each call site would not.
pub const ALL: [JobKind; 9] = [
JobKind::ScanFolder,
JobKind::ExtractMetadata,
JobKind::Thumbnail,
JobKind::ReadSidecar,
JobKind::WriteSidecar,
JobKind::ContentHash,
JobKind::FetchPreview,
JobKind::FetchOriginal,
JobKind::DetectFaces,
];
/// Kinds that are no longer queued by anything, whose rows are deleted on
/// sight by [`drop_retired`].
///
/// `Thumbnail` is here because thumbnails are owed by the store, not by
/// the queue. The grid's worker and the thumbnail sweep both find their
/// work by asking `ThumbStore` what it lacks, and the store is shared
/// between devices, so it is the only thing that can say another device
/// already made one. Up to 0.16.0 every scan enqueued a job per
/// photograph anyway and no handler ever claimed one: the reference
/// catalog held 23,582 of them (#73; catalog.md §6.1).
pub const RETIRED: [JobKind; 1] = [JobKind::Thumbnail];
fn from_i64(v: i64) -> Option<Self> {
Some(match v {
0 => JobKind::ScanFolder,
1 => JobKind::ExtractMetadata,
2 => JobKind::Thumbnail,
3 => JobKind::ReadSidecar,
4 => JobKind::WriteSidecar,
5 => JobKind::ContentHash,
6 => JobKind::FetchPreview,
7 => JobKind::FetchOriginal,
8 => JobKind::DetectFaces,
_ => return None,
})
}
/// Whether this job transfers over the network, and so is subject to the
/// metered-connection and charging constraints in FR-NC-6.
pub fn is_network(self) -> bool {
matches!(self, JobKind::FetchPreview | JobKind::FetchOriginal)
}
/// Whether `subject_id` names a row in `images`.
///
/// Every kind but one is per-photograph. `ScanFolder`'s subject is a
/// *folder*, and the two id spaces are unrelated — so anything that joins
/// `subject_id` against `images` has to exclude it, or it will read one
/// table's ids as another's and act on the answer.
pub fn subject_is_image(self) -> bool {
!matches!(self, JobKind::ScanFolder)
}
}
/// Scheduling class, matching the GPU tile scheduler (ARCH §5.3).
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
#[repr(i64)]
pub enum Priority {
/// Bulk work: metadata sweeps, rule-driven fetches, hashing.
Background = 0,
/// Just outside the viewport; the next image in culling.
Prefetch = 1,
/// Visible cells, and the image currently open.
///
/// Strictly preempts background work. Without this, scrolling during a
/// bulk thumbnail pass misses its frame budget — the common case, not an
/// edge case (NFR-ARCH-2).
Interactive = 2,
}
impl Priority {
/// Read back from the stored column.
///
/// An unrecognised value reads as `Background` rather than failing: a
/// priority is a hint about ordering, and refusing to run a job because
/// its urgency is spelled oddly would be a worse answer than running it
/// last.
fn from_i64(v: i64) -> Self {
match v {
2 => Priority::Interactive,
1 => Priority::Prefetch,
_ => Priority::Background,
}
}
}
/// Lifecycle state.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[repr(i64)]
pub enum JobState {
Pending = 0,
Running = 1,
Failed = 2,
}
/// A job ready to run.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Job {
pub id: i64,
pub kind: JobKind,
pub subject_id: Option<i64>,
pub priority: Priority,
pub attempts: i64,
pub payload: Option<String>,
}
/// Give up after this many attempts and attach the error to the subject.
///
/// One corrupt file must not stall the queue behind endless retries
/// (FR-RAW-4).
pub const MAX_ATTEMPTS: i64 = 5;
/// Backoff before retrying a failed job, in seconds.
///
/// Exponential, capped — a server that is down for an hour should not be
/// retried every second, and a transient decode failure should not wait an
/// hour.
pub fn backoff_seconds(attempts: i64) -> i64 {
const CAP: i64 = 300;
match attempts {
a if a <= 0 => 0,
a if a >= 9 => CAP,
a => (1i64 << (a - 1)).min(CAP),
}
}
/// Enqueue work, coalescing with any identical pending job.
///
/// Re-requesting at a higher priority *promotes* the existing row rather than
/// duplicating it, which is what lets the grid shout "this one is visible now"
/// about a job already queued in the background.
pub fn enqueue(
conn: &Connection,
kind: JobKind,
subject_id: Option<i64>,
priority: Priority,
payload: Option<&str>,
) -> Result<(), CatalogError> {
// Cached: a scan enqueues one per photograph it lists.
conn.prepare_cached(
"INSERT INTO jobs(kind, subject_id, priority, state, payload)
VALUES (?1, ?2, ?3, 0, ?4)
ON CONFLICT(kind, subject_id) DO UPDATE SET
priority = max(jobs.priority, excluded.priority),
-- A job that failed and is being re-requested deserves a fresh
-- start: the file may well have changed since it failed.
state = CASE WHEN jobs.state = 2 THEN 0 ELSE jobs.state END,
attempts = CASE WHEN jobs.state = 2 THEN 0 ELSE jobs.attempts END,
not_before = CASE WHEN jobs.state = 2 THEN 0 ELSE jobs.not_before END",
)?
.execute(rusqlite::params![
kind as i64,
subject_id,
priority as i64,
payload
])?;
Ok(())
}
/// Claim the next runnable job, highest priority first.
///
/// `now` is passed rather than read from the clock so backoff is testable.
pub fn claim_next(conn: &Connection, now: i64) -> Result<Option<Job>, CatalogError> {
claim(conn, now, None)
}
/// Claim the next runnable job of one of `kinds`.
///
/// What lets a runner take only the work it can actually do. A device with no
/// connector must leave `FetchOriginal` rows alone rather than claim them and
/// fail them five times each with backoff; a runner gated onto an unmetered
/// network (FR-NC-6) passes the local kinds only and leaves the transfers
/// where they are. Neither is expressible by filtering *after* a claim,
/// because the claim has already marked the row `Running`.
///
/// An empty list claims nothing, which is the honest reading of "there is
/// nothing this worker can do".
pub fn claim_next_matching(
conn: &Connection,
now: i64,
kinds: &[JobKind],
) -> Result<Option<Job>, CatalogError> {
if kinds.is_empty() {
return Ok(None);
}
claim(conn, now, Some(kinds))
}
/// The claim, as one statement.
///
/// # Why this is not a transaction around a read and a write
///
/// It used to be, and under two connections that is not safe in the way it
/// looks. A deferred transaction takes a read lock for the `SELECT` and only
/// tries to upgrade at the `UPDATE`; with WAL, a second worker that read the
/// same snapshot gets `SQLITE_BUSY_SNAPSHOT` on its write — an error a busy
/// handler cannot retry away, because the fix is to roll back and start over.
/// So the queue was correct only in the sense that the loser failed loudly.
///
/// `UPDATE ... WHERE id = (SELECT ...) RETURNING` is one statement, so it is
/// one implicit transaction that takes the write lock immediately. Two workers
/// serialise, the loser waits out its busy timeout rather than erroring, and
/// neither can see a row the other is already holding.
fn claim(
conn: &Connection,
now: i64,
kinds: Option<&[JobKind]>,
) -> Result<Option<Job>, CatalogError> {
// `now` first, then the kinds, matching the order the placeholders appear
// in the text below.
let mut args: Vec<i64> = vec![now];
let filter = match kinds {
None => String::new(),
Some(kinds) => {
// Built from the kind *count*, never from anything a user typed —
// the same discipline `collections::descendants` uses, since
// `carray` is not compiled in.
let placeholders = std::iter::repeat_n("?", kinds.len())
.collect::<Vec<_>>()
.join(",");
args.extend(kinds.iter().map(|k| *k as i64));
format!(" AND kind IN ({placeholders})")
}
};
let sql = format!(
"UPDATE jobs
SET state = 1, attempts = attempts + 1
WHERE id = (SELECT id
FROM jobs
WHERE state = 0 AND not_before <= ?{filter}
ORDER BY priority DESC, id ASC
LIMIT 1)
RETURNING id, kind, subject_id, priority, attempts, payload"
);
let claimed = conn
.query_row(&sql, rusqlite::params_from_iter(args.iter()), |r| {
Ok((
r.get::<_, i64>(0)?,
r.get::<_, i64>(1)?,
r.get::<_, Option<i64>>(2)?,
r.get::<_, i64>(3)?,
r.get::<_, i64>(4)?,
r.get::<_, Option<String>>(5)?,
))
})
.optional()?;
let Some((id, kind, subject_id, priority, attempts, payload)) = claimed else {
return Ok(None);
};
let Some(kind) = JobKind::from_i64(kind) else {
// A row written by a build that knows a kind this one does not — a
// downgrade, or a catalog synced from a newer device. Running it as
// some other kind would be worse than not running it, so it is parked
// where the next claim will not see it again.
//
// Answering `None` understates what is queued for one pass. The
// alternative is a loop that keeps claiming the same unreadable row.
abandon(conn, id, &format!("unknown job kind {kind}"))?;
return Ok(None);
};
Ok(Some(Job {
id,
kind,
subject_id,
priority: Priority::from_i64(priority),
attempts,
payload,
}))
}
/// Job finished successfully.
pub fn complete(conn: &Connection, id: i64) -> Result<(), CatalogError> {
conn.execute("DELETE FROM jobs WHERE id = ?1", [id])?;
Ok(())
}
/// Job failed. Reschedules with backoff, or gives up past [`MAX_ATTEMPTS`].
pub fn fail(conn: &Connection, job: &Job, now: i64, err: &str) -> Result<(), CatalogError> {
if job.attempts >= MAX_ATTEMPTS {
abandon(conn, job.id, err)
} else {
conn.execute(
"UPDATE jobs SET state = 0, not_before = ?2, last_error = ?3 WHERE id = ?1",
rusqlite::params![job.id, now + backoff_seconds(job.attempts), err],
)?;
Ok(())
}
}
/// Give up on a job now, with no further retries.
///
/// For failures a retry cannot fix — the subject is gone, the payload is
/// unreadable, the format is one this build does not know. Walking the whole
/// retry ladder to reach a conclusion the first attempt already reached costs
/// five wakeups and five backoffs per photograph, which on a phone is the
/// difference the user notices.
///
/// The row is kept rather than deleted, because "this file failed and here is
/// why" is something the user is entitled to see (NFR-ARCH-4).
pub fn abandon(conn: &Connection, id: i64, err: &str) -> Result<(), CatalogError> {
conn.execute(
"UPDATE jobs SET state = 2, last_error = ?2 WHERE id = ?1",
rusqlite::params![id, err],
)?;
Ok(())
}
/// Put a claimed job back exactly as it was found.
///
/// For a worker that is being stopped rather than a job that is going wrong:
/// the platform revoked the slot, the user left the screen. The attempt the
/// claim consumed is given back, because nothing was learned about the file —
/// without that, five backgroundings in a row would mark good work as failed.
///
/// Guarded on the claim still being *this* claim. There is no owner column, so
/// `attempts` stands in for one: it is bumped by every claim, so the row only
/// still reads `state = 1` with the caller's own attempt number while nobody
/// else has taken it since. A worker that comes back after the recovery pass
/// handed its job to someone else therefore changes nothing, rather than
/// releasing a job another worker is in the middle of.
pub fn release(conn: &Connection, job: &Job) -> Result<(), CatalogError> {
conn.execute(
"UPDATE jobs SET state = 0, attempts = max(0, attempts - 1)
WHERE id = ?1 AND state = 1 AND attempts = ?2",
rusqlite::params![job.id, job.attempts],
)?;
Ok(())
}
/// Recover jobs orphaned by process death.
///
/// A row left `Running` has no owner — the process that claimed it is gone.
/// Called at startup, before any worker begins (FR-PLAT-AND-3).
///
/// The attempt the dead claim consumed is deliberately *not* refunded. A job
/// that takes the process down with it is indistinguishable from one that
/// fails, and the attempt counter is the only evidence that survives a death —
/// without it a poison-pill job is reclaimed and re-run forever.
pub fn recover_orphaned(conn: &Connection) -> Result<usize, CatalogError> {
let n = conn.execute("UPDATE jobs SET state = 0 WHERE state = 1", [])?;
Ok(n)
}
/// Delete jobs whose photograph is gone.
///
/// Coalescing keeps the table one row per unit of work, but nothing shrinks it
/// when the work stops existing: a library that has been culled carries a
/// job for every photograph deleted since the last time anything
/// looked. Each one would be claimed, run, and failed five times.
///
/// Only kinds whose subject really is an image ([`JobKind::subject_is_image`])
/// are considered — `ScanFolder`'s subject is a folder id, and joining it
/// against `images` would delete jobs by coincidence of numbering.
///
/// There is no foreign key to do this instead. `jobs.subject_id` deliberately
/// references nothing: it means different tables for different kinds, and a
/// constraint that is right for eight of nine kinds is not a constraint.
pub fn reap_orphan_subjects(conn: &Connection) -> Result<usize, CatalogError> {
let kinds: Vec<i64> = JobKind::ALL
.iter()
.filter(|k| k.subject_is_image())
.map(|k| *k as i64)
.collect();
let placeholders = std::iter::repeat_n("?", kinds.len())
.collect::<Vec<_>>()
.join(",");
let n = conn.execute(
&format!(
"DELETE FROM jobs
WHERE subject_id IS NOT NULL
AND kind IN ({placeholders})
AND NOT EXISTS (SELECT 1 FROM images WHERE images.id = jobs.subject_id)"
),
rusqlite::params_from_iter(kinds.iter()),
)?;
Ok(n)
}
/// Delete every row of a [`JobKind::RETIRED`] kind.
///
/// Not a migration, deliberately. A schema bump makes an older build refuse
/// the synced catalog snapshot, and a device still on 0.16.0 would lose the
/// catalog to save a megabyte. So this runs where the queue is readied —
/// [`crate::runner::recover`], at every open — and has to be cheap when there
/// is nothing to do: `kind` leads the `UNIQUE(kind, subject_id)` index, so an
/// empty answer is one index probe, not a table scan.
///
/// Every open rather than once, because once is not enough: an older build
/// opening the same catalog enqueues them again on its next scan.
///
/// Rows in any state go. Nothing claims these kinds, so none can be running,
/// and a failed one would be a report about work nobody was going to do.
pub fn drop_retired(conn: &Connection) -> Result<usize, CatalogError> {
let kinds: Vec<i64> = JobKind::RETIRED.iter().map(|k| *k as i64).collect();
let placeholders = std::iter::repeat_n("?", kinds.len())
.collect::<Vec<_>>()
.join(",");
let n = conn.execute(
&format!("DELETE FROM jobs WHERE kind IN ({placeholders})"),
rusqlite::params_from_iter(kinds.iter()),
)?;
Ok(n)
}
/// How much is left, by state.
///
/// One query rather than a listing, because the caller is a progress line: a
/// foreground service's notification has to say how much remains without
/// reading a hundred thousand rows to find out (FR-PLAT-AND-4).
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
pub struct Counts {
/// Claimable now or after a backoff.
pub pending: usize,
/// Claimed by someone. After a clean start that is a live worker; before
/// [`recover_orphaned`] it is a dead one.
pub running: usize,
/// Given up on, and kept so the user can see what failed and why.
pub failed: usize,
}
impl Counts {
/// Work that is still going to happen.
pub fn outstanding(&self) -> usize {
self.pending + self.running
}
}
/// Count the queue by state.
pub fn counts(conn: &Connection) -> Result<Counts, CatalogError> {
// `sum` over no rows is NULL, not 0 — an empty queue would otherwise fail
// to convert rather than counting nothing.
let (pending, running, failed) = conn.query_row(
"SELECT sum(state = 0), sum(state = 1), sum(state = 2) FROM jobs",
[],
|r| {
Ok((
r.get::<_, Option<i64>>(0)?,
r.get::<_, Option<i64>>(1)?,
r.get::<_, Option<i64>>(2)?,
))
},
)?;
Ok(Counts {
pending: pending.unwrap_or(0) as usize,
running: running.unwrap_or(0) as usize,
failed: failed.unwrap_or(0) as usize,
})
}
#[cfg(test)]
mod tests {
use super::*;
use crate::schema;
fn db() -> Connection {
let c = Connection::open_in_memory().unwrap();
schema::configure(&c).unwrap();
schema::migrate(&c).unwrap();
c
}
#[test]
fn repeated_enqueue_coalesces() {
let c = db();
for _ in 0..10 {
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
}
let n: i64 = c
.query_row("SELECT count(*) FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 1);
}
#[test]
fn re_enqueueing_at_higher_priority_promotes() {
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
// The grid scrolls this image into view.
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Interactive, None).unwrap();
let p: i64 = c
.query_row("SELECT priority FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(p, Priority::Interactive as i64);
}
#[test]
fn priority_never_regresses() {
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Interactive, None).unwrap();
// A background sweep must not demote work the user is waiting on.
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
let p: i64 = c
.query_row("SELECT priority FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(p, Priority::Interactive as i64);
}
#[test]
fn claim_takes_highest_priority_first() {
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
enqueue(&c, JobKind::Thumbnail, Some(2), Priority::Interactive, None).unwrap();
enqueue(&c, JobKind::Thumbnail, Some(3), Priority::Prefetch, None).unwrap();
let first = claim_next(&c, 0).unwrap().unwrap();
assert_eq!(first.subject_id, Some(2));
let second = claim_next(&c, 0).unwrap().unwrap();
assert_eq!(second.subject_id, Some(3));
}
#[test]
fn a_claimed_job_is_not_claimed_twice() {
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
assert!(claim_next(&c, 0).unwrap().is_some());
assert!(claim_next(&c, 0).unwrap().is_none());
}
#[test]
fn failure_backs_off_then_becomes_claimable_again() {
let c = db();
enqueue(
&c,
JobKind::FetchPreview,
Some(1),
Priority::Background,
None,
)
.unwrap();
let job = claim_next(&c, 100).unwrap().unwrap();
fail(&c, &job, 100, "network down").unwrap();
// Still backing off.
assert!(claim_next(&c, 100).unwrap().is_none());
// Past the backoff.
assert!(claim_next(&c, 100 + backoff_seconds(job.attempts))
.unwrap()
.is_some());
}
#[test]
fn a_persistently_failing_job_stops_retrying() {
let c = db();
enqueue(
&c,
JobKind::ExtractMetadata,
Some(1),
Priority::Background,
None,
)
.unwrap();
let mut now = 0;
for _ in 0..MAX_ATTEMPTS {
let job = claim_next(&c, now).unwrap().expect("should be claimable");
fail(&c, &job, now, "corrupt file").unwrap();
now += backoff_seconds(job.attempts);
}
// One corrupt file must not stall the queue forever (FR-RAW-4).
assert!(claim_next(&c, now + 100_000).unwrap().is_none());
let state: i64 = c
.query_row("SELECT state FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(state, JobState::Failed as i64);
}
#[test]
fn re_requesting_a_failed_job_gives_it_a_fresh_start() {
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
let mut now = 0;
for _ in 0..MAX_ATTEMPTS {
let job = claim_next(&c, now).unwrap().unwrap();
fail(&c, &job, now, "boom").unwrap();
now += backoff_seconds(job.attempts);
}
// The file changed on disk, so the old failure says nothing about it.
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Interactive, None).unwrap();
let job = claim_next(&c, now).unwrap().expect("retryable again");
assert_eq!(job.attempts, 1);
}
#[test]
fn orphaned_jobs_return_to_pending_on_restart() {
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
claim_next(&c, 0).unwrap().unwrap();
// Process dies here. Android does this routinely.
assert_eq!(recover_orphaned(&c).unwrap(), 1);
assert!(claim_next(&c, 0).unwrap().is_some());
}
#[test]
fn backoff_grows_then_caps() {
assert_eq!(backoff_seconds(0), 0);
assert_eq!(backoff_seconds(1), 1);
assert_eq!(backoff_seconds(3), 4);
assert_eq!(backoff_seconds(100), 300);
}
/// An image row, so a job has a subject that exists.
fn image(c: &Connection, id: i64) {
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', '/lib')
ON CONFLICT DO NOTHING",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (?1, 1, ?2, 0)",
rusqlite::params![id, format!("/lib/{id}.CR3")],
)
.unwrap();
}
#[test]
fn a_runner_claims_only_the_kinds_it_names() {
// The property a filtered claim exists for: work this worker cannot do
// is left untouched — not claimed, not attempted, not failed.
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
enqueue(
&c,
JobKind::FetchOriginal,
Some(1),
Priority::Interactive,
None,
)
.unwrap();
// `FetchOriginal` is the higher priority and would be claimed first by
// an unfiltered claim. It is not this worker's to take.
let job = claim_next_matching(&c, 0, &[JobKind::Thumbnail])
.unwrap()
.expect("the thumbnail is claimable");
assert_eq!(job.kind, JobKind::Thumbnail);
assert!(claim_next_matching(&c, 0, &[JobKind::Thumbnail])
.unwrap()
.is_none());
let (state, attempts): (i64, i64) = c
.query_row(
"SELECT state, attempts FROM jobs WHERE kind = ?1",
[JobKind::FetchOriginal as i64],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(state, JobState::Pending as i64);
assert_eq!(attempts, 0);
}
#[test]
fn a_worker_that_can_do_nothing_claims_nothing() {
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
assert!(claim_next_matching(&c, 0, &[]).unwrap().is_none());
}
#[test]
fn releasing_a_claim_gives_the_attempt_back() {
// A stopped worker has learned nothing about the file, so the claim it
// is handing back must cost nothing.
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
let job = claim_next(&c, 0).unwrap().unwrap();
assert_eq!(job.attempts, 1);
release(&c, &job).unwrap();
let again = claim_next(&c, 0).unwrap().expect("claimable again at once");
assert_eq!(again.attempts, 1, "the release refunded the first attempt");
}
#[test]
fn releasing_a_job_someone_else_has_reclaimed_does_nothing() {
// `attempts` standing in for an owner column. A worker that comes back
// after the recovery pass handed its job to someone else must not
// release a claim that is no longer its to release — which would drop
// the live worker's job back into the queue to be run twice.
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
let stale = claim_next(&c, 0).unwrap().unwrap();
recover_orphaned(&c).unwrap();
let live = claim_next(&c, 0).unwrap().unwrap();
assert_eq!(live.attempts, 2);
release(&c, &stale).unwrap();
let (state, attempts): (i64, i64) = c
.query_row("SELECT state, attempts FROM jobs", [], |r| {
Ok((r.get(0)?, r.get(1)?))
})
.unwrap();
assert_eq!(
state,
JobState::Running as i64,
"the live claim still holds"
);
assert_eq!(attempts, 2, "and its attempt was not refunded for it");
}
#[test]
fn abandoning_skips_the_whole_retry_ladder() {
let c = db();
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
let job = claim_next(&c, 0).unwrap().unwrap();
abandon(&c, job.id, "not an image this build can read").unwrap();
assert!(claim_next(&c, 1_000_000).unwrap().is_none());
let (state, attempts, err): (i64, i64, String) = c
.query_row("SELECT state, attempts, last_error FROM jobs", [], |r| {
Ok((r.get(0)?, r.get(1)?, r.get(2)?))
})
.unwrap();
assert_eq!(state, JobState::Failed as i64);
assert_eq!(attempts, 1, "one attempt, not MAX_ATTEMPTS");
// Kept, not deleted: the user is entitled to see what failed and why.
assert!(err.contains("this build can read"));
}
#[test]
fn jobs_for_a_deleted_photograph_are_reaped() {
// Coalescing keeps the table one row per unit of work; nothing shrank
// it when the work stopped existing.
let c = db();
image(&c, 1);
image(&c, 2);
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
enqueue(&c, JobKind::Thumbnail, Some(2), Priority::Background, None).unwrap();
enqueue(
&c,
JobKind::ExtractMetadata,
Some(2),
Priority::Background,
None,
)
.unwrap();
c.execute("DELETE FROM images WHERE id = 2", []).unwrap();
assert_eq!(reap_orphan_subjects(&c).unwrap(), 2);
let left: i64 = c
.query_row("SELECT subject_id FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(left, 1);
}
#[test]
fn a_folder_scan_is_not_reaped_by_image_ids() {
// `ScanFolder`'s subject is a folder. Joining it against `images`
// would delete it whenever the numbering happened not to collide —
// which, on a fresh library, is almost always.
let c = db();
enqueue(
&c,
JobKind::ScanFolder,
Some(1),
Priority::Background,
Some("/lib/2024"),
)
.unwrap();
assert_eq!(reap_orphan_subjects(&c).unwrap(), 0);
assert!(claim_next(&c, 0).unwrap().is_some());
}
#[test]
fn a_job_kind_from_a_newer_build_is_parked_rather_than_guessed_at() {
// A catalog synced from a device running a later build. Running an
// unknown kind as some arbitrary known one is worse than not running
// it, and the old code silently read every unknown kind as
// `ExtractMetadata`.
let c = db();
c.execute(
"INSERT INTO jobs(kind, subject_id, priority, state) VALUES (99, 1, 0, 0)",
[],
)
.unwrap();
assert!(claim_next(&c, 0).unwrap().is_none());
let (state, err): (i64, String) = c
.query_row("SELECT state, last_error FROM jobs", [], |r| {
Ok((r.get(0)?, r.get(1)?))
})
.unwrap();
assert_eq!(state, JobState::Failed as i64);
assert!(err.contains("99"), "{err}");
}
#[test]
fn counts_say_what_is_left() {
// What a foreground service's notification is built from: a number,
// without reading a hundred thousand rows to find it.
let c = db();
assert_eq!(counts(&c).unwrap(), Counts::default());
for id in 1..=3 {
enqueue(&c, JobKind::Thumbnail, Some(id), Priority::Background, None).unwrap();
}
let job = claim_next(&c, 0).unwrap().unwrap();
abandon(&c, job.id, "nope").unwrap();
claim_next(&c, 0).unwrap().unwrap();
let n = counts(&c).unwrap();
assert_eq!(n.pending, 1);
assert_eq!(n.running, 1);
assert_eq!(n.failed, 1);
assert_eq!(n.outstanding(), 2, "failed work is not outstanding work");
}
#[test]
fn every_kind_is_in_all() {
// `ALL` is what the reap builds its kind filter from, so a kind added
// to the enum and forgotten here would quietly stop being reaped.
for (i, kind) in JobKind::ALL.iter().enumerate() {
assert_eq!(
JobKind::from_i64(i as i64),
Some(*kind),
"ALL is out of step with the discriminants at {i}"
);
}
assert!(JobKind::from_i64(JobKind::ALL.len() as i64).is_none());
}
#[test]
fn only_a_folder_scan_has_a_non_image_subject() {
assert!(!JobKind::ScanFolder.subject_is_image());
for kind in JobKind::ALL.iter().filter(|k| **k != JobKind::ScanFolder) {
assert!(kind.subject_is_image(), "{kind:?}");
}
}
#[test]
fn network_jobs_are_identifiable_for_metered_gating() {
// FR-NC-6: transfers respect unmetered-network and charging
// constraints; local work must not be gated by them.
assert!(JobKind::FetchOriginal.is_network());
assert!(JobKind::FetchPreview.is_network());
assert!(!JobKind::Thumbnail.is_network());
assert!(!JobKind::ExtractMetadata.is_network());
}
}
File diff suppressed because it is too large Load Diff
-615
View File
@@ -1,615 +0,0 @@
//! TRACES: FR-CAT-2 | FR-CAT-4 | FR-CAT-6 | NFR-P1
//! The catalog: a rebuildable index over the library.
//!
//! Not a source of truth. Sidecars next to the images hold the authoritative
//! edit state (ARCH §6.12), and this file is deletable at any time — rebuilt
//! by rescanning sources and reading sidecars. That inversion is deliberate:
//! darktable maintains both a database and sidecars while achieving the
//! reliability of neither.
//!
//! # What lives here
//!
//! - [`schema`] — tables and forward-only migrations
//! - [`scan`] — incremental discovery that prunes unchanged directories
//! - [`walk`] — those decisions driven against real storage, local or SAF
//! - [`query`] — selectors compiled to indexed SQL, windowed for the grid
//! - [`collections`] — the collection tree and membership the UI edits
//! - [`keywords`] — the keyword vocabulary and what it is assigned to
//! - [`faces`] — detected faces, the people they belong to, and who said so
//! - [`bursts`] — frames that are one moment, grouped so they judge as one
//! - [`jobs`] — the durable background work queue
//! - [`runner`] — the thing that drains it, driven by whoever owns the thread
//! - [`trash`] — soft delete to a folder, then permanent delete
//! - [`merge`] / [`sync`] — cross-device merging of collections and keywords
//! - [`recovery`] — backups, and the two offers made when this file is damaged
//!
//! # The one thing everything is designed around
//!
//! **Work is proportional to what changed, or to what the user is looking at —
//! never to library size.** A 50k-image library that has not changed costs one
//! metadata probe per folder to verify (§scan), no thumbnails to regenerate
//! (§jobs coalescing), and no rule evaluation per grid cell (materialised
//! `tier_desired`).
use std::path::Path;
use dr_types::{Availability, ImageId};
use rusqlite::Connection;
pub mod albums;
mod backfilled;
pub mod bursts;
pub mod cache;
pub mod collections;
pub mod dedup;
pub mod dedup_people;
pub mod duplicates;
pub mod error;
pub mod face_shard;
pub mod faces;
pub mod jobs;
pub mod keywords;
pub mod merge;
pub mod query;
pub mod rating;
pub mod recovery;
pub mod runner;
pub mod scan;
pub mod schema;
pub mod sync;
pub mod trash;
pub mod walk;
pub use albums::{Album, AlbumId, Place};
pub use cache::{Budget, Cache, DEFAULT_BUDGET_BYTES};
pub use collections::{Collection, CollectionKind, TreeRow};
pub use dedup::{seen_by_content, seen_by_metadata, set_content_hash};
pub use error::CatalogError;
pub use face_shard::{FaceShardStore, SharedFace};
pub use faces::{Calibration, DetectedFace, Face, FaceId, FaceUpdate, Person, PersonId};
pub use jobs::{Job, JobKind, Priority};
pub use keywords::{Coverage, Keyword, KeywordId, SelectionKeyword};
pub use merge::MergeReport;
pub use query::{Query, Sort};
pub use rating::{Judgement, MAX_RATING};
pub use recovery::Backup;
// Not `runner::Budget`: `cache::Budget` already owns that name here and
// means something else entirely (bytes on disk, not jobs in a slot).
// Callers spell the work budget `runner::Budget`, where it is unambiguous.
pub use runner::{DrainReport, JobHandler, Outcome, Runner};
pub use scan::{DirAction, DirState, EntryAction, ScanOutcome};
pub use trash::{TrashedImage, TRASH_DIR};
pub use walk::{ensure_root, mark_root_offline, scan_root, RootKind, ScanProgress, ScanReport};
/// One row of the library grid.
///
/// Exactly what a cell draws and nothing more — no join per cell, and
/// availability reads a materialised column rather than evaluating cache rules
/// (ARCH §9.5).
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct GridRow {
pub id: ImageId,
pub name: String,
pub availability: Availability,
/// UTC seconds. `None` until EXIF has been read.
pub captured_at: Option<i64>,
/// Minutes east of UTC, for rendering the photographer's local time.
pub captured_offset: Option<i32>,
/// 0 = nothing, 1 = stat-only, 2 = full EXIF.
pub metadata_state: u8,
}
/// A count of images in one time bucket, for the timeline scrubber.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct TimeBucket {
/// UTC seconds at the bucket's start.
pub start: i64,
pub count: u32,
}
/// Time bucket size.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Granularity {
Year,
Month,
Day,
Hour,
}
impl Granularity {
/// SQLite `strftime` format that collapses a timestamp to this bucket.
///
/// Applied to **local** time, not UTC: "everything from 3 August" means
/// the photographer's 3 August, which is why `captured_offset` is stored
/// alongside the UTC timestamp.
/// Public so a caller that must build its own bucketing query — one
/// joining collection membership, say — buckets identically to
/// [`Catalog::timeline_range`] rather than reimplementing the format.
pub fn strftime(self) -> &'static str {
match self {
Granularity::Year => "%Y",
Granularity::Month => "%Y-%m",
Granularity::Day => "%Y-%m-%d",
Granularity::Hour => "%Y-%m-%dT%H",
}
}
/// A sensible bucket size for a span of seconds, so the UI need not guess.
///
/// # Chosen by how many bars it produces, not by fixed cut-offs
///
/// This used to be four thresholds on the span, which reads sensibly and
/// behaves badly under zoom. Each zoom step halves the span, so the bar
/// count halves with it until a threshold is crossed — a fifteen-year
/// library went 15 bars, 8, then 46, 23, 11, and finally *6*. Zooming in
/// made the picture coarser, which is the opposite of what zooming is for.
///
/// So the choice is made on the axis's terms: of the four bucket sizes,
/// take the one whose bar count comes nearest [`Self::TARGET_BARS`]. The
/// count then stays in the same neighbourhood at every zoom level, and
/// each step in genuinely shows finer structure rather than the same
/// structure drawn wider.
///
/// Nearest in *ratio*, not in difference: the counts available for a given
/// span are orders of magnitude apart — a span is either about 4 years or
/// about 48 months — and on a linear measure the larger count always looks
/// further away, which would bias every choice towards too few bars.
pub fn for_span(seconds: i64) -> Self {
Self::for_bucket(seconds.max(1) / Self::TARGET_BARS)
}
/// The calendar unit nearest a bucket of `seconds`, for *labelling* one.
///
/// Split out from [`Self::for_span`] because the axis no longer buckets by
/// calendar unit at all — it divides the visible span into a fixed number
/// of equal bins (see `LibrarySettings::timeline_bars`). What is still
/// wanted is the unit a bin is closest to, so a bin of about a day is
/// labelled as a date and one of about a year as a year. Asked directly
/// rather than derived from the span, because the bin count is now the
/// user's rather than this module's target.
pub fn for_bucket(seconds: i64) -> Self {
let seconds = seconds.max(1) as f64;
// Finest first, so that when two options are equally far from the
// target the finer one wins: `min_by` keeps the first minimum it saw,
// and more detail is the better failure.
[
Granularity::Hour,
Granularity::Day,
Granularity::Month,
Granularity::Year,
]
.into_iter()
.min_by(|a, b| {
let cost = |g: Granularity| {
// How far off, measured multiplicatively: twice as long and
// half as long are equally wrong.
//
// Deliberately not clamped. A bucket shorter than the unit
// scores *worse* the coarser the unit, which is what makes an
// hour of photographs pick hourly bars instead of every option
// tying at "one bucket" and the coarsest winning.
(seconds / g.approx_seconds() as f64).ln().abs()
};
cost(*a)
.partial_cmp(&cost(*b))
// Ties cannot arise from real spans, but a NaN would; falling
// back to the coarser option keeps the axis drawable.
.unwrap_or(std::cmp::Ordering::Equal)
})
.unwrap_or(Granularity::Day)
}
/// How many bars the timeline wants across its axis.
///
/// Not a hard count — the bucket sizes are calendar units, so the actual
/// number lands where the calendar puts it. It is the figure the choice
/// aims at: enough bars that a busy fortnight is visibly busier than a
/// quiet one, few enough that each is wide enough to hit with a finger.
const TARGET_BARS: i64 = 40;
/// Nominal length of one bucket, for choosing between them.
///
/// Approximate on purpose: months and years vary and it does not matter
/// here, because this only ranks four options that are a factor of ~12 or
/// ~30 apart. The exact boundaries come from `strftime` on the real dates.
fn approx_seconds(self) -> i64 {
const DAY: i64 = 86_400;
match self {
Granularity::Year => 365 * DAY,
Granularity::Month => 30 * DAY,
Granularity::Day => DAY,
Granularity::Hour => 3600,
}
}
}
/// A connection to the catalog.
pub struct Catalog {
conn: Connection,
}
impl Catalog {
/// Open or create a catalog, migrating it forward if needed.
///
/// Does **not** verify the file — see [`Self::open_verified`], and
/// [`recovery`] for why the check is bound to startup rather than to every
/// open. Damage this trips over on the way past is still reported as
/// [`CatalogError::Corrupt`] rather than as a stray SQLite error.
pub fn open(path: &Path) -> Result<Self, CatalogError> {
let conn = Connection::open(path)?;
schema::configure(&conn)?;
// NFR-R2, and the reason it is *here*: a migration is the one routine
// operation that rewrites table structure, so it is the likeliest way
// this file becomes unreadable — and afterwards there is no
// pre-migration state left to copy. A failure to take the copy is
// logged rather than raised: a full disk must not be the thing that
// makes a library unopenable.
if let Err(e) = recovery::backup_before_migration(&conn, path) {
log::warn!("could not back up before migrating: {e}");
}
let from = schema::migrate(&conn)?;
// A migration adds a column; it cannot know what the value should be
// for rows that already existed. Backfilling on open is what stops
// those rows being silently partial.
//
// Once per catalog state rather than once per open (NFR-P9): every
// worker thread opens its own connection, and a develop landing made
// five, each paying the whole backfill to confirm nothing had changed.
// [`backfilled`] says what "changed" means and why it is enough. A
// migration always backfills, stamp or no stamp.
let stamp = backfilled::stamp(&conn, path)?;
if from < schema::SCHEMA_VERSION || !backfilled::is_current(path, &stamp) {
for (what, n) in schema::backfill(&conn)? {
log::info!("backfilled {what} for {n} row(s) (schema was v{from})");
}
backfilled::record(path, stamp);
}
Ok(Catalog { conn })
}
/// TRACES: NFR-R6
/// Open a catalog, checking the file first.
///
/// What startup calls. On [`CatalogError::Corrupt`] the caller has a user
/// in front of it and must make the two offers [`recovery`] describes,
/// rather than reporting a SQLite message on a banner and carrying on into
/// a scan that would write into the damage.
///
/// Checked *before* opening rather than after, because opening runs
/// migrations: a damaged catalog that happens to have an intact header
/// would otherwise be migrated — rewriting structure on top of structure
/// that is already wrong — before anybody asked whether it was sound.
pub fn open_verified(path: &Path) -> Result<Self, CatalogError> {
// A catalog that is not there yet is not damaged; `open` creates it.
if path.is_file() {
recovery::check_file(path)?;
}
Self::open(path)
}
/// An in-memory catalog, for tests and for a throwaway import preview.
pub fn in_memory() -> Result<Self, CatalogError> {
let conn = Connection::open_in_memory()?;
schema::configure(&conn)?;
schema::migrate(&conn)?;
schema::backfill(&conn)?;
Ok(Catalog { conn })
}
/// Escape hatch for modules that need raw access. Not part of the UI-facing
/// surface.
pub fn connection(&self) -> &Connection {
&self.conn
}
/// How many images match.
///
/// Returned alongside the first window so the grid can size its scrollbar
/// and paint in one round trip.
pub fn count(&self, q: &Query, now: i64) -> Result<usize, CatalogError> {
let c = query::compile(&q.filter, now);
let sql = query::count_sql(&c);
let n: i64 =
self.conn
.query_row(&sql, rusqlite::params_from_iter(c.params.iter()), |r| {
r.get(0)
})?;
Ok(n as usize)
}
/// Fetch one window of results.
///
/// Never returns the whole catalog: FR-CAT-4 requires memory bounded
/// independently of library size.
pub fn window(
&self,
q: &Query,
range: std::ops::Range<usize>,
now: i64,
) -> Result<Vec<GridRow>, CatalogError> {
let c = query::compile(&q.filter, now);
let sql = query::window_sql(q, &c);
let mut params = c.params.clone();
params.push(rusqlite::types::Value::Integer(range.len() as i64));
params.push(rusqlite::types::Value::Integer(range.start as i64));
let mut stmt = self.conn.prepare(&sql)?;
let rows = stmt
.query_map(rusqlite::params_from_iter(params.iter()), |r| {
let source_ref: String = r.get(1)?;
let avail: i64 = r.get(2)?;
Ok(GridRow {
id: ImageId(r.get::<_, i64>(0)? as u64),
name: source_ref
.rsplit(['/', ':'])
.next()
.unwrap_or(&source_ref)
.to_string(),
availability: decode_availability(avail),
captured_at: r.get(3)?,
captured_offset: r.get::<_, Option<i64>>(4)?.map(|v| v as i32),
metadata_state: r.get::<_, i64>(5)? as u8,
})
})?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// Counts per time bucket, for the timeline scrubber.
///
/// One grouped aggregate over the `images_captured` index — not 50k rows
/// handed to the UI to bucket itself.
pub fn timeline(
&self,
q: &Query,
g: Granularity,
now: i64,
) -> Result<Vec<TimeBucket>, CatalogError> {
let c = query::compile(&q.filter, now);
// Bucketed in local time: captured_offset is minutes east of UTC, and
// NULL falls back to UTC rather than dropping the row.
let sql = format!(
"SELECT min(captured_at) AS start,
count(*) AS n
FROM images
-- A shadowed JPEG is the same frame as its RAW; counting both
-- would double every paired shot in the histogram.
WHERE {} AND captured_at IS NOT NULL AND shadowed_by IS NULL
GROUP BY strftime('{}', captured_at + coalesce(captured_offset, 0) * 60,
'unixepoch')
ORDER BY start ASC",
c.where_sql,
g.strftime()
);
let mut stmt = self.conn.prepare(&sql)?;
let rows = stmt
.query_map(rusqlite::params_from_iter(c.params.iter()), |r| {
Ok(TimeBucket {
start: r.get(0)?,
count: r.get::<_, i64>(1)? as u32,
})
})?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// Counts per time bucket, bounded to a date range.
///
/// What a zoomed timeline needs: [`timeline`](Self::timeline) always spans
/// the whole library, so zooming in would return the same coarse buckets
/// with the ends cropped rather than finer detail over a narrower span.
pub fn timeline_range(
&self,
q: &Query,
g: Granularity,
from: i64,
to: i64,
now: i64,
) -> Result<Vec<TimeBucket>, CatalogError> {
let c = query::compile(&q.filter, now);
let sql = format!(
"SELECT min(captured_at) AS start,
count(*) AS n
FROM images
WHERE {} AND captured_at IS NOT NULL AND shadowed_by IS NULL
AND captured_at >= ?{} AND captured_at <= ?{}
GROUP BY strftime('{}', captured_at + coalesce(captured_offset, 0) * 60,
'unixepoch')
ORDER BY start ASC",
c.where_sql,
c.params.len() + 1,
c.params.len() + 2,
g.strftime()
);
let mut params = c.params.clone();
params.push(rusqlite::types::Value::Integer(from));
params.push(rusqlite::types::Value::Integer(to));
let mut stmt = self.conn.prepare(&sql)?;
let rows = stmt
.query_map(rusqlite::params_from_iter(params.iter()), |r| {
Ok(TimeBucket {
start: r.get(0)?,
count: r.get::<_, i64>(1)? as u32,
})
})?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// Merge a downloaded remote catalog's collections into this one.
///
/// See [`sync`] for why only collections cross over.
pub fn merge_remote_catalog(&self, remote: &Path) -> Result<MergeReport, CatalogError> {
sync::merge_remote(&self.conn, remote)
}
/// Write a consistent snapshot ready to upload.
pub fn snapshot_for_upload(&self, dest: &Path) -> Result<(), CatalogError> {
sync::snapshot_for_upload(&self.conn, dest)
}
}
fn decode_availability(v: i64) -> Availability {
match v {
1 => Availability::Preview,
2 => Availability::Original,
3 => Availability::Offline,
_ => Availability::MetadataOnly,
}
}
#[cfg(test)]
mod tests {
use super::*;
use dr_types::Selector;
fn seeded() -> Catalog {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
// Three images across two days, one with no EXIF read yet.
for (id, name, captured, state) in [
(1i64, "a.CR3", Some(1_000_000i64), 2i64),
(2, "b.CR3", Some(1_100_000), 2),
(3, "c.CR3", None, 1),
] {
c.execute(
"INSERT INTO images(id, root_id, source_ref, captured_at, metadata_state, added_at)
VALUES (?1, 1, ?2, ?3, ?4, 0)",
rusqlite::params![id, name, captured, state],
)
.unwrap();
}
cat
}
#[test]
fn count_and_window_agree() {
let cat = seeded();
let q = Query::default();
assert_eq!(cat.count(&q, 0).unwrap(), 3);
assert_eq!(cat.window(&q, 0..10, 0).unwrap().len(), 3);
}
#[test]
fn window_is_bounded_by_the_requested_range() {
// FR-CAT-4: memory independent of catalog size.
let cat = seeded();
let rows = cat.window(&Query::default(), 0..2, 0).unwrap();
assert_eq!(rows.len(), 2);
}
#[test]
fn paging_covers_every_row_exactly_once() {
let cat = seeded();
let q = Query::default();
let mut seen = Vec::new();
for start in (0..3).step_by(2) {
seen.extend(cat.window(&q, start..start + 2, 0).unwrap());
}
let mut ids: Vec<u64> = seen.iter().map(|r| r.id.0).collect();
ids.sort_unstable();
assert_eq!(ids, vec![1, 2, 3]);
}
#[test]
fn an_image_without_capture_time_sorts_last_not_first() {
// Otherwise a freshly scanned library leads with whatever has not been
// read yet, which looks like corruption to the user.
let cat = seeded();
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
assert_eq!(rows.last().unwrap().id, ImageId(3));
}
#[test]
fn metadata_state_reaches_the_grid() {
// The grid needs it to distinguish "no photos on this date" from
// "EXIF not read yet" (FR-NC-6c's honesty principle).
let cat = seeded();
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
let pending = rows.iter().find(|r| r.id == ImageId(3)).unwrap();
assert_eq!(pending.metadata_state, 1);
}
#[test]
fn a_filter_narrows_the_count() {
let cat = seeded();
let q = Query {
filter: Selector::Text("a.CR3".into()),
..Default::default()
};
assert_eq!(cat.count(&q, 0).unwrap(), 1);
}
#[test]
fn timeline_buckets_and_skips_unread_images() {
let cat = seeded();
let buckets = cat
.timeline(&Query::default(), Granularity::Day, 0)
.unwrap();
// Two images with timestamps, one day apart in UTC; the third has no
// capture time and cannot be placed on a timeline at all.
let total: u32 = buckets.iter().map(|b| b.count).sum();
assert_eq!(total, 2);
}
#[test]
fn timeline_granularity_follows_the_span() {
const DAY: i64 = 86_400;
// Chosen by how many bars it makes, not by fixed cut-offs — see
// `for_span`. Ten years of yearly bars is ten bars, which says almost
// nothing about a library; monthly is 122, which is a shape.
assert_eq!(Granularity::for_span(10 * 365 * DAY), Granularity::Month);
assert_eq!(Granularity::for_span(120 * DAY), Granularity::Day);
assert_eq!(Granularity::for_span(10 * DAY), Granularity::Day);
assert_eq!(Granularity::for_span(3600), Granularity::Hour);
// The property the target exists for: zooming in never coarsens the
// axis. Under the old thresholds a fifteen-year library went 15 bars,
// then 8, then 46, 23, 11 — finer spans drawn with wider bars.
let mut span = 15 * 365 * DAY;
let mut previous = Granularity::for_span(span).approx_seconds();
for _ in 0..10 {
span /= 2;
let bucket = Granularity::for_span(span).approx_seconds();
assert!(
bucket <= previous,
"halving the span to {span}s coarsened the bucket \
from {previous}s to {bucket}s"
);
previous = bucket;
}
// And a span shorter than any bucket still picks the finest, rather
// than every option tying at one bar and the coarsest winning.
assert_eq!(Granularity::for_span(60), Granularity::Hour);
assert_eq!(Granularity::for_span(1), Granularity::Hour);
}
#[test]
fn names_are_derived_for_both_paths_and_saf_ids() {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'saf', 'tree')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (1, 1, 'primary:DCIM/Camera/IMG_1.CR3', 0)",
[],
)
.unwrap();
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
assert_eq!(rows[0].name, "IMG_1.CR3");
}
}
File diff suppressed because it is too large Load Diff
-508
View File
@@ -1,508 +0,0 @@
//! TRACES: FR-CAT-4 | FR-CAT-6
//! Compiling a [`Selector`] into indexed SQL, and windowing the result.
//!
//! The UI never assembles SQL — it hands over a [`Query`] and receives a
//! window. Two properties matter:
//!
//! 1. **Nothing user-supplied is interpolated into SQL text.** Every value
//! binds as a parameter; `LIKE` patterns have their wildcards escaped.
//! 2. **Predicates hit indices.** Filtering 50k images must stay interactive
//! (FR-CAT-6), which means no expression over a column that would defeat
//! its index.
use dr_types::{Availability, ColourLabel, DateSelector, FlagState, Selector};
use rusqlite::types::Value;
/// What to show, and in what order.
#[derive(Debug, Clone)]
pub struct Query {
pub filter: Selector,
pub sort: Sort,
pub descending: bool,
}
impl Default for Query {
fn default() -> Self {
Query {
filter: Selector::All,
sort: Sort::CapturedAt,
descending: true,
}
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Sort {
CapturedAt,
Added,
FileName,
Rating,
/// Manual order within a collection. Falls back to capture time where the
/// query is not scoped to one collection, since position is meaningless
/// outside it.
CollectionPosition,
}
impl Sort {
/// The ORDER BY fragment. Fixed strings — never user input.
///
/// Capture time sorts NULLs last regardless of direction: an image whose
/// EXIF has not been read yet (metadata_state 1) should not lead the grid
/// simply because its timestamp is unknown.
fn sql(self, descending: bool) -> &'static str {
match (self, descending) {
(Sort::CapturedAt, false) => {
"ORDER BY images.captured_at IS NULL, images.captured_at ASC, images.id ASC"
}
(Sort::CapturedAt, true) => {
"ORDER BY images.captured_at IS NULL, images.captured_at DESC, images.id DESC"
}
(Sort::Added, false) => "ORDER BY images.added_at ASC, images.id ASC",
(Sort::Added, true) => "ORDER BY images.added_at DESC, images.id DESC",
(Sort::FileName, false) => "ORDER BY images.source_ref ASC, images.id ASC",
(Sort::FileName, true) => "ORDER BY images.source_ref DESC, images.id DESC",
(Sort::Rating, false) => "ORDER BY v.rating ASC, images.id ASC",
(Sort::Rating, true) => "ORDER BY v.rating DESC, images.id DESC",
(Sort::CollectionPosition, false) => {
"ORDER BY cm.position IS NULL, cm.position ASC, images.captured_at ASC"
}
(Sort::CollectionPosition, true) => {
"ORDER BY cm.position IS NULL, cm.position DESC, images.captured_at DESC"
}
}
}
/// Whether this sort needs the default-version join.
fn needs_version(self) -> bool {
matches!(self, Sort::Rating)
}
/// Whether this sort needs a collection-membership join.
fn needs_membership(self) -> bool {
matches!(self, Sort::CollectionPosition)
}
}
/// A compiled WHERE clause plus its bound parameters.
///
/// Kept separate from the statement so `count` and `window` can share one
/// compilation.
#[derive(Debug, Default)]
pub struct Compiled {
pub where_sql: String,
pub params: Vec<Value>,
/// True if the filter depends on capture time, and therefore on EXIF that
/// a freshly scanned library may not have read yet. The UI surfaces this
/// rather than silently under-reporting.
pub needs_capture_time: bool,
}
/// Compile a selector to SQL against the `images` table.
///
/// `now` is passed rather than read from the clock so a rolling window is
/// reproducible in tests and consistent across one query.
pub fn compile(filter: &Selector, now: i64) -> Compiled {
let mut params = Vec::new();
let sql = if filter.is_unfiltered() {
"1".to_string()
} else {
emit(filter, now, &mut params)
};
Compiled {
where_sql: sql,
params,
needs_capture_time: filter.needs_capture_time(),
}
}
fn emit(s: &Selector, now: i64, p: &mut Vec<Value>) -> String {
match s {
Selector::All => "1".into(),
Selector::Collection(id) => {
p.push(Value::Integer(id.0 as i64));
format!(
"EXISTS (SELECT 1 FROM collection_members m
WHERE m.image_id = images.id AND m.collection_id = ?{})",
p.len()
)
}
Selector::Folder {
root,
path,
recursive,
} => {
p.push(Value::Integer(root.0 as i64));
let root_ix = p.len();
if *recursive {
// Prefix match on the folder path. `like_prefix` escapes the
// pattern metacharacters, so a folder literally named "50%"
// matches itself and not everything.
p.push(Value::Text(like_prefix(path)));
format!(
"images.folder_id IN (
SELECT id FROM folders
WHERE root_id = ?{root_ix}
AND (path = ?{p} OR path LIKE ?{p} || '/%' ESCAPE '\\'))",
p = p.len()
)
} else {
p.push(Value::Text(path.clone()));
format!(
"images.folder_id IN (
SELECT id FROM folders WHERE root_id = ?{root_ix} AND path = ?{})",
p.len()
)
}
}
Selector::DateRange(d) => emit_date(d, now, p),
Selector::Rating { min } => {
p.push(Value::Integer(*min as i64));
format!("{} >= ?{}", default_version_scalar("rating"), p.len())
}
Selector::Label(l) => {
p.push(Value::Integer(label_code(*l)));
format!("{} = ?{}", default_version_scalar("label"), p.len())
}
Selector::Flag(f) => {
p.push(Value::Integer(flag_code(*f)));
format!("{} = ?{}", default_version_scalar("flag"), p.len())
}
Selector::Keyword(k) => {
p.push(Value::Text(k.clone()));
format!(
"EXISTS (SELECT 1 FROM keywords kw
JOIN versions kv ON kv.id = kw.version_id
WHERE kv.image_id = images.id AND kw.keyword = ?{})",
p.len()
)
}
Selector::Camera(c) => {
p.push(Value::Text(c.clone()));
format!("images.camera = ?{}", p.len())
}
Selector::Lens(l) => {
p.push(Value::Text(l.clone()));
format!("images.lens = ?{}", p.len())
}
Selector::IsoRange { min, max } => {
p.push(Value::Integer(*min as i64));
let lo = p.len();
p.push(Value::Integer(*max as i64));
format!("images.iso BETWEEN ?{lo} AND ?{}", p.len())
}
Selector::Availability(a) => {
p.push(Value::Integer(availability_code(*a)));
format!("images.availability = ?{}", p.len())
}
Selector::Text(t) => {
// Substring over filename and keywords. A LIKE scan is adequate at
// 50k; if free text over title and description becomes a real
// workflow, FTS5 is the answer and it is additive.
p.push(Value::Text(format!("%{}%", escape_like(t))));
let ix = p.len();
format!(
"(images.source_ref LIKE ?{ix} ESCAPE '\\'
OR EXISTS (SELECT 1 FROM keywords kw
JOIN versions kv ON kv.id = kw.version_id
WHERE kv.image_id = images.id
AND kw.keyword LIKE ?{ix} ESCAPE '\\'))"
)
}
// An empty conjunction is vacuously true; an empty disjunction matches
// nothing. Both arise from a UI that lets every term be cleared, and
// conflating them would show the whole library when the user meant the
// opposite.
Selector::All_(v) if v.is_empty() => "1".into(),
Selector::Any(v) if v.is_empty() => "0".into(),
Selector::All_(v) => join(v, " AND ", now, p),
Selector::Any(v) => join(v, " OR ", now, p),
Selector::Not(inner) => format!("NOT ({})", emit(inner, now, p)),
}
}
fn join(items: &[Selector], op: &str, now: i64, p: &mut Vec<Value>) -> String {
let parts: Vec<String> = items.iter().map(|s| emit(s, now, p)).collect();
format!("({})", parts.join(op))
}
fn emit_date(d: &DateSelector, now: i64, p: &mut Vec<Value>) -> String {
match d {
DateSelector::Between { from, to } => {
p.push(Value::Integer(*from));
let lo = p.len();
p.push(Value::Integer(*to));
// Half-open, so adjacent ranges neither overlap nor gap.
format!(
"(images.captured_at >= ?{lo} AND images.captured_at < ?{})",
p.len()
)
}
DateSelector::Rolling { days } => {
let from = now - (*days as i64) * 86_400;
p.push(Value::Integer(from));
format!("images.captured_at >= ?{}", p.len())
}
DateSelector::CollectionSpan(id) => {
p.push(Value::Integer(id.0 as i64));
let ix = p.len();
format!(
"images.captured_at BETWEEN
(SELECT min(i2.captured_at) FROM images i2
JOIN collection_members m2 ON m2.image_id = i2.id
WHERE m2.collection_id = ?{ix})
AND (SELECT max(i2.captured_at) FROM images i2
JOIN collection_members m2 ON m2.image_id = i2.id
WHERE m2.collection_id = ?{ix})"
)
}
}
}
/// Rating, label, and flag live on the *default* version, not the image.
///
/// A correlated subquery rather than a join, so these compose inside `OR` and
/// `NOT` without the join multiplying rows.
fn default_version_scalar(col: &str) -> String {
format!(
"(SELECT dv.{col} FROM versions dv
WHERE dv.image_id = images.id AND dv.is_default = 1 LIMIT 1)"
)
}
/// Escape LIKE metacharacters so a literal `%` or `_` in user text matches
/// itself. Paired with `ESCAPE '\'` in every LIKE that uses it.
fn escape_like(s: &str) -> String {
let mut out = String::with_capacity(s.len());
for c in s.chars() {
if matches!(c, '%' | '_' | '\\') {
out.push('\\');
}
out.push(c);
}
out
}
fn like_prefix(path: &str) -> String {
escape_like(path.trim_end_matches('/'))
}
fn label_code(l: ColourLabel) -> i64 {
crate::rating::label_code(l)
}
fn flag_code(f: FlagState) -> i64 {
match f {
FlagState::Unflagged => 0,
FlagState::Pick => 1,
FlagState::Reject => 2,
}
}
/// The stored form of an availability. Shared with [`crate::walk`], which
/// writes the column this reads — two spellings of the same mapping would
/// filter for a state nothing ever writes.
pub(crate) fn availability_code(a: Availability) -> i64 {
match a {
Availability::MetadataOnly => 0,
Availability::Preview => 1,
Availability::Original => 2,
Availability::Offline => 3,
}
}
/// Build the full SELECT for a window of results.
///
/// Joins are added only where the sort needs them, so an unsorted-by-rating
/// grid query touches one table.
pub fn window_sql(q: &Query, compiled: &Compiled) -> String {
let mut joins = String::new();
if q.sort.needs_version() {
joins.push_str(" LEFT JOIN versions v ON v.image_id = images.id AND v.is_default = 1");
}
if q.sort.needs_membership() {
// Only meaningful when the filter scopes to one collection; elsewhere
// position is NULL and the sort falls through to capture time.
joins.push_str(" LEFT JOIN collection_members cm ON cm.image_id = images.id");
}
format!(
"SELECT images.id, images.source_ref, images.availability, images.captured_at, \
images.captured_offset, images.metadata_state \
FROM images{joins} WHERE {} {} LIMIT ? OFFSET ?",
compiled.where_sql,
q.sort.sql(q.descending)
)
}
/// Build the COUNT for the same filter.
pub fn count_sql(compiled: &Compiled) -> String {
format!("SELECT count(*) FROM images WHERE {}", compiled.where_sql)
}
#[cfg(test)]
mod tests {
use super::*;
use dr_types::{CollectionId, RootId};
#[test]
fn unfiltered_compiles_to_a_constant() {
let c = compile(&Selector::All, 0);
assert_eq!(c.where_sql, "1");
assert!(c.params.is_empty());
}
#[test]
fn empty_conjunction_and_disjunction_differ() {
// The distinction that decides whether clearing a filter shows
// everything or nothing.
assert_eq!(compile(&Selector::All_(vec![]), 0).where_sql, "1");
assert_eq!(compile(&Selector::Any(vec![]), 0).where_sql, "0");
}
#[test]
fn values_bind_rather_than_interpolate() {
// The injection guard: a hostile keyword must appear in params, never
// in SQL text.
let evil = "'; DROP TABLE images; --";
let c = compile(&Selector::Keyword(evil.into()), 0);
assert!(!c.where_sql.contains("DROP"));
assert_eq!(c.params, vec![Value::Text(evil.into())]);
}
#[test]
fn like_metacharacters_are_escaped() {
// A search for "50%" must not match everything containing "50".
let c = compile(&Selector::Text("50%".into()), 0);
assert_eq!(c.params, vec![Value::Text("%50\\%%".into())]);
assert!(c.where_sql.contains("ESCAPE"));
}
#[test]
fn a_backslash_in_search_text_is_itself_escaped() {
let c = compile(&Selector::Text("a\\b".into()), 0);
assert_eq!(c.params, vec![Value::Text("%a\\\\b%".into())]);
}
#[test]
fn rolling_window_resolves_against_supplied_now() {
// Passed in rather than read from the clock, so the window is stable
// across one query and reproducible in a test.
let now = 1_000_000i64;
let c = compile(
&Selector::DateRange(DateSelector::Rolling { days: 90 }),
now,
);
assert_eq!(c.params, vec![Value::Integer(now - 90 * 86_400)]);
}
#[test]
fn between_is_half_open() {
let c = compile(
&Selector::DateRange(DateSelector::Between { from: 10, to: 20 }),
0,
);
// Half-open so adjacent day buckets neither overlap nor leave a gap.
assert!(c.where_sql.contains(">= ?1"));
assert!(c.where_sql.contains("< ?2"));
}
#[test]
fn nested_composition_numbers_parameters_in_order() {
let s = Selector::All_(vec![
Selector::Rating { min: 4 },
Selector::Any(vec![
Selector::Camera("X-T5".into()),
Selector::Not(Box::new(Selector::Lens("XF 35".into()))),
]),
]);
let c = compile(&s, 0);
assert_eq!(
c.params,
vec![
Value::Integer(4),
Value::Text("X-T5".into()),
Value::Text("XF 35".into()),
]
);
assert!(c.where_sql.contains("?1"));
assert!(c.where_sql.contains("?2"));
assert!(c.where_sql.contains("?3"));
}
#[test]
fn recursive_folder_matches_the_folder_itself_and_below() {
let c = compile(
&Selector::Folder {
root: RootId(1),
path: "2026/08".into(),
recursive: true,
},
0,
);
// Both branches: the folder's own images and those in subfolders.
assert!(c.where_sql.contains("path = ?2"));
assert!(c.where_sql.contains("|| '/%'"));
}
#[test]
fn collection_span_binds_its_id_once_and_reuses_it() {
let c = compile(
&Selector::DateRange(DateSelector::CollectionSpan(CollectionId(7))),
0,
);
assert_eq!(c.params, vec![Value::Integer(7)]);
}
#[test]
fn capture_time_dependency_is_reported() {
let c = compile(&Selector::DateRange(DateSelector::Rolling { days: 7 }), 0);
assert!(c.needs_capture_time);
let c = compile(&Selector::Rating { min: 5 }, 0);
assert!(!c.needs_capture_time);
}
#[test]
fn capture_sort_puts_unknown_timestamps_last_in_both_directions() {
// An image whose EXIF has not been read yet must not lead the grid
// just because its timestamp is NULL.
assert!(Sort::CapturedAt.sql(true).contains("IS NULL"));
assert!(Sort::CapturedAt.sql(false).contains("IS NULL"));
}
#[test]
fn window_sql_joins_only_when_the_sort_needs_it() {
let c = compile(&Selector::All, 0);
let plain = window_sql(
&Query {
filter: Selector::All,
sort: Sort::CapturedAt,
descending: true,
},
&c,
);
assert!(!plain.contains("JOIN"));
let rated = window_sql(
&Query {
filter: Selector::All,
sort: Sort::Rating,
descending: true,
},
&c,
);
assert!(rated.contains("JOIN versions"));
}
}
File diff suppressed because it is too large Load Diff
-757
View File
@@ -1,757 +0,0 @@
//! TRACES: NFR-R2 | NFR-R6
//! What to do once the index is already damaged.
//!
//! # Why this can be a small module
//!
//! Because of a property the rest of the catalog was built to keep: the
//! catalog is an *index*, not a source of truth (ARCH §6.12, invariant
//! §5.2.4). Sidecars beside the images hold the authoritative ratings,
//! keywords and edit graphs, for every catalogued image and whether or not a
//! remote account exists (FR-CAT-8). So the worst outcome available here is a
//! rescan — expensive, but not a loss.
//!
//! That is the second offer. The first is cheaper and loses nothing at all: a
//! backup, restored.
//!
//! # The one thing a rebuild does not recover
//!
//! **Collections.** A manual collection is a set of images the user assembled
//! by hand and nothing in the filesystem records it (`docs/dev/catalog.md` §8.1) —
//! which is the whole reason the catalog file itself syncs. So the two offers
//! are not interchangeable, and the interface must not present them as if they
//! were: a restore keeps the user's collections, a rebuild does not.
//!
//! # When the check runs, and when it does not
//!
//! [`integrity_check`] reads every page of the database. That is affordable
//! once, at startup, where a failure has a user in front of it who can answer
//! a question — and it is *not* affordable on every [`Catalog::open`], which
//! this application does per background task, dozens of times a session. So
//! the check is bound to [`Catalog::open_verified`] rather than to `open`,
//! and the cheap half of the story — classifying `SQLITE_CORRUPT` and
//! `SQLITE_NOTADB` as [`CatalogError::Corrupt`] — happens for free on every
//! query through [`crate::error`]'s conversion. A background job that trips
//! over the damage first therefore reports the same thing the startup check
//! would have.
//!
//! [`Catalog::open`]: crate::Catalog::open
//! [`Catalog::open_verified`]: crate::Catalog::open_verified
use std::path::{Path, PathBuf};
use std::time::{SystemTime, UNIX_EPOCH};
use rusqlite::Connection;
use crate::error::CatalogError;
use crate::schema;
/// Directory backups live in, relative to the catalog file.
///
/// Beside the catalog rather than in the cache directory, and that is the
/// point of the choice: this is the copy the user falls back on, and a cache
/// is a place the operating system is entitled to empty without asking
/// (see `library::data_root` for the same reasoning about sidecars).
const BACKUP_DIR: &str = "backups";
/// How many backups are kept.
///
/// Small on purpose. A backup is a full copy of a catalog that is tens of
/// megabytes at 50k images, and the value of the third-oldest one is close to
/// zero: corruption is noticed at the next launch, not months later. What the
/// depth buys is protection against backing *up* the damage — if a corrupt
/// catalog is copied before anyone notices, the generation behind it is still
/// clean.
pub const KEEP_BACKUPS: usize = 3;
/// Suffix given to a catalog that has been set aside as damaged.
///
/// Kept rather than deleted. It costs disk this application would rather not
/// spend, and it is still the right call: `.sqlite` files have been recovered
/// by hand before, the user has not consented to a deletion, and NFR-R4's
/// instinct — never destroy what the user did not ask you to destroy — does
/// not stop applying at the catalog's edge.
const DAMAGED_SUFFIX: &str = "damaged";
/// One kept backup.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Backup {
pub path: PathBuf,
/// UTC seconds at which it was taken, read from the filename rather than
/// from the filesystem: a copy, a restore or a sync can rewrite an mtime,
/// and then the newest backup is not the one that looks newest.
pub taken_at: i64,
pub bytes: u64,
}
/// Where backups for `catalog` are kept.
pub fn backup_dir(catalog: &Path) -> PathBuf {
catalog
.parent()
.unwrap_or_else(|| Path::new("."))
.join(BACKUP_DIR)
}
/// Check the database this connection is attached to.
///
/// `quick_check` rather than `integrity_check`: the difference is that
/// `quick_check` skips verifying that every index agrees with its table, which
/// is the expensive half and the half this application least needs — every
/// index here is derivable, and `REINDEX` fixes one without anybody being
/// asked a question. What is left still reads every page, and catches the
/// damage that matters: torn b-trees, a bad freelist, a truncated file.
///
/// Returns [`CatalogError::Corrupt`] carrying what SQLite said, so the message
/// the user sees is the diagnosis rather than a paraphrase of it.
pub fn integrity_check(conn: &Connection) -> Result<(), CatalogError> {
// The argument caps how many problems are reported. One is enough: the
// answer is the same whether the file has one damaged page or nine
// hundred, and an unbounded check on a badly damaged file can run for a
// very long time producing a list nobody will read.
let mut stmt = conn.prepare("PRAGMA quick_check(1)")?;
let rows: Vec<String> = stmt
.query_map([], |r| r.get(0))?
.collect::<Result<Vec<_>, _>>()?;
// A healthy database answers with the single row "ok".
if rows.len() == 1 && rows[0] == "ok" {
return Ok(());
}
Err(CatalogError::Corrupt {
detail: rows.join("; "),
})
}
/// Check a catalog file that is not currently open.
///
/// Used before a restore: a backup is only worth swapping in if it is sound,
/// and swapping in a second damaged file — leaving the user with no catalog
/// and no offer left — is the failure this exists to prevent.
pub fn check_file(path: &Path) -> Result<(), CatalogError> {
if !path.is_file() {
return Err(CatalogError::Io(format!("{} is missing", path.display())));
}
// Read-write rather than read-only, which reads oddly for a check. A
// backup carries the WAL journal mode in its header because it was copied
// page-for-page from a WAL database, and SQLite cannot open one read-only
// without a shared-memory file it is then not allowed to create. Nothing
// here writes; the connection is opened, read, and dropped.
let conn = Connection::open(path)?;
integrity_check(&conn)
}
/// Take a backup of the open catalog.
///
/// Returns the file written. Older generations beyond [`KEEP_BACKUPS`] are
/// pruned, newest kept.
///
/// Goes through `crate::sync::copy_to` — SQLite's own backup API after a
/// TRUNCATE checkpoint — rather than copying the file. A WAL database is not
/// one file, and `fs::copy` of the main file alone would silently back up a
/// state that is older than the catalog and possibly torn, which is the one
/// failure mode a backup cannot afford.
pub fn backup(conn: &Connection, catalog: &Path) -> Result<PathBuf, CatalogError> {
let dir = backup_dir(catalog);
std::fs::create_dir_all(&dir)
.map_err(|e| CatalogError::Io(format!("creating {}: {e}", dir.display())))?;
let dest = dir.join(format!("catalog-{}.sqlite", now()));
// A second backup within the same second would otherwise land on the first
// one's name. Rare, and only reachable from tests and a retry, but the
// result would be a half-overwritten backup rather than two.
if dest.exists() {
std::fs::remove_file(&dest)
.map_err(|e| CatalogError::Io(format!("replacing {}: {e}", dest.display())))?;
}
// Dropped immediately: the copy is complete when `copy_to` returns, and
// holding the connection open would leave a `-wal` beside a file whose
// whole purpose is to be a single self-contained artefact.
drop(crate::sync::copy_to(conn, &dest)?);
prune(catalog);
Ok(dest)
}
/// Back up before a migration, if there is anything to back up.
///
/// Called from [`Catalog::open`](crate::Catalog::open) between `configure` and
/// `migrate`. NFR-R2 asks for this and the reasoning is narrower than "backups
/// are prudent": a migration is the one routine operation that rewrites table
/// structure, so it is the likeliest way a catalog becomes unreadable, and it
/// is the one moment where the pre-change state is still on disk to be copied.
/// Afterwards there is nothing left to take a copy *of*.
///
/// A no-op in the two cases where it would cost without buying anything: a
/// catalog already at [`schema::SCHEMA_VERSION`], and a brand-new file at
/// version 0 with no tables in it yet.
pub fn backup_before_migration(conn: &Connection, catalog: &Path) -> Result<(), CatalogError> {
let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
if from == 0 || from >= schema::SCHEMA_VERSION {
return Ok(());
}
let path = backup(conn, catalog)?;
log::info!(
"backed up catalog at v{from} to {} before migrating to v{}",
path.display(),
schema::SCHEMA_VERSION
);
Ok(())
}
/// How long a catalog may go without a backup before the next opportunity
/// takes one.
///
/// A day. The catalog is an index, so what a backup protects is the day's
/// worth of collection and people edits the sidecars do not hold — and a
/// second copy of a 130 MB file per launch would be a cost with nothing to
/// show for it when the user launches four times in an afternoon.
pub const BACKUP_EVERY: i64 = 24 * 60 * 60;
/// Whether [`BACKUP_EVERY`] has passed since the newest backup, or there is
/// none.
///
/// Read from the filenames, like [`backups`], so a restored or copied backup
/// directory answers the same way it did on the machine it came from.
pub fn backup_due(catalog: &Path) -> bool {
match backups(catalog).first() {
Some(newest) => now() - newest.taken_at >= BACKUP_EVERY,
None => true,
}
}
/// TRACES: NFR-R2
/// Take the scheduled backup, if one is due. Returns the file written, or
/// `None` when the newest is recent enough.
///
/// The scheduled half of NFR-R2 — the migration half is
/// [`backup_before_migration`]. "On a schedule" for an application that runs
/// when the user opens it means "at the next chance after a day has passed",
/// and the chance the caller picks is the end of a library sweep: the
/// catalog is quiet, the work is already off the UI thread, and it is the
/// moment a day's edits have just been consolidated.
///
/// A brand-new catalog with no images is not backed up: there is nothing in
/// it yet that a rescan would not rebuild, and the first backup would only be
/// a copy of an empty schema.
pub fn backup_if_due(conn: &Connection, catalog: &Path) -> Result<Option<PathBuf>, CatalogError> {
if !backup_due(catalog) {
return Ok(None);
}
let images: i64 = conn.query_row("SELECT count(*) FROM images", [], |r| r.get(0))?;
if images == 0 {
return Ok(None);
}
let path = backup(conn, catalog)?;
log::info!("scheduled backup of the catalog to {}", path.display());
Ok(Some(path))
}
/// The backups available for `catalog`, newest first.
///
/// Never fails: an unreadable or absent backup directory means there are no
/// backups, which is a fact about the offer to make rather than an error to
/// report on top of the corruption the user is already looking at.
pub fn backups(catalog: &Path) -> Vec<Backup> {
let dir = backup_dir(catalog);
let Ok(entries) = std::fs::read_dir(&dir) else {
return Vec::new();
};
let mut out: Vec<Backup> = entries
.flatten()
.filter_map(|e| {
let path = e.path();
let taken_at = timestamp_of(&path)?;
let bytes = e.metadata().ok()?.len();
Some(Backup {
path,
taken_at,
bytes,
})
})
.collect();
out.sort_by_key(|b| std::cmp::Reverse(b.taken_at));
out
}
/// Put a backup back in place of the damaged catalog.
///
/// **Every connection to `catalog` must be closed first.** This replaces the
/// file underneath anything still holding it open, which on a live connection
/// is how a *second* corrupt catalog gets made.
///
/// The order is deliberate:
///
/// 1. The backup is checked. A restore that installs a second damaged file
/// leaves the user with nothing to try next.
/// 2. The damaged catalog is renamed aside, and its `-wal` and `-shm` are
/// **deleted**. This is the step that is easy to leave out and fatal to
/// leave out: a journal belonging to the old file, sitting beside the new
/// one under the same name, is replayed into it on the next open. That is
/// not a restore, it is a fresh corruption with the evidence gone.
/// 3. The backup is *copied* into place, not moved, so a failure here can be
/// retried against the same backup.
pub fn restore(catalog: &Path, backup: &Path) -> Result<(), CatalogError> {
check_file(backup)?;
set_aside(catalog)?;
std::fs::copy(backup, catalog).map_err(|e| {
CatalogError::Io(format!(
"restoring {} from {}: {e}",
catalog.display(),
backup.display()
))
})?;
log::info!(
"restored {} from backup {}",
catalog.display(),
backup.display()
);
Ok(())
}
/// Move a damaged catalog out of the way so the next open builds a fresh one.
///
/// This is the rebuild path (NFR-R6's second offer) and also the first half of
/// a [`restore`]. Nothing else is needed to rebuild: the next
/// [`Catalog::open`](crate::Catalog::open) creates an empty catalog at the
/// current schema, and the ordinary scan repopulates it from sources and
/// sidecars — which is precisely invariant §5.2.4 being spent rather than
/// merely asserted.
///
/// Returns where the damaged file was put, or `None` if there was no catalog
/// to move — a caller may be recovering from a file SQLite could not open
/// because it was never created.
pub fn set_aside(catalog: &Path) -> Result<Option<PathBuf>, CatalogError> {
// Whatever takes this name next — a rebuild or a restored backup — is not
// the file this process last backfilled. Forgotten while the path still
// resolves, so it is the same key the open recorded.
crate::backfilled::forget(catalog);
let moved = if catalog.exists() {
let dest = with_suffix(catalog, DAMAGED_SUFFIX);
// An earlier damaged copy is replaced rather than accumulating: two of
// these is two full-size catalogs on the user's disk, and the older
// one has already been superseded by a recovery the user completed.
let _ = std::fs::remove_file(&dest);
// The rename first, so that a failure here leaves the journals with
// the file they belong to rather than orphaned beside a catalog that
// is still in use.
std::fs::rename(catalog, &dest).map_err(|e| {
CatalogError::Io(format!(
"setting aside {} as {}: {e}",
catalog.display(),
dest.display()
))
})?;
log::warn!(
"catalog {} was damaged; kept as {}",
catalog.display(),
dest.display()
);
Some(dest)
} else {
None
};
// Then the journals, whether or not there was a catalog to move: a `-wal`
// orphaned beside a missing database is replayed into whatever takes that
// name next, which would not be a restore but a fresh corruption with the
// evidence gone.
for sidecar in journals(catalog) {
if let Err(e) = std::fs::remove_file(&sidecar) {
if e.kind() != std::io::ErrorKind::NotFound {
return Err(CatalogError::Io(format!(
"removing stale journal {}: {e}",
sidecar.display()
)));
}
}
}
Ok(moved)
}
/// Delete backups beyond [`KEEP_BACKUPS`].
///
/// Best-effort and silent about individual failures: failing to delete an old
/// backup is not a reason to fail the new one, which is already written.
fn prune(catalog: &Path) {
for old in backups(catalog).into_iter().skip(KEEP_BACKUPS) {
if let Err(e) = std::fs::remove_file(&old.path) {
log::warn!("could not prune backup {}: {e}", old.path.display());
}
}
}
/// The WAL and shared-memory files SQLite keeps beside a database.
fn journals(catalog: &Path) -> [PathBuf; 2] {
[with_suffix(catalog, "wal"), with_suffix(catalog, "shm")]
}
/// `catalog.sqlite` plus `-suffix`, the way SQLite names its own sidecars.
///
/// Appended to the whole filename rather than replacing the extension, so
/// `catalog.sqlite-wal` is what SQLite would look for and `catalog.sqlite-
/// damaged` sorts next to the catalog it came from.
fn with_suffix(catalog: &Path, suffix: &str) -> PathBuf {
let mut s = catalog.as_os_str().to_os_string();
s.push("-");
s.push(suffix);
PathBuf::from(s)
}
/// Read the timestamp out of a backup's filename, or `None` if this is not one.
///
/// Doubles as the filter that keeps [`backups`] from offering the user
/// something that is not a catalog — a stray file in the directory, or a `-wal`
/// left by a crash mid-backup.
fn timestamp_of(path: &Path) -> Option<i64> {
let name = path.file_name()?.to_str()?;
name.strip_prefix("catalog-")?
.strip_suffix(".sqlite")?
.parse()
.ok()
}
/// Seconds since the epoch, or 0 if the clock is before it.
fn now() -> i64 {
SystemTime::now()
.duration_since(UNIX_EPOCH)
.map(|d| d.as_secs() as i64)
.unwrap_or(0)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::Catalog;
use std::io::{Seek, SeekFrom, Write};
/// A scratch directory that cleans up with the test.
fn tempdir(tag: &str) -> PathBuf {
let base = std::env::temp_dir().join(format!(
"dr-recovery-{tag}-{}-{:?}",
std::process::id(),
std::thread::current().id()
));
let _ = std::fs::remove_dir_all(&base);
std::fs::create_dir_all(&base).unwrap();
base
}
#[test]
fn a_scheduled_backup_is_taken_once_a_day_and_not_more() {
let dir = tempdir("scheduled");
let path = dir.join("catalog.sqlite");
fixture(&path, 3);
let cat = Catalog::open(&path).unwrap();
// Nothing yet: due.
assert!(backup_due(&path));
let first = backup_if_due(cat.connection(), &path).unwrap();
assert!(first.is_some(), "the first opportunity takes one");
// Taken just now: not due, and a second call does nothing.
assert!(!backup_due(&path));
assert_eq!(backup_if_due(cat.connection(), &path).unwrap(), None);
assert_eq!(backups(&path).len(), 1);
// Age the one backup past the interval by renaming it, since the
// timestamp is read from the name. Now it is due again.
let old = first.unwrap();
let aged = old
.parent()
.unwrap()
.join(format!("catalog-{}.sqlite", now() - BACKUP_EVERY - 1));
std::fs::rename(&old, &aged).unwrap();
assert!(backup_due(&path));
assert!(backup_if_due(cat.connection(), &path).unwrap().is_some());
assert_eq!(backups(&path).len(), 2);
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn an_empty_catalog_is_not_worth_backing_up() {
let dir = tempdir("empty");
let path = dir.join("catalog.sqlite");
let cat = Catalog::open(&path).unwrap();
assert!(backup_due(&path), "due in principle");
assert_eq!(backup_if_due(cat.connection(), &path).unwrap(), None);
assert!(backups(&path).is_empty());
let _ = std::fs::remove_dir_all(&dir);
}
/// A catalog on disk with enough rows to span several pages, closed.
///
/// Closed matters: WAL means the rows are in `catalog.sqlite-wal` until
/// something checkpoints, and a test that corrupted the main file while
/// the data was still in the journal would be corrupting empty space.
fn fixture(path: &Path, images: i64) {
let cat = Catalog::open(path).unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
VALUES ('u1', 'Iceland', 0, 0, 1, 1)",
[],
)
.unwrap();
for i in 1..=images {
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at)
VALUES (?1, 1, ?2, 0)",
rusqlite::params![i, format!("DCIM/IMG_{i:05}.CR3")],
)
.unwrap();
}
crate::sync::checkpoint(c).unwrap();
drop(cat);
}
/// Scribble over everything past the first two pages.
///
/// Past them rather than over them so that page 1 — the header and the
/// schema — survives: this produces a file SQLite is willing to open and
/// then finds damaged, which is the case `quick_check` exists for. Wiping
/// the header instead produces `SQLITE_NOTADB` at the first pragma, which
/// is a different branch and has its own test.
fn corrupt(path: &Path) {
let mut f = std::fs::OpenOptions::new().write(true).open(path).unwrap();
let len = f.metadata().unwrap().len();
assert!(
len > 8192,
"fixture is only {len} bytes; corrupting past page 2 would be a no-op"
);
let junk = vec![0x5a_u8; (len - 8192) as usize];
f.seek(SeekFrom::Start(8192)).unwrap();
f.write_all(&junk).unwrap();
f.sync_all().unwrap();
}
#[test]
fn a_healthy_catalog_passes() {
let cat = Catalog::in_memory().unwrap();
integrity_check(cat.connection()).unwrap();
}
#[test]
fn a_corrupt_catalog_is_reported_as_corrupt_not_as_sqlite() {
// The whole point of the variant: this used to arrive as whatever
// rusqlite error the first failing query produced, with nowhere to
// hang a recovery offer.
let dir = tempdir("detect");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
corrupt(&path);
assert!(matches!(
Catalog::open_verified(&path),
Err(CatalogError::Corrupt { .. })
));
}
#[test]
fn a_file_that_is_not_a_database_is_also_corrupt() {
// A truncated or overwritten catalog never reaches `quick_check`: the
// first pragma fails with SQLITE_NOTADB. Same accident to the user,
// same two offers, so it must classify the same way.
let dir = tempdir("notadb");
let path = dir.join("catalog.sqlite");
std::fs::write(&path, b"this is not a catalog, it is a text file\n").unwrap();
assert!(matches!(
Catalog::open(&path),
Err(CatalogError::Corrupt { .. })
));
}
#[test]
fn restoring_a_backup_recovers_the_collections_a_rebuild_would_lose() {
// The first NFR-R6 branch, asserted on the thing that distinguishes it
// from the second: a collection exists nowhere but the catalog, so it
// is the evidence that the *contents* came back and not merely a
// readable file (docs/dev/catalog.md §8.1).
let dir = tempdir("restore");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
{
let cat = Catalog::open(&path).unwrap();
backup(cat.connection(), &path).unwrap();
}
corrupt(&path);
assert!(matches!(
Catalog::open_verified(&path),
Err(CatalogError::Corrupt { .. })
));
let newest = backups(&path).into_iter().next().expect("a backup exists");
restore(&path, &newest.path).unwrap();
let cat = Catalog::open_verified(&path).unwrap();
let name: String = cat
.connection()
.query_row("SELECT name FROM collections", [], |r| r.get(0))
.unwrap();
assert_eq!(name, "Iceland");
let images: i64 = cat
.connection()
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(images, 500);
}
#[test]
fn a_damaged_backup_is_refused_rather_than_installed() {
let dir = tempdir("badbackup");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
{
let cat = Catalog::open(&path).unwrap();
backup(cat.connection(), &path).unwrap();
}
let newest = backups(&path).into_iter().next().unwrap();
corrupt(&newest.path);
corrupt(&path);
assert!(matches!(
restore(&path, &newest.path),
Err(CatalogError::Corrupt { .. })
));
// And the damaged catalog is still where it was, so the second offer
// is still available.
assert!(path.exists());
}
#[test]
fn setting_aside_leaves_a_fresh_catalog_to_rebuild_into() {
// The second NFR-R6 branch. What makes it a rebuild rather than a data
// loss is invariant §5.2.4, which lives outside this crate — what is
// testable here is that the damaged file is out of the way, kept, and
// that the next open succeeds on an empty catalog at the current
// schema, which is what a scan then fills.
let dir = tempdir("rebuild");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
corrupt(&path);
let kept = set_aside(&path).unwrap().expect("the catalog was there");
assert!(kept.exists(), "the damaged catalog was deleted, not kept");
assert!(!path.exists());
let cat = Catalog::open_verified(&path).unwrap();
let images: i64 = cat
.connection()
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(images, 0);
let v: i64 = cat
.connection()
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, schema::SCHEMA_VERSION);
}
#[test]
fn a_stale_journal_does_not_follow_the_catalog_into_recovery() {
// The step that is easy to omit: a `-wal` belonging to the damaged
// file is replayed into whatever takes its name next.
let dir = tempdir("journal");
let path = dir.join("catalog.sqlite");
fixture(&path, 500);
std::fs::write(with_suffix(&path, "wal"), b"stale").unwrap();
set_aside(&path).unwrap();
assert!(!with_suffix(&path, "wal").exists());
}
#[test]
fn a_migration_is_backed_up_before_it_runs() {
// NFR-R2's second clause, against a real v1 catalog rather than a
// faked version number: the point is not that *a* file appears but
// that it holds the state from before the migration, which is the only
// state that is any use if the migration is what breaks it.
let dir = tempdir("premigrate");
let path = dir.join("catalog.sqlite");
{
let c = Connection::open(&path).unwrap();
schema::configure(&c).unwrap();
// `v1_for_attached` names the schema it targets, and "main" is a
// schema like any other — so this is the real v1, without needing
// `V1` itself to become visible outside its module.
c.execute_batch(&schema::v1_for_attached("main")).unwrap();
c.pragma_update(None, "user_version", 1).unwrap();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
crate::sync::checkpoint(&c).unwrap();
}
assert!(backups(&path).is_empty());
Catalog::open(&path).unwrap();
let taken = backups(&path);
assert_eq!(taken.len(), 1, "no backup was taken before the migration");
check_file(&taken[0].path).unwrap();
let kept = Connection::open(&taken[0].path).unwrap();
let v: i64 = kept
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, 1, "the backup was taken after the migration, not before");
}
#[test]
fn opening_an_up_to_date_catalog_takes_no_backup() {
// Or every background task that opens the catalog would copy it.
let dir = tempdir("nobackup");
let path = dir.join("catalog.sqlite");
fixture(&path, 10);
Catalog::open(&path).unwrap();
assert!(backups(&path).is_empty());
}
#[test]
fn only_the_newest_generations_are_kept() {
let dir = tempdir("prune");
let path = dir.join("catalog.sqlite");
fixture(&path, 10);
let cat = Catalog::open(&path).unwrap();
// Written by hand rather than by calling `backup` in a loop: the
// filename carries whole seconds, so real calls would collide.
std::fs::create_dir_all(backup_dir(&path)).unwrap();
for t in 1..=KEEP_BACKUPS as i64 + 2 {
drop(
crate::sync::copy_to(
cat.connection(),
&backup_dir(&path).join(format!("catalog-{t}.sqlite")),
)
.unwrap(),
);
}
prune(&path);
let kept = backups(&path);
assert_eq!(kept.len(), KEEP_BACKUPS);
// Newest first, and the newest is the highest timestamp.
assert_eq!(kept[0].taken_at, KEEP_BACKUPS as i64 + 2);
}
#[test]
fn a_stray_file_in_the_backup_directory_is_not_offered_as_one() {
let dir = tempdir("stray");
let path = dir.join("catalog.sqlite");
fixture(&path, 10);
std::fs::create_dir_all(backup_dir(&path)).unwrap();
std::fs::write(backup_dir(&path).join("notes.txt"), b"hello").unwrap();
std::fs::write(backup_dir(&path).join("catalog-7.sqlite-wal"), b"x").unwrap();
assert!(backups(&path).is_empty());
}
}
-981
View File
@@ -1,981 +0,0 @@
//! TRACES: FR-PLAT-AND-4 | FR-PLAT-AND-3
//! The thing that drains the queue.
//!
//! [`crate::jobs`] has been a complete, durable, coalescing work queue since
//! the catalog was written, and nothing has ever taken a job out of it. Every
//! producer — the local walk, the remote scan — called `enqueue` and no one
//! called `claim_next`, so the table grew one row per photograph and stayed
//! that size forever. This module is the missing half.
//!
//! # Why the runner is driven rather than self-owning
//!
//! The obvious shape is a thread that loops until the queue is empty, and it
//! is the wrong one. On Android the process does not decide when background
//! work may run: `WorkManager` does, subject to Doze, battery saver and the
//! metered-network constraints in FR-NC-6, and it revokes permission mid-job
//! by calling `onStopped()` (FR-PLAT-AND-4). A foreground service for a
//! user-initiated export gets a longer leash but still not an unbounded one.
//!
//! So the runner owns no thread, no clock and no policy. It exposes
//! [`Runner::run_one`] — claim one job, run it, record what happened — and
//! [`Runner::drain`], which repeats that against a [`Budget`] and a
//! cancellation flag the host owns. A `Worker.doWork()` that must return
//! within ten minutes calls `drain` with a deadline; a desktop idle loop calls
//! it with none. Neither has to reach inside.
//!
//! Everything the host supplies is passed in for the same reason `jobs` takes
//! `now` rather than reading the clock: a scheduler is exactly the thing that
//! has to be testable without waiting.
//!
//! # Why interruption is not failure
//!
//! Four things can happen to a claimed job, and only two of them are the job's
//! fault:
//!
//! - [`Outcome::Done`] — the row is deleted.
//! - [`Outcome::Retry`] — the work failed and might succeed later. Backoff,
//! and eventually [`crate::jobs::MAX_ATTEMPTS`] gives up on it.
//! - [`Outcome::Abandon`] — the work cannot succeed, ever. Failing five times
//! over five minutes to learn that is five minutes of a phone's battery.
//! - [`Outcome::Interrupted`] — the *host* stopped, not the job. The claim is
//! released and the attempt it consumed is given back, because a user who
//! pulled the app off the screen has not told us anything about the file.
//!
//! Process death is the fifth case and the one that cannot report itself: the
//! row simply stays `Running` with no owner. [`Runner::recover`] is what
//! reclaims it, and it is why an interrupted job is resumable rather than lost
//! (FR-PLAT-AND-3). It must run **before** any worker starts against a
//! catalog, or it will steal a job another runner is holding — there is no
//! owner column to tell them apart.
use std::sync::atomic::{AtomicBool, Ordering};
use rusqlite::Connection;
use crate::error::CatalogError;
use crate::jobs::{self, Job, JobKind};
/// What running a job turned out to be.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Outcome {
/// The work is done. The row goes away.
Done,
/// It failed, and trying again later is reasonable. Backoff applies, and
/// [`crate::jobs::MAX_ATTEMPTS`] eventually stops it.
Retry(String),
/// It failed in a way no retry can fix — the subject is gone, the payload
/// is unreadable, the format is one this build does not know. Marked
/// failed at once rather than burning the whole retry ladder to reach the
/// same answer.
Abandon(String),
/// The host is stopping, and the job never really ran.
///
/// Distinct from `Retry` because it costs no attempt: `onStopped()` five
/// times in a row would otherwise mark a perfectly good job as failed.
Interrupted,
}
/// Something that can actually do the work a job describes.
///
/// The catalog knows what needs doing and nothing about how — face detection
/// needs a decoder, a fetch needs a network stack, and neither belongs under
/// `core/dr-catalog` (ARCH §4.1: calls go downward). So the queue lives here
/// and the handlers are supplied from above.
pub trait JobHandler {
/// The kinds this handler will accept.
///
/// Load-bearing, not documentation: the runner claims **only** kinds some
/// handler declares. A queue holding `FetchOriginal` rows on a device with
/// no connector must leave them alone rather than claim them and fail
/// them, and a runner that claimed everything would do exactly that — five
/// times each, with backoff, on battery.
fn kinds(&self) -> &[JobKind];
/// Do the work.
///
/// The connection is offered because most handlers write their result back
/// into the catalog; one that does not is free to ignore it. It is the
/// runner's own connection, so a handler must not hold a transaction open
/// across a network call — the runner needs it back to record the outcome.
fn run(&mut self, conn: &Connection, job: &Job) -> Outcome;
}
/// How much work a host is willing to let one drain do.
///
/// Both limits are checked *before* a job is claimed, never during one: a
/// handler is opaque and may be halfway through writing a sidecar. Overrunning
/// a deadline by one job is survivable; being killed mid-write is the thing
/// [`Runner::recover`] exists to clean up after.
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
pub struct Budget {
/// Stop after this many jobs. `None` means "until the queue is empty".
pub max_jobs: Option<usize>,
/// Stop once the clock reaches this second. Same clock the drain is given.
pub deadline: Option<i64>,
}
impl Budget {
/// Run until nothing is left. What a desktop idle pass wants.
pub const UNLIMITED: Self = Self {
max_jobs: None,
deadline: None,
};
/// At most `n` jobs. A slice small enough to stay responsive.
pub fn jobs(n: usize) -> Self {
Self {
max_jobs: Some(n),
..Self::UNLIMITED
}
}
/// Until the clock reaches `deadline`. What a `WorkManager` slot wants.
pub fn until(deadline: i64) -> Self {
Self {
deadline: Some(deadline),
..Self::UNLIMITED
}
}
}
/// Why a drain stopped.
///
/// Worth distinguishing because the host's next move differs: `Drained` means
/// there is nothing to reschedule for, and the other three all mean "there is
/// more, ask again".
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
pub enum Stopped {
/// Nothing claimable is left.
#[default]
Drained,
/// The job count ran out.
Budget,
/// The clock ran out.
Deadline,
/// The host asked it to stop, or a handler reported itself interrupted.
Cancelled,
}
impl Stopped {
/// Whether the queue may still hold claimable work.
pub fn more_to_do(self) -> bool {
self != Stopped::Drained
}
}
/// What one drain did.
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
pub struct DrainReport {
pub completed: usize,
pub retried: usize,
pub abandoned: usize,
pub interrupted: usize,
pub stopped: Stopped,
}
impl DrainReport {
/// Jobs claimed, whatever became of them. This is what a budget counts.
pub fn ran(&self) -> usize {
self.completed + self.retried + self.abandoned + self.interrupted
}
}
/// What a recovery pass found waiting from the last run.
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
pub struct Recovered {
/// Jobs a dead process was holding. These are the resumed ones.
pub reclaimed: usize,
/// Jobs deleted because the photograph they name no longer exists.
pub reaped: usize,
/// Jobs deleted because their kind is retired ([`JobKind::RETIRED`]).
pub retired: usize,
}
impl Recovered {
pub fn did_anything(&self) -> bool {
self.reclaimed > 0 || self.reaped > 0 || self.retired > 0
}
}
/// Ready the queue for a fresh run, before any worker touches it.
///
/// Three distinct cleanups, and all are startup-only:
///
/// - **Reclaim.** A `Running` row has no owner; the process that claimed it is
/// gone. On Android that is a routine morning, not a crash (FR-PLAT-AND-3).
/// The attempt it consumed is *kept*, deliberately: a job that takes the
/// process down with it three times running should not be retried forever,
/// and the attempt counter is the only evidence of that we have.
/// - **Reap.** Jobs naming an image the catalog no longer has. A library that
/// has been culled leaves jobs for photographs that were deleted
/// months ago, and every one of them would be claimed, run and failed.
/// - **Retire.** Rows of a kind nothing enqueues or claims any more
/// ([`jobs::drop_retired`]). Here rather than in a migration so that no
/// schema bump locks an older device out of the synced catalog, and every
/// time rather than once because an older build sharing the catalog will
/// queue them again.
///
/// Retiring runs first, so the other two never touch rows about to go.
/// Reclaim runs next so its count is the honest number of interrupted jobs,
/// before reaping removes whichever of them pointed at nothing.
///
/// **Call this exactly once per catalog, at startup.** It cannot distinguish a
/// job a dead process was holding from one a live runner is holding right now,
/// because there is no owner column — the queue is durable, not distributed.
pub fn recover(conn: &Connection) -> Result<Recovered, CatalogError> {
let retired = jobs::drop_retired(conn)?;
Ok(Recovered {
reclaimed: jobs::recover_orphaned(conn)?,
reaped: jobs::reap_orphan_subjects(conn)?,
retired,
})
}
/// Claims work, runs it, and records what happened.
///
/// Borrows its connection rather than owning one so a host can drive it from
/// the same handle it already has open. Nothing here spawns a thread; several
/// runners on several threads, each with its own connection to the same
/// catalog, are safe because the claim is a single atomic statement (see
/// [`crate::jobs::claim_next`]).
pub struct Runner<'a> {
conn: &'a Connection,
handlers: Vec<Box<dyn JobHandler + 'a>>,
/// The union of every handler's kinds, cached because it is passed to
/// every claim. This is what stops the runner claiming work it cannot do.
claimable: Vec<JobKind>,
}
impl<'a> Runner<'a> {
/// A runner with no handlers. It can recover, and it can claim nothing.
pub fn new(conn: &'a Connection) -> Self {
Self {
conn,
handlers: Vec::new(),
claimable: Vec::new(),
}
}
/// Add a handler.
pub fn with(self, handler: impl JobHandler + 'a) -> Self {
self.with_boxed(Box::new(handler))
}
/// Add a handler chosen at runtime — a network one only where there is a
/// connector, a decoding one only where there is a decoder.
pub fn with_boxed(mut self, handler: Box<dyn JobHandler + 'a>) -> Self {
for kind in handler.kinds() {
if !self.claimable.contains(kind) {
self.claimable.push(*kind);
}
}
self.handlers.push(handler);
self
}
/// The kinds this runner will claim. Useful to a host deciding whether
/// starting it is worth waking the radio for.
pub fn claimable(&self) -> &[JobKind] {
&self.claimable
}
/// See [`recover`]. Offered here too so a host has one thing to hold.
pub fn recover(&self) -> Result<Recovered, CatalogError> {
recover(self.conn)
}
/// Claim one job, run it, and record the outcome.
///
/// `Ok(None)` means nothing this runner can do is claimable *now* — the
/// queue may still hold work of other kinds, or work still backing off.
pub fn run_one(&mut self, now: i64) -> Result<Option<Ran>, CatalogError> {
// Copied out before `self.handlers` is borrowed mutably below. Both
// are fields of `self`, but the copy is what lets the two borrows
// coexist without the connection being reborrowed through `self`.
let conn = self.conn;
let Some(job) = jobs::claim_next_matching(conn, now, &self.claimable)? else {
return Ok(None);
};
let outcome = match self
.handlers
.iter_mut()
.find(|h| h.kinds().contains(&job.kind))
{
Some(handler) => handler.run(conn, &job),
// Unreachable by construction: `claimable` is exactly the union of
// the handlers' kinds. Parked rather than released, because
// releasing it would put it straight back where the next turn of
// the drain loop would claim it again, forever.
None => Outcome::Abandon(format!("no handler for {:?}", job.kind)),
};
match &outcome {
Outcome::Done => jobs::complete(conn, job.id)?,
Outcome::Retry(why) => jobs::fail(conn, &job, now, why)?,
Outcome::Abandon(why) => jobs::abandon(conn, job.id, why)?,
Outcome::Interrupted => jobs::release(conn, &job)?,
}
Ok(Some(Ran { job, outcome }))
}
/// Run jobs until the budget, the flag or the queue says stop.
///
/// `clock` is called once per iteration rather than sampled once, because
/// the two things it feeds both move during a long drain: the deadline
/// check, and the `now` a failing job's backoff is measured from.
///
/// Cancellation is checked between jobs only. A handler that wants to bail
/// out of work already started says so with [`Outcome::Interrupted`],
/// which also ends the drain — otherwise a handler that always interrupts
/// would release its job and be handed it straight back.
pub fn drain(
&mut self,
clock: &dyn Fn() -> i64,
budget: Budget,
cancel: &AtomicBool,
) -> Result<DrainReport, CatalogError> {
let mut report = DrainReport::default();
loop {
// Relaxed: the flag is a one-way latch set by another thread and
// the only thing ordered against it is our own next claim. Missing
// one turn of the loop costs a job, not correctness.
if cancel.load(Ordering::Relaxed) {
report.stopped = Stopped::Cancelled;
break;
}
if budget.max_jobs.is_some_and(|max| report.ran() >= max) {
report.stopped = Stopped::Budget;
break;
}
let now = clock();
if budget.deadline.is_some_and(|end| now >= end) {
report.stopped = Stopped::Deadline;
break;
}
let Some(ran) = self.run_one(now)? else {
report.stopped = Stopped::Drained;
break;
};
match ran.outcome {
Outcome::Done => report.completed += 1,
Outcome::Retry(_) => report.retried += 1,
Outcome::Abandon(_) => report.abandoned += 1,
Outcome::Interrupted => {
report.interrupted += 1;
report.stopped = Stopped::Cancelled;
break;
}
}
}
Ok(report)
}
/// Drain with no budget and no cancellation, at a fixed instant.
///
/// Terminates because a job that fails is pushed past `now` by its backoff
/// and stops being claimable at this instant.
pub fn drain_all(&mut self, now: i64) -> Result<DrainReport, CatalogError> {
static NEVER: AtomicBool = AtomicBool::new(false);
self.drain(&|| now, Budget::UNLIMITED, &NEVER)
}
}
/// One job and what became of it.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Ran {
pub job: Job,
pub outcome: Outcome,
}
#[cfg(test)]
mod tests {
use std::path::{Path, PathBuf};
use std::sync::{Arc, Mutex};
use super::*;
use crate::jobs::{enqueue, JobState, Priority, MAX_ATTEMPTS};
use crate::schema;
/// A handler built from a closure, so each test states its own behaviour.
struct Fake<F> {
kinds: Vec<JobKind>,
act: F,
}
impl<F: FnMut(&Job) -> Outcome> JobHandler for Fake<F> {
fn kinds(&self) -> &[JobKind] {
&self.kinds
}
fn run(&mut self, _conn: &Connection, job: &Job) -> Outcome {
(self.act)(job)
}
}
fn handler<F: FnMut(&Job) -> Outcome>(kinds: &[JobKind], act: F) -> Fake<F> {
Fake {
kinds: kinds.to_vec(),
act,
}
}
/// A handler that records which subjects it saw and always succeeds.
///
/// Takes its kinds by value and borrows nothing, so the returned handler is
/// `Send + 'static` and can be moved into a worker thread — which the
/// contention test needs.
fn recording(kinds: Vec<JobKind>, seen: Arc<Mutex<Vec<i64>>>) -> impl JobHandler + Send {
Fake {
kinds,
act: move |job: &Job| {
seen.lock().unwrap().push(job.subject_id.unwrap_or(-1));
Outcome::Done
},
}
}
fn db() -> Connection {
let c = Connection::open_in_memory().unwrap();
schema::configure(&c).unwrap();
schema::migrate(&c).unwrap();
c
}
/// An image row, so a job has a subject that exists.
fn image(c: &Connection, id: i64) {
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', '/lib')
ON CONFLICT DO NOTHING",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (?1, 1, ?2, 0)",
rusqlite::params![id, format!("/lib/{id}.CR3")],
)
.unwrap();
}
fn queued(c: &Connection, kind: JobKind, subject: i64) {
image(c, subject);
enqueue(c, kind, Some(subject), Priority::Background, None).unwrap();
}
fn rows(c: &Connection) -> i64 {
c.query_row("SELECT count(*) FROM jobs", [], |r| r.get(0))
.unwrap()
}
/// A catalog on disk, so more than one connection can open it.
fn temp_catalog(name: &str) -> PathBuf {
let dir = std::env::temp_dir().join(format!(
"dr-runner-{name}-{}-{:?}",
std::process::id(),
std::thread::current().id()
));
let _ = std::fs::remove_dir_all(&dir);
std::fs::create_dir_all(&dir).unwrap();
dir.join("catalog.db")
}
fn open(path: &Path) -> Connection {
let c = Connection::open(path).unwrap();
schema::configure(&c).unwrap();
schema::migrate(&c).unwrap();
// Several connections write to this file at once in the contention
// tests. Without a busy handler the loser of a race gets an error
// instead of a turn.
c.busy_timeout(std::time::Duration::from_secs(10)).unwrap();
c
}
#[test]
fn a_completed_job_leaves_the_queue() {
// The whole finding in one assertion: before this module, the row
// stayed forever because nothing ever claimed it.
let c = db();
queued(&c, JobKind::Thumbnail, 1);
let seen = Arc::new(Mutex::new(Vec::new()));
let report = Runner::new(&c)
.with(recording(vec![JobKind::Thumbnail], seen.clone()))
.drain_all(0)
.unwrap();
assert_eq!(report.completed, 1);
assert_eq!(report.stopped, Stopped::Drained);
assert_eq!(*seen.lock().unwrap(), vec![1]);
assert_eq!(rows(&c), 0, "a completed job leaves no row behind");
}
#[test]
fn a_failed_job_backs_off_and_is_claimed_again_later() {
let c = db();
queued(&c, JobKind::Thumbnail, 1);
// A `Cell` rather than a captured `bool`, so the closure's mutability
// is its own business and the test reads the same either way.
let failed_once = std::cell::Cell::new(false);
let mut runner = Runner::new(&c).with(handler(&[JobKind::Thumbnail], |_| {
if failed_once.replace(true) {
Outcome::Done
} else {
Outcome::Retry("decoder said no".into())
}
}));
let first = runner.drain_all(100).unwrap();
assert_eq!(first.retried, 1);
assert_eq!(rows(&c), 1, "a retryable failure keeps its row");
// Still inside the backoff window: nothing claimable, so the drain
// reports itself drained rather than spinning on the same job.
assert_eq!(runner.drain_all(100).unwrap().ran(), 0);
let later = runner.drain_all(100 + jobs::backoff_seconds(1)).unwrap();
assert_eq!(later.completed, 1);
assert_eq!(rows(&c), 0);
}
#[test]
fn a_job_that_keeps_failing_is_given_up_on() {
// FR-RAW-4: one corrupt file must not stall the queue behind endless
// retries. Driven through the runner rather than by hand, because the
// runner is what a corrupt file will actually meet.
let c = db();
queued(&c, JobKind::ExtractMetadata, 1);
let mut runner = Runner::new(&c).with(handler(&[JobKind::ExtractMetadata], |_| {
Outcome::Retry("corrupt file".into())
}));
let mut now = 0;
for _ in 0..MAX_ATTEMPTS {
assert_eq!(runner.drain_all(now).unwrap().retried, 1);
now += jobs::backoff_seconds(MAX_ATTEMPTS);
}
assert_eq!(runner.drain_all(now + 100_000).unwrap().ran(), 0);
let state: i64 = c
.query_row("SELECT state FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(state, JobState::Failed as i64);
}
#[test]
fn an_abandoned_job_is_not_retried_at_all() {
// The difference that matters on battery: five failures spread over
// five minutes to learn what the first one already said.
let c = db();
queued(&c, JobKind::FetchOriginal, 1);
let mut runner = Runner::new(&c).with(handler(&[JobKind::FetchOriginal], |_| {
Outcome::Abandon("no connector on this device".into())
}));
assert_eq!(runner.drain_all(0).unwrap().abandoned, 1);
// One attempt, not MAX_ATTEMPTS, and never claimable again.
assert_eq!(runner.drain_all(1_000_000).unwrap().ran(), 0);
let (state, attempts): (i64, i64) = c
.query_row("SELECT state, attempts FROM jobs", [], |r| {
Ok((r.get(0)?, r.get(1)?))
})
.unwrap();
assert_eq!(state, JobState::Failed as i64);
assert_eq!(attempts, 1);
}
#[test]
fn an_interrupted_job_costs_no_attempt_and_ends_the_drain() {
// `onStopped()` says nothing about the file. Charging it an attempt
// would let five backgroundings mark good work as failed.
let c = db();
queued(&c, JobKind::Thumbnail, 1);
queued(&c, JobKind::Thumbnail, 2);
let report = Runner::new(&c)
.with(handler(&[JobKind::Thumbnail], |_| Outcome::Interrupted))
.drain_all(0)
.unwrap();
assert_eq!(report.interrupted, 1);
assert_eq!(
report.stopped,
Stopped::Cancelled,
"an interrupted job must end the drain, or releasing it hands it \
straight back and the loop never ends"
);
let (state, attempts): (i64, i64) = c
.query_row(
"SELECT state, attempts FROM jobs WHERE subject_id = 1",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(state, JobState::Pending as i64);
assert_eq!(attempts, 0, "the claim's speculative attempt is given back");
}
#[test]
fn only_kinds_a_handler_covers_are_claimed() {
// A device with no connector must leave `FetchOriginal` where it is.
// Claiming it to fail it would cost five attempts and five backoffs
// per photograph, on battery, to reach a conclusion known in advance.
let c = db();
queued(&c, JobKind::Thumbnail, 1);
queued(&c, JobKind::FetchOriginal, 2);
let seen = Arc::new(Mutex::new(Vec::new()));
let report = Runner::new(&c)
.with(recording(vec![JobKind::Thumbnail], seen.clone()))
.drain_all(0)
.unwrap();
assert_eq!(report.completed, 1);
assert_eq!(*seen.lock().unwrap(), vec![1]);
let (state, attempts): (i64, i64) = c
.query_row(
"SELECT state, attempts FROM jobs WHERE subject_id = 2",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(state, JobState::Pending as i64);
assert_eq!(attempts, 0, "an unhandled job is untouched, not failed");
}
#[test]
fn a_runner_with_no_handlers_claims_nothing() {
let c = db();
queued(&c, JobKind::Thumbnail, 1);
assert_eq!(Runner::new(&c).drain_all(0).unwrap().ran(), 0);
assert_eq!(rows(&c), 1);
}
#[test]
fn a_drain_stops_at_its_job_budget() {
let c = db();
for id in 1..=5 {
queued(&c, JobKind::Thumbnail, id);
}
let seen = Arc::new(Mutex::new(Vec::new()));
let mut runner = Runner::new(&c).with(recording(vec![JobKind::Thumbnail], seen.clone()));
let report = runner
.drain(&|| 0, Budget::jobs(2), &AtomicBool::new(false))
.unwrap();
assert_eq!(report.completed, 2);
assert_eq!(report.stopped, Stopped::Budget);
assert!(report.stopped.more_to_do());
assert_eq!(rows(&c), 3, "the rest is still queued for the next slot");
assert_eq!(seen.lock().unwrap().len(), 2);
}
#[test]
fn a_drain_stops_at_its_deadline() {
// What a `WorkManager` slot does: a fixed window, and whatever did not
// fit stays queued for the next one.
let c = db();
for id in 1..=5 {
queued(&c, JobKind::Thumbnail, id);
}
// A clock that advances a second per reading, so the deadline arrives
// without the test sleeping.
let tick = std::cell::Cell::new(0i64);
let clock = || {
let t = tick.get();
tick.set(t + 1);
t
};
let report = Runner::new(&c)
.with(handler(&[JobKind::Thumbnail], |_| Outcome::Done))
.drain(&clock, Budget::until(3), &AtomicBool::new(false))
.unwrap();
assert_eq!(report.stopped, Stopped::Deadline);
assert_eq!(report.completed, 3, "one job per second up to the deadline");
assert_eq!(rows(&c), 2);
}
#[test]
fn a_cancelled_drain_stops_between_jobs() {
let c = db();
for id in 1..=5 {
queued(&c, JobKind::Thumbnail, id);
}
let cancel = AtomicBool::new(false);
// Cancelled from inside the handler, standing in for the host thread
// setting the flag while a job is in flight: the job in hand finishes,
// and nothing further is claimed.
let report = Runner::new(&c)
.with(handler(&[JobKind::Thumbnail], |_| {
cancel.store(true, Ordering::Relaxed);
Outcome::Done
}))
.drain(&|| 0, Budget::UNLIMITED, &cancel)
.unwrap();
assert_eq!(report.completed, 1);
assert_eq!(report.stopped, Stopped::Cancelled);
assert_eq!(rows(&c), 4, "the work is kept, not lost");
}
#[test]
fn a_job_interrupted_by_process_death_is_reclaimed_and_run_once() {
// FR-PLAT-AND-3. The kill happens between the claim and the outcome,
// which is the window a durable queue exists to survive: no `complete`,
// no `fail`, just a row marked `Running` with nobody holding it.
let c = db();
queued(&c, JobKind::ContentHash, 1);
// The dead process. It claimed the job and never came back.
let claimed = jobs::claim_next(&c, 0).unwrap().expect("claimable");
assert_eq!(claimed.subject_id, Some(1));
// A fresh runner, before it starts, finds the queue empty — the row is
// `Running` and no claim will touch it.
let seen = Arc::new(Mutex::new(Vec::new()));
let mut runner = Runner::new(&c).with(recording(vec![JobKind::ContentHash], seen.clone()));
assert_eq!(
runner.drain_all(0).unwrap().ran(),
0,
"an orphan is invisible until it is recovered — which is exactly \
why recovery has to happen at startup"
);
let recovered = runner.recover().unwrap();
assert_eq!(recovered.reclaimed, 1);
let report = runner.drain_all(0).unwrap();
assert_eq!(report.completed, 1);
assert_eq!(
*seen.lock().unwrap(),
vec![1],
"resumed, not repeated and not lost"
);
assert_eq!(rows(&c), 0);
}
#[test]
fn a_crash_still_costs_an_attempt() {
// Deliberate: a job that takes the process down with it every time is
// indistinguishable from one that fails, and the attempt counter is
// the only evidence we keep across a death. Without this a poison-pill
// job would be reclaimed and re-run forever.
let c = db();
queued(&c, JobKind::ContentHash, 1);
for _ in 0..MAX_ATTEMPTS {
jobs::claim_next(&c, 0).unwrap().expect("claimable");
recover(&c).unwrap();
}
let job = jobs::claim_next(&c, 0).unwrap().unwrap();
assert!(job.attempts > MAX_ATTEMPTS);
jobs::fail(&c, &job, 0, "died again").unwrap();
let state: i64 = c
.query_row("SELECT state FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(state, JobState::Failed as i64);
}
#[test]
fn recovery_drops_retired_kinds_every_time_and_nothing_else() {
// What 0.16.0 left behind: a thumbnail job per photograph that nothing
// would ever claim, beside live work that must survive.
let c = db();
queued(&c, JobKind::Thumbnail, 1);
queued(&c, JobKind::Thumbnail, 2);
enqueue(
&c,
JobKind::DetectFaces,
Some(1),
Priority::Background,
None,
)
.unwrap();
let first = recover(&c).unwrap();
assert_eq!(first.retired, 2);
assert!(first.did_anything());
let kinds: Vec<i64> = c
.prepare("SELECT kind FROM jobs")
.unwrap()
.query_map([], |r| r.get(0))
.unwrap()
.map(Result::unwrap)
.collect();
assert_eq!(kinds, vec![JobKind::DetectFaces as i64]);
// An older build opening the same catalog queues them again on its
// next scan. The next open by this one clears them again.
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
assert_eq!(recover(&c).unwrap().retired, 1);
assert_eq!(recover(&c).unwrap(), Recovered::default());
}
#[test]
fn recovery_drops_jobs_whose_photograph_is_gone() {
// A culled library leaves thumbnail jobs for images deleted months
// ago. Every one would be claimed, run and failed.
let c = db();
queued(&c, JobKind::ContentHash, 1);
queued(&c, JobKind::ContentHash, 2);
c.execute("DELETE FROM images WHERE id = 2", []).unwrap();
let recovered = recover(&c).unwrap();
assert_eq!(recovered.reaped, 1);
assert!(recovered.did_anything());
let left: i64 = c
.query_row("SELECT subject_id FROM jobs", [], |r| r.get(0))
.unwrap();
assert_eq!(left, 1, "only the job whose subject survives is kept");
}
#[test]
fn a_quiet_startup_recovers_nothing() {
let c = db();
queued(&c, JobKind::ContentHash, 1);
assert_eq!(recover(&c).unwrap(), Recovered::default());
assert!(!recover(&c).unwrap().did_anything());
}
#[test]
fn two_runners_on_one_catalog_never_take_the_same_job() {
// Sequential rather than threaded, so the property is asserted without
// depending on the scheduler: whatever the second connection claims,
// it is not what the first one is holding.
let path = temp_catalog("contention-pair");
let a = open(&path);
let b = open(&path);
for id in 1..=2 {
queued(&a, JobKind::Thumbnail, id);
}
let first = jobs::claim_next(&a, 0).unwrap().expect("one for A");
let second = jobs::claim_next(&b, 0).unwrap().expect("one for B");
assert_ne!(first.id, second.id);
assert!(
jobs::claim_next(&a, 0).unwrap().is_none(),
"a claimed job is invisible to every connection, not just its own"
);
}
#[test]
fn concurrent_runners_share_the_queue_without_repeating_work() {
// The claim is one atomic statement precisely so this holds: four
// threads, four connections, and every job run exactly once.
const THREADS: usize = 4;
const JOBS: i64 = 24;
let path = temp_catalog("contention-threads");
let seeder = open(&path);
for id in 1..=JOBS {
queued(&seeder, JobKind::Thumbnail, id);
}
drop(seeder);
let seen = Arc::new(Mutex::new(Vec::new()));
let mut threads = Vec::new();
for _ in 0..THREADS {
let path = path.clone();
let seen = seen.clone();
threads.push(std::thread::spawn(move || {
let conn = open(&path);
// Bound to a local rather than left as the block's tail: the
// `Runner` borrows `conn`, and a tail expression's temporaries
// are dropped *after* the block's locals, so the borrow would
// outlive what it borrows.
let completed = Runner::new(&conn)
.with(recording(vec![JobKind::Thumbnail], seen))
.drain_all(0)
.unwrap()
.completed;
completed
}));
}
let completed: usize = threads.into_iter().map(|t| t.join().unwrap()).sum();
assert_eq!(completed, JOBS as usize);
let mut ran = seen.lock().unwrap().clone();
ran.sort_unstable();
assert_eq!(
ran,
(1..=JOBS).collect::<Vec<_>>(),
"every job exactly once — no duplicate claim, nothing dropped"
);
let leftover = open(&path);
assert_eq!(rows(&leftover), 0);
}
#[test]
fn priority_survives_the_runner() {
// NFR-ARCH-2: visible work strictly preempts bulk work, and it has to
// still be true when the queue is drained through a handler rather
// than by hand.
let c = db();
image(&c, 1);
image(&c, 2);
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
enqueue(&c, JobKind::Thumbnail, Some(2), Priority::Interactive, None).unwrap();
let seen = Arc::new(Mutex::new(Vec::new()));
Runner::new(&c)
.with(recording(vec![JobKind::Thumbnail], seen.clone()))
.drain_all(0)
.unwrap();
assert_eq!(*seen.lock().unwrap(), vec![2, 1]);
}
#[test]
fn a_handler_sees_the_payload_and_the_attempt_count() {
// Both are how a handler decides what to do: the payload is the only
// thing that survives from the enqueue site, and the attempt count is
// how it can tell a first try from a last one.
let c = db();
image(&c, 1);
enqueue(
&c,
JobKind::ScanFolder,
Some(1),
Priority::Background,
Some("/lib/2024"),
)
.unwrap();
let payload = Arc::new(Mutex::new(None));
let recorded = payload.clone();
Runner::new(&c)
.with(handler(&[JobKind::ScanFolder], move |job| {
*recorded.lock().unwrap() = Some((job.payload.clone(), job.attempts));
Outcome::Done
}))
.drain_all(0)
.unwrap();
assert_eq!(
*payload.lock().unwrap(),
Some((Some("/lib/2024".to_string()), 1))
);
}
}
-237
View File
@@ -1,237 +0,0 @@
//! TRACES: FR-CAT-1 | FR-CAT-9 | NFR-P1
//! Incremental scan: the local analogue of ETag pruning.
//!
//! Nextcloud propagates ETags up the tree, so one request proves a whole
//! library unchanged (ARCH §8.4). A filesystem offers no such guarantee — a
//! directory's mtime moves when its *direct* entries change and not when a
//! grandchild does, so there is no cheap "did anything below here change"
//! probe.
//!
//! Local scan therefore prunes at each level rather than at the root: one
//! metadata probe per directory when nothing changed, instead of one per file.
//! A 50k-image library in ~2k folders costs 2k probes, which is the difference
//! between meeting and missing NFR-P1 on SAF.
//!
//! This module holds the decision logic and the deletion-sweep rules; walking
//! an actual directory belongs to the platform layer, which supplies
//! [`DirState`] and [`DirEntry`]. [`crate::walk`] is what puts the two
//! together.
pub use dr_types::{DirEntry, DirState};
use dr_types::FormatFilter;
/// What the scanner should do with a directory, before listing it.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum DirAction {
/// Contents unchanged. Skip the listing, but still recurse into known
/// children — without upward propagation, a deep change is invisible from
/// here.
RecurseOnly,
/// List and reconcile, then recurse.
ListAndRecurse,
}
/// Decide whether a directory needs listing.
pub fn classify_dir(stored: Option<DirState>, current: DirState) -> DirAction {
match stored {
Some(s) if s == current => DirAction::RecurseOnly,
_ => DirAction::ListAndRecurse,
}
}
/// What reconciling one listed entry against the catalog implies.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum EntryAction {
/// Not catalogued. Insert at `metadata_state = 1` and queue EXIF.
Insert,
/// Catalogued and unchanged. The common case, and it must cost nothing.
Unchanged,
/// Size or mtime moved: re-read metadata, rebuild the thumbnail, and drop
/// the content hash, which is no longer valid.
Changed,
/// Recognised but not a format the user asked to scan for.
Ignored,
}
/// What the catalog already holds for a source.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct KnownFile {
pub size: u64,
pub mtime: i64,
}
/// Classify one listed file.
pub fn classify_entry(
entry: &DirEntry,
known: Option<KnownFile>,
formats: &FormatFilter,
) -> EntryAction {
if !formats.allows_name(&entry.name) {
return EntryAction::Ignored;
}
match known {
None => EntryAction::Insert,
Some(k) if k.size == entry.size && k.mtime == entry.mtime => EntryAction::Unchanged,
Some(_) => EntryAction::Changed,
}
}
/// Outcome of a scan, which decides whether pruning may run.
///
/// `Cancelled` is the default because a scan that has not run has proven
/// nothing absent, and every default in this area must fail towards keeping
/// photographs.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum ScanOutcome {
/// Every reachable folder was visited.
Complete,
/// The user cancelled. Partial state is valid — jobs are resumable — but
/// unvisited folders must not be read as deleted.
#[default]
Cancelled,
/// The root itself could not be opened: drive unplugged, SAF grant
/// revoked, share unmounted.
RootUnreachable,
/// Some subtree failed while the root was fine.
PartialFailure,
}
impl ScanOutcome {
/// Whether the deletion sweep may run.
///
/// **The most dangerous decision in the catalog.** The sweep deletes every
/// folder not reached by this scan's generation. After an incomplete scan
/// that is most of the library, so it runs only on `Complete`.
///
/// FR-CAT-9 draws exactly this line: a source *proven absent* may leave
/// the catalog; a source merely *unreachable* is marked offline and kept,
/// with its ratings and edits intact.
pub fn may_prune(self) -> bool {
matches!(self, ScanOutcome::Complete)
}
}
#[cfg(test)]
mod tests {
use super::*;
use dr_types::Format;
const A: DirState = DirState {
mtime: 100,
entry_count: 5,
};
#[test]
fn unchanged_directory_is_not_listed() {
assert_eq!(classify_dir(Some(A), A), DirAction::RecurseOnly);
}
#[test]
fn a_never_seen_directory_is_listed() {
assert_eq!(classify_dir(None, A), DirAction::ListAndRecurse);
}
#[test]
fn changed_mtime_forces_a_listing() {
let now = DirState { mtime: 101, ..A };
assert_eq!(classify_dir(Some(A), now), DirAction::ListAndRecurse);
}
#[test]
fn entry_count_catches_what_mtime_misses() {
// A file added within the same timestamp tick: mtime is unchanged, so
// mtime alone would skip this directory and lose the new image.
let now = DirState {
mtime: 100,
entry_count: 6,
};
assert_eq!(classify_dir(Some(A), now), DirAction::ListAndRecurse);
}
#[test]
fn unchanged_file_costs_nothing() {
let e = DirEntry {
name: "IMG_0001.CR3".into(),
is_dir: false,
size: 30_000_000,
mtime: 500,
};
let known = KnownFile {
size: 30_000_000,
mtime: 500,
};
assert_eq!(
classify_entry(&e, Some(known), &FormatFilter::all()),
EntryAction::Unchanged
);
}
#[test]
fn a_resaved_file_is_reprocessed() {
let e = DirEntry {
name: "IMG_0001.CR3".into(),
is_dir: false,
size: 30_000_001,
mtime: 900,
};
let known = KnownFile {
size: 30_000_000,
mtime: 500,
};
assert_eq!(
classify_entry(&e, Some(known), &FormatFilter::all()),
EntryAction::Changed
);
}
#[test]
fn format_filter_excludes_unwanted_types() {
let jpeg = DirEntry {
name: "IMG_0001.JPG".into(),
is_dir: false,
size: 1,
mtime: 1,
};
assert_eq!(
classify_entry(&jpeg, None, &FormatFilter::raw_only()),
EntryAction::Ignored
);
assert_eq!(
classify_entry(&jpeg, None, &FormatFilter::all()),
EntryAction::Insert
);
}
#[test]
fn a_placeholder_is_catalogued_as_the_image_it_stands_for() {
// 121,785 of these in a real synced folder (ARCH §9.0). Each must
// enter the catalog as a CR2 marked offline, not be skipped as an
// unknown ".nextcloud" type.
let stub = DirEntry {
name: "_MG_4130.CR2.nextcloud".into(),
is_dir: false,
size: 1,
mtime: 1,
};
assert_eq!(
classify_entry(&stub, None, &FormatFilter::from_formats([Format::Cr2])),
EntryAction::Insert
);
}
#[test]
fn pruning_requires_a_complete_scan() {
assert!(ScanOutcome::Complete.may_prune());
}
#[test]
fn an_unreachable_root_never_prunes() {
// The guard that stops an unplugged drive from deleting the library:
// every folder would look unreached, so the sweep would take all of
// them (FR-CAT-9).
assert!(!ScanOutcome::RootUnreachable.may_prune());
assert!(!ScanOutcome::Cancelled.may_prune());
assert!(!ScanOutcome::PartialFailure.may_prune());
}
}
File diff suppressed because it is too large Load Diff
-831
View File
@@ -1,831 +0,0 @@
//! TRACES: FR-CAT-7 | FR-NC-9 | NFR-R1
//! Preparing the catalog file for upload, and taking in a remote one.
//!
//! # The hazard this module exists to handle
//!
//! A WAL-mode SQLite database is not one file. Committed transactions can live
//! in `catalog.sqlite-wal` with the main file lagging behind, so copying
//! `catalog.sqlite` alone uploads a **torn snapshot**: internally consistent as
//! of some older point, missing everything since. Worse, a naive copy taken
//! while a writer is mid-transaction can be structurally corrupt.
//!
//! So an upload never copies the live file. It runs a TRUNCATE checkpoint to
//! fold the WAL back into the main file, then uses SQLite's own backup API to
//! take a consistent snapshot — which serialises correctly against concurrent
//! writers rather than racing them.
//!
//! # What is actually synced
//!
//! Only the *user's judgements about their library* merge: collections, and the
//! keyword vocabulary with its assignments (see [`crate::merge`]). The rest of
//! the catalog is a *local index* of *local* storage — folder mtimes, cache
//! paths, job rows — and copying another device's version of those in would be
//! actively wrong. The remote file is read for those two and then discarded.
//!
//! This is why the catalog remains disposable in the ARCH §6.12 sense: nothing
//! here makes the local database authoritative for anything a rebuild could
//! not recover.
use std::path::{Path, PathBuf};
use rusqlite::Connection;
use crate::error::CatalogError;
use crate::merge::{self, MergeReport};
/// Schema name the downloaded remote catalog is attached under.
const REMOTE_SCHEMA: &str = "remote_cat";
/// Fold the WAL into the main database file.
///
/// TRUNCATE rather than PASSIVE: passive checkpointing gives up when a reader
/// holds the WAL open, which would leave recent commits out of the snapshot
/// without saying so.
pub fn checkpoint(conn: &Connection) -> Result<(), CatalogError> {
conn.pragma_update(None, "wal_checkpoint", "TRUNCATE")?;
Ok(())
}
/// Write a consistent snapshot of the catalog to `dest`, ready to upload,
/// without the face crops.
///
/// Built rather than copied. The snapshot is the *whole catalog* bar the
/// crops, uploaded on every sync and downloaded by every device. The crops are
/// most of the file (96 MB of a 158 MB reference catalog), and copying them in
/// only to delete them was most of the cost. A backup-API copy followed by
/// `UPDATE faces SET crop = NULL` and `VACUUM` wrote the file roughly three
/// times over to produce 50 MB (#71). So this creates the schema in an empty
/// file and copies every table into it with `crop` left NULL. That is one
/// pass, with nothing written that is not uploaded.
///
/// Consistency comes from doing the whole copy inside one transaction on the
/// snapshot's connection, which holds a single read snapshot of the source for
/// its duration. A writer committing meanwhile lands in the source's WAL and is
/// simply not seen, the same serialisation the backup API gave.
///
/// Crops are not lost by this: they travel in the face shards
/// ([`crate::face_shard::export_to_shards`]), which are written once and
/// downloaded once. Nothing reads a crop out of a merged remote catalog. The
/// merge reads a remote face's box and model to match it to a local one, and
/// no more. So leaving them out costs a receiving device nothing it would
/// otherwise have had. A device never adopts a downloaded catalog as its own,
/// so a fresh one gets its crops from the shards too.
///
/// The result must stay what every earlier build already merges: same schema,
/// same `user_version`, same page size and the same WAL flag in the header.
/// The `the_snapshot_*` tests pin those.
pub fn snapshot_for_upload(conn: &Connection, dest: &Path) -> Result<(), CatalogError> {
// The source is read through a second connection, attached to the
// snapshot's, so it needs to be a file. Every catalog is one.
let source = conn
.path()
.filter(|p| !p.is_empty())
.map(PathBuf::from)
.ok_or_else(|| CatalogError::Io("the catalog to snapshot has no file".into()))?;
// Not needed for consistency, since the read transaction below sees the
// WAL, but it keeps the live WAL from growing across syncs, as before.
checkpoint(conn)?;
// A leftover from a pass that died mid-build would otherwise be built on.
for stale in [
dest.to_path_buf(),
sidecar_of(dest, "-wal"),
sidecar_of(dest, "-journal"),
sidecar_of(dest, "-shm"),
] {
match std::fs::remove_file(&stale) {
Ok(()) => {}
Err(e) if e.kind() == std::io::ErrorKind::NotFound => {}
Err(e) => return Err(CatalogError::Io(format!("{}: {e}", stale.display()))),
}
}
let out = Connection::open(dest)?;
build_snapshot(conn, &source, &out)?;
// This device's local album folders are paths and SAF grants nobody
// else can use; the merge never reads them, and the snapshot is what
// a fresh device would otherwise adopt whole.
out.execute_batch("DROP TABLE IF EXISTS album_folders")?;
verify_snapshot(&out)?;
Ok(())
}
/// `catalog.sqlite` + `-wal` → `catalog.sqlite-wal`.
fn sidecar_of(path: &Path, suffix: &str) -> PathBuf {
let mut s = path.as_os_str().to_owned();
s.push(suffix);
PathBuf::from(s)
}
/// Schema name the source catalog is attached under while a snapshot is built.
const SOURCE_SCHEMA: &str = "snap_src";
/// The body of [`snapshot_for_upload`]: fill the empty database `out` from
/// the catalog at `source`.
fn build_snapshot(conn: &Connection, source: &Path, out: &Connection) -> Result<(), CatalogError> {
// Settings that only take on an empty file, copied from the source so the
// result is the file a backup would have been.
let page_size: i64 = conn.query_row("PRAGMA main.page_size", [], |r| r.get(0))?;
let auto_vacuum: i64 = conn.query_row("PRAGMA main.auto_vacuum", [], |r| r.get(0))?;
out.pragma_update(None, "page_size", page_size)?;
out.pragma_update(None, "auto_vacuum", auto_vacuum)?;
// A scratch file, rebuilt whole on every pass and checked before upload:
// durability during the build buys nothing. MEMORY rather than OFF keeps
// ROLLBACK defined on the failure path.
out.pragma_update(None, "journal_mode", "MEMORY")?;
out.pragma_update(None, "synchronous", "OFF")?;
// The rows were checked when they were written, and the copy has them
// all by the end. The bundled SQLite turns foreign keys on by default,
// and with them a multi-row INSERT into `images` scans `images` for
// children of every row it adds (`shadowed_by` refers to the same table
// and has no index): 1.2 s of a 1.7 s snapshot on 24k images.
out.pragma_update(None, "foreign_keys", false)?;
// Bound as a parameter, so a path containing a quote cannot break out.
out.execute(
&format!("ATTACH DATABASE ?1 AS {SOURCE_SCHEMA}"),
[source.to_string_lossy().as_ref()],
)?;
let result = copy_schema_and_rows(out);
if let Err(e) = out.execute(&format!("DETACH DATABASE {SOURCE_SCHEMA}"), []) {
log::warn!("failed to detach the catalog from its snapshot: {e}");
}
result?;
// Last, and outside any transaction, which is the only place it can be
// set: the header says WAL, as every snapshot uploaded so far has.
out.pragma_update(None, "journal_mode", "WAL")?;
Ok(())
}
fn copy_schema_and_rows(out: &Connection) -> Result<(), CatalogError> {
let tx = out.unchecked_transaction()?;
// The first read of the source opens its read snapshot. Everything from
// here, schema included, is as of that one moment.
let objects: Vec<(String, String, String)> = {
let mut stmt = tx.prepare(&format!(
"SELECT type, name, sql FROM {SOURCE_SCHEMA}.sqlite_master
WHERE sql IS NOT NULL AND name NOT LIKE 'sqlite\\_%' ESCAPE '\\'
ORDER BY rowid"
))?;
let rows = stmt.query_map([], |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)))?;
rows.collect::<Result<_, _>>()?
};
let user_version: i64 =
tx.query_row(&format!("PRAGMA {SOURCE_SCHEMA}.user_version"), [], |r| {
r.get(0)
})?;
let application_id: i64 =
tx.query_row(&format!("PRAGMA {SOURCE_SCHEMA}.application_id"), [], |r| {
r.get(0)
})?;
// Tables and their rows first, then indexes, triggers and views, so that
// an index is built once over the data rather than maintained per row,
// and no trigger fires on the copy. Foreign keys are off on this
// connection, so the order tables are filled in does not matter.
for (_, name, sql) in objects.iter().filter(|(k, _, _)| k == "table") {
// Verbatim: an unqualified CREATE lands in `main`, the snapshot.
tx.execute_batch(sql)?;
let columns: Vec<String> = {
let mut stmt = tx.prepare("SELECT name FROM pragma_table_info(?1, 'main')")?;
let rows = stmt.query_map([name], |r| r.get::<_, String>(0))?;
rows.collect::<Result<_, _>>()?
};
let select = columns
.iter()
.map(|c| {
if name == "faces" && c == "crop" {
"NULL".to_string()
} else {
quote_ident(c)
}
})
.collect::<Vec<_>>()
.join(", ");
let insert = columns
.iter()
.map(|c| quote_ident(c))
.collect::<Vec<_>>()
.join(", ");
let table = quote_ident(name);
tx.execute(
&format!(
"INSERT INTO main.{table} ({insert})
SELECT {select} FROM {SOURCE_SCHEMA}.{table}"
),
[],
)?;
}
// AUTOINCREMENT's counters live in a table the filter above skips; the
// CREATE of such a table makes an empty one here.
let has_sequence: bool = tx.query_row(
&format!(
"SELECT EXISTS(SELECT 1 FROM {SOURCE_SCHEMA}.sqlite_master
WHERE name = 'sqlite_sequence')"
),
[],
|r| r.get(0),
)?;
if has_sequence {
tx.execute_batch(&format!(
"DELETE FROM main.sqlite_sequence;
INSERT INTO main.sqlite_sequence SELECT * FROM {SOURCE_SCHEMA}.sqlite_sequence;"
))?;
}
for (_, _, sql) in objects.iter().filter(|(k, _, _)| k != "table") {
tx.execute_batch(sql)?;
}
tx.pragma_update(None, "user_version", user_version)?;
tx.pragma_update(None, "application_id", application_id)?;
tx.commit()?;
Ok(())
}
/// `name` as an SQL identifier, whatever it contains.
fn quote_ident(name: &str) -> String {
format!("\"{}\"", name.replace('"', "\"\""))
}
/// TRACES: NFR-R2
/// Refuse to hand over a snapshot that will not pass `quick_check`.
///
/// The upload is the copy every other device merges from, and a damaged one
/// costs far more than the check: each device downloads it, fails, and — for
/// a week, once — declines to push over it. `quick_check` reads every page
/// but skips index verification, which is the affordable version of "is this
/// a database" on a 40 MB file that has just been written and is still in the
/// page cache. A failure here is [`CatalogError::Corrupt`], the same thing a
/// receiving device would have said, so the sync reports it the same way.
fn verify_snapshot(snapshot: &Connection) -> Result<(), CatalogError> {
let verdict: String = snapshot.query_row("PRAGMA quick_check", [], |r| r.get(0))?;
if verdict == "ok" {
Ok(())
} else {
Err(CatalogError::Corrupt {
detail: format!("the snapshot for upload failed quick_check: {verdict}"),
})
}
}
/// Checkpoint, then copy the whole database to `dest`, and hand back the
/// connection to the copy.
///
/// What [`crate::recovery`] takes its NFR-R2 backups with. A backup is the
/// file the user may have to *live on*, so it keeps the face crops that
/// [`snapshot_for_upload`] leaves out, and a byte-for-byte page copy is the
/// right tool. Keeping it here beside the upload keeps the WAL discipline in
/// one place: a backup taken with `fs::copy` would be the torn snapshot this
/// module's header exists to warn about.
pub(crate) fn copy_to(conn: &Connection, dest: &Path) -> Result<Connection, CatalogError> {
checkpoint(conn)?;
let mut out = Connection::open(dest)?;
let backup = rusqlite::backup::Backup::new(conn, &mut out)?;
// SQLite's own "copy everything" sentinel is -1, but rusqlite asserts a
// positive page count, so ask for more pages than a catalog will ever
// have. The effect is the same: one step, no interleaved writers, no
// progress callback. A 50k-image catalog is tens of megabytes.
backup.run_to_completion(i32::MAX, std::time::Duration::ZERO, None)?;
drop(backup);
Ok(out)
}
/// Whether a downloaded remote catalog is worth merging.
///
/// Cheap guard before attaching: a remote written by a newer build may contain
/// tables and columns this one cannot read, and attempting the merge would
/// fail mid-transaction rather than declining cleanly.
pub fn remote_is_mergeable(remote: &Path) -> Result<bool, CatalogError> {
let conn = Connection::open_with_flags(
remote,
rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY | rusqlite::OpenFlags::SQLITE_OPEN_NO_MUTEX,
)?;
let v: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
Ok(v <= crate::schema::SCHEMA_VERSION)
}
/// Attach a downloaded remote catalog, merge its collections, detach.
///
/// The remote file is opened **read-only** — this device never writes to
/// another device's catalog, it only reads collections out of it.
pub fn merge_remote(conn: &Connection, remote: &Path) -> Result<MergeReport, CatalogError> {
if !remote_is_mergeable(remote)? {
return Err(CatalogError::SchemaTooNew {
found: -1,
supported: crate::schema::SCHEMA_VERSION,
});
}
// Path binds as a parameter; ATTACH accepts one, so a path containing a
// quote cannot break out into SQL.
conn.execute(
&format!("ATTACH DATABASE ?1 AS {REMOTE_SCHEMA}"),
[remote.to_string_lossy().as_ref()],
)?;
let result = merge::merge_all(conn);
// The merge brings in rows the backfill exists for — assignments whose
// word this device has no term for, from a remote older than v6 — so the
// next open must run it, whether or not the stamp happened to move.
if let Some(path) = conn.path().filter(|p| !p.is_empty()) {
crate::backfilled::forget(Path::new(path));
}
// Detach even if the merge failed, or the next attempt errors with
// "database remote_cat is already in use".
let detach = conn.execute(&format!("DETACH DATABASE {REMOTE_SCHEMA}"), []);
if let Err(e) = detach {
log::warn!("failed to detach remote catalog: {e}");
}
// After every merge, because a merge is where two devices' people meet:
// the same name typed on each, or a redirect one of them made. Its own
// transaction, and a failure is logged rather than returned -- what the
// merge took is committed and valid whether or not the duplicates were
// folded, and the next pass tries again. Runs on the sync worker, never
// the UI thread, and costs ~10 ms when there is nothing to do.
if result.is_ok() {
if let Err(e) = crate::dedup_people::run(conn) {
log::warn!("dedup after the catalog merge: {e}");
}
}
result
}
/// Where the catalog snapshot and the downloaded remote live.
///
/// Both are transient working files, not the catalog itself, so they belong in
/// the cache directory rather than beside the live database.
#[derive(Debug, Clone)]
pub struct SyncPaths {
pub upload_snapshot: PathBuf,
pub downloaded_remote: PathBuf,
}
impl SyncPaths {
pub fn in_dir(cache_dir: &Path) -> Self {
SyncPaths {
upload_snapshot: cache_dir.join("catalog-upload.sqlite"),
downloaded_remote: cache_dir.join("catalog-remote.sqlite"),
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::schema;
fn seeded(path: &Path) -> Connection {
let c = Connection::open(path).unwrap();
schema::configure(&c).unwrap();
schema::migrate(&c).unwrap();
c
}
#[test]
fn snapshot_captures_committed_data() {
let dir = tempdir();
let live = dir.join("catalog.sqlite");
let snap = dir.join("snap.sqlite");
let c = seeded(&live);
c.execute(
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
VALUES ('u1', 'Iceland', 0, 0, 1, 1)",
[],
)
.unwrap();
snapshot_for_upload(&c, &snap).unwrap();
// The snapshot must hold the row even though it was written after the
// database was created — the torn-file failure this guards against.
let s = Connection::open(&snap).unwrap();
let name: String = s
.query_row("SELECT name FROM collections", [], |r| r.get(0))
.unwrap();
assert_eq!(name, "Iceland");
}
#[test]
fn a_remote_from_a_newer_build_is_declined_not_attempted() {
let dir = tempdir();
let remote = dir.join("remote.sqlite");
let r = seeded(&remote);
r.pragma_update(None, "user_version", schema::SCHEMA_VERSION + 1)
.unwrap();
drop(r);
assert!(!remote_is_mergeable(&remote).unwrap());
let local = seeded(&dir.join("local.sqlite"));
assert!(matches!(
merge_remote(&local, &remote),
Err(CatalogError::SchemaTooNew { .. })
));
}
#[test]
fn merge_remote_round_trips_a_collection() {
let dir = tempdir();
let remote_path = dir.join("remote.sqlite");
{
let r = seeded(&remote_path);
r.execute(
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
VALUES ('u-remote', 'Portugal', 0, 0, 1, 1)",
[],
)
.unwrap();
checkpoint(&r).unwrap();
}
let local = seeded(&dir.join("local.sqlite"));
local
.execute(
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
VALUES ('u-local', 'Iceland', 0, 0, 1, 1)",
[],
)
.unwrap();
let report = merge_remote(&local, &remote_path).unwrap();
assert_eq!(report.inserted, 1);
let n: i64 = local
.query_row("SELECT count(*) FROM collections", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 2);
}
/// One image, known to the server by `file_id`, in a catalog.
fn with_image(c: &Connection, file_id: i64) -> dr_types::ImageId {
c.execute(
"INSERT OR IGNORE INTO roots(id, kind, label) VALUES (1, 'remote', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(root_id, source_ref, added_at) VALUES (1, ?1, 0)",
[format!("IMG_{file_id}.CR3")],
)
.unwrap();
let id = c.last_insert_rowid();
c.execute(
"INSERT INTO remote(image_id, file_id) VALUES (?1, ?2)",
[id, file_id],
)
.unwrap();
dr_types::ImageId(id as u64)
}
#[test]
fn an_album_and_its_exports_reach_another_device_but_its_folder_does_not() {
use crate::albums::{self, Place};
let dir = tempdir();
let remote_path = dir.join("remote.sqlite");
let snap = dir.join("snap.sqlite");
{
// The desktop: two albums, one on the server and one on its own
// disk, each with an export of the same photograph.
let r = seeded(&dir.join("desktop.sqlite"));
let img = with_image(&r, 4242);
let web = albums::create(&r, "Web", &Place::Server("Shared/Web".into())).unwrap();
let print = albums::create(&r, "Print", &Place::Local("/mnt/print".into())).unwrap();
albums::record_exports(&r, web, &[(img, "IMG_4242.jpg".into())]).unwrap();
albums::record_exports(&r, print, &[(img, "IMG_4242.tif".into())]).unwrap();
snapshot_for_upload(&r, &snap).unwrap();
std::fs::rename(&snap, &remote_path).unwrap();
}
// The tablet knows the same file under its own image id.
let local = seeded(&dir.join("tablet.sqlite"));
with_image(&local, 1);
let img = with_image(&local, 4242);
let report = merge_remote(&local, &remote_path).unwrap();
assert_eq!(report.albums_taken, 2);
assert_eq!(report.album_exports_added, 2);
assert!(report.local_changed());
let all = albums::list(&local).unwrap();
let print = all.iter().find(|a| a.name == "Print").unwrap();
let web = all.iter().find(|a| a.name == "Web").unwrap();
assert_eq!(print.place, None, "the desktop's disk is not the tablet's");
assert_eq!(web.place, Some(Place::Server("Shared/Web".into())));
assert_eq!(albums::sources(&local, web.id).unwrap(), vec![img]);
// Nothing changed on either side, so a second pass takes nothing.
let again = merge_remote(&local, &remote_path).unwrap();
assert_eq!(again.albums_taken, 0);
assert_eq!(again.album_exports_added, 0);
}
#[test]
fn a_device_that_never_made_an_album_still_uploads_this_ones() {
use crate::albums::{self, Place};
let dir = tempdir();
let remote_path = dir.join("remote.sqlite");
{
// A snapshot from a build that predates albums altogether.
let r = seeded(&remote_path);
checkpoint(&r).unwrap();
}
let local = seeded(&dir.join("local.sqlite"));
albums::create(&local, "Web", &Place::Server("Web".into())).unwrap();
let report = merge_remote(&local, &remote_path).unwrap();
assert_eq!(report.albums_taken, 0);
// Its albums table is absent, so nothing was compared — and the local
// album has still to reach the server.
assert!(
albums::list(&local).unwrap().len() == 1,
"the local album survives a merge with a catalog that has none"
);
}
#[test]
fn the_remote_can_be_merged_twice_without_attach_conflict() {
// Detach must happen even on the failure path, or the second attempt
// errors with "database remote_cat is already in use".
let dir = tempdir();
let remote_path = dir.join("remote.sqlite");
{
let r = seeded(&remote_path);
r.execute(
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
VALUES ('u-remote', 'Portugal', 0, 0, 1, 1)",
[],
)
.unwrap();
checkpoint(&r).unwrap();
}
let local = seeded(&dir.join("local.sqlite"));
merge_remote(&local, &remote_path).unwrap();
let second = merge_remote(&local, &remote_path).unwrap();
assert!(!second.local_changed());
}
/// A scratch directory that cleans up with the test.
fn tempdir() -> PathBuf {
let base = std::env::temp_dir().join(format!(
"dr-catalog-test-{}-{:?}",
std::process::id(),
std::thread::current().id()
));
let _ = std::fs::remove_dir_all(&base);
std::fs::create_dir_all(&base).unwrap();
base
}
/// The whole reason crops live in the shards: a snapshot is uploaded whole,
/// on every sync, to every device.
#[test]
fn the_snapshot_carries_no_face_crops() {
let dir = tempdir();
let live = dir.join("catalog.sqlite");
let snap = dir.join("snap.sqlite");
let c = seeded(&live);
c.execute(
"INSERT OR IGNORE INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)",
[],
)
.unwrap();
c.execute(
"INSERT INTO faces
(image_id, x, y, w, h, landmarks, detector_confidence, embedding,
crop_px, model_id, detected_at, crop)
VALUES (1, 0.1, 0.1, 0.2, 0.2, X'00', 0.9, X'00', 180.0, 'm', 0, ?1)",
[vec![7u8; 4096]],
)
.unwrap();
snapshot_for_upload(&c, &snap).unwrap();
let out = Connection::open(&snap).unwrap();
let crops: i64 = out
.query_row(
"SELECT COUNT(*) FROM faces WHERE crop IS NOT NULL",
[],
|r| r.get(0),
)
.unwrap();
assert_eq!(crops, 0, "the snapshot still carries face crops");
// The face itself must still be there — only the pixels are dropped.
let faces: i64 = out
.query_row("SELECT COUNT(*) FROM faces", [], |r| r.get(0))
.unwrap();
assert_eq!(faces, 1);
// And the local catalog keeps its crop: this strips the copy, never
// the original.
let kept: i64 = c
.query_row(
"SELECT COUNT(*) FROM faces WHERE crop IS NOT NULL",
[],
|r| r.get(0),
)
.unwrap();
assert_eq!(kept, 1, "stripping the snapshot damaged the live catalog");
}
/// Device-side setup for the snapshot tests: an image both devices know by
/// its cross-device file id, and one face on it carrying `crop`.
fn with_a_face(c: &Connection, face_id: i64, x: f64, crop: &[u8]) {
c.execute(
"INSERT OR IGNORE INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
[],
)
.unwrap();
c.execute(
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)",
[],
)
.unwrap();
c.execute("INSERT INTO remote(image_id, file_id) VALUES (1, 5000)", [])
.unwrap();
c.execute(
"INSERT INTO faces
(id, image_id, x, y, w, h, landmarks, detector_confidence, embedding,
crop_px, model_id, detected_at, crop)
VALUES (?1, 1, ?2, 0.2, 0.2, 0.2, X'00', 0.9, X'00', 150.0, 'w600k_mbf', 0, ?3)",
rusqlite::params![face_id, x, crop],
)
.unwrap();
}
fn crop_of(c: &Connection, face_id: i64) -> Option<Vec<u8>> {
c.query_row("SELECT crop FROM faces WHERE id = ?1", [face_id], |r| {
r.get(0)
})
.unwrap()
}
/// The snapshot is built table by table rather than copied, so what has to
/// hold is that it is still the same database bar the crops: every table,
/// index and row, and the header fields an older build checks before it
/// will merge (`user_version`) or open it the way it always has (the WAL
/// flag and page size a backup-API copy carried).
#[test]
fn the_snapshot_is_the_catalog_bar_the_crops() {
let dir = tempdir();
let live = dir.join("catalog.sqlite");
let snap = dir.join("snap.sqlite");
let c = seeded(&live);
with_a_face(&c, 7, 0.3, &[7u8; 4096]);
c.execute(
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
VALUES ('u1', 'Iceland', 0, 0, 1, 1)",
[],
)
.unwrap();
// Created on first use rather than by a migration: the copy must not
// depend on the migrations knowing every table.
crate::duplicates::ensure_probe_table(&c).unwrap();
snapshot_for_upload(&c, &snap).unwrap();
let out = Connection::open(&snap).unwrap();
let objects = |conn: &Connection| -> Vec<(String, String)> {
let mut stmt = conn
.prepare("SELECT type, name FROM sqlite_master ORDER BY type, name")
.unwrap();
let rows = stmt
.query_map([], |r| Ok((r.get(0).unwrap(), r.get(1).unwrap())))
.unwrap();
rows.map(Result::unwrap).collect()
};
assert_eq!(objects(&out), objects(&c), "the snapshot's schema differs");
for (kind, table) in objects(&c) {
if kind != "table" {
continue;
}
let count = |conn: &Connection| -> i64 {
conn.query_row(&format!("SELECT COUNT(*) FROM \"{table}\""), [], |r| {
r.get(0)
})
.unwrap()
};
assert_eq!(count(&out), count(&c), "rows differ in {table}");
}
for pragma in [
"user_version",
"application_id",
"page_size",
"journal_mode",
] {
let read = |conn: &Connection| -> String {
conn.query_row(&format!("PRAGMA {pragma}"), [], |r| {
r.get::<_, rusqlite::types::Value>(0)
})
.map(|v| format!("{v:?}"))
.unwrap()
};
assert_eq!(read(&out), read(&c), "{pragma} differs");
}
drop(out);
// Bytes 18 and 19 of the header are 2 for a WAL database, which is
// what every snapshot uploaded before this one said.
let header = std::fs::read(&snap).unwrap();
assert_eq!(&header[18..20], &[2, 2], "the snapshot is not WAL-flagged");
// The face is there; its pixels are not.
let out = Connection::open(&snap).unwrap();
assert_eq!(crop_of(&out, 7), None);
}
/// A pass that died mid-build leaves a file behind; the next one must
/// build afresh rather than on top of it.
#[test]
fn a_leftover_snapshot_is_replaced_not_built_on() {
let dir = tempdir();
let live = dir.join("catalog.sqlite");
let snap = dir.join("snap.sqlite");
let c = seeded(&live);
std::fs::write(&snap, b"not a database").unwrap();
snapshot_for_upload(&c, &snap).unwrap();
// And twice over a good one, which is the steady state.
snapshot_for_upload(&c, &snap).unwrap();
let out = Connection::open(&snap).unwrap();
let v: i64 = out
.query_row("PRAGMA user_version", [], |r| r.get(0))
.unwrap();
assert_eq!(v, schema::SCHEMA_VERSION);
}
/// The merge side of a crop-less snapshot: the other device's names still
/// cross over — the match is by box, not by pixels — and this device's
/// own crop is left exactly as it was, never replaced by the snapshot's
/// NULL.
#[test]
fn merging_a_crop_less_snapshot_keeps_local_crops_and_takes_the_names() {
let dir = tempdir();
let snap = dir.join("snap.sqlite");
let desktop = seeded(&dir.join("desktop.sqlite"));
with_a_face(&desktop, 42, 0.31, &[1u8; 3000]);
desktop
.execute(
"INSERT INTO people(id, uuid, name, ignored, created, revision, modified)
VALUES (3, 'u-anna', 'Anna', 0, 0, 1, 1)",
[],
)
.unwrap();
desktop
.execute(
"INSERT INTO face_person(face_id, person_id, probability, confirmed)
VALUES (42, 3, 0.9, 1)",
[],
)
.unwrap();
snapshot_for_upload(&desktop, &snap).unwrap();
// The tablet found the same face itself, under its own row id, and has
// its own crop of it — from its own detection or from the shards.
let tablet = seeded(&dir.join("tablet.sqlite"));
let mine = vec![9u8; 2500];
with_a_face(&tablet, 7, 0.30, &mine);
let report = merge_remote(&tablet, &snap).unwrap();
assert_eq!(report.people_inserted, 1);
assert_eq!(report.faces_assigned, 1);
let named: (String, bool) = tablet
.query_row(
"SELECT p.name, fp.confirmed
FROM face_person fp JOIN people p ON p.id = fp.person_id
WHERE fp.face_id = 7",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(named, ("Anna".to_string(), true));
assert_eq!(
crop_of(&tablet, 7),
Some(mine),
"the merge touched a local crop"
);
// Idempotent over a crop-less remote too.
let again = merge_remote(&tablet, &snap).unwrap();
assert!(!again.local_changed());
}
}
-648
View File
@@ -1,648 +0,0 @@
//! TRACES: FR-CAT-15 | NFR-R2
//! Soft delete, restore, and the permanent delete that follows.
//!
//! # Why the trash is a folder and not a flag
//!
//! The catalog is a *rebuildable index* (ARCH §6.12): delete `catalog.sqlite`
//! and it is reconstructed by rescanning sources. A trash implemented as a
//! column alone would therefore not survive its own design — a rebuild would
//! find every trashed file still sitting in the library and re-index it as an
//! ordinary photograph, silently undoing every delete the user had made.
//!
//! So a soft delete **moves the file** into `.darkroom-trash/` under the library
//! root, and the catalog merely records that this happened. The folder is the
//! durable fact; the row is the convenience. Recovering by hand needs no
//! DarkRoom at all, which is the property that matters when the thing being
//! risked is a photograph.
//!
//! `dr_sync::scan::is_excluded` keeps the scanner out of that folder. Without
//! it the next scan re-indexes the trash and the delete comes undone — the two
//! halves are one mechanism and neither works alone.
//!
//! # The two steps
//!
//! **Soft** ([`trash`]) — `MOVE` to the trash folder, record `trashed_at` and
//! the path it came from. Reversible by [`restore`], which is why the original
//! path has to be remembered: the trash is flat, and the folder structure cannot
//! be recovered from the trashed name.
//!
//! **Hard** ([`purge`]) — `DELETE` the file, then delete the row. Irreversible
//! from DarkRoom's side, though the server's own trashbin may still hold it.
//! Ordered file-first deliberately: see [`purge_order`].
//!
//! # What this module does not do
//!
//! It performs no I/O. Every function here records or reads catalog state, and
//! the caller pairs it with the remote operation — because the remote call is
//! async and the catalog is not, and because the *order* of the two is a
//! correctness property that belongs in one visible place rather than buried in
//! a transaction.
use rusqlite::{Connection, OptionalExtension};
use dr_types::ImageId;
use crate::error::CatalogError;
/// Directory holding soft-deleted images, under the library root.
///
/// The same constant `dr_sync::scan` excludes. Duplicated as a `const` here
/// rather than depended upon because `dr-catalog` does not (and should not)
/// depend on `dr-sync`; the pairing is asserted by a test.
pub const TRASH_DIR: &str = ".darkroom-trash";
/// One trashed image, as the trash view lists it.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct TrashedImage {
pub image_id: ImageId,
/// Where the file is *now* — inside the trash folder.
pub source_ref: String,
/// Where it was before, and where [`restore`] will put it back.
pub trashed_from: String,
/// UTC seconds when it was trashed.
pub trashed_at: i64,
/// `oc:fileid`, preserved across the move. What the thumbnail store keys on,
/// and what makes a restore free rather than a re-download.
pub file_id: Option<u64>,
pub size: u64,
}
/// The path a soft-deleted image should be moved to.
///
/// Flat: the trash is a holding area, not an archive, and mirroring the library
/// tree inside it would mean creating directories on the way to deleting things.
/// The original path is remembered in the catalog instead, which is what
/// [`restore`] reads.
///
/// **Collisions are resolved rather than allowed to overwrite.** Two files named
/// `IMG_0001.CR2` from different folders are different photographs, and a `MOVE`
/// onto an existing name would destroy one of them — the precise failure a trash
/// exists to prevent. The image id disambiguates, and being already unique it
/// needs no retry loop.
pub fn trash_path(root: &str, image: ImageId, original: &str) -> String {
let name = original.rsplit(['/', ':']).next().unwrap_or(original);
let prefix = if root.is_empty() {
String::new()
} else {
format!("{root}/")
};
format!("{prefix}{TRASH_DIR}/{}-{name}", image.0)
}
/// Where a trashed image goes back to.
///
/// The stored original path, verbatim. Returns `None` where the image is not
/// trashed, so a caller cannot restore something that was never deleted.
pub fn restore_path(conn: &Connection, image: ImageId) -> Result<Option<String>, CatalogError> {
let path: Option<String> = conn
.query_row(
"SELECT trashed_from FROM images
WHERE id = ?1 AND trashed_at IS NOT NULL",
[image.0 as i64],
|r| r.get(0),
)
.optional()?
.flatten();
Ok(path)
}
/// Record that images have been moved to the trash.
///
/// Call **after** the move succeeds. Recording first and moving second would
/// leave the catalog claiming a file is trashed while it sits in the library,
/// where the next scan finds it — and since the scan excludes the trash folder,
/// the row would never be corrected.
///
/// `moved` pairs each image with the path it now occupies, which is what
/// [`trash_path`] produced for it.
///
/// Idempotent on `trashed_at`: re-trashing an already-trashed image keeps the
/// *original* timestamp and original path, so a retry after a partial failure
/// cannot rewrite `trashed_from` to a path inside the trash — which would make
/// the image unrestorable.
pub fn record_trashed(
conn: &Connection,
moved: &[(ImageId, String)],
now: i64,
) -> Result<usize, CatalogError> {
if moved.is_empty() {
return Ok(0);
}
let tx = conn.unchecked_transaction()?;
let n = record_trashed_within(&tx, moved, now)?;
tx.commit()?;
Ok(n)
}
/// [`record_trashed`] inside a transaction the caller owns.
///
/// For a caller whose trash is one half of a larger write that must land
/// whole or not at all — consolidating duplicates (`crate::duplicates`)
/// merges a copy's judgements onto the survivor and trashes the copy in one
/// commit. `unchecked_transaction` cannot nest, so this is offered here
/// rather than wrapped from above.
pub fn record_trashed_within(
tx: &Connection,
moved: &[(ImageId, String)],
now: i64,
) -> Result<usize, CatalogError> {
let mut n = 0;
let mut stmt = tx.prepare(
"UPDATE images
SET trashed_from = CASE
WHEN trashed_at IS NULL THEN source_ref
ELSE trashed_from
END,
source_ref = ?2,
trashed_at = coalesce(trashed_at, ?3)
WHERE id = ?1",
)?;
for (image, path) in moved {
n += stmt.execute(rusqlite::params![image.0 as i64, path, now])?;
}
Ok(n)
}
/// Record that images have been moved back out of the trash.
///
/// Call after the move succeeds, for the same reason as [`record_trashed`].
/// Clears both columns: a restored image is an ordinary one, and leaving
/// `trashed_from` set would make the next trash-and-restore cycle restore it to
/// a stale location.
pub fn record_restored(
conn: &Connection,
restored: &[(ImageId, String)],
) -> Result<usize, CatalogError> {
if restored.is_empty() {
return Ok(0);
}
let tx = conn.unchecked_transaction()?;
let mut n = 0;
{
let mut stmt = tx.prepare(
"UPDATE images
SET source_ref = ?2, trashed_at = NULL, trashed_from = NULL
WHERE id = ?1 AND trashed_at IS NOT NULL",
)?;
for (image, path) in restored {
n += stmt.execute(rusqlite::params![image.0 as i64, path])?;
}
}
tx.commit()?;
Ok(n)
}
/// Forget images whose files have been permanently deleted.
///
/// Call **after** the remote delete succeeds — see [`purge_order`].
///
/// Deletes the catalog rows outright rather than tombstoning them. There is
/// nothing to merge: unlike a collection, an image row is derived from a file
/// that no longer exists, so a rescan on another device will not reintroduce it
/// and needs no tombstone to be told so. `ON DELETE CASCADE` takes the versions,
/// keywords, remote mapping and cache rows with it.
///
/// Returns how many rows went.
pub fn forget(conn: &Connection, images: &[ImageId]) -> Result<usize, CatalogError> {
if images.is_empty() {
return Ok(0);
}
let tx = conn.unchecked_transaction()?;
let mut n = 0;
{
let mut stmt = tx.prepare("DELETE FROM images WHERE id = ?1")?;
for image in images {
n += stmt.execute([image.0 as i64])?;
}
}
tx.commit()?;
Ok(n)
}
/// Why the file is deleted before the row.
///
/// Not a function — a note with a name, so the reasoning is findable from the
/// call site.
///
/// **File first, then the row.** If the delete succeeds and the process dies
/// before the row goes, the catalog holds a trashed row whose file is gone; the
/// user sees it in the trash, empties again, gets a `404`, and it is treated as
/// already-deleted (see [`is_already_gone`]). Recoverable, and visible.
///
/// The other order loses the file silently. Dropping the row first and dying
/// before the delete leaves an orphan in `.darkroom-trash/` that nothing in the
/// UI lists, nothing counts, and no scan will ever find — because the scanner
/// excludes that folder. It consumes quota forever and the user has no way to
/// learn it is there.
pub const fn purge_order() {}
/// Whether a delete failure means the file was already gone.
///
/// A `404` on the way to deleting something is success: the goal state is
/// "this file does not exist", and it does not. Treating it as an error would
/// wedge an empty-trash operation on a file the user had removed by hand, and
/// no amount of retrying would clear it.
pub fn is_already_gone(status: Option<u16>) -> bool {
matches!(status, Some(404) | Some(410))
}
/// List what is in the trash, newest first.
///
/// Newest first because the trash is reviewed to undo a recent mistake, not
/// browsed chronologically.
pub fn list(conn: &Connection, limit: usize) -> Result<Vec<TrashedImage>, CatalogError> {
let mut stmt = conn.prepare(
"SELECT i.id, i.source_ref, i.trashed_from, i.trashed_at, r.file_id, i.file_size
FROM images i
LEFT JOIN remote r ON r.image_id = i.id
WHERE i.trashed_at IS NOT NULL
ORDER BY i.trashed_at DESC, i.id DESC
LIMIT ?1",
)?;
let rows = stmt
.query_map([limit as i64], |r| {
let source_ref: String = r.get(1)?;
Ok(TrashedImage {
image_id: ImageId(r.get::<_, i64>(0)? as u64),
// A row with no `trashed_from` predates nothing — it cannot
// happen through this module — but a hand-edited or
// partially-migrated catalog could produce one. Falling back to
// the current path keeps it listed and deletable rather than
// invisible; a restore to the trash folder is a no-op the user
// can see, where a hidden row is not.
trashed_from: r
.get::<_, Option<String>>(2)?
.unwrap_or_else(|| source_ref.clone()),
source_ref,
trashed_at: r.get(3)?,
file_id: r.get::<_, Option<i64>>(4)?.map(|v| v as u64),
size: r.get::<_, Option<i64>>(5)?.unwrap_or(0) as u64,
})
})?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// Every trashed image id, for emptying the whole trash.
///
/// Separate from [`list`] because emptying needs all of them, not a window, and
/// wants no per-row detail.
pub fn all_trashed(conn: &Connection) -> Result<Vec<ImageId>, CatalogError> {
let mut stmt = conn.prepare("SELECT id FROM images WHERE trashed_at IS NOT NULL")?;
let rows = stmt
.query_map([], |r| Ok(ImageId(r.get::<_, i64>(0)? as u64)))?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
/// How many images are in the trash, and how many bytes they hold.
///
/// The bytes are the point: "empty trash" is a destructive action, and the
/// amount being freed is what tells the user whether they meant it.
pub fn summary(conn: &Connection) -> Result<(usize, u64), CatalogError> {
let (n, bytes): (i64, i64) = conn.query_row(
"SELECT count(*), coalesce(sum(file_size), 0)
FROM images WHERE trashed_at IS NOT NULL",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)?;
Ok((n as usize, bytes as u64))
}
/// `oc:fileid`s of trashed images, so their thumbnails can be dropped.
///
/// The thumbnail store is keyed on the stable file id and shared with other
/// clients, so a purge that left its entries behind would keep serving previews
/// of photographs that no longer exist — and the shards sync, so it would keep
/// doing so on every other device too.
pub fn file_ids_for(conn: &Connection, images: &[ImageId]) -> Result<Vec<u64>, CatalogError> {
if images.is_empty() {
return Ok(Vec::new());
}
let placeholders = std::iter::repeat_n("?", images.len())
.collect::<Vec<_>>()
.join(",");
let sql = format!("SELECT file_id FROM remote WHERE image_id IN ({placeholders})");
let params: Vec<rusqlite::types::Value> = images
.iter()
.map(|i| rusqlite::types::Value::Integer(i.0 as i64))
.collect();
let mut stmt = conn.prepare(&sql)?;
let rows = stmt
.query_map(rusqlite::params_from_iter(params.iter()), |r| {
Ok(r.get::<_, i64>(0)? as u64)
})?
.collect::<Result<Vec<_>, _>>()?;
Ok(rows)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::Catalog;
fn seeded() -> Catalog {
let cat = Catalog::in_memory().unwrap();
let c = cat.connection();
c.execute(
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'PhotosRaw')",
[],
)
.unwrap();
for i in 1..=4i64 {
c.execute(
"INSERT INTO images(id, root_id, source_ref, file_size, added_at)
VALUES (?1, 1, ?2, ?3, 0)",
rusqlite::params![i, format!("PhotosRaw/2019/IMG_{i:04}.CR2"), 30_000_000 * i],
)
.unwrap();
c.execute(
"INSERT INTO remote(image_id, file_id) VALUES (?1, ?2)",
rusqlite::params![i, 1000 + i],
)
.unwrap();
}
cat
}
fn img(i: u64) -> ImageId {
ImageId(i)
}
/// Trash one image the way the UI does: compute the path, then record.
fn do_trash(cat: &Catalog, i: u64, now: i64) -> String {
let c = cat.connection();
let original: String = c
.query_row(
"SELECT source_ref FROM images WHERE id = ?1",
[i as i64],
|r| r.get(0),
)
.unwrap();
let to = trash_path("PhotosRaw", img(i), &original);
record_trashed(c, &[(img(i), to.clone())], now).unwrap();
to
}
#[test]
fn the_trash_directory_matches_the_one_the_scanner_excludes() {
// These are two constants in two crates that must agree, or the scan
// re-indexes the trash and every soft delete comes undone.
assert_eq!(TRASH_DIR, dr_sync_trash_dir());
}
/// The scanner's constant, quoted rather than imported — `dr-catalog` does
/// not depend on `dr-sync`, and adding that dependency for one string would
/// invert the layering.
fn dr_sync_trash_dir() -> &'static str {
".darkroom-trash"
}
#[test]
fn trashing_moves_the_path_and_remembers_where_it_came_from() {
let cat = seeded();
let c = cat.connection();
do_trash(&cat, 1, 5_000);
let (source, from, at): (String, String, i64) = c
.query_row(
"SELECT source_ref, trashed_from, trashed_at FROM images WHERE id = 1",
[],
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)),
)
.unwrap();
// `source_ref` follows the bytes: this is where a fetch must now look.
assert!(source.contains(TRASH_DIR), "{source}");
// And the original is remembered, or a restore has nowhere to go.
assert_eq!(from, "PhotosRaw/2019/IMG_0001.CR2");
assert_eq!(at, 5_000);
}
#[test]
fn the_trash_path_keeps_the_original_filename_recognisable() {
// The user reviewing the trash needs to recognise the photograph; an
// opaque id alone would make the list unreadable.
let p = trash_path("PhotosRaw", img(7), "PhotosRaw/2019/IMG_0042.CR2");
assert!(p.ends_with("IMG_0042.CR2"), "{p}");
assert!(p.starts_with("PhotosRaw/.darkroom-trash/"), "{p}");
}
#[test]
fn two_files_with_the_same_name_do_not_collide_in_the_trash() {
// The failure a trash exists to prevent: a MOVE onto an existing name
// destroys one of two different photographs.
let a = trash_path("PhotosRaw", img(1), "PhotosRaw/2019/IMG_0001.CR2");
let b = trash_path("PhotosRaw", img(2), "PhotosRaw/2024/IMG_0001.CR2");
assert_ne!(a, b);
}
#[test]
fn a_whole_account_root_yields_no_leading_slash() {
// The root is empty when the library is the whole account; a path
// beginning "/" would resolve differently on the server.
let p = trash_path("", img(3), "2019/IMG_0003.CR2");
assert_eq!(p, ".darkroom-trash/3-IMG_0003.CR2");
}
#[test]
fn restoring_puts_the_original_path_back_and_clears_the_flag() {
let cat = seeded();
let c = cat.connection();
do_trash(&cat, 1, 5_000);
let back = restore_path(c, img(1))
.unwrap()
.expect("knows where it came from");
assert_eq!(back, "PhotosRaw/2019/IMG_0001.CR2");
record_restored(c, &[(img(1), back.clone())]).unwrap();
let (source, at): (String, Option<i64>) = c
.query_row(
"SELECT source_ref, trashed_at FROM images WHERE id = 1",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(source, back);
assert_eq!(at, None, "a restored image is an ordinary one");
assert!(restore_path(c, img(1)).unwrap().is_none());
}
#[test]
fn a_trash_restore_trash_cycle_restores_to_the_right_place_twice() {
// If `trashed_from` were not cleared on restore, the second trash would
// record a stale origin and the second restore would put the file
// somewhere it never was.
let cat = seeded();
let c = cat.connection();
do_trash(&cat, 1, 1_000);
let first = restore_path(c, img(1)).unwrap().unwrap();
record_restored(c, &[(img(1), first.clone())]).unwrap();
do_trash(&cat, 1, 2_000);
let second = restore_path(c, img(1)).unwrap().unwrap();
assert_eq!(
first, second,
"the origin is the library path, not the trash"
);
}
#[test]
fn re_trashing_does_not_overwrite_the_original_path() {
// A retry after a partial failure must not record a trash-folder path as
// the origin — that makes the image unrestorable.
let cat = seeded();
let c = cat.connection();
let to = do_trash(&cat, 1, 1_000);
// Second attempt, as a retry would do.
record_trashed(c, &[(img(1), to)], 9_999).unwrap();
let (from, at): (String, i64) = c
.query_row(
"SELECT trashed_from, trashed_at FROM images WHERE id = 1",
[],
|r| Ok((r.get(0)?, r.get(1)?)),
)
.unwrap();
assert_eq!(from, "PhotosRaw/2019/IMG_0001.CR2");
assert_eq!(at, 1_000, "the original timestamp survives a retry");
}
#[test]
fn restoring_something_that_was_never_trashed_does_nothing() {
let cat = seeded();
let c = cat.connection();
assert!(restore_path(c, img(2)).unwrap().is_none());
assert_eq!(
record_restored(c, &[(img(2), "elsewhere".into())]).unwrap(),
0
);
// And its path is untouched.
let source: String = c
.query_row("SELECT source_ref FROM images WHERE id = 2", [], |r| {
r.get(0)
})
.unwrap();
assert_eq!(source, "PhotosRaw/2019/IMG_0002.CR2");
}
#[test]
fn the_trash_lists_newest_first() {
// Reviewed to undo a recent mistake, not browsed chronologically.
let cat = seeded();
do_trash(&cat, 1, 1_000);
do_trash(&cat, 2, 3_000);
do_trash(&cat, 3, 2_000);
let listed = list(cat.connection(), 100).unwrap();
let order: Vec<u64> = listed.iter().map(|t| t.image_id.0).collect();
assert_eq!(order, vec![2, 3, 1]);
}
#[test]
fn the_trash_list_carries_the_file_id_a_restore_needs() {
// Without it a restore cannot find the thumbnail it already has, and
// re-downloads a preview it is holding.
let cat = seeded();
do_trash(&cat, 1, 1_000);
let listed = list(cat.connection(), 10).unwrap();
assert_eq!(listed[0].file_id, Some(1001));
}
#[test]
fn the_summary_reports_what_emptying_would_free() {
// "Empty trash" is destructive; the size is what tells the user whether
// they meant it.
let cat = seeded();
do_trash(&cat, 1, 1_000);
do_trash(&cat, 2, 1_000);
let (n, bytes) = summary(cat.connection()).unwrap();
assert_eq!(n, 2);
assert_eq!(bytes, 30_000_000 + 60_000_000);
}
#[test]
fn an_empty_trash_summarises_as_zero_rather_than_erroring() {
let cat = seeded();
assert_eq!(summary(cat.connection()).unwrap(), (0, 0));
assert!(all_trashed(cat.connection()).unwrap().is_empty());
}
#[test]
fn purging_removes_the_row_and_everything_hanging_off_it() {
let cat = seeded();
let c = cat.connection();
do_trash(&cat, 1, 1_000);
assert_eq!(forget(c, &[img(1)]).unwrap(), 1);
let n: i64 = c
.query_row("SELECT count(*) FROM images WHERE id = 1", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 0);
// The remote mapping must go too, or a later scan could pair a new file
// with a dead image's id.
let n: i64 = c
.query_row("SELECT count(*) FROM remote WHERE image_id = 1", [], |r| {
r.get(0)
})
.unwrap();
assert_eq!(n, 0, "cascaded");
}
#[test]
fn purging_leaves_untrashed_images_alone() {
let cat = seeded();
let c = cat.connection();
do_trash(&cat, 1, 1_000);
forget(c, &all_trashed(c).unwrap()).unwrap();
let n: i64 = c
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
.unwrap();
assert_eq!(n, 3, "only the trashed one went");
}
#[test]
fn file_ids_are_collected_so_thumbnails_can_be_dropped() {
// The shards sync to the server; a purge that left them would serve
// previews of deleted photographs on every device.
let cat = seeded();
let c = cat.connection();
do_trash(&cat, 1, 1_000);
do_trash(&cat, 2, 1_000);
let mut ids = file_ids_for(c, &[img(1), img(2)]).unwrap();
ids.sort_unstable();
assert_eq!(ids, vec![1001, 1002]);
}
#[test]
fn a_missing_file_counts_as_already_deleted() {
// Otherwise one file removed by hand wedges every future empty-trash,
// and no amount of retrying clears it.
assert!(is_already_gone(Some(404)));
assert!(is_already_gone(Some(410)));
assert!(!is_already_gone(Some(403)), "a permission failure is real");
assert!(!is_already_gone(Some(500)));
assert!(!is_already_gone(None));
}
#[test]
fn empty_batches_are_no_ops_rather_than_errors() {
// The UI can reach these with nothing selected.
let cat = seeded();
let c = cat.connection();
assert_eq!(record_trashed(c, &[], 0).unwrap(), 0);
assert_eq!(record_restored(c, &[]).unwrap(), 0);
assert_eq!(forget(c, &[]).unwrap(), 0);
assert!(file_ids_for(c, &[]).unwrap().is_empty());
}
}
File diff suppressed because it is too large Load Diff
@@ -1,341 +0,0 @@
// TRACES: FR-PLAT-AND-3
//! No job kind is enqueued without something that claims it.
//!
//! The queue coalesces, so a producer with no consumer does not fail — it
//! just leaves a row per subject for ever. That is how the reference catalog
//! came to hold 23,582 `Thumbnail` jobs, one per photograph, re-coalesced on
//! every scan, with no handler for the kind anywhere in the tree (#73). Nothing
//! at runtime notices: the rows are cheap one at a time and invisible in the
//! interface. So the pairing is checked here, over the source, instead.
//!
//! ## What counts
//!
//! In shipping code under `core/`, `ui/`, `apps/` and `platform/` — every
//! `src/` tree, with `#[cfg(test)]` items dropped:
//!
//! - **Enqueued**: the `JobKind::X` named in the arguments of a call to
//! `enqueue(`. An enqueue whose kind is not spelled there — passed in a
//! variable — is refused outright, because this scan could not say what it
//! queues.
//! - **Claimed**: the `JobKind::X` in the body of a `fn kinds(` (what a
//! `JobHandler` declares, and all a `Runner` claims), or named in a call to
//! `claim_next_matching(`. A call to `claim_next(` claims every kind.
//! `JobKind::ALL` in either place means every kind.
//!
//! Tests and examples are left out on purpose: a unit test of the queue's
//! mechanics enqueues and claims whatever it likes, and proves nothing about
//! the app.
use std::collections::BTreeSet;
use std::fs;
use std::path::{Path, PathBuf};
/// Every `.rs` file under each `src/` of each crate in `group`.
fn crate_sources(group: &Path, out: &mut Vec<PathBuf>) {
let Ok(crates) = fs::read_dir(group) else {
return;
};
for krate in crates {
let src = krate.expect("read dir entry").path().join("src");
if src.is_dir() {
rust_files(&src, out);
}
}
}
fn rust_files(dir: &Path, out: &mut Vec<PathBuf>) {
for entry in fs::read_dir(dir).unwrap_or_else(|e| panic!("cannot read {}: {e}", dir.display()))
{
let path = entry.expect("read dir entry").path();
if path.is_dir() {
rust_files(&path, out);
} else if path.extension().and_then(|e| e.to_str()) == Some("rs") {
out.push(path);
}
}
}
/// Blank out string literals and line comments, so neither a brace nor a
/// `JobKind::` inside prose is read as code.
fn strip_literals_and_comments(line: &str) -> String {
let mut out = String::with_capacity(line.len());
let mut chars = line.chars().peekable();
let mut in_string = false;
while let Some(c) = chars.next() {
if in_string {
match c {
'\\' => {
chars.next();
}
'"' => in_string = false,
_ => {}
}
continue;
}
match c {
'"' => in_string = true,
'/' if chars.peek() == Some(&'/') => break,
_ => out.push(c),
}
}
out
}
/// The shipping code of a file, comments and strings blanked, with every
/// `#[cfg(test)]` item dropped. The attribute must be the whole line, so a
/// doc comment mentioning it is not mistaken for one.
fn shipping_code(text: &str) -> String {
let lines: Vec<String> = text.lines().map(strip_literals_and_comments).collect();
let mut out = String::new();
let mut i = 0;
while i < lines.len() {
if lines[i].trim() != "#[cfg(test)]" {
out.push_str(&lines[i]);
out.push('\n');
i += 1;
continue;
}
let (mut j, mut depth, mut opened) = (i + 1, 0i32, false);
while j < lines.len() {
depth += lines[j].matches('{').count() as i32;
depth -= lines[j].matches('}').count() as i32;
opened |= lines[j].contains('{');
if (opened && depth <= 0) || (!opened && lines[j].contains(';')) {
break;
}
j += 1;
}
i = j + 1;
}
out
}
/// The text from `start` (just past an opening delimiter) to its matching
/// close.
fn balanced(code: &str, start: usize, open: char, close: char) -> &str {
let mut depth = 1;
for (i, c) in code[start..].char_indices() {
if c == open {
depth += 1;
} else if c == close {
depth -= 1;
if depth == 0 {
return &code[start..start + i];
}
}
}
&code[start..]
}
/// Each call of `name(` in `code` that is a call rather than the function's
/// own definition or a longer name ending in it, as its argument text.
fn calls<'a>(code: &'a str, name: &str) -> Vec<&'a str> {
let needle = format!("{name}(");
let mut found = Vec::new();
for (at, _) in code.match_indices(&needle) {
let before = &code[..at];
let prev = before.chars().next_back();
if prev.is_some_and(|c| c.is_alphanumeric() || c == '_') {
continue;
}
if before.trim_end().ends_with("fn") {
continue;
}
found.push(balanced(code, at + needle.len(), '(', ')'));
}
found
}
/// Bodies of every `fn kinds(` that has one — a trait declaration ending in
/// `;` has none.
fn kinds_bodies(code: &str) -> Vec<&str> {
let mut found = Vec::new();
for (at, _) in code.match_indices("fn kinds(") {
let rest = &code[at..];
let (Some(brace), semi) = (rest.find('{'), rest.find(';')) else {
continue;
};
if semi.is_some_and(|s| s < brace) {
continue;
}
found.push(balanced(code, at + brace + 1, '{', '}'));
}
found
}
/// The `X` of each `JobKind::X` in `text`.
fn kinds_named(text: &str) -> Vec<String> {
text.match_indices("JobKind::")
.map(|(at, m)| {
text[at + m.len()..]
.chars()
.take_while(|c| c.is_alphanumeric() || *c == '_')
.collect()
})
.collect()
}
#[derive(Default, Debug)]
struct Ledger {
/// Kind → where it is enqueued.
enqueued: Vec<(String, String)>,
/// Enqueue calls whose kind could not be read.
unreadable: Vec<String>,
claimed: BTreeSet<String>,
claims_everything: bool,
}
fn read(files: &[(String, String)]) -> Ledger {
let mut ledger = Ledger::default();
for (name, text) in files {
let code = shipping_code(text);
for args in calls(&code, "enqueue") {
let kinds = kinds_named(args);
if kinds.is_empty() {
ledger
.unreadable
.push(format!("{name}: enqueue({})", args.trim()));
}
for k in kinds {
ledger.enqueued.push((k, name.clone()));
}
}
let claimed = kinds_bodies(&code)
.into_iter()
.chain(calls(&code, "claim_next_matching"));
for text in claimed {
for k in kinds_named(text) {
if k == "ALL" {
ledger.claims_everything = true;
} else {
ledger.claimed.insert(k);
}
}
}
if !calls(&code, "claim_next").is_empty() {
ledger.claims_everything = true;
}
}
ledger
}
#[test]
fn every_kind_enqueued_is_claimed_by_something() {
let repo = Path::new(env!("CARGO_MANIFEST_DIR"))
.parent()
.and_then(Path::parent)
.expect("core/dr-catalog has a grandparent");
let mut paths = Vec::new();
for group in ["core", "ui", "apps", "platform"] {
crate_sources(&repo.join(group), &mut paths);
}
let files: Vec<(String, String)> = paths
.iter()
.map(|p| {
let text = fs::read_to_string(p)
.unwrap_or_else(|e| panic!("cannot read {}: {e}", p.display()));
let name = p.strip_prefix(repo).unwrap_or(p).display().to_string();
(name, text)
})
.collect();
// A scan over nothing passes for the wrong reason. The queue's own file
// and the scan that used to feed it must both have been read, and the
// queue's definitions found in them.
for must in [
"core/dr-catalog/src/jobs.rs",
"core/dr-catalog/src/runner.rs",
"ui/dr-ui/src/library/scan.rs",
] {
assert!(
files.iter().any(|(n, _)| n == must),
"{must} was not scanned — the source walk is wrong, not the code"
);
}
let jobs = &files
.iter()
.find(|(n, _)| n == "core/dr-catalog/src/jobs.rs")
.unwrap()
.1;
assert!(shipping_code(jobs).contains("pub fn enqueue("));
let ledger = read(&files);
assert!(
ledger.unreadable.is_empty(),
"\n\nThese enqueue calls do not name their JobKind, so this test cannot \
check that anything claims it. Spell the kind at the call:\n {}\n",
ledger.unreadable.join("\n ")
);
if ledger.claims_everything {
return;
}
let orphans: Vec<String> = ledger
.enqueued
.iter()
.filter(|(k, _)| !ledger.claimed.contains(k))
.map(|(k, at)| format!("JobKind::{k}, enqueued in {at}"))
.collect();
assert!(
orphans.is_empty(),
"\n\nEnqueued, and claimed by nothing (claimed: {:?}):\n {}\n\n\
A kind nobody claims is a row per subject that stays for ever — the \
queue coalesces, so it never fails, it only grows (#73). Register a \
JobHandler for the kind, or stop enqueueing it and add it to \
JobKind::RETIRED so the rows already queued are dropped.\n",
ledger.claimed,
orphans.join("\n ")
);
}
/// The reader itself, on code whose answer is known — so a parsing bug shows
/// up as this failing, rather than the real check quietly finding nothing.
#[test]
fn the_reader_sees_producers_and_consumers() {
let producer = r#"
use dr_catalog::jobs;
fn persist(tx: &Connection, id: i64) {
// jobs::enqueue(tx, JobKind::ContentHash, ...) in a comment is not a call
let _ = dr_catalog::jobs::enqueue(
tx,
JobKind::Thumbnail,
Some(id),
Priority::Background,
None,
);
jobs::enqueue(tx, kind, Some(id), Priority::Background, None)?;
}
pub fn enqueue(conn: &Connection, kind: JobKind) {}
#[cfg(test)]
mod tests {
fn t() { enqueue(&c, JobKind::FetchOriginal, None, P, None); }
}
"#;
let consumer = r#"
impl JobHandler for Faces {
fn kinds(&self) -> &[JobKind] {
&[JobKind::DetectFaces]
}
fn run(&mut self) {}
}
trait JobHandler { fn kinds(&self) -> &[JobKind]; }
fn pull(c: &Connection) { claim_next_matching(c, 0, &[JobKind::FetchPreview]); }
"#;
let ledger = read(&[
("producer.rs".into(), producer.into()),
("consumer.rs".into(), consumer.into()),
]);
let enqueued: Vec<&str> = ledger.enqueued.iter().map(|(k, _)| k.as_str()).collect();
assert_eq!(enqueued, vec!["Thumbnail"]);
assert_eq!(ledger.unreadable.len(), 1, "{:?}", ledger.unreadable);
assert_eq!(
ledger.claimed,
BTreeSet::from(["DetectFaces".to_string(), "FetchPreview".to_string()])
);
assert!(!ledger.claims_everything);
}
-23
View File
@@ -1,23 +0,0 @@
[package]
name = "dr-decode"
version.workspace = true
edition.workspace = true
rust-version.workspace = true
license.workspace = true
[dependencies]
dr-types.workspace = true
rawler.workspace = true
# The camera profile database is data, not code (FR-DEV-3e): a YAML file that
# ships with the binary and is superseded by a newer one on disk. serde_norway
# is the workspace's YAML crate — the fork still receiving releases — and it is
# already in the tree for `dr-pipeline`'s node declarations and `dr-ui`'s style
# tokens. Pure Rust, so it costs nothing under the Android NDK.
serde = { workspace = true }
serde_norway.workspace = true
zune-jpeg.workspace = true
thiserror.workspace = true
log.workspace = true
[dev-dependencies]
env_logger.workspace = true
-52
View File
@@ -1,52 +0,0 @@
//! Report the defect map a raw file carries, if it carries one.
//!
//! ```text
//! cargo run -p dr-decode --example defects -- IMG_6320.dng photo.cr2
//! ```
//!
//! Exists because whether this is worth building a correction stage for is a
//! question about *your files*, not about the specification: DNGs written by
//! cameras that map their own sensors carry `OpcodeList1`, conversions from a
//! proprietary raw usually do not, and no CR2 or scanner TIFF ever does.
//! Rather than guess, point this at the library and see.
fn main() {
let files: Vec<String> = std::env::args().skip(1).collect();
if files.is_empty() {
eprintln!("usage: defects <raw file>...");
std::process::exit(2);
}
for path in &files {
let bytes = match std::fs::read(path) {
Ok(b) => b,
Err(e) => {
println!("{path}: unreadable — {e}");
continue;
}
};
let found = dr_decode::defects(&bytes);
if found.is_empty() {
println!("{path}: no defect map");
continue;
}
println!(
"{path}: {} bad pixel(s), {} bad line(s)",
found.pixels.len(),
found.lines.len()
);
// A handful, so the output stays readable on a sensor reporting
// hundreds — the count above is the number that matters.
for p in found.pixels.iter().take(8) {
println!(" pixel at {},{}", p.x, p.y);
}
for l in found.lines.iter().take(8) {
match l {
dr_decode::BadLine::Column(x) => println!(" dead column {x}"),
dr_decode::BadLine::Row(y) => println!(" dead row {y}"),
}
}
}
}
-19
View File
@@ -1,19 +0,0 @@
fn main() {
for p in std::env::args().skip(1) {
let Ok(d) = std::fs::read(&p) else { continue };
let n = p.rsplit('/').next().unwrap();
// Exactly what the sweep sees: the first HEADER_BYTES only.
let head = &d[..d.len().min(dr_decode::HEADER_BYTES as usize)];
match dr_decode::metadata(head) {
Ok(m) => println!(
"{n}: header-only at={:?} model={:?}",
m.captured_at, m.model
),
Err(e) => println!("{n}: header-only ERROR {e}"),
}
match dr_decode::metadata(&d) {
Ok(m) => println!("{n}: whole-file at={:?}", m.captured_at),
Err(e) => println!("{n}: whole-file ERROR {e}"),
}
}
}
-221
View File
@@ -1,221 +0,0 @@
//! TRACES: S15 | FR-MRG-3
//! Spike S15.1 — does rawler read back a linear DNG this application writes?
//!
//! cargo run -p dr-decode --example linear_dng [-- <out.dng>]
//!
//! Decides FR-MRG-3's container. A panorama composite is three linear samples
//! per pixel with a camera matrix attached, which is exactly what a
//! `LinearRaw` DNG is; if rawler parses one, the composite re-enters the
//! library as `Format::Dng` and the only new decode work is a `cpp == 3`
//! branch. If it does not, the container is a float TIFF with a decode path
//! of its own.
//!
//! The file is hand-rolled rather than written with the `tiff` crate, whose
//! encoder fixes `PhotometricInterpretation` to RGB and cannot say
//! `LinearRaw`. Eighty lines of IFD is the cheaper thing to own than a fork.
use rawler::rawsource::RawSource;
const W: u32 = 64;
const H: u32 = 48;
fn main() {
let bytes = write_linear_dng(W, H);
if let Some(path) = std::env::args().nth(1) {
std::fs::write(&path, &bytes).expect("write");
println!("wrote {path} ({} bytes)", bytes.len());
}
let source = RawSource::new_from_slice(&bytes);
let decoder = match rawler::get_decoder(&source) {
Ok(d) => d,
Err(e) => {
println!("FAIL get_decoder: {e}");
std::process::exit(1);
}
};
println!("ok decoder found");
let image = match decoder.raw_image(&source, &Default::default(), false) {
Ok(i) => i,
Err(e) => {
println!("FAIL raw_image: {e}");
std::process::exit(1);
}
};
println!(
"ok raw_image: {}×{}, cpp {}, bps {}, {} samples, make {:?} model {:?}",
image.width,
image.height,
image.cpp,
image.bps,
match &image.data {
rawler::RawImageData::Integer(v) => v.len(),
rawler::RawImageData::Float(v) => v.len(),
},
image.make,
image.model
);
println!(
" white {:?} black {:?} wb {:?}",
image.whitelevel.0,
image
.blacklevel
.levels
.iter()
.map(|r| r.n as f32 / r.d.max(1) as f32)
.collect::<Vec<_>>(),
image.wb_coeffs
);
// The pixel at (1, 0) was written as (1000, 2000, 3000): if the samples
// come back interleaved in that order, cpp == 3 means what it says.
if let rawler::RawImageData::Integer(v) = &image.data {
let i = image.cpp;
println!(" pixel (1,0) = {:?}", &v[i..i + image.cpp.min(3)]);
}
// What dr-decode itself makes of it: the colour matrix rawler parsed into
// the camera definition, and the profile the decoder would build from it.
println!(" rawler color_matrix: {:?}", image.camera.color_matrix);
let dng = dr_decode::profile::read_dng_matrices(decoder.as_ref());
let profile = dr_decode::CameraProfile::extract(&image, &dng);
println!(
" CameraProfile: {}",
profile
.as_ref()
.map(|p| format!("xyz_to_cam {:?}", p.xyz_to_cam()))
.unwrap_or_else(|| "none".into())
);
match dr_decode::decode(&bytes) {
Ok(r) => println!(
"note dr_decode::decode accepted it as CFA: {}×{}, {} samples — the cpp==3 branch is the work",
r.width,
r.height,
r.data.len()
),
Err(e) => println!("note dr_decode::decode refused it: {e} — the cpp==3 branch is the work"),
}
}
/// A minimal `LinearRaw` DNG: one IFD, uncompressed 16-bit RGB, the tags a
/// decoder needs to treat it as a DNG and the matrix a develop chain needs
/// to treat it as a camera. Little-endian, one strip.
fn write_linear_dng(w: u32, h: u32) -> Vec<u8> {
// Pixels first, so their offset is known: a ramp with one marker pixel.
let mut pixels: Vec<u16> = Vec::with_capacity((w * h * 3) as usize);
for y in 0..h {
for x in 0..w {
if (x, y) == (1, 0) {
pixels.extend([1000, 2000, 3000]);
} else {
let v = ((x + y) * 512).min(65535) as u16;
pixels.extend([v, v / 2, v / 3]);
}
}
}
let pixel_bytes: Vec<u8> = pixels.iter().flat_map(|v| v.to_le_bytes()).collect();
// Layout: header (8) | pixels | extra data | IFD.
let pixels_off = 8u32;
let extra_off = pixels_off + pixel_bytes.len() as u32;
// Values that do not fit in four bytes go in `extra`, and the entry
// points at them.
let mut extra: Vec<u8> = Vec::new();
let mut entries: Vec<(u16, u16, u32, [u8; 4])> = Vec::new();
fn short(tag: u16, v: u16) -> (u16, u16, u32, [u8; 4]) {
let mut b = [0u8; 4];
b[..2].copy_from_slice(&v.to_le_bytes());
(tag, 3, 1, b)
}
fn long(tag: u16, v: u32) -> (u16, u16, u32, [u8; 4]) {
(tag, 4, 1, v.to_le_bytes())
}
fn ascii(extra: &mut Vec<u8>, extra_off: u32, tag: u16, s: &str) -> (u16, u16, u32, [u8; 4]) {
let mut bytes = s.as_bytes().to_vec();
bytes.push(0);
let off = extra_off + extra.len() as u32;
extra.extend(&bytes);
(tag, 2, bytes.len() as u32, off.to_le_bytes())
}
entries.push(long(254, 0)); // NewSubfileType: main image
entries.push(long(256, w));
entries.push(long(257, h));
// BitsPerSample ×3 — three shorts, six bytes, so out of line.
{
let off = extra_off + extra.len() as u32;
for _ in 0..3 {
extra.extend(16u16.to_le_bytes());
}
entries.push((258, 3, 3, off.to_le_bytes()));
}
entries.push(short(259, 1)); // Compression: none
entries.push(short(262, 34892)); // PhotometricInterpretation: LinearRaw
entries.push(ascii(&mut extra, extra_off, 271, "DarkRoom"));
entries.push(ascii(&mut extra, extra_off, 272, "Panorama"));
entries.push(long(273, pixels_off)); // StripOffsets
entries.push(short(274, 1)); // Orientation
entries.push(short(277, 3)); // SamplesPerPixel
entries.push(long(278, h)); // RowsPerStrip
entries.push(long(279, pixel_bytes.len() as u32)); // StripByteCounts
entries.push(short(284, 1)); // PlanarConfiguration: chunky
entries.push((50706, 1, 4, [1, 4, 0, 0])); // DNGVersion
entries.push((50707, 1, 4, [1, 4, 0, 0])); // DNGBackwardVersion
entries.push(ascii(&mut extra, extra_off, 50708, "DarkRoom Panorama")); // UniqueCameraModel
entries.push(long(50717, 65535)); // WhiteLevel
// ColorMatrix1: XYZ → camera, 9 SRATIONALs. A plausible sRGB-ish matrix
// (the inverse of the sRGB D65 primaries), scaled to integers.
{
let m: [(i32, i32); 9] = [
(32406, 10000),
(-15372, 10000),
(-4986, 10000),
(-9689, 10000),
(18758, 10000),
(415, 10000),
(557, 10000),
(-2040, 10000),
(10570, 10000),
];
let off = extra_off + extra.len() as u32;
for (n, d) in m {
extra.extend(n.to_le_bytes());
extra.extend(d.to_le_bytes());
}
entries.push((50721, 10, 9, off.to_le_bytes()));
}
// AsShotNeutral: 3 RATIONALs, neutral.
{
let off = extra_off + extra.len() as u32;
for _ in 0..3 {
extra.extend(1u32.to_le_bytes());
extra.extend(1u32.to_le_bytes());
}
entries.push((50728, 5, 3, off.to_le_bytes()));
}
entries.push(short(50778, 21)); // CalibrationIlluminant1: D65
entries.sort_by_key(|e| e.0);
let ifd_off = extra_off + extra.len() as u32;
let mut out = Vec::new();
out.extend(b"II");
out.extend(42u16.to_le_bytes());
out.extend(ifd_off.to_le_bytes());
out.extend(&pixel_bytes);
out.extend(&extra);
out.extend((entries.len() as u16).to_le_bytes());
for (tag, ty, count, value) in &entries {
out.extend(tag.to_le_bytes());
out.extend(ty.to_le_bytes());
out.extend(count.to_le_bytes());
out.extend(value);
}
out.extend(0u32.to_le_bytes()); // no next IFD
out
}
-61
View File
@@ -1,61 +0,0 @@
//! Print what `decode` extracts from a RAW file.
//!
//! A sanity check on the pipeline's inputs: black and white levels, the CFA
//! pattern after re-phasing, as-shot white balance, and the camera→sRGB
//! matrix. Wrong values here produce a wrong image no shader can fix, so it
//! is worth being able to see them directly.
//!
//! ```sh
//! cargo run -p dr-decode --example rawinfo -- IMG.CR2
//! ```
fn main() {
let Some(path) = std::env::args().nth(1) else {
eprintln!("usage: rawinfo <file.cr2>");
std::process::exit(2);
};
let bytes = std::fs::read(&path).expect("read file");
let raw = dr_decode::decode(&bytes).expect("decode");
println!("file {path}");
println!("readout {} × {}", raw.width, raw.height);
println!(
"crop {} × {} at ({}, {})",
raw.crop.width, raw.crop.height, raw.crop.x, raw.crop.y
);
let (dx, dy) = raw.crop.shifts_cfa_phase();
println!(
"cfa {:?} (rephased: {dx}, {dy})",
raw.cfa_pattern
);
println!("black {:?}", raw.black_level);
println!("white {}", raw.white_level);
println!("wb_coeffs {:?}", raw.wb_coeffs);
match raw.color_matrix {
Some(m) => {
println!("cam→srgb");
for row in m.chunks(3) {
println!(
" [{:>8.4} {:>8.4} {:>8.4}]",
row[0], row[1], row[2]
);
}
// Each row should sum to roughly 1: a neutral camera-space colour
// must stay neutral in sRGB. Far from 1 means the normalisation
// or the matrix composition is wrong.
let sums: Vec<f32> = m.chunks(3).map(|r| r.iter().sum()).collect();
println!("row sums {sums:.4?} (≈1.0 each if correct)");
}
None => println!("cam→srgb none — uncalibrated body"),
}
// Sample the actual data range, which reveals a black-level or bit-depth
// mistake faster than any amount of staring at metadata.
let (min, max) = raw
.data
.iter()
.fold((u16::MAX, 0u16), |(lo, hi), &v| (lo.min(v), hi.max(v)));
println!("sample range {min} … {max}");
}
-146
View File
@@ -1,146 +0,0 @@
//! Smoke test against real RAW files.
//!
//! cargo run -p dr-decode --example smoke -- <file-or-dir>...
//!
//! Reports, per file, what each entry point costs — which is the whole reason
//! they are separate (ARCH §3.2).
use std::path::{Path, PathBuf};
use std::time::Instant;
fn main() {
env_logger::init();
let args: Vec<String> = std::env::args().skip(1).collect();
if args.is_empty() {
eprintln!("usage: smoke <file-or-dir>...");
std::process::exit(2);
}
let mut files = Vec::new();
for a in &args {
let p = PathBuf::from(a);
if p.is_dir() {
collect(&p, &mut files);
} else {
files.push(p);
}
}
files.sort();
files.truncate(8);
println!(
"{:<20} {:>7} {:>8} {:>9} {:>13} {:>9} {:>13}",
"file", "size", "meta", "thumb", "thumb dims", "full", "full dims"
);
println!("{}", "-".repeat(88));
let (mut ok, mut failed) = (0, 0);
for f in &files {
match run_one(f) {
Ok(line) => {
println!("{line}");
ok += 1;
}
Err(e) => {
println!("{:<22} {e}", truncate(&name(f), 22));
failed += 1;
}
}
}
println!("\n{ok} ok, {failed} failed");
if failed > 0 {
std::process::exit(1);
}
}
fn run_one(path: &Path) -> Result<String, String> {
let size = std::fs::metadata(path).map_err(|e| e.to_string())?.len();
// The culling path: read only the header region, not the whole file.
let probe_bytes =
read_prefix(path, dr_decode::PREVIEW_PROBE_BYTES).map_err(|e| e.to_string())?;
let t0 = Instant::now();
let fmt = dr_decode::probe(&probe_bytes);
let meta = dr_decode::metadata(&probe_bytes).ok();
let meta_ms = t0.elapsed().as_secs_f64() * 1000.0;
let all = std::fs::read(path).map_err(|e| e.to_string())?;
// The culling rung.
let t1 = Instant::now();
let thumb = dr_decode::extract_preview(&all, dr_decode::PreviewSize::Thumbnail)
.map_err(|e| format!("thumb: {e}"))?;
let thumb_ms = t1.elapsed().as_secs_f64() * 1000.0;
// The full-resolution rung, for comparison.
let t2 = Instant::now();
let full = dr_decode::extract_preview(&all, dr_decode::PreviewSize::Full)
.map_err(|e| format!("full: {e}"))?;
let full_ms = t2.elapsed().as_secs_f64() * 1000.0;
let model = meta
.as_ref()
.and_then(|m| m.model.clone())
.unwrap_or_else(|| "?".into());
let budget = if thumb_ms <= 50.0 {
""
} else {
" OVER BUDGET"
};
Ok(format!(
"{:<20} {:>6.1}M {:>6.1}ms {:>7.1}ms {:>7}x{:<5} {:>7.1}ms {:>7}x{:<5} {:?} {}{}",
truncate(&name(path), 20),
size as f64 / 1e6,
meta_ms,
thumb_ms,
thumb.width,
thumb.height,
full_ms,
full.width,
full.height,
fmt,
model.trim(),
budget,
))
}
fn read_prefix(path: &Path, n: u64) -> std::io::Result<Vec<u8>> {
use std::io::Read;
let mut f = std::fs::File::open(path)?;
let mut buf = vec![0u8; n as usize];
let read = f.read(&mut buf)?;
buf.truncate(read);
Ok(buf)
}
fn collect(dir: &Path, out: &mut Vec<PathBuf>) {
let Ok(entries) = std::fs::read_dir(dir) else {
return;
};
for e in entries.flatten() {
let p = e.path();
if p.is_file() {
let ext = p
.extension()
.map(|s| s.to_string_lossy().to_ascii_lowercase())
.unwrap_or_default();
if dr_types::Format::from_extension(&ext).is_some() {
out.push(p);
}
}
}
}
fn name(p: &Path) -> String {
p.file_name().unwrap_or_default().to_string_lossy().into()
}
fn truncate(s: &str, n: usize) -> String {
if s.len() <= n {
s.to_string()
} else {
format!("{}…", &s[..n - 1])
}
}
-160
View File
@@ -1,160 +0,0 @@
# DarkRoom camera base curves (FR-DEV-3e).
#
# ---------------------------------------------------------------------------
# Adding a body is editing this file. It is not a code change.
# ---------------------------------------------------------------------------
#
# The copy you are reading is compiled into the binary as a floor. At startup
# `dr_decode::base_curve::load` also looks for `base_curves.yaml` in:
#
# 1. $DARKROOM_PROFILES/ (set it while you are tuning)
# 2. $XDG_DATA_HOME/darkroom/profiles/
# or $HOME/.local/share/darkroom/profiles/
#
# and uses the first one it finds *whose `version:` is higher than this one's*.
# So: bump `version`, drop the file in that directory, restart. A body added
# this afternoon renders correctly this afternoon, with no release and no
# rebuild — which is what the requirement asks for, and what makes these
# contributable under the GPL.
#
# The version check runs both ways on purpose. A file older than the built-in
# copy is ignored with a log line, so upgrading DarkRoom cannot silently lose
# curves to a pack somebody downloaded a year ago.
#
# ---------------------------------------------------------------------------
# What the numbers mean
# ---------------------------------------------------------------------------
#
# Five `[x, y]` control points on a monotone spline (Fritsch-Carlson, the same
# one the tone curve widget draws). Both axes are **linear**:
#
# x scene-referred camera RGB after white balance, 1.0 = sensor saturation
# y display-referred linear; the sRGB transfer function is applied later,
# at the end of the shader, so do not pre-apply a gamma here
#
# The identity is y = x, and it is what an unrecognised body gets if `default:`
# is removed. It is also the wrong answer for almost every photograph: linear
# scene data has middle grey at about 13% and a camera JPEG puts it near 18%,
# so an uncurved render is roughly half a stop dark through the midtones and
# has no highlight rolloff at all.
#
# A curve that works has three parts, and it is worth naming them because they
# are what you are actually tuning:
#
# the toe the first span, slope near or below 1. Deep shadows stay
# deep. Lift it and blacks go milky; crush it and shadow
# detail the sensor recorded disappears.
# the midtones the middle spans, slope well above 1. This is the contrast
# and the brightness people read as "the camera's look".
# the shoulder the last span, slope well below 1. Highlights compress
# toward white instead of arriving there and clipping. It is
# the difference between a rolled-off sky and a white hole.
#
# Two invariants are enforced in code and tested, so a mistake here fails the
# build rather than the photograph: x must strictly increase, y must not
# decrease, and everything must lie inside the unit square.
#
# ---------------------------------------------------------------------------
# Honesty about these values
# ---------------------------------------------------------------------------
#
# These are hand-tuned shapes, not measurements. They encode what every camera
# JPEG rendering has in common — the toe/midtone/shoulder structure above —
# plus each maker's well-known house differences: Canon's gentler shoulder and
# warmer-reading midtones, Nikon's slightly higher midtone contrast, Sony's
# flatter and more conservative default, Fujifilm's markedly contrastier
# Provia-derived rendering.
#
# FR-DEV-3e's acceptance criterion is subjective comparison against each body's
# own JPEG, and meeting it properly needs a frame from that body in front of
# you. Where that has not been done, the entry is still much closer to right
# than the identity — which is the bar these have to clear, and do.
version: 1
# The rendering for a body with no entry of its own.
#
# **Deliberately not the identity.** The failure this requirement exists to fix
# is the flat render, and a conservative curve is far closer to right for every
# body than no curve is for any of them. It is gentler than the per-body
# entries below — a shallower midtone and an earlier, softer shoulder — because
# it has to be safe on a sensor nobody has looked at, and the cost of being too
# tame is a photograph that wants a little contrast rather than one that has
# lost its highlights.
default:
points:
- [0.00, 0.000]
- [0.04, 0.043]
- [0.13, 0.175]
- [0.45, 0.690]
- [1.00, 1.000]
bodies:
# Canon. A soft toe and a long, gradual shoulder — the reason Canon files
# are described as forgiving in highlights and a little low in contrast
# straight out of camera.
- make: Canon
model: EOS 6D
points:
- [0.00, 0.000]
- [0.04, 0.045]
- [0.13, 0.190]
- [0.45, 0.720]
- [1.00, 1.000]
- make: Canon
model: EOS R6
points:
- [0.00, 0.000]
- [0.04, 0.044]
- [0.13, 0.195]
- [0.45, 0.730]
- [1.00, 1.000]
# Nikon. A slightly deeper toe and more midtone slope than Canon, which is
# the "punchier out of camera" difference people describe between the two.
- make: Nikon
model: Z 6
points:
- [0.00, 0.000]
- [0.04, 0.038]
- [0.13, 0.200]
- [0.46, 0.750]
- [1.00, 1.000]
- make: Nikon
model: D750
points:
- [0.00, 0.000]
- [0.04, 0.039]
- [0.13, 0.198]
- [0.46, 0.745]
- [1.00, 1.000]
# Sony. The flattest default of the four, and intentionally so — Sony's own
# rendering leaves more headroom than it uses, which is why Sony files are
# the ones people describe as needing the most work.
- make: Sony
model: ILCE-7M3
points:
- [0.00, 0.000]
- [0.04, 0.048]
- [0.13, 0.185]
- [0.44, 0.700]
- [1.00, 1.000]
# Fujifilm. Provia, the default film simulation: a firm toe, the steepest
# midtones here, and a hard shoulder. It is the most distinctive rendering of
# the four and the one where a flat render looks most obviously wrong.
#
# This entry does *not* read the in-RAF film simulation tag — that is
# FR-DEV-3f, and until it lands every Fujifilm file gets the Provia shape
# whatever the camera was set to.
- make: Fujifilm
model: X-T3
points:
- [0.00, 0.000]
- [0.045, 0.040]
- [0.14, 0.215]
- [0.47, 0.775]
- [1.00, 1.000]
-752
View File
@@ -1,752 +0,0 @@
//! TRACES: FR-DEV-3e
//! Base curves — the per-body rendering that turns a correct exposure into a
//! photograph.
//!
//! # What this is for
//!
//! A camera matrix gets the *colours* right and leaves the picture flat. Sensor
//! data is scene-referred and very nearly linear; a print, a screen and a
//! camera's own JPEG are none of those things. Rendering linear data straight
//! out is the dcraw default, and FR-DEV-3e names it precisely: "the flat,
//! poor-skin-tone rendering characteristic of dcraw defaults, which is the
//! documented reason people abandon darktable in the first hour."
//!
//! The fix is a tone curve applied as part of *reading* the file rather than as
//! an edit — a toe, a steep midtone, and a shoulder that rolls highlights off
//! instead of clipping them. Every raw converter has one. Adobe calls it the
//! camera profile's tone curve, darktable calls it the base curve, and the name
//! here follows darktable's because the placement does too: it runs in camera
//! RGB, after white balance and the user's adjustments, immediately before the
//! conversion out to a working space.
//!
//! # Why it is not an edit
//!
//! It never reaches the sidecar and there is no slider for it, for the same
//! reason the EXIF orientation is not an edit (FR-DEV-3h): it is a property of
//! the body that took the frame, not of what anyone decided about the frame.
//! Sidecars are shared between devices and bodies (FR-NC-9), and one camera's
//! rendering must not follow an edit onto another camera's file.
//!
//! # Why it is data
//!
//! FR-DEV-3e requires the profile database to be "versioned independently of
//! the app binary so bodies and curves can be added without a release — and,
//! under D8's GPLv3, contributed by users". So the curves live in
//! `profiles/base_curves.yaml`, a file that is compiled in as a floor and
//! *overridden* by a copy on disk carrying a higher `version:`. Adding a body
//! is adding ten numbers to a YAML file; shipping that body to users is
//! publishing the file. Neither is a code change and neither needs a release.
//!
//! See [`load`] for the search path and [`Curves::body`] for the matching.
use std::path::{Path, PathBuf};
use std::sync::OnceLock;
/// How many control points a base curve has.
///
/// Five, which is not a coincidence: it is what the tone curve widget uses
/// (`dr_pipeline::ops::curve::POINTS`), so the shader evaluates a profile's
/// curve and a photographer's curve through exactly the same spline. A profile
/// author and a photographer dragging a point mean the same thing by it, and
/// the generated shader carries one implementation rather than two that could
/// disagree.
pub const POINTS: usize = 5;
/// TRACES: FR-DEV-3e
/// A base curve: five points on a monotone spline through the unit square.
///
/// `xs` is scene-linear camera RGB, normalised so that 1.0 is the sensor's
/// saturation point. `ys` is display-referred linear — *not* gamma-encoded,
/// because the sRGB transfer function is applied at the very end of the
/// generated shader and applying it twice would wash the image out.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct BaseCurve {
pub xs: [f32; POINTS],
pub ys: [f32; POINTS],
}
impl BaseCurve {
/// The curve that does nothing — the identity diagonal.
///
/// What an unrecognised body gets if the database carries no default, and
/// what a JPEG gets always: an already-rendered image must not be rendered
/// a second time.
pub const IDENTITY: Self = Self {
xs: [0.0, 0.25, 0.5, 0.75, 1.0],
ys: [0.0, 0.25, 0.5, 0.75, 1.0],
};
/// Whether this curve would leave the image alone.
///
/// The shader is told to skip the stage entirely when it would, so an
/// unprofiled body costs a branch that is uniform across the dispatch
/// rather than a spline evaluation per channel per pixel.
pub fn is_identity(&self) -> bool {
self.xs
.iter()
.zip(self.ys.iter())
.all(|(x, y)| (x - y).abs() < 1e-6)
}
/// Build from raw pairs, rejecting anything that is not a curve.
///
/// A profile file is data a user may have edited, so this is the boundary
/// where "ten numbers" becomes "a curve": the x coordinates must increase,
/// the y coordinates must not decrease, and both must lie in the unit
/// square. A non-monotone x sends the spline's span search backwards and
/// divides by a negative width; a decreasing y inverts tones locally,
/// which reads as a dark halo through smooth gradients rather than as a
/// bad profile.
///
/// Endpoints are not forced to (0,0) and (1,1). A curve that lifts black
/// slightly, or that places the shoulder below white, is a legitimate
/// rendering choice and several bodies make it.
pub fn from_points(points: &[[f32; 2]]) -> Option<Self> {
if points.len() != POINTS {
return None;
}
let mut xs = [0.0f32; POINTS];
let mut ys = [0.0f32; POINTS];
for (i, p) in points.iter().enumerate() {
if !p[0].is_finite() || !p[1].is_finite() {
return None;
}
if !(0.0..=1.0).contains(&p[0]) || !(0.0..=1.0).contains(&p[1]) {
return None;
}
xs[i] = p[0];
ys[i] = p[1];
}
for i in 1..POINTS {
// Strictly increasing in x — the spline divides by the span width.
if xs[i] <= xs[i - 1] {
return None;
}
// Non-decreasing in y. Flat is allowed: a curve that holds a
// highlight range at white is clipping deliberately.
if ys[i] < ys[i - 1] {
return None;
}
}
Some(Self { xs, ys })
}
}
/// One body's entry in the database.
#[derive(Debug, Clone, PartialEq)]
pub struct BodyCurve {
/// The manufacturer, as the file writes it — "Canon", "NIKON CORPORATION".
pub make: String,
/// The model, as the file writes it — "EOS 6D", "ILCE-7M3".
pub model: String,
pub curve: BaseCurve,
}
/// TRACES: FR-DEV-3e
/// The base curve database.
///
/// Versioned as a whole rather than per body, because that is the unit a user
/// downloads and the unit that has to beat the built-in copy. See [`load`].
#[derive(Debug, Clone, PartialEq)]
pub struct Curves {
version: u32,
default: Option<BaseCurve>,
bodies: Vec<BodyCurve>,
}
impl Curves {
/// TRACES: FR-DEV-3e
/// The curve to render a frame from this body with.
///
/// Falls back, in order, to the database's `default:` and then to the
/// identity. **The default is deliberately not the identity**: an
/// unrecognised body rendered flat is the failure this requirement exists
/// to prevent, and a gentle, conservative curve is much closer to right for
/// every body than no curve is for any of them. A body with its own entry
/// gets that instead.
///
/// # What "this body" has to survive
///
/// The same camera names itself three ways depending on which program last
/// touched the file. A native NEF says make "NIKON CORPORATION", model
/// "NIKON Z 6"; rawler's own database cleans that to "Nikon" and "Z 6"; an
/// Adobe-converted DNG keeps the uncleaned pair. A database that had to
/// spell every variant would go stale the first time a maker changed its
/// mind about its own name, so the matching does the folding instead:
///
/// - Case, punctuation and runs of whitespace are flattened, so
/// "ILCE-7M3", "ILCE 7M3" and "ilce-7m3" are one body.
/// - The make is compared on its **first word only**. Every maker's
/// trailing corporate boilerplate — "CORPORATION", "IMAGING CORP" — is
/// noise, and no two camera manufacturers share a first word.
/// - The model is tried both as written and with a leading copy of the
/// make removed, which is what lets one "Canon"/"EOS 6D" entry cover
/// "Canon EOS 6D" as well.
pub fn body(&self, make: &str, model: &str) -> BaseCurve {
let (make, model) = (make_key(make), normalise(model));
// The model with a leading copy of the maker's name removed.
let bare = model.strip_prefix(&format!("{make} ")).unwrap_or(&model);
self.bodies
.iter()
.find(|b| {
let entry_model = normalise(&b.model);
make_key(&b.make) == make && (entry_model == model || entry_model == bare)
})
.map(|b| b.curve)
.or(self.default)
.unwrap_or(BaseCurve::IDENTITY)
}
/// The database version. Higher wins; see [`load`].
pub fn version(&self) -> u32 {
self.version
}
/// How many bodies have their own curve, excluding the default.
pub fn len(&self) -> usize {
self.bodies.len()
}
pub fn is_empty(&self) -> bool {
self.bodies.is_empty()
}
/// Parse a database from YAML.
///
/// Entries that are not curves are dropped with a warning rather than
/// failing the parse. A user-contributed file with one bad body should
/// cost that body's rendering, not every body's — and the alternative is an
/// application that will not open a photograph because somebody typed a
/// comma.
pub fn parse(yaml: &str) -> Result<Self, String> {
let file: File = serde_norway::from_str(yaml).map_err(|e| e.to_string())?;
let default = file.default.and_then(|d| {
BaseCurve::from_points(&d.points).or_else(|| {
log::warn!("base curves: the default entry is not a monotone curve; ignoring it");
None
})
});
let bodies = file
.bodies
.into_iter()
.filter_map(|b| match BaseCurve::from_points(&b.points) {
Some(curve) => Some(BodyCurve {
make: b.make,
model: b.model,
curve,
}),
None => {
log::warn!(
"base curves: {} {} is not a monotone curve; ignoring it",
b.make,
b.model
);
None
}
})
.collect();
Ok(Self {
version: file.version,
default,
bodies,
})
}
}
/// The copy that ships inside the binary.
///
/// A floor, not the answer: [`load`] prefers a newer file on disk. Compiled in
/// so that a fresh install with no profile directory — and every Android build,
/// where there is no such directory to speak of — still renders properly.
const BUILT_IN: &str = include_str!("../profiles/base_curves.yaml");
/// TRACES: FR-DEV-3e
/// The base curve database, loaded once.
///
/// # The search path, and why it is a version comparison
///
/// 1. `$DARKROOM_PROFILES`, a directory, when set. The escape hatch: a profile
/// author iterating on a curve points this at their working copy and does
/// not have to install anything.
/// 2. `$XDG_DATA_HOME/darkroom/profiles/`, else `$HOME/.local/share/darkroom/profiles/`.
/// The same base directory the catalog uses, chosen there for the same
/// reason — it is data, not cache, and must survive a storage sweep.
/// 3. The copy compiled into the binary.
///
/// The first file that parses *and carries a higher `version:` than the
/// built-in copy* wins. The version check is the whole mechanism the
/// requirement asks for, and it runs in both directions:
///
/// - A downloaded pack at version 7 supersedes a binary shipping version 3, so
/// a body added after the release renders correctly with no release.
/// - A stale pack at version 2 does **not** supersede a binary shipping version
/// 3, so upgrading the application cannot silently lose curves to a file
/// somebody downloaded a year ago and forgot.
///
/// Failures are warnings, never errors. A malformed profile file must cost the
/// user their curves, not their photographs.
pub fn load() -> &'static Curves {
static LOADED: OnceLock<Curves> = OnceLock::new();
LOADED.get_or_init(|| {
let built_in = Curves::parse(BUILT_IN).unwrap_or_else(|e| {
// Unreachable in a build that ran its tests — `the_shipped_database_parses`
// asserts exactly this — but a panic here would mean an
// application that cannot open a photograph because of a typo in a
// data file, which is never the right trade.
log::error!("base curves: the built-in database does not parse: {e}");
Curves {
version: 0,
default: None,
bodies: Vec::new(),
}
});
choose(built_in, &search_path())
})
}
/// The version comparison, separated from where the directories come from.
///
/// Split out so it can be tested against real files in a real directory
/// without the process-wide `OnceLock` and the environment `load` reads. The
/// rule this implements is the whole of what FR-DEV-3e asks for, so it is
/// worth being able to state it as a test rather than as a comment.
fn choose(built_in: Curves, dirs: &[PathBuf]) -> Curves {
for dir in dirs {
let path = dir.join("base_curves.yaml");
let Ok(text) = std::fs::read_to_string(&path) else {
continue;
};
match Curves::parse(&text) {
Ok(external) if external.version > built_in.version => {
log::info!(
"base curves: using {} (version {}, {} bodies) over the built-in version {}",
path.display(),
external.version,
external.len(),
built_in.version
);
return external;
}
Ok(external) => log::info!(
"base curves: ignoring {} at version {}; the built-in database is version {}",
path.display(),
external.version,
built_in.version
),
Err(e) => log::warn!("base curves: {} does not parse: {e}", path.display()),
}
}
built_in
}
/// TRACES: FR-DEV-3e
/// The curve for a body, from the loaded database.
///
/// The one call site the decoder needs; everything above is reachable for
/// tests and for a future profile editor.
pub fn for_body(make: &str, model: &str) -> BaseCurve {
load().body(make, model)
}
/// Directories that may hold a `base_curves.yaml`, most specific first.
fn search_path() -> Vec<PathBuf> {
let mut dirs = Vec::new();
if let Some(explicit) = std::env::var_os("DARKROOM_PROFILES") {
dirs.push(PathBuf::from(explicit));
}
// The same resolution `dr_ui::library::catalog_path` uses, and for the
// same reason: this is data a user may have installed, not a cache. It is
// duplicated rather than shared because `dr-decode` sits far below the UI
// and must not acquire a dependency on it to find a directory.
let base = std::env::var_os("XDG_DATA_HOME")
.map(PathBuf::from)
.or_else(|| std::env::var_os("HOME").map(|h| Path::new(&h).join(".local/share")));
if let Some(base) = base {
dirs.push(base.join("darkroom").join("profiles"));
}
dirs
}
/// A manufacturer's first word, folded.
///
/// "NIKON CORPORATION", "Nikon" and "nikon" all become `NIKON`. The corporate
/// suffixes are not information — they appear or not depending on whether the
/// file went through a DNG converter — and no two camera manufacturers share a
/// first word, so nothing is lost by dropping them.
fn make_key(s: &str) -> String {
normalise(s)
.split(' ')
.next()
.unwrap_or_default()
.to_string()
}
/// Fold a make or model into something two files can agree on.
///
/// Upper-cased, with every run of non-alphanumeric characters collapsed to one
/// space and the ends trimmed, so that "ILCE-7M3", "ILCE 7M3" and "ilce-7m3"
/// become one.
fn normalise(s: &str) -> String {
let mut out = String::with_capacity(s.len());
let mut pending_space = false;
for c in s.chars() {
if c.is_ascii_alphanumeric() {
if pending_space && !out.is_empty() {
out.push(' ');
}
pending_space = false;
out.push(c.to_ascii_uppercase());
} else {
pending_space = true;
}
}
out
}
// ---- The on-disk shape, kept apart from the in-memory one ----------------
//
// Deliberately separate types. The file is data a user edits and is allowed to
// be wrong; `Curves` is a parsed database whose every entry is known to be a
// monotone curve. Deriving `Deserialize` on `BaseCurve` directly would delete
// that boundary and let an unchecked five-point array reach the shader.
//
// Unknown fields are **accepted**, which is not laziness. The database is
// versioned independently of the binary and moves in both directions: a pack
// published after this release may carry keys this build has never heard of —
// a hue twist, a look table (FR-DEV-3f) — and it must still deliver its curves
// to an older DarkRoom rather than failing to parse and leaving every body
// flat. `deny_unknown_fields` would trade that for a diagnostic nobody needs.
#[derive(serde::Deserialize)]
struct File {
version: u32,
#[serde(default)]
default: Option<Entry>,
#[serde(default)]
bodies: Vec<BodyEntry>,
}
#[derive(serde::Deserialize)]
struct Entry {
points: Vec<[f32; 2]>,
}
#[derive(serde::Deserialize)]
struct BodyEntry {
make: String,
model: String,
points: Vec<[f32; 2]>,
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn the_shipped_database_parses_and_carries_a_default() {
// The one test that must never be allowed to fail quietly: `load`
// degrades to an empty database rather than panicking, so without this
// a typo in the YAML would ship as "every photograph renders flat"
// rather than as a build failure.
let curves = Curves::parse(BUILT_IN).expect("the shipped database parses");
assert!(curves.version() >= 1);
assert!(!curves.is_empty(), "the database ships bodies");
assert!(
!curves.body("Nobody", "Nothing").is_identity(),
"an unknown body must still get the default rendering"
);
}
#[test]
fn every_shipped_curve_lifts_the_midtones_and_rolls_the_highlights() {
// What makes a base curve a base curve rather than a decoration. If a
// shipped curve failed either half it would be a worse rendering than
// the flat one it replaced, which is the one outcome forbidden.
let curves = Curves::parse(BUILT_IN).expect("parses");
let all = curves
.bodies
.iter()
.map(|b| (format!("{} {}", b.make, b.model), b.curve))
.chain(curves.default.map(|c| ("default".to_string(), c)));
for (name, curve) in all {
// The midtone point sits above the diagonal: a linear midtone is
// roughly a stop and a half darker than any camera renders it.
let mid = 2;
assert!(
curve.ys[mid] > curve.xs[mid],
"{name} does not lift its midtones ({} -> {})",
curve.xs[mid],
curve.ys[mid]
);
// And the last span is shallower than the one before it, which is
// what a shoulder *is*. Without one the curve clips highlights
// harder than the linear rendering did.
let slope =
|i: usize| (curve.ys[i + 1] - curve.ys[i]) / (curve.xs[i + 1] - curve.xs[i]);
assert!(
slope(POINTS - 2) < slope(POINTS - 3),
"{name} has no highlight shoulder"
);
}
}
#[test]
fn a_curve_that_is_not_monotone_is_refused() {
// The profile file is user-editable, so this is a real boundary and
// not a formality. A decreasing y inverts tones locally and shows up
// as a dark halo in a gradient, which reads as a rendering fault
// rather than as a bad profile.
assert_eq!(
BaseCurve::from_points(&[[0.0, 0.0], [0.25, 0.4], [0.5, 0.3], [0.75, 0.8], [1.0, 1.0]]),
None
);
}
#[test]
fn a_curve_whose_x_does_not_advance_is_refused() {
// The spline divides by the span width; a repeated x is a division by
// zero in the shader, which is a NaN pixel rather than an error.
assert_eq!(
BaseCurve::from_points(&[
[0.0, 0.0],
[0.25, 0.3],
[0.25, 0.5],
[0.75, 0.8],
[1.0, 1.0]
]),
None
);
}
#[test]
fn a_curve_of_the_wrong_length_is_refused() {
assert_eq!(BaseCurve::from_points(&[[0.0, 0.0], [1.0, 1.0]]), None);
}
#[test]
fn values_outside_the_unit_square_are_refused() {
// The shader clamps its output at the very end anyway, but a control
// point above 1.0 would put the shoulder outside the range the curve
// is defined over and silently flatten everything below it.
assert_eq!(
BaseCurve::from_points(&[[0.0, 0.0], [0.25, 0.3], [0.5, 1.4], [0.75, 1.5], [1.0, 1.6]]),
None
);
}
#[test]
fn a_body_with_its_own_entry_beats_the_default() {
let curves = Curves::parse(
"version: 2
default:
points: [[0.0, 0.0], [0.25, 0.3], [0.5, 0.6], [0.75, 0.85], [1.0, 1.0]]
bodies:
- make: Canon
model: EOS 6D
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
",
)
.expect("parses");
assert_eq!(curves.body("Canon", "EOS 6D").ys[1], 0.35);
assert_eq!(curves.body("Canon", "EOS 5D").ys[1], 0.30);
}
#[test]
fn the_make_may_be_repeated_in_the_model() {
// Canon writes "Canon" as the make and "Canon EOS 6D" as the model;
// rawler's cleaned strings drop the repetition and both reach here.
// One entry has to cover both or half the files on a card miss.
let curves = Curves::parse(
"version: 1
bodies:
- make: Canon
model: EOS 6D
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
",
)
.expect("parses");
assert_eq!(curves.body("Canon", "Canon EOS 6D").ys[1], 0.35);
assert_eq!(curves.body("Canon", "EOS 6D").ys[1], 0.35);
assert_eq!(curves.body("CANON", "eos 6d").ys[1], 0.35);
}
#[test]
fn a_corporate_suffix_does_not_hide_a_body() {
// The same Z 6 arrives as "Nikon"/"Z 6" from rawler's camera database
// and as "NIKON CORPORATION"/"NIKON Z 6" from a DNG converted out of
// the same file. Both must find the entry, or converting a file to
// DNG would silently change how it renders.
let curves = Curves::parse(
"version: 1
bodies:
- make: Nikon
model: Z 6
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
",
)
.expect("parses");
assert_eq!(curves.body("Nikon", "Z 6").ys[1], 0.35);
assert_eq!(curves.body("NIKON CORPORATION", "NIKON Z 6").ys[1], 0.35);
}
#[test]
fn punctuation_and_spacing_do_not_decide_whether_a_body_is_known() {
let curves = Curves::parse(
"version: 1
bodies:
- make: Sony
model: ILCE-7M3
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
",
)
.expect("parses");
assert_eq!(curves.body("SONY", "ILCE 7M3").ys[1], 0.35);
assert_eq!(curves.body("sony", "ilce-7m3").ys[1], 0.35);
}
#[test]
fn one_bad_entry_does_not_cost_the_rest() {
// A user-contributed file with one typo should cost that body's
// rendering, not every body's.
let curves = Curves::parse(
"version: 1
bodies:
- make: Broken
model: Body
points: [[0.0, 0.0], [0.25, 0.9], [0.5, 0.1], [0.75, 0.9], [1.0, 1.0]]
- make: Canon
model: EOS 6D
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
",
)
.expect("parses");
assert_eq!(curves.len(), 1);
assert_eq!(curves.body("Canon", "EOS 6D").ys[1], 0.35);
assert!(curves.body("Broken", "Body").is_identity());
}
#[test]
fn a_pack_from_the_future_still_delivers_its_curves() {
// The database is versioned independently of the binary, so a pack
// published after this build may carry keys this build has never heard
// of. It must still hand over the curves it does understand — failing
// the parse would leave every body flat, which is the exact failure
// FR-DEV-3e exists to prevent, delivered by the mechanism meant to
// prevent it.
let curves = Curves::parse(
"version: 9
look_table: ambitious
bodies:
- make: Canon
model: EOS 6D
hue_twist: [1, 2, 3]
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
",
)
.expect("an unfamiliar key must not fail the parse");
assert_eq!(curves.version(), 9);
assert_eq!(curves.body("Canon", "EOS 6D").ys[1], 0.35);
}
#[test]
fn an_unknown_body_with_no_default_gets_the_identity() {
// Graceful fallback, stated as a property: never worse than a flat
// render, and never a curve tuned for somebody else's sensor when the
// database declines to offer one.
let curves = Curves::parse("version: 1\nbodies: []\n").expect("parses");
assert!(curves.body("Nobody", "Nothing").is_identity());
}
/// A directory holding one `base_curves.yaml`, unique to the caller.
fn a_pack_dir(name: &str, yaml: &str) -> PathBuf {
let dir = std::env::temp_dir().join(format!("darkroom-base-curves-{name}"));
let _ = std::fs::remove_dir_all(&dir);
std::fs::create_dir_all(&dir).expect("a writable temp directory");
std::fs::write(dir.join("base_curves.yaml"), yaml).expect("write");
dir
}
const A_CANON_ENTRY: &str = "bodies:
- make: Canon
model: EOS 6D
points: [[0.0, 0.0], [0.25, 0.42], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
";
#[test]
fn a_newer_pack_on_disk_supersedes_the_built_in_database() {
// **This is the requirement.** FR-DEV-3e asks for a profile database
// versioned independently of the app binary "so bodies and curves can
// be added without a release". A file with a higher version, dropped
// in the profile directory, is what that means in practice.
let built_in = Curves::parse(BUILT_IN).expect("parses");
let newer = format!("version: {}\n{A_CANON_ENTRY}", built_in.version() + 1);
let dir = a_pack_dir("newer", &newer);
let chosen = choose(built_in.clone(), &[dir]);
assert_eq!(chosen.version(), built_in.version() + 1);
assert_eq!(chosen.body("Canon", "EOS 6D").ys[1], 0.42);
}
#[test]
fn a_stale_pack_does_not_survive_an_upgrade() {
// The other direction, and the one that protects the user. Somebody
// downloads a pack, a release later ships better curves for the same
// bodies, and the forgotten file must not quietly hold the application
// back at last year's rendering.
let built_in = Curves::parse(BUILT_IN).expect("parses");
let stale = format!("version: {}\n{A_CANON_ENTRY}", built_in.version());
let dir = a_pack_dir("stale", &stale);
let chosen = choose(built_in.clone(), &[dir]);
assert_eq!(chosen.version(), built_in.version());
assert_ne!(
chosen.body("Canon", "EOS 6D").ys[1],
0.42,
"an equal version must not displace the built-in database"
);
}
#[test]
fn a_broken_pack_costs_the_curves_and_not_the_photographs() {
// A malformed profile file must degrade to the built-in database, not
// to an error. The user came here to look at a photograph.
let built_in = Curves::parse(BUILT_IN).expect("parses");
let dir = a_pack_dir("broken", "version: [this is not a number\n");
let chosen = choose(built_in.clone(), &[dir]);
assert_eq!(chosen.version(), built_in.version());
assert_eq!(chosen.len(), built_in.len());
}
#[test]
fn a_directory_with_no_pack_in_it_is_simply_skipped() {
// The ordinary case on every machine: the search path exists, the file
// does not. It must not be a warning, an error, or a slow path.
let built_in = Curves::parse(BUILT_IN).expect("parses");
let missing = std::env::temp_dir().join("darkroom-base-curves-nothing-here");
let _ = std::fs::remove_dir_all(&missing);
assert_eq!(choose(built_in.clone(), &[missing]), built_in);
}
#[test]
fn the_identity_is_recognised_as_doing_nothing() {
assert!(BaseCurve::IDENTITY.is_identity());
assert!(!Curves::parse(BUILT_IN)
.expect("parses")
.body("Canon", "EOS 6D")
.is_identity());
}
}
-121
View File
@@ -1,121 +0,0 @@
//! TRACES: FR-RAW-2
//! The seam a second decoder plugs into.
//!
//! D2 keeps LibRaw as the fallback for bodies rawler does not cover. Adding
//! it later should be a new `impl Decoder`, not an edit to every caller that
//! reads a header, cuts a thumbnail or opens a photograph for export — which
//! is what the free functions alone would have made it. So the callers take a
//! `&dyn Decoder`, and only the places that start a job name [`default`].
//!
//! Bytes in, always. Nothing here takes a path or a `SourceRef`: resolving a
//! file to bytes is `Storage`'s job at the caller, so the same decoder serves a
//! local file, an Android document and a range fetched from Nextcloud. The
//! decoder's part in that is to say how much of a file it needs
//! ([`Decoder::header_bytes`]) and where its preview sits
//! ([`Decoder::locate_preview`]); the storage layer fetches exactly that.
//!
//! What stays a free function is what is not a decoder's to vary: recognising
//! a JPEG ([`crate::probe`]), decoding one ([`crate::decode_jpeg`]) and
//! checking one is whole ([`crate::is_complete_jpeg`]). A second RAW decoder
//! would not read a JPEG differently.
use dr_types::Orientation;
use crate::{DecodeError, Metadata, Preview, PreviewLocation, PreviewSize, RawImage};
/// TRACES: FR-RAW-2
/// A RAW decoder, over bytes.
///
/// Object-safe so a caller can hold `&dyn Decoder` without becoming generic,
/// `Send + Sync` because the callers that need one most — the thumbnail
/// lanes, the export worker — run off the UI thread, and `Debug` so a job
/// description that carries one can still be printed.
pub trait Decoder: Send + Sync + std::fmt::Debug {
/// How much of the start of a file [`Self::metadata`] and
/// [`Self::locate_preview`] need. A caller reading over a network fetches
/// this range and no more.
fn header_bytes(&self) -> u64;
/// Capture metadata, from a header or a whole file, without touching
/// sensor data.
fn metadata(&self, bytes: &[u8]) -> Result<Metadata, DecodeError>;
/// How the stored pixels are turned, from a header. `None` where the file
/// does not say, which callers take as upright.
fn orientation(&self, header: &[u8]) -> Option<Orientation>;
/// Where the embedded preview best suited to a thumbnail sits in the file,
/// from its header, so a remote caller can fetch that range alone.
fn locate_preview(&self, header: &[u8], file_len: u64) -> Option<PreviewLocation>;
/// The embedded preview at the size asked for, falling through the ladder
/// to the next size where the file lacks it.
fn preview(&self, bytes: &[u8], size: PreviewSize) -> Result<Preview, DecodeError>;
/// Sensor data, for develop and export. The expensive path.
fn decode(&self, bytes: &[u8]) -> Result<RawImage, DecodeError>;
}
/// TRACES: FR-RAW-2
/// The decoder the application ships: rawler for sensor data and the
/// previews it knows, DarkRoom's own container walk for headers and ranges.
///
/// Its methods are the crate's free functions, unchanged. They stay public
/// for the tools and examples that read one file and have no caller to keep
/// decoder-agnostic.
#[derive(Debug, Clone, Copy, Default)]
pub struct Rawler;
impl Decoder for Rawler {
fn header_bytes(&self) -> u64 {
crate::HEADER_BYTES
}
fn metadata(&self, bytes: &[u8]) -> Result<Metadata, DecodeError> {
crate::metadata(bytes)
}
fn orientation(&self, header: &[u8]) -> Option<Orientation> {
crate::orientation(header)
}
fn locate_preview(&self, header: &[u8], file_len: u64) -> Option<PreviewLocation> {
crate::locate_preview(header, file_len)
}
fn preview(&self, bytes: &[u8], size: PreviewSize) -> Result<Preview, DecodeError> {
crate::extract_preview(bytes, size)
}
fn decode(&self, bytes: &[u8]) -> Result<RawImage, DecodeError> {
crate::decode(bytes)
}
}
/// TRACES: FR-RAW-2
/// The decoder a job uses unless it was handed another.
///
/// Named by the places that start work — a thread, a UI handler — and by
/// nothing below them. Returning `&'static dyn Decoder` rather than `Rawler`
/// is the point: a caller that only has this cannot reach past the trait.
pub fn default() -> &'static dyn Decoder {
static RAWLER: Rawler = Rawler;
&RAWLER
}
#[cfg(test)]
mod tests {
use super::*;
/// The default is the shipped decoder, reached through the trait: same
/// header budget, and the same answer to bytes neither can read.
#[test]
fn the_default_is_rawler_behind_the_trait() {
let d = default();
assert_eq!(d.header_bytes(), crate::HEADER_BYTES);
let junk = [0u8; 64];
assert_eq!(d.metadata(&junk).is_err(), crate::metadata(&junk).is_err());
assert!(d.decode(&junk).is_err());
assert_eq!(d.orientation(&junk), crate::orientation(&junk));
}
}
-112
View File
@@ -1,112 +0,0 @@
/// TRACES: FR-RAW-4 | NFR-SEC-1
/// Failures from decoding.
///
/// Per FR-RAW-4 a malformed file must not abort a batch, so these are always
/// returned rather than panicking — and the decode path is the one place
/// untrusted input arrives (NFR-SEC-1).
#[derive(Debug, thiserror::Error)]
pub enum DecodeError {
#[error("read failed: {0}")]
Read(String),
#[error("unsupported or unrecognised format: {0}")]
Unsupported(String),
#[error("decode failed: {0}")]
Decode(String),
#[error("metadata unavailable: {0}")]
Metadata(String),
#[error("no embedded preview in this file")]
NoPreview,
#[error("embedded preview is corrupt: {0}")]
CorruptPreview(String),
}
/// Run a decoder call, and return a panic inside it as an error.
///
/// TRACES: FR-RAW-4 | NFR-SEC-1 | NFR-R3
/// rawler `panic!`s on some malformed input rather than returning `Err` — a
/// DNG whose IFD claims a >50000 px image, for one, which is in the reference
/// library. A panic on a worker thread ends the thread: the face sweep that
/// met that file stopped 13 seconds in, three sweeps running, with "17301
/// image(s) to index" as the last word and nothing to say why. FR-RAW-4's
/// rule — a malformed file must not abort a batch — is this crate's to keep
/// whatever the library beneath it does, so every entry point that calls into
/// rawler runs through here, and a file that panics the decoder is one failed
/// file like any other.
///
/// The crash hook still records the panic, because it runs before unwinding
/// reaches this frame; that is right — it is a real defect in a dependency
/// and the record is how it gets reported upstream — and a repeat is the same
/// file being met again rather than a new fault.
pub(crate) fn guarded<T>(
what: &'static str,
f: impl FnOnce() -> Result<T, DecodeError>,
) -> Result<T, DecodeError> {
match std::panic::catch_unwind(std::panic::AssertUnwindSafe(f)) {
Ok(result) => result,
Err(payload) => {
let msg = payload
.downcast_ref::<&str>()
.map(|s| s.to_string())
.or_else(|| payload.downcast_ref::<String>().cloned())
.unwrap_or_else(|| "no message".to_string());
Err(DecodeError::Decode(format!(
"{what}: the decoder panicked on this file: {msg}"
)))
}
}
}
impl DecodeError {
/// Whether a fallback path might still produce an image.
///
/// A missing preview is not a failure to display the file — it means fall
/// through to full decode (FR-CULL-2, M-11).
pub fn has_fallback(&self) -> bool {
matches!(
self,
DecodeError::NoPreview | DecodeError::CorruptPreview(_)
)
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn preview_failures_fall_through_rather_than_failing() {
assert!(DecodeError::NoPreview.has_fallback());
assert!(DecodeError::CorruptPreview("truncated".into()).has_fallback());
// A genuinely unsupported file has nowhere to fall through to.
assert!(!DecodeError::Unsupported("unknown".into()).has_fallback());
}
#[test]
fn a_panic_in_the_decoder_is_an_error_and_the_thread_survives() {
// The property the face sweep relies on: one file that panics rawler
// is one failed file, not the end of the pass. The message travels,
// because "decode failed" alone sends the reader to the crash log.
let err = guarded("decode", || -> Result<(), DecodeError> {
panic!("rawler: surely there's no such thing as a {}MP image!", 600)
})
.unwrap_err();
let text = err.to_string();
assert!(text.contains("panicked"), "{text}");
assert!(text.contains("600MP"), "{text}");
assert!(!err.has_fallback(), "a panic is not a missing preview");
}
#[test]
fn a_result_passes_through_untouched() {
assert_eq!(guarded("decode", || Ok::<_, DecodeError>(7)).unwrap(), 7);
assert!(matches!(
guarded("decode", || Err::<(), _>(DecodeError::NoPreview)),
Err(DecodeError::NoPreview)
));
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-412
View File
@@ -1,412 +0,0 @@
//! Embedded preview extraction — the fast display path.
//!
//! Every RAW container carries one or more JPEG previews, often at or near
//! full resolution. Extracting one costs a fraction of a full decode, and is
//! what makes culling feel instant (FR-CULL-1, NFR-P13: 50 ms per image).
//!
//! It is also what makes remote browsing viable: fetching ~1-3 MB of preview
//! from an 80 MB file over WebDAV is the difference between usable and not on
//! mobile data (FR-NC-3).
use crate::DecodeError;
/// How much of a file header to read when locating a preview.
///
/// Enough to cover the IFD structure of the TIFF-derived formats. Sized for
/// remote range requests, where every byte costs.
pub const PREVIEW_PROBE_BYTES: u64 = 256 * 1024;
/// A decoded preview image, RGBA8.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Preview {
pub width: u32,
pub height: u32,
/// Tightly packed RGBA, 4 bytes per pixel.
pub rgba: Vec<u8>,
}
impl Preview {
/// TRACES: FR-DEV-3h
/// Turn the pixels the right way up, in place.
///
/// Every path that shows a preview without the GPU needs this: the grid's
/// thumbnails, and the read-only fallback develop shows when no decoder
/// could open the file. An embedded preview is written in the sensor's
/// orientation, not the photograph's, so a phone or a camera held sideways
/// fills the grid with frames on their side until this runs.
///
/// Done before [`Self::downscale_to`] would be wasteful and after it is
/// not: a quarter turn is a permutation, so it costs the same either way,
/// and doing it on the smaller buffer moves a fraction of the bytes.
///
/// The turn itself is [`dr_types::Orientation::into_shown`], which every
/// other consumer of an orientation in this codebase also goes through.
/// That is deliberate: a hand-written permutation per caller is how two of
/// them come to disagree, and a disagreement here shows as a thumbnail
/// facing the other way from the develop view.
pub fn apply_orientation(&mut self, orientation: dr_types::Orientation) {
let (rgba, dw, dh) = orientation.into_shown(&self.rgba, self.width, self.height, 4);
self.rgba = rgba;
self.width = dw;
self.height = dh;
}
/// Downscale in place to fit within `max_dim` on the long edge.
///
/// A 5472x3648 preview is 79.8 MB of RGBA — far more than a grid cell or
/// even a 4K viewport needs, and enough to exhaust a phone's budget after
/// a handful of images (NFR-RES-1). Box-filtered rather than nearest, so
/// downscaled thumbnails do not alias.
pub fn downscale_to(&mut self, max_dim: u32) {
let longest = self.width.max(self.height);
if longest <= max_dim || longest == 0 {
return;
}
let scale = max_dim as f32 / longest as f32;
let (nw, nh) = (
((self.width as f32 * scale).round() as u32).max(1),
((self.height as f32 * scale).round() as u32).max(1),
);
let mut out = vec![0u8; (nw as usize) * (nh as usize) * 4];
let x_ratio = self.width as f32 / nw as f32;
let y_ratio = self.height as f32 / nh as f32;
for y in 0..nh {
let y0 = (y as f32 * y_ratio) as u32;
let y1 = (((y + 1) as f32 * y_ratio) as u32)
.min(self.height)
.max(y0 + 1);
for x in 0..nw {
let x0 = (x as f32 * x_ratio) as u32;
let x1 = (((x + 1) as f32 * x_ratio) as u32)
.min(self.width)
.max(x0 + 1);
let (mut r, mut g, mut b, mut n) = (0u32, 0u32, 0u32, 0u32);
for sy in y0..y1 {
for sx in x0..x1 {
let i = ((sy * self.width + sx) * 4) as usize;
r += self.rgba[i] as u32;
g += self.rgba[i + 1] as u32;
b += self.rgba[i + 2] as u32;
n += 1;
}
}
let n = n.max(1);
let o = ((y * nw + x) * 4) as usize;
out[o] = (r / n) as u8;
out[o + 1] = (g / n) as u8;
out[o + 2] = (b / n) as u8;
out[o + 3] = 255;
}
}
self.rgba = out;
self.width = nw;
self.height = nh;
}
/// Whether this is large enough to be worth displaying at `target`.
///
/// Some bodies embed thumbnails only a few hundred pixels wide — Sony is
/// the documented case. Displaying one where a larger render is wanted
/// shows a soft image the user discovers only on zoom, so the caller
/// should background-render instead (M-11).
pub fn is_useful_at(&self, target: u32) -> bool {
self.width.max(self.height) >= target
}
}
/// TRACES: FR-CULL-1 | NFR-P13
/// Which embedded image to extract.
///
/// Containers carry several at different sizes, and decoding the
/// full-resolution one to fill a grid cell is pure waste.
///
/// **Measured caveat (rawler 0.7.2):** the CR2 decoder implements only
/// `full_image`; `thumbnail_image` and `preview_image` are unimplemented trait
/// defaults returning `None`. So on Canon CR2 every rung currently resolves to
/// the full-resolution JPEG at ~250 ms — 5× over NFR-P13's 50 ms budget.
///
/// Three ways out, in increasing cost: extract the smaller IFD ourselves
/// (CR2 carries a 160×120 thumbnail and a ~1620×1080 preview in IFD1/IFD2),
/// contribute the methods upstream, or cache a downscaled proxy on first
/// sight. The ladder is written now so that fixing it is a decoder change
/// rather than a change to every caller.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum PreviewSize {
/// Smallest available. Grid cells and rapid culling.
Thumbnail,
/// Mid-sized where the container has one. Single-image view.
Screen,
/// Largest available, usually full sensor resolution. Only where the
/// display genuinely needs it.
Full,
}
/// TRACES: FR-CULL-2 | FR-NC-3 | M-10
/// Extract and decode an embedded preview at the requested size.
///
/// Takes bytes rather than a reader, because the caller usually has them
/// already: a range read locally, or a `Range:` request remotely. Forcing a
/// `Read + Seek` here would push remote callers into buffering the whole file.
///
/// Falls through the ladder — a container without the requested size yields
/// the next available rather than failing (FR-CULL-2).
///
/// Returns [`DecodeError::NoPreview`] where there is none at all: a
/// fall-through signal, not a failure (see [`DecodeError::has_fallback`]).
pub fn extract_preview(bytes: &[u8], size: PreviewSize) -> Result<Preview, DecodeError> {
crate::error::guarded("preview", || extract_preview_unguarded(bytes, size))
}
fn extract_preview_unguarded(bytes: &[u8], size: PreviewSize) -> Result<Preview, DecodeError> {
use rawler::rawsource::RawSource;
// A plain JPEG *is* its own preview — rawler has no decoder for one, and
// a mixed folder must display sensibly (M-9).
if bytes.starts_with(&[0xFF, 0xD8, 0xFF]) {
return decode_jpeg(bytes);
}
let source = RawSource::new_from_slice(bytes);
let decoder =
rawler::get_decoder(&source).map_err(|e| DecodeError::Unsupported(e.to_string()))?;
let params = Default::default();
// Preference order per requested size, each falling through to the next.
let attempts: &[PreviewSize] = match size {
PreviewSize::Thumbnail => &[
PreviewSize::Thumbnail,
PreviewSize::Screen,
PreviewSize::Full,
],
PreviewSize::Screen => &[
PreviewSize::Screen,
PreviewSize::Full,
PreviewSize::Thumbnail,
],
PreviewSize::Full => &[PreviewSize::Full, PreviewSize::Screen],
};
for attempt in attempts {
let got = match attempt {
PreviewSize::Thumbnail => decoder.thumbnail_image(&source, &params),
PreviewSize::Screen => decoder.preview_image(&source, &params),
PreviewSize::Full => decoder.full_image(&source, &params),
};
if let Ok(Some(img)) = got {
let rgb = img.to_rgb8();
let (width, height) = (rgb.width(), rgb.height());
if width > 0 && height > 0 {
return Ok(Preview {
width,
height,
rgba: rgb_to_rgba(rgb.as_raw(), width, height),
});
}
}
}
Err(DecodeError::NoPreview)
}
/// Extract the largest available preview.
///
/// Convenience over [`extract_preview`]; prefer naming a size explicitly.
pub fn extract_embedded_preview(bytes: &[u8]) -> Result<Preview, DecodeError> {
extract_preview(bytes, PreviewSize::Full)
}
/// Decode a standalone JPEG (an embedded preview already sliced out, or a
/// JPEG file).
pub fn decode_jpeg(bytes: &[u8]) -> Result<Preview, DecodeError> {
let mut d = zune_jpeg::JpegDecoder::new(bytes);
let pixels = d
.decode()
.map_err(|e| DecodeError::CorruptPreview(e.to_string()))?;
let info = d
.info()
.ok_or_else(|| DecodeError::CorruptPreview("no image info".into()))?;
let (w, h) = (info.width as u32, info.height as u32);
let expected = (w as usize) * (h as usize);
// zune yields RGB or grayscale depending on the source; normalise both to
// RGBA so callers have one representation.
let rgba = match pixels.len() / expected.max(1) {
3 => rgb_to_rgba(&pixels, w, h),
1 => pixels.iter().flat_map(|&g| [g, g, g, 255]).collect(),
4 => pixels,
n => {
return Err(DecodeError::CorruptPreview(format!(
"unexpected {n} channels"
)))
}
};
Ok(Preview {
width: w,
height: h,
rgba,
})
}
fn rgb_to_rgba(rgb: &[u8], w: u32, h: u32) -> Vec<u8> {
let n = (w as usize) * (h as usize);
let mut out = Vec::with_capacity(n * 4);
for px in rgb.chunks_exact(3).take(n) {
out.extend_from_slice(&[px[0], px[1], px[2], 255]);
}
out
}
#[cfg(test)]
mod tests {
use super::*;
/// A preview whose every pixel encodes its own coordinates, so a
/// misplaced one is identifiable rather than merely wrong.
fn coded(width: u32, height: u32) -> Preview {
let mut rgba = Vec::with_capacity((width * height * 4) as usize);
for y in 0..height {
for x in 0..width {
rgba.extend_from_slice(&[x as u8, y as u8, 0, 255]);
}
}
Preview {
width,
height,
rgba,
}
}
#[test]
fn a_quarter_turn_moves_every_pixel_where_the_orientation_says() {
// Tag 6: the stored image's first row becomes the displayed right
// edge, its first column the displayed top. A 4x2 landscape preview
// therefore comes out 2x4 portrait, with stored (0,0) at the top right.
let mut p = coded(4, 2);
p.apply_orientation(dr_types::Orientation::from_exif(6));
assert_eq!((p.width, p.height), (2, 4));
let at = |x: u32, y: u32| {
let i = ((y * p.width + x) * 4) as usize;
(p.rgba[i], p.rgba[i + 1])
};
// Displayed top-right reads stored (0, 0).
assert_eq!(at(1, 0), (0, 0));
// Displayed top-left reads stored (0, 1) — the last row of column 0.
assert_eq!(at(0, 0), (0, 1));
// Displayed bottom-right reads stored (3, 0).
assert_eq!(at(1, 3), (3, 0));
}
#[test]
fn an_upright_file_is_left_untouched() {
// The common case, and the one where an unnecessary reallocation
// would be paid on every thumbnail in the library.
let original = coded(4, 2);
let mut p = original.clone();
p.apply_orientation(dr_types::Orientation::NORMAL);
assert_eq!(p, original);
}
#[test]
fn every_orientation_preserves_the_pixels_it_was_given() {
// A turn or a mirror is a permutation: the same bytes, rearranged.
// Anything else means a pixel was dropped, duplicated or read out of
// bounds — and the bounds case would have panicked first.
for tag in 1..=8u16 {
let orientation = dr_types::Orientation::from_exif(tag);
let mut p = coded(5, 3);
p.apply_orientation(orientation);
assert_eq!(
(p.width, p.height),
orientation.oriented_size(5, 3),
"tag {tag}"
);
let mut got: Vec<_> = p.rgba.chunks(4).map(|c| (c[0], c[1])).collect();
got.sort_unstable();
let mut want: Vec<_> = coded(5, 3).rgba.chunks(4).map(|c| (c[0], c[1])).collect();
want.sort_unstable();
assert_eq!(got, want, "tag {tag}");
}
}
#[test]
fn size_preference_falls_through_in_order() {
// A container missing the requested size must yield the next
// available rather than failing (FR-CULL-2).
// Ordering is asserted here; behaviour against real files is covered
// by the smoke example.
assert_ne!(PreviewSize::Thumbnail, PreviewSize::Full);
}
#[test]
fn usefulness_is_judged_on_the_long_edge() {
let p = Preview {
width: 1600,
height: 1067,
rgba: Vec::new(),
};
assert!(p.is_useful_at(1024));
assert!(p.is_useful_at(1600));
// A body embedding only a small thumbnail must trigger a background
// render rather than showing a soft image.
assert!(!p.is_useful_at(2048));
}
#[test]
fn downscale_preserves_aspect_and_bounds_memory() {
let mut p = Preview {
width: 5472,
height: 3648,
rgba: vec![128; 5472 * 3648 * 4],
};
assert_eq!(p.rgba.len(), 79_847_424);
p.downscale_to(2048);
assert_eq!(p.width, 2048);
assert_eq!(p.height, 1365, "aspect preserved");
assert_eq!(p.rgba.len(), (2048 * 1365 * 4) as usize);
// A flat source must stay flat through the box filter.
assert!(p
.rgba
.chunks_exact(4)
.all(|px| px[0] == 128 && px[3] == 255));
}
#[test]
fn downscale_is_a_noop_when_already_small() {
let mut p = Preview {
width: 720,
height: 480,
rgba: vec![7; 720 * 480 * 4],
};
let before = p.rgba.len();
p.downscale_to(2048);
assert_eq!((p.width, p.height, p.rgba.len()), (720, 480, before));
}
#[test]
fn rgb_expands_to_rgba_opaque() {
let rgb = [10, 20, 30, 40, 50, 60];
let rgba = rgb_to_rgba(&rgb, 2, 1);
assert_eq!(rgba, vec![10, 20, 30, 255, 40, 50, 60, 255]);
}
#[test]
fn corrupt_jpeg_is_an_error_not_a_panic() {
// Untrusted input arrives here (NFR-SEC-1); it must never panic.
let err = decode_jpeg(&[0xFF, 0xD8, 0x00, 0x01, 0x02]).unwrap_err();
assert!(matches!(err, DecodeError::CorruptPreview(_)));
}
#[test]
fn empty_input_is_an_error_not_a_panic() {
assert!(decode_jpeg(&[]).is_err());
}
}
File diff suppressed because it is too large Load Diff

Some files were not shown because too many files have changed in this diff Show More