Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
83bcf09be1 |
@@ -1,41 +0,0 @@
|
||||
{
|
||||
"permissions": {
|
||||
"allow": [
|
||||
"Bash(curl -s \"https://api.github.com/search/code?q=WrapTexture+org:Noesis\" -H \"Accept: application/vnd.github+json\")",
|
||||
"Bash(curl -s \"https://api.github.com/orgs/Noesis/repos?per_page=100\")",
|
||||
"WebFetch(domain:wiki.wxwidgets.org)",
|
||||
"Bash(curl -sS \"https://gitlab.com/sciter-engine/sciter-js-sdk/-/raw/main/samples.gpu/hello-es-triangle.htm\")",
|
||||
"Bash(curl -sS \"https://gitlab.com/api/v4/projects/sciter-engine%2Fsciter-js-sdk/repository/tree?path=include&recursive=true&per_page=100&ref=main\")",
|
||||
"Bash(python3 -c ' *)",
|
||||
"Bash(curl -sL --max-time 40 \"https://api.github.com/repos/wxWidgets/wxWidgets/contents/src?ref=master\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://api.github.com/repos/wxWidgets/wxWidgets/contents/include/wx/android?ref=master\")",
|
||||
"Bash(curl -sL --max-time 40 -H \"Accept: application/vnd.github.text-match+json\" \"https://api.github.com/search/code?q=vulkan+repo:wxWidgets/wxWidgets\")",
|
||||
"Bash(curl -sS \"https://gitlab.com/sciter-engine/sciter-js-sdk/-/raw/main/include/sciter-x-video-api.h\")",
|
||||
"WebFetch(domain:docs.wxwidgets.org)",
|
||||
"Bash(curl -sS \"https://gitlab.com/sciter-engine/sciter-js-sdk/-/raw/main/CHANGELOG.md\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://raw.githubusercontent.com/wxWidgets/wxWidgets/master/docs/readme.txt\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://raw.githubusercontent.com/wxWidgets/wxWidgets/master/docs/licence.txt\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://raw.githubusercontent.com/wxWidgets/wxWidgets/master/docs/licendu.txt\")",
|
||||
"Bash(curl -sS \"https://gitlab.com/api/v4/projects/sciter-engine%2Fsciter-js-sdk/repository/tree?path=build&recursive=true&per_page=100&ref=main\")",
|
||||
"Bash(curl -sS \"https://gitlab.com/sciter-engine/sciter-js-sdk/-/raw/main/premake5.lua\")",
|
||||
"WebFetch(domain:slack-chats.kotlinlang.org)",
|
||||
"Bash(curl -sS -L \"https://sciter.com/\")",
|
||||
"Bash(curl -sL --max-time 45 \"https://api.github.com/orgs/ultralight-ux/repos?per_page=100&sort=pushed\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/GNOME/gtk/-/raw/main/gdk/gdkdmabuftexturebuilder.h\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/GNOME/gtk/-/raw/main/gdk/gdkgltexturebuilder.h\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/api/v4/projects/GNOME%2Fgtk/repository/tree?path=gdk&ref=main&per_page=100\")",
|
||||
"WebFetch(domain:docs.slint.dev)",
|
||||
"WebFetch(domain:releases.slint.dev)",
|
||||
"WebFetch(domain:flutter.dev)",
|
||||
"Bash(curl -sS -L \"https://sciter.com/forums/topic/status-of-quark-sciter-lite-sciterjs-android-ios/\")",
|
||||
"Bash(curl -sL --max-time 45 \"https://api.github.com/repos/ultralight-ux/AppCore/git/trees/master?recursive=1\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/GNOME/gtk/-/raw/main/gdk/meson.build\")",
|
||||
"WebFetch(domain:www.jetbrains.com)",
|
||||
"Bash(curl -sS \"https://gitlab.com/api/v4/projects/sciter-engine%2Fsciter-js-sdk/repository/tree?path=demos.lite&recursive=true&per_page=100&ref=main\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/api/v4/projects/GNOME%2Fgtk/repository/commits?path=gdk/android/gdkandroidglcontext.c&ref_name=main&per_page=20\")",
|
||||
"Bash(curl -sL --max-time 40 \"https://gitlab.gnome.org/GNOME/gtk/-/raw/main/gdk/android/meson.build\")",
|
||||
"WebFetch(domain:docs.sciter.com)",
|
||||
"Bash(curl -sS -L \"https://sciter.com/support-of-displayflex-and-displaygrid-in-sciter/\")"
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -1,12 +0,0 @@
|
||||
# Model weights live in LFS.
|
||||
#
|
||||
# `models/**/*.onnx` is tens of MB of binary that changes wholesale
|
||||
# when it changes at all. In ordinary git objects every future revision of it
|
||||
# would be stored in full, in every clone, forever — and the one thing nobody
|
||||
# can do with it is a useful diff.
|
||||
#
|
||||
# Consequence worth knowing before it bites: a clone without git-lfs gets a
|
||||
# ~130-byte pointer file where the model should be. `dr-segment`'s build script
|
||||
# detects exactly that and fails with an instruction rather than embedding the
|
||||
# pointer and failing at inference time.
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
@@ -1,170 +0,0 @@
|
||||
name: '🐳 Android image'
|
||||
|
||||
# Builds and pushes gitea.tourolle.paris/dtourolle/darkroom-android, the job
|
||||
# container for the Android leg of build-and-test.yml.
|
||||
#
|
||||
# It exists because that image previously lived only on a developer's laptop:
|
||||
# the workflow referenced a tag that had never been pushed, and every Android
|
||||
# job died at `docker pull` with "manifest unknown" before running a step. The
|
||||
# image is now reproducible from the repo rather than from one machine.
|
||||
#
|
||||
# Called by build-and-test.yml on every push, and runnable by hand via
|
||||
# workflow_dispatch. It is cheap when nothing changed — see the guard below.
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
force:
|
||||
description: 'Rebuild even if the registry already has this image ("true"/"false")'
|
||||
type: string
|
||||
default: 'false'
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
force:
|
||||
description: 'Rebuild even if the registry already has this image ("true"/"false")'
|
||||
type: string
|
||||
default: 'false'
|
||||
|
||||
# Gitea's act_runner mangles boolean workflow inputs passed through an
|
||||
# expression — they arrive as false regardless of what was sent. Every input
|
||||
# here is a string compared with == 'true', as in KPN's docker.yaml.
|
||||
|
||||
env:
|
||||
IMAGE: gitea.tourolle.paris/dtourolle/darkroom-android
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: linux/amd64
|
||||
name: Build and push
|
||||
# Deliberately NOT in a container: this job needs the host Docker daemon to
|
||||
# build an image, and the host's cached ~/.docker/config.json to push it.
|
||||
# That is also why there is no `docker login` step — the runner host was
|
||||
# authenticated to the registry during setup.
|
||||
|
||||
steps:
|
||||
# The host has no Node, so the JS-based actions/checkout cannot run here.
|
||||
# A minimal shallow fetch with plain git gets the same tree.
|
||||
- name: Checkout
|
||||
run: |
|
||||
set -e
|
||||
git init -q .
|
||||
git remote add origin "${{ github.server_url }}/${{ github.repository }}.git"
|
||||
git -c http.extraheader="AUTHORIZATION: basic $(printf '%s' '${{ github.actor }}:${{ github.token }}' | base64 -w0)" \
|
||||
fetch --depth 1 origin "${{ github.sha }}"
|
||||
git checkout -q FETCH_HEAD
|
||||
|
||||
# The image is tagged by the content of docker/android, not by the commit
|
||||
# that happened to touch it. `git rev-parse HEAD:<dir>` is the tree object
|
||||
# id — it changes when and only when a file in that directory changes, so
|
||||
# an unrelated push reuses the existing image and a Dockerfile edit can
|
||||
# never silently keep serving a stale `latest`.
|
||||
#
|
||||
# Using the commit sha instead would rebuild 7 GB on every push; using a
|
||||
# paths-filter action would need a container that has Node, and the only
|
||||
# one this repo would reach for is the very image being built.
|
||||
- name: Resolve image tag
|
||||
id: tag
|
||||
run: |
|
||||
set -e
|
||||
TREE=$(git rev-parse HEAD:docker/android)
|
||||
echo "tree=$TREE" >> "$GITHUB_OUTPUT"
|
||||
echo "docker/android tree: $TREE"
|
||||
|
||||
# Skip the build when the registry already holds this exact content. This
|
||||
# is what keeps the job a few seconds long on a normal push, and what
|
||||
# makes it self-healing: if the tag is missing for any reason, including
|
||||
# the image having never been pushed at all, it gets built here.
|
||||
#
|
||||
# The probe is curl against the registry API, NOT `docker manifest
|
||||
# inspect`. The latter exits 1 on this registry even for tags that are
|
||||
# demonstrably present — jellytau-builder:latest answers HTTP 200 to the
|
||||
# API while `docker manifest inspect` reports "manifest unknown" for it.
|
||||
# Trusting that would have rebuilt 7 GB on every single push.
|
||||
#
|
||||
# A HEAD request also gives the digest for free, which is how the repoint
|
||||
# decision below is made without pulling any layers.
|
||||
- name: Query registry
|
||||
id: check
|
||||
env:
|
||||
# The runner's own credentials, so this does not depend on how the
|
||||
# host's ~/.docker/config.json happens to be set up.
|
||||
REG_USER: ${{ github.actor }}
|
||||
REG_PASS: ${{ github.token }}
|
||||
TREE: ${{ steps.tag.outputs.tree }}
|
||||
run: |
|
||||
set -eu
|
||||
ACCEPT='application/vnd.oci.image.index.v1+json,application/vnd.docker.distribution.manifest.v2+json,application/vnd.oci.image.manifest.v1+json,application/vnd.docker.distribution.manifest.list.v2+json'
|
||||
API="https://gitea.tourolle.paris/v2/dtourolle/darkroom-android/manifests"
|
||||
|
||||
# Prints "<http-status> <digest-or-empty>" for a tag.
|
||||
probe() {
|
||||
curl -sI -u "$REG_USER:$REG_PASS" -H "Accept: $ACCEPT" "$API/$1" \
|
||||
| tr -d '\r' \
|
||||
| awk 'BEGIN{s="000";d=""} /^HTTP/{s=$2} tolower($1)=="docker-content-digest:"{d=$2} END{print s, d}'
|
||||
}
|
||||
|
||||
read -r TREE_STATUS TREE_DIGEST <<EOF
|
||||
$(probe "$TREE")
|
||||
EOF
|
||||
read -r LATEST_STATUS LATEST_DIGEST <<EOF
|
||||
$(probe latest)
|
||||
EOF
|
||||
|
||||
echo "tag $TREE -> HTTP $TREE_STATUS ${TREE_DIGEST:-(no digest)}"
|
||||
echo "tag latest -> HTTP $LATEST_STATUS ${LATEST_DIGEST:-(no digest)}"
|
||||
|
||||
# Build unless the registry definitively confirms this content is
|
||||
# already there. An auth failure or an unreachable registry lands
|
||||
# here too, and rebuilding needlessly is the safe direction to fail —
|
||||
# skipping a build that was needed is what breaks the Android job.
|
||||
if [ "${{ inputs.force }}" = "true" ]; then
|
||||
echo "forced rebuild requested"
|
||||
echo "build=true" >> "$GITHUB_OUTPUT"
|
||||
echo "repoint=false" >> "$GITHUB_OUTPUT"
|
||||
elif [ "$TREE_STATUS" != "200" ]; then
|
||||
echo "registry does not have this content — building"
|
||||
echo "build=true" >> "$GITHUB_OUTPUT"
|
||||
echo "repoint=false" >> "$GITHUB_OUTPUT"
|
||||
elif [ -n "$TREE_DIGEST" ] && [ "$TREE_DIGEST" = "$LATEST_DIGEST" ]; then
|
||||
echo "registry is already correct — nothing to do"
|
||||
echo "build=false" >> "$GITHUB_OUTPUT"
|
||||
echo "repoint=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "content is present but latest points elsewhere — repointing"
|
||||
echo "build=false" >> "$GITHUB_OUTPUT"
|
||||
echo "repoint=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# Context is docker/android, matching the README's build command. The
|
||||
# Dockerfile COPYs nothing from the repo, so it needs no wider context —
|
||||
# and a narrow context keeps the daemon from tarring up the whole tree,
|
||||
# target/ included.
|
||||
- name: Build
|
||||
if: ${{ steps.check.outputs.build == 'true' }}
|
||||
run: |
|
||||
set -e
|
||||
docker build \
|
||||
-t "$IMAGE:${{ steps.tag.outputs.tree }}" \
|
||||
-t "$IMAGE:latest" \
|
||||
docker/android
|
||||
|
||||
# Both tags are pushed: the tree tag is what the guard above looks for on
|
||||
# the next run, and `latest` is what build-and-test.yml pulls.
|
||||
- name: Push
|
||||
if: ${{ steps.check.outputs.build == 'true' }}
|
||||
run: |
|
||||
set -e
|
||||
docker push "$IMAGE:${{ steps.tag.outputs.tree }}"
|
||||
docker push "$IMAGE:latest"
|
||||
|
||||
# A cache hit on the tree tag says nothing about where `latest` points — a
|
||||
# reverted Dockerfile or a build from another branch can leave it on
|
||||
# different content. This runs only when the digests above actually
|
||||
# disagree, so the common case costs nothing; the layers are already in
|
||||
# the registry, so the push that follows uploads a manifest, not 7 GB.
|
||||
- name: Repoint latest
|
||||
if: ${{ steps.check.outputs.repoint == 'true' }}
|
||||
run: |
|
||||
set -e
|
||||
docker pull "$IMAGE:${{ steps.tag.outputs.tree }}"
|
||||
docker tag "$IMAGE:${{ steps.tag.outputs.tree }}" "$IMAGE:latest"
|
||||
docker push "$IMAGE:latest"
|
||||
@@ -1,193 +0,0 @@
|
||||
name: Benchmarks
|
||||
|
||||
# The suite docs/requirements.md §8 has been promising since it was written:
|
||||
# "an automated benchmark suite against a synthetic 50k catalog, run per-commit
|
||||
# … A regression beyond stated tolerance fails the build."
|
||||
#
|
||||
# Its own workflow rather than a step inside build-and-test.yml, and the reason
|
||||
# is what a failure here means. A red `Build and test` says the code is wrong; a
|
||||
# red `Benchmarks` says the code is slower than it was, which is a different
|
||||
# conversation, is read by different people, and must not be reachable by
|
||||
# retrying a flaky compile.
|
||||
#
|
||||
# # Why this is split in two
|
||||
#
|
||||
# §8 names "the reference desktop", not CI, and it is right to. So:
|
||||
#
|
||||
# cpu — runs on every push. It needs no adapter and no display, and the
|
||||
# budgets it asserts (a 50k catalog opening inside two seconds) have
|
||||
# two orders of magnitude of headroom, so a modest runner can be held
|
||||
# to them honestly. Machine-sensitive budgets — throughput targets
|
||||
# written for a 24-thread desktop — are reported here rather than
|
||||
# asserted; `dr-bench` decides that per metric and says so in its
|
||||
# report. Asserting them on a two-core container would produce exactly
|
||||
# what core/dr-gpu/tests/frame_budget.rs refused to produce: a red gate
|
||||
# everybody learns to ignore.
|
||||
#
|
||||
# gpu — the frame budget, which already exists and already skips itself where
|
||||
# there is no adapter. Not on push: it would build wgpu and naga on
|
||||
# every commit to establish, every time, that this runner has no GPU. It
|
||||
# runs on demand (Actions → Run workflow) so that a runner that *does*
|
||||
# have one can be pointed at it, and the numbers it produces belong in
|
||||
# docs/frame-budget.md by hand, as they already are.
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main, master, develop]
|
||||
pull_request:
|
||||
branches: [main, master, develop]
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
cpu:
|
||||
runs-on: linux/amd64
|
||||
name: CPU and I/O (per commit)
|
||||
# Node for actions/checkout and actions/cache, which the bare runner image
|
||||
# cannot execute. Rust is installed below.
|
||||
container:
|
||||
image: catthehacker/ubuntu:act-latest
|
||||
|
||||
env:
|
||||
# Same reasoning as the desktop job in build-and-test.yml: incremental
|
||||
# state exists to make the *second* build in a working tree fast, which is
|
||||
# not a thing a fresh checkout has, and it fills the runner's disk.
|
||||
CARGO_INCREMENTAL: 0
|
||||
# The fixture, out of the workspace so actions/cache never picks it up.
|
||||
# A 14 MB synthetic catalog is two seconds to regenerate and would
|
||||
# otherwise be uploaded and downloaded on every push to save them.
|
||||
DR_BENCH_DIR: /tmp/darkroom-bench
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# No `git lfs pull` here, deliberately. `dr-bench` depends on the catalog,
|
||||
# the decoder, the thumbnail store and the encoder, and on nothing that
|
||||
# reaches `dr-segment` — so the model this repository keeps in LFS is not
|
||||
# part of this job's dependency graph and fetching it would be a minute
|
||||
# spent on a file nothing opens.
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
target
|
||||
key: bench-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
|
||||
|
||||
# Pinned to the workspace rust-version, as every other job here is: a
|
||||
# floating toolchain turns an unrelated push into a mystery failure, and
|
||||
# for a benchmark it would turn one into a mystery *regression*.
|
||||
#
|
||||
# rust-analyzer is named for the reason build-and-test.yml gives: rustup
|
||||
# reconciles rust-toolchain.toml on the first cargo call whatever this
|
||||
# step asks for, so naming it keeps the download inside the step that says
|
||||
# it is installing things.
|
||||
- name: Install Rust 1.92.0
|
||||
run: |
|
||||
set -e
|
||||
curl -fsSL https://sh.rustup.rs | sh -s -- \
|
||||
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
|
||||
--component rust-analyzer
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
|
||||
# `-p dr-bench`, not `--workspace`. The whole point of that crate having
|
||||
# no GPU and no UI dependency is that this job resolves the catalog, the
|
||||
# decoder and the encoders and stops there — a few minutes rather than the
|
||||
# release build of Slint and wgpu the desktop job pays for.
|
||||
#
|
||||
# Release, and it is not optional: the workspace builds its own crates at
|
||||
# opt-level = 0 in dev, and every figure this produces is dominated by
|
||||
# this workspace's own code. A debug run would measure rustc.
|
||||
- name: Build the suite
|
||||
run: cargo build --release -p dr-bench
|
||||
|
||||
# Exit 1 is a violated budget or a regression past tolerance; exit 2 is
|
||||
# the harness failing to run at all. Both fail the job, and the report
|
||||
# above the failure says which.
|
||||
- name: Measure, and gate
|
||||
run: cargo run --release -p dr-bench -- check
|
||||
|
||||
- name: Disk after
|
||||
if: always()
|
||||
run: df -h /workspace 2>/dev/null || df -h .
|
||||
|
||||
gpu:
|
||||
# On demand only — see the header. A runner with a Vulkan device can be
|
||||
# pointed at this; one without will skip the measurement and say so, which
|
||||
# is the same posture the rest of this repository's device tests take.
|
||||
if: github.event_name == 'workflow_dispatch'
|
||||
runs-on: linux/amd64
|
||||
name: Frame budget (on demand)
|
||||
container:
|
||||
image: catthehacker/ubuntu:act-latest
|
||||
|
||||
env:
|
||||
CARGO_INCREMENTAL: 0
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# `dr-gpu` depends on `dr-segment` for the watershed's pixel passes. Its
|
||||
# default features are off, so no weights are compiled in — but the fetch
|
||||
# is cheap insurance and its failure is not fatal. The header of the same
|
||||
# step in build-and-test.yml explains why the extraheader is stripped
|
||||
# rather than reused: two Authorization headers is a 400 from Gitea, one
|
||||
# step after the batch call that had just succeeded.
|
||||
- name: Fetch the segmentation model
|
||||
continue-on-error: true
|
||||
env:
|
||||
LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
|
||||
run: |
|
||||
set -e
|
||||
git lfs install --local
|
||||
git config --local --get-regexp '^http\..*extraheader$' \
|
||||
| cut -d' ' -f1 | sort -u \
|
||||
| while read -r key; do git config --local --unset-all "$key"; done || true
|
||||
git config --local lfs.url \
|
||||
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
|
||||
git lfs pull
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
target
|
||||
key: bench-gpu-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
|
||||
|
||||
- name: Build dependencies
|
||||
run: |
|
||||
apt-get update -qq
|
||||
apt-get install -y -qq pkg-config libfontconfig1-dev libxkbcommon-dev
|
||||
|
||||
- name: Install Rust 1.92.0
|
||||
run: |
|
||||
set -e
|
||||
curl -fsSL https://sh.rustup.rs | sh -s -- \
|
||||
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
|
||||
--component rust-analyzer
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
|
||||
# The guard, in release. Its own module documentation is explicit that a
|
||||
# release run checks strictly more than a dev one: the CPU half of a frame
|
||||
# is shader-string assembly, which is several times slower unoptimised, so
|
||||
# it is folded into the assertion only when debug_assertions is off.
|
||||
#
|
||||
# With no adapter this prints "skipping: no GPU adapter" and passes. A
|
||||
# test that cannot run is not evidence either way, and turning that into a
|
||||
# failure would make the job useless on the runner it usually lands on.
|
||||
- name: Frame budget (FR-DSP-3)
|
||||
run: cargo test --release -p dr-gpu --test frame_budget -- --nocapture
|
||||
|
||||
# The instrument behind docs/frame-budget.md. It exits non-zero with no
|
||||
# adapter, which is right for a tool a person runs deliberately and wrong
|
||||
# for a job that usually has none — hence continue-on-error. Its table is
|
||||
# in the log for whoever asked for this run; the committed numbers are
|
||||
# still updated by hand, as that file says.
|
||||
- name: Frame budget table
|
||||
continue-on-error: true
|
||||
run: cargo run --release -p dr-gpu --example frame_budget
|
||||
@@ -1,410 +0,0 @@
|
||||
name: Build and test
|
||||
|
||||
# Desktop and Android are built on every push, per the v0.1 decision to carry
|
||||
# both platforms from the first commit. An Android break is then caught the day
|
||||
# it lands rather than at a porting milestone.
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main, master, develop]
|
||||
pull_request:
|
||||
branches: [main, master, develop]
|
||||
|
||||
jobs:
|
||||
# The Android job runs inside an image that this repo builds. Ensure it is in
|
||||
# the registry before anything tries to pull it — see android-image.yml for
|
||||
# why this is a job rather than a documented manual step. It is a no-op of a
|
||||
# few seconds unless docker/android actually changed.
|
||||
android-image:
|
||||
uses: ./.gitea/workflows/android-image.yml
|
||||
|
||||
desktop:
|
||||
runs-on: linux/amd64
|
||||
name: Desktop (Linux)
|
||||
# actions/checkout and actions/cache are JavaScript actions: the runner
|
||||
# executes them with Node from inside this container. The bare runner image
|
||||
# has none, so the job failed at checkout before reaching any build step.
|
||||
container:
|
||||
image: catthehacker/ubuntu:act-latest
|
||||
|
||||
# This job filled the runner's disk and died mid-link with "No space left
|
||||
# on device" — LLVM reporting an IO failure on its output stream, which
|
||||
# reads like a compiler crash and is not one.
|
||||
#
|
||||
# `target/debug` was 24 GB against `target/release`'s 2.6 GB: 15 GB of it
|
||||
# debug info in `debug/deps`, 3.6 GB incremental state. Neither earns its
|
||||
# space here. Nothing attaches a debugger to a CI run, and incremental
|
||||
# compilation exists to make the *second* build in a working tree fast,
|
||||
# which is not a thing a fresh checkout has. Turning both off is the
|
||||
# standard CI setting rather than a trick.
|
||||
#
|
||||
# Measured on this workspace: the same `cargo test --workspace --no-run`
|
||||
# tree goes from 24 GB to 3.3 GB, `debug/deps` from 15 GB to 2.8 GB.
|
||||
#
|
||||
# Backtraces still name functions without debug info; they lose file and
|
||||
# line numbers. If a test failure ever needs those, drop DEBUG to 1
|
||||
# (line-tables-only) rather than back to 2.
|
||||
#
|
||||
# This is a mitigation, not a fix. If the runner is full of anything other
|
||||
# than this job's own output, it will still be full afterwards.
|
||||
env:
|
||||
CARGO_INCREMENTAL: 0
|
||||
CARGO_PROFILE_DEV_DEBUG: 0
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# The model, which is in LFS and is not optional.
|
||||
#
|
||||
# `models/**/*.onnx` is tracked in LFS (.gitattributes), so a
|
||||
# plain checkout writes a ~130-byte pointer where 11 MB should be, and
|
||||
# `dr-segment`'s build script panics by design rather than embedding a
|
||||
# pointer and failing at inference. That failure reads like a broken build
|
||||
# instead of a missing fetch, which is how it went unnoticed.
|
||||
#
|
||||
# Not `lfs: true` on the checkout above, and no `Authorization` header
|
||||
# here either. Both install a blanket header for every request to this
|
||||
# host, and the object download is the one request that already carries
|
||||
# one: `git lfs pull` asks `/info/lfs/objects/batch` first, and Gitea
|
||||
# answers with a short-lived `Bearer` JWT scoped to that object. git-lfs
|
||||
# then sends the JWT *and* the configured header, and two `Authorization`
|
||||
# headers is a 400 from Gitea — reported as
|
||||
# LFS: Client error: .../info/lfs/objects/<oid>
|
||||
# one step after the batch call that had just succeeded, which reads like
|
||||
# a rejected credential rather than a duplicated one. A lone token header
|
||||
# is understood fine; it is only the collision that fails.
|
||||
#
|
||||
# So: strip the headers and hand the token to git-lfs as an ordinary
|
||||
# credential instead. It authenticates the batch call and leaves the
|
||||
# per-object JWT untouched. Gitea authenticates on the password, so the
|
||||
# username is a placeholder. Nothing later in this job talks to the
|
||||
# remote, so dropping checkout's header costs us nothing.
|
||||
#
|
||||
# `continue-on-error` deliberately: if this cannot authenticate, the build
|
||||
# below still runs and fails with the build script's own message, which
|
||||
# names the real problem. A checkout that dies here says nothing.
|
||||
- name: Fetch the segmentation model
|
||||
continue-on-error: true
|
||||
env:
|
||||
LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
|
||||
run: |
|
||||
set -e
|
||||
git lfs install --local
|
||||
git config --local --get-regexp '^http\..*extraheader$' \
|
||||
| cut -d' ' -f1 | sort -u \
|
||||
| while read -r key; do git config --local --unset-all "$key"; done || true
|
||||
git config --local lfs.url \
|
||||
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
|
||||
git lfs pull
|
||||
ls -lR models/
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
target
|
||||
key: desktop-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
|
||||
|
||||
# Slint and winit need these at build time; the runner image is minimal.
|
||||
- name: Build dependencies
|
||||
run: |
|
||||
apt-get update -qq
|
||||
apt-get install -y -qq pkg-config libfontconfig1-dev libxkbcommon-dev
|
||||
|
||||
# The act image ships Node but no Rust. Pinned to the workspace
|
||||
# rust-version so CI, the Android image, and local builds agree — a
|
||||
# floating toolchain turns an unrelated push into a mystery failure.
|
||||
#
|
||||
# The component list mirrors rust-toolchain.toml's, rust-analyzer
|
||||
# included, even though nothing in this job runs it. rustup reconciles
|
||||
# that file against the installed toolchain on the first cargo call in
|
||||
# the work tree and fetches whatever is missing — so leaving it out does
|
||||
# not save the download, it only moves it into the middle of a build
|
||||
# step where it is nobody's line item. Naming it here keeps every fetch
|
||||
# inside the step whose name says it is installing things.
|
||||
- name: Install Rust 1.92.0
|
||||
run: |
|
||||
set -e
|
||||
curl -fsSL https://sh.rustup.rs | sh -s -- \
|
||||
-y --no-modify-path --profile minimal \
|
||||
--default-toolchain 1.92.0 --component rustfmt,clippy,rust-analyzer
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
|
||||
# Free space before and after the expensive steps, so a repeat of the
|
||||
# disk exhaustion above is one line to diagnose instead of a puzzling
|
||||
# LLVM error.
|
||||
- name: Disk before
|
||||
run: df -h /workspace 2>/dev/null || df -h .
|
||||
|
||||
- name: Format
|
||||
run: cargo fmt --all -- --check
|
||||
|
||||
- name: Clippy
|
||||
run: cargo clippy --workspace --all-targets -- -D warnings
|
||||
|
||||
# GPU tests skip themselves where no adapter is present rather than
|
||||
# failing — CI runners generally have none, and a test that cannot run is
|
||||
# not evidence either way.
|
||||
- name: Test
|
||||
run: cargo test --workspace
|
||||
|
||||
- name: Build
|
||||
run: cargo build --workspace --release
|
||||
|
||||
- name: Disk after
|
||||
if: always()
|
||||
run: df -h /workspace 2>/dev/null || df -h .
|
||||
|
||||
android:
|
||||
runs-on: linux/amd64
|
||||
name: Android (aarch64)
|
||||
# Waits for the image build. Without this the pull races the push and the
|
||||
# job dies with "manifest unknown" before its first step, which is the
|
||||
# failure mode this ordering exists to remove.
|
||||
needs: android-image
|
||||
container:
|
||||
image: gitea.tourolle.paris/dtourolle/darkroom-android:latest
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# The model, which is in LFS and is not optional.
|
||||
#
|
||||
# `models/**/*.onnx` is tracked in LFS (.gitattributes), so a
|
||||
# plain checkout writes a ~130-byte pointer where 11 MB should be, and
|
||||
# `dr-segment`'s build script panics by design rather than embedding a
|
||||
# pointer and failing at inference. That failure reads like a broken build
|
||||
# instead of a missing fetch, which is how it went unnoticed.
|
||||
#
|
||||
# Not `lfs: true` on the checkout above, and no `Authorization` header
|
||||
# here either. Both install a blanket header for every request to this
|
||||
# host, and the object download is the one request that already carries
|
||||
# one: `git lfs pull` asks `/info/lfs/objects/batch` first, and Gitea
|
||||
# answers with a short-lived `Bearer` JWT scoped to that object. git-lfs
|
||||
# then sends the JWT *and* the configured header, and two `Authorization`
|
||||
# headers is a 400 from Gitea — reported as
|
||||
# LFS: Client error: .../info/lfs/objects/<oid>
|
||||
# one step after the batch call that had just succeeded, which reads like
|
||||
# a rejected credential rather than a duplicated one. A lone token header
|
||||
# is understood fine; it is only the collision that fails.
|
||||
#
|
||||
# So: strip the headers and hand the token to git-lfs as an ordinary
|
||||
# credential instead. It authenticates the batch call and leaves the
|
||||
# per-object JWT untouched. Gitea authenticates on the password, so the
|
||||
# username is a placeholder. Nothing later in this job talks to the
|
||||
# remote, so dropping checkout's header costs us nothing.
|
||||
#
|
||||
# `continue-on-error` deliberately: if this cannot authenticate, the build
|
||||
# below still runs and fails with the build script's own message, which
|
||||
# names the real problem. A checkout that dies here says nothing.
|
||||
- name: Fetch the segmentation model
|
||||
continue-on-error: true
|
||||
env:
|
||||
LFS_TOKEN: ${{ secrets.GITEA_TOKEN || github.token }}
|
||||
run: |
|
||||
set -e
|
||||
git lfs install --local
|
||||
git config --local --get-regexp '^http\..*extraheader$' \
|
||||
| cut -d' ' -f1 | sort -u \
|
||||
| while read -r key; do git config --local --unset-all "$key"; done || true
|
||||
git config --local lfs.url \
|
||||
"https://x-access-token:${LFS_TOKEN}@gitea.tourolle.paris/dtourolle/DarkRoom.git/info/lfs"
|
||||
git lfs pull
|
||||
ls -lR models/
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
/opt/cargo/registry
|
||||
target-android
|
||||
key: android-${{ hashFiles('**/Cargo.lock') }}
|
||||
|
||||
# A fast gate on the crates most likely to break the cross-compile, run
|
||||
# before the expensive part. It is `cargo check`, so it type-checks
|
||||
# without linking and returns in a fraction of the time the step below
|
||||
# takes.
|
||||
#
|
||||
# Not a statement that only these crates cross-compile — `darkroom-android`
|
||||
# and the whole UI stack beneath it build for aarch64 too, which is what
|
||||
# the API-level step below does. This one exists to fail fast and name a
|
||||
# smaller suspect when it does.
|
||||
- name: Cross-compile core
|
||||
env:
|
||||
CARGO_TARGET_DIR: target-android
|
||||
run: cargo check -p dr-types -p dr-gpu -p dr-sync --target aarch64-linux-android
|
||||
|
||||
# The linker targets MIN_API, not the compile SDK. cargo-ndk otherwise
|
||||
# defaults to API 21, far below the Vulkan floor this app needs — and the
|
||||
# mismatch is invisible until a device refuses to install.
|
||||
#
|
||||
# Look under the target triple, and fail on a mismatch. Searching the
|
||||
# whole target dir for the first `*.so` found the host proc-macro
|
||||
# libraries in target-android/debug/deps instead — x86-64 objects built
|
||||
# by the runner's gcc, whose .comment section says nothing about Android
|
||||
# and can never contradict the expected API. The step passed regardless
|
||||
# of what the linker actually did, which is the one thing it exists to
|
||||
# rule out.
|
||||
- name: Verify minimum API level
|
||||
env:
|
||||
CARGO_TARGET_DIR: target-android
|
||||
run: |
|
||||
set -e
|
||||
# `darkroom-android`, not a core crate: this step reads the API level
|
||||
# out of a *linked* object, and only that crate produces one. It is
|
||||
# the workspace's single `crate-type = ["cdylib"]`; a library crate
|
||||
# builds an rlib, which is an archive of object files that no linker
|
||||
# has yet touched and that `file` therefore has nothing to say about.
|
||||
# Asking for `-p dr-gpu` here could only ever reach the "no aarch64
|
||||
# .so was produced" branch below, whatever the linker did.
|
||||
#
|
||||
# It is also the honest artefact to check: the .so this names is the
|
||||
# one that ships in the APK, so the API level verified here is the
|
||||
# API level a device will refuse to install against.
|
||||
cargo ndk -t arm64-v8a -o target-android/jniLibs \
|
||||
build -p darkroom-android --release
|
||||
MIN_API=$(sed -n 's/^ARG MIN_API=\([0-9]*\).*/\1/p' docker/android/Dockerfile)
|
||||
# Empty on both sides would compare equal and pass, so neither side
|
||||
# is allowed to be the result of a failed parse.
|
||||
if [ -z "$MIN_API" ]; then
|
||||
echo "no ARG MIN_API= in docker/android/Dockerfile"
|
||||
exit 1
|
||||
fi
|
||||
SO=$(find target-android/aarch64-linux-android/release -maxdepth 1 -name '*.so' | head -1)
|
||||
if [ -z "$SO" ]; then
|
||||
echo "no aarch64 .so was produced"
|
||||
exit 1
|
||||
fi
|
||||
echo "checking $SO"
|
||||
# `file` is kept for the log — it names the NDK that built this — but
|
||||
# the check no longer depends on it.
|
||||
file "$SO" || true
|
||||
# The API level is the first word of the `.note.android.ident` ELF
|
||||
# note, little-endian. Read the note rather than asking `file` for it:
|
||||
# `file` only prints "for Android 28" when its magic database is new
|
||||
# enough to decode that note, and this image's is not. The parse then
|
||||
# produced nothing, `${API:-unknown}` reported "unknown", and every
|
||||
# push failed here for weeks on a .so that was linked perfectly
|
||||
# correctly. A note read straight out of the ELF cannot go stale that
|
||||
# way.
|
||||
readelf -n "$SO" | sed -n '/android.ident/,+3p'
|
||||
HEX=$(readelf -n "$SO" 2>/dev/null \
|
||||
| awk '/description data:/ { print $6 $5 $4 $3; exit }')
|
||||
if [ -z "$HEX" ]; then
|
||||
echo "FAIL: no .note.android.ident in $SO — nothing states an API level"
|
||||
exit 1
|
||||
fi
|
||||
API=$(( 0x$HEX ))
|
||||
if [ "$API" != "$MIN_API" ]; then
|
||||
echo "FAIL: linked for Android $API, expected $MIN_API"
|
||||
exit 1
|
||||
fi
|
||||
echo "OK: linked for Android $API"
|
||||
|
||||
# The APK itself, so a run leaves something installable behind rather
|
||||
# than only the knowledge that it would have linked. The assembly is
|
||||
# `docker/android/assemble-apk.sh`, shared with `package.sh` so the file
|
||||
# a device gets from `package.sh --install` and the file published here
|
||||
# are built by the same code — see that script's header.
|
||||
#
|
||||
# `KEYSTORE` deliberately points at a throwaway directory instead of its
|
||||
# default under `target-android`: that directory is what `actions/cache`
|
||||
# restores and saves, and a signing key has no business in a build cache
|
||||
# or in anything this job uploads. A fresh debug key per run is the right
|
||||
# trade for an artefact whose purpose is getting the app onto a test
|
||||
# device; nothing upgrades in place over it, which is the one thing a
|
||||
# stable key would buy.
|
||||
- name: Package the APK
|
||||
env:
|
||||
CARGO_TARGET_DIR: target-android
|
||||
# Absent secrets mean a debug signature, which is what a fork or a
|
||||
# branch build should get. Set all three (see docs/android-signing.md)
|
||||
# and the same job produces a release-signed APK instead.
|
||||
ANDROID_KEYSTORE_BASE64: ${{ secrets.ANDROID_KEYSTORE_BASE64 }}
|
||||
KEYSTORE_PASS: ${{ secrets.ANDROID_KEYSTORE_PASSWORD }}
|
||||
KEY_PASS: ${{ secrets.ANDROID_KEY_PASSWORD }}
|
||||
KEY_ALIAS: ${{ secrets.ANDROID_KEY_ALIAS }}
|
||||
run: |
|
||||
set -e
|
||||
KEYDIR="$(mktemp -d)"
|
||||
chmod 700 "$KEYDIR"
|
||||
trap 'rm -rf "$KEYDIR"' EXIT
|
||||
|
||||
if [ -n "$ANDROID_KEYSTORE_BASE64" ]; then
|
||||
# The keystore reaches the runner base64-encoded because a secret
|
||||
# is a string. It is written under a 0700 mktemp directory, never
|
||||
# into the workspace: `target-android` is what actions/cache saves,
|
||||
# and the upload step globs the workspace.
|
||||
printf '%s' "$ANDROID_KEYSTORE_BASE64" | base64 -d > "$KEYDIR/release.keystore"
|
||||
export KEYSTORE="$KEYDIR/release.keystore"
|
||||
else
|
||||
# Not an error. Unset the rest so assemble-apk.sh takes its debug
|
||||
# path cleanly rather than seeing a half-configured release one.
|
||||
export KEYSTORE="$KEYDIR/debug.keystore"
|
||||
unset KEYSTORE_PASS KEY_PASS KEY_ALIAS
|
||||
fi
|
||||
|
||||
REPO="$PWD" TARGET_DIR="$PWD/target-android" \
|
||||
bash docker/android/assemble-apk.sh
|
||||
|
||||
# v3, not v4. v4 is untested against this Gitea and its runner; v3 is
|
||||
# what JellyTau uploads its APK with on this same runner, so it is the
|
||||
# version known to work here rather than the version that ought to.
|
||||
#
|
||||
# `if-no-files-found: error` because the failure this guards against is
|
||||
# a green run with an empty artefact list, which reads as success until
|
||||
# somebody goes looking for the file.
|
||||
- name: Upload the APK
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: darkroom-arm64-v8a-apk
|
||||
path: target-android/apk/darkroom.apk
|
||||
if-no-files-found: error
|
||||
|
||||
layering:
|
||||
runs-on: linux/amd64
|
||||
name: Layer separation
|
||||
# Node for the JS actions, as above. cargo comes from rustup below.
|
||||
container:
|
||||
image: catthehacker/ubuntu:act-latest
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# `cargo tree` resolves the dependency graph, so it needs the registry
|
||||
# index but no system libraries — this job builds nothing.
|
||||
#
|
||||
# rust-analyzer is named for the reason given in the desktop job: rustup
|
||||
# installs rust-toolchain.toml's components on the first cargo call
|
||||
# whether or not this step asks for them, and an unasked-for download is
|
||||
# the one nobody can find in the log.
|
||||
- name: Install Rust 1.92.0
|
||||
run: |
|
||||
set -e
|
||||
curl -fsSL https://sh.rustup.rs | sh -s -- \
|
||||
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
|
||||
--component rust-analyzer
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
|
||||
# ARCH §6.5a: no core/ crate may depend on the UI toolkit. One stray
|
||||
# `use slint::` costs headless golden-image testing and the
|
||||
# one-operation-two-presentations property together, and nothing else
|
||||
# would notice.
|
||||
- name: Core crates must not depend on the UI
|
||||
run: |
|
||||
set -e
|
||||
FAILED=0
|
||||
for crate in dr-types dr-gpu dr-sync; do
|
||||
if cargo tree -p "$crate" -e normal 2>/dev/null | grep -qE '\bslint\b|\bi-slint'; then
|
||||
echo "FAIL: $crate depends on Slint (ARCH §6.5a)"
|
||||
FAILED=1
|
||||
else
|
||||
echo "ok: $crate"
|
||||
fi
|
||||
done
|
||||
exit $FAILED
|
||||
@@ -1,130 +0,0 @@
|
||||
name: Traceability
|
||||
|
||||
# Mirrors JellyTau's traceability gate, including the reason it exists.
|
||||
#
|
||||
# That gate divided a traced count by frozen literal denominators while the
|
||||
# requirements file grew past them, reported 158% coverage, and so could never
|
||||
# fail its own threshold. Two rules follow, and the extractor's own tests
|
||||
# enforce both:
|
||||
#
|
||||
# 1. Denominators are parsed from docs/requirements.md at run time.
|
||||
# 2. Coverage is |traced ∩ defined| / |defined|, never a raw traced count.
|
||||
#
|
||||
# This job is static analysis of source comments plus markdown parsing, so it
|
||||
# needs no GPU and no Android SDK — only the Rust toolchain.
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main, master, develop]
|
||||
pull_request:
|
||||
branches: [main, master, develop]
|
||||
|
||||
jobs:
|
||||
traceability:
|
||||
runs-on: linux/amd64
|
||||
name: Requirement traces
|
||||
# Node for actions/checkout and actions/cache, which the bare runner image
|
||||
# cannot execute. Rust is installed below.
|
||||
container:
|
||||
image: catthehacker/ubuntu:act-latest
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Cache cargo
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
target
|
||||
key: traces-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}
|
||||
|
||||
# Source-comment and markdown parsing only, so the minimal profile is
|
||||
# enough — no system libraries and nothing this job itself needs beyond
|
||||
# cargo. rust-analyzer is here anyway because rust-toolchain.toml lists
|
||||
# it: rustup installs that file's components on the first cargo call in
|
||||
# the work tree regardless, and a download named in the install step
|
||||
# beats the same download appearing unannounced inside the gate.
|
||||
- name: Install Rust 1.92.0
|
||||
run: |
|
||||
set -e
|
||||
curl -fsSL https://sh.rustup.rs | sh -s -- \
|
||||
-y --no-modify-path --profile minimal --default-toolchain 1.92.0 \
|
||||
--component rust-analyzer
|
||||
echo "$HOME/.cargo/bin" >> "$GITHUB_PATH"
|
||||
|
||||
# The gate's own arithmetic is the thing being trusted, so its tests run
|
||||
# before it does. Untested gate logic is exactly how JellyTau's 158% went
|
||||
# unnoticed for months.
|
||||
- name: Test the extractor
|
||||
run: cargo test -p traceability
|
||||
|
||||
# Structural failures are unconditional and do not depend on the coverage
|
||||
# threshold: zero requirements parsed, zero files scanned, a ratio above
|
||||
# 100%, or any orphan tag all fail the build. A misconfigured run must not
|
||||
# report a plausible-looking 0%.
|
||||
- name: Traceability gate
|
||||
run: cargo run -q -p traceability -- check
|
||||
|
||||
- name: Regenerate matrix and check it is committed
|
||||
run: |
|
||||
set -e
|
||||
cargo run -q -p traceability -- report
|
||||
if ! git diff --quiet docs/traceability.md; then
|
||||
echo ""
|
||||
echo "docs/traceability.md is out of date."
|
||||
echo "Run: cargo run -p traceability -- report"
|
||||
git diff --stat docs/traceability.md
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The gesture vocabulary, from the same scanner and under the same rule.
|
||||
#
|
||||
# Blocking, and for a sharper reason than the matrix: these two artefacts
|
||||
# are not only read, one of them is *shown to the user*. A stale
|
||||
# `gesture_book.rs` is a help sheet in the application telling somebody to
|
||||
# perform a gesture that was removed — worse than no help sheet, because
|
||||
# they will conclude the application is broken rather than the page.
|
||||
#
|
||||
# This also fails on a malformed tag, so a typo costs a gesture its
|
||||
# desktop half loudly rather than silently.
|
||||
- name: Regenerate the gesture vocabulary and check it is committed
|
||||
run: cargo run -q -p traceability -- gestures-check
|
||||
|
||||
# Advisory, not blocking: not every file implements a requirement, and a
|
||||
# tag on every function is noise that rots faster than it helps. Tag the
|
||||
# unit that decides.
|
||||
- name: Check changed files for tags
|
||||
if: github.event_name == 'pull_request'
|
||||
run: |
|
||||
set -e
|
||||
CHANGED=$(git diff --name-only "origin/${{ github.base_ref }}...HEAD" \
|
||||
| grep -E '\.(rs|slint|wgsl)$' || true)
|
||||
[ -z "$CHANGED" ] && { echo "No source files changed."; exit 0; }
|
||||
|
||||
MISSING=0
|
||||
for file in $CHANGED; do
|
||||
case "$file" in
|
||||
*/tests/*|*/test_*|tools/*) continue ;;
|
||||
esac
|
||||
[ -f "$file" ] || continue
|
||||
if ! grep -q 'TRACES:' "$file"; then
|
||||
echo " no TRACES tag: $file"
|
||||
MISSING=$((MISSING + 1))
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$MISSING" -gt 0 ]; then
|
||||
echo ""
|
||||
echo "$MISSING changed file(s) carry no requirement tag."
|
||||
echo "Format: /// TRACES: FR-CAT-1, FR-CAT-2 | NFR-P1"
|
||||
echo " (comma separates IDs, pipe groups types)"
|
||||
fi
|
||||
|
||||
- name: Summary
|
||||
if: always()
|
||||
run: head -30 docs/traceability.md || true
|
||||
@@ -1,3 +0,0 @@
|
||||
#!/bin/sh
|
||||
command -v git-lfs >/dev/null 2>&1 || { printf >&2 "\n%s\n\n" "This repository is configured for Git LFS but 'git-lfs' was not found on your path. If you no longer wish to use Git LFS, remove this hook by deleting the 'post-checkout' file in the hooks directory (set by 'core.hookspath'; usually '.git/hooks')."; exit 2; }
|
||||
git lfs post-checkout "$@"
|
||||
@@ -1,3 +0,0 @@
|
||||
#!/bin/sh
|
||||
command -v git-lfs >/dev/null 2>&1 || { printf >&2 "\n%s\n\n" "This repository is configured for Git LFS but 'git-lfs' was not found on your path. If you no longer wish to use Git LFS, remove this hook by deleting the 'post-commit' file in the hooks directory (set by 'core.hookspath'; usually '.git/hooks')."; exit 2; }
|
||||
git lfs post-commit "$@"
|
||||
@@ -1,3 +0,0 @@
|
||||
#!/bin/sh
|
||||
command -v git-lfs >/dev/null 2>&1 || { printf >&2 "\n%s\n\n" "This repository is configured for Git LFS but 'git-lfs' was not found on your path. If you no longer wish to use Git LFS, remove this hook by deleting the 'post-merge' file in the hooks directory (set by 'core.hookspath'; usually '.git/hooks')."; exit 2; }
|
||||
git lfs post-merge "$@"
|
||||
@@ -1,67 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Keep the generated artefacts in step with the tags in the tree.
|
||||
#
|
||||
# Two of them now, from the same scanner: the requirements matrix and the
|
||||
# gesture vocabulary. Both are generated *from* the tree and cite line numbers
|
||||
# in it, so both go stale on any commit that moves a line — a `cargo fmt` sweep
|
||||
# above all, but equally a commit that merely adds a paragraph above a tag.
|
||||
#
|
||||
# The gate regenerates the matrix in CI and fails if the result differs from
|
||||
# what is committed. That is the right check — a matrix that disagrees with the
|
||||
# tree is worse than none, because it is read as current — but it fails *after*
|
||||
# a push, on a commit that is otherwise fine, and it has now done so on six
|
||||
# commits in a row because adding a `TRACES:` tag and regenerating the matrix
|
||||
# are two actions and only the first is on anyone's mind.
|
||||
#
|
||||
# So it happens here instead, where the tags are being changed.
|
||||
#
|
||||
# Only when something that can carry a tag is staged: a commit touching
|
||||
# workflows, packaging or the matrix itself pays nothing.
|
||||
set -euo pipefail
|
||||
|
||||
staged="$(git diff --cached --name-only --diff-filter=ACMR)"
|
||||
if ! grep -qE '\.(rs|slint|yaml|md)$' <<< "${staged}"; then
|
||||
exit 0
|
||||
fi
|
||||
# The artefacts are generated from the tree, so regenerating them because one
|
||||
# was itself edited would be circular.
|
||||
case "$(tr -d '[:space:]' <<< "${staged}")" in
|
||||
docs/traceability.md | docs/gestures.md | ui/dr-ui/src/gesture_book.rs)
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
|
||||
repo="$(git rev-parse --show-toplevel)"
|
||||
cd "${repo}"
|
||||
|
||||
# Quiet unless it has something to say. A hook that prints on every commit is
|
||||
# a hook people start passing --no-verify to.
|
||||
if ! cargo run -q -p traceability -- report >/dev/null 2>&1; then
|
||||
echo "pre-commit: could not run the traceability report; leaving the matrix alone" >&2
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if ! git diff --quiet -- docs/traceability.md; then
|
||||
git add docs/traceability.md
|
||||
echo "pre-commit: regenerated docs/traceability.md and staged it"
|
||||
fi
|
||||
|
||||
# The gesture vocabulary, same discipline.
|
||||
#
|
||||
# **Failure here is reported and not swallowed**, unlike the matrix above. A
|
||||
# matrix that will not build leaves the previous one in place, which is merely
|
||||
# stale; a malformed `GESTURE:` block means a gesture the user is about to be
|
||||
# told about in the wrong words, or not at all. The gate would catch it in CI
|
||||
# either way — this is only about catching it a push earlier.
|
||||
if ! out="$(cargo run -q -p traceability -- gestures 2>&1)"; then
|
||||
echo "pre-commit: the gesture scan failed — the tags below need fixing" >&2
|
||||
echo "${out}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
for f in docs/gestures.md ui/dr-ui/src/gesture_book.rs; do
|
||||
if ! git diff --quiet -- "${f}"; then
|
||||
git add "${f}"
|
||||
echo "pre-commit: regenerated ${f} and staged it"
|
||||
fi
|
||||
done
|
||||
@@ -1,3 +0,0 @@
|
||||
#!/bin/sh
|
||||
command -v git-lfs >/dev/null 2>&1 || { printf >&2 "\n%s\n\n" "This repository is configured for Git LFS but 'git-lfs' was not found on your path. If you no longer wish to use Git LFS, remove this hook by deleting the 'pre-push' file in the hooks directory (set by 'core.hookspath'; usually '.git/hooks')."; exit 2; }
|
||||
git lfs pre-push "$@"
|
||||
@@ -1,25 +0,0 @@
|
||||
/target
|
||||
/target-android
|
||||
Cargo.lock.bak
|
||||
*.log
|
||||
|
||||
# makepkg build products. `packaging/PKGBUILD` and the .desktop entry are
|
||||
# sources and belong in the tree; everything makepkg derives from them does
|
||||
# not — `pkg/` and `src/` are staging directories it recreates on every run,
|
||||
# and the package itself is 33 MB of compiled output.
|
||||
/packaging/pkg/
|
||||
/packaging/src/
|
||||
/packaging/*.pkg.tar.*
|
||||
/packaging/*.log
|
||||
|
||||
# Cached upstream film profiles, re-fetchable with
|
||||
# tools/film-profiles/convert.py --fetch. Not source: the converted
|
||||
# profiles in core/dr-film/profiles are.
|
||||
tools/film-profiles/upstream/
|
||||
|
||||
# flatpak-builder's cache and its output tree. `packaging/flatpak/` holds the
|
||||
# manifest, which is source; everything a build derives from it is not — and
|
||||
# `.flatpak-builder/` in particular caches an unpacked copy of the whole
|
||||
# checkout, so it is larger than the repository it sits in.
|
||||
/.flatpak-builder/
|
||||
/build/
|
||||
@@ -1,180 +0,0 @@
|
||||
# Contributing to DarkRoom
|
||||
|
||||
There is a lot of documentation here — 14 documents and 177 numbered
|
||||
requirements — and almost all of it is written for someone who has already
|
||||
decided to work on this. This file is the other thing: how to get a first
|
||||
change landed without reading any of it.
|
||||
|
||||
## The shortest useful contribution
|
||||
|
||||
**A develop operation is one file.** Not one file plus a registration, plus a
|
||||
shader edit, plus a control in the UI — one file:
|
||||
|
||||
```
|
||||
core/dr-pipeline/ops/split_toning.yaml
|
||||
```
|
||||
|
||||
`build.rs` finds it with `read_dir`, compiles it into Rust implementing
|
||||
`Operation`, and from there it is indistinguishable from a hand-written node.
|
||||
It arrives with controls built from its declared parameter kinds, a place in
|
||||
the chain from `order:`, a place in the panel from `attributes:`, sidecar
|
||||
persistence, and its own tests — which are declared in the same file and run
|
||||
under `cargo test`.
|
||||
|
||||
Read [`core/dr-pipeline/ops/README.md`](core/dr-pipeline/ops/README.md) and
|
||||
copy [`exposure.yaml`](core/dr-pipeline/ops/exposure.yaml). Split toning,
|
||||
colour zones, selective colour and channel-mixer variants are all pure point
|
||||
operations, which means all of them are declarations rather than code.
|
||||
|
||||
If you want to understand one thing about the architecture before starting,
|
||||
make it this: **the core describes its capabilities and the interface composes
|
||||
them.** No code in `ui/` names an operation, and a test enforces that
|
||||
(`ui/dr-ui/tests/ui_names_no_operation.rs`). It is why your node needs no UI
|
||||
change.
|
||||
|
||||
## Getting it to build
|
||||
|
||||
**Git LFS is required.** Model weights are stored in LFS, and a clone made
|
||||
without it leaves a ~130-byte text pointer where an 11 MB model should be:
|
||||
|
||||
```bash
|
||||
git lfs install && git lfs pull
|
||||
```
|
||||
|
||||
Forget this and `dr-segment`'s build script stops with an instruction rather
|
||||
than embedding the pointer and failing at inference time — but it is easier to
|
||||
run the two commands now.
|
||||
|
||||
**The toolchain pins itself.** `rust-toolchain.toml` selects 1.92.0 and rustup
|
||||
fetches it on first use. Do not override it; `cargo fmt` and `clippy` are both
|
||||
version-sensitive and CI runs exactly this version.
|
||||
|
||||
**System packages.** Slint and winit need these at build time. On Debian or
|
||||
Ubuntu:
|
||||
|
||||
```bash
|
||||
sudo apt-get install pkg-config libfontconfig1-dev libxkbcommon-dev
|
||||
```
|
||||
|
||||
**Then:**
|
||||
|
||||
```bash
|
||||
cargo run -p darkroom-desktop
|
||||
```
|
||||
|
||||
The first build resolves 826 crates and takes a while — on a laptop, long
|
||||
enough to look like a hang. It is not one.
|
||||
|
||||
Android is a containerised toolchain and is not needed for most work; see
|
||||
[`docker/android/README.md`](docker/android/README.md) if you get there.
|
||||
|
||||
## What CI will check
|
||||
|
||||
All four of these run on every push, so run them before you send anything:
|
||||
|
||||
```bash
|
||||
cargo fmt --all -- --check
|
||||
cargo clippy --workspace --all-targets -- -D warnings
|
||||
cargo test --workspace
|
||||
cargo build --workspace --release
|
||||
```
|
||||
|
||||
GPU tests skip themselves where there is no adapter rather than failing — a
|
||||
test that cannot run is not evidence either way — so a green run on a machine
|
||||
without a GPU is expected, and does not mean the GPU paths were exercised.
|
||||
|
||||
There is a fifth check, and it is not in that list because you are unlikely to
|
||||
break it by accident:
|
||||
|
||||
```bash
|
||||
cargo run --release -p dr-bench -- check
|
||||
```
|
||||
|
||||
That is the benchmark suite (`docs/requirements.md` §8), which builds a
|
||||
synthetic 50,000-image catalog and fails the build if a performance target is
|
||||
missed or a measurement has drifted past its tolerance. It runs on every push in
|
||||
its own workflow. [`docs/benchmarks.md`](docs/benchmarks.md) says what it
|
||||
measures, what it deliberately does not, and how to read a failure. If you have
|
||||
touched the catalog, the decoder, the thumbnail store or the exporter, run it
|
||||
before you send.
|
||||
|
||||
## Requirements and traceability
|
||||
|
||||
[`requirements.md`](docs/requirements.md) is the register of record.
|
||||
[`traceability.md`](docs/traceability.md) is generated from `TRACES:` tags in
|
||||
the source and must never be hand-edited:
|
||||
|
||||
```rust
|
||||
// TRACES: FR-DEV-3a | FR-DEV-3c
|
||||
```
|
||||
|
||||
Tags are read from `.rs`, `.slint`, `.wgsl` and `.yaml` — the last so a
|
||||
declared operation can record the requirement it satisfies, since the Rust it
|
||||
generates lands in `OUT_DIR` and is not scanned.
|
||||
|
||||
A pre-commit hook regenerates the matrix and stages it whenever you touch
|
||||
something that can carry a tag, so you should not have to think about it. If
|
||||
you do need to run it by hand:
|
||||
|
||||
```bash
|
||||
cargo run -p traceability -- report
|
||||
```
|
||||
|
||||
Note that it tracks line numbers, so a change that only moves code still moves
|
||||
the matrix. Never regenerate it with a stale prebuilt binary.
|
||||
|
||||
**One convention that the tooling cannot enforce.** A tag proves that a tag
|
||||
exists, not that the code under it does the thing — `docs/code-health.md`
|
||||
CH-4 has the details, and two requirements currently read as covered on the
|
||||
strength of plumbing a future feature would use. So: **close a requirement
|
||||
with a test that would fail if the behaviour were removed.** Coverage that
|
||||
moves slowly and means something beats coverage that moves quickly.
|
||||
|
||||
## Two invariants the build defends
|
||||
|
||||
Worth knowing before you trip one, because both failures name a requirement
|
||||
rather than a line:
|
||||
|
||||
- **No operation may be named in `ui/`** (FR-DEV-3a). Special-casing one
|
||||
operation in the panel to fix a layout problem is how a generated interface
|
||||
stops being generated. If a node needs presentation the panel cannot give it,
|
||||
the answer is a `presentation:` hint in the declaration and a `WidgetKind`,
|
||||
not a branch in `develop.rs`.
|
||||
- **The operation schema rejects ambiguity at build time**: a duplicate
|
||||
`order:`, a filename disagreeing with its `id:`, a default outside its own
|
||||
range, an expression naming something that is not a parameter. Each error
|
||||
names the key you got wrong and exits rather than panicking.
|
||||
|
||||
## Commit messages
|
||||
|
||||
Imperative subject describing the change from the reader's side — "Offer the
|
||||
merge when two people turn out to share a name", not "fix: merge dialog". No
|
||||
conventional-commits prefixes.
|
||||
|
||||
The body is where the reasoning goes, and it is expected to be substantial when
|
||||
the change is. This codebase records *why* far more than most, in commits and
|
||||
in comments alike, and that is the single habit most worth adopting: the
|
||||
constraint you worked around is invisible to whoever reads the diff next.
|
||||
|
||||
One commit per change. If you fixed two things, that is two commits.
|
||||
|
||||
## Where to read next, in order
|
||||
|
||||
| Document | Read it when |
|
||||
|---|---|
|
||||
| [`core/dr-pipeline/ops/README.md`](core/dr-pipeline/ops/README.md) | Adding or changing a develop operation — start here regardless |
|
||||
| [`docs/architecture.md`](docs/architecture.md) | Anything touching the render path, catalog or sync |
|
||||
| [`docs/code-health.md`](docs/code-health.md) | Deciding what to work on; grades each seam by what it costs |
|
||||
| [`docs/benchmarks.md`](docs/benchmarks.md) | A change that could plausibly cost time or memory |
|
||||
| [`docs/technical-debt.md`](docs/technical-debt.md) | Something looks wrong — check it was not chosen |
|
||||
| [`docs/distribution.md`](docs/distribution.md) | Packaging a build, or adding a permission to one |
|
||||
| [`docs/requirements.md`](docs/requirements.md) | Reference, not reading |
|
||||
|
||||
`technical-debt.md` is the one to check before "fixing" anything surprising.
|
||||
It records compromises that were deliberate, each with the reasoning and a
|
||||
falsifiable condition for when it stops being one — the point being that you
|
||||
can tell a constraint from an accident without asking.
|
||||
|
||||
## Licence
|
||||
|
||||
GPL-3.0-or-later. By contributing you agree your work is licensed the same way.
|
||||
@@ -1,257 +0,0 @@
|
||||
[workspace]
|
||||
resolver = "2"
|
||||
members = [
|
||||
"core/dr-types",
|
||||
"core/dr-catalog",
|
||||
"core/dr-thumbs",
|
||||
"core/dr-decode",
|
||||
"core/dr-export",
|
||||
"core/dr-face",
|
||||
"core/dr-film",
|
||||
"core/dr-ingest",
|
||||
"core/dr-gpu",
|
||||
"core/dr-lens",
|
||||
"core/dr-pipeline",
|
||||
"core/dr-preset-xmp",
|
||||
"core/dr-segment",
|
||||
"core/dr-sync",
|
||||
"core/dr-sync-folder",
|
||||
"core/dr-sync-nextcloud",
|
||||
"core/dr-xmp",
|
||||
"platform/dr-plat",
|
||||
"ui/dr-ui",
|
||||
"apps/darkroom-desktop",
|
||||
"apps/darkroom-android",
|
||||
"tools/bench",
|
||||
"tools/traceability",
|
||||
]
|
||||
|
||||
[workspace.package]
|
||||
version = "0.12.0"
|
||||
edition = "2021"
|
||||
rust-version = "1.92"
|
||||
license = "GPL-3.0-or-later"
|
||||
repository = "https://github.com/dtourolle/DarkRoom"
|
||||
|
||||
[workspace.dependencies]
|
||||
# Internal
|
||||
dr-types = { path = "core/dr-types" }
|
||||
dr-catalog = { path = "core/dr-catalog" }
|
||||
dr-thumbs = { path = "core/dr-thumbs" }
|
||||
dr-decode = { path = "core/dr-decode" }
|
||||
dr-export = { path = "core/dr-export" }
|
||||
# Stated explicitly for the same reason as `dr-segment` below: no dependant
|
||||
# should drag in an ONNX runtime by accident. Members opt in with
|
||||
# `features = ["inference"]`.
|
||||
dr-face = { path = "core/dr-face", default-features = false }
|
||||
dr-film = { path = "core/dr-film" }
|
||||
dr-ingest = { path = "core/dr-ingest" }
|
||||
dr-gpu = { path = "core/dr-gpu" }
|
||||
dr-lens = { path = "core/dr-lens" }
|
||||
dr-pipeline = { path = "core/dr-pipeline" }
|
||||
dr-preset-xmp = { path = "core/dr-preset-xmp" }
|
||||
# `default-features = false` belongs *here*, not on each dependant: a member
|
||||
# inheriting a workspace dependency cannot turn its default features off, so
|
||||
# writing it below would silently do nothing and every crate touching
|
||||
# `dr-segment` would drag in tract and 11 MB of weights. Members opt in with
|
||||
# `features = ["semantic", "embedded-model"]` instead.
|
||||
dr-segment = { path = "core/dr-segment", default-features = false }
|
||||
dr-plat = { path = "platform/dr-plat" }
|
||||
dr-sync = { path = "core/dr-sync" }
|
||||
dr-sync-folder = { path = "core/dr-sync-folder" }
|
||||
dr-sync-nextcloud = { path = "core/dr-sync-nextcloud" }
|
||||
dr-xmp = { path = "core/dr-xmp" }
|
||||
dr-ui = { path = "ui/dr-ui" }
|
||||
|
||||
# GPU + UI
|
||||
#
|
||||
# The wgpu version is not a free choice: it is dictated by Slint. Importing a
|
||||
# texture into the scene (ARCH §6.1, spike S1) requires it to come from the
|
||||
# *same* `wgpu::Device` Slint renders with, and Slint will only hand out a
|
||||
# device of the version it was compiled against. Slint 1.17 offers
|
||||
# `unstable-wgpu-28` and `unstable-wgpu-29` and nothing older, so 29 it is —
|
||||
# pinned to the same `29.0.4` floor Slint itself requires, because two
|
||||
# semver-compatible-but-different wgpu crates in one tree are two *types*, and
|
||||
# the device would not typecheck across them.
|
||||
#
|
||||
# Consequently: bumping Slint may force a wgpu bump, and wgpu cannot be bumped
|
||||
# on its own. They move together or not at all.
|
||||
wgpu = "29.0.4"
|
||||
slint = { version = "1.17", default-features = false }
|
||||
slint-build = "1.17"
|
||||
|
||||
# UI token codegen (S2): style.yaml -> theme.slint. serde_yaml was deprecated
|
||||
# by its maintainer in 2024 and serde_yml, the first fork, has since been
|
||||
# deprecated too; serde_norway is the fork still receiving releases. Its
|
||||
# mappings preserve insertion order, which is what lets the generated Slint
|
||||
# keep the token ordering the YAML author chose.
|
||||
serde_norway = "0.9"
|
||||
|
||||
# Foundations
|
||||
anyhow = "1"
|
||||
thiserror = "2"
|
||||
log = "0.4"
|
||||
env_logger = "0.11"
|
||||
pollster = "0.4"
|
||||
|
||||
# Networking — no mature Nextcloud crate exists; the connector is hand-rolled
|
||||
# over reqwest (D7). reqwest_dav was evaluated and is too thin to build on.
|
||||
# `rustls-no-provider` rather than `rustls`: the latter defaults to the
|
||||
# aws-lc-rs crypto provider, whose aws-lc-sys crate is C and fails to
|
||||
# cross-compile for Android — precisely the NDK pain D1 chose Rust to avoid.
|
||||
# ring is pure Rust apart from a small asm core that does build under the NDK.
|
||||
#
|
||||
# `rustls-tls-webpki-roots-no-provider` rather than plain `rustls-no-provider`:
|
||||
# the latter verifies against rustls-platform-verifier, which reaches the
|
||||
# Android trust store over JNI and panics mid-handshake unless initialised from
|
||||
# Java first — the crash D7 predicted and spike S3 exists to resolve properly.
|
||||
# The panic surfaces inside tokio, which catches task panics itself, so it
|
||||
# reaches the UI as a worker that stopped rather than as an error.
|
||||
#
|
||||
# webpki-roots is the escape hatch D7 records: a root store compiled into the
|
||||
# binary, no JNI, identical on both platforms. The trade is real and belongs in
|
||||
# S3's scope — user-installed and enterprise CAs are not consulted, and the
|
||||
# roots go stale with the release rather than with the OS.
|
||||
reqwest = { version = "0.13", default-features = false, features = ["rustls-no-provider", "webpki-roots", "stream", "json"] }
|
||||
rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] }
|
||||
quick-xml = "0.41"
|
||||
tokio = { version = "1", features = ["rt-multi-thread", "macros", "sync", "time"] }
|
||||
url = "2.5"
|
||||
async-trait = "0.1"
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
base64 = "0.23"
|
||||
|
||||
# Display-server clients, for FR-DSP-8's per-display profile acquisition.
|
||||
#
|
||||
# Neither is a new cost: winit already builds both, so the versions are the
|
||||
# ones Slint's backend has resolved to and pinning anything else here would
|
||||
# compile a second copy. Both are pure Rust — x11rb speaks the X11 wire
|
||||
# protocol itself rather than binding libxcb, and wayland-client binds
|
||||
# libwayland only under a feature that is off — which keeps the Android
|
||||
# cross-compile a plain Rust dependency graph, the same criterion as the TLS
|
||||
# and SQLite choices above. They are declared under a target predicate that
|
||||
# excludes Android, where neither display server exists.
|
||||
#
|
||||
# `staging` on wayland-protocols is what carries `wp_color_manager_v1`: the
|
||||
# colour-management extension is still staging upstream, which is the
|
||||
# protocol-level statement of the thing FR-DSP-8 anticipates when it says
|
||||
# Wayland's colour management "is not universally available".
|
||||
x11rb = { version = "0.13", features = ["randr"] }
|
||||
wayland-client = "0.31"
|
||||
wayland-protocols = { version = "0.32", features = ["client", "staging"] }
|
||||
|
||||
# Platform secure storage: Secret Service on Linux, Keystore on Android
|
||||
# (FR-NC-2). Credentials never touch the catalog or a plain file.
|
||||
# keyring 4 restructured its features: `v1` is the default set and brings
|
||||
# the zbus Secret Service backend, which is what GNOME Keyring and KWallet
|
||||
# (via ksecretd) both speak.
|
||||
keyring = { version = "4", features = ["v1"] }
|
||||
|
||||
# The Android half of the same project: a keyring-core CredentialStore backed
|
||||
# by AndroidKeyStore AES-GCM over SharedPreferences (FR-PLAT-AND-1). It reads
|
||||
# the JavaVM and Context from ndk-context, which android-activity populates
|
||||
# before `android_main` runs, so no Kotlin shim of our own is needed.
|
||||
#
|
||||
# This is the keyring-core API, not the v1 `Entry` API the Linux path uses;
|
||||
# the two impls are deliberately separate rather than sharing a code path.
|
||||
android-native-keyring-store = "1.0.0"
|
||||
keyring-core = "1"
|
||||
|
||||
# Decode. rawler is the pure-Rust decoder (D2); zune-jpeg decodes the
|
||||
# embedded previews rawler extracts.
|
||||
# Catalog. `bundled` compiles SQLite from source rather than linking the
|
||||
# system library — the same cross-compilation reasoning as the TLS choice
|
||||
# above: no system dependency to satisfy under the Android NDK.
|
||||
#
|
||||
# `backup` is not optional in practice: it is what takes a consistent snapshot
|
||||
# of a live WAL database for upload. A filesystem copy of `catalog.sqlite`
|
||||
# while a `-wal` exists beside it uploads a torn file.
|
||||
rusqlite = { version = "0.40", features = ["bundled", "backup"] }
|
||||
|
||||
rawler = "0.7"
|
||||
zune-jpeg = "0.4.21"
|
||||
# Thumbnails are stored encoded, not as raw RGBA: a 256px RGBA buffer is
|
||||
# ~256 KB against ~20 KB as JPEG, and the store syncs to Nextcloud where that
|
||||
# 13× is transfer cost on every client. Pure Rust, no C dependency — the same
|
||||
# criterion behind the TLS and SQLite choices above.
|
||||
jpeg-encoder = "0.7"
|
||||
bytemuck = { version = "1", features = ["derive"] }
|
||||
|
||||
# Lens correction profiles. A pure-Rust port of Lensfun rather than a binding
|
||||
# to the C library, for the same cross-compilation reason as the TLS and
|
||||
# SQLite choices above: liblensfun would be a third C dependency to satisfy
|
||||
# under the Android NDK.
|
||||
#
|
||||
# The database ships *inside* the crate — 56 XML files, gzipped at build time
|
||||
# and decompressed on first lookup. That matters beyond convenience: Android
|
||||
# gives us no filesystem path (ARCH §6.9), so a database loaded from a
|
||||
# system directory would have nowhere to live there.
|
||||
#
|
||||
# Licence: LGPL-3.0-or-later, which upgrades cleanly into our GPLv3 (D8).
|
||||
# The upstream Lensfun *database* is CC-BY-SA and is redistributed by the
|
||||
# crate; attribution belongs in the about screen.
|
||||
#
|
||||
# Caveat worth remembering: this is a third-party port at 0.7.0, not upstream
|
||||
# Lensfun. Verified working against the bundled database (interpolation
|
||||
# between calibration points, and an unknown lens returning empty rather than
|
||||
# panicking), but the pipeline talks to it through its own profile types so
|
||||
# swapping it out is not a pipeline change.
|
||||
lensfun = "0.7"
|
||||
|
||||
# Neural inference for semantic segmentation (S15 arm B, D14).
|
||||
#
|
||||
# D13 framed this as a choice between `ort` (fast, best operator coverage, and
|
||||
# a C++ dependency to cross-compile under the NDK) and a pure-Rust runtime
|
||||
# (policy-compliant, unproven coverage). That framing turned out to be a false
|
||||
# choice: `ort` 2.0's `alternative-backend` feature *disables the linking
|
||||
# entirely* and lets a different engine supply the `OrtApi`, and `ort-tract` —
|
||||
# same authors, MIT/Apache — supplies it from `tract`, which is pure Rust.
|
||||
#
|
||||
# So we get `ort`'s API with no C at all. `download-binaries` and `tls-native`
|
||||
# are off with `default-features = false`, which is the point: nothing is
|
||||
# fetched at build time and nothing is linked, so the Android cross-compile
|
||||
# sees an ordinary Rust dependency graph. That is the same reasoning as rustls
|
||||
# over aws-lc-rs and bundled SQLite, applied to inference — D13's largest
|
||||
# tolerated exception turns out not to be needed.
|
||||
#
|
||||
# The trade is real and belongs on the record: tract is slower than the C++
|
||||
# runtime and covers fewer operators. Both were measured rather than assumed
|
||||
# before this landed — yolo26n-seg loads with **zero unsupported operators**
|
||||
# and runs 640x640 in ~470 ms on the reference desktop's CPU. That is fine for
|
||||
# a once-per-image precompute off the frame path (ARCH §6.1) and would not be
|
||||
# fine for anything per-frame, which is a constraint on what may be built on
|
||||
# top rather than on this choice.
|
||||
#
|
||||
# Pinned to an rc: `ort` 2.0 has been in rc for a long while and `ort-tract`
|
||||
# exists only against it. Worth revisiting at 2.0 final.
|
||||
ort = { version = "2.0.0-rc.13", default-features = false, features = ["alternative-backend", "ndarray", "std"] }
|
||||
ort-tract = "0.4"
|
||||
# Not a free choice: it is the version `ort` exposes its tensors through, so
|
||||
# two semver-incompatible ndarrays would not typecheck across the boundary —
|
||||
# the same coupling wgpu has with Slint above.
|
||||
ndarray = "0.17"
|
||||
|
||||
[profile.dev]
|
||||
# Dev builds are tuned for how fast they *compile*, not for how fast they run.
|
||||
# Optimisation is a release concern; `[profile.release]` below is where it
|
||||
# belongs.
|
||||
#
|
||||
# This deliberately reverses an earlier choice. Dependencies used to be built
|
||||
# at `opt-level = 2` here, because wgpu and image decoding are slow without it.
|
||||
# That is still true, and it is the price: a debug run of the app, and the
|
||||
# decode- and GPU-heavy tests, are slower than they were. What it buys is that
|
||||
# nothing has to be optimised before it can be compiled — which is the cost
|
||||
# paid on every edit, by every worktree, rather than only when something is
|
||||
# actually run.
|
||||
#
|
||||
# If a particular crate turns out to be the one that makes a test unbearable,
|
||||
# raise it alone rather than restoring the blanket rule:
|
||||
#
|
||||
# [profile.dev.package.zune-jpeg]
|
||||
# opt-level = 2
|
||||
opt-level = 0
|
||||
|
||||
[profile.release]
|
||||
lto = "thin"
|
||||
codegen-units = 1
|
||||
@@ -1,84 +0,0 @@
|
||||
# DarkRoom
|
||||
|
||||
A cross-platform, non-destructive RAW photo editor for Linux and Android.
|
||||
|
||||
**Status:** 0.9.0, and no longer a spike. A library opens, culls, develops and
|
||||
exports on both platforms, across eight tagged releases. What is *not*
|
||||
built is written down rather than merely absent — see
|
||||
[docs/outstanding.md](docs/outstanding.md) for the requirements that have no
|
||||
implementation and why, and [docs/technical-debt.md](docs/technical-debt.md)
|
||||
for the compromises that were chosen.
|
||||
|
||||
## Documentation
|
||||
|
||||
| Document | Contents |
|
||||
|---|---|
|
||||
| [CONTRIBUTING.md](CONTRIBUTING.md) | How to land a first change without reading the rest |
|
||||
| [requirements.md](docs/requirements.md) | What the software must do — 179 numbered requirements |
|
||||
| [architecture.md](docs/architecture.md) | How it is built — crates, GPU pipeline, data model, sync |
|
||||
| [technical-debt.md](docs/technical-debt.md) | Compromises taken deliberately, each with the condition that retires it |
|
||||
| [outstanding.md](docs/outstanding.md) | What is not built, and whether that is a decision or a gap |
|
||||
| [code-health.md](docs/code-health.md) | What a contribution costs, per seam, measured |
|
||||
| [traceability.md](docs/traceability.md) | Generated: which requirement is claimed by which file |
|
||||
| [faces.md](docs/faces.md) | Face detection and identity — the models, the licence problem, and what S14 measured |
|
||||
|
||||
## Building
|
||||
|
||||
Desktop:
|
||||
|
||||
```bash
|
||||
cargo run -p darkroom-desktop
|
||||
```
|
||||
|
||||
Android (containerised toolchain, see [docker/android](docker/android/README.md)):
|
||||
|
||||
```bash
|
||||
./docker/android/build.sh cargo ndk -t arm64-v8a build --release
|
||||
```
|
||||
|
||||
Git LFS is required for the model weights, and the toolchain pins itself.
|
||||
[CONTRIBUTING.md](CONTRIBUTING.md) has the details and the four commands CI
|
||||
will run against what you send.
|
||||
|
||||
## Current state
|
||||
|
||||
**Working.** A catalog over a local folder, a Nextcloud account, or a folder a
|
||||
sync client keeps in virtual-files mode — where a placeholder is treated as the
|
||||
photograph rather than as a one-byte file. A virtualised library grid with a
|
||||
capture-time timeline, ratings, labels, keywords, collections and a trash that
|
||||
survives a crash mid-operation. Card ingest. Face detection and identity, with
|
||||
the index syncing between devices. A develop pipeline of fifteen declared
|
||||
operations fused into a single compute dispatch, plus the neighbourhood
|
||||
operations that cannot be — clarity, texture, capture sharpening, noise
|
||||
reduction, lens correction, spectral film simulation. Crop, straighten, spot
|
||||
removal, gradient and subject-segmentation masks, named presets, and a
|
||||
generated panel that no operation in `ui/` is allowed to name. Export to JPEG,
|
||||
PNG and 8- or 16-bit TIFF with resize and output sharpening.
|
||||
|
||||
**The zero-copy display path works on desktop.** The compute pass writes a
|
||||
texture that Slint composites directly, which is what
|
||||
[ARCH §6.1](docs/architecture.md) requires; the readback it forbids costs 96%
|
||||
of frame time at 4K, and
|
||||
|
||||
```bash
|
||||
cargo run -p dr-gpu --example bench --features readback
|
||||
```
|
||||
|
||||
still reproduces that measurement. **The one exception is the Android develop
|
||||
view**, which reads the frame back through the CPU because zero-copy there
|
||||
needs wgpu's Vulkan swapchain, and that tears a portrait window on a tablet
|
||||
whose panel is mounted landscape. It is debt, not a revision of the rule: the
|
||||
reasoning, the on-device measurements that forced it, and the three separate
|
||||
things any one of which would remove it are in
|
||||
[technical-debt.md TD-1](docs/technical-debt.md).
|
||||
|
||||
**Not built.** Plugins, compare and survey culling, focus peaking, burst
|
||||
grouping, AI denoise, tiled and progressive rendering, and most of the Android
|
||||
platform integration beyond running. The performance targets in §4.1 are
|
||||
unverified rather than unmet — the per-commit benchmark suite §8 requires does
|
||||
not exist, so nothing fails a build on a regression.
|
||||
[docs/outstanding.md](docs/outstanding.md) is the list, with the reasoning.
|
||||
|
||||
## Licence
|
||||
|
||||
GPL-3.0-or-later.
|
||||
@@ -1,54 +0,0 @@
|
||||
[package]
|
||||
name = "darkroom-android"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
license.workspace = true
|
||||
|
||||
# A cdylib, not a bin: Android loads the app as a shared library and calls
|
||||
# `android_main` through android-activity's glue. Nothing execs a binary, so
|
||||
# there is no `main` to provide.
|
||||
[lib]
|
||||
name = "darkroom"
|
||||
crate-type = ["cdylib"]
|
||||
|
||||
[dependencies]
|
||||
# No backend feature to select: dr-ui picks its Slint backend from the target,
|
||||
# so building for aarch64-linux-android gets android-activity automatically.
|
||||
dr-ui.workspace = true
|
||||
# For `account::set_data_dir`: only the platform entry point knows where Android
|
||||
# lets this app keep files, and it must be set before any store is opened.
|
||||
dr-sync.workspace = true
|
||||
# For the panic hook, for `state::set_state_dir` and for
|
||||
# `diagnostics::install`. Android has no XDG directories, so the entry point is
|
||||
# the only place that knows where a crash record or a log file may be written,
|
||||
# and both have to be in place before anything can fail.
|
||||
dr-plat.workspace = true
|
||||
# Directly, not just through dr-ui: `android_main` takes an `AndroidApp` and
|
||||
# calls `slint::android::init`, both of which come from this crate. The backend
|
||||
# feature comes from dr-ui's target-specific dependency.
|
||||
slint.workspace = true
|
||||
log.workspace = true
|
||||
android_logger = "0.15"
|
||||
|
||||
# The launch Intent and the share sheet are Java-only surfaces — see `intents`
|
||||
# — and JNI is the only way to reach them.
|
||||
#
|
||||
# Target-gated because the crate still has to compile on the host: it is a
|
||||
# workspace member, `cargo test --workspace` builds it, and the manifest tests
|
||||
# in `lib.rs` are the one part of it that runs there.
|
||||
#
|
||||
# 0.21 rather than the 0.22 that android-activity 0.6 uses. Both are already in
|
||||
# the lock — Slint's Android backend depends on two major versions of
|
||||
# android-activity and pulls both — so this adds nothing to the build either
|
||||
# way, and every object here comes from a raw pointer rather than from a type
|
||||
# android-activity handed over, so the two never have to agree.
|
||||
[target.'cfg(target_os = "android")'.dependencies]
|
||||
jni = "0.21"
|
||||
|
||||
[features]
|
||||
# Mirrors darkroom-desktop: the CPU readback path is gone since S1 landed
|
||||
# zero-copy. It mattered more here than on desktop — the same wrong path with
|
||||
# far less memory bandwidth to absorb it (ARCH §6.1) — but it is untested on a
|
||||
# device, since S1 was verified on desktop only.
|
||||
default = []
|
||||
@@ -1,171 +0,0 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<!--
|
||||
DarkRoom Android manifest.
|
||||
|
||||
Deliberately minimal: this packages the viewer for on-device testing (spike
|
||||
S2 needs Adreno and Mali hardware, which no emulator represents). Nothing
|
||||
here is a distribution manifest yet. Only network access is declared: file
|
||||
access needs no manifest permission because the library grid reads through
|
||||
SAF, which grants per-tree at runtime (ARCH §6.9).
|
||||
|
||||
Minimal is not the same as empty, and the entries below that are not the
|
||||
activity are the difference. A manifest is the only place a component can be
|
||||
declared: an intent filter is how the system learns this app is worth
|
||||
offering for a photograph, and a provider is how it learns the class exists
|
||||
at all. Neither can be moved into code (FR-PLAT-AND-6).
|
||||
-->
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
|
||||
package="paris.tourolle.darkroom">
|
||||
|
||||
<!-- Everything the app does with a server needs this: Login Flow v2, the
|
||||
WebDAV listing, thumbnail and image fetches. Without it Android refuses
|
||||
socket creation outright, and the failure is invisible — no panic to
|
||||
catch, no log line, just a worker thread that stops. Storage is the
|
||||
separate case that genuinely needs no permission here, because SAF
|
||||
grants per-tree at runtime (ARCH §6.9). -->
|
||||
<uses-permission android:name="android.permission.INTERNET" />
|
||||
<!-- Read before deciding whether a sync may run: FR-NC-6 gates background
|
||||
work on unmetered-and-charging, which means knowing the network type. -->
|
||||
<uses-permission android:name="android.permission.ACCESS_NETWORK_STATE" />
|
||||
|
||||
<!-- Vulkan 1.1 is what wgpu needs; the API 28 floor is where support is
|
||||
dependable (NFR-COMPAT-1). Marked required so an unsupported device
|
||||
fails at install rather than at first frame. -->
|
||||
<uses-feature
|
||||
android:name="android.hardware.vulkan.version"
|
||||
android:version="0x00401000"
|
||||
android:required="true" />
|
||||
|
||||
<!-- One name covers both icon generations, which is the point of the
|
||||
`anydpi-v26` qualifier: @mipmap/ic_launcher resolves to the adaptive
|
||||
icon at res/mipmap-anydpi-v26/ic_launcher.xml on API 26 and up, and to
|
||||
the density-matched ic_launcher.png below that. Since minSdk is 28 the
|
||||
PNGs are only ever reached by tooling, but they cost little and aapt2
|
||||
wants a real drawable behind the name. `roundIcon` is deliberately
|
||||
absent: it predates adaptive icons and a launcher that reads it would
|
||||
also be one that ignores the XML, which no device here is.
|
||||
|
||||
The adaptive icon has three layers rather than two. The third,
|
||||
monochrome, is what lets Android 13's themed-icon setting recolour it
|
||||
instead of dropping the app out of the themed set. -->
|
||||
<application
|
||||
android:label="DarkRoom"
|
||||
android:icon="@mipmap/ic_launcher"
|
||||
android:hasCode="true"
|
||||
android:allowBackup="false"
|
||||
android:supportsRtl="true">
|
||||
|
||||
<!-- NativeActivity rather than a Kotlin Activity: android-activity's
|
||||
glue loads libdarkroom.so and calls android_main. `android.app.lib_name`
|
||||
is how it learns which library to load, and must match [lib].name.
|
||||
|
||||
`singleTask` because a second instance of this activity is not
|
||||
survivable. The intent filters below mean another app can now
|
||||
launch it while it is already running, and under the default
|
||||
launch mode that starts a *second* NativeActivity — in the
|
||||
caller's task, in this same process, calling android_main again.
|
||||
Two Slint backends and two wgpu devices in one process is not a
|
||||
degraded experience, it is a failed second launch on top of a
|
||||
working first one.
|
||||
|
||||
What it costs, stated plainly: a share that arrives while DarkRoom
|
||||
is already running brings it forward without opening the image.
|
||||
The Intent goes to `onNewIntent`, and android-activity's event
|
||||
stream has no variant for it (MainEvent in 0.6 stops at Destroy),
|
||||
so nothing native ever sees it. Reading it would mean a Java
|
||||
Activity subclass forwarding it across JNI — the same shape of
|
||||
change FR-PLAT-AND-5 declined for onTrimMemory, and for the same
|
||||
reason. Launched from cold, which is the ordinary case for "open
|
||||
this photograph", the Intent is on getIntent() and is read. -->
|
||||
<activity
|
||||
android:name="android.app.NativeActivity"
|
||||
android:exported="true"
|
||||
android:launchMode="singleTask"
|
||||
android:configChanges="orientation|keyboardHidden|screenSize|screenLayout|density|uiMode"
|
||||
android:windowSoftInputMode="adjustResize">
|
||||
|
||||
<meta-data
|
||||
android:name="android.app.lib_name"
|
||||
android:value="darkroom" />
|
||||
|
||||
<intent-filter>
|
||||
<action android:name="android.intent.action.MAIN" />
|
||||
<category android:name="android.intent.category.LAUNCHER" />
|
||||
</intent-filter>
|
||||
|
||||
<!-- FR-PLAT-AND-6, inbound. The traceability tool reads .rs,
|
||||
.slint, .wgsl and .yaml, so this is a reference and not a
|
||||
tag; the tag that counts is on the test in lib.rs that
|
||||
asserts these declarations are still here.
|
||||
|
||||
Opening a photograph from a gallery, a file manager or a
|
||||
download. `android_main` reads the launch Intent through
|
||||
`Intents.receive` and the named images become the browsing
|
||||
list, exactly as paths on the desktop command line do.
|
||||
|
||||
`image/*` and not a wider match, even though it misses raws:
|
||||
a provider that does not recognise `.CR3` reports it as
|
||||
`application/octet-stream`, and claiming that type would put
|
||||
DarkRoom in the chooser for every unidentified binary on the
|
||||
device — an APK, a database, a partial download. Being absent
|
||||
from one gallery's menu is a smaller failure than being
|
||||
present in all of them. DNG, which providers do know as
|
||||
`image/x-adobe-dng`, matches here already.
|
||||
|
||||
BROWSABLE is what lets a browser's finished download and a
|
||||
link hand the file over; without it those routes silently do
|
||||
not list the app. -->
|
||||
<intent-filter>
|
||||
<action android:name="android.intent.action.VIEW" />
|
||||
<category android:name="android.intent.category.DEFAULT" />
|
||||
<category android:name="android.intent.category.BROWSABLE" />
|
||||
<data android:mimeType="image/*" />
|
||||
</intent-filter>
|
||||
|
||||
<!-- The share sheet, one photograph or a selection of them.
|
||||
SEND_MULTIPLE is declared because the sheet offers this app
|
||||
for a multi-selection only if it says it accepts one, and a
|
||||
culling tool that can be sent a single frame and not a burst
|
||||
is the wrong way round.
|
||||
|
||||
ACTION_EDIT is deliberately not here. It is a promise to write
|
||||
the result back to the URI it was handed, and nothing in this
|
||||
app does: an edit lands in a sidecar beside the original
|
||||
(FR-CAT-8). Registering for it would put DarkRoom in the "edit
|
||||
with" menu and lose the user's work every time. -->
|
||||
<intent-filter>
|
||||
<action android:name="android.intent.action.SEND" />
|
||||
<action android:name="android.intent.action.SEND_MULTIPLE" />
|
||||
<category android:name="android.intent.category.DEFAULT" />
|
||||
<data android:mimeType="image/*" />
|
||||
</intent-filter>
|
||||
</activity>
|
||||
|
||||
<!-- FR-PLAT-AND-6, outbound. Android has refused file:// URIs
|
||||
between apps since API 24 — handing one out raises
|
||||
FileUriExposedException in *this* process — so an exported JPEG
|
||||
reaches the share sheet as a content:// URI or not at all.
|
||||
|
||||
Not AndroidX's FileProvider: that is a Maven artefact, and this
|
||||
build has no Gradle and no dependency resolver (docker/android/
|
||||
README.md). ExportProvider does the same hundred lines against one
|
||||
fixed root.
|
||||
|
||||
`exported="false"` with `grantUriPermissions="true"` is the whole
|
||||
security model, and the two halves are not redundant. Exported
|
||||
false means no app may address the provider on its own account;
|
||||
the grant flag means a URI this app puts in an Intent carries a
|
||||
read permission for that one file, for the lifetime of the
|
||||
receiving task. Without the grant flag the share sheet opens and
|
||||
every target fails with SecurityException; with `exported="true"`
|
||||
instead, every app on the device could read the app's private
|
||||
directory. The authority must equal ExportProvider.AUTHORITY — a
|
||||
mismatch is a SecurityException in somebody else's app, so a test
|
||||
in lib.rs compares the two strings. -->
|
||||
<provider
|
||||
android:name="paris.tourolle.darkroom.ExportProvider"
|
||||
android:authorities="paris.tourolle.darkroom.exports"
|
||||
android:exported="false"
|
||||
android:grantUriPermissions="true" />
|
||||
</application>
|
||||
</manifest>
|
||||
@@ -1,260 +0,0 @@
|
||||
package paris.tourolle.darkroom;
|
||||
|
||||
import android.content.ContentProvider;
|
||||
import android.content.ContentValues;
|
||||
import android.content.Context;
|
||||
import android.database.Cursor;
|
||||
import android.database.MatrixCursor;
|
||||
import android.net.Uri;
|
||||
import android.os.ParcelFileDescriptor;
|
||||
import android.provider.OpenableColumns;
|
||||
import android.util.Log;
|
||||
import android.webkit.MimeTypeMap;
|
||||
|
||||
import java.io.File;
|
||||
import java.io.FileNotFoundException;
|
||||
import java.io.IOException;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
|
||||
/**
|
||||
* Hands an exported file to another app, and hands out nothing else.
|
||||
*
|
||||
* <p>FR-PLAT-AND-6's outbound half. Android has refused {@code file://} URIs
|
||||
* between apps since API 24 — passing one raises {@code FileUriExposedException}
|
||||
* in the *sending* process — so the only way to give a photo to the share sheet
|
||||
* is a {@code content://} URI backed by a provider, plus a per-Intent read
|
||||
* grant that expires with the task that received it.
|
||||
*
|
||||
* <h2>Why this is not AndroidX's FileProvider</h2>
|
||||
*
|
||||
* <p>Because AndroidX is a Maven artefact and this build has no Gradle and no
|
||||
* dependency resolver (see docker/android/README.md). Pulling in the one class
|
||||
* would mean adopting the whole mechanism that fetches it. What
|
||||
* {@code FileProvider} does is a hundred lines — map a request path onto a
|
||||
* directory, refuse anything outside it, answer the two columns the share sheet
|
||||
* reads — and those lines are below. The configuration it takes as an XML
|
||||
* {@code <meta-data>} resource is a constant here instead, because there is
|
||||
* exactly one directory worth serving and a second place to state it is a
|
||||
* second place for it to be wrong.
|
||||
*
|
||||
* <h2>The one directory</h2>
|
||||
*
|
||||
* <p>{@code getFilesDir()}, which is the same directory the Rust side calls
|
||||
* {@code internal_data_path} and passes to {@code dr_sync::account::set_data_dir}
|
||||
* — {@code ANativeActivity.internalDataPath} and {@code Context.getFilesDir()}
|
||||
* are the same path. Everything the app writes for itself, the export outbox
|
||||
* included, is under it. Nothing else is reachable: a request is resolved
|
||||
* against the real filesystem with {@link File#getCanonicalFile()} and then
|
||||
* checked to be *inside* that root, so {@code ../} and a symlink planted in the
|
||||
* outbox are refused by the same test. Serving a path the caller composed,
|
||||
* unchecked, would turn a share button into a reader for every file this app
|
||||
* can see, which on Android includes credentials and the whole catalog.
|
||||
*
|
||||
* <p>{@code android:exported="false"} in the manifest is the outer half of the
|
||||
* same rule: no app can address this provider at all except through a URI this
|
||||
* app handed it with a read grant attached.
|
||||
*/
|
||||
public final class ExportProvider extends ContentProvider {
|
||||
private static final String TAG = "DarkRoom";
|
||||
|
||||
/**
|
||||
* Must equal {@code android:authorities} in AndroidManifest.xml.
|
||||
*
|
||||
* <p>A mismatch is not a build error and not a runtime error here: it is a
|
||||
* {@code SecurityException} in whichever app opened the share sheet, naming
|
||||
* an authority that does not exist. A test in {@code lib.rs} asserts the
|
||||
* two strings are the same for that reason.
|
||||
*/
|
||||
public static final String AUTHORITY = "paris.tourolle.darkroom.exports";
|
||||
|
||||
/** Nothing to set up; the root is resolved per request against the context. */
|
||||
@Override
|
||||
public boolean onCreate() {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code content://} URI for a file, or null if it is not one this
|
||||
* provider may serve.
|
||||
*
|
||||
* <p>Returning null rather than an unusable URI keeps the refusal at the
|
||||
* point where the path is known. A URI for a file outside the root would be
|
||||
* rejected later by {@link #openFile}, in the *receiving* app's stack trace,
|
||||
* where nothing says which of our files was asked for.
|
||||
*/
|
||||
public static Uri uriFor(Context context, File file) {
|
||||
try {
|
||||
File root = root(context);
|
||||
File target = file.getCanonicalFile();
|
||||
String relative = within(root, target);
|
||||
if (relative == null) {
|
||||
Log.w(TAG, "not shareable, outside " + root + ": " + target);
|
||||
return null;
|
||||
}
|
||||
// Built segment by segment rather than with a composed path
|
||||
// string: appendPath percent-encodes, and getPathSegments below
|
||||
// decodes symmetrically. A file called "Rue d'Alésia.jpg" survives
|
||||
// the round trip only because both halves agree.
|
||||
Uri.Builder builder = new Uri.Builder().scheme("content").authority(AUTHORITY);
|
||||
for (String segment : relative.split("/")) {
|
||||
if (!segment.isEmpty()) {
|
||||
builder.appendPath(segment);
|
||||
}
|
||||
}
|
||||
return builder.build();
|
||||
} catch (IOException e) {
|
||||
Log.w(TAG, "cannot resolve " + file + " for sharing: " + e);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The two columns a share target actually reads.
|
||||
*
|
||||
* <p>Without {@code _display_name} the receiving app shows the URI's last
|
||||
* segment, and without {@code _size} a mail client cannot tell whether the
|
||||
* attachment fits before it starts reading. Both are optional in the sense
|
||||
* that the transfer still works; both are the difference between "DSC_4471
|
||||
* final.jpg, 8.2 MB" and an unnamed blob.
|
||||
*/
|
||||
@Override
|
||||
public Cursor query(Uri uri, String[] projection, String selection,
|
||||
String[] selectionArgs, String sortOrder) {
|
||||
File file = resolve(uri);
|
||||
if (file == null) {
|
||||
return null;
|
||||
}
|
||||
String[] columns = projection != null
|
||||
? projection
|
||||
: new String[] {OpenableColumns.DISPLAY_NAME, OpenableColumns.SIZE};
|
||||
MatrixCursor cursor = new MatrixCursor(columns, 1);
|
||||
MatrixCursor.RowBuilder row = cursor.newRow();
|
||||
for (String column : columns) {
|
||||
if (OpenableColumns.DISPLAY_NAME.equals(column)) {
|
||||
row.add(file.getName());
|
||||
} else if (OpenableColumns.SIZE.equals(column)) {
|
||||
row.add(file.length());
|
||||
} else {
|
||||
// A column we do not have. Null rather than omitted: a cursor
|
||||
// whose row is shorter than its projection throws in the
|
||||
// caller, which is a crash in someone else's app.
|
||||
row.add(null);
|
||||
}
|
||||
}
|
||||
return cursor;
|
||||
}
|
||||
|
||||
/**
|
||||
* From the extension, because that is all there is.
|
||||
*
|
||||
* <p>The type decides which apps the chooser offers, so guessing wrong
|
||||
* narrows the sheet rather than breaking the transfer. Exports are JPEG,
|
||||
* PNG or TIFF and {@code MimeTypeMap} knows all three.
|
||||
*/
|
||||
@Override
|
||||
public String getType(Uri uri) {
|
||||
File file = resolve(uri);
|
||||
if (file == null) {
|
||||
return null;
|
||||
}
|
||||
String name = file.getName();
|
||||
int dot = name.lastIndexOf('.');
|
||||
if (dot >= 0 && dot < name.length() - 1) {
|
||||
String extension = name.substring(dot + 1).toLowerCase(Locale.ROOT);
|
||||
String type = MimeTypeMap.getSingleton().getMimeTypeFromExtension(extension);
|
||||
if (type != null) {
|
||||
return type;
|
||||
}
|
||||
}
|
||||
return "application/octet-stream";
|
||||
}
|
||||
|
||||
/**
|
||||
* Read-only, always.
|
||||
*
|
||||
* <p>A write mode is refused rather than quietly downgraded: a caller that
|
||||
* asked for "rw" intends to save something back, and letting it open the
|
||||
* file read-only would fail at its first write with an error about a
|
||||
* descriptor rather than about permission. Nothing this app shares is meant
|
||||
* to be edited in place by the app it was shared with.
|
||||
*/
|
||||
@Override
|
||||
public ParcelFileDescriptor openFile(Uri uri, String mode) throws FileNotFoundException {
|
||||
if (!"r".equals(mode)) {
|
||||
throw new SecurityException("this provider is read-only, asked for '" + mode + "'");
|
||||
}
|
||||
File file = resolve(uri);
|
||||
if (file == null) {
|
||||
throw new FileNotFoundException("no such export: " + uri);
|
||||
}
|
||||
return ParcelFileDescriptor.open(file, ParcelFileDescriptor.MODE_READ_ONLY);
|
||||
}
|
||||
|
||||
@Override
|
||||
public Uri insert(Uri uri, ContentValues values) {
|
||||
throw new UnsupportedOperationException("exports are written by the app, not through it");
|
||||
}
|
||||
|
||||
@Override
|
||||
public int update(Uri uri, ContentValues values, String selection, String[] selectionArgs) {
|
||||
throw new UnsupportedOperationException("exports are written by the app, not through it");
|
||||
}
|
||||
|
||||
@Override
|
||||
public int delete(Uri uri, String selection, String[] selectionArgs) {
|
||||
throw new UnsupportedOperationException("exports are deleted by the app, not through it");
|
||||
}
|
||||
|
||||
/** The served root, resolved through the filesystem so the check below is real. */
|
||||
private static File root(Context context) throws IOException {
|
||||
return context.getFilesDir().getCanonicalFile();
|
||||
}
|
||||
|
||||
/** The file a request names, or null if it names anything else. */
|
||||
private File resolve(Uri uri) {
|
||||
Context context = getContext();
|
||||
if (context == null) {
|
||||
return null;
|
||||
}
|
||||
List<String> segments = uri.getPathSegments();
|
||||
if (segments.isEmpty()) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
File root = root(context);
|
||||
File candidate = root;
|
||||
for (String segment : segments) {
|
||||
candidate = new File(candidate, segment);
|
||||
}
|
||||
candidate = candidate.getCanonicalFile();
|
||||
if (within(root, candidate) == null || !candidate.isFile()) {
|
||||
Log.w(TAG, "refused " + uri);
|
||||
return null;
|
||||
}
|
||||
return candidate;
|
||||
} catch (IOException e) {
|
||||
Log.w(TAG, "refused " + uri + ": " + e);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code target}'s path relative to {@code root}, or null if it is not
|
||||
* under it.
|
||||
*
|
||||
* <p>Both sides are canonical by the time they get here, which is what
|
||||
* makes one string comparison enough for {@code ../} and for a symlink
|
||||
* alike. The trailing separator matters: without it a sibling directory
|
||||
* whose name merely starts with the root's — {@code /data/.../files.old} —
|
||||
* passes.
|
||||
*/
|
||||
private static String within(File root, File target) {
|
||||
String rootPath = root.getPath() + File.separator;
|
||||
String targetPath = target.getPath();
|
||||
if (!targetPath.startsWith(rootPath)) {
|
||||
return null;
|
||||
}
|
||||
return targetPath.substring(rootPath.length());
|
||||
}
|
||||
}
|
||||
@@ -1,320 +0,0 @@
|
||||
package paris.tourolle.darkroom;
|
||||
|
||||
import android.app.Activity;
|
||||
import android.content.ActivityNotFoundException;
|
||||
import android.content.ContentResolver;
|
||||
import android.content.Context;
|
||||
import android.content.Intent;
|
||||
import android.database.Cursor;
|
||||
import android.net.Uri;
|
||||
import android.provider.OpenableColumns;
|
||||
import android.util.Log;
|
||||
|
||||
import java.io.File;
|
||||
import java.io.FileOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.io.OutputStream;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* The two directions of FR-PLAT-AND-6: what the app was opened *with*, and
|
||||
* handing a finished export to somebody else.
|
||||
*
|
||||
* <h2>Why this is Java and not JNI in lib.rs</h2>
|
||||
*
|
||||
* <p>Every call below is reachable over JNI, and doing it that way would be
|
||||
* roughly forty {@code call_method} invocations with their signatures written
|
||||
* out as strings — each one a name Java checks at run time and nothing checks
|
||||
* at build time. The Rust side would then hold the exact logic that is here,
|
||||
* expressed less clearly, and a typo in {@code "()Landroid/content/Intent;"}
|
||||
* would surface on a device as a {@code NoSuchMethodError} rather than at the
|
||||
* compiler. So the platform work stays on the platform's side and the JNI
|
||||
* surface is two calls, both taking and returning strings.
|
||||
*
|
||||
* <p>The class is only reachable because the APK now compiles Java at all; see
|
||||
* docker/android/assemble-apk.sh.
|
||||
*/
|
||||
public final class Intents {
|
||||
private static final String TAG = "DarkRoom";
|
||||
|
||||
/**
|
||||
* Where incoming images are copied, under {@code getCacheDir()}.
|
||||
*
|
||||
* <p>The cache and not the data directory, deliberately: these are copies
|
||||
* of somebody else's file, the app has no claim on them once the session
|
||||
* ends, and the cache is the one place Android may reclaim under storage
|
||||
* pressure without the user being asked. Putting them in the data
|
||||
* directory would grow the app's footprint by a RAW file per share, for
|
||||
* ever, with nothing that ever deletes them.
|
||||
*/
|
||||
private static final String INBOX = "incoming";
|
||||
|
||||
private Intents() {
|
||||
}
|
||||
|
||||
/**
|
||||
* The images this launch was asked to open, as paths the decoder can read.
|
||||
*
|
||||
* <p>Empty for an ordinary launch from the launcher, which is the common
|
||||
* case and not a failure.
|
||||
*
|
||||
* <h3>Why the bytes are copied</h3>
|
||||
*
|
||||
* <p>A share arrives as a {@code content://} URI, which is a handle into
|
||||
* another app's provider and not a path — there is no filename behind it to
|
||||
* open, and the grant that makes it readable belongs to this task and dies
|
||||
* with it. DarkRoom's decoders take paths (ARCH §6.9 is the note that
|
||||
* Android has no paths to give), so the choice is to copy or to teach the
|
||||
* whole read path about URIs, and the second is FR-PLAT-AND-1's SAF
|
||||
* connector, which is not built.
|
||||
*
|
||||
* <p>So it is a copy, and the cost is honest: a 60 MB raw file is written
|
||||
* once, to the cache, before the viewer opens. It is bounded by the share
|
||||
* being a deliberate act — a person picked these files — rather than by
|
||||
* anything this code does.
|
||||
*
|
||||
* <p>The inbox is emptied first. Without that, every share ever received
|
||||
* accumulates until the platform decides the cache is too large, and the
|
||||
* files are indistinguishable from each other by then.
|
||||
*/
|
||||
public static String[] receive(Activity activity) {
|
||||
List<Uri> uris = incoming(activity.getIntent());
|
||||
if (uris.isEmpty()) {
|
||||
return new String[0];
|
||||
}
|
||||
|
||||
File inbox = new File(activity.getCacheDir(), INBOX);
|
||||
empty(inbox);
|
||||
if (!inbox.mkdirs() && !inbox.isDirectory()) {
|
||||
Log.e(TAG, "cannot create " + inbox + "; the launch intent is dropped");
|
||||
return new String[0];
|
||||
}
|
||||
|
||||
List<String> paths = new ArrayList<String>();
|
||||
for (Uri uri : uris) {
|
||||
String path = localise(activity, uri, inbox, paths.size());
|
||||
if (path != null) {
|
||||
paths.add(path);
|
||||
}
|
||||
}
|
||||
Log.i(TAG, "launch intent carried " + paths.size() + " of " + uris.size() + " image(s)");
|
||||
return paths.toArray(new String[0]);
|
||||
}
|
||||
|
||||
/**
|
||||
* Offer a file this app produced to whatever else is installed.
|
||||
*
|
||||
* <p>Returns false when there is nothing to offer it to, or when the file
|
||||
* is not one {@link ExportProvider} may serve — both of which the caller
|
||||
* has to be able to say out loud, because from the user's side a share
|
||||
* button that does nothing is indistinguishable from one that failed.
|
||||
*
|
||||
* <p>{@code FLAG_GRANT_READ_URI_PERMISSION} is the whole security model:
|
||||
* the provider is not exported, so the receiving app can reach this one
|
||||
* file, for as long as its task lives, and nothing else ever.
|
||||
*/
|
||||
public static boolean share(Activity activity, String path, String mimeType) {
|
||||
Uri uri = ExportProvider.uriFor(activity, new File(path));
|
||||
if (uri == null) {
|
||||
return false;
|
||||
}
|
||||
|
||||
Intent send = new Intent(Intent.ACTION_SEND);
|
||||
send.setType(mimeType != null && !mimeType.isEmpty() ? mimeType : "image/*");
|
||||
send.putExtra(Intent.EXTRA_STREAM, uri);
|
||||
send.addFlags(Intent.FLAG_GRANT_READ_URI_PERMISSION);
|
||||
|
||||
// Always a chooser, never a direct start. Android's "remembered
|
||||
// default" for ACTION_SEND is a per-user setting this app has no
|
||||
// business consuming: the app a photograph should go to differs every
|
||||
// time, and the one time it does not, the sheet is one extra tap.
|
||||
Intent chooser = Intent.createChooser(send, null);
|
||||
try {
|
||||
activity.startActivity(chooser);
|
||||
return true;
|
||||
} catch (ActivityNotFoundException e) {
|
||||
Log.w(TAG, "nothing installed accepts " + mimeType + ": " + e);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The URIs an Intent carries, by the action that carried them.
|
||||
*
|
||||
* <p>Only the actions the manifest registers for. An action we did not
|
||||
* declare cannot arrive, so handling one here would be code that reads as
|
||||
* support for something the launcher will never offer.
|
||||
*/
|
||||
@SuppressWarnings("deprecation")
|
||||
private static List<Uri> incoming(Intent intent) {
|
||||
List<Uri> uris = new ArrayList<Uri>();
|
||||
if (intent == null) {
|
||||
return uris;
|
||||
}
|
||||
String action = intent.getAction();
|
||||
if (Intent.ACTION_VIEW.equals(action)) {
|
||||
add(uris, intent.getData());
|
||||
} else if (Intent.ACTION_SEND.equals(action)) {
|
||||
// The typed getParcelableExtra(String, Class) overload is API 33,
|
||||
// and minSdk is 28. The deprecated form is the only one that exists
|
||||
// on every device this APK installs on.
|
||||
add(uris, (Uri) intent.getParcelableExtra(Intent.EXTRA_STREAM));
|
||||
} else if (Intent.ACTION_SEND_MULTIPLE.equals(action)) {
|
||||
ArrayList<Uri> many = intent.getParcelableArrayListExtra(Intent.EXTRA_STREAM);
|
||||
if (many != null) {
|
||||
for (Uri uri : many) {
|
||||
add(uris, uri);
|
||||
}
|
||||
}
|
||||
}
|
||||
return uris;
|
||||
}
|
||||
|
||||
private static void add(List<Uri> uris, Uri uri) {
|
||||
if (uri != null) {
|
||||
uris.add(uri);
|
||||
}
|
||||
}
|
||||
|
||||
/** A URI as a readable path, copying it into the inbox if it is not one already. */
|
||||
private static String localise(Context context, Uri uri, File inbox, int index) {
|
||||
// A file:// URI is already a path, and copying it would double a raw
|
||||
// file on disk to no end. Rare — the platform has refused file:// URIs
|
||||
// between apps since API 24 — but it is what a shell `am start -d
|
||||
// file:///sdcard/…` produces, which is how this path gets tested
|
||||
// without a second app installed.
|
||||
if (ContentResolver.SCHEME_FILE.equals(uri.getScheme())) {
|
||||
String path = uri.getPath();
|
||||
if (path != null && new File(path).canRead()) {
|
||||
return path;
|
||||
}
|
||||
Log.w(TAG, "cannot read " + uri);
|
||||
return null;
|
||||
}
|
||||
|
||||
File dest = new File(inbox, unique(inbox, displayName(context, uri), index));
|
||||
InputStream in = null;
|
||||
OutputStream out = null;
|
||||
try {
|
||||
in = context.getContentResolver().openInputStream(uri);
|
||||
if (in == null) {
|
||||
Log.w(TAG, "no stream behind " + uri);
|
||||
return null;
|
||||
}
|
||||
out = new FileOutputStream(dest);
|
||||
byte[] buffer = new byte[64 * 1024];
|
||||
int read;
|
||||
while ((read = in.read(buffer)) > 0) {
|
||||
out.write(buffer, 0, read);
|
||||
}
|
||||
out.flush();
|
||||
return dest.getAbsolutePath();
|
||||
} catch (IOException e) {
|
||||
Log.w(TAG, "cannot copy " + uri + ": " + e);
|
||||
// The partial copy is removed rather than left: it has the name and
|
||||
// the extension of a photograph and none of the bytes, and the
|
||||
// decoder would report it as a corrupt file rather than a failed
|
||||
// transfer.
|
||||
dest.delete();
|
||||
return null;
|
||||
} catch (SecurityException e) {
|
||||
// The grant on a shared URI dies with the task that received it.
|
||||
// A process resumed from a saved state can find itself holding a
|
||||
// URI it may no longer read (FR-PLAT-AND-3), and that is a lost
|
||||
// permission rather than a broken file.
|
||||
Log.w(TAG, "no longer permitted to read " + uri + ": " + e);
|
||||
dest.delete();
|
||||
return null;
|
||||
} finally {
|
||||
close(in);
|
||||
close(out);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* What the sending app calls the file, reduced to something safe to write.
|
||||
*
|
||||
* <p>The name is chosen by another application and lands in a path this one
|
||||
* composes, so it is filtered rather than trusted: a name containing a
|
||||
* separator would place the copy outside the inbox, and one beginning with
|
||||
* a dot would hide it from everything that lists the directory. What
|
||||
* survives is the part a photographer recognises — {@code DSC_4471.NEF} —
|
||||
* which is the only reason to use the sender's name at all.
|
||||
*/
|
||||
private static String displayName(Context context, Uri uri) {
|
||||
String name = null;
|
||||
Cursor cursor = null;
|
||||
try {
|
||||
cursor = context.getContentResolver().query(
|
||||
uri, new String[] {OpenableColumns.DISPLAY_NAME}, null, null, null);
|
||||
if (cursor != null && cursor.moveToFirst() && !cursor.isNull(0)) {
|
||||
name = cursor.getString(0);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
// Providers are other people's code and any of them may throw.
|
||||
// A name is a convenience; failing the whole open over it is not.
|
||||
Log.d(TAG, "no display name for " + uri + ": " + e);
|
||||
} finally {
|
||||
if (cursor != null) {
|
||||
cursor.close();
|
||||
}
|
||||
}
|
||||
if (name == null) {
|
||||
name = uri.getLastPathSegment();
|
||||
}
|
||||
if (name == null) {
|
||||
return "shared";
|
||||
}
|
||||
StringBuilder safe = new StringBuilder(name.length());
|
||||
for (int i = 0; i < name.length(); i++) {
|
||||
char c = name.charAt(i);
|
||||
boolean ok = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z')
|
||||
|| (c >= '0' && c <= '9') || c == '.' || c == '-' || c == '_';
|
||||
safe.append(ok ? c : '_');
|
||||
}
|
||||
while (safe.length() > 0 && safe.charAt(0) == '.') {
|
||||
safe.deleteCharAt(0);
|
||||
}
|
||||
return safe.length() > 0 ? safe.toString() : "shared";
|
||||
}
|
||||
|
||||
/**
|
||||
* A name nothing in the inbox has yet.
|
||||
*
|
||||
* <p>A multi-image share of a burst arrives as several files a camera named
|
||||
* the same thing in different folders, and the second one silently
|
||||
* overwriting the first would show the user one photograph where they
|
||||
* picked four.
|
||||
*/
|
||||
private static String unique(File inbox, String name, int index) {
|
||||
if (!new File(inbox, name).exists()) {
|
||||
return name;
|
||||
}
|
||||
return index + "-" + name;
|
||||
}
|
||||
|
||||
private static void close(java.io.Closeable stream) {
|
||||
if (stream != null) {
|
||||
try {
|
||||
stream.close();
|
||||
} catch (IOException e) {
|
||||
Log.d(TAG, "close failed: " + e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Delete the inbox's contents, one level deep, which is all it ever has. */
|
||||
private static void empty(File inbox) {
|
||||
File[] stale = inbox.listFiles();
|
||||
if (stale == null) {
|
||||
return;
|
||||
}
|
||||
for (File file : stale) {
|
||||
if (!file.delete()) {
|
||||
Log.d(TAG, "could not remove stale " + file);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,6 +0,0 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<adaptive-icon xmlns:android="http://schemas.android.com/apk/res/android">
|
||||
<background android:drawable="@mipmap/ic_launcher_background"/>
|
||||
<foreground android:drawable="@mipmap/ic_launcher_foreground"/>
|
||||
<monochrome android:drawable="@mipmap/ic_launcher_monochrome"/>
|
||||
</adaptive-icon>
|
||||
|
Before Width: | Height: | Size: 9.5 KiB |
|
Before Width: | Height: | Size: 518 B |
|
Before Width: | Height: | Size: 25 KiB |
|
Before Width: | Height: | Size: 25 KiB |
|
Before Width: | Height: | Size: 5.0 KiB |
|
Before Width: | Height: | Size: 343 B |
|
Before Width: | Height: | Size: 13 KiB |
|
Before Width: | Height: | Size: 13 KiB |
|
Before Width: | Height: | Size: 15 KiB |
|
Before Width: | Height: | Size: 680 B |
|
Before Width: | Height: | Size: 43 KiB |
|
Before Width: | Height: | Size: 43 KiB |
|
Before Width: | Height: | Size: 30 KiB |
|
Before Width: | Height: | Size: 1005 B |
|
Before Width: | Height: | Size: 87 KiB |
|
Before Width: | Height: | Size: 87 KiB |
|
Before Width: | Height: | Size: 51 KiB |
|
Before Width: | Height: | Size: 1.6 KiB |
|
Before Width: | Height: | Size: 151 KiB |
|
Before Width: | Height: | Size: 151 KiB |
@@ -1,192 +0,0 @@
|
||||
//! What the app was launched with, and handing a finished export back out.
|
||||
//!
|
||||
//! FR-PLAT-AND-6's Rust side, which is deliberately the thin side. Both
|
||||
//! directions are implemented in `android/java/paris/tourolle/darkroom/` and
|
||||
//! everything here is the two calls that reach them; `Intents.java` carries the
|
||||
//! reasoning for the split. The short version is that a JNI method signature is
|
||||
//! a string Java resolves at run time and nothing checks at build time, so
|
||||
//! forty of them is forty ways for a rename to become a `NoSuchMethodError` on
|
||||
//! somebody's tablet. Two is two.
|
||||
//!
|
||||
//! # Nothing here fails loudly
|
||||
//!
|
||||
//! A class the loader cannot see, a pending Java exception, a shared URI whose
|
||||
//! grant died with the task that received it: each ends as a log line and an
|
||||
//! empty result. This runs on the way to [`dr_ui::run`], before a window
|
||||
//! exists, and the alternative to opening with an empty browsing list is not
|
||||
//! opening at all.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use jni::errors::Result as JniResult;
|
||||
use jni::objects::{JClass, JObject, JObjectArray, JString, JValue};
|
||||
use jni::{JNIEnv, JavaVM};
|
||||
|
||||
/// The class both directions live in, named the way `loadClass` wants it —
|
||||
/// dots, not slashes. `find_class` takes the other form, and this code calls
|
||||
/// neither by accident; see [`load_class`].
|
||||
const INTENTS: &str = "paris.tourolle.darkroom.Intents";
|
||||
|
||||
/// The images this launch was asked to open, already local and readable.
|
||||
///
|
||||
/// Empty for an ordinary launch from the launcher, which is the common case
|
||||
/// and not a failure. What comes back is passed to `dr_ui::run` exactly as
|
||||
/// command-line paths are on the desktop, so a shared photograph becomes the
|
||||
/// browsing list and `startup_action` shows it rather than the launch screen.
|
||||
pub fn launch_images(app: &slint::android::AndroidApp) -> Vec<PathBuf> {
|
||||
with_activity(app, "reading the launch intent", |env, activity| {
|
||||
let class = load_class(env, activity, INTENTS)?;
|
||||
let returned = env
|
||||
.call_static_method(
|
||||
&class,
|
||||
"receive",
|
||||
"(Landroid/app/Activity;)[Ljava/lang/String;",
|
||||
&[JValue::Object(activity)],
|
||||
)?
|
||||
.l()?;
|
||||
|
||||
let array = JObjectArray::from(returned);
|
||||
let count = env.get_array_length(&array)?;
|
||||
let mut paths = Vec::with_capacity(count as usize);
|
||||
for i in 0..count {
|
||||
let element = env.get_object_array_element(&array, i)?;
|
||||
let text: String = env.get_string(&JString::from(element))?.into();
|
||||
paths.push(PathBuf::from(text));
|
||||
}
|
||||
Ok(paths)
|
||||
})
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Offer a file this app produced to whatever else is installed.
|
||||
///
|
||||
/// `false` means the sheet did not open — the file is not under the directory
|
||||
/// [`ExportProvider`] serves, or nothing installed accepts the type. Both are
|
||||
/// answers a caller has to be able to give the user, because a share control
|
||||
/// that silently does nothing is indistinguishable from one that failed.
|
||||
///
|
||||
/// **This half has no caller yet, and that is the honest state of it.** The
|
||||
/// provider, the URI grant and the chooser are all here and are what
|
||||
/// FR-PLAT-AND-6 asks for; what is missing is a share control in the interface,
|
||||
/// which lives in `ui/dr-ui` and needs one thing this signature shows: an
|
||||
/// `AndroidApp` to call through. Wiring it means keeping a clone of the app —
|
||||
/// it is `Clone` and cheap — somewhere `ui/` can reach, which is a change to
|
||||
/// how the platform entry point talks to the interface rather than a change
|
||||
/// here. Until that exists this function is reachable and untested, and it is
|
||||
/// deliberately not tagged as covering the requirement.
|
||||
///
|
||||
/// `mime` decides which applications the chooser offers; the empty string
|
||||
/// falls back to `image/*` on the Java side.
|
||||
pub fn share(app: &slint::android::AndroidApp, file: &Path, mime: &str) -> bool {
|
||||
with_activity(app, "opening the share sheet", |env, activity| {
|
||||
let class = load_class(env, activity, INTENTS)?;
|
||||
let path = env.new_string(file.to_string_lossy().as_ref())?;
|
||||
let mime = env.new_string(mime)?;
|
||||
env.call_static_method(
|
||||
&class,
|
||||
"share",
|
||||
"(Landroid/app/Activity;Ljava/lang/String;Ljava/lang/String;)Z",
|
||||
&[
|
||||
JValue::Object(activity),
|
||||
JValue::Object(&path),
|
||||
JValue::Object(&mime),
|
||||
],
|
||||
)?
|
||||
.z()
|
||||
})
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
/// Attach to the JVM, borrow the activity, and run `body` against both.
|
||||
///
|
||||
/// Shared by the two entry points because the three steps before the
|
||||
/// interesting one are identical and each has its own way of failing. `body`
|
||||
/// returning `Err` is reported here, once, in the one place that can also clear
|
||||
/// a pending Java exception — see [`report`].
|
||||
fn with_activity<T>(
|
||||
app: &slint::android::AndroidApp,
|
||||
doing: &str,
|
||||
body: impl FnOnce(&mut JNIEnv, &JObject) -> JniResult<T>,
|
||||
) -> Option<T> {
|
||||
let vm = match unsafe { JavaVM::from_raw(app.vm_as_ptr().cast()) } {
|
||||
Ok(vm) => vm,
|
||||
Err(e) => {
|
||||
log::error!("no JVM handle, so {doing} is skipped: {e}");
|
||||
return None;
|
||||
}
|
||||
};
|
||||
// Cheap when the thread is already attached, which it is: the glue
|
||||
// attached it before it called `android_main`. The guard exists for the
|
||||
// case where it is not, and costs a lookup where it is.
|
||||
let mut env = match vm.attach_current_thread() {
|
||||
Ok(env) => env,
|
||||
Err(e) => {
|
||||
log::error!("cannot attach to the JVM, so {doing} is skipped: {e}");
|
||||
return None;
|
||||
}
|
||||
};
|
||||
|
||||
// SAFETY: `activity_as_ptr` documents this as an unowned JNI *global*
|
||||
// reference to the Activity, valid for as long as the `AndroidApp` it came
|
||||
// from. `JObject` in jni 0.21 is a plain wrapper with no `Drop`, so
|
||||
// borrowing it here cannot delete a reference this code does not own — the
|
||||
// one way to get this wrong is `AutoLocal` or a `GlobalRef`, both of which
|
||||
// would free it out from under android-activity.
|
||||
let activity = unsafe { JObject::from_raw(app.activity_as_ptr().cast()) };
|
||||
|
||||
match body(&mut env, &activity) {
|
||||
Ok(value) => Some(value),
|
||||
Err(e) => {
|
||||
report(&mut env, doing, &e);
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Look an app class up through the *activity's* class loader.
|
||||
///
|
||||
/// `find_class` is the obvious call and the wrong one. JNI resolves a class
|
||||
/// against the loader belonging to the Java frame beneath the call, and on this
|
||||
/// thread there is no such frame: `android_main` runs on a thread the native
|
||||
/// glue created and attached itself, so the loader in scope is the system one.
|
||||
/// It knows every class in the platform and nothing at all from this APK, and
|
||||
/// says so as a `ClassNotFoundException` naming a class that is plainly in the
|
||||
/// dex — which reads as a broken build rather than as the wrong loader.
|
||||
///
|
||||
/// The activity is a Java object, so its loader is the app's.
|
||||
fn load_class<'local>(
|
||||
env: &mut JNIEnv<'local>,
|
||||
activity: &JObject,
|
||||
name: &str,
|
||||
) -> JniResult<JClass<'local>> {
|
||||
let loader = env
|
||||
.call_method(activity, "getClassLoader", "()Ljava/lang/ClassLoader;", &[])?
|
||||
.l()?;
|
||||
let name = env.new_string(name)?;
|
||||
let class = env
|
||||
.call_method(
|
||||
&loader,
|
||||
"loadClass",
|
||||
"(Ljava/lang/String;)Ljava/lang/Class;",
|
||||
&[JValue::Object(&name)],
|
||||
)?
|
||||
.l()?;
|
||||
Ok(JClass::from(class))
|
||||
}
|
||||
|
||||
/// Log a JNI failure, and clear the exception behind it if there is one.
|
||||
///
|
||||
/// The clearing is not tidiness. A Java exception raised through JNI stays
|
||||
/// *pending* on the thread, and the next JNI call made while one is pending
|
||||
/// aborts the process — so a swallowed exception here would come back as a
|
||||
/// crash somewhere unrelated, most likely inside Slint. `exception_describe`
|
||||
/// first, because the trace it prints to logcat is the only place the Java
|
||||
/// class and line survive; `jni::errors::Error::JavaException` on its own says
|
||||
/// neither.
|
||||
fn report(env: &mut JNIEnv, doing: &str, e: &jni::errors::Error) {
|
||||
log::error!("{doing} failed: {e}");
|
||||
if let Ok(true) = env.exception_check() {
|
||||
let _ = env.exception_describe();
|
||||
let _ = env.exception_clear();
|
||||
}
|
||||
}
|
||||
@@ -1,558 +0,0 @@
|
||||
//! DarkRoom Android entry point.
|
||||
//!
|
||||
//! The counterpart to `darkroom-desktop`'s `main`, with two differences that
|
||||
//! come from the platform rather than from choice:
|
||||
//!
|
||||
//! * There are no command-line paths. Android's SAF hands out document URIs,
|
||||
//! not filesystem paths (ARCH §6.9), so the viewer opens with an empty
|
||||
//! browsing list and the library grid is the only way in.
|
||||
//! * Logging goes to logcat *and* to a file. `env_logger` writes to stderr,
|
||||
//! which Android discards; logcat replaces it, and a rotating file beside it
|
||||
//! replaces the thing logcat cannot be — a record that outlives the session
|
||||
//! and can be sent to somebody (NFR-OPS-1, [`dr_plat::diagnostics`]).
|
||||
//!
|
||||
//! The first of those has one exception, and it is the launch `Intent`: a
|
||||
//! gallery, a file manager or the share sheet can name images to open, and
|
||||
//! those arrive as URIs on an `Intent` rather than as words on a command line.
|
||||
//! [`intents`] turns them into paths, and from there they are the same list
|
||||
//! the desktop builds from `argv` (FR-PLAT-AND-6).
|
||||
|
||||
// The whole module is JNI against classes that exist only in the APK, so it
|
||||
// is gated with everything else that cannot compile off-device.
|
||||
#[cfg(target_os = "android")]
|
||||
mod intents;
|
||||
|
||||
// `slint::android` exists only when compiling for Android, so the whole entry
|
||||
// point is gated on the target rather than on a feature. Without this the
|
||||
// crate is still a workspace member on the host, and `cargo test --workspace`
|
||||
// fails to compile it — a build break that only ever appears off-device.
|
||||
#[cfg(target_os = "android")]
|
||||
/// TRACES: M-13 | M-14
|
||||
/// Android application entry point, called by android-activity's glue.
|
||||
#[no_mangle]
|
||||
fn android_main(app: slint::android::AndroidApp) {
|
||||
// **The first statement in the process, and it has to be.** Everything
|
||||
// between here and `install` returning runs with no logger installed at
|
||||
// all: asking the activity for its external directory, `create_dir_all`
|
||||
// and an `open` on a FUSE-backed volume the system may still be mounting.
|
||||
// A failure or a stall in any of it is invisible on every surface there
|
||||
// is — no file yet, and nothing in logcat either — which is precisely the
|
||||
// kind of launch logcat exists to debug.
|
||||
//
|
||||
// `AndroidLogger` rather than `init_once`, so logcat can be *teed* rather
|
||||
// than replaced: `init_once` installs itself as the global logger and
|
||||
// there is only one of those. Everything that reached logcat before the
|
||||
// file existed still reaches it, at the same level and under the same tag;
|
||||
// the file is strictly additional.
|
||||
let console = android_logger::AndroidLogger::new(
|
||||
android_logger::Config::default()
|
||||
.with_max_level(log::LevelFilter::Info)
|
||||
.with_tag("DarkRoom"),
|
||||
);
|
||||
|
||||
// Handed to the logger directly, and **not** written as `log::info!`,
|
||||
// which here would compile and emit nothing: the facade's maximum level is
|
||||
// `Off` until `diagnostics::install` sets it, and the macro tests that
|
||||
// before it reaches any logger at all. This call skips the facade and
|
||||
// reaches `__android_log_write` with nothing in between.
|
||||
//
|
||||
// That independence is the second reason for it. When the log is silent,
|
||||
// this line is what says which half is at fault: present here and absent
|
||||
// below means the `log` wiring, absent in both means liblog is not
|
||||
// delivering this process's records — a question about the device, which
|
||||
// no amount of reading this file can answer.
|
||||
log::Log::log(
|
||||
&console,
|
||||
&log::Record::builder()
|
||||
.level(log::Level::Info)
|
||||
.target(module_path!())
|
||||
.module_path(Some(module_path!()))
|
||||
.args(format_args!(
|
||||
"DarkRoom v{} starting; logcat only until the log file opens",
|
||||
env!("CARGO_PKG_VERSION")
|
||||
))
|
||||
.build(),
|
||||
);
|
||||
|
||||
// Before the file logger, because it needs somewhere to write.
|
||||
//
|
||||
// **The external directory, not the internal one, and the difference is
|
||||
// the entire point of the file.** Both are app-private and both survive
|
||||
// backgrounding — the volatile one is the *cache* directory, which is not
|
||||
// in play here. What separates them is retrieval:
|
||||
// `/data/data/<pkg>/files` needs `run-as` against a debuggable build or
|
||||
// root to read, and `/sdcard/Android/data/<pkg>/files` is a plain
|
||||
// `adb pull` from any build, needing no permission since API 19. A log
|
||||
// nobody can get off the device does not do the job NFR-OPS-1 describes.
|
||||
//
|
||||
// The consequence is that anyone holding the tablet can read it, which is
|
||||
// why `dr_plat::diagnostics` redacts at the sink and why configuration —
|
||||
// the account list, and the credential reference beside it — stays on
|
||||
// `internal_data_path` below rather than moving here (NFR-SEC-2).
|
||||
let external = app.external_data_path();
|
||||
if let Some(dir) = external.clone().or_else(|| app.internal_data_path()) {
|
||||
dr_plat::set_state_dir(dir);
|
||||
}
|
||||
|
||||
let logging = dr_plat::diagnostics::install(Box::new(console), log::LevelFilter::Info);
|
||||
|
||||
// Panics go to stderr, and Android discards stderr. Without this hook a
|
||||
// worker thread that panics is invisible: the process survives, the
|
||||
// channel it was writing to closes, and the UI reports only that
|
||||
// something "failed unexpectedly" with no way to find out what.
|
||||
//
|
||||
// This used to be one `log::error!` of the raw panic, which had two
|
||||
// problems: logcat is a ring buffer that is gone by the time a user
|
||||
// reports anything, and the raw message can carry a document URI naming
|
||||
// their library or a credential a library interpolated into an error
|
||||
// (NFR-SEC-2). `dr_plat::crash` writes a redacted record to disk and logs
|
||||
// the redacted form. Nothing uploads it.
|
||||
//
|
||||
// Before `set_state_dir` on purpose: the hook resolves the directory when
|
||||
// it fires, so installing it first covers the startup below rather than
|
||||
// leaving it uncovered.
|
||||
dr_plat::crash::install(env!("CARGO_PKG_VERSION"));
|
||||
|
||||
log::info!("DarkRoom v{}", env!("CARGO_PKG_VERSION"));
|
||||
|
||||
// Said in logcat as well as in the file, because the first thing anybody
|
||||
// asked for a log needs is where it is — and on a device that is a path
|
||||
// nobody can guess and a command nobody remembers.
|
||||
match &logging {
|
||||
dr_plat::Installed::ToFile(path) => {
|
||||
log::info!("logging to {}", path.display());
|
||||
if external.is_none() {
|
||||
log::warn!(
|
||||
"no external storage; the log is app-private and needs \
|
||||
`adb shell run-as paris.tourolle.darkroom cat files/darkroom.log` \
|
||||
on a debuggable build"
|
||||
);
|
||||
}
|
||||
}
|
||||
dr_plat::Installed::ConsoleOnly(why) => {
|
||||
log::warn!("no log file this session, only logcat: {why}");
|
||||
}
|
||||
}
|
||||
|
||||
// Before anything opens a store: Android has no $HOME and no XDG
|
||||
// directories, so the default guess resolves to a path the app cannot
|
||||
// write. Nothing failed loudly — the session list went to a doomed path, so
|
||||
// the account survived only as long as the process and backgrounding the app
|
||||
// lost the sign-in. `internal_data_path` is the app's private directory
|
||||
// (ARCH §6.9).
|
||||
match app.internal_data_path() {
|
||||
Some(dir) => {
|
||||
log::info!("data dir: {}", dir.display());
|
||||
// Crash records go beside the account data rather than under it:
|
||||
// both are app-private and neither is a cache, which is the whole
|
||||
// distinction that matters here (see `dr_plat::crash::state_dir`).
|
||||
dr_plat::crash::set_state_dir(dir.join("state"));
|
||||
dr_sync::account::set_data_dir(dir);
|
||||
}
|
||||
None => log::error!("no internal data path; settings will not persist"),
|
||||
}
|
||||
|
||||
// After the data dir, because it writes beside the catalog. **Not** before
|
||||
// the first frame any more — it starts a worker and returns; see the
|
||||
// function for what it used to cost the launch.
|
||||
install_bundled_models(app.clone());
|
||||
|
||||
// Before `init_with_event_listener`, which takes `app` by value and is the
|
||||
// last moment anything can ask the activity a question. Not an ordering
|
||||
// preference — after that line there is no `app` left to read the Intent
|
||||
// through.
|
||||
let opened_with = intents::launch_images(&app);
|
||||
|
||||
// TRACES: FR-PLAT-AND-5
|
||||
// The listener is the whole reason this is not the one-line
|
||||
// `slint::android::init(app)`. Slint owns the event loop on Android, so
|
||||
// the platform's lifecycle and memory events reach the application only if
|
||||
// it asks for them here — and it must ask *before* the loop starts, which
|
||||
// is why this sits between the data directory and `dr_ui::run`.
|
||||
//
|
||||
// The listener runs inside `poll_events`, on the same thread the event
|
||||
// loop and every interface cache live on, which is what lets
|
||||
// `dr_ui::memory` be a thread-local registry of plain `Fn()` rather than a
|
||||
// cross-thread channel (see its module documentation).
|
||||
//
|
||||
// # Why two events and not eight
|
||||
//
|
||||
// FR-PLAT-AND-5 names `onTrimMemory`, whose `TRIM_MEMORY_*` levels grade
|
||||
// how badly the system wants the memory back. Those levels do not exist
|
||||
// here: `ComponentCallbacks2` is a Java interface implemented by an
|
||||
// `Activity` or `Application`, and this app has neither — it is a bare
|
||||
// `NativeActivity`, whose native callback table offers only the ungraded
|
||||
// `onLowMemory`. android-activity surfaces exactly that as `LowMemory`.
|
||||
// Reading the grades would mean shipping a Java subclass to forward them,
|
||||
// which is a distribution-manifest change and not this one.
|
||||
//
|
||||
// `Stop` recovers the one grade that matters most anyway, and for free.
|
||||
// It is the moment the activity stops being visible — `TRIM_MEMORY_UI_HIDDEN`
|
||||
// in all but name — and it is the cheapest possible time to give memory
|
||||
// back, because nothing that is freed has to be drawn again before anyone
|
||||
// sees it. `Pause` deliberately does not qualify: a permission dialog or
|
||||
// the share sheet pauses an activity that is still on screen behind it,
|
||||
// and throwing away its render pipeline would make every such interruption
|
||||
// cost a full re-render.
|
||||
use slint::android::android_activity::{MainEvent, PollEvent};
|
||||
if let Err(e) = slint::android::init_with_event_listener(app, |event| match event {
|
||||
PollEvent::Main(MainEvent::LowMemory) => {
|
||||
dr_ui::memory::relieve(dr_ui::memory::Level::Critical);
|
||||
}
|
||||
PollEvent::Main(MainEvent::Stop) => {
|
||||
dr_ui::memory::relieve(dr_ui::memory::Level::UiHidden);
|
||||
}
|
||||
_ => {}
|
||||
}) {
|
||||
log::error!("Slint Android backend failed to initialise: {e}");
|
||||
return;
|
||||
}
|
||||
|
||||
// The launch Intent's images, where there were any, standing in for the
|
||||
// desktop's argv — `launch::startup_action` treats a non-empty list as
|
||||
// "the user asked for these specifically", which is exactly what a share
|
||||
// or a tap in a gallery is. Empty for an ordinary launch, and the library
|
||||
// opens as before.
|
||||
//
|
||||
// Returning from `android_main` ends the process, so a failure here is
|
||||
// logged rather than propagated — there is no shell to show `Err` to.
|
||||
if let Err(e) = dr_ui::run(opened_with) {
|
||||
log::error!("DarkRoom exited with error: {e:#}");
|
||||
}
|
||||
}
|
||||
|
||||
/// Unpack the models the APK carries, if it carries any.
|
||||
///
|
||||
/// # Why Android needs this and no other platform does
|
||||
///
|
||||
/// A desktop build reads its models from a path — the account's directory, the
|
||||
/// shared one, or `$XDG_DATA_DIRS` where a package put them. **Android has no
|
||||
/// such path.** `internal_data_path` is app-private, `run-as` needs a
|
||||
/// debuggable build, and an asset inside a package is not a path anything can
|
||||
/// open (ARCH §6.9), so a phone had no way to reach a model at all.
|
||||
///
|
||||
/// So the APK carries them in `assets/models/` and this copies them out, once,
|
||||
/// into the same shared directory a desktop install uses. After that every
|
||||
/// lookup in `dr_ui::library` finds them exactly where it finds a desktop
|
||||
/// user's.
|
||||
///
|
||||
/// # The two sets are not the same kind of thing
|
||||
///
|
||||
/// **Face weights are absent from the repository by design.** The InsightFace
|
||||
/// grant is research-only and incompatible with this project's licence
|
||||
/// (docs/faces.md §2), so a desktop user fetches them, runs
|
||||
/// `tools/fix-face-model-shapes.sh` over them, and drops the result in. A build
|
||||
/// that carries none is the ordinary case and face indexing simply stays off.
|
||||
///
|
||||
/// **The scene model is committed** (AGPL, compatible — `models/LICENCE.md`),
|
||||
/// so a build carrying none means a checkout without `git lfs pull` rather than
|
||||
/// a deliberate omission. It is still not an error here: the scene tab reports
|
||||
/// itself unavailable the same way face indexing does, because a photo editor
|
||||
/// that refuses to start over a missing grading feature is worse than one that
|
||||
/// starts without it.
|
||||
///
|
||||
/// # Why it is not `include_bytes!` like the instance model
|
||||
///
|
||||
/// Size. The instance model is 11 MB and compiled in; the scene model is 24 MB
|
||||
/// on top of that, and a 35 MB constant in the binary is paid by every install
|
||||
/// whether or not the tab is opened. Assets are also *stored* rather than
|
||||
/// deflated in the APK (see `assemble-apk.sh`), so unpacking is a copy rather
|
||||
/// than an inflate.
|
||||
///
|
||||
/// # Why it returns before it has done anything
|
||||
///
|
||||
/// **`android_main` runs with the input channel unserviced.** Nothing drains
|
||||
/// it until Slint reaches `poll_events`, and Slint does not reach `poll_events`
|
||||
/// until `dr_ui::run` calls `window.run()`, which is the last line of it. So
|
||||
/// every millisecond spent between the top of `android_main` and that line is a
|
||||
/// millisecond in which Android's input dispatcher gets no answer, and five
|
||||
/// thousand of them is an ANR by definition — the system puts "DarkRoom isn't
|
||||
/// responding" over a window that has never painted, and offers to kill it.
|
||||
///
|
||||
/// This copied **41 MB** on the first launch after an install: 24.9 MB of scene
|
||||
/// model, 13.6 MB of embedder, 2.5 MB of detector, each read whole out of the
|
||||
/// APK and written to `/data`. v0.10.0 added the scene model, which was 60% of
|
||||
/// that total; v0.10.0 is the release the ANR appeared in, and the 8,010 minor
|
||||
/// faults in its report are what 41 MB of freshly touched pages looks like.
|
||||
/// The two further detectors the settings page offers since have made it
|
||||
/// 61 MB, which is the same argument with a larger number.
|
||||
///
|
||||
/// So it runs on a worker (NFR-ARCH-1: nothing blocking on the UI executor) and
|
||||
/// this function returns as soon as the thread is running. Nothing on the
|
||||
/// launch path waits for it, and no other startup step needs its result.
|
||||
///
|
||||
/// # The window in which a model looks absent, and why that is honest enough
|
||||
///
|
||||
/// Until the copy finishes, `library::face_models` and `library::scene_model`
|
||||
/// answer `is_file()` about files that are not written yet, so both report
|
||||
/// their feature unavailable — the same answer they give a build carrying no
|
||||
/// weights at all, which is the ordinary case this whole path was written
|
||||
/// around. It is briefly pessimistic rather than wrong, it lasts about as long
|
||||
/// as it takes to read one screenful of the grid, and the temporary name
|
||||
/// [`unpack_bundled_models`] writes under is what stops it being worse than
|
||||
/// pessimistic: a lookup never sees a half-written file, only an absent one.
|
||||
#[cfg(target_os = "android")]
|
||||
fn install_bundled_models(app: slint::android::AndroidApp) {
|
||||
// Detached rather than joined: there is no later moment on the launch path
|
||||
// that wants the answer, and a handle nobody joins is a handle nobody can
|
||||
// forget to. `AndroidApp` is documented `Send` and `Sync` and is an `Arc`
|
||||
// internally, so the clone costs a refcount; `asset_manager` is asked for
|
||||
// on the worker because `AAssetManager` is thread-safe by contract and
|
||||
// reading the pointer takes only the app's read lock, which `poll_events`
|
||||
// also only ever holds shared.
|
||||
std::thread::spawn(move || unpack_bundled_models(&app));
|
||||
}
|
||||
|
||||
/// The copy itself, on the worker [`install_bundled_models`] starts.
|
||||
#[cfg(target_os = "android")]
|
||||
fn unpack_bundled_models(app: &slint::android::AndroidApp) {
|
||||
use std::io::Read;
|
||||
|
||||
let started = std::time::Instant::now();
|
||||
|
||||
// The face names are the **shape-fixed** exports, matching what
|
||||
// `library::face_models` looks for: tract cannot parse either InsightFace
|
||||
// graph with its dynamic input dimension, so what ships here has already
|
||||
// been through `tools/fix-face-model-shapes.sh`.
|
||||
//
|
||||
// The scene entries are three files rather than one because the graph alone
|
||||
// decodes to 150 anonymous channels — `library::scene_model` wants the
|
||||
// vocabulary and the category descriptor beside it, and requires all three
|
||||
// before it reports the tab available.
|
||||
//
|
||||
// Three detectors, because which one runs is a setting
|
||||
// (`FaceDetector`, docs/faces.md §12.3) and a tablet has no other way to
|
||||
// obtain the one it was not shipped with. Twenty megabytes of APK for
|
||||
// the choice; the embedder is the same for all three.
|
||||
const BUNDLED: [(&std::ffi::CStr, &str); 7] = [
|
||||
(c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"),
|
||||
(c"models/scrfd_2.5g_640.onnx", "scrfd_2.5g_640.onnx"),
|
||||
(c"models/scrfd_10g_640.onnx", "scrfd_10g_640.onnx"),
|
||||
(c"models/arcface_mbf_b1.onnx", "arcface_mbf_b1.onnx"),
|
||||
(c"models/yolo26s-sem-ade20k.onnx", "yolo26s-sem-ade20k.onnx"),
|
||||
(
|
||||
c"models/yolo26s-sem-ade20k.classes.json",
|
||||
"yolo26s-sem-ade20k.classes.json",
|
||||
),
|
||||
(c"models/categories.txt", "categories.txt"),
|
||||
];
|
||||
|
||||
let dir = dr_ui::shared_face_models_dir();
|
||||
let assets = app.asset_manager();
|
||||
let mut copied = 0u64;
|
||||
|
||||
for (asset_path, name) in BUNDLED {
|
||||
let dest = dir.join(name);
|
||||
// Already unpacked. Not re-read on every launch: this is 61 MB of
|
||||
// copying across the seven entries, and the file does not change without
|
||||
// the APK changing, at which point the install wiped it anyway. It
|
||||
// matters more now than it did — a launch that skips every entry here
|
||||
// costs nothing at all, which is what makes the second launch after an
|
||||
// install cheap even though the first one is not.
|
||||
if dest.is_file() {
|
||||
continue;
|
||||
}
|
||||
let Some(mut asset) = assets.open(asset_path) else {
|
||||
log::info!("no bundled {name} in this APK; the feature needing it stays off");
|
||||
continue;
|
||||
};
|
||||
let mut bytes = Vec::new();
|
||||
if let Err(e) = asset.read_to_end(&mut bytes) {
|
||||
log::error!("bundled {name} could not be read: {e}");
|
||||
continue;
|
||||
}
|
||||
if let Err(e) = std::fs::create_dir_all(&dir) {
|
||||
log::error!("cannot create {}: {e}", dir.display());
|
||||
return;
|
||||
}
|
||||
// Written under a temporary name and renamed, because
|
||||
// `library::face_models` and `library::scene_model` both decide a
|
||||
// feature is available on `is_file()` alone. A truncated write — the process backgrounded and
|
||||
// killed mid-copy — would otherwise leave a file that passes that test
|
||||
// and fails inside tract, reported to the user as a broken model rather
|
||||
// than a missing one.
|
||||
let part = dir.join(format!("{name}.part"));
|
||||
match std::fs::write(&part, &bytes).and_then(|()| std::fs::rename(&part, &dest)) {
|
||||
Ok(()) => {
|
||||
copied += bytes.len() as u64;
|
||||
log::info!("installed bundled {name} ({} bytes)", bytes.len());
|
||||
}
|
||||
Err(e) => {
|
||||
log::error!("cannot install {name}: {e}");
|
||||
let _ = std::fs::remove_file(&part);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The figure this whole function is about. Said even when it is zero, so a
|
||||
// launch that ANRs anyway can be told apart from one that spent its six
|
||||
// seconds here — on a second launch there is nothing left to copy and the
|
||||
// line reads `0 bytes`.
|
||||
log::info!(
|
||||
"bundled models ready: {copied} bytes copied in {} ms",
|
||||
started.elapsed().as_millis()
|
||||
);
|
||||
}
|
||||
|
||||
/// TRACES: FR-PLAT-AND-6
|
||||
/// The declarations that make this app a receiver, held to on the host.
|
||||
///
|
||||
/// Everything FR-PLAT-AND-6 does on a device is unreachable from `cargo test`:
|
||||
/// there is no `Intent` off-device and no `ContentProvider` to instantiate. But
|
||||
/// the requirement is not only behaviour — half of it is *declaration*, and a
|
||||
/// declaration can be wrong in ways that compile perfectly and fail silently.
|
||||
/// An intent filter that is deleted takes the app out of every gallery's "open
|
||||
/// with" menu with nothing to notice; an authority that stops matching the
|
||||
/// class it names raises a `SecurityException` in whichever other app opened
|
||||
/// the share sheet, which is the last place anybody would look for it.
|
||||
///
|
||||
/// The manifest is read by aapt2 and the Java by javac, so a Rust build sees
|
||||
/// neither. `include_str!` is what puts them where a test can reach them, and
|
||||
/// this is the only place in the workspace that does.
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
/// The manifest with its comments removed and its whitespace flattened, so
|
||||
/// a match is about the declaration and not about how it is indented.
|
||||
fn manifest() -> String {
|
||||
let xml = include_str!("../android/AndroidManifest.xml");
|
||||
let mut out = String::with_capacity(xml.len());
|
||||
let mut rest = xml;
|
||||
// Comments first, and not by regex over the whole file: several of them
|
||||
// quote the very attribute names the assertions below look for, so a
|
||||
// test that read them would pass on the strength of the prose
|
||||
// explaining an entry that had been deleted.
|
||||
while let Some(start) = rest.find("<!--") {
|
||||
out.push_str(&rest[..start]);
|
||||
match rest[start..].find("-->") {
|
||||
Some(end) => rest = &rest[start + end + 3..],
|
||||
None => {
|
||||
rest = "";
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
out.push_str(rest);
|
||||
out.split_whitespace().collect::<Vec<_>>().join(" ")
|
||||
}
|
||||
|
||||
/// The body of each `<intent-filter>`, so an action and a MIME type are
|
||||
/// checked to be in the *same* filter. Two filters, one naming the action
|
||||
/// and one naming the type, register for neither.
|
||||
fn intent_filters(manifest: &str) -> Vec<&str> {
|
||||
manifest
|
||||
.split("<intent-filter>")
|
||||
.skip(1)
|
||||
.filter_map(|filter| filter.split("</intent-filter>").next())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The single `<provider>` element, attributes and all.
|
||||
fn provider(manifest: &str) -> String {
|
||||
let start = manifest
|
||||
.find("<provider")
|
||||
.expect("no <provider> in the manifest");
|
||||
let rest = &manifest[start..];
|
||||
let end = rest.find("/>").expect("unterminated <provider> element");
|
||||
rest[..end + 2].to_string()
|
||||
}
|
||||
|
||||
fn attribute(element: &str, name: &str) -> Option<String> {
|
||||
let key = format!("{name}=\"");
|
||||
let start = element.find(&key)? + key.len();
|
||||
let value = element[start..].split('"').next()?;
|
||||
Some(value.to_string())
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_gallery_can_open_a_photograph_in_this_app() {
|
||||
let manifest = manifest();
|
||||
let registered = intent_filters(&manifest).iter().any(|filter| {
|
||||
filter.contains("android.intent.action.VIEW")
|
||||
&& filter.contains("android.intent.category.DEFAULT")
|
||||
&& filter.contains(r#"android:mimeType="image/*""#)
|
||||
});
|
||||
assert!(
|
||||
registered,
|
||||
"no VIEW filter for image/*: nothing will offer DarkRoom for a photograph"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_share_sheet_can_send_one_image_or_several() {
|
||||
let manifest = manifest();
|
||||
let registered = intent_filters(&manifest).iter().any(|filter| {
|
||||
// The closing quote matters: SEND is a prefix of SEND_MULTIPLE, so
|
||||
// a bare substring test passes on a filter that declares only the
|
||||
// second and would not be offered for a single photograph.
|
||||
filter.contains(r#"android.intent.action.SEND""#)
|
||||
&& filter.contains(r#"android.intent.action.SEND_MULTIPLE""#)
|
||||
&& filter.contains("android.intent.category.DEFAULT")
|
||||
&& filter.contains(r#"android:mimeType="image/*""#)
|
||||
});
|
||||
assert!(
|
||||
registered,
|
||||
"no SEND/SEND_MULTIPLE filter for image/*, so the share sheet will not list DarkRoom"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn one_activity_ever_so_a_second_launch_cannot_start_a_second_one() {
|
||||
// Not style. Another app can now launch this activity while it is
|
||||
// already running, and the default launch mode answers that by
|
||||
// creating a second NativeActivity in this process — a second
|
||||
// android_main, a second Slint backend, a second wgpu device.
|
||||
assert!(
|
||||
manifest().contains(r#"android:launchMode="singleTask""#),
|
||||
"the activity must be singleTask; see the manifest comment"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_provider_authority_is_the_one_the_class_answers_to() {
|
||||
let manifest = manifest();
|
||||
let element = provider(&manifest);
|
||||
|
||||
let declared =
|
||||
attribute(&element, "android:authorities").expect("the provider declares no authority");
|
||||
let java = include_str!("../android/java/paris/tourolle/darkroom/ExportProvider.java");
|
||||
let constant = java
|
||||
.split("AUTHORITY = \"")
|
||||
.nth(1)
|
||||
.and_then(|rest| rest.split('"').next())
|
||||
.expect("ExportProvider declares no AUTHORITY constant");
|
||||
|
||||
assert_eq!(
|
||||
declared, constant,
|
||||
"the manifest and ExportProvider disagree about the authority; \
|
||||
a share would fail as a SecurityException inside the receiving app"
|
||||
);
|
||||
|
||||
let class = attribute(&element, "android:name").expect("the provider declares no class");
|
||||
let (package, _) = class
|
||||
.rsplit_once('.')
|
||||
.expect("the provider class is unqualified");
|
||||
assert!(
|
||||
java.contains(&format!("package {package};")),
|
||||
"the manifest names {class}, which is not the class in ExportProvider.java"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_provider_hands_out_one_file_at_a_time_and_nothing_by_itself() {
|
||||
let manifest = manifest();
|
||||
let element = provider(&manifest);
|
||||
// The two halves are not redundant. Without the grant, every share
|
||||
// target fails; exported, every app on the device could read this
|
||||
// app's private directory.
|
||||
assert_eq!(
|
||||
attribute(&element, "android:exported").as_deref(),
|
||||
Some("false"),
|
||||
"an exported provider would serve the app's private directory to anything installed"
|
||||
);
|
||||
assert_eq!(
|
||||
attribute(&element, "android:grantUriPermissions").as_deref(),
|
||||
Some("true"),
|
||||
"without URI grants the share sheet opens and every target fails to read the file"
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,19 +0,0 @@
|
||||
[package]
|
||||
name = "darkroom-desktop"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
license.workspace = true
|
||||
|
||||
[dependencies]
|
||||
dr-ui = { workspace = true, features = ["scene-model"] }
|
||||
# For the panic hook and the log sink, directly rather than through dr-ui:
|
||||
# both have to be installed before `dr_ui::run`, because a panic during startup
|
||||
# is exactly the one they exist to catch and record (NFR-OPS-1, NFR-OPS-2).
|
||||
dr-plat.workspace = true
|
||||
anyhow.workspace = true
|
||||
env_logger.workspace = true
|
||||
log.workspace = true
|
||||
|
||||
[features]
|
||||
default = []
|
||||
@@ -1,52 +0,0 @@
|
||||
//! DarkRoom desktop entry point.
|
||||
//!
|
||||
//! darkroom-desktop <file-or-directory>...
|
||||
|
||||
use std::path::PathBuf;
|
||||
|
||||
use dr_plat::diagnostics::Installed;
|
||||
|
||||
fn main() -> anyhow::Result<()> {
|
||||
// Built rather than `init`ed, so the same logger can be handed to the
|
||||
// diagnostics tee: `env_logger` keeps writing to stderr exactly as before,
|
||||
// and every record it accepts is also appended to the on-disk log
|
||||
// (NFR-OPS-1). `filter()` is asked afterwards because the environment may
|
||||
// have overridden the default below, and the file must not be quieter than
|
||||
// the terminal.
|
||||
let console = env_logger::Builder::from_env(env_logger::Env::default().default_filter_or(
|
||||
"info,wgpu_core=warn,wgpu_hal=warn,zbus=warn,tracing=warn,calloop=warn,rawler=warn",
|
||||
))
|
||||
.build();
|
||||
let level = console.filter();
|
||||
let logging = dr_plat::diagnostics::install(Box::new(console), level);
|
||||
|
||||
// Immediately after the logger and before anything that could fail. Until
|
||||
// now a panic on desktop went to stderr and died with the terminal, which
|
||||
// means every panic a user has ever hit was unreportable: the process
|
||||
// survives (the panicking worker does not), a control goes dead, and there
|
||||
// is nothing on disk to say why. The record is local and stays local —
|
||||
// there is no upload path, by design; see `dr_plat::crash`.
|
||||
dr_plat::crash::install(env!("CARGO_PKG_VERSION"));
|
||||
|
||||
log::info!("DarkRoom v{}", env!("CARGO_PKG_VERSION"));
|
||||
// First thing in the file, so a user asked for "the log" can find it
|
||||
// without being told a path over the phone.
|
||||
match &logging {
|
||||
Installed::ToFile(path) => log::info!("logging to {}", path.display()),
|
||||
Installed::ConsoleOnly(why) => log::warn!("no log file this session: {why}"),
|
||||
}
|
||||
|
||||
let paths: Vec<PathBuf> = std::env::args().skip(1).map(PathBuf::from).collect();
|
||||
if paths.is_empty() {
|
||||
eprintln!("usage: darkroom-desktop <file-or-directory>...");
|
||||
}
|
||||
|
||||
dr_ui::run(paths)?;
|
||||
|
||||
// Skip Rust's normal static/thread-local teardown on the way out: a
|
||||
// background zbus/keyring connection opened by dr_ui::launch_ui can
|
||||
// still be alive here, and unwinding through it races its async-io
|
||||
// reactor thread, panicking with "thread local ... during or after
|
||||
// destruction" when the window is closed.
|
||||
std::process::exit(0);
|
||||
}
|
||||
@@ -1,34 +0,0 @@
|
||||
[package]
|
||||
name = "dr-catalog"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
license.workspace = true
|
||||
|
||||
[dependencies]
|
||||
dr-types.workspace = true
|
||||
# The face subsystem's arithmetic — `Calibration` in particular, so the sigmoid
|
||||
# that turns a cosine into a probability has exactly one definition. Default
|
||||
# features are off, so this brings in no ONNX runtime and no weights: only the
|
||||
# model-free half compiles here.
|
||||
dr-face.workspace = true
|
||||
# For `SHARD_MAX_BYTES` alone. The face shards are capped at the same 25 MB the
|
||||
# thumbnail shards are, and sharing the constant is what keeps them from
|
||||
# drifting apart — the cap is a statement about sync cost, not about thumbnails.
|
||||
dr-thumbs.workspace = true
|
||||
# The `Storage` trait, and nothing else from it. A scan has to read a real
|
||||
# directory, and this is how `core/` reaches the platform without a
|
||||
# `#[cfg(target_os)]` of its own (ARCH §4.1: calls go downward).
|
||||
dr-plat.workspace = true
|
||||
rusqlite.workspace = true
|
||||
thiserror.workspace = true
|
||||
log.workspace = true
|
||||
# `collections.selector_json` — the stored form of a smart collection's
|
||||
# selector. The column predates this dependency; nothing else here is JSON.
|
||||
serde_json.workspace = true
|
||||
|
||||
# For the `scan_local` example only, which is a diagnostic tool: what it is
|
||||
# diagnosing is often a folder the scan warned about and skipped, and those
|
||||
# warnings go to `log`.
|
||||
[dev-dependencies]
|
||||
env_logger.workspace = true
|
||||
@@ -1,420 +0,0 @@
|
||||
//! What the suggestion confidence would say about a real library.
|
||||
//!
|
||||
//! cargo run --release -p dr-catalog --example face_confidence -- CATALOG.sqlite [--full]
|
||||
//!
|
||||
//! Read-only: it writes nothing to the catalog, so it can be pointed at a copy
|
||||
//! of a live library and re-run at will.
|
||||
//!
|
||||
//! # What it measures
|
||||
//!
|
||||
//! The user's own confirmations are the only ground truth a library has, so
|
||||
//! the evaluation is leave-one-out over them: hide one confirmed face, ask the
|
||||
//! scorer which of the confirmed identities it belongs to, and compare with
|
||||
//! what the user said. Faces from the same photograph are excluded exactly as
|
||||
//! the clusterer excludes them, so nothing is scored against a co-occurrence
|
||||
//! that would never have been allowed to merge.
|
||||
//!
|
||||
//! Two numbers are compared on that task: the share (`dr_face::assign`) and the
|
||||
//! mean-within-group figure it replaced. Accuracy says which one picks the
|
||||
//! right person; the reliability table says whether the percentage the user is
|
||||
//! shown means what it claims — which is the question FR-CULL-9 exists for.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use dr_catalog::faces::{self, PersonId};
|
||||
use dr_catalog::Catalog;
|
||||
|
||||
const MODEL_ID: &str = "w600k_mbf";
|
||||
const TOP: usize = 10;
|
||||
|
||||
struct Known {
|
||||
image: u64,
|
||||
person: PersonId,
|
||||
embedding: Vec<f32>,
|
||||
crop_px: f32,
|
||||
}
|
||||
|
||||
/// What the catalog holds per face, decoded: photograph, vector, size,
|
||||
/// quality.
|
||||
type Decoded = (u64, Vec<f32>, f32, Option<f32>);
|
||||
|
||||
fn main() {
|
||||
let args: Vec<String> = std::env::args().skip(1).collect();
|
||||
let Some(path) = args.first() else {
|
||||
eprintln!("usage: face_confidence CATALOG.sqlite [--full]");
|
||||
std::process::exit(2);
|
||||
};
|
||||
|
||||
let catalog = Catalog::open(std::path::Path::new(path)).expect("open catalog");
|
||||
let conn = catalog.connection();
|
||||
|
||||
let cal = match faces::calibration(conn, MODEL_ID) {
|
||||
Ok(Some((c, _))) => c,
|
||||
_ => dr_face::Calibration::default(),
|
||||
};
|
||||
println!(
|
||||
"calibration: a={:.2} b={:.2} w_size={:.3} valid={} (P=0.5 at cosine {:.3})",
|
||||
cal.a,
|
||||
cal.b,
|
||||
cal.w_size,
|
||||
cal.valid,
|
||||
cal.boundary_at(0.5, 150.0, 0.0)
|
||||
);
|
||||
|
||||
let model = dr_face::ModelId::new(MODEL_ID.to_string());
|
||||
let stored = faces::embeddings(conn, MODEL_ID).expect("embeddings");
|
||||
let mut embedding_of: HashMap<faces::FaceId, Decoded> = HashMap::new();
|
||||
for f in stored {
|
||||
if let Some(e) = dr_face::Embedding::from_f16_bytes(model.clone(), &f.embedding) {
|
||||
embedding_of.insert(f.face, (f.image.0, e.v.to_vec(), f.crop_px, f.quality));
|
||||
}
|
||||
}
|
||||
println!("faces with embeddings: {}", embedding_of.len());
|
||||
|
||||
// The ground truth: every confirmed face, under the person the user put it
|
||||
// on. Identities with a single confirmation are dropped — leaving one out
|
||||
// leaves that identity with no evidence at all, so they measure nothing.
|
||||
let people = faces::people(conn).expect("people");
|
||||
let mut known: Vec<Known> = Vec::new();
|
||||
let mut identities = 0usize;
|
||||
for p in &people {
|
||||
if p.confirmed_faces < 2 {
|
||||
continue;
|
||||
}
|
||||
let mut mine = Vec::new();
|
||||
for f in faces::for_person(conn, p.id, false).expect("faces") {
|
||||
if !f.confirmed {
|
||||
continue;
|
||||
}
|
||||
if let Some((image, embedding, crop_px, _)) = embedding_of.get(&f.id) {
|
||||
mine.push(Known {
|
||||
image: *image,
|
||||
person: p.id,
|
||||
embedding: embedding.clone(),
|
||||
crop_px: *crop_px,
|
||||
});
|
||||
}
|
||||
}
|
||||
if mine.len() >= 2 {
|
||||
identities += 1;
|
||||
known.extend(mine);
|
||||
}
|
||||
}
|
||||
println!(
|
||||
"ground truth: {} confirmed faces across {identities} identities\n",
|
||||
known.len()
|
||||
);
|
||||
if known.len() < 2 {
|
||||
println!("not enough confirmations to evaluate.");
|
||||
return;
|
||||
}
|
||||
|
||||
let mut share_right = 0usize;
|
||||
let mut mean_right = 0usize;
|
||||
// (share of the winner, was the winner correct)
|
||||
let mut reliability: Vec<(f32, bool)> = Vec::with_capacity(known.len());
|
||||
// What each scorer would have *displayed* for the correct answer.
|
||||
let mut shown_share = Vec::with_capacity(known.len());
|
||||
let mut shown_mean = Vec::with_capacity(known.len());
|
||||
|
||||
for (i, me) in known.iter().enumerate() {
|
||||
let mut per_person: HashMap<PersonId, Vec<f32>> = HashMap::new();
|
||||
// The mean baseline is the old code's, which had no floor: it averaged
|
||||
// over every member of the group.
|
||||
let mut all_person: HashMap<PersonId, Vec<f32>> = HashMap::new();
|
||||
for (j, them) in known.iter().enumerate() {
|
||||
if i == j || me.image == them.image {
|
||||
continue;
|
||||
}
|
||||
let cos: f32 = me
|
||||
.embedding
|
||||
.iter()
|
||||
.zip(&them.embedding)
|
||||
.map(|(a, b)| a * b)
|
||||
.sum();
|
||||
// The floor the real scorer sees: `cluster_scored` scans at
|
||||
// `RIVAL_FLOOR` and `identity_shares` never learns about a pair
|
||||
// below it. Summing the near-orthogonal ones here instead of
|
||||
// dropping them is not a stricter test, it is a different
|
||||
// function — fifty identities contributing their *upper tail* of
|
||||
// noise outweigh one contributing a real match.
|
||||
let probability = cal.probability(cos, me.crop_px.min(them.crop_px), 0.0);
|
||||
if probability >= dr_face::RIVAL_FLOOR {
|
||||
per_person.entry(them.person).or_default().push(probability);
|
||||
}
|
||||
all_person.entry(them.person).or_default().push(probability);
|
||||
}
|
||||
|
||||
// The share: sum of the best TOP matches per identity, normalised.
|
||||
// Evidence, and the coherence that goes with it: the sum of the best
|
||||
// TOP matches, and their mean. dr_face::assign shows the product of
|
||||
// that mean and the identity's share of the total.
|
||||
let mut evidence: Vec<(PersonId, f32, f32)> = per_person
|
||||
.iter()
|
||||
.map(|(&p, probabilities)| {
|
||||
let mut v = probabilities.clone();
|
||||
v.sort_by(|a, b| b.total_cmp(a));
|
||||
let counted = v.len().min(TOP);
|
||||
let sum = v.iter().take(TOP).sum::<f32>();
|
||||
(p, sum, sum / counted as f32)
|
||||
})
|
||||
.collect();
|
||||
let total: f32 = evidence.iter().map(|(_, s, _)| *s).sum();
|
||||
evidence.sort_by(|a, b| b.1.total_cmp(&a.1));
|
||||
|
||||
// The number it replaced: the mean over every member of the identity.
|
||||
let mut means: Vec<(PersonId, f32)> = all_person
|
||||
.iter()
|
||||
.map(|(&p, probabilities)| {
|
||||
(
|
||||
p,
|
||||
probabilities.iter().sum::<f32>() / probabilities.len() as f32,
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
means.sort_by(|a, b| b.1.total_cmp(&a.1));
|
||||
|
||||
if let (Some(&(winner, score, coherence)), true) = (evidence.first(), total > 0.0) {
|
||||
let correct = winner == me.person;
|
||||
share_right += correct as usize;
|
||||
// What the screen would say about the identity it picked.
|
||||
reliability.push((coherence * score / total, correct));
|
||||
let ours = evidence
|
||||
.iter()
|
||||
.find(|(p, _, _)| *p == me.person)
|
||||
.map(|(_, s, c)| c * s / total)
|
||||
.unwrap_or(0.0);
|
||||
shown_share.push(ours);
|
||||
}
|
||||
if let Some(&(winner, _)) = means.first() {
|
||||
mean_right += (winner == me.person) as usize;
|
||||
shown_mean.push(
|
||||
means
|
||||
.iter()
|
||||
.find(|(p, _)| *p == me.person)
|
||||
.map(|(_, s)| *s)
|
||||
.unwrap_or(0.0),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let n = known.len() as f64;
|
||||
println!("which identity does this face belong to? (leave-one-out, top-1)");
|
||||
println!(
|
||||
" share of evidence {:>6.2}% ({share_right}/{})",
|
||||
100.0 * share_right as f64 / n,
|
||||
known.len()
|
||||
);
|
||||
println!(
|
||||
" mean within group {:>6.2}% ({mean_right}/{})\n",
|
||||
100.0 * mean_right as f64 / n,
|
||||
known.len()
|
||||
);
|
||||
|
||||
println!("what the screen would show for the answer the user gave:");
|
||||
band(" share ", &shown_share);
|
||||
band(" mean ", &shown_mean);
|
||||
|
||||
println!("\nreliability of the share — is a stated {{n}}% right {{n}}% of the time?");
|
||||
println!(
|
||||
" {:>12} {:>7} {:>9} {:>8}",
|
||||
"stated", "faces", "correct", "gap"
|
||||
);
|
||||
for (lo, hi) in [
|
||||
(0.0, 0.5),
|
||||
(0.5, 0.6),
|
||||
(0.6, 0.7),
|
||||
(0.7, 0.8),
|
||||
(0.8, 0.9),
|
||||
(0.9, 0.95),
|
||||
(0.95, 1.001),
|
||||
] {
|
||||
let bucket: Vec<bool> = reliability
|
||||
.iter()
|
||||
.filter(|(s, _)| *s >= lo && *s < hi)
|
||||
.map(|(_, c)| *c)
|
||||
.collect();
|
||||
if bucket.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let observed = bucket.iter().filter(|c| **c).count() as f64 / bucket.len() as f64;
|
||||
let stated = reliability
|
||||
.iter()
|
||||
.filter(|(s, _)| *s >= lo && *s < hi)
|
||||
.map(|(s, _)| *s as f64)
|
||||
.sum::<f64>()
|
||||
/ bucket.len() as f64;
|
||||
println!(
|
||||
" {:>5.0}–{:>3.0}% {:>9} {:>8.1}% {:>+7.1}",
|
||||
lo * 100.0,
|
||||
hi.min(1.0) * 100.0,
|
||||
bucket.len(),
|
||||
100.0 * observed,
|
||||
100.0 * (observed - stated)
|
||||
);
|
||||
}
|
||||
|
||||
if args.iter().any(|a| a == "--full") {
|
||||
// The confirmations go in as anchors, exactly as `recluster` sends
|
||||
// them: they are what makes a group a named identity, and therefore
|
||||
// what makes it a rival.
|
||||
let mut confirmed = HashMap::new();
|
||||
for p in &people {
|
||||
for f in faces::for_person(conn, p.id, false).expect("faces") {
|
||||
if f.confirmed {
|
||||
confirmed.insert(f.id, p.id.0);
|
||||
}
|
||||
}
|
||||
}
|
||||
full_library(&embedding_of, &confirmed, &cal);
|
||||
}
|
||||
}
|
||||
|
||||
/// Where a set of confidences actually falls.
|
||||
fn band(label: &str, v: &[f32]) {
|
||||
if v.is_empty() {
|
||||
return;
|
||||
}
|
||||
let mut s = v.to_vec();
|
||||
s.sort_by(|a, b| a.total_cmp(b));
|
||||
let pct = |q: f64| s[((s.len() - 1) as f64 * q) as usize];
|
||||
let mean = s.iter().sum::<f32>() / s.len() as f32;
|
||||
println!(
|
||||
"{label} median {:>5.1}% mean {:>5.1}% p10 {:>5.1}% p90 {:>5.1}% under 50%: {:>5.1}%",
|
||||
100.0 * pct(0.5),
|
||||
100.0 * mean,
|
||||
100.0 * pct(0.10),
|
||||
100.0 * pct(0.90),
|
||||
100.0 * s.iter().filter(|x| **x < 0.5).count() as f32 / s.len() as f32
|
||||
);
|
||||
}
|
||||
|
||||
/// The whole library through the real clusterer, for the numbers it would
|
||||
/// actually write.
|
||||
fn full_library(
|
||||
embedding_of: &HashMap<faces::FaceId, Decoded>,
|
||||
confirmed: &HashMap<faces::FaceId, u64>,
|
||||
cal: &dr_face::Calibration,
|
||||
) {
|
||||
let mut candidates: Vec<dr_face::Candidate> = embedding_of
|
||||
.iter()
|
||||
.map(
|
||||
|(id, (image, embedding, crop_px, quality))| dr_face::Candidate {
|
||||
face: id.0,
|
||||
image: *image,
|
||||
embedding: embedding.clone(),
|
||||
crop_px: *crop_px,
|
||||
quality: *quality,
|
||||
confirmed_person: confirmed.get(id).copied(),
|
||||
},
|
||||
)
|
||||
.collect();
|
||||
candidates.sort_by_key(|c| c.face);
|
||||
|
||||
println!("\nthe whole library, at the default merge probability:");
|
||||
|
||||
// The three phases, separately, because "a regroup takes n seconds" does
|
||||
// not tell anyone which half to optimise — and the answer differs between
|
||||
// a desktop and a tablet (docs/faces.md §9).
|
||||
{
|
||||
let dim = candidates.first().map(|c| c.embedding.len()).unwrap_or(0);
|
||||
let flat: Vec<f32> = candidates
|
||||
.iter()
|
||||
.flat_map(|c| c.embedding.clone())
|
||||
.collect();
|
||||
let crop_px: Vec<f32> = candidates.iter().map(|c| c.crop_px).collect();
|
||||
let images: Vec<u64> = candidates.iter().map(|c| c.image).collect();
|
||||
let gallery: Vec<bool> = candidates.iter().map(|c| c.in_gallery()).collect();
|
||||
let view = dr_face::neighbours::Faces {
|
||||
embeddings: &flat,
|
||||
dim,
|
||||
crop_px: &crop_px,
|
||||
images: &images,
|
||||
gallery: &gallery,
|
||||
};
|
||||
|
||||
let t = std::time::Instant::now();
|
||||
let evidence = dr_face::neighbours::above_threshold(&view, cal, dr_face::RIVAL_FLOOR);
|
||||
let scan = t.elapsed().as_secs_f64();
|
||||
|
||||
// `cluster` runs its own scan at the merge threshold, so the
|
||||
// agglomeration is what is left after taking one scan off the total.
|
||||
let t = std::time::Instant::now();
|
||||
let clusters = dr_face::cluster(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY);
|
||||
let agglomerate = t.elapsed().as_secs_f64() - scan;
|
||||
|
||||
let t = std::time::Instant::now();
|
||||
let _ = dr_face::identity_shares(&gallery, &clusters, &evidence, dr_face::TOP_MATCHES);
|
||||
println!(
|
||||
" scan {scan:.2}s ({} evidence pairs) · agglomerate {agglomerate:.2}s · score {:.2}s",
|
||||
evidence.len(),
|
||||
t.elapsed().as_secs_f64()
|
||||
);
|
||||
}
|
||||
|
||||
let start = std::time::Instant::now();
|
||||
let grouping = dr_face::cluster_scored(&candidates, cal, dr_face::DEFAULT_MERGE_PROBABILITY);
|
||||
let real: Vec<_> = grouping
|
||||
.clusters
|
||||
.iter()
|
||||
.filter(|c| c.members.len() >= 2)
|
||||
.collect();
|
||||
let grouped: usize = real.iter().map(|c| c.members.len()).sum();
|
||||
println!(
|
||||
" {} face(s) → {} group(s) of two or more, holding {grouped} faces ({:.0}%), in {:.1}s",
|
||||
candidates.len(),
|
||||
real.len(),
|
||||
100.0 * grouped as f64 / candidates.len() as f64,
|
||||
start.elapsed().as_secs_f64()
|
||||
);
|
||||
|
||||
let named: Vec<_> = real.iter().filter(|c| c.person.is_some()).collect();
|
||||
println!(
|
||||
" {} of those group(s) carry a confirmation, holding {} faces",
|
||||
named.len(),
|
||||
named.iter().map(|c| c.members.len()).sum::<usize>()
|
||||
);
|
||||
|
||||
let shown: Vec<f32> = real
|
||||
.iter()
|
||||
.flat_map(|c| c.members.iter().map(|&m| grouping.confidence[m]))
|
||||
.collect();
|
||||
band(" new, all groups ", &shown);
|
||||
let onto_people: Vec<f32> = named
|
||||
.iter()
|
||||
.flat_map(|c| c.members.iter().map(|&m| grouping.confidence[m]))
|
||||
.collect();
|
||||
band(" new, onto a person", &onto_people);
|
||||
|
||||
// The number the old code would have written for the same grouping.
|
||||
let means: Vec<f32> = real
|
||||
.iter()
|
||||
.flat_map(|c| {
|
||||
c.members.iter().map(|&m| {
|
||||
let me = &candidates[m];
|
||||
let mut sum = 0.0;
|
||||
let mut n = 0.0;
|
||||
for &other in &c.members {
|
||||
if other == m {
|
||||
continue;
|
||||
}
|
||||
let them = &candidates[other];
|
||||
let cos: f32 = me
|
||||
.embedding
|
||||
.iter()
|
||||
.zip(&them.embedding)
|
||||
.map(|(a, b)| a * b)
|
||||
.sum();
|
||||
sum += cal.probability(cos, me.crop_px.min(them.crop_px), 0.0);
|
||||
n += 1.0;
|
||||
}
|
||||
if n == 0.0 {
|
||||
1.0
|
||||
} else {
|
||||
sum / n
|
||||
}
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
band(" old, all groups ", &means);
|
||||
}
|
||||
@@ -1,111 +0,0 @@
|
||||
//! Scan a real folder on this machine into a catalog, and say what it cost.
|
||||
//!
|
||||
//! cargo run -p dr-catalog --example scan_local -- ~/Pictures [catalog.sqlite]
|
||||
//!
|
||||
//! **Run it twice.** The first run is a full walk; the second is the one worth
|
||||
//! watching, because on an unchanged library it should list no directories at
|
||||
//! all and take a fraction of the time. That difference is NFR-P1, and a
|
||||
//! synthetic test cannot show it at the scale a real library does — 121,785
|
||||
//! files in a synced folder is a different question from twenty in a temporary
|
||||
//! directory.
|
||||
//!
|
||||
//! Writes only to the catalog file, which defaults to a fixed path in the
|
||||
//! system temporary directory so a second run has something to compare
|
||||
//! against. Nothing in the scanned folder is touched.
|
||||
|
||||
use std::path::PathBuf;
|
||||
|
||||
use dr_catalog::walk::{ensure_root, scan_root, RootKind};
|
||||
use dr_catalog::Catalog;
|
||||
use dr_plat::LocalStorage;
|
||||
use dr_types::FormatFilter;
|
||||
|
||||
fn main() {
|
||||
env_logger::init();
|
||||
|
||||
let mut args = std::env::args().skip(1);
|
||||
let Some(dir) = args.next().map(PathBuf::from) else {
|
||||
eprintln!("usage: scan_local <directory> [catalog.sqlite]");
|
||||
std::process::exit(2);
|
||||
};
|
||||
let catalog_path = args
|
||||
.next()
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| std::env::temp_dir().join("darkroom-scan-local.sqlite"));
|
||||
|
||||
let catalog = match Catalog::open(&catalog_path) {
|
||||
Ok(c) => c,
|
||||
Err(e) => {
|
||||
eprintln!("cannot open {}: {e}", catalog_path.display());
|
||||
std::process::exit(1);
|
||||
}
|
||||
};
|
||||
println!("catalog: {}", catalog_path.display());
|
||||
|
||||
// The label is how the grant is spelled, and the only place a path is
|
||||
// written down. Everything after this line addresses files by `RootId`.
|
||||
let label = dir.display().to_string();
|
||||
let root = match ensure_root(catalog.connection(), RootKind::Local, &label) {
|
||||
Ok(r) => r,
|
||||
Err(e) => {
|
||||
eprintln!("cannot record the root: {e}");
|
||||
std::process::exit(1);
|
||||
}
|
||||
};
|
||||
let storage = LocalStorage::with_root(root, &dir);
|
||||
|
||||
let now = std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_secs() as i64)
|
||||
.unwrap_or(0);
|
||||
|
||||
let started = std::time::Instant::now();
|
||||
let report = match scan_root(
|
||||
catalog.connection(),
|
||||
&storage,
|
||||
root,
|
||||
&FormatFilter::all(),
|
||||
now,
|
||||
|| false,
|
||||
|p| {
|
||||
// One line per hundred directories: enough to show it is alive on a
|
||||
// large library, not enough to be the thing that slows it down.
|
||||
let visited = p.directories_listed + p.directories_pruned;
|
||||
if visited % 100 == 0 {
|
||||
println!(
|
||||
" … {visited} directories ({} pruned), {} images",
|
||||
p.directories_pruned, p.images_found
|
||||
);
|
||||
}
|
||||
},
|
||||
) {
|
||||
Ok(r) => r,
|
||||
Err(e) => {
|
||||
eprintln!("scan failed: {e}");
|
||||
std::process::exit(1);
|
||||
}
|
||||
};
|
||||
let elapsed = started.elapsed();
|
||||
|
||||
let total: i64 = catalog
|
||||
.connection()
|
||||
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
|
||||
.unwrap_or(-1);
|
||||
|
||||
println!("\noutcome: {:?}", report.outcome);
|
||||
println!(
|
||||
"directories: {} listed, {} pruned",
|
||||
report.progress.directories_listed, report.progress.directories_pruned
|
||||
);
|
||||
println!(
|
||||
"images: {} new, {} changed, {} unchanged, {} removed",
|
||||
report.inserted, report.updated, report.unchanged, report.images_removed
|
||||
);
|
||||
println!("folders: {} removed", report.folders_removed);
|
||||
println!("catalogued: {total} in total");
|
||||
println!("took: {:.2?}", elapsed);
|
||||
|
||||
if report.progress.directories_listed == 0 && report.progress.directories_pruned > 0 {
|
||||
println!("\nnothing had changed: every folder was proven unchanged by one probe");
|
||||
}
|
||||
}
|
||||
@@ -1,111 +0,0 @@
|
||||
//! What a second device ends up with after adopting this library.
|
||||
//!
|
||||
//! Stands up an empty catalog, gives it the images the real one has, adopts the
|
||||
//! face shards into it exactly as a sync would, merges the real catalog in as a
|
||||
//! remote — and then counts. The point is to answer "why does the tablet show
|
||||
//! fewer faces for this person" without needing the tablet.
|
||||
//!
|
||||
//! cargo run -p dr-catalog --example sync_probe -- CATALOG.sqlite FACES_DIR
|
||||
|
||||
use std::path::PathBuf;
|
||||
|
||||
use dr_catalog::face_shard::{self, FaceShardStore};
|
||||
use dr_catalog::Catalog;
|
||||
|
||||
const MODEL: &str = "w600k_mbf";
|
||||
|
||||
fn main() {
|
||||
let args: Vec<String> = std::env::args().skip(1).collect();
|
||||
if args.len() < 2 {
|
||||
eprintln!("usage: sync_probe CATALOG.sqlite FACES_DIR");
|
||||
std::process::exit(2);
|
||||
}
|
||||
let source = PathBuf::from(&args[0]);
|
||||
let faces_dir = PathBuf::from(&args[1]);
|
||||
|
||||
let dir = std::env::temp_dir().join(format!("dr-sync-probe-{}", std::process::id()));
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
let dest = dir.join("catalog.sqlite");
|
||||
|
||||
let far = Catalog::open(&dest).expect("fresh catalog");
|
||||
let conn = far.connection();
|
||||
|
||||
// The images a scan would have found. Nothing else: no faces, no people.
|
||||
conn.execute(
|
||||
"ATTACH DATABASE ?1 AS src",
|
||||
[source.to_string_lossy().as_ref()],
|
||||
)
|
||||
.unwrap();
|
||||
// Foreign keys off for the copy: `images` carries self-references
|
||||
// (`shadowed_by`) that are only consistent once every row is in, and this
|
||||
// is a bulk clone rather than an edit.
|
||||
conn.execute_batch(
|
||||
"PRAGMA foreign_keys = OFF;
|
||||
INSERT INTO roots SELECT * FROM src.roots;
|
||||
INSERT INTO images SELECT * FROM src.images;
|
||||
INSERT INTO remote SELECT * FROM src.remote;
|
||||
PRAGMA foreign_keys = ON;",
|
||||
)
|
||||
.unwrap();
|
||||
let images: i64 = conn
|
||||
.query_row("SELECT COUNT(*) FROM images", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
conn.execute_batch("DETACH DATABASE src").unwrap();
|
||||
println!("second device starts with {images} image(s), no faces");
|
||||
|
||||
// Adopt every shard, which is what a completed face sync leaves behind.
|
||||
let store = FaceShardStore::open(&faces_dir).expect("shard store");
|
||||
let adopted = face_shard::import_from_shards(conn, &store, MODEL).expect("import");
|
||||
let faces: i64 = conn
|
||||
.query_row("SELECT COUNT(*) FROM faces", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
println!("adopted {adopted} image(s) from the shards -> {faces} face(s)");
|
||||
|
||||
// Then the catalog merge, which is where people and their judgements come.
|
||||
let report = dr_catalog::sync::merge_remote(conn, &source).expect("merge");
|
||||
println!(
|
||||
"merge: {} people in, {} updated, {} kept local, {} face(s) assigned, \
|
||||
{} kept local, {} rejection(s)",
|
||||
report.people_inserted,
|
||||
report.people_updated,
|
||||
report.people_kept_local,
|
||||
report.faces_assigned,
|
||||
report.faces_kept_local,
|
||||
report.faces_rejected,
|
||||
);
|
||||
|
||||
// Per person, against what the source holds.
|
||||
conn.execute(
|
||||
"ATTACH DATABASE ?1 AS src",
|
||||
[source.to_string_lossy().as_ref()],
|
||||
)
|
||||
.unwrap();
|
||||
let mut q = conn
|
||||
.prepare(
|
||||
"SELECT p.name,
|
||||
(SELECT COUNT(*) FROM src.face_person sfp
|
||||
JOIN src.people sp ON sp.id = sfp.person_id
|
||||
WHERE sp.uuid = p.uuid) AS there,
|
||||
(SELECT COUNT(*) FROM face_person fp WHERE fp.person_id = p.id) AS here
|
||||
FROM people p
|
||||
WHERE p.name != ''
|
||||
ORDER BY there DESC LIMIT 12",
|
||||
)
|
||||
.unwrap();
|
||||
println!("\n{:<24} {:>8} {:>8}", "person", "source", "here");
|
||||
let rows = q
|
||||
.query_map([], |r| {
|
||||
Ok((
|
||||
r.get::<_, String>(0)?,
|
||||
r.get::<_, i64>(1)?,
|
||||
r.get::<_, i64>(2)?,
|
||||
))
|
||||
})
|
||||
.unwrap();
|
||||
for row in rows.flatten() {
|
||||
println!("{:<24} {:>8} {:>8}", row.0, row.1, row.2);
|
||||
}
|
||||
|
||||
println!("\nprobe catalog left at {}", dest.display());
|
||||
}
|
||||
@@ -1,308 +0,0 @@
|
||||
//! TRACES: FR-CAT-11
|
||||
//! Has this photograph been imported before?
|
||||
//!
|
||||
//! Two tiers, because neither alone is enough and they cost very different
|
||||
//! amounts. The metadata tier — capture time, camera, size, the name the
|
||||
//! camera gave it — is answerable from the catalog before a byte leaves the
|
||||
//! card, which is what makes re-inserting an already-imported card cost a
|
||||
//! metadata read per file rather than a full transfer. The content tier
|
||||
//! catches what the first misses: the same frame arriving under a different
|
||||
//! name, from a second card, or after somebody renamed it.
|
||||
//!
|
||||
//! # Why the filename is compared here rather than in SQL
|
||||
//!
|
||||
//! `images.source_ref` holds the whole opaque key — a relative path on Linux,
|
||||
//! a document id on SAF — and the camera's filename is only its last
|
||||
//! component. Matching that in SQL means `LIKE '%/IMG_0001.CR3'`, which cannot
|
||||
//! use an index, scans the whole table, and is wrong on SAF where the
|
||||
//! separator is not `/`. So the query narrows on the indexed columns and the
|
||||
//! handful of rows that survive are compared in Rust, the same way the grid
|
||||
//! already derives a display name.
|
||||
//!
|
||||
//! # Filename alone is never sufficient
|
||||
//!
|
||||
//! Camera filenames wrap at `IMG_9999` and start again, so a library of any
|
||||
//! age holds several unrelated `IMG_0001.CR3`. That is why the cheap tier
|
||||
//! carries capture time and camera as well, and why the expensive tier exists
|
||||
//! at all.
|
||||
|
||||
use rusqlite::Connection;
|
||||
|
||||
use crate::CatalogError;
|
||||
|
||||
/// The last component of a stored source reference.
|
||||
///
|
||||
/// Splits on both separators for the same reason `Catalog::window` does: the
|
||||
/// key's shape belongs to the storage that produced it, and a SAF document id
|
||||
/// is delimited with `:`.
|
||||
fn file_name(source_ref: &str) -> &str {
|
||||
source_ref.rsplit(['/', ':']).next().unwrap_or(source_ref)
|
||||
}
|
||||
|
||||
/// Whether the catalog already holds this photograph, on metadata alone.
|
||||
///
|
||||
/// `camera` is the joined make-and-model string the scan stores, not the raw
|
||||
/// EXIF pair — the caller composes it the same way, or the comparison is
|
||||
/// always false.
|
||||
///
|
||||
/// A `captured_at` of `None` makes this answer `false` rather than matching
|
||||
/// every undated image in the library: without a capture time the key is
|
||||
/// filename plus size, which two frames from the same body collide on
|
||||
/// routinely. An undated file falls through to the content tier, which is
|
||||
/// slower and right.
|
||||
pub fn seen_by_metadata(
|
||||
conn: &Connection,
|
||||
captured_at: Option<i64>,
|
||||
camera: Option<&str>,
|
||||
size: u64,
|
||||
original_name: &str,
|
||||
) -> Result<bool, CatalogError> {
|
||||
let Some(captured_at) = captured_at else {
|
||||
return Ok(false);
|
||||
};
|
||||
|
||||
// `images_captured` indexes the capture time, so this reads a few rows
|
||||
// even in a library of fifty thousand: one instant to the second holds
|
||||
// one frame, or a handful on a body shooting a burst.
|
||||
let mut stmt = conn.prepare(
|
||||
"SELECT source_ref FROM images
|
||||
WHERE captured_at = ?1
|
||||
AND (?2 IS NULL OR camera IS ?2)
|
||||
AND (file_size IS NULL OR file_size = ?3)",
|
||||
)?;
|
||||
let mut rows = stmt.query(rusqlite::params![captured_at, camera, size as i64])?;
|
||||
while let Some(row) = rows.next()? {
|
||||
let source_ref: String = row.get(0)?;
|
||||
if file_name(&source_ref).eq_ignore_ascii_case(original_name) {
|
||||
return Ok(true);
|
||||
}
|
||||
}
|
||||
Ok(false)
|
||||
}
|
||||
|
||||
/// Whether these exact bytes are already in the library.
|
||||
///
|
||||
/// The tier that costs a read of the file. Cheap here — `images_hash` is a
|
||||
/// partial index over the rows that have one — and expensive for the caller,
|
||||
/// which had to hash something to ask.
|
||||
pub fn seen_by_content(conn: &Connection, digest: &str) -> Result<bool, CatalogError> {
|
||||
let n: i64 = conn.query_row(
|
||||
"SELECT COUNT(*) FROM images WHERE content_hash = ?1",
|
||||
[digest],
|
||||
|r| r.get(0),
|
||||
)?;
|
||||
Ok(n > 0)
|
||||
}
|
||||
|
||||
/// Record the digest of a file the import computed.
|
||||
///
|
||||
/// An import reads every byte anyway, so the hash is free at that moment and
|
||||
/// costs a full read of an 80 MB file at any other. Storing it is what lets
|
||||
/// the *next* import answer [`seen_by_content`] without reading anything.
|
||||
///
|
||||
/// Matched on `source_ref` within a root, which is how the scan that just
|
||||
/// catalogued the imported file identifies it. Returns how many rows were
|
||||
/// updated: zero means the scan has not reached the file yet, which is a
|
||||
/// normal race and not an error.
|
||||
pub fn set_content_hash(
|
||||
conn: &Connection,
|
||||
root_id: u64,
|
||||
source_ref: &str,
|
||||
digest: &str,
|
||||
) -> Result<usize, CatalogError> {
|
||||
Ok(conn.execute(
|
||||
"UPDATE images SET content_hash = ?3
|
||||
WHERE root_id = ?1 AND source_ref = ?2",
|
||||
rusqlite::params![root_id as i64, source_ref, digest],
|
||||
)?)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::Catalog;
|
||||
|
||||
/// A catalog holding one photograph, as a scan plus a metadata pass would
|
||||
/// leave it.
|
||||
fn with_one() -> Catalog {
|
||||
let cat = Catalog::in_memory().unwrap();
|
||||
let c = cat.connection();
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO images(root_id, source_ref, captured_at, camera, file_size,
|
||||
content_hash, added_at)
|
||||
VALUES (1, '2026/2026-08-22/IMG_0001.CR3', 1787407200, 'Canon EOS R5',
|
||||
9, 'deadbeef', 0)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
cat
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn re_inserting_the_same_card_is_recognised_before_a_transfer() {
|
||||
let cat = with_one();
|
||||
assert!(seen_by_metadata(
|
||||
cat.connection(),
|
||||
Some(1_787_407_200),
|
||||
Some("Canon EOS R5"),
|
||||
9,
|
||||
"IMG_0001.CR3"
|
||||
)
|
||||
.unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_different_frame_at_the_same_instant_is_not_a_duplicate() {
|
||||
// Two bodies firing together, or a burst. The name separates them.
|
||||
let cat = with_one();
|
||||
assert!(!seen_by_metadata(
|
||||
cat.connection(),
|
||||
Some(1_787_407_200),
|
||||
Some("Canon EOS R5"),
|
||||
9,
|
||||
"IMG_0002.CR3"
|
||||
)
|
||||
.unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_same_name_from_a_different_camera_is_not_a_duplicate() {
|
||||
// IMG_0001.CR3 exists on every card ever formatted.
|
||||
let cat = with_one();
|
||||
assert!(!seen_by_metadata(
|
||||
cat.connection(),
|
||||
Some(1_787_407_200),
|
||||
Some("NIKON Z 9"),
|
||||
9,
|
||||
"IMG_0001.CR3"
|
||||
)
|
||||
.unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_same_name_at_a_different_time_is_not_a_duplicate() {
|
||||
// The IMG_9999 wrap: the library holds an unrelated IMG_0001.CR3 from
|
||||
// four years ago, and matching on name alone would refuse to import
|
||||
// today's.
|
||||
let cat = with_one();
|
||||
assert!(!seen_by_metadata(
|
||||
cat.connection(),
|
||||
Some(1_600_000_000),
|
||||
Some("Canon EOS R5"),
|
||||
9,
|
||||
"IMG_0001.CR3"
|
||||
)
|
||||
.unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_undated_file_falls_through_to_the_content_tier() {
|
||||
// Not "matches everything undated" — that would silently refuse to
|
||||
// import a whole card of scanned film.
|
||||
let cat = with_one();
|
||||
assert!(!seen_by_metadata(cat.connection(), None, None, 9, "IMG_0001.CR3").unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_file_that_grew_is_not_the_one_already_held() {
|
||||
// A truncated earlier import, or a different rendition of the same
|
||||
// frame. Same instant, same camera, same name, different bytes.
|
||||
let cat = with_one();
|
||||
assert!(!seen_by_metadata(
|
||||
cat.connection(),
|
||||
Some(1_787_407_200),
|
||||
Some("Canon EOS R5"),
|
||||
1234,
|
||||
"IMG_0001.CR3"
|
||||
)
|
||||
.unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_row_with_no_recorded_size_still_matches() {
|
||||
// The scan stores a size, but a row merged from another device may
|
||||
// not have one, and refusing to match it would re-import the library.
|
||||
let cat = with_one();
|
||||
cat.connection()
|
||||
.execute("UPDATE images SET file_size = NULL", [])
|
||||
.unwrap();
|
||||
assert!(seen_by_metadata(
|
||||
cat.connection(),
|
||||
Some(1_787_407_200),
|
||||
Some("Canon EOS R5"),
|
||||
9,
|
||||
"IMG_0001.CR3"
|
||||
)
|
||||
.unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_same_frame_renamed_is_caught_by_its_bytes() {
|
||||
let cat = with_one();
|
||||
// The metadata tier misses it...
|
||||
assert!(!seen_by_metadata(
|
||||
cat.connection(),
|
||||
Some(1_787_407_200),
|
||||
Some("Canon EOS R5"),
|
||||
9,
|
||||
"holiday-42.CR3"
|
||||
)
|
||||
.unwrap());
|
||||
// ...and the content tier does not.
|
||||
assert!(seen_by_content(cat.connection(), "deadbeef").unwrap());
|
||||
assert!(!seen_by_content(cat.connection(), "cafe").unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_digest_recorded_now_answers_the_next_import() {
|
||||
let cat = Catalog::in_memory().unwrap();
|
||||
let c = cat.connection();
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO images(root_id, source_ref, added_at)
|
||||
VALUES (1, '2026/2026-08-22/IMG_0001.CR3', 0)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert!(!seen_by_content(c, "abc123").unwrap());
|
||||
let n = set_content_hash(c, 1, "2026/2026-08-22/IMG_0001.CR3", "abc123").unwrap();
|
||||
assert_eq!(n, 1);
|
||||
assert!(seen_by_content(c, "abc123").unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recording_a_digest_before_the_scan_arrives_is_not_an_error() {
|
||||
// The import writes the file and the scan catalogues it; between those
|
||||
// two moments there is no row to update, and that is a race rather
|
||||
// than a failure.
|
||||
let cat = Catalog::in_memory().unwrap();
|
||||
let c = cat.connection();
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
set_content_hash(c, 1, "not/scanned/yet.CR3", "abc").unwrap(),
|
||||
0
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_name_is_the_last_component_of_either_kind_of_key() {
|
||||
assert_eq!(file_name("2026/2026-08-22/IMG_0001.CR3"), "IMG_0001.CR3");
|
||||
// A SAF document id delimits with a colon.
|
||||
assert_eq!(file_name("primary:DCIM/Camera/IMG_1.CR3"), "IMG_1.CR3");
|
||||
assert_eq!(file_name("IMG_0001.CR3"), "IMG_0001.CR3");
|
||||
}
|
||||
}
|
||||
@@ -1,134 +0,0 @@
|
||||
//! TRACES: NFR-ARCH-4 | NFR-R5 | NFR-R6
|
||||
//! Catalog errors.
|
||||
//!
|
||||
//! Typed and attached to the affected subject rather than panicking — a
|
||||
//! corrupt row or a failed job marks one image and lets the batch continue.
|
||||
//!
|
||||
//! # Why `From<rusqlite::Error>` is written by hand
|
||||
//!
|
||||
//! One class of SQLite failure is not about the statement that hit it: when
|
||||
//! the file itself is damaged, *every* query fails, and which one the user
|
||||
//! happened to trigger first says nothing. Before this, corruption reached the
|
||||
//! interface as whatever `Sqlite(...)` the first failing query produced —
|
||||
//! "database disk image is malformed" attached to a thumbnail refresh — and
|
||||
//! there was nowhere to hang a recovery offer.
|
||||
//!
|
||||
//! So the conversion classifies rather than wraps: `SQLITE_CORRUPT` and
|
||||
//! `SQLITE_NOTADB` become [`CatalogError::Corrupt`] wherever they arise, which
|
||||
//! means a background job that trips over the damage reports the same thing
|
||||
//! the startup check does (see [`crate::recovery`]).
|
||||
|
||||
/// Something went wrong talking to the catalog.
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum CatalogError {
|
||||
#[error("sqlite: {0}")]
|
||||
Sqlite(#[source] rusqlite::Error),
|
||||
|
||||
/// The catalog file is damaged.
|
||||
///
|
||||
/// Its own variant because it is the one error with a *user-facing
|
||||
/// remedy*: restore the NFR-R2 backup, or discard the index and rebuild it
|
||||
/// from sources plus sidecars (NFR-R6, invariant §5.2.4). Every other
|
||||
/// variant here is either a caller's mistake or a fact about one row.
|
||||
#[error("the catalog file is damaged: {detail}")]
|
||||
Corrupt { detail: String },
|
||||
|
||||
/// The catalog was written by a newer build.
|
||||
///
|
||||
/// Opening it read-write would corrupt state this build cannot represent,
|
||||
/// so the app refuses and says so (NFR-R5).
|
||||
#[error("catalog schema v{found} is newer than this build supports (v{supported})")]
|
||||
SchemaTooNew { found: i64, supported: i64 },
|
||||
|
||||
/// A scan could not reach a root at all.
|
||||
///
|
||||
/// Distinct from "files are missing": this aborts the scan *before* the
|
||||
/// deletion sweep, because every folder would look unreached and the sweep
|
||||
/// would delete the whole library (FR-CAT-9).
|
||||
#[error("root {0} is unreachable; scan aborted without pruning")]
|
||||
RootUnreachable(u64),
|
||||
|
||||
/// A scan was asked for a root the catalog has no row for.
|
||||
///
|
||||
/// A caller's mistake rather than a user's: the row is created when the
|
||||
/// grant is obtained, because the label — the path, the tree URI — is known
|
||||
/// only there. Inventing one here would file the library under a name
|
||||
/// nothing else would look it up by.
|
||||
#[error("no such root: {0}")]
|
||||
NoSuchRoot(u64),
|
||||
|
||||
/// A smart collection whose selector references itself, directly or via
|
||||
/// another collection.
|
||||
#[error("collection {0} would form a cycle")]
|
||||
CollectionCycle(u64),
|
||||
|
||||
#[error("no such collection: {0}")]
|
||||
NoSuchCollection(u64),
|
||||
|
||||
/// A keyword the caller named is gone — deleted, or fused into another by a
|
||||
/// merge while its id sat in a UI model.
|
||||
///
|
||||
/// Its own variant rather than a silent no-op because the two are different
|
||||
/// answers to the user: a rename that quietly did nothing looks exactly like
|
||||
/// a rename that did not take.
|
||||
#[error("no such keyword: {0}")]
|
||||
NoSuchKeyword(u64),
|
||||
|
||||
/// Images were dropped onto a smart collection.
|
||||
///
|
||||
/// A smart collection's membership *is* its selector, so member rows would
|
||||
/// be a second source of truth that nothing reads. Refused rather than
|
||||
/// silently discarded, so the UI can say why the drop did nothing.
|
||||
#[error("collection {0} is a saved filter; its contents cannot be edited by hand")]
|
||||
SmartCollectionNotEditable(u64),
|
||||
|
||||
#[error("malformed stored selector: {0}")]
|
||||
BadSelector(String),
|
||||
|
||||
/// A name the user typed that cannot be stored — blank, or one a sibling
|
||||
/// already holds.
|
||||
///
|
||||
/// Its own variant rather than a reused `BadSelector`, because this one is
|
||||
/// shown to the user verbatim: it has to read as a sentence about their
|
||||
/// collection, not as a diagnostic about a stored selector.
|
||||
#[error("{0}")]
|
||||
BadName(String),
|
||||
|
||||
#[error("io: {0}")]
|
||||
Io(String),
|
||||
}
|
||||
|
||||
impl From<rusqlite::Error> for CatalogError {
|
||||
fn from(e: rusqlite::Error) -> Self {
|
||||
if is_corruption(&e) {
|
||||
// `to_string` rather than keeping the error: the detail is going
|
||||
// into a dialog and into a log line, and the recovery path has no
|
||||
// use for the rusqlite type once it knows the file is damaged.
|
||||
CatalogError::Corrupt {
|
||||
detail: e.to_string(),
|
||||
}
|
||||
} else {
|
||||
CatalogError::Sqlite(e)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a SQLite failure means the *file* is damaged rather than the
|
||||
/// statement wrong.
|
||||
///
|
||||
/// `SQLITE_NOTADB` is included because it is what a truncated or overwritten
|
||||
/// catalog produces — SQLite cannot read the header, so it declines to call it
|
||||
/// a database at all. To a user those are the same accident, and the same two
|
||||
/// offers answer both.
|
||||
///
|
||||
/// Deliberately *not* included: `SQLITE_CANTOPEN` (a missing file, which
|
||||
/// `Connection::open` fixes by creating one), `SQLITE_BUSY`, and
|
||||
/// `SQLITE_IOERR` — a failing disk or a dropped network mount is a different
|
||||
/// problem, and telling the user to rebuild their index would be a wrong
|
||||
/// answer delivered confidently.
|
||||
fn is_corruption(e: &rusqlite::Error) -> bool {
|
||||
matches!(
|
||||
e.sqlite_error_code(),
|
||||
Some(rusqlite::ErrorCode::DatabaseCorrupt) | Some(rusqlite::ErrorCode::NotADatabase)
|
||||
)
|
||||
}
|
||||
@@ -1,871 +0,0 @@
|
||||
//! TRACES: FR-CAT-3 | NFR-ARCH-2 | FR-PLAT-AND-3
|
||||
//! The background work queue.
|
||||
//!
|
||||
//! Jobs live in the catalog, so they survive process death — routine on
|
||||
//! Android rather than exceptional (FR-PLAT-AND-3). Two properties carry the
|
||||
//! design:
|
||||
//!
|
||||
//! - **Coalescing.** `UNIQUE(kind, subject_id)` makes enqueueing idempotent,
|
||||
//! so every code path that notices a change can just enqueue and let the
|
||||
//! table absorb the redundancy.
|
||||
//! - **Priority shared with the GPU scheduler** (ARCH §5.3), so one notion of
|
||||
//! urgency governs the whole app and visible work always preempts bulk work.
|
||||
//!
|
||||
//! Nothing in this file runs a job. [`crate::runner`] is the other half — the
|
||||
//! one that claims from this table, does the work through a handler, and
|
||||
//! reports back. Worth knowing because for a long time it did not exist: every
|
||||
//! producer called [`enqueue`] and nothing ever called [`claim_next`], so the
|
||||
//! table only ever grew.
|
||||
|
||||
use rusqlite::{Connection, OptionalExtension};
|
||||
|
||||
use crate::error::CatalogError;
|
||||
|
||||
/// What a job does.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[repr(i64)]
|
||||
pub enum JobKind {
|
||||
/// Recursive incremental scan from a folder (§scan).
|
||||
ScanFolder = 0,
|
||||
/// Promote an image from stat-only to full EXIF.
|
||||
ExtractMetadata = 1,
|
||||
/// Build or rebuild a thumbnail.
|
||||
Thumbnail = 2,
|
||||
/// A sidecar on disk is newer than what the catalog read.
|
||||
ReadSidecar = 3,
|
||||
/// Flush a local edit to its sidecar. Debounced, never per slider tick.
|
||||
WriteSidecar = 4,
|
||||
/// Whole-file hash. On demand only — import dedup, reconnect-by-hash.
|
||||
ContentHash = 5,
|
||||
/// Range-extract an embedded preview from a remote file (FR-NC-3).
|
||||
FetchPreview = 6,
|
||||
/// Fetch a full original: pinned by rule, or explicitly asked for.
|
||||
FetchOriginal = 7,
|
||||
/// Detect and embed the faces in one image (FR-CULL-8).
|
||||
///
|
||||
/// One job does both, rather than splitting them: the proxy is already
|
||||
/// decoded and in memory, and the natural unit of resumable work is one
|
||||
/// photograph. Splitting would double the queue's row count for nothing.
|
||||
///
|
||||
/// Runs against the proxy tier, never a full decode — a library that has
|
||||
/// been browsed has already paid for its proxies, so face indexing adds no
|
||||
/// RAW decodes that were not already happening.
|
||||
DetectFaces = 8,
|
||||
}
|
||||
|
||||
impl JobKind {
|
||||
/// Every kind, so code that has to enumerate them cannot quietly miss one
|
||||
/// that was added later. A `match` would catch that; a hand-written array
|
||||
/// at each call site would not.
|
||||
pub const ALL: [JobKind; 9] = [
|
||||
JobKind::ScanFolder,
|
||||
JobKind::ExtractMetadata,
|
||||
JobKind::Thumbnail,
|
||||
JobKind::ReadSidecar,
|
||||
JobKind::WriteSidecar,
|
||||
JobKind::ContentHash,
|
||||
JobKind::FetchPreview,
|
||||
JobKind::FetchOriginal,
|
||||
JobKind::DetectFaces,
|
||||
];
|
||||
|
||||
fn from_i64(v: i64) -> Option<Self> {
|
||||
Some(match v {
|
||||
0 => JobKind::ScanFolder,
|
||||
1 => JobKind::ExtractMetadata,
|
||||
2 => JobKind::Thumbnail,
|
||||
3 => JobKind::ReadSidecar,
|
||||
4 => JobKind::WriteSidecar,
|
||||
5 => JobKind::ContentHash,
|
||||
6 => JobKind::FetchPreview,
|
||||
7 => JobKind::FetchOriginal,
|
||||
8 => JobKind::DetectFaces,
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether this job transfers over the network, and so is subject to the
|
||||
/// metered-connection and charging constraints in FR-NC-6.
|
||||
pub fn is_network(self) -> bool {
|
||||
matches!(self, JobKind::FetchPreview | JobKind::FetchOriginal)
|
||||
}
|
||||
|
||||
/// Whether `subject_id` names a row in `images`.
|
||||
///
|
||||
/// Every kind but one is per-photograph. `ScanFolder`'s subject is a
|
||||
/// *folder*, and the two id spaces are unrelated — so anything that joins
|
||||
/// `subject_id` against `images` has to exclude it, or it will read one
|
||||
/// table's ids as another's and act on the answer.
|
||||
pub fn subject_is_image(self) -> bool {
|
||||
!matches!(self, JobKind::ScanFolder)
|
||||
}
|
||||
}
|
||||
|
||||
/// Scheduling class, matching the GPU tile scheduler (ARCH §5.3).
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
|
||||
#[repr(i64)]
|
||||
pub enum Priority {
|
||||
/// Bulk work: metadata sweeps, rule-driven fetches, hashing.
|
||||
Background = 0,
|
||||
/// Just outside the viewport; the next image in culling.
|
||||
Prefetch = 1,
|
||||
/// Visible cells, and the image currently open.
|
||||
///
|
||||
/// Strictly preempts background work. Without this, scrolling during a
|
||||
/// bulk thumbnail pass misses its frame budget — the common case, not an
|
||||
/// edge case (NFR-ARCH-2).
|
||||
Interactive = 2,
|
||||
}
|
||||
|
||||
impl Priority {
|
||||
/// Read back from the stored column.
|
||||
///
|
||||
/// An unrecognised value reads as `Background` rather than failing: a
|
||||
/// priority is a hint about ordering, and refusing to run a job because
|
||||
/// its urgency is spelled oddly would be a worse answer than running it
|
||||
/// last.
|
||||
fn from_i64(v: i64) -> Self {
|
||||
match v {
|
||||
2 => Priority::Interactive,
|
||||
1 => Priority::Prefetch,
|
||||
_ => Priority::Background,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Lifecycle state.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[repr(i64)]
|
||||
pub enum JobState {
|
||||
Pending = 0,
|
||||
Running = 1,
|
||||
Failed = 2,
|
||||
}
|
||||
|
||||
/// A job ready to run.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Job {
|
||||
pub id: i64,
|
||||
pub kind: JobKind,
|
||||
pub subject_id: Option<i64>,
|
||||
pub priority: Priority,
|
||||
pub attempts: i64,
|
||||
pub payload: Option<String>,
|
||||
}
|
||||
|
||||
/// Give up after this many attempts and attach the error to the subject.
|
||||
///
|
||||
/// One corrupt file must not stall the queue behind endless retries
|
||||
/// (FR-RAW-4).
|
||||
pub const MAX_ATTEMPTS: i64 = 5;
|
||||
|
||||
/// Backoff before retrying a failed job, in seconds.
|
||||
///
|
||||
/// Exponential, capped — a server that is down for an hour should not be
|
||||
/// retried every second, and a transient decode failure should not wait an
|
||||
/// hour.
|
||||
pub fn backoff_seconds(attempts: i64) -> i64 {
|
||||
const CAP: i64 = 300;
|
||||
match attempts {
|
||||
a if a <= 0 => 0,
|
||||
a if a >= 9 => CAP,
|
||||
a => (1i64 << (a - 1)).min(CAP),
|
||||
}
|
||||
}
|
||||
|
||||
/// Enqueue work, coalescing with any identical pending job.
|
||||
///
|
||||
/// Re-requesting at a higher priority *promotes* the existing row rather than
|
||||
/// duplicating it, which is what lets the grid shout "this one is visible now"
|
||||
/// about a job already queued in the background.
|
||||
pub fn enqueue(
|
||||
conn: &Connection,
|
||||
kind: JobKind,
|
||||
subject_id: Option<i64>,
|
||||
priority: Priority,
|
||||
payload: Option<&str>,
|
||||
) -> Result<(), CatalogError> {
|
||||
conn.execute(
|
||||
"INSERT INTO jobs(kind, subject_id, priority, state, payload)
|
||||
VALUES (?1, ?2, ?3, 0, ?4)
|
||||
ON CONFLICT(kind, subject_id) DO UPDATE SET
|
||||
priority = max(jobs.priority, excluded.priority),
|
||||
-- A job that failed and is being re-requested deserves a fresh
|
||||
-- start: the file may well have changed since it failed.
|
||||
state = CASE WHEN jobs.state = 2 THEN 0 ELSE jobs.state END,
|
||||
attempts = CASE WHEN jobs.state = 2 THEN 0 ELSE jobs.attempts END,
|
||||
not_before = CASE WHEN jobs.state = 2 THEN 0 ELSE jobs.not_before END",
|
||||
rusqlite::params![kind as i64, subject_id, priority as i64, payload],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Claim the next runnable job, highest priority first.
|
||||
///
|
||||
/// `now` is passed rather than read from the clock so backoff is testable.
|
||||
pub fn claim_next(conn: &Connection, now: i64) -> Result<Option<Job>, CatalogError> {
|
||||
claim(conn, now, None)
|
||||
}
|
||||
|
||||
/// Claim the next runnable job of one of `kinds`.
|
||||
///
|
||||
/// What lets a runner take only the work it can actually do. A device with no
|
||||
/// connector must leave `FetchOriginal` rows alone rather than claim them and
|
||||
/// fail them five times each with backoff; a runner gated onto an unmetered
|
||||
/// network (FR-NC-6) passes the local kinds only and leaves the transfers
|
||||
/// where they are. Neither is expressible by filtering *after* a claim,
|
||||
/// because the claim has already marked the row `Running`.
|
||||
///
|
||||
/// An empty list claims nothing, which is the honest reading of "there is
|
||||
/// nothing this worker can do".
|
||||
pub fn claim_next_matching(
|
||||
conn: &Connection,
|
||||
now: i64,
|
||||
kinds: &[JobKind],
|
||||
) -> Result<Option<Job>, CatalogError> {
|
||||
if kinds.is_empty() {
|
||||
return Ok(None);
|
||||
}
|
||||
claim(conn, now, Some(kinds))
|
||||
}
|
||||
|
||||
/// The claim, as one statement.
|
||||
///
|
||||
/// # Why this is not a transaction around a read and a write
|
||||
///
|
||||
/// It used to be, and under two connections that is not safe in the way it
|
||||
/// looks. A deferred transaction takes a read lock for the `SELECT` and only
|
||||
/// tries to upgrade at the `UPDATE`; with WAL, a second worker that read the
|
||||
/// same snapshot gets `SQLITE_BUSY_SNAPSHOT` on its write — an error a busy
|
||||
/// handler cannot retry away, because the fix is to roll back and start over.
|
||||
/// So the queue was correct only in the sense that the loser failed loudly.
|
||||
///
|
||||
/// `UPDATE ... WHERE id = (SELECT ...) RETURNING` is one statement, so it is
|
||||
/// one implicit transaction that takes the write lock immediately. Two workers
|
||||
/// serialise, the loser waits out its busy timeout rather than erroring, and
|
||||
/// neither can see a row the other is already holding.
|
||||
fn claim(
|
||||
conn: &Connection,
|
||||
now: i64,
|
||||
kinds: Option<&[JobKind]>,
|
||||
) -> Result<Option<Job>, CatalogError> {
|
||||
// `now` first, then the kinds, matching the order the placeholders appear
|
||||
// in the text below.
|
||||
let mut args: Vec<i64> = vec![now];
|
||||
let filter = match kinds {
|
||||
None => String::new(),
|
||||
Some(kinds) => {
|
||||
// Built from the kind *count*, never from anything a user typed —
|
||||
// the same discipline `collections::descendants` uses, since
|
||||
// `carray` is not compiled in.
|
||||
let placeholders = std::iter::repeat_n("?", kinds.len())
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
args.extend(kinds.iter().map(|k| *k as i64));
|
||||
format!(" AND kind IN ({placeholders})")
|
||||
}
|
||||
};
|
||||
|
||||
let sql = format!(
|
||||
"UPDATE jobs
|
||||
SET state = 1, attempts = attempts + 1
|
||||
WHERE id = (SELECT id
|
||||
FROM jobs
|
||||
WHERE state = 0 AND not_before <= ?{filter}
|
||||
ORDER BY priority DESC, id ASC
|
||||
LIMIT 1)
|
||||
RETURNING id, kind, subject_id, priority, attempts, payload"
|
||||
);
|
||||
|
||||
let claimed = conn
|
||||
.query_row(&sql, rusqlite::params_from_iter(args.iter()), |r| {
|
||||
Ok((
|
||||
r.get::<_, i64>(0)?,
|
||||
r.get::<_, i64>(1)?,
|
||||
r.get::<_, Option<i64>>(2)?,
|
||||
r.get::<_, i64>(3)?,
|
||||
r.get::<_, i64>(4)?,
|
||||
r.get::<_, Option<String>>(5)?,
|
||||
))
|
||||
})
|
||||
.optional()?;
|
||||
|
||||
let Some((id, kind, subject_id, priority, attempts, payload)) = claimed else {
|
||||
return Ok(None);
|
||||
};
|
||||
|
||||
let Some(kind) = JobKind::from_i64(kind) else {
|
||||
// A row written by a build that knows a kind this one does not — a
|
||||
// downgrade, or a catalog synced from a newer device. Running it as
|
||||
// some other kind would be worse than not running it, so it is parked
|
||||
// where the next claim will not see it again.
|
||||
//
|
||||
// Answering `None` understates what is queued for one pass. The
|
||||
// alternative is a loop that keeps claiming the same unreadable row.
|
||||
abandon(conn, id, &format!("unknown job kind {kind}"))?;
|
||||
return Ok(None);
|
||||
};
|
||||
|
||||
Ok(Some(Job {
|
||||
id,
|
||||
kind,
|
||||
subject_id,
|
||||
priority: Priority::from_i64(priority),
|
||||
attempts,
|
||||
payload,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Job finished successfully.
|
||||
pub fn complete(conn: &Connection, id: i64) -> Result<(), CatalogError> {
|
||||
conn.execute("DELETE FROM jobs WHERE id = ?1", [id])?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Job failed. Reschedules with backoff, or gives up past [`MAX_ATTEMPTS`].
|
||||
pub fn fail(conn: &Connection, job: &Job, now: i64, err: &str) -> Result<(), CatalogError> {
|
||||
if job.attempts >= MAX_ATTEMPTS {
|
||||
abandon(conn, job.id, err)
|
||||
} else {
|
||||
conn.execute(
|
||||
"UPDATE jobs SET state = 0, not_before = ?2, last_error = ?3 WHERE id = ?1",
|
||||
rusqlite::params![job.id, now + backoff_seconds(job.attempts), err],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Give up on a job now, with no further retries.
|
||||
///
|
||||
/// For failures a retry cannot fix — the subject is gone, the payload is
|
||||
/// unreadable, the format is one this build does not know. Walking the whole
|
||||
/// retry ladder to reach a conclusion the first attempt already reached costs
|
||||
/// five wakeups and five backoffs per photograph, which on a phone is the
|
||||
/// difference the user notices.
|
||||
///
|
||||
/// The row is kept rather than deleted, because "this file failed and here is
|
||||
/// why" is something the user is entitled to see (NFR-ARCH-4).
|
||||
pub fn abandon(conn: &Connection, id: i64, err: &str) -> Result<(), CatalogError> {
|
||||
conn.execute(
|
||||
"UPDATE jobs SET state = 2, last_error = ?2 WHERE id = ?1",
|
||||
rusqlite::params![id, err],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Put a claimed job back exactly as it was found.
|
||||
///
|
||||
/// For a worker that is being stopped rather than a job that is going wrong:
|
||||
/// the platform revoked the slot, the user left the screen. The attempt the
|
||||
/// claim consumed is given back, because nothing was learned about the file —
|
||||
/// without that, five backgroundings in a row would mark good work as failed.
|
||||
///
|
||||
/// Guarded on the claim still being *this* claim. There is no owner column, so
|
||||
/// `attempts` stands in for one: it is bumped by every claim, so the row only
|
||||
/// still reads `state = 1` with the caller's own attempt number while nobody
|
||||
/// else has taken it since. A worker that comes back after the recovery pass
|
||||
/// handed its job to someone else therefore changes nothing, rather than
|
||||
/// releasing a job another worker is in the middle of.
|
||||
pub fn release(conn: &Connection, job: &Job) -> Result<(), CatalogError> {
|
||||
conn.execute(
|
||||
"UPDATE jobs SET state = 0, attempts = max(0, attempts - 1)
|
||||
WHERE id = ?1 AND state = 1 AND attempts = ?2",
|
||||
rusqlite::params![job.id, job.attempts],
|
||||
)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Recover jobs orphaned by process death.
|
||||
///
|
||||
/// A row left `Running` has no owner — the process that claimed it is gone.
|
||||
/// Called at startup, before any worker begins (FR-PLAT-AND-3).
|
||||
///
|
||||
/// The attempt the dead claim consumed is deliberately *not* refunded. A job
|
||||
/// that takes the process down with it is indistinguishable from one that
|
||||
/// fails, and the attempt counter is the only evidence that survives a death —
|
||||
/// without it a poison-pill job is reclaimed and re-run forever.
|
||||
pub fn recover_orphaned(conn: &Connection) -> Result<usize, CatalogError> {
|
||||
let n = conn.execute("UPDATE jobs SET state = 0 WHERE state = 1", [])?;
|
||||
Ok(n)
|
||||
}
|
||||
|
||||
/// Delete jobs whose photograph is gone.
|
||||
///
|
||||
/// Coalescing keeps the table one row per unit of work, but nothing shrinks it
|
||||
/// when the work stops existing: a library that has been culled carries a
|
||||
/// thumbnail job for every photograph deleted since the last time anything
|
||||
/// looked. Each one would be claimed, run, and failed five times.
|
||||
///
|
||||
/// Only kinds whose subject really is an image ([`JobKind::subject_is_image`])
|
||||
/// are considered — `ScanFolder`'s subject is a folder id, and joining it
|
||||
/// against `images` would delete jobs by coincidence of numbering.
|
||||
///
|
||||
/// There is no foreign key to do this instead. `jobs.subject_id` deliberately
|
||||
/// references nothing: it means different tables for different kinds, and a
|
||||
/// constraint that is right for eight of nine kinds is not a constraint.
|
||||
pub fn reap_orphan_subjects(conn: &Connection) -> Result<usize, CatalogError> {
|
||||
let kinds: Vec<i64> = JobKind::ALL
|
||||
.iter()
|
||||
.filter(|k| k.subject_is_image())
|
||||
.map(|k| *k as i64)
|
||||
.collect();
|
||||
let placeholders = std::iter::repeat_n("?", kinds.len())
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
|
||||
let n = conn.execute(
|
||||
&format!(
|
||||
"DELETE FROM jobs
|
||||
WHERE subject_id IS NOT NULL
|
||||
AND kind IN ({placeholders})
|
||||
AND NOT EXISTS (SELECT 1 FROM images WHERE images.id = jobs.subject_id)"
|
||||
),
|
||||
rusqlite::params_from_iter(kinds.iter()),
|
||||
)?;
|
||||
Ok(n)
|
||||
}
|
||||
|
||||
/// How much is left, by state.
|
||||
///
|
||||
/// One query rather than a listing, because the caller is a progress line: a
|
||||
/// foreground service's notification has to say how much remains without
|
||||
/// reading a hundred thousand rows to find out (FR-PLAT-AND-4).
|
||||
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Counts {
|
||||
/// Claimable now or after a backoff.
|
||||
pub pending: usize,
|
||||
/// Claimed by someone. After a clean start that is a live worker; before
|
||||
/// [`recover_orphaned`] it is a dead one.
|
||||
pub running: usize,
|
||||
/// Given up on, and kept so the user can see what failed and why.
|
||||
pub failed: usize,
|
||||
}
|
||||
|
||||
impl Counts {
|
||||
/// Work that is still going to happen.
|
||||
pub fn outstanding(&self) -> usize {
|
||||
self.pending + self.running
|
||||
}
|
||||
}
|
||||
|
||||
/// Count the queue by state.
|
||||
pub fn counts(conn: &Connection) -> Result<Counts, CatalogError> {
|
||||
// `sum` over no rows is NULL, not 0 — an empty queue would otherwise fail
|
||||
// to convert rather than counting nothing.
|
||||
let (pending, running, failed) = conn.query_row(
|
||||
"SELECT sum(state = 0), sum(state = 1), sum(state = 2) FROM jobs",
|
||||
[],
|
||||
|r| {
|
||||
Ok((
|
||||
r.get::<_, Option<i64>>(0)?,
|
||||
r.get::<_, Option<i64>>(1)?,
|
||||
r.get::<_, Option<i64>>(2)?,
|
||||
))
|
||||
},
|
||||
)?;
|
||||
Ok(Counts {
|
||||
pending: pending.unwrap_or(0) as usize,
|
||||
running: running.unwrap_or(0) as usize,
|
||||
failed: failed.unwrap_or(0) as usize,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::schema;
|
||||
|
||||
fn db() -> Connection {
|
||||
let c = Connection::open_in_memory().unwrap();
|
||||
schema::configure(&c).unwrap();
|
||||
schema::migrate(&c).unwrap();
|
||||
c
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_enqueue_coalesces() {
|
||||
let c = db();
|
||||
for _ in 0..10 {
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
}
|
||||
let n: i64 = c
|
||||
.query_row("SELECT count(*) FROM jobs", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(n, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn re_enqueueing_at_higher_priority_promotes() {
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
// The grid scrolls this image into view.
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Interactive, None).unwrap();
|
||||
|
||||
let p: i64 = c
|
||||
.query_row("SELECT priority FROM jobs", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(p, Priority::Interactive as i64);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn priority_never_regresses() {
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Interactive, None).unwrap();
|
||||
// A background sweep must not demote work the user is waiting on.
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
|
||||
let p: i64 = c
|
||||
.query_row("SELECT priority FROM jobs", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(p, Priority::Interactive as i64);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn claim_takes_highest_priority_first() {
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(2), Priority::Interactive, None).unwrap();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(3), Priority::Prefetch, None).unwrap();
|
||||
|
||||
let first = claim_next(&c, 0).unwrap().unwrap();
|
||||
assert_eq!(first.subject_id, Some(2));
|
||||
let second = claim_next(&c, 0).unwrap().unwrap();
|
||||
assert_eq!(second.subject_id, Some(3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_claimed_job_is_not_claimed_twice() {
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
assert!(claim_next(&c, 0).unwrap().is_some());
|
||||
assert!(claim_next(&c, 0).unwrap().is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn failure_backs_off_then_becomes_claimable_again() {
|
||||
let c = db();
|
||||
enqueue(
|
||||
&c,
|
||||
JobKind::FetchPreview,
|
||||
Some(1),
|
||||
Priority::Background,
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
let job = claim_next(&c, 100).unwrap().unwrap();
|
||||
fail(&c, &job, 100, "network down").unwrap();
|
||||
|
||||
// Still backing off.
|
||||
assert!(claim_next(&c, 100).unwrap().is_none());
|
||||
// Past the backoff.
|
||||
assert!(claim_next(&c, 100 + backoff_seconds(job.attempts))
|
||||
.unwrap()
|
||||
.is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_persistently_failing_job_stops_retrying() {
|
||||
let c = db();
|
||||
enqueue(
|
||||
&c,
|
||||
JobKind::ExtractMetadata,
|
||||
Some(1),
|
||||
Priority::Background,
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let mut now = 0;
|
||||
for _ in 0..MAX_ATTEMPTS {
|
||||
let job = claim_next(&c, now).unwrap().expect("should be claimable");
|
||||
fail(&c, &job, now, "corrupt file").unwrap();
|
||||
now += backoff_seconds(job.attempts);
|
||||
}
|
||||
|
||||
// One corrupt file must not stall the queue forever (FR-RAW-4).
|
||||
assert!(claim_next(&c, now + 100_000).unwrap().is_none());
|
||||
let state: i64 = c
|
||||
.query_row("SELECT state FROM jobs", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Failed as i64);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn re_requesting_a_failed_job_gives_it_a_fresh_start() {
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
let mut now = 0;
|
||||
for _ in 0..MAX_ATTEMPTS {
|
||||
let job = claim_next(&c, now).unwrap().unwrap();
|
||||
fail(&c, &job, now, "boom").unwrap();
|
||||
now += backoff_seconds(job.attempts);
|
||||
}
|
||||
// The file changed on disk, so the old failure says nothing about it.
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Interactive, None).unwrap();
|
||||
let job = claim_next(&c, now).unwrap().expect("retryable again");
|
||||
assert_eq!(job.attempts, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn orphaned_jobs_return_to_pending_on_restart() {
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
claim_next(&c, 0).unwrap().unwrap();
|
||||
// Process dies here. Android does this routinely.
|
||||
assert_eq!(recover_orphaned(&c).unwrap(), 1);
|
||||
assert!(claim_next(&c, 0).unwrap().is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn backoff_grows_then_caps() {
|
||||
assert_eq!(backoff_seconds(0), 0);
|
||||
assert_eq!(backoff_seconds(1), 1);
|
||||
assert_eq!(backoff_seconds(3), 4);
|
||||
assert_eq!(backoff_seconds(100), 300);
|
||||
}
|
||||
|
||||
/// An image row, so a job has a subject that exists.
|
||||
fn image(c: &Connection, id: i64) {
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', '/lib')
|
||||
ON CONFLICT DO NOTHING",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (?1, 1, ?2, 0)",
|
||||
rusqlite::params![id, format!("/lib/{id}.CR3")],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_runner_claims_only_the_kinds_it_names() {
|
||||
// The property a filtered claim exists for: work this worker cannot do
|
||||
// is left untouched — not claimed, not attempted, not failed.
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
enqueue(
|
||||
&c,
|
||||
JobKind::FetchOriginal,
|
||||
Some(1),
|
||||
Priority::Interactive,
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// `FetchOriginal` is the higher priority and would be claimed first by
|
||||
// an unfiltered claim. It is not this worker's to take.
|
||||
let job = claim_next_matching(&c, 0, &[JobKind::Thumbnail])
|
||||
.unwrap()
|
||||
.expect("the thumbnail is claimable");
|
||||
assert_eq!(job.kind, JobKind::Thumbnail);
|
||||
assert!(claim_next_matching(&c, 0, &[JobKind::Thumbnail])
|
||||
.unwrap()
|
||||
.is_none());
|
||||
|
||||
let (state, attempts): (i64, i64) = c
|
||||
.query_row(
|
||||
"SELECT state, attempts FROM jobs WHERE kind = ?1",
|
||||
[JobKind::FetchOriginal as i64],
|
||||
|r| Ok((r.get(0)?, r.get(1)?)),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Pending as i64);
|
||||
assert_eq!(attempts, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_worker_that_can_do_nothing_claims_nothing() {
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
assert!(claim_next_matching(&c, 0, &[]).unwrap().is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn releasing_a_claim_gives_the_attempt_back() {
|
||||
// A stopped worker has learned nothing about the file, so the claim it
|
||||
// is handing back must cost nothing.
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
|
||||
let job = claim_next(&c, 0).unwrap().unwrap();
|
||||
assert_eq!(job.attempts, 1);
|
||||
release(&c, &job).unwrap();
|
||||
|
||||
let again = claim_next(&c, 0).unwrap().expect("claimable again at once");
|
||||
assert_eq!(again.attempts, 1, "the release refunded the first attempt");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn releasing_a_job_someone_else_has_reclaimed_does_nothing() {
|
||||
// `attempts` standing in for an owner column. A worker that comes back
|
||||
// after the recovery pass handed its job to someone else must not
|
||||
// release a claim that is no longer its to release — which would drop
|
||||
// the live worker's job back into the queue to be run twice.
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
|
||||
let stale = claim_next(&c, 0).unwrap().unwrap();
|
||||
recover_orphaned(&c).unwrap();
|
||||
let live = claim_next(&c, 0).unwrap().unwrap();
|
||||
assert_eq!(live.attempts, 2);
|
||||
|
||||
release(&c, &stale).unwrap();
|
||||
|
||||
let (state, attempts): (i64, i64) = c
|
||||
.query_row("SELECT state, attempts FROM jobs", [], |r| {
|
||||
Ok((r.get(0)?, r.get(1)?))
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
state,
|
||||
JobState::Running as i64,
|
||||
"the live claim still holds"
|
||||
);
|
||||
assert_eq!(attempts, 2, "and its attempt was not refunded for it");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn abandoning_skips_the_whole_retry_ladder() {
|
||||
let c = db();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
let job = claim_next(&c, 0).unwrap().unwrap();
|
||||
|
||||
abandon(&c, job.id, "not an image this build can read").unwrap();
|
||||
|
||||
assert!(claim_next(&c, 1_000_000).unwrap().is_none());
|
||||
let (state, attempts, err): (i64, i64, String) = c
|
||||
.query_row("SELECT state, attempts, last_error FROM jobs", [], |r| {
|
||||
Ok((r.get(0)?, r.get(1)?, r.get(2)?))
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Failed as i64);
|
||||
assert_eq!(attempts, 1, "one attempt, not MAX_ATTEMPTS");
|
||||
// Kept, not deleted: the user is entitled to see what failed and why.
|
||||
assert!(err.contains("this build can read"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn jobs_for_a_deleted_photograph_are_reaped() {
|
||||
// Coalescing keeps the table one row per unit of work; nothing shrank
|
||||
// it when the work stopped existing.
|
||||
let c = db();
|
||||
image(&c, 1);
|
||||
image(&c, 2);
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(2), Priority::Background, None).unwrap();
|
||||
enqueue(
|
||||
&c,
|
||||
JobKind::ExtractMetadata,
|
||||
Some(2),
|
||||
Priority::Background,
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
c.execute("DELETE FROM images WHERE id = 2", []).unwrap();
|
||||
assert_eq!(reap_orphan_subjects(&c).unwrap(), 2);
|
||||
|
||||
let left: i64 = c
|
||||
.query_row("SELECT subject_id FROM jobs", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(left, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_folder_scan_is_not_reaped_by_image_ids() {
|
||||
// `ScanFolder`'s subject is a folder. Joining it against `images`
|
||||
// would delete it whenever the numbering happened not to collide —
|
||||
// which, on a fresh library, is almost always.
|
||||
let c = db();
|
||||
enqueue(
|
||||
&c,
|
||||
JobKind::ScanFolder,
|
||||
Some(1),
|
||||
Priority::Background,
|
||||
Some("/lib/2024"),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(reap_orphan_subjects(&c).unwrap(), 0);
|
||||
assert!(claim_next(&c, 0).unwrap().is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_job_kind_from_a_newer_build_is_parked_rather_than_guessed_at() {
|
||||
// A catalog synced from a device running a later build. Running an
|
||||
// unknown kind as some arbitrary known one is worse than not running
|
||||
// it, and the old code silently read every unknown kind as
|
||||
// `ExtractMetadata`.
|
||||
let c = db();
|
||||
c.execute(
|
||||
"INSERT INTO jobs(kind, subject_id, priority, state) VALUES (99, 1, 0, 0)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert!(claim_next(&c, 0).unwrap().is_none());
|
||||
|
||||
let (state, err): (i64, String) = c
|
||||
.query_row("SELECT state, last_error FROM jobs", [], |r| {
|
||||
Ok((r.get(0)?, r.get(1)?))
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Failed as i64);
|
||||
assert!(err.contains("99"), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn counts_say_what_is_left() {
|
||||
// What a foreground service's notification is built from: a number,
|
||||
// without reading a hundred thousand rows to find it.
|
||||
let c = db();
|
||||
assert_eq!(counts(&c).unwrap(), Counts::default());
|
||||
|
||||
for id in 1..=3 {
|
||||
enqueue(&c, JobKind::Thumbnail, Some(id), Priority::Background, None).unwrap();
|
||||
}
|
||||
let job = claim_next(&c, 0).unwrap().unwrap();
|
||||
abandon(&c, job.id, "nope").unwrap();
|
||||
claim_next(&c, 0).unwrap().unwrap();
|
||||
|
||||
let n = counts(&c).unwrap();
|
||||
assert_eq!(n.pending, 1);
|
||||
assert_eq!(n.running, 1);
|
||||
assert_eq!(n.failed, 1);
|
||||
assert_eq!(n.outstanding(), 2, "failed work is not outstanding work");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_kind_is_in_all() {
|
||||
// `ALL` is what the reap builds its kind filter from, so a kind added
|
||||
// to the enum and forgotten here would quietly stop being reaped.
|
||||
for (i, kind) in JobKind::ALL.iter().enumerate() {
|
||||
assert_eq!(
|
||||
JobKind::from_i64(i as i64),
|
||||
Some(*kind),
|
||||
"ALL is out of step with the discriminants at {i}"
|
||||
);
|
||||
}
|
||||
assert!(JobKind::from_i64(JobKind::ALL.len() as i64).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_a_folder_scan_has_a_non_image_subject() {
|
||||
assert!(!JobKind::ScanFolder.subject_is_image());
|
||||
for kind in JobKind::ALL.iter().filter(|k| **k != JobKind::ScanFolder) {
|
||||
assert!(kind.subject_is_image(), "{kind:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn network_jobs_are_identifiable_for_metered_gating() {
|
||||
// FR-NC-6: transfers respect unmetered-network and charging
|
||||
// constraints; local work must not be gated by them.
|
||||
assert!(JobKind::FetchOriginal.is_network());
|
||||
assert!(JobKind::FetchPreview.is_network());
|
||||
assert!(!JobKind::Thumbnail.is_network());
|
||||
assert!(!JobKind::ExtractMetadata.is_network());
|
||||
}
|
||||
}
|
||||
@@ -1,600 +0,0 @@
|
||||
//! TRACES: FR-CAT-2 | FR-CAT-4 | FR-CAT-6 | NFR-P1
|
||||
//! The catalog: a rebuildable index over the library.
|
||||
//!
|
||||
//! Not a source of truth. Sidecars next to the images hold the authoritative
|
||||
//! edit state (ARCH §6.12), and this file is deletable at any time — rebuilt
|
||||
//! by rescanning sources and reading sidecars. That inversion is deliberate:
|
||||
//! darktable maintains both a database and sidecars while achieving the
|
||||
//! reliability of neither.
|
||||
//!
|
||||
//! # What lives here
|
||||
//!
|
||||
//! - [`schema`] — tables and forward-only migrations
|
||||
//! - [`scan`] — incremental discovery that prunes unchanged directories
|
||||
//! - [`walk`] — those decisions driven against real storage, local or SAF
|
||||
//! - [`query`] — selectors compiled to indexed SQL, windowed for the grid
|
||||
//! - [`collections`] — the collection tree and membership the UI edits
|
||||
//! - [`keywords`] — the keyword vocabulary and what it is assigned to
|
||||
//! - [`faces`] — detected faces, the people they belong to, and who said so
|
||||
//! - [`bursts`] — frames that are one moment, grouped so they judge as one
|
||||
//! - [`jobs`] — the durable background work queue
|
||||
//! - [`runner`] — the thing that drains it, driven by whoever owns the thread
|
||||
//! - [`trash`] — soft delete to a folder, then permanent delete
|
||||
//! - [`merge`] / [`sync`] — cross-device merging of collections and keywords
|
||||
//! - [`recovery`] — backups, and the two offers made when this file is damaged
|
||||
//!
|
||||
//! # The one thing everything is designed around
|
||||
//!
|
||||
//! **Work is proportional to what changed, or to what the user is looking at —
|
||||
//! never to library size.** A 50k-image library that has not changed costs one
|
||||
//! metadata probe per folder to verify (§scan), no thumbnails to regenerate
|
||||
//! (§jobs coalescing), and no rule evaluation per grid cell (materialised
|
||||
//! `tier_desired`).
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use dr_types::{Availability, ImageId};
|
||||
use rusqlite::Connection;
|
||||
|
||||
pub mod bursts;
|
||||
pub mod cache;
|
||||
pub mod collections;
|
||||
pub mod dedup;
|
||||
pub mod error;
|
||||
pub mod face_shard;
|
||||
pub mod faces;
|
||||
pub mod jobs;
|
||||
pub mod keywords;
|
||||
pub mod merge;
|
||||
pub mod query;
|
||||
pub mod rating;
|
||||
pub mod recovery;
|
||||
pub mod runner;
|
||||
pub mod scan;
|
||||
pub mod schema;
|
||||
pub mod sync;
|
||||
pub mod trash;
|
||||
pub mod walk;
|
||||
|
||||
pub use cache::{Budget, Cache, DEFAULT_BUDGET_BYTES};
|
||||
pub use collections::{Collection, CollectionKind, TreeRow};
|
||||
pub use dedup::{seen_by_content, seen_by_metadata, set_content_hash};
|
||||
pub use error::CatalogError;
|
||||
pub use face_shard::{FaceShardStore, SharedFace};
|
||||
pub use faces::{Calibration, DetectedFace, Face, FaceId, Measurement, Person, PersonId};
|
||||
pub use jobs::{Job, JobKind, Priority};
|
||||
pub use keywords::{Coverage, Keyword, KeywordId, SelectionKeyword};
|
||||
pub use merge::MergeReport;
|
||||
pub use query::{Query, Sort};
|
||||
pub use rating::{Judgement, MAX_RATING};
|
||||
pub use recovery::Backup;
|
||||
// Not `runner::Budget`: `cache::Budget` already owns that name here and
|
||||
// means something else entirely (bytes on disk, not jobs in a slot).
|
||||
// Callers spell the work budget `runner::Budget`, where it is unambiguous.
|
||||
pub use runner::{DrainReport, JobHandler, Outcome, Runner};
|
||||
pub use scan::{DirAction, DirState, EntryAction, ScanOutcome};
|
||||
pub use trash::{TrashedImage, TRASH_DIR};
|
||||
pub use walk::{ensure_root, mark_root_offline, scan_root, RootKind, ScanProgress, ScanReport};
|
||||
|
||||
/// One row of the library grid.
|
||||
///
|
||||
/// Exactly what a cell draws and nothing more — no join per cell, and
|
||||
/// availability reads a materialised column rather than evaluating cache rules
|
||||
/// (ARCH §9.5).
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct GridRow {
|
||||
pub id: ImageId,
|
||||
pub name: String,
|
||||
pub availability: Availability,
|
||||
/// UTC seconds. `None` until EXIF has been read.
|
||||
pub captured_at: Option<i64>,
|
||||
/// Minutes east of UTC, for rendering the photographer's local time.
|
||||
pub captured_offset: Option<i32>,
|
||||
/// 0 = nothing, 1 = stat-only, 2 = full EXIF.
|
||||
pub metadata_state: u8,
|
||||
}
|
||||
|
||||
/// A count of images in one time bucket, for the timeline scrubber.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct TimeBucket {
|
||||
/// UTC seconds at the bucket's start.
|
||||
pub start: i64,
|
||||
pub count: u32,
|
||||
}
|
||||
|
||||
/// Time bucket size.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Granularity {
|
||||
Year,
|
||||
Month,
|
||||
Day,
|
||||
Hour,
|
||||
}
|
||||
|
||||
impl Granularity {
|
||||
/// SQLite `strftime` format that collapses a timestamp to this bucket.
|
||||
///
|
||||
/// Applied to **local** time, not UTC: "everything from 3 August" means
|
||||
/// the photographer's 3 August, which is why `captured_offset` is stored
|
||||
/// alongside the UTC timestamp.
|
||||
/// Public so a caller that must build its own bucketing query — one
|
||||
/// joining collection membership, say — buckets identically to
|
||||
/// [`Catalog::timeline_range`] rather than reimplementing the format.
|
||||
pub fn strftime(self) -> &'static str {
|
||||
match self {
|
||||
Granularity::Year => "%Y",
|
||||
Granularity::Month => "%Y-%m",
|
||||
Granularity::Day => "%Y-%m-%d",
|
||||
Granularity::Hour => "%Y-%m-%dT%H",
|
||||
}
|
||||
}
|
||||
|
||||
/// A sensible bucket size for a span of seconds, so the UI need not guess.
|
||||
///
|
||||
/// # Chosen by how many bars it produces, not by fixed cut-offs
|
||||
///
|
||||
/// This used to be four thresholds on the span, which reads sensibly and
|
||||
/// behaves badly under zoom. Each zoom step halves the span, so the bar
|
||||
/// count halves with it until a threshold is crossed — a fifteen-year
|
||||
/// library went 15 bars, 8, then 46, 23, 11, and finally *6*. Zooming in
|
||||
/// made the picture coarser, which is the opposite of what zooming is for.
|
||||
///
|
||||
/// So the choice is made on the axis's terms: of the four bucket sizes,
|
||||
/// take the one whose bar count comes nearest [`Self::TARGET_BARS`]. The
|
||||
/// count then stays in the same neighbourhood at every zoom level, and
|
||||
/// each step in genuinely shows finer structure rather than the same
|
||||
/// structure drawn wider.
|
||||
///
|
||||
/// Nearest in *ratio*, not in difference: the counts available for a given
|
||||
/// span are orders of magnitude apart — a span is either about 4 years or
|
||||
/// about 48 months — and on a linear measure the larger count always looks
|
||||
/// further away, which would bias every choice towards too few bars.
|
||||
pub fn for_span(seconds: i64) -> Self {
|
||||
Self::for_bucket(seconds.max(1) / Self::TARGET_BARS)
|
||||
}
|
||||
|
||||
/// The calendar unit nearest a bucket of `seconds`, for *labelling* one.
|
||||
///
|
||||
/// Split out from [`Self::for_span`] because the axis no longer buckets by
|
||||
/// calendar unit at all — it divides the visible span into a fixed number
|
||||
/// of equal bins (see `LibrarySettings::timeline_bars`). What is still
|
||||
/// wanted is the unit a bin is closest to, so a bin of about a day is
|
||||
/// labelled as a date and one of about a year as a year. Asked directly
|
||||
/// rather than derived from the span, because the bin count is now the
|
||||
/// user's rather than this module's target.
|
||||
pub fn for_bucket(seconds: i64) -> Self {
|
||||
let seconds = seconds.max(1) as f64;
|
||||
// Finest first, so that when two options are equally far from the
|
||||
// target the finer one wins: `min_by` keeps the first minimum it saw,
|
||||
// and more detail is the better failure.
|
||||
[
|
||||
Granularity::Hour,
|
||||
Granularity::Day,
|
||||
Granularity::Month,
|
||||
Granularity::Year,
|
||||
]
|
||||
.into_iter()
|
||||
.min_by(|a, b| {
|
||||
let cost = |g: Granularity| {
|
||||
// How far off, measured multiplicatively: twice as long and
|
||||
// half as long are equally wrong.
|
||||
//
|
||||
// Deliberately not clamped. A bucket shorter than the unit
|
||||
// scores *worse* the coarser the unit, which is what makes an
|
||||
// hour of photographs pick hourly bars instead of every option
|
||||
// tying at "one bucket" and the coarsest winning.
|
||||
(seconds / g.approx_seconds() as f64).ln().abs()
|
||||
};
|
||||
cost(*a)
|
||||
.partial_cmp(&cost(*b))
|
||||
// Ties cannot arise from real spans, but a NaN would; falling
|
||||
// back to the coarser option keeps the axis drawable.
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
})
|
||||
.unwrap_or(Granularity::Day)
|
||||
}
|
||||
|
||||
/// How many bars the timeline wants across its axis.
|
||||
///
|
||||
/// Not a hard count — the bucket sizes are calendar units, so the actual
|
||||
/// number lands where the calendar puts it. It is the figure the choice
|
||||
/// aims at: enough bars that a busy fortnight is visibly busier than a
|
||||
/// quiet one, few enough that each is wide enough to hit with a finger.
|
||||
const TARGET_BARS: i64 = 40;
|
||||
|
||||
/// Nominal length of one bucket, for choosing between them.
|
||||
///
|
||||
/// Approximate on purpose: months and years vary and it does not matter
|
||||
/// here, because this only ranks four options that are a factor of ~12 or
|
||||
/// ~30 apart. The exact boundaries come from `strftime` on the real dates.
|
||||
fn approx_seconds(self) -> i64 {
|
||||
const DAY: i64 = 86_400;
|
||||
match self {
|
||||
Granularity::Year => 365 * DAY,
|
||||
Granularity::Month => 30 * DAY,
|
||||
Granularity::Day => DAY,
|
||||
Granularity::Hour => 3600,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A connection to the catalog.
|
||||
pub struct Catalog {
|
||||
conn: Connection,
|
||||
}
|
||||
|
||||
impl Catalog {
|
||||
/// Open or create a catalog, migrating it forward if needed.
|
||||
///
|
||||
/// Does **not** verify the file — see [`Self::open_verified`], and
|
||||
/// [`recovery`] for why the check is bound to startup rather than to every
|
||||
/// open. Damage this trips over on the way past is still reported as
|
||||
/// [`CatalogError::Corrupt`] rather than as a stray SQLite error.
|
||||
pub fn open(path: &Path) -> Result<Self, CatalogError> {
|
||||
let conn = Connection::open(path)?;
|
||||
schema::configure(&conn)?;
|
||||
// NFR-R2, and the reason it is *here*: a migration is the one routine
|
||||
// operation that rewrites table structure, so it is the likeliest way
|
||||
// this file becomes unreadable — and afterwards there is no
|
||||
// pre-migration state left to copy. A failure to take the copy is
|
||||
// logged rather than raised: a full disk must not be the thing that
|
||||
// makes a library unopenable.
|
||||
if let Err(e) = recovery::backup_before_migration(&conn, path) {
|
||||
log::warn!("could not back up before migrating: {e}");
|
||||
}
|
||||
let from = schema::migrate(&conn)?;
|
||||
// A migration adds a column; it cannot know what the value should be
|
||||
// for rows that already existed. Backfilling on open is what stops
|
||||
// those rows being silently partial.
|
||||
for (what, n) in schema::backfill(&conn)? {
|
||||
log::info!("backfilled {what} for {n} row(s) (schema was v{from})");
|
||||
}
|
||||
Ok(Catalog { conn })
|
||||
}
|
||||
|
||||
/// TRACES: NFR-R6
|
||||
/// Open a catalog, checking the file first.
|
||||
///
|
||||
/// What startup calls. On [`CatalogError::Corrupt`] the caller has a user
|
||||
/// in front of it and must make the two offers [`recovery`] describes,
|
||||
/// rather than reporting a SQLite message on a banner and carrying on into
|
||||
/// a scan that would write into the damage.
|
||||
///
|
||||
/// Checked *before* opening rather than after, because opening runs
|
||||
/// migrations: a damaged catalog that happens to have an intact header
|
||||
/// would otherwise be migrated — rewriting structure on top of structure
|
||||
/// that is already wrong — before anybody asked whether it was sound.
|
||||
pub fn open_verified(path: &Path) -> Result<Self, CatalogError> {
|
||||
// A catalog that is not there yet is not damaged; `open` creates it.
|
||||
if path.is_file() {
|
||||
recovery::check_file(path)?;
|
||||
}
|
||||
Self::open(path)
|
||||
}
|
||||
|
||||
/// An in-memory catalog, for tests and for a throwaway import preview.
|
||||
pub fn in_memory() -> Result<Self, CatalogError> {
|
||||
let conn = Connection::open_in_memory()?;
|
||||
schema::configure(&conn)?;
|
||||
schema::migrate(&conn)?;
|
||||
schema::backfill(&conn)?;
|
||||
Ok(Catalog { conn })
|
||||
}
|
||||
|
||||
/// Escape hatch for modules that need raw access. Not part of the UI-facing
|
||||
/// surface.
|
||||
pub fn connection(&self) -> &Connection {
|
||||
&self.conn
|
||||
}
|
||||
|
||||
/// How many images match.
|
||||
///
|
||||
/// Returned alongside the first window so the grid can size its scrollbar
|
||||
/// and paint in one round trip.
|
||||
pub fn count(&self, q: &Query, now: i64) -> Result<usize, CatalogError> {
|
||||
let c = query::compile(&q.filter, now);
|
||||
let sql = query::count_sql(&c);
|
||||
let n: i64 =
|
||||
self.conn
|
||||
.query_row(&sql, rusqlite::params_from_iter(c.params.iter()), |r| {
|
||||
r.get(0)
|
||||
})?;
|
||||
Ok(n as usize)
|
||||
}
|
||||
|
||||
/// Fetch one window of results.
|
||||
///
|
||||
/// Never returns the whole catalog: FR-CAT-4 requires memory bounded
|
||||
/// independently of library size.
|
||||
pub fn window(
|
||||
&self,
|
||||
q: &Query,
|
||||
range: std::ops::Range<usize>,
|
||||
now: i64,
|
||||
) -> Result<Vec<GridRow>, CatalogError> {
|
||||
let c = query::compile(&q.filter, now);
|
||||
let sql = query::window_sql(q, &c);
|
||||
|
||||
let mut params = c.params.clone();
|
||||
params.push(rusqlite::types::Value::Integer(range.len() as i64));
|
||||
params.push(rusqlite::types::Value::Integer(range.start as i64));
|
||||
|
||||
let mut stmt = self.conn.prepare(&sql)?;
|
||||
let rows = stmt
|
||||
.query_map(rusqlite::params_from_iter(params.iter()), |r| {
|
||||
let source_ref: String = r.get(1)?;
|
||||
let avail: i64 = r.get(2)?;
|
||||
Ok(GridRow {
|
||||
id: ImageId(r.get::<_, i64>(0)? as u64),
|
||||
name: source_ref
|
||||
.rsplit(['/', ':'])
|
||||
.next()
|
||||
.unwrap_or(&source_ref)
|
||||
.to_string(),
|
||||
availability: decode_availability(avail),
|
||||
captured_at: r.get(3)?,
|
||||
captured_offset: r.get::<_, Option<i64>>(4)?.map(|v| v as i32),
|
||||
metadata_state: r.get::<_, i64>(5)? as u8,
|
||||
})
|
||||
})?
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
/// Counts per time bucket, for the timeline scrubber.
|
||||
///
|
||||
/// One grouped aggregate over the `images_captured` index — not 50k rows
|
||||
/// handed to the UI to bucket itself.
|
||||
pub fn timeline(
|
||||
&self,
|
||||
q: &Query,
|
||||
g: Granularity,
|
||||
now: i64,
|
||||
) -> Result<Vec<TimeBucket>, CatalogError> {
|
||||
let c = query::compile(&q.filter, now);
|
||||
// Bucketed in local time: captured_offset is minutes east of UTC, and
|
||||
// NULL falls back to UTC rather than dropping the row.
|
||||
let sql = format!(
|
||||
"SELECT min(captured_at) AS start,
|
||||
count(*) AS n
|
||||
FROM images
|
||||
-- A shadowed JPEG is the same frame as its RAW; counting both
|
||||
-- would double every paired shot in the histogram.
|
||||
WHERE {} AND captured_at IS NOT NULL AND shadowed_by IS NULL
|
||||
GROUP BY strftime('{}', captured_at + coalesce(captured_offset, 0) * 60,
|
||||
'unixepoch')
|
||||
ORDER BY start ASC",
|
||||
c.where_sql,
|
||||
g.strftime()
|
||||
);
|
||||
|
||||
let mut stmt = self.conn.prepare(&sql)?;
|
||||
let rows = stmt
|
||||
.query_map(rusqlite::params_from_iter(c.params.iter()), |r| {
|
||||
Ok(TimeBucket {
|
||||
start: r.get(0)?,
|
||||
count: r.get::<_, i64>(1)? as u32,
|
||||
})
|
||||
})?
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
/// Counts per time bucket, bounded to a date range.
|
||||
///
|
||||
/// What a zoomed timeline needs: [`timeline`](Self::timeline) always spans
|
||||
/// the whole library, so zooming in would return the same coarse buckets
|
||||
/// with the ends cropped rather than finer detail over a narrower span.
|
||||
pub fn timeline_range(
|
||||
&self,
|
||||
q: &Query,
|
||||
g: Granularity,
|
||||
from: i64,
|
||||
to: i64,
|
||||
now: i64,
|
||||
) -> Result<Vec<TimeBucket>, CatalogError> {
|
||||
let c = query::compile(&q.filter, now);
|
||||
let sql = format!(
|
||||
"SELECT min(captured_at) AS start,
|
||||
count(*) AS n
|
||||
FROM images
|
||||
WHERE {} AND captured_at IS NOT NULL AND shadowed_by IS NULL
|
||||
AND captured_at >= ?{} AND captured_at <= ?{}
|
||||
GROUP BY strftime('{}', captured_at + coalesce(captured_offset, 0) * 60,
|
||||
'unixepoch')
|
||||
ORDER BY start ASC",
|
||||
c.where_sql,
|
||||
c.params.len() + 1,
|
||||
c.params.len() + 2,
|
||||
g.strftime()
|
||||
);
|
||||
|
||||
let mut params = c.params.clone();
|
||||
params.push(rusqlite::types::Value::Integer(from));
|
||||
params.push(rusqlite::types::Value::Integer(to));
|
||||
|
||||
let mut stmt = self.conn.prepare(&sql)?;
|
||||
let rows = stmt
|
||||
.query_map(rusqlite::params_from_iter(params.iter()), |r| {
|
||||
Ok(TimeBucket {
|
||||
start: r.get(0)?,
|
||||
count: r.get::<_, i64>(1)? as u32,
|
||||
})
|
||||
})?
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
/// Merge a downloaded remote catalog's collections into this one.
|
||||
///
|
||||
/// See [`sync`] for why only collections cross over.
|
||||
pub fn merge_remote_catalog(&self, remote: &Path) -> Result<MergeReport, CatalogError> {
|
||||
sync::merge_remote(&self.conn, remote)
|
||||
}
|
||||
|
||||
/// Write a consistent snapshot ready to upload.
|
||||
pub fn snapshot_for_upload(&self, dest: &Path) -> Result<(), CatalogError> {
|
||||
sync::snapshot_for_upload(&self.conn, dest)
|
||||
}
|
||||
}
|
||||
|
||||
fn decode_availability(v: i64) -> Availability {
|
||||
match v {
|
||||
1 => Availability::Preview,
|
||||
2 => Availability::Original,
|
||||
3 => Availability::Offline,
|
||||
_ => Availability::MetadataOnly,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use dr_types::Selector;
|
||||
|
||||
fn seeded() -> Catalog {
|
||||
let cat = Catalog::in_memory().unwrap();
|
||||
let c = cat.connection();
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
// Three images across two days, one with no EXIF read yet.
|
||||
for (id, name, captured, state) in [
|
||||
(1i64, "a.CR3", Some(1_000_000i64), 2i64),
|
||||
(2, "b.CR3", Some(1_100_000), 2),
|
||||
(3, "c.CR3", None, 1),
|
||||
] {
|
||||
c.execute(
|
||||
"INSERT INTO images(id, root_id, source_ref, captured_at, metadata_state, added_at)
|
||||
VALUES (?1, 1, ?2, ?3, ?4, 0)",
|
||||
rusqlite::params![id, name, captured, state],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
cat
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn count_and_window_agree() {
|
||||
let cat = seeded();
|
||||
let q = Query::default();
|
||||
assert_eq!(cat.count(&q, 0).unwrap(), 3);
|
||||
assert_eq!(cat.window(&q, 0..10, 0).unwrap().len(), 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn window_is_bounded_by_the_requested_range() {
|
||||
// FR-CAT-4: memory independent of catalog size.
|
||||
let cat = seeded();
|
||||
let rows = cat.window(&Query::default(), 0..2, 0).unwrap();
|
||||
assert_eq!(rows.len(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paging_covers_every_row_exactly_once() {
|
||||
let cat = seeded();
|
||||
let q = Query::default();
|
||||
let mut seen = Vec::new();
|
||||
for start in (0..3).step_by(2) {
|
||||
seen.extend(cat.window(&q, start..start + 2, 0).unwrap());
|
||||
}
|
||||
let mut ids: Vec<u64> = seen.iter().map(|r| r.id.0).collect();
|
||||
ids.sort_unstable();
|
||||
assert_eq!(ids, vec![1, 2, 3]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_image_without_capture_time_sorts_last_not_first() {
|
||||
// Otherwise a freshly scanned library leads with whatever has not been
|
||||
// read yet, which looks like corruption to the user.
|
||||
let cat = seeded();
|
||||
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
|
||||
assert_eq!(rows.last().unwrap().id, ImageId(3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn metadata_state_reaches_the_grid() {
|
||||
// The grid needs it to distinguish "no photos on this date" from
|
||||
// "EXIF not read yet" (FR-NC-6c's honesty principle).
|
||||
let cat = seeded();
|
||||
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
|
||||
let pending = rows.iter().find(|r| r.id == ImageId(3)).unwrap();
|
||||
assert_eq!(pending.metadata_state, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_filter_narrows_the_count() {
|
||||
let cat = seeded();
|
||||
let q = Query {
|
||||
filter: Selector::Text("a.CR3".into()),
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(cat.count(&q, 0).unwrap(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn timeline_buckets_and_skips_unread_images() {
|
||||
let cat = seeded();
|
||||
let buckets = cat
|
||||
.timeline(&Query::default(), Granularity::Day, 0)
|
||||
.unwrap();
|
||||
// Two images with timestamps, one day apart in UTC; the third has no
|
||||
// capture time and cannot be placed on a timeline at all.
|
||||
let total: u32 = buckets.iter().map(|b| b.count).sum();
|
||||
assert_eq!(total, 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn timeline_granularity_follows_the_span() {
|
||||
const DAY: i64 = 86_400;
|
||||
// Chosen by how many bars it makes, not by fixed cut-offs — see
|
||||
// `for_span`. Ten years of yearly bars is ten bars, which says almost
|
||||
// nothing about a library; monthly is 122, which is a shape.
|
||||
assert_eq!(Granularity::for_span(10 * 365 * DAY), Granularity::Month);
|
||||
assert_eq!(Granularity::for_span(120 * DAY), Granularity::Day);
|
||||
assert_eq!(Granularity::for_span(10 * DAY), Granularity::Day);
|
||||
assert_eq!(Granularity::for_span(3600), Granularity::Hour);
|
||||
|
||||
// The property the target exists for: zooming in never coarsens the
|
||||
// axis. Under the old thresholds a fifteen-year library went 15 bars,
|
||||
// then 8, then 46, 23, 11 — finer spans drawn with wider bars.
|
||||
let mut span = 15 * 365 * DAY;
|
||||
let mut previous = Granularity::for_span(span).approx_seconds();
|
||||
for _ in 0..10 {
|
||||
span /= 2;
|
||||
let bucket = Granularity::for_span(span).approx_seconds();
|
||||
assert!(
|
||||
bucket <= previous,
|
||||
"halving the span to {span}s coarsened the bucket \
|
||||
from {previous}s to {bucket}s"
|
||||
);
|
||||
previous = bucket;
|
||||
}
|
||||
|
||||
// And a span shorter than any bucket still picks the finest, rather
|
||||
// than every option tying at one bar and the coarsest winning.
|
||||
assert_eq!(Granularity::for_span(60), Granularity::Hour);
|
||||
assert_eq!(Granularity::for_span(1), Granularity::Hour);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn names_are_derived_for_both_paths_and_saf_ids() {
|
||||
let cat = Catalog::in_memory().unwrap();
|
||||
let c = cat.connection();
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'saf', 'tree')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO images(id, root_id, source_ref, added_at)
|
||||
VALUES (1, 1, 'primary:DCIM/Camera/IMG_1.CR3', 0)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
let rows = cat.window(&Query::default(), 0..10, 0).unwrap();
|
||||
assert_eq!(rows[0].name, "IMG_1.CR3");
|
||||
}
|
||||
}
|
||||
@@ -1,514 +0,0 @@
|
||||
//! TRACES: FR-CAT-4 | FR-CAT-6
|
||||
//! Compiling a [`Selector`] into indexed SQL, and windowing the result.
|
||||
//!
|
||||
//! The UI never assembles SQL — it hands over a [`Query`] and receives a
|
||||
//! window. Two properties matter:
|
||||
//!
|
||||
//! 1. **Nothing user-supplied is interpolated into SQL text.** Every value
|
||||
//! binds as a parameter; `LIKE` patterns have their wildcards escaped.
|
||||
//! 2. **Predicates hit indices.** Filtering 50k images must stay interactive
|
||||
//! (FR-CAT-6), which means no expression over a column that would defeat
|
||||
//! its index.
|
||||
|
||||
use dr_types::{Availability, ColourLabel, DateSelector, FlagState, Selector};
|
||||
use rusqlite::types::Value;
|
||||
|
||||
/// What to show, and in what order.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Query {
|
||||
pub filter: Selector,
|
||||
pub sort: Sort,
|
||||
pub descending: bool,
|
||||
}
|
||||
|
||||
impl Default for Query {
|
||||
fn default() -> Self {
|
||||
Query {
|
||||
filter: Selector::All,
|
||||
sort: Sort::CapturedAt,
|
||||
descending: true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Sort {
|
||||
CapturedAt,
|
||||
Added,
|
||||
FileName,
|
||||
Rating,
|
||||
/// Manual order within a collection. Falls back to capture time where the
|
||||
/// query is not scoped to one collection, since position is meaningless
|
||||
/// outside it.
|
||||
CollectionPosition,
|
||||
}
|
||||
|
||||
impl Sort {
|
||||
/// The ORDER BY fragment. Fixed strings — never user input.
|
||||
///
|
||||
/// Capture time sorts NULLs last regardless of direction: an image whose
|
||||
/// EXIF has not been read yet (metadata_state 1) should not lead the grid
|
||||
/// simply because its timestamp is unknown.
|
||||
fn sql(self, descending: bool) -> &'static str {
|
||||
match (self, descending) {
|
||||
(Sort::CapturedAt, false) => {
|
||||
"ORDER BY images.captured_at IS NULL, images.captured_at ASC, images.id ASC"
|
||||
}
|
||||
(Sort::CapturedAt, true) => {
|
||||
"ORDER BY images.captured_at IS NULL, images.captured_at DESC, images.id DESC"
|
||||
}
|
||||
(Sort::Added, false) => "ORDER BY images.added_at ASC, images.id ASC",
|
||||
(Sort::Added, true) => "ORDER BY images.added_at DESC, images.id DESC",
|
||||
(Sort::FileName, false) => "ORDER BY images.source_ref ASC, images.id ASC",
|
||||
(Sort::FileName, true) => "ORDER BY images.source_ref DESC, images.id DESC",
|
||||
(Sort::Rating, false) => "ORDER BY v.rating ASC, images.id ASC",
|
||||
(Sort::Rating, true) => "ORDER BY v.rating DESC, images.id DESC",
|
||||
(Sort::CollectionPosition, false) => {
|
||||
"ORDER BY cm.position IS NULL, cm.position ASC, images.captured_at ASC"
|
||||
}
|
||||
(Sort::CollectionPosition, true) => {
|
||||
"ORDER BY cm.position IS NULL, cm.position DESC, images.captured_at DESC"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether this sort needs the default-version join.
|
||||
fn needs_version(self) -> bool {
|
||||
matches!(self, Sort::Rating)
|
||||
}
|
||||
|
||||
/// Whether this sort needs a collection-membership join.
|
||||
fn needs_membership(self) -> bool {
|
||||
matches!(self, Sort::CollectionPosition)
|
||||
}
|
||||
}
|
||||
|
||||
/// A compiled WHERE clause plus its bound parameters.
|
||||
///
|
||||
/// Kept separate from the statement so `count` and `window` can share one
|
||||
/// compilation.
|
||||
#[derive(Debug, Default)]
|
||||
pub struct Compiled {
|
||||
pub where_sql: String,
|
||||
pub params: Vec<Value>,
|
||||
/// True if the filter depends on capture time, and therefore on EXIF that
|
||||
/// a freshly scanned library may not have read yet. The UI surfaces this
|
||||
/// rather than silently under-reporting.
|
||||
pub needs_capture_time: bool,
|
||||
}
|
||||
|
||||
/// Compile a selector to SQL against the `images` table.
|
||||
///
|
||||
/// `now` is passed rather than read from the clock so a rolling window is
|
||||
/// reproducible in tests and consistent across one query.
|
||||
pub fn compile(filter: &Selector, now: i64) -> Compiled {
|
||||
let mut params = Vec::new();
|
||||
let sql = if filter.is_unfiltered() {
|
||||
"1".to_string()
|
||||
} else {
|
||||
emit(filter, now, &mut params)
|
||||
};
|
||||
Compiled {
|
||||
where_sql: sql,
|
||||
params,
|
||||
needs_capture_time: filter.needs_capture_time(),
|
||||
}
|
||||
}
|
||||
|
||||
fn emit(s: &Selector, now: i64, p: &mut Vec<Value>) -> String {
|
||||
match s {
|
||||
Selector::All => "1".into(),
|
||||
|
||||
Selector::Collection(id) => {
|
||||
p.push(Value::Integer(id.0 as i64));
|
||||
format!(
|
||||
"EXISTS (SELECT 1 FROM collection_members m
|
||||
WHERE m.image_id = images.id AND m.collection_id = ?{})",
|
||||
p.len()
|
||||
)
|
||||
}
|
||||
|
||||
Selector::Folder {
|
||||
root,
|
||||
path,
|
||||
recursive,
|
||||
} => {
|
||||
p.push(Value::Integer(root.0 as i64));
|
||||
let root_ix = p.len();
|
||||
if *recursive {
|
||||
// Prefix match on the folder path. `like_prefix` escapes the
|
||||
// pattern metacharacters, so a folder literally named "50%"
|
||||
// matches itself and not everything.
|
||||
p.push(Value::Text(like_prefix(path)));
|
||||
format!(
|
||||
"images.folder_id IN (
|
||||
SELECT id FROM folders
|
||||
WHERE root_id = ?{root_ix}
|
||||
AND (path = ?{p} OR path LIKE ?{p} || '/%' ESCAPE '\\'))",
|
||||
p = p.len()
|
||||
)
|
||||
} else {
|
||||
p.push(Value::Text(path.clone()));
|
||||
format!(
|
||||
"images.folder_id IN (
|
||||
SELECT id FROM folders WHERE root_id = ?{root_ix} AND path = ?{})",
|
||||
p.len()
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
Selector::DateRange(d) => emit_date(d, now, p),
|
||||
|
||||
Selector::Rating { min } => {
|
||||
p.push(Value::Integer(*min as i64));
|
||||
format!("{} >= ?{}", default_version_scalar("rating"), p.len())
|
||||
}
|
||||
|
||||
Selector::Label(l) => {
|
||||
p.push(Value::Integer(label_code(*l)));
|
||||
format!("{} = ?{}", default_version_scalar("label"), p.len())
|
||||
}
|
||||
|
||||
Selector::Flag(f) => {
|
||||
p.push(Value::Integer(flag_code(*f)));
|
||||
format!("{} = ?{}", default_version_scalar("flag"), p.len())
|
||||
}
|
||||
|
||||
Selector::Keyword(k) => {
|
||||
p.push(Value::Text(k.clone()));
|
||||
format!(
|
||||
"EXISTS (SELECT 1 FROM keywords kw
|
||||
JOIN versions kv ON kv.id = kw.version_id
|
||||
WHERE kv.image_id = images.id AND kw.keyword = ?{})",
|
||||
p.len()
|
||||
)
|
||||
}
|
||||
|
||||
Selector::Camera(c) => {
|
||||
p.push(Value::Text(c.clone()));
|
||||
format!("images.camera = ?{}", p.len())
|
||||
}
|
||||
|
||||
Selector::Lens(l) => {
|
||||
p.push(Value::Text(l.clone()));
|
||||
format!("images.lens = ?{}", p.len())
|
||||
}
|
||||
|
||||
Selector::IsoRange { min, max } => {
|
||||
p.push(Value::Integer(*min as i64));
|
||||
let lo = p.len();
|
||||
p.push(Value::Integer(*max as i64));
|
||||
format!("images.iso BETWEEN ?{lo} AND ?{}", p.len())
|
||||
}
|
||||
|
||||
Selector::Availability(a) => {
|
||||
p.push(Value::Integer(availability_code(*a)));
|
||||
format!("images.availability = ?{}", p.len())
|
||||
}
|
||||
|
||||
Selector::Text(t) => {
|
||||
// Substring over filename and keywords. A LIKE scan is adequate at
|
||||
// 50k; if free text over title and description becomes a real
|
||||
// workflow, FTS5 is the answer and it is additive.
|
||||
p.push(Value::Text(format!("%{}%", escape_like(t))));
|
||||
let ix = p.len();
|
||||
format!(
|
||||
"(images.source_ref LIKE ?{ix} ESCAPE '\\'
|
||||
OR EXISTS (SELECT 1 FROM keywords kw
|
||||
JOIN versions kv ON kv.id = kw.version_id
|
||||
WHERE kv.image_id = images.id
|
||||
AND kw.keyword LIKE ?{ix} ESCAPE '\\'))"
|
||||
)
|
||||
}
|
||||
|
||||
// An empty conjunction is vacuously true; an empty disjunction matches
|
||||
// nothing. Both arise from a UI that lets every term be cleared, and
|
||||
// conflating them would show the whole library when the user meant the
|
||||
// opposite.
|
||||
Selector::All_(v) if v.is_empty() => "1".into(),
|
||||
Selector::Any(v) if v.is_empty() => "0".into(),
|
||||
|
||||
Selector::All_(v) => join(v, " AND ", now, p),
|
||||
Selector::Any(v) => join(v, " OR ", now, p),
|
||||
Selector::Not(inner) => format!("NOT ({})", emit(inner, now, p)),
|
||||
}
|
||||
}
|
||||
|
||||
fn join(items: &[Selector], op: &str, now: i64, p: &mut Vec<Value>) -> String {
|
||||
let parts: Vec<String> = items.iter().map(|s| emit(s, now, p)).collect();
|
||||
format!("({})", parts.join(op))
|
||||
}
|
||||
|
||||
fn emit_date(d: &DateSelector, now: i64, p: &mut Vec<Value>) -> String {
|
||||
match d {
|
||||
DateSelector::Between { from, to } => {
|
||||
p.push(Value::Integer(*from));
|
||||
let lo = p.len();
|
||||
p.push(Value::Integer(*to));
|
||||
// Half-open, so adjacent ranges neither overlap nor gap.
|
||||
format!(
|
||||
"(images.captured_at >= ?{lo} AND images.captured_at < ?{})",
|
||||
p.len()
|
||||
)
|
||||
}
|
||||
DateSelector::Rolling { days } => {
|
||||
let from = now - (*days as i64) * 86_400;
|
||||
p.push(Value::Integer(from));
|
||||
format!("images.captured_at >= ?{}", p.len())
|
||||
}
|
||||
DateSelector::CollectionSpan(id) => {
|
||||
p.push(Value::Integer(id.0 as i64));
|
||||
let ix = p.len();
|
||||
format!(
|
||||
"images.captured_at BETWEEN
|
||||
(SELECT min(i2.captured_at) FROM images i2
|
||||
JOIN collection_members m2 ON m2.image_id = i2.id
|
||||
WHERE m2.collection_id = ?{ix})
|
||||
AND (SELECT max(i2.captured_at) FROM images i2
|
||||
JOIN collection_members m2 ON m2.image_id = i2.id
|
||||
WHERE m2.collection_id = ?{ix})"
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Rating, label, and flag live on the *default* version, not the image.
|
||||
///
|
||||
/// A correlated subquery rather than a join, so these compose inside `OR` and
|
||||
/// `NOT` without the join multiplying rows.
|
||||
fn default_version_scalar(col: &str) -> String {
|
||||
format!(
|
||||
"(SELECT dv.{col} FROM versions dv
|
||||
WHERE dv.image_id = images.id AND dv.is_default = 1 LIMIT 1)"
|
||||
)
|
||||
}
|
||||
|
||||
/// Escape LIKE metacharacters so a literal `%` or `_` in user text matches
|
||||
/// itself. Paired with `ESCAPE '\'` in every LIKE that uses it.
|
||||
fn escape_like(s: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len());
|
||||
for c in s.chars() {
|
||||
if matches!(c, '%' | '_' | '\\') {
|
||||
out.push('\\');
|
||||
}
|
||||
out.push(c);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
fn like_prefix(path: &str) -> String {
|
||||
escape_like(path.trim_end_matches('/'))
|
||||
}
|
||||
|
||||
fn label_code(l: ColourLabel) -> i64 {
|
||||
match l {
|
||||
ColourLabel::Red => 1,
|
||||
ColourLabel::Yellow => 2,
|
||||
ColourLabel::Green => 3,
|
||||
ColourLabel::Blue => 4,
|
||||
ColourLabel::Purple => 5,
|
||||
}
|
||||
}
|
||||
|
||||
fn flag_code(f: FlagState) -> i64 {
|
||||
match f {
|
||||
FlagState::Unflagged => 0,
|
||||
FlagState::Pick => 1,
|
||||
FlagState::Reject => 2,
|
||||
}
|
||||
}
|
||||
|
||||
/// The stored form of an availability. Shared with [`crate::walk`], which
|
||||
/// writes the column this reads — two spellings of the same mapping would
|
||||
/// filter for a state nothing ever writes.
|
||||
pub(crate) fn availability_code(a: Availability) -> i64 {
|
||||
match a {
|
||||
Availability::MetadataOnly => 0,
|
||||
Availability::Preview => 1,
|
||||
Availability::Original => 2,
|
||||
Availability::Offline => 3,
|
||||
}
|
||||
}
|
||||
|
||||
/// Build the full SELECT for a window of results.
|
||||
///
|
||||
/// Joins are added only where the sort needs them, so an unsorted-by-rating
|
||||
/// grid query touches one table.
|
||||
pub fn window_sql(q: &Query, compiled: &Compiled) -> String {
|
||||
let mut joins = String::new();
|
||||
if q.sort.needs_version() {
|
||||
joins.push_str(" LEFT JOIN versions v ON v.image_id = images.id AND v.is_default = 1");
|
||||
}
|
||||
if q.sort.needs_membership() {
|
||||
// Only meaningful when the filter scopes to one collection; elsewhere
|
||||
// position is NULL and the sort falls through to capture time.
|
||||
joins.push_str(" LEFT JOIN collection_members cm ON cm.image_id = images.id");
|
||||
}
|
||||
format!(
|
||||
"SELECT images.id, images.source_ref, images.availability, images.captured_at, \
|
||||
images.captured_offset, images.metadata_state \
|
||||
FROM images{joins} WHERE {} {} LIMIT ? OFFSET ?",
|
||||
compiled.where_sql,
|
||||
q.sort.sql(q.descending)
|
||||
)
|
||||
}
|
||||
|
||||
/// Build the COUNT for the same filter.
|
||||
pub fn count_sql(compiled: &Compiled) -> String {
|
||||
format!("SELECT count(*) FROM images WHERE {}", compiled.where_sql)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use dr_types::{CollectionId, RootId};
|
||||
|
||||
#[test]
|
||||
fn unfiltered_compiles_to_a_constant() {
|
||||
let c = compile(&Selector::All, 0);
|
||||
assert_eq!(c.where_sql, "1");
|
||||
assert!(c.params.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_conjunction_and_disjunction_differ() {
|
||||
// The distinction that decides whether clearing a filter shows
|
||||
// everything or nothing.
|
||||
assert_eq!(compile(&Selector::All_(vec![]), 0).where_sql, "1");
|
||||
assert_eq!(compile(&Selector::Any(vec![]), 0).where_sql, "0");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn values_bind_rather_than_interpolate() {
|
||||
// The injection guard: a hostile keyword must appear in params, never
|
||||
// in SQL text.
|
||||
let evil = "'; DROP TABLE images; --";
|
||||
let c = compile(&Selector::Keyword(evil.into()), 0);
|
||||
assert!(!c.where_sql.contains("DROP"));
|
||||
assert_eq!(c.params, vec![Value::Text(evil.into())]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn like_metacharacters_are_escaped() {
|
||||
// A search for "50%" must not match everything containing "50".
|
||||
let c = compile(&Selector::Text("50%".into()), 0);
|
||||
assert_eq!(c.params, vec![Value::Text("%50\\%%".into())]);
|
||||
assert!(c.where_sql.contains("ESCAPE"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_backslash_in_search_text_is_itself_escaped() {
|
||||
let c = compile(&Selector::Text("a\\b".into()), 0);
|
||||
assert_eq!(c.params, vec![Value::Text("%a\\\\b%".into())]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rolling_window_resolves_against_supplied_now() {
|
||||
// Passed in rather than read from the clock, so the window is stable
|
||||
// across one query and reproducible in a test.
|
||||
let now = 1_000_000i64;
|
||||
let c = compile(
|
||||
&Selector::DateRange(DateSelector::Rolling { days: 90 }),
|
||||
now,
|
||||
);
|
||||
assert_eq!(c.params, vec![Value::Integer(now - 90 * 86_400)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn between_is_half_open() {
|
||||
let c = compile(
|
||||
&Selector::DateRange(DateSelector::Between { from: 10, to: 20 }),
|
||||
0,
|
||||
);
|
||||
// Half-open so adjacent day buckets neither overlap nor leave a gap.
|
||||
assert!(c.where_sql.contains(">= ?1"));
|
||||
assert!(c.where_sql.contains("< ?2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nested_composition_numbers_parameters_in_order() {
|
||||
let s = Selector::All_(vec![
|
||||
Selector::Rating { min: 4 },
|
||||
Selector::Any(vec![
|
||||
Selector::Camera("X-T5".into()),
|
||||
Selector::Not(Box::new(Selector::Lens("XF 35".into()))),
|
||||
]),
|
||||
]);
|
||||
let c = compile(&s, 0);
|
||||
assert_eq!(
|
||||
c.params,
|
||||
vec![
|
||||
Value::Integer(4),
|
||||
Value::Text("X-T5".into()),
|
||||
Value::Text("XF 35".into()),
|
||||
]
|
||||
);
|
||||
assert!(c.where_sql.contains("?1"));
|
||||
assert!(c.where_sql.contains("?2"));
|
||||
assert!(c.where_sql.contains("?3"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recursive_folder_matches_the_folder_itself_and_below() {
|
||||
let c = compile(
|
||||
&Selector::Folder {
|
||||
root: RootId(1),
|
||||
path: "2026/08".into(),
|
||||
recursive: true,
|
||||
},
|
||||
0,
|
||||
);
|
||||
// Both branches: the folder's own images and those in subfolders.
|
||||
assert!(c.where_sql.contains("path = ?2"));
|
||||
assert!(c.where_sql.contains("|| '/%'"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn collection_span_binds_its_id_once_and_reuses_it() {
|
||||
let c = compile(
|
||||
&Selector::DateRange(DateSelector::CollectionSpan(CollectionId(7))),
|
||||
0,
|
||||
);
|
||||
assert_eq!(c.params, vec![Value::Integer(7)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn capture_time_dependency_is_reported() {
|
||||
let c = compile(&Selector::DateRange(DateSelector::Rolling { days: 7 }), 0);
|
||||
assert!(c.needs_capture_time);
|
||||
let c = compile(&Selector::Rating { min: 5 }, 0);
|
||||
assert!(!c.needs_capture_time);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn capture_sort_puts_unknown_timestamps_last_in_both_directions() {
|
||||
// An image whose EXIF has not been read yet must not lead the grid
|
||||
// just because its timestamp is NULL.
|
||||
assert!(Sort::CapturedAt.sql(true).contains("IS NULL"));
|
||||
assert!(Sort::CapturedAt.sql(false).contains("IS NULL"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn window_sql_joins_only_when_the_sort_needs_it() {
|
||||
let c = compile(&Selector::All, 0);
|
||||
let plain = window_sql(
|
||||
&Query {
|
||||
filter: Selector::All,
|
||||
sort: Sort::CapturedAt,
|
||||
descending: true,
|
||||
},
|
||||
&c,
|
||||
);
|
||||
assert!(!plain.contains("JOIN"));
|
||||
|
||||
let rated = window_sql(
|
||||
&Query {
|
||||
filter: Selector::All,
|
||||
sort: Sort::Rating,
|
||||
descending: true,
|
||||
},
|
||||
&c,
|
||||
);
|
||||
assert!(rated.contains("JOIN versions"));
|
||||
}
|
||||
}
|
||||
@@ -1,662 +0,0 @@
|
||||
//! TRACES: NFR-R2 | NFR-R6
|
||||
//! What to do once the index is already damaged.
|
||||
//!
|
||||
//! # Why this can be a small module
|
||||
//!
|
||||
//! Because of a property the rest of the catalog was built to keep: the
|
||||
//! catalog is an *index*, not a source of truth (ARCH §6.12, invariant
|
||||
//! §5.2.4). Sidecars beside the images hold the authoritative ratings,
|
||||
//! keywords and edit graphs, for every catalogued image and whether or not a
|
||||
//! remote account exists (FR-CAT-8). So the worst outcome available here is a
|
||||
//! rescan — expensive, but not a loss.
|
||||
//!
|
||||
//! That is the second offer. The first is cheaper and loses nothing at all: a
|
||||
//! backup, restored.
|
||||
//!
|
||||
//! # The one thing a rebuild does not recover
|
||||
//!
|
||||
//! **Collections.** A manual collection is a set of images the user assembled
|
||||
//! by hand and nothing in the filesystem records it (`docs/catalog.md` §8.1) —
|
||||
//! which is the whole reason the catalog file itself syncs. So the two offers
|
||||
//! are not interchangeable, and the interface must not present them as if they
|
||||
//! were: a restore keeps the user's collections, a rebuild does not.
|
||||
//!
|
||||
//! # When the check runs, and when it does not
|
||||
//!
|
||||
//! [`integrity_check`] reads every page of the database. That is affordable
|
||||
//! once, at startup, where a failure has a user in front of it who can answer
|
||||
//! a question — and it is *not* affordable on every [`Catalog::open`], which
|
||||
//! this application does per background task, dozens of times a session. So
|
||||
//! the check is bound to [`Catalog::open_verified`] rather than to `open`,
|
||||
//! and the cheap half of the story — classifying `SQLITE_CORRUPT` and
|
||||
//! `SQLITE_NOTADB` as [`CatalogError::Corrupt`] — happens for free on every
|
||||
//! query through [`crate::error`]'s conversion. A background job that trips
|
||||
//! over the damage first therefore reports the same thing the startup check
|
||||
//! would have.
|
||||
//!
|
||||
//! [`Catalog::open`]: crate::Catalog::open
|
||||
//! [`Catalog::open_verified`]: crate::Catalog::open_verified
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use rusqlite::Connection;
|
||||
|
||||
use crate::error::CatalogError;
|
||||
use crate::schema;
|
||||
|
||||
/// Directory backups live in, relative to the catalog file.
|
||||
///
|
||||
/// Beside the catalog rather than in the cache directory, and that is the
|
||||
/// point of the choice: this is the copy the user falls back on, and a cache
|
||||
/// is a place the operating system is entitled to empty without asking
|
||||
/// (see `library::data_root` for the same reasoning about sidecars).
|
||||
const BACKUP_DIR: &str = "backups";
|
||||
|
||||
/// How many backups are kept.
|
||||
///
|
||||
/// Small on purpose. A backup is a full copy of a catalog that is tens of
|
||||
/// megabytes at 50k images, and the value of the third-oldest one is close to
|
||||
/// zero: corruption is noticed at the next launch, not months later. What the
|
||||
/// depth buys is protection against backing *up* the damage — if a corrupt
|
||||
/// catalog is copied before anyone notices, the generation behind it is still
|
||||
/// clean.
|
||||
pub const KEEP_BACKUPS: usize = 3;
|
||||
|
||||
/// Suffix given to a catalog that has been set aside as damaged.
|
||||
///
|
||||
/// Kept rather than deleted. It costs disk this application would rather not
|
||||
/// spend, and it is still the right call: `.sqlite` files have been recovered
|
||||
/// by hand before, the user has not consented to a deletion, and NFR-R4's
|
||||
/// instinct — never destroy what the user did not ask you to destroy — does
|
||||
/// not stop applying at the catalog's edge.
|
||||
const DAMAGED_SUFFIX: &str = "damaged";
|
||||
|
||||
/// One kept backup.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Backup {
|
||||
pub path: PathBuf,
|
||||
/// UTC seconds at which it was taken, read from the filename rather than
|
||||
/// from the filesystem: a copy, a restore or a sync can rewrite an mtime,
|
||||
/// and then the newest backup is not the one that looks newest.
|
||||
pub taken_at: i64,
|
||||
pub bytes: u64,
|
||||
}
|
||||
|
||||
/// Where backups for `catalog` are kept.
|
||||
pub fn backup_dir(catalog: &Path) -> PathBuf {
|
||||
catalog
|
||||
.parent()
|
||||
.unwrap_or_else(|| Path::new("."))
|
||||
.join(BACKUP_DIR)
|
||||
}
|
||||
|
||||
/// Check the database this connection is attached to.
|
||||
///
|
||||
/// `quick_check` rather than `integrity_check`: the difference is that
|
||||
/// `quick_check` skips verifying that every index agrees with its table, which
|
||||
/// is the expensive half and the half this application least needs — every
|
||||
/// index here is derivable, and `REINDEX` fixes one without anybody being
|
||||
/// asked a question. What is left still reads every page, and catches the
|
||||
/// damage that matters: torn b-trees, a bad freelist, a truncated file.
|
||||
///
|
||||
/// Returns [`CatalogError::Corrupt`] carrying what SQLite said, so the message
|
||||
/// the user sees is the diagnosis rather than a paraphrase of it.
|
||||
pub fn integrity_check(conn: &Connection) -> Result<(), CatalogError> {
|
||||
// The argument caps how many problems are reported. One is enough: the
|
||||
// answer is the same whether the file has one damaged page or nine
|
||||
// hundred, and an unbounded check on a badly damaged file can run for a
|
||||
// very long time producing a list nobody will read.
|
||||
let mut stmt = conn.prepare("PRAGMA quick_check(1)")?;
|
||||
let rows: Vec<String> = stmt
|
||||
.query_map([], |r| r.get(0))?
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
|
||||
// A healthy database answers with the single row "ok".
|
||||
if rows.len() == 1 && rows[0] == "ok" {
|
||||
return Ok(());
|
||||
}
|
||||
Err(CatalogError::Corrupt {
|
||||
detail: rows.join("; "),
|
||||
})
|
||||
}
|
||||
|
||||
/// Check a catalog file that is not currently open.
|
||||
///
|
||||
/// Used before a restore: a backup is only worth swapping in if it is sound,
|
||||
/// and swapping in a second damaged file — leaving the user with no catalog
|
||||
/// and no offer left — is the failure this exists to prevent.
|
||||
pub fn check_file(path: &Path) -> Result<(), CatalogError> {
|
||||
if !path.is_file() {
|
||||
return Err(CatalogError::Io(format!("{} is missing", path.display())));
|
||||
}
|
||||
// Read-write rather than read-only, which reads oddly for a check. A
|
||||
// backup carries the WAL journal mode in its header because it was copied
|
||||
// page-for-page from a WAL database, and SQLite cannot open one read-only
|
||||
// without a shared-memory file it is then not allowed to create. Nothing
|
||||
// here writes; the connection is opened, read, and dropped.
|
||||
let conn = Connection::open(path)?;
|
||||
integrity_check(&conn)
|
||||
}
|
||||
|
||||
/// Take a backup of the open catalog.
|
||||
///
|
||||
/// Returns the file written. Older generations beyond [`KEEP_BACKUPS`] are
|
||||
/// pruned, newest kept.
|
||||
///
|
||||
/// Goes through `crate::sync::copy_to` — SQLite's own backup API after a
|
||||
/// TRUNCATE checkpoint — rather than copying the file. A WAL database is not
|
||||
/// one file, and `fs::copy` of the main file alone would silently back up a
|
||||
/// state that is older than the catalog and possibly torn, which is the one
|
||||
/// failure mode a backup cannot afford.
|
||||
pub fn backup(conn: &Connection, catalog: &Path) -> Result<PathBuf, CatalogError> {
|
||||
let dir = backup_dir(catalog);
|
||||
std::fs::create_dir_all(&dir)
|
||||
.map_err(|e| CatalogError::Io(format!("creating {}: {e}", dir.display())))?;
|
||||
|
||||
let dest = dir.join(format!("catalog-{}.sqlite", now()));
|
||||
// A second backup within the same second would otherwise land on the first
|
||||
// one's name. Rare, and only reachable from tests and a retry, but the
|
||||
// result would be a half-overwritten backup rather than two.
|
||||
if dest.exists() {
|
||||
std::fs::remove_file(&dest)
|
||||
.map_err(|e| CatalogError::Io(format!("replacing {}: {e}", dest.display())))?;
|
||||
}
|
||||
|
||||
// Dropped immediately: the copy is complete when `copy_to` returns, and
|
||||
// holding the connection open would leave a `-wal` beside a file whose
|
||||
// whole purpose is to be a single self-contained artefact.
|
||||
drop(crate::sync::copy_to(conn, &dest)?);
|
||||
|
||||
prune(catalog);
|
||||
Ok(dest)
|
||||
}
|
||||
|
||||
/// Back up before a migration, if there is anything to back up.
|
||||
///
|
||||
/// Called from [`Catalog::open`](crate::Catalog::open) between `configure` and
|
||||
/// `migrate`. NFR-R2 asks for this and the reasoning is narrower than "backups
|
||||
/// are prudent": a migration is the one routine operation that rewrites table
|
||||
/// structure, so it is the likeliest way a catalog becomes unreadable, and it
|
||||
/// is the one moment where the pre-change state is still on disk to be copied.
|
||||
/// Afterwards there is nothing left to take a copy *of*.
|
||||
///
|
||||
/// A no-op in the two cases where it would cost without buying anything: a
|
||||
/// catalog already at [`schema::SCHEMA_VERSION`], and a brand-new file at
|
||||
/// version 0 with no tables in it yet.
|
||||
pub fn backup_before_migration(conn: &Connection, catalog: &Path) -> Result<(), CatalogError> {
|
||||
let from: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
|
||||
if from == 0 || from >= schema::SCHEMA_VERSION {
|
||||
return Ok(());
|
||||
}
|
||||
let path = backup(conn, catalog)?;
|
||||
log::info!(
|
||||
"backed up catalog at v{from} to {} before migrating to v{}",
|
||||
path.display(),
|
||||
schema::SCHEMA_VERSION
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The backups available for `catalog`, newest first.
|
||||
///
|
||||
/// Never fails: an unreadable or absent backup directory means there are no
|
||||
/// backups, which is a fact about the offer to make rather than an error to
|
||||
/// report on top of the corruption the user is already looking at.
|
||||
pub fn backups(catalog: &Path) -> Vec<Backup> {
|
||||
let dir = backup_dir(catalog);
|
||||
let Ok(entries) = std::fs::read_dir(&dir) else {
|
||||
return Vec::new();
|
||||
};
|
||||
|
||||
let mut out: Vec<Backup> = entries
|
||||
.flatten()
|
||||
.filter_map(|e| {
|
||||
let path = e.path();
|
||||
let taken_at = timestamp_of(&path)?;
|
||||
let bytes = e.metadata().ok()?.len();
|
||||
Some(Backup {
|
||||
path,
|
||||
taken_at,
|
||||
bytes,
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
out.sort_by_key(|b| std::cmp::Reverse(b.taken_at));
|
||||
out
|
||||
}
|
||||
|
||||
/// Put a backup back in place of the damaged catalog.
|
||||
///
|
||||
/// **Every connection to `catalog` must be closed first.** This replaces the
|
||||
/// file underneath anything still holding it open, which on a live connection
|
||||
/// is how a *second* corrupt catalog gets made.
|
||||
///
|
||||
/// The order is deliberate:
|
||||
///
|
||||
/// 1. The backup is checked. A restore that installs a second damaged file
|
||||
/// leaves the user with nothing to try next.
|
||||
/// 2. The damaged catalog is renamed aside, and its `-wal` and `-shm` are
|
||||
/// **deleted**. This is the step that is easy to leave out and fatal to
|
||||
/// leave out: a journal belonging to the old file, sitting beside the new
|
||||
/// one under the same name, is replayed into it on the next open. That is
|
||||
/// not a restore, it is a fresh corruption with the evidence gone.
|
||||
/// 3. The backup is *copied* into place, not moved, so a failure here can be
|
||||
/// retried against the same backup.
|
||||
pub fn restore(catalog: &Path, backup: &Path) -> Result<(), CatalogError> {
|
||||
check_file(backup)?;
|
||||
set_aside(catalog)?;
|
||||
std::fs::copy(backup, catalog).map_err(|e| {
|
||||
CatalogError::Io(format!(
|
||||
"restoring {} from {}: {e}",
|
||||
catalog.display(),
|
||||
backup.display()
|
||||
))
|
||||
})?;
|
||||
log::info!(
|
||||
"restored {} from backup {}",
|
||||
catalog.display(),
|
||||
backup.display()
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Move a damaged catalog out of the way so the next open builds a fresh one.
|
||||
///
|
||||
/// This is the rebuild path (NFR-R6's second offer) and also the first half of
|
||||
/// a [`restore`]. Nothing else is needed to rebuild: the next
|
||||
/// [`Catalog::open`](crate::Catalog::open) creates an empty catalog at the
|
||||
/// current schema, and the ordinary scan repopulates it from sources and
|
||||
/// sidecars — which is precisely invariant §5.2.4 being spent rather than
|
||||
/// merely asserted.
|
||||
///
|
||||
/// Returns where the damaged file was put, or `None` if there was no catalog
|
||||
/// to move — a caller may be recovering from a file SQLite could not open
|
||||
/// because it was never created.
|
||||
pub fn set_aside(catalog: &Path) -> Result<Option<PathBuf>, CatalogError> {
|
||||
let moved = if catalog.exists() {
|
||||
let dest = with_suffix(catalog, DAMAGED_SUFFIX);
|
||||
// An earlier damaged copy is replaced rather than accumulating: two of
|
||||
// these is two full-size catalogs on the user's disk, and the older
|
||||
// one has already been superseded by a recovery the user completed.
|
||||
let _ = std::fs::remove_file(&dest);
|
||||
// The rename first, so that a failure here leaves the journals with
|
||||
// the file they belong to rather than orphaned beside a catalog that
|
||||
// is still in use.
|
||||
std::fs::rename(catalog, &dest).map_err(|e| {
|
||||
CatalogError::Io(format!(
|
||||
"setting aside {} as {}: {e}",
|
||||
catalog.display(),
|
||||
dest.display()
|
||||
))
|
||||
})?;
|
||||
log::warn!(
|
||||
"catalog {} was damaged; kept as {}",
|
||||
catalog.display(),
|
||||
dest.display()
|
||||
);
|
||||
Some(dest)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
// Then the journals, whether or not there was a catalog to move: a `-wal`
|
||||
// orphaned beside a missing database is replayed into whatever takes that
|
||||
// name next, which would not be a restore but a fresh corruption with the
|
||||
// evidence gone.
|
||||
for sidecar in journals(catalog) {
|
||||
if let Err(e) = std::fs::remove_file(&sidecar) {
|
||||
if e.kind() != std::io::ErrorKind::NotFound {
|
||||
return Err(CatalogError::Io(format!(
|
||||
"removing stale journal {}: {e}",
|
||||
sidecar.display()
|
||||
)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(moved)
|
||||
}
|
||||
|
||||
/// Delete backups beyond [`KEEP_BACKUPS`].
|
||||
///
|
||||
/// Best-effort and silent about individual failures: failing to delete an old
|
||||
/// backup is not a reason to fail the new one, which is already written.
|
||||
fn prune(catalog: &Path) {
|
||||
for old in backups(catalog).into_iter().skip(KEEP_BACKUPS) {
|
||||
if let Err(e) = std::fs::remove_file(&old.path) {
|
||||
log::warn!("could not prune backup {}: {e}", old.path.display());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The WAL and shared-memory files SQLite keeps beside a database.
|
||||
fn journals(catalog: &Path) -> [PathBuf; 2] {
|
||||
[with_suffix(catalog, "wal"), with_suffix(catalog, "shm")]
|
||||
}
|
||||
|
||||
/// `catalog.sqlite` plus `-suffix`, the way SQLite names its own sidecars.
|
||||
///
|
||||
/// Appended to the whole filename rather than replacing the extension, so
|
||||
/// `catalog.sqlite-wal` is what SQLite would look for and `catalog.sqlite-
|
||||
/// damaged` sorts next to the catalog it came from.
|
||||
fn with_suffix(catalog: &Path, suffix: &str) -> PathBuf {
|
||||
let mut s = catalog.as_os_str().to_os_string();
|
||||
s.push("-");
|
||||
s.push(suffix);
|
||||
PathBuf::from(s)
|
||||
}
|
||||
|
||||
/// Read the timestamp out of a backup's filename, or `None` if this is not one.
|
||||
///
|
||||
/// Doubles as the filter that keeps [`backups`] from offering the user
|
||||
/// something that is not a catalog — a stray file in the directory, or a `-wal`
|
||||
/// left by a crash mid-backup.
|
||||
fn timestamp_of(path: &Path) -> Option<i64> {
|
||||
let name = path.file_name()?.to_str()?;
|
||||
name.strip_prefix("catalog-")?
|
||||
.strip_suffix(".sqlite")?
|
||||
.parse()
|
||||
.ok()
|
||||
}
|
||||
|
||||
/// Seconds since the epoch, or 0 if the clock is before it.
|
||||
fn now() -> i64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.map(|d| d.as_secs() as i64)
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::Catalog;
|
||||
use std::io::{Seek, SeekFrom, Write};
|
||||
|
||||
/// A scratch directory that cleans up with the test.
|
||||
fn tempdir(tag: &str) -> PathBuf {
|
||||
let base = std::env::temp_dir().join(format!(
|
||||
"dr-recovery-{tag}-{}-{:?}",
|
||||
std::process::id(),
|
||||
std::thread::current().id()
|
||||
));
|
||||
let _ = std::fs::remove_dir_all(&base);
|
||||
std::fs::create_dir_all(&base).unwrap();
|
||||
base
|
||||
}
|
||||
|
||||
/// A catalog on disk with enough rows to span several pages, closed.
|
||||
///
|
||||
/// Closed matters: WAL means the rows are in `catalog.sqlite-wal` until
|
||||
/// something checkpoints, and a test that corrupted the main file while
|
||||
/// the data was still in the journal would be corrupting empty space.
|
||||
fn fixture(path: &Path, images: i64) {
|
||||
let cat = Catalog::open(path).unwrap();
|
||||
let c = cat.connection();
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
|
||||
VALUES ('u1', 'Iceland', 0, 0, 1, 1)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1..=images {
|
||||
c.execute(
|
||||
"INSERT INTO images(id, root_id, source_ref, added_at)
|
||||
VALUES (?1, 1, ?2, 0)",
|
||||
rusqlite::params![i, format!("DCIM/IMG_{i:05}.CR3")],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
crate::sync::checkpoint(c).unwrap();
|
||||
drop(cat);
|
||||
}
|
||||
|
||||
/// Scribble over everything past the first two pages.
|
||||
///
|
||||
/// Past them rather than over them so that page 1 — the header and the
|
||||
/// schema — survives: this produces a file SQLite is willing to open and
|
||||
/// then finds damaged, which is the case `quick_check` exists for. Wiping
|
||||
/// the header instead produces `SQLITE_NOTADB` at the first pragma, which
|
||||
/// is a different branch and has its own test.
|
||||
fn corrupt(path: &Path) {
|
||||
let mut f = std::fs::OpenOptions::new().write(true).open(path).unwrap();
|
||||
let len = f.metadata().unwrap().len();
|
||||
assert!(
|
||||
len > 8192,
|
||||
"fixture is only {len} bytes; corrupting past page 2 would be a no-op"
|
||||
);
|
||||
let junk = vec![0x5a_u8; (len - 8192) as usize];
|
||||
f.seek(SeekFrom::Start(8192)).unwrap();
|
||||
f.write_all(&junk).unwrap();
|
||||
f.sync_all().unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_healthy_catalog_passes() {
|
||||
let cat = Catalog::in_memory().unwrap();
|
||||
integrity_check(cat.connection()).unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_corrupt_catalog_is_reported_as_corrupt_not_as_sqlite() {
|
||||
// The whole point of the variant: this used to arrive as whatever
|
||||
// rusqlite error the first failing query produced, with nowhere to
|
||||
// hang a recovery offer.
|
||||
let dir = tempdir("detect");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
fixture(&path, 500);
|
||||
corrupt(&path);
|
||||
|
||||
assert!(matches!(
|
||||
Catalog::open_verified(&path),
|
||||
Err(CatalogError::Corrupt { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_file_that_is_not_a_database_is_also_corrupt() {
|
||||
// A truncated or overwritten catalog never reaches `quick_check`: the
|
||||
// first pragma fails with SQLITE_NOTADB. Same accident to the user,
|
||||
// same two offers, so it must classify the same way.
|
||||
let dir = tempdir("notadb");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
std::fs::write(&path, b"this is not a catalog, it is a text file\n").unwrap();
|
||||
|
||||
assert!(matches!(
|
||||
Catalog::open(&path),
|
||||
Err(CatalogError::Corrupt { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn restoring_a_backup_recovers_the_collections_a_rebuild_would_lose() {
|
||||
// The first NFR-R6 branch, asserted on the thing that distinguishes it
|
||||
// from the second: a collection exists nowhere but the catalog, so it
|
||||
// is the evidence that the *contents* came back and not merely a
|
||||
// readable file (docs/catalog.md §8.1).
|
||||
let dir = tempdir("restore");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
fixture(&path, 500);
|
||||
|
||||
{
|
||||
let cat = Catalog::open(&path).unwrap();
|
||||
backup(cat.connection(), &path).unwrap();
|
||||
}
|
||||
corrupt(&path);
|
||||
assert!(matches!(
|
||||
Catalog::open_verified(&path),
|
||||
Err(CatalogError::Corrupt { .. })
|
||||
));
|
||||
|
||||
let newest = backups(&path).into_iter().next().expect("a backup exists");
|
||||
restore(&path, &newest.path).unwrap();
|
||||
|
||||
let cat = Catalog::open_verified(&path).unwrap();
|
||||
let name: String = cat
|
||||
.connection()
|
||||
.query_row("SELECT name FROM collections", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(name, "Iceland");
|
||||
let images: i64 = cat
|
||||
.connection()
|
||||
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(images, 500);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_damaged_backup_is_refused_rather_than_installed() {
|
||||
let dir = tempdir("badbackup");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
fixture(&path, 500);
|
||||
{
|
||||
let cat = Catalog::open(&path).unwrap();
|
||||
backup(cat.connection(), &path).unwrap();
|
||||
}
|
||||
let newest = backups(&path).into_iter().next().unwrap();
|
||||
corrupt(&newest.path);
|
||||
corrupt(&path);
|
||||
|
||||
assert!(matches!(
|
||||
restore(&path, &newest.path),
|
||||
Err(CatalogError::Corrupt { .. })
|
||||
));
|
||||
// And the damaged catalog is still where it was, so the second offer
|
||||
// is still available.
|
||||
assert!(path.exists());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn setting_aside_leaves_a_fresh_catalog_to_rebuild_into() {
|
||||
// The second NFR-R6 branch. What makes it a rebuild rather than a data
|
||||
// loss is invariant §5.2.4, which lives outside this crate — what is
|
||||
// testable here is that the damaged file is out of the way, kept, and
|
||||
// that the next open succeeds on an empty catalog at the current
|
||||
// schema, which is what a scan then fills.
|
||||
let dir = tempdir("rebuild");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
fixture(&path, 500);
|
||||
corrupt(&path);
|
||||
|
||||
let kept = set_aside(&path).unwrap().expect("the catalog was there");
|
||||
assert!(kept.exists(), "the damaged catalog was deleted, not kept");
|
||||
assert!(!path.exists());
|
||||
|
||||
let cat = Catalog::open_verified(&path).unwrap();
|
||||
let images: i64 = cat
|
||||
.connection()
|
||||
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(images, 0);
|
||||
let v: i64 = cat
|
||||
.connection()
|
||||
.query_row("PRAGMA user_version", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(v, schema::SCHEMA_VERSION);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_stale_journal_does_not_follow_the_catalog_into_recovery() {
|
||||
// The step that is easy to omit: a `-wal` belonging to the damaged
|
||||
// file is replayed into whatever takes its name next.
|
||||
let dir = tempdir("journal");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
fixture(&path, 500);
|
||||
std::fs::write(with_suffix(&path, "wal"), b"stale").unwrap();
|
||||
|
||||
set_aside(&path).unwrap();
|
||||
assert!(!with_suffix(&path, "wal").exists());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_migration_is_backed_up_before_it_runs() {
|
||||
// NFR-R2's second clause, against a real v1 catalog rather than a
|
||||
// faked version number: the point is not that *a* file appears but
|
||||
// that it holds the state from before the migration, which is the only
|
||||
// state that is any use if the migration is what breaks it.
|
||||
let dir = tempdir("premigrate");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
{
|
||||
let c = Connection::open(&path).unwrap();
|
||||
schema::configure(&c).unwrap();
|
||||
// `v1_for_attached` names the schema it targets, and "main" is a
|
||||
// schema like any other — so this is the real v1, without needing
|
||||
// `V1` itself to become visible outside its module.
|
||||
c.execute_batch(&schema::v1_for_attached("main")).unwrap();
|
||||
c.pragma_update(None, "user_version", 1).unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
crate::sync::checkpoint(&c).unwrap();
|
||||
}
|
||||
assert!(backups(&path).is_empty());
|
||||
|
||||
Catalog::open(&path).unwrap();
|
||||
|
||||
let taken = backups(&path);
|
||||
assert_eq!(taken.len(), 1, "no backup was taken before the migration");
|
||||
check_file(&taken[0].path).unwrap();
|
||||
let kept = Connection::open(&taken[0].path).unwrap();
|
||||
let v: i64 = kept
|
||||
.query_row("PRAGMA user_version", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(v, 1, "the backup was taken after the migration, not before");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn opening_an_up_to_date_catalog_takes_no_backup() {
|
||||
// Or every background task that opens the catalog would copy it.
|
||||
let dir = tempdir("nobackup");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
fixture(&path, 10);
|
||||
Catalog::open(&path).unwrap();
|
||||
assert!(backups(&path).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_the_newest_generations_are_kept() {
|
||||
let dir = tempdir("prune");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
fixture(&path, 10);
|
||||
let cat = Catalog::open(&path).unwrap();
|
||||
|
||||
// Written by hand rather than by calling `backup` in a loop: the
|
||||
// filename carries whole seconds, so real calls would collide.
|
||||
std::fs::create_dir_all(backup_dir(&path)).unwrap();
|
||||
for t in 1..=KEEP_BACKUPS as i64 + 2 {
|
||||
drop(
|
||||
crate::sync::copy_to(
|
||||
cat.connection(),
|
||||
&backup_dir(&path).join(format!("catalog-{t}.sqlite")),
|
||||
)
|
||||
.unwrap(),
|
||||
);
|
||||
}
|
||||
prune(&path);
|
||||
|
||||
let kept = backups(&path);
|
||||
assert_eq!(kept.len(), KEEP_BACKUPS);
|
||||
// Newest first, and the newest is the highest timestamp.
|
||||
assert_eq!(kept[0].taken_at, KEEP_BACKUPS as i64 + 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_stray_file_in_the_backup_directory_is_not_offered_as_one() {
|
||||
let dir = tempdir("stray");
|
||||
let path = dir.join("catalog.sqlite");
|
||||
fixture(&path, 10);
|
||||
std::fs::create_dir_all(backup_dir(&path)).unwrap();
|
||||
std::fs::write(backup_dir(&path).join("notes.txt"), b"hello").unwrap();
|
||||
std::fs::write(backup_dir(&path).join("catalog-7.sqlite-wal"), b"x").unwrap();
|
||||
|
||||
assert!(backups(&path).is_empty());
|
||||
}
|
||||
}
|
||||
@@ -1,935 +0,0 @@
|
||||
//! TRACES: FR-PLAT-AND-4 | FR-PLAT-AND-3
|
||||
//! The thing that drains the queue.
|
||||
//!
|
||||
//! [`crate::jobs`] has been a complete, durable, coalescing work queue since
|
||||
//! the catalog was written, and nothing has ever taken a job out of it. Every
|
||||
//! producer — the local walk, the remote scan — called `enqueue` and no one
|
||||
//! called `claim_next`, so the table grew one row per photograph and stayed
|
||||
//! that size forever. This module is the missing half.
|
||||
//!
|
||||
//! # Why the runner is driven rather than self-owning
|
||||
//!
|
||||
//! The obvious shape is a thread that loops until the queue is empty, and it
|
||||
//! is the wrong one. On Android the process does not decide when background
|
||||
//! work may run: `WorkManager` does, subject to Doze, battery saver and the
|
||||
//! metered-network constraints in FR-NC-6, and it revokes permission mid-job
|
||||
//! by calling `onStopped()` (FR-PLAT-AND-4). A foreground service for a
|
||||
//! user-initiated export gets a longer leash but still not an unbounded one.
|
||||
//!
|
||||
//! So the runner owns no thread, no clock and no policy. It exposes
|
||||
//! [`Runner::run_one`] — claim one job, run it, record what happened — and
|
||||
//! [`Runner::drain`], which repeats that against a [`Budget`] and a
|
||||
//! cancellation flag the host owns. A `Worker.doWork()` that must return
|
||||
//! within ten minutes calls `drain` with a deadline; a desktop idle loop calls
|
||||
//! it with none. Neither has to reach inside.
|
||||
//!
|
||||
//! Everything the host supplies is passed in for the same reason `jobs` takes
|
||||
//! `now` rather than reading the clock: a scheduler is exactly the thing that
|
||||
//! has to be testable without waiting.
|
||||
//!
|
||||
//! # Why interruption is not failure
|
||||
//!
|
||||
//! Four things can happen to a claimed job, and only two of them are the job's
|
||||
//! fault:
|
||||
//!
|
||||
//! - [`Outcome::Done`] — the row is deleted.
|
||||
//! - [`Outcome::Retry`] — the work failed and might succeed later. Backoff,
|
||||
//! and eventually [`crate::jobs::MAX_ATTEMPTS`] gives up on it.
|
||||
//! - [`Outcome::Abandon`] — the work cannot succeed, ever. Failing five times
|
||||
//! over five minutes to learn that is five minutes of a phone's battery.
|
||||
//! - [`Outcome::Interrupted`] — the *host* stopped, not the job. The claim is
|
||||
//! released and the attempt it consumed is given back, because a user who
|
||||
//! pulled the app off the screen has not told us anything about the file.
|
||||
//!
|
||||
//! Process death is the fifth case and the one that cannot report itself: the
|
||||
//! row simply stays `Running` with no owner. [`Runner::recover`] is what
|
||||
//! reclaims it, and it is why an interrupted job is resumable rather than lost
|
||||
//! (FR-PLAT-AND-3). It must run **before** any worker starts against a
|
||||
//! catalog, or it will steal a job another runner is holding — there is no
|
||||
//! owner column to tell them apart.
|
||||
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
|
||||
use rusqlite::Connection;
|
||||
|
||||
use crate::error::CatalogError;
|
||||
use crate::jobs::{self, Job, JobKind};
|
||||
|
||||
/// What running a job turned out to be.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Outcome {
|
||||
/// The work is done. The row goes away.
|
||||
Done,
|
||||
/// It failed, and trying again later is reasonable. Backoff applies, and
|
||||
/// [`crate::jobs::MAX_ATTEMPTS`] eventually stops it.
|
||||
Retry(String),
|
||||
/// It failed in a way no retry can fix — the subject is gone, the payload
|
||||
/// is unreadable, the format is one this build does not know. Marked
|
||||
/// failed at once rather than burning the whole retry ladder to reach the
|
||||
/// same answer.
|
||||
Abandon(String),
|
||||
/// The host is stopping, and the job never really ran.
|
||||
///
|
||||
/// Distinct from `Retry` because it costs no attempt: `onStopped()` five
|
||||
/// times in a row would otherwise mark a perfectly good job as failed.
|
||||
Interrupted,
|
||||
}
|
||||
|
||||
/// Something that can actually do the work a job describes.
|
||||
///
|
||||
/// The catalog knows what needs doing and nothing about how — a thumbnail
|
||||
/// needs a decoder, a fetch needs a network stack, and neither belongs under
|
||||
/// `core/dr-catalog` (ARCH §4.1: calls go downward). So the queue lives here
|
||||
/// and the handlers are supplied from above.
|
||||
pub trait JobHandler {
|
||||
/// The kinds this handler will accept.
|
||||
///
|
||||
/// Load-bearing, not documentation: the runner claims **only** kinds some
|
||||
/// handler declares. A queue holding `FetchOriginal` rows on a device with
|
||||
/// no connector must leave them alone rather than claim them and fail
|
||||
/// them, and a runner that claimed everything would do exactly that — five
|
||||
/// times each, with backoff, on battery.
|
||||
fn kinds(&self) -> &[JobKind];
|
||||
|
||||
/// Do the work.
|
||||
///
|
||||
/// The connection is offered because most handlers write their result back
|
||||
/// into the catalog; one that does not is free to ignore it. It is the
|
||||
/// runner's own connection, so a handler must not hold a transaction open
|
||||
/// across a network call — the runner needs it back to record the outcome.
|
||||
fn run(&mut self, conn: &Connection, job: &Job) -> Outcome;
|
||||
}
|
||||
|
||||
/// How much work a host is willing to let one drain do.
|
||||
///
|
||||
/// Both limits are checked *before* a job is claimed, never during one: a
|
||||
/// handler is opaque and may be halfway through writing a sidecar. Overrunning
|
||||
/// a deadline by one job is survivable; being killed mid-write is the thing
|
||||
/// [`Runner::recover`] exists to clean up after.
|
||||
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Budget {
|
||||
/// Stop after this many jobs. `None` means "until the queue is empty".
|
||||
pub max_jobs: Option<usize>,
|
||||
/// Stop once the clock reaches this second. Same clock the drain is given.
|
||||
pub deadline: Option<i64>,
|
||||
}
|
||||
|
||||
impl Budget {
|
||||
/// Run until nothing is left. What a desktop idle pass wants.
|
||||
pub const UNLIMITED: Self = Self {
|
||||
max_jobs: None,
|
||||
deadline: None,
|
||||
};
|
||||
|
||||
/// At most `n` jobs. A slice small enough to stay responsive.
|
||||
pub fn jobs(n: usize) -> Self {
|
||||
Self {
|
||||
max_jobs: Some(n),
|
||||
..Self::UNLIMITED
|
||||
}
|
||||
}
|
||||
|
||||
/// Until the clock reaches `deadline`. What a `WorkManager` slot wants.
|
||||
pub fn until(deadline: i64) -> Self {
|
||||
Self {
|
||||
deadline: Some(deadline),
|
||||
..Self::UNLIMITED
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Why a drain stopped.
|
||||
///
|
||||
/// Worth distinguishing because the host's next move differs: `Drained` means
|
||||
/// there is nothing to reschedule for, and the other three all mean "there is
|
||||
/// more, ask again".
|
||||
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Stopped {
|
||||
/// Nothing claimable is left.
|
||||
#[default]
|
||||
Drained,
|
||||
/// The job count ran out.
|
||||
Budget,
|
||||
/// The clock ran out.
|
||||
Deadline,
|
||||
/// The host asked it to stop, or a handler reported itself interrupted.
|
||||
Cancelled,
|
||||
}
|
||||
|
||||
impl Stopped {
|
||||
/// Whether the queue may still hold claimable work.
|
||||
pub fn more_to_do(self) -> bool {
|
||||
self != Stopped::Drained
|
||||
}
|
||||
}
|
||||
|
||||
/// What one drain did.
|
||||
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct DrainReport {
|
||||
pub completed: usize,
|
||||
pub retried: usize,
|
||||
pub abandoned: usize,
|
||||
pub interrupted: usize,
|
||||
pub stopped: Stopped,
|
||||
}
|
||||
|
||||
impl DrainReport {
|
||||
/// Jobs claimed, whatever became of them. This is what a budget counts.
|
||||
pub fn ran(&self) -> usize {
|
||||
self.completed + self.retried + self.abandoned + self.interrupted
|
||||
}
|
||||
}
|
||||
|
||||
/// What a recovery pass found waiting from the last run.
|
||||
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Recovered {
|
||||
/// Jobs a dead process was holding. These are the resumed ones.
|
||||
pub reclaimed: usize,
|
||||
/// Jobs deleted because the photograph they name no longer exists.
|
||||
pub reaped: usize,
|
||||
}
|
||||
|
||||
impl Recovered {
|
||||
pub fn did_anything(&self) -> bool {
|
||||
self.reclaimed > 0 || self.reaped > 0
|
||||
}
|
||||
}
|
||||
|
||||
/// Ready the queue for a fresh run, before any worker touches it.
|
||||
///
|
||||
/// Two distinct cleanups, and both are startup-only:
|
||||
///
|
||||
/// - **Reclaim.** A `Running` row has no owner; the process that claimed it is
|
||||
/// gone. On Android that is a routine morning, not a crash (FR-PLAT-AND-3).
|
||||
/// The attempt it consumed is *kept*, deliberately: a job that takes the
|
||||
/// process down with it three times running should not be retried forever,
|
||||
/// and the attempt counter is the only evidence of that we have.
|
||||
/// - **Reap.** Jobs naming an image the catalog no longer has. A library that
|
||||
/// has been culled leaves thumbnail jobs for photographs that were deleted
|
||||
/// months ago, and every one of them would be claimed, run and failed.
|
||||
///
|
||||
/// Reclaim runs first so its count is the honest number of interrupted jobs,
|
||||
/// before reaping removes whichever of them pointed at nothing.
|
||||
///
|
||||
/// **Call this exactly once per catalog, at startup.** It cannot distinguish a
|
||||
/// job a dead process was holding from one a live runner is holding right now,
|
||||
/// because there is no owner column — the queue is durable, not distributed.
|
||||
pub fn recover(conn: &Connection) -> Result<Recovered, CatalogError> {
|
||||
Ok(Recovered {
|
||||
reclaimed: jobs::recover_orphaned(conn)?,
|
||||
reaped: jobs::reap_orphan_subjects(conn)?,
|
||||
})
|
||||
}
|
||||
|
||||
/// Claims work, runs it, and records what happened.
|
||||
///
|
||||
/// Borrows its connection rather than owning one so a host can drive it from
|
||||
/// the same handle it already has open. Nothing here spawns a thread; several
|
||||
/// runners on several threads, each with its own connection to the same
|
||||
/// catalog, are safe because the claim is a single atomic statement (see
|
||||
/// [`crate::jobs::claim_next`]).
|
||||
pub struct Runner<'a> {
|
||||
conn: &'a Connection,
|
||||
handlers: Vec<Box<dyn JobHandler + 'a>>,
|
||||
/// The union of every handler's kinds, cached because it is passed to
|
||||
/// every claim. This is what stops the runner claiming work it cannot do.
|
||||
claimable: Vec<JobKind>,
|
||||
}
|
||||
|
||||
impl<'a> Runner<'a> {
|
||||
/// A runner with no handlers. It can recover, and it can claim nothing.
|
||||
pub fn new(conn: &'a Connection) -> Self {
|
||||
Self {
|
||||
conn,
|
||||
handlers: Vec::new(),
|
||||
claimable: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Add a handler.
|
||||
pub fn with(self, handler: impl JobHandler + 'a) -> Self {
|
||||
self.with_boxed(Box::new(handler))
|
||||
}
|
||||
|
||||
/// Add a handler chosen at runtime — a network one only where there is a
|
||||
/// connector, a decoding one only where there is a decoder.
|
||||
pub fn with_boxed(mut self, handler: Box<dyn JobHandler + 'a>) -> Self {
|
||||
for kind in handler.kinds() {
|
||||
if !self.claimable.contains(kind) {
|
||||
self.claimable.push(*kind);
|
||||
}
|
||||
}
|
||||
self.handlers.push(handler);
|
||||
self
|
||||
}
|
||||
|
||||
/// The kinds this runner will claim. Useful to a host deciding whether
|
||||
/// starting it is worth waking the radio for.
|
||||
pub fn claimable(&self) -> &[JobKind] {
|
||||
&self.claimable
|
||||
}
|
||||
|
||||
/// See [`recover`]. Offered here too so a host has one thing to hold.
|
||||
pub fn recover(&self) -> Result<Recovered, CatalogError> {
|
||||
recover(self.conn)
|
||||
}
|
||||
|
||||
/// Claim one job, run it, and record the outcome.
|
||||
///
|
||||
/// `Ok(None)` means nothing this runner can do is claimable *now* — the
|
||||
/// queue may still hold work of other kinds, or work still backing off.
|
||||
pub fn run_one(&mut self, now: i64) -> Result<Option<Ran>, CatalogError> {
|
||||
// Copied out before `self.handlers` is borrowed mutably below. Both
|
||||
// are fields of `self`, but the copy is what lets the two borrows
|
||||
// coexist without the connection being reborrowed through `self`.
|
||||
let conn = self.conn;
|
||||
|
||||
let Some(job) = jobs::claim_next_matching(conn, now, &self.claimable)? else {
|
||||
return Ok(None);
|
||||
};
|
||||
|
||||
let outcome = match self
|
||||
.handlers
|
||||
.iter_mut()
|
||||
.find(|h| h.kinds().contains(&job.kind))
|
||||
{
|
||||
Some(handler) => handler.run(conn, &job),
|
||||
// Unreachable by construction: `claimable` is exactly the union of
|
||||
// the handlers' kinds. Parked rather than released, because
|
||||
// releasing it would put it straight back where the next turn of
|
||||
// the drain loop would claim it again, forever.
|
||||
None => Outcome::Abandon(format!("no handler for {:?}", job.kind)),
|
||||
};
|
||||
|
||||
match &outcome {
|
||||
Outcome::Done => jobs::complete(conn, job.id)?,
|
||||
Outcome::Retry(why) => jobs::fail(conn, &job, now, why)?,
|
||||
Outcome::Abandon(why) => jobs::abandon(conn, job.id, why)?,
|
||||
Outcome::Interrupted => jobs::release(conn, &job)?,
|
||||
}
|
||||
|
||||
Ok(Some(Ran { job, outcome }))
|
||||
}
|
||||
|
||||
/// Run jobs until the budget, the flag or the queue says stop.
|
||||
///
|
||||
/// `clock` is called once per iteration rather than sampled once, because
|
||||
/// the two things it feeds both move during a long drain: the deadline
|
||||
/// check, and the `now` a failing job's backoff is measured from.
|
||||
///
|
||||
/// Cancellation is checked between jobs only. A handler that wants to bail
|
||||
/// out of work already started says so with [`Outcome::Interrupted`],
|
||||
/// which also ends the drain — otherwise a handler that always interrupts
|
||||
/// would release its job and be handed it straight back.
|
||||
pub fn drain(
|
||||
&mut self,
|
||||
clock: &dyn Fn() -> i64,
|
||||
budget: Budget,
|
||||
cancel: &AtomicBool,
|
||||
) -> Result<DrainReport, CatalogError> {
|
||||
let mut report = DrainReport::default();
|
||||
|
||||
loop {
|
||||
// Relaxed: the flag is a one-way latch set by another thread and
|
||||
// the only thing ordered against it is our own next claim. Missing
|
||||
// one turn of the loop costs a job, not correctness.
|
||||
if cancel.load(Ordering::Relaxed) {
|
||||
report.stopped = Stopped::Cancelled;
|
||||
break;
|
||||
}
|
||||
if budget.max_jobs.is_some_and(|max| report.ran() >= max) {
|
||||
report.stopped = Stopped::Budget;
|
||||
break;
|
||||
}
|
||||
|
||||
let now = clock();
|
||||
if budget.deadline.is_some_and(|end| now >= end) {
|
||||
report.stopped = Stopped::Deadline;
|
||||
break;
|
||||
}
|
||||
|
||||
let Some(ran) = self.run_one(now)? else {
|
||||
report.stopped = Stopped::Drained;
|
||||
break;
|
||||
};
|
||||
|
||||
match ran.outcome {
|
||||
Outcome::Done => report.completed += 1,
|
||||
Outcome::Retry(_) => report.retried += 1,
|
||||
Outcome::Abandon(_) => report.abandoned += 1,
|
||||
Outcome::Interrupted => {
|
||||
report.interrupted += 1;
|
||||
report.stopped = Stopped::Cancelled;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(report)
|
||||
}
|
||||
|
||||
/// Drain with no budget and no cancellation, at a fixed instant.
|
||||
///
|
||||
/// Terminates because a job that fails is pushed past `now` by its backoff
|
||||
/// and stops being claimable at this instant.
|
||||
pub fn drain_all(&mut self, now: i64) -> Result<DrainReport, CatalogError> {
|
||||
static NEVER: AtomicBool = AtomicBool::new(false);
|
||||
self.drain(&|| now, Budget::UNLIMITED, &NEVER)
|
||||
}
|
||||
}
|
||||
|
||||
/// One job and what became of it.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Ran {
|
||||
pub job: Job,
|
||||
pub outcome: Outcome,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use super::*;
|
||||
use crate::jobs::{enqueue, JobState, Priority, MAX_ATTEMPTS};
|
||||
use crate::schema;
|
||||
|
||||
/// A handler built from a closure, so each test states its own behaviour.
|
||||
struct Fake<F> {
|
||||
kinds: Vec<JobKind>,
|
||||
act: F,
|
||||
}
|
||||
|
||||
impl<F: FnMut(&Job) -> Outcome> JobHandler for Fake<F> {
|
||||
fn kinds(&self) -> &[JobKind] {
|
||||
&self.kinds
|
||||
}
|
||||
fn run(&mut self, _conn: &Connection, job: &Job) -> Outcome {
|
||||
(self.act)(job)
|
||||
}
|
||||
}
|
||||
|
||||
fn handler<F: FnMut(&Job) -> Outcome>(kinds: &[JobKind], act: F) -> Fake<F> {
|
||||
Fake {
|
||||
kinds: kinds.to_vec(),
|
||||
act,
|
||||
}
|
||||
}
|
||||
|
||||
/// A handler that records which subjects it saw and always succeeds.
|
||||
///
|
||||
/// Takes its kinds by value and borrows nothing, so the returned handler is
|
||||
/// `Send + 'static` and can be moved into a worker thread — which the
|
||||
/// contention test needs.
|
||||
fn recording(kinds: Vec<JobKind>, seen: Arc<Mutex<Vec<i64>>>) -> impl JobHandler + Send {
|
||||
Fake {
|
||||
kinds,
|
||||
act: move |job: &Job| {
|
||||
seen.lock().unwrap().push(job.subject_id.unwrap_or(-1));
|
||||
Outcome::Done
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn db() -> Connection {
|
||||
let c = Connection::open_in_memory().unwrap();
|
||||
schema::configure(&c).unwrap();
|
||||
schema::migrate(&c).unwrap();
|
||||
c
|
||||
}
|
||||
|
||||
/// An image row, so a job has a subject that exists.
|
||||
fn image(c: &Connection, id: i64) {
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'local', '/lib')
|
||||
ON CONFLICT DO NOTHING",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (?1, 1, ?2, 0)",
|
||||
rusqlite::params![id, format!("/lib/{id}.CR3")],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
fn queued(c: &Connection, kind: JobKind, subject: i64) {
|
||||
image(c, subject);
|
||||
enqueue(c, kind, Some(subject), Priority::Background, None).unwrap();
|
||||
}
|
||||
|
||||
fn rows(c: &Connection) -> i64 {
|
||||
c.query_row("SELECT count(*) FROM jobs", [], |r| r.get(0))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// A catalog on disk, so more than one connection can open it.
|
||||
fn temp_catalog(name: &str) -> PathBuf {
|
||||
let dir = std::env::temp_dir().join(format!(
|
||||
"dr-runner-{name}-{}-{:?}",
|
||||
std::process::id(),
|
||||
std::thread::current().id()
|
||||
));
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
dir.join("catalog.db")
|
||||
}
|
||||
|
||||
fn open(path: &Path) -> Connection {
|
||||
let c = Connection::open(path).unwrap();
|
||||
schema::configure(&c).unwrap();
|
||||
schema::migrate(&c).unwrap();
|
||||
// Several connections write to this file at once in the contention
|
||||
// tests. Without a busy handler the loser of a race gets an error
|
||||
// instead of a turn.
|
||||
c.busy_timeout(std::time::Duration::from_secs(10)).unwrap();
|
||||
c
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_completed_job_leaves_the_queue() {
|
||||
// The whole finding in one assertion: before this module, the row
|
||||
// stayed forever because nothing ever claimed it.
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
let report = Runner::new(&c)
|
||||
.with(recording(vec![JobKind::Thumbnail], seen.clone()))
|
||||
.drain_all(0)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(report.completed, 1);
|
||||
assert_eq!(report.stopped, Stopped::Drained);
|
||||
assert_eq!(*seen.lock().unwrap(), vec![1]);
|
||||
assert_eq!(rows(&c), 0, "a completed job leaves no row behind");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_failed_job_backs_off_and_is_claimed_again_later() {
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
|
||||
// A `Cell` rather than a captured `bool`, so the closure's mutability
|
||||
// is its own business and the test reads the same either way.
|
||||
let failed_once = std::cell::Cell::new(false);
|
||||
let mut runner = Runner::new(&c).with(handler(&[JobKind::Thumbnail], |_| {
|
||||
if failed_once.replace(true) {
|
||||
Outcome::Done
|
||||
} else {
|
||||
Outcome::Retry("decoder said no".into())
|
||||
}
|
||||
}));
|
||||
|
||||
let first = runner.drain_all(100).unwrap();
|
||||
assert_eq!(first.retried, 1);
|
||||
assert_eq!(rows(&c), 1, "a retryable failure keeps its row");
|
||||
|
||||
// Still inside the backoff window: nothing claimable, so the drain
|
||||
// reports itself drained rather than spinning on the same job.
|
||||
assert_eq!(runner.drain_all(100).unwrap().ran(), 0);
|
||||
|
||||
let later = runner.drain_all(100 + jobs::backoff_seconds(1)).unwrap();
|
||||
assert_eq!(later.completed, 1);
|
||||
assert_eq!(rows(&c), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_job_that_keeps_failing_is_given_up_on() {
|
||||
// FR-RAW-4: one corrupt file must not stall the queue behind endless
|
||||
// retries. Driven through the runner rather than by hand, because the
|
||||
// runner is what a corrupt file will actually meet.
|
||||
let c = db();
|
||||
queued(&c, JobKind::ExtractMetadata, 1);
|
||||
|
||||
let mut runner = Runner::new(&c).with(handler(&[JobKind::ExtractMetadata], |_| {
|
||||
Outcome::Retry("corrupt file".into())
|
||||
}));
|
||||
|
||||
let mut now = 0;
|
||||
for _ in 0..MAX_ATTEMPTS {
|
||||
assert_eq!(runner.drain_all(now).unwrap().retried, 1);
|
||||
now += jobs::backoff_seconds(MAX_ATTEMPTS);
|
||||
}
|
||||
|
||||
assert_eq!(runner.drain_all(now + 100_000).unwrap().ran(), 0);
|
||||
let state: i64 = c
|
||||
.query_row("SELECT state FROM jobs", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Failed as i64);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_abandoned_job_is_not_retried_at_all() {
|
||||
// The difference that matters on battery: five failures spread over
|
||||
// five minutes to learn what the first one already said.
|
||||
let c = db();
|
||||
queued(&c, JobKind::FetchOriginal, 1);
|
||||
|
||||
let mut runner = Runner::new(&c).with(handler(&[JobKind::FetchOriginal], |_| {
|
||||
Outcome::Abandon("no connector on this device".into())
|
||||
}));
|
||||
|
||||
assert_eq!(runner.drain_all(0).unwrap().abandoned, 1);
|
||||
// One attempt, not MAX_ATTEMPTS, and never claimable again.
|
||||
assert_eq!(runner.drain_all(1_000_000).unwrap().ran(), 0);
|
||||
let (state, attempts): (i64, i64) = c
|
||||
.query_row("SELECT state, attempts FROM jobs", [], |r| {
|
||||
Ok((r.get(0)?, r.get(1)?))
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Failed as i64);
|
||||
assert_eq!(attempts, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_interrupted_job_costs_no_attempt_and_ends_the_drain() {
|
||||
// `onStopped()` says nothing about the file. Charging it an attempt
|
||||
// would let five backgroundings mark good work as failed.
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
queued(&c, JobKind::Thumbnail, 2);
|
||||
|
||||
let report = Runner::new(&c)
|
||||
.with(handler(&[JobKind::Thumbnail], |_| Outcome::Interrupted))
|
||||
.drain_all(0)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(report.interrupted, 1);
|
||||
assert_eq!(
|
||||
report.stopped,
|
||||
Stopped::Cancelled,
|
||||
"an interrupted job must end the drain, or releasing it hands it \
|
||||
straight back and the loop never ends"
|
||||
);
|
||||
let (state, attempts): (i64, i64) = c
|
||||
.query_row(
|
||||
"SELECT state, attempts FROM jobs WHERE subject_id = 1",
|
||||
[],
|
||||
|r| Ok((r.get(0)?, r.get(1)?)),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Pending as i64);
|
||||
assert_eq!(attempts, 0, "the claim's speculative attempt is given back");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_kinds_a_handler_covers_are_claimed() {
|
||||
// A device with no connector must leave `FetchOriginal` where it is.
|
||||
// Claiming it to fail it would cost five attempts and five backoffs
|
||||
// per photograph, on battery, to reach a conclusion known in advance.
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
queued(&c, JobKind::FetchOriginal, 2);
|
||||
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
let report = Runner::new(&c)
|
||||
.with(recording(vec![JobKind::Thumbnail], seen.clone()))
|
||||
.drain_all(0)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(report.completed, 1);
|
||||
assert_eq!(*seen.lock().unwrap(), vec![1]);
|
||||
|
||||
let (state, attempts): (i64, i64) = c
|
||||
.query_row(
|
||||
"SELECT state, attempts FROM jobs WHERE subject_id = 2",
|
||||
[],
|
||||
|r| Ok((r.get(0)?, r.get(1)?)),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Pending as i64);
|
||||
assert_eq!(attempts, 0, "an unhandled job is untouched, not failed");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_runner_with_no_handlers_claims_nothing() {
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
|
||||
assert_eq!(Runner::new(&c).drain_all(0).unwrap().ran(), 0);
|
||||
assert_eq!(rows(&c), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_drain_stops_at_its_job_budget() {
|
||||
let c = db();
|
||||
for id in 1..=5 {
|
||||
queued(&c, JobKind::Thumbnail, id);
|
||||
}
|
||||
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
let mut runner = Runner::new(&c).with(recording(vec![JobKind::Thumbnail], seen.clone()));
|
||||
let report = runner
|
||||
.drain(&|| 0, Budget::jobs(2), &AtomicBool::new(false))
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(report.completed, 2);
|
||||
assert_eq!(report.stopped, Stopped::Budget);
|
||||
assert!(report.stopped.more_to_do());
|
||||
assert_eq!(rows(&c), 3, "the rest is still queued for the next slot");
|
||||
assert_eq!(seen.lock().unwrap().len(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_drain_stops_at_its_deadline() {
|
||||
// What a `WorkManager` slot does: a fixed window, and whatever did not
|
||||
// fit stays queued for the next one.
|
||||
let c = db();
|
||||
for id in 1..=5 {
|
||||
queued(&c, JobKind::Thumbnail, id);
|
||||
}
|
||||
|
||||
// A clock that advances a second per reading, so the deadline arrives
|
||||
// without the test sleeping.
|
||||
let tick = std::cell::Cell::new(0i64);
|
||||
let clock = || {
|
||||
let t = tick.get();
|
||||
tick.set(t + 1);
|
||||
t
|
||||
};
|
||||
|
||||
let report = Runner::new(&c)
|
||||
.with(handler(&[JobKind::Thumbnail], |_| Outcome::Done))
|
||||
.drain(&clock, Budget::until(3), &AtomicBool::new(false))
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(report.stopped, Stopped::Deadline);
|
||||
assert_eq!(report.completed, 3, "one job per second up to the deadline");
|
||||
assert_eq!(rows(&c), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_cancelled_drain_stops_between_jobs() {
|
||||
let c = db();
|
||||
for id in 1..=5 {
|
||||
queued(&c, JobKind::Thumbnail, id);
|
||||
}
|
||||
|
||||
let cancel = AtomicBool::new(false);
|
||||
// Cancelled from inside the handler, standing in for the host thread
|
||||
// setting the flag while a job is in flight: the job in hand finishes,
|
||||
// and nothing further is claimed.
|
||||
let report = Runner::new(&c)
|
||||
.with(handler(&[JobKind::Thumbnail], |_| {
|
||||
cancel.store(true, Ordering::Relaxed);
|
||||
Outcome::Done
|
||||
}))
|
||||
.drain(&|| 0, Budget::UNLIMITED, &cancel)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(report.completed, 1);
|
||||
assert_eq!(report.stopped, Stopped::Cancelled);
|
||||
assert_eq!(rows(&c), 4, "the work is kept, not lost");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_job_interrupted_by_process_death_is_reclaimed_and_run_once() {
|
||||
// FR-PLAT-AND-3. The kill happens between the claim and the outcome,
|
||||
// which is the window a durable queue exists to survive: no `complete`,
|
||||
// no `fail`, just a row marked `Running` with nobody holding it.
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
|
||||
// The dead process. It claimed the job and never came back.
|
||||
let claimed = jobs::claim_next(&c, 0).unwrap().expect("claimable");
|
||||
assert_eq!(claimed.subject_id, Some(1));
|
||||
|
||||
// A fresh runner, before it starts, finds the queue empty — the row is
|
||||
// `Running` and no claim will touch it.
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
let mut runner = Runner::new(&c).with(recording(vec![JobKind::Thumbnail], seen.clone()));
|
||||
assert_eq!(
|
||||
runner.drain_all(0).unwrap().ran(),
|
||||
0,
|
||||
"an orphan is invisible until it is recovered — which is exactly \
|
||||
why recovery has to happen at startup"
|
||||
);
|
||||
|
||||
let recovered = runner.recover().unwrap();
|
||||
assert_eq!(recovered.reclaimed, 1);
|
||||
|
||||
let report = runner.drain_all(0).unwrap();
|
||||
assert_eq!(report.completed, 1);
|
||||
assert_eq!(
|
||||
*seen.lock().unwrap(),
|
||||
vec![1],
|
||||
"resumed, not repeated and not lost"
|
||||
);
|
||||
assert_eq!(rows(&c), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_crash_still_costs_an_attempt() {
|
||||
// Deliberate: a job that takes the process down with it every time is
|
||||
// indistinguishable from one that fails, and the attempt counter is
|
||||
// the only evidence we keep across a death. Without this a poison-pill
|
||||
// job would be reclaimed and re-run forever.
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
|
||||
for _ in 0..MAX_ATTEMPTS {
|
||||
jobs::claim_next(&c, 0).unwrap().expect("claimable");
|
||||
recover(&c).unwrap();
|
||||
}
|
||||
|
||||
let job = jobs::claim_next(&c, 0).unwrap().unwrap();
|
||||
assert!(job.attempts > MAX_ATTEMPTS);
|
||||
jobs::fail(&c, &job, 0, "died again").unwrap();
|
||||
let state: i64 = c
|
||||
.query_row("SELECT state FROM jobs", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(state, JobState::Failed as i64);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_drops_jobs_whose_photograph_is_gone() {
|
||||
// A culled library leaves thumbnail jobs for images deleted months
|
||||
// ago. Every one would be claimed, run and failed.
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
queued(&c, JobKind::Thumbnail, 2);
|
||||
c.execute("DELETE FROM images WHERE id = 2", []).unwrap();
|
||||
|
||||
let recovered = recover(&c).unwrap();
|
||||
assert_eq!(recovered.reaped, 1);
|
||||
assert!(recovered.did_anything());
|
||||
|
||||
let left: i64 = c
|
||||
.query_row("SELECT subject_id FROM jobs", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(left, 1, "only the job whose subject survives is kept");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_quiet_startup_recovers_nothing() {
|
||||
let c = db();
|
||||
queued(&c, JobKind::Thumbnail, 1);
|
||||
assert_eq!(recover(&c).unwrap(), Recovered::default());
|
||||
assert!(!recover(&c).unwrap().did_anything());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn two_runners_on_one_catalog_never_take_the_same_job() {
|
||||
// Sequential rather than threaded, so the property is asserted without
|
||||
// depending on the scheduler: whatever the second connection claims,
|
||||
// it is not what the first one is holding.
|
||||
let path = temp_catalog("contention-pair");
|
||||
let a = open(&path);
|
||||
let b = open(&path);
|
||||
|
||||
for id in 1..=2 {
|
||||
queued(&a, JobKind::Thumbnail, id);
|
||||
}
|
||||
|
||||
let first = jobs::claim_next(&a, 0).unwrap().expect("one for A");
|
||||
let second = jobs::claim_next(&b, 0).unwrap().expect("one for B");
|
||||
|
||||
assert_ne!(first.id, second.id);
|
||||
assert!(
|
||||
jobs::claim_next(&a, 0).unwrap().is_none(),
|
||||
"a claimed job is invisible to every connection, not just its own"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn concurrent_runners_share_the_queue_without_repeating_work() {
|
||||
// The claim is one atomic statement precisely so this holds: four
|
||||
// threads, four connections, and every job run exactly once.
|
||||
const THREADS: usize = 4;
|
||||
const JOBS: i64 = 24;
|
||||
|
||||
let path = temp_catalog("contention-threads");
|
||||
let seeder = open(&path);
|
||||
for id in 1..=JOBS {
|
||||
queued(&seeder, JobKind::Thumbnail, id);
|
||||
}
|
||||
drop(seeder);
|
||||
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
let mut threads = Vec::new();
|
||||
for _ in 0..THREADS {
|
||||
let path = path.clone();
|
||||
let seen = seen.clone();
|
||||
threads.push(std::thread::spawn(move || {
|
||||
let conn = open(&path);
|
||||
// Bound to a local rather than left as the block's tail: the
|
||||
// `Runner` borrows `conn`, and a tail expression's temporaries
|
||||
// are dropped *after* the block's locals, so the borrow would
|
||||
// outlive what it borrows.
|
||||
let completed = Runner::new(&conn)
|
||||
.with(recording(vec![JobKind::Thumbnail], seen))
|
||||
.drain_all(0)
|
||||
.unwrap()
|
||||
.completed;
|
||||
completed
|
||||
}));
|
||||
}
|
||||
|
||||
let completed: usize = threads.into_iter().map(|t| t.join().unwrap()).sum();
|
||||
assert_eq!(completed, JOBS as usize);
|
||||
|
||||
let mut ran = seen.lock().unwrap().clone();
|
||||
ran.sort_unstable();
|
||||
assert_eq!(
|
||||
ran,
|
||||
(1..=JOBS).collect::<Vec<_>>(),
|
||||
"every job exactly once — no duplicate claim, nothing dropped"
|
||||
);
|
||||
|
||||
let leftover = open(&path);
|
||||
assert_eq!(rows(&leftover), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn priority_survives_the_runner() {
|
||||
// NFR-ARCH-2: visible work strictly preempts bulk work, and it has to
|
||||
// still be true when the queue is drained through a handler rather
|
||||
// than by hand.
|
||||
let c = db();
|
||||
image(&c, 1);
|
||||
image(&c, 2);
|
||||
enqueue(&c, JobKind::Thumbnail, Some(1), Priority::Background, None).unwrap();
|
||||
enqueue(&c, JobKind::Thumbnail, Some(2), Priority::Interactive, None).unwrap();
|
||||
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
Runner::new(&c)
|
||||
.with(recording(vec![JobKind::Thumbnail], seen.clone()))
|
||||
.drain_all(0)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(*seen.lock().unwrap(), vec![2, 1]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_handler_sees_the_payload_and_the_attempt_count() {
|
||||
// Both are how a handler decides what to do: the payload is the only
|
||||
// thing that survives from the enqueue site, and the attempt count is
|
||||
// how it can tell a first try from a last one.
|
||||
let c = db();
|
||||
image(&c, 1);
|
||||
enqueue(
|
||||
&c,
|
||||
JobKind::ScanFolder,
|
||||
Some(1),
|
||||
Priority::Background,
|
||||
Some("/lib/2024"),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let payload = Arc::new(Mutex::new(None));
|
||||
let recorded = payload.clone();
|
||||
Runner::new(&c)
|
||||
.with(handler(&[JobKind::ScanFolder], move |job| {
|
||||
*recorded.lock().unwrap() = Some((job.payload.clone(), job.attempts));
|
||||
Outcome::Done
|
||||
}))
|
||||
.drain_all(0)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
*payload.lock().unwrap(),
|
||||
Some((Some("/lib/2024".to_string()), 1))
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,237 +0,0 @@
|
||||
//! TRACES: FR-CAT-1 | FR-CAT-9 | NFR-P1
|
||||
//! Incremental scan: the local analogue of ETag pruning.
|
||||
//!
|
||||
//! Nextcloud propagates ETags up the tree, so one request proves a whole
|
||||
//! library unchanged (ARCH §8.4). A filesystem offers no such guarantee — a
|
||||
//! directory's mtime moves when its *direct* entries change and not when a
|
||||
//! grandchild does, so there is no cheap "did anything below here change"
|
||||
//! probe.
|
||||
//!
|
||||
//! Local scan therefore prunes at each level rather than at the root: one
|
||||
//! metadata probe per directory when nothing changed, instead of one per file.
|
||||
//! A 50k-image library in ~2k folders costs 2k probes, which is the difference
|
||||
//! between meeting and missing NFR-P1 on SAF.
|
||||
//!
|
||||
//! This module holds the decision logic and the deletion-sweep rules; walking
|
||||
//! an actual directory belongs to the platform layer, which supplies
|
||||
//! [`DirState`] and [`DirEntry`]. [`crate::walk`] is what puts the two
|
||||
//! together.
|
||||
|
||||
pub use dr_types::{DirEntry, DirState};
|
||||
|
||||
use dr_types::FormatFilter;
|
||||
|
||||
/// What the scanner should do with a directory, before listing it.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum DirAction {
|
||||
/// Contents unchanged. Skip the listing, but still recurse into known
|
||||
/// children — without upward propagation, a deep change is invisible from
|
||||
/// here.
|
||||
RecurseOnly,
|
||||
/// List and reconcile, then recurse.
|
||||
ListAndRecurse,
|
||||
}
|
||||
|
||||
/// Decide whether a directory needs listing.
|
||||
pub fn classify_dir(stored: Option<DirState>, current: DirState) -> DirAction {
|
||||
match stored {
|
||||
Some(s) if s == current => DirAction::RecurseOnly,
|
||||
_ => DirAction::ListAndRecurse,
|
||||
}
|
||||
}
|
||||
|
||||
/// What reconciling one listed entry against the catalog implies.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum EntryAction {
|
||||
/// Not catalogued. Insert at `metadata_state = 1` and queue EXIF.
|
||||
Insert,
|
||||
/// Catalogued and unchanged. The common case, and it must cost nothing.
|
||||
Unchanged,
|
||||
/// Size or mtime moved: re-read metadata, rebuild the thumbnail, and drop
|
||||
/// the content hash, which is no longer valid.
|
||||
Changed,
|
||||
/// Recognised but not a format the user asked to scan for.
|
||||
Ignored,
|
||||
}
|
||||
|
||||
/// What the catalog already holds for a source.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct KnownFile {
|
||||
pub size: u64,
|
||||
pub mtime: i64,
|
||||
}
|
||||
|
||||
/// Classify one listed file.
|
||||
pub fn classify_entry(
|
||||
entry: &DirEntry,
|
||||
known: Option<KnownFile>,
|
||||
formats: &FormatFilter,
|
||||
) -> EntryAction {
|
||||
if !formats.allows_name(&entry.name) {
|
||||
return EntryAction::Ignored;
|
||||
}
|
||||
match known {
|
||||
None => EntryAction::Insert,
|
||||
Some(k) if k.size == entry.size && k.mtime == entry.mtime => EntryAction::Unchanged,
|
||||
Some(_) => EntryAction::Changed,
|
||||
}
|
||||
}
|
||||
|
||||
/// Outcome of a scan, which decides whether pruning may run.
|
||||
///
|
||||
/// `Cancelled` is the default because a scan that has not run has proven
|
||||
/// nothing absent, and every default in this area must fail towards keeping
|
||||
/// photographs.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
||||
pub enum ScanOutcome {
|
||||
/// Every reachable folder was visited.
|
||||
Complete,
|
||||
/// The user cancelled. Partial state is valid — jobs are resumable — but
|
||||
/// unvisited folders must not be read as deleted.
|
||||
#[default]
|
||||
Cancelled,
|
||||
/// The root itself could not be opened: drive unplugged, SAF grant
|
||||
/// revoked, share unmounted.
|
||||
RootUnreachable,
|
||||
/// Some subtree failed while the root was fine.
|
||||
PartialFailure,
|
||||
}
|
||||
|
||||
impl ScanOutcome {
|
||||
/// Whether the deletion sweep may run.
|
||||
///
|
||||
/// **The most dangerous decision in the catalog.** The sweep deletes every
|
||||
/// folder not reached by this scan's generation. After an incomplete scan
|
||||
/// that is most of the library, so it runs only on `Complete`.
|
||||
///
|
||||
/// FR-CAT-9 draws exactly this line: a source *proven absent* may leave
|
||||
/// the catalog; a source merely *unreachable* is marked offline and kept,
|
||||
/// with its ratings and edits intact.
|
||||
pub fn may_prune(self) -> bool {
|
||||
matches!(self, ScanOutcome::Complete)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use dr_types::Format;
|
||||
|
||||
const A: DirState = DirState {
|
||||
mtime: 100,
|
||||
entry_count: 5,
|
||||
};
|
||||
|
||||
#[test]
|
||||
fn unchanged_directory_is_not_listed() {
|
||||
assert_eq!(classify_dir(Some(A), A), DirAction::RecurseOnly);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_never_seen_directory_is_listed() {
|
||||
assert_eq!(classify_dir(None, A), DirAction::ListAndRecurse);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn changed_mtime_forces_a_listing() {
|
||||
let now = DirState { mtime: 101, ..A };
|
||||
assert_eq!(classify_dir(Some(A), now), DirAction::ListAndRecurse);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entry_count_catches_what_mtime_misses() {
|
||||
// A file added within the same timestamp tick: mtime is unchanged, so
|
||||
// mtime alone would skip this directory and lose the new image.
|
||||
let now = DirState {
|
||||
mtime: 100,
|
||||
entry_count: 6,
|
||||
};
|
||||
assert_eq!(classify_dir(Some(A), now), DirAction::ListAndRecurse);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unchanged_file_costs_nothing() {
|
||||
let e = DirEntry {
|
||||
name: "IMG_0001.CR3".into(),
|
||||
is_dir: false,
|
||||
size: 30_000_000,
|
||||
mtime: 500,
|
||||
};
|
||||
let known = KnownFile {
|
||||
size: 30_000_000,
|
||||
mtime: 500,
|
||||
};
|
||||
assert_eq!(
|
||||
classify_entry(&e, Some(known), &FormatFilter::all()),
|
||||
EntryAction::Unchanged
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_resaved_file_is_reprocessed() {
|
||||
let e = DirEntry {
|
||||
name: "IMG_0001.CR3".into(),
|
||||
is_dir: false,
|
||||
size: 30_000_001,
|
||||
mtime: 900,
|
||||
};
|
||||
let known = KnownFile {
|
||||
size: 30_000_000,
|
||||
mtime: 500,
|
||||
};
|
||||
assert_eq!(
|
||||
classify_entry(&e, Some(known), &FormatFilter::all()),
|
||||
EntryAction::Changed
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_filter_excludes_unwanted_types() {
|
||||
let jpeg = DirEntry {
|
||||
name: "IMG_0001.JPG".into(),
|
||||
is_dir: false,
|
||||
size: 1,
|
||||
mtime: 1,
|
||||
};
|
||||
assert_eq!(
|
||||
classify_entry(&jpeg, None, &FormatFilter::raw_only()),
|
||||
EntryAction::Ignored
|
||||
);
|
||||
assert_eq!(
|
||||
classify_entry(&jpeg, None, &FormatFilter::all()),
|
||||
EntryAction::Insert
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_placeholder_is_catalogued_as_the_image_it_stands_for() {
|
||||
// 121,785 of these in a real synced folder (ARCH §9.0). Each must
|
||||
// enter the catalog as a CR2 marked offline, not be skipped as an
|
||||
// unknown ".nextcloud" type.
|
||||
let stub = DirEntry {
|
||||
name: "_MG_4130.CR2.nextcloud".into(),
|
||||
is_dir: false,
|
||||
size: 1,
|
||||
mtime: 1,
|
||||
};
|
||||
assert_eq!(
|
||||
classify_entry(&stub, None, &FormatFilter::from_formats([Format::Cr2])),
|
||||
EntryAction::Insert
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pruning_requires_a_complete_scan() {
|
||||
assert!(ScanOutcome::Complete.may_prune());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unreachable_root_never_prunes() {
|
||||
// The guard that stops an unplugged drive from deleting the library:
|
||||
// every folder would look unreached, so the sweep would take all of
|
||||
// them (FR-CAT-9).
|
||||
assert!(!ScanOutcome::RootUnreachable.may_prune());
|
||||
assert!(!ScanOutcome::Cancelled.may_prune());
|
||||
assert!(!ScanOutcome::PartialFailure.may_prune());
|
||||
}
|
||||
}
|
||||
@@ -1,361 +0,0 @@
|
||||
//! TRACES: FR-CAT-7 | FR-NC-9 | NFR-R1
|
||||
//! Preparing the catalog file for upload, and taking in a remote one.
|
||||
//!
|
||||
//! # The hazard this module exists to handle
|
||||
//!
|
||||
//! A WAL-mode SQLite database is not one file. Committed transactions can live
|
||||
//! in `catalog.sqlite-wal` with the main file lagging behind, so copying
|
||||
//! `catalog.sqlite` alone uploads a **torn snapshot**: internally consistent as
|
||||
//! of some older point, missing everything since. Worse, a naive copy taken
|
||||
//! while a writer is mid-transaction can be structurally corrupt.
|
||||
//!
|
||||
//! So an upload never copies the live file. It runs a TRUNCATE checkpoint to
|
||||
//! fold the WAL back into the main file, then uses SQLite's own backup API to
|
||||
//! take a consistent snapshot — which serialises correctly against concurrent
|
||||
//! writers rather than racing them.
|
||||
//!
|
||||
//! # What is actually synced
|
||||
//!
|
||||
//! Only the *user's judgements about their library* merge: collections, and the
|
||||
//! keyword vocabulary with its assignments (see [`crate::merge`]). The rest of
|
||||
//! the catalog is a *local index* of *local* storage — folder mtimes, cache
|
||||
//! paths, job rows — and copying another device's version of those in would be
|
||||
//! actively wrong. The remote file is read for those two and then discarded.
|
||||
//!
|
||||
//! This is why the catalog remains disposable in the ARCH §6.12 sense: nothing
|
||||
//! here makes the local database authoritative for anything a rebuild could
|
||||
//! not recover.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use rusqlite::Connection;
|
||||
|
||||
use crate::error::CatalogError;
|
||||
use crate::merge::{self, MergeReport};
|
||||
|
||||
/// Schema name the downloaded remote catalog is attached under.
|
||||
const REMOTE_SCHEMA: &str = "remote_cat";
|
||||
|
||||
/// Fold the WAL into the main database file.
|
||||
///
|
||||
/// TRUNCATE rather than PASSIVE: passive checkpointing gives up when a reader
|
||||
/// holds the WAL open, which would leave recent commits out of the snapshot
|
||||
/// without saying so.
|
||||
pub fn checkpoint(conn: &Connection) -> Result<(), CatalogError> {
|
||||
conn.pragma_update(None, "wal_checkpoint", "TRUNCATE")?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Write a consistent snapshot of the catalog to `dest`, ready to upload.
|
||||
///
|
||||
/// Uses the backup API rather than a filesystem copy so the snapshot is
|
||||
/// coherent even with writers active. Callers should still prefer a quiet
|
||||
/// moment — this competes with background jobs for the write lock.
|
||||
pub fn snapshot_for_upload(conn: &Connection, dest: &Path) -> Result<(), CatalogError> {
|
||||
let out = copy_to(conn, dest)?;
|
||||
strip_face_crops(&out)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Checkpoint, then copy the whole database to `dest`, and hand back the
|
||||
/// connection to the copy.
|
||||
///
|
||||
/// Split out from [`snapshot_for_upload`] because [`crate::recovery`] wants
|
||||
/// exactly this and none of what follows it there: an NFR-R2 backup is the
|
||||
/// file the user may have to *live on*, so it keeps the face crops that an
|
||||
/// upload strips. Sharing the copy rather than reimplementing it is what keeps
|
||||
/// the WAL discipline in one place — a backup taken with `fs::copy` would be
|
||||
/// the torn snapshot this module's header exists to warn about.
|
||||
pub(crate) fn copy_to(conn: &Connection, dest: &Path) -> Result<Connection, CatalogError> {
|
||||
checkpoint(conn)?;
|
||||
|
||||
let mut out = Connection::open(dest)?;
|
||||
let backup = rusqlite::backup::Backup::new(conn, &mut out)?;
|
||||
// SQLite's own "copy everything" sentinel is -1, but rusqlite asserts a
|
||||
// positive page count, so ask for more pages than a catalog will ever
|
||||
// have. The effect is the same: one step, no interleaved writers, no
|
||||
// progress callback. A 50k-image catalog is tens of megabytes.
|
||||
backup.run_to_completion(i32::MAX, std::time::Duration::ZERO, None)?;
|
||||
drop(backup);
|
||||
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// Drop the stored face crops from a snapshot before it is uploaded.
|
||||
///
|
||||
/// The snapshot is the *whole catalog*, uploaded on every sync and downloaded
|
||||
/// by every device. Face crops are a few KB each and a fully indexed library
|
||||
/// holds tens of thousands of them, so leaving them in would put tens of MB on
|
||||
/// every round trip — the exact cost `face_shard`'s 25 MB cap exists to bound,
|
||||
/// and the reason the bulk per-face data lives in shards in the first place.
|
||||
///
|
||||
/// Crops are not lost by this: they travel in the face shards
|
||||
/// ([`crate::face_shard::export_to_shards`]), which are written once and
|
||||
/// downloaded once. Nothing reads a crop out of a merged remote catalog —
|
||||
/// [`merge_all`] touches collections and keywords only — so removing them here
|
||||
/// costs a receiving device nothing it would otherwise have had.
|
||||
///
|
||||
/// `VACUUM` afterwards because SQLite does not return freed pages to the file
|
||||
/// on its own, and an upload sized by the file rather than by its contents
|
||||
/// would keep paying for bytes that are no longer there.
|
||||
fn strip_face_crops(snapshot: &Connection) -> Result<(), CatalogError> {
|
||||
// A catalog older than the crop column is a legitimate input here — a
|
||||
// snapshot taken mid-migration, or a test fixture built from an earlier
|
||||
// schema — so an absent column is nothing to fail over.
|
||||
let has_crop = snapshot
|
||||
.prepare("SELECT crop FROM faces LIMIT 1")
|
||||
.map(|_| true)
|
||||
.unwrap_or(false);
|
||||
if !has_crop {
|
||||
return Ok(());
|
||||
}
|
||||
snapshot.execute("UPDATE faces SET crop = NULL WHERE crop IS NOT NULL", [])?;
|
||||
snapshot.execute_batch("VACUUM")?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Whether a downloaded remote catalog is worth merging.
|
||||
///
|
||||
/// Cheap guard before attaching: a remote written by a newer build may contain
|
||||
/// tables and columns this one cannot read, and attempting the merge would
|
||||
/// fail mid-transaction rather than declining cleanly.
|
||||
pub fn remote_is_mergeable(remote: &Path) -> Result<bool, CatalogError> {
|
||||
let conn = Connection::open_with_flags(
|
||||
remote,
|
||||
rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY | rusqlite::OpenFlags::SQLITE_OPEN_NO_MUTEX,
|
||||
)?;
|
||||
let v: i64 = conn.query_row("PRAGMA user_version", [], |r| r.get(0))?;
|
||||
Ok(v <= crate::schema::SCHEMA_VERSION)
|
||||
}
|
||||
|
||||
/// Attach a downloaded remote catalog, merge its collections, detach.
|
||||
///
|
||||
/// The remote file is opened **read-only** — this device never writes to
|
||||
/// another device's catalog, it only reads collections out of it.
|
||||
pub fn merge_remote(conn: &Connection, remote: &Path) -> Result<MergeReport, CatalogError> {
|
||||
if !remote_is_mergeable(remote)? {
|
||||
return Err(CatalogError::SchemaTooNew {
|
||||
found: -1,
|
||||
supported: crate::schema::SCHEMA_VERSION,
|
||||
});
|
||||
}
|
||||
|
||||
// Path binds as a parameter; ATTACH accepts one, so a path containing a
|
||||
// quote cannot break out into SQL.
|
||||
conn.execute(
|
||||
&format!("ATTACH DATABASE ?1 AS {REMOTE_SCHEMA}"),
|
||||
[remote.to_string_lossy().as_ref()],
|
||||
)?;
|
||||
|
||||
let result = merge::merge_all(conn);
|
||||
|
||||
// Detach even if the merge failed, or the next attempt errors with
|
||||
// "database remote_cat is already in use".
|
||||
let detach = conn.execute(&format!("DETACH DATABASE {REMOTE_SCHEMA}"), []);
|
||||
if let Err(e) = detach {
|
||||
log::warn!("failed to detach remote catalog: {e}");
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
/// Where the catalog snapshot and the downloaded remote live.
|
||||
///
|
||||
/// Both are transient working files, not the catalog itself, so they belong in
|
||||
/// the cache directory rather than beside the live database.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct SyncPaths {
|
||||
pub upload_snapshot: PathBuf,
|
||||
pub downloaded_remote: PathBuf,
|
||||
}
|
||||
|
||||
impl SyncPaths {
|
||||
pub fn in_dir(cache_dir: &Path) -> Self {
|
||||
SyncPaths {
|
||||
upload_snapshot: cache_dir.join("catalog-upload.sqlite"),
|
||||
downloaded_remote: cache_dir.join("catalog-remote.sqlite"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::schema;
|
||||
|
||||
fn seeded(path: &Path) -> Connection {
|
||||
let c = Connection::open(path).unwrap();
|
||||
schema::configure(&c).unwrap();
|
||||
schema::migrate(&c).unwrap();
|
||||
c
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn snapshot_captures_committed_data() {
|
||||
let dir = tempdir();
|
||||
let live = dir.join("catalog.sqlite");
|
||||
let snap = dir.join("snap.sqlite");
|
||||
|
||||
let c = seeded(&live);
|
||||
c.execute(
|
||||
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
|
||||
VALUES ('u1', 'Iceland', 0, 0, 1, 1)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
snapshot_for_upload(&c, &snap).unwrap();
|
||||
|
||||
// The snapshot must hold the row even though it was written after the
|
||||
// database was created — the torn-file failure this guards against.
|
||||
let s = Connection::open(&snap).unwrap();
|
||||
let name: String = s
|
||||
.query_row("SELECT name FROM collections", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(name, "Iceland");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_remote_from_a_newer_build_is_declined_not_attempted() {
|
||||
let dir = tempdir();
|
||||
let remote = dir.join("remote.sqlite");
|
||||
let r = seeded(&remote);
|
||||
r.pragma_update(None, "user_version", schema::SCHEMA_VERSION + 1)
|
||||
.unwrap();
|
||||
drop(r);
|
||||
|
||||
assert!(!remote_is_mergeable(&remote).unwrap());
|
||||
|
||||
let local = seeded(&dir.join("local.sqlite"));
|
||||
assert!(matches!(
|
||||
merge_remote(&local, &remote),
|
||||
Err(CatalogError::SchemaTooNew { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_remote_round_trips_a_collection() {
|
||||
let dir = tempdir();
|
||||
let remote_path = dir.join("remote.sqlite");
|
||||
{
|
||||
let r = seeded(&remote_path);
|
||||
r.execute(
|
||||
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
|
||||
VALUES ('u-remote', 'Portugal', 0, 0, 1, 1)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
checkpoint(&r).unwrap();
|
||||
}
|
||||
|
||||
let local = seeded(&dir.join("local.sqlite"));
|
||||
local
|
||||
.execute(
|
||||
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
|
||||
VALUES ('u-local', 'Iceland', 0, 0, 1, 1)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let report = merge_remote(&local, &remote_path).unwrap();
|
||||
assert_eq!(report.inserted, 1);
|
||||
|
||||
let n: i64 = local
|
||||
.query_row("SELECT count(*) FROM collections", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(n, 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_remote_can_be_merged_twice_without_attach_conflict() {
|
||||
// Detach must happen even on the failure path, or the second attempt
|
||||
// errors with "database remote_cat is already in use".
|
||||
let dir = tempdir();
|
||||
let remote_path = dir.join("remote.sqlite");
|
||||
{
|
||||
let r = seeded(&remote_path);
|
||||
r.execute(
|
||||
"INSERT INTO collections(uuid, name, kind, created, revision, modified)
|
||||
VALUES ('u-remote', 'Portugal', 0, 0, 1, 1)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
checkpoint(&r).unwrap();
|
||||
}
|
||||
let local = seeded(&dir.join("local.sqlite"));
|
||||
|
||||
merge_remote(&local, &remote_path).unwrap();
|
||||
let second = merge_remote(&local, &remote_path).unwrap();
|
||||
assert!(!second.local_changed());
|
||||
}
|
||||
|
||||
/// A scratch directory that cleans up with the test.
|
||||
fn tempdir() -> PathBuf {
|
||||
let base = std::env::temp_dir().join(format!(
|
||||
"dr-catalog-test-{}-{:?}",
|
||||
std::process::id(),
|
||||
std::thread::current().id()
|
||||
));
|
||||
let _ = std::fs::remove_dir_all(&base);
|
||||
std::fs::create_dir_all(&base).unwrap();
|
||||
base
|
||||
}
|
||||
|
||||
/// The whole reason crops live in the shards: a snapshot is uploaded whole,
|
||||
/// on every sync, to every device.
|
||||
#[test]
|
||||
fn the_snapshot_carries_no_face_crops() {
|
||||
let dir = tempdir();
|
||||
let live = dir.join("catalog.sqlite");
|
||||
let snap = dir.join("snap.sqlite");
|
||||
|
||||
let c = seeded(&live);
|
||||
c.execute(
|
||||
"INSERT OR IGNORE INTO roots(id, kind, label) VALUES (1, 'local', 'lib')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO images(id, root_id, source_ref, added_at) VALUES (1, 1, 'a.CR3', 0)",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO faces
|
||||
(image_id, x, y, w, h, landmarks, detector_confidence, embedding,
|
||||
crop_px, model_id, detected_at, crop)
|
||||
VALUES (1, 0.1, 0.1, 0.2, 0.2, X'00', 0.9, X'00', 180.0, 'm', 0, ?1)",
|
||||
[vec![7u8; 4096]],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
snapshot_for_upload(&c, &snap).unwrap();
|
||||
|
||||
let out = Connection::open(&snap).unwrap();
|
||||
let crops: i64 = out
|
||||
.query_row(
|
||||
"SELECT COUNT(*) FROM faces WHERE crop IS NOT NULL",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(crops, 0, "the snapshot still carries face crops");
|
||||
|
||||
// The face itself must still be there — only the pixels are dropped.
|
||||
let faces: i64 = out
|
||||
.query_row("SELECT COUNT(*) FROM faces", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(faces, 1);
|
||||
|
||||
// And the local catalog keeps its crop: this strips the copy, never
|
||||
// the original.
|
||||
let kept: i64 = c
|
||||
.query_row(
|
||||
"SELECT COUNT(*) FROM faces WHERE crop IS NOT NULL",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(kept, 1, "stripping the snapshot damaged the live catalog");
|
||||
}
|
||||
}
|
||||
@@ -1,636 +0,0 @@
|
||||
//! TRACES: FR-CAT-15 | NFR-R2
|
||||
//! Soft delete, restore, and the permanent delete that follows.
|
||||
//!
|
||||
//! # Why the trash is a folder and not a flag
|
||||
//!
|
||||
//! The catalog is a *rebuildable index* (ARCH §6.12): delete `catalog.sqlite`
|
||||
//! and it is reconstructed by rescanning sources. A trash implemented as a
|
||||
//! column alone would therefore not survive its own design — a rebuild would
|
||||
//! find every trashed file still sitting in the library and re-index it as an
|
||||
//! ordinary photograph, silently undoing every delete the user had made.
|
||||
//!
|
||||
//! So a soft delete **moves the file** into `.darkroom-trash/` under the library
|
||||
//! root, and the catalog merely records that this happened. The folder is the
|
||||
//! durable fact; the row is the convenience. Recovering by hand needs no
|
||||
//! DarkRoom at all, which is the property that matters when the thing being
|
||||
//! risked is a photograph.
|
||||
//!
|
||||
//! `dr_sync::scan::is_excluded` keeps the scanner out of that folder. Without
|
||||
//! it the next scan re-indexes the trash and the delete comes undone — the two
|
||||
//! halves are one mechanism and neither works alone.
|
||||
//!
|
||||
//! # The two steps
|
||||
//!
|
||||
//! **Soft** ([`trash`]) — `MOVE` to the trash folder, record `trashed_at` and
|
||||
//! the path it came from. Reversible by [`restore`], which is why the original
|
||||
//! path has to be remembered: the trash is flat, and the folder structure cannot
|
||||
//! be recovered from the trashed name.
|
||||
//!
|
||||
//! **Hard** ([`purge`]) — `DELETE` the file, then delete the row. Irreversible
|
||||
//! from DarkRoom's side, though the server's own trashbin may still hold it.
|
||||
//! Ordered file-first deliberately: see [`purge_order`].
|
||||
//!
|
||||
//! # What this module does not do
|
||||
//!
|
||||
//! It performs no I/O. Every function here records or reads catalog state, and
|
||||
//! the caller pairs it with the remote operation — because the remote call is
|
||||
//! async and the catalog is not, and because the *order* of the two is a
|
||||
//! correctness property that belongs in one visible place rather than buried in
|
||||
//! a transaction.
|
||||
|
||||
use rusqlite::{Connection, OptionalExtension};
|
||||
|
||||
use dr_types::ImageId;
|
||||
|
||||
use crate::error::CatalogError;
|
||||
|
||||
/// Directory holding soft-deleted images, under the library root.
|
||||
///
|
||||
/// The same constant `dr_sync::scan` excludes. Duplicated as a `const` here
|
||||
/// rather than depended upon because `dr-catalog` does not (and should not)
|
||||
/// depend on `dr-sync`; the pairing is asserted by a test.
|
||||
pub const TRASH_DIR: &str = ".darkroom-trash";
|
||||
|
||||
/// One trashed image, as the trash view lists it.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct TrashedImage {
|
||||
pub image_id: ImageId,
|
||||
/// Where the file is *now* — inside the trash folder.
|
||||
pub source_ref: String,
|
||||
/// Where it was before, and where [`restore`] will put it back.
|
||||
pub trashed_from: String,
|
||||
/// UTC seconds when it was trashed.
|
||||
pub trashed_at: i64,
|
||||
/// `oc:fileid`, preserved across the move. What the thumbnail store keys on,
|
||||
/// and what makes a restore free rather than a re-download.
|
||||
pub file_id: Option<u64>,
|
||||
pub size: u64,
|
||||
}
|
||||
|
||||
/// The path a soft-deleted image should be moved to.
|
||||
///
|
||||
/// Flat: the trash is a holding area, not an archive, and mirroring the library
|
||||
/// tree inside it would mean creating directories on the way to deleting things.
|
||||
/// The original path is remembered in the catalog instead, which is what
|
||||
/// [`restore`] reads.
|
||||
///
|
||||
/// **Collisions are resolved rather than allowed to overwrite.** Two files named
|
||||
/// `IMG_0001.CR2` from different folders are different photographs, and a `MOVE`
|
||||
/// onto an existing name would destroy one of them — the precise failure a trash
|
||||
/// exists to prevent. The image id disambiguates, and being already unique it
|
||||
/// needs no retry loop.
|
||||
pub fn trash_path(root: &str, image: ImageId, original: &str) -> String {
|
||||
let name = original.rsplit(['/', ':']).next().unwrap_or(original);
|
||||
let prefix = if root.is_empty() {
|
||||
String::new()
|
||||
} else {
|
||||
format!("{root}/")
|
||||
};
|
||||
format!("{prefix}{TRASH_DIR}/{}-{name}", image.0)
|
||||
}
|
||||
|
||||
/// Where a trashed image goes back to.
|
||||
///
|
||||
/// The stored original path, verbatim. Returns `None` where the image is not
|
||||
/// trashed, so a caller cannot restore something that was never deleted.
|
||||
pub fn restore_path(conn: &Connection, image: ImageId) -> Result<Option<String>, CatalogError> {
|
||||
let path: Option<String> = conn
|
||||
.query_row(
|
||||
"SELECT trashed_from FROM images
|
||||
WHERE id = ?1 AND trashed_at IS NOT NULL",
|
||||
[image.0 as i64],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.optional()?
|
||||
.flatten();
|
||||
Ok(path)
|
||||
}
|
||||
|
||||
/// Record that images have been moved to the trash.
|
||||
///
|
||||
/// Call **after** the move succeeds. Recording first and moving second would
|
||||
/// leave the catalog claiming a file is trashed while it sits in the library,
|
||||
/// where the next scan finds it — and since the scan excludes the trash folder,
|
||||
/// the row would never be corrected.
|
||||
///
|
||||
/// `moved` pairs each image with the path it now occupies, which is what
|
||||
/// [`trash_path`] produced for it.
|
||||
///
|
||||
/// Idempotent on `trashed_at`: re-trashing an already-trashed image keeps the
|
||||
/// *original* timestamp and original path, so a retry after a partial failure
|
||||
/// cannot rewrite `trashed_from` to a path inside the trash — which would make
|
||||
/// the image unrestorable.
|
||||
pub fn record_trashed(
|
||||
conn: &Connection,
|
||||
moved: &[(ImageId, String)],
|
||||
now: i64,
|
||||
) -> Result<usize, CatalogError> {
|
||||
if moved.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
let tx = conn.unchecked_transaction()?;
|
||||
let mut n = 0;
|
||||
|
||||
{
|
||||
let mut stmt = tx.prepare(
|
||||
"UPDATE images
|
||||
SET trashed_from = CASE
|
||||
WHEN trashed_at IS NULL THEN source_ref
|
||||
ELSE trashed_from
|
||||
END,
|
||||
source_ref = ?2,
|
||||
trashed_at = coalesce(trashed_at, ?3)
|
||||
WHERE id = ?1",
|
||||
)?;
|
||||
for (image, path) in moved {
|
||||
n += stmt.execute(rusqlite::params![image.0 as i64, path, now])?;
|
||||
}
|
||||
}
|
||||
|
||||
tx.commit()?;
|
||||
Ok(n)
|
||||
}
|
||||
|
||||
/// Record that images have been moved back out of the trash.
|
||||
///
|
||||
/// Call after the move succeeds, for the same reason as [`record_trashed`].
|
||||
/// Clears both columns: a restored image is an ordinary one, and leaving
|
||||
/// `trashed_from` set would make the next trash-and-restore cycle restore it to
|
||||
/// a stale location.
|
||||
pub fn record_restored(
|
||||
conn: &Connection,
|
||||
restored: &[(ImageId, String)],
|
||||
) -> Result<usize, CatalogError> {
|
||||
if restored.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
let tx = conn.unchecked_transaction()?;
|
||||
let mut n = 0;
|
||||
|
||||
{
|
||||
let mut stmt = tx.prepare(
|
||||
"UPDATE images
|
||||
SET source_ref = ?2, trashed_at = NULL, trashed_from = NULL
|
||||
WHERE id = ?1 AND trashed_at IS NOT NULL",
|
||||
)?;
|
||||
for (image, path) in restored {
|
||||
n += stmt.execute(rusqlite::params![image.0 as i64, path])?;
|
||||
}
|
||||
}
|
||||
|
||||
tx.commit()?;
|
||||
Ok(n)
|
||||
}
|
||||
|
||||
/// Forget images whose files have been permanently deleted.
|
||||
///
|
||||
/// Call **after** the remote delete succeeds — see [`purge_order`].
|
||||
///
|
||||
/// Deletes the catalog rows outright rather than tombstoning them. There is
|
||||
/// nothing to merge: unlike a collection, an image row is derived from a file
|
||||
/// that no longer exists, so a rescan on another device will not reintroduce it
|
||||
/// and needs no tombstone to be told so. `ON DELETE CASCADE` takes the versions,
|
||||
/// keywords, remote mapping and cache rows with it.
|
||||
///
|
||||
/// Returns how many rows went.
|
||||
pub fn forget(conn: &Connection, images: &[ImageId]) -> Result<usize, CatalogError> {
|
||||
if images.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
let tx = conn.unchecked_transaction()?;
|
||||
let mut n = 0;
|
||||
{
|
||||
let mut stmt = tx.prepare("DELETE FROM images WHERE id = ?1")?;
|
||||
for image in images {
|
||||
n += stmt.execute([image.0 as i64])?;
|
||||
}
|
||||
}
|
||||
tx.commit()?;
|
||||
Ok(n)
|
||||
}
|
||||
|
||||
/// Why the file is deleted before the row.
|
||||
///
|
||||
/// Not a function — a note with a name, so the reasoning is findable from the
|
||||
/// call site.
|
||||
///
|
||||
/// **File first, then the row.** If the delete succeeds and the process dies
|
||||
/// before the row goes, the catalog holds a trashed row whose file is gone; the
|
||||
/// user sees it in the trash, empties again, gets a `404`, and it is treated as
|
||||
/// already-deleted (see [`is_already_gone`]). Recoverable, and visible.
|
||||
///
|
||||
/// The other order loses the file silently. Dropping the row first and dying
|
||||
/// before the delete leaves an orphan in `.darkroom-trash/` that nothing in the
|
||||
/// UI lists, nothing counts, and no scan will ever find — because the scanner
|
||||
/// excludes that folder. It consumes quota forever and the user has no way to
|
||||
/// learn it is there.
|
||||
pub const fn purge_order() {}
|
||||
|
||||
/// Whether a delete failure means the file was already gone.
|
||||
///
|
||||
/// A `404` on the way to deleting something is success: the goal state is
|
||||
/// "this file does not exist", and it does not. Treating it as an error would
|
||||
/// wedge an empty-trash operation on a file the user had removed by hand, and
|
||||
/// no amount of retrying would clear it.
|
||||
pub fn is_already_gone(status: Option<u16>) -> bool {
|
||||
matches!(status, Some(404) | Some(410))
|
||||
}
|
||||
|
||||
/// List what is in the trash, newest first.
|
||||
///
|
||||
/// Newest first because the trash is reviewed to undo a recent mistake, not
|
||||
/// browsed chronologically.
|
||||
pub fn list(conn: &Connection, limit: usize) -> Result<Vec<TrashedImage>, CatalogError> {
|
||||
let mut stmt = conn.prepare(
|
||||
"SELECT i.id, i.source_ref, i.trashed_from, i.trashed_at, r.file_id, i.file_size
|
||||
FROM images i
|
||||
LEFT JOIN remote r ON r.image_id = i.id
|
||||
WHERE i.trashed_at IS NOT NULL
|
||||
ORDER BY i.trashed_at DESC, i.id DESC
|
||||
LIMIT ?1",
|
||||
)?;
|
||||
let rows = stmt
|
||||
.query_map([limit as i64], |r| {
|
||||
let source_ref: String = r.get(1)?;
|
||||
Ok(TrashedImage {
|
||||
image_id: ImageId(r.get::<_, i64>(0)? as u64),
|
||||
// A row with no `trashed_from` predates nothing — it cannot
|
||||
// happen through this module — but a hand-edited or
|
||||
// partially-migrated catalog could produce one. Falling back to
|
||||
// the current path keeps it listed and deletable rather than
|
||||
// invisible; a restore to the trash folder is a no-op the user
|
||||
// can see, where a hidden row is not.
|
||||
trashed_from: r
|
||||
.get::<_, Option<String>>(2)?
|
||||
.unwrap_or_else(|| source_ref.clone()),
|
||||
source_ref,
|
||||
trashed_at: r.get(3)?,
|
||||
file_id: r.get::<_, Option<i64>>(4)?.map(|v| v as u64),
|
||||
size: r.get::<_, Option<i64>>(5)?.unwrap_or(0) as u64,
|
||||
})
|
||||
})?
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
/// Every trashed image id, for emptying the whole trash.
|
||||
///
|
||||
/// Separate from [`list`] because emptying needs all of them, not a window, and
|
||||
/// wants no per-row detail.
|
||||
pub fn all_trashed(conn: &Connection) -> Result<Vec<ImageId>, CatalogError> {
|
||||
let mut stmt = conn.prepare("SELECT id FROM images WHERE trashed_at IS NOT NULL")?;
|
||||
let rows = stmt
|
||||
.query_map([], |r| Ok(ImageId(r.get::<_, i64>(0)? as u64)))?
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
/// How many images are in the trash, and how many bytes they hold.
|
||||
///
|
||||
/// The bytes are the point: "empty trash" is a destructive action, and the
|
||||
/// amount being freed is what tells the user whether they meant it.
|
||||
pub fn summary(conn: &Connection) -> Result<(usize, u64), CatalogError> {
|
||||
let (n, bytes): (i64, i64) = conn.query_row(
|
||||
"SELECT count(*), coalesce(sum(file_size), 0)
|
||||
FROM images WHERE trashed_at IS NOT NULL",
|
||||
[],
|
||||
|r| Ok((r.get(0)?, r.get(1)?)),
|
||||
)?;
|
||||
Ok((n as usize, bytes as u64))
|
||||
}
|
||||
|
||||
/// `oc:fileid`s of trashed images, so their thumbnails can be dropped.
|
||||
///
|
||||
/// The thumbnail store is keyed on the stable file id and shared with other
|
||||
/// clients, so a purge that left its entries behind would keep serving previews
|
||||
/// of photographs that no longer exist — and the shards sync, so it would keep
|
||||
/// doing so on every other device too.
|
||||
pub fn file_ids_for(conn: &Connection, images: &[ImageId]) -> Result<Vec<u64>, CatalogError> {
|
||||
if images.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
let placeholders = std::iter::repeat_n("?", images.len())
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let sql = format!("SELECT file_id FROM remote WHERE image_id IN ({placeholders})");
|
||||
let params: Vec<rusqlite::types::Value> = images
|
||||
.iter()
|
||||
.map(|i| rusqlite::types::Value::Integer(i.0 as i64))
|
||||
.collect();
|
||||
|
||||
let mut stmt = conn.prepare(&sql)?;
|
||||
let rows = stmt
|
||||
.query_map(rusqlite::params_from_iter(params.iter()), |r| {
|
||||
Ok(r.get::<_, i64>(0)? as u64)
|
||||
})?
|
||||
.collect::<Result<Vec<_>, _>>()?;
|
||||
Ok(rows)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::Catalog;
|
||||
|
||||
fn seeded() -> Catalog {
|
||||
let cat = Catalog::in_memory().unwrap();
|
||||
let c = cat.connection();
|
||||
c.execute(
|
||||
"INSERT INTO roots(id, kind, label) VALUES (1, 'remote', 'PhotosRaw')",
|
||||
[],
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1..=4i64 {
|
||||
c.execute(
|
||||
"INSERT INTO images(id, root_id, source_ref, file_size, added_at)
|
||||
VALUES (?1, 1, ?2, ?3, 0)",
|
||||
rusqlite::params![i, format!("PhotosRaw/2019/IMG_{i:04}.CR2"), 30_000_000 * i],
|
||||
)
|
||||
.unwrap();
|
||||
c.execute(
|
||||
"INSERT INTO remote(image_id, file_id) VALUES (?1, ?2)",
|
||||
rusqlite::params![i, 1000 + i],
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
cat
|
||||
}
|
||||
|
||||
fn img(i: u64) -> ImageId {
|
||||
ImageId(i)
|
||||
}
|
||||
|
||||
/// Trash one image the way the UI does: compute the path, then record.
|
||||
fn do_trash(cat: &Catalog, i: u64, now: i64) -> String {
|
||||
let c = cat.connection();
|
||||
let original: String = c
|
||||
.query_row(
|
||||
"SELECT source_ref FROM images WHERE id = ?1",
|
||||
[i as i64],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
let to = trash_path("PhotosRaw", img(i), &original);
|
||||
record_trashed(c, &[(img(i), to.clone())], now).unwrap();
|
||||
to
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_trash_directory_matches_the_one_the_scanner_excludes() {
|
||||
// These are two constants in two crates that must agree, or the scan
|
||||
// re-indexes the trash and every soft delete comes undone.
|
||||
assert_eq!(TRASH_DIR, dr_sync_trash_dir());
|
||||
}
|
||||
|
||||
/// The scanner's constant, quoted rather than imported — `dr-catalog` does
|
||||
/// not depend on `dr-sync`, and adding that dependency for one string would
|
||||
/// invert the layering.
|
||||
fn dr_sync_trash_dir() -> &'static str {
|
||||
".darkroom-trash"
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trashing_moves_the_path_and_remembers_where_it_came_from() {
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
do_trash(&cat, 1, 5_000);
|
||||
|
||||
let (source, from, at): (String, String, i64) = c
|
||||
.query_row(
|
||||
"SELECT source_ref, trashed_from, trashed_at FROM images WHERE id = 1",
|
||||
[],
|
||||
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// `source_ref` follows the bytes: this is where a fetch must now look.
|
||||
assert!(source.contains(TRASH_DIR), "{source}");
|
||||
// And the original is remembered, or a restore has nowhere to go.
|
||||
assert_eq!(from, "PhotosRaw/2019/IMG_0001.CR2");
|
||||
assert_eq!(at, 5_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_trash_path_keeps_the_original_filename_recognisable() {
|
||||
// The user reviewing the trash needs to recognise the photograph; an
|
||||
// opaque id alone would make the list unreadable.
|
||||
let p = trash_path("PhotosRaw", img(7), "PhotosRaw/2019/IMG_0042.CR2");
|
||||
assert!(p.ends_with("IMG_0042.CR2"), "{p}");
|
||||
assert!(p.starts_with("PhotosRaw/.darkroom-trash/"), "{p}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn two_files_with_the_same_name_do_not_collide_in_the_trash() {
|
||||
// The failure a trash exists to prevent: a MOVE onto an existing name
|
||||
// destroys one of two different photographs.
|
||||
let a = trash_path("PhotosRaw", img(1), "PhotosRaw/2019/IMG_0001.CR2");
|
||||
let b = trash_path("PhotosRaw", img(2), "PhotosRaw/2024/IMG_0001.CR2");
|
||||
assert_ne!(a, b);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_whole_account_root_yields_no_leading_slash() {
|
||||
// The root is empty when the library is the whole account; a path
|
||||
// beginning "/" would resolve differently on the server.
|
||||
let p = trash_path("", img(3), "2019/IMG_0003.CR2");
|
||||
assert_eq!(p, ".darkroom-trash/3-IMG_0003.CR2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn restoring_puts_the_original_path_back_and_clears_the_flag() {
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
do_trash(&cat, 1, 5_000);
|
||||
|
||||
let back = restore_path(c, img(1))
|
||||
.unwrap()
|
||||
.expect("knows where it came from");
|
||||
assert_eq!(back, "PhotosRaw/2019/IMG_0001.CR2");
|
||||
|
||||
record_restored(c, &[(img(1), back.clone())]).unwrap();
|
||||
|
||||
let (source, at): (String, Option<i64>) = c
|
||||
.query_row(
|
||||
"SELECT source_ref, trashed_at FROM images WHERE id = 1",
|
||||
[],
|
||||
|r| Ok((r.get(0)?, r.get(1)?)),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(source, back);
|
||||
assert_eq!(at, None, "a restored image is an ordinary one");
|
||||
assert!(restore_path(c, img(1)).unwrap().is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_trash_restore_trash_cycle_restores_to_the_right_place_twice() {
|
||||
// If `trashed_from` were not cleared on restore, the second trash would
|
||||
// record a stale origin and the second restore would put the file
|
||||
// somewhere it never was.
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
|
||||
do_trash(&cat, 1, 1_000);
|
||||
let first = restore_path(c, img(1)).unwrap().unwrap();
|
||||
record_restored(c, &[(img(1), first.clone())]).unwrap();
|
||||
|
||||
do_trash(&cat, 1, 2_000);
|
||||
let second = restore_path(c, img(1)).unwrap().unwrap();
|
||||
assert_eq!(
|
||||
first, second,
|
||||
"the origin is the library path, not the trash"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn re_trashing_does_not_overwrite_the_original_path() {
|
||||
// A retry after a partial failure must not record a trash-folder path as
|
||||
// the origin — that makes the image unrestorable.
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
let to = do_trash(&cat, 1, 1_000);
|
||||
// Second attempt, as a retry would do.
|
||||
record_trashed(c, &[(img(1), to)], 9_999).unwrap();
|
||||
|
||||
let (from, at): (String, i64) = c
|
||||
.query_row(
|
||||
"SELECT trashed_from, trashed_at FROM images WHERE id = 1",
|
||||
[],
|
||||
|r| Ok((r.get(0)?, r.get(1)?)),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(from, "PhotosRaw/2019/IMG_0001.CR2");
|
||||
assert_eq!(at, 1_000, "the original timestamp survives a retry");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn restoring_something_that_was_never_trashed_does_nothing() {
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
assert!(restore_path(c, img(2)).unwrap().is_none());
|
||||
assert_eq!(
|
||||
record_restored(c, &[(img(2), "elsewhere".into())]).unwrap(),
|
||||
0
|
||||
);
|
||||
// And its path is untouched.
|
||||
let source: String = c
|
||||
.query_row("SELECT source_ref FROM images WHERE id = 2", [], |r| {
|
||||
r.get(0)
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(source, "PhotosRaw/2019/IMG_0002.CR2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_trash_lists_newest_first() {
|
||||
// Reviewed to undo a recent mistake, not browsed chronologically.
|
||||
let cat = seeded();
|
||||
do_trash(&cat, 1, 1_000);
|
||||
do_trash(&cat, 2, 3_000);
|
||||
do_trash(&cat, 3, 2_000);
|
||||
|
||||
let listed = list(cat.connection(), 100).unwrap();
|
||||
let order: Vec<u64> = listed.iter().map(|t| t.image_id.0).collect();
|
||||
assert_eq!(order, vec![2, 3, 1]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_trash_list_carries_the_file_id_a_restore_needs() {
|
||||
// Without it a restore cannot find the thumbnail it already has, and
|
||||
// re-downloads a preview it is holding.
|
||||
let cat = seeded();
|
||||
do_trash(&cat, 1, 1_000);
|
||||
let listed = list(cat.connection(), 10).unwrap();
|
||||
assert_eq!(listed[0].file_id, Some(1001));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_summary_reports_what_emptying_would_free() {
|
||||
// "Empty trash" is destructive; the size is what tells the user whether
|
||||
// they meant it.
|
||||
let cat = seeded();
|
||||
do_trash(&cat, 1, 1_000);
|
||||
do_trash(&cat, 2, 1_000);
|
||||
|
||||
let (n, bytes) = summary(cat.connection()).unwrap();
|
||||
assert_eq!(n, 2);
|
||||
assert_eq!(bytes, 30_000_000 + 60_000_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_empty_trash_summarises_as_zero_rather_than_erroring() {
|
||||
let cat = seeded();
|
||||
assert_eq!(summary(cat.connection()).unwrap(), (0, 0));
|
||||
assert!(all_trashed(cat.connection()).unwrap().is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn purging_removes_the_row_and_everything_hanging_off_it() {
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
do_trash(&cat, 1, 1_000);
|
||||
|
||||
assert_eq!(forget(c, &[img(1)]).unwrap(), 1);
|
||||
|
||||
let n: i64 = c
|
||||
.query_row("SELECT count(*) FROM images WHERE id = 1", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(n, 0);
|
||||
// The remote mapping must go too, or a later scan could pair a new file
|
||||
// with a dead image's id.
|
||||
let n: i64 = c
|
||||
.query_row("SELECT count(*) FROM remote WHERE image_id = 1", [], |r| {
|
||||
r.get(0)
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(n, 0, "cascaded");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn purging_leaves_untrashed_images_alone() {
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
do_trash(&cat, 1, 1_000);
|
||||
forget(c, &all_trashed(c).unwrap()).unwrap();
|
||||
|
||||
let n: i64 = c
|
||||
.query_row("SELECT count(*) FROM images", [], |r| r.get(0))
|
||||
.unwrap();
|
||||
assert_eq!(n, 3, "only the trashed one went");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn file_ids_are_collected_so_thumbnails_can_be_dropped() {
|
||||
// The shards sync to the server; a purge that left them would serve
|
||||
// previews of deleted photographs on every device.
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
do_trash(&cat, 1, 1_000);
|
||||
do_trash(&cat, 2, 1_000);
|
||||
|
||||
let mut ids = file_ids_for(c, &[img(1), img(2)]).unwrap();
|
||||
ids.sort_unstable();
|
||||
assert_eq!(ids, vec![1001, 1002]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_missing_file_counts_as_already_deleted() {
|
||||
// Otherwise one file removed by hand wedges every future empty-trash,
|
||||
// and no amount of retrying clears it.
|
||||
assert!(is_already_gone(Some(404)));
|
||||
assert!(is_already_gone(Some(410)));
|
||||
assert!(!is_already_gone(Some(403)), "a permission failure is real");
|
||||
assert!(!is_already_gone(Some(500)));
|
||||
assert!(!is_already_gone(None));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_batches_are_no_ops_rather_than_errors() {
|
||||
// The UI can reach these with nothing selected.
|
||||
let cat = seeded();
|
||||
let c = cat.connection();
|
||||
assert_eq!(record_trashed(c, &[], 0).unwrap(), 0);
|
||||
assert_eq!(record_restored(c, &[]).unwrap(), 0);
|
||||
assert_eq!(forget(c, &[]).unwrap(), 0);
|
||||
assert!(file_ids_for(c, &[]).unwrap().is_empty());
|
||||
}
|
||||
}
|
||||
@@ -1,23 +0,0 @@
|
||||
[package]
|
||||
name = "dr-decode"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
license.workspace = true
|
||||
|
||||
[dependencies]
|
||||
dr-types.workspace = true
|
||||
rawler.workspace = true
|
||||
# The camera profile database is data, not code (FR-DEV-3e): a YAML file that
|
||||
# ships with the binary and is superseded by a newer one on disk. serde_norway
|
||||
# is the workspace's YAML crate — the fork still receiving releases — and it is
|
||||
# already in the tree for `dr-pipeline`'s node declarations and `dr-ui`'s style
|
||||
# tokens. Pure Rust, so it costs nothing under the Android NDK.
|
||||
serde = { workspace = true }
|
||||
serde_norway.workspace = true
|
||||
zune-jpeg.workspace = true
|
||||
thiserror.workspace = true
|
||||
log.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
env_logger.workspace = true
|
||||
@@ -1,52 +0,0 @@
|
||||
//! Report the defect map a raw file carries, if it carries one.
|
||||
//!
|
||||
//! ```text
|
||||
//! cargo run -p dr-decode --example defects -- IMG_6320.dng photo.cr2
|
||||
//! ```
|
||||
//!
|
||||
//! Exists because whether this is worth building a correction stage for is a
|
||||
//! question about *your files*, not about the specification: DNGs written by
|
||||
//! cameras that map their own sensors carry `OpcodeList1`, conversions from a
|
||||
//! proprietary raw usually do not, and no CR2 or scanner TIFF ever does.
|
||||
//! Rather than guess, point this at the library and see.
|
||||
|
||||
fn main() {
|
||||
let files: Vec<String> = std::env::args().skip(1).collect();
|
||||
if files.is_empty() {
|
||||
eprintln!("usage: defects <raw file>...");
|
||||
std::process::exit(2);
|
||||
}
|
||||
|
||||
for path in &files {
|
||||
let bytes = match std::fs::read(path) {
|
||||
Ok(b) => b,
|
||||
Err(e) => {
|
||||
println!("{path}: unreadable — {e}");
|
||||
continue;
|
||||
}
|
||||
};
|
||||
|
||||
let found = dr_decode::defects(&bytes);
|
||||
if found.is_empty() {
|
||||
println!("{path}: no defect map");
|
||||
continue;
|
||||
}
|
||||
|
||||
println!(
|
||||
"{path}: {} bad pixel(s), {} bad line(s)",
|
||||
found.pixels.len(),
|
||||
found.lines.len()
|
||||
);
|
||||
// A handful, so the output stays readable on a sensor reporting
|
||||
// hundreds — the count above is the number that matters.
|
||||
for p in found.pixels.iter().take(8) {
|
||||
println!(" pixel at {},{}", p.x, p.y);
|
||||
}
|
||||
for l in found.lines.iter().take(8) {
|
||||
match l {
|
||||
dr_decode::BadLine::Column(x) => println!(" dead column {x}"),
|
||||
dr_decode::BadLine::Row(y) => println!(" dead row {y}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,19 +0,0 @@
|
||||
fn main() {
|
||||
for p in std::env::args().skip(1) {
|
||||
let Ok(d) = std::fs::read(&p) else { continue };
|
||||
let n = p.rsplit('/').next().unwrap();
|
||||
// Exactly what the sweep sees: the first HEADER_BYTES only.
|
||||
let head = &d[..d.len().min(dr_decode::HEADER_BYTES as usize)];
|
||||
match dr_decode::metadata(head) {
|
||||
Ok(m) => println!(
|
||||
"{n}: header-only at={:?} model={:?}",
|
||||
m.captured_at, m.model
|
||||
),
|
||||
Err(e) => println!("{n}: header-only ERROR {e}"),
|
||||
}
|
||||
match dr_decode::metadata(&d) {
|
||||
Ok(m) => println!("{n}: whole-file at={:?}", m.captured_at),
|
||||
Err(e) => println!("{n}: whole-file ERROR {e}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,61 +0,0 @@
|
||||
//! Print what `decode` extracts from a RAW file.
|
||||
//!
|
||||
//! A sanity check on the pipeline's inputs: black and white levels, the CFA
|
||||
//! pattern after re-phasing, as-shot white balance, and the camera→sRGB
|
||||
//! matrix. Wrong values here produce a wrong image no shader can fix, so it
|
||||
//! is worth being able to see them directly.
|
||||
//!
|
||||
//! ```sh
|
||||
//! cargo run -p dr-decode --example rawinfo -- IMG.CR2
|
||||
//! ```
|
||||
|
||||
fn main() {
|
||||
let Some(path) = std::env::args().nth(1) else {
|
||||
eprintln!("usage: rawinfo <file.cr2>");
|
||||
std::process::exit(2);
|
||||
};
|
||||
|
||||
let bytes = std::fs::read(&path).expect("read file");
|
||||
let raw = dr_decode::decode(&bytes).expect("decode");
|
||||
|
||||
println!("file {path}");
|
||||
println!("readout {} × {}", raw.width, raw.height);
|
||||
println!(
|
||||
"crop {} × {} at ({}, {})",
|
||||
raw.crop.width, raw.crop.height, raw.crop.x, raw.crop.y
|
||||
);
|
||||
let (dx, dy) = raw.crop.shifts_cfa_phase();
|
||||
println!(
|
||||
"cfa {:?} (rephased: {dx}, {dy})",
|
||||
raw.cfa_pattern
|
||||
);
|
||||
println!("black {:?}", raw.black_level);
|
||||
println!("white {}", raw.white_level);
|
||||
println!("wb_coeffs {:?}", raw.wb_coeffs);
|
||||
|
||||
match raw.color_matrix {
|
||||
Some(m) => {
|
||||
println!("cam→srgb");
|
||||
for row in m.chunks(3) {
|
||||
println!(
|
||||
" [{:>8.4} {:>8.4} {:>8.4}]",
|
||||
row[0], row[1], row[2]
|
||||
);
|
||||
}
|
||||
// Each row should sum to roughly 1: a neutral camera-space colour
|
||||
// must stay neutral in sRGB. Far from 1 means the normalisation
|
||||
// or the matrix composition is wrong.
|
||||
let sums: Vec<f32> = m.chunks(3).map(|r| r.iter().sum()).collect();
|
||||
println!("row sums {sums:.4?} (≈1.0 each if correct)");
|
||||
}
|
||||
None => println!("cam→srgb none — uncalibrated body"),
|
||||
}
|
||||
|
||||
// Sample the actual data range, which reveals a black-level or bit-depth
|
||||
// mistake faster than any amount of staring at metadata.
|
||||
let (min, max) = raw
|
||||
.data
|
||||
.iter()
|
||||
.fold((u16::MAX, 0u16), |(lo, hi), &v| (lo.min(v), hi.max(v)));
|
||||
println!("sample range {min} … {max}");
|
||||
}
|
||||
@@ -1,146 +0,0 @@
|
||||
//! Smoke test against real RAW files.
|
||||
//!
|
||||
//! cargo run -p dr-decode --example smoke -- <file-or-dir>...
|
||||
//!
|
||||
//! Reports, per file, what each entry point costs — which is the whole reason
|
||||
//! they are separate (ARCH §3.2).
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::time::Instant;
|
||||
|
||||
fn main() {
|
||||
env_logger::init();
|
||||
let args: Vec<String> = std::env::args().skip(1).collect();
|
||||
if args.is_empty() {
|
||||
eprintln!("usage: smoke <file-or-dir>...");
|
||||
std::process::exit(2);
|
||||
}
|
||||
|
||||
let mut files = Vec::new();
|
||||
for a in &args {
|
||||
let p = PathBuf::from(a);
|
||||
if p.is_dir() {
|
||||
collect(&p, &mut files);
|
||||
} else {
|
||||
files.push(p);
|
||||
}
|
||||
}
|
||||
files.sort();
|
||||
files.truncate(8);
|
||||
|
||||
println!(
|
||||
"{:<20} {:>7} {:>8} {:>9} {:>13} {:>9} {:>13}",
|
||||
"file", "size", "meta", "thumb", "thumb dims", "full", "full dims"
|
||||
);
|
||||
println!("{}", "-".repeat(88));
|
||||
|
||||
let (mut ok, mut failed) = (0, 0);
|
||||
for f in &files {
|
||||
match run_one(f) {
|
||||
Ok(line) => {
|
||||
println!("{line}");
|
||||
ok += 1;
|
||||
}
|
||||
Err(e) => {
|
||||
println!("{:<22} {e}", truncate(&name(f), 22));
|
||||
failed += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
println!("\n{ok} ok, {failed} failed");
|
||||
if failed > 0 {
|
||||
std::process::exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
fn run_one(path: &Path) -> Result<String, String> {
|
||||
let size = std::fs::metadata(path).map_err(|e| e.to_string())?.len();
|
||||
|
||||
// The culling path: read only the header region, not the whole file.
|
||||
let probe_bytes =
|
||||
read_prefix(path, dr_decode::PREVIEW_PROBE_BYTES).map_err(|e| e.to_string())?;
|
||||
let t0 = Instant::now();
|
||||
let fmt = dr_decode::probe(&probe_bytes);
|
||||
let meta = dr_decode::metadata(&probe_bytes).ok();
|
||||
let meta_ms = t0.elapsed().as_secs_f64() * 1000.0;
|
||||
|
||||
let all = std::fs::read(path).map_err(|e| e.to_string())?;
|
||||
|
||||
// The culling rung.
|
||||
let t1 = Instant::now();
|
||||
let thumb = dr_decode::extract_preview(&all, dr_decode::PreviewSize::Thumbnail)
|
||||
.map_err(|e| format!("thumb: {e}"))?;
|
||||
let thumb_ms = t1.elapsed().as_secs_f64() * 1000.0;
|
||||
|
||||
// The full-resolution rung, for comparison.
|
||||
let t2 = Instant::now();
|
||||
let full = dr_decode::extract_preview(&all, dr_decode::PreviewSize::Full)
|
||||
.map_err(|e| format!("full: {e}"))?;
|
||||
let full_ms = t2.elapsed().as_secs_f64() * 1000.0;
|
||||
|
||||
let model = meta
|
||||
.as_ref()
|
||||
.and_then(|m| m.model.clone())
|
||||
.unwrap_or_else(|| "?".into());
|
||||
let budget = if thumb_ms <= 50.0 {
|
||||
""
|
||||
} else {
|
||||
" OVER BUDGET"
|
||||
};
|
||||
|
||||
Ok(format!(
|
||||
"{:<20} {:>6.1}M {:>6.1}ms {:>7.1}ms {:>7}x{:<5} {:>7.1}ms {:>7}x{:<5} {:?} {}{}",
|
||||
truncate(&name(path), 20),
|
||||
size as f64 / 1e6,
|
||||
meta_ms,
|
||||
thumb_ms,
|
||||
thumb.width,
|
||||
thumb.height,
|
||||
full_ms,
|
||||
full.width,
|
||||
full.height,
|
||||
fmt,
|
||||
model.trim(),
|
||||
budget,
|
||||
))
|
||||
}
|
||||
|
||||
fn read_prefix(path: &Path, n: u64) -> std::io::Result<Vec<u8>> {
|
||||
use std::io::Read;
|
||||
let mut f = std::fs::File::open(path)?;
|
||||
let mut buf = vec![0u8; n as usize];
|
||||
let read = f.read(&mut buf)?;
|
||||
buf.truncate(read);
|
||||
Ok(buf)
|
||||
}
|
||||
|
||||
fn collect(dir: &Path, out: &mut Vec<PathBuf>) {
|
||||
let Ok(entries) = std::fs::read_dir(dir) else {
|
||||
return;
|
||||
};
|
||||
for e in entries.flatten() {
|
||||
let p = e.path();
|
||||
if p.is_file() {
|
||||
let ext = p
|
||||
.extension()
|
||||
.map(|s| s.to_string_lossy().to_ascii_lowercase())
|
||||
.unwrap_or_default();
|
||||
if dr_types::Format::from_extension(&ext).is_some() {
|
||||
out.push(p);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn name(p: &Path) -> String {
|
||||
p.file_name().unwrap_or_default().to_string_lossy().into()
|
||||
}
|
||||
|
||||
fn truncate(s: &str, n: usize) -> String {
|
||||
if s.len() <= n {
|
||||
s.to_string()
|
||||
} else {
|
||||
format!("{}…", &s[..n - 1])
|
||||
}
|
||||
}
|
||||
@@ -1,160 +0,0 @@
|
||||
# DarkRoom camera base curves (FR-DEV-3e).
|
||||
#
|
||||
# ---------------------------------------------------------------------------
|
||||
# Adding a body is editing this file. It is not a code change.
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# The copy you are reading is compiled into the binary as a floor. At startup
|
||||
# `dr_decode::base_curve::load` also looks for `base_curves.yaml` in:
|
||||
#
|
||||
# 1. $DARKROOM_PROFILES/ (set it while you are tuning)
|
||||
# 2. $XDG_DATA_HOME/darkroom/profiles/
|
||||
# or $HOME/.local/share/darkroom/profiles/
|
||||
#
|
||||
# and uses the first one it finds *whose `version:` is higher than this one's*.
|
||||
# So: bump `version`, drop the file in that directory, restart. A body added
|
||||
# this afternoon renders correctly this afternoon, with no release and no
|
||||
# rebuild — which is what the requirement asks for, and what makes these
|
||||
# contributable under the GPL.
|
||||
#
|
||||
# The version check runs both ways on purpose. A file older than the built-in
|
||||
# copy is ignored with a log line, so upgrading DarkRoom cannot silently lose
|
||||
# curves to a pack somebody downloaded a year ago.
|
||||
#
|
||||
# ---------------------------------------------------------------------------
|
||||
# What the numbers mean
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# Five `[x, y]` control points on a monotone spline (Fritsch-Carlson, the same
|
||||
# one the tone curve widget draws). Both axes are **linear**:
|
||||
#
|
||||
# x scene-referred camera RGB after white balance, 1.0 = sensor saturation
|
||||
# y display-referred linear; the sRGB transfer function is applied later,
|
||||
# at the end of the shader, so do not pre-apply a gamma here
|
||||
#
|
||||
# The identity is y = x, and it is what an unrecognised body gets if `default:`
|
||||
# is removed. It is also the wrong answer for almost every photograph: linear
|
||||
# scene data has middle grey at about 13% and a camera JPEG puts it near 18%,
|
||||
# so an uncurved render is roughly half a stop dark through the midtones and
|
||||
# has no highlight rolloff at all.
|
||||
#
|
||||
# A curve that works has three parts, and it is worth naming them because they
|
||||
# are what you are actually tuning:
|
||||
#
|
||||
# the toe the first span, slope near or below 1. Deep shadows stay
|
||||
# deep. Lift it and blacks go milky; crush it and shadow
|
||||
# detail the sensor recorded disappears.
|
||||
# the midtones the middle spans, slope well above 1. This is the contrast
|
||||
# and the brightness people read as "the camera's look".
|
||||
# the shoulder the last span, slope well below 1. Highlights compress
|
||||
# toward white instead of arriving there and clipping. It is
|
||||
# the difference between a rolled-off sky and a white hole.
|
||||
#
|
||||
# Two invariants are enforced in code and tested, so a mistake here fails the
|
||||
# build rather than the photograph: x must strictly increase, y must not
|
||||
# decrease, and everything must lie inside the unit square.
|
||||
#
|
||||
# ---------------------------------------------------------------------------
|
||||
# Honesty about these values
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# These are hand-tuned shapes, not measurements. They encode what every camera
|
||||
# JPEG rendering has in common — the toe/midtone/shoulder structure above —
|
||||
# plus each maker's well-known house differences: Canon's gentler shoulder and
|
||||
# warmer-reading midtones, Nikon's slightly higher midtone contrast, Sony's
|
||||
# flatter and more conservative default, Fujifilm's markedly contrastier
|
||||
# Provia-derived rendering.
|
||||
#
|
||||
# FR-DEV-3e's acceptance criterion is subjective comparison against each body's
|
||||
# own JPEG, and meeting it properly needs a frame from that body in front of
|
||||
# you. Where that has not been done, the entry is still much closer to right
|
||||
# than the identity — which is the bar these have to clear, and do.
|
||||
|
||||
version: 1
|
||||
|
||||
# The rendering for a body with no entry of its own.
|
||||
#
|
||||
# **Deliberately not the identity.** The failure this requirement exists to fix
|
||||
# is the flat render, and a conservative curve is far closer to right for every
|
||||
# body than no curve is for any of them. It is gentler than the per-body
|
||||
# entries below — a shallower midtone and an earlier, softer shoulder — because
|
||||
# it has to be safe on a sensor nobody has looked at, and the cost of being too
|
||||
# tame is a photograph that wants a little contrast rather than one that has
|
||||
# lost its highlights.
|
||||
default:
|
||||
points:
|
||||
- [0.00, 0.000]
|
||||
- [0.04, 0.043]
|
||||
- [0.13, 0.175]
|
||||
- [0.45, 0.690]
|
||||
- [1.00, 1.000]
|
||||
|
||||
bodies:
|
||||
# Canon. A soft toe and a long, gradual shoulder — the reason Canon files
|
||||
# are described as forgiving in highlights and a little low in contrast
|
||||
# straight out of camera.
|
||||
- make: Canon
|
||||
model: EOS 6D
|
||||
points:
|
||||
- [0.00, 0.000]
|
||||
- [0.04, 0.045]
|
||||
- [0.13, 0.190]
|
||||
- [0.45, 0.720]
|
||||
- [1.00, 1.000]
|
||||
|
||||
- make: Canon
|
||||
model: EOS R6
|
||||
points:
|
||||
- [0.00, 0.000]
|
||||
- [0.04, 0.044]
|
||||
- [0.13, 0.195]
|
||||
- [0.45, 0.730]
|
||||
- [1.00, 1.000]
|
||||
|
||||
# Nikon. A slightly deeper toe and more midtone slope than Canon, which is
|
||||
# the "punchier out of camera" difference people describe between the two.
|
||||
- make: Nikon
|
||||
model: Z 6
|
||||
points:
|
||||
- [0.00, 0.000]
|
||||
- [0.04, 0.038]
|
||||
- [0.13, 0.200]
|
||||
- [0.46, 0.750]
|
||||
- [1.00, 1.000]
|
||||
|
||||
- make: Nikon
|
||||
model: D750
|
||||
points:
|
||||
- [0.00, 0.000]
|
||||
- [0.04, 0.039]
|
||||
- [0.13, 0.198]
|
||||
- [0.46, 0.745]
|
||||
- [1.00, 1.000]
|
||||
|
||||
# Sony. The flattest default of the four, and intentionally so — Sony's own
|
||||
# rendering leaves more headroom than it uses, which is why Sony files are
|
||||
# the ones people describe as needing the most work.
|
||||
- make: Sony
|
||||
model: ILCE-7M3
|
||||
points:
|
||||
- [0.00, 0.000]
|
||||
- [0.04, 0.048]
|
||||
- [0.13, 0.185]
|
||||
- [0.44, 0.700]
|
||||
- [1.00, 1.000]
|
||||
|
||||
# Fujifilm. Provia, the default film simulation: a firm toe, the steepest
|
||||
# midtones here, and a hard shoulder. It is the most distinctive rendering of
|
||||
# the four and the one where a flat render looks most obviously wrong.
|
||||
#
|
||||
# This entry does *not* read the in-RAF film simulation tag — that is
|
||||
# FR-DEV-3f, and until it lands every Fujifilm file gets the Provia shape
|
||||
# whatever the camera was set to.
|
||||
- make: Fujifilm
|
||||
model: X-T3
|
||||
points:
|
||||
- [0.00, 0.000]
|
||||
- [0.045, 0.040]
|
||||
- [0.14, 0.215]
|
||||
- [0.47, 0.775]
|
||||
- [1.00, 1.000]
|
||||
@@ -1,752 +0,0 @@
|
||||
//! TRACES: FR-DEV-3e
|
||||
//! Base curves — the per-body rendering that turns a correct exposure into a
|
||||
//! photograph.
|
||||
//!
|
||||
//! # What this is for
|
||||
//!
|
||||
//! A camera matrix gets the *colours* right and leaves the picture flat. Sensor
|
||||
//! data is scene-referred and very nearly linear; a print, a screen and a
|
||||
//! camera's own JPEG are none of those things. Rendering linear data straight
|
||||
//! out is the dcraw default, and FR-DEV-3e names it precisely: "the flat,
|
||||
//! poor-skin-tone rendering characteristic of dcraw defaults, which is the
|
||||
//! documented reason people abandon darktable in the first hour."
|
||||
//!
|
||||
//! The fix is a tone curve applied as part of *reading* the file rather than as
|
||||
//! an edit — a toe, a steep midtone, and a shoulder that rolls highlights off
|
||||
//! instead of clipping them. Every raw converter has one. Adobe calls it the
|
||||
//! camera profile's tone curve, darktable calls it the base curve, and the name
|
||||
//! here follows darktable's because the placement does too: it runs in camera
|
||||
//! RGB, after white balance and the user's adjustments, immediately before the
|
||||
//! conversion out to a working space.
|
||||
//!
|
||||
//! # Why it is not an edit
|
||||
//!
|
||||
//! It never reaches the sidecar and there is no slider for it, for the same
|
||||
//! reason the EXIF orientation is not an edit (FR-DEV-3h): it is a property of
|
||||
//! the body that took the frame, not of what anyone decided about the frame.
|
||||
//! Sidecars are shared between devices and bodies (FR-NC-9), and one camera's
|
||||
//! rendering must not follow an edit onto another camera's file.
|
||||
//!
|
||||
//! # Why it is data
|
||||
//!
|
||||
//! FR-DEV-3e requires the profile database to be "versioned independently of
|
||||
//! the app binary so bodies and curves can be added without a release — and,
|
||||
//! under D8's GPLv3, contributed by users". So the curves live in
|
||||
//! `profiles/base_curves.yaml`, a file that is compiled in as a floor and
|
||||
//! *overridden* by a copy on disk carrying a higher `version:`. Adding a body
|
||||
//! is adding ten numbers to a YAML file; shipping that body to users is
|
||||
//! publishing the file. Neither is a code change and neither needs a release.
|
||||
//!
|
||||
//! See [`load`] for the search path and [`Curves::body`] for the matching.
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::OnceLock;
|
||||
|
||||
/// How many control points a base curve has.
|
||||
///
|
||||
/// Five, which is not a coincidence: it is what the tone curve widget uses
|
||||
/// (`dr_pipeline::ops::curve::POINTS`), so the shader evaluates a profile's
|
||||
/// curve and a photographer's curve through exactly the same spline. A profile
|
||||
/// author and a photographer dragging a point mean the same thing by it, and
|
||||
/// the generated shader carries one implementation rather than two that could
|
||||
/// disagree.
|
||||
pub const POINTS: usize = 5;
|
||||
|
||||
/// TRACES: FR-DEV-3e
|
||||
/// A base curve: five points on a monotone spline through the unit square.
|
||||
///
|
||||
/// `xs` is scene-linear camera RGB, normalised so that 1.0 is the sensor's
|
||||
/// saturation point. `ys` is display-referred linear — *not* gamma-encoded,
|
||||
/// because the sRGB transfer function is applied at the very end of the
|
||||
/// generated shader and applying it twice would wash the image out.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub struct BaseCurve {
|
||||
pub xs: [f32; POINTS],
|
||||
pub ys: [f32; POINTS],
|
||||
}
|
||||
|
||||
impl BaseCurve {
|
||||
/// The curve that does nothing — the identity diagonal.
|
||||
///
|
||||
/// What an unrecognised body gets if the database carries no default, and
|
||||
/// what a JPEG gets always: an already-rendered image must not be rendered
|
||||
/// a second time.
|
||||
pub const IDENTITY: Self = Self {
|
||||
xs: [0.0, 0.25, 0.5, 0.75, 1.0],
|
||||
ys: [0.0, 0.25, 0.5, 0.75, 1.0],
|
||||
};
|
||||
|
||||
/// Whether this curve would leave the image alone.
|
||||
///
|
||||
/// The shader is told to skip the stage entirely when it would, so an
|
||||
/// unprofiled body costs a branch that is uniform across the dispatch
|
||||
/// rather than a spline evaluation per channel per pixel.
|
||||
pub fn is_identity(&self) -> bool {
|
||||
self.xs
|
||||
.iter()
|
||||
.zip(self.ys.iter())
|
||||
.all(|(x, y)| (x - y).abs() < 1e-6)
|
||||
}
|
||||
|
||||
/// Build from raw pairs, rejecting anything that is not a curve.
|
||||
///
|
||||
/// A profile file is data a user may have edited, so this is the boundary
|
||||
/// where "ten numbers" becomes "a curve": the x coordinates must increase,
|
||||
/// the y coordinates must not decrease, and both must lie in the unit
|
||||
/// square. A non-monotone x sends the spline's span search backwards and
|
||||
/// divides by a negative width; a decreasing y inverts tones locally,
|
||||
/// which reads as a dark halo through smooth gradients rather than as a
|
||||
/// bad profile.
|
||||
///
|
||||
/// Endpoints are not forced to (0,0) and (1,1). A curve that lifts black
|
||||
/// slightly, or that places the shoulder below white, is a legitimate
|
||||
/// rendering choice and several bodies make it.
|
||||
pub fn from_points(points: &[[f32; 2]]) -> Option<Self> {
|
||||
if points.len() != POINTS {
|
||||
return None;
|
||||
}
|
||||
let mut xs = [0.0f32; POINTS];
|
||||
let mut ys = [0.0f32; POINTS];
|
||||
for (i, p) in points.iter().enumerate() {
|
||||
if !p[0].is_finite() || !p[1].is_finite() {
|
||||
return None;
|
||||
}
|
||||
if !(0.0..=1.0).contains(&p[0]) || !(0.0..=1.0).contains(&p[1]) {
|
||||
return None;
|
||||
}
|
||||
xs[i] = p[0];
|
||||
ys[i] = p[1];
|
||||
}
|
||||
for i in 1..POINTS {
|
||||
// Strictly increasing in x — the spline divides by the span width.
|
||||
if xs[i] <= xs[i - 1] {
|
||||
return None;
|
||||
}
|
||||
// Non-decreasing in y. Flat is allowed: a curve that holds a
|
||||
// highlight range at white is clipping deliberately.
|
||||
if ys[i] < ys[i - 1] {
|
||||
return None;
|
||||
}
|
||||
}
|
||||
Some(Self { xs, ys })
|
||||
}
|
||||
}
|
||||
|
||||
/// One body's entry in the database.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct BodyCurve {
|
||||
/// The manufacturer, as the file writes it — "Canon", "NIKON CORPORATION".
|
||||
pub make: String,
|
||||
/// The model, as the file writes it — "EOS 6D", "ILCE-7M3".
|
||||
pub model: String,
|
||||
pub curve: BaseCurve,
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-3e
|
||||
/// The base curve database.
|
||||
///
|
||||
/// Versioned as a whole rather than per body, because that is the unit a user
|
||||
/// downloads and the unit that has to beat the built-in copy. See [`load`].
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub struct Curves {
|
||||
version: u32,
|
||||
default: Option<BaseCurve>,
|
||||
bodies: Vec<BodyCurve>,
|
||||
}
|
||||
|
||||
impl Curves {
|
||||
/// TRACES: FR-DEV-3e
|
||||
/// The curve to render a frame from this body with.
|
||||
///
|
||||
/// Falls back, in order, to the database's `default:` and then to the
|
||||
/// identity. **The default is deliberately not the identity**: an
|
||||
/// unrecognised body rendered flat is the failure this requirement exists
|
||||
/// to prevent, and a gentle, conservative curve is much closer to right for
|
||||
/// every body than no curve is for any of them. A body with its own entry
|
||||
/// gets that instead.
|
||||
///
|
||||
/// # What "this body" has to survive
|
||||
///
|
||||
/// The same camera names itself three ways depending on which program last
|
||||
/// touched the file. A native NEF says make "NIKON CORPORATION", model
|
||||
/// "NIKON Z 6"; rawler's own database cleans that to "Nikon" and "Z 6"; an
|
||||
/// Adobe-converted DNG keeps the uncleaned pair. A database that had to
|
||||
/// spell every variant would go stale the first time a maker changed its
|
||||
/// mind about its own name, so the matching does the folding instead:
|
||||
///
|
||||
/// - Case, punctuation and runs of whitespace are flattened, so
|
||||
/// "ILCE-7M3", "ILCE 7M3" and "ilce-7m3" are one body.
|
||||
/// - The make is compared on its **first word only**. Every maker's
|
||||
/// trailing corporate boilerplate — "CORPORATION", "IMAGING CORP" — is
|
||||
/// noise, and no two camera manufacturers share a first word.
|
||||
/// - The model is tried both as written and with a leading copy of the
|
||||
/// make removed, which is what lets one "Canon"/"EOS 6D" entry cover
|
||||
/// "Canon EOS 6D" as well.
|
||||
pub fn body(&self, make: &str, model: &str) -> BaseCurve {
|
||||
let (make, model) = (make_key(make), normalise(model));
|
||||
// The model with a leading copy of the maker's name removed.
|
||||
let bare = model.strip_prefix(&format!("{make} ")).unwrap_or(&model);
|
||||
|
||||
self.bodies
|
||||
.iter()
|
||||
.find(|b| {
|
||||
let entry_model = normalise(&b.model);
|
||||
make_key(&b.make) == make && (entry_model == model || entry_model == bare)
|
||||
})
|
||||
.map(|b| b.curve)
|
||||
.or(self.default)
|
||||
.unwrap_or(BaseCurve::IDENTITY)
|
||||
}
|
||||
|
||||
/// The database version. Higher wins; see [`load`].
|
||||
pub fn version(&self) -> u32 {
|
||||
self.version
|
||||
}
|
||||
|
||||
/// How many bodies have their own curve, excluding the default.
|
||||
pub fn len(&self) -> usize {
|
||||
self.bodies.len()
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.bodies.is_empty()
|
||||
}
|
||||
|
||||
/// Parse a database from YAML.
|
||||
///
|
||||
/// Entries that are not curves are dropped with a warning rather than
|
||||
/// failing the parse. A user-contributed file with one bad body should
|
||||
/// cost that body's rendering, not every body's — and the alternative is an
|
||||
/// application that will not open a photograph because somebody typed a
|
||||
/// comma.
|
||||
pub fn parse(yaml: &str) -> Result<Self, String> {
|
||||
let file: File = serde_norway::from_str(yaml).map_err(|e| e.to_string())?;
|
||||
|
||||
let default = file.default.and_then(|d| {
|
||||
BaseCurve::from_points(&d.points).or_else(|| {
|
||||
log::warn!("base curves: the default entry is not a monotone curve; ignoring it");
|
||||
None
|
||||
})
|
||||
});
|
||||
|
||||
let bodies = file
|
||||
.bodies
|
||||
.into_iter()
|
||||
.filter_map(|b| match BaseCurve::from_points(&b.points) {
|
||||
Some(curve) => Some(BodyCurve {
|
||||
make: b.make,
|
||||
model: b.model,
|
||||
curve,
|
||||
}),
|
||||
None => {
|
||||
log::warn!(
|
||||
"base curves: {} {} is not a monotone curve; ignoring it",
|
||||
b.make,
|
||||
b.model
|
||||
);
|
||||
None
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
Ok(Self {
|
||||
version: file.version,
|
||||
default,
|
||||
bodies,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// The copy that ships inside the binary.
|
||||
///
|
||||
/// A floor, not the answer: [`load`] prefers a newer file on disk. Compiled in
|
||||
/// so that a fresh install with no profile directory — and every Android build,
|
||||
/// where there is no such directory to speak of — still renders properly.
|
||||
const BUILT_IN: &str = include_str!("../profiles/base_curves.yaml");
|
||||
|
||||
/// TRACES: FR-DEV-3e
|
||||
/// The base curve database, loaded once.
|
||||
///
|
||||
/// # The search path, and why it is a version comparison
|
||||
///
|
||||
/// 1. `$DARKROOM_PROFILES`, a directory, when set. The escape hatch: a profile
|
||||
/// author iterating on a curve points this at their working copy and does
|
||||
/// not have to install anything.
|
||||
/// 2. `$XDG_DATA_HOME/darkroom/profiles/`, else `$HOME/.local/share/darkroom/profiles/`.
|
||||
/// The same base directory the catalog uses, chosen there for the same
|
||||
/// reason — it is data, not cache, and must survive a storage sweep.
|
||||
/// 3. The copy compiled into the binary.
|
||||
///
|
||||
/// The first file that parses *and carries a higher `version:` than the
|
||||
/// built-in copy* wins. The version check is the whole mechanism the
|
||||
/// requirement asks for, and it runs in both directions:
|
||||
///
|
||||
/// - A downloaded pack at version 7 supersedes a binary shipping version 3, so
|
||||
/// a body added after the release renders correctly with no release.
|
||||
/// - A stale pack at version 2 does **not** supersede a binary shipping version
|
||||
/// 3, so upgrading the application cannot silently lose curves to a file
|
||||
/// somebody downloaded a year ago and forgot.
|
||||
///
|
||||
/// Failures are warnings, never errors. A malformed profile file must cost the
|
||||
/// user their curves, not their photographs.
|
||||
pub fn load() -> &'static Curves {
|
||||
static LOADED: OnceLock<Curves> = OnceLock::new();
|
||||
LOADED.get_or_init(|| {
|
||||
let built_in = Curves::parse(BUILT_IN).unwrap_or_else(|e| {
|
||||
// Unreachable in a build that ran its tests — `the_shipped_database_parses`
|
||||
// asserts exactly this — but a panic here would mean an
|
||||
// application that cannot open a photograph because of a typo in a
|
||||
// data file, which is never the right trade.
|
||||
log::error!("base curves: the built-in database does not parse: {e}");
|
||||
Curves {
|
||||
version: 0,
|
||||
default: None,
|
||||
bodies: Vec::new(),
|
||||
}
|
||||
});
|
||||
|
||||
choose(built_in, &search_path())
|
||||
})
|
||||
}
|
||||
|
||||
/// The version comparison, separated from where the directories come from.
|
||||
///
|
||||
/// Split out so it can be tested against real files in a real directory
|
||||
/// without the process-wide `OnceLock` and the environment `load` reads. The
|
||||
/// rule this implements is the whole of what FR-DEV-3e asks for, so it is
|
||||
/// worth being able to state it as a test rather than as a comment.
|
||||
fn choose(built_in: Curves, dirs: &[PathBuf]) -> Curves {
|
||||
for dir in dirs {
|
||||
let path = dir.join("base_curves.yaml");
|
||||
let Ok(text) = std::fs::read_to_string(&path) else {
|
||||
continue;
|
||||
};
|
||||
match Curves::parse(&text) {
|
||||
Ok(external) if external.version > built_in.version => {
|
||||
log::info!(
|
||||
"base curves: using {} (version {}, {} bodies) over the built-in version {}",
|
||||
path.display(),
|
||||
external.version,
|
||||
external.len(),
|
||||
built_in.version
|
||||
);
|
||||
return external;
|
||||
}
|
||||
Ok(external) => log::info!(
|
||||
"base curves: ignoring {} at version {}; the built-in database is version {}",
|
||||
path.display(),
|
||||
external.version,
|
||||
built_in.version
|
||||
),
|
||||
Err(e) => log::warn!("base curves: {} does not parse: {e}", path.display()),
|
||||
}
|
||||
}
|
||||
built_in
|
||||
}
|
||||
|
||||
/// TRACES: FR-DEV-3e
|
||||
/// The curve for a body, from the loaded database.
|
||||
///
|
||||
/// The one call site the decoder needs; everything above is reachable for
|
||||
/// tests and for a future profile editor.
|
||||
pub fn for_body(make: &str, model: &str) -> BaseCurve {
|
||||
load().body(make, model)
|
||||
}
|
||||
|
||||
/// Directories that may hold a `base_curves.yaml`, most specific first.
|
||||
fn search_path() -> Vec<PathBuf> {
|
||||
let mut dirs = Vec::new();
|
||||
if let Some(explicit) = std::env::var_os("DARKROOM_PROFILES") {
|
||||
dirs.push(PathBuf::from(explicit));
|
||||
}
|
||||
// The same resolution `dr_ui::library::catalog_path` uses, and for the
|
||||
// same reason: this is data a user may have installed, not a cache. It is
|
||||
// duplicated rather than shared because `dr-decode` sits far below the UI
|
||||
// and must not acquire a dependency on it to find a directory.
|
||||
let base = std::env::var_os("XDG_DATA_HOME")
|
||||
.map(PathBuf::from)
|
||||
.or_else(|| std::env::var_os("HOME").map(|h| Path::new(&h).join(".local/share")));
|
||||
if let Some(base) = base {
|
||||
dirs.push(base.join("darkroom").join("profiles"));
|
||||
}
|
||||
dirs
|
||||
}
|
||||
|
||||
/// A manufacturer's first word, folded.
|
||||
///
|
||||
/// "NIKON CORPORATION", "Nikon" and "nikon" all become `NIKON`. The corporate
|
||||
/// suffixes are not information — they appear or not depending on whether the
|
||||
/// file went through a DNG converter — and no two camera manufacturers share a
|
||||
/// first word, so nothing is lost by dropping them.
|
||||
fn make_key(s: &str) -> String {
|
||||
normalise(s)
|
||||
.split(' ')
|
||||
.next()
|
||||
.unwrap_or_default()
|
||||
.to_string()
|
||||
}
|
||||
|
||||
/// Fold a make or model into something two files can agree on.
|
||||
///
|
||||
/// Upper-cased, with every run of non-alphanumeric characters collapsed to one
|
||||
/// space and the ends trimmed, so that "ILCE-7M3", "ILCE 7M3" and "ilce-7m3"
|
||||
/// become one.
|
||||
fn normalise(s: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len());
|
||||
let mut pending_space = false;
|
||||
for c in s.chars() {
|
||||
if c.is_ascii_alphanumeric() {
|
||||
if pending_space && !out.is_empty() {
|
||||
out.push(' ');
|
||||
}
|
||||
pending_space = false;
|
||||
out.push(c.to_ascii_uppercase());
|
||||
} else {
|
||||
pending_space = true;
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// ---- The on-disk shape, kept apart from the in-memory one ----------------
|
||||
//
|
||||
// Deliberately separate types. The file is data a user edits and is allowed to
|
||||
// be wrong; `Curves` is a parsed database whose every entry is known to be a
|
||||
// monotone curve. Deriving `Deserialize` on `BaseCurve` directly would delete
|
||||
// that boundary and let an unchecked five-point array reach the shader.
|
||||
//
|
||||
// Unknown fields are **accepted**, which is not laziness. The database is
|
||||
// versioned independently of the binary and moves in both directions: a pack
|
||||
// published after this release may carry keys this build has never heard of —
|
||||
// a hue twist, a look table (FR-DEV-3f) — and it must still deliver its curves
|
||||
// to an older DarkRoom rather than failing to parse and leaving every body
|
||||
// flat. `deny_unknown_fields` would trade that for a diagnostic nobody needs.
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct File {
|
||||
version: u32,
|
||||
#[serde(default)]
|
||||
default: Option<Entry>,
|
||||
#[serde(default)]
|
||||
bodies: Vec<BodyEntry>,
|
||||
}
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct Entry {
|
||||
points: Vec<[f32; 2]>,
|
||||
}
|
||||
|
||||
#[derive(serde::Deserialize)]
|
||||
struct BodyEntry {
|
||||
make: String,
|
||||
model: String,
|
||||
points: Vec<[f32; 2]>,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn the_shipped_database_parses_and_carries_a_default() {
|
||||
// The one test that must never be allowed to fail quietly: `load`
|
||||
// degrades to an empty database rather than panicking, so without this
|
||||
// a typo in the YAML would ship as "every photograph renders flat"
|
||||
// rather than as a build failure.
|
||||
let curves = Curves::parse(BUILT_IN).expect("the shipped database parses");
|
||||
assert!(curves.version() >= 1);
|
||||
assert!(!curves.is_empty(), "the database ships bodies");
|
||||
assert!(
|
||||
!curves.body("Nobody", "Nothing").is_identity(),
|
||||
"an unknown body must still get the default rendering"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_shipped_curve_lifts_the_midtones_and_rolls_the_highlights() {
|
||||
// What makes a base curve a base curve rather than a decoration. If a
|
||||
// shipped curve failed either half it would be a worse rendering than
|
||||
// the flat one it replaced, which is the one outcome forbidden.
|
||||
let curves = Curves::parse(BUILT_IN).expect("parses");
|
||||
let all = curves
|
||||
.bodies
|
||||
.iter()
|
||||
.map(|b| (format!("{} {}", b.make, b.model), b.curve))
|
||||
.chain(curves.default.map(|c| ("default".to_string(), c)));
|
||||
|
||||
for (name, curve) in all {
|
||||
// The midtone point sits above the diagonal: a linear midtone is
|
||||
// roughly a stop and a half darker than any camera renders it.
|
||||
let mid = 2;
|
||||
assert!(
|
||||
curve.ys[mid] > curve.xs[mid],
|
||||
"{name} does not lift its midtones ({} -> {})",
|
||||
curve.xs[mid],
|
||||
curve.ys[mid]
|
||||
);
|
||||
// And the last span is shallower than the one before it, which is
|
||||
// what a shoulder *is*. Without one the curve clips highlights
|
||||
// harder than the linear rendering did.
|
||||
let slope =
|
||||
|i: usize| (curve.ys[i + 1] - curve.ys[i]) / (curve.xs[i + 1] - curve.xs[i]);
|
||||
assert!(
|
||||
slope(POINTS - 2) < slope(POINTS - 3),
|
||||
"{name} has no highlight shoulder"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_curve_that_is_not_monotone_is_refused() {
|
||||
// The profile file is user-editable, so this is a real boundary and
|
||||
// not a formality. A decreasing y inverts tones locally and shows up
|
||||
// as a dark halo in a gradient, which reads as a rendering fault
|
||||
// rather than as a bad profile.
|
||||
assert_eq!(
|
||||
BaseCurve::from_points(&[[0.0, 0.0], [0.25, 0.4], [0.5, 0.3], [0.75, 0.8], [1.0, 1.0]]),
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_curve_whose_x_does_not_advance_is_refused() {
|
||||
// The spline divides by the span width; a repeated x is a division by
|
||||
// zero in the shader, which is a NaN pixel rather than an error.
|
||||
assert_eq!(
|
||||
BaseCurve::from_points(&[
|
||||
[0.0, 0.0],
|
||||
[0.25, 0.3],
|
||||
[0.25, 0.5],
|
||||
[0.75, 0.8],
|
||||
[1.0, 1.0]
|
||||
]),
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_curve_of_the_wrong_length_is_refused() {
|
||||
assert_eq!(BaseCurve::from_points(&[[0.0, 0.0], [1.0, 1.0]]), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn values_outside_the_unit_square_are_refused() {
|
||||
// The shader clamps its output at the very end anyway, but a control
|
||||
// point above 1.0 would put the shoulder outside the range the curve
|
||||
// is defined over and silently flatten everything below it.
|
||||
assert_eq!(
|
||||
BaseCurve::from_points(&[[0.0, 0.0], [0.25, 0.3], [0.5, 1.4], [0.75, 1.5], [1.0, 1.6]]),
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_body_with_its_own_entry_beats_the_default() {
|
||||
let curves = Curves::parse(
|
||||
"version: 2
|
||||
default:
|
||||
points: [[0.0, 0.0], [0.25, 0.3], [0.5, 0.6], [0.75, 0.85], [1.0, 1.0]]
|
||||
bodies:
|
||||
- make: Canon
|
||||
model: EOS 6D
|
||||
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
|
||||
",
|
||||
)
|
||||
.expect("parses");
|
||||
|
||||
assert_eq!(curves.body("Canon", "EOS 6D").ys[1], 0.35);
|
||||
assert_eq!(curves.body("Canon", "EOS 5D").ys[1], 0.30);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_make_may_be_repeated_in_the_model() {
|
||||
// Canon writes "Canon" as the make and "Canon EOS 6D" as the model;
|
||||
// rawler's cleaned strings drop the repetition and both reach here.
|
||||
// One entry has to cover both or half the files on a card miss.
|
||||
let curves = Curves::parse(
|
||||
"version: 1
|
||||
bodies:
|
||||
- make: Canon
|
||||
model: EOS 6D
|
||||
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
|
||||
",
|
||||
)
|
||||
.expect("parses");
|
||||
|
||||
assert_eq!(curves.body("Canon", "Canon EOS 6D").ys[1], 0.35);
|
||||
assert_eq!(curves.body("Canon", "EOS 6D").ys[1], 0.35);
|
||||
assert_eq!(curves.body("CANON", "eos 6d").ys[1], 0.35);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_corporate_suffix_does_not_hide_a_body() {
|
||||
// The same Z 6 arrives as "Nikon"/"Z 6" from rawler's camera database
|
||||
// and as "NIKON CORPORATION"/"NIKON Z 6" from a DNG converted out of
|
||||
// the same file. Both must find the entry, or converting a file to
|
||||
// DNG would silently change how it renders.
|
||||
let curves = Curves::parse(
|
||||
"version: 1
|
||||
bodies:
|
||||
- make: Nikon
|
||||
model: Z 6
|
||||
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
|
||||
",
|
||||
)
|
||||
.expect("parses");
|
||||
|
||||
assert_eq!(curves.body("Nikon", "Z 6").ys[1], 0.35);
|
||||
assert_eq!(curves.body("NIKON CORPORATION", "NIKON Z 6").ys[1], 0.35);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn punctuation_and_spacing_do_not_decide_whether_a_body_is_known() {
|
||||
let curves = Curves::parse(
|
||||
"version: 1
|
||||
bodies:
|
||||
- make: Sony
|
||||
model: ILCE-7M3
|
||||
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
|
||||
",
|
||||
)
|
||||
.expect("parses");
|
||||
|
||||
assert_eq!(curves.body("SONY", "ILCE 7M3").ys[1], 0.35);
|
||||
assert_eq!(curves.body("sony", "ilce-7m3").ys[1], 0.35);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn one_bad_entry_does_not_cost_the_rest() {
|
||||
// A user-contributed file with one typo should cost that body's
|
||||
// rendering, not every body's.
|
||||
let curves = Curves::parse(
|
||||
"version: 1
|
||||
bodies:
|
||||
- make: Broken
|
||||
model: Body
|
||||
points: [[0.0, 0.0], [0.25, 0.9], [0.5, 0.1], [0.75, 0.9], [1.0, 1.0]]
|
||||
- make: Canon
|
||||
model: EOS 6D
|
||||
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
|
||||
",
|
||||
)
|
||||
.expect("parses");
|
||||
|
||||
assert_eq!(curves.len(), 1);
|
||||
assert_eq!(curves.body("Canon", "EOS 6D").ys[1], 0.35);
|
||||
assert!(curves.body("Broken", "Body").is_identity());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_pack_from_the_future_still_delivers_its_curves() {
|
||||
// The database is versioned independently of the binary, so a pack
|
||||
// published after this build may carry keys this build has never heard
|
||||
// of. It must still hand over the curves it does understand — failing
|
||||
// the parse would leave every body flat, which is the exact failure
|
||||
// FR-DEV-3e exists to prevent, delivered by the mechanism meant to
|
||||
// prevent it.
|
||||
let curves = Curves::parse(
|
||||
"version: 9
|
||||
look_table: ambitious
|
||||
bodies:
|
||||
- make: Canon
|
||||
model: EOS 6D
|
||||
hue_twist: [1, 2, 3]
|
||||
points: [[0.0, 0.0], [0.25, 0.35], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
|
||||
",
|
||||
)
|
||||
.expect("an unfamiliar key must not fail the parse");
|
||||
|
||||
assert_eq!(curves.version(), 9);
|
||||
assert_eq!(curves.body("Canon", "EOS 6D").ys[1], 0.35);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unknown_body_with_no_default_gets_the_identity() {
|
||||
// Graceful fallback, stated as a property: never worse than a flat
|
||||
// render, and never a curve tuned for somebody else's sensor when the
|
||||
// database declines to offer one.
|
||||
let curves = Curves::parse("version: 1\nbodies: []\n").expect("parses");
|
||||
assert!(curves.body("Nobody", "Nothing").is_identity());
|
||||
}
|
||||
|
||||
/// A directory holding one `base_curves.yaml`, unique to the caller.
|
||||
fn a_pack_dir(name: &str, yaml: &str) -> PathBuf {
|
||||
let dir = std::env::temp_dir().join(format!("darkroom-base-curves-{name}"));
|
||||
let _ = std::fs::remove_dir_all(&dir);
|
||||
std::fs::create_dir_all(&dir).expect("a writable temp directory");
|
||||
std::fs::write(dir.join("base_curves.yaml"), yaml).expect("write");
|
||||
dir
|
||||
}
|
||||
|
||||
const A_CANON_ENTRY: &str = "bodies:
|
||||
- make: Canon
|
||||
model: EOS 6D
|
||||
points: [[0.0, 0.0], [0.25, 0.42], [0.5, 0.7], [0.75, 0.9], [1.0, 1.0]]
|
||||
";
|
||||
|
||||
#[test]
|
||||
fn a_newer_pack_on_disk_supersedes_the_built_in_database() {
|
||||
// **This is the requirement.** FR-DEV-3e asks for a profile database
|
||||
// versioned independently of the app binary "so bodies and curves can
|
||||
// be added without a release". A file with a higher version, dropped
|
||||
// in the profile directory, is what that means in practice.
|
||||
let built_in = Curves::parse(BUILT_IN).expect("parses");
|
||||
let newer = format!("version: {}\n{A_CANON_ENTRY}", built_in.version() + 1);
|
||||
let dir = a_pack_dir("newer", &newer);
|
||||
|
||||
let chosen = choose(built_in.clone(), &[dir]);
|
||||
assert_eq!(chosen.version(), built_in.version() + 1);
|
||||
assert_eq!(chosen.body("Canon", "EOS 6D").ys[1], 0.42);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_stale_pack_does_not_survive_an_upgrade() {
|
||||
// The other direction, and the one that protects the user. Somebody
|
||||
// downloads a pack, a release later ships better curves for the same
|
||||
// bodies, and the forgotten file must not quietly hold the application
|
||||
// back at last year's rendering.
|
||||
let built_in = Curves::parse(BUILT_IN).expect("parses");
|
||||
let stale = format!("version: {}\n{A_CANON_ENTRY}", built_in.version());
|
||||
let dir = a_pack_dir("stale", &stale);
|
||||
|
||||
let chosen = choose(built_in.clone(), &[dir]);
|
||||
assert_eq!(chosen.version(), built_in.version());
|
||||
assert_ne!(
|
||||
chosen.body("Canon", "EOS 6D").ys[1],
|
||||
0.42,
|
||||
"an equal version must not displace the built-in database"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_broken_pack_costs_the_curves_and_not_the_photographs() {
|
||||
// A malformed profile file must degrade to the built-in database, not
|
||||
// to an error. The user came here to look at a photograph.
|
||||
let built_in = Curves::parse(BUILT_IN).expect("parses");
|
||||
let dir = a_pack_dir("broken", "version: [this is not a number\n");
|
||||
|
||||
let chosen = choose(built_in.clone(), &[dir]);
|
||||
assert_eq!(chosen.version(), built_in.version());
|
||||
assert_eq!(chosen.len(), built_in.len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_directory_with_no_pack_in_it_is_simply_skipped() {
|
||||
// The ordinary case on every machine: the search path exists, the file
|
||||
// does not. It must not be a warning, an error, or a slow path.
|
||||
let built_in = Curves::parse(BUILT_IN).expect("parses");
|
||||
let missing = std::env::temp_dir().join("darkroom-base-curves-nothing-here");
|
||||
let _ = std::fs::remove_dir_all(&missing);
|
||||
|
||||
assert_eq!(choose(built_in.clone(), &[missing]), built_in);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_identity_is_recognised_as_doing_nothing() {
|
||||
assert!(BaseCurve::IDENTITY.is_identity());
|
||||
assert!(!Curves::parse(BUILT_IN)
|
||||
.expect("parses")
|
||||
.body("Canon", "EOS 6D")
|
||||
.is_identity());
|
||||
}
|
||||
}
|
||||
@@ -1,52 +0,0 @@
|
||||
/// TRACES: FR-RAW-4 | NFR-SEC-1
|
||||
/// Failures from decoding.
|
||||
///
|
||||
/// Per FR-RAW-4 a malformed file must not abort a batch, so these are always
|
||||
/// returned rather than panicking — and the decode path is the one place
|
||||
/// untrusted input arrives (NFR-SEC-1).
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum DecodeError {
|
||||
#[error("read failed: {0}")]
|
||||
Read(String),
|
||||
|
||||
#[error("unsupported or unrecognised format: {0}")]
|
||||
Unsupported(String),
|
||||
|
||||
#[error("decode failed: {0}")]
|
||||
Decode(String),
|
||||
|
||||
#[error("metadata unavailable: {0}")]
|
||||
Metadata(String),
|
||||
|
||||
#[error("no embedded preview in this file")]
|
||||
NoPreview,
|
||||
|
||||
#[error("embedded preview is corrupt: {0}")]
|
||||
CorruptPreview(String),
|
||||
}
|
||||
|
||||
impl DecodeError {
|
||||
/// Whether a fallback path might still produce an image.
|
||||
///
|
||||
/// A missing preview is not a failure to display the file — it means fall
|
||||
/// through to full decode (FR-CULL-2, M-11).
|
||||
pub fn has_fallback(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
DecodeError::NoPreview | DecodeError::CorruptPreview(_)
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn preview_failures_fall_through_rather_than_failing() {
|
||||
assert!(DecodeError::NoPreview.has_fallback());
|
||||
assert!(DecodeError::CorruptPreview("truncated".into()).has_fallback());
|
||||
// A genuinely unsupported file has nowhere to fall through to.
|
||||
assert!(!DecodeError::Unsupported("unknown".into()).has_fallback());
|
||||
}
|
||||
}
|
||||
@@ -1,408 +0,0 @@
|
||||
//! Embedded preview extraction — the fast display path.
|
||||
//!
|
||||
//! Every RAW container carries one or more JPEG previews, often at or near
|
||||
//! full resolution. Extracting one costs a fraction of a full decode, and is
|
||||
//! what makes culling feel instant (FR-CULL-1, NFR-P13: 50 ms per image).
|
||||
//!
|
||||
//! It is also what makes remote browsing viable: fetching ~1-3 MB of preview
|
||||
//! from an 80 MB file over WebDAV is the difference between usable and not on
|
||||
//! mobile data (FR-NC-3).
|
||||
|
||||
use crate::DecodeError;
|
||||
|
||||
/// How much of a file header to read when locating a preview.
|
||||
///
|
||||
/// Enough to cover the IFD structure of the TIFF-derived formats. Sized for
|
||||
/// remote range requests, where every byte costs.
|
||||
pub const PREVIEW_PROBE_BYTES: u64 = 256 * 1024;
|
||||
|
||||
/// A decoded preview image, RGBA8.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Preview {
|
||||
pub width: u32,
|
||||
pub height: u32,
|
||||
/// Tightly packed RGBA, 4 bytes per pixel.
|
||||
pub rgba: Vec<u8>,
|
||||
}
|
||||
|
||||
impl Preview {
|
||||
/// TRACES: FR-DEV-3h
|
||||
/// Turn the pixels the right way up, in place.
|
||||
///
|
||||
/// Every path that shows a preview without the GPU needs this: the grid's
|
||||
/// thumbnails, and the read-only fallback develop shows when no decoder
|
||||
/// could open the file. An embedded preview is written in the sensor's
|
||||
/// orientation, not the photograph's, so a phone or a camera held sideways
|
||||
/// fills the grid with frames on their side until this runs.
|
||||
///
|
||||
/// Done before [`Self::downscale_to`] would be wasteful and after it is
|
||||
/// not: a quarter turn is a permutation, so it costs the same either way,
|
||||
/// and doing it on the smaller buffer moves a fraction of the bytes.
|
||||
///
|
||||
/// The turn itself is [`dr_types::Orientation::into_shown`], which every
|
||||
/// other consumer of an orientation in this codebase also goes through.
|
||||
/// That is deliberate: a hand-written permutation per caller is how two of
|
||||
/// them come to disagree, and a disagreement here shows as a thumbnail
|
||||
/// facing the other way from the develop view.
|
||||
pub fn apply_orientation(&mut self, orientation: dr_types::Orientation) {
|
||||
let (rgba, dw, dh) = orientation.into_shown(&self.rgba, self.width, self.height, 4);
|
||||
self.rgba = rgba;
|
||||
self.width = dw;
|
||||
self.height = dh;
|
||||
}
|
||||
|
||||
/// Downscale in place to fit within `max_dim` on the long edge.
|
||||
///
|
||||
/// A 5472x3648 preview is 79.8 MB of RGBA — far more than a grid cell or
|
||||
/// even a 4K viewport needs, and enough to exhaust a phone's budget after
|
||||
/// a handful of images (NFR-RES-1). Box-filtered rather than nearest, so
|
||||
/// downscaled thumbnails do not alias.
|
||||
pub fn downscale_to(&mut self, max_dim: u32) {
|
||||
let longest = self.width.max(self.height);
|
||||
if longest <= max_dim || longest == 0 {
|
||||
return;
|
||||
}
|
||||
let scale = max_dim as f32 / longest as f32;
|
||||
let (nw, nh) = (
|
||||
((self.width as f32 * scale).round() as u32).max(1),
|
||||
((self.height as f32 * scale).round() as u32).max(1),
|
||||
);
|
||||
|
||||
let mut out = vec![0u8; (nw as usize) * (nh as usize) * 4];
|
||||
let x_ratio = self.width as f32 / nw as f32;
|
||||
let y_ratio = self.height as f32 / nh as f32;
|
||||
|
||||
for y in 0..nh {
|
||||
let y0 = (y as f32 * y_ratio) as u32;
|
||||
let y1 = (((y + 1) as f32 * y_ratio) as u32)
|
||||
.min(self.height)
|
||||
.max(y0 + 1);
|
||||
for x in 0..nw {
|
||||
let x0 = (x as f32 * x_ratio) as u32;
|
||||
let x1 = (((x + 1) as f32 * x_ratio) as u32)
|
||||
.min(self.width)
|
||||
.max(x0 + 1);
|
||||
|
||||
let (mut r, mut g, mut b, mut n) = (0u32, 0u32, 0u32, 0u32);
|
||||
for sy in y0..y1 {
|
||||
for sx in x0..x1 {
|
||||
let i = ((sy * self.width + sx) * 4) as usize;
|
||||
r += self.rgba[i] as u32;
|
||||
g += self.rgba[i + 1] as u32;
|
||||
b += self.rgba[i + 2] as u32;
|
||||
n += 1;
|
||||
}
|
||||
}
|
||||
let n = n.max(1);
|
||||
let o = ((y * nw + x) * 4) as usize;
|
||||
out[o] = (r / n) as u8;
|
||||
out[o + 1] = (g / n) as u8;
|
||||
out[o + 2] = (b / n) as u8;
|
||||
out[o + 3] = 255;
|
||||
}
|
||||
}
|
||||
|
||||
self.rgba = out;
|
||||
self.width = nw;
|
||||
self.height = nh;
|
||||
}
|
||||
|
||||
/// Whether this is large enough to be worth displaying at `target`.
|
||||
///
|
||||
/// Some bodies embed thumbnails only a few hundred pixels wide — Sony is
|
||||
/// the documented case. Displaying one where a larger render is wanted
|
||||
/// shows a soft image the user discovers only on zoom, so the caller
|
||||
/// should background-render instead (M-11).
|
||||
pub fn is_useful_at(&self, target: u32) -> bool {
|
||||
self.width.max(self.height) >= target
|
||||
}
|
||||
}
|
||||
|
||||
/// TRACES: FR-CULL-1 | NFR-P13
|
||||
/// Which embedded image to extract.
|
||||
///
|
||||
/// Containers carry several at different sizes, and decoding the
|
||||
/// full-resolution one to fill a grid cell is pure waste.
|
||||
///
|
||||
/// **Measured caveat (rawler 0.7.2):** the CR2 decoder implements only
|
||||
/// `full_image`; `thumbnail_image` and `preview_image` are unimplemented trait
|
||||
/// defaults returning `None`. So on Canon CR2 every rung currently resolves to
|
||||
/// the full-resolution JPEG at ~250 ms — 5× over NFR-P13's 50 ms budget.
|
||||
///
|
||||
/// Three ways out, in increasing cost: extract the smaller IFD ourselves
|
||||
/// (CR2 carries a 160×120 thumbnail and a ~1620×1080 preview in IFD1/IFD2),
|
||||
/// contribute the methods upstream, or cache a downscaled proxy on first
|
||||
/// sight. The ladder is written now so that fixing it is a decoder change
|
||||
/// rather than a change to every caller.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum PreviewSize {
|
||||
/// Smallest available. Grid cells and rapid culling.
|
||||
Thumbnail,
|
||||
/// Mid-sized where the container has one. Single-image view.
|
||||
Screen,
|
||||
/// Largest available, usually full sensor resolution. Only where the
|
||||
/// display genuinely needs it.
|
||||
Full,
|
||||
}
|
||||
|
||||
/// TRACES: FR-CULL-2 | FR-NC-3 | M-10
|
||||
/// Extract and decode an embedded preview at the requested size.
|
||||
///
|
||||
/// Takes bytes rather than a reader, because the caller usually has them
|
||||
/// already: a range read locally, or a `Range:` request remotely. Forcing a
|
||||
/// `Read + Seek` here would push remote callers into buffering the whole file.
|
||||
///
|
||||
/// Falls through the ladder — a container without the requested size yields
|
||||
/// the next available rather than failing (FR-CULL-2).
|
||||
///
|
||||
/// Returns [`DecodeError::NoPreview`] where there is none at all: a
|
||||
/// fall-through signal, not a failure (see [`DecodeError::has_fallback`]).
|
||||
pub fn extract_preview(bytes: &[u8], size: PreviewSize) -> Result<Preview, DecodeError> {
|
||||
use rawler::rawsource::RawSource;
|
||||
|
||||
// A plain JPEG *is* its own preview — rawler has no decoder for one, and
|
||||
// a mixed folder must display sensibly (M-9).
|
||||
if bytes.starts_with(&[0xFF, 0xD8, 0xFF]) {
|
||||
return decode_jpeg(bytes);
|
||||
}
|
||||
|
||||
let source = RawSource::new_from_slice(bytes);
|
||||
let decoder =
|
||||
rawler::get_decoder(&source).map_err(|e| DecodeError::Unsupported(e.to_string()))?;
|
||||
let params = Default::default();
|
||||
|
||||
// Preference order per requested size, each falling through to the next.
|
||||
let attempts: &[PreviewSize] = match size {
|
||||
PreviewSize::Thumbnail => &[
|
||||
PreviewSize::Thumbnail,
|
||||
PreviewSize::Screen,
|
||||
PreviewSize::Full,
|
||||
],
|
||||
PreviewSize::Screen => &[
|
||||
PreviewSize::Screen,
|
||||
PreviewSize::Full,
|
||||
PreviewSize::Thumbnail,
|
||||
],
|
||||
PreviewSize::Full => &[PreviewSize::Full, PreviewSize::Screen],
|
||||
};
|
||||
|
||||
for attempt in attempts {
|
||||
let got = match attempt {
|
||||
PreviewSize::Thumbnail => decoder.thumbnail_image(&source, ¶ms),
|
||||
PreviewSize::Screen => decoder.preview_image(&source, ¶ms),
|
||||
PreviewSize::Full => decoder.full_image(&source, ¶ms),
|
||||
};
|
||||
if let Ok(Some(img)) = got {
|
||||
let rgb = img.to_rgb8();
|
||||
let (width, height) = (rgb.width(), rgb.height());
|
||||
if width > 0 && height > 0 {
|
||||
return Ok(Preview {
|
||||
width,
|
||||
height,
|
||||
rgba: rgb_to_rgba(rgb.as_raw(), width, height),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Err(DecodeError::NoPreview)
|
||||
}
|
||||
|
||||
/// Extract the largest available preview.
|
||||
///
|
||||
/// Convenience over [`extract_preview`]; prefer naming a size explicitly.
|
||||
pub fn extract_embedded_preview(bytes: &[u8]) -> Result<Preview, DecodeError> {
|
||||
extract_preview(bytes, PreviewSize::Full)
|
||||
}
|
||||
|
||||
/// Decode a standalone JPEG (an embedded preview already sliced out, or a
|
||||
/// JPEG file).
|
||||
pub fn decode_jpeg(bytes: &[u8]) -> Result<Preview, DecodeError> {
|
||||
let mut d = zune_jpeg::JpegDecoder::new(bytes);
|
||||
let pixels = d
|
||||
.decode()
|
||||
.map_err(|e| DecodeError::CorruptPreview(e.to_string()))?;
|
||||
let info = d
|
||||
.info()
|
||||
.ok_or_else(|| DecodeError::CorruptPreview("no image info".into()))?;
|
||||
|
||||
let (w, h) = (info.width as u32, info.height as u32);
|
||||
let expected = (w as usize) * (h as usize);
|
||||
|
||||
// zune yields RGB or grayscale depending on the source; normalise both to
|
||||
// RGBA so callers have one representation.
|
||||
let rgba = match pixels.len() / expected.max(1) {
|
||||
3 => rgb_to_rgba(&pixels, w, h),
|
||||
1 => pixels.iter().flat_map(|&g| [g, g, g, 255]).collect(),
|
||||
4 => pixels,
|
||||
n => {
|
||||
return Err(DecodeError::CorruptPreview(format!(
|
||||
"unexpected {n} channels"
|
||||
)))
|
||||
}
|
||||
};
|
||||
|
||||
Ok(Preview {
|
||||
width: w,
|
||||
height: h,
|
||||
rgba,
|
||||
})
|
||||
}
|
||||
|
||||
fn rgb_to_rgba(rgb: &[u8], w: u32, h: u32) -> Vec<u8> {
|
||||
let n = (w as usize) * (h as usize);
|
||||
let mut out = Vec::with_capacity(n * 4);
|
||||
for px in rgb.chunks_exact(3).take(n) {
|
||||
out.extend_from_slice(&[px[0], px[1], px[2], 255]);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// A preview whose every pixel encodes its own coordinates, so a
|
||||
/// misplaced one is identifiable rather than merely wrong.
|
||||
fn coded(width: u32, height: u32) -> Preview {
|
||||
let mut rgba = Vec::with_capacity((width * height * 4) as usize);
|
||||
for y in 0..height {
|
||||
for x in 0..width {
|
||||
rgba.extend_from_slice(&[x as u8, y as u8, 0, 255]);
|
||||
}
|
||||
}
|
||||
Preview {
|
||||
width,
|
||||
height,
|
||||
rgba,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_quarter_turn_moves_every_pixel_where_the_orientation_says() {
|
||||
// Tag 6: the stored image's first row becomes the displayed right
|
||||
// edge, its first column the displayed top. A 4x2 landscape preview
|
||||
// therefore comes out 2x4 portrait, with stored (0,0) at the top right.
|
||||
let mut p = coded(4, 2);
|
||||
p.apply_orientation(dr_types::Orientation::from_exif(6));
|
||||
|
||||
assert_eq!((p.width, p.height), (2, 4));
|
||||
let at = |x: u32, y: u32| {
|
||||
let i = ((y * p.width + x) * 4) as usize;
|
||||
(p.rgba[i], p.rgba[i + 1])
|
||||
};
|
||||
// Displayed top-right reads stored (0, 0).
|
||||
assert_eq!(at(1, 0), (0, 0));
|
||||
// Displayed top-left reads stored (0, 1) — the last row of column 0.
|
||||
assert_eq!(at(0, 0), (0, 1));
|
||||
// Displayed bottom-right reads stored (3, 0).
|
||||
assert_eq!(at(1, 3), (3, 0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_upright_file_is_left_untouched() {
|
||||
// The common case, and the one where an unnecessary reallocation
|
||||
// would be paid on every thumbnail in the library.
|
||||
let original = coded(4, 2);
|
||||
let mut p = original.clone();
|
||||
p.apply_orientation(dr_types::Orientation::NORMAL);
|
||||
assert_eq!(p, original);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_orientation_preserves_the_pixels_it_was_given() {
|
||||
// A turn or a mirror is a permutation: the same bytes, rearranged.
|
||||
// Anything else means a pixel was dropped, duplicated or read out of
|
||||
// bounds — and the bounds case would have panicked first.
|
||||
for tag in 1..=8u16 {
|
||||
let orientation = dr_types::Orientation::from_exif(tag);
|
||||
let mut p = coded(5, 3);
|
||||
p.apply_orientation(orientation);
|
||||
|
||||
assert_eq!(
|
||||
(p.width, p.height),
|
||||
orientation.oriented_size(5, 3),
|
||||
"tag {tag}"
|
||||
);
|
||||
let mut got: Vec<_> = p.rgba.chunks(4).map(|c| (c[0], c[1])).collect();
|
||||
got.sort_unstable();
|
||||
let mut want: Vec<_> = coded(5, 3).rgba.chunks(4).map(|c| (c[0], c[1])).collect();
|
||||
want.sort_unstable();
|
||||
assert_eq!(got, want, "tag {tag}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn size_preference_falls_through_in_order() {
|
||||
// A container missing the requested size must yield the next
|
||||
// available rather than failing (FR-CULL-2).
|
||||
// Ordering is asserted here; behaviour against real files is covered
|
||||
// by the smoke example.
|
||||
assert_ne!(PreviewSize::Thumbnail, PreviewSize::Full);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn usefulness_is_judged_on_the_long_edge() {
|
||||
let p = Preview {
|
||||
width: 1600,
|
||||
height: 1067,
|
||||
rgba: Vec::new(),
|
||||
};
|
||||
assert!(p.is_useful_at(1024));
|
||||
assert!(p.is_useful_at(1600));
|
||||
// A body embedding only a small thumbnail must trigger a background
|
||||
// render rather than showing a soft image.
|
||||
assert!(!p.is_useful_at(2048));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn downscale_preserves_aspect_and_bounds_memory() {
|
||||
let mut p = Preview {
|
||||
width: 5472,
|
||||
height: 3648,
|
||||
rgba: vec![128; 5472 * 3648 * 4],
|
||||
};
|
||||
assert_eq!(p.rgba.len(), 79_847_424);
|
||||
|
||||
p.downscale_to(2048);
|
||||
assert_eq!(p.width, 2048);
|
||||
assert_eq!(p.height, 1365, "aspect preserved");
|
||||
assert_eq!(p.rgba.len(), (2048 * 1365 * 4) as usize);
|
||||
// A flat source must stay flat through the box filter.
|
||||
assert!(p
|
||||
.rgba
|
||||
.chunks_exact(4)
|
||||
.all(|px| px[0] == 128 && px[3] == 255));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn downscale_is_a_noop_when_already_small() {
|
||||
let mut p = Preview {
|
||||
width: 720,
|
||||
height: 480,
|
||||
rgba: vec![7; 720 * 480 * 4],
|
||||
};
|
||||
let before = p.rgba.len();
|
||||
p.downscale_to(2048);
|
||||
assert_eq!((p.width, p.height, p.rgba.len()), (720, 480, before));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rgb_expands_to_rgba_opaque() {
|
||||
let rgb = [10, 20, 30, 40, 50, 60];
|
||||
let rgba = rgb_to_rgba(&rgb, 2, 1);
|
||||
assert_eq!(rgba, vec![10, 20, 30, 255, 40, 50, 60, 255]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn corrupt_jpeg_is_an_error_not_a_panic() {
|
||||
// Untrusted input arrives here (NFR-SEC-1); it must never panic.
|
||||
let err = decode_jpeg(&[0xFF, 0xD8, 0x00, 0x01, 0x02]).unwrap_err();
|
||||
assert!(matches!(err, DecodeError::CorruptPreview(_)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_input_is_an_error_not_a_panic() {
|
||||
assert!(decode_jpeg(&[]).is_err());
|
||||
}
|
||||
}
|
||||
@@ -1,41 +0,0 @@
|
||||
[package]
|
||||
name = "dr-export"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
license.workspace = true
|
||||
|
||||
# No platform dependency and no filesystem, deliberately. This crate turns a
|
||||
# rendered frame into *bytes* and a *name*; where those bytes go is the
|
||||
# caller's problem, because the answer differs by more than a path. On Linux
|
||||
# it is a file, on Android a SAF document descriptor with no path at all
|
||||
# (ARCH §6.9), and on either it may be a `PUT` to the server. A crate that
|
||||
# took a `Path` would work on exactly one of the three.
|
||||
[dependencies]
|
||||
dr-types.workspace = true
|
||||
log.workspace = true
|
||||
thiserror.workspace = true
|
||||
|
||||
# Encoders. All three are pure Rust and already in the tree, which is the same
|
||||
# criterion that chose rustls, bundled SQLite and the Lensfun port: a C
|
||||
# dependency here would be one more thing to satisfy under the Android NDK.
|
||||
#
|
||||
# AVIF and JPEG XL (FR-EXP-1) are deliberately absent. The mature encoders for
|
||||
# both are C or C++ — libaom and libjxl — and ravif, the pure-Rust AVIF path,
|
||||
# is slow enough to change what a batch export feels like. Neither belongs in
|
||||
# the first version; see `format` in lib.rs for what happens when one is asked
|
||||
# for.
|
||||
jpeg-encoder.workspace = true
|
||||
png = "0.18"
|
||||
tiff = "0.11"
|
||||
|
||||
# The example runs the whole path — decode, GPU render, read back, encode,
|
||||
# write — so it needs what the library deliberately does not: a GPU, a
|
||||
# pipeline and a decoder. Dev-only, so none of it reaches a dependent.
|
||||
[dev-dependencies]
|
||||
dr-decode.workspace = true
|
||||
dr-gpu.workspace = true
|
||||
dr-pipeline.workspace = true
|
||||
env_logger.workspace = true
|
||||
pollster.workspace = true
|
||||
zune-jpeg.workspace = true
|
||||
@@ -1,211 +0,0 @@
|
||||
//! Export a real file, end to end, from a real image.
|
||||
//!
|
||||
//! cargo run -p dr-export --example export -- <file.jpg|file.cr2> [out-dir]
|
||||
//!
|
||||
//! Deliberately the *whole* path and not a unit test of the encoder: decode,
|
||||
//! demosaic or upload, run the develop chain on the GPU at full resolution,
|
||||
//! read the result back through `AdjustPass::export_pixels`, resize, sharpen,
|
||||
//! encode, and write. A test can prove the JPEG has the right magic bytes; it
|
||||
//! cannot tell anyone whether the picture came out looking like the picture.
|
||||
|
||||
use std::path::PathBuf;
|
||||
|
||||
use dr_export::{export, Frame, NameContext, SourceMetadata};
|
||||
use dr_gpu::{AdjustPass, DemosaicedImage, Demosaicer, GpuContext};
|
||||
use dr_pipeline::EditGraph;
|
||||
use dr_types::{ColourSpace, ExportFormat, ExportSettings, OutputSharpening, SizingMode};
|
||||
|
||||
fn main() {
|
||||
env_logger::Builder::from_env(env_logger::Env::default().default_filter_or("info,wgpu=warn"))
|
||||
.init();
|
||||
|
||||
let mut args = std::env::args().skip(1);
|
||||
let Some(input) = args.next() else {
|
||||
eprintln!("usage: export <file.jpg|file.cr2> [out-dir]");
|
||||
std::process::exit(2);
|
||||
};
|
||||
let out_dir = PathBuf::from(args.next().unwrap_or_else(|| ".".into()));
|
||||
let input = PathBuf::from(input);
|
||||
|
||||
let ctx = pollster::block_on(GpuContext::new_headless()).expect("gpu");
|
||||
println!("gpu: {} ({:?})", ctx.adapter_name(), ctx.backend());
|
||||
|
||||
// Decode. A RAW goes through the demosaicer; a JPEG is already RGB and
|
||||
// takes the same path every operation after the sensor stage does.
|
||||
let bytes = std::fs::read(&input).expect("read input");
|
||||
// From content, not from the extension — dr-decode is emphatic that an
|
||||
// extension is only a hint. Its own `probe` reports a crate-private
|
||||
// `Format`, so the SOI marker is checked directly here rather than
|
||||
// widening that API for an example.
|
||||
let is_jpeg = bytes.starts_with(&[0xFF, 0xD8, 0xFF]);
|
||||
let source = if !is_jpeg {
|
||||
let raw = dr_decode::decode(&bytes).expect("decode raw");
|
||||
let demosaicer = Demosaicer::new(&ctx).expect("demosaicer");
|
||||
demosaicer.run(&raw).expect("demosaic")
|
||||
} else {
|
||||
let (rgba, w, h) = decode_jpeg(&bytes);
|
||||
DemosaicedImage::from_rgba8(&ctx, &rgba, w, h).expect("upload")
|
||||
};
|
||||
|
||||
// An edit worth seeing in the output, so a broken pipeline is obvious
|
||||
// rather than subtle.
|
||||
let mut graph = EditGraph::default_chain();
|
||||
graph.set_param(
|
||||
dr_pipeline::ops::exposure::ID,
|
||||
dr_pipeline::ops::exposure::EXPOSURE,
|
||||
0.35,
|
||||
);
|
||||
graph.set_param(
|
||||
dr_pipeline::ops::contrast::ID,
|
||||
dr_pipeline::ops::contrast::CONTRAST,
|
||||
18.0,
|
||||
);
|
||||
graph.set_param(
|
||||
dr_pipeline::ops::saturation::ID,
|
||||
dr_pipeline::ops::saturation::SATURATION,
|
||||
12.0,
|
||||
);
|
||||
|
||||
// Full resolution, not the viewport (FR-EXP-9). This is the one thing an
|
||||
// export must not economise on.
|
||||
let (sw, sh) = source.size();
|
||||
let (fw, fh) = graph.output_size(sw, sh);
|
||||
println!("source {sw}×{sh}, framed {fw}×{fh}");
|
||||
|
||||
// The output space is chosen *here*, before the render, because that is
|
||||
// where it takes effect: the primaries conversion and the encode are the
|
||||
// last two lines of the generated shader (FR-EXP-2). Asking for it at the
|
||||
// encoder would be too late — the pixels would already be clipped.
|
||||
let space = ColourSpace::DisplayP3;
|
||||
|
||||
let mut adjust = AdjustPass::new(&ctx);
|
||||
let shader = graph.compose_for(space);
|
||||
let t = std::time::Instant::now();
|
||||
adjust.render(&source, &shader, fw, fh).expect("render");
|
||||
let (pixels, w, h) = adjust.export_pixels().expect("read back");
|
||||
println!(
|
||||
"rendered {w}×{h} in {:.0} ms as {}",
|
||||
t.elapsed().as_secs_f32() * 1000.0,
|
||||
space.label()
|
||||
);
|
||||
|
||||
let frame = Frame::in_space(w, h, pixels, space).expect("well-formed frame");
|
||||
|
||||
let stem = input
|
||||
.file_stem()
|
||||
.map(|s| s.to_string_lossy().into_owned())
|
||||
.unwrap_or_else(|| "export".into());
|
||||
|
||||
// TRACES: FR-EXP-8
|
||||
// What the input said about itself, transcribed field by field into the
|
||||
// allowlist `dr-export` will write from. The example passes it because
|
||||
// this is the one place in the tree that produces files a person can open
|
||||
// in exiftool — a unit test can prove a GPS directory is absent from a
|
||||
// byte slice, but only a real export proves that a real photograph comes
|
||||
// out of the far end still knowing which camera took it.
|
||||
//
|
||||
// The defaults apply, so the files written here carry the camera, the
|
||||
// lens, the exposure and the rights statement, and carry no coordinates.
|
||||
let meta = dr_decode::metadata(&bytes).unwrap_or_default();
|
||||
let source_metadata = SourceMetadata {
|
||||
make: meta.make.clone(),
|
||||
model: meta.model.clone(),
|
||||
lens: meta.lens.clone(),
|
||||
shutter: meta.shutter,
|
||||
aperture: meta.aperture,
|
||||
iso: meta.iso,
|
||||
focal_length: meta.focal_length,
|
||||
captured_at: meta.captured_at,
|
||||
captured_offset: meta.captured_offset,
|
||||
artist: meta.artist.clone(),
|
||||
copyright: meta.copyright.clone(),
|
||||
location: meta.location,
|
||||
};
|
||||
|
||||
// One of each format, so the run exercises every encoder that exists.
|
||||
for (format, sizing, sharpening) in [
|
||||
(
|
||||
ExportFormat::Jpeg,
|
||||
SizingMode::Original,
|
||||
OutputSharpening::None,
|
||||
),
|
||||
(
|
||||
ExportFormat::Jpeg,
|
||||
SizingMode::LongEdge(1200),
|
||||
OutputSharpening::Screen,
|
||||
),
|
||||
(
|
||||
ExportFormat::Png,
|
||||
SizingMode::LongEdge(600),
|
||||
OutputSharpening::Screen,
|
||||
),
|
||||
(
|
||||
ExportFormat::Tiff8,
|
||||
SizingMode::Percentage(25),
|
||||
OutputSharpening::MattePaper,
|
||||
),
|
||||
(
|
||||
ExportFormat::Tiff16,
|
||||
SizingMode::Percentage(25),
|
||||
OutputSharpening::MattePaper,
|
||||
),
|
||||
] {
|
||||
let settings = ExportSettings {
|
||||
format,
|
||||
sizing,
|
||||
sharpening,
|
||||
colour_space: space,
|
||||
filename_template: "{name}-{dimensions}".into(),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
// The size has to be known before the name, because `{dimensions}` is
|
||||
// part of it — which is why sizing is resolved here and not inside
|
||||
// `export`.
|
||||
let (tw, th) = dr_export::target_size(w, h, sizing, settings.allow_upscaling);
|
||||
let ctx = NameContext {
|
||||
source_stem: &stem,
|
||||
sequence: 1,
|
||||
date: "",
|
||||
width: tw,
|
||||
height: th,
|
||||
preset: "",
|
||||
};
|
||||
let name = dr_export::resolve_name(
|
||||
&settings.filename_template,
|
||||
&ctx,
|
||||
format,
|
||||
settings.collision,
|
||||
&|n| out_dir.join(n).exists(),
|
||||
)
|
||||
.expect("a free name");
|
||||
|
||||
let t = std::time::Instant::now();
|
||||
let out = export(&frame, &settings, name, Some(&source_metadata)).expect("export");
|
||||
let path = out_dir.join(&out.name);
|
||||
std::fs::write(&path, &out.bytes).expect("write");
|
||||
println!(
|
||||
"{:>10} {:>5}×{:<5} {:>8} KB {:>5.0} ms {}",
|
||||
format.label(),
|
||||
out.width,
|
||||
out.height,
|
||||
out.bytes.len() / 1024,
|
||||
t.elapsed().as_secs_f32() * 1000.0,
|
||||
path.display()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
fn decode_jpeg(bytes: &[u8]) -> (Vec<u8>, u32, u32) {
|
||||
let mut decoder = zune_jpeg::JpegDecoder::new(bytes);
|
||||
let pixels = decoder.decode().expect("decode jpeg");
|
||||
let info = decoder.info().expect("jpeg info");
|
||||
let (w, h) = (u32::from(info.width), u32::from(info.height));
|
||||
|
||||
// zune gives RGB; the GPU upload wants RGBA.
|
||||
let mut rgba = Vec::with_capacity((w * h * 4) as usize);
|
||||
for px in pixels.chunks_exact(3) {
|
||||
rgba.extend_from_slice(&[px[0], px[1], px[2], 255]);
|
||||
}
|
||||
(rgba, w, h)
|
||||
}
|
||||
@@ -1,46 +0,0 @@
|
||||
//! TRACES: NFR-ARCH-4
|
||||
//! Typed export failures.
|
||||
//!
|
||||
//! Every variant is something a caller can act on or report. A batch export
|
||||
//! runs unattended over hundreds of frames (FR-EXP-7), so "what went wrong
|
||||
//! with which file" has to survive as data rather than as a log line.
|
||||
|
||||
use dr_types::{ColourSpace, ExportFormat};
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum ExportError {
|
||||
#[error("frame buffer is {got} bytes, expected {expected}")]
|
||||
FrameSize { expected: usize, got: usize },
|
||||
|
||||
#[error("frame has no pixels")]
|
||||
EmptyFrame,
|
||||
|
||||
/// Asked for a format with no encoder in this build.
|
||||
///
|
||||
/// Not a panic and not a silent substitution: the settings page offers
|
||||
/// AVIF and JPEG XL because FR-EXP-1 lists them, and a build without them
|
||||
/// should say so rather than quietly writing a JPEG under a `.avif` name.
|
||||
#[error("{} export is not supported yet", .0.label())]
|
||||
FormatUnsupported(ExportFormat),
|
||||
|
||||
/// TRACES: FR-EXP-2
|
||||
/// The frame was rendered into one colour space and asked to be labelled
|
||||
/// another.
|
||||
///
|
||||
/// Not a limitation of the encoders — all four spaces embed a correct
|
||||
/// profile. It is that the conversion happens in the shader, before the
|
||||
/// clip to 0..1, so a frame is in exactly one space by the time it gets
|
||||
/// here. The caller composes with `EditGraph::compose_for` to change which.
|
||||
#[error(
|
||||
"the frame was rendered in {} but a {} file was asked for",
|
||||
.rendered.label(),
|
||||
.requested.label()
|
||||
)]
|
||||
ColourSpaceMismatch {
|
||||
rendered: ColourSpace,
|
||||
requested: ColourSpace,
|
||||
},
|
||||
|
||||
#[error("encoding failed: {0}")]
|
||||
Encode(String),
|
||||
}
|
||||
@@ -1,518 +0,0 @@
|
||||
//! TRACES: FR-EXP-8
|
||||
//! Building an EXIF block, rather than copying one.
|
||||
//!
|
||||
//! # Why this is written by hand and not with a crate
|
||||
//!
|
||||
//! Two reasons, in order of importance.
|
||||
//!
|
||||
//! The first is the privacy behaviour. Every EXIF library worth using offers a
|
||||
//! "load the source block, delete these tags, write it back" shape, and that
|
||||
//! shape is the wrong one here: it makes the file that leaves the machine a
|
||||
//! copy of the source's metadata *minus what we thought to remove*, so every
|
||||
//! tag nobody has thought about — a vendor's proprietary sub-directory, a
|
||||
//! serial number under a tag id this build has never seen — travels by
|
||||
//! default. Constructing the block from a fixed list of parsed values inverts
|
||||
//! that. What is written is exactly what appears in [`crate::SourceMetadata`],
|
||||
//! and a tag that is not in this file cannot end up in the output no matter
|
||||
//! what the source contained. The allowlist *is* the implementation.
|
||||
//!
|
||||
//! The second is the dependency policy. The root `Cargo.toml` explains why
|
||||
//! nothing here may link C — this tree has to build under the Android NDK —
|
||||
//! and the mature EXIF writers are bindings. This is a couple of hundred
|
||||
//! lines of offset arithmetic against a specification that has not changed
|
||||
//! since 2010, and it is the same TIFF structure `dr-decode` already reads.
|
||||
//!
|
||||
//! # What the block is
|
||||
//!
|
||||
//! A complete little-endian TIFF: an 8-byte header, IFD0 with the identity
|
||||
//! and rights tags, an Exif sub-IFD with the capture tags, optionally a GPS
|
||||
//! sub-IFD, and a heap of values too long to sit inside an entry. JPEG carries
|
||||
//! it in an APP1 segment behind the marker `Exif\0\0`; PNG carries the same
|
||||
//! bytes in an `eXIf` chunk with no marker. TIFF does not use this at all —
|
||||
//! its own directory *is* the EXIF, so `encode.rs` writes the tags there
|
||||
//! directly.
|
||||
|
||||
use crate::metadata::SourceMetadata;
|
||||
|
||||
/// One entry's value, in the handful of TIFF types this writer emits.
|
||||
enum Value {
|
||||
/// NUL-terminated, as the specification requires; the terminator is
|
||||
/// counted, which is the detail readers trip over when it is missing.
|
||||
Ascii(String),
|
||||
Byte(Vec<u8>),
|
||||
Short(u16),
|
||||
Long(u32),
|
||||
/// Type 7. Used only for `ExifVersion`, which is four characters that are
|
||||
/// deliberately *not* a string.
|
||||
Undefined(&'static [u8]),
|
||||
/// Numerator and denominator pairs. A coordinate is three of them.
|
||||
Rational(Vec<(u32, u32)>),
|
||||
}
|
||||
|
||||
impl Value {
|
||||
fn field_type(&self) -> u16 {
|
||||
match self {
|
||||
Value::Byte(_) => 1,
|
||||
Value::Ascii(_) => 2,
|
||||
Value::Short(_) => 3,
|
||||
Value::Long(_) => 4,
|
||||
Value::Rational(_) => 5,
|
||||
Value::Undefined(_) => 7,
|
||||
}
|
||||
}
|
||||
|
||||
/// The element count, which is not the byte length: a rational counts as
|
||||
/// one element per eight bytes.
|
||||
fn count(&self) -> u32 {
|
||||
match self {
|
||||
Value::Ascii(s) => s.len() as u32 + 1,
|
||||
Value::Byte(b) => b.len() as u32,
|
||||
Value::Undefined(b) => b.len() as u32,
|
||||
Value::Short(_) | Value::Long(_) => 1,
|
||||
Value::Rational(r) => r.len() as u32,
|
||||
}
|
||||
}
|
||||
|
||||
/// The payload, in file order.
|
||||
fn payload(&self) -> Vec<u8> {
|
||||
match self {
|
||||
Value::Ascii(s) => {
|
||||
let mut out = s.as_bytes().to_vec();
|
||||
out.push(0);
|
||||
out
|
||||
}
|
||||
Value::Byte(b) => b.clone(),
|
||||
Value::Undefined(b) => b.to_vec(),
|
||||
Value::Short(v) => v.to_le_bytes().to_vec(),
|
||||
Value::Long(v) => v.to_le_bytes().to_vec(),
|
||||
Value::Rational(r) => r
|
||||
.iter()
|
||||
.flat_map(|(n, d)| {
|
||||
let mut b = n.to_le_bytes().to_vec();
|
||||
b.extend_from_slice(&d.to_le_bytes());
|
||||
b
|
||||
})
|
||||
.collect(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// An IFD under construction.
|
||||
type Entries = Vec<(u16, Value)>;
|
||||
|
||||
/// Tag numbers. Named rather than inlined because a mistyped one produces a
|
||||
/// file that still parses and says something else entirely.
|
||||
pub(crate) mod tag {
|
||||
pub(crate) const MAKE: u16 = 0x010F;
|
||||
pub(crate) const MODEL: u16 = 0x0110;
|
||||
pub(crate) const SOFTWARE: u16 = 0x0131;
|
||||
pub(crate) const DATE_TIME: u16 = 0x0132;
|
||||
pub(crate) const ARTIST: u16 = 0x013B;
|
||||
pub(crate) const COPYRIGHT: u16 = 0x8298;
|
||||
pub(crate) const EXIF_IFD: u16 = 0x8769;
|
||||
pub(crate) const GPS_IFD: u16 = 0x8825;
|
||||
|
||||
pub(crate) const EXPOSURE_TIME: u16 = 0x829A;
|
||||
pub(crate) const FNUMBER: u16 = 0x829D;
|
||||
pub(crate) const ISO: u16 = 0x8827;
|
||||
pub(crate) const EXIF_VERSION: u16 = 0x9000;
|
||||
pub(crate) const DATE_TIME_ORIGINAL: u16 = 0x9003;
|
||||
pub(crate) const OFFSET_TIME_ORIGINAL: u16 = 0x9011;
|
||||
pub(crate) const FOCAL_LENGTH: u16 = 0x920A;
|
||||
pub(crate) const PIXEL_X: u16 = 0xA002;
|
||||
pub(crate) const PIXEL_Y: u16 = 0xA003;
|
||||
pub(crate) const LENS_MODEL: u16 = 0xA434;
|
||||
|
||||
pub(crate) const GPS_VERSION_ID: u16 = 0x0000;
|
||||
pub(crate) const GPS_LATITUDE_REF: u16 = 0x0001;
|
||||
pub(crate) const GPS_LATITUDE: u16 = 0x0002;
|
||||
pub(crate) const GPS_LONGITUDE_REF: u16 = 0x0003;
|
||||
pub(crate) const GPS_LONGITUDE: u16 = 0x0004;
|
||||
pub(crate) const GPS_ALTITUDE_REF: u16 = 0x0005;
|
||||
pub(crate) const GPS_ALTITUDE: u16 = 0x0006;
|
||||
}
|
||||
|
||||
/// What DarkRoom calls itself in a file it wrote.
|
||||
///
|
||||
/// Not vanity: an export is a derived file, and a reader that knows which
|
||||
/// program produced it can tell a camera original from a rendition without
|
||||
/// guessing from the absence of a maker note.
|
||||
pub(crate) const SOFTWARE: &str = "DarkRoom";
|
||||
|
||||
/// The complete EXIF block for JPEG's APP1 and PNG's `eXIf`.
|
||||
///
|
||||
/// `width`/`height` are the *exported* dimensions, not the source's: the
|
||||
/// pixel-dimension tags describe the file they are in, and a reader that
|
||||
/// trusts them after a resize would report the wrong size for the image it is
|
||||
/// holding.
|
||||
///
|
||||
/// `None` where there is nothing to say. An empty EXIF block is not the same
|
||||
/// as no EXIF block — it is a structure a reader must parse to discover it
|
||||
/// learned nothing — and the second is the better file.
|
||||
pub(crate) fn block(md: &SourceMetadata, width: u32, height: u32) -> Option<Vec<u8>> {
|
||||
let ifd0 = main_entries(md);
|
||||
let exif = exif_entries(md, width, height);
|
||||
let gps = gps_entries(md);
|
||||
if ifd0.is_empty() && exif.is_empty() && gps.is_empty() {
|
||||
return None;
|
||||
}
|
||||
Some(assemble(ifd0, exif, gps))
|
||||
}
|
||||
|
||||
/// Lay the three directories and their heap out in the block.
|
||||
///
|
||||
/// The order is fixed — IFD0, Exif, GPS, heap — because the pointers have to
|
||||
/// be known before IFD0 is serialised, and an IFD's size is decided by its
|
||||
/// entry count alone: two bytes of count, twelve per entry, four for the link
|
||||
/// to the next directory.
|
||||
fn assemble(mut ifd0: Entries, exif: Entries, gps: Entries) -> Vec<u8> {
|
||||
const HEADER: u32 = 8;
|
||||
let size = |n: usize| 2 + 12 * n as u32 + 4;
|
||||
|
||||
// The pointer entries are part of IFD0's count, so they have to be added
|
||||
// before its size is taken — a chicken-and-egg the specification resolves
|
||||
// by making entry size fixed.
|
||||
let pointers = usize::from(!exif.is_empty()) + usize::from(!gps.is_empty());
|
||||
let ifd0_size = size(ifd0.len() + pointers);
|
||||
|
||||
let exif_offset = HEADER + ifd0_size;
|
||||
let gps_offset = exif_offset + if exif.is_empty() { 0 } else { size(exif.len()) };
|
||||
let heap_base = gps_offset + if gps.is_empty() { 0 } else { size(gps.len()) };
|
||||
|
||||
if !exif.is_empty() {
|
||||
ifd0.push((tag::EXIF_IFD, Value::Long(exif_offset)));
|
||||
}
|
||||
if !gps.is_empty() {
|
||||
ifd0.push((tag::GPS_IFD, Value::Long(gps_offset)));
|
||||
}
|
||||
|
||||
let mut heap = Vec::new();
|
||||
let ifd0_bytes = directory(ifd0, heap_base, &mut heap);
|
||||
let exif_bytes = directory(exif, heap_base, &mut heap);
|
||||
let gps_bytes = directory(gps, heap_base, &mut heap);
|
||||
|
||||
let mut out = Vec::with_capacity(HEADER as usize + heap.len() + 128);
|
||||
// Little-endian, magic 42, first directory at byte 8. Little-endian
|
||||
// because every value written below is, and a header that disagreed with
|
||||
// its own body is the one corruption a reader cannot recover from.
|
||||
out.extend_from_slice(b"II");
|
||||
out.extend_from_slice(&42u16.to_le_bytes());
|
||||
out.extend_from_slice(&HEADER.to_le_bytes());
|
||||
out.extend_from_slice(&ifd0_bytes);
|
||||
out.extend_from_slice(&exif_bytes);
|
||||
out.extend_from_slice(&gps_bytes);
|
||||
out.extend_from_slice(&heap);
|
||||
out
|
||||
}
|
||||
|
||||
/// Serialise one directory, spilling long values onto the shared heap.
|
||||
///
|
||||
/// Entries are sorted by tag: TIFF requires ascending order within a
|
||||
/// directory, and while most readers cope with any order, the ones that
|
||||
/// binary-search stop at the first tag they cannot place.
|
||||
fn directory(mut entries: Entries, heap_base: u32, heap: &mut Vec<u8>) -> Vec<u8> {
|
||||
if entries.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
entries.sort_by_key(|(tag, _)| *tag);
|
||||
|
||||
let mut out = Vec::with_capacity(2 + entries.len() * 12 + 4);
|
||||
out.extend_from_slice(&(entries.len() as u16).to_le_bytes());
|
||||
for (tag, value) in &entries {
|
||||
out.extend_from_slice(&tag.to_le_bytes());
|
||||
out.extend_from_slice(&value.field_type().to_le_bytes());
|
||||
out.extend_from_slice(&value.count().to_le_bytes());
|
||||
|
||||
let payload = value.payload();
|
||||
if payload.len() <= 4 {
|
||||
// Four bytes or fewer live in the entry itself, left-justified and
|
||||
// zero-padded.
|
||||
let mut inline = payload.clone();
|
||||
inline.resize(4, 0);
|
||||
out.extend_from_slice(&inline);
|
||||
} else {
|
||||
out.extend_from_slice(&(heap_base + heap.len() as u32).to_le_bytes());
|
||||
heap.extend_from_slice(&payload);
|
||||
// Values start on even offsets. Not every reader cares; the ones
|
||||
// that do read a short from an odd address and get nonsense.
|
||||
if heap.len() % 2 == 1 {
|
||||
heap.push(0);
|
||||
}
|
||||
}
|
||||
}
|
||||
// No directory follows this one. The Exif and GPS sub-directories are
|
||||
// pointed at, not chained, so this is zero in all three.
|
||||
out.extend_from_slice(&0u32.to_le_bytes());
|
||||
out
|
||||
}
|
||||
|
||||
/// IFD0: who took it, with what, and who owns it.
|
||||
///
|
||||
/// **No orientation tag, deliberately.** The frame reaching the encoder has
|
||||
/// already had the source's orientation applied by the pipeline — it is
|
||||
/// upright pixels — so copying the source's tag across would tell every
|
||||
/// reader to rotate an image that is already the right way up. A portrait
|
||||
/// frame would come out on its side in exactly the viewers that honour the
|
||||
/// tag, which is most of them.
|
||||
fn main_entries(md: &SourceMetadata) -> Entries {
|
||||
let mut e = Entries::new();
|
||||
push_ascii(&mut e, tag::MAKE, md.make.as_deref());
|
||||
push_ascii(&mut e, tag::MODEL, md.model.as_deref());
|
||||
push_ascii(&mut e, tag::ARTIST, md.artist.as_deref());
|
||||
push_ascii(&mut e, tag::COPYRIGHT, md.copyright.as_deref());
|
||||
e.push((tag::SOFTWARE, Value::Ascii(SOFTWARE.to_string())));
|
||||
// IFD0's `DateTime` is nominally when the file was written, and this is
|
||||
// the capture time instead. That is what the rest of the world does —
|
||||
// and it is what `dr-decode` falls back to for scanner output that has no
|
||||
// `DateTimeOriginal` — so a re-import of an export lands on the timeline
|
||||
// where the original did rather than on the day it was exported.
|
||||
if let Some(t) = md.captured_at.map(datetime) {
|
||||
e.push((tag::DATE_TIME, Value::Ascii(t)));
|
||||
}
|
||||
e
|
||||
}
|
||||
|
||||
/// The Exif sub-IFD: the exposure, and what made it.
|
||||
fn exif_entries(md: &SourceMetadata, width: u32, height: u32) -> Entries {
|
||||
let mut e = Entries::new();
|
||||
// "0232" is Exif 2.32. A sub-directory without a version is technically
|
||||
// malformed, and some readers refuse the whole block over it.
|
||||
e.push((tag::EXIF_VERSION, Value::Undefined(b"0232")));
|
||||
e.push((tag::PIXEL_X, Value::Long(width)));
|
||||
e.push((tag::PIXEL_Y, Value::Long(height)));
|
||||
push_ascii(&mut e, tag::LENS_MODEL, md.lens.as_deref());
|
||||
if let Some(t) = md.captured_at.map(datetime) {
|
||||
e.push((tag::DATE_TIME_ORIGINAL, Value::Ascii(t)));
|
||||
}
|
||||
if let Some(o) = md.captured_offset.map(offset) {
|
||||
e.push((tag::OFFSET_TIME_ORIGINAL, Value::Ascii(o)));
|
||||
}
|
||||
if let Some(s) = md.shutter.filter(|s| *s > 0.0) {
|
||||
e.push((tag::EXPOSURE_TIME, Value::Rational(vec![shutter(s)])));
|
||||
}
|
||||
if let Some(f) = md.aperture.filter(|f| *f > 0.0) {
|
||||
e.push((tag::FNUMBER, Value::Rational(vec![tenths(f)])));
|
||||
}
|
||||
if let Some(f) = md.focal_length.filter(|f| *f > 0.0) {
|
||||
e.push((tag::FOCAL_LENGTH, Value::Rational(vec![tenths(f)])));
|
||||
}
|
||||
// The tag is a SHORT, so a sensitivity above 65535 has no representation
|
||||
// in it. Dropped rather than truncated: ISO 102400 written as 36864 is a
|
||||
// lie, and an absent tag is not.
|
||||
if let Some(iso) = md.iso.filter(|v| *v <= u32::from(u16::MAX)) {
|
||||
e.push((tag::ISO, Value::Short(iso as u16)));
|
||||
}
|
||||
e
|
||||
}
|
||||
|
||||
/// The GPS sub-IFD.
|
||||
///
|
||||
/// Empty unless the caller has already decided that coordinates may be
|
||||
/// written — see [`SourceMetadata::sanitised`], which is where the stripping
|
||||
/// happens. Nothing in this file consults the settings, so there is exactly
|
||||
/// one place to look to answer "can this export carry a location".
|
||||
fn gps_entries(md: &SourceMetadata) -> Entries {
|
||||
let Some(loc) = md.location else {
|
||||
return Entries::new();
|
||||
};
|
||||
let mut e = Entries::new();
|
||||
// 2.3.0.0, the current GPS tag version.
|
||||
e.push((tag::GPS_VERSION_ID, Value::Byte(vec![2, 3, 0, 0])));
|
||||
e.push((
|
||||
tag::GPS_LATITUDE_REF,
|
||||
Value::Ascii(if loc.latitude < 0.0 { "S" } else { "N" }.into()),
|
||||
));
|
||||
e.push((tag::GPS_LATITUDE, Value::Rational(dms(loc.latitude))));
|
||||
e.push((
|
||||
tag::GPS_LONGITUDE_REF,
|
||||
Value::Ascii(if loc.longitude < 0.0 { "W" } else { "E" }.into()),
|
||||
));
|
||||
e.push((tag::GPS_LONGITUDE, Value::Rational(dms(loc.longitude))));
|
||||
if let Some(alt) = loc.altitude {
|
||||
// The altitude itself is unsigned; below sea level is a separate byte.
|
||||
e.push((
|
||||
tag::GPS_ALTITUDE_REF,
|
||||
Value::Byte(vec![u8::from(alt < 0.0)]),
|
||||
));
|
||||
e.push((
|
||||
tag::GPS_ALTITUDE,
|
||||
Value::Rational(vec![((alt.abs() * 100.0).round() as u32, 100)]),
|
||||
));
|
||||
}
|
||||
e
|
||||
}
|
||||
|
||||
fn push_ascii(entries: &mut Entries, tag: u16, value: Option<&str>) {
|
||||
// An empty string is a tag saying nothing, which is worse than no tag: it
|
||||
// overwrites whatever a reader would otherwise have inferred.
|
||||
if let Some(v) = value.map(str::trim).filter(|v| !v.is_empty()) {
|
||||
entries.push((tag, Value::Ascii(v.to_string())));
|
||||
}
|
||||
}
|
||||
|
||||
/// Signed degrees back into the tag's degrees/minutes/seconds.
|
||||
///
|
||||
/// The sign is carried by the hemisphere letter, so this takes the magnitude.
|
||||
/// Seconds keep four decimal places, which is about 3 mm — far finer than any
|
||||
/// consumer fix, and enough that a round trip through the tag does not move
|
||||
/// the pin.
|
||||
pub(crate) fn dms(degrees: f64) -> Vec<(u32, u32)> {
|
||||
let d = degrees.abs();
|
||||
let whole = d.trunc();
|
||||
let minutes = (d - whole) * 60.0;
|
||||
let seconds = (minutes - minutes.trunc()) * 60.0;
|
||||
vec![
|
||||
(whole as u32, 1),
|
||||
(minutes.trunc() as u32, 1),
|
||||
((seconds * 10_000.0).round() as u32, 10_000),
|
||||
]
|
||||
}
|
||||
|
||||
/// A shutter speed as the fraction a photographer would recognise.
|
||||
///
|
||||
/// `1/250`, not `4/1000`. Both are the same number and every reader computes
|
||||
/// the same exposure from either, but the first is what the camera wrote and
|
||||
/// what a properties panel displays verbatim.
|
||||
pub(crate) fn shutter(seconds: f32) -> (u32, u32) {
|
||||
if seconds < 1.0 {
|
||||
(1, (1.0 / seconds).round().max(1.0) as u32)
|
||||
} else {
|
||||
((seconds * 10.0).round() as u32, 10)
|
||||
}
|
||||
}
|
||||
|
||||
/// f/2.8 and 85 mm as tenths, which is how cameras write both.
|
||||
pub(crate) fn tenths(value: f32) -> (u32, u32) {
|
||||
((value * 10.0).round().max(0.0) as u32, 10)
|
||||
}
|
||||
|
||||
/// Unix seconds as EXIF's `"YYYY:MM:DD HH:MM:SS"`.
|
||||
///
|
||||
/// The reading is a wall clock with no zone — that is what the tag means, and
|
||||
/// what `dr-decode` parsed it as — so this is the exact inverse of that parse
|
||||
/// and involves no timezone conversion. The zone, where the source recorded
|
||||
/// one, travels separately in `OffsetTimeOriginal`.
|
||||
pub(crate) fn datetime(unix: i64) -> String {
|
||||
let days = unix.div_euclid(86_400);
|
||||
let secs = unix.rem_euclid(86_400);
|
||||
|
||||
// Howard Hinnant's civil-from-days, the inverse of the days-from-civil
|
||||
// that `dr-decode` uses to parse. Eras of 400 years, shifted so that the
|
||||
// arithmetic never sees a negative.
|
||||
let z = days + 719_468;
|
||||
let era = z.div_euclid(146_097);
|
||||
let doe = z.rem_euclid(146_097);
|
||||
let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365;
|
||||
let y = yoe + era * 400;
|
||||
let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
|
||||
let mp = (5 * doy + 2) / 153;
|
||||
let d = doy - (153 * mp + 2) / 5 + 1;
|
||||
let m = if mp < 10 { mp + 3 } else { mp - 9 };
|
||||
let y = if m <= 2 { y + 1 } else { y };
|
||||
|
||||
format!(
|
||||
"{y:04}:{m:02}:{d:02} {:02}:{:02}:{:02}",
|
||||
secs / 3600,
|
||||
(secs / 60) % 60,
|
||||
secs % 60
|
||||
)
|
||||
}
|
||||
|
||||
/// Minutes east of UTC as EXIF's `"+HH:MM"`.
|
||||
pub(crate) fn offset(minutes: i32) -> String {
|
||||
let sign = if minutes < 0 { '-' } else { '+' };
|
||||
let m = minutes.unsigned_abs();
|
||||
format!("{sign}{:02}:{:02}", m / 60, m % 60)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use dr_types::Location;
|
||||
|
||||
#[test]
|
||||
fn a_capture_time_survives_the_round_trip_through_the_tag() {
|
||||
// The parse side lives in `dr-decode` and is exercised against real
|
||||
// files; this is the inverse, and the two meeting in the middle is
|
||||
// what keeps an exported frame on the same point of the timeline as
|
||||
// the original.
|
||||
assert_eq!(datetime(1_372_462_374), "2013:06:28 23:32:54");
|
||||
assert_eq!(datetime(0), "1970:01:01 00:00:00");
|
||||
// A leap day, which is where a hand-rolled calendar goes wrong.
|
||||
assert_eq!(datetime(1_709_164_800), "2024:02:29 00:00:00");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_zone_is_written_the_way_the_tag_spells_it() {
|
||||
assert_eq!(offset(120), "+02:00");
|
||||
assert_eq!(offset(-330), "-05:30");
|
||||
assert_eq!(offset(0), "+00:00");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_shutter_speed_keeps_the_photographers_fraction() {
|
||||
assert_eq!(shutter(1.0 / 250.0), (1, 250));
|
||||
assert_eq!(shutter(2.5), (25, 10));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn degrees_round_trip_through_the_tags_triple() {
|
||||
// 48.8582 N is the Eiffel Tower; the check is that the three-part
|
||||
// form comes back to the same place, to well under a metre.
|
||||
for degrees in [48.8582_f64, -33.8568, 0.0, 179.999] {
|
||||
let parts = dms(degrees);
|
||||
let back = parts[0].0 as f64
|
||||
+ parts[1].0 as f64 / 60.0
|
||||
+ (parts[2].0 as f64 / parts[2].1 as f64) / 3600.0;
|
||||
assert!(
|
||||
(back - degrees.abs()).abs() < 1e-6,
|
||||
"{degrees} came back as {back}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_empty_source_produces_no_block_at_all() {
|
||||
// Every field absent means the only entries would be the ones this
|
||||
// writer adds itself. That is still worth writing — `Software` and
|
||||
// the pixel dimensions are true statements — so the block exists; what
|
||||
// must not happen is a *malformed* one.
|
||||
let md = SourceMetadata::default();
|
||||
let bytes = block(&md, 100, 50).expect("the writer's own tags");
|
||||
assert!(bytes.starts_with(b"II*\0"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_gps_directory_is_absent_when_there_is_no_position() {
|
||||
let md = SourceMetadata {
|
||||
make: Some("Canon".into()),
|
||||
..Default::default()
|
||||
};
|
||||
let bytes = block(&md, 10, 10).unwrap();
|
||||
assert!(!contains_entry(&bytes, tag::GPS_IFD));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_gps_directory_is_present_when_there_is_one() {
|
||||
// The counterpart of the test above: a strip test that passed because
|
||||
// the writer could never emit GPS at all would prove nothing.
|
||||
let md = SourceMetadata {
|
||||
location: Location::new(48.8582, 2.2945, Some(35.0)),
|
||||
..Default::default()
|
||||
};
|
||||
let bytes = block(&md, 10, 10).unwrap();
|
||||
assert!(contains_entry(&bytes, tag::GPS_IFD));
|
||||
}
|
||||
|
||||
/// Whether a directory entry for `tag` appears anywhere in the block.
|
||||
///
|
||||
/// Byte-level on purpose: an entry is a tag, a type and a count, and
|
||||
/// searching for that twelve-byte shape's first eight bytes is a far
|
||||
/// stronger statement than asking a parser that might have skipped the
|
||||
/// directory the tag was in.
|
||||
fn contains_entry(bytes: &[u8], tag: u16) -> bool {
|
||||
bytes
|
||||
.windows(4)
|
||||
.any(|w| w[..2] == tag.to_le_bytes() && (w[2] == 4 || w[2] == 13) && w[3] == 0)
|
||||
}
|
||||
}
|
||||
@@ -1,483 +0,0 @@
|
||||
//! TRACES: FR-EXP-2
|
||||
//! Minimal ICC v2 matrix/TRC profiles, generated.
|
||||
//!
|
||||
//! # Why generated rather than shipped
|
||||
//!
|
||||
//! A profile is a description of what the pixels in a file mean, and the
|
||||
//! pixels here were produced by [`dr_types::colour`]'s matrices. Embedding a
|
||||
//! profile downloaded from elsewhere would mean two independent statements
|
||||
//! about the same space, agreeing until one of them was revised. Deriving both
|
||||
//! from the same primaries makes agreement structural.
|
||||
//!
|
||||
//! It is also the only pure-Rust route. Little-CMS is the obvious library and
|
||||
//! it is C, which the whole workspace avoids so it cross-compiles under the
|
||||
//! Android NDK — the same reasoning behind rustls, bundled SQLite and the
|
||||
//! Lensfun port.
|
||||
//!
|
||||
//! # What "minimal" leaves out
|
||||
//!
|
||||
//! A matrix/TRC display profile and nothing else: three colorants, three tone
|
||||
//! curves, a white point and the chromatic adaptation that got it there. No
|
||||
//! A2B/B2A lookup tables, no gamut tag, no named colours. That is the whole of
|
||||
//! what an RGB working space *is*, and it is what every reader — a browser, an
|
||||
//! operating system compositor, Photoshop — takes from a profile like this
|
||||
//! one. The tags omitted describe device behaviour these spaces do not have.
|
||||
//!
|
||||
//! Profiles come out around 2 KB, which matters more than it sounds: a JPEG
|
||||
//! carries the profile in APP2 segments capped at 64 KB each, and one that fits
|
||||
//! in a single segment avoids the chunked form that some older readers
|
||||
//! mishandle.
|
||||
|
||||
use dr_types::{ColourSpace, Transfer};
|
||||
|
||||
/// The ICC profile describing `space`, ready to embed.
|
||||
///
|
||||
/// Deterministic — the same space always produces the same bytes. Two exports
|
||||
/// of the same frame must be byte-identical files, which a creation timestamp
|
||||
/// read from the clock would quietly break, along with any deduplication
|
||||
/// downstream of it.
|
||||
pub fn profile(space: ColourSpace) -> Vec<u8> {
|
||||
let colorants = space.to_pcs_xyz();
|
||||
let trc = trc_curve(space.transfer());
|
||||
|
||||
// Sorted by signature, as the specification asks a tag table to be. Some
|
||||
// readers binary-search it.
|
||||
let mut tags: Vec<(&[u8; 4], Vec<u8>)> = vec![
|
||||
(b"bTRC", trc.clone()),
|
||||
// Columns, not rows: a colorant tag is where one primary lands in XYZ.
|
||||
(b"bXYZ", xyz_type(colorants[2], colorants[5], colorants[8])),
|
||||
(b"cprt", text_type(COPYRIGHT)),
|
||||
(b"desc", description_type(&description(space))),
|
||||
(b"gTRC", trc.clone()),
|
||||
(b"gXYZ", xyz_type(colorants[1], colorants[4], colorants[7])),
|
||||
(b"rTRC", trc),
|
||||
(b"rXYZ", xyz_type(colorants[0], colorants[3], colorants[6])),
|
||||
// The PCS illuminant itself, not the space's own white. The space's
|
||||
// white is recoverable from this and `chad`, and a profile that put
|
||||
// its native white here would have every reader adapt it twice.
|
||||
(b"wtpt", xyz_type(PCS_D50[0], PCS_D50[1], PCS_D50[2])),
|
||||
];
|
||||
|
||||
// Only where there is an adaptation to declare. ProPhoto is a D50 space
|
||||
// already, and an identity `chad` is a tag saying nothing.
|
||||
let adaptation = space.adaptation_to_pcs();
|
||||
if !is_identity(&adaptation) {
|
||||
tags.push((b"chad", sf32_type(&adaptation)));
|
||||
}
|
||||
tags.sort_by_key(|(sig, _)| **sig);
|
||||
|
||||
assemble(&tags)
|
||||
}
|
||||
|
||||
/// What a colour-management dialogue will show this profile as.
|
||||
///
|
||||
/// Deliberately not the canonical names. "sRGB IEC61966-2.1" is the reference
|
||||
/// profile, and this is not it — it is a profile derived from the same
|
||||
/// primaries, which is a different and weaker claim. "Adobe RGB (1998)" is
|
||||
/// additionally a name belonging to someone else. A distinct name also tells a
|
||||
/// user opening the file where the profile came from, which is the question
|
||||
/// they are asking when they look.
|
||||
fn description(space: ColourSpace) -> String {
|
||||
format!("DarkRoom {}", space.label())
|
||||
}
|
||||
|
||||
/// The copyright tag, which ICC requires a profile to carry.
|
||||
///
|
||||
/// A set of chromaticity coordinates from a published specification is not
|
||||
/// something to claim rights over, and a profile nobody may redistribute would
|
||||
/// make the files carrying it awkward to share — which is the whole purpose of
|
||||
/// an export.
|
||||
const COPYRIGHT: &str = "Generated by DarkRoom. No rights reserved.";
|
||||
|
||||
/// The profile connection space illuminant, as s15Fixed16 exactly.
|
||||
const PCS_D50: [f32; 3] = [0.9642, 1.0, 0.8249];
|
||||
|
||||
/// Samples in a tabulated tone curve.
|
||||
///
|
||||
/// 1024 is what the reference sRGB profiles use. The curve is interpolated
|
||||
/// linearly between samples, so this is far finer than the 8-bit values it
|
||||
/// describes; halving it would still be adequate and would save a kilobyte
|
||||
/// nobody is counting.
|
||||
const TRC_SAMPLES: usize = 1024;
|
||||
|
||||
/// A tone reproduction curve for the space's transfer function.
|
||||
///
|
||||
/// ICC curves run *towards* the connection space — device value to linear —
|
||||
/// which is the opposite direction from the shader's final encode. Getting it
|
||||
/// backwards produces a file that looks washed out or crushed by exactly the
|
||||
/// amount the curve bends.
|
||||
fn trc_curve(transfer: Transfer) -> Vec<u8> {
|
||||
// A pure power curve has an exact representation: a single u8Fixed8
|
||||
// gamma. Adobe RGB's 563/256 lands on it precisely, where a 1024-entry
|
||||
// table would be an approximation of a number the format can hold.
|
||||
if let Transfer::Gamma(g) = transfer {
|
||||
let mut out = tag_header(b"curv");
|
||||
out.extend_from_slice(&1u32.to_be_bytes());
|
||||
out.extend_from_slice(&((g * 256.0).round() as u16).to_be_bytes());
|
||||
return out;
|
||||
}
|
||||
|
||||
let mut out = tag_header(b"curv");
|
||||
out.extend_from_slice(&(TRC_SAMPLES as u32).to_be_bytes());
|
||||
for i in 0..TRC_SAMPLES {
|
||||
let device = i as f32 / (TRC_SAMPLES - 1) as f32;
|
||||
let linear = transfer.decode(device);
|
||||
out.extend_from_slice(&((linear * 65535.0).round() as u16).to_be_bytes());
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// An `XYZType` tag: one colour in the connection space.
|
||||
fn xyz_type(x: f32, y: f32, z: f32) -> Vec<u8> {
|
||||
let mut out = tag_header(b"XYZ ");
|
||||
for v in [x, y, z] {
|
||||
out.extend_from_slice(&s15_fixed16(v).to_be_bytes());
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// An `s15Fixed16ArrayType` tag, which is how `chad` is stored.
|
||||
fn sf32_type(m: &[f32; 9]) -> Vec<u8> {
|
||||
let mut out = tag_header(b"sf32");
|
||||
for v in m {
|
||||
out.extend_from_slice(&s15_fixed16(*v).to_be_bytes());
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// A `textType` tag: ASCII with a terminating NUL.
|
||||
fn text_type(s: &str) -> Vec<u8> {
|
||||
let mut out = tag_header(b"text");
|
||||
out.extend_from_slice(s.as_bytes());
|
||||
out.push(0);
|
||||
out
|
||||
}
|
||||
|
||||
/// A `textDescriptionType` tag — the v2 profile's name field.
|
||||
///
|
||||
/// Baroque, and not optional: v2 has no plain `mluc`, and the ASCII string is
|
||||
/// followed by empty Unicode and ScriptCode blocks that a reader will walk
|
||||
/// whether or not they hold anything. The 67-byte Macintosh field is fixed
|
||||
/// width by specification, so it is written out zeroed rather than omitted.
|
||||
fn description_type(s: &str) -> Vec<u8> {
|
||||
let ascii = s.as_bytes();
|
||||
let mut out = tag_header(b"desc");
|
||||
out.extend_from_slice(&(ascii.len() as u32 + 1).to_be_bytes());
|
||||
out.extend_from_slice(ascii);
|
||||
out.push(0);
|
||||
// Unicode language code, then Unicode character count: none of either.
|
||||
out.extend_from_slice(&[0; 8]);
|
||||
// ScriptCode code (u16), length (u8), and the fixed 67-byte field.
|
||||
out.extend_from_slice(&[0; 3]);
|
||||
out.extend_from_slice(&[0; 67]);
|
||||
out
|
||||
}
|
||||
|
||||
/// Every tag element opens with its type signature and four reserved bytes.
|
||||
fn tag_header(sig: &[u8; 4]) -> Vec<u8> {
|
||||
let mut out = Vec::from(*sig);
|
||||
out.extend_from_slice(&[0; 4]);
|
||||
out
|
||||
}
|
||||
|
||||
/// ICC's fixed-point number: 16 integer bits, 16 fractional.
|
||||
fn s15_fixed16(v: f32) -> i32 {
|
||||
(f64::from(v) * 65536.0).round() as i32
|
||||
}
|
||||
|
||||
fn is_identity(m: &[f32; 9]) -> bool {
|
||||
const IDENTITY: [f32; 9] = [1.0, 0.0, 0.0, 0.0, 1.0, 0.0, 0.0, 0.0, 1.0];
|
||||
// One step of s15Fixed16, the format the matrix would be stored in. Below
|
||||
// that it *is* the identity — ProPhoto's own white and the PCS illuminant
|
||||
// differ in the sixth decimal place, and a `chad` recording that would be
|
||||
// nine copies of 1.0000 and 0.0000 dressed up as information.
|
||||
const STEP: f32 = 1.0 / 65536.0;
|
||||
m.iter().zip(IDENTITY).all(|(a, b)| (a - b).abs() < STEP)
|
||||
}
|
||||
|
||||
/// Header, tag table, and the tag data, with the size written back in.
|
||||
fn assemble(tags: &[(&[u8; 4], Vec<u8>)]) -> Vec<u8> {
|
||||
let mut out = header();
|
||||
|
||||
out.extend_from_slice(&(tags.len() as u32).to_be_bytes());
|
||||
let table_at = out.len();
|
||||
out.resize(table_at + tags.len() * 12, 0);
|
||||
|
||||
for (i, (sig, data)) in tags.iter().enumerate() {
|
||||
// Identical elements share one copy, which the specification allows
|
||||
// explicitly. The three tone curves of a grey-balanced space are the
|
||||
// same 2 KB table, so this is two thirds of the profile.
|
||||
let offset = find(&out, data).unwrap_or_else(|| {
|
||||
let at = out.len();
|
||||
out.extend_from_slice(data);
|
||||
// Every element starts on a four-byte boundary.
|
||||
while !out.len().is_multiple_of(4) {
|
||||
out.push(0);
|
||||
}
|
||||
at
|
||||
});
|
||||
|
||||
let entry = table_at + i * 12;
|
||||
out[entry..entry + 4].copy_from_slice(*sig);
|
||||
out[entry + 4..entry + 8].copy_from_slice(&(offset as u32).to_be_bytes());
|
||||
out[entry + 8..entry + 12].copy_from_slice(&(data.len() as u32).to_be_bytes());
|
||||
}
|
||||
|
||||
let size = out.len() as u32;
|
||||
out[0..4].copy_from_slice(&size.to_be_bytes());
|
||||
out
|
||||
}
|
||||
|
||||
/// Where `needle` already sits in `haystack`, if it does.
|
||||
///
|
||||
/// Only ever called with tag elements, which begin on four-byte boundaries and
|
||||
/// start with a type signature — so a match cannot be a coincidental overlap
|
||||
/// of two other tags' bytes.
|
||||
fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
|
||||
haystack
|
||||
.windows(needle.len())
|
||||
.position(|w| w == needle)
|
||||
.filter(|at| at.is_multiple_of(4))
|
||||
}
|
||||
|
||||
/// The fixed 128-byte profile header.
|
||||
fn header() -> Vec<u8> {
|
||||
let mut h = Vec::with_capacity(128);
|
||||
// Size, filled in once the profile is complete.
|
||||
h.extend_from_slice(&[0; 4]);
|
||||
// Preferred CMM: no preference.
|
||||
h.extend_from_slice(&[0; 4]);
|
||||
// Version 2.1.0. v2 rather than v4 because it is what every reader
|
||||
// handles, and because nothing here needs a v4 tag type.
|
||||
h.extend_from_slice(&[0x02, 0x10, 0x00, 0x00]);
|
||||
h.extend_from_slice(b"mntr");
|
||||
h.extend_from_slice(b"RGB ");
|
||||
h.extend_from_slice(b"XYZ ");
|
||||
// Creation date. Fixed, for the determinism the module docs describe.
|
||||
for field in [2025u16, 1, 1, 0, 0, 0] {
|
||||
h.extend_from_slice(&field.to_be_bytes());
|
||||
}
|
||||
h.extend_from_slice(b"acsp");
|
||||
// Primary platform, flags, manufacturer, model, attributes: unspecified.
|
||||
h.extend_from_slice(&[0; 24]);
|
||||
// Rendering intent: perceptual, as the reference RGB working-space
|
||||
// profiles declare. For a matrix/TRC profile the field is advisory —
|
||||
// there is only one transform in here to apply.
|
||||
h.extend_from_slice(&[0; 4]);
|
||||
for v in PCS_D50 {
|
||||
h.extend_from_slice(&s15_fixed16(v).to_be_bytes());
|
||||
}
|
||||
// Creator, profile ID, and the reserved tail.
|
||||
h.extend_from_slice(&[0; 4]);
|
||||
h.extend_from_slice(&[0; 16]);
|
||||
h.extend_from_slice(&[0; 28]);
|
||||
debug_assert_eq!(h.len(), 128);
|
||||
h
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// A tag's element data, located through the profile's own tag table —
|
||||
/// so these tests read the profile the way a colour engine would rather
|
||||
/// than the way it was written.
|
||||
fn tag<'a>(profile: &'a [u8], want: &[u8; 4]) -> Option<&'a [u8]> {
|
||||
let count = u32::from_be_bytes(profile[128..132].try_into().unwrap()) as usize;
|
||||
for i in 0..count {
|
||||
let at = 132 + i * 12;
|
||||
if &profile[at..at + 4] == want {
|
||||
let off = u32::from_be_bytes(profile[at + 4..at + 8].try_into().unwrap()) as usize;
|
||||
let len = u32::from_be_bytes(profile[at + 8..at + 12].try_into().unwrap()) as usize;
|
||||
return Some(&profile[off..off + len]);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn xyz(data: &[u8]) -> [f32; 3] {
|
||||
let read =
|
||||
|at: usize| i32::from_be_bytes(data[at..at + 4].try_into().unwrap()) as f32 / 65536.0;
|
||||
[read(8), read(12), read(16)]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_profile_declares_its_own_length() {
|
||||
// The first field a reader trusts. A profile whose header says it is
|
||||
// longer than the buffer is one a strict parser rejects outright and a
|
||||
// lax one reads past the end of.
|
||||
for space in ColourSpace::ALL {
|
||||
let p = profile(space);
|
||||
let declared = u32::from_be_bytes(p[0..4].try_into().unwrap()) as usize;
|
||||
assert_eq!(declared, p.len(), "{space:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_profile_carries_the_signature_that_identifies_it_as_one() {
|
||||
// `acsp` at offset 36 is how every reader recognises an ICC profile.
|
||||
for space in ColourSpace::ALL {
|
||||
assert_eq!(&profile(space)[36..40], b"acsp", "{space:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_tag_lies_inside_the_profile_and_on_a_boundary() {
|
||||
// A tag table is offsets and lengths, and nothing checks them for us.
|
||||
// An off-by-four here produces a profile that parses as far as the
|
||||
// tag a reader happens to want.
|
||||
for space in ColourSpace::ALL {
|
||||
let p = profile(space);
|
||||
let count = u32::from_be_bytes(p[128..132].try_into().unwrap()) as usize;
|
||||
for i in 0..count {
|
||||
let at = 132 + i * 12;
|
||||
let off = u32::from_be_bytes(p[at + 4..at + 8].try_into().unwrap()) as usize;
|
||||
let len = u32::from_be_bytes(p[at + 8..at + 12].try_into().unwrap()) as usize;
|
||||
assert!(off.is_multiple_of(4), "{space:?} tag {i} starts at {off}");
|
||||
assert!(off + len <= p.len(), "{space:?} tag {i} runs off the end");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_profile_carries_the_tags_a_matrix_trc_profile_requires() {
|
||||
// The ICC v2 required set for a display profile. A reader missing any
|
||||
// one of these falls back to assuming sRGB, which is the silent
|
||||
// failure this whole feature exists to prevent.
|
||||
for space in ColourSpace::ALL {
|
||||
let p = profile(space);
|
||||
for required in [
|
||||
b"desc", b"cprt", b"wtpt", b"rXYZ", b"gXYZ", b"bXYZ", b"rTRC", b"gTRC", b"bTRC",
|
||||
] {
|
||||
assert!(tag(&p, required).is_some(), "{space:?} has no {required:?}");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_colorants_are_the_ones_the_shader_encoded_with() {
|
||||
// The property the file's honesty rests on. The composer converts the
|
||||
// pixels with `to_pcs_xyz`'s primaries; if the profile described any
|
||||
// others the file would be a precise, confident lie.
|
||||
for space in ColourSpace::ALL {
|
||||
let p = profile(space);
|
||||
let want = space.to_pcs_xyz();
|
||||
for (i, sig) in [b"rXYZ", b"gXYZ", b"bXYZ"].into_iter().enumerate() {
|
||||
let got = xyz(tag(&p, sig).expect("colorant"));
|
||||
for (row, g) in got.iter().enumerate() {
|
||||
let expected = want[row * 3 + i];
|
||||
assert!(
|
||||
(g - expected).abs() < 1e-4,
|
||||
"{space:?} {sig:?} row {row}: profile says {g}, shader used {expected}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_white_point_is_the_connection_space_illuminant() {
|
||||
// Not the space's own white. ProPhoto's is D50 anyway, but P3's is
|
||||
// D65, and a profile advertising D65 as its media white would have
|
||||
// every neutral adapted a second time.
|
||||
for space in ColourSpace::ALL {
|
||||
let got = xyz(tag(&profile(space), b"wtpt").expect("wtpt"));
|
||||
for (i, want) in PCS_D50.iter().enumerate() {
|
||||
assert!((got[i] - want).abs() < 1e-4, "{space:?} white {i}: {got:?}");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_tabulated_curve_reproduces_the_transfer_function_it_came_from() {
|
||||
// Read back out of the profile and compared against the function the
|
||||
// shader encodes with. The curve runs device-to-linear, and writing it
|
||||
// the other way round would still produce a monotonic curve of the
|
||||
// right length — this is what catches the direction.
|
||||
for space in [ColourSpace::Srgb, ColourSpace::ProPhoto] {
|
||||
let p = profile(space);
|
||||
let curve = tag(&p, b"rTRC").expect("rTRC");
|
||||
let count = u32::from_be_bytes(curve[8..12].try_into().unwrap()) as usize;
|
||||
assert_eq!(count, TRC_SAMPLES, "{space:?}");
|
||||
|
||||
let transfer = space.transfer();
|
||||
for i in [0, 1, count / 4, count / 2, count - 1] {
|
||||
let at = 12 + i * 2;
|
||||
let got =
|
||||
f32::from(u16::from_be_bytes(curve[at..at + 2].try_into().unwrap())) / 65535.0;
|
||||
let want = transfer.decode(i as f32 / (count - 1) as f32);
|
||||
assert!(
|
||||
(got - want).abs() < 1e-4,
|
||||
"{space:?} sample {i}: profile {got}, transfer {want}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adobe_rgb_stores_its_gamma_exactly_rather_than_sampling_it() {
|
||||
// 563/256 is representable in a u8Fixed8, so the curve is one number.
|
||||
// A 1024-entry table would approximate a value the format can hold
|
||||
// exactly, and would round-trip through other software as 2.2.
|
||||
let p = profile(ColourSpace::AdobeRgb);
|
||||
let curve = tag(&p, b"rTRC").expect("rTRC");
|
||||
assert_eq!(u32::from_be_bytes(curve[8..12].try_into().unwrap()), 1);
|
||||
assert_eq!(u16::from_be_bytes(curve[12..14].try_into().unwrap()), 563);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_three_tone_curves_share_one_copy() {
|
||||
// Not a size optimisation for its own sake: it keeps the profile under
|
||||
// the 64 KB a single JPEG APP2 segment holds, so the chunked form that
|
||||
// older readers mishandle is never needed.
|
||||
let p = profile(ColourSpace::Srgb);
|
||||
let count = u32::from_be_bytes(p[128..132].try_into().unwrap()) as usize;
|
||||
let offsets: Vec<u32> = ["rTRC", "gTRC", "bTRC"]
|
||||
.iter()
|
||||
.map(|sig| {
|
||||
(0..count)
|
||||
.map(|i| 132 + i * 12)
|
||||
.find(|at| &p[*at..at + 4] == sig.as_bytes())
|
||||
.map(|at| u32::from_be_bytes(p[at + 4..at + 8].try_into().unwrap()))
|
||||
.expect("curve present")
|
||||
})
|
||||
.collect();
|
||||
assert_eq!(offsets[0], offsets[1]);
|
||||
assert_eq!(offsets[1], offsets[2]);
|
||||
assert!(p.len() < 8 * 1024, "{} bytes is too large", p.len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_d65_space_declares_its_adaptation_and_a_d50_space_does_not() {
|
||||
// `chad` is what lets a reader recover the space's native white from
|
||||
// colorants that have already been adapted. Without it, D65 primaries
|
||||
// adapted to D50 and genuine D50 primaries are the same nine numbers.
|
||||
assert!(tag(&profile(ColourSpace::DisplayP3), b"chad").is_some());
|
||||
assert!(
|
||||
tag(&profile(ColourSpace::ProPhoto), b"chad").is_none(),
|
||||
"ProPhoto is a D50 space; an identity chad says nothing"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_same_space_always_produces_the_same_bytes() {
|
||||
// Two exports of one frame must be identical files. A creation
|
||||
// timestamp from the clock is the obvious way to lose that.
|
||||
for space in ColourSpace::ALL {
|
||||
assert_eq!(profile(space), profile(space), "{space:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn each_space_is_described_by_its_own_name() {
|
||||
// A file whose profile says "sRGB" while carrying P3 pixels is exactly
|
||||
// as misleading as no profile at all, and harder to notice.
|
||||
for space in ColourSpace::ALL {
|
||||
let p = profile(space);
|
||||
let desc = tag(&p, b"desc").expect("desc");
|
||||
let len = u32::from_be_bytes(desc[8..12].try_into().unwrap()) as usize;
|
||||
let name = std::str::from_utf8(&desc[12..12 + len - 1]).expect("ascii");
|
||||
assert_eq!(name, format!("DarkRoom {}", space.label()));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,392 +0,0 @@
|
||||
//! TRACES: FR-EXP-1 | FR-EXP-2 | FR-EXP-3 | FR-EXP-4 | FR-EXP-6 | FR-EXP-9 | R3
|
||||
//! Turning a rendered frame into a file's worth of bytes.
|
||||
//!
|
||||
//! # What this crate is, and is not
|
||||
//!
|
||||
//! It is: resize, output sharpening, encode, and the name the result should
|
||||
//! be given. It is not: a filesystem, a network client, or a job queue.
|
||||
//! [`export`] returns [`Encoded`] — bytes and a filename — and the caller
|
||||
//! decides where that lands.
|
||||
//!
|
||||
//! That boundary is not fastidiousness. An export has three possible
|
||||
//! destinations and they have nothing in common: a path on Linux, a Storage
|
||||
//! Access Framework document on Android where there *is* no path
|
||||
//! (ARCH §6.9), and a `PUT` to a Nextcloud folder. A crate that wrote the
|
||||
//! file itself would serve one of them and be rewritten for the other two.
|
||||
//!
|
||||
//! # Order of operations
|
||||
//!
|
||||
//! Resize, then sharpen, then encode. Sharpening after the resize is the
|
||||
//! whole point of output sharpening (FR-EXP-4): it compensates for the
|
||||
//! softening the resample introduced, so its strength has to scale with how
|
||||
//! much scaling actually happened. Sharpening first and then shrinking would
|
||||
//! throw the sharpened detail away.
|
||||
|
||||
use dr_types::{ColourSpace, ExportFormat, ExportSettings};
|
||||
|
||||
mod encode;
|
||||
mod error;
|
||||
mod exif;
|
||||
pub mod icc;
|
||||
mod metadata;
|
||||
mod name;
|
||||
mod sharpen;
|
||||
mod size;
|
||||
|
||||
pub use error::ExportError;
|
||||
pub use metadata::SourceMetadata;
|
||||
pub use name::{resolve_name, NameContext};
|
||||
pub use size::target_size;
|
||||
|
||||
/// A rendered frame, as the adjust pass produced it.
|
||||
///
|
||||
/// 8-bit RGBA, display-encoded in [`Self::space`] — the format
|
||||
/// [`dr_gpu::AdjustPass`](../dr_gpu/struct.AdjustPass.html) writes. Alpha is
|
||||
/// carried but never meaningful: the pipeline writes 1.0 everywhere, and no
|
||||
/// operation produces transparency.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Frame {
|
||||
pub width: u32,
|
||||
pub height: u32,
|
||||
/// Tightly packed RGBA8, `width * height * 4` bytes.
|
||||
pub rgba: Vec<u8>,
|
||||
/// TRACES: FR-EXP-2
|
||||
/// The space the shader encoded these pixels into.
|
||||
///
|
||||
/// Travels with the pixels rather than being asserted at the point of
|
||||
/// encoding, because it is a fact about them and not a preference. The
|
||||
/// conversion happened in the generated shader, before the clip to 0..1,
|
||||
/// and nothing downstream can undo or redo it — a frame clipped to sRGB
|
||||
/// has already lost whatever a wider space would have carried.
|
||||
///
|
||||
/// Making it a field is what lets [`export`] refuse to label a frame as
|
||||
/// something it is not, rather than trusting a caller to have rendered
|
||||
/// what it asked for.
|
||||
pub space: ColourSpace,
|
||||
}
|
||||
|
||||
impl Frame {
|
||||
/// A frame the pipeline rendered in sRGB — what
|
||||
/// [`EditGraph::compose`](../dr_pipeline/struct.EditGraph.html#method.compose)
|
||||
/// produces, and so what the display path hands over.
|
||||
///
|
||||
/// An export in a wider space must render its own frame with
|
||||
/// `compose_for` and declare it through [`Self::in_space`]. Defaulting
|
||||
/// here rather than demanding the space at every call site keeps the
|
||||
/// common case honest by construction: a caller that has not thought
|
||||
/// about colour is describing sRGB, and sRGB is what it rendered.
|
||||
pub fn new(width: u32, height: u32, rgba: Vec<u8>) -> Result<Self, ExportError> {
|
||||
Self::in_space(width, height, rgba, ColourSpace::Srgb)
|
||||
}
|
||||
|
||||
/// A frame rendered into a stated colour space.
|
||||
pub fn in_space(
|
||||
width: u32,
|
||||
height: u32,
|
||||
rgba: Vec<u8>,
|
||||
space: ColourSpace,
|
||||
) -> Result<Self, ExportError> {
|
||||
let expected = width as usize * height as usize * 4;
|
||||
if rgba.len() != expected {
|
||||
return Err(ExportError::FrameSize {
|
||||
expected,
|
||||
got: rgba.len(),
|
||||
});
|
||||
}
|
||||
if width == 0 || height == 0 {
|
||||
return Err(ExportError::EmptyFrame);
|
||||
}
|
||||
Ok(Self {
|
||||
width,
|
||||
height,
|
||||
rgba,
|
||||
space,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// The finished article: what to write, and what to call it.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Encoded {
|
||||
/// Filename including extension. Never a path — the destination folder is
|
||||
/// the caller's, and on Android it is not expressible as one anyway.
|
||||
pub name: String,
|
||||
pub bytes: Vec<u8>,
|
||||
/// What the image was actually written at, after sizing and the upscaling
|
||||
/// guard. Worth reporting: a batch that silently exported at source size
|
||||
/// because the request was larger has done something the user should know.
|
||||
pub width: u32,
|
||||
pub height: u32,
|
||||
}
|
||||
|
||||
/// Resize, sharpen and encode one frame.
|
||||
///
|
||||
/// `name` is the filename already resolved by [`resolve_name`] — passed in
|
||||
/// rather than derived here because resolving it needs to know what is
|
||||
/// already in the destination, which this crate cannot see.
|
||||
///
|
||||
/// TRACES: FR-EXP-9
|
||||
/// The frame is expected to be a **full-resolution** render. Nothing here
|
||||
/// enforces that, because nothing here can tell a full render from a
|
||||
/// viewport-sized one; the caller renders at the framed output size and this
|
||||
/// resamples down from it. Exporting from the display proxy would silently
|
||||
/// produce a soft file, which is why the develop session's export path renders
|
||||
/// its own frame rather than reusing the one on screen.
|
||||
///
|
||||
/// TRACES: FR-EXP-8
|
||||
/// `source` is what the photograph's own file said about itself, or `None`
|
||||
/// where the caller has nothing — a frame that came from somewhere other than
|
||||
/// a decoded file, or a caller that has not yet been taught to pass it.
|
||||
///
|
||||
/// **A parameter rather than a field on [`Frame`]**, because it is not a fact
|
||||
/// about the pixels: two exports of the same frame can legitimately disclose
|
||||
/// different amounts, and the settings that decide how much travel beside it.
|
||||
/// It is also why this is an argument and not an `Option` with a default — a
|
||||
/// caller that has the source metadata should have to decide, in one visible
|
||||
/// place, to hand it over.
|
||||
pub fn export(
|
||||
frame: &Frame,
|
||||
settings: &ExportSettings,
|
||||
name: String,
|
||||
source: Option<&SourceMetadata>,
|
||||
) -> Result<Encoded, ExportError> {
|
||||
// TRACES: FR-EXP-2
|
||||
// Refused rather than mislabelled. Every space the settings page offers
|
||||
// now works, but only if the *frame* was rendered into it: the conversion
|
||||
// and the clip both happen in the generated shader, so pixels that arrive
|
||||
// clipped to sRGB have already lost whatever a wider space would have
|
||||
// carried, and no amount of profile-writing here brings it back.
|
||||
//
|
||||
// The caller's fix is to compose with `EditGraph::compose_for(space)`
|
||||
// before rendering. Until it does, this is an accurate error where the
|
||||
// alternative would be a file that claims a gamut it does not contain —
|
||||
// and that claim survives into everything downstream.
|
||||
if frame.space != settings.colour_space {
|
||||
return Err(ExportError::ColourSpaceMismatch {
|
||||
rendered: frame.space,
|
||||
requested: settings.colour_space,
|
||||
});
|
||||
}
|
||||
|
||||
if matches!(settings.format, ExportFormat::Avif | ExportFormat::JpegXl) {
|
||||
return Err(ExportError::FormatUnsupported(settings.format));
|
||||
}
|
||||
|
||||
let (width, height) = size::target_size(
|
||||
frame.width,
|
||||
frame.height,
|
||||
settings.sizing,
|
||||
settings.allow_upscaling,
|
||||
);
|
||||
|
||||
// TRACES: FR-EXP-3
|
||||
// One mode promises exact dimensions rather than a bound on them, and it
|
||||
// is the only place the fit/fill distinction survives: `target_size` has
|
||||
// already reported what the file will be either way.
|
||||
let resized = if settings.sizing.crops_to_fill() {
|
||||
size::resample_filling(frame, width, height)
|
||||
} else {
|
||||
size::resample(frame, width, height)
|
||||
};
|
||||
|
||||
// Scaled by how much the image actually shrank: a full-size export needs
|
||||
// no compensation, and a thumbnail needs a great deal.
|
||||
let scale = width as f32 / frame.width.max(1) as f32;
|
||||
let sharpened = sharpen::apply(resized, width, height, settings.sharpening, scale);
|
||||
|
||||
let bytes = encode::encode(&sharpened, width, height, settings, source)?;
|
||||
|
||||
Ok(Encoded {
|
||||
name,
|
||||
bytes,
|
||||
width,
|
||||
height,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use dr_types::SizingMode;
|
||||
|
||||
/// A frame with a recognisable gradient, so a resample can be checked for
|
||||
/// having done something rather than merely returned the right length.
|
||||
pub(crate) fn frame(w: u32, h: u32) -> Frame {
|
||||
let mut rgba = Vec::with_capacity((w * h * 4) as usize);
|
||||
for y in 0..h {
|
||||
for x in 0..w {
|
||||
rgba.push((x * 255 / w.max(1)) as u8);
|
||||
rgba.push((y * 255 / h.max(1)) as u8);
|
||||
rgba.push(128);
|
||||
rgba.push(255);
|
||||
}
|
||||
}
|
||||
Frame::new(w, h, rgba).expect("well-formed")
|
||||
}
|
||||
|
||||
fn settings(format: ExportFormat) -> ExportSettings {
|
||||
ExportSettings {
|
||||
format,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_frame_rejects_a_buffer_of_the_wrong_length() {
|
||||
// The one error that would otherwise surface as a panic deep in an
|
||||
// encoder, or worse, as a file of garbage.
|
||||
assert!(matches!(
|
||||
Frame::new(4, 4, vec![0; 10]),
|
||||
Err(ExportError::FrameSize { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn jpeg_export_produces_a_jpeg() {
|
||||
let out = export(
|
||||
&frame(64, 48),
|
||||
&settings(ExportFormat::Jpeg),
|
||||
"a.jpg".into(),
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
// SOI marker. Cheap, and it catches an encoder wired to the wrong
|
||||
// format far more directly than a byte count would.
|
||||
assert_eq!(&out.bytes[..2], &[0xFF, 0xD8]);
|
||||
assert_eq!((out.width, out.height), (64, 48));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn png_export_produces_a_png() {
|
||||
let out = export(
|
||||
&frame(32, 32),
|
||||
&settings(ExportFormat::Png),
|
||||
"a.png".into(),
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(&out.bytes[..8], b"\x89PNG\r\n\x1a\n");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiff_exports_produce_a_tiff() {
|
||||
for format in [ExportFormat::Tiff8, ExportFormat::Tiff16] {
|
||||
let out = export(&frame(16, 16), &settings(format), "a.tif".into(), None).unwrap();
|
||||
// Either byte order is a valid TIFF; the crate writes little-endian.
|
||||
assert!(
|
||||
out.bytes.starts_with(b"II*\0") || out.bytes.starts_with(b"MM\0*"),
|
||||
"{format:?} did not produce a TIFF header"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_sixteen_bit_tiff_is_larger_than_an_eight_bit_one() {
|
||||
// Both are uncompressed RGB; the only difference is the sample width,
|
||||
// so this is what proves the 16-bit path is not quietly writing 8.
|
||||
let eight = export(
|
||||
&frame(16, 16),
|
||||
&settings(ExportFormat::Tiff8),
|
||||
"a".into(),
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
let sixteen = export(
|
||||
&frame(16, 16),
|
||||
&settings(ExportFormat::Tiff16),
|
||||
"a".into(),
|
||||
None,
|
||||
)
|
||||
.unwrap();
|
||||
assert!(sixteen.bytes.len() > eight.bytes.len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn quality_changes_the_size_of_a_jpeg() {
|
||||
// The setting is plumbed all the way to the encoder rather than
|
||||
// accepted and dropped, which a size-independent output would show.
|
||||
let mut low = settings(ExportFormat::Jpeg);
|
||||
low.quality = 20;
|
||||
let mut high = settings(ExportFormat::Jpeg);
|
||||
high.quality = 98;
|
||||
|
||||
let small = export(&frame(128, 128), &low, "a".into(), None).unwrap();
|
||||
let large = export(&frame(128, 128), &high, "a".into(), None).unwrap();
|
||||
assert!(
|
||||
large.bytes.len() > small.bytes.len(),
|
||||
"quality 98 produced {} bytes against quality 20's {}",
|
||||
large.bytes.len(),
|
||||
small.bytes.len()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_long_edge_export_lands_on_the_requested_size() {
|
||||
let mut s = settings(ExportFormat::Png);
|
||||
s.sizing = SizingMode::LongEdge(32);
|
||||
let out = export(&frame(128, 64), &s, "a".into(), None).unwrap();
|
||||
assert_eq!((out.width, out.height), (32, 16));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_frame_rendered_in_one_space_is_not_labelled_another() {
|
||||
// A file tagged Display P3 carrying sRGB-clipped pixels is a lie that
|
||||
// survives into everything downstream. The frame carries the space it
|
||||
// was rendered in precisely so this cannot be waved through.
|
||||
let mut s = settings(ExportFormat::Jpeg);
|
||||
s.colour_space = ColourSpace::DisplayP3;
|
||||
assert!(matches!(
|
||||
export(&frame(8, 8), &s, "a".into(), None),
|
||||
Err(ExportError::ColourSpaceMismatch { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_colour_space_exports_when_the_frame_was_rendered_in_it() {
|
||||
// The other side of the refusal above, and what FR-EXP-2 actually
|
||||
// asks for: a frame the pipeline encoded into a wide space reaches a
|
||||
// file, in every format that has an encoder.
|
||||
for space in ColourSpace::ALL {
|
||||
for format in [
|
||||
ExportFormat::Jpeg,
|
||||
ExportFormat::Png,
|
||||
ExportFormat::Tiff8,
|
||||
ExportFormat::Tiff16,
|
||||
] {
|
||||
let mut s = settings(format);
|
||||
s.colour_space = space;
|
||||
let mut f = frame(8, 8);
|
||||
f.space = space;
|
||||
let out = export(&f, &s, "a".into(), None)
|
||||
.unwrap_or_else(|e| panic!("{space:?} as {format:?}: {e}"));
|
||||
assert!(!out.bytes.is_empty());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_formats_without_an_encoder_say_so() {
|
||||
for format in [ExportFormat::Avif, ExportFormat::JpegXl] {
|
||||
assert!(
|
||||
matches!(
|
||||
export(&frame(8, 8), &settings(format), "a".into(), None),
|
||||
Err(ExportError::FormatUnsupported(_))
|
||||
),
|
||||
"{format:?} should report that it has no encoder yet"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_offered_format_either_encodes_or_explains_itself() {
|
||||
// Walks `ExportFormat::ALL`, so a format added to the settings page
|
||||
// cannot quietly reach an encoder that does not handle it.
|
||||
for format in ExportFormat::ALL {
|
||||
match export(&frame(8, 8), &settings(format), "a".into(), None) {
|
||||
Ok(out) => assert!(!out.bytes.is_empty(), "{format:?} encoded to nothing"),
|
||||
Err(ExportError::FormatUnsupported(f)) => assert_eq!(f, format),
|
||||
Err(e) => panic!("{format:?} failed unexpectedly: {e}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,106 +0,0 @@
|
||||
//! TRACES: FR-EXP-8
|
||||
//! What an export is allowed to say about where it came from.
|
||||
//!
|
||||
//! # An allowlist, not a filter
|
||||
//!
|
||||
//! [`SourceMetadata`] is the whole of what can reach a file this crate writes.
|
||||
//! It is populated field by field from whatever the caller decoded, and
|
||||
//! nothing else travels — not because each unwanted tag is removed, but
|
||||
//! because there is nowhere in this type for one to sit. That is the
|
||||
//! difference between "we strip GPS" and "GPS cannot be written unless
|
||||
//! [`SourceMetadata::location`] is `Some`", and only the second survives
|
||||
//! somebody adding a field to the decoder next year.
|
||||
//!
|
||||
//! # What is deliberately not here
|
||||
//!
|
||||
//! **The maker note** (EXIF `0x927C`). It is an opaque vendor blob with no
|
||||
//! public format, and its contents differ by body and firmware. Canon's
|
||||
//! carries the body serial number and the shutter count; several bodies put a
|
||||
//! *duplicate copy of the GPS fix* inside it, which is the specific reason it
|
||||
//! cannot be passed through as an unexamined byte range: an export that
|
||||
//! stripped the GPS directory and copied the maker note would have published
|
||||
//! the coordinates anyway, while reporting itself as private. Parsing it per
|
||||
//! vendor to decide what is safe is a research project with a permanent
|
||||
//! maintenance cost, and the value on the other side is a few tags a
|
||||
//! photographer rarely misses. So it is dropped, in both directions, whatever
|
||||
//! the settings say.
|
||||
//!
|
||||
//! **Serial numbers and owner name** (`BodySerialNumber` 0xA431,
|
||||
//! `LensSerialNumber` 0xA435, `CameraOwnerName` 0xA430). These identify a
|
||||
//! person and a specific piece of equipment, and a serial number in a
|
||||
//! published file links every photograph that person has ever posted. They
|
||||
//! have no field here, so no export writes them.
|
||||
//!
|
||||
//! **IPTC and XMP.** FR-EXP-8 names both. Neither is read by `dr-decode`
|
||||
//! today, so there is nothing to carry through; when there is, it arrives as
|
||||
//! fields on this type and is written from them, and the same allowlist
|
||||
//! reasoning applies unchanged.
|
||||
|
||||
use dr_types::Location;
|
||||
|
||||
/// TRACES: FR-EXP-8
|
||||
/// The source metadata an export may carry.
|
||||
///
|
||||
/// Every field is optional because every field is genuinely absent from some
|
||||
/// real file: scanner output has no aperture, a JPEG from a phone has no lens
|
||||
/// model, and most photographs have no copyright statement at all.
|
||||
///
|
||||
/// Built by the caller, which is the only place that has both the decoded
|
||||
/// source and the crate that decoded it — `dr-export` deliberately depends on
|
||||
/// no decoder (see the crate docs), so the copy is made one field at a time
|
||||
/// where both types are in scope. That transcription is a feature: it is the
|
||||
/// point where somebody has to decide, in writing, that a newly-parsed piece
|
||||
/// of the source is allowed to leave the machine.
|
||||
#[derive(Debug, Clone, Default, PartialEq)]
|
||||
pub struct SourceMetadata {
|
||||
pub make: Option<String>,
|
||||
pub model: Option<String>,
|
||||
pub lens: Option<String>,
|
||||
/// Exposure time in seconds.
|
||||
pub shutter: Option<f32>,
|
||||
/// The f-number, as in f/2.8.
|
||||
pub aperture: Option<f32>,
|
||||
pub iso: Option<u32>,
|
||||
/// Millimetres, as marked on the lens rather than 35 mm equivalent.
|
||||
pub focal_length: Option<f32>,
|
||||
/// When the shutter fired, as Unix seconds read as a wall clock.
|
||||
pub captured_at: Option<i64>,
|
||||
/// Minutes east of UTC, where the camera recorded a zone.
|
||||
pub captured_offset: Option<i32>,
|
||||
/// Who made the photograph.
|
||||
pub artist: Option<String>,
|
||||
/// The rights statement.
|
||||
pub copyright: Option<String>,
|
||||
/// TRACES: FR-EXP-8
|
||||
/// Where the shutter fired.
|
||||
///
|
||||
/// The one field the strip option is about. It is carried this far so that
|
||||
/// a photographer who *wants* their coordinates can have them; by the time
|
||||
/// the encoder sees the record this field has already been through
|
||||
/// [`Self::sanitised`], and is `None` unless the user turned stripping
|
||||
/// off.
|
||||
pub location: Option<Location>,
|
||||
}
|
||||
|
||||
impl SourceMetadata {
|
||||
/// This record as the settings permit it to be written.
|
||||
///
|
||||
/// **The single place stripping happens.** The encoders below take a
|
||||
/// record and write what is in it, with no view on privacy; concentrating
|
||||
/// the decision here means there is one function to read to know what an
|
||||
/// export can disclose, and no format can quietly disagree with the
|
||||
/// others — the failure mode where JPEG honours the setting and TIFF, five
|
||||
/// hundred lines away, does not.
|
||||
///
|
||||
/// Stripping empties the field rather than blanking it. A `GPSLatitude` of
|
||||
/// `0/0` still announces that the camera had a fix and that this file has
|
||||
/// been through a scrubber; an absent directory says nothing at all, and
|
||||
/// says it in the same shape as the millions of files that never had one.
|
||||
pub(crate) fn sanitised(&self, strip_location: bool) -> Self {
|
||||
let mut out = self.clone();
|
||||
if strip_location {
|
||||
out.location = None;
|
||||
}
|
||||
out
|
||||
}
|
||||
}
|
||||
@@ -1,316 +0,0 @@
|
||||
//! TRACES: FR-EXP-6
|
||||
//! Filename templates and what to do when the name is taken.
|
||||
//!
|
||||
//! # Why the caller supplies the "does this exist" test
|
||||
//!
|
||||
//! [`resolve_name`] takes a closure rather than looking at a directory,
|
||||
//! because there is no directory it could look at that would work everywhere.
|
||||
//! A destination is a path on Linux, a Storage Access Framework tree on
|
||||
//! Android with no path at all (ARCH §6.9), or a folder on a Nextcloud
|
||||
//! server reached by PROPFIND. All three can answer "is this name taken",
|
||||
//! and none of them can be asked the same way.
|
||||
//!
|
||||
//! It matters most on Android, where the platform actively works against us:
|
||||
//! `DocumentsContract.createDocument` renames on collision *by itself*,
|
||||
//! appending ` (1)` and returning a URI with a name nobody asked for, and it
|
||||
//! cannot overwrite at all. So every one of the three [`CollisionPolicy`]
|
||||
//! settings requires knowing the answer before creating anything — which is
|
||||
//! exactly what this function is shaped for.
|
||||
|
||||
use dr_types::{CollisionPolicy, ExportFormat};
|
||||
|
||||
/// What a template can refer to.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct NameContext<'a> {
|
||||
/// The source image's name, without extension — `{name}`.
|
||||
pub source_stem: &'a str,
|
||||
/// Position in the batch, 1-based — `{seq}`.
|
||||
pub sequence: u32,
|
||||
/// Capture date as `YYYY-MM-DD` — `{date}`. Empty where unknown.
|
||||
pub date: &'a str,
|
||||
/// The export's pixel dimensions — `{dimensions}`.
|
||||
pub width: u32,
|
||||
pub height: u32,
|
||||
/// The preset that produced this export — `{preset}`. Empty where none.
|
||||
pub preset: &'a str,
|
||||
}
|
||||
|
||||
/// Expand a template into a filename stem.
|
||||
///
|
||||
/// Unknown tokens are left verbatim rather than dropped. A user who typed
|
||||
/// `{nmae}` should see it in the output and understand what happened; a
|
||||
/// silently empty filename is a puzzle, and a template that quietly loses a
|
||||
/// token produces a directory of files named the same thing.
|
||||
pub fn expand(template: &str, ctx: &NameContext<'_>) -> String {
|
||||
let seq = ctx.sequence.to_string();
|
||||
let dimensions = format!("{}x{}", ctx.width, ctx.height);
|
||||
|
||||
let mut out = String::with_capacity(template.len() + 16);
|
||||
let mut rest = template;
|
||||
while let Some(open) = rest.find('{') {
|
||||
out.push_str(&rest[..open]);
|
||||
let Some(close) = rest[open..].find('}') else {
|
||||
// An unclosed brace is literal text; there is nothing to expand
|
||||
// and dropping the remainder would truncate the name. Consumed
|
||||
// here rather than left for the tail append below, which has
|
||||
// already had everything before the brace taken from it.
|
||||
out.push_str(&rest[open..]);
|
||||
rest = "";
|
||||
break;
|
||||
};
|
||||
let token = &rest[open + 1..open + close];
|
||||
match token {
|
||||
"name" => out.push_str(ctx.source_stem),
|
||||
"seq" => out.push_str(&seq),
|
||||
"date" => out.push_str(ctx.date),
|
||||
"dimensions" => out.push_str(&dimensions),
|
||||
"preset" => out.push_str(ctx.preset),
|
||||
_ => out.push_str(&rest[open..open + close + 1]),
|
||||
}
|
||||
rest = &rest[open + close + 1..];
|
||||
}
|
||||
out.push_str(rest);
|
||||
|
||||
let cleaned = sanitise(&out);
|
||||
if cleaned.is_empty() {
|
||||
// Every token was empty — a template of `{preset}` with no preset, on
|
||||
// an image with no date. Falling back to the source name is the one
|
||||
// answer that is always available and never collides more than the
|
||||
// source files themselves do.
|
||||
return sanitise(ctx.source_stem);
|
||||
}
|
||||
cleaned
|
||||
}
|
||||
|
||||
/// Strip what no filesystem, SAF provider or WebDAV server will take.
|
||||
///
|
||||
/// The intersection of three sets of rules rather than any one of them: an
|
||||
/// export written to a Nextcloud folder may later sync down to a Windows
|
||||
/// client, and a name that was legal where it was created is not much comfort
|
||||
/// on the machine that cannot open it.
|
||||
fn sanitise(stem: &str) -> String {
|
||||
let mut out: String = stem
|
||||
.chars()
|
||||
.map(|c| match c {
|
||||
'/' | '\\' | ':' | '*' | '?' | '"' | '<' | '>' | '|' => '-',
|
||||
c if (c as u32) < 0x20 => '-',
|
||||
c => c,
|
||||
})
|
||||
.collect();
|
||||
// Trailing dots and spaces are legal on Linux and rejected by Windows,
|
||||
// and a name ending in one is almost always an accident of a template
|
||||
// whose last token expanded to nothing.
|
||||
while out.ends_with('.') || out.ends_with(' ') {
|
||||
out.pop();
|
||||
}
|
||||
out.trim_start().to_string()
|
||||
}
|
||||
|
||||
/// The filename this export should be written under, honouring the collision
|
||||
/// policy.
|
||||
///
|
||||
/// `taken` answers whether a name already exists in the destination. Returns
|
||||
/// `None` for [`CollisionPolicy::Skip`] when the name is in use — the caller
|
||||
/// writes nothing and moves on, which is the whole point of that setting.
|
||||
pub fn resolve_name(
|
||||
template: &str,
|
||||
ctx: &NameContext<'_>,
|
||||
format: ExportFormat,
|
||||
collision: CollisionPolicy,
|
||||
taken: &dyn Fn(&str) -> bool,
|
||||
) -> Option<String> {
|
||||
let stem = expand(template, ctx);
|
||||
let ext = format.extension();
|
||||
let first = format!("{stem}.{ext}");
|
||||
|
||||
if !taken(&first) {
|
||||
return Some(first);
|
||||
}
|
||||
|
||||
match collision {
|
||||
CollisionPolicy::Overwrite => Some(first),
|
||||
CollisionPolicy::Skip => None,
|
||||
CollisionPolicy::Increment => {
|
||||
// Bounded. An unbounded search would spin forever against a
|
||||
// destination that reports everything as taken — a permission
|
||||
// error misread as existence, say — and a batch that hangs is
|
||||
// worse than one that reports a failure.
|
||||
for n in 1..10_000 {
|
||||
let candidate = format!("{stem}-{n}.{ext}");
|
||||
if !taken(&candidate) {
|
||||
return Some(candidate);
|
||||
}
|
||||
}
|
||||
log::warn!("{stem}: ten thousand names taken; skipping");
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn ctx() -> NameContext<'static> {
|
||||
NameContext {
|
||||
source_stem: "IMG_1234",
|
||||
sequence: 7,
|
||||
date: "2026-08-16",
|
||||
width: 2048,
|
||||
height: 1365,
|
||||
preset: "Web",
|
||||
}
|
||||
}
|
||||
|
||||
fn free(_: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_default_template_is_the_source_name() {
|
||||
assert_eq!(expand("{name}", &ctx()), "IMG_1234");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn every_documented_token_expands() {
|
||||
// The settings page advertises these five in its hint; a token listed
|
||||
// there and unhandled here would reach the filename verbatim.
|
||||
assert_eq!(expand("{name}", &ctx()), "IMG_1234");
|
||||
assert_eq!(expand("{seq}", &ctx()), "7");
|
||||
assert_eq!(expand("{date}", &ctx()), "2026-08-16");
|
||||
assert_eq!(expand("{dimensions}", &ctx()), "2048x1365");
|
||||
assert_eq!(expand("{preset}", &ctx()), "Web");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tokens_combine_with_literal_text() {
|
||||
assert_eq!(
|
||||
expand("{date}_{name}_{dimensions}", &ctx()),
|
||||
"2026-08-16_IMG_1234_2048x1365"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unknown_token_survives_verbatim() {
|
||||
// A typo the user can see and fix, rather than a name that silently
|
||||
// lost a component and now collides with every other export.
|
||||
assert_eq!(expand("{nmae}-x", &ctx()), "{nmae}-x");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unclosed_brace_is_literal_text() {
|
||||
assert_eq!(expand("{name", &ctx()), "{name");
|
||||
assert_eq!(expand("a{name}b{", &ctx()), "aIMG_1234b{");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_template_that_expands_to_nothing_falls_back_to_the_source_name() {
|
||||
// `{preset}` with no preset selected. An empty filename is not a file.
|
||||
let mut c = ctx();
|
||||
c.preset = "";
|
||||
assert_eq!(expand("{preset}", &c), "IMG_1234");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn path_separators_cannot_escape_the_destination() {
|
||||
// `{name}` comes from a source filename, and a template is user text.
|
||||
// Either could carry a slash, and an export must not write outside
|
||||
// the folder that was chosen — nor create a subfolder on the server.
|
||||
let mut c = ctx();
|
||||
c.source_stem = "holiday/2026";
|
||||
assert_eq!(expand("{name}", &c), "holiday-2026");
|
||||
assert_eq!(expand("../../etc/passwd", &ctx()), "..-..-etc-passwd");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn characters_windows_rejects_are_replaced() {
|
||||
// An export may sync down to a Windows client through Nextcloud, and
|
||||
// a name that was legal where it was written is no comfort there.
|
||||
assert_eq!(expand(r#"a:b*c?d"e<f>g|h\i"#, &ctx()), "a-b-c-d-e-f-g-h-i");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trailing_dots_and_spaces_are_trimmed() {
|
||||
let mut c = ctx();
|
||||
c.preset = "";
|
||||
assert_eq!(expand("{name}.{preset}", &c), "IMG_1234");
|
||||
assert_eq!(expand("{name} ", &ctx()), "IMG_1234");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_free_name_is_used_as_is() {
|
||||
let got = resolve_name(
|
||||
"{name}",
|
||||
&ctx(),
|
||||
ExportFormat::Jpeg,
|
||||
CollisionPolicy::Increment,
|
||||
&free,
|
||||
);
|
||||
assert_eq!(got.as_deref(), Some("IMG_1234.jpg"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_extension_follows_the_format() {
|
||||
for (format, ext) in [
|
||||
(ExportFormat::Jpeg, "jpg"),
|
||||
(ExportFormat::Png, "png"),
|
||||
(ExportFormat::Tiff16, "tif"),
|
||||
] {
|
||||
let got = resolve_name("{name}", &ctx(), format, CollisionPolicy::Skip, &free);
|
||||
assert_eq!(got.as_deref(), Some(&*format!("IMG_1234.{ext}")));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn increment_finds_the_first_free_suffix() {
|
||||
let taken = |n: &str| matches!(n, "IMG_1234.jpg" | "IMG_1234-1.jpg" | "IMG_1234-2.jpg");
|
||||
let got = resolve_name(
|
||||
"{name}",
|
||||
&ctx(),
|
||||
ExportFormat::Jpeg,
|
||||
CollisionPolicy::Increment,
|
||||
&taken,
|
||||
);
|
||||
assert_eq!(got.as_deref(), Some("IMG_1234-3.jpg"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn skip_returns_nothing_when_the_name_is_taken() {
|
||||
// The caller writes no file at all — that is what Skip means, and it
|
||||
// is why this returns an Option rather than always a name.
|
||||
let got = resolve_name(
|
||||
"{name}",
|
||||
&ctx(),
|
||||
ExportFormat::Jpeg,
|
||||
CollisionPolicy::Skip,
|
||||
&|_| true,
|
||||
);
|
||||
assert_eq!(got, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn overwrite_returns_the_taken_name() {
|
||||
let got = resolve_name(
|
||||
"{name}",
|
||||
&ctx(),
|
||||
ExportFormat::Jpeg,
|
||||
CollisionPolicy::Overwrite,
|
||||
&|_| true,
|
||||
);
|
||||
assert_eq!(got.as_deref(), Some("IMG_1234.jpg"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn increment_gives_up_rather_than_spinning_forever() {
|
||||
// A destination that reports every name as taken — a permission error
|
||||
// misread as existence — must not hang the batch.
|
||||
let got = resolve_name(
|
||||
"{name}",
|
||||
&ctx(),
|
||||
ExportFormat::Jpeg,
|
||||
CollisionPolicy::Increment,
|
||||
&|_| true,
|
||||
);
|
||||
assert_eq!(got, None);
|
||||
}
|
||||
}
|
||||
@@ -1,197 +0,0 @@
|
||||
//! TRACES: FR-EXP-4
|
||||
//! Output sharpening, scaled by how far the image was resized.
|
||||
//!
|
||||
//! # Why an export needs this at all
|
||||
//!
|
||||
//! Downsampling averages neighbouring pixels, and averaging is a low-pass
|
||||
//! filter: a 24 MP frame reduced to 2048px comes out measurably softer than
|
||||
//! the same scene shot at 2048px would be. Output sharpening puts back the
|
||||
//! acuity the resample removed. It is not creative sharpening — that belongs
|
||||
//! in the develop pipeline, acts on the full-resolution image, and is a
|
||||
//! different control entirely.
|
||||
//!
|
||||
//! # Why the strength depends on the medium
|
||||
//!
|
||||
//! The three settings are not intensities dressed up as names. A screen shows
|
||||
//! a pixel as a pixel, so it needs the least. Ink spreads into paper — dot
|
||||
//! gain — and matte stock spreads it further than glossy, so a print needs
|
||||
//! more compensation to arrive looking the same. That is why the paper
|
||||
//! options are stronger, and why "more" is not simply a slider.
|
||||
|
||||
use dr_types::OutputSharpening;
|
||||
|
||||
/// Radius of the unsharp mask, in pixels.
|
||||
///
|
||||
/// Fixed at a small value rather than scaled with the image: output
|
||||
/// sharpening compensates for the *resample*, which softens over a pixel or
|
||||
/// two whatever the size of the frame. A radius that grew with the image
|
||||
/// would produce haloes on a large export.
|
||||
const RADIUS: i32 = 1;
|
||||
|
||||
/// Per-setting strength. Applied on top of the resize-derived scaling below.
|
||||
fn strength(setting: OutputSharpening) -> f32 {
|
||||
match setting {
|
||||
OutputSharpening::None => 0.0,
|
||||
OutputSharpening::Screen => 0.55,
|
||||
// Ink spread. Matte stock absorbs more than glossy, so it needs the
|
||||
// heavier hand of the two.
|
||||
OutputSharpening::GlossyPaper => 0.85,
|
||||
OutputSharpening::MattePaper => 1.15,
|
||||
}
|
||||
}
|
||||
|
||||
/// Sharpen in place-ish: takes the resized buffer and returns it, sharpened.
|
||||
///
|
||||
/// `scale` is the resize factor — destination width over source width. Below
|
||||
/// 1 the image was reduced and needs compensation; at or above 1 nothing was
|
||||
/// averaged away and the sharpening is skipped, because sharpening an image
|
||||
/// that was not softened only adds haloes.
|
||||
pub fn apply(
|
||||
mut rgba: Vec<u8>,
|
||||
width: u32,
|
||||
height: u32,
|
||||
setting: OutputSharpening,
|
||||
scale: f32,
|
||||
) -> Vec<u8> {
|
||||
let base = strength(setting);
|
||||
if base == 0.0 || scale >= 1.0 || width < 3 || height < 3 {
|
||||
return rgba;
|
||||
}
|
||||
|
||||
// A frame reduced to a tenth lost far more than one reduced to nine
|
||||
// tenths, so the compensation follows the reduction. Capped at the base
|
||||
// strength: past a point more sharpening is just edge artefacts, and a
|
||||
// thumbnail is the case where that shows most.
|
||||
let amount = base * (1.0 - scale).clamp(0.0, 1.0);
|
||||
|
||||
let src = rgba.clone();
|
||||
let (w, h) = (width as i32, height as i32);
|
||||
|
||||
for y in 0..h {
|
||||
for x in 0..w {
|
||||
for c in 0..3 {
|
||||
// A 3×3 box blur is the mask. Gaussian would be more correct
|
||||
// and, at radius 1, indistinguishable — the kernel is nine
|
||||
// pixels either way.
|
||||
let mut sum = 0.0f32;
|
||||
let mut n = 0.0f32;
|
||||
for dy in -RADIUS..=RADIUS {
|
||||
for dx in -RADIUS..=RADIUS {
|
||||
let sx = (x + dx).clamp(0, w - 1);
|
||||
let sy = (y + dy).clamp(0, h - 1);
|
||||
sum += f32::from(src[((sy * w + sx) * 4 + c) as usize]);
|
||||
n += 1.0;
|
||||
}
|
||||
}
|
||||
let blurred = sum / n;
|
||||
let p = ((y * w + x) * 4 + c) as usize;
|
||||
let original = f32::from(src[p]);
|
||||
// Unsharp mask: the original plus its difference from a
|
||||
// blurred copy, which is the high-frequency detail.
|
||||
let sharpened = original + (original - blurred) * amount;
|
||||
rgba[p] = sharpened.round().clamp(0.0, 255.0) as u8;
|
||||
}
|
||||
}
|
||||
}
|
||||
rgba
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// A frame split down the middle: dark left, light right. One vertical
|
||||
/// edge, which is what sharpening acts on.
|
||||
fn edge(w: u32, h: u32) -> Vec<u8> {
|
||||
let mut v = Vec::new();
|
||||
for _ in 0..h {
|
||||
for x in 0..w {
|
||||
let level = if x < w / 2 { 60 } else { 190 };
|
||||
v.extend_from_slice(&[level, level, level, 255]);
|
||||
}
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
fn at(buf: &[u8], w: u32, x: u32, y: u32) -> u8 {
|
||||
buf[((y * w + x) * 4) as usize]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn none_leaves_the_image_exactly_as_it_was() {
|
||||
let src = edge(16, 8);
|
||||
let out = apply(src.clone(), 16, 8, OutputSharpening::None, 0.5);
|
||||
assert_eq!(out, src);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_unresized_export_is_not_sharpened() {
|
||||
// Nothing was averaged away, so there is nothing to compensate for
|
||||
// and sharpening would only add haloes.
|
||||
let src = edge(16, 8);
|
||||
assert_eq!(
|
||||
apply(src.clone(), 16, 8, OutputSharpening::MattePaper, 1.0),
|
||||
src
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sharpening_increases_contrast_across_an_edge() {
|
||||
// The property, stated directly: the dark side of the edge gets
|
||||
// darker and the light side lighter.
|
||||
let src = edge(16, 8);
|
||||
let out = apply(src.clone(), 16, 8, OutputSharpening::Screen, 0.4);
|
||||
let (before_dark, before_light) = (at(&src, 16, 7, 4), at(&src, 16, 8, 4));
|
||||
let (after_dark, after_light) = (at(&out, 16, 7, 4), at(&out, 16, 8, 4));
|
||||
assert!(after_dark < before_dark, "the dark side should deepen");
|
||||
assert!(after_light > before_light, "the light side should lift");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paper_sharpens_harder_than_screen() {
|
||||
// Ink spreads; the settings are about the medium, not taste.
|
||||
let src = edge(16, 8);
|
||||
let screen = apply(src.clone(), 16, 8, OutputSharpening::Screen, 0.4);
|
||||
let matte = apply(src.clone(), 16, 8, OutputSharpening::MattePaper, 0.4);
|
||||
assert!(at(&matte, 16, 8, 4) > at(&screen, 16, 8, 4));
|
||||
assert!(strength(OutputSharpening::MattePaper) > strength(OutputSharpening::GlossyPaper));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_bigger_reduction_sharpens_more() {
|
||||
let src = edge(16, 8);
|
||||
let mild = apply(src.clone(), 16, 8, OutputSharpening::Screen, 0.9);
|
||||
let severe = apply(src.clone(), 16, 8, OutputSharpening::Screen, 0.1);
|
||||
assert!(at(&severe, 16, 8, 4) >= at(&mild, 16, 8, 4));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_flat_field_is_untouched() {
|
||||
// No detail means no high frequencies to amplify. If this drifts, the
|
||||
// mask is not centred and every sky gains a gradient.
|
||||
let flat = vec![128u8; 16 * 16 * 4];
|
||||
assert_eq!(
|
||||
apply(flat.clone(), 16, 16, OutputSharpening::MattePaper, 0.3),
|
||||
flat
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn alpha_is_never_touched() {
|
||||
// The loop runs over three channels for a reason: sharpening alpha
|
||||
// would put a halo in the transparency of an image that has none.
|
||||
let out = apply(edge(16, 8), 16, 8, OutputSharpening::MattePaper, 0.2);
|
||||
for px in out.chunks_exact(4) {
|
||||
assert_eq!(px[3], 255);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_frame_too_small_to_have_neighbours_is_left_alone() {
|
||||
let tiny = vec![10u8; 2 * 2 * 4];
|
||||
assert_eq!(
|
||||
apply(tiny.clone(), 2, 2, OutputSharpening::Screen, 0.5),
|
||||
tiny
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,520 +0,0 @@
|
||||
//! TRACES: FR-EXP-3 | FR-EXP-4
|
||||
//! Output sizing and resampling.
|
||||
//!
|
||||
//! # Why Lanczos
|
||||
//!
|
||||
//! FR-EXP-4 asks for "a quality resampler (Lanczos or better)", and the
|
||||
//! reason is what a cheap one does to a photograph. Box or bilinear
|
||||
//! downsampling of a 24 MP frame to 2048px averages away detail the sensor
|
||||
//! resolved and aliases what is left — a brick wall or a distant fence comes
|
||||
//! back as moiré. Lanczos's negative lobes preserve edge acuity through a
|
||||
//! large reduction, which is exactly the operation an export performs.
|
||||
//!
|
||||
//! Separable: a horizontal pass then a vertical one, which turns an `a²`
|
||||
//! kernel into `2a` taps per pixel. At the sizes involved that is the
|
||||
//! difference between an export that feels instant and one that does not.
|
||||
|
||||
use dr_types::SizingMode;
|
||||
|
||||
use crate::Frame;
|
||||
|
||||
/// The Lanczos window. 3 is the photographic default — 2 is softer, and
|
||||
/// beyond 3 the extra lobes buy ringing rather than detail.
|
||||
const A: f32 = 3.0;
|
||||
|
||||
/// TRACES: FR-EXP-3
|
||||
/// Resolve the requested sizing against a source, honouring the upscale rule.
|
||||
///
|
||||
/// Aspect is preserved in every mode. In all but one that means only a single
|
||||
/// dimension is ever the requested one; [`SizingMode::FillBox`] is the
|
||||
/// exception, and it keeps the aspect by *discarding* the overhang rather than
|
||||
/// by settling for a smaller box — see [`resample_filling`], which does the
|
||||
/// discarding.
|
||||
///
|
||||
/// **Upscaling is refused by clamping, never by failing.** FR-EXP-3 makes
|
||||
/// upscaling opt-in, and a batch of mixed frames must not abort because one
|
||||
/// was smaller than the target — the user asked for a set of exports, and
|
||||
/// stopping the run over a frame that came out at source size would be a
|
||||
/// worse answer than the file itself.
|
||||
pub fn target_size(
|
||||
src_w: u32,
|
||||
src_h: u32,
|
||||
sizing: SizingMode,
|
||||
allow_upscaling: bool,
|
||||
) -> (u32, u32) {
|
||||
let (src_w, src_h) = (src_w.max(1), src_h.max(1));
|
||||
|
||||
let (w, h) = match sizing {
|
||||
SizingMode::Original => (src_w, src_h),
|
||||
SizingMode::LongEdge(n) => scale_to(src_w, src_h, n, src_w >= src_h),
|
||||
SizingMode::ShortEdge(n) => scale_to(src_w, src_h, n, src_w < src_h),
|
||||
// Fit is a ceiling on both axes, so the smaller factor wins and the
|
||||
// result touches the box on one axis only.
|
||||
SizingMode::FitBox(bw, bh) => {
|
||||
scale_by(src_w, src_h, box_factor(src_w, src_h, bw, bh, f64::min))
|
||||
}
|
||||
// Fill is the box, exactly. The scale that covers it is the larger
|
||||
// factor, and the overhang is taken off in `resample_filling` — this
|
||||
// reports what the file will be, which is the whole reason the mode
|
||||
// exists.
|
||||
SizingMode::FillBox(bw, bh) => (bw.max(1), bh.max(1)),
|
||||
SizingMode::Percentage(p) => {
|
||||
let f = f64::from(p) / 100.0;
|
||||
scale_by(src_w, src_h, f)
|
||||
}
|
||||
};
|
||||
|
||||
if !allow_upscaling && (w > src_w || h > src_h) {
|
||||
// A fill box has to keep its shape even when it cannot keep its size:
|
||||
// the mode's promise is an exact aspect ratio at exact dimensions, and
|
||||
// falling back to the source's own shape would quietly export a 3:2
|
||||
// file where a 16:9 one was asked for. So the *box* is scaled down to
|
||||
// what the source can cover, rather than abandoned.
|
||||
if let SizingMode::FillBox(bw, bh) = sizing {
|
||||
let cover = box_factor(src_w, src_h, bw, bh, f64::max);
|
||||
if cover > 1.0 {
|
||||
return scale_by(bw.max(1), bh.max(1), 1.0 / cover);
|
||||
}
|
||||
}
|
||||
return (src_w, src_h);
|
||||
}
|
||||
(w.max(1), h.max(1))
|
||||
}
|
||||
|
||||
/// TRACES: FR-EXP-3
|
||||
/// The scale that puts `src` against a `bw × bh` box, `choose` deciding which
|
||||
/// axis governs: `f64::min` fits inside it, `f64::max` covers it.
|
||||
fn box_factor(src_w: u32, src_h: u32, bw: u32, bh: u32, choose: fn(f64, f64) -> f64) -> f64 {
|
||||
let fw = f64::from(bw.max(1)) / f64::from(src_w.max(1));
|
||||
let fh = f64::from(bh.max(1)) / f64::from(src_h.max(1));
|
||||
choose(fw, fh)
|
||||
}
|
||||
|
||||
/// Both axes by one factor, never rounding away to nothing.
|
||||
fn scale_by(w: u32, h: u32, factor: f64) -> (u32, u32) {
|
||||
(
|
||||
((f64::from(w) * factor).round() as u32).max(1),
|
||||
((f64::from(h) * factor).round() as u32).max(1),
|
||||
)
|
||||
}
|
||||
|
||||
/// Scale so that the chosen edge lands on `n`.
|
||||
fn scale_to(src_w: u32, src_h: u32, n: u32, width_is_the_edge: bool) -> (u32, u32) {
|
||||
let n = n.max(1);
|
||||
if width_is_the_edge {
|
||||
let h = (f64::from(n) * f64::from(src_h) / f64::from(src_w)).round() as u32;
|
||||
(n, h.max(1))
|
||||
} else {
|
||||
let w = (f64::from(n) * f64::from(src_w) / f64::from(src_h)).round() as u32;
|
||||
(w.max(1), n)
|
||||
}
|
||||
}
|
||||
|
||||
/// Resample to `(dst_w, dst_h)`, returning tightly packed RGBA8.
|
||||
///
|
||||
/// Returns the source buffer untouched where no scaling is needed, which is
|
||||
/// the `SizingMode::Original` case and therefore the common one.
|
||||
pub fn resample(frame: &Frame, dst_w: u32, dst_h: u32) -> Vec<u8> {
|
||||
if dst_w == frame.width && dst_h == frame.height {
|
||||
return frame.rgba.clone();
|
||||
}
|
||||
|
||||
// Horizontal, then vertical. The intermediate is the destination width by
|
||||
// the *source* height, so the second pass works on as little data as the
|
||||
// first can leave it.
|
||||
let horizontal = pass(
|
||||
&frame.rgba,
|
||||
frame.width,
|
||||
frame.height,
|
||||
dst_w,
|
||||
frame.height,
|
||||
true,
|
||||
);
|
||||
pass(&horizontal, dst_w, frame.height, dst_w, dst_h, false)
|
||||
}
|
||||
|
||||
/// TRACES: FR-EXP-3
|
||||
/// Resample onto exactly `(dst_w, dst_h)`, covering the box and cutting the
|
||||
/// overhang off the middle.
|
||||
///
|
||||
/// The other half of [`SizingMode::FillBox`]. [`resample`] alone would do the
|
||||
/// job by *stretching* the frame onto the box, which is the one outcome a
|
||||
/// photographer would never accept — a 3:2 photograph squeezed onto a 16:9
|
||||
/// panel is visibly wrong in a way no amount of resolution fixes.
|
||||
///
|
||||
/// So the frame is scaled until it covers the box, on whichever axis needs the
|
||||
/// most, and the surplus is taken symmetrically off the other. Centred rather
|
||||
/// than anchored: the crop tool is where a photographer decides *which* part
|
||||
/// of the frame survives, and this stage guessing differently would fight it.
|
||||
/// Locking the crop to the export's ratio leaves nothing here to cut.
|
||||
pub fn resample_filling(frame: &Frame, dst_w: u32, dst_h: u32) -> Vec<u8> {
|
||||
let (dst_w, dst_h) = (dst_w.max(1), dst_h.max(1));
|
||||
|
||||
// Rounded *up*, and floored at the destination: a cover scale that rounds
|
||||
// down leaves the box a pixel short on one axis, and the crop below would
|
||||
// then read past the end of the buffer.
|
||||
let cover = box_factor(frame.width, frame.height, dst_w, dst_h, f64::max);
|
||||
let cw = (((f64::from(frame.width) * cover).ceil()) as u32).max(dst_w);
|
||||
let ch = (((f64::from(frame.height) * cover).ceil()) as u32).max(dst_h);
|
||||
|
||||
let covered = resample(frame, cw, ch);
|
||||
if cw == dst_w && ch == dst_h {
|
||||
return covered;
|
||||
}
|
||||
|
||||
let (x0, y0) = ((cw - dst_w) / 2, (ch - dst_h) / 2);
|
||||
let mut out = vec![0u8; (dst_w as usize) * (dst_h as usize) * 4];
|
||||
for y in 0..dst_h as usize {
|
||||
let src = ((y + y0 as usize) * cw as usize + x0 as usize) * 4;
|
||||
let dst = y * dst_w as usize * 4;
|
||||
let run = dst_w as usize * 4;
|
||||
out[dst..dst + run].copy_from_slice(&covered[src..src + run]);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// One separable pass. `horizontal` picks the axis being resampled.
|
||||
fn pass(src: &[u8], src_w: u32, src_h: u32, dst_w: u32, dst_h: u32, horizontal: bool) -> Vec<u8> {
|
||||
let (src_len, dst_len) = if horizontal {
|
||||
(src_w, dst_w)
|
||||
} else {
|
||||
(src_h, dst_h)
|
||||
};
|
||||
let ratio = f64::from(src_len) / f64::from(dst_len);
|
||||
|
||||
// Enlarging samples the source at its own frequency; shrinking has to
|
||||
// widen the kernel to average the pixels being discarded, or the result
|
||||
// aliases. This is the whole difference between a resample and a
|
||||
// subsample.
|
||||
let filter_scale = ratio.max(1.0);
|
||||
let support = A as f64 * filter_scale;
|
||||
|
||||
let mut out = vec![0u8; (dst_w * dst_h * 4) as usize];
|
||||
|
||||
for i in 0..dst_len {
|
||||
// Centre of the destination sample, in source coordinates.
|
||||
let centre = (f64::from(i) + 0.5) * ratio - 0.5;
|
||||
let first = ((centre - support).ceil() as i64).max(0);
|
||||
let last = ((centre + support).floor() as i64).min(i64::from(src_len) - 1);
|
||||
|
||||
// Weights once per output row/column rather than per pixel: they
|
||||
// depend only on the axis position, and recomputing them per channel
|
||||
// was most of the cost when this was written the obvious way.
|
||||
let mut weights = Vec::with_capacity((last - first + 1).max(0) as usize);
|
||||
let mut total = 0.0f64;
|
||||
for s in first..=last {
|
||||
let w = lanczos((f64::from(s as i32) - centre) / filter_scale);
|
||||
weights.push(w);
|
||||
total += w;
|
||||
}
|
||||
if total == 0.0 {
|
||||
total = 1.0;
|
||||
}
|
||||
|
||||
let other = if horizontal { dst_h } else { dst_w };
|
||||
for j in 0..other {
|
||||
let mut acc = [0.0f64; 4];
|
||||
for (k, w) in weights.iter().enumerate() {
|
||||
let s = first as u32 + k as u32;
|
||||
let (x, y) = if horizontal { (s, j) } else { (j, s) };
|
||||
let p = ((y * src_w + x) * 4) as usize;
|
||||
for c in 0..4 {
|
||||
acc[c] += f64::from(src[p + c]) * w;
|
||||
}
|
||||
}
|
||||
let (x, y) = if horizontal { (i, j) } else { (j, i) };
|
||||
let p = ((y * dst_w + x) * 4) as usize;
|
||||
for c in 0..4 {
|
||||
// Lanczos overshoots at edges — that is what makes it look
|
||||
// sharp — so the result must be clamped rather than wrapped.
|
||||
out[p + c] = (acc[c] / total).round().clamp(0.0, 255.0) as u8;
|
||||
}
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// The Lanczos kernel, `sinc(x) * sinc(x / a)`.
|
||||
fn lanczos(x: f64) -> f64 {
|
||||
let x = x.abs();
|
||||
if x < 1e-9 {
|
||||
return 1.0;
|
||||
}
|
||||
if x >= f64::from(A) {
|
||||
return 0.0;
|
||||
}
|
||||
let px = std::f64::consts::PI * x;
|
||||
(px.sin() / px) * ((px / f64::from(A)).sin() / (px / f64::from(A)))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::tests::frame;
|
||||
|
||||
#[test]
|
||||
fn original_is_the_source_size() {
|
||||
assert_eq!(
|
||||
target_size(6000, 4000, SizingMode::Original, false),
|
||||
(6000, 4000)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_edge_picks_the_longer_dimension_either_way_round() {
|
||||
assert_eq!(
|
||||
target_size(6000, 4000, SizingMode::LongEdge(3000), false),
|
||||
(3000, 2000)
|
||||
);
|
||||
// Portrait: the long edge is now the height.
|
||||
assert_eq!(
|
||||
target_size(4000, 6000, SizingMode::LongEdge(3000), false),
|
||||
(2000, 3000)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn short_edge_picks_the_shorter_dimension_either_way_round() {
|
||||
assert_eq!(
|
||||
target_size(6000, 4000, SizingMode::ShortEdge(2000), false),
|
||||
(3000, 2000)
|
||||
);
|
||||
assert_eq!(
|
||||
target_size(4000, 6000, SizingMode::ShortEdge(2000), false),
|
||||
(2000, 3000)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_percentage_scales_both_dimensions() {
|
||||
assert_eq!(
|
||||
target_size(4000, 3000, SizingMode::Percentage(50), false),
|
||||
(2000, 1500)
|
||||
);
|
||||
assert_eq!(
|
||||
target_size(4000, 3000, SizingMode::Percentage(100), false),
|
||||
(4000, 3000)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_fit_box_stays_inside_the_box_and_keeps_its_shape() {
|
||||
// Fit is a ceiling on both axes, so a 3:2 frame in a 16:9 box comes
|
||||
// back short of the box's width, never past its height.
|
||||
assert_eq!(
|
||||
target_size(6000, 4000, SizingMode::FitBox(3840, 2160), false),
|
||||
(3240, 2160)
|
||||
);
|
||||
// Portrait into the same box: now the height governs nothing and the
|
||||
// width does.
|
||||
assert_eq!(
|
||||
target_size(4000, 6000, SizingMode::FitBox(3840, 2160), false),
|
||||
(1440, 2160)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_fill_box_is_the_box_exactly() {
|
||||
// The whole point of the mode. A display that accepts one resolution
|
||||
// and rejects everything else has to get that resolution whatever the
|
||||
// photograph's own shape is.
|
||||
for (w, h) in [(6000u32, 4000u32), (4000, 6000), (5000, 5000)] {
|
||||
assert_eq!(
|
||||
target_size(w, h, SizingMode::FillBox(3840, 2160), false),
|
||||
(3840, 2160),
|
||||
"{w}x{h} did not fill the box"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_fill_box_too_large_for_the_source_keeps_its_shape_not_the_sources() {
|
||||
// Upscaling off, and the source cannot cover 4K. Falling back to the
|
||||
// source's own size would export a 3:2 file where 16:9 was asked
|
||||
// for — silently wrong in exactly the way the mode exists to prevent.
|
||||
// The box shrinks instead.
|
||||
let (w, h) = target_size(1600, 1200, SizingMode::FillBox(3840, 2160), false);
|
||||
assert!(w <= 1600 && h <= 1200, "upscaled to {w}x{h}");
|
||||
let want = 3840.0 / 2160.0;
|
||||
assert!(
|
||||
((w as f64 / h as f64) / want - 1.0).abs() < 0.01,
|
||||
"{w}x{h} is not the box's shape"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_fill_box_within_the_source_is_honoured_with_upscaling_off() {
|
||||
// The ordinary case: a 24 MP frame has pixels to spare for a 4K panel,
|
||||
// so nothing is being enlarged and the clamp must not fire.
|
||||
assert_eq!(
|
||||
target_size(6000, 4000, SizingMode::FillBox(3840, 2160), false),
|
||||
(3840, 2160)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_filled_frame_comes_back_at_exactly_the_box() {
|
||||
let f = frame(128, 64);
|
||||
// Wider than the source's 2:1, so the crop comes off the width.
|
||||
assert_eq!(resample_filling(&f, 40, 40).len(), 40 * 40 * 4);
|
||||
assert_eq!(resample_filling(&f, 100, 25).len(), 100 * 25 * 4);
|
||||
// Already the box: no work, and no drift.
|
||||
assert_eq!(resample_filling(&f, 128, 64), f.rgba);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_fill_crops_rather_than_stretching() {
|
||||
// The property that separates fill from handing the box straight to
|
||||
// `resample`. The test frame ramps red left to right, so a 2:1 source
|
||||
// squeezed into a square would compress that ramp into the full
|
||||
// width — where a centre crop keeps its middle, and therefore starts
|
||||
// and ends well inside the source's own range.
|
||||
let f = frame(128, 128);
|
||||
let square = resample(&f, 64, 64);
|
||||
let filled = resample_filling(&f, 32, 64);
|
||||
|
||||
let left = |b: &[u8]| b[0];
|
||||
let right = |b: &[u8], w: usize| b[(w - 1) * 4];
|
||||
|
||||
assert!(
|
||||
left(&filled) > left(&square),
|
||||
"a centre crop must start further into the ramp"
|
||||
);
|
||||
assert!(
|
||||
right(&filled, 32) < right(&square, 64),
|
||||
"a centre crop must end further from the ramp's end"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_fill_takes_the_overhang_evenly_off_both_sides() {
|
||||
// Centred, not anchored: the crop tool is where a photographer decides
|
||||
// which part of the frame survives, and this stage must not have an
|
||||
// opinion of its own.
|
||||
let f = frame(128, 128);
|
||||
let filled = resample_filling(&f, 32, 64);
|
||||
let px = |x: usize| filled[x * 4];
|
||||
// The ramp is horizontal, so a centred crop is symmetric about the
|
||||
// frame's own midpoint: the two ends should sit equally far from it.
|
||||
let mid = i32::from(resample(&f, 128, 128)[64 * 4]);
|
||||
let lo = i32::from(px(0));
|
||||
let hi = i32::from(px(31));
|
||||
assert!(
|
||||
((mid - lo) - (hi - mid)).abs() < 8,
|
||||
"not centred: {lo} .. {mid} .. {hi}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_flat_field_survives_a_fill_unchanged() {
|
||||
// Same guard as the fit path: any deviation means the cover scale and
|
||||
// the crop disagree about where the pixels are.
|
||||
let flat = Frame::new(64, 48, vec![200; 64 * 48 * 4]).unwrap();
|
||||
for byte in resample_filling(&flat, 30, 30) {
|
||||
assert_eq!(byte, 200);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn upscaling_is_refused_by_clamping_rather_than_failing() {
|
||||
// FR-EXP-3: opt-in, and a batch must not abort over one small frame.
|
||||
assert_eq!(
|
||||
target_size(800, 600, SizingMode::LongEdge(4000), false),
|
||||
(800, 600)
|
||||
);
|
||||
assert_eq!(
|
||||
target_size(800, 600, SizingMode::Percentage(400), false),
|
||||
(800, 600)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn upscaling_is_honoured_when_asked_for() {
|
||||
assert_eq!(
|
||||
target_size(800, 600, SizingMode::LongEdge(1600), true),
|
||||
(1600, 1200)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_square_frame_treats_either_edge_as_the_long_one() {
|
||||
// The tie has to resolve somewhere, and both answers are the same
|
||||
// size — but it must not produce a zero or a panic.
|
||||
assert_eq!(
|
||||
target_size(1000, 1000, SizingMode::LongEdge(500), false),
|
||||
(500, 500)
|
||||
);
|
||||
assert_eq!(
|
||||
target_size(1000, 1000, SizingMode::ShortEdge(500), false),
|
||||
(500, 500)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_size_can_never_round_down_to_nothing() {
|
||||
// A 1% export of a small frame rounds toward zero, and a zero-pixel
|
||||
// image is not a file anyone can open.
|
||||
let (w, h) = target_size(50, 30, SizingMode::Percentage(1), false);
|
||||
assert!(w >= 1 && h >= 1, "got {w}x{h}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resampling_to_the_same_size_changes_nothing() {
|
||||
// The `Original` path, which is the common one — it must not spend a
|
||||
// Lanczos pass to return what it was given.
|
||||
let f = frame(32, 24);
|
||||
assert_eq!(resample(&f, 32, 24), f.rgba);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_resample_produces_the_right_number_of_pixels() {
|
||||
let f = frame(64, 48);
|
||||
assert_eq!(resample(&f, 32, 24).len(), 32 * 24 * 4);
|
||||
assert_eq!(resample(&f, 100, 75).len(), 100 * 75 * 4);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_downscale_preserves_the_gradient_it_was_given() {
|
||||
// The check that separates a real resample from a buffer of the right
|
||||
// length: the test frame ramps red left-to-right, so the output must
|
||||
// too, and its corners must still be near the source's.
|
||||
let f = frame(128, 128);
|
||||
let small = resample(&f, 32, 32);
|
||||
let px = |x: usize, y: usize| small[(y * 32 + x) * 4];
|
||||
assert!(px(0, 0) < px(16, 0), "red should rise across the frame");
|
||||
assert!(px(16, 0) < px(31, 0));
|
||||
// Row-invariant in red, since the ramp is horizontal.
|
||||
assert!((i32::from(px(16, 0)) - i32::from(px(16, 31))).abs() < 8);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_flat_field_survives_a_resample_unchanged() {
|
||||
// Lanczos rings on edges, which is intended — but a constant field
|
||||
// has no edges, and any deviation here means the weights do not sum
|
||||
// to one. That error is invisible on a photograph and glaring on a
|
||||
// sky.
|
||||
let flat = Frame::new(64, 64, vec![200; 64 * 64 * 4]).unwrap();
|
||||
for byte in resample(&flat, 21, 21) {
|
||||
assert_eq!(byte, 200, "a constant field must resample to itself");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_upscale_also_holds_a_flat_field() {
|
||||
let flat = Frame::new(16, 16, vec![64; 16 * 16 * 4]).unwrap();
|
||||
for byte in resample(&flat, 40, 40) {
|
||||
assert_eq!(byte, 64);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_kernel_is_one_at_the_centre_and_zero_past_its_window() {
|
||||
assert!((lanczos(0.0) - 1.0).abs() < 1e-9);
|
||||
assert_eq!(lanczos(3.0), 0.0);
|
||||
assert_eq!(lanczos(4.5), 0.0);
|
||||
// Zero at the integers inside the window, which is what makes an
|
||||
// unscaled resample an identity.
|
||||
assert!(lanczos(1.0).abs() < 1e-9);
|
||||
assert!(lanczos(2.0).abs() < 1e-9);
|
||||
}
|
||||
}
|
||||
@@ -1,47 +0,0 @@
|
||||
[package]
|
||||
name = "dr-face"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
license.workspace = true
|
||||
|
||||
[dependencies]
|
||||
thiserror.workspace = true
|
||||
log.workspace = true
|
||||
|
||||
# Inference. `ort` is the API; **tract is the engine** — see the workspace
|
||||
# manifest, and docs/faces.md §3, for why the C++ ONNX Runtime is not linked.
|
||||
ort = { workspace = true, optional = true }
|
||||
ort-tract = { workspace = true, optional = true }
|
||||
ndarray = { workspace = true, optional = true }
|
||||
|
||||
[dev-dependencies]
|
||||
zune-jpeg.workspace = true
|
||||
env_logger.workspace = true
|
||||
# The M1 probe drives `ort` directly so it can print the raw load error.
|
||||
ort = { workspace = true }
|
||||
ort-tract = { workspace = true }
|
||||
|
||||
[[example]]
|
||||
name = "probe"
|
||||
required-features = ["inference"]
|
||||
|
||||
[[example]]
|
||||
name = "faces"
|
||||
required-features = ["inference"]
|
||||
|
||||
[features]
|
||||
# Nothing on by default, and in particular **no `embedded-model`**: the weights
|
||||
# are not a build input and never become one (docs/faces.md §2.2). A feature
|
||||
# flag that *could* embed them is a flag someone eventually sets in a packaging
|
||||
# script, and the InsightFace grant does not survive that.
|
||||
default = []
|
||||
|
||||
# The ONNX runtime, and the two stages that need it.
|
||||
#
|
||||
# Separable because the accuracy of this subsystem lives in `calibrate` and
|
||||
# `cluster`, which are arithmetic over embeddings with no model in them. They
|
||||
# must be testable against synthetic embeddings on a machine with no weights on
|
||||
# it — a test suite that needs a research-licensed download is a test suite
|
||||
# that does not run in CI.
|
||||
inference = ["dep:ort", "dep:ort-tract", "dep:ndarray"]
|
||||
@@ -1,128 +0,0 @@
|
||||
//! Detect, align and embed the faces in a JPEG.
|
||||
//!
|
||||
//! The thing worth looking at is whether the landmarks land on a real
|
||||
//! photograph — the same reason `dr-segment` has `examples/detect.rs`.
|
||||
//!
|
||||
//! cargo run -p dr-face --features inference --example faces -- \
|
||||
//! DET.onnx EMB.onnx photo.jpg [photo.jpg ...]
|
||||
//!
|
||||
//! The models must have had their input dims frozen first; see
|
||||
//! `tools/fix-face-model-shapes.sh` and docs/faces.md §12 M1.
|
||||
|
||||
use std::time::Instant;
|
||||
|
||||
use dr_face::{align, DetectOptions, Detector, Embedder, ModelId};
|
||||
|
||||
fn main() {
|
||||
env_logger::init();
|
||||
|
||||
let args: Vec<String> = std::env::args().skip(1).collect();
|
||||
if args.len() < 3 {
|
||||
eprintln!("usage: faces DET.onnx EMB.onnx IMAGE.jpg [IMAGE.jpg ...]");
|
||||
std::process::exit(2);
|
||||
}
|
||||
|
||||
let t = Instant::now();
|
||||
let mut detector = Detector::from_path(&args[0]).expect("load detector");
|
||||
let mut embedder =
|
||||
Embedder::from_path(&args[1], ModelId::new("w600k_mbf")).expect("load embedder");
|
||||
println!(
|
||||
"loaded both models in {:?} (strides {:?})",
|
||||
t.elapsed(),
|
||||
detector.strides()
|
||||
);
|
||||
|
||||
let opts = DetectOptions::default();
|
||||
let mut all = Vec::new();
|
||||
|
||||
for path in &args[2..] {
|
||||
let (rgb, w, h) = match load_jpeg(path) {
|
||||
Ok(v) => v,
|
||||
Err(e) => {
|
||||
println!("{path}: {e}");
|
||||
continue;
|
||||
}
|
||||
};
|
||||
|
||||
let t = Instant::now();
|
||||
let dets = detector.detect(&rgb, w, h, &opts).expect("detect");
|
||||
let detect_ms = t.elapsed().as_secs_f64() * 1e3;
|
||||
|
||||
println!(
|
||||
"\n{path} ({w}×{h}) {} face(s) in {detect_ms:.0} ms",
|
||||
dets.len()
|
||||
);
|
||||
|
||||
for (i, d) in dets.iter().enumerate() {
|
||||
let Some(aligned) = align::warp(&rgb, w, h, &d.landmarks) else {
|
||||
println!(" [{i}] degenerate landmarks, skipped");
|
||||
continue;
|
||||
};
|
||||
let t = Instant::now();
|
||||
let emb = embedder.embed(&aligned).expect("embed");
|
||||
let embed_ms = t.elapsed().as_secs_f64() * 1e3;
|
||||
|
||||
println!(
|
||||
" [{i}] conf {:.3} box {:.0},{:.0} {:.0}×{:.0} crop_px {:.0} quality {:.1} embed {embed_ms:.0} ms",
|
||||
d.confidence,
|
||||
d.bbox.0,
|
||||
d.bbox.1,
|
||||
d.width(),
|
||||
d.height(),
|
||||
aligned.source_px(),
|
||||
emb.quality,
|
||||
);
|
||||
all.push((path.clone(), i, emb.embedding));
|
||||
}
|
||||
}
|
||||
|
||||
// Every pair, so the numbers can be eyeballed against the expectation that
|
||||
// faces from one identity's folder score high and everything else low.
|
||||
if all.len() > 1 {
|
||||
println!("\ncosine similarity");
|
||||
for i in 0..all.len() {
|
||||
for j in i + 1..all.len() {
|
||||
let cos = all[i].2.cosine(&all[j].2).expect("same model");
|
||||
println!(
|
||||
" {:.4} {}#{} vs {}#{}",
|
||||
cos,
|
||||
short(&all[i].0),
|
||||
all[i].1,
|
||||
short(&all[j].0),
|
||||
all[j].1
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn short(path: &str) -> String {
|
||||
let p = std::path::Path::new(path);
|
||||
let file = p.file_name().unwrap_or_default().to_string_lossy();
|
||||
match p.parent().and_then(|d| d.file_name()) {
|
||||
Some(dir) => format!("{}/{file}", dir.to_string_lossy()),
|
||||
None => file.into_owned(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Decode to the tightly packed `f32` RGB `0.0..=1.0` the crate expects.
|
||||
fn load_jpeg(path: &str) -> Result<(Vec<f32>, usize, usize), String> {
|
||||
let bytes = std::fs::read(path).map_err(|e| e.to_string())?;
|
||||
let mut dec = zune_jpeg::JpegDecoder::new(&bytes);
|
||||
let px = dec.decode().map_err(|e| e.to_string())?;
|
||||
let info = dec.info().ok_or("no jpeg header")?;
|
||||
let (w, h) = (info.width as usize, info.height as usize);
|
||||
|
||||
let rgb: Vec<f32> = match px.len() / (w * h) {
|
||||
3 => px.iter().map(|&v| v as f32 / 255.0).collect(),
|
||||
1 => px
|
||||
.iter()
|
||||
.flat_map(|&v| {
|
||||
let g = v as f32 / 255.0;
|
||||
[g, g, g]
|
||||
})
|
||||
.collect(),
|
||||
n => return Err(format!("{n} components per pixel, expected 1 or 3")),
|
||||
};
|
||||
Ok((rgb, w, h))
|
||||
}
|
||||
@@ -1,58 +0,0 @@
|
||||
//! M1 (docs/faces.md §12) — will tract load these graphs at all?
|
||||
//!
|
||||
//! The one measurement everything else in the face subsystem is conditional
|
||||
//! on. `det_500m.onnx` has a dynamic H/W input, which is exactly what tract
|
||||
//! failed on for YOLO26n-seg, so a plain "no" here is the expected outcome and
|
||||
//! the interesting part is the error it gives.
|
||||
//!
|
||||
//! cargo run -p dr-face --features inference --example probe -- MODEL...
|
||||
|
||||
fn main() {
|
||||
env_logger::init();
|
||||
|
||||
let paths: Vec<String> = std::env::args().skip(1).collect();
|
||||
if paths.is_empty() {
|
||||
eprintln!("usage: probe MODEL.onnx [MODEL.onnx ...]");
|
||||
std::process::exit(2);
|
||||
}
|
||||
|
||||
let mut failures = 0;
|
||||
for path in &paths {
|
||||
println!("\n=== {path} ===");
|
||||
let bytes = match std::fs::read(path) {
|
||||
Ok(b) => b,
|
||||
Err(e) => {
|
||||
println!(" UNREADABLE: {e}");
|
||||
failures += 1;
|
||||
continue;
|
||||
}
|
||||
};
|
||||
println!(" {} bytes", bytes.len());
|
||||
|
||||
dr_face::install_backend_for_probe();
|
||||
|
||||
let session =
|
||||
ort::session::Session::builder().and_then(|mut b| b.commit_from_memory(&bytes));
|
||||
|
||||
match session {
|
||||
Err(e) => {
|
||||
println!(" LOAD FAILED: {e}");
|
||||
failures += 1;
|
||||
}
|
||||
Ok(s) => {
|
||||
println!(" LOADED");
|
||||
for i in s.inputs() {
|
||||
println!(" in {:<24} {:?}", i.name(), i.dtype().tensor_shape());
|
||||
}
|
||||
for o in s.outputs() {
|
||||
println!(" out {:<24} {:?}", o.name(), o.dtype().tensor_shape());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
println!("\n{} of {} failed", failures, paths.len());
|
||||
if failures > 0 {
|
||||
std::process::exit(1);
|
||||
}
|
||||
}
|
||||
@@ -1,112 +0,0 @@
|
||||
//! How fast this machine can scan a library for face pairs.
|
||||
//!
|
||||
//! cargo run --release -p dr-face --example scan_bench [-- FACES…]
|
||||
//!
|
||||
//! Synthetic embeddings, because the scan's cost is `n²/2` dot products and
|
||||
//! does not care what the vectors mean — which is what makes this runnable on
|
||||
//! a phone with no library on it, over `adb shell`, beside
|
||||
//! `tools/face-tests-on-device.sh`.
|
||||
//!
|
||||
//! # What it is for
|
||||
//!
|
||||
//! docs/faces.md §9 has the desktop numbers and the question they leave open:
|
||||
//! a GPU GEMM is worth roughly 1.5× of a regroup on a twenty-core desktop,
|
||||
//! because the scan is under a third of the pass there. On a tablet the CPU is
|
||||
//! several times slower and the GPU is not, so the same optimisation is worth
|
||||
//! something different — and nobody had measured which.
|
||||
//!
|
||||
//! No weights, no catalog, no display: it needs nothing but the binary.
|
||||
|
||||
use dr_face::neighbours::{above_threshold, Faces};
|
||||
use dr_face::{Calibration, EMBEDDING_DIM, RIVAL_FLOOR};
|
||||
|
||||
/// Sizes to time, unless the command line names others.
|
||||
const DEFAULT_SIZES: [usize; 4] = [2_000, 4_000, 8_000, 18_000];
|
||||
|
||||
fn main() {
|
||||
let sizes: Vec<usize> = {
|
||||
let given: Vec<usize> = std::env::args()
|
||||
.skip(1)
|
||||
.filter_map(|a| a.parse().ok())
|
||||
.collect();
|
||||
if given.is_empty() {
|
||||
DEFAULT_SIZES.to_vec()
|
||||
} else {
|
||||
given
|
||||
}
|
||||
};
|
||||
|
||||
// The reference curve, so the cosine floor is the one a real library with
|
||||
// no fit of its own would scan at.
|
||||
let cal = Calibration::default();
|
||||
|
||||
println!(
|
||||
"{:>8} {:>9} {:>10} {:>9}",
|
||||
"faces", "scan", "pairs", "GFLOP/s"
|
||||
);
|
||||
for n in sizes {
|
||||
let (embeddings, crop_px, images) = population(n);
|
||||
let gallery = vec![true; n];
|
||||
let faces = Faces {
|
||||
embeddings: &embeddings,
|
||||
dim: EMBEDDING_DIM,
|
||||
crop_px: &crop_px,
|
||||
images: &images,
|
||||
gallery: &gallery,
|
||||
};
|
||||
|
||||
let start = std::time::Instant::now();
|
||||
let pairs = above_threshold(&faces, &cal, RIVAL_FLOOR);
|
||||
let secs = start.elapsed().as_secs_f64();
|
||||
|
||||
let flop = n as f64 * n as f64 / 2.0 * EMBEDDING_DIM as f64 * 2.0;
|
||||
println!(
|
||||
"{n:>8} {secs:>8.2}s {:>10} {:>9.1}",
|
||||
pairs.len(),
|
||||
flop / secs / 1e9
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// `n` L2-normalised embeddings in a handful of loose clusters.
|
||||
///
|
||||
/// Clustered rather than uniform so the scan finds a plausible number of pairs
|
||||
/// to keep — a population where nothing survives the threshold would time the
|
||||
/// rejection path alone, which is not the path that matters. Hashed from an
|
||||
/// index rather than drawn from an RNG, so a number is reproducible from the
|
||||
/// command that produced it.
|
||||
fn population(n: usize) -> (Vec<f32>, Vec<f32>, Vec<u64>) {
|
||||
let identities = (n / 12).max(1);
|
||||
let mut embeddings = Vec::with_capacity(n * EMBEDDING_DIM);
|
||||
for i in 0..n {
|
||||
let mut v = unit(i % identities);
|
||||
let noise = unit(i + 1_000_000);
|
||||
for (x, e) in v.iter_mut().zip(&noise) {
|
||||
*x = 0.75 * *x + 0.25 * e;
|
||||
}
|
||||
embeddings.extend(normalise(v));
|
||||
}
|
||||
// Every face in its own photograph: the co-occurrence rule skips pairs
|
||||
// rather than scoring them, and skipped pairs are not what is being timed.
|
||||
((embeddings), vec![150.0; n], (0..n as u64).collect())
|
||||
}
|
||||
|
||||
fn unit(seed: usize) -> Vec<f32> {
|
||||
let mut s = (seed as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15) | 1;
|
||||
let mut v = Vec::with_capacity(EMBEDDING_DIM);
|
||||
for _ in 0..EMBEDDING_DIM {
|
||||
s ^= s << 13;
|
||||
s ^= s >> 7;
|
||||
s ^= s << 17;
|
||||
v.push(((s >> 11) as f64 / (1u64 << 53) as f64) as f32 - 0.5);
|
||||
}
|
||||
normalise(v)
|
||||
}
|
||||
|
||||
fn normalise(mut v: Vec<f32>) -> Vec<f32> {
|
||||
let len = v.iter().map(|x| x * x).sum::<f32>().sqrt();
|
||||
for x in &mut v {
|
||||
*x /= len;
|
||||
}
|
||||
v
|
||||
}
|
||||
@@ -1,659 +0,0 @@
|
||||
//! Five-point face alignment (docs/faces.md §5).
|
||||
//!
|
||||
//! ArcFace embeddings are trained on faces warped to a canonical 112×112
|
||||
//! arrangement. Feeding the model a plain bounding-box crop *works* — it
|
||||
//! produces 512 numbers, they are unit-norm, and cosine similarities between
|
||||
//! them look entirely reasonable. They are just much worse, and nothing in the
|
||||
//! system reports it.
|
||||
//!
|
||||
//! That is the whole reason this module exists, and the reason [`Aligned112`]
|
||||
//! is a newtype only [`warp`] can construct: the mistake is not one a reviewer
|
||||
//! catches, so the type system catches it instead.
|
||||
//!
|
||||
//! Model-free, so it builds and tests without the `inference` feature.
|
||||
|
||||
/// Canonical landmark positions for a 112×112 ArcFace crop.
|
||||
///
|
||||
/// # The naming is a trap; the order is not
|
||||
///
|
||||
/// Point 0 sits at x=38 on a 112-wide canvas — left of centre *in the image*,
|
||||
/// which is the subject's **right** eye. Both namings are in circulation and
|
||||
/// they are opposite, so the array is written in the detector's order and the
|
||||
/// comment says whose left is whose:
|
||||
///
|
||||
/// ```text
|
||||
/// 0 subject's right eye (image-left)
|
||||
/// 1 subject's left eye (image-right)
|
||||
/// 2 nose tip
|
||||
/// 3 subject's right mouth corner
|
||||
/// 4 subject's left mouth corner
|
||||
/// ```
|
||||
///
|
||||
/// SCRFD emits its five points in this same order, so the correct amount of
|
||||
/// reordering between detector and template is **none**. A detector with a
|
||||
/// different order carries its own permutation beside its model id rather than
|
||||
/// this constant growing an assumption.
|
||||
pub const ARCFACE_TEMPLATE: [(f32, f32); 5] = [
|
||||
(38.2946, 51.6963),
|
||||
(73.5318, 51.5014),
|
||||
(56.0252, 71.7366),
|
||||
(41.5493, 92.3655),
|
||||
(70.7299, 92.2041),
|
||||
];
|
||||
|
||||
/// Edge of the aligned crop, in pixels. Fixed by the embedder's input.
|
||||
pub const ALIGNED_EDGE: usize = 112;
|
||||
|
||||
/// A face warped to [`ARCFACE_TEMPLATE`], ready for the embedder.
|
||||
///
|
||||
/// Constructible only by [`warp`]. That is the point: an `Embedder` that took
|
||||
/// a plain `&[f32]` would accept an unaligned bounding-box crop and silently
|
||||
/// return worse embeddings, which is a failure no test of the embedder itself
|
||||
/// would catch.
|
||||
pub struct Aligned112 {
|
||||
/// `112 × 112 × 3`, row-major RGB in `0.0..=1.0`.
|
||||
pixels: Vec<f32>,
|
||||
/// Source pixels across the crop before warping — `crop_px` in the catalog.
|
||||
///
|
||||
/// Carried here rather than recomputed later because the scale factor is
|
||||
/// known exactly at warp time and only approximately from the box
|
||||
/// afterwards. §7: it is the honest quality signal, and a feature in the
|
||||
/// calibration.
|
||||
source_px: f32,
|
||||
}
|
||||
|
||||
impl Aligned112 {
|
||||
pub fn pixels(&self) -> &[f32] {
|
||||
&self.pixels
|
||||
}
|
||||
|
||||
/// Source pixels spanned by the 112-pixel crop.
|
||||
///
|
||||
/// Below ~112 the face was upsampled to reach the embedder and the
|
||||
/// embedding is correspondingly weaker; above it, downsampled and healthy.
|
||||
pub fn source_px(&self) -> f32 {
|
||||
self.source_px
|
||||
}
|
||||
|
||||
/// How sharp the face the embedder is about to see actually is.
|
||||
///
|
||||
/// # Why size is not enough
|
||||
///
|
||||
/// A face can be large and useless. A subject walking through a half-second
|
||||
/// exposure, a frame focused on the person behind them, a hand-held shot at
|
||||
/// 1/15 — all yield a big box, a confident detection and five landmarks in
|
||||
/// plausible places. The embedding that comes back is not *wrong* in any
|
||||
/// way the system can see: it is unit-norm and its cosines look ordinary.
|
||||
/// It is simply an embedding of a blur, and blurs resemble each other more
|
||||
/// than they resemble the people they were, so they cluster together and
|
||||
/// bridge identities that have nothing to do with one another.
|
||||
///
|
||||
/// That is the failure this exists to prevent, and it is the same class of
|
||||
/// fault as the unaligned-crop one the [`Aligned112`] newtype guards
|
||||
/// against: plausible output, no error, worse results, nothing reported.
|
||||
///
|
||||
/// # The measure
|
||||
///
|
||||
/// Variance of the Laplacian — the standard blur metric — **divided by the
|
||||
/// variance of the luma it was taken over**. The division is what makes it
|
||||
/// usable here. Raw Laplacian variance scales with contrast, so a sharp
|
||||
/// face in flat, hazy or backlit light scores like a blurred one in hard
|
||||
/// light, and a threshold on it would quietly throw away every face shot
|
||||
/// against a bright sky. The ratio asks the question that actually matters
|
||||
/// — *how much of this crop's variation is edges rather than broad
|
||||
/// gradients* — and is invariant to exposure and contrast.
|
||||
///
|
||||
/// Computed on luma over the interior, so the 3x3 kernel never needs a
|
||||
/// border rule. Returns 0.0 for a crop with no variation at all, which is
|
||||
/// a flat patch and correctly unusable rather than infinitely sharp.
|
||||
///
|
||||
/// # This is not independent of size
|
||||
///
|
||||
/// A face smaller than 112 pixels was *upsampled* to reach the embedder,
|
||||
/// and upsampling invents no edges — so a small face scores low here even
|
||||
/// when the original was perfectly sharp. That is not a flaw to correct: it
|
||||
/// is the honest statement that the embedder is looking at a soft image.
|
||||
/// The size floor and this one overlap deliberately, and
|
||||
/// `face_index --quality` prints the joint distribution so the two are
|
||||
/// chosen together rather than each in ignorance of the other.
|
||||
pub fn sharpness(&self) -> f32 {
|
||||
let e = ALIGNED_EDGE;
|
||||
let luma: Vec<f32> = self
|
||||
.pixels
|
||||
.chunks_exact(3)
|
||||
.map(|p| 0.2126 * p[0] + 0.7152 * p[1] + 0.0722 * p[2])
|
||||
.collect();
|
||||
|
||||
let (mut lap_sum, mut lap_sq) = (0.0_f64, 0.0_f64);
|
||||
let (mut lum_sum, mut lum_sq) = (0.0_f64, 0.0_f64);
|
||||
let mut n = 0.0_f64;
|
||||
|
||||
for y in 1..e - 1 {
|
||||
for x in 1..e - 1 {
|
||||
let i = y * e + x;
|
||||
// Four-neighbour Laplacian. The 8-neighbour form is more
|
||||
// sensitive to diagonal detail and also to noise, which on a
|
||||
// high-ISO frame is exactly the thing that must not read as
|
||||
// sharpness.
|
||||
let lap = 4.0 * luma[i] - luma[i - 1] - luma[i + 1] - luma[i - e] - luma[i + e];
|
||||
let lap = lap as f64;
|
||||
lap_sum += lap;
|
||||
lap_sq += lap * lap;
|
||||
|
||||
let l = luma[i] as f64;
|
||||
lum_sum += l;
|
||||
lum_sq += l * l;
|
||||
n += 1.0;
|
||||
}
|
||||
}
|
||||
|
||||
if n == 0.0 {
|
||||
return 0.0;
|
||||
}
|
||||
let lap_var = (lap_sq / n - (lap_sum / n).powi(2)).max(0.0);
|
||||
let lum_var = (lum_sq / n - (lum_sum / n).powi(2)).max(0.0);
|
||||
|
||||
// A crop with no luma variation has no edges to find either, so the
|
||||
// ratio is 0/0. Zero is the right answer: nothing there is a face.
|
||||
if lum_var <= 1e-9 {
|
||||
return 0.0;
|
||||
}
|
||||
(lap_var / lum_var) as f32
|
||||
}
|
||||
}
|
||||
|
||||
/// A similarity transform: rotation, uniform scale, translation.
|
||||
///
|
||||
/// Stored as the four independent parameters rather than a 2×3 matrix so that
|
||||
/// [`Similarity::scale`] is readable without a decomposition.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub struct Similarity {
|
||||
a: f32,
|
||||
b: f32,
|
||||
tx: f32,
|
||||
ty: f32,
|
||||
}
|
||||
|
||||
impl Similarity {
|
||||
/// `x' = a·x − b·y + tx`, `y' = b·x + a·y + ty`.
|
||||
pub fn apply(&self, x: f32, y: f32) -> (f32, f32) {
|
||||
(
|
||||
self.a * x - self.b * y + self.tx,
|
||||
self.b * x + self.a * y + self.ty,
|
||||
)
|
||||
}
|
||||
|
||||
/// Uniform scale factor — destination pixels per source pixel.
|
||||
pub fn scale(&self) -> f32 {
|
||||
(self.a * self.a + self.b * self.b).sqrt()
|
||||
}
|
||||
|
||||
fn invert(&self, u: f32, v: f32) -> (f32, f32) {
|
||||
let det = self.a * self.a + self.b * self.b;
|
||||
let du = u - self.tx;
|
||||
let dv = v - self.ty;
|
||||
(
|
||||
(self.a * du + self.b * dv) / det,
|
||||
(-self.b * du + self.a * dv) / det,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/// Least-squares similarity transform from `src` onto `dst`.
|
||||
///
|
||||
/// # Why least squares and not RANSAC
|
||||
///
|
||||
/// The reference C++ implementation (docs/faces.md §1.1) fits this with
|
||||
/// OpenCV's `estimateAffinePartial2D` under RANSAC. RANSAC over five points is
|
||||
/// a strange fit: the minimal sample for a similarity is two, so it can discard
|
||||
/// landmarks it judges outliers and solve from a subset — and on a profile face
|
||||
/// the "outlier" is as likely to be the correct geometry as the wrong one.
|
||||
/// InsightFace's own pipeline uses plain least squares over all five points,
|
||||
/// which cannot silently drop anything, and that is what this is.
|
||||
///
|
||||
/// # The closed form
|
||||
///
|
||||
/// A 2-D similarity is linear in its four parameters:
|
||||
///
|
||||
/// ```text
|
||||
/// x' = a·x − b·y + tx
|
||||
/// y' = b·x + a·y + ty
|
||||
/// ```
|
||||
///
|
||||
/// so this is an ordinary linear least-squares problem, not an SVD one.
|
||||
/// Centring both point sets kills `tx`/`ty` from the normal equations and
|
||||
/// leaves `a` and `b` as two dot products over a common denominator — which is
|
||||
/// why there is no matrix decomposition anywhere in this function.
|
||||
///
|
||||
/// Returns `None` when the source points are degenerate (coincident or
|
||||
/// collinear to within f32), which does happen: a detector firing on a
|
||||
/// motion-blurred profile can put all five landmarks on a line.
|
||||
pub fn fit_similarity(src: &[(f32, f32); 5], dst: &[(f32, f32); 5]) -> Option<Similarity> {
|
||||
let n = 5.0_f32;
|
||||
let (mut sx, mut sy, mut dx, mut dy) = (0.0, 0.0, 0.0, 0.0);
|
||||
for i in 0..5 {
|
||||
sx += src[i].0;
|
||||
sy += src[i].1;
|
||||
dx += dst[i].0;
|
||||
dy += dst[i].1;
|
||||
}
|
||||
let (sx, sy, dx, dy) = (sx / n, sy / n, dx / n, dy / n);
|
||||
|
||||
let mut var = 0.0_f32;
|
||||
let mut num_a = 0.0_f32;
|
||||
let mut num_b = 0.0_f32;
|
||||
for i in 0..5 {
|
||||
let (px, py) = (src[i].0 - sx, src[i].1 - sy);
|
||||
let (qx, qy) = (dst[i].0 - dx, dst[i].1 - dy);
|
||||
var += px * px + py * py;
|
||||
num_a += px * qx + py * qy;
|
||||
num_b += px * qy - py * qx;
|
||||
}
|
||||
|
||||
// Degenerate: every landmark on one point. Collinear input still solves,
|
||||
// but with a scale that can be absurd, so the caller's sanity check on
|
||||
// `scale()` is what catches that case.
|
||||
if var <= f32::EPSILON {
|
||||
return None;
|
||||
}
|
||||
|
||||
let a = num_a / var;
|
||||
let b = num_b / var;
|
||||
if !a.is_finite() || !b.is_finite() || (a * a + b * b) <= f32::EPSILON {
|
||||
return None;
|
||||
}
|
||||
|
||||
Some(Similarity {
|
||||
a,
|
||||
b,
|
||||
tx: dx - (a * sx - b * sy),
|
||||
ty: dy - (b * sx + a * sy),
|
||||
})
|
||||
}
|
||||
|
||||
/// Warp a face onto the canonical 112×112 arrangement.
|
||||
///
|
||||
/// `rgb` is tightly packed `f32` RGB in `0.0..=1.0`, row-major — the same
|
||||
/// convention `dr-segment` uses, so both read the same proxy.
|
||||
///
|
||||
/// Sampling is bilinear **from the source in one step**: never crop-then-warp,
|
||||
/// which resamples twice and throws away detail the warp could have used.
|
||||
/// Pixels falling outside the source read as black.
|
||||
pub fn warp(
|
||||
rgb: &[f32],
|
||||
width: usize,
|
||||
height: usize,
|
||||
landmarks: &[(f32, f32); 5],
|
||||
) -> Option<Aligned112> {
|
||||
warp_pixels(Pixels::RgbF32(rgb), width, height, landmarks)
|
||||
}
|
||||
|
||||
/// TRACES: FR-CULL-8
|
||||
/// What the warp may sample, in whichever layout the caller already holds.
|
||||
///
|
||||
/// # Why the 8-bit variant exists
|
||||
///
|
||||
/// FR-CULL-8 requires the crop to come from the **native** render, and a native
|
||||
/// render is large: a 24 MP frame is 96 MB as `RGBA8` and 288 MB converted to
|
||||
/// the `f32` RGB this module was originally written against. Converting the
|
||||
/// whole frame to sample 112×112 from it is three hundred megabytes allocated
|
||||
/// to read about forty thousand pixels, per image, on a pass that runs over a
|
||||
/// whole library — and on Android it is NFR-RES-2's budget spent outright.
|
||||
///
|
||||
/// So the warp reads whatever the caller has instead. It touches so few pixels
|
||||
/// that the per-sample conversion is free, and the buffer never has to be
|
||||
/// duplicated in another layout.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub enum Pixels<'a> {
|
||||
/// Tightly packed `f32` RGB in `0.0..=1.0`, row-major.
|
||||
RgbF32(&'a [f32]),
|
||||
/// Tightly packed 8-bit RGBA, row-major. Alpha is ignored: a face crop has
|
||||
/// no use for it and carrying it would change what the embedder receives.
|
||||
Rgba8(&'a [u8]),
|
||||
}
|
||||
|
||||
impl Pixels<'_> {
|
||||
/// Whether the buffer is the size `width × height` implies.
|
||||
fn fits(&self, width: usize, height: usize) -> bool {
|
||||
match self {
|
||||
Pixels::RgbF32(v) => v.len() == width * height * 3,
|
||||
Pixels::Rgba8(v) => v.len() == width * height * 4,
|
||||
}
|
||||
}
|
||||
|
||||
/// One channel of one pixel, as `0.0..=1.0`. Outside the buffer reads black.
|
||||
///
|
||||
/// Public because the face *crop* stored for the People screen is cut from
|
||||
/// the same buffer by the same caller, and it should not need a second
|
||||
/// copy of this to do it.
|
||||
pub fn channel(&self, w: usize, h: usize, x: isize, y: isize, c: usize) -> f32 {
|
||||
if x < 0 || y < 0 || x >= w as isize || y >= h as isize {
|
||||
return 0.0;
|
||||
}
|
||||
let i = y as usize * w + x as usize;
|
||||
match self {
|
||||
Pixels::RgbF32(v) => v[i * 3 + c],
|
||||
Pixels::Rgba8(v) => v[i * 4 + c] as f32 / 255.0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// [`warp`], over any layout [`Pixels`] describes.
|
||||
pub fn warp_pixels(
|
||||
px: Pixels<'_>,
|
||||
width: usize,
|
||||
height: usize,
|
||||
landmarks: &[(f32, f32); 5],
|
||||
) -> Option<Aligned112> {
|
||||
if !px.fits(width, height) {
|
||||
return None;
|
||||
}
|
||||
let m = fit_similarity(landmarks, &ARCFACE_TEMPLATE)?;
|
||||
|
||||
let e = ALIGNED_EDGE;
|
||||
let mut pixels = vec![0.0_f32; e * e * 3];
|
||||
for v in 0..e {
|
||||
for u in 0..e {
|
||||
// Pixel centres, so the transform is not off by half a pixel —
|
||||
// which is small enough to survive review and large enough to
|
||||
// matter on a 40-pixel face.
|
||||
let (x, y) = m.invert(u as f32 + 0.5, v as f32 + 0.5);
|
||||
let (x, y) = (x - 0.5, y - 0.5);
|
||||
let out = (v * e + u) * 3;
|
||||
sample_bilinear(px, width, height, x, y, &mut pixels[out..out + 3]);
|
||||
}
|
||||
}
|
||||
|
||||
Some(Aligned112 {
|
||||
pixels,
|
||||
// The warp maps `scale` source pixels to one destination pixel, so the
|
||||
// crop spans 112/scale of the source.
|
||||
source_px: ALIGNED_EDGE as f32 / m.scale(),
|
||||
})
|
||||
}
|
||||
|
||||
fn sample_bilinear(px: Pixels<'_>, w: usize, h: usize, x: f32, y: f32, out: &mut [f32]) {
|
||||
let x0 = x.floor();
|
||||
let y0 = y.floor();
|
||||
let fx = x - x0;
|
||||
let fy = y - y0;
|
||||
let x0 = x0 as isize;
|
||||
let y0 = y0 as isize;
|
||||
|
||||
for (c, o) in out.iter_mut().enumerate() {
|
||||
let get = |xi: isize, yi: isize| -> f32 { px.channel(w, h, xi, yi, c) };
|
||||
let top = get(x0, y0) * (1.0 - fx) + get(x0 + 1, y0) * fx;
|
||||
let bot = get(x0, y0 + 1) * (1.0 - fx) + get(x0 + 1, y0 + 1) * fx;
|
||||
*o = top * (1.0 - fy) + bot * fy;
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn shifted_scaled(scale: f32, dx: f32, dy: f32, rot: f32) -> [(f32, f32); 5] {
|
||||
let (s, c) = (rot.sin(), rot.cos());
|
||||
let mut out = [(0.0, 0.0); 5];
|
||||
for (i, &(x, y)) in ARCFACE_TEMPLATE.iter().enumerate() {
|
||||
out[i] = (scale * (c * x - s * y) + dx, scale * (s * x + c * y) + dy);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn template_onto_itself_is_the_identity() {
|
||||
let m = fit_similarity(&ARCFACE_TEMPLATE, &ARCFACE_TEMPLATE).unwrap();
|
||||
for &(x, y) in &ARCFACE_TEMPLATE {
|
||||
let (u, v) = m.apply(x, y);
|
||||
assert!((u - x).abs() < 1e-3, "{u} vs {x}");
|
||||
assert!((v - y).abs() < 1e-3, "{v} vs {y}");
|
||||
}
|
||||
assert!((m.scale() - 1.0).abs() < 1e-4);
|
||||
}
|
||||
|
||||
/// The property that matters: whatever similarity the face was seen under,
|
||||
/// the fit must undo it and land the landmarks back on the template. This
|
||||
/// is the test that fails if the transform is ever "simplified" into an
|
||||
/// affine or a bare scale-and-translate.
|
||||
#[test]
|
||||
fn any_similarity_of_the_template_maps_back_onto_it() {
|
||||
for &(scale, dx, dy, rot) in &[
|
||||
(1.0_f32, 0.0_f32, 0.0_f32, 0.0_f32),
|
||||
(2.5, 100.0, -40.0, 0.0),
|
||||
(0.4, -12.0, 300.0, 0.6),
|
||||
(1.7, 5.0, 5.0, -1.2),
|
||||
] {
|
||||
let observed = shifted_scaled(scale, dx, dy, rot);
|
||||
let m = fit_similarity(&observed, &ARCFACE_TEMPLATE).unwrap();
|
||||
for (i, &(tx, ty)) in ARCFACE_TEMPLATE.iter().enumerate() {
|
||||
let (u, v) = m.apply(observed[i].0, observed[i].1);
|
||||
assert!(
|
||||
(u - tx).abs() < 1e-2 && (v - ty).abs() < 1e-2,
|
||||
"scale={scale} rot={rot}: point {i} landed at ({u}, {v}), want ({tx}, {ty})"
|
||||
);
|
||||
}
|
||||
assert!(
|
||||
(m.scale() - 1.0 / scale).abs() < 1e-3,
|
||||
"scale {} should invert {scale}",
|
||||
m.scale()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn coincident_landmarks_are_rejected_rather_than_producing_a_crop() {
|
||||
let degenerate = [(50.0, 50.0); 5];
|
||||
assert!(fit_similarity(°enerate, &ARCFACE_TEMPLATE).is_none());
|
||||
let rgb = vec![0.5_f32; 64 * 64 * 3];
|
||||
assert!(warp(&rgb, 64, 64, °enerate).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn source_px_reports_the_face_size_the_embedder_actually_saw() {
|
||||
let rgb = vec![0.5_f32; 400 * 400 * 3];
|
||||
// A face twice the template's size spans 224 source pixels.
|
||||
let big = shifted_scaled(2.0, 80.0, 80.0, 0.0);
|
||||
let a = warp(&rgb, 400, 400, &big).unwrap();
|
||||
assert!((a.source_px() - 224.0).abs() < 0.5, "{}", a.source_px());
|
||||
|
||||
// Half-size: 56 source pixels upsampled to 112, which §7 calls the
|
||||
// degraded bucket.
|
||||
let small = shifted_scaled(0.5, 10.0, 10.0, 0.0);
|
||||
let a = warp(&rgb, 400, 400, &small).unwrap();
|
||||
assert!((a.source_px() - 56.0).abs() < 0.5, "{}", a.source_px());
|
||||
}
|
||||
|
||||
/// A white square on black, warped by a transform that should centre it:
|
||||
/// checks the sampler's geometry rather than the fit's algebra.
|
||||
#[test]
|
||||
fn warp_resamples_the_right_pixels() {
|
||||
let (w, h) = (224, 224);
|
||||
let mut rgb = vec![0.0_f32; w * h * 3];
|
||||
for y in 0..h {
|
||||
for x in 0..w {
|
||||
if (56..168).contains(&x) && (56..168).contains(&y) {
|
||||
for c in 0..3 {
|
||||
rgb[(y * w + x) * 3 + c] = 1.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Landmarks placed so the fit is a pure translation of (56, 56):
|
||||
// the white square maps exactly onto the 112×112 output.
|
||||
let lm = shifted_scaled(1.0, 56.0, 56.0, 0.0);
|
||||
let a = warp(&rgb, w, h, &lm).unwrap();
|
||||
let px = a.pixels();
|
||||
for (i, v) in px.iter().enumerate() {
|
||||
assert!((v - 1.0).abs() < 1e-3, "pixel {i} is {v}, expected white");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn out_of_bounds_samples_read_black_rather_than_wrapping() {
|
||||
let rgb = vec![1.0_f32; 32 * 32 * 3];
|
||||
// Face far outside the image: every sample is out of bounds.
|
||||
let lm = shifted_scaled(1.0, 5000.0, 5000.0, 0.0);
|
||||
let a = warp(&rgb, 32, 32, &lm).unwrap();
|
||||
assert!(a.pixels().iter().all(|&v| v == 0.0));
|
||||
}
|
||||
|
||||
// ── sharpness ─────────────────────────────────────────────────────────
|
||||
|
||||
/// An image of `edge` square, filled by `f(x, y) -> luma`.
|
||||
fn image(edge: usize, f: impl Fn(usize, usize) -> f32) -> Vec<f32> {
|
||||
let mut v = Vec::with_capacity(edge * edge * 3);
|
||||
for y in 0..edge {
|
||||
for x in 0..edge {
|
||||
let l = f(x, y);
|
||||
v.extend_from_slice(&[l, l, l]);
|
||||
}
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
/// One box-blur pass, which is enough to move the metric a long way.
|
||||
fn blur(rgb: &[f32], edge: usize) -> Vec<f32> {
|
||||
let mut out = rgb.to_vec();
|
||||
for y in 1..edge - 1 {
|
||||
for x in 1..edge - 1 {
|
||||
for c in 0..3 {
|
||||
let mut sum = 0.0;
|
||||
for dy in -1isize..=1 {
|
||||
for dx in -1isize..=1 {
|
||||
let i = (((y as isize + dy) as usize) * edge
|
||||
+ ((x as isize + dx) as usize))
|
||||
* 3
|
||||
+ c;
|
||||
sum += rgb[i];
|
||||
}
|
||||
}
|
||||
out[(y * edge + x) * 3 + c] = sum / 9.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Landmarks placing the template into a larger image at scale 1, so the
|
||||
/// warp resamples one-to-one and the metric sees the source detail.
|
||||
fn centred(edge: usize) -> [(f32, f32); 5] {
|
||||
let off = (edge as f32 - ALIGNED_EDGE as f32) / 2.0;
|
||||
shifted_scaled(1.0, off, off, 0.0)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_blurred_face_scores_lower_than_a_sharp_one() {
|
||||
let edge = 200;
|
||||
let sharp = image(
|
||||
edge,
|
||||
|x, y| if (x / 3 + y / 3) % 2 == 0 { 0.9 } else { 0.1 },
|
||||
);
|
||||
let soft = blur(&blur(&sharp, edge), edge);
|
||||
|
||||
let a = warp(&sharp, edge, edge, ¢red(edge))
|
||||
.unwrap()
|
||||
.sharpness();
|
||||
let b = warp(&soft, edge, edge, ¢red(edge)).unwrap().sharpness();
|
||||
assert!(a > b * 2.0, "sharp {a} should clearly beat blurred {b}");
|
||||
}
|
||||
|
||||
/// The reason for dividing by luma variance. A sharp face photographed
|
||||
/// against a bright sky is low-contrast, and a raw Laplacian variance would
|
||||
/// reject it as blurred — which would quietly throw away every backlit
|
||||
/// portrait in the library.
|
||||
#[test]
|
||||
fn both_pixel_layouts_warp_to_the_same_crop() {
|
||||
// The 8-bit path exists so a native render need not be converted to
|
||||
// f32 whole; it has to agree with the path it replaces to within the
|
||||
// quantisation it introduces.
|
||||
let (w, h) = (64usize, 64usize);
|
||||
let mut rgba = vec![0u8; w * h * 4];
|
||||
let mut rgb = vec![0.0f32; w * h * 3];
|
||||
for y in 0..h {
|
||||
for x in 0..w {
|
||||
let v = [
|
||||
(x * 4 % 256) as u8,
|
||||
(y * 4 % 256) as u8,
|
||||
((x + y) % 256) as u8,
|
||||
];
|
||||
for c in 0..3 {
|
||||
rgba[(y * w + x) * 4 + c] = v[c];
|
||||
rgb[(y * w + x) * 3 + c] = v[c] as f32 / 255.0;
|
||||
}
|
||||
rgba[(y * w + x) * 4 + 3] = 255;
|
||||
}
|
||||
}
|
||||
let lm = shifted_scaled(0.35, 32.0, 32.0, 0.2);
|
||||
let a = warp_pixels(Pixels::RgbF32(&rgb), w, h, &lm).unwrap();
|
||||
let b = warp_pixels(Pixels::Rgba8(&rgba), w, h, &lm).unwrap();
|
||||
assert_eq!(a.source_px(), b.source_px());
|
||||
for (x, y) in a.pixels().iter().zip(b.pixels()) {
|
||||
assert!((x - y).abs() < 1e-6, "{x} vs {y}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sharpness_survives_the_contrast_being_halved() {
|
||||
let edge = 200;
|
||||
let full = image(
|
||||
edge,
|
||||
|x, y| if (x / 3 + y / 3) % 2 == 0 { 0.9 } else { 0.1 },
|
||||
);
|
||||
// Same detail, half the contrast, lifted so it does not clip.
|
||||
let flat = image(
|
||||
edge,
|
||||
|x, y| {
|
||||
if (x / 3 + y / 3) % 2 == 0 {
|
||||
0.55
|
||||
} else {
|
||||
0.45
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
let a = warp(&full, edge, edge, ¢red(edge)).unwrap().sharpness();
|
||||
let b = warp(&flat, edge, edge, ¢red(edge)).unwrap().sharpness();
|
||||
let ratio = a / b;
|
||||
assert!(
|
||||
(0.5..2.0).contains(&ratio),
|
||||
"contrast changed the score {ratio}x ({a} vs {b})"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_flat_crop_has_no_sharpness() {
|
||||
let edge = 200;
|
||||
let flat = image(edge, |_, _| 0.5);
|
||||
assert_eq!(
|
||||
warp(&flat, edge, edge, ¢red(edge)).unwrap().sharpness(),
|
||||
0.0
|
||||
);
|
||||
}
|
||||
|
||||
/// Upsampling invents no detail, so a face that had to be stretched to
|
||||
/// reach the embedder scores lower than the same face at full size. That
|
||||
/// overlap with the size floor is deliberate and documented; this pins it
|
||||
/// so a future change cannot quietly remove it.
|
||||
#[test]
|
||||
fn an_upsampled_face_scores_lower_than_the_same_face_at_full_size() {
|
||||
let edge = 200;
|
||||
let src = image(
|
||||
edge,
|
||||
|x, y| if (x / 3 + y / 3) % 2 == 0 { 0.9 } else { 0.1 },
|
||||
);
|
||||
|
||||
let full = warp(&src, edge, edge, ¢red(edge)).unwrap();
|
||||
// Half scale: the crop spans 56 source pixels and is stretched to 112.
|
||||
let off = (edge as f32 - ALIGNED_EDGE as f32 / 2.0) / 2.0;
|
||||
let small = warp(&src, edge, edge, &shifted_scaled(0.5, off, off, 0.0)).unwrap();
|
||||
|
||||
assert!(small.source_px() < full.source_px());
|
||||
assert!(
|
||||
small.sharpness() < full.sharpness(),
|
||||
"upsampled {} should be softer than full {}",
|
||||
small.sharpness(),
|
||||
full.sharpness()
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,399 +0,0 @@
|
||||
//! TRACES: FR-CULL-9 | FR-CULL-10
|
||||
//! How sure a *suggestion* is: an identity's share of the evidence for a face.
|
||||
//!
|
||||
//! [`crate::cluster`] decides which people exist; this decides what number to
|
||||
//! put beside "we think this is Anna". They are not the same question, and the
|
||||
//! answer to the second used to be a by-product of the first — the mean
|
||||
//! calibrated probability between a face and *every* other member of its group.
|
||||
//!
|
||||
//! # Why a mean over the group is the wrong number
|
||||
//!
|
||||
//! It measures the wrong thing twice over.
|
||||
//!
|
||||
//! **It punishes large, well-photographed people.** Anna has two hundred faces
|
||||
//! spanning fifteen years; a new photograph of her matches thirty of them
|
||||
//! strongly and is near-orthogonal to the rest, because a face at 8 and a face
|
||||
//! at 23 genuinely are. The mean lands around 0.2 and the interface reports a
|
||||
//! correct suggestion as a doubtful one. The better a person is covered, the
|
||||
//! worse their confidences get, which is exactly backwards.
|
||||
//!
|
||||
//! **It never asks who else it could be.** A face that matches Anna at 0.95 and
|
||||
//! matches nobody else at all, and a face that matches Anna at 0.95 *and her
|
||||
//! sister at 0.93*, are the same number under a within-group mean. The second
|
||||
//! is the one the user actually needs to look at, and it was indistinguishable
|
||||
//! from the first.
|
||||
//!
|
||||
//! # Coherence, times uniqueness
|
||||
//!
|
||||
//! Two questions, and the number is their product because they are genuinely
|
||||
//! independent: *is this the same person at all*, and *of the people we know,
|
||||
//! is it uniquely this one*.
|
||||
//!
|
||||
//! ```text
|
||||
//! evidence(P) = Σ of the top n of { P(same | this face, f) : f ∈ P }
|
||||
//! coherence = evidence(own) / (however many of the top n there were)
|
||||
//! uniqueness = evidence(own) / (evidence(own) + Σ evidence(named rivals))
|
||||
//! confidence = coherence × uniqueness
|
||||
//! ```
|
||||
//!
|
||||
//! **Coherence** is a mean, like the old number, but over the face's best
|
||||
//! [`TOP_MATCHES`] matches into the identity rather than over all of them. That single cap is what
|
||||
//! stops a well-photographed person scoring worse than a thin one: the two
|
||||
//! hundred faces a given photograph is legitimately orthogonal to no longer
|
||||
//! count against it.
|
||||
//!
|
||||
//! **Uniqueness** is the competition. A sole strong match leaves it at 1 and
|
||||
//! the confidence is the coherence; two identities matching equally well pull
|
||||
//! it to 0.5 each, and the screen has told the user the truth, which is that
|
||||
//! this face is a coin toss between two people.
|
||||
//!
|
||||
//! # Only the people the user has named compete
|
||||
//!
|
||||
//! Measured on a real 18,000-face library, normalising across *every* group
|
||||
//! made the number useless: the median suggestion read 21% and four in five
|
||||
//! read under half. The cause is not a bug in the arithmetic but a fact about
|
||||
//! clustering — one person is spread across many groups, since the pairs that
|
||||
//! would have joined them are the ones that fell short of the merge threshold.
|
||||
//! Normalising over groups therefore makes a face compete against *itself*,
|
||||
//! and the better covered the person, the more fragments there are to lose to.
|
||||
//!
|
||||
//! A fragment is not a rival. An identity the user has actually asserted is, so
|
||||
//! the denominator counts only the people they have ruled on — a group carries
|
||||
//! a [`Cluster::person`] when it holds a confirmation, a name, or an ignore —
|
||||
//! and counts them **per person, not per group**, since one person is left in
|
||||
//! several anchored groups for the same reason. Keying it by group had
|
||||
//! Catherine competing with Catherine and put the median suggestion onto a
|
||||
//! named person at 39%; keying it by person put it at 99.5%.
|
||||
//!
|
||||
//! Leave-one-out over that library's 2,702 confirmations across 54 named
|
||||
//! people, this is the regime where the number is worth having: 99.3% of faces
|
||||
//! are placed on the right person (the old mean managed 99.15%), the stated
|
||||
//! percentage is monotone in being right, and it errs low — 100% correct
|
||||
//! wherever it states 80% or more, 84% correct where it states under half.
|
||||
//! Understating is the safe direction for a screen whose whole purpose is
|
||||
//! deciding what to look at first, but it is *not* calibrated in the low bands
|
||||
//! and should not be read as though it were.
|
||||
//!
|
||||
//! Two faces of one unnamed group cannot be told from two fragments of one
|
||||
//! person by similarity alone; that is exactly why clustering stopped where it
|
||||
//! did. So the module does not pretend to: where nobody is named, uniqueness is
|
||||
//! 1 and the number falls back to plain coherence.
|
||||
//!
|
||||
//! # What it does not do
|
||||
//!
|
||||
//! It is not a merge threshold and must not become one. Clustering keeps
|
||||
//! deciding on the pairwise calibrated probability: uniqueness is *relative*,
|
||||
//! so a library with one named person in it would hand every stray face a
|
||||
//! uniqueness of 1. The absolute question ("is this the same person at all")
|
||||
//! and the comparative one ("of the people we know, which") are different, and
|
||||
//! the product is what keeps both in the answer.
|
||||
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
use crate::cluster::Cluster;
|
||||
use crate::neighbours::Pair;
|
||||
|
||||
/// Who the evidence is for.
|
||||
///
|
||||
/// Ordered rather than hashed so the sums below are reproducible; the ordering
|
||||
/// itself carries no meaning.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
|
||||
enum Identity {
|
||||
/// A person the user has confirmed a face onto. Every group anchored to
|
||||
/// them is the same identity, however many of them the clusterer left.
|
||||
Person(u64),
|
||||
/// A group nobody has ruled on. It stands for itself and competes with
|
||||
/// nothing.
|
||||
Group(usize),
|
||||
}
|
||||
|
||||
/// How many of an identity's best matches count as its evidence.
|
||||
///
|
||||
/// The cap is the whole reason the sum works: uncapped, evidence would grow
|
||||
/// with a person's face count and the largest group in the library would win
|
||||
/// every contest. Ten is enough that a person photographed from several angles
|
||||
/// contributes more than one lucky frame, and small enough that the hundred
|
||||
/// mediocre matches inside a well-covered identity cannot add up to a strong
|
||||
/// one. It is a starting point, not a measured optimum — M7's corpus is where
|
||||
/// it would be tuned.
|
||||
pub const TOP_MATCHES: usize = 10;
|
||||
|
||||
/// The weakest match that counts as evidence for an identity.
|
||||
///
|
||||
/// Rivals are half the point of this module, so the evidence scan has to reach
|
||||
/// *below* the merge threshold — a named person who matches at 0.6 will never
|
||||
/// be merged into but is precisely the competition a suggestion should be
|
||||
/// discounted for. Even odds is the natural floor: below it a pair is more
|
||||
/// likely different people than the same, and it is not a small distinction —
|
||||
/// summing the near-orthogonal pairs instead of dropping them lets fifty
|
||||
/// identities' worth of noise, each contributing its *upper tail*, outweigh one
|
||||
/// real match. Measured on a real library, that alone moved the median stated
|
||||
/// confidence from 100% to 31%.
|
||||
pub const RIVAL_FLOOR: f32 = 0.5;
|
||||
|
||||
/// How confident each face's placement is, indexed like the face slice the
|
||||
/// clusters came from.
|
||||
///
|
||||
/// A face in no group, or one with no evidence for anybody, scores 0.
|
||||
///
|
||||
/// `gallery` is one flag per face — which faces may be evidence at all
|
||||
/// ([`crate::embedding::MIN_GALLERY_QUALITY`]). Its length is the face count.
|
||||
/// A pair is evidence *about* either face but only *from* a gallery one: a
|
||||
/// probe learns from the references it matched, and a reference learns nothing
|
||||
/// from a probe that happened to match it, however well. Without that, the one
|
||||
/// short vector in a group would be the strongest match every face in it had.
|
||||
///
|
||||
/// `pairs` must be the *evidence* list — scanned at [`RIVAL_FLOOR`], not at the
|
||||
/// merge threshold. Passing the merge list still works but silently removes
|
||||
/// every rival weaker than a merge, which is most of them, and every uniqueness
|
||||
/// collapses to 1.
|
||||
pub fn identity_shares(
|
||||
gallery: &[bool],
|
||||
clusters: &[Cluster],
|
||||
pairs: &[Pair],
|
||||
top: usize,
|
||||
) -> Vec<f32> {
|
||||
let faces = gallery.len();
|
||||
// An identity is a *person*, not a group. One person routinely holds
|
||||
// several anchored groups — the same reason they hold several unnamed ones
|
||||
// — and keying this by group had Catherine competing with Catherine, which
|
||||
// on the library it was measured against put the median suggestion onto a
|
||||
// named person at 39%.
|
||||
let key_of: Vec<Identity> = clusters
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(g, c)| match c.person {
|
||||
Some(p) => Identity::Person(p),
|
||||
None => Identity::Group(g),
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut group_of = vec![usize::MAX; faces];
|
||||
for (g, c) in clusters.iter().enumerate() {
|
||||
for &m in &c.members {
|
||||
if m < faces {
|
||||
group_of[m] = g;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Ordered, not hashed: the numbers are sums of floats over these buckets
|
||||
// and this module inherits [`crate::cluster`]'s promise that the same input
|
||||
// yields the same output, bit for bit.
|
||||
let mut evidence: Vec<BTreeMap<Identity, Vec<f32>>> = vec![BTreeMap::new(); faces];
|
||||
for p in pairs {
|
||||
if p.i >= faces || p.j >= faces {
|
||||
continue;
|
||||
}
|
||||
// A pair is evidence in both directions: j's identity hears about i,
|
||||
// and i's identity hears about j. The pair list holds each unordered
|
||||
// pair once, so both have to be recorded here — each only where the
|
||||
// face doing the telling is in the gallery.
|
||||
let (gi, gj) = (group_of[p.i], group_of[p.j]);
|
||||
if gj != usize::MAX && gallery[p.j] {
|
||||
evidence[p.i]
|
||||
.entry(key_of[gj])
|
||||
.or_default()
|
||||
.push(p.probability);
|
||||
}
|
||||
if gi != usize::MAX && gallery[p.i] {
|
||||
evidence[p.j]
|
||||
.entry(key_of[gi])
|
||||
.or_default()
|
||||
.push(p.probability);
|
||||
}
|
||||
}
|
||||
|
||||
let mut out = vec![0.0; faces];
|
||||
for (i, buckets) in evidence.iter_mut().enumerate() {
|
||||
let mine = group_of[i];
|
||||
if mine == usize::MAX {
|
||||
continue;
|
||||
}
|
||||
let mine = key_of[mine];
|
||||
let mut coherence = 0.0;
|
||||
let mut ours = 0.0;
|
||||
let mut rivals = 0.0;
|
||||
for (&who, probabilities) in buckets.iter_mut() {
|
||||
// Descending, and the ties broken by nothing: equal probabilities
|
||||
// sum the same whichever order they land in.
|
||||
probabilities.sort_by(|a, b| b.total_cmp(a));
|
||||
let counted = probabilities.len().min(top);
|
||||
let score: f32 = probabilities.iter().take(top).sum();
|
||||
if who == mine {
|
||||
ours = score;
|
||||
coherence = score / counted as f32;
|
||||
} else if matches!(who, Identity::Person(_)) {
|
||||
// Only an identity the user has ruled on competes. A group
|
||||
// nobody has ruled on and that matches this face is far more
|
||||
// likely to be another fragment of the same person than a
|
||||
// different one — see the module note, and the library it was
|
||||
// measured on.
|
||||
rivals += score;
|
||||
}
|
||||
}
|
||||
let total = ours + rivals;
|
||||
// No evidence at all: a face anchored into a group it has no measured
|
||||
// similarity to. Nothing honest to report, so nothing is claimed.
|
||||
out[i] = if total > 0.0 {
|
||||
coherence * (ours / total)
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// An unnamed group: nobody has ruled on it, so it competes with nothing.
|
||||
fn cluster(members: &[usize]) -> Cluster {
|
||||
Cluster {
|
||||
members: members.to_vec(),
|
||||
person: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// A group the user has confirmed a face onto — an identity, and therefore
|
||||
/// a rival.
|
||||
fn named(members: &[usize], person: u64) -> Cluster {
|
||||
Cluster {
|
||||
members: members.to_vec(),
|
||||
person: Some(person),
|
||||
}
|
||||
}
|
||||
|
||||
fn pair(i: usize, j: usize, probability: f32) -> Pair {
|
||||
Pair { i, j, probability }
|
||||
}
|
||||
|
||||
/// `n` faces, every one of them fit to be compared against.
|
||||
fn all(n: usize) -> Vec<bool> {
|
||||
vec![true; n]
|
||||
}
|
||||
|
||||
/// The failure the module exists to fix: face 0 matches its own group's
|
||||
/// three members strongly, and the group has forty more it is unrelated to.
|
||||
/// The old within-group mean reported ~0.07 for this.
|
||||
#[test]
|
||||
fn a_large_group_does_not_dilute_a_strong_match() {
|
||||
let members: Vec<usize> = (0..44).collect();
|
||||
let clusters = vec![cluster(&members)];
|
||||
let pairs = vec![pair(0, 1, 0.99), pair(0, 2, 0.97), pair(0, 3, 0.95)];
|
||||
|
||||
let shares = identity_shares(&all(44), &clusters, &pairs, TOP_MATCHES);
|
||||
assert!(
|
||||
(shares[0] - 0.97).abs() < 1e-6,
|
||||
"the mean of its three real matches, undiluted: {}",
|
||||
shares[0]
|
||||
);
|
||||
}
|
||||
|
||||
/// Two named people matching equally well is a coin toss, and saying so is
|
||||
/// the point — this is the sibling case FR-CULL-10 warns about.
|
||||
#[test]
|
||||
fn an_ambiguous_face_splits_its_confidence_between_the_rivals() {
|
||||
let clusters = vec![named(&[0, 1, 2], 1), named(&[3, 4], 2)];
|
||||
let pairs = vec![
|
||||
pair(0, 1, 0.90),
|
||||
pair(0, 2, 0.90),
|
||||
pair(0, 3, 0.90),
|
||||
pair(0, 4, 0.90),
|
||||
];
|
||||
|
||||
let shares = identity_shares(&all(5), &clusters, &pairs, TOP_MATCHES);
|
||||
// Coherent at 0.90, and only half of the evidence is its own.
|
||||
assert!(
|
||||
(shares[0] - 0.45).abs() < 1e-6,
|
||||
"even evidence both ways: {}",
|
||||
shares[0]
|
||||
);
|
||||
}
|
||||
|
||||
/// A rival below the merge threshold still has to count, which is why the
|
||||
/// evidence scan reaches down to [`RIVAL_FLOOR`].
|
||||
#[test]
|
||||
fn a_rival_too_weak_to_merge_still_lowers_the_confidence() {
|
||||
let clusters = vec![named(&[0, 1], 1), named(&[2, 3], 2)];
|
||||
let sure = identity_shares(&all(4), &clusters, &[pair(0, 1, 0.95)], TOP_MATCHES);
|
||||
let contested = identity_shares(
|
||||
&all(4),
|
||||
&clusters,
|
||||
&[pair(0, 1, 0.95), pair(0, 2, 0.60)],
|
||||
TOP_MATCHES,
|
||||
);
|
||||
|
||||
assert_eq!(sure[0], 0.95, "nobody else to be: its coherence stands");
|
||||
assert!(
|
||||
contested[0] < 0.59 && contested[0] > 0.57,
|
||||
"0.95 coherent, but 0.95 against 0.60: {}",
|
||||
contested[0]
|
||||
);
|
||||
}
|
||||
|
||||
/// A fragment of the same person is not a rival. Measured on a real
|
||||
/// library, counting unnamed groups as competition put four suggestions in
|
||||
/// five under half — see the module note.
|
||||
#[test]
|
||||
fn an_unnamed_group_is_not_treated_as_competition() {
|
||||
let clusters = vec![cluster(&[0, 1]), cluster(&[2, 3])];
|
||||
let shares = identity_shares(
|
||||
&all(4),
|
||||
&clusters,
|
||||
&[pair(0, 1, 0.95), pair(0, 2, 0.90)],
|
||||
TOP_MATCHES,
|
||||
);
|
||||
assert_eq!(
|
||||
shares[0], 0.95,
|
||||
"an unnamed group took evidence off a suggestion"
|
||||
);
|
||||
}
|
||||
|
||||
/// The cap, doing its job: an identity with fifty mediocre matches must not
|
||||
/// beat one with ten strong ones on volume alone.
|
||||
#[test]
|
||||
fn evidence_is_capped_so_the_biggest_group_cannot_win_on_volume() {
|
||||
let small: Vec<usize> = (0..11).collect();
|
||||
let large: Vec<usize> = (11..62).collect();
|
||||
let clusters = vec![named(&small, 1), named(&large, 2)];
|
||||
|
||||
let mut pairs: Vec<Pair> = (1..11).map(|j| pair(0, j, 0.90)).collect();
|
||||
pairs.extend((11..62).map(|j| pair(0, j, 0.55)));
|
||||
|
||||
let shares = identity_shares(&all(62), &clusters, &pairs, TOP_MATCHES);
|
||||
// Ten at 0.90 against ten at 0.55 — not fifty-one at 0.55.
|
||||
assert!(
|
||||
(shares[0] - 0.90 * (9.0 / 14.5)).abs() < 1e-5,
|
||||
"capped at ten either side: {}",
|
||||
shares[0]
|
||||
);
|
||||
}
|
||||
|
||||
/// A probe learns from the references it matched; a reference learns
|
||||
/// nothing from a probe. The pair is the same pair — what differs is who
|
||||
/// is doing the telling.
|
||||
#[test]
|
||||
fn a_face_outside_the_gallery_is_nobody_s_evidence() {
|
||||
let clusters = vec![named(&[0, 1, 2], 1)];
|
||||
let gallery = vec![true, true, false];
|
||||
let pairs = vec![pair(0, 1, 0.80), pair(0, 2, 0.99), pair(1, 2, 0.99)];
|
||||
|
||||
let shares = identity_shares(&gallery, &clusters, &pairs, TOP_MATCHES);
|
||||
// Faces 0 and 1 hear only from each other: the 0.99 the probe offered
|
||||
// them is not counted.
|
||||
assert!((shares[0] - 0.80).abs() < 1e-6, "{}", shares[0]);
|
||||
assert!((shares[1] - 0.80).abs() < 1e-6, "{}", shares[1]);
|
||||
// The probe hears from both references.
|
||||
assert!((shares[2] - 0.99).abs() < 1e-6, "{}", shares[2]);
|
||||
}
|
||||
|
||||
/// A face nothing has any evidence about claims nothing.
|
||||
#[test]
|
||||
fn a_face_with_no_evidence_reports_no_confidence() {
|
||||
let clusters = vec![cluster(&[0, 1])];
|
||||
let shares = identity_shares(&all(2), &clusters, &[], TOP_MATCHES);
|
||||
assert_eq!(shares, vec![0.0, 0.0]);
|
||||
}
|
||||
}
|
||||
@@ -1,452 +0,0 @@
|
||||
//! Cosine to probability (docs/faces.md §8, FR-CULL-9).
|
||||
//!
|
||||
//! FR-CULL-9 is a hard requirement rather than an implementation detail: no
|
||||
//! code path may threshold a bare cosine, every threshold in the subsystem is
|
||||
//! stated as a probability, and the fit is per library and reports its own
|
||||
//! validity. The failure it guards against is invisible — a raw cosine means
|
||||
//! something different for every model, every population and every face size,
|
||||
//! and an uncalibrated similarity still *looks* like a plausible number all the
|
||||
//! way to the user interface.
|
||||
//!
|
||||
//! Model-free, so the part of this subsystem most likely to be subtly wrong is
|
||||
//! testable on synthetic embeddings with no weights on the machine.
|
||||
//!
|
||||
//! # Where the training pairs come from
|
||||
//!
|
||||
//! **Negatives are free and abundant.** Two faces detected in *the same
|
||||
//! photograph* are almost never the same person, which hands every multi-face
|
||||
//! image in the library a full set of negative pairs at no labelling cost — and
|
||||
//! they are *hard* negatives, from the same camera, lighting and processing,
|
||||
//! which is exactly the population where a threshold tuned on easy negatives
|
||||
//! fails. The exceptions (mirrors, photographs of photographs, collages) are
|
||||
//! rare enough to be noise at this scale.
|
||||
//!
|
||||
//! **Positives have to be earned.** A positive is a pair of faces the user has
|
||||
//! confirmed onto one person (FR-CULL-10), and there is no second source: the
|
||||
//! only labelling this subsystem has is the labelling somebody did by hand.
|
||||
//! Bootstrapping positives from a high cosine is circular — it fits the
|
||||
//! calibration to the belief it was supposed to test — and that is the whole
|
||||
//! of the alternative.
|
||||
//!
|
||||
//! docs/faces.md §8.1 names one more that would cost no labelling at all: two
|
||||
//! faces in adjacent frames of one burst are near-certainly the same person,
|
||||
//! and FR-CULL-5's grouping is sitting there. Nothing draws on it. This crate
|
||||
//! cannot see a catalog, let alone the bursts in one — it is handed cosines by
|
||||
//! whoever assembled the pair — and no caller does that assembly on its behalf
|
||||
//! yet. Until one does, and until the purity of a burst pair is *measured*
|
||||
//! rather than assumed, the positives are the confirmations and nothing else.
|
||||
//!
|
||||
//! Which is why a fresh library has **no valid calibration** — no fit of its
|
||||
//! own — and says so. It is not left without a curve: it uses the reference
|
||||
//! implementation's fitted one ([`Calibration::default`]), which is a published
|
||||
//! operating point rather than an invention, and the interface reports which of
|
||||
//! the two it is speaking from.
|
||||
|
||||
/// Bins over cosine ∈ [-1, 1].
|
||||
///
|
||||
/// 200 is the reference implementation's figure and the resolution is not
|
||||
/// critical; what matters is that there *is* a histogram. See [`Pairs`].
|
||||
const BINS: usize = 200;
|
||||
|
||||
/// Minimum evidence before a fit is trusted.
|
||||
///
|
||||
/// Far stricter than the reference implementation's floor of two positives and
|
||||
/// one negative. That floor is reasonable there: its pairs come from a curated
|
||||
/// gallery of labelled reference portraits, where a positive pair is
|
||||
/// trustworthy by construction. Here every positive is a pair somebody
|
||||
/// confirmed while working through a young library's suggestions — a handful,
|
||||
/// arriving slowly — and the whole risk is fitting confidently to too few of
|
||||
/// them.
|
||||
pub const MIN_POSITIVE_PAIRS: u64 = 200;
|
||||
pub const MIN_NEGATIVE_PAIRS: u64 = 2_000;
|
||||
|
||||
/// A fitted `P(same person | cosine, face size)`.
|
||||
///
|
||||
/// The single definition of what a similarity means in this subsystem. The
|
||||
/// catalog stores its parameters; nothing re-implements the sigmoid.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub struct Calibration {
|
||||
pub a: f32,
|
||||
pub b: f32,
|
||||
/// Weight on `log2(min crop_px)` — the face-size term FR-CULL-9 asks for.
|
||||
pub w_size: f32,
|
||||
/// Whether there was enough evidence to fit this library's own curve.
|
||||
///
|
||||
/// False means the numbers came from the built-in reference curve, and the
|
||||
/// UI says so — once, at the screen level. It does **not** mean confidences
|
||||
/// are withheld: FR-CULL-9's distinction is between presenting an untuned
|
||||
/// default *as though it were measured* and presenting it as what it is,
|
||||
/// and only the first is forbidden.
|
||||
pub valid: bool,
|
||||
pub positive_pairs: u64,
|
||||
pub negative_pairs: u64,
|
||||
}
|
||||
|
||||
impl Default for Calibration {
|
||||
/// The reference implementation's fitted MBF curve (docs/faces.md §1):
|
||||
/// steepness 16.2, P=0.5 at cosine 0.267.
|
||||
///
|
||||
/// **`valid` is false**, and that is the point. It is a documented
|
||||
/// operating point rather than an invented one, so a library with no fit of
|
||||
/// its own can both cluster and quote a probability from it — what it may
|
||||
/// not do is call that probability a measurement of *this* library.
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
a: 16.2,
|
||||
b: -16.2 * 0.267,
|
||||
w_size: 0.0,
|
||||
valid: false,
|
||||
positive_pairs: 0,
|
||||
negative_pairs: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Calibration {
|
||||
/// P(same person), shifted by a base-rate prior.
|
||||
///
|
||||
/// `log_prior_odds` is applied at evaluation rather than folded into the
|
||||
/// fit, so one stored calibration serves every context: the odds that two
|
||||
/// faces in a 40-image album match are not the odds in a 40,000-image
|
||||
/// archive. Folding a prior in would need a refit per context and would
|
||||
/// make the stored parameters mean different things depending on where they
|
||||
/// came from.
|
||||
pub fn probability(&self, cosine: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 {
|
||||
sigmoid(self.logit(cosine, min_crop_px) + log_prior_odds)
|
||||
}
|
||||
|
||||
fn logit(&self, cosine: f32, min_crop_px: f32) -> f32 {
|
||||
self.a * cosine + self.b + self.w_size * min_crop_px.max(1.0).log2()
|
||||
}
|
||||
|
||||
/// The cosine at which [`Calibration::probability`] crosses `p`.
|
||||
///
|
||||
/// What turns "merge above 0.9" into one comparison against a stored
|
||||
/// similarity, rather than a sigmoid evaluated per candidate edge.
|
||||
pub fn boundary_at(&self, p: f32, min_crop_px: f32, log_prior_odds: f32) -> f32 {
|
||||
((p / (1.0 - p)).ln() - self.b - self.w_size * min_crop_px.max(1.0).log2() - log_prior_odds)
|
||||
/ self.a
|
||||
}
|
||||
}
|
||||
|
||||
fn sigmoid(z: f32) -> f32 {
|
||||
// Branch on the sign so neither tail overflows: exp(-z) for large positive
|
||||
// z, exp(z) for large negative.
|
||||
if z >= 0.0 {
|
||||
1.0 / (1.0 + (-z).exp())
|
||||
} else {
|
||||
let e = z.exp();
|
||||
e / (1.0 + e)
|
||||
}
|
||||
}
|
||||
|
||||
/// Accumulated pair evidence, as a histogram rather than a list.
|
||||
///
|
||||
/// # Why a histogram
|
||||
///
|
||||
/// A 25,000-face library has ~3×10⁸ pairs and no gradient descent is running
|
||||
/// over that. Bucketing them costs 200 counters per class and reduces the fit
|
||||
/// to two parameters against per-bin totals; the expensive part becomes the
|
||||
/// similarity matrix, which is one blocked GEMM. This is the trick that makes a
|
||||
/// per-library fit affordable at all, and it is not obvious from outside.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Pairs {
|
||||
positive: Vec<f64>,
|
||||
negative: Vec<f64>,
|
||||
n_pos: u64,
|
||||
n_neg: u64,
|
||||
}
|
||||
|
||||
impl Default for Pairs {
|
||||
fn default() -> Self {
|
||||
Self::new()
|
||||
}
|
||||
}
|
||||
|
||||
impl Pairs {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
positive: vec![0.0; BINS],
|
||||
negative: vec![0.0; BINS],
|
||||
n_pos: 0,
|
||||
n_neg: 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Record a pair known to be the same person.
|
||||
pub fn push_positive(&mut self, cosine: f32) {
|
||||
self.positive[bin(cosine)] += 1.0;
|
||||
self.n_pos += 1;
|
||||
}
|
||||
|
||||
/// Record a pair known to be different people.
|
||||
pub fn push_negative(&mut self, cosine: f32) {
|
||||
self.negative[bin(cosine)] += 1.0;
|
||||
self.n_neg += 1;
|
||||
}
|
||||
|
||||
pub fn positives(&self) -> u64 {
|
||||
self.n_pos
|
||||
}
|
||||
pub fn negatives(&self) -> u64 {
|
||||
self.n_neg
|
||||
}
|
||||
|
||||
/// Fit `P(same) = σ(a·cos + b)` by weighted logistic regression.
|
||||
///
|
||||
/// Class weights are explicit because negatives outnumber positives by
|
||||
/// orders of magnitude, and an unweighted fit produces a well-shaped curve
|
||||
/// sitting at the wrong height — precisely the "plausible number all the
|
||||
/// way to the user interface" failure FR-CULL-9 describes.
|
||||
///
|
||||
/// Returns a calibration with `valid` set only if there was enough
|
||||
/// evidence; the parameters are filled in either way so a caller with no
|
||||
/// better option can still cluster at a documented operating point.
|
||||
pub fn fit(&self) -> Calibration {
|
||||
let base = Calibration {
|
||||
positive_pairs: self.n_pos,
|
||||
negative_pairs: self.n_neg,
|
||||
..Calibration::default()
|
||||
};
|
||||
if self.n_pos < MIN_POSITIVE_PAIRS || self.n_neg < MIN_NEGATIVE_PAIRS {
|
||||
return base;
|
||||
}
|
||||
|
||||
let total = self.n_pos as f64 + self.n_neg as f64;
|
||||
let w_pos = total / (2.0 * self.n_pos as f64);
|
||||
let w_neg = total / (2.0 * self.n_neg as f64);
|
||||
|
||||
// Start from the reference's fitted MBF curve rather than from zero:
|
||||
// it is the right order of magnitude for every model in this family,
|
||||
// so descent converges in far fewer steps and cannot wander into a
|
||||
// sign-flipped solution on thin evidence.
|
||||
let mut a = base.a as f64;
|
||||
let mut b = base.b as f64;
|
||||
const LR: f64 = 0.05;
|
||||
const MAX_ITER: usize = 20_000;
|
||||
const TOL: f64 = 1e-7;
|
||||
|
||||
for _ in 0..MAX_ITER {
|
||||
let (mut da, mut db) = (0.0, 0.0);
|
||||
for i in 0..BINS {
|
||||
let x = bin_centre(i) as f64;
|
||||
let s = 1.0 / (1.0 + (-(a * x + b)).exp());
|
||||
if self.positive[i] > 0.0 {
|
||||
let e = (s - 1.0) * w_pos * self.positive[i];
|
||||
da += e * x;
|
||||
db += e;
|
||||
}
|
||||
if self.negative[i] > 0.0 {
|
||||
let e = s * w_neg * self.negative[i];
|
||||
da += e * x;
|
||||
db += e;
|
||||
}
|
||||
}
|
||||
da /= total;
|
||||
db /= total;
|
||||
a -= LR * da;
|
||||
b -= LR * db;
|
||||
if da * da + db * db < TOL * TOL {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
Calibration {
|
||||
a: a as f32,
|
||||
b: b as f32,
|
||||
w_size: 0.0,
|
||||
valid: true,
|
||||
positive_pairs: self.n_pos,
|
||||
negative_pairs: self.n_neg,
|
||||
}
|
||||
}
|
||||
|
||||
/// How well the fit predicts the evidence, as a reliability diagram.
|
||||
///
|
||||
/// FR-CULL-9's acceptance criterion is exactly this and not a single
|
||||
/// accuracy figure: for each populated probability band, the observed match
|
||||
/// rate against the predicted one. Returned rather than asserted so the
|
||||
/// caller can show it, log it, or fail a test on it.
|
||||
pub fn reliability(&self, cal: &Calibration, bands: usize) -> Vec<ReliabilityBand> {
|
||||
let mut out = vec![
|
||||
ReliabilityBand {
|
||||
predicted: 0.0,
|
||||
observed: 0.0,
|
||||
count: 0
|
||||
};
|
||||
bands
|
||||
];
|
||||
let mut sum_pred = vec![0.0_f64; bands];
|
||||
for i in 0..BINS {
|
||||
let n_pos = self.positive[i];
|
||||
let n_neg = self.negative[i];
|
||||
if n_pos + n_neg == 0.0 {
|
||||
continue;
|
||||
}
|
||||
let p = cal.probability(bin_centre(i), 112.0, 0.0) as f64;
|
||||
let band = ((p * bands as f64) as usize).min(bands - 1);
|
||||
sum_pred[band] += p * (n_pos + n_neg);
|
||||
out[band].observed += n_pos as f32;
|
||||
out[band].count += (n_pos + n_neg) as u64;
|
||||
}
|
||||
for (band, o) in out.iter_mut().enumerate() {
|
||||
if o.count > 0 {
|
||||
o.predicted = (sum_pred[band] / o.count as f64) as f32;
|
||||
o.observed /= o.count as f32;
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
}
|
||||
|
||||
/// One row of a reliability diagram.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub struct ReliabilityBand {
|
||||
/// Mean probability the calibration predicted for pairs in this band.
|
||||
pub predicted: f32,
|
||||
/// Fraction of them that were actually the same person.
|
||||
pub observed: f32,
|
||||
pub count: u64,
|
||||
}
|
||||
|
||||
fn bin(cosine: f32) -> usize {
|
||||
let width = 2.0 / BINS as f32;
|
||||
(((cosine + 1.0) / width) as usize).min(BINS - 1)
|
||||
}
|
||||
|
||||
fn bin_centre(i: usize) -> f32 {
|
||||
let width = 2.0 / BINS as f32;
|
||||
-1.0 + (i as f32 + 0.5) * width
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Synthesise pairs from two well-separated cosine distributions, the way
|
||||
/// a real embedding space behaves: positives near 0.6, negatives near 0.05
|
||||
/// — the numbers our own end-to-end run actually produced.
|
||||
fn realistic_pairs(n_pos: u64, n_neg: u64) -> Pairs {
|
||||
let mut p = Pairs::new();
|
||||
let mut s = 12345_u32;
|
||||
let mut rand = move || {
|
||||
s = s.wrapping_mul(1_664_525).wrapping_add(1_013_904_223);
|
||||
(s >> 8) as f32 / (1u32 << 24) as f32
|
||||
};
|
||||
for _ in 0..n_pos {
|
||||
// ~N(0.60, 0.12), by summing uniforms.
|
||||
let g = (0..4).map(|_| rand()).sum::<f32>() / 4.0 - 0.5;
|
||||
p.push_positive((0.60 + g * 0.48).clamp(-1.0, 1.0));
|
||||
}
|
||||
for _ in 0..n_neg {
|
||||
let g = (0..4).map(|_| rand()).sum::<f32>() / 4.0 - 0.5;
|
||||
p.push_negative((0.05 + g * 0.32).clamp(-1.0, 1.0));
|
||||
}
|
||||
p
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_empty_library_has_no_valid_calibration() {
|
||||
let cal = Pairs::new().fit();
|
||||
assert!(!cal.valid, "a fit with no evidence must not claim validity");
|
||||
assert_eq!(cal.positive_pairs, 0);
|
||||
}
|
||||
|
||||
/// The exact case FR-CULL-9 legislates: enough negatives, too few
|
||||
/// positives. The answer is "unavailable", not a plausible-looking curve.
|
||||
#[test]
|
||||
fn too_few_positives_is_invalid_however_many_negatives_there_are() {
|
||||
let p = realistic_pairs(MIN_POSITIVE_PAIRS - 1, MIN_NEGATIVE_PAIRS * 10);
|
||||
assert!(!p.fit().valid);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn too_few_negatives_is_invalid_too() {
|
||||
let p = realistic_pairs(MIN_POSITIVE_PAIRS * 10, MIN_NEGATIVE_PAIRS - 1);
|
||||
assert!(!p.fit().valid);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_well_separated_library_fits_a_usable_curve() {
|
||||
let cal = realistic_pairs(2_000, 40_000).fit();
|
||||
assert!(cal.valid);
|
||||
assert!(cal.a > 0.0, "steepness must be positive: {}", cal.a);
|
||||
|
||||
// The decision boundary lands between the two populations.
|
||||
let boundary = cal.boundary_at(0.5, 112.0, 0.0);
|
||||
assert!(
|
||||
boundary > 0.05 && boundary < 0.60,
|
||||
"boundary {boundary} is not between the negative and positive modes"
|
||||
);
|
||||
|
||||
// And the measured cosines from the real end-to-end run fall the
|
||||
// right side of it.
|
||||
assert!(cal.probability(0.596, 200.0, 0.0) > 0.9);
|
||||
assert!(cal.probability(0.050, 200.0, 0.0) < 0.1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn probability_and_boundary_are_inverses() {
|
||||
let cal = realistic_pairs(2_000, 40_000).fit();
|
||||
for &p in &[0.1_f32, 0.5, 0.9, 0.99] {
|
||||
let cos = cal.boundary_at(p, 112.0, 0.0);
|
||||
assert!((cal.probability(cos, 112.0, 0.0) - p).abs() < 1e-3);
|
||||
}
|
||||
}
|
||||
|
||||
/// A base rate shifts the answer without a refit — the property that lets
|
||||
/// one stored calibration serve a small album and a large archive.
|
||||
#[test]
|
||||
fn a_prior_moves_the_boundary_in_the_right_direction() {
|
||||
let cal = realistic_pairs(2_000, 40_000).fit();
|
||||
let neutral = cal.probability(0.4, 112.0, 0.0);
|
||||
let pessimistic = cal.probability(0.4, 112.0, -2.0);
|
||||
let optimistic = cal.probability(0.4, 112.0, 2.0);
|
||||
assert!(pessimistic < neutral && neutral < optimistic);
|
||||
}
|
||||
|
||||
/// FR-CULL-9's acceptance criterion, run against the fit's own evidence:
|
||||
/// in every populated band, the stated probability should track the
|
||||
/// observed match rate.
|
||||
#[test]
|
||||
fn the_fit_is_reliable_on_the_evidence_it_was_fitted_to() {
|
||||
let pairs = realistic_pairs(4_000, 40_000);
|
||||
let cal = pairs.fit();
|
||||
let bands = pairs.reliability(&cal, 10);
|
||||
|
||||
let mut checked = 0;
|
||||
for b in &bands {
|
||||
// Thinly populated bands are noise, not evidence.
|
||||
if b.count < 200 {
|
||||
continue;
|
||||
}
|
||||
checked += 1;
|
||||
assert!(
|
||||
(b.predicted - b.observed).abs() < 0.15,
|
||||
"band predicted {:.3} but observed {:.3} over {} pairs",
|
||||
b.predicted,
|
||||
b.observed,
|
||||
b.count
|
||||
);
|
||||
}
|
||||
assert!(
|
||||
checked >= 2,
|
||||
"only {checked} bands had enough pairs to check"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_default_curve_is_the_references_and_is_not_marked_valid() {
|
||||
let cal = Calibration::default();
|
||||
assert!(!cal.valid);
|
||||
assert!((cal.boundary_at(0.5, 112.0, 0.0) - 0.267).abs() < 1e-3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bins_cover_the_cosine_range_without_overflowing() {
|
||||
assert_eq!(bin(-1.0), 0);
|
||||
assert_eq!(bin(1.0), BINS - 1);
|
||||
assert_eq!(bin(2.0), BINS - 1, "an out-of-range cosine must not panic");
|
||||
assert!((bin_centre(bin(0.5)) - 0.5).abs() < 0.01);
|
||||
}
|
||||
}
|
||||