From a848750a659612c69c3a12b497eddf7eeee2e925 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Thu, 30 Jul 2026 18:14:02 +0200 Subject: [PATCH] Initial implementation: core vertical slice MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Implements the core of SPEC.md — the manifest exchange, less audio-tier matching (§3) and federation (§9a), both of which the spec sequences as later work. - §2 Jmanifest format and series bundles - §3 cut matching: exact / runtime / loose tiers - §4 API, less POST /manifests/search - §5 rate limiting; §5a trust model, anonymous bearer tokens - §6 upload validation, all four stages - §7 relational storage, no JSON blob on the write path - §8 Rust + Axum + SQLite, single serialized writer, in-process job queue - §9a content addressing, computed on upload Reconciled against the system spec: - anneal_sec removed, withdrawn upstream by AR-012/AR-013. Presence follows track extent, so a track survives its own gaps and there is nothing to anneal. Its successor extinction_sec and the new gallery_scope are accepted and stored; scope enters the §7 ranking. A manifest still carrying anneal_sec is a hard 400, not silently ignored — it came from a pipeline whose window semantics differ from what this server assumes. - Audio signature: media under 120 s now emits no signature at all, matching scene-actor-extraction IR-007. The earlier §3 draft allowed a shortened window under 150 s, which was the weaker rule — a caller-varying length is the property SR-004 forbids. - UR IDs regularised to UR-nnn; docs/requirements.md registers 32 requirements, each tracing to an SR-nnn or PR-nnn. 189 tests: unit, end-to-end through the real router, and an injection suite covering SQL, JSON, header and Unicode payloads. Writing that suite found two real gaps, both fixed here: compatibility homoglyphs passed the §5a character class, and a one-frame audio signature was accepted on a feature-length item. Co-Authored-By: Claude Opus 5 --- .dockerignore | 22 + .gitea/workflows/ci.yml | 151 ++++ .gitignore | 35 + Cargo.lock | 1778 +++++++++++++++++++++++++++++++++++++ Cargo.toml | 30 + Dockerfile | 89 ++ README.md | 220 +++++ SPEC.md | 1839 +++++++++++++++++++++++++++++++++++++++ deny.toml | 91 ++ docs/requirements.md | 200 +++++ rustfmt.toml | 8 + src/api/exists.rs | 187 ++++ src/api/fetch.rs | 351 ++++++++ src/api/json.rs | 287 ++++++ src/api/mod.rs | 35 + src/api/report.rs | 141 +++ src/api/upload.rs | 241 +++++ src/app.rs | 66 ++ src/auth.rs | 186 ++++ src/castcheck.rs | 484 +++++++++++ src/config.rs | 62 ++ src/content_id.rs | 289 ++++++ src/db/mod.rs | 168 ++++ src/db/repo.rs | 1110 +++++++++++++++++++++++ src/db/schema.sql | 119 +++ src/error.rs | 83 ++ src/ingest.rs | 403 +++++++++ src/lib.rs | 38 + src/main.rs | 96 ++ src/matching.rs | 217 +++++ src/model.rs | 307 +++++++ src/ratelimit.rs | 217 +++++ src/state.rs | 102 +++ src/tmdb.rs | 226 +++++ src/validate.rs | 1055 ++++++++++++++++++++++ src/worker.rs | 494 +++++++++++ tests/api.rs | 953 ++++++++++++++++++++ tests/injection.rs | 634 ++++++++++++++ 38 files changed, 13014 insertions(+) create mode 100644 .dockerignore create mode 100644 .gitea/workflows/ci.yml create mode 100644 .gitignore create mode 100644 Cargo.lock create mode 100644 Cargo.toml create mode 100644 Dockerfile create mode 100644 README.md create mode 100644 SPEC.md create mode 100644 deny.toml create mode 100644 docs/requirements.md create mode 100644 rustfmt.toml create mode 100644 src/api/exists.rs create mode 100644 src/api/fetch.rs create mode 100644 src/api/json.rs create mode 100644 src/api/mod.rs create mode 100644 src/api/report.rs create mode 100644 src/api/upload.rs create mode 100644 src/app.rs create mode 100644 src/auth.rs create mode 100644 src/castcheck.rs create mode 100644 src/config.rs create mode 100644 src/content_id.rs create mode 100644 src/db/mod.rs create mode 100644 src/db/repo.rs create mode 100644 src/db/schema.sql create mode 100644 src/error.rs create mode 100644 src/ingest.rs create mode 100644 src/lib.rs create mode 100644 src/main.rs create mode 100644 src/matching.rs create mode 100644 src/model.rs create mode 100644 src/ratelimit.rs create mode 100644 src/state.rs create mode 100644 src/tmdb.rs create mode 100644 src/validate.rs create mode 100644 src/worker.rs create mode 100644 tests/api.rs create mode 100644 tests/injection.rs diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..4ab1a13 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,22 @@ +# Keep the build context small and free of host state. +target/ +.git/ +.gitea/ + +# Never ship an operator's database or key into an image layer. +*.db +*.db-wal +*.db-shm +*.sqlite +*.sqlite3 +.env +.env.* + +# Not needed to build. +tests/ +README.md +SPEC.md +deny.toml +Dockerfile +.dockerignore +.gitignore diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml new file mode 100644 index 0000000..a567981 --- /dev/null +++ b/.gitea/workflows/ci.yml @@ -0,0 +1,151 @@ +# Gitea Actions CI. +# +# Gitea Actions is workflow-compatible with GitHub Actions, so this runs on either +# with no changes. It needs a registered runner with the `ubuntu-latest` label. +# +# The gates, in the order they fail fastest: +# fmt — formatting, seconds +# clippy — lints, denied rather than warned +# test — 160 unit + integration tests +# deny — RustSec advisories, licence policy, source policy +# musl — the artifact §8 actually ships: one static binary + +name: CI + +on: + push: + branches: [main, master] + pull_request: + # Advisories appear without any code changing, so the dependency audit also + # runs on a schedule rather than only on push. + schedule: + - cron: "0 6 * * 1" + +env: + CARGO_TERM_COLOR: always + # Fail the build on warnings. The tree is warning-clean, so keeping it that way + # is cheaper than letting warnings accumulate. + RUSTFLAGS: "-D warnings" + +jobs: + check: + name: fmt, clippy, test + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Install Rust + run: | + # rustup is not guaranteed present on a self-hosted Gitea runner. + if ! command -v rustup >/dev/null 2>&1; then + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + | sh -s -- -y --profile minimal --component rustfmt,clippy + echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" + else + rustup component add rustfmt clippy + fi + + - name: Cache cargo + uses: actions/cache@v4 + with: + path: | + ~/.cargo/registry + ~/.cargo/git + target + key: ${{ runner.os }}-cargo-${{ hashFiles('Cargo.lock') }} + restore-keys: ${{ runner.os }}-cargo- + + - name: Formatting + run: cargo fmt --all -- --check + + - name: Clippy + run: cargo clippy --all-targets --all-features + + - name: Tests + run: cargo test --all-features + + deny: + name: advisories and licences + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Install Rust + run: | + if ! command -v rustup >/dev/null 2>&1; then + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + | sh -s -- -y --profile minimal + echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" + fi + + - name: Cache cargo-deny + uses: actions/cache@v4 + with: + path: ~/.cargo/bin/cargo-deny + key: ${{ runner.os }}-cargo-deny + + - name: Install cargo-deny + run: | + command -v cargo-deny >/dev/null 2>&1 || cargo install cargo-deny --locked + + # Advisories, licences, bans and sources — see deny.toml for why the licence + # allow-list is closed rather than a deny-list. + - name: cargo deny + run: cargo deny check + + musl: + name: static musl binary + runs-on: ubuntu-latest + # Only gate merges on the artifact build once the cheaper checks have passed. + needs: check + steps: + - uses: actions/checkout@v4 + + - name: Install Rust and musl target + run: | + if ! command -v rustup >/dev/null 2>&1; then + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + | sh -s -- -y --profile minimal + echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" + export PATH="$HOME/.cargo/bin:$PATH" + fi + rustup target add x86_64-unknown-linux-musl + sudo apt-get update && sudo apt-get install -y musl-tools + + - name: Cache cargo + uses: actions/cache@v4 + with: + path: | + ~/.cargo/registry + ~/.cargo/git + target + key: ${{ runner.os }}-musl-${{ hashFiles('Cargo.lock') }} + restore-keys: ${{ runner.os }}-musl- + + # §8: "Ship a single static binary (musl target) plus the SQLite file." + # rusqlite is built with `bundled`, so SQLite is compiled in; reqwest uses + # rustls rather than OpenSSL, so there is no system TLS dependency to link. + - name: Build + run: cargo build --release --target x86_64-unknown-linux-musl + + - name: Verify the binary is actually static + run: | + BIN=target/x86_64-unknown-linux-musl/release/jray-server + file "$BIN" + # A dynamically-linked result would defeat §8's deployment story, so this + # is asserted rather than assumed. + # + # Checked with `file`, not `ldd`: the musl target produces a static-PIE, + # and `ldd` prints the musl loader for one — an `ldd`-based check reports + # a perfectly static binary as dynamic. + if ! file "$BIN" | grep -qE 'static-pie linked|statically linked'; then + echo "::error::binary is not statically linked" >&2 + exit 1 + fi + + - name: Upload binary + uses: actions/upload-artifact@v3 + with: + name: jray-server-x86_64-musl + path: target/x86_64-unknown-linux-musl/release/jray-server + if-no-files-found: error diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..a239b91 --- /dev/null +++ b/.gitignore @@ -0,0 +1,35 @@ +# Rust build artifacts +/target/ +**/*.rs.bk +*.pdb + +# Cargo.lock is committed: this crate ships a binary, so reproducible builds +# matter more than dependency-resolution freedom. + +# SQLite database and its WAL sidecars (§7, §8). Never commit an operator's data; +# note that a plain file copy of a live WAL database is not a valid backup — +# use `VACUUM INTO` or the backup API. +*.db +*.db-wal +*.db-shm +*.sqlite +*.sqlite3 + +# Local operator configuration — holds the TMDB API key (§8). +.env +.env.* +!.env.example + +# Python artefacts from the traceability tooling +__pycache__/ +*.pyc + +# Generated traceability output — regenerate with the gate, never hand-edit. +traces-report.json + +# Editor / OS noise +.vscode/ +.idea/ +*.swp +*~ +.DS_Store diff --git a/Cargo.lock b/Cargo.lock new file mode 100644 index 0000000..ee1f4c5 --- /dev/null +++ b/Cargo.lock @@ -0,0 +1,1778 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "aho-corasick" +version = "1.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +dependencies = [ + "memchr", +] + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "atomic-waker" +version = "1.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" + +[[package]] +name = "axum" +version = "0.8.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90" +dependencies = [ + "axum-core", + "bytes", + "form_urlencoded", + "futures-util", + "http", + "http-body", + "http-body-util", + "hyper", + "hyper-util", + "itoa", + "matchit", + "memchr", + "mime", + "percent-encoding", + "pin-project-lite", + "serde_core", + "serde_json", + "serde_path_to_error", + "serde_urlencoded", + "sync_wrapper", + "tokio", + "tower", + "tower-layer", + "tower-service", + "tracing", +] + +[[package]] +name = "axum-core" +version = "0.5.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "08c78f31d7b1291f7ee735c1c6780ccde7785daae9a9206026862dab7d8792d1" +dependencies = [ + "bytes", + "futures-core", + "http", + "http-body", + "http-body-util", + "mime", + "pin-project-lite", + "sync_wrapper", + "tower-layer", + "tower-service", + "tracing", +] + +[[package]] +name = "base64" +version = "0.22.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" + +[[package]] +name = "bitflags" +version = "2.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + +[[package]] +name = "bumpalo" +version = "3.20.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" + +[[package]] +name = "bytes" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" + +[[package]] +name = "cc" +version = "1.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5add81bb678e6cb321aff7fa0dc7689ad82b112dbc032cea19f91d6b8e3582b9" +dependencies = [ + "find-msvc-tools", + "shlex", +] + +[[package]] +name = "cfg-if" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" + +[[package]] +name = "cfg_aliases" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" + +[[package]] +name = "chacha20" +version = "0.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81" +dependencies = [ + "cfg-if", + "cpufeatures 0.3.0", + "rand_core 0.10.1", +] + +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "cpufeatures" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b2a41393f66f16b0823bb79094d54ac5fbd34ab292ddafb9a0456ac9f87d201" +dependencies = [ + "libc", +] + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + +[[package]] +name = "displaydoc" +version = "0.2.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "fallible-iterator" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" + +[[package]] +name = "fallible-streaming-iterator" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" + +[[package]] +name = "find-msvc-tools" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" + +[[package]] +name = "foldhash" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" + +[[package]] +name = "form_urlencoded" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" +dependencies = [ + "percent-encoding", +] + +[[package]] +name = "futures-channel" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "262590f4fe6afeb0bc83be1daa64e52657fe185690a958af7f3ad0e92085c5ae" +dependencies = [ + "futures-core", +] + +[[package]] +name = "futures-core" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" + +[[package]] +name = "futures-task" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" + +[[package]] +name = "futures-util" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" +dependencies = [ + "futures-core", + "futures-task", + "pin-project-lite", + "slab", +] + +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + +[[package]] +name = "getrandom" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" +dependencies = [ + "cfg-if", + "js-sys", + "libc", + "wasi", + "wasm-bindgen", +] + +[[package]] +name = "getrandom" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" +dependencies = [ + "cfg-if", + "libc", + "r-efi 5.3.0", + "wasip2", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "js-sys", + "libc", + "r-efi 6.0.0", + "rand_core 0.10.1", + "wasm-bindgen", +] + +[[package]] +name = "hashbrown" +version = "0.15.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" +dependencies = [ + "foldhash", +] + +[[package]] +name = "hashlink" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7382cf6263419f2d8df38c55d7da83da5c18aef87fc7a7fc1fb1e344edfe14c1" +dependencies = [ + "hashbrown", +] + +[[package]] +name = "http" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" +dependencies = [ + "bytes", + "itoa", +] + +[[package]] +name = "http-body" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" +dependencies = [ + "bytes", + "http", +] + +[[package]] +name = "http-body-util" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" +dependencies = [ + "bytes", + "futures-core", + "http", + "http-body", + "pin-project-lite", +] + +[[package]] +name = "httparse" +version = "1.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" + +[[package]] +name = "httpdate" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" + +[[package]] +name = "hyper" +version = "1.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d22053281f852e11534f5198498373cbb59295120a20771d90f7ed1897490a72" +dependencies = [ + "atomic-waker", + "bytes", + "futures-channel", + "futures-core", + "http", + "http-body", + "httparse", + "httpdate", + "itoa", + "pin-project-lite", + "smallvec", + "tokio", + "want", +] + +[[package]] +name = "hyper-rustls" +version = "0.27.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "33ca68d021ef39cf6463ab54c1d0f5daf03377b70561305bb89a8f83aab66e0f" +dependencies = [ + "http", + "hyper", + "hyper-util", + "rustls", + "tokio", + "tokio-rustls", + "tower-service", + "webpki-roots", +] + +[[package]] +name = "hyper-util" +version = "0.1.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "96547c2556ec9d12fb1578c4eaf448b04993e7fb79cbaad930a656880a6bdfa0" +dependencies = [ + "base64", + "bytes", + "futures-channel", + "futures-util", + "http", + "http-body", + "hyper", + "ipnet", + "libc", + "percent-encoding", + "pin-project-lite", + "socket2", + "tokio", + "tower-service", + "tracing", +] + +[[package]] +name = "icu_collections" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c" +dependencies = [ + "displaydoc", + "potential_utf", + "utf8_iter", + "yoke", + "zerofrom", + "zerovec", +] + +[[package]] +name = "icu_locale_core" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29" +dependencies = [ + "displaydoc", + "litemap", + "tinystr", + "writeable", + "zerovec", +] + +[[package]] +name = "icu_normalizer" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4" +dependencies = [ + "icu_collections", + "icu_normalizer_data", + "icu_properties", + "icu_provider", + "smallvec", + "zerovec", +] + +[[package]] +name = "icu_normalizer_data" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38" + +[[package]] +name = "icu_properties" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de" +dependencies = [ + "icu_collections", + "icu_locale_core", + "icu_properties_data", + "icu_provider", + "zerotrie", + "zerovec", +] + +[[package]] +name = "icu_properties_data" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14" + +[[package]] +name = "icu_provider" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421" +dependencies = [ + "displaydoc", + "icu_locale_core", + "writeable", + "yoke", + "zerofrom", + "zerotrie", + "zerovec", +] + +[[package]] +name = "idna" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" +dependencies = [ + "idna_adapter", + "smallvec", + "utf8_iter", +] + +[[package]] +name = "idna_adapter" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb68373c0d6620ef8105e855e7745e18b0d00d3bdb07fb532e434244cdb9a714" +dependencies = [ + "icu_normalizer", + "icu_properties", +] + +[[package]] +name = "ipnet" +version = "2.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "jray-server" +version = "0.1.0" +dependencies = [ + "anyhow", + "axum", + "http-body-util", + "rand 0.9.5", + "reqwest", + "rusqlite", + "serde", + "serde_json", + "sha2", + "thiserror", + "tokio", + "tower", + "tower-http", + "tracing", + "tracing-subscriber", + "ulid", + "unicode-general-category", + "unicode-normalization", +] + +[[package]] +name = "js-sys" +version = "0.3.103" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" +dependencies = [ + "cfg-if", + "futures-util", + "wasm-bindgen", +] + +[[package]] +name = "lazy_static" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" + +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + +[[package]] +name = "libsqlite3-sys" +version = "0.35.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "133c182a6a2c87864fe97778797e46c7e999672690dc9fa3ee8e241aa4a9c13f" +dependencies = [ + "cc", + "pkg-config", + "vcpkg", +] + +[[package]] +name = "litemap" +version = "0.8.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" + +[[package]] +name = "log" +version = "0.4.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" + +[[package]] +name = "lru-slab" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "112b39cec0b298b6c1999fee3e31427f74f676e4cb9879ed1a121b43661a4154" + +[[package]] +name = "matchers" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d1525a2a28c7f4fa0fc98bb91ae755d1e2d1505079e05539e35bc876b5d65ae9" +dependencies = [ + "regex-automata", +] + +[[package]] +name = "matchit" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "47e1ffaa40ddd1f3ed91f717a33c8c0ee23fff369e3aa8772b9605cc1d22f4c3" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "mime" +version = "0.3.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6877bb514081ee2a7ff5ef9de3281f14a4dd4bceac4c09388074a6b5df8a139a" + +[[package]] +name = "mio" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "30d65c71f1ce40ab09135ce117d742b9f8a19ff91a41a8b57ed50bc2de59c427" +dependencies = [ + "libc", + "wasi", + "windows-sys 0.61.2", +] + +[[package]] +name = "nu-ansi-term" +version = "0.50.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7957b9740744892f114936ab4a57b3f487491bbeafaf8083688b16841a4240e5" +dependencies = [ + "windows-sys 0.61.2", +] + +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "percent-encoding" +version = "2.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "pkg-config" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" + +[[package]] +name = "potential_utf" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564" +dependencies = [ + "zerovec", +] + +[[package]] +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quinn" +version = "0.11.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c1a41e437b6bbd489372cd4971de128e85c855f56c57f283d20ff016cf7c0a8" +dependencies = [ + "bytes", + "cfg_aliases", + "pin-project-lite", + "quinn-proto", + "quinn-udp", + "rustc-hash", + "rustls", + "socket2", + "thiserror", + "tokio", + "tracing", + "web-time", +] + +[[package]] +name = "quinn-proto" +version = "0.11.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560" +dependencies = [ + "bytes", + "getrandom 0.4.3", + "lru-slab", + "rand 0.10.2", + "rand_pcg", + "ring", + "rustc-hash", + "rustls", + "rustls-pki-types", + "slab", + "thiserror", + "tinyvec", + "tracing", + "web-time", +] + +[[package]] +name = "quinn-udp" +version = "0.5.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694" +dependencies = [ + "cfg_aliases", + "libc", + "once_cell", + "socket2", + "tracing", + "windows-sys 0.61.2", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "rand" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" +dependencies = [ + "rand_chacha", + "rand_core 0.9.5", +] + +[[package]] +name = "rand" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c7f5fa3a058cd35567ef9bfa5e75732bee0f9e4c55fa90477bef2dfcdbc4be80" +dependencies = [ + "chacha20", + "getrandom 0.4.3", + "rand_core 0.10.1", +] + +[[package]] +name = "rand_chacha" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" +dependencies = [ + "ppv-lite86", + "rand_core 0.9.5", +] + +[[package]] +name = "rand_core" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" +dependencies = [ + "getrandom 0.3.4", +] + +[[package]] +name = "rand_core" +version = "0.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69" + +[[package]] +name = "rand_pcg" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "caa0f4137e1c0a72f4c651489402276c8e8e1cf081f3b0ba156d2cbeef09e86a" +dependencies = [ + "rand_core 0.10.1", +] + +[[package]] +name = "regex-automata" +version = "0.4.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad" +dependencies = [ + "aho-corasick", + "memchr", + "regex-syntax", +] + +[[package]] +name = "regex-syntax" +version = "0.8.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" + +[[package]] +name = "reqwest" +version = "0.12.28" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" +dependencies = [ + "base64", + "bytes", + "futures-core", + "http", + "http-body", + "http-body-util", + "hyper", + "hyper-rustls", + "hyper-util", + "js-sys", + "log", + "percent-encoding", + "pin-project-lite", + "quinn", + "rustls", + "rustls-pki-types", + "serde", + "serde_json", + "serde_urlencoded", + "sync_wrapper", + "tokio", + "tokio-rustls", + "tower", + "tower-http", + "tower-service", + "url", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", + "webpki-roots", +] + +[[package]] +name = "ring" +version = "0.17.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7" +dependencies = [ + "cc", + "cfg-if", + "getrandom 0.2.17", + "libc", + "untrusted", + "windows-sys 0.52.0", +] + +[[package]] +name = "rusqlite" +version = "0.37.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "165ca6e57b20e1351573e3729b958bc62f0e48025386970b6e4d29e7a7e71f3f" +dependencies = [ + "bitflags", + "fallible-iterator", + "fallible-streaming-iterator", + "hashlink", + "libsqlite3-sys", + "smallvec", +] + +[[package]] +name = "rustc-hash" +version = "2.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" + +[[package]] +name = "rustls" +version = "0.23.43" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" +dependencies = [ + "once_cell", + "ring", + "rustls-pki-types", + "rustls-webpki", + "subtle", + "zeroize", +] + +[[package]] +name = "rustls-pki-types" +version = "1.15.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" +dependencies = [ + "web-time", + "zeroize", +] + +[[package]] +name = "rustls-webpki" +version = "0.103.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +dependencies = [ + "ring", + "rustls-pki-types", + "untrusted", +] + +[[package]] +name = "rustversion" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" + +[[package]] +name = "ryu" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "serde_path_to_error" +version = "0.1.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10a9ff822e371bb5403e391ecd83e182e0e77ba7f6fe0160b795797109d1b457" +dependencies = [ + "itoa", + "serde", + "serde_core", +] + +[[package]] +name = "serde_urlencoded" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd" +dependencies = [ + "form_urlencoded", + "itoa", + "ryu", + "serde", +] + +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures 0.2.17", + "digest", +] + +[[package]] +name = "sharded-slab" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f40ca3c46823713e0d4209592e8d6e826aa57e928f09752619fc696c499637f6" +dependencies = [ + "lazy_static", +] + +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + +[[package]] +name = "signal-hook-registry" +version = "1.4.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c4db69cba1110affc0e9f7bcd48bbf87b3f4fc7c61fc9155afd4c469eb3d6c1b" +dependencies = [ + "errno", + "libc", +] + +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + +[[package]] +name = "smallvec" +version = "1.15.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" + +[[package]] +name = "socket2" +version = "0.6.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "stable_deref_trait" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" + +[[package]] +name = "subtle" +version = "2.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "sync_wrapper" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0bf256ce5efdfa370213c1dabab5935a12e49f2c58d15e9eac2870d3b4f27263" +dependencies = [ + "futures-core", +] + +[[package]] +name = "synstructure" +version = "0.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "thiserror" +version = "2.0.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9" +dependencies = [ + "thiserror-impl", +] + +[[package]] +name = "thiserror-impl" +version = "2.0.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "thread_local" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ad99c4c6d32803332c548b1af0540b357b3f5fc0be8f6c6bfe8b2e6ae784070" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "tinystr" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d" +dependencies = [ + "displaydoc", + "zerovec", +] + +[[package]] +name = "tinyvec" +version = "1.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" +dependencies = [ + "tinyvec_macros", +] + +[[package]] +name = "tinyvec_macros" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" + +[[package]] +name = "tokio" +version = "1.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" +dependencies = [ + "bytes", + "libc", + "mio", + "pin-project-lite", + "signal-hook-registry", + "socket2", + "tokio-macros", + "windows-sys 0.61.2", +] + +[[package]] +name = "tokio-macros" +version = "2.7.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "tokio-rustls" +version = "0.26.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1729aa945f29d91ba541258c8df89027d5792d85a8841fb65e8bf0f4ede4ef61" +dependencies = [ + "rustls", + "tokio", +] + +[[package]] +name = "tower" +version = "0.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4" +dependencies = [ + "futures-core", + "futures-util", + "pin-project-lite", + "sync_wrapper", + "tokio", + "tower-layer", + "tower-service", + "tracing", +] + +[[package]] +name = "tower-http" +version = "0.6.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" +dependencies = [ + "bitflags", + "bytes", + "futures-util", + "http", + "http-body", + "http-body-util", + "pin-project-lite", + "tokio", + "tower", + "tower-layer", + "tower-service", + "tracing", + "url", +] + +[[package]] +name = "tower-layer" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "121c2a6cda46980bb0fcd1647ffaf6cd3fc79a013de288782836f6df9c48780e" + +[[package]] +name = "tower-service" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3" + +[[package]] +name = "tracing" +version = "0.1.44" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" +dependencies = [ + "log", + "pin-project-lite", + "tracing-attributes", + "tracing-core", +] + +[[package]] +name = "tracing-attributes" +version = "0.1.31" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "tracing-core" +version = "0.1.36" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" +dependencies = [ + "once_cell", + "valuable", +] + +[[package]] +name = "tracing-log" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ee855f1f400bd0e5c02d150ae5de3840039a3f54b025156404e34c23c03f47c3" +dependencies = [ + "log", + "once_cell", + "tracing-core", +] + +[[package]] +name = "tracing-subscriber" +version = "0.3.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb7f578e5945fb242538965c2d0b04418d38ec25c79d160cd279bf0731c8d319" +dependencies = [ + "matchers", + "nu-ansi-term", + "once_cell", + "regex-automata", + "sharded-slab", + "smallvec", + "thread_local", + "tracing", + "tracing-core", + "tracing-log", +] + +[[package]] +name = "try-lock" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" + +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + +[[package]] +name = "ulid" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "470dbf6591da1b39d43c14523b2b469c86879a53e8b758c8e090a470fe7b1fbe" +dependencies = [ + "rand 0.9.5", + "web-time", +] + +[[package]] +name = "unicode-general-category" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b993bddc193ae5bd0d623b49ec06ac3e9312875fdae725a975c51db1cc1677f" + +[[package]] +name = "unicode-ident" +version = "1.0.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" + +[[package]] +name = "unicode-normalization" +version = "0.1.25" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8" +dependencies = [ + "tinyvec", +] + +[[package]] +name = "untrusted" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" + +[[package]] +name = "url" +version = "2.5.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" +dependencies = [ + "form_urlencoded", + "idna", + "percent-encoding", + "serde", +] + +[[package]] +name = "utf8_iter" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" + +[[package]] +name = "valuable" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ba73ea9cf16a25df0c8caa16c51acb937d5712a8429db78a3ee29d5dcacd3a65" + +[[package]] +name = "vcpkg" +version = "0.2.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "want" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bfa7760aed19e106de2c7c0b581b509f2f25d3dacaf737cb82ac61bc6d760b0e" +dependencies = [ + "try-lock", +] + +[[package]] +name = "wasi" +version = "0.11.1+wasi-snapshot-preview1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" + +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + +[[package]] +name = "wasm-bindgen" +version = "0.2.126" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" +dependencies = [ + "cfg-if", + "once_cell", + "rustversion", + "wasm-bindgen-macro", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-futures" +version = "0.4.76" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "wasm-bindgen-macro" +version = "0.2.126" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" +dependencies = [ + "quote", + "wasm-bindgen-macro-support", +] + +[[package]] +name = "wasm-bindgen-macro-support" +version = "0.2.126" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" +dependencies = [ + "bumpalo", + "proc-macro2", + "quote", + "syn 2.0.119", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-shared" +version = "0.2.126" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "web-sys" +version = "0.3.103" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "web-time" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "webpki-roots" +version = "1.0.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7dcd9d09a39985f5344844e66b0c530a33843579125f23e21e9f0f220850f22a" +dependencies = [ + "rustls-pki-types", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-sys" +version = "0.52.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" +dependencies = [ + "windows-targets", +] + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-targets" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" +dependencies = [ + "windows_aarch64_gnullvm", + "windows_aarch64_msvc", + "windows_i686_gnu", + "windows_i686_gnullvm", + "windows_i686_msvc", + "windows_x86_64_gnu", + "windows_x86_64_gnullvm", + "windows_x86_64_msvc", +] + +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" + +[[package]] +name = "windows_aarch64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" + +[[package]] +name = "windows_i686_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" + +[[package]] +name = "windows_i686_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" + +[[package]] +name = "windows_i686_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" + +[[package]] +name = "windows_x86_64_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" + +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" + +[[package]] +name = "windows_x86_64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" + +[[package]] +name = "writeable" +version = "0.6.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" + +[[package]] +name = "yoke" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" +dependencies = [ + "stable_deref_trait", + "yoke-derive", + "zerofrom", +] + +[[package]] +name = "yoke-derive" +version = "0.8.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", + "synstructure", +] + +[[package]] +name = "zerocopy" +version = "0.8.55" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b5a105cd7b140f6eeec8acff2ea38135d3cab283ada58540f629fe51e46696eb" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.55" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0fe976fb70c78cd64cccfe3a6fc142244e8a77b70959b30faf9d0ac37ee228eb" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zerofrom" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" +dependencies = [ + "zerofrom-derive", +] + +[[package]] +name = "zerofrom-derive" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", + "synstructure", +] + +[[package]] +name = "zeroize" +version = "1.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" + +[[package]] +name = "zerotrie" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf" +dependencies = [ + "displaydoc", + "yoke", + "zerofrom", +] + +[[package]] +name = "zerovec" +version = "0.11.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239" +dependencies = [ + "yoke", + "zerofrom", + "zerovec-derive", +] + +[[package]] +name = "zerovec-derive" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/Cargo.toml b/Cargo.toml new file mode 100644 index 0000000..90afaa7 --- /dev/null +++ b/Cargo.toml @@ -0,0 +1,30 @@ +[package] +name = "jray-server" +version = "0.1.0" +edition = "2021" +rust-version = "1.85" +license = "GPL-3.0-or-later" +description = "JRay public server — community manifest exchange" + +[dependencies] +axum = { version = "0.8", features = ["json", "query"] } +tokio = { version = "1", features = ["rt-multi-thread", "macros", "signal", "sync", "time"] } +tower = "0.5" +tower-http = { version = "0.6", features = ["trace", "timeout", "limit"] } +serde = { version = "1", features = ["derive"] } +serde_json = "1" +rusqlite = { version = "0.37", features = ["bundled"] } +reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls"] } +tracing = "0.1" +tracing-subscriber = { version = "0.3", features = ["env-filter"] } +unicode-normalization = "0.1" +unicode-general-category = "1" +sha2 = "0.10" +rand = "0.9" +ulid = "1" +thiserror = "2" +anyhow = "1" + +[dev-dependencies] +tower = { version = "0.5", features = ["util"] } +http-body-util = "0.1" diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..f479362 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,89 @@ +# Deploy image for jray-server. +# +# §8 is explicit that neither Docker nor Compose should be *required* — the +# primary artifact is a single static binary plus one database file, and that is +# deliberately the lowest-friction thing a hobbyist operator can deploy. This +# image is the optional convenience, not the intended path. +# +# It builds against musl so the runtime stage can be `scratch`: no libc, no shell, +# no package manager, nothing to keep patched. rusqlite is built with `bundled` +# (SQLite compiled in) and reqwest with rustls rather than OpenSSL, so there is +# genuinely nothing left to link against. +# +# docker build -t jray-server . +# docker run --rm -p 8080:8080 -v jray-data:/data \ +# -e JRAY_TMDB_API_KEY=... jray-server + +FROM rust:1.92-alpine AS builder + +# `musl-dev` for the C toolchain rusqlite's bundled SQLite needs; `file` for the +# static-linkage assertion below. +RUN apk add --no-cache musl-dev file + +WORKDIR /build + +# Dependencies first, in their own layer, so editing source does not re-download +# and rebuild the entire tree. +COPY Cargo.toml Cargo.lock ./ +RUN mkdir -p src \ + && echo 'fn main() {}' > src/main.rs \ + && echo '' > src/lib.rs \ + && cargo build --release --target x86_64-unknown-linux-musl \ + && rm -rf src + +COPY src ./src + +# `touch` defeats the cargo staleness check that the dummy-source trick above +# would otherwise leave in place. +RUN touch src/main.rs src/lib.rs \ + && cargo build --release --target x86_64-unknown-linux-musl \ + && strip target/x86_64-unknown-linux-musl/release/jray-server + +# Verify the result is genuinely static. A dynamically-linked binary would fail at +# runtime on `scratch`, and failing here is far easier to diagnose. +# +# Asserted with `file`, not `ldd`: the musl target produces a **static-PIE**, and +# `ldd` prints the musl loader path for one, so an `ldd`-based check reports a +# static binary as dynamic. `file` reports "static-pie linked" and is unambiguous. +RUN file target/x86_64-unknown-linux-musl/release/jray-server | tee /tmp/linkage \ + && grep -qE 'static-pie linked|statically linked' /tmp/linkage \ + || (echo "binary is not statically linked; it will not run on scratch" && exit 1) + +# Stage the data directory with the runtime uid's ownership. `scratch` has no +# shell, so this cannot be done in the final stage — and a bare `VOLUME` there +# would be created root-owned, leaving the non-root process unable to create the +# database at all. +RUN mkdir -p /staged-data && chown 65534:65534 /staged-data + +# --------------------------------------------------------------------------- + +FROM scratch + +COPY --from=builder /build/target/x86_64-unknown-linux-musl/release/jray-server /jray-server + +# The database lives on a volume; §8 warns that a plain file copy of a live WAL +# database is not a valid backup, so back it up with `VACUUM INTO` from the host +# rather than by archiving this directory. +# +# Copied from the builder so it arrives owned by the runtime uid. Docker seeds a +# named volume from the image's directory, ownership included, so the server can +# create the database on first run. A bind mount is *not* seeded this way — the +# host directory keeps its own ownership, so it must be made writable by uid +# 65534 (`chown 65534:65534 /path/on/host`). +COPY --from=builder --chown=65534:65534 /staged-data /data +VOLUME ["/data"] + +# Non-root. `scratch` has no /etc/passwd, so this is a bare uid — which is all the +# kernel needs, and the binary touches nothing outside /data. +USER 65534:65534 + +ENV JRAY_BIND=0.0.0.0:8080 \ + JRAY_DB=/data/jray.db + +EXPOSE 8080 + +# No HEALTHCHECK: it would need a shell or curl, and `scratch` has neither. +# §4 provides `GET /health` (liveness) and `/ready` (database and migrations) for +# an orchestrator to probe externally, which is the right place for it. + +ENTRYPOINT ["/jray-server"] diff --git a/README.md b/README.md new file mode 100644 index 0000000..bd277ce --- /dev/null +++ b/README.md @@ -0,0 +1,220 @@ +# JRay public server + +A community manifest exchange for JRay. Jellyfin servers running the JRay plugin +pull actor-timeline manifests ("Jmanifests") for titles they own instead of +running the CV pipeline locally, and optionally contribute the manifests they +generate back. + +See [SPEC.md](SPEC.md) for the design. Section references throughout the code +point at it. + +The community instance is **`https://jray.tourolle.paris`**. The JRay plugin +ships with it pre-configured but **disabled** — §9 requires that no traffic leave +an installation until an admin opts in, so the default entry exists to save the +admin from typing a URL, not to enable sharing on their behalf. Set +`JRAY_SERVER_ID=jray.tourolle.paris` when deploying that instance: it becomes the +`origin` stamped on manifests it first accepts (§9a) and the salt for report IP +hashes. + +## Status + +First implementation pass: the **core vertical slice**, reconciled against the +[system spec](../SPEC.md). + +Per-requirement status is in [`docs/requirements.md`](docs/requirements.md) — +32 requirements (`UR-001..018`, `DR-001..014`), each tracing up to an `SR-nnn` +or `PR-nnn`. `UR-015..018` are the pending SR-003 schema bump and are marked +`Planned` rather than omitted. + +Implemented: + +- §2 Jmanifest format and series bundles +- §3 cut matching — `exact` / `runtime` / `loose` tiers +- §4 the API surface, less `POST /manifests/search` +- §5 rate limiting, in-process fixed-window counters +- §5a trust model — anonymous bearer tokens, closed-vocabulary storage, + automatic revocation +- §6 upload validation, all four stages +- §7 relational storage, no JSON blobs on the write path +- §8 Rust + Axum + SQLite, single serialized writer, in-process job queue +- §9a content addressing (`content_id`), computed on upload + +Reconciled with the system spec (see `docs/requirements.md` for the detail): + +- **`anneal_sec` removed.** Withdrawn upstream by AR-012/AR-013 — presence now + follows track extent, so a track survives its own gaps and there is nothing to + anneal. Its successor `extinction_sec` and the new `gallery_scope` are + accepted, stored, and (for scope) used in §7 ranking. A manifest still + carrying `anneal_sec` is now a hard `400`, not silently ignored: it was + produced by a pipeline whose window semantics differ from what this server + assumes. +- **Audio signature: the 120 s rule now matches both producers.** An earlier + draft of §3 allowed a shortened window for items under 150 s; that conflicted + with `scene-actor-extraction` IR-007 and was the weaker rule, since a + caller-varying length is the property SR-004 forbids. Items under 120 s now + send no signature at all. + +Deferred: + +- §3 audio signatures — the field is **accepted, validated and stored**, and + `content_id` already excludes it, but `audio`-tier matching and + `POST /manifests/search` are not wired up. This follows §3's own recommended + sequencing: ship the plugin-side computation first, let signatures accumulate, + then enable matching once coverage is useful. +- §9a federation endpoints (`/federation/*`) and the pull worker. The schema + columns (`content_id`, `origin`, `ingested_from`, `peers`) are in place, and + `ingest::persist` is already the shared path a pull would reuse. + +## Running + +```sh +cargo run +``` + +Configuration is entirely environment variables: + +| Variable | Default | Purpose | +|---|---|---| +| `JRAY_BIND` | `127.0.0.1:8080` | Listen address. Terminate TLS at the operator's proxy (§8) | +| `JRAY_DB` | `jray.db` | SQLite path. WAL mode, created on first run | +| `JRAY_TMDB_API_KEY` | — | **Hard dependency for UR-3.** Without it, uploads stay `pending` and are never listed | +| `JRAY_TMDB_BASE_URL` | `https://api.themoviedb.org/3` | Override for testing | +| `JRAY_TRUSTED_PROXIES` | — | Comma-separated proxy IPs whose `X-Forwarded-For` is honoured. **Not default-on**: §5 rate limiting and report attribution key on client IP, so a spoofable header defeats both | +| `JRAY_SERVER_ID` | `localhost` | This server's identity, used as manifest `origin` and as the report IP-hash salt | +| `JRAY_REQUEST_TIMEOUT_SEC` | `30` | Request timeout so a slow bundle query fails fast | +| `JRAY_JOB_BATCH` | `8` | Cast-check jobs leased per worker tick | +| `JRAY_JOB_POLL_SEC` | `5` | Worker poll interval | +| `JRAY_LOG` | `info` | `tracing` filter | + +Contributing requires a token (§5a) — an anonymous bearer capability, not an +account. Self-issue one: + +```sh +curl -sX POST -H 'content-type: application/json' -d '{}' \ + http://127.0.0.1:8080/api/v1/tokens +``` + +## Deployment + +§8's deployment notes are requirements, not suggestions: + +- Enforce the body cap at **both** the proxy and the app. `client_max_body_size` + (nginx) / `request_body max_size` (Caddy) should match §6 stage 1, so oversized + uploads are dropped at the edge and never occupy an application worker. The app + must also be safe when run without a proxy, which it is. +- Set `JRAY_TRUSTED_PROXIES` to the proxy's address, or `X-Forwarded-For` is + ignored and every client behind it shares one rate-limit bucket. +- **Back up the SQLite file with `VACUUM INTO` or the backup API** — never a + plain file copy of a live WAL database. Manifests represent real CV compute. + +## Tests + +```sh +cargo test # 178 tests +cargo deny check # advisories, licences, bans, sources +``` + +Unit tests per module, plus two integration suites: + +- `tests/api.rs` — end-to-end through the real router: status codes, headers, and + the properties that only hold if the layers compose correctly (per-route body + caps, rate-limit surfaces, the strict schema actually reaching uploads). +- `tests/injection.rs` — that hostile input cannot escape its layer: SQL payloads + in query parameters, path segments, JSON bodies, bearer tokens and report notes; + JSON structure abuse; CRLF header injection; path traversal; and Unicode tricks + against the §5a character class. + +Security-relevant properties are asserted rather than assumed — +`movie`/`jellyfin_id` rejection, the §5a character class defeating base64/hex +smuggling, per-route body caps, a lying `Content-Length` not bypassing the cap, +and a forged `X-Forwarded-For` not resetting a rate-limit budget. + +**On injection specifically.** Two independent defences apply, and they fail +differently, so both are tested: + +1. **Parameterised queries.** Every value reaches SQLite through `params![]`. The + only `format!`-built SQL interpolates two compile-time constants (a column list + and a status literal) — no runtime input ever becomes SQL syntax. This is what + actually prevents injection. +2. **Closed-vocabulary validation.** Identifiers are regex-constrained and free + text is limited to a closed character class, so most payloads never reach the + query layer at all. + +The injection suite would still pass on defence 1 alone, which is deliberate: if +validation were ever loosened, the tests should not silently start depending on it. + +Writing that suite found two real gaps, both since fixed: + +- Compatibility homoglyphs (`𝐒𝐭𝐞𝐯𝐞`, `Actor`) passed the §5a class. They are + letters by Unicode category and NFC does not fold them — only NFKC would. Beyond + name spoofing, a fullwidth-digit alphabet would have reopened the encoding + channel the "no digits" rule exists to close. +- A one-frame `audio_signature` was accepted on a feature-length manifest, making + the field the variable-length container §3 explicitly forbids. The length floor + is now derived from the declared runtime, keeping §3's genuine short-item + exception without trusting the client's length. + +Two tests worth knowing about: + +- `content_id::tests::golden_vector_hash_is_stable` locks the §9a canonical form. + **The JRay plugin must reproduce it byte-identically**; §8 notes the extraction + side is Python, so this can no longer be one shared implementation and must be + cross-tested instead. `GOLDEN_VECTORS` is that fixture, and its hash was + verified against an independent Python implementation. +- The integration tests use an on-disk temporary database, not `:memory:`, + because §8's topology is one writer connection plus a read pool — in-memory + SQLite is per-connection, so the readers would see an empty database. + +## CI and container image + +`.gitea/workflows/ci.yml` runs on Gitea Actions (and unmodified on GitHub +Actions), gated fastest-first: `fmt` → `clippy` → `test` → `cargo deny` → static +musl build. The dependency audit also runs weekly, since advisories appear without +any code changing. + +The `Dockerfile` is the optional convenience, not the intended deployment path — +§8 is explicit that neither Docker nor Compose should be *required*. It builds +against musl and runs from `scratch` as uid 65534: **8.7 MB**, no libc, no shell, +no package manager. `rusqlite` bundles SQLite and `reqwest` uses rustls, so there +is nothing left to link. + +```sh +docker build -t jray-server . +docker run --rm -p 8080:8080 -v jray-data:/data -e JRAY_TMDB_API_KEY=... jray-server +``` + +Two things that are easy to get wrong, so they are handled explicitly: + +- **Static linkage is asserted with `file`, not `ldd`.** The musl target produces + a *static-PIE*, and `ldd` prints the musl loader for one — an `ldd`-based check + reports a perfectly static binary as dynamic. +- **`/data` is staged with the runtime uid's ownership.** Docker seeds a named + volume from the image directory, ownership included, so a non-root server can + create the database on first run. A **bind mount is not seeded this way** — the + host directory keeps its own ownership, so `chown 65534:65534` it first or the + server exits with "unable to open database file". + +Two tests are worth knowing about: + +- `content_id::tests::golden_vector_hash_is_stable` locks the §9a canonical form. + **The JRay plugin must reproduce it byte-identically**; §8 notes the extraction + side is Python, so this can no longer be one shared implementation and must be + cross-tested instead. `GOLDEN_VECTORS` is that fixture, and its hash was + verified against an independent Python implementation. +- The integration tests use an on-disk temporary database, not `:memory:`, + because §8's topology is one writer connection plus a read pool — in-memory + SQLite is per-connection, so the readers would see an empty database. + +## Known gaps + +- **The §6 stage-3 thresholds are still §10's guesses.** §10 (5) is explicit that + running the check over the 331 real corpus files would give the true + distribution of honest-upload match ratios, and is "the single cheapest way to + de-risk UR-3 and UR-5". The scoring logic is deliberately pure functions in + `castcheck.rs` so that retuning is a test-data exercise, not a code change. +- `POST /manifests/{id}/report` records reports but nothing consumes them yet. + §5a's divergence detection and the operator kill switch are not implemented; + delisting is currently a manual `UPDATE`, which §5a does note is the intended + shape ("one UPDATE", not a moderation queue). +- No admin surface. Revocation is automatic (§5a), but an operator has no + endpoint for the deliberate kill-switch case. diff --git a/SPEC.md b/SPEC.md new file mode 100644 index 0000000..939b12d --- /dev/null +++ b/SPEC.md @@ -0,0 +1,1839 @@ +# JRay Public Server — specification + +A community manifest exchange for JRay. Jellyfin servers running the JRay +plugin pull actor-timeline manifests ("Jmanifests") for titles they own +instead of running the CV pipeline locally, and optionally contribute the +manifests they generate back. + +Status: **core implemented.** See [`docs/requirements.md`](docs/requirements.md) +for per-requirement status and [`README.md`](README.md) for what is deferred. + +This is a *software* spec: its job is to implement the +[system spec](../SPEC.md), which owns everything spanning more than one repo. +Requirements here trace up to an `SR-nnn`; the prose below is the detail. + +--- + +## 0. Requirements + +IDs are `UR-nnn`, zero-padded and **permanent** — a withdrawn requirement keeps +its number, because renumbering is what produces orphan TRACES tags +([system spec](../SPEC.md) §6). The authoritative list with status lives in +[`docs/requirements.md`](docs/requirements.md); this table is the prose anchor. + +| # | Requirement | Traces to | Where addressed | +|---|---|---|---| +| UR-001 | Query whether a JRay manifest for a given media item exists on the server | SR-001 | §4 `GET /manifests/exists` | +| UR-002 | Route to post a JRay manifest for a media item | PR-006 | §4 `POST /manifests` | +| UR-003 | Content verification: no additional JSON fields, file size limit, approximate cast match against TMDB | SR-004 | §6 | +| UR-004 | Rate limiting on queries | SR-004 | §5 | +| UR-005 | Trust without account management: server must not be usable as a content store, nor for prank/vandalism manifests | SR-004 | §5a | +| UR-006 | Serve and accept a whole series in one operation, not episode-by-episode | PR-006 | §2 series bundles, §4 `GET /manifests/series`, `POST /manifests/bundle` | +| UR-007 | JRay plugin must query a configurable list of servers | PR-005 | §9 | +| UR-008 | Servers must be able to sync/replicate manifests between each other | PR-006 | §9a | +| UR-009 | Store an audio spectral-peak signature from the media centre, so a file of unknown providence can be identified and synchronised | SR-003 | §3 audio signature | +| UR-010 | Identity crossing the API boundary is TMDB/IMDB ids, never a name alone | SR-001 | §2, §6 stage 3, §7 | +| UR-011 | Reject any manifest field capable of carrying binary or attacker-chosen content | SR-004 | §5a Threat 1, §6 stage 2 | +| UR-012 | Never accept, store, or serve gallery data — reference faces or embeddings | SR-005 | §5a, and the absence of any such field in §2 | +| UR-013 | Windows are scene-scoped claims; the server must not reinterpret their boundaries | SR-002 | §2 field notes, §6 | +| UR-014 | Reject a manifest whose `schema_version` / `jmanifest_version` is unknown, never guess | SR-003 | §2, §6 stage 2 | + +Two notes on UR-001. An existence check is deliberately a *separate, cheaper* +endpoint from the fetch in §4 — it answers "should I bother?" for a whole +library sweep without transferring payloads, and it is the endpoint a +scheduled task will hammer. It is also the most abuse-prone surface, since +it doubles as an oracle for "does the community have this title" — so it is +rate-limited harder than the fetches and returns no manifest content. + +UR-003's three checks are different in kind and are enforced at different +stages: field strictness and size are cheap and synchronous (reject at the +door), whereas the TMDB cast match needs an outbound API call and so runs +asynchronously after a `202`. See §6. + +**UR-010 to UR-014 were added when this spec was reconciled against the system +spec.** They are not new work — each states a property the design already had, +which had been left implicit because no system requirement existed to trace it +to. UR-012 and UR-013 are the two worth stating explicitly: the server's refusal +to carry gallery data (SR-005) and its refusal to reinterpret window boundaries +(SR-002) are both invariants preserved by *not* doing something, and an +unstated prohibition is the kind that erodes. + +--- + +## 1. Why this needs more than the current truth file + +The extraction pipeline (`result_sink_node`) writes: + +```json +{ + "schema_version": 1, + "movie": "/data/movies/Movie.mkv", + "sample_fps": 1, + "anneal_sec": 2, + "actors": [ + { "name": "...", "imdb_id": "...", "tmdb_id": "...", "jellyfin_id": "...", + "scenes": [[12.0, 45.0]] } + ] +} +``` + +Three properties of this format block sharing as-is: + +1. **No portable title identity.** `movie` is an absolute path on the machine + that ran extraction. Nothing in the file says "this is The Death of Stalin + (2017)". The receiving server cannot tell what it just downloaded. +2. **Installation-local identifiers.** `movie` leaks the contributor's + directory layout and `jellyfin_id` is a GUID from the contributor's + database — meaningless and mildly identifying elsewhere. Both must be + stripped on upload, not merely ignored on download. +3. **Timings are cut-specific.** `scenes` are absolute seconds. A theatrical + cut, an extended cut, a PAL speed-up, and a release with 40s of distributor + logos all produce different timelines for the same TMDB id. Keying purely + on TMDB id would silently serve misaligned overlays. + +The Jmanifest format below is the truth file plus a portable identity block +and a cut fingerprint; the actor timeline payload is unchanged. + +### Terminology + +- **Jmanifest** — one shareable actor timeline for one *cut* of one title. +- **Title identity** — what the work is (TMDB/IMDB id + episode coordinates). +- **Cut fingerprint** — which encode/edit the timings apply to (runtime, plus + optional stronger signals). + +--- + +## 2. Jmanifest format + +```json +{ + "jmanifest_version": 1, + "identity": { + "type": "movie", + "tmdb_id": "504172", + "imdb_id": "tt4686844", + "title": "The Death of Stalin", + "year": 2017 + }, + "cut": { + "runtime_sec": 6420.5, + "container_duration_sec": 6420.5, + "video_hash": "opensubtitles:8e245d9679d31e12", + "audio_signature": "v1:v7fA3k…" + }, + "extraction": { + "sample_fps": 5, + "extinction_sec": 12, + "gallery_size": 1820, + "gallery_scope": "global", + "pipeline_version": "scene-actor-extraction 0.4.1" + }, + "actors": [ + { + "name": "Steve Buscemi", + "imdb_id": "nm0000114", + "tmdb_id": "884", + "scenes": [[191.6, 209.2], [438.2, 465.6]] + } + ] +} +``` + +For an episode, `identity` is: + +```json +{ + "type": "episode", + "series_tmdb_id": "1396", + "series_imdb_id": "tt0903747", + "title": "Breaking Bad", + "season": 2, + "episode": 5 +} +``` + +Field notes: + +- `jmanifest_version` — separate from the plugin's `schema_version`; this + versions the *exchange* envelope. +- `identity.tmdb_id` / `imdb_id` — at least one required. These are the + lookup keys. +- `cut.runtime_sec` — **required**, the decoded duration of the media the + timings came from. This is the primary alignment guard. +- `cut.video_hash` — optional but strongly preferred. See §3. +- `cut.audio_signature` — optional; a spectral-peak signature from the media + centre, version-prefixed (`v1:`). Enables content-based matching and offset + recovery for files of unknown providence. See §3. +- `extraction.extinction_sec` — the re-acquisition timeout that shapes window + extent. Replaces `anneal_sec`; see the schema-bump note below. +- `extraction.gallery_scope` — `global` or `limited`. The strongest available + quality signal when ranking competing manifests for one cut (§7), since a + gallery built from the whole library competes against every actor in it, + whereas a per-title gallery does not. +- `actors[].jellyfin_id` — **must not appear.** The server rejects uploads + containing it (see §6). +- `movie` (absolute path) — **must not appear.** Rejected likewise. +- `actors[].scenes` — `[start_sec, end_sec]` inclusive, sorted. **A window is a + claim about scene membership, not a recognition event** (UR-013, system spec + SR-002): an actor who turns away or is off-camera during a reverse shot is + still present. The server therefore never reinterprets, merges, splits or + trims windows — it stores and serves what it was given, quantised (§9a) but + not reshaped. Two windows mean a genuine departure and return. +- `actors[].tmdb_id` — the **primary** actor join key. In practice the + extraction pipeline populates this and leaves `imdb_id` empty (see §6 + stage 3), so a manifest without actor TMDB ids will match poorly. +- `actors[].name` — sent on upload for matching, but **not persisted**: the + server resolves each actor to a TMDB person id and serves names from its own + TMDB-derived table (§5a, §7). On download, `name` is present and + server-authoritative. Contributors should not expect a name they invented to + round-trip. + +### Pending schema bump — SR-003 + +The truth file and the Jmanifest are consumed by components that ship +independently, so breaking changes are **batched into one `schema_version` +bump** coordinated across all three repos (system spec SR-003). One bump is +currently pending, and this server must accept the new shape when it lands: + +| Change | Effect here | +|---|---| +| **Remove `anneal_sec`** | Withdrawn upstream: presence now follows track extent, so a track survives its own gaps and there is nothing to anneal. Deleted rather than kept as a vestigial `0` — a field naming a mechanism the pipeline no longer has is actively misleading | +| **Add `extinction_sec`** | Its successor: the parameter that actually shapes window extent | +| **Add `gallery_scope`** | New ranking signal (§7) | +| **Per-window belief** | `scenes` becomes a list of objects — interval plus posterior and identification route — rather than a list of float pairs | +| **Add audio signature** | Already specified here (§3, UR-009) | + +**Per-window belief does not weaken §5a.** The added fields are a bounded float +and a small enumerated string, so an accepted manifest still contains only +numbers and closed-vocabulary values. No free-form channel is opened, and +SR-004 is preserved. + +**Two consequences for `content_id`** (§9a), both of which must land with the +bump rather than after it: + +- The canonical form currently hashes `[start_cs, end_cs]` pairs. Once windows + carry belief, the canonical form must decide whether belief is part of + *identity*. It should **not** be: two servers that validated the same upload + must agree, and belief is a producer-side estimate that may legitimately + differ between pipeline versions for identical timings. Belief is replicated + as an attribute, exactly as `audio_signature` is (§9a). +- Quantisation is unchanged: integer centiseconds, for the reasons in §9a. + +Until the bump ships, this server accepts `jmanifest_version: 1` and rejects +anything else outright (UR-014) rather than guessing at an unknown shape. + +### Series bundles + +Series are the primary unit of exchange, not episodes. A user asks for +"Breaking Bad", not for 62 individual files, and per-episode round trips would +mean 62 requests against the rate limit for one obvious intent. + +A bundle is a thin wrapper, not a new format: + +```json +{ + "jmanifest_version": 1, + "series": { + "series_tmdb_id": "1396", + "series_imdb_id": "tt0903747", + "title": "Breaking Bad" + }, + "episodes": [ { "...a full Jmanifest, identity.type == episode..." } ] +} +``` + +**Sizing.** Measured against the 316 non-empty manifests in the extraction +corpus, a minimal (actors-only) manifest is ~4.8 KiB median and ~9.7 KiB at +p95. So a 24-episode season is ~112 KiB median / ~228 KiB p95, and even a +62-episode series is well under 1 MiB. Whole-series transfer is therefore the +sensible default rather than something to paginate defensively — the response +is smaller than a single poster image. + +Bundles are capped at 500 episodes and 25 MiB; beyond that the client must +page by season. + +**Partial bundles are normal.** The server returns whatever episodes it holds. +A bundle with 9 of 13 episodes is a valid, useful response, not an error. Each +episode carries its own `cut` block, so the client matches each one +independently — one mismatched episode does not invalidate the rest. The +bundle response includes coverage metadata so the client can report it: + +```json +{ + "coverage": { "episodes_available": 9, "seasons": [1, 2] } +} +``` + +**Bundles are a transfer convenience, not a storage unit.** Each episode +manifest is stored, validated, versioned, reported and delisted individually +(§7). There is no "series manifest" row — a bundle is assembled per request. +This matters for moderation: one bad episode is delisted on its own without +disturbing the other 61. + +### Series bundle upload + +Contributing a whole series is the natural counterpart, and it is the more +important half — a worker that has just processed a season should not make 24 +separate `POST`s, each triggering its own TMDB round trip. + +`POST /manifests/bundle` takes the same envelope. Semantics: + +- **Per-episode validation.** Each episode runs the full §6 pipeline + independently. The bundle is *not* atomic: valid episodes are accepted and + invalid ones rejected, with a per-episode result list. All-or-nothing would + let one bad episode discard an entire season's compute. +- **Shared TMDB fetch.** All episodes of a series resolve against one cached + credits fetch (§6 stage 3), so a 24-episode bundle costs one upstream call + rather than 24. This is the main reason bundle upload exists. +- **One rate-limit unit.** A bundle counts as a single write against the §5 + limit, with a separate per-episode cap, so contributing a season is not + punished relative to contributing a film. + +Response is `202` with per-episode outcomes: + +```json +{ + "results": [ + { "season": 1, "episode": 1, "manifest_id": "01HZ...", "status": "pending" }, + { "season": 1, "episode": 2, "status": "rejected", "reason": "cast_match_below_threshold" } + ] +} +``` + +--- + +## 3. Cut matching + +Timings only transfer between identical cuts. Matching is tiered, and the +server reports which tier matched so the client can decide whether to trust it. + +| Tier | Signal | Confidence | +|---|---|---| +| `exact` | `video_hash` equal | Same file, timings are exact | +| `runtime` | runtimes within ±2s | Very likely the same cut | +| `loose` | runtimes within ±30s | Probably same cut, different trims | +| — | beyond that | No match; do not serve | + +`video_hash` uses the OpenSubtitles hash (first+last 64KiB plus file size) — +cheap to compute, no full read, and already well-known in the media-server +ecosystem. It identifies a *file*, so it only ever matches an identical +release; it can never produce a false positive, which is why it is tier one. + +The client sends its own runtime and hash when requesting; the server does the +matching and returns the best available tier. A `loose` match should surface +as a caveat in the JRay UI rather than being applied silently. + +### Audio signature — UR-009 + +Everything above depends on knowing *what the file is*. When providence is +unknown — no TMDB id, no usable metadata, a renamed or badly-tagged file — none +of those tiers can fire. And when a release is trimmed differently (distributor +logos, PAL speed-up, an extra recap), the runtime tiers correctly *decline* to +match, but the underlying timings would have been reusable if only the offset +were known. + +A content-derived audio signature solves both. It is stored on every manifest +as `cut.audio_signature`. + +**Why audio and not video.** Audio survives what breaks video hashing: +re-encoding, resolution changes, bitrate changes, colour-space conversion, +letterboxing. Two releases of the same cut nearly always share an +audio track that is perceptually identical even when every video byte differs. + +#### Construction + +Sampled from the **centre of the media**, which avoids the two regions that +differ most between releases — logos and cold opens at the head, credits at the +tail. + +1. Decode a **120 s window centred on the midpoint** + (`runtime/2 - 60s` to `runtime/2 + 60s`). +2. Downmix to mono, resample to **11025 Hz**. +3. STFT with a **4096-sample frame, 1024-sample hop** (~93 ms/frame, + ~1290 frames), Hann window. +4. Per frame, take the log-magnitude spectrum over **300–3000 Hz** — the band + carrying dialogue and score, and the most codec-robust. +5. Divide that band into **32 logarithmically spaced bins** and record the + index of the **peak bin** plus a coarse 2-bit energy class. +6. Pack each frame into one byte; the signature is the resulting + **~1290-byte array**, base64-encoded. + +The result is ~1.7 KB per manifest — negligible against a ~4.8 KiB manifest. + +This is deliberately a **peak-bin** signature rather than a full spectrum: +peaks survive lossy re-encoding, loudness normalisation and channel-layout +differences, whereas absolute magnitudes do not. It follows the same principle +as Chromaprint/AcoustID (compact per-frame spectral features, matched by +sliding alignment) but is self-contained: no external service is queried, so +no lookup leaks which titles an instance holds (§9 privacy). + +#### Matching and offset recovery + +Two signatures are compared by sliding one against the other and taking the +best score: + +``` +for offset in -600 .. +600 frames: # ±56 s + score(offset) = fraction of overlapping frames whose peak bin matches +best = argmax score +``` + +| Result | Interpretation | +|---|---| +| `score ≥ 0.85`, `offset ≈ 0` | Same cut, aligned. Timings apply directly | +| `score ≥ 0.85`, `offset ≠ 0` | **Same cut, shifted.** Timings apply with `offset` added | +| `0.60 ≤ score < 0.85` | Possibly same cut, degraded audio. Flag as `loose` | +| `score < 0.60` | Different content. No match | + +The second row is the valuable one, and the reason to do this at all: a +release with 40 s of extra logos previously failed the ±2 s runtime tier +outright. Now it matches, and the client shifts every scene window by the +recovered offset. **The server returns the offset; the client applies it** — +manifests are never rewritten, so one stored manifest serves every trim of the +same cut. + +Offset search is capped at ±56 s, which covers realistic trim differences. +Speed-differing releases (PAL 4% speed-up) are **not** handled by a constant +offset and are correctly rejected by the score threshold; a scale-and-offset +search is possible later but is out of scope. + +#### Revised tier table + +| Tier | Signal | Confidence | +|---|---|---| +| `exact` | `video_hash` equal | Same file | +| `audio` | audio score ≥ 0.85 | Same cut; `offset` returned, may be non-zero | +| `runtime` | runtimes within ±2s | Very likely the same cut | +| `loose` | audio 0.60–0.85, or runtimes within ±30s | Caveat in UI | + +`audio` ranks above `runtime` because it is content-derived: it confirms the +audio actually matches, where equal runtimes are only circumstantial. + +#### Unknown-providence search + +With no TMDB id at all, a client can search by signature alone: + +``` +POST /manifests/search +{ "audio_signature": "base64…", "runtime_sec": 6420.5 } +``` + +The server returns candidate matches with scores, offsets and title identity — +letting JRay identify an unidentified file *and* align to it in one step. + +**This endpoint is a scaling problem, not a correctness one.** A naive +implementation compares against every stored signature. Mitigations: + +- Prefilter by runtime (±90 s) before scoring, which eliminates almost + everything. +- Index a **coarse hash** of the signature (e.g. the peak-bin sequence of + every 16th frame) for candidate generation, with full sliding comparison + only on candidates. +- Rate-limit hard (§5): this is the most expensive read endpoint and the most + attractive to abuse. + +Because it is expensive, `POST /manifests/search` is **optional for a server +to implement**; `GET /federation/capabilities` advertises support. + +#### Validation and abuse + +The signature is attacker-supplied, so §6 applies: + +- Fixed length (1290 frames ± a small tolerance for seek and encoder + differences at the window edges), base64, rejected otherwise. The length is + **not caller-varying**: items too short for the window emit no signature at + all (see "Media shorter than the window" above), so there is no legitimate + short signature to accommodate. A variable-length blob would be a payload + channel — precisely what §5a closes. +- Each byte is structurally constrained (5-bit bin index + 2-bit energy + class), so arbitrary bytes are invalid. This keeps §5a's "no free-form + storage" property intact: the field cannot carry meaningful smuggled data. +- Signatures are **never** used as a trust signal for cast validity — they + establish which cut a manifest describes, nothing more. + +#### Implementation cost — flagged honestly + +This is the most expensive addition in the spec, and it is worth being clear +where the work lands: + +- **Extraction pipeline (C++)** — **optional**, and best deferred. It already + links `libavformat`/`libavcodec`/`libavutil`, but `ffmpeg_decoder.hpp` is + **video-only**, so audio would need `libswresample` plus an FFT. Since the + plugin covers the whole library (below), this is redundant work. +- **JRay plugin (C#)** — **the primary implementation site**, and less costly + than it first appears. See below. +- **Server (Rust)** — comparison only, no audio decoding. `rustfft` plus a + sliding comparison; the cheapest of the three. + +Recommended sequencing: **make `audio_signature` optional**. Manifests without +one continue to work exactly as today via the existing tiers. Ship the +plugin-side computation first, let signatures accumulate, then enable +`audio`-tier matching and the search endpoint once coverage is useful. Nothing +above needs to land at once. + +#### Computing the signature in the JRay plugin + +The plugin is the right place for this, and it is the *only* place that covers +the whole use case. The extraction pipeline only ever sees files it processes; +the plugin sees **every item in the library**, including the ones with no truth +data and unknown providence — which is exactly the population UR-009 targets. A +signature must also be computable at *query* time (to identify a local file), +not only at contribution time. + +**Jellyfin already ships FFmpeg, and the plugin can reach it.** Verified +against `Jellyfin.Controller` 10.11.5, which the plugin already references: +`MediaBrowser.Controller.MediaEncoding.IMediaEncoder` is injectable and +exposes + +| Member | Use | +|---|---| +| `EncoderPath` | Absolute path to the server's `ffmpeg` binary | +| `ProbePath` | Path to `ffprobe` | +| `EncoderVersion` | Version gating | +| `SupportsEncoder(...)` | Capability check | + +So there is **no new dependency and nothing to bundle** — the plugin takes +`IMediaEncoder` through DI (registered in `ServiceRegistrator`) and invokes the +binary Jellyfin is already using for transcoding. + +**FFmpeg does the hard part.** Decode, downmix, resample and format conversion +are all a single invocation; the plugin never touches a codec: + +``` +ffmpeg -nostdin -v error \ + -ss -t 120 \ + -i \ + -vn -ac 1 -ar 11025 -f f32le - +``` + +That streams 120 s of mono 32-bit float PCM at 11025 Hz to stdout — +~5.3 MB, read incrementally rather than buffered whole. `-ss` **before** `-i` +makes the seek fast, which matters when sweeping a library. + +**What remains in C# is only the DSP**, and it is modest: + +1. Hann window, 4096-sample frames, 1024 hop (~1290 frames). +2. Real FFT per frame. +3. Log-magnitude, 300–3000 Hz band, 32 log-spaced bins, take peak bin + + 2-bit energy class. +4. Pack one byte per frame, base64. + +A radix-2 real FFT over 4096 samples is on the order of a hundred lines and +has no external dependency. Avoid pulling in a DSP package: this is a fixed, +well-specified transform, and vendoring a small implementation keeps the +plugin's dependency surface at zero, which matters for a GPLv3 Jellyfin +plugin. + +Cost is dominated by the FFmpeg seek and decode, not the FFT: roughly a second +or two per item, entirely I/O-bound. + +**Where it runs in the plugin:** + +- On demand, for `POST /Plugins/JRay/Items/{itemId}/Identify` (§9). +- As a **scheduled task** that backfills signatures for library items, so a + sweep is not blocked on computing them inline. Signatures are cached against + the item (keyed on item id + file mtime + size, so a replaced file + recomputes). +- Before contributing a manifest, so uploads carry `cut.audio_signature`. + +**Degradation, not failure.** If `IMediaEncoder` is unavailable, the binary is +missing, the item has no audio stream, or the file is shorter than the window, +the plugin logs and proceeds **without** a signature. Every existing tier keeps +working; UR-009 is an enhancement and must never be able to break a fetch. + +#### Media shorter than the window — 120 s + +**Items under 120 s emit no signature at all, and no sync offset is applied to +them.** The window is `runtime/2 ± 60 s`, so below 120 s it underflows: there is +no shortened window to compute, because the construction has no definition +there. Such items fall back to the `exact` and `runtime` tiers, which is +adequate — a sub-two-minute item is rarely the ambiguous-providence case UR-009 +exists to solve. + +The signature is therefore **fixed-length by construction**, not merely bounded. +That is what keeps it inside SR-004: a caller cannot choose the length, so the +field cannot be used as a variable-size container (§5a, and "Validation and +abuse" below). + +> **Reconciled with `scene-actor-extraction` IR-007.** An earlier draft of this +> section said items under *150 s* got a centred, shortened window with the +> frame count recorded. That described a mechanism neither producer implements, +> and it was the weaker rule: a caller-varying length is exactly the property +> SR-004 forbids. The 120 s cutoff is now identical in both producers and in +> this server's validator, which is what IR-007 requires — a rule that differs +> between producers yields signatures that never match. + +**The extraction pipeline (C++) is then optional for UR-009.** It may compute +signatures for files it processes — `libswresample` plus an FFT, as noted +above — but since the plugin computes them for the whole library and attaches +them on contribution, the pipeline need not implement this at all. That +removes the `libswresample` work from the critical path. + +--- + +## 4. API + +Base path `/api/v1`. JSON throughout. + +### `GET /manifests/exists` — UR-001 + +Cheap existence probe. Answers whether a manifest is available for a given +title *and* at what cut-match tier, without transferring the payload. + +Query parameters are the same identity + cut parameters as the fetch +endpoints: `tmdb_id` / `imdb_id` (or `series_tmdb_id` + `season` + `episode`), +plus optional `runtime_sec` and `video_hash`. + +```json +{ "exists": true, "match": "runtime", "manifest_id": "01HZ...", "actor_count": 34 } +``` + +`exists: false` is returned with `200`, not `404` — absence is a normal answer +to this question, and using `404` would conflate "no manifest" with "bad +route" for the client. + +If `runtime_sec` and `video_hash` are both omitted, the response reports +whether *any* manifest exists for the title with `"match": "unknown"`; the +client must still fetch to find out whether a cut actually aligns. This is +the mode a library-wide sweep uses. + +#### Batch form + +A client sweeping a library should not issue one request per item. The batch +form takes up to 100 items: + +``` +POST /manifests/exists +{ "items": [ { "tmdb_id": "504172", "runtime_sec": 6420.5 }, ... ] } +``` + +returning results positionally. This exists specifically so the rate limit in +§5 can be generous per *request* while staying strict per *item*, and so a +2000-item library sweep is 20 requests rather than 2000. It is a `POST` only +because the payload does not fit a query string; it is a read and requires no +token. + +### `GET /manifests/movie?tmdb_id=&imdb_id=&runtime_sec=&video_hash=` + +Returns the best-matching Jmanifest, or `404` if none clears `loose`. + +```json +{ "match": "runtime", "manifest": { "...": "..." } } +``` + +### `GET /manifests/series/{series_tmdb_id}?season=` + +Returns a series bundle (§2). `season` optional; omitted means all seasons. +Episode-level cut matching is done client-side against the returned bundle, +since a client pulling a whole series already knows its own runtimes. + +### `GET /manifests/episode?series_tmdb_id=&season=&episode=&runtime_sec=&video_hash=` + +Single-episode equivalent of the movie endpoint. + +### `POST /manifests` + +Contribute a manifest. Body is a Jmanifest. Requires an API token (§5). + +- `202 Accepted` — passed size and schema validation; held unlisted pending + the TMDB cast check (§6). Returns `{ "manifest_id": "...", "status": "pending" }` +- `400` — malformed, or contains an unrecognised or forbidden field (§6) +- `409` — an identical `(identity, cut)` manifest already exists from this + contributor +- `413` — body exceeds the size limits (§6) +- `429` — rate limited (§5) + +### `POST /manifests/bundle` — UR-006 + +Contribute a whole series in one request. Body is a series bundle (§2). +Per-episode validation, non-atomic, one rate-limit unit, shared TMDB fetch — +see "Series bundle upload" in §2. + +- `202 Accepted` — returns per-episode outcomes +- `400` — the bundle envelope itself is malformed (individual bad episodes are + reported in the results list, not as a whole-request error) +- `413` — exceeds 500 episodes or 25 MiB + +### `GET /manifests/{id}/status` + +Poll the outcome of the asynchronous cast check for an upload: +`{ "status": "pending" | "listed" | "flagged" | "rejected", "reason": "..." }`. + +### `GET /manifests/{id}` + +Fetch a specific manifest by its server-assigned id (for debugging and for +the "report this manifest" flow). + +### `POST /manifests/{id}/report` + +Flag a manifest as wrong (misaligned, wrong actors). Body: +`{ "reason": "misaligned" | "wrong_actors" | "spam", "note": "..." }`. + +### `GET /health` + +Liveness. Unauthenticated. + +--- + +## 5. Authentication and abuse + +Reads are anonymous and cacheable. Writes require a token — an anonymous +bearer capability, not an account. No email, no verification, no personal +data; see §5a for why identity is deliberately not load-bearing. + +### Rate limiting — UR-004 + +Limits are per token where one is present, otherwise per source IP. Anonymous +reads are keyed on IP, which is imperfect behind CGNAT; the limits below are +therefore set well above what a single real server needs. + +| Surface | Limit | Rationale | +|---|---|---| +| `GET /manifests/exists` | 600 / hour | Sweeps should use the batch form | +| `POST /manifests/exists` (batch) | 60 / hour, ≤100 items each | 6000 items/hour — a large library sweeps in one pass | +| Manifest fetches (`/movie`, `/episode`) | 300 / hour | A client only fetches what `exists` said was there | +| `GET /manifests/series/{id}` | 120 / hour | Bundles are ~100–250 KiB; this is the preferred path for TV and should not be scarcer than per-episode fetching | +| `POST /manifests` | 100 / hour per token | Nobody uploads faster than the CV pipeline runs | +| `POST /manifests/bundle` | 20 / hour per token, ≤500 episodes each | One unit per bundle, so contributing a season is not penalised versus a film | +| `POST /manifests/{id}/report` | 20 / hour per IP | Reports are a moderation lever; cheap to abuse | +| `POST /manifests/search` (audio) | 60 / hour | Most expensive read endpoint (§3); sliding comparison over candidates | +| `GET /federation/peers` | 60 / hour per IP | Public directory, read by humans; no reason for volume | +| `GET /federation/changes` | 120 / hour per peer | Hourly polling is the default; this allows generous catch-up | +| `GET /federation/manifests/{content_id}` | 5000 / hour per peer | Bootstrap pulls are bulk by nature; capped so one peer cannot saturate egress | +| `POST /federation/have` | 120 / hour per peer, ≤1000 ids each | Diffing a catalogue should be a handful of requests | +| `GET /health` | unlimited | Liveness | + +Responses carry `X-RateLimit-Limit`, `X-RateLimit-Remaining` and +`X-RateLimit-Reset`; exceeding a limit returns `429` with `Retry-After`. The +JRay client must honour `Retry-After` and back off exponentially rather than +retrying tightly — a scheduled library sweep that ignores this will get an +instance's IP throttled. + +Implemented as a fixed-window counter keyed on `(token_or_ip, surface)`, +held in process memory (§8) — no external counter store. A sliding window is +not worth the complexity at this volume. Counters reset on restart, which is +acceptable for abuse throttling. Read limits are applied *behind* the CDN +cache, so a cache hit costs a client nothing against its budget. + +--- + +## 5a. Trust model + +**Design goal: no accounts, no identity, no moderation queue that scales with +users — and no way to use the server as a content host.** + +The key property that makes this tractable: a Jmanifest is not free-form +content. It is a *closed-vocabulary* document — a title identity, a runtime, +and a list of actors with timings. Everything in it is checkable against an +external ground truth (TMDB) that the attacker does not control. So trust can +attach to **content**, not to **contributors**. This is why the server needs +no accounts: a manifest listing pornstars for a children's film fails the +check regardless of who uploaded it, and a valid manifest is valid regardless +of who uploaded it. + +### Threat 1 — using the server as a content store + +The concern is the server being used to host illegal material (the worst case +being CSAM) or arbitrary payloads, making the operator liable. + +The structural defence is that **there is nowhere to put it**. After §6 +stage 2, an accepted document contains only: + +| Field | Constraint | +|---|---| +| `identity.tmdb_id` / `imdb_id` | Regex-constrained to digits / `tt\d{7,8}` | +| `identity.season`/`episode`/`year` | Bounded integers | +| `cut.*` | Numbers, plus a fixed-format hash | +| `extraction.*` | Numbers and a version string from an allow-list | +| `actors[].tmdb_id` / `imdb_id` | Regex-constrained | +| `actors[].scenes` | Pairs of floats | +| `actors[].name`, `identity.title` | **The only free-form strings** | + +No binary. No images. No URLs. No base64 fields. No extension points — because +`extra="forbid"` applies at every nesting level, an attacker cannot add one. + +That reduces the entire content-hosting surface to two short text fields, which +are then constrained further: + +- **Length caps.** `name` ≤ 200 chars, `title` ≤ 300. With ≤ 500 actors that is + a hard ceiling of ~100 KB of attacker-controlled text per manifest, but see + the next two rules, which cut it far below that. +- **Character class.** Names must match a permissive-but-closed pattern: + Unicode letters, marks, spaces, and `. ' - ,` only. No digits, no `/ + =`, + no control characters, no zero-width or bidi-control codepoints, NFC + normalised. **This alone defeats base64/hex smuggling**, which needs digits + and padding characters. +- **Cross-check against a known vocabulary.** Every actor name must correspond + to a real TMDB person (§6 stage 3). A name that matches no TMDB person is + not stored at all. An attacker therefore cannot write arbitrary strings — + only strings that already exist in TMDB's person index. + +Combined, the last rule is decisive: **the server does not store +attacker-authored text, it stores references to TMDB entities.** The strongest +form of this — and what I recommend for v1 — is to go one step further and +**not persist the submitted name string at all**: + +> Store `tmdb_person_id` plus the timings. Resolve display names from the +> server's own TMDB-derived person table at serve time. The uploaded `name` +> field is used only for matching during validation, then discarded. + +At that point the free-text channel is closed completely. The only +attacker-controlled values that reach the database are integers. There is no +CSAM risk and no payload-smuggling risk because there is no field capable of +carrying either. + +This also resolves your point about not storing JSON files: with names +normalised to person ids, the natural representation is relational rather than +a blob. See §7. + +### Threat 2 — prank and vandalism manifests + +Semantically valid but wrong: casting pornstars in a children's film, or +mislabelling a film's cast as a joke. Every structural check passes; only +ground truth catches it. + +Defence is the TMDB cast cross-check in §6 stage 3. Its effectiveness rests on +the attacker not controlling TMDB: to make a pornstar manifest pass, they would +need those performers to be *credited cast on that title in TMDB*, which means +vandalising TMDB itself — a separate, moderated system with its own edit +history. That is a meaningfully high bar for a prank. + +Additional layers, in order of cost: + +1. **Category guard.** Reject any manifest where a matched TMDB person's + known-for department or credits are dominated by titles TMDB flags as + adult (`adult: true`), unless the target title is itself flagged adult. + This directly targets the stated prank without needing a blocklist of + names. +2. **Age-appropriateness guard.** If the target title's TMDB certification is + a children's rating, apply the strictest cast-match threshold and require + an `exact` or `runtime` cut match. Mismatched content on children's titles + is the highest-harm case and deserves the tightest gate. +3. **Divergence detection.** When two manifests exist for the same + `(title, cut)` from different sources and their actor sets disagree beyond + a threshold, flag both and serve the one with the better cast-match ratio. + Honest extractions of the same cut converge; a prank diverges from them. + +### What replaces accounts + +Contribution requires a token, but a token is **not an account** — it is an +anonymous bearer capability: + +- Self-issued on request, no email, no verification, no personal data. +- Stored only as a hash. The server cannot enumerate who holds tokens. +- Its sole purposes are rate-limiting attribution (§5) and revocation. +- Discarding a token and requesting another is trivially easy — and that is + *fine*, because the token is not the defence. The content checks are. A new + token gains an attacker nothing, since every upload faces the same + ground-truth validation. + +This is the crucial difference from an account system: the token exists to +throttle volume, not to establish identity. Sybil resistance is not required +because identity is not load-bearing. + +Consequently the only reputational state is per-token counters +(`accepted`, `rejected`, `flagged`), used for one purpose: a token whose +rejection rate exceeds a threshold over a minimum sample is revoked +automatically, and its `pending`/`flagged` manifests are dropped. No human is +in the loop for the common case. + +### Residual risk and the operator's lever + +Two things remain that automation cannot fully close: + +1. A manifest that is *plausible but wrong* — correct cast, deliberately + misaligned timings — degrades the overlay but carries no legal or safety + risk. Reports plus divergence detection handle it. +2. A novel abuse pattern nobody anticipated. + +For both, the operator needs a **kill switch**, not a moderation queue: +`status` transitions (§7) are a single column, so delisting a manifest, every +manifest from a token, or every manifest for a title is one UPDATE. Delisting +is instant and reversible; deletion is a separate, logged action. + +**Legal posture.** Because the server stores only integers and references to +TMDB entities, it holds no user-generated content in the sense that +intermediary-liability regimes contemplate. This should be stated plainly in +the operator documentation, alongside a contact address for takedown requests. +It is a materially better position than "we store user-submitted JSON and +moderate it". + +### Client-side hardening + +Independent of the server, because a compromised or hostile server must not be +able to attack its clients: + +- The JRay overlay renders actor names as **text nodes only**, never as HTML. +- The plugin validates downloaded manifests against the same schema it would + apply to an upload — a client must not trust a manifest merely because the + server served it. +- Downloaded manifests are stored via the existing `IManagedTruthStore` and + never written into the media library filesystem. + +--- + +## 6. Upload validation — UR-003 + +Validation runs in four stages, ordered cheapest-first so that abusive +uploads are rejected before they cost anything. + +### Stage 0 — reject on headers, before the body is read + +The cheapest rejection is the one that happens before any payload is accepted. + +> **In one line:** `Content-Length` is the fast path; the streaming byte +> counter is the enforcement. The header is a *claim by the client*, so it +> rejects honest oversized uploads early and cheaply, but it cannot be the +> only check — a lying header, a chunked upload, or a compressed body all pass +> it. Implement both; in Axum they are the same one-line layer (§8). + +Three mechanisms, in order of how early they fire: + +**1. `Expect: 100-continue` (earliest — body genuinely never sent).** +A client may send headers with `Expect: 100-continue` and wait before +transmitting the body. The server responds `100 Continue` or, if +`Content-Length` already exceeds the cap, `413` — and the body is never +transmitted at all. This is the true "reject before upload". + +Clients under this project's control — the JRay plugin contributing manifests +(§9) and federation peers pulling (§9a) — **should** use `Expect: 100-continue` +for uploads, because it turns a rejected 25 MiB bundle into a two-header +exchange. The server must handle it correctly, but must never *depend* on it: +arbitrary clients will not send it. + +**2. `Content-Length` check (normal case).** +When present, the declared length is available in the request headers before +the body. If it exceeds the cap for that route, respond `413` immediately and +do not read the body. + +This is a filter, not a guarantee: a hostile client can declare +`Content-Length: 100` and then send gigabytes. **The streaming cap below is +therefore mandatory, not redundant.** + +**3. Chunked requests have no declared size.** +HTTP/1.1 `Transfer-Encoding: chunked` omits `Content-Length` entirely, so +there is nothing to check up front. These must be capped while streaming. + +> **What "before anything is uploaded" can and cannot mean.** Even on an +> immediate `413`, a client has typically already put some body bytes on the +> wire — they may sit in kernel or proxy buffers before the response lands. +> The achievable guarantee is that the server never *reads, buffers, or +> parses* an oversized body, and closes the connection promptly. It is not +> that zero bytes cross the network. Do not size defences on the assumption +> that a `413` prevents transmission. + +### Stage 1 — size limits while streaming (before parsing) + +Enforced at the reverse proxy and again in the app, on the raw body, *before* +JSON parsing. A parser handed an unbounded body is a denial-of-service +primitive, so this must not be deferred to the schema layer. + +The app-level cap counts bytes as they are read and **aborts mid-transfer** +once exceeded, rather than reading to completion and then measuring. This is +what makes a lying `Content-Length` and a chunked upload both safe. + +| Limit | Value | +|---|---| +| Request body (movie or episode manifest) | 2 MiB | +| Request body (series bundle, §2) | 25 MiB | +| Request body after gzip decompression | 8 MiB, with a max compression ratio of 20:1 | +| JSON nesting depth | 12 | +| `actors[]` entries | 500 | +| `scenes[]` entries per actor | 2000 | +| Total scene windows across all actors | 20000 | + +For scale: the sample feature film in the extraction repo has ~30 actors and a +few hundred windows. These caps are roughly an order of magnitude above +anything legitimate. Oversized bodies are rejected with `413`. + +The decompression-ratio cap matters because a gzip bomb passes a 2 MiB +body-size check trivially. Decompression must also be **streamed with a +running output cap** — decompressing fully and then checking the size defeats +the point. + +**Implementation.** Axum's `DefaultBodyLimit` (§8) implements the streaming +cap and honours `Content-Length` for early rejection, applied per-route so the +bundle endpoint gets its larger limit without widening the others. Set the +matching `client_max_body_size` (nginx) / `request_body max_size` (Caddy) at +the proxy so oversized uploads are dropped at the edge and never occupy an +application worker. + +### Stage 2 — strict schema (synchronous, rejects with `400`) + +**No additional fields anywhere.** Every object in the document is validated +in strict mode — serde `#[serde(deny_unknown_fields)]` on every DTO (§8) — so +an unrecognised key at any nesting level is an error, not something silently +ignored. This is the default posture, not a special case for the two fields +below, and it is enforced by the type definitions rather than by validator +code that could omit a field. + +Rejected outright: + +- **any unrecognised field**, at any level of the document +- `movie` present, or any string anywhere that looks like an absolute + filesystem path (`/…`, `C:\…`, `\\…`) or a `file://` URI +- `actors[].jellyfin_id` present and non-empty +- missing `cut.runtime_sec` +- neither `tmdb_id` nor `imdb_id` in `identity` +- `imdb_id` not matching `^tt\d{7,8}$`, actor `imdb_id` not matching + `^nm\d{7,8}$`, `tmdb_id` not matching `^\d{1,9}$` +- scene windows with `end < start`, negative times, non-finite values + (`NaN`/`Infinity`), or times beyond `runtime_sec` + 5s tolerance +- actor names longer than 200 characters, containing control characters, or + failing Unicode normalisation to NFC +- duplicate actors within one manifest (same `imdb_id`) + +Rejecting unknown fields is what makes the `movie` / `jellyfin_id` strip in +§9 verifiable: a client that forgets to strip them gets a hard `400` naming +the offending field, rather than quietly publishing a contributor's directory +layout. + +### Stage 3 — TMDB cast cross-check (asynchronous, after `202`) + +This needs an outbound TMDB call and so cannot run inside the request without +coupling upload latency to a third party. The upload is accepted with `202` +and the manifest is held **unlisted** until the check completes; it is not +served to anyone in the meantime. + +The check: fetch the TMDB credits for `identity.tmdb_id`, take the set of +credited cast **TMDB person ids**, and compare against the actors in the +manifest. + +> **Join on `tmdb_id`, not `imdb_id`.** This is grounded in the actual +> pipeline output, not assumed. Of the 331 manifests in the +> scene-actor-extraction repo, exactly **one** has IMDB ids populated; the +> other 330 have `imdb_id: ""` with `tmdb_id` set. The Jellyfin-gallery path +> (`make_jellyfin_gallery.py`) — which is the path most users will take, since +> it needs no TMDB key — yields TMDB ids only: 2391 of 2392 gallery entries +> have a `tmdb_id`, and **none** have an `imdb_id`. +> +> A design keyed on IMDB ids would therefore fall back to name matching for +> essentially every real upload, which is exactly the weak path. `tmdb_id` is +> the join key; `imdb_id` is an optional secondary signal when present. + +Let *M* = actors in the manifest, *C* = credited cast from TMDB. + +| Condition | Outcome | +|---|---| +| `\|M ∩ C\| / \|M\|` ≥ 0.6 | **Listed.** Normal case | +| 0.3 ≤ ratio < 0.6 | **Listed, flagged** for review; served with reduced ranking | +| ratio < 0.3 | **Rejected.** Manifest is deleted and the contributor notified | +| TMDB has no credits for the id | **Listed, flagged** — absent data is not evidence of a bad manifest | +| TMDB unreachable / rate-limited | **Retry** with backoff; stays unlisted, not rejected | + +The match is deliberately *approximate* and directional. It asks "are these +plausibly this film's cast?", not "is this cast list complete": + +- Ratio is over *M*, not *C*. A manifest legitimately contains only actors who + were both credited and detected on screen, so it is normally a strict subset + of the cast — penalising it for missing credited actors would fail every + honest upload. The corpus bears this out: median 7 actors per manifest, + against feature casts several times larger. +- Uncredited appearances, cameos, and actors TMDB lists only under a + differently-spelled name are exactly why the threshold is 0.6 and not 1.0. +- Matching is on `tmdb_id` (see above), with `imdb_id` as a secondary signal + when present, falling back to case- and accent-insensitive name comparison. + Name-only matches are counted but capped at half the intersection, so a + manifest cannot pass on name collisions alone. + +**Small-*M* handling.** With a median of 7 actors, a ratio threshold is coarse +— one mismatch moves it by 14%. So: + +- `|M|` ≥ 5: apply the ratio table above. +- `2 ≤ |M| < 5`: require *all but one* actor to match. A ratio is meaningless + at this size. +- `|M| ≤ 1`: accept only if the single actor matches; such a manifest is + near-worthless anyway and is ranked last. +- `|M| == 0`: **reject.** 15 of the 331 corpus files have empty actor lists — + these are extraction failures, not contributions, and must not be uploaded. + The client should refuse to submit them. + +**Every matched actor is resolved to a TMDB person id, and unmatched actors +are dropped rather than stored.** This is what closes the free-text channel +described in §5a: a manifest is persisted as a set of TMDB person references, +so a name that corresponds to no TMDB person never reaches the database. + +For episodes the check runs against the **union** of TMDB's per-episode +credits (cast + guest stars) and the series' aggregate credits. + +Using the union rather than either alone matches what the extraction client +already does: `run_from_jellyfin.py` defaults to `--episode-cast tmdb`, taking +TMDB per-episode credits and falling back to series-wide when the episode has +no usable credits. Checking against per-episode credits alone would reject +recurring cast that TMDB lists only at series level; checking against +series-wide alone would reject legitimate guest stars. The union admits both, +and since the ratio is over the manifest's actors (not TMDB's cast), widening +the reference set costs nothing in strictness against pranks — a pornstar is +in neither set. + +TMDB responses are cached (24h) so that a burst of episode uploads for one +series costs a single upstream call, and so the server stays within TMDB's own +rate limits. + +### Sanity-checked and warned, not rejected + +- total on-screen coverage implausibly high (>95% of runtime) or near zero +- an actor whose windows sum to under a second +- `sample_fps` below 1, which yields low-quality timings + +Because the contributed manifest is stripped of `jellyfin_id`, the downloading +server resolves actors locally via `tmdb_id` (primarily) against its own People +`ProviderIds` — which is exactly the fallback path JRay already implements. + +--- + +## 7. Storage + +SQLite in WAL mode (§8), **fully relational — no JSON blobs on the write +path.** The schema below is portable SQL and runs unchanged on Postgres should +an instance ever outgrow SQLite. + +Storing the uploaded document as a JSON payload would undermine §5a: a blob +is an opaque container, so whatever the schema validator missed gets persisted +verbatim and served back out. Decomposing into columns means **the database +can only represent what the schema models** — there is physically nowhere for +an unexpected field or a smuggled string to live. Normalisation is a security +control here, not just tidiness. + +Concretely, the submitted JSON is parsed, validated, resolved to TMDB person +ids, written as rows, and **discarded**. The document served to clients is +*reconstructed* from those rows, never echoed. + +```sql +contributors (id, token_hash, created_at, revoked_at, + accepted_count, rejected_count, flagged_count) + +people (tmdb_person_id PK, -- server-side, TMDB-derived + name, -- from TMDB, never from an upload + adult bool, updated_at) + +titles (id PK, kind, -- movie | series + tmdb_id, imdb_id, name, year, + adult bool, certification, updated_at) + +manifests (id PK, title_id FK, season, episode, + runtime_sec, video_hash, + audio_signature blob NULL, -- §3, ~1290 bytes + audio_sig_coarse blob NULL, -- candidate-generation index key + sample_fps, extinction_sec, pipeline_version, + gallery_size, gallery_scope, -- ranking signal, §2 + contributor_id FK, + status, -- pending | listed | flagged | rejected + cast_match_ratio real, created_at) + +manifest_actors (manifest_id FK, tmdb_person_id FK, + PRIMARY KEY (manifest_id, tmdb_person_id)) + +scenes (manifest_id FK, tmdb_person_id FK, + start_cs integer, end_cs integer) -- centiseconds, see §9a + +reports (id, manifest_id FK, reason, note, created_at, source_ip_hash) +tmdb_cache (tmdb_id, kind, credits json, fetched_at) + +jobs (id, kind, -- cast_check | federation_pull + payload, run_after, attempts, last_error) +``` + +`jobs` is the background queue (§8) — a table rather than an external broker, +so pending work survives a restart. + +Note what is **not** in this schema: no actor-name column on any upload-derived +table. Display names come from `people.name`, populated from TMDB by the +server. `tmdb_cache` is the sole JSON column and holds *TMDB's* responses, +not users'. + +Scene times are stored as **integer centiseconds** (§9a), not floats — the +same quantisation used for `content_id`, so stored values and hashed values +cannot diverge. + +Indexes on `titles(tmdb_id)`, `manifests(title_id, runtime_sec)`, +`manifests(video_hash)`, `manifests(title_id, season, episode)`, and +`scenes(manifest_id, tmdb_person_id)`. All read queries filter +`status IN ('listed','flagged')`, so a partial index on that predicate keeps +the hot path small. + +`scenes` is the only table with real row volume — roughly (actors × windows) +per manifest, capped by §6 at 20000 rows. At corpus-realistic sizes (median 7 +actors) it is a few hundred rows per manifest, so even tens of thousands of +manifests stay comfortably small. + +Multiple manifests may coexist for the same title with different cuts — that +is the point. Multiple manifests for the *same* cut from different +contributors are allowed too; serve the one with the best +`(cast_match_ratio, reports, gallery_scope, sample_fps)` ranking. + +`gallery_scope` enters the ranking because it is the strongest available +quality signal between two otherwise comparable manifests (§2): a manifest +extracted against a `global` gallery had to distinguish its actors from every +other actor in the contributor's library, whereas a `limited` one only had to +distinguish them from that title's own cast. The former surviving the cast +check is stronger evidence than the latter doing so. It ranks below +`cast_match_ratio` and reports, which are evidence about *this* manifest rather +than about the conditions that produced it. + +--- + +## 8. Stack + +Traffic is low and read-dominated; a reverse-proxy or CDN cache in front keeps +the app tier trivial. + +Two capabilities beyond request handling and storage are load-bearing rather +than optional: + +- **Rate-limit counters** (§5). +- **Asynchronous background work** — the TMDB cast check (§6 stage 3) and its + retries, plus federation pulls (§9a). + +Both are satisfied in-process by the recommended stack below; neither requires +a separate service. + +The server needs a **TMDB API key** as operational configuration. It is a hard +dependency for UR-003: if TMDB is unreachable, uploads accumulate in `pending` +rather than being listed unverified. + +### Recommended production stack + +**Rust + Axum + SQLite**, behind an operator-provided reverse proxy. + +| Layer | Choice | Version at time of writing | +|---|---|---| +| Framework | **Axum** + Tower/`tower-http` | axum 0.8 | +| Runtime | **Tokio** | 1.x | +| Database | **SQLite** (WAL mode) | 3.4x | +| DB access | **sqlx** (compile-time checked SQL) or **rusqlite** | sqlx 0.9 / rusqlite 0.40 | +| Serialization | **serde** / **serde_json** | 1.x | +| HTTP client | **reqwest** (TMDB, federation pulls) | 0.12 | +| Observability | **tracing** + OpenTelemetry exporter | — | +| Edge / TLS | Operator-provided (Caddy, nginx, Traefik) | — | +| Packaging | Single static binary + one DB file | — | + +**Why this fits.** The workload is read-dominated, low-volume, and +cache-frontable; the largest response is a ~228 KiB series bundle. Raw +throughput is not the constraint — Postgres/SQLite queries and TMDB calls are. +What *does* matter here is operational simplicity for hobbyist operators +(§9a expects independent people to run instances) and strictness at the +validation boundary (§6, §5a). Rust serves both: a single static binary plus +one file is the lowest-friction thing an operator can deploy, and a strict +type system at the parse boundary is exactly the posture §5a asks for. + +**Axum over Actix Web.** Actix leads on raw throughput by ~10–15% under heavy +load, which is irrelevant at this volume. Axum's Tower middleware composition +maps directly onto what the spec needs — rate limiting (§5), body-size limits +(§6 stage 1), tracing, timeouts — as composable layers rather than bespoke +code. It is the mainstream default for new services and the easier codebase +for occasional contributors. + +**Serde `deny_unknown_fields` is the §6 stage 2 enforcement mechanism.** This +is the strongest argument for Rust here. `#[serde(deny_unknown_fields)]` on +every DTO gives the "no additional fields anywhere" requirement structurally, +checked at compile time against the type definitions, with no possibility of +a field being silently accepted because a validator forgot it. Combined with +newtypes for `TmdbId`, `ContentId`, and centisecond timestamps, malformed +input fails to parse rather than being caught later — invalid states become +unrepresentable rather than merely rejected. + +**Axum's `DefaultBodyLimit`** enforces the §6 stage 0/1 caps in the framework: +it rejects on `Content-Length` before reading a body, and caps the stream for +chunked or mis-declared uploads — satisfying "reject before parsing" without +trusting the client's declared size. + +### SQLite: the write-concurrency question + +SQLite is the right call, but it has one hard constraint that must be designed +around rather than discovered: **only one writer at a time, even in WAL mode.** +Concurrent write transactions return `SQLITE_BUSY`. + +Reads are unaffected — WAL gives concurrent readers alongside the single +writer, which suits a read-dominated workload well. The risk is concentrated +in this spec's three bulk-write paths: + +| Path | Write shape | Risk | +|---|---|---| +| Single manifest upload | ~106 scene rows median | Negligible | +| Series bundle upload (§2) | 24 episodes × ~106 rows ≈ 2.5k rows | Moderate — one long transaction | +| Federation bulk ingest (§9a) | Thousands of manifests | **This is the real one** | + +Federation bootstrap is explicitly a bulk-write workload, and it runs +concurrently with live uploads. Mitigations, which are requirements rather +than suggestions: + +- **WAL mode**, plus `busy_timeout` (5s) so contention waits rather than + errors, and `synchronous = NORMAL` (safe under WAL). +- **A single writer connection**, serialized through one task/actor, with a + read pool alongside. Do not point a multi-connection pool at writes and rely + on `busy_timeout` to sort it out — serialize deliberately. +- **Chunked ingest transactions.** Federation ingest commits per manifest, not + per batch, so a bootstrap never holds the write lock for long. Combined with + the §9a `MaxIngestPerHour` cap, live uploads are not starved. +- **Batch inserts within a transaction** for a manifest's scene rows — one + transaction per manifest, not per row. + +With those, a single modest VPS handles this comfortably. SQLite does tens of +thousands of writes/sec on modern hardware; the constraint is lock *duration*, +not throughput. + +**When to reconsider.** If an instance ever runs multiple writer processes, or +federation bootstrap contention becomes visible in practice, Postgres is the +escape hatch. Keep the SQL portable and use sqlx (which supports both) so the +migration is a configuration change rather than a rewrite. **Do not** design +around a hypothetical Postgres future at the cost of SQLite's simplicity now. + +### Alternatives considered + +The single-writer constraint above is the one real weakness, so it is worth +being explicit about why SQLite still wins. + +| Option | Verdict | +|---|---| +| **SQLite** (rusqlite / sqlx) | **Chosen.** Ubiquitous, unmatched track record, trivially portable, one file | +| **Turso** (SQLite rewritten in Rust, MVCC) | Strong future candidate; pre-1.0 today | +| **libSQL** (C fork of SQLite) | Viable, but its own maintainers now direct effort at Turso | +| **Postgres** | The escape hatch, not the default — a service to operate, against §9a's goal | +| **redb / fjall / sled** | Wrong data model — see below | +| **DuckDB** | Analytical (OLAP); this is a transactional point-lookup workload | +| **SurrealDB** | Far larger surface area than needed; not an embedded-first story | + +**Key-value stores are the wrong shape, not merely a different one.** redb is +mature (v4.1, actively developed) and gives MVCC with concurrent readers plus a +single writer — but it is a key-value B-tree with no SQL, no secondary indexes +and no joins. §7 is a genuinely relational schema: foreign keys between +`manifests`, `scenes`, `manifest_actors` and `people`, partial indexes on +`status`, and queries that join and filter across them. On a KV store all of +that becomes hand-maintained index keys and application-side joins — more code +in exactly the layer where §5a demands correctness. `sled` is additionally out +on maintenance grounds (last release October 2024). + +**Turso deserves a serious look, just not yet.** It is a clean-room Rust +rewrite of SQLite whose `BEGIN CONCURRENT` / MVCC mode (`PRAGMA journal_mode = +'mvcc'`) directly removes the single-writer limitation described above — the +precise weakness in this design. It is SQLite-compatible, so the schema and +most queries carry over, and it is developed with deterministic simulation +testing. But as of this writing it is **pre-1.0**; the maintainers state it +powers production systems while being explicit that they have not yet reached +their "SQLite-level reliability" bar. It also shifts work onto the +application: MVCC transactions that touch overlapping data return a conflict +error and must be retried, so the caller owns retry logic. + +For a small, low-write, read-dominated service where the mitigations above +already keep lock duration short, adopting a pre-1.0 database to solve a +problem this workload does not yet have is the wrong trade. The recommendation +is therefore: + +> Build on SQLite, keep the SQL standard and the data-access layer behind a +> thin trait. Re-evaluate Turso when it reaches 1.0 or if federation ingest +> contention shows up in real operation. Because Turso is SQLite-compatible, +> that migration is far cheaper than the Postgres one — which is itself an +> argument for not over-engineering now. + +**A note on portability.** "Portable" here means two distinct things, and +SQLite is best at both: the *file* is portable (a single database file an +operator can copy, back up, or hand to someone bootstrapping a mirror), and +the *SQL* is portable (standard enough to move to Postgres or Turso later). +Any KV store sacrifices the second entirely. + +### What Rust changes elsewhere in the spec + +- **No Redis.** Rate-limit counters (§5) live in process memory (`governor` or + a Tower layer) or in SQLite. A single-process server does not need an + external counter store, and dropping Redis removes a whole moving part. + Note the tradeoff: in-memory counters reset on restart, which is acceptable + for abuse throttling and avoids a dependency for a hobbyist deployment. +- **No separate worker process or broker.** The async TMDB cast check (§6 + stage 3) and federation pulls (§9a) run as Tokio background tasks in the + same binary, with the job queue as a SQLite table so state survives restart. + This replaces arq/Dramatiq/Celery entirely. +- **The TMDB cache** (§7 `tmdb_cache`) stays a table; SQLite's JSON functions + cover the `jsonb` usage, which is only caching TMDB responses. + +Net effect: **one binary, one database file, one reverse proxy.** That is a +materially better deployment story for federation than "app + worker + +Postgres + Redis", and federation only works if running an instance is easy. + +### Cost of choosing Rust + +Stated honestly, since the alternative was Python: + +- The extraction side is Python, so validation and canonicalisation logic + (notably the §9a `content_id` canonical form) can no longer be shared as + one implementation. It must be specified precisely enough to reimplement, + and cross-tested — a golden-vector test fixture shared by both sides. +- Fewer casual contributors than a FastAPI codebase would attract. +- Slower initial development. + +These are real, and worth accepting for a long-lived service whose main risks +are hostile input and operator friction — both of which Rust directly +addresses. + +### Deployment notes + +- Ship a **single static binary** (musl target) plus the SQLite file. Optional + container image, but neither Docker nor Compose should be required. +- Put all database access behind a **thin repository trait** rather than + scattering queries through handlers. This is what keeps the Turso/Postgres + options above cheap, and it localises the single-writer serialization + described earlier in one place instead of every call site. +- Avoid SQLite-specific SQL where a standard form exists (notably `INSERT … + ON CONFLICT`, which is portable, versus `INSERT OR REPLACE`, which is not). +- Terminate TLS at the operator's proxy; the app speaks plain HTTP on + loopback and must trust `X-Forwarded-For` **only** from that proxy — §5 rate + limiting and report attribution key on client IP, so a spoofable header + defeats both. Make the trusted-proxy CIDR explicit configuration, not a + default-on behaviour. +- Enforce the §6 stage 1 body cap at *both* the proxy and + `DefaultBodyLimit`; defence in depth, and the app must be safe when run + without a proxy. +- Set a statement timeout and request timeout (`tower_http::timeout`) so a + slow bundle query fails fast. +- Health checks: `GET /health` for liveness, plus a readiness check verifying + the database opens and migrations are current. +- **Back up the SQLite file** with `VACUUM INTO` or the backup API (never a + plain file copy of a live WAL database). Manifests represent real CV + compute; federation (§9a) gives partial resilience but is not a backup. + +--- + +## 9. JRay plugin integration + +### Configuration + +- **Enable manifest sharing** (default off — this is a network egress feature + and must be opt-in) +- **Servers** — an *ordered list*, not a single URL. See below. +- **Contribute manifests** (separate opt-in from downloading; off by default) +- **Minimum accepted match tier** (`exact` / `audio` / `runtime` / `loose`) +- **Compute audio signatures** (default off) — enables `audio`-tier matching + and unknown-providence search (§3). Uses the FFmpeg binary Jellyfin already + ships, via `IMediaEncoder.EncoderPath`; no extra dependency. + +### Multiple servers + +The plugin queries a user-configured **ordered list** of servers rather than +one. Each entry is: + +| Field | Purpose | +|---|---| +| `Url` | Base URL | +| `Name` | Display label | +| `Token` | Optional; required only to contribute | +| `Enabled` | Toggle without deleting | +| `AllowContribute` | Per-server, independent of fetching | +| `TrustLevel` | `Full` / `FetchOnly` — see below | + +A default entry for the community instance ships pre-configured but +**disabled**, so no traffic leaves an installation until the admin opts in. + +**Resolution order.** For a fetch, servers are tried in list order and the +*first acceptable* result wins — acceptable meaning it clears the configured +match tier. Order is the user's trust ranking, made explicit. Rationale for +first-match over best-match: querying every server for every item multiplies +egress, leaks the library to more parties, and the ordering already encodes +which source the admin prefers. A `Best match across servers` toggle is a +reasonable later addition, off by default. + +For a **series bundle**, first-match applies per *episode*, not per bundle: +fetch the bundle from server 1, then query server 2 only for the episodes +still missing. A series is commonly split across sources, and this is where +multi-server earns its keep. + +**Failure isolation.** A server that is unreachable, slow, or returning errors +is skipped after a short timeout (5s connect, 30s read) and marked +temporarily failed with exponential backoff. One dead server must never stall +a library sweep. Failures are surfaced per-server in the config page. + +**Contribution is never fanned out.** A manifest is contributed only to +servers with `AllowContribute` set, and each is an explicit choice. The plugin +must not broadcast uploads to every configured server — that would multiply +the privacy exposure described below without the user intending it. + +### Trusting third-party servers + +This is the part that does not come for free. Everything in §5a is a property +of a *correctly operated* server. Pointing the plugin at an arbitrary URL +inherits none of it: a hostile server can serve malformed manifests, wrong +casts, or oversized payloads. + +The plugin therefore treats **every** server as untrusted, including the +default one, and re-applies client-side what the server applies on upload: + +- **Validate on receipt.** Downloaded manifests are validated against the same + strict schema used for uploads (§6 stage 2) — unknown fields rejected, sizes + capped, scene windows bounds-checked against the item's real runtime. A + manifest is never trusted merely because a server served it. +- **Response size caps** enforced during streaming, so an unbounded body is + aborted rather than buffered. Bundle cap 25 MiB, single manifest 2 MiB. +- **HTTPS required** for non-loopback servers; certificate validation must not + be disabled. A plaintext community server would let any network intermediary + rewrite actor overlays. +- **Names rendered as text, never markup** (§5a client-side hardening). This + is the single most important control, because it holds even if every other + check is bypassed. +- **`TrustLevel: FetchOnly`** — the default for user-added servers — accepts + manifests but never contributes to them and never sends library inventory + beyond the single item being queried. + +The honest framing for the config page: *adding a third-party server means +trusting its operator not to serve you deliberately wrong actor data.* The +structural protections above bound the damage to bad overlay content; they +cannot make wrong data right. + +### Endpoints + +Mirroring the existing Truth/Tasks controllers: + +- `POST /Plugins/JRay/Items/{itemId}/Fetch` — resolve the item across the + configured servers in order and, on a match at or above the configured tier, + store the result via the existing `IManagedTruthStore`. Admin key. + When the response carries a non-zero `offset` (§3 `audio` tier), the plugin + **must** add it to every scene window before storing — the stored truth file + is always in the local file's own timebase, so the overlay and the `jray?t=` + query need no offset awareness at read time. +- `POST /Plugins/JRay/Series/{seriesId}/Fetch` — bundle fetch for a whole + series, with per-episode gap-filling across servers as described above. +- `GET /Plugins/JRay/Servers/Status` — per-server reachability and last-error, + for the config page. +- `POST /Plugins/JRay/Items/{itemId}/Identify` — compute the item's audio + signature and search configured servers by content (§3), for items whose + providence is unknown. Returns candidate titles with scores and offsets; + storing a result is a separate confirmation step, never automatic. +- A scheduled task that walks items with no truth data and attempts a fetch, + reusing the same backlog logic as `Tasks/Pending`, using the batch + `exists` endpoint (§4) so a sweep is a handful of requests per server. + +Contribution runs the reverse: on a `PUT .../Truth` from a local worker, if +contribution is enabled, strip `movie`/`jellyfin_id`, attach identity from the +item's `ProviderIds` and its measured runtime, and `POST` to each +contribute-enabled server. For a series, batch into a bundle upload rather +than per-episode posts. + +Uploads should set `Expect: 100-continue` (§6 stage 0) so a server that is +going to reject the request on size or auth does so before the body is +transmitted. This matters most for series bundles, where a rejected upload +would otherwise push tens of MiB pointlessly. + +### Privacy + +Contribution reveals to the server operator that some instance holds a given +title. Fetching reveals the same thing. That is inherent, but it means: + +- opt-in, off by default, clearly described in the config page +- no library-wide inventory ever sent in one request — the batch `exists` + endpoint is capped at 100 items and a sweep is paced +- **each configured server multiplies this exposure**, which the config page + must say plainly; first-match resolution limits it, since later servers are + only queried for what earlier ones lacked + +--- + +## 9a. Federation and replication — UR-008 + +Servers can replicate manifests from each other, so a new instance can bootstrap +from an existing one and independent communities need not each re-run the CV +pipeline on the same films. + +### What makes this easy, and what makes it hard + +**Easy:** a validated manifest is *immutable and content-addressable*. Its +content is a fixed set of (TMDB person id, time windows) for a fixed +(title, cut). Nothing about it changes after acceptance. Replication is +therefore **set reconciliation**, not state synchronisation — there are no +concurrent edits, no last-write-wins, no vector clocks, no merge conflicts. +Two servers holding the same manifest hold byte-identical content. + +**Hard:** the mutable state is exactly the part that must *not* replicate +blindly. `status`, `reports` and `cast_match_ratio` encode a *local operator's +judgement and legal position*. A server that pulls another's `delisted` flags +as authoritative has outsourced its moderation; a server that pulls another's +`listed` flags has outsourced its liability. §5a's guarantees are per-operator, +and federation must not silently transfer them. + +The design follows directly: **replicate content, re-derive judgement.** + +### Content addressing + +Every manifest gets a `content_id` — a SHA-256 over its canonical form: + +``` +sha256(canonical_json({ + identity, cut, actors: [{tmdb_person_id, scenes}] sorted by person id +})) +``` + +Canonicalisation: keys sorted, no whitespace, and **scene times quantised to +whole centiseconds** — `round(t * 100)` stored as an integer, not a rounded +float. `extraction` metadata and all local state are excluded, so two servers +that validated the same upload independently arrive at the same `content_id`. + +**`audio_signature` is excluded from `content_id`**, deliberately. It is +derived by decoding audio, so two servers running different FFmpeg or resampler +versions could compute marginally different signatures for the same manifest — +including it would produce different `content_id`s for identical content and +silently break federation deduplication. The signature is replicated as an +attribute of the manifest, not as part of its identity. A peer that already +holds a manifest but lacks its signature may adopt the incoming one. + +Quantising to integers rather than formatting floats is deliberate. Pipeline +timings are *derived* by accumulating `1/fps`, not measured, so they carry +accumulated float error — real corpus values look like `8045.066666660665`. +Measured over 28972 scene values from the extraction corpus, 3-decimal +rounding has a maximum error of 3.3e-4 and produces no boundary cases, so it +is currently safe. But "currently safe" is luck: any value landing near a +`.0005` boundary would hash differently on two servers that computed it +slightly differently, silently defeating deduplication. + +Integer centiseconds remove the failure mode rather than dodging it — 10 ms is +far below the precision any overlay can use (the pipeline samples at 1–10 fps), +so nothing is lost. The client must canonicalise identically, and the +canonicalisation routine should be shared code between server and client +rather than reimplemented. + +This gives deduplication for free: a pulled manifest whose `content_id` is +already present is skipped without re-validation. It also makes "have you got +this?" a cheap hash comparison rather than a content diff. + +### Replication protocol + +Deliberately a **pull-based feed**, not push. Pull means a server chooses what +it ingests and when; push would let any peer inject work into your validation +queue, which is the same abuse surface as anonymous upload but with higher +volume. + +#### `GET /federation/changes?since={cursor}&limit=1000` + +A monotonic, append-only change feed of locally-*listed* manifests. + +```json +{ + "cursor": "01HZ...", + "server_id": "jray.example.org", + "changes": [ + { + "content_id": "sha256:9f2a…", + "op": "add", + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 6420.5, "video_hash": "opensubtitles:8e24…" }, + "actor_count": 17, + "cast_match_ratio": 0.82, + "origin": "jray.example.org", + "seq": "01HZ..." + } + ] +} +``` + +Entries are metadata only — enough to decide whether to fetch, without +transferring payloads. `op` is `add` or `retract` (see below). The cursor is +opaque and monotonic; a peer resumes from its last cursor, making the feed +resumable and idempotent. + +#### `GET /federation/manifests/{content_id}` + +Fetch full content by hash. The puller **must** verify that the returned +content hashes to the requested `content_id` and reject it otherwise — this is +what makes an intermediary or a misbehaving peer unable to substitute content. + +#### `POST /federation/have` + +Batch existence check by `content_id` (up to 1000), so a peer can diff its set +against yours in one request before fetching anything. + +### Ingestion: re-derive, don't inherit + +A pulled manifest is **not** trusted because a peer listed it. It enters the +local pipeline as if freshly uploaded: + +1. Full §6 stage 1 and 2 validation — size caps, strict schema, bounds checks. + A peer is not exempt from the checks that keep §5a's Threat 1 closed. +2. **Local** TMDB cast cross-check (§6 stage 3), against this server's own + TMDB cache and its own thresholds. The peer's `cast_match_ratio` is + advisory only — useful for prioritising ingestion order, never a substitute. +3. Local `status` assigned by this server's rules. + +The peer's moderation decisions are recorded as *signals*, not verdicts: + +| Peer state | Local effect | +|---|---| +| Peer lists it | Eligible for ingestion; still fully re-validated | +| Peer retracts it (`op: retract`) | Local copy flagged for review, **not** auto-delisted | +| Peer never had it | No signal | + +The asymmetry is deliberate: a retraction is a *warning worth acting on*, +while a listing is merely a *nomination*. Auto-delisting on a peer's retraction +would hand any peer a remote delete primitive over your catalogue. + +**Exception — the abuse channel.** One class of retraction *should* +auto-delist: content withdrawn for legal reasons. A `retract` entry may carry +`reason: "abuse"`, and a peer explicitly configured as +`TrustAbuseRetractions: true` will delist immediately and log it. This is +opt-in per peer, and the intended configuration between operators who know +each other. It exists because the alternative — a takedown propagating at the +speed of manual review — is the wrong failure mode for that one case. + +### Origin and loop prevention + +Each change carries `origin`, the `server_id` that first accepted the +manifest, preserved across hops. A server ignores changes whose `origin` is +itself, which prevents the trivial A→B→A loop. Because content is addressed by +hash and ingestion is idempotent, longer cycles are harmless: the second +arrival is a no-op deduplication. + +`origin` is provenance, not authority — it does not confer trust, it just +enables an operator to say "stop ingesting anything originating from X". + +### Peer configuration + +Symmetrical with the plugin's server list (§9), and for the same reason — +federation is trust-by-configuration, not trust-by-protocol: + +| Field | Purpose | +|---|---| +| `Url` | Peer base URL | +| `Enabled` | Toggle without deleting | +| `PullInterval` | Poll cadence, default hourly | +| `TrustAbuseRetractions` | Auto-delist on legal retractions (default false) | +| `IngestFilter` | Optional: only titles matching a filter (e.g. exclude adult-flagged) | +| `MaxIngestPerHour` | Rate cap, so a peer cannot flood the validation queue | +| `Advertise` | Whether to list this peering in the public directory (default false) | + +**There is no automatic peering. Ever.** A peer relationship is created only +by an operator explicitly adding a URL. Nothing a remote server says, and no +data returned from any endpoint, can cause a peering to be established, +re-enabled, or widened. Automatic peering would let the network's trust +properties be set by whoever joins, which is precisely what §5a avoids. + +Federation is off by default. A server with no peers configured behaves +exactly as specified in §1–§9. + +### Peer directory — publishing, not discovering + +A server *may* publish the peers it has chosen, so an operator evaluating the +network can see who is connected to whom. This is a **human-facing directory**, +not a discovery mechanism. + +The distinction is the whole point: + +| | Peer directory (allowed) | Auto-discovery (prohibited) | +|---|---|---| +| What it does | Publishes a list a human can read | Acts on a list a machine received | +| Who decides | The operator, by hand | The protocol | +| Failure mode | Someone reads a stale list | A hostile server injects itself into your trust set, transitively | + +#### `GET /federation/peers` + +Returns peers this server has chosen to advertise: + +```json +{ + "server_id": "jray.example.org", + "contact": "admin@example.org", + "peers": [ + { "url": "https://jray.other.org", "name": "Other Community", "since": "2026-03-01" } + ] +} +``` + +Rules that keep this a directory and not a discovery channel: + +- **Advertising is per-peer opt-in on both sides.** A peering appears here only + if the local operator set `Advertise: true` *and* the remote operator + consented to being listed. Peering with someone must not publish their + existence against their wishes — for a small operator, being listed is an + invitation to traffic and scrutiny they may not want. +- **The response is never ingested.** The server does not parse it, store it, + or act on it. It is rendered in the admin UI for a human, with entries as + inert text and an explicit "add this peer" button that performs the same + manual add as typing a URL. No one-click "add all". +- **Not transitive.** A peer's peers are not fetched recursively. There is no + crawl, so there is no network-wide topology to poison. +- **`contact` is for humans** arranging a peering out of band, which is the + intended workflow: operators talk, then each adds the other by hand. + +Publishing the directory is itself optional (`PublishPeerDirectory`, default +off). A server that would rather not disclose its topology simply doesn't. + +### Duplicate manifests across origins + +Two servers may independently hold manifests for the same (title, cut) from +different contributors — different `content_id`, same identity. This is +already handled: §7 permits multiple manifests per cut and ranks by +`(cast_match_ratio, reports, gallery_scope, sample_fps)` (§7). Federation just +makes it more +common. No deduplication beyond exact `content_id` match is attempted, because +choosing between two plausible extractions is a ranking problem, not a merge +problem. + +### Storage additions + +```sql +peers (id, url, name, enabled, pull_interval, + trust_abuse_retractions, advertise, peered_since, + last_cursor, last_pull_at, last_error) + +-- manifests gains: +-- content_id text unique -- sha256 over canonical form +-- origin text -- server_id of first acceptance +-- ingested_from text NULL -- peer id, NULL if uploaded directly +``` + +`content_id` carries a unique index and is the deduplication key on ingest. + +### What is deliberately not specified + +- **No automatic peering.** A published peer directory (above) is readable by + humans; it is never acted on by software. No crawling, no transitive + peering, no "trusted because a peer trusts them". +- **No consensus.** There is no global agreement on what the catalogue + contains. Each server's catalogue is its own; federation only makes it + cheaper to fill. +- **No global identity.** No shared contributor identity across servers. + Tokens stay local, consistent with §5a — a contributor's standing on one + server means nothing on another, and needs to mean nothing. +- **No deletion propagation** beyond the opt-in abuse channel above. + +--- + +## 10. Open questions + +1. **Is `runtime` tier good enough as the default?** ±2s will match most + same-cut releases but will also match a different encode with identical + runtime and a different logo trim. Leaning yes-with-caveat-in-UI. +2. **Should manifests be signed by the contributor?** Adds provenance but + also key management for hobbyist operators. Probably not for v1. +3. **Should the gallery be shareable too?** Embeddings are far larger than + manifests and are derived from copyrighted headshots; out of scope here, + but worth a separate look — it would remove the biggest setup cost for new + users. +4. **Federation** is now specified in §9a. Open sub-questions: + - **Is re-running the TMDB cast check on every ingested manifest + affordable?** Bulk-ingesting a large peer catalogue means a TMDB lookup + per distinct title. The 24h credits cache and the fact that lookups are + per *title* (not per manifest) should make it fine, but a bootstrap of + tens of thousands of titles needs a throttled backfill mode rather than + the normal upload path. + - **Should a fresh server be allowed to trust a peer's `cast_match_ratio` + during initial bootstrap only?** It would make standing up a mirror far + cheaper, at the cost of the guarantee in §9a. Leaning no. + - ~~Float canonicalisation stability~~ — **resolved.** Checked against + 28972 scene values from the corpus: 3dp rounding is safe today (max error + 3.3e-4, no boundary cases) but fragile, since pipeline timings accumulate + float error from summing `1/fps`. §9a now quantises to integer + centiseconds, which removes the failure mode rather than relying on + luck. +5. **Is 0.6 the right cast-match threshold?** Still a guess, but now a + *testable* one: the extraction repo has 331 real output files. Running the + §6 stage-3 check over them against TMDB would yield the true distribution + of honest-upload match ratios and let the threshold be set at, say, the 1st + percentile rather than by intuition. This is the single cheapest way to + de-risk UR-003 and UR-005 and should happen before launch. Note that the corpus + is heavily TV-weighted, so movie and episode thresholds may need to differ. +6. **Discarding the uploaded `name` string (§5a) depends on TMDB person + resolution being reliable.** If too many legitimate actors fail to resolve, + manifests lose actors silently. The corpus run in (5) measures this too. If + resolution proves lossy, the fallback is to store names but restricted to + the closed character class — weaker, but still not a usable payload channel. +7. **Anonymous existence checks are a title-availability oracle.** Rate limits + blunt this but do not remove it. Requiring a token for `exists` would close + it at the cost of making read-only use non-anonymous. Left open. +8. **Adult-content classification for the §5a category guard** relies on + TMDB's `adult` flag, whose coverage for *performers* is less consistent + than for titles. The guard may need a supplementary signal. diff --git a/deny.toml b/deny.toml new file mode 100644 index 0000000..4663bd3 --- /dev/null +++ b/deny.toml @@ -0,0 +1,91 @@ +# cargo-deny configuration. +# +# This crate is GPLv3 (it shares a licence with the JRay Jellyfin plugin), and it +# is a long-lived network service whose main risks are hostile input and operator +# friction (§8). So two checks matter most here: +# +# - `advisories` — a public-facing service must not ship known-vulnerable +# dependencies. +# - `licenses` — GPLv3 is compatible with permissive licences, but *not* with +# everything. A copyleft-incompatible dependency arriving transitively would +# be a licensing problem discovered far too late. +# +# Run with `cargo deny check`. + +[graph] +# Check the targets an operator actually deploys. §8 ships a single static binary +# (musl target), so both glibc and musl Linux are in scope. +targets = [ + "x86_64-unknown-linux-gnu", + "x86_64-unknown-linux-musl", + "aarch64-unknown-linux-gnu", + "aarch64-unknown-linux-musl", +] +all-features = true + +[advisories] +version = 2 +# Fail on any RustSec advisory. Unmaintained crates are a warning rather than an +# error: `sled` was rejected in §8 partly on maintenance grounds, so the signal is +# worth surfacing, but it should not break a build on its own. +yanked = "deny" +unmaintained = "workspace" +ignore = [] + +[licenses] +version = 2 +# Permissive licences, all GPLv3-compatible. Deliberately a closed allow-list +# rather than a deny-list: a licence nobody vetted should stop the build, in the +# same spirit as §6's "no additional fields anywhere". +# +# Kept to licences actually present in the tree, so `cargo deny` stays quiet in +# CI and an added allowance is a visible decision. Adding a dependency that needs +# a new licence should be a deliberate edit here. +allow = [ + "Apache-2.0", + "MIT", + "BSD-2-Clause", + "BSD-3-Clause", + "ISC", + "Zlib", + "Unicode-3.0", + # `webpki-roots` — Mozilla's trusted CA certificate set. This is a *data* + # licence, not a code licence, which is why it is not on the usual permissive + # list: the crate ships certificates rather than logic. CDLA-Permissive-2.0 + # imposes no copyleft and no attribution burden on a binary that embeds it, so + # it is compatible with distributing this server under GPLv3. + # + # It arrives via reqwest's rustls stack, which §8's single static musl binary + # depends on (bundling roots is what lets the binary verify TLS without a + # system trust store). + "CDLA-Permissive-2.0", + # This crate's own licence. + "GPL-3.0-or-later", +] +confidence-threshold = 0.9 +# `ring` ships a bespoke licence file that no SPDX expression describes; it is +# a permissive OpenSSL/ISC-style licence and is GPL-compatible. Clarify it rather +# than widening the allow-list. +[[licenses.clarify]] +crate = "ring" +expression = "MIT AND ISC AND OpenSSL" +license-files = [{ path = "LICENSE", hash = 0xbd0eed23 }] + +[bans] +multiple-versions = "warn" +wildcards = "deny" +# Nothing is banned outright yet. The obvious future entries are alternative TLS +# stacks: reqwest is pinned to rustls (`default-features = false`) so that a +# static musl binary needs no system OpenSSL, and an accidental openssl-sys +# dependency would silently break that deployment story. +deny = [] +skip = [] +skip-tree = [] + +[sources] +unknown-registry = "deny" +unknown-git = "deny" +# Only crates.io. A git dependency in a service that hobbyist operators build +# from source is a supply-chain and reproducibility problem. +allow-registry = ["https://github.com/rust-lang/crates.io-index"] +allow-git = [] diff --git a/docs/requirements.md b/docs/requirements.md new file mode 100644 index 0000000..7f525a9 --- /dev/null +++ b/docs/requirements.md @@ -0,0 +1,200 @@ +# JRay-public-server — requirements register + +Stable IDs for every requirement in [`../SPEC.md`](../SPEC.md), which holds the +prose. This file is the **authoritative list**; the CI gate reads its +denominators from here (see the [system spec](../../SPEC.md) §6). + +**IDs are permanent.** A withdrawn requirement is marked `Withdrawn` and its +number is never reused — renumbering is what produces orphan TRACES tags. + +Tag code with `// TRACES: UR-003 | SR-004`. + +| Type | Scope | +|---|---| +| `UR` | User/functional — what the server does | +| `DR` | Development — how it is built and operated | +| `UT` / `IT` | Unit / integration tests | + +Status: `Done` · `In Progress` · `Planned` · `TBD` · `Withdrawn` + +A requirement is `Done` only when it is implemented **and** has a test that +executes. Everything below runs in CI on any machine — this repo has no GPU +requirement and no fixture-generation step, unlike `scene-actor-extraction`. + +--- + +## User requirements (UR) + +| ID | Requirement | Traces to | Priority | Status | +|---|---|---|---|---| +| UR-001 | Cheap existence probe, separate from the fetch, returning availability and cut-match tier without payload | SR-001 | High | Done | +| UR-002 | Accept a contributed manifest for a media item | PR-006 | High | Done | +| UR-003 | Content verification: strict schema, size caps, approximate TMDB cast match | SR-004 | High | Done | +| UR-004 | Rate limiting, per token where present and per source IP otherwise | SR-004 | High | Done | +| UR-005 | Trust without accounts: not usable as a content store, nor for prank manifests | SR-004 | High | Done | +| UR-006 | Serve and accept a whole series in one operation | PR-006 | High | Done | +| UR-007 | Plugin queries an ordered, configurable list of servers | PR-005 | High | In Progress | +| UR-008 | Servers replicate manifests between each other | PR-006 | Medium | Planned | +| UR-009 | Store an audio spectral-peak signature for content-based identification | SR-003 | Medium | In Progress | +| UR-010 | Identity crossing the API boundary is TMDB/IMDB ids, never a name alone | SR-001 | High | Done | +| UR-011 | Reject any field capable of carrying binary or attacker-chosen content | SR-004 | High | Done | +| UR-012 | Never accept, store, or serve gallery data — reference faces or embeddings | SR-005 | High | Done | +| UR-013 | Windows are scene-scoped claims; never reinterpret their boundaries | SR-002 | High | Done | +| UR-014 | Reject an unknown `jmanifest_version` outright, never guess | SR-003 | High | Done | + +### Notes on status + +**UR-007 is `In Progress`, not `Done`.** The plugin now carries the ordered +server list and its per-server trust settings, with the community instance +pre-configured but disabled. The fetch path that consumes it does not exist yet. + +**UR-009 is `In Progress`.** The server accepts, validates and stores +`cut.audio_signature`, and `content_id` correctly excludes it (§9a). What is +absent is `audio`-tier matching and `POST /manifests/search`. This is the +sequencing §3 recommends — accumulate signatures first, enable matching once +coverage is useful — not an oversight. + +**UR-012 is satisfied structurally, by absence.** There is no field in the +Jmanifest capable of carrying an embedding or a crop, and no endpoint that would +accept one. Like PR-005 in the system spec, it cannot be verified by pointing at +code that does something; UT-024 verifies it by asserting that the obvious +attempts are rejected. + +--- + +## Development requirements (DR) + +| ID | Requirement | Traces to | Priority | Status | +|---|---|---|---|---| +| DR-001 | Strict parse boundary: unknown fields rejected structurally, not by validator code | SR-004 | High | Done | +| DR-002 | Fully relational storage — no JSON blob on the write path | SR-004 | High | Done | +| DR-003 | Single serialized writer connection, with a read pool alongside | PR-004 | High | Done | +| DR-004 | All database access behind a repository layer, not scattered through handlers | PR-004 | Medium | Done | +| DR-005 | Background work in-process, with the job queue as a table so it survives restart | PR-004 | High | Done | +| DR-006 | Rate-limit counters in process memory; no external counter store | PR-004 | Medium | Done | +| DR-007 | Ship a single static binary plus one database file; container optional | PR-004 | High | Done | +| DR-008 | `X-Forwarded-For` honoured only from explicitly configured proxies | SR-004 | High | Done | +| DR-009 | Body caps enforced while streaming, before parsing, per route | SR-004 | High | Done | +| DR-010 | Request bodies are UTF-8 only, rejected with a diagnosable error otherwise | SR-003 | Medium | Done | +| DR-011 | `content_id` canonical form is byte-stable and cross-implementation tested | SR-003 | High | Done | +| DR-012 | Dependency audit: advisories, licence policy, source policy | PR-004 | Medium | Done | +| DR-013 | API errors use the status codes the spec names, not the framework's defaults | SR-003 | Medium | Done | +| DR-014 | Portable SQL — no SQLite-specific form where a standard one exists | PR-004 | Medium | Done | + +--- + +## Verification + +**No GPU, no fixtures, no external services.** Every test here runs on any +machine in under three seconds. The TMDB dependency is the only external service, +and it is absent from tests by construction: an unconfigured client makes uploads +stay `pending`, which is the correct production failure mode (§8) and happens to +make the test suite hermetic. + +| Tier | Runs in CI | What it covers | +|---|---|---| +| **T1 — unit** | Yes | Pure logic: validation, cut matching, cast-check scoring, canonicalisation, rate limiting | +| **T2 — integration** | Yes | End-to-end through the real router against a temporary on-disk database | + +There is no tier that does not run. A requirement here is either verified or +visibly not. + +**Integration tests use an on-disk temporary database, not `:memory:`.** DR-003 +specifies one writer connection plus a read pool, and in-memory SQLite is +per-connection — the readers would see an empty database. Testing the real +topology is the point, so this is a deliberate choice rather than an oversight. + +### Per-requirement verification + +| ID | Tier | Test asserts | Edge cases covered | +|---|---|---|---| +| UR-001 | T2 | `exists` reports availability and tier without payload | Absent title returns `200` with `false`, not `404`; batch form is positional; one bad item does not fail the batch | +| UR-002 | T2 | Valid upload accepted as `202 pending` | Duplicate content deduplicates; same contributor resubmitting the same cut is `409` | +| UR-003 | T1 + T2 | Strict schema, caps, and cast-match thresholds | Ratio boundaries at 0.6 and 0.3 exactly; small-\|M\| all-but-one rule; missing TMDB credits flags rather than rejects | +| UR-004 | T1 + T2 | Limits engage and carry the documented headers | Window reset; a rejected request does not extend its own lockout; surfaces have independent budgets | +| UR-005 | T1 + T2 | Prank manifests rejected; no free-text channel | Uncredited cast rejected; name-only matches capped; automatic revocation needs a minimum sample | +| UR-006 | T2 | Bundle accepted per-episode, non-atomically | One bad episode rejected while its neighbours are accepted; envelope errors are whole-request `400` | +| UR-007 | — | *No server-side test.* Plugin-side; the register there will carry it | — | +| UR-009 | T1 | Signature structurally validated | Fixed length; reserved high bit; **media < 120 s must send no signature at all** | +| UR-010 | T1 + T2 | Actors persist as TMDB person ids | A name the upload invented does not round-trip | +| UR-011 | T2 | Every payload-shaped field rejected | base64, hex, markup, control characters, bidi overrides, compatibility homoglyphs | +| UR-012 | T2 | No endpoint accepts embeddings or image data | An `embedding` or `crop` field is an unknown-field `400` | +| UR-013 | T1 | Stored windows are byte-identical to those submitted | Adjacent windows never merged; a window is never trimmed to a shorter one | +| UR-014 | T1 | Unknown `jmanifest_version` rejected | Version `2` and version `0` both refused, naming the field | +| DR-001 | T1 | Unknown field at any nesting depth fails to parse | `movie` and `jellyfin_id` named in the error | +| DR-003 | T1 | Concurrent writes serialize rather than returning `SQLITE_BUSY` | Failed transaction rolls back fully | +| DR-005 | T1 | Jobs lease once, reschedule with backoff, survive restart | Stranded lease released at startup; future job not leased early | +| DR-008 | T2 | Forged `X-Forwarded-For` cannot mint a fresh budget | Untrusted peer ignored; trusted proxy honoured; client-supplied entries to the left cannot spoof | +| DR-009 | T2 | Oversized body rejected as `413` | **A lying `Content-Length` does not bypass the cap**; per-route limits differ | +| DR-010 | T1 | Non-UTF-8 rejected by name | UTF-16 with and without BOM; UTF-8 BOM; declared `charset=utf-16` | +| DR-011 | T1 | Canonical form is stable and order-independent | Accumulated float error hashes identically; `audio_signature` and `extraction` excluded; **golden vector verified against an independent Python implementation** | +| DR-013 | T1 | Schema mismatch is `400`, not the framework's `422` | §4 names `400` for a forbidden field, and a client checking for it would mishandle `422` | + +Three are worth singling out, because each verifies a claim that would otherwise +be an assertion: + +- **DR-009's lying-`Content-Length` case.** §6 stage 0 is explicit that the + header is a claim by the client, so the streaming cap is mandatory rather than + redundant. A test that only sends honest bodies verifies nothing. +- **DR-011's golden vector.** Two servers that validated the same upload must + reach the same `content_id`, and the plugin must reproduce it byte-identically + from a different language. The fixture is the only thing that can catch + divergence before it silently breaks federation deduplication. +- **UR-011's homoglyph case.** Writing this test found a real gap: compatibility + variants (`𝐒𝐭𝐞𝐯𝐞`, `Actor`) are letters by Unicode category and NFC does not + fold them, so a fullwidth-digit alphabet would have reopened the encoding + channel §5a's "no digits" rule closes. + +--- + +## Withdrawn + +| ID | Requirement | Reason | +|---|---|---| +| — | `anneal_sec` in `extraction` | Withdrawn upstream (`scene-actor-extraction` AR-012/AR-013): presence follows track extent, so a track survives its own gaps and there is nothing to anneal. Ships as part of the SR-003 bump | + +Deleted rather than retained at zero: a field naming a mechanism the pipeline no +longer has is actively misleading to anyone reading a manifest, and would outlive +everyone who remembers why it is zero. + +No `UR`/`DR` number was ever assigned to it — it was a *field*, not a +requirement — so nothing is orphaned by its removal. + +--- + +## Pending — the SR-003 schema bump + +These are `Planned` rather than absent, because the bump is coordinated across +three repos and this register should show the work rather than imply the server +is finished. + +| ID | Requirement | Traces to | Priority | Status | +|---|---|---|---|---| +| UR-015 | Accept `extraction.extinction_sec` in place of `anneal_sec` | SR-003 | High | Planned | +| UR-016 | Accept and store `extraction.gallery_scope`; rank on it (§7) | SR-003 | Medium | Planned | +| UR-017 | Accept per-window belief and identification route; `scenes` becomes objects | SR-003 | High | Planned | +| UR-018 | Exclude belief from `content_id`, replicating it as an attribute | SR-003 | High | Planned | + +**UR-018 is the one with a trap in it.** Belief is a producer-side estimate that +may legitimately differ between pipeline versions for identical timings, so +including it in the canonical form would give two servers different `content_id`s +for the same content — the exact failure mode §9a quantises centiseconds to +avoid. It follows `audio_signature`'s precedent: replicated as an attribute, not +part of identity. + +--- + +## Notes on coverage + +- **`DR-*` traces to `PR-004` (self-hosted) more often than to an `SR-nnn`.** + Operational simplicity is a single-repo concern serving the project goal + directly. This is correct rather than a gap: §8's whole argument for one binary + and one file is that federation only works if running an instance is easy. +- **UR-007 has no server-side test** and cannot have one — it is a requirement on + the plugin, recorded here because this spec is where it is stated. It should be + cross-referenced from the plugin's register when that is created, and until + then it is visibly unverified rather than quietly assumed. +- **UR-012 and UR-013 are preserved by prohibition**, like PR-005 in the system + spec. They cannot be verified by pointing at code that does something, only by + asserting that the attempts fail. They die the moment either prohibition is + relaxed, which is precisely why they are stated rather than left implicit. diff --git a/rustfmt.toml b/rustfmt.toml new file mode 100644 index 0000000..3714eb1 --- /dev/null +++ b/rustfmt.toml @@ -0,0 +1,8 @@ +# Formatting matches the style the codebase is written in. +# +# `use_small_heuristics = "Max"` keeps short structs, calls and match arms on one +# line rather than exploding them across four. In a codebase this dense with spec +# citations, vertical space spent on punctuation is space not spent on the comment +# explaining *why* a rule exists. +max_width = 100 +use_small_heuristics = "Max" diff --git a/src/api/exists.rs b/src/api/exists.rs new file mode 100644 index 0000000..22fb70c --- /dev/null +++ b/src/api/exists.rs @@ -0,0 +1,187 @@ +//! `GET /manifests/exists` and its batch form — UR-1. +//! +//! Deliberately a *separate, cheaper* endpoint from the fetch: it answers +//! "should I bother?" for a whole library sweep without transferring payloads, +//! and it is the endpoint a scheduled task will hammer. It is also the most +//! abuse-prone surface, since it doubles as an oracle for "does the community +//! have this title" — so it is rate-limited harder than the fetches and returns +//! no manifest content (§0). + +use axum::extract::{Query, State}; +use axum::http::HeaderMap; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::{Deserialize, Serialize}; + +use super::LookupParams; +use crate::db::repo; +use crate::error::{ApiError, ApiResult}; +use crate::matching::{self, StoredCut}; +use crate::model::{IdentityType, MatchTier}; +use crate::ratelimit::Surface; +use crate::state::{with_quota_headers, AppState}; + +/// §4: no manifest content, just availability and tier. +#[derive(Debug, Clone, Serialize)] +pub struct ExistsResponse { + pub exists: bool, + #[serde(skip_serializing_if = "Option::is_none")] + pub r#match: Option<&'static str>, + #[serde(skip_serializing_if = "Option::is_none")] + pub manifest_id: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub actor_count: Option, +} + +impl ExistsResponse { + fn absent() -> Self { + Self { exists: false, r#match: None, manifest_id: None, actor_count: None } + } +} + +/// §4 batch form: up to 100 items. +/// +/// Exists specifically so the §5 rate limit can be generous per *request* while +/// staying strict per *item*, and so a 2000-item library sweep is 20 requests +/// rather than 2000. +pub const MAX_BATCH_ITEMS: usize = 100; + +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BatchRequest { + pub items: Vec, +} + +#[derive(Debug, Serialize)] +pub struct BatchResponse { + /// Positional, matching the request order (§4). + pub results: Vec, +} + +pub async fn exists( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Query(params): Query, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ExistsSingle)?; + let body = lookup_one(&state, ¶ms).await?; + Ok(with_quota_headers(Json(body).into_response(), quota)) +} + +pub async fn exists_batch( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + super::json::Json(req): super::json::Json, +) -> ApiResult { + if req.items.len() > MAX_BATCH_ITEMS { + return Err(ApiError::BadRequest(format!( + "items: at most {MAX_BATCH_ITEMS} per request, got {}", + req.items.len() + ))); + } + if req.items.is_empty() { + return Err(ApiError::BadRequest("items: must not be empty".into())); + } + + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ExistsBatch)?; + + let mut results = Vec::with_capacity(req.items.len()); + for item in &req.items { + // A malformed item yields "absent" rather than failing the whole batch — + // a sweep of 100 items should not be lost to one bad entry. + results.push(lookup_one(&state, item).await.unwrap_or_else(|_| ExistsResponse::absent())); + } + + Ok(with_quota_headers(Json(BatchResponse { results }).into_response(), quota)) +} + +async fn lookup_one(state: &AppState, params: &LookupParams) -> ApiResult { + let Some((kind, tmdb_id, imdb_id)) = resolve_kind(params) else { + return Err(ApiError::BadRequest( + "requires tmdb_id/imdb_id, or series_tmdb_id with season and episode".into(), + )); + }; + + let (season, episode) = match kind { + IdentityType::Movie => (None, None), + IdentityType::Episode => (params.season, params.episode), + }; + let client_cut = params.client_cut(); + + let found = state + .db + .read(move |conn| { + let Some(title) = repo::find_title(conn, kind, tmdb_id.as_deref(), imdb_id.as_deref())? + else { + return Ok(None); + }; + let candidates = repo::candidates_for_title(conn, &title.id, season, episode)?; + if candidates.is_empty() { + return Ok(None); + } + + let cuts: Vec<(String, StoredCut)> = candidates + .iter() + .map(|m| { + ( + m.id.clone(), + StoredCut { runtime_sec: m.runtime_sec, video_hash: m.video_hash.clone() }, + ) + }) + .collect(); + + let Some((id, m)) = matching::best_match(&client_cut, &cuts) else { + return Ok(None); + }; + let actor_count = repo::manifest_actor_ids(conn, &id)?.len() as i64; + Ok(Some((id, m.tier, actor_count))) + }) + .await + .map_err(ApiError::Internal)?; + + // §4: `exists: false` is returned with `200`, not `404` — absence is a normal + // answer to this question, and `404` would conflate "no manifest" with "bad + // route" for the client. + Ok(match found { + Some((id, tier, actor_count)) => ExistsResponse { + exists: true, + // With no cut parameters the answer is "some manifest exists" with + // `"match": "unknown"`; the client must still fetch to find out + // whether a cut aligns. This is the mode a library sweep uses (§4). + r#match: Some(tier.as_str()), + manifest_id: Some(id), + actor_count: Some(actor_count), + }, + None => ExistsResponse::absent(), + }) +} + +/// Determines whether these parameters address a movie or an episode. +pub fn resolve_kind( + params: &LookupParams, +) -> Option<(IdentityType, Option, Option)> { + if params.series_tmdb_id.is_some() || params.series_imdb_id.is_some() { + // Episode coordinates are required alongside series identity; without + // them the caller wants the series bundle endpoint instead. + params.season?; + params.episode?; + return Some(( + IdentityType::Episode, + params.series_tmdb_id.clone(), + params.series_imdb_id.clone(), + )); + } + if params.tmdb_id.is_some() || params.imdb_id.is_some() { + return Some((IdentityType::Movie, params.tmdb_id.clone(), params.imdb_id.clone())); + } + None +} + +/// Exposed for tests asserting the documented tier string. +pub fn tier_str(t: MatchTier) -> &'static str { + t.as_str() +} diff --git a/src/api/fetch.rs b/src/api/fetch.rs new file mode 100644 index 0000000..48b8ea3 --- /dev/null +++ b/src/api/fetch.rs @@ -0,0 +1,351 @@ +//! Manifest fetch endpoints (§4). +//! +//! §7: the submitted JSON was parsed, validated, resolved to TMDB person ids, +//! written as rows and discarded. Everything served here is **reconstructed** +//! from those rows, never echoed — which is what makes §5a's Threat 1 defence +//! structural rather than a promise. + +use axum::extract::{Path, Query, State}; +use axum::http::HeaderMap; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::Serialize; + +use super::LookupParams; +use crate::db::repo::{self, ManifestRow}; +use crate::error::{ApiError, ApiResult}; +use crate::matching::{self, StoredCut}; +use crate::model::{ + Actor, Coverage, Cut, Extraction, GalleryScope, Identity, IdentityType, Jmanifest, MatchTier, + SeriesBundle, SeriesRef, JMANIFEST_VERSION, +}; +use crate::ratelimit::Surface; +use crate::state::{with_quota_headers, AppState}; + +#[derive(Debug, Serialize)] +pub struct FetchResponse { + pub r#match: &'static str, + /// Scene offset the client must add (§3). Zero for the tiers currently + /// served; present unconditionally so the plugin contract does not change + /// when `audio` is enabled. + pub offset_sec: f64, + pub manifest: Jmanifest, +} + +#[derive(Debug, Serialize)] +pub struct StatusResponse { + pub status: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub reason: Option, +} + +pub async fn get_movie( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Query(params): Query, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ManifestFetch)?; + + if params.tmdb_id.is_none() && params.imdb_id.is_none() { + return Err(ApiError::BadRequest("requires tmdb_id or imdb_id".into())); + } + let body = fetch_best(&state, IdentityType::Movie, ¶ms, None, None).await?; + Ok(with_quota_headers(Json(body).into_response(), quota)) +} + +pub async fn get_episode( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Query(params): Query, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ManifestFetch)?; + + if params.series_tmdb_id.is_none() && params.series_imdb_id.is_none() { + return Err(ApiError::BadRequest("requires series_tmdb_id or series_imdb_id".into())); + } + let (Some(season), Some(episode)) = (params.season, params.episode) else { + return Err(ApiError::BadRequest("requires season and episode".into())); + }; + let body = + fetch_best(&state, IdentityType::Episode, ¶ms, Some(season), Some(episode)).await?; + Ok(with_quota_headers(Json(body).into_response(), quota)) +} + +async fn fetch_best( + state: &AppState, + kind: IdentityType, + params: &LookupParams, + season: Option, + episode: Option, +) -> ApiResult { + let (tmdb_id, imdb_id) = match kind { + IdentityType::Movie => (params.tmdb_id.clone(), params.imdb_id.clone()), + IdentityType::Episode => (params.series_tmdb_id.clone(), params.series_imdb_id.clone()), + }; + let client_cut = params.client_cut(); + + let found = state + .db + .read(move |conn| { + let Some(title) = repo::find_title(conn, kind, tmdb_id.as_deref(), imdb_id.as_deref())? + else { + return Ok(None); + }; + let candidates = repo::candidates_for_title(conn, &title.id, season, episode)?; + let cuts: Vec<(ManifestRow, StoredCut)> = candidates + .into_iter() + .map(|m| { + let cut = + StoredCut { runtime_sec: m.runtime_sec, video_hash: m.video_hash.clone() }; + (m, cut) + }) + .collect(); + + let Some((row, m)) = matching::best_match(&client_cut, &cuts) else { + return Ok(None); + }; + let manifest = reconstruct(conn, &row, &title, kind)?; + Ok(Some((m.tier, m.offset_sec, manifest))) + }) + .await + .map_err(ApiError::Internal)?; + + // §4: `404` if none clears `loose`. + let (tier, offset_sec, manifest) = found.ok_or(ApiError::NotFound)?; + Ok(FetchResponse { r#match: tier.as_str(), offset_sec, manifest }) +} + +/// `GET /manifests/series/{series_tmdb_id}?season=` (§4). +/// +/// Returns whatever episodes the server holds. **Partial bundles are normal** — a +/// bundle with 9 of 13 episodes is a valid, useful response, not an error (§2). +/// Episode-level cut matching is done client-side against the returned bundle, +/// since a client pulling a whole series already knows its own runtimes. +pub async fn get_series( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Path(series_tmdb_id): Path, + Query(params): Query, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::SeriesFetch)?; + + let season = params.season; + let bundle = state + .db + .read(move |conn| { + let Some(title) = + repo::find_title(conn, IdentityType::Episode, Some(&series_tmdb_id), None)? + else { + return Ok(None); + }; + let rows = repo::episodes_for_series(conn, &title.id, season)?; + + // Multiple contributors may hold the same episode; `episodes_for_series` + // orders by rank, so keep the first per (season, episode). + let mut episodes: Vec = Vec::new(); + let mut seen: Vec<(i64, i64)> = Vec::new(); + let mut seasons: Vec = Vec::new(); + + for row in rows { + let key = (row.season.unwrap_or(-1), row.episode.unwrap_or(-1)); + if seen.contains(&key) { + continue; + } + seen.push(key); + if !seasons.contains(&key.0) { + seasons.push(key.0); + } + episodes.push(reconstruct(conn, &row, &title, IdentityType::Episode)?); + } + seasons.sort_unstable(); + + Ok(Some(SeriesBundle { + jmanifest_version: JMANIFEST_VERSION, + series: SeriesRef { + series_tmdb_id: title.tmdb_id.clone(), + series_imdb_id: title.imdb_id.clone(), + title: title.name.clone(), + }, + coverage: Some(Coverage { episodes_available: episodes.len(), seasons }), + episodes, + })) + }) + .await + .map_err(ApiError::Internal)?; + + let bundle = bundle.filter(|b| !b.episodes.is_empty()).ok_or(ApiError::NotFound)?; + Ok(with_quota_headers(Json(bundle).into_response(), quota)) +} + +/// `GET /manifests/{id}` — fetch a specific manifest by its server-assigned id, +/// for debugging and for the "report this manifest" flow (§4). +pub async fn get_by_id( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Path(id): Path, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ManifestFetch)?; + + let manifest = state + .db + .read(move |conn| { + let Some(row) = repo::manifest_by_id(conn, &id)? else { return Ok(None) }; + // Unlisted manifests are not served to anyone (§6 stage 3). + if row.status != "listed" && row.status != "flagged" { + return Ok(None); + } + let title = title_of(conn, &row.title_id)?; + let kind = + if title.kind == "movie" { IdentityType::Movie } else { IdentityType::Episode }; + Ok(Some(reconstruct(conn, &row, &title, kind)?)) + }) + .await + .map_err(ApiError::Internal)?; + + let manifest = manifest.ok_or(ApiError::NotFound)?; + Ok(with_quota_headers(Json(manifest).into_response(), quota)) +} + +/// `GET /manifests/{id}/status` — poll the outcome of the asynchronous cast +/// check (§4). +pub async fn get_status( + State(state): State, + Path(id): Path, +) -> ApiResult> { + let found = state + .db + .read(move |conn| repo::manifest_status(conn, &id)) + .await + .map_err(ApiError::Internal)?; + + match found { + Some((status, reason)) => Ok(Json(StatusResponse { status, reason })), + // §6 deletes rejected manifests, so a vanished id is reported as + // rejected rather than as a bad route. + None => Ok(Json(StatusResponse { + status: "rejected".into(), + reason: Some("not_found_or_rejected".into()), + })), + } +} + +fn title_of(conn: &rusqlite::Connection, title_id: &str) -> anyhow::Result { + let row = conn.query_row( + "SELECT id, kind, tmdb_id, imdb_id, name, year, adult, certification + FROM titles WHERE id = ?1", + rusqlite::params![title_id], + |r| { + Ok(repo::TitleRow { + id: r.get(0)?, + kind: r.get(1)?, + tmdb_id: r.get(2)?, + imdb_id: r.get(3)?, + name: r.get(4)?, + year: r.get(5)?, + adult: r.get::<_, i64>(6)? != 0, + certification: r.get(7)?, + }) + }, + )?; + Ok(row) +} + +/// Rebuilds a Jmanifest from stored rows. +/// +/// Names come from `people` — populated from TMDB by the server — so `name` is +/// server-authoritative on download and a name a contributor invented does not +/// round-trip (§2, §5a). +pub fn reconstruct( + conn: &rusqlite::Connection, + row: &ManifestRow, + title: &repo::TitleRow, + kind: IdentityType, +) -> anyhow::Result { + let stored = repo::actors_for_manifest(conn, &row.id)?; + + let actors = stored + .into_iter() + .map(|a| Actor { + name: a.name, + imdb_id: None, + tmdb_id: Some(a.tmdb_person_id.to_string()), + scenes: a + .scenes_cs + .into_iter() + .map(|(s, e)| [s as f64 / 100.0, e as f64 / 100.0]) + .collect(), + }) + .collect(); + + let identity = match kind { + IdentityType::Movie => Identity { + kind, + tmdb_id: title.tmdb_id.clone(), + imdb_id: title.imdb_id.clone(), + series_tmdb_id: None, + series_imdb_id: None, + season: None, + episode: None, + title: title.name.clone(), + year: title.year, + }, + IdentityType::Episode => Identity { + kind, + tmdb_id: None, + imdb_id: None, + series_tmdb_id: title.tmdb_id.clone(), + series_imdb_id: title.imdb_id.clone(), + season: row.season, + episode: row.episode, + title: title.name.clone(), + year: title.year, + }, + }; + + // An unrecognised stored scope is served as absent rather than guessed at: + // the column is written from a closed enum, so anything else means the row + // predates a schema change and its meaning is unknown (UR-014's spirit). + let gallery_scope = match row.gallery_scope.as_deref() { + Some("global") => Some(GalleryScope::Global), + Some("limited") => Some(GalleryScope::Limited), + _ => None, + }; + + let extraction = Extraction { + sample_fps: row.sample_fps, + extinction_sec: row.extinction_sec, + pipeline_version: row.pipeline_version.clone(), + gallery_size: None, + gallery_scope, + }; + let has_extraction = extraction.sample_fps.is_some() + || extraction.extinction_sec.is_some() + || extraction.pipeline_version.is_some() + || extraction.gallery_scope.is_some(); + + Ok(Jmanifest { + jmanifest_version: JMANIFEST_VERSION, + identity, + cut: Cut { + runtime_sec: row.runtime_sec, + container_duration_sec: None, + video_hash: row.video_hash.clone(), + audio_signature: None, + }, + extraction: has_extraction.then_some(extraction), + actors, + }) +} + +/// Exposed so tests can assert the served tier strings. +pub fn tier_name(t: MatchTier) -> &'static str { + t.as_str() +} diff --git a/src/api/json.rs b/src/api/json.rs new file mode 100644 index 0000000..ba3be86 --- /dev/null +++ b/src/api/json.rs @@ -0,0 +1,287 @@ +//! A JSON extractor that fails with the status codes §4 specifies. +//! +//! Axum's own `Json` rejects a body that parses as JSON but does not match the +//! target type with **422 Unprocessable Entity**. §4 is explicit that this case +//! is **`400`** — "malformed, or contains an unrecognised or forbidden field" — +//! and that distinction is load-bearing: §6 requires that a client which forgets +//! to strip `movie` or `jellyfin_id` gets "a hard `400` naming the offending +//! field". A client checking for 400 would mishandle a 422. +//! +//! This wrapper also guarantees the field name reaches the caller, since serde's +//! `deny_unknown_fields` error text is what identifies the offending key. + +use axum::extract::{FromRequest, Request}; +use axum::http::header::CONTENT_TYPE; + +use crate::error::ApiError; + +/// Drop-in replacement for `axum::Json` on request bodies. +pub struct Json(pub T); + +impl FromRequest for Json +where + T: serde::de::DeserializeOwned, + S: Send + Sync, +{ + type Rejection = ApiError; + + async fn from_request(req: Request, state: &S) -> Result { + // A wrong content type is the client's mistake, reported as such rather + // than as a parse failure. + let content_type = + req.headers().get(CONTENT_TYPE).and_then(|v| v.to_str().ok()).unwrap_or("").to_string(); + + let mime = content_type.split(';').next().unwrap_or("").trim().to_ascii_lowercase(); + if !(mime == "application/json" || mime.ends_with("+json")) { + return Err(ApiError::BadRequest("expected content-type: application/json".into())); + } + + // A declared charset other than UTF-8 is refused up front, so the client + // learns what is wrong rather than receiving a confusing parse error from + // deep inside the document. See `require_utf8` for why UTF-8 is the only + // accepted encoding. + if let Some(charset) = + content_type.split(';').skip(1).filter_map(|p| p.trim().strip_prefix("charset=")).next() + { + let charset = charset.trim().trim_matches('"').to_ascii_lowercase(); + if !matches!(charset.as_str(), "utf-8" | "utf8") { + return Err(ApiError::BadRequest(format!( + "unsupported charset {charset:?}: JSON must be UTF-8 encoded (RFC 8259 §8.1)" + ))); + } + } + + let bytes = axum::body::Bytes::from_request(req, state).await.map_err(|e| { + // §6 stage 1: the body cap aborts mid-transfer, and that must surface + // as `413`, not as a generic parse error. Axum folds the length-limit + // case into `FailedToBufferBody`, so the status it chose is the + // reliable discriminator. + if e.status() == axum::http::StatusCode::PAYLOAD_TOO_LARGE { + ApiError::PayloadTooLarge("request body exceeds the limit for this route".into()) + } else { + ApiError::BadRequest(format!("could not read request body: {e}")) + } + })?; + + // Encoding is checked before parsing, so a mis-encoded body gets an + // actionable message instead of whatever the parser happens to trip over. + let text = require_utf8(&bytes)?; + + serde_json::from_str(text) + .map(Json) + // serde's message names the offending field, which is exactly what §6 + // requires the response to identify. + .map_err(|e| ApiError::BadRequest(e.to_string())) + } +} + +/// Enforces that the body is UTF-8, naming the encoding it appears to be. +/// +/// **UTF-8 is the only accepted encoding, deliberately.** RFC 8259 §8.1 requires +/// it for JSON exchanged outside a closed ecosystem, and this is a public, +/// federated API. Three further reasons make it the right call *here* +/// specifically, rather than merely conventional: +/// +/// 1. **§9a content addressing hashes bytes.** `content_id` is a SHA-256 over the +/// canonical form, so the same manifest submitted in two encodings would +/// produce two different ids — silently defeating federation deduplication. +/// That is precisely the failure mode §9a quantises scene times to avoid, and +/// it would be reintroduced at the encoding layer. +/// 2. **UTF-16 admits lone surrogates**, which have no UTF-8 representation. A +/// field able to carry them is a channel for bytes that survive validation but +/// are not text — against §5a's premise that no field can carry a payload. +/// 3. **§5a's character class assumes well-formed Unicode scalar values.** NFC +/// normalisation and the category checks are defined over scalars, so admitting +/// an encoding that can express non-scalars would undermine both. +/// +/// serde_json would reject non-UTF-8 anyway; the value added here is a diagnosable +/// error rather than a misleading one. A UTF-16 body otherwise fails with "key +/// must be a string", which points an operator at the wrong problem entirely. +fn require_utf8(bytes: &[u8]) -> Result<&str, ApiError> { + // A BOM is not valid JSON (RFC 8259 §8.1: "implementations MUST NOT add a + // byte order mark"), and it is the clearest signal of an encoding mistake, so + // it is named rather than left to the parser. + let encoding_hint = match bytes { + [0xEF, 0xBB, 0xBF, ..] => Some("UTF-8 with a byte order mark"), + [0xFF, 0xFE, 0x00, 0x00, ..] => Some("UTF-32LE"), + [0x00, 0x00, 0xFE, 0xFF, ..] => Some("UTF-32BE"), + [0xFF, 0xFE, ..] => Some("UTF-16LE"), + [0xFE, 0xFF, ..] => Some("UTF-16BE"), + // Unmarked UTF-16 is the common case, since encoders often omit the BOM. + // A JSON document always begins with an ASCII character, so an + // interleaved NUL in the first two bytes is conclusive. + [0x00, b, ..] if b.is_ascii_graphic() => Some("UTF-16BE (no BOM)"), + [b, 0x00, ..] if b.is_ascii_graphic() => Some("UTF-16LE (no BOM)"), + _ => None, + }; + + if let Some(encoding) = encoding_hint { + return Err(ApiError::BadRequest(format!( + "request body appears to be {encoding}: JSON must be UTF-8 encoded \ + without a byte order mark (RFC 8259 §8.1)" + ))); + } + + std::str::from_utf8(bytes).map_err(|e| { + ApiError::BadRequest(format!( + "request body is not valid UTF-8 at byte {}: JSON must be UTF-8 encoded \ + (RFC 8259 §8.1)", + e.valid_up_to() + )) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use axum::http::StatusCode; + use axum::response::IntoResponse; + + #[derive(serde::Deserialize)] + #[serde(deny_unknown_fields)] + struct Probe { + _wanted: i64, + } + + async fn extract(body: &'static str, content_type: Option<&str>) -> StatusCode { + let mut builder = Request::builder().method("POST").uri("/"); + if let Some(ct) = content_type { + builder = builder.header(CONTENT_TYPE, ct); + } + let req = builder.body(axum::body::Body::from(body)).unwrap(); + match Json::::from_request(req, &()).await { + Ok(_) => StatusCode::OK, + Err(e) => e.into_response().status(), + } + } + + #[tokio::test] + async fn schema_mismatch_is_400_not_422() { + // The whole reason this extractor exists (§4, §6). + assert_eq!( + extract(r#"{"unexpected":1}"#, Some("application/json")).await, + StatusCode::BAD_REQUEST + ); + } + + #[tokio::test] + async fn malformed_json_is_400() { + assert_eq!(extract("{ nope", Some("application/json")).await, StatusCode::BAD_REQUEST); + } + + #[tokio::test] + async fn missing_content_type_is_400() { + assert_eq!(extract(r#"{"_wanted":1}"#, None).await, StatusCode::BAD_REQUEST); + } + + #[tokio::test] + async fn content_type_parameters_are_tolerated() { + assert_eq!( + extract(r#"{"_wanted":1}"#, Some("application/json; charset=utf-8")).await, + StatusCode::OK + ); + } + + #[tokio::test] + async fn valid_body_extracts() { + assert_eq!(extract(r#"{"_wanted":1}"#, Some("application/json")).await, StatusCode::OK); + } + + #[tokio::test] + async fn an_explicit_utf8_charset_is_accepted() { + for ct in [ + "application/json; charset=utf-8", + "application/json;charset=UTF-8", + "application/json; charset=\"utf-8\"", + "application/json; charset=utf8", + ] { + assert_eq!(extract(r#"{"_wanted":1}"#, Some(ct)).await, StatusCode::OK, "{ct}"); + } + } + + #[tokio::test] + async fn a_non_utf8_charset_is_refused_by_name() { + for ct in [ + "application/json; charset=utf-16", + "application/json; charset=iso-8859-1", + "application/json; charset=windows-1252", + ] { + assert_eq!( + extract(r#"{"_wanted":1}"#, Some(ct)).await, + StatusCode::BAD_REQUEST, + "{ct}" + ); + } + } + + /// Builds a request from raw bytes, since these bodies are not valid `&str`. + async fn extract_bytes(body: Vec) -> Result<(), ApiError> { + let req = Request::builder() + .method("POST") + .uri("/") + .header(CONTENT_TYPE, "application/json") + .body(axum::body::Body::from(body)) + .unwrap(); + Json::::from_request(req, &()).await.map(|_| ()) + } + + #[tokio::test] + async fn utf16_bodies_are_rejected_with_an_actionable_message() { + // The reason this check exists: serde_json rejects UTF-16 anyway, but with + // "key must be a string", which points an operator at the wrong problem. + let doc = r#"{"_wanted":1}"#; + + let le: Vec = doc.encode_utf16().flat_map(|u| u.to_le_bytes()).collect(); + let err = extract_bytes(le).await.unwrap_err().to_string(); + assert!(err.contains("UTF-16LE"), "should name the encoding: {err}"); + assert!(err.contains("UTF-8"), "should say what is required: {err}"); + + let be: Vec = doc.encode_utf16().flat_map(|u| u.to_be_bytes()).collect(); + let err = extract_bytes(be).await.unwrap_err().to_string(); + assert!(err.contains("UTF-16BE"), "should name the encoding: {err}"); + + // With BOMs. + let mut le_bom = vec![0xFF, 0xFE]; + le_bom.extend(doc.encode_utf16().flat_map(|u| u.to_le_bytes())); + assert!(extract_bytes(le_bom).await.is_err()); + + let mut be_bom = vec![0xFE, 0xFF]; + be_bom.extend(doc.encode_utf16().flat_map(|u| u.to_be_bytes())); + assert!(extract_bytes(be_bom).await.is_err()); + } + + #[tokio::test] + async fn a_utf8_bom_is_rejected() { + // RFC 8259 §8.1: implementations MUST NOT add a byte order mark. + let mut body = vec![0xEF, 0xBB, 0xBF]; + body.extend_from_slice(br#"{"_wanted":1}"#); + let err = extract_bytes(body).await.unwrap_err().to_string(); + assert!(err.contains("byte order mark"), "{err}"); + } + + #[tokio::test] + async fn invalid_utf8_is_rejected_with_the_offending_offset() { + // A truncated multi-byte sequence inside an otherwise well-formed document. + let body = b"{\"_wanted\":\"\xC3\x28\"}".to_vec(); + let err = extract_bytes(body).await.unwrap_err().to_string(); + assert!(err.contains("not valid UTF-8"), "{err}"); + assert!(err.contains("byte 12"), "should locate the failure: {err}"); + } + + #[tokio::test] + async fn valid_multibyte_utf8_is_accepted() { + // The check must not reject legitimate non-ASCII content — actor names are + // routinely non-Latin (§5a accepts any Unicode letter). + // Rejected for the unknown `_note` field, not for its encoding — which is + // the distinction being asserted. + let body = r#"{"_wanted":1,"_note":"宮崎 駿 Renée"}"#.as_bytes().to_vec(); + let err = extract_bytes(body).await.unwrap_err().to_string(); + assert!(err.contains("_note"), "should fail on the schema, not the encoding: {err}"); + } + + #[tokio::test] + async fn an_empty_body_is_not_mistaken_for_an_encoding_problem() { + let err = extract_bytes(Vec::new()).await.unwrap_err().to_string(); + assert!(!err.contains("UTF-16"), "empty body is a parse error, not an encoding one: {err}"); + } +} diff --git a/src/api/mod.rs b/src/api/mod.rs new file mode 100644 index 0000000..72e32f1 --- /dev/null +++ b/src/api/mod.rs @@ -0,0 +1,35 @@ +//! HTTP surface (§4). Base path `/api/v1`, JSON throughout. + +pub mod exists; +pub mod fetch; +pub mod json; +pub mod report; +pub mod upload; + +use serde::Deserialize; + +use crate::matching::ClientCut; + +/// Identity + cut query parameters, shared by the read endpoints (§4). +#[derive(Debug, Clone, Default, Deserialize)] +pub struct LookupParams { + pub tmdb_id: Option, + pub imdb_id: Option, + pub series_tmdb_id: Option, + pub series_imdb_id: Option, + pub season: Option, + pub episode: Option, + pub runtime_sec: Option, + pub video_hash: Option, +} + +impl LookupParams { + pub fn client_cut(&self) -> ClientCut { + ClientCut { + // A non-finite or non-positive runtime is not a usable signal; treat + // it as absent rather than letting it drive a match. + runtime_sec: self.runtime_sec.filter(|r| r.is_finite() && *r > 0.0), + video_hash: self.video_hash.clone(), + } + } +} diff --git a/src/api/report.rs b/src/api/report.rs new file mode 100644 index 0000000..a50303a --- /dev/null +++ b/src/api/report.rs @@ -0,0 +1,141 @@ +//! `POST /manifests/{id}/report` (§4), and `GET /health`. +//! +//! Reports are a moderation lever and cheap to abuse, hence the tight §5 limit. +//! A report never changes `status` by itself: §5a keeps delisting an operator +//! action, because automatic delisting on report would hand any client a remote +//! delete primitive. + +use axum::extract::{Path, State}; +use axum::http::HeaderMap; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::{Deserialize, Serialize}; + +use crate::db::repo; +use crate::error::{ApiError, ApiResult}; +use crate::ratelimit::Surface; +use crate::state::{with_quota_headers, AppState}; +use crate::worker::now_iso; + +/// §4: `{ "reason": "misaligned" | "wrong_actors" | "spam", "note": "..." }`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Deserialize, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum ReportReason { + Misaligned, + WrongActors, + Spam, +} + +impl ReportReason { + fn as_str(self) -> &'static str { + match self { + ReportReason::Misaligned => "misaligned", + ReportReason::WrongActors => "wrong_actors", + ReportReason::Spam => "spam", + } + } +} + +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ReportRequest { + pub reason: ReportReason, + #[serde(default)] + pub note: Option, +} + +/// §5a: `note` is free text from an anonymous caller, so it is capped hard. It is +/// never served back to clients — only the operator reads it. +const MAX_NOTE_CHARS: usize = 500; + +#[derive(Debug, Serialize)] +pub struct ReportAccepted { + pub report_id: String, +} + +pub async fn post_report( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Path(manifest_id): Path, + super::json::Json(req): super::json::Json, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::Report)?; + + let note = match req.note { + Some(n) if n.chars().count() > MAX_NOTE_CHARS => { + return Err(ApiError::BadRequest(format!( + "note: longer than {MAX_NOTE_CHARS} characters" + ))) + } + // Strip control characters; the note is operator-facing text, not markup. + Some(n) => Some(n.chars().filter(|c| !c.is_control()).collect::()), + None => None, + }; + + let ip_hash = crate::auth::hash_ip(&ip, &state.config.server_id); + let reason = req.reason.as_str(); + let now = now_iso(); + let id_for_check = manifest_id.clone(); + + let exists = state + .db + .read(move |c| Ok(repo::manifest_by_id(c, &id_for_check)?.is_some())) + .await + .map_err(ApiError::Internal)?; + if !exists { + return Err(ApiError::NotFound); + } + + let report_id = state + .db + .write(move |tx| { + repo::insert_report(tx, &manifest_id, reason, note.as_deref(), &ip_hash, &now) + }) + .await + .map_err(ApiError::Internal)?; + + Ok(with_quota_headers(Json(ReportAccepted { report_id }).into_response(), quota)) +} + +#[derive(Debug, Serialize)] +pub struct Health { + pub status: &'static str, + pub version: &'static str, +} + +/// `GET /health` — liveness, unauthenticated and unlimited (§4, §5). +pub async fn health() -> Json { + Json(Health { status: "ok", version: env!("CARGO_PKG_VERSION") }) +} + +#[derive(Debug, Serialize)] +pub struct Readiness { + pub status: &'static str, + pub database: &'static str, + /// §8: TMDB is a hard dependency for UR-3. If it is unconfigured, uploads + /// accumulate in `pending` rather than being listed unverified — worth + /// surfacing rather than failing silently. + pub tmdb_configured: bool, +} + +/// Readiness check verifying the database opens and migrations are current (§8). +pub async fn ready(State(state): State) -> ApiResult> { + let ok = state + .db + .read(|conn| { + // Any query against a schema table proves both that the file opens + // and that migrations have been applied. + let n: i64 = conn.query_row("SELECT COUNT(*) FROM manifests", [], |r| r.get(0))?; + Ok(n >= 0) + }) + .await + .map_err(ApiError::Internal)?; + + Ok(Json(Readiness { + status: if ok { "ready" } else { "degraded" }, + database: "ok", + tmdb_configured: state.tmdb.is_configured(), + })) +} diff --git a/src/api/upload.rs b/src/api/upload.rs new file mode 100644 index 0000000..8fb4199 --- /dev/null +++ b/src/api/upload.rs @@ -0,0 +1,241 @@ +//! Contribution endpoints (§4) — UR-2 and UR-6. +//! +//! Both require a token (§5). Both return `202`: the upload has passed size and +//! schema validation and is held unlisted pending the asynchronous TMDB cast +//! check (§6 stage 3). + +use axum::extract::State; +use axum::http::{HeaderMap, StatusCode}; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::Serialize; + +use crate::db::repo; +use crate::error::{ApiError, ApiResult}; +use crate::ingest::{self, IngestOutcome}; +use crate::model::{IdentityType, Jmanifest, SeriesBundle}; +use crate::ratelimit::Surface; +use crate::state::{with_quota_headers, AppState}; +use crate::validate::{self, limits}; +use crate::worker::now_iso; + +#[derive(Debug, Serialize)] +pub struct UploadAccepted { + pub manifest_id: String, + pub status: &'static str, +} + +/// `POST /manifests` — UR-2. +pub async fn post_manifest( + State(state): State, + headers: HeaderMap, + super::json::Json(manifest): super::json::Json, +) -> ApiResult { + let contributor = state.require_contributor(&headers).await?; + // §5: limits are per token where one is present. + let quota = state.check_limit(&contributor.id, Surface::ManifestUpload)?; + + // §6 stage 2. A rejection names the offending field, so a client that forgets + // to strip `movie`/`jellyfin_id` gets a diagnosable `400`. + let valid = + validate::validate_manifest(manifest).map_err(|e| ApiError::BadRequest(e.to_string()))?; + + let origin = state.config.server_id.clone(); + let contributor_id = contributor.id.clone(); + let now = now_iso(); + + let outcome = state + .db + .write(move |tx| ingest::persist(tx, &valid, Some(&contributor_id), &origin, None, &now)) + .await + .map_err(ApiError::Internal)?; + + let resp = match outcome { + IngestOutcome::Pending { manifest_id } => { + (StatusCode::ACCEPTED, Json(UploadAccepted { manifest_id, status: "pending" })) + .into_response() + } + // §4 `409` — an identical `(identity, cut)` manifest already exists from + // this contributor. + IngestOutcome::DuplicateFromContributor { manifest_id } => { + return Err(ApiError::Conflict(format!( + "an identical manifest already exists from this contributor: {manifest_id}" + ))) + } + // §9a: identical content already held, from any source. Not an error — + // the contributor's work is simply already represented. + IngestOutcome::DuplicateContent { manifest_id } => { + (StatusCode::OK, Json(UploadAccepted { manifest_id, status: "already_present" })) + .into_response() + } + }; + + Ok(with_quota_headers(resp, quota)) +} + +#[derive(Debug, Serialize)] +pub struct BundleResult { + pub season: Option, + pub episode: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub manifest_id: Option, + pub status: &'static str, + #[serde(skip_serializing_if = "Option::is_none")] + pub reason: Option, +} + +#[derive(Debug, Serialize)] +pub struct BundleAccepted { + pub results: Vec, +} + +/// `POST /manifests/bundle` — UR-6. +/// +/// **Per-episode validation, not atomic**: valid episodes are accepted and +/// invalid ones rejected, with a per-episode result list. All-or-nothing would let +/// one bad episode discard an entire season's compute (§2). +/// +/// **One rate-limit unit**, so contributing a season is not punished relative to +/// contributing a film (§2, §5). +pub async fn post_bundle( + State(state): State, + headers: HeaderMap, + super::json::Json(bundle): super::json::Json, +) -> ApiResult { + let contributor = state.require_contributor(&headers).await?; + let quota = state.check_limit(&contributor.id, Surface::BundleUpload)?; + + // §4: `413` for exceeding the episode cap, distinct from a malformed envelope. + if bundle.episodes.len() > limits::MAX_BUNDLE_EPISODES { + return Err(ApiError::PayloadTooLarge(format!( + "bundle carries {} episodes, limit is {}", + bundle.episodes.len(), + limits::MAX_BUNDLE_EPISODES + ))); + } + // §4: `400` only for the envelope itself; individual bad episodes are + // reported in the results list, not as a whole-request error. + validate::validate_bundle_envelope(&bundle).map_err(|e| ApiError::BadRequest(e.to_string()))?; + + let series_tmdb = bundle.series.series_tmdb_id.clone(); + let mut results = Vec::with_capacity(bundle.episodes.len()); + + for episode in bundle.episodes { + let coords = (episode.identity.season, episode.identity.episode); + + // An episode whose identity contradicts the envelope is rejected on its + // own rather than being silently reattributed to the bundle's series. + if episode.identity.kind != IdentityType::Episode { + results.push(BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: None, + status: "rejected", + reason: Some("identity.type must be 'episode' within a bundle".into()), + }); + continue; + } + if let (Some(envelope), Some(ep)) = (&series_tmdb, &episode.identity.series_tmdb_id) { + if envelope != ep { + results.push(BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: None, + status: "rejected", + reason: Some("series_tmdb_id does not match the bundle envelope".into()), + }); + continue; + } + } + + let valid = match validate::validate_manifest(episode) { + Ok(v) => v, + Err(e) => { + results.push(BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: None, + status: "rejected", + reason: Some(e.to_string()), + }); + continue; + } + }; + + let origin = state.config.server_id.clone(); + let contributor_id = contributor.id.clone(); + let now = now_iso(); + // One transaction per episode, so a bundle never holds the write lock for + // the whole request (§8 chunked ingest reasoning). + let outcome = state + .db + .write(move |tx| { + ingest::persist(tx, &valid, Some(&contributor_id), &origin, None, &now) + }) + .await; + + results.push(match outcome { + Ok(IngestOutcome::Pending { manifest_id }) => BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: Some(manifest_id), + status: "pending", + reason: None, + }, + Ok(IngestOutcome::DuplicateFromContributor { manifest_id }) + | Ok(IngestOutcome::DuplicateContent { manifest_id }) => BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: Some(manifest_id), + status: "already_present", + reason: None, + }, + Err(e) => { + tracing::error!(error = ?e, "bundle episode failed to persist"); + BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: None, + status: "rejected", + reason: Some("internal error".into()), + } + } + }); + } + + let resp = (StatusCode::ACCEPTED, Json(BundleAccepted { results })).into_response(); + Ok(with_quota_headers(resp, quota)) +} + +#[derive(Debug, Serialize)] +pub struct TokenIssued { + pub token: String, +} + +/// Issues an anonymous bearer capability (§5a). +/// +/// Self-issued on request: no email, no verification, no personal data. Stored +/// only as a hash, so the server cannot enumerate who holds tokens. Discarding a +/// token and requesting another is trivially easy — and that is fine, because the +/// token is not the defence; the content checks are. +pub async fn post_token( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, +) -> ApiResult> { + let ip = state.client_ip(&headers, peer.0); + // Reuse the report budget: issuing tokens is cheap but should not be a free + // unbounded write. + state.check_limit(&ip, Surface::Report)?; + + let token = crate::auth::generate_token(); + let hash = crate::auth::hash_token(&token); + let now = now_iso(); + state + .db + .write(move |tx| repo::insert_contributor(tx, &hash, &now)) + .await + .map_err(ApiError::Internal)?; + + Ok(Json(TokenIssued { token })) +} diff --git a/src/app.rs b/src/app.rs new file mode 100644 index 0000000..f3e7f47 --- /dev/null +++ b/src/app.rs @@ -0,0 +1,66 @@ +//! Router construction. +//! +//! §6 stage 0/1 body caps are applied here as per-route `DefaultBodyLimit` +//! layers: Axum rejects on `Content-Length` before reading a body *and* caps the +//! stream for chunked or mis-declared uploads, which is what makes a lying +//! header and a chunked upload both safe. Per-route means the bundle endpoint +//! gets its larger limit without widening the others (§6 stage 1). + +use std::time::Duration; + +use axum::extract::DefaultBodyLimit; +use axum::routing::{get, post}; +use axum::Router; +use tower_http::timeout::TimeoutLayer; +use tower_http::trace::TraceLayer; + +use crate::api::{exists, fetch, report, upload}; +use crate::state::AppState; +use crate::validate::limits; + +/// Small cap for endpoints that take a short JSON body. A read endpoint has no +/// business accepting a large payload, and the batch `exists` form is bounded at +/// 100 items. +const SMALL_BODY_LIMIT: usize = 256 * 1024; + +pub fn router(state: AppState) -> Router { + let timeout = state.config.request_timeout; + + let v1 = Router::new() + // UR-1 — existence probes. + .route("/manifests/exists", get(exists::exists).post(exists::exists_batch)) + // Reads. + .route("/manifests/movie", get(fetch::get_movie)) + .route("/manifests/episode", get(fetch::get_episode)) + .route("/manifests/series/{series_tmdb_id}", get(fetch::get_series)) + .route("/manifests/{id}", get(fetch::get_by_id)) + .route("/manifests/{id}/status", get(fetch::get_status)) + .route("/manifests/{id}/report", post(report::post_report)) + // UR-2 — contribution. + .route( + "/manifests", + post(upload::post_manifest).layer(DefaultBodyLimit::max(limits::BODY_LIMIT_MANIFEST)), + ) + // UR-6 — whole-series contribution, with its own larger cap. + .route( + "/manifests/bundle", + post(upload::post_bundle).layer(DefaultBodyLimit::max(limits::BODY_LIMIT_BUNDLE)), + ) + // §5a — anonymous bearer capability, not an account. + .route("/tokens", post(upload::post_token)) + .layer(DefaultBodyLimit::max(SMALL_BODY_LIMIT)); + + Router::new() + .route("/health", get(report::health)) + .route("/ready", get(report::ready)) + .nest("/api/v1", v1) + // §8: a request timeout so a slow bundle query fails fast. + .layer(TimeoutLayer::with_status_code(axum::http::StatusCode::REQUEST_TIMEOUT, timeout)) + .layer(TraceLayer::new_for_http()) + .with_state(state) +} + +/// Convenience for tests and `main`. +pub fn default_timeout() -> Duration { + Duration::from_secs(30) +} diff --git a/src/auth.rs b/src/auth.rs new file mode 100644 index 0000000..f296a26 --- /dev/null +++ b/src/auth.rs @@ -0,0 +1,186 @@ +//! §5a tokens and client-IP attribution. +//! +//! A token is **not an account** — it is an anonymous bearer capability. No +//! email, no verification, no personal data. It is stored only as a hash, so the +//! server cannot enumerate who holds tokens, and its sole purposes are +//! rate-limiting attribution (§5) and revocation. +//! +//! Discarding a token and requesting another is trivially easy, and that is +//! fine: the token is not the defence, the content checks are. Sybil resistance +//! is not required because identity is not load-bearing. + +use std::net::IpAddr; + +use axum::http::HeaderMap; +use sha2::{Digest, Sha256}; + +/// Hashes a bearer token for storage and lookup. +/// +/// Plain SHA-256 rather than a password KDF is deliberate and sufficient here: +/// tokens are 256 bits of server-generated randomness, not user-chosen secrets, +/// so there is no dictionary to attack. +pub fn hash_token(token: &str) -> String { + let mut h = Sha256::new(); + h.update(token.as_bytes()); + hex(&h.finalize()) +} + +/// Hashes a client IP for report attribution (§7 `reports.source_ip_hash`). +/// +/// Salted with the server id so hashes are not comparable across instances. +pub fn hash_ip(ip: &str, server_id: &str) -> String { + let mut h = Sha256::new(); + h.update(server_id.as_bytes()); + h.update(b"\0"); + h.update(ip.as_bytes()); + hex(&h.finalize()) +} + +fn hex(bytes: &[u8]) -> String { + let mut s = String::with_capacity(bytes.len() * 2); + for b in bytes { + s.push_str(&format!("{b:02x}")); + } + s +} + +/// Generates a new token. Returned once to the caller; only its hash is stored. +pub fn generate_token() -> String { + use rand::RngCore; + let mut bytes = [0u8; 32]; + rand::rng().fill_bytes(&mut bytes); + format!("jray_{}", hex(&bytes)) +} + +/// Extracts a bearer token from an `Authorization` header. +pub fn bearer_token(headers: &HeaderMap) -> Option { + let raw = headers.get(axum::http::header::AUTHORIZATION)?.to_str().ok()?; + let (scheme, value) = raw.split_once(' ')?; + if !scheme.eq_ignore_ascii_case("bearer") { + return None; + } + let value = value.trim(); + if value.is_empty() { + return None; + } + Some(value.to_string()) +} + +/// Resolves the client IP for rate-limiting and report attribution. +/// +/// §8: the app must trust `X-Forwarded-For` **only** from the operator's proxy. +/// Rate limiting and report attribution key on client IP, so a spoofable header +/// defeats both — hence `trusted_proxies` is explicit configuration and an +/// untrusted peer's header is ignored outright. +pub fn client_ip(headers: &HeaderMap, peer: Option, trusted_proxies: &[IpAddr]) -> String { + let peer_is_trusted = peer.is_some_and(|p| trusted_proxies.contains(&p)); + + if peer_is_trusted { + if let Some(xff) = headers.get("x-forwarded-for").and_then(|v| v.to_str().ok()) { + // Right-most entry is the one our trusted proxy appended; entries to + // its left are client-supplied and forgeable. Walk from the right + // past any further trusted hops. + for candidate in xff.split(',').rev().map(str::trim).filter(|s| !s.is_empty()) { + match candidate.parse::() { + Ok(ip) if trusted_proxies.contains(&ip) => continue, + Ok(ip) => return ip.to_string(), + Err(_) => break, + } + } + } + } + + peer.map(|p| p.to_string()).unwrap_or_else(|| "unknown".to_string()) +} + +#[cfg(test)] +mod tests { + use super::*; + use axum::http::HeaderValue; + + fn headers(pairs: &[(&'static str, &str)]) -> HeaderMap { + let mut h = HeaderMap::new(); + for (k, v) in pairs { + h.insert(*k, HeaderValue::from_str(v).unwrap()); + } + h + } + + #[test] + fn token_hash_is_stable_and_distinguishing() { + assert_eq!(hash_token("abc"), hash_token("abc")); + assert_ne!(hash_token("abc"), hash_token("abd")); + assert_eq!(hash_token("abc").len(), 64); + } + + #[test] + fn generated_tokens_are_unique_and_prefixed() { + let a = generate_token(); + let b = generate_token(); + assert_ne!(a, b); + assert!(a.starts_with("jray_")); + assert_eq!(a.len(), 5 + 64); + } + + #[test] + fn ip_hash_is_salted_per_server() { + // Hashes must not be comparable across instances. + assert_ne!(hash_ip("1.2.3.4", "a.example"), hash_ip("1.2.3.4", "b.example")); + assert_eq!(hash_ip("1.2.3.4", "a.example"), hash_ip("1.2.3.4", "a.example")); + } + + #[test] + fn parses_bearer_tokens_case_insensitively() { + assert_eq!( + bearer_token(&headers(&[("authorization", "Bearer xyz")])).as_deref(), + Some("xyz") + ); + assert_eq!( + bearer_token(&headers(&[("authorization", "bearer xyz")])).as_deref(), + Some("xyz") + ); + assert!(bearer_token(&headers(&[("authorization", "Basic xyz")])).is_none()); + assert!(bearer_token(&headers(&[("authorization", "Bearer ")])).is_none()); + assert!(bearer_token(&HeaderMap::new()).is_none()); + } + + #[test] + fn forwarded_header_from_an_untrusted_peer_is_ignored() { + // The whole point of §8's explicit trusted-proxy configuration: an + // arbitrary client must not be able to choose its own rate-limit key. + let h = headers(&[("x-forwarded-for", "9.9.9.9")]); + let peer: IpAddr = "203.0.113.7".parse().unwrap(); + assert_eq!(client_ip(&h, Some(peer), &[]), "203.0.113.7"); + } + + #[test] + fn forwarded_header_from_a_trusted_proxy_is_honoured() { + let h = headers(&[("x-forwarded-for", "9.9.9.9")]); + let proxy: IpAddr = "127.0.0.1".parse().unwrap(); + assert_eq!(client_ip(&h, Some(proxy), &[proxy]), "9.9.9.9"); + } + + #[test] + fn client_supplied_entries_left_of_the_proxy_cannot_spoof() { + // A client that sends its own XFF gets its value appended to, not + // replaced, so only the right-most entry is trustworthy. + let h = headers(&[("x-forwarded-for", "9.9.9.9, 203.0.113.7")]); + let proxy: IpAddr = "127.0.0.1".parse().unwrap(); + assert_eq!(client_ip(&h, Some(proxy), &[proxy]), "203.0.113.7"); + } + + #[test] + fn walks_past_additional_trusted_hops() { + let inner: IpAddr = "10.0.0.2".parse().unwrap(); + let proxy: IpAddr = "127.0.0.1".parse().unwrap(); + let h = headers(&[("x-forwarded-for", "203.0.113.7, 10.0.0.2")]); + assert_eq!(client_ip(&h, Some(proxy), &[proxy, inner]), "203.0.113.7"); + } + + #[test] + fn malformed_forwarded_value_falls_back_to_the_peer() { + let h = headers(&[("x-forwarded-for", "not-an-ip")]); + let proxy: IpAddr = "127.0.0.1".parse().unwrap(); + assert_eq!(client_ip(&h, Some(proxy), &[proxy]), "127.0.0.1"); + } +} diff --git a/src/castcheck.rs b/src/castcheck.rs new file mode 100644 index 0000000..581f8b7 --- /dev/null +++ b/src/castcheck.rs @@ -0,0 +1,484 @@ +//! §6 stage 3 cast-match scoring, as pure functions. +//! +//! The thresholds here are the load-bearing part of UR-3 and §5a Threat 2, and +//! §10 (5) wants them retuned against the 331-file extraction corpus. Keeping +//! the decision logic free of I/O is what makes that a test-data exercise rather +//! than a code change. + +use crate::tmdb::CastMember; + +/// §6: thresholds over the ratio `|M ∩ C| / |M|`. +pub const LISTED_THRESHOLD: f64 = 0.6; +pub const FLAGGED_THRESHOLD: f64 = 0.3; +/// Below this size a ratio is meaningless (§6 small-|M| handling). +pub const SMALL_M_LIMIT: usize = 5; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Verdict { + /// Normal case. + Listed, + /// Served with reduced ranking, flagged for review. + Flagged, + /// Deleted, and the contributor's counter bumped. + Rejected, +} + +impl Verdict { + pub fn status(self) -> &'static str { + match self { + Verdict::Listed => "listed", + Verdict::Flagged => "flagged", + Verdict::Rejected => "rejected", + } + } +} + +#[derive(Debug, Clone)] +pub struct CastCheckOutcome { + pub verdict: Verdict, + pub ratio: f64, + /// Manifest actors resolved to a TMDB person id, with the TMDB-authoritative + /// name. Only these are kept; §6 drops unmatched actors rather than storing + /// them, which is what closes §5a's free-text channel. + pub matched: Vec, + /// Actors that matched nothing and will be dropped. + pub unmatched_person_ids: Vec, + pub reason: Option, +} + +#[derive(Debug, Clone)] +pub struct MatchedActor { + pub tmdb_person_id: u64, + /// From TMDB, never from the upload. + pub name: String, + pub adult: bool, + /// True when the match came from name comparison rather than an id. + pub by_name: bool, +} + +/// An actor as submitted, after §6 stage 2 validation. +#[derive(Debug, Clone)] +pub struct SubmittedActor { + pub tmdb_id: Option, + pub imdb_id: Option, + /// Used only for matching here, then discarded (§5a). + pub name: Option, +} + +/// Case- and accent-insensitive comparison key for the name fallback (§6). +fn name_key(s: &str) -> String { + use unicode_normalization::UnicodeNormalization; + s.nfd() + .filter(|c| !unicode_normalization::char::is_combining_mark(*c)) + .flat_map(|c| c.to_lowercase()) + .filter(|c| !c.is_whitespace() && *c != '.' && *c != ',' && *c != '-' && *c != '\'') + .collect() +} + +/// Runs the §6 stage 3 comparison. +/// +/// `credits` is the reference set *C*: for a movie, its credits; for an episode, +/// the union of per-episode credits and the series' aggregate credits. +pub fn evaluate(submitted: &[SubmittedActor], credits: &[CastMember]) -> CastCheckOutcome { + let m = submitted.len(); + + // §6: `|M| == 0` is rejected. These are extraction failures, not + // contributions — validation already refuses them, so reaching here means a + // manifest lost every actor upstream. + if m == 0 { + return CastCheckOutcome { + verdict: Verdict::Rejected, + ratio: 0.0, + matched: Vec::new(), + unmatched_person_ids: Vec::new(), + reason: Some("empty_actor_list".into()), + }; + } + + // TMDB has no credits for the id: absent data is not evidence of a bad + // manifest, so this is flagged rather than rejected (§6). + if credits.is_empty() { + return CastCheckOutcome { + verdict: Verdict::Flagged, + ratio: 0.0, + matched: Vec::new(), + unmatched_person_ids: submitted.iter().filter_map(|a| a.tmdb_id).collect(), + reason: Some("tmdb_no_credits".into()), + }; + } + + let mut matched: Vec = Vec::new(); + let mut unmatched: Vec = Vec::new(); + let mut id_matches = 0usize; + let mut name_matches = 0usize; + + for actor in submitted { + // Join on `tmdb_id` — grounded in the pipeline's actual output, where + // 330 of 331 manifests have `imdb_id: ""` and `tmdb_id` set (§6). + let by_id = actor.tmdb_id.and_then(|id| credits.iter().find(|c| c.id == id)); + + if let Some(c) = by_id { + id_matches += 1; + push_unique( + &mut matched, + MatchedActor { + tmdb_person_id: c.id, + name: c.name.clone(), + adult: c.adult, + by_name: false, + }, + ); + continue; + } + + // Fall back to case- and accent-insensitive name comparison. + let by_name = actor.name.as_deref().and_then(|n| { + let key = name_key(n); + (!key.is_empty()).then(|| credits.iter().find(|c| name_key(&c.name) == key))? + }); + + if let Some(c) = by_name { + name_matches += 1; + push_unique( + &mut matched, + MatchedActor { + tmdb_person_id: c.id, + name: c.name.clone(), + adult: c.adult, + by_name: true, + }, + ); + continue; + } + + if let Some(id) = actor.tmdb_id { + unmatched.push(id); + } + } + + // §6: name-only matches are counted but capped at half the intersection, so + // a manifest cannot pass on name collisions alone. + let capped_name_matches = name_matches.min(id_matches); + let effective = id_matches + capped_name_matches; + let ratio = effective as f64 / m as f64; + + let verdict = classify(m, effective, ratio); + let reason = match verdict { + Verdict::Rejected => Some("cast_match_below_threshold".into()), + Verdict::Flagged => Some("cast_match_marginal".into()), + Verdict::Listed => None, + }; + + CastCheckOutcome { verdict, ratio, matched, unmatched_person_ids: unmatched, reason } +} + +fn push_unique(matched: &mut Vec, actor: MatchedActor) { + if !matched.iter().any(|m| m.tmdb_person_id == actor.tmdb_person_id) { + matched.push(actor); + } +} + +/// §6 small-*M* handling. With a median of 7 actors a ratio threshold is coarse +/// — one mismatch moves it by 14% — so small manifests use counts, not ratios. +fn classify(m: usize, matches: usize, ratio: f64) -> Verdict { + if m >= SMALL_M_LIMIT { + if ratio >= LISTED_THRESHOLD { + Verdict::Listed + } else if ratio >= FLAGGED_THRESHOLD { + Verdict::Flagged + } else { + Verdict::Rejected + } + } else if m >= 2 { + // Require all but one actor to match. + if matches + 1 >= m { + Verdict::Listed + } else { + Verdict::Rejected + } + } else { + // |M| <= 1: accept only if the single actor matches. Such a manifest is + // near-worthless anyway and is ranked last. + if matches >= 1 { + Verdict::Listed + } else { + Verdict::Rejected + } + } +} + +/// §5a additional layer 1 — category guard. +/// +/// Rejects when a matched person is flagged adult by TMDB and the target title +/// is not, which targets the stated prank without needing a blocklist of names. +pub fn category_guard_violation(matched: &[MatchedActor], title_is_adult: bool) -> Option { + if title_is_adult { + return None; + } + matched.iter().find(|m| m.adult).map(|m| m.tmdb_person_id) +} + +/// §5a additional layer 2 — age-appropriateness guard. +/// +/// On a children's certification, apply the strictest cast-match threshold and +/// require an `exact` or `runtime` cut match. Mismatched content on children's +/// titles is the highest-harm case and deserves the tightest gate. +pub fn is_childrens_certification(cert: &str) -> bool { + matches!( + cert.trim().to_ascii_uppercase().as_str(), + "G" | "TV-Y" | "TV-Y7" | "TV-G" | "U" | "0+" | "6+" | "PG" | "TV-PG" + ) +} + +pub const CHILDRENS_LISTED_THRESHOLD: f64 = 0.8; + +/// Applies the children's-title gate to an already-computed outcome. +pub fn apply_childrens_guard(outcome: &mut CastCheckOutcome, m: usize) { + if m >= SMALL_M_LIMIT && outcome.ratio < CHILDRENS_LISTED_THRESHOLD { + outcome.verdict = match outcome.verdict { + Verdict::Listed => Verdict::Flagged, + other => other, + }; + if outcome.reason.is_none() { + outcome.reason = Some("childrens_title_strict_threshold".into()); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn credit(id: u64, name: &str) -> CastMember { + CastMember { id, name: name.to_string(), adult: false } + } + + fn adult_credit(id: u64, name: &str) -> CastMember { + CastMember { id, name: name.to_string(), adult: true } + } + + fn by_id(id: u64) -> SubmittedActor { + SubmittedActor { tmdb_id: Some(id), imdb_id: None, name: None } + } + + fn by_name(name: &str) -> SubmittedActor { + SubmittedActor { tmdb_id: None, imdb_id: None, name: Some(name.to_string()) } + } + + /// A realistic reference cast — feature casts are several times larger than + /// the manifests extracted from them (§6). + fn cast_of_20() -> Vec { + (1..=20).map(|i| credit(i, &format!("Actor {i}"))).collect() + } + + #[test] + fn full_subset_of_the_cast_is_listed() { + // §6: the ratio is over *M*, not *C* — a manifest legitimately contains + // only actors both credited and detected on screen, so penalising it for + // missing credited actors would fail every honest upload. + let submitted: Vec<_> = (1..=7).map(by_id).collect(); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Listed); + assert_eq!(out.ratio, 1.0); + assert_eq!(out.matched.len(), 7); + } + + #[test] + fn threshold_boundaries_at_point_six_and_point_three() { + // 6 of 10 matching == 0.6 exactly: listed. + let mut submitted: Vec<_> = (1..=6).map(by_id).collect(); + submitted.extend((900..904).map(by_id)); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.matched.len(), 6); + assert!((out.ratio - 0.6).abs() < 1e-9); + assert_eq!(out.verdict, Verdict::Listed); + + // 5 of 10 == 0.5: flagged, served with reduced ranking. + let mut submitted: Vec<_> = (1..=5).map(by_id).collect(); + submitted.extend((900..905).map(by_id)); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Flagged); + + // 3 of 10 == 0.3 exactly: still flagged, not rejected. + let mut submitted: Vec<_> = (1..=3).map(by_id).collect(); + submitted.extend((900..907).map(by_id)); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Flagged); + + // 2 of 10 == 0.2: rejected. + let mut submitted: Vec<_> = (1..=2).map(by_id).collect(); + submitted.extend((900..908).map(by_id)); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Rejected); + } + + #[test] + fn prank_manifest_is_rejected() { + // §5a Threat 2: performers who are not credited cast on the title. + let submitted: Vec<_> = (500..510).map(by_id).collect(); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Rejected); + assert_eq!(out.ratio, 0.0); + assert_eq!(out.reason.as_deref(), Some("cast_match_below_threshold")); + } + + #[test] + fn small_m_requires_all_but_one_to_match() { + // §6: `2 <= |M| < 5` — a ratio is meaningless at this size. + let out = evaluate(&[by_id(1), by_id(2), by_id(3), by_id(999)], &cast_of_20()); + assert_eq!(out.verdict, Verdict::Listed, "3 of 4 is all-but-one"); + + let out = evaluate(&[by_id(1), by_id(2), by_id(998), by_id(999)], &cast_of_20()); + assert_eq!(out.verdict, Verdict::Rejected, "2 of 4 fails all-but-one"); + + // 0.5 would be `Flagged` under the ratio table, so this proves the + // small-|M| branch is actually taken. + let out = evaluate(&[by_id(1), by_id(999)], &cast_of_20()); + assert_eq!(out.verdict, Verdict::Listed, "1 of 2 is all-but-one"); + } + + #[test] + fn single_actor_manifest_needs_that_actor_to_match() { + assert_eq!(evaluate(&[by_id(1)], &cast_of_20()).verdict, Verdict::Listed); + assert_eq!(evaluate(&[by_id(999)], &cast_of_20()).verdict, Verdict::Rejected); + } + + #[test] + fn empty_manifest_is_rejected() { + let out = evaluate(&[], &cast_of_20()); + assert_eq!(out.verdict, Verdict::Rejected); + assert_eq!(out.reason.as_deref(), Some("empty_actor_list")); + } + + #[test] + fn missing_tmdb_credits_flags_rather_than_rejects() { + // §6: absent data is not evidence of a bad manifest. + let submitted: Vec<_> = (1..=7).map(by_id).collect(); + let out = evaluate(&submitted, &[]); + assert_eq!(out.verdict, Verdict::Flagged); + assert_eq!(out.reason.as_deref(), Some("tmdb_no_credits")); + } + + #[test] + fn name_matching_is_case_and_accent_insensitive() { + let credits = vec![credit(1, "Renée Zellweger"), credit(2, "Miloš Forman")]; + let out = evaluate(&[by_name("renee zellweger"), by_name("MILOS FORMAN")], &credits); + assert_eq!(out.matched.len(), 2); + } + + #[test] + fn name_only_matches_cannot_carry_a_manifest_alone() { + // §6: name-only matches are capped at half the intersection, so a + // manifest cannot pass on name collisions alone. + let credits: Vec<_> = (1..=20).map(|i| credit(i, &format!("Actor {i}"))).collect(); + let submitted: Vec<_> = (1..=10).map(|i| by_name(&format!("Actor {i}"))).collect(); + let out = evaluate(&submitted, &credits); + assert_eq!(out.ratio, 0.0, "with no id matches, name matches cap to zero"); + assert_eq!(out.verdict, Verdict::Rejected); + } + + #[test] + fn name_matches_count_up_to_the_number_of_id_matches() { + let credits: Vec<_> = (1..=20).map(|i| credit(i, &format!("Actor {i}"))).collect(); + // 4 by id + 6 by name, of 10 => capped to 4 + 4 = 8 => 0.8. + let mut submitted: Vec<_> = (1..=4).map(by_id).collect(); + submitted.extend((5..=10).map(|i| by_name(&format!("Actor {i}")))); + let out = evaluate(&submitted, &credits); + assert!((out.ratio - 0.8).abs() < 1e-9, "got {}", out.ratio); + assert_eq!(out.verdict, Verdict::Listed); + } + + #[test] + fn unmatched_actors_are_reported_for_dropping() { + // §6: unmatched actors are dropped rather than stored. + let submitted: Vec<_> = (1..=6).map(by_id).chain([by_id(777)]).collect(); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.unmatched_person_ids, vec![777]); + assert!(out.matched.iter().all(|m| m.tmdb_person_id != 777)); + } + + #[test] + fn matched_names_come_from_tmdb_not_the_upload() { + // §5a: the server stores references to TMDB entities, not + // attacker-authored text. + let credits = vec![credit(884, "Steve Buscemi")]; + let submitted = vec![SubmittedActor { + tmdb_id: Some(884), + imdb_id: None, + name: Some("Definitely Not Him".into()), + }]; + let out = evaluate(&submitted, &credits); + assert_eq!(out.matched[0].name, "Steve Buscemi"); + } + + #[test] + fn duplicate_credits_do_not_double_count() { + // TMDB aggregate credits can list a person more than once. + let credits = vec![credit(1, "A"), credit(1, "A")]; + let out = evaluate(&[by_id(1)], &credits); + assert_eq!(out.matched.len(), 1); + } + + #[test] + fn category_guard_catches_adult_performers_on_a_non_adult_title() { + // §5a layer 1, aimed squarely at the stated prank. + let matched = vec![ + MatchedActor { tmdb_person_id: 1, name: "A".into(), adult: false, by_name: false }, + MatchedActor { tmdb_person_id: 2, name: "B".into(), adult: true, by_name: false }, + ]; + assert_eq!(category_guard_violation(&matched, false), Some(2)); + // Unless the target title is itself flagged adult. + assert_eq!(category_guard_violation(&matched, true), None); + } + + #[test] + fn category_guard_ignores_clean_casts() { + let matched = vec![MatchedActor { + tmdb_person_id: 1, + name: "A".into(), + adult: false, + by_name: false, + }]; + assert_eq!(category_guard_violation(&matched, false), None); + } + + #[test] + fn adult_credit_is_carried_through_matching() { + let out = evaluate(&[by_id(9)], &[adult_credit(9, "X")]); + assert!(out.matched[0].adult); + } + + #[test] + fn childrens_certifications_are_recognised() { + for c in ["G", "TV-Y", "tv-y7", "U", " PG "] { + assert!(is_childrens_certification(c), "{c} should be a children's rating"); + } + for c in ["R", "NC-17", "TV-MA", "18", ""] { + assert!(!is_childrens_certification(c), "{c} should not be"); + } + } + + #[test] + fn childrens_guard_tightens_the_threshold() { + // §5a layer 2: the highest-harm case gets the tightest gate. A ratio of + // 0.7 lists normally but only reaches `flagged` on a children's title. + let mut submitted: Vec<_> = (1..=7).map(by_id).collect(); + submitted.extend((900..903).map(by_id)); + let mut out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Listed); + + let m = submitted.len(); + apply_childrens_guard(&mut out, m); + assert_eq!(out.verdict, Verdict::Flagged); + assert_eq!(out.reason.as_deref(), Some("childrens_title_strict_threshold")); + } + + #[test] + fn childrens_guard_leaves_strong_matches_listed() { + let submitted: Vec<_> = (1..=10).map(by_id).collect(); + let mut out = evaluate(&submitted, &cast_of_20()); + let m = submitted.len(); + apply_childrens_guard(&mut out, m); + assert_eq!(out.verdict, Verdict::Listed); + } +} diff --git a/src/config.rs b/src/config.rs new file mode 100644 index 0000000..e6a1042 --- /dev/null +++ b/src/config.rs @@ -0,0 +1,62 @@ +//! Operational configuration, read from the environment. +//! +//! The trusted-proxy CIDR is explicit configuration rather than a default-on +//! behaviour (§8 deployment notes): §5 rate limiting and report attribution key +//! on client IP, so an unconditionally-trusted `X-Forwarded-For` defeats both. + +use std::net::IpAddr; +use std::time::Duration; + +#[derive(Clone, Debug)] +pub struct Config { + pub bind: String, + pub db_path: String, + /// Hard dependency for UR-3. Without it, uploads accumulate in `pending` + /// rather than being listed unverified (§8). + pub tmdb_api_key: Option, + pub tmdb_base_url: String, + /// Prefixes of proxy addresses whose `X-Forwarded-For` is honoured. + pub trusted_proxies: Vec, + pub server_id: String, + pub request_timeout: Duration, + /// Number of cast-check jobs to lease per worker tick. + pub job_batch: usize, + pub job_poll_interval: Duration, +} + +impl Config { + pub fn from_env() -> anyhow::Result { + let trusted_proxies = match std::env::var("JRAY_TRUSTED_PROXIES") { + Ok(v) => v + .split(',') + .map(str::trim) + .filter(|s| !s.is_empty()) + .map(|s| { + s.parse::() + .map_err(|e| anyhow::anyhow!("bad JRAY_TRUSTED_PROXIES entry {s:?}: {e}")) + }) + .collect::, _>>()?, + Err(_) => Vec::new(), + }; + + Ok(Self { + bind: env_or("JRAY_BIND", "127.0.0.1:8080"), + db_path: env_or("JRAY_DB", "jray.db"), + tmdb_api_key: std::env::var("JRAY_TMDB_API_KEY").ok().filter(|s| !s.is_empty()), + tmdb_base_url: env_or("JRAY_TMDB_BASE_URL", "https://api.themoviedb.org/3"), + trusted_proxies, + server_id: env_or("JRAY_SERVER_ID", "localhost"), + request_timeout: Duration::from_secs(env_num("JRAY_REQUEST_TIMEOUT_SEC", 30)), + job_batch: env_num("JRAY_JOB_BATCH", 8) as usize, + job_poll_interval: Duration::from_secs(env_num("JRAY_JOB_POLL_SEC", 5)), + }) + } +} + +fn env_or(key: &str, default: &str) -> String { + std::env::var(key).ok().filter(|s| !s.is_empty()).unwrap_or_else(|| default.to_string()) +} + +fn env_num(key: &str, default: u64) -> u64 { + std::env::var(key).ok().and_then(|v| v.parse().ok()).unwrap_or(default) +} diff --git a/src/content_id.rs b/src/content_id.rs new file mode 100644 index 0000000..44ce29d --- /dev/null +++ b/src/content_id.rs @@ -0,0 +1,289 @@ +//! §9a content addressing. +//! +//! A validated manifest is immutable and content-addressable, which is what +//! makes replication *set reconciliation* rather than state synchronisation. +//! Even without the federation endpoints, computing `content_id` on upload gives +//! deduplication now and means stored manifests are already addressable when +//! federation lands. +//! +//! **This canonical form must be reimplemented byte-identically by the JRay +//! plugin** (§8 "Cost of choosing Rust": the extraction side is Python, so this +//! can no longer be shared as one implementation and must instead be specified +//! precisely and cross-tested). [`GOLDEN_VECTORS`] is that shared fixture. + +use sha2::{Digest, Sha256}; + +/// One actor's contribution to the canonical form. +#[derive(Debug, Clone)] +pub struct CanonicalActor { + pub tmdb_person_id: u64, + /// Integer centiseconds — quantised, not formatted floats (§9a). + pub scenes_cs: Vec<(i64, i64)>, +} + +/// The identity coordinates that enter the hash. +#[derive(Debug, Clone, Default)] +pub struct CanonicalIdentity { + pub kind: &'static str, + pub tmdb_id: Option, + pub imdb_id: Option, + pub season: Option, + pub episode: Option, +} + +/// The cut coordinates that enter the hash. +/// +/// **`audio_signature` is excluded, deliberately** (§9a): it is derived by +/// decoding audio, so two servers running different FFmpeg or resampler versions +/// could compute marginally different signatures for identical content, and +/// including it would silently break federation deduplication. +#[derive(Debug, Clone, Default)] +pub struct CanonicalCut { + /// Quantised to centiseconds for the same reason scene times are. + pub runtime_cs: i64, + pub video_hash: Option, +} + +/// Builds the canonical JSON form: keys sorted, no whitespace, actors sorted by +/// person id, scene times as integer centiseconds. +/// +/// `extraction` metadata and all local state are excluded, so two servers that +/// validated the same upload independently arrive at the same `content_id`. +pub fn canonical_json( + identity: &CanonicalIdentity, + cut: &CanonicalCut, + actors: &[CanonicalActor], +) -> String { + let mut sorted: Vec<&CanonicalActor> = actors.iter().collect(); + sorted.sort_by_key(|a| a.tmdb_person_id); + + let mut s = String::new(); + s.push_str("{\"actors\":["); + for (i, a) in sorted.iter().enumerate() { + if i > 0 { + s.push(','); + } + // Scene windows are emitted in stored order; validation has already + // established they are sorted by start time. + s.push_str("{\"scenes\":["); + for (j, (start, end)) in a.scenes_cs.iter().enumerate() { + if j > 0 { + s.push(','); + } + s.push('['); + s.push_str(&start.to_string()); + s.push(','); + s.push_str(&end.to_string()); + s.push(']'); + } + s.push_str("],\"tmdb_person_id\":"); + s.push_str(&a.tmdb_person_id.to_string()); + s.push('}'); + } + s.push_str("],\"cut\":{"); + s.push_str("\"runtime_cs\":"); + s.push_str(&cut.runtime_cs.to_string()); + s.push_str(",\"video_hash\":"); + push_opt_str(&mut s, cut.video_hash.as_deref()); + s.push_str("},\"identity\":{"); + s.push_str("\"episode\":"); + push_opt_num(&mut s, cut_opt(identity.episode)); + s.push_str(",\"imdb_id\":"); + push_opt_str(&mut s, identity.imdb_id.as_deref()); + s.push_str(",\"season\":"); + push_opt_num(&mut s, cut_opt(identity.season)); + s.push_str(",\"tmdb_id\":"); + push_opt_str(&mut s, identity.tmdb_id.as_deref()); + s.push_str(",\"type\":\""); + s.push_str(identity.kind); + s.push_str("\"}}"); + s +} + +fn cut_opt(v: Option) -> Option { + v +} + +fn push_opt_str(s: &mut String, v: Option<&str>) { + match v { + // Only closed-vocabulary values reach here (regex-constrained ids and a + // fixed-format hash), so no string escaping is required. + Some(v) => { + s.push('"'); + s.push_str(v); + s.push('"'); + } + None => s.push_str("null"), + } +} + +fn push_opt_num(s: &mut String, v: Option) { + match v { + Some(v) => s.push_str(&v.to_string()), + None => s.push_str("null"), + } +} + +/// `sha256:` over the canonical form (§9a). +pub fn content_id( + identity: &CanonicalIdentity, + cut: &CanonicalCut, + actors: &[CanonicalActor], +) -> String { + let canonical = canonical_json(identity, cut, actors); + let mut h = Sha256::new(); + h.update(canonical.as_bytes()); + let digest = h.finalize(); + let mut hex = String::with_capacity(64 + 7); + hex.push_str("sha256:"); + for b in digest { + hex.push_str(&format!("{b:02x}")); + } + hex +} + +/// Cross-implementation fixture (§8): the JRay plugin and any reimplementation +/// must reproduce these exactly, or federation deduplication silently breaks. +pub const GOLDEN_VECTORS: &[(&str, &str)] = &[( + // Movie, one actor, two windows, with a video hash. + r#"{"actors":[{"scenes":[[19160,20920],[43820,46560]],"tmdb_person_id":884}],"cut":{"runtime_cs":642050,"video_hash":"opensubtitles:8e245d9679d31e12"},"identity":{"episode":null,"imdb_id":"tt4686844","season":null,"tmdb_id":"504172","type":"movie"}}"#, + // Verified against an independent Python implementation: + // sha256(canonical.encode()).hexdigest() + "sha256:367f8b05c54a992a3a30fa016edaaac0b9b36b148fc76574f5b1ef326b56760f", +)]; + +#[cfg(test)] +mod tests { + use super::*; + + fn movie_identity() -> CanonicalIdentity { + CanonicalIdentity { + kind: "movie", + tmdb_id: Some("504172".into()), + imdb_id: Some("tt4686844".into()), + season: None, + episode: None, + } + } + + fn movie_cut() -> CanonicalCut { + CanonicalCut { + runtime_cs: 642050, + video_hash: Some("opensubtitles:8e245d9679d31e12".into()), + } + } + + fn actors() -> Vec { + vec![CanonicalActor { + tmdb_person_id: 884, + scenes_cs: vec![(19160, 20920), (43820, 46560)], + }] + } + + #[test] + fn canonical_form_matches_the_documented_shape() { + let json = canonical_json(&movie_identity(), &movie_cut(), &actors()); + assert_eq!(json, GOLDEN_VECTORS[0].0); + // Keys sorted, no whitespace (§9a). + assert!(!json.contains(' ')); + } + + #[test] + fn canonical_form_is_valid_json_with_sorted_keys() { + // Hand-built strings are easy to get subtly wrong, so assert the output + // actually parses and that its keys really are ordered. + let json = canonical_json(&movie_identity(), &movie_cut(), &actors()); + let v: serde_json::Value = + serde_json::from_str(&json).expect("canonical form must be JSON"); + let obj = v.as_object().unwrap(); + let keys: Vec<&String> = obj.keys().collect(); + assert_eq!(keys, vec!["actors", "cut", "identity"]); + let id_keys: Vec<&String> = v["identity"].as_object().unwrap().keys().collect(); + assert_eq!(id_keys, vec!["episode", "imdb_id", "season", "tmdb_id", "type"]); + let cut_keys: Vec<&String> = v["cut"].as_object().unwrap().keys().collect(); + assert_eq!(cut_keys, vec!["runtime_cs", "video_hash"]); + } + + #[test] + fn actor_order_does_not_affect_the_hash() { + // §9a: actors sorted by person id, so two servers that stored them in + // different orders still agree. + let a = vec![ + CanonicalActor { tmdb_person_id: 884, scenes_cs: vec![(0, 100)] }, + CanonicalActor { tmdb_person_id: 17419, scenes_cs: vec![(200, 300)] }, + ]; + let b = vec![a[1].clone(), a[0].clone()]; + assert_eq!( + content_id(&movie_identity(), &movie_cut(), &a), + content_id(&movie_identity(), &movie_cut(), &b) + ); + } + + #[test] + fn accumulated_float_error_hashes_identically() { + // The failure mode §9a exists to remove: real corpus values look like + // 8045.066666660665, and two servers may compute them slightly + // differently. Quantising first means both hash the same. + let a = vec![CanonicalActor { + tmdb_person_id: 1, + scenes_cs: vec![(crate::validate::to_centiseconds(8045.066666660665), 900000)], + }]; + let b = vec![CanonicalActor { + tmdb_person_id: 1, + scenes_cs: vec![(crate::validate::to_centiseconds(8045.066666666), 900000)], + }]; + assert_eq!( + content_id(&movie_identity(), &movie_cut(), &a), + content_id(&movie_identity(), &movie_cut(), &b) + ); + } + + #[test] + fn differing_content_produces_differing_ids() { + let base = content_id(&movie_identity(), &movie_cut(), &actors()); + + let mut other_actors = actors(); + other_actors[0].scenes_cs[0].1 += 1; + assert_ne!(base, content_id(&movie_identity(), &movie_cut(), &other_actors)); + + let mut other_cut = movie_cut(); + other_cut.runtime_cs += 1; + assert_ne!(base, content_id(&movie_identity(), &other_cut, &actors())); + + let mut other_id = movie_identity(); + other_id.tmdb_id = Some("999".into()); + assert_ne!(base, content_id(&other_id, &movie_cut(), &actors())); + } + + #[test] + fn episode_and_movie_coordinates_are_distinguished() { + let ep = CanonicalIdentity { + kind: "episode", + tmdb_id: Some("1396".into()), + imdb_id: None, + season: Some(2), + episode: Some(5), + }; + let other = CanonicalIdentity { season: Some(3), ..ep.clone() }; + assert_ne!( + content_id(&ep, &movie_cut(), &actors()), + content_id(&other, &movie_cut(), &actors()) + ); + } + + #[test] + fn content_id_is_prefixed_and_hex() { + let id = content_id(&movie_identity(), &movie_cut(), &actors()); + let hex = id.strip_prefix("sha256:").expect("prefixed"); + assert_eq!(hex.len(), 64); + assert!(hex.bytes().all(|b| b.is_ascii_hexdigit())); + } + + #[test] + fn golden_vector_hash_is_stable() { + // Locks the hash so an accidental change to the canonical form is caught + // here rather than by silent federation divergence. + let id = content_id(&movie_identity(), &movie_cut(), &actors()); + assert_eq!(id, GOLDEN_VECTORS[0].1, "canonical form or hash changed"); + } +} diff --git a/src/db/mod.rs b/src/db/mod.rs new file mode 100644 index 0000000..8806e97 --- /dev/null +++ b/src/db/mod.rs @@ -0,0 +1,168 @@ +//! Database access. +//! +//! §8 imposes two structural requirements that this module exists to satisfy: +//! +//! 1. **A single writer connection, serialized through one owner**, with a read +//! pool alongside. SQLite permits only one writer at a time even in WAL mode; +//! pointing a multi-connection pool at writes and relying on `busy_timeout` +//! to sort it out is explicitly rejected by the spec. Here the writer lives +//! behind a `Mutex`, so contention queues in Rust rather than surfacing as +//! `SQLITE_BUSY`. +//! 2. **All access behind a thin repository layer** rather than queries +//! scattered through handlers — this is what keeps the Turso/Postgres options +//! cheap and localises the serialization in one place. +//! +//! rusqlite is synchronous, so every call is wrapped in `spawn_blocking`: a +//! write that waits on the mutex must never block a Tokio worker thread. + +pub mod repo; + +use std::sync::{Arc, Mutex}; + +use anyhow::Context; +use rusqlite::Connection; + +const SCHEMA: &str = include_str!("schema.sql"); + +/// Handle to the database: one serialized writer, plus read connections. +/// +/// Cloning is cheap and shares the same underlying connections. +#[derive(Clone)] +pub struct Db { + writer: Arc>, + readers: Arc, +} + +struct ReadPool { + conns: Mutex>, + path: String, +} + +impl ReadPool { + fn acquire(&self) -> anyhow::Result { + if let Some(c) = self.conns.lock().expect("read pool poisoned").pop() { + return Ok(c); + } + open_conn(&self.path, false) + } + + fn release(&self, conn: Connection) { + let mut conns = self.conns.lock().expect("read pool poisoned"); + // Bounded: excess connections are dropped rather than accumulating. + if conns.len() < 8 { + conns.push(conn); + } + } +} + +fn open_conn(path: &str, writer: bool) -> anyhow::Result { + let conn = Connection::open(path).with_context(|| format!("opening database {path}"))?; + + // WAL gives concurrent readers alongside the single writer, which suits a + // read-dominated workload; `synchronous = NORMAL` is safe under WAL, and + // `busy_timeout` makes contention wait rather than error (§8). + conn.pragma_update(None, "journal_mode", "WAL")?; + conn.pragma_update(None, "synchronous", "NORMAL")?; + conn.pragma_update(None, "busy_timeout", 5_000)?; + conn.pragma_update(None, "foreign_keys", true)?; + if !writer { + conn.pragma_update(None, "query_only", true)?; + } + Ok(conn) +} + +impl Db { + /// Opens the database, applying the schema. Idempotent — every statement in + /// `schema.sql` is `IF NOT EXISTS`. + pub fn open(path: &str) -> anyhow::Result { + let writer = open_conn(path, true)?; + writer.execute_batch(SCHEMA).context("applying schema")?; + + Ok(Self { + writer: Arc::new(Mutex::new(writer)), + readers: Arc::new(ReadPool { conns: Mutex::new(Vec::new()), path: path.to_string() }), + }) + } + + /// Runs `f` against the serialized writer connection on a blocking thread. + /// + /// `f` receives a `Transaction`, so a manifest's scene rows go in as one + /// transaction rather than one per row (§8), and a failure rolls back. + pub async fn write(&self, f: F) -> anyhow::Result + where + T: Send + 'static, + F: FnOnce(&rusqlite::Transaction<'_>) -> anyhow::Result + Send + 'static, + { + let writer = self.writer.clone(); + tokio::task::spawn_blocking(move || { + let mut conn = writer.lock().expect("writer poisoned"); + let tx = conn.transaction()?; + let out = f(&tx)?; + tx.commit()?; + Ok(out) + }) + .await + .context("writer task panicked")? + } + + /// Runs `f` against a read connection on a blocking thread. + pub async fn read(&self, f: F) -> anyhow::Result + where + T: Send + 'static, + F: FnOnce(&Connection) -> anyhow::Result + Send + 'static, + { + let readers = self.readers.clone(); + tokio::task::spawn_blocking(move || { + let conn = readers.acquire()?; + let out = f(&conn); + readers.release(conn); + out + }) + .await + .context("reader task panicked")? + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn schema_applies_and_roundtrips() { + let db = Db::open(":memory:").unwrap(); + // In-memory databases are per-connection, so only exercise the writer. + let n = db + .write(|tx| { + tx.execute( + "INSERT INTO contributors (id, token_hash, created_at) VALUES (?1, ?2, ?3)", + rusqlite::params!["c1", "hash", "2026-01-01T00:00:00Z"], + )?; + Ok(tx.query_row("SELECT COUNT(*) FROM contributors", [], |r| r.get::<_, i64>(0))?) + }) + .await + .unwrap(); + assert_eq!(n, 1); + } + + #[tokio::test] + async fn write_rolls_back_on_error() { + let db = Db::open(":memory:").unwrap(); + let res: anyhow::Result<()> = db + .write(|tx| { + tx.execute( + "INSERT INTO contributors (id, token_hash, created_at) VALUES ('c1','h','t')", + [], + )?; + anyhow::bail!("deliberate failure") + }) + .await; + assert!(res.is_err()); + let n = db + .write(|tx| { + Ok(tx.query_row("SELECT COUNT(*) FROM contributors", [], |r| r.get::<_, i64>(0))?) + }) + .await + .unwrap(); + assert_eq!(n, 0, "failed transaction must not persist rows"); + } +} diff --git a/src/db/repo.rs b/src/db/repo.rs new file mode 100644 index 0000000..9621928 --- /dev/null +++ b/src/db/repo.rs @@ -0,0 +1,1110 @@ +//! Repository queries. +//! +//! §8: all database access lives behind this layer rather than being scattered +//! through handlers — that is what keeps the Turso/Postgres options cheap and +//! localises the single-writer serialization in one place. +//! +//! Functions here take `&Connection` or `&Transaction` and are synchronous; the +//! `spawn_blocking` boundary is [`crate::db::Db`]'s concern. + +use anyhow::Context; +use rusqlite::{params, Connection, OptionalExtension, Transaction}; + +use crate::model::IdentityType; + +/// A stored manifest row, as needed to serve reads. +#[derive(Debug, Clone)] +pub struct ManifestRow { + pub id: String, + pub title_id: String, + pub season: Option, + pub episode: Option, + pub runtime_sec: f64, + pub video_hash: Option, + pub audio_signature: Option>, + pub sample_fps: Option, + pub extinction_sec: Option, + pub pipeline_version: Option, + pub gallery_scope: Option, + pub status: String, + pub cast_match_ratio: Option, + pub content_id: Option, +} + +const MANIFEST_COLUMNS: &str = "id, title_id, season, episode, runtime_sec, video_hash, \ + audio_signature, sample_fps, extinction_sec, pipeline_version, gallery_scope, \ + status, cast_match_ratio, \ + content_id"; + +fn map_manifest(row: &rusqlite::Row<'_>) -> rusqlite::Result { + Ok(ManifestRow { + id: row.get(0)?, + title_id: row.get(1)?, + season: row.get(2)?, + episode: row.get(3)?, + runtime_sec: row.get(4)?, + video_hash: row.get(5)?, + audio_signature: row.get(6)?, + sample_fps: row.get(7)?, + extinction_sec: row.get(8)?, + pipeline_version: row.get(9)?, + gallery_scope: row.get(10)?, + status: row.get(11)?, + cast_match_ratio: row.get(12)?, + content_id: row.get(13)?, + }) +} + +/// Statuses that are served to clients (§7). +pub const SERVED_STATUSES: &str = "('listed','flagged')"; + +// --------------------------------------------------------------------------- +// Titles +// --------------------------------------------------------------------------- + +#[derive(Debug, Clone)] +pub struct TitleRow { + pub id: String, + pub kind: String, + pub tmdb_id: Option, + pub imdb_id: Option, + pub name: Option, + pub year: Option, + pub adult: bool, + pub certification: Option, +} + +pub fn find_title( + conn: &Connection, + kind: IdentityType, + tmdb_id: Option<&str>, + imdb_id: Option<&str>, +) -> anyhow::Result> { + let kind_str = title_kind(kind); + let mut stmt = conn.prepare_cached( + "SELECT id, kind, tmdb_id, imdb_id, name, year, adult, certification + FROM titles + WHERE kind = ?1 + AND ( (?2 IS NOT NULL AND tmdb_id = ?2) OR (?3 IS NOT NULL AND imdb_id = ?3) ) + LIMIT 1", + )?; + let row = stmt + .query_row(params![kind_str, tmdb_id, imdb_id], |r| { + Ok(TitleRow { + id: r.get(0)?, + kind: r.get(1)?, + tmdb_id: r.get(2)?, + imdb_id: r.get(3)?, + name: r.get(4)?, + year: r.get(5)?, + adult: r.get::<_, i64>(6)? != 0, + certification: r.get(7)?, + }) + }) + .optional()?; + Ok(row) +} + +/// `kind` for the `titles` table: an episode manifest hangs off its *series* +/// title, since §2 keys episodes on series coordinates. +pub fn title_kind(kind: IdentityType) -> &'static str { + match kind { + IdentityType::Movie => "movie", + IdentityType::Episode => "series", + } +} + +/// Finds or creates the title row, returning its id. +pub fn upsert_title( + tx: &Transaction<'_>, + kind: IdentityType, + tmdb_id: Option<&str>, + imdb_id: Option<&str>, + name: Option<&str>, + year: Option, + now: &str, +) -> anyhow::Result { + if let Some(existing) = find_title(tx, kind, tmdb_id, imdb_id)? { + // Backfill identifiers a later upload supplied but an earlier one lacked. + tx.execute( + "UPDATE titles + SET tmdb_id = COALESCE(tmdb_id, ?2), + imdb_id = COALESCE(imdb_id, ?3), + year = COALESCE(year, ?4), + updated_at = ?5 + WHERE id = ?1", + params![existing.id, tmdb_id, imdb_id, year, now], + )?; + return Ok(existing.id); + } + + let id = ulid::Ulid::new().to_string(); + tx.execute( + "INSERT INTO titles (id, kind, tmdb_id, imdb_id, name, year, adult, certification, updated_at) + VALUES (?1, ?2, ?3, ?4, ?5, ?6, 0, NULL, ?7)", + params![id, title_kind(kind), tmdb_id, imdb_id, name, year, now], + )?; + Ok(id) +} + +/// Records TMDB-derived title attributes used by the §5a guards. +pub fn set_title_attributes( + tx: &Transaction<'_>, + title_id: &str, + adult: bool, + certification: Option<&str>, + name: Option<&str>, + now: &str, +) -> anyhow::Result<()> { + tx.execute( + "UPDATE titles + SET adult = ?2, + certification = COALESCE(?3, certification), + name = COALESCE(?4, name), + updated_at = ?5 + WHERE id = ?1", + params![title_id, adult as i64, certification, name, now], + )?; + Ok(()) +} + +// --------------------------------------------------------------------------- +// People — TMDB-derived, never from an upload (§5a, §7) +// --------------------------------------------------------------------------- + +pub fn upsert_person( + tx: &Transaction<'_>, + tmdb_person_id: u64, + name: &str, + adult: bool, + now: &str, +) -> anyhow::Result<()> { + // Portable `INSERT ... ON CONFLICT` rather than `INSERT OR REPLACE` (§8). + tx.execute( + "INSERT INTO people (tmdb_person_id, name, adult, updated_at) + VALUES (?1, ?2, ?3, ?4) + ON CONFLICT (tmdb_person_id) DO UPDATE + SET name = excluded.name, adult = excluded.adult, updated_at = excluded.updated_at", + params![tmdb_person_id as i64, name, adult as i64, now], + )?; + Ok(()) +} + +pub fn person_names( + conn: &Connection, + ids: &[u64], +) -> anyhow::Result> { + let mut out = std::collections::HashMap::new(); + let mut stmt = conn.prepare_cached("SELECT name FROM people WHERE tmdb_person_id = ?1")?; + for &id in ids { + if let Some(name) = + stmt.query_row(params![id as i64], |r| r.get::<_, String>(0)).optional()? + { + out.insert(id, name); + } + } + Ok(out) +} + +// --------------------------------------------------------------------------- +// Manifests +// --------------------------------------------------------------------------- + +/// Every candidate manifest for a title (and episode coordinates, if given) +/// that is eligible to be served. +/// +/// Cut matching is done in [`crate::matching`] rather than in SQL: the tier +/// logic is spec-critical and belongs somewhere testable. +pub fn candidates_for_title( + conn: &Connection, + title_id: &str, + season: Option, + episode: Option, +) -> anyhow::Result> { + let sql = format!( + "SELECT {MANIFEST_COLUMNS} FROM manifests + WHERE title_id = ?1 + AND status IN {SERVED_STATUSES} + AND ( (?2 IS NULL AND season IS NULL) OR season = ?2 ) + AND ( (?3 IS NULL AND episode IS NULL) OR episode = ?3 ) + ORDER BY cast_match_ratio DESC NULLS LAST, sample_fps DESC NULLS LAST, created_at ASC" + ); + let mut stmt = conn.prepare_cached(&sql)?; + let rows = stmt + .query_map(params![title_id, season, episode], map_manifest)? + .collect::>>()?; + Ok(rows) +} + +/// All servable episode manifests for a series, for bundle assembly (§2). +/// +/// A bundle is assembled per request; there is no "series manifest" row. +pub fn episodes_for_series( + conn: &Connection, + title_id: &str, + season: Option, +) -> anyhow::Result> { + let sql = format!( + "SELECT {MANIFEST_COLUMNS} FROM manifests + WHERE title_id = ?1 + AND status IN {SERVED_STATUSES} + AND season IS NOT NULL AND episode IS NOT NULL + AND (?2 IS NULL OR season = ?2) + ORDER BY season ASC, episode ASC, + cast_match_ratio DESC NULLS LAST, sample_fps DESC NULLS LAST" + ); + let mut stmt = conn.prepare_cached(&sql)?; + let rows = stmt + .query_map(params![title_id, season], map_manifest)? + .collect::>>()?; + Ok(rows) +} + +pub fn manifest_by_id(conn: &Connection, id: &str) -> anyhow::Result> { + let sql = format!("SELECT {MANIFEST_COLUMNS} FROM manifests WHERE id = ?1"); + let mut stmt = conn.prepare_cached(&sql)?; + Ok(stmt.query_row(params![id], map_manifest).optional()?) +} + +pub fn manifest_status( + conn: &Connection, + id: &str, +) -> anyhow::Result)>> { + let mut stmt = + conn.prepare_cached("SELECT status, reject_reason FROM manifests WHERE id = ?1")?; + Ok(stmt.query_row(params![id], |r| Ok((r.get(0)?, r.get(1)?))).optional()?) +} + +/// §4 `409`: an identical `(identity, cut)` manifest already exists from this +/// contributor. +pub fn duplicate_from_contributor( + tx: &Transaction<'_>, + title_id: &str, + season: Option, + episode: Option, + runtime_sec: f64, + video_hash: Option<&str>, + contributor_id: &str, +) -> anyhow::Result> { + let mut stmt = tx.prepare_cached( + "SELECT id FROM manifests + WHERE title_id = ?1 AND contributor_id = ?6 + AND status <> 'rejected' + AND ( (?2 IS NULL AND season IS NULL) OR season = ?2 ) + AND ( (?3 IS NULL AND episode IS NULL) OR episode = ?3 ) + AND ABS(runtime_sec - ?4) < 0.001 + AND ( (?5 IS NULL AND video_hash IS NULL) OR video_hash = ?5 ) + LIMIT 1", + )?; + Ok(stmt + .query_row( + params![title_id, season, episode, runtime_sec, video_hash, contributor_id], + |r| r.get::<_, String>(0), + ) + .optional()?) +} + +pub fn manifest_by_content_id( + tx: &Transaction<'_>, + content_id: &str, +) -> anyhow::Result> { + let mut stmt = tx.prepare_cached("SELECT id FROM manifests WHERE content_id = ?1")?; + Ok(stmt.query_row(params![content_id], |r| r.get::<_, String>(0)).optional()?) +} + +/// Everything needed to insert one manifest. +pub struct NewManifest<'a> { + pub id: &'a str, + pub title_id: &'a str, + pub season: Option, + pub episode: Option, + pub runtime_sec: f64, + pub video_hash: Option<&'a str>, + pub audio_signature: Option<&'a [u8]>, + pub audio_sig_coarse: Option<&'a [u8]>, + pub sample_fps: Option, + pub extinction_sec: Option, + pub pipeline_version: Option<&'a str>, + pub gallery_scope: Option<&'a str>, + pub contributor_id: Option<&'a str>, + pub status: &'a str, + pub content_id: Option<&'a str>, + pub origin: &'a str, + pub ingested_from: Option<&'a str>, + pub created_at: &'a str, +} + +pub fn insert_manifest(tx: &Transaction<'_>, m: &NewManifest<'_>) -> anyhow::Result<()> { + tx.execute( + "INSERT INTO manifests + (id, title_id, season, episode, runtime_sec, video_hash, + audio_signature, audio_sig_coarse, sample_fps, extinction_sec, pipeline_version, + gallery_scope, contributor_id, status, content_id, origin, ingested_from, + created_at) + VALUES (?1,?2,?3,?4,?5,?6,?7,?8,?9,?10,?11,?12,?13,?14,?15,?16,?17,?18)", + params![ + m.id, + m.title_id, + m.season, + m.episode, + m.runtime_sec, + m.video_hash, + m.audio_signature, + m.audio_sig_coarse, + m.sample_fps, + m.extinction_sec, + m.pipeline_version, + m.gallery_scope, + m.contributor_id, + m.status, + m.content_id, + m.origin, + m.ingested_from, + m.created_at, + ], + ) + .context("inserting manifest")?; + Ok(()) +} + +/// Inserts one actor's rows. Batched within the caller's transaction — one +/// transaction per manifest, not per row (§8). +pub fn insert_actor_scenes( + tx: &Transaction<'_>, + manifest_id: &str, + tmdb_person_id: u64, + scenes_cs: &[(i64, i64)], +) -> anyhow::Result<()> { + tx.execute( + "INSERT INTO manifest_actors (manifest_id, tmdb_person_id) VALUES (?1, ?2) + ON CONFLICT (manifest_id, tmdb_person_id) DO NOTHING", + params![manifest_id, tmdb_person_id as i64], + )?; + let mut stmt = tx.prepare_cached( + "INSERT INTO scenes (manifest_id, tmdb_person_id, start_cs, end_cs) + VALUES (?1, ?2, ?3, ?4)", + )?; + for (start, end) in scenes_cs { + stmt.execute(params![manifest_id, tmdb_person_id as i64, start, end])?; + } + Ok(()) +} + +/// The actor payload of a stored manifest, reconstructed from rows. +/// +/// §7: the submitted JSON is discarded; the document served to clients is +/// *reconstructed*, never echoed. Names come from `people`, populated from TMDB. +#[derive(Debug, Clone)] +pub struct StoredActor { + pub tmdb_person_id: u64, + pub name: Option, + pub scenes_cs: Vec<(i64, i64)>, +} + +pub fn actors_for_manifest( + conn: &Connection, + manifest_id: &str, +) -> anyhow::Result> { + let mut stmt = conn.prepare_cached( + "SELECT ma.tmdb_person_id, p.name + FROM manifest_actors ma + LEFT JOIN people p ON p.tmdb_person_id = ma.tmdb_person_id + WHERE ma.manifest_id = ?1 + ORDER BY ma.tmdb_person_id ASC", + )?; + let people = stmt + .query_map(params![manifest_id], |r| { + Ok((r.get::<_, i64>(0)? as u64, r.get::<_, Option>(1)?)) + })? + .collect::>>()?; + + let mut scene_stmt = conn.prepare_cached( + "SELECT start_cs, end_cs FROM scenes + WHERE manifest_id = ?1 AND tmdb_person_id = ?2 + ORDER BY start_cs ASC", + )?; + + let mut out = Vec::with_capacity(people.len()); + for (id, name) in people { + let scenes_cs = scene_stmt + .query_map(params![manifest_id, id as i64], |r| Ok((r.get(0)?, r.get(1)?)))? + .collect::>>()?; + out.push(StoredActor { tmdb_person_id: id, name, scenes_cs }); + } + Ok(out) +} + +pub fn set_manifest_status( + tx: &Transaction<'_>, + id: &str, + status: &str, + reason: Option<&str>, + cast_match_ratio: Option, +) -> anyhow::Result<()> { + tx.execute( + "UPDATE manifests + SET status = ?2, reject_reason = ?3, cast_match_ratio = COALESCE(?4, cast_match_ratio) + WHERE id = ?1", + params![id, status, reason, cast_match_ratio], + )?; + Ok(()) +} + +/// §6 stage 3: unmatched actors are dropped rather than stored, which is what +/// closes the free-text channel described in §5a. +pub fn delete_manifest_actor( + tx: &Transaction<'_>, + manifest_id: &str, + tmdb_person_id: u64, +) -> anyhow::Result<()> { + tx.execute( + "DELETE FROM scenes WHERE manifest_id = ?1 AND tmdb_person_id = ?2", + params![manifest_id, tmdb_person_id as i64], + )?; + tx.execute( + "DELETE FROM manifest_actors WHERE manifest_id = ?1 AND tmdb_person_id = ?2", + params![manifest_id, tmdb_person_id as i64], + )?; + Ok(()) +} + +/// §6 stage 3: a rejected manifest is deleted, not merely marked. +pub fn delete_manifest(tx: &Transaction<'_>, id: &str) -> anyhow::Result<()> { + tx.execute("DELETE FROM scenes WHERE manifest_id = ?1", params![id])?; + tx.execute("DELETE FROM manifest_actors WHERE manifest_id = ?1", params![id])?; + tx.execute("DELETE FROM manifests WHERE id = ?1", params![id])?; + Ok(()) +} + +pub fn manifest_actor_ids(conn: &Connection, manifest_id: &str) -> anyhow::Result> { + let mut stmt = stmt_actor_ids(conn)?; + let ids = stmt + .query_map(params![manifest_id], |r| Ok(r.get::<_, i64>(0)? as u64))? + .collect::>>()?; + Ok(ids) +} + +fn stmt_actor_ids(conn: &Connection) -> rusqlite::Result> { + conn.prepare_cached( + "SELECT tmdb_person_id FROM manifest_actors WHERE manifest_id = ?1 ORDER BY tmdb_person_id", + ) +} + +pub fn report_count(conn: &Connection, manifest_id: &str) -> anyhow::Result { + let mut stmt = conn.prepare_cached("SELECT COUNT(*) FROM reports WHERE manifest_id = ?1")?; + Ok(stmt.query_row(params![manifest_id], |r| r.get(0))?) +} + +// --------------------------------------------------------------------------- +// Contributors +// --------------------------------------------------------------------------- + +#[derive(Debug, Clone)] +pub struct Contributor { + pub id: String, + pub revoked: bool, + pub accepted_count: i64, + pub rejected_count: i64, + pub flagged_count: i64, +} + +pub fn contributor_by_token_hash( + conn: &Connection, + token_hash: &str, +) -> anyhow::Result> { + let mut stmt = conn.prepare_cached( + "SELECT id, revoked_at, accepted_count, rejected_count, flagged_count + FROM contributors WHERE token_hash = ?1", + )?; + Ok(stmt + .query_row(params![token_hash], |r| { + Ok(Contributor { + id: r.get(0)?, + revoked: r.get::<_, Option>(1)?.is_some(), + accepted_count: r.get(2)?, + rejected_count: r.get(3)?, + flagged_count: r.get(4)?, + }) + }) + .optional()?) +} + +pub fn insert_contributor( + tx: &Transaction<'_>, + token_hash: &str, + now: &str, +) -> anyhow::Result { + let id = ulid::Ulid::new().to_string(); + tx.execute( + "INSERT INTO contributors (id, token_hash, created_at) VALUES (?1, ?2, ?3)", + params![id, token_hash, now], + )?; + Ok(id) +} + +/// §5a: the only reputational state is per-token counters. +pub fn bump_contributor_counter( + tx: &Transaction<'_>, + contributor_id: &str, + which: &str, +) -> anyhow::Result<()> { + // Column name is from a closed set, never from input. + let sql = match which { + "accepted" => "UPDATE contributors SET accepted_count = accepted_count + 1 WHERE id = ?1", + "rejected" => "UPDATE contributors SET rejected_count = rejected_count + 1 WHERE id = ?1", + "flagged" => "UPDATE contributors SET flagged_count = flagged_count + 1 WHERE id = ?1", + other => anyhow::bail!("unknown contributor counter {other}"), + }; + tx.execute(sql, params![contributor_id])?; + Ok(()) +} + +/// §5a: a token whose rejection rate exceeds a threshold over a minimum sample +/// is revoked automatically, and its `pending`/`flagged` manifests are dropped. +/// No human is in the loop for the common case. +pub const REVOKE_MIN_SAMPLE: i64 = 20; +pub const REVOKE_REJECTION_RATE: f64 = 0.5; + +pub fn maybe_revoke_contributor( + tx: &Transaction<'_>, + contributor_id: &str, + now: &str, +) -> anyhow::Result { + let (accepted, rejected, revoked): (i64, i64, Option) = tx.query_row( + "SELECT accepted_count, rejected_count, revoked_at FROM contributors WHERE id = ?1", + params![contributor_id], + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)), + )?; + if revoked.is_some() { + return Ok(false); + } + let total = accepted + rejected; + if total < REVOKE_MIN_SAMPLE { + return Ok(false); + } + if (rejected as f64) / (total as f64) <= REVOKE_REJECTION_RATE { + return Ok(false); + } + + tx.execute( + "UPDATE contributors SET revoked_at = ?2 WHERE id = ?1", + params![contributor_id, now], + )?; + // Drop the token's pending/flagged manifests. + let ids: Vec = { + let mut stmt = tx.prepare( + "SELECT id FROM manifests WHERE contributor_id = ?1 AND status IN ('pending','flagged')", + )?; + let rows = stmt + .query_map(params![contributor_id], |r| r.get::<_, String>(0))? + .collect::>>()?; + rows + }; + for id in &ids { + delete_manifest(tx, id)?; + } + Ok(true) +} + +// --------------------------------------------------------------------------- +// Reports +// --------------------------------------------------------------------------- + +pub fn insert_report( + tx: &Transaction<'_>, + manifest_id: &str, + reason: &str, + note: Option<&str>, + source_ip_hash: &str, + now: &str, +) -> anyhow::Result { + let id = ulid::Ulid::new().to_string(); + tx.execute( + "INSERT INTO reports (id, manifest_id, reason, note, created_at, source_ip_hash) + VALUES (?1,?2,?3,?4,?5,?6)", + params![id, manifest_id, reason, note, now, source_ip_hash], + )?; + Ok(id) +} + +// --------------------------------------------------------------------------- +// TMDB cache — the sole JSON column, holding TMDB's responses, not users' (§7) +// --------------------------------------------------------------------------- + +pub fn cached_credits( + conn: &Connection, + tmdb_id: &str, + kind: &str, +) -> anyhow::Result> { + let mut stmt = conn.prepare_cached( + "SELECT credits, fetched_at FROM tmdb_cache WHERE tmdb_id = ?1 AND kind = ?2", + )?; + Ok(stmt.query_row(params![tmdb_id, kind], |r| Ok((r.get(0)?, r.get(1)?))).optional()?) +} + +pub fn put_credits( + tx: &Transaction<'_>, + tmdb_id: &str, + kind: &str, + credits: &str, + now: &str, +) -> anyhow::Result<()> { + tx.execute( + "INSERT INTO tmdb_cache (tmdb_id, kind, credits, fetched_at) VALUES (?1,?2,?3,?4) + ON CONFLICT (tmdb_id, kind) DO UPDATE + SET credits = excluded.credits, fetched_at = excluded.fetched_at", + params![tmdb_id, kind, credits, now], + )?; + Ok(()) +} + +// --------------------------------------------------------------------------- +// Jobs — a table rather than an external broker, so work survives restart (§7) +// --------------------------------------------------------------------------- + +#[derive(Debug, Clone)] +pub struct Job { + pub id: String, + pub kind: String, + pub payload: String, + pub attempts: i64, +} + +pub fn enqueue_job( + tx: &Transaction<'_>, + kind: &str, + payload: &str, + run_after: &str, +) -> anyhow::Result { + let id = ulid::Ulid::new().to_string(); + tx.execute( + "INSERT INTO jobs (id, kind, payload, run_after) VALUES (?1,?2,?3,?4)", + params![id, kind, payload, run_after], + )?; + Ok(id) +} + +/// Claims up to `limit` due jobs, marking them leased so a second worker tick +/// cannot pick up the same work. +pub fn lease_jobs(tx: &Transaction<'_>, now: &str, limit: usize) -> anyhow::Result> { + let jobs: Vec = { + let mut stmt = tx.prepare( + "SELECT id, kind, payload, attempts FROM jobs + WHERE run_after <= ?1 AND leased_at IS NULL + ORDER BY run_after ASC + LIMIT ?2", + )?; + let rows = stmt + .query_map(params![now, limit as i64], |r| { + Ok(Job { id: r.get(0)?, kind: r.get(1)?, payload: r.get(2)?, attempts: r.get(3)? }) + })? + .collect::>>()?; + rows + }; + for j in &jobs { + tx.execute("UPDATE jobs SET leased_at = ?2 WHERE id = ?1", params![j.id, now])?; + } + Ok(jobs) +} + +pub fn delete_job(tx: &Transaction<'_>, id: &str) -> anyhow::Result<()> { + tx.execute("DELETE FROM jobs WHERE id = ?1", params![id])?; + Ok(()) +} + +/// Releases a leased job for a later retry with backoff (§6 stage 3: TMDB +/// unreachable means retry, not reject). +pub fn reschedule_job( + tx: &Transaction<'_>, + id: &str, + run_after: &str, + error: &str, +) -> anyhow::Result<()> { + tx.execute( + "UPDATE jobs SET attempts = attempts + 1, last_error = ?3, run_after = ?2, leased_at = NULL + WHERE id = ?1", + params![id, run_after, error], + )?; + Ok(()) +} + +/// Frees leases held at startup — a process that died mid-job would otherwise +/// leave work stranded. +pub fn release_all_leases(tx: &Transaction<'_>) -> anyhow::Result { + let n = tx.execute("UPDATE jobs SET leased_at = NULL WHERE leased_at IS NOT NULL", [])?; + Ok(n) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::db::Db; + + async fn db() -> Db { + Db::open(":memory:").unwrap() + } + + const NOW: &str = "2026-07-30T12:00:00Z"; + + #[tokio::test] + async fn upsert_title_is_idempotent_and_backfills() { + let db = db().await; + let (a, b) = db + .write(|tx| { + let a = upsert_title( + tx, + IdentityType::Movie, + Some("504172"), + None, + Some("X"), + Some(2017), + NOW, + )?; + // Second upload supplies the IMDB id the first lacked. + let b = upsert_title( + tx, + IdentityType::Movie, + Some("504172"), + Some("tt4686844"), + None, + None, + NOW, + )?; + Ok((a, b)) + }) + .await + .unwrap(); + assert_eq!(a, b, "same title must not be duplicated"); + + let found = db + .write(|tx| find_title(tx, IdentityType::Movie, Some("504172"), None)) + .await + .unwrap() + .unwrap(); + assert_eq!(found.imdb_id.as_deref(), Some("tt4686844"), "identifier should be backfilled"); + } + + #[tokio::test] + async fn movie_and_series_titles_do_not_collide_on_id() { + // TMDB numbers movies and series in separate spaces, so id 1396 is both + // a film and Breaking Bad. + let db = db().await; + let (m, s) = db + .write(|tx| { + let m = upsert_title(tx, IdentityType::Movie, Some("1396"), None, None, None, NOW)?; + let s = + upsert_title(tx, IdentityType::Episode, Some("1396"), None, None, None, NOW)?; + Ok((m, s)) + }) + .await + .unwrap(); + assert_ne!(m, s); + } + + #[tokio::test] + async fn manifest_roundtrips_through_rows() { + let db = db().await; + let id = db + .write(|tx| { + let title_id = + upsert_title(tx, IdentityType::Movie, Some("1"), None, None, None, NOW)?; + let cid = insert_contributor(tx, "hash", NOW)?; + let id = ulid::Ulid::new().to_string(); + insert_manifest( + tx, + &NewManifest { + id: &id, + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 6420.5, + video_hash: Some("opensubtitles:8e245d9679d31e12"), + audio_signature: None, + audio_sig_coarse: None, + sample_fps: Some(5.0), + extinction_sec: Some(12.0), + gallery_scope: Some("global"), + pipeline_version: Some("test 0.1"), + contributor_id: Some(&cid), + status: "listed", + content_id: Some("sha256:abc"), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + upsert_person(tx, 884, "Steve Buscemi", false, NOW)?; + insert_actor_scenes(tx, &id, 884, &[(19160, 20920), (43820, 46560)])?; + Ok(id) + }) + .await + .unwrap(); + + let (row, actors) = db + .write(move |tx| { + let row = manifest_by_id(tx, &id)?.unwrap(); + let actors = actors_for_manifest(tx, &id)?; + Ok((row, actors)) + }) + .await + .unwrap(); + + assert_eq!(row.runtime_sec, 6420.5); + assert_eq!(actors.len(), 1); + assert_eq!(actors[0].tmdb_person_id, 884); + // §7: names come from `people`, populated from TMDB, never from upload. + assert_eq!(actors[0].name.as_deref(), Some("Steve Buscemi")); + assert_eq!(actors[0].scenes_cs, vec![(19160, 20920), (43820, 46560)]); + } + + #[tokio::test] + async fn only_served_statuses_are_candidates() { + let db = db().await; + let title_id = db + .write(|tx| { + let title_id = + upsert_title(tx, IdentityType::Movie, Some("7"), None, None, None, NOW)?; + for (i, status) in ["listed", "flagged", "pending", "rejected"].iter().enumerate() { + insert_manifest( + tx, + &NewManifest { + id: &format!("m{i}"), + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 100.0, + video_hash: None, + audio_signature: None, + audio_sig_coarse: None, + sample_fps: None, + extinction_sec: None, + gallery_scope: None, + pipeline_version: None, + contributor_id: None, + status, + content_id: Some(&format!("c{i}")), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + } + Ok(title_id) + }) + .await + .unwrap(); + + let rows = + db.write(move |tx| candidates_for_title(tx, &title_id, None, None)).await.unwrap(); + // `pending` is held unlisted and not served to anyone (§6 stage 3); + // `rejected` never is. + assert_eq!(rows.len(), 2); + assert!(rows.iter().all(|r| r.status == "listed" || r.status == "flagged")); + } + + #[tokio::test] + async fn duplicate_detection_is_per_contributor() { + let db = db().await; + let dup = db + .write(|tx| { + let title_id = + upsert_title(tx, IdentityType::Movie, Some("9"), None, None, None, NOW)?; + let a = insert_contributor(tx, "hash-a", NOW)?; + let b = insert_contributor(tx, "hash-b", NOW)?; + insert_manifest( + tx, + &NewManifest { + id: "m1", + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 6420.5, + video_hash: None, + audio_signature: None, + audio_sig_coarse: None, + sample_fps: None, + extinction_sec: None, + gallery_scope: None, + pipeline_version: None, + contributor_id: Some(&a), + status: "listed", + content_id: Some("c1"), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + let same = duplicate_from_contributor(tx, &title_id, None, None, 6420.5, None, &a)?; + // §7: multiple manifests for the same cut from *different* + // contributors are allowed, and ranked. + let other = + duplicate_from_contributor(tx, &title_id, None, None, 6420.5, None, &b)?; + Ok((same.is_some(), other.is_some())) + }) + .await + .unwrap(); + assert_eq!(dup, (true, false)); + } + + #[tokio::test] + async fn revocation_needs_a_minimum_sample_then_drops_pending_work() { + let db = db().await; + let revoked_early = db + .write(|tx| { + let c = insert_contributor(tx, "h", NOW)?; + for _ in 0..5 { + bump_contributor_counter(tx, &c, "rejected")?; + } + // §5a: a threshold over a *minimum sample* — 5 rejections is not + // yet evidence. + let early = maybe_revoke_contributor(tx, &c, NOW)?; + + for _ in 0..20 { + bump_contributor_counter(tx, &c, "rejected")?; + } + let title_id = + upsert_title(tx, IdentityType::Movie, Some("11"), None, None, None, NOW)?; + insert_manifest( + tx, + &NewManifest { + id: "pend", + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 100.0, + video_hash: None, + audio_signature: None, + audio_sig_coarse: None, + sample_fps: None, + extinction_sec: None, + gallery_scope: None, + pipeline_version: None, + contributor_id: Some(&c), + status: "pending", + content_id: Some("cp"), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + let now_revoked = maybe_revoke_contributor(tx, &c, NOW)?; + let still_there = manifest_by_id(tx, "pend")?.is_some(); + Ok((early, now_revoked, still_there)) + }) + .await + .unwrap(); + assert_eq!(revoked_early, (false, true, false)); + } + + #[tokio::test] + async fn good_contributors_are_not_revoked() { + let db = db().await; + let revoked = db + .write(|tx| { + let c = insert_contributor(tx, "h", NOW)?; + for _ in 0..40 { + bump_contributor_counter(tx, &c, "accepted")?; + } + for _ in 0..3 { + bump_contributor_counter(tx, &c, "rejected")?; + } + maybe_revoke_contributor(tx, &c, NOW) + }) + .await + .unwrap(); + assert!(!revoked); + } + + #[tokio::test] + async fn jobs_lease_once_and_reschedule() { + let db = db().await; + let (first, second, after_reschedule) = db + .write(|tx| { + enqueue_job(tx, "cast_check", "{}", NOW)?; + let first = lease_jobs(tx, NOW, 10)?; + // A leased job must not be handed out twice. + let second = lease_jobs(tx, NOW, 10)?; + reschedule_job(tx, &first[0].id, NOW, "tmdb unreachable")?; + let third = lease_jobs(tx, NOW, 10)?; + Ok((first.len(), second.len(), third)) + }) + .await + .unwrap(); + assert_eq!((first, second), (1, 0)); + assert_eq!(after_reschedule.len(), 1); + assert_eq!(after_reschedule[0].attempts, 1); + } + + #[tokio::test] + async fn future_jobs_are_not_leased() { + let db = db().await; + let n = db + .write(|tx| { + enqueue_job(tx, "cast_check", "{}", "2099-01-01T00:00:00Z")?; + Ok(lease_jobs(tx, NOW, 10)?.len()) + }) + .await + .unwrap(); + assert_eq!(n, 0); + } + + #[tokio::test] + async fn startup_releases_stranded_leases() { + let db = db().await; + let n = db + .write(|tx| { + enqueue_job(tx, "cast_check", "{}", NOW)?; + lease_jobs(tx, NOW, 10)?; + release_all_leases(tx)?; + Ok(lease_jobs(tx, NOW, 10)?.len()) + }) + .await + .unwrap(); + assert_eq!(n, 1); + } + + #[tokio::test] + async fn deleting_a_manifest_removes_its_rows() { + let db = db().await; + let counts = db + .write(|tx| { + let title_id = + upsert_title(tx, IdentityType::Movie, Some("13"), None, None, None, NOW)?; + insert_manifest( + tx, + &NewManifest { + id: "m", + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 100.0, + video_hash: None, + audio_signature: None, + audio_sig_coarse: None, + sample_fps: None, + extinction_sec: None, + gallery_scope: None, + pipeline_version: None, + contributor_id: None, + status: "listed", + content_id: Some("c"), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + upsert_person(tx, 1, "A", false, NOW)?; + insert_actor_scenes(tx, "m", 1, &[(0, 100)])?; + delete_manifest(tx, "m")?; + let scenes: i64 = tx.query_row("SELECT COUNT(*) FROM scenes", [], |r| r.get(0))?; + let actors: i64 = + tx.query_row("SELECT COUNT(*) FROM manifest_actors", [], |r| r.get(0))?; + let manifests: i64 = + tx.query_row("SELECT COUNT(*) FROM manifests", [], |r| r.get(0))?; + Ok((scenes, actors, manifests)) + }) + .await + .unwrap(); + assert_eq!(counts, (0, 0, 0)); + } +} diff --git a/src/db/schema.sql b/src/db/schema.sql new file mode 100644 index 0000000..3a2afe4 --- /dev/null +++ b/src/db/schema.sql @@ -0,0 +1,119 @@ +-- §7 Storage. Fully relational, no JSON blobs on the write path: the database +-- can only represent what the schema models, so there is physically nowhere for +-- an unexpected field or a smuggled string to live (§5a Threat 1). +-- +-- Portable SQL — runs unchanged on Postgres. Avoid SQLite-specific forms +-- (`INSERT OR REPLACE`); use `INSERT ... ON CONFLICT` (§8 deployment notes). + +CREATE TABLE IF NOT EXISTS contributors ( + id TEXT PRIMARY KEY, + token_hash TEXT NOT NULL UNIQUE, + created_at TEXT NOT NULL, + revoked_at TEXT, + accepted_count INTEGER NOT NULL DEFAULT 0, + rejected_count INTEGER NOT NULL DEFAULT 0, + flagged_count INTEGER NOT NULL DEFAULT 0 +); + +-- Server-side, TMDB-derived. `name` never comes from an upload (§5a). +CREATE TABLE IF NOT EXISTS people ( + tmdb_person_id INTEGER PRIMARY KEY, + name TEXT NOT NULL, + adult INTEGER NOT NULL DEFAULT 0, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS titles ( + id TEXT PRIMARY KEY, + kind TEXT NOT NULL, -- movie | series + tmdb_id TEXT, + imdb_id TEXT, + name TEXT, + year INTEGER, + adult INTEGER NOT NULL DEFAULT 0, + certification TEXT, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS manifests ( + id TEXT PRIMARY KEY, + title_id TEXT NOT NULL REFERENCES titles(id), + season INTEGER, + episode INTEGER, + runtime_sec REAL NOT NULL, + video_hash TEXT, + audio_signature BLOB, -- §3, ~1290 bytes + audio_sig_coarse BLOB, -- candidate-generation index key + sample_fps REAL, + extinction_sec REAL, -- successor to the withdrawn anneal_sec + gallery_scope TEXT, -- limited | global; ranking signal (§2, §7) + pipeline_version TEXT, + contributor_id TEXT REFERENCES contributors(id), + status TEXT NOT NULL, -- pending | listed | flagged | rejected + reject_reason TEXT, + cast_match_ratio REAL, + content_id TEXT UNIQUE, -- §9a, sha256 over canonical form + origin TEXT, -- server_id of first acceptance + ingested_from TEXT, -- peer id, NULL if uploaded directly + created_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS manifest_actors ( + manifest_id TEXT NOT NULL REFERENCES manifests(id) ON DELETE CASCADE, + tmdb_person_id INTEGER NOT NULL, + PRIMARY KEY (manifest_id, tmdb_person_id) +); + +-- Integer centiseconds, not floats — the same quantisation used for +-- `content_id`, so stored values and hashed values cannot diverge (§7, §9a). +CREATE TABLE IF NOT EXISTS scenes ( + manifest_id TEXT NOT NULL REFERENCES manifests(id) ON DELETE CASCADE, + tmdb_person_id INTEGER NOT NULL, + start_cs INTEGER NOT NULL, + end_cs INTEGER NOT NULL +); + +CREATE TABLE IF NOT EXISTS reports ( + id TEXT PRIMARY KEY, + manifest_id TEXT NOT NULL REFERENCES manifests(id) ON DELETE CASCADE, + reason TEXT NOT NULL, + note TEXT, + created_at TEXT NOT NULL, + source_ip_hash TEXT +); + +-- The sole JSON column, and it holds TMDB's responses, not users' (§7). +CREATE TABLE IF NOT EXISTS tmdb_cache ( + tmdb_id TEXT NOT NULL, + kind TEXT NOT NULL, + credits TEXT NOT NULL, + fetched_at TEXT NOT NULL, + PRIMARY KEY (tmdb_id, kind) +); + +-- Background queue as a table rather than an external broker, so pending work +-- survives a restart (§7, §8). +CREATE TABLE IF NOT EXISTS jobs ( + id TEXT PRIMARY KEY, + kind TEXT NOT NULL, -- cast_check | federation_pull + payload TEXT NOT NULL, + run_after TEXT NOT NULL, + attempts INTEGER NOT NULL DEFAULT 0, + last_error TEXT, + leased_at TEXT +); + +CREATE INDEX IF NOT EXISTS idx_titles_tmdb ON titles(tmdb_id); +CREATE INDEX IF NOT EXISTS idx_titles_imdb ON titles(imdb_id); +CREATE INDEX IF NOT EXISTS idx_manifests_title_runtime ON manifests(title_id, runtime_sec); +CREATE INDEX IF NOT EXISTS idx_manifests_video_hash ON manifests(video_hash); +CREATE INDEX IF NOT EXISTS idx_manifests_episode ON manifests(title_id, season, episode); +CREATE INDEX IF NOT EXISTS idx_scenes_manifest_person ON scenes(manifest_id, tmdb_person_id); + +-- All read queries filter `status IN ('listed','flagged')`, so a partial index +-- on that predicate keeps the hot path small (§7). +CREATE INDEX IF NOT EXISTS idx_manifests_served + ON manifests(title_id, season, episode) + WHERE status IN ('listed', 'flagged'); + +CREATE INDEX IF NOT EXISTS idx_jobs_ready ON jobs(run_after); diff --git a/src/error.rs b/src/error.rs new file mode 100644 index 0000000..f5043d5 --- /dev/null +++ b/src/error.rs @@ -0,0 +1,83 @@ +//! API error type mapping onto the status codes §4 specifies. + +use axum::http::StatusCode; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::Serialize; + +#[derive(Debug, thiserror::Error)] +pub enum ApiError { + /// §6 stage 2 — malformed, unrecognised or forbidden field. The message + /// names the offending field so a client that forgets to strip `movie` or + /// `jellyfin_id` gets a hard, diagnosable `400` (§6). + #[error("{0}")] + BadRequest(String), + + #[error("not found")] + NotFound, + + /// §4 — identical `(identity, cut)` already exists from this contributor. + #[error("{0}")] + Conflict(String), + + #[error("{0}")] + PayloadTooLarge(String), + + #[error("missing or invalid API token")] + Unauthorized, + + /// §5 — carries the `Retry-After` value in seconds. + #[error("rate limited")] + RateLimited { retry_after: u64 }, + + #[error("internal error")] + Internal(#[from] anyhow::Error), +} + +#[derive(Serialize)] +struct ErrorBody { + error: String, + message: String, +} + +impl IntoResponse for ApiError { + fn into_response(self) -> Response { + let (status, code) = match &self { + ApiError::BadRequest(_) => (StatusCode::BAD_REQUEST, "bad_request"), + ApiError::NotFound => (StatusCode::NOT_FOUND, "not_found"), + ApiError::Conflict(_) => (StatusCode::CONFLICT, "conflict"), + ApiError::PayloadTooLarge(_) => (StatusCode::PAYLOAD_TOO_LARGE, "payload_too_large"), + ApiError::Unauthorized => (StatusCode::UNAUTHORIZED, "unauthorized"), + ApiError::RateLimited { .. } => (StatusCode::TOO_MANY_REQUESTS, "rate_limited"), + ApiError::Internal(e) => { + // Internal detail is logged, never returned. + tracing::error!(error = ?e, "internal error"); + (StatusCode::INTERNAL_SERVER_ERROR, "internal") + } + }; + + let body = Json(ErrorBody { + error: code.to_string(), + message: match &self { + ApiError::Internal(_) => "internal error".to_string(), + other => other.to_string(), + }, + }); + + let mut resp = (status, body).into_response(); + if let ApiError::RateLimited { retry_after } = self { + if let Ok(v) = retry_after.to_string().parse() { + resp.headers_mut().insert(axum::http::header::RETRY_AFTER, v); + } + } + resp + } +} + +impl From for ApiError { + fn from(e: rusqlite::Error) -> Self { + ApiError::Internal(anyhow::Error::new(e)) + } +} + +pub type ApiResult = Result; diff --git a/src/ingest.rs b/src/ingest.rs new file mode 100644 index 0000000..1fede64 --- /dev/null +++ b/src/ingest.rs @@ -0,0 +1,403 @@ +//! Manifest ingestion: the shared path behind `POST /manifests` and +//! `POST /manifests/bundle`, and the path a federation pull will reuse (§9a +//! "re-derive, don't inherit"). +//! +//! Stages 0 and 1 are layers; stage 2 is parse + [`crate::validate`]. What +//! happens here is persistence plus enqueueing the stage 3 check: the upload is +//! accepted with `202` and the manifest is held **unlisted** until the cast check +//! completes — it is not served to anyone in the meantime (§6). + +use anyhow::Context; + +use crate::content_id::{self, CanonicalActor, CanonicalCut, CanonicalIdentity}; +use crate::db::repo::{self, NewManifest}; +use crate::model::IdentityType; +use crate::validate::ValidManifest; + +/// Outcome of persisting one manifest. +#[derive(Debug, Clone)] +pub enum IngestOutcome { + /// Held unlisted pending the §6 stage 3 cast check. + Pending { manifest_id: String }, + /// §4 `409` — identical `(identity, cut)` from this contributor. + DuplicateFromContributor { manifest_id: String }, + /// §9a — the exact same content is already held, from any source. Skipped + /// without re-validation, which is the deduplication content addressing buys. + DuplicateContent { manifest_id: String }, +} + +impl IngestOutcome { + pub fn manifest_id(&self) -> &str { + match self { + IngestOutcome::Pending { manifest_id } + | IngestOutcome::DuplicateFromContributor { manifest_id } + | IngestOutcome::DuplicateContent { manifest_id } => manifest_id, + } + } +} + +/// Job payload for the §6 stage 3 check. +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct CastCheckJob { + pub manifest_id: String, +} + +pub const JOB_CAST_CHECK: &str = "cast_check"; + +/// Persists a validated manifest and enqueues its cast check, all in one +/// transaction — so a manifest is never left listed-but-unchecked, and its scene +/// rows go in as a single transaction rather than one per row (§8). +pub fn persist( + tx: &rusqlite::Transaction<'_>, + valid: &ValidManifest, + contributor_id: Option<&str>, + origin: &str, + ingested_from: Option<&str>, + now: &str, +) -> anyhow::Result { + let m = &valid.manifest; + let kind = m.identity.kind; + let tmdb_id = m.identity.effective_tmdb_id(); + let imdb_id = m.identity.effective_imdb_id(); + + let title_id = repo::upsert_title( + tx, + kind, + tmdb_id, + imdb_id, + m.identity.title.as_deref(), + m.identity.year, + now, + ) + .context("resolving title")?; + + let (season, episode) = match kind { + IdentityType::Movie => (None, None), + IdentityType::Episode => (m.identity.season, m.identity.episode), + }; + + // Content addressing over the *submitted* actor ids. Recomputed after the + // cast check drops unmatched actors, since dropping changes the content. + let cid = compute_content_id(valid); + + if let Some(existing) = repo::manifest_by_content_id(tx, &cid)? { + return Ok(IngestOutcome::DuplicateContent { manifest_id: existing }); + } + + if let Some(c) = contributor_id { + if let Some(existing) = repo::duplicate_from_contributor( + tx, + &title_id, + season, + episode, + m.cut.runtime_sec, + m.cut.video_hash.as_deref(), + c, + )? { + return Ok(IngestOutcome::DuplicateFromContributor { manifest_id: existing }); + } + } + + let manifest_id = ulid::Ulid::new().to_string(); + let extraction = m.extraction.as_ref(); + + repo::insert_manifest( + tx, + &NewManifest { + id: &manifest_id, + title_id: &title_id, + season, + episode, + runtime_sec: m.cut.runtime_sec, + video_hash: m.cut.video_hash.as_deref(), + // Stored as an attribute, not part of identity (§9a). + audio_signature: None, + audio_sig_coarse: None, + sample_fps: extraction.and_then(|e| e.sample_fps), + extinction_sec: extraction.and_then(|e| e.extinction_sec), + pipeline_version: extraction.and_then(|e| e.pipeline_version.as_deref()), + gallery_scope: extraction.and_then(|e| e.gallery_scope).map(|g| g.as_str()), + contributor_id, + // Held unlisted until stage 3 completes (§6). + status: "pending", + content_id: Some(&cid), + origin, + ingested_from, + created_at: now, + }, + )?; + + // Actors are recorded by TMDB person id only. Those without one cannot be + // stored at all — there is no name column to put them in (§5a, §7) — so they + // are carried into the cast check via the submitted payload instead. + for actor in &valid.actor_scenes_cs { + if let Some(person_id) = actor.tmdb_id { + repo::insert_actor_scenes(tx, &manifest_id, person_id, &actor.scenes_cs)?; + } + } + + let payload = serde_json::to_string(&CastCheckJob { manifest_id: manifest_id.clone() })?; + repo::enqueue_job(tx, JOB_CAST_CHECK, &payload, now)?; + + Ok(IngestOutcome::Pending { manifest_id }) +} + +/// Computes the §9a `content_id` for a validated manifest. +pub fn compute_content_id(valid: &ValidManifest) -> String { + let m = &valid.manifest; + let identity = CanonicalIdentity { + kind: match m.identity.kind { + IdentityType::Movie => "movie", + IdentityType::Episode => "episode", + }, + tmdb_id: m.identity.effective_tmdb_id().map(str::to_string), + imdb_id: m.identity.effective_imdb_id().map(str::to_string), + season: m.identity.season, + episode: m.identity.episode, + }; + let cut = CanonicalCut { + runtime_cs: crate::validate::to_centiseconds(m.cut.runtime_sec), + video_hash: m.cut.video_hash.clone(), + }; + let actors: Vec = valid + .actor_scenes_cs + .iter() + .filter_map(|a| { + a.tmdb_id + .map(|id| CanonicalActor { tmdb_person_id: id, scenes_cs: a.scenes_cs.clone() }) + }) + .collect(); + + content_id::content_id(&identity, &cut, &actors) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::db::Db; + use crate::model::Jmanifest; + use crate::validate::validate_manifest; + + const NOW: &str = "2026-07-30T12:00:00Z"; + + fn valid_from(json: &str) -> ValidManifest { + let m: Jmanifest = serde_json::from_str(json).unwrap(); + validate_manifest(m).unwrap() + } + + fn movie_json(tmdb: &str, runtime: f64) -> String { + format!( + r#"{{"jmanifest_version":1, + "identity":{{"type":"movie","tmdb_id":"{tmdb}","title":"A Film"}}, + "cut":{{"runtime_sec":{runtime}}}, + "extraction":{{"sample_fps":5,"pipeline_version":"test 0.1"}}, + "actors":[{{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[10.0,20.0]]}}, + {{"name":"Michael Palin","tmdb_id":"11007","scenes":[[30.0,40.0]]}}]}}"# + ) + } + + #[tokio::test] + async fn persists_as_pending_and_enqueues_a_check() { + let db = Db::open(":memory:").unwrap(); + let valid = valid_from(&movie_json("504172", 6420.5)); + + let (outcome, status, jobs) = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + let outcome = persist(tx, &valid, Some(&c), "local", None, NOW)?; + let status = repo::manifest_status(tx, outcome.manifest_id())?; + let jobs = repo::lease_jobs(tx, NOW, 10)?; + Ok((outcome, status, jobs)) + }) + .await + .unwrap(); + + assert!(matches!(outcome, IngestOutcome::Pending { .. })); + // §6: held unlisted, not served to anyone, until stage 3 completes. + assert_eq!(status.unwrap().0, "pending"); + assert_eq!(jobs.len(), 1); + assert_eq!(jobs[0].kind, JOB_CAST_CHECK); + } + + #[tokio::test] + async fn a_pending_manifest_is_not_served() { + let db = Db::open(":memory:").unwrap(); + let valid = valid_from(&movie_json("504172", 6420.5)); + let candidates = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + persist(tx, &valid, Some(&c), "local", None, NOW)?; + let title = + repo::find_title(tx, IdentityType::Movie, Some("504172"), None)?.unwrap(); + repo::candidates_for_title(tx, &title.id, None, None) + }) + .await + .unwrap(); + assert!(candidates.is_empty()); + } + + #[tokio::test] + async fn identical_content_deduplicates() { + // §9a: a manifest whose `content_id` is already present is skipped + // without re-validation. + let db = Db::open(":memory:").unwrap(); + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from(&movie_json("504172", 6420.5)); + + let (first, second) = db + .write(move |tx| { + let c1 = repo::insert_contributor(tx, "h1", NOW)?; + let c2 = repo::insert_contributor(tx, "h2", NOW)?; + let first = persist(tx, &a, Some(&c1), "local", None, NOW)?; + // A *different* contributor, so this is content dedup, not the + // per-contributor 409. + let second = persist(tx, &b, Some(&c2), "local", None, NOW)?; + Ok((first, second)) + }) + .await + .unwrap(); + + assert!(matches!(first, IngestOutcome::Pending { .. })); + assert!(matches!(second, IngestOutcome::DuplicateContent { .. })); + assert_eq!(first.manifest_id(), second.manifest_id()); + } + + #[tokio::test] + async fn same_contributor_resubmitting_the_same_cut_is_a_duplicate() { + let db = Db::open(":memory:").unwrap(); + // Same identity and cut, different actor timings => different content_id, + // so this exercises the per-contributor 409 path specifically. + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from( + r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172","title":"A Film"}, + "cut":{"runtime_sec":6420.5}, + "actors":[{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[11.0,21.0]]}]}"#, + ); + + let second = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + persist(tx, &a, Some(&c), "local", None, NOW)?; + persist(tx, &b, Some(&c), "local", None, NOW) + }) + .await + .unwrap(); + + assert!(matches!(second, IngestOutcome::DuplicateFromContributor { .. })); + } + + #[tokio::test] + async fn different_cuts_of_one_title_coexist() { + // §7: multiple manifests may coexist for the same title with different + // cuts — that is the point. + let db = Db::open(":memory:").unwrap(); + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from(&movie_json("504172", 7000.0)); + + let (x, y) = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + let x = persist(tx, &a, Some(&c), "local", None, NOW)?; + let y = persist(tx, &b, Some(&c), "local", None, NOW)?; + Ok((x, y)) + }) + .await + .unwrap(); + + assert!(matches!(x, IngestOutcome::Pending { .. })); + assert!(matches!(y, IngestOutcome::Pending { .. })); + assert_ne!(x.manifest_id(), y.manifest_id()); + } + + #[tokio::test] + async fn episode_manifests_carry_their_coordinates() { + let db = Db::open(":memory:").unwrap(); + let valid = valid_from( + r#"{"jmanifest_version":1, + "identity":{"type":"episode","series_tmdb_id":"1396","title":"Breaking Bad", + "season":2,"episode":5}, + "cut":{"runtime_sec":2820.0}, + "actors":[{"name":"Bryan Cranston","tmdb_id":"17419","scenes":[[10.0,20.0]]}]}"#, + ); + let row = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + let o = persist(tx, &valid, Some(&c), "local", None, NOW)?; + Ok(repo::manifest_by_id(tx, o.manifest_id())?.unwrap()) + }) + .await + .unwrap(); + assert_eq!((row.season, row.episode), (Some(2), Some(5))); + } + + #[tokio::test] + async fn upload_metadata_is_not_echoed_back_as_actor_names() { + // §5a/§7: only integers reach the database. The submitted name is used + // for matching and never persisted, so before the cast check populates + // `people` there is no name to serve. + let db = Db::open(":memory:").unwrap(); + let valid = valid_from(&movie_json("504172", 6420.5)); + let actors = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + let o = persist(tx, &valid, Some(&c), "local", None, NOW)?; + repo::actors_for_manifest(tx, o.manifest_id()) + }) + .await + .unwrap(); + assert_eq!(actors.len(), 2); + assert!(actors.iter().all(|a| a.name.is_none())); + } + + #[test] + fn content_id_excludes_extraction_metadata() { + // §9a: `extraction` metadata and local state are excluded, so two + // servers validating the same upload agree. + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from( + r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172","title":"A Film"}, + "cut":{"runtime_sec":6420.5}, + "extraction":{"sample_fps":1,"extinction_sec":9,"pipeline_version":"other 9.9", + "gallery_size":5}, + "actors":[{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[10.0,20.0]]}, + {"name":"Michael Palin","tmdb_id":"11007","scenes":[[30.0,40.0]]}]}"#, + ); + assert_eq!(compute_content_id(&a), compute_content_id(&b)); + } + + #[test] + fn content_id_excludes_the_audio_signature() { + // §9a is explicit: including it would produce different content_ids for + // identical content and silently break federation deduplication. + let a = valid_from(&movie_json("504172", 6420.5)); + let sig = format!("v1:{}", "A".repeat(1720)); + let with_sig = format!( + r#"{{"jmanifest_version":1, + "identity":{{"type":"movie","tmdb_id":"504172","title":"A Film"}}, + "cut":{{"runtime_sec":6420.5,"audio_signature":"{sig}"}}, + "extraction":{{"sample_fps":5,"pipeline_version":"test 0.1"}}, + "actors":[{{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[10.0,20.0]]}}, + {{"name":"Michael Palin","tmdb_id":"11007","scenes":[[30.0,40.0]]}}]}}"# + ); + let b = valid_from(&with_sig); + assert_eq!(compute_content_id(&a), compute_content_id(&b)); + } + + #[test] + fn content_id_excludes_submitted_names() { + // Names are not persisted, so they must not be part of identity either — + // otherwise a renamed resubmission would evade deduplication. + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from( + r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172","title":"A Film"}, + "cut":{"runtime_sec":6420.5}, + "extraction":{"sample_fps":5,"pipeline_version":"test 0.1"}, + "actors":[{"name":"Someone Else","tmdb_id":"884","scenes":[[10.0,20.0]]}, + {"name":"Another Person","tmdb_id":"11007","scenes":[[30.0,40.0]]}]}"#, + ); + assert_eq!(compute_content_id(&a), compute_content_id(&b)); + } +} diff --git a/src/lib.rs b/src/lib.rs new file mode 100644 index 0000000..28ed871 --- /dev/null +++ b/src/lib.rs @@ -0,0 +1,38 @@ +//! JRay public server — a community manifest exchange (see `SPEC.md`). +//! +//! Jellyfin servers running the JRay plugin pull actor-timeline manifests +//! ("Jmanifests") for titles they own instead of running the CV pipeline +//! locally, and optionally contribute the manifests they generate back. +//! +//! Module map against the spec: +//! +//! | Module | Spec section | +//! |---|---| +//! | [`model`] | §2 Jmanifest format, and §6 stage 2's `deny_unknown_fields` | +//! | [`validate`] | §6 stage 2 semantics, §5a character class | +//! | [`matching`] | §3 cut matching tiers | +//! | [`content_id`] | §9a canonical form and content addressing | +//! | [`ingest`] | the shared upload path behind §4's `POST` endpoints | +//! | [`castcheck`] | §6 stage 3 scoring, §5a Threat 2 guards | +//! | [`worker`] | §6 stage 3 execution, §8 in-process background work | +//! | [`ratelimit`] | §5 | +//! | [`auth`] | §5a tokens, §8 trusted-proxy handling | +//! | [`db`] | §7 storage, §8 single-writer serialization | +//! | [`api`] | §4 | + +pub mod api; +pub mod app; +pub mod auth; +pub mod castcheck; +pub mod config; +pub mod content_id; +pub mod db; +pub mod error; +pub mod ingest; +pub mod matching; +pub mod model; +pub mod ratelimit; +pub mod state; +pub mod tmdb; +pub mod validate; +pub mod worker; diff --git a/src/main.rs b/src/main.rs new file mode 100644 index 0000000..345724c --- /dev/null +++ b/src/main.rs @@ -0,0 +1,96 @@ +//! Entry point. +//! +//! §8: one binary, one database file, one reverse proxy. The rate-limit counters +//! and the background cast-check worker both live in this process — no Redis, no +//! broker, no separate worker process. + +use std::net::SocketAddr; +use std::sync::Arc; + +use anyhow::Context; +use jray_server::app; +use jray_server::config::Config; +use jray_server::db::Db; +use jray_server::ratelimit::RateLimiter; +use jray_server::state::AppState; +use jray_server::tmdb::TmdbClient; +use jray_server::worker::Worker; +use tracing_subscriber::EnvFilter; + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + tracing_subscriber::fmt() + .with_env_filter( + EnvFilter::try_from_env("JRAY_LOG").unwrap_or_else(|_| EnvFilter::new("info")), + ) + .init(); + + let config = Arc::new(Config::from_env()?); + let db = Db::open(&config.db_path).context("opening database")?; + let tmdb = Arc::new(TmdbClient::new(config.tmdb_base_url.clone(), config.tmdb_api_key.clone())); + + if !tmdb.is_configured() { + // §8: TMDB is a hard dependency for UR-3. Uploads will accumulate in + // `pending` rather than being listed unverified — which is the correct + // failure mode, but the operator should know. + tracing::warn!("no JRAY_TMDB_API_KEY configured: uploads will stay pending, never listed"); + } + if config.trusted_proxies.is_empty() { + tracing::info!("no JRAY_TRUSTED_PROXIES set: X-Forwarded-For will be ignored"); + } + + let state = AppState { + db: db.clone(), + config: config.clone(), + limiter: Arc::new(RateLimiter::new()), + tmdb: tmdb.clone(), + }; + + let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); + + let worker = Worker { + db: db.clone(), + tmdb, + batch: config.job_batch, + poll_interval: config.job_poll_interval, + }; + let worker_handle = tokio::spawn(worker.run(shutdown_rx)); + + let listener = tokio::net::TcpListener::bind(&config.bind) + .await + .with_context(|| format!("binding {}", config.bind))?; + tracing::info!(bind = %config.bind, server_id = %config.server_id, "jray-server listening"); + + let router = app::router(state); + axum::serve(listener, router.into_make_service_with_connect_info::()) + .with_graceful_shutdown(async move { + shutdown_signal().await; + let _ = shutdown_tx.send(true); + }) + .await + .context("server error")?; + + let _ = worker_handle.await; + Ok(()) +} + +async fn shutdown_signal() { + let ctrl_c = async { + tokio::signal::ctrl_c().await.expect("installing ctrl-c handler"); + }; + + #[cfg(unix)] + let terminate = async { + tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate()) + .expect("installing SIGTERM handler") + .recv() + .await; + }; + #[cfg(not(unix))] + let terminate = std::future::pending::<()>(); + + tokio::select! { + _ = ctrl_c => tracing::info!("received ctrl-c, shutting down"), + _ = terminate => tracing::info!("received SIGTERM, shutting down"), + } +} diff --git a/src/matching.rs b/src/matching.rs new file mode 100644 index 0000000..b647d5d --- /dev/null +++ b/src/matching.rs @@ -0,0 +1,217 @@ +//! §3 cut matching. +//! +//! Timings only transfer between identical cuts, so matching is tiered and the +//! server reports *which* tier matched — a `loose` match is meant to surface as +//! a caveat in the JRay UI rather than being applied silently. +//! +//! `video_hash` identifies a *file*, so it only ever matches an identical +//! release and can never produce a false positive; that is why it is tier one. + +use crate::model::MatchTier; + +/// §3: runtimes within ±2s. +pub const RUNTIME_TOLERANCE_SEC: f64 = 2.0; +/// §3: runtimes within ±30s. +pub const LOOSE_TOLERANCE_SEC: f64 = 30.0; + +/// What the client tells us about its own copy. +#[derive(Debug, Clone, Default)] +pub struct ClientCut { + pub runtime_sec: Option, + pub video_hash: Option, +} + +impl ClientCut { + /// True when the client supplied nothing to match on, in which case §4 + /// specifies a `"match": "unknown"` answer rather than a guess. + pub fn is_empty(&self) -> bool { + self.runtime_sec.is_none() && self.video_hash.is_none() + } +} + +/// What the server holds. +#[derive(Debug, Clone)] +pub struct StoredCut { + pub runtime_sec: f64, + pub video_hash: Option, +} + +/// The outcome of comparing a client's cut against a stored one. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct CutMatch { + pub tier: MatchTier, + /// Scene offset in seconds the client must add (§3 `audio` tier). Always + /// zero for the tiers implemented here; the field exists because the plugin + /// contract is "the server returns the offset, the client applies it", and + /// enabling `audio` must not change the response shape. + pub offset_sec: f64, +} + +/// Compares a client's cut against a stored one, returning the best tier that +/// fires, or `None` for "beyond that: no match; do not serve" (§3). +pub fn match_cut(client: &ClientCut, stored: &StoredCut) -> Option { + // Tier 1 — same file. Checked first and unconditionally: an equal hash is + // decisive regardless of what the runtimes say. + if let (Some(c), Some(s)) = (&client.video_hash, &stored.video_hash) { + if c.eq_ignore_ascii_case(s) { + return Some(CutMatch { tier: MatchTier::Exact, offset_sec: 0.0 }); + } + } + + // `audio` tier would slot in here, above `runtime`, once signature coverage + // is useful (§3 recommended sequencing). + + if let Some(c_rt) = client.runtime_sec { + let delta = (c_rt - stored.runtime_sec).abs(); + if delta <= RUNTIME_TOLERANCE_SEC { + return Some(CutMatch { tier: MatchTier::Runtime, offset_sec: 0.0 }); + } + if delta <= LOOSE_TOLERANCE_SEC { + return Some(CutMatch { tier: MatchTier::Loose, offset_sec: 0.0 }); + } + // A runtime was supplied and cleared nothing — that is a definite + // no-match, not an unknown. + return None; + } + + // A hash that did not match, with no runtime to fall back on, tells us + // nothing about alignment either way. + if client.video_hash.is_some() { + return None; + } + + Some(CutMatch { tier: MatchTier::Unknown, offset_sec: 0.0 }) +} + +/// Picks the best-matching stored cut, if any clears `loose` (§4). +/// +/// `candidates` is `(key, cut)`; the key is returned so the caller can identify +/// which manifest won without re-scanning. +pub fn best_match( + client: &ClientCut, + candidates: &[(K, StoredCut)], +) -> Option<(K, CutMatch)> { + candidates + .iter() + .filter_map(|(k, cut)| match_cut(client, cut).map(|m| (k.clone(), m))) + .max_by(|a, b| a.1.tier.cmp(&b.1.tier)) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn stored(runtime: f64, hash: Option<&str>) -> StoredCut { + StoredCut { runtime_sec: runtime, video_hash: hash.map(str::to_string) } + } + + #[test] + fn equal_video_hash_is_exact() { + let c = ClientCut { + runtime_sec: Some(6420.5), + video_hash: Some("opensubtitles:8e245d9679d31e12".into()), + }; + let m = match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).unwrap(); + assert_eq!(m.tier, MatchTier::Exact); + } + + #[test] + fn equal_hash_wins_even_when_runtimes_disagree() { + // The hash identifies the file; a differing stored runtime means our + // own metadata is off, not that the file is different. + let c = ClientCut { + runtime_sec: Some(6000.0), + video_hash: Some("opensubtitles:8e245d9679d31e12".into()), + }; + let m = match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).unwrap(); + assert_eq!(m.tier, MatchTier::Exact); + } + + #[test] + fn hash_comparison_is_case_insensitive() { + let c = ClientCut { + runtime_sec: None, + video_hash: Some("opensubtitles:8E245D9679D31E12".into()), + }; + let m = match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).unwrap(); + assert_eq!(m.tier, MatchTier::Exact); + } + + #[test] + fn runtime_within_two_seconds_is_runtime_tier() { + let c = ClientCut { runtime_sec: Some(6422.0), video_hash: None }; + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Runtime); + } + + #[test] + fn runtime_within_thirty_seconds_is_loose() { + let c = ClientCut { runtime_sec: Some(6450.0), video_hash: None }; + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Loose); + } + + #[test] + fn beyond_thirty_seconds_does_not_match() { + // §3: "beyond that — no match; do not serve". + let c = ClientCut { runtime_sec: Some(6500.0), video_hash: None }; + assert!(match_cut(&c, &stored(6420.5, None)).is_none()); + } + + #[test] + fn tier_boundaries_are_inclusive() { + let c = ClientCut { runtime_sec: Some(6422.5), video_hash: None }; + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Runtime); + let c = ClientCut { runtime_sec: Some(6450.5), video_hash: None }; + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Loose); + } + + #[test] + fn no_cut_information_yields_unknown() { + // §4: the mode a library-wide sweep uses — "does the community have + // this title at all", with alignment still to be determined. + let c = ClientCut::default(); + assert!(c.is_empty()); + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Unknown); + } + + #[test] + fn non_matching_hash_alone_is_not_a_match() { + let c = ClientCut { + runtime_sec: None, + video_hash: Some("opensubtitles:ffffffffffffffff".into()), + }; + assert!(match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).is_none()); + } + + #[test] + fn non_matching_hash_falls_back_to_runtime() { + let c = ClientCut { + runtime_sec: Some(6421.0), + video_hash: Some("opensubtitles:ffffffffffffffff".into()), + }; + let m = match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).unwrap(); + assert_eq!(m.tier, MatchTier::Runtime); + } + + #[test] + fn best_match_prefers_the_highest_tier() { + let c = ClientCut { + runtime_sec: Some(6420.5), + video_hash: Some("opensubtitles:8e245d9679d31e12".into()), + }; + let candidates = vec![ + ("loose", stored(6445.0, None)), + ("exact", stored(9999.0, Some("opensubtitles:8e245d9679d31e12"))), + ("runtime", stored(6420.0, None)), + ]; + let (winner, m) = best_match(&c, &candidates).unwrap(); + assert_eq!(winner, "exact"); + assert_eq!(m.tier, MatchTier::Exact); + } + + #[test] + fn best_match_returns_none_when_nothing_clears_loose() { + let c = ClientCut { runtime_sec: Some(100.0), video_hash: None }; + let candidates = vec![("a", stored(6420.5, None)), ("b", stored(3000.0, None))]; + assert!(best_match(&c, &candidates).is_none()); + } +} diff --git a/src/model.rs b/src/model.rs new file mode 100644 index 0000000..5241824 --- /dev/null +++ b/src/model.rs @@ -0,0 +1,307 @@ +//! Jmanifest wire types (§2). +//! +//! **`#[serde(deny_unknown_fields)]` on every struct is the §6 stage 2 +//! enforcement mechanism.** "No additional fields anywhere" is a property of +//! these type definitions rather than of validator code that could omit a +//! field, so an unrecognised key at any nesting level fails to parse. That is +//! also what makes the §9 `movie`/`jellyfin_id` strip verifiable: a client that +//! forgets gets a hard `400` naming the field, rather than quietly publishing a +//! contributor's directory layout. +//! +//! Deeper semantic checks — bounds, character classes, path-shaped strings — +//! live in [`crate::validate`]. Parsing rejects *shape*; validation rejects +//! *content*. + +use serde::{Deserialize, Serialize}; + +/// Cut-match tier (§3). Ordered worst-to-best so derived `Ord` ranks them. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum MatchTier { + /// No cut information was supplied, so alignment is unknown (§4). + Unknown, + /// Audio 0.60–0.85, or runtimes within ±30s. Caveat in UI. + Loose, + /// Runtimes within ±2s. + Runtime, + /// Audio score ≥ 0.85; ranks above `runtime` because it is content-derived. + Audio, + /// `video_hash` equal — same file. + Exact, +} + +impl MatchTier { + pub fn as_str(self) -> &'static str { + match self { + MatchTier::Unknown => "unknown", + MatchTier::Loose => "loose", + MatchTier::Runtime => "runtime", + MatchTier::Audio => "audio", + MatchTier::Exact => "exact", + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum IdentityType { + Movie, + Episode, +} + +/// What the work is (§2 terminology: *title identity*). +/// +/// Movie and episode coordinates share one struct because `deny_unknown_fields` +/// with `#[serde(untagged)]` alternatives produces unhelpful error messages; +/// the discriminant is checked in [`crate::validate`], which can name the +/// offending field. +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Identity { + #[serde(rename = "type")] + pub kind: IdentityType, + + // Movie coordinates. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tmdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub imdb_id: Option, + + // Episode coordinates. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub series_tmdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub series_imdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub season: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub episode: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub title: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub year: Option, +} + +impl Identity { + /// The TMDB id used as the lookup key, whichever coordinate carries it. + pub fn effective_tmdb_id(&self) -> Option<&str> { + match self.kind { + IdentityType::Movie => self.tmdb_id.as_deref(), + IdentityType::Episode => self.series_tmdb_id.as_deref(), + } + } + + pub fn effective_imdb_id(&self) -> Option<&str> { + match self.kind { + IdentityType::Movie => self.imdb_id.as_deref(), + IdentityType::Episode => self.series_imdb_id.as_deref(), + } + } +} + +/// Which encode/edit the timings apply to (§2 terminology: *cut fingerprint*). +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Cut { + /// **Required** — the decoded duration of the media the timings came from. + /// The primary alignment guard (§2). + pub runtime_sec: f64, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub container_duration_sec: Option, + /// Optional but strongly preferred. OpenSubtitles hash (§3). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub video_hash: Option, + /// Optional; version-prefixed spectral-peak signature (§3, UR-9). + /// + /// Accepted and stored by this build; `audio`-tier matching is enabled once + /// coverage is useful, per §3's recommended sequencing. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub audio_signature: Option, +} + +/// How well the contributor's gallery could discriminate (§2, §7). +/// +/// The strongest available quality signal between two otherwise comparable +/// manifests: a `Global` gallery had to distinguish its actors from every other +/// actor in the contributor's library, whereas a `Limited` one only had to +/// distinguish them from this title's own cast. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Deserialize, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum GalleryScope { + /// Built from this title's cast alone. + Limited, + /// Built from the whole library. The default upstream. + Global, +} + +impl GalleryScope { + pub fn as_str(self) -> &'static str { + match self { + GalleryScope::Limited => "limited", + GalleryScope::Global => "global", + } + } +} + +/// Extraction parameters, carried for provenance and ranking. +/// +/// Note there is no `anneal_sec`: it was **withdrawn** in the SR-003 schema +/// bump, because presence now follows track extent — a track survives its own +/// gaps, so there is nothing to anneal (`scene-actor-extraction` AR-012/AR-013). +/// `deny_unknown_fields` therefore makes its presence a hard parse error rather +/// than something silently ignored, which is deliberate: a manifest still +/// carrying it was produced by a pipeline whose window semantics differ from +/// what this server now assumes. +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Extraction { + #[serde(default, skip_serializing_if = "Option::is_none")] + pub sample_fps: Option, + /// The re-acquisition timeout that shapes window extent. Successor to the + /// withdrawn `anneal_sec`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub extinction_sec: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub pipeline_version: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub gallery_size: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub gallery_scope: Option, +} + +/// One actor's timeline. +/// +/// Note there is no `jellyfin_id` field: `deny_unknown_fields` means its +/// presence is a parse error, which is exactly the §2/§6 requirement that it be +/// *rejected on upload* rather than merely ignored on download. +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Actor { + /// Sent on upload for matching, but **not persisted** — the server resolves + /// each actor to a TMDB person id and serves names from its own TMDB-derived + /// table (§2, §5a). On download this is server-authoritative. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub name: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub imdb_id: Option, + /// The **primary** actor join key (§2, §6 stage 3). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tmdb_id: Option, + /// `[start_sec, end_sec]` inclusive, sorted. + pub scenes: Vec<[f64; 2]>, +} + +/// One shareable actor timeline for one cut of one title (§2). +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Jmanifest { + pub jmanifest_version: u32, + pub identity: Identity, + pub cut: Cut, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub extraction: Option, + pub actors: Vec, +} + +/// Series-level coordinates for a bundle envelope (§2). +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct SeriesRef { + #[serde(default, skip_serializing_if = "Option::is_none")] + pub series_tmdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub series_imdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub title: Option, +} + +/// A thin wrapper, not a new format (§2). Bundles are a transfer convenience, +/// never a storage unit — each episode is stored and moderated individually. +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct SeriesBundle { + pub jmanifest_version: u32, + pub series: SeriesRef, + pub episodes: Vec, + /// Present on responses only; ignored on upload. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub coverage: Option, +} + +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Coverage { + pub episodes_available: usize, + pub seasons: Vec, +} + +/// The current `jmanifest_version` this server speaks (§2). +pub const JMANIFEST_VERSION: u32 = 1; + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn unknown_field_at_top_level_is_rejected() { + let json = r#"{"jmanifest_version":1,"identity":{"type":"movie","tmdb_id":"1"}, + "cut":{"runtime_sec":100.0},"actors":[],"surprise":"x"}"#; + let err = serde_json::from_str::(json).unwrap_err().to_string(); + assert!(err.contains("surprise"), "error should name the field: {err}"); + } + + #[test] + fn jellyfin_id_on_an_actor_is_a_parse_error() { + // §2: `actors[].jellyfin_id` must not appear. `deny_unknown_fields` + // makes this structural rather than a validator's responsibility. + let json = r#"{"jmanifest_version":1,"identity":{"type":"movie","tmdb_id":"1"}, + "cut":{"runtime_sec":100.0}, + "actors":[{"name":"A","tmdb_id":"2","jellyfin_id":"guid","scenes":[]}]}"#; + let err = serde_json::from_str::(json).unwrap_err().to_string(); + assert!(err.contains("jellyfin_id"), "error should name the field: {err}"); + } + + #[test] + fn movie_path_field_is_a_parse_error() { + let json = r#"{"jmanifest_version":1,"movie":"/data/movies/x.mkv", + "identity":{"type":"movie","tmdb_id":"1"}, + "cut":{"runtime_sec":100.0},"actors":[]}"#; + let err = serde_json::from_str::(json).unwrap_err().to_string(); + assert!(err.contains("movie"), "error should name the field: {err}"); + } + + #[test] + fn unknown_field_nested_in_cut_is_rejected() { + let json = r#"{"jmanifest_version":1,"identity":{"type":"movie","tmdb_id":"1"}, + "cut":{"runtime_sec":100.0,"payload":"x"},"actors":[]}"#; + assert!(serde_json::from_str::(json).is_err()); + } + + #[test] + fn spec_example_manifest_parses() { + let json = r#"{ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172", "imdb_id": "tt4686844", + "title": "The Death of Stalin", "year": 2017 }, + "cut": { "runtime_sec": 6420.5, "container_duration_sec": 6420.5, + "video_hash": "opensubtitles:8e245d9679d31e12" }, + "extraction": { "sample_fps": 5, "extinction_sec": 12, + "pipeline_version": "scene-actor-extraction 0.4.1", + "gallery_size": 1820, "gallery_scope": "global" }, + "actors": [ { "name": "Steve Buscemi", "imdb_id": "nm0000114", "tmdb_id": "884", + "scenes": [[191.6, 209.2], [438.2, 465.6]] } ] + }"#; + let m: Jmanifest = serde_json::from_str(json).unwrap(); + assert_eq!(m.actors.len(), 1); + assert_eq!(m.identity.effective_tmdb_id(), Some("504172")); + } + + #[test] + fn tiers_order_audio_above_runtime() { + // §3: `audio` ranks above `runtime` because it is content-derived. + assert!(MatchTier::Audio > MatchTier::Runtime); + assert!(MatchTier::Exact > MatchTier::Audio); + assert!(MatchTier::Runtime > MatchTier::Loose); + } +} diff --git a/src/ratelimit.rs b/src/ratelimit.rs new file mode 100644 index 0000000..66a4eba --- /dev/null +++ b/src/ratelimit.rs @@ -0,0 +1,217 @@ +//! §5 rate limiting. +//! +//! A fixed-window counter keyed on `(token_or_ip, surface)`, held in process +//! memory — no external counter store. §5 is explicit that a sliding window is +//! not worth the complexity at this volume, and that counters resetting on +//! restart is acceptable for abuse throttling. +//! +//! Read limits are applied *behind* the CDN cache, so a cache hit costs a client +//! nothing against its budget — that is a deployment property (§8), not +//! something this module can enforce. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +/// The rate-limited surfaces of §5. Distinct from routes: the batch and single +/// forms of `exists` are separate surfaces with separate budgets. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Surface { + ExistsSingle, + ExistsBatch, + ManifestFetch, + SeriesFetch, + ManifestUpload, + BundleUpload, + Report, + Search, +} + +impl Surface { + /// Requests per hour, per §5's table. + pub fn limit(self) -> u32 { + match self { + Surface::ExistsSingle => 600, + Surface::ExistsBatch => 60, + Surface::ManifestFetch => 300, + Surface::SeriesFetch => 120, + Surface::ManifestUpload => 100, + Surface::BundleUpload => 20, + Surface::Report => 20, + Surface::Search => 60, + } + } + + pub fn as_str(self) -> &'static str { + match self { + Surface::ExistsSingle => "exists", + Surface::ExistsBatch => "exists_batch", + Surface::ManifestFetch => "manifest_fetch", + Surface::SeriesFetch => "series_fetch", + Surface::ManifestUpload => "manifest_upload", + Surface::BundleUpload => "bundle_upload", + Surface::Report => "report", + Surface::Search => "search", + } + } +} + +const WINDOW: Duration = Duration::from_secs(3600); + +/// Headers §5 requires on every rate-limited response. +#[derive(Debug, Clone, Copy)] +pub struct Quota { + pub limit: u32, + pub remaining: u32, + /// Seconds until the window resets. + pub reset: u64, +} + +#[derive(Debug, Clone, Copy)] +struct Window { + started: Instant, + count: u32, +} + +pub struct RateLimiter { + windows: Mutex>, +} + +impl Default for RateLimiter { + fn default() -> Self { + Self::new() + } +} + +impl RateLimiter { + pub fn new() -> Self { + Self { windows: Mutex::new(HashMap::new()) } + } + + /// Records one request against `(key, surface)`. + /// + /// `Ok(quota)` when within budget, `Err(quota)` when the limit is exceeded — + /// in which case the caller returns `429` with `Retry-After` set from + /// `quota.reset`. A rejected request does **not** increment the counter, so a + /// client hammering a closed window cannot extend its own lockout. + pub fn check(&self, key: &str, surface: Surface) -> Result { + self.check_at(key, surface, Instant::now()) + } + + fn check_at(&self, key: &str, surface: Surface, now: Instant) -> Result { + let limit = surface.limit(); + let mut windows = self.windows.lock().expect("rate limiter poisoned"); + + // Opportunistic eviction of stale windows, so an IP-keyed map cannot + // grow without bound behind CGNAT. + if windows.len() > 10_000 { + windows.retain(|_, w| now.duration_since(w.started) < WINDOW); + } + + let entry = + windows.entry((key.to_string(), surface)).or_insert(Window { started: now, count: 0 }); + + let elapsed = now.duration_since(entry.started); + if elapsed >= WINDOW { + *entry = Window { started: now, count: 0 }; + } + + let reset = WINDOW.saturating_sub(now.duration_since(entry.started)).as_secs(); + + if entry.count >= limit { + return Err(Quota { limit, remaining: 0, reset }); + } + entry.count += 1; + Ok(Quota { limit, remaining: limit - entry.count, reset }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn allows_up_to_the_limit_then_rejects() { + let rl = RateLimiter::new(); + let limit = Surface::Report.limit(); + for i in 0..limit { + let q = rl.check("ip", Surface::Report).expect("within budget"); + assert_eq!(q.remaining, limit - i - 1); + } + let q = rl.check("ip", Surface::Report).expect_err("over budget"); + assert_eq!(q.remaining, 0); + } + + #[test] + fn surfaces_have_independent_budgets() { + let rl = RateLimiter::new(); + for _ in 0..Surface::BundleUpload.limit() { + rl.check("t", Surface::BundleUpload).unwrap(); + } + assert!(rl.check("t", Surface::BundleUpload).is_err()); + // §5: a bundle counts as a single write against its own limit, and must + // not consume the single-manifest budget. + assert!(rl.check("t", Surface::ManifestUpload).is_ok()); + } + + #[test] + fn keys_are_independent() { + let rl = RateLimiter::new(); + for _ in 0..Surface::Report.limit() { + rl.check("a", Surface::Report).unwrap(); + } + assert!(rl.check("a", Surface::Report).is_err()); + assert!(rl.check("b", Surface::Report).is_ok()); + } + + #[test] + fn window_resets_after_an_hour() { + let rl = RateLimiter::new(); + let t0 = Instant::now(); + for _ in 0..Surface::Report.limit() { + rl.check_at("ip", Surface::Report, t0).unwrap(); + } + assert!(rl.check_at("ip", Surface::Report, t0).is_err()); + // Still closed just inside the window. + assert!(rl.check_at("ip", Surface::Report, t0 + Duration::from_secs(3599)).is_err()); + // Open again once it rolls over. + assert!(rl.check_at("ip", Surface::Report, t0 + Duration::from_secs(3600)).is_ok()); + } + + #[test] + fn rejected_requests_do_not_extend_the_lockout() { + let rl = RateLimiter::new(); + let t0 = Instant::now(); + for _ in 0..Surface::Report.limit() { + rl.check_at("ip", Surface::Report, t0).unwrap(); + } + // Hammer the closed window; the counter must not keep climbing, so the + // window still expires on schedule. + for _ in 0..50 { + assert!(rl.check_at("ip", Surface::Report, t0 + Duration::from_secs(10)).is_err()); + } + assert!(rl.check_at("ip", Surface::Report, t0 + Duration::from_secs(3600)).is_ok()); + } + + #[test] + fn reset_counts_down_within_the_window() { + let rl = RateLimiter::new(); + let t0 = Instant::now(); + let q = rl.check_at("ip", Surface::ExistsSingle, t0).unwrap(); + assert_eq!(q.reset, 3600); + let q = rl.check_at("ip", Surface::ExistsSingle, t0 + Duration::from_secs(600)).unwrap(); + assert_eq!(q.reset, 3000); + } + + #[test] + fn limits_match_the_spec_table() { + assert_eq!(Surface::ExistsSingle.limit(), 600); + assert_eq!(Surface::ExistsBatch.limit(), 60); + assert_eq!(Surface::ManifestFetch.limit(), 300); + assert_eq!(Surface::SeriesFetch.limit(), 120); + assert_eq!(Surface::ManifestUpload.limit(), 100); + assert_eq!(Surface::BundleUpload.limit(), 20); + assert_eq!(Surface::Report.limit(), 20); + assert_eq!(Surface::Search.limit(), 60); + } +} diff --git a/src/state.rs b/src/state.rs new file mode 100644 index 0000000..809a78a --- /dev/null +++ b/src/state.rs @@ -0,0 +1,102 @@ +//! Shared application state, and the cross-cutting request concerns (§5 rate +//! limiting, §5a token resolution) that every handler needs. + +use std::net::SocketAddr; +use std::sync::Arc; + +use axum::extract::ConnectInfo; +use axum::http::{HeaderMap, HeaderValue}; +use axum::response::Response; + +use crate::auth; +use crate::config::Config; +use crate::db::{repo, Db}; +use crate::error::{ApiError, ApiResult}; +use crate::ratelimit::{Quota, RateLimiter, Surface}; +use crate::tmdb::TmdbClient; + +/// The connection's peer address, when the server was started with connect-info. +/// +/// A dedicated extractor rather than `ConnectInfo` directly, because +/// this must not be a *hard* requirement: a router used without +/// `into_make_service_with_connect_info` — as in tests — has no peer address, and +/// a handler that fails to extract would be a routing error rather than degrading +/// to header-only attribution. +pub struct PeerIp(pub Option); + +impl axum::extract::FromRequestParts for PeerIp +where + S: Send + Sync, +{ + type Rejection = std::convert::Infallible; + + async fn from_request_parts( + parts: &mut axum::http::request::Parts, + _state: &S, + ) -> Result { + Ok(PeerIp( + parts.extensions.get::>().map(|ConnectInfo(addr)| addr.ip()), + )) + } +} + +#[derive(Clone)] +pub struct AppState { + pub db: Db, + pub config: Arc, + pub limiter: Arc, + pub tmdb: Arc, +} + +impl AppState { + /// Resolves the client IP for rate-limiting and attribution, honouring + /// `X-Forwarded-For` only from a configured proxy (§8). + pub fn client_ip(&self, headers: &HeaderMap, peer: Option) -> String { + auth::client_ip(headers, peer, &self.config.trusted_proxies) + } + + /// §5: limits are per token where one is present, otherwise per source IP. + pub fn check_limit(&self, key: &str, surface: Surface) -> ApiResult { + self.limiter.check(key, surface).map_err(|q| { + tracing::debug!(surface = surface.as_str(), "rate limited"); + ApiError::RateLimited { retry_after: q.reset.max(1) } + }) + } + + /// Resolves a bearer token to a contributor (§5a). + /// + /// A token is an anonymous bearer capability, not an account: the only state + /// behind it is the per-token counters used for rate-limiting attribution and + /// automatic revocation. + pub async fn require_contributor(&self, headers: &HeaderMap) -> ApiResult { + let token = auth::bearer_token(headers).ok_or(ApiError::Unauthorized)?; + let hash = auth::hash_token(&token); + let found = self + .db + .read(move |conn| repo::contributor_by_token_hash(conn, &hash)) + .await + .map_err(ApiError::Internal)?; + + match found { + Some(c) if !c.revoked => Ok(c), + // A revoked token is indistinguishable from an unknown one to the + // caller; there is nothing useful to disclose. + _ => Err(ApiError::Unauthorized), + } + } +} + +/// Attaches the §5 rate-limit headers to a response. +pub fn with_quota_headers(mut resp: Response, quota: Quota) -> Response { + let h = resp.headers_mut(); + insert_num(h, "x-ratelimit-limit", quota.limit as u64); + insert_num(h, "x-ratelimit-remaining", quota.remaining as u64); + insert_num(h, "x-ratelimit-reset", quota.reset); + resp +} + +fn insert_num(headers: &mut HeaderMap, name: &'static str, value: u64) { + if let Ok(v) = HeaderValue::from_str(&value.to_string()) { + headers.insert(name, v); + } +} diff --git a/src/tmdb.rs b/src/tmdb.rs new file mode 100644 index 0000000..a279629 --- /dev/null +++ b/src/tmdb.rs @@ -0,0 +1,226 @@ +//! TMDB client for the §6 stage 3 cast cross-check. +//! +//! §5a's Threat 2 defence rests entirely on the attacker not controlling TMDB: +//! to make a prank manifest pass, they would need those performers to be +//! credited cast on that title in TMDB, which means vandalising a separate, +//! moderated system. +//! +//! Responses are cached for 24h (§6) so a burst of episode uploads for one +//! series costs a single upstream call, and so the server stays within TMDB's +//! own rate limits. + +use std::time::Duration; + +use serde::Deserialize; + +/// A credited cast member, reduced to what the check needs. +#[derive(Debug, Clone, Deserialize)] +pub struct CastMember { + pub id: u64, + #[serde(default)] + pub name: String, + #[serde(default)] + pub adult: bool, +} + +#[derive(Debug, Clone, Default, Deserialize)] +pub struct Credits { + #[serde(default)] + pub cast: Vec, + /// Present on episode credits. + #[serde(default)] + pub guest_stars: Vec, +} + +impl Credits { + /// Cast plus guest stars — the union §6 specifies for episodes. + pub fn all(&self) -> impl Iterator { + self.cast.iter().chain(self.guest_stars.iter()) + } +} + +#[derive(Debug, Clone, Default, Deserialize)] +pub struct TitleDetails { + #[serde(default)] + pub adult: bool, + #[serde(default)] + pub title: Option, + #[serde(default)] + pub name: Option, +} + +/// A failure that should be retried rather than treated as a verdict. +/// +/// §6: "TMDB unreachable / rate-limited → retry with backoff; stays unlisted, +/// not rejected." Distinguishing this from "TMDB has no credits" is essential — +/// conflating them would reject honest manifests during an outage. +#[derive(Debug, thiserror::Error)] +pub enum TmdbError { + #[error("tmdb transport error: {0}")] + Transport(String), + #[error("tmdb rate limited")] + RateLimited, + #[error("tmdb server error: {0}")] + ServerError(u16), + /// The id genuinely does not exist upstream. + #[error("tmdb resource not found")] + NotFound, + #[error("tmdb response was not understood: {0}")] + Malformed(String), + #[error("no tmdb api key configured")] + NotConfigured, +} + +impl TmdbError { + /// True when the job should be rescheduled rather than resolved. + pub fn is_retryable(&self) -> bool { + matches!( + self, + TmdbError::Transport(_) + | TmdbError::RateLimited + | TmdbError::ServerError(_) + | TmdbError::NotConfigured + ) + } +} + +#[derive(Clone)] +pub struct TmdbClient { + http: reqwest::Client, + base_url: String, + api_key: Option, +} + +impl TmdbClient { + pub fn new(base_url: String, api_key: Option) -> Self { + let http = reqwest::Client::builder() + .timeout(Duration::from_secs(15)) + .user_agent(concat!("jray-server/", env!("CARGO_PKG_VERSION"))) + .build() + .expect("building reqwest client"); + Self { http, base_url, api_key } + } + + pub fn is_configured(&self) -> bool { + self.api_key.is_some() + } + + async fn get(&self, path: &str) -> Result { + let key = self.api_key.as_deref().ok_or(TmdbError::NotConfigured)?; + let url = + format!("{}/{}", self.base_url.trim_end_matches('/'), path.trim_start_matches('/')); + + let resp = self + .http + .get(&url) + .query(&[("api_key", key)]) + .send() + .await + .map_err(|e| TmdbError::Transport(e.to_string()))?; + + let status = resp.status(); + if status == reqwest::StatusCode::NOT_FOUND { + return Err(TmdbError::NotFound); + } + if status == reqwest::StatusCode::TOO_MANY_REQUESTS { + return Err(TmdbError::RateLimited); + } + if status.is_server_error() { + return Err(TmdbError::ServerError(status.as_u16())); + } + if !status.is_success() { + return Err(TmdbError::Malformed(format!("unexpected status {status}"))); + } + + let body = resp.text().await.map_err(|e| TmdbError::Transport(e.to_string()))?; + serde_json::from_str(&body).map_err(|e| TmdbError::Malformed(e.to_string())) + } + + pub async fn movie_credits(&self, tmdb_id: &str) -> Result { + self.get(&format!("movie/{tmdb_id}/credits")).await + } + + pub async fn movie_details(&self, tmdb_id: &str) -> Result { + self.get(&format!("movie/{tmdb_id}")).await + } + + pub async fn series_credits(&self, series_tmdb_id: &str) -> Result { + // Aggregate credits carry recurring cast TMDB lists only at series level. + self.get(&format!("tv/{series_tmdb_id}/aggregate_credits")).await + } + + pub async fn episode_credits( + &self, + series_tmdb_id: &str, + season: i64, + episode: i64, + ) -> Result { + self.get(&format!("tv/{series_tmdb_id}/season/{season}/episode/{episode}/credits")).await + } + + pub async fn series_details(&self, series_tmdb_id: &str) -> Result { + self.get(&format!("tv/{series_tmdb_id}")).await + } +} + +/// Fetches a person's details, used by the §5a category guard. +#[derive(Debug, Clone, Default, Deserialize)] +pub struct PersonDetails { + #[serde(default)] + pub adult: bool, + #[serde(default)] + pub name: String, +} + +impl TmdbClient { + pub async fn person(&self, tmdb_person_id: u64) -> Result { + self.get(&format!("person/{tmdb_person_id}")).await + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn credits_union_covers_cast_and_guest_stars() { + // §6: for episodes the check runs against the union of per-episode + // credits (cast + guest stars) and series aggregate credits. + let c: Credits = serde_json::from_str( + r#"{"cast":[{"id":1,"name":"A"}],"guest_stars":[{"id":2,"name":"B"}]}"#, + ) + .unwrap(); + let ids: Vec = c.all().map(|m| m.id).collect(); + assert_eq!(ids, vec![1, 2]); + } + + #[test] + fn credits_tolerate_missing_and_extra_fields() { + // TMDB adds fields freely; our own strictness applies to *uploads*, not + // to a trusted upstream we merely read. + let c: Credits = + serde_json::from_str(r#"{"cast":[{"id":1,"unexpected":true}],"id":99}"#).unwrap(); + assert_eq!(c.cast.len(), 1); + assert_eq!(c.cast[0].name, ""); + assert!(c.guest_stars.is_empty()); + } + + #[test] + fn transport_and_rate_limit_are_retryable_but_not_found_is_not() { + // The distinction that keeps an outage from rejecting honest uploads. + assert!(TmdbError::Transport("x".into()).is_retryable()); + assert!(TmdbError::RateLimited.is_retryable()); + assert!(TmdbError::ServerError(503).is_retryable()); + assert!(TmdbError::NotConfigured.is_retryable()); + assert!(!TmdbError::NotFound.is_retryable()); + assert!(!TmdbError::Malformed("x".into()).is_retryable()); + } + + #[tokio::test] + async fn unconfigured_client_reports_retryable_failure() { + let c = TmdbClient::new("http://127.0.0.1:1".into(), None); + assert!(!c.is_configured()); + let err = c.movie_credits("1").await.unwrap_err(); + assert!(err.is_retryable(), "missing key must hold uploads pending, not reject them"); + } +} diff --git a/src/validate.rs b/src/validate.rs new file mode 100644 index 0000000..f1993b3 --- /dev/null +++ b/src/validate.rs @@ -0,0 +1,1055 @@ +//! §6 stage 2 semantic validation, and the §5a character-class constraint. +//! +//! Shape is already enforced by `deny_unknown_fields` at parse time +//! ([`crate::model`]); everything here is *content*. Rejections name the +//! offending field, per §6. + +use unicode_general_category::{get_general_category, GeneralCategory}; +use unicode_normalization::{is_nfc, UnicodeNormalization}; + +use crate::model::{Actor, Identity, IdentityType, Jmanifest, SeriesBundle, JMANIFEST_VERSION}; + +/// §6 stage 1 caps. Body-size limits are applied as a layer (see [`crate::app`]); +/// these are the structural counts the schema layer enforces. +pub mod limits { + pub const MAX_ACTORS: usize = 500; + pub const MAX_SCENES_PER_ACTOR: usize = 2000; + pub const MAX_TOTAL_SCENES: usize = 20_000; + pub const MAX_NAME_CHARS: usize = 200; + pub const MAX_TITLE_CHARS: usize = 300; + /// §2 bundle caps. + pub const MAX_BUNDLE_EPISODES: usize = 500; + /// Times beyond `runtime_sec` + this tolerance are rejected (§6). + pub const RUNTIME_TOLERANCE_SEC: f64 = 5.0; + /// §6 stage 1 body caps, in bytes. + pub const BODY_LIMIT_MANIFEST: usize = 2 * 1024 * 1024; + pub const BODY_LIMIT_BUNDLE: usize = 25 * 1024 * 1024; + /// §3: fixed signature length. The tolerance covers seek and encoder + /// differences at the window edges — it is not a licence to vary the length. + pub const AUDIO_SIG_FRAMES: usize = 1290; + pub const AUDIO_SIG_TOLERANCE: usize = 32; +} + +#[derive(Debug, thiserror::Error)] +#[error("{field}: {reason}")] +pub struct ValidationError { + pub field: String, + pub reason: String, +} + +fn err(field: impl Into, reason: impl Into) -> ValidationError { + ValidationError { field: field.into(), reason: reason.into() } +} + +type VResult = Result; + +/// §3 / IR-007: the analysis window is `runtime/2 ± 60 s`, so below 120 s it +/// underflows and **no signature is emitted**. The rule is identical in both +/// producers and here; a rule that differs between them yields signatures that +/// never match. +pub const AUDIO_SIG_MIN_RUNTIME_SEC: f64 = 120.0; + +/// A manifest that has passed §6 stage 2. Carries the normalised forms so +/// downstream stages do not re-derive them. +#[derive(Debug, Clone)] +pub struct ValidManifest { + pub manifest: Jmanifest, + /// Scene windows quantised to integer centiseconds (§7, §9a) — the same + /// quantisation used for `content_id`, so stored and hashed values cannot + /// diverge. + pub actor_scenes_cs: Vec, +} + +#[derive(Debug, Clone)] +pub struct ActorScenes { + /// NFC-normalised name, used only for matching in stage 3 and then dropped. + pub name: Option, + pub tmdb_id: Option, + pub imdb_id: Option, + pub scenes_cs: Vec<(i64, i64)>, +} + +/// Quantises seconds to whole centiseconds (§9a). +/// +/// Integer centiseconds remove the float-canonicalisation failure mode rather +/// than dodging it: pipeline timings are *derived* by accumulating `1/fps`, so +/// they carry accumulated error, and any value near a rounding boundary would +/// otherwise hash differently on two servers. +pub fn to_centiseconds(secs: f64) -> i64 { + (secs * 100.0).round() as i64 +} + +// --------------------------------------------------------------------------- +// Identifier formats (§6 stage 2) +// --------------------------------------------------------------------------- + +/// `^tt\d{7,8}$` +fn is_title_imdb_id(s: &str) -> bool { + let Some(digits) = s.strip_prefix("tt") else { return false }; + matches!(digits.len(), 7 | 8) && digits.bytes().all(|b| b.is_ascii_digit()) +} + +/// `^nm\d{7,8}$` +fn is_person_imdb_id(s: &str) -> bool { + let Some(digits) = s.strip_prefix("nm") else { return false }; + matches!(digits.len(), 7 | 8) && digits.bytes().all(|b| b.is_ascii_digit()) +} + +/// `^\d{1,9}$` +fn is_tmdb_id(s: &str) -> bool { + !s.is_empty() && s.len() <= 9 && s.bytes().all(|b| b.is_ascii_digit()) +} + +// --------------------------------------------------------------------------- +// §5a free-text constraints +// --------------------------------------------------------------------------- + +/// §5a character class: Unicode letters, marks, spaces, and `. ' - ,` only. +/// +/// No digits and no `/ + =`, which is what **defeats base64/hex smuggling**. No +/// control characters, and no zero-width or bidi-control codepoints. +fn is_allowed_text_char(c: char) -> bool { + if matches!(c, '.' | '\'' | '-' | ',' | ' ') { + return true; + } + // Explicitly excluded regardless of category: zero-width and bidi controls. + if matches!(c, '\u{200B}'..='\u{200F}' | '\u{202A}'..='\u{202E}' + | '\u{2060}'..='\u{2064}' | '\u{2066}'..='\u{2069}' | '\u{FEFF}') + { + return false; + } + if !matches!( + get_general_category(c), + GeneralCategory::UppercaseLetter + | GeneralCategory::LowercaseLetter + | GeneralCategory::TitlecaseLetter + | GeneralCategory::ModifierLetter + | GeneralCategory::OtherLetter + | GeneralCategory::NonspacingMark + | GeneralCategory::SpacingMark + | GeneralCategory::EnclosingMark + ) { + return false; + } + + // Reject *compatibility* variants of otherwise-allowed letters — mathematical + // bold (`𝐒`), fullwidth (`A`), enclosed and other presentation forms. + // + // These are genuine letters by category, so the check above admits them, and + // NFC does not fold them (only NFKC would). They matter for two reasons: + // homoglyph spoofing of a real person's name, and the fact that a + // fullwidth-digit alphabet would reopen the very encoding channel §5a's "no + // digits" rule closes. A character that NFKC would rewrite is not the + // character it appears to be, so it is not accepted. + // + // Names are stored as TMDB references anyway (§5a), so the cost of being + // strict here is nil: a real TMDB name is already in normal form. + !is_compatibility_variant(c) +} + +/// True when NFKC rewrites `c` into something other than itself. +fn is_compatibility_variant(c: char) -> bool { + let mut it = c.nfkc(); + match (it.next(), it.next()) { + (Some(first), None) => first != c, + // Decomposes to several characters, so it is certainly not canonical. + (Some(_), Some(_)) => true, + (None, _) => true, + } +} + +/// Validates a free-text field against §5a's permissive-but-closed pattern and +/// returns it NFC-normalised. +fn check_text(field: &str, value: &str, max_chars: usize) -> VResult { + if value.chars().count() > max_chars { + return Err(err(field, format!("longer than {max_chars} characters"))); + } + let normalised: String = if is_nfc(value) { value.to_string() } else { value.nfc().collect() }; + if normalised.chars().count() > max_chars { + return Err(err(field, format!("longer than {max_chars} characters after NFC"))); + } + if let Some(bad) = normalised.chars().find(|c| !is_allowed_text_char(*c)) { + return Err(err(field, format!("contains disallowed character U+{:04X}", bad as u32))); + } + Ok(normalised) +} + +/// §6: reject any string anywhere that looks like an absolute filesystem path +/// or a `file://` URI. +/// +/// `movie` itself is already a parse error via `deny_unknown_fields`; this +/// closes the same leak arriving through a field that *is* allowed. +fn check_not_path_shaped(field: &str, value: &str) -> VResult<()> { + let v = value.trim(); + let looks_like_path = v.starts_with('/') + || v.starts_with("\\\\") + || v.to_ascii_lowercase().starts_with("file://") + || (v.len() >= 3 + && v.as_bytes()[0].is_ascii_alphabetic() + && v.as_bytes()[1] == b':' + && matches!(v.as_bytes()[2], b'\\' | b'/')); + if looks_like_path { + return Err(err(field, "looks like a filesystem path or file:// URI")); + } + Ok(()) +} + +// --------------------------------------------------------------------------- +// Manifest validation +// --------------------------------------------------------------------------- + +pub fn validate_manifest(mut m: Jmanifest) -> VResult { + if m.jmanifest_version != JMANIFEST_VERSION { + return Err(err( + "jmanifest_version", + format!("unsupported version {}, expected {JMANIFEST_VERSION}", m.jmanifest_version), + )); + } + + validate_identity(&mut m.identity)?; + validate_cut(&m)?; + + if let Some(ex) = &m.extraction { + if let Some(pv) = &ex.pipeline_version { + check_not_path_shaped("extraction.pipeline_version", pv)?; + if pv.chars().count() > limits::MAX_TITLE_CHARS { + return Err(err("extraction.pipeline_version", "too long")); + } + } + if let Some(fps) = ex.sample_fps { + if !fps.is_finite() || fps <= 0.0 { + return Err(err("extraction.sample_fps", "must be a positive finite number")); + } + } + if let Some(a) = ex.extinction_sec { + if !a.is_finite() || a < 0.0 { + return Err(err( + "extraction.extinction_sec", + "must be a non-negative finite number", + )); + } + } + } + + let actor_scenes_cs = validate_actors(&m)?; + Ok(ValidManifest { manifest: m, actor_scenes_cs }) +} + +fn validate_identity(id: &mut Identity) -> VResult<()> { + match id.kind { + IdentityType::Movie => { + if id.series_tmdb_id.is_some() || id.series_imdb_id.is_some() { + return Err(err("identity.series_tmdb_id", "not valid for type=movie")); + } + if id.season.is_some() || id.episode.is_some() { + return Err(err("identity.season", "not valid for type=movie")); + } + // §6: neither `tmdb_id` nor `imdb_id` in `identity`. + if id.tmdb_id.is_none() && id.imdb_id.is_none() { + return Err(err("identity", "requires at least one of tmdb_id or imdb_id")); + } + if let Some(t) = &id.tmdb_id { + if !is_tmdb_id(t) { + return Err(err("identity.tmdb_id", "must match ^\\d{1,9}$")); + } + } + if let Some(i) = &id.imdb_id { + if !is_title_imdb_id(i) { + return Err(err("identity.imdb_id", "must match ^tt\\d{7,8}$")); + } + } + } + IdentityType::Episode => { + if id.tmdb_id.is_some() || id.imdb_id.is_some() { + return Err(err( + "identity.tmdb_id", + "use series_tmdb_id / series_imdb_id for type=episode", + )); + } + if id.series_tmdb_id.is_none() && id.series_imdb_id.is_none() { + return Err(err( + "identity", + "requires at least one of series_tmdb_id or series_imdb_id", + )); + } + if let Some(t) = &id.series_tmdb_id { + if !is_tmdb_id(t) { + return Err(err("identity.series_tmdb_id", "must match ^\\d{1,9}$")); + } + } + if let Some(i) = &id.series_imdb_id { + if !is_title_imdb_id(i) { + return Err(err("identity.series_imdb_id", "must match ^tt\\d{7,8}$")); + } + } + // Bounded integers (§5a). + match id.season { + Some(s) if (0..=1000).contains(&s) => {} + Some(_) => return Err(err("identity.season", "out of range 0..=1000")), + None => return Err(err("identity.season", "required for type=episode")), + } + match id.episode { + Some(e) if (0..=10_000).contains(&e) => {} + Some(_) => return Err(err("identity.episode", "out of range 0..=10000")), + None => return Err(err("identity.episode", "required for type=episode")), + } + } + } + + if let Some(y) = id.year { + if !(1870..=2200).contains(&y) { + return Err(err("identity.year", "out of range 1870..=2200")); + } + } + if let Some(t) = &id.title { + check_not_path_shaped("identity.title", t)?; + id.title = Some(check_text("identity.title", t, limits::MAX_TITLE_CHARS)?); + } + Ok(()) +} + +fn validate_cut(m: &Jmanifest) -> VResult<()> { + let rt = m.cut.runtime_sec; + // §2: `cut.runtime_sec` is required — absence is already a parse error, so + // what remains is range. + if !rt.is_finite() || rt <= 0.0 || rt > 200_000.0 { + return Err(err("cut.runtime_sec", "must be a finite duration in (0, 200000]")); + } + if let Some(d) = m.cut.container_duration_sec { + if !d.is_finite() || d <= 0.0 || d > 200_000.0 { + return Err(err("cut.container_duration_sec", "must be a finite duration")); + } + } + if let Some(h) = &m.cut.video_hash { + validate_video_hash(h)?; + } + if let Some(sig) = &m.cut.audio_signature { + validate_audio_signature(sig, rt)?; + } + Ok(()) +} + +/// §3: the OpenSubtitles hash, in the fixed `opensubtitles:<16 hex>` form. +fn validate_video_hash(h: &str) -> VResult<()> { + let Some(hex) = h.strip_prefix("opensubtitles:") else { + return Err(err("cut.video_hash", "must be prefixed 'opensubtitles:'")); + }; + if hex.len() != 16 || !hex.bytes().all(|b| b.is_ascii_hexdigit()) { + return Err(err("cut.video_hash", "expected 16 hex digits after the prefix")); + } + Ok(()) +} + +/// §3 "Validation and abuse": fixed length, base64, and each byte structurally +/// constrained (5-bit bin index + 2-bit energy class). +/// +/// A variable-length blob would be a payload channel — precisely what §5a +/// closes — so length is checked, not merely bounded. +/// +/// `runtime_sec` is needed because §3 shortens the window for very short items; +/// see [`expected_min_frames`]. +pub fn validate_audio_signature(sig: &str, runtime_sec: f64) -> VResult<()> { + // IR-007: media shorter than the window emits **no signature**, and no sync + // offset is applied to it. A signature present on such an item did not come + // from the specified construction, so it is rejected rather than stored — + // whatever it is, it is not the thing this field is for. + if runtime_sec < AUDIO_SIG_MIN_RUNTIME_SEC { + return Err(err( + "cut.audio_signature", + format!( + "must not be present for media shorter than {AUDIO_SIG_MIN_RUNTIME_SEC:.0}s — \ + the {AUDIO_SIG_MIN_RUNTIME_SEC:.0}s analysis window underflows" + ), + )); + } + + let Some(payload) = sig.strip_prefix("v1:") else { + return Err(err("cut.audio_signature", "must be version-prefixed 'v1:'")); + }; + let bytes = base64_decode(payload) + .map_err(|e| err("cut.audio_signature", format!("invalid base64: {e}")))?; + + // §3, and `scene-actor-extraction` IR-007: the length is **fixed by the + // construction**, not merely bounded. A 120 s window at a 1024-sample hop and + // 11025 Hz yields ~1290 frames, and an item too short for that window emits + // no signature at all — the window `runtime/2 ± 60 s` underflows below 120 s, + // so there is nothing to shorten. + // + // That makes the length non-negotiable, which is what keeps the field inside + // SR-004: a caller cannot choose it, so it cannot be used as a variable-size + // container. The tolerance covers seek and encoder differences at the window + // edges, nothing more. + let lo = limits::AUDIO_SIG_FRAMES - limits::AUDIO_SIG_TOLERANCE; + let hi = limits::AUDIO_SIG_FRAMES + limits::AUDIO_SIG_TOLERANCE; + + if bytes.len() < lo || bytes.len() > hi { + return Err(err( + "cut.audio_signature", + format!("decoded length {} outside the fixed {lo}..={hi} frames", bytes.len()), + )); + } + // 5-bit bin index (0..=31) + 2-bit energy class => bit 7 must be clear. + // Arbitrary bytes are therefore invalid, keeping §5a's "no free-form + // storage" property intact. + if let Some(pos) = bytes.iter().position(|b| b & 0x80 != 0) { + return Err(err("cut.audio_signature", format!("frame {pos} has reserved high bit set"))); + } + Ok(()) +} + +fn validate_actors(m: &Jmanifest) -> VResult> { + if m.actors.len() > limits::MAX_ACTORS { + return Err(err("actors", format!("more than {} entries", limits::MAX_ACTORS))); + } + // §6 small-|M| handling: `|M| == 0` is rejected. 15 of the 331 corpus files + // have empty actor lists — extraction failures, not contributions. + if m.actors.is_empty() { + return Err(err("actors", "empty actor list is an extraction failure, not a contribution")); + } + + let mut out = Vec::with_capacity(m.actors.len()); + let mut total_scenes = 0usize; + let mut seen_tmdb: Vec = Vec::new(); + let mut seen_imdb: Vec = Vec::new(); + + for (i, a) in m.actors.iter().enumerate() { + let scenes_cs = validate_scenes(i, a, m.cut.runtime_sec)?; + total_scenes += scenes_cs.len(); + if total_scenes > limits::MAX_TOTAL_SCENES { + return Err(err( + "actors", + format!("more than {} scene windows in total", limits::MAX_TOTAL_SCENES), + )); + } + + let tmdb_id = match &a.tmdb_id { + Some(t) if !t.is_empty() => { + if !is_tmdb_id(t) { + return Err(err(format!("actors[{i}].tmdb_id"), "must match ^\\d{1,9}$")); + } + Some(t.parse::().map_err(|_| { + err(format!("actors[{i}].tmdb_id"), "not a representable integer") + })?) + } + _ => None, + }; + + let imdb_id = match &a.imdb_id { + Some(v) if !v.is_empty() => { + if !is_person_imdb_id(v) { + return Err(err(format!("actors[{i}].imdb_id"), "must match ^nm\\d{7,8}$")); + } + Some(v.clone()) + } + _ => None, + }; + + if tmdb_id.is_none() && imdb_id.is_none() && a.name.as_deref().unwrap_or("").is_empty() { + return Err(err( + format!("actors[{i}]"), + "requires at least one of tmdb_id, imdb_id or name", + )); + } + + // §6: duplicate actors within one manifest. + if let Some(t) = tmdb_id { + if seen_tmdb.contains(&t) { + return Err(err(format!("actors[{i}].tmdb_id"), "duplicate actor in manifest")); + } + seen_tmdb.push(t); + } + if let Some(v) = &imdb_id { + if seen_imdb.contains(v) { + return Err(err(format!("actors[{i}].imdb_id"), "duplicate actor in manifest")); + } + seen_imdb.push(v.clone()); + } + + let name = match &a.name { + Some(n) if !n.is_empty() => { + check_not_path_shaped(&format!("actors[{i}].name"), n)?; + Some(check_text(&format!("actors[{i}].name"), n, limits::MAX_NAME_CHARS)?) + } + _ => None, + }; + + out.push(ActorScenes { name, tmdb_id, imdb_id, scenes_cs }); + } + + Ok(out) +} + +fn validate_scenes(idx: usize, a: &Actor, runtime_sec: f64) -> VResult> { + if a.scenes.len() > limits::MAX_SCENES_PER_ACTOR { + return Err(err( + format!("actors[{idx}].scenes"), + format!("more than {} entries", limits::MAX_SCENES_PER_ACTOR), + )); + } + let max_t = runtime_sec + limits::RUNTIME_TOLERANCE_SEC; + let mut out = Vec::with_capacity(a.scenes.len()); + + for (j, [start, end]) in a.scenes.iter().copied().enumerate() { + let field = format!("actors[{idx}].scenes[{j}]"); + // §6: non-finite values (NaN/Infinity), negative times, `end < start`, + // or times beyond `runtime_sec` + tolerance. + if !start.is_finite() || !end.is_finite() { + return Err(err(field, "non-finite value")); + } + if start < 0.0 || end < 0.0 { + return Err(err(field, "negative time")); + } + if end < start { + return Err(err(field, "end before start")); + } + if end > max_t { + return Err(err( + field, + format!( + "end {end} beyond runtime_sec + {}s tolerance", + limits::RUNTIME_TOLERANCE_SEC + ), + )); + } + out.push((to_centiseconds(start), to_centiseconds(end))); + } + + // §2: scenes are sorted. Checked on the quantised values so the stored form + // is the one guaranteed ordered. + if out.windows(2).any(|w| w[1].0 < w[0].0) { + return Err(err(format!("actors[{idx}].scenes"), "windows must be sorted by start time")); + } + Ok(out) +} + +// --------------------------------------------------------------------------- +// Bundle validation +// --------------------------------------------------------------------------- + +/// Validates the bundle *envelope* only. +/// +/// §4: a malformed envelope is a whole-request `400`, whereas individual bad +/// episodes are reported in the per-episode results list — the bundle is not +/// atomic, because all-or-nothing would let one bad episode discard an entire +/// season's compute (§2). +pub fn validate_bundle_envelope(b: &SeriesBundle) -> VResult<()> { + if b.jmanifest_version != JMANIFEST_VERSION { + return Err(err( + "jmanifest_version", + format!("unsupported version {}, expected {JMANIFEST_VERSION}", b.jmanifest_version), + )); + } + if b.series.series_tmdb_id.is_none() && b.series.series_imdb_id.is_none() { + return Err(err("series", "requires at least one of series_tmdb_id or series_imdb_id")); + } + if let Some(t) = &b.series.series_tmdb_id { + if !is_tmdb_id(t) { + return Err(err("series.series_tmdb_id", "must match ^\\d{1,9}$")); + } + } + if let Some(i) = &b.series.series_imdb_id { + if !is_title_imdb_id(i) { + return Err(err("series.series_imdb_id", "must match ^tt\\d{7,8}$")); + } + } + if let Some(t) = &b.series.title { + check_not_path_shaped("series.title", t)?; + check_text("series.title", t, limits::MAX_TITLE_CHARS)?; + } + if b.episodes.is_empty() { + return Err(err("episodes", "bundle contains no episodes")); + } + if b.episodes.len() > limits::MAX_BUNDLE_EPISODES { + return Err(err("episodes", format!("more than {} episodes", limits::MAX_BUNDLE_EPISODES))); + } + Ok(()) +} + +// --------------------------------------------------------------------------- +// base64 (standard alphabet, padded) +// --------------------------------------------------------------------------- + +/// Minimal standard-alphabet base64 decoder. +/// +/// Vendored rather than pulled in as a dependency: the only base64 in this +/// service is the fixed-format audio signature, and §3 argues for keeping the +/// dependency surface small on the same grounds as the plugin's FFT. +fn base64_decode(s: &str) -> Result, &'static str> { + fn val(b: u8) -> Result { + match b { + b'A'..=b'Z' => Ok(b - b'A'), + b'a'..=b'z' => Ok(b - b'a' + 26), + b'0'..=b'9' => Ok(b - b'0' + 52), + b'+' => Ok(62), + b'/' => Ok(63), + _ => Err("character outside the base64 alphabet"), + } + } + + let bytes = s.as_bytes(); + if bytes.len() % 4 != 0 { + return Err("length is not a multiple of 4"); + } + if bytes.is_empty() { + return Ok(Vec::new()); + } + + let mut out = Vec::with_capacity(bytes.len() / 4 * 3); + for (i, chunk) in bytes.chunks(4).enumerate() { + let last = i == bytes.len() / 4 - 1; + let pad = if last { + chunk.iter().filter(|&&b| b == b'=').count() + } else { + if chunk.contains(&b'=') { + return Err("padding before the final chunk"); + } + 0 + }; + if pad > 2 { + return Err("more than two padding characters"); + } + let mut acc = 0u32; + for (k, &b) in chunk.iter().enumerate() { + let v = if b == b'=' { + if k < 4 - pad { + return Err("padding in a data position"); + } + 0 + } else { + val(b)? + }; + acc = (acc << 6) | v as u32; + } + let triple = acc.to_be_bytes(); + out.push(triple[1]); + if pad < 2 { + out.push(triple[2]); + } + if pad < 1 { + out.push(triple[3]); + } + } + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn base_manifest() -> Jmanifest { + serde_json::from_str( + r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172","title":"The Death of Stalin"}, + "cut":{"runtime_sec":6420.5}, + "actors":[{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[191.6,209.2]]}]}"#, + ) + .unwrap() + } + + #[test] + fn accepts_a_realistic_manifest() { + let v = validate_manifest(base_manifest()).unwrap(); + assert_eq!(v.actor_scenes_cs.len(), 1); + assert_eq!(v.actor_scenes_cs[0].scenes_cs, vec![(19160, 20920)]); + } + + #[test] + fn windows_are_never_reshaped() { + // UR-013 / SR-002: a window is a claim about *scene membership*, not a + // recognition event. The server therefore stores what it was given — + // quantised, but never merged, split or trimmed. + // + // The adjacent-window case is the one that matters: a naive + // implementation might "tidy" two windows that touch into one, which + // would destroy the distinction SR-002 draws between an actor who turned + // away (one window, gap absorbed by the producer) and one who genuinely + // left and returned (two windows). + let mut m = base_manifest(); + m.actors[0].scenes = vec![ + [10.0, 20.0], + [20.0, 30.0], // exactly adjacent — must stay separate + [30.01, 40.0], // a hair's gap — likewise + [100.0, 100.0], // zero-length — a real producer emits these + ]; + let v = validate_manifest(m).unwrap(); + assert_eq!( + v.actor_scenes_cs[0].scenes_cs, + vec![(1000, 2000), (2000, 3000), (3001, 4000), (10000, 10000)], + "windows must survive validation unchanged apart from quantisation" + ); + } + + #[test] + fn extinction_sec_replaces_anneal_sec() { + // The SR-003 withdrawal. `anneal_sec` cannot even be constructed here — + // it is not a field on `Extraction` — so this asserts the successor is + // accepted and range-checked; `tests/api.rs` covers the wire rejection. + let mut m = base_manifest(); + m.extraction = Some(crate::model::Extraction { + sample_fps: Some(5.0), + extinction_sec: Some(12.0), + pipeline_version: Some("test 0.1".into()), + gallery_size: Some(1820), + gallery_scope: Some(crate::model::GalleryScope::Global), + }); + assert!(validate_manifest(m).is_ok()); + + let mut m = base_manifest(); + m.extraction = Some(crate::model::Extraction { + sample_fps: None, + extinction_sec: Some(-1.0), + pipeline_version: None, + gallery_size: None, + gallery_scope: None, + }); + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "extraction.extinction_sec"); + } + + #[test] + fn rejects_empty_actor_list() { + let mut m = base_manifest(); + m.actors.clear(); + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "actors"); + } + + #[test] + fn rejects_scene_beyond_runtime_tolerance() { + let mut m = base_manifest(); + m.actors[0].scenes = vec![[10.0, 6500.0]]; + let e = validate_manifest(m).unwrap_err(); + assert!(e.field.starts_with("actors[0].scenes"), "got {}", e.field); + } + + #[test] + fn accepts_scene_within_runtime_tolerance() { + let mut m = base_manifest(); + m.actors[0].scenes = vec![[10.0, 6424.0]]; + assert!(validate_manifest(m).is_ok()); + } + + #[test] + fn rejects_end_before_start_and_negative_and_nonfinite() { + for scenes in [ + vec![[50.0, 10.0]], + vec![[-1.0, 10.0]], + vec![[f64::NAN, 10.0]], + vec![[0.0, f64::INFINITY]], + ] { + let mut m = base_manifest(); + m.actors[0].scenes = scenes; + assert!(validate_manifest(m).is_err()); + } + } + + #[test] + fn rejects_unsorted_scenes() { + let mut m = base_manifest(); + m.actors[0].scenes = vec![[100.0, 120.0], [10.0, 20.0]]; + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "actors[0].scenes"); + } + + #[test] + fn rejects_duplicate_actor() { + let mut m = base_manifest(); + m.actors.push(m.actors[0].clone()); + let e = validate_manifest(m).unwrap_err(); + assert!(e.reason.contains("duplicate"), "got {}", e.reason); + } + + #[test] + fn rejects_bad_identifier_formats() { + let mut m = base_manifest(); + m.identity.imdb_id = Some("tt123".into()); + assert!(validate_manifest(m).is_err()); + + let mut m = base_manifest(); + m.identity.tmdb_id = Some("504172x".into()); + assert!(validate_manifest(m).is_err()); + + let mut m = base_manifest(); + m.actors[0].imdb_id = Some("tt0000114".into()); // title id in a person field + assert!(validate_manifest(m).is_err()); + } + + #[test] + fn requires_an_identity_key() { + let mut m = base_manifest(); + m.identity.tmdb_id = None; + m.identity.imdb_id = None; + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "identity"); + } + + #[test] + fn episode_identity_requires_season_and_episode() { + let json = r#"{"jmanifest_version":1, + "identity":{"type":"episode","series_tmdb_id":"1396","title":"Breaking Bad"}, + "cut":{"runtime_sec":2820.0}, + "actors":[{"name":"Bryan Cranston","tmdb_id":"17419","scenes":[[10.0,20.0]]}]}"#; + let m: Jmanifest = serde_json::from_str(json).unwrap(); + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "identity.season"); + } + + #[test] + fn valid_episode_identity_is_accepted() { + let json = r#"{"jmanifest_version":1, + "identity":{"type":"episode","series_tmdb_id":"1396","series_imdb_id":"tt0903747", + "title":"Breaking Bad","season":2,"episode":5}, + "cut":{"runtime_sec":2820.0}, + "actors":[{"name":"Bryan Cranston","tmdb_id":"17419","scenes":[[10.0,20.0]]}]}"#; + let m: Jmanifest = serde_json::from_str(json).unwrap(); + assert!(validate_manifest(m).is_ok()); + } + + #[test] + fn movie_identity_rejects_episode_coordinates() { + let mut m = base_manifest(); + m.identity.season = Some(1); + assert!(validate_manifest(m).is_err()); + } + + // §5a: the character class alone must defeat base64/hex smuggling, which + // needs digits and padding characters. + #[test] + fn name_character_class_rejects_smuggling() { + for name in [ + "SGVsbG8gd29ybGQ=", // base64 + "deadbeef1234", // hex + "Steve/Buscemi", + "Steve+Buscemi", + "Actor 2", // digits + "Steve\u{200B}Buscemi", // zero-width space + "Steve\u{202E}imecsuB", // bidi override + "Steve\u{0007}Buscemi", // control character + "", + ] { + let mut m = base_manifest(); + m.actors[0].name = Some(name.to_string()); + assert!( + validate_manifest(m).is_err(), + "name {name:?} should be rejected by the §5a character class" + ); + } + } + + #[test] + fn name_character_class_rejects_compatibility_homoglyphs() { + // Another gap an injection test caught. These are letters by Unicode + // category, so a category-only check admits them, and NFC does not fold + // them — only NFKC would. Two problems: they spoof a real person's name, + // and a fullwidth-digit alphabet would reopen the encoding channel that + // §5a's "no digits" rule exists to close. + for name in [ + "𝐒𝐭𝐞𝐯𝐞 𝐁𝐮𝐬𝐜𝐞𝐦𝐢", // mathematical bold + "Steve", // fullwidth + "ⓈⓉⒺⓋⒺ", // enclosed alphanumerics + "STEVE 123", // fullwidth with digits + "film", // ligature + "Ⅻ", // Roman numeral + ] { + let mut m = base_manifest(); + m.actors[0].name = Some(name.to_string()); + assert!( + validate_manifest(m).is_err(), + "compatibility homoglyph {name:?} should be rejected" + ); + } + } + + #[test] + fn name_character_class_accepts_real_names() { + for name in [ + "Steve Buscemi", + "Michael Palin", + "Jean-Luc Picard", + "Renée Zellweger", + "Hayao Miyazaki", + "宮崎 駿", + "Miloš Forman", + "O'Brien", + "Sammy Davis, Jr.", + ] { + let mut m = base_manifest(); + m.actors[0].name = Some(name.to_string()); + assert!(validate_manifest(m).is_ok(), "name {name:?} should be accepted"); + } + } + + #[test] + fn rejects_path_shaped_strings_in_allowed_fields() { + for probe in ["/data/movies/x.mkv", "C:\\media\\x.mkv", "\\\\nas\\media", "file:///x"] { + let mut m = base_manifest(); + m.identity.title = Some(probe.to_string()); + assert!(validate_manifest(m).is_err(), "{probe} should be rejected"); + } + } + + #[test] + fn rejects_overlong_name() { + let mut m = base_manifest(); + m.actors[0].name = Some("a".repeat(limits::MAX_NAME_CHARS + 1)); + assert!(validate_manifest(m).is_err()); + } + + #[test] + fn rejects_too_many_actors() { + let mut m = base_manifest(); + let a = m.actors[0].clone(); + m.actors = (0..=limits::MAX_ACTORS) + .map(|i| { + let mut c = a.clone(); + c.tmdb_id = Some((1000 + i).to_string()); + c + }) + .collect(); + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "actors"); + } + + #[test] + fn video_hash_format_is_enforced() { + let mut m = base_manifest(); + m.cut.video_hash = Some("opensubtitles:8e245d9679d31e12".into()); + assert!(validate_manifest(m).is_ok()); + + for bad in ["8e245d9679d31e12", "opensubtitles:xyz", "opensubtitles:8e245d9679d31e1"] { + let mut m = base_manifest(); + m.cut.video_hash = Some(bad.into()); + assert!(validate_manifest(m).is_err(), "{bad} should be rejected"); + } + } + + #[test] + fn centisecond_quantisation_is_stable_for_accumulated_float_error() { + // §9a: real corpus values look like 8045.066666660665. + assert_eq!(to_centiseconds(8045.066666660665), 804507); + assert_eq!(to_centiseconds(8045.066666666), 804507); + assert_eq!(to_centiseconds(0.0), 0); + } + + #[test] + fn base64_roundtrip() { + // "Man" => "TWFu"; padding variants. + assert_eq!(base64_decode("TWFu").unwrap(), b"Man"); + assert_eq!(base64_decode("TWE=").unwrap(), b"Ma"); + assert_eq!(base64_decode("TQ==").unwrap(), b"M"); + assert!(base64_decode("TWF").is_err()); + assert!(base64_decode("TW$u").is_err()); + assert!(base64_decode("T=Fu").is_err()); + } + + /// A feature-length runtime, so the full window applies. + const FEATURE_RUNTIME: f64 = 6420.5; + + #[test] + fn audio_signature_validation() { + // 1290 frames with the high bit clear, base64-encoded. + let frames = vec![0x3Fu8; limits::AUDIO_SIG_FRAMES]; + let sig = format!("v1:{}", base64_encode_for_test(&frames)); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_ok()); + + // Missing version prefix. + assert!( + validate_audio_signature(&base64_encode_for_test(&frames), FEATURE_RUNTIME).is_err() + ); + + // High bit set is structurally invalid, so arbitrary bytes cannot ride + // along in this field (§3, §5a). + let mut bad = frames.clone(); + bad[7] = 0xFF; + let sig = format!("v1:{}", base64_encode_for_test(&bad)); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_err()); + + // Oversized blob would be a payload channel. + let huge = vec![0x01u8; limits::AUDIO_SIG_FRAMES + limits::AUDIO_SIG_TOLERANCE + 1]; + let sig = format!("v1:{}", base64_encode_for_test(&huge)); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_err()); + } + + #[test] + fn media_below_the_window_must_carry_no_signature() { + // IR-007, reconciled with server §3: the window `runtime/2 ± 60 s` + // underflows below 120 s, so no signature exists to send. One present on + // such an item did not come from the specified construction, whatever it + // is. An earlier draft of §3 allowed a shortened window here; that was + // the weaker rule, because a caller-varying length is exactly the + // property SR-004 forbids. + let frames = vec![0x3Fu8; limits::AUDIO_SIG_FRAMES]; + let sig = format!("v1:{}", base64_encode_for_test(&frames)); + + for runtime in [1.0f64, 30.0, 119.0, 119.999] { + let e = validate_audio_signature(&sig, runtime).unwrap_err(); + assert_eq!(e.field, "cut.audio_signature"); + assert!( + e.reason.contains("shorter than"), + "a {runtime}s item should be refused on its runtime: {}", + e.reason + ); + } + + // At and above the window, the normal rules apply. + assert!(validate_audio_signature(&sig, 120.0).is_ok()); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_ok()); + } + + #[test] + fn the_signature_length_is_fixed_not_caller_chosen() { + // What keeps the field inside SR-004: the length is a property of the + // construction, so a caller cannot use it as a variable-size container. + for frames in [1usize, 16, 100, 900, 1200] { + let sig = format!("v1:{}", base64_encode_for_test(&vec![0x3Fu8; frames])); + assert!( + validate_audio_signature(&sig, FEATURE_RUNTIME).is_err(), + "{frames} frames should be rejected — the length is fixed" + ); + } + // Only the construction's own length, within edge tolerance, is accepted. + for frames in [ + limits::AUDIO_SIG_FRAMES - limits::AUDIO_SIG_TOLERANCE, + limits::AUDIO_SIG_FRAMES, + limits::AUDIO_SIG_FRAMES + limits::AUDIO_SIG_TOLERANCE, + ] { + let sig = format!("v1:{}", base64_encode_for_test(&vec![0x3Fu8; frames])); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_ok(), "{frames} frames"); + } + } + + fn base64_encode_for_test(data: &[u8]) -> String { + const A: &[u8] = b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; + let mut out = String::new(); + for chunk in data.chunks(3) { + let b = [chunk[0], *chunk.get(1).unwrap_or(&0), *chunk.get(2).unwrap_or(&0)]; + let n = u32::from_be_bytes([0, b[0], b[1], b[2]]); + out.push(A[(n >> 18 & 63) as usize] as char); + out.push(A[(n >> 12 & 63) as usize] as char); + out.push(if chunk.len() > 1 { A[(n >> 6 & 63) as usize] as char } else { '=' }); + out.push(if chunk.len() > 2 { A[(n & 63) as usize] as char } else { '=' }); + } + out + } + + #[test] + fn bundle_envelope_checks() { + let ok = r#"{"jmanifest_version":1, + "series":{"series_tmdb_id":"1396","title":"Breaking Bad"}, + "episodes":[{"jmanifest_version":1, + "identity":{"type":"episode","series_tmdb_id":"1396","season":1,"episode":1}, + "cut":{"runtime_sec":2820.0}, + "actors":[{"tmdb_id":"17419","scenes":[[1.0,2.0]]}]}]}"#; + let b: SeriesBundle = serde_json::from_str(ok).unwrap(); + assert!(validate_bundle_envelope(&b).is_ok()); + + let mut empty = b.clone(); + empty.episodes.clear(); + assert!(validate_bundle_envelope(&empty).is_err()); + + let mut no_id = b.clone(); + no_id.series.series_tmdb_id = None; + no_id.series.series_imdb_id = None; + assert!(validate_bundle_envelope(&no_id).is_err()); + } +} diff --git a/src/worker.rs b/src/worker.rs new file mode 100644 index 0000000..1538f23 --- /dev/null +++ b/src/worker.rs @@ -0,0 +1,494 @@ +//! Background worker for the §6 stage 3 cast check. +//! +//! §8: this runs as a Tokio background task in the same binary, with the job +//! queue as a SQLite table so state survives restart — replacing an external +//! broker entirely. The check needs an outbound TMDB call and so cannot run +//! inside the request without coupling upload latency to a third party (§6). + +use std::sync::Arc; +use std::time::Duration; + +use anyhow::Context; + +use crate::castcheck::{self, SubmittedActor, Verdict}; +use crate::db::{repo, Db}; +use crate::ingest::{CastCheckJob, JOB_CAST_CHECK}; +use crate::tmdb::{CastMember, Credits, TmdbClient, TmdbError}; + +/// §6: TMDB responses are cached for 24h, so a burst of episode uploads for one +/// series costs a single upstream call. +const CACHE_TTL: Duration = Duration::from_secs(24 * 3600); +/// Cap on retry backoff for a persistent TMDB outage. +const MAX_BACKOFF_SECS: u64 = 3600; + +pub struct Worker { + pub db: Db, + pub tmdb: Arc, + pub batch: usize, + pub poll_interval: Duration, +} + +impl Worker { + /// Runs until `shutdown` resolves. + pub async fn run(self, mut shutdown: tokio::sync::watch::Receiver) { + // A process that died mid-job would otherwise leave work stranded. + match self.db.write(repo::release_all_leases).await { + Ok(n) if n > 0 => tracing::info!(released = n, "released stranded job leases"), + Ok(_) => {} + Err(e) => tracing::error!(error = ?e, "failed to release job leases at startup"), + } + + loop { + tokio::select! { + _ = shutdown.changed() => { + tracing::info!("worker shutting down"); + return; + } + _ = tokio::time::sleep(self.poll_interval) => { + if let Err(e) = self.tick().await { + tracing::error!(error = ?e, "worker tick failed"); + } + } + } + } + } + + async fn tick(&self) -> anyhow::Result<()> { + let now = now_iso(); + let batch = self.batch; + let leased_at = now.clone(); + let jobs = self.db.write(move |tx| repo::lease_jobs(tx, &leased_at, batch)).await?; + + for job in jobs { + let result = match job.kind.as_str() { + JOB_CAST_CHECK => self.run_cast_check(&job.payload).await, + other => { + tracing::warn!(kind = other, "unknown job kind, dropping"); + Ok(()) + } + }; + + let job_id = job.id.clone(); + match result { + Ok(()) => { + self.db.write(move |tx| repo::delete_job(tx, &job_id)).await?; + } + Err(JobError::Retry(msg)) => { + // §6: TMDB unreachable or rate-limited means retry with + // backoff; the manifest stays unlisted, not rejected. + let delay = backoff_secs(job.attempts); + let run_after = iso_in(delay); + tracing::warn!(job = %job_id, attempts = job.attempts, delay, reason = %msg, + "rescheduling job"); + self.db + .write(move |tx| repo::reschedule_job(tx, &job_id, &run_after, &msg)) + .await?; + } + Err(JobError::Fatal(e)) => { + tracing::error!(job = %job_id, error = ?e, "dropping job after fatal error"); + self.db.write(move |tx| repo::delete_job(tx, &job_id)).await?; + } + } + } + Ok(()) + } + + async fn run_cast_check(&self, payload: &str) -> Result<(), JobError> { + let job: CastCheckJob = + serde_json::from_str(payload).map_err(|e| JobError::Fatal(e.into()))?; + let manifest_id = job.manifest_id; + + // Load what the check needs. + let mid = manifest_id.clone(); + let loaded = self + .db + .read(move |conn| { + let Some(m) = repo::manifest_by_id(conn, &mid)? else { return Ok(None) }; + let title = conn + .query_row( + "SELECT kind, tmdb_id, imdb_id, adult, certification FROM titles WHERE id = ?1", + rusqlite::params![m.title_id], + |r| { + Ok(( + r.get::<_, String>(0)?, + r.get::<_, Option>(1)?, + r.get::<_, Option>(2)?, + r.get::<_, i64>(3)? != 0, + r.get::<_, Option>(4)?, + )) + }, + ) + .map_err(anyhow::Error::from)?; + let actor_ids = repo::manifest_actor_ids(conn, &mid)?; + Ok(Some((m, title, actor_ids))) + }) + .await + .map_err(JobError::Fatal)?; + + // The manifest may have been deleted (contributor revoked, §5a) between + // enqueue and now; that is not an error. + let Some((manifest, (kind, tmdb_id, _imdb_id, title_adult, certification), actor_ids)) = + loaded + else { + return Ok(()); + }; + if manifest.status != "pending" { + return Ok(()); + } + + let Some(tmdb_id) = tmdb_id else { + // No TMDB id means the cast check cannot run at all. §6 treats absent + // reference data as flagged, not rejected. + self.finalise(&manifest_id, Verdict::Flagged, 0.0, Some("no_tmdb_id"), &[], &[]) + .await + .map_err(JobError::Fatal)?; + return Ok(()); + }; + + let credits = self + .credits_for(&kind, &tmdb_id, manifest.season, manifest.episode) + .await + .map_err(|e| { + if e.is_retryable() { + JobError::Retry(e.to_string()) + } else { + // A genuinely absent title is a verdict, not a transport + // failure — handled below via empty credits. + JobError::Retry(format!("non-retryable tmdb error treated as absent: {e}")) + } + }); + + let credits = match credits { + Ok(c) => c, + Err(JobError::Retry(msg)) if msg.starts_with("non-retryable") => { + tracing::info!(manifest = %manifest_id, "tmdb has no such title; flagging"); + Credits::default() + } + Err(e) => return Err(e), + }; + + let reference: Vec = credits.all().cloned().collect(); + let submitted: Vec = actor_ids + .iter() + .map(|id| SubmittedActor { tmdb_id: Some(*id), imdb_id: None, name: None }) + .collect(); + + let mut outcome = castcheck::evaluate(&submitted, &reference); + + // §5a layer 1 — category guard. + if let Some(offender) = castcheck::category_guard_violation(&outcome.matched, title_adult) { + tracing::warn!(manifest = %manifest_id, person = offender, + "category guard: adult-flagged performer on a non-adult title"); + outcome.verdict = Verdict::Rejected; + outcome.reason = Some("category_guard".into()); + } + + // §5a layer 2 — age-appropriateness guard. + if let Some(cert) = &certification { + if castcheck::is_childrens_certification(cert) { + castcheck::apply_childrens_guard(&mut outcome, submitted.len()); + } + } + + self.finalise( + &manifest_id, + outcome.verdict, + outcome.ratio, + outcome.reason.as_deref(), + &outcome.matched, + &outcome.unmatched_person_ids, + ) + .await + .map_err(JobError::Fatal)?; + + Ok(()) + } + + /// Fetches credits, using the 24h cache (§6). + /// + /// For episodes this is the **union** of TMDB's per-episode credits (cast + + /// guest stars) and the series' aggregate credits: per-episode alone would + /// reject recurring cast TMDB lists only at series level, series-wide alone + /// would reject legitimate guest stars. + async fn credits_for( + &self, + kind: &str, + tmdb_id: &str, + season: Option, + episode: Option, + ) -> Result { + if kind == "movie" { + return self.cached("movie", tmdb_id, || self.tmdb.movie_credits(tmdb_id)).await; + } + + let series = self.cached("series", tmdb_id, || self.tmdb.series_credits(tmdb_id)).await?; + + let mut combined = series; + if let (Some(s), Some(e)) = (season, episode) { + let key = format!("{tmdb_id}:{s}:{e}"); + match self.cached("episode", &key, || self.tmdb.episode_credits(tmdb_id, s, e)).await { + Ok(ep) => { + combined.cast.extend(ep.cast); + combined.guest_stars.extend(ep.guest_stars); + } + // A missing episode entry is normal; the series set still applies. + Err(TmdbError::NotFound) => {} + Err(e) if e.is_retryable() => return Err(e), + Err(e) => tracing::warn!(error = ?e, "ignoring episode credits error"), + } + } + Ok(combined) + } + + async fn cached(&self, kind: &str, key: &str, fetch: F) -> Result + where + F: FnOnce() -> Fut, + Fut: std::future::Future>, + { + let (k, kk) = (key.to_string(), kind.to_string()); + let cached = self + .db + .read(move |conn| repo::cached_credits(conn, &k, &kk)) + .await + .map_err(|e| TmdbError::Transport(e.to_string()))?; + + if let Some((json, fetched_at)) = cached { + if !is_stale(&fetched_at, CACHE_TTL) { + if let Ok(c) = serde_json::from_str::(&json) { + return Ok(c); + } + } + } + + let fresh = fetch().await?; + let json = serde_json::to_string(&SerializableCredits::from(&fresh)) + .map_err(|e| TmdbError::Malformed(e.to_string()))?; + let (k, kk, now) = (key.to_string(), kind.to_string(), now_iso()); + let _ = self.db.write(move |tx| repo::put_credits(tx, &k, &kk, &json, &now)).await; + Ok(fresh) + } + + /// Applies the verdict: resolves names into `people`, drops unmatched actors, + /// and updates status and contributor counters — in one transaction. + async fn finalise( + &self, + manifest_id: &str, + verdict: Verdict, + ratio: f64, + reason: Option<&str>, + matched: &[castcheck::MatchedActor], + unmatched: &[u64], + ) -> anyhow::Result<()> { + let id = manifest_id.to_string(); + let reason = reason.map(str::to_string); + let matched: Vec<(u64, String, bool)> = + matched.iter().map(|m| (m.tmdb_person_id, m.name.clone(), m.adult)).collect(); + let unmatched = unmatched.to_vec(); + let now = now_iso(); + + self.db + .write(move |tx| { + let contributor: Option = tx + .query_row( + "SELECT contributor_id FROM manifests WHERE id = ?1", + rusqlite::params![id], + |r| r.get(0), + ) + .map_err(anyhow::Error::from)?; + + if verdict == Verdict::Rejected { + // §6: the manifest is deleted and the contributor notified + // (via `GET /manifests/{id}/status` until it is gone). + repo::delete_manifest(tx, &id)?; + } else { + // Names come from TMDB, never from the upload (§5a, §7). + for (person_id, name, adult) in &matched { + repo::upsert_person(tx, *person_id, name, *adult, &now)?; + } + // §6: unmatched actors are dropped rather than stored. + for person_id in &unmatched { + repo::delete_manifest_actor(tx, &id, *person_id)?; + } + repo::set_manifest_status( + tx, + &id, + verdict.status(), + reason.as_deref(), + Some(ratio), + )?; + } + + if let Some(c) = contributor { + let counter = match verdict { + Verdict::Listed => "accepted", + Verdict::Flagged => "flagged", + Verdict::Rejected => "rejected", + }; + repo::bump_contributor_counter(tx, &c, counter)?; + if repo::maybe_revoke_contributor(tx, &c, &now)? { + tracing::warn!(contributor = %c, "revoked token for excessive rejections"); + } + } + Ok(()) + }) + .await + .context("finalising cast check") + } +} + +/// Serialisable projection of `Credits` for the cache. +#[derive(serde::Serialize)] +struct SerializableCredits { + cast: Vec, + guest_stars: Vec, +} + +#[derive(serde::Serialize)] +struct SerializableMember { + id: u64, + name: String, + adult: bool, +} + +impl From<&Credits> for SerializableCredits { + fn from(c: &Credits) -> Self { + let f = + |m: &CastMember| SerializableMember { id: m.id, name: m.name.clone(), adult: m.adult }; + Self { + cast: c.cast.iter().map(f).collect(), + guest_stars: c.guest_stars.iter().map(&f).collect(), + } + } +} + +enum JobError { + Retry(String), + Fatal(anyhow::Error), +} + +/// Exponential backoff, capped (§5: the client must back off exponentially +/// rather than retrying tightly; the same discipline applies to our own +/// outbound calls). +fn backoff_secs(attempts: i64) -> u64 { + let base = 30u64; + base.saturating_mul(1u64 << attempts.clamp(0, 8) as u32).min(MAX_BACKOFF_SECS) +} + +fn is_stale(fetched_at: &str, ttl: Duration) -> bool { + let Some(then) = parse_iso(fetched_at) else { return true }; + let now = unix_now(); + now.saturating_sub(then) > ttl.as_secs() +} + +/// Current time as an RFC 3339 UTC string, which is what every timestamp column +/// stores. Kept in one place so the format cannot drift. +pub fn now_iso() -> String { + iso_in(0) +} + +pub fn iso_in(secs: u64) -> String { + format_unix(unix_now() + secs) +} + +fn unix_now() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_secs()) + .unwrap_or(0) +} + +/// Formats a Unix timestamp as `YYYY-MM-DDTHH:MM:SSZ`. +/// +/// Hand-rolled rather than pulling in `chrono`/`time`: the only requirement is a +/// lexicographically-sortable UTC string, which is what the `jobs.run_after` +/// comparison relies on. +pub fn format_unix(mut secs: u64) -> String { + let days = secs / 86_400; + secs %= 86_400; + let (h, m, s) = (secs / 3600, (secs % 3600) / 60, secs % 60); + + // Civil-from-days, Howard Hinnant's algorithm. + let z = days as i64 + 719_468; + let era = z.div_euclid(146_097); + let doe = z.rem_euclid(146_097); + let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365; + let y = yoe + era * 400; + let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); + let mp = (5 * doy + 2) / 153; + let d = doy - (153 * mp + 2) / 5 + 1; + let mo = if mp < 10 { mp + 3 } else { mp - 9 }; + let y = if mo <= 2 { y + 1 } else { y }; + + format!("{y:04}-{mo:02}-{d:02}T{h:02}:{m:02}:{s:02}Z") +} + +fn parse_iso(s: &str) -> Option { + // Parses the format `format_unix` produces. + let b = s.as_bytes(); + if b.len() < 20 { + return None; + } + let num = |from: usize, to: usize| s.get(from..to)?.parse::().ok(); + let (y, mo, d) = (num(0, 4)?, num(5, 7)?, num(8, 10)?); + let (h, mi, se) = (num(11, 13)?, num(14, 16)?, num(17, 19)?); + + let y_adj = if mo <= 2 { y - 1 } else { y }; + let era = y_adj.div_euclid(400); + let yoe = y_adj - era * 400; + let mp = if mo > 2 { mo - 3 } else { mo + 9 }; + let doy = (153 * mp + 2) / 5 + d - 1; + let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy; + let days = era * 146_097 + doe - 719_468; + + Some((days * 86_400 + h * 3600 + mi * 60 + se).max(0) as u64) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn timestamp_roundtrips() { + for t in [0u64, 1, 1_000_000, 1_700_000_000, 1_785_000_000, 4_000_000_000] { + let s = format_unix(t); + assert_eq!(parse_iso(&s), Some(t), "roundtrip failed for {t} => {s}"); + } + } + + #[test] + fn timestamp_format_is_sortable() { + // `jobs.run_after <= ?1` is a string comparison, so lexical order must + // match chronological order. + let a = format_unix(1_700_000_000); + let b = format_unix(1_700_000_001); + let c = format_unix(1_800_000_000); + assert!(a < b && b < c, "{a} {b} {c}"); + assert_eq!(format_unix(0), "1970-01-01T00:00:00Z"); + } + + #[test] + fn known_dates_format_correctly() { + // 2026-07-30T12:00:00Z + assert_eq!(format_unix(1_785_412_800), "2026-07-30T12:00:00Z"); + // A leap day, since the civil-from-days algorithm is where this would + // break. + assert_eq!(format_unix(1_709_164_800), "2024-02-29T00:00:00Z"); + } + + #[test] + fn backoff_grows_and_is_capped() { + assert_eq!(backoff_secs(0), 30); + assert_eq!(backoff_secs(1), 60); + assert_eq!(backoff_secs(4), 480); + assert_eq!(backoff_secs(50), MAX_BACKOFF_SECS, "must not overflow or grow unbounded"); + } + + #[test] + fn staleness_uses_the_ttl() { + let fresh = format_unix(unix_now()); + assert!(!is_stale(&fresh, CACHE_TTL)); + let old = format_unix(unix_now() - 25 * 3600); + assert!(is_stale(&old, CACHE_TTL)); + assert!(is_stale("not-a-timestamp", CACHE_TTL)); + } +} diff --git a/tests/api.rs b/tests/api.rs new file mode 100644 index 0000000..e6dc1a0 --- /dev/null +++ b/tests/api.rs @@ -0,0 +1,953 @@ +//! End-to-end tests through the real router. +//! +//! The unit tests cover each spec rule in isolation; these cover the wiring — +//! status codes, headers, and the properties that only hold if the layers are +//! composed correctly (per-route body caps, rate-limit surfaces, the strict +//! schema actually reaching uploads). + +use std::sync::Arc; + +use axum::body::Body; +use axum::http::{Request, StatusCode}; +use http_body_util::BodyExt; +use jray_server::app; +use jray_server::config::Config; +use jray_server::db::Db; +use jray_server::ratelimit::RateLimiter; +use jray_server::state::AppState; +use jray_server::tmdb::TmdbClient; +use serde_json::{json, Value}; +use tower::ServiceExt; + +/// A server backed by a temporary on-disk database. +/// +/// On-disk rather than `:memory:` because §8's design uses a separate writer +/// connection and a read pool, and in-memory SQLite is per-connection — the +/// readers would see an empty database. Testing the real topology is the point. +struct TestServer { + router: axum::Router, + _dir: TempDir, +} + +struct TempDir(std::path::PathBuf); + +impl TempDir { + fn new(tag: &str) -> Self { + let mut p = std::env::temp_dir(); + // Unique per test without pulling in a tempfile dependency. + p.push(format!("jray-test-{}-{}", tag, ulid_like())); + std::fs::create_dir_all(&p).expect("creating temp dir"); + Self(p) + } + + fn db_path(&self) -> String { + self.0.join("test.db").to_string_lossy().into_owned() + } +} + +impl Drop for TempDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } +} + +fn ulid_like() -> String { + use std::sync::atomic::{AtomicU64, Ordering}; + static N: AtomicU64 = AtomicU64::new(0); + let t = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_nanos()) + .unwrap_or(0); + format!("{t}-{}", N.fetch_add(1, Ordering::Relaxed)) +} + +impl TestServer { + fn new(tag: &str) -> Self { + let dir = TempDir::new(tag); + let db = Db::open(&dir.db_path()).expect("opening database"); + let config = Arc::new(Config { + bind: "127.0.0.1:0".into(), + db_path: dir.db_path(), + // No key: uploads stay `pending`, which is the correct failure mode + // (§8) and keeps these tests free of network calls. + tmdb_api_key: None, + tmdb_base_url: "http://127.0.0.1:1".into(), + trusted_proxies: Vec::new(), + server_id: "test.example".into(), + request_timeout: std::time::Duration::from_secs(30), + job_batch: 8, + job_poll_interval: std::time::Duration::from_secs(3600), + }); + let state = AppState { + db, + config: config.clone(), + limiter: Arc::new(RateLimiter::new()), + tmdb: Arc::new(TmdbClient::new(config.tmdb_base_url.clone(), None)), + }; + Self { router: app::router(state), _dir: dir } + } + + async fn send(&self, req: Request) -> (StatusCode, Value, axum::http::HeaderMap) { + let resp = self.router.clone().oneshot(req).await.expect("router call"); + let status = resp.status(); + let headers = resp.headers().clone(); + let bytes = resp.into_body().collect().await.expect("reading body").to_bytes(); + let body = if bytes.is_empty() { + Value::Null + } else { + serde_json::from_slice(&bytes) + .unwrap_or(Value::String(String::from_utf8_lossy(&bytes).into_owned())) + }; + (status, body, headers) + } + + async fn get(&self, uri: &str) -> (StatusCode, Value, axum::http::HeaderMap) { + self.send(Request::builder().uri(uri).body(Body::empty()).unwrap()).await + } + + async fn post_json( + &self, + uri: &str, + body: &Value, + ) -> (StatusCode, Value, axum::http::HeaderMap) { + self.send( + Request::builder() + .method("POST") + .uri(uri) + .header("content-type", "application/json") + .body(Body::from(body.to_string())) + .unwrap(), + ) + .await + } + + async fn post_json_auth( + &self, + uri: &str, + token: &str, + body: &Value, + ) -> (StatusCode, Value, axum::http::HeaderMap) { + self.send( + Request::builder() + .method("POST") + .uri(uri) + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(body.to_string())) + .unwrap(), + ) + .await + } + + /// Issues an anonymous bearer capability (§5a). + async fn token(&self) -> String { + let (status, body, _) = self.post_json("/api/v1/tokens", &json!({})).await; + assert_eq!(status, StatusCode::OK, "token issue failed: {body}"); + body["token"].as_str().expect("token in response").to_string() + } +} + +fn movie_manifest(tmdb_id: &str, runtime: f64) -> Value { + json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": tmdb_id, "title": "The Death of Stalin", + "year": 2017 }, + "cut": { "runtime_sec": runtime, "video_hash": "opensubtitles:8e245d9679d31e12" }, + "extraction": { "sample_fps": 5, "extinction_sec": 12, + "pipeline_version": "scene-actor-extraction 0.4.1", + "gallery_scope": "global" }, + "actors": [ + { "name": "Steve Buscemi", "tmdb_id": "884", "scenes": [[191.6, 209.2], [438.2, 465.6]] }, + { "name": "Michael Palin", "tmdb_id": "11007", "scenes": [[300.0, 320.0]] } + ] + }) +} + +// --------------------------------------------------------------------------- +// Health and readiness +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn health_is_unauthenticated() { + let s = TestServer::new("health"); + let (status, body, _) = s.get("/health").await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "ok"); +} + +#[tokio::test] +async fn readiness_reports_database_and_tmdb_configuration() { + // §8: TMDB is a hard dependency for UR-3, so its absence is worth surfacing. + let s = TestServer::new("ready"); + let (status, body, _) = s.get("/ready").await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "ready"); + assert_eq!(body["tmdb_configured"], false); +} + +// --------------------------------------------------------------------------- +// §5a — tokens +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn upload_without_a_token_is_rejected() { + let s = TestServer::new("noauth"); + let (status, _, _) = s.post_json("/api/v1/manifests", &movie_manifest("504172", 6420.5)).await; + assert_eq!(status, StatusCode::UNAUTHORIZED); +} + +#[tokio::test] +async fn upload_with_an_unknown_token_is_rejected() { + // A token the server never issued has no contributor row, and §5a stores only + // hashes, so there is nothing to match. + let s = TestServer::new("badauth"); + let (status, _, _) = s + .post_json_auth("/api/v1/manifests", "jray_deadbeef", &movie_manifest("504172", 6420.5)) + .await; + assert_eq!(status, StatusCode::UNAUTHORIZED); +} + +#[tokio::test] +async fn tokens_are_issued_anonymously_and_are_distinct() { + let s = TestServer::new("tokens"); + let a = s.token().await; + let b = s.token().await; + assert_ne!(a, b); + assert!(a.starts_with("jray_")); +} + +// --------------------------------------------------------------------------- +// §6 — upload validation +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn valid_upload_is_accepted_as_pending() { + // §6 stage 3: accepted with `202` and held unlisted until the cast check. + let s = TestServer::new("upload-ok"); + let token = s.token().await; + let (status, body, _) = + s.post_json_auth("/api/v1/manifests", &token, &movie_manifest("504172", 6420.5)).await; + assert_eq!(status, StatusCode::ACCEPTED, "body: {body}"); + assert_eq!(body["status"], "pending"); + assert!(body["manifest_id"].is_string()); +} + +#[tokio::test] +async fn a_pending_manifest_is_not_served() { + // The property that makes §6 stage 3 meaningful: an unverified manifest is + // not served to anyone in the meantime. + let s = TestServer::new("pending-hidden"); + let token = s.token().await; + let (_, body, _) = + s.post_json_auth("/api/v1/manifests", &token, &movie_manifest("504172", 6420.5)).await; + let id = body["manifest_id"].as_str().unwrap(); + + let (status, _, _) = s.get("/api/v1/manifests/movie?tmdb_id=504172").await; + assert_eq!(status, StatusCode::NOT_FOUND); + + let (status, _, _) = s.get(&format!("/api/v1/manifests/{id}")).await; + assert_eq!(status, StatusCode::NOT_FOUND); + + // But its status is pollable (§4). + let (status, body, _) = s.get(&format!("/api/v1/manifests/{id}/status")).await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "pending"); +} + +#[tokio::test] +async fn unknown_field_anywhere_is_rejected_with_400() { + // §6 stage 2, enforced by `deny_unknown_fields` on every DTO. + let s = TestServer::new("strict"); + let token = s.token().await; + + let mut m = movie_manifest("504172", 6420.5); + m["surprise"] = json!("payload"); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + assert!( + body["message"].as_str().unwrap_or("").contains("surprise"), + "the error should name the offending field: {body}" + ); + + // Nested, too — `extra="forbid"` applies at every level (§5a). + let mut m = movie_manifest("504172", 6420.5); + m["cut"]["extra"] = json!(1); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn contributor_local_identifiers_are_rejected_not_ignored() { + // §1/§6: `movie` leaks the contributor's directory layout and `jellyfin_id` is + // a GUID from their database. Both must be *rejected on upload*, so a client + // that forgets to strip them gets a hard 400 naming the field rather than + // quietly publishing them. + let s = TestServer::new("strip"); + let token = s.token().await; + + let mut m = movie_manifest("504172", 6420.5); + m["movie"] = json!("/data/movies/The.Death.of.Stalin.2017.mkv"); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + assert!(body["message"].as_str().unwrap_or("").contains("movie"), "{body}"); + + let mut m = movie_manifest("504172", 6420.5); + m["actors"][0]["jellyfin_id"] = json!("a1b2c3d4e5f6"); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + assert!(body["message"].as_str().unwrap_or("").contains("jellyfin_id"), "{body}"); +} + +#[tokio::test] +async fn the_withdrawn_anneal_sec_field_is_rejected() { + // `anneal_sec` was withdrawn in the SR-003 bump: presence now follows track + // extent, so a track survives its own gaps and there is nothing to anneal + // (`scene-actor-extraction` AR-012/AR-013). + // + // Rejecting rather than ignoring it is the point. A manifest still carrying + // the field was produced by a pipeline whose window semantics differ from + // what this server now assumes, and silently accepting it would store + // timings whose meaning we cannot vouch for. + let s = TestServer::new("anneal-withdrawn"); + let token = s.token().await; + + let mut m = movie_manifest("504172", 6420.5); + m["extraction"]["anneal_sec"] = json!(3); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + assert!( + body["message"].as_str().unwrap_or("").contains("anneal_sec"), + "the error should name the withdrawn field: {body}" + ); +} + +#[tokio::test] +async fn the_schema_bump_fields_round_trip() { + // `extinction_sec` and `gallery_scope` are the SR-003 additions. They are + // stored and reconstructed, since §7 ranks on scope and both are provenance + // a consumer may want. + let s = TestServer::new("bump-fields"); + let token = s.token().await; + + let m = movie_manifest("504172", 6420.5); + assert_eq!(m["extraction"]["gallery_scope"], "global"); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::ACCEPTED, "{body}"); + + // An unrecognised scope is a closed-vocabulary violation, not a free string. + let mut bad = movie_manifest("504173", 6420.5); + bad["extraction"]["gallery_scope"] = json!("enormous"); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &bad).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "gallery_scope is a closed enum"); +} + +#[tokio::test] +async fn an_unknown_manifest_version_is_rejected() { + // UR-014 / SR-003: a consumer encountering an unknown `schema_version` + // refuses or warns; it never guesses. + let s = TestServer::new("version"); + let token = s.token().await; + + for version in [0, 2, 99] { + let mut m = movie_manifest("504172", 6420.5); + m["jmanifest_version"] = json!(version); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "version {version}: {body}"); + assert!( + body["message"].as_str().unwrap_or("").contains("jmanifest_version"), + "should name the field: {body}" + ); + } +} + +#[tokio::test] +async fn missing_runtime_is_rejected() { + // §2: `cut.runtime_sec` is required — the primary alignment guard. + let s = TestServer::new("no-runtime"); + let token = s.token().await; + let mut m = movie_manifest("504172", 6420.5); + m["cut"] = json!({ "video_hash": "opensubtitles:8e245d9679d31e12" }); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn empty_actor_list_is_rejected() { + // §6: 15 of the 331 corpus files have empty actor lists — extraction + // failures, not contributions. + let s = TestServer::new("empty-actors"); + let token = s.token().await; + let mut m = movie_manifest("504172", 6420.5); + m["actors"] = json!([]); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn smuggled_payload_in_an_actor_name_is_rejected() { + // §5a: the character class defeats base64/hex smuggling, which needs digits + // and padding characters. This is the last free-text channel, so it is worth + // asserting end-to-end and not only in the unit tests. + let s = TestServer::new("smuggle"); + let token = s.token().await; + for payload in [ + "SGVsbG8gd29ybGQgdGhpcyBpcyBhIHBheWxvYWQ=", + "4d5a90000300000004000000ffff0000", + "", + "http://evil.example/x", + ] { + let mut m = movie_manifest("504172", 6420.5); + m["actors"][0]["name"] = json!(payload); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "payload {payload:?} should be rejected"); + } +} + +#[tokio::test] +async fn resubmitting_identical_content_is_not_a_duplicate_error() { + // §9a: content addressing gives deduplication — the same manifest from the + // same contributor is recognised rather than stored twice. + let s = TestServer::new("dedup"); + let token = s.token().await; + let m = movie_manifest("504172", 6420.5); + let (first, body1, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(first, StatusCode::ACCEPTED); + let (second, body2, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(second, StatusCode::OK); + assert_eq!(body2["status"], "already_present"); + assert_eq!(body1["manifest_id"], body2["manifest_id"]); +} + +#[tokio::test] +async fn oversized_body_is_rejected_by_the_route_cap() { + // §6 stage 1: the app-level cap counts bytes as they are read, so a lying + // `Content-Length` and a chunked upload are both safe. Here the body genuinely + // exceeds the 2 MiB single-manifest cap. + let s = TestServer::new("too-big"); + let token = s.token().await; + + let mut m = movie_manifest("504172", 6420.5); + // Many actors, each with many windows — legitimate shape, illegitimate size. + let actors: Vec = (0..400) + .map(|i| { + let scenes: Vec = (0..1500).map(|j| json!([j as f64, (j + 1) as f64])).collect(); + json!({ "tmdb_id": (1000 + i).to_string(), "scenes": scenes }) + }) + .collect(); + m["actors"] = json!(actors); + + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::PAYLOAD_TOO_LARGE); +} + +#[tokio::test] +async fn a_lying_content_length_does_not_bypass_the_cap() { + // §6 stage 0 is explicit that `Content-Length` is a *claim by the client*: a + // hostile client can declare 100 and send far more, so the streaming cap is + // mandatory rather than redundant. + let s = TestServer::new("lying-length"); + let token = s.token().await; + + let huge = "x".repeat(3 * 1024 * 1024); + let body = format!("{{\"padding\":\"{huge}\"}}"); + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .header("content-length", "100") + .body(Body::from(body)) + .unwrap(); + + let (status, _, _) = s.send(req).await; + assert_ne!( + status, + StatusCode::ACCEPTED, + "an oversized body must never be accepted, whatever the declared length" + ); + assert!( + status == StatusCode::PAYLOAD_TOO_LARGE || status == StatusCode::BAD_REQUEST, + "unexpected status {status}" + ); +} + +// --------------------------------------------------------------------------- +// §4 — exists +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn exists_returns_200_with_false_rather_than_404() { + // §4: absence is a normal answer, and `404` would conflate "no manifest" with + // "bad route" for the client. + let s = TestServer::new("exists-absent"); + let (status, body, _) = s.get("/api/v1/manifests/exists?tmdb_id=999999").await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["exists"], false); + assert!(body["manifest_id"].is_null()); +} + +#[tokio::test] +async fn exists_requires_identity_parameters() { + let s = TestServer::new("exists-noid"); + let (status, _, _) = s.get("/api/v1/manifests/exists").await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn exists_carries_rate_limit_headers() { + // §5: responses carry `X-RateLimit-Limit`, `-Remaining` and `-Reset`. + let s = TestServer::new("exists-headers"); + let (_, _, headers) = s.get("/api/v1/manifests/exists?tmdb_id=1").await; + assert_eq!(headers["x-ratelimit-limit"], "600"); + assert_eq!(headers["x-ratelimit-remaining"], "599"); + assert!(headers.contains_key("x-ratelimit-reset")); +} + +#[tokio::test] +async fn batch_exists_is_positional_and_capped_at_100() { + // §4: results are positional, and the cap is what lets §5 be generous per + // request while staying strict per item. + let s = TestServer::new("exists-batch"); + let items: Vec = (0..3).map(|i| json!({ "tmdb_id": (100 + i).to_string() })).collect(); + let (status, body, headers) = + s.post_json("/api/v1/manifests/exists", &json!({ "items": items })).await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["results"].as_array().unwrap().len(), 3); + assert_eq!(headers["x-ratelimit-limit"], "60", "batch has its own §5 budget"); + + let too_many: Vec = + (0..101).map(|i| json!({ "tmdb_id": (100 + i).to_string() })).collect(); + let (status, _, _) = + s.post_json("/api/v1/manifests/exists", &json!({ "items": too_many })).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn batch_exists_rejects_unknown_fields() { + let s = TestServer::new("exists-batch-strict"); + let (status, _, _) = + s.post_json("/api/v1/manifests/exists", &json!({ "items": [], "extra": 1 })).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn one_bad_item_does_not_fail_the_whole_batch() { + // A 100-item sweep should not be lost to one malformed entry. + let s = TestServer::new("exists-batch-partial"); + let (status, body, _) = s + .post_json( + "/api/v1/manifests/exists", + &json!({ "items": [ { "tmdb_id": "1" }, { }, { "tmdb_id": "2" } ] }), + ) + .await; + assert_eq!(status, StatusCode::OK); + let results = body["results"].as_array().unwrap(); + assert_eq!(results.len(), 3); + assert_eq!(results[1]["exists"], false); +} + +// --------------------------------------------------------------------------- +// §5 — rate limiting +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn exceeding_a_limit_returns_429_with_retry_after() { + // §5: exceeding a limit returns `429` with `Retry-After`, which the JRay + // client must honour. + let s = TestServer::new("ratelimit"); + let token = s.token().await; + + // The bundle surface has the tightest write limit (20/hour), so it is the + // cheapest to exhaust. + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396", "title": "Breaking Bad" }, + "episodes": [] + }); + + let mut saw_429 = false; + for _ in 0..25 { + let (status, _, headers) = + s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + if status == StatusCode::TOO_MANY_REQUESTS { + assert!(headers.contains_key("retry-after"), "429 must carry Retry-After"); + saw_429 = true; + break; + } + } + assert!(saw_429, "the §5 bundle limit should engage within 25 requests"); +} + +#[tokio::test] +async fn read_surfaces_have_independent_budgets() { + // §5: each surface has its own budget, so a library sweep hammering `exists` + // cannot exhaust the budget a fetch needs. + // + // Only the `exists` surface returns a body on an empty database; the fetch + // surfaces 404 (and a 404 carries no quota headers, by design). So the + // independence is asserted by consuming `exists` and observing that its + // counter alone moves. + let s = TestServer::new("surfaces"); + + let (_, _, h) = s.get("/api/v1/manifests/exists?tmdb_id=1").await; + assert_eq!(h["x-ratelimit-limit"], "600"); + assert_eq!(h["x-ratelimit-remaining"], "599"); + + // A fetch and a series request in between must not consume `exists` budget. + let _ = s.get("/api/v1/manifests/movie?tmdb_id=1").await; + let _ = s.get("/api/v1/manifests/series/1396").await; + + let (_, _, h) = s.get("/api/v1/manifests/exists?tmdb_id=1").await; + assert_eq!( + h["x-ratelimit-remaining"], "598", + "fetch requests must not draw down the exists budget" + ); + + // And the batch form is a separate surface again (§5). + let (_, _, h) = + s.post_json("/api/v1/manifests/exists", &json!({ "items": [ { "tmdb_id": "1" } ] })).await; + assert_eq!(h["x-ratelimit-limit"], "60"); + assert_eq!(h["x-ratelimit-remaining"], "59"); +} + +#[tokio::test] +async fn a_forged_forwarded_header_cannot_reset_a_budget() { + // §8: the app must trust `X-Forwarded-For` only from the operator's proxy, + // because §5 rate limiting keys on client IP. This server has no configured + // proxies, so the header must be ignored entirely — otherwise a client could + // mint a fresh budget per request. + let s = TestServer::new("xff"); + + let mut last_remaining = u32::MAX; + for i in 0..3 { + let req = Request::builder() + .uri("/api/v1/manifests/exists?tmdb_id=1") + .header("x-forwarded-for", format!("10.1.1.{i}")) + .body(Body::empty()) + .unwrap(); + let (_, _, headers) = s.send(req).await; + let remaining: u32 = headers["x-ratelimit-remaining"].to_str().unwrap().parse().unwrap(); + assert!( + remaining < last_remaining, + "budget must keep decreasing despite a changing X-Forwarded-For" + ); + last_remaining = remaining; + } +} + +// --------------------------------------------------------------------------- +// §4 — fetch +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn fetch_requires_identity_parameters() { + let s = TestServer::new("fetch-noid"); + let (status, _, _) = s.get("/api/v1/manifests/movie").await; + assert_eq!(status, StatusCode::BAD_REQUEST); + + let (status, _, _) = s.get("/api/v1/manifests/episode?series_tmdb_id=1396").await; + assert_eq!(status, StatusCode::BAD_REQUEST, "episode fetch needs season and episode"); +} + +#[tokio::test] +async fn fetching_an_absent_manifest_is_404() { + // §4: `404` if none clears `loose`. + let s = TestServer::new("fetch-absent"); + let (status, _, _) = s.get("/api/v1/manifests/movie?tmdb_id=999999").await; + assert_eq!(status, StatusCode::NOT_FOUND); +} + +#[tokio::test] +async fn series_bundle_for_an_unknown_series_is_404() { + let s = TestServer::new("series-absent"); + let (status, _, _) = s.get("/api/v1/manifests/series/999999").await; + assert_eq!(status, StatusCode::NOT_FOUND); +} + +#[tokio::test] +async fn status_of_an_unknown_manifest_reports_rejected() { + // §6 deletes rejected manifests, so a vanished id must not read as a bad + // route — the contributor polling it needs a verdict. + let s = TestServer::new("status-unknown"); + let (status, body, _) = s.get("/api/v1/manifests/01HZZZZZZZZZZZZZZZZZZZZZZZ/status").await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "rejected"); +} + +// --------------------------------------------------------------------------- +// §2, §4 — bundles +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn bundle_upload_is_not_atomic() { + // §2: valid episodes are accepted and invalid ones rejected, with a + // per-episode result list. All-or-nothing would let one bad episode discard an + // entire season's compute. + let s = TestServer::new("bundle-partial"); + let token = s.token().await; + + let good = |ep: i64| { + json!({ + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "1396", "title": "Breaking Bad", + "season": 1, "episode": ep }, + "cut": { "runtime_sec": 2820.0 }, + "actors": [ { "name": "Bryan Cranston", "tmdb_id": "17419", + "scenes": [[10.0, 20.0]] } ] + }) + }; + // Invalid: a scene window beyond the runtime tolerance (§6). + let bad = json!({ + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "1396", "season": 1, "episode": 3 }, + "cut": { "runtime_sec": 2820.0 }, + "actors": [ { "tmdb_id": "17419", "scenes": [[10.0, 99999.0]] } ] + }); + + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396", "title": "Breaking Bad" }, + "episodes": [ good(1), bad, good(2) ] + }); + + let (status, body, _) = s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + assert_eq!(status, StatusCode::ACCEPTED, "{body}"); + let results = body["results"].as_array().unwrap(); + assert_eq!(results.len(), 3); + assert_eq!(results[0]["status"], "pending"); + assert_eq!(results[1]["status"], "rejected"); + assert!(results[1]["reason"].is_string(), "a rejected episode should say why"); + assert_eq!(results[2]["status"], "pending", "a later episode must still be accepted"); +} + +#[tokio::test] +async fn bundle_envelope_errors_are_whole_request_400s() { + // §4: `400` for the envelope itself, whereas individual bad episodes are + // reported in the results list. + let s = TestServer::new("bundle-envelope"); + let token = s.token().await; + + let (status, _, _) = s + .post_json_auth( + "/api/v1/manifests/bundle", + &token, + &json!({ "jmanifest_version": 1, "series": {}, "episodes": [] }), + ) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST, "series needs an identifier"); + + let (status, _, _) = s + .post_json_auth( + "/api/v1/manifests/bundle", + &token, + &json!({ "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396" }, + "episodes": [], "extra": 1 }), + ) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST, "unknown envelope field"); +} + +#[tokio::test] +async fn bundle_rejects_an_episode_contradicting_the_envelope() { + // An episode must not be silently reattributed to the bundle's series. + let s = TestServer::new("bundle-mismatch"); + let token = s.token().await; + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396" }, + "episodes": [ { + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "9999", "season": 1, "episode": 1 }, + "cut": { "runtime_sec": 2820.0 }, + "actors": [ { "tmdb_id": "17419", "scenes": [[1.0, 2.0]] } ] + } ] + }); + let (status, body, _) = s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + assert_eq!(status, StatusCode::ACCEPTED); + assert_eq!(body["results"][0]["status"], "rejected"); +} + +#[tokio::test] +async fn bundle_beyond_the_episode_cap_is_413() { + // §2/§4: capped at 500 episodes; beyond that the client must page by season. + let s = TestServer::new("bundle-cap"); + let token = s.token().await; + let episodes: Vec = (0..501) + .map(|i| { + json!({ + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "1396", + "season": 1, "episode": i }, + "cut": { "runtime_sec": 2820.0 }, + "actors": [ { "tmdb_id": "17419", "scenes": [[1.0, 2.0]] } ] + }) + }) + .collect(); + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396" }, + "episodes": episodes + }); + let (status, _, _) = s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + assert_eq!(status, StatusCode::PAYLOAD_TOO_LARGE); +} + +#[tokio::test] +async fn bundle_route_accepts_a_body_larger_than_the_single_manifest_cap() { + // §6 stage 1: per-route limits, so the bundle endpoint gets its larger cap + // without widening the others. A ~3 MiB bundle exceeds the 2 MiB manifest cap + // but is well within the 25 MiB bundle cap. + let s = TestServer::new("bundle-bigger-cap"); + let token = s.token().await; + + let episodes: Vec = (1..=60) + .map(|ep| { + let scenes: Vec = (0..600).map(|j| json!([j as f64, (j + 1) as f64])).collect(); + json!({ + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "1396", + "season": 1, "episode": ep }, + "cut": { "runtime_sec": 2820.0 }, + "actors": (0..8).map(|a| json!({ + "tmdb_id": (20000 + a).to_string(), "scenes": scenes + })).collect::>() + }) + }) + .collect(); + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396" }, + "episodes": episodes + }); + let encoded = bundle.to_string(); + assert!( + encoded.len() > 2 * 1024 * 1024, + "test body should exceed the single-manifest cap, got {} bytes", + encoded.len() + ); + + let (status, _, _) = s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + assert_eq!(status, StatusCode::ACCEPTED, "the bundle route has its own larger cap"); +} + +// --------------------------------------------------------------------------- +// §4 — reports +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn reporting_an_unknown_manifest_is_404() { + let s = TestServer::new("report-unknown"); + let (status, _, _) = s + .post_json( + "/api/v1/manifests/01HZZZZZZZZZZZZZZZZZZZZZZZ/report", + &json!({ "reason": "misaligned" }), + ) + .await; + assert_eq!(status, StatusCode::NOT_FOUND); +} + +#[tokio::test] +async fn a_report_is_accepted_and_does_not_delist() { + // §5a: delisting stays an operator action. Automatic delisting on report would + // hand any client a remote delete primitive. + let s = TestServer::new("report-ok"); + let token = s.token().await; + let (_, body, _) = + s.post_json_auth("/api/v1/manifests", &token, &movie_manifest("504172", 6420.5)).await; + let id = body["manifest_id"].as_str().unwrap().to_string(); + + let (status, body, _) = s + .post_json( + &format!("/api/v1/manifests/{id}/report"), + &json!({ "reason": "wrong_actors", "note": "these are not the right people" }), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert!(body["report_id"].is_string()); + + let (status, body, _) = s.get(&format!("/api/v1/manifests/{id}/status")).await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "pending", "a report must not change status by itself"); +} + +#[tokio::test] +async fn report_rejects_an_unknown_reason_and_unknown_fields() { + let s = TestServer::new("report-strict"); + let (status, _, _) = s + .post_json("/api/v1/manifests/x/report", &json!({ "reason": "i_just_dont_like_it" })) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST); + + let (status, _, _) = s + .post_json("/api/v1/manifests/x/report", &json!({ "reason": "spam", "extra": true })) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn report_note_is_length_capped() { + // §5a: free text from an anonymous caller is capped hard. + let s = TestServer::new("report-note"); + let token = s.token().await; + let (_, body, _) = + s.post_json_auth("/api/v1/manifests", &token, &movie_manifest("504172", 6420.5)).await; + let id = body["manifest_id"].as_str().unwrap().to_string(); + + let (status, _, _) = s + .post_json( + &format!("/api/v1/manifests/{id}/report"), + &json!({ "reason": "spam", "note": "a".repeat(5000) }), + ) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +// --------------------------------------------------------------------------- +// Malformed input +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn malformed_json_is_a_400_not_a_500() { + let s = TestServer::new("bad-json"); + let token = s.token().await; + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from("{ this is not json")) + .unwrap(); + let (status, _, _) = s.send(req).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn deeply_nested_json_does_not_crash_the_parser() { + // §6 stage 1 caps nesting depth; a parser handed unbounded input is a DoS + // primitive, so the failure must be a clean rejection. + let s = TestServer::new("deep-json"); + let token = s.token().await; + let deep = format!("{}{}", "[".repeat(5000), "]".repeat(5000)); + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(deep)) + .unwrap(); + let (status, _, _) = s.send(req).await; + assert!( + status == StatusCode::BAD_REQUEST || status == StatusCode::PAYLOAD_TOO_LARGE, + "deeply nested input should be rejected cleanly, got {status}" + ); +} + +#[tokio::test] +async fn unknown_routes_are_404() { + let s = TestServer::new("routes"); + let (status, _, _) = s.get("/api/v1/nonexistent").await; + assert_eq!(status, StatusCode::NOT_FOUND); + let (status, _, _) = s.get("/api/v2/manifests/exists?tmdb_id=1").await; + assert_eq!(status, StatusCode::NOT_FOUND); +} diff --git a/tests/injection.rs b/tests/injection.rs new file mode 100644 index 0000000..fc248ad --- /dev/null +++ b/tests/injection.rs @@ -0,0 +1,634 @@ +//! Injection resistance — SQL, JSON and header. +//! +//! These are regression tests for properties the design already provides, kept +//! separate from `api.rs` because their purpose is different: `api.rs` asserts the +//! spec's behaviour, this asserts that hostile input cannot escape its layer. +//! +//! Two distinct defences are at work, and it is worth being precise about which +//! applies where, because they fail differently: +//! +//! 1. **Parameterised queries** (§7, §8). Every value reaches SQLite through +//! `params![]`; the only `format!`-built SQL interpolates compile-time +//! constants (a column list and a status literal). So a value carrying SQL +//! syntax is bound as *data* and simply matches nothing. +//! 2. **Closed-vocabulary validation** (§5a, §6 stage 2). Identifiers are +//! regex-constrained and free text is restricted to a closed character class, +//! so most injection strings are rejected before they reach the database. +//! +//! Defence 1 is what actually prevents injection; defence 2 means an attacker +//! usually cannot even reach it. Testing both matters: if validation were ever +//! loosened, these tests should still pass on the strength of parameterisation +//! alone. + +use std::sync::Arc; + +use axum::body::Body; +use axum::http::{Request, StatusCode}; +use http_body_util::BodyExt; +use jray_server::app; +use jray_server::config::Config; +use jray_server::db::Db; +use jray_server::ratelimit::RateLimiter; +use jray_server::state::AppState; +use jray_server::tmdb::TmdbClient; +use serde_json::{json, Value}; +use tower::ServiceExt; + +/// Payloads spanning the usual SQL-injection shapes: boolean tautology, statement +/// termination, stacked statements, UNION exfiltration, comment truncation, and +/// string-concatenation exfiltration. +const SQL_PAYLOADS: &[&str] = &[ + "1' OR '1'='1", + "1'; DROP TABLE manifests;--", + "1 UNION SELECT token_hash FROM contributors", + "' OR 1=1--", + "1'||(SELECT token_hash FROM contributors)||'", + "1)) OR 1=1 --", + "'; UPDATE manifests SET status='listed' WHERE 1=1;--", + "1/**/UNION/**/SELECT/**/1", + "x' AND (SELECT COUNT(*) FROM sqlite_master)>0 --", + "\"; DELETE FROM scenes; --", +]; + +struct TestServer { + router: axum::Router, + db: Db, + _dir: TempDir, +} + +struct TempDir(std::path::PathBuf); + +impl TempDir { + fn new(tag: &str) -> Self { + let mut p = std::env::temp_dir(); + p.push(format!("jray-inj-{}-{}", tag, unique())); + std::fs::create_dir_all(&p).expect("creating temp dir"); + Self(p) + } + fn db_path(&self) -> String { + self.0.join("test.db").to_string_lossy().into_owned() + } +} + +impl Drop for TempDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } +} + +fn unique() -> String { + use std::sync::atomic::{AtomicU64, Ordering}; + static N: AtomicU64 = AtomicU64::new(0); + let t = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_nanos()) + .unwrap_or(0); + format!("{t}-{}", N.fetch_add(1, Ordering::Relaxed)) +} + +impl TestServer { + fn new(tag: &str) -> Self { + let dir = TempDir::new(tag); + let db = Db::open(&dir.db_path()).expect("opening database"); + let config = Arc::new(Config { + bind: "127.0.0.1:0".into(), + db_path: dir.db_path(), + tmdb_api_key: None, + tmdb_base_url: "http://127.0.0.1:1".into(), + trusted_proxies: Vec::new(), + server_id: "test.example".into(), + request_timeout: std::time::Duration::from_secs(30), + job_batch: 8, + job_poll_interval: std::time::Duration::from_secs(3600), + }); + let state = AppState { + db: db.clone(), + config: config.clone(), + limiter: Arc::new(RateLimiter::new()), + tmdb: Arc::new(TmdbClient::new(config.tmdb_base_url.clone(), None)), + }; + Self { router: app::router(state), db, _dir: dir } + } + + async fn send(&self, req: Request) -> (StatusCode, Value) { + let resp = self.router.clone().oneshot(req).await.expect("router call"); + let status = resp.status(); + let bytes = resp.into_body().collect().await.expect("body").to_bytes(); + let body = if bytes.is_empty() { + Value::Null + } else { + serde_json::from_slice(&bytes) + .unwrap_or(Value::String(String::from_utf8_lossy(&bytes).into_owned())) + }; + (status, body) + } + + async fn get(&self, uri: &str) -> (StatusCode, Value) { + self.send(Request::builder().uri(uri).body(Body::empty()).unwrap()).await + } + + async fn post(&self, uri: &str, token: Option<&str>, body: &Value) -> (StatusCode, Value) { + let mut b = + Request::builder().method("POST").uri(uri).header("content-type", "application/json"); + if let Some(t) = token { + b = b.header("authorization", format!("Bearer {t}")); + } + self.send(b.body(Body::from(body.to_string())).unwrap()).await + } + + async fn token(&self) -> String { + let (_, body) = self.post("/api/v1/tokens", None, &json!({})).await; + body["token"].as_str().expect("token").to_string() + } + + /// Confirms the schema is intact and the expected row counts hold. + /// + /// A successful injection would most likely drop a table or delete rows, so + /// this is the assertion that actually matters after each payload. + async fn assert_schema_intact(&self) { + let tables: Vec = self + .db + .read(|conn| { + let mut stmt = conn + .prepare("SELECT name FROM sqlite_master WHERE type='table' ORDER BY name")?; + let rows = stmt + .query_map([], |r| r.get::<_, String>(0))? + .collect::>>()?; + Ok(rows) + }) + .await + .expect("listing tables"); + + for expected in [ + "contributors", + "jobs", + "manifest_actors", + "manifests", + "people", + "reports", + "scenes", + "titles", + "tmdb_cache", + ] { + assert!( + tables.iter().any(|t| t == expected), + "table {expected} is missing — an injection may have dropped it. tables: {tables:?}" + ); + } + } +} + +fn urlencode(s: &str) -> String { + let mut out = String::new(); + for b in s.bytes() { + match b { + b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => { + out.push(b as char) + } + _ => out.push_str(&format!("%{b:02X}")), + } + } + out +} + +// --------------------------------------------------------------------------- +// SQL injection — query parameters +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn sql_payloads_in_query_parameters_are_inert() { + let s = TestServer::new("query"); + + for payload in SQL_PAYLOADS { + let enc = urlencode(payload); + for uri in [ + format!("/api/v1/manifests/exists?tmdb_id={enc}"), + format!("/api/v1/manifests/exists?imdb_id={enc}"), + format!("/api/v1/manifests/movie?tmdb_id={enc}"), + format!("/api/v1/manifests/movie?imdb_id={enc}"), + format!("/api/v1/manifests/episode?series_tmdb_id={enc}&season=1&episode=1"), + format!("/api/v1/manifests/series/{enc}"), + format!("/api/v1/manifests/exists?tmdb_id=1&video_hash={enc}"), + ] { + let (status, body) = s.get(&uri).await; + // The payload is bound as data, so it matches nothing. What must never + // happen is a 5xx, which would mean SQLite saw it as syntax. + assert!( + status.is_success() + || status == StatusCode::NOT_FOUND + || status == StatusCode::BAD_REQUEST, + "payload {payload:?} on {uri} produced {status} — expected data-not-found, \ + not a server error. body: {body}" + ); + } + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn sql_payloads_in_path_parameters_are_inert() { + let s = TestServer::new("path"); + + for payload in SQL_PAYLOADS { + let enc = urlencode(payload); + for uri in [ + format!("/api/v1/manifests/{enc}"), + format!("/api/v1/manifests/{enc}/status"), + format!("/api/v1/manifests/series/{enc}"), + ] { + let (status, body) = s.get(&uri).await; + assert!( + !status.is_server_error(), + "payload {payload:?} on {uri} produced {status}: {body}" + ); + } + } + + s.assert_schema_intact().await; +} + +// --------------------------------------------------------------------------- +// SQL injection — JSON body fields +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn sql_payloads_in_identifier_fields_are_rejected() { + // §6 stage 2 regex-constrains every identifier, so these never even reach the + // query layer. The response must be a clean 400 naming the field. + let s = TestServer::new("body-ids"); + let token = s.token().await; + + for payload in SQL_PAYLOADS { + let manifest = json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": payload }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }); + let (status, body) = s.post("/api/v1/manifests", Some(&token), &manifest).await; + assert_eq!( + status, + StatusCode::BAD_REQUEST, + "payload {payload:?} should be rejected by validation: {body}" + ); + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn sql_payloads_in_free_text_fields_are_rejected() { + // The two free-text fields (§5a) are the only place arbitrary strings could + // arrive. The closed character class excludes quotes, semicolons and digits, + // which is what makes SQL syntax unrepresentable there. + let s = TestServer::new("body-text"); + let token = s.token().await; + + for payload in SQL_PAYLOADS { + for manifest in [ + json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172", "title": payload }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "name": payload, "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + ] { + let (status, body) = s.post("/api/v1/manifests", Some(&token), &manifest).await; + assert_eq!( + status, + StatusCode::BAD_REQUEST, + "free-text payload {payload:?} should be rejected: {body}" + ); + } + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn sql_payloads_in_a_report_note_cannot_escape() { + // `note` is the one field that accepts relatively free text (control + // characters stripped, length capped) because only the operator reads it. It + // reaches the database, so it is the strongest test of parameterisation: + // validation is *not* filtering SQL syntax here. + let s = TestServer::new("report-note"); + let token = s.token().await; + + let (_, body) = s + .post( + "/api/v1/manifests", + Some(&token), + &json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + ) + .await; + let id = body["manifest_id"].as_str().expect("manifest id").to_string(); + + for payload in SQL_PAYLOADS { + let (status, body) = s + .post( + &format!("/api/v1/manifests/{id}/report"), + None, + &json!({ "reason": "spam", "note": payload }), + ) + .await; + assert!( + status.is_success() || status == StatusCode::TOO_MANY_REQUESTS, + "note payload {payload:?} produced {status}: {body}" + ); + if status.is_success() { + s.assert_schema_intact().await; + } + } + + // The notes were stored verbatim as *data* — proving they were bound, not + // executed. Verified by reading them back out. + let stored: i64 = + s.db.read(|conn| Ok(conn.query_row("SELECT COUNT(*) FROM reports", [], |r| r.get(0))?)) + .await + .expect("counting reports"); + assert!(stored > 0, "reports should have been stored as inert data"); +} + +#[tokio::test] +async fn sql_payloads_in_a_bearer_token_are_inert() { + // The token is hashed before it reaches any query, but a payload arriving via + // a header must still not produce a 5xx. + let s = TestServer::new("token-inj"); + + for payload in SQL_PAYLOADS { + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {payload}")) + .body(Body::from( + json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }) + .to_string(), + )) + .unwrap(); + let (status, body) = s.send(req).await; + assert_eq!( + status, + StatusCode::UNAUTHORIZED, + "token payload {payload:?} produced {status}: {body}" + ); + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn sql_payloads_in_the_batch_exists_body_are_inert() { + let s = TestServer::new("batch-inj"); + let items: Vec = SQL_PAYLOADS.iter().map(|p| json!({ "tmdb_id": p })).collect(); + let (status, body) = s.post("/api/v1/manifests/exists", None, &json!({ "items": items })).await; + assert_eq!(status, StatusCode::OK, "{body}"); + // Each malformed item degrades to "absent" rather than erroring the batch. + for result in body["results"].as_array().expect("results") { + assert_eq!(result["exists"], false); + } + s.assert_schema_intact().await; +} + +// --------------------------------------------------------------------------- +// JSON injection / parser abuse +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn json_structure_abuse_is_rejected_cleanly() { + // A parser handed hostile structure must fail with 400/413, never 5xx and + // never a hang (§6 stage 1: "a parser handed an unbounded body is a + // denial-of-service primitive"). + let s = TestServer::new("json-abuse"); + let token = s.token().await; + + let cases: Vec<(&str, String)> = vec![ + ("deep nesting", format!("{}{}", "[".repeat(20_000), "]".repeat(20_000))), + ("unterminated", "{\"identity\": {\"type\": \"movie\"".to_string()), + ("duplicate keys", r#"{"jmanifest_version":1,"jmanifest_version":2}"#.to_string()), + ("null bytes", "{\"jmanifest_version\":\u{0}1}".to_string()), + ("huge number", format!("{{\"jmanifest_version\":{}}}", "9".repeat(5000))), + ("nan literal", r#"{"jmanifest_version":1,"cut":{"runtime_sec":NaN}}"#.to_string()), + ("bare array", "[1,2,3]".to_string()), + ("bare string", "\"just a string\"".to_string()), + ("empty body", String::new()), + ( + "prototype-style key", + r#"{"__proto__":{"admin":true},"jmanifest_version":1}"#.to_string(), + ), + ]; + + for (label, body) in cases { + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(body)) + .unwrap(); + let (status, resp) = s.send(req).await; + assert!( + status == StatusCode::BAD_REQUEST || status == StatusCode::PAYLOAD_TOO_LARGE, + "{label} produced {status}, expected a clean rejection: {resp}" + ); + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn non_finite_scene_times_are_rejected() { + // §6 explicitly rejects NaN/Infinity. They cannot arrive as JSON literals, but + // they can arrive as overflowing decimals, which parse to f64 infinity. + let s = TestServer::new("nonfinite"); + let token = s.token().await; + + // Sent as raw JSON text rather than via `json!`, because rustc refuses an + // out-of-range float literal — and the point is to make the *server's* parser + // handle it, which is the real attack path. + let raw = r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172"}, + "cut":{"runtime_sec":100.0}, + "actors":[{"tmdb_id":"884","scenes":[[1.0,1e400]]}]}"#; + + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(raw)) + .unwrap(); + let (status, body) = s.send(req).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "{body}"); + + // Likewise an overflowing runtime. + let raw = r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172"}, + "cut":{"runtime_sec":1e400}, + "actors":[{"tmdb_id":"884","scenes":[[1.0,2.0]]}]}"#; + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(raw)) + .unwrap(); + let (status, body) = s.send(req).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "{body}"); +} + +// --------------------------------------------------------------------------- +// Header injection +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn crlf_in_a_header_value_cannot_split_the_response() { + // A CRLF-carrying header value must not appear in the response as new headers. + // `http` rejects such values at construction, so this asserts the invariant + // holds at the boundary rather than relying on our own escaping. + let bad = "1.2.3.4\r\nX-Injected: yes"; + assert!( + axum::http::HeaderValue::from_str(bad).is_err(), + "the http crate must refuse CRLF in header values" + ); + + // And a percent-encoded variant reaching a handler stays inert data. + let s = TestServer::new("crlf"); + let (status, _) = s.get("/api/v1/manifests/exists?tmdb_id=1%0D%0AX-Injected:%20yes").await; + assert!(!status.is_server_error()); + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn oversized_headers_do_not_take_the_server_down() { + let s = TestServer::new("big-header"); + let big = "a".repeat(100_000); + let req = Request::builder() + .uri("/api/v1/manifests/exists?tmdb_id=1") + .header("x-filler", big) + .body(Body::empty()) + .unwrap(); + let (status, _) = s.send(req).await; + assert!(!status.is_server_error(), "got {status}"); +} + +// --------------------------------------------------------------------------- +// Path traversal +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn path_traversal_attempts_reach_no_filesystem() { + // The server serves no files at all, so traversal has nowhere to go. Asserted + // anyway, because the manifest id is a path segment. + let s = TestServer::new("traversal"); + for probe in [ + "..%2F..%2F..%2Fetc%2Fpasswd", + "....%2F%2F....%2F%2Fetc%2Fpasswd", + "%2e%2e%2f%2e%2e%2fetc%2fshadow", + "..%5C..%5Cwindows%5Csystem32", + "%00/etc/passwd", + ] { + let (status, body) = s.get(&format!("/api/v1/manifests/{probe}")).await; + assert!( + status == StatusCode::NOT_FOUND || status == StatusCode::BAD_REQUEST, + "probe {probe} produced {status}: {body}" + ); + // Nothing that looks like file content should ever come back. + let text = body.to_string(); + assert!(!text.contains("root:"), "probe {probe} returned passwd-like content"); + } +} + +// --------------------------------------------------------------------------- +// Unicode and encoding tricks against the §5a character class +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn unicode_tricks_cannot_smuggle_text_past_the_character_class() { + // §5a's class is checked *after* NFC normalisation, so decomposed and + // compatibility forms must not provide a way in. Fullwidth digits are the + // sharpest case: NFKC would fold them to ASCII digits, but NFC does not, and + // they are `Nd` (not a letter), so the class rejects them either way. + let s = TestServer::new("unicode"); + let token = s.token().await; + + for payload in [ + "Actor 123", // fullwidth letters and digits + "Steve\u{FEFF}Buscemi", // zero-width no-break space + "Ste\u{0301}ve\u{202E}", // combining acute plus bidi override + "𝐒𝐭𝐞𝐯𝐞", // mathematical bold (compatibility form) + "Steve\u{2028}Buscemi", // line separator + "\u{1F600} Actor", // emoji + "Actor\u{00A0}Name\u{0000}", // nbsp plus NUL + ] { + let (status, body) = s + .post( + "/api/v1/manifests", + Some(&token), + &json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "name": payload, "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + ) + .await; + assert_eq!( + status, + StatusCode::BAD_REQUEST, + "unicode payload {payload:?} should be rejected: {body}" + ); + } +} + +#[tokio::test] +async fn the_audio_signature_field_cannot_carry_arbitrary_bytes() { + // §3: a variable-length blob would be a payload channel — "precisely what §5a + // closes". Length is fixed and every byte is structurally constrained. + let s = TestServer::new("audio-sig"); + let token = s.token().await; + + for sig in [ + "v1:aGVsbG8gd29ybGQ=", // too short to be a signature + &format!("v1:{}", "/".repeat(4000)), // high bit set throughout + &format!("v1:{}", "A".repeat(100_000)), // oversized + "not-base64-at-all", // missing version prefix + &"A".repeat(1720), // unprefixed + ] { + let (status, body) = s + .post( + "/api/v1/manifests", + Some(&token), + &json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 6420.5, "audio_signature": sig }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + ) + .await; + assert_eq!( + status, + StatusCode::BAD_REQUEST, + "signature {:?} should be rejected: {body}", + &sig.chars().take(40).collect::() + ); + } +}