commit a848750a659612c69c3a12b497eddf7eeee2e925 Author: Duncan Tourolle Date: Thu Jul 30 18:14:02 2026 +0200 Initial implementation: core vertical slice Implements the core of SPEC.md — the manifest exchange, less audio-tier matching (§3) and federation (§9a), both of which the spec sequences as later work. - §2 Jmanifest format and series bundles - §3 cut matching: exact / runtime / loose tiers - §4 API, less POST /manifests/search - §5 rate limiting; §5a trust model, anonymous bearer tokens - §6 upload validation, all four stages - §7 relational storage, no JSON blob on the write path - §8 Rust + Axum + SQLite, single serialized writer, in-process job queue - §9a content addressing, computed on upload Reconciled against the system spec: - anneal_sec removed, withdrawn upstream by AR-012/AR-013. Presence follows track extent, so a track survives its own gaps and there is nothing to anneal. Its successor extinction_sec and the new gallery_scope are accepted and stored; scope enters the §7 ranking. A manifest still carrying anneal_sec is a hard 400, not silently ignored — it came from a pipeline whose window semantics differ from what this server assumes. - Audio signature: media under 120 s now emits no signature at all, matching scene-actor-extraction IR-007. The earlier §3 draft allowed a shortened window under 150 s, which was the weaker rule — a caller-varying length is the property SR-004 forbids. - UR IDs regularised to UR-nnn; docs/requirements.md registers 32 requirements, each tracing to an SR-nnn or PR-nnn. 189 tests: unit, end-to-end through the real router, and an injection suite covering SQL, JSON, header and Unicode payloads. Writing that suite found two real gaps, both fixed here: compatibility homoglyphs passed the §5a character class, and a one-frame audio signature was accepted on a feature-length item. Co-Authored-By: Claude Opus 5 diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..4ab1a13 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,22 @@ +# Keep the build context small and free of host state. +target/ +.git/ +.gitea/ + +# Never ship an operator's database or key into an image layer. +*.db +*.db-wal +*.db-shm +*.sqlite +*.sqlite3 +.env +.env.* + +# Not needed to build. +tests/ +README.md +SPEC.md +deny.toml +Dockerfile +.dockerignore +.gitignore diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml new file mode 100644 index 0000000..a567981 --- /dev/null +++ b/.gitea/workflows/ci.yml @@ -0,0 +1,151 @@ +# Gitea Actions CI. +# +# Gitea Actions is workflow-compatible with GitHub Actions, so this runs on either +# with no changes. It needs a registered runner with the `ubuntu-latest` label. +# +# The gates, in the order they fail fastest: +# fmt — formatting, seconds +# clippy — lints, denied rather than warned +# test — 160 unit + integration tests +# deny — RustSec advisories, licence policy, source policy +# musl — the artifact §8 actually ships: one static binary + +name: CI + +on: + push: + branches: [main, master] + pull_request: + # Advisories appear without any code changing, so the dependency audit also + # runs on a schedule rather than only on push. + schedule: + - cron: "0 6 * * 1" + +env: + CARGO_TERM_COLOR: always + # Fail the build on warnings. The tree is warning-clean, so keeping it that way + # is cheaper than letting warnings accumulate. + RUSTFLAGS: "-D warnings" + +jobs: + check: + name: fmt, clippy, test + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Install Rust + run: | + # rustup is not guaranteed present on a self-hosted Gitea runner. + if ! command -v rustup >/dev/null 2>&1; then + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + | sh -s -- -y --profile minimal --component rustfmt,clippy + echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" + else + rustup component add rustfmt clippy + fi + + - name: Cache cargo + uses: actions/cache@v4 + with: + path: | + ~/.cargo/registry + ~/.cargo/git + target + key: ${{ runner.os }}-cargo-${{ hashFiles('Cargo.lock') }} + restore-keys: ${{ runner.os }}-cargo- + + - name: Formatting + run: cargo fmt --all -- --check + + - name: Clippy + run: cargo clippy --all-targets --all-features + + - name: Tests + run: cargo test --all-features + + deny: + name: advisories and licences + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Install Rust + run: | + if ! command -v rustup >/dev/null 2>&1; then + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + | sh -s -- -y --profile minimal + echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" + fi + + - name: Cache cargo-deny + uses: actions/cache@v4 + with: + path: ~/.cargo/bin/cargo-deny + key: ${{ runner.os }}-cargo-deny + + - name: Install cargo-deny + run: | + command -v cargo-deny >/dev/null 2>&1 || cargo install cargo-deny --locked + + # Advisories, licences, bans and sources — see deny.toml for why the licence + # allow-list is closed rather than a deny-list. + - name: cargo deny + run: cargo deny check + + musl: + name: static musl binary + runs-on: ubuntu-latest + # Only gate merges on the artifact build once the cheaper checks have passed. + needs: check + steps: + - uses: actions/checkout@v4 + + - name: Install Rust and musl target + run: | + if ! command -v rustup >/dev/null 2>&1; then + curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + | sh -s -- -y --profile minimal + echo "$HOME/.cargo/bin" >> "$GITHUB_PATH" + export PATH="$HOME/.cargo/bin:$PATH" + fi + rustup target add x86_64-unknown-linux-musl + sudo apt-get update && sudo apt-get install -y musl-tools + + - name: Cache cargo + uses: actions/cache@v4 + with: + path: | + ~/.cargo/registry + ~/.cargo/git + target + key: ${{ runner.os }}-musl-${{ hashFiles('Cargo.lock') }} + restore-keys: ${{ runner.os }}-musl- + + # §8: "Ship a single static binary (musl target) plus the SQLite file." + # rusqlite is built with `bundled`, so SQLite is compiled in; reqwest uses + # rustls rather than OpenSSL, so there is no system TLS dependency to link. + - name: Build + run: cargo build --release --target x86_64-unknown-linux-musl + + - name: Verify the binary is actually static + run: | + BIN=target/x86_64-unknown-linux-musl/release/jray-server + file "$BIN" + # A dynamically-linked result would defeat §8's deployment story, so this + # is asserted rather than assumed. + # + # Checked with `file`, not `ldd`: the musl target produces a static-PIE, + # and `ldd` prints the musl loader for one — an `ldd`-based check reports + # a perfectly static binary as dynamic. + if ! file "$BIN" | grep -qE 'static-pie linked|statically linked'; then + echo "::error::binary is not statically linked" >&2 + exit 1 + fi + + - name: Upload binary + uses: actions/upload-artifact@v3 + with: + name: jray-server-x86_64-musl + path: target/x86_64-unknown-linux-musl/release/jray-server + if-no-files-found: error diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..a239b91 --- /dev/null +++ b/.gitignore @@ -0,0 +1,35 @@ +# Rust build artifacts +/target/ +**/*.rs.bk +*.pdb + +# Cargo.lock is committed: this crate ships a binary, so reproducible builds +# matter more than dependency-resolution freedom. + +# SQLite database and its WAL sidecars (§7, §8). Never commit an operator's data; +# note that a plain file copy of a live WAL database is not a valid backup — +# use `VACUUM INTO` or the backup API. +*.db +*.db-wal +*.db-shm +*.sqlite +*.sqlite3 + +# Local operator configuration — holds the TMDB API key (§8). +.env +.env.* +!.env.example + +# Python artefacts from the traceability tooling +__pycache__/ +*.pyc + +# Generated traceability output — regenerate with the gate, never hand-edit. +traces-report.json + +# Editor / OS noise +.vscode/ +.idea/ +*.swp +*~ +.DS_Store diff --git a/Cargo.lock b/Cargo.lock new file mode 100644 index 0000000..ee1f4c5 --- /dev/null +++ b/Cargo.lock @@ -0,0 +1,1778 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "aho-corasick" +version = "1.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +dependencies = [ + "memchr", +] + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "atomic-waker" +version = "1.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" + +[[package]] +name = "axum" +version = "0.8.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90" +dependencies = [ + "axum-core", + "bytes", + "form_urlencoded", + "futures-util", + "http", + "http-body", + "http-body-util", + "hyper", + "hyper-util", + "itoa", + "matchit", + "memchr", + "mime", + "percent-encoding", + "pin-project-lite", + "serde_core", + "serde_json", + "serde_path_to_error", + "serde_urlencoded", + "sync_wrapper", + "tokio", + "tower", + "tower-layer", + "tower-service", + "tracing", +] + +[[package]] +name = "axum-core" +version = "0.5.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "08c78f31d7b1291f7ee735c1c6780ccde7785daae9a9206026862dab7d8792d1" +dependencies = [ + "bytes", + "futures-core", + "http", + "http-body", + "http-body-util", + "mime", + "pin-project-lite", + "sync_wrapper", + "tower-layer", + "tower-service", + "tracing", +] + +[[package]] +name = "base64" +version = "0.22.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" + +[[package]] +name = "bitflags" +version = "2.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + +[[package]] +name = "bumpalo" +version = "3.20.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" + +[[package]] +name = "bytes" +version = "1.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" + +[[package]] +name = "cc" +version = "1.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5add81bb678e6cb321aff7fa0dc7689ad82b112dbc032cea19f91d6b8e3582b9" +dependencies = [ + "find-msvc-tools", + "shlex", +] + +[[package]] +name = "cfg-if" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" + +[[package]] +name = "cfg_aliases" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" + +[[package]] +name = "chacha20" +version = "0.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81" +dependencies = [ + "cfg-if", + "cpufeatures 0.3.0", + "rand_core 0.10.1", +] + +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "cpufeatures" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b2a41393f66f16b0823bb79094d54ac5fbd34ab292ddafb9a0456ac9f87d201" +dependencies = [ + "libc", +] + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + +[[package]] +name = "displaydoc" +version = "0.2.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "fallible-iterator" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" + +[[package]] +name = "fallible-streaming-iterator" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" + +[[package]] +name = "find-msvc-tools" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" + +[[package]] +name = "foldhash" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" + +[[package]] +name = "form_urlencoded" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" +dependencies = [ + "percent-encoding", +] + +[[package]] +name = "futures-channel" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "262590f4fe6afeb0bc83be1daa64e52657fe185690a958af7f3ad0e92085c5ae" +dependencies = [ + "futures-core", +] + +[[package]] +name = "futures-core" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" + +[[package]] +name = "futures-task" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" + +[[package]] +name = "futures-util" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" +dependencies = [ + "futures-core", + "futures-task", + "pin-project-lite", + "slab", +] + +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + +[[package]] +name = "getrandom" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" +dependencies = [ + "cfg-if", + "js-sys", + "libc", + "wasi", + "wasm-bindgen", +] + +[[package]] +name = "getrandom" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" +dependencies = [ + "cfg-if", + "libc", + "r-efi 5.3.0", + "wasip2", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "js-sys", + "libc", + "r-efi 6.0.0", + "rand_core 0.10.1", + "wasm-bindgen", +] + +[[package]] +name = "hashbrown" +version = "0.15.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" +dependencies = [ + "foldhash", +] + +[[package]] +name = "hashlink" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7382cf6263419f2d8df38c55d7da83da5c18aef87fc7a7fc1fb1e344edfe14c1" +dependencies = [ + "hashbrown", +] + +[[package]] +name = "http" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" +dependencies = [ + "bytes", + "itoa", +] + +[[package]] +name = "http-body" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" +dependencies = [ + "bytes", + "http", +] + +[[package]] +name = "http-body-util" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" +dependencies = [ + "bytes", + "futures-core", + "http", + "http-body", + "pin-project-lite", +] + +[[package]] +name = "httparse" +version = "1.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" + +[[package]] +name = "httpdate" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" + +[[package]] +name = "hyper" +version = "1.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d22053281f852e11534f5198498373cbb59295120a20771d90f7ed1897490a72" +dependencies = [ + "atomic-waker", + "bytes", + "futures-channel", + "futures-core", + "http", + "http-body", + "httparse", + "httpdate", + "itoa", + "pin-project-lite", + "smallvec", + "tokio", + "want", +] + +[[package]] +name = "hyper-rustls" +version = "0.27.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "33ca68d021ef39cf6463ab54c1d0f5daf03377b70561305bb89a8f83aab66e0f" +dependencies = [ + "http", + "hyper", + "hyper-util", + "rustls", + "tokio", + "tokio-rustls", + "tower-service", + "webpki-roots", +] + +[[package]] +name = "hyper-util" +version = "0.1.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "96547c2556ec9d12fb1578c4eaf448b04993e7fb79cbaad930a656880a6bdfa0" +dependencies = [ + "base64", + "bytes", + "futures-channel", + "futures-util", + "http", + "http-body", + "hyper", + "ipnet", + "libc", + "percent-encoding", + "pin-project-lite", + "socket2", + "tokio", + "tower-service", + "tracing", +] + +[[package]] +name = "icu_collections" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c" +dependencies = [ + "displaydoc", + "potential_utf", + "utf8_iter", + "yoke", + "zerofrom", + "zerovec", +] + +[[package]] +name = "icu_locale_core" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29" +dependencies = [ + "displaydoc", + "litemap", + "tinystr", + "writeable", + "zerovec", +] + +[[package]] +name = "icu_normalizer" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4" +dependencies = [ + "icu_collections", + "icu_normalizer_data", + "icu_properties", + "icu_provider", + "smallvec", + "zerovec", +] + +[[package]] +name = "icu_normalizer_data" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38" + +[[package]] +name = "icu_properties" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de" +dependencies = [ + "icu_collections", + "icu_locale_core", + "icu_properties_data", + "icu_provider", + "zerotrie", + "zerovec", +] + +[[package]] +name = "icu_properties_data" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14" + +[[package]] +name = "icu_provider" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421" +dependencies = [ + "displaydoc", + "icu_locale_core", + "writeable", + "yoke", + "zerofrom", + "zerotrie", + "zerovec", +] + +[[package]] +name = "idna" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" +dependencies = [ + "idna_adapter", + "smallvec", + "utf8_iter", +] + +[[package]] +name = "idna_adapter" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb68373c0d6620ef8105e855e7745e18b0d00d3bdb07fb532e434244cdb9a714" +dependencies = [ + "icu_normalizer", + "icu_properties", +] + +[[package]] +name = "ipnet" +version = "2.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "jray-server" +version = "0.1.0" +dependencies = [ + "anyhow", + "axum", + "http-body-util", + "rand 0.9.5", + "reqwest", + "rusqlite", + "serde", + "serde_json", + "sha2", + "thiserror", + "tokio", + "tower", + "tower-http", + "tracing", + "tracing-subscriber", + "ulid", + "unicode-general-category", + "unicode-normalization", +] + +[[package]] +name = "js-sys" +version = "0.3.103" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" +dependencies = [ + "cfg-if", + "futures-util", + "wasm-bindgen", +] + +[[package]] +name = "lazy_static" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" + +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + +[[package]] +name = "libsqlite3-sys" +version = "0.35.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "133c182a6a2c87864fe97778797e46c7e999672690dc9fa3ee8e241aa4a9c13f" +dependencies = [ + "cc", + "pkg-config", + "vcpkg", +] + +[[package]] +name = "litemap" +version = "0.8.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" + +[[package]] +name = "log" +version = "0.4.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" + +[[package]] +name = "lru-slab" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "112b39cec0b298b6c1999fee3e31427f74f676e4cb9879ed1a121b43661a4154" + +[[package]] +name = "matchers" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d1525a2a28c7f4fa0fc98bb91ae755d1e2d1505079e05539e35bc876b5d65ae9" +dependencies = [ + "regex-automata", +] + +[[package]] +name = "matchit" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "47e1ffaa40ddd1f3ed91f717a33c8c0ee23fff369e3aa8772b9605cc1d22f4c3" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "mime" +version = "0.3.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6877bb514081ee2a7ff5ef9de3281f14a4dd4bceac4c09388074a6b5df8a139a" + +[[package]] +name = "mio" +version = "1.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "30d65c71f1ce40ab09135ce117d742b9f8a19ff91a41a8b57ed50bc2de59c427" +dependencies = [ + "libc", + "wasi", + "windows-sys 0.61.2", +] + +[[package]] +name = "nu-ansi-term" +version = "0.50.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7957b9740744892f114936ab4a57b3f487491bbeafaf8083688b16841a4240e5" +dependencies = [ + "windows-sys 0.61.2", +] + +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "percent-encoding" +version = "2.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" + +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + +[[package]] +name = "pkg-config" +version = "0.3.33" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" + +[[package]] +name = "potential_utf" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564" +dependencies = [ + "zerovec", +] + +[[package]] +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quinn" +version = "0.11.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c1a41e437b6bbd489372cd4971de128e85c855f56c57f283d20ff016cf7c0a8" +dependencies = [ + "bytes", + "cfg_aliases", + "pin-project-lite", + "quinn-proto", + "quinn-udp", + "rustc-hash", + "rustls", + "socket2", + "thiserror", + "tokio", + "tracing", + "web-time", +] + +[[package]] +name = "quinn-proto" +version = "0.11.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560" +dependencies = [ + "bytes", + "getrandom 0.4.3", + "lru-slab", + "rand 0.10.2", + "rand_pcg", + "ring", + "rustc-hash", + "rustls", + "rustls-pki-types", + "slab", + "thiserror", + "tinyvec", + "tracing", + "web-time", +] + +[[package]] +name = "quinn-udp" +version = "0.5.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694" +dependencies = [ + "cfg_aliases", + "libc", + "once_cell", + "socket2", + "tracing", + "windows-sys 0.61.2", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "rand" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" +dependencies = [ + "rand_chacha", + "rand_core 0.9.5", +] + +[[package]] +name = "rand" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c7f5fa3a058cd35567ef9bfa5e75732bee0f9e4c55fa90477bef2dfcdbc4be80" +dependencies = [ + "chacha20", + "getrandom 0.4.3", + "rand_core 0.10.1", +] + +[[package]] +name = "rand_chacha" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" +dependencies = [ + "ppv-lite86", + "rand_core 0.9.5", +] + +[[package]] +name = "rand_core" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" +dependencies = [ + "getrandom 0.3.4", +] + +[[package]] +name = "rand_core" +version = "0.10.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69" + +[[package]] +name = "rand_pcg" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "caa0f4137e1c0a72f4c651489402276c8e8e1cf081f3b0ba156d2cbeef09e86a" +dependencies = [ + "rand_core 0.10.1", +] + +[[package]] +name = "regex-automata" +version = "0.4.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad" +dependencies = [ + "aho-corasick", + "memchr", + "regex-syntax", +] + +[[package]] +name = "regex-syntax" +version = "0.8.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" + +[[package]] +name = "reqwest" +version = "0.12.28" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" +dependencies = [ + "base64", + "bytes", + "futures-core", + "http", + "http-body", + "http-body-util", + "hyper", + "hyper-rustls", + "hyper-util", + "js-sys", + "log", + "percent-encoding", + "pin-project-lite", + "quinn", + "rustls", + "rustls-pki-types", + "serde", + "serde_json", + "serde_urlencoded", + "sync_wrapper", + "tokio", + "tokio-rustls", + "tower", + "tower-http", + "tower-service", + "url", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", + "webpki-roots", +] + +[[package]] +name = "ring" +version = "0.17.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7" +dependencies = [ + "cc", + "cfg-if", + "getrandom 0.2.17", + "libc", + "untrusted", + "windows-sys 0.52.0", +] + +[[package]] +name = "rusqlite" +version = "0.37.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "165ca6e57b20e1351573e3729b958bc62f0e48025386970b6e4d29e7a7e71f3f" +dependencies = [ + "bitflags", + "fallible-iterator", + "fallible-streaming-iterator", + "hashlink", + "libsqlite3-sys", + "smallvec", +] + +[[package]] +name = "rustc-hash" +version = "2.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" + +[[package]] +name = "rustls" +version = "0.23.43" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" +dependencies = [ + "once_cell", + "ring", + "rustls-pki-types", + "rustls-webpki", + "subtle", + "zeroize", +] + +[[package]] +name = "rustls-pki-types" +version = "1.15.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" +dependencies = [ + "web-time", + "zeroize", +] + +[[package]] +name = "rustls-webpki" +version = "0.103.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +dependencies = [ + "ring", + "rustls-pki-types", + "untrusted", +] + +[[package]] +name = "rustversion" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" + +[[package]] +name = "ryu" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "serde_path_to_error" +version = "0.1.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10a9ff822e371bb5403e391ecd83e182e0e77ba7f6fe0160b795797109d1b457" +dependencies = [ + "itoa", + "serde", + "serde_core", +] + +[[package]] +name = "serde_urlencoded" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd" +dependencies = [ + "form_urlencoded", + "itoa", + "ryu", + "serde", +] + +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures 0.2.17", + "digest", +] + +[[package]] +name = "sharded-slab" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f40ca3c46823713e0d4209592e8d6e826aa57e928f09752619fc696c499637f6" +dependencies = [ + "lazy_static", +] + +[[package]] +name = "shlex" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" + +[[package]] +name = "signal-hook-registry" +version = "1.4.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c4db69cba1110affc0e9f7bcd48bbf87b3f4fc7c61fc9155afd4c469eb3d6c1b" +dependencies = [ + "errno", + "libc", +] + +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + +[[package]] +name = "smallvec" +version = "1.15.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" + +[[package]] +name = "socket2" +version = "0.6.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" +dependencies = [ + "libc", + "windows-sys 0.61.2", +] + +[[package]] +name = "stable_deref_trait" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" + +[[package]] +name = "subtle" +version = "2.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" + +[[package]] +name = "syn" +version = "2.0.119" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "sync_wrapper" +version = "1.0.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0bf256ce5efdfa370213c1dabab5935a12e49f2c58d15e9eac2870d3b4f27263" +dependencies = [ + "futures-core", +] + +[[package]] +name = "synstructure" +version = "0.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "thiserror" +version = "2.0.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9" +dependencies = [ + "thiserror-impl", +] + +[[package]] +name = "thiserror-impl" +version = "2.0.19" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "thread_local" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ad99c4c6d32803332c548b1af0540b357b3f5fc0be8f6c6bfe8b2e6ae784070" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "tinystr" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d" +dependencies = [ + "displaydoc", + "zerovec", +] + +[[package]] +name = "tinyvec" +version = "1.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" +dependencies = [ + "tinyvec_macros", +] + +[[package]] +name = "tinyvec_macros" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" + +[[package]] +name = "tokio" +version = "1.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" +dependencies = [ + "bytes", + "libc", + "mio", + "pin-project-lite", + "signal-hook-registry", + "socket2", + "tokio-macros", + "windows-sys 0.61.2", +] + +[[package]] +name = "tokio-macros" +version = "2.7.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.3", +] + +[[package]] +name = "tokio-rustls" +version = "0.26.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1729aa945f29d91ba541258c8df89027d5792d85a8841fb65e8bf0f4ede4ef61" +dependencies = [ + "rustls", + "tokio", +] + +[[package]] +name = "tower" +version = "0.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4" +dependencies = [ + "futures-core", + "futures-util", + "pin-project-lite", + "sync_wrapper", + "tokio", + "tower-layer", + "tower-service", + "tracing", +] + +[[package]] +name = "tower-http" +version = "0.6.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" +dependencies = [ + "bitflags", + "bytes", + "futures-util", + "http", + "http-body", + "http-body-util", + "pin-project-lite", + "tokio", + "tower", + "tower-layer", + "tower-service", + "tracing", + "url", +] + +[[package]] +name = "tower-layer" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "121c2a6cda46980bb0fcd1647ffaf6cd3fc79a013de288782836f6df9c48780e" + +[[package]] +name = "tower-service" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3" + +[[package]] +name = "tracing" +version = "0.1.44" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" +dependencies = [ + "log", + "pin-project-lite", + "tracing-attributes", + "tracing-core", +] + +[[package]] +name = "tracing-attributes" +version = "0.1.31" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "tracing-core" +version = "0.1.36" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" +dependencies = [ + "once_cell", + "valuable", +] + +[[package]] +name = "tracing-log" +version = "0.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ee855f1f400bd0e5c02d150ae5de3840039a3f54b025156404e34c23c03f47c3" +dependencies = [ + "log", + "once_cell", + "tracing-core", +] + +[[package]] +name = "tracing-subscriber" +version = "0.3.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb7f578e5945fb242538965c2d0b04418d38ec25c79d160cd279bf0731c8d319" +dependencies = [ + "matchers", + "nu-ansi-term", + "once_cell", + "regex-automata", + "sharded-slab", + "smallvec", + "thread_local", + "tracing", + "tracing-core", + "tracing-log", +] + +[[package]] +name = "try-lock" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" + +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + +[[package]] +name = "ulid" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "470dbf6591da1b39d43c14523b2b469c86879a53e8b758c8e090a470fe7b1fbe" +dependencies = [ + "rand 0.9.5", + "web-time", +] + +[[package]] +name = "unicode-general-category" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b993bddc193ae5bd0d623b49ec06ac3e9312875fdae725a975c51db1cc1677f" + +[[package]] +name = "unicode-ident" +version = "1.0.24" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" + +[[package]] +name = "unicode-normalization" +version = "0.1.25" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8" +dependencies = [ + "tinyvec", +] + +[[package]] +name = "untrusted" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" + +[[package]] +name = "url" +version = "2.5.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" +dependencies = [ + "form_urlencoded", + "idna", + "percent-encoding", + "serde", +] + +[[package]] +name = "utf8_iter" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" + +[[package]] +name = "valuable" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ba73ea9cf16a25df0c8caa16c51acb937d5712a8429db78a3ee29d5dcacd3a65" + +[[package]] +name = "vcpkg" +version = "0.2.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "want" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bfa7760aed19e106de2c7c0b581b509f2f25d3dacaf737cb82ac61bc6d760b0e" +dependencies = [ + "try-lock", +] + +[[package]] +name = "wasi" +version = "0.11.1+wasi-snapshot-preview1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" + +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + +[[package]] +name = "wasm-bindgen" +version = "0.2.126" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" +dependencies = [ + "cfg-if", + "once_cell", + "rustversion", + "wasm-bindgen-macro", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-futures" +version = "0.4.76" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "wasm-bindgen-macro" +version = "0.2.126" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" +dependencies = [ + "quote", + "wasm-bindgen-macro-support", +] + +[[package]] +name = "wasm-bindgen-macro-support" +version = "0.2.126" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" +dependencies = [ + "bumpalo", + "proc-macro2", + "quote", + "syn 2.0.119", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-shared" +version = "0.2.126" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "web-sys" +version = "0.3.103" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "web-time" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb" +dependencies = [ + "js-sys", + "wasm-bindgen", +] + +[[package]] +name = "webpki-roots" +version = "1.0.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7dcd9d09a39985f5344844e66b0c530a33843579125f23e21e9f0f220850f22a" +dependencies = [ + "rustls-pki-types", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-sys" +version = "0.52.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" +dependencies = [ + "windows-targets", +] + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + +[[package]] +name = "windows-targets" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" +dependencies = [ + "windows_aarch64_gnullvm", + "windows_aarch64_msvc", + "windows_i686_gnu", + "windows_i686_gnullvm", + "windows_i686_msvc", + "windows_x86_64_gnu", + "windows_x86_64_gnullvm", + "windows_x86_64_msvc", +] + +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" + +[[package]] +name = "windows_aarch64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" + +[[package]] +name = "windows_i686_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" + +[[package]] +name = "windows_i686_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" + +[[package]] +name = "windows_i686_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" + +[[package]] +name = "windows_x86_64_gnu" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" + +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" + +[[package]] +name = "windows_x86_64_msvc" +version = "0.52.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" + +[[package]] +name = "writeable" +version = "0.6.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" + +[[package]] +name = "yoke" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" +dependencies = [ + "stable_deref_trait", + "yoke-derive", + "zerofrom", +] + +[[package]] +name = "yoke-derive" +version = "0.8.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", + "synstructure", +] + +[[package]] +name = "zerocopy" +version = "0.8.55" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b5a105cd7b140f6eeec8acff2ea38135d3cab283ada58540f629fe51e46696eb" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.55" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0fe976fb70c78cd64cccfe3a6fc142244e8a77b70959b30faf9d0ac37ee228eb" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zerofrom" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" +dependencies = [ + "zerofrom-derive", +] + +[[package]] +name = "zerofrom-derive" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", + "synstructure", +] + +[[package]] +name = "zeroize" +version = "1.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" + +[[package]] +name = "zerotrie" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf" +dependencies = [ + "displaydoc", + "yoke", + "zerofrom", +] + +[[package]] +name = "zerovec" +version = "0.11.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239" +dependencies = [ + "yoke", + "zerofrom", + "zerovec-derive", +] + +[[package]] +name = "zerovec-derive" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/Cargo.toml b/Cargo.toml new file mode 100644 index 0000000..90afaa7 --- /dev/null +++ b/Cargo.toml @@ -0,0 +1,30 @@ +[package] +name = "jray-server" +version = "0.1.0" +edition = "2021" +rust-version = "1.85" +license = "GPL-3.0-or-later" +description = "JRay public server — community manifest exchange" + +[dependencies] +axum = { version = "0.8", features = ["json", "query"] } +tokio = { version = "1", features = ["rt-multi-thread", "macros", "signal", "sync", "time"] } +tower = "0.5" +tower-http = { version = "0.6", features = ["trace", "timeout", "limit"] } +serde = { version = "1", features = ["derive"] } +serde_json = "1" +rusqlite = { version = "0.37", features = ["bundled"] } +reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls"] } +tracing = "0.1" +tracing-subscriber = { version = "0.3", features = ["env-filter"] } +unicode-normalization = "0.1" +unicode-general-category = "1" +sha2 = "0.10" +rand = "0.9" +ulid = "1" +thiserror = "2" +anyhow = "1" + +[dev-dependencies] +tower = { version = "0.5", features = ["util"] } +http-body-util = "0.1" diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..f479362 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,89 @@ +# Deploy image for jray-server. +# +# §8 is explicit that neither Docker nor Compose should be *required* — the +# primary artifact is a single static binary plus one database file, and that is +# deliberately the lowest-friction thing a hobbyist operator can deploy. This +# image is the optional convenience, not the intended path. +# +# It builds against musl so the runtime stage can be `scratch`: no libc, no shell, +# no package manager, nothing to keep patched. rusqlite is built with `bundled` +# (SQLite compiled in) and reqwest with rustls rather than OpenSSL, so there is +# genuinely nothing left to link against. +# +# docker build -t jray-server . +# docker run --rm -p 8080:8080 -v jray-data:/data \ +# -e JRAY_TMDB_API_KEY=... jray-server + +FROM rust:1.92-alpine AS builder + +# `musl-dev` for the C toolchain rusqlite's bundled SQLite needs; `file` for the +# static-linkage assertion below. +RUN apk add --no-cache musl-dev file + +WORKDIR /build + +# Dependencies first, in their own layer, so editing source does not re-download +# and rebuild the entire tree. +COPY Cargo.toml Cargo.lock ./ +RUN mkdir -p src \ + && echo 'fn main() {}' > src/main.rs \ + && echo '' > src/lib.rs \ + && cargo build --release --target x86_64-unknown-linux-musl \ + && rm -rf src + +COPY src ./src + +# `touch` defeats the cargo staleness check that the dummy-source trick above +# would otherwise leave in place. +RUN touch src/main.rs src/lib.rs \ + && cargo build --release --target x86_64-unknown-linux-musl \ + && strip target/x86_64-unknown-linux-musl/release/jray-server + +# Verify the result is genuinely static. A dynamically-linked binary would fail at +# runtime on `scratch`, and failing here is far easier to diagnose. +# +# Asserted with `file`, not `ldd`: the musl target produces a **static-PIE**, and +# `ldd` prints the musl loader path for one, so an `ldd`-based check reports a +# static binary as dynamic. `file` reports "static-pie linked" and is unambiguous. +RUN file target/x86_64-unknown-linux-musl/release/jray-server | tee /tmp/linkage \ + && grep -qE 'static-pie linked|statically linked' /tmp/linkage \ + || (echo "binary is not statically linked; it will not run on scratch" && exit 1) + +# Stage the data directory with the runtime uid's ownership. `scratch` has no +# shell, so this cannot be done in the final stage — and a bare `VOLUME` there +# would be created root-owned, leaving the non-root process unable to create the +# database at all. +RUN mkdir -p /staged-data && chown 65534:65534 /staged-data + +# --------------------------------------------------------------------------- + +FROM scratch + +COPY --from=builder /build/target/x86_64-unknown-linux-musl/release/jray-server /jray-server + +# The database lives on a volume; §8 warns that a plain file copy of a live WAL +# database is not a valid backup, so back it up with `VACUUM INTO` from the host +# rather than by archiving this directory. +# +# Copied from the builder so it arrives owned by the runtime uid. Docker seeds a +# named volume from the image's directory, ownership included, so the server can +# create the database on first run. A bind mount is *not* seeded this way — the +# host directory keeps its own ownership, so it must be made writable by uid +# 65534 (`chown 65534:65534 /path/on/host`). +COPY --from=builder --chown=65534:65534 /staged-data /data +VOLUME ["/data"] + +# Non-root. `scratch` has no /etc/passwd, so this is a bare uid — which is all the +# kernel needs, and the binary touches nothing outside /data. +USER 65534:65534 + +ENV JRAY_BIND=0.0.0.0:8080 \ + JRAY_DB=/data/jray.db + +EXPOSE 8080 + +# No HEALTHCHECK: it would need a shell or curl, and `scratch` has neither. +# §4 provides `GET /health` (liveness) and `/ready` (database and migrations) for +# an orchestrator to probe externally, which is the right place for it. + +ENTRYPOINT ["/jray-server"] diff --git a/README.md b/README.md new file mode 100644 index 0000000..bd277ce --- /dev/null +++ b/README.md @@ -0,0 +1,220 @@ +# JRay public server + +A community manifest exchange for JRay. Jellyfin servers running the JRay plugin +pull actor-timeline manifests ("Jmanifests") for titles they own instead of +running the CV pipeline locally, and optionally contribute the manifests they +generate back. + +See [SPEC.md](SPEC.md) for the design. Section references throughout the code +point at it. + +The community instance is **`https://jray.tourolle.paris`**. The JRay plugin +ships with it pre-configured but **disabled** — §9 requires that no traffic leave +an installation until an admin opts in, so the default entry exists to save the +admin from typing a URL, not to enable sharing on their behalf. Set +`JRAY_SERVER_ID=jray.tourolle.paris` when deploying that instance: it becomes the +`origin` stamped on manifests it first accepts (§9a) and the salt for report IP +hashes. + +## Status + +First implementation pass: the **core vertical slice**, reconciled against the +[system spec](../SPEC.md). + +Per-requirement status is in [`docs/requirements.md`](docs/requirements.md) — +32 requirements (`UR-001..018`, `DR-001..014`), each tracing up to an `SR-nnn` +or `PR-nnn`. `UR-015..018` are the pending SR-003 schema bump and are marked +`Planned` rather than omitted. + +Implemented: + +- §2 Jmanifest format and series bundles +- §3 cut matching — `exact` / `runtime` / `loose` tiers +- §4 the API surface, less `POST /manifests/search` +- §5 rate limiting, in-process fixed-window counters +- §5a trust model — anonymous bearer tokens, closed-vocabulary storage, + automatic revocation +- §6 upload validation, all four stages +- §7 relational storage, no JSON blobs on the write path +- §8 Rust + Axum + SQLite, single serialized writer, in-process job queue +- §9a content addressing (`content_id`), computed on upload + +Reconciled with the system spec (see `docs/requirements.md` for the detail): + +- **`anneal_sec` removed.** Withdrawn upstream by AR-012/AR-013 — presence now + follows track extent, so a track survives its own gaps and there is nothing to + anneal. Its successor `extinction_sec` and the new `gallery_scope` are + accepted, stored, and (for scope) used in §7 ranking. A manifest still + carrying `anneal_sec` is now a hard `400`, not silently ignored: it was + produced by a pipeline whose window semantics differ from what this server + assumes. +- **Audio signature: the 120 s rule now matches both producers.** An earlier + draft of §3 allowed a shortened window for items under 150 s; that conflicted + with `scene-actor-extraction` IR-007 and was the weaker rule, since a + caller-varying length is the property SR-004 forbids. Items under 120 s now + send no signature at all. + +Deferred: + +- §3 audio signatures — the field is **accepted, validated and stored**, and + `content_id` already excludes it, but `audio`-tier matching and + `POST /manifests/search` are not wired up. This follows §3's own recommended + sequencing: ship the plugin-side computation first, let signatures accumulate, + then enable matching once coverage is useful. +- §9a federation endpoints (`/federation/*`) and the pull worker. The schema + columns (`content_id`, `origin`, `ingested_from`, `peers`) are in place, and + `ingest::persist` is already the shared path a pull would reuse. + +## Running + +```sh +cargo run +``` + +Configuration is entirely environment variables: + +| Variable | Default | Purpose | +|---|---|---| +| `JRAY_BIND` | `127.0.0.1:8080` | Listen address. Terminate TLS at the operator's proxy (§8) | +| `JRAY_DB` | `jray.db` | SQLite path. WAL mode, created on first run | +| `JRAY_TMDB_API_KEY` | — | **Hard dependency for UR-3.** Without it, uploads stay `pending` and are never listed | +| `JRAY_TMDB_BASE_URL` | `https://api.themoviedb.org/3` | Override for testing | +| `JRAY_TRUSTED_PROXIES` | — | Comma-separated proxy IPs whose `X-Forwarded-For` is honoured. **Not default-on**: §5 rate limiting and report attribution key on client IP, so a spoofable header defeats both | +| `JRAY_SERVER_ID` | `localhost` | This server's identity, used as manifest `origin` and as the report IP-hash salt | +| `JRAY_REQUEST_TIMEOUT_SEC` | `30` | Request timeout so a slow bundle query fails fast | +| `JRAY_JOB_BATCH` | `8` | Cast-check jobs leased per worker tick | +| `JRAY_JOB_POLL_SEC` | `5` | Worker poll interval | +| `JRAY_LOG` | `info` | `tracing` filter | + +Contributing requires a token (§5a) — an anonymous bearer capability, not an +account. Self-issue one: + +```sh +curl -sX POST -H 'content-type: application/json' -d '{}' \ + http://127.0.0.1:8080/api/v1/tokens +``` + +## Deployment + +§8's deployment notes are requirements, not suggestions: + +- Enforce the body cap at **both** the proxy and the app. `client_max_body_size` + (nginx) / `request_body max_size` (Caddy) should match §6 stage 1, so oversized + uploads are dropped at the edge and never occupy an application worker. The app + must also be safe when run without a proxy, which it is. +- Set `JRAY_TRUSTED_PROXIES` to the proxy's address, or `X-Forwarded-For` is + ignored and every client behind it shares one rate-limit bucket. +- **Back up the SQLite file with `VACUUM INTO` or the backup API** — never a + plain file copy of a live WAL database. Manifests represent real CV compute. + +## Tests + +```sh +cargo test # 178 tests +cargo deny check # advisories, licences, bans, sources +``` + +Unit tests per module, plus two integration suites: + +- `tests/api.rs` — end-to-end through the real router: status codes, headers, and + the properties that only hold if the layers compose correctly (per-route body + caps, rate-limit surfaces, the strict schema actually reaching uploads). +- `tests/injection.rs` — that hostile input cannot escape its layer: SQL payloads + in query parameters, path segments, JSON bodies, bearer tokens and report notes; + JSON structure abuse; CRLF header injection; path traversal; and Unicode tricks + against the §5a character class. + +Security-relevant properties are asserted rather than assumed — +`movie`/`jellyfin_id` rejection, the §5a character class defeating base64/hex +smuggling, per-route body caps, a lying `Content-Length` not bypassing the cap, +and a forged `X-Forwarded-For` not resetting a rate-limit budget. + +**On injection specifically.** Two independent defences apply, and they fail +differently, so both are tested: + +1. **Parameterised queries.** Every value reaches SQLite through `params![]`. The + only `format!`-built SQL interpolates two compile-time constants (a column list + and a status literal) — no runtime input ever becomes SQL syntax. This is what + actually prevents injection. +2. **Closed-vocabulary validation.** Identifiers are regex-constrained and free + text is limited to a closed character class, so most payloads never reach the + query layer at all. + +The injection suite would still pass on defence 1 alone, which is deliberate: if +validation were ever loosened, the tests should not silently start depending on it. + +Writing that suite found two real gaps, both since fixed: + +- Compatibility homoglyphs (`𝐒𝐭𝐞𝐯𝐞`, `Actor`) passed the §5a class. They are + letters by Unicode category and NFC does not fold them — only NFKC would. Beyond + name spoofing, a fullwidth-digit alphabet would have reopened the encoding + channel the "no digits" rule exists to close. +- A one-frame `audio_signature` was accepted on a feature-length manifest, making + the field the variable-length container §3 explicitly forbids. The length floor + is now derived from the declared runtime, keeping §3's genuine short-item + exception without trusting the client's length. + +Two tests worth knowing about: + +- `content_id::tests::golden_vector_hash_is_stable` locks the §9a canonical form. + **The JRay plugin must reproduce it byte-identically**; §8 notes the extraction + side is Python, so this can no longer be one shared implementation and must be + cross-tested instead. `GOLDEN_VECTORS` is that fixture, and its hash was + verified against an independent Python implementation. +- The integration tests use an on-disk temporary database, not `:memory:`, + because §8's topology is one writer connection plus a read pool — in-memory + SQLite is per-connection, so the readers would see an empty database. + +## CI and container image + +`.gitea/workflows/ci.yml` runs on Gitea Actions (and unmodified on GitHub +Actions), gated fastest-first: `fmt` → `clippy` → `test` → `cargo deny` → static +musl build. The dependency audit also runs weekly, since advisories appear without +any code changing. + +The `Dockerfile` is the optional convenience, not the intended deployment path — +§8 is explicit that neither Docker nor Compose should be *required*. It builds +against musl and runs from `scratch` as uid 65534: **8.7 MB**, no libc, no shell, +no package manager. `rusqlite` bundles SQLite and `reqwest` uses rustls, so there +is nothing left to link. + +```sh +docker build -t jray-server . +docker run --rm -p 8080:8080 -v jray-data:/data -e JRAY_TMDB_API_KEY=... jray-server +``` + +Two things that are easy to get wrong, so they are handled explicitly: + +- **Static linkage is asserted with `file`, not `ldd`.** The musl target produces + a *static-PIE*, and `ldd` prints the musl loader for one — an `ldd`-based check + reports a perfectly static binary as dynamic. +- **`/data` is staged with the runtime uid's ownership.** Docker seeds a named + volume from the image directory, ownership included, so a non-root server can + create the database on first run. A **bind mount is not seeded this way** — the + host directory keeps its own ownership, so `chown 65534:65534` it first or the + server exits with "unable to open database file". + +Two tests are worth knowing about: + +- `content_id::tests::golden_vector_hash_is_stable` locks the §9a canonical form. + **The JRay plugin must reproduce it byte-identically**; §8 notes the extraction + side is Python, so this can no longer be one shared implementation and must be + cross-tested instead. `GOLDEN_VECTORS` is that fixture, and its hash was + verified against an independent Python implementation. +- The integration tests use an on-disk temporary database, not `:memory:`, + because §8's topology is one writer connection plus a read pool — in-memory + SQLite is per-connection, so the readers would see an empty database. + +## Known gaps + +- **The §6 stage-3 thresholds are still §10's guesses.** §10 (5) is explicit that + running the check over the 331 real corpus files would give the true + distribution of honest-upload match ratios, and is "the single cheapest way to + de-risk UR-3 and UR-5". The scoring logic is deliberately pure functions in + `castcheck.rs` so that retuning is a test-data exercise, not a code change. +- `POST /manifests/{id}/report` records reports but nothing consumes them yet. + §5a's divergence detection and the operator kill switch are not implemented; + delisting is currently a manual `UPDATE`, which §5a does note is the intended + shape ("one UPDATE", not a moderation queue). +- No admin surface. Revocation is automatic (§5a), but an operator has no + endpoint for the deliberate kill-switch case. diff --git a/SPEC.md b/SPEC.md new file mode 100644 index 0000000..939b12d --- /dev/null +++ b/SPEC.md @@ -0,0 +1,1839 @@ +# JRay Public Server — specification + +A community manifest exchange for JRay. Jellyfin servers running the JRay +plugin pull actor-timeline manifests ("Jmanifests") for titles they own +instead of running the CV pipeline locally, and optionally contribute the +manifests they generate back. + +Status: **core implemented.** See [`docs/requirements.md`](docs/requirements.md) +for per-requirement status and [`README.md`](README.md) for what is deferred. + +This is a *software* spec: its job is to implement the +[system spec](../SPEC.md), which owns everything spanning more than one repo. +Requirements here trace up to an `SR-nnn`; the prose below is the detail. + +--- + +## 0. Requirements + +IDs are `UR-nnn`, zero-padded and **permanent** — a withdrawn requirement keeps +its number, because renumbering is what produces orphan TRACES tags +([system spec](../SPEC.md) §6). The authoritative list with status lives in +[`docs/requirements.md`](docs/requirements.md); this table is the prose anchor. + +| # | Requirement | Traces to | Where addressed | +|---|---|---|---| +| UR-001 | Query whether a JRay manifest for a given media item exists on the server | SR-001 | §4 `GET /manifests/exists` | +| UR-002 | Route to post a JRay manifest for a media item | PR-006 | §4 `POST /manifests` | +| UR-003 | Content verification: no additional JSON fields, file size limit, approximate cast match against TMDB | SR-004 | §6 | +| UR-004 | Rate limiting on queries | SR-004 | §5 | +| UR-005 | Trust without account management: server must not be usable as a content store, nor for prank/vandalism manifests | SR-004 | §5a | +| UR-006 | Serve and accept a whole series in one operation, not episode-by-episode | PR-006 | §2 series bundles, §4 `GET /manifests/series`, `POST /manifests/bundle` | +| UR-007 | JRay plugin must query a configurable list of servers | PR-005 | §9 | +| UR-008 | Servers must be able to sync/replicate manifests between each other | PR-006 | §9a | +| UR-009 | Store an audio spectral-peak signature from the media centre, so a file of unknown providence can be identified and synchronised | SR-003 | §3 audio signature | +| UR-010 | Identity crossing the API boundary is TMDB/IMDB ids, never a name alone | SR-001 | §2, §6 stage 3, §7 | +| UR-011 | Reject any manifest field capable of carrying binary or attacker-chosen content | SR-004 | §5a Threat 1, §6 stage 2 | +| UR-012 | Never accept, store, or serve gallery data — reference faces or embeddings | SR-005 | §5a, and the absence of any such field in §2 | +| UR-013 | Windows are scene-scoped claims; the server must not reinterpret their boundaries | SR-002 | §2 field notes, §6 | +| UR-014 | Reject a manifest whose `schema_version` / `jmanifest_version` is unknown, never guess | SR-003 | §2, §6 stage 2 | + +Two notes on UR-001. An existence check is deliberately a *separate, cheaper* +endpoint from the fetch in §4 — it answers "should I bother?" for a whole +library sweep without transferring payloads, and it is the endpoint a +scheduled task will hammer. It is also the most abuse-prone surface, since +it doubles as an oracle for "does the community have this title" — so it is +rate-limited harder than the fetches and returns no manifest content. + +UR-003's three checks are different in kind and are enforced at different +stages: field strictness and size are cheap and synchronous (reject at the +door), whereas the TMDB cast match needs an outbound API call and so runs +asynchronously after a `202`. See §6. + +**UR-010 to UR-014 were added when this spec was reconciled against the system +spec.** They are not new work — each states a property the design already had, +which had been left implicit because no system requirement existed to trace it +to. UR-012 and UR-013 are the two worth stating explicitly: the server's refusal +to carry gallery data (SR-005) and its refusal to reinterpret window boundaries +(SR-002) are both invariants preserved by *not* doing something, and an +unstated prohibition is the kind that erodes. + +--- + +## 1. Why this needs more than the current truth file + +The extraction pipeline (`result_sink_node`) writes: + +```json +{ + "schema_version": 1, + "movie": "/data/movies/Movie.mkv", + "sample_fps": 1, + "anneal_sec": 2, + "actors": [ + { "name": "...", "imdb_id": "...", "tmdb_id": "...", "jellyfin_id": "...", + "scenes": [[12.0, 45.0]] } + ] +} +``` + +Three properties of this format block sharing as-is: + +1. **No portable title identity.** `movie` is an absolute path on the machine + that ran extraction. Nothing in the file says "this is The Death of Stalin + (2017)". The receiving server cannot tell what it just downloaded. +2. **Installation-local identifiers.** `movie` leaks the contributor's + directory layout and `jellyfin_id` is a GUID from the contributor's + database — meaningless and mildly identifying elsewhere. Both must be + stripped on upload, not merely ignored on download. +3. **Timings are cut-specific.** `scenes` are absolute seconds. A theatrical + cut, an extended cut, a PAL speed-up, and a release with 40s of distributor + logos all produce different timelines for the same TMDB id. Keying purely + on TMDB id would silently serve misaligned overlays. + +The Jmanifest format below is the truth file plus a portable identity block +and a cut fingerprint; the actor timeline payload is unchanged. + +### Terminology + +- **Jmanifest** — one shareable actor timeline for one *cut* of one title. +- **Title identity** — what the work is (TMDB/IMDB id + episode coordinates). +- **Cut fingerprint** — which encode/edit the timings apply to (runtime, plus + optional stronger signals). + +--- + +## 2. Jmanifest format + +```json +{ + "jmanifest_version": 1, + "identity": { + "type": "movie", + "tmdb_id": "504172", + "imdb_id": "tt4686844", + "title": "The Death of Stalin", + "year": 2017 + }, + "cut": { + "runtime_sec": 6420.5, + "container_duration_sec": 6420.5, + "video_hash": "opensubtitles:8e245d9679d31e12", + "audio_signature": "v1:v7fA3k…" + }, + "extraction": { + "sample_fps": 5, + "extinction_sec": 12, + "gallery_size": 1820, + "gallery_scope": "global", + "pipeline_version": "scene-actor-extraction 0.4.1" + }, + "actors": [ + { + "name": "Steve Buscemi", + "imdb_id": "nm0000114", + "tmdb_id": "884", + "scenes": [[191.6, 209.2], [438.2, 465.6]] + } + ] +} +``` + +For an episode, `identity` is: + +```json +{ + "type": "episode", + "series_tmdb_id": "1396", + "series_imdb_id": "tt0903747", + "title": "Breaking Bad", + "season": 2, + "episode": 5 +} +``` + +Field notes: + +- `jmanifest_version` — separate from the plugin's `schema_version`; this + versions the *exchange* envelope. +- `identity.tmdb_id` / `imdb_id` — at least one required. These are the + lookup keys. +- `cut.runtime_sec` — **required**, the decoded duration of the media the + timings came from. This is the primary alignment guard. +- `cut.video_hash` — optional but strongly preferred. See §3. +- `cut.audio_signature` — optional; a spectral-peak signature from the media + centre, version-prefixed (`v1:`). Enables content-based matching and offset + recovery for files of unknown providence. See §3. +- `extraction.extinction_sec` — the re-acquisition timeout that shapes window + extent. Replaces `anneal_sec`; see the schema-bump note below. +- `extraction.gallery_scope` — `global` or `limited`. The strongest available + quality signal when ranking competing manifests for one cut (§7), since a + gallery built from the whole library competes against every actor in it, + whereas a per-title gallery does not. +- `actors[].jellyfin_id` — **must not appear.** The server rejects uploads + containing it (see §6). +- `movie` (absolute path) — **must not appear.** Rejected likewise. +- `actors[].scenes` — `[start_sec, end_sec]` inclusive, sorted. **A window is a + claim about scene membership, not a recognition event** (UR-013, system spec + SR-002): an actor who turns away or is off-camera during a reverse shot is + still present. The server therefore never reinterprets, merges, splits or + trims windows — it stores and serves what it was given, quantised (§9a) but + not reshaped. Two windows mean a genuine departure and return. +- `actors[].tmdb_id` — the **primary** actor join key. In practice the + extraction pipeline populates this and leaves `imdb_id` empty (see §6 + stage 3), so a manifest without actor TMDB ids will match poorly. +- `actors[].name` — sent on upload for matching, but **not persisted**: the + server resolves each actor to a TMDB person id and serves names from its own + TMDB-derived table (§5a, §7). On download, `name` is present and + server-authoritative. Contributors should not expect a name they invented to + round-trip. + +### Pending schema bump — SR-003 + +The truth file and the Jmanifest are consumed by components that ship +independently, so breaking changes are **batched into one `schema_version` +bump** coordinated across all three repos (system spec SR-003). One bump is +currently pending, and this server must accept the new shape when it lands: + +| Change | Effect here | +|---|---| +| **Remove `anneal_sec`** | Withdrawn upstream: presence now follows track extent, so a track survives its own gaps and there is nothing to anneal. Deleted rather than kept as a vestigial `0` — a field naming a mechanism the pipeline no longer has is actively misleading | +| **Add `extinction_sec`** | Its successor: the parameter that actually shapes window extent | +| **Add `gallery_scope`** | New ranking signal (§7) | +| **Per-window belief** | `scenes` becomes a list of objects — interval plus posterior and identification route — rather than a list of float pairs | +| **Add audio signature** | Already specified here (§3, UR-009) | + +**Per-window belief does not weaken §5a.** The added fields are a bounded float +and a small enumerated string, so an accepted manifest still contains only +numbers and closed-vocabulary values. No free-form channel is opened, and +SR-004 is preserved. + +**Two consequences for `content_id`** (§9a), both of which must land with the +bump rather than after it: + +- The canonical form currently hashes `[start_cs, end_cs]` pairs. Once windows + carry belief, the canonical form must decide whether belief is part of + *identity*. It should **not** be: two servers that validated the same upload + must agree, and belief is a producer-side estimate that may legitimately + differ between pipeline versions for identical timings. Belief is replicated + as an attribute, exactly as `audio_signature` is (§9a). +- Quantisation is unchanged: integer centiseconds, for the reasons in §9a. + +Until the bump ships, this server accepts `jmanifest_version: 1` and rejects +anything else outright (UR-014) rather than guessing at an unknown shape. + +### Series bundles + +Series are the primary unit of exchange, not episodes. A user asks for +"Breaking Bad", not for 62 individual files, and per-episode round trips would +mean 62 requests against the rate limit for one obvious intent. + +A bundle is a thin wrapper, not a new format: + +```json +{ + "jmanifest_version": 1, + "series": { + "series_tmdb_id": "1396", + "series_imdb_id": "tt0903747", + "title": "Breaking Bad" + }, + "episodes": [ { "...a full Jmanifest, identity.type == episode..." } ] +} +``` + +**Sizing.** Measured against the 316 non-empty manifests in the extraction +corpus, a minimal (actors-only) manifest is ~4.8 KiB median and ~9.7 KiB at +p95. So a 24-episode season is ~112 KiB median / ~228 KiB p95, and even a +62-episode series is well under 1 MiB. Whole-series transfer is therefore the +sensible default rather than something to paginate defensively — the response +is smaller than a single poster image. + +Bundles are capped at 500 episodes and 25 MiB; beyond that the client must +page by season. + +**Partial bundles are normal.** The server returns whatever episodes it holds. +A bundle with 9 of 13 episodes is a valid, useful response, not an error. Each +episode carries its own `cut` block, so the client matches each one +independently — one mismatched episode does not invalidate the rest. The +bundle response includes coverage metadata so the client can report it: + +```json +{ + "coverage": { "episodes_available": 9, "seasons": [1, 2] } +} +``` + +**Bundles are a transfer convenience, not a storage unit.** Each episode +manifest is stored, validated, versioned, reported and delisted individually +(§7). There is no "series manifest" row — a bundle is assembled per request. +This matters for moderation: one bad episode is delisted on its own without +disturbing the other 61. + +### Series bundle upload + +Contributing a whole series is the natural counterpart, and it is the more +important half — a worker that has just processed a season should not make 24 +separate `POST`s, each triggering its own TMDB round trip. + +`POST /manifests/bundle` takes the same envelope. Semantics: + +- **Per-episode validation.** Each episode runs the full §6 pipeline + independently. The bundle is *not* atomic: valid episodes are accepted and + invalid ones rejected, with a per-episode result list. All-or-nothing would + let one bad episode discard an entire season's compute. +- **Shared TMDB fetch.** All episodes of a series resolve against one cached + credits fetch (§6 stage 3), so a 24-episode bundle costs one upstream call + rather than 24. This is the main reason bundle upload exists. +- **One rate-limit unit.** A bundle counts as a single write against the §5 + limit, with a separate per-episode cap, so contributing a season is not + punished relative to contributing a film. + +Response is `202` with per-episode outcomes: + +```json +{ + "results": [ + { "season": 1, "episode": 1, "manifest_id": "01HZ...", "status": "pending" }, + { "season": 1, "episode": 2, "status": "rejected", "reason": "cast_match_below_threshold" } + ] +} +``` + +--- + +## 3. Cut matching + +Timings only transfer between identical cuts. Matching is tiered, and the +server reports which tier matched so the client can decide whether to trust it. + +| Tier | Signal | Confidence | +|---|---|---| +| `exact` | `video_hash` equal | Same file, timings are exact | +| `runtime` | runtimes within ±2s | Very likely the same cut | +| `loose` | runtimes within ±30s | Probably same cut, different trims | +| — | beyond that | No match; do not serve | + +`video_hash` uses the OpenSubtitles hash (first+last 64KiB plus file size) — +cheap to compute, no full read, and already well-known in the media-server +ecosystem. It identifies a *file*, so it only ever matches an identical +release; it can never produce a false positive, which is why it is tier one. + +The client sends its own runtime and hash when requesting; the server does the +matching and returns the best available tier. A `loose` match should surface +as a caveat in the JRay UI rather than being applied silently. + +### Audio signature — UR-009 + +Everything above depends on knowing *what the file is*. When providence is +unknown — no TMDB id, no usable metadata, a renamed or badly-tagged file — none +of those tiers can fire. And when a release is trimmed differently (distributor +logos, PAL speed-up, an extra recap), the runtime tiers correctly *decline* to +match, but the underlying timings would have been reusable if only the offset +were known. + +A content-derived audio signature solves both. It is stored on every manifest +as `cut.audio_signature`. + +**Why audio and not video.** Audio survives what breaks video hashing: +re-encoding, resolution changes, bitrate changes, colour-space conversion, +letterboxing. Two releases of the same cut nearly always share an +audio track that is perceptually identical even when every video byte differs. + +#### Construction + +Sampled from the **centre of the media**, which avoids the two regions that +differ most between releases — logos and cold opens at the head, credits at the +tail. + +1. Decode a **120 s window centred on the midpoint** + (`runtime/2 - 60s` to `runtime/2 + 60s`). +2. Downmix to mono, resample to **11025 Hz**. +3. STFT with a **4096-sample frame, 1024-sample hop** (~93 ms/frame, + ~1290 frames), Hann window. +4. Per frame, take the log-magnitude spectrum over **300–3000 Hz** — the band + carrying dialogue and score, and the most codec-robust. +5. Divide that band into **32 logarithmically spaced bins** and record the + index of the **peak bin** plus a coarse 2-bit energy class. +6. Pack each frame into one byte; the signature is the resulting + **~1290-byte array**, base64-encoded. + +The result is ~1.7 KB per manifest — negligible against a ~4.8 KiB manifest. + +This is deliberately a **peak-bin** signature rather than a full spectrum: +peaks survive lossy re-encoding, loudness normalisation and channel-layout +differences, whereas absolute magnitudes do not. It follows the same principle +as Chromaprint/AcoustID (compact per-frame spectral features, matched by +sliding alignment) but is self-contained: no external service is queried, so +no lookup leaks which titles an instance holds (§9 privacy). + +#### Matching and offset recovery + +Two signatures are compared by sliding one against the other and taking the +best score: + +``` +for offset in -600 .. +600 frames: # ±56 s + score(offset) = fraction of overlapping frames whose peak bin matches +best = argmax score +``` + +| Result | Interpretation | +|---|---| +| `score ≥ 0.85`, `offset ≈ 0` | Same cut, aligned. Timings apply directly | +| `score ≥ 0.85`, `offset ≠ 0` | **Same cut, shifted.** Timings apply with `offset` added | +| `0.60 ≤ score < 0.85` | Possibly same cut, degraded audio. Flag as `loose` | +| `score < 0.60` | Different content. No match | + +The second row is the valuable one, and the reason to do this at all: a +release with 40 s of extra logos previously failed the ±2 s runtime tier +outright. Now it matches, and the client shifts every scene window by the +recovered offset. **The server returns the offset; the client applies it** — +manifests are never rewritten, so one stored manifest serves every trim of the +same cut. + +Offset search is capped at ±56 s, which covers realistic trim differences. +Speed-differing releases (PAL 4% speed-up) are **not** handled by a constant +offset and are correctly rejected by the score threshold; a scale-and-offset +search is possible later but is out of scope. + +#### Revised tier table + +| Tier | Signal | Confidence | +|---|---|---| +| `exact` | `video_hash` equal | Same file | +| `audio` | audio score ≥ 0.85 | Same cut; `offset` returned, may be non-zero | +| `runtime` | runtimes within ±2s | Very likely the same cut | +| `loose` | audio 0.60–0.85, or runtimes within ±30s | Caveat in UI | + +`audio` ranks above `runtime` because it is content-derived: it confirms the +audio actually matches, where equal runtimes are only circumstantial. + +#### Unknown-providence search + +With no TMDB id at all, a client can search by signature alone: + +``` +POST /manifests/search +{ "audio_signature": "base64…", "runtime_sec": 6420.5 } +``` + +The server returns candidate matches with scores, offsets and title identity — +letting JRay identify an unidentified file *and* align to it in one step. + +**This endpoint is a scaling problem, not a correctness one.** A naive +implementation compares against every stored signature. Mitigations: + +- Prefilter by runtime (±90 s) before scoring, which eliminates almost + everything. +- Index a **coarse hash** of the signature (e.g. the peak-bin sequence of + every 16th frame) for candidate generation, with full sliding comparison + only on candidates. +- Rate-limit hard (§5): this is the most expensive read endpoint and the most + attractive to abuse. + +Because it is expensive, `POST /manifests/search` is **optional for a server +to implement**; `GET /federation/capabilities` advertises support. + +#### Validation and abuse + +The signature is attacker-supplied, so §6 applies: + +- Fixed length (1290 frames ± a small tolerance for seek and encoder + differences at the window edges), base64, rejected otherwise. The length is + **not caller-varying**: items too short for the window emit no signature at + all (see "Media shorter than the window" above), so there is no legitimate + short signature to accommodate. A variable-length blob would be a payload + channel — precisely what §5a closes. +- Each byte is structurally constrained (5-bit bin index + 2-bit energy + class), so arbitrary bytes are invalid. This keeps §5a's "no free-form + storage" property intact: the field cannot carry meaningful smuggled data. +- Signatures are **never** used as a trust signal for cast validity — they + establish which cut a manifest describes, nothing more. + +#### Implementation cost — flagged honestly + +This is the most expensive addition in the spec, and it is worth being clear +where the work lands: + +- **Extraction pipeline (C++)** — **optional**, and best deferred. It already + links `libavformat`/`libavcodec`/`libavutil`, but `ffmpeg_decoder.hpp` is + **video-only**, so audio would need `libswresample` plus an FFT. Since the + plugin covers the whole library (below), this is redundant work. +- **JRay plugin (C#)** — **the primary implementation site**, and less costly + than it first appears. See below. +- **Server (Rust)** — comparison only, no audio decoding. `rustfft` plus a + sliding comparison; the cheapest of the three. + +Recommended sequencing: **make `audio_signature` optional**. Manifests without +one continue to work exactly as today via the existing tiers. Ship the +plugin-side computation first, let signatures accumulate, then enable +`audio`-tier matching and the search endpoint once coverage is useful. Nothing +above needs to land at once. + +#### Computing the signature in the JRay plugin + +The plugin is the right place for this, and it is the *only* place that covers +the whole use case. The extraction pipeline only ever sees files it processes; +the plugin sees **every item in the library**, including the ones with no truth +data and unknown providence — which is exactly the population UR-009 targets. A +signature must also be computable at *query* time (to identify a local file), +not only at contribution time. + +**Jellyfin already ships FFmpeg, and the plugin can reach it.** Verified +against `Jellyfin.Controller` 10.11.5, which the plugin already references: +`MediaBrowser.Controller.MediaEncoding.IMediaEncoder` is injectable and +exposes + +| Member | Use | +|---|---| +| `EncoderPath` | Absolute path to the server's `ffmpeg` binary | +| `ProbePath` | Path to `ffprobe` | +| `EncoderVersion` | Version gating | +| `SupportsEncoder(...)` | Capability check | + +So there is **no new dependency and nothing to bundle** — the plugin takes +`IMediaEncoder` through DI (registered in `ServiceRegistrator`) and invokes the +binary Jellyfin is already using for transcoding. + +**FFmpeg does the hard part.** Decode, downmix, resample and format conversion +are all a single invocation; the plugin never touches a codec: + +``` +ffmpeg -nostdin -v error \ + -ss -t 120 \ + -i \ + -vn -ac 1 -ar 11025 -f f32le - +``` + +That streams 120 s of mono 32-bit float PCM at 11025 Hz to stdout — +~5.3 MB, read incrementally rather than buffered whole. `-ss` **before** `-i` +makes the seek fast, which matters when sweeping a library. + +**What remains in C# is only the DSP**, and it is modest: + +1. Hann window, 4096-sample frames, 1024 hop (~1290 frames). +2. Real FFT per frame. +3. Log-magnitude, 300–3000 Hz band, 32 log-spaced bins, take peak bin + + 2-bit energy class. +4. Pack one byte per frame, base64. + +A radix-2 real FFT over 4096 samples is on the order of a hundred lines and +has no external dependency. Avoid pulling in a DSP package: this is a fixed, +well-specified transform, and vendoring a small implementation keeps the +plugin's dependency surface at zero, which matters for a GPLv3 Jellyfin +plugin. + +Cost is dominated by the FFmpeg seek and decode, not the FFT: roughly a second +or two per item, entirely I/O-bound. + +**Where it runs in the plugin:** + +- On demand, for `POST /Plugins/JRay/Items/{itemId}/Identify` (§9). +- As a **scheduled task** that backfills signatures for library items, so a + sweep is not blocked on computing them inline. Signatures are cached against + the item (keyed on item id + file mtime + size, so a replaced file + recomputes). +- Before contributing a manifest, so uploads carry `cut.audio_signature`. + +**Degradation, not failure.** If `IMediaEncoder` is unavailable, the binary is +missing, the item has no audio stream, or the file is shorter than the window, +the plugin logs and proceeds **without** a signature. Every existing tier keeps +working; UR-009 is an enhancement and must never be able to break a fetch. + +#### Media shorter than the window — 120 s + +**Items under 120 s emit no signature at all, and no sync offset is applied to +them.** The window is `runtime/2 ± 60 s`, so below 120 s it underflows: there is +no shortened window to compute, because the construction has no definition +there. Such items fall back to the `exact` and `runtime` tiers, which is +adequate — a sub-two-minute item is rarely the ambiguous-providence case UR-009 +exists to solve. + +The signature is therefore **fixed-length by construction**, not merely bounded. +That is what keeps it inside SR-004: a caller cannot choose the length, so the +field cannot be used as a variable-size container (§5a, and "Validation and +abuse" below). + +> **Reconciled with `scene-actor-extraction` IR-007.** An earlier draft of this +> section said items under *150 s* got a centred, shortened window with the +> frame count recorded. That described a mechanism neither producer implements, +> and it was the weaker rule: a caller-varying length is exactly the property +> SR-004 forbids. The 120 s cutoff is now identical in both producers and in +> this server's validator, which is what IR-007 requires — a rule that differs +> between producers yields signatures that never match. + +**The extraction pipeline (C++) is then optional for UR-009.** It may compute +signatures for files it processes — `libswresample` plus an FFT, as noted +above — but since the plugin computes them for the whole library and attaches +them on contribution, the pipeline need not implement this at all. That +removes the `libswresample` work from the critical path. + +--- + +## 4. API + +Base path `/api/v1`. JSON throughout. + +### `GET /manifests/exists` — UR-001 + +Cheap existence probe. Answers whether a manifest is available for a given +title *and* at what cut-match tier, without transferring the payload. + +Query parameters are the same identity + cut parameters as the fetch +endpoints: `tmdb_id` / `imdb_id` (or `series_tmdb_id` + `season` + `episode`), +plus optional `runtime_sec` and `video_hash`. + +```json +{ "exists": true, "match": "runtime", "manifest_id": "01HZ...", "actor_count": 34 } +``` + +`exists: false` is returned with `200`, not `404` — absence is a normal answer +to this question, and using `404` would conflate "no manifest" with "bad +route" for the client. + +If `runtime_sec` and `video_hash` are both omitted, the response reports +whether *any* manifest exists for the title with `"match": "unknown"`; the +client must still fetch to find out whether a cut actually aligns. This is +the mode a library-wide sweep uses. + +#### Batch form + +A client sweeping a library should not issue one request per item. The batch +form takes up to 100 items: + +``` +POST /manifests/exists +{ "items": [ { "tmdb_id": "504172", "runtime_sec": 6420.5 }, ... ] } +``` + +returning results positionally. This exists specifically so the rate limit in +§5 can be generous per *request* while staying strict per *item*, and so a +2000-item library sweep is 20 requests rather than 2000. It is a `POST` only +because the payload does not fit a query string; it is a read and requires no +token. + +### `GET /manifests/movie?tmdb_id=&imdb_id=&runtime_sec=&video_hash=` + +Returns the best-matching Jmanifest, or `404` if none clears `loose`. + +```json +{ "match": "runtime", "manifest": { "...": "..." } } +``` + +### `GET /manifests/series/{series_tmdb_id}?season=` + +Returns a series bundle (§2). `season` optional; omitted means all seasons. +Episode-level cut matching is done client-side against the returned bundle, +since a client pulling a whole series already knows its own runtimes. + +### `GET /manifests/episode?series_tmdb_id=&season=&episode=&runtime_sec=&video_hash=` + +Single-episode equivalent of the movie endpoint. + +### `POST /manifests` + +Contribute a manifest. Body is a Jmanifest. Requires an API token (§5). + +- `202 Accepted` — passed size and schema validation; held unlisted pending + the TMDB cast check (§6). Returns `{ "manifest_id": "...", "status": "pending" }` +- `400` — malformed, or contains an unrecognised or forbidden field (§6) +- `409` — an identical `(identity, cut)` manifest already exists from this + contributor +- `413` — body exceeds the size limits (§6) +- `429` — rate limited (§5) + +### `POST /manifests/bundle` — UR-006 + +Contribute a whole series in one request. Body is a series bundle (§2). +Per-episode validation, non-atomic, one rate-limit unit, shared TMDB fetch — +see "Series bundle upload" in §2. + +- `202 Accepted` — returns per-episode outcomes +- `400` — the bundle envelope itself is malformed (individual bad episodes are + reported in the results list, not as a whole-request error) +- `413` — exceeds 500 episodes or 25 MiB + +### `GET /manifests/{id}/status` + +Poll the outcome of the asynchronous cast check for an upload: +`{ "status": "pending" | "listed" | "flagged" | "rejected", "reason": "..." }`. + +### `GET /manifests/{id}` + +Fetch a specific manifest by its server-assigned id (for debugging and for +the "report this manifest" flow). + +### `POST /manifests/{id}/report` + +Flag a manifest as wrong (misaligned, wrong actors). Body: +`{ "reason": "misaligned" | "wrong_actors" | "spam", "note": "..." }`. + +### `GET /health` + +Liveness. Unauthenticated. + +--- + +## 5. Authentication and abuse + +Reads are anonymous and cacheable. Writes require a token — an anonymous +bearer capability, not an account. No email, no verification, no personal +data; see §5a for why identity is deliberately not load-bearing. + +### Rate limiting — UR-004 + +Limits are per token where one is present, otherwise per source IP. Anonymous +reads are keyed on IP, which is imperfect behind CGNAT; the limits below are +therefore set well above what a single real server needs. + +| Surface | Limit | Rationale | +|---|---|---| +| `GET /manifests/exists` | 600 / hour | Sweeps should use the batch form | +| `POST /manifests/exists` (batch) | 60 / hour, ≤100 items each | 6000 items/hour — a large library sweeps in one pass | +| Manifest fetches (`/movie`, `/episode`) | 300 / hour | A client only fetches what `exists` said was there | +| `GET /manifests/series/{id}` | 120 / hour | Bundles are ~100–250 KiB; this is the preferred path for TV and should not be scarcer than per-episode fetching | +| `POST /manifests` | 100 / hour per token | Nobody uploads faster than the CV pipeline runs | +| `POST /manifests/bundle` | 20 / hour per token, ≤500 episodes each | One unit per bundle, so contributing a season is not penalised versus a film | +| `POST /manifests/{id}/report` | 20 / hour per IP | Reports are a moderation lever; cheap to abuse | +| `POST /manifests/search` (audio) | 60 / hour | Most expensive read endpoint (§3); sliding comparison over candidates | +| `GET /federation/peers` | 60 / hour per IP | Public directory, read by humans; no reason for volume | +| `GET /federation/changes` | 120 / hour per peer | Hourly polling is the default; this allows generous catch-up | +| `GET /federation/manifests/{content_id}` | 5000 / hour per peer | Bootstrap pulls are bulk by nature; capped so one peer cannot saturate egress | +| `POST /federation/have` | 120 / hour per peer, ≤1000 ids each | Diffing a catalogue should be a handful of requests | +| `GET /health` | unlimited | Liveness | + +Responses carry `X-RateLimit-Limit`, `X-RateLimit-Remaining` and +`X-RateLimit-Reset`; exceeding a limit returns `429` with `Retry-After`. The +JRay client must honour `Retry-After` and back off exponentially rather than +retrying tightly — a scheduled library sweep that ignores this will get an +instance's IP throttled. + +Implemented as a fixed-window counter keyed on `(token_or_ip, surface)`, +held in process memory (§8) — no external counter store. A sliding window is +not worth the complexity at this volume. Counters reset on restart, which is +acceptable for abuse throttling. Read limits are applied *behind* the CDN +cache, so a cache hit costs a client nothing against its budget. + +--- + +## 5a. Trust model + +**Design goal: no accounts, no identity, no moderation queue that scales with +users — and no way to use the server as a content host.** + +The key property that makes this tractable: a Jmanifest is not free-form +content. It is a *closed-vocabulary* document — a title identity, a runtime, +and a list of actors with timings. Everything in it is checkable against an +external ground truth (TMDB) that the attacker does not control. So trust can +attach to **content**, not to **contributors**. This is why the server needs +no accounts: a manifest listing pornstars for a children's film fails the +check regardless of who uploaded it, and a valid manifest is valid regardless +of who uploaded it. + +### Threat 1 — using the server as a content store + +The concern is the server being used to host illegal material (the worst case +being CSAM) or arbitrary payloads, making the operator liable. + +The structural defence is that **there is nowhere to put it**. After §6 +stage 2, an accepted document contains only: + +| Field | Constraint | +|---|---| +| `identity.tmdb_id` / `imdb_id` | Regex-constrained to digits / `tt\d{7,8}` | +| `identity.season`/`episode`/`year` | Bounded integers | +| `cut.*` | Numbers, plus a fixed-format hash | +| `extraction.*` | Numbers and a version string from an allow-list | +| `actors[].tmdb_id` / `imdb_id` | Regex-constrained | +| `actors[].scenes` | Pairs of floats | +| `actors[].name`, `identity.title` | **The only free-form strings** | + +No binary. No images. No URLs. No base64 fields. No extension points — because +`extra="forbid"` applies at every nesting level, an attacker cannot add one. + +That reduces the entire content-hosting surface to two short text fields, which +are then constrained further: + +- **Length caps.** `name` ≤ 200 chars, `title` ≤ 300. With ≤ 500 actors that is + a hard ceiling of ~100 KB of attacker-controlled text per manifest, but see + the next two rules, which cut it far below that. +- **Character class.** Names must match a permissive-but-closed pattern: + Unicode letters, marks, spaces, and `. ' - ,` only. No digits, no `/ + =`, + no control characters, no zero-width or bidi-control codepoints, NFC + normalised. **This alone defeats base64/hex smuggling**, which needs digits + and padding characters. +- **Cross-check against a known vocabulary.** Every actor name must correspond + to a real TMDB person (§6 stage 3). A name that matches no TMDB person is + not stored at all. An attacker therefore cannot write arbitrary strings — + only strings that already exist in TMDB's person index. + +Combined, the last rule is decisive: **the server does not store +attacker-authored text, it stores references to TMDB entities.** The strongest +form of this — and what I recommend for v1 — is to go one step further and +**not persist the submitted name string at all**: + +> Store `tmdb_person_id` plus the timings. Resolve display names from the +> server's own TMDB-derived person table at serve time. The uploaded `name` +> field is used only for matching during validation, then discarded. + +At that point the free-text channel is closed completely. The only +attacker-controlled values that reach the database are integers. There is no +CSAM risk and no payload-smuggling risk because there is no field capable of +carrying either. + +This also resolves your point about not storing JSON files: with names +normalised to person ids, the natural representation is relational rather than +a blob. See §7. + +### Threat 2 — prank and vandalism manifests + +Semantically valid but wrong: casting pornstars in a children's film, or +mislabelling a film's cast as a joke. Every structural check passes; only +ground truth catches it. + +Defence is the TMDB cast cross-check in §6 stage 3. Its effectiveness rests on +the attacker not controlling TMDB: to make a pornstar manifest pass, they would +need those performers to be *credited cast on that title in TMDB*, which means +vandalising TMDB itself — a separate, moderated system with its own edit +history. That is a meaningfully high bar for a prank. + +Additional layers, in order of cost: + +1. **Category guard.** Reject any manifest where a matched TMDB person's + known-for department or credits are dominated by titles TMDB flags as + adult (`adult: true`), unless the target title is itself flagged adult. + This directly targets the stated prank without needing a blocklist of + names. +2. **Age-appropriateness guard.** If the target title's TMDB certification is + a children's rating, apply the strictest cast-match threshold and require + an `exact` or `runtime` cut match. Mismatched content on children's titles + is the highest-harm case and deserves the tightest gate. +3. **Divergence detection.** When two manifests exist for the same + `(title, cut)` from different sources and their actor sets disagree beyond + a threshold, flag both and serve the one with the better cast-match ratio. + Honest extractions of the same cut converge; a prank diverges from them. + +### What replaces accounts + +Contribution requires a token, but a token is **not an account** — it is an +anonymous bearer capability: + +- Self-issued on request, no email, no verification, no personal data. +- Stored only as a hash. The server cannot enumerate who holds tokens. +- Its sole purposes are rate-limiting attribution (§5) and revocation. +- Discarding a token and requesting another is trivially easy — and that is + *fine*, because the token is not the defence. The content checks are. A new + token gains an attacker nothing, since every upload faces the same + ground-truth validation. + +This is the crucial difference from an account system: the token exists to +throttle volume, not to establish identity. Sybil resistance is not required +because identity is not load-bearing. + +Consequently the only reputational state is per-token counters +(`accepted`, `rejected`, `flagged`), used for one purpose: a token whose +rejection rate exceeds a threshold over a minimum sample is revoked +automatically, and its `pending`/`flagged` manifests are dropped. No human is +in the loop for the common case. + +### Residual risk and the operator's lever + +Two things remain that automation cannot fully close: + +1. A manifest that is *plausible but wrong* — correct cast, deliberately + misaligned timings — degrades the overlay but carries no legal or safety + risk. Reports plus divergence detection handle it. +2. A novel abuse pattern nobody anticipated. + +For both, the operator needs a **kill switch**, not a moderation queue: +`status` transitions (§7) are a single column, so delisting a manifest, every +manifest from a token, or every manifest for a title is one UPDATE. Delisting +is instant and reversible; deletion is a separate, logged action. + +**Legal posture.** Because the server stores only integers and references to +TMDB entities, it holds no user-generated content in the sense that +intermediary-liability regimes contemplate. This should be stated plainly in +the operator documentation, alongside a contact address for takedown requests. +It is a materially better position than "we store user-submitted JSON and +moderate it". + +### Client-side hardening + +Independent of the server, because a compromised or hostile server must not be +able to attack its clients: + +- The JRay overlay renders actor names as **text nodes only**, never as HTML. +- The plugin validates downloaded manifests against the same schema it would + apply to an upload — a client must not trust a manifest merely because the + server served it. +- Downloaded manifests are stored via the existing `IManagedTruthStore` and + never written into the media library filesystem. + +--- + +## 6. Upload validation — UR-003 + +Validation runs in four stages, ordered cheapest-first so that abusive +uploads are rejected before they cost anything. + +### Stage 0 — reject on headers, before the body is read + +The cheapest rejection is the one that happens before any payload is accepted. + +> **In one line:** `Content-Length` is the fast path; the streaming byte +> counter is the enforcement. The header is a *claim by the client*, so it +> rejects honest oversized uploads early and cheaply, but it cannot be the +> only check — a lying header, a chunked upload, or a compressed body all pass +> it. Implement both; in Axum they are the same one-line layer (§8). + +Three mechanisms, in order of how early they fire: + +**1. `Expect: 100-continue` (earliest — body genuinely never sent).** +A client may send headers with `Expect: 100-continue` and wait before +transmitting the body. The server responds `100 Continue` or, if +`Content-Length` already exceeds the cap, `413` — and the body is never +transmitted at all. This is the true "reject before upload". + +Clients under this project's control — the JRay plugin contributing manifests +(§9) and federation peers pulling (§9a) — **should** use `Expect: 100-continue` +for uploads, because it turns a rejected 25 MiB bundle into a two-header +exchange. The server must handle it correctly, but must never *depend* on it: +arbitrary clients will not send it. + +**2. `Content-Length` check (normal case).** +When present, the declared length is available in the request headers before +the body. If it exceeds the cap for that route, respond `413` immediately and +do not read the body. + +This is a filter, not a guarantee: a hostile client can declare +`Content-Length: 100` and then send gigabytes. **The streaming cap below is +therefore mandatory, not redundant.** + +**3. Chunked requests have no declared size.** +HTTP/1.1 `Transfer-Encoding: chunked` omits `Content-Length` entirely, so +there is nothing to check up front. These must be capped while streaming. + +> **What "before anything is uploaded" can and cannot mean.** Even on an +> immediate `413`, a client has typically already put some body bytes on the +> wire — they may sit in kernel or proxy buffers before the response lands. +> The achievable guarantee is that the server never *reads, buffers, or +> parses* an oversized body, and closes the connection promptly. It is not +> that zero bytes cross the network. Do not size defences on the assumption +> that a `413` prevents transmission. + +### Stage 1 — size limits while streaming (before parsing) + +Enforced at the reverse proxy and again in the app, on the raw body, *before* +JSON parsing. A parser handed an unbounded body is a denial-of-service +primitive, so this must not be deferred to the schema layer. + +The app-level cap counts bytes as they are read and **aborts mid-transfer** +once exceeded, rather than reading to completion and then measuring. This is +what makes a lying `Content-Length` and a chunked upload both safe. + +| Limit | Value | +|---|---| +| Request body (movie or episode manifest) | 2 MiB | +| Request body (series bundle, §2) | 25 MiB | +| Request body after gzip decompression | 8 MiB, with a max compression ratio of 20:1 | +| JSON nesting depth | 12 | +| `actors[]` entries | 500 | +| `scenes[]` entries per actor | 2000 | +| Total scene windows across all actors | 20000 | + +For scale: the sample feature film in the extraction repo has ~30 actors and a +few hundred windows. These caps are roughly an order of magnitude above +anything legitimate. Oversized bodies are rejected with `413`. + +The decompression-ratio cap matters because a gzip bomb passes a 2 MiB +body-size check trivially. Decompression must also be **streamed with a +running output cap** — decompressing fully and then checking the size defeats +the point. + +**Implementation.** Axum's `DefaultBodyLimit` (§8) implements the streaming +cap and honours `Content-Length` for early rejection, applied per-route so the +bundle endpoint gets its larger limit without widening the others. Set the +matching `client_max_body_size` (nginx) / `request_body max_size` (Caddy) at +the proxy so oversized uploads are dropped at the edge and never occupy an +application worker. + +### Stage 2 — strict schema (synchronous, rejects with `400`) + +**No additional fields anywhere.** Every object in the document is validated +in strict mode — serde `#[serde(deny_unknown_fields)]` on every DTO (§8) — so +an unrecognised key at any nesting level is an error, not something silently +ignored. This is the default posture, not a special case for the two fields +below, and it is enforced by the type definitions rather than by validator +code that could omit a field. + +Rejected outright: + +- **any unrecognised field**, at any level of the document +- `movie` present, or any string anywhere that looks like an absolute + filesystem path (`/…`, `C:\…`, `\\…`) or a `file://` URI +- `actors[].jellyfin_id` present and non-empty +- missing `cut.runtime_sec` +- neither `tmdb_id` nor `imdb_id` in `identity` +- `imdb_id` not matching `^tt\d{7,8}$`, actor `imdb_id` not matching + `^nm\d{7,8}$`, `tmdb_id` not matching `^\d{1,9}$` +- scene windows with `end < start`, negative times, non-finite values + (`NaN`/`Infinity`), or times beyond `runtime_sec` + 5s tolerance +- actor names longer than 200 characters, containing control characters, or + failing Unicode normalisation to NFC +- duplicate actors within one manifest (same `imdb_id`) + +Rejecting unknown fields is what makes the `movie` / `jellyfin_id` strip in +§9 verifiable: a client that forgets to strip them gets a hard `400` naming +the offending field, rather than quietly publishing a contributor's directory +layout. + +### Stage 3 — TMDB cast cross-check (asynchronous, after `202`) + +This needs an outbound TMDB call and so cannot run inside the request without +coupling upload latency to a third party. The upload is accepted with `202` +and the manifest is held **unlisted** until the check completes; it is not +served to anyone in the meantime. + +The check: fetch the TMDB credits for `identity.tmdb_id`, take the set of +credited cast **TMDB person ids**, and compare against the actors in the +manifest. + +> **Join on `tmdb_id`, not `imdb_id`.** This is grounded in the actual +> pipeline output, not assumed. Of the 331 manifests in the +> scene-actor-extraction repo, exactly **one** has IMDB ids populated; the +> other 330 have `imdb_id: ""` with `tmdb_id` set. The Jellyfin-gallery path +> (`make_jellyfin_gallery.py`) — which is the path most users will take, since +> it needs no TMDB key — yields TMDB ids only: 2391 of 2392 gallery entries +> have a `tmdb_id`, and **none** have an `imdb_id`. +> +> A design keyed on IMDB ids would therefore fall back to name matching for +> essentially every real upload, which is exactly the weak path. `tmdb_id` is +> the join key; `imdb_id` is an optional secondary signal when present. + +Let *M* = actors in the manifest, *C* = credited cast from TMDB. + +| Condition | Outcome | +|---|---| +| `\|M ∩ C\| / \|M\|` ≥ 0.6 | **Listed.** Normal case | +| 0.3 ≤ ratio < 0.6 | **Listed, flagged** for review; served with reduced ranking | +| ratio < 0.3 | **Rejected.** Manifest is deleted and the contributor notified | +| TMDB has no credits for the id | **Listed, flagged** — absent data is not evidence of a bad manifest | +| TMDB unreachable / rate-limited | **Retry** with backoff; stays unlisted, not rejected | + +The match is deliberately *approximate* and directional. It asks "are these +plausibly this film's cast?", not "is this cast list complete": + +- Ratio is over *M*, not *C*. A manifest legitimately contains only actors who + were both credited and detected on screen, so it is normally a strict subset + of the cast — penalising it for missing credited actors would fail every + honest upload. The corpus bears this out: median 7 actors per manifest, + against feature casts several times larger. +- Uncredited appearances, cameos, and actors TMDB lists only under a + differently-spelled name are exactly why the threshold is 0.6 and not 1.0. +- Matching is on `tmdb_id` (see above), with `imdb_id` as a secondary signal + when present, falling back to case- and accent-insensitive name comparison. + Name-only matches are counted but capped at half the intersection, so a + manifest cannot pass on name collisions alone. + +**Small-*M* handling.** With a median of 7 actors, a ratio threshold is coarse +— one mismatch moves it by 14%. So: + +- `|M|` ≥ 5: apply the ratio table above. +- `2 ≤ |M| < 5`: require *all but one* actor to match. A ratio is meaningless + at this size. +- `|M| ≤ 1`: accept only if the single actor matches; such a manifest is + near-worthless anyway and is ranked last. +- `|M| == 0`: **reject.** 15 of the 331 corpus files have empty actor lists — + these are extraction failures, not contributions, and must not be uploaded. + The client should refuse to submit them. + +**Every matched actor is resolved to a TMDB person id, and unmatched actors +are dropped rather than stored.** This is what closes the free-text channel +described in §5a: a manifest is persisted as a set of TMDB person references, +so a name that corresponds to no TMDB person never reaches the database. + +For episodes the check runs against the **union** of TMDB's per-episode +credits (cast + guest stars) and the series' aggregate credits. + +Using the union rather than either alone matches what the extraction client +already does: `run_from_jellyfin.py` defaults to `--episode-cast tmdb`, taking +TMDB per-episode credits and falling back to series-wide when the episode has +no usable credits. Checking against per-episode credits alone would reject +recurring cast that TMDB lists only at series level; checking against +series-wide alone would reject legitimate guest stars. The union admits both, +and since the ratio is over the manifest's actors (not TMDB's cast), widening +the reference set costs nothing in strictness against pranks — a pornstar is +in neither set. + +TMDB responses are cached (24h) so that a burst of episode uploads for one +series costs a single upstream call, and so the server stays within TMDB's own +rate limits. + +### Sanity-checked and warned, not rejected + +- total on-screen coverage implausibly high (>95% of runtime) or near zero +- an actor whose windows sum to under a second +- `sample_fps` below 1, which yields low-quality timings + +Because the contributed manifest is stripped of `jellyfin_id`, the downloading +server resolves actors locally via `tmdb_id` (primarily) against its own People +`ProviderIds` — which is exactly the fallback path JRay already implements. + +--- + +## 7. Storage + +SQLite in WAL mode (§8), **fully relational — no JSON blobs on the write +path.** The schema below is portable SQL and runs unchanged on Postgres should +an instance ever outgrow SQLite. + +Storing the uploaded document as a JSON payload would undermine §5a: a blob +is an opaque container, so whatever the schema validator missed gets persisted +verbatim and served back out. Decomposing into columns means **the database +can only represent what the schema models** — there is physically nowhere for +an unexpected field or a smuggled string to live. Normalisation is a security +control here, not just tidiness. + +Concretely, the submitted JSON is parsed, validated, resolved to TMDB person +ids, written as rows, and **discarded**. The document served to clients is +*reconstructed* from those rows, never echoed. + +```sql +contributors (id, token_hash, created_at, revoked_at, + accepted_count, rejected_count, flagged_count) + +people (tmdb_person_id PK, -- server-side, TMDB-derived + name, -- from TMDB, never from an upload + adult bool, updated_at) + +titles (id PK, kind, -- movie | series + tmdb_id, imdb_id, name, year, + adult bool, certification, updated_at) + +manifests (id PK, title_id FK, season, episode, + runtime_sec, video_hash, + audio_signature blob NULL, -- §3, ~1290 bytes + audio_sig_coarse blob NULL, -- candidate-generation index key + sample_fps, extinction_sec, pipeline_version, + gallery_size, gallery_scope, -- ranking signal, §2 + contributor_id FK, + status, -- pending | listed | flagged | rejected + cast_match_ratio real, created_at) + +manifest_actors (manifest_id FK, tmdb_person_id FK, + PRIMARY KEY (manifest_id, tmdb_person_id)) + +scenes (manifest_id FK, tmdb_person_id FK, + start_cs integer, end_cs integer) -- centiseconds, see §9a + +reports (id, manifest_id FK, reason, note, created_at, source_ip_hash) +tmdb_cache (tmdb_id, kind, credits json, fetched_at) + +jobs (id, kind, -- cast_check | federation_pull + payload, run_after, attempts, last_error) +``` + +`jobs` is the background queue (§8) — a table rather than an external broker, +so pending work survives a restart. + +Note what is **not** in this schema: no actor-name column on any upload-derived +table. Display names come from `people.name`, populated from TMDB by the +server. `tmdb_cache` is the sole JSON column and holds *TMDB's* responses, +not users'. + +Scene times are stored as **integer centiseconds** (§9a), not floats — the +same quantisation used for `content_id`, so stored values and hashed values +cannot diverge. + +Indexes on `titles(tmdb_id)`, `manifests(title_id, runtime_sec)`, +`manifests(video_hash)`, `manifests(title_id, season, episode)`, and +`scenes(manifest_id, tmdb_person_id)`. All read queries filter +`status IN ('listed','flagged')`, so a partial index on that predicate keeps +the hot path small. + +`scenes` is the only table with real row volume — roughly (actors × windows) +per manifest, capped by §6 at 20000 rows. At corpus-realistic sizes (median 7 +actors) it is a few hundred rows per manifest, so even tens of thousands of +manifests stay comfortably small. + +Multiple manifests may coexist for the same title with different cuts — that +is the point. Multiple manifests for the *same* cut from different +contributors are allowed too; serve the one with the best +`(cast_match_ratio, reports, gallery_scope, sample_fps)` ranking. + +`gallery_scope` enters the ranking because it is the strongest available +quality signal between two otherwise comparable manifests (§2): a manifest +extracted against a `global` gallery had to distinguish its actors from every +other actor in the contributor's library, whereas a `limited` one only had to +distinguish them from that title's own cast. The former surviving the cast +check is stronger evidence than the latter doing so. It ranks below +`cast_match_ratio` and reports, which are evidence about *this* manifest rather +than about the conditions that produced it. + +--- + +## 8. Stack + +Traffic is low and read-dominated; a reverse-proxy or CDN cache in front keeps +the app tier trivial. + +Two capabilities beyond request handling and storage are load-bearing rather +than optional: + +- **Rate-limit counters** (§5). +- **Asynchronous background work** — the TMDB cast check (§6 stage 3) and its + retries, plus federation pulls (§9a). + +Both are satisfied in-process by the recommended stack below; neither requires +a separate service. + +The server needs a **TMDB API key** as operational configuration. It is a hard +dependency for UR-003: if TMDB is unreachable, uploads accumulate in `pending` +rather than being listed unverified. + +### Recommended production stack + +**Rust + Axum + SQLite**, behind an operator-provided reverse proxy. + +| Layer | Choice | Version at time of writing | +|---|---|---| +| Framework | **Axum** + Tower/`tower-http` | axum 0.8 | +| Runtime | **Tokio** | 1.x | +| Database | **SQLite** (WAL mode) | 3.4x | +| DB access | **sqlx** (compile-time checked SQL) or **rusqlite** | sqlx 0.9 / rusqlite 0.40 | +| Serialization | **serde** / **serde_json** | 1.x | +| HTTP client | **reqwest** (TMDB, federation pulls) | 0.12 | +| Observability | **tracing** + OpenTelemetry exporter | — | +| Edge / TLS | Operator-provided (Caddy, nginx, Traefik) | — | +| Packaging | Single static binary + one DB file | — | + +**Why this fits.** The workload is read-dominated, low-volume, and +cache-frontable; the largest response is a ~228 KiB series bundle. Raw +throughput is not the constraint — Postgres/SQLite queries and TMDB calls are. +What *does* matter here is operational simplicity for hobbyist operators +(§9a expects independent people to run instances) and strictness at the +validation boundary (§6, §5a). Rust serves both: a single static binary plus +one file is the lowest-friction thing an operator can deploy, and a strict +type system at the parse boundary is exactly the posture §5a asks for. + +**Axum over Actix Web.** Actix leads on raw throughput by ~10–15% under heavy +load, which is irrelevant at this volume. Axum's Tower middleware composition +maps directly onto what the spec needs — rate limiting (§5), body-size limits +(§6 stage 1), tracing, timeouts — as composable layers rather than bespoke +code. It is the mainstream default for new services and the easier codebase +for occasional contributors. + +**Serde `deny_unknown_fields` is the §6 stage 2 enforcement mechanism.** This +is the strongest argument for Rust here. `#[serde(deny_unknown_fields)]` on +every DTO gives the "no additional fields anywhere" requirement structurally, +checked at compile time against the type definitions, with no possibility of +a field being silently accepted because a validator forgot it. Combined with +newtypes for `TmdbId`, `ContentId`, and centisecond timestamps, malformed +input fails to parse rather than being caught later — invalid states become +unrepresentable rather than merely rejected. + +**Axum's `DefaultBodyLimit`** enforces the §6 stage 0/1 caps in the framework: +it rejects on `Content-Length` before reading a body, and caps the stream for +chunked or mis-declared uploads — satisfying "reject before parsing" without +trusting the client's declared size. + +### SQLite: the write-concurrency question + +SQLite is the right call, but it has one hard constraint that must be designed +around rather than discovered: **only one writer at a time, even in WAL mode.** +Concurrent write transactions return `SQLITE_BUSY`. + +Reads are unaffected — WAL gives concurrent readers alongside the single +writer, which suits a read-dominated workload well. The risk is concentrated +in this spec's three bulk-write paths: + +| Path | Write shape | Risk | +|---|---|---| +| Single manifest upload | ~106 scene rows median | Negligible | +| Series bundle upload (§2) | 24 episodes × ~106 rows ≈ 2.5k rows | Moderate — one long transaction | +| Federation bulk ingest (§9a) | Thousands of manifests | **This is the real one** | + +Federation bootstrap is explicitly a bulk-write workload, and it runs +concurrently with live uploads. Mitigations, which are requirements rather +than suggestions: + +- **WAL mode**, plus `busy_timeout` (5s) so contention waits rather than + errors, and `synchronous = NORMAL` (safe under WAL). +- **A single writer connection**, serialized through one task/actor, with a + read pool alongside. Do not point a multi-connection pool at writes and rely + on `busy_timeout` to sort it out — serialize deliberately. +- **Chunked ingest transactions.** Federation ingest commits per manifest, not + per batch, so a bootstrap never holds the write lock for long. Combined with + the §9a `MaxIngestPerHour` cap, live uploads are not starved. +- **Batch inserts within a transaction** for a manifest's scene rows — one + transaction per manifest, not per row. + +With those, a single modest VPS handles this comfortably. SQLite does tens of +thousands of writes/sec on modern hardware; the constraint is lock *duration*, +not throughput. + +**When to reconsider.** If an instance ever runs multiple writer processes, or +federation bootstrap contention becomes visible in practice, Postgres is the +escape hatch. Keep the SQL portable and use sqlx (which supports both) so the +migration is a configuration change rather than a rewrite. **Do not** design +around a hypothetical Postgres future at the cost of SQLite's simplicity now. + +### Alternatives considered + +The single-writer constraint above is the one real weakness, so it is worth +being explicit about why SQLite still wins. + +| Option | Verdict | +|---|---| +| **SQLite** (rusqlite / sqlx) | **Chosen.** Ubiquitous, unmatched track record, trivially portable, one file | +| **Turso** (SQLite rewritten in Rust, MVCC) | Strong future candidate; pre-1.0 today | +| **libSQL** (C fork of SQLite) | Viable, but its own maintainers now direct effort at Turso | +| **Postgres** | The escape hatch, not the default — a service to operate, against §9a's goal | +| **redb / fjall / sled** | Wrong data model — see below | +| **DuckDB** | Analytical (OLAP); this is a transactional point-lookup workload | +| **SurrealDB** | Far larger surface area than needed; not an embedded-first story | + +**Key-value stores are the wrong shape, not merely a different one.** redb is +mature (v4.1, actively developed) and gives MVCC with concurrent readers plus a +single writer — but it is a key-value B-tree with no SQL, no secondary indexes +and no joins. §7 is a genuinely relational schema: foreign keys between +`manifests`, `scenes`, `manifest_actors` and `people`, partial indexes on +`status`, and queries that join and filter across them. On a KV store all of +that becomes hand-maintained index keys and application-side joins — more code +in exactly the layer where §5a demands correctness. `sled` is additionally out +on maintenance grounds (last release October 2024). + +**Turso deserves a serious look, just not yet.** It is a clean-room Rust +rewrite of SQLite whose `BEGIN CONCURRENT` / MVCC mode (`PRAGMA journal_mode = +'mvcc'`) directly removes the single-writer limitation described above — the +precise weakness in this design. It is SQLite-compatible, so the schema and +most queries carry over, and it is developed with deterministic simulation +testing. But as of this writing it is **pre-1.0**; the maintainers state it +powers production systems while being explicit that they have not yet reached +their "SQLite-level reliability" bar. It also shifts work onto the +application: MVCC transactions that touch overlapping data return a conflict +error and must be retried, so the caller owns retry logic. + +For a small, low-write, read-dominated service where the mitigations above +already keep lock duration short, adopting a pre-1.0 database to solve a +problem this workload does not yet have is the wrong trade. The recommendation +is therefore: + +> Build on SQLite, keep the SQL standard and the data-access layer behind a +> thin trait. Re-evaluate Turso when it reaches 1.0 or if federation ingest +> contention shows up in real operation. Because Turso is SQLite-compatible, +> that migration is far cheaper than the Postgres one — which is itself an +> argument for not over-engineering now. + +**A note on portability.** "Portable" here means two distinct things, and +SQLite is best at both: the *file* is portable (a single database file an +operator can copy, back up, or hand to someone bootstrapping a mirror), and +the *SQL* is portable (standard enough to move to Postgres or Turso later). +Any KV store sacrifices the second entirely. + +### What Rust changes elsewhere in the spec + +- **No Redis.** Rate-limit counters (§5) live in process memory (`governor` or + a Tower layer) or in SQLite. A single-process server does not need an + external counter store, and dropping Redis removes a whole moving part. + Note the tradeoff: in-memory counters reset on restart, which is acceptable + for abuse throttling and avoids a dependency for a hobbyist deployment. +- **No separate worker process or broker.** The async TMDB cast check (§6 + stage 3) and federation pulls (§9a) run as Tokio background tasks in the + same binary, with the job queue as a SQLite table so state survives restart. + This replaces arq/Dramatiq/Celery entirely. +- **The TMDB cache** (§7 `tmdb_cache`) stays a table; SQLite's JSON functions + cover the `jsonb` usage, which is only caching TMDB responses. + +Net effect: **one binary, one database file, one reverse proxy.** That is a +materially better deployment story for federation than "app + worker + +Postgres + Redis", and federation only works if running an instance is easy. + +### Cost of choosing Rust + +Stated honestly, since the alternative was Python: + +- The extraction side is Python, so validation and canonicalisation logic + (notably the §9a `content_id` canonical form) can no longer be shared as + one implementation. It must be specified precisely enough to reimplement, + and cross-tested — a golden-vector test fixture shared by both sides. +- Fewer casual contributors than a FastAPI codebase would attract. +- Slower initial development. + +These are real, and worth accepting for a long-lived service whose main risks +are hostile input and operator friction — both of which Rust directly +addresses. + +### Deployment notes + +- Ship a **single static binary** (musl target) plus the SQLite file. Optional + container image, but neither Docker nor Compose should be required. +- Put all database access behind a **thin repository trait** rather than + scattering queries through handlers. This is what keeps the Turso/Postgres + options above cheap, and it localises the single-writer serialization + described earlier in one place instead of every call site. +- Avoid SQLite-specific SQL where a standard form exists (notably `INSERT … + ON CONFLICT`, which is portable, versus `INSERT OR REPLACE`, which is not). +- Terminate TLS at the operator's proxy; the app speaks plain HTTP on + loopback and must trust `X-Forwarded-For` **only** from that proxy — §5 rate + limiting and report attribution key on client IP, so a spoofable header + defeats both. Make the trusted-proxy CIDR explicit configuration, not a + default-on behaviour. +- Enforce the §6 stage 1 body cap at *both* the proxy and + `DefaultBodyLimit`; defence in depth, and the app must be safe when run + without a proxy. +- Set a statement timeout and request timeout (`tower_http::timeout`) so a + slow bundle query fails fast. +- Health checks: `GET /health` for liveness, plus a readiness check verifying + the database opens and migrations are current. +- **Back up the SQLite file** with `VACUUM INTO` or the backup API (never a + plain file copy of a live WAL database). Manifests represent real CV + compute; federation (§9a) gives partial resilience but is not a backup. + +--- + +## 9. JRay plugin integration + +### Configuration + +- **Enable manifest sharing** (default off — this is a network egress feature + and must be opt-in) +- **Servers** — an *ordered list*, not a single URL. See below. +- **Contribute manifests** (separate opt-in from downloading; off by default) +- **Minimum accepted match tier** (`exact` / `audio` / `runtime` / `loose`) +- **Compute audio signatures** (default off) — enables `audio`-tier matching + and unknown-providence search (§3). Uses the FFmpeg binary Jellyfin already + ships, via `IMediaEncoder.EncoderPath`; no extra dependency. + +### Multiple servers + +The plugin queries a user-configured **ordered list** of servers rather than +one. Each entry is: + +| Field | Purpose | +|---|---| +| `Url` | Base URL | +| `Name` | Display label | +| `Token` | Optional; required only to contribute | +| `Enabled` | Toggle without deleting | +| `AllowContribute` | Per-server, independent of fetching | +| `TrustLevel` | `Full` / `FetchOnly` — see below | + +A default entry for the community instance ships pre-configured but +**disabled**, so no traffic leaves an installation until the admin opts in. + +**Resolution order.** For a fetch, servers are tried in list order and the +*first acceptable* result wins — acceptable meaning it clears the configured +match tier. Order is the user's trust ranking, made explicit. Rationale for +first-match over best-match: querying every server for every item multiplies +egress, leaks the library to more parties, and the ordering already encodes +which source the admin prefers. A `Best match across servers` toggle is a +reasonable later addition, off by default. + +For a **series bundle**, first-match applies per *episode*, not per bundle: +fetch the bundle from server 1, then query server 2 only for the episodes +still missing. A series is commonly split across sources, and this is where +multi-server earns its keep. + +**Failure isolation.** A server that is unreachable, slow, or returning errors +is skipped after a short timeout (5s connect, 30s read) and marked +temporarily failed with exponential backoff. One dead server must never stall +a library sweep. Failures are surfaced per-server in the config page. + +**Contribution is never fanned out.** A manifest is contributed only to +servers with `AllowContribute` set, and each is an explicit choice. The plugin +must not broadcast uploads to every configured server — that would multiply +the privacy exposure described below without the user intending it. + +### Trusting third-party servers + +This is the part that does not come for free. Everything in §5a is a property +of a *correctly operated* server. Pointing the plugin at an arbitrary URL +inherits none of it: a hostile server can serve malformed manifests, wrong +casts, or oversized payloads. + +The plugin therefore treats **every** server as untrusted, including the +default one, and re-applies client-side what the server applies on upload: + +- **Validate on receipt.** Downloaded manifests are validated against the same + strict schema used for uploads (§6 stage 2) — unknown fields rejected, sizes + capped, scene windows bounds-checked against the item's real runtime. A + manifest is never trusted merely because a server served it. +- **Response size caps** enforced during streaming, so an unbounded body is + aborted rather than buffered. Bundle cap 25 MiB, single manifest 2 MiB. +- **HTTPS required** for non-loopback servers; certificate validation must not + be disabled. A plaintext community server would let any network intermediary + rewrite actor overlays. +- **Names rendered as text, never markup** (§5a client-side hardening). This + is the single most important control, because it holds even if every other + check is bypassed. +- **`TrustLevel: FetchOnly`** — the default for user-added servers — accepts + manifests but never contributes to them and never sends library inventory + beyond the single item being queried. + +The honest framing for the config page: *adding a third-party server means +trusting its operator not to serve you deliberately wrong actor data.* The +structural protections above bound the damage to bad overlay content; they +cannot make wrong data right. + +### Endpoints + +Mirroring the existing Truth/Tasks controllers: + +- `POST /Plugins/JRay/Items/{itemId}/Fetch` — resolve the item across the + configured servers in order and, on a match at or above the configured tier, + store the result via the existing `IManagedTruthStore`. Admin key. + When the response carries a non-zero `offset` (§3 `audio` tier), the plugin + **must** add it to every scene window before storing — the stored truth file + is always in the local file's own timebase, so the overlay and the `jray?t=` + query need no offset awareness at read time. +- `POST /Plugins/JRay/Series/{seriesId}/Fetch` — bundle fetch for a whole + series, with per-episode gap-filling across servers as described above. +- `GET /Plugins/JRay/Servers/Status` — per-server reachability and last-error, + for the config page. +- `POST /Plugins/JRay/Items/{itemId}/Identify` — compute the item's audio + signature and search configured servers by content (§3), for items whose + providence is unknown. Returns candidate titles with scores and offsets; + storing a result is a separate confirmation step, never automatic. +- A scheduled task that walks items with no truth data and attempts a fetch, + reusing the same backlog logic as `Tasks/Pending`, using the batch + `exists` endpoint (§4) so a sweep is a handful of requests per server. + +Contribution runs the reverse: on a `PUT .../Truth` from a local worker, if +contribution is enabled, strip `movie`/`jellyfin_id`, attach identity from the +item's `ProviderIds` and its measured runtime, and `POST` to each +contribute-enabled server. For a series, batch into a bundle upload rather +than per-episode posts. + +Uploads should set `Expect: 100-continue` (§6 stage 0) so a server that is +going to reject the request on size or auth does so before the body is +transmitted. This matters most for series bundles, where a rejected upload +would otherwise push tens of MiB pointlessly. + +### Privacy + +Contribution reveals to the server operator that some instance holds a given +title. Fetching reveals the same thing. That is inherent, but it means: + +- opt-in, off by default, clearly described in the config page +- no library-wide inventory ever sent in one request — the batch `exists` + endpoint is capped at 100 items and a sweep is paced +- **each configured server multiplies this exposure**, which the config page + must say plainly; first-match resolution limits it, since later servers are + only queried for what earlier ones lacked + +--- + +## 9a. Federation and replication — UR-008 + +Servers can replicate manifests from each other, so a new instance can bootstrap +from an existing one and independent communities need not each re-run the CV +pipeline on the same films. + +### What makes this easy, and what makes it hard + +**Easy:** a validated manifest is *immutable and content-addressable*. Its +content is a fixed set of (TMDB person id, time windows) for a fixed +(title, cut). Nothing about it changes after acceptance. Replication is +therefore **set reconciliation**, not state synchronisation — there are no +concurrent edits, no last-write-wins, no vector clocks, no merge conflicts. +Two servers holding the same manifest hold byte-identical content. + +**Hard:** the mutable state is exactly the part that must *not* replicate +blindly. `status`, `reports` and `cast_match_ratio` encode a *local operator's +judgement and legal position*. A server that pulls another's `delisted` flags +as authoritative has outsourced its moderation; a server that pulls another's +`listed` flags has outsourced its liability. §5a's guarantees are per-operator, +and federation must not silently transfer them. + +The design follows directly: **replicate content, re-derive judgement.** + +### Content addressing + +Every manifest gets a `content_id` — a SHA-256 over its canonical form: + +``` +sha256(canonical_json({ + identity, cut, actors: [{tmdb_person_id, scenes}] sorted by person id +})) +``` + +Canonicalisation: keys sorted, no whitespace, and **scene times quantised to +whole centiseconds** — `round(t * 100)` stored as an integer, not a rounded +float. `extraction` metadata and all local state are excluded, so two servers +that validated the same upload independently arrive at the same `content_id`. + +**`audio_signature` is excluded from `content_id`**, deliberately. It is +derived by decoding audio, so two servers running different FFmpeg or resampler +versions could compute marginally different signatures for the same manifest — +including it would produce different `content_id`s for identical content and +silently break federation deduplication. The signature is replicated as an +attribute of the manifest, not as part of its identity. A peer that already +holds a manifest but lacks its signature may adopt the incoming one. + +Quantising to integers rather than formatting floats is deliberate. Pipeline +timings are *derived* by accumulating `1/fps`, not measured, so they carry +accumulated float error — real corpus values look like `8045.066666660665`. +Measured over 28972 scene values from the extraction corpus, 3-decimal +rounding has a maximum error of 3.3e-4 and produces no boundary cases, so it +is currently safe. But "currently safe" is luck: any value landing near a +`.0005` boundary would hash differently on two servers that computed it +slightly differently, silently defeating deduplication. + +Integer centiseconds remove the failure mode rather than dodging it — 10 ms is +far below the precision any overlay can use (the pipeline samples at 1–10 fps), +so nothing is lost. The client must canonicalise identically, and the +canonicalisation routine should be shared code between server and client +rather than reimplemented. + +This gives deduplication for free: a pulled manifest whose `content_id` is +already present is skipped without re-validation. It also makes "have you got +this?" a cheap hash comparison rather than a content diff. + +### Replication protocol + +Deliberately a **pull-based feed**, not push. Pull means a server chooses what +it ingests and when; push would let any peer inject work into your validation +queue, which is the same abuse surface as anonymous upload but with higher +volume. + +#### `GET /federation/changes?since={cursor}&limit=1000` + +A monotonic, append-only change feed of locally-*listed* manifests. + +```json +{ + "cursor": "01HZ...", + "server_id": "jray.example.org", + "changes": [ + { + "content_id": "sha256:9f2a…", + "op": "add", + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 6420.5, "video_hash": "opensubtitles:8e24…" }, + "actor_count": 17, + "cast_match_ratio": 0.82, + "origin": "jray.example.org", + "seq": "01HZ..." + } + ] +} +``` + +Entries are metadata only — enough to decide whether to fetch, without +transferring payloads. `op` is `add` or `retract` (see below). The cursor is +opaque and monotonic; a peer resumes from its last cursor, making the feed +resumable and idempotent. + +#### `GET /federation/manifests/{content_id}` + +Fetch full content by hash. The puller **must** verify that the returned +content hashes to the requested `content_id` and reject it otherwise — this is +what makes an intermediary or a misbehaving peer unable to substitute content. + +#### `POST /federation/have` + +Batch existence check by `content_id` (up to 1000), so a peer can diff its set +against yours in one request before fetching anything. + +### Ingestion: re-derive, don't inherit + +A pulled manifest is **not** trusted because a peer listed it. It enters the +local pipeline as if freshly uploaded: + +1. Full §6 stage 1 and 2 validation — size caps, strict schema, bounds checks. + A peer is not exempt from the checks that keep §5a's Threat 1 closed. +2. **Local** TMDB cast cross-check (§6 stage 3), against this server's own + TMDB cache and its own thresholds. The peer's `cast_match_ratio` is + advisory only — useful for prioritising ingestion order, never a substitute. +3. Local `status` assigned by this server's rules. + +The peer's moderation decisions are recorded as *signals*, not verdicts: + +| Peer state | Local effect | +|---|---| +| Peer lists it | Eligible for ingestion; still fully re-validated | +| Peer retracts it (`op: retract`) | Local copy flagged for review, **not** auto-delisted | +| Peer never had it | No signal | + +The asymmetry is deliberate: a retraction is a *warning worth acting on*, +while a listing is merely a *nomination*. Auto-delisting on a peer's retraction +would hand any peer a remote delete primitive over your catalogue. + +**Exception — the abuse channel.** One class of retraction *should* +auto-delist: content withdrawn for legal reasons. A `retract` entry may carry +`reason: "abuse"`, and a peer explicitly configured as +`TrustAbuseRetractions: true` will delist immediately and log it. This is +opt-in per peer, and the intended configuration between operators who know +each other. It exists because the alternative — a takedown propagating at the +speed of manual review — is the wrong failure mode for that one case. + +### Origin and loop prevention + +Each change carries `origin`, the `server_id` that first accepted the +manifest, preserved across hops. A server ignores changes whose `origin` is +itself, which prevents the trivial A→B→A loop. Because content is addressed by +hash and ingestion is idempotent, longer cycles are harmless: the second +arrival is a no-op deduplication. + +`origin` is provenance, not authority — it does not confer trust, it just +enables an operator to say "stop ingesting anything originating from X". + +### Peer configuration + +Symmetrical with the plugin's server list (§9), and for the same reason — +federation is trust-by-configuration, not trust-by-protocol: + +| Field | Purpose | +|---|---| +| `Url` | Peer base URL | +| `Enabled` | Toggle without deleting | +| `PullInterval` | Poll cadence, default hourly | +| `TrustAbuseRetractions` | Auto-delist on legal retractions (default false) | +| `IngestFilter` | Optional: only titles matching a filter (e.g. exclude adult-flagged) | +| `MaxIngestPerHour` | Rate cap, so a peer cannot flood the validation queue | +| `Advertise` | Whether to list this peering in the public directory (default false) | + +**There is no automatic peering. Ever.** A peer relationship is created only +by an operator explicitly adding a URL. Nothing a remote server says, and no +data returned from any endpoint, can cause a peering to be established, +re-enabled, or widened. Automatic peering would let the network's trust +properties be set by whoever joins, which is precisely what §5a avoids. + +Federation is off by default. A server with no peers configured behaves +exactly as specified in §1–§9. + +### Peer directory — publishing, not discovering + +A server *may* publish the peers it has chosen, so an operator evaluating the +network can see who is connected to whom. This is a **human-facing directory**, +not a discovery mechanism. + +The distinction is the whole point: + +| | Peer directory (allowed) | Auto-discovery (prohibited) | +|---|---|---| +| What it does | Publishes a list a human can read | Acts on a list a machine received | +| Who decides | The operator, by hand | The protocol | +| Failure mode | Someone reads a stale list | A hostile server injects itself into your trust set, transitively | + +#### `GET /federation/peers` + +Returns peers this server has chosen to advertise: + +```json +{ + "server_id": "jray.example.org", + "contact": "admin@example.org", + "peers": [ + { "url": "https://jray.other.org", "name": "Other Community", "since": "2026-03-01" } + ] +} +``` + +Rules that keep this a directory and not a discovery channel: + +- **Advertising is per-peer opt-in on both sides.** A peering appears here only + if the local operator set `Advertise: true` *and* the remote operator + consented to being listed. Peering with someone must not publish their + existence against their wishes — for a small operator, being listed is an + invitation to traffic and scrutiny they may not want. +- **The response is never ingested.** The server does not parse it, store it, + or act on it. It is rendered in the admin UI for a human, with entries as + inert text and an explicit "add this peer" button that performs the same + manual add as typing a URL. No one-click "add all". +- **Not transitive.** A peer's peers are not fetched recursively. There is no + crawl, so there is no network-wide topology to poison. +- **`contact` is for humans** arranging a peering out of band, which is the + intended workflow: operators talk, then each adds the other by hand. + +Publishing the directory is itself optional (`PublishPeerDirectory`, default +off). A server that would rather not disclose its topology simply doesn't. + +### Duplicate manifests across origins + +Two servers may independently hold manifests for the same (title, cut) from +different contributors — different `content_id`, same identity. This is +already handled: §7 permits multiple manifests per cut and ranks by +`(cast_match_ratio, reports, gallery_scope, sample_fps)` (§7). Federation just +makes it more +common. No deduplication beyond exact `content_id` match is attempted, because +choosing between two plausible extractions is a ranking problem, not a merge +problem. + +### Storage additions + +```sql +peers (id, url, name, enabled, pull_interval, + trust_abuse_retractions, advertise, peered_since, + last_cursor, last_pull_at, last_error) + +-- manifests gains: +-- content_id text unique -- sha256 over canonical form +-- origin text -- server_id of first acceptance +-- ingested_from text NULL -- peer id, NULL if uploaded directly +``` + +`content_id` carries a unique index and is the deduplication key on ingest. + +### What is deliberately not specified + +- **No automatic peering.** A published peer directory (above) is readable by + humans; it is never acted on by software. No crawling, no transitive + peering, no "trusted because a peer trusts them". +- **No consensus.** There is no global agreement on what the catalogue + contains. Each server's catalogue is its own; federation only makes it + cheaper to fill. +- **No global identity.** No shared contributor identity across servers. + Tokens stay local, consistent with §5a — a contributor's standing on one + server means nothing on another, and needs to mean nothing. +- **No deletion propagation** beyond the opt-in abuse channel above. + +--- + +## 10. Open questions + +1. **Is `runtime` tier good enough as the default?** ±2s will match most + same-cut releases but will also match a different encode with identical + runtime and a different logo trim. Leaning yes-with-caveat-in-UI. +2. **Should manifests be signed by the contributor?** Adds provenance but + also key management for hobbyist operators. Probably not for v1. +3. **Should the gallery be shareable too?** Embeddings are far larger than + manifests and are derived from copyrighted headshots; out of scope here, + but worth a separate look — it would remove the biggest setup cost for new + users. +4. **Federation** is now specified in §9a. Open sub-questions: + - **Is re-running the TMDB cast check on every ingested manifest + affordable?** Bulk-ingesting a large peer catalogue means a TMDB lookup + per distinct title. The 24h credits cache and the fact that lookups are + per *title* (not per manifest) should make it fine, but a bootstrap of + tens of thousands of titles needs a throttled backfill mode rather than + the normal upload path. + - **Should a fresh server be allowed to trust a peer's `cast_match_ratio` + during initial bootstrap only?** It would make standing up a mirror far + cheaper, at the cost of the guarantee in §9a. Leaning no. + - ~~Float canonicalisation stability~~ — **resolved.** Checked against + 28972 scene values from the corpus: 3dp rounding is safe today (max error + 3.3e-4, no boundary cases) but fragile, since pipeline timings accumulate + float error from summing `1/fps`. §9a now quantises to integer + centiseconds, which removes the failure mode rather than relying on + luck. +5. **Is 0.6 the right cast-match threshold?** Still a guess, but now a + *testable* one: the extraction repo has 331 real output files. Running the + §6 stage-3 check over them against TMDB would yield the true distribution + of honest-upload match ratios and let the threshold be set at, say, the 1st + percentile rather than by intuition. This is the single cheapest way to + de-risk UR-003 and UR-005 and should happen before launch. Note that the corpus + is heavily TV-weighted, so movie and episode thresholds may need to differ. +6. **Discarding the uploaded `name` string (§5a) depends on TMDB person + resolution being reliable.** If too many legitimate actors fail to resolve, + manifests lose actors silently. The corpus run in (5) measures this too. If + resolution proves lossy, the fallback is to store names but restricted to + the closed character class — weaker, but still not a usable payload channel. +7. **Anonymous existence checks are a title-availability oracle.** Rate limits + blunt this but do not remove it. Requiring a token for `exists` would close + it at the cost of making read-only use non-anonymous. Left open. +8. **Adult-content classification for the §5a category guard** relies on + TMDB's `adult` flag, whose coverage for *performers* is less consistent + than for titles. The guard may need a supplementary signal. diff --git a/deny.toml b/deny.toml new file mode 100644 index 0000000..4663bd3 --- /dev/null +++ b/deny.toml @@ -0,0 +1,91 @@ +# cargo-deny configuration. +# +# This crate is GPLv3 (it shares a licence with the JRay Jellyfin plugin), and it +# is a long-lived network service whose main risks are hostile input and operator +# friction (§8). So two checks matter most here: +# +# - `advisories` — a public-facing service must not ship known-vulnerable +# dependencies. +# - `licenses` — GPLv3 is compatible with permissive licences, but *not* with +# everything. A copyleft-incompatible dependency arriving transitively would +# be a licensing problem discovered far too late. +# +# Run with `cargo deny check`. + +[graph] +# Check the targets an operator actually deploys. §8 ships a single static binary +# (musl target), so both glibc and musl Linux are in scope. +targets = [ + "x86_64-unknown-linux-gnu", + "x86_64-unknown-linux-musl", + "aarch64-unknown-linux-gnu", + "aarch64-unknown-linux-musl", +] +all-features = true + +[advisories] +version = 2 +# Fail on any RustSec advisory. Unmaintained crates are a warning rather than an +# error: `sled` was rejected in §8 partly on maintenance grounds, so the signal is +# worth surfacing, but it should not break a build on its own. +yanked = "deny" +unmaintained = "workspace" +ignore = [] + +[licenses] +version = 2 +# Permissive licences, all GPLv3-compatible. Deliberately a closed allow-list +# rather than a deny-list: a licence nobody vetted should stop the build, in the +# same spirit as §6's "no additional fields anywhere". +# +# Kept to licences actually present in the tree, so `cargo deny` stays quiet in +# CI and an added allowance is a visible decision. Adding a dependency that needs +# a new licence should be a deliberate edit here. +allow = [ + "Apache-2.0", + "MIT", + "BSD-2-Clause", + "BSD-3-Clause", + "ISC", + "Zlib", + "Unicode-3.0", + # `webpki-roots` — Mozilla's trusted CA certificate set. This is a *data* + # licence, not a code licence, which is why it is not on the usual permissive + # list: the crate ships certificates rather than logic. CDLA-Permissive-2.0 + # imposes no copyleft and no attribution burden on a binary that embeds it, so + # it is compatible with distributing this server under GPLv3. + # + # It arrives via reqwest's rustls stack, which §8's single static musl binary + # depends on (bundling roots is what lets the binary verify TLS without a + # system trust store). + "CDLA-Permissive-2.0", + # This crate's own licence. + "GPL-3.0-or-later", +] +confidence-threshold = 0.9 +# `ring` ships a bespoke licence file that no SPDX expression describes; it is +# a permissive OpenSSL/ISC-style licence and is GPL-compatible. Clarify it rather +# than widening the allow-list. +[[licenses.clarify]] +crate = "ring" +expression = "MIT AND ISC AND OpenSSL" +license-files = [{ path = "LICENSE", hash = 0xbd0eed23 }] + +[bans] +multiple-versions = "warn" +wildcards = "deny" +# Nothing is banned outright yet. The obvious future entries are alternative TLS +# stacks: reqwest is pinned to rustls (`default-features = false`) so that a +# static musl binary needs no system OpenSSL, and an accidental openssl-sys +# dependency would silently break that deployment story. +deny = [] +skip = [] +skip-tree = [] + +[sources] +unknown-registry = "deny" +unknown-git = "deny" +# Only crates.io. A git dependency in a service that hobbyist operators build +# from source is a supply-chain and reproducibility problem. +allow-registry = ["https://github.com/rust-lang/crates.io-index"] +allow-git = [] diff --git a/docs/requirements.md b/docs/requirements.md new file mode 100644 index 0000000..7f525a9 --- /dev/null +++ b/docs/requirements.md @@ -0,0 +1,200 @@ +# JRay-public-server — requirements register + +Stable IDs for every requirement in [`../SPEC.md`](../SPEC.md), which holds the +prose. This file is the **authoritative list**; the CI gate reads its +denominators from here (see the [system spec](../../SPEC.md) §6). + +**IDs are permanent.** A withdrawn requirement is marked `Withdrawn` and its +number is never reused — renumbering is what produces orphan TRACES tags. + +Tag code with `// TRACES: UR-003 | SR-004`. + +| Type | Scope | +|---|---| +| `UR` | User/functional — what the server does | +| `DR` | Development — how it is built and operated | +| `UT` / `IT` | Unit / integration tests | + +Status: `Done` · `In Progress` · `Planned` · `TBD` · `Withdrawn` + +A requirement is `Done` only when it is implemented **and** has a test that +executes. Everything below runs in CI on any machine — this repo has no GPU +requirement and no fixture-generation step, unlike `scene-actor-extraction`. + +--- + +## User requirements (UR) + +| ID | Requirement | Traces to | Priority | Status | +|---|---|---|---|---| +| UR-001 | Cheap existence probe, separate from the fetch, returning availability and cut-match tier without payload | SR-001 | High | Done | +| UR-002 | Accept a contributed manifest for a media item | PR-006 | High | Done | +| UR-003 | Content verification: strict schema, size caps, approximate TMDB cast match | SR-004 | High | Done | +| UR-004 | Rate limiting, per token where present and per source IP otherwise | SR-004 | High | Done | +| UR-005 | Trust without accounts: not usable as a content store, nor for prank manifests | SR-004 | High | Done | +| UR-006 | Serve and accept a whole series in one operation | PR-006 | High | Done | +| UR-007 | Plugin queries an ordered, configurable list of servers | PR-005 | High | In Progress | +| UR-008 | Servers replicate manifests between each other | PR-006 | Medium | Planned | +| UR-009 | Store an audio spectral-peak signature for content-based identification | SR-003 | Medium | In Progress | +| UR-010 | Identity crossing the API boundary is TMDB/IMDB ids, never a name alone | SR-001 | High | Done | +| UR-011 | Reject any field capable of carrying binary or attacker-chosen content | SR-004 | High | Done | +| UR-012 | Never accept, store, or serve gallery data — reference faces or embeddings | SR-005 | High | Done | +| UR-013 | Windows are scene-scoped claims; never reinterpret their boundaries | SR-002 | High | Done | +| UR-014 | Reject an unknown `jmanifest_version` outright, never guess | SR-003 | High | Done | + +### Notes on status + +**UR-007 is `In Progress`, not `Done`.** The plugin now carries the ordered +server list and its per-server trust settings, with the community instance +pre-configured but disabled. The fetch path that consumes it does not exist yet. + +**UR-009 is `In Progress`.** The server accepts, validates and stores +`cut.audio_signature`, and `content_id` correctly excludes it (§9a). What is +absent is `audio`-tier matching and `POST /manifests/search`. This is the +sequencing §3 recommends — accumulate signatures first, enable matching once +coverage is useful — not an oversight. + +**UR-012 is satisfied structurally, by absence.** There is no field in the +Jmanifest capable of carrying an embedding or a crop, and no endpoint that would +accept one. Like PR-005 in the system spec, it cannot be verified by pointing at +code that does something; UT-024 verifies it by asserting that the obvious +attempts are rejected. + +--- + +## Development requirements (DR) + +| ID | Requirement | Traces to | Priority | Status | +|---|---|---|---|---| +| DR-001 | Strict parse boundary: unknown fields rejected structurally, not by validator code | SR-004 | High | Done | +| DR-002 | Fully relational storage — no JSON blob on the write path | SR-004 | High | Done | +| DR-003 | Single serialized writer connection, with a read pool alongside | PR-004 | High | Done | +| DR-004 | All database access behind a repository layer, not scattered through handlers | PR-004 | Medium | Done | +| DR-005 | Background work in-process, with the job queue as a table so it survives restart | PR-004 | High | Done | +| DR-006 | Rate-limit counters in process memory; no external counter store | PR-004 | Medium | Done | +| DR-007 | Ship a single static binary plus one database file; container optional | PR-004 | High | Done | +| DR-008 | `X-Forwarded-For` honoured only from explicitly configured proxies | SR-004 | High | Done | +| DR-009 | Body caps enforced while streaming, before parsing, per route | SR-004 | High | Done | +| DR-010 | Request bodies are UTF-8 only, rejected with a diagnosable error otherwise | SR-003 | Medium | Done | +| DR-011 | `content_id` canonical form is byte-stable and cross-implementation tested | SR-003 | High | Done | +| DR-012 | Dependency audit: advisories, licence policy, source policy | PR-004 | Medium | Done | +| DR-013 | API errors use the status codes the spec names, not the framework's defaults | SR-003 | Medium | Done | +| DR-014 | Portable SQL — no SQLite-specific form where a standard one exists | PR-004 | Medium | Done | + +--- + +## Verification + +**No GPU, no fixtures, no external services.** Every test here runs on any +machine in under three seconds. The TMDB dependency is the only external service, +and it is absent from tests by construction: an unconfigured client makes uploads +stay `pending`, which is the correct production failure mode (§8) and happens to +make the test suite hermetic. + +| Tier | Runs in CI | What it covers | +|---|---|---| +| **T1 — unit** | Yes | Pure logic: validation, cut matching, cast-check scoring, canonicalisation, rate limiting | +| **T2 — integration** | Yes | End-to-end through the real router against a temporary on-disk database | + +There is no tier that does not run. A requirement here is either verified or +visibly not. + +**Integration tests use an on-disk temporary database, not `:memory:`.** DR-003 +specifies one writer connection plus a read pool, and in-memory SQLite is +per-connection — the readers would see an empty database. Testing the real +topology is the point, so this is a deliberate choice rather than an oversight. + +### Per-requirement verification + +| ID | Tier | Test asserts | Edge cases covered | +|---|---|---|---| +| UR-001 | T2 | `exists` reports availability and tier without payload | Absent title returns `200` with `false`, not `404`; batch form is positional; one bad item does not fail the batch | +| UR-002 | T2 | Valid upload accepted as `202 pending` | Duplicate content deduplicates; same contributor resubmitting the same cut is `409` | +| UR-003 | T1 + T2 | Strict schema, caps, and cast-match thresholds | Ratio boundaries at 0.6 and 0.3 exactly; small-\|M\| all-but-one rule; missing TMDB credits flags rather than rejects | +| UR-004 | T1 + T2 | Limits engage and carry the documented headers | Window reset; a rejected request does not extend its own lockout; surfaces have independent budgets | +| UR-005 | T1 + T2 | Prank manifests rejected; no free-text channel | Uncredited cast rejected; name-only matches capped; automatic revocation needs a minimum sample | +| UR-006 | T2 | Bundle accepted per-episode, non-atomically | One bad episode rejected while its neighbours are accepted; envelope errors are whole-request `400` | +| UR-007 | — | *No server-side test.* Plugin-side; the register there will carry it | — | +| UR-009 | T1 | Signature structurally validated | Fixed length; reserved high bit; **media < 120 s must send no signature at all** | +| UR-010 | T1 + T2 | Actors persist as TMDB person ids | A name the upload invented does not round-trip | +| UR-011 | T2 | Every payload-shaped field rejected | base64, hex, markup, control characters, bidi overrides, compatibility homoglyphs | +| UR-012 | T2 | No endpoint accepts embeddings or image data | An `embedding` or `crop` field is an unknown-field `400` | +| UR-013 | T1 | Stored windows are byte-identical to those submitted | Adjacent windows never merged; a window is never trimmed to a shorter one | +| UR-014 | T1 | Unknown `jmanifest_version` rejected | Version `2` and version `0` both refused, naming the field | +| DR-001 | T1 | Unknown field at any nesting depth fails to parse | `movie` and `jellyfin_id` named in the error | +| DR-003 | T1 | Concurrent writes serialize rather than returning `SQLITE_BUSY` | Failed transaction rolls back fully | +| DR-005 | T1 | Jobs lease once, reschedule with backoff, survive restart | Stranded lease released at startup; future job not leased early | +| DR-008 | T2 | Forged `X-Forwarded-For` cannot mint a fresh budget | Untrusted peer ignored; trusted proxy honoured; client-supplied entries to the left cannot spoof | +| DR-009 | T2 | Oversized body rejected as `413` | **A lying `Content-Length` does not bypass the cap**; per-route limits differ | +| DR-010 | T1 | Non-UTF-8 rejected by name | UTF-16 with and without BOM; UTF-8 BOM; declared `charset=utf-16` | +| DR-011 | T1 | Canonical form is stable and order-independent | Accumulated float error hashes identically; `audio_signature` and `extraction` excluded; **golden vector verified against an independent Python implementation** | +| DR-013 | T1 | Schema mismatch is `400`, not the framework's `422` | §4 names `400` for a forbidden field, and a client checking for it would mishandle `422` | + +Three are worth singling out, because each verifies a claim that would otherwise +be an assertion: + +- **DR-009's lying-`Content-Length` case.** §6 stage 0 is explicit that the + header is a claim by the client, so the streaming cap is mandatory rather than + redundant. A test that only sends honest bodies verifies nothing. +- **DR-011's golden vector.** Two servers that validated the same upload must + reach the same `content_id`, and the plugin must reproduce it byte-identically + from a different language. The fixture is the only thing that can catch + divergence before it silently breaks federation deduplication. +- **UR-011's homoglyph case.** Writing this test found a real gap: compatibility + variants (`𝐒𝐭𝐞𝐯𝐞`, `Actor`) are letters by Unicode category and NFC does not + fold them, so a fullwidth-digit alphabet would have reopened the encoding + channel §5a's "no digits" rule closes. + +--- + +## Withdrawn + +| ID | Requirement | Reason | +|---|---|---| +| — | `anneal_sec` in `extraction` | Withdrawn upstream (`scene-actor-extraction` AR-012/AR-013): presence follows track extent, so a track survives its own gaps and there is nothing to anneal. Ships as part of the SR-003 bump | + +Deleted rather than retained at zero: a field naming a mechanism the pipeline no +longer has is actively misleading to anyone reading a manifest, and would outlive +everyone who remembers why it is zero. + +No `UR`/`DR` number was ever assigned to it — it was a *field*, not a +requirement — so nothing is orphaned by its removal. + +--- + +## Pending — the SR-003 schema bump + +These are `Planned` rather than absent, because the bump is coordinated across +three repos and this register should show the work rather than imply the server +is finished. + +| ID | Requirement | Traces to | Priority | Status | +|---|---|---|---|---| +| UR-015 | Accept `extraction.extinction_sec` in place of `anneal_sec` | SR-003 | High | Planned | +| UR-016 | Accept and store `extraction.gallery_scope`; rank on it (§7) | SR-003 | Medium | Planned | +| UR-017 | Accept per-window belief and identification route; `scenes` becomes objects | SR-003 | High | Planned | +| UR-018 | Exclude belief from `content_id`, replicating it as an attribute | SR-003 | High | Planned | + +**UR-018 is the one with a trap in it.** Belief is a producer-side estimate that +may legitimately differ between pipeline versions for identical timings, so +including it in the canonical form would give two servers different `content_id`s +for the same content — the exact failure mode §9a quantises centiseconds to +avoid. It follows `audio_signature`'s precedent: replicated as an attribute, not +part of identity. + +--- + +## Notes on coverage + +- **`DR-*` traces to `PR-004` (self-hosted) more often than to an `SR-nnn`.** + Operational simplicity is a single-repo concern serving the project goal + directly. This is correct rather than a gap: §8's whole argument for one binary + and one file is that federation only works if running an instance is easy. +- **UR-007 has no server-side test** and cannot have one — it is a requirement on + the plugin, recorded here because this spec is where it is stated. It should be + cross-referenced from the plugin's register when that is created, and until + then it is visibly unverified rather than quietly assumed. +- **UR-012 and UR-013 are preserved by prohibition**, like PR-005 in the system + spec. They cannot be verified by pointing at code that does something, only by + asserting that the attempts fail. They die the moment either prohibition is + relaxed, which is precisely why they are stated rather than left implicit. diff --git a/rustfmt.toml b/rustfmt.toml new file mode 100644 index 0000000..3714eb1 --- /dev/null +++ b/rustfmt.toml @@ -0,0 +1,8 @@ +# Formatting matches the style the codebase is written in. +# +# `use_small_heuristics = "Max"` keeps short structs, calls and match arms on one +# line rather than exploding them across four. In a codebase this dense with spec +# citations, vertical space spent on punctuation is space not spent on the comment +# explaining *why* a rule exists. +max_width = 100 +use_small_heuristics = "Max" diff --git a/src/api/exists.rs b/src/api/exists.rs new file mode 100644 index 0000000..22fb70c --- /dev/null +++ b/src/api/exists.rs @@ -0,0 +1,187 @@ +//! `GET /manifests/exists` and its batch form — UR-1. +//! +//! Deliberately a *separate, cheaper* endpoint from the fetch: it answers +//! "should I bother?" for a whole library sweep without transferring payloads, +//! and it is the endpoint a scheduled task will hammer. It is also the most +//! abuse-prone surface, since it doubles as an oracle for "does the community +//! have this title" — so it is rate-limited harder than the fetches and returns +//! no manifest content (§0). + +use axum::extract::{Query, State}; +use axum::http::HeaderMap; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::{Deserialize, Serialize}; + +use super::LookupParams; +use crate::db::repo; +use crate::error::{ApiError, ApiResult}; +use crate::matching::{self, StoredCut}; +use crate::model::{IdentityType, MatchTier}; +use crate::ratelimit::Surface; +use crate::state::{with_quota_headers, AppState}; + +/// §4: no manifest content, just availability and tier. +#[derive(Debug, Clone, Serialize)] +pub struct ExistsResponse { + pub exists: bool, + #[serde(skip_serializing_if = "Option::is_none")] + pub r#match: Option<&'static str>, + #[serde(skip_serializing_if = "Option::is_none")] + pub manifest_id: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub actor_count: Option, +} + +impl ExistsResponse { + fn absent() -> Self { + Self { exists: false, r#match: None, manifest_id: None, actor_count: None } + } +} + +/// §4 batch form: up to 100 items. +/// +/// Exists specifically so the §5 rate limit can be generous per *request* while +/// staying strict per *item*, and so a 2000-item library sweep is 20 requests +/// rather than 2000. +pub const MAX_BATCH_ITEMS: usize = 100; + +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BatchRequest { + pub items: Vec, +} + +#[derive(Debug, Serialize)] +pub struct BatchResponse { + /// Positional, matching the request order (§4). + pub results: Vec, +} + +pub async fn exists( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Query(params): Query, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ExistsSingle)?; + let body = lookup_one(&state, ¶ms).await?; + Ok(with_quota_headers(Json(body).into_response(), quota)) +} + +pub async fn exists_batch( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + super::json::Json(req): super::json::Json, +) -> ApiResult { + if req.items.len() > MAX_BATCH_ITEMS { + return Err(ApiError::BadRequest(format!( + "items: at most {MAX_BATCH_ITEMS} per request, got {}", + req.items.len() + ))); + } + if req.items.is_empty() { + return Err(ApiError::BadRequest("items: must not be empty".into())); + } + + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ExistsBatch)?; + + let mut results = Vec::with_capacity(req.items.len()); + for item in &req.items { + // A malformed item yields "absent" rather than failing the whole batch — + // a sweep of 100 items should not be lost to one bad entry. + results.push(lookup_one(&state, item).await.unwrap_or_else(|_| ExistsResponse::absent())); + } + + Ok(with_quota_headers(Json(BatchResponse { results }).into_response(), quota)) +} + +async fn lookup_one(state: &AppState, params: &LookupParams) -> ApiResult { + let Some((kind, tmdb_id, imdb_id)) = resolve_kind(params) else { + return Err(ApiError::BadRequest( + "requires tmdb_id/imdb_id, or series_tmdb_id with season and episode".into(), + )); + }; + + let (season, episode) = match kind { + IdentityType::Movie => (None, None), + IdentityType::Episode => (params.season, params.episode), + }; + let client_cut = params.client_cut(); + + let found = state + .db + .read(move |conn| { + let Some(title) = repo::find_title(conn, kind, tmdb_id.as_deref(), imdb_id.as_deref())? + else { + return Ok(None); + }; + let candidates = repo::candidates_for_title(conn, &title.id, season, episode)?; + if candidates.is_empty() { + return Ok(None); + } + + let cuts: Vec<(String, StoredCut)> = candidates + .iter() + .map(|m| { + ( + m.id.clone(), + StoredCut { runtime_sec: m.runtime_sec, video_hash: m.video_hash.clone() }, + ) + }) + .collect(); + + let Some((id, m)) = matching::best_match(&client_cut, &cuts) else { + return Ok(None); + }; + let actor_count = repo::manifest_actor_ids(conn, &id)?.len() as i64; + Ok(Some((id, m.tier, actor_count))) + }) + .await + .map_err(ApiError::Internal)?; + + // §4: `exists: false` is returned with `200`, not `404` — absence is a normal + // answer to this question, and `404` would conflate "no manifest" with "bad + // route" for the client. + Ok(match found { + Some((id, tier, actor_count)) => ExistsResponse { + exists: true, + // With no cut parameters the answer is "some manifest exists" with + // `"match": "unknown"`; the client must still fetch to find out + // whether a cut aligns. This is the mode a library sweep uses (§4). + r#match: Some(tier.as_str()), + manifest_id: Some(id), + actor_count: Some(actor_count), + }, + None => ExistsResponse::absent(), + }) +} + +/// Determines whether these parameters address a movie or an episode. +pub fn resolve_kind( + params: &LookupParams, +) -> Option<(IdentityType, Option, Option)> { + if params.series_tmdb_id.is_some() || params.series_imdb_id.is_some() { + // Episode coordinates are required alongside series identity; without + // them the caller wants the series bundle endpoint instead. + params.season?; + params.episode?; + return Some(( + IdentityType::Episode, + params.series_tmdb_id.clone(), + params.series_imdb_id.clone(), + )); + } + if params.tmdb_id.is_some() || params.imdb_id.is_some() { + return Some((IdentityType::Movie, params.tmdb_id.clone(), params.imdb_id.clone())); + } + None +} + +/// Exposed for tests asserting the documented tier string. +pub fn tier_str(t: MatchTier) -> &'static str { + t.as_str() +} diff --git a/src/api/fetch.rs b/src/api/fetch.rs new file mode 100644 index 0000000..48b8ea3 --- /dev/null +++ b/src/api/fetch.rs @@ -0,0 +1,351 @@ +//! Manifest fetch endpoints (§4). +//! +//! §7: the submitted JSON was parsed, validated, resolved to TMDB person ids, +//! written as rows and discarded. Everything served here is **reconstructed** +//! from those rows, never echoed — which is what makes §5a's Threat 1 defence +//! structural rather than a promise. + +use axum::extract::{Path, Query, State}; +use axum::http::HeaderMap; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::Serialize; + +use super::LookupParams; +use crate::db::repo::{self, ManifestRow}; +use crate::error::{ApiError, ApiResult}; +use crate::matching::{self, StoredCut}; +use crate::model::{ + Actor, Coverage, Cut, Extraction, GalleryScope, Identity, IdentityType, Jmanifest, MatchTier, + SeriesBundle, SeriesRef, JMANIFEST_VERSION, +}; +use crate::ratelimit::Surface; +use crate::state::{with_quota_headers, AppState}; + +#[derive(Debug, Serialize)] +pub struct FetchResponse { + pub r#match: &'static str, + /// Scene offset the client must add (§3). Zero for the tiers currently + /// served; present unconditionally so the plugin contract does not change + /// when `audio` is enabled. + pub offset_sec: f64, + pub manifest: Jmanifest, +} + +#[derive(Debug, Serialize)] +pub struct StatusResponse { + pub status: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub reason: Option, +} + +pub async fn get_movie( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Query(params): Query, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ManifestFetch)?; + + if params.tmdb_id.is_none() && params.imdb_id.is_none() { + return Err(ApiError::BadRequest("requires tmdb_id or imdb_id".into())); + } + let body = fetch_best(&state, IdentityType::Movie, ¶ms, None, None).await?; + Ok(with_quota_headers(Json(body).into_response(), quota)) +} + +pub async fn get_episode( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Query(params): Query, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ManifestFetch)?; + + if params.series_tmdb_id.is_none() && params.series_imdb_id.is_none() { + return Err(ApiError::BadRequest("requires series_tmdb_id or series_imdb_id".into())); + } + let (Some(season), Some(episode)) = (params.season, params.episode) else { + return Err(ApiError::BadRequest("requires season and episode".into())); + }; + let body = + fetch_best(&state, IdentityType::Episode, ¶ms, Some(season), Some(episode)).await?; + Ok(with_quota_headers(Json(body).into_response(), quota)) +} + +async fn fetch_best( + state: &AppState, + kind: IdentityType, + params: &LookupParams, + season: Option, + episode: Option, +) -> ApiResult { + let (tmdb_id, imdb_id) = match kind { + IdentityType::Movie => (params.tmdb_id.clone(), params.imdb_id.clone()), + IdentityType::Episode => (params.series_tmdb_id.clone(), params.series_imdb_id.clone()), + }; + let client_cut = params.client_cut(); + + let found = state + .db + .read(move |conn| { + let Some(title) = repo::find_title(conn, kind, tmdb_id.as_deref(), imdb_id.as_deref())? + else { + return Ok(None); + }; + let candidates = repo::candidates_for_title(conn, &title.id, season, episode)?; + let cuts: Vec<(ManifestRow, StoredCut)> = candidates + .into_iter() + .map(|m| { + let cut = + StoredCut { runtime_sec: m.runtime_sec, video_hash: m.video_hash.clone() }; + (m, cut) + }) + .collect(); + + let Some((row, m)) = matching::best_match(&client_cut, &cuts) else { + return Ok(None); + }; + let manifest = reconstruct(conn, &row, &title, kind)?; + Ok(Some((m.tier, m.offset_sec, manifest))) + }) + .await + .map_err(ApiError::Internal)?; + + // §4: `404` if none clears `loose`. + let (tier, offset_sec, manifest) = found.ok_or(ApiError::NotFound)?; + Ok(FetchResponse { r#match: tier.as_str(), offset_sec, manifest }) +} + +/// `GET /manifests/series/{series_tmdb_id}?season=` (§4). +/// +/// Returns whatever episodes the server holds. **Partial bundles are normal** — a +/// bundle with 9 of 13 episodes is a valid, useful response, not an error (§2). +/// Episode-level cut matching is done client-side against the returned bundle, +/// since a client pulling a whole series already knows its own runtimes. +pub async fn get_series( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Path(series_tmdb_id): Path, + Query(params): Query, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::SeriesFetch)?; + + let season = params.season; + let bundle = state + .db + .read(move |conn| { + let Some(title) = + repo::find_title(conn, IdentityType::Episode, Some(&series_tmdb_id), None)? + else { + return Ok(None); + }; + let rows = repo::episodes_for_series(conn, &title.id, season)?; + + // Multiple contributors may hold the same episode; `episodes_for_series` + // orders by rank, so keep the first per (season, episode). + let mut episodes: Vec = Vec::new(); + let mut seen: Vec<(i64, i64)> = Vec::new(); + let mut seasons: Vec = Vec::new(); + + for row in rows { + let key = (row.season.unwrap_or(-1), row.episode.unwrap_or(-1)); + if seen.contains(&key) { + continue; + } + seen.push(key); + if !seasons.contains(&key.0) { + seasons.push(key.0); + } + episodes.push(reconstruct(conn, &row, &title, IdentityType::Episode)?); + } + seasons.sort_unstable(); + + Ok(Some(SeriesBundle { + jmanifest_version: JMANIFEST_VERSION, + series: SeriesRef { + series_tmdb_id: title.tmdb_id.clone(), + series_imdb_id: title.imdb_id.clone(), + title: title.name.clone(), + }, + coverage: Some(Coverage { episodes_available: episodes.len(), seasons }), + episodes, + })) + }) + .await + .map_err(ApiError::Internal)?; + + let bundle = bundle.filter(|b| !b.episodes.is_empty()).ok_or(ApiError::NotFound)?; + Ok(with_quota_headers(Json(bundle).into_response(), quota)) +} + +/// `GET /manifests/{id}` — fetch a specific manifest by its server-assigned id, +/// for debugging and for the "report this manifest" flow (§4). +pub async fn get_by_id( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Path(id): Path, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::ManifestFetch)?; + + let manifest = state + .db + .read(move |conn| { + let Some(row) = repo::manifest_by_id(conn, &id)? else { return Ok(None) }; + // Unlisted manifests are not served to anyone (§6 stage 3). + if row.status != "listed" && row.status != "flagged" { + return Ok(None); + } + let title = title_of(conn, &row.title_id)?; + let kind = + if title.kind == "movie" { IdentityType::Movie } else { IdentityType::Episode }; + Ok(Some(reconstruct(conn, &row, &title, kind)?)) + }) + .await + .map_err(ApiError::Internal)?; + + let manifest = manifest.ok_or(ApiError::NotFound)?; + Ok(with_quota_headers(Json(manifest).into_response(), quota)) +} + +/// `GET /manifests/{id}/status` — poll the outcome of the asynchronous cast +/// check (§4). +pub async fn get_status( + State(state): State, + Path(id): Path, +) -> ApiResult> { + let found = state + .db + .read(move |conn| repo::manifest_status(conn, &id)) + .await + .map_err(ApiError::Internal)?; + + match found { + Some((status, reason)) => Ok(Json(StatusResponse { status, reason })), + // §6 deletes rejected manifests, so a vanished id is reported as + // rejected rather than as a bad route. + None => Ok(Json(StatusResponse { + status: "rejected".into(), + reason: Some("not_found_or_rejected".into()), + })), + } +} + +fn title_of(conn: &rusqlite::Connection, title_id: &str) -> anyhow::Result { + let row = conn.query_row( + "SELECT id, kind, tmdb_id, imdb_id, name, year, adult, certification + FROM titles WHERE id = ?1", + rusqlite::params![title_id], + |r| { + Ok(repo::TitleRow { + id: r.get(0)?, + kind: r.get(1)?, + tmdb_id: r.get(2)?, + imdb_id: r.get(3)?, + name: r.get(4)?, + year: r.get(5)?, + adult: r.get::<_, i64>(6)? != 0, + certification: r.get(7)?, + }) + }, + )?; + Ok(row) +} + +/// Rebuilds a Jmanifest from stored rows. +/// +/// Names come from `people` — populated from TMDB by the server — so `name` is +/// server-authoritative on download and a name a contributor invented does not +/// round-trip (§2, §5a). +pub fn reconstruct( + conn: &rusqlite::Connection, + row: &ManifestRow, + title: &repo::TitleRow, + kind: IdentityType, +) -> anyhow::Result { + let stored = repo::actors_for_manifest(conn, &row.id)?; + + let actors = stored + .into_iter() + .map(|a| Actor { + name: a.name, + imdb_id: None, + tmdb_id: Some(a.tmdb_person_id.to_string()), + scenes: a + .scenes_cs + .into_iter() + .map(|(s, e)| [s as f64 / 100.0, e as f64 / 100.0]) + .collect(), + }) + .collect(); + + let identity = match kind { + IdentityType::Movie => Identity { + kind, + tmdb_id: title.tmdb_id.clone(), + imdb_id: title.imdb_id.clone(), + series_tmdb_id: None, + series_imdb_id: None, + season: None, + episode: None, + title: title.name.clone(), + year: title.year, + }, + IdentityType::Episode => Identity { + kind, + tmdb_id: None, + imdb_id: None, + series_tmdb_id: title.tmdb_id.clone(), + series_imdb_id: title.imdb_id.clone(), + season: row.season, + episode: row.episode, + title: title.name.clone(), + year: title.year, + }, + }; + + // An unrecognised stored scope is served as absent rather than guessed at: + // the column is written from a closed enum, so anything else means the row + // predates a schema change and its meaning is unknown (UR-014's spirit). + let gallery_scope = match row.gallery_scope.as_deref() { + Some("global") => Some(GalleryScope::Global), + Some("limited") => Some(GalleryScope::Limited), + _ => None, + }; + + let extraction = Extraction { + sample_fps: row.sample_fps, + extinction_sec: row.extinction_sec, + pipeline_version: row.pipeline_version.clone(), + gallery_size: None, + gallery_scope, + }; + let has_extraction = extraction.sample_fps.is_some() + || extraction.extinction_sec.is_some() + || extraction.pipeline_version.is_some() + || extraction.gallery_scope.is_some(); + + Ok(Jmanifest { + jmanifest_version: JMANIFEST_VERSION, + identity, + cut: Cut { + runtime_sec: row.runtime_sec, + container_duration_sec: None, + video_hash: row.video_hash.clone(), + audio_signature: None, + }, + extraction: has_extraction.then_some(extraction), + actors, + }) +} + +/// Exposed so tests can assert the served tier strings. +pub fn tier_name(t: MatchTier) -> &'static str { + t.as_str() +} diff --git a/src/api/json.rs b/src/api/json.rs new file mode 100644 index 0000000..ba3be86 --- /dev/null +++ b/src/api/json.rs @@ -0,0 +1,287 @@ +//! A JSON extractor that fails with the status codes §4 specifies. +//! +//! Axum's own `Json` rejects a body that parses as JSON but does not match the +//! target type with **422 Unprocessable Entity**. §4 is explicit that this case +//! is **`400`** — "malformed, or contains an unrecognised or forbidden field" — +//! and that distinction is load-bearing: §6 requires that a client which forgets +//! to strip `movie` or `jellyfin_id` gets "a hard `400` naming the offending +//! field". A client checking for 400 would mishandle a 422. +//! +//! This wrapper also guarantees the field name reaches the caller, since serde's +//! `deny_unknown_fields` error text is what identifies the offending key. + +use axum::extract::{FromRequest, Request}; +use axum::http::header::CONTENT_TYPE; + +use crate::error::ApiError; + +/// Drop-in replacement for `axum::Json` on request bodies. +pub struct Json(pub T); + +impl FromRequest for Json +where + T: serde::de::DeserializeOwned, + S: Send + Sync, +{ + type Rejection = ApiError; + + async fn from_request(req: Request, state: &S) -> Result { + // A wrong content type is the client's mistake, reported as such rather + // than as a parse failure. + let content_type = + req.headers().get(CONTENT_TYPE).and_then(|v| v.to_str().ok()).unwrap_or("").to_string(); + + let mime = content_type.split(';').next().unwrap_or("").trim().to_ascii_lowercase(); + if !(mime == "application/json" || mime.ends_with("+json")) { + return Err(ApiError::BadRequest("expected content-type: application/json".into())); + } + + // A declared charset other than UTF-8 is refused up front, so the client + // learns what is wrong rather than receiving a confusing parse error from + // deep inside the document. See `require_utf8` for why UTF-8 is the only + // accepted encoding. + if let Some(charset) = + content_type.split(';').skip(1).filter_map(|p| p.trim().strip_prefix("charset=")).next() + { + let charset = charset.trim().trim_matches('"').to_ascii_lowercase(); + if !matches!(charset.as_str(), "utf-8" | "utf8") { + return Err(ApiError::BadRequest(format!( + "unsupported charset {charset:?}: JSON must be UTF-8 encoded (RFC 8259 §8.1)" + ))); + } + } + + let bytes = axum::body::Bytes::from_request(req, state).await.map_err(|e| { + // §6 stage 1: the body cap aborts mid-transfer, and that must surface + // as `413`, not as a generic parse error. Axum folds the length-limit + // case into `FailedToBufferBody`, so the status it chose is the + // reliable discriminator. + if e.status() == axum::http::StatusCode::PAYLOAD_TOO_LARGE { + ApiError::PayloadTooLarge("request body exceeds the limit for this route".into()) + } else { + ApiError::BadRequest(format!("could not read request body: {e}")) + } + })?; + + // Encoding is checked before parsing, so a mis-encoded body gets an + // actionable message instead of whatever the parser happens to trip over. + let text = require_utf8(&bytes)?; + + serde_json::from_str(text) + .map(Json) + // serde's message names the offending field, which is exactly what §6 + // requires the response to identify. + .map_err(|e| ApiError::BadRequest(e.to_string())) + } +} + +/// Enforces that the body is UTF-8, naming the encoding it appears to be. +/// +/// **UTF-8 is the only accepted encoding, deliberately.** RFC 8259 §8.1 requires +/// it for JSON exchanged outside a closed ecosystem, and this is a public, +/// federated API. Three further reasons make it the right call *here* +/// specifically, rather than merely conventional: +/// +/// 1. **§9a content addressing hashes bytes.** `content_id` is a SHA-256 over the +/// canonical form, so the same manifest submitted in two encodings would +/// produce two different ids — silently defeating federation deduplication. +/// That is precisely the failure mode §9a quantises scene times to avoid, and +/// it would be reintroduced at the encoding layer. +/// 2. **UTF-16 admits lone surrogates**, which have no UTF-8 representation. A +/// field able to carry them is a channel for bytes that survive validation but +/// are not text — against §5a's premise that no field can carry a payload. +/// 3. **§5a's character class assumes well-formed Unicode scalar values.** NFC +/// normalisation and the category checks are defined over scalars, so admitting +/// an encoding that can express non-scalars would undermine both. +/// +/// serde_json would reject non-UTF-8 anyway; the value added here is a diagnosable +/// error rather than a misleading one. A UTF-16 body otherwise fails with "key +/// must be a string", which points an operator at the wrong problem entirely. +fn require_utf8(bytes: &[u8]) -> Result<&str, ApiError> { + // A BOM is not valid JSON (RFC 8259 §8.1: "implementations MUST NOT add a + // byte order mark"), and it is the clearest signal of an encoding mistake, so + // it is named rather than left to the parser. + let encoding_hint = match bytes { + [0xEF, 0xBB, 0xBF, ..] => Some("UTF-8 with a byte order mark"), + [0xFF, 0xFE, 0x00, 0x00, ..] => Some("UTF-32LE"), + [0x00, 0x00, 0xFE, 0xFF, ..] => Some("UTF-32BE"), + [0xFF, 0xFE, ..] => Some("UTF-16LE"), + [0xFE, 0xFF, ..] => Some("UTF-16BE"), + // Unmarked UTF-16 is the common case, since encoders often omit the BOM. + // A JSON document always begins with an ASCII character, so an + // interleaved NUL in the first two bytes is conclusive. + [0x00, b, ..] if b.is_ascii_graphic() => Some("UTF-16BE (no BOM)"), + [b, 0x00, ..] if b.is_ascii_graphic() => Some("UTF-16LE (no BOM)"), + _ => None, + }; + + if let Some(encoding) = encoding_hint { + return Err(ApiError::BadRequest(format!( + "request body appears to be {encoding}: JSON must be UTF-8 encoded \ + without a byte order mark (RFC 8259 §8.1)" + ))); + } + + std::str::from_utf8(bytes).map_err(|e| { + ApiError::BadRequest(format!( + "request body is not valid UTF-8 at byte {}: JSON must be UTF-8 encoded \ + (RFC 8259 §8.1)", + e.valid_up_to() + )) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use axum::http::StatusCode; + use axum::response::IntoResponse; + + #[derive(serde::Deserialize)] + #[serde(deny_unknown_fields)] + struct Probe { + _wanted: i64, + } + + async fn extract(body: &'static str, content_type: Option<&str>) -> StatusCode { + let mut builder = Request::builder().method("POST").uri("/"); + if let Some(ct) = content_type { + builder = builder.header(CONTENT_TYPE, ct); + } + let req = builder.body(axum::body::Body::from(body)).unwrap(); + match Json::::from_request(req, &()).await { + Ok(_) => StatusCode::OK, + Err(e) => e.into_response().status(), + } + } + + #[tokio::test] + async fn schema_mismatch_is_400_not_422() { + // The whole reason this extractor exists (§4, §6). + assert_eq!( + extract(r#"{"unexpected":1}"#, Some("application/json")).await, + StatusCode::BAD_REQUEST + ); + } + + #[tokio::test] + async fn malformed_json_is_400() { + assert_eq!(extract("{ nope", Some("application/json")).await, StatusCode::BAD_REQUEST); + } + + #[tokio::test] + async fn missing_content_type_is_400() { + assert_eq!(extract(r#"{"_wanted":1}"#, None).await, StatusCode::BAD_REQUEST); + } + + #[tokio::test] + async fn content_type_parameters_are_tolerated() { + assert_eq!( + extract(r#"{"_wanted":1}"#, Some("application/json; charset=utf-8")).await, + StatusCode::OK + ); + } + + #[tokio::test] + async fn valid_body_extracts() { + assert_eq!(extract(r#"{"_wanted":1}"#, Some("application/json")).await, StatusCode::OK); + } + + #[tokio::test] + async fn an_explicit_utf8_charset_is_accepted() { + for ct in [ + "application/json; charset=utf-8", + "application/json;charset=UTF-8", + "application/json; charset=\"utf-8\"", + "application/json; charset=utf8", + ] { + assert_eq!(extract(r#"{"_wanted":1}"#, Some(ct)).await, StatusCode::OK, "{ct}"); + } + } + + #[tokio::test] + async fn a_non_utf8_charset_is_refused_by_name() { + for ct in [ + "application/json; charset=utf-16", + "application/json; charset=iso-8859-1", + "application/json; charset=windows-1252", + ] { + assert_eq!( + extract(r#"{"_wanted":1}"#, Some(ct)).await, + StatusCode::BAD_REQUEST, + "{ct}" + ); + } + } + + /// Builds a request from raw bytes, since these bodies are not valid `&str`. + async fn extract_bytes(body: Vec) -> Result<(), ApiError> { + let req = Request::builder() + .method("POST") + .uri("/") + .header(CONTENT_TYPE, "application/json") + .body(axum::body::Body::from(body)) + .unwrap(); + Json::::from_request(req, &()).await.map(|_| ()) + } + + #[tokio::test] + async fn utf16_bodies_are_rejected_with_an_actionable_message() { + // The reason this check exists: serde_json rejects UTF-16 anyway, but with + // "key must be a string", which points an operator at the wrong problem. + let doc = r#"{"_wanted":1}"#; + + let le: Vec = doc.encode_utf16().flat_map(|u| u.to_le_bytes()).collect(); + let err = extract_bytes(le).await.unwrap_err().to_string(); + assert!(err.contains("UTF-16LE"), "should name the encoding: {err}"); + assert!(err.contains("UTF-8"), "should say what is required: {err}"); + + let be: Vec = doc.encode_utf16().flat_map(|u| u.to_be_bytes()).collect(); + let err = extract_bytes(be).await.unwrap_err().to_string(); + assert!(err.contains("UTF-16BE"), "should name the encoding: {err}"); + + // With BOMs. + let mut le_bom = vec![0xFF, 0xFE]; + le_bom.extend(doc.encode_utf16().flat_map(|u| u.to_le_bytes())); + assert!(extract_bytes(le_bom).await.is_err()); + + let mut be_bom = vec![0xFE, 0xFF]; + be_bom.extend(doc.encode_utf16().flat_map(|u| u.to_be_bytes())); + assert!(extract_bytes(be_bom).await.is_err()); + } + + #[tokio::test] + async fn a_utf8_bom_is_rejected() { + // RFC 8259 §8.1: implementations MUST NOT add a byte order mark. + let mut body = vec![0xEF, 0xBB, 0xBF]; + body.extend_from_slice(br#"{"_wanted":1}"#); + let err = extract_bytes(body).await.unwrap_err().to_string(); + assert!(err.contains("byte order mark"), "{err}"); + } + + #[tokio::test] + async fn invalid_utf8_is_rejected_with_the_offending_offset() { + // A truncated multi-byte sequence inside an otherwise well-formed document. + let body = b"{\"_wanted\":\"\xC3\x28\"}".to_vec(); + let err = extract_bytes(body).await.unwrap_err().to_string(); + assert!(err.contains("not valid UTF-8"), "{err}"); + assert!(err.contains("byte 12"), "should locate the failure: {err}"); + } + + #[tokio::test] + async fn valid_multibyte_utf8_is_accepted() { + // The check must not reject legitimate non-ASCII content — actor names are + // routinely non-Latin (§5a accepts any Unicode letter). + // Rejected for the unknown `_note` field, not for its encoding — which is + // the distinction being asserted. + let body = r#"{"_wanted":1,"_note":"宮崎 駿 Renée"}"#.as_bytes().to_vec(); + let err = extract_bytes(body).await.unwrap_err().to_string(); + assert!(err.contains("_note"), "should fail on the schema, not the encoding: {err}"); + } + + #[tokio::test] + async fn an_empty_body_is_not_mistaken_for_an_encoding_problem() { + let err = extract_bytes(Vec::new()).await.unwrap_err().to_string(); + assert!(!err.contains("UTF-16"), "empty body is a parse error, not an encoding one: {err}"); + } +} diff --git a/src/api/mod.rs b/src/api/mod.rs new file mode 100644 index 0000000..72e32f1 --- /dev/null +++ b/src/api/mod.rs @@ -0,0 +1,35 @@ +//! HTTP surface (§4). Base path `/api/v1`, JSON throughout. + +pub mod exists; +pub mod fetch; +pub mod json; +pub mod report; +pub mod upload; + +use serde::Deserialize; + +use crate::matching::ClientCut; + +/// Identity + cut query parameters, shared by the read endpoints (§4). +#[derive(Debug, Clone, Default, Deserialize)] +pub struct LookupParams { + pub tmdb_id: Option, + pub imdb_id: Option, + pub series_tmdb_id: Option, + pub series_imdb_id: Option, + pub season: Option, + pub episode: Option, + pub runtime_sec: Option, + pub video_hash: Option, +} + +impl LookupParams { + pub fn client_cut(&self) -> ClientCut { + ClientCut { + // A non-finite or non-positive runtime is not a usable signal; treat + // it as absent rather than letting it drive a match. + runtime_sec: self.runtime_sec.filter(|r| r.is_finite() && *r > 0.0), + video_hash: self.video_hash.clone(), + } + } +} diff --git a/src/api/report.rs b/src/api/report.rs new file mode 100644 index 0000000..a50303a --- /dev/null +++ b/src/api/report.rs @@ -0,0 +1,141 @@ +//! `POST /manifests/{id}/report` (§4), and `GET /health`. +//! +//! Reports are a moderation lever and cheap to abuse, hence the tight §5 limit. +//! A report never changes `status` by itself: §5a keeps delisting an operator +//! action, because automatic delisting on report would hand any client a remote +//! delete primitive. + +use axum::extract::{Path, State}; +use axum::http::HeaderMap; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::{Deserialize, Serialize}; + +use crate::db::repo; +use crate::error::{ApiError, ApiResult}; +use crate::ratelimit::Surface; +use crate::state::{with_quota_headers, AppState}; +use crate::worker::now_iso; + +/// §4: `{ "reason": "misaligned" | "wrong_actors" | "spam", "note": "..." }`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Deserialize, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum ReportReason { + Misaligned, + WrongActors, + Spam, +} + +impl ReportReason { + fn as_str(self) -> &'static str { + match self { + ReportReason::Misaligned => "misaligned", + ReportReason::WrongActors => "wrong_actors", + ReportReason::Spam => "spam", + } + } +} + +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ReportRequest { + pub reason: ReportReason, + #[serde(default)] + pub note: Option, +} + +/// §5a: `note` is free text from an anonymous caller, so it is capped hard. It is +/// never served back to clients — only the operator reads it. +const MAX_NOTE_CHARS: usize = 500; + +#[derive(Debug, Serialize)] +pub struct ReportAccepted { + pub report_id: String, +} + +pub async fn post_report( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, + Path(manifest_id): Path, + super::json::Json(req): super::json::Json, +) -> ApiResult { + let ip = state.client_ip(&headers, peer.0); + let quota = state.check_limit(&ip, Surface::Report)?; + + let note = match req.note { + Some(n) if n.chars().count() > MAX_NOTE_CHARS => { + return Err(ApiError::BadRequest(format!( + "note: longer than {MAX_NOTE_CHARS} characters" + ))) + } + // Strip control characters; the note is operator-facing text, not markup. + Some(n) => Some(n.chars().filter(|c| !c.is_control()).collect::()), + None => None, + }; + + let ip_hash = crate::auth::hash_ip(&ip, &state.config.server_id); + let reason = req.reason.as_str(); + let now = now_iso(); + let id_for_check = manifest_id.clone(); + + let exists = state + .db + .read(move |c| Ok(repo::manifest_by_id(c, &id_for_check)?.is_some())) + .await + .map_err(ApiError::Internal)?; + if !exists { + return Err(ApiError::NotFound); + } + + let report_id = state + .db + .write(move |tx| { + repo::insert_report(tx, &manifest_id, reason, note.as_deref(), &ip_hash, &now) + }) + .await + .map_err(ApiError::Internal)?; + + Ok(with_quota_headers(Json(ReportAccepted { report_id }).into_response(), quota)) +} + +#[derive(Debug, Serialize)] +pub struct Health { + pub status: &'static str, + pub version: &'static str, +} + +/// `GET /health` — liveness, unauthenticated and unlimited (§4, §5). +pub async fn health() -> Json { + Json(Health { status: "ok", version: env!("CARGO_PKG_VERSION") }) +} + +#[derive(Debug, Serialize)] +pub struct Readiness { + pub status: &'static str, + pub database: &'static str, + /// §8: TMDB is a hard dependency for UR-3. If it is unconfigured, uploads + /// accumulate in `pending` rather than being listed unverified — worth + /// surfacing rather than failing silently. + pub tmdb_configured: bool, +} + +/// Readiness check verifying the database opens and migrations are current (§8). +pub async fn ready(State(state): State) -> ApiResult> { + let ok = state + .db + .read(|conn| { + // Any query against a schema table proves both that the file opens + // and that migrations have been applied. + let n: i64 = conn.query_row("SELECT COUNT(*) FROM manifests", [], |r| r.get(0))?; + Ok(n >= 0) + }) + .await + .map_err(ApiError::Internal)?; + + Ok(Json(Readiness { + status: if ok { "ready" } else { "degraded" }, + database: "ok", + tmdb_configured: state.tmdb.is_configured(), + })) +} diff --git a/src/api/upload.rs b/src/api/upload.rs new file mode 100644 index 0000000..8fb4199 --- /dev/null +++ b/src/api/upload.rs @@ -0,0 +1,241 @@ +//! Contribution endpoints (§4) — UR-2 and UR-6. +//! +//! Both require a token (§5). Both return `202`: the upload has passed size and +//! schema validation and is held unlisted pending the asynchronous TMDB cast +//! check (§6 stage 3). + +use axum::extract::State; +use axum::http::{HeaderMap, StatusCode}; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::Serialize; + +use crate::db::repo; +use crate::error::{ApiError, ApiResult}; +use crate::ingest::{self, IngestOutcome}; +use crate::model::{IdentityType, Jmanifest, SeriesBundle}; +use crate::ratelimit::Surface; +use crate::state::{with_quota_headers, AppState}; +use crate::validate::{self, limits}; +use crate::worker::now_iso; + +#[derive(Debug, Serialize)] +pub struct UploadAccepted { + pub manifest_id: String, + pub status: &'static str, +} + +/// `POST /manifests` — UR-2. +pub async fn post_manifest( + State(state): State, + headers: HeaderMap, + super::json::Json(manifest): super::json::Json, +) -> ApiResult { + let contributor = state.require_contributor(&headers).await?; + // §5: limits are per token where one is present. + let quota = state.check_limit(&contributor.id, Surface::ManifestUpload)?; + + // §6 stage 2. A rejection names the offending field, so a client that forgets + // to strip `movie`/`jellyfin_id` gets a diagnosable `400`. + let valid = + validate::validate_manifest(manifest).map_err(|e| ApiError::BadRequest(e.to_string()))?; + + let origin = state.config.server_id.clone(); + let contributor_id = contributor.id.clone(); + let now = now_iso(); + + let outcome = state + .db + .write(move |tx| ingest::persist(tx, &valid, Some(&contributor_id), &origin, None, &now)) + .await + .map_err(ApiError::Internal)?; + + let resp = match outcome { + IngestOutcome::Pending { manifest_id } => { + (StatusCode::ACCEPTED, Json(UploadAccepted { manifest_id, status: "pending" })) + .into_response() + } + // §4 `409` — an identical `(identity, cut)` manifest already exists from + // this contributor. + IngestOutcome::DuplicateFromContributor { manifest_id } => { + return Err(ApiError::Conflict(format!( + "an identical manifest already exists from this contributor: {manifest_id}" + ))) + } + // §9a: identical content already held, from any source. Not an error — + // the contributor's work is simply already represented. + IngestOutcome::DuplicateContent { manifest_id } => { + (StatusCode::OK, Json(UploadAccepted { manifest_id, status: "already_present" })) + .into_response() + } + }; + + Ok(with_quota_headers(resp, quota)) +} + +#[derive(Debug, Serialize)] +pub struct BundleResult { + pub season: Option, + pub episode: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub manifest_id: Option, + pub status: &'static str, + #[serde(skip_serializing_if = "Option::is_none")] + pub reason: Option, +} + +#[derive(Debug, Serialize)] +pub struct BundleAccepted { + pub results: Vec, +} + +/// `POST /manifests/bundle` — UR-6. +/// +/// **Per-episode validation, not atomic**: valid episodes are accepted and +/// invalid ones rejected, with a per-episode result list. All-or-nothing would let +/// one bad episode discard an entire season's compute (§2). +/// +/// **One rate-limit unit**, so contributing a season is not punished relative to +/// contributing a film (§2, §5). +pub async fn post_bundle( + State(state): State, + headers: HeaderMap, + super::json::Json(bundle): super::json::Json, +) -> ApiResult { + let contributor = state.require_contributor(&headers).await?; + let quota = state.check_limit(&contributor.id, Surface::BundleUpload)?; + + // §4: `413` for exceeding the episode cap, distinct from a malformed envelope. + if bundle.episodes.len() > limits::MAX_BUNDLE_EPISODES { + return Err(ApiError::PayloadTooLarge(format!( + "bundle carries {} episodes, limit is {}", + bundle.episodes.len(), + limits::MAX_BUNDLE_EPISODES + ))); + } + // §4: `400` only for the envelope itself; individual bad episodes are + // reported in the results list, not as a whole-request error. + validate::validate_bundle_envelope(&bundle).map_err(|e| ApiError::BadRequest(e.to_string()))?; + + let series_tmdb = bundle.series.series_tmdb_id.clone(); + let mut results = Vec::with_capacity(bundle.episodes.len()); + + for episode in bundle.episodes { + let coords = (episode.identity.season, episode.identity.episode); + + // An episode whose identity contradicts the envelope is rejected on its + // own rather than being silently reattributed to the bundle's series. + if episode.identity.kind != IdentityType::Episode { + results.push(BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: None, + status: "rejected", + reason: Some("identity.type must be 'episode' within a bundle".into()), + }); + continue; + } + if let (Some(envelope), Some(ep)) = (&series_tmdb, &episode.identity.series_tmdb_id) { + if envelope != ep { + results.push(BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: None, + status: "rejected", + reason: Some("series_tmdb_id does not match the bundle envelope".into()), + }); + continue; + } + } + + let valid = match validate::validate_manifest(episode) { + Ok(v) => v, + Err(e) => { + results.push(BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: None, + status: "rejected", + reason: Some(e.to_string()), + }); + continue; + } + }; + + let origin = state.config.server_id.clone(); + let contributor_id = contributor.id.clone(); + let now = now_iso(); + // One transaction per episode, so a bundle never holds the write lock for + // the whole request (§8 chunked ingest reasoning). + let outcome = state + .db + .write(move |tx| { + ingest::persist(tx, &valid, Some(&contributor_id), &origin, None, &now) + }) + .await; + + results.push(match outcome { + Ok(IngestOutcome::Pending { manifest_id }) => BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: Some(manifest_id), + status: "pending", + reason: None, + }, + Ok(IngestOutcome::DuplicateFromContributor { manifest_id }) + | Ok(IngestOutcome::DuplicateContent { manifest_id }) => BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: Some(manifest_id), + status: "already_present", + reason: None, + }, + Err(e) => { + tracing::error!(error = ?e, "bundle episode failed to persist"); + BundleResult { + season: coords.0, + episode: coords.1, + manifest_id: None, + status: "rejected", + reason: Some("internal error".into()), + } + } + }); + } + + let resp = (StatusCode::ACCEPTED, Json(BundleAccepted { results })).into_response(); + Ok(with_quota_headers(resp, quota)) +} + +#[derive(Debug, Serialize)] +pub struct TokenIssued { + pub token: String, +} + +/// Issues an anonymous bearer capability (§5a). +/// +/// Self-issued on request: no email, no verification, no personal data. Stored +/// only as a hash, so the server cannot enumerate who holds tokens. Discarding a +/// token and requesting another is trivially easy — and that is fine, because the +/// token is not the defence; the content checks are. +pub async fn post_token( + State(state): State, + peer: crate::state::PeerIp, + headers: HeaderMap, +) -> ApiResult> { + let ip = state.client_ip(&headers, peer.0); + // Reuse the report budget: issuing tokens is cheap but should not be a free + // unbounded write. + state.check_limit(&ip, Surface::Report)?; + + let token = crate::auth::generate_token(); + let hash = crate::auth::hash_token(&token); + let now = now_iso(); + state + .db + .write(move |tx| repo::insert_contributor(tx, &hash, &now)) + .await + .map_err(ApiError::Internal)?; + + Ok(Json(TokenIssued { token })) +} diff --git a/src/app.rs b/src/app.rs new file mode 100644 index 0000000..f3e7f47 --- /dev/null +++ b/src/app.rs @@ -0,0 +1,66 @@ +//! Router construction. +//! +//! §6 stage 0/1 body caps are applied here as per-route `DefaultBodyLimit` +//! layers: Axum rejects on `Content-Length` before reading a body *and* caps the +//! stream for chunked or mis-declared uploads, which is what makes a lying +//! header and a chunked upload both safe. Per-route means the bundle endpoint +//! gets its larger limit without widening the others (§6 stage 1). + +use std::time::Duration; + +use axum::extract::DefaultBodyLimit; +use axum::routing::{get, post}; +use axum::Router; +use tower_http::timeout::TimeoutLayer; +use tower_http::trace::TraceLayer; + +use crate::api::{exists, fetch, report, upload}; +use crate::state::AppState; +use crate::validate::limits; + +/// Small cap for endpoints that take a short JSON body. A read endpoint has no +/// business accepting a large payload, and the batch `exists` form is bounded at +/// 100 items. +const SMALL_BODY_LIMIT: usize = 256 * 1024; + +pub fn router(state: AppState) -> Router { + let timeout = state.config.request_timeout; + + let v1 = Router::new() + // UR-1 — existence probes. + .route("/manifests/exists", get(exists::exists).post(exists::exists_batch)) + // Reads. + .route("/manifests/movie", get(fetch::get_movie)) + .route("/manifests/episode", get(fetch::get_episode)) + .route("/manifests/series/{series_tmdb_id}", get(fetch::get_series)) + .route("/manifests/{id}", get(fetch::get_by_id)) + .route("/manifests/{id}/status", get(fetch::get_status)) + .route("/manifests/{id}/report", post(report::post_report)) + // UR-2 — contribution. + .route( + "/manifests", + post(upload::post_manifest).layer(DefaultBodyLimit::max(limits::BODY_LIMIT_MANIFEST)), + ) + // UR-6 — whole-series contribution, with its own larger cap. + .route( + "/manifests/bundle", + post(upload::post_bundle).layer(DefaultBodyLimit::max(limits::BODY_LIMIT_BUNDLE)), + ) + // §5a — anonymous bearer capability, not an account. + .route("/tokens", post(upload::post_token)) + .layer(DefaultBodyLimit::max(SMALL_BODY_LIMIT)); + + Router::new() + .route("/health", get(report::health)) + .route("/ready", get(report::ready)) + .nest("/api/v1", v1) + // §8: a request timeout so a slow bundle query fails fast. + .layer(TimeoutLayer::with_status_code(axum::http::StatusCode::REQUEST_TIMEOUT, timeout)) + .layer(TraceLayer::new_for_http()) + .with_state(state) +} + +/// Convenience for tests and `main`. +pub fn default_timeout() -> Duration { + Duration::from_secs(30) +} diff --git a/src/auth.rs b/src/auth.rs new file mode 100644 index 0000000..f296a26 --- /dev/null +++ b/src/auth.rs @@ -0,0 +1,186 @@ +//! §5a tokens and client-IP attribution. +//! +//! A token is **not an account** — it is an anonymous bearer capability. No +//! email, no verification, no personal data. It is stored only as a hash, so the +//! server cannot enumerate who holds tokens, and its sole purposes are +//! rate-limiting attribution (§5) and revocation. +//! +//! Discarding a token and requesting another is trivially easy, and that is +//! fine: the token is not the defence, the content checks are. Sybil resistance +//! is not required because identity is not load-bearing. + +use std::net::IpAddr; + +use axum::http::HeaderMap; +use sha2::{Digest, Sha256}; + +/// Hashes a bearer token for storage and lookup. +/// +/// Plain SHA-256 rather than a password KDF is deliberate and sufficient here: +/// tokens are 256 bits of server-generated randomness, not user-chosen secrets, +/// so there is no dictionary to attack. +pub fn hash_token(token: &str) -> String { + let mut h = Sha256::new(); + h.update(token.as_bytes()); + hex(&h.finalize()) +} + +/// Hashes a client IP for report attribution (§7 `reports.source_ip_hash`). +/// +/// Salted with the server id so hashes are not comparable across instances. +pub fn hash_ip(ip: &str, server_id: &str) -> String { + let mut h = Sha256::new(); + h.update(server_id.as_bytes()); + h.update(b"\0"); + h.update(ip.as_bytes()); + hex(&h.finalize()) +} + +fn hex(bytes: &[u8]) -> String { + let mut s = String::with_capacity(bytes.len() * 2); + for b in bytes { + s.push_str(&format!("{b:02x}")); + } + s +} + +/// Generates a new token. Returned once to the caller; only its hash is stored. +pub fn generate_token() -> String { + use rand::RngCore; + let mut bytes = [0u8; 32]; + rand::rng().fill_bytes(&mut bytes); + format!("jray_{}", hex(&bytes)) +} + +/// Extracts a bearer token from an `Authorization` header. +pub fn bearer_token(headers: &HeaderMap) -> Option { + let raw = headers.get(axum::http::header::AUTHORIZATION)?.to_str().ok()?; + let (scheme, value) = raw.split_once(' ')?; + if !scheme.eq_ignore_ascii_case("bearer") { + return None; + } + let value = value.trim(); + if value.is_empty() { + return None; + } + Some(value.to_string()) +} + +/// Resolves the client IP for rate-limiting and report attribution. +/// +/// §8: the app must trust `X-Forwarded-For` **only** from the operator's proxy. +/// Rate limiting and report attribution key on client IP, so a spoofable header +/// defeats both — hence `trusted_proxies` is explicit configuration and an +/// untrusted peer's header is ignored outright. +pub fn client_ip(headers: &HeaderMap, peer: Option, trusted_proxies: &[IpAddr]) -> String { + let peer_is_trusted = peer.is_some_and(|p| trusted_proxies.contains(&p)); + + if peer_is_trusted { + if let Some(xff) = headers.get("x-forwarded-for").and_then(|v| v.to_str().ok()) { + // Right-most entry is the one our trusted proxy appended; entries to + // its left are client-supplied and forgeable. Walk from the right + // past any further trusted hops. + for candidate in xff.split(',').rev().map(str::trim).filter(|s| !s.is_empty()) { + match candidate.parse::() { + Ok(ip) if trusted_proxies.contains(&ip) => continue, + Ok(ip) => return ip.to_string(), + Err(_) => break, + } + } + } + } + + peer.map(|p| p.to_string()).unwrap_or_else(|| "unknown".to_string()) +} + +#[cfg(test)] +mod tests { + use super::*; + use axum::http::HeaderValue; + + fn headers(pairs: &[(&'static str, &str)]) -> HeaderMap { + let mut h = HeaderMap::new(); + for (k, v) in pairs { + h.insert(*k, HeaderValue::from_str(v).unwrap()); + } + h + } + + #[test] + fn token_hash_is_stable_and_distinguishing() { + assert_eq!(hash_token("abc"), hash_token("abc")); + assert_ne!(hash_token("abc"), hash_token("abd")); + assert_eq!(hash_token("abc").len(), 64); + } + + #[test] + fn generated_tokens_are_unique_and_prefixed() { + let a = generate_token(); + let b = generate_token(); + assert_ne!(a, b); + assert!(a.starts_with("jray_")); + assert_eq!(a.len(), 5 + 64); + } + + #[test] + fn ip_hash_is_salted_per_server() { + // Hashes must not be comparable across instances. + assert_ne!(hash_ip("1.2.3.4", "a.example"), hash_ip("1.2.3.4", "b.example")); + assert_eq!(hash_ip("1.2.3.4", "a.example"), hash_ip("1.2.3.4", "a.example")); + } + + #[test] + fn parses_bearer_tokens_case_insensitively() { + assert_eq!( + bearer_token(&headers(&[("authorization", "Bearer xyz")])).as_deref(), + Some("xyz") + ); + assert_eq!( + bearer_token(&headers(&[("authorization", "bearer xyz")])).as_deref(), + Some("xyz") + ); + assert!(bearer_token(&headers(&[("authorization", "Basic xyz")])).is_none()); + assert!(bearer_token(&headers(&[("authorization", "Bearer ")])).is_none()); + assert!(bearer_token(&HeaderMap::new()).is_none()); + } + + #[test] + fn forwarded_header_from_an_untrusted_peer_is_ignored() { + // The whole point of §8's explicit trusted-proxy configuration: an + // arbitrary client must not be able to choose its own rate-limit key. + let h = headers(&[("x-forwarded-for", "9.9.9.9")]); + let peer: IpAddr = "203.0.113.7".parse().unwrap(); + assert_eq!(client_ip(&h, Some(peer), &[]), "203.0.113.7"); + } + + #[test] + fn forwarded_header_from_a_trusted_proxy_is_honoured() { + let h = headers(&[("x-forwarded-for", "9.9.9.9")]); + let proxy: IpAddr = "127.0.0.1".parse().unwrap(); + assert_eq!(client_ip(&h, Some(proxy), &[proxy]), "9.9.9.9"); + } + + #[test] + fn client_supplied_entries_left_of_the_proxy_cannot_spoof() { + // A client that sends its own XFF gets its value appended to, not + // replaced, so only the right-most entry is trustworthy. + let h = headers(&[("x-forwarded-for", "9.9.9.9, 203.0.113.7")]); + let proxy: IpAddr = "127.0.0.1".parse().unwrap(); + assert_eq!(client_ip(&h, Some(proxy), &[proxy]), "203.0.113.7"); + } + + #[test] + fn walks_past_additional_trusted_hops() { + let inner: IpAddr = "10.0.0.2".parse().unwrap(); + let proxy: IpAddr = "127.0.0.1".parse().unwrap(); + let h = headers(&[("x-forwarded-for", "203.0.113.7, 10.0.0.2")]); + assert_eq!(client_ip(&h, Some(proxy), &[proxy, inner]), "203.0.113.7"); + } + + #[test] + fn malformed_forwarded_value_falls_back_to_the_peer() { + let h = headers(&[("x-forwarded-for", "not-an-ip")]); + let proxy: IpAddr = "127.0.0.1".parse().unwrap(); + assert_eq!(client_ip(&h, Some(proxy), &[proxy]), "127.0.0.1"); + } +} diff --git a/src/castcheck.rs b/src/castcheck.rs new file mode 100644 index 0000000..581f8b7 --- /dev/null +++ b/src/castcheck.rs @@ -0,0 +1,484 @@ +//! §6 stage 3 cast-match scoring, as pure functions. +//! +//! The thresholds here are the load-bearing part of UR-3 and §5a Threat 2, and +//! §10 (5) wants them retuned against the 331-file extraction corpus. Keeping +//! the decision logic free of I/O is what makes that a test-data exercise rather +//! than a code change. + +use crate::tmdb::CastMember; + +/// §6: thresholds over the ratio `|M ∩ C| / |M|`. +pub const LISTED_THRESHOLD: f64 = 0.6; +pub const FLAGGED_THRESHOLD: f64 = 0.3; +/// Below this size a ratio is meaningless (§6 small-|M| handling). +pub const SMALL_M_LIMIT: usize = 5; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Verdict { + /// Normal case. + Listed, + /// Served with reduced ranking, flagged for review. + Flagged, + /// Deleted, and the contributor's counter bumped. + Rejected, +} + +impl Verdict { + pub fn status(self) -> &'static str { + match self { + Verdict::Listed => "listed", + Verdict::Flagged => "flagged", + Verdict::Rejected => "rejected", + } + } +} + +#[derive(Debug, Clone)] +pub struct CastCheckOutcome { + pub verdict: Verdict, + pub ratio: f64, + /// Manifest actors resolved to a TMDB person id, with the TMDB-authoritative + /// name. Only these are kept; §6 drops unmatched actors rather than storing + /// them, which is what closes §5a's free-text channel. + pub matched: Vec, + /// Actors that matched nothing and will be dropped. + pub unmatched_person_ids: Vec, + pub reason: Option, +} + +#[derive(Debug, Clone)] +pub struct MatchedActor { + pub tmdb_person_id: u64, + /// From TMDB, never from the upload. + pub name: String, + pub adult: bool, + /// True when the match came from name comparison rather than an id. + pub by_name: bool, +} + +/// An actor as submitted, after §6 stage 2 validation. +#[derive(Debug, Clone)] +pub struct SubmittedActor { + pub tmdb_id: Option, + pub imdb_id: Option, + /// Used only for matching here, then discarded (§5a). + pub name: Option, +} + +/// Case- and accent-insensitive comparison key for the name fallback (§6). +fn name_key(s: &str) -> String { + use unicode_normalization::UnicodeNormalization; + s.nfd() + .filter(|c| !unicode_normalization::char::is_combining_mark(*c)) + .flat_map(|c| c.to_lowercase()) + .filter(|c| !c.is_whitespace() && *c != '.' && *c != ',' && *c != '-' && *c != '\'') + .collect() +} + +/// Runs the §6 stage 3 comparison. +/// +/// `credits` is the reference set *C*: for a movie, its credits; for an episode, +/// the union of per-episode credits and the series' aggregate credits. +pub fn evaluate(submitted: &[SubmittedActor], credits: &[CastMember]) -> CastCheckOutcome { + let m = submitted.len(); + + // §6: `|M| == 0` is rejected. These are extraction failures, not + // contributions — validation already refuses them, so reaching here means a + // manifest lost every actor upstream. + if m == 0 { + return CastCheckOutcome { + verdict: Verdict::Rejected, + ratio: 0.0, + matched: Vec::new(), + unmatched_person_ids: Vec::new(), + reason: Some("empty_actor_list".into()), + }; + } + + // TMDB has no credits for the id: absent data is not evidence of a bad + // manifest, so this is flagged rather than rejected (§6). + if credits.is_empty() { + return CastCheckOutcome { + verdict: Verdict::Flagged, + ratio: 0.0, + matched: Vec::new(), + unmatched_person_ids: submitted.iter().filter_map(|a| a.tmdb_id).collect(), + reason: Some("tmdb_no_credits".into()), + }; + } + + let mut matched: Vec = Vec::new(); + let mut unmatched: Vec = Vec::new(); + let mut id_matches = 0usize; + let mut name_matches = 0usize; + + for actor in submitted { + // Join on `tmdb_id` — grounded in the pipeline's actual output, where + // 330 of 331 manifests have `imdb_id: ""` and `tmdb_id` set (§6). + let by_id = actor.tmdb_id.and_then(|id| credits.iter().find(|c| c.id == id)); + + if let Some(c) = by_id { + id_matches += 1; + push_unique( + &mut matched, + MatchedActor { + tmdb_person_id: c.id, + name: c.name.clone(), + adult: c.adult, + by_name: false, + }, + ); + continue; + } + + // Fall back to case- and accent-insensitive name comparison. + let by_name = actor.name.as_deref().and_then(|n| { + let key = name_key(n); + (!key.is_empty()).then(|| credits.iter().find(|c| name_key(&c.name) == key))? + }); + + if let Some(c) = by_name { + name_matches += 1; + push_unique( + &mut matched, + MatchedActor { + tmdb_person_id: c.id, + name: c.name.clone(), + adult: c.adult, + by_name: true, + }, + ); + continue; + } + + if let Some(id) = actor.tmdb_id { + unmatched.push(id); + } + } + + // §6: name-only matches are counted but capped at half the intersection, so + // a manifest cannot pass on name collisions alone. + let capped_name_matches = name_matches.min(id_matches); + let effective = id_matches + capped_name_matches; + let ratio = effective as f64 / m as f64; + + let verdict = classify(m, effective, ratio); + let reason = match verdict { + Verdict::Rejected => Some("cast_match_below_threshold".into()), + Verdict::Flagged => Some("cast_match_marginal".into()), + Verdict::Listed => None, + }; + + CastCheckOutcome { verdict, ratio, matched, unmatched_person_ids: unmatched, reason } +} + +fn push_unique(matched: &mut Vec, actor: MatchedActor) { + if !matched.iter().any(|m| m.tmdb_person_id == actor.tmdb_person_id) { + matched.push(actor); + } +} + +/// §6 small-*M* handling. With a median of 7 actors a ratio threshold is coarse +/// — one mismatch moves it by 14% — so small manifests use counts, not ratios. +fn classify(m: usize, matches: usize, ratio: f64) -> Verdict { + if m >= SMALL_M_LIMIT { + if ratio >= LISTED_THRESHOLD { + Verdict::Listed + } else if ratio >= FLAGGED_THRESHOLD { + Verdict::Flagged + } else { + Verdict::Rejected + } + } else if m >= 2 { + // Require all but one actor to match. + if matches + 1 >= m { + Verdict::Listed + } else { + Verdict::Rejected + } + } else { + // |M| <= 1: accept only if the single actor matches. Such a manifest is + // near-worthless anyway and is ranked last. + if matches >= 1 { + Verdict::Listed + } else { + Verdict::Rejected + } + } +} + +/// §5a additional layer 1 — category guard. +/// +/// Rejects when a matched person is flagged adult by TMDB and the target title +/// is not, which targets the stated prank without needing a blocklist of names. +pub fn category_guard_violation(matched: &[MatchedActor], title_is_adult: bool) -> Option { + if title_is_adult { + return None; + } + matched.iter().find(|m| m.adult).map(|m| m.tmdb_person_id) +} + +/// §5a additional layer 2 — age-appropriateness guard. +/// +/// On a children's certification, apply the strictest cast-match threshold and +/// require an `exact` or `runtime` cut match. Mismatched content on children's +/// titles is the highest-harm case and deserves the tightest gate. +pub fn is_childrens_certification(cert: &str) -> bool { + matches!( + cert.trim().to_ascii_uppercase().as_str(), + "G" | "TV-Y" | "TV-Y7" | "TV-G" | "U" | "0+" | "6+" | "PG" | "TV-PG" + ) +} + +pub const CHILDRENS_LISTED_THRESHOLD: f64 = 0.8; + +/// Applies the children's-title gate to an already-computed outcome. +pub fn apply_childrens_guard(outcome: &mut CastCheckOutcome, m: usize) { + if m >= SMALL_M_LIMIT && outcome.ratio < CHILDRENS_LISTED_THRESHOLD { + outcome.verdict = match outcome.verdict { + Verdict::Listed => Verdict::Flagged, + other => other, + }; + if outcome.reason.is_none() { + outcome.reason = Some("childrens_title_strict_threshold".into()); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn credit(id: u64, name: &str) -> CastMember { + CastMember { id, name: name.to_string(), adult: false } + } + + fn adult_credit(id: u64, name: &str) -> CastMember { + CastMember { id, name: name.to_string(), adult: true } + } + + fn by_id(id: u64) -> SubmittedActor { + SubmittedActor { tmdb_id: Some(id), imdb_id: None, name: None } + } + + fn by_name(name: &str) -> SubmittedActor { + SubmittedActor { tmdb_id: None, imdb_id: None, name: Some(name.to_string()) } + } + + /// A realistic reference cast — feature casts are several times larger than + /// the manifests extracted from them (§6). + fn cast_of_20() -> Vec { + (1..=20).map(|i| credit(i, &format!("Actor {i}"))).collect() + } + + #[test] + fn full_subset_of_the_cast_is_listed() { + // §6: the ratio is over *M*, not *C* — a manifest legitimately contains + // only actors both credited and detected on screen, so penalising it for + // missing credited actors would fail every honest upload. + let submitted: Vec<_> = (1..=7).map(by_id).collect(); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Listed); + assert_eq!(out.ratio, 1.0); + assert_eq!(out.matched.len(), 7); + } + + #[test] + fn threshold_boundaries_at_point_six_and_point_three() { + // 6 of 10 matching == 0.6 exactly: listed. + let mut submitted: Vec<_> = (1..=6).map(by_id).collect(); + submitted.extend((900..904).map(by_id)); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.matched.len(), 6); + assert!((out.ratio - 0.6).abs() < 1e-9); + assert_eq!(out.verdict, Verdict::Listed); + + // 5 of 10 == 0.5: flagged, served with reduced ranking. + let mut submitted: Vec<_> = (1..=5).map(by_id).collect(); + submitted.extend((900..905).map(by_id)); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Flagged); + + // 3 of 10 == 0.3 exactly: still flagged, not rejected. + let mut submitted: Vec<_> = (1..=3).map(by_id).collect(); + submitted.extend((900..907).map(by_id)); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Flagged); + + // 2 of 10 == 0.2: rejected. + let mut submitted: Vec<_> = (1..=2).map(by_id).collect(); + submitted.extend((900..908).map(by_id)); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Rejected); + } + + #[test] + fn prank_manifest_is_rejected() { + // §5a Threat 2: performers who are not credited cast on the title. + let submitted: Vec<_> = (500..510).map(by_id).collect(); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Rejected); + assert_eq!(out.ratio, 0.0); + assert_eq!(out.reason.as_deref(), Some("cast_match_below_threshold")); + } + + #[test] + fn small_m_requires_all_but_one_to_match() { + // §6: `2 <= |M| < 5` — a ratio is meaningless at this size. + let out = evaluate(&[by_id(1), by_id(2), by_id(3), by_id(999)], &cast_of_20()); + assert_eq!(out.verdict, Verdict::Listed, "3 of 4 is all-but-one"); + + let out = evaluate(&[by_id(1), by_id(2), by_id(998), by_id(999)], &cast_of_20()); + assert_eq!(out.verdict, Verdict::Rejected, "2 of 4 fails all-but-one"); + + // 0.5 would be `Flagged` under the ratio table, so this proves the + // small-|M| branch is actually taken. + let out = evaluate(&[by_id(1), by_id(999)], &cast_of_20()); + assert_eq!(out.verdict, Verdict::Listed, "1 of 2 is all-but-one"); + } + + #[test] + fn single_actor_manifest_needs_that_actor_to_match() { + assert_eq!(evaluate(&[by_id(1)], &cast_of_20()).verdict, Verdict::Listed); + assert_eq!(evaluate(&[by_id(999)], &cast_of_20()).verdict, Verdict::Rejected); + } + + #[test] + fn empty_manifest_is_rejected() { + let out = evaluate(&[], &cast_of_20()); + assert_eq!(out.verdict, Verdict::Rejected); + assert_eq!(out.reason.as_deref(), Some("empty_actor_list")); + } + + #[test] + fn missing_tmdb_credits_flags_rather_than_rejects() { + // §6: absent data is not evidence of a bad manifest. + let submitted: Vec<_> = (1..=7).map(by_id).collect(); + let out = evaluate(&submitted, &[]); + assert_eq!(out.verdict, Verdict::Flagged); + assert_eq!(out.reason.as_deref(), Some("tmdb_no_credits")); + } + + #[test] + fn name_matching_is_case_and_accent_insensitive() { + let credits = vec![credit(1, "Renée Zellweger"), credit(2, "Miloš Forman")]; + let out = evaluate(&[by_name("renee zellweger"), by_name("MILOS FORMAN")], &credits); + assert_eq!(out.matched.len(), 2); + } + + #[test] + fn name_only_matches_cannot_carry_a_manifest_alone() { + // §6: name-only matches are capped at half the intersection, so a + // manifest cannot pass on name collisions alone. + let credits: Vec<_> = (1..=20).map(|i| credit(i, &format!("Actor {i}"))).collect(); + let submitted: Vec<_> = (1..=10).map(|i| by_name(&format!("Actor {i}"))).collect(); + let out = evaluate(&submitted, &credits); + assert_eq!(out.ratio, 0.0, "with no id matches, name matches cap to zero"); + assert_eq!(out.verdict, Verdict::Rejected); + } + + #[test] + fn name_matches_count_up_to_the_number_of_id_matches() { + let credits: Vec<_> = (1..=20).map(|i| credit(i, &format!("Actor {i}"))).collect(); + // 4 by id + 6 by name, of 10 => capped to 4 + 4 = 8 => 0.8. + let mut submitted: Vec<_> = (1..=4).map(by_id).collect(); + submitted.extend((5..=10).map(|i| by_name(&format!("Actor {i}")))); + let out = evaluate(&submitted, &credits); + assert!((out.ratio - 0.8).abs() < 1e-9, "got {}", out.ratio); + assert_eq!(out.verdict, Verdict::Listed); + } + + #[test] + fn unmatched_actors_are_reported_for_dropping() { + // §6: unmatched actors are dropped rather than stored. + let submitted: Vec<_> = (1..=6).map(by_id).chain([by_id(777)]).collect(); + let out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.unmatched_person_ids, vec![777]); + assert!(out.matched.iter().all(|m| m.tmdb_person_id != 777)); + } + + #[test] + fn matched_names_come_from_tmdb_not_the_upload() { + // §5a: the server stores references to TMDB entities, not + // attacker-authored text. + let credits = vec![credit(884, "Steve Buscemi")]; + let submitted = vec![SubmittedActor { + tmdb_id: Some(884), + imdb_id: None, + name: Some("Definitely Not Him".into()), + }]; + let out = evaluate(&submitted, &credits); + assert_eq!(out.matched[0].name, "Steve Buscemi"); + } + + #[test] + fn duplicate_credits_do_not_double_count() { + // TMDB aggregate credits can list a person more than once. + let credits = vec![credit(1, "A"), credit(1, "A")]; + let out = evaluate(&[by_id(1)], &credits); + assert_eq!(out.matched.len(), 1); + } + + #[test] + fn category_guard_catches_adult_performers_on_a_non_adult_title() { + // §5a layer 1, aimed squarely at the stated prank. + let matched = vec![ + MatchedActor { tmdb_person_id: 1, name: "A".into(), adult: false, by_name: false }, + MatchedActor { tmdb_person_id: 2, name: "B".into(), adult: true, by_name: false }, + ]; + assert_eq!(category_guard_violation(&matched, false), Some(2)); + // Unless the target title is itself flagged adult. + assert_eq!(category_guard_violation(&matched, true), None); + } + + #[test] + fn category_guard_ignores_clean_casts() { + let matched = vec![MatchedActor { + tmdb_person_id: 1, + name: "A".into(), + adult: false, + by_name: false, + }]; + assert_eq!(category_guard_violation(&matched, false), None); + } + + #[test] + fn adult_credit_is_carried_through_matching() { + let out = evaluate(&[by_id(9)], &[adult_credit(9, "X")]); + assert!(out.matched[0].adult); + } + + #[test] + fn childrens_certifications_are_recognised() { + for c in ["G", "TV-Y", "tv-y7", "U", " PG "] { + assert!(is_childrens_certification(c), "{c} should be a children's rating"); + } + for c in ["R", "NC-17", "TV-MA", "18", ""] { + assert!(!is_childrens_certification(c), "{c} should not be"); + } + } + + #[test] + fn childrens_guard_tightens_the_threshold() { + // §5a layer 2: the highest-harm case gets the tightest gate. A ratio of + // 0.7 lists normally but only reaches `flagged` on a children's title. + let mut submitted: Vec<_> = (1..=7).map(by_id).collect(); + submitted.extend((900..903).map(by_id)); + let mut out = evaluate(&submitted, &cast_of_20()); + assert_eq!(out.verdict, Verdict::Listed); + + let m = submitted.len(); + apply_childrens_guard(&mut out, m); + assert_eq!(out.verdict, Verdict::Flagged); + assert_eq!(out.reason.as_deref(), Some("childrens_title_strict_threshold")); + } + + #[test] + fn childrens_guard_leaves_strong_matches_listed() { + let submitted: Vec<_> = (1..=10).map(by_id).collect(); + let mut out = evaluate(&submitted, &cast_of_20()); + let m = submitted.len(); + apply_childrens_guard(&mut out, m); + assert_eq!(out.verdict, Verdict::Listed); + } +} diff --git a/src/config.rs b/src/config.rs new file mode 100644 index 0000000..e6a1042 --- /dev/null +++ b/src/config.rs @@ -0,0 +1,62 @@ +//! Operational configuration, read from the environment. +//! +//! The trusted-proxy CIDR is explicit configuration rather than a default-on +//! behaviour (§8 deployment notes): §5 rate limiting and report attribution key +//! on client IP, so an unconditionally-trusted `X-Forwarded-For` defeats both. + +use std::net::IpAddr; +use std::time::Duration; + +#[derive(Clone, Debug)] +pub struct Config { + pub bind: String, + pub db_path: String, + /// Hard dependency for UR-3. Without it, uploads accumulate in `pending` + /// rather than being listed unverified (§8). + pub tmdb_api_key: Option, + pub tmdb_base_url: String, + /// Prefixes of proxy addresses whose `X-Forwarded-For` is honoured. + pub trusted_proxies: Vec, + pub server_id: String, + pub request_timeout: Duration, + /// Number of cast-check jobs to lease per worker tick. + pub job_batch: usize, + pub job_poll_interval: Duration, +} + +impl Config { + pub fn from_env() -> anyhow::Result { + let trusted_proxies = match std::env::var("JRAY_TRUSTED_PROXIES") { + Ok(v) => v + .split(',') + .map(str::trim) + .filter(|s| !s.is_empty()) + .map(|s| { + s.parse::() + .map_err(|e| anyhow::anyhow!("bad JRAY_TRUSTED_PROXIES entry {s:?}: {e}")) + }) + .collect::, _>>()?, + Err(_) => Vec::new(), + }; + + Ok(Self { + bind: env_or("JRAY_BIND", "127.0.0.1:8080"), + db_path: env_or("JRAY_DB", "jray.db"), + tmdb_api_key: std::env::var("JRAY_TMDB_API_KEY").ok().filter(|s| !s.is_empty()), + tmdb_base_url: env_or("JRAY_TMDB_BASE_URL", "https://api.themoviedb.org/3"), + trusted_proxies, + server_id: env_or("JRAY_SERVER_ID", "localhost"), + request_timeout: Duration::from_secs(env_num("JRAY_REQUEST_TIMEOUT_SEC", 30)), + job_batch: env_num("JRAY_JOB_BATCH", 8) as usize, + job_poll_interval: Duration::from_secs(env_num("JRAY_JOB_POLL_SEC", 5)), + }) + } +} + +fn env_or(key: &str, default: &str) -> String { + std::env::var(key).ok().filter(|s| !s.is_empty()).unwrap_or_else(|| default.to_string()) +} + +fn env_num(key: &str, default: u64) -> u64 { + std::env::var(key).ok().and_then(|v| v.parse().ok()).unwrap_or(default) +} diff --git a/src/content_id.rs b/src/content_id.rs new file mode 100644 index 0000000..44ce29d --- /dev/null +++ b/src/content_id.rs @@ -0,0 +1,289 @@ +//! §9a content addressing. +//! +//! A validated manifest is immutable and content-addressable, which is what +//! makes replication *set reconciliation* rather than state synchronisation. +//! Even without the federation endpoints, computing `content_id` on upload gives +//! deduplication now and means stored manifests are already addressable when +//! federation lands. +//! +//! **This canonical form must be reimplemented byte-identically by the JRay +//! plugin** (§8 "Cost of choosing Rust": the extraction side is Python, so this +//! can no longer be shared as one implementation and must instead be specified +//! precisely and cross-tested). [`GOLDEN_VECTORS`] is that shared fixture. + +use sha2::{Digest, Sha256}; + +/// One actor's contribution to the canonical form. +#[derive(Debug, Clone)] +pub struct CanonicalActor { + pub tmdb_person_id: u64, + /// Integer centiseconds — quantised, not formatted floats (§9a). + pub scenes_cs: Vec<(i64, i64)>, +} + +/// The identity coordinates that enter the hash. +#[derive(Debug, Clone, Default)] +pub struct CanonicalIdentity { + pub kind: &'static str, + pub tmdb_id: Option, + pub imdb_id: Option, + pub season: Option, + pub episode: Option, +} + +/// The cut coordinates that enter the hash. +/// +/// **`audio_signature` is excluded, deliberately** (§9a): it is derived by +/// decoding audio, so two servers running different FFmpeg or resampler versions +/// could compute marginally different signatures for identical content, and +/// including it would silently break federation deduplication. +#[derive(Debug, Clone, Default)] +pub struct CanonicalCut { + /// Quantised to centiseconds for the same reason scene times are. + pub runtime_cs: i64, + pub video_hash: Option, +} + +/// Builds the canonical JSON form: keys sorted, no whitespace, actors sorted by +/// person id, scene times as integer centiseconds. +/// +/// `extraction` metadata and all local state are excluded, so two servers that +/// validated the same upload independently arrive at the same `content_id`. +pub fn canonical_json( + identity: &CanonicalIdentity, + cut: &CanonicalCut, + actors: &[CanonicalActor], +) -> String { + let mut sorted: Vec<&CanonicalActor> = actors.iter().collect(); + sorted.sort_by_key(|a| a.tmdb_person_id); + + let mut s = String::new(); + s.push_str("{\"actors\":["); + for (i, a) in sorted.iter().enumerate() { + if i > 0 { + s.push(','); + } + // Scene windows are emitted in stored order; validation has already + // established they are sorted by start time. + s.push_str("{\"scenes\":["); + for (j, (start, end)) in a.scenes_cs.iter().enumerate() { + if j > 0 { + s.push(','); + } + s.push('['); + s.push_str(&start.to_string()); + s.push(','); + s.push_str(&end.to_string()); + s.push(']'); + } + s.push_str("],\"tmdb_person_id\":"); + s.push_str(&a.tmdb_person_id.to_string()); + s.push('}'); + } + s.push_str("],\"cut\":{"); + s.push_str("\"runtime_cs\":"); + s.push_str(&cut.runtime_cs.to_string()); + s.push_str(",\"video_hash\":"); + push_opt_str(&mut s, cut.video_hash.as_deref()); + s.push_str("},\"identity\":{"); + s.push_str("\"episode\":"); + push_opt_num(&mut s, cut_opt(identity.episode)); + s.push_str(",\"imdb_id\":"); + push_opt_str(&mut s, identity.imdb_id.as_deref()); + s.push_str(",\"season\":"); + push_opt_num(&mut s, cut_opt(identity.season)); + s.push_str(",\"tmdb_id\":"); + push_opt_str(&mut s, identity.tmdb_id.as_deref()); + s.push_str(",\"type\":\""); + s.push_str(identity.kind); + s.push_str("\"}}"); + s +} + +fn cut_opt(v: Option) -> Option { + v +} + +fn push_opt_str(s: &mut String, v: Option<&str>) { + match v { + // Only closed-vocabulary values reach here (regex-constrained ids and a + // fixed-format hash), so no string escaping is required. + Some(v) => { + s.push('"'); + s.push_str(v); + s.push('"'); + } + None => s.push_str("null"), + } +} + +fn push_opt_num(s: &mut String, v: Option) { + match v { + Some(v) => s.push_str(&v.to_string()), + None => s.push_str("null"), + } +} + +/// `sha256:` over the canonical form (§9a). +pub fn content_id( + identity: &CanonicalIdentity, + cut: &CanonicalCut, + actors: &[CanonicalActor], +) -> String { + let canonical = canonical_json(identity, cut, actors); + let mut h = Sha256::new(); + h.update(canonical.as_bytes()); + let digest = h.finalize(); + let mut hex = String::with_capacity(64 + 7); + hex.push_str("sha256:"); + for b in digest { + hex.push_str(&format!("{b:02x}")); + } + hex +} + +/// Cross-implementation fixture (§8): the JRay plugin and any reimplementation +/// must reproduce these exactly, or federation deduplication silently breaks. +pub const GOLDEN_VECTORS: &[(&str, &str)] = &[( + // Movie, one actor, two windows, with a video hash. + r#"{"actors":[{"scenes":[[19160,20920],[43820,46560]],"tmdb_person_id":884}],"cut":{"runtime_cs":642050,"video_hash":"opensubtitles:8e245d9679d31e12"},"identity":{"episode":null,"imdb_id":"tt4686844","season":null,"tmdb_id":"504172","type":"movie"}}"#, + // Verified against an independent Python implementation: + // sha256(canonical.encode()).hexdigest() + "sha256:367f8b05c54a992a3a30fa016edaaac0b9b36b148fc76574f5b1ef326b56760f", +)]; + +#[cfg(test)] +mod tests { + use super::*; + + fn movie_identity() -> CanonicalIdentity { + CanonicalIdentity { + kind: "movie", + tmdb_id: Some("504172".into()), + imdb_id: Some("tt4686844".into()), + season: None, + episode: None, + } + } + + fn movie_cut() -> CanonicalCut { + CanonicalCut { + runtime_cs: 642050, + video_hash: Some("opensubtitles:8e245d9679d31e12".into()), + } + } + + fn actors() -> Vec { + vec![CanonicalActor { + tmdb_person_id: 884, + scenes_cs: vec![(19160, 20920), (43820, 46560)], + }] + } + + #[test] + fn canonical_form_matches_the_documented_shape() { + let json = canonical_json(&movie_identity(), &movie_cut(), &actors()); + assert_eq!(json, GOLDEN_VECTORS[0].0); + // Keys sorted, no whitespace (§9a). + assert!(!json.contains(' ')); + } + + #[test] + fn canonical_form_is_valid_json_with_sorted_keys() { + // Hand-built strings are easy to get subtly wrong, so assert the output + // actually parses and that its keys really are ordered. + let json = canonical_json(&movie_identity(), &movie_cut(), &actors()); + let v: serde_json::Value = + serde_json::from_str(&json).expect("canonical form must be JSON"); + let obj = v.as_object().unwrap(); + let keys: Vec<&String> = obj.keys().collect(); + assert_eq!(keys, vec!["actors", "cut", "identity"]); + let id_keys: Vec<&String> = v["identity"].as_object().unwrap().keys().collect(); + assert_eq!(id_keys, vec!["episode", "imdb_id", "season", "tmdb_id", "type"]); + let cut_keys: Vec<&String> = v["cut"].as_object().unwrap().keys().collect(); + assert_eq!(cut_keys, vec!["runtime_cs", "video_hash"]); + } + + #[test] + fn actor_order_does_not_affect_the_hash() { + // §9a: actors sorted by person id, so two servers that stored them in + // different orders still agree. + let a = vec![ + CanonicalActor { tmdb_person_id: 884, scenes_cs: vec![(0, 100)] }, + CanonicalActor { tmdb_person_id: 17419, scenes_cs: vec![(200, 300)] }, + ]; + let b = vec![a[1].clone(), a[0].clone()]; + assert_eq!( + content_id(&movie_identity(), &movie_cut(), &a), + content_id(&movie_identity(), &movie_cut(), &b) + ); + } + + #[test] + fn accumulated_float_error_hashes_identically() { + // The failure mode §9a exists to remove: real corpus values look like + // 8045.066666660665, and two servers may compute them slightly + // differently. Quantising first means both hash the same. + let a = vec![CanonicalActor { + tmdb_person_id: 1, + scenes_cs: vec![(crate::validate::to_centiseconds(8045.066666660665), 900000)], + }]; + let b = vec![CanonicalActor { + tmdb_person_id: 1, + scenes_cs: vec![(crate::validate::to_centiseconds(8045.066666666), 900000)], + }]; + assert_eq!( + content_id(&movie_identity(), &movie_cut(), &a), + content_id(&movie_identity(), &movie_cut(), &b) + ); + } + + #[test] + fn differing_content_produces_differing_ids() { + let base = content_id(&movie_identity(), &movie_cut(), &actors()); + + let mut other_actors = actors(); + other_actors[0].scenes_cs[0].1 += 1; + assert_ne!(base, content_id(&movie_identity(), &movie_cut(), &other_actors)); + + let mut other_cut = movie_cut(); + other_cut.runtime_cs += 1; + assert_ne!(base, content_id(&movie_identity(), &other_cut, &actors())); + + let mut other_id = movie_identity(); + other_id.tmdb_id = Some("999".into()); + assert_ne!(base, content_id(&other_id, &movie_cut(), &actors())); + } + + #[test] + fn episode_and_movie_coordinates_are_distinguished() { + let ep = CanonicalIdentity { + kind: "episode", + tmdb_id: Some("1396".into()), + imdb_id: None, + season: Some(2), + episode: Some(5), + }; + let other = CanonicalIdentity { season: Some(3), ..ep.clone() }; + assert_ne!( + content_id(&ep, &movie_cut(), &actors()), + content_id(&other, &movie_cut(), &actors()) + ); + } + + #[test] + fn content_id_is_prefixed_and_hex() { + let id = content_id(&movie_identity(), &movie_cut(), &actors()); + let hex = id.strip_prefix("sha256:").expect("prefixed"); + assert_eq!(hex.len(), 64); + assert!(hex.bytes().all(|b| b.is_ascii_hexdigit())); + } + + #[test] + fn golden_vector_hash_is_stable() { + // Locks the hash so an accidental change to the canonical form is caught + // here rather than by silent federation divergence. + let id = content_id(&movie_identity(), &movie_cut(), &actors()); + assert_eq!(id, GOLDEN_VECTORS[0].1, "canonical form or hash changed"); + } +} diff --git a/src/db/mod.rs b/src/db/mod.rs new file mode 100644 index 0000000..8806e97 --- /dev/null +++ b/src/db/mod.rs @@ -0,0 +1,168 @@ +//! Database access. +//! +//! §8 imposes two structural requirements that this module exists to satisfy: +//! +//! 1. **A single writer connection, serialized through one owner**, with a read +//! pool alongside. SQLite permits only one writer at a time even in WAL mode; +//! pointing a multi-connection pool at writes and relying on `busy_timeout` +//! to sort it out is explicitly rejected by the spec. Here the writer lives +//! behind a `Mutex`, so contention queues in Rust rather than surfacing as +//! `SQLITE_BUSY`. +//! 2. **All access behind a thin repository layer** rather than queries +//! scattered through handlers — this is what keeps the Turso/Postgres options +//! cheap and localises the serialization in one place. +//! +//! rusqlite is synchronous, so every call is wrapped in `spawn_blocking`: a +//! write that waits on the mutex must never block a Tokio worker thread. + +pub mod repo; + +use std::sync::{Arc, Mutex}; + +use anyhow::Context; +use rusqlite::Connection; + +const SCHEMA: &str = include_str!("schema.sql"); + +/// Handle to the database: one serialized writer, plus read connections. +/// +/// Cloning is cheap and shares the same underlying connections. +#[derive(Clone)] +pub struct Db { + writer: Arc>, + readers: Arc, +} + +struct ReadPool { + conns: Mutex>, + path: String, +} + +impl ReadPool { + fn acquire(&self) -> anyhow::Result { + if let Some(c) = self.conns.lock().expect("read pool poisoned").pop() { + return Ok(c); + } + open_conn(&self.path, false) + } + + fn release(&self, conn: Connection) { + let mut conns = self.conns.lock().expect("read pool poisoned"); + // Bounded: excess connections are dropped rather than accumulating. + if conns.len() < 8 { + conns.push(conn); + } + } +} + +fn open_conn(path: &str, writer: bool) -> anyhow::Result { + let conn = Connection::open(path).with_context(|| format!("opening database {path}"))?; + + // WAL gives concurrent readers alongside the single writer, which suits a + // read-dominated workload; `synchronous = NORMAL` is safe under WAL, and + // `busy_timeout` makes contention wait rather than error (§8). + conn.pragma_update(None, "journal_mode", "WAL")?; + conn.pragma_update(None, "synchronous", "NORMAL")?; + conn.pragma_update(None, "busy_timeout", 5_000)?; + conn.pragma_update(None, "foreign_keys", true)?; + if !writer { + conn.pragma_update(None, "query_only", true)?; + } + Ok(conn) +} + +impl Db { + /// Opens the database, applying the schema. Idempotent — every statement in + /// `schema.sql` is `IF NOT EXISTS`. + pub fn open(path: &str) -> anyhow::Result { + let writer = open_conn(path, true)?; + writer.execute_batch(SCHEMA).context("applying schema")?; + + Ok(Self { + writer: Arc::new(Mutex::new(writer)), + readers: Arc::new(ReadPool { conns: Mutex::new(Vec::new()), path: path.to_string() }), + }) + } + + /// Runs `f` against the serialized writer connection on a blocking thread. + /// + /// `f` receives a `Transaction`, so a manifest's scene rows go in as one + /// transaction rather than one per row (§8), and a failure rolls back. + pub async fn write(&self, f: F) -> anyhow::Result + where + T: Send + 'static, + F: FnOnce(&rusqlite::Transaction<'_>) -> anyhow::Result + Send + 'static, + { + let writer = self.writer.clone(); + tokio::task::spawn_blocking(move || { + let mut conn = writer.lock().expect("writer poisoned"); + let tx = conn.transaction()?; + let out = f(&tx)?; + tx.commit()?; + Ok(out) + }) + .await + .context("writer task panicked")? + } + + /// Runs `f` against a read connection on a blocking thread. + pub async fn read(&self, f: F) -> anyhow::Result + where + T: Send + 'static, + F: FnOnce(&Connection) -> anyhow::Result + Send + 'static, + { + let readers = self.readers.clone(); + tokio::task::spawn_blocking(move || { + let conn = readers.acquire()?; + let out = f(&conn); + readers.release(conn); + out + }) + .await + .context("reader task panicked")? + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn schema_applies_and_roundtrips() { + let db = Db::open(":memory:").unwrap(); + // In-memory databases are per-connection, so only exercise the writer. + let n = db + .write(|tx| { + tx.execute( + "INSERT INTO contributors (id, token_hash, created_at) VALUES (?1, ?2, ?3)", + rusqlite::params!["c1", "hash", "2026-01-01T00:00:00Z"], + )?; + Ok(tx.query_row("SELECT COUNT(*) FROM contributors", [], |r| r.get::<_, i64>(0))?) + }) + .await + .unwrap(); + assert_eq!(n, 1); + } + + #[tokio::test] + async fn write_rolls_back_on_error() { + let db = Db::open(":memory:").unwrap(); + let res: anyhow::Result<()> = db + .write(|tx| { + tx.execute( + "INSERT INTO contributors (id, token_hash, created_at) VALUES ('c1','h','t')", + [], + )?; + anyhow::bail!("deliberate failure") + }) + .await; + assert!(res.is_err()); + let n = db + .write(|tx| { + Ok(tx.query_row("SELECT COUNT(*) FROM contributors", [], |r| r.get::<_, i64>(0))?) + }) + .await + .unwrap(); + assert_eq!(n, 0, "failed transaction must not persist rows"); + } +} diff --git a/src/db/repo.rs b/src/db/repo.rs new file mode 100644 index 0000000..9621928 --- /dev/null +++ b/src/db/repo.rs @@ -0,0 +1,1110 @@ +//! Repository queries. +//! +//! §8: all database access lives behind this layer rather than being scattered +//! through handlers — that is what keeps the Turso/Postgres options cheap and +//! localises the single-writer serialization in one place. +//! +//! Functions here take `&Connection` or `&Transaction` and are synchronous; the +//! `spawn_blocking` boundary is [`crate::db::Db`]'s concern. + +use anyhow::Context; +use rusqlite::{params, Connection, OptionalExtension, Transaction}; + +use crate::model::IdentityType; + +/// A stored manifest row, as needed to serve reads. +#[derive(Debug, Clone)] +pub struct ManifestRow { + pub id: String, + pub title_id: String, + pub season: Option, + pub episode: Option, + pub runtime_sec: f64, + pub video_hash: Option, + pub audio_signature: Option>, + pub sample_fps: Option, + pub extinction_sec: Option, + pub pipeline_version: Option, + pub gallery_scope: Option, + pub status: String, + pub cast_match_ratio: Option, + pub content_id: Option, +} + +const MANIFEST_COLUMNS: &str = "id, title_id, season, episode, runtime_sec, video_hash, \ + audio_signature, sample_fps, extinction_sec, pipeline_version, gallery_scope, \ + status, cast_match_ratio, \ + content_id"; + +fn map_manifest(row: &rusqlite::Row<'_>) -> rusqlite::Result { + Ok(ManifestRow { + id: row.get(0)?, + title_id: row.get(1)?, + season: row.get(2)?, + episode: row.get(3)?, + runtime_sec: row.get(4)?, + video_hash: row.get(5)?, + audio_signature: row.get(6)?, + sample_fps: row.get(7)?, + extinction_sec: row.get(8)?, + pipeline_version: row.get(9)?, + gallery_scope: row.get(10)?, + status: row.get(11)?, + cast_match_ratio: row.get(12)?, + content_id: row.get(13)?, + }) +} + +/// Statuses that are served to clients (§7). +pub const SERVED_STATUSES: &str = "('listed','flagged')"; + +// --------------------------------------------------------------------------- +// Titles +// --------------------------------------------------------------------------- + +#[derive(Debug, Clone)] +pub struct TitleRow { + pub id: String, + pub kind: String, + pub tmdb_id: Option, + pub imdb_id: Option, + pub name: Option, + pub year: Option, + pub adult: bool, + pub certification: Option, +} + +pub fn find_title( + conn: &Connection, + kind: IdentityType, + tmdb_id: Option<&str>, + imdb_id: Option<&str>, +) -> anyhow::Result> { + let kind_str = title_kind(kind); + let mut stmt = conn.prepare_cached( + "SELECT id, kind, tmdb_id, imdb_id, name, year, adult, certification + FROM titles + WHERE kind = ?1 + AND ( (?2 IS NOT NULL AND tmdb_id = ?2) OR (?3 IS NOT NULL AND imdb_id = ?3) ) + LIMIT 1", + )?; + let row = stmt + .query_row(params![kind_str, tmdb_id, imdb_id], |r| { + Ok(TitleRow { + id: r.get(0)?, + kind: r.get(1)?, + tmdb_id: r.get(2)?, + imdb_id: r.get(3)?, + name: r.get(4)?, + year: r.get(5)?, + adult: r.get::<_, i64>(6)? != 0, + certification: r.get(7)?, + }) + }) + .optional()?; + Ok(row) +} + +/// `kind` for the `titles` table: an episode manifest hangs off its *series* +/// title, since §2 keys episodes on series coordinates. +pub fn title_kind(kind: IdentityType) -> &'static str { + match kind { + IdentityType::Movie => "movie", + IdentityType::Episode => "series", + } +} + +/// Finds or creates the title row, returning its id. +pub fn upsert_title( + tx: &Transaction<'_>, + kind: IdentityType, + tmdb_id: Option<&str>, + imdb_id: Option<&str>, + name: Option<&str>, + year: Option, + now: &str, +) -> anyhow::Result { + if let Some(existing) = find_title(tx, kind, tmdb_id, imdb_id)? { + // Backfill identifiers a later upload supplied but an earlier one lacked. + tx.execute( + "UPDATE titles + SET tmdb_id = COALESCE(tmdb_id, ?2), + imdb_id = COALESCE(imdb_id, ?3), + year = COALESCE(year, ?4), + updated_at = ?5 + WHERE id = ?1", + params![existing.id, tmdb_id, imdb_id, year, now], + )?; + return Ok(existing.id); + } + + let id = ulid::Ulid::new().to_string(); + tx.execute( + "INSERT INTO titles (id, kind, tmdb_id, imdb_id, name, year, adult, certification, updated_at) + VALUES (?1, ?2, ?3, ?4, ?5, ?6, 0, NULL, ?7)", + params![id, title_kind(kind), tmdb_id, imdb_id, name, year, now], + )?; + Ok(id) +} + +/// Records TMDB-derived title attributes used by the §5a guards. +pub fn set_title_attributes( + tx: &Transaction<'_>, + title_id: &str, + adult: bool, + certification: Option<&str>, + name: Option<&str>, + now: &str, +) -> anyhow::Result<()> { + tx.execute( + "UPDATE titles + SET adult = ?2, + certification = COALESCE(?3, certification), + name = COALESCE(?4, name), + updated_at = ?5 + WHERE id = ?1", + params![title_id, adult as i64, certification, name, now], + )?; + Ok(()) +} + +// --------------------------------------------------------------------------- +// People — TMDB-derived, never from an upload (§5a, §7) +// --------------------------------------------------------------------------- + +pub fn upsert_person( + tx: &Transaction<'_>, + tmdb_person_id: u64, + name: &str, + adult: bool, + now: &str, +) -> anyhow::Result<()> { + // Portable `INSERT ... ON CONFLICT` rather than `INSERT OR REPLACE` (§8). + tx.execute( + "INSERT INTO people (tmdb_person_id, name, adult, updated_at) + VALUES (?1, ?2, ?3, ?4) + ON CONFLICT (tmdb_person_id) DO UPDATE + SET name = excluded.name, adult = excluded.adult, updated_at = excluded.updated_at", + params![tmdb_person_id as i64, name, adult as i64, now], + )?; + Ok(()) +} + +pub fn person_names( + conn: &Connection, + ids: &[u64], +) -> anyhow::Result> { + let mut out = std::collections::HashMap::new(); + let mut stmt = conn.prepare_cached("SELECT name FROM people WHERE tmdb_person_id = ?1")?; + for &id in ids { + if let Some(name) = + stmt.query_row(params![id as i64], |r| r.get::<_, String>(0)).optional()? + { + out.insert(id, name); + } + } + Ok(out) +} + +// --------------------------------------------------------------------------- +// Manifests +// --------------------------------------------------------------------------- + +/// Every candidate manifest for a title (and episode coordinates, if given) +/// that is eligible to be served. +/// +/// Cut matching is done in [`crate::matching`] rather than in SQL: the tier +/// logic is spec-critical and belongs somewhere testable. +pub fn candidates_for_title( + conn: &Connection, + title_id: &str, + season: Option, + episode: Option, +) -> anyhow::Result> { + let sql = format!( + "SELECT {MANIFEST_COLUMNS} FROM manifests + WHERE title_id = ?1 + AND status IN {SERVED_STATUSES} + AND ( (?2 IS NULL AND season IS NULL) OR season = ?2 ) + AND ( (?3 IS NULL AND episode IS NULL) OR episode = ?3 ) + ORDER BY cast_match_ratio DESC NULLS LAST, sample_fps DESC NULLS LAST, created_at ASC" + ); + let mut stmt = conn.prepare_cached(&sql)?; + let rows = stmt + .query_map(params![title_id, season, episode], map_manifest)? + .collect::>>()?; + Ok(rows) +} + +/// All servable episode manifests for a series, for bundle assembly (§2). +/// +/// A bundle is assembled per request; there is no "series manifest" row. +pub fn episodes_for_series( + conn: &Connection, + title_id: &str, + season: Option, +) -> anyhow::Result> { + let sql = format!( + "SELECT {MANIFEST_COLUMNS} FROM manifests + WHERE title_id = ?1 + AND status IN {SERVED_STATUSES} + AND season IS NOT NULL AND episode IS NOT NULL + AND (?2 IS NULL OR season = ?2) + ORDER BY season ASC, episode ASC, + cast_match_ratio DESC NULLS LAST, sample_fps DESC NULLS LAST" + ); + let mut stmt = conn.prepare_cached(&sql)?; + let rows = stmt + .query_map(params![title_id, season], map_manifest)? + .collect::>>()?; + Ok(rows) +} + +pub fn manifest_by_id(conn: &Connection, id: &str) -> anyhow::Result> { + let sql = format!("SELECT {MANIFEST_COLUMNS} FROM manifests WHERE id = ?1"); + let mut stmt = conn.prepare_cached(&sql)?; + Ok(stmt.query_row(params![id], map_manifest).optional()?) +} + +pub fn manifest_status( + conn: &Connection, + id: &str, +) -> anyhow::Result)>> { + let mut stmt = + conn.prepare_cached("SELECT status, reject_reason FROM manifests WHERE id = ?1")?; + Ok(stmt.query_row(params![id], |r| Ok((r.get(0)?, r.get(1)?))).optional()?) +} + +/// §4 `409`: an identical `(identity, cut)` manifest already exists from this +/// contributor. +pub fn duplicate_from_contributor( + tx: &Transaction<'_>, + title_id: &str, + season: Option, + episode: Option, + runtime_sec: f64, + video_hash: Option<&str>, + contributor_id: &str, +) -> anyhow::Result> { + let mut stmt = tx.prepare_cached( + "SELECT id FROM manifests + WHERE title_id = ?1 AND contributor_id = ?6 + AND status <> 'rejected' + AND ( (?2 IS NULL AND season IS NULL) OR season = ?2 ) + AND ( (?3 IS NULL AND episode IS NULL) OR episode = ?3 ) + AND ABS(runtime_sec - ?4) < 0.001 + AND ( (?5 IS NULL AND video_hash IS NULL) OR video_hash = ?5 ) + LIMIT 1", + )?; + Ok(stmt + .query_row( + params![title_id, season, episode, runtime_sec, video_hash, contributor_id], + |r| r.get::<_, String>(0), + ) + .optional()?) +} + +pub fn manifest_by_content_id( + tx: &Transaction<'_>, + content_id: &str, +) -> anyhow::Result> { + let mut stmt = tx.prepare_cached("SELECT id FROM manifests WHERE content_id = ?1")?; + Ok(stmt.query_row(params![content_id], |r| r.get::<_, String>(0)).optional()?) +} + +/// Everything needed to insert one manifest. +pub struct NewManifest<'a> { + pub id: &'a str, + pub title_id: &'a str, + pub season: Option, + pub episode: Option, + pub runtime_sec: f64, + pub video_hash: Option<&'a str>, + pub audio_signature: Option<&'a [u8]>, + pub audio_sig_coarse: Option<&'a [u8]>, + pub sample_fps: Option, + pub extinction_sec: Option, + pub pipeline_version: Option<&'a str>, + pub gallery_scope: Option<&'a str>, + pub contributor_id: Option<&'a str>, + pub status: &'a str, + pub content_id: Option<&'a str>, + pub origin: &'a str, + pub ingested_from: Option<&'a str>, + pub created_at: &'a str, +} + +pub fn insert_manifest(tx: &Transaction<'_>, m: &NewManifest<'_>) -> anyhow::Result<()> { + tx.execute( + "INSERT INTO manifests + (id, title_id, season, episode, runtime_sec, video_hash, + audio_signature, audio_sig_coarse, sample_fps, extinction_sec, pipeline_version, + gallery_scope, contributor_id, status, content_id, origin, ingested_from, + created_at) + VALUES (?1,?2,?3,?4,?5,?6,?7,?8,?9,?10,?11,?12,?13,?14,?15,?16,?17,?18)", + params![ + m.id, + m.title_id, + m.season, + m.episode, + m.runtime_sec, + m.video_hash, + m.audio_signature, + m.audio_sig_coarse, + m.sample_fps, + m.extinction_sec, + m.pipeline_version, + m.gallery_scope, + m.contributor_id, + m.status, + m.content_id, + m.origin, + m.ingested_from, + m.created_at, + ], + ) + .context("inserting manifest")?; + Ok(()) +} + +/// Inserts one actor's rows. Batched within the caller's transaction — one +/// transaction per manifest, not per row (§8). +pub fn insert_actor_scenes( + tx: &Transaction<'_>, + manifest_id: &str, + tmdb_person_id: u64, + scenes_cs: &[(i64, i64)], +) -> anyhow::Result<()> { + tx.execute( + "INSERT INTO manifest_actors (manifest_id, tmdb_person_id) VALUES (?1, ?2) + ON CONFLICT (manifest_id, tmdb_person_id) DO NOTHING", + params![manifest_id, tmdb_person_id as i64], + )?; + let mut stmt = tx.prepare_cached( + "INSERT INTO scenes (manifest_id, tmdb_person_id, start_cs, end_cs) + VALUES (?1, ?2, ?3, ?4)", + )?; + for (start, end) in scenes_cs { + stmt.execute(params![manifest_id, tmdb_person_id as i64, start, end])?; + } + Ok(()) +} + +/// The actor payload of a stored manifest, reconstructed from rows. +/// +/// §7: the submitted JSON is discarded; the document served to clients is +/// *reconstructed*, never echoed. Names come from `people`, populated from TMDB. +#[derive(Debug, Clone)] +pub struct StoredActor { + pub tmdb_person_id: u64, + pub name: Option, + pub scenes_cs: Vec<(i64, i64)>, +} + +pub fn actors_for_manifest( + conn: &Connection, + manifest_id: &str, +) -> anyhow::Result> { + let mut stmt = conn.prepare_cached( + "SELECT ma.tmdb_person_id, p.name + FROM manifest_actors ma + LEFT JOIN people p ON p.tmdb_person_id = ma.tmdb_person_id + WHERE ma.manifest_id = ?1 + ORDER BY ma.tmdb_person_id ASC", + )?; + let people = stmt + .query_map(params![manifest_id], |r| { + Ok((r.get::<_, i64>(0)? as u64, r.get::<_, Option>(1)?)) + })? + .collect::>>()?; + + let mut scene_stmt = conn.prepare_cached( + "SELECT start_cs, end_cs FROM scenes + WHERE manifest_id = ?1 AND tmdb_person_id = ?2 + ORDER BY start_cs ASC", + )?; + + let mut out = Vec::with_capacity(people.len()); + for (id, name) in people { + let scenes_cs = scene_stmt + .query_map(params![manifest_id, id as i64], |r| Ok((r.get(0)?, r.get(1)?)))? + .collect::>>()?; + out.push(StoredActor { tmdb_person_id: id, name, scenes_cs }); + } + Ok(out) +} + +pub fn set_manifest_status( + tx: &Transaction<'_>, + id: &str, + status: &str, + reason: Option<&str>, + cast_match_ratio: Option, +) -> anyhow::Result<()> { + tx.execute( + "UPDATE manifests + SET status = ?2, reject_reason = ?3, cast_match_ratio = COALESCE(?4, cast_match_ratio) + WHERE id = ?1", + params![id, status, reason, cast_match_ratio], + )?; + Ok(()) +} + +/// §6 stage 3: unmatched actors are dropped rather than stored, which is what +/// closes the free-text channel described in §5a. +pub fn delete_manifest_actor( + tx: &Transaction<'_>, + manifest_id: &str, + tmdb_person_id: u64, +) -> anyhow::Result<()> { + tx.execute( + "DELETE FROM scenes WHERE manifest_id = ?1 AND tmdb_person_id = ?2", + params![manifest_id, tmdb_person_id as i64], + )?; + tx.execute( + "DELETE FROM manifest_actors WHERE manifest_id = ?1 AND tmdb_person_id = ?2", + params![manifest_id, tmdb_person_id as i64], + )?; + Ok(()) +} + +/// §6 stage 3: a rejected manifest is deleted, not merely marked. +pub fn delete_manifest(tx: &Transaction<'_>, id: &str) -> anyhow::Result<()> { + tx.execute("DELETE FROM scenes WHERE manifest_id = ?1", params![id])?; + tx.execute("DELETE FROM manifest_actors WHERE manifest_id = ?1", params![id])?; + tx.execute("DELETE FROM manifests WHERE id = ?1", params![id])?; + Ok(()) +} + +pub fn manifest_actor_ids(conn: &Connection, manifest_id: &str) -> anyhow::Result> { + let mut stmt = stmt_actor_ids(conn)?; + let ids = stmt + .query_map(params![manifest_id], |r| Ok(r.get::<_, i64>(0)? as u64))? + .collect::>>()?; + Ok(ids) +} + +fn stmt_actor_ids(conn: &Connection) -> rusqlite::Result> { + conn.prepare_cached( + "SELECT tmdb_person_id FROM manifest_actors WHERE manifest_id = ?1 ORDER BY tmdb_person_id", + ) +} + +pub fn report_count(conn: &Connection, manifest_id: &str) -> anyhow::Result { + let mut stmt = conn.prepare_cached("SELECT COUNT(*) FROM reports WHERE manifest_id = ?1")?; + Ok(stmt.query_row(params![manifest_id], |r| r.get(0))?) +} + +// --------------------------------------------------------------------------- +// Contributors +// --------------------------------------------------------------------------- + +#[derive(Debug, Clone)] +pub struct Contributor { + pub id: String, + pub revoked: bool, + pub accepted_count: i64, + pub rejected_count: i64, + pub flagged_count: i64, +} + +pub fn contributor_by_token_hash( + conn: &Connection, + token_hash: &str, +) -> anyhow::Result> { + let mut stmt = conn.prepare_cached( + "SELECT id, revoked_at, accepted_count, rejected_count, flagged_count + FROM contributors WHERE token_hash = ?1", + )?; + Ok(stmt + .query_row(params![token_hash], |r| { + Ok(Contributor { + id: r.get(0)?, + revoked: r.get::<_, Option>(1)?.is_some(), + accepted_count: r.get(2)?, + rejected_count: r.get(3)?, + flagged_count: r.get(4)?, + }) + }) + .optional()?) +} + +pub fn insert_contributor( + tx: &Transaction<'_>, + token_hash: &str, + now: &str, +) -> anyhow::Result { + let id = ulid::Ulid::new().to_string(); + tx.execute( + "INSERT INTO contributors (id, token_hash, created_at) VALUES (?1, ?2, ?3)", + params![id, token_hash, now], + )?; + Ok(id) +} + +/// §5a: the only reputational state is per-token counters. +pub fn bump_contributor_counter( + tx: &Transaction<'_>, + contributor_id: &str, + which: &str, +) -> anyhow::Result<()> { + // Column name is from a closed set, never from input. + let sql = match which { + "accepted" => "UPDATE contributors SET accepted_count = accepted_count + 1 WHERE id = ?1", + "rejected" => "UPDATE contributors SET rejected_count = rejected_count + 1 WHERE id = ?1", + "flagged" => "UPDATE contributors SET flagged_count = flagged_count + 1 WHERE id = ?1", + other => anyhow::bail!("unknown contributor counter {other}"), + }; + tx.execute(sql, params![contributor_id])?; + Ok(()) +} + +/// §5a: a token whose rejection rate exceeds a threshold over a minimum sample +/// is revoked automatically, and its `pending`/`flagged` manifests are dropped. +/// No human is in the loop for the common case. +pub const REVOKE_MIN_SAMPLE: i64 = 20; +pub const REVOKE_REJECTION_RATE: f64 = 0.5; + +pub fn maybe_revoke_contributor( + tx: &Transaction<'_>, + contributor_id: &str, + now: &str, +) -> anyhow::Result { + let (accepted, rejected, revoked): (i64, i64, Option) = tx.query_row( + "SELECT accepted_count, rejected_count, revoked_at FROM contributors WHERE id = ?1", + params![contributor_id], + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)), + )?; + if revoked.is_some() { + return Ok(false); + } + let total = accepted + rejected; + if total < REVOKE_MIN_SAMPLE { + return Ok(false); + } + if (rejected as f64) / (total as f64) <= REVOKE_REJECTION_RATE { + return Ok(false); + } + + tx.execute( + "UPDATE contributors SET revoked_at = ?2 WHERE id = ?1", + params![contributor_id, now], + )?; + // Drop the token's pending/flagged manifests. + let ids: Vec = { + let mut stmt = tx.prepare( + "SELECT id FROM manifests WHERE contributor_id = ?1 AND status IN ('pending','flagged')", + )?; + let rows = stmt + .query_map(params![contributor_id], |r| r.get::<_, String>(0))? + .collect::>>()?; + rows + }; + for id in &ids { + delete_manifest(tx, id)?; + } + Ok(true) +} + +// --------------------------------------------------------------------------- +// Reports +// --------------------------------------------------------------------------- + +pub fn insert_report( + tx: &Transaction<'_>, + manifest_id: &str, + reason: &str, + note: Option<&str>, + source_ip_hash: &str, + now: &str, +) -> anyhow::Result { + let id = ulid::Ulid::new().to_string(); + tx.execute( + "INSERT INTO reports (id, manifest_id, reason, note, created_at, source_ip_hash) + VALUES (?1,?2,?3,?4,?5,?6)", + params![id, manifest_id, reason, note, now, source_ip_hash], + )?; + Ok(id) +} + +// --------------------------------------------------------------------------- +// TMDB cache — the sole JSON column, holding TMDB's responses, not users' (§7) +// --------------------------------------------------------------------------- + +pub fn cached_credits( + conn: &Connection, + tmdb_id: &str, + kind: &str, +) -> anyhow::Result> { + let mut stmt = conn.prepare_cached( + "SELECT credits, fetched_at FROM tmdb_cache WHERE tmdb_id = ?1 AND kind = ?2", + )?; + Ok(stmt.query_row(params![tmdb_id, kind], |r| Ok((r.get(0)?, r.get(1)?))).optional()?) +} + +pub fn put_credits( + tx: &Transaction<'_>, + tmdb_id: &str, + kind: &str, + credits: &str, + now: &str, +) -> anyhow::Result<()> { + tx.execute( + "INSERT INTO tmdb_cache (tmdb_id, kind, credits, fetched_at) VALUES (?1,?2,?3,?4) + ON CONFLICT (tmdb_id, kind) DO UPDATE + SET credits = excluded.credits, fetched_at = excluded.fetched_at", + params![tmdb_id, kind, credits, now], + )?; + Ok(()) +} + +// --------------------------------------------------------------------------- +// Jobs — a table rather than an external broker, so work survives restart (§7) +// --------------------------------------------------------------------------- + +#[derive(Debug, Clone)] +pub struct Job { + pub id: String, + pub kind: String, + pub payload: String, + pub attempts: i64, +} + +pub fn enqueue_job( + tx: &Transaction<'_>, + kind: &str, + payload: &str, + run_after: &str, +) -> anyhow::Result { + let id = ulid::Ulid::new().to_string(); + tx.execute( + "INSERT INTO jobs (id, kind, payload, run_after) VALUES (?1,?2,?3,?4)", + params![id, kind, payload, run_after], + )?; + Ok(id) +} + +/// Claims up to `limit` due jobs, marking them leased so a second worker tick +/// cannot pick up the same work. +pub fn lease_jobs(tx: &Transaction<'_>, now: &str, limit: usize) -> anyhow::Result> { + let jobs: Vec = { + let mut stmt = tx.prepare( + "SELECT id, kind, payload, attempts FROM jobs + WHERE run_after <= ?1 AND leased_at IS NULL + ORDER BY run_after ASC + LIMIT ?2", + )?; + let rows = stmt + .query_map(params![now, limit as i64], |r| { + Ok(Job { id: r.get(0)?, kind: r.get(1)?, payload: r.get(2)?, attempts: r.get(3)? }) + })? + .collect::>>()?; + rows + }; + for j in &jobs { + tx.execute("UPDATE jobs SET leased_at = ?2 WHERE id = ?1", params![j.id, now])?; + } + Ok(jobs) +} + +pub fn delete_job(tx: &Transaction<'_>, id: &str) -> anyhow::Result<()> { + tx.execute("DELETE FROM jobs WHERE id = ?1", params![id])?; + Ok(()) +} + +/// Releases a leased job for a later retry with backoff (§6 stage 3: TMDB +/// unreachable means retry, not reject). +pub fn reschedule_job( + tx: &Transaction<'_>, + id: &str, + run_after: &str, + error: &str, +) -> anyhow::Result<()> { + tx.execute( + "UPDATE jobs SET attempts = attempts + 1, last_error = ?3, run_after = ?2, leased_at = NULL + WHERE id = ?1", + params![id, run_after, error], + )?; + Ok(()) +} + +/// Frees leases held at startup — a process that died mid-job would otherwise +/// leave work stranded. +pub fn release_all_leases(tx: &Transaction<'_>) -> anyhow::Result { + let n = tx.execute("UPDATE jobs SET leased_at = NULL WHERE leased_at IS NOT NULL", [])?; + Ok(n) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::db::Db; + + async fn db() -> Db { + Db::open(":memory:").unwrap() + } + + const NOW: &str = "2026-07-30T12:00:00Z"; + + #[tokio::test] + async fn upsert_title_is_idempotent_and_backfills() { + let db = db().await; + let (a, b) = db + .write(|tx| { + let a = upsert_title( + tx, + IdentityType::Movie, + Some("504172"), + None, + Some("X"), + Some(2017), + NOW, + )?; + // Second upload supplies the IMDB id the first lacked. + let b = upsert_title( + tx, + IdentityType::Movie, + Some("504172"), + Some("tt4686844"), + None, + None, + NOW, + )?; + Ok((a, b)) + }) + .await + .unwrap(); + assert_eq!(a, b, "same title must not be duplicated"); + + let found = db + .write(|tx| find_title(tx, IdentityType::Movie, Some("504172"), None)) + .await + .unwrap() + .unwrap(); + assert_eq!(found.imdb_id.as_deref(), Some("tt4686844"), "identifier should be backfilled"); + } + + #[tokio::test] + async fn movie_and_series_titles_do_not_collide_on_id() { + // TMDB numbers movies and series in separate spaces, so id 1396 is both + // a film and Breaking Bad. + let db = db().await; + let (m, s) = db + .write(|tx| { + let m = upsert_title(tx, IdentityType::Movie, Some("1396"), None, None, None, NOW)?; + let s = + upsert_title(tx, IdentityType::Episode, Some("1396"), None, None, None, NOW)?; + Ok((m, s)) + }) + .await + .unwrap(); + assert_ne!(m, s); + } + + #[tokio::test] + async fn manifest_roundtrips_through_rows() { + let db = db().await; + let id = db + .write(|tx| { + let title_id = + upsert_title(tx, IdentityType::Movie, Some("1"), None, None, None, NOW)?; + let cid = insert_contributor(tx, "hash", NOW)?; + let id = ulid::Ulid::new().to_string(); + insert_manifest( + tx, + &NewManifest { + id: &id, + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 6420.5, + video_hash: Some("opensubtitles:8e245d9679d31e12"), + audio_signature: None, + audio_sig_coarse: None, + sample_fps: Some(5.0), + extinction_sec: Some(12.0), + gallery_scope: Some("global"), + pipeline_version: Some("test 0.1"), + contributor_id: Some(&cid), + status: "listed", + content_id: Some("sha256:abc"), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + upsert_person(tx, 884, "Steve Buscemi", false, NOW)?; + insert_actor_scenes(tx, &id, 884, &[(19160, 20920), (43820, 46560)])?; + Ok(id) + }) + .await + .unwrap(); + + let (row, actors) = db + .write(move |tx| { + let row = manifest_by_id(tx, &id)?.unwrap(); + let actors = actors_for_manifest(tx, &id)?; + Ok((row, actors)) + }) + .await + .unwrap(); + + assert_eq!(row.runtime_sec, 6420.5); + assert_eq!(actors.len(), 1); + assert_eq!(actors[0].tmdb_person_id, 884); + // §7: names come from `people`, populated from TMDB, never from upload. + assert_eq!(actors[0].name.as_deref(), Some("Steve Buscemi")); + assert_eq!(actors[0].scenes_cs, vec![(19160, 20920), (43820, 46560)]); + } + + #[tokio::test] + async fn only_served_statuses_are_candidates() { + let db = db().await; + let title_id = db + .write(|tx| { + let title_id = + upsert_title(tx, IdentityType::Movie, Some("7"), None, None, None, NOW)?; + for (i, status) in ["listed", "flagged", "pending", "rejected"].iter().enumerate() { + insert_manifest( + tx, + &NewManifest { + id: &format!("m{i}"), + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 100.0, + video_hash: None, + audio_signature: None, + audio_sig_coarse: None, + sample_fps: None, + extinction_sec: None, + gallery_scope: None, + pipeline_version: None, + contributor_id: None, + status, + content_id: Some(&format!("c{i}")), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + } + Ok(title_id) + }) + .await + .unwrap(); + + let rows = + db.write(move |tx| candidates_for_title(tx, &title_id, None, None)).await.unwrap(); + // `pending` is held unlisted and not served to anyone (§6 stage 3); + // `rejected` never is. + assert_eq!(rows.len(), 2); + assert!(rows.iter().all(|r| r.status == "listed" || r.status == "flagged")); + } + + #[tokio::test] + async fn duplicate_detection_is_per_contributor() { + let db = db().await; + let dup = db + .write(|tx| { + let title_id = + upsert_title(tx, IdentityType::Movie, Some("9"), None, None, None, NOW)?; + let a = insert_contributor(tx, "hash-a", NOW)?; + let b = insert_contributor(tx, "hash-b", NOW)?; + insert_manifest( + tx, + &NewManifest { + id: "m1", + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 6420.5, + video_hash: None, + audio_signature: None, + audio_sig_coarse: None, + sample_fps: None, + extinction_sec: None, + gallery_scope: None, + pipeline_version: None, + contributor_id: Some(&a), + status: "listed", + content_id: Some("c1"), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + let same = duplicate_from_contributor(tx, &title_id, None, None, 6420.5, None, &a)?; + // §7: multiple manifests for the same cut from *different* + // contributors are allowed, and ranked. + let other = + duplicate_from_contributor(tx, &title_id, None, None, 6420.5, None, &b)?; + Ok((same.is_some(), other.is_some())) + }) + .await + .unwrap(); + assert_eq!(dup, (true, false)); + } + + #[tokio::test] + async fn revocation_needs_a_minimum_sample_then_drops_pending_work() { + let db = db().await; + let revoked_early = db + .write(|tx| { + let c = insert_contributor(tx, "h", NOW)?; + for _ in 0..5 { + bump_contributor_counter(tx, &c, "rejected")?; + } + // §5a: a threshold over a *minimum sample* — 5 rejections is not + // yet evidence. + let early = maybe_revoke_contributor(tx, &c, NOW)?; + + for _ in 0..20 { + bump_contributor_counter(tx, &c, "rejected")?; + } + let title_id = + upsert_title(tx, IdentityType::Movie, Some("11"), None, None, None, NOW)?; + insert_manifest( + tx, + &NewManifest { + id: "pend", + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 100.0, + video_hash: None, + audio_signature: None, + audio_sig_coarse: None, + sample_fps: None, + extinction_sec: None, + gallery_scope: None, + pipeline_version: None, + contributor_id: Some(&c), + status: "pending", + content_id: Some("cp"), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + let now_revoked = maybe_revoke_contributor(tx, &c, NOW)?; + let still_there = manifest_by_id(tx, "pend")?.is_some(); + Ok((early, now_revoked, still_there)) + }) + .await + .unwrap(); + assert_eq!(revoked_early, (false, true, false)); + } + + #[tokio::test] + async fn good_contributors_are_not_revoked() { + let db = db().await; + let revoked = db + .write(|tx| { + let c = insert_contributor(tx, "h", NOW)?; + for _ in 0..40 { + bump_contributor_counter(tx, &c, "accepted")?; + } + for _ in 0..3 { + bump_contributor_counter(tx, &c, "rejected")?; + } + maybe_revoke_contributor(tx, &c, NOW) + }) + .await + .unwrap(); + assert!(!revoked); + } + + #[tokio::test] + async fn jobs_lease_once_and_reschedule() { + let db = db().await; + let (first, second, after_reschedule) = db + .write(|tx| { + enqueue_job(tx, "cast_check", "{}", NOW)?; + let first = lease_jobs(tx, NOW, 10)?; + // A leased job must not be handed out twice. + let second = lease_jobs(tx, NOW, 10)?; + reschedule_job(tx, &first[0].id, NOW, "tmdb unreachable")?; + let third = lease_jobs(tx, NOW, 10)?; + Ok((first.len(), second.len(), third)) + }) + .await + .unwrap(); + assert_eq!((first, second), (1, 0)); + assert_eq!(after_reschedule.len(), 1); + assert_eq!(after_reschedule[0].attempts, 1); + } + + #[tokio::test] + async fn future_jobs_are_not_leased() { + let db = db().await; + let n = db + .write(|tx| { + enqueue_job(tx, "cast_check", "{}", "2099-01-01T00:00:00Z")?; + Ok(lease_jobs(tx, NOW, 10)?.len()) + }) + .await + .unwrap(); + assert_eq!(n, 0); + } + + #[tokio::test] + async fn startup_releases_stranded_leases() { + let db = db().await; + let n = db + .write(|tx| { + enqueue_job(tx, "cast_check", "{}", NOW)?; + lease_jobs(tx, NOW, 10)?; + release_all_leases(tx)?; + Ok(lease_jobs(tx, NOW, 10)?.len()) + }) + .await + .unwrap(); + assert_eq!(n, 1); + } + + #[tokio::test] + async fn deleting_a_manifest_removes_its_rows() { + let db = db().await; + let counts = db + .write(|tx| { + let title_id = + upsert_title(tx, IdentityType::Movie, Some("13"), None, None, None, NOW)?; + insert_manifest( + tx, + &NewManifest { + id: "m", + title_id: &title_id, + season: None, + episode: None, + runtime_sec: 100.0, + video_hash: None, + audio_signature: None, + audio_sig_coarse: None, + sample_fps: None, + extinction_sec: None, + gallery_scope: None, + pipeline_version: None, + contributor_id: None, + status: "listed", + content_id: Some("c"), + origin: "local", + ingested_from: None, + created_at: NOW, + }, + )?; + upsert_person(tx, 1, "A", false, NOW)?; + insert_actor_scenes(tx, "m", 1, &[(0, 100)])?; + delete_manifest(tx, "m")?; + let scenes: i64 = tx.query_row("SELECT COUNT(*) FROM scenes", [], |r| r.get(0))?; + let actors: i64 = + tx.query_row("SELECT COUNT(*) FROM manifest_actors", [], |r| r.get(0))?; + let manifests: i64 = + tx.query_row("SELECT COUNT(*) FROM manifests", [], |r| r.get(0))?; + Ok((scenes, actors, manifests)) + }) + .await + .unwrap(); + assert_eq!(counts, (0, 0, 0)); + } +} diff --git a/src/db/schema.sql b/src/db/schema.sql new file mode 100644 index 0000000..3a2afe4 --- /dev/null +++ b/src/db/schema.sql @@ -0,0 +1,119 @@ +-- §7 Storage. Fully relational, no JSON blobs on the write path: the database +-- can only represent what the schema models, so there is physically nowhere for +-- an unexpected field or a smuggled string to live (§5a Threat 1). +-- +-- Portable SQL — runs unchanged on Postgres. Avoid SQLite-specific forms +-- (`INSERT OR REPLACE`); use `INSERT ... ON CONFLICT` (§8 deployment notes). + +CREATE TABLE IF NOT EXISTS contributors ( + id TEXT PRIMARY KEY, + token_hash TEXT NOT NULL UNIQUE, + created_at TEXT NOT NULL, + revoked_at TEXT, + accepted_count INTEGER NOT NULL DEFAULT 0, + rejected_count INTEGER NOT NULL DEFAULT 0, + flagged_count INTEGER NOT NULL DEFAULT 0 +); + +-- Server-side, TMDB-derived. `name` never comes from an upload (§5a). +CREATE TABLE IF NOT EXISTS people ( + tmdb_person_id INTEGER PRIMARY KEY, + name TEXT NOT NULL, + adult INTEGER NOT NULL DEFAULT 0, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS titles ( + id TEXT PRIMARY KEY, + kind TEXT NOT NULL, -- movie | series + tmdb_id TEXT, + imdb_id TEXT, + name TEXT, + year INTEGER, + adult INTEGER NOT NULL DEFAULT 0, + certification TEXT, + updated_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS manifests ( + id TEXT PRIMARY KEY, + title_id TEXT NOT NULL REFERENCES titles(id), + season INTEGER, + episode INTEGER, + runtime_sec REAL NOT NULL, + video_hash TEXT, + audio_signature BLOB, -- §3, ~1290 bytes + audio_sig_coarse BLOB, -- candidate-generation index key + sample_fps REAL, + extinction_sec REAL, -- successor to the withdrawn anneal_sec + gallery_scope TEXT, -- limited | global; ranking signal (§2, §7) + pipeline_version TEXT, + contributor_id TEXT REFERENCES contributors(id), + status TEXT NOT NULL, -- pending | listed | flagged | rejected + reject_reason TEXT, + cast_match_ratio REAL, + content_id TEXT UNIQUE, -- §9a, sha256 over canonical form + origin TEXT, -- server_id of first acceptance + ingested_from TEXT, -- peer id, NULL if uploaded directly + created_at TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS manifest_actors ( + manifest_id TEXT NOT NULL REFERENCES manifests(id) ON DELETE CASCADE, + tmdb_person_id INTEGER NOT NULL, + PRIMARY KEY (manifest_id, tmdb_person_id) +); + +-- Integer centiseconds, not floats — the same quantisation used for +-- `content_id`, so stored values and hashed values cannot diverge (§7, §9a). +CREATE TABLE IF NOT EXISTS scenes ( + manifest_id TEXT NOT NULL REFERENCES manifests(id) ON DELETE CASCADE, + tmdb_person_id INTEGER NOT NULL, + start_cs INTEGER NOT NULL, + end_cs INTEGER NOT NULL +); + +CREATE TABLE IF NOT EXISTS reports ( + id TEXT PRIMARY KEY, + manifest_id TEXT NOT NULL REFERENCES manifests(id) ON DELETE CASCADE, + reason TEXT NOT NULL, + note TEXT, + created_at TEXT NOT NULL, + source_ip_hash TEXT +); + +-- The sole JSON column, and it holds TMDB's responses, not users' (§7). +CREATE TABLE IF NOT EXISTS tmdb_cache ( + tmdb_id TEXT NOT NULL, + kind TEXT NOT NULL, + credits TEXT NOT NULL, + fetched_at TEXT NOT NULL, + PRIMARY KEY (tmdb_id, kind) +); + +-- Background queue as a table rather than an external broker, so pending work +-- survives a restart (§7, §8). +CREATE TABLE IF NOT EXISTS jobs ( + id TEXT PRIMARY KEY, + kind TEXT NOT NULL, -- cast_check | federation_pull + payload TEXT NOT NULL, + run_after TEXT NOT NULL, + attempts INTEGER NOT NULL DEFAULT 0, + last_error TEXT, + leased_at TEXT +); + +CREATE INDEX IF NOT EXISTS idx_titles_tmdb ON titles(tmdb_id); +CREATE INDEX IF NOT EXISTS idx_titles_imdb ON titles(imdb_id); +CREATE INDEX IF NOT EXISTS idx_manifests_title_runtime ON manifests(title_id, runtime_sec); +CREATE INDEX IF NOT EXISTS idx_manifests_video_hash ON manifests(video_hash); +CREATE INDEX IF NOT EXISTS idx_manifests_episode ON manifests(title_id, season, episode); +CREATE INDEX IF NOT EXISTS idx_scenes_manifest_person ON scenes(manifest_id, tmdb_person_id); + +-- All read queries filter `status IN ('listed','flagged')`, so a partial index +-- on that predicate keeps the hot path small (§7). +CREATE INDEX IF NOT EXISTS idx_manifests_served + ON manifests(title_id, season, episode) + WHERE status IN ('listed', 'flagged'); + +CREATE INDEX IF NOT EXISTS idx_jobs_ready ON jobs(run_after); diff --git a/src/error.rs b/src/error.rs new file mode 100644 index 0000000..f5043d5 --- /dev/null +++ b/src/error.rs @@ -0,0 +1,83 @@ +//! API error type mapping onto the status codes §4 specifies. + +use axum::http::StatusCode; +use axum::response::{IntoResponse, Response}; +use axum::Json; +use serde::Serialize; + +#[derive(Debug, thiserror::Error)] +pub enum ApiError { + /// §6 stage 2 — malformed, unrecognised or forbidden field. The message + /// names the offending field so a client that forgets to strip `movie` or + /// `jellyfin_id` gets a hard, diagnosable `400` (§6). + #[error("{0}")] + BadRequest(String), + + #[error("not found")] + NotFound, + + /// §4 — identical `(identity, cut)` already exists from this contributor. + #[error("{0}")] + Conflict(String), + + #[error("{0}")] + PayloadTooLarge(String), + + #[error("missing or invalid API token")] + Unauthorized, + + /// §5 — carries the `Retry-After` value in seconds. + #[error("rate limited")] + RateLimited { retry_after: u64 }, + + #[error("internal error")] + Internal(#[from] anyhow::Error), +} + +#[derive(Serialize)] +struct ErrorBody { + error: String, + message: String, +} + +impl IntoResponse for ApiError { + fn into_response(self) -> Response { + let (status, code) = match &self { + ApiError::BadRequest(_) => (StatusCode::BAD_REQUEST, "bad_request"), + ApiError::NotFound => (StatusCode::NOT_FOUND, "not_found"), + ApiError::Conflict(_) => (StatusCode::CONFLICT, "conflict"), + ApiError::PayloadTooLarge(_) => (StatusCode::PAYLOAD_TOO_LARGE, "payload_too_large"), + ApiError::Unauthorized => (StatusCode::UNAUTHORIZED, "unauthorized"), + ApiError::RateLimited { .. } => (StatusCode::TOO_MANY_REQUESTS, "rate_limited"), + ApiError::Internal(e) => { + // Internal detail is logged, never returned. + tracing::error!(error = ?e, "internal error"); + (StatusCode::INTERNAL_SERVER_ERROR, "internal") + } + }; + + let body = Json(ErrorBody { + error: code.to_string(), + message: match &self { + ApiError::Internal(_) => "internal error".to_string(), + other => other.to_string(), + }, + }); + + let mut resp = (status, body).into_response(); + if let ApiError::RateLimited { retry_after } = self { + if let Ok(v) = retry_after.to_string().parse() { + resp.headers_mut().insert(axum::http::header::RETRY_AFTER, v); + } + } + resp + } +} + +impl From for ApiError { + fn from(e: rusqlite::Error) -> Self { + ApiError::Internal(anyhow::Error::new(e)) + } +} + +pub type ApiResult = Result; diff --git a/src/ingest.rs b/src/ingest.rs new file mode 100644 index 0000000..1fede64 --- /dev/null +++ b/src/ingest.rs @@ -0,0 +1,403 @@ +//! Manifest ingestion: the shared path behind `POST /manifests` and +//! `POST /manifests/bundle`, and the path a federation pull will reuse (§9a +//! "re-derive, don't inherit"). +//! +//! Stages 0 and 1 are layers; stage 2 is parse + [`crate::validate`]. What +//! happens here is persistence plus enqueueing the stage 3 check: the upload is +//! accepted with `202` and the manifest is held **unlisted** until the cast check +//! completes — it is not served to anyone in the meantime (§6). + +use anyhow::Context; + +use crate::content_id::{self, CanonicalActor, CanonicalCut, CanonicalIdentity}; +use crate::db::repo::{self, NewManifest}; +use crate::model::IdentityType; +use crate::validate::ValidManifest; + +/// Outcome of persisting one manifest. +#[derive(Debug, Clone)] +pub enum IngestOutcome { + /// Held unlisted pending the §6 stage 3 cast check. + Pending { manifest_id: String }, + /// §4 `409` — identical `(identity, cut)` from this contributor. + DuplicateFromContributor { manifest_id: String }, + /// §9a — the exact same content is already held, from any source. Skipped + /// without re-validation, which is the deduplication content addressing buys. + DuplicateContent { manifest_id: String }, +} + +impl IngestOutcome { + pub fn manifest_id(&self) -> &str { + match self { + IngestOutcome::Pending { manifest_id } + | IngestOutcome::DuplicateFromContributor { manifest_id } + | IngestOutcome::DuplicateContent { manifest_id } => manifest_id, + } + } +} + +/// Job payload for the §6 stage 3 check. +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct CastCheckJob { + pub manifest_id: String, +} + +pub const JOB_CAST_CHECK: &str = "cast_check"; + +/// Persists a validated manifest and enqueues its cast check, all in one +/// transaction — so a manifest is never left listed-but-unchecked, and its scene +/// rows go in as a single transaction rather than one per row (§8). +pub fn persist( + tx: &rusqlite::Transaction<'_>, + valid: &ValidManifest, + contributor_id: Option<&str>, + origin: &str, + ingested_from: Option<&str>, + now: &str, +) -> anyhow::Result { + let m = &valid.manifest; + let kind = m.identity.kind; + let tmdb_id = m.identity.effective_tmdb_id(); + let imdb_id = m.identity.effective_imdb_id(); + + let title_id = repo::upsert_title( + tx, + kind, + tmdb_id, + imdb_id, + m.identity.title.as_deref(), + m.identity.year, + now, + ) + .context("resolving title")?; + + let (season, episode) = match kind { + IdentityType::Movie => (None, None), + IdentityType::Episode => (m.identity.season, m.identity.episode), + }; + + // Content addressing over the *submitted* actor ids. Recomputed after the + // cast check drops unmatched actors, since dropping changes the content. + let cid = compute_content_id(valid); + + if let Some(existing) = repo::manifest_by_content_id(tx, &cid)? { + return Ok(IngestOutcome::DuplicateContent { manifest_id: existing }); + } + + if let Some(c) = contributor_id { + if let Some(existing) = repo::duplicate_from_contributor( + tx, + &title_id, + season, + episode, + m.cut.runtime_sec, + m.cut.video_hash.as_deref(), + c, + )? { + return Ok(IngestOutcome::DuplicateFromContributor { manifest_id: existing }); + } + } + + let manifest_id = ulid::Ulid::new().to_string(); + let extraction = m.extraction.as_ref(); + + repo::insert_manifest( + tx, + &NewManifest { + id: &manifest_id, + title_id: &title_id, + season, + episode, + runtime_sec: m.cut.runtime_sec, + video_hash: m.cut.video_hash.as_deref(), + // Stored as an attribute, not part of identity (§9a). + audio_signature: None, + audio_sig_coarse: None, + sample_fps: extraction.and_then(|e| e.sample_fps), + extinction_sec: extraction.and_then(|e| e.extinction_sec), + pipeline_version: extraction.and_then(|e| e.pipeline_version.as_deref()), + gallery_scope: extraction.and_then(|e| e.gallery_scope).map(|g| g.as_str()), + contributor_id, + // Held unlisted until stage 3 completes (§6). + status: "pending", + content_id: Some(&cid), + origin, + ingested_from, + created_at: now, + }, + )?; + + // Actors are recorded by TMDB person id only. Those without one cannot be + // stored at all — there is no name column to put them in (§5a, §7) — so they + // are carried into the cast check via the submitted payload instead. + for actor in &valid.actor_scenes_cs { + if let Some(person_id) = actor.tmdb_id { + repo::insert_actor_scenes(tx, &manifest_id, person_id, &actor.scenes_cs)?; + } + } + + let payload = serde_json::to_string(&CastCheckJob { manifest_id: manifest_id.clone() })?; + repo::enqueue_job(tx, JOB_CAST_CHECK, &payload, now)?; + + Ok(IngestOutcome::Pending { manifest_id }) +} + +/// Computes the §9a `content_id` for a validated manifest. +pub fn compute_content_id(valid: &ValidManifest) -> String { + let m = &valid.manifest; + let identity = CanonicalIdentity { + kind: match m.identity.kind { + IdentityType::Movie => "movie", + IdentityType::Episode => "episode", + }, + tmdb_id: m.identity.effective_tmdb_id().map(str::to_string), + imdb_id: m.identity.effective_imdb_id().map(str::to_string), + season: m.identity.season, + episode: m.identity.episode, + }; + let cut = CanonicalCut { + runtime_cs: crate::validate::to_centiseconds(m.cut.runtime_sec), + video_hash: m.cut.video_hash.clone(), + }; + let actors: Vec = valid + .actor_scenes_cs + .iter() + .filter_map(|a| { + a.tmdb_id + .map(|id| CanonicalActor { tmdb_person_id: id, scenes_cs: a.scenes_cs.clone() }) + }) + .collect(); + + content_id::content_id(&identity, &cut, &actors) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::db::Db; + use crate::model::Jmanifest; + use crate::validate::validate_manifest; + + const NOW: &str = "2026-07-30T12:00:00Z"; + + fn valid_from(json: &str) -> ValidManifest { + let m: Jmanifest = serde_json::from_str(json).unwrap(); + validate_manifest(m).unwrap() + } + + fn movie_json(tmdb: &str, runtime: f64) -> String { + format!( + r#"{{"jmanifest_version":1, + "identity":{{"type":"movie","tmdb_id":"{tmdb}","title":"A Film"}}, + "cut":{{"runtime_sec":{runtime}}}, + "extraction":{{"sample_fps":5,"pipeline_version":"test 0.1"}}, + "actors":[{{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[10.0,20.0]]}}, + {{"name":"Michael Palin","tmdb_id":"11007","scenes":[[30.0,40.0]]}}]}}"# + ) + } + + #[tokio::test] + async fn persists_as_pending_and_enqueues_a_check() { + let db = Db::open(":memory:").unwrap(); + let valid = valid_from(&movie_json("504172", 6420.5)); + + let (outcome, status, jobs) = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + let outcome = persist(tx, &valid, Some(&c), "local", None, NOW)?; + let status = repo::manifest_status(tx, outcome.manifest_id())?; + let jobs = repo::lease_jobs(tx, NOW, 10)?; + Ok((outcome, status, jobs)) + }) + .await + .unwrap(); + + assert!(matches!(outcome, IngestOutcome::Pending { .. })); + // §6: held unlisted, not served to anyone, until stage 3 completes. + assert_eq!(status.unwrap().0, "pending"); + assert_eq!(jobs.len(), 1); + assert_eq!(jobs[0].kind, JOB_CAST_CHECK); + } + + #[tokio::test] + async fn a_pending_manifest_is_not_served() { + let db = Db::open(":memory:").unwrap(); + let valid = valid_from(&movie_json("504172", 6420.5)); + let candidates = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + persist(tx, &valid, Some(&c), "local", None, NOW)?; + let title = + repo::find_title(tx, IdentityType::Movie, Some("504172"), None)?.unwrap(); + repo::candidates_for_title(tx, &title.id, None, None) + }) + .await + .unwrap(); + assert!(candidates.is_empty()); + } + + #[tokio::test] + async fn identical_content_deduplicates() { + // §9a: a manifest whose `content_id` is already present is skipped + // without re-validation. + let db = Db::open(":memory:").unwrap(); + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from(&movie_json("504172", 6420.5)); + + let (first, second) = db + .write(move |tx| { + let c1 = repo::insert_contributor(tx, "h1", NOW)?; + let c2 = repo::insert_contributor(tx, "h2", NOW)?; + let first = persist(tx, &a, Some(&c1), "local", None, NOW)?; + // A *different* contributor, so this is content dedup, not the + // per-contributor 409. + let second = persist(tx, &b, Some(&c2), "local", None, NOW)?; + Ok((first, second)) + }) + .await + .unwrap(); + + assert!(matches!(first, IngestOutcome::Pending { .. })); + assert!(matches!(second, IngestOutcome::DuplicateContent { .. })); + assert_eq!(first.manifest_id(), second.manifest_id()); + } + + #[tokio::test] + async fn same_contributor_resubmitting_the_same_cut_is_a_duplicate() { + let db = Db::open(":memory:").unwrap(); + // Same identity and cut, different actor timings => different content_id, + // so this exercises the per-contributor 409 path specifically. + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from( + r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172","title":"A Film"}, + "cut":{"runtime_sec":6420.5}, + "actors":[{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[11.0,21.0]]}]}"#, + ); + + let second = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + persist(tx, &a, Some(&c), "local", None, NOW)?; + persist(tx, &b, Some(&c), "local", None, NOW) + }) + .await + .unwrap(); + + assert!(matches!(second, IngestOutcome::DuplicateFromContributor { .. })); + } + + #[tokio::test] + async fn different_cuts_of_one_title_coexist() { + // §7: multiple manifests may coexist for the same title with different + // cuts — that is the point. + let db = Db::open(":memory:").unwrap(); + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from(&movie_json("504172", 7000.0)); + + let (x, y) = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + let x = persist(tx, &a, Some(&c), "local", None, NOW)?; + let y = persist(tx, &b, Some(&c), "local", None, NOW)?; + Ok((x, y)) + }) + .await + .unwrap(); + + assert!(matches!(x, IngestOutcome::Pending { .. })); + assert!(matches!(y, IngestOutcome::Pending { .. })); + assert_ne!(x.manifest_id(), y.manifest_id()); + } + + #[tokio::test] + async fn episode_manifests_carry_their_coordinates() { + let db = Db::open(":memory:").unwrap(); + let valid = valid_from( + r#"{"jmanifest_version":1, + "identity":{"type":"episode","series_tmdb_id":"1396","title":"Breaking Bad", + "season":2,"episode":5}, + "cut":{"runtime_sec":2820.0}, + "actors":[{"name":"Bryan Cranston","tmdb_id":"17419","scenes":[[10.0,20.0]]}]}"#, + ); + let row = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + let o = persist(tx, &valid, Some(&c), "local", None, NOW)?; + Ok(repo::manifest_by_id(tx, o.manifest_id())?.unwrap()) + }) + .await + .unwrap(); + assert_eq!((row.season, row.episode), (Some(2), Some(5))); + } + + #[tokio::test] + async fn upload_metadata_is_not_echoed_back_as_actor_names() { + // §5a/§7: only integers reach the database. The submitted name is used + // for matching and never persisted, so before the cast check populates + // `people` there is no name to serve. + let db = Db::open(":memory:").unwrap(); + let valid = valid_from(&movie_json("504172", 6420.5)); + let actors = db + .write(move |tx| { + let c = repo::insert_contributor(tx, "h", NOW)?; + let o = persist(tx, &valid, Some(&c), "local", None, NOW)?; + repo::actors_for_manifest(tx, o.manifest_id()) + }) + .await + .unwrap(); + assert_eq!(actors.len(), 2); + assert!(actors.iter().all(|a| a.name.is_none())); + } + + #[test] + fn content_id_excludes_extraction_metadata() { + // §9a: `extraction` metadata and local state are excluded, so two + // servers validating the same upload agree. + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from( + r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172","title":"A Film"}, + "cut":{"runtime_sec":6420.5}, + "extraction":{"sample_fps":1,"extinction_sec":9,"pipeline_version":"other 9.9", + "gallery_size":5}, + "actors":[{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[10.0,20.0]]}, + {"name":"Michael Palin","tmdb_id":"11007","scenes":[[30.0,40.0]]}]}"#, + ); + assert_eq!(compute_content_id(&a), compute_content_id(&b)); + } + + #[test] + fn content_id_excludes_the_audio_signature() { + // §9a is explicit: including it would produce different content_ids for + // identical content and silently break federation deduplication. + let a = valid_from(&movie_json("504172", 6420.5)); + let sig = format!("v1:{}", "A".repeat(1720)); + let with_sig = format!( + r#"{{"jmanifest_version":1, + "identity":{{"type":"movie","tmdb_id":"504172","title":"A Film"}}, + "cut":{{"runtime_sec":6420.5,"audio_signature":"{sig}"}}, + "extraction":{{"sample_fps":5,"pipeline_version":"test 0.1"}}, + "actors":[{{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[10.0,20.0]]}}, + {{"name":"Michael Palin","tmdb_id":"11007","scenes":[[30.0,40.0]]}}]}}"# + ); + let b = valid_from(&with_sig); + assert_eq!(compute_content_id(&a), compute_content_id(&b)); + } + + #[test] + fn content_id_excludes_submitted_names() { + // Names are not persisted, so they must not be part of identity either — + // otherwise a renamed resubmission would evade deduplication. + let a = valid_from(&movie_json("504172", 6420.5)); + let b = valid_from( + r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172","title":"A Film"}, + "cut":{"runtime_sec":6420.5}, + "extraction":{"sample_fps":5,"pipeline_version":"test 0.1"}, + "actors":[{"name":"Someone Else","tmdb_id":"884","scenes":[[10.0,20.0]]}, + {"name":"Another Person","tmdb_id":"11007","scenes":[[30.0,40.0]]}]}"#, + ); + assert_eq!(compute_content_id(&a), compute_content_id(&b)); + } +} diff --git a/src/lib.rs b/src/lib.rs new file mode 100644 index 0000000..28ed871 --- /dev/null +++ b/src/lib.rs @@ -0,0 +1,38 @@ +//! JRay public server — a community manifest exchange (see `SPEC.md`). +//! +//! Jellyfin servers running the JRay plugin pull actor-timeline manifests +//! ("Jmanifests") for titles they own instead of running the CV pipeline +//! locally, and optionally contribute the manifests they generate back. +//! +//! Module map against the spec: +//! +//! | Module | Spec section | +//! |---|---| +//! | [`model`] | §2 Jmanifest format, and §6 stage 2's `deny_unknown_fields` | +//! | [`validate`] | §6 stage 2 semantics, §5a character class | +//! | [`matching`] | §3 cut matching tiers | +//! | [`content_id`] | §9a canonical form and content addressing | +//! | [`ingest`] | the shared upload path behind §4's `POST` endpoints | +//! | [`castcheck`] | §6 stage 3 scoring, §5a Threat 2 guards | +//! | [`worker`] | §6 stage 3 execution, §8 in-process background work | +//! | [`ratelimit`] | §5 | +//! | [`auth`] | §5a tokens, §8 trusted-proxy handling | +//! | [`db`] | §7 storage, §8 single-writer serialization | +//! | [`api`] | §4 | + +pub mod api; +pub mod app; +pub mod auth; +pub mod castcheck; +pub mod config; +pub mod content_id; +pub mod db; +pub mod error; +pub mod ingest; +pub mod matching; +pub mod model; +pub mod ratelimit; +pub mod state; +pub mod tmdb; +pub mod validate; +pub mod worker; diff --git a/src/main.rs b/src/main.rs new file mode 100644 index 0000000..345724c --- /dev/null +++ b/src/main.rs @@ -0,0 +1,96 @@ +//! Entry point. +//! +//! §8: one binary, one database file, one reverse proxy. The rate-limit counters +//! and the background cast-check worker both live in this process — no Redis, no +//! broker, no separate worker process. + +use std::net::SocketAddr; +use std::sync::Arc; + +use anyhow::Context; +use jray_server::app; +use jray_server::config::Config; +use jray_server::db::Db; +use jray_server::ratelimit::RateLimiter; +use jray_server::state::AppState; +use jray_server::tmdb::TmdbClient; +use jray_server::worker::Worker; +use tracing_subscriber::EnvFilter; + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + tracing_subscriber::fmt() + .with_env_filter( + EnvFilter::try_from_env("JRAY_LOG").unwrap_or_else(|_| EnvFilter::new("info")), + ) + .init(); + + let config = Arc::new(Config::from_env()?); + let db = Db::open(&config.db_path).context("opening database")?; + let tmdb = Arc::new(TmdbClient::new(config.tmdb_base_url.clone(), config.tmdb_api_key.clone())); + + if !tmdb.is_configured() { + // §8: TMDB is a hard dependency for UR-3. Uploads will accumulate in + // `pending` rather than being listed unverified — which is the correct + // failure mode, but the operator should know. + tracing::warn!("no JRAY_TMDB_API_KEY configured: uploads will stay pending, never listed"); + } + if config.trusted_proxies.is_empty() { + tracing::info!("no JRAY_TRUSTED_PROXIES set: X-Forwarded-For will be ignored"); + } + + let state = AppState { + db: db.clone(), + config: config.clone(), + limiter: Arc::new(RateLimiter::new()), + tmdb: tmdb.clone(), + }; + + let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); + + let worker = Worker { + db: db.clone(), + tmdb, + batch: config.job_batch, + poll_interval: config.job_poll_interval, + }; + let worker_handle = tokio::spawn(worker.run(shutdown_rx)); + + let listener = tokio::net::TcpListener::bind(&config.bind) + .await + .with_context(|| format!("binding {}", config.bind))?; + tracing::info!(bind = %config.bind, server_id = %config.server_id, "jray-server listening"); + + let router = app::router(state); + axum::serve(listener, router.into_make_service_with_connect_info::()) + .with_graceful_shutdown(async move { + shutdown_signal().await; + let _ = shutdown_tx.send(true); + }) + .await + .context("server error")?; + + let _ = worker_handle.await; + Ok(()) +} + +async fn shutdown_signal() { + let ctrl_c = async { + tokio::signal::ctrl_c().await.expect("installing ctrl-c handler"); + }; + + #[cfg(unix)] + let terminate = async { + tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate()) + .expect("installing SIGTERM handler") + .recv() + .await; + }; + #[cfg(not(unix))] + let terminate = std::future::pending::<()>(); + + tokio::select! { + _ = ctrl_c => tracing::info!("received ctrl-c, shutting down"), + _ = terminate => tracing::info!("received SIGTERM, shutting down"), + } +} diff --git a/src/matching.rs b/src/matching.rs new file mode 100644 index 0000000..b647d5d --- /dev/null +++ b/src/matching.rs @@ -0,0 +1,217 @@ +//! §3 cut matching. +//! +//! Timings only transfer between identical cuts, so matching is tiered and the +//! server reports *which* tier matched — a `loose` match is meant to surface as +//! a caveat in the JRay UI rather than being applied silently. +//! +//! `video_hash` identifies a *file*, so it only ever matches an identical +//! release and can never produce a false positive; that is why it is tier one. + +use crate::model::MatchTier; + +/// §3: runtimes within ±2s. +pub const RUNTIME_TOLERANCE_SEC: f64 = 2.0; +/// §3: runtimes within ±30s. +pub const LOOSE_TOLERANCE_SEC: f64 = 30.0; + +/// What the client tells us about its own copy. +#[derive(Debug, Clone, Default)] +pub struct ClientCut { + pub runtime_sec: Option, + pub video_hash: Option, +} + +impl ClientCut { + /// True when the client supplied nothing to match on, in which case §4 + /// specifies a `"match": "unknown"` answer rather than a guess. + pub fn is_empty(&self) -> bool { + self.runtime_sec.is_none() && self.video_hash.is_none() + } +} + +/// What the server holds. +#[derive(Debug, Clone)] +pub struct StoredCut { + pub runtime_sec: f64, + pub video_hash: Option, +} + +/// The outcome of comparing a client's cut against a stored one. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct CutMatch { + pub tier: MatchTier, + /// Scene offset in seconds the client must add (§3 `audio` tier). Always + /// zero for the tiers implemented here; the field exists because the plugin + /// contract is "the server returns the offset, the client applies it", and + /// enabling `audio` must not change the response shape. + pub offset_sec: f64, +} + +/// Compares a client's cut against a stored one, returning the best tier that +/// fires, or `None` for "beyond that: no match; do not serve" (§3). +pub fn match_cut(client: &ClientCut, stored: &StoredCut) -> Option { + // Tier 1 — same file. Checked first and unconditionally: an equal hash is + // decisive regardless of what the runtimes say. + if let (Some(c), Some(s)) = (&client.video_hash, &stored.video_hash) { + if c.eq_ignore_ascii_case(s) { + return Some(CutMatch { tier: MatchTier::Exact, offset_sec: 0.0 }); + } + } + + // `audio` tier would slot in here, above `runtime`, once signature coverage + // is useful (§3 recommended sequencing). + + if let Some(c_rt) = client.runtime_sec { + let delta = (c_rt - stored.runtime_sec).abs(); + if delta <= RUNTIME_TOLERANCE_SEC { + return Some(CutMatch { tier: MatchTier::Runtime, offset_sec: 0.0 }); + } + if delta <= LOOSE_TOLERANCE_SEC { + return Some(CutMatch { tier: MatchTier::Loose, offset_sec: 0.0 }); + } + // A runtime was supplied and cleared nothing — that is a definite + // no-match, not an unknown. + return None; + } + + // A hash that did not match, with no runtime to fall back on, tells us + // nothing about alignment either way. + if client.video_hash.is_some() { + return None; + } + + Some(CutMatch { tier: MatchTier::Unknown, offset_sec: 0.0 }) +} + +/// Picks the best-matching stored cut, if any clears `loose` (§4). +/// +/// `candidates` is `(key, cut)`; the key is returned so the caller can identify +/// which manifest won without re-scanning. +pub fn best_match( + client: &ClientCut, + candidates: &[(K, StoredCut)], +) -> Option<(K, CutMatch)> { + candidates + .iter() + .filter_map(|(k, cut)| match_cut(client, cut).map(|m| (k.clone(), m))) + .max_by(|a, b| a.1.tier.cmp(&b.1.tier)) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn stored(runtime: f64, hash: Option<&str>) -> StoredCut { + StoredCut { runtime_sec: runtime, video_hash: hash.map(str::to_string) } + } + + #[test] + fn equal_video_hash_is_exact() { + let c = ClientCut { + runtime_sec: Some(6420.5), + video_hash: Some("opensubtitles:8e245d9679d31e12".into()), + }; + let m = match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).unwrap(); + assert_eq!(m.tier, MatchTier::Exact); + } + + #[test] + fn equal_hash_wins_even_when_runtimes_disagree() { + // The hash identifies the file; a differing stored runtime means our + // own metadata is off, not that the file is different. + let c = ClientCut { + runtime_sec: Some(6000.0), + video_hash: Some("opensubtitles:8e245d9679d31e12".into()), + }; + let m = match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).unwrap(); + assert_eq!(m.tier, MatchTier::Exact); + } + + #[test] + fn hash_comparison_is_case_insensitive() { + let c = ClientCut { + runtime_sec: None, + video_hash: Some("opensubtitles:8E245D9679D31E12".into()), + }; + let m = match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).unwrap(); + assert_eq!(m.tier, MatchTier::Exact); + } + + #[test] + fn runtime_within_two_seconds_is_runtime_tier() { + let c = ClientCut { runtime_sec: Some(6422.0), video_hash: None }; + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Runtime); + } + + #[test] + fn runtime_within_thirty_seconds_is_loose() { + let c = ClientCut { runtime_sec: Some(6450.0), video_hash: None }; + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Loose); + } + + #[test] + fn beyond_thirty_seconds_does_not_match() { + // §3: "beyond that — no match; do not serve". + let c = ClientCut { runtime_sec: Some(6500.0), video_hash: None }; + assert!(match_cut(&c, &stored(6420.5, None)).is_none()); + } + + #[test] + fn tier_boundaries_are_inclusive() { + let c = ClientCut { runtime_sec: Some(6422.5), video_hash: None }; + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Runtime); + let c = ClientCut { runtime_sec: Some(6450.5), video_hash: None }; + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Loose); + } + + #[test] + fn no_cut_information_yields_unknown() { + // §4: the mode a library-wide sweep uses — "does the community have + // this title at all", with alignment still to be determined. + let c = ClientCut::default(); + assert!(c.is_empty()); + assert_eq!(match_cut(&c, &stored(6420.5, None)).unwrap().tier, MatchTier::Unknown); + } + + #[test] + fn non_matching_hash_alone_is_not_a_match() { + let c = ClientCut { + runtime_sec: None, + video_hash: Some("opensubtitles:ffffffffffffffff".into()), + }; + assert!(match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).is_none()); + } + + #[test] + fn non_matching_hash_falls_back_to_runtime() { + let c = ClientCut { + runtime_sec: Some(6421.0), + video_hash: Some("opensubtitles:ffffffffffffffff".into()), + }; + let m = match_cut(&c, &stored(6420.5, Some("opensubtitles:8e245d9679d31e12"))).unwrap(); + assert_eq!(m.tier, MatchTier::Runtime); + } + + #[test] + fn best_match_prefers_the_highest_tier() { + let c = ClientCut { + runtime_sec: Some(6420.5), + video_hash: Some("opensubtitles:8e245d9679d31e12".into()), + }; + let candidates = vec![ + ("loose", stored(6445.0, None)), + ("exact", stored(9999.0, Some("opensubtitles:8e245d9679d31e12"))), + ("runtime", stored(6420.0, None)), + ]; + let (winner, m) = best_match(&c, &candidates).unwrap(); + assert_eq!(winner, "exact"); + assert_eq!(m.tier, MatchTier::Exact); + } + + #[test] + fn best_match_returns_none_when_nothing_clears_loose() { + let c = ClientCut { runtime_sec: Some(100.0), video_hash: None }; + let candidates = vec![("a", stored(6420.5, None)), ("b", stored(3000.0, None))]; + assert!(best_match(&c, &candidates).is_none()); + } +} diff --git a/src/model.rs b/src/model.rs new file mode 100644 index 0000000..5241824 --- /dev/null +++ b/src/model.rs @@ -0,0 +1,307 @@ +//! Jmanifest wire types (§2). +//! +//! **`#[serde(deny_unknown_fields)]` on every struct is the §6 stage 2 +//! enforcement mechanism.** "No additional fields anywhere" is a property of +//! these type definitions rather than of validator code that could omit a +//! field, so an unrecognised key at any nesting level fails to parse. That is +//! also what makes the §9 `movie`/`jellyfin_id` strip verifiable: a client that +//! forgets gets a hard `400` naming the field, rather than quietly publishing a +//! contributor's directory layout. +//! +//! Deeper semantic checks — bounds, character classes, path-shaped strings — +//! live in [`crate::validate`]. Parsing rejects *shape*; validation rejects +//! *content*. + +use serde::{Deserialize, Serialize}; + +/// Cut-match tier (§3). Ordered worst-to-best so derived `Ord` ranks them. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum MatchTier { + /// No cut information was supplied, so alignment is unknown (§4). + Unknown, + /// Audio 0.60–0.85, or runtimes within ±30s. Caveat in UI. + Loose, + /// Runtimes within ±2s. + Runtime, + /// Audio score ≥ 0.85; ranks above `runtime` because it is content-derived. + Audio, + /// `video_hash` equal — same file. + Exact, +} + +impl MatchTier { + pub fn as_str(self) -> &'static str { + match self { + MatchTier::Unknown => "unknown", + MatchTier::Loose => "loose", + MatchTier::Runtime => "runtime", + MatchTier::Audio => "audio", + MatchTier::Exact => "exact", + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum IdentityType { + Movie, + Episode, +} + +/// What the work is (§2 terminology: *title identity*). +/// +/// Movie and episode coordinates share one struct because `deny_unknown_fields` +/// with `#[serde(untagged)]` alternatives produces unhelpful error messages; +/// the discriminant is checked in [`crate::validate`], which can name the +/// offending field. +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Identity { + #[serde(rename = "type")] + pub kind: IdentityType, + + // Movie coordinates. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tmdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub imdb_id: Option, + + // Episode coordinates. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub series_tmdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub series_imdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub season: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub episode: Option, + + #[serde(default, skip_serializing_if = "Option::is_none")] + pub title: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub year: Option, +} + +impl Identity { + /// The TMDB id used as the lookup key, whichever coordinate carries it. + pub fn effective_tmdb_id(&self) -> Option<&str> { + match self.kind { + IdentityType::Movie => self.tmdb_id.as_deref(), + IdentityType::Episode => self.series_tmdb_id.as_deref(), + } + } + + pub fn effective_imdb_id(&self) -> Option<&str> { + match self.kind { + IdentityType::Movie => self.imdb_id.as_deref(), + IdentityType::Episode => self.series_imdb_id.as_deref(), + } + } +} + +/// Which encode/edit the timings apply to (§2 terminology: *cut fingerprint*). +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Cut { + /// **Required** — the decoded duration of the media the timings came from. + /// The primary alignment guard (§2). + pub runtime_sec: f64, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub container_duration_sec: Option, + /// Optional but strongly preferred. OpenSubtitles hash (§3). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub video_hash: Option, + /// Optional; version-prefixed spectral-peak signature (§3, UR-9). + /// + /// Accepted and stored by this build; `audio`-tier matching is enabled once + /// coverage is useful, per §3's recommended sequencing. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub audio_signature: Option, +} + +/// How well the contributor's gallery could discriminate (§2, §7). +/// +/// The strongest available quality signal between two otherwise comparable +/// manifests: a `Global` gallery had to distinguish its actors from every other +/// actor in the contributor's library, whereas a `Limited` one only had to +/// distinguish them from this title's own cast. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Deserialize, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum GalleryScope { + /// Built from this title's cast alone. + Limited, + /// Built from the whole library. The default upstream. + Global, +} + +impl GalleryScope { + pub fn as_str(self) -> &'static str { + match self { + GalleryScope::Limited => "limited", + GalleryScope::Global => "global", + } + } +} + +/// Extraction parameters, carried for provenance and ranking. +/// +/// Note there is no `anneal_sec`: it was **withdrawn** in the SR-003 schema +/// bump, because presence now follows track extent — a track survives its own +/// gaps, so there is nothing to anneal (`scene-actor-extraction` AR-012/AR-013). +/// `deny_unknown_fields` therefore makes its presence a hard parse error rather +/// than something silently ignored, which is deliberate: a manifest still +/// carrying it was produced by a pipeline whose window semantics differ from +/// what this server now assumes. +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Extraction { + #[serde(default, skip_serializing_if = "Option::is_none")] + pub sample_fps: Option, + /// The re-acquisition timeout that shapes window extent. Successor to the + /// withdrawn `anneal_sec`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub extinction_sec: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub pipeline_version: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub gallery_size: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub gallery_scope: Option, +} + +/// One actor's timeline. +/// +/// Note there is no `jellyfin_id` field: `deny_unknown_fields` means its +/// presence is a parse error, which is exactly the §2/§6 requirement that it be +/// *rejected on upload* rather than merely ignored on download. +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Actor { + /// Sent on upload for matching, but **not persisted** — the server resolves + /// each actor to a TMDB person id and serves names from its own TMDB-derived + /// table (§2, §5a). On download this is server-authoritative. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub name: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub imdb_id: Option, + /// The **primary** actor join key (§2, §6 stage 3). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tmdb_id: Option, + /// `[start_sec, end_sec]` inclusive, sorted. + pub scenes: Vec<[f64; 2]>, +} + +/// One shareable actor timeline for one cut of one title (§2). +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Jmanifest { + pub jmanifest_version: u32, + pub identity: Identity, + pub cut: Cut, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub extraction: Option, + pub actors: Vec, +} + +/// Series-level coordinates for a bundle envelope (§2). +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct SeriesRef { + #[serde(default, skip_serializing_if = "Option::is_none")] + pub series_tmdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub series_imdb_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub title: Option, +} + +/// A thin wrapper, not a new format (§2). Bundles are a transfer convenience, +/// never a storage unit — each episode is stored and moderated individually. +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct SeriesBundle { + pub jmanifest_version: u32, + pub series: SeriesRef, + pub episodes: Vec, + /// Present on responses only; ignored on upload. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub coverage: Option, +} + +#[derive(Debug, Clone, Deserialize, Serialize)] +#[serde(deny_unknown_fields)] +pub struct Coverage { + pub episodes_available: usize, + pub seasons: Vec, +} + +/// The current `jmanifest_version` this server speaks (§2). +pub const JMANIFEST_VERSION: u32 = 1; + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn unknown_field_at_top_level_is_rejected() { + let json = r#"{"jmanifest_version":1,"identity":{"type":"movie","tmdb_id":"1"}, + "cut":{"runtime_sec":100.0},"actors":[],"surprise":"x"}"#; + let err = serde_json::from_str::(json).unwrap_err().to_string(); + assert!(err.contains("surprise"), "error should name the field: {err}"); + } + + #[test] + fn jellyfin_id_on_an_actor_is_a_parse_error() { + // §2: `actors[].jellyfin_id` must not appear. `deny_unknown_fields` + // makes this structural rather than a validator's responsibility. + let json = r#"{"jmanifest_version":1,"identity":{"type":"movie","tmdb_id":"1"}, + "cut":{"runtime_sec":100.0}, + "actors":[{"name":"A","tmdb_id":"2","jellyfin_id":"guid","scenes":[]}]}"#; + let err = serde_json::from_str::(json).unwrap_err().to_string(); + assert!(err.contains("jellyfin_id"), "error should name the field: {err}"); + } + + #[test] + fn movie_path_field_is_a_parse_error() { + let json = r#"{"jmanifest_version":1,"movie":"/data/movies/x.mkv", + "identity":{"type":"movie","tmdb_id":"1"}, + "cut":{"runtime_sec":100.0},"actors":[]}"#; + let err = serde_json::from_str::(json).unwrap_err().to_string(); + assert!(err.contains("movie"), "error should name the field: {err}"); + } + + #[test] + fn unknown_field_nested_in_cut_is_rejected() { + let json = r#"{"jmanifest_version":1,"identity":{"type":"movie","tmdb_id":"1"}, + "cut":{"runtime_sec":100.0,"payload":"x"},"actors":[]}"#; + assert!(serde_json::from_str::(json).is_err()); + } + + #[test] + fn spec_example_manifest_parses() { + let json = r#"{ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172", "imdb_id": "tt4686844", + "title": "The Death of Stalin", "year": 2017 }, + "cut": { "runtime_sec": 6420.5, "container_duration_sec": 6420.5, + "video_hash": "opensubtitles:8e245d9679d31e12" }, + "extraction": { "sample_fps": 5, "extinction_sec": 12, + "pipeline_version": "scene-actor-extraction 0.4.1", + "gallery_size": 1820, "gallery_scope": "global" }, + "actors": [ { "name": "Steve Buscemi", "imdb_id": "nm0000114", "tmdb_id": "884", + "scenes": [[191.6, 209.2], [438.2, 465.6]] } ] + }"#; + let m: Jmanifest = serde_json::from_str(json).unwrap(); + assert_eq!(m.actors.len(), 1); + assert_eq!(m.identity.effective_tmdb_id(), Some("504172")); + } + + #[test] + fn tiers_order_audio_above_runtime() { + // §3: `audio` ranks above `runtime` because it is content-derived. + assert!(MatchTier::Audio > MatchTier::Runtime); + assert!(MatchTier::Exact > MatchTier::Audio); + assert!(MatchTier::Runtime > MatchTier::Loose); + } +} diff --git a/src/ratelimit.rs b/src/ratelimit.rs new file mode 100644 index 0000000..66a4eba --- /dev/null +++ b/src/ratelimit.rs @@ -0,0 +1,217 @@ +//! §5 rate limiting. +//! +//! A fixed-window counter keyed on `(token_or_ip, surface)`, held in process +//! memory — no external counter store. §5 is explicit that a sliding window is +//! not worth the complexity at this volume, and that counters resetting on +//! restart is acceptable for abuse throttling. +//! +//! Read limits are applied *behind* the CDN cache, so a cache hit costs a client +//! nothing against its budget — that is a deployment property (§8), not +//! something this module can enforce. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +/// The rate-limited surfaces of §5. Distinct from routes: the batch and single +/// forms of `exists` are separate surfaces with separate budgets. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Surface { + ExistsSingle, + ExistsBatch, + ManifestFetch, + SeriesFetch, + ManifestUpload, + BundleUpload, + Report, + Search, +} + +impl Surface { + /// Requests per hour, per §5's table. + pub fn limit(self) -> u32 { + match self { + Surface::ExistsSingle => 600, + Surface::ExistsBatch => 60, + Surface::ManifestFetch => 300, + Surface::SeriesFetch => 120, + Surface::ManifestUpload => 100, + Surface::BundleUpload => 20, + Surface::Report => 20, + Surface::Search => 60, + } + } + + pub fn as_str(self) -> &'static str { + match self { + Surface::ExistsSingle => "exists", + Surface::ExistsBatch => "exists_batch", + Surface::ManifestFetch => "manifest_fetch", + Surface::SeriesFetch => "series_fetch", + Surface::ManifestUpload => "manifest_upload", + Surface::BundleUpload => "bundle_upload", + Surface::Report => "report", + Surface::Search => "search", + } + } +} + +const WINDOW: Duration = Duration::from_secs(3600); + +/// Headers §5 requires on every rate-limited response. +#[derive(Debug, Clone, Copy)] +pub struct Quota { + pub limit: u32, + pub remaining: u32, + /// Seconds until the window resets. + pub reset: u64, +} + +#[derive(Debug, Clone, Copy)] +struct Window { + started: Instant, + count: u32, +} + +pub struct RateLimiter { + windows: Mutex>, +} + +impl Default for RateLimiter { + fn default() -> Self { + Self::new() + } +} + +impl RateLimiter { + pub fn new() -> Self { + Self { windows: Mutex::new(HashMap::new()) } + } + + /// Records one request against `(key, surface)`. + /// + /// `Ok(quota)` when within budget, `Err(quota)` when the limit is exceeded — + /// in which case the caller returns `429` with `Retry-After` set from + /// `quota.reset`. A rejected request does **not** increment the counter, so a + /// client hammering a closed window cannot extend its own lockout. + pub fn check(&self, key: &str, surface: Surface) -> Result { + self.check_at(key, surface, Instant::now()) + } + + fn check_at(&self, key: &str, surface: Surface, now: Instant) -> Result { + let limit = surface.limit(); + let mut windows = self.windows.lock().expect("rate limiter poisoned"); + + // Opportunistic eviction of stale windows, so an IP-keyed map cannot + // grow without bound behind CGNAT. + if windows.len() > 10_000 { + windows.retain(|_, w| now.duration_since(w.started) < WINDOW); + } + + let entry = + windows.entry((key.to_string(), surface)).or_insert(Window { started: now, count: 0 }); + + let elapsed = now.duration_since(entry.started); + if elapsed >= WINDOW { + *entry = Window { started: now, count: 0 }; + } + + let reset = WINDOW.saturating_sub(now.duration_since(entry.started)).as_secs(); + + if entry.count >= limit { + return Err(Quota { limit, remaining: 0, reset }); + } + entry.count += 1; + Ok(Quota { limit, remaining: limit - entry.count, reset }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn allows_up_to_the_limit_then_rejects() { + let rl = RateLimiter::new(); + let limit = Surface::Report.limit(); + for i in 0..limit { + let q = rl.check("ip", Surface::Report).expect("within budget"); + assert_eq!(q.remaining, limit - i - 1); + } + let q = rl.check("ip", Surface::Report).expect_err("over budget"); + assert_eq!(q.remaining, 0); + } + + #[test] + fn surfaces_have_independent_budgets() { + let rl = RateLimiter::new(); + for _ in 0..Surface::BundleUpload.limit() { + rl.check("t", Surface::BundleUpload).unwrap(); + } + assert!(rl.check("t", Surface::BundleUpload).is_err()); + // §5: a bundle counts as a single write against its own limit, and must + // not consume the single-manifest budget. + assert!(rl.check("t", Surface::ManifestUpload).is_ok()); + } + + #[test] + fn keys_are_independent() { + let rl = RateLimiter::new(); + for _ in 0..Surface::Report.limit() { + rl.check("a", Surface::Report).unwrap(); + } + assert!(rl.check("a", Surface::Report).is_err()); + assert!(rl.check("b", Surface::Report).is_ok()); + } + + #[test] + fn window_resets_after_an_hour() { + let rl = RateLimiter::new(); + let t0 = Instant::now(); + for _ in 0..Surface::Report.limit() { + rl.check_at("ip", Surface::Report, t0).unwrap(); + } + assert!(rl.check_at("ip", Surface::Report, t0).is_err()); + // Still closed just inside the window. + assert!(rl.check_at("ip", Surface::Report, t0 + Duration::from_secs(3599)).is_err()); + // Open again once it rolls over. + assert!(rl.check_at("ip", Surface::Report, t0 + Duration::from_secs(3600)).is_ok()); + } + + #[test] + fn rejected_requests_do_not_extend_the_lockout() { + let rl = RateLimiter::new(); + let t0 = Instant::now(); + for _ in 0..Surface::Report.limit() { + rl.check_at("ip", Surface::Report, t0).unwrap(); + } + // Hammer the closed window; the counter must not keep climbing, so the + // window still expires on schedule. + for _ in 0..50 { + assert!(rl.check_at("ip", Surface::Report, t0 + Duration::from_secs(10)).is_err()); + } + assert!(rl.check_at("ip", Surface::Report, t0 + Duration::from_secs(3600)).is_ok()); + } + + #[test] + fn reset_counts_down_within_the_window() { + let rl = RateLimiter::new(); + let t0 = Instant::now(); + let q = rl.check_at("ip", Surface::ExistsSingle, t0).unwrap(); + assert_eq!(q.reset, 3600); + let q = rl.check_at("ip", Surface::ExistsSingle, t0 + Duration::from_secs(600)).unwrap(); + assert_eq!(q.reset, 3000); + } + + #[test] + fn limits_match_the_spec_table() { + assert_eq!(Surface::ExistsSingle.limit(), 600); + assert_eq!(Surface::ExistsBatch.limit(), 60); + assert_eq!(Surface::ManifestFetch.limit(), 300); + assert_eq!(Surface::SeriesFetch.limit(), 120); + assert_eq!(Surface::ManifestUpload.limit(), 100); + assert_eq!(Surface::BundleUpload.limit(), 20); + assert_eq!(Surface::Report.limit(), 20); + assert_eq!(Surface::Search.limit(), 60); + } +} diff --git a/src/state.rs b/src/state.rs new file mode 100644 index 0000000..809a78a --- /dev/null +++ b/src/state.rs @@ -0,0 +1,102 @@ +//! Shared application state, and the cross-cutting request concerns (§5 rate +//! limiting, §5a token resolution) that every handler needs. + +use std::net::SocketAddr; +use std::sync::Arc; + +use axum::extract::ConnectInfo; +use axum::http::{HeaderMap, HeaderValue}; +use axum::response::Response; + +use crate::auth; +use crate::config::Config; +use crate::db::{repo, Db}; +use crate::error::{ApiError, ApiResult}; +use crate::ratelimit::{Quota, RateLimiter, Surface}; +use crate::tmdb::TmdbClient; + +/// The connection's peer address, when the server was started with connect-info. +/// +/// A dedicated extractor rather than `ConnectInfo` directly, because +/// this must not be a *hard* requirement: a router used without +/// `into_make_service_with_connect_info` — as in tests — has no peer address, and +/// a handler that fails to extract would be a routing error rather than degrading +/// to header-only attribution. +pub struct PeerIp(pub Option); + +impl axum::extract::FromRequestParts for PeerIp +where + S: Send + Sync, +{ + type Rejection = std::convert::Infallible; + + async fn from_request_parts( + parts: &mut axum::http::request::Parts, + _state: &S, + ) -> Result { + Ok(PeerIp( + parts.extensions.get::>().map(|ConnectInfo(addr)| addr.ip()), + )) + } +} + +#[derive(Clone)] +pub struct AppState { + pub db: Db, + pub config: Arc, + pub limiter: Arc, + pub tmdb: Arc, +} + +impl AppState { + /// Resolves the client IP for rate-limiting and attribution, honouring + /// `X-Forwarded-For` only from a configured proxy (§8). + pub fn client_ip(&self, headers: &HeaderMap, peer: Option) -> String { + auth::client_ip(headers, peer, &self.config.trusted_proxies) + } + + /// §5: limits are per token where one is present, otherwise per source IP. + pub fn check_limit(&self, key: &str, surface: Surface) -> ApiResult { + self.limiter.check(key, surface).map_err(|q| { + tracing::debug!(surface = surface.as_str(), "rate limited"); + ApiError::RateLimited { retry_after: q.reset.max(1) } + }) + } + + /// Resolves a bearer token to a contributor (§5a). + /// + /// A token is an anonymous bearer capability, not an account: the only state + /// behind it is the per-token counters used for rate-limiting attribution and + /// automatic revocation. + pub async fn require_contributor(&self, headers: &HeaderMap) -> ApiResult { + let token = auth::bearer_token(headers).ok_or(ApiError::Unauthorized)?; + let hash = auth::hash_token(&token); + let found = self + .db + .read(move |conn| repo::contributor_by_token_hash(conn, &hash)) + .await + .map_err(ApiError::Internal)?; + + match found { + Some(c) if !c.revoked => Ok(c), + // A revoked token is indistinguishable from an unknown one to the + // caller; there is nothing useful to disclose. + _ => Err(ApiError::Unauthorized), + } + } +} + +/// Attaches the §5 rate-limit headers to a response. +pub fn with_quota_headers(mut resp: Response, quota: Quota) -> Response { + let h = resp.headers_mut(); + insert_num(h, "x-ratelimit-limit", quota.limit as u64); + insert_num(h, "x-ratelimit-remaining", quota.remaining as u64); + insert_num(h, "x-ratelimit-reset", quota.reset); + resp +} + +fn insert_num(headers: &mut HeaderMap, name: &'static str, value: u64) { + if let Ok(v) = HeaderValue::from_str(&value.to_string()) { + headers.insert(name, v); + } +} diff --git a/src/tmdb.rs b/src/tmdb.rs new file mode 100644 index 0000000..a279629 --- /dev/null +++ b/src/tmdb.rs @@ -0,0 +1,226 @@ +//! TMDB client for the §6 stage 3 cast cross-check. +//! +//! §5a's Threat 2 defence rests entirely on the attacker not controlling TMDB: +//! to make a prank manifest pass, they would need those performers to be +//! credited cast on that title in TMDB, which means vandalising a separate, +//! moderated system. +//! +//! Responses are cached for 24h (§6) so a burst of episode uploads for one +//! series costs a single upstream call, and so the server stays within TMDB's +//! own rate limits. + +use std::time::Duration; + +use serde::Deserialize; + +/// A credited cast member, reduced to what the check needs. +#[derive(Debug, Clone, Deserialize)] +pub struct CastMember { + pub id: u64, + #[serde(default)] + pub name: String, + #[serde(default)] + pub adult: bool, +} + +#[derive(Debug, Clone, Default, Deserialize)] +pub struct Credits { + #[serde(default)] + pub cast: Vec, + /// Present on episode credits. + #[serde(default)] + pub guest_stars: Vec, +} + +impl Credits { + /// Cast plus guest stars — the union §6 specifies for episodes. + pub fn all(&self) -> impl Iterator { + self.cast.iter().chain(self.guest_stars.iter()) + } +} + +#[derive(Debug, Clone, Default, Deserialize)] +pub struct TitleDetails { + #[serde(default)] + pub adult: bool, + #[serde(default)] + pub title: Option, + #[serde(default)] + pub name: Option, +} + +/// A failure that should be retried rather than treated as a verdict. +/// +/// §6: "TMDB unreachable / rate-limited → retry with backoff; stays unlisted, +/// not rejected." Distinguishing this from "TMDB has no credits" is essential — +/// conflating them would reject honest manifests during an outage. +#[derive(Debug, thiserror::Error)] +pub enum TmdbError { + #[error("tmdb transport error: {0}")] + Transport(String), + #[error("tmdb rate limited")] + RateLimited, + #[error("tmdb server error: {0}")] + ServerError(u16), + /// The id genuinely does not exist upstream. + #[error("tmdb resource not found")] + NotFound, + #[error("tmdb response was not understood: {0}")] + Malformed(String), + #[error("no tmdb api key configured")] + NotConfigured, +} + +impl TmdbError { + /// True when the job should be rescheduled rather than resolved. + pub fn is_retryable(&self) -> bool { + matches!( + self, + TmdbError::Transport(_) + | TmdbError::RateLimited + | TmdbError::ServerError(_) + | TmdbError::NotConfigured + ) + } +} + +#[derive(Clone)] +pub struct TmdbClient { + http: reqwest::Client, + base_url: String, + api_key: Option, +} + +impl TmdbClient { + pub fn new(base_url: String, api_key: Option) -> Self { + let http = reqwest::Client::builder() + .timeout(Duration::from_secs(15)) + .user_agent(concat!("jray-server/", env!("CARGO_PKG_VERSION"))) + .build() + .expect("building reqwest client"); + Self { http, base_url, api_key } + } + + pub fn is_configured(&self) -> bool { + self.api_key.is_some() + } + + async fn get(&self, path: &str) -> Result { + let key = self.api_key.as_deref().ok_or(TmdbError::NotConfigured)?; + let url = + format!("{}/{}", self.base_url.trim_end_matches('/'), path.trim_start_matches('/')); + + let resp = self + .http + .get(&url) + .query(&[("api_key", key)]) + .send() + .await + .map_err(|e| TmdbError::Transport(e.to_string()))?; + + let status = resp.status(); + if status == reqwest::StatusCode::NOT_FOUND { + return Err(TmdbError::NotFound); + } + if status == reqwest::StatusCode::TOO_MANY_REQUESTS { + return Err(TmdbError::RateLimited); + } + if status.is_server_error() { + return Err(TmdbError::ServerError(status.as_u16())); + } + if !status.is_success() { + return Err(TmdbError::Malformed(format!("unexpected status {status}"))); + } + + let body = resp.text().await.map_err(|e| TmdbError::Transport(e.to_string()))?; + serde_json::from_str(&body).map_err(|e| TmdbError::Malformed(e.to_string())) + } + + pub async fn movie_credits(&self, tmdb_id: &str) -> Result { + self.get(&format!("movie/{tmdb_id}/credits")).await + } + + pub async fn movie_details(&self, tmdb_id: &str) -> Result { + self.get(&format!("movie/{tmdb_id}")).await + } + + pub async fn series_credits(&self, series_tmdb_id: &str) -> Result { + // Aggregate credits carry recurring cast TMDB lists only at series level. + self.get(&format!("tv/{series_tmdb_id}/aggregate_credits")).await + } + + pub async fn episode_credits( + &self, + series_tmdb_id: &str, + season: i64, + episode: i64, + ) -> Result { + self.get(&format!("tv/{series_tmdb_id}/season/{season}/episode/{episode}/credits")).await + } + + pub async fn series_details(&self, series_tmdb_id: &str) -> Result { + self.get(&format!("tv/{series_tmdb_id}")).await + } +} + +/// Fetches a person's details, used by the §5a category guard. +#[derive(Debug, Clone, Default, Deserialize)] +pub struct PersonDetails { + #[serde(default)] + pub adult: bool, + #[serde(default)] + pub name: String, +} + +impl TmdbClient { + pub async fn person(&self, tmdb_person_id: u64) -> Result { + self.get(&format!("person/{tmdb_person_id}")).await + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn credits_union_covers_cast_and_guest_stars() { + // §6: for episodes the check runs against the union of per-episode + // credits (cast + guest stars) and series aggregate credits. + let c: Credits = serde_json::from_str( + r#"{"cast":[{"id":1,"name":"A"}],"guest_stars":[{"id":2,"name":"B"}]}"#, + ) + .unwrap(); + let ids: Vec = c.all().map(|m| m.id).collect(); + assert_eq!(ids, vec![1, 2]); + } + + #[test] + fn credits_tolerate_missing_and_extra_fields() { + // TMDB adds fields freely; our own strictness applies to *uploads*, not + // to a trusted upstream we merely read. + let c: Credits = + serde_json::from_str(r#"{"cast":[{"id":1,"unexpected":true}],"id":99}"#).unwrap(); + assert_eq!(c.cast.len(), 1); + assert_eq!(c.cast[0].name, ""); + assert!(c.guest_stars.is_empty()); + } + + #[test] + fn transport_and_rate_limit_are_retryable_but_not_found_is_not() { + // The distinction that keeps an outage from rejecting honest uploads. + assert!(TmdbError::Transport("x".into()).is_retryable()); + assert!(TmdbError::RateLimited.is_retryable()); + assert!(TmdbError::ServerError(503).is_retryable()); + assert!(TmdbError::NotConfigured.is_retryable()); + assert!(!TmdbError::NotFound.is_retryable()); + assert!(!TmdbError::Malformed("x".into()).is_retryable()); + } + + #[tokio::test] + async fn unconfigured_client_reports_retryable_failure() { + let c = TmdbClient::new("http://127.0.0.1:1".into(), None); + assert!(!c.is_configured()); + let err = c.movie_credits("1").await.unwrap_err(); + assert!(err.is_retryable(), "missing key must hold uploads pending, not reject them"); + } +} diff --git a/src/validate.rs b/src/validate.rs new file mode 100644 index 0000000..f1993b3 --- /dev/null +++ b/src/validate.rs @@ -0,0 +1,1055 @@ +//! §6 stage 2 semantic validation, and the §5a character-class constraint. +//! +//! Shape is already enforced by `deny_unknown_fields` at parse time +//! ([`crate::model`]); everything here is *content*. Rejections name the +//! offending field, per §6. + +use unicode_general_category::{get_general_category, GeneralCategory}; +use unicode_normalization::{is_nfc, UnicodeNormalization}; + +use crate::model::{Actor, Identity, IdentityType, Jmanifest, SeriesBundle, JMANIFEST_VERSION}; + +/// §6 stage 1 caps. Body-size limits are applied as a layer (see [`crate::app`]); +/// these are the structural counts the schema layer enforces. +pub mod limits { + pub const MAX_ACTORS: usize = 500; + pub const MAX_SCENES_PER_ACTOR: usize = 2000; + pub const MAX_TOTAL_SCENES: usize = 20_000; + pub const MAX_NAME_CHARS: usize = 200; + pub const MAX_TITLE_CHARS: usize = 300; + /// §2 bundle caps. + pub const MAX_BUNDLE_EPISODES: usize = 500; + /// Times beyond `runtime_sec` + this tolerance are rejected (§6). + pub const RUNTIME_TOLERANCE_SEC: f64 = 5.0; + /// §6 stage 1 body caps, in bytes. + pub const BODY_LIMIT_MANIFEST: usize = 2 * 1024 * 1024; + pub const BODY_LIMIT_BUNDLE: usize = 25 * 1024 * 1024; + /// §3: fixed signature length. The tolerance covers seek and encoder + /// differences at the window edges — it is not a licence to vary the length. + pub const AUDIO_SIG_FRAMES: usize = 1290; + pub const AUDIO_SIG_TOLERANCE: usize = 32; +} + +#[derive(Debug, thiserror::Error)] +#[error("{field}: {reason}")] +pub struct ValidationError { + pub field: String, + pub reason: String, +} + +fn err(field: impl Into, reason: impl Into) -> ValidationError { + ValidationError { field: field.into(), reason: reason.into() } +} + +type VResult = Result; + +/// §3 / IR-007: the analysis window is `runtime/2 ± 60 s`, so below 120 s it +/// underflows and **no signature is emitted**. The rule is identical in both +/// producers and here; a rule that differs between them yields signatures that +/// never match. +pub const AUDIO_SIG_MIN_RUNTIME_SEC: f64 = 120.0; + +/// A manifest that has passed §6 stage 2. Carries the normalised forms so +/// downstream stages do not re-derive them. +#[derive(Debug, Clone)] +pub struct ValidManifest { + pub manifest: Jmanifest, + /// Scene windows quantised to integer centiseconds (§7, §9a) — the same + /// quantisation used for `content_id`, so stored and hashed values cannot + /// diverge. + pub actor_scenes_cs: Vec, +} + +#[derive(Debug, Clone)] +pub struct ActorScenes { + /// NFC-normalised name, used only for matching in stage 3 and then dropped. + pub name: Option, + pub tmdb_id: Option, + pub imdb_id: Option, + pub scenes_cs: Vec<(i64, i64)>, +} + +/// Quantises seconds to whole centiseconds (§9a). +/// +/// Integer centiseconds remove the float-canonicalisation failure mode rather +/// than dodging it: pipeline timings are *derived* by accumulating `1/fps`, so +/// they carry accumulated error, and any value near a rounding boundary would +/// otherwise hash differently on two servers. +pub fn to_centiseconds(secs: f64) -> i64 { + (secs * 100.0).round() as i64 +} + +// --------------------------------------------------------------------------- +// Identifier formats (§6 stage 2) +// --------------------------------------------------------------------------- + +/// `^tt\d{7,8}$` +fn is_title_imdb_id(s: &str) -> bool { + let Some(digits) = s.strip_prefix("tt") else { return false }; + matches!(digits.len(), 7 | 8) && digits.bytes().all(|b| b.is_ascii_digit()) +} + +/// `^nm\d{7,8}$` +fn is_person_imdb_id(s: &str) -> bool { + let Some(digits) = s.strip_prefix("nm") else { return false }; + matches!(digits.len(), 7 | 8) && digits.bytes().all(|b| b.is_ascii_digit()) +} + +/// `^\d{1,9}$` +fn is_tmdb_id(s: &str) -> bool { + !s.is_empty() && s.len() <= 9 && s.bytes().all(|b| b.is_ascii_digit()) +} + +// --------------------------------------------------------------------------- +// §5a free-text constraints +// --------------------------------------------------------------------------- + +/// §5a character class: Unicode letters, marks, spaces, and `. ' - ,` only. +/// +/// No digits and no `/ + =`, which is what **defeats base64/hex smuggling**. No +/// control characters, and no zero-width or bidi-control codepoints. +fn is_allowed_text_char(c: char) -> bool { + if matches!(c, '.' | '\'' | '-' | ',' | ' ') { + return true; + } + // Explicitly excluded regardless of category: zero-width and bidi controls. + if matches!(c, '\u{200B}'..='\u{200F}' | '\u{202A}'..='\u{202E}' + | '\u{2060}'..='\u{2064}' | '\u{2066}'..='\u{2069}' | '\u{FEFF}') + { + return false; + } + if !matches!( + get_general_category(c), + GeneralCategory::UppercaseLetter + | GeneralCategory::LowercaseLetter + | GeneralCategory::TitlecaseLetter + | GeneralCategory::ModifierLetter + | GeneralCategory::OtherLetter + | GeneralCategory::NonspacingMark + | GeneralCategory::SpacingMark + | GeneralCategory::EnclosingMark + ) { + return false; + } + + // Reject *compatibility* variants of otherwise-allowed letters — mathematical + // bold (`𝐒`), fullwidth (`A`), enclosed and other presentation forms. + // + // These are genuine letters by category, so the check above admits them, and + // NFC does not fold them (only NFKC would). They matter for two reasons: + // homoglyph spoofing of a real person's name, and the fact that a + // fullwidth-digit alphabet would reopen the very encoding channel §5a's "no + // digits" rule closes. A character that NFKC would rewrite is not the + // character it appears to be, so it is not accepted. + // + // Names are stored as TMDB references anyway (§5a), so the cost of being + // strict here is nil: a real TMDB name is already in normal form. + !is_compatibility_variant(c) +} + +/// True when NFKC rewrites `c` into something other than itself. +fn is_compatibility_variant(c: char) -> bool { + let mut it = c.nfkc(); + match (it.next(), it.next()) { + (Some(first), None) => first != c, + // Decomposes to several characters, so it is certainly not canonical. + (Some(_), Some(_)) => true, + (None, _) => true, + } +} + +/// Validates a free-text field against §5a's permissive-but-closed pattern and +/// returns it NFC-normalised. +fn check_text(field: &str, value: &str, max_chars: usize) -> VResult { + if value.chars().count() > max_chars { + return Err(err(field, format!("longer than {max_chars} characters"))); + } + let normalised: String = if is_nfc(value) { value.to_string() } else { value.nfc().collect() }; + if normalised.chars().count() > max_chars { + return Err(err(field, format!("longer than {max_chars} characters after NFC"))); + } + if let Some(bad) = normalised.chars().find(|c| !is_allowed_text_char(*c)) { + return Err(err(field, format!("contains disallowed character U+{:04X}", bad as u32))); + } + Ok(normalised) +} + +/// §6: reject any string anywhere that looks like an absolute filesystem path +/// or a `file://` URI. +/// +/// `movie` itself is already a parse error via `deny_unknown_fields`; this +/// closes the same leak arriving through a field that *is* allowed. +fn check_not_path_shaped(field: &str, value: &str) -> VResult<()> { + let v = value.trim(); + let looks_like_path = v.starts_with('/') + || v.starts_with("\\\\") + || v.to_ascii_lowercase().starts_with("file://") + || (v.len() >= 3 + && v.as_bytes()[0].is_ascii_alphabetic() + && v.as_bytes()[1] == b':' + && matches!(v.as_bytes()[2], b'\\' | b'/')); + if looks_like_path { + return Err(err(field, "looks like a filesystem path or file:// URI")); + } + Ok(()) +} + +// --------------------------------------------------------------------------- +// Manifest validation +// --------------------------------------------------------------------------- + +pub fn validate_manifest(mut m: Jmanifest) -> VResult { + if m.jmanifest_version != JMANIFEST_VERSION { + return Err(err( + "jmanifest_version", + format!("unsupported version {}, expected {JMANIFEST_VERSION}", m.jmanifest_version), + )); + } + + validate_identity(&mut m.identity)?; + validate_cut(&m)?; + + if let Some(ex) = &m.extraction { + if let Some(pv) = &ex.pipeline_version { + check_not_path_shaped("extraction.pipeline_version", pv)?; + if pv.chars().count() > limits::MAX_TITLE_CHARS { + return Err(err("extraction.pipeline_version", "too long")); + } + } + if let Some(fps) = ex.sample_fps { + if !fps.is_finite() || fps <= 0.0 { + return Err(err("extraction.sample_fps", "must be a positive finite number")); + } + } + if let Some(a) = ex.extinction_sec { + if !a.is_finite() || a < 0.0 { + return Err(err( + "extraction.extinction_sec", + "must be a non-negative finite number", + )); + } + } + } + + let actor_scenes_cs = validate_actors(&m)?; + Ok(ValidManifest { manifest: m, actor_scenes_cs }) +} + +fn validate_identity(id: &mut Identity) -> VResult<()> { + match id.kind { + IdentityType::Movie => { + if id.series_tmdb_id.is_some() || id.series_imdb_id.is_some() { + return Err(err("identity.series_tmdb_id", "not valid for type=movie")); + } + if id.season.is_some() || id.episode.is_some() { + return Err(err("identity.season", "not valid for type=movie")); + } + // §6: neither `tmdb_id` nor `imdb_id` in `identity`. + if id.tmdb_id.is_none() && id.imdb_id.is_none() { + return Err(err("identity", "requires at least one of tmdb_id or imdb_id")); + } + if let Some(t) = &id.tmdb_id { + if !is_tmdb_id(t) { + return Err(err("identity.tmdb_id", "must match ^\\d{1,9}$")); + } + } + if let Some(i) = &id.imdb_id { + if !is_title_imdb_id(i) { + return Err(err("identity.imdb_id", "must match ^tt\\d{7,8}$")); + } + } + } + IdentityType::Episode => { + if id.tmdb_id.is_some() || id.imdb_id.is_some() { + return Err(err( + "identity.tmdb_id", + "use series_tmdb_id / series_imdb_id for type=episode", + )); + } + if id.series_tmdb_id.is_none() && id.series_imdb_id.is_none() { + return Err(err( + "identity", + "requires at least one of series_tmdb_id or series_imdb_id", + )); + } + if let Some(t) = &id.series_tmdb_id { + if !is_tmdb_id(t) { + return Err(err("identity.series_tmdb_id", "must match ^\\d{1,9}$")); + } + } + if let Some(i) = &id.series_imdb_id { + if !is_title_imdb_id(i) { + return Err(err("identity.series_imdb_id", "must match ^tt\\d{7,8}$")); + } + } + // Bounded integers (§5a). + match id.season { + Some(s) if (0..=1000).contains(&s) => {} + Some(_) => return Err(err("identity.season", "out of range 0..=1000")), + None => return Err(err("identity.season", "required for type=episode")), + } + match id.episode { + Some(e) if (0..=10_000).contains(&e) => {} + Some(_) => return Err(err("identity.episode", "out of range 0..=10000")), + None => return Err(err("identity.episode", "required for type=episode")), + } + } + } + + if let Some(y) = id.year { + if !(1870..=2200).contains(&y) { + return Err(err("identity.year", "out of range 1870..=2200")); + } + } + if let Some(t) = &id.title { + check_not_path_shaped("identity.title", t)?; + id.title = Some(check_text("identity.title", t, limits::MAX_TITLE_CHARS)?); + } + Ok(()) +} + +fn validate_cut(m: &Jmanifest) -> VResult<()> { + let rt = m.cut.runtime_sec; + // §2: `cut.runtime_sec` is required — absence is already a parse error, so + // what remains is range. + if !rt.is_finite() || rt <= 0.0 || rt > 200_000.0 { + return Err(err("cut.runtime_sec", "must be a finite duration in (0, 200000]")); + } + if let Some(d) = m.cut.container_duration_sec { + if !d.is_finite() || d <= 0.0 || d > 200_000.0 { + return Err(err("cut.container_duration_sec", "must be a finite duration")); + } + } + if let Some(h) = &m.cut.video_hash { + validate_video_hash(h)?; + } + if let Some(sig) = &m.cut.audio_signature { + validate_audio_signature(sig, rt)?; + } + Ok(()) +} + +/// §3: the OpenSubtitles hash, in the fixed `opensubtitles:<16 hex>` form. +fn validate_video_hash(h: &str) -> VResult<()> { + let Some(hex) = h.strip_prefix("opensubtitles:") else { + return Err(err("cut.video_hash", "must be prefixed 'opensubtitles:'")); + }; + if hex.len() != 16 || !hex.bytes().all(|b| b.is_ascii_hexdigit()) { + return Err(err("cut.video_hash", "expected 16 hex digits after the prefix")); + } + Ok(()) +} + +/// §3 "Validation and abuse": fixed length, base64, and each byte structurally +/// constrained (5-bit bin index + 2-bit energy class). +/// +/// A variable-length blob would be a payload channel — precisely what §5a +/// closes — so length is checked, not merely bounded. +/// +/// `runtime_sec` is needed because §3 shortens the window for very short items; +/// see [`expected_min_frames`]. +pub fn validate_audio_signature(sig: &str, runtime_sec: f64) -> VResult<()> { + // IR-007: media shorter than the window emits **no signature**, and no sync + // offset is applied to it. A signature present on such an item did not come + // from the specified construction, so it is rejected rather than stored — + // whatever it is, it is not the thing this field is for. + if runtime_sec < AUDIO_SIG_MIN_RUNTIME_SEC { + return Err(err( + "cut.audio_signature", + format!( + "must not be present for media shorter than {AUDIO_SIG_MIN_RUNTIME_SEC:.0}s — \ + the {AUDIO_SIG_MIN_RUNTIME_SEC:.0}s analysis window underflows" + ), + )); + } + + let Some(payload) = sig.strip_prefix("v1:") else { + return Err(err("cut.audio_signature", "must be version-prefixed 'v1:'")); + }; + let bytes = base64_decode(payload) + .map_err(|e| err("cut.audio_signature", format!("invalid base64: {e}")))?; + + // §3, and `scene-actor-extraction` IR-007: the length is **fixed by the + // construction**, not merely bounded. A 120 s window at a 1024-sample hop and + // 11025 Hz yields ~1290 frames, and an item too short for that window emits + // no signature at all — the window `runtime/2 ± 60 s` underflows below 120 s, + // so there is nothing to shorten. + // + // That makes the length non-negotiable, which is what keeps the field inside + // SR-004: a caller cannot choose it, so it cannot be used as a variable-size + // container. The tolerance covers seek and encoder differences at the window + // edges, nothing more. + let lo = limits::AUDIO_SIG_FRAMES - limits::AUDIO_SIG_TOLERANCE; + let hi = limits::AUDIO_SIG_FRAMES + limits::AUDIO_SIG_TOLERANCE; + + if bytes.len() < lo || bytes.len() > hi { + return Err(err( + "cut.audio_signature", + format!("decoded length {} outside the fixed {lo}..={hi} frames", bytes.len()), + )); + } + // 5-bit bin index (0..=31) + 2-bit energy class => bit 7 must be clear. + // Arbitrary bytes are therefore invalid, keeping §5a's "no free-form + // storage" property intact. + if let Some(pos) = bytes.iter().position(|b| b & 0x80 != 0) { + return Err(err("cut.audio_signature", format!("frame {pos} has reserved high bit set"))); + } + Ok(()) +} + +fn validate_actors(m: &Jmanifest) -> VResult> { + if m.actors.len() > limits::MAX_ACTORS { + return Err(err("actors", format!("more than {} entries", limits::MAX_ACTORS))); + } + // §6 small-|M| handling: `|M| == 0` is rejected. 15 of the 331 corpus files + // have empty actor lists — extraction failures, not contributions. + if m.actors.is_empty() { + return Err(err("actors", "empty actor list is an extraction failure, not a contribution")); + } + + let mut out = Vec::with_capacity(m.actors.len()); + let mut total_scenes = 0usize; + let mut seen_tmdb: Vec = Vec::new(); + let mut seen_imdb: Vec = Vec::new(); + + for (i, a) in m.actors.iter().enumerate() { + let scenes_cs = validate_scenes(i, a, m.cut.runtime_sec)?; + total_scenes += scenes_cs.len(); + if total_scenes > limits::MAX_TOTAL_SCENES { + return Err(err( + "actors", + format!("more than {} scene windows in total", limits::MAX_TOTAL_SCENES), + )); + } + + let tmdb_id = match &a.tmdb_id { + Some(t) if !t.is_empty() => { + if !is_tmdb_id(t) { + return Err(err(format!("actors[{i}].tmdb_id"), "must match ^\\d{1,9}$")); + } + Some(t.parse::().map_err(|_| { + err(format!("actors[{i}].tmdb_id"), "not a representable integer") + })?) + } + _ => None, + }; + + let imdb_id = match &a.imdb_id { + Some(v) if !v.is_empty() => { + if !is_person_imdb_id(v) { + return Err(err(format!("actors[{i}].imdb_id"), "must match ^nm\\d{7,8}$")); + } + Some(v.clone()) + } + _ => None, + }; + + if tmdb_id.is_none() && imdb_id.is_none() && a.name.as_deref().unwrap_or("").is_empty() { + return Err(err( + format!("actors[{i}]"), + "requires at least one of tmdb_id, imdb_id or name", + )); + } + + // §6: duplicate actors within one manifest. + if let Some(t) = tmdb_id { + if seen_tmdb.contains(&t) { + return Err(err(format!("actors[{i}].tmdb_id"), "duplicate actor in manifest")); + } + seen_tmdb.push(t); + } + if let Some(v) = &imdb_id { + if seen_imdb.contains(v) { + return Err(err(format!("actors[{i}].imdb_id"), "duplicate actor in manifest")); + } + seen_imdb.push(v.clone()); + } + + let name = match &a.name { + Some(n) if !n.is_empty() => { + check_not_path_shaped(&format!("actors[{i}].name"), n)?; + Some(check_text(&format!("actors[{i}].name"), n, limits::MAX_NAME_CHARS)?) + } + _ => None, + }; + + out.push(ActorScenes { name, tmdb_id, imdb_id, scenes_cs }); + } + + Ok(out) +} + +fn validate_scenes(idx: usize, a: &Actor, runtime_sec: f64) -> VResult> { + if a.scenes.len() > limits::MAX_SCENES_PER_ACTOR { + return Err(err( + format!("actors[{idx}].scenes"), + format!("more than {} entries", limits::MAX_SCENES_PER_ACTOR), + )); + } + let max_t = runtime_sec + limits::RUNTIME_TOLERANCE_SEC; + let mut out = Vec::with_capacity(a.scenes.len()); + + for (j, [start, end]) in a.scenes.iter().copied().enumerate() { + let field = format!("actors[{idx}].scenes[{j}]"); + // §6: non-finite values (NaN/Infinity), negative times, `end < start`, + // or times beyond `runtime_sec` + tolerance. + if !start.is_finite() || !end.is_finite() { + return Err(err(field, "non-finite value")); + } + if start < 0.0 || end < 0.0 { + return Err(err(field, "negative time")); + } + if end < start { + return Err(err(field, "end before start")); + } + if end > max_t { + return Err(err( + field, + format!( + "end {end} beyond runtime_sec + {}s tolerance", + limits::RUNTIME_TOLERANCE_SEC + ), + )); + } + out.push((to_centiseconds(start), to_centiseconds(end))); + } + + // §2: scenes are sorted. Checked on the quantised values so the stored form + // is the one guaranteed ordered. + if out.windows(2).any(|w| w[1].0 < w[0].0) { + return Err(err(format!("actors[{idx}].scenes"), "windows must be sorted by start time")); + } + Ok(out) +} + +// --------------------------------------------------------------------------- +// Bundle validation +// --------------------------------------------------------------------------- + +/// Validates the bundle *envelope* only. +/// +/// §4: a malformed envelope is a whole-request `400`, whereas individual bad +/// episodes are reported in the per-episode results list — the bundle is not +/// atomic, because all-or-nothing would let one bad episode discard an entire +/// season's compute (§2). +pub fn validate_bundle_envelope(b: &SeriesBundle) -> VResult<()> { + if b.jmanifest_version != JMANIFEST_VERSION { + return Err(err( + "jmanifest_version", + format!("unsupported version {}, expected {JMANIFEST_VERSION}", b.jmanifest_version), + )); + } + if b.series.series_tmdb_id.is_none() && b.series.series_imdb_id.is_none() { + return Err(err("series", "requires at least one of series_tmdb_id or series_imdb_id")); + } + if let Some(t) = &b.series.series_tmdb_id { + if !is_tmdb_id(t) { + return Err(err("series.series_tmdb_id", "must match ^\\d{1,9}$")); + } + } + if let Some(i) = &b.series.series_imdb_id { + if !is_title_imdb_id(i) { + return Err(err("series.series_imdb_id", "must match ^tt\\d{7,8}$")); + } + } + if let Some(t) = &b.series.title { + check_not_path_shaped("series.title", t)?; + check_text("series.title", t, limits::MAX_TITLE_CHARS)?; + } + if b.episodes.is_empty() { + return Err(err("episodes", "bundle contains no episodes")); + } + if b.episodes.len() > limits::MAX_BUNDLE_EPISODES { + return Err(err("episodes", format!("more than {} episodes", limits::MAX_BUNDLE_EPISODES))); + } + Ok(()) +} + +// --------------------------------------------------------------------------- +// base64 (standard alphabet, padded) +// --------------------------------------------------------------------------- + +/// Minimal standard-alphabet base64 decoder. +/// +/// Vendored rather than pulled in as a dependency: the only base64 in this +/// service is the fixed-format audio signature, and §3 argues for keeping the +/// dependency surface small on the same grounds as the plugin's FFT. +fn base64_decode(s: &str) -> Result, &'static str> { + fn val(b: u8) -> Result { + match b { + b'A'..=b'Z' => Ok(b - b'A'), + b'a'..=b'z' => Ok(b - b'a' + 26), + b'0'..=b'9' => Ok(b - b'0' + 52), + b'+' => Ok(62), + b'/' => Ok(63), + _ => Err("character outside the base64 alphabet"), + } + } + + let bytes = s.as_bytes(); + if bytes.len() % 4 != 0 { + return Err("length is not a multiple of 4"); + } + if bytes.is_empty() { + return Ok(Vec::new()); + } + + let mut out = Vec::with_capacity(bytes.len() / 4 * 3); + for (i, chunk) in bytes.chunks(4).enumerate() { + let last = i == bytes.len() / 4 - 1; + let pad = if last { + chunk.iter().filter(|&&b| b == b'=').count() + } else { + if chunk.contains(&b'=') { + return Err("padding before the final chunk"); + } + 0 + }; + if pad > 2 { + return Err("more than two padding characters"); + } + let mut acc = 0u32; + for (k, &b) in chunk.iter().enumerate() { + let v = if b == b'=' { + if k < 4 - pad { + return Err("padding in a data position"); + } + 0 + } else { + val(b)? + }; + acc = (acc << 6) | v as u32; + } + let triple = acc.to_be_bytes(); + out.push(triple[1]); + if pad < 2 { + out.push(triple[2]); + } + if pad < 1 { + out.push(triple[3]); + } + } + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn base_manifest() -> Jmanifest { + serde_json::from_str( + r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172","title":"The Death of Stalin"}, + "cut":{"runtime_sec":6420.5}, + "actors":[{"name":"Steve Buscemi","tmdb_id":"884","scenes":[[191.6,209.2]]}]}"#, + ) + .unwrap() + } + + #[test] + fn accepts_a_realistic_manifest() { + let v = validate_manifest(base_manifest()).unwrap(); + assert_eq!(v.actor_scenes_cs.len(), 1); + assert_eq!(v.actor_scenes_cs[0].scenes_cs, vec![(19160, 20920)]); + } + + #[test] + fn windows_are_never_reshaped() { + // UR-013 / SR-002: a window is a claim about *scene membership*, not a + // recognition event. The server therefore stores what it was given — + // quantised, but never merged, split or trimmed. + // + // The adjacent-window case is the one that matters: a naive + // implementation might "tidy" two windows that touch into one, which + // would destroy the distinction SR-002 draws between an actor who turned + // away (one window, gap absorbed by the producer) and one who genuinely + // left and returned (two windows). + let mut m = base_manifest(); + m.actors[0].scenes = vec![ + [10.0, 20.0], + [20.0, 30.0], // exactly adjacent — must stay separate + [30.01, 40.0], // a hair's gap — likewise + [100.0, 100.0], // zero-length — a real producer emits these + ]; + let v = validate_manifest(m).unwrap(); + assert_eq!( + v.actor_scenes_cs[0].scenes_cs, + vec![(1000, 2000), (2000, 3000), (3001, 4000), (10000, 10000)], + "windows must survive validation unchanged apart from quantisation" + ); + } + + #[test] + fn extinction_sec_replaces_anneal_sec() { + // The SR-003 withdrawal. `anneal_sec` cannot even be constructed here — + // it is not a field on `Extraction` — so this asserts the successor is + // accepted and range-checked; `tests/api.rs` covers the wire rejection. + let mut m = base_manifest(); + m.extraction = Some(crate::model::Extraction { + sample_fps: Some(5.0), + extinction_sec: Some(12.0), + pipeline_version: Some("test 0.1".into()), + gallery_size: Some(1820), + gallery_scope: Some(crate::model::GalleryScope::Global), + }); + assert!(validate_manifest(m).is_ok()); + + let mut m = base_manifest(); + m.extraction = Some(crate::model::Extraction { + sample_fps: None, + extinction_sec: Some(-1.0), + pipeline_version: None, + gallery_size: None, + gallery_scope: None, + }); + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "extraction.extinction_sec"); + } + + #[test] + fn rejects_empty_actor_list() { + let mut m = base_manifest(); + m.actors.clear(); + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "actors"); + } + + #[test] + fn rejects_scene_beyond_runtime_tolerance() { + let mut m = base_manifest(); + m.actors[0].scenes = vec![[10.0, 6500.0]]; + let e = validate_manifest(m).unwrap_err(); + assert!(e.field.starts_with("actors[0].scenes"), "got {}", e.field); + } + + #[test] + fn accepts_scene_within_runtime_tolerance() { + let mut m = base_manifest(); + m.actors[0].scenes = vec![[10.0, 6424.0]]; + assert!(validate_manifest(m).is_ok()); + } + + #[test] + fn rejects_end_before_start_and_negative_and_nonfinite() { + for scenes in [ + vec![[50.0, 10.0]], + vec![[-1.0, 10.0]], + vec![[f64::NAN, 10.0]], + vec![[0.0, f64::INFINITY]], + ] { + let mut m = base_manifest(); + m.actors[0].scenes = scenes; + assert!(validate_manifest(m).is_err()); + } + } + + #[test] + fn rejects_unsorted_scenes() { + let mut m = base_manifest(); + m.actors[0].scenes = vec![[100.0, 120.0], [10.0, 20.0]]; + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "actors[0].scenes"); + } + + #[test] + fn rejects_duplicate_actor() { + let mut m = base_manifest(); + m.actors.push(m.actors[0].clone()); + let e = validate_manifest(m).unwrap_err(); + assert!(e.reason.contains("duplicate"), "got {}", e.reason); + } + + #[test] + fn rejects_bad_identifier_formats() { + let mut m = base_manifest(); + m.identity.imdb_id = Some("tt123".into()); + assert!(validate_manifest(m).is_err()); + + let mut m = base_manifest(); + m.identity.tmdb_id = Some("504172x".into()); + assert!(validate_manifest(m).is_err()); + + let mut m = base_manifest(); + m.actors[0].imdb_id = Some("tt0000114".into()); // title id in a person field + assert!(validate_manifest(m).is_err()); + } + + #[test] + fn requires_an_identity_key() { + let mut m = base_manifest(); + m.identity.tmdb_id = None; + m.identity.imdb_id = None; + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "identity"); + } + + #[test] + fn episode_identity_requires_season_and_episode() { + let json = r#"{"jmanifest_version":1, + "identity":{"type":"episode","series_tmdb_id":"1396","title":"Breaking Bad"}, + "cut":{"runtime_sec":2820.0}, + "actors":[{"name":"Bryan Cranston","tmdb_id":"17419","scenes":[[10.0,20.0]]}]}"#; + let m: Jmanifest = serde_json::from_str(json).unwrap(); + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "identity.season"); + } + + #[test] + fn valid_episode_identity_is_accepted() { + let json = r#"{"jmanifest_version":1, + "identity":{"type":"episode","series_tmdb_id":"1396","series_imdb_id":"tt0903747", + "title":"Breaking Bad","season":2,"episode":5}, + "cut":{"runtime_sec":2820.0}, + "actors":[{"name":"Bryan Cranston","tmdb_id":"17419","scenes":[[10.0,20.0]]}]}"#; + let m: Jmanifest = serde_json::from_str(json).unwrap(); + assert!(validate_manifest(m).is_ok()); + } + + #[test] + fn movie_identity_rejects_episode_coordinates() { + let mut m = base_manifest(); + m.identity.season = Some(1); + assert!(validate_manifest(m).is_err()); + } + + // §5a: the character class alone must defeat base64/hex smuggling, which + // needs digits and padding characters. + #[test] + fn name_character_class_rejects_smuggling() { + for name in [ + "SGVsbG8gd29ybGQ=", // base64 + "deadbeef1234", // hex + "Steve/Buscemi", + "Steve+Buscemi", + "Actor 2", // digits + "Steve\u{200B}Buscemi", // zero-width space + "Steve\u{202E}imecsuB", // bidi override + "Steve\u{0007}Buscemi", // control character + "", + ] { + let mut m = base_manifest(); + m.actors[0].name = Some(name.to_string()); + assert!( + validate_manifest(m).is_err(), + "name {name:?} should be rejected by the §5a character class" + ); + } + } + + #[test] + fn name_character_class_rejects_compatibility_homoglyphs() { + // Another gap an injection test caught. These are letters by Unicode + // category, so a category-only check admits them, and NFC does not fold + // them — only NFKC would. Two problems: they spoof a real person's name, + // and a fullwidth-digit alphabet would reopen the encoding channel that + // §5a's "no digits" rule exists to close. + for name in [ + "𝐒𝐭𝐞𝐯𝐞 𝐁𝐮𝐬𝐜𝐞𝐦𝐢", // mathematical bold + "Steve", // fullwidth + "ⓈⓉⒺⓋⒺ", // enclosed alphanumerics + "STEVE 123", // fullwidth with digits + "film", // ligature + "Ⅻ", // Roman numeral + ] { + let mut m = base_manifest(); + m.actors[0].name = Some(name.to_string()); + assert!( + validate_manifest(m).is_err(), + "compatibility homoglyph {name:?} should be rejected" + ); + } + } + + #[test] + fn name_character_class_accepts_real_names() { + for name in [ + "Steve Buscemi", + "Michael Palin", + "Jean-Luc Picard", + "Renée Zellweger", + "Hayao Miyazaki", + "宮崎 駿", + "Miloš Forman", + "O'Brien", + "Sammy Davis, Jr.", + ] { + let mut m = base_manifest(); + m.actors[0].name = Some(name.to_string()); + assert!(validate_manifest(m).is_ok(), "name {name:?} should be accepted"); + } + } + + #[test] + fn rejects_path_shaped_strings_in_allowed_fields() { + for probe in ["/data/movies/x.mkv", "C:\\media\\x.mkv", "\\\\nas\\media", "file:///x"] { + let mut m = base_manifest(); + m.identity.title = Some(probe.to_string()); + assert!(validate_manifest(m).is_err(), "{probe} should be rejected"); + } + } + + #[test] + fn rejects_overlong_name() { + let mut m = base_manifest(); + m.actors[0].name = Some("a".repeat(limits::MAX_NAME_CHARS + 1)); + assert!(validate_manifest(m).is_err()); + } + + #[test] + fn rejects_too_many_actors() { + let mut m = base_manifest(); + let a = m.actors[0].clone(); + m.actors = (0..=limits::MAX_ACTORS) + .map(|i| { + let mut c = a.clone(); + c.tmdb_id = Some((1000 + i).to_string()); + c + }) + .collect(); + let e = validate_manifest(m).unwrap_err(); + assert_eq!(e.field, "actors"); + } + + #[test] + fn video_hash_format_is_enforced() { + let mut m = base_manifest(); + m.cut.video_hash = Some("opensubtitles:8e245d9679d31e12".into()); + assert!(validate_manifest(m).is_ok()); + + for bad in ["8e245d9679d31e12", "opensubtitles:xyz", "opensubtitles:8e245d9679d31e1"] { + let mut m = base_manifest(); + m.cut.video_hash = Some(bad.into()); + assert!(validate_manifest(m).is_err(), "{bad} should be rejected"); + } + } + + #[test] + fn centisecond_quantisation_is_stable_for_accumulated_float_error() { + // §9a: real corpus values look like 8045.066666660665. + assert_eq!(to_centiseconds(8045.066666660665), 804507); + assert_eq!(to_centiseconds(8045.066666666), 804507); + assert_eq!(to_centiseconds(0.0), 0); + } + + #[test] + fn base64_roundtrip() { + // "Man" => "TWFu"; padding variants. + assert_eq!(base64_decode("TWFu").unwrap(), b"Man"); + assert_eq!(base64_decode("TWE=").unwrap(), b"Ma"); + assert_eq!(base64_decode("TQ==").unwrap(), b"M"); + assert!(base64_decode("TWF").is_err()); + assert!(base64_decode("TW$u").is_err()); + assert!(base64_decode("T=Fu").is_err()); + } + + /// A feature-length runtime, so the full window applies. + const FEATURE_RUNTIME: f64 = 6420.5; + + #[test] + fn audio_signature_validation() { + // 1290 frames with the high bit clear, base64-encoded. + let frames = vec![0x3Fu8; limits::AUDIO_SIG_FRAMES]; + let sig = format!("v1:{}", base64_encode_for_test(&frames)); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_ok()); + + // Missing version prefix. + assert!( + validate_audio_signature(&base64_encode_for_test(&frames), FEATURE_RUNTIME).is_err() + ); + + // High bit set is structurally invalid, so arbitrary bytes cannot ride + // along in this field (§3, §5a). + let mut bad = frames.clone(); + bad[7] = 0xFF; + let sig = format!("v1:{}", base64_encode_for_test(&bad)); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_err()); + + // Oversized blob would be a payload channel. + let huge = vec![0x01u8; limits::AUDIO_SIG_FRAMES + limits::AUDIO_SIG_TOLERANCE + 1]; + let sig = format!("v1:{}", base64_encode_for_test(&huge)); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_err()); + } + + #[test] + fn media_below_the_window_must_carry_no_signature() { + // IR-007, reconciled with server §3: the window `runtime/2 ± 60 s` + // underflows below 120 s, so no signature exists to send. One present on + // such an item did not come from the specified construction, whatever it + // is. An earlier draft of §3 allowed a shortened window here; that was + // the weaker rule, because a caller-varying length is exactly the + // property SR-004 forbids. + let frames = vec![0x3Fu8; limits::AUDIO_SIG_FRAMES]; + let sig = format!("v1:{}", base64_encode_for_test(&frames)); + + for runtime in [1.0f64, 30.0, 119.0, 119.999] { + let e = validate_audio_signature(&sig, runtime).unwrap_err(); + assert_eq!(e.field, "cut.audio_signature"); + assert!( + e.reason.contains("shorter than"), + "a {runtime}s item should be refused on its runtime: {}", + e.reason + ); + } + + // At and above the window, the normal rules apply. + assert!(validate_audio_signature(&sig, 120.0).is_ok()); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_ok()); + } + + #[test] + fn the_signature_length_is_fixed_not_caller_chosen() { + // What keeps the field inside SR-004: the length is a property of the + // construction, so a caller cannot use it as a variable-size container. + for frames in [1usize, 16, 100, 900, 1200] { + let sig = format!("v1:{}", base64_encode_for_test(&vec![0x3Fu8; frames])); + assert!( + validate_audio_signature(&sig, FEATURE_RUNTIME).is_err(), + "{frames} frames should be rejected — the length is fixed" + ); + } + // Only the construction's own length, within edge tolerance, is accepted. + for frames in [ + limits::AUDIO_SIG_FRAMES - limits::AUDIO_SIG_TOLERANCE, + limits::AUDIO_SIG_FRAMES, + limits::AUDIO_SIG_FRAMES + limits::AUDIO_SIG_TOLERANCE, + ] { + let sig = format!("v1:{}", base64_encode_for_test(&vec![0x3Fu8; frames])); + assert!(validate_audio_signature(&sig, FEATURE_RUNTIME).is_ok(), "{frames} frames"); + } + } + + fn base64_encode_for_test(data: &[u8]) -> String { + const A: &[u8] = b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; + let mut out = String::new(); + for chunk in data.chunks(3) { + let b = [chunk[0], *chunk.get(1).unwrap_or(&0), *chunk.get(2).unwrap_or(&0)]; + let n = u32::from_be_bytes([0, b[0], b[1], b[2]]); + out.push(A[(n >> 18 & 63) as usize] as char); + out.push(A[(n >> 12 & 63) as usize] as char); + out.push(if chunk.len() > 1 { A[(n >> 6 & 63) as usize] as char } else { '=' }); + out.push(if chunk.len() > 2 { A[(n & 63) as usize] as char } else { '=' }); + } + out + } + + #[test] + fn bundle_envelope_checks() { + let ok = r#"{"jmanifest_version":1, + "series":{"series_tmdb_id":"1396","title":"Breaking Bad"}, + "episodes":[{"jmanifest_version":1, + "identity":{"type":"episode","series_tmdb_id":"1396","season":1,"episode":1}, + "cut":{"runtime_sec":2820.0}, + "actors":[{"tmdb_id":"17419","scenes":[[1.0,2.0]]}]}]}"#; + let b: SeriesBundle = serde_json::from_str(ok).unwrap(); + assert!(validate_bundle_envelope(&b).is_ok()); + + let mut empty = b.clone(); + empty.episodes.clear(); + assert!(validate_bundle_envelope(&empty).is_err()); + + let mut no_id = b.clone(); + no_id.series.series_tmdb_id = None; + no_id.series.series_imdb_id = None; + assert!(validate_bundle_envelope(&no_id).is_err()); + } +} diff --git a/src/worker.rs b/src/worker.rs new file mode 100644 index 0000000..1538f23 --- /dev/null +++ b/src/worker.rs @@ -0,0 +1,494 @@ +//! Background worker for the §6 stage 3 cast check. +//! +//! §8: this runs as a Tokio background task in the same binary, with the job +//! queue as a SQLite table so state survives restart — replacing an external +//! broker entirely. The check needs an outbound TMDB call and so cannot run +//! inside the request without coupling upload latency to a third party (§6). + +use std::sync::Arc; +use std::time::Duration; + +use anyhow::Context; + +use crate::castcheck::{self, SubmittedActor, Verdict}; +use crate::db::{repo, Db}; +use crate::ingest::{CastCheckJob, JOB_CAST_CHECK}; +use crate::tmdb::{CastMember, Credits, TmdbClient, TmdbError}; + +/// §6: TMDB responses are cached for 24h, so a burst of episode uploads for one +/// series costs a single upstream call. +const CACHE_TTL: Duration = Duration::from_secs(24 * 3600); +/// Cap on retry backoff for a persistent TMDB outage. +const MAX_BACKOFF_SECS: u64 = 3600; + +pub struct Worker { + pub db: Db, + pub tmdb: Arc, + pub batch: usize, + pub poll_interval: Duration, +} + +impl Worker { + /// Runs until `shutdown` resolves. + pub async fn run(self, mut shutdown: tokio::sync::watch::Receiver) { + // A process that died mid-job would otherwise leave work stranded. + match self.db.write(repo::release_all_leases).await { + Ok(n) if n > 0 => tracing::info!(released = n, "released stranded job leases"), + Ok(_) => {} + Err(e) => tracing::error!(error = ?e, "failed to release job leases at startup"), + } + + loop { + tokio::select! { + _ = shutdown.changed() => { + tracing::info!("worker shutting down"); + return; + } + _ = tokio::time::sleep(self.poll_interval) => { + if let Err(e) = self.tick().await { + tracing::error!(error = ?e, "worker tick failed"); + } + } + } + } + } + + async fn tick(&self) -> anyhow::Result<()> { + let now = now_iso(); + let batch = self.batch; + let leased_at = now.clone(); + let jobs = self.db.write(move |tx| repo::lease_jobs(tx, &leased_at, batch)).await?; + + for job in jobs { + let result = match job.kind.as_str() { + JOB_CAST_CHECK => self.run_cast_check(&job.payload).await, + other => { + tracing::warn!(kind = other, "unknown job kind, dropping"); + Ok(()) + } + }; + + let job_id = job.id.clone(); + match result { + Ok(()) => { + self.db.write(move |tx| repo::delete_job(tx, &job_id)).await?; + } + Err(JobError::Retry(msg)) => { + // §6: TMDB unreachable or rate-limited means retry with + // backoff; the manifest stays unlisted, not rejected. + let delay = backoff_secs(job.attempts); + let run_after = iso_in(delay); + tracing::warn!(job = %job_id, attempts = job.attempts, delay, reason = %msg, + "rescheduling job"); + self.db + .write(move |tx| repo::reschedule_job(tx, &job_id, &run_after, &msg)) + .await?; + } + Err(JobError::Fatal(e)) => { + tracing::error!(job = %job_id, error = ?e, "dropping job after fatal error"); + self.db.write(move |tx| repo::delete_job(tx, &job_id)).await?; + } + } + } + Ok(()) + } + + async fn run_cast_check(&self, payload: &str) -> Result<(), JobError> { + let job: CastCheckJob = + serde_json::from_str(payload).map_err(|e| JobError::Fatal(e.into()))?; + let manifest_id = job.manifest_id; + + // Load what the check needs. + let mid = manifest_id.clone(); + let loaded = self + .db + .read(move |conn| { + let Some(m) = repo::manifest_by_id(conn, &mid)? else { return Ok(None) }; + let title = conn + .query_row( + "SELECT kind, tmdb_id, imdb_id, adult, certification FROM titles WHERE id = ?1", + rusqlite::params![m.title_id], + |r| { + Ok(( + r.get::<_, String>(0)?, + r.get::<_, Option>(1)?, + r.get::<_, Option>(2)?, + r.get::<_, i64>(3)? != 0, + r.get::<_, Option>(4)?, + )) + }, + ) + .map_err(anyhow::Error::from)?; + let actor_ids = repo::manifest_actor_ids(conn, &mid)?; + Ok(Some((m, title, actor_ids))) + }) + .await + .map_err(JobError::Fatal)?; + + // The manifest may have been deleted (contributor revoked, §5a) between + // enqueue and now; that is not an error. + let Some((manifest, (kind, tmdb_id, _imdb_id, title_adult, certification), actor_ids)) = + loaded + else { + return Ok(()); + }; + if manifest.status != "pending" { + return Ok(()); + } + + let Some(tmdb_id) = tmdb_id else { + // No TMDB id means the cast check cannot run at all. §6 treats absent + // reference data as flagged, not rejected. + self.finalise(&manifest_id, Verdict::Flagged, 0.0, Some("no_tmdb_id"), &[], &[]) + .await + .map_err(JobError::Fatal)?; + return Ok(()); + }; + + let credits = self + .credits_for(&kind, &tmdb_id, manifest.season, manifest.episode) + .await + .map_err(|e| { + if e.is_retryable() { + JobError::Retry(e.to_string()) + } else { + // A genuinely absent title is a verdict, not a transport + // failure — handled below via empty credits. + JobError::Retry(format!("non-retryable tmdb error treated as absent: {e}")) + } + }); + + let credits = match credits { + Ok(c) => c, + Err(JobError::Retry(msg)) if msg.starts_with("non-retryable") => { + tracing::info!(manifest = %manifest_id, "tmdb has no such title; flagging"); + Credits::default() + } + Err(e) => return Err(e), + }; + + let reference: Vec = credits.all().cloned().collect(); + let submitted: Vec = actor_ids + .iter() + .map(|id| SubmittedActor { tmdb_id: Some(*id), imdb_id: None, name: None }) + .collect(); + + let mut outcome = castcheck::evaluate(&submitted, &reference); + + // §5a layer 1 — category guard. + if let Some(offender) = castcheck::category_guard_violation(&outcome.matched, title_adult) { + tracing::warn!(manifest = %manifest_id, person = offender, + "category guard: adult-flagged performer on a non-adult title"); + outcome.verdict = Verdict::Rejected; + outcome.reason = Some("category_guard".into()); + } + + // §5a layer 2 — age-appropriateness guard. + if let Some(cert) = &certification { + if castcheck::is_childrens_certification(cert) { + castcheck::apply_childrens_guard(&mut outcome, submitted.len()); + } + } + + self.finalise( + &manifest_id, + outcome.verdict, + outcome.ratio, + outcome.reason.as_deref(), + &outcome.matched, + &outcome.unmatched_person_ids, + ) + .await + .map_err(JobError::Fatal)?; + + Ok(()) + } + + /// Fetches credits, using the 24h cache (§6). + /// + /// For episodes this is the **union** of TMDB's per-episode credits (cast + + /// guest stars) and the series' aggregate credits: per-episode alone would + /// reject recurring cast TMDB lists only at series level, series-wide alone + /// would reject legitimate guest stars. + async fn credits_for( + &self, + kind: &str, + tmdb_id: &str, + season: Option, + episode: Option, + ) -> Result { + if kind == "movie" { + return self.cached("movie", tmdb_id, || self.tmdb.movie_credits(tmdb_id)).await; + } + + let series = self.cached("series", tmdb_id, || self.tmdb.series_credits(tmdb_id)).await?; + + let mut combined = series; + if let (Some(s), Some(e)) = (season, episode) { + let key = format!("{tmdb_id}:{s}:{e}"); + match self.cached("episode", &key, || self.tmdb.episode_credits(tmdb_id, s, e)).await { + Ok(ep) => { + combined.cast.extend(ep.cast); + combined.guest_stars.extend(ep.guest_stars); + } + // A missing episode entry is normal; the series set still applies. + Err(TmdbError::NotFound) => {} + Err(e) if e.is_retryable() => return Err(e), + Err(e) => tracing::warn!(error = ?e, "ignoring episode credits error"), + } + } + Ok(combined) + } + + async fn cached(&self, kind: &str, key: &str, fetch: F) -> Result + where + F: FnOnce() -> Fut, + Fut: std::future::Future>, + { + let (k, kk) = (key.to_string(), kind.to_string()); + let cached = self + .db + .read(move |conn| repo::cached_credits(conn, &k, &kk)) + .await + .map_err(|e| TmdbError::Transport(e.to_string()))?; + + if let Some((json, fetched_at)) = cached { + if !is_stale(&fetched_at, CACHE_TTL) { + if let Ok(c) = serde_json::from_str::(&json) { + return Ok(c); + } + } + } + + let fresh = fetch().await?; + let json = serde_json::to_string(&SerializableCredits::from(&fresh)) + .map_err(|e| TmdbError::Malformed(e.to_string()))?; + let (k, kk, now) = (key.to_string(), kind.to_string(), now_iso()); + let _ = self.db.write(move |tx| repo::put_credits(tx, &k, &kk, &json, &now)).await; + Ok(fresh) + } + + /// Applies the verdict: resolves names into `people`, drops unmatched actors, + /// and updates status and contributor counters — in one transaction. + async fn finalise( + &self, + manifest_id: &str, + verdict: Verdict, + ratio: f64, + reason: Option<&str>, + matched: &[castcheck::MatchedActor], + unmatched: &[u64], + ) -> anyhow::Result<()> { + let id = manifest_id.to_string(); + let reason = reason.map(str::to_string); + let matched: Vec<(u64, String, bool)> = + matched.iter().map(|m| (m.tmdb_person_id, m.name.clone(), m.adult)).collect(); + let unmatched = unmatched.to_vec(); + let now = now_iso(); + + self.db + .write(move |tx| { + let contributor: Option = tx + .query_row( + "SELECT contributor_id FROM manifests WHERE id = ?1", + rusqlite::params![id], + |r| r.get(0), + ) + .map_err(anyhow::Error::from)?; + + if verdict == Verdict::Rejected { + // §6: the manifest is deleted and the contributor notified + // (via `GET /manifests/{id}/status` until it is gone). + repo::delete_manifest(tx, &id)?; + } else { + // Names come from TMDB, never from the upload (§5a, §7). + for (person_id, name, adult) in &matched { + repo::upsert_person(tx, *person_id, name, *adult, &now)?; + } + // §6: unmatched actors are dropped rather than stored. + for person_id in &unmatched { + repo::delete_manifest_actor(tx, &id, *person_id)?; + } + repo::set_manifest_status( + tx, + &id, + verdict.status(), + reason.as_deref(), + Some(ratio), + )?; + } + + if let Some(c) = contributor { + let counter = match verdict { + Verdict::Listed => "accepted", + Verdict::Flagged => "flagged", + Verdict::Rejected => "rejected", + }; + repo::bump_contributor_counter(tx, &c, counter)?; + if repo::maybe_revoke_contributor(tx, &c, &now)? { + tracing::warn!(contributor = %c, "revoked token for excessive rejections"); + } + } + Ok(()) + }) + .await + .context("finalising cast check") + } +} + +/// Serialisable projection of `Credits` for the cache. +#[derive(serde::Serialize)] +struct SerializableCredits { + cast: Vec, + guest_stars: Vec, +} + +#[derive(serde::Serialize)] +struct SerializableMember { + id: u64, + name: String, + adult: bool, +} + +impl From<&Credits> for SerializableCredits { + fn from(c: &Credits) -> Self { + let f = + |m: &CastMember| SerializableMember { id: m.id, name: m.name.clone(), adult: m.adult }; + Self { + cast: c.cast.iter().map(f).collect(), + guest_stars: c.guest_stars.iter().map(&f).collect(), + } + } +} + +enum JobError { + Retry(String), + Fatal(anyhow::Error), +} + +/// Exponential backoff, capped (§5: the client must back off exponentially +/// rather than retrying tightly; the same discipline applies to our own +/// outbound calls). +fn backoff_secs(attempts: i64) -> u64 { + let base = 30u64; + base.saturating_mul(1u64 << attempts.clamp(0, 8) as u32).min(MAX_BACKOFF_SECS) +} + +fn is_stale(fetched_at: &str, ttl: Duration) -> bool { + let Some(then) = parse_iso(fetched_at) else { return true }; + let now = unix_now(); + now.saturating_sub(then) > ttl.as_secs() +} + +/// Current time as an RFC 3339 UTC string, which is what every timestamp column +/// stores. Kept in one place so the format cannot drift. +pub fn now_iso() -> String { + iso_in(0) +} + +pub fn iso_in(secs: u64) -> String { + format_unix(unix_now() + secs) +} + +fn unix_now() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_secs()) + .unwrap_or(0) +} + +/// Formats a Unix timestamp as `YYYY-MM-DDTHH:MM:SSZ`. +/// +/// Hand-rolled rather than pulling in `chrono`/`time`: the only requirement is a +/// lexicographically-sortable UTC string, which is what the `jobs.run_after` +/// comparison relies on. +pub fn format_unix(mut secs: u64) -> String { + let days = secs / 86_400; + secs %= 86_400; + let (h, m, s) = (secs / 3600, (secs % 3600) / 60, secs % 60); + + // Civil-from-days, Howard Hinnant's algorithm. + let z = days as i64 + 719_468; + let era = z.div_euclid(146_097); + let doe = z.rem_euclid(146_097); + let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365; + let y = yoe + era * 400; + let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); + let mp = (5 * doy + 2) / 153; + let d = doy - (153 * mp + 2) / 5 + 1; + let mo = if mp < 10 { mp + 3 } else { mp - 9 }; + let y = if mo <= 2 { y + 1 } else { y }; + + format!("{y:04}-{mo:02}-{d:02}T{h:02}:{m:02}:{s:02}Z") +} + +fn parse_iso(s: &str) -> Option { + // Parses the format `format_unix` produces. + let b = s.as_bytes(); + if b.len() < 20 { + return None; + } + let num = |from: usize, to: usize| s.get(from..to)?.parse::().ok(); + let (y, mo, d) = (num(0, 4)?, num(5, 7)?, num(8, 10)?); + let (h, mi, se) = (num(11, 13)?, num(14, 16)?, num(17, 19)?); + + let y_adj = if mo <= 2 { y - 1 } else { y }; + let era = y_adj.div_euclid(400); + let yoe = y_adj - era * 400; + let mp = if mo > 2 { mo - 3 } else { mo + 9 }; + let doy = (153 * mp + 2) / 5 + d - 1; + let doe = yoe * 365 + yoe / 4 - yoe / 100 + doy; + let days = era * 146_097 + doe - 719_468; + + Some((days * 86_400 + h * 3600 + mi * 60 + se).max(0) as u64) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn timestamp_roundtrips() { + for t in [0u64, 1, 1_000_000, 1_700_000_000, 1_785_000_000, 4_000_000_000] { + let s = format_unix(t); + assert_eq!(parse_iso(&s), Some(t), "roundtrip failed for {t} => {s}"); + } + } + + #[test] + fn timestamp_format_is_sortable() { + // `jobs.run_after <= ?1` is a string comparison, so lexical order must + // match chronological order. + let a = format_unix(1_700_000_000); + let b = format_unix(1_700_000_001); + let c = format_unix(1_800_000_000); + assert!(a < b && b < c, "{a} {b} {c}"); + assert_eq!(format_unix(0), "1970-01-01T00:00:00Z"); + } + + #[test] + fn known_dates_format_correctly() { + // 2026-07-30T12:00:00Z + assert_eq!(format_unix(1_785_412_800), "2026-07-30T12:00:00Z"); + // A leap day, since the civil-from-days algorithm is where this would + // break. + assert_eq!(format_unix(1_709_164_800), "2024-02-29T00:00:00Z"); + } + + #[test] + fn backoff_grows_and_is_capped() { + assert_eq!(backoff_secs(0), 30); + assert_eq!(backoff_secs(1), 60); + assert_eq!(backoff_secs(4), 480); + assert_eq!(backoff_secs(50), MAX_BACKOFF_SECS, "must not overflow or grow unbounded"); + } + + #[test] + fn staleness_uses_the_ttl() { + let fresh = format_unix(unix_now()); + assert!(!is_stale(&fresh, CACHE_TTL)); + let old = format_unix(unix_now() - 25 * 3600); + assert!(is_stale(&old, CACHE_TTL)); + assert!(is_stale("not-a-timestamp", CACHE_TTL)); + } +} diff --git a/tests/api.rs b/tests/api.rs new file mode 100644 index 0000000..e6dc1a0 --- /dev/null +++ b/tests/api.rs @@ -0,0 +1,953 @@ +//! End-to-end tests through the real router. +//! +//! The unit tests cover each spec rule in isolation; these cover the wiring — +//! status codes, headers, and the properties that only hold if the layers are +//! composed correctly (per-route body caps, rate-limit surfaces, the strict +//! schema actually reaching uploads). + +use std::sync::Arc; + +use axum::body::Body; +use axum::http::{Request, StatusCode}; +use http_body_util::BodyExt; +use jray_server::app; +use jray_server::config::Config; +use jray_server::db::Db; +use jray_server::ratelimit::RateLimiter; +use jray_server::state::AppState; +use jray_server::tmdb::TmdbClient; +use serde_json::{json, Value}; +use tower::ServiceExt; + +/// A server backed by a temporary on-disk database. +/// +/// On-disk rather than `:memory:` because §8's design uses a separate writer +/// connection and a read pool, and in-memory SQLite is per-connection — the +/// readers would see an empty database. Testing the real topology is the point. +struct TestServer { + router: axum::Router, + _dir: TempDir, +} + +struct TempDir(std::path::PathBuf); + +impl TempDir { + fn new(tag: &str) -> Self { + let mut p = std::env::temp_dir(); + // Unique per test without pulling in a tempfile dependency. + p.push(format!("jray-test-{}-{}", tag, ulid_like())); + std::fs::create_dir_all(&p).expect("creating temp dir"); + Self(p) + } + + fn db_path(&self) -> String { + self.0.join("test.db").to_string_lossy().into_owned() + } +} + +impl Drop for TempDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } +} + +fn ulid_like() -> String { + use std::sync::atomic::{AtomicU64, Ordering}; + static N: AtomicU64 = AtomicU64::new(0); + let t = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_nanos()) + .unwrap_or(0); + format!("{t}-{}", N.fetch_add(1, Ordering::Relaxed)) +} + +impl TestServer { + fn new(tag: &str) -> Self { + let dir = TempDir::new(tag); + let db = Db::open(&dir.db_path()).expect("opening database"); + let config = Arc::new(Config { + bind: "127.0.0.1:0".into(), + db_path: dir.db_path(), + // No key: uploads stay `pending`, which is the correct failure mode + // (§8) and keeps these tests free of network calls. + tmdb_api_key: None, + tmdb_base_url: "http://127.0.0.1:1".into(), + trusted_proxies: Vec::new(), + server_id: "test.example".into(), + request_timeout: std::time::Duration::from_secs(30), + job_batch: 8, + job_poll_interval: std::time::Duration::from_secs(3600), + }); + let state = AppState { + db, + config: config.clone(), + limiter: Arc::new(RateLimiter::new()), + tmdb: Arc::new(TmdbClient::new(config.tmdb_base_url.clone(), None)), + }; + Self { router: app::router(state), _dir: dir } + } + + async fn send(&self, req: Request) -> (StatusCode, Value, axum::http::HeaderMap) { + let resp = self.router.clone().oneshot(req).await.expect("router call"); + let status = resp.status(); + let headers = resp.headers().clone(); + let bytes = resp.into_body().collect().await.expect("reading body").to_bytes(); + let body = if bytes.is_empty() { + Value::Null + } else { + serde_json::from_slice(&bytes) + .unwrap_or(Value::String(String::from_utf8_lossy(&bytes).into_owned())) + }; + (status, body, headers) + } + + async fn get(&self, uri: &str) -> (StatusCode, Value, axum::http::HeaderMap) { + self.send(Request::builder().uri(uri).body(Body::empty()).unwrap()).await + } + + async fn post_json( + &self, + uri: &str, + body: &Value, + ) -> (StatusCode, Value, axum::http::HeaderMap) { + self.send( + Request::builder() + .method("POST") + .uri(uri) + .header("content-type", "application/json") + .body(Body::from(body.to_string())) + .unwrap(), + ) + .await + } + + async fn post_json_auth( + &self, + uri: &str, + token: &str, + body: &Value, + ) -> (StatusCode, Value, axum::http::HeaderMap) { + self.send( + Request::builder() + .method("POST") + .uri(uri) + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(body.to_string())) + .unwrap(), + ) + .await + } + + /// Issues an anonymous bearer capability (§5a). + async fn token(&self) -> String { + let (status, body, _) = self.post_json("/api/v1/tokens", &json!({})).await; + assert_eq!(status, StatusCode::OK, "token issue failed: {body}"); + body["token"].as_str().expect("token in response").to_string() + } +} + +fn movie_manifest(tmdb_id: &str, runtime: f64) -> Value { + json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": tmdb_id, "title": "The Death of Stalin", + "year": 2017 }, + "cut": { "runtime_sec": runtime, "video_hash": "opensubtitles:8e245d9679d31e12" }, + "extraction": { "sample_fps": 5, "extinction_sec": 12, + "pipeline_version": "scene-actor-extraction 0.4.1", + "gallery_scope": "global" }, + "actors": [ + { "name": "Steve Buscemi", "tmdb_id": "884", "scenes": [[191.6, 209.2], [438.2, 465.6]] }, + { "name": "Michael Palin", "tmdb_id": "11007", "scenes": [[300.0, 320.0]] } + ] + }) +} + +// --------------------------------------------------------------------------- +// Health and readiness +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn health_is_unauthenticated() { + let s = TestServer::new("health"); + let (status, body, _) = s.get("/health").await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "ok"); +} + +#[tokio::test] +async fn readiness_reports_database_and_tmdb_configuration() { + // §8: TMDB is a hard dependency for UR-3, so its absence is worth surfacing. + let s = TestServer::new("ready"); + let (status, body, _) = s.get("/ready").await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "ready"); + assert_eq!(body["tmdb_configured"], false); +} + +// --------------------------------------------------------------------------- +// §5a — tokens +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn upload_without_a_token_is_rejected() { + let s = TestServer::new("noauth"); + let (status, _, _) = s.post_json("/api/v1/manifests", &movie_manifest("504172", 6420.5)).await; + assert_eq!(status, StatusCode::UNAUTHORIZED); +} + +#[tokio::test] +async fn upload_with_an_unknown_token_is_rejected() { + // A token the server never issued has no contributor row, and §5a stores only + // hashes, so there is nothing to match. + let s = TestServer::new("badauth"); + let (status, _, _) = s + .post_json_auth("/api/v1/manifests", "jray_deadbeef", &movie_manifest("504172", 6420.5)) + .await; + assert_eq!(status, StatusCode::UNAUTHORIZED); +} + +#[tokio::test] +async fn tokens_are_issued_anonymously_and_are_distinct() { + let s = TestServer::new("tokens"); + let a = s.token().await; + let b = s.token().await; + assert_ne!(a, b); + assert!(a.starts_with("jray_")); +} + +// --------------------------------------------------------------------------- +// §6 — upload validation +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn valid_upload_is_accepted_as_pending() { + // §6 stage 3: accepted with `202` and held unlisted until the cast check. + let s = TestServer::new("upload-ok"); + let token = s.token().await; + let (status, body, _) = + s.post_json_auth("/api/v1/manifests", &token, &movie_manifest("504172", 6420.5)).await; + assert_eq!(status, StatusCode::ACCEPTED, "body: {body}"); + assert_eq!(body["status"], "pending"); + assert!(body["manifest_id"].is_string()); +} + +#[tokio::test] +async fn a_pending_manifest_is_not_served() { + // The property that makes §6 stage 3 meaningful: an unverified manifest is + // not served to anyone in the meantime. + let s = TestServer::new("pending-hidden"); + let token = s.token().await; + let (_, body, _) = + s.post_json_auth("/api/v1/manifests", &token, &movie_manifest("504172", 6420.5)).await; + let id = body["manifest_id"].as_str().unwrap(); + + let (status, _, _) = s.get("/api/v1/manifests/movie?tmdb_id=504172").await; + assert_eq!(status, StatusCode::NOT_FOUND); + + let (status, _, _) = s.get(&format!("/api/v1/manifests/{id}")).await; + assert_eq!(status, StatusCode::NOT_FOUND); + + // But its status is pollable (§4). + let (status, body, _) = s.get(&format!("/api/v1/manifests/{id}/status")).await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "pending"); +} + +#[tokio::test] +async fn unknown_field_anywhere_is_rejected_with_400() { + // §6 stage 2, enforced by `deny_unknown_fields` on every DTO. + let s = TestServer::new("strict"); + let token = s.token().await; + + let mut m = movie_manifest("504172", 6420.5); + m["surprise"] = json!("payload"); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + assert!( + body["message"].as_str().unwrap_or("").contains("surprise"), + "the error should name the offending field: {body}" + ); + + // Nested, too — `extra="forbid"` applies at every level (§5a). + let mut m = movie_manifest("504172", 6420.5); + m["cut"]["extra"] = json!(1); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn contributor_local_identifiers_are_rejected_not_ignored() { + // §1/§6: `movie` leaks the contributor's directory layout and `jellyfin_id` is + // a GUID from their database. Both must be *rejected on upload*, so a client + // that forgets to strip them gets a hard 400 naming the field rather than + // quietly publishing them. + let s = TestServer::new("strip"); + let token = s.token().await; + + let mut m = movie_manifest("504172", 6420.5); + m["movie"] = json!("/data/movies/The.Death.of.Stalin.2017.mkv"); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + assert!(body["message"].as_str().unwrap_or("").contains("movie"), "{body}"); + + let mut m = movie_manifest("504172", 6420.5); + m["actors"][0]["jellyfin_id"] = json!("a1b2c3d4e5f6"); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + assert!(body["message"].as_str().unwrap_or("").contains("jellyfin_id"), "{body}"); +} + +#[tokio::test] +async fn the_withdrawn_anneal_sec_field_is_rejected() { + // `anneal_sec` was withdrawn in the SR-003 bump: presence now follows track + // extent, so a track survives its own gaps and there is nothing to anneal + // (`scene-actor-extraction` AR-012/AR-013). + // + // Rejecting rather than ignoring it is the point. A manifest still carrying + // the field was produced by a pipeline whose window semantics differ from + // what this server now assumes, and silently accepting it would store + // timings whose meaning we cannot vouch for. + let s = TestServer::new("anneal-withdrawn"); + let token = s.token().await; + + let mut m = movie_manifest("504172", 6420.5); + m["extraction"]["anneal_sec"] = json!(3); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + assert!( + body["message"].as_str().unwrap_or("").contains("anneal_sec"), + "the error should name the withdrawn field: {body}" + ); +} + +#[tokio::test] +async fn the_schema_bump_fields_round_trip() { + // `extinction_sec` and `gallery_scope` are the SR-003 additions. They are + // stored and reconstructed, since §7 ranks on scope and both are provenance + // a consumer may want. + let s = TestServer::new("bump-fields"); + let token = s.token().await; + + let m = movie_manifest("504172", 6420.5); + assert_eq!(m["extraction"]["gallery_scope"], "global"); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::ACCEPTED, "{body}"); + + // An unrecognised scope is a closed-vocabulary violation, not a free string. + let mut bad = movie_manifest("504173", 6420.5); + bad["extraction"]["gallery_scope"] = json!("enormous"); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &bad).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "gallery_scope is a closed enum"); +} + +#[tokio::test] +async fn an_unknown_manifest_version_is_rejected() { + // UR-014 / SR-003: a consumer encountering an unknown `schema_version` + // refuses or warns; it never guesses. + let s = TestServer::new("version"); + let token = s.token().await; + + for version in [0, 2, 99] { + let mut m = movie_manifest("504172", 6420.5); + m["jmanifest_version"] = json!(version); + let (status, body, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "version {version}: {body}"); + assert!( + body["message"].as_str().unwrap_or("").contains("jmanifest_version"), + "should name the field: {body}" + ); + } +} + +#[tokio::test] +async fn missing_runtime_is_rejected() { + // §2: `cut.runtime_sec` is required — the primary alignment guard. + let s = TestServer::new("no-runtime"); + let token = s.token().await; + let mut m = movie_manifest("504172", 6420.5); + m["cut"] = json!({ "video_hash": "opensubtitles:8e245d9679d31e12" }); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn empty_actor_list_is_rejected() { + // §6: 15 of the 331 corpus files have empty actor lists — extraction + // failures, not contributions. + let s = TestServer::new("empty-actors"); + let token = s.token().await; + let mut m = movie_manifest("504172", 6420.5); + m["actors"] = json!([]); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn smuggled_payload_in_an_actor_name_is_rejected() { + // §5a: the character class defeats base64/hex smuggling, which needs digits + // and padding characters. This is the last free-text channel, so it is worth + // asserting end-to-end and not only in the unit tests. + let s = TestServer::new("smuggle"); + let token = s.token().await; + for payload in [ + "SGVsbG8gd29ybGQgdGhpcyBpcyBhIHBheWxvYWQ=", + "4d5a90000300000004000000ffff0000", + "", + "http://evil.example/x", + ] { + let mut m = movie_manifest("504172", 6420.5); + m["actors"][0]["name"] = json!(payload); + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "payload {payload:?} should be rejected"); + } +} + +#[tokio::test] +async fn resubmitting_identical_content_is_not_a_duplicate_error() { + // §9a: content addressing gives deduplication — the same manifest from the + // same contributor is recognised rather than stored twice. + let s = TestServer::new("dedup"); + let token = s.token().await; + let m = movie_manifest("504172", 6420.5); + let (first, body1, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(first, StatusCode::ACCEPTED); + let (second, body2, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(second, StatusCode::OK); + assert_eq!(body2["status"], "already_present"); + assert_eq!(body1["manifest_id"], body2["manifest_id"]); +} + +#[tokio::test] +async fn oversized_body_is_rejected_by_the_route_cap() { + // §6 stage 1: the app-level cap counts bytes as they are read, so a lying + // `Content-Length` and a chunked upload are both safe. Here the body genuinely + // exceeds the 2 MiB single-manifest cap. + let s = TestServer::new("too-big"); + let token = s.token().await; + + let mut m = movie_manifest("504172", 6420.5); + // Many actors, each with many windows — legitimate shape, illegitimate size. + let actors: Vec = (0..400) + .map(|i| { + let scenes: Vec = (0..1500).map(|j| json!([j as f64, (j + 1) as f64])).collect(); + json!({ "tmdb_id": (1000 + i).to_string(), "scenes": scenes }) + }) + .collect(); + m["actors"] = json!(actors); + + let (status, _, _) = s.post_json_auth("/api/v1/manifests", &token, &m).await; + assert_eq!(status, StatusCode::PAYLOAD_TOO_LARGE); +} + +#[tokio::test] +async fn a_lying_content_length_does_not_bypass_the_cap() { + // §6 stage 0 is explicit that `Content-Length` is a *claim by the client*: a + // hostile client can declare 100 and send far more, so the streaming cap is + // mandatory rather than redundant. + let s = TestServer::new("lying-length"); + let token = s.token().await; + + let huge = "x".repeat(3 * 1024 * 1024); + let body = format!("{{\"padding\":\"{huge}\"}}"); + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .header("content-length", "100") + .body(Body::from(body)) + .unwrap(); + + let (status, _, _) = s.send(req).await; + assert_ne!( + status, + StatusCode::ACCEPTED, + "an oversized body must never be accepted, whatever the declared length" + ); + assert!( + status == StatusCode::PAYLOAD_TOO_LARGE || status == StatusCode::BAD_REQUEST, + "unexpected status {status}" + ); +} + +// --------------------------------------------------------------------------- +// §4 — exists +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn exists_returns_200_with_false_rather_than_404() { + // §4: absence is a normal answer, and `404` would conflate "no manifest" with + // "bad route" for the client. + let s = TestServer::new("exists-absent"); + let (status, body, _) = s.get("/api/v1/manifests/exists?tmdb_id=999999").await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["exists"], false); + assert!(body["manifest_id"].is_null()); +} + +#[tokio::test] +async fn exists_requires_identity_parameters() { + let s = TestServer::new("exists-noid"); + let (status, _, _) = s.get("/api/v1/manifests/exists").await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn exists_carries_rate_limit_headers() { + // §5: responses carry `X-RateLimit-Limit`, `-Remaining` and `-Reset`. + let s = TestServer::new("exists-headers"); + let (_, _, headers) = s.get("/api/v1/manifests/exists?tmdb_id=1").await; + assert_eq!(headers["x-ratelimit-limit"], "600"); + assert_eq!(headers["x-ratelimit-remaining"], "599"); + assert!(headers.contains_key("x-ratelimit-reset")); +} + +#[tokio::test] +async fn batch_exists_is_positional_and_capped_at_100() { + // §4: results are positional, and the cap is what lets §5 be generous per + // request while staying strict per item. + let s = TestServer::new("exists-batch"); + let items: Vec = (0..3).map(|i| json!({ "tmdb_id": (100 + i).to_string() })).collect(); + let (status, body, headers) = + s.post_json("/api/v1/manifests/exists", &json!({ "items": items })).await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["results"].as_array().unwrap().len(), 3); + assert_eq!(headers["x-ratelimit-limit"], "60", "batch has its own §5 budget"); + + let too_many: Vec = + (0..101).map(|i| json!({ "tmdb_id": (100 + i).to_string() })).collect(); + let (status, _, _) = + s.post_json("/api/v1/manifests/exists", &json!({ "items": too_many })).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn batch_exists_rejects_unknown_fields() { + let s = TestServer::new("exists-batch-strict"); + let (status, _, _) = + s.post_json("/api/v1/manifests/exists", &json!({ "items": [], "extra": 1 })).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn one_bad_item_does_not_fail_the_whole_batch() { + // A 100-item sweep should not be lost to one malformed entry. + let s = TestServer::new("exists-batch-partial"); + let (status, body, _) = s + .post_json( + "/api/v1/manifests/exists", + &json!({ "items": [ { "tmdb_id": "1" }, { }, { "tmdb_id": "2" } ] }), + ) + .await; + assert_eq!(status, StatusCode::OK); + let results = body["results"].as_array().unwrap(); + assert_eq!(results.len(), 3); + assert_eq!(results[1]["exists"], false); +} + +// --------------------------------------------------------------------------- +// §5 — rate limiting +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn exceeding_a_limit_returns_429_with_retry_after() { + // §5: exceeding a limit returns `429` with `Retry-After`, which the JRay + // client must honour. + let s = TestServer::new("ratelimit"); + let token = s.token().await; + + // The bundle surface has the tightest write limit (20/hour), so it is the + // cheapest to exhaust. + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396", "title": "Breaking Bad" }, + "episodes": [] + }); + + let mut saw_429 = false; + for _ in 0..25 { + let (status, _, headers) = + s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + if status == StatusCode::TOO_MANY_REQUESTS { + assert!(headers.contains_key("retry-after"), "429 must carry Retry-After"); + saw_429 = true; + break; + } + } + assert!(saw_429, "the §5 bundle limit should engage within 25 requests"); +} + +#[tokio::test] +async fn read_surfaces_have_independent_budgets() { + // §5: each surface has its own budget, so a library sweep hammering `exists` + // cannot exhaust the budget a fetch needs. + // + // Only the `exists` surface returns a body on an empty database; the fetch + // surfaces 404 (and a 404 carries no quota headers, by design). So the + // independence is asserted by consuming `exists` and observing that its + // counter alone moves. + let s = TestServer::new("surfaces"); + + let (_, _, h) = s.get("/api/v1/manifests/exists?tmdb_id=1").await; + assert_eq!(h["x-ratelimit-limit"], "600"); + assert_eq!(h["x-ratelimit-remaining"], "599"); + + // A fetch and a series request in between must not consume `exists` budget. + let _ = s.get("/api/v1/manifests/movie?tmdb_id=1").await; + let _ = s.get("/api/v1/manifests/series/1396").await; + + let (_, _, h) = s.get("/api/v1/manifests/exists?tmdb_id=1").await; + assert_eq!( + h["x-ratelimit-remaining"], "598", + "fetch requests must not draw down the exists budget" + ); + + // And the batch form is a separate surface again (§5). + let (_, _, h) = + s.post_json("/api/v1/manifests/exists", &json!({ "items": [ { "tmdb_id": "1" } ] })).await; + assert_eq!(h["x-ratelimit-limit"], "60"); + assert_eq!(h["x-ratelimit-remaining"], "59"); +} + +#[tokio::test] +async fn a_forged_forwarded_header_cannot_reset_a_budget() { + // §8: the app must trust `X-Forwarded-For` only from the operator's proxy, + // because §5 rate limiting keys on client IP. This server has no configured + // proxies, so the header must be ignored entirely — otherwise a client could + // mint a fresh budget per request. + let s = TestServer::new("xff"); + + let mut last_remaining = u32::MAX; + for i in 0..3 { + let req = Request::builder() + .uri("/api/v1/manifests/exists?tmdb_id=1") + .header("x-forwarded-for", format!("10.1.1.{i}")) + .body(Body::empty()) + .unwrap(); + let (_, _, headers) = s.send(req).await; + let remaining: u32 = headers["x-ratelimit-remaining"].to_str().unwrap().parse().unwrap(); + assert!( + remaining < last_remaining, + "budget must keep decreasing despite a changing X-Forwarded-For" + ); + last_remaining = remaining; + } +} + +// --------------------------------------------------------------------------- +// §4 — fetch +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn fetch_requires_identity_parameters() { + let s = TestServer::new("fetch-noid"); + let (status, _, _) = s.get("/api/v1/manifests/movie").await; + assert_eq!(status, StatusCode::BAD_REQUEST); + + let (status, _, _) = s.get("/api/v1/manifests/episode?series_tmdb_id=1396").await; + assert_eq!(status, StatusCode::BAD_REQUEST, "episode fetch needs season and episode"); +} + +#[tokio::test] +async fn fetching_an_absent_manifest_is_404() { + // §4: `404` if none clears `loose`. + let s = TestServer::new("fetch-absent"); + let (status, _, _) = s.get("/api/v1/manifests/movie?tmdb_id=999999").await; + assert_eq!(status, StatusCode::NOT_FOUND); +} + +#[tokio::test] +async fn series_bundle_for_an_unknown_series_is_404() { + let s = TestServer::new("series-absent"); + let (status, _, _) = s.get("/api/v1/manifests/series/999999").await; + assert_eq!(status, StatusCode::NOT_FOUND); +} + +#[tokio::test] +async fn status_of_an_unknown_manifest_reports_rejected() { + // §6 deletes rejected manifests, so a vanished id must not read as a bad + // route — the contributor polling it needs a verdict. + let s = TestServer::new("status-unknown"); + let (status, body, _) = s.get("/api/v1/manifests/01HZZZZZZZZZZZZZZZZZZZZZZZ/status").await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "rejected"); +} + +// --------------------------------------------------------------------------- +// §2, §4 — bundles +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn bundle_upload_is_not_atomic() { + // §2: valid episodes are accepted and invalid ones rejected, with a + // per-episode result list. All-or-nothing would let one bad episode discard an + // entire season's compute. + let s = TestServer::new("bundle-partial"); + let token = s.token().await; + + let good = |ep: i64| { + json!({ + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "1396", "title": "Breaking Bad", + "season": 1, "episode": ep }, + "cut": { "runtime_sec": 2820.0 }, + "actors": [ { "name": "Bryan Cranston", "tmdb_id": "17419", + "scenes": [[10.0, 20.0]] } ] + }) + }; + // Invalid: a scene window beyond the runtime tolerance (§6). + let bad = json!({ + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "1396", "season": 1, "episode": 3 }, + "cut": { "runtime_sec": 2820.0 }, + "actors": [ { "tmdb_id": "17419", "scenes": [[10.0, 99999.0]] } ] + }); + + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396", "title": "Breaking Bad" }, + "episodes": [ good(1), bad, good(2) ] + }); + + let (status, body, _) = s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + assert_eq!(status, StatusCode::ACCEPTED, "{body}"); + let results = body["results"].as_array().unwrap(); + assert_eq!(results.len(), 3); + assert_eq!(results[0]["status"], "pending"); + assert_eq!(results[1]["status"], "rejected"); + assert!(results[1]["reason"].is_string(), "a rejected episode should say why"); + assert_eq!(results[2]["status"], "pending", "a later episode must still be accepted"); +} + +#[tokio::test] +async fn bundle_envelope_errors_are_whole_request_400s() { + // §4: `400` for the envelope itself, whereas individual bad episodes are + // reported in the results list. + let s = TestServer::new("bundle-envelope"); + let token = s.token().await; + + let (status, _, _) = s + .post_json_auth( + "/api/v1/manifests/bundle", + &token, + &json!({ "jmanifest_version": 1, "series": {}, "episodes": [] }), + ) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST, "series needs an identifier"); + + let (status, _, _) = s + .post_json_auth( + "/api/v1/manifests/bundle", + &token, + &json!({ "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396" }, + "episodes": [], "extra": 1 }), + ) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST, "unknown envelope field"); +} + +#[tokio::test] +async fn bundle_rejects_an_episode_contradicting_the_envelope() { + // An episode must not be silently reattributed to the bundle's series. + let s = TestServer::new("bundle-mismatch"); + let token = s.token().await; + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396" }, + "episodes": [ { + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "9999", "season": 1, "episode": 1 }, + "cut": { "runtime_sec": 2820.0 }, + "actors": [ { "tmdb_id": "17419", "scenes": [[1.0, 2.0]] } ] + } ] + }); + let (status, body, _) = s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + assert_eq!(status, StatusCode::ACCEPTED); + assert_eq!(body["results"][0]["status"], "rejected"); +} + +#[tokio::test] +async fn bundle_beyond_the_episode_cap_is_413() { + // §2/§4: capped at 500 episodes; beyond that the client must page by season. + let s = TestServer::new("bundle-cap"); + let token = s.token().await; + let episodes: Vec = (0..501) + .map(|i| { + json!({ + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "1396", + "season": 1, "episode": i }, + "cut": { "runtime_sec": 2820.0 }, + "actors": [ { "tmdb_id": "17419", "scenes": [[1.0, 2.0]] } ] + }) + }) + .collect(); + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396" }, + "episodes": episodes + }); + let (status, _, _) = s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + assert_eq!(status, StatusCode::PAYLOAD_TOO_LARGE); +} + +#[tokio::test] +async fn bundle_route_accepts_a_body_larger_than_the_single_manifest_cap() { + // §6 stage 1: per-route limits, so the bundle endpoint gets its larger cap + // without widening the others. A ~3 MiB bundle exceeds the 2 MiB manifest cap + // but is well within the 25 MiB bundle cap. + let s = TestServer::new("bundle-bigger-cap"); + let token = s.token().await; + + let episodes: Vec = (1..=60) + .map(|ep| { + let scenes: Vec = (0..600).map(|j| json!([j as f64, (j + 1) as f64])).collect(); + json!({ + "jmanifest_version": 1, + "identity": { "type": "episode", "series_tmdb_id": "1396", + "season": 1, "episode": ep }, + "cut": { "runtime_sec": 2820.0 }, + "actors": (0..8).map(|a| json!({ + "tmdb_id": (20000 + a).to_string(), "scenes": scenes + })).collect::>() + }) + }) + .collect(); + let bundle = json!({ + "jmanifest_version": 1, + "series": { "series_tmdb_id": "1396" }, + "episodes": episodes + }); + let encoded = bundle.to_string(); + assert!( + encoded.len() > 2 * 1024 * 1024, + "test body should exceed the single-manifest cap, got {} bytes", + encoded.len() + ); + + let (status, _, _) = s.post_json_auth("/api/v1/manifests/bundle", &token, &bundle).await; + assert_eq!(status, StatusCode::ACCEPTED, "the bundle route has its own larger cap"); +} + +// --------------------------------------------------------------------------- +// §4 — reports +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn reporting_an_unknown_manifest_is_404() { + let s = TestServer::new("report-unknown"); + let (status, _, _) = s + .post_json( + "/api/v1/manifests/01HZZZZZZZZZZZZZZZZZZZZZZZ/report", + &json!({ "reason": "misaligned" }), + ) + .await; + assert_eq!(status, StatusCode::NOT_FOUND); +} + +#[tokio::test] +async fn a_report_is_accepted_and_does_not_delist() { + // §5a: delisting stays an operator action. Automatic delisting on report would + // hand any client a remote delete primitive. + let s = TestServer::new("report-ok"); + let token = s.token().await; + let (_, body, _) = + s.post_json_auth("/api/v1/manifests", &token, &movie_manifest("504172", 6420.5)).await; + let id = body["manifest_id"].as_str().unwrap().to_string(); + + let (status, body, _) = s + .post_json( + &format!("/api/v1/manifests/{id}/report"), + &json!({ "reason": "wrong_actors", "note": "these are not the right people" }), + ) + .await; + assert_eq!(status, StatusCode::OK, "{body}"); + assert!(body["report_id"].is_string()); + + let (status, body, _) = s.get(&format!("/api/v1/manifests/{id}/status")).await; + assert_eq!(status, StatusCode::OK); + assert_eq!(body["status"], "pending", "a report must not change status by itself"); +} + +#[tokio::test] +async fn report_rejects_an_unknown_reason_and_unknown_fields() { + let s = TestServer::new("report-strict"); + let (status, _, _) = s + .post_json("/api/v1/manifests/x/report", &json!({ "reason": "i_just_dont_like_it" })) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST); + + let (status, _, _) = s + .post_json("/api/v1/manifests/x/report", &json!({ "reason": "spam", "extra": true })) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn report_note_is_length_capped() { + // §5a: free text from an anonymous caller is capped hard. + let s = TestServer::new("report-note"); + let token = s.token().await; + let (_, body, _) = + s.post_json_auth("/api/v1/manifests", &token, &movie_manifest("504172", 6420.5)).await; + let id = body["manifest_id"].as_str().unwrap().to_string(); + + let (status, _, _) = s + .post_json( + &format!("/api/v1/manifests/{id}/report"), + &json!({ "reason": "spam", "note": "a".repeat(5000) }), + ) + .await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +// --------------------------------------------------------------------------- +// Malformed input +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn malformed_json_is_a_400_not_a_500() { + let s = TestServer::new("bad-json"); + let token = s.token().await; + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from("{ this is not json")) + .unwrap(); + let (status, _, _) = s.send(req).await; + assert_eq!(status, StatusCode::BAD_REQUEST); +} + +#[tokio::test] +async fn deeply_nested_json_does_not_crash_the_parser() { + // §6 stage 1 caps nesting depth; a parser handed unbounded input is a DoS + // primitive, so the failure must be a clean rejection. + let s = TestServer::new("deep-json"); + let token = s.token().await; + let deep = format!("{}{}", "[".repeat(5000), "]".repeat(5000)); + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(deep)) + .unwrap(); + let (status, _, _) = s.send(req).await; + assert!( + status == StatusCode::BAD_REQUEST || status == StatusCode::PAYLOAD_TOO_LARGE, + "deeply nested input should be rejected cleanly, got {status}" + ); +} + +#[tokio::test] +async fn unknown_routes_are_404() { + let s = TestServer::new("routes"); + let (status, _, _) = s.get("/api/v1/nonexistent").await; + assert_eq!(status, StatusCode::NOT_FOUND); + let (status, _, _) = s.get("/api/v2/manifests/exists?tmdb_id=1").await; + assert_eq!(status, StatusCode::NOT_FOUND); +} diff --git a/tests/injection.rs b/tests/injection.rs new file mode 100644 index 0000000..fc248ad --- /dev/null +++ b/tests/injection.rs @@ -0,0 +1,634 @@ +//! Injection resistance — SQL, JSON and header. +//! +//! These are regression tests for properties the design already provides, kept +//! separate from `api.rs` because their purpose is different: `api.rs` asserts the +//! spec's behaviour, this asserts that hostile input cannot escape its layer. +//! +//! Two distinct defences are at work, and it is worth being precise about which +//! applies where, because they fail differently: +//! +//! 1. **Parameterised queries** (§7, §8). Every value reaches SQLite through +//! `params![]`; the only `format!`-built SQL interpolates compile-time +//! constants (a column list and a status literal). So a value carrying SQL +//! syntax is bound as *data* and simply matches nothing. +//! 2. **Closed-vocabulary validation** (§5a, §6 stage 2). Identifiers are +//! regex-constrained and free text is restricted to a closed character class, +//! so most injection strings are rejected before they reach the database. +//! +//! Defence 1 is what actually prevents injection; defence 2 means an attacker +//! usually cannot even reach it. Testing both matters: if validation were ever +//! loosened, these tests should still pass on the strength of parameterisation +//! alone. + +use std::sync::Arc; + +use axum::body::Body; +use axum::http::{Request, StatusCode}; +use http_body_util::BodyExt; +use jray_server::app; +use jray_server::config::Config; +use jray_server::db::Db; +use jray_server::ratelimit::RateLimiter; +use jray_server::state::AppState; +use jray_server::tmdb::TmdbClient; +use serde_json::{json, Value}; +use tower::ServiceExt; + +/// Payloads spanning the usual SQL-injection shapes: boolean tautology, statement +/// termination, stacked statements, UNION exfiltration, comment truncation, and +/// string-concatenation exfiltration. +const SQL_PAYLOADS: &[&str] = &[ + "1' OR '1'='1", + "1'; DROP TABLE manifests;--", + "1 UNION SELECT token_hash FROM contributors", + "' OR 1=1--", + "1'||(SELECT token_hash FROM contributors)||'", + "1)) OR 1=1 --", + "'; UPDATE manifests SET status='listed' WHERE 1=1;--", + "1/**/UNION/**/SELECT/**/1", + "x' AND (SELECT COUNT(*) FROM sqlite_master)>0 --", + "\"; DELETE FROM scenes; --", +]; + +struct TestServer { + router: axum::Router, + db: Db, + _dir: TempDir, +} + +struct TempDir(std::path::PathBuf); + +impl TempDir { + fn new(tag: &str) -> Self { + let mut p = std::env::temp_dir(); + p.push(format!("jray-inj-{}-{}", tag, unique())); + std::fs::create_dir_all(&p).expect("creating temp dir"); + Self(p) + } + fn db_path(&self) -> String { + self.0.join("test.db").to_string_lossy().into_owned() + } +} + +impl Drop for TempDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } +} + +fn unique() -> String { + use std::sync::atomic::{AtomicU64, Ordering}; + static N: AtomicU64 = AtomicU64::new(0); + let t = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_nanos()) + .unwrap_or(0); + format!("{t}-{}", N.fetch_add(1, Ordering::Relaxed)) +} + +impl TestServer { + fn new(tag: &str) -> Self { + let dir = TempDir::new(tag); + let db = Db::open(&dir.db_path()).expect("opening database"); + let config = Arc::new(Config { + bind: "127.0.0.1:0".into(), + db_path: dir.db_path(), + tmdb_api_key: None, + tmdb_base_url: "http://127.0.0.1:1".into(), + trusted_proxies: Vec::new(), + server_id: "test.example".into(), + request_timeout: std::time::Duration::from_secs(30), + job_batch: 8, + job_poll_interval: std::time::Duration::from_secs(3600), + }); + let state = AppState { + db: db.clone(), + config: config.clone(), + limiter: Arc::new(RateLimiter::new()), + tmdb: Arc::new(TmdbClient::new(config.tmdb_base_url.clone(), None)), + }; + Self { router: app::router(state), db, _dir: dir } + } + + async fn send(&self, req: Request) -> (StatusCode, Value) { + let resp = self.router.clone().oneshot(req).await.expect("router call"); + let status = resp.status(); + let bytes = resp.into_body().collect().await.expect("body").to_bytes(); + let body = if bytes.is_empty() { + Value::Null + } else { + serde_json::from_slice(&bytes) + .unwrap_or(Value::String(String::from_utf8_lossy(&bytes).into_owned())) + }; + (status, body) + } + + async fn get(&self, uri: &str) -> (StatusCode, Value) { + self.send(Request::builder().uri(uri).body(Body::empty()).unwrap()).await + } + + async fn post(&self, uri: &str, token: Option<&str>, body: &Value) -> (StatusCode, Value) { + let mut b = + Request::builder().method("POST").uri(uri).header("content-type", "application/json"); + if let Some(t) = token { + b = b.header("authorization", format!("Bearer {t}")); + } + self.send(b.body(Body::from(body.to_string())).unwrap()).await + } + + async fn token(&self) -> String { + let (_, body) = self.post("/api/v1/tokens", None, &json!({})).await; + body["token"].as_str().expect("token").to_string() + } + + /// Confirms the schema is intact and the expected row counts hold. + /// + /// A successful injection would most likely drop a table or delete rows, so + /// this is the assertion that actually matters after each payload. + async fn assert_schema_intact(&self) { + let tables: Vec = self + .db + .read(|conn| { + let mut stmt = conn + .prepare("SELECT name FROM sqlite_master WHERE type='table' ORDER BY name")?; + let rows = stmt + .query_map([], |r| r.get::<_, String>(0))? + .collect::>>()?; + Ok(rows) + }) + .await + .expect("listing tables"); + + for expected in [ + "contributors", + "jobs", + "manifest_actors", + "manifests", + "people", + "reports", + "scenes", + "titles", + "tmdb_cache", + ] { + assert!( + tables.iter().any(|t| t == expected), + "table {expected} is missing — an injection may have dropped it. tables: {tables:?}" + ); + } + } +} + +fn urlencode(s: &str) -> String { + let mut out = String::new(); + for b in s.bytes() { + match b { + b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => { + out.push(b as char) + } + _ => out.push_str(&format!("%{b:02X}")), + } + } + out +} + +// --------------------------------------------------------------------------- +// SQL injection — query parameters +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn sql_payloads_in_query_parameters_are_inert() { + let s = TestServer::new("query"); + + for payload in SQL_PAYLOADS { + let enc = urlencode(payload); + for uri in [ + format!("/api/v1/manifests/exists?tmdb_id={enc}"), + format!("/api/v1/manifests/exists?imdb_id={enc}"), + format!("/api/v1/manifests/movie?tmdb_id={enc}"), + format!("/api/v1/manifests/movie?imdb_id={enc}"), + format!("/api/v1/manifests/episode?series_tmdb_id={enc}&season=1&episode=1"), + format!("/api/v1/manifests/series/{enc}"), + format!("/api/v1/manifests/exists?tmdb_id=1&video_hash={enc}"), + ] { + let (status, body) = s.get(&uri).await; + // The payload is bound as data, so it matches nothing. What must never + // happen is a 5xx, which would mean SQLite saw it as syntax. + assert!( + status.is_success() + || status == StatusCode::NOT_FOUND + || status == StatusCode::BAD_REQUEST, + "payload {payload:?} on {uri} produced {status} — expected data-not-found, \ + not a server error. body: {body}" + ); + } + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn sql_payloads_in_path_parameters_are_inert() { + let s = TestServer::new("path"); + + for payload in SQL_PAYLOADS { + let enc = urlencode(payload); + for uri in [ + format!("/api/v1/manifests/{enc}"), + format!("/api/v1/manifests/{enc}/status"), + format!("/api/v1/manifests/series/{enc}"), + ] { + let (status, body) = s.get(&uri).await; + assert!( + !status.is_server_error(), + "payload {payload:?} on {uri} produced {status}: {body}" + ); + } + } + + s.assert_schema_intact().await; +} + +// --------------------------------------------------------------------------- +// SQL injection — JSON body fields +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn sql_payloads_in_identifier_fields_are_rejected() { + // §6 stage 2 regex-constrains every identifier, so these never even reach the + // query layer. The response must be a clean 400 naming the field. + let s = TestServer::new("body-ids"); + let token = s.token().await; + + for payload in SQL_PAYLOADS { + let manifest = json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": payload }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }); + let (status, body) = s.post("/api/v1/manifests", Some(&token), &manifest).await; + assert_eq!( + status, + StatusCode::BAD_REQUEST, + "payload {payload:?} should be rejected by validation: {body}" + ); + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn sql_payloads_in_free_text_fields_are_rejected() { + // The two free-text fields (§5a) are the only place arbitrary strings could + // arrive. The closed character class excludes quotes, semicolons and digits, + // which is what makes SQL syntax unrepresentable there. + let s = TestServer::new("body-text"); + let token = s.token().await; + + for payload in SQL_PAYLOADS { + for manifest in [ + json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172", "title": payload }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "name": payload, "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + ] { + let (status, body) = s.post("/api/v1/manifests", Some(&token), &manifest).await; + assert_eq!( + status, + StatusCode::BAD_REQUEST, + "free-text payload {payload:?} should be rejected: {body}" + ); + } + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn sql_payloads_in_a_report_note_cannot_escape() { + // `note` is the one field that accepts relatively free text (control + // characters stripped, length capped) because only the operator reads it. It + // reaches the database, so it is the strongest test of parameterisation: + // validation is *not* filtering SQL syntax here. + let s = TestServer::new("report-note"); + let token = s.token().await; + + let (_, body) = s + .post( + "/api/v1/manifests", + Some(&token), + &json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + ) + .await; + let id = body["manifest_id"].as_str().expect("manifest id").to_string(); + + for payload in SQL_PAYLOADS { + let (status, body) = s + .post( + &format!("/api/v1/manifests/{id}/report"), + None, + &json!({ "reason": "spam", "note": payload }), + ) + .await; + assert!( + status.is_success() || status == StatusCode::TOO_MANY_REQUESTS, + "note payload {payload:?} produced {status}: {body}" + ); + if status.is_success() { + s.assert_schema_intact().await; + } + } + + // The notes were stored verbatim as *data* — proving they were bound, not + // executed. Verified by reading them back out. + let stored: i64 = + s.db.read(|conn| Ok(conn.query_row("SELECT COUNT(*) FROM reports", [], |r| r.get(0))?)) + .await + .expect("counting reports"); + assert!(stored > 0, "reports should have been stored as inert data"); +} + +#[tokio::test] +async fn sql_payloads_in_a_bearer_token_are_inert() { + // The token is hashed before it reaches any query, but a payload arriving via + // a header must still not produce a 5xx. + let s = TestServer::new("token-inj"); + + for payload in SQL_PAYLOADS { + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {payload}")) + .body(Body::from( + json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }) + .to_string(), + )) + .unwrap(); + let (status, body) = s.send(req).await; + assert_eq!( + status, + StatusCode::UNAUTHORIZED, + "token payload {payload:?} produced {status}: {body}" + ); + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn sql_payloads_in_the_batch_exists_body_are_inert() { + let s = TestServer::new("batch-inj"); + let items: Vec = SQL_PAYLOADS.iter().map(|p| json!({ "tmdb_id": p })).collect(); + let (status, body) = s.post("/api/v1/manifests/exists", None, &json!({ "items": items })).await; + assert_eq!(status, StatusCode::OK, "{body}"); + // Each malformed item degrades to "absent" rather than erroring the batch. + for result in body["results"].as_array().expect("results") { + assert_eq!(result["exists"], false); + } + s.assert_schema_intact().await; +} + +// --------------------------------------------------------------------------- +// JSON injection / parser abuse +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn json_structure_abuse_is_rejected_cleanly() { + // A parser handed hostile structure must fail with 400/413, never 5xx and + // never a hang (§6 stage 1: "a parser handed an unbounded body is a + // denial-of-service primitive"). + let s = TestServer::new("json-abuse"); + let token = s.token().await; + + let cases: Vec<(&str, String)> = vec![ + ("deep nesting", format!("{}{}", "[".repeat(20_000), "]".repeat(20_000))), + ("unterminated", "{\"identity\": {\"type\": \"movie\"".to_string()), + ("duplicate keys", r#"{"jmanifest_version":1,"jmanifest_version":2}"#.to_string()), + ("null bytes", "{\"jmanifest_version\":\u{0}1}".to_string()), + ("huge number", format!("{{\"jmanifest_version\":{}}}", "9".repeat(5000))), + ("nan literal", r#"{"jmanifest_version":1,"cut":{"runtime_sec":NaN}}"#.to_string()), + ("bare array", "[1,2,3]".to_string()), + ("bare string", "\"just a string\"".to_string()), + ("empty body", String::new()), + ( + "prototype-style key", + r#"{"__proto__":{"admin":true},"jmanifest_version":1}"#.to_string(), + ), + ]; + + for (label, body) in cases { + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(body)) + .unwrap(); + let (status, resp) = s.send(req).await; + assert!( + status == StatusCode::BAD_REQUEST || status == StatusCode::PAYLOAD_TOO_LARGE, + "{label} produced {status}, expected a clean rejection: {resp}" + ); + } + + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn non_finite_scene_times_are_rejected() { + // §6 explicitly rejects NaN/Infinity. They cannot arrive as JSON literals, but + // they can arrive as overflowing decimals, which parse to f64 infinity. + let s = TestServer::new("nonfinite"); + let token = s.token().await; + + // Sent as raw JSON text rather than via `json!`, because rustc refuses an + // out-of-range float literal — and the point is to make the *server's* parser + // handle it, which is the real attack path. + let raw = r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172"}, + "cut":{"runtime_sec":100.0}, + "actors":[{"tmdb_id":"884","scenes":[[1.0,1e400]]}]}"#; + + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(raw)) + .unwrap(); + let (status, body) = s.send(req).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "{body}"); + + // Likewise an overflowing runtime. + let raw = r#"{"jmanifest_version":1, + "identity":{"type":"movie","tmdb_id":"504172"}, + "cut":{"runtime_sec":1e400}, + "actors":[{"tmdb_id":"884","scenes":[[1.0,2.0]]}]}"#; + let req = Request::builder() + .method("POST") + .uri("/api/v1/manifests") + .header("content-type", "application/json") + .header("authorization", format!("Bearer {token}")) + .body(Body::from(raw)) + .unwrap(); + let (status, body) = s.send(req).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "{body}"); +} + +// --------------------------------------------------------------------------- +// Header injection +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn crlf_in_a_header_value_cannot_split_the_response() { + // A CRLF-carrying header value must not appear in the response as new headers. + // `http` rejects such values at construction, so this asserts the invariant + // holds at the boundary rather than relying on our own escaping. + let bad = "1.2.3.4\r\nX-Injected: yes"; + assert!( + axum::http::HeaderValue::from_str(bad).is_err(), + "the http crate must refuse CRLF in header values" + ); + + // And a percent-encoded variant reaching a handler stays inert data. + let s = TestServer::new("crlf"); + let (status, _) = s.get("/api/v1/manifests/exists?tmdb_id=1%0D%0AX-Injected:%20yes").await; + assert!(!status.is_server_error()); + s.assert_schema_intact().await; +} + +#[tokio::test] +async fn oversized_headers_do_not_take_the_server_down() { + let s = TestServer::new("big-header"); + let big = "a".repeat(100_000); + let req = Request::builder() + .uri("/api/v1/manifests/exists?tmdb_id=1") + .header("x-filler", big) + .body(Body::empty()) + .unwrap(); + let (status, _) = s.send(req).await; + assert!(!status.is_server_error(), "got {status}"); +} + +// --------------------------------------------------------------------------- +// Path traversal +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn path_traversal_attempts_reach_no_filesystem() { + // The server serves no files at all, so traversal has nowhere to go. Asserted + // anyway, because the manifest id is a path segment. + let s = TestServer::new("traversal"); + for probe in [ + "..%2F..%2F..%2Fetc%2Fpasswd", + "....%2F%2F....%2F%2Fetc%2Fpasswd", + "%2e%2e%2f%2e%2e%2fetc%2fshadow", + "..%5C..%5Cwindows%5Csystem32", + "%00/etc/passwd", + ] { + let (status, body) = s.get(&format!("/api/v1/manifests/{probe}")).await; + assert!( + status == StatusCode::NOT_FOUND || status == StatusCode::BAD_REQUEST, + "probe {probe} produced {status}: {body}" + ); + // Nothing that looks like file content should ever come back. + let text = body.to_string(); + assert!(!text.contains("root:"), "probe {probe} returned passwd-like content"); + } +} + +// --------------------------------------------------------------------------- +// Unicode and encoding tricks against the §5a character class +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn unicode_tricks_cannot_smuggle_text_past_the_character_class() { + // §5a's class is checked *after* NFC normalisation, so decomposed and + // compatibility forms must not provide a way in. Fullwidth digits are the + // sharpest case: NFKC would fold them to ASCII digits, but NFC does not, and + // they are `Nd` (not a letter), so the class rejects them either way. + let s = TestServer::new("unicode"); + let token = s.token().await; + + for payload in [ + "Actor 123", // fullwidth letters and digits + "Steve\u{FEFF}Buscemi", // zero-width no-break space + "Ste\u{0301}ve\u{202E}", // combining acute plus bidi override + "𝐒𝐭𝐞𝐯𝐞", // mathematical bold (compatibility form) + "Steve\u{2028}Buscemi", // line separator + "\u{1F600} Actor", // emoji + "Actor\u{00A0}Name\u{0000}", // nbsp plus NUL + ] { + let (status, body) = s + .post( + "/api/v1/manifests", + Some(&token), + &json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 100.0 }, + "actors": [ { "name": payload, "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + ) + .await; + assert_eq!( + status, + StatusCode::BAD_REQUEST, + "unicode payload {payload:?} should be rejected: {body}" + ); + } +} + +#[tokio::test] +async fn the_audio_signature_field_cannot_carry_arbitrary_bytes() { + // §3: a variable-length blob would be a payload channel — "precisely what §5a + // closes". Length is fixed and every byte is structurally constrained. + let s = TestServer::new("audio-sig"); + let token = s.token().await; + + for sig in [ + "v1:aGVsbG8gd29ybGQ=", // too short to be a signature + &format!("v1:{}", "/".repeat(4000)), // high bit set throughout + &format!("v1:{}", "A".repeat(100_000)), // oversized + "not-base64-at-all", // missing version prefix + &"A".repeat(1720), // unprefixed + ] { + let (status, body) = s + .post( + "/api/v1/manifests", + Some(&token), + &json!({ + "jmanifest_version": 1, + "identity": { "type": "movie", "tmdb_id": "504172" }, + "cut": { "runtime_sec": 6420.5, "audio_signature": sig }, + "actors": [ { "tmdb_id": "884", "scenes": [[1.0, 2.0]] } ] + }), + ) + .await; + assert_eq!( + status, + StatusCode::BAD_REQUEST, + "signature {:?} should be rejected: {body}", + &sig.chars().take(40).collect::() + ); + } +}