Add the v1 audio signature to the pipeline (IR-004, IR-005, IR-007, IR-008)

Implements the content-derived spectral-peak signature from
JRay-public-server/SPEC.md §3 so a truth file is self-identifying: 120 s
window centred on the media midpoint, mono at 11025 Hz, 4096/1024 Hann
STFT, 32 log-spaced bins over 300-3000 Hz, one byte per frame (5-bit peak
band + 2-bit energy class), base64, `v1:` prefix.

Audio decode is a second stream from the FFmpeg libraries the pipeline
already links for video; libswresample is added to the existing
ffmpeg_libs interface target. The FFT is written out rather than pulled
from a library for the same reason the plugin vendors one: the output has
to be bit-identical across two languages, so a dependency whose version
could change the numerics is a liability.

The server spec fixes the geometry but not enough to reproduce a byte
stream — Hann periodicity, band aggregation, the energy-class definition,
tie-breaking and the base64 alphabet are all unconstrained by it. Those
are pinned in audio_signature.hpp and mirrored in the golden fixture, so
the plugin can be implemented from the fixture alone.

IR-005: tests/fixtures/audio/ carries a deterministic 120 s tone (FLAC —
lossless, so identical PCM to the WAV make_fixture.py emits, and 3.5x
smaller in git) plus the signature it must produce, the decoded-PCM
checksum and the full parameter contract. That directory is the artefact
shared with the plugin repo; the PCM checksum is separate from the
signature so a codec-level difference is distinguishable from a DSP one.

IR-007: media under 120 s emits no signature. Same for a file with no
audio stream or one that will not open — UR-9 is an enhancement and must
never be able to break a fetch.

Verified against an independent Python reference implementation: same
bytes. All 32 bands and all 4 energy classes appear in the golden vector,
and the window-centring test wraps the fixture in 90 s of silence either
side and requires the golden value back.

Not wired into the truth-file output yet — that is the schema_version
bump under IR-002/IR-003 and is deliberately out of scope here.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-30 18:25:37 +02:00
co-authored by Claude Opus 5
parent 43d2c976c3
commit 45ef7c1916
9 changed files with 1283 additions and 9 deletions
+4
View File
@@ -19,6 +19,10 @@ compile_commands.json
# coverage). Regenerate with scripts/docs/run_holdout_all_models.py and # coverage). Regenerate with scripts/docs/run_holdout_all_models.py and
# scripts/docs/gallery_coverage_per_film.py. # scripts/docs/gallery_coverage_per_film.py.
!docs_data/*.json !docs_data/*.json
# Exception: test fixtures are inputs, not build output. The audio golden
# vector (IR-005) is shared verbatim with the jRay plugin repo, so it has to be
# tracked. Regenerate the media with tests/fixtures/audio/make_fixture.py.
!tests/fixtures/**
# Video files # Video files
*.mp4 *.mp4
*.mkv *.mkv
+13 -4
View File
@@ -196,23 +196,31 @@ endif()
# FFmpeg (hwaccel video decode: CUDA/VAAPI, runtime-detected + swscale colour # FFmpeg (hwaccel video decode: CUDA/VAAPI, runtime-detected + swscale colour
# conversion). Hwaccel support is built into libavcodec/libavutil; no extra # conversion). Hwaccel support is built into libavcodec/libavutil; no extra
# libraries are needed here. # libraries are needed here.
# libswresample is the audio side of the same dependency — downmix + resample
# for the audio signature (IR-004, src/audio_signature.cpp). Not a new project
# dependency: it ships with the libav* set already required above.
find_package(PkgConfig REQUIRED) find_package(PkgConfig REQUIRED)
pkg_check_modules(AVFORMAT REQUIRED libavformat) pkg_check_modules(AVFORMAT REQUIRED libavformat)
pkg_check_modules(AVCODEC REQUIRED libavcodec) pkg_check_modules(AVCODEC REQUIRED libavcodec)
pkg_check_modules(AVUTIL REQUIRED libavutil) pkg_check_modules(AVUTIL REQUIRED libavutil)
pkg_check_modules(SWSCALE REQUIRED libswscale) pkg_check_modules(SWSCALE REQUIRED libswscale)
pkg_check_modules(SWRESAMPLE REQUIRED libswresample)
add_library(ffmpeg_libs INTERFACE) add_library(ffmpeg_libs INTERFACE)
target_compile_options(ffmpeg_libs INTERFACE target_compile_options(ffmpeg_libs INTERFACE
${AVFORMAT_CFLAGS_OTHER} ${AVCODEC_CFLAGS_OTHER} ${AVFORMAT_CFLAGS_OTHER} ${AVCODEC_CFLAGS_OTHER}
${AVUTIL_CFLAGS_OTHER} ${SWSCALE_CFLAGS_OTHER}) ${AVUTIL_CFLAGS_OTHER} ${SWSCALE_CFLAGS_OTHER}
${SWRESAMPLE_CFLAGS_OTHER})
target_include_directories(ffmpeg_libs INTERFACE target_include_directories(ffmpeg_libs INTERFACE
${AVFORMAT_INCLUDE_DIRS} ${AVCODEC_INCLUDE_DIRS} ${AVFORMAT_INCLUDE_DIRS} ${AVCODEC_INCLUDE_DIRS}
${AVUTIL_INCLUDE_DIRS} ${SWSCALE_INCLUDE_DIRS}) ${AVUTIL_INCLUDE_DIRS} ${SWSCALE_INCLUDE_DIRS}
${SWRESAMPLE_INCLUDE_DIRS})
target_link_libraries(ffmpeg_libs INTERFACE target_link_libraries(ffmpeg_libs INTERFACE
${AVFORMAT_LIBRARIES} ${AVCODEC_LIBRARIES} ${AVFORMAT_LIBRARIES} ${AVCODEC_LIBRARIES}
${AVUTIL_LIBRARIES} ${SWSCALE_LIBRARIES}) ${AVUTIL_LIBRARIES} ${SWSCALE_LIBRARIES}
message(STATUS "FFmpeg: avformat=${AVFORMAT_VERSION} avcodec=${AVCODEC_VERSION}") ${SWRESAMPLE_LIBRARIES})
message(STATUS "FFmpeg: avformat=${AVFORMAT_VERSION} avcodec=${AVCODEC_VERSION} "
"swresample=${SWRESAMPLE_VERSION}")
# nlohmann/json (gallery + output serialisation) # nlohmann/json (gallery + output serialisation)
include(FetchContent) include(FetchContent)
@@ -249,6 +257,7 @@ find_package(HDF5 REQUIRED COMPONENTS CXX)
add_library(sae_gallery STATIC add_library(sae_gallery STATIC
src/gallery/gallery_store.cpp src/gallery/gallery_store.cpp
src/gallery/gallery_builder.cpp src/gallery/gallery_builder.cpp
src/audio_signature.cpp # IR-004 — content-derived audio signature
) )
set_target_properties(sae_gallery PROPERTIES POSITION_INDEPENDENT_CODE ON) set_target_properties(sae_gallery PROPERTIES POSITION_INDEPENDENT_CODE ON)
target_include_directories(sae_gallery PUBLIC src ${HDF5_INCLUDE_DIRS}) target_include_directories(sae_gallery PUBLIC src ${HDF5_INCLUDE_DIRS})
+428
View File
@@ -0,0 +1,428 @@
// ── JRay audio signature, v1 — implementation ────────────────────────────────
//
/// TRACES: IR-004, IR-007, IR-008 | SR-003
//
// The contract this implements is documented in full in audio_signature.hpp;
// read that before changing anything here. Every constant is load-bearing: the
// JRay Jellyfin plugin computes the same bytes in C#, and a signature that
// differs in any parameter simply does not match.
//
// Audio decode is a *second stream from an existing dependency* — the pipeline
// already links libavformat/libavcodec/libavutil for video (ffmpeg_decoder.hpp);
// this adds libswresample for the downmix+resample, no new project dependency.
// The FFT is written out here rather than pulled from a library for the same
// reason the plugin vendors one: it is a fixed, fully specified transform, and
// a dependency whose version could change the numerics is a liability when the
// output has to be bit-identical across two languages.
#include "audio_signature.hpp"
extern "C" {
#include <libavcodec/avcodec.h>
#include <libavformat/avformat.h>
#include <libavutil/avutil.h>
#include <libavutil/channel_layout.h>
#include <libavutil/opt.h>
#include <libavutil/samplefmt.h>
#include <libswresample/swresample.h>
}
#include <algorithm>
#include <cmath>
#include <cstring>
#include <vector>
namespace sae::audio {
namespace {
constexpr double kPi = 3.14159265358979323846;
// ── Band table ───────────────────────────────────────────────────────────────
// edge[b] = 300 * 10^(b/32); band b owns FFT bins [k_lo[b], k_lo[b+1]).
// ceil() of the edge in bins, so membership is decided once by integers rather
// than by a float comparison per bin per frame. The bands tile [112, 1115)
// contiguously with no gap and no overlap, which is what lets the frame energy
// below be accumulated from the per-band sums.
std::array<std::pair<int, int>, kNumBands> build_band_table() {
const double hz_per_bin = static_cast<double>(kSampleRate) / kFrameSize;
std::array<int, kNumBands + 1> k{};
for (int b = 0; b <= kNumBands; ++b) {
const double edge = kBandLoHz * std::pow(kBandHiHz / kBandLoHz,
static_cast<double>(b) / kNumBands);
k[b] = static_cast<int>(std::ceil(edge / hz_per_bin));
}
std::array<std::pair<int, int>, kNumBands> tbl{};
for (int b = 0; b < kNumBands; ++b) tbl[b] = {k[b], k[b + 1]};
return tbl;
}
// Hann, periodic: w[n] = 0.5 * (1 - cos(2*pi*n/N)). Not the symmetric (N-1)
// variant — the two differ, and the difference is observable.
const std::vector<double>& hann_window() {
static const std::vector<double> w = [] {
std::vector<double> v(kFrameSize);
for (int n = 0; n < kFrameSize; ++n)
v[n] = 0.5 * (1.0 - std::cos(2.0 * kPi * n / kFrameSize));
return v;
}();
return w;
}
// ── Radix-2 decimation-in-time complex FFT, in place, no normalisation ──────
// Twiddles are precomputed per stage from cos/sin of -2*pi*j/len so the angle
// is an exactly reproducible double in any language and only the libm rounding
// of cos/sin (≤1 ulp) can differ — orders of magnitude below the decision
// margins in the golden fixture.
struct FftTables {
std::vector<int> rev; // bit-reversal permutation
std::vector<std::vector<double>> wr, wi; // per stage
};
const FftTables& fft_tables() {
static const FftTables t = [] {
FftTables f;
f.rev.resize(kFrameSize);
int bits = 0;
while ((1 << bits) < kFrameSize) ++bits;
for (int i = 0; i < kFrameSize; ++i) {
int r = 0;
for (int b = 0; b < bits; ++b)
if (i & (1 << b)) r |= 1 << (bits - 1 - b);
f.rev[i] = r;
}
for (int len = 2; len <= kFrameSize; len <<= 1) {
const int half = len / 2;
std::vector<double> cr(half), ci(half);
for (int j = 0; j < half; ++j) {
const double ang = -2.0 * kPi * j / len;
cr[j] = std::cos(ang);
ci[j] = std::sin(ang);
}
f.wr.push_back(std::move(cr));
f.wi.push_back(std::move(ci));
}
return f;
}();
return t;
}
void fft_4096(std::vector<double>& re, std::vector<double>& im) {
const FftTables& t = fft_tables();
for (int i = 0; i < kFrameSize; ++i) {
const int j = t.rev[i];
if (i < j) { std::swap(re[i], re[j]); std::swap(im[i], im[j]); }
}
int stage = 0;
for (int len = 2; len <= kFrameSize; len <<= 1, ++stage) {
const int half = len / 2;
const std::vector<double>& wr = t.wr[stage];
const std::vector<double>& wi = t.wi[stage];
for (int base = 0; base < kFrameSize; base += len) {
for (int j = 0; j < half; ++j) {
const int a = base + j;
const int b = a + half;
const double tr = re[b] * wr[j] - im[b] * wi[j];
const double ti = re[b] * wi[j] + im[b] * wr[j];
re[b] = re[a] - tr; im[b] = im[a] - ti;
re[a] = re[a] + tr; im[a] = im[a] + ti;
}
}
}
}
int energy_class(double r) {
if (r < kEnergyClassEdges[0]) return 0;
if (r < kEnergyClassEdges[1]) return 1;
if (r < kEnergyClassEdges[2]) return 2;
return 3;
}
// ── FFmpeg RAII ─────────────────────────────────────────────────────────────
struct DecodeCtx {
AVFormatContext* fmt = nullptr;
AVCodecContext* dec = nullptr;
SwrContext* swr = nullptr;
AVFrame* frm = nullptr;
AVPacket* pkt = nullptr;
~DecodeCtx() {
if (swr) swr_free(&swr);
if (frm) av_frame_free(&frm);
if (pkt) av_packet_free(&pkt);
if (dec) avcodec_free_context(&dec);
if (fmt) avformat_close_input(&fmt);
}
};
bool open_resampler(DecodeCtx& c, const AVFrame* f) {
#if LIBAVUTIL_VERSION_INT >= AV_VERSION_INT(57, 24, 100)
AVChannelLayout out_layout;
av_channel_layout_default(&out_layout, 1); // mono
AVChannelLayout in_layout;
if (av_channel_layout_copy(&in_layout, &f->ch_layout) < 0) return false;
if (in_layout.nb_channels <= 0) {
av_channel_layout_uninit(&in_layout);
av_channel_layout_default(&in_layout, 1);
}
const int rc = swr_alloc_set_opts2(
&c.swr,
&out_layout, AV_SAMPLE_FMT_FLT, kSampleRate,
&in_layout, static_cast<AVSampleFormat>(f->format),
f->sample_rate ? f->sample_rate : kSampleRate,
0, nullptr);
av_channel_layout_uninit(&in_layout);
av_channel_layout_uninit(&out_layout);
if (rc < 0 || !c.swr) return false;
#else
const int64_t in_layout = f->channel_layout
? static_cast<int64_t>(f->channel_layout)
: av_get_default_channel_layout(f->channels ? f->channels : 1);
c.swr = swr_alloc_set_opts(
nullptr,
AV_CH_LAYOUT_MONO, AV_SAMPLE_FMT_FLT, kSampleRate,
in_layout, static_cast<AVSampleFormat>(f->format),
f->sample_rate ? f->sample_rate : kSampleRate,
0, nullptr);
if (!c.swr) return false;
#endif
return swr_init(c.swr) >= 0;
}
// Push one decoded frame (or a flush) through the resampler, dropping the
// leading `to_skip` output samples, and append to `out`.
void drain(SwrContext* swr, const AVFrame* f, int in_rate,
std::size_t& to_skip, std::vector<float>& out) {
const int64_t delay = swr_get_delay(swr, in_rate ? in_rate : kSampleRate);
const int in_n = f ? f->nb_samples : 0;
const int max_out = static_cast<int>(av_rescale_rnd(
delay + in_n, kSampleRate, in_rate ? in_rate : kSampleRate, AV_ROUND_UP)) + 32;
if (max_out <= 0) return;
std::vector<float> buf(static_cast<std::size_t>(max_out));
uint8_t* dst = reinterpret_cast<uint8_t*>(buf.data());
const int n = swr_convert(swr, &dst, max_out,
f ? const_cast<const uint8_t**>(f->extended_data) : nullptr,
in_n);
if (n <= 0) return;
std::size_t produced = static_cast<std::size_t>(n);
std::size_t off = 0;
if (to_skip) {
const std::size_t drop = std::min(to_skip, produced);
to_skip -= drop;
off = drop;
produced -= drop;
}
if (produced)
out.insert(out.end(), buf.begin() + off, buf.begin() + off + produced);
}
} // namespace
// ── Public surface ──────────────────────────────────────────────────────────
const std::array<std::pair<int, int>, kNumBands>& band_fft_bins() {
static const std::array<std::pair<int, int>, kNumBands> tbl = build_band_table();
return tbl;
}
std::string base64_encode(const std::uint8_t* data, std::size_t n) {
static constexpr char kAlphabet[] =
"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/";
std::string out;
out.reserve(((n + 2) / 3) * 4);
std::size_t i = 0;
for (; i + 3 <= n; i += 3) {
const std::uint32_t v = (std::uint32_t(data[i]) << 16) |
(std::uint32_t(data[i + 1]) << 8) |
std::uint32_t(data[i + 2]);
out += kAlphabet[(v >> 18) & 0x3F];
out += kAlphabet[(v >> 12) & 0x3F];
out += kAlphabet[(v >> 6) & 0x3F];
out += kAlphabet[v & 0x3F];
}
if (i < n) {
const bool two = (n - i) == 2;
const std::uint32_t v = (std::uint32_t(data[i]) << 16) |
(two ? (std::uint32_t(data[i + 1]) << 8) : 0u);
out += kAlphabet[(v >> 18) & 0x3F];
out += kAlphabet[(v >> 12) & 0x3F];
out += two ? kAlphabet[(v >> 6) & 0x3F] : '=';
out += '=';
}
return out;
}
std::uint64_t fnv1a64(const void* data, std::size_t n) {
const auto* p = static_cast<const std::uint8_t*>(data);
std::uint64_t h = 0xcbf29ce484222325ULL;
for (std::size_t i = 0; i < n; ++i) {
h ^= p[i];
h *= 0x100000001b3ULL;
}
return h;
}
/// TRACES: IR-004
std::vector<std::uint8_t> pack_frames(const std::vector<float>& mono) {
if (mono.size() < static_cast<std::size_t>(kFrameSize)) return {};
const std::size_t nframes = 1 + (mono.size() - kFrameSize) / kHopSize;
const auto& bands = band_fft_bins();
const auto& win = hann_window();
const int k_lo = bands.front().first;
const int k_hi = bands.back().second; // exclusive
const double bin_count = static_cast<double>(k_hi - k_lo);
std::vector<double> re(kFrameSize), im(kFrameSize);
std::vector<std::uint8_t> peak(nframes);
std::vector<double> energy(nframes);
for (std::size_t f = 0; f < nframes; ++f) {
const float* src = mono.data() + f * kHopSize;
for (int n = 0; n < kFrameSize; ++n) {
re[n] = static_cast<double>(src[n]) * win[n];
im[n] = 0.0;
}
fft_4096(re, im);
// Per-band mean magnitude; the bands tile the 3003000 Hz range with no
// gaps, so the frame's band-limited energy is the sum of the band sums.
double best = -1.0, total = 0.0;
int best_b = 0;
for (int b = 0; b < kNumBands; ++b) {
double sum = 0.0;
for (int k = bands[b].first; k < bands[b].second; ++k)
sum += std::sqrt(re[k] * re[k] + im[k] * im[k]);
total += sum;
const double mean = sum / (bands[b].second - bands[b].first);
if (mean > best) { best = mean; best_b = b; } // ties → lowest index
}
peak[f] = static_cast<std::uint8_t>(best_b);
energy[f] = total / bin_count;
}
// Reference is the upper median of the frame energies: an actually observed
// value (no averaging of the two middle samples), so it is bit-reproducible,
// gain-invariant and barely moves when the window is trimmed.
std::vector<double> sorted = energy;
std::sort(sorted.begin(), sorted.end());
const double ref = sorted[sorted.size() / 2];
std::vector<std::uint8_t> out(nframes);
for (std::size_t f = 0; f < nframes; ++f) {
const double r = std::log10((energy[f] + kEnergyEps) / (ref + kEnergyEps));
out[f] = static_cast<std::uint8_t>(((peak[f] & 0x1F) << 2) |
(energy_class(r) & 0x03));
}
return out;
}
/// TRACES: IR-004, IR-008
std::optional<std::string> signature_from_mono(const std::vector<float>& mono) {
const std::vector<std::uint8_t> packed = pack_frames(mono);
if (packed.empty()) return std::nullopt;
return std::string(kVersionPrefix) + base64_encode(packed.data(), packed.size());
}
/// TRACES: IR-004, IR-007
std::optional<std::vector<float>> decode_centre_window(const std::string& path) {
av_log_set_level(AV_LOG_ERROR);
DecodeCtx c;
if (avformat_open_input(&c.fmt, path.c_str(), nullptr, nullptr) < 0)
return std::nullopt;
if (avformat_find_stream_info(c.fmt, nullptr) < 0) return std::nullopt;
if (c.fmt->duration == AV_NOPTS_VALUE) return std::nullopt;
const double duration = static_cast<double>(c.fmt->duration) / AV_TIME_BASE;
// IR-007 — the window underflows, so there is no signature and no sync
// offset downstream. The plugin applies the identical rule.
if (duration < kWindowSec) return std::nullopt;
const int idx = av_find_best_stream(c.fmt, AVMEDIA_TYPE_AUDIO, -1, -1, nullptr, 0);
if (idx < 0) return std::nullopt; // no audio → no signature
AVStream* st = c.fmt->streams[idx];
const AVCodec* codec = avcodec_find_decoder(st->codecpar->codec_id);
if (!codec) return std::nullopt;
c.dec = avcodec_alloc_context3(codec);
if (!c.dec) return std::nullopt;
if (avcodec_parameters_to_context(c.dec, st->codecpar) < 0) return std::nullopt;
c.dec->thread_count = 0;
if (avcodec_open2(c.dec, codec, nullptr) < 0) return std::nullopt;
const double start_sec = duration / 2.0 - kWindowSec / 2.0;
// Seek to a packet at or before the window start; the exact start is then
// reached by discarding the leading output samples, which is what
// `ffmpeg -ss <t> -i <file>` does and therefore what the plugin sees.
if (start_sec > 0.0) {
const int64_t tgt = av_rescale_q(
static_cast<int64_t>(start_sec * AV_TIME_BASE), AV_TIME_BASE_Q, st->time_base);
if (av_seek_frame(c.fmt, idx, tgt, AVSEEK_FLAG_BACKWARD) >= 0)
avcodec_flush_buffers(c.dec);
}
c.frm = av_frame_alloc();
c.pkt = av_packet_alloc();
if (!c.frm || !c.pkt) return std::nullopt;
std::vector<float> mono;
mono.reserve(kWindowSamples + kSampleRate);
std::size_t to_skip = 0;
bool have_swr = false;
int in_rate = kSampleRate;
bool eof = false;
while (mono.size() < kWindowSamples && !eof) {
const int rr = av_read_frame(c.fmt, c.pkt);
if (rr < 0) {
eof = true;
avcodec_send_packet(c.dec, nullptr); // flush the decoder
} else if (c.pkt->stream_index != idx) {
av_packet_unref(c.pkt);
continue;
} else {
avcodec_send_packet(c.dec, c.pkt);
av_packet_unref(c.pkt);
}
while (avcodec_receive_frame(c.dec, c.frm) == 0) {
if (!have_swr) {
if (!open_resampler(c, c.frm)) return std::nullopt;
have_swr = true;
in_rate = c.frm->sample_rate ? c.frm->sample_rate : kSampleRate;
int64_t pts = c.frm->best_effort_timestamp;
if (pts == AV_NOPTS_VALUE) pts = c.frm->pts;
const double t0 = (pts == AV_NOPTS_VALUE)
? start_sec : av_q2d(st->time_base) * static_cast<double>(pts);
const double lead = start_sec - t0;
to_skip = lead > 0.0
? static_cast<std::size_t>(std::llround(lead * kSampleRate)) : 0;
}
drain(c.swr, c.frm, in_rate, to_skip, mono);
av_frame_unref(c.frm);
if (mono.size() >= kWindowSamples) break;
}
}
if (have_swr && mono.size() < kWindowSamples)
drain(c.swr, nullptr, in_rate, to_skip, mono); // flush the resampler
if (mono.empty()) return std::nullopt;
// Truncate to exactly 120.000 s so the frame count is 1288 for every input
// and does not wobble with seek granularity or the resampler tail.
if (mono.size() > kWindowSamples) mono.resize(kWindowSamples);
return mono;
}
/// TRACES: IR-004, IR-005, IR-007, IR-008
std::optional<std::string> compute_signature(const std::string& path) {
const std::optional<std::vector<float>> mono = decode_centre_window(path);
if (!mono) return std::nullopt;
return signature_from_mono(*mono);
}
} // namespace sae::audio
+158
View File
@@ -0,0 +1,158 @@
#pragma once
// ── JRay audio signature, v1 ─────────────────────────────────────────────────
//
/// TRACES: IR-004, IR-005, IR-007, IR-008 | SR-003
//
// A content-derived spectral-peak signature taken from the *centre* of the
// media, so a truth file is self-identifying: a consumer can tell whether a
// local file is the same cut as the one a manifest describes, and recover the
// frame offset when it is the same cut trimmed differently.
//
// The construction is owned by `JRay-public-server/SPEC.md` §3 and is
// reproduced by the JRay Jellyfin plugin in C#. **The two implementations must
// agree byte for byte** — a signature that differs in any parameter simply does
// not match, which defeats the entire point. Every deviation is therefore a
// breaking change and must go through the `v1:` prefix (see kVersionPrefix).
//
// Server spec §3, restated:
//
// 1. Decode a 120 s window centred on the midpoint (runtime/2 ± 60 s).
// 2. Downmix to mono, resample to 11025 Hz.
// 3. STFT: 4096-sample frame, 1024-sample hop, Hann window (~1290 frames).
// 4. Per frame, log-magnitude spectrum over 3003000 Hz.
// 5. 32 logarithmically spaced bins; peak bin index + coarse 2-bit energy
// class.
// 6. Pack one byte per frame; base64-encode.
// 7. Prefix `v1:`.
//
// ── Details the server spec leaves open, pinned here for v1 ──────────────────
//
// The prose above is not sufficient to reproduce a byte stream, so the choices
// below are the contract. They are mirrored in
// `tests/fixtures/audio/jray_audio_v1_golden.json`, which is the artefact
// shared with the plugin repo (IR-005).
//
// Arithmetic All DSP in IEEE-754 **double**. float32 is not sufficient:
// the golden fixture has frames whose two strongest bands are
// within 1.3% of each other, which double resolves identically
// everywhere and float32 does not.
// Sample scale FFmpeg's native s16→flt conversion, x * (1/32768), then
// widened to double. Values in [-1, 1).
// Framing Only whole frames: n_frames = 1 + (n_samples - 4096) / 1024,
// integer division, 0 when n_samples < 4096. A 120.000 s
// window is 1 323 000 samples → **1288 frames**.
// ("~1290" in the spec; the server accepts a tolerance.)
// Window Hann, **periodic**: w[n] = 0.5 * (1 - cos(2*pi*n/4096)).
// Not the symmetric (N-1) variant.
// Transform Plain radix-2 decimation-in-time complex FFT over 4096 real
// samples (imag = 0), no normalisation. Magnitude is
// sqrt(re² + im²). Twiddles from cos/sin of
// -2*pi*k/len computed in double.
// Band edges edge[b] = 300 * (3000/300)^(b/32), b = 0..32. Band b spans
// FFT bins [k_lo[b], k_lo[b+1]) with
// k_lo[b] = ceil(edge[b] * 4096 / 11025) — i.e. bins 112..1114
// inclusive, 8 bins in the narrowest band. Precomputed as an
// integer table so no float comparison decides membership.
// Band value **Mean** of the linear magnitudes in the band. Mean, not
// sum, so a wide high band is not favoured over a narrow low
// one; magnitude, not power, because it is an energy proxy and
// more codec-robust than a single bin's peak.
// Peak bin argmax over the 32 band values; ties resolve to the **lowest
// index**. The log of step 4 is a monotone squash and so
// cannot change an argmax — it is applied only where it is
// observable, in the energy class below.
// Energy class The spec says "coarse 2-bit energy class" and no more. v1
// defines it as the frame's band-limited energy relative to
// the window, which is invariant to gain (loudness
// normalisation must not change a signature) and robust to
// trimming (the median barely moves):
// E_f = mean magnitude over *all* FFT bins 112..1114
// Eref = median over frames of E_f, taken as the upper
// median sorted[n/2] — no averaging of the two middle
// values, so the reference is always an actual
// observed value and is bit-reproducible
// r = log10((E_f + 1e-12) / (Eref + 1e-12))
// class = 0 if r < -0.6, 1 if r < -0.2, 2 if r < 0.2, else 3
// The thresholds deliberately straddle r = 0 rather than sit
// on it, so the median frame itself is not on a boundary.
// Byte layout bit 7 = 0 (reserved), bits 6..2 = 5-bit band index,
// bits 1..0 = 2-bit energy class:
// byte = (band << 2) | class → always 0..127
// This is the structural constraint the server validates on
// upload (§3 "Validation and abuse").
// Base64 Standard alphabet AZaz09+/ with '=' padding.
//
// ── Short media (IR-007) ─────────────────────────────────────────────────────
//
// `runtime/2 ± 60 s` underflows below 120 s, so **no signature is emitted** and
// no sync offset is applied downstream. Both producers apply the identical
// rule; diverging here would break exactly the short items most likely to be
// misidentified. `compute_signature` returns `std::nullopt`.
//
// The same nullopt is returned for a file with no audio stream, an unopenable
// file, or an unknown duration. UR-9 is an enhancement and must never be able
// to break a fetch — degradation, not failure.
#include <array>
#include <cstddef>
#include <cstdint>
#include <optional>
#include <string>
#include <utility>
#include <vector>
namespace sae::audio {
// ── Contract constants — changing any of these is a `v1:` bump ───────────────
inline constexpr int kSampleRate = 11025;
inline constexpr int kFrameSize = 4096;
inline constexpr int kHopSize = 1024;
inline constexpr int kNumBands = 32;
inline constexpr double kBandLoHz = 300.0;
inline constexpr double kBandHiHz = 3000.0;
inline constexpr double kWindowSec = 120.0;
inline constexpr double kEnergyEps = 1e-12;
// Class thresholds on log10(E_frame / E_median); see the header comment.
inline constexpr double kEnergyClassEdges[3] = {-0.6, -0.2, 0.2};
// 120.000 s at 11025 Hz. The decoded window is truncated to exactly this so the
// frame count does not wobble with seek granularity or resampler tail.
inline constexpr std::size_t kWindowSamples =
static_cast<std::size_t>(kWindowSec * kSampleRate); // 1 323 000
inline constexpr std::size_t kExpectedFrames =
1 + (kWindowSamples - kFrameSize) / kHopSize; // 1288
static_assert(kWindowSamples == 1323000, "120 s at 11025 Hz");
static_assert(kExpectedFrames == 1288, "server spec's ~1290 frames");
/// The version prefix is the signature's own, separate from `schema_version`:
/// a future change to the DSP chain must be *detectable* rather than silently
/// producing non-matching signatures (IR-008).
inline constexpr const char* kVersionPrefix = "v1:";
/// FFT bin range [first, last) for each of the 32 log-spaced bands.
/// Computed once from the constants above; exposed so the golden fixture can
/// assert the table itself, not merely the signature it produces.
const std::array<std::pair<int, int>, kNumBands>& band_fft_bins();
/// Decode the centre window of `path` as mono float PCM at 11025 Hz.
/// nullopt when the media is shorter than 120 s (IR-007), has no audio stream,
/// or cannot be opened. Never throws.
std::optional<std::vector<float>> decode_centre_window(const std::string& path);
/// One packed byte per whole STFT frame. Empty when `mono` is shorter than one
/// frame. This is the payload that gets base64-encoded.
std::vector<std::uint8_t> pack_frames(const std::vector<float>& mono);
/// `v1:` + base64(pack_frames(mono)). nullopt when no whole frame fits.
std::optional<std::string> signature_from_mono(const std::vector<float>& mono);
/// Decode + sign. The one call the pipeline makes. nullopt per IR-007 and on
/// any decode failure — degradation, not failure.
std::optional<std::string> compute_signature(const std::string& path);
// ── Small utilities, exposed for the golden-fixture test ────────────────────
std::string base64_encode(const std::uint8_t* data, std::size_t n);
/// FNV-1a 64. Used only to pin the *decoded PCM* in the golden fixture, so a
/// codec-level difference is distinguishable from a DSP-level one.
std::uint64_t fnv1a64(const void* data, std::size_t n);
} // namespace sae::audio
+9 -1
View File
@@ -21,21 +21,29 @@ add_executable(sae_tests
test_face_utils.cpp test_face_utils.cpp
test_track_gallery.cpp test_track_gallery.cpp
test_face_tracker.cpp test_face_tracker.cpp
test_audio_signature.cpp
${CMAKE_SOURCE_DIR}/src/backends/gemm_backend.cpp ${CMAKE_SOURCE_DIR}/src/backends/gemm_backend.cpp
${CMAKE_SOURCE_DIR}/src/gallery/gallery_store.cpp ${CMAKE_SOURCE_DIR}/src/gallery/gallery_store.cpp
${CMAKE_SOURCE_DIR}/src/audio_signature.cpp
) )
target_include_directories(sae_tests PRIVATE ${CMAKE_SOURCE_DIR}/src) target_include_directories(sae_tests PRIVATE ${CMAKE_SOURCE_DIR}/src)
# SAE_GEMM_CPU: build the CPU reference GEMM regardless of the main backend. # SAE_GEMM_CPU: build the CPU reference GEMM regardless of the main backend.
# SAE_MODELS_DIR: config.hpp (pulled in by track_gallery.hpp) bakes model paths. # SAE_MODELS_DIR: config.hpp (pulled in by track_gallery.hpp) bakes model paths.
# SAE_TEST_FIXTURES_DIR: the audio golden vector is read from the source tree,
# not copied, so the file the plugin repo shares is the file under test.
target_compile_definitions(sae_tests PRIVATE target_compile_definitions(sae_tests PRIVATE
SAE_GEMM_CPU SAE_GEMM_CPU
SAE_MODELS_DIR="${SAE_MODELS_DIR}") SAE_MODELS_DIR="${SAE_MODELS_DIR}"
SAE_TEST_FIXTURES_DIR="${CMAKE_CURRENT_SOURCE_DIR}/fixtures")
# gallery_store.cpp + gallery_calibration.hpp use nlohmann/json and HDF5 # gallery_store.cpp + gallery_calibration.hpp use nlohmann/json and HDF5
# (galleries are HDF5-native, see src/gallery/gallery_store.cpp); face_utils.hpp # (galleries are HDF5-native, see src/gallery/gallery_store.cpp); face_utils.hpp
# and the calibration GEMM pull in OpenCV (calib3d/imgproc/core) via types.hpp. # and the calibration GEMM pull in OpenCV (calib3d/imgproc/core) via types.hpp.
# ffmpeg_libs: audio_signature.cpp decodes the golden fixture (avformat/avcodec/
# avutil/swresample). Still GPU-free — the audio path is pure CPU.
target_link_libraries(sae_tests PRIVATE target_link_libraries(sae_tests PRIVATE
Catch2::Catch2WithMain Catch2::Catch2WithMain
nlohmann_json::nlohmann_json nlohmann_json::nlohmann_json
ffmpeg_libs
${OpenCV_LIBS} ${OpenCV_LIBS}
${HDF5_CXX_LIBRARIES}) ${HDF5_CXX_LIBRARIES})
target_include_directories(sae_tests PRIVATE ${HDF5_INCLUDE_DIRS}) target_include_directories(sae_tests PRIVATE ${HDF5_INCLUDE_DIRS})
+194
View File
@@ -0,0 +1,194 @@
{
"_": "Golden vector for the JRay v1 audio signature (JRay-public-server SPEC.md \u00a73). Shared verbatim between scene-actor-extraction (C++) and the jRay Jellyfin plugin (C#) so the two implementations can be proven bit-identical. IR-004, IR-005, IR-007, IR-008.",
"version": "v1",
"signature": "v1:AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAeHx8eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh8fHzk5OTk5OTk5OTk5OTk5OTk5OTk5OTk5OTk5OTk5V1dXV1dXV1dXV1dXV1dXV1dXV1dXV1dXV1dXV1dXV1dXV1dycnJycnJycnJycnJycnJycnJycnJycnJycnJycnMPDgwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMKysrKysrKysrKysrKysrKysrKysrKysrKysrKysrKysrKytFRUVFRUVFRUVFRUVFRUVFRUVFRUVFRUVFRUVFRWNjY2NjY2NjY2NjY2NjY2NjY2NjY2NjY2NjY2NjY2NjY2Njfn5+fn5+fn5+fn5+fn5+fn5+fn5+fn5+fn5+fn5/GxoZGRkZGRkZGRkZGRkZGRkZGRkZGRkZGRkZGRkZGTc3Nzc3Nzc3Nzc3Nzc3Nzc3Nzc3Nzc3Nzc3Nzc3Nzc3Nzc3UlJSUlJSUlJSUlJSUlJSUlJSUlJSUlJSUlJSUlJSU1JsbGxsbGxsbGxsbGxsbGxsbGxsbGxsbGxsbGxsbAoLCwoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCwsLJSUlJSUlJSUlJSUlJSUlJSUlJSUlJSUlJSUlJSVDQ0NDQ0NDQ0NDQ0NDQ0NDQ0NDQ0NDQ0NDQ0NDQ0NDQ0NDQ15eXl5eXl5eXl5eXl5eXl5eXl5eXl5eXl5eXl5eX19eeXl5eXl5eXl5eXl5eXl5eXl5eXl5eXl5eXl5eXkXFxcXFxcXFxcXFxcXFxcXFxcXFxcXFxcXFxcXFxcXFxcXFzExMTExMTExMTExMTExMTExMTExMTExMTExMTExT09PT09PT09PT09PT09PT09PT09PT09PT09PT09PT09PT09qampqampqampqampqampqampqampqampqampqamsHBwUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyM+Pj4+Pj4+Pj4+Pj4+Pj4+Pj4+Pj4+Pj4+Pj4+Pj4/PlhYWFhYWFhYWFhYWFhYWFhYWFhYWFhYWFhYWFhYd3d3d3d3d3d3d3d3d3d3d3d3d3d3d3d3d3d3d3d3d3d3ExMRERERERERERERERERERERERERERERERERERERES8vLy8vLy8vLy8vLy8vLy8vLy8vLy8vLy8vLy8vLy8vLy8vSkpKSkpKSkpKSkpKSkpKSkpKSkpKSkpKSkpKSkpLS0plZWVlZWVlZWVlZWVlZWVlZWVlZWVlZWVlZWVlZQMDAwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMDHR0dHR0dHR0dHR0dHR0dHR0dHR0dHR0dHR0dHR0eHh44ODg4ODg4ODg4ODg4ODg4ODg4ODg4ODg4ODg4OFdXV1ZWVlZWVlZWVlZWVlZWVlZWVlZWVlZWVlZWVlZWV1dXcXFxcXFxcXFxcXFxcXFxcXFxcXFxcXFxcXFxcXEPDw8PDw8PDw8PDw8PDw8PDw8PDw8PDw8PDw8PDw8PDw8PDyoqKioqKioqKioqKioqKioqKioqKioqKioqKioqKysqRERERERERERERERERERERERERERERERERERERERjY2NjY2NjY2NjYw==",
"frame_count": 1288,
"media": {
"file": "jray_audio_v1_tone.flac",
"generator": "make_fixture.py",
"container": "FLAC (lossless \u2014 decodes to exactly the PCM make_fixture.py emits)",
"duration_sec": 120.0,
"sample_rate": 11025,
"channels": 1,
"sample_format": "s16",
"sha256": "912ecd426cd426dccb37753e0249694227619c701cb9f533502b37da0fbe8096",
"bytes": 585142
},
"decoded_window": {
"_": "Checksums of the 120 s centre window after downmix to mono and resample to 11025 Hz, i.e. exactly the stream `ffmpeg -ss <mid-60> -t 120 -i <file> -vn -ac 1 -ar 11025 -f f32le -` produces. Check these first: a mismatch here is a decode problem, not a DSP one.",
"samples": 1323000,
"f32le_fnv1a64": "0x1ef7899cd4d12662",
"s16le_fnv1a64": "0xf824fa56f125c0dc"
},
"params": {
"window_sec": 120.0,
"window_centre": "runtime/2, i.e. samples from runtime/2 - 60 s; truncated to exactly 1323000 samples",
"min_duration_sec": 120.0,
"min_duration_rule": "IR-007 \u2014 below this emit NO signature and apply no sync offset",
"sample_rate": 11025,
"channels": 1,
"arithmetic": "IEEE-754 double throughout; float32 is not sufficient",
"sample_scale": "s16 * (1/32768), FFmpeg's native s16->flt",
"frame_size": 4096,
"hop_size": 1024,
"frame_count_rule": "1 + (n_samples - 4096) / 1024, integer division; whole frames only",
"window_fn": "Hann, PERIODIC: w[n] = 0.5 * (1 - cos(2*pi*n/4096))",
"transform": "radix-2 DIT complex FFT over the 4096 real samples (imag=0), no normalisation",
"magnitude": "sqrt(re^2 + im^2), linear",
"band_lo_hz": 300.0,
"band_hi_hz": 3000.0,
"num_bands": 32,
"band_edges": "edge[b] = 300 * (3000/300)^(b/32), b = 0..32",
"band_bins": "band b owns FFT bins [k_lo[b], k_lo[b+1]) with k_lo[b] = ceil(edge[b] * 4096 / 11025); see band_fft_bins",
"band_value": "MEAN of the linear magnitudes in the band (not sum, not max)",
"peak_bin": "argmax over the 32 band values; ties resolve to the LOWEST index",
"energy_metric": "E = mean magnitude over all FFT bins 112..1114, i.e. the whole 300-3000 Hz band",
"energy_reference": "upper median of E over all frames: sorted[n/2], no averaging of the two middle values",
"energy_ratio": "r = log10((E + 1e-12) / (E_ref + 1e-12))",
"energy_class_edges": [
-0.6,
-0.2,
0.2
],
"energy_class": "0 if r < -0.6, 1 if r < -0.2, 2 if r < 0.2, else 3",
"byte_layout": "bit7 = 0 (reserved), bits6..2 = 5-bit band index, bits1..0 = 2-bit energy class; byte = (band << 2) | class",
"base64": "standard alphabet A-Za-z0-9+/ with '=' padding",
"prefix": "v1:"
},
"band_fft_bins": [
[
112,
120
],
[
120,
129
],
[
129,
139
],
[
139,
149
],
[
149,
160
],
[
160,
172
],
[
172,
185
],
[
185,
199
],
[
199,
213
],
[
213,
229
],
[
229,
246
],
[
246,
265
],
[
265,
285
],
[
285,
306
],
[
306,
328
],
[
328,
353
],
[
353,
379
],
[
379,
408
],
[
408,
438
],
[
438,
471
],
[
471,
506
],
[
506,
543
],
[
543,
584
],
[
584,
627
],
[
627,
674
],
[
674,
724
],
[
724,
778
],
[
778,
836
],
[
836,
899
],
[
899,
966
],
[
966,
1038
],
[
1038,
1115
]
],
"notes": [
"The server spec fixes the window, rate, STFT geometry, band and the 5+2 bit packing. Everything under params beyond that (Hann periodicity, band aggregation, the energy-class definition, tie-breaking, base64 alphabet) is pinned HERE for v1 \u2014 the spec does not constrain it, and two implementations that guess differently produce non-matching signatures.",
"Decision margins on this fixture: the two strongest bands are within 1.3% on the closest frame, and the closest frame to an energy-class edge is 3.6e-3 away in log10. Both are many orders of magnitude above double-precision FFT differences, so any two correct double- precision implementations agree; a float32 implementation is not guaranteed to.",
"Coverage: all 32 bands and all 4 energy classes appear in the golden signature.",
"Robustness observed on this fixture: identical peak-bin sequence after a stereo/44100 Hz round trip and after AAC 128 kbit/s re-encoding."
]
}
Binary file not shown.
+114
View File
@@ -0,0 +1,114 @@
#!/usr/bin/env python3
"""Regenerate the JRay audio-signature golden fixture.
python3 make_fixture.py # writes jray_audio_v1_tone.flac here
This is the *source of truth* for the fixture media: `jray_audio_v1_tone.flac`
is a lossless FLAC encoding of exactly the PCM this script emits, so any repo
that wants to check its own audio-signature implementation against the golden
vector in `jray_audio_v1_golden.json` can regenerate the input from scratch and
confirm it is byte-identical (the golden file records `pcm_fnv1a64`, a hash of
the decoded 16-bit samples).
Deliberately dependency-free (no numpy) and written in plain arithmetic so it
ports to any language in ~20 lines.
Signal — 120.000 s, mono, 11025 Hz, 16-bit signed PCM:
* split into segments of 32768 samples (~2.97 s), 40.4 segments in total;
* segment `s` carries one sine at the geometric centre of log-band
`(s * 7) mod 32` of the 300-3000 Hz band, so all 32 bands are exercised;
* its amplitude walks a golden-ratio low-discrepancy sequence over
[10^-1.55, 10^-0.02] so frame energies spread continuously across ~1.5
decades and all four energy classes are exercised, without a dense cluster
of frames sitting on a class boundary;
* phase is carried across segment boundaries (no clicks);
* a constant, far quieter 777 Hz tone sits underneath so no frame is
degenerate;
* samples are quantised with floor(x * 32767 + 0.5).
Why FLAC and not WAV: 120 s of 11025 Hz 16-bit PCM is 2.6 MB and does not
compress in git. FLAC is lossless — FFmpeg decodes it to exactly the PCM
written here — and is ~3.5x smaller. `--wav` writes the uncompressed original
if you want to diff it.
"""
import math
import struct
import subprocess
import sys
import os
SAMPLE_RATE = 11025
DURATION_SEC = 120.0
SEGMENT = 32768 # samples per tone segment
BAND_STRIDE = 7 # coprime with 32 -> visits every band
BAND_LO_HZ = 300.0
BAND_HI_HZ = 3000.0
NUM_BANDS = 32
AMP_LOG_MIN = -1.55 # 10^-1.55 ~= 0.028
AMP_LOG_SPAN = 1.53 # up to 10^-0.02 ~= 0.955
PHI_FRAC = 0.6180339887498949
BG_HZ = 777.0
BG_AMP = 0.004
OUT_FLAC = "jray_audio_v1_tone.flac"
OUT_WAV = "jray_audio_v1_tone.wav"
def generate():
"""Return the 120 s signal as a list of int16 sample values."""
n = int(round(SAMPLE_RATE * DURATION_SEC))
out = [0] * n
phase = 0.0
two_pi = 2.0 * math.pi
for start in range(0, n, SEGMENT):
s = start // SEGMENT
end = min(n, start + SEGMENT)
band = (s * BAND_STRIDE) % NUM_BANDS
# geometric centre of log-band `band`
freq = BAND_LO_HZ * (BAND_HI_HZ / BAND_LO_HZ) ** ((band + 0.5) / NUM_BANDS)
amp = 10.0 ** (AMP_LOG_MIN + AMP_LOG_SPAN * ((s * PHI_FRAC) % 1.0))
step = two_pi * freq / SAMPLE_RATE
for k in range(end - start):
i = start + k
x = amp * math.sin(phase + step * k)
x += BG_AMP * math.sin(two_pi * BG_HZ * i / SAMPLE_RATE)
if x > 1.0:
x = 1.0
elif x < -1.0:
x = -1.0
out[i] = int(math.floor(x * 32767.0 + 0.5))
phase = (phase + step * (end - start)) % two_pi
return out
def write_wav(path, samples):
data = struct.pack("<%dh" % len(samples), *samples)
hdr = b"RIFF" + struct.pack("<I", 36 + len(data)) + b"WAVE"
hdr += b"fmt " + struct.pack("<IHHIIHH", 16, 1, 1, SAMPLE_RATE,
SAMPLE_RATE * 2, 2, 16)
hdr += b"data" + struct.pack("<I", len(data))
with open(path, "wb") as fh:
fh.write(hdr + data)
def main():
here = os.path.dirname(os.path.abspath(__file__))
samples = generate()
wav = os.path.join(here, OUT_WAV)
write_wav(wav, samples)
if "--wav" in sys.argv:
print("wrote", wav)
return
flac = os.path.join(here, OUT_FLAC)
# -compression_level 12 is deterministic for a given libFLAC/ffmpeg build;
# only the container bytes vary, never the decoded PCM.
subprocess.run(["ffmpeg", "-nostdin", "-v", "error", "-y", "-i", wav,
"-c:a", "flac", "-compression_level", "12", flac],
check=True)
os.remove(wav)
print("wrote", flac)
if __name__ == "__main__":
main()
+359
View File
@@ -0,0 +1,359 @@
// Unit tests for the JRay v1 audio signature (src/audio_signature.*).
//
/// TRACES: UT-101, UT-102, UT-103, UT-104 | IR-004, IR-005, IR-007, IR-008
//
// The headline test is the golden vector: a deterministic tone fixture checked
// into tests/fixtures/audio/ together with the signature it must produce. That
// fixture is the artefact shared with the jRay plugin repo, and it is what
// makes "both producers agree bit-for-bit" a checked claim rather than an
// assertion (IR-005).
//
// GPU-free, model-free, no network. Pure CPU DSP plus an FFmpeg decode of a
// 585 KB file — which is precisely why this is the right cross-repo check: it
// runs anywhere, including the N100 CI host.
#include <catch2/catch_test_macros.hpp>
#include <catch2/matchers/catch_matchers_floating_point.hpp>
#include "audio_signature.hpp"
#include <nlohmann/json.hpp>
#include <cmath>
#include <cstdint>
#include <filesystem>
#include <fstream>
#include <stdexcept>
#include <string>
#include <vector>
using namespace sae::audio;
namespace fs = std::filesystem;
namespace {
const std::string kFixtureDir = SAE_TEST_FIXTURES_DIR "/audio";
const std::string kGoldenPath = kFixtureDir + "/jray_audio_v1_golden.json";
const std::string kMediaPath = kFixtureDir + "/jray_audio_v1_tone.flac";
const nlohmann::json& golden() {
static const nlohmann::json j = [] {
std::ifstream in(kGoldenPath);
if (!in.good())
throw std::runtime_error("golden fixture not found: " + kGoldenPath);
nlohmann::json parsed;
in >> parsed;
return parsed;
}();
return j;
}
std::uint64_t hex64(const std::string& s) {
return std::stoull(s, nullptr, 16);
}
// The fixture is 120 s of audio: decoding and signing it is the expensive part
// of this file, so both results are computed once and shared. Every test below
// still asserts against the on-disk golden values, not against each other.
const std::optional<std::vector<float>>& fixture_window() {
static const std::optional<std::vector<float>> w = decode_centre_window(kMediaPath);
return w;
}
const std::optional<std::string>& fixture_signature() {
static const std::optional<std::string> s = compute_signature(kMediaPath);
return s;
}
// ── Minimal WAV writer, so the short-media and resample cases need no fixture ─
// 16-bit PCM, interleaved.
struct TempWav {
fs::path path;
explicit TempWav(const std::string& name)
: path(fs::temp_directory_path() / ("sae_audio_test_" + name + ".wav")) {}
~TempWav() { std::error_code ec; fs::remove(path, ec); }
void write(const std::vector<std::int16_t>& samples, int rate, int channels) const {
const std::uint32_t bytes = static_cast<std::uint32_t>(samples.size() * 2);
const std::uint32_t byte_rate = static_cast<std::uint32_t>(rate * channels * 2);
std::ofstream out(path, std::ios::binary);
auto u32 = [&](std::uint32_t v) { out.write(reinterpret_cast<const char*>(&v), 4); };
auto u16 = [&](std::uint16_t v) { out.write(reinterpret_cast<const char*>(&v), 2); };
out.write("RIFF", 4); u32(36 + bytes); out.write("WAVE", 4);
out.write("fmt ", 4); u32(16); u16(1); u16(static_cast<std::uint16_t>(channels));
u32(static_cast<std::uint32_t>(rate)); u32(byte_rate);
u16(static_cast<std::uint16_t>(channels * 2)); u16(16);
out.write("data", 4); u32(bytes);
out.write(reinterpret_cast<const char*>(samples.data()), bytes);
}
};
// A plain 1 kHz tone, mono, at the signature's own rate.
std::vector<std::int16_t> tone(double seconds, int rate = kSampleRate) {
const std::size_t n = static_cast<std::size_t>(std::llround(seconds * rate));
std::vector<std::int16_t> s(n);
for (std::size_t i = 0; i < n; ++i)
s[i] = static_cast<std::int16_t>(std::llround(
20000.0 * std::sin(2.0 * 3.14159265358979323846 * 1000.0 * double(i) / rate)));
return s;
}
std::vector<std::uint8_t> base64_decode(const std::string& in) {
auto val = [](char c) -> int {
if (c >= 'A' && c <= 'Z') return c - 'A';
if (c >= 'a' && c <= 'z') return c - 'a' + 26;
if (c >= '0' && c <= '9') return c - '0' + 52;
if (c == '+') return 62;
if (c == '/') return 63;
return -1;
};
std::vector<std::uint8_t> out;
std::uint32_t acc = 0;
int bits = 0;
for (char c : in) {
const int v = val(c);
if (v < 0) continue; // '=' padding
acc = (acc << 6) | static_cast<std::uint32_t>(v);
bits += 6;
if (bits >= 8) {
bits -= 8;
out.push_back(static_cast<std::uint8_t>((acc >> bits) & 0xFF));
}
}
return out;
}
} // namespace
// ── UT-101 — the golden vector ──────────────────────────────────────────────
/// TRACES: UT-101 | IR-004, IR-005, IR-008
TEST_CASE("signature of the golden fixture matches the recorded value exactly",
"[audio_signature][golden]") {
REQUIRE(fs::exists(kMediaPath));
const std::optional<std::string>& sig = fixture_signature();
REQUIRE(sig.has_value());
CHECK(*sig == golden()["signature"].get<std::string>());
}
/// TRACES: UT-101 | IR-005
TEST_CASE("decoded centre window matches the recorded PCM checksum",
"[audio_signature][golden]") {
// Checked separately from the signature so a codec-level difference is
// distinguishable from a DSP-level one: if this passes and the signature
// test fails, the DSP diverged; if this fails, the decode did.
const std::optional<std::vector<float>>& mono = fixture_window();
REQUIRE(mono.has_value());
CHECK(mono->size() == golden()["decoded_window"]["samples"].get<std::size_t>());
CHECK(fnv1a64(mono->data(), mono->size() * sizeof(float)) ==
hex64(golden()["decoded_window"]["f32le_fnv1a64"].get<std::string>()));
}
/// TRACES: UT-101 | IR-004
TEST_CASE("log-spaced band table matches the recorded one", "[audio_signature][golden]") {
// The band->FFT-bin table is the part of the construction most likely to
// drift between two implementations, so it is pinned independently of the
// signature it produces.
const auto& tbl = band_fft_bins();
const auto& want = golden()["band_fft_bins"];
REQUIRE(want.size() == tbl.size());
for (std::size_t b = 0; b < tbl.size(); ++b) {
CHECK(tbl[b].first == want[b][0].get<int>());
CHECK(tbl[b].second == want[b][1].get<int>());
CHECK(tbl[b].second > tbl[b].first); // no empty band
if (b) CHECK(tbl[b].first == tbl[b - 1].second); // contiguous, no overlap
}
}
/// TRACES: UT-101 | IR-004, IR-008
TEST_CASE("signature is well-formed: v1 prefix, 1288 frames, structural bytes",
"[audio_signature][golden]") {
const std::optional<std::string>& sig = fixture_signature();
REQUIRE(sig.has_value());
// IR-008 — the signature carries its own version, separate from
// schema_version, so a future DSP change is detectable rather than silently
// producing non-matching signatures.
REQUIRE(sig->rfind(kVersionPrefix, 0) == 0);
const std::vector<std::uint8_t> bytes = base64_decode(sig->substr(3));
CHECK(bytes.size() == kExpectedFrames);
CHECK(bytes.size() == golden()["frame_count"].get<std::size_t>());
// The server validates this structure on upload (server SPEC §3): each byte
// is a 5-bit band index plus a 2-bit energy class, so bit 7 is always clear
// and arbitrary bytes are invalid. That is what keeps the field from being
// a payload channel.
bool bands_seen[kNumBands] = {};
bool classes_seen[4] = {};
for (std::uint8_t b : bytes) {
REQUIRE((b & 0x80) == 0);
bands_seen[(b >> 2) & 0x1F] = true;
classes_seen[b & 0x03] = true;
}
// The fixture is built to exercise the whole output alphabet — if it ever
// stops doing so, the golden vector has become a weaker check than it looks.
for (bool seen : bands_seen) CHECK(seen);
for (bool seen : classes_seen) CHECK(seen);
}
// ── UT-102 — IR-007, media shorter than the window ──────────────────────────
/// TRACES: UT-102 | IR-007
TEST_CASE("media shorter than 120 s emits no signature", "[audio_signature][short]") {
// The window runtime/2 ± 60 s underflows, so there is no signature and no
// sync offset downstream. Both producers must apply the identical rule or
// they diverge on exactly the short items most likely to be misidentified.
SECTION("30 s") {
TempWav w("short30");
w.write(tone(30.0), kSampleRate, 1);
CHECK_FALSE(compute_signature(w.path.string()).has_value());
CHECK_FALSE(decode_centre_window(w.path.string()).has_value());
}
SECTION("just under the boundary") {
TempWav w("short11999");
w.write(tone(119.99), kSampleRate, 1);
CHECK_FALSE(compute_signature(w.path.string()).has_value());
}
}
/// TRACES: UT-102 | IR-007
TEST_CASE("media of exactly 120 s emits a full-length signature",
"[audio_signature][short]") {
TempWav w("exact120");
w.write(tone(120.0), kSampleRate, 1);
const std::optional<std::string> sig = compute_signature(w.path.string());
REQUIRE(sig.has_value());
CHECK(base64_decode(sig->substr(3)).size() == kExpectedFrames);
}
/// TRACES: UT-102 | IR-007
TEST_CASE("unreadable media degrades to no signature rather than failing",
"[audio_signature][short]") {
// UR-9 is an enhancement and must never be able to break a fetch.
CHECK_FALSE(compute_signature("/nonexistent/definitely-not-here.mkv").has_value());
}
/// TRACES: UT-102 | IR-004
TEST_CASE("the window is taken from the centre, not the head",
"[audio_signature][centre]") {
// Sampling from the centre is the whole reason the construction avoids the
// head and tail (logos, cold opens, credits), so it needs its own check:
// wrap the fixture's own 120 s in 90 s of silence either side and the
// signature of the 300 s file must be the golden value, byte for byte.
// Nothing else pins the seek offset — a head-anchored window would pass
// every other test in this file.
const std::optional<std::vector<float>>& mono = fixture_window();
REQUIRE(mono.has_value());
const std::size_t pad = 90 * kSampleRate;
std::vector<std::int16_t> padded(pad * 2 + mono->size(), 0);
for (std::size_t i = 0; i < mono->size(); ++i)
padded[pad + i] = static_cast<std::int16_t>(std::llround(double((*mono)[i]) * 32768.0));
TempWav w("centred300");
w.write(padded, kSampleRate, 1);
const std::optional<std::string> sig = compute_signature(w.path.string());
REQUIRE(sig.has_value());
CHECK(*sig == golden()["signature"].get<std::string>());
}
// ── UT-103 — downmix and resample ───────────────────────────────────────────
/// TRACES: UT-103 | IR-004
TEST_CASE("stereo, non-native sample rate yields the same peak-bin sequence",
"[audio_signature][resample]") {
// The golden fixture is already mono at 11025 Hz so the golden vector does
// not depend on the resampler's version. This case exercises the path that
// real media takes — downmix plus resample — by rebuilding the fixture's own
// audio as 22050 Hz stereo and checking the peak bins survive it.
const std::optional<std::vector<float>>& mono = fixture_window();
REQUIRE(mono.has_value());
std::vector<std::int16_t> stereo;
stereo.reserve(mono->size() * 4);
for (float f : *mono) {
const auto s = static_cast<std::int16_t>(std::llround(double(f) * 32768.0));
stereo.push_back(s); stereo.push_back(s); // sample 1, L/R
stereo.push_back(s); stereo.push_back(s); // sample 2 (zero-order hold)
}
TempWav w("stereo22050");
w.write(stereo, 2 * kSampleRate, 2);
const std::optional<std::string> sig = compute_signature(w.path.string());
REQUIRE(sig.has_value());
const std::vector<std::uint8_t> got = base64_decode(sig->substr(3));
const std::vector<std::uint8_t> want =
base64_decode(golden()["signature"].get<std::string>().substr(3));
REQUIRE(got.size() == want.size());
std::size_t agree = 0;
for (std::size_t i = 0; i < got.size(); ++i)
agree += ((got[i] >> 2) == (want[i] >> 2)) ? 1 : 0;
// The server treats ≥ 0.85 as the `audio` match tier; this path scores 1.0
// in practice, and the margin is left for libswresample version drift.
CHECK(double(agree) / double(got.size()) >= 0.85);
}
// ── UT-104 — the pure DSP surface ───────────────────────────────────────────
/// TRACES: UT-104 | IR-004
TEST_CASE("pack_frames uses whole frames only", "[audio_signature][dsp]") {
CHECK(pack_frames(std::vector<float>(kFrameSize - 1, 0.f)).empty());
CHECK(pack_frames(std::vector<float>(kFrameSize, 0.f)).size() == 1);
CHECK(pack_frames(std::vector<float>(kFrameSize + kHopSize - 1, 0.f)).size() == 1);
CHECK(pack_frames(std::vector<float>(kFrameSize + kHopSize, 0.f)).size() == 2);
// The full 120 s window is 1288 frames — asserted as a constant rather than
// by running the DSP over 1.3 M zeros, which is the same claim for free.
CHECK(kWindowSamples == 1323000u);
CHECK(kExpectedFrames == 1288u);
CHECK_FALSE(signature_from_mono(std::vector<float>(kFrameSize - 1, 0.f)).has_value());
}
/// TRACES: UT-104 | IR-004
TEST_CASE("a pure tone lands in the band that contains it", "[audio_signature][dsp]") {
// 1000 Hz sits in log-band floor(32 * log10(1000/300)) = 16.
const int expect = static_cast<int>(std::floor(
kNumBands * std::log10(1000.0 / kBandLoHz) / std::log10(kBandHiHz / kBandLoHz)));
std::vector<float> mono(kWindowSamples / 100);
for (std::size_t i = 0; i < mono.size(); ++i)
mono[i] = static_cast<float>(0.5 * std::sin(
2.0 * 3.14159265358979323846 * 1000.0 * double(i) / kSampleRate));
const std::vector<std::uint8_t> packed = pack_frames(mono);
REQUIRE_FALSE(packed.empty());
for (std::uint8_t b : packed) CHECK(((b >> 2) & 0x1F) == expect);
}
/// TRACES: UT-104 | IR-004
TEST_CASE("signature is invariant to overall gain", "[audio_signature][dsp]") {
// Loudness normalisation between two releases of the same cut must not
// change the signature — that is why the energy class is relative.
std::vector<float> a(kWindowSamples / 50);
for (std::size_t i = 0; i < a.size(); ++i) {
const double t = double(i) / kSampleRate;
a[i] = static_cast<float>(0.4 * std::sin(2.0 * 3.14159265358979323846 * 640.0 * t) +
0.2 * std::sin(2.0 * 3.14159265358979323846 * 1900.0 * t) *
std::sin(2.0 * 3.14159265358979323846 * 0.7 * t));
}
std::vector<float> b(a.size());
for (std::size_t i = 0; i < a.size(); ++i) b[i] = a[i] * 0.25f;
CHECK(pack_frames(a) == pack_frames(b));
}
/// TRACES: UT-104 | IR-004
TEST_CASE("base64 encoder matches the standard alphabet and padding",
"[audio_signature][dsp]") {
auto enc = [](const std::string& s) {
return base64_encode(reinterpret_cast<const std::uint8_t*>(s.data()), s.size());
};
CHECK(enc("") == "");
CHECK(enc("f") == "Zg==");
CHECK(enc("fo") == "Zm8=");
CHECK(enc("foo") == "Zm9v");
CHECK(enc("foob") == "Zm9vYg==");
CHECK(enc("fooba") == "Zm9vYmE=");
CHECK(enc("foobar") == "Zm9vYmFy");
const std::uint8_t all[] = {0xFB, 0xFF, 0xBF}; // exercises '+' and '/'
CHECK(base64_encode(all, 3) == "+/+/");
}