feat(engine): add Python replay bindings, gallery pose-expansion, scene detection, embedding dumps

New C++ sources:
- kpn_bindings.cpp (sae_kpn): assembles the real face_tracker/identity_matcher/
  scene_tracker nodes inside a Python-driven KPN network via nanobind, for
  offline threshold-sweep replay against dumped embeddings (scripts/optimizer/).
- track_gallery.hpp: per-film gallery expansion — promotes a confidently-
  identified track's novel-pose reference views into an in-memory annex so
  later frames/tracks of that actor at similar poses are recognised, without
  touching the baked gallery.
- dump_embeddings.cpp: standalone exe that runs detect→embed only (no gallery,
  no matching) and dumps per-frame face embeddings + metadata to HDF5, so a
  parameter sweep can replay the expensive half once and vary tracking/matching
  config freely downstream.
- scene_detector.hpp / scene_detector_node.hpp: TransNetV2-based shot-boundary
  detection, opt-in alongside the always-on histogram cut detector.
- camera_position_change_detector_node.hpp, embedding_dump_node.hpp: supporting
  nodes for the above.
This commit is contained in:
2026-07-19 19:05:05 +02:00
parent 41a277bc19
commit 26139ffe8a
7 changed files with 1013 additions and 0 deletions
+124
View File
@@ -0,0 +1,124 @@
#pragma once
#include "types.hpp"
#include "config.hpp"
#include <H5Cpp.h>
#include <cstdint>
#include <iostream>
#include <string>
#include <vector>
// ── EmbeddingDumpFunc ─────────────────────────────────────────────────────────
// KPN sink that taps the EmbeddedSceneFrame channel and writes the per-frame face
// metadata + embeddings to one HDF5 file (schema: scripts/optimizer/SCHEMA.md).
// The dump is the expensive, parameter-independent half of the pipeline
// (decode→detect→align→embed); replaying it lets a threshold sweep re-run the cheap
// downstream nodes thousands of times with no GPU. See sae_kpn / scripts/optimizer.
//
// Accumulates in flat/ragged arrays and writes once on EOF.
struct EmbeddingDumpFunc {
static constexpr std::string_view label() { return "embedding_dump"; }
EmbeddingDumpFunc(const Config& cfg, std::atomic<bool>& done)
: path_(cfg.dump_embeddings_path), movie_(cfg.movie_path),
sample_fps_(cfg.sample_fps), done_(done)
{
std::cerr << "[embedding_dump] writing " << path_ << "\n";
}
void operator()(EmbeddedSceneFrame ef) {
if (ef.source.eof) { flush(); return; }
const int32_t n = static_cast<int32_t>(ef.faces.size());
ts_.push_back(ef.source.timestamp_sec);
fidx_.push_back(ef.source.frame_idx);
is_cut_.push_back(ef.source.is_cut ? 1 : 0);
is_bnd_.push_back(ef.source.is_scene_boundary ? 1 : 0);
face_off_.push_back(static_cast<int64_t>(conf_.size()));
face_cnt_.push_back(n);
for (int i = 0; i < n; ++i) {
const auto& f = ef.faces[i];
bbox_.insert(bbox_.end(), {f.bbox.x, f.bbox.y, f.bbox.width, f.bbox.height});
for (int k = 0; k < 5; ++k) {
lmk_.push_back(f.landmarks[k].x);
lmk_.push_back(f.landmarks[k].y);
}
conf_.push_back(f.confidence);
const auto& e = ef.embeddings[i];
emb_.insert(emb_.end(), e.begin(), e.end());
}
}
void flush() {
if (written_.exchange(true)) return;
try {
write_hdf5();
} catch (const H5::Exception& e) {
std::cerr << "[embedding_dump] HDF5 error: " << e.getDetailMsg() << "\n";
}
done_.store(true, std::memory_order_release);
}
private:
static constexpr int kSchemaVersion = 1;
static constexpr int kEmbedDim = 512;
template<typename T>
void write_vec(H5::Group& g, const char* name, const std::vector<T>& v,
const H5::PredType& dtype, hsize_t cols = 0) {
hsize_t rows = cols ? v.size() / cols : v.size();
std::vector<hsize_t> dims = cols ? std::vector<hsize_t>{rows, cols}
: std::vector<hsize_t>{rows};
H5::DataSpace space(static_cast<int>(dims.size()), dims.data());
auto ds = g.createDataSet(name, dtype, space);
if (!v.empty()) ds.write(v.data(), dtype);
}
void write_hdf5() {
H5::H5File file(path_, H5F_ACC_TRUNC);
// root attrs
auto scalar = H5::DataSpace(H5S_SCALAR);
auto ver = file.createAttribute("schema_version", H5::PredType::NATIVE_INT, scalar);
int sv = kSchemaVersion; ver.write(H5::PredType::NATIVE_INT, &sv);
auto ed = file.createAttribute("embed_dim", H5::PredType::NATIVE_INT, scalar);
int dim = kEmbedDim; ed.write(H5::PredType::NATIVE_INT, &dim);
auto fps = file.createAttribute("sample_fps", H5::PredType::NATIVE_FLOAT, scalar);
fps.write(H5::PredType::NATIVE_FLOAT, &sample_fps_);
H5::StrType str(H5::PredType::C_S1, H5T_VARIABLE);
auto mv = file.createAttribute("movie", str, scalar);
mv.write(str, movie_);
H5::Group frames = file.createGroup("frames");
write_vec(frames, "timestamp_sec", ts_, H5::PredType::NATIVE_DOUBLE);
write_vec(frames, "frame_idx", fidx_, H5::PredType::NATIVE_INT64);
write_vec(frames, "is_cut", is_cut_, H5::PredType::NATIVE_UINT8);
write_vec(frames, "is_scene_boundary", is_bnd_, H5::PredType::NATIVE_UINT8);
write_vec(frames, "face_offset", face_off_, H5::PredType::NATIVE_INT64);
write_vec(frames, "face_count", face_cnt_, H5::PredType::NATIVE_INT32);
H5::Group faces = file.createGroup("faces");
write_vec(faces, "embedding", emb_, H5::PredType::NATIVE_FLOAT, kEmbedDim);
write_vec(faces, "bbox", bbox_, H5::PredType::NATIVE_FLOAT, 4);
write_vec(faces, "landmarks", lmk_, H5::PredType::NATIVE_FLOAT, 10);
write_vec(faces, "confidence", conf_, H5::PredType::NATIVE_FLOAT);
std::cerr << "[embedding_dump] wrote " << ts_.size() << " frames, "
<< conf_.size() << " faces → " << path_ << "\n";
}
std::string path_, movie_;
float sample_fps_;
std::atomic<bool>& done_;
std::atomic<bool> written_{false};
std::vector<double> ts_;
std::vector<int64_t> fidx_;
std::vector<uint8_t> is_cut_, is_bnd_;
std::vector<int64_t> face_off_;
std::vector<int32_t> face_cnt_;
std::vector<float> emb_, bbox_, lmk_, conf_;
};