feat(engine): add Python replay bindings, gallery pose-expansion, scene detection, embedding dumps

New C++ sources:
- kpn_bindings.cpp (sae_kpn): assembles the real face_tracker/identity_matcher/
  scene_tracker nodes inside a Python-driven KPN network via nanobind, for
  offline threshold-sweep replay against dumped embeddings (scripts/optimizer/).
- track_gallery.hpp: per-film gallery expansion — promotes a confidently-
  identified track's novel-pose reference views into an in-memory annex so
  later frames/tracks of that actor at similar poses are recognised, without
  touching the baked gallery.
- dump_embeddings.cpp: standalone exe that runs detect→embed only (no gallery,
  no matching) and dumps per-frame face embeddings + metadata to HDF5, so a
  parameter sweep can replay the expensive half once and vary tracking/matching
  config freely downstream.
- scene_detector.hpp / scene_detector_node.hpp: TransNetV2-based shot-boundary
  detection, opt-in alongside the always-on histogram cut detector.
- camera_position_change_detector_node.hpp, embedding_dump_node.hpp: supporting
  nodes for the above.
This commit is contained in:
2026-07-19 19:05:05 +02:00
parent 41a277bc19
commit 26139ffe8a
7 changed files with 1013 additions and 0 deletions
+179
View File
@@ -0,0 +1,179 @@
#pragma once
#include "types.hpp"
#include "config.hpp"
#include "inference/scene_detector.hpp"
#include <nlohmann/json.hpp>
#include <algorithm>
#include <atomic>
#include <deque>
#include <fstream>
#include <iostream>
#include <string>
#include <vector>
// ── SceneDetectorFunc ─────────────────────────────────────────────────────────
// KPN sink node: TransNetV2 shot-boundary detection on the dense frame stream.
//
// Buffers incoming (dense, native-rate) Frames into a rolling window of
// ISceneDetector::kWindow (=100) frames. Every `stride` frames it runs one
// inference and reads back per-frame boundary probabilities, but only trusts the
// central region of each window — TransNetV2 (like most sliding-window boundary
// models) is unreliable near the window edges where it lacks temporal context.
// Overlapping windows by (kWindow - stride) frames means every frame is scored
// from at least one window's trusted centre.
//
// Boundaries (prob > scene_threshold, local maxima) are collected with their
// timestamps and written to scenes.json alongside the main annotations output on
// EOF. This branch is terminal: it produces no pipeline messages, only a file.
struct SceneDetectorFunc {
static constexpr std::string_view label() { return "scene_detector"; }
SceneDetectorFunc(const Config& cfg, std::atomic<bool>& done)
: detector_(make_scene_detector(cfg))
, threshold_(cfg.scene_threshold)
, stride_(std::clamp(cfg.scene_stride, 1, ISceneDetector::kWindow))
, output_path_(scenes_path(cfg.output_path))
, movie_path_(cfg.movie_path)
, done_(done)
{
// Trusted centre half of each window. Frames outside [guard, kWindow-guard)
// are re-scored by an adjacent window, so we ignore them here to avoid
// edge artefacts and double-counting.
guard_ = (ISceneDetector::kWindow - stride_) / 2;
std::cerr << "[scene_detector] threshold=" << threshold_
<< " stride=" << stride_
<< " guard=" << guard_
<< " output=" << output_path_ << "\n";
}
void operator()(Frame f) {
if (f.eof) {
flush_remaining();
write_output();
done_.store(true, std::memory_order_release);
return;
}
images_.push_back(f.image);
times_.push_back(f.timestamp_sec);
// Once we have a full window, score it and slide forward by `stride`.
while (static_cast<int>(images_.size()) >= ISceneDetector::kWindow) {
score_window();
for (int i = 0; i < stride_; ++i) {
images_.pop_front();
times_.pop_front();
}
window_base_ += stride_;
}
}
private:
// Run TransNetV2 on the leading kWindow frames of the buffer and record any
// boundaries found within the trusted centre region.
void score_window() {
std::vector<cv::Mat> win(images_.begin(),
images_.begin() + ISceneDetector::kWindow);
std::vector<float> probs = detector_->detect_window(win);
// On the very first window there is no preceding window, so trust from 0;
// otherwise skip the leading guard already covered by the previous window.
const int lo = (window_base_ == 0) ? 0 : guard_;
const int hi = ISceneDetector::kWindow - guard_;
for (int i = lo; i < hi; ++i) {
if (probs[i] <= threshold_) continue;
// Local maximum → the boundary frame (avoid a run of high scores
// registering as several adjacent cuts).
const bool peak =
(i == 0 || probs[i] >= probs[i-1]) &&
(i == kLast_() || probs[i] >= probs[i+1]);
if (peak)
boundaries_.push_back({times_[i], probs[i]});
}
}
// At EOF the tail (< kWindow frames) never formed a full window. Pad it out
// to kWindow by repeating the last frame so the final real frames still get
// scored, then take only the region past what earlier windows covered.
void flush_remaining() {
const int n = static_cast<int>(images_.size());
if (n == 0) return;
std::vector<cv::Mat> win(images_.begin(), images_.end());
cv::Mat last = win.back();
while (static_cast<int>(win.size()) < ISceneDetector::kWindow)
win.push_back(last);
std::vector<float> probs = detector_->detect_window(win);
const int lo = (window_base_ == 0) ? 0 : guard_;
for (int i = lo; i < n; ++i) { // only real (non-padded) frames
if (probs[i] <= threshold_) continue;
const bool peak =
(i == 0 || probs[i] >= probs[i-1]) &&
(i == n - 1 || probs[i] >= probs[i+1]);
if (peak)
boundaries_.push_back({times_[i], probs[i]});
}
}
void write_output() {
if (written_) return;
written_ = true;
// Merge boundaries closer than one frame apart (dedup across window seams).
std::sort(boundaries_.begin(), boundaries_.end(),
[](const Boundary& a, const Boundary& b) {
return a.t < b.t;
});
nlohmann::json root;
root["schema_version"] = 1;
root["movie"] = movie_path_;
root["model"] = "transnetv2";
root["threshold"] = threshold_;
nlohmann::json cuts = nlohmann::json::array();
double last_t = -1e9;
for (const auto& b : boundaries_) {
if (b.t - last_t < 0.04) continue; // ~1 frame @25fps dedup
cuts.push_back({{"t", b.t}, {"probability", b.prob}});
last_t = b.t;
}
root["cuts"] = std::move(cuts);
std::ofstream f(output_path_);
if (!f.is_open()) {
std::cerr << "\n[scene_detector] ERROR: cannot write "
<< output_path_ << "\n";
return;
}
f << root.dump(2) << "\n";
std::cerr << "\n[scene_detector] wrote " << root["cuts"].size()
<< " boundaries → " << output_path_ << "\n";
}
static int kLast_() { return ISceneDetector::kWindow - 1; }
// annotations.json → annotations.scenes.json (or scenes.json for bare names)
static std::string scenes_path(const std::string& out) {
auto dot = out.find_last_of('.');
if (dot == std::string::npos) return out + ".scenes.json";
return out.substr(0, dot) + ".scenes.json";
}
struct Boundary { double t; float prob; };
std::unique_ptr<ISceneDetector> detector_;
float threshold_;
int stride_;
int guard_{0};
std::string output_path_;
std::string movie_path_;
std::atomic<bool>& done_;
std::deque<cv::Mat> images_;
std::deque<double> times_;
int64_t window_base_{0}; // frame index of images_.front()
std::vector<Boundary> boundaries_;
bool written_{false};
};