Initial commit: scene-actor-extraction pipeline
Source (KPN++ pipeline nodes, ArcFace embedders, SCRFD/YuNet detectors, gallery builder), build scripts, and eval artifacts. - external/KPN as a git submodule (gitea.tourolle.paris/dtourolle/KPN) - ONNX models tracked via Git LFS (models/*.onnx) - generated outputs, TensorRT engines, reference repos, and media ignored
This commit is contained in:
+124
@@ -0,0 +1,124 @@
|
||||
#pragma once
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
|
||||
// ── Embedding ─────────────────────────────────────────────────────────────────
|
||||
// 512-dim L2-normalised ArcFace embedding
|
||||
using Embedding = std::array<float, 512>;
|
||||
|
||||
inline float cosine_similarity(const Embedding& a, const Embedding& b) {
|
||||
float dot = 0.f;
|
||||
for (int i = 0; i < 512; ++i) dot += a[i] * b[i];
|
||||
return dot;
|
||||
}
|
||||
|
||||
// ── Frame ─────────────────────────────────────────────────────────────────────
|
||||
// Raw sampled frame from the movie. eof=true is the pipeline shutdown sentinel:
|
||||
// every node must forward it immediately without processing.
|
||||
struct Frame {
|
||||
cv::Mat image;
|
||||
double timestamp_sec{0.0};
|
||||
int64_t frame_idx{-1};
|
||||
bool eof{false};
|
||||
bool is_cut{false}; // true when a hard scene cut was detected before this frame
|
||||
};
|
||||
|
||||
// ── ArcFace alignment ─────────────────────────────────────────────────────────
|
||||
// Canonical 5-point target positions for a 112×112 ArcFace crop.
|
||||
// Landmark order: right-eye, left-eye, nose, right-mouth, left-mouth
|
||||
// (matches SCRFD output order — no reordering needed).
|
||||
inline constexpr float kArcFaceRef[5][2] = {
|
||||
{38.2946f, 51.6963f},
|
||||
{73.5318f, 51.5014f},
|
||||
{56.0252f, 71.7366f},
|
||||
{41.5493f, 92.3655f},
|
||||
{70.7299f, 92.2041f},
|
||||
};
|
||||
|
||||
// ── DetectedFace ──────────────────────────────────────────────────────────────
|
||||
// One face found by SCRFD in a Frame.
|
||||
// Landmark order matches ArcFace convention (same as SCRFD output order):
|
||||
// [0] right-eye-centre [1] left-eye-centre [2] nose
|
||||
// [3] right-mouth [4] left-mouth
|
||||
struct DetectedFace {
|
||||
cv::Rect2f bbox;
|
||||
std::array<cv::Point2f, 5> landmarks;
|
||||
float confidence{0.f};
|
||||
};
|
||||
|
||||
// ── Pipeline messages ─────────────────────────────────────────────────────────
|
||||
|
||||
struct SceneFrame {
|
||||
Frame source;
|
||||
std::vector<DetectedFace> faces; // empty when no faces detected (or eof)
|
||||
};
|
||||
|
||||
struct AlignedSceneFrame {
|
||||
Frame source;
|
||||
std::vector<DetectedFace> faces;
|
||||
std::vector<cv::Mat> crops; // 112×112 BGR, ArcFace-ready; parallel to faces
|
||||
};
|
||||
|
||||
struct EmbeddedSceneFrame {
|
||||
Frame source;
|
||||
std::vector<DetectedFace> faces;
|
||||
std::vector<cv::Mat> crops; // forwarded for debug rendering downstream
|
||||
std::vector<Embedding> embeddings;
|
||||
};
|
||||
|
||||
// ── Face tracking ─────────────────────────────────────────────────────────────
|
||||
// Output of FaceTrackerFunc — EmbeddedSceneFrame augmented with per-detection
|
||||
// track context. track_embeddings[i] is the L2-normalised running mean across
|
||||
// the track's history; use it for identity matching when track_mature[i] is true.
|
||||
|
||||
struct TrackedSceneFrame {
|
||||
Frame source;
|
||||
std::vector<DetectedFace> faces;
|
||||
std::vector<cv::Mat> crops;
|
||||
std::vector<int> track_ids; // -1 = brand-new track this frame
|
||||
std::vector<Embedding> embeddings; // per-frame raw (from embedder)
|
||||
std::vector<Embedding> track_embeddings; // accumulated mean per track
|
||||
std::vector<bool> track_mature; // true once track has ≥ min_frames obs.
|
||||
};
|
||||
|
||||
// ── Identity matching ─────────────────────────────────────────────────────────
|
||||
|
||||
struct IdentifiedActor {
|
||||
int actor_idx{-1}; // index into ActorGallery::actors; -1 = unknown
|
||||
int track_id{-1}; // face track ID from FaceTrackerFunc
|
||||
std::string name;
|
||||
std::string imdb_id;
|
||||
float similarity{0.f}; // calibrated P(match) or cosine similarity; 0 for unknowns
|
||||
cv::Rect2f bbox;
|
||||
cv::Mat crop; // 112×112 aligned crop (stored as shared_ptr by KPN)
|
||||
};
|
||||
|
||||
struct MatchedSceneFrame {
|
||||
Frame source;
|
||||
std::vector<IdentifiedActor> actors; // includes unknowns (actor_idx == -1)
|
||||
};
|
||||
|
||||
// ── Scene annotation ──────────────────────────────────────────────────────────
|
||||
// Output of the scene tracker: one per sampled frame.
|
||||
// visible_actors contains all actors still within their extinction window.
|
||||
struct SceneAnnotation {
|
||||
double timestamp_sec{0.0};
|
||||
std::vector<IdentifiedActor> visible_actors;
|
||||
bool eof{false};
|
||||
};
|
||||
|
||||
// ── Actor gallery ─────────────────────────────────────────────────────────────
|
||||
// Loaded once at startup; baked into the identity matcher.
|
||||
struct ActorGallery {
|
||||
struct Actor {
|
||||
std::string imdb_id;
|
||||
std::string name;
|
||||
std::vector<Embedding> embeddings; // one per reference image
|
||||
std::vector<std::string> source_images;
|
||||
};
|
||||
std::vector<Actor> actors;
|
||||
};
|
||||
Reference in New Issue
Block a user