#pragma once #include #include #include #include #include #include "gallery/embedder_stamp.hpp" // ── Embedding ───────────────────────────────────────────────────────────────── // 512-dim L2-normalised ArcFace embedding using Embedding = std::array; inline float cosine_similarity(const Embedding& a, const Embedding& b) { float dot = 0.f; for (int i = 0; i < 512; ++i) dot += a[i] * b[i]; return dot; } // ── Frame ───────────────────────────────────────────────────────────────────── // Raw sampled frame from the movie. eof=true is the pipeline shutdown sentinel: // every node must forward it immediately without processing. struct Frame { cv::Mat image; double timestamp_sec{0.0}; int64_t frame_idx{-1}; bool eof{false}; bool is_cut{false}; // histogram: intra-scene camera-angle change (tracker reset) bool is_scene_boundary{false}; // TransNetV2: true shot/scene boundary (opt-in) float cut_score{0.f}; // histogram cut score = 1 - hist_corr (0=identical, ~1=cut); HUD/debug float bbox_upscale{1.f}; // multiply detector bboxes/landmarks by this to map back to // original video resolution (>1 when dense_scale downscaled the frame) }; // ── CutEvent ────────────────────────────────────────────────────────────────── // Emitted by SceneDetectorFunc when TransNetV2 localises a shot boundary, keyed // by the boundary frame's timestamp. eof=true is the shutdown sentinel. struct CutEvent { double timestamp_sec{0.0}; float probability{0.f}; // sigmoid boundary score at the peak bool eof{false}; }; // ── ArcFace alignment ───────────────────────────────────────────────────────── // Canonical 5-point target positions for a 112×112 ArcFace crop. // Landmark order: right-eye, left-eye, nose, right-mouth, left-mouth // (matches SCRFD output order — no reordering needed). inline constexpr float kArcFaceRef[5][2] = { {38.2946f, 51.6963f}, {73.5318f, 51.5014f}, {56.0252f, 71.7366f}, {41.5493f, 92.3655f}, {70.7299f, 92.2041f}, }; // ── DetectedFace ────────────────────────────────────────────────────────────── // One face found by SCRFD in a Frame. // Landmark order matches ArcFace convention (same as SCRFD output order): // [0] right-eye-centre [1] left-eye-centre [2] nose // [3] right-mouth [4] left-mouth struct DetectedFace { cv::Rect2f bbox; std::array landmarks; float confidence{0.f}; // AR-030 visibility: RMS landmark misfit, in canonical 112×112 pixels, left // over after the best similarity fit to the ArcFace template. Rises with // out-of-plane pose and with occlusion; blind to in-plane roll and to face // size, both of which the fit absorbs. Set by the aligner, which is where // the transform is computed; -1 until then. float alignment_residual{-1.f}; }; // ── Pipeline messages ───────────────────────────────────────────────────────── struct SceneFrame { Frame source; std::vector faces; // empty when no faces detected (or eof) }; struct AlignedSceneFrame { Frame source; std::vector faces; std::vector crops; // 112×112 BGR, ArcFace-ready; parallel to faces }; struct EmbeddedSceneFrame { Frame source; std::vector faces; std::vector crops; // forwarded for debug rendering downstream std::vector embeddings; }; // ── Face tracking ───────────────────────────────────────────────────────────── // Output of FaceTrackerFunc — EmbeddedSceneFrame augmented with per-detection // track context. struct TrackedSceneFrame { Frame source; std::vector faces; std::vector crops; std::vector track_ids; // -1 = brand-new track this frame std::vector embeddings; // per-frame raw (from embedder) }; // ── Identity matching ───────────────────────────────────────────────────────── struct IdentifiedActor { int actor_idx{-1}; // index into ActorGallery::actors; -1 = unknown int track_id{-1}; // face track ID from FaceTrackerFunc std::string name; std::string imdb_id; std::string tmdb_id; std::string jellyfin_id; // Jellyfin Person item GUID, if gallery was built from Jellyfin float similarity{0.f}; // calibrated P(match) or cosine similarity; 0 for unknowns cv::Rect2f bbox; cv::Mat crop; // 112×112 aligned crop (stored as shared_ptr by KPN) }; struct MatchedSceneFrame { Frame source; std::vector actors; // includes unknowns (actor_idx == -1) }; // ── Scene annotation ────────────────────────────────────────────────────────── // Output of the scene tracker: one per sampled frame. // visible_actors contains all actors still within their extinction window. struct SceneAnnotation { double timestamp_sec{0.0}; std::vector visible_actors; bool eof{false}; }; // ── Actor gallery ───────────────────────────────────────────────────────────── // Loaded once at startup; baked into the identity matcher. struct ActorGallery { struct Actor { std::string imdb_id; std::string tmdb_id; std::string jellyfin_id; // Jellyfin Person item GUID, if known std::string name; std::vector embeddings; // one per reference image std::vector source_images; }; std::vector actors; /// TRACES: GR-004 | SR-001 // Which embedder produced every embedding above. Empty == the file predates // model binding; see gallery/embedder_stamp.hpp for what is checked and why. EmbedderStamp embedder; // Cached Platt-sigmoid calibration (see gallery/gallery_calibration.hpp), // stored alongside the gallery in HDF5 so it never needs recomputing // unless the reference embeddings actually change. calib_valid=false and // calib_hash=0 means "not present in this file, compute it." float calib_a{10.f}; float calib_b{-5.f}; bool calib_valid{false}; uint64_t calib_hash{0}; };