Every embedding now carries the quality of the input it came from. Both axes fall out of the AR-005 warp for free: crop_sharpness() is the normalised Laplacian variance over the aligned 112x112, so contrast and size cannot leak into it, and the alignment residual is the part of the landmark deformation a similarity transform cannot explain, so in-plane roll reads as zero and foreshortening does not. Carried, not consumed. Nothing discounts or thresholds on either number yet -- that is AR-030 and VR-012, and the knee has to be located against recorded data before a gate is chosen. What this change buys is that the data exists to locate it with. No face is admitted unscored: the -1 sentinel is preserved rather than clamped, and a degenerate landmark fit is counted rather than silently dropped. Takes the VR-001 dump to schema_version 2. The bump is not for readers, which check for the datasets by name and replay a v1 dump unchanged; it is so a consumer can tell "never scored" from "scored zero", which is not recoverable from the arrays afterwards. TRACES: AR-028, AR-029, AR-030 | VR-001 | SR-002
319 lines
16 KiB
C++
319 lines
16 KiB
C++
#pragma once
|
||
/// TRACES: AR-028 | VR-001, VR-010 | PR-002
|
||
#include "types.hpp"
|
||
#include "config.hpp"
|
||
#include "gallery/embedder_stamp.hpp"
|
||
|
||
#include <H5Cpp.h>
|
||
|
||
#include <atomic>
|
||
#include <cstdint>
|
||
#include <iostream>
|
||
#include <optional>
|
||
#include <string>
|
||
#include <type_traits>
|
||
#include <vector>
|
||
|
||
// ── DumpProvenance ────────────────────────────────────────────────────────────
|
||
/// TRACES: VR-010 | PR-002
|
||
// Everything that determined a dump's *content*, read back tolerantly.
|
||
//
|
||
// Two dumps of the same film with different detector thresholds, a different
|
||
// `dense_scale`, or scene detection on versus off are different measurements of
|
||
// different things — but they are byte-shaped identically, so a consumer that
|
||
// mixes them gets a plausible number from an incoherent input. GR-004 closed the
|
||
// worst case (a cross-model replay, where every cosine is meaningless); this
|
||
// closes the rest.
|
||
//
|
||
// Every field is optional because dumps written before VR-010 lack the
|
||
// attributes. A missing field reads as *unknown*, never as a default — a
|
||
// silently-defaulted `detector_conf` is exactly the fabricated provenance the
|
||
// requirement exists to prevent ("a fixture whose provenance is unknown is worse
|
||
// than no fixture, because it will be trusted").
|
||
struct DumpProvenance {
|
||
// Model identity
|
||
std::optional<std::string> embedder_model; // GR-004
|
||
std::optional<std::string> embedder_sha256; // GR-004
|
||
std::optional<std::string> detector_model;
|
||
|
||
// Sampling
|
||
std::optional<std::string> movie;
|
||
std::optional<float> sample_fps;
|
||
std::optional<double> start_sec;
|
||
std::optional<double> end_sec; // -1 = to end of file
|
||
|
||
// Detection — what the run admitted into the dump
|
||
std::optional<float> detector_conf;
|
||
std::optional<float> detector_nms;
|
||
std::optional<float> min_face_px;
|
||
std::optional<int> max_faces; // 0 = uncapped (AR-003)
|
||
|
||
// Frame geometry
|
||
std::optional<float> dense_scale;
|
||
std::optional<float> bbox_upscale; // faces/bbox × this = original-resolution px
|
||
std::optional<float> cut_threshold;
|
||
|
||
// Scene detection. The reason this flag exists: `is_scene_boundary` is
|
||
// all-zero both when TransNetV2 found no boundaries and when it never ran,
|
||
// and no amount of staring at the array distinguishes them.
|
||
std::optional<bool> scene_detect;
|
||
|
||
// Downstream knob that shaped nothing in the dump but everything a replay is
|
||
// compared against — recorded so a sweep can be told apart from the baseline.
|
||
std::optional<float> track_assoc_min_prob;
|
||
};
|
||
|
||
// Read whatever provenance a dump carries. Never throws on a missing attribute;
|
||
// an old dump simply yields a DumpProvenance full of empty optionals.
|
||
inline DumpProvenance read_dump_provenance(const H5::H5File& f) {
|
||
DumpProvenance p;
|
||
auto str = [&](const char* n, std::optional<std::string>& out) {
|
||
if (!f.attrExists(n)) return;
|
||
// Written as a variable-length string, so the read must name the same
|
||
// type explicitly — the default would truncate to a fixed length.
|
||
H5::StrType vlen(H5::PredType::C_S1, H5T_VARIABLE);
|
||
std::string v;
|
||
f.openAttribute(n).read(vlen, v);
|
||
out = v;
|
||
};
|
||
auto num = [&](const char* n, const H5::PredType& dt, auto& out) {
|
||
if (!f.attrExists(n)) return;
|
||
typename std::decay_t<decltype(out)>::value_type v{};
|
||
f.openAttribute(n).read(dt, &v);
|
||
out = v;
|
||
};
|
||
|
||
str("embedder_model", p.embedder_model);
|
||
str("embedder_sha256", p.embedder_sha256);
|
||
str("detector_model", p.detector_model);
|
||
str("movie", p.movie);
|
||
|
||
num("sample_fps", H5::PredType::NATIVE_FLOAT, p.sample_fps);
|
||
num("start_sec", H5::PredType::NATIVE_DOUBLE, p.start_sec);
|
||
num("end_sec", H5::PredType::NATIVE_DOUBLE, p.end_sec);
|
||
num("detector_conf", H5::PredType::NATIVE_FLOAT, p.detector_conf);
|
||
num("detector_nms", H5::PredType::NATIVE_FLOAT, p.detector_nms);
|
||
num("min_face_px", H5::PredType::NATIVE_FLOAT, p.min_face_px);
|
||
num("max_faces", H5::PredType::NATIVE_INT, p.max_faces);
|
||
num("dense_scale", H5::PredType::NATIVE_FLOAT, p.dense_scale);
|
||
num("bbox_upscale", H5::PredType::NATIVE_FLOAT, p.bbox_upscale);
|
||
num("cut_threshold", H5::PredType::NATIVE_FLOAT, p.cut_threshold);
|
||
num("track_assoc_min_prob", H5::PredType::NATIVE_FLOAT, p.track_assoc_min_prob);
|
||
|
||
if (f.attrExists("scene_detect")) {
|
||
uint8_t v = 0;
|
||
f.openAttribute("scene_detect").read(H5::PredType::NATIVE_UINT8, &v);
|
||
p.scene_detect = (v != 0);
|
||
}
|
||
return p;
|
||
}
|
||
|
||
// ── EmbeddingDumpFunc ─────────────────────────────────────────────────────────
|
||
// KPN sink that taps the EmbeddedSceneFrame channel and writes the per-frame face
|
||
// metadata + embeddings to one HDF5 file (schema: scripts/optimizer/SCHEMA.md).
|
||
// The dump is the expensive, parameter-independent half of the pipeline
|
||
// (decode→detect→align→embed); replaying it lets a threshold sweep re-run the cheap
|
||
// downstream nodes thousands of times with no GPU. See sae_kpn / scripts/optimizer.
|
||
//
|
||
// Accumulates in flat/ragged arrays and writes once on EOF.
|
||
|
||
struct EmbeddingDumpFunc {
|
||
static constexpr std::string_view label() { return "embedding_dump"; }
|
||
|
||
EmbeddingDumpFunc(const Config& cfg, std::atomic<bool>& done)
|
||
: path_(cfg.dump_embeddings_path), movie_(cfg.movie_path),
|
||
sample_fps_(cfg.sample_fps), done_(done)
|
||
{
|
||
/// TRACES: GR-004 | SR-001
|
||
// A dump is a bag of embeddings with no model attached, replayed against a
|
||
// gallery hours or weeks later — the same silent cross-model hazard as the
|
||
// gallery itself, so it carries the same stamp.
|
||
stamp_ = make_embedder_stamp(cfg.arcface_model);
|
||
|
||
/// TRACES: VR-010 | PR-002
|
||
// The rest of what determined this file's content. Captured from the live
|
||
// Config at construction, so it describes the run that is being written
|
||
// rather than whatever config happens to be lying around at read time.
|
||
prov_.detector_model = basename_of(cfg.detector_model);
|
||
prov_.detector_conf = cfg.detector_conf;
|
||
prov_.detector_nms = cfg.detector_nms;
|
||
prov_.min_face_px = cfg.min_face_px;
|
||
prov_.max_faces = cfg.max_faces;
|
||
prov_.cut_threshold = cfg.cut_threshold;
|
||
prov_.dense_scale = cfg.dense_scale;
|
||
prov_.start_sec = cfg.start_sec;
|
||
prov_.end_sec = cfg.end_sec;
|
||
prov_.scene_detect = cfg.scene_detect;
|
||
prov_.track_assoc_min_prob = cfg.track_assoc_min_prob;
|
||
|
||
std::cerr << "[embedding_dump] writing " << path_
|
||
<< " embedder: " << stamp_.describe()
|
||
<< " detector: " << *prov_.detector_model
|
||
<< " @conf " << cfg.detector_conf
|
||
<< " scene_detect=" << (cfg.scene_detect ? "on" : "off") << "\n";
|
||
}
|
||
|
||
void operator()(EmbeddedSceneFrame ef) {
|
||
if (ef.source.eof) { flush(); return; }
|
||
|
||
/// TRACES: VR-010 | PR-002
|
||
// Taken from the frames themselves, not recomputed from dense_scale — the
|
||
// factor the source actually stamped on them is the one that maps
|
||
// faces/bbox back to original resolution, whatever rule produced it.
|
||
if (!prov_.bbox_upscale) prov_.bbox_upscale = ef.source.bbox_upscale;
|
||
|
||
const int32_t n = static_cast<int32_t>(ef.faces.size());
|
||
ts_.push_back(ef.source.timestamp_sec);
|
||
fidx_.push_back(ef.source.frame_idx);
|
||
is_cut_.push_back(ef.source.is_cut ? 1 : 0);
|
||
is_bnd_.push_back(ef.source.is_scene_boundary ? 1 : 0);
|
||
face_off_.push_back(static_cast<int64_t>(conf_.size()));
|
||
face_cnt_.push_back(n);
|
||
|
||
for (int i = 0; i < n; ++i) {
|
||
const auto& f = ef.faces[i];
|
||
bbox_.insert(bbox_.end(), {f.bbox.x, f.bbox.y, f.bbox.width, f.bbox.height});
|
||
for (int k = 0; k < 5; ++k) {
|
||
lmk_.push_back(f.landmarks[k].x);
|
||
lmk_.push_back(f.landmarks[k].y);
|
||
}
|
||
conf_.push_back(f.confidence);
|
||
/// TRACES: AR-028 | SR-002
|
||
// The quality vector, carried rather than consumed: written beside
|
||
// the embedding it describes so VR-012 can locate its knees against
|
||
// recorded data instead of by re-running video. Size is the third
|
||
// axis and is already here as bbox + the bbox_upscale attribute.
|
||
// Both are -1 only if a face reached the dump unscored, which the
|
||
// aligner does not allow — the sentinel is preserved rather than
|
||
// clamped so that a future path which did would be visible.
|
||
sharp_.push_back(f.sharpness);
|
||
resid_.push_back(f.alignment_residual);
|
||
const auto& e = ef.embeddings[i];
|
||
emb_.insert(emb_.end(), e.begin(), e.end());
|
||
}
|
||
}
|
||
|
||
void flush() {
|
||
if (written_.exchange(true)) return;
|
||
try {
|
||
write_hdf5();
|
||
} catch (const H5::Exception& e) {
|
||
std::cerr << "[embedding_dump] HDF5 error: " << e.getDetailMsg() << "\n";
|
||
}
|
||
done_.store(true, std::memory_order_release);
|
||
}
|
||
|
||
private:
|
||
// Root attributes are additive: schema_version stayed 1 across VR-010, because
|
||
// every reader takes attributes by name with a default (replay.py) or an
|
||
// existence check (read_dump_provenance), so an old dump loses nothing and a
|
||
// new dump breaks nothing. A bump is for a change to the *datasets*.
|
||
//
|
||
// v2 is that change: AR-028 adds faces/sharpness and faces/alignment_residual.
|
||
// The bump is not about readers — those check for the datasets by name, and a
|
||
// v1 dump still replays. It is so a *consumer of the quality vector* can tell
|
||
// "this film's faces were never scored" from "this film's faces scored zero",
|
||
// which is the same distinction scene_detect exists to make and is likewise
|
||
// not recoverable from the arrays. A v1 dump reports the vector as unknown;
|
||
// re-dump to acquire it, since nobody can assert after the fact how sharp a
|
||
// face was.
|
||
static constexpr int kSchemaVersion = 2;
|
||
static constexpr int kEmbedDim = 512;
|
||
|
||
static std::string basename_of(const std::string& path) {
|
||
const auto slash = path.find_last_of("/\\");
|
||
return slash == std::string::npos ? path : path.substr(slash + 1);
|
||
}
|
||
|
||
template<typename T>
|
||
void write_vec(H5::Group& g, const char* name, const std::vector<T>& v,
|
||
const H5::PredType& dtype, hsize_t cols = 0) {
|
||
hsize_t rows = cols ? v.size() / cols : v.size();
|
||
std::vector<hsize_t> dims = cols ? std::vector<hsize_t>{rows, cols}
|
||
: std::vector<hsize_t>{rows};
|
||
H5::DataSpace space(static_cast<int>(dims.size()), dims.data());
|
||
auto ds = g.createDataSet(name, dtype, space);
|
||
if (!v.empty()) ds.write(v.data(), dtype);
|
||
}
|
||
|
||
static void attr_str(H5::H5File& f, const char* name, const std::string& v) {
|
||
H5::StrType str(H5::PredType::C_S1, H5T_VARIABLE);
|
||
f.createAttribute(name, str, H5::DataSpace(H5S_SCALAR)).write(str, v);
|
||
}
|
||
|
||
template<typename T>
|
||
static void attr_num(H5::H5File& f, const char* name, const H5::PredType& dt, T v) {
|
||
f.createAttribute(name, dt, H5::DataSpace(H5S_SCALAR)).write(dt, &v);
|
||
}
|
||
|
||
void write_hdf5() {
|
||
H5::H5File file(path_, H5F_ACC_TRUNC);
|
||
|
||
// root attrs
|
||
attr_num(file, "schema_version", H5::PredType::NATIVE_INT, kSchemaVersion);
|
||
attr_num(file, "embed_dim", H5::PredType::NATIVE_INT, kEmbedDim);
|
||
attr_num(file, "sample_fps", H5::PredType::NATIVE_FLOAT, sample_fps_);
|
||
attr_str(file, "movie", movie_);
|
||
/// TRACES: GR-004 | SR-001
|
||
attr_str(file, "embedder_model", stamp_.model_name);
|
||
attr_str(file, "embedder_sha256", stamp_.model_sha256);
|
||
|
||
/// TRACES: VR-010 | PR-002
|
||
attr_str(file, "detector_model", prov_.detector_model.value_or(""));
|
||
attr_num(file, "detector_conf", H5::PredType::NATIVE_FLOAT, *prov_.detector_conf);
|
||
attr_num(file, "detector_nms", H5::PredType::NATIVE_FLOAT, *prov_.detector_nms);
|
||
attr_num(file, "min_face_px", H5::PredType::NATIVE_FLOAT, *prov_.min_face_px);
|
||
attr_num(file, "max_faces", H5::PredType::NATIVE_INT, *prov_.max_faces);
|
||
attr_num(file, "cut_threshold", H5::PredType::NATIVE_FLOAT, *prov_.cut_threshold);
|
||
attr_num(file, "dense_scale", H5::PredType::NATIVE_FLOAT, *prov_.dense_scale);
|
||
// Recorded, NOT applied — faces/bbox stays in the detector's own frame
|
||
// space so a replay feeds the tracker exactly what the live run fed it.
|
||
attr_num(file, "bbox_upscale", H5::PredType::NATIVE_FLOAT,
|
||
prov_.bbox_upscale.value_or(1.f));
|
||
attr_num(file, "start_sec", H5::PredType::NATIVE_DOUBLE, *prov_.start_sec);
|
||
attr_num(file, "end_sec", H5::PredType::NATIVE_DOUBLE, *prov_.end_sec);
|
||
attr_num(file, "track_assoc_min_prob", H5::PredType::NATIVE_FLOAT,
|
||
*prov_.track_assoc_min_prob);
|
||
// 0/1, matching the uint8 booleans in frames/. Tells "TransNetV2 found no
|
||
// boundaries" apart from "TransNetV2 never ran", which is/was the same
|
||
// all-zero is_scene_boundary array either way.
|
||
attr_num(file, "scene_detect", H5::PredType::NATIVE_UINT8,
|
||
static_cast<uint8_t>(*prov_.scene_detect ? 1 : 0));
|
||
|
||
H5::Group frames = file.createGroup("frames");
|
||
write_vec(frames, "timestamp_sec", ts_, H5::PredType::NATIVE_DOUBLE);
|
||
write_vec(frames, "frame_idx", fidx_, H5::PredType::NATIVE_INT64);
|
||
write_vec(frames, "is_cut", is_cut_, H5::PredType::NATIVE_UINT8);
|
||
write_vec(frames, "is_scene_boundary", is_bnd_, H5::PredType::NATIVE_UINT8);
|
||
write_vec(frames, "face_offset", face_off_, H5::PredType::NATIVE_INT64);
|
||
write_vec(frames, "face_count", face_cnt_, H5::PredType::NATIVE_INT32);
|
||
|
||
H5::Group faces = file.createGroup("faces");
|
||
write_vec(faces, "embedding", emb_, H5::PredType::NATIVE_FLOAT, kEmbedDim);
|
||
write_vec(faces, "bbox", bbox_, H5::PredType::NATIVE_FLOAT, 4);
|
||
write_vec(faces, "landmarks", lmk_, H5::PredType::NATIVE_FLOAT, 10);
|
||
write_vec(faces, "confidence", conf_, H5::PredType::NATIVE_FLOAT);
|
||
/// TRACES: AR-028 | SR-002
|
||
write_vec(faces, "sharpness", sharp_, H5::PredType::NATIVE_FLOAT);
|
||
write_vec(faces, "alignment_residual", resid_, H5::PredType::NATIVE_FLOAT);
|
||
|
||
std::cerr << "[embedding_dump] wrote " << ts_.size() << " frames, "
|
||
<< conf_.size() << " faces → " << path_ << "\n";
|
||
}
|
||
|
||
std::string path_, movie_;
|
||
EmbedderStamp stamp_;
|
||
DumpProvenance prov_;
|
||
float sample_fps_;
|
||
std::atomic<bool>& done_;
|
||
std::atomic<bool> written_{false};
|
||
|
||
std::vector<double> ts_;
|
||
std::vector<int64_t> fidx_;
|
||
std::vector<uint8_t> is_cut_, is_bnd_;
|
||
std::vector<int64_t> face_off_;
|
||
std::vector<int32_t> face_cnt_;
|
||
std::vector<float> emb_, bbox_, lmk_, conf_;
|
||
std::vector<float> sharp_, resid_; // AR-028 quality vector, parallel to conf_
|
||
};
|