Files
scene-actor-extraction/src/nodes/embedding_dump_node.hpp
T
dtourolleandClaude Opus 5 b35d49c772 docs: tag the implemented core with its requirement IDs
Adds TRACES tags to code that already satisfies a Done requirement, so coverage
reflects what exists rather than starting from zero:

AR-001 face detection, AR-005 ArcFace alignment, AR-023 calibration fit,
DP-001/DP-002 the single analysis core behind the CLI, IR-001 truth-file
emission, IR-006 the Jellyfin round trip, GR-001/GR-002 gallery build and
incremental merge, VR-001 the embedding dump, VR-002 replay through the real
nodes, VR-003 per-second scoring.

Only Done requirements are tagged. A tag on Planned work would inflate coverage
with fiction that looks plausible — the same failure family as a gate that
cannot fail, and harder to spot.

GR-005 (gallery never leaves the instance) stays untagged deliberately: it is a
prohibition satisfied by the absence of an egress path, so there is no unit that
decides it. Same shape as PR-005 in the system spec, which has no software row
for the same reason. A goal held only by prohibitions cannot be verified by
pointing at code.

Coverage 5/63 to 14/63. The three VR tags are reported as tagged-but-unexecuted
and excluded from the numerator, since their tier cannot run on the CI host —
tagging deliberately cannot raise the number on its own.

Suite still 64 cases, 3199 assertions.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

TRACES: AR-001, AR-005, AR-023, DP-001, DP-002, IR-001, IR-006, GR-001, GR-002, VR-001, VR-002, VR-003
2026-07-30 21:20:27 +02:00

137 lines
5.9 KiB
C++

#pragma once
/// TRACES: VR-001 | PR-002
#include "types.hpp"
#include "config.hpp"
#include "gallery/embedder_stamp.hpp"
#include <H5Cpp.h>
#include <cstdint>
#include <iostream>
#include <string>
#include <vector>
// ── EmbeddingDumpFunc ─────────────────────────────────────────────────────────
// KPN sink that taps the EmbeddedSceneFrame channel and writes the per-frame face
// metadata + embeddings to one HDF5 file (schema: scripts/optimizer/SCHEMA.md).
// The dump is the expensive, parameter-independent half of the pipeline
// (decode→detect→align→embed); replaying it lets a threshold sweep re-run the cheap
// downstream nodes thousands of times with no GPU. See sae_kpn / scripts/optimizer.
//
// Accumulates in flat/ragged arrays and writes once on EOF.
struct EmbeddingDumpFunc {
static constexpr std::string_view label() { return "embedding_dump"; }
EmbeddingDumpFunc(const Config& cfg, std::atomic<bool>& done)
: path_(cfg.dump_embeddings_path), movie_(cfg.movie_path),
sample_fps_(cfg.sample_fps), done_(done)
{
/// TRACES: GR-004 | SR-001
// A dump is a bag of embeddings with no model attached, replayed against a
// gallery hours or weeks later — the same silent cross-model hazard as the
// gallery itself, so it carries the same stamp.
stamp_ = make_embedder_stamp(cfg.arcface_model);
std::cerr << "[embedding_dump] writing " << path_
<< " embedder: " << stamp_.describe() << "\n";
}
void operator()(EmbeddedSceneFrame ef) {
if (ef.source.eof) { flush(); return; }
const int32_t n = static_cast<int32_t>(ef.faces.size());
ts_.push_back(ef.source.timestamp_sec);
fidx_.push_back(ef.source.frame_idx);
is_cut_.push_back(ef.source.is_cut ? 1 : 0);
is_bnd_.push_back(ef.source.is_scene_boundary ? 1 : 0);
face_off_.push_back(static_cast<int64_t>(conf_.size()));
face_cnt_.push_back(n);
for (int i = 0; i < n; ++i) {
const auto& f = ef.faces[i];
bbox_.insert(bbox_.end(), {f.bbox.x, f.bbox.y, f.bbox.width, f.bbox.height});
for (int k = 0; k < 5; ++k) {
lmk_.push_back(f.landmarks[k].x);
lmk_.push_back(f.landmarks[k].y);
}
conf_.push_back(f.confidence);
const auto& e = ef.embeddings[i];
emb_.insert(emb_.end(), e.begin(), e.end());
}
}
void flush() {
if (written_.exchange(true)) return;
try {
write_hdf5();
} catch (const H5::Exception& e) {
std::cerr << "[embedding_dump] HDF5 error: " << e.getDetailMsg() << "\n";
}
done_.store(true, std::memory_order_release);
}
private:
static constexpr int kSchemaVersion = 1;
static constexpr int kEmbedDim = 512;
template<typename T>
void write_vec(H5::Group& g, const char* name, const std::vector<T>& v,
const H5::PredType& dtype, hsize_t cols = 0) {
hsize_t rows = cols ? v.size() / cols : v.size();
std::vector<hsize_t> dims = cols ? std::vector<hsize_t>{rows, cols}
: std::vector<hsize_t>{rows};
H5::DataSpace space(static_cast<int>(dims.size()), dims.data());
auto ds = g.createDataSet(name, dtype, space);
if (!v.empty()) ds.write(v.data(), dtype);
}
void write_hdf5() {
H5::H5File file(path_, H5F_ACC_TRUNC);
// root attrs
auto scalar = H5::DataSpace(H5S_SCALAR);
auto ver = file.createAttribute("schema_version", H5::PredType::NATIVE_INT, scalar);
int sv = kSchemaVersion; ver.write(H5::PredType::NATIVE_INT, &sv);
auto ed = file.createAttribute("embed_dim", H5::PredType::NATIVE_INT, scalar);
int dim = kEmbedDim; ed.write(H5::PredType::NATIVE_INT, &dim);
auto fps = file.createAttribute("sample_fps", H5::PredType::NATIVE_FLOAT, scalar);
fps.write(H5::PredType::NATIVE_FLOAT, &sample_fps_);
H5::StrType str(H5::PredType::C_S1, H5T_VARIABLE);
auto mv = file.createAttribute("movie", str, scalar);
mv.write(str, movie_);
/// TRACES: GR-004 | SR-001
file.createAttribute("embedder_model", str, scalar).write(str, stamp_.model_name);
file.createAttribute("embedder_sha256", str, scalar).write(str, stamp_.model_sha256);
H5::Group frames = file.createGroup("frames");
write_vec(frames, "timestamp_sec", ts_, H5::PredType::NATIVE_DOUBLE);
write_vec(frames, "frame_idx", fidx_, H5::PredType::NATIVE_INT64);
write_vec(frames, "is_cut", is_cut_, H5::PredType::NATIVE_UINT8);
write_vec(frames, "is_scene_boundary", is_bnd_, H5::PredType::NATIVE_UINT8);
write_vec(frames, "face_offset", face_off_, H5::PredType::NATIVE_INT64);
write_vec(frames, "face_count", face_cnt_, H5::PredType::NATIVE_INT32);
H5::Group faces = file.createGroup("faces");
write_vec(faces, "embedding", emb_, H5::PredType::NATIVE_FLOAT, kEmbedDim);
write_vec(faces, "bbox", bbox_, H5::PredType::NATIVE_FLOAT, 4);
write_vec(faces, "landmarks", lmk_, H5::PredType::NATIVE_FLOAT, 10);
write_vec(faces, "confidence", conf_, H5::PredType::NATIVE_FLOAT);
std::cerr << "[embedding_dump] wrote " << ts_.size() << " frames, "
<< conf_.size() << " faces → " << path_ << "\n";
}
std::string path_, movie_;
EmbedderStamp stamp_;
float sample_fps_;
std::atomic<bool>& done_;
std::atomic<bool> written_{false};
std::vector<double> ts_;
std::vector<int64_t> fidx_;
std::vector<uint8_t> is_cut_, is_bnd_;
std::vector<int64_t> face_off_;
std::vector<int32_t> face_cnt_;
std::vector<float> emb_, bbox_, lmk_, conf_;
};