From a667caa3139fc14e76536764dbee3e99c7d92195 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Fri, 31 Jul 2026 15:24:06 +0200 Subject: [PATCH] feat: provenance attributes on the embedding dump MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit VR-010 — a dump made with one detector/embedder pair was byte-indistinguishable from one made with another, except for the two attributes GR-004 added. Replayed against a gallery from a different model, cosine similarities are meaningless but look entirely plausible. The register states the principle directly: a fixture whose provenance is unknown is worse than no fixture, because it will be trusted. Sixteen attributes now record everything that determines the dump's content: detector model and thresholds, min_face_px, max_faces, cut_threshold, dense_scale, bbox_upscale, start/end, track_assoc_min_prob, and scene_detect. scene_detect is the one that matters most. is_scene_boundary is all-zero both when the detector found nothing and when it never ran, and those mean completely different things to a consumer — without the flag they are indistinguishable. No schema_version bump: new root attributes are additive and replay.py already reads attributes with a default, so older dumps stay readable and the committed fixtures — which predate this — still load. Also corrects SCHEMA.md, which claimed bbox was already mapped to original resolution at dump time. It is not; the upscale is applied downstream in the matcher, after the dump tap. Harmless while dense_scale is 1 and silently wrong otherwise, so bbox_upscale is now recorded and the doc says what the code does. Verified end to end: all sixteen attributes present and correct on a freshly generated dump. Suite: 92 cases, 6136 assertions. Co-Authored-By: Claude Opus 5 TRACES: VR-010, VR-001 | PR-002 --- docs/SPEC.md | 8 +- scripts/optimizer/SCHEMA.md | 96 +++++++++++++-- src/nodes/embedding_dump_node.hpp | 187 +++++++++++++++++++++++++++--- 3 files changed, 266 insertions(+), 25 deletions(-) diff --git a/docs/SPEC.md b/docs/SPEC.md index c1f9358..0cd62ae 100644 --- a/docs/SPEC.md +++ b/docs/SPEC.md @@ -1360,9 +1360,11 @@ Persist pipeline state at the point where the expensive work ends. variable-length HDF5 types and reads straight into numpy. - Stores per frame: `timestamp_sec`, `frame_idx`, `is_cut`, `is_scene_boundary`. Per face: `embedding` [N,512], `bbox` [N,4], `landmarks` [N,10], `confidence`. -- Invariants: embeddings unit-norm; `face_offset` contiguous; bboxes already in - original resolution; frames with no faces still get a row so timestamps stay - dense; EOF sentinels not written. +- Invariants: embeddings unit-norm; `face_offset` contiguous; bboxes and + landmarks in **decoded-frame** pixels with `bbox_upscale` recorded alongside + (the dump is a faithful tap, so it does not transform what the tracker saw — + see VR-010); frames with no faces still get a row so timestamps stay dense; + EOF sentinels not written. - Enabled by `--dump-embeddings out.h5`; teeing must not perturb the live result. Schema owned by [`scripts/optimizer/SCHEMA.md`](../scripts/optimizer/SCHEMA.md). diff --git a/scripts/optimizer/SCHEMA.md b/scripts/optimizer/SCHEMA.md index 105cfca..e710f79 100644 --- a/scripts/optimizer/SCHEMA.md +++ b/scripts/optimizer/SCHEMA.md @@ -19,11 +19,32 @@ variable-length HDF5 types and reads straight into numpy. / (root) attrs: schema_version : int = 1 - movie : str (source video path) - sample_fps : float embed_dim : int = 512 - embedder_model : str basename of the embedding model (GR-004) - embedder_sha256: str SHA-256 of that model file (GR-004) + + # ── what produced the vectors (GR-004) ────────────────────────────────── + embedder_model : str basename of the embedding model + embedder_sha256: str SHA-256 of that model file + + # ── what produced the faces (VR-010) ──────────────────────────────────── + detector_model : str basename of the detector .onnx + detector_conf : float score floor a detection had to clear to be dumped + detector_nms : float NMS IoU threshold + min_face_px : float minimum box side, ORIGINAL-resolution px (AR-002) + max_faces : int per-frame cap; 0 = uncapped, the default (AR-003) + + # ── what produced the frames (VR-010) ─────────────────────────────────── + movie : str source video path + sample_fps : float frames analysed per second of movie + start_sec : float seek point + end_sec : float stop point; -1 = end of file + cut_threshold : float histogram correlation below which is_cut fires + dense_scale : float decoded-frame downscale in dense mode; 1 = off + bbox_upscale : float multiply faces/bbox and faces/landmarks by this to + reach original video pixels; 1 when dense_scale is 1 + scene_detect : uint8 0/1 — was TransNetV2 running at all (see below) + + # ── downstream setting recorded for comparability (VR-010) ────────────── + track_assoc_min_prob : float the run's tracker admission probability frames/ group — one row per sampled frame timestamp_sec : float64 [F] @@ -35,14 +56,47 @@ variable-length HDF5 types and reads straight into numpy. faces/ group — one row per detected face, concatenated embedding : float32 [N, 512] L2-normalised ArcFace embedding - bbox : float32 [N, 4] x, y, w, h in original video pixels - landmarks : float32 [N, 10] 5 (x,y) pairs, SCRFD/ArcFace order + bbox : float32 [N, 4] x, y, w, h in DECODED-frame pixels + landmarks : float32 [N, 10] 5 (x,y) pairs, SCRFD/ArcFace order, + same space as bbox confidence : float32 [N] detector confidence ``` `F` = number of sampled frames, `N` = total faces (= sum of face_count). Frame *i*'s faces are `faces/*[ face_offset[i] : face_offset[i]+face_count[i] ]`. +## Provenance (VR-010) + +The attributes above are not documentation; they are the only thing that makes a +dump interpretable. Two dumps of the same film at `detector_conf` 0.5 and 0.7, or +at `dense_scale` 1.0 and 0.5, or with scene detection on and off, are different +measurements of different things — and they are byte-shaped identically. Without +provenance a consumer that mixes them gets a plausible number from an incoherent +input, and nothing anywhere reports a problem. + +**`scene_detect` is the one that cannot be inferred.** `is_scene_boundary` is +all-zero both when TransNetV2 found no boundaries in the clip and when it was +never enabled, and those mean opposite things: the first says *this footage has +no shot changes*, the second says *nobody looked*. A consumer that reads the +array alone must guess. The flag is what removes the guess. (`dump_embeddings` +has no `--scene-detect`, so every dump it writes records `false` — which is +exactly the fact the committed fixtures needed to state.) + +**`bbox_upscale` is recorded, not applied.** See the coordinate-space note below. + +Reading is by name with a default or an existence check on **both** sides — +`replay.py` (`f.attrs.get(...)`) and `read_dump_provenance()` in +`src/nodes/embedding_dump_node.hpp` (`attrExists`). So the attributes are +additive and `schema_version` stays 1: a pre-VR-010 dump still loads, and a +post-VR-010 dump still reads on old code. + +A missing attribute means **unknown**, never a default value. Substituting +`detector_conf = 0.5` for a dump that does not say so manufactures the provenance +the requirement exists to prevent — per `docs/requirements.md`, *"a fixture whose +provenance is unknown is worse than no fixture, because it will be trusted."* +The committed `tests/fixtures/dumps/*.h5` predate VR-010 and carry none of these +attributes; re-dump to bind them, as with GR-004. + ## Model binding (GR-004) `embedder_model` / `embedder_sha256` record which embedder produced every vector @@ -57,10 +111,36 @@ warning, or a hard error under `SAE_REQUIRE_GALLERY_STAMP=1`) rather than as a pass. Re-dump to bind an old dump; there is no in-place migration, because unlike a gallery nobody can assert after the fact which model produced a vector. +## Coordinate space — `bbox`, `landmarks`, `bbox_upscale` + +`bbox` and `landmarks` are in **decoded-frame pixels**: exactly the numbers SCRFD +produced, untransformed. To reach original video pixels, multiply by +`bbox_upscale`. With `dense_scale == 1` (the default, and every committed +fixture) `bbox_upscale == 1` and the two spaces coincide. + +> Earlier revisions of this document claimed the upscale was applied at dump time. +> It never was. `embedding_dump_node.hpp` writes `f.bbox` raw; the upscale lives +> in `identity_matcher_node.hpp`, which is *downstream* of the dump tap. The +> claim was harmless only because `dense_scale` was 1 in practice. + +The fix is to record the factor rather than to apply it, because the dump's whole +contract is to be a **faithful tap** at the `EmbeddedSceneFrame` channel — VR-002 +requires replay to drive the real nodes, and a replay is only equivalent to the +live run if the tracker is fed the geometry the live tracker saw. Rescaling at +the tap would break that: the replayed tracker would associate on boxes the live +one never received. Two further reasons: + +- The matcher's upscale is applied to `bbox` **only**, not to `landmarks`. + Pre-multiplying at the tap would leave the two arrays in different coordinate + spaces inside one file — a worse trap than the one being fixed. +- Pre-multiplying is lossy in the sense that matters: a dump that had been + upscaled would be indistinguishable from one taken at `dense_scale == 1`, so + you would have to record `bbox_upscale` anyway to know which you were holding. + ## Invariants - `embedding` rows are unit-norm (cosine == dot product against the gallery). - `face_offset[0] == 0`; `face_offset[i+1] == face_offset[i] + face_count[i]`. -- `bbox` is already mapped to original resolution (bbox_upscale applied at dump time), - matching what the identity matcher would emit. +- `bbox` and `landmarks` share one coordinate space; `bbox_upscale` maps both to + original resolution (see above). - A frame with no faces has `face_count == 0` (still gets a row, so timestamps stay dense). - EOF sentinel frames are NOT written. diff --git a/src/nodes/embedding_dump_node.hpp b/src/nodes/embedding_dump_node.hpp index 91ae7a5..969865d 100644 --- a/src/nodes/embedding_dump_node.hpp +++ b/src/nodes/embedding_dump_node.hpp @@ -1,16 +1,113 @@ #pragma once -/// TRACES: VR-001 | PR-002 +/// TRACES: VR-001, VR-010 | PR-002 #include "types.hpp" #include "config.hpp" #include "gallery/embedder_stamp.hpp" #include +#include #include #include +#include #include +#include #include +// ── DumpProvenance ──────────────────────────────────────────────────────────── +/// TRACES: VR-010 | PR-002 +// Everything that determined a dump's *content*, read back tolerantly. +// +// Two dumps of the same film with different detector thresholds, a different +// `dense_scale`, or scene detection on versus off are different measurements of +// different things — but they are byte-shaped identically, so a consumer that +// mixes them gets a plausible number from an incoherent input. GR-004 closed the +// worst case (a cross-model replay, where every cosine is meaningless); this +// closes the rest. +// +// Every field is optional because dumps written before VR-010 lack the +// attributes. A missing field reads as *unknown*, never as a default — a +// silently-defaulted `detector_conf` is exactly the fabricated provenance the +// requirement exists to prevent ("a fixture whose provenance is unknown is worse +// than no fixture, because it will be trusted"). +struct DumpProvenance { + // Model identity + std::optional embedder_model; // GR-004 + std::optional embedder_sha256; // GR-004 + std::optional detector_model; + + // Sampling + std::optional movie; + std::optional sample_fps; + std::optional start_sec; + std::optional end_sec; // -1 = to end of file + + // Detection — what the run admitted into the dump + std::optional detector_conf; + std::optional detector_nms; + std::optional min_face_px; + std::optional max_faces; // 0 = uncapped (AR-003) + + // Frame geometry + std::optional dense_scale; + std::optional bbox_upscale; // faces/bbox × this = original-resolution px + std::optional cut_threshold; + + // Scene detection. The reason this flag exists: `is_scene_boundary` is + // all-zero both when TransNetV2 found no boundaries and when it never ran, + // and no amount of staring at the array distinguishes them. + std::optional scene_detect; + + // Downstream knob that shaped nothing in the dump but everything a replay is + // compared against — recorded so a sweep can be told apart from the baseline. + std::optional track_assoc_min_prob; +}; + +// Read whatever provenance a dump carries. Never throws on a missing attribute; +// an old dump simply yields a DumpProvenance full of empty optionals. +inline DumpProvenance read_dump_provenance(const H5::H5File& f) { + DumpProvenance p; + auto str = [&](const char* n, std::optional& out) { + if (!f.attrExists(n)) return; + // Written as a variable-length string, so the read must name the same + // type explicitly — the default would truncate to a fixed length. + H5::StrType vlen(H5::PredType::C_S1, H5T_VARIABLE); + std::string v; + f.openAttribute(n).read(vlen, v); + out = v; + }; + auto num = [&](const char* n, const H5::PredType& dt, auto& out) { + if (!f.attrExists(n)) return; + typename std::decay_t::value_type v{}; + f.openAttribute(n).read(dt, &v); + out = v; + }; + + str("embedder_model", p.embedder_model); + str("embedder_sha256", p.embedder_sha256); + str("detector_model", p.detector_model); + str("movie", p.movie); + + num("sample_fps", H5::PredType::NATIVE_FLOAT, p.sample_fps); + num("start_sec", H5::PredType::NATIVE_DOUBLE, p.start_sec); + num("end_sec", H5::PredType::NATIVE_DOUBLE, p.end_sec); + num("detector_conf", H5::PredType::NATIVE_FLOAT, p.detector_conf); + num("detector_nms", H5::PredType::NATIVE_FLOAT, p.detector_nms); + num("min_face_px", H5::PredType::NATIVE_FLOAT, p.min_face_px); + num("max_faces", H5::PredType::NATIVE_INT, p.max_faces); + num("dense_scale", H5::PredType::NATIVE_FLOAT, p.dense_scale); + num("bbox_upscale", H5::PredType::NATIVE_FLOAT, p.bbox_upscale); + num("cut_threshold", H5::PredType::NATIVE_FLOAT, p.cut_threshold); + num("track_assoc_min_prob", H5::PredType::NATIVE_FLOAT, p.track_assoc_min_prob); + + if (f.attrExists("scene_detect")) { + uint8_t v = 0; + f.openAttribute("scene_detect").read(H5::PredType::NATIVE_UINT8, &v); + p.scene_detect = (v != 0); + } + return p; +} + // ── EmbeddingDumpFunc ───────────────────────────────────────────────────────── // KPN sink that taps the EmbeddedSceneFrame channel and writes the per-frame face // metadata + embeddings to one HDF5 file (schema: scripts/optimizer/SCHEMA.md). @@ -32,13 +129,39 @@ struct EmbeddingDumpFunc { // gallery hours or weeks later — the same silent cross-model hazard as the // gallery itself, so it carries the same stamp. stamp_ = make_embedder_stamp(cfg.arcface_model); + + /// TRACES: VR-010 | PR-002 + // The rest of what determined this file's content. Captured from the live + // Config at construction, so it describes the run that is being written + // rather than whatever config happens to be lying around at read time. + prov_.detector_model = basename_of(cfg.detector_model); + prov_.detector_conf = cfg.detector_conf; + prov_.detector_nms = cfg.detector_nms; + prov_.min_face_px = cfg.min_face_px; + prov_.max_faces = cfg.max_faces; + prov_.cut_threshold = cfg.cut_threshold; + prov_.dense_scale = cfg.dense_scale; + prov_.start_sec = cfg.start_sec; + prov_.end_sec = cfg.end_sec; + prov_.scene_detect = cfg.scene_detect; + prov_.track_assoc_min_prob = cfg.track_assoc_min_prob; + std::cerr << "[embedding_dump] writing " << path_ - << " embedder: " << stamp_.describe() << "\n"; + << " embedder: " << stamp_.describe() + << " detector: " << *prov_.detector_model + << " @conf " << cfg.detector_conf + << " scene_detect=" << (cfg.scene_detect ? "on" : "off") << "\n"; } void operator()(EmbeddedSceneFrame ef) { if (ef.source.eof) { flush(); return; } + /// TRACES: VR-010 | PR-002 + // Taken from the frames themselves, not recomputed from dense_scale — the + // factor the source actually stamped on them is the one that maps + // faces/bbox back to original resolution, whatever rule produced it. + if (!prov_.bbox_upscale) prov_.bbox_upscale = ef.source.bbox_upscale; + const int32_t n = static_cast(ef.faces.size()); ts_.push_back(ef.source.timestamp_sec); fidx_.push_back(ef.source.frame_idx); @@ -71,9 +194,18 @@ struct EmbeddingDumpFunc { } private: + // Root attributes are additive: schema_version stays 1 across VR-010, because + // every reader takes attributes by name with a default (replay.py) or an + // existence check (read_dump_provenance), so an old dump loses nothing and a + // new dump breaks nothing. A bump is for a change to the *datasets*. static constexpr int kSchemaVersion = 1; static constexpr int kEmbedDim = 512; + static std::string basename_of(const std::string& path) { + const auto slash = path.find_last_of("/\\"); + return slash == std::string::npos ? path : path.substr(slash + 1); + } + template void write_vec(H5::Group& g, const char* name, const std::vector& v, const H5::PredType& dtype, hsize_t cols = 0) { @@ -85,23 +217,49 @@ private: if (!v.empty()) ds.write(v.data(), dtype); } + static void attr_str(H5::H5File& f, const char* name, const std::string& v) { + H5::StrType str(H5::PredType::C_S1, H5T_VARIABLE); + f.createAttribute(name, str, H5::DataSpace(H5S_SCALAR)).write(str, v); + } + + template + static void attr_num(H5::H5File& f, const char* name, const H5::PredType& dt, T v) { + f.createAttribute(name, dt, H5::DataSpace(H5S_SCALAR)).write(dt, &v); + } + void write_hdf5() { H5::H5File file(path_, H5F_ACC_TRUNC); // root attrs - auto scalar = H5::DataSpace(H5S_SCALAR); - auto ver = file.createAttribute("schema_version", H5::PredType::NATIVE_INT, scalar); - int sv = kSchemaVersion; ver.write(H5::PredType::NATIVE_INT, &sv); - auto ed = file.createAttribute("embed_dim", H5::PredType::NATIVE_INT, scalar); - int dim = kEmbedDim; ed.write(H5::PredType::NATIVE_INT, &dim); - auto fps = file.createAttribute("sample_fps", H5::PredType::NATIVE_FLOAT, scalar); - fps.write(H5::PredType::NATIVE_FLOAT, &sample_fps_); - H5::StrType str(H5::PredType::C_S1, H5T_VARIABLE); - auto mv = file.createAttribute("movie", str, scalar); - mv.write(str, movie_); + attr_num(file, "schema_version", H5::PredType::NATIVE_INT, kSchemaVersion); + attr_num(file, "embed_dim", H5::PredType::NATIVE_INT, kEmbedDim); + attr_num(file, "sample_fps", H5::PredType::NATIVE_FLOAT, sample_fps_); + attr_str(file, "movie", movie_); /// TRACES: GR-004 | SR-001 - file.createAttribute("embedder_model", str, scalar).write(str, stamp_.model_name); - file.createAttribute("embedder_sha256", str, scalar).write(str, stamp_.model_sha256); + attr_str(file, "embedder_model", stamp_.model_name); + attr_str(file, "embedder_sha256", stamp_.model_sha256); + + /// TRACES: VR-010 | PR-002 + attr_str(file, "detector_model", prov_.detector_model.value_or("")); + attr_num(file, "detector_conf", H5::PredType::NATIVE_FLOAT, *prov_.detector_conf); + attr_num(file, "detector_nms", H5::PredType::NATIVE_FLOAT, *prov_.detector_nms); + attr_num(file, "min_face_px", H5::PredType::NATIVE_FLOAT, *prov_.min_face_px); + attr_num(file, "max_faces", H5::PredType::NATIVE_INT, *prov_.max_faces); + attr_num(file, "cut_threshold", H5::PredType::NATIVE_FLOAT, *prov_.cut_threshold); + attr_num(file, "dense_scale", H5::PredType::NATIVE_FLOAT, *prov_.dense_scale); + // Recorded, NOT applied — faces/bbox stays in the detector's own frame + // space so a replay feeds the tracker exactly what the live run fed it. + attr_num(file, "bbox_upscale", H5::PredType::NATIVE_FLOAT, + prov_.bbox_upscale.value_or(1.f)); + attr_num(file, "start_sec", H5::PredType::NATIVE_DOUBLE, *prov_.start_sec); + attr_num(file, "end_sec", H5::PredType::NATIVE_DOUBLE, *prov_.end_sec); + attr_num(file, "track_assoc_min_prob", H5::PredType::NATIVE_FLOAT, + *prov_.track_assoc_min_prob); + // 0/1, matching the uint8 booleans in frames/. Tells "TransNetV2 found no + // boundaries" apart from "TransNetV2 never ran", which is/was the same + // all-zero is_scene_boundary array either way. + attr_num(file, "scene_detect", H5::PredType::NATIVE_UINT8, + static_cast(*prov_.scene_detect ? 1 : 0)); H5::Group frames = file.createGroup("frames"); write_vec(frames, "timestamp_sec", ts_, H5::PredType::NATIVE_DOUBLE); @@ -123,6 +281,7 @@ private: std::string path_, movie_; EmbedderStamp stamp_; + DumpProvenance prov_; float sample_fps_; std::atomic& done_; std::atomic written_{false};