Initial commit: scene-actor-extraction pipeline

Source (KPN++ pipeline nodes, ArcFace embedders, SCRFD/YuNet detectors,
gallery builder), build scripts, and eval artifacts.

- external/KPN as a git submodule (gitea.tourolle.paris/dtourolle/KPN)
- ONNX models tracked via Git LFS (models/*.onnx)
- generated outputs, TensorRT engines, reference repos, and media ignored
This commit is contained in:
2026-06-12 15:29:01 +02:00
commit d753062c6c
50 changed files with 10100 additions and 0 deletions
+115
View File
@@ -0,0 +1,115 @@
#include "gallery_builder.hpp"
#include "arcface_embedder.hpp"
#include "face_utils.hpp"
#include "ort_provider.hpp"
#include "scrfd_decoder.hpp"
#include <opencv2/imgcodecs.hpp>
#include <algorithm>
#include <filesystem>
#include <iostream>
#include <stdexcept>
#include <string>
namespace fs = std::filesystem;
// ── Parse "nm0000093_Brad_Pitt" → ("nm0000093", "Brad Pitt") ─────────────────
static std::pair<std::string, std::string> parse_dir_name(const std::string& dirname) {
auto pos = dirname.find('_');
if (pos == std::string::npos) return {dirname, dirname};
std::string imdb_id = dirname.substr(0, pos);
std::string raw = dirname.substr(pos + 1);
std::string name;
name.reserve(raw.size());
for (char c : raw)
name += (c == '_' ? ' ' : c);
return {imdb_id, name};
}
// ── Public API ────────────────────────────────────────────────────────────────
ActorGallery build_gallery(const BuildConfig& cfg) {
const OrtProvider provider = detect_ort_provider();
std::cerr << "[build_gallery] inference provider: " << provider_name(provider) << "\n";
SCRFDDecoder decoder(cfg.detector_model, cfg.detector_conf, cfg.detector_nms, provider);
ArcFaceEmbedder arcface(cfg.arcface_model, provider);
ActorGallery gallery;
for (const auto& actor_dir : fs::directory_iterator(cfg.gallery_root)) {
if (!actor_dir.is_directory()) continue;
auto [imdb_id, name] = parse_dir_name(actor_dir.path().filename().string());
std::cerr << "[build_gallery] " << name << " (" << imdb_id << ")\n";
ActorGallery::Actor actor;
actor.imdb_id = imdb_id;
actor.name = name;
static const std::vector<std::string> kExts{".jpg", ".jpeg", ".png", ".webp"};
for (const auto& img_file : fs::directory_iterator(actor_dir.path())) {
if (!img_file.is_regular_file()) continue;
std::string ext = img_file.path().extension().string();
std::transform(ext.begin(), ext.end(), ext.begin(), ::tolower);
if (std::find(kExts.begin(), kExts.end(), ext) == kExts.end()) continue;
cv::Mat img = cv::imread(img_file.path().string());
if (img.empty()) {
std::cerr << " [skip] cannot read " << img_file.path().filename() << "\n";
continue;
}
if (cfg.max_side > 0) {
const int big = std::max(img.cols, img.rows);
if (big > cfg.max_side) {
const double s = static_cast<double>(cfg.max_side) / big;
cv::resize(img, img, {}, s, s, cv::INTER_AREA);
}
}
auto faces = decoder.detect(img);
if (faces.empty()) {
std::cerr << " [skip] no face: " << img_file.path().filename() << "\n";
continue;
}
if (faces.size() > 1) {
std::cerr << " [warn] " << faces.size() << " faces, using highest confidence: "
<< img_file.path().filename() << "\n";
}
const auto& best = *std::max_element(
faces.begin(), faces.end(),
[](const DetectedFace& a, const DetectedFace& b) {
return a.confidence < b.confidence;
});
cv::Mat crop = align_face(img, best.landmarks);
if (crop.empty()) {
std::cerr << " [skip] alignment failed: " << img_file.path().filename() << "\n";
continue;
}
Embedding emb = arcface.embed_one(crop);
actor.embeddings.push_back(emb);
actor.source_images.push_back(img_file.path().filename().string());
std::cerr << " [ok] " << img_file.path().filename()
<< " conf=" << best.confidence << "\n";
}
if (actor.embeddings.empty()) {
std::cerr << " [warn] no valid embeddings for " << name << " — skipped\n";
continue;
}
std::cerr << "" << actor.embeddings.size() << " embeddings\n";
gallery.actors.push_back(std::move(actor));
}
std::cerr << "[build_gallery] total: " << gallery.actors.size() << " actors\n";
return gallery;
}