Initial commit: scene-actor-extraction pipeline

Source (KPN++ pipeline nodes, ArcFace embedders, SCRFD/YuNet detectors,
gallery builder), build scripts, and eval artifacts.

- external/KPN as a git submodule (gitea.tourolle.paris/dtourolle/KPN)
- ONNX models tracked via Git LFS (models/*.onnx)
- generated outputs, TensorRT engines, reference repos, and media ignored
This commit is contained in:
2026-06-12 15:29:01 +02:00
commit d753062c6c
50 changed files with 10100 additions and 0 deletions
+251
View File
@@ -0,0 +1,251 @@
// embed_faces — run SCRFD-500MF + ArcFace on a list of image files and write
// embeddings as JSON to stdout.
//
// Usage:
// embed_faces [--detector <path>] [--arcface <path>] [--conf <f>] [--nms <f>]
// image1.jpg image2.jpg ...
//
// Output (stdout): JSON array, one object per input image:
// [
// {
// "image": "actor.jpg",
// "embedding": [0.012, -0.034, ...], // 512 floats, L2-normalised
// "bbox": [x, y, w, h],
// "confidence": 0.91
// },
// {
// "image": "bad.jpg",
// "embedding": null, // no face detected / alignment failed
// "error": "no face detected"
// }
// ]
//
// Design: each image is processed independently. If multiple faces are
// detected the one with the highest confidence is used (gallery images are
// expected to contain exactly one subject). A warning is printed to stderr
// when more than one face is found.
//
// This binary is intentionally a thin wrapper around the same ONNX models
// used by scene_analyze, so embeddings are guaranteed compatible.
#include "arcface_embedder.hpp"
#include "trt_arcface_embedder.hpp"
#include "trt_scrfd_decoder.hpp"
#include "face_utils.hpp"
#include "ort_provider.hpp"
#include "scrfd_decoder.hpp"
#include "config.hpp"
#include <opencv2/imgcodecs.hpp>
#include <opencv2/imgproc.hpp>
#include <nlohmann/json.hpp>
#include <algorithm>
#include <cstring>
#include <filesystem>
#include <functional>
#include <iostream>
#include <memory>
#include <stdexcept>
#include <string>
#include <vector>
namespace fs = std::filesystem;
using json = nlohmann::json;
// ── Per-image result ──────────────────────────────────────────────────────────
struct FaceResult {
std::string image_path;
bool ok{false};
std::string error;
Embedding embedding{};
float confidence{0.f};
float bbox[4]{}; // x, y, w, h
std::array<cv::Point2f, 5> landmarks{};
};
// ── Debug rendering ───────────────────────────────────────────────────────────
// Writes <dir>/<stem>_annotated.jpg (input with bbox + 5 landmarks) and
// <dir>/<stem>_aligned.jpg (112×112 aligned crop). Stem is derived from the
// parent directory and filename so images from different actor folders don't
// collide when fed into a single debug dir.
static std::string debug_stem(const std::string& path) {
fs::path p(path);
std::string parent = p.parent_path().filename().string();
std::string stem = p.stem().string();
return parent.empty() ? stem : parent + "_" + stem;
}
static void save_debug(const std::string& dir,
const std::string& src_path,
const cv::Mat& img,
const DetectedFace& face,
const cv::Mat& aligned) {
fs::create_directories(dir);
cv::Mat annotated = img.clone();
cv::rectangle(annotated, face.bbox, {0, 255, 0}, 2);
static const cv::Scalar colors[5] = {
{ 0, 0, 255}, // right eye — red
{255, 0, 0}, // left eye — blue
{ 0, 255, 255}, // nose — yellow
{ 0, 255, 0}, // right mouth — green
{255, 0, 255}, // left mouth — magenta
};
for (int i = 0; i < 5; ++i)
cv::circle(annotated, face.landmarks[i], 4, colors[i], -1);
const std::string stem = debug_stem(src_path);
cv::imwrite(dir + "/" + stem + "_annotated.jpg", annotated);
cv::imwrite(dir + "/" + stem + "_aligned.jpg", aligned);
}
// ── Process one image ─────────────────────────────────────────────────────────
static FaceResult process(const std::string& path,
const std::function<std::vector<DetectedFace>(const cv::Mat&)>& detect,
const std::function<Embedding(const cv::Mat&)>& embed_one,
int max_side,
const std::string& debug_dir = "") {
FaceResult res;
res.image_path = path;
cv::Mat img = cv::imread(path);
if (img.empty()) {
res.error = "cannot read image";
return res;
}
if (max_side > 0) {
const int big = std::max(img.cols, img.rows);
if (big > max_side) {
const double s = static_cast<double>(max_side) / big;
cv::resize(img, img, {}, s, s, cv::INTER_AREA);
}
}
std::vector<DetectedFace> faces = detect(img);
if (faces.empty()) {
res.error = "no face detected";
return res;
}
if (faces.size() > 1)
std::cerr << "[warn] " << path << ": " << faces.size()
<< " faces detected, using highest-confidence one\n";
const auto& best = *std::max_element(
faces.begin(), faces.end(),
[](const DetectedFace& a, const DetectedFace& b) {
return a.confidence < b.confidence;
});
cv::Mat crop = align_face(img, best.landmarks);
if (crop.empty()) {
res.error = "alignment failed";
return res;
}
res.ok = true;
res.embedding = embed_one(crop);
res.confidence = best.confidence;
res.bbox[0] = best.bbox.x;
res.bbox[1] = best.bbox.y;
res.bbox[2] = best.bbox.width;
res.bbox[3] = best.bbox.height;
res.landmarks = best.landmarks;
if (!debug_dir.empty())
save_debug(debug_dir, path, img, best, crop);
return res;
}
// ── Main ──────────────────────────────────────────────────────────────────────
int main(int argc, char** argv) {
std::string detector_model = kDefaultDetectorModel;
std::string detector_engine;
std::string arcface_model = kDefaultArcfaceModel;
std::string arcface_engine;
std::string debug_dir;
float conf = 0.5f, nms = 0.4f;
int max_side = 500;
std::vector<std::string> images;
for (int i = 1; i < argc; ++i) {
if (std::strcmp(argv[i], "--detector") == 0 && i+1 < argc) { detector_model = argv[++i]; }
else if (std::strcmp(argv[i], "--detector-engine") == 0 && i+1 < argc) { detector_engine = argv[++i]; }
else if (std::strcmp(argv[i], "--arcface") == 0 && i+1 < argc) { arcface_model = argv[++i]; }
else if (std::strcmp(argv[i], "--arcface-engine") == 0 && i+1 < argc) { arcface_engine = argv[++i]; }
else if (std::strcmp(argv[i], "--conf") == 0 && i+1 < argc) { conf = std::stof(argv[++i]); }
else if (std::strcmp(argv[i], "--nms") == 0 && i+1 < argc) { nms = std::stof(argv[++i]); }
else if (std::strcmp(argv[i], "--save-debug") == 0 && i+1 < argc) { debug_dir = argv[++i]; }
else if (std::strcmp(argv[i], "--max-side") == 0 && i+1 < argc) { max_side = std::stoi(argv[++i]); }
else if (argv[i][0] != '-') { images.push_back(argv[i]); }
else { std::cerr << "[warn] unknown flag: " << argv[i] << "\n"; }
}
if (images.empty()) {
std::cerr << "Usage: embed_faces [--detector <path>] [--arcface <path>] "
"[--save-debug <dir>] [--max-side <N>] image1.jpg ...\n";
return 1;
}
const OrtProvider provider = detect_ort_provider();
std::cerr << "[embed_faces] inference provider: " << provider_name(provider) << "\n";
std::unique_ptr<SCRFDDecoder> ort_det;
std::unique_ptr<TrtScrfdDecoder> trt_det;
std::function<std::vector<DetectedFace>(const cv::Mat&)> detect;
if (!detector_engine.empty()) {
trt_det = std::make_unique<TrtScrfdDecoder>(detector_engine, conf, nms);
detect = [&](const cv::Mat& im) { return trt_det->detect(im); };
} else {
ort_det = std::make_unique<SCRFDDecoder>(detector_model, conf, nms, provider);
detect = [&](const cv::Mat& im) { return ort_det->detect(im); };
}
std::unique_ptr<ArcFaceEmbedder> ort_emb;
std::unique_ptr<TrtArcFaceEmbedder> trt_emb;
std::function<Embedding(const cv::Mat&)> embed_one;
if (!arcface_engine.empty()) {
trt_emb = std::make_unique<TrtArcFaceEmbedder>(arcface_engine);
embed_one = [&](const cv::Mat& c) { return trt_emb->embed({c})[0]; };
} else {
ort_emb = std::make_unique<ArcFaceEmbedder>(arcface_model, provider);
embed_one = [&](const cv::Mat& c) { return ort_emb->embed_one(c); };
}
// Process images and build JSON output
json output = json::array();
for (const auto& path : images) {
std::cerr << "[embed_faces] " << path << "\n";
FaceResult res = process(path, detect, embed_one, max_side, debug_dir);
json entry;
entry["image"] = res.image_path;
if (res.ok) {
entry["embedding"] = std::vector<float>(res.embedding.begin(),
res.embedding.end());
entry["confidence"] = res.confidence;
entry["bbox"] = {res.bbox[0], res.bbox[1], res.bbox[2], res.bbox[3]};
json lms = json::array();
for (const auto& pt : res.landmarks) lms.push_back({pt.x, pt.y});
entry["landmarks"] = std::move(lms);
} else {
entry["embedding"] = nullptr;
entry["error"] = res.error;
std::cerr << " [skip] " << res.error << "\n";
}
output.push_back(std::move(entry));
}
std::cout << output.dump() << "\n";
return 0;
}