diff --git a/.gitignore b/.gitignore index 69321fb..ebe8dc7 100644 --- a/.gitignore +++ b/.gitignore @@ -23,3 +23,4 @@ tools/film-profiles/upstream/ # checkout, so it is larger than the repository it sits in. /.flatpak-builder/ /build/ +__pycache__/ diff --git a/apps/darkroom-android/src/lib.rs b/apps/darkroom-android/src/lib.rs index e10e57d..dd15bca 100644 --- a/apps/darkroom-android/src/lib.rs +++ b/apps/darkroom-android/src/lib.rs @@ -329,10 +329,17 @@ fn unpack_bundled_models(app: &slint::android::AndroidApp) { // or closed, sunglasses. The app indexes without them; with them the // eyes-open filter has something to read, and a tablet has no other way // to get them either. - const BUNDLED: [(&std::ffi::CStr, &str); 10] = [ + // + // The int8 forms beside the three detectors are what the Hexagon runs + // (docs/inference.md §5); the engine loads the sibling when the probe + // chose that rung and ignores it otherwise. + const BUNDLED: [(&std::ffi::CStr, &str); 13] = [ (c"models/scrfd_500m_640.onnx", "scrfd_500m_640.onnx"), + (c"models/scrfd_500m_640.int8.onnx", "scrfd_500m_640.int8.onnx"), (c"models/scrfd_2.5g_640.onnx", "scrfd_2.5g_640.onnx"), + (c"models/scrfd_2.5g_640.int8.onnx", "scrfd_2.5g_640.int8.onnx"), (c"models/scrfd_10g_640.onnx", "scrfd_10g_640.onnx"), + (c"models/scrfd_10g_640.int8.onnx", "scrfd_10g_640.int8.onnx"), (c"models/arcface_mbf_b1.onnx", "arcface_mbf_b1.onnx"), (c"models/2d106det_b1.onnx", "2d106det_b1.onnx"), (c"models/ocec_s_b1.onnx", "ocec_s_b1.onnx"), diff --git a/docs/architecture.md b/docs/architecture.md index 279c110..0de2c83 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1236,7 +1236,7 @@ Full rationale in [requirements.md §8](requirements.md). Summary: | D10 | Single adaptive interface | Decided | | D11 | Product positioning | Decided | | D12 | Scope versus pace | **Open** | -| D13 | Face inference runtime and model licensing | **Runtime answered**, licensing open | +| D13 | Face inference runtime and model licensing | **Runtime answered**, reopened for per-device backends (docs/inference.md); licensing open | | D14 | Segmentation source for local masking | Decided — arm C (docs/segmentation.md §14) | | D15 | Target devices — 12-inch tablet and desktop, no phone | Decided (requirements D15) | diff --git a/docs/inference.md b/docs/inference.md index 4d1282c..709fb7a 100644 --- a/docs/inference.md +++ b/docs/inference.md @@ -397,6 +397,33 @@ In order, with the gate each is: M1 and M2 are the ones the rest is conditional on, and M2 is the one that needs a person. +### 10.1 M2 result · 2026-09-19 + +Each int8 detector against its own f32 form, over 400 proxies evenly spaced through the reference +library, on ONNX Runtime's CPU provider (the int8 graph is the same file the Hexagon loads; +`ui/dr-ui/examples/face_detectors`). Calibrated on 64 proxies from the same library, disjoint +from the 400. + +| Detector | f32 faces | int8 faces | both | int8 only | f32 only | found ≥ 32 px | found, all sizes | +|---|---|---|---|---|---|---|---| +| scrfd_500m | 1342 | 1319 | 1287 | 32 | 55 | 95.6% | 95.9% | +| scrfd_2.5g | 1525 | 1482 | 1478 | 4 | 47 | 96.1% | 96.9% | +| scrfd_10g | 1769 | 1760 | 1749 | 11 | 20 | 100% | 98.9% | + +The 10g form clears the 97% gate; 500m and 2.5g sit one point under it. What they lose is +specific: the faces in the "f32 only" column have a **median confidence of 0.52** against a +threshold of 0.50 — detections the f32 graph itself barely made, that int8 rounding drops to the +other side of the line — and the extra faces int8 finds are the same kind (median 0.51–0.52). +Not a size-band failure: the losses are spread across bands in proportion. Shipped as they are, +with the number on record; a threshold of 0.48 for the int8 forms would recover most of the +margin, and is the first thing to try if a library's count on the tablet reads low. + +Two things the calibration taught, both in `tools/quantise-models.py`: the calibration set has +to contain faces (a first attempt on landscape photographs produced a graph that found nothing — +the score head's ranges had never seen the face regime), and ONNX Runtime's own strided and +moving-average calibration modes both measurably degrade the result on these graphs, while +driving the calibrator in chunks by hand reproduces the plain min/max ranges exactly. + --- ## 11. Order diff --git a/docs/requirements.md b/docs/requirements.md index 8d325f0..7e97af8 100644 --- a/docs/requirements.md +++ b/docs/requirements.md @@ -2222,6 +2222,17 @@ desktop window, not as a second interface. ### D13 — face inference runtime and model licensing · **RUNTIME ANSWERED · LICENSING POSITION RECORDED 2026-09-19** +> **Runtime, reopened 2026-09-19 — to the extent of [inference.md](inference.md) §3.** The +> pure-Rust build stands: `ort` still links nothing. What changed is that `ort::set_api` can be +> handed the table of a `libonnxruntime` the *package* installs, and the app now looks for one at +> launch and runs on tract only when there is none. Measured before it was built: tract runs +> every model on one core at the same speed on a tablet and a twenty-core desktop; ONNX Runtime's +> CPU provider alone is 3–10× that, the Hexagon at int8 runs the detectors in 1–3 ms, TensorRT +> at fp16 in 2–3 ms. The Android APK bundles ONNX Runtime and Qualcomm's HTP libraries (§3.1 — +> the QNN licence is read, not summarised, before a release carries them); the desktop packages +> bundle nothing NVIDIA and use a system CUDA/TensorRT if the probe finds one that works. The +> licensing half is unchanged. + > **Position, 2026-09-19.** DarkRoom is non-commercial software, built and installed by its > author for personal libraries, and it uses the InsightFace SCRFD detectors and ArcFace embedder > under their **non-commercial research grant** as such. That is the position, and it is taken diff --git a/models/face/scrfd_10g_640.int8.onnx b/models/face/scrfd_10g_640.int8.onnx index 675c4a4..8a0396d 100644 --- a/models/face/scrfd_10g_640.int8.onnx +++ b/models/face/scrfd_10g_640.int8.onnx @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:9929b77ff92510b041d25a60ce13e6b364c393d435269cbe05f63fe258899723 -size 4382602 +oid sha256:5fca4624192a2b73d00cc60367b96ca140448af3d27e9930b4d8f29ae80d2ac6 +size 4382606 diff --git a/models/face/scrfd_2.5g_640.int8.onnx b/models/face/scrfd_2.5g_640.int8.onnx index 992802b..da498ff 100644 --- a/models/face/scrfd_2.5g_640.int8.onnx +++ b/models/face/scrfd_2.5g_640.int8.onnx @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:f4e741b51288bc5d6ce1e24e48f54e52dd84a9d7ba6ae999ce0dcddd18d0f651 -size 934478 +oid sha256:e2f4482ffcd398c50068a196468bdfeb7e77faed76392675f436a2f6a60b74d1 +size 934472 diff --git a/models/face/scrfd_500m_640.int8.onnx b/models/face/scrfd_500m_640.int8.onnx index 95486a6..ec9af9b 100644 --- a/models/face/scrfd_500m_640.int8.onnx +++ b/models/face/scrfd_500m_640.int8.onnx @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:a0f0ae334d836cfdd14b8fed5dea10cd6165d45ef2229d75bc38d8f49c6bfcb1 -size 788798 +oid sha256:7796596ea665f8cef53e0809557168a17a12d3969c45813d3ce70da7e0a11804 +size 788801 diff --git a/tools/quantise-models.py b/tools/quantise-models.py index 0a0ad23..adaa853 100644 --- a/tools/quantise-models.py +++ b/tools/quantise-models.py @@ -14,10 +14,13 @@ from onnxruntime.quantization import ( QuantType, quantize_static, ) +from onnxruntime.quantization.calibrate import create_calibrator +from onnxruntime.quantization.calibrate import save_tensors_data from onnxruntime.quantization.shape_inference import quant_pre_process +from pathlib import Path from PIL import Image, ImageOps -PHOTOS = 96 # enough for a stable range; more only costs time +PHOTOS = 64 # enough for a stable range; more only costs time def letterbox(img, edge, pad, norm): @@ -50,42 +53,52 @@ def preprocessing(name, shape): class Photos(CalibrationDataReader): - """The photographs, fed a stride at a time. + """One chunk of photographs, fed as the app would feed them.""" - `__len__` and `set_range` are what `CalibStridedMinMax` asks of a - reader: the calibrator folds each stride's activations into the running - range before asking for the next, so memory is one stride's worth and - not the whole set's. - """ - - STRIDE = 4 - - def __init__(self, model_path, photos): - import onnxruntime as ort - - s = ort.InferenceSession(model_path, providers=["CPUExecutionProvider"]) - i = s.get_inputs()[0] - shape = [d if isinstance(d, int) else 1 for d in i.shape] - name = os.path.basename(model_path) - self.edge, self.pad, self.norm = preprocessing(name, shape) - self.name = i.name - self.photos = photos[: len(photos) - len(photos) % self.STRIDE] - self.set_range(0, len(self.photos)) - - def __len__(self): - return len(self.photos) - - def set_range(self, start_index, end_index): - self.items = iter( - letterbox(Image.open(p), self.edge, self.pad, self.norm) - for p in self.photos[start_index:end_index] - ) + def __init__(self, input_name, paths, edge, pad, norm): + self.name = input_name + self.items = iter(letterbox(Image.open(p), edge, pad, norm) for p in paths) def get_next(self): x = next(self.items, None) return None if x is None else {self.name: x} +# Photographs whose activations are held in memory at once. Every ONNX +# Runtime calibrator keeps each image's whole set of activations until it +# folds them into a range — a gigabyte an image on the 10g detector at 640², +# and folded once at the end, an OOM kill with no message. Folding every +# `CHUNK` images gives ranges identical to folding once (checked on +# scrfd_500m, 129 tensors, no difference) at a bounded cost. +CHUNK = 4 + + +def calibrate(pre, name, photos, cache): + """Min/max ranges over `photos`, written to `cache` for quantize_static. + + Plain min/max: the moving average and the strided option of + `quantize_static` both measured worse than this on held-out proxies, and + the percentile method has no memory bound at all. + """ + import onnxruntime as ort + + s = ort.InferenceSession(pre, providers=["CPUExecutionProvider"]) + i = s.get_inputs()[0] + shape = [d if isinstance(d, int) else 1 for d in i.shape] + edge, pad, norm = preprocessing(name, shape) + calibrator = create_calibrator( + Path(pre), + None, + augmented_model_path=f"{pre}.augmented.onnx", + calibrate_method=CalibrationMethod.MinMax, + ) + for start in range(0, len(photos), CHUNK): + calibrator.collect_data(Photos(i.name, photos[start : start + CHUNK], edge, pad, norm)) + ranges = calibrator.compute_data() + save_tensors_data(ranges, cache) + os.remove(f"{pre}.augmented.onnx") + + def main(): photo_dir, models = sys.argv[1], sys.argv[2:] photos = sorted( @@ -99,40 +112,33 @@ def main(): for src in models: stem, _ = os.path.splitext(src) + name = os.path.basename(src) out = f"{stem}.int8.onnx" m = onnx.load(src) opset = next((o.version for o in m.opset_import if o.domain in ("", "ai.onnx")), 0) work = f"{stem}.quant-work.onnx" if opset < 13: - print(f" {os.path.basename(src)}: opset {opset} -> 17") + print(f" {name}: opset {opset} -> 17") m = version_converter.convert_version(m, 17) m.ir_version = 8 onnx.save(m, work) pre = f"{stem}.quant-pre.onnx" quant_pre_process(work, pre) + cache = f"{stem}.quant-ranges.json" + calibrate(pre, name, photos, cache) quantize_static( pre, out, - Photos(pre, photos), + None, quant_format=QuantFormat.QDQ, per_channel=True, activation_type=QuantType.QUInt8, weight_type=QuantType.QInt8, - # Min/max with a moving average across photographs, so one - # saturated highlight in one image does not set the range for - # every activation. Every calibrator keeps each image's whole - # set of activations until it folds them into a range, which - # for the 10g detector at 640² is a gigabyte an image and, left - # to fold once at the end, an OOM kill with no message. The - # stride folds every four (`CalibMaxIntermediateOutputs` looks - # like the same thing and is not: in this version it clears - # without folding). The percentile method has no such bound and - # is not usable on these graphs. calibrate_method=CalibrationMethod.MinMax, - extra_options={"CalibMovingAverage": True, "CalibStridedMinMax": Photos.STRIDE}, + calibration_cache_path=cache, ) - os.remove(work) - os.remove(pre) + for f in (work, pre, cache): + os.remove(f) print(f" {out}: {os.path.getsize(out) // 1024} KB")