diff --git a/models/face/README.md b/models/face/README.md index 8e1712d..ddb9674 100644 --- a/models/face/README.md +++ b/models/face/README.md @@ -28,6 +28,16 @@ every reader treats "never read" as unknown, never as closed. They are found in directory as the pair, so a hand-placed pair does not pick up a package's eye models from a directory it otherwise outranks. + scrfd_500m_640.int8.onnx 0.8 MB the same three, in the form the Hexagon NPU takes + scrfd_2.5g_640.int8.onnx 0.9 MB (docs/inference.md §5) — opset 17, per-channel int8 + scrfd_10g_640.int8.onnx 4.3 MB weights, uint8 activations, calibrated on 96 photographs + +The int8 files are **derived** by `tools/quantise-models.sh` from the f32 ones beside them and +travel with them: the engine loads the `.int8.onnx` sibling when the device's backend wants it and +the canonical file otherwise, and a library indexed on the int8 form records it as a different +detector (`scrfd_500m_i8+w600k_mbf`), because it finds a different set of faces. Every other +platform ignores them. The embedder has no int8 form and never will (§7 of the same document). + **A clone without git-lfs gets a ~130-byte pointer where each model should be.** Both packagers check for exactly that and refuse, rather than shipping the pointer and failing inside tract on the user's machine. Fix it with `git lfs pull`. diff --git a/models/face/scrfd_10g_640.int8.onnx b/models/face/scrfd_10g_640.int8.onnx new file mode 100644 index 0000000..675c4a4 --- /dev/null +++ b/models/face/scrfd_10g_640.int8.onnx @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9929b77ff92510b041d25a60ce13e6b364c393d435269cbe05f63fe258899723 +size 4382602 diff --git a/models/face/scrfd_2.5g_640.int8.onnx b/models/face/scrfd_2.5g_640.int8.onnx new file mode 100644 index 0000000..992802b --- /dev/null +++ b/models/face/scrfd_2.5g_640.int8.onnx @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f4e741b51288bc5d6ce1e24e48f54e52dd84a9d7ba6ae999ce0dcddd18d0f651 +size 934478 diff --git a/models/face/scrfd_500m_640.int8.onnx b/models/face/scrfd_500m_640.int8.onnx new file mode 100644 index 0000000..95486a6 --- /dev/null +++ b/models/face/scrfd_500m_640.int8.onnx @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a0f0ae334d836cfdd14b8fed5dea10cd6165d45ef2229d75bc38d8f49c6bfcb1 +size 788798 diff --git a/tools/quantise-models.py b/tools/quantise-models.py new file mode 100644 index 0000000..0a0ad23 --- /dev/null +++ b/tools/quantise-models.py @@ -0,0 +1,140 @@ +"""The body of quantise-models.sh; see there. Run through it, not directly.""" + +import glob +import os +import sys + +import numpy as np +import onnx +from onnx import version_converter +from onnxruntime.quantization import ( + CalibrationDataReader, + CalibrationMethod, + QuantFormat, + QuantType, + quantize_static, +) +from onnxruntime.quantization.shape_inference import quant_pre_process +from PIL import Image, ImageOps + +PHOTOS = 96 # enough for a stable range; more only costs time + + +def letterbox(img, edge, pad, norm): + """The app's Letterbox::sample: fit the long side to `edge`, centre, pad.""" + img = ImageOps.exif_transpose(img).convert("RGB") + w, h = img.size + scale = edge / max(w, h) + nw, nh = max(1, round(w * scale)), max(1, round(h * scale)) + img = img.resize((nw, nh), Image.BILINEAR) + canvas = Image.new("RGB", (edge, edge), (pad, pad, pad)) + canvas.paste(img, ((edge - nw) // 2, (edge - nh) // 2)) + x = np.asarray(canvas, dtype=np.float32) # HWC, 0..255 + x = norm(x) + return np.ascontiguousarray(x.transpose(2, 0, 1))[None] # NCHW + + +def preprocessing(name, shape): + """Which normalisation this model is fed in the app. + + SCRFD (`dr-face::detect`): `(v - 127.5) / 128`, padded with 114. + ArcFace (`dr-face::embed`): the same, on an aligned 112 crop — a + letterboxed photograph is the wrong distribution, but the embedder is + never quantised (§7), so this is only ever a fallback. + YOLO (`dr-segment`): `v / 255`, padded with 0.5. + """ + edge = shape[-1] + if name.startswith("scrfd") or name.startswith("arcface"): + return edge, 114, lambda x: (x - 127.5) / 128.0 + return edge, 128, lambda x: x / 255.0 + + +class Photos(CalibrationDataReader): + """The photographs, fed a stride at a time. + + `__len__` and `set_range` are what `CalibStridedMinMax` asks of a + reader: the calibrator folds each stride's activations into the running + range before asking for the next, so memory is one stride's worth and + not the whole set's. + """ + + STRIDE = 4 + + def __init__(self, model_path, photos): + import onnxruntime as ort + + s = ort.InferenceSession(model_path, providers=["CPUExecutionProvider"]) + i = s.get_inputs()[0] + shape = [d if isinstance(d, int) else 1 for d in i.shape] + name = os.path.basename(model_path) + self.edge, self.pad, self.norm = preprocessing(name, shape) + self.name = i.name + self.photos = photos[: len(photos) - len(photos) % self.STRIDE] + self.set_range(0, len(self.photos)) + + def __len__(self): + return len(self.photos) + + def set_range(self, start_index, end_index): + self.items = iter( + letterbox(Image.open(p), self.edge, self.pad, self.norm) + for p in self.photos[start_index:end_index] + ) + + def get_next(self): + x = next(self.items, None) + return None if x is None else {self.name: x} + + +def main(): + photo_dir, models = sys.argv[1], sys.argv[2:] + photos = sorted( + p + for ext in ("jpg", "jpeg", "JPG", "JPEG", "png") + for p in glob.glob(os.path.join(photo_dir, "**", f"*.{ext}"), recursive=True) + )[:PHOTOS] + if len(photos) < 20: + sys.exit(f"only {len(photos)} photographs under {photo_dir}; calibration wants dozens") + print(f"==> calibrating on {len(photos)} photographs") + + for src in models: + stem, _ = os.path.splitext(src) + out = f"{stem}.int8.onnx" + m = onnx.load(src) + opset = next((o.version for o in m.opset_import if o.domain in ("", "ai.onnx")), 0) + work = f"{stem}.quant-work.onnx" + if opset < 13: + print(f" {os.path.basename(src)}: opset {opset} -> 17") + m = version_converter.convert_version(m, 17) + m.ir_version = 8 + onnx.save(m, work) + pre = f"{stem}.quant-pre.onnx" + quant_pre_process(work, pre) + quantize_static( + pre, + out, + Photos(pre, photos), + quant_format=QuantFormat.QDQ, + per_channel=True, + activation_type=QuantType.QUInt8, + weight_type=QuantType.QInt8, + # Min/max with a moving average across photographs, so one + # saturated highlight in one image does not set the range for + # every activation. Every calibrator keeps each image's whole + # set of activations until it folds them into a range, which + # for the 10g detector at 640² is a gigabyte an image and, left + # to fold once at the end, an OOM kill with no message. The + # stride folds every four (`CalibMaxIntermediateOutputs` looks + # like the same thing and is not: in this version it clears + # without folding). The percentile method has no such bound and + # is not usable on these graphs. + calibrate_method=CalibrationMethod.MinMax, + extra_options={"CalibMovingAverage": True, "CalibStridedMinMax": Photos.STRIDE}, + ) + os.remove(work) + os.remove(pre) + print(f" {out}: {os.path.getsize(out) // 1024} KB") + + +if __name__ == "__main__": + main() diff --git a/tools/quantise-models.sh b/tools/quantise-models.sh new file mode 100755 index 0000000..91703c4 --- /dev/null +++ b/tools/quantise-models.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash +# Produce the int8 form of a model for the Hexagon (docs/inference.md §5). +# +# ./tools/quantise-models.sh PHOTO_DIR MODEL.onnx [MODEL.onnx ...] +# +# Writes `MODEL.int8.onnx` beside each input: a QDQ graph, per-channel int8 +# weights, uint8 activations — the form QNN's HTP backend takes whole. The +# activations' ranges come from running the f32 model over the photographs in +# PHOTO_DIR, fed exactly as the app feeds them (letterboxed to the model's +# input, the detector's `(x - 127.5) / 128` normalisation), which is why +# this is a release-time step and not something the device does: it needs +# real photographs and, after it, a person reading §10 M2's numbers. +# +# The SCRFD and ArcFace exports are opset 11; per-channel QDQ needs 13, so a +# model below 13 is first upgraded to 17. That changes only the graph's +# spelling, not a weight — and it is what `tools/fix-face-model-shapes.sh` +# will do to the canonical files in the same model release. +# +# A venv per run, like fix-face-model-shapes.sh: the tools are not a build +# input and nothing in the tree should have them on its path. +set -euo pipefail +if [ "$#" -lt 2 ]; then + sed -n '2,20p' "$0" >&2 + exit 2 +fi +PHOTOS="$1"; shift +[ -d "${PHOTOS}" ] || { echo "no such directory: ${PHOTOS}" >&2; exit 1; } + +WORK="$(mktemp -d -p /var/tmp quantise-models.XXXXXX)" +trap 'rm -rf "${WORK}"' EXIT +echo "==> venv in ${WORK}" +uv venv --python 3.12 "${WORK}/venv" >/dev/null +VIRTUAL_ENV="${WORK}/venv" uv pip install --quiet onnx onnxruntime pillow numpy sympy + +exec "${WORK}/venv/bin/python" "$(dirname "$0")/quantise-models.py" "${PHOTOS}" "$@"