#!/usr/bin/env bash # Produce the int8 form of a model for the Hexagon (docs/dev/inference.md §5). # # ./tools/quantise-models.sh PHOTO_DIR MODEL.onnx [MODEL.onnx ...] # # Writes `MODEL.int8.onnx` beside each input: a QDQ graph, per-channel int8 # weights, uint8 activations — the form QNN's HTP backend takes whole. The # activations' ranges come from running the f32 model over the photographs in # PHOTO_DIR, fed exactly as the app feeds them (letterboxed to the model's # input, the detector's `(x - 127.5) / 128` normalisation), which is why # this is a release-time step and not something the device does: it needs # real photographs and, after it, a person reading §10 M2's numbers. # # The SCRFD and ArcFace exports are opset 11; per-channel QDQ needs 13, so a # model below 13 is first upgraded to 17. That changes only the graph's # spelling, not a weight — and it is what `tools/fix-face-model-shapes.sh` # will do to the canonical files in the same model release. # # A venv per run, like fix-face-model-shapes.sh: the tools are not a build # input and nothing in the tree should have them on its path. set -euo pipefail if [ "$#" -lt 2 ]; then sed -n '2,20p' "$0" >&2 exit 2 fi PHOTOS="$1"; shift [ -d "${PHOTOS}" ] || { echo "no such directory: ${PHOTOS}" >&2; exit 1; } WORK="$(mktemp -d -p /var/tmp quantise-models.XXXXXX)" trap 'rm -rf "${WORK}"' EXIT echo "==> venv in ${WORK}" uv venv --python 3.12 "${WORK}/venv" >/dev/null VIRTUAL_ENV="${WORK}/venv" uv pip install --quiet onnx onnxruntime pillow numpy sympy exec "${WORK}/venv/bin/python" "$(dirname "$0")/quantise-models.py" "${PHOTOS}" "$@"