Calibrate the int8 detectors on library proxies, in chunks, and measure them

The first int8 files found no faces at all, and for two reasons the
tool now guards against. The calibration set was landscape photographs
with no faces in them, so the score head's ranges had never seen the
face regime; the set is now proxies from the library itself. And ONNX
Runtime's strided and moving-average calibration modes both degrade
these graphs measurably (a quarter of the faces at eight images, none
at ninety-six), while driving the calibrator in chunks by hand gives
ranges identical to a single pass — so the tool does that, four images
at a time, and feeds quantize_static through its range cache.

Measured against f32 over 400 proxies (docs/inference.md §10.1): the
10g form finds every face above 32 px the f32 form finds; 500m and
2.5g find 96%, and what they lose sits at a median confidence of 0.52
against the 0.50 threshold. Shipped with the number on record.

The Android unpack list gains the three int8 files; without that the
tablet never saw them. D13's runtime half records the reopening.
This commit is contained in:
2026-09-19 16:02:44 +02:00
parent 4ed29b9d81
commit 76bc5652d7
9 changed files with 105 additions and 53 deletions
+51 -45
View File
@@ -14,10 +14,13 @@ from onnxruntime.quantization import (
QuantType,
quantize_static,
)
from onnxruntime.quantization.calibrate import create_calibrator
from onnxruntime.quantization.calibrate import save_tensors_data
from onnxruntime.quantization.shape_inference import quant_pre_process
from pathlib import Path
from PIL import Image, ImageOps
PHOTOS = 96 # enough for a stable range; more only costs time
PHOTOS = 64 # enough for a stable range; more only costs time
def letterbox(img, edge, pad, norm):
@@ -50,42 +53,52 @@ def preprocessing(name, shape):
class Photos(CalibrationDataReader):
"""The photographs, fed a stride at a time.
"""One chunk of photographs, fed as the app would feed them."""
`__len__` and `set_range` are what `CalibStridedMinMax` asks of a
reader: the calibrator folds each stride's activations into the running
range before asking for the next, so memory is one stride's worth and
not the whole set's.
"""
STRIDE = 4
def __init__(self, model_path, photos):
import onnxruntime as ort
s = ort.InferenceSession(model_path, providers=["CPUExecutionProvider"])
i = s.get_inputs()[0]
shape = [d if isinstance(d, int) else 1 for d in i.shape]
name = os.path.basename(model_path)
self.edge, self.pad, self.norm = preprocessing(name, shape)
self.name = i.name
self.photos = photos[: len(photos) - len(photos) % self.STRIDE]
self.set_range(0, len(self.photos))
def __len__(self):
return len(self.photos)
def set_range(self, start_index, end_index):
self.items = iter(
letterbox(Image.open(p), self.edge, self.pad, self.norm)
for p in self.photos[start_index:end_index]
)
def __init__(self, input_name, paths, edge, pad, norm):
self.name = input_name
self.items = iter(letterbox(Image.open(p), edge, pad, norm) for p in paths)
def get_next(self):
x = next(self.items, None)
return None if x is None else {self.name: x}
# Photographs whose activations are held in memory at once. Every ONNX
# Runtime calibrator keeps each image's whole set of activations until it
# folds them into a range — a gigabyte an image on the 10g detector at 640²,
# and folded once at the end, an OOM kill with no message. Folding every
# `CHUNK` images gives ranges identical to folding once (checked on
# scrfd_500m, 129 tensors, no difference) at a bounded cost.
CHUNK = 4
def calibrate(pre, name, photos, cache):
"""Min/max ranges over `photos`, written to `cache` for quantize_static.
Plain min/max: the moving average and the strided option of
`quantize_static` both measured worse than this on held-out proxies, and
the percentile method has no memory bound at all.
"""
import onnxruntime as ort
s = ort.InferenceSession(pre, providers=["CPUExecutionProvider"])
i = s.get_inputs()[0]
shape = [d if isinstance(d, int) else 1 for d in i.shape]
edge, pad, norm = preprocessing(name, shape)
calibrator = create_calibrator(
Path(pre),
None,
augmented_model_path=f"{pre}.augmented.onnx",
calibrate_method=CalibrationMethod.MinMax,
)
for start in range(0, len(photos), CHUNK):
calibrator.collect_data(Photos(i.name, photos[start : start + CHUNK], edge, pad, norm))
ranges = calibrator.compute_data()
save_tensors_data(ranges, cache)
os.remove(f"{pre}.augmented.onnx")
def main():
photo_dir, models = sys.argv[1], sys.argv[2:]
photos = sorted(
@@ -99,40 +112,33 @@ def main():
for src in models:
stem, _ = os.path.splitext(src)
name = os.path.basename(src)
out = f"{stem}.int8.onnx"
m = onnx.load(src)
opset = next((o.version for o in m.opset_import if o.domain in ("", "ai.onnx")), 0)
work = f"{stem}.quant-work.onnx"
if opset < 13:
print(f" {os.path.basename(src)}: opset {opset} -> 17")
print(f" {name}: opset {opset} -> 17")
m = version_converter.convert_version(m, 17)
m.ir_version = 8
onnx.save(m, work)
pre = f"{stem}.quant-pre.onnx"
quant_pre_process(work, pre)
cache = f"{stem}.quant-ranges.json"
calibrate(pre, name, photos, cache)
quantize_static(
pre,
out,
Photos(pre, photos),
None,
quant_format=QuantFormat.QDQ,
per_channel=True,
activation_type=QuantType.QUInt8,
weight_type=QuantType.QInt8,
# Min/max with a moving average across photographs, so one
# saturated highlight in one image does not set the range for
# every activation. Every calibrator keeps each image's whole
# set of activations until it folds them into a range, which
# for the 10g detector at 640² is a gigabyte an image and, left
# to fold once at the end, an OOM kill with no message. The
# stride folds every four (`CalibMaxIntermediateOutputs` looks
# like the same thing and is not: in this version it clears
# without folding). The percentile method has no such bound and
# is not usable on these graphs.
calibrate_method=CalibrationMethod.MinMax,
extra_options={"CalibMovingAverage": True, "CalibStridedMinMax": Photos.STRIDE},
calibration_cache_path=cache,
)
os.remove(work)
os.remove(pre)
for f in (work, pre, cache):
os.remove(f)
print(f" {out}: {os.path.getsize(out) // 1024} KB")