diff --git a/tools/htp_graph.py b/tools/htp_graph.py new file mode 100644 index 0000000..b86f1c1 --- /dev/null +++ b/tools/htp_graph.py @@ -0,0 +1,202 @@ +"""Exact rewrites that make a graph one QNN's HTP can hold (docs/dev/inference.md §1.5). + +Each rewrite spells the same arithmetic in operators the Hexagon runs, and +`rewrite` checks the result against the input graph on ONNX Runtime's CPU +before returning it — a rewrite that changes an output by more than float +rounding is refused, not shipped. + +- **6-D Bayer pack → SpaceToDepth(2).** The denoiser packs the mosaic with + `Reshape(1,C,H/2,2,W/2,2) → Transpose(0,1,3,5,2,4) → Reshape(1,4C,…)`. + QNN's tensors stop at rank 5 (error 6007 at compose); for C = 1 that + sequence *is* SpaceToDepth. +- **Unfold → SpaceToDepth(8).** XFeat spells its 8×8 unfold as 224 Slices, + 225 Transposes and two 6-D Concats; it is SpaceToDepth(8) of the + normalised image, 736 nodes down to 60. +- **Computed reshape targets → constants.** `Shape → Slice → Concat` feeding + a Reshape, where shape inference proves the answer. +- **Bilinear Resize → two MatMuls.** The HTP refuses `ResizeBilinear` at the + sizes XFeat uses (3110); a half-pixel bilinear resize between fixed sizes + is `X · Rxᵀ` then `Ry ·`, with ONNX's own edge-clamped weights. +- **InstanceNormalization → primitives** is not here: it is exact, but the + two full-image reductions it becomes cost the HTP more than it saves. XFeat + keeps its normalisation in int8, which the HTP takes. +""" +import numpy as np +import onnx +import onnxruntime as ort +from onnx import helper, numpy_helper, shape_inference + +TOLERANCE = 1e-4 # relative to each output's largest magnitude + + +def _prune(g): + """Drop nodes nobody reads and initialisers nobody uses.""" + while True: + used = {i for n in g.node for i in n.input} | {o.name for o in g.output} + keep = [n for n in g.node if any(o in used for o in n.output)] + if len(keep) == len(g.node): + break + del g.node[:] + g.node.extend(keep) + used = {i for n in g.node for i in n.input} + keep = [i for i in g.initializer if i.name in used] + del g.initializer[:] + g.initializer.extend(keep) + + +def bayer_pack(m): + g = m.graph + prod = {o: n for n in g.node for o in n.output} + users = {} + for n in g.node: + for i in n.input: + users.setdefault(i, []).append(n) + swaps = {} + for n in g.node: + if n.op_type != "Transpose": + continue + perm = next(a.ints for a in n.attribute if a.name == "perm") + r1, u = prod.get(n.input[0]), users.get(n.output[0], []) + if list(perm) != [0, 1, 3, 5, 2, 4] or not r1 or r1.op_type != "Reshape": + continue + if len(u) != 1 or u[0].op_type != "Reshape": + continue + s2d = helper.make_node("SpaceToDepth", [r1.input[0]], [u[0].output[0]], name=n.name + "_s2d", blocksize=2) + swaps[id(r1)] = s2d + swaps[id(n)] = swaps[id(u[0])] = None + nodes = [swaps.get(id(n), n) for n in g.node if swaps.get(id(n), n) is not None] + del g.node[:] + g.node.extend(nodes) + return m + + +def fold_reshapes(m): + m = shape_inference.infer_shapes(m) + g = m.graph + vi = {v.name: v for v in list(g.value_info) + list(g.output) + list(g.input)} + inits = {i.name for i in g.initializer} + dims = lambda t: [d.dim_value for d in vi[t].type.tensor_type.shape.dim] if t in vi else [] + for n in g.node: + if n.op_type != "Reshape" or n.input[1] in inits: + continue + shape = dims(n.output[0]) + if not shape or 0 in shape: + src = dims(n.input[0]) + if len(src) != 5 or 0 in src: # (1,C,k,H,W) -> (1,C·k,H,W), the denoiser's tile + continue + shape = [src[0], src[1] * src[2], src[3], src[4]] + name = n.output[0] + "_shape" + g.initializer.append(numpy_helper.from_array(np.array(shape, np.int64), name)) + n.input[1] = name + _prune(g) + del g.value_info[:] + return m + + +def unfold(m, block=8): + """Replace the Slice/Transpose/Concat region ending in the Reshape that + produces the (1, block², H/block, W/block) tensor with SpaceToDepth.""" + g = m.graph + prod = {o: n for n in g.node for o in n.output} + region_ops = ("Slice", "Transpose", "Concat", "Unsqueeze", "Reshape") + inits = {i.name for i in g.initializer} + m_inf = shape_inference.infer_shapes(m) + vi = {v.name: [d.dim_value for d in v.type.tensor_type.shape.dim] for v in m_inf.graph.value_info} + for end in g.node: + out = vi.get(end.output[0], []) + if end.op_type != "Reshape" or len(out) != 4 or out[1] != block * block: + continue + seen, stack, leaves = set(), [end.input[0]], set() + while stack: + t = stack.pop() + if t in seen or t in inits: + continue + seen.add(t) + n = prod.get(t) + if n is not None and n.op_type in region_ops: + stack += list(n.input) + else: + leaves.add(t) + if len(leaves) != 1 or len(seen) < 100: # the hand-written unfold, not an ordinary reshape + continue + region = {id(end)} | {id(prod[t]) for t in seen if t in prod and prod[t].op_type in region_ops} + nodes = [] + for n in g.node: + if id(n) in region: + if n is end: + nodes.append(helper.make_node("SpaceToDepth", [next(iter(leaves))], [end.output[0]], + name=end.name + "_s2d", blocksize=block)) + continue + nodes.append(n) + del g.node[:] + g.node.extend(nodes) + _prune(g) + del g.value_info[:] + return m + return m + + +def _bilinear(n_in, n_out): + r = np.zeros((n_out, n_in), np.float32) + for o in range(n_out): + x = min(max((o + 0.5) * n_in / n_out - 0.5, 0), n_in - 1) + i0 = int(np.floor(x)) + f = x - i0 + r[o, i0] += 1 - f + r[o, min(i0 + 1, n_in - 1)] += f + return r + + +def resize_matmul(m): + """Every linear, half-pixel Resize between fixed NCHW sizes.""" + m = shape_inference.infer_shapes(m) + g = m.graph + vi = {v.name: [d.dim_value for d in v.type.tensor_type.shape.dim] for v in list(g.value_info) + list(g.output)} + nodes = [] + for n in g.node: + a = {x.name: helper.get_attribute_value(x) for x in n.attribute} + ok = (n.op_type == "Resize" and a.get("mode") == b"linear" + and a.get("coordinate_transformation_mode", b"half_pixel") == b"half_pixel" + and len(vi.get(n.input[0], [])) == 4 and len(vi.get(n.output[0], [])) == 4) + if not ok: + nodes.append(n) + continue + (_, _, h, w), (_, _, h2, w2) = vi[n.input[0]], vi[n.output[0]] + t = n.name + rx = numpy_helper.from_array(_bilinear(w, w2).T.copy(), t + "_rxT") + ry = numpy_helper.from_array(_bilinear(h, h2).T.copy(), t + "_ryT") + g.initializer.extend([rx, ry]) + nodes += [ + helper.make_node("MatMul", [n.input[0], rx.name], [t + "_w"], name=t + "_mw"), + helper.make_node("Transpose", [t + "_w"], [t + "_t"], name=t + "_t1", perm=[0, 1, 3, 2]), + helper.make_node("MatMul", [t + "_t", ry.name], [t + "_h"], name=t + "_mh"), + helper.make_node("Transpose", [t + "_h"], [n.output[0]], name=t + "_t2", perm=[0, 1, 3, 2]), + ] + del g.node[:] + g.node.extend(nodes) + _prune(g) + del g.value_info[:] + return m + + +REWRITES = {"bayer": [bayer_pack, fold_reshapes], "unfold": [unfold], "resize": [resize_matmul]} + + +def rewrite(path, names): + """The graph at `path` with the named rewrites applied, checked exact.""" + m = onnx.load(path) + before = len(m.graph.node) + for name in names: + for step in REWRITES[name]: + m = step(m) + onnx.checker.check_model(m) + a = ort.InferenceSession(path, providers=["CPUExecutionProvider"]) + b = ort.InferenceSession(m.SerializeToString(), providers=["CPUExecutionProvider"]) + rng = np.random.default_rng(3) + feed = {i.name: rng.random([d if isinstance(d, int) else 1 for d in i.shape], dtype=np.float32) for i in a.get_inputs()} + for o, x, y in zip(a.get_outputs(), a.run(None, feed), b.run(None, feed)): + err = float(np.abs(x - y).max()) / max(float(np.abs(x).max()), 1e-6) + if err > TOLERANCE: + raise SystemExit(f"{path}: rewrite {names} moved output {o.name} by {err:.2e} (relative)") + print(f" rewrites {'+'.join(names)}: {before} -> {len(m.graph.node)} nodes, exact") + return m diff --git a/tools/quantise-models.py b/tools/quantise-models.py index adaa853..9ccd165 100644 --- a/tools/quantise-models.py +++ b/tools/quantise-models.py @@ -3,143 +3,258 @@ import glob import os import sys +from pathlib import Path import numpy as np import onnx +import onnxruntime as ort from onnx import version_converter -from onnxruntime.quantization import ( - CalibrationDataReader, - CalibrationMethod, - QuantFormat, - QuantType, - quantize_static, -) -from onnxruntime.quantization.calibrate import create_calibrator -from onnxruntime.quantization.calibrate import save_tensors_data -from onnxruntime.quantization.shape_inference import quant_pre_process -from pathlib import Path -from PIL import Image, ImageOps +from onnxruntime.quantization import CalibrationDataReader, CalibrationMethod, QuantType, quantize_static +from onnxruntime.quantization.calibrate import create_calibrator, save_tensors_data +from onnxruntime.quantization.execution_providers.qnn import get_qnn_qdq_config +from PIL import Image -PHOTOS = 64 # enough for a stable range; more only costs time +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import htp_graph # noqa: E402 + +MODELS = Path(__file__).resolve().parents[1] / "models" +PHOTOS = 300 # calibration photographs; face crops and keypoints come from fewer +CHUNK = 4 # inputs whose activations are held at once (scrfd_10g: ~1 GB each) + +Q = QuantType +FORMS = {"int8": (Q.QUInt8, Q.QInt8), "a16w8": (Q.QUInt16, Q.QInt8), "a16w16": (Q.QUInt16, Q.QInt16)} + +# Per model: where it lives, the form `Rung::form` gives its role, the exact +# rewrites its graph needs, and nodes that stay float on the CPU because one +# scale cannot serve the tensor (the segmenter's rows: boxes in pixels beside +# scores in 0..1) or because the HTP's 16-bit arithmetic drifts there (the +# scene model's attention). Measured, inference.md §1.5. +TABLE = { + "scrfd_500m_640": dict(dir="face", form="a16w8", feed="scrfd"), + "scrfd_2.5g_640": dict(dir="face", form="a16w8", feed="scrfd"), + "scrfd_10g_640": dict(dir="face", form="a16w8", feed="scrfd"), + "2d106det_b1": dict(dir="face", form="a16w8", feed="landmarks"), + "yolo26n-seg": dict(dir="segment", form="a16w16", feed="yolo", float_from="/model.23/Concat_4"), + "yolo26s-sem-ade20k": dict(dir="scene", form="a16w16", feed="yolo", + float_nodes=["/model.10/m/m.0/attn/MatMul", "/model.10/m/m.0/attn/Softmax", + "/model.10/m/m.0/attn/MatMul_1"]), + "migan-512": dict(dir="inpaint", form="a16w16", feed="migan"), + "xfeat-1024": dict(dir="keypoints", form="int8", feed="xfeat", rewrites=["unfold", "resize"]), + "xfeat-768": dict(dir="keypoints", form="int8", feed="xfeat", rewrites=["unfold", "resize"]), + "mosaic-1408": dict(dir="denoise", form="a16w16", feed=None, rewrites=["bayer"]), +} -def letterbox(img, edge, pad, norm): - """The app's Letterbox::sample: fit the long side to `edge`, centre, pad.""" - img = ImageOps.exif_transpose(img).convert("RGB") - w, h = img.size - scale = edge / max(w, h) - nw, nh = max(1, round(w * scale)), max(1, round(h * scale)) - img = img.resize((nw, nh), Image.BILINEAR) - canvas = Image.new("RGB", (edge, edge), (pad, pad, pad)) - canvas.paste(img, ((edge - nw) // 2, (edge - nh) // 2)) - x = np.asarray(canvas, dtype=np.float32) # HWC, 0..255 - x = norm(x) - return np.ascontiguousarray(x.transpose(2, 0, 1))[None] # NCHW +# ---- the app's samplers (dr-face Letterbox, dr-segment Letterbox::sample, align.rs) ---- + +def load(p): + return np.asarray(Image.open(p).convert("RGB"), np.float32) / 255.0 -def preprocessing(name, shape): - """Which normalisation this model is fed in the app. - - SCRFD (`dr-face::detect`): `(v - 127.5) / 128`, padded with 114. - ArcFace (`dr-face::embed`): the same, on an aligned 112 crop — a - letterboxed photograph is the wrong distribution, but the embedder is - never quantised (§7), so this is only ever a fallback. - YOLO (`dr-segment`): `v / 255`, padded with 0.5. - """ - edge = shape[-1] - if name.startswith("scrfd") or name.startswith("arcface"): - return edge, 114, lambda x: (x - 127.5) / 128.0 - return edge, 128, lambda x: x / 255.0 +def bilinear(img, sx, sy): + h, w = img.shape[:2] + x0, y0 = np.floor(sx).astype(np.int64), np.floor(sy).astype(np.int64) + fx, fy = (sx - x0)[..., None], (sy - y0)[..., None] + c = lambda a, n: np.clip(a, 0, n - 1) # noqa: E731 - neighbours clamp at the edge + top = img[c(y0, h), c(x0, w)] * (1 - fx) + img[c(y0, h), c(x0 + 1, w)] * fx + bot = img[c(y0 + 1, h), c(x0, w)] * (1 - fx) + img[c(y0 + 1, h), c(x0 + 1, w)] * fx + return top * (1 - fy) + bot * fy -class Photos(CalibrationDataReader): - """One chunk of photographs, fed as the app would feed them.""" +def chw(x): + return np.ascontiguousarray(x.transpose(2, 0, 1))[None].astype(np.float32) - def __init__(self, input_name, paths, edge, pad, norm): - self.name = input_name - self.items = iter(letterbox(Image.open(p), edge, pad, norm) for p in paths) + +def letterbox(img, edge, yolo): + h, w = img.shape[:2] + s = min(edge / w, edge / h) + px, py = (edge - w * s) / 2, (edge - h * s) / 2 + ix, iy = np.meshgrid(np.arange(edge) + 0.5, np.arange(edge) + 0.5) + if yolo: # semantic.rs: no -0.5, pad 0.5, 0..1 + sx, sy = (ix - px) / s, (iy - py) / s + out = bilinear(img, sx, sy) + out[(sx < 0) | (sx >= w) | (sy < 0) | (sy >= h)] = 0.5 + else: # detect.rs: -0.5, pad 114, (v·255 − 127.5)/128 + sx, sy = (ix - px) / s - 0.5, (iy - py) / s - 0.5 + out = (bilinear(img, sx, sy) * 255 - 127.5) / 128 + out[(sx < -0.5) | (sx > w - 0.5) | (sy < -0.5) | (sy > h - 0.5)] = (114 - 127.5) / 128 + return chw(out), (s, px, py) + + +def crop_box(img, x0, y0, bw, bh, ow, oh): + """align.rs crop_box: output (u+.5) → source, −0.5, bilinear, outside black.""" + u, v = np.meshgrid(np.arange(ow) + 0.5, np.arange(oh) + 0.5) + sx, sy = x0 + u * bw / ow - 0.5, y0 + v * bh / oh - 0.5 + out = bilinear(img, sx, sy) + h, w = img.shape[:2] + out[(sx < -1) | (sx > w) | (sy < -1) | (sy > h)] = 0 + return out + + +def scrfd_boxes(outs, s, px, py): + """detect.rs decode: score ≥ 0.5, greedy NMS at 0.4, min side 24 px.""" + fmc = len(outs) // 3 + boxes, scores = [], [] + for i, st in enumerate([8, 16, 32, 64][:fmc]): + sc, bx = outs[i].reshape(-1), outs[fmc + i].reshape(-1, 4) + idx = np.nonzero(sc >= 0.5)[0] + cell = idx // 2 + cx, cy = (cell % (640 // st)) * st, (cell // (640 // st)) * st + boxes.append(np.stack([cx - bx[idx, 0] * st, cy - bx[idx, 1] * st, cx + bx[idx, 2] * st, cy + bx[idx, 3] * st], 1)) + scores.append(sc[idx]) + b, sc = np.concatenate(boxes), np.concatenate(scores) + keep = [] + for i in np.argsort(-sc): + x0 = np.maximum(b[i, 0], b[keep, 0]); y0 = np.maximum(b[i, 1], b[keep, 1]) + x1 = np.minimum(b[i, 2], b[keep, 2]); y1 = np.minimum(b[i, 3], b[keep, 3]) + inter = np.clip(x1 - x0, 0, None) * np.clip(y1 - y0, 0, None) + area = lambda r: (r[..., 2] - r[..., 0]) * (r[..., 3] - r[..., 1]) # noqa: E731 + if not keep or (inter / (area(b[i]) + area(b[keep]) - inter)).max() <= 0.4: + keep.append(i) + b = (b[keep] - [px, py, px, py]) / s + return b[np.minimum(b[:, 2] - b[:, 0], b[:, 3] - b[:, 1]) >= 32] + + +# ---- one calibration input per photograph (or per face), as the app makes it ---- + +def feeds(kind, photos, model): + name = model.get_inputs()[0].name + if kind == "scrfd": + for p in photos: + yield {name: letterbox(load(p), 640, False)[0]} + elif kind == "landmarks": # landmarks.rs: 1.5× the box, square, 0..255 + det = ort.InferenceSession(str(MODELS / "face/scrfd_10g_640.onnx"), providers=["CPUExecutionProvider"]) + for p in photos: + img = load(p) + x, ctx = letterbox(img, 640, False) + for b in scrfd_boxes(det.run(None, {"input.1": x}), *ctx): + cx, cy, side = (b[0] + b[2]) / 2, (b[1] + b[3]) / 2, 1.5 * max(b[2] - b[0], b[3] - b[1]) + yield {name: chw(crop_box(img, cx - side / 2, cy - side / 2, side, side, 192, 192) * 255)} + elif kind == "yolo": + for p in photos: + yield {name: letterbox(load(p), 640, True)[0]} + elif kind == "migan": # migan.rs: ch0 = known − 0.5, ch1–3 = (rgb·2 − 1)·known + rng = np.random.default_rng(7) + for p in photos: + img = load(p) + h, w = img.shape[:2] + e = min(h, w) + sq = Image.fromarray((img[(h - e) // 2:(h + e) // 2, (w - e) // 2:(w + e) // 2] * 255).astype(np.uint8)) + img = np.asarray(sq.resize((512, 512), Image.BILINEAR), np.float32) / 255 + known = np.ones((512, 512), np.float32) + for _ in range(rng.integers(1, 3)): # a panorama's unknown border: a wedge along one edge + side = rng.integers(4) + depth = np.linspace(rng.integers(20, 110), rng.integers(20, 110), 512).astype(int) + edge = np.arange(512)[:, None] < depth[None, :] # [depth, along]: inside the wedge + wedge = edge if side % 2 == 0 else edge[::-1] # top / bottom of a column + known[wedge if side < 2 else wedge.T] = 0 # or left / right of a row + x = np.concatenate([(known - 0.5)[None], ((img * 2 - 1) * known[..., None]).transpose(2, 0, 1)])[None] + yield {name: x.astype(np.float32)} + elif kind == "xfeat": # xfeat.rs: grey 0..1, shrink to fit, top-left, zero pad + _, _, H, W = [d if isinstance(d, int) else 1 for d in model.get_inputs()[0].shape] + for p in photos: + g = load(p).mean(2) # a display-rendered photograph is already the app's (R+G+B)/3 ^ 1/2.2 + h, w = g.shape + s = min(W / w, H / h, 1.0) + if s < 1: + g = np.asarray(Image.fromarray(g).resize((round(w * s), round(h * s)), Image.BOX)) + pad = np.zeros((H, W), np.float32) + pad[:g.shape[0], :g.shape[1]] = g + yield {name: pad[None, None]} + + +class Items(CalibrationDataReader): + def __init__(self, items): + self.it = iter(items) def get_next(self): - x = next(self.items, None) - return None if x is None else {self.name: x} + return next(self.it, None) -# Photographs whose activations are held in memory at once. Every ONNX -# Runtime calibrator keeps each image's whole set of activations until it -# folds them into a range — a gigabyte an image on the 10g detector at 640², -# and folded once at the end, an OOM kill with no message. Folding every -# `CHUNK` images gives ranges identical to folding once (checked on -# scrfd_500m, 129 tensors, no difference) at a bounded cost. -CHUNK = 4 +def calibrate(path, items, cache): + """Min/max ranges in chunks — every ORT calibrator holds all activations + until it folds them, and the others measurably degrade the result.""" + cal = create_calibrator(Path(path), None, augmented_model_path=f"{path}.aug.onnx", + calibrate_method=CalibrationMethod.MinMax) + batch, n = [], 0 + for item in items: + batch.append(item) + n += 1 + if len(batch) == CHUNK: + cal.collect_data(Items(batch)) + batch = [] + if batch: + cal.collect_data(Items(batch)) + save_tensors_data(cal.compute_data(), cache) + os.remove(f"{path}.aug.onnx") + return n -def calibrate(pre, name, photos, cache): - """Min/max ranges over `photos`, written to `cache` for quantize_static. - - Plain min/max: the moving average and the strided option of - `quantize_static` both measured worse than this on held-out proxies, and - the percentile method has no memory bound at all. - """ - import onnxruntime as ort - - s = ort.InferenceSession(pre, providers=["CPUExecutionProvider"]) - i = s.get_inputs()[0] - shape = [d if isinstance(d, int) else 1 for d in i.shape] - edge, pad, norm = preprocessing(name, shape) - calibrator = create_calibrator( - Path(pre), - None, - augmented_model_path=f"{pre}.augmented.onnx", - calibrate_method=CalibrationMethod.MinMax, - ) - for start in range(0, len(photos), CHUNK): - calibrator.collect_data(Photos(i.name, photos[start : start + CHUNK], edge, pad, norm)) - ranges = calibrator.compute_data() - save_tensors_data(ranges, cache) - os.remove(f"{pre}.augmented.onnx") +def downstream(m, start): + names, live = set(), set() + for n in m.graph.node: + if n.name == start or any(i in live for i in n.input): + names.add(n.name) + live.update(n.output) + return sorted(names) def main(): - photo_dir, models = sys.argv[1], sys.argv[2:] - photos = sorted( - p - for ext in ("jpg", "jpeg", "JPG", "JPEG", "png") - for p in glob.glob(os.path.join(photo_dir, "**", f"*.{ext}"), recursive=True) - )[:PHOTOS] - if len(photos) < 20: - sys.exit(f"only {len(photos)} photographs under {photo_dir}; calibration wants dozens") - print(f"==> calibrating on {len(photos)} photographs") + args = sys.argv[1:] + ranges = None + if args[:1] == ["--ranges"]: + ranges, args = args[1], args[2:] + photo_dir = None + else: + photo_dir, args = args[0], args[1:] + names = args or [n for n in TABLE if TABLE[n]["feed"]] + photos = [] + if photo_dir: + photos = sorted(p for e in ("jpg", "jpeg", "JPG", "JPEG", "png") + for p in glob.glob(os.path.join(photo_dir, "**", f"*.{e}"), recursive=True))[:PHOTOS] + if len(photos) < 50: + sys.exit(f"only {len(photos)} photographs under {photo_dir}; calibration wants hundreds") + print(f"==> calibrating on {len(photos)} photographs") - for src in models: - stem, _ = os.path.splitext(src) - name = os.path.basename(src) - out = f"{stem}.int8.onnx" + for stem in names: + spec = TABLE[stem] + src = MODELS / spec["dir"] / f"{stem}.onnx" + out = src.with_name(f"{stem}.{spec['form']}.onnx") + work = src.with_name(f"{stem}.quant-work.onnx") + print(f" {stem} -> {out.name}") m = onnx.load(src) - opset = next((o.version for o in m.opset_import if o.domain in ("", "ai.onnx")), 0) - work = f"{stem}.quant-work.onnx" - if opset < 13: - print(f" {name}: opset {opset} -> 17") - m = version_converter.convert_version(m, 17) + if next(o.version for o in m.opset_import if o.domain in ("", "ai.onnx")) < 13: + m = version_converter.convert_version(m, 17) # per-channel QDQ needs 13 m.ir_version = 8 onnx.save(m, work) - pre = f"{stem}.quant-pre.onnx" - quant_pre_process(work, pre) - cache = f"{stem}.quant-ranges.json" - calibrate(pre, name, photos, cache) - quantize_static( - pre, - out, - None, - quant_format=QuantFormat.QDQ, - per_channel=True, - activation_type=QuantType.QUInt8, - weight_type=QuantType.QInt8, - calibrate_method=CalibrationMethod.MinMax, - calibration_cache_path=cache, - ) - for f in (work, pre, cache): - os.remove(f) - print(f" {out}: {os.path.getsize(out) // 1024} KB") + if spec.get("rewrites"): + onnx.save(htp_graph.rewrite(str(work), spec["rewrites"]), work) + model = ort.InferenceSession(str(work), providers=["CPUExecutionProvider"]) + cache = str(work) + ".ranges" + if spec["feed"] is None: + if not ranges: + sys.exit(f"{stem} is calibrated on mosaics, not photographs: pass --ranges") + cache = ranges + else: + n = calibrate(str(work), feeds(spec["feed"], photos, model), cache) + print(f" {n} calibration inputs") + act, wt = FORMS[spec["form"]] + # The config needs a reader only to exist; the ranges come from `cache`. + zeros = {i.name: np.zeros([d if isinstance(d, int) else 1 for d in i.shape], np.float32) + for i in model.get_inputs()} + cfg = get_qnn_qdq_config(str(work), Items([zeros]), activation_type=act, weight_type=wt, per_channel=True) + exclude = list(cfg.nodes_to_exclude or []) + spec.get("float_nodes", []) + if spec.get("float_from"): + exclude += downstream(onnx.load(work), spec["float_from"]) + quantize_static(str(work), str(out), None, quant_format=cfg.quant_format, + op_types_to_quantize=cfg.op_types_to_quantize, per_channel=True, + activation_type=act, weight_type=wt, nodes_to_exclude=exclude, + calibrate_method=CalibrationMethod.MinMax, extra_options=cfg.extra_options, + calibration_cache_path=cache) + os.remove(work) + if cache != ranges: + os.remove(cache) + print(f" {out.name}: {out.stat().st_size // 1024} KB") if __name__ == "__main__": diff --git a/tools/quantise-models.sh b/tools/quantise-models.sh index 8d67e71..48abe22 100755 --- a/tools/quantise-models.sh +++ b/tools/quantise-models.sh @@ -1,30 +1,34 @@ #!/usr/bin/env bash -# Produce the int8 form of a model for the Hexagon (docs/dev/inference.md §5). +# Produce the Hexagon's form of each model (docs/dev/inference.md §1.5, §5). # -# ./tools/quantise-models.sh PHOTO_DIR MODEL.onnx [MODEL.onnx ...] +# ./tools/quantise-models.sh PHOTO_DIR [MODEL ...] +# ./tools/quantise-models.sh --ranges RANGES.json mosaic-1408 # -# Writes `MODEL.int8.onnx` beside each input: a QDQ graph, per-channel int8 -# weights, uint8 activations — the form QNN's HTP backend takes whole. The -# activations' ranges come from running the f32 model over the photographs in -# PHOTO_DIR, fed exactly as the app feeds them (letterboxed to the model's -# input, the detector's `(x - 127.5) / 128` normalisation), which is why -# this is a release-time step and not something the device does: it needs -# real photographs and, after it, a person reading §10 M2's numbers. +# Writes `.
.onnx` beside each canonical file under models/: a QDQ +# graph from QNN's own quantisation config, per-channel weights, in the form +# the engine's `Rung::form` names for that role — A16W8, A16W16 or int8, each +# the narrowest that held the model's accuracy on the tablet. The activation +# ranges come from running the f32 model over the photographs in PHOTO_DIR, +# fed exactly as the app feeds them (letterbox maths, pads, normalisation, +# face crops through the app's own similarity), which is why this is a +# release-time step and not something the device does. With no MODEL, every +# model in the table. # -# The SCRFD and ArcFace exports are opset 11; per-channel QDQ needs 13, so a -# model below 13 is first upgraded to 17. That changes only the graph's -# spelling, not a weight — and it is what `tools/fix-face-model-shapes.sh` -# will do to the canonical files in the same model release. +# The denoiser is calibrated on noisy mosaics, not photographs: its ranges +# come from darkroom-denoise's precision gate (`--ranges`), computed on a +# smaller tile of the same network — activation ranges do not depend on the +# tile's size, and the tensor names match. +# +# Then measure before shipping: a quantised form is a different network, and +# the numbers in inference.md §1.5 are what each one had to hold. # # A venv per run, like fix-face-model-shapes.sh: the tools are not a build # input and nothing in the tree should have them on its path. set -euo pipefail -if [ "$#" -lt 2 ]; then - sed -n '2,20p' "$0" >&2 +if [ "$#" -lt 1 ]; then + sed -n '2,27p' "$0" >&2 exit 2 fi -PHOTOS="$1"; shift -[ -d "${PHOTOS}" ] || { echo "no such directory: ${PHOTOS}" >&2; exit 1; } WORK="$(mktemp -d -p /var/tmp quantise-models.XXXXXX)" trap 'rm -rf "${WORK}"' EXIT @@ -32,4 +36,4 @@ echo "==> venv in ${WORK}" uv venv --python 3.12 "${WORK}/venv" >/dev/null VIRTUAL_ENV="${WORK}/venv" uv pip install --quiet onnx onnxruntime pillow numpy sympy -exec "${WORK}/venv/bin/python" "$(dirname "$0")/quantise-models.py" "${PHOTOS}" "$@" +exec "${WORK}/venv/bin/python" "$(dirname "$0")/quantise-models.py" "$@"