#!/usr/bin/env bash # Re-export the segmentation models that ship in models/ at the repository root. # # The .onnx is committed (D14), so this is not part of any build — it exists so # the committed artefact is reproducible rather than a binary someone once # produced and nobody can regenerate. Run it when bumping a model. # # ./tools/export-seg-model.sh # instance -> models/segment/ # ./tools/export-seg-model.sh yolo26s-sem-ade20k # semantic -> models/scene/ # # Requires `uv`. Everything else is fetched into a throwaway venv. # # ## The two models, and why they are both here # # `yolo26n-seg` is COCO instance segmentation: it separates *things*, so # clicking one of three people selects that person. `yolo26s-sem-ade20k` is # ADE20K semantic segmentation: it labels every pixel with one of 150 classes # including the *stuff* — sky, vegetation, water — that COCO has no word for, # but it merges same-class pixels into one region and so cannot tell those # three people apart. Neither substitutes for the other; the scene tab wants # the second and local adjustments want the first. # # ## Why these export flags # # `dynamic=False` is not a default we failed to change: **tract cannot parse # the dynamic-shape graph at all**, failing shape inference on the neck's # Concat. A fixed input shape is a hard requirement of the pure-Rust backend # (see the workspace manifest for why that backend was chosen), and it is what # makes the tiling option in `semantic.rs` the only route to more resolution. # # `imgsz` square rather than a rectangle matched to 3:2: one graph has to # serve portrait, landscape, square crops and panoramas. A landscape-shaped # graph trades letterbox waste on 3:2 for worse waste on everything else. # # ## Why the semantic export is truncated # # Ultralytics ends the `-sem-` graph with `Resize -> ArgMax -> Cast`, handing # back a `[1, 640, 640]` u8 label map. Two reasons that tail is cut here: # # 1. **Cost.** The Resize materialises 150 x 640 x 640 x f32 — *246 MB* — and # ArgMax then reduces across the channel axis, which in NCHW strides # 409,600 elements per comparison. Measured on one machine it was roughly # four fifths of total runtime, for work the app throws away. # 2. **Softness.** ArgMax destroys the per-class scores. The scene tab needs # them: softmax over the 150 channels, summed within each photographic # category, gives per-category weights that sum to 1 at every pixel — a # partition of unity. Feathering those cannot double-grade a boundary, # where feathering hard labels outward from two adjacent categories does. # # The upsample is not information: the graph's true spatial resolution is the # logit grid (80x80 at imgsz=640), and the app can resample from that itself. set -euo pipefail HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" REPO="$(cd "${HERE}/.." && pwd)" MODEL="${1:-yolo26n-seg}" IMGSZ="${2:-640}" # Semantic models are the scene tab's; instance models are the selection path's. case "${MODEL}" in *-sem-*|*-sem) OUT="${REPO}/models/scene"; SEMANTIC=1 ;; *) OUT="${REPO}/models/segment"; SEMANTIC=0 ;; esac # Not `mktemp -d`: the default TMPDIR is `/tmp`, which on most current Linux # distributions is a tmpfs — RAM, sized at half of physical memory. The venv # below pulls torch, several gigabytes of it, and installing that into RAM # either evicts the user's page cache or fails outright with ENOSPC on a # machine that has hundreds of gigabytes of actual disk free. WORK="$(mktemp -d -p "${TMPDIR:-/var/tmp}")" trap 'rm -rf "${WORK}"' EXIT echo "==> exporting ${MODEL} at imgsz=${IMGSZ} in ${WORK}" cd "${WORK}" uv venv --python 3.12 venv VIRTUAL_ENV="${WORK}/venv" uv pip install ultralytics onnx onnxslim VIRTUAL_ENV="${WORK}/venv" "${WORK}/venv/bin/python" - "${MODEL}" "${IMGSZ}" "${SEMANTIC}" <<'PY' import sys, json import onnx from onnx import helper from ultralytics import YOLO name, imgsz, semantic = sys.argv[1], int(sys.argv[2]), sys.argv[3] == "1" m = YOLO(f"{name}.pt") path = m.export(format="onnx", opset=17, simplify=True, imgsz=imgsz, dynamic=False) print("ONNX:", path) # The class names travel with the model rather than being retyped into Rust — # a hand-copied vocabulary is a silent mismatch waiting to happen when the # model is bumped. with open("classes.json", "w") as f: json.dump([m.names[i] for i in range(len(m.names))], f, indent=1) print("classes:", len(m.names)) if semantic: # Drop `Resize -> ArgMax -> Cast` and expose the classifier's logits. See # the header for why. Matched by op type rather than by node name so a # re-export under a different naming scheme still works, and asserted # rather than assumed so an upstream graph change fails loudly here # instead of silently shipping a differently-shaped model. g = onnx.load(path).graph tail = [n.op_type for n in g.node[-3:]] assert tail == ["Resize", "ArgMax", "Cast"], f"unexpected graph tail: {tail}" logits = g.node[-3].input[0] del g.node[-3:] del g.output[:] g.output.extend([helper.make_tensor_value_info(logits, onnx.TensorProto.FLOAT, None)]) model = onnx.shape_inference.infer_shapes(helper.make_model(g, opset_imports=[helper.make_opsetid("", 17)])) onnx.checker.check_model(model) onnx.save(model, path) shape = [d.dim_value for d in model.graph.output[0].type.tensor_type.shape.dim] print("truncated to logits:", logits, shape) PY mkdir -p "${OUT}" cp "${WORK}/${MODEL}.onnx" "${OUT}/${MODEL}.onnx" cp "${WORK}/classes.json" "${OUT}/${MODEL}.classes.json" echo "==> wrote:" ls -la "${OUT}" echo echo "Remember: these weights are AGPL-3.0 (see ${REPO}/models/LICENCE.md)."