diff --git a/models/LICENCE.md b/models/LICENCE.md index 83235e8..803bf1a 100644 --- a/models/LICENCE.md +++ b/models/LICENCE.md @@ -1,9 +1,16 @@ # Model weights — licensing -`yolo26n-seg.onnx` is exported from Ultralytics YOLO26n-seg -(`https://huggingface.co/Ultralytics/YOLO26`, `yolo26n-seg.pt`) by -`tools/export-seg-model.sh`. `yolo26n-seg.classes.json` is that checkpoint's -class vocabulary, written out by the same script. +Two Ultralytics checkpoints ship here, both exported by +`tools/export-seg-model.sh`, each with its class vocabulary written out by the +same script: + +| File | Checkpoint | Trained on | Used by | +|---|---|---|---| +| `segment/yolo26n-seg.onnx` | `yolo26n-seg.pt` | COCO, 80 *thing* classes | local adjustments, subject selection | +| `scene/yolo26s-sem-ade20k.onnx` | `yolo26s-sem-ade20k.pt` | ADE20K, 150 classes | the scene tab's per-category grades | + +Both come from `https://huggingface.co/Ultralytics/YOLO26`. The face weights in +`face/` are a separate matter with a separate grant — see `face/README.md`. ## The grant @@ -38,17 +45,27 @@ This was decided deliberately (D14), not arrived at by accident, and classes include the *stuff* categories that matter most in photography — sky, vegetation, water, wall, mountain. -**No such model exists in usable form.** Checked 2026-08-21: Ultralytics ships -YOLO26-seg trained on **COCO**, whose 80 classes are all *things* — person, -dog, car, bird, potted plant — and the one HuggingFace repository claiming a -YOLO/ADE20K combination (`laxmacl/yolov8-ade20k`) is empty. ADE20K semantic -models do exist, but as SegFormer/OneFormer/MaskFormer transformers, not YOLO. +**This was true when written and is not any more.** Checked 2026-08-21, no +YOLO/ADE20K combination existed: Ultralytics shipped YOLO26-seg on **COCO** +only, and the one HuggingFace repository claiming otherwise +(`laxmacl/yolov8-ade20k`) was empty. Re-checked 2026-08-30: Ultralytics now +ships a `semantic` task with ADE20K checkpoints +(`https://docs.ultralytics.com/tasks/semantic`), and `yolo26s-sem-ade20k` is +what `scene/` holds. -So the shipped vocabulary selects **subjects**, not **stuff**. "Select the -person" works; "select the sky" does not come from the model and must come from -the watershed hierarchy instead. That is a narrower arm B than §4 assumed, and -it raises rather than lowers the importance of arm C. +So the two vocabularies divide the work rather than compete: -The loader treats the vocabulary as model metadata rather than compiled-in -knowledge, so adding a stuff-class model later is a file plus a descriptor, not -a code change. +- **`segment/`, COCO, 80 things.** Separates *instances* — clicking one of + three people selects that person. This is what local adjustments need, and a + semantic model cannot do it: it would return one "person" region covering all + three. +- **`scene/`, ADE20K, 150 classes.** Labels every pixel, including the *stuff* + COCO has no word for — sky, vegetation, water, mountain, wall. This is what + the scene tab's per-category grades need, and it does not care that instances + are merged, because a per-category grade applies to the whole category. + +Neither replaces the other. Keeping both is the deliberate choice. + +The loader treats each vocabulary as model metadata rather than compiled-in +knowledge, which is what made adding the second model a file plus a descriptor +rather than a code change — as this document predicted it would be. diff --git a/models/scene/yolo26s-sem-ade20k.classes.json b/models/scene/yolo26s-sem-ade20k.classes.json new file mode 100644 index 0000000..e5357c4 --- /dev/null +++ b/models/scene/yolo26s-sem-ade20k.classes.json @@ -0,0 +1,152 @@ +[ + "wall", + "building", + "sky", + "floor", + "tree", + "ceiling", + "road", + "bed", + "windowpane", + "grass", + "cabinet", + "sidewalk", + "person", + "earth", + "door", + "table", + "mountain", + "plant", + "curtain", + "chair", + "car", + "water", + "painting", + "sofa", + "shelf", + "house", + "sea", + "mirror", + "rug", + "field", + "armchair", + "seat", + "fence", + "desk", + "rock", + "wardrobe", + "lamp", + "bathtub", + "railing", + "cushion", + "base", + "box", + "column", + "signboard", + "chest of drawers", + "counter", + "sand", + "sink", + "skyscraper", + "fireplace", + "refrigerator", + "grandstand", + "path", + "stairs", + "runway", + "case", + "pool table", + "pillow", + "screen door", + "stairway", + "river", + "bridge", + "bookcase", + "blind", + "coffee table", + "toilet", + "flower", + "book", + "hill", + "bench", + "countertop", + "stove", + "palm", + "kitchen island", + "computer", + "swivel chair", + "boat", + "bar", + "arcade machine", + "hovel", + "bus", + "towel", + "light", + "truck", + "tower", + "chandelier", + "awning", + "streetlight", + "booth", + "television receiver", + "airplane", + "dirt track", + "apparel", + "pole", + "land", + "bannister", + "escalator", + "ottoman", + "bottle", + "buffet", + "poster", + "stage", + "van", + "ship", + "fountain", + "conveyor belt", + "canopy", + "washer", + "plaything", + "swimming pool", + "stool", + "barrel", + "basket", + "waterfall", + "tent", + "bag", + "minibike", + "cradle", + "oven", + "ball", + "food", + "step", + "tank", + "trade name", + "microwave", + "pot", + "animal", + "bicycle", + "lake", + "dishwasher", + "screen", + "blanket", + "sculpture", + "hood", + "sconce", + "vase", + "traffic light", + "tray", + "ashcan", + "fan", + "pier", + "crt screen", + "plate", + "monitor", + "bulletin board", + "shower", + "radiator", + "glass", + "clock", + "flag" +] \ No newline at end of file diff --git a/models/scene/yolo26s-sem-ade20k.onnx b/models/scene/yolo26s-sem-ade20k.onnx new file mode 100644 index 0000000..b5d4946 --- /dev/null +++ b/models/scene/yolo26s-sem-ade20k.onnx @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:296cbbf983a7dbd53eb551f4bcb2ce28d6503bb5ddd30a60c4a13b44087e01d7 +size 24921749 diff --git a/tools/export-seg-model.sh b/tools/export-seg-model.sh index 261fad5..de12bce 100755 --- a/tools/export-seg-model.sh +++ b/tools/export-seg-model.sh @@ -1,14 +1,25 @@ #!/usr/bin/env bash -# Re-export the segmentation model that ships in models/ at the repository root. +# Re-export the segmentation models that ship in models/ at the repository root. # # The .onnx is committed (D14), so this is not part of any build — it exists so # the committed artefact is reproducible rather than a binary someone once -# produced and nobody can regenerate. Run it when bumping the model. +# produced and nobody can regenerate. Run it when bumping a model. # -# ./tools/export-seg-model.sh +# ./tools/export-seg-model.sh # instance -> models/segment/ +# ./tools/export-seg-model.sh yolo26s-sem-ade20k # semantic -> models/scene/ # # Requires `uv`. Everything else is fetched into a throwaway venv. # +# ## The two models, and why they are both here +# +# `yolo26n-seg` is COCO instance segmentation: it separates *things*, so +# clicking one of three people selects that person. `yolo26s-sem-ade20k` is +# ADE20K semantic segmentation: it labels every pixel with one of 150 classes +# including the *stuff* — sky, vegetation, water — that COCO has no word for, +# but it merges same-class pixels into one region and so cannot tell those +# three people apart. Neither substitutes for the other; the scene tab wants +# the second and local adjustments want the first. +# # ## Why these export flags # # `dynamic=False` is not a default we failed to change: **tract cannot parse @@ -20,14 +31,38 @@ # `imgsz` square rather than a rectangle matched to 3:2: one graph has to # serve portrait, landscape, square crops and panoramas. A landscape-shaped # graph trades letterbox waste on 3:2 for worse waste on everything else. +# +# ## Why the semantic export is truncated +# +# Ultralytics ends the `-sem-` graph with `Resize -> ArgMax -> Cast`, handing +# back a `[1, 640, 640]` u8 label map. Two reasons that tail is cut here: +# +# 1. **Cost.** The Resize materialises 150 x 640 x 640 x f32 — *246 MB* — and +# ArgMax then reduces across the channel axis, which in NCHW strides +# 409,600 elements per comparison. Measured on one machine it was roughly +# four fifths of total runtime, for work the app throws away. +# 2. **Softness.** ArgMax destroys the per-class scores. The scene tab needs +# them: softmax over the 150 channels, summed within each photographic +# category, gives per-category weights that sum to 1 at every pixel — a +# partition of unity. Feathering those cannot double-grade a boundary, +# where feathering hard labels outward from two adjacent categories does. +# +# The upsample is not information: the graph's true spatial resolution is the +# logit grid (80x80 at imgsz=640), and the app can resample from that itself. set -euo pipefail HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" REPO="$(cd "${HERE}/.." && pwd)" -OUT="${REPO}/models/segment" MODEL="${1:-yolo26n-seg}" IMGSZ="${2:-640}" + +# Semantic models are the scene tab's; instance models are the selection path's. +case "${MODEL}" in + *-sem-*|*-sem) OUT="${REPO}/models/scene"; SEMANTIC=1 ;; + *) OUT="${REPO}/models/segment"; SEMANTIC=0 ;; +esac + # Not `mktemp -d`: the default TMPDIR is `/tmp`, which on most current Linux # distributions is a tmpfs — RAM, sized at half of physical memory. The venv # below pulls torch, several gigabytes of it, and installing that into RAM @@ -41,12 +76,13 @@ cd "${WORK}" uv venv --python 3.12 venv VIRTUAL_ENV="${WORK}/venv" uv pip install ultralytics onnx onnxslim -VIRTUAL_ENV="${WORK}/venv" "${WORK}/venv/bin/python" - "${MODEL}" "${IMGSZ}" <<'PY' +VIRTUAL_ENV="${WORK}/venv" "${WORK}/venv/bin/python" - "${MODEL}" "${IMGSZ}" "${SEMANTIC}" <<'PY' import sys, json +import onnx +from onnx import helper from ultralytics import YOLO -name = sys.argv[1] -imgsz = int(sys.argv[2]) +name, imgsz, semantic = sys.argv[1], int(sys.argv[2]), sys.argv[3] == "1" m = YOLO(f"{name}.pt") path = m.export(format="onnx", opset=17, simplify=True, imgsz=imgsz, dynamic=False) print("ONNX:", path) @@ -57,6 +93,25 @@ print("ONNX:", path) with open("classes.json", "w") as f: json.dump([m.names[i] for i in range(len(m.names))], f, indent=1) print("classes:", len(m.names)) + +if semantic: + # Drop `Resize -> ArgMax -> Cast` and expose the classifier's logits. See + # the header for why. Matched by op type rather than by node name so a + # re-export under a different naming scheme still works, and asserted + # rather than assumed so an upstream graph change fails loudly here + # instead of silently shipping a differently-shaped model. + g = onnx.load(path).graph + tail = [n.op_type for n in g.node[-3:]] + assert tail == ["Resize", "ArgMax", "Cast"], f"unexpected graph tail: {tail}" + logits = g.node[-3].input[0] + del g.node[-3:] + del g.output[:] + g.output.extend([helper.make_tensor_value_info(logits, onnx.TensorProto.FLOAT, None)]) + model = onnx.shape_inference.infer_shapes(helper.make_model(g, opset_imports=[helper.make_opsetid("", 17)])) + onnx.checker.check_model(model) + onnx.save(model, path) + shape = [d.dim_value for d in model.graph.output[0].type.tensor_type.shape.dim] + print("truncated to logits:", logits, shape) PY mkdir -p "${OUT}" @@ -66,4 +121,4 @@ cp "${WORK}/classes.json" "${OUT}/${MODEL}.classes.json" echo "==> wrote:" ls -la "${OUT}" echo -echo "Remember: these weights are AGPL-3.0 (see ${OUT}/LICENCE.md)." +echo "Remember: these weights are AGPL-3.0 (see ${REPO}/models/LICENCE.md)."