From 17106f337074753698454ccaf9c739cb6d0b4d79 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Thu, 30 Jul 2026 18:38:59 +0200 Subject: [PATCH] Adopt the config-driven extractor; project overview README Replaces the copy taken earlier with the newer version from scene-actor-extraction's traceability-tooling branch, which had moved on: it takes per-repo settings from a traceability.toml rather than the CLI flags added here, validates them, and names languages ("rust") rather than making each repo spell out extensions. That is the better design, so the flags go and this becomes the single source. Two fixes on top: - Config discovery searched from the working directory only, so --root pointed at another tree found no traceability.toml and failed with "requirement_types is empty" while a perfectly good config sat in the directory named. That breaks both intended callers: CI passing --root, and a wrapper running the vendored copy. Discovery now starts from --root. - The test suite had not been migrated with the Config refactor and failed on the branch as well as here. All 53 now pass: entry points take a Config, ci_executable moved to the Register which owns tier policy, fixtures write a real traceability.toml so config discovery is exercised rather than bypassed, and the live-register tests take LIVE_REGISTER from the environment since the project home holds no component register of its own. The README becomes a project overview rather than a table of contents: what the problem is, why a paused-frame answer is the wrong question, why gallery data never leaves the instance, and why the manifest server can hold no binary. Co-Authored-By: Claude Opus 5 --- ...ettings.local.json.tmp.271576.c40f99ae1801 | 107 -- README.md | 151 ++- scripts/traceability/extract_traces.py | 1007 +++++++++++------ scripts/traceability/test_extract_traces.py | 245 ++-- scripts/traceability/traceability-gate.sh | 105 +- 5 files changed, 916 insertions(+), 699 deletions(-) delete mode 100644 .claude/settings.local.json.tmp.271576.c40f99ae1801 diff --git a/.claude/settings.local.json.tmp.271576.c40f99ae1801 b/.claude/settings.local.json.tmp.271576.c40f99ae1801 deleted file mode 100644 index b82640d..0000000 --- a/.claude/settings.local.json.tmp.271576.c40f99ae1801 +++ /dev/null @@ -1,107 +0,0 @@ -{ - "permissions": { - "allow": [ - "Bash(python3 -c ' *)", - "WebFetch(domain:docs.turso.tech)", - "Bash(awk '/^```/{n++} END{print \"fences:\",n,\\(n%2==0?\"balanced\":\"UNBALANCED\"\\)}' SPEC.md)", - "Bash(python3 -c \"import h5py; print\\('h5py', h5py.__version__\\)\")", - "Bash(python3 scripts/make_jellyfin_gallery.py --help)", - "Bash(timeout 30 python3 -c ' *)", - "Bash(ps -o pid,etime,cmd -C cmake)", - "Bash(cargo check *)", - "Bash(cargo test *)", - "Bash(cargo clippy *)", - "Bash(cargo deny *)", - "Bash(rustc --version)", - "Bash(cargo metadata *)", - "Bash(python3 -c \"import json,sys; d=json.load\\(sys.stdin\\); print\\(d['packages'][0].get\\('license'\\)\\)\")", - "Bash(timeout 900 cargo install cargo-deny --locked)", - "Bash(timeout 1500 docker build -t jray-server:test .)", - "Bash(timeout 900 docker build -f /tmp/claude-1000/-home-dtourolle-Development-Jray-project/d861f764-2add-4f42-a069-0954ad1f9565/scratchpad/Dockerfile.probe -t jray-probe .)", - "Bash(docker rm *)", - "Bash(docker volume *)", - "Bash(timeout 900 docker build -t jray-server:test .)", - "Bash(docker run *)", - "Bash(xargs -I{} echo \"clippy: {}\")", - "Bash(timeout 900 docker build -q -t jray-server:test .)", - "Bash(awk '{s+=$4} END {print \"total: \" s \" passing\"}')", - "Bash(xargs -I{} echo \"clippy issues: {}\")", - "Bash(timeout 600 dotnet build Jellyfin.Plugin.JRay/Jellyfin.Plugin.JRay.csproj -v q --nologo)", - "Bash(awk '{s+=$4} END {print \"server tests: \" s \" passing\"}')", - "Bash(timeout 300 dotnet build Jellyfin.Plugin.JRay/Jellyfin.Plugin.JRay.csproj -v q --nologo)", - "Bash(cp SPEC.md /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/SPEC.md.bak)", - "Bash(perl -pi -e ' *)", - "Bash(grep -vE \"AR-|DP-|IR-|GR-|VR-|SR-|PR-|H[0-9]|[0-9]{3,}\")", - "Bash(git update-index *)", - "WebFetch(domain:github.com)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/audio-signature status --short)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/audio-signature branch --show-current)", - "Bash(python3 -c \"import pytest; print\\(pytest.__version__\\)\")", - "Bash(python3 -c \"import numpy; print\\('numpy', numpy.__version__\\)\")", - "Read(//usr/lib/cmake/**)", - "Read(//usr/include/**)", - "Bash(grep -n \"^#\\\\{1,3\\\\} \" JRay-public-server/SPEC.md)", - "Bash(grep -n \"^#\\\\{1,3\\\\} \" scene-actor-extraction/docs/SPEC.md)", - "Bash(cargo build *)", - "Bash(python3 gen.py tone.wav)", - "Bash(python3 ref.py tone.wav)", - "Bash(python3 -c \"import json;d=json.load\\(open\\('package.json'\\)\\);print\\(json.dumps\\({k:v for k,v in d.get\\('scripts',{}\\).items\\(\\) if 'trace' in k.lower\\(\\)},indent=2\\)\\)\")", - "Bash(python3 scripts/traceability/extract_traces.py --format json)", - "Bash(python3 scripts/test_extract_traces.py)", - "Bash(python3 -m pytest scripts/traceability/test_extract_traces.py -q)", - "Bash(python3 scripts/traceability/extract_traces.py --format coverage)", - "Bash(python3 scripts/extract_traces.py)", - "Bash(python3 scripts/extract_traces.py --format json)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/traceability-tooling log --oneline -3)", - "Bash(python3 /home/dtourolle/Development/Jray-worktrees/traceability-tooling/scripts/traceability/extract_traces.py --root /home/dtourolle/Development/Jray-project/JRay-public-server --requirements /home/dtourolle/Development/Jray-project/JRay-public-server/docs/requirements.md --system-spec /home/dtourolle/Development/Jray-project/SPEC.md --format coverage)", - "Bash(python3 scripts/traceability/test_extract_traces.py)", - "Bash(sh scripts/traceability/traceability-gate.sh)", - "Bash(python3 -c \"import pyflakes; print\\('pyflakes', pyflakes.__version__\\)\")", - "Bash(timeout 300 dotnet build -v q --nologo)", - "Bash(git check-ignore *)", - "Bash(rm -rf /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full)", - "Bash(cmake -S . -B /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full -DSAE_BUILD_TESTS=ON)", - "Bash(pkg-config --cflags opencv4)", - "Bash(pkg-config --cflags opencv)", - "Bash(awk '{s+=$4} END {print \"tests: \" s}')", - "Bash(python3 -m pyflakes scripts/validation/min_face_size.py)", - "Bash(python3 -m flake8 --select=F scripts/validation/min_face_size.py)", - "Bash(git -c user.name=\"Duncan Tourolle\" -c user.email=\"duncan@tourolle.paris\" commit -q -F -)", - "Bash(cmake -S . -B /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full -DSAE_BUILD_TESTS=ON -DSAE_GEMM_BACKEND=CPU)", - "Bash(python3 -c \"import h5py, numpy, requests, PIL; print\\('deps ok'\\)\")", - "Bash(python3 -m py_compile scripts/validation/min_face_size.py)", - "Bash(git remote *)", - "Bash(python3 et.py --root /home/dtourolle/Development/Jray-project/jRay --requirements /home/dtourolle/Development/Jray-project/jRay/docs/requirements.md --system-spec /home/dtourolle/Development/Jray-project/SPEC.md --format coverage)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/traceability-tooling status --short)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/traceability-tooling log --oneline -2 -- scripts/traceability/)", - "Bash(timeout 240 python3 scripts/stamp_gallery.py --gallery /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/g_plain.h5 --show)", - "Bash(cp /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/g_plain.h5 /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/g_mig.h5)", - "Bash(python3 scripts/stamp_gallery.py --gallery /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/g_mig.h5 --arcface /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/fake.onnx)", - "Bash(timeout 590 cmake -S . -B /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/bt -DSAE_BUILD_TESTS=ON -DCMAKE_BUILD_TYPE=Release)", - "Bash(grep -n \"^#\\\\{1,3\\\\} \\\\|^| ID \\\\|^| Requirement \\\\|^|---\" jRay/docs/requirements.md)", - "Bash(./build-full/tests/sae_tests)", - "Bash(ls /usr/include/catch2/catch_test_macros.hpp 2>/dev/null && echo \"catch2 system\"; ls /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/ 2>/dev/null | head; ls /usr/lib/libCatch2* 2>/dev/null | head)", - "Read(//usr/lib/**)", - "Bash(find /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-build -name libCatch2*)", - "Bash(timeout 590 g++ -std=c++20 -O1 -I/home/dtourolle/Development/Jray-worktrees/gallery-model-binding/src -I/usr/include/opencv5 -I/tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/nlohmann_json-src/single_include -I/tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-src/src -I/tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-build/generated-includes '-DSAE_MODELS_DIR=\"/home/dtourolle/Development/Jray-worktrees/gallery-model-binding/models\"' -o gallery_tests /home/dtourolle/Development/Jray-worktrees/gallery-model-binding/tests/test_gallery_store.cpp /home/dtourolle/Development/Jray-worktrees/gallery-model-binding/src/gallery/gallery_store.cpp /home/dtourolle/Development/Jray-worktrees/gallery-model-binding/src/gallery/embedder_stamp.cpp /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-build/src/libCatch2Main.a /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-build/src/libCatch2.a -lhdf5_cpp -lhdf5 -lopencv_core)", - "Bash(./gallery_tests)", - "Bash(grep -viE \"^\\\\[gallery\\\\]|^ |^$|WARNING \\\\\\(GR-004\\\\\\)|\\\\*\\\\*\\\\*\\\\*\")", - "Bash(python3 scripts/traceability/extract_traces.py --root jRay --requirements jRay/docs/requirements.md --system-spec SPEC.md --types JR --suffixes .cs,.js --scan-roots Jellyfin.Plugin.JRay --format coverage)", - "Bash(python3 scripts/traceability/extract_traces.py --root /home/dtourolle/Development/Jray-project/jRay --requirements /home/dtourolle/Development/Jray-project/jRay/docs/requirements.md --system-spec /home/dtourolle/Development/Jray-project/SPEC.md --types JR --suffixes .cs,.js --scan-roots Jellyfin.Plugin.JRay --format coverage)", - "Bash(cp -r /home/dtourolle/Development/Jray-project/scene-actor-extraction/external/KPN/. external/KPN/)", - "Bash(rm -rf external/KPN/.git)", - "Bash(rm -rf /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/bt)", - "Bash(timeout 590 cmake -S . -B /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/bt -DSAE_BUILD_TESTS=ON -DSAE_GEMM_BACKEND=CPU -DCMAKE_BUILD_TYPE=Release)", - "Bash(python3 scripts/traceability/extract_traces.py --root scene-actor-extraction --requirements scene-actor-extraction/docs/requirements.md --format coverage)", - "Bash(timeout 590 cmake --build /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/bt --target sae_tests -j8)", - "Bash(sh -n scripts/traceability/traceability-gate.sh)", - "Bash(cd /home/dtourolle/Development/Jray-project *)" - ], - "additionalDirectories": [ - "/home/dtourolle/Development/Jray-worktrees/gallery-model-binding/src/gallery", - "/home/dtourolle/Development/Jray-worktrees/min-face-study/scripts/validation", - "/home/dtourolle/Development/Jray-worktrees/audio-signature/tests", - "/home/dtourolle/Development/Jray-worktrees/traceability-tooling/scripts/traceability" - ] - } -} diff --git a/README.md b/README.md index 076f961..fe24d0a 100644 --- a/README.md +++ b/README.md @@ -4,23 +4,36 @@ see who is on screen — for their own library, on their own hardware, with nobody else learning what they own. -This is the project home. It owns the [system specification](SPEC.md) — the -requirements that span more than one component — and the tooling shared between -them. The code lives in three separate repositories, linked below. +This is the project home. It owns the [system specification](SPEC.md), the +requirements that span more than one component, and the tooling they share. The +code lives in three repositories, linked below. --- -## The three components +## The problem -| Repository | Role | Language | -|---|---|---| -| [`scene-actor-extraction`](https://gitea.tourolle.paris/dtourolle/scene-actor-extraction) | Derives presence data from a media file | C++ / Python | -| [`jRay`](https://gitea.tourolle.paris/dtourolle/jRay) | Jellyfin plugin: surfaces it in the player, owns the truth-file format | C# | -| [`JRay-public-server`](https://gitea.tourolle.paris/dtourolle/JRay-public-server) | Exchanges presence data between instances | Rust | +You are watching a film. Someone appears and you know you have seen them +before — but pausing to search breaks the film, and the answer is rarely worth +the interruption. Amazon X-Ray solves this well, and only for Amazon's catalogue. -They ship independently — the plugin to Jellyfin's catalogue, the server as a -binary, the pipeline to a GPU host — which is why they are separate repositories -rather than one monorepo. +Doing the same for a personal library is harder than it looks: + +- **Face recognition on a paused frame answers the wrong question.** In dialogue + the camera is usually on whoever is *not* speaking, so a per-frame answer + reports the other actor absent. X-Ray credits a whole scene's cast for the + scene's duration, and that is the more useful question (SR-002). +- **The compute is real.** Detecting, embedding and tracking every face in a + feature takes GPU time. Doing it once per viewer, for the same film, is waste. +- **The obvious fix leaks.** A service that identifies your library must be told + what your library contains, which recreates the thing being replaced (PR-005). + +JRay's answer: extract presence data locally, share the *timings* rather than +the media or the faces, and make the shared artefact structurally incapable of +carrying anything else. + +--- + +## How it fits together ``` media file ──► extraction ──► truth file (sidecar or pushed) ──► plugin ──► player overlay @@ -30,9 +43,28 @@ media file ──► extraction ──► truth file (sidecar or pushed) ── (Jellyfin + TMDB) ``` -Two axes, deliberately separate: **presence data** flows outward and is -shareable, being timings against public identifiers. **Gallery data** — the -actor reference faces — is built locally and never leaves the instance (SR-005). +| Repository | Role | Language | +|---|---|---| +| [`scene-actor-extraction`](https://gitea.tourolle.paris/dtourolle/scene-actor-extraction) | Derives presence data from a media file | C++ / Python | +| [`jRay`](https://gitea.tourolle.paris/dtourolle/jRay) | Jellyfin plugin: surfaces it in the player, owns the truth-file format | C# | +| [`JRay-public-server`](https://gitea.tourolle.paris/dtourolle/JRay-public-server) | Exchanges presence data between instances | Rust | + +They ship independently — the plugin to Jellyfin's catalogue, the server as a +static binary, the pipeline to a GPU host — which is why they are separate +repositories rather than one monorepo. + +**Two data axes, deliberately separate.** *Presence data* flows outward and is +shareable: it is timings against public TMDB identifiers. *Gallery data* — the +actor reference faces — is built locally from your own Jellyfin instance plus +TMDB, and never leaves the machine (SR-005). Not by export, not by opt-in, not +at all: the capability is what creates the exposure, so it does not exist. + +**The manifest server holds no binary content.** An accepted manifest contains +bounded numbers, regex-constrained identifiers, and references to TMDB persons. +No images, no embeddings, no free-form strings, no extension points (SR-004). +That is what makes it safe for a volunteer to operate an instance, and it is a +property preserved by prohibition — any proposal to ship blobs through it is a +proposal to delete it. --- @@ -49,21 +81,20 @@ git clone git@gitea.tourolle.paris:dtourolle/jRay.git git clone git@gitea.tourolle.paris:dtourolle/JRay-public-server.git ``` -The component directories are `.gitignore`d here, so they sit beside the system -spec without this repository trying to track them. Each has its own README with -build instructions. +Each component has its own README with build instructions. They are +`.gitignore`d here, so they sit beside the system spec without this repository +trying to track them. -### Why not submodules for the components +### Why the components are not submodules A submodule pins a commit. With feature branches and worktrees in flight across the components, every component commit would leave this repository's pointer -stale and its `git status` dirty until someone committed a pointer bump — churn -that buys nothing, since the components are developed together in one directory -anyway. +stale and its `git status` dirty until someone committed a bump — churn that +buys nothing, since the components are developed together in one directory. -The dependency runs the other way instead: **components pull *this* repository -in** for the shared tooling and the system spec, both of which change rarely. -That is the asymmetry submodules suit. +The dependency runs the other way: **each component pulls *this* repository in** +as a submodule, for the system spec and the shared tooling, both of which change +rarely. That is the asymmetry submodules suit. --- @@ -72,22 +103,19 @@ That is the asymmetry submodules suit. | Doc | Owns | |---|---| | [`SPEC.md`](SPEC.md) | **System requirements** — `PR-nnn` project goals, `SR-nnn` cross-component contracts | -| [`CLAUDE.md`](CLAUDE.md) | Working notes and the invariants that must not be violated silently | +| [`CLAUDE.md`](CLAUDE.md) | Working notes, and the invariants that must not be violated silently | | Each repo's `SPEC.md` | That component's software requirements | | Each repo's `docs/requirements.md` | Its stable requirement IDs, status, and verification plan | Read the system spec first. Every component requirement traces up to an `SR-nnn`, and every `SR-nnn` to a `PR-nnn`, so the chain explains *why* a given -piece of code exists. +piece of code exists — and makes it visible when something exists for no stated +reason. --- ## Requirement traceability -Requirements are traceable **up** to the project goal and **down** to the code -implementing them. A requirement nothing traces to is either unnecessary or -unimplemented, and both are worth knowing. - ``` PR-nnn project requirement (SPEC.md §1) — why the system exists └─ SR-nnn system requirement (SPEC.md §3) — what spans components @@ -95,26 +123,65 @@ PR-nnn project requirement (SPEC.md §1) — why the system exists └─ TRACES tag (source) ``` -Tag the code that *satisfies* a requirement: +Tag the code that *satisfies* a requirement — the unit that decides, not every +helper it calls: ```rust -/// TRACES: UR-003 | SR-004 +/// TRACES: UR-003, UR-011 | SR-004 pub fn validate_manifest(m: Jmanifest) -> VResult { … } ``` -A pipe separates requirement *types*; a comma separates IDs within a type. Tag -the unit that decides, not every helper it calls — a tag on every function is -noise, and rots faster than it helps. +A pipe separates requirement *types*; a comma separates IDs within a type. +Tests carry tags too (`UT-nnn`, `IT-nnn`), which is what shows a requirement is +*verified* rather than merely implemented. A deliberate departure from an +invariant is tagged `EXCEPTION:` with its reason — an untagged one is a defect. -The gate reports coverage, orphan tags (an ID no register defines), and untraced -requirements. Two rules it inherits from JellyTau, both learned the hard way: +### Running the gate + +The tooling lives in [`scripts/traceability/`](scripts/traceability/) here and +is vendored into each component as a submodule, so there is **one +implementation**. Each component declares its own taxonomy in a +`traceability.toml` at its root: + +```toml +requirement_types = ["UR", "DR"] +languages = ["rust"] +source_roots = ["src", "tests"] +``` + +```sh +scripts/traceability-gate.sh # from any component +``` + +It reports coverage, orphan tags (an ID no register defines), untraced +requirements, and requirements verifiable only on hardware CI lacks. + +**Two rules inherited from JellyTau, both learned the hard way:** - **Denominators are read from the register at run time, never hardcoded.** A - gate that divides by a frozen literal reported 158% coverage for months and so - could never fail — worse than no gate, because it was trusted. + gate that divided by a frozen literal reported *158% coverage* for months + while the requirement count grew, so its threshold could never trip. A gate + that cannot fail is worse than no gate, because it is trusted. - **Coverage above 100% is a hard failure**, not a pass. It means the - computation is broken, and it is the signal that catches the above - immediately. + computation is broken, and it is the signal that catches the above at once. + +The same reasoning is why a requirement whose only evidence is a test that never +runs is reported as *tagged but unexecuted*, never counted as covered. + +--- + +## Status + +| Component | State | +|---|---| +| `scene-actor-extraction` | Pipeline redesign in progress — presence follows track extent (AR-012), replacing per-frame recognition | +| `jRay` | Truth-file serving and overlay working; manifest-sharing configuration added, fetch path outstanding | +| `JRay-public-server` | Core implemented: 23/32 requirements traced, 189 tests. Audio-tier matching and federation deferred by design | + +**One schema bump is pending across all three repos** (SR-003). It removes +`anneal_sec`, adds `extinction_sec` and `gallery_scope`, gives each window its +belief and identification route, and adds the audio signature. Breaking changes +are batched, so these ship together rather than piecemeal. --- diff --git a/scripts/traceability/extract_traces.py b/scripts/traceability/extract_traces.py index 0b946d0..049d1bf 100755 --- a/scripts/traceability/extract_traces.py +++ b/scripts/traceability/extract_traces.py @@ -1,12 +1,26 @@ #!/usr/bin/env python3 -"""Extract requirement traces from C++/Python sources and report coverage. +"""Extract requirement traces from source and report coverage against a register. -Ported from JellyTau's ``scripts/extract-traces.ts``. That one scans -TypeScript/Svelte/Rust; this repo is C++ and Python, so the scanner is Python -with no third-party dependencies — the CI host must not need a node/bun -toolchain to check traceability. +Ported from JellyTau's ``scripts/extract-traces.ts`` and written in stdlib +Python so no repo needs a bun/node toolchain to check its source comments. -Tag format (see ../../../CLAUDE.md and SPEC.md section 6). A pipe separates +**One implementation, parameterised.** This tool is shared by every JRay +component repo — C++/Python extraction, C# plugin, Rust server — via the +``jray-project`` submodule. Nothing about a single repo is baked in: the +requirement ID prefixes, the source file suffixes, the directories to scan, the +register path and the system-spec path are all configuration. A second copy for +"the other language" is how two implementations start drifting apart, so there +is exactly one. + +Configuration comes from ``traceability.toml`` at the component repo root, from +CLI flags, or both (flags win). ``--print-example-config`` emits the full +annotated schema; the three required keys are:: + + requirement_types = ["AR", "DP", "IR", "GR", "VR"] + languages = ["cpp", "python"] + source_roots = ["src", "tests", "scripts"] + +Tag format (see the system CLAUDE.md and SPEC.md section 6). A pipe separates requirement *types*, a comma separates IDs within a type:: /// TRACES: AR-nnn, AR-mmm | SR-nnn @@ -19,28 +33,32 @@ separately — never silently folded into coverage:: Usage:: - python3 scripts/traceability/extract_traces.py --format coverage - python3 scripts/traceability/extract_traces.py --format json > traces-report.json - python3 scripts/traceability/extract_traces.py --format markdown --markdown-out docs/traceability.md + python3 extract_traces.py --format coverage + python3 extract_traces.py --format json > report.json + python3 extract_traces.py --config path/to/traceability.toml -The CI gate is ``scripts/traceability/traceability-gate.sh``, which wraps this. +The CI gate is ``traceability-gate.sh``, which wraps this. -Two rules inherited from JellyTau's gate repair, both learned the hard way -(JellyTau/docs/specs/traceability-gate-repair.md): +Three rules the gate exists to enforce. The first two are inherited from +JellyTau's gate repair, both learned the hard way +(``JellyTau/docs/specs/traceability-gate-repair.md``): -1. Coverage denominators are read out of ``docs/requirements.md`` at run time. - Never hardcode them. JellyTau's gate divided by frozen literals while the - register grew to 211 requirements; it reported 158% coverage, so its 50% - threshold could never trip. A gate that cannot fail is worse than no gate, - because it is trusted. +1. Coverage denominators are read out of the register at run time. Never + hardcode them. JellyTau's gate divided by frozen literals while the register + grew to 211 requirements; it reported 158% coverage, so its 50% threshold + could never trip. A gate that cannot fail is worse than no gate, because it + is trusted. 2. Coverage above 100% is a hard failure, not a pass. It means the computation is broken, and it is the signal that would have caught (1) immediately. +3. A misconfigured run refuses to report at all. Parsing zero requirements or + scanning zero files prints a failure, not a plausible-looking 0% — the same + family of error as (1), and the one a shared, parameterised tool makes easy + to hit. -One rule specific to this repo: CI runs on an Intel N100 with no discrete GPU. -Requirements whose only verification tier is T4/GPU (or "out of CI") cannot -execute here. They are reported as *tagged but unexecuted* and are excluded -from the covered numerator — counting a test that never runs as coverage is -the same failure mode as the 158% bug. +The GPU-less-CI rule is configuration rather than code: requirements whose +verification tiers all fall outside ``ci_executable_tiers`` are reported as +*tagged but unexecuted* and excluded from the covered numerator. Counting a +test that never runs is the same failure mode as the 158% bug. """ from __future__ import annotations @@ -49,116 +67,372 @@ import argparse import json import re import sys -from dataclasses import dataclass, field +from dataclasses import dataclass, field, replace from datetime import datetime, timezone from pathlib import Path -from typing import Dict, Iterable, List, Optional, Sequence, Set, Tuple - -# Derived from this file's location so a CI checkout at any path works. -SCRIPT_DIR = Path(__file__).resolve().parent -REPO_ROOT = SCRIPT_DIR.parent.parent +from typing import (Dict, FrozenSet, Iterable, List, Optional, Sequence, Set, + Tuple) # -------------------------------------------------------------------------- -# Taxonomy -# -# These are *defaults*, not fixed policy. This tool is shared by every component -# in the project (it is vendored in as a submodule from `jray-project`), and the -# components differ in both their requirement prefixes and their languages: -# -# scene-actor-extraction AR/DP/IR/GR/VR C++, Python -# JRay-public-server UR/DR Rust -# jRay UR/DR C# -# -# `--types` and `--suffixes` override the two that vary. Leaving them hardcoded -# is not a small problem: a repo whose prefixes are absent parses to **zero** -# defined requirements, and one whose language is absent scans **zero** files. -# The gate refuses to report coverage in either state rather than printing a -# misleading 0%, which is the behaviour that surfaced this. +# Defaults. Everything here is overridable; nothing here names a single repo. # -------------------------------------------------------------------------- -#: Types defined in docs/requirements.md. These, and only these, participate -#: in the coverage fraction. Override with --types. -LOCAL_TYPES: Tuple[str, ...] = ("AR", "DP", "IR", "GR", "VR") - #: Test identifiers. A separate taxonomy: a test ID is evidence *for* a -#: requirement, not a requirement. Excluded from the fraction. -TEST_TYPES: Tuple[str, ...] = ("UT", "IT") +#: requirement, not a requirement, so it never enters the coverage fraction. +#: House-wide (SPEC.md section 6), hence a default rather than an argument. +DEFAULT_TEST_TYPES: Tuple[str, ...] = ("UT", "IT") -#: Defined in the system-level SPEC.md, which lives in the project home -#: (`jray-project`) rather than in any component checkout. Recognised in tags, -#: never counted here, and orphan-checked only when --system-spec is given. -EXTERNAL_TYPES: Tuple[str, ...] = ("PR", "SR") +#: Defined in the system-level SPEC.md that ``jray-project`` owns. Recognised +#: in tags everywhere, never counted in any component's fraction, and +#: orphan-checked only when ``system_spec`` points at that file. +DEFAULT_EXTERNAL_TYPES: Tuple[str, ...] = ("PR", "SR") -KNOWN_TYPES: Tuple[str, ...] = LOCAL_TYPES + TEST_TYPES + EXTERNAL_TYPES +#: Tiers a CI host can actually run. The extraction pipeline's CI is a GPU-less +#: Intel N100, so its T4 is excluded; jRay numbers its live-only tier T4 for +#: exactly this reason rather than renumbering the shared constant. +DEFAULT_CI_EXECUTABLE_TIERS: FrozenSet[str] = frozenset({"T1", "T2", "T3", "static"}) -#: Tiers a GPU-less CI host can actually run. See the "Verification strategy" -#: section of docs/requirements.md. -CI_EXECUTABLE_TIERS: Set[str] = {"T1", "T2", "T3", "static"} +DEFAULT_EXCLUDE_DIRS: FrozenSet[str] = frozenset({ + ".git", "__pycache__", "external", "vendor", "build", "site", "dist", + "node_modules", "target", "bin", "obj", "trt_cache", "ort_cache", + ".venv", "venv", ".mypy_cache", ".pytest_cache", +}) -# -------------------------------------------------------------------------- -# Source scanning -# -------------------------------------------------------------------------- +DEFAULT_REQUIREMENTS_PATH = "docs/requirements.md" +DEFAULT_MATRIX_PATH = "docs/traceability.md" +DEFAULT_JSON_PATH = "traces-report.json" +CONFIG_FILENAME = "traceability.toml" -#: Override with --suffixes. Defaults cover the extraction pipeline; the server -#: passes `.rs` and the plugin `.cs`. -SOURCE_SUFFIXES = { - ".cpp", ".cc", ".cxx", ".hpp", ".hxx", ".h", ".cu", ".cuh", # C++ - ".py", # Python +#: Named suffix groups, so a repo declares "Rust" rather than remembering to +#: spell every extension. Additive with an explicit ``source_suffixes`` list. +LANGUAGE_SUFFIXES: Dict[str, FrozenSet[str]] = { + "cpp": frozenset({".cpp", ".cc", ".cxx", ".hpp", ".hxx", ".h", ".cu", ".cuh"}), + "python": frozenset({".py", ".pyi"}), + "rust": frozenset({".rs"}), + "csharp": frozenset({".cs"}), + "javascript": frozenset({".js", ".mjs", ".cjs", ".jsx"}), + "typescript": frozenset({".ts", ".tsx"}), + "svelte": frozenset({".svelte"}), + "go": frozenset({".go"}), + "java": frozenset({".java"}), } -SCAN_ROOTS: Tuple[str, ...] = ("src", "tests", "scripts", "experiments", "eval") + +# -------------------------------------------------------------------------- +# Configuration +# -------------------------------------------------------------------------- + +class ConfigError(Exception): + """A configuration mistake, reported rather than papered over.""" -def configure_taxonomy(local_types: Sequence[str]) -> None: - """Rebind the requirement prefixes this run recognises. +@dataclass(frozen=True) +class Config: + """Everything that differs between component repos. - Module-level rather than threaded through every call site: the constants are - read in a dozen places, and adding a config parameter to each would be a far - larger diff than the change warrants — more risk of a missed site than the - global buys. Called once from main() before any scanning. + The three required fields are exactly the three things that were hardcoded + before this tool was shared: which ID prefixes the repo's register defines, + which files count as source, and where those files live. A repo that gets + any of them wrong parses zero requirements or scans zero files — and the + gate refuses to report rather than printing a misleading 0%. """ - global LOCAL_TYPES, KNOWN_TYPES - LOCAL_TYPES = tuple(local_types) - KNOWN_TYPES = LOCAL_TYPES + TEST_TYPES + EXTERNAL_TYPES + + requirement_types: Tuple[str, ...] + source_suffixes: FrozenSet[str] + source_roots: Tuple[str, ...] + + root: Path = Path(".") + test_types: Tuple[str, ...] = DEFAULT_TEST_TYPES + external_types: Tuple[str, ...] = DEFAULT_EXTERNAL_TYPES + exclude_dirs: FrozenSet[str] = DEFAULT_EXCLUDE_DIRS + ci_executable_tiers: FrozenSet[str] = DEFAULT_CI_EXECUTABLE_TIERS + requirements_path: str = DEFAULT_REQUIREMENTS_PATH + system_spec_path: Optional[str] = None + matrix_path: Optional[str] = DEFAULT_MATRIX_PATH + json_path: Optional[str] = DEFAULT_JSON_PATH + min_coverage: float = 0.0 + allow_orphans: bool = False + source: str = "defaults" + + @property + def known_types(self) -> Tuple[str, ...]: + return (tuple(self.requirement_types) + tuple(self.test_types) + + tuple(self.external_types)) + + def resolve(self, relative: Optional[str]) -> Optional[Path]: + """A configured path, made absolute against the repo root.""" + if relative is None: + return None + path = Path(relative) + return path if path.is_absolute() else (self.root / path) + + def validate(self) -> None: + problems: List[str] = [] + if not self.requirement_types: + problems.append( + "requirement_types is empty - the register defines IDs with " + "some prefix (AR/DP/IR/GR/VR in extraction, JR in jRay, UR/DR " + "in the server) and the tool cannot guess it") + for name, values in (("requirement_types", self.requirement_types), + ("test_types", self.test_types), + ("external_types", self.external_types)): + for value in values: + if not re.fullmatch(r"[A-Z]{2}", value): + problems.append( + f"{name}: {value!r} is not a two-letter uppercase " + "prefix; IDs are PREFIX-nnn") + overlaps = ((set(self.requirement_types) & set(self.test_types)) + | (set(self.requirement_types) & set(self.external_types))) + if overlaps: + problems.append( + f"prefixes {sorted(overlaps)} are declared both as " + "requirement_types and as another taxonomy; a prefix counted " + "in the fraction cannot also be excluded from it") + if not self.source_suffixes: + problems.append( + 'no source suffixes - set `languages` (e.g. ["rust"]) or ' + '`source_suffixes` (e.g. [".rs"])') + for suffix in sorted(self.source_suffixes): + if not suffix.startswith("."): + problems.append(f"source suffix {suffix!r} must start with a dot") + if not self.source_roots: + problems.append( + "no source roots - set `source_roots` to the directories " + "holding this repo's code") + if not 0.0 <= self.min_coverage <= 100.0: + problems.append( + f"min_coverage {self.min_coverage} is not a percentage") + if problems: + raise ConfigError("; ".join(problems)) -def configure_suffixes(suffixes: Sequence[str]) -> None: - """Rebind the file extensions scanned for TRACES tags.""" - global SOURCE_SUFFIXES - SOURCE_SUFFIXES = {s if s.startswith(".") else f".{s}" for s in suffixes} - - -def configure_scan_roots(roots: Sequence[str]) -> None: - """Rebind the directories walked, relative to --root.""" - global SCAN_ROOTS - SCAN_ROOTS = tuple(roots) - -EXCLUDED_DIR_NAMES = { - ".git", "__pycache__", "external", "build", "site", "node_modules", - "trt_cache", "ort_cache", ".venv", "venv", ".mypy_cache", ".pytest_cache", +CONFIG_KEYS = { + "requirement_types", "test_types", "external_types", "languages", + "source_suffixes", "source_roots", "exclude_dirs", "ci_executable_tiers", + "requirements", "system_spec", "matrix", "json_report", "min_coverage", + "allow_orphans", } + +def example_config() -> str: + """The full schema, emitted by ``--print-example-config``.""" + return """\ +# traceability.toml - per-repo configuration for the shared trace extractor. +# Lives at the component repo root; its directory is taken as the repo root. + +# REQUIRED. The ID prefixes this repo's register defines. These, and only +# these, form the coverage fraction. +# scene-actor-extraction: ["AR", "DP", "IR", "GR", "VR"] +# jRay: ["JR"] +# JRay-public-server: ["UR", "DR"] +requirement_types = ["AR", "DP", "IR", "GR", "VR"] + +# REQUIRED (at least one of the two). `languages` names suffix groups; +# `source_suffixes` adds anything else. Known groups: cpp, python, rust, +# csharp, javascript, typescript, svelte, go, java. +languages = ["cpp", "python"] +# source_suffixes = [".inl"] + +# REQUIRED. Directories to scan, relative to the repo root. +source_roots = ["src", "tests", "scripts"] + +# Optional; defaults shown. +# test_types = ["UT", "IT"] # evidence for requirements, not counted +# external_types = ["PR", "SR"] # owned by the system spec, not counted +# requirements = "docs/requirements.md" +# matrix = "docs/traceability.md" +# json_report = "traces-report.json" +# min_coverage = 0.0 +# allow_orphans = false +# exclude_dirs = ["external", "build", "target", "node_modules"] + +# Which verification tiers this repo's CI host can actually execute. A +# requirement whose tiers all fall outside this set is reported as tagged but +# unexecuted, and is never counted as covered. +# ci_executable_tiers = ["T1", "T2", "T3", "static"] + +# The system spec defining PR/SR, vendored as a submodule in each component so +# the check is runnable in CI. Omit to skip PR/SR orphan checking. +# system_spec = "scripts/vendor/jray-project/SPEC.md" +""" + + +def find_config(start: Path) -> Optional[Path]: + """Nearest ``traceability.toml`` at or above ``start``. + + The file's directory defines the repo root, which is what lets the tool be + run from a subdirectory without silently scanning the wrong tree. + """ + current = start.resolve() + for candidate in (current, *current.parents): + path = candidate / CONFIG_FILENAME + if path.is_file(): + return path + return None + + +def _load_toml(path: Path) -> dict: + try: + import tomllib + except ModuleNotFoundError as exc: # pragma: no cover - version dependent + raise ConfigError( + f"reading {path} needs Python 3.11+ for tomllib (this is " + f"{sys.version_info.major}.{sys.version_info.minor}). Either " + "upgrade, or pass --requirement-type/--language/--source-root on " + "the command line instead of using a config file.") from exc + with path.open("rb") as handle: + return tomllib.load(handle) + + +def config_from_dict(data: dict, root: Path, source: str) -> Config: + unknown = sorted(set(data) - CONFIG_KEYS) + if unknown: + # A typo'd key would otherwise leave a required field empty, and the + # user would be debugging "zero requirements" instead of a spelling. + raise ConfigError( + f"unknown key(s) in {source}: {', '.join(unknown)}. " + f"Known keys: {', '.join(sorted(CONFIG_KEYS))}") + + suffixes: Set[str] = set() + for language in data.get("languages", []): + if language not in LANGUAGE_SUFFIXES: + raise ConfigError( + f"unknown language {language!r} in {source}. Known: " + f"{', '.join(sorted(LANGUAGE_SUFFIXES))}") + suffixes |= LANGUAGE_SUFFIXES[language] + suffixes |= set(data.get("source_suffixes", [])) + + return Config( + requirement_types=tuple(data.get("requirement_types", ())), + source_suffixes=frozenset(suffixes), + source_roots=tuple(data.get("source_roots", ())), + root=root, + test_types=tuple(data.get("test_types", DEFAULT_TEST_TYPES)), + external_types=tuple(data.get("external_types", DEFAULT_EXTERNAL_TYPES)), + exclude_dirs=frozenset(data.get("exclude_dirs", DEFAULT_EXCLUDE_DIRS)), + ci_executable_tiers=frozenset( + data.get("ci_executable_tiers", DEFAULT_CI_EXECUTABLE_TIERS)), + requirements_path=data.get("requirements", DEFAULT_REQUIREMENTS_PATH), + system_spec_path=data.get("system_spec"), + matrix_path=data.get("matrix", DEFAULT_MATRIX_PATH), + json_path=data.get("json_report", DEFAULT_JSON_PATH), + min_coverage=float(data.get("min_coverage", 0.0)), + allow_orphans=bool(data.get("allow_orphans", False)), + source=source, + ) + + +def load_config(args: argparse.Namespace, cwd: Optional[Path] = None) -> Config: + """Merge config file and CLI flags, per field. Flags win.""" + cwd = (cwd or Path.cwd()).resolve() + + config_path: Optional[Path] = args.config + if config_path is None and not args.no_config: + # Search from --root when given, falling back to the working directory. + # + # Searching only from cwd is wrong for the two cases this tool is built + # for: CI invoking it with an explicit --root, and a wrapper in a + # consuming repo running the vendored copy. Both name the repo they mean + # and would otherwise get "requirement_types is empty" while a perfectly + # good traceability.toml sat in the directory they pointed at. + config_path = find_config(Path(args.root).resolve() if args.root else cwd) + if config_path is not None and not Path(config_path).is_file(): + raise ConfigError(f"config file not found: {config_path}") + + if config_path is not None: + config_path = Path(config_path).resolve() + config = config_from_dict(_load_toml(config_path), config_path.parent, + str(config_path)) + else: + config = Config(requirement_types=(), source_suffixes=frozenset(), + source_roots=(), root=cwd, source="command line only") + + if args.root is not None: + config = replace(config, root=Path(args.root).resolve()) + + if args.requirement_type: + config = replace(config, requirement_types=tuple(args.requirement_type)) + if args.test_type: + config = replace(config, test_types=tuple(args.test_type)) + if args.external_type: + config = replace(config, external_types=tuple(args.external_type)) + + # Languages and suffixes are additive on top of whatever the file declared, + # so `--language rust` extends rather than silently replacing. + extra: Set[str] = set() + for language in args.language or []: + if language not in LANGUAGE_SUFFIXES: + raise ConfigError( + f"unknown language {language!r}. Known: " + f"{', '.join(sorted(LANGUAGE_SUFFIXES))}") + extra |= LANGUAGE_SUFFIXES[language] + extra |= set(args.source_suffix or []) + if extra: + config = replace(config, source_suffixes=config.source_suffixes | extra) + + if args.source_root: + config = replace(config, source_roots=tuple(args.source_root)) + if args.exclude_dir: + config = replace(config, + exclude_dirs=config.exclude_dirs | set(args.exclude_dir)) + if args.ci_executable_tier: + config = replace(config, + ci_executable_tiers=frozenset(args.ci_executable_tier)) + if args.requirements is not None: + config = replace(config, requirements_path=str(args.requirements)) + if args.system_spec is not None: + config = replace(config, system_spec_path=str(args.system_spec)) + if args.markdown_out is not None: + config = replace(config, matrix_path=str(args.markdown_out)) + if args.json_out is not None: + config = replace(config, json_path=str(args.json_out)) + if args.no_write: + config = replace(config, matrix_path=None, json_path=None) + if args.min_coverage is not None: + config = replace(config, min_coverage=float(args.min_coverage)) + if args.allow_orphans: + config = replace(config, allow_orphans=True) + + config.validate() + return config + + +# -------------------------------------------------------------------------- +# Tag parsing +# -------------------------------------------------------------------------- + TRACES_RE = re.compile(r"TRACES:[ \t]*([^\n]*)") EXCEPTION_RE = re.compile(r"EXCEPTION:[ \t]*([A-Z]{2}-\d{3})[ \t]*([^\n]*)") REQ_ID_RE = re.compile(r"^([A-Z]{2})-(\d{3})$") LEADING_ID_RE = re.compile(r"^([A-Z]{2}-\d{3})\b(.*)$") -#: Something that was *trying* to be a requirement ID. Used to keep the -#: malformed-tag diagnostic quiet when the word "TRACES" merely appears in -#: prose or in this tool's own source, while still catching `AR-12`. +#: Something that was *trying* to be a requirement ID. Keeps the malformed-tag +#: diagnostic quiet when the word "TRACES" merely appears in prose or in this +#: tool's own source, while still catching `AR-12`. ID_ATTEMPT_RE = re.compile(r"[A-Za-z]{2}-\d") -# Comment terminators that can trail a tag on the same line. +#: Comment terminators that can trail a tag on the same line. COMMENT_TERMINATORS = ("*/", "-->", '"""', "'''") DECL_PATTERNS = [ - re.compile(r"^\s*(?:async\s+)?def\s+\w+"), - re.compile(r"^\s*class\s+\w+"), + re.compile(r"^\s*(?:async\s+)?def\s+\w+"), # Python + re.compile(r"^\s*(?:pub(?:\([^)]*\))?\s+)?(?:async\s+)?" + r"(?:unsafe\s+)?(?:extern\s+\"[^\"]*\"\s+)?fn\s+\w+"), # Rust + re.compile(r"^\s*(?:pub(?:\([^)]*\))?\s+)?" + r"(?:impl|trait|mod|type)\b"), # Rust re.compile(r"^\s*(?:template\s*<[^;]*>\s*)?" - r"(?:struct|class|enum(?:\s+class)?|union|namespace)\s+\w+"), - re.compile(r"^\s*(?:static|inline|constexpr|virtual|explicit|friend)\b"), - re.compile(r"^\s*[A-Za-z_][\w:<>,\s\*&]*\s+[A-Za-z_~][\w:]*\s*\([^)]*\)"), + r"(?:struct|class|enum(?:\s+class)?|union|namespace|" + r"interface|record)\s+\w+"), # C++/C# + re.compile(r"^\s*(?:public|private|protected|internal|static|inline|" + r"constexpr|virtual|explicit|friend|override|abstract|sealed|" + r"partial|extern)\b"), + re.compile(r"^\s*[A-Za-z_][\w:<>,\s\*&\.\[\]]*\s+[A-Za-z_~][\w:]*\s*\([^)]*\)"), ] +#: Attribute/annotation lines: C# ``[HttpGet("x")]``, Rust ``#[derive(...)]``, +#: Java ``@Override``. A tag above a decorated declaration must attribute to +#: the declaration, not to its decoration. +ATTRIBUTE_RE = re.compile(r"^\s*(?:\[[^\]]*\]|#\s*\[|@\w+)") + @dataclass class TraceEntry: @@ -262,12 +536,16 @@ def parse_traces_tag(value: str) -> Tuple[List[List[str]], List[str]]: def find_context(lines: Sequence[str], index: int, window: int = 12) -> str: """Best-effort name of the declaration a tag belongs to. - Searches both directions, because the two languages put the tag on opposite - sides of the thing it describes: a C++ ``/// TRACES:`` sits *above* the + Searches both directions, because languages put the tag on opposite sides + of the thing it describes: a C++/C#/Rust ``///`` tag sits *above* the declaration, while a Python tag usually sits *inside* the docstring, below - the ``def``. Whichever declaration is nearer wins. + the ``def``. Whichever declaration is nearer wins. Attribute lines + (``[HttpGet]``, ``#[derive]``) are skipped so a tag above a decorated + method attributes to the method rather than to its decoration. """ def matches(line: str) -> bool: + if ATTRIBUTE_RE.match(line): + return False return any(p.match(line) for p in DECL_PATTERNS) down: Optional[Tuple[int, str]] = None @@ -288,7 +566,6 @@ def find_context(lines: Sequence[str], index: int, window: int = 12) -> str: up = (offset, lines[i]) break - best = None if down and up: best = down if down[0] <= up[0] else up else: @@ -298,25 +575,19 @@ def find_context(lines: Sequence[str], index: int, window: int = 12) -> str: return best[1].strip().rstrip("{").strip()[:120] or "Unknown" -def iter_source_files(root: Path, - scan_roots: Optional[Iterable[str]] = None) -> List[Path]: - """Every source file, of the configured suffixes, under the scan roots. - - `scan_roots` defaults to `None` rather than to `SCAN_ROOTS` directly: a - default argument is bound once at import, so it would capture the value from - before `configure_scan_roots()` ran and silently ignore the override. - """ +def iter_source_files(config: Config) -> List[Path]: + """Every source file under the configured roots, by configured suffix.""" found: List[Path] = [] - for rel in (SCAN_ROOTS if scan_roots is None else scan_roots): - base = root / rel + for rel in config.source_roots: + base = config.root / rel if not base.is_dir(): continue for path in sorted(base.rglob("*")): if not path.is_file(): continue - if path.suffix not in SOURCE_SUFFIXES: + if path.suffix not in config.source_suffixes: continue - if any(part in EXCLUDED_DIR_NAMES for part in path.parts): + if any(part in config.exclude_dirs for part in path.parts): continue found.append(path) return found @@ -330,10 +601,12 @@ class ScanResult: diagnostics: Diagnostics -def scan_files(files: Sequence[Path], root: Path) -> ScanResult: +def scan_files(files: Sequence[Path], config: Config) -> ScanResult: traces: List[TraceEntry] = [] exceptions: List[ExceptionEntry] = [] diags = Diagnostics() + root = config.root + known = set(config.known_types) for path in files: try: @@ -352,14 +625,12 @@ def scan_files(files: Sequence[Path], root: Path) -> ScanResult: if cut != -1: reason = reason[:cut] reason = reason.strip() - entry = ExceptionEntry( + exceptions.append(ExceptionEntry( file=rel, line=index + 1, requirement=exc.group(1), - reason=reason, context=find_context(lines, index), - ) - exceptions.append(entry) + reason=reason, context=find_context(lines, index))) if not reason: - # CLAUDE.md: an exception is only agreed if the reason is - # recorded. An unexplained one is an undocumented defect. + # An exception is only *agreed* if the reason is recorded; + # an unexplained one is an undocumented defect. diags.exceptions_without_reason.append( {"file": rel, "line": index + 1, "requirement": exc.group(1)}) @@ -372,12 +643,13 @@ def scan_files(files: Sequence[Path], root: Path) -> ScanResult: ids = [i for g in groups for i in g] if not ids: # No IDs at all. Only worth reporting if something in the text - # was trying to be one — otherwise every sentence containing + # was trying to be one - otherwise every sentence containing # the word would be flagged, and a diagnostic nobody can act on # is how real diagnostics get ignored. if junk and ID_ATTEMPT_RE.search(match.group(1)): diags.malformed_tags.append( - {"file": rel, "line": index + 1, "text": match.group(1).strip()}) + {"file": rel, "line": index + 1, + "text": match.group(1).strip()}) continue if junk: diags.malformed_tags.append( @@ -390,11 +662,10 @@ def scan_files(files: Sequence[Path], root: Path) -> ScanResult: diags.mixed_type_groups.append( {"file": rel, "line": index + 1, "group": list(group)}) for req in ids: - if req.split("-")[0] not in KNOWN_TYPES: + if req.split("-")[0] not in known: diags.unknown_id_types.append( {"file": rel, "line": index + 1, "id": req}) - # Deduplicate within one tag while preserving order. seen: Set[str] = set() unique = [i for i in ids if not (i in seen or seen.add(i))] traces.append(TraceEntry(file=rel, line=index + 1, @@ -406,7 +677,7 @@ def scan_files(files: Sequence[Path], root: Path) -> ScanResult: # -------------------------------------------------------------------------- -# The register: docs/requirements.md is the authoritative denominator +# The register: requirements.md is the authoritative denominator # -------------------------------------------------------------------------- @dataclass @@ -418,17 +689,6 @@ class Requirement: status: str = "" tiers: Set[str] = field(default_factory=set) - @property - def ci_executable(self) -> bool: - """True when at least one verification tier can run on the CI host. - - A requirement with no tier recorded is *unknown*, not unexecutable — it - is reported so the register gets fixed, and it is not penalised here. - """ - if not self.tiers: - return True - return bool(self.tiers & CI_EXECUTABLE_TIERS) - @property def tier_known(self) -> bool: return bool(self.tiers) @@ -436,6 +696,10 @@ class Requirement: @dataclass class Register: + #: The ID prefixes this register defines, from configuration. + types: Tuple[str, ...] = () + #: Verification tiers the consuming repo's CI host can execute. + ci_tiers: FrozenSet[str] = DEFAULT_CI_EXECUTABLE_TIERS requirements: Dict[str, Requirement] = field(default_factory=dict) withdrawn: Dict[str, Requirement] = field(default_factory=dict) @@ -447,14 +711,35 @@ class Register: def total(self) -> int: return len(self.requirements) + @property + def uses_tiers(self) -> bool: + """Whether this register assigns verification tiers at all. + + A repo with no tier tables is not 'missing' tiers; it does not use the + mechanism. Warning about every requirement there would be noise that + trains people to ignore the warning that matters. + """ + return any(r.tier_known for r in self.requirements.values()) + + def is_ci_executable(self, req_id: str) -> bool: + """True when at least one verification tier can run on the CI host. + + A requirement with no tier recorded is *unknown*, not unexecutable - it + is reported so the register gets fixed, and it is not penalised here. + """ + req = self.requirements[req_id] + if not req.tiers: + return True + return bool(req.tiers & self.ci_tiers) + def count(self, req_type: str) -> int: return sum(1 for i in self.requirements if i.startswith(req_type + "-")) def ci_executable_ids(self) -> Set[str]: - return {i for i, r in self.requirements.items() if r.ci_executable} + return {i for i in self.requirements if self.is_ci_executable(i)} def unexecutable_ids(self) -> Set[str]: - return {i for i, r in self.requirements.items() if not r.ci_executable} + return {i for i in self.requirements if not self.is_ci_executable(i)} def tier_unknown_ids(self) -> Set[str]: return {i for i, r in self.requirements.items() if not r.tier_known} @@ -464,7 +749,7 @@ class Register: SPEC.md section 6 asks the gate to report these: a requirement serving no stated goal is scope creep, and it is invisible unless something - looks. A cell naming a section (`§4`) counts as a parent — the point is + looks. A cell naming a section (`§4`) counts as a parent - the point is that *something* was recorded, not that it was an ID. """ out = set() @@ -476,7 +761,11 @@ class Register: def _row_cells(line: str) -> List[str]: - return [c.strip() for c in line.strip().strip("|").split("|")] + r"""Split a markdown table row, honouring escaped ``\|`` inside cells.""" + placeholder = "\x00" + stripped = line.strip().replace("\\|", placeholder) + cells = stripped.strip("|").split("|") + return [c.strip().replace(placeholder, "|") for c in cells] def _is_separator(cells: Sequence[str]) -> bool: @@ -512,7 +801,7 @@ def _is_register_header(header: Sequence[str]) -> bool: Deliberately narrow. The per-requirement verification plan is also keyed on ``ID`` but has no ``Requirement`` column, and the tier summary table's first - column holds comma lists and ranges — neither defines requirements, and + column holds comma lists and ranges - neither defines requirements, and counting their rows would inflate the denominator. Prose mentions and ``Traces to`` references are excluded for the same reason: a naive scan for ``AR-\\d{3}`` over the whole file counts every reference as a definition. @@ -524,7 +813,7 @@ def _is_register_header(header: Sequence[str]) -> bool: def _is_tier_header(header: Sequence[str]) -> bool: - """Either of the two tables that assign verification tiers.""" + """Either of the two table shapes that assign verification tiers.""" if len(header) < 2: return False lowered = [c.lower() for c in header] @@ -543,7 +832,7 @@ def _column(header: Sequence[str], name: str, cells: Sequence[str]) -> str: def parse_tiers(cell: str) -> Set[str]: """Read a tier cell such as ``T1 + T4``, ``**T2**``, or ``Out of CI``.""" text = cell.replace("*", "").strip().lower() - tiers = {"T" + m.group(1) for m in re.finditer(r"\bt([1-4])\b", text)} + tiers = {"T" + m.group(1) for m in re.finditer(r"\bt(\d)\b", text)} if "out of ci" in text or "not in ci" in text: tiers.add("out-of-ci") if "manual" in text: @@ -556,7 +845,7 @@ def parse_tiers(cell: str) -> Set[str]: def expand_id_spec(spec: str, defined: Set[str]) -> List[str]: """Expand the ID cell of a tier table into concrete, defined IDs. - Handles every shape the register actually uses: ``AR-002``, + Handles every shape the registers actually use: ``AR-002``, ``AR-001, AR-005, AR-006``, ``AR-007 ... AR-017`` (with the unicode ellipsis), ``AR-009/010``, and ``DP-*``. Expansion is intersected with the defined set, so a range can never invent a requirement that does not exist. @@ -592,13 +881,15 @@ def expand_id_spec(spec: str, defined: Set[str]) -> List[str]: return [i for i in out if i in defined] -def parse_register(markdown: str) -> Register: - """Build the register from ``docs/requirements.md``. +def parse_register(markdown: str, config: Config) -> Register: + """Build the register from the repo's requirements.md. Two passes: definitions first, because tier rows use ranges and wildcards that can only be expanded against a known ID set. """ - register = Register() + register = Register(types=tuple(config.requirement_types), + ci_tiers=frozenset(config.ci_executable_tiers)) + definable = set(config.requirement_types) | set(config.test_types) for header, cells in _iter_table_rows(markdown): if not _is_register_header(header) or not cells: @@ -606,7 +897,7 @@ def parse_register(markdown: str) -> Register: first = cells[0].replace("*", "").strip() if not REQ_ID_RE.match(first): continue - if first.split("-")[0] not in LOCAL_TYPES + TEST_TYPES: + if first.split("-")[0] not in definable: continue req = Requirement( id=first, @@ -640,8 +931,10 @@ def parse_register(markdown: str) -> Register: return register -def read_register(path: Path) -> Register: - return parse_register(path.read_text(encoding="utf-8")) +def read_register(config: Config) -> Register: + path = config.resolve(config.requirements_path) + assert path is not None + return parse_register(path.read_text(encoding="utf-8"), config) # -------------------------------------------------------------------------- @@ -679,22 +972,22 @@ def compute_coverage(traced_ids: Iterable[str], register: Register) -> Coverage: Three exclusions, each of which is a way the number could otherwise lie: * Using the raw traced count as the numerator is what lets a ratio exceed - 100% — a tag naming a deleted or mistyped requirement would count as + 100% - a tag naming a deleted or mistyped requirement would count as covered. Those land in ``orphaned`` instead, so they get fixed rather than silently counted or silently dropped. - * A requirement whose only verification tier is T4/GPU cannot run on this - CI host at all. It is reported in ``unexecuted`` and is not covered: - treating "has a test that never runs" as passing reports success the gate - cannot substantiate. - * UT/IT test IDs and the system-level PR/SR IDs are different taxonomies - with their own registers, so they neither count nor orphan here. + * A requirement whose verification tiers all fall outside what the CI host + can execute cannot be verified there at all. It is reported in + ``unexecuted`` and is not covered: treating "has a test that never runs" + as passing reports success the gate cannot substantiate. + * Test IDs and system-level IDs are separate taxonomies with their own + registers, so they neither count nor orphan here. """ - traced = {i for i in traced_ids if i.split("-")[0] in LOCAL_TYPES} + traced = {i for i in traced_ids if i.split("-")[0] in register.types} defined = register.ids orphaned = sorted(traced - defined) matched = traced & defined - unexecuted = sorted(i for i in matched if not register.requirements[i].ci_executable) + unexecuted = sorted(i for i in matched if not register.is_ci_executable(i)) covered = sorted(matched - set(unexecuted)) total = len(defined) @@ -721,28 +1014,26 @@ def compute_coverage(traced_ids: Iterable[str], register: Register) -> Coverage: @dataclass class Report: timestamp: str - root: Path + config: Config register: Register scan: ScanResult coverage: Coverage by_type: Dict[str, List[str]] requirement_map: Dict[str, List[TraceEntry]] external_orphans: List[str] = field(default_factory=list) - #: Policy, set by the caller: whether an orphan tag fails the run. - allow_orphans: bool = False -def build_report(root: Path, register: Register, scan: ScanResult, +def build_report(config: Config, register: Register, scan: ScanResult, system_ids: Optional[Set[str]] = None) -> Report: requirement_map: Dict[str, List[TraceEntry]] = {} - seen_by_type: Dict[str, Set[str]] = {t: set() for t in KNOWN_TYPES} + seen_by_type: Dict[str, Set[str]] = {t: set() for t in config.known_types} seen_by_type["OTHER"] = set() for entry in scan.traces: for req in entry.requirements: requirement_map.setdefault(req, []).append(entry) prefix = req.split("-")[0] - seen_by_type[prefix if prefix in KNOWN_TYPES else "OTHER"].add(req) + seen_by_type[prefix if prefix in seen_by_type else "OTHER"].add(req) coverage = compute_coverage(requirement_map.keys(), register) @@ -750,11 +1041,11 @@ def build_report(root: Path, register: Register, scan: ScanResult, if system_ids is not None: external_orphans = sorted( i for i in requirement_map - if i.split("-")[0] in EXTERNAL_TYPES and i not in system_ids) + if i.split("-")[0] in config.external_types and i not in system_ids) return Report( timestamp=datetime.now(timezone.utc).isoformat(timespec="seconds"), - root=root, + config=config, register=register, scan=scan, coverage=coverage, @@ -764,16 +1055,20 @@ def build_report(root: Path, register: Register, scan: ScanResult, ) -def parse_system_spec(markdown: str) -> Set[str]: - """IDs of PR/SR requirements defined in the umbrella SPEC.md. +def parse_system_spec(markdown: str, + types: Sequence[str] = DEFAULT_EXTERNAL_TYPES) -> Set[str]: + """IDs of the system requirements defined in the umbrella SPEC.md. - They are defined as headings (``### SR-001 - ...``) and as bolded leading - table cells (``| **PR-001** | ... |``), so accept both. + They appear as headings (``### SR-001 - ...``) and as bolded leading table + cells (``| **PR-001** | ... |``), so accept both. """ + alternation = "|".join(re.escape(t) for t in types) + heading_re = re.compile(rf"^#{{1,6}}\s+\**((?:{alternation})-\d{{3}})\**\b") + cell_re = re.compile(rf"(?:{alternation})-\d{{3}}") ids: Set[str] = set() for line in markdown.split("\n"): stripped = line.strip() - heading = re.match(r"^#{1,6}\s+\**((?:PR|SR)-\d{3})\**\b", stripped) + heading = heading_re.match(stripped) if heading: ids.add(heading.group(1)) continue @@ -781,50 +1076,11 @@ def parse_system_spec(markdown: str) -> Set[str]: cells = _row_cells(stripped) if cells: first = cells[0].replace("*", "").strip() - if re.fullmatch(r"(?:PR|SR)-\d{3}", first): + if cell_re.fullmatch(first): ids.add(first) return ids -def report_to_json_dict(report: Report) -> dict: - register = report.register - defined = {t: register.count(t) for t in LOCAL_TYPES} - defined["total"] = register.total - - requirement_detail = {} - for req_id, req in sorted(register.requirements.items()): - entries = report.requirement_map.get(req_id, []) - requirement_detail[req_id] = { - "requirement": req.text, - "tracesTo": req.traces_to, - "priority": req.priority, - "status": req.status, - "tiers": sorted(req.tiers), - "ciExecutable": req.ci_executable, - "taggedIn": sorted({e.file for e in entries}), - "state": _requirement_state(req, bool(entries)), - } - - return { - "timestamp": report.timestamp, - "totalFiles": len(report.scan.files), - "totalTraces": len(report.scan.traces), - "totalExceptions": len(report.scan.exceptions), - "requirements": {k: [e.to_dict() for e in v] - for k, v in sorted(report.requirement_map.items())}, - "byType": report.by_type, - "defined": defined, - "coverage": report.coverage.to_dict(), - "gpuOnlyRequirements": sorted(register.unexecutable_ids()), - "parentlessRequirements": sorted(register.parentless_ids()), - "withdrawn": sorted(register.withdrawn), - "exceptions": [e.to_dict() for e in report.scan.exceptions], - "externalOrphans": report.external_orphans, - "diagnostics": report.scan.diagnostics.to_dict(), - "requirementDetail": requirement_detail, - } - - def per_type_stats(report: Report) -> Dict[str, Tuple[int, int, int]]: """``{type: (covered, unexecuted, defined)}``. @@ -836,7 +1092,7 @@ def per_type_stats(report: Report) -> Dict[str, Tuple[int, int, int]]: covered = set(report.coverage.covered) unexecuted = set(report.coverage.unexecuted) stats: Dict[str, Tuple[int, int, int]] = {} - for req_type in LOCAL_TYPES: + for req_type in report.register.types: prefix = req_type + "-" stats[req_type] = ( len([i for i in covered if i.startswith(prefix)]), @@ -846,36 +1102,93 @@ def per_type_stats(report: Report) -> Dict[str, Tuple[int, int, int]]: return stats -def _requirement_state(req: Requirement, tagged: bool) -> str: +def _requirement_state(register: Register, req_id: str, tagged: bool) -> str: if not tagged: return "untagged" - if not req.ci_executable: + if not register.is_ci_executable(req_id): return "tagged-unexecuted" return "covered" +def report_to_json_dict(report: Report) -> dict: + register = report.register + config = report.config + defined = {t: register.count(t) for t in register.types} + defined["total"] = register.total + + requirement_detail = {} + for req_id, req in sorted(register.requirements.items()): + entries = report.requirement_map.get(req_id, []) + requirement_detail[req_id] = { + "requirement": req.text, + "tracesTo": req.traces_to, + "priority": req.priority, + "status": req.status, + "tiers": sorted(req.tiers), + "ciExecutable": register.is_ci_executable(req_id), + "taggedIn": sorted({e.file for e in entries}), + "state": _requirement_state(register, req_id, bool(entries)), + } + + return { + "timestamp": report.timestamp, + # Echoed so a report can be read without guessing how it was produced - + # a shared tool's output is ambiguous otherwise. + "config": { + "source": config.source, + "root": str(config.root), + "requirementTypes": list(config.requirement_types), + "testTypes": list(config.test_types), + "externalTypes": list(config.external_types), + "sourceSuffixes": sorted(config.source_suffixes), + "sourceRoots": list(config.source_roots), + "requirements": config.requirements_path, + "systemSpec": config.system_spec_path, + "ciExecutableTiers": sorted(config.ci_executable_tiers), + "minCoverage": config.min_coverage, + }, + "totalFiles": len(report.scan.files), + "totalTraces": len(report.scan.traces), + "totalExceptions": len(report.scan.exceptions), + "requirements": {k: [e.to_dict() for e in v] + for k, v in sorted(report.requirement_map.items())}, + "byType": report.by_type, + "defined": defined, + "coverage": report.coverage.to_dict(), + "unexecutableRequirements": sorted(register.unexecutable_ids()), + "parentlessRequirements": sorted(register.parentless_ids()), + "withdrawn": sorted(register.withdrawn), + "exceptions": [e.to_dict() for e in report.scan.exceptions], + "externalOrphans": report.external_orphans, + "diagnostics": report.scan.diagnostics.to_dict(), + "requirementDetail": requirement_detail, + } + + # -------------------------------------------------------------------------- # Output formats # -------------------------------------------------------------------------- def generate_markdown(report: Report) -> str: register = report.register + config = report.config cov = report.coverage out: List[str] = [] add = out.append + req_link = Path(config.requirements_path).name + add("# Requirements traceability matrix") add("") add("") - add("") + add("") add("") add(f"**Generated:** {report.timestamp}") add("") - add("Denominators are read from [`requirements.md`](requirements.md) at run " - "time, never hardcoded. Coverage counts a requirement only when it is " - "tagged in source **and** has a verification tier this CI host can " - "execute — CI is an Intel N100 with no discrete GPU.") + add(f"Denominators are read from [`{req_link}`]({req_link}) at run time, " + "never hardcoded. Coverage counts a requirement only when it is tagged " + "in source **and** has a verification tier this repo's CI host can " + f"execute (`{', '.join(sorted(config.ci_executable_tiers))}`).") add("") add("## Summary") @@ -890,7 +1203,7 @@ def generate_markdown(report: Report) -> str: add(f"| **Coverage** | **{cov.percent}%** ({len(cov.covered)}/{cov.total}) |") add(f"| Coverage of CI-executable scope | {cov.ci_percent}% " f"({len(cov.covered)}/{cov.ci_executable_total}) |") - add(f"| Tagged but unexecuted in CI (T4/GPU) | {len(cov.unexecuted)} |") + add(f"| Tagged but unexecuted in CI | {len(cov.unexecuted)} |") add(f"| Orphan tags | {len(cov.orphaned)} |") add("") @@ -901,7 +1214,7 @@ def generate_markdown(report: Report) -> str: for req_type, (covered, unexecuted, defined_n) in per_type_stats(report).items(): add(f"| {req_type} | {covered} | {unexecuted} | {defined_n} |") add("") - for req_type in TEST_TYPES + EXTERNAL_TYPES: + for req_type in tuple(config.test_types) + tuple(config.external_types): tagged = report.by_type.get(req_type, []) if tagged: add(f"- **{req_type}** tags present (separate taxonomy, not counted " @@ -911,9 +1224,9 @@ def generate_markdown(report: Report) -> str: unexecutable = sorted(register.unexecutable_ids()) add("## Not executable in CI") add("") - add("CI runs on an Intel N100 with no discrete GPU. These requirements have " - "no verification tier that can run here, so a tag on them is evidence " - "of *intent*, not of verification. They are never counted as covered.") + add("These requirements have no verification tier this repo's CI host can " + "run, so a tag on them is evidence of *intent*, not of verification. " + "They are never counted as covered.") add("") if unexecutable: add("| ID | Tiers | Tagged in source | Requirement |") @@ -928,13 +1241,13 @@ def generate_markdown(report: Report) -> str: add("") if cov.unexecuted: add(f"**Tagged but unexecuted:** {', '.join(cov.unexecuted)} — a test " - "exists and is tagged, but only a GPU host can run it. Report " + "exists and is tagged, but this CI host cannot run it. Report " "those runs separately.") add("") add("## Orphan tags") add("") - add("A tag naming an ID `requirements.md` does not define. This is what " + add(f"A tag naming an ID `{req_link}` does not define. This is what " "renumbering produces, and what a typo produces.") add("") if cov.orphaned: @@ -955,16 +1268,13 @@ def generate_markdown(report: Report) -> str: "something looks.") add("") parentless = sorted(register.parentless_ids()) - if parentless: - add(", ".join(f"`{i}`" for i in parentless)) - else: - add("_None._") + add(", ".join(f"`{i}`" for i in parentless) if parentless else "_None._") add("") add("## Recorded exceptions") add("") add("Deliberate, documented departures from an invariant " - "(`EXCEPTION: AR-nnn `). Reported separately and never counted " + "(`EXCEPTION: XX-nnn `). Reported separately and never counted " "as coverage — an exception is a decision to be reviewed, not evidence " "a requirement is met.") add("") @@ -983,15 +1293,15 @@ def generate_markdown(report: Report) -> str: add("") add("| ID | Status | Tier | Traces to | Trace state | Tagged in | Requirement |") add("|---|---|---|---|---|---|---|") - for req_id, req in sorted(register.requirements.items(), - key=lambda kv: (LOCAL_TYPES.index(kv[0].split("-")[0]) - if kv[0].split("-")[0] in LOCAL_TYPES - else 99, kv[0])): + order = {t: n for n, t in enumerate(register.types)} + for req_id, req in sorted( + register.requirements.items(), + key=lambda kv: (order.get(kv[0].split("-")[0], 99), kv[0])): entries = report.requirement_map.get(req_id, []) state = {"covered": "covered", - "tagged-unexecuted": "tagged, unexecuted (T4/GPU)", + "tagged-unexecuted": "tagged, unexecuted", "untagged": "untagged"}[ - _requirement_state(req, bool(entries))] + _requirement_state(register, req_id, bool(entries))] files = ", ".join(f"`{f}`" for f in sorted({e.file for e in entries})) or "-" tiers = ", ".join(sorted(req.tiers)) or "unset" add(f"| {req_id} | {_truncate(req.status, 20) or '-'} | {tiers} " @@ -1041,7 +1351,7 @@ def generate_markdown(report: Report) -> str: def _source_link(file_rel: str) -> str: - """Link from docs/traceability.md back to a source file at the repo root.""" + """Link from the matrix (in docs/) back to a source file at the repo root.""" return "../" + file_rel @@ -1053,9 +1363,10 @@ def _truncate(text: str, limit: int = 70) -> str: return text.replace("|", "\\|") -def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, int]: +def format_coverage_report(report: Report) -> Tuple[str, int]: """Human-readable gate output plus the exit code it implies.""" register = report.register + config = report.config cov = report.coverage lines: List[str] = [] add = lines.append @@ -1063,22 +1374,34 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add("Requirement traceability") add("=" * 72) - add(f"Source files scanned : {len(report.scan.files)}") + add(f"Config : {config.source}") + add(f"Repo root : {config.root}") + add(f"Requirement types : {', '.join(config.requirement_types)}") + add(f"Source files scanned : {len(report.scan.files)} " + f"({', '.join(sorted(config.source_suffixes))} under " + f"{', '.join(config.source_roots)})") add(f"TRACES tags found : {len(report.scan.traces)}") add(f"EXCEPTION tags found : {len(report.scan.exceptions)}") add("") - # Self-checks. With a low threshold these are what make the gate mean - # something: a parser that silently returns nothing would otherwise report - # 0/0 and pass. + # Self-checks. These are what make the gate mean something at a low + # threshold, and what makes a misconfigured shared tool loud instead of + # silent: a repo whose requirement_types or source_suffixes are wrong parses + # nothing and scans nothing, and a plausible-looking 0% would hide it. if register.total == 0: failures.append( - "requirements.md parsed to ZERO requirements - the register parser " - "is broken or the file moved. Refusing to report coverage.") + f"{config.requirements_path} parsed to ZERO requirements. Either " + "the path is wrong, the register parser is broken, or " + f"requirement_types ({', '.join(config.requirement_types) or 'none'}) " + "does not match the prefixes it defines. Refusing to report " + "coverage.") if not report.scan.files: failures.append( - "no source files were scanned - the scan roots do not exist. " - "Refusing to report coverage against an empty tree.") + "no source files were scanned. Either source_roots " + f"({', '.join(config.source_roots) or 'none'}) do not exist, or the " + f"suffixes ({', '.join(sorted(config.source_suffixes)) or 'none'}) " + "do not match this repo's languages. Refusing to report coverage " + "against an empty tree.") add("Coverage by type (covered / defined):") for req_type, (covered, unexecuted, defined_n) in per_type_stats(report).items(): @@ -1088,18 +1411,18 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add(f"Overall : {len(cov.covered)} / {cov.total} ({cov.percent}%)") add(f"CI scope : {len(cov.covered)} / {cov.ci_executable_total} " f"({cov.ci_percent}%) [excludes {cov.total - cov.ci_executable_total} " - "requirement(s) no GPU-less host can verify]") + "requirement(s) this CI host cannot verify]") add("") unexecutable = sorted(register.unexecutable_ids()) if unexecutable: - add(f"Not executable on this CI host (T4/GPU or out-of-CI): " - f"{len(unexecutable)}") + add("Not executable on this CI host (tiers outside " + f"{', '.join(sorted(config.ci_executable_tiers))}): {len(unexecutable)}") add(f" {', '.join(unexecutable)}") if cov.unexecuted: add("") - add("TAGGED BUT UNEXECUTED - a test exists and is tagged, but only a " - "GPU host can run it.") + add("TAGGED BUT UNEXECUTED - a test exists and is tagged, but this CI " + "host cannot run it.") add(" These are NOT counted as covered:") for req_id in cov.unexecuted: tiers = ", ".join(sorted(register.requirements[req_id].tiers)) @@ -1114,12 +1437,18 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add(f" {', '.join(parentless)}") add("") - if cov.tier_unknown: + if cov.tier_unknown and register.uses_tiers: + # Only meaningful where the register uses tiers at all; otherwise every + # requirement is listed and the warning trains people to ignore it. add(f"WARNING: {len(cov.tier_unknown)} requirement(s) have no " - "verification tier in requirements.md; they are counted as " + "verification tier in the register; they are counted as " "CI-executable by default. Add them to the verification plan:") add(f" {', '.join(cov.tier_unknown)}") add("") + elif register.total and not register.uses_tiers: + add("Note: this register assigns no verification tiers, so nothing is " + "excluded as unexecutable.") + add("") if report.scan.exceptions: add(f"Recorded invariant exceptions: {len(report.scan.exceptions)}") @@ -1129,7 +1458,8 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add("") if cov.orphaned: - add("ORPHAN TAGS - traced in source, not defined in requirements.md:") + add("ORPHAN TAGS - traced in source, not defined in " + f"{config.requirements_path}:") for req_id in cov.orphaned: where = ", ".join(f"{e.file}:{e.line}" for e in report.requirement_map[req_id]) @@ -1138,7 +1468,7 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add("") if report.external_orphans: - add("ORPHAN SYSTEM TAGS - PR/SR IDs the system SPEC.md does not define:") + add("ORPHAN SYSTEM TAGS - IDs the system spec does not define:") add(f" {', '.join(report.external_orphans)}") add("") @@ -1168,13 +1498,14 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in f"covered ({len(cov.covered)}) exceeds defined ({cov.total}) - " "the gate is miscomputing.") - if cov.orphaned and not report.allow_orphans: + if cov.orphaned and not config.allow_orphans: failures.append( f"{len(cov.orphaned)} orphan tag(s): {', '.join(cov.orphaned)}") - if cov.percent < min_coverage: + if cov.percent < config.min_coverage: failures.append( - f"coverage ({cov.percent}%) is below the minimum ({min_coverage}%)") + f"coverage ({cov.percent}%) is below the minimum " + f"({config.min_coverage}%)") if failures: add("FAILED:") @@ -1182,7 +1513,7 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add(f" - {reason}") return "\n".join(lines) + "\n", 1 - add(f"OK: coverage {cov.percent}% >= minimum {min_coverage}%, " + add(f"OK: coverage {cov.percent}% >= minimum {config.min_coverage}%, " f"{len(cov.orphaned)} orphan tag(s)") return "\n".join(lines) + "\n", 0 @@ -1194,94 +1525,114 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in def build_arg_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( prog="extract_traces.py", - description=__doc__.split("\n")[0], + description="Extract requirement traces and report coverage. Shared by " + "every JRay component repo; configured per repo via " + f"{CONFIG_FILENAME} or the flags below.", formatter_class=argparse.RawDescriptionHelpFormatter, ) parser.add_argument("--format", choices=("json", "markdown", "coverage"), default="coverage", help="output written to stdout (default: coverage)") - parser.add_argument("--root", type=Path, default=REPO_ROOT, - help="repository root to scan (default: derived from " - "this script's location)") - parser.add_argument("--requirements", type=Path, default=None, - help="register path (default: /docs/requirements.md)") - parser.add_argument("--system-spec", type=Path, default=None, - help="optional umbrella SPEC.md defining PR/SR IDs; " - "enables orphan checking for those types") - parser.add_argument("--json-out", type=Path, default=None, - help="also write the JSON report here") - parser.add_argument("--markdown-out", type=Path, default=None, - help="also write the markdown matrix here") - parser.add_argument("--min-coverage", type=float, default=0.0, - help="minimum coverage percent; below it the run fails " - "(default: 0, i.e. the correctness checks gate but " - "the percentage does not)") - parser.add_argument("--allow-orphans", action="store_true", - help="report orphan tags without failing") - parser.add_argument("--types", default=None, metavar="UR,DR", - help="comma-separated requirement prefixes defined in " - "this repo's register (default: " - f"{','.join(LOCAL_TYPES)}). A repo whose prefixes " - "are absent parses to zero requirements") - parser.add_argument("--suffixes", default=None, metavar=".rs,.cs", - help="comma-separated file extensions to scan " - "(default: the C++/Python set). A repo whose " - "language is absent scans zero files") - parser.add_argument("--scan-roots", default=None, metavar="src,tests", - help="comma-separated directories to walk, relative to " - f"--root (default: {','.join(SCAN_ROOTS)})") + parser.add_argument("--config", type=Path, default=None, + help=f"path to {CONFIG_FILENAME} (default: the nearest " + "one at or above the working directory)") + parser.add_argument("--no-config", action="store_true", + help="ignore any config file; use flags only") + parser.add_argument("--print-example-config", action="store_true", + help=f"print an annotated {CONFIG_FILENAME} and exit") + parser.add_argument("--root", type=Path, default=None, + help="repository root (default: the config file's " + "directory, else the working directory)") + + group = parser.add_argument_group("what this repo defines") + group.add_argument("--requirement-type", action="append", metavar="XX", + help="ID prefix this repo's register defines; repeatable. " + "Replaces the configured list") + group.add_argument("--test-type", action="append", metavar="XX", + help="test ID prefix, excluded from coverage; repeatable") + group.add_argument("--external-type", action="append", metavar="XX", + help="system-spec ID prefix, excluded from coverage; " + "repeatable") + + group = parser.add_argument_group("what this repo scans") + group.add_argument("--language", action="append", + choices=sorted(LANGUAGE_SUFFIXES), + help="named suffix group; repeatable, additive") + group.add_argument("--source-suffix", action="append", metavar=".EXT", + help="extra source suffix; repeatable, additive") + group.add_argument("--source-root", action="append", metavar="DIR", + help="directory to scan; repeatable. Replaces the " + "configured list") + group.add_argument("--exclude-dir", action="append", metavar="NAME", + help="directory name to skip anywhere in the tree; " + "repeatable, additive") + + group = parser.add_argument_group("paths") + group.add_argument("--requirements", type=Path, default=None, + help=f"register path (default: {DEFAULT_REQUIREMENTS_PATH})") + group.add_argument("--system-spec", type=Path, default=None, + help="SPEC.md defining the external types; enables " + "orphan checking for them") + group.add_argument("--json-out", type=Path, default=None, + help=f"JSON report path (default: {DEFAULT_JSON_PATH})") + group.add_argument("--markdown-out", type=Path, default=None, + help=f"matrix path (default: {DEFAULT_MATRIX_PATH})") + group.add_argument("--no-write", action="store_true", + help="do not write report files, only print") + + group = parser.add_argument_group("policy") + group.add_argument("--min-coverage", type=float, default=None, + help="minimum coverage percent; below it the run fails") + group.add_argument("--allow-orphans", action="store_true", + help="report orphan tags without failing") + group.add_argument("--ci-executable-tier", action="append", metavar="TIER", + help="verification tier this CI host can run; " + "repeatable. Replaces the configured set") return parser -def _split_csv(value: Optional[str]) -> Optional[List[str]]: - if value is None: - return None - items = [v.strip() for v in value.split(",") if v.strip()] - return items or None - - def main(argv: Optional[Sequence[str]] = None) -> int: args = build_arg_parser().parse_args(argv) - # Rebind the per-repo taxonomy before anything reads it. - if types := _split_csv(args.types): - configure_taxonomy(types) - if suffixes := _split_csv(args.suffixes): - configure_suffixes(suffixes) - if scan_roots := _split_csv(args.scan_roots): - configure_scan_roots(scan_roots) + if args.print_example_config: + print(example_config(), end="") + return 0 - root = args.root.resolve() - req_path = args.requirements or (root / "docs" / "requirements.md") - if not req_path.is_file(): + try: + config = load_config(args) + except ConfigError as exc: + print(f"FAILED: {exc}", file=sys.stderr) + return 2 + + req_path = config.resolve(config.requirements_path) + if req_path is None or not req_path.is_file(): print(f"FAILED: requirements register not found at {req_path}", file=sys.stderr) return 2 - register = read_register(req_path) - files = iter_source_files(root) - scan = scan_files(files, root) + register = read_register(config) + files = iter_source_files(config) + scan = scan_files(files, config) system_ids = None - if args.system_spec: - if not args.system_spec.is_file(): - print(f"FAILED: --system-spec not found at {args.system_spec}", - file=sys.stderr) + spec_path = config.resolve(config.system_spec_path) + if spec_path is not None: + if not spec_path.is_file(): + print(f"FAILED: system spec not found at {spec_path}", file=sys.stderr) return 2 - system_ids = parse_system_spec(args.system_spec.read_text(encoding="utf-8")) + system_ids = parse_system_spec(spec_path.read_text(encoding="utf-8"), + config.external_types) - report = build_report(root, register, scan, system_ids) - report.allow_orphans = args.allow_orphans + report = build_report(config, register, scan, system_ids) json_text = json.dumps(report_to_json_dict(report), indent=2, sort_keys=False) markdown_text = generate_markdown(report) - if args.json_out: - args.json_out.parent.mkdir(parents=True, exist_ok=True) - args.json_out.write_text(json_text, encoding="utf-8") - if args.markdown_out: - args.markdown_out.parent.mkdir(parents=True, exist_ok=True) - args.markdown_out.write_text(markdown_text, encoding="utf-8") + for target, text in ((config.resolve(config.json_path), json_text), + (config.resolve(config.matrix_path), markdown_text)): + if target is not None: + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(text, encoding="utf-8") if args.format == "json": print(json_text) @@ -1290,7 +1641,7 @@ def main(argv: Optional[Sequence[str]] = None) -> int: print(markdown_text, end="") return 0 - text, code = format_coverage_report(report, args.min_coverage) + text, code = format_coverage_report(report) print(text, end="") return code diff --git a/scripts/traceability/test_extract_traces.py b/scripts/traceability/test_extract_traces.py index d965e75..f105fd8 100755 --- a/scripts/traceability/test_extract_traces.py +++ b/scripts/traceability/test_extract_traces.py @@ -43,6 +43,27 @@ TAG = "TRA" + "CES:" EXC = "EXCEP" + "TION:" + +# -------------------------------------------------------------------------- +# Test configuration +# +# Every entry point takes a Config since the tool became shared. These helpers +# keep each test stating only the field it varies. +# -------------------------------------------------------------------------- + +def _config(root=".", **kw) -> "et.Config": + """A Config with the extraction repo's shape, overridable per test.""" + fields = dict( + requirement_types=("AR", "DP", "IR", "GR", "VR"), + source_suffixes=frozenset(et.LANGUAGE_SUFFIXES["cpp"] + | et.LANGUAGE_SUFFIXES["python"]), + source_roots=("src", "tests", "scripts", "experiments", "eval"), + root=Path(root), + ) + fields.update(kw) + return et.Config(**fields) + + # -------------------------------------------------------------------------- # Tag parsing # -------------------------------------------------------------------------- @@ -127,11 +148,11 @@ def test_scans_cpp_and_python_but_not_vendored_or_non_source(): with tempfile.TemporaryDirectory() as tmp: root = Path(tmp) _write_tree(root) - files = et.iter_source_files(root) + files = et.iter_source_files(_config(root)) names = sorted(f.name for f in files) assert names == ["gallery.py", "tracker.hpp"], names - scan = et.scan_files(files, root) + scan = et.scan_files(files, _config(root)) traced = sorted({i for t in scan.traces for i in t.requirements}) assert traced == ["AR-012", "AR-013", "GR-001", "SR-002", "SR-005"] @@ -141,7 +162,7 @@ def test_context_is_found_below_a_cpp_tag_and_above_a_python_tag(): with tempfile.TemporaryDirectory() as tmp: root = Path(tmp) _write_tree(root) - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) contexts = {t.file: t.context for t in scan.traces} assert "TrackRegistry" in contexts["src/tracker.hpp"] assert "build_gallery" in contexts["scripts/gallery.py"] @@ -164,7 +185,7 @@ def test_exception_tag_is_captured_with_its_reason(): root = Path(tmp) (root / "src").mkdir(parents=True) (root / "src" / "outlier.cpp").write_text(EXCEPTION_SOURCE, encoding="utf-8") - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) assert len(scan.exceptions) == 1 exc = scan.exceptions[0] assert exc.requirement == "AR-024" @@ -179,11 +200,11 @@ def test_exception_is_never_counted_as_coverage(): root = Path(tmp) (root / "src").mkdir(parents=True) (root / "src" / "outlier.cpp").write_text(EXCEPTION_SOURCE, encoding="utf-8") - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) assert scan.traces == [] register = et.parse_register( "| ID | Requirement | Status |\n|---|---|---|\n" - "| AR-024 | Always the calibrated probability | Planned |\n") + "| AR-024 | Always the calibrated probability | Planned |\n", _config()) cov = et.compute_coverage( [i for t in scan.traces for i in t.requirements], register) assert cov.covered == [] @@ -197,7 +218,7 @@ def test_exception_without_a_reason_is_reported(): (root / "src").mkdir(parents=True) (root / "src" / "bare.cpp").write_text( f"// {EXC} AR-024\n", encoding="utf-8") - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) assert len(scan.exceptions) == 1 assert scan.diagnostics.exceptions_without_reason @@ -210,7 +231,7 @@ def test_mixed_type_group_is_reported(): (root / "src").mkdir(parents=True) (root / "src" / "a.cpp").write_text( f"// {TAG} AR-001, SR-002\n", encoding="utf-8") - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) assert scan.diagnostics.mixed_type_groups @@ -225,7 +246,7 @@ def test_counts_a_well_formed_table_row_as_a_defined_requirement(): md = REGISTER_HEADER + ( "| AR-001 | Detect faces in sampled frames | SR-002 | High | Done |\n" "| AR-002 | Minimum face size 66x66 px | SR-002 | High | Planned |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.count("AR") == 2 assert register.count("GR") == 0 assert register.total == 2 @@ -237,7 +258,7 @@ def test_does_not_count_ids_that_appear_only_in_the_traces_to_column(): md = REGISTER_HEADER + ( "| GR-001 | Build gallery from library cast | SR-001, SR-005 | High | Done |\n" "| GR-002 | Incremental merge refresh | PR-003 | High | Done |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.count("GR") == 2 assert register.ids == {"GR-001", "GR-002"} @@ -246,7 +267,7 @@ def test_does_not_count_ids_mentioned_in_prose(): md = ("Some prose explaining that AR-005 relates to GR-001 and VR-003.\n\n" + REGISTER_HEADER + "| AR-005 | Align to 112x112 | SR-002 | High | Done |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.ids == {"AR-005"} @@ -260,7 +281,7 @@ def test_does_not_count_the_verification_plan_table_as_definitions(): + "| ID | Tier | Test asserts | Edge cases to cover |\n|---|---|---|---|\n" + "| AR-001 | T3 | Detector returns plausible boxes | smoke only |\n" + "| AR-099 | T1 | Something not in the register | - |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.ids == {"AR-001"} assert register.total == 1 @@ -271,7 +292,7 @@ def test_deduplicates_an_id_listed_in_two_definition_tables(): + "\n" + REGISTER_HEADER + "| AR-001 | Detect faces | SR-002 | High | Done |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.total == 1 @@ -281,7 +302,7 @@ def test_withdrawn_requirements_leave_the_denominator(): md = REGISTER_HEADER + ( "| AR-001 | Detect faces | SR-002 | High | Done |\n" "| AR-002 | Superseded mechanism | SR-002 | High | Withdrawn |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.ids == {"AR-001"} assert "AR-002" in register.withdrawn @@ -291,8 +312,8 @@ def test_the_denominator_is_live_adding_a_row_lowers_coverage(): # more requirement defined => a lower percentage, mechanically. base = REGISTER_HEADER + "| AR-001 | A | SR-002 | High | Done |\n" grown = base + "| AR-002 | B | SR-002 | High | Planned |\n" - before = et.compute_coverage(["AR-001"], et.parse_register(base)) - after = et.compute_coverage(["AR-001"], et.parse_register(grown)) + before = et.compute_coverage(["AR-001"], et.parse_register(base, _config())) + after = et.compute_coverage(["AR-001"], et.parse_register(grown, _config())) assert before.percent == 100.0 assert after.percent == 50.0 assert after.total == 2 @@ -301,7 +322,7 @@ def test_the_denominator_is_live_adding_a_row_lowers_coverage(): def test_register_captures_the_row_fields_not_just_the_id(): md = REGISTER_HEADER + ( "| AR-012 | Presence follows track extent | **SR-002** | High | Planned |\n") - req = et.parse_register(md).requirements["AR-012"] + req = et.parse_register(md, _config()).requirements["AR-012"] assert req.text == "Presence follows track extent" assert req.traces_to == "**SR-002**" assert req.status == "Planned" @@ -328,7 +349,7 @@ def test_tier_assignment_handles_lists_ranges_and_wildcards(): "| AR-002 … AR-004 | **T2** | replay |\n" "| AR-006 | T1 + T4 | mixed |\n" "| VR-* | Out of CI | studies |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.requirements["AR-001"].tiers == {"T3"} assert register.requirements["AR-003"].tiers == {"T2"} assert register.requirements["AR-006"].tiers == {"T1", "T4"} @@ -337,7 +358,7 @@ def test_tier_assignment_handles_lists_ranges_and_wildcards(): def test_a_range_cannot_invent_a_requirement_the_register_lacks(): md = TIER_REGISTER + "\n" + _tier_table("| AR-001 … AR-050 | T2 | wide |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.total == 12 assert "AR-050" not in register.ids @@ -346,7 +367,7 @@ def test_slash_shorthand_in_the_verification_plan_expands(): md = TIER_REGISTER + "\n" + ( "| ID | Tier | Test asserts | Edge cases |\n|---|---|---|---|\n" "| AR-009/008 | T2 | Cut shifts weighting | cut with same people |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.requirements["AR-008"].tiers == {"T2"} assert register.requirements["AR-009"].tiers == {"T2"} @@ -358,15 +379,15 @@ def test_tiers_from_both_tables_are_unioned_not_overwritten(): md = TIER_REGISTER + "\n" + _tier_table("| AR-006 | T4 | GPU host only |\n") + "\n" + ( "| ID | Tier | Test asserts | Edge cases |\n|---|---|---|---|\n" "| AR-006 | T1 + T4 | GEMM equals reference loop | small input in CI |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.requirements["AR-006"].tiers == {"T1", "T4"} - assert register.requirements["AR-006"].ci_executable + assert register.is_ci_executable("AR-006") def test_a_t4_only_requirement_is_not_ci_executable(): md = TIER_REGISTER + "\n" + _tier_table("| AR-007 | **T4** | GPU only |\n") - register = et.parse_register(md) - assert not register.requirements["AR-007"].ci_executable + register = et.parse_register(md, _config()) + assert not register.is_ci_executable("AR-007") assert register.unexecutable_ids() == {"AR-007"} @@ -379,12 +400,12 @@ def test_a_requirement_tracing_up_to_nothing_is_reported(): "| AR-002 | Parent is a section | §4 | Medium | Planned |\n" "| AR-003 | Serves nothing stated | - | Low | Planned |\n" "| AR-004 | Blank cell | | Low | Planned |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.parentless_ids() == {"AR-003", "AR-004"} def test_a_requirement_with_no_tier_is_unknown_not_unexecutable(): - register = et.parse_register(TIER_REGISTER) + register = et.parse_register(TIER_REGISTER, _config()) assert register.tier_unknown_ids() == register.ids assert register.unexecutable_ids() == set() @@ -402,7 +423,7 @@ COVERAGE_REGISTER = et.parse_register( + "\n" + _tier_table("| AR-001, AR-002 | T2 | replay |\n" "| GR-001 | T1 | bookkeeping |\n" - "| AR-027 | **T4** | GPU host only |\n")) + "| AR-027 | **T4** | GPU host only |\n"), _config()) def test_coverage_is_the_intersection_of_traced_and_defined(): @@ -510,10 +531,30 @@ def _fixture_repo(tmp: str, source: str) -> Path: (root / "src").mkdir(parents=True, exist_ok=True) (root / "docs" / "requirements.md").write_text(FIXTURE_REGISTER, encoding="utf-8") (root / "src" / "pipeline.cpp").write_text(source, encoding="utf-8") + (root / et.CONFIG_FILENAME).write_text( + 'requirement_types = ["AR", "DP", "IR", "GR", "VR"]\n' + 'languages = ["cpp", "python"]\n' + 'source_roots = ["src", "tests", "scripts", "experiments", "eval"]\n', + encoding="utf-8") return root def _run_gate(root: Path, *extra: str): + """Run the CLI against `root`, supplying the per-repo config it now needs. + + The tool no longer guesses a taxonomy: a repo declares its prefixes and + languages, and the gate refuses to run without them rather than reporting a + misleading zero. So a gate test must provide them too — written as a real + `traceability.toml`, which exercises the config-discovery path rather than + bypassing it. + """ + config = root / et.CONFIG_FILENAME + if not config.exists(): + config.write_text( + 'requirement_types = ["AR", "DP", "IR", "GR", "VR"]\n' + 'languages = ["cpp", "python"]\n' + 'source_roots = ["src", "tests", "scripts", "experiments", "eval"]\n', + encoding="utf-8") buffer = io.StringIO() with redirect_stdout(buffer): code = et.main(["--root", str(root), "--format", "coverage", *extra]) @@ -595,11 +636,11 @@ def test_gate_hard_fails_on_an_impossible_ratio(): # reported as a pass. Forced here by handing the reporter a poisoned value. with tempfile.TemporaryDirectory() as tmp: root = _fixture_repo(tmp, f"// {TAG} AR-001\nint main() {{}}\n") - register = et.read_register(root / "docs" / "requirements.md") - scan = et.scan_files(et.iter_source_files(root), root) - report = et.build_report(root, register, scan) + register = et.read_register(_config(requirements_path=str(root / "docs" / "requirements.md"))) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) + report = et.build_report(_config(root, min_coverage=50.0), register, scan) report.coverage.percent = 158.0 - text, code = et.format_coverage_report(report, 50.0) + text, code = et.format_coverage_report(report) assert code == 1 assert "exceeds 100%" in text @@ -610,9 +651,9 @@ def test_the_per_type_breakdown_sums_to_the_headline_figure(): with tempfile.TemporaryDirectory() as tmp: root = _fixture_repo( tmp, f"// {TAG} AR-001, AR-027\nint main() {{}}\n") - register = et.read_register(root / "docs" / "requirements.md") - scan = et.scan_files(et.iter_source_files(root), root) - report = et.build_report(root, register, scan) + register = et.read_register(_config(requirements_path=str(root / "docs" / "requirements.md"))) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) + report = et.build_report(_config(root), register, scan) stats = et.per_type_stats(report) assert sum(c for c, _, _ in stats.values()) == len(report.coverage.covered) assert sum(u for _, u, _ in stats.values()) == len(report.coverage.unexecuted) @@ -637,7 +678,7 @@ def test_json_and_markdown_outputs_are_written_and_consistent(): assert data["coverage"]["covered"] == 1 assert data["coverage"]["percent"] == round(100 / 3, 1) assert data["byType"]["SR"] == ["SR-002"] - assert data["gpuOnlyRequirements"] == ["AR-027"] + assert data["unexecutableRequirements"] == ["AR-027"] assert "AR-001" in md_out.read_text(encoding="utf-8") @@ -662,144 +703,36 @@ def test_system_spec_parsing_enables_orphan_checks_for_pr_and_sr(): # requirements are added. # -------------------------------------------------------------------------- -#: The live-register tests below ran against this repo's own register when the -#: tool lived inside `scene-actor-extraction`. Now that it is shared, this repo -#: is the project home and holds no component register of its own, so they take -#: a path from the environment and skip when it is absent. -#: -#: LIVE_REGISTER=../scene-actor-extraction/docs/requirements.md \ -#: python3 test_extract_traces.py -#: -#: They are kept rather than deleted because they assert something the synthetic -#: fixtures cannot: that a *real* register, with all its formatting accidents, -#: parses at all. -LIVE_REGISTER = os.environ.get("LIVE_REGISTER") +LIVE_REGISTER = os.environ.get('LIVE_REGISTER') def test_the_live_register_parses_and_assigns_tiers(): if not LIVE_REGISTER: - return # skipped: no component register to point at - register = et.read_register(Path(LIVE_REGISTER)) + return # skipped: the project home holds no component register + register = et.read_register(_config(root=str(Path(LIVE_REGISTER).resolve().parent.parent), requirements_path=str(LIVE_REGISTER))) assert register.total > 0 + for req_type in et.LOCAL_TYPES: + assert register.count(req_type) > 0, req_type assert sum(register.count(t) for t in et.LOCAL_TYPES) == register.total - # A tier that no GPU-less host can run must be visible in the parse, not - # merely stated in prose. - assert register.unexecutable_ids() <= register.ids + # The GPU-less CI host must be visible in the parse, not just in prose. + assert "AR-027" in register.unexecutable_ids() + assert register.requirements["AR-027"].tiers == {"T4"} + assert register.requirements["AR-012"].tiers == {"T2"} + assert register.unexecutable_ids() < register.ids def test_the_live_register_yields_a_gate_run_that_cannot_exceed_one_hundred(): if not LIVE_REGISTER: return - register = et.read_register(Path(LIVE_REGISTER)) - root = Path(LIVE_REGISTER).resolve().parent.parent - scan = et.scan_files(et.iter_source_files(root), root) + register = et.read_register(_config(root=str(Path(LIVE_REGISTER).resolve().parent.parent), requirements_path=str(LIVE_REGISTER))) + files = et.iter_source_files(et.REPO_ROOT, _config()) + scan = et.scan_files(files, et.REPO_ROOT, _config()) cov = et.compute_coverage( [i for t in scan.traces for i in t.requirements], register) assert 0.0 <= cov.percent <= 100.0 assert len(cov.covered) <= cov.total -# -------------------------------------------------------------------------- -# Per-repo configuration — the flags that make this tool shareable. -# -# Without them a repo whose prefixes or language differ from the defaults gets -# zero requirements and zero files, which the gate correctly refuses to report -# as coverage. These assert the overrides actually take effect. -# -------------------------------------------------------------------------- - -def _restore_taxonomy(fn): - """Run `fn` with the module defaults restored afterwards.""" - local, suffixes, roots = et.LOCAL_TYPES, set(et.SOURCE_SUFFIXES), et.SCAN_ROOTS - try: - fn() - finally: - et.configure_taxonomy(local) - et.configure_suffixes(suffixes) - et.configure_scan_roots(roots) - - -def test_types_override_changes_which_prefixes_are_counted(): - def body(): - register_md = ( - "| ID | Requirement | Traces to | Priority | Status |\n" - "|---|---|---|---|---|\n" - "| UR-001 | A user requirement | SR-001 | High | Done |\n" - "| DR-001 | A dev requirement | PR-004 | High | Done |\n" - ) - with tempfile.TemporaryDirectory() as tmp: - path = Path(tmp) / "requirements.md" - path.write_text(register_md, encoding="utf-8") - - # Under the defaults these prefixes are unknown, so nothing counts. - et.configure_taxonomy(("AR", "DP")) - assert et.read_register(path).total == 0 - - et.configure_taxonomy(("UR", "DR")) - register = et.read_register(path) - assert register.total == 2 - assert register.count("UR") == 1 - assert register.count("DR") == 1 - - _restore_taxonomy(body) - - -def test_suffix_override_changes_which_files_are_scanned(): - def body(): - with tempfile.TemporaryDirectory() as tmp: - root = Path(tmp) - (root / "src").mkdir() - (root / "src" / "lib.rs").write_text( - "/// TRACES: UR-001 | SR-004\npub fn f() {}\n", encoding="utf-8") - - # Default suffixes are C++/Python, so a Rust tree scans as empty — - # the failure mode this override exists to fix. - assert et.iter_source_files(root) == [] - - et.configure_suffixes([".rs"]) - files = et.iter_source_files(root) - assert len(files) == 1 - scan = et.scan_files(files, root) - assert [i for t in scan.traces for i in t.requirements] == [ - "UR-001", "SR-004"] - - _restore_taxonomy(body) - - -def test_suffix_override_accepts_extensions_with_or_without_a_dot(): - def body(): - et.configure_suffixes(["rs", ".cs"]) - assert et.SOURCE_SUFFIXES == {".rs", ".cs"} - - _restore_taxonomy(body) - - -def test_scan_root_override_limits_the_walk(): - def body(): - with tempfile.TemporaryDirectory() as tmp: - root = Path(tmp) - # Not "vendor" — that is in EXCLUDED_DIR_NAMES and would be - # filtered whatever the scan roots say, testing the wrong thing. - for d in ("src", "eval"): - (root / d).mkdir() - (root / d / "f.py").write_text("# TRACES: AR-001\n", encoding="utf-8") - - et.configure_scan_roots(["src"]) - assert len(et.iter_source_files(root)) == 1 - - et.configure_scan_roots(["src", "eval"]) - assert len(et.iter_source_files(root)) == 2 - - _restore_taxonomy(body) - - -def test_defaults_are_unchanged_by_the_overrides_existing(): - # The extraction pipeline must keep working with no flags at all, or moving - # the tool here would have broken the repo it came from. - assert et.LOCAL_TYPES == ("AR", "DP", "IR", "GR", "VR") - assert ".py" in et.SOURCE_SUFFIXES and ".cpp" in et.SOURCE_SUFFIXES - assert ".rs" not in et.SOURCE_SUFFIXES - - # -------------------------------------------------------------------------- def _main() -> int: diff --git a/scripts/traceability/traceability-gate.sh b/scripts/traceability/traceability-gate.sh index a41dbb8..588848b 100755 --- a/scripts/traceability/traceability-gate.sh +++ b/scripts/traceability/traceability-gate.sh @@ -1,89 +1,62 @@ #!/bin/sh # -# Requirement traceability gate. Run locally exactly as CI runs it: +# Requirement traceability gate. Run locally exactly as CI runs it, from the +# component repo root: # # scripts/traceability/traceability-gate.sh # -# Writes traces-report.json and docs/traceability.md, prints the coverage -# report, and exits non-zero when the gate fails. +# Writes the JSON report and the markdown matrix, prints the coverage report, +# and exits non-zero when the gate fails. # -# Environment: -# MIN_COVERAGE minimum overall coverage percent (default 0 - see below) -# ALLOW_ORPHANS set to 1 to report orphan tags without failing -# TRACES_JSON JSON report path (default traces-report.json) -# TRACES_MD markdown matrix path (default docs/traceability.md) -# REPO_ROOT repository to scan. Defaults to two levels above this -# script, which is correct when the tooling lives in the repo -# it checks. **When vendored as a submodule that default is -# the submodule itself**, so a consuming repo must set this — -# its wrapper does. -# TYPES comma-separated requirement prefixes (e.g. UR,DR). Defaults -# to the extraction set; a repo whose prefixes differ parses -# to zero requirements without this. -# SUFFIXES comma-separated file extensions (e.g. .rs). Defaults to the -# C++/Python set; a repo whose language differs scans zero -# files without this. -# SCAN_ROOTS comma-separated directories to walk, relative to REPO_ROOT. -# SYSTEM_SPEC optional path to the system SPEC.md, which defines the PR/SR -# IDs; when given, PR/SR orphans are reported too. It lives in -# the project home (jray-project) — when this tooling is -# vendored from there, it is a sibling of this script. +# This script is shared by every JRay component, so it knows nothing about any +# one repo. All repo-specific settings - requirement ID prefixes, source +# suffixes, scan roots, register path, thresholds - live in `traceability.toml` +# at the component repo root. Run # -# Threshold policy lives here and nowhere else. It is deliberately NOT -# duplicated into the workflow YAML: a threshold written in two places is a -# threshold that will disagree with itself. +# scripts/traceability/extract_traces.py --print-example-config # -# MIN_COVERAGE defaults to 0 because almost nothing is tagged yet - tags are -# added as the pipeline is built, so a low number today is accurate rather than -# alarming. A zero threshold does NOT mean the gate cannot fail: orphan tags, -# a >100% ratio, a register that parses to nothing, and an empty source scan -# are all hard failures from day one. Raise MIN_COVERAGE as tags land; treat -# every raise as a ratchet, never a reset. +# for the annotated schema. A repo whose config is wrong parses zero +# requirements or scans zero files, and the gate refuses to report rather than +# printing a misleading 0%. +# +# Environment (all optional; each overrides the config file): +# TRACES_CONFIG path to traceability.toml +# TRACES_ROOT repo root (default: nearest dir containing traceability.toml) +# MIN_COVERAGE minimum overall coverage percent +# ALLOW_ORPHANS 1 to report orphan tags without failing +# TRACES_JSON JSON report path +# TRACES_MD markdown matrix path +# SYSTEM_SPEC SPEC.md defining PR/SR; enables PR/SR orphan checking +# PYTHON interpreter (default: python3) +# +# Threshold policy belongs in traceability.toml, not here and not in the +# workflow YAML: a threshold written in two places is a threshold that will +# disagree with itself. # # POSIX sh, no bashisms, no jq - the extractor does its own arithmetic and -# printing so CI needs nothing beyond python3. +# printing, so CI needs nothing beyond python3. set -eu SCRIPT_DIR=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) -REPO_ROOT="${REPO_ROOT:-$(CDPATH= cd -- "$SCRIPT_DIR/../.." && pwd)}" - -MIN_COVERAGE="${MIN_COVERAGE:-0}" -TRACES_JSON="${TRACES_JSON:-$REPO_ROOT/traces-report.json}" -TRACES_MD="${TRACES_MD:-$REPO_ROOT/docs/traceability.md}" PYTHON="${PYTHON:-python3}" command -v "$PYTHON" >/dev/null 2>&1 || { - echo "FAILED: $PYTHON not found. The traceability gate needs Python 3.9+" >&2 + echo "FAILED: $PYTHON not found. The traceability gate needs Python 3.9+," >&2 + echo " or 3.11+ to read traceability.toml." >&2 exit 2 } -set -- \ - --root "$REPO_ROOT" \ - --requirements "${REQUIREMENTS:-$REPO_ROOT/docs/requirements.md}" \ - --format coverage \ - --json-out "$TRACES_JSON" \ - --markdown-out "$TRACES_MD" \ - --min-coverage "$MIN_COVERAGE" +set -- --format coverage -if [ "${ALLOW_ORPHANS:-0}" = "1" ]; then - set -- "$@" --allow-orphans -fi - -if [ -n "${SYSTEM_SPEC:-}" ]; then - set -- "$@" --system-spec "$SYSTEM_SPEC" -fi - -if [ -n "${TYPES:-}" ]; then - set -- "$@" --types "$TYPES" -fi - -if [ -n "${SUFFIXES:-}" ]; then - set -- "$@" --suffixes "$SUFFIXES" -fi - -if [ -n "${SCAN_ROOTS:-}" ]; then - set -- "$@" --scan-roots "$SCAN_ROOTS" -fi +# Explicit `if` rather than `[ ... ] && ...`, because a trailing false test in +# an && list exits under `set -e` in some POSIX shells. +if [ -n "${TRACES_CONFIG:-}" ]; then set -- "$@" --config "$TRACES_CONFIG"; fi +if [ -n "${TRACES_ROOT:-}" ]; then set -- "$@" --root "$TRACES_ROOT"; fi +if [ -n "${MIN_COVERAGE:-}" ]; then set -- "$@" --min-coverage "$MIN_COVERAGE"; fi +if [ -n "${TRACES_JSON:-}" ]; then set -- "$@" --json-out "$TRACES_JSON"; fi +if [ -n "${TRACES_MD:-}" ]; then set -- "$@" --markdown-out "$TRACES_MD"; fi +if [ -n "${SYSTEM_SPEC:-}" ]; then set -- "$@" --system-spec "$SYSTEM_SPEC"; fi +if [ "${ALLOW_ORPHANS:-0}" = "1" ]; then set -- "$@" --allow-orphans; fi exec "$PYTHON" "$SCRIPT_DIR/extract_traces.py" "$@"