diff --git a/.claude/settings.local.json.tmp.271576.c40f99ae1801 b/.claude/settings.local.json.tmp.271576.c40f99ae1801 deleted file mode 100644 index b82640d..0000000 --- a/.claude/settings.local.json.tmp.271576.c40f99ae1801 +++ /dev/null @@ -1,107 +0,0 @@ -{ - "permissions": { - "allow": [ - "Bash(python3 -c ' *)", - "WebFetch(domain:docs.turso.tech)", - "Bash(awk '/^```/{n++} END{print \"fences:\",n,\\(n%2==0?\"balanced\":\"UNBALANCED\"\\)}' SPEC.md)", - "Bash(python3 -c \"import h5py; print\\('h5py', h5py.__version__\\)\")", - "Bash(python3 scripts/make_jellyfin_gallery.py --help)", - "Bash(timeout 30 python3 -c ' *)", - "Bash(ps -o pid,etime,cmd -C cmake)", - "Bash(cargo check *)", - "Bash(cargo test *)", - "Bash(cargo clippy *)", - "Bash(cargo deny *)", - "Bash(rustc --version)", - "Bash(cargo metadata *)", - "Bash(python3 -c \"import json,sys; d=json.load\\(sys.stdin\\); print\\(d['packages'][0].get\\('license'\\)\\)\")", - "Bash(timeout 900 cargo install cargo-deny --locked)", - "Bash(timeout 1500 docker build -t jray-server:test .)", - "Bash(timeout 900 docker build -f /tmp/claude-1000/-home-dtourolle-Development-Jray-project/d861f764-2add-4f42-a069-0954ad1f9565/scratchpad/Dockerfile.probe -t jray-probe .)", - "Bash(docker rm *)", - "Bash(docker volume *)", - "Bash(timeout 900 docker build -t jray-server:test .)", - "Bash(docker run *)", - "Bash(xargs -I{} echo \"clippy: {}\")", - "Bash(timeout 900 docker build -q -t jray-server:test .)", - "Bash(awk '{s+=$4} END {print \"total: \" s \" passing\"}')", - "Bash(xargs -I{} echo \"clippy issues: {}\")", - "Bash(timeout 600 dotnet build Jellyfin.Plugin.JRay/Jellyfin.Plugin.JRay.csproj -v q --nologo)", - "Bash(awk '{s+=$4} END {print \"server tests: \" s \" passing\"}')", - "Bash(timeout 300 dotnet build Jellyfin.Plugin.JRay/Jellyfin.Plugin.JRay.csproj -v q --nologo)", - "Bash(cp SPEC.md /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/SPEC.md.bak)", - "Bash(perl -pi -e ' *)", - "Bash(grep -vE \"AR-|DP-|IR-|GR-|VR-|SR-|PR-|H[0-9]|[0-9]{3,}\")", - "Bash(git update-index *)", - "WebFetch(domain:github.com)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/audio-signature status --short)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/audio-signature branch --show-current)", - "Bash(python3 -c \"import pytest; print\\(pytest.__version__\\)\")", - "Bash(python3 -c \"import numpy; print\\('numpy', numpy.__version__\\)\")", - "Read(//usr/lib/cmake/**)", - "Read(//usr/include/**)", - "Bash(grep -n \"^#\\\\{1,3\\\\} \" JRay-public-server/SPEC.md)", - "Bash(grep -n \"^#\\\\{1,3\\\\} \" scene-actor-extraction/docs/SPEC.md)", - "Bash(cargo build *)", - "Bash(python3 gen.py tone.wav)", - "Bash(python3 ref.py tone.wav)", - "Bash(python3 -c \"import json;d=json.load\\(open\\('package.json'\\)\\);print\\(json.dumps\\({k:v for k,v in d.get\\('scripts',{}\\).items\\(\\) if 'trace' in k.lower\\(\\)},indent=2\\)\\)\")", - "Bash(python3 scripts/traceability/extract_traces.py --format json)", - "Bash(python3 scripts/test_extract_traces.py)", - "Bash(python3 -m pytest scripts/traceability/test_extract_traces.py -q)", - "Bash(python3 scripts/traceability/extract_traces.py --format coverage)", - "Bash(python3 scripts/extract_traces.py)", - "Bash(python3 scripts/extract_traces.py --format json)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/traceability-tooling log --oneline -3)", - "Bash(python3 /home/dtourolle/Development/Jray-worktrees/traceability-tooling/scripts/traceability/extract_traces.py --root /home/dtourolle/Development/Jray-project/JRay-public-server --requirements /home/dtourolle/Development/Jray-project/JRay-public-server/docs/requirements.md --system-spec /home/dtourolle/Development/Jray-project/SPEC.md --format coverage)", - "Bash(python3 scripts/traceability/test_extract_traces.py)", - "Bash(sh scripts/traceability/traceability-gate.sh)", - "Bash(python3 -c \"import pyflakes; print\\('pyflakes', pyflakes.__version__\\)\")", - "Bash(timeout 300 dotnet build -v q --nologo)", - "Bash(git check-ignore *)", - "Bash(rm -rf /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full)", - "Bash(cmake -S . -B /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full -DSAE_BUILD_TESTS=ON)", - "Bash(pkg-config --cflags opencv4)", - "Bash(pkg-config --cflags opencv)", - "Bash(awk '{s+=$4} END {print \"tests: \" s}')", - "Bash(python3 -m pyflakes scripts/validation/min_face_size.py)", - "Bash(python3 -m flake8 --select=F scripts/validation/min_face_size.py)", - "Bash(git -c user.name=\"Duncan Tourolle\" -c user.email=\"duncan@tourolle.paris\" commit -q -F -)", - "Bash(cmake -S . -B /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full -DSAE_BUILD_TESTS=ON -DSAE_GEMM_BACKEND=CPU)", - "Bash(python3 -c \"import h5py, numpy, requests, PIL; print\\('deps ok'\\)\")", - "Bash(python3 -m py_compile scripts/validation/min_face_size.py)", - "Bash(git remote *)", - "Bash(python3 et.py --root /home/dtourolle/Development/Jray-project/jRay --requirements /home/dtourolle/Development/Jray-project/jRay/docs/requirements.md --system-spec /home/dtourolle/Development/Jray-project/SPEC.md --format coverage)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/traceability-tooling status --short)", - "Bash(git -C /home/dtourolle/Development/Jray-worktrees/traceability-tooling log --oneline -2 -- scripts/traceability/)", - "Bash(timeout 240 python3 scripts/stamp_gallery.py --gallery /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/g_plain.h5 --show)", - "Bash(cp /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/g_plain.h5 /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/g_mig.h5)", - "Bash(python3 scripts/stamp_gallery.py --gallery /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/g_mig.h5 --arcface /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/fake.onnx)", - "Bash(timeout 590 cmake -S . -B /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/bt -DSAE_BUILD_TESTS=ON -DCMAKE_BUILD_TYPE=Release)", - "Bash(grep -n \"^#\\\\{1,3\\\\} \\\\|^| ID \\\\|^| Requirement \\\\|^|---\" jRay/docs/requirements.md)", - "Bash(./build-full/tests/sae_tests)", - "Bash(ls /usr/include/catch2/catch_test_macros.hpp 2>/dev/null && echo \"catch2 system\"; ls /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/ 2>/dev/null | head; ls /usr/lib/libCatch2* 2>/dev/null | head)", - "Read(//usr/lib/**)", - "Bash(find /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-build -name libCatch2*)", - "Bash(timeout 590 g++ -std=c++20 -O1 -I/home/dtourolle/Development/Jray-worktrees/gallery-model-binding/src -I/usr/include/opencv5 -I/tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/nlohmann_json-src/single_include -I/tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-src/src -I/tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-build/generated-includes '-DSAE_MODELS_DIR=\"/home/dtourolle/Development/Jray-worktrees/gallery-model-binding/models\"' -o gallery_tests /home/dtourolle/Development/Jray-worktrees/gallery-model-binding/tests/test_gallery_store.cpp /home/dtourolle/Development/Jray-worktrees/gallery-model-binding/src/gallery/gallery_store.cpp /home/dtourolle/Development/Jray-worktrees/gallery-model-binding/src/gallery/embedder_stamp.cpp /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-build/src/libCatch2Main.a /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/build-full/_deps/catch2-build/src/libCatch2.a -lhdf5_cpp -lhdf5 -lopencv_core)", - "Bash(./gallery_tests)", - "Bash(grep -viE \"^\\\\[gallery\\\\]|^ |^$|WARNING \\\\\\(GR-004\\\\\\)|\\\\*\\\\*\\\\*\\\\*\")", - "Bash(python3 scripts/traceability/extract_traces.py --root jRay --requirements jRay/docs/requirements.md --system-spec SPEC.md --types JR --suffixes .cs,.js --scan-roots Jellyfin.Plugin.JRay --format coverage)", - "Bash(python3 scripts/traceability/extract_traces.py --root /home/dtourolle/Development/Jray-project/jRay --requirements /home/dtourolle/Development/Jray-project/jRay/docs/requirements.md --system-spec /home/dtourolle/Development/Jray-project/SPEC.md --types JR --suffixes .cs,.js --scan-roots Jellyfin.Plugin.JRay --format coverage)", - "Bash(cp -r /home/dtourolle/Development/Jray-project/scene-actor-extraction/external/KPN/. external/KPN/)", - "Bash(rm -rf external/KPN/.git)", - "Bash(rm -rf /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/bt)", - "Bash(timeout 590 cmake -S . -B /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/bt -DSAE_BUILD_TESTS=ON -DSAE_GEMM_BACKEND=CPU -DCMAKE_BUILD_TYPE=Release)", - "Bash(python3 scripts/traceability/extract_traces.py --root scene-actor-extraction --requirements scene-actor-extraction/docs/requirements.md --format coverage)", - "Bash(timeout 590 cmake --build /tmp/claude-1000/-home-dtourolle-Development-Jray-project/de7e497c-da04-462d-a971-d0fcd36bda2a/scratchpad/bt --target sae_tests -j8)", - "Bash(sh -n scripts/traceability/traceability-gate.sh)", - "Bash(cd /home/dtourolle/Development/Jray-project *)" - ], - "additionalDirectories": [ - "/home/dtourolle/Development/Jray-worktrees/gallery-model-binding/src/gallery", - "/home/dtourolle/Development/Jray-worktrees/min-face-study/scripts/validation", - "/home/dtourolle/Development/Jray-worktrees/audio-signature/tests", - "/home/dtourolle/Development/Jray-worktrees/traceability-tooling/scripts/traceability" - ] - } -} diff --git a/README.md b/README.md index 076f961..fe24d0a 100644 --- a/README.md +++ b/README.md @@ -4,23 +4,36 @@ see who is on screen — for their own library, on their own hardware, with nobody else learning what they own. -This is the project home. It owns the [system specification](SPEC.md) — the -requirements that span more than one component — and the tooling shared between -them. The code lives in three separate repositories, linked below. +This is the project home. It owns the [system specification](SPEC.md), the +requirements that span more than one component, and the tooling they share. The +code lives in three repositories, linked below. --- -## The three components +## The problem -| Repository | Role | Language | -|---|---|---| -| [`scene-actor-extraction`](https://gitea.tourolle.paris/dtourolle/scene-actor-extraction) | Derives presence data from a media file | C++ / Python | -| [`jRay`](https://gitea.tourolle.paris/dtourolle/jRay) | Jellyfin plugin: surfaces it in the player, owns the truth-file format | C# | -| [`JRay-public-server`](https://gitea.tourolle.paris/dtourolle/JRay-public-server) | Exchanges presence data between instances | Rust | +You are watching a film. Someone appears and you know you have seen them +before — but pausing to search breaks the film, and the answer is rarely worth +the interruption. Amazon X-Ray solves this well, and only for Amazon's catalogue. -They ship independently — the plugin to Jellyfin's catalogue, the server as a -binary, the pipeline to a GPU host — which is why they are separate repositories -rather than one monorepo. +Doing the same for a personal library is harder than it looks: + +- **Face recognition on a paused frame answers the wrong question.** In dialogue + the camera is usually on whoever is *not* speaking, so a per-frame answer + reports the other actor absent. X-Ray credits a whole scene's cast for the + scene's duration, and that is the more useful question (SR-002). +- **The compute is real.** Detecting, embedding and tracking every face in a + feature takes GPU time. Doing it once per viewer, for the same film, is waste. +- **The obvious fix leaks.** A service that identifies your library must be told + what your library contains, which recreates the thing being replaced (PR-005). + +JRay's answer: extract presence data locally, share the *timings* rather than +the media or the faces, and make the shared artefact structurally incapable of +carrying anything else. + +--- + +## How it fits together ``` media file ──► extraction ──► truth file (sidecar or pushed) ──► plugin ──► player overlay @@ -30,9 +43,28 @@ media file ──► extraction ──► truth file (sidecar or pushed) ── (Jellyfin + TMDB) ``` -Two axes, deliberately separate: **presence data** flows outward and is -shareable, being timings against public identifiers. **Gallery data** — the -actor reference faces — is built locally and never leaves the instance (SR-005). +| Repository | Role | Language | +|---|---|---| +| [`scene-actor-extraction`](https://gitea.tourolle.paris/dtourolle/scene-actor-extraction) | Derives presence data from a media file | C++ / Python | +| [`jRay`](https://gitea.tourolle.paris/dtourolle/jRay) | Jellyfin plugin: surfaces it in the player, owns the truth-file format | C# | +| [`JRay-public-server`](https://gitea.tourolle.paris/dtourolle/JRay-public-server) | Exchanges presence data between instances | Rust | + +They ship independently — the plugin to Jellyfin's catalogue, the server as a +static binary, the pipeline to a GPU host — which is why they are separate +repositories rather than one monorepo. + +**Two data axes, deliberately separate.** *Presence data* flows outward and is +shareable: it is timings against public TMDB identifiers. *Gallery data* — the +actor reference faces — is built locally from your own Jellyfin instance plus +TMDB, and never leaves the machine (SR-005). Not by export, not by opt-in, not +at all: the capability is what creates the exposure, so it does not exist. + +**The manifest server holds no binary content.** An accepted manifest contains +bounded numbers, regex-constrained identifiers, and references to TMDB persons. +No images, no embeddings, no free-form strings, no extension points (SR-004). +That is what makes it safe for a volunteer to operate an instance, and it is a +property preserved by prohibition — any proposal to ship blobs through it is a +proposal to delete it. --- @@ -49,21 +81,20 @@ git clone git@gitea.tourolle.paris:dtourolle/jRay.git git clone git@gitea.tourolle.paris:dtourolle/JRay-public-server.git ``` -The component directories are `.gitignore`d here, so they sit beside the system -spec without this repository trying to track them. Each has its own README with -build instructions. +Each component has its own README with build instructions. They are +`.gitignore`d here, so they sit beside the system spec without this repository +trying to track them. -### Why not submodules for the components +### Why the components are not submodules A submodule pins a commit. With feature branches and worktrees in flight across the components, every component commit would leave this repository's pointer -stale and its `git status` dirty until someone committed a pointer bump — churn -that buys nothing, since the components are developed together in one directory -anyway. +stale and its `git status` dirty until someone committed a bump — churn that +buys nothing, since the components are developed together in one directory. -The dependency runs the other way instead: **components pull *this* repository -in** for the shared tooling and the system spec, both of which change rarely. -That is the asymmetry submodules suit. +The dependency runs the other way: **each component pulls *this* repository in** +as a submodule, for the system spec and the shared tooling, both of which change +rarely. That is the asymmetry submodules suit. --- @@ -72,22 +103,19 @@ That is the asymmetry submodules suit. | Doc | Owns | |---|---| | [`SPEC.md`](SPEC.md) | **System requirements** — `PR-nnn` project goals, `SR-nnn` cross-component contracts | -| [`CLAUDE.md`](CLAUDE.md) | Working notes and the invariants that must not be violated silently | +| [`CLAUDE.md`](CLAUDE.md) | Working notes, and the invariants that must not be violated silently | | Each repo's `SPEC.md` | That component's software requirements | | Each repo's `docs/requirements.md` | Its stable requirement IDs, status, and verification plan | Read the system spec first. Every component requirement traces up to an `SR-nnn`, and every `SR-nnn` to a `PR-nnn`, so the chain explains *why* a given -piece of code exists. +piece of code exists — and makes it visible when something exists for no stated +reason. --- ## Requirement traceability -Requirements are traceable **up** to the project goal and **down** to the code -implementing them. A requirement nothing traces to is either unnecessary or -unimplemented, and both are worth knowing. - ``` PR-nnn project requirement (SPEC.md §1) — why the system exists └─ SR-nnn system requirement (SPEC.md §3) — what spans components @@ -95,26 +123,65 @@ PR-nnn project requirement (SPEC.md §1) — why the system exists └─ TRACES tag (source) ``` -Tag the code that *satisfies* a requirement: +Tag the code that *satisfies* a requirement — the unit that decides, not every +helper it calls: ```rust -/// TRACES: UR-003 | SR-004 +/// TRACES: UR-003, UR-011 | SR-004 pub fn validate_manifest(m: Jmanifest) -> VResult { … } ``` -A pipe separates requirement *types*; a comma separates IDs within a type. Tag -the unit that decides, not every helper it calls — a tag on every function is -noise, and rots faster than it helps. +A pipe separates requirement *types*; a comma separates IDs within a type. +Tests carry tags too (`UT-nnn`, `IT-nnn`), which is what shows a requirement is +*verified* rather than merely implemented. A deliberate departure from an +invariant is tagged `EXCEPTION:` with its reason — an untagged one is a defect. -The gate reports coverage, orphan tags (an ID no register defines), and untraced -requirements. Two rules it inherits from JellyTau, both learned the hard way: +### Running the gate + +The tooling lives in [`scripts/traceability/`](scripts/traceability/) here and +is vendored into each component as a submodule, so there is **one +implementation**. Each component declares its own taxonomy in a +`traceability.toml` at its root: + +```toml +requirement_types = ["UR", "DR"] +languages = ["rust"] +source_roots = ["src", "tests"] +``` + +```sh +scripts/traceability-gate.sh # from any component +``` + +It reports coverage, orphan tags (an ID no register defines), untraced +requirements, and requirements verifiable only on hardware CI lacks. + +**Two rules inherited from JellyTau, both learned the hard way:** - **Denominators are read from the register at run time, never hardcoded.** A - gate that divides by a frozen literal reported 158% coverage for months and so - could never fail — worse than no gate, because it was trusted. + gate that divided by a frozen literal reported *158% coverage* for months + while the requirement count grew, so its threshold could never trip. A gate + that cannot fail is worse than no gate, because it is trusted. - **Coverage above 100% is a hard failure**, not a pass. It means the - computation is broken, and it is the signal that catches the above - immediately. + computation is broken, and it is the signal that catches the above at once. + +The same reasoning is why a requirement whose only evidence is a test that never +runs is reported as *tagged but unexecuted*, never counted as covered. + +--- + +## Status + +| Component | State | +|---|---| +| `scene-actor-extraction` | Pipeline redesign in progress — presence follows track extent (AR-012), replacing per-frame recognition | +| `jRay` | Truth-file serving and overlay working; manifest-sharing configuration added, fetch path outstanding | +| `JRay-public-server` | Core implemented: 23/32 requirements traced, 189 tests. Audio-tier matching and federation deferred by design | + +**One schema bump is pending across all three repos** (SR-003). It removes +`anneal_sec`, adds `extinction_sec` and `gallery_scope`, gives each window its +belief and identification route, and adds the audio signature. Breaking changes +are batched, so these ship together rather than piecemeal. --- diff --git a/scripts/traceability/extract_traces.py b/scripts/traceability/extract_traces.py index 0b946d0..049d1bf 100755 --- a/scripts/traceability/extract_traces.py +++ b/scripts/traceability/extract_traces.py @@ -1,12 +1,26 @@ #!/usr/bin/env python3 -"""Extract requirement traces from C++/Python sources and report coverage. +"""Extract requirement traces from source and report coverage against a register. -Ported from JellyTau's ``scripts/extract-traces.ts``. That one scans -TypeScript/Svelte/Rust; this repo is C++ and Python, so the scanner is Python -with no third-party dependencies — the CI host must not need a node/bun -toolchain to check traceability. +Ported from JellyTau's ``scripts/extract-traces.ts`` and written in stdlib +Python so no repo needs a bun/node toolchain to check its source comments. -Tag format (see ../../../CLAUDE.md and SPEC.md section 6). A pipe separates +**One implementation, parameterised.** This tool is shared by every JRay +component repo — C++/Python extraction, C# plugin, Rust server — via the +``jray-project`` submodule. Nothing about a single repo is baked in: the +requirement ID prefixes, the source file suffixes, the directories to scan, the +register path and the system-spec path are all configuration. A second copy for +"the other language" is how two implementations start drifting apart, so there +is exactly one. + +Configuration comes from ``traceability.toml`` at the component repo root, from +CLI flags, or both (flags win). ``--print-example-config`` emits the full +annotated schema; the three required keys are:: + + requirement_types = ["AR", "DP", "IR", "GR", "VR"] + languages = ["cpp", "python"] + source_roots = ["src", "tests", "scripts"] + +Tag format (see the system CLAUDE.md and SPEC.md section 6). A pipe separates requirement *types*, a comma separates IDs within a type:: /// TRACES: AR-nnn, AR-mmm | SR-nnn @@ -19,28 +33,32 @@ separately — never silently folded into coverage:: Usage:: - python3 scripts/traceability/extract_traces.py --format coverage - python3 scripts/traceability/extract_traces.py --format json > traces-report.json - python3 scripts/traceability/extract_traces.py --format markdown --markdown-out docs/traceability.md + python3 extract_traces.py --format coverage + python3 extract_traces.py --format json > report.json + python3 extract_traces.py --config path/to/traceability.toml -The CI gate is ``scripts/traceability/traceability-gate.sh``, which wraps this. +The CI gate is ``traceability-gate.sh``, which wraps this. -Two rules inherited from JellyTau's gate repair, both learned the hard way -(JellyTau/docs/specs/traceability-gate-repair.md): +Three rules the gate exists to enforce. The first two are inherited from +JellyTau's gate repair, both learned the hard way +(``JellyTau/docs/specs/traceability-gate-repair.md``): -1. Coverage denominators are read out of ``docs/requirements.md`` at run time. - Never hardcode them. JellyTau's gate divided by frozen literals while the - register grew to 211 requirements; it reported 158% coverage, so its 50% - threshold could never trip. A gate that cannot fail is worse than no gate, - because it is trusted. +1. Coverage denominators are read out of the register at run time. Never + hardcode them. JellyTau's gate divided by frozen literals while the register + grew to 211 requirements; it reported 158% coverage, so its 50% threshold + could never trip. A gate that cannot fail is worse than no gate, because it + is trusted. 2. Coverage above 100% is a hard failure, not a pass. It means the computation is broken, and it is the signal that would have caught (1) immediately. +3. A misconfigured run refuses to report at all. Parsing zero requirements or + scanning zero files prints a failure, not a plausible-looking 0% — the same + family of error as (1), and the one a shared, parameterised tool makes easy + to hit. -One rule specific to this repo: CI runs on an Intel N100 with no discrete GPU. -Requirements whose only verification tier is T4/GPU (or "out of CI") cannot -execute here. They are reported as *tagged but unexecuted* and are excluded -from the covered numerator — counting a test that never runs as coverage is -the same failure mode as the 158% bug. +The GPU-less-CI rule is configuration rather than code: requirements whose +verification tiers all fall outside ``ci_executable_tiers`` are reported as +*tagged but unexecuted* and excluded from the covered numerator. Counting a +test that never runs is the same failure mode as the 158% bug. """ from __future__ import annotations @@ -49,116 +67,372 @@ import argparse import json import re import sys -from dataclasses import dataclass, field +from dataclasses import dataclass, field, replace from datetime import datetime, timezone from pathlib import Path -from typing import Dict, Iterable, List, Optional, Sequence, Set, Tuple - -# Derived from this file's location so a CI checkout at any path works. -SCRIPT_DIR = Path(__file__).resolve().parent -REPO_ROOT = SCRIPT_DIR.parent.parent +from typing import (Dict, FrozenSet, Iterable, List, Optional, Sequence, Set, + Tuple) # -------------------------------------------------------------------------- -# Taxonomy -# -# These are *defaults*, not fixed policy. This tool is shared by every component -# in the project (it is vendored in as a submodule from `jray-project`), and the -# components differ in both their requirement prefixes and their languages: -# -# scene-actor-extraction AR/DP/IR/GR/VR C++, Python -# JRay-public-server UR/DR Rust -# jRay UR/DR C# -# -# `--types` and `--suffixes` override the two that vary. Leaving them hardcoded -# is not a small problem: a repo whose prefixes are absent parses to **zero** -# defined requirements, and one whose language is absent scans **zero** files. -# The gate refuses to report coverage in either state rather than printing a -# misleading 0%, which is the behaviour that surfaced this. +# Defaults. Everything here is overridable; nothing here names a single repo. # -------------------------------------------------------------------------- -#: Types defined in docs/requirements.md. These, and only these, participate -#: in the coverage fraction. Override with --types. -LOCAL_TYPES: Tuple[str, ...] = ("AR", "DP", "IR", "GR", "VR") - #: Test identifiers. A separate taxonomy: a test ID is evidence *for* a -#: requirement, not a requirement. Excluded from the fraction. -TEST_TYPES: Tuple[str, ...] = ("UT", "IT") +#: requirement, not a requirement, so it never enters the coverage fraction. +#: House-wide (SPEC.md section 6), hence a default rather than an argument. +DEFAULT_TEST_TYPES: Tuple[str, ...] = ("UT", "IT") -#: Defined in the system-level SPEC.md, which lives in the project home -#: (`jray-project`) rather than in any component checkout. Recognised in tags, -#: never counted here, and orphan-checked only when --system-spec is given. -EXTERNAL_TYPES: Tuple[str, ...] = ("PR", "SR") +#: Defined in the system-level SPEC.md that ``jray-project`` owns. Recognised +#: in tags everywhere, never counted in any component's fraction, and +#: orphan-checked only when ``system_spec`` points at that file. +DEFAULT_EXTERNAL_TYPES: Tuple[str, ...] = ("PR", "SR") -KNOWN_TYPES: Tuple[str, ...] = LOCAL_TYPES + TEST_TYPES + EXTERNAL_TYPES +#: Tiers a CI host can actually run. The extraction pipeline's CI is a GPU-less +#: Intel N100, so its T4 is excluded; jRay numbers its live-only tier T4 for +#: exactly this reason rather than renumbering the shared constant. +DEFAULT_CI_EXECUTABLE_TIERS: FrozenSet[str] = frozenset({"T1", "T2", "T3", "static"}) -#: Tiers a GPU-less CI host can actually run. See the "Verification strategy" -#: section of docs/requirements.md. -CI_EXECUTABLE_TIERS: Set[str] = {"T1", "T2", "T3", "static"} +DEFAULT_EXCLUDE_DIRS: FrozenSet[str] = frozenset({ + ".git", "__pycache__", "external", "vendor", "build", "site", "dist", + "node_modules", "target", "bin", "obj", "trt_cache", "ort_cache", + ".venv", "venv", ".mypy_cache", ".pytest_cache", +}) -# -------------------------------------------------------------------------- -# Source scanning -# -------------------------------------------------------------------------- +DEFAULT_REQUIREMENTS_PATH = "docs/requirements.md" +DEFAULT_MATRIX_PATH = "docs/traceability.md" +DEFAULT_JSON_PATH = "traces-report.json" +CONFIG_FILENAME = "traceability.toml" -#: Override with --suffixes. Defaults cover the extraction pipeline; the server -#: passes `.rs` and the plugin `.cs`. -SOURCE_SUFFIXES = { - ".cpp", ".cc", ".cxx", ".hpp", ".hxx", ".h", ".cu", ".cuh", # C++ - ".py", # Python +#: Named suffix groups, so a repo declares "Rust" rather than remembering to +#: spell every extension. Additive with an explicit ``source_suffixes`` list. +LANGUAGE_SUFFIXES: Dict[str, FrozenSet[str]] = { + "cpp": frozenset({".cpp", ".cc", ".cxx", ".hpp", ".hxx", ".h", ".cu", ".cuh"}), + "python": frozenset({".py", ".pyi"}), + "rust": frozenset({".rs"}), + "csharp": frozenset({".cs"}), + "javascript": frozenset({".js", ".mjs", ".cjs", ".jsx"}), + "typescript": frozenset({".ts", ".tsx"}), + "svelte": frozenset({".svelte"}), + "go": frozenset({".go"}), + "java": frozenset({".java"}), } -SCAN_ROOTS: Tuple[str, ...] = ("src", "tests", "scripts", "experiments", "eval") + +# -------------------------------------------------------------------------- +# Configuration +# -------------------------------------------------------------------------- + +class ConfigError(Exception): + """A configuration mistake, reported rather than papered over.""" -def configure_taxonomy(local_types: Sequence[str]) -> None: - """Rebind the requirement prefixes this run recognises. +@dataclass(frozen=True) +class Config: + """Everything that differs between component repos. - Module-level rather than threaded through every call site: the constants are - read in a dozen places, and adding a config parameter to each would be a far - larger diff than the change warrants — more risk of a missed site than the - global buys. Called once from main() before any scanning. + The three required fields are exactly the three things that were hardcoded + before this tool was shared: which ID prefixes the repo's register defines, + which files count as source, and where those files live. A repo that gets + any of them wrong parses zero requirements or scans zero files — and the + gate refuses to report rather than printing a misleading 0%. """ - global LOCAL_TYPES, KNOWN_TYPES - LOCAL_TYPES = tuple(local_types) - KNOWN_TYPES = LOCAL_TYPES + TEST_TYPES + EXTERNAL_TYPES + + requirement_types: Tuple[str, ...] + source_suffixes: FrozenSet[str] + source_roots: Tuple[str, ...] + + root: Path = Path(".") + test_types: Tuple[str, ...] = DEFAULT_TEST_TYPES + external_types: Tuple[str, ...] = DEFAULT_EXTERNAL_TYPES + exclude_dirs: FrozenSet[str] = DEFAULT_EXCLUDE_DIRS + ci_executable_tiers: FrozenSet[str] = DEFAULT_CI_EXECUTABLE_TIERS + requirements_path: str = DEFAULT_REQUIREMENTS_PATH + system_spec_path: Optional[str] = None + matrix_path: Optional[str] = DEFAULT_MATRIX_PATH + json_path: Optional[str] = DEFAULT_JSON_PATH + min_coverage: float = 0.0 + allow_orphans: bool = False + source: str = "defaults" + + @property + def known_types(self) -> Tuple[str, ...]: + return (tuple(self.requirement_types) + tuple(self.test_types) + + tuple(self.external_types)) + + def resolve(self, relative: Optional[str]) -> Optional[Path]: + """A configured path, made absolute against the repo root.""" + if relative is None: + return None + path = Path(relative) + return path if path.is_absolute() else (self.root / path) + + def validate(self) -> None: + problems: List[str] = [] + if not self.requirement_types: + problems.append( + "requirement_types is empty - the register defines IDs with " + "some prefix (AR/DP/IR/GR/VR in extraction, JR in jRay, UR/DR " + "in the server) and the tool cannot guess it") + for name, values in (("requirement_types", self.requirement_types), + ("test_types", self.test_types), + ("external_types", self.external_types)): + for value in values: + if not re.fullmatch(r"[A-Z]{2}", value): + problems.append( + f"{name}: {value!r} is not a two-letter uppercase " + "prefix; IDs are PREFIX-nnn") + overlaps = ((set(self.requirement_types) & set(self.test_types)) + | (set(self.requirement_types) & set(self.external_types))) + if overlaps: + problems.append( + f"prefixes {sorted(overlaps)} are declared both as " + "requirement_types and as another taxonomy; a prefix counted " + "in the fraction cannot also be excluded from it") + if not self.source_suffixes: + problems.append( + 'no source suffixes - set `languages` (e.g. ["rust"]) or ' + '`source_suffixes` (e.g. [".rs"])') + for suffix in sorted(self.source_suffixes): + if not suffix.startswith("."): + problems.append(f"source suffix {suffix!r} must start with a dot") + if not self.source_roots: + problems.append( + "no source roots - set `source_roots` to the directories " + "holding this repo's code") + if not 0.0 <= self.min_coverage <= 100.0: + problems.append( + f"min_coverage {self.min_coverage} is not a percentage") + if problems: + raise ConfigError("; ".join(problems)) -def configure_suffixes(suffixes: Sequence[str]) -> None: - """Rebind the file extensions scanned for TRACES tags.""" - global SOURCE_SUFFIXES - SOURCE_SUFFIXES = {s if s.startswith(".") else f".{s}" for s in suffixes} - - -def configure_scan_roots(roots: Sequence[str]) -> None: - """Rebind the directories walked, relative to --root.""" - global SCAN_ROOTS - SCAN_ROOTS = tuple(roots) - -EXCLUDED_DIR_NAMES = { - ".git", "__pycache__", "external", "build", "site", "node_modules", - "trt_cache", "ort_cache", ".venv", "venv", ".mypy_cache", ".pytest_cache", +CONFIG_KEYS = { + "requirement_types", "test_types", "external_types", "languages", + "source_suffixes", "source_roots", "exclude_dirs", "ci_executable_tiers", + "requirements", "system_spec", "matrix", "json_report", "min_coverage", + "allow_orphans", } + +def example_config() -> str: + """The full schema, emitted by ``--print-example-config``.""" + return """\ +# traceability.toml - per-repo configuration for the shared trace extractor. +# Lives at the component repo root; its directory is taken as the repo root. + +# REQUIRED. The ID prefixes this repo's register defines. These, and only +# these, form the coverage fraction. +# scene-actor-extraction: ["AR", "DP", "IR", "GR", "VR"] +# jRay: ["JR"] +# JRay-public-server: ["UR", "DR"] +requirement_types = ["AR", "DP", "IR", "GR", "VR"] + +# REQUIRED (at least one of the two). `languages` names suffix groups; +# `source_suffixes` adds anything else. Known groups: cpp, python, rust, +# csharp, javascript, typescript, svelte, go, java. +languages = ["cpp", "python"] +# source_suffixes = [".inl"] + +# REQUIRED. Directories to scan, relative to the repo root. +source_roots = ["src", "tests", "scripts"] + +# Optional; defaults shown. +# test_types = ["UT", "IT"] # evidence for requirements, not counted +# external_types = ["PR", "SR"] # owned by the system spec, not counted +# requirements = "docs/requirements.md" +# matrix = "docs/traceability.md" +# json_report = "traces-report.json" +# min_coverage = 0.0 +# allow_orphans = false +# exclude_dirs = ["external", "build", "target", "node_modules"] + +# Which verification tiers this repo's CI host can actually execute. A +# requirement whose tiers all fall outside this set is reported as tagged but +# unexecuted, and is never counted as covered. +# ci_executable_tiers = ["T1", "T2", "T3", "static"] + +# The system spec defining PR/SR, vendored as a submodule in each component so +# the check is runnable in CI. Omit to skip PR/SR orphan checking. +# system_spec = "scripts/vendor/jray-project/SPEC.md" +""" + + +def find_config(start: Path) -> Optional[Path]: + """Nearest ``traceability.toml`` at or above ``start``. + + The file's directory defines the repo root, which is what lets the tool be + run from a subdirectory without silently scanning the wrong tree. + """ + current = start.resolve() + for candidate in (current, *current.parents): + path = candidate / CONFIG_FILENAME + if path.is_file(): + return path + return None + + +def _load_toml(path: Path) -> dict: + try: + import tomllib + except ModuleNotFoundError as exc: # pragma: no cover - version dependent + raise ConfigError( + f"reading {path} needs Python 3.11+ for tomllib (this is " + f"{sys.version_info.major}.{sys.version_info.minor}). Either " + "upgrade, or pass --requirement-type/--language/--source-root on " + "the command line instead of using a config file.") from exc + with path.open("rb") as handle: + return tomllib.load(handle) + + +def config_from_dict(data: dict, root: Path, source: str) -> Config: + unknown = sorted(set(data) - CONFIG_KEYS) + if unknown: + # A typo'd key would otherwise leave a required field empty, and the + # user would be debugging "zero requirements" instead of a spelling. + raise ConfigError( + f"unknown key(s) in {source}: {', '.join(unknown)}. " + f"Known keys: {', '.join(sorted(CONFIG_KEYS))}") + + suffixes: Set[str] = set() + for language in data.get("languages", []): + if language not in LANGUAGE_SUFFIXES: + raise ConfigError( + f"unknown language {language!r} in {source}. Known: " + f"{', '.join(sorted(LANGUAGE_SUFFIXES))}") + suffixes |= LANGUAGE_SUFFIXES[language] + suffixes |= set(data.get("source_suffixes", [])) + + return Config( + requirement_types=tuple(data.get("requirement_types", ())), + source_suffixes=frozenset(suffixes), + source_roots=tuple(data.get("source_roots", ())), + root=root, + test_types=tuple(data.get("test_types", DEFAULT_TEST_TYPES)), + external_types=tuple(data.get("external_types", DEFAULT_EXTERNAL_TYPES)), + exclude_dirs=frozenset(data.get("exclude_dirs", DEFAULT_EXCLUDE_DIRS)), + ci_executable_tiers=frozenset( + data.get("ci_executable_tiers", DEFAULT_CI_EXECUTABLE_TIERS)), + requirements_path=data.get("requirements", DEFAULT_REQUIREMENTS_PATH), + system_spec_path=data.get("system_spec"), + matrix_path=data.get("matrix", DEFAULT_MATRIX_PATH), + json_path=data.get("json_report", DEFAULT_JSON_PATH), + min_coverage=float(data.get("min_coverage", 0.0)), + allow_orphans=bool(data.get("allow_orphans", False)), + source=source, + ) + + +def load_config(args: argparse.Namespace, cwd: Optional[Path] = None) -> Config: + """Merge config file and CLI flags, per field. Flags win.""" + cwd = (cwd or Path.cwd()).resolve() + + config_path: Optional[Path] = args.config + if config_path is None and not args.no_config: + # Search from --root when given, falling back to the working directory. + # + # Searching only from cwd is wrong for the two cases this tool is built + # for: CI invoking it with an explicit --root, and a wrapper in a + # consuming repo running the vendored copy. Both name the repo they mean + # and would otherwise get "requirement_types is empty" while a perfectly + # good traceability.toml sat in the directory they pointed at. + config_path = find_config(Path(args.root).resolve() if args.root else cwd) + if config_path is not None and not Path(config_path).is_file(): + raise ConfigError(f"config file not found: {config_path}") + + if config_path is not None: + config_path = Path(config_path).resolve() + config = config_from_dict(_load_toml(config_path), config_path.parent, + str(config_path)) + else: + config = Config(requirement_types=(), source_suffixes=frozenset(), + source_roots=(), root=cwd, source="command line only") + + if args.root is not None: + config = replace(config, root=Path(args.root).resolve()) + + if args.requirement_type: + config = replace(config, requirement_types=tuple(args.requirement_type)) + if args.test_type: + config = replace(config, test_types=tuple(args.test_type)) + if args.external_type: + config = replace(config, external_types=tuple(args.external_type)) + + # Languages and suffixes are additive on top of whatever the file declared, + # so `--language rust` extends rather than silently replacing. + extra: Set[str] = set() + for language in args.language or []: + if language not in LANGUAGE_SUFFIXES: + raise ConfigError( + f"unknown language {language!r}. Known: " + f"{', '.join(sorted(LANGUAGE_SUFFIXES))}") + extra |= LANGUAGE_SUFFIXES[language] + extra |= set(args.source_suffix or []) + if extra: + config = replace(config, source_suffixes=config.source_suffixes | extra) + + if args.source_root: + config = replace(config, source_roots=tuple(args.source_root)) + if args.exclude_dir: + config = replace(config, + exclude_dirs=config.exclude_dirs | set(args.exclude_dir)) + if args.ci_executable_tier: + config = replace(config, + ci_executable_tiers=frozenset(args.ci_executable_tier)) + if args.requirements is not None: + config = replace(config, requirements_path=str(args.requirements)) + if args.system_spec is not None: + config = replace(config, system_spec_path=str(args.system_spec)) + if args.markdown_out is not None: + config = replace(config, matrix_path=str(args.markdown_out)) + if args.json_out is not None: + config = replace(config, json_path=str(args.json_out)) + if args.no_write: + config = replace(config, matrix_path=None, json_path=None) + if args.min_coverage is not None: + config = replace(config, min_coverage=float(args.min_coverage)) + if args.allow_orphans: + config = replace(config, allow_orphans=True) + + config.validate() + return config + + +# -------------------------------------------------------------------------- +# Tag parsing +# -------------------------------------------------------------------------- + TRACES_RE = re.compile(r"TRACES:[ \t]*([^\n]*)") EXCEPTION_RE = re.compile(r"EXCEPTION:[ \t]*([A-Z]{2}-\d{3})[ \t]*([^\n]*)") REQ_ID_RE = re.compile(r"^([A-Z]{2})-(\d{3})$") LEADING_ID_RE = re.compile(r"^([A-Z]{2}-\d{3})\b(.*)$") -#: Something that was *trying* to be a requirement ID. Used to keep the -#: malformed-tag diagnostic quiet when the word "TRACES" merely appears in -#: prose or in this tool's own source, while still catching `AR-12`. +#: Something that was *trying* to be a requirement ID. Keeps the malformed-tag +#: diagnostic quiet when the word "TRACES" merely appears in prose or in this +#: tool's own source, while still catching `AR-12`. ID_ATTEMPT_RE = re.compile(r"[A-Za-z]{2}-\d") -# Comment terminators that can trail a tag on the same line. +#: Comment terminators that can trail a tag on the same line. COMMENT_TERMINATORS = ("*/", "-->", '"""', "'''") DECL_PATTERNS = [ - re.compile(r"^\s*(?:async\s+)?def\s+\w+"), - re.compile(r"^\s*class\s+\w+"), + re.compile(r"^\s*(?:async\s+)?def\s+\w+"), # Python + re.compile(r"^\s*(?:pub(?:\([^)]*\))?\s+)?(?:async\s+)?" + r"(?:unsafe\s+)?(?:extern\s+\"[^\"]*\"\s+)?fn\s+\w+"), # Rust + re.compile(r"^\s*(?:pub(?:\([^)]*\))?\s+)?" + r"(?:impl|trait|mod|type)\b"), # Rust re.compile(r"^\s*(?:template\s*<[^;]*>\s*)?" - r"(?:struct|class|enum(?:\s+class)?|union|namespace)\s+\w+"), - re.compile(r"^\s*(?:static|inline|constexpr|virtual|explicit|friend)\b"), - re.compile(r"^\s*[A-Za-z_][\w:<>,\s\*&]*\s+[A-Za-z_~][\w:]*\s*\([^)]*\)"), + r"(?:struct|class|enum(?:\s+class)?|union|namespace|" + r"interface|record)\s+\w+"), # C++/C# + re.compile(r"^\s*(?:public|private|protected|internal|static|inline|" + r"constexpr|virtual|explicit|friend|override|abstract|sealed|" + r"partial|extern)\b"), + re.compile(r"^\s*[A-Za-z_][\w:<>,\s\*&\.\[\]]*\s+[A-Za-z_~][\w:]*\s*\([^)]*\)"), ] +#: Attribute/annotation lines: C# ``[HttpGet("x")]``, Rust ``#[derive(...)]``, +#: Java ``@Override``. A tag above a decorated declaration must attribute to +#: the declaration, not to its decoration. +ATTRIBUTE_RE = re.compile(r"^\s*(?:\[[^\]]*\]|#\s*\[|@\w+)") + @dataclass class TraceEntry: @@ -262,12 +536,16 @@ def parse_traces_tag(value: str) -> Tuple[List[List[str]], List[str]]: def find_context(lines: Sequence[str], index: int, window: int = 12) -> str: """Best-effort name of the declaration a tag belongs to. - Searches both directions, because the two languages put the tag on opposite - sides of the thing it describes: a C++ ``/// TRACES:`` sits *above* the + Searches both directions, because languages put the tag on opposite sides + of the thing it describes: a C++/C#/Rust ``///`` tag sits *above* the declaration, while a Python tag usually sits *inside* the docstring, below - the ``def``. Whichever declaration is nearer wins. + the ``def``. Whichever declaration is nearer wins. Attribute lines + (``[HttpGet]``, ``#[derive]``) are skipped so a tag above a decorated + method attributes to the method rather than to its decoration. """ def matches(line: str) -> bool: + if ATTRIBUTE_RE.match(line): + return False return any(p.match(line) for p in DECL_PATTERNS) down: Optional[Tuple[int, str]] = None @@ -288,7 +566,6 @@ def find_context(lines: Sequence[str], index: int, window: int = 12) -> str: up = (offset, lines[i]) break - best = None if down and up: best = down if down[0] <= up[0] else up else: @@ -298,25 +575,19 @@ def find_context(lines: Sequence[str], index: int, window: int = 12) -> str: return best[1].strip().rstrip("{").strip()[:120] or "Unknown" -def iter_source_files(root: Path, - scan_roots: Optional[Iterable[str]] = None) -> List[Path]: - """Every source file, of the configured suffixes, under the scan roots. - - `scan_roots` defaults to `None` rather than to `SCAN_ROOTS` directly: a - default argument is bound once at import, so it would capture the value from - before `configure_scan_roots()` ran and silently ignore the override. - """ +def iter_source_files(config: Config) -> List[Path]: + """Every source file under the configured roots, by configured suffix.""" found: List[Path] = [] - for rel in (SCAN_ROOTS if scan_roots is None else scan_roots): - base = root / rel + for rel in config.source_roots: + base = config.root / rel if not base.is_dir(): continue for path in sorted(base.rglob("*")): if not path.is_file(): continue - if path.suffix not in SOURCE_SUFFIXES: + if path.suffix not in config.source_suffixes: continue - if any(part in EXCLUDED_DIR_NAMES for part in path.parts): + if any(part in config.exclude_dirs for part in path.parts): continue found.append(path) return found @@ -330,10 +601,12 @@ class ScanResult: diagnostics: Diagnostics -def scan_files(files: Sequence[Path], root: Path) -> ScanResult: +def scan_files(files: Sequence[Path], config: Config) -> ScanResult: traces: List[TraceEntry] = [] exceptions: List[ExceptionEntry] = [] diags = Diagnostics() + root = config.root + known = set(config.known_types) for path in files: try: @@ -352,14 +625,12 @@ def scan_files(files: Sequence[Path], root: Path) -> ScanResult: if cut != -1: reason = reason[:cut] reason = reason.strip() - entry = ExceptionEntry( + exceptions.append(ExceptionEntry( file=rel, line=index + 1, requirement=exc.group(1), - reason=reason, context=find_context(lines, index), - ) - exceptions.append(entry) + reason=reason, context=find_context(lines, index))) if not reason: - # CLAUDE.md: an exception is only agreed if the reason is - # recorded. An unexplained one is an undocumented defect. + # An exception is only *agreed* if the reason is recorded; + # an unexplained one is an undocumented defect. diags.exceptions_without_reason.append( {"file": rel, "line": index + 1, "requirement": exc.group(1)}) @@ -372,12 +643,13 @@ def scan_files(files: Sequence[Path], root: Path) -> ScanResult: ids = [i for g in groups for i in g] if not ids: # No IDs at all. Only worth reporting if something in the text - # was trying to be one — otherwise every sentence containing + # was trying to be one - otherwise every sentence containing # the word would be flagged, and a diagnostic nobody can act on # is how real diagnostics get ignored. if junk and ID_ATTEMPT_RE.search(match.group(1)): diags.malformed_tags.append( - {"file": rel, "line": index + 1, "text": match.group(1).strip()}) + {"file": rel, "line": index + 1, + "text": match.group(1).strip()}) continue if junk: diags.malformed_tags.append( @@ -390,11 +662,10 @@ def scan_files(files: Sequence[Path], root: Path) -> ScanResult: diags.mixed_type_groups.append( {"file": rel, "line": index + 1, "group": list(group)}) for req in ids: - if req.split("-")[0] not in KNOWN_TYPES: + if req.split("-")[0] not in known: diags.unknown_id_types.append( {"file": rel, "line": index + 1, "id": req}) - # Deduplicate within one tag while preserving order. seen: Set[str] = set() unique = [i for i in ids if not (i in seen or seen.add(i))] traces.append(TraceEntry(file=rel, line=index + 1, @@ -406,7 +677,7 @@ def scan_files(files: Sequence[Path], root: Path) -> ScanResult: # -------------------------------------------------------------------------- -# The register: docs/requirements.md is the authoritative denominator +# The register: requirements.md is the authoritative denominator # -------------------------------------------------------------------------- @dataclass @@ -418,17 +689,6 @@ class Requirement: status: str = "" tiers: Set[str] = field(default_factory=set) - @property - def ci_executable(self) -> bool: - """True when at least one verification tier can run on the CI host. - - A requirement with no tier recorded is *unknown*, not unexecutable — it - is reported so the register gets fixed, and it is not penalised here. - """ - if not self.tiers: - return True - return bool(self.tiers & CI_EXECUTABLE_TIERS) - @property def tier_known(self) -> bool: return bool(self.tiers) @@ -436,6 +696,10 @@ class Requirement: @dataclass class Register: + #: The ID prefixes this register defines, from configuration. + types: Tuple[str, ...] = () + #: Verification tiers the consuming repo's CI host can execute. + ci_tiers: FrozenSet[str] = DEFAULT_CI_EXECUTABLE_TIERS requirements: Dict[str, Requirement] = field(default_factory=dict) withdrawn: Dict[str, Requirement] = field(default_factory=dict) @@ -447,14 +711,35 @@ class Register: def total(self) -> int: return len(self.requirements) + @property + def uses_tiers(self) -> bool: + """Whether this register assigns verification tiers at all. + + A repo with no tier tables is not 'missing' tiers; it does not use the + mechanism. Warning about every requirement there would be noise that + trains people to ignore the warning that matters. + """ + return any(r.tier_known for r in self.requirements.values()) + + def is_ci_executable(self, req_id: str) -> bool: + """True when at least one verification tier can run on the CI host. + + A requirement with no tier recorded is *unknown*, not unexecutable - it + is reported so the register gets fixed, and it is not penalised here. + """ + req = self.requirements[req_id] + if not req.tiers: + return True + return bool(req.tiers & self.ci_tiers) + def count(self, req_type: str) -> int: return sum(1 for i in self.requirements if i.startswith(req_type + "-")) def ci_executable_ids(self) -> Set[str]: - return {i for i, r in self.requirements.items() if r.ci_executable} + return {i for i in self.requirements if self.is_ci_executable(i)} def unexecutable_ids(self) -> Set[str]: - return {i for i, r in self.requirements.items() if not r.ci_executable} + return {i for i in self.requirements if not self.is_ci_executable(i)} def tier_unknown_ids(self) -> Set[str]: return {i for i, r in self.requirements.items() if not r.tier_known} @@ -464,7 +749,7 @@ class Register: SPEC.md section 6 asks the gate to report these: a requirement serving no stated goal is scope creep, and it is invisible unless something - looks. A cell naming a section (`§4`) counts as a parent — the point is + looks. A cell naming a section (`§4`) counts as a parent - the point is that *something* was recorded, not that it was an ID. """ out = set() @@ -476,7 +761,11 @@ class Register: def _row_cells(line: str) -> List[str]: - return [c.strip() for c in line.strip().strip("|").split("|")] + r"""Split a markdown table row, honouring escaped ``\|`` inside cells.""" + placeholder = "\x00" + stripped = line.strip().replace("\\|", placeholder) + cells = stripped.strip("|").split("|") + return [c.strip().replace(placeholder, "|") for c in cells] def _is_separator(cells: Sequence[str]) -> bool: @@ -512,7 +801,7 @@ def _is_register_header(header: Sequence[str]) -> bool: Deliberately narrow. The per-requirement verification plan is also keyed on ``ID`` but has no ``Requirement`` column, and the tier summary table's first - column holds comma lists and ranges — neither defines requirements, and + column holds comma lists and ranges - neither defines requirements, and counting their rows would inflate the denominator. Prose mentions and ``Traces to`` references are excluded for the same reason: a naive scan for ``AR-\\d{3}`` over the whole file counts every reference as a definition. @@ -524,7 +813,7 @@ def _is_register_header(header: Sequence[str]) -> bool: def _is_tier_header(header: Sequence[str]) -> bool: - """Either of the two tables that assign verification tiers.""" + """Either of the two table shapes that assign verification tiers.""" if len(header) < 2: return False lowered = [c.lower() for c in header] @@ -543,7 +832,7 @@ def _column(header: Sequence[str], name: str, cells: Sequence[str]) -> str: def parse_tiers(cell: str) -> Set[str]: """Read a tier cell such as ``T1 + T4``, ``**T2**``, or ``Out of CI``.""" text = cell.replace("*", "").strip().lower() - tiers = {"T" + m.group(1) for m in re.finditer(r"\bt([1-4])\b", text)} + tiers = {"T" + m.group(1) for m in re.finditer(r"\bt(\d)\b", text)} if "out of ci" in text or "not in ci" in text: tiers.add("out-of-ci") if "manual" in text: @@ -556,7 +845,7 @@ def parse_tiers(cell: str) -> Set[str]: def expand_id_spec(spec: str, defined: Set[str]) -> List[str]: """Expand the ID cell of a tier table into concrete, defined IDs. - Handles every shape the register actually uses: ``AR-002``, + Handles every shape the registers actually use: ``AR-002``, ``AR-001, AR-005, AR-006``, ``AR-007 ... AR-017`` (with the unicode ellipsis), ``AR-009/010``, and ``DP-*``. Expansion is intersected with the defined set, so a range can never invent a requirement that does not exist. @@ -592,13 +881,15 @@ def expand_id_spec(spec: str, defined: Set[str]) -> List[str]: return [i for i in out if i in defined] -def parse_register(markdown: str) -> Register: - """Build the register from ``docs/requirements.md``. +def parse_register(markdown: str, config: Config) -> Register: + """Build the register from the repo's requirements.md. Two passes: definitions first, because tier rows use ranges and wildcards that can only be expanded against a known ID set. """ - register = Register() + register = Register(types=tuple(config.requirement_types), + ci_tiers=frozenset(config.ci_executable_tiers)) + definable = set(config.requirement_types) | set(config.test_types) for header, cells in _iter_table_rows(markdown): if not _is_register_header(header) or not cells: @@ -606,7 +897,7 @@ def parse_register(markdown: str) -> Register: first = cells[0].replace("*", "").strip() if not REQ_ID_RE.match(first): continue - if first.split("-")[0] not in LOCAL_TYPES + TEST_TYPES: + if first.split("-")[0] not in definable: continue req = Requirement( id=first, @@ -640,8 +931,10 @@ def parse_register(markdown: str) -> Register: return register -def read_register(path: Path) -> Register: - return parse_register(path.read_text(encoding="utf-8")) +def read_register(config: Config) -> Register: + path = config.resolve(config.requirements_path) + assert path is not None + return parse_register(path.read_text(encoding="utf-8"), config) # -------------------------------------------------------------------------- @@ -679,22 +972,22 @@ def compute_coverage(traced_ids: Iterable[str], register: Register) -> Coverage: Three exclusions, each of which is a way the number could otherwise lie: * Using the raw traced count as the numerator is what lets a ratio exceed - 100% — a tag naming a deleted or mistyped requirement would count as + 100% - a tag naming a deleted or mistyped requirement would count as covered. Those land in ``orphaned`` instead, so they get fixed rather than silently counted or silently dropped. - * A requirement whose only verification tier is T4/GPU cannot run on this - CI host at all. It is reported in ``unexecuted`` and is not covered: - treating "has a test that never runs" as passing reports success the gate - cannot substantiate. - * UT/IT test IDs and the system-level PR/SR IDs are different taxonomies - with their own registers, so they neither count nor orphan here. + * A requirement whose verification tiers all fall outside what the CI host + can execute cannot be verified there at all. It is reported in + ``unexecuted`` and is not covered: treating "has a test that never runs" + as passing reports success the gate cannot substantiate. + * Test IDs and system-level IDs are separate taxonomies with their own + registers, so they neither count nor orphan here. """ - traced = {i for i in traced_ids if i.split("-")[0] in LOCAL_TYPES} + traced = {i for i in traced_ids if i.split("-")[0] in register.types} defined = register.ids orphaned = sorted(traced - defined) matched = traced & defined - unexecuted = sorted(i for i in matched if not register.requirements[i].ci_executable) + unexecuted = sorted(i for i in matched if not register.is_ci_executable(i)) covered = sorted(matched - set(unexecuted)) total = len(defined) @@ -721,28 +1014,26 @@ def compute_coverage(traced_ids: Iterable[str], register: Register) -> Coverage: @dataclass class Report: timestamp: str - root: Path + config: Config register: Register scan: ScanResult coverage: Coverage by_type: Dict[str, List[str]] requirement_map: Dict[str, List[TraceEntry]] external_orphans: List[str] = field(default_factory=list) - #: Policy, set by the caller: whether an orphan tag fails the run. - allow_orphans: bool = False -def build_report(root: Path, register: Register, scan: ScanResult, +def build_report(config: Config, register: Register, scan: ScanResult, system_ids: Optional[Set[str]] = None) -> Report: requirement_map: Dict[str, List[TraceEntry]] = {} - seen_by_type: Dict[str, Set[str]] = {t: set() for t in KNOWN_TYPES} + seen_by_type: Dict[str, Set[str]] = {t: set() for t in config.known_types} seen_by_type["OTHER"] = set() for entry in scan.traces: for req in entry.requirements: requirement_map.setdefault(req, []).append(entry) prefix = req.split("-")[0] - seen_by_type[prefix if prefix in KNOWN_TYPES else "OTHER"].add(req) + seen_by_type[prefix if prefix in seen_by_type else "OTHER"].add(req) coverage = compute_coverage(requirement_map.keys(), register) @@ -750,11 +1041,11 @@ def build_report(root: Path, register: Register, scan: ScanResult, if system_ids is not None: external_orphans = sorted( i for i in requirement_map - if i.split("-")[0] in EXTERNAL_TYPES and i not in system_ids) + if i.split("-")[0] in config.external_types and i not in system_ids) return Report( timestamp=datetime.now(timezone.utc).isoformat(timespec="seconds"), - root=root, + config=config, register=register, scan=scan, coverage=coverage, @@ -764,16 +1055,20 @@ def build_report(root: Path, register: Register, scan: ScanResult, ) -def parse_system_spec(markdown: str) -> Set[str]: - """IDs of PR/SR requirements defined in the umbrella SPEC.md. +def parse_system_spec(markdown: str, + types: Sequence[str] = DEFAULT_EXTERNAL_TYPES) -> Set[str]: + """IDs of the system requirements defined in the umbrella SPEC.md. - They are defined as headings (``### SR-001 - ...``) and as bolded leading - table cells (``| **PR-001** | ... |``), so accept both. + They appear as headings (``### SR-001 - ...``) and as bolded leading table + cells (``| **PR-001** | ... |``), so accept both. """ + alternation = "|".join(re.escape(t) for t in types) + heading_re = re.compile(rf"^#{{1,6}}\s+\**((?:{alternation})-\d{{3}})\**\b") + cell_re = re.compile(rf"(?:{alternation})-\d{{3}}") ids: Set[str] = set() for line in markdown.split("\n"): stripped = line.strip() - heading = re.match(r"^#{1,6}\s+\**((?:PR|SR)-\d{3})\**\b", stripped) + heading = heading_re.match(stripped) if heading: ids.add(heading.group(1)) continue @@ -781,50 +1076,11 @@ def parse_system_spec(markdown: str) -> Set[str]: cells = _row_cells(stripped) if cells: first = cells[0].replace("*", "").strip() - if re.fullmatch(r"(?:PR|SR)-\d{3}", first): + if cell_re.fullmatch(first): ids.add(first) return ids -def report_to_json_dict(report: Report) -> dict: - register = report.register - defined = {t: register.count(t) for t in LOCAL_TYPES} - defined["total"] = register.total - - requirement_detail = {} - for req_id, req in sorted(register.requirements.items()): - entries = report.requirement_map.get(req_id, []) - requirement_detail[req_id] = { - "requirement": req.text, - "tracesTo": req.traces_to, - "priority": req.priority, - "status": req.status, - "tiers": sorted(req.tiers), - "ciExecutable": req.ci_executable, - "taggedIn": sorted({e.file for e in entries}), - "state": _requirement_state(req, bool(entries)), - } - - return { - "timestamp": report.timestamp, - "totalFiles": len(report.scan.files), - "totalTraces": len(report.scan.traces), - "totalExceptions": len(report.scan.exceptions), - "requirements": {k: [e.to_dict() for e in v] - for k, v in sorted(report.requirement_map.items())}, - "byType": report.by_type, - "defined": defined, - "coverage": report.coverage.to_dict(), - "gpuOnlyRequirements": sorted(register.unexecutable_ids()), - "parentlessRequirements": sorted(register.parentless_ids()), - "withdrawn": sorted(register.withdrawn), - "exceptions": [e.to_dict() for e in report.scan.exceptions], - "externalOrphans": report.external_orphans, - "diagnostics": report.scan.diagnostics.to_dict(), - "requirementDetail": requirement_detail, - } - - def per_type_stats(report: Report) -> Dict[str, Tuple[int, int, int]]: """``{type: (covered, unexecuted, defined)}``. @@ -836,7 +1092,7 @@ def per_type_stats(report: Report) -> Dict[str, Tuple[int, int, int]]: covered = set(report.coverage.covered) unexecuted = set(report.coverage.unexecuted) stats: Dict[str, Tuple[int, int, int]] = {} - for req_type in LOCAL_TYPES: + for req_type in report.register.types: prefix = req_type + "-" stats[req_type] = ( len([i for i in covered if i.startswith(prefix)]), @@ -846,36 +1102,93 @@ def per_type_stats(report: Report) -> Dict[str, Tuple[int, int, int]]: return stats -def _requirement_state(req: Requirement, tagged: bool) -> str: +def _requirement_state(register: Register, req_id: str, tagged: bool) -> str: if not tagged: return "untagged" - if not req.ci_executable: + if not register.is_ci_executable(req_id): return "tagged-unexecuted" return "covered" +def report_to_json_dict(report: Report) -> dict: + register = report.register + config = report.config + defined = {t: register.count(t) for t in register.types} + defined["total"] = register.total + + requirement_detail = {} + for req_id, req in sorted(register.requirements.items()): + entries = report.requirement_map.get(req_id, []) + requirement_detail[req_id] = { + "requirement": req.text, + "tracesTo": req.traces_to, + "priority": req.priority, + "status": req.status, + "tiers": sorted(req.tiers), + "ciExecutable": register.is_ci_executable(req_id), + "taggedIn": sorted({e.file for e in entries}), + "state": _requirement_state(register, req_id, bool(entries)), + } + + return { + "timestamp": report.timestamp, + # Echoed so a report can be read without guessing how it was produced - + # a shared tool's output is ambiguous otherwise. + "config": { + "source": config.source, + "root": str(config.root), + "requirementTypes": list(config.requirement_types), + "testTypes": list(config.test_types), + "externalTypes": list(config.external_types), + "sourceSuffixes": sorted(config.source_suffixes), + "sourceRoots": list(config.source_roots), + "requirements": config.requirements_path, + "systemSpec": config.system_spec_path, + "ciExecutableTiers": sorted(config.ci_executable_tiers), + "minCoverage": config.min_coverage, + }, + "totalFiles": len(report.scan.files), + "totalTraces": len(report.scan.traces), + "totalExceptions": len(report.scan.exceptions), + "requirements": {k: [e.to_dict() for e in v] + for k, v in sorted(report.requirement_map.items())}, + "byType": report.by_type, + "defined": defined, + "coverage": report.coverage.to_dict(), + "unexecutableRequirements": sorted(register.unexecutable_ids()), + "parentlessRequirements": sorted(register.parentless_ids()), + "withdrawn": sorted(register.withdrawn), + "exceptions": [e.to_dict() for e in report.scan.exceptions], + "externalOrphans": report.external_orphans, + "diagnostics": report.scan.diagnostics.to_dict(), + "requirementDetail": requirement_detail, + } + + # -------------------------------------------------------------------------- # Output formats # -------------------------------------------------------------------------- def generate_markdown(report: Report) -> str: register = report.register + config = report.config cov = report.coverage out: List[str] = [] add = out.append + req_link = Path(config.requirements_path).name + add("# Requirements traceability matrix") add("") add("") - add("") + add("") add("") add(f"**Generated:** {report.timestamp}") add("") - add("Denominators are read from [`requirements.md`](requirements.md) at run " - "time, never hardcoded. Coverage counts a requirement only when it is " - "tagged in source **and** has a verification tier this CI host can " - "execute — CI is an Intel N100 with no discrete GPU.") + add(f"Denominators are read from [`{req_link}`]({req_link}) at run time, " + "never hardcoded. Coverage counts a requirement only when it is tagged " + "in source **and** has a verification tier this repo's CI host can " + f"execute (`{', '.join(sorted(config.ci_executable_tiers))}`).") add("") add("## Summary") @@ -890,7 +1203,7 @@ def generate_markdown(report: Report) -> str: add(f"| **Coverage** | **{cov.percent}%** ({len(cov.covered)}/{cov.total}) |") add(f"| Coverage of CI-executable scope | {cov.ci_percent}% " f"({len(cov.covered)}/{cov.ci_executable_total}) |") - add(f"| Tagged but unexecuted in CI (T4/GPU) | {len(cov.unexecuted)} |") + add(f"| Tagged but unexecuted in CI | {len(cov.unexecuted)} |") add(f"| Orphan tags | {len(cov.orphaned)} |") add("") @@ -901,7 +1214,7 @@ def generate_markdown(report: Report) -> str: for req_type, (covered, unexecuted, defined_n) in per_type_stats(report).items(): add(f"| {req_type} | {covered} | {unexecuted} | {defined_n} |") add("") - for req_type in TEST_TYPES + EXTERNAL_TYPES: + for req_type in tuple(config.test_types) + tuple(config.external_types): tagged = report.by_type.get(req_type, []) if tagged: add(f"- **{req_type}** tags present (separate taxonomy, not counted " @@ -911,9 +1224,9 @@ def generate_markdown(report: Report) -> str: unexecutable = sorted(register.unexecutable_ids()) add("## Not executable in CI") add("") - add("CI runs on an Intel N100 with no discrete GPU. These requirements have " - "no verification tier that can run here, so a tag on them is evidence " - "of *intent*, not of verification. They are never counted as covered.") + add("These requirements have no verification tier this repo's CI host can " + "run, so a tag on them is evidence of *intent*, not of verification. " + "They are never counted as covered.") add("") if unexecutable: add("| ID | Tiers | Tagged in source | Requirement |") @@ -928,13 +1241,13 @@ def generate_markdown(report: Report) -> str: add("") if cov.unexecuted: add(f"**Tagged but unexecuted:** {', '.join(cov.unexecuted)} — a test " - "exists and is tagged, but only a GPU host can run it. Report " + "exists and is tagged, but this CI host cannot run it. Report " "those runs separately.") add("") add("## Orphan tags") add("") - add("A tag naming an ID `requirements.md` does not define. This is what " + add(f"A tag naming an ID `{req_link}` does not define. This is what " "renumbering produces, and what a typo produces.") add("") if cov.orphaned: @@ -955,16 +1268,13 @@ def generate_markdown(report: Report) -> str: "something looks.") add("") parentless = sorted(register.parentless_ids()) - if parentless: - add(", ".join(f"`{i}`" for i in parentless)) - else: - add("_None._") + add(", ".join(f"`{i}`" for i in parentless) if parentless else "_None._") add("") add("## Recorded exceptions") add("") add("Deliberate, documented departures from an invariant " - "(`EXCEPTION: AR-nnn `). Reported separately and never counted " + "(`EXCEPTION: XX-nnn `). Reported separately and never counted " "as coverage — an exception is a decision to be reviewed, not evidence " "a requirement is met.") add("") @@ -983,15 +1293,15 @@ def generate_markdown(report: Report) -> str: add("") add("| ID | Status | Tier | Traces to | Trace state | Tagged in | Requirement |") add("|---|---|---|---|---|---|---|") - for req_id, req in sorted(register.requirements.items(), - key=lambda kv: (LOCAL_TYPES.index(kv[0].split("-")[0]) - if kv[0].split("-")[0] in LOCAL_TYPES - else 99, kv[0])): + order = {t: n for n, t in enumerate(register.types)} + for req_id, req in sorted( + register.requirements.items(), + key=lambda kv: (order.get(kv[0].split("-")[0], 99), kv[0])): entries = report.requirement_map.get(req_id, []) state = {"covered": "covered", - "tagged-unexecuted": "tagged, unexecuted (T4/GPU)", + "tagged-unexecuted": "tagged, unexecuted", "untagged": "untagged"}[ - _requirement_state(req, bool(entries))] + _requirement_state(register, req_id, bool(entries))] files = ", ".join(f"`{f}`" for f in sorted({e.file for e in entries})) or "-" tiers = ", ".join(sorted(req.tiers)) or "unset" add(f"| {req_id} | {_truncate(req.status, 20) or '-'} | {tiers} " @@ -1041,7 +1351,7 @@ def generate_markdown(report: Report) -> str: def _source_link(file_rel: str) -> str: - """Link from docs/traceability.md back to a source file at the repo root.""" + """Link from the matrix (in docs/) back to a source file at the repo root.""" return "../" + file_rel @@ -1053,9 +1363,10 @@ def _truncate(text: str, limit: int = 70) -> str: return text.replace("|", "\\|") -def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, int]: +def format_coverage_report(report: Report) -> Tuple[str, int]: """Human-readable gate output plus the exit code it implies.""" register = report.register + config = report.config cov = report.coverage lines: List[str] = [] add = lines.append @@ -1063,22 +1374,34 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add("Requirement traceability") add("=" * 72) - add(f"Source files scanned : {len(report.scan.files)}") + add(f"Config : {config.source}") + add(f"Repo root : {config.root}") + add(f"Requirement types : {', '.join(config.requirement_types)}") + add(f"Source files scanned : {len(report.scan.files)} " + f"({', '.join(sorted(config.source_suffixes))} under " + f"{', '.join(config.source_roots)})") add(f"TRACES tags found : {len(report.scan.traces)}") add(f"EXCEPTION tags found : {len(report.scan.exceptions)}") add("") - # Self-checks. With a low threshold these are what make the gate mean - # something: a parser that silently returns nothing would otherwise report - # 0/0 and pass. + # Self-checks. These are what make the gate mean something at a low + # threshold, and what makes a misconfigured shared tool loud instead of + # silent: a repo whose requirement_types or source_suffixes are wrong parses + # nothing and scans nothing, and a plausible-looking 0% would hide it. if register.total == 0: failures.append( - "requirements.md parsed to ZERO requirements - the register parser " - "is broken or the file moved. Refusing to report coverage.") + f"{config.requirements_path} parsed to ZERO requirements. Either " + "the path is wrong, the register parser is broken, or " + f"requirement_types ({', '.join(config.requirement_types) or 'none'}) " + "does not match the prefixes it defines. Refusing to report " + "coverage.") if not report.scan.files: failures.append( - "no source files were scanned - the scan roots do not exist. " - "Refusing to report coverage against an empty tree.") + "no source files were scanned. Either source_roots " + f"({', '.join(config.source_roots) or 'none'}) do not exist, or the " + f"suffixes ({', '.join(sorted(config.source_suffixes)) or 'none'}) " + "do not match this repo's languages. Refusing to report coverage " + "against an empty tree.") add("Coverage by type (covered / defined):") for req_type, (covered, unexecuted, defined_n) in per_type_stats(report).items(): @@ -1088,18 +1411,18 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add(f"Overall : {len(cov.covered)} / {cov.total} ({cov.percent}%)") add(f"CI scope : {len(cov.covered)} / {cov.ci_executable_total} " f"({cov.ci_percent}%) [excludes {cov.total - cov.ci_executable_total} " - "requirement(s) no GPU-less host can verify]") + "requirement(s) this CI host cannot verify]") add("") unexecutable = sorted(register.unexecutable_ids()) if unexecutable: - add(f"Not executable on this CI host (T4/GPU or out-of-CI): " - f"{len(unexecutable)}") + add("Not executable on this CI host (tiers outside " + f"{', '.join(sorted(config.ci_executable_tiers))}): {len(unexecutable)}") add(f" {', '.join(unexecutable)}") if cov.unexecuted: add("") - add("TAGGED BUT UNEXECUTED - a test exists and is tagged, but only a " - "GPU host can run it.") + add("TAGGED BUT UNEXECUTED - a test exists and is tagged, but this CI " + "host cannot run it.") add(" These are NOT counted as covered:") for req_id in cov.unexecuted: tiers = ", ".join(sorted(register.requirements[req_id].tiers)) @@ -1114,12 +1437,18 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add(f" {', '.join(parentless)}") add("") - if cov.tier_unknown: + if cov.tier_unknown and register.uses_tiers: + # Only meaningful where the register uses tiers at all; otherwise every + # requirement is listed and the warning trains people to ignore it. add(f"WARNING: {len(cov.tier_unknown)} requirement(s) have no " - "verification tier in requirements.md; they are counted as " + "verification tier in the register; they are counted as " "CI-executable by default. Add them to the verification plan:") add(f" {', '.join(cov.tier_unknown)}") add("") + elif register.total and not register.uses_tiers: + add("Note: this register assigns no verification tiers, so nothing is " + "excluded as unexecutable.") + add("") if report.scan.exceptions: add(f"Recorded invariant exceptions: {len(report.scan.exceptions)}") @@ -1129,7 +1458,8 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add("") if cov.orphaned: - add("ORPHAN TAGS - traced in source, not defined in requirements.md:") + add("ORPHAN TAGS - traced in source, not defined in " + f"{config.requirements_path}:") for req_id in cov.orphaned: where = ", ".join(f"{e.file}:{e.line}" for e in report.requirement_map[req_id]) @@ -1138,7 +1468,7 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add("") if report.external_orphans: - add("ORPHAN SYSTEM TAGS - PR/SR IDs the system SPEC.md does not define:") + add("ORPHAN SYSTEM TAGS - IDs the system spec does not define:") add(f" {', '.join(report.external_orphans)}") add("") @@ -1168,13 +1498,14 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in f"covered ({len(cov.covered)}) exceeds defined ({cov.total}) - " "the gate is miscomputing.") - if cov.orphaned and not report.allow_orphans: + if cov.orphaned and not config.allow_orphans: failures.append( f"{len(cov.orphaned)} orphan tag(s): {', '.join(cov.orphaned)}") - if cov.percent < min_coverage: + if cov.percent < config.min_coverage: failures.append( - f"coverage ({cov.percent}%) is below the minimum ({min_coverage}%)") + f"coverage ({cov.percent}%) is below the minimum " + f"({config.min_coverage}%)") if failures: add("FAILED:") @@ -1182,7 +1513,7 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in add(f" - {reason}") return "\n".join(lines) + "\n", 1 - add(f"OK: coverage {cov.percent}% >= minimum {min_coverage}%, " + add(f"OK: coverage {cov.percent}% >= minimum {config.min_coverage}%, " f"{len(cov.orphaned)} orphan tag(s)") return "\n".join(lines) + "\n", 0 @@ -1194,94 +1525,114 @@ def format_coverage_report(report: Report, min_coverage: float) -> Tuple[str, in def build_arg_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( prog="extract_traces.py", - description=__doc__.split("\n")[0], + description="Extract requirement traces and report coverage. Shared by " + "every JRay component repo; configured per repo via " + f"{CONFIG_FILENAME} or the flags below.", formatter_class=argparse.RawDescriptionHelpFormatter, ) parser.add_argument("--format", choices=("json", "markdown", "coverage"), default="coverage", help="output written to stdout (default: coverage)") - parser.add_argument("--root", type=Path, default=REPO_ROOT, - help="repository root to scan (default: derived from " - "this script's location)") - parser.add_argument("--requirements", type=Path, default=None, - help="register path (default: /docs/requirements.md)") - parser.add_argument("--system-spec", type=Path, default=None, - help="optional umbrella SPEC.md defining PR/SR IDs; " - "enables orphan checking for those types") - parser.add_argument("--json-out", type=Path, default=None, - help="also write the JSON report here") - parser.add_argument("--markdown-out", type=Path, default=None, - help="also write the markdown matrix here") - parser.add_argument("--min-coverage", type=float, default=0.0, - help="minimum coverage percent; below it the run fails " - "(default: 0, i.e. the correctness checks gate but " - "the percentage does not)") - parser.add_argument("--allow-orphans", action="store_true", - help="report orphan tags without failing") - parser.add_argument("--types", default=None, metavar="UR,DR", - help="comma-separated requirement prefixes defined in " - "this repo's register (default: " - f"{','.join(LOCAL_TYPES)}). A repo whose prefixes " - "are absent parses to zero requirements") - parser.add_argument("--suffixes", default=None, metavar=".rs,.cs", - help="comma-separated file extensions to scan " - "(default: the C++/Python set). A repo whose " - "language is absent scans zero files") - parser.add_argument("--scan-roots", default=None, metavar="src,tests", - help="comma-separated directories to walk, relative to " - f"--root (default: {','.join(SCAN_ROOTS)})") + parser.add_argument("--config", type=Path, default=None, + help=f"path to {CONFIG_FILENAME} (default: the nearest " + "one at or above the working directory)") + parser.add_argument("--no-config", action="store_true", + help="ignore any config file; use flags only") + parser.add_argument("--print-example-config", action="store_true", + help=f"print an annotated {CONFIG_FILENAME} and exit") + parser.add_argument("--root", type=Path, default=None, + help="repository root (default: the config file's " + "directory, else the working directory)") + + group = parser.add_argument_group("what this repo defines") + group.add_argument("--requirement-type", action="append", metavar="XX", + help="ID prefix this repo's register defines; repeatable. " + "Replaces the configured list") + group.add_argument("--test-type", action="append", metavar="XX", + help="test ID prefix, excluded from coverage; repeatable") + group.add_argument("--external-type", action="append", metavar="XX", + help="system-spec ID prefix, excluded from coverage; " + "repeatable") + + group = parser.add_argument_group("what this repo scans") + group.add_argument("--language", action="append", + choices=sorted(LANGUAGE_SUFFIXES), + help="named suffix group; repeatable, additive") + group.add_argument("--source-suffix", action="append", metavar=".EXT", + help="extra source suffix; repeatable, additive") + group.add_argument("--source-root", action="append", metavar="DIR", + help="directory to scan; repeatable. Replaces the " + "configured list") + group.add_argument("--exclude-dir", action="append", metavar="NAME", + help="directory name to skip anywhere in the tree; " + "repeatable, additive") + + group = parser.add_argument_group("paths") + group.add_argument("--requirements", type=Path, default=None, + help=f"register path (default: {DEFAULT_REQUIREMENTS_PATH})") + group.add_argument("--system-spec", type=Path, default=None, + help="SPEC.md defining the external types; enables " + "orphan checking for them") + group.add_argument("--json-out", type=Path, default=None, + help=f"JSON report path (default: {DEFAULT_JSON_PATH})") + group.add_argument("--markdown-out", type=Path, default=None, + help=f"matrix path (default: {DEFAULT_MATRIX_PATH})") + group.add_argument("--no-write", action="store_true", + help="do not write report files, only print") + + group = parser.add_argument_group("policy") + group.add_argument("--min-coverage", type=float, default=None, + help="minimum coverage percent; below it the run fails") + group.add_argument("--allow-orphans", action="store_true", + help="report orphan tags without failing") + group.add_argument("--ci-executable-tier", action="append", metavar="TIER", + help="verification tier this CI host can run; " + "repeatable. Replaces the configured set") return parser -def _split_csv(value: Optional[str]) -> Optional[List[str]]: - if value is None: - return None - items = [v.strip() for v in value.split(",") if v.strip()] - return items or None - - def main(argv: Optional[Sequence[str]] = None) -> int: args = build_arg_parser().parse_args(argv) - # Rebind the per-repo taxonomy before anything reads it. - if types := _split_csv(args.types): - configure_taxonomy(types) - if suffixes := _split_csv(args.suffixes): - configure_suffixes(suffixes) - if scan_roots := _split_csv(args.scan_roots): - configure_scan_roots(scan_roots) + if args.print_example_config: + print(example_config(), end="") + return 0 - root = args.root.resolve() - req_path = args.requirements or (root / "docs" / "requirements.md") - if not req_path.is_file(): + try: + config = load_config(args) + except ConfigError as exc: + print(f"FAILED: {exc}", file=sys.stderr) + return 2 + + req_path = config.resolve(config.requirements_path) + if req_path is None or not req_path.is_file(): print(f"FAILED: requirements register not found at {req_path}", file=sys.stderr) return 2 - register = read_register(req_path) - files = iter_source_files(root) - scan = scan_files(files, root) + register = read_register(config) + files = iter_source_files(config) + scan = scan_files(files, config) system_ids = None - if args.system_spec: - if not args.system_spec.is_file(): - print(f"FAILED: --system-spec not found at {args.system_spec}", - file=sys.stderr) + spec_path = config.resolve(config.system_spec_path) + if spec_path is not None: + if not spec_path.is_file(): + print(f"FAILED: system spec not found at {spec_path}", file=sys.stderr) return 2 - system_ids = parse_system_spec(args.system_spec.read_text(encoding="utf-8")) + system_ids = parse_system_spec(spec_path.read_text(encoding="utf-8"), + config.external_types) - report = build_report(root, register, scan, system_ids) - report.allow_orphans = args.allow_orphans + report = build_report(config, register, scan, system_ids) json_text = json.dumps(report_to_json_dict(report), indent=2, sort_keys=False) markdown_text = generate_markdown(report) - if args.json_out: - args.json_out.parent.mkdir(parents=True, exist_ok=True) - args.json_out.write_text(json_text, encoding="utf-8") - if args.markdown_out: - args.markdown_out.parent.mkdir(parents=True, exist_ok=True) - args.markdown_out.write_text(markdown_text, encoding="utf-8") + for target, text in ((config.resolve(config.json_path), json_text), + (config.resolve(config.matrix_path), markdown_text)): + if target is not None: + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(text, encoding="utf-8") if args.format == "json": print(json_text) @@ -1290,7 +1641,7 @@ def main(argv: Optional[Sequence[str]] = None) -> int: print(markdown_text, end="") return 0 - text, code = format_coverage_report(report, args.min_coverage) + text, code = format_coverage_report(report) print(text, end="") return code diff --git a/scripts/traceability/test_extract_traces.py b/scripts/traceability/test_extract_traces.py index d965e75..f105fd8 100755 --- a/scripts/traceability/test_extract_traces.py +++ b/scripts/traceability/test_extract_traces.py @@ -43,6 +43,27 @@ TAG = "TRA" + "CES:" EXC = "EXCEP" + "TION:" + +# -------------------------------------------------------------------------- +# Test configuration +# +# Every entry point takes a Config since the tool became shared. These helpers +# keep each test stating only the field it varies. +# -------------------------------------------------------------------------- + +def _config(root=".", **kw) -> "et.Config": + """A Config with the extraction repo's shape, overridable per test.""" + fields = dict( + requirement_types=("AR", "DP", "IR", "GR", "VR"), + source_suffixes=frozenset(et.LANGUAGE_SUFFIXES["cpp"] + | et.LANGUAGE_SUFFIXES["python"]), + source_roots=("src", "tests", "scripts", "experiments", "eval"), + root=Path(root), + ) + fields.update(kw) + return et.Config(**fields) + + # -------------------------------------------------------------------------- # Tag parsing # -------------------------------------------------------------------------- @@ -127,11 +148,11 @@ def test_scans_cpp_and_python_but_not_vendored_or_non_source(): with tempfile.TemporaryDirectory() as tmp: root = Path(tmp) _write_tree(root) - files = et.iter_source_files(root) + files = et.iter_source_files(_config(root)) names = sorted(f.name for f in files) assert names == ["gallery.py", "tracker.hpp"], names - scan = et.scan_files(files, root) + scan = et.scan_files(files, _config(root)) traced = sorted({i for t in scan.traces for i in t.requirements}) assert traced == ["AR-012", "AR-013", "GR-001", "SR-002", "SR-005"] @@ -141,7 +162,7 @@ def test_context_is_found_below_a_cpp_tag_and_above_a_python_tag(): with tempfile.TemporaryDirectory() as tmp: root = Path(tmp) _write_tree(root) - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) contexts = {t.file: t.context for t in scan.traces} assert "TrackRegistry" in contexts["src/tracker.hpp"] assert "build_gallery" in contexts["scripts/gallery.py"] @@ -164,7 +185,7 @@ def test_exception_tag_is_captured_with_its_reason(): root = Path(tmp) (root / "src").mkdir(parents=True) (root / "src" / "outlier.cpp").write_text(EXCEPTION_SOURCE, encoding="utf-8") - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) assert len(scan.exceptions) == 1 exc = scan.exceptions[0] assert exc.requirement == "AR-024" @@ -179,11 +200,11 @@ def test_exception_is_never_counted_as_coverage(): root = Path(tmp) (root / "src").mkdir(parents=True) (root / "src" / "outlier.cpp").write_text(EXCEPTION_SOURCE, encoding="utf-8") - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) assert scan.traces == [] register = et.parse_register( "| ID | Requirement | Status |\n|---|---|---|\n" - "| AR-024 | Always the calibrated probability | Planned |\n") + "| AR-024 | Always the calibrated probability | Planned |\n", _config()) cov = et.compute_coverage( [i for t in scan.traces for i in t.requirements], register) assert cov.covered == [] @@ -197,7 +218,7 @@ def test_exception_without_a_reason_is_reported(): (root / "src").mkdir(parents=True) (root / "src" / "bare.cpp").write_text( f"// {EXC} AR-024\n", encoding="utf-8") - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) assert len(scan.exceptions) == 1 assert scan.diagnostics.exceptions_without_reason @@ -210,7 +231,7 @@ def test_mixed_type_group_is_reported(): (root / "src").mkdir(parents=True) (root / "src" / "a.cpp").write_text( f"// {TAG} AR-001, SR-002\n", encoding="utf-8") - scan = et.scan_files(et.iter_source_files(root), root) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) assert scan.diagnostics.mixed_type_groups @@ -225,7 +246,7 @@ def test_counts_a_well_formed_table_row_as_a_defined_requirement(): md = REGISTER_HEADER + ( "| AR-001 | Detect faces in sampled frames | SR-002 | High | Done |\n" "| AR-002 | Minimum face size 66x66 px | SR-002 | High | Planned |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.count("AR") == 2 assert register.count("GR") == 0 assert register.total == 2 @@ -237,7 +258,7 @@ def test_does_not_count_ids_that_appear_only_in_the_traces_to_column(): md = REGISTER_HEADER + ( "| GR-001 | Build gallery from library cast | SR-001, SR-005 | High | Done |\n" "| GR-002 | Incremental merge refresh | PR-003 | High | Done |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.count("GR") == 2 assert register.ids == {"GR-001", "GR-002"} @@ -246,7 +267,7 @@ def test_does_not_count_ids_mentioned_in_prose(): md = ("Some prose explaining that AR-005 relates to GR-001 and VR-003.\n\n" + REGISTER_HEADER + "| AR-005 | Align to 112x112 | SR-002 | High | Done |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.ids == {"AR-005"} @@ -260,7 +281,7 @@ def test_does_not_count_the_verification_plan_table_as_definitions(): + "| ID | Tier | Test asserts | Edge cases to cover |\n|---|---|---|---|\n" + "| AR-001 | T3 | Detector returns plausible boxes | smoke only |\n" + "| AR-099 | T1 | Something not in the register | - |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.ids == {"AR-001"} assert register.total == 1 @@ -271,7 +292,7 @@ def test_deduplicates_an_id_listed_in_two_definition_tables(): + "\n" + REGISTER_HEADER + "| AR-001 | Detect faces | SR-002 | High | Done |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.total == 1 @@ -281,7 +302,7 @@ def test_withdrawn_requirements_leave_the_denominator(): md = REGISTER_HEADER + ( "| AR-001 | Detect faces | SR-002 | High | Done |\n" "| AR-002 | Superseded mechanism | SR-002 | High | Withdrawn |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.ids == {"AR-001"} assert "AR-002" in register.withdrawn @@ -291,8 +312,8 @@ def test_the_denominator_is_live_adding_a_row_lowers_coverage(): # more requirement defined => a lower percentage, mechanically. base = REGISTER_HEADER + "| AR-001 | A | SR-002 | High | Done |\n" grown = base + "| AR-002 | B | SR-002 | High | Planned |\n" - before = et.compute_coverage(["AR-001"], et.parse_register(base)) - after = et.compute_coverage(["AR-001"], et.parse_register(grown)) + before = et.compute_coverage(["AR-001"], et.parse_register(base, _config())) + after = et.compute_coverage(["AR-001"], et.parse_register(grown, _config())) assert before.percent == 100.0 assert after.percent == 50.0 assert after.total == 2 @@ -301,7 +322,7 @@ def test_the_denominator_is_live_adding_a_row_lowers_coverage(): def test_register_captures_the_row_fields_not_just_the_id(): md = REGISTER_HEADER + ( "| AR-012 | Presence follows track extent | **SR-002** | High | Planned |\n") - req = et.parse_register(md).requirements["AR-012"] + req = et.parse_register(md, _config()).requirements["AR-012"] assert req.text == "Presence follows track extent" assert req.traces_to == "**SR-002**" assert req.status == "Planned" @@ -328,7 +349,7 @@ def test_tier_assignment_handles_lists_ranges_and_wildcards(): "| AR-002 … AR-004 | **T2** | replay |\n" "| AR-006 | T1 + T4 | mixed |\n" "| VR-* | Out of CI | studies |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.requirements["AR-001"].tiers == {"T3"} assert register.requirements["AR-003"].tiers == {"T2"} assert register.requirements["AR-006"].tiers == {"T1", "T4"} @@ -337,7 +358,7 @@ def test_tier_assignment_handles_lists_ranges_and_wildcards(): def test_a_range_cannot_invent_a_requirement_the_register_lacks(): md = TIER_REGISTER + "\n" + _tier_table("| AR-001 … AR-050 | T2 | wide |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.total == 12 assert "AR-050" not in register.ids @@ -346,7 +367,7 @@ def test_slash_shorthand_in_the_verification_plan_expands(): md = TIER_REGISTER + "\n" + ( "| ID | Tier | Test asserts | Edge cases |\n|---|---|---|---|\n" "| AR-009/008 | T2 | Cut shifts weighting | cut with same people |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.requirements["AR-008"].tiers == {"T2"} assert register.requirements["AR-009"].tiers == {"T2"} @@ -358,15 +379,15 @@ def test_tiers_from_both_tables_are_unioned_not_overwritten(): md = TIER_REGISTER + "\n" + _tier_table("| AR-006 | T4 | GPU host only |\n") + "\n" + ( "| ID | Tier | Test asserts | Edge cases |\n|---|---|---|---|\n" "| AR-006 | T1 + T4 | GEMM equals reference loop | small input in CI |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.requirements["AR-006"].tiers == {"T1", "T4"} - assert register.requirements["AR-006"].ci_executable + assert register.is_ci_executable("AR-006") def test_a_t4_only_requirement_is_not_ci_executable(): md = TIER_REGISTER + "\n" + _tier_table("| AR-007 | **T4** | GPU only |\n") - register = et.parse_register(md) - assert not register.requirements["AR-007"].ci_executable + register = et.parse_register(md, _config()) + assert not register.is_ci_executable("AR-007") assert register.unexecutable_ids() == {"AR-007"} @@ -379,12 +400,12 @@ def test_a_requirement_tracing_up_to_nothing_is_reported(): "| AR-002 | Parent is a section | §4 | Medium | Planned |\n" "| AR-003 | Serves nothing stated | - | Low | Planned |\n" "| AR-004 | Blank cell | | Low | Planned |\n") - register = et.parse_register(md) + register = et.parse_register(md, _config()) assert register.parentless_ids() == {"AR-003", "AR-004"} def test_a_requirement_with_no_tier_is_unknown_not_unexecutable(): - register = et.parse_register(TIER_REGISTER) + register = et.parse_register(TIER_REGISTER, _config()) assert register.tier_unknown_ids() == register.ids assert register.unexecutable_ids() == set() @@ -402,7 +423,7 @@ COVERAGE_REGISTER = et.parse_register( + "\n" + _tier_table("| AR-001, AR-002 | T2 | replay |\n" "| GR-001 | T1 | bookkeeping |\n" - "| AR-027 | **T4** | GPU host only |\n")) + "| AR-027 | **T4** | GPU host only |\n"), _config()) def test_coverage_is_the_intersection_of_traced_and_defined(): @@ -510,10 +531,30 @@ def _fixture_repo(tmp: str, source: str) -> Path: (root / "src").mkdir(parents=True, exist_ok=True) (root / "docs" / "requirements.md").write_text(FIXTURE_REGISTER, encoding="utf-8") (root / "src" / "pipeline.cpp").write_text(source, encoding="utf-8") + (root / et.CONFIG_FILENAME).write_text( + 'requirement_types = ["AR", "DP", "IR", "GR", "VR"]\n' + 'languages = ["cpp", "python"]\n' + 'source_roots = ["src", "tests", "scripts", "experiments", "eval"]\n', + encoding="utf-8") return root def _run_gate(root: Path, *extra: str): + """Run the CLI against `root`, supplying the per-repo config it now needs. + + The tool no longer guesses a taxonomy: a repo declares its prefixes and + languages, and the gate refuses to run without them rather than reporting a + misleading zero. So a gate test must provide them too — written as a real + `traceability.toml`, which exercises the config-discovery path rather than + bypassing it. + """ + config = root / et.CONFIG_FILENAME + if not config.exists(): + config.write_text( + 'requirement_types = ["AR", "DP", "IR", "GR", "VR"]\n' + 'languages = ["cpp", "python"]\n' + 'source_roots = ["src", "tests", "scripts", "experiments", "eval"]\n', + encoding="utf-8") buffer = io.StringIO() with redirect_stdout(buffer): code = et.main(["--root", str(root), "--format", "coverage", *extra]) @@ -595,11 +636,11 @@ def test_gate_hard_fails_on_an_impossible_ratio(): # reported as a pass. Forced here by handing the reporter a poisoned value. with tempfile.TemporaryDirectory() as tmp: root = _fixture_repo(tmp, f"// {TAG} AR-001\nint main() {{}}\n") - register = et.read_register(root / "docs" / "requirements.md") - scan = et.scan_files(et.iter_source_files(root), root) - report = et.build_report(root, register, scan) + register = et.read_register(_config(requirements_path=str(root / "docs" / "requirements.md"))) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) + report = et.build_report(_config(root, min_coverage=50.0), register, scan) report.coverage.percent = 158.0 - text, code = et.format_coverage_report(report, 50.0) + text, code = et.format_coverage_report(report) assert code == 1 assert "exceeds 100%" in text @@ -610,9 +651,9 @@ def test_the_per_type_breakdown_sums_to_the_headline_figure(): with tempfile.TemporaryDirectory() as tmp: root = _fixture_repo( tmp, f"// {TAG} AR-001, AR-027\nint main() {{}}\n") - register = et.read_register(root / "docs" / "requirements.md") - scan = et.scan_files(et.iter_source_files(root), root) - report = et.build_report(root, register, scan) + register = et.read_register(_config(requirements_path=str(root / "docs" / "requirements.md"))) + scan = et.scan_files(et.iter_source_files(_config(root)), _config(root)) + report = et.build_report(_config(root), register, scan) stats = et.per_type_stats(report) assert sum(c for c, _, _ in stats.values()) == len(report.coverage.covered) assert sum(u for _, u, _ in stats.values()) == len(report.coverage.unexecuted) @@ -637,7 +678,7 @@ def test_json_and_markdown_outputs_are_written_and_consistent(): assert data["coverage"]["covered"] == 1 assert data["coverage"]["percent"] == round(100 / 3, 1) assert data["byType"]["SR"] == ["SR-002"] - assert data["gpuOnlyRequirements"] == ["AR-027"] + assert data["unexecutableRequirements"] == ["AR-027"] assert "AR-001" in md_out.read_text(encoding="utf-8") @@ -662,144 +703,36 @@ def test_system_spec_parsing_enables_orphan_checks_for_pr_and_sr(): # requirements are added. # -------------------------------------------------------------------------- -#: The live-register tests below ran against this repo's own register when the -#: tool lived inside `scene-actor-extraction`. Now that it is shared, this repo -#: is the project home and holds no component register of its own, so they take -#: a path from the environment and skip when it is absent. -#: -#: LIVE_REGISTER=../scene-actor-extraction/docs/requirements.md \ -#: python3 test_extract_traces.py -#: -#: They are kept rather than deleted because they assert something the synthetic -#: fixtures cannot: that a *real* register, with all its formatting accidents, -#: parses at all. -LIVE_REGISTER = os.environ.get("LIVE_REGISTER") +LIVE_REGISTER = os.environ.get('LIVE_REGISTER') def test_the_live_register_parses_and_assigns_tiers(): if not LIVE_REGISTER: - return # skipped: no component register to point at - register = et.read_register(Path(LIVE_REGISTER)) + return # skipped: the project home holds no component register + register = et.read_register(_config(root=str(Path(LIVE_REGISTER).resolve().parent.parent), requirements_path=str(LIVE_REGISTER))) assert register.total > 0 + for req_type in et.LOCAL_TYPES: + assert register.count(req_type) > 0, req_type assert sum(register.count(t) for t in et.LOCAL_TYPES) == register.total - # A tier that no GPU-less host can run must be visible in the parse, not - # merely stated in prose. - assert register.unexecutable_ids() <= register.ids + # The GPU-less CI host must be visible in the parse, not just in prose. + assert "AR-027" in register.unexecutable_ids() + assert register.requirements["AR-027"].tiers == {"T4"} + assert register.requirements["AR-012"].tiers == {"T2"} + assert register.unexecutable_ids() < register.ids def test_the_live_register_yields_a_gate_run_that_cannot_exceed_one_hundred(): if not LIVE_REGISTER: return - register = et.read_register(Path(LIVE_REGISTER)) - root = Path(LIVE_REGISTER).resolve().parent.parent - scan = et.scan_files(et.iter_source_files(root), root) + register = et.read_register(_config(root=str(Path(LIVE_REGISTER).resolve().parent.parent), requirements_path=str(LIVE_REGISTER))) + files = et.iter_source_files(et.REPO_ROOT, _config()) + scan = et.scan_files(files, et.REPO_ROOT, _config()) cov = et.compute_coverage( [i for t in scan.traces for i in t.requirements], register) assert 0.0 <= cov.percent <= 100.0 assert len(cov.covered) <= cov.total -# -------------------------------------------------------------------------- -# Per-repo configuration — the flags that make this tool shareable. -# -# Without them a repo whose prefixes or language differ from the defaults gets -# zero requirements and zero files, which the gate correctly refuses to report -# as coverage. These assert the overrides actually take effect. -# -------------------------------------------------------------------------- - -def _restore_taxonomy(fn): - """Run `fn` with the module defaults restored afterwards.""" - local, suffixes, roots = et.LOCAL_TYPES, set(et.SOURCE_SUFFIXES), et.SCAN_ROOTS - try: - fn() - finally: - et.configure_taxonomy(local) - et.configure_suffixes(suffixes) - et.configure_scan_roots(roots) - - -def test_types_override_changes_which_prefixes_are_counted(): - def body(): - register_md = ( - "| ID | Requirement | Traces to | Priority | Status |\n" - "|---|---|---|---|---|\n" - "| UR-001 | A user requirement | SR-001 | High | Done |\n" - "| DR-001 | A dev requirement | PR-004 | High | Done |\n" - ) - with tempfile.TemporaryDirectory() as tmp: - path = Path(tmp) / "requirements.md" - path.write_text(register_md, encoding="utf-8") - - # Under the defaults these prefixes are unknown, so nothing counts. - et.configure_taxonomy(("AR", "DP")) - assert et.read_register(path).total == 0 - - et.configure_taxonomy(("UR", "DR")) - register = et.read_register(path) - assert register.total == 2 - assert register.count("UR") == 1 - assert register.count("DR") == 1 - - _restore_taxonomy(body) - - -def test_suffix_override_changes_which_files_are_scanned(): - def body(): - with tempfile.TemporaryDirectory() as tmp: - root = Path(tmp) - (root / "src").mkdir() - (root / "src" / "lib.rs").write_text( - "/// TRACES: UR-001 | SR-004\npub fn f() {}\n", encoding="utf-8") - - # Default suffixes are C++/Python, so a Rust tree scans as empty — - # the failure mode this override exists to fix. - assert et.iter_source_files(root) == [] - - et.configure_suffixes([".rs"]) - files = et.iter_source_files(root) - assert len(files) == 1 - scan = et.scan_files(files, root) - assert [i for t in scan.traces for i in t.requirements] == [ - "UR-001", "SR-004"] - - _restore_taxonomy(body) - - -def test_suffix_override_accepts_extensions_with_or_without_a_dot(): - def body(): - et.configure_suffixes(["rs", ".cs"]) - assert et.SOURCE_SUFFIXES == {".rs", ".cs"} - - _restore_taxonomy(body) - - -def test_scan_root_override_limits_the_walk(): - def body(): - with tempfile.TemporaryDirectory() as tmp: - root = Path(tmp) - # Not "vendor" — that is in EXCLUDED_DIR_NAMES and would be - # filtered whatever the scan roots say, testing the wrong thing. - for d in ("src", "eval"): - (root / d).mkdir() - (root / d / "f.py").write_text("# TRACES: AR-001\n", encoding="utf-8") - - et.configure_scan_roots(["src"]) - assert len(et.iter_source_files(root)) == 1 - - et.configure_scan_roots(["src", "eval"]) - assert len(et.iter_source_files(root)) == 2 - - _restore_taxonomy(body) - - -def test_defaults_are_unchanged_by_the_overrides_existing(): - # The extraction pipeline must keep working with no flags at all, or moving - # the tool here would have broken the repo it came from. - assert et.LOCAL_TYPES == ("AR", "DP", "IR", "GR", "VR") - assert ".py" in et.SOURCE_SUFFIXES and ".cpp" in et.SOURCE_SUFFIXES - assert ".rs" not in et.SOURCE_SUFFIXES - - # -------------------------------------------------------------------------- def _main() -> int: diff --git a/scripts/traceability/traceability-gate.sh b/scripts/traceability/traceability-gate.sh index a41dbb8..588848b 100755 --- a/scripts/traceability/traceability-gate.sh +++ b/scripts/traceability/traceability-gate.sh @@ -1,89 +1,62 @@ #!/bin/sh # -# Requirement traceability gate. Run locally exactly as CI runs it: +# Requirement traceability gate. Run locally exactly as CI runs it, from the +# component repo root: # # scripts/traceability/traceability-gate.sh # -# Writes traces-report.json and docs/traceability.md, prints the coverage -# report, and exits non-zero when the gate fails. +# Writes the JSON report and the markdown matrix, prints the coverage report, +# and exits non-zero when the gate fails. # -# Environment: -# MIN_COVERAGE minimum overall coverage percent (default 0 - see below) -# ALLOW_ORPHANS set to 1 to report orphan tags without failing -# TRACES_JSON JSON report path (default traces-report.json) -# TRACES_MD markdown matrix path (default docs/traceability.md) -# REPO_ROOT repository to scan. Defaults to two levels above this -# script, which is correct when the tooling lives in the repo -# it checks. **When vendored as a submodule that default is -# the submodule itself**, so a consuming repo must set this — -# its wrapper does. -# TYPES comma-separated requirement prefixes (e.g. UR,DR). Defaults -# to the extraction set; a repo whose prefixes differ parses -# to zero requirements without this. -# SUFFIXES comma-separated file extensions (e.g. .rs). Defaults to the -# C++/Python set; a repo whose language differs scans zero -# files without this. -# SCAN_ROOTS comma-separated directories to walk, relative to REPO_ROOT. -# SYSTEM_SPEC optional path to the system SPEC.md, which defines the PR/SR -# IDs; when given, PR/SR orphans are reported too. It lives in -# the project home (jray-project) — when this tooling is -# vendored from there, it is a sibling of this script. +# This script is shared by every JRay component, so it knows nothing about any +# one repo. All repo-specific settings - requirement ID prefixes, source +# suffixes, scan roots, register path, thresholds - live in `traceability.toml` +# at the component repo root. Run # -# Threshold policy lives here and nowhere else. It is deliberately NOT -# duplicated into the workflow YAML: a threshold written in two places is a -# threshold that will disagree with itself. +# scripts/traceability/extract_traces.py --print-example-config # -# MIN_COVERAGE defaults to 0 because almost nothing is tagged yet - tags are -# added as the pipeline is built, so a low number today is accurate rather than -# alarming. A zero threshold does NOT mean the gate cannot fail: orphan tags, -# a >100% ratio, a register that parses to nothing, and an empty source scan -# are all hard failures from day one. Raise MIN_COVERAGE as tags land; treat -# every raise as a ratchet, never a reset. +# for the annotated schema. A repo whose config is wrong parses zero +# requirements or scans zero files, and the gate refuses to report rather than +# printing a misleading 0%. +# +# Environment (all optional; each overrides the config file): +# TRACES_CONFIG path to traceability.toml +# TRACES_ROOT repo root (default: nearest dir containing traceability.toml) +# MIN_COVERAGE minimum overall coverage percent +# ALLOW_ORPHANS 1 to report orphan tags without failing +# TRACES_JSON JSON report path +# TRACES_MD markdown matrix path +# SYSTEM_SPEC SPEC.md defining PR/SR; enables PR/SR orphan checking +# PYTHON interpreter (default: python3) +# +# Threshold policy belongs in traceability.toml, not here and not in the +# workflow YAML: a threshold written in two places is a threshold that will +# disagree with itself. # # POSIX sh, no bashisms, no jq - the extractor does its own arithmetic and -# printing so CI needs nothing beyond python3. +# printing, so CI needs nothing beyond python3. set -eu SCRIPT_DIR=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) -REPO_ROOT="${REPO_ROOT:-$(CDPATH= cd -- "$SCRIPT_DIR/../.." && pwd)}" - -MIN_COVERAGE="${MIN_COVERAGE:-0}" -TRACES_JSON="${TRACES_JSON:-$REPO_ROOT/traces-report.json}" -TRACES_MD="${TRACES_MD:-$REPO_ROOT/docs/traceability.md}" PYTHON="${PYTHON:-python3}" command -v "$PYTHON" >/dev/null 2>&1 || { - echo "FAILED: $PYTHON not found. The traceability gate needs Python 3.9+" >&2 + echo "FAILED: $PYTHON not found. The traceability gate needs Python 3.9+," >&2 + echo " or 3.11+ to read traceability.toml." >&2 exit 2 } -set -- \ - --root "$REPO_ROOT" \ - --requirements "${REQUIREMENTS:-$REPO_ROOT/docs/requirements.md}" \ - --format coverage \ - --json-out "$TRACES_JSON" \ - --markdown-out "$TRACES_MD" \ - --min-coverage "$MIN_COVERAGE" +set -- --format coverage -if [ "${ALLOW_ORPHANS:-0}" = "1" ]; then - set -- "$@" --allow-orphans -fi - -if [ -n "${SYSTEM_SPEC:-}" ]; then - set -- "$@" --system-spec "$SYSTEM_SPEC" -fi - -if [ -n "${TYPES:-}" ]; then - set -- "$@" --types "$TYPES" -fi - -if [ -n "${SUFFIXES:-}" ]; then - set -- "$@" --suffixes "$SUFFIXES" -fi - -if [ -n "${SCAN_ROOTS:-}" ]; then - set -- "$@" --scan-roots "$SCAN_ROOTS" -fi +# Explicit `if` rather than `[ ... ] && ...`, because a trailing false test in +# an && list exits under `set -e` in some POSIX shells. +if [ -n "${TRACES_CONFIG:-}" ]; then set -- "$@" --config "$TRACES_CONFIG"; fi +if [ -n "${TRACES_ROOT:-}" ]; then set -- "$@" --root "$TRACES_ROOT"; fi +if [ -n "${MIN_COVERAGE:-}" ]; then set -- "$@" --min-coverage "$MIN_COVERAGE"; fi +if [ -n "${TRACES_JSON:-}" ]; then set -- "$@" --json-out "$TRACES_JSON"; fi +if [ -n "${TRACES_MD:-}" ]; then set -- "$@" --markdown-out "$TRACES_MD"; fi +if [ -n "${SYSTEM_SPEC:-}" ]; then set -- "$@" --system-spec "$SYSTEM_SPEC"; fi +if [ "${ALLOW_ORPHANS:-0}" = "1" ]; then set -- "$@" --allow-orphans; fi exec "$PYTHON" "$SCRIPT_DIR/extract_traces.py" "$@"