//! Requirements traceability for DarkRoom. //! //! Extracts `TRACES:` tags from source, counts what `requirements.md` actually //! defines, and reports coverage as the intersection of the two. //! //! # Why the arithmetic is written this way //! //! Adapted from the JellyTau tooling, including the bug it was repaired for. //! That gate divided a traced count by *frozen literal* denominators; the //! requirements file grew past them, and it reported **158% coverage**. A gate //! reporting over 100% cannot fail its own threshold, so it silently stopped //! being a gate at all. //! //! Two rules follow, and both are enforced by tests here: //! //! 1. **Denominators are parsed from `requirements.md` at run time.** Never //! hardcoded, never cached. //! 2. **Coverage is `|traced ∩ defined| / |defined|`.** Using the raw traced //! count as the numerator is precisely what lets a ratio exceed 100%, since //! a tag naming a deleted requirement would count as covered. Such tags are //! reported as orphans instead. use std::collections::{BTreeMap, BTreeSet}; use std::path::{Path, PathBuf}; use serde::Serialize; /// Requirement ID prefixes that participate in coverage. /// /// Test identifiers (UT, IT) are a separate taxonomy: they are *evidence* for /// requirements, not requirements themselves. Counting them would inflate both /// numerator and denominator, and flagging them as orphans would bury real /// typos in noise. pub const REQUIREMENT_TYPES: &[&str] = &["FR", "NFR", "R"]; /// Prefixes recognised in tags but deliberately excluded from coverage. /// /// `D` are decisions, `S` spikes, `M` milestone items, `AC` acceptance /// criteria, `UT`/`IT` tests. All are legitimate things to tag against, and /// none is a requirement — counting them would inflate the denominator by 25 /// and make coverage look worse than it is, which is the same class of defect /// as JellyTau's inflated ratio, just in the other direction. pub const NON_REQUIREMENT_TYPES: &[&str] = &["UT", "IT", "AC", "M", "S", "D"]; /// One `TRACES:` tag found in source. #[derive(Debug, Clone, Serialize, PartialEq, Eq)] pub struct TraceEntry { pub file: String, pub line: usize, pub context: String, pub requirements: Vec, } /// What `requirements.md` defines — the coverage denominators. #[derive(Debug, Clone, Default)] pub struct DefinedRequirements { pub ids: BTreeSet, pub by_type: BTreeMap, } impl DefinedRequirements { pub fn total(&self) -> usize { self.ids.len() } } /// The coverage result. #[derive(Debug, Clone, Serialize, PartialEq)] pub struct Coverage { pub covered: usize, pub total: usize, pub percent: f64, /// Tagged in source but absent from `requirements.md` — a typo, or a /// requirement that was renumbered or deleted. Never counted as covered. pub orphaned: Vec, /// Defined but never tagged anywhere. pub untraced: Vec, } /// Parse the requirement IDs a markdown document *defines*. /// /// A requirement is defined by a bolded heading-style declaration /// (`**FR-CAT-1 — …**`) or as the leading cell of a table row (`| R1 | … |`). /// Both forms appear in DarkRoom's requirements.md. /// /// Deliberately *not* a bare scan for anything matching the ID shape: that /// counts cross-references in prose and in "relates to" columns as /// definitions, inflating the denominator. IDs are deduplicated because a /// requirement may legitimately appear in both a definition and a summary /// table. pub fn parse_defined_requirements(markdown: &str) -> DefinedRequirements { let mut ids = BTreeSet::new(); for line in markdown.lines() { let trimmed = line.trim_start(); // Form 1: a bolded definition, e.g. `**FR-CAT-1 — Scan.**` if let Some(rest) = trimmed.strip_prefix("**") { if let Some(id) = leading_id(rest) { ids.insert(id); continue; } } // Form 2: leading table cell, e.g. `| **R1** | … |` or `| R1 | … |` if let Some(rest) = trimmed.strip_prefix('|') { let cell = rest.trim().trim_start_matches("**"); if let Some(id) = leading_id(cell) { ids.insert(id); } } } // Only requirement types enter the register. Decisions, spikes, milestone // items and test ids are all taggable, but none is a requirement, and // counting them would inflate the denominator. let ids: BTreeSet = ids.into_iter().filter(|id| is_requirement(id)).collect(); let mut by_type: BTreeMap = BTreeMap::new(); for id in &ids { *by_type.entry(type_of(id).to_string()).or_insert(0) += 1; } DefinedRequirements { ids, by_type } } /// Extract a requirement ID anchored at the start of `s`. /// /// Accepts `FR-CAT-1`, `NFR-P13`, `R1`, `FR-DEV-3a` — DarkRoom uses /// alphanumeric segments and an optional trailing letter, not the fixed /// three-digit form JellyTau assumed. fn leading_id(s: &str) -> Option { let bytes = s.as_bytes(); if bytes.is_empty() || !bytes[0].is_ascii_uppercase() { return None; } let mut end = 0; let mut seen_digit = false; for (i, c) in s.char_indices() { match c { 'A'..='Z' | '0'..='9' | '-' => { if c.is_ascii_digit() { seen_digit = true; } end = i + c.len_utf8(); } 'a'..='z' if seen_digit => { // Trailing variant letter, e.g. FR-DEV-3a. end = i + c.len_utf8(); } _ => break, } } if end == 0 || !seen_digit { return None; } let id = s[..end].trim_end_matches('-').to_string(); // Must have a recognised prefix, or arbitrary capitalised words match. let ty = type_of(&id); if REQUIREMENT_TYPES.contains(&ty) || NON_REQUIREMENT_TYPES.contains(&ty) { Some(id) } else { None } } /// The type prefix of an ID: `FR-CAT-1` → `FR`, `R1` → `R`. pub fn type_of(id: &str) -> &str { let end = id .find(|c: char| !c.is_ascii_uppercase()) .unwrap_or(id.len()); &id[..end] } /// Whether an ID participates in coverage. pub fn is_requirement(id: &str) -> bool { REQUIREMENT_TYPES.contains(&type_of(id)) } /// Extract every `TRACES:` tag from a source file's text. pub fn extract_from_text(text: &str, path: &str) -> Vec { let lines: Vec<&str> = text.lines().collect(); let mut out = Vec::new(); for (idx, line) in lines.iter().enumerate() { let Some(pos) = line.find("TRACES:") else { continue; }; let tail = &line[pos + "TRACES:".len()..]; let ids = parse_ids(tail); if ids.is_empty() { continue; } out.push(TraceEntry { file: path.to_string(), line: idx + 1, context: find_context(&lines, idx), requirements: ids, }); } out } /// Parse the ID list from a tag body: `FR-CAT-1, FR-CAT-2 | NFR-P1`. /// /// The pipe groups types for readability; both separators are treated alike. fn parse_ids(s: &str) -> Vec { s.split(['|', ',']) .filter_map(|part| leading_id(part.trim())) .collect() } /// The nearest preceding declaration, for the report's context column. fn find_context(lines: &[&str], from: usize) -> String { const MARKERS: &[&str] = &[ "pub fn ", "fn ", "pub struct ", "struct ", "pub enum ", "enum ", "impl ", "pub trait ", "trait ", "#[test]", "component ", "export ", ]; // Look forward first — a doc comment precedes what it documents. for line in lines.iter().skip(from + 1).take(6) { if MARKERS.iter().any(|m| line.contains(m)) { return trim_body(line); } } // Then backward, for tags placed inside a body. for i in (from.saturating_sub(6)..from).rev() { if MARKERS.iter().any(|m| lines[i].contains(m)) { return lines[i].trim().trim_end_matches('{').trim().to_string(); } } "—".to_string() } /// Strip a trailing body opener so the context reads as a signature. /// /// Handles both `fn f() {` and `fn f() {}` — the latter is why this is a /// helper rather than a single `trim_end_matches('{')`. fn trim_body(line: &str) -> String { line.trim() .trim_end_matches("{}") .trim_end() .trim_end_matches('{') .trim_end() .to_string() } /// Compute coverage as the intersection of traced and defined IDs. /// /// This signature is the fix for the 158% bug: `defined` is required, so there /// is nowhere for a frozen denominator to hide. /// TRACES: NFR-OPS-1 pub fn compute_coverage(traced: &BTreeSet, defined: &DefinedRequirements) -> Coverage { let traced_reqs: BTreeSet<&String> = traced.iter().filter(|id| is_requirement(id)).collect(); let covered: Vec<&String> = traced_reqs .iter() .filter(|id| defined.ids.contains(**id)) .copied() .collect(); let orphaned: Vec = traced_reqs .iter() .filter(|id| !defined.ids.contains(**id)) .map(|s| s.to_string()) .collect(); let untraced: Vec = defined .ids .iter() .filter(|id| !traced.contains(*id)) .cloned() .collect(); let total = defined.total(); Coverage { covered: covered.len(), total, percent: if total == 0 { 0.0 } else { (covered.len() as f64 / total as f64) * 100.0 }, orphaned, untraced, } } /// Source file extensions scanned for tags. /// /// `.yaml` is here because a develop operation is now declared rather than /// written: `core/dr-pipeline/ops/.yaml` is the whole node, and the Rust /// implementing it is generated into `OUT_DIR`, which is not scanned and could /// not be linked to from the report if it were. Without this a node would have /// nowhere to record the requirement it satisfies. pub const SOURCE_SUFFIXES: &[&str] = &[".rs", ".slint", ".wgsl", ".yaml"]; /// Directories never scanned. const EXCLUDED: &[&str] = &["target", "target-android", ".git", "node_modules", "temp"]; /// Walk `roots` under `base`, returning every source file. pub fn collect_sources(base: &Path, roots: &[&str]) -> Vec { let mut out = Vec::new(); for root in roots { let dir = base.join(root); if dir.exists() { walk(&dir, &mut out); } } out.sort(); out } fn walk(dir: &Path, out: &mut Vec) { let Ok(entries) = std::fs::read_dir(dir) else { return; }; for entry in entries.flatten() { let path = entry.path(); let name = entry.file_name(); let name = name.to_string_lossy(); if EXCLUDED.iter().any(|e| *e == name) { continue; } if path.is_dir() { walk(&path, out); } else if SOURCE_SUFFIXES.iter().any(|s| name.ends_with(s)) { out.push(path); } } } #[cfg(test)] mod tests { use super::*; // ---- the 158% regression ------------------------------------------ #[test] fn coverage_never_exceeds_one_hundred_percent() { // The JellyTau failure, reproduced: more tags than the register // defines. Every extra tag must land in `orphaned`, never inflate // the numerator. let defined = parse_defined_requirements("**FR-CAT-1 — Scan.**\n"); let traced: BTreeSet = ["FR-CAT-1", "FR-CAT-2", "FR-CAT-3", "FR-CAT-4"] .iter() .map(|s| s.to_string()) .collect(); let cov = compute_coverage(&traced, &defined); assert_eq!(cov.covered, 1); assert_eq!(cov.total, 1); assert_eq!(cov.percent, 100.0); assert!(cov.percent <= 100.0, "coverage must never exceed 100%"); assert_eq!(cov.orphaned, vec!["FR-CAT-2", "FR-CAT-3", "FR-CAT-4"]); } #[test] fn orphan_tags_are_reported_not_counted() { let defined = parse_defined_requirements("**FR-CAT-1 — Scan.**\n"); let traced: BTreeSet = ["FR-CAT-1", "FR-TYPO-9"] .iter() .map(|s| s.to_string()) .collect(); let cov = compute_coverage(&traced, &defined); assert_eq!(cov.covered, 1); assert_eq!(cov.orphaned, vec!["FR-TYPO-9"]); } #[test] fn empty_register_is_zero_not_a_division_by_zero() { let defined = parse_defined_requirements(""); let traced = BTreeSet::new(); let cov = compute_coverage(&traced, &defined); assert_eq!(cov.percent, 0.0); assert_eq!(cov.total, 0); } // ---- denominator parsing ------------------------------------------ #[test] fn definitions_come_from_declarations_not_cross_references() { // Only the first line defines FR-CAT-1. The prose reference and the // "relates to" cell must not each count as another definition. let md = "\ **FR-CAT-1 — Scan.** The app shall scan roots. This is discussed in FR-CAT-1 and also FR-CAT-1 again. | Spike | Answers | Relates to | |---|---|---| | S1 | whether it works | FR-CAT-1 | "; let defined = parse_defined_requirements(md); // FR-CAT-1 once, despite three mentions. S1 is a spike, not a // requirement, so it does not enter the register at all. assert_eq!(defined.total(), 1); assert!(defined.ids.contains("FR-CAT-1")); } #[test] fn decisions_and_spikes_are_not_requirements() { // D and S are taggable but are not requirements; counting them would // inflate the denominator and understate real coverage. let md = "\ **FR-CAT-1 — Scan.** | D1 | Language | Rust | | S1 | Zero-copy spike | proves ARCH 6.1 | "; let defined = parse_defined_requirements(md); assert_eq!(defined.total(), 1, "only FR-CAT-1 is a requirement"); assert!(defined.ids.contains("FR-CAT-1")); assert!(!defined.ids.contains("D1")); assert!(!defined.ids.contains("S1")); } #[test] fn tagging_a_decision_neither_covers_nor_orphans() { let defined = parse_defined_requirements("**FR-CAT-1 — Scan.**\n"); let traced: BTreeSet = ["FR-CAT-1", "D1"].iter().map(|s| s.to_string()).collect(); let cov = compute_coverage(&traced, &defined); assert_eq!(cov.covered, 1); assert!(cov.orphaned.is_empty(), "D1 is a valid tag, not an orphan"); } #[test] fn table_row_ids_are_definitions() { let md = "| **R1** | Cross-platform | Same core on both |\n\ | R2 | Efficient display | 60fps |\n"; let defined = parse_defined_requirements(md); assert!(defined.ids.contains("R1")); assert!(defined.ids.contains("R2")); } #[test] fn darkroom_id_shapes_parse() { // Not JellyTau's fixed three-digit form. assert_eq!(leading_id("FR-CAT-1 — Scan").as_deref(), Some("FR-CAT-1")); assert_eq!( leading_id("NFR-P13 — Next image").as_deref(), Some("NFR-P13") ); assert_eq!( leading_id("FR-DEV-3a — Descriptors").as_deref(), Some("FR-DEV-3a") ); assert_eq!(leading_id("R1 | Cross-platform").as_deref(), Some("R1")); } #[test] fn prose_is_not_an_id() { assert_eq!(leading_id("The app shall scan"), None); assert_eq!(leading_id("GPU results never"), None); // A recognised prefix with no digits is not an ID either. assert_eq!(leading_id("FR without a number"), None); } // ---- tag extraction ----------------------------------------------- #[test] fn extracts_tags_with_both_separators() { let src = "\ /// TRACES: FR-CAT-1, FR-CAT-2 | NFR-P1 pub fn scan() {} "; let traces = extract_from_text(src, "x.rs"); assert_eq!(traces.len(), 1); assert_eq!( traces[0].requirements, vec!["FR-CAT-1", "FR-CAT-2", "NFR-P1"] ); assert_eq!(traces[0].line, 1); assert_eq!(traces[0].context, "pub fn scan()"); } #[test] fn context_looks_forward_then_backward() { // Doc comments precede their item. let fwd = extract_from_text("// TRACES: R1\npub struct Catalog;", "x.rs"); assert_eq!(fwd[0].context, "pub struct Catalog;"); // A tag inside a body refers to the enclosing item. let back = extract_from_text("pub fn render() {\n // TRACES: R1\n}", "x.rs"); assert_eq!(back[0].context, "pub fn render()"); } #[test] fn a_tag_with_no_ids_is_ignored() { let traces = extract_from_text("// TRACES: see the design doc\n", "x.rs"); assert!(traces.is_empty()); } #[test] fn test_ids_are_extracted_but_not_counted_as_requirements() { let traces = extract_from_text("// TRACES: FR-CAT-1 | UT-001\nfn f(){}", "x.rs"); assert_eq!(traces[0].requirements, vec!["FR-CAT-1", "UT-001"]); // UT is a separate taxonomy: evidence, not a requirement. assert!(is_requirement("FR-CAT-1")); assert!(!is_requirement("UT-001")); // So it neither covers nor orphans. let defined = parse_defined_requirements("**FR-CAT-1 — Scan.**\n"); let traced: BTreeSet = ["FR-CAT-1", "UT-001"] .iter() .map(|s| s.to_string()) .collect(); let cov = compute_coverage(&traced, &defined); assert_eq!(cov.covered, 1); assert!( cov.orphaned.is_empty(), "UT must not be reported as orphaned" ); } #[test] fn type_prefixes_split_correctly() { assert_eq!(type_of("FR-CAT-1"), "FR"); assert_eq!(type_of("NFR-P13"), "NFR"); assert_eq!(type_of("R1"), "R"); assert_eq!(type_of("UT-001"), "UT"); } #[test] fn untraced_requirements_are_listed() { let defined = parse_defined_requirements("**FR-A-1 — One.**\n**FR-B-2 — Two.**\n"); let traced: BTreeSet = ["FR-A-1"].iter().map(|s| s.to_string()).collect(); let cov = compute_coverage(&traced, &defined); assert_eq!(cov.untraced, vec!["FR-B-2"]); } }