1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
//! Guard: no tracked doc under `docs/` states a stale hard-coded "N
//! analyses" count that disagrees with the live `AnalysisName` registry.
//!
//! Reference docs describe the registry by shape ("the full analysis
//! registry, enumerated by `AnalysisName::all()`") rather than by a pinned
//! number, precisely because the registry grows. A hard-coded count left
//! behind after that growth is the same class of rot the project already
//! guards against for stale test counts and version numbers — this test
//! closes the same gap for analysis counts, so it can't silently regress.
//!
//! Scope: every `.md` file under `docs/`, excluding documents that
//! legitimately narrate a point-in-time count rather than a claim about
//! the current registry: `docs/superpowers/` (dated plans and specs),
//! `docs/reports/` (dated audit reports that quote whatever counts a doc
//! held at audit time — including stale ones, as findings to reconcile),
//! `docs/RELEASING.md` (its "how we got here" section narrates past
//! release milestones, changelog-style), and `docs/maximum-feature-plan.md`
//! (marked fully shipped — a frozen plan that predates the
//! `docs/superpowers/` convention).
use std::path::{Path, PathBuf};
use codelore_lib::analysis::AnalysisName;
/// Path prefixes excluded from the scan (see module doc for rationale).
const EXCLUDED_PREFIXES: &[&str] = &["docs/superpowers/", "docs/reports/"];
/// Exact file paths excluded from the scan (see module doc for rationale).
const EXCLUDED_FILES: &[&str] = &["docs/RELEASING.md", "docs/maximum-feature-plan.md"];
/// `CARGO_MANIFEST_DIR` is `<root>/crates/codelore-lib`; two levels up is
/// the workspace root. Embedded at compile time, so it resolves under CI
/// too.
fn workspace_root() -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR"))
.ancestors()
.nth(2)
.expect("workspace root two levels above crates/codelore-lib")
.to_path_buf()
}
fn collect_md_files(dir: &Path, out: &mut Vec<PathBuf>) {
let Ok(entries) = std::fs::read_dir(dir) else {
return; // a missing root is fine — just nothing to scan
};
for entry in entries.flatten() {
let path = entry.path();
if path.is_dir() {
collect_md_files(&path, out);
} else if path.extension().and_then(|e| e.to_str()) == Some("md") {
out.push(path);
}
}
}
fn is_excluded(rel: &Path) -> bool {
let rel_str = rel.to_string_lossy().replace('\\', "/");
EXCLUDED_PREFIXES.iter().any(|p| rel_str.starts_with(p))
|| EXCLUDED_FILES.iter().any(|f| rel_str == *f)
}
/// If `line` contains a hard-coded "<digits> analyses" count that
/// disagrees with `real_count`, returns a description for the violation
/// report. Matches a run of ASCII digits, optional spaces, then the word
/// "analyses" (case-insensitive) at a word boundary — this catches both
/// plain prose ("the 54 analyses") and Markdown emphasis
/// ("**54 analyses**"), since the `**` markers sit outside the matched
/// span.
fn stale_count_in_line(line: &str, real_count: usize) -> Option<String> {
const WORD: &str = "analyses";
let bytes = line.as_bytes();
let mut i = 0;
while i < bytes.len() {
if !bytes[i].is_ascii_digit() {
i += 1;
continue;
}
let start = i;
while i < bytes.len() && bytes[i].is_ascii_digit() {
i += 1;
}
let digits = &line[start..i];
let mut j = i;
while j < bytes.len() && bytes[j] == b' ' {
j += 1;
}
let rest = &line[j..];
// `.get()`, not byte-index slicing: `rest` may put the WORD.len()
// cut point inside a multi-byte UTF-8 character (docs use math
// symbols like 'σ'), which would panic on a raw `&rest[..N]`.
let matches_word = rest
.get(..WORD.len())
.is_some_and(|candidate| candidate.eq_ignore_ascii_case(WORD));
// Word boundary: the char after "analyses" must not be alphanumeric
// (so "analyses" matches but "analysesx" does not).
let boundary_ok = rest
.as_bytes()
.get(WORD.len())
.is_none_or(|b| !b.is_ascii_alphanumeric());
if matches_word && boundary_ok {
match digits.parse::<usize>() {
Ok(n) if n != real_count => {
return Some(format!(
"\"{digits} analyses\" (registry currently has {real_count})"
));
}
_ => {}
}
}
}
None
}
#[test]
fn no_stale_analysis_count_in_docs() {
let real_count = AnalysisName::all().len();
let root = workspace_root();
let mut files = Vec::new();
collect_md_files(&root.join("docs"), &mut files);
assert!(
!files.is_empty(),
"scanned zero docs/*.md files — doc-path resolution is broken"
);
let mut violations = Vec::new();
for file in &files {
let rel = file.strip_prefix(&root).unwrap_or(file);
if is_excluded(rel) {
continue;
}
let text = std::fs::read_to_string(file).expect("read doc file");
for (line_idx, line) in text.lines().enumerate() {
if let Some(found) = stale_count_in_line(line, real_count) {
violations.push(format!("{}:{}: {found}", rel.display(), line_idx + 1));
}
}
}
assert!(
violations.is_empty(),
"found {} stale hard-coded analysis count(s) in docs. Describe the registry by \
shape instead (e.g. \"the full analysis registry, enumerated by \
`AnalysisName::all()`\") so the doc can't drift again:\n{}",
violations.len(),
violations.join("\n"),
);
}