1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
//! Guard: no tracked doc under `docs/` states a stale hard-coded "N
//! analyses" count that disagrees with the live `AnalysisName` registry.
//!
//! Reference docs describe the registry by shape ("the full analysis
//! registry, enumerated by `AnalysisName::all()`") rather than by a pinned
//! number, precisely because the registry grows. A hard-coded count left
//! behind after that growth is the same class of rot the project already
//! guards against for stale test counts and version numbers — this test
//! closes the same gap for analysis counts, so it can't silently regress.
//!
//! Scope: every `.md` file under `docs/`, excluding documents that
//! legitimately narrate a point-in-time count rather than a claim about
//! the current registry: `docs/superpowers/` (dated plans and specs),
//! `docs/reports/` (dated audit reports that quote whatever counts a doc
//! held at audit time — including stale ones, as findings to reconcile),
//! `docs/RELEASING.md` (its "how we got here" section narrates past
//! release milestones, changelog-style), and `docs/maximum-feature-plan.md`
//! (marked fully shipped — a frozen plan that predates the
//! `docs/superpowers/` convention).
use std::path::{Path, PathBuf};
use codelore_lib::analysis::AnalysisName;
/// Path prefixes excluded from the scan (see module doc for rationale).
const EXCLUDED_PREFIXES: &[&str] = &["docs/superpowers/", "docs/reports/"];
/// Exact file paths excluded from the scan (see module doc for rationale).
const EXCLUDED_FILES: &[&str] = &["docs/RELEASING.md", "docs/maximum-feature-plan.md"];
/// `CARGO_MANIFEST_DIR` is `<root>/crates/codelore-lib`; two levels up is
/// the workspace root. Embedded at compile time, so it resolves under CI
/// too.
fn workspace_root() -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR"))
.ancestors()
.nth(2)
.expect("workspace root two levels above crates/codelore-lib")
.to_path_buf()
}
fn collect_md_files(dir: &Path, out: &mut Vec<PathBuf>) {
let Ok(entries) = std::fs::read_dir(dir) else {
return; // a missing root is fine — just nothing to scan
};
for entry in entries.flatten() {
let path = entry.path();
if path.is_dir() {
collect_md_files(&path, out);
} else if path.extension().and_then(|e| e.to_str()) == Some("md") {
out.push(path);
}
}
}
fn is_excluded(rel: &Path) -> bool {
let rel_str = rel.to_string_lossy().replace('\\', "/");
EXCLUDED_PREFIXES.iter().any(|p| rel_str.starts_with(p))
|| EXCLUDED_FILES.iter().any(|f| rel_str == *f)
}
/// Byte offset just past a single `adjective ` token starting at `from`,
/// or `from` unchanged when what follows is not a plain word followed by a
/// space. Lets the scan see the noun through one qualifier.
fn skip_one_word(line: &str, from: usize) -> usize {
let bytes = line.as_bytes();
let mut k = from;
while k < bytes.len() && (bytes[k].is_ascii_alphabetic() || bytes[k] == b'-') {
k += 1;
}
if k == from || bytes.get(k) != Some(&b' ') {
return from;
}
while k < bytes.len() && bytes[k] == b' ' {
k += 1;
}
k
}
/// If `line` contains a hard-coded "<digits> analyses" count that
/// disagrees with `real_count`, returns a description for the violation
/// report. Matches a run of ASCII digits, optional spaces, an optional
/// single qualifier, then the word "analyses" (case-insensitive) at a word
/// boundary — this catches plain prose ("the 54 analyses"), Markdown
/// emphasis ("**54 analyses**") since the `**` markers sit outside the
/// matched span, and the qualified form the architecture diagrams use
/// ("54 behavioral analyses"), which is how a stale count previously
/// survived a sweep that only corrected the number.
fn stale_count_in_line(line: &str, real_count: usize) -> Option<String> {
const WORD: &str = "analyses";
let bytes = line.as_bytes();
let mut i = 0;
while i < bytes.len() {
if !bytes[i].is_ascii_digit() {
i += 1;
continue;
}
let start = i;
while i < bytes.len() && bytes[i].is_ascii_digit() {
i += 1;
}
let digits = &line[start..i];
let mut j = i;
while j < bytes.len() && bytes[j] == b' ' {
j += 1;
}
// Probe the noun directly, then again past one qualifier, so both
// "57 analyses" and "57 behavioral analyses" are seen.
for probe in [j, skip_one_word(line, j)] {
let rest = &line[probe..];
// `.get()`, not byte-index slicing: `rest` may put the WORD.len()
// cut point inside a multi-byte UTF-8 character (docs use math
// symbols like 'σ'), which would panic on a raw `&rest[..N]`.
let matches_word = rest
.get(..WORD.len())
.is_some_and(|candidate| candidate.eq_ignore_ascii_case(WORD));
// Word boundary: the char after "analyses" must not be alphanumeric
// (so "analyses" matches but "analysesx" does not).
let boundary_ok = rest
.as_bytes()
.get(WORD.len())
.is_none_or(|b| !b.is_ascii_alphanumeric());
if matches_word && boundary_ok {
match digits.parse::<usize>() {
Ok(n) if n != real_count => {
let matched = &line[start..probe + WORD.len()];
return Some(format!(
"\"{matched}\" (registry currently has {real_count})"
));
}
_ => {}
}
}
}
}
None
}
#[test]
fn no_stale_analysis_count_in_docs() {
let real_count = AnalysisName::all().len();
let root = workspace_root();
let mut files = Vec::new();
collect_md_files(&root.join("docs"), &mut files);
assert!(
!files.is_empty(),
"scanned zero docs/*.md files — doc-path resolution is broken"
);
let mut violations = Vec::new();
for file in &files {
let rel = file.strip_prefix(&root).unwrap_or(file);
if is_excluded(rel) {
continue;
}
let text = std::fs::read_to_string(file).expect("read doc file");
for (line_idx, line) in text.lines().enumerate() {
if let Some(found) = stale_count_in_line(line, real_count) {
violations.push(format!("{}:{}: {found}", rel.display(), line_idx + 1));
}
}
}
assert!(
violations.is_empty(),
"found {} stale hard-coded analysis count(s) in docs. Describe the registry by \
shape instead (e.g. \"the full analysis registry, enumerated by \
`AnalysisName::all()`\") so the doc can't drift again:\n{}",
violations.len(),
violations.join("\n"),
);
}