use crate::finding::{Finding, Severity};
use crate::instance::Outcome;
use crate::scope::{CorpusCheck, CorpusView};
use headwater_graph::links::{Binding, Link};
use std::collections::{HashMap, HashSet};
pub const RULE: &str = "link.fragment.unresolved";
const NO_LINKS: &str = "the view carries no bound prose links for this corpus";
const NO_ANCHORS: &str = "the view carries no document headings for this corpus";
pub struct Fragments;
impl CorpusCheck for Fragments {
const RULE: &'static str = self::RULE;
const VERSION: u32 = 3;
const NEEDS_LINKS: bool = true;
const NEEDS_ANCHORS: bool = true;
fn evaluate(&self, view: &CorpusView<'_>) -> Outcome {
let Some(links) = view.links() else {
return Outcome::Skipped(NO_LINKS.to_string());
};
let Some(anchors) = view.anchors() else {
return Outcome::Skipped(NO_ANCHORS.to_string());
};
Outcome::failed(
links
.iter()
.filter_map(|link| finding(link, anchors))
.collect(),
)
}
}
fn finding(link: &Link, anchors: &Anchors) -> Option<Finding> {
let fragment = link.fragment.as_deref().filter(|it| !it.is_empty())?;
let target = match &link.binding {
Binding::SameDocument => link.source_path.as_str(),
Binding::Corpus { path, .. } => path.as_str(),
Binding::Missing { .. }
| Binding::Unnormalizable { .. }
| Binding::Repository { .. }
| Binding::External => return None,
};
if anchors.resolves(target, fragment)? {
return None;
}
let (message, remediation) = match &link.binding {
Binding::SameDocument => (
format!("`{}` names no heading of this document", link.destination),
format!("point it at a heading of {target}, or write the heading it names"),
),
_ => (
format!("`{}` names no heading of `{target}`", link.destination),
format!(
"point it at a heading that `{target}` has, or write the heading it names there"
),
),
};
Some(Finding {
rule: self::RULE,
severity: Severity::Error,
obligation: None,
path: link.source_path.clone(),
line: link.span.start.line,
column: link.span.start.col,
message,
remediation,
patch: None,
})
}
pub struct Anchors {
by_path: Vec<(String, Vec<String>)>,
}
impl Anchors {
pub fn of(census: &headwater_census::census::Census) -> Anchors {
Anchors {
by_path: census
.rows
.iter()
.filter_map(|row| {
let document = row.document.as_ref()?;
Some((row.path.clone(), anchors(&document.body)))
})
.collect(),
}
}
#[cfg(test)]
pub(crate) fn of_pairs(pairs: &[(&str, &[&str])]) -> Anchors {
let mut by_path: Vec<(String, Vec<String>)> = pairs
.iter()
.map(|(path, anchors)| {
(
(*path).to_string(),
anchors.iter().map(|it| (*it).to_string()).collect(),
)
})
.collect();
by_path.sort();
Anchors { by_path }
}
fn resolves(&self, path: &str, fragment: &str) -> Option<bool> {
let at = self
.by_path
.binary_search_by(|(known, _)| known.as_str().cmp(path))
.ok()?;
let wanted = fragment.to_lowercase();
Some(self.by_path[at].1.contains(&wanted))
}
}
fn anchors(body: &headwater_doc::Body) -> Vec<String> {
let mut anchors: Vec<String> = Vec::new();
let mut issued: HashSet<String> = HashSet::new();
let mut repeats: HashMap<String, usize> = HashMap::new();
for heading in body.headings() {
let slug = slug(&heading.text());
let mut n = repeats.get(&slug).copied().unwrap_or(0);
let anchor = loop {
let candidate = if n == 0 {
slug.clone()
} else {
format!("{slug}-{n}")
};
n += 1;
if !issued.contains(&candidate) {
break candidate;
}
};
repeats.insert(slug, n);
issued.insert(anchor.clone());
anchors.push(anchor);
}
anchors
}
fn slug(text: &str) -> String {
text.trim()
.to_lowercase()
.chars()
.filter(|c| c.is_alphanumeric() || *c == ' ' || *c == '-' || *c == '_')
.map(|c| if c == ' ' { '-' } else { c })
.collect()
}
#[cfg(test)]
mod comment_links {
use super::{anchors, slug};
use std::path::{Path, PathBuf};
fn engine_root() -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR"))
.join("../..")
.canonicalize()
.expect("the engine root")
}
fn repository_root() -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR"))
.join("../../..")
.canonicalize()
.expect("the repository root")
}
enum State {
Code,
Block(usize),
Str,
Raw(usize),
}
fn comment_markdown(src: &str) -> String {
let mut out = String::new();
let mut state = State::Code;
for line in src.lines() {
let mut comment = String::new();
let bytes: Vec<char> = line.chars().collect();
let mut i = 0;
while i < bytes.len() {
match state {
State::Block(depth) => {
if bytes[i..].starts_with(&['*', '/']) {
state = if depth == 1 {
State::Code
} else {
State::Block(depth - 1)
};
i += 2;
} else if bytes[i..].starts_with(&['/', '*']) {
state = State::Block(depth + 1);
comment.push_str("/*");
i += 2;
} else {
comment.push(bytes[i]);
i += 1;
}
}
State::Str => {
if bytes[i] == '\\' {
i += 2;
} else if bytes[i] == '"' {
state = State::Code;
i += 1;
} else {
i += 1;
}
}
State::Raw(hashes) => {
if bytes[i] == '"'
&& bytes[i + 1..]
.iter()
.take(hashes)
.filter(|c| **c == '#')
.count()
== hashes
{
state = State::Code;
i += 1 + hashes;
} else {
i += 1;
}
}
State::Code => {
if bytes[i..].starts_with(&['/', '/']) {
comment.push_str(&bytes[i..].iter().collect::<String>());
break;
} else if bytes[i..].starts_with(&['/', '*']) {
state = State::Block(1);
i += 2;
} else if bytes[i] == '"' {
state = State::Str;
i += 1;
} else if bytes[i] == 'r'
&& bytes[i + 1..]
.first()
.is_some_and(|c| *c == '"' || *c == '#')
{
let hashes = bytes[i + 1..].iter().take_while(|c| **c == '#').count();
if bytes.get(i + 1 + hashes) == Some(&'"') {
state = State::Raw(hashes);
i += 2 + hashes;
} else {
i += 1;
}
} else if bytes[i] == '\'' {
let closes = match bytes.get(i + 1) {
Some('\\') => bytes.get(i + 3) == Some(&'\''),
Some(_) => bytes.get(i + 2) == Some(&'\''),
None => false,
};
i += if closes {
if bytes.get(i + 1) == Some(&'\\') {
4
} else {
3
}
} else {
1
};
} else {
i += 1;
}
}
}
}
let text = comment.trim_start();
let text = text
.strip_prefix("///")
.or_else(|| text.strip_prefix("//!"))
.or_else(|| text.strip_prefix("//"))
.unwrap_or(text);
out.push_str(text.trim_start());
out.push('\n');
}
out
}
fn sources(dir: &Path, out: &mut Vec<PathBuf>) {
let mut entries: Vec<_> = std::fs::read_dir(dir)
.expect("the engine tree is readable")
.filter_map(Result::ok)
.map(|e| e.path())
.collect();
entries.sort();
for path in entries {
if path.is_dir() {
if path.file_name().is_some_and(|n| n == "target") {
continue;
}
sources(&path, out);
} else if path.extension().is_some_and(|e| e == "rs") {
out.push(path);
}
}
}
fn anchors_of(path: &Path) -> Vec<String> {
let source = std::fs::read_to_string(path).expect("a document this engine cites");
match headwater_doc::split::split(&source) {
Ok(split) => anchors(&headwater_doc::body::scan(
&source,
split.body,
split.body_offset,
)),
Err(_) => anchors(&headwater_doc::body::scan(&source, &source, 0)),
}
}
fn broken(walk: &Path, root: &Path) -> Vec<String> {
let mut files = Vec::new();
sources(walk, &mut files);
let mut out = Vec::new();
for file in files {
let src = std::fs::read_to_string(&file).expect("a source of this engine");
let markdown = comment_markdown(&src);
let body = headwater_doc::body::scan(&markdown, &markdown, 0);
let dir = file.parent().expect("a source has a directory");
for link in &body.links {
if link.image
|| link.destination.contains("://")
|| link.destination.starts_with('/')
{
continue;
}
let (target, fragment) = match link.destination.split_once('#') {
Some((target, fragment)) => (target, Some(fragment)),
None => (link.destination.as_str(), None),
};
if !target.contains("docs/") {
continue;
}
let resolved = normalize(&dir.join(target));
let where_ = format!(
"{}:{}",
file.strip_prefix(root).unwrap_or(&file).display(),
link.span.start.line
);
if !resolved.exists() {
out.push(format!("{where_}: no such file: {}", link.destination));
continue;
}
let Some(fragment) = fragment else { continue };
if resolved.extension().is_some_and(|e| e == "md")
&& !anchors_of(&resolved).contains(&fragment.to_lowercase())
{
out.push(format!("{where_}: no such heading: {}", link.destination));
}
}
}
out
}
fn normalize(path: &Path) -> PathBuf {
let mut out = PathBuf::new();
for part in path.components() {
match part {
std::path::Component::ParentDir => {
out.pop();
}
std::path::Component::CurDir => {}
other => out.push(other),
}
}
out
}
#[test]
fn every_comment_link_into_docs_resolves() {
let root = repository_root();
let broken = broken(&engine_root(), &root);
assert!(
broken.is_empty(),
"{} comment links into docs/ do not resolve:\n{}",
broken.len(),
broken.join("\n")
);
}
#[test]
fn a_broken_link_in_a_comment_is_named() {
let depth = comment_markdown(
"//! see [spec 2](../../../docs/spec/02-taxonomy-model.md) for the rule\n",
);
let scanned = headwater_doc::body::scan(&depth, &depth, 0);
assert_eq!(scanned.links.len(), 1, "the link is found in the comment");
assert_eq!(
scanned.links[0].destination,
"../../../docs/spec/02-taxonomy-model.md"
);
let quoted = comment_markdown("let s = \"[a](../../../docs/nope.md)\"; // and\n");
let scanned = headwater_doc::body::scan("ed, "ed, 0);
assert!(
scanned.links.is_empty(),
"a docs path inside a string literal is not a link a reader follows"
);
assert_eq!(slug("Q4 — Relation storage"), "q4--relation-storage");
let real = anchors_of(&repository_root().join("docs/spec/09-decisions.md"));
assert!(real.contains(&"q4--relation-storage".to_string()));
assert!(!real.contains(&"the-four-scopes".to_string()));
}
fn scratch(name: &str) -> PathBuf {
let nanos = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.expect("a clock later than the epoch")
.as_nanos();
let dir = std::env::temp_dir().join(format!(
"headwater-fragment-{name}-{}-{nanos}",
std::process::id()
));
std::fs::create_dir_all(&dir).expect("a scratch directory");
dir
}
fn write(path: &Path, body: &str) {
std::fs::create_dir_all(path.parent().expect("a file has a directory"))
.expect("a fixture directory");
std::fs::write(path, body).expect("a fixture file");
}
#[test]
fn the_filter_chain_names_what_it_holds_and_passes_what_it_does_not() {
let dir = scratch("filter-chain");
write(
&dir.join("docs/spec/02-taxonomy-model.md"),
"# A model\n\nThe body.\n\n## The heading that is here\n\nMore body.\n",
);
write(
&dir.join("engine/crates/check/src/sample.rs"),
concat!(
"//! [here](../../../../docs/spec/02-taxonomy-model.md#the-heading-that-is-here)\n",
"//! [gone](../../../../docs/spec/99-not-a-document.md)\n",
"//! [retitled](../../../../docs/spec/02-taxonomy-model.md#the-heading-that-is-not)\n",
"//! [rustdoc](../../headwater_doc/struct.Body.html)\n",
"//! [remote](https://example.invalid/docs/spec/02-taxonomy-model.md)\n",
),
);
write(
&dir.join("engine/docs/spec/02-taxonomy-model.md"),
"# A model\n\nThe body.\n\n## The heading that is here\n\nMore body.\n",
);
write(
&dir.join("engine/probe.rs"),
concat!(
"//! [here](docs/spec/02-taxonomy-model.md#the-heading-that-is-here)\n",
"//! [retitled](docs/spec/02-taxonomy-model.md#the-heading-that-is-not)\n",
),
);
let mut found = broken(&dir, &dir);
found.sort();
std::fs::remove_dir_all(&dir).ok();
assert_eq!(
found,
[
"engine/crates/check/src/sample.rs:2: no such file: \
../../../../docs/spec/99-not-a-document.md",
"engine/crates/check/src/sample.rs:3: no such heading: \
../../../../docs/spec/02-taxonomy-model.md#the-heading-that-is-not",
"engine/probe.rs:2: no such heading: \
docs/spec/02-taxonomy-model.md#the-heading-that-is-not",
]
);
}
}
#[cfg(test)]
mod tests {
use super::*;
use headwater_doc::body::scan;
#[test]
fn an_em_dash_leaves_the_spaces_around_it() {
assert_eq!(
slug("Q5 — Voice checking depth"),
"q5--voice-checking-depth"
);
assert_eq!(slug("`$package.optional`"), "packageoptional");
assert_eq!(
slug("Two phases, and why the order matters"),
"two-phases-and-why-the-order-matters"
);
}
fn anchors_of(source: &str) -> Vec<String> {
anchors(&scan(source, source, 0))
}
#[test]
fn a_repeated_heading_takes_a_numeric_suffix() {
let body = scan("# One\n\n# One\n\n# One\n", "# One\n\n# One\n\n# One\n", 0);
assert_eq!(anchors(&body), ["one", "one-1", "one-2"]);
}
#[test]
fn a_longer_heading_is_not_a_repeat_of_the_one_it_extends() {
assert_eq!(
anchors_of("## Edit sites, as spec 2 stands\n\n## Edit sites\n\n## Edit sites\n"),
["edit-sites-as-spec-2-stands", "edit-sites", "edit-sites-1"]
);
}
#[test]
fn a_suffix_an_earlier_heading_took_is_stepped_past() {
assert_eq!(
anchors_of("## One\n\n## One\n\n## One-1\n\n## One\n"),
["one", "one-1", "one-1-1", "one-2"]
);
}
#[test]
fn two_headings_that_differ_never_share_an_anchor() {
let anchors = anchors_of("## One-1\n\n## One\n\n## One\n");
assert_eq!(anchors, ["one-1", "one", "one-2"]);
let issued: std::collections::HashSet<&String> = anchors.iter().collect();
assert_eq!(issued.len(), anchors.len(), "an anchor is issued once");
}
}
#[cfg(test)]
mod arms {
use super::*;
use crate::scope::CorpusView;
use headwater_doc::LinkForm;
use headwater_yaml::{Position, Span};
const CITER: &str = "docs/spec/01-conceptual-model.md";
const TARGET: &str = "docs/spec/glossary.md";
fn span() -> Span {
Span {
start: Position {
line: 30,
col: 5,
offset: 0,
},
end: Position {
line: 30,
col: 6,
offset: 1,
},
}
}
fn link(destination: &str, fragment: Option<&str>, binding: Binding) -> Link {
Link {
source_path: CITER.to_string(),
destination: destination.to_string(),
fragment: fragment.map(str::to_string),
form: LinkForm::Inline,
span: span(),
binding,
}
}
fn corpus(path: &str) -> Binding {
Binding::Corpus {
path: path.to_string(),
class: "typed",
id: None,
}
}
fn index() -> Anchors {
Anchors::of_pairs(&[
(CITER, &["a-heading-of-the-citer"]),
(TARGET, &["projection"]),
])
}
#[test]
fn a_fragment_that_names_no_heading_of_the_target_is_an_error_at_the_citing_line() {
let found = finding(
&link(
"glossary.md#projections",
Some("projections"),
corpus(TARGET),
),
&index(),
)
.expect("a finding");
assert_eq!(found.rule, self::RULE);
assert_eq!(found.severity, Severity::Error);
assert_eq!(found.path, CITER);
assert_eq!(found.line, 30);
assert_eq!(found.column, 5);
assert!(found.message.contains(TARGET), "{found:#?}");
assert!(!found.message.contains("this document"), "{found:#?}");
assert!(!found.fixable(), "{found:#?}");
}
#[test]
fn the_near_arm_says_this_document_and_the_far_arm_names_the_file() {
let near = finding(
&link(
"#no-such-heading",
Some("no-such-heading"),
Binding::SameDocument,
),
&index(),
)
.expect("a finding");
assert!(near.message.contains("this document"), "{near:#?}");
assert!(!near.message.contains(TARGET), "{near:#?}");
assert_eq!(near.path, CITER);
}
#[test]
fn a_fragment_that_resolves_is_not_a_finding_on_either_arm() {
assert!(finding(
&link("glossary.md#projection", Some("projection"), corpus(TARGET)),
&index()
)
.is_none());
assert!(finding(
&link(
"#a-heading-of-the-citer",
Some("a-heading-of-the-citer"),
Binding::SameDocument
),
&index()
)
.is_none());
}
#[test]
fn a_fragment_resolves_whatever_case_the_author_wrote_it_in() {
assert!(finding(
&link("glossary.md#Projection", Some("Projection"), corpus(TARGET)),
&index()
)
.is_none());
}
#[test]
fn every_binding_that_is_not_this_rule_produces_nothing() {
for binding in [
Binding::Missing {
path: "docs/spec/gone.md".to_string(),
},
Binding::Unnormalizable {
why: "climbs above the repository root".to_string(),
},
Binding::Repository {
path: "CLAUDE.md".to_string(),
},
Binding::External,
] {
assert!(finding(&link("x#y", Some("y"), binding), &index()).is_none());
}
}
#[test]
fn a_link_carrying_no_fragment_is_not_this_rules_business() {
assert!(finding(&link("glossary.md", None, corpus(TARGET)), &index()).is_none());
assert!(finding(&link("glossary.md#", Some(""), corpus(TARGET)), &index()).is_none());
}
#[test]
fn a_corpus_path_with_no_parsed_document_gets_no_verdict_rather_than_a_guess() {
let untyped = "docs/spec/notes.txt";
assert_eq!(index().resolves(untyped, "anything"), None);
assert!(finding(
&link("notes.txt#anything", Some("anything"), corpus(untyped)),
&index()
)
.is_none());
}
#[test]
fn a_view_with_no_links_skips_rather_than_passes() {
let view = CorpusView::only_links_and_anchors(None, None);
assert!(matches!(Fragments.evaluate(&view), Outcome::Skipped(why) if why == NO_LINKS));
}
#[test]
fn a_view_with_links_and_no_anchors_skips_with_the_other_reason() {
let links = vec![link(
"glossary.md#projection",
Some("projection"),
corpus(TARGET),
)];
let view = CorpusView::only_links_and_anchors(Some(&links), None);
assert!(matches!(Fragments.evaluate(&view), Outcome::Skipped(why) if why == NO_ANCHORS));
}
#[test]
fn the_unresolved_fragments_of_the_view_become_the_findings() {
let anchors = index();
let broken = vec![
link(
"glossary.md#projections",
Some("projections"),
corpus(TARGET),
),
link("#nope", Some("nope"), Binding::SameDocument),
link("glossary.md#projection", Some("projection"), corpus(TARGET)),
link("https://example.com", None, Binding::External),
];
let view = CorpusView::only_links_and_anchors(Some(&broken), Some(&anchors));
let Outcome::Failed(found) = Fragments.evaluate(&view) else {
panic!("two unresolved fragments are two findings");
};
assert_eq!(found.len(), 2, "{found:#?}");
let clean = vec![link(
"glossary.md#projection",
Some("projection"),
corpus(TARGET),
)];
let view = CorpusView::only_links_and_anchors(Some(&clean), Some(&anchors));
assert!(matches!(Fragments.evaluate(&view), Outcome::Passed));
}
}