headwater-check 0.4.0

Generates the rules from the taxonomy, runs them, computes coverage against the census, and keys each instance on what it read and on the clock it was handed
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
// SPDX-License-Identifier: Apache-2.0
//! A check at corpus grain: a prose link's fragment names a heading of the
//! document it points at, whether that is the citing document or another one.
//!
//! [Spec 12](../../../../docs/spec/12-check-layer.md#the-five-origins-of-a-check)
//! lists prose-link resolution among the Document-origin examples, and
//! [`headwater_graph::links`] left the far half deliberately undone: "whether
//! `#q4--relation-storage` names a heading of the target is a Document check
//! ... a graph build that started checking headings would be the check layer
//! with no scope declaration."
//!
//! # Both halves are here now, and neither of them is a Document check
//!
//! A fragment with no path names a heading of the document that wrote it, and
//! that half did sit at `Document` grain for two editions.
//!
//! A fragment on a path names a heading of **another** document, and no
//! document-grained view carries one.
//! [HW-OBL-0077](../../../../docs/obligations/0077-a-fragment-on-a-path-needs-a-grain-that-no-scope-supplies.md)
//! recorded that gap, and it read as a missing fifth grain. It is not. Handing
//! a document-scoped rule the target documents as a declared input is unsound
//! for a reason the code states rather than argues: `Instance::paths()` is the
//! read set, and [`crate::coverage`] routes an instance to every path in
//! `paths()`. They are one list by construction. So every target document the
//! read set must name — and soundness demands it, or a renamed heading in the
//! target leaves a cached pass standing — is thereby counted as *checked by
//! this rule*, which is the miscount [`crate::coverage`] exists to name.
//!
//! `Corpus` is the grain, and [`crate::link_path`] already argues it word for
//! word for the same reference class: the defect is a relation between two
//! files, a rename touches no byte of the citing document, so a
//! document-scoped key would not move and a cached pass would outlive the
//! rename that made it false. One instance, whose findings each name the
//! citing document and the line the author wrote.
//!
//! # What this widening costs [`crate::coverage`], measured on this corpus
//!
//! A corpus-scoped instance routes to no document, so the 312 document-routed
//! instances this rule made become one. A classified document whose only
//! routed check was this one would become a `coverage.document_unchecked`
//! finding. That cannot happen here: `facet.required.missing` reports one
//! instance against every classified document and skips none of them, so every
//! classified document is already routed by an evaluated document-scoped rule
//! that is not this one.
//!
//! **That is a fact about this corpus and not a property of the engine.** A
//! thin shelf in an adopting corpus, whose kind requires no facet, could lose
//! its only routed check to this move.
//! [HW-OBL-0071](../../../../docs/obligations/0071-a-corpus-scoped-check-makes-the-coverage-rule-unreachable.md)
//! is the record of that shape, and this change makes it one instance larger.
//!
//! # Why the anchors are computed rather than read
//!
//! A Markdown heading has no anchor in the source. The anchor is what a
//! renderer derives from the heading text, so a check that resolves a fragment
//! has to derive it the same way, and two renderers do not agree. The rule
//! below is the one GitHub applies, because that is where this corpus is read,
//! and it is stated in one function so that an adopter reading a wrong verdict
//! finds one place to look.
//!
//! One place in this engine, and a second across a language boundary. The site
//! derives its own anchors with `pymdownx.slugs.slugify(case='lower')`, which
//! `mkdocs.yml` names and `.github/workflows/ci.yml` pins, and neither
//! implementation can call the other. `mod renderer_slug` at the foot of this
//! file is what now holds the two to one answer, over every heading the site
//! renders, and it fails naming the document, the heading and both slugs.
//!
//! # The one instance says what it read
//!
//! A run whose scope admitted no links, and a run that reached no document
//! body, each **skip with that reason** rather than passing. A pass would
//! report a corpus this rule never read as clean, which is
//! [`crate::link_path`]'s `NO_LINKS` and [`crate::duplicate`]'s `NO_REPORT` at
//! the same grain.
//!
//! The population itself is stated by the graph report, which counts the
//! fragment-bearing links by arm beside the bindings it already prints. It is
//! there rather than here because a passing outcome carries no sentence, and a
//! reader who wants to know how much of the corpus this rule reads should not
//! have to break it to find out.
//!
//! # What is not this rule
//!
//! `Binding::Missing` and `Binding::Unnormalizable` are a path that resolved to
//! nothing, and they belong to [`crate::link_path`]. `Binding::Repository` and
//! `Binding::External` name a body this engine never read, so this rule has
//! nothing to compare a fragment against and says nothing.
//!
//! `Binding::Corpus` on a file the census walked but parsed no document out of
//! — an untyped or an excluded file — is the one case inside this rule's
//! bindings that it passes over. No anchor list exists for such a path, so a
//! finding there would be a guess. See [`Anchors::of`].

use crate::finding::{Finding, Severity};
use crate::instance::Outcome;
use crate::scope::{CorpusCheck, CorpusView};
use headwater_graph::links::{Binding, Link};
use std::collections::{HashMap, HashSet};

pub const RULE: &str = "link.fragment.unresolved";

/// A run whose scope did not admit the links has nothing to read, and it says
/// so rather than passing. [`crate::link_path::NO_LINKS`] verbatim, at the same
/// grain and for the same reason.
const NO_LINKS: &str = "the view carries no bound prose links for this corpus";

/// A run that reached no document body cannot derive an anchor, so it has no
/// second half to compare a fragment against.
const NO_ANCHORS: &str = "the view carries no document headings for this corpus";

/// The check. It reads no declaration, and the module comment says why: the
/// taxonomy language has no member that turns prose-link resolution on or off,
/// because [spec 1](../../../../docs/spec/01-conceptual-model.md#prose-links-are-not-relations)
/// makes it a property of a corpus rather than of a kind.
pub struct Fragments;

impl CorpusCheck for Fragments {
    const RULE: &'static str = self::RULE;
    /// The fourth edition, and this one stopped case folding the citation, so
    /// it reports links that every earlier edition passed. A warm cache holding
    /// an edition-3 verdict holds a verdict reached by a wider comparison, and
    /// it would serve that pass forever over the citations this edition reads
    /// as dead. The third edition widened the rule to cross-document
    /// fragments; the second renumbered a repeated heading; the first counted
    /// every earlier anchor that began with its slug and a hyphen, which is not
    /// what a renderer does.
    const VERSION: u32 = 4;
    const NEEDS_LINKS: bool = true;
    const NEEDS_ANCHORS: bool = true;

    fn evaluate(&self, view: &CorpusView<'_>) -> Outcome {
        let Some(links) = view.links() else {
            return Outcome::Skipped(NO_LINKS.to_string());
        };
        let Some(anchors) = view.anchors() else {
            return Outcome::Skipped(NO_ANCHORS.to_string());
        };
        // A quoted link belongs to another author, and an image is a reference
        // to an asset. Both were dropped by the graph build, which is the one
        // reader of this corpus's prose links, so this rule inherits that
        // answer rather than writing a second one.
        Outcome::failed(
            links
                .iter()
                .filter_map(|link| finding(link, anchors))
                .collect(),
        )
    }
}

/// One finding per fragment that names no heading of the document it points at,
/// and nothing for every other link.
fn finding(link: &Link, anchors: &Anchors) -> Option<Finding> {
    let fragment = link.fragment.as_deref().filter(|it| !it.is_empty())?;
    // The two arms this rule reads. Every other binding is somebody else's, and
    // the module comment enumerates which.
    let target = match &link.binding {
        Binding::SameDocument => link.source_path.as_str(),
        Binding::Corpus { path, .. } => path.as_str(),
        Binding::Missing { .. }
        | Binding::Unnormalizable { .. }
        | Binding::Repository { .. }
        | Binding::External => return None,
    };
    // A corpus path the census walked and parsed no document out of. There is
    // no heading list, so there is no verdict, and a finding would be a guess.
    if anchors.resolves(target, fragment)? {
        return None;
    }
    let (message, remediation) = match &link.binding {
        Binding::SameDocument => (
            format!("`{}` names no heading of this document", link.destination),
            format!("point it at a heading of {target}, or write the heading it names"),
        ),
        _ => (
            format!("`{}` names no heading of `{target}`", link.destination),
            format!(
                "point it at a heading that `{target}` has, or write the heading it names there"
            ),
        ),
    };
    Some(Finding {
        rule: self::RULE,
        severity: Severity::Error,
        obligation: None,
        // The citing document and the line the author wrote, on
        // [`crate::link_path`]'s terms: the instance is one, and an author
        // still reads the defect where they made it.
        path: link.source_path.clone(),
        line: link.span.start.line,
        column: link.span.start.col,
        message,
        remediation,
        // The heading an author meant is a guess among the headings the target
        // has, and spec 12 admits a fix only where one outcome is derivable
        // without judgment.
        patch: None,
    })
}

/// Every anchor of every document of this corpus, by path.
///
/// # Why this is built beside the read set rather than injected into it
///
/// A corpus-scoped instance's read set is already every census row that carries
/// a document, so the bytes these anchors derive from are named in the key
/// before this index exists. A second key component would hash the same bytes
/// twice. The declaration is still made — [`CorpusCheck::NEEDS_ANCHORS`] —
/// because a rule that reads an input it never declared is the thing the scope
/// declarations exist to stop, and a reader of the trait should be able to see
/// every input this rule takes without reading its body.
pub struct Anchors {
    /// In the census's own order, which is path order, so a lookup is a binary
    /// search rather than a scan of the corpus per link.
    by_path: Vec<(String, Vec<String>)>,
}

impl Anchors {
    /// Derived from the rows that carry a document, which is exactly the read
    /// set of a corpus-scoped instance.
    ///
    /// A row that carries none contributed nothing and could not have: there is
    /// no parsed body, so there is no heading, so a link into that path has no
    /// answer here rather than a false one.
    pub fn of(census: &headwater_census::census::Census) -> Anchors {
        Anchors {
            by_path: census
                .rows
                .iter()
                .filter_map(|row| {
                    let document = row.document.as_ref()?;
                    Some((row.path.clone(), anchors(&document.body)))
                })
                .collect(),
        }
    }

    /// Whether `fragment` names a heading of `path`.
    ///
    /// `None` where this corpus holds no heading list for `path` at all, which
    /// is a path outside the corpus or a file the census parsed no document
    /// out of. That is the absence of a verdict rather than a passing one, and
    /// the caller reports nothing on it.
    /// An index built from a list, for the tests of the rule that reads one.
    /// `cfg(test)` so that no shipped path can build an index whose anchors did
    /// not come from a census.
    #[cfg(test)]
    pub(crate) fn of_pairs(pairs: &[(&str, &[&str])]) -> Anchors {
        let mut by_path: Vec<(String, Vec<String>)> = pairs
            .iter()
            .map(|(path, anchors)| {
                (
                    (*path).to_string(),
                    anchors.iter().map(|it| (*it).to_string()).collect(),
                )
            })
            .collect();
        // The shipped constructor inherits the census's path order, and
        // `resolves` binary searches. A test index that was not sorted would
        // answer `None` for a path it holds.
        by_path.sort();
        Anchors { by_path }
    }

    fn resolves(&self, path: &str, fragment: &str) -> Option<bool> {
        let at = self
            .by_path
            .binary_search_by(|(known, _)| known.as_str().cmp(path))
            .ok()?;
        // Exactly, and not folded. [`anchors`] folds the *heading text*
        // because GitHub's slugger does, and the result is the identifier a
        // browser is handed. The citation is then matched against it byte for
        // byte, because "find a potential indicated element" matches an `id`
        // exactly and falls back to an `a` element's `name` exactly. Folding
        // here failed open in one direction: a dead link passed, and no live
        // link was ever reported. The case table carries what Chrome did.
        Some(self.by_path[at].1.iter().any(|it| it == fragment))
    }
}

/// Every anchor a renderer gives this document, in heading order.
///
/// A repeated heading takes a numeric suffix, first occurrence bare. Two
/// `## Consequences` headings therefore give `consequences` and
/// `consequences-1`, and a link to the second resolves.
///
/// # Two rules, and either one alone gets a document of this corpus wrong
///
/// The suffix counts how many anchors this **exact** slug has already been the
/// base of. A longer heading whose slug merely begins with a shorter one is
/// not a repeat of it, so `## Edit sites, as spec 2 stands` standing above two
/// `## Edit sites` leaves them `edit-sites` and `edit-sites-1`.
/// `docs/evaluations/schema-format-walkthrough.md` is that document.
///
/// Then a candidate an earlier heading already took is stepped past, and the
/// counter resumes from where the search stopped. So `## One`, `## One` and
/// `## One-1` give `one`, `one-1` and `one-1-1`, and a fourth `## One` gives
/// `one-2` rather than colliding with the third heading.
///
/// That second half is what makes this function injective: two headings that
/// differ never receive one anchor, because an anchor is issued once. Counting
/// a prefix as a repeat was not injective, and the witness is short —
/// `## One-1` above two `## One` gave the first two headings the anchor
/// `one-1` each.
///
/// Both halves are settled against GitHub's own renderer rather than against a
/// second implementation of the rule, by posting headings to its `/markdown`
/// endpoint and reading the `user-content-` identifiers it returns. The tests
/// below carry the cases that endpoint answered.
fn anchors(body: &headwater_doc::Body) -> Vec<String> {
    let mut anchors: Vec<String> = Vec::new();
    let mut issued: HashSet<String> = HashSet::new();
    let mut repeats: HashMap<String, usize> = HashMap::new();
    for heading in body.headings() {
        let slug = slug(&heading.text());
        let mut n = repeats.get(&slug).copied().unwrap_or(0);
        let anchor = loop {
            let candidate = if n == 0 {
                slug.clone()
            } else {
                format!("{slug}-{n}")
            };
            n += 1;
            if !issued.contains(&candidate) {
                break candidate;
            }
        };
        repeats.insert(slug, n);
        issued.insert(anchor.clone());
        anchors.push(anchor);
    }
    anchors
}

/// The anchor a renderer derives from a heading.
///
/// Lower case; every character that is not a letter, a digit, a space or a
/// hyphen removed; then each space becomes a hyphen. An em dash therefore
/// leaves the two spaces around it, and `Q5 — Voice checking depth` resolves as
/// `q5--voice-checking-depth`, which is what this corpus writes.
fn slug(text: &str) -> String {
    text.trim()
        .to_lowercase()
        .chars()
        .filter(|c| c.is_alphanumeric() || *c == ' ' || *c == '-' || *c == '_')
        .map(|c| if c == ' ' { '-' } else { c })
        .collect()
}

/// Every relative link this engine's own comments write into `docs/`.
///
/// # This is a test and not a rule, and the module above says why
///
/// A fragment on a path names a heading of another document, and no scope this
/// engine has carries one. So the thing checked here cannot be a check-layer
/// rule at all: `corpus.root` is `docs`, a `.rs` file has no front matter and
/// therefore no kind, and the census never reaches one. What this material is
/// closer to is `cargo fmt` — a property of the engine's own source, held by
/// the engine's own suite.
///
/// It lives inside this module rather than beside it because the slugging rule
/// is here. `CLAUDE.md` states that no second copy of a rule lives in a script,
/// and a comment-link checker with its own slugger would be exactly that: two
/// definitions of what a heading anchor is, agreeing today and drifting on the
/// first edit. This test calls [`anchors`] and [`slug`], so a corpus link and a
/// comment link resolve by one rule or by neither.
///
/// # What a broken link costs
///
/// Nothing else reads a `.rs` comment. No check, no gate and no continuous
/// integration step, which is why twenty-nine of these rotted in silence:
/// twenty at the wrong `../` depth, six naming a heading that was retitled, and
/// three naming a document that was.
#[cfg(test)]
mod comment_links {
    use super::{anchors, slug};
    use std::path::{Path, PathBuf};

    /// The engine's own tree, four levels above this crate.
    fn engine_root() -> PathBuf {
        Path::new(env!("CARGO_MANIFEST_DIR"))
            .join("../..")
            .canonicalize()
            .expect("the engine root")
    }

    pub(super) fn repository_root() -> PathBuf {
        Path::new(env!("CARGO_MANIFEST_DIR"))
            .join("../../..")
            .canonicalize()
            .expect("the repository root")
    }

    /// The comment text of a Rust file, one output line per input line.
    ///
    /// [`headwater_graph::comments::rust_comment_text`] is the shared walk.
    /// `CommentScan`, the shipped `comment-scan` resolver, is the other
    /// reader of it, and this module's header states why one walk serves
    /// both rather than a second copy of it living here.
    use headwater_graph::comments::rust_comment_text as comment_markdown;

    /// Every `.rs` file of the engine, `target/` excluded.
    fn sources(dir: &Path, out: &mut Vec<PathBuf>) {
        let mut entries: Vec<_> = std::fs::read_dir(dir)
            .expect("the engine tree is readable")
            .filter_map(Result::ok)
            .map(|e| e.path())
            .collect();
        entries.sort();
        for path in entries {
            if path.is_dir() {
                if path.file_name().is_some_and(|n| n == "target") {
                    continue;
                }
                sources(&path, out);
            } else if path.extension().is_some_and(|e| e == "rs") {
                out.push(path);
            }
        }
    }

    /// The anchor set of a Markdown file on disk.
    ///
    /// The front matter is split off first. A `---` fence read as body would
    /// give the line above it a setext heading and an anchor that no renderer
    /// produces, and an anchor set that is too large is an anchor set that
    /// confirms a broken link.
    fn anchors_of(path: &Path) -> Vec<String> {
        let source = std::fs::read_to_string(path).expect("a document this engine cites");
        match headwater_doc::split::split(&source) {
            Ok(split) => anchors(&headwater_doc::body::scan(
                &source,
                split.body,
                split.body_offset,
            )),
            Err(_) => anchors(&headwater_doc::body::scan(&source, &source, 0)),
        }
    }

    /// Every broken link, as `path:line: target`.
    ///
    /// `walk` is the tree of Rust sources to read and `root` is what a reported
    /// path is relative to. They are two parameters rather than one so that the
    /// whole chain below can be pointed at a tree a test wrote. A function that
    /// walked a hardcoded root could only ever run over material that is clean,
    /// and a filter inverted inside it would read nothing while every test here
    /// stayed green.
    fn broken(walk: &Path, root: &Path) -> Vec<String> {
        let mut files = Vec::new();
        sources(walk, &mut files);
        let mut out = Vec::new();
        for file in files {
            let src = std::fs::read_to_string(&file).expect("a source of this engine");
            let markdown = comment_markdown(&src);
            let body = headwater_doc::body::scan(&markdown, &markdown, 0);
            let dir = file.parent().expect("a source has a directory");
            for link in &body.links {
                // The class this holds is the one a reader follows into
                // this repository, and two classes are skipped for reasons
                // that differ. An absolute URL and an absolute path leave the
                // checkout, so nothing here can resolve them. A relative
                // destination naming no `docs/` component is rustdoc's own
                // output tree, which resolves against `target/doc` and not
                // against the source, so a filesystem test of one would report
                // a defect that is not one. The `docs/` test below is what
                // holds that class out.
                //
                // A dotless relative destination such as
                // `docs/spec/02-taxonomy-model.md#x` is neither. It resolves
                // against the source file's own directory exactly as
                // `./docs/...` does, in a browser and here, so it is checked
                // rather than skipped. Requiring the leading `.` skipped it
                // silently, which is the second half of #206.
                if link.image
                    || link.destination.contains("://")
                    || link.destination.starts_with('/')
                {
                    continue;
                }
                let (target, fragment) = match link.destination.split_once('#') {
                    Some((target, fragment)) => (target, Some(fragment)),
                    None => (link.destination.as_str(), None),
                };
                if !target.contains("docs/") {
                    continue;
                }
                let resolved = normalize(&dir.join(target));
                let where_ = format!(
                    "{}:{}",
                    file.strip_prefix(root).unwrap_or(&file).display(),
                    link.span.start.line
                );
                // The path is resolved before the fragment. They are two
                // defects, and a checker that skipped an unresolved file would
                // see neither the depth error nor the retitled document.
                if !resolved.exists() {
                    out.push(format!("{where_}: no such file: {}", link.destination));
                    continue;
                }
                let Some(fragment) = fragment else { continue };
                if resolved.extension().is_some_and(|e| e == "md")
                    && !anchors_of(&resolved).iter().any(|it| it == fragment)
                {
                    out.push(format!("{where_}: no such heading: {}", link.destination));
                }
            }
        }
        out
    }

    /// `..` resolved textually, because the path a comment writes may not exist
    /// and `canonicalize` refuses one that does not.
    fn normalize(path: &Path) -> PathBuf {
        let mut out = PathBuf::new();
        for part in path.components() {
            match part {
                std::path::Component::ParentDir => {
                    out.pop();
                }
                std::path::Component::CurDir => {}
                other => out.push(other),
            }
        }
        out
    }

    #[test]
    fn every_comment_link_into_docs_resolves() {
        let root = repository_root();
        let broken = broken(&engine_root(), &root);
        assert!(
            broken.is_empty(),
            "{} comment links into docs/ do not resolve:\n{}",
            broken.len(),
            broken.join("\n")
        );
    }

    /// The instrument fires, and it names the file, the line and the target.
    ///
    /// A checker over material that is clean is indistinguishable from a
    /// checker that reads nothing, so the two defect classes are provoked here
    /// rather than trusted. Both inputs are written in this test, so nothing in
    /// the tree has to be broken to hold it.
    #[test]
    fn a_broken_link_in_a_comment_is_named() {
        let depth = comment_markdown(
            "//! see [spec 2](../../../docs/spec/02-taxonomy-model.md) for the rule\n",
        );
        let scanned = headwater_doc::body::scan(&depth, &depth, 0);
        assert_eq!(scanned.links.len(), 1, "the link is found in the comment");
        assert_eq!(
            scanned.links[0].destination,
            "../../../docs/spec/02-taxonomy-model.md"
        );

        let quoted = comment_markdown("let s = \"[a](../../../docs/nope.md)\"; // and\n");
        let scanned = headwater_doc::body::scan(&quoted, &quoted, 0);
        assert!(
            scanned.links.is_empty(),
            "a docs path inside a string literal is not a link a reader follows"
        );

        // The anchor half, against a heading this repository has. The slugger
        // is the rule above, so this also pins the em-dash reading that made
        // twenty-seven of these look broken to a checker that kept the dash.
        assert_eq!(slug("Q4 — Relation storage"), "q4--relation-storage");
        let real = anchors_of(&repository_root().join("docs/spec/09-decisions.md"));
        assert!(real.contains(&"q4--relation-storage".to_string()));
        assert!(!real.contains(&"the-four-scopes".to_string()));
    }

    /// A directory under the temporary directory that is removed when this value
    /// is dropped, so a case that fails an assertion leaves nothing behind (#1158).
    struct Scratch(std::path::PathBuf);

    impl Drop for Scratch {
        fn drop(&mut self) {
            let _ = std::fs::remove_dir_all(&self.0);
        }
    }

    impl std::ops::Deref for Scratch {
        type Target = std::path::Path;
        fn deref(&self) -> &std::path::Path {
            &self.0
        }
    }

    impl AsRef<std::path::Path> for Scratch {
        fn as_ref(&self) -> &std::path::Path {
            &self.0
        }
    }

    impl AsRef<std::ffi::OsStr> for Scratch {
        fn as_ref(&self) -> &std::ffi::OsStr {
            self.0.as_os_str()
        }
    }

    /// A directory nothing else in this process writes into.
    ///
    /// The process identifier alone is not a key here, because cargo runs the
    /// cases of one target as threads of one process (#189). The test's own
    /// name and the clock carry the rest.
    fn scratch(name: &str) -> Scratch {
        let nanos = std::time::SystemTime::now()
            .duration_since(std::time::UNIX_EPOCH)
            .expect("a clock later than the epoch")
            .as_nanos();
        let dir = std::env::temp_dir().join(format!(
            "headwater-fragment-{name}-{}-{nanos}",
            std::process::id()
        ));
        std::fs::create_dir_all(&dir).expect("a scratch directory");
        Scratch(dir)
    }

    fn write(path: &Path, body: &str) {
        std::fs::create_dir_all(path.parent().expect("a file has a directory"))
            .expect("a fixture directory");
        std::fs::write(path, body).expect("a fixture file");
    }

    /// The comment guard compares a fragment exactly, as the rule above does.
    ///
    /// A second copy of the comparison lives here, so a case that only the
    /// corpus rule carried would leave this one folding forever. The heading is
    /// `## The heading that is here` and the citation differs from its anchor
    /// in nothing but capitalization.
    #[test]
    fn a_comment_fragment_that_differs_only_in_case_is_named() {
        let dir = scratch("comment-case");
        write(
            &dir.join("docs/spec/02-taxonomy-model.md"),
            "# A model\n\nThe body.\n\n## The heading that is here\n\nMore body.\n",
        );
        write(
            &dir.join("engine/crates/check/src/sample.rs"),
            concat!(
                "//! [exact](../../../../docs/spec/02-taxonomy-model.md#the-heading-that-is-here)\n",
                "//! [cased](../../../../docs/spec/02-taxonomy-model.md#The-Heading-That-Is-Here)\n",
            ),
        );

        let found = broken(&dir, &dir);
        std::fs::remove_dir_all(&dir).ok();

        let expected = "engine/crates/check/src/sample.rs:2: no such heading: \
                        ../../../../docs/spec/02-taxonomy-model.md#The-Heading-That-Is-Here";
        assert_eq!(found, [expected]);
    }

    /// The whole chain, against a tree written here, one link of each class.
    ///
    /// [`every_comment_link_into_docs_resolves`] runs over material that is
    /// clean, so it returns the same empty list whether the chain reads the
    /// tree or reads nothing at all. Invert a filter inside `broken` and that
    /// test stays green. This one gives the chain five links whose classes are
    /// known and asserts exactly which of them come back, so a filter that
    /// stopped holding its class fails here in one direction or the other.
    #[test]
    fn the_filter_chain_names_what_it_holds_and_passes_what_it_does_not() {
        let dir = scratch("filter-chain");
        write(
            &dir.join("docs/spec/02-taxonomy-model.md"),
            "# A model\n\nThe body.\n\n## The heading that is here\n\nMore body.\n",
        );
        // Four levels below the tree root, as this crate's own sources are.
        write(
            &dir.join("engine/crates/check/src/sample.rs"),
            concat!(
                "//! [here](../../../../docs/spec/02-taxonomy-model.md#the-heading-that-is-here)\n",
                "//! [gone](../../../../docs/spec/99-not-a-document.md)\n",
                "//! [retitled](../../../../docs/spec/02-taxonomy-model.md#the-heading-that-is-not)\n",
                "//! [rustdoc](../../headwater_doc/struct.Body.html)\n",
                "//! [remote](https://example.invalid/docs/spec/02-taxonomy-model.md)\n",
            ),
        );
        // The dotless class, both directions, one level below the tree root.
        write(
            &dir.join("engine/docs/spec/02-taxonomy-model.md"),
            "# A model\n\nThe body.\n\n## The heading that is here\n\nMore body.\n",
        );
        write(
            &dir.join("engine/probe.rs"),
            concat!(
                "//! [here](docs/spec/02-taxonomy-model.md#the-heading-that-is-here)\n",
                "//! [retitled](docs/spec/02-taxonomy-model.md#the-heading-that-is-not)\n",
            ),
        );

        let mut found = broken(&dir, &dir);
        found.sort();
        std::fs::remove_dir_all(&dir).ok();

        assert_eq!(
            found,
            [
                "engine/crates/check/src/sample.rs:2: no such file: \
                 ../../../../docs/spec/99-not-a-document.md",
                "engine/crates/check/src/sample.rs:3: no such heading: \
                 ../../../../docs/spec/02-taxonomy-model.md#the-heading-that-is-not",
                "engine/probe.rs:2: no such heading: \
                 docs/spec/02-taxonomy-model.md#the-heading-that-is-not",
            ]
        );
    }
}

/// This engine's slug rule held against the renderer's, over every heading the
/// site serves.
///
/// # Two implementations of one rule, in two languages
///
/// [`slug`] above is this engine's. The site's is
/// `pymdownx.slugs.slugify(case='lower')`, which `mkdocs.yml` names and
/// `.github/workflows/ci.yml` pins at `pymdown-extensions==10.11.2`. Neither
/// can call the other — this engine cannot reach Python while it checks, and
/// MkDocs cannot reach Rust while it builds — so two implementations is the
/// situation rather than the defect. The defect was that nothing compared
/// them, which is [#545](https://github.com/headwater-ai/headwater/issues/545).
///
/// What the absence cost is measured rather than argued. A release of
/// `pymdown-extensions` that moved that function would move every anchor the
/// site serves with no commit in this repository, and the first report would
/// arrive as *N dead links across the corpus*. That is the shape #431 took,
/// and it cost three weeks and two agents to reduce to one sentence about em
/// dashes. This says instead: these two rules disagree, on this heading of
/// this document, and here are the two slugs.
///
/// # What this compares, and what it does not
///
/// [`slug`], which is the whole of the derivation a renderer performs on one
/// heading. Not the repeat suffix [`anchors`] adds, which is a second rule
/// settled against GitHub's own renderer, where this corpus is read, rather
/// than against a renderer it is not. Not the `id=` attribute of the built
/// site either: `tools/site/check-site-fragments.py` reads those and states in
/// its own header that it asserts nothing about how an anchor is derived.
///
/// # The population is derived and never carried
///
/// `mkdocs.yml` declares `docs_dir` and `exclude_docs`, and this reads both
/// rather than repeating either. A literal file list would go stale on the
/// next document, and a literal heading count on the next heading. The
/// exclusions are not cosmetic: `docs/taxonomies/` carries third-party source
/// fixtures whose `<a name="...">` headings disagree by construction, and the
/// site never renders one of them.
///
/// # Skipping, and what turns a skip into a failure
///
/// The comparison reaches `python3` and `pymdownx`. Absent either, it skips
/// with a printed reason, on the terms `crates/hash/tests/oracle.rs` and the
/// two stock-validator differentials already take. Set `HEADWATER_SLUG_ORACLE`
/// and the skip becomes a failure. Continuous integration sets it, so this
/// comparison cannot go quiet by losing a dependency.
#[cfg(test)]
mod renderer_slug {
    use super::comment_links::repository_root;
    use super::slug;
    use std::io::Write;
    use std::path::{Path, PathBuf};
    use std::process::{Command, Stdio};

    /// The Python that answers, one slugger per name.
    ///
    /// `pymdownx` is the site's own. `default` is Python-Markdown's, which
    /// this corpus is **not** rendered with, and it is here so that the
    /// comparison can be watched refusing something: spec 12 refuses a check
    /// that ships with no failing fixture, and a gate that has refused nothing
    /// is a gate nobody has seen work. `pymdown-extensions` depends on
    /// `Markdown`, so the second slugger needs no second pin.
    ///
    /// Headings travel separated by a NUL in both directions. No heading can
    /// hold one, so nothing has to be escaped and nothing can be mis-split.
    const SLUGGERS: &str = r#"
import sys
which = sys.argv[1]
if which == "pymdownx":
    from pymdownx.slugs import slugify
    fn = slugify(case="lower")
elif which == "default":
    from markdown.extensions.toc import slugify as fn
else:
    raise SystemExit("no slugger is named " + which)
raw = sys.stdin.buffer.read().decode("utf-8")
texts = raw.split("\0") if raw else []
sys.stdout.buffer.write("\0".join(fn(text, "-") for text in texts).encode("utf-8"))
"#;

    /// Every heading the site renders, and what the walk that found them read.
    struct Population {
        /// `(path relative to the repository root, heading text)`, in path
        /// order and then in document order.
        headings: Vec<(String, String)>,
        documents: usize,
        excluded: usize,
    }

    /// A heading the two rules do not agree on.
    struct Disagreement {
        document: String,
        heading: String,
        ours: String,
        theirs: String,
    }

    impl Disagreement {
        /// The four facts a reader needs to act, and no fifth.
        fn report(&self) -> String {
            format!(
                "{}\n    heading   `{}`\n    engine    {}\n    renderer  {}",
                self.document, self.heading, self.ours, self.theirs
            )
        }
    }

    /// `docs_dir` and the `exclude_docs` patterns, as `mkdocs.yml` declares
    /// them.
    ///
    /// Read by hand rather than loaded as YAML because `mkdocs.yml` carries
    /// Python object tags that no loader in this workspace accepts. Both keys
    /// must be found: a rename that left this reading nothing would compare an
    /// empty corpus and report a pass.
    fn site_declaration(root: &Path) -> (PathBuf, Vec<String>) {
        let source = std::fs::read_to_string(root.join("mkdocs.yml"))
            .expect("`mkdocs.yml`, which declares what the site renders");
        let mut docs_dir: Option<String> = None;
        let mut exclude: Vec<String> = Vec::new();
        let mut in_block = false;
        for line in source.lines() {
            if let Some(rest) = line.strip_prefix("docs_dir:") {
                docs_dir = Some(rest.trim().to_string());
            }
            if in_block {
                if line.starts_with(' ') || line.starts_with('\t') {
                    let pattern = line.trim();
                    if !pattern.is_empty() && !pattern.starts_with('#') {
                        exclude.push(pattern.to_string());
                    }
                    continue;
                }
                in_block = false;
            }
            if line.starts_with("exclude_docs:") {
                in_block = true;
            }
        }
        let docs_dir = docs_dir.expect(
            "`mkdocs.yml` declares no `docs_dir`, so what the site renders is not stated \
             where this reads it",
        );
        assert!(
            !exclude.is_empty(),
            "`mkdocs.yml` declares no `exclude_docs` pattern this reads, so this would \
             compare files the site never serves"
        );
        (root.join(docs_dir), exclude)
    }

    /// Whether `exclude_docs` names a path, which is given relative to
    /// `docs_dir` with `/` separators.
    ///
    /// A pattern that ends in `/` names a directory and excludes everything
    /// under a directory of that name at any depth, which is the gitignore
    /// shape MkDocs matches these with. Anything else names a file.
    ///
    /// A pattern with a `/` before its end is anchored at `docs_dir`, as
    /// gitignore anchors one, and `*` in it stands for one whole path
    /// component. That is the shape `taxonomies/*/fixtures/` takes since
    /// #350 served the library's doctrine and kept its fixtures out.
    fn is_excluded(relative: &str, patterns: &[String]) -> bool {
        patterns.iter().any(|pattern| {
            let body = pattern.strip_suffix('/').unwrap_or(pattern);
            if body.contains('/') {
                return is_excluded_anchored(relative, body, pattern.ends_with('/'));
            }
            match pattern.strip_suffix('/') {
                Some(directory) => relative
                    .split('/')
                    .rev()
                    .skip(1)
                    .any(|component| component == directory),
                // The bare name matches at any depth and the whole path
                // matches once, which is how `LICENSE` reaches
                // `docs/LICENSE`.
                None => {
                    relative == pattern || relative.rsplit('/').next() == Some(pattern.as_str())
                }
            }
        })
    }

    /// Whether an anchored pattern names `relative`: each of its components
    /// is a literal or `*`, and it matches the leading components of the
    /// path. A directory pattern needs a component of the path beyond it, and
    /// a file pattern needs the whole path.
    fn is_excluded_anchored(relative: &str, pattern: &str, directory: bool) -> bool {
        let wanted: Vec<&str> = pattern.split('/').collect();
        let path: Vec<&str> = relative.split('/').collect();
        let fits = if directory {
            path.len() > wanted.len()
        } else {
            path.len() == wanted.len()
        };
        fits && wanted
            .iter()
            .zip(&path)
            .all(|(want, got)| *want == "*" || want == got)
    }

    #[test]
    fn an_anchored_exclude_docs_pattern_names_one_component_per_star() {
        let patterns = vec![
            "taxonomies/*/fixtures/".to_string(),
            "taxonomies/*/bundle.yml".to_string(),
        ];
        assert!(is_excluded(
            "taxonomies/design-spec/fixtures/corpus/a.md",
            &patterns
        ));
        assert!(is_excluded("taxonomies/design-spec/bundle.yml", &patterns));
        assert!(!is_excluded(
            "taxonomies/design-spec/doctrine.md",
            &patterns
        ));
        assert!(!is_excluded("taxonomies/README.md", &patterns));
        assert!(!is_excluded("spec/fixtures/a.md", &patterns));
        assert!(!is_excluded("taxonomies/a/b/fixtures/c.md", &patterns));
    }

    /// Every Markdown file under a directory, in path order.
    fn markdown(dir: &Path, out: &mut Vec<PathBuf>) {
        let mut entries: Vec<PathBuf> = std::fs::read_dir(dir)
            .expect("the directory the site renders")
            .filter_map(Result::ok)
            .map(|entry| entry.path())
            .collect();
        entries.sort();
        for path in entries {
            if path.is_dir() {
                markdown(&path, out);
            } else if path.extension().is_some_and(|it| it == "md") {
                out.push(path);
            }
        }
    }

    /// The headings of the documents the site renders.
    ///
    /// The front matter is split off first, for the reason
    /// `comment_links::anchors_of` states: a `---` fence read as body gives the
    /// line above it a setext heading that no renderer produces.
    fn population(root: &Path) -> Population {
        let (docs_dir, patterns) = site_declaration(root);
        let mut files = Vec::new();
        markdown(&docs_dir, &mut files);

        let mut headings = Vec::new();
        let mut documents = 0usize;
        let mut excluded = 0usize;
        for file in files {
            let relative = file
                .strip_prefix(&docs_dir)
                .expect("a file this walk found under `docs_dir`")
                .to_string_lossy()
                .replace('\\', "/");
            if is_excluded(&relative, &patterns) {
                excluded += 1;
                continue;
            }
            documents += 1;
            let reported = file
                .strip_prefix(root)
                .unwrap_or(&file)
                .to_string_lossy()
                .replace('\\', "/");
            let source = std::fs::read_to_string(&file).expect("a document the site renders");
            let body = match headwater_doc::split::split(&source) {
                Ok(split) => headwater_doc::body::scan(&source, split.body, split.body_offset),
                Err(_) => headwater_doc::body::scan(&source, &source, 0),
            };
            for heading in body.headings() {
                headings.push((reported.clone(), heading.text()));
            }
        }
        Population {
            headings,
            documents,
            excluded,
        }
    }

    /// What the named slugger makes of each heading, or why it could not be
    /// asked.
    fn renderer(which: &str, texts: &[String]) -> Result<Vec<String>, String> {
        let mut child = Command::new("python3")
            .arg("-c")
            .arg(SLUGGERS)
            .arg(which)
            .stdin(Stdio::piped())
            .stdout(Stdio::piped())
            .stderr(Stdio::piped())
            .spawn()
            .map_err(|error| format!("python3 did not run: {error}"))?;
        let payload = texts.join("\0").into_bytes();
        let mut stdin = child.stdin.take().expect("a piped standard input");
        // Written from a second thread, and not inline. This corpus's headings
        // are larger than a pipe buffer, so a writer that filled one while
        // this thread waited on the reader would deadlock both halves.
        let writing = std::thread::spawn(move || stdin.write_all(&payload));
        let ran = child
            .wait_with_output()
            .map_err(|error| format!("python3 did not finish: {error}"))?;
        let _ = writing.join();
        if !ran.status.success() {
            return Err(format!(
                "python3 exited {}: {}",
                ran.status,
                String::from_utf8_lossy(&ran.stderr).trim()
            ));
        }
        let text = String::from_utf8(ran.stdout)
            .map_err(|error| format!("python3 wrote bytes this cannot read: {error}"))?;
        let slugs: Vec<String> = if text.is_empty() {
            Vec::new()
        } else {
            text.split('\0').map(str::to_string).collect()
        };
        if slugs.len() != texts.len() {
            return Err(format!(
                "{} headings were sent and {} slugs came back, so no heading here is \
                 held against the slug that belongs to it",
                texts.len(),
                slugs.len()
            ));
        }
        Ok(slugs)
    }

    /// Every heading the named slugger and [`slug`] answer differently.
    fn disagreements(which: &str, population: &Population) -> Result<Vec<Disagreement>, String> {
        let texts: Vec<String> = population
            .headings
            .iter()
            .map(|(_, text)| text.clone())
            .collect();
        let theirs = renderer(which, &texts)?;
        Ok(population
            .headings
            .iter()
            .zip(theirs)
            .filter_map(|((document, heading), their)| {
                let ours = slug(heading);
                (ours != their).then(|| Disagreement {
                    document: document.clone(),
                    heading: heading.clone(),
                    ours,
                    theirs: their,
                })
            })
            .collect())
    }

    /// The population, and the denominators that say it is one.
    ///
    /// A run that compared nothing must not read as a run that found nothing,
    /// which is the assertion `tools/site/site-fragments-fixtures.sh` makes of
    /// the site checker in the same words. Each figure is held against zero
    /// rather than against a literal, because a literal would be a measurement
    /// of one day committed as a contract.
    fn measured(root: &Path) -> Population {
        let population = population(root);
        assert!(
            population.documents > 0,
            "no document under `docs_dir` survived the `exclude_docs` patterns, so this \
             compared nothing"
        );
        assert!(
            !population.headings.is_empty(),
            "{} documents carried no heading between them, so this compared nothing",
            population.documents
        );
        assert!(
            population.excluded > 0,
            "`exclude_docs` excluded none of the {} files under `docs_dir`, so either the \
             declaration in `mkdocs.yml` moved or this stopped reading it, and material \
             the site never renders is being held to the site's rule",
            population.documents
        );
        eprintln!(
            "note: {} headings across {} rendered documents, {} files excluded by \
             `exclude_docs`",
            population.headings.len(),
            population.documents,
            population.excluded
        );
        population
    }

    /// Every heading the site renders takes the same slug under both rules.
    #[test]
    fn the_renderer_slugs_every_rendered_heading_as_this_engine_does() {
        let required = std::env::var_os("HEADWATER_SLUG_ORACLE").is_some();
        let population = measured(&repository_root());

        let found = match disagreements("pymdownx", &population) {
            Ok(found) => found,
            Err(reason) => {
                assert!(
                    !required,
                    "HEADWATER_SLUG_ORACLE is set and the site's slugger did not run, so \
                     nothing holds this engine's heading anchors against the ones the site \
                     serves: {reason}"
                );
                eprintln!(
                    "note: the site's slugger did not run, so this slug rule is held only \
                     against its own cases. Install `pymdown-extensions`, or set \
                     HEADWATER_SLUG_ORACLE to make its absence a failure.\n{reason}"
                );
                return;
            }
        };

        assert!(
            found.is_empty(),
            "the engine's slug rule and the site's disagree on {} of {} rendered \
             headings:\n{}",
            found.len(),
            population.headings.len(),
            found
                .iter()
                .map(Disagreement::report)
                .collect::<Vec<_>>()
                .join("\n")
        );
    }

    /// The same comparison, pointed at a slugger this corpus is not rendered
    /// with, names the document, the heading and both slugs.
    ///
    /// This is the failing fixture the comparison above ships with. It provokes
    /// the one thing #545 exists to catch — a slugger that moved under a corpus
    /// nobody edited — and it holds the report to the diagnosis rather than to
    /// a count of dead links downstream of it.
    #[test]
    fn a_slugger_this_corpus_is_not_rendered_with_is_named_heading_by_heading() {
        let required = std::env::var_os("HEADWATER_SLUG_ORACLE").is_some();
        let population = measured(&repository_root());

        let found = match disagreements("default", &population) {
            Ok(found) => found,
            Err(reason) => {
                assert!(
                    !required,
                    "HEADWATER_SLUG_ORACLE is set and Python-Markdown's own slugger did \
                     not run, so the comparison above has not been seen refusing \
                     anything: {reason}"
                );
                eprintln!(
                    "note: Python-Markdown's slugger did not run, so the comparison above \
                     has not been seen refusing anything here. Install \
                     `pymdown-extensions`, or set HEADWATER_SLUG_ORACLE to make its \
                     absence a failure.\n{reason}"
                );
                return;
            }
        };

        assert!(
            !found.is_empty(),
            "Python-Markdown's own slugger agreed with this engine on all {} rendered \
             headings, which no corpus of this one's shape does. The comparison has \
             stopped comparing, or it is reading a slugger it did not ask for",
            population.headings.len()
        );

        let first = &found[0];
        assert!(
            first.document.ends_with(".md"),
            "a disagreement names no document: {}",
            first.report()
        );
        assert!(
            !first.heading.is_empty(),
            "a disagreement names no heading: {}",
            first.report()
        );
        assert!(
            !first.ours.is_empty() && first.ours != first.theirs,
            "a disagreement carries no pair of slugs to compare: {}",
            first.report()
        );
        let report = first.report();
        for fact in [
            first.document.as_str(),
            first.heading.as_str(),
            first.ours.as_str(),
            first.theirs.as_str(),
        ] {
            assert!(
                report.contains(fact),
                "the report drops one of the four facts it exists to carry: {report}"
            );
        }
    }
}

#[cfg(test)]
mod tests {
    use super::*;
    use headwater_doc::body::scan;

    #[test]
    fn an_em_dash_leaves_the_spaces_around_it() {
        assert_eq!(
            slug("Q5 — Voice checking depth"),
            "q5--voice-checking-depth"
        );
        assert_eq!(slug("`$package.optional`"), "packageoptional");
        assert_eq!(
            slug("Two phases, and why the order matters"),
            "two-phases-and-why-the-order-matters"
        );
    }

    /// The anchors of a document that is these headings and nothing else.
    fn anchors_of(source: &str) -> Vec<String> {
        anchors(&scan(source, source, 0))
    }

    #[test]
    fn a_repeated_heading_takes_a_numeric_suffix() {
        let body = scan("# One\n\n# One\n\n# One\n", "# One\n\n# One\n\n# One\n", 0);
        assert_eq!(anchors(&body), ["one", "one-1", "one-2"]);
    }

    /// A longer heading is not a repeat of the shorter one its slug extends.
    ///
    /// Three identical headings cannot see this rule, because counting an
    /// exact repeat and counting a hyphen-prefixed one give the same three
    /// anchors. The case below is the shape that separates them, and
    /// `docs/evaluations/schema-format-walkthrough.md` carries it: the first
    /// heading is a prefix extension of the two that follow, so a rule that
    /// counted it moved every later suffix up by one.
    ///
    /// GitHub's `/markdown` endpoint returns these three identifiers for this
    /// input.
    #[test]
    fn a_longer_heading_is_not_a_repeat_of_the_one_it_extends() {
        assert_eq!(
            anchors_of("## Edit sites, as spec 2 stands\n\n## Edit sites\n\n## Edit sites\n"),
            ["edit-sites-as-spec-2-stands", "edit-sites", "edit-sites-1"]
        );
    }

    /// A candidate an earlier heading took is stepped past, and the count
    /// resumes from where the search stopped.
    ///
    /// This is the half that counting exact repeats alone still gets wrong.
    /// The third heading wants `one-1`, which the second heading holds, so it
    /// takes `one-1-1`. The fourth heading is the second repeat of `one` and
    /// takes `one-2`, which no rule reading only the anchors already issued
    /// arrives at. GitHub's `/markdown` endpoint returns these four.
    #[test]
    fn a_suffix_an_earlier_heading_took_is_stepped_past() {
        assert_eq!(
            anchors_of("## One\n\n## One\n\n## One-1\n\n## One\n"),
            ["one", "one-1", "one-1-1", "one-2"]
        );
    }

    /// Two headings that differ never receive one anchor.
    ///
    /// The review question this answers is whether two states that must differ
    /// can produce one key. Counting a hyphen-prefixed anchor as a repeat
    /// could: on this input it gave the first two headings `one-1` each, so a
    /// link to `#one-1` named two places and the second was unreachable.
    /// GitHub's `/markdown` endpoint returns `one-1`, `one`, `one-2`.
    #[test]
    fn two_headings_that_differ_never_share_an_anchor() {
        let anchors = anchors_of("## One-1\n\n## One\n\n## One\n");
        assert_eq!(anchors, ["one-1", "one", "one-2"]);
        let issued: std::collections::HashSet<&String> = anchors.iter().collect();
        assert_eq!(issued.len(), anchors.len(), "an anchor is issued once");
    }
}

#[cfg(test)]
mod arms {
    use super::*;
    use crate::scope::CorpusView;
    use headwater_doc::LinkForm;
    use headwater_yaml::{Position, Span};

    const CITER: &str = "docs/spec/01-conceptual-model.md";
    const TARGET: &str = "docs/spec/glossary.md";

    fn span() -> Span {
        Span {
            start: Position {
                line: 30,
                col: 5,
                offset: 0,
            },
            end: Position {
                line: 30,
                col: 6,
                offset: 1,
            },
        }
    }

    fn link(destination: &str, fragment: Option<&str>, binding: Binding) -> Link {
        Link {
            source_path: CITER.to_string(),
            destination: destination.to_string(),
            fragment: fragment.map(str::to_string),
            form: LinkForm::Inline,
            span: span(),
            binding,
        }
    }

    fn corpus(path: &str) -> Binding {
        Binding::Corpus {
            path: path.to_string(),
            class: "typed",
            id: None,
        }
    }

    fn index() -> Anchors {
        Anchors::of_pairs(&[
            (CITER, &["a-heading-of-the-citer"]),
            (TARGET, &["projection"]),
        ])
    }

    /// The defect this rule was widened for. The finding is written at the
    /// citing document and line, and it names the target rather than saying
    /// "this document", which would send an author to the wrong file.
    #[test]
    fn a_fragment_that_names_no_heading_of_the_target_is_an_error_at_the_citing_line() {
        let found = finding(
            &link(
                "glossary.md#projections",
                Some("projections"),
                corpus(TARGET),
            ),
            &index(),
        )
        .expect("a finding");
        assert_eq!(found.rule, self::RULE);
        assert_eq!(found.severity, Severity::Error);
        assert_eq!(found.path, CITER);
        assert_eq!(found.line, 30);
        assert_eq!(found.column, 5);
        assert!(found.message.contains(TARGET), "{found:#?}");
        assert!(!found.message.contains("this document"), "{found:#?}");
        assert!(!found.fixable(), "{found:#?}");
    }

    /// The near arm keeps its own sentence, and it names no path.
    #[test]
    fn the_near_arm_says_this_document_and_the_far_arm_names_the_file() {
        let near = finding(
            &link(
                "#no-such-heading",
                Some("no-such-heading"),
                Binding::SameDocument,
            ),
            &index(),
        )
        .expect("a finding");
        assert!(near.message.contains("this document"), "{near:#?}");
        assert!(!near.message.contains(TARGET), "{near:#?}");
        assert_eq!(near.path, CITER);
    }

    /// A fragment that resolves is nothing to this rule, on both arms.
    #[test]
    fn a_fragment_that_resolves_is_not_a_finding_on_either_arm() {
        assert!(finding(
            &link("glossary.md#projection", Some("projection"), corpus(TARGET)),
            &index()
        )
        .is_none());
        assert!(finding(
            &link(
                "#a-heading-of-the-citer",
                Some("a-heading-of-the-citer"),
                Binding::SameDocument
            ),
            &index()
        )
        .is_none());
    }

    /// The comparison is exact, which is what a browser does.
    ///
    /// What a renderer *writes* and what a browser *resolves* are two
    /// questions, and this rule answers the second one. [`anchors`] case folds
    /// the heading text because GitHub's slugger does; the citation is compared
    /// to the result byte for byte, because the HTML standard's "find a
    /// potential indicated element" matches an `id` exactly and falls back to
    /// an `a` element's `name` exactly, and neither step case folds. Chrome 153
    /// was driven against a local page carrying `id="edit-sites"` to settle it
    /// rather than to read it: `#edit-sites` scrolled the document to 5020 and
    /// `:target` named the heading, while `#Edit-Sites` and `#EDIT-SITES` each
    /// left the scroll at 0 with no `:target` at all. The `name` fallback
    /// answered the same way. So a citation that differs from its heading only
    /// in capitalization is a dead link, and folding it here passed one.
    ///
    /// Both arms carry a case, because the two comparisons are separate lines
    /// and one of them was already changed without the other (#206).
    #[test]
    fn a_fragment_that_differs_from_its_heading_only_in_case_is_a_finding() {
        let across = finding(
            &link("glossary.md#Projection", Some("Projection"), corpus(TARGET)),
            &index(),
        )
        .expect("a finding");
        assert!(across.message.contains(TARGET), "{across:#?}");

        let near = finding(
            &link(
                "#A-Heading-Of-The-Citer",
                Some("A-Heading-Of-The-Citer"),
                Binding::SameDocument,
            ),
            &index(),
        )
        .expect("a finding");
        assert!(near.message.contains("this document"), "{near:#?}");
    }

    /// The bindings this rule passes over, enumerated rather than sampled.
    ///
    /// Two belong to [`crate::link_path`], and two name a body this engine
    /// never read. A finding on any of the four would be a second rule
    /// reporting one defect, or a guess about bytes nobody opened.
    #[test]
    fn every_binding_that_is_not_this_rule_produces_nothing() {
        for binding in [
            Binding::Missing {
                path: "docs/spec/gone.md".to_string(),
            },
            Binding::Unnormalizable {
                why: "climbs above the repository root".to_string(),
            },
            Binding::Repository {
                path: "CLAUDE.md".to_string(),
            },
            Binding::External,
        ] {
            assert!(finding(&link("x#y", Some("y"), binding), &index()).is_none());
        }
    }

    /// A link with no fragment, and one with an empty fragment, are both
    /// nothing to this rule. `split_fragment` gives `Some("")` for a
    /// destination that ends in a bare `#`, so the emptiness has to be tested
    /// rather than the option.
    #[test]
    fn a_link_carrying_no_fragment_is_not_this_rules_business() {
        assert!(finding(&link("glossary.md", None, corpus(TARGET)), &index()).is_none());
        assert!(finding(&link("glossary.md#", Some(""), corpus(TARGET)), &index()).is_none());
    }

    /// The case the module comment records as passed over: a corpus path the
    /// census parsed no document out of. There is no heading list, so there is
    /// no verdict, and a finding would be a guess about a file nobody read.
    #[test]
    fn a_corpus_path_with_no_parsed_document_gets_no_verdict_rather_than_a_guess() {
        let untyped = "docs/spec/notes.txt";
        assert_eq!(index().resolves(untyped, "anything"), None);
        assert!(finding(
            &link("notes.txt#anything", Some("anything"), corpus(untyped)),
            &index()
        )
        .is_none());
    }

    /// A run whose scope admitted no links skips with the reason rather than
    /// passing. A pass would report a corpus this rule never read as clean.
    #[test]
    fn a_view_with_no_links_skips_rather_than_passes() {
        let view = CorpusView::only_links_and_anchors(None, None);
        assert!(matches!(Fragments.evaluate(&view), Outcome::Skipped(why) if why == NO_LINKS));
    }

    /// And a run that reached the links but no anchors says the other thing.
    /// Two absences that became one reason would leave an author guessing which
    /// half of the rule was unavailable.
    #[test]
    fn a_view_with_links_and_no_anchors_skips_with_the_other_reason() {
        let links = vec![link(
            "glossary.md#projection",
            Some("projection"),
            corpus(TARGET),
        )];
        let view = CorpusView::only_links_and_anchors(Some(&links), None);
        assert!(matches!(Fragments.evaluate(&view), Outcome::Skipped(why) if why == NO_ANCHORS));
    }

    /// The whole set, and one finding per unresolved member of it. A corpus
    /// whose fragments all resolve passes, which is what lets this rule be an
    /// error at all.
    #[test]
    fn the_unresolved_fragments_of_the_view_become_the_findings() {
        let anchors = index();
        let broken = vec![
            link(
                "glossary.md#projections",
                Some("projections"),
                corpus(TARGET),
            ),
            link("#nope", Some("nope"), Binding::SameDocument),
            link("glossary.md#projection", Some("projection"), corpus(TARGET)),
            link("https://example.com", None, Binding::External),
        ];
        let view = CorpusView::only_links_and_anchors(Some(&broken), Some(&anchors));
        let Outcome::Failed(found) = Fragments.evaluate(&view) else {
            panic!("two unresolved fragments are two findings");
        };
        assert_eq!(found.len(), 2, "{found:#?}");

        let clean = vec![link(
            "glossary.md#projection",
            Some("projection"),
            corpus(TARGET),
        )];
        let view = CorpusView::only_links_and_anchors(Some(&clean), Some(&anchors));
        assert!(matches!(Fragments.evaluate(&view), Outcome::Passed));
    }
}