links_notation/comments.rs
1//! Comments, and the rule that decides where one starts.
2//!
3//! A comment starts at a `#` written where a line or a token starts and runs to
4//! the end of the line. The parser never sees it: [`strip_comments`] replaces
5//! every byte of every comment with a space before the document is parsed, so
6//! the notation itself stays exactly as it was and a position reported by the
7//! parser still points at the same character of the document the caller wrote.
8
9use crate::quotes::{DelimitedReferences, QUOTES};
10
11/// The character that opens a comment.
12pub const COMMENT: char = '#';
13
14/// What can stand before a delimited reference: the reference is the first
15/// thing on a line, follows a space, opens a group or follows a colon.
16const BEFORE_REFERENCE: [u8; 6] = *b" \t\n\r(:";
17
18/// What can stand before a comment: the comment is the first thing on a line,
19/// or it follows whitespace. A `#` inside a word is part of that word, so
20/// `issue#1047` is a reference and not the start of a comment.
21const BEFORE_COMMENT: [u8; 4] = *b" \t\n\r";
22
23/// Blanks out every comment in `document`, keeping every other byte where it
24/// was.
25///
26/// Comments are replaced rather than removed so that a byte offset in the
27/// result is the same byte offset in `document`: the line and column a parse
28/// error reports are the line and column the writer sees in their file.
29///
30/// A `#` inside a delimited reference is content, so `"# not a comment"` is
31/// still one reference.
32///
33/// # Examples
34/// ```
35/// use links_notation::comments::strip_comments;
36///
37/// assert_eq!(strip_comments("a: b # why\n"), "a: b \n");
38/// assert_eq!(strip_comments("\"# kept\"\n"), "\"# kept\"\n");
39/// assert_eq!(strip_comments("issue#1047\n"), "issue#1047\n");
40/// ```
41pub fn strip_comments(document: &str) -> String {
42 let mut bytes = document.as_bytes().to_vec();
43 let references = DelimitedReferences::new(document);
44 let mut position = 0;
45
46 while position < bytes.len() {
47 let byte = bytes[position];
48
49 if QUOTES.contains(&byte) && follows(&bytes, position, &BEFORE_REFERENCE) {
50 match references.end_at(document, position) {
51 Some(end) => position = end,
52 None => position += 1,
53 }
54 continue;
55 }
56
57 if byte == COMMENT as u8 && follows(&bytes, position, &BEFORE_COMMENT) {
58 while position < bytes.len() && bytes[position] != b'\n' && bytes[position] != b'\r' {
59 bytes[position] = b' ';
60 position += 1;
61 }
62 continue;
63 }
64
65 position += 1;
66 }
67
68 // Only whole comments were replaced, and only by spaces, so what is left is
69 // the document it was read from with some of its bytes blanked.
70 String::from_utf8(bytes).expect("blanking comment bytes keeps the document valid UTF-8")
71}
72
73/// Reports whether the byte before `position` is one of `allowed`, treating the
74/// start of the document as one of them.
75fn follows(bytes: &[u8], position: usize, allowed: &[u8]) -> bool {
76 match position {
77 0 => true,
78 _ => allowed.contains(&bytes[position - 1]),
79 }
80}