Skip to main content

links_notation/
comments.rs

1//! Comments, and the rule that decides where one starts.
2//!
3//! A comment starts at a `#` written where a line or a token starts and runs to
4//! the end of the line. The parser never sees it: [`strip_comments`] replaces
5//! every byte of every comment with a space before the document is parsed, so
6//! the notation itself stays exactly as it was and a position reported by the
7//! parser still points at the same character of the document the caller wrote.
8
9use crate::quotes::{DelimitedReferences, QUOTES};
10
11/// The character that opens a comment.
12pub const COMMENT: char = '#';
13
14/// What can stand before a delimited reference: the reference is the first
15/// thing on a line, follows a space, opens a group or follows a colon.
16const BEFORE_REFERENCE: [u8; 6] = *b" \t\n\r(:";
17
18/// What can stand before a comment: the comment is the first thing on a line,
19/// or it follows whitespace. A `#` inside a word is part of that word, so
20/// `issue#1047` is a reference and not the start of a comment.
21const BEFORE_COMMENT: [u8; 4] = *b" \t\n\r";
22
23/// Blanks out every comment in `document`, keeping every other byte where it
24/// was.
25///
26/// Comments are replaced rather than removed so that a byte offset in the
27/// result is the same byte offset in `document`: the line and column a parse
28/// error reports are the line and column the writer sees in their file.
29///
30/// A `#` inside a delimited reference is content, so `"# not a comment"` is
31/// still one reference.
32///
33/// # Examples
34/// ```
35/// use links_notation::comments::strip_comments;
36///
37/// assert_eq!(strip_comments("a: b # why\n"), "a: b      \n");
38/// assert_eq!(strip_comments("\"# kept\"\n"), "\"# kept\"\n");
39/// assert_eq!(strip_comments("issue#1047\n"), "issue#1047\n");
40/// ```
41pub fn strip_comments(document: &str) -> String {
42    let mut bytes = document.as_bytes().to_vec();
43    let references = DelimitedReferences::new(document);
44    let mut position = 0;
45
46    while position < bytes.len() {
47        let byte = bytes[position];
48
49        if QUOTES.contains(&byte) && follows(&bytes, position, &BEFORE_REFERENCE) {
50            match references.end_at(document, position) {
51                Some(end) => position = end,
52                None => position += 1,
53            }
54            continue;
55        }
56
57        if byte == COMMENT as u8 && follows(&bytes, position, &BEFORE_COMMENT) {
58            while position < bytes.len() && bytes[position] != b'\n' && bytes[position] != b'\r' {
59                bytes[position] = b' ';
60                position += 1;
61            }
62            continue;
63        }
64
65        position += 1;
66    }
67
68    // Only whole comments were replaced, and only by spaces, so what is left is
69    // the document it was read from with some of its bytes blanked.
70    String::from_utf8(bytes).expect("blanking comment bytes keeps the document valid UTF-8")
71}
72
73/// Reports whether the byte before `position` is one of `allowed`, treating the
74/// start of the document as one of them.
75fn follows(bytes: &[u8], position: usize, allowed: &[u8]) -> bool {
76    match position {
77        0 => true,
78        _ => allowed.contains(&bytes[position - 1]),
79    }
80}