uncomment 3.10.0

A CLI tool to remove comments from code using tree-sitter for accurate parsing
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
use crate::languages::handlers::CommentNodeVerdict;
use crate::languages::{LanguageHandler, get_handler};
use crate::rules::preservation::PreservationRule;
use tree_sitter::Node;

#[derive(Debug, Clone, PartialEq, Eq)]
pub struct CommentInfo {
    pub start_byte: usize,
    pub end_byte: usize,
    pub start_row: usize,
    pub end_row: usize,
    pub node_type: String,
    pub should_preserve: bool,
    pub is_documentation: bool,
}

impl CommentInfo {
    #[must_use]
    pub fn new(node: Node) -> Self {
        Self {
            start_byte: node.start_byte(),
            end_byte: node.end_byte(),
            start_row: node.start_position().row,
            end_row: node.end_position().row,
            node_type: node.kind().to_string(),
            should_preserve: false,
            is_documentation: false,
        }
    }

    #[must_use]
    pub const fn with_documentation(mut self, is_documentation: bool) -> Self {
        self.is_documentation = is_documentation;
        self
    }

    #[must_use]
    pub const fn with_preservation(mut self, should_preserve: bool) -> Self {
        self.should_preserve = should_preserve;
        self
    }

    /// Narrow this comment to `start_byte..end_byte`, a range inside the node it was built from,
    /// recomputing the rows so a caller reporting line numbers still sees the truth.
    ///
    /// Used where a grammar's comment node carries more than the comment — markdown's `html_block`
    /// holds the comment's indentation and its trailing newline — so that removal expands from the
    /// comment's own bounds rather than the node's.
    ///
    /// A range that is not a valid slice of `source` — reversed, out of bounds, or landing inside a
    /// multi-byte character — is ignored rather than allowed to panic the visit, leaving the comment
    /// on the node's own span. [`LanguageHandler`] is a public extension point, so a handler that
    /// miscalculates must not be able to bring down a run over unrelated files.
    #[must_use]
    fn narrowed(mut self, start_byte: usize, end_byte: usize, source: &str) -> Self {
        let (Some(skipped), Some(kept)) = (
            source.get(self.start_byte..start_byte),
            source.get(start_byte..end_byte),
        ) else {
            return self;
        };
        let newlines = |range: &str| range.bytes().filter(|&byte| byte == b'\n').count();
        self.start_row += newlines(skipped);
        self.end_row = self.start_row + newlines(kept);
        self.start_byte = start_byte;
        self.end_byte = end_byte;
        self
    }

    /// Grow this comment's span to `start..end`, which must contain the span it currently holds.
    ///
    /// The mirror image of [`Self::narrowed`], and rows are corrected the same way: by counting line
    /// breaks across the two added stretches only, which are a delimiter wide, so this stays cheap
    /// where re-deriving a row from the file start would not. The widened text becomes the comment's
    /// [`Self::content`], which is what the whole preservation path reads: a `~keep` inside the
    /// comment still reads as one, and a delimiter that is now part of the text lets
    /// [`crate::processor::CommentKind`] name the shape correctly.
    ///
    /// A range that is not a valid slice of `source` is ignored rather than allowed to panic the
    /// visit, on the same reasoning as [`Self::narrowed`]: [`crate::languages::handlers::LanguageHandler`]
    /// is a public extension point, so a handler that miscalculates must not bring down a run.
    #[must_use]
    fn widened(mut self, start: usize, end: usize, source: &str) -> Self {
        let newlines = |range: &str| range.bytes().filter(|&byte| byte == b'\n').count();

        if let Some(prefix) = source.get(start..self.start_byte) {
            self.start_row = self.start_row.saturating_sub(newlines(prefix));
            self.start_byte = start;
        }
        if let Some(suffix) = source.get(self.end_byte..end) {
            self.end_row += newlines(suffix);
            self.end_byte = end;
        }
        self
    }

    /// Extract comment content from source by byte range.
    #[inline]
    pub fn content<'a>(&self, source: &'a str) -> &'a str {
        &source[self.start_byte..self.end_byte]
    }
}

/// What the visit pass concluded about one comment, recorded per comment so a caller
/// can name *why* it survived. Captured before the `~keep` extension passes run, so a
/// comment kept only by a neighbour's marker has no `matched_rule` here — see
/// [`CommentVisitor::is_extended`].
#[derive(Debug, Clone, Copy, Default)]
pub struct VisitDecision<'a> {
    /// First preservation rule whose test the comment's own text passed.
    pub matched_rule: Option<&'a PreservationRule>,
    /// Whether the grammar handler forced preservation from surrounding syntax — a Go
    /// build/embed directive, a cgo preamble, a trailing preprocessor comment, a Ruby
    /// magic comment.
    pub forced_by_grammar: bool,
}

pub struct CommentVisitor<'a> {
    source: &'a str,
    preservation_rules: &'a [PreservationRule],
    comments: Vec<CommentInfo>,
    /// Parallel to `comments`: why each one was kept, as decided during the visit.
    decisions: Vec<VisitDecision<'a>>,
    comment_node_types: &'a [String],
    doc_comment_node_types: &'a [String],
    language_handler: Box<dyn LanguageHandler>,
    /// Indices of comments preserved *by* a `~keep` on a different comment
    /// (block extension or an above-line marker), rather than by a marker of
    /// their own. Used to tell a load-bearing marker from a redundant one.
    extended: std::collections::HashSet<usize>,
}

impl<'a> CommentVisitor<'a> {
    #[must_use]
    pub fn new_with_language(
        source: &'a str,
        preservation_rules: &'a [PreservationRule],
        comment_node_types: &'a [String],
        doc_comment_node_types: &'a [String],
        language_name: &str,
    ) -> Self {
        let language_handler = get_handler(language_name);
        Self {
            source,
            preservation_rules,
            comments: Vec::with_capacity(32),
            decisions: Vec::with_capacity(32),
            comment_node_types,
            doc_comment_node_types,
            language_handler,
            extended: std::collections::HashSet::new(),
        }
    }

    pub fn visit_node(&mut self, node: Node) {
        self.visit_node_recursive(node, None);
    }

    fn visit_node_recursive(&mut self, node: Node, parent: Option<Node>) {
        if let Some((start_byte, end_byte)) = self.comment_span(&node, parent) {
            let mut comment_info = CommentInfo::new(node);
            if (start_byte, end_byte) != (node.start_byte(), node.end_byte()) {
                comment_info = comment_info.narrowed(start_byte, end_byte, self.source);
            }

            // A grammar may model a comment's own delimiters as siblings rather than as part of the
            // comment, in which case deleting the comment node alone leaves a syntax fragment
            // behind. The handler reports the span that actually has to go.
            if let Some((start, end)) = self.language_handler.removal_span(&node, self.source) {
                comment_info = comment_info.widened(start, end, self.source);
            }

            if let Some(is_doc) = self
                .language_handler
                .is_documentation_comment(&node, parent, self.source)
            {
                comment_info = comment_info.with_documentation(is_doc);
            }

            let forced_by_grammar = self
                .language_handler
                .should_preserve_comment(&node, parent, self.source)
                .unwrap_or(false);

            let content = comment_info.content(self.source);
            let matched_rule = self.matching_preservation_rule(&comment_info, content);
            let should_preserve = forced_by_grammar || matched_rule.is_some();
            let comment_with_preservation = comment_info.with_preservation(should_preserve);
            self.comments.push(comment_with_preservation);
            self.decisions.push(VisitDecision {
                matched_rule,
                forced_by_grammar,
            });
        }

        let mut cursor = node.walk();
        for child in node.children(&mut cursor) {
            self.visit_node_recursive(child, Some(node));
        }
    }

    #[must_use]
    pub fn get_comments_to_remove(&self) -> Vec<&CommentInfo> {
        self.comments
            .iter()
            .filter(|comment| !comment.should_preserve)
            .collect()
    }

    /// Every comment the visit collected, in discovery order — preserved and removable
    /// alike. Index into it to pair an entry with [`Self::decision`] and
    /// [`Self::is_extended`].
    #[must_use]
    pub fn comments(&self) -> &[CommentInfo] {
        &self.comments
    }

    /// What the visit pass concluded about the comment at `index`. An index past the
    /// end reports "nothing matched" rather than panicking.
    #[must_use]
    pub fn decision(&self, index: usize) -> VisitDecision<'a> {
        self.decisions.get(index).copied().unwrap_or_default()
    }

    /// Whether the comment at `index` owes its survival to a `~keep` carried by a
    /// *different* comment rather than one of its own.
    #[must_use]
    pub fn is_extended(&self, index: usize) -> bool {
        self.extended.contains(&index)
    }

    /// The byte range `node` contributes as a comment, or `None` when it is not one.
    ///
    /// The configured node kinds decide first, as they do for every grammar. A language handler may
    /// then reject the node or narrow the range — the escape hatch for a grammar whose comment kind
    /// is ambiguous, markdown's `html_block` being the case it exists for. A handler with no opinion
    /// leaves the node's own span, so this is a no-op for every other language.
    fn comment_span(&self, node: &Node, parent: Option<Node>) -> Option<(usize, usize)> {
        let kind = node.kind();
        let is_comment_kind = self.comment_node_types.iter().any(|node_type| node_type == kind);
        let is_doc_kind = self.doc_comment_node_types.iter().any(|node_type| node_type == kind);

        if !is_comment_kind && !is_doc_kind {
            return None;
        }

        if !is_comment_kind
            && let Some(false) = self
                .language_handler
                .is_documentation_comment(node, parent, self.source)
        {
            return None;
        }

        match self.language_handler.classify_comment_node(node, parent, self.source) {
            Some(CommentNodeVerdict::Rejected) => None,
            Some(CommentNodeVerdict::Accepted { start_byte, end_byte }) => Some((start_byte, end_byte)),
            None => Some((node.start_byte(), node.end_byte())),
        }
    }

    /// The first configured rule whose test `content` passes, or `None` when no rule
    /// keeps this comment. Rule order is meaningful: the caller reports the winner as
    /// the reason, and `~keep` sits near the front so a marker is named as a marker
    /// rather than as whatever else the text happens to match.
    fn matching_preservation_rule(&self, comment: &CommentInfo, content: &str) -> Option<&'a PreservationRule> {
        self.preservation_rules
            .iter()
            .find(|rule| rule.matches(comment, content))
    }

    /// Extend `~keep` preservation across contiguous single-line comment blocks.
    ///
    /// A rationale comment often spans several consecutive `//` (or `#`, `--`, …)
    /// lines that tree-sitter models as one node *per line*, so a per-comment
    /// `~keep` would preserve only the marked line and strip the rest, gutting the
    /// block. This pass groups **standalone single-line comments on consecutive
    /// rows** into blocks and, when any line in a block carries `~keep`, preserves
    /// the whole block.
    ///
    /// Scope is deliberately narrow so the behaviour is unsurprising:
    /// - Only `~keep` extends — other preservation rules (TODO, patterns,
    ///   directives) stay per-comment.
    /// - Only *standalone* comments join a block; a trailing comment (`code // x`)
    ///   never anchors or joins one, and never drags in the line below.
    /// - Only *single-line* comment nodes group; a `/* … */` block comment is
    ///   already one node, so `~keep` inside it preserves it without this pass.
    /// - A blank line (non-consecutive rows) or any code between comments ends the
    ///   block.
    ///
    /// The pass is purely additive: it only ever sets `should_preserve = true`,
    /// never clears it, so running it after the per-comment decisions is safe.
    pub fn extend_keep_blocks(&mut self) {
        // Standalone single-line comments, in source order.
        let mut indices: Vec<usize> = (0..self.comments.len())
            .filter(|&i| self.is_standalone_single_line(&self.comments[i]))
            .collect();
        indices.sort_by_key(|&i| self.comments[i].start_byte);

        let mut run_start = 0;
        while run_start < indices.len() {
            // Extend the run while the next comment sits on the immediately
            // following row (consecutive standalone single-line comments).
            let mut run_end = run_start;
            while run_end + 1 < indices.len()
                && self.comments[indices[run_end + 1]].start_row == self.comments[indices[run_end]].start_row + 1
            {
                run_end += 1;
            }

            let has_keep = indices[run_start..=run_end]
                .iter()
                .any(|&i| self.comments[i].content(self.source).contains("~keep"));
            if has_keep {
                for &i in &indices[run_start..=run_end] {
                    // A comment without a marker of its own owes its survival to
                    // a neighbour's marker, which makes that marker load-bearing.
                    if !self.comments[i].content(self.source).contains("~keep") {
                        self.extended.insert(i);
                    }
                    self.comments[i].should_preserve = true;
                }
            }

            run_start = run_end + 1;
        }
    }

    /// Extend `~keep` preservation from a marker comment down to the comment
    /// directly beneath it.
    ///
    /// `~keep` is plain comment text, so a marker written *inside* a doc comment
    /// is republished by every tool that consumes doc comments — rustdoc, OpenAPI
    /// schemas generated from `utoipa`, generated API clients, editor hover text.
    /// This pass gives the marker a home that never renders: a plain line comment
    /// on its own line, directly above the comment it protects.
    ///
    /// ```text
    /// // ~keep
    /// /// Parent element ID for hierarchical relationships.
    /// pub parent_id: Option<String>,
    /// ```
    ///
    /// [`Self::extend_keep_blocks`] already covers the case where both the marker
    /// and its target are standalone *single-line* comments. It cannot cover this
    /// one: a `///` node spans two rows (it swallows its trailing newline), so it
    /// never joins a single-line run. This pass matches on byte adjacency instead
    /// of row arithmetic, so it reaches doc comments and block comments alike.
    ///
    /// Scope mirrors the block pass:
    /// - Only a *standalone, non-documentation* comment acts as a marker, since a
    ///   doc comment carrying `~keep` is the very thing this exists to avoid.
    /// - Preservation runs forward through comments separated by nothing but
    ///   whitespace spanning at most one newline, so a blank line or any code
    ///   between comments ends the run.
    ///
    /// Purely additive: only ever sets `should_preserve = true`.
    pub fn extend_keep_above(&mut self) {
        let mut indices: Vec<usize> = (0..self.comments.len()).collect();
        indices.sort_by_key(|&i| (self.comments[i].start_byte, self.comments[i].end_byte));

        for position in 0..indices.len() {
            let marker = indices[position];
            if !self.is_keep_marker_line(&self.comments[marker]) {
                continue;
            }

            let mut previous_end = self.comments[marker].end_byte;
            for &next in &indices[position + 1..] {
                if !Self::gap_is_adjacent(
                    &self.source[previous_end.min(self.comments[next].start_byte)..self.comments[next].start_byte],
                ) {
                    break;
                }
                if !self.comments[next].should_preserve {
                    self.extended.insert(next);
                }
                self.comments[next].should_preserve = true;
                previous_end = previous_end.max(self.comments[next].end_byte);
            }
        }
    }

    /// Whether `comment` is a standalone, non-documentation comment whose text
    /// carries `~keep` — the form that may protect the comment below it.
    fn is_keep_marker_line(&self, comment: &CommentInfo) -> bool {
        let content = comment.content(self.source);
        !Self::is_doc_comment(comment, content) && self.is_standalone(comment) && content.contains("~keep")
    }

    /// Whether a comment is documentation, by the same content-based test the
    /// [`PreservationRule::Documentation`] rule applies. Grammar handlers leave
    /// [`CommentInfo::is_documentation`] unset for most languages — Rust records a
    /// `///` line as a plain `line_comment` — so the flag alone under-reports.
    pub(crate) fn is_doc_comment(comment: &CommentInfo, content: &str) -> bool {
        PreservationRule::documentation().matches(comment, content)
    }

    /// Whether the source between two comments separates them by nothing but the
    /// line break, so they read as one contiguous run. Any code, or a blank line
    /// (two or more newlines), ends the run.
    fn gap_is_adjacent(gap: &str) -> bool {
        gap.bytes().all(|byte| byte.is_ascii_whitespace()) && gap.bytes().filter(|&b| b == b'\n').count() <= 1
    }

    /// Byte ranges of `~keep` markers that sit inside a preserved documentation
    /// comment and do nothing there, together with the surrounding space that
    /// should collapse with them.
    ///
    /// Such a marker is redundant: doc comments are preserved anyway unless
    /// `--remove-doc` is set, so the token only travels outward into rendered
    /// documentation. Callers pass `remove_docs` so the one case where an in-doc
    /// marker *is* load-bearing — it is the only thing protecting this doc comment
    /// from `--remove-doc` — is left untouched.
    ///
    /// Ranges are deduplicated: a `///` line is commonly recorded twice, once as
    /// the outer comment node and once as its inner doc node.
    #[must_use]
    pub fn redundant_keep_markers(&self, remove_docs: bool) -> Vec<(usize, usize)> {
        if remove_docs {
            return Vec::new();
        }

        // A `///` line is commonly recorded twice: once as the outer comment node,
        // which still carries the `///`, and once as an inner doc node whose text
        // begins after it. The two disagree about where the doc marker ends, so a
        // span counts as a marker only when no node covering it objects.
        let mut ranges: Vec<(usize, usize)> = Vec::new();
        let mut rejected: std::collections::HashSet<usize> = std::collections::HashSet::new();
        for (index, comment) in self.comments.iter().enumerate() {
            let content = comment.content(self.source);
            if !comment.should_preserve || !Self::is_doc_comment(comment, content) {
                continue;
            }
            // A marker that is holding a neighbouring comment alive is doing work,
            // even from inside a doc comment. Leave it be.
            if self.run_members(index).any(|member| self.extended.contains(&member)) {
                continue;
            }
            if !content.contains("~keep") {
                continue;
            }
            for (offset, _) in content.match_indices("~keep") {
                let start = comment.start_byte + offset;
                if Self::is_marker_occurrence(content, offset) {
                    ranges.push(Self::widen_marker(self.source, start, start + "~keep".len()));
                } else {
                    rejected.insert(start);
                }
            }
        }

        ranges.retain(|&(start, end)| !rejected.contains(&(end - "~keep".len())) && !rejected.contains(&start));
        ranges.sort_unstable();
        ranges.dedup();
        ranges
    }

    /// Indices of comments that share a contiguous run with `index`, itself
    /// included — the comments whose survival a marker on `index` could explain.
    fn run_members(&self, index: usize) -> impl Iterator<Item = usize> + '_ {
        let anchor = &self.comments[index];
        let (low, high) = (anchor.start_row.saturating_sub(1), anchor.end_row + 1);
        (0..self.comments.len())
            .filter(move |&other| self.comments[other].start_row >= low && self.comments[other].start_row <= high)
    }

    /// Whether the `~keep` at `offset` in a doc comment is an actual marker
    /// rather than prose *about* the marker.
    ///
    /// Documentation that explains `~keep` — this crate's own docs, a README
    /// excerpt, a rustdoc example — mentions the token constantly, and rewriting
    /// those mentions would corrupt the very documentation this feature exists to
    /// protect. Three cheap signals separate the two:
    ///
    /// - A marker is a whole word. `~keepsake` is not one.
    /// - A line carrying a backtick is discussing the token, not using it, so
    ///   `` `~keep` `` and `` `/// ~keep Parent element ID.` `` are both left alone.
    /// - A `//` or `#` *inside* the doc text (past the doc marker itself) means the
    ///   line is a commented-out code sample, as in a rustdoc fenced block.
    ///
    /// The bias is deliberate: leaving a real marker in place is a cosmetic miss,
    /// while stripping a word out of prose is data loss.
    fn is_marker_occurrence(content: &str, offset: usize) -> bool {
        let end = offset + "~keep".len();
        let preceded_by_word = content[..offset]
            .chars()
            .next_back()
            .is_some_and(|c| !c.is_whitespace());
        let followed_by_word = content[end..].chars().next().is_some_and(|c| !c.is_whitespace());
        if preceded_by_word || followed_by_word {
            return false;
        }

        let line_start = content[..offset].rfind('\n').map_or(0, |pos| pos + 1);
        let line_end = content[offset..].find('\n').map_or(content.len(), |pos| offset + pos);
        let line = &content[line_start..line_end];
        if line.contains('`') {
            return false;
        }

        // Skip the doc marker that opens the line (`///`, `//!`, `##`, `/**`, `*`)
        // so only a comment marker *within* the documented text counts.
        let body = line.trim_start();
        let body_offset = line.len() - body.len();
        let text = body.trim_start_matches(['/', '!', '*', '#']);
        let text_start = line_start + body_offset + (body.len() - text.len());
        if text_start > offset {
            return false;
        }
        let before = &content[text_start..offset];
        !before.contains("//") && !before.contains('#')
    }

    /// Grow a `~keep` span to swallow one adjacent space, so stripping the token
    /// from `/// ~keep Parent element ID.` leaves `/// Parent element ID.` rather
    /// than a doubled space. Prefers the space after the marker; falls back to the
    /// one before it when the marker ends the line.
    fn widen_marker(source: &str, start: usize, end: usize) -> (usize, usize) {
        if source[end..].starts_with(' ') {
            return (start, end + 1);
        }
        if source[..start].ends_with(' ') {
            return (start - 1, end);
        }
        (start, end)
    }

    /// Whether `comment` occupies its line alone (only whitespace precedes it).
    fn is_standalone(&self, comment: &CommentInfo) -> bool {
        let line_start = self.source[..comment.start_byte].rfind('\n').map_or(0, |pos| pos + 1);
        self.source[line_start..comment.start_byte]
            .bytes()
            .all(|byte| byte.is_ascii_whitespace())
    }

    /// Whether `comment` is a single-line comment node that occupies its line
    /// alone (only whitespace precedes it). Trailing comments and multi-line
    /// (block) comment nodes return `false`.
    fn is_standalone_single_line(&self, comment: &CommentInfo) -> bool {
        comment.start_row == comment.end_row && self.is_standalone(comment)
    }
}

#[cfg(test)]
mod tests {
    use super::*;
    use crate::rules::preservation::PreservationRule;

    fn create_mock_comment(node_type: &str) -> CommentInfo {
        CommentInfo {
            start_byte: 0,
            end_byte: 0,
            start_row: 0,
            end_row: 0,
            node_type: node_type.to_string(),
            should_preserve: false,
            is_documentation: false,
        }
    }

    #[test]
    fn test_comment_info_creation() {
        let comment = create_mock_comment("line_comment");
        assert_eq!(comment.node_type, "line_comment");
        assert!(!comment.should_preserve);
    }

    #[test]
    fn test_comment_preservation() {
        let comment = create_mock_comment("line_comment");
        let preserved_comment = comment.with_preservation(true);
        assert!(preserved_comment.should_preserve);
    }

    #[test]
    fn test_visitor_creation() {
        let source = "// Test\nfn main() {}";
        let rules = vec![PreservationRule::pattern("TODO")];
        let comment_types = vec!["comment".to_string(), "line_comment".to_string()];
        let doc_types = vec!["doc_comment".to_string()];
        let visitor = CommentVisitor::new_with_language(source, &rules, &comment_types, &doc_types, "test");
        assert_eq!(visitor.source, source);
        assert_eq!(visitor.comments.len(), 0);
    }

    #[test]
    fn test_get_comments_to_remove() {
        let source = "// Test";
        let rules = vec![PreservationRule::pattern("TODO")];
        let comment_types = vec!["comment".to_string(), "line_comment".to_string()];
        let doc_types = vec!["doc_comment".to_string()];
        let mut visitor = CommentVisitor::new_with_language(source, &rules, &comment_types, &doc_types, "test");

        visitor
            .comments
            .push(create_mock_comment("line_comment").with_preservation(true));
        visitor
            .comments
            .push(create_mock_comment("line_comment").with_preservation(false));

        let to_remove = visitor.get_comments_to_remove();
        assert_eq!(to_remove.len(), 1);
        assert!(!to_remove[0].should_preserve);
    }
}