Skip to main content

rumdl_lib/rules/
md034_no_bare_urls.rs

1/// Rule MD034: No unformatted URLs
2///
3/// See [docs/md034.md](../../docs/md034.md) for full documentation, configuration, and examples.
4use std::sync::LazyLock;
5
6use regex::Regex;
7
8use crate::rule::{Fix, LintError, LintResult, LintWarning, Rule, RuleCategory, Severity};
9use crate::utils::range_utils::{LineIndex, calculate_url_range};
10use crate::utils::regex_cache::{
11    EMAIL_PATTERN, URL_IPV6_REGEX, URL_QUICK_CHECK_REGEX, URL_STANDARD_REGEX, URL_WWW_REGEX, XMPP_URI_REGEX,
12};
13
14use crate::filtered_lines::FilteredLinesExt;
15use crate::lint_context::LintContext;
16
17// MD034-specific pre-compiled regex patterns for markdown constructs
18static CUSTOM_PROTOCOL_REGEX: LazyLock<Regex> = LazyLock::new(|| {
19    Regex::new(r#"(?:grpc|ws|wss|ssh|git|svn|file|data|javascript|vscode|chrome|about|slack|discord|matrix|irc|redis|mongodb|postgresql|mysql|kafka|nats|amqp|mqtt|custom|app|api|service)://"#).unwrap()
20});
21static MARKDOWN_LINK_REGEX: LazyLock<Regex> = LazyLock::new(|| {
22    Regex::new(r#"\[(?:[^\[\]]|\[[^\]]*\])*\]\(([^)\s]+)(?:\s+(?:\"[^\"]*\"|\'[^\']*\'))?\)"#).unwrap()
23});
24static MARKDOWN_EMPTY_LINK_REGEX: LazyLock<Regex> =
25    LazyLock::new(|| Regex::new(r#"\[(?:[^\[\]]|\[[^\]]*\])*\]\(\)"#).unwrap());
26static MARKDOWN_EMPTY_REF_REGEX: LazyLock<Regex> =
27    LazyLock::new(|| Regex::new(r#"\[(?:[^\[\]]|\[[^\]]*\])*\]\[\]"#).unwrap());
28static ANGLE_LINK_REGEX: LazyLock<Regex> = LazyLock::new(|| {
29    Regex::new(
30        r#"<((?:https?|ftps?)://(?:\[[0-9a-fA-F:]+(?:%[a-zA-Z0-9]+)?\]|[^>]+)|xmpp:[^>]+|[^@\s]+@[^@\s]+\.[^@\s>]+)>"#,
31    )
32    .unwrap()
33});
34static BADGE_LINK_LINE_REGEX: LazyLock<Regex> =
35    LazyLock::new(|| Regex::new(r#"^\s*\[!\[[^\]]*\]\([^)]*\)\]\([^)]*\)\s*$"#).unwrap());
36static MARKDOWN_IMAGE_REGEX: LazyLock<Regex> =
37    LazyLock::new(|| Regex::new(r#"!\s*\[([^\]]*)\]\s*\(([^)\s]+)(?:\s+(?:\"[^\"]*\"|\'[^\']*\'))?\)"#).unwrap());
38static MULTILINE_LINK_CONTINUATION_REGEX: LazyLock<Regex> = LazyLock::new(|| Regex::new(r#"^[^\[]*\]\(.*\)"#).unwrap());
39static SHORTCUT_REF_REGEX: LazyLock<Regex> = LazyLock::new(|| Regex::new(r#"\[([^\[\]]+)\]"#).unwrap());
40
41/// Reusable buffers for check_line to reduce allocations
42#[derive(Default)]
43struct LineCheckBuffers {
44    markdown_link_ranges: Vec<(usize, usize)>,
45    image_ranges: Vec<(usize, usize)>,
46    urls_found: Vec<(usize, usize, String)>,
47}
48
49#[derive(Default, Clone)]
50pub struct MD034NoBareUrls;
51
52impl MD034NoBareUrls {
53    #[inline]
54    pub fn should_skip_content(&self, content: &str) -> bool {
55        // Skip if content has no URLs, XMPP URIs, or email addresses
56        // Fast byte scanning for common URL/email/xmpp indicators
57        let bytes = content.as_bytes();
58        let has_colon = bytes.contains(&b':');
59        let has_at = bytes.contains(&b'@');
60        let has_www = content.contains("www.");
61        !has_colon && !has_at && !has_www
62    }
63
64    /// Remove trailing punctuation that is likely sentence punctuation, not part of the URL
65    fn trim_trailing_punctuation<'a>(&self, url: &'a str) -> &'a str {
66        let mut trimmed = url;
67
68        // Check for balanced parentheses - if we have unmatched closing parens, they're likely punctuation
69        let open_parens = url.chars().filter(|&c| c == '(').count();
70        let close_parens = url.chars().filter(|&c| c == ')').count();
71
72        if close_parens > open_parens {
73            // Find the last balanced closing paren position
74            let mut balance = 0;
75            let mut last_balanced_pos = url.len();
76
77            for (byte_idx, c) in url.char_indices() {
78                if c == '(' {
79                    balance += 1;
80                } else if c == ')' {
81                    balance -= 1;
82                    if balance < 0 {
83                        // Found an unmatched closing paren
84                        last_balanced_pos = byte_idx;
85                        break;
86                    }
87                }
88            }
89
90            trimmed = &trimmed[..last_balanced_pos];
91        }
92
93        // Trim specific punctuation only if not followed by more URL-like chars
94        while let Some(last_char) = trimmed.chars().last() {
95            if matches!(last_char, '.' | ',' | ';' | ':' | '!' | '?') {
96                // Check if this looks like it could be part of the URL
97                // For ':' specifically, keep it if followed by digits (port number)
98                if last_char == ':' && trimmed.len() > 1 {
99                    // Don't trim
100                    break;
101                }
102                trimmed = &trimmed[..trimmed.len() - 1];
103            } else {
104                break;
105            }
106        }
107
108        trimmed
109    }
110
111    fn check_line(
112        &self,
113        line: &str,
114        ctx: &LintContext,
115        line_number: usize,
116        code_spans: &[crate::lint_context::CodeSpan],
117        buffers: &mut LineCheckBuffers,
118        line_index: &LineIndex,
119    ) -> Vec<LintWarning> {
120        let mut warnings = Vec::new();
121
122        // Skip lines inside HTML blocks - URLs in HTML attributes should not be linted
123        if ctx.line_info(line_number).is_some_and(|info| info.in_html_block) {
124            return warnings;
125        }
126
127        // Skip lines that are continuations of multiline markdown links
128        // Pattern: text](url) without a leading [
129        if MULTILINE_LINK_CONTINUATION_REGEX.is_match(line) {
130            return warnings;
131        }
132
133        // Quick check - does this line potentially have a URL or email?
134        let has_quick_check = URL_QUICK_CHECK_REGEX.is_match(line);
135        let has_www = line.contains("www.");
136        let has_at = line.contains('@');
137
138        if !has_quick_check && !has_at && !has_www {
139            return warnings;
140        }
141
142        // Clear and reuse buffers instead of allocating new ones
143        buffers.markdown_link_ranges.clear();
144        buffers.image_ranges.clear();
145
146        let has_bracket = line.contains('[');
147        let has_angle = line.contains('<');
148        let has_bang = line.contains('!');
149
150        if has_bracket {
151            for mat in MARKDOWN_LINK_REGEX.find_iter(line) {
152                buffers.markdown_link_ranges.push((mat.start(), mat.end()));
153            }
154
155            // Also include empty link patterns like [text]() and [text][]
156            for mat in MARKDOWN_EMPTY_LINK_REGEX.find_iter(line) {
157                buffers.markdown_link_ranges.push((mat.start(), mat.end()));
158            }
159
160            for mat in MARKDOWN_EMPTY_REF_REGEX.find_iter(line) {
161                buffers.markdown_link_ranges.push((mat.start(), mat.end()));
162            }
163
164            // Also exclude shortcut reference links like [URL]
165            for mat in SHORTCUT_REF_REGEX.find_iter(line) {
166                let end = mat.end();
167                let next_non_ws = line[end..].bytes().find(|b| !b.is_ascii_whitespace());
168                if next_non_ws == Some(b'(') || next_non_ws == Some(b'[') {
169                    continue;
170                }
171                buffers.markdown_link_ranges.push((mat.start(), mat.end()));
172            }
173
174            // Check if this line contains only a badge link (common pattern)
175            if has_bang && BADGE_LINK_LINE_REGEX.is_match(line) {
176                return warnings;
177            }
178        }
179
180        if has_angle {
181            for mat in ANGLE_LINK_REGEX.find_iter(line) {
182                buffers.markdown_link_ranges.push((mat.start(), mat.end()));
183            }
184        }
185
186        // Find all markdown images for exclusion
187        if has_bang && has_bracket {
188            for mat in MARKDOWN_IMAGE_REGEX.find_iter(line) {
189                buffers.image_ranges.push((mat.start(), mat.end()));
190            }
191        }
192
193        // Find bare URLs
194        buffers.urls_found.clear();
195
196        // First, find IPv6 URLs (they need special handling)
197        for mat in URL_IPV6_REGEX.find_iter(line) {
198            let url_str = mat.as_str();
199            buffers.urls_found.push((mat.start(), mat.end(), url_str.to_string()));
200        }
201
202        // Then find regular URLs
203        for mat in URL_STANDARD_REGEX.find_iter(line) {
204            let url_str = mat.as_str();
205
206            // Skip if it's an IPv6 URL (already handled)
207            if url_str.contains("://[") {
208                continue;
209            }
210
211            // Skip malformed IPv6-like URLs
212            // Check for IPv6-like patterns that are malformed
213            if let Some(host_start) = url_str.find("://") {
214                let after_protocol = &url_str[host_start + 3..];
215                // If it looks like IPv6 (has :: or multiple :) but no brackets, skip if followed by ]
216                if after_protocol.contains("::") || after_protocol.chars().filter(|&c| c == ':').count() > 1 {
217                    // Check if the next byte after our match is ] (ASCII, so byte check is safe)
218                    if line.as_bytes().get(mat.end()) == Some(&b']') {
219                        // This is likely a malformed IPv6 URL like "https://::1]:8080"
220                        continue;
221                    }
222                }
223            }
224
225            buffers.urls_found.push((mat.start(), mat.end(), url_str.to_string()));
226        }
227
228        // Find www URLs without protocol (e.g., www.example.com)
229        for mat in URL_WWW_REGEX.find_iter(line) {
230            let url_str = mat.as_str();
231            let start_pos = mat.start();
232            let end_pos = mat.end();
233
234            // Skip if preceded by / or @ (likely part of a full URL)
235            if start_pos > 0 {
236                let prev_char = line.as_bytes().get(start_pos - 1).copied();
237                if prev_char == Some(b'/') || prev_char == Some(b'@') {
238                    continue;
239                }
240            }
241
242            // Skip if inside angle brackets (autolink syntax like <www.example.com>)
243            if start_pos > 0 && end_pos < line.len() {
244                let prev_char = line.as_bytes().get(start_pos - 1).copied();
245                let next_char = line.as_bytes().get(end_pos).copied();
246                if prev_char == Some(b'<') && next_char == Some(b'>') {
247                    continue;
248                }
249            }
250
251            buffers.urls_found.push((start_pos, end_pos, url_str.to_string()));
252        }
253
254        // Find XMPP URIs (GFM extended autolinks: xmpp:user@domain/resource)
255        for mat in XMPP_URI_REGEX.find_iter(line) {
256            let uri_str = mat.as_str();
257            let start_pos = mat.start();
258            let end_pos = mat.end();
259
260            // Skip if inside angle brackets (already properly formatted: <xmpp:user@domain>)
261            if start_pos > 0 && end_pos < line.len() {
262                let prev_char = line.as_bytes().get(start_pos - 1).copied();
263                let next_char = line.as_bytes().get(end_pos).copied();
264                if prev_char == Some(b'<') && next_char == Some(b'>') {
265                    continue;
266                }
267            }
268
269            buffers.urls_found.push((start_pos, end_pos, uri_str.to_string()));
270        }
271
272        // Process found URLs
273        for &(start, _end, ref url_str) in &buffers.urls_found {
274            // Skip custom protocols
275            if CUSTOM_PROTOCOL_REGEX.is_match(url_str) {
276                continue;
277            }
278
279            // Check if this URL is inside a markdown link, angle bracket, or image
280            // We check if the URL starts within a construct, not if it's entirely contained.
281            // This handles cases where URL detection may include trailing characters
282            // that extend past the construct boundary (e.g., parentheses).
283            // Linear scan is correct here because ranges can overlap/nest (e.g., [[1]](url))
284            let is_inside_construct = buffers
285                .markdown_link_ranges
286                .iter()
287                .any(|&(s, e)| start >= s && start < e)
288                || buffers.image_ranges.iter().any(|&(s, e)| start >= s && start < e);
289
290            if is_inside_construct {
291                continue;
292            }
293
294            // Calculate absolute byte position for context-aware checks
295            let line_start_byte = line_index.get_line_start_byte(line_number).unwrap_or(0);
296            let absolute_pos = line_start_byte + start;
297
298            // Check if URL is inside an HTML tag (handles multiline tags correctly)
299            if ctx.is_in_html_tag(absolute_pos) {
300                continue;
301            }
302
303            // Check if URL is a JSX component attribute value (e.g. `<Card href="..."/>`).
304            // These are string props, not bare prose; wrapping them in angle brackets
305            // would produce invalid JSX. No-op for non-JSX flavors.
306            if ctx.is_in_jsx_component_tag(absolute_pos) {
307                continue;
308            }
309
310            // Check if we're inside an HTML comment
311            if ctx.is_in_html_comment(absolute_pos) || ctx.is_in_mdx_comment(absolute_pos) {
312                continue;
313            }
314
315            // Check if we're inside a Hugo/Quarto shortcode
316            if ctx.is_in_shortcode(absolute_pos) {
317                continue;
318            }
319
320            // Skip URLs inside Pandoc line blocks (`| text`) or YAML metadata blocks.
321            // Both constructs treat their content as literal/structured text where bare
322            // URLs are intentional and should not be reformatted.
323            if ctx.flavor.is_pandoc_compatible()
324                && (ctx.is_in_line_block(absolute_pos) || ctx.is_in_pandoc_metadata(absolute_pos))
325            {
326                continue;
327            }
328
329            // Clean up the URL by removing trailing punctuation
330            let trimmed_url = self.trim_trailing_punctuation(url_str);
331
332            // Only report if we have a valid URL after trimming
333            if !trimmed_url.is_empty() && trimmed_url != "//" {
334                let trimmed_len = trimmed_url.len();
335                let (start_line, start_col, end_line, end_col) =
336                    calculate_url_range(line_number, line, start, trimmed_len);
337
338                // For www URLs without protocol, add https:// prefix in the fix
339                let replacement = if trimmed_url.starts_with("www.") {
340                    format!("<https://{trimmed_url}>")
341                } else {
342                    format!("<{trimmed_url}>")
343                };
344
345                warnings.push(LintWarning {
346                    rule_name: Some("MD034".to_string()),
347                    line: start_line,
348                    column: start_col,
349                    end_line,
350                    end_column: end_col,
351                    message: format!("URL without angle brackets or link formatting: '{trimmed_url}'"),
352                    severity: Severity::Warning,
353                    fix: Some(Fix::new(
354                        {
355                            let line_start_byte = line_index.get_line_start_byte(line_number).unwrap_or(0);
356                            (line_start_byte + start)..(line_start_byte + start + trimmed_len)
357                        },
358                        replacement,
359                    )),
360                });
361            }
362        }
363
364        // Check for bare email addresses
365        for cap in EMAIL_PATTERN.captures_iter(line) {
366            if let Some(mat) = cap.get(0) {
367                let email = mat.as_str();
368                let start = mat.start();
369                let end = mat.end();
370
371                // Skip if email is part of an XMPP URI (xmpp:user@domain)
372                // Check character boundary to avoid panics with multi-byte UTF-8
373                if start >= 5 && line.is_char_boundary(start - 5) && &line[start - 5..start] == "xmpp:" {
374                    continue;
375                }
376
377                // Check if email is inside angle brackets or markdown link
378                let mut is_inside_construct = false;
379                for &(link_start, link_end) in &buffers.markdown_link_ranges {
380                    if start >= link_start && end <= link_end {
381                        is_inside_construct = true;
382                        break;
383                    }
384                }
385
386                if !is_inside_construct {
387                    // Calculate absolute byte position for context-aware checks
388                    let line_start_byte = line_index.get_line_start_byte(line_number).unwrap_or(0);
389                    let absolute_pos = line_start_byte + start;
390
391                    // Check if email is inside an HTML tag (handles multiline tags)
392                    if ctx.is_in_html_tag(absolute_pos) {
393                        continue;
394                    }
395
396                    // Check if email is a JSX component attribute value (e.g.
397                    // `<Contact email="..."/>`). No-op for non-JSX flavors.
398                    if ctx.is_in_jsx_component_tag(absolute_pos) {
399                        continue;
400                    }
401
402                    // Skip emails inside Pandoc line blocks or YAML metadata blocks.
403                    if ctx.flavor.is_pandoc_compatible()
404                        && (ctx.is_in_line_block(absolute_pos) || ctx.is_in_pandoc_metadata(absolute_pos))
405                    {
406                        continue;
407                    }
408
409                    // Check if email is inside a code span (byte offsets handle multi-line spans)
410                    let is_in_code_span = code_spans
411                        .iter()
412                        .any(|span| absolute_pos >= span.byte_offset && absolute_pos < span.byte_end);
413
414                    if !is_in_code_span {
415                        let email_len = end - start;
416                        let (start_line, start_col, end_line, end_col) =
417                            calculate_url_range(line_number, line, start, email_len);
418
419                        warnings.push(LintWarning {
420                            rule_name: Some("MD034".to_string()),
421                            line: start_line,
422                            column: start_col,
423                            end_line,
424                            end_column: end_col,
425                            message: format!("Email address without angle brackets or link formatting: '{email}'"),
426                            severity: Severity::Warning,
427                            fix: Some(Fix::new(
428                                (line_start_byte + start)..(line_start_byte + end),
429                                format!("<{email}>"),
430                            )),
431                        });
432                    }
433                }
434            }
435        }
436
437        warnings
438    }
439}
440
441impl Rule for MD034NoBareUrls {
442    #[inline]
443    fn name(&self) -> &'static str {
444        "MD034"
445    }
446
447    fn as_any(&self) -> &dyn std::any::Any {
448        self
449    }
450
451    fn from_config(_config: &crate::config::Config) -> Box<dyn Rule>
452    where
453        Self: Sized,
454    {
455        Box::new(MD034NoBareUrls)
456    }
457
458    #[inline]
459    fn category(&self) -> RuleCategory {
460        RuleCategory::Link
461    }
462
463    fn should_skip(&self, ctx: &crate::lint_context::LintContext) -> bool {
464        !ctx.likely_has_links_or_images() && self.should_skip_content(ctx.content)
465    }
466
467    #[inline]
468    fn description(&self) -> &'static str {
469        "No bare URLs - wrap URLs in angle brackets"
470    }
471
472    fn check(&self, ctx: &LintContext) -> LintResult {
473        let mut warnings = Vec::new();
474        let content = ctx.content;
475
476        // Quick skip for content without URLs
477        if self.should_skip_content(content) {
478            return Ok(warnings);
479        }
480
481        // Create LineIndex for correct byte position calculations across all line ending types
482        let line_index = &ctx.line_index;
483
484        // Get code spans for exclusion
485        let code_spans = ctx.code_spans();
486
487        // Reference-definition lines are detected by rumdl's shared parser (which
488        // understands blockquote-prefixed definitions and the full CommonMark
489        // grammar), so their destination URLs are not flagged as bare URLs.
490        let ref_def_lines: std::collections::HashSet<usize> = ctx.reference_defs.iter().map(|def| def.line).collect();
491
492        // Allocate reusable buffers once instead of per-line to reduce allocations
493        let mut buffers = LineCheckBuffers::default();
494
495        // Iterate over content lines, automatically skipping front matter, code blocks,
496        // and Obsidian comments (when in Obsidian flavor)
497        // This uses the filtered iterator API which centralizes the skip logic
498        for line in ctx
499            .filtered_lines()
500            .skip_front_matter()
501            .skip_code_blocks()
502            .skip_jsx_expressions()
503            .skip_mdx_comments()
504            .skip_obsidian_comments()
505        {
506            // Skip MyST colon-fence directive openers (`:::{name} <arg>`). The text
507            // after the directive name is an opaque argument (a URL, path, or label),
508            // not markdown prose, so a bare URL there must not be wrapped in angle
509            // brackets. Directive body lines are not openers, so they fall through to
510            // `check_line` and are linted as usual.
511            if ctx.is_myst_colon_directive_opener_line(line.line_num) {
512                continue;
513            }
514
515            // Skip reference-definition lines (`[id]: url`, including inside blockquotes).
516            if ref_def_lines.contains(&line.line_num) {
517                continue;
518            }
519
520            let mut line_warnings =
521                self.check_line(line.content, ctx, line.line_num, &code_spans, &mut buffers, line_index);
522
523            // Filter out warnings that are inside code spans (handles multi-line spans via byte offsets)
524            line_warnings.retain(|warning| {
525                !code_spans.iter().any(|span| {
526                    if let Some(fix) = &warning.fix {
527                        // Byte-offset check handles both single-line and multi-line code spans
528                        fix.range.start >= span.byte_offset && fix.range.start < span.byte_end
529                    } else {
530                        span.line == warning.line
531                            && span.end_line == warning.line
532                            && warning.column > 0
533                            && (warning.column - 1) >= span.start_col
534                            && (warning.column - 1) < span.end_col
535                    }
536                })
537            });
538
539            // Filter out warnings where the URL is inside a parsed link
540            // This handles cases like [text]( https://url ) where the URL has leading whitespace
541            // pulldown-cmark correctly parses these as valid links even though our regex misses them
542            line_warnings.retain(|warning| {
543                if let Some(fix) = &warning.fix {
544                    // Check if the fix range falls inside any parsed link's byte range
545                    !ctx.links
546                        .iter()
547                        .any(|link| fix.range.start >= link.byte_offset && fix.range.end <= link.byte_end)
548                } else {
549                    true
550                }
551            });
552
553            // Filter out warnings where the URL is inside an Obsidian comment (%%...%%)
554            // This handles inline comments like: text %%https://hidden.com%% text
555            line_warnings.retain(|warning| !ctx.is_position_in_obsidian_comment(warning.line, warning.column));
556
557            warnings.extend(line_warnings);
558        }
559
560        Ok(warnings)
561    }
562
563    fn fix(&self, ctx: &LintContext) -> Result<String, LintError> {
564        let mut content = ctx.content.to_string();
565        let warnings = self.check(ctx)?;
566        let mut warnings =
567            crate::utils::fix_utils::filter_warnings_by_inline_config(warnings, ctx.inline_config(), self.name());
568
569        // Sort warnings by position to ensure consistent fix application
570        warnings.sort_by_key(|w| w.fix.as_ref().map_or(0, |f| f.range.start));
571
572        // Apply fixes in reverse order to maintain positions
573        for warning in warnings.iter().rev() {
574            if let Some(fix) = &warning.fix {
575                let start = fix.range.start;
576                let end = fix.range.end;
577                content.replace_range(start..end, &fix.replacement);
578            }
579        }
580
581        Ok(content)
582    }
583}
584
585#[cfg(test)]
586mod tests {
587    use super::*;
588
589    #[test]
590    fn test_shortcut_ref_at_end_of_line_no_trailing_chars() {
591        let rule = MD034NoBareUrls;
592        let content = "See [https://example.com]";
593        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::Standard, None);
594        let result = rule.check(&ctx).unwrap();
595        assert!(
596            result.is_empty(),
597            "[URL] at end of line should be treated as shortcut ref: {result:?}"
598        );
599    }
600
601    #[test]
602    fn test_shortcut_ref_multiple_spaces_before_paren() {
603        let rule = MD034NoBareUrls;
604        let content = "[text]  (https://example.com)";
605        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::Standard, None);
606        let result = rule.check(&ctx).unwrap();
607        // [text]  (url) — the spaces between ] and ( mean this should be treated
608        // as shortcut ref then bare parens, NOT a markdown link. URL may still be bare.
609        // This test verifies consistent behavior with the FancyRegex that had (?!\s*[\[(])
610        let _ = result; // Just verify no panic; the exact warning count depends on other rules
611    }
612
613    #[test]
614    fn test_shortcut_ref_tab_before_bracket() {
615        let rule = MD034NoBareUrls;
616        let content = "[https://example.com]\t[other]";
617        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::Standard, None);
618        let result = rule.check(&ctx).unwrap();
619        // Tab between ] and [ does not form a full reference link in Markdown.
620        // The first [URL] is a shortcut ref containing a bare URL, so MD034 warns.
621        // This test verifies consistent behavior and no panic with tab characters.
622        assert_eq!(
623            result.len(),
624            1,
625            "Bare URL inside shortcut ref should be detected: {result:?}"
626        );
627    }
628
629    #[test]
630    fn test_shortcut_ref_followed_by_punctuation() {
631        let rule = MD034NoBareUrls;
632        let content = "[https://example.com], see also other things.";
633        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::Standard, None);
634        let result = rule.check(&ctx).unwrap();
635        assert!(
636            result.is_empty(),
637            "[URL] followed by comma should be treated as shortcut ref: {result:?}"
638        );
639    }
640
641    #[test]
642    fn test_url_in_backticks_inside_mdx_component_not_flagged() {
643        // Exact reproduction from issue #572: URL inside inline code within an MDX
644        // component body must not be flagged. The same URL in backticks outside the
645        // component is already handled correctly and serves as a control.
646        let rule = MD034NoBareUrls;
647        let content = "# Test\n\nControl: `https://rumdl.example.com/` is fine here.\n\n<ParamField path=\"--stuff\">\n  This URL `https://rumdl.example.com/` must not be flagged.\n</ParamField>\n";
648        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
649        let result = rule.check(&ctx).unwrap();
650        assert!(
651            result.is_empty(),
652            "URL in backticks inside MDX component must not be flagged: {result:?}"
653        );
654    }
655
656    #[test]
657    fn test_bare_url_inside_mdx_component_still_flagged() {
658        // A bare URL (not in backticks) inside an MDX component body must still be flagged.
659        // This ensures the fix for issue #572 only suppresses properly code-spanned URLs.
660        let rule = MD034NoBareUrls;
661        let content =
662            "# Test\n\n<ParamField path=\"--stuff\">\n  Visit https://rumdl.example.com/ for details.\n</ParamField>\n";
663        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
664        let result = rule.check(&ctx).unwrap();
665        assert_eq!(
666            result.len(),
667            1,
668            "Bare URL in MDX component body must still be flagged: {result:?}"
669        );
670    }
671
672    #[test]
673    fn test_url_in_backticks_inside_nested_mdx_component_not_flagged() {
674        // Nested MDX components must also respect code spans.
675        let rule = MD034NoBareUrls;
676        let content = "<Outer>\n  <Inner>\n    Check `https://example.com/` here.\n  </Inner>\n</Outer>\n";
677        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
678        let result = rule.check(&ctx).unwrap();
679        assert!(
680            result.is_empty(),
681            "URL in backticks inside nested MDX component must not be flagged: {result:?}"
682        );
683    }
684
685    /// Issue #678: a URL inside a fenced code block that is nested within a JSX/MDX
686    /// component (e.g. `<Steps><Step>`) is code, not bare prose. It must not be
687    /// flagged, and `fix` must not rewrite it (which would corrupt the command).
688    #[test]
689    fn test_url_in_fenced_code_block_inside_jsx_not_flagged() {
690        let rule = MD034NoBareUrls;
691        let content = "# Title\n\n<Steps>\n  <Step title=\"Send a request\">\n```bash\ncurl https://example.com/api\n```\n  </Step>\n</Steps>\n";
692        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
693        let result = rule.check(&ctx).unwrap();
694        assert!(
695            result.is_empty(),
696            "URL in a fenced code block nested in a JSX component must not be flagged: {result:?}"
697        );
698    }
699
700    /// The same code block must be left byte-for-byte intact by `fix` (no
701    /// `<https://...>` rewrite that breaks a copy-pasteable command).
702    #[test]
703    fn test_fix_does_not_rewrite_url_in_fenced_code_block_inside_jsx() {
704        let rule = MD034NoBareUrls;
705        let content = "# Title\n\n<Steps>\n  <Step title=\"Send a request\">\n```bash\ncurl https://example.com/api\n```\n  </Step>\n</Steps>\n";
706        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
707        let fixed = rule.fix(&ctx).unwrap();
708        assert_eq!(
709            fixed, content,
710            "fix must not rewrite a URL inside a JSX-nested fenced code block"
711        );
712    }
713
714    /// Control: a bare URL in the JSX *body* (outside any fence) is genuine prose
715    /// and must still be flagged, so the fence exemption is not over-broad.
716    #[test]
717    fn test_bare_url_in_jsx_body_outside_fence_still_flagged() {
718        let rule = MD034NoBareUrls;
719        let content = "# Title\n\n<Steps>\n  <Step title=\"Send a request\">\n  Visit https://example.com/api now.\n  </Step>\n</Steps>\n";
720        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
721        let result = rule.check(&ctx).unwrap();
722        assert_eq!(
723            result.len(),
724            1,
725            "A bare URL in the JSX body (not in a fence) must still be flagged: {result:?}"
726        );
727    }
728
729    /// A `<!--` inside a fenced code block is literal, not a comment opener, so it
730    /// must not pair with a later `-->` to form a comment range that masks a real
731    /// bare URL between them (the code-block counterpart to the code-span fix).
732    #[test]
733    fn test_bare_url_not_masked_by_comment_delimiter_in_code_block() {
734        let rule = MD034NoBareUrls;
735        let content =
736            "# T\n\n```text\n<!-- literal opener, not a comment\n```\n\nhttps://example.com should be flagged\n\n-->\n";
737        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::Standard, None);
738        let result = rule.check(&ctx).unwrap();
739        assert_eq!(result.len(), 1, "the bare URL must still be flagged: {result:?}");
740        assert!(
741            result[0].message.contains("example.com"),
742            "the flagged URL must be the bare one: {result:?}"
743        );
744    }
745
746    /// Only *fenced* code blocks suppress `<!--`/`-->` as literal. A real HTML
747    /// comment indented inside a MkDocs admonition (which pulldown-cmark
748    /// misclassifies as an indented code block) must still be recognized as a
749    /// comment, so its bare URL stays skipped.
750    #[test]
751    fn test_bare_url_in_indented_comment_in_admonition_still_skipped() {
752        let rule = MD034NoBareUrls;
753        let content = "# T\n\n!!! note\n    Some text.\n\n    <!--\n    https://example.com\n    -->\n";
754        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MkDocs, None);
755        let result = rule.check(&ctx).unwrap();
756        assert!(
757            result.is_empty(),
758            "URL inside an indented HTML comment in an admonition must not be flagged: {result:?}"
759        );
760    }
761
762    /// Issue #649: a URL that is a JSX component attribute value (e.g. `href="..."`)
763    /// is a string prop, not bare prose. Wrapping it in angle brackets produces
764    /// invalid JSX, so MD034 must not flag it under the MDX flavor.
765    #[test]
766    fn test_url_in_jsx_component_attribute_not_flagged() {
767        let rule = MD034NoBareUrls;
768        let content = "<Card title=\"Docs\" href=\"https://example.com/docs\" />\n";
769        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
770        let result = rule.check(&ctx).unwrap();
771        assert!(
772            result.is_empty(),
773            "URL in a JSX component attribute must not be flagged: {result:?}"
774        );
775    }
776
777    /// The same exemption must apply when the JSX opening tag spans multiple lines.
778    #[test]
779    fn test_url_in_multiline_jsx_component_attribute_not_flagged() {
780        let rule = MD034NoBareUrls;
781        let content = "<Card\n  title=\"Docs\"\n  href=\"https://example.com/docs\"\n/>\n";
782        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
783        let result = rule.check(&ctx).unwrap();
784        assert!(
785            result.is_empty(),
786            "URL in a multi-line JSX component attribute must not be flagged: {result:?}"
787        );
788    }
789
790    /// The exemption is surgical: a URL in the component's *attributes* is skipped,
791    /// but a bare URL in the component's *body* is genuine prose and still flagged.
792    #[test]
793    fn test_jsx_attribute_url_skipped_but_body_url_flagged() {
794        let rule = MD034NoBareUrls;
795        let content = "<Card href=\"https://attr.example.com\">\n  Visit https://body.example.com now.\n</Card>\n";
796        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
797        let result = rule.check(&ctx).unwrap();
798        assert_eq!(
799            result.len(),
800            1,
801            "Only the body URL must be flagged, not the attribute URL: {result:?}"
802        );
803        assert!(
804            result[0].message.contains("body.example.com"),
805            "The flagged URL must be the body one: {result:?}"
806        );
807    }
808
809    /// The email path has the same JSX-attribute blind spot; an email used as a
810    /// JSX component attribute value must not be flagged either.
811    #[test]
812    fn test_email_in_jsx_component_attribute_not_flagged() {
813        let rule = MD034NoBareUrls;
814        let content = "<Contact email=\"hello@example.com\" />\n";
815        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
816        let result = rule.check(&ctx).unwrap();
817        assert!(
818            result.is_empty(),
819            "Email in a JSX component attribute must not be flagged: {result:?}"
820        );
821    }
822
823    /// Control: under the Standard flavor `<Card .../>` is parsed as an HTML tag,
824    /// so the attribute URL is already covered by the existing HTML-tag guard.
825    /// This locks in that the two flavors agree.
826    #[test]
827    fn test_jsx_attribute_url_not_flagged_in_standard_flavor() {
828        let rule = MD034NoBareUrls;
829        let content = "<Card href=\"https://example.com/docs\" />\n";
830        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::Standard, None);
831        let result = rule.check(&ctx).unwrap();
832        assert!(
833            result.is_empty(),
834            "URL in a tag attribute must not be flagged under Standard flavor either: {result:?}"
835        );
836    }
837
838    /// URLs inside Pandoc line blocks (`| text`) must not be flagged as bare URLs.
839    #[test]
840    fn test_pandoc_skips_urls_in_line_blocks() {
841        use crate::config::MarkdownFlavor;
842        use crate::lint_context::LintContext;
843        let rule = MD034NoBareUrls;
844        let content = "| See https://example.com\n| For details\n";
845        let ctx = LintContext::new(content, MarkdownFlavor::Pandoc, None);
846        let result = rule.check(&ctx).unwrap();
847        assert!(
848            result.is_empty(),
849            "MD034 should skip URLs in Pandoc line blocks: {result:?}"
850        );
851    }
852
853    /// URLs inside Pandoc YAML metadata blocks must not be flagged.
854    #[test]
855    fn test_pandoc_skips_urls_in_metadata() {
856        use crate::config::MarkdownFlavor;
857        use crate::lint_context::LintContext;
858        let rule = MD034NoBareUrls;
859        let content = "---\nhomepage: https://example.com\n---\n\nBody.\n";
860        let ctx = LintContext::new(content, MarkdownFlavor::Pandoc, None);
861        let result = rule.check(&ctx).unwrap();
862        assert!(
863            result.is_empty(),
864            "MD034 should skip URLs in Pandoc YAML metadata: {result:?}"
865        );
866    }
867
868    /// Standard flavor must still flag bare URLs in lines starting with `|`
869    /// (which are not interpreted as line blocks).
870    #[test]
871    fn test_standard_still_flags_urls_in_pipe_prefixed_lines() {
872        use crate::config::MarkdownFlavor;
873        use crate::lint_context::LintContext;
874        let rule = MD034NoBareUrls;
875        let content = "| See https://example.com\n";
876        let ctx = LintContext::new(content, MarkdownFlavor::Standard, None);
877        let result = rule.check(&ctx).unwrap();
878        assert!(
879            !result.is_empty(),
880            "MD034 should still flag URLs in pipe-prefixed lines under Standard flavor"
881        );
882    }
883
884    #[test]
885    fn test_url_in_backticks_after_fenced_code_block_inside_mdx_not_flagged() {
886        // A fenced code block inside a JSX component must not misalign the code-span
887        // offset map. The URL in backticks that appears *after* the code block must
888        // still be recognised as being inside a code span.
889        let rule = MD034NoBareUrls;
890        let content = "\
891<Component>
892Some intro text.
893
894```
895example code here
896```
897
898Check `https://example.com/` here.
899</Component>
900";
901        let ctx = crate::lint_context::LintContext::new(content, crate::config::MarkdownFlavor::MDX, None);
902        let result = rule.check(&ctx).unwrap();
903        assert!(
904            result.is_empty(),
905            "URL in backticks after a fenced code block inside MDX must not be flagged: {result:?}"
906        );
907    }
908
909    /// Issue #642: a URL given as the argument of a MyST colon-fence directive
910    /// (`:::{name} <url>`) is the directive's opaque argument, not markdown prose,
911    /// and must not be wrapped in angle brackets.
912    #[test]
913    fn test_myst_colon_directive_argument_url_not_flagged() {
914        use crate::config::MarkdownFlavor;
915        use crate::lint_context::LintContext;
916        let rule = MD034NoBareUrls;
917        let content = "\
918:::{anywidget} https://cdn.jsdelivr.net/npm/repo-review-webapp@1.1.3/dist/repo-review-anywidget.mjs
919{
920  \"deps\": [\"repo-review~=1.1.0\"]
921}
922:::
923";
924        let ctx = LintContext::new(content, MarkdownFlavor::MyST, None);
925        let result = rule.check(&ctx).unwrap();
926        assert!(
927            result.is_empty(),
928            "URL argument on a MyST colon directive opener must not be flagged: {result:?}"
929        );
930    }
931
932    /// A nested MyST colon directive opener also carries an opaque argument.
933    #[test]
934    fn test_myst_nested_colon_directive_argument_url_not_flagged() {
935        use crate::config::MarkdownFlavor;
936        use crate::lint_context::LintContext;
937        let rule = MD034NoBareUrls;
938        let content = "\
939::::{grid}
940:::{card} https://example.com/card-target
941Some caption.
942:::
943::::
944";
945        let ctx = LintContext::new(content, MarkdownFlavor::MyST, None);
946        let result = rule.check(&ctx).unwrap();
947        assert!(
948            result.is_empty(),
949            "URL argument on a nested MyST colon directive opener must not be flagged: {result:?}"
950        );
951    }
952
953    /// A bare URL in the *body* of a content directive (e.g. `{note}`) is genuine
954    /// prose and must still be flagged. The opener exemption must not leak to the body.
955    #[test]
956    fn test_myst_directive_body_url_still_flagged() {
957        use crate::config::MarkdownFlavor;
958        use crate::lint_context::LintContext;
959        let rule = MD034NoBareUrls;
960        let content = "\
961:::{note}
962See https://example.com/docs for more details.
963:::
964";
965        let ctx = LintContext::new(content, MarkdownFlavor::MyST, None);
966        let result = rule.check(&ctx).unwrap();
967        assert_eq!(
968            result.len(),
969            1,
970            "Bare URL in a MyST directive body must still be flagged: {result:?}"
971        );
972    }
973
974    /// An unclosed colon directive (no terminating `:::`) still has its opener
975    /// argument treated as opaque: the URL must not be flagged.
976    #[test]
977    fn test_myst_unclosed_colon_directive_argument_url_not_flagged() {
978        use crate::config::MarkdownFlavor;
979        use crate::lint_context::LintContext;
980        let rule = MD034NoBareUrls;
981        let content = "\
982:::{anywidget} https://example.com/widget.mjs
983Some trailing content with no closing fence.
984";
985        let ctx = LintContext::new(content, MarkdownFlavor::MyST, None);
986        let result = rule.check(&ctx).unwrap();
987        assert!(
988            result.is_empty(),
989            "URL argument on an unclosed MyST colon directive opener must not be flagged: {result:?}"
990        );
991    }
992
993    /// The colon-directive exemption is MyST-specific: under the Standard flavor a
994    /// `:::{...}` line is ordinary text and a bare URL on it must still be flagged.
995    #[test]
996    fn test_colon_directive_url_flagged_in_standard_flavor() {
997        use crate::config::MarkdownFlavor;
998        use crate::lint_context::LintContext;
999        let rule = MD034NoBareUrls;
1000        let content = ":::{anywidget} https://example.com/widget.mjs\n";
1001        let ctx = LintContext::new(content, MarkdownFlavor::Standard, None);
1002        let result = rule.check(&ctx).unwrap();
1003        assert_eq!(
1004            result.len(),
1005            1,
1006            "Under Standard flavor a bare URL on a `:::` line must still be flagged: {result:?}"
1007        );
1008    }
1009}