1pub mod types;
2pub use types::*;
3
4mod element_parsers;
5mod flavor_detection;
6mod heading_detection;
7mod line_computation;
8mod link_parser;
9mod list_blocks;
10#[cfg(test)]
11mod tests;
12
13use crate::config::MarkdownFlavor;
14use crate::inline_config::InlineConfig;
15use crate::rules::front_matter_utils::FrontMatterUtils;
16use crate::utils::code_block_utils::{CodeBlockDetail, CodeBlockUtils};
17use crate::utils::range_utils::byte_to_char_count;
18use std::collections::HashMap;
19use std::path::PathBuf;
20
21#[cfg(not(target_arch = "wasm32"))]
23macro_rules! profile_section {
24 ($name:expr, $profile:expr, $code:expr) => {{
25 let start = std::time::Instant::now();
26 let result = $code;
27 if $profile {
28 eprintln!("[PROFILE] {}: {:?}", $name, start.elapsed());
29 }
30 result
31 }};
32}
33
34#[cfg(target_arch = "wasm32")]
35macro_rules! profile_section {
36 ($name:expr, $profile:expr, $code:expr) => {{ $code }};
37}
38
39pub(super) struct SkipByteRanges<'a> {
42 pub(super) html_comment_ranges: &'a [crate::utils::skip_context::ByteRange],
43 pub(super) autodoc_ranges: &'a [crate::utils::skip_context::ByteRange],
44 pub(super) pandoc_div_ranges: &'a [crate::utils::skip_context::ByteRange],
45 pub(super) pymdown_block_ranges: &'a [crate::utils::skip_context::ByteRange],
46}
47
48use std::sync::{Arc, OnceLock};
49
50pub(super) type ListItemMap = std::collections::HashMap<usize, (bool, String, usize, usize, Option<usize>)>;
52
53pub(super) type ByteRanges = Vec<(usize, usize)>;
55
56pub struct LintContext<'a> {
57 pub content: &'a str,
58 content_lines: Vec<&'a str>, pub line_offsets: Vec<usize>,
60 pub code_blocks: Vec<(usize, usize)>, pub code_block_details: Vec<CodeBlockDetail>, pub strong_spans: Vec<crate::utils::code_block_utils::StrongSpanDetail>, pub line_to_list: crate::utils::code_block_utils::LineToListMap, pub list_start_values: crate::utils::code_block_utils::ListStartValues, pub lines: Vec<LineInfo>, pub links: Vec<ParsedLink<'a>>, pub images: Vec<ParsedImage<'a>>, pub broken_links: Vec<BrokenLinkInfo>, pub footnote_refs: Vec<FootnoteRef>, pub reference_defs: Vec<ReferenceDef>, reference_defs_map: HashMap<String, usize>, code_spans_cache: OnceLock<Arc<Vec<CodeSpan>>>, math_spans_cache: OnceLock<Arc<Vec<MathSpan>>>, math_byte_ranges_cache: OnceLock<Vec<(usize, usize)>>, pub list_blocks: Vec<ListBlock>, pub char_frequency: CharFrequency, html_tags_cache: OnceLock<Arc<Vec<HtmlTag>>>, jsx_component_tags_cache: OnceLock<Arc<Vec<HtmlTag>>>, emphasis_spans_cache: OnceLock<Arc<Vec<EmphasisSpan>>>, bare_urls_cache: OnceLock<Arc<Vec<BareUrl>>>, has_mixed_list_nesting_cache: OnceLock<bool>, html_comment_ranges: Vec<crate::utils::skip_context::ByteRange>, pub table_blocks: Vec<crate::utils::table_utils::TableBlock>, pub line_index: crate::utils::range_utils::LineIndex<'a>, jinja_ranges: Vec<(usize, usize)>, pub flavor: MarkdownFlavor, pub source_file: Option<PathBuf>, jsx_expression_ranges: Vec<(usize, usize)>, mdx_comment_ranges: Vec<(usize, usize)>, citation_ranges: Vec<crate::utils::skip_context::ByteRange>, pandoc_div_ranges: Vec<crate::utils::skip_context::ByteRange>, colon_fence_ranges: Vec<(usize, usize)>, inline_footnote_ranges: Vec<crate::utils::skip_context::ByteRange>, pandoc_header_slugs: std::collections::HashSet<String>, example_list_marker_ranges: Vec<crate::utils::skip_context::ByteRange>, example_reference_ranges: Vec<crate::utils::skip_context::ByteRange>, sub_super_ranges: Vec<crate::utils::skip_context::ByteRange>, inline_code_attr_ranges: Vec<crate::utils::skip_context::ByteRange>, bracketed_span_ranges: Vec<crate::utils::skip_context::ByteRange>, line_block_ranges: Vec<crate::utils::skip_context::ByteRange>, pipe_table_caption_ranges: Vec<crate::utils::skip_context::ByteRange>, pandoc_metadata_ranges: Vec<crate::utils::skip_context::ByteRange>, grid_table_ranges: Vec<crate::utils::skip_context::ByteRange>, multi_line_table_ranges: Vec<crate::utils::skip_context::ByteRange>, shortcode_ranges: Vec<(usize, usize)>, link_title_ranges: Vec<(usize, usize)>, code_span_byte_ranges: Vec<(usize, usize)>, inline_config: InlineConfig, obsidian_comment_ranges: Vec<(usize, usize)>, unterminated_html_comment: Option<usize>, unterminated_obsidian_comment: Option<usize>, lazy_cont_lines_cache: OnceLock<Arc<Vec<LazyContLine>>>, myst_directive_ranges: Vec<(usize, usize)>, myst_comment_ranges: Vec<(usize, usize)>, myst_role_ranges: Vec<(usize, usize)>, front_matter_end: usize, }
118
119pub fn code_block_ranges(content: &str, flavor: MarkdownFlavor) -> Vec<(usize, usize)> {
130 LintContext::new(content, flavor, None).code_blocks
131}
132
133impl<'a> LintContext<'a> {
134 pub fn new(content: &'a str, flavor: MarkdownFlavor, source_file: Option<PathBuf>) -> Self {
135 #[cfg(not(target_arch = "wasm32"))]
136 let profile = std::env::var("RUMDL_PROFILE_QUADRATIC").is_ok();
137
138 let line_offsets = profile_section!("Line offsets", profile, {
139 let mut offsets = vec![0];
140 for (i, c) in content.char_indices() {
141 if c == '\n' {
142 offsets.push(i + 1);
143 }
144 }
145 offsets
146 });
147
148 let content_lines: Vec<&str> = content.lines().collect();
150
151 #[allow(clippy::disallowed_methods)]
155 let front_matter_end = FrontMatterUtils::get_front_matter_end_line(content);
156
157 let parse_result = profile_section!(
159 "Code blocks",
160 profile,
161 CodeBlockUtils::detect_code_blocks_and_spans(content)
162 );
163 let mut code_blocks = parse_result.code_blocks;
164 let code_span_ranges = parse_result.code_spans;
165 let code_block_details = parse_result.code_block_details;
166 let strong_spans = parse_result.strong_spans;
167 let line_to_list = parse_result.line_to_list;
168 let list_start_values = parse_result.list_start_values;
169 let html_blocks = parse_result.html_blocks;
170
171 let containers = profile_section!(
174 "Container lines",
175 profile,
176 flavor_detection::detect_container_lines(&content_lines, flavor)
177 );
178
179 let comment_code_block_ranges: Vec<(usize, usize)> = code_block_details
188 .iter()
189 .flat_map(|detail| {
190 if detail.is_fenced {
191 return vec![(detail.start, detail.end)];
192 }
193 let start_line = line_offsets
194 .partition_point(|&offset| offset <= detail.start)
195 .saturating_sub(1);
196 let end_line = line_offsets.partition_point(|&offset| offset < detail.end);
197 containers
198 .code_line_spans_in(start_line..end_line)
199 .into_iter()
200 .map(|span| {
201 let start = line_offsets[span.start].max(detail.start);
202 let end = line_offsets
203 .get(span.end)
204 .copied()
205 .unwrap_or(content.len())
206 .min(detail.end);
207 (start, end)
208 })
209 .collect()
210 })
211 .collect();
212 let body_start = line_offsets.get(front_matter_end).copied().unwrap_or(content.len());
218 let html_comment_scan = profile_section!(
219 "HTML comment ranges",
220 profile,
221 crate::utils::skip_context::scan_html_comments(
222 content,
223 &code_span_ranges,
224 &comment_code_block_ranges,
225 body_start
226 )
227 );
228 let mut html_comment_ranges = html_comment_scan.ranges;
229 let unterminated_html_comment = html_comment_scan.unterminated;
230
231 let autodoc_ranges = profile_section!("Autodoc block ranges", profile, {
235 if flavor.supports_colon_code_fences() || flavor.supports_myst_directives() {
236 Vec::new()
237 } else {
238 crate::utils::mkdocstrings_refs::detect_autodoc_block_ranges(content)
239 }
240 });
241
242 let pandoc_div_ranges = profile_section!("Pandoc div ranges", profile, {
244 if flavor.is_pandoc_compatible() {
245 crate::utils::pandoc::detect_div_block_ranges(content)
246 } else {
247 Vec::new()
248 }
249 });
250
251 let pymdown_block_ranges = profile_section!("PyMdown block ranges", profile, {
253 if flavor == MarkdownFlavor::MkDocs {
254 crate::utils::pymdown_blocks::detect_block_ranges(content)
255 } else {
256 Vec::new()
257 }
258 });
259
260 let skip_ranges = SkipByteRanges {
263 html_comment_ranges: &html_comment_ranges,
264 autodoc_ranges: &autodoc_ranges,
265 pandoc_div_ranges: &pandoc_div_ranges,
266 pymdown_block_ranges: &pymdown_block_ranges,
267 };
268 let (mut lines, emphasis_spans) = profile_section!(
269 "Basic line info",
270 profile,
271 line_computation::compute_basic_line_info(
272 content,
273 &content_lines,
274 &line_offsets,
275 &code_blocks,
276 flavor,
277 &skip_ranges,
278 front_matter_end,
279 )
280 );
281
282 profile_section!(
284 "HTML blocks",
285 profile,
286 heading_detection::detect_html_blocks(content, &mut lines)
287 );
288
289 profile_section!(
291 "ESM blocks",
292 profile,
293 flavor_detection::detect_esm_blocks(content, &mut lines, flavor)
294 );
295
296 profile_section!(
298 "JSX block detection",
299 profile,
300 flavor_detection::detect_jsx_blocks(content, &mut lines, flavor)
301 );
302
303 let (jsx_expression_ranges, mdx_comment_ranges) = profile_section!(
305 "JSX/MDX detection",
306 profile,
307 flavor_detection::detect_jsx_and_mdx_comments(content, &mut lines, flavor, &code_blocks)
308 );
309
310 profile_section!(
315 "Markdown-in-HTML blocks",
316 profile,
317 flavor_detection::detect_markdown_html_blocks(&mut lines, &containers)
318 );
319
320 profile_section!(
322 "MkDocs constructs",
323 profile,
324 flavor_detection::detect_mkdocs_line_info(&content_lines, &mut lines, flavor, &containers)
325 );
326
327 profile_section!(
332 "Footnote definitions",
333 profile,
334 detect_footnote_definitions(content, &mut lines, &line_offsets)
335 );
336
337 {
340 let mut new_code_blocks = Vec::with_capacity(code_blocks.len());
341 for &(start, end) in &code_blocks {
342 let start_line = line_offsets
343 .partition_point(|&offset| offset <= start)
344 .saturating_sub(1);
345 let end_line = line_offsets.partition_point(|&offset| offset < end).min(lines.len());
346
347 let mut sub_start: Option<usize> = None;
348 for (i, &offset) in line_offsets[start_line..end_line]
349 .iter()
350 .enumerate()
351 .map(|(j, o)| (j + start_line, o))
352 {
353 let is_real_code = lines.get(i).is_some_and(|info| info.in_code_block);
354 if is_real_code && sub_start.is_none() {
355 let byte_start = if i == start_line { start } else { offset };
356 sub_start = Some(byte_start);
357 } else if !is_real_code && sub_start.is_some() {
358 new_code_blocks.push((sub_start.unwrap(), offset));
359 sub_start = None;
360 }
361 }
362 if let Some(s) = sub_start {
363 new_code_blocks.push((s, end));
364 }
365 }
366 code_blocks = new_code_blocks;
367 }
368
369 let has_markdown_html = lines.iter().any(|l| l.in_mkdocs_html_markdown);
377 if flavor == MarkdownFlavor::MkDocs || has_markdown_html {
378 let mut new_code_blocks = Vec::with_capacity(code_blocks.len());
379 for &(start, end) in &code_blocks {
380 let start_line = line_offsets
381 .partition_point(|&offset| offset <= start)
382 .saturating_sub(1);
383 let end_line = line_offsets.partition_point(|&offset| offset < end).min(lines.len());
384
385 let mut sub_start: Option<usize> = None;
387 for (i, &offset) in line_offsets[start_line..end_line]
388 .iter()
389 .enumerate()
390 .map(|(j, o)| (j + start_line, o))
391 {
392 let is_real_code = lines.get(i).is_some_and(|info| info.in_code_block);
393 if is_real_code && sub_start.is_none() {
394 let byte_start = if i == start_line { start } else { offset };
395 sub_start = Some(byte_start);
396 } else if !is_real_code && sub_start.is_some() {
397 new_code_blocks.push((sub_start.unwrap(), offset));
398 sub_start = None;
399 }
400 }
401 if let Some(s) = sub_start {
402 new_code_blocks.push((s, end));
403 }
404 }
405 code_blocks = new_code_blocks;
406 }
407
408 if flavor.supports_jsx() {
412 let mut new_code_blocks = Vec::with_capacity(code_blocks.len());
413 for &(start, end) in &code_blocks {
414 let start_line = line_offsets
415 .partition_point(|&offset| offset <= start)
416 .saturating_sub(1);
417 let end_line = line_offsets.partition_point(|&offset| offset < end).min(lines.len());
418
419 let mut sub_start: Option<usize> = None;
420 for (i, &offset) in line_offsets[start_line..end_line]
421 .iter()
422 .enumerate()
423 .map(|(j, o)| (j + start_line, o))
424 {
425 let is_real_code = lines.get(i).is_some_and(|info| info.in_code_block);
426 if is_real_code && sub_start.is_none() {
427 let byte_start = if i == start_line { start } else { offset };
428 sub_start = Some(byte_start);
429 } else if !is_real_code && sub_start.is_some() {
430 new_code_blocks.push((sub_start.unwrap(), offset));
431 sub_start = None;
432 }
433 }
434 if let Some(s) = sub_start {
435 new_code_blocks.push((s, end));
436 }
437 }
438 code_blocks = new_code_blocks;
439
440 let mut jsx_fence_ranges: Vec<(usize, usize)> = Vec::new();
447 let mut run: Option<(usize, usize)> = None;
448 for line in &lines {
449 if line.in_jsx_block && line.in_code_block {
450 let line_end = line.byte_offset + line.byte_len;
451 match &mut run {
452 Some((_, end)) => *end = line_end,
453 None => run = Some((line.byte_offset, line_end)),
454 }
455 } else if let Some(r) = run.take() {
456 jsx_fence_ranges.push(r);
457 }
458 }
459 if let Some(r) = run.take() {
460 jsx_fence_ranges.push(r);
461 }
462 if !jsx_fence_ranges.is_empty() {
463 code_blocks.extend(jsx_fence_ranges);
464 code_blocks.sort_by_key(|&(start, _)| start);
465 }
466 }
467
468 let colon_fence_ranges = profile_section!(
471 "Azure colon fence detection",
472 profile,
473 flavor_detection::detect_azure_colon_fences(content, &mut lines, flavor)
474 );
475 if !colon_fence_ranges.is_empty() {
476 code_blocks.extend(colon_fence_ranges.iter().copied());
477 code_blocks.sort_by_key(|&(start, _)| start);
478 }
479
480 let myst_directive_ranges = profile_section!(
483 "MyST colon directives",
484 profile,
485 flavor_detection::detect_myst_colon_directives(content, &mut lines, flavor)
486 );
487
488 let myst_comment_ranges = profile_section!(
490 "MyST comments",
491 profile,
492 flavor_detection::detect_myst_comments(content, &mut lines, flavor)
493 );
494
495 profile_section!(
498 "MyST backtick directives",
499 profile,
500 flavor_detection::detect_myst_backtick_directives(
501 content,
502 &mut lines,
503 flavor,
504 &code_block_details,
505 &line_offsets
506 )
507 );
508
509 if flavor.supports_myst_directives() {
512 let mut new_code_blocks = Vec::with_capacity(code_blocks.len());
513 for &(start, end) in &code_blocks {
514 let start_line = line_offsets
515 .partition_point(|&offset| offset <= start)
516 .saturating_sub(1);
517 let end_line = line_offsets.partition_point(|&offset| offset < end).min(lines.len());
518
519 let mut sub_start: Option<usize> = None;
520 for (i, &offset) in line_offsets[start_line..end_line]
521 .iter()
522 .enumerate()
523 .map(|(j, o)| (j + start_line, o))
524 {
525 let is_real_code = lines.get(i).is_some_and(|info| info.in_code_block);
526 if is_real_code && sub_start.is_none() {
527 let byte_start = if i == start_line { start } else { offset };
528 sub_start = Some(byte_start);
529 } else if !is_real_code && sub_start.is_some() {
530 new_code_blocks.push((sub_start.unwrap(), offset));
531 sub_start = None;
532 }
533 }
534 if let Some(s) = sub_start {
535 new_code_blocks.push((s, end));
536 }
537 }
538 code_blocks = new_code_blocks;
539 }
540
541 profile_section!(
543 "Kramdown constructs",
544 profile,
545 flavor_detection::detect_kramdown_line_info(content, &mut lines, flavor)
546 );
547
548 for line in &mut lines {
553 if line.in_kramdown_extension_block {
554 line.list_item = None;
555 line.is_horizontal_rule = false;
556 line.blockquote = None;
557 line.is_kramdown_block_ial = false;
558 }
559 }
560
561 let obsidian_comment_scan = profile_section!(
563 "Obsidian comments",
564 profile,
565 flavor_detection::detect_obsidian_comments(
566 content,
567 &mut lines,
568 flavor,
569 &code_span_ranges,
570 &html_comment_ranges,
571 body_start
572 )
573 );
574 let mut obsidian_comment_ranges = obsidian_comment_scan.ranges;
575 let mut unterminated_obsidian_comment = obsidian_comment_scan.unterminated;
576
577 let unterminated_html_comment = crate::utils::skip_context::unterminated_html_comment_outside(
582 unterminated_html_comment,
583 &obsidian_comment_ranges,
584 content,
585 &code_span_ranges,
586 &comment_code_block_ranges,
587 body_start,
588 );
589
590 if let Some(range) = unterminated_html_comment.and_then(|opener| {
603 crate::utils::skip_context::unterminated_comment_range(opener, &html_blocks)
604 .or_else(|| container_comment_range(opener, &containers, &lines, content))
605 }) {
606 html_comment_ranges.push(range);
609
610 for line in &mut lines {
616 let text = line.content(content);
617 let content_start = line.byte_offset + line.indent;
618 let content_end = line.byte_offset + text.trim_end().len();
619 line.in_html_comment = crate::utils::skip_context::is_line_entirely_in_html_comment(
620 &html_comment_ranges,
621 content_start,
622 content_end,
623 );
624 line.in_obsidian_comment = false;
625 }
626
627 let obsidian_rescan = flavor_detection::detect_obsidian_comments(
638 content,
639 &mut lines,
640 flavor,
641 &code_span_ranges,
642 &html_comment_ranges,
643 body_start,
644 );
645 obsidian_comment_ranges = obsidian_rescan.ranges;
646 unterminated_obsidian_comment = obsidian_rescan.unterminated;
647 }
648
649 let myst_role_ranges = profile_section!(
651 "MyST roles",
652 profile,
653 flavor_detection::detect_myst_role_ranges(content, &lines, flavor, &code_blocks)
654 );
655
656 let pulldown_result = profile_section!(
660 "Links, images & link ranges",
661 profile,
662 link_parser::parse_links_images_pulldown(content, &lines, &code_blocks, flavor, &html_comment_ranges)
663 );
664
665 profile_section!(
667 "Headings & blockquotes",
668 profile,
669 heading_detection::detect_headings_and_blockquotes(
670 &content_lines,
671 &mut lines,
672 flavor,
673 &html_comment_ranges,
674 &pulldown_result.link_byte_ranges,
675 front_matter_end,
676 )
677 );
678
679 for line in &mut lines {
681 if line.in_kramdown_extension_block {
682 line.heading = None;
683 }
684 }
685
686 for line in &mut lines {
697 if line.is_horizontal_rule
698 && (line.in_code_block
699 || line.in_html_block
700 || line.in_html_comment
701 || line.in_math_block
702 || line.in_mdx_comment
703 || line.in_obsidian_comment)
704 {
705 line.is_horizontal_rule = false;
706 }
707 }
708
709 let mut code_spans = profile_section!(
711 "Code spans",
712 profile,
713 element_parsers::build_code_spans_from_ranges(content, &lines, &code_span_ranges)
714 );
715
716 if flavor == MarkdownFlavor::MkDocs {
720 let extra = profile_section!(
721 "MkDocs code spans",
722 profile,
723 element_parsers::scan_mkdocs_container_code_spans(content, &lines, &code_span_ranges,)
724 );
725 if !extra.is_empty() {
726 code_spans.extend(extra);
727 code_spans.sort_by_key(|span| span.byte_offset);
728 }
729 }
730
731 if flavor == MarkdownFlavor::MDX {
736 let extra = profile_section!(
737 "MDX JSX code spans",
738 profile,
739 element_parsers::scan_jsx_block_code_spans(content, &lines, &code_span_ranges)
740 );
741 if !extra.is_empty() {
742 code_spans.extend(extra);
743 code_spans.sort_by_key(|span| span.byte_offset);
744 }
745 }
746
747 for span in &code_spans {
750 if span.end_line > span.line {
751 for line_num in (span.line + 1)..=span.end_line {
753 if let Some(line_info) = lines.get_mut(line_num - 1) {
754 line_info.in_code_span_continuation = true;
755 }
756 }
757 }
758 }
759
760 let (links, images, broken_links, footnote_refs) = profile_section!(
762 "Links & images finalize",
763 profile,
764 link_parser::finalize_links_and_images(
765 content,
766 &lines,
767 &code_blocks,
768 &code_spans,
769 flavor,
770 &html_comment_ranges,
771 pulldown_result
772 )
773 );
774
775 let reference_defs = profile_section!(
776 "Reference defs",
777 profile,
778 link_parser::parse_reference_defs(content, &lines)
779 );
780
781 let list_blocks = profile_section!("List blocks", profile, list_blocks::parse_list_blocks(content, &lines));
782
783 let char_frequency = profile_section!(
785 "Char frequency",
786 profile,
787 line_computation::compute_char_frequency(content)
788 );
789
790 let table_blocks = profile_section!(
792 "Table blocks",
793 profile,
794 crate::utils::table_utils::TableUtils::find_table_blocks_with_code_info(
795 content,
796 &code_blocks,
797 &code_spans,
798 &html_comment_ranges,
799 )
800 );
801
802 let links = links
805 .into_iter()
806 .filter(|link| !lines.get(link.line - 1).is_some_and(|l| l.in_kramdown_extension_block))
807 .collect::<Vec<_>>();
808 let images = images
809 .into_iter()
810 .filter(|img| !lines.get(img.line - 1).is_some_and(|l| l.in_kramdown_extension_block))
811 .collect::<Vec<_>>();
812 let broken_links = broken_links
813 .into_iter()
814 .filter(|bl| {
815 let line_idx = line_offsets
817 .partition_point(|&offset| offset <= bl.span.start)
818 .saturating_sub(1);
819 !lines.get(line_idx).is_some_and(|l| l.in_kramdown_extension_block)
820 })
821 .collect::<Vec<_>>();
822 let footnote_refs = footnote_refs
823 .into_iter()
824 .filter(|fr| !lines.get(fr.line - 1).is_some_and(|l| l.in_kramdown_extension_block))
825 .collect::<Vec<_>>();
826 let reference_defs = reference_defs
827 .into_iter()
828 .filter(|def| !lines.get(def.line - 1).is_some_and(|l| l.in_kramdown_extension_block))
829 .collect::<Vec<_>>();
830 let list_blocks = list_blocks
831 .into_iter()
832 .filter(|block| {
833 !lines
834 .get(block.start_line - 1)
835 .is_some_and(|l| l.in_kramdown_extension_block)
836 })
837 .collect::<Vec<_>>();
838 let table_blocks = table_blocks
839 .into_iter()
840 .filter(|block| {
841 !lines
843 .get(block.start_line)
844 .is_some_and(|l| l.in_kramdown_extension_block)
845 })
846 .collect::<Vec<_>>();
847 let emphasis_spans = emphasis_spans
848 .into_iter()
849 .filter(|span| !lines.get(span.line - 1).is_some_and(|l| l.in_kramdown_extension_block))
850 .collect::<Vec<_>>();
851
852 for block in &list_blocks {
856 for line_num in block.start_line..=block.end_line {
858 if let Some(li) = lines.get_mut(line_num - 1) {
859 li.in_list_block = true;
860 }
861 }
862 }
863 for block in &table_blocks {
864 for idx in block.start_line..=block.end_line {
866 if let Some(li) = lines.get_mut(idx) {
867 li.in_table_block = true;
868 }
869 }
870 }
871
872 let reference_defs_map: HashMap<String, usize> = reference_defs
874 .iter()
875 .enumerate()
876 .map(|(idx, def)| (def.id.to_lowercase(), idx))
877 .collect();
878
879 let link_title_ranges: Vec<(usize, usize)> = reference_defs
881 .iter()
882 .filter_map(|def| match (def.title_byte_start, def.title_byte_end) {
883 (Some(start), Some(end)) => Some((start, end)),
884 _ => None,
885 })
886 .collect();
887
888 let line_index = profile_section!(
890 "Line index",
891 profile,
892 crate::utils::range_utils::LineIndex::with_line_starts_and_code_blocks(
893 content,
894 line_offsets.clone(),
895 &code_blocks,
896 )
897 );
898
899 let jinja_ranges = profile_section!(
901 "Jinja ranges",
902 profile,
903 crate::utils::jinja_utils::find_jinja_ranges(content)
904 );
905
906 let citation_ranges = profile_section!("Citation ranges", profile, {
908 if flavor.is_pandoc_compatible() {
909 crate::utils::pandoc::find_citation_ranges(content)
910 } else {
911 Vec::new()
912 }
913 });
914
915 let inline_footnote_ranges = profile_section!("Inline footnote ranges", profile, {
917 if flavor.is_pandoc_compatible() {
918 crate::utils::pandoc::detect_inline_footnote_ranges(content)
919 } else {
920 Vec::new()
921 }
922 });
923
924 let pandoc_header_slugs = profile_section!("Pandoc header slugs", profile, {
926 if flavor.is_pandoc_compatible() {
927 crate::utils::pandoc::collect_pandoc_header_slugs(content)
928 } else {
929 std::collections::HashSet::new()
930 }
931 });
932
933 let example_list_marker_ranges = profile_section!("Example list markers", profile, {
935 if flavor.is_pandoc_compatible() {
936 crate::utils::pandoc::detect_example_list_marker_ranges(content)
937 } else {
938 Vec::new()
939 }
940 });
941
942 let example_reference_ranges = profile_section!("Example references", profile, {
944 if flavor.is_pandoc_compatible() {
945 crate::utils::pandoc::detect_example_reference_ranges(content, &example_list_marker_ranges)
946 } else {
947 Vec::new()
948 }
949 });
950
951 let sub_super_ranges = profile_section!("Subscript/superscript ranges", profile, {
953 if flavor.is_pandoc_compatible() {
954 crate::utils::pandoc::detect_subscript_superscript_ranges(content)
955 } else {
956 Vec::new()
957 }
958 });
959
960 let inline_code_attr_ranges = profile_section!("Inline code attribute ranges", profile, {
962 if flavor.is_pandoc_compatible() {
963 crate::utils::pandoc::detect_inline_code_attr_ranges(content)
964 } else {
965 Vec::new()
966 }
967 });
968
969 let bracketed_span_ranges = profile_section!("Bracketed span ranges", profile, {
971 if flavor.is_pandoc_compatible() {
972 crate::utils::pandoc::detect_bracketed_span_ranges(content)
973 } else {
974 Vec::new()
975 }
976 });
977
978 let line_block_ranges = profile_section!("Line block ranges", profile, {
980 if flavor.is_pandoc_compatible() {
981 crate::utils::pandoc::detect_line_block_ranges(content)
982 } else {
983 Vec::new()
984 }
985 });
986
987 let pipe_table_caption_ranges = profile_section!("Pipe-table caption ranges", profile, {
989 if flavor.is_pandoc_compatible() {
990 crate::utils::pandoc::detect_pipe_table_caption_ranges(content)
991 } else {
992 Vec::new()
993 }
994 });
995
996 let pandoc_metadata_ranges = profile_section!("Pandoc metadata ranges", profile, {
998 if flavor.is_pandoc_compatible() {
999 crate::utils::pandoc::detect_yaml_metadata_block_ranges(content)
1000 } else {
1001 Vec::new()
1002 }
1003 });
1004
1005 let grid_table_ranges = profile_section!("Grid table ranges", profile, {
1007 if flavor.is_pandoc_compatible() {
1008 crate::utils::pandoc::detect_grid_table_ranges(content)
1009 } else {
1010 Vec::new()
1011 }
1012 });
1013
1014 let multi_line_table_ranges = profile_section!("Multi-line table ranges", profile, {
1016 if flavor.is_pandoc_compatible() {
1017 crate::utils::pandoc::detect_multi_line_table_ranges(content)
1018 } else {
1019 Vec::new()
1020 }
1021 });
1022
1023 let shortcode_ranges = profile_section!("Shortcode ranges", profile, {
1025 use crate::utils::regex_cache::HUGO_SHORTCODE_REGEX;
1026 let mut ranges = Vec::new();
1027 for mat in HUGO_SHORTCODE_REGEX.find_iter(content) {
1028 ranges.push((mat.start(), mat.end()));
1029 }
1030 ranges
1031 });
1032
1033 let inline_config = InlineConfig::from_content_with_code_blocks(content, &code_blocks);
1034
1035 Self {
1036 content,
1037 content_lines,
1038 line_offsets,
1039 code_blocks,
1040 code_block_details,
1041 strong_spans,
1042 line_to_list,
1043 list_start_values,
1044 lines,
1045 links,
1046 images,
1047 broken_links,
1048 footnote_refs,
1049 reference_defs,
1050 reference_defs_map,
1051 code_spans_cache: OnceLock::from(Arc::new(code_spans)),
1052 math_spans_cache: OnceLock::new(), math_byte_ranges_cache: OnceLock::new(), list_blocks,
1055 char_frequency,
1056 html_tags_cache: OnceLock::new(),
1057 jsx_component_tags_cache: OnceLock::new(),
1058 emphasis_spans_cache: OnceLock::from(Arc::new(emphasis_spans)),
1059 bare_urls_cache: OnceLock::new(),
1060 has_mixed_list_nesting_cache: OnceLock::new(),
1061 html_comment_ranges,
1062 table_blocks,
1063 line_index,
1064 jinja_ranges,
1065 flavor,
1066 source_file,
1067 jsx_expression_ranges,
1068 mdx_comment_ranges,
1069 citation_ranges,
1070 pandoc_div_ranges,
1071 colon_fence_ranges,
1072 inline_footnote_ranges,
1073 pandoc_header_slugs,
1074 example_list_marker_ranges,
1075 example_reference_ranges,
1076 sub_super_ranges,
1077 inline_code_attr_ranges,
1078 bracketed_span_ranges,
1079 line_block_ranges,
1080 pipe_table_caption_ranges,
1081 pandoc_metadata_ranges,
1082 grid_table_ranges,
1083 multi_line_table_ranges,
1084 shortcode_ranges,
1085 link_title_ranges,
1086 code_span_byte_ranges: code_span_ranges,
1087 inline_config,
1088 obsidian_comment_ranges,
1089 unterminated_html_comment,
1090 unterminated_obsidian_comment,
1091 lazy_cont_lines_cache: OnceLock::new(),
1092 myst_directive_ranges,
1093 myst_comment_ranges,
1094 myst_role_ranges,
1095 front_matter_end,
1096 }
1097 }
1098
1099 pub fn front_matter_end_line(&self) -> usize {
1104 self.front_matter_end
1105 }
1106
1107 #[inline]
1110 fn binary_search_ranges(ranges: &[(usize, usize)], pos: usize) -> bool {
1111 let idx = ranges.partition_point(|&(start, _)| start <= pos);
1113 idx > 0 && pos < ranges[idx - 1].1
1115 }
1116
1117 pub fn is_in_code_span_byte(&self, pos: usize) -> bool {
1119 Self::binary_search_ranges(&self.code_span_byte_ranges, pos)
1120 }
1121
1122 pub fn is_in_link(&self, pos: usize) -> bool {
1124 let idx = self.links.partition_point(|link| link.byte_offset <= pos);
1125 if idx > 0 && pos < self.links[idx - 1].byte_end {
1126 return true;
1127 }
1128 let idx = self.images.partition_point(|img| img.byte_offset <= pos);
1129 if idx > 0 && pos < self.images[idx - 1].byte_end {
1130 return true;
1131 }
1132 self.is_in_reference_def(pos)
1133 }
1134
1135 pub fn is_in_bare_url(&self, pos: usize) -> bool {
1137 let bare_urls = self.bare_urls();
1138 let idx = bare_urls.partition_point(|url| url.byte_offset <= pos);
1140 idx > 0 && pos < bare_urls[idx - 1].byte_end
1141 }
1142
1143 pub fn inline_config(&self) -> &InlineConfig {
1145 &self.inline_config
1146 }
1147
1148 pub fn colon_fence_ranges(&self) -> &[(usize, usize)] {
1151 &self.colon_fence_ranges
1152 }
1153
1154 pub fn raw_lines(&self) -> &[&'a str] {
1158 &self.content_lines
1159 }
1160
1161 pub fn is_rule_disabled(&self, rule_name: &str, line_number: usize) -> bool {
1166 self.inline_config.is_rule_disabled(rule_name, line_number)
1167 }
1168
1169 pub fn code_spans(&self) -> Arc<Vec<CodeSpan>> {
1171 Arc::clone(
1172 self.code_spans_cache
1173 .get_or_init(|| Arc::new(element_parsers::parse_code_spans(self.content, &self.lines))),
1174 )
1175 }
1176
1177 pub fn math_byte_ranges(&self) -> &[(usize, usize)] {
1181 self.math_byte_ranges_cache
1182 .get_or_init(|| crate::utils::skip_context::math_byte_ranges(self.content))
1183 }
1184
1185 pub fn math_spans(&self) -> Arc<Vec<MathSpan>> {
1187 Arc::clone(
1188 self.math_spans_cache
1189 .get_or_init(|| Arc::new(element_parsers::parse_math_spans(self.content, &self.lines))),
1190 )
1191 }
1192
1193 pub fn is_in_math_span(&self, byte_pos: usize) -> bool {
1195 let math_spans = self.math_spans();
1196 let idx = math_spans.partition_point(|span| span.byte_offset <= byte_pos);
1198 idx > 0 && byte_pos < math_spans[idx - 1].byte_end
1199 }
1200
1201 pub fn html_comment_ranges(&self) -> &[crate::utils::skip_context::ByteRange] {
1203 &self.html_comment_ranges
1204 }
1205
1206 pub fn unterminated_html_comment(&self) -> Option<usize> {
1211 self.unterminated_html_comment
1212 }
1213
1214 pub fn unterminated_obsidian_comment(&self) -> Option<usize> {
1218 self.unterminated_obsidian_comment
1219 }
1220
1221 pub fn is_in_obsidian_comment(&self, byte_pos: usize) -> bool {
1225 Self::binary_search_ranges(&self.obsidian_comment_ranges, byte_pos)
1226 }
1227
1228 pub fn is_position_in_obsidian_comment(&self, line_num: usize, col: usize) -> bool {
1233 if self.obsidian_comment_ranges.is_empty() {
1234 return false;
1235 }
1236
1237 let byte_pos = self.line_index.line_col_to_byte_range(line_num, col).start;
1239 self.is_in_obsidian_comment(byte_pos)
1240 }
1241
1242 pub fn myst_directive_ranges(&self) -> &[(usize, usize)] {
1244 &self.myst_directive_ranges
1245 }
1246
1247 pub fn is_in_myst_role(&self, byte_pos: usize) -> bool {
1249 Self::binary_search_ranges(&self.myst_role_ranges, byte_pos)
1250 }
1251
1252 pub fn is_in_myst_comment(&self, byte_pos: usize) -> bool {
1254 Self::binary_search_ranges(&self.myst_comment_ranges, byte_pos)
1255 }
1256
1257 pub fn is_myst_colon_directive_opener_line(&self, line_num: usize) -> bool {
1264 if !self.flavor.supports_myst_directives() {
1265 return false;
1266 }
1267 self.lines.get(line_num.wrapping_sub(1)).is_some_and(|info| {
1268 info.in_myst_directive
1269 && flavor_detection::myst_colon_directive_opener(info.content(self.content)).is_some()
1270 })
1271 }
1272
1273 fn filter_kramdown_tags(&self, tags: Vec<HtmlTag>) -> Vec<HtmlTag> {
1275 tags.into_iter()
1276 .filter(|tag| {
1277 !self
1278 .lines
1279 .get(tag.line - 1)
1280 .is_some_and(|l| l.in_kramdown_extension_block)
1281 })
1282 .collect()
1283 }
1284
1285 pub fn html_tags(&self) -> Arc<Vec<HtmlTag>> {
1291 Arc::clone(self.html_tags_cache.get_or_init(|| {
1292 let (html_tags, jsx_component_tags) =
1293 element_parsers::parse_html_tags(self.content, &self.lines, &self.code_blocks, self.flavor);
1294 let _ = self
1296 .jsx_component_tags_cache
1297 .set(Arc::new(self.filter_kramdown_tags(jsx_component_tags)));
1298 Arc::new(self.filter_kramdown_tags(html_tags))
1299 }))
1300 }
1301
1302 pub fn jsx_component_tags(&self) -> Arc<Vec<HtmlTag>> {
1305 if let Some(cached) = self.jsx_component_tags_cache.get() {
1306 return Arc::clone(cached);
1307 }
1308 let _ = self.html_tags();
1310 Arc::clone(
1311 self.jsx_component_tags_cache
1312 .get()
1313 .expect("html_tags() populates jsx_component_tags_cache"),
1314 )
1315 }
1316
1317 pub fn emphasis_spans(&self) -> Arc<Vec<EmphasisSpan>> {
1319 Arc::clone(
1320 self.emphasis_spans_cache
1321 .get()
1322 .expect("emphasis_spans_cache initialized during construction"),
1323 )
1324 }
1325
1326 pub fn bare_urls(&self) -> Arc<Vec<BareUrl>> {
1328 Arc::clone(self.bare_urls_cache.get_or_init(|| {
1329 Arc::new(element_parsers::parse_bare_urls(
1330 self.content,
1331 &self.lines,
1332 &self.code_blocks,
1333 ))
1334 }))
1335 }
1336
1337 pub fn lazy_continuation_lines(&self) -> Arc<Vec<LazyContLine>> {
1339 Arc::clone(self.lazy_cont_lines_cache.get_or_init(|| {
1340 Arc::new(element_parsers::detect_lazy_continuation_lines(
1341 self.content,
1342 &self.lines,
1343 &self.line_offsets,
1344 ))
1345 }))
1346 }
1347
1348 pub fn has_mixed_list_nesting(&self) -> bool {
1352 *self
1353 .has_mixed_list_nesting_cache
1354 .get_or_init(|| self.compute_mixed_list_nesting())
1355 }
1356
1357 fn compute_mixed_list_nesting(&self) -> bool {
1359 let mut stack: Vec<(usize, bool)> = Vec::new();
1364 let mut last_was_blank = false;
1365
1366 for line_info in &self.lines {
1367 if line_info.in_code_block
1369 || line_info.in_front_matter
1370 || line_info.in_mkdocstrings
1371 || line_info.in_html_comment
1372 || line_info.in_mdx_comment
1373 || line_info.in_esm_block
1374 {
1375 continue;
1376 }
1377
1378 if line_info.is_blank {
1380 last_was_blank = true;
1381 continue;
1382 }
1383
1384 if let Some(list_item) = &line_info.list_item {
1385 let current_pos = if list_item.marker_column == 1 {
1387 0
1388 } else {
1389 list_item.marker_column
1390 };
1391
1392 if last_was_blank && current_pos == 0 {
1394 stack.clear();
1395 }
1396 last_was_blank = false;
1397
1398 while let Some(&(pos, _)) = stack.last() {
1400 if pos >= current_pos {
1401 stack.pop();
1402 } else {
1403 break;
1404 }
1405 }
1406
1407 if let Some(&(_, parent_is_ordered)) = stack.last()
1409 && parent_is_ordered != list_item.is_ordered
1410 {
1411 return true; }
1413
1414 stack.push((current_pos, list_item.is_ordered));
1415 } else {
1416 last_was_blank = false;
1418 }
1419 }
1420
1421 false
1422 }
1423
1424 pub fn offset_to_line_col(&self, offset: usize) -> (usize, usize) {
1430 match self.line_offsets.binary_search(&offset) {
1431 Ok(line) => (line + 1, 1),
1432 Err(line) => {
1433 let line_start = self.line_offsets.get(line.wrapping_sub(1)).copied().unwrap_or(0);
1434 let col = byte_to_char_count(&self.content[line_start..], offset.saturating_sub(line_start));
1436 (line, col)
1437 }
1438 }
1439 }
1440
1441 pub fn is_in_code_block_or_span(&self, pos: usize) -> bool {
1443 if CodeBlockUtils::is_in_code_block_or_span(&self.code_blocks, pos) {
1445 return true;
1446 }
1447
1448 self.is_byte_offset_in_code_span(pos)
1450 }
1451
1452 pub fn line_info(&self, line_num: usize) -> Option<&LineInfo> {
1454 if line_num > 0 {
1455 self.lines.get(line_num - 1)
1456 } else {
1457 None
1458 }
1459 }
1460
1461 pub fn get_reference_url(&self, ref_id: &str) -> Option<&str> {
1463 let normalized_id = ref_id.to_lowercase();
1464 self.reference_defs_map
1465 .get(&normalized_id)
1466 .map(|&idx| self.reference_defs[idx].url.as_str())
1467 }
1468
1469 pub fn is_in_list_block(&self, line_num: usize) -> bool {
1471 if line_num == 0 || line_num > self.lines.len() {
1472 return false;
1473 }
1474 self.lines[line_num - 1].in_list_block
1475 }
1476
1477 pub fn is_in_html_block(&self, line_num: usize) -> bool {
1479 if line_num == 0 || line_num > self.lines.len() {
1480 return false;
1481 }
1482 self.lines[line_num - 1].in_html_block
1483 }
1484
1485 pub fn is_in_table_block(&self, line_num: usize) -> bool {
1491 if line_num == 0 || line_num > self.lines.len() {
1492 return false;
1493 }
1494 self.lines[line_num - 1].in_table_block
1495 }
1496
1497 pub fn is_in_code_span(&self, line_num: usize, col: usize) -> bool {
1499 if line_num == 0 || line_num > self.lines.len() {
1500 return false;
1501 }
1502
1503 let col_0indexed = if col > 0 { col - 1 } else { 0 };
1507 let code_spans = self.code_spans();
1508 code_spans.iter().any(|span| {
1509 if line_num < span.line || line_num > span.end_line {
1511 return false;
1512 }
1513
1514 if span.line == span.end_line {
1515 col_0indexed >= span.start_col && col_0indexed < span.end_col
1517 } else if line_num == span.line {
1518 col_0indexed >= span.start_col
1520 } else if line_num == span.end_line {
1521 col_0indexed < span.end_col
1523 } else {
1524 true
1526 }
1527 })
1528 }
1529
1530 #[inline]
1532 pub fn is_byte_offset_in_code_span(&self, byte_offset: usize) -> bool {
1533 let code_spans = self.code_spans();
1534 let idx = code_spans.partition_point(|span| span.byte_offset <= byte_offset);
1535 idx > 0 && byte_offset < code_spans[idx - 1].byte_end
1536 }
1537
1538 #[inline]
1540 pub fn is_in_reference_def(&self, byte_pos: usize) -> bool {
1541 let idx = self.reference_defs.partition_point(|rd| rd.byte_offset <= byte_pos);
1542 idx > 0 && byte_pos < self.reference_defs[idx - 1].byte_end
1543 }
1544
1545 #[inline]
1547 pub fn is_in_html_comment(&self, byte_pos: usize) -> bool {
1548 let idx = self.html_comment_ranges.partition_point(|r| r.start <= byte_pos);
1549 idx > 0 && byte_pos < self.html_comment_ranges[idx - 1].end
1550 }
1551
1552 #[inline]
1555 pub fn is_in_html_tag(&self, byte_pos: usize) -> bool {
1556 let tags = self.html_tags();
1557 let idx = tags.partition_point(|tag| tag.byte_offset <= byte_pos);
1558 idx > 0 && byte_pos < tags[idx - 1].byte_end
1559 }
1560
1561 #[inline]
1565 pub fn is_in_jsx_component_tag(&self, byte_pos: usize) -> bool {
1566 if !self.flavor.supports_jsx() {
1567 return false;
1568 }
1569 let tags = self.jsx_component_tags();
1570 let idx = tags.partition_point(|tag| tag.byte_offset <= byte_pos);
1571 idx > 0 && byte_pos < tags[idx - 1].byte_end
1572 }
1573
1574 pub fn is_in_jinja_range(&self, byte_pos: usize) -> bool {
1576 Self::binary_search_ranges(&self.jinja_ranges, byte_pos)
1577 }
1578
1579 #[inline]
1581 pub fn is_in_jsx_expression(&self, byte_pos: usize) -> bool {
1582 Self::binary_search_ranges(&self.jsx_expression_ranges, byte_pos)
1583 }
1584
1585 #[inline]
1587 pub fn is_in_mdx_comment(&self, byte_pos: usize) -> bool {
1588 Self::binary_search_ranges(&self.mdx_comment_ranges, byte_pos)
1589 }
1590
1591 #[inline]
1594 pub fn is_in_citation(&self, byte_pos: usize) -> bool {
1595 let idx = self.citation_ranges.partition_point(|r| r.start <= byte_pos);
1596 idx > 0 && byte_pos < self.citation_ranges[idx - 1].end
1597 }
1598
1599 #[inline]
1601 pub fn citation_ranges(&self) -> &[crate::utils::skip_context::ByteRange] {
1602 &self.citation_ranges
1603 }
1604
1605 #[inline]
1608 pub fn is_in_div_block(&self, byte_pos: usize) -> bool {
1609 let idx = self.pandoc_div_ranges.partition_point(|r| r.start <= byte_pos);
1610 idx > 0 && byte_pos < self.pandoc_div_ranges[idx - 1].end
1611 }
1612
1613 #[inline]
1616 pub fn is_in_inline_footnote(&self, byte_pos: usize) -> bool {
1617 let idx = self.inline_footnote_ranges.partition_point(|r| r.start <= byte_pos);
1618 idx > 0 && byte_pos < self.inline_footnote_ranges[idx - 1].end
1619 }
1620
1621 #[inline]
1624 pub fn is_in_example_list_marker(&self, byte_pos: usize) -> bool {
1625 let idx = self.example_list_marker_ranges.partition_point(|r| r.start <= byte_pos);
1626 idx > 0 && byte_pos < self.example_list_marker_ranges[idx - 1].end
1627 }
1628
1629 #[inline]
1632 pub fn is_in_example_reference(&self, byte_pos: usize) -> bool {
1633 let idx = self.example_reference_ranges.partition_point(|r| r.start <= byte_pos);
1634 idx > 0 && byte_pos < self.example_reference_ranges[idx - 1].end
1635 }
1636
1637 #[inline]
1640 pub fn is_in_subscript_or_superscript(&self, byte_pos: usize) -> bool {
1641 let idx = self.sub_super_ranges.partition_point(|r| r.start <= byte_pos);
1642 idx > 0 && byte_pos < self.sub_super_ranges[idx - 1].end
1643 }
1644
1645 #[inline]
1649 pub fn is_in_inline_code_attr(&self, byte_pos: usize) -> bool {
1650 let idx = self.inline_code_attr_ranges.partition_point(|r| r.start <= byte_pos);
1651 idx > 0 && byte_pos < self.inline_code_attr_ranges[idx - 1].end
1652 }
1653
1654 #[inline]
1657 pub fn is_in_bracketed_span(&self, byte_pos: usize) -> bool {
1658 let idx = self.bracketed_span_ranges.partition_point(|r| r.start <= byte_pos);
1659 idx > 0 && byte_pos < self.bracketed_span_ranges[idx - 1].end
1660 }
1661
1662 #[inline]
1665 pub fn is_in_line_block(&self, byte_pos: usize) -> bool {
1666 let idx = self.line_block_ranges.partition_point(|r| r.start <= byte_pos);
1667 idx > 0 && byte_pos < self.line_block_ranges[idx - 1].end
1668 }
1669
1670 #[inline]
1674 pub fn is_in_pipe_table_caption(&self, byte_pos: usize) -> bool {
1675 let idx = self.pipe_table_caption_ranges.partition_point(|r| r.start <= byte_pos);
1676 idx > 0 && byte_pos < self.pipe_table_caption_ranges[idx - 1].end
1677 }
1678
1679 #[inline]
1682 pub fn is_in_pandoc_metadata(&self, byte_pos: usize) -> bool {
1683 let idx = self.pandoc_metadata_ranges.partition_point(|r| r.start <= byte_pos);
1684 idx > 0 && byte_pos < self.pandoc_metadata_ranges[idx - 1].end
1685 }
1686
1687 #[inline]
1690 pub fn is_in_grid_table(&self, byte_pos: usize) -> bool {
1691 let idx = self.grid_table_ranges.partition_point(|r| r.start <= byte_pos);
1692 idx > 0 && byte_pos < self.grid_table_ranges[idx - 1].end
1693 }
1694
1695 #[inline]
1698 pub fn is_in_multi_line_table(&self, byte_pos: usize) -> bool {
1699 let idx = self.multi_line_table_ranges.partition_point(|r| r.start <= byte_pos);
1700 idx > 0 && byte_pos < self.multi_line_table_ranges[idx - 1].end
1701 }
1702
1703 pub fn matches_implicit_header_reference(&self, link_text: &str) -> bool {
1708 let slug = crate::utils::pandoc::pandoc_header_slug(link_text);
1709 self.pandoc_header_slugs.contains(&slug)
1710 }
1711
1712 #[inline]
1718 pub fn has_pandoc_slug(&self, slug: &str) -> bool {
1719 self.pandoc_header_slugs.contains(slug)
1720 }
1721
1722 #[inline]
1724 pub fn is_in_shortcode(&self, byte_pos: usize) -> bool {
1725 Self::binary_search_ranges(&self.shortcode_ranges, byte_pos)
1726 }
1727
1728 #[inline]
1730 pub fn shortcode_ranges(&self) -> &[(usize, usize)] {
1731 &self.shortcode_ranges
1732 }
1733
1734 pub fn is_in_link_title(&self, byte_pos: usize) -> bool {
1736 Self::binary_search_ranges(&self.link_title_ranges, byte_pos)
1737 }
1738
1739 pub fn has_char(&self, ch: char) -> bool {
1741 match ch {
1742 '#' => self.char_frequency.hash_count > 0,
1743 '*' => self.char_frequency.asterisk_count > 0,
1744 '_' => self.char_frequency.underscore_count > 0,
1745 '-' => self.char_frequency.hyphen_count > 0,
1746 '+' => self.char_frequency.plus_count > 0,
1747 '>' => self.char_frequency.gt_count > 0,
1748 '|' => self.char_frequency.pipe_count > 0,
1749 '[' => self.char_frequency.bracket_count > 0,
1750 '`' => self.char_frequency.backtick_count > 0,
1751 '<' => self.char_frequency.lt_count > 0,
1752 '!' => self.char_frequency.exclamation_count > 0,
1753 '\n' => self.char_frequency.newline_count > 0,
1754 _ => self.content.contains(ch), }
1756 }
1757
1758 pub fn char_count(&self, ch: char) -> usize {
1760 match ch {
1761 '#' => self.char_frequency.hash_count,
1762 '*' => self.char_frequency.asterisk_count,
1763 '_' => self.char_frequency.underscore_count,
1764 '-' => self.char_frequency.hyphen_count,
1765 '+' => self.char_frequency.plus_count,
1766 '>' => self.char_frequency.gt_count,
1767 '|' => self.char_frequency.pipe_count,
1768 '[' => self.char_frequency.bracket_count,
1769 '`' => self.char_frequency.backtick_count,
1770 '<' => self.char_frequency.lt_count,
1771 '!' => self.char_frequency.exclamation_count,
1772 '\n' => self.char_frequency.newline_count,
1773 _ => self.content.matches(ch).count(), }
1775 }
1776
1777 pub fn likely_has_headings(&self) -> bool {
1779 self.char_frequency.hash_count > 0 || self.char_frequency.hyphen_count > 2 || self.content.contains('=') }
1781
1782 pub fn likely_has_lists(&self) -> bool {
1784 self.char_frequency.asterisk_count > 0
1785 || self.char_frequency.hyphen_count > 0
1786 || self.char_frequency.plus_count > 0
1787 }
1788
1789 pub fn likely_has_emphasis(&self) -> bool {
1791 self.char_frequency.asterisk_count > 1 || self.char_frequency.underscore_count > 1
1792 }
1793
1794 pub fn likely_has_tables(&self) -> bool {
1796 self.char_frequency.pipe_count > 2
1797 }
1798
1799 pub fn likely_has_blockquotes(&self) -> bool {
1801 self.char_frequency.gt_count > 0
1802 }
1803
1804 pub fn likely_has_code(&self) -> bool {
1806 self.char_frequency.backtick_count > 0
1807 }
1808
1809 pub fn likely_has_links_or_images(&self) -> bool {
1811 self.char_frequency.bracket_count > 0 || self.char_frequency.exclamation_count > 0
1812 }
1813
1814 pub fn likely_has_html(&self) -> bool {
1816 self.char_frequency.lt_count > 0
1817 }
1818
1819 pub fn blockquote_prefix_for_blank_line(&self, line_idx: usize) -> String {
1824 if let Some(line_info) = self.lines.get(line_idx)
1825 && let Some(ref bq) = line_info.blockquote
1826 {
1827 bq.prefix.trim_end().to_string()
1828 } else {
1829 String::new()
1830 }
1831 }
1832
1833 #[inline]
1844 fn find_line_for_offset(lines: &[LineInfo], content: &str, byte_offset: usize) -> (usize, usize, usize) {
1845 let idx = match lines.binary_search_by(|line| {
1847 if byte_offset < line.byte_offset {
1848 std::cmp::Ordering::Greater
1849 } else if byte_offset > line.byte_offset + line.byte_len {
1850 std::cmp::Ordering::Less
1851 } else {
1852 std::cmp::Ordering::Equal
1853 }
1854 }) {
1855 Ok(idx) => idx,
1856 Err(idx) => idx.saturating_sub(1),
1857 };
1858
1859 let line = &lines[idx];
1860 let line_num = idx + 1;
1861 let byte_col = byte_offset.saturating_sub(line.byte_offset);
1862 let col = byte_to_char_count(line.content(content), byte_col) - 1;
1865
1866 (idx, line_num, col)
1867 }
1868
1869 #[inline]
1871 fn is_offset_in_code_span(code_spans: &[CodeSpan], offset: usize) -> bool {
1872 let idx = code_spans.partition_point(|span| span.byte_offset <= offset);
1874
1875 if idx > 0 {
1877 let span = &code_spans[idx - 1];
1878 if offset >= span.byte_offset && offset < span.byte_end {
1879 return true;
1880 }
1881 }
1882
1883 false
1884 }
1885
1886 #[must_use]
1906 pub fn valid_headings(&self) -> ValidHeadingsIter<'_> {
1907 ValidHeadingsIter::new(&self.lines)
1908 }
1909
1910 #[must_use]
1914 pub fn has_valid_headings(&self) -> bool {
1915 self.lines
1916 .iter()
1917 .any(|line| line.heading.as_ref().is_some_and(|h| h.is_valid))
1918 }
1919}
1920
1921fn container_comment_range(
1933 opener: usize,
1934 containers: &flavor_detection::ContainerLines,
1935 lines: &[types::LineInfo],
1936 content: &str,
1937) -> Option<crate::utils::skip_context::ByteRange> {
1938 let line_index = lines
1939 .partition_point(|line| line.byte_offset <= opener)
1940 .checked_sub(1)?;
1941 let line = lines.get(line_index)?;
1942 if line.byte_offset + line.indent != opener {
1943 return None;
1944 }
1945 if !containers.is_container_body(line_index) {
1946 return None;
1947 }
1948 let end_line = lines.get(containers.body_end_line(line_index)?)?;
1949 Some(crate::utils::skip_context::ByteRange {
1950 start: opener,
1951 end: (end_line.byte_offset + end_line.byte_len).min(content.len()),
1952 })
1953}
1954
1955fn detect_footnote_definitions(content: &str, lines: &mut [types::LineInfo], line_offsets: &[usize]) {
1964 use pulldown_cmark::{CodeBlockKind, Event, Parser, Tag, TagEnd};
1965
1966 let options = crate::utils::rumdl_parser_options();
1967 let parser = Parser::new_ext(content, options).into_offset_iter();
1968
1969 let mut footnote_ranges: Vec<(usize, usize)> = Vec::new();
1971 let mut fenced_code_ranges: Vec<(usize, usize)> = Vec::new();
1972 let mut in_footnote = false;
1973
1974 for (event, range) in parser {
1975 match event {
1976 Event::Start(Tag::FootnoteDefinition(_)) => {
1977 in_footnote = true;
1978 footnote_ranges.push((range.start, range.end));
1979 }
1980 Event::End(TagEnd::FootnoteDefinition) => {
1981 in_footnote = false;
1982 }
1983 Event::Start(Tag::CodeBlock(CodeBlockKind::Fenced(_))) if in_footnote => {
1984 fenced_code_ranges.push((range.start, range.end));
1985 }
1986 _ => {}
1987 }
1988 }
1989
1990 let byte_to_line = |byte_offset: usize| -> usize {
1991 line_offsets
1992 .partition_point(|&offset| offset <= byte_offset)
1993 .saturating_sub(1)
1994 };
1995
1996 for &(start, end) in &footnote_ranges {
1998 let start_line = byte_to_line(start);
1999 let end_line = line_offsets.partition_point(|&offset| offset < end).min(lines.len());
2000
2001 for line in &mut lines[start_line..end_line] {
2002 line.in_footnote_definition = true;
2003 line.in_code_block = false;
2004 }
2005 }
2006
2007 for &(start, end) in &fenced_code_ranges {
2009 let start_line = byte_to_line(start);
2010 let end_line = line_offsets.partition_point(|&offset| offset < end).min(lines.len());
2011
2012 for line in &mut lines[start_line..end_line] {
2013 line.in_code_block = true;
2014 }
2015 }
2016}