Skip to main content

tpnote_lib/
html.rs

1//! Helper functions dealing with HTML conversion.
2use crate::clone_ext::CloneExt;
3use crate::error::InputStreamError;
4use crate::filename::{NotePath, NotePathStr};
5use crate::{
6    config::{HeadingIdPolicy, LocalLinkKind},
7    error::NoteError,
8};
9use html_escape;
10use parking_lot::RwLock;
11use parse_hyperlinks::parser::Link;
12use parse_hyperlinks_extras::iterator_html::HtmlLinkInlineImage;
13use percent_encoding::{AsciiSet, CONTROLS, percent_decode_str, utf8_percent_encode};
14use std::path::MAIN_SEPARATOR_STR;
15use std::{
16    borrow::Cow,
17    collections::HashSet,
18    path::{Component, Path, PathBuf},
19    sync::Arc,
20};
21
22pub(crate) const HTML_EXT: &str = ".html";
23
24/// A local path can carry a format string at the end. This is the separator
25/// character.
26const FORMAT_SEPARATOR: char = '?';
27
28/// If followed directly after FORMAT_SEPARATOR, it selects the sort-tag
29/// for further matching.
30const FORMAT_ONLY_SORT_TAG: char = '#';
31
32/// If followed directly after FORMAT_SEPARATOR, it selects the whole filename
33/// for further matching.
34const FORMAT_COMPLETE_FILENAME: &str = "?";
35
36/// A format string can be separated in a _from_ and _to_ part. This
37/// optional separator is placed after `FORMAT_SEPARATOR` and separates
38/// the _from_ and _to_ pattern.
39const FORMAT_FROM_TO_SEPARATOR: char = ':';
40
41/// Bytes that must be percent-encoded when a filesystem path segment is
42/// embedded in an `href`/`src` attribute. `#` and `?` are URL syntax
43/// (fragment and query introducers): left as-is, a literal one in a
44/// directory or file name is read by the browser as the end of the path
45/// and everything after it never reaches the server. `%` must be in this
46/// set too, so `percent_encode_path()` escapes an existing `%` in a file
47/// name before anything can be mistaken for one of its own escapes. The
48/// space is encoded for consistency, even though browsers already encode
49/// a literal space themselves before sending it.
50static PATH_SEGMENT: &AsciiSet = &CONTROLS.add(b'#').add(b'?').add(b'%').add(b' ');
51
52/// Splits `dest` into a filesystem path and a trailing URL fragment the
53/// author wrote (`note.md#anchor`), mirroring the heuristic used
54/// throughout this module: the last `#` starts a fragment only if it
55/// falls in the final path segment, i.e. after the last `/` or `\`, or
56/// there is no separator at all. A `#` that is part of a directory name
57/// (`Meeting #12/notes.md`) precedes a later separator and is therefore
58/// left in the path half. The returned fragment, if any, keeps its
59/// leading `#`.
60fn split_path_and_fragment(dest: &str) -> (&str, &str) {
61    match (dest.rfind('#'), dest.rfind(['/', '\\'])) {
62        (Some(n), sep) if sep.is_some_and(|sep| n > sep) || sep.is_none() => {
63            (&dest[..n], &dest[n..])
64        }
65        _ => (dest, ""),
66    }
67}
68
69/// Percent-encodes `path` for safe embedding in an `href`/`src` attribute,
70/// segment by segment. Encoding is applied per segment, not to the joined
71/// string, so the `/` separators — including a leading one — never need to
72/// be exempted afterwards, which would risk exempting a `/` that was
73/// actually part of a name. Bytes outside ASCII are always percent-encoded
74/// by `utf8_percent_encode` as their UTF-8 octets.
75fn percent_encode_path(path: &str) -> String {
76    path.split('/')
77        .map(|segment| utf8_percent_encode(segment, PATH_SEGMENT).to_string())
78        .collect::<Vec<_>>()
79        .join("/")
80}
81
82/// If `rewrite_rel_path` and `dest` is relative, concatenate `docdir` and
83/// `dest`, then strip `root_path` from the left before returning.
84/// If not `rewrite_rel_path` and `dest` is relative, return `dest`.
85/// If `rewrite_abs_path` and `dest` is absolute, concatenate and return
86/// `root_path` and `dest`.
87/// If not `rewrite_abs_path` and `dest` is absolute, return `dest`.
88/// The `dest` portion of the output is always canonicalized.
89/// Return the assembled path, when in `root_path`, or `None` otherwise.
90/// Asserts in debug mode, that `doc_dir` is in `root_path`.
91fn assemble_link(
92    root_path: &Path,
93    docdir: &Path,
94    dest: &Path,
95    rewrite_rel_paths: bool,
96    rewrite_abs_paths: bool,
97) -> Option<PathBuf> {
98    ///
99    /// Concatenate `path` and `append`.
100    /// The `append` portion of the output is if possible canonicalized.
101    /// In case of underflow of an absolute link, the returned path is empty.
102    fn append(path: &mut PathBuf, append: &Path) {
103        // Append `dest` to `link` and canonicalize.
104        for dir in append.components() {
105            match dir {
106                Component::ParentDir => {
107                    if !path.pop() {
108                        let path_is_relative = {
109                            let mut c = path.components();
110                            !(c.next() == Some(Component::RootDir)
111                                || c.next() == Some(Component::RootDir))
112                        };
113                        if path_is_relative {
114                            path.push(Component::ParentDir.as_os_str());
115                        } else {
116                            path.clear();
117                            break;
118                        }
119                    }
120                }
121                Component::Normal(c) => path.push(c),
122                _ => {}
123            }
124        }
125    }
126
127    // Under Windows `.is_relative()` does not detect `Component::RootDir`
128    let dest_is_relative = {
129        let mut c = dest.components();
130        !(c.next() == Some(Component::RootDir) || c.next() == Some(Component::RootDir))
131    };
132
133    // Check if the link points into `root_path`, reject otherwise
134    // (strip_prefix will not work).
135    debug_assert!(docdir.starts_with(root_path));
136
137    // Calculate the output.
138    let mut link = match (rewrite_rel_paths, rewrite_abs_paths, dest_is_relative) {
139        // *** Relative links.
140        // Result: "/" + docdir.strip(root_path) + dest
141        (true, false, true) => {
142            let link = PathBuf::from(Component::RootDir.as_os_str());
143            link.join(docdir.strip_prefix(root_path).ok()?)
144        }
145        // Result: docdir + dest
146        (true, true, true) => docdir.to_path_buf(),
147        // Result: dest
148        (false, _, true) => PathBuf::new(),
149        // *** Absolute links.
150        // Result: "/" + dest
151        (_, false, false) => PathBuf::from(Component::RootDir.as_os_str()),
152        // Result: "/" + root_path
153        (_, true, false) => root_path.to_path_buf(),
154    };
155    append(&mut link, dest);
156
157    if link.as_os_str().is_empty() {
158        None
159    } else {
160        Some(link)
161    }
162}
163
164trait Hyperlink {
165    /// A helper function, that first HTML escape decodes all strings of the
166    /// link. Then it percent decodes the link destination (and the
167    /// link text in case of an autolink).
168    fn decode_ampersand_and_percent(&mut self);
169
170    /// True if the value is a local link.
171    #[allow(clippy::ptr_arg)]
172    fn is_local_fn(value: &Cow<str>) -> bool;
173
174    /// * `Link::Text2Dest`: strips a possible scheme in local `dest`.
175    /// * `Link::Image2Dest`: strip local scheme in `dest`.
176    /// * `Link::Image`: strip local scheme in `src`.
177    ///
178    ///  No action if not local.
179    fn strip_local_scheme(&mut self);
180
181    /// Helper function that strips a possible scheme in `input`.
182    fn strip_scheme_fn(input: &mut Cow<str>);
183
184    /// True if the link is:
185    /// * `Link::Text2Dest` and the link text equals the link destination, or
186    /// * `Link::Image` and the links `alt` equals the link source.
187    ///
188    /// WARNING: place this test after `decode_html_escape_and_percent()`
189    /// and before: `rebase_local_link`, `expand_shorthand_link`,
190    /// `rewrite_autolink` and `apply_format_attribute`.
191    fn is_autolink(&self) -> bool;
192
193    /// A method that converts the relative URLs (local links) in `self`.
194    /// If successful, it returns `Ok(Some(URL))`, otherwise
195    /// `Err(NoteError::InvalidLocalLink)`.
196    /// If `self` contains an absolute URL, no conversion is performed and the
197    /// return value is `Ok(())`.
198    ///
199    /// Conversion details:
200    /// The base path for this conversion (usually where the HTML file resides),
201    /// is `docdir`. If not `rewrite_rel_links`, relative local links are not
202    /// converted. Furthermore, all local links starting with `/` are prepended
203    /// with `root_path`. All absolute URLs always remain untouched.
204    ///
205    /// Algorithm:
206    /// 1. If `rewrite_abs_links==true` and `link` starts with `/`, concatenate
207    ///    and return `root_path` and `dest`.
208    /// 2. If `rewrite_abs_links==false` and `dest` does not start wit `/`,
209    ///    return `dest`.
210    /// 3. If `rewrite_ext==true` and the link points to a known Tp-Note file
211    ///    extension, then `.html` is appended to the converted link.
212    ///
213    /// Remark: The _anchor's text property_ is never changed. However, there
214    /// is one exception: when the text contains a URL starting with `http:` or
215    /// `https:`, only the file stem is kept. Example, the anchor text property:
216    /// `<a ...>http:dir/my file.md</a>` is rewritten into `<a ...>my file</a>`.
217    ///
218    /// Contracts:
219    /// 1. `link` may have a scheme.
220    /// 2. `link` is `Link::Text2Dest` or `Link::Image`
221    /// 3. `root_path` and `docdir` are absolute paths to directories.
222    /// 4. `root_path` is never empty `""`. It can be `"/"`.
223    fn rebase_local_link(
224        &mut self,
225        root_path: &Path,
226        docdir: &Path,
227        rewrite_rel_paths: bool,
228        rewrite_abs_paths: bool,
229    ) -> Result<(), NoteError>;
230
231    /// If `dest` in `Link::Text2Dest` contains only a sort
232    /// tag as filename, expand the latter to a full filename.
233    /// Otherwise, no action.
234    /// This method accesses the filesystem. Therefore sometimes `prepend_path`
235    /// is needed as parameter and prepended.
236    fn expand_shorthand_link(&mut self, prepend_path: Option<&Path>) -> Result<(), NoteError>;
237
238    /// This removes a possible scheme in `text`.
239    /// Call this method only when you sure that this
240    /// is an autolink by testing with `is_autolink()`.
241    fn rewrite_autolink(&mut self);
242
243    /// A formatting attribute is a format string starting with `?` followed
244    /// by one or two patterns. It is appended to `dest` or `src`.
245    /// Processing details:
246    /// 1. Extract some a possible formatting attribute string in `dest`
247    ///    (`Link::Text2Dest`) or `src` (`Link::Image`) after `?`.
248    /// 2. Extract the _path_ before `?` in `dest` or `src`.
249    /// 3. Apply the formatting to _path_.
250    /// 4. Store the result by overwriting `text` or `alt`.
251    fn apply_format_attribute(&mut self);
252
253    /// If the link destination `dest` is a local path, return it.
254    /// Otherwise return `None`.
255    /// Acts on `Link:Text2Dest` and `Link::Imgage2Dest` only.
256    fn get_local_link_dest_path(&self) -> Option<&Path>;
257
258    /// If `dest` or `src` is a local path, return it.
259    /// Otherwise return `None`.
260    /// Acts an `Link:Image` and `Link::Image2Dest` only.
261    fn get_local_link_src_path(&self) -> Option<&Path>;
262
263    /// If the extension of a local path in `dest` is some Tp-Note
264    /// extension, append `.html` to the path. Otherwise silently return.
265    /// Acts on `Link:Text2Dest` only.
266    fn append_html_ext(&mut self);
267
268    /// Renders `Link::Text2Dest`, `Link::Image2Dest` and `Link::Image`
269    /// to HTML. Some characters in `dest` or `src` might be HTML
270    /// escape encoded. This does not percent encode at all, because
271    /// we know, that the result will be inserted later in a UTF-8 template.
272    fn to_html(&self) -> String;
273}
274
275impl Hyperlink for Link<'_> {
276    #[inline]
277    fn decode_ampersand_and_percent(&mut self) {
278        // HTML escape decode value.
279        fn dec_amp(val: &mut Cow<str>) {
280            let decoded_text = html_escape::decode_html_entities(val);
281            if matches!(&decoded_text, Cow::Owned(..)) {
282                // Does nothing, but satisfying the borrow checker. Does not `clone()`.
283                let decoded_text = Cow::Owned(decoded_text.into_owned());
284                // Store result.
285                let _ = std::mem::replace(val, decoded_text);
286            }
287        }
288
289        // HTML escape decode and percent decode value.
290        fn dec_amp_percent(val: &mut Cow<str>) {
291            dec_amp(val);
292            let decoded_dest = percent_decode_str(val.as_ref()).decode_utf8().unwrap();
293            if matches!(&decoded_dest, Cow::Owned(..)) {
294                // Does nothing, but satisfying the borrow checker. Does not `clone()`.
295                let decoded_dest = Cow::Owned(decoded_dest.into_owned());
296                // Store result.
297                let _ = std::mem::replace(val, decoded_dest);
298            }
299        }
300
301        match self {
302            Link::Text2Dest(text1, dest, title) => {
303                dec_amp(text1);
304                dec_amp_percent(dest);
305                dec_amp(title);
306            }
307            Link::Image(alt, src) => {
308                dec_amp(alt);
309                dec_amp_percent(src);
310            }
311            Link::Image2Dest(text1, alt, src, text2, dest, title) => {
312                dec_amp(text1);
313                dec_amp(alt);
314                dec_amp_percent(src);
315                dec_amp(text2);
316                dec_amp_percent(dest);
317                dec_amp(title);
318            }
319            _ => unimplemented!(),
320        };
321    }
322
323    //
324    fn is_local_fn(dest: &Cow<str>) -> bool {
325        !((dest.contains("://") && !dest.contains(":///"))
326            || dest.starts_with("mailto:")
327            || dest.starts_with("tel:"))
328    }
329
330    //
331    fn strip_local_scheme(&mut self) {
332        fn strip(dest: &mut Cow<str>) {
333            if <Link<'_> as Hyperlink>::is_local_fn(dest) {
334                <Link<'_> as Hyperlink>::strip_scheme_fn(dest);
335            }
336        }
337
338        match self {
339            Link::Text2Dest(_, dest, _title) => strip(dest),
340            Link::Image2Dest(_, _, src, _, dest, _) => {
341                strip(src);
342                strip(dest);
343            }
344            Link::Image(_, src) => strip(src),
345            _ => {}
346        };
347    }
348
349    //
350    fn strip_scheme_fn(inout: &mut Cow<str>) {
351        let output = inout
352            .trim_start_matches("https://")
353            .trim_start_matches("https:")
354            .trim_start_matches("http://")
355            .trim_start_matches("http:")
356            .trim_start_matches("tpnote:")
357            .trim_start_matches("mailto:")
358            .trim_start_matches("tel:");
359        if output != inout.as_ref() {
360            let _ = std::mem::replace(inout, Cow::Owned(output.to_string()));
361        }
362    }
363
364    //
365    fn is_autolink(&self) -> bool {
366        let (text, dest) = match self {
367            Link::Text2Dest(text, dest, _title) => (text, dest),
368            Link::Image(alt, source) => (alt, source),
369            // `Link::Image2Dest` is never an autolink.
370            _ => return false,
371        };
372        text == dest
373    }
374
375    //
376    fn rebase_local_link(
377        &mut self,
378        root_path: &Path,
379        docdir: &Path,
380        rewrite_rel_paths: bool,
381        rewrite_abs_paths: bool,
382    ) -> Result<(), NoteError> {
383        let do_rebase = |path: &mut Cow<str>| -> Result<(), NoteError> {
384            if <Link as Hyperlink>::is_local_fn(path) {
385                let (path_part, fragment) = split_path_and_fragment(path.as_ref());
386                if path_part.is_empty() {
387                    // A bare fragment (`#ch1`) denotes the current document.
388                    // There is no path to rebase.
389                    return Ok(());
390                }
391
392                let dest_out = assemble_link(
393                    root_path,
394                    docdir,
395                    Path::new(path_part),
396                    rewrite_rel_paths,
397                    rewrite_abs_paths,
398                )
399                .ok_or(NoteError::InvalidLocalPath {
400                    path: path.as_ref().to_string(),
401                })?;
402
403                // Store result.
404                let mut new_dest = dest_out.to_str().unwrap_or_default().to_string();
405                new_dest.push_str(fragment);
406                let _ = std::mem::replace(path, Cow::Owned(new_dest));
407            }
408            Ok(())
409        };
410
411        match self {
412            Link::Text2Dest(_, dest, _) => do_rebase(dest),
413            Link::Image2Dest(_, _, src, _, dest, _) => do_rebase(src).and_then(|_| do_rebase(dest)),
414            Link::Image(_, src) => do_rebase(src),
415            _ => unimplemented!(),
416        }
417    }
418
419    //
420    fn expand_shorthand_link(&mut self, prepend_path: Option<&Path>) -> Result<(), NoteError> {
421        let shorthand_link = match self {
422            Link::Text2Dest(_, dest, _) => dest,
423            Link::Image2Dest(_, _, _, _, dest, _) => dest,
424            _ => return Ok(()),
425        };
426
427        if !<Link as Hyperlink>::is_local_fn(shorthand_link) {
428            return Ok(());
429        }
430
431        let (shorthand_str, shorthand_format) = match shorthand_link.split_once(FORMAT_SEPARATOR) {
432            Some((path, fmt)) => (path, Some(fmt)),
433            None => (shorthand_link.as_ref(), None),
434        };
435
436        let shorthand_path = Path::new(shorthand_str);
437
438        if let Some(sort_tag) = shorthand_str.is_valid_sort_tag() {
439            let full_shorthand_path = if let Some(root_path) = prepend_path {
440                // Concatenate `root_path` and `shorthand_path`.
441                let shorthand_path = shorthand_path
442                    .strip_prefix(MAIN_SEPARATOR_STR)
443                    .unwrap_or(shorthand_path);
444                Cow::Owned(root_path.join(shorthand_path))
445            } else {
446                Cow::Borrowed(shorthand_path)
447            };
448
449            // Search for the file.
450            let found = full_shorthand_path
451                .parent()
452                .and_then(|dir| dir.find_file_with_sort_tag(sort_tag));
453
454            if let Some(path) = found {
455                // We prepended `root_path` before, we can safely strip it
456                // and unwrap.
457                let found_link = path
458                    .strip_prefix(prepend_path.unwrap_or(Path::new("")))
459                    .unwrap();
460                // Prepend `/`.
461                let mut found_link = Path::new(MAIN_SEPARATOR_STR)
462                    .join(found_link)
463                    .to_str()
464                    .unwrap_or_default()
465                    .to_string();
466
467                if let Some(fmt) = shorthand_format {
468                    found_link.push(FORMAT_SEPARATOR);
469                    found_link.push_str(fmt);
470                }
471
472                // Store result.
473                let _ = std::mem::replace(shorthand_link, Cow::Owned(found_link));
474            } else {
475                return Err(NoteError::CanNotExpandShorthandLink {
476                    path: full_shorthand_path.to_string_lossy().into_owned(),
477                });
478            }
479        }
480        Ok(())
481    }
482
483    //
484    fn rewrite_autolink(&mut self) {
485        let text = match self {
486            Link::Text2Dest(text, _, _) => text,
487            Link::Image(alt, _) => alt,
488            _ => return,
489        };
490
491        <Link as Hyperlink>::strip_scheme_fn(text);
492    }
493
494    //
495    fn apply_format_attribute(&mut self) {
496        // Is this an absolute URL?
497
498        let (text, dest) = match self {
499            Link::Text2Dest(text, dest, _) => (text, dest),
500            Link::Image(alt, source) => (alt, source),
501            _ => return,
502        };
503
504        if !<Link as Hyperlink>::is_local_fn(dest) {
505            return;
506        }
507
508        // We assume, that `dest` had been expanded already, so we can extract
509        // the full filename here.
510        // If ever it ends with a format string we apply it. Otherwise we quit
511        // the method and do nothing.
512        let (path, format) = match dest.split_once(FORMAT_SEPARATOR) {
513            Some(s) => s,
514            None => return,
515        };
516
517        let mut short_text = Path::new(path)
518            .file_name()
519            .unwrap_or_default()
520            .to_str()
521            .unwrap_or_default();
522
523        // Select what to match:
524        let format = if format.starts_with(FORMAT_COMPLETE_FILENAME) {
525            // Keep complete filename.
526            format
527                .strip_prefix(FORMAT_COMPLETE_FILENAME)
528                .unwrap_or(format)
529        } else if format.starts_with(FORMAT_ONLY_SORT_TAG) {
530            // Keep only format-tag.
531            short_text = Path::new(path).disassemble().0;
532            format.strip_prefix(FORMAT_ONLY_SORT_TAG).unwrap_or(format)
533        } else {
534            // Keep only stem.
535            short_text = Path::new(path).disassemble().2;
536            format
537        };
538
539        match format.split_once(FORMAT_FROM_TO_SEPARATOR) {
540            // No `:`
541            None => {
542                if !format.is_empty()
543                    && let Some(idx) = short_text.find(format) {
544                        short_text = &short_text[..idx];
545                    };
546            }
547            // Some `:`
548            Some((from, to)) => {
549                if !from.is_empty()
550                    && let Some(idx) = short_text.find(from) {
551                        short_text = &short_text[(idx + from.len())..];
552                    };
553                if !to.is_empty()
554                    && let Some(idx) = short_text.find(to) {
555                        short_text = &short_text[..idx];
556                    };
557            }
558        }
559        // Store the result.
560        let _ = std::mem::replace(text, Cow::Owned(short_text.to_string()));
561        let _ = std::mem::replace(dest, Cow::Owned(path.to_string()));
562    }
563
564    //
565    fn get_local_link_dest_path(&self) -> Option<&Path> {
566        let dest = match self {
567            Link::Text2Dest(_, dest, _) => dest,
568            Link::Image2Dest(_, _, _, _, dest, _) => dest,
569            _ => return None,
570        };
571        if <Link as Hyperlink>::is_local_fn(dest) {
572            let path = split_path_and_fragment(dest.as_ref()).0;
573            (!path.is_empty()).then(|| Path::new(path))
574        } else {
575            None
576        }
577    }
578
579    //
580    fn get_local_link_src_path(&self) -> Option<&Path> {
581        let src = match self {
582            Link::Image2Dest(_, _, src, _, _, _) => src,
583            Link::Image(_, src) => src,
584            _ => return None,
585        };
586        if <Link as Hyperlink>::is_local_fn(src) {
587            Some(Path::new(src.as_ref()))
588        } else {
589            None
590        }
591    }
592
593    //
594    fn append_html_ext(&mut self) {
595        let dest = match self {
596            Link::Text2Dest(_, dest, _) => dest,
597            Link::Image2Dest(_, _, _, _, dest, _) => dest,
598            _ => return,
599        };
600        if <Link as Hyperlink>::is_local_fn(dest) {
601            let (path, fragment) = split_path_and_fragment(dest.as_ref());
602            if path.has_tpnote_ext() {
603                let mut newpath = path.to_string();
604                newpath.push_str(HTML_EXT);
605                newpath.push_str(fragment);
606
607                let _ = std::mem::replace(dest, Cow::Owned(newpath));
608            }
609        }
610    }
611
612    //
613    fn to_html(&self) -> String {
614        // HTML escape encode double quoted attributes
615        fn enc_amp(val: Cow<str>) -> Cow<str> {
616            let s = html_escape::encode_double_quoted_attribute(val.as_ref());
617            if s == val {
618                val
619            } else {
620                // No cloning happens here, because we own `s` already.
621                Cow::Owned(s.into_owned())
622            }
623        }
624        // Replace Windows backslash, percent-encode the path (keeping a
625        // written fragment untouched), then HTML escape encode.
626        fn repl_backspace_enc_amp(val: Cow<str>) -> Cow<str> {
627            // Under Windows `\` is a path separator, not data: normalize it
628            // to `/` before `split_path_and_fragment`/`percent_encode_path`
629            // treat it as one.
630            let val = if val.as_ref().contains('\\') {
631                Cow::Owned(val.to_string().replace('\\', "/"))
632            } else {
633                val
634            };
635            let (path, fragment) = split_path_and_fragment(val.as_ref());
636            let encoded = format!("{}{}", percent_encode_path(path), fragment);
637            let s = html_escape::encode_double_quoted_attribute(&encoded);
638            Cow::Owned(s.into_owned())
639        }
640
641        match self {
642            Link::Text2Dest(text, dest, title) => {
643                // Format title.
644                let title_html = if !title.is_empty() {
645                    format!(" title=\"{}\"", enc_amp(title.shallow_clone()))
646                } else {
647                    "".to_string()
648                };
649
650                format!(
651                    "<a href=\"{}\"{}>{}</a>",
652                    repl_backspace_enc_amp(dest.shallow_clone()),
653                    title_html,
654                    text
655                )
656            }
657            Link::Image2Dest(text1, alt, src, text2, dest, title) => {
658                // Format title.
659                let title_html = if !title.is_empty() {
660                    format!(" title=\"{}\"", enc_amp(title.shallow_clone()))
661                } else {
662                    "".to_string()
663                };
664
665                format!(
666                    "<a href=\"{}\"{}>{}<img src=\"{}\" alt=\"{}\">{}</a>",
667                    repl_backspace_enc_amp(dest.shallow_clone()),
668                    title_html,
669                    text1,
670                    repl_backspace_enc_amp(src.shallow_clone()),
671                    enc_amp(alt.shallow_clone()),
672                    text2
673                )
674            }
675            Link::Image(alt, src) => {
676                format!(
677                    "<img src=\"{}\" alt=\"{}\">",
678                    repl_backspace_enc_amp(src.shallow_clone()),
679                    enc_amp(alt.shallow_clone())
680                )
681            }
682            _ => unimplemented!(),
683        }
684    }
685}
686
687#[inline]
688/// A helper function that scans the input HTML document in `html_input` for
689/// HTML hyperlinks. When it finds a relative URL (local link), it analyzes it's
690/// path. Depending on the `local_link_kind` configuration, relative local
691/// links are converted into absolute local links and eventually rebased.
692///
693/// In order to achieve this, the user must respect the following convention
694/// concerning absolute local links in Tp-Note documents:
695/// 1. When a document contains a local link with an absolute path (absolute
696///    local link), the base of this path is considered to be the directory
697///    where the project configuration file ‘tpnote.toml’ resides (or ‘/’ in
698///    non exists). The project configuration file directory is `root_path`.
699/// 2. Furthermore, the parameter `docdir` contains the absolute path of the
700///    directory of the currently processed HTML document. The user guarantees
701///    that `docdir` is the base for all relative local links in the document.
702///    Note: `docdir` must always start with `root_path`.
703///
704/// If `LocalLinkKind::Off`, relative local links are not converted.
705/// If `LocalLinkKind::Short`, relative local links are converted into an
706/// absolute local links with `root_path` as base directory.
707/// If `LocalLinkKind::Long`, in addition to the above, the resulting absolute
708/// local link is prepended with `root_path`.
709///
710/// If `rewrite_ext` is true and a local link points to a known
711/// Tp-Note file extension, then `.html` is appended to the converted link.
712///
713/// Remark: The link's text property is never changed. However, there is
714/// one exception: when the link's text contains a string similar to URLs,
715/// starting with `http:` or `tpnote:`. In this case, the string is interpreted
716/// as URL and only the stem of the filename is displayed, e.g.
717/// `<a ...>http:dir/my file.md</a>` is replaced with `<a ...>my file</a>`.
718///
719/// Finally, before a converted local link is reinserted in the output HTML, a
720/// copy of that link is kept in `allowed_local_links` for further bookkeeping.
721///
722/// NB: All absolute URLs (starting with a domain) always remain untouched.
723///
724/// NB2: It is guaranteed, that the resulting HTML document contains only local
725/// links to other documents within `root_path`. Deviant links displayed as
726/// `INVALID LOCAL LINK` and URL is discarded.
727pub fn rewrite_links(
728    html_input: String,
729    root_path: &Path,
730    docdir: &Path,
731    local_link_kind: LocalLinkKind,
732    rewrite_ext: bool,
733    allowed_local_links: Arc<RwLock<HashSet<PathBuf>>>,
734) -> String {
735    let (rewrite_rel_paths, rewrite_abs_paths) = match local_link_kind {
736        LocalLinkKind::Off => (false, false),
737        LocalLinkKind::Short => (true, false),
738        LocalLinkKind::Long => (true, true),
739    };
740
741    // Search for hyperlinks and inline images in the HTML rendition
742    // of this note.
743    let mut rest = &*html_input;
744    let mut html_out = String::new();
745    for ((skipped, _consumed, remaining), mut link) in HtmlLinkInlineImage::new(&html_input) {
746        html_out.push_str(skipped);
747        rest = remaining;
748
749        // Check if `text` = `dest`.
750        let mut link_is_autolink = link.is_autolink();
751
752        // Percent decode link destination.
753        link.decode_ampersand_and_percent();
754
755        // Check again if `text` = `dest`.
756        link_is_autolink = link_is_autolink || link.is_autolink();
757
758        link.strip_local_scheme();
759
760        // Rewrite the local link.
761        match link
762            .rebase_local_link(root_path, docdir, rewrite_rel_paths, rewrite_abs_paths)
763            .and_then(|_| {
764                link.expand_shorthand_link(
765                    (matches!(local_link_kind, LocalLinkKind::Short)).then_some(root_path),
766                )
767            }) {
768            Ok(()) => {}
769            Err(e) => {
770                let e = e.to_string();
771                let e = html_escape::encode_text(&e);
772                html_out.push_str(&format!("<i>{}</i>", e));
773                continue;
774            }
775        };
776
777        if link_is_autolink {
778            link.rewrite_autolink();
779        }
780
781        link.apply_format_attribute();
782
783        if let Some(dest_path) = link.get_local_link_dest_path() {
784            allowed_local_links.write().insert(dest_path.to_path_buf());
785        };
786        if let Some(src_path) = link.get_local_link_src_path() {
787            allowed_local_links.write().insert(src_path.to_path_buf());
788        };
789
790        if rewrite_ext {
791            link.append_html_ext();
792        }
793        html_out.push_str(&link.to_html());
794    }
795    // Add the last `remaining`.
796    html_out.push_str(rest);
797
798    log::trace!(
799        "Viewer: referenced allowed local files: {}",
800        allowed_local_links
801            .read_recursive()
802            .iter()
803            .map(|p| {
804                let mut s = "\n    '".to_string();
805                s.push_str(&p.display().to_string());
806                s
807            })
808            .collect::<String>()
809    );
810
811    html_out
812    // The `RwLockWriteGuard` is released here.
813}
814
815/// One `<h1>`-`<h6>` heading found while scanning rendered HTML, in
816/// document order.
817struct HeadingMatch {
818    /// Byte offset of the opening tag's terminating `>`.
819    tag_close: usize,
820    /// The opening tag's `id="..."` attribute value, if it already has one.
821    existing_id: Option<String>,
822    /// Tag-stripped, entity-decoded text content of the heading.
823    text: String,
824}
825
826/// Scans `html` for every heading, in document order. Mirrors the
827/// tag-finding approach of `filter::FirstHtmlHeading` (which stops at the
828/// first heading; this collects all of them) and additionally extracts a
829/// pre-existing `id="..."` attribute value, if present. Headings never
830/// nest — CommonMark's grammar and RST's section model both forbid it — so
831/// a simple "next matching closing tag" scan is safe.
832fn scan_headings(html: &str) -> Vec<HeadingMatch> {
833    const OPENING: &[&str; 6] = &["<h1", "<h2", "<h3", "<h4", "<h5", "<h6"];
834    const CLOSING: &[&str; 6] = &["</h1>", "</h2>", "</h3>", "</h4>", "</h5>", "</h6>"];
835
836    let mut headings = Vec::new();
837    let mut i = 0;
838    while let Some(mut tag_start) = html[i..].find('<') {
839        let Some(mut tag_end) = html[i + tag_start..].find('>') else {
840            break;
841        };
842        tag_end += 1;
843        // Move on if there is another opening bracket.
844        if let Some(new_start) = html[i + tag_start + 1..i + tag_start + tag_end].rfind('<') {
845            tag_start += new_start + 1;
846            tag_end -= new_start + 1;
847        }
848
849        let tag_str = &html[i + tag_start..i + tag_start + tag_end];
850        if !OPENING.iter().any(|&pat| tag_str.starts_with(pat)) {
851            i += tag_start + tag_end;
852            continue;
853        }
854
855        // Index right after the opening tag's `>`, and of the `>` itself.
856        let heading_start = i + tag_start + tag_end;
857        let tag_close = heading_start - 1;
858
859        let existing_id = tag_str.find("id=\"").map(|p| {
860            let rest = &tag_str[p + 4..];
861            let end = rest.find('"').unwrap_or(rest.len());
862            rest[..end].to_string()
863        });
864
865        // Find the matching closing tag.
866        let mut k = heading_start;
867        let mut heading_end = None;
868        while let Some(mut cs) = html[k..].find('<') {
869            let Some(mut ce) = html[k + cs..].find('>') else {
870                break;
871            };
872            ce += 1;
873            if let Some(new_start) = html[k + cs + 1..k + cs + ce].rfind('<') {
874                cs += new_start + 1;
875                ce -= new_start + 1;
876            }
877            if CLOSING.iter().any(|&pat| html[k + cs..k + cs + ce].starts_with(pat)) {
878                heading_end = Some(k + cs);
879                break;
880            }
881            k += cs + ce;
882        }
883
884        let Some(heading_end) = heading_end else {
885            i = heading_start;
886            continue;
887        };
888
889        // Remove HTML tags inside the heading, then decode entities.
890        let mut cleaned = String::new();
891        let mut inside_tag = false;
892        for c in html[heading_start..heading_end].chars() {
893            if c == '<' {
894                inside_tag = true;
895            } else if c == '>' {
896                inside_tag = false;
897            } else if !inside_tag {
898                cleaned.push(c);
899            }
900        }
901        let text = html_escape::decode_html_entities(&cleaned).into_owned();
902
903        headings.push(HeadingMatch { tag_close, existing_id, text });
904
905        i = heading_end;
906    }
907    headings
908}
909
910/// GitHub/GitLab-style slug (see `HeadingIdPolicy::Gfm`): lowercase, keep
911/// only Unicode letters/digits/`-`/`_`/space, convert each remaining space
912/// to a hyphen individually (two adjacent spaces become two adjacent
913/// hyphens, not one collapsed hyphen — this is what turns an em dash
914/// surrounded by spaces into a double hyphen once the dash itself is
915/// dropped), then trim stray leading/trailing hyphens.
916fn slugify_gfm(text: &str) -> String {
917    let lower = text.to_lowercase();
918    let filtered: String = lower
919        .chars()
920        .filter(|c| c.is_alphanumeric() || *c == '-' || *c == '_' || *c == ' ')
921        .map(|c| if c == ' ' { '-' } else { c })
922        .collect();
923    filtered.trim_matches('-').to_string()
924}
925
926/// Pandoc's `auto_identifiers` algorithm (see `HeadingIdPolicy::Pandoc`):
927/// like `slugify_gfm`, but periods are also kept, and any leading run of
928/// non-letter characters is stripped (`2. Section` becomes `section`, not
929/// `2-section`).
930fn slugify_pandoc(text: &str) -> String {
931    let lower = text.to_lowercase();
932    let filtered: String = lower
933        .chars()
934        .filter(|c| c.is_alphanumeric() || *c == '-' || *c == '_' || *c == '.' || *c == ' ')
935        .map(|c| if c == ' ' { '-' } else { c })
936        .collect();
937    filtered
938        .trim_start_matches(|c: char| !c.is_alphabetic())
939        .trim_matches('-')
940        .to_string()
941}
942
943/// Appends `-1`, `-2`, ... to `base` until the result isn't already in
944/// `seen`, records the result in `seen`, and returns it.
945fn disambiguate(base: String, seen: &mut HashSet<String>) -> String {
946    if seen.insert(base.clone()) {
947        return base;
948    }
949    let mut n = 1;
950    loop {
951        let candidate = format!("{base}-{n}");
952        if seen.insert(candidate.clone()) {
953            return candidate;
954        }
955        n += 1;
956    }
957}
958
959/// Assigns an `id` attribute to every heading in `html` that doesn't
960/// already have one, according to `policy`. A heading's existing `id` —
961/// whether an explicit Markdown `{#id}` heading attribute or one another
962/// renderer already assigned (e.g. RST's own auto-ids) — always wins and is
963/// never touched. Markup-language agnostic: this runs on the final
964/// rendered HTML, after `markup_to_html` has already produced `<h1>`-`<h6>`
965/// tags, regardless of which renderer produced them.
966pub fn assign_heading_ids(html: String, policy: HeadingIdPolicy) -> String {
967    if policy == HeadingIdPolicy::Off {
968        return html;
969    }
970
971    let headings = scan_headings(&html);
972    if headings.is_empty() {
973        return html;
974    }
975
976    let mut seen: HashSet<String> = HashSet::new();
977    for h in &headings {
978        if let Some(id) = &h.existing_id {
979            seen.insert(id.clone());
980        }
981    }
982
983    // `(tag_close offset, new id)` for every heading that needs one, in
984    // document order.
985    let mut assignments: Vec<(usize, String)> = Vec::new();
986    for h in &headings {
987        if h.existing_id.is_some() {
988            continue;
989        }
990        let slug = match policy {
991            HeadingIdPolicy::Gfm => slugify_gfm(&h.text),
992            HeadingIdPolicy::Pandoc => slugify_pandoc(&h.text),
993            HeadingIdPolicy::Off => unreachable!(),
994        };
995        let slug = if slug.is_empty() {
996            match policy {
997                // Pandoc documents this fallback explicitly.
998                HeadingIdPolicy::Pandoc => "section".to_string(),
999                // No spec for this case in GFM/GitLab: leave the heading
1000                // id-less rather than fabricate one.
1001                _ => continue,
1002            }
1003        } else {
1004            slug
1005        };
1006        assignments.push((h.tag_close, disambiguate(slug, &mut seen)));
1007    }
1008
1009    if assignments.is_empty() {
1010        return html;
1011    }
1012
1013    // Splice `id="..."` into each opening tag right before its `>`. Offsets
1014    // are into the original `html`, in ascending order, so a running
1015    // cursor over spans of the original string is enough to rebuild it.
1016    let mut out = String::with_capacity(html.len() + assignments.len() * 16);
1017    let mut cursor = 0;
1018    for (tag_close, id) in assignments {
1019        out.push_str(&html[cursor..tag_close]);
1020        out.push_str(&format!(" id=\"{id}\""));
1021        cursor = tag_close;
1022    }
1023    out.push_str(&html[cursor..]);
1024    out
1025}
1026
1027/// This trait deals with tagged HTML `&str` data.
1028pub trait HtmlStr {
1029    /// Lowercase pattern to check if this is a Doctype tag.
1030    const TAG_DOCTYPE_PAT: &'static str = "<!doctype";
1031    /// Lowercase pattern to check if this Doctype is HTML.
1032    const TAG_DOCTYPE_HTML_PAT: &'static str = "<!doctype html";
1033    /// Doctype HTML tag. This is inserted by
1034    /// `<HtmlString>.prepend_html_start_tag()`
1035    const TAG_DOCTYPE_HTML: &'static str = "<!DOCTYPE html>";
1036    /// Pattern to check if f this is an HTML start tag.
1037    const START_TAG_HTML_PAT: &'static str = "<html";
1038    /// HTML end tag.
1039    const END_TAG_HTML: &'static str = "</html>";
1040
1041    /// We consider `self` empty, when it equals to `<!DOCTYPE html...>` or
1042    /// when it is empty.
1043    fn is_empty_html(&self) -> bool;
1044
1045    /// We consider `html` empty, when it equals to `<!DOCTYPE html...>` or
1046    /// when it is empty.
1047    /// This is identical to `is_empty_html()`, but does not pull in
1048    /// additional trait bounds.
1049    fn is_empty_html2(html: &str) -> bool {
1050        html.is_empty_html()
1051    }
1052
1053    /// True if stream starts with `<!DOCTYPE html...>`.
1054    fn has_html_start_tag(&self) -> bool;
1055
1056    /// True if `html` starts with `<!DOCTYPE html...>`.
1057    /// This is identical to `has_html_start_tag()`, but does not pull in
1058    /// additional trait bounds.
1059    fn has_html_start_tag2(html: &str) -> bool {
1060        html.has_html_start_tag()
1061    }
1062
1063    /// Some heuristics to guess if the input stream contains HTML.
1064    /// Current implementation:
1065    /// True if:
1066    ///
1067    /// * The stream starts with `<!DOCTYPE html ...>`, or
1068    /// * the stream starts with `<html ...>`    
1069    ///
1070    /// This function does not check if the recognized HTML is valid.
1071    fn is_html_unchecked(&self) -> bool;
1072}
1073
1074impl HtmlStr for str {
1075    fn is_empty_html(&self) -> bool {
1076        if self.is_empty() {
1077            return true;
1078        }
1079
1080        let html = self
1081            .trim_start()
1082            .lines()
1083            .next()
1084            .map(|l| l.to_ascii_lowercase())
1085            .unwrap_or_default();
1086
1087        html.as_str().starts_with(Self::TAG_DOCTYPE_HTML_PAT)
1088            // The next closing bracket must be in last position.
1089            && html.find('>').unwrap_or_default() == html.len()-1
1090    }
1091
1092    fn has_html_start_tag(&self) -> bool {
1093        let html = self
1094            .trim_start()
1095            .lines()
1096            .next()
1097            .map(|l| l.to_ascii_lowercase());
1098        html.as_ref()
1099            .is_some_and(|l| l.starts_with(Self::TAG_DOCTYPE_HTML_PAT))
1100    }
1101
1102    fn is_html_unchecked(&self) -> bool {
1103        let html = self
1104            .trim_start()
1105            .lines()
1106            .next()
1107            .map(|l| l.to_ascii_lowercase());
1108        html.as_ref().is_some_and(|l| {
1109            (l.starts_with(Self::TAG_DOCTYPE_HTML_PAT)
1110                && l[Self::TAG_DOCTYPE_HTML_PAT.len()..].contains('>'))
1111                || (l.starts_with(Self::START_TAG_HTML_PAT)
1112                    && l[Self::START_TAG_HTML_PAT.len()..].contains('>'))
1113        })
1114    }
1115}
1116
1117/// This trait deals with tagged HTML `String` data.
1118pub trait HtmlString: Sized {
1119    /// If the input does not start with `<!DOCTYPE html`
1120    /// (or lowercase variants), then insert `<!DOCTYPE html>`.
1121    /// Returns `InputStreamError::NonHtmlDoctype` if there is another Doctype
1122    /// already.
1123    fn prepend_html_start_tag(self) -> Result<Self, InputStreamError>;
1124}
1125
1126impl HtmlString for String {
1127    fn prepend_html_start_tag(self) -> Result<Self, InputStreamError> {
1128        // Bring `HtmlStr` methods into scope.
1129        use crate::html::HtmlStr;
1130
1131        let html2 = self
1132            .trim_start()
1133            .lines()
1134            .next()
1135            .map(|l| l.to_ascii_lowercase())
1136            .unwrap_or_default();
1137
1138        if html2.starts_with(<str as HtmlStr>::TAG_DOCTYPE_HTML_PAT) {
1139            // Has a start tag already.
1140            Ok(self)
1141        } else if !html2.starts_with(<str as HtmlStr>::TAG_DOCTYPE_PAT) {
1142            // Insert HTML Doctype tag.
1143            let mut html = self;
1144            html.insert_str(0, <str as HtmlStr>::TAG_DOCTYPE_HTML);
1145            Ok(html)
1146        } else {
1147            // There is a Doctype other than HTML.
1148            Err(InputStreamError::NonHtmlDoctype {
1149                html: self.chars().take(25).collect::<String>(),
1150            })
1151        }
1152    }
1153}
1154
1155#[cfg(test)]
1156mod tests {
1157
1158    use crate::error::InputStreamError;
1159    use crate::error::NoteError;
1160    use crate::html::Hyperlink;
1161    use crate::html::assemble_link;
1162    use crate::html::assign_heading_ids;
1163    use crate::html::rewrite_links;
1164    use parking_lot::RwLock;
1165    use parse_hyperlinks::parser::Link;
1166    use parse_hyperlinks_extras::parser::parse_html::take_link;
1167    use std::borrow::Cow;
1168    use std::{
1169        collections::HashSet,
1170        path::{Path, PathBuf},
1171        sync::Arc,
1172    };
1173
1174    #[test]
1175    fn test_assemble_link() {
1176        // `rewrite_rel_links=true`
1177        let output = assemble_link(
1178            Path::new("/my"),
1179            Path::new("/my/doc/path"),
1180            Path::new("../local/link to/note.md"),
1181            true,
1182            false,
1183        )
1184        .unwrap();
1185        assert_eq!(output, Path::new("/doc/local/link to/note.md"));
1186
1187        // `rewrite_rel_links=false`
1188        let output = assemble_link(
1189            Path::new("/my"),
1190            Path::new("/my/doc/path"),
1191            Path::new("../local/link to/note.md"),
1192            false,
1193            false,
1194        )
1195        .unwrap();
1196        assert_eq!(output, Path::new("../local/link to/note.md"));
1197
1198        // Absolute `dest`.
1199        let output = assemble_link(
1200            Path::new("/my"),
1201            Path::new("/my/doc/path"),
1202            Path::new("/test/../abs/local/link to/note.md"),
1203            false,
1204            false,
1205        )
1206        .unwrap();
1207        assert_eq!(output, Path::new("/abs/local/link to/note.md"));
1208
1209        // Underflow.
1210        let output = assemble_link(
1211            Path::new("/my"),
1212            Path::new("/my/doc/path"),
1213            Path::new("/../local/link to/note.md"),
1214            false,
1215            false,
1216        );
1217        assert_eq!(output, None);
1218
1219        // Absolute `dest`, `rewrite_abs_links=true`.
1220        let output = assemble_link(
1221            Path::new("/my"),
1222            Path::new("/my/doc/path"),
1223            Path::new("/abs/local/link to/note.md"),
1224            false,
1225            true,
1226        )
1227        .unwrap();
1228        assert_eq!(output, Path::new("/my/abs/local/link to/note.md"));
1229
1230        // Absolute `dest`, `rewrite_abs_links=false`.
1231        let output = assemble_link(
1232            Path::new("/my"),
1233            Path::new("/my/doc/path"),
1234            Path::new("/test/../abs/local/link to/note.md"),
1235            false,
1236            false,
1237        )
1238        .unwrap();
1239        assert_eq!(output, Path::new("/abs/local/link to/note.md"));
1240
1241        // Absolute `dest`, `rewrite` both.
1242        let output = assemble_link(
1243            Path::new("/my"),
1244            Path::new("/my/doc/path"),
1245            Path::new("abs/local/link to/note.md"),
1246            true,
1247            true,
1248        )
1249        .unwrap();
1250        assert_eq!(output, Path::new("/my/doc/path/abs/local/link to/note.md"));
1251    }
1252
1253    #[test]
1254    fn test_decode_html_escape_and_percent() {
1255        //
1256        let mut input = Link::Text2Dest(Cow::from("text"), Cow::from("dest"), Cow::from("title"));
1257        let expected = Link::Text2Dest(Cow::from("text"), Cow::from("dest"), Cow::from("title"));
1258        input.decode_ampersand_and_percent();
1259        let output = input;
1260        assert_eq!(output, expected);
1261
1262        //
1263        let mut input = Link::Text2Dest(
1264            Cow::from("te%20xt"),
1265            Cow::from("de%20st"),
1266            Cow::from("title"),
1267        );
1268        let expected =
1269            Link::Text2Dest(Cow::from("te%20xt"), Cow::from("de st"), Cow::from("title"));
1270        input.decode_ampersand_and_percent();
1271        let output = input;
1272        assert_eq!(output, expected);
1273
1274        //
1275        let mut input =
1276            Link::Text2Dest(Cow::from("text"), Cow::from("d:e%20st"), Cow::from("title"));
1277        let expected = Link::Text2Dest(Cow::from("text"), Cow::from("d:e st"), Cow::from("title"));
1278        input.decode_ampersand_and_percent();
1279        let output = input;
1280        assert_eq!(output, expected);
1281
1282        let mut input = Link::Text2Dest(
1283            Cow::from("a&amp;&quot;lt"),
1284            Cow::from("a&amp;&quot;lt"),
1285            Cow::from("a&amp;&quot;lt"),
1286        );
1287        let expected = Link::Text2Dest(
1288            Cow::from("a&\"lt"),
1289            Cow::from("a&\"lt"),
1290            Cow::from("a&\"lt"),
1291        );
1292        input.decode_ampersand_and_percent();
1293        let output = input;
1294        assert_eq!(output, expected);
1295
1296        //
1297        let mut input = Link::Image(Cow::from("al%20t"), Cow::from("de%20st"));
1298        let expected = Link::Image(Cow::from("al%20t"), Cow::from("de st"));
1299        input.decode_ampersand_and_percent();
1300        let output = input;
1301        assert_eq!(output, expected);
1302
1303        //
1304        let mut input = Link::Image(Cow::from("a\\lt"), Cow::from("d\\est"));
1305        let expected = Link::Image(Cow::from("a\\lt"), Cow::from("d\\est"));
1306        input.decode_ampersand_and_percent();
1307        let output = input;
1308        assert_eq!(output, expected);
1309
1310        //
1311        let mut input = Link::Image(Cow::from("a&amp;&quot;lt"), Cow::from("a&amp;&quot;lt"));
1312        let expected = Link::Image(Cow::from("a&\"lt"), Cow::from("a&\"lt"));
1313        input.decode_ampersand_and_percent();
1314        let output = input;
1315        assert_eq!(output, expected);
1316    }
1317
1318    #[test]
1319    fn test_is_local() {
1320        let input = Cow::from("/path/My doc.md");
1321        assert!(<Link as Hyperlink>::is_local_fn(&input));
1322
1323        let input = Cow::from("tpnote:path/My doc.md");
1324        assert!(<Link as Hyperlink>::is_local_fn(&input));
1325
1326        let input = Cow::from("tpnote:/path/My doc.md");
1327        assert!(<Link as Hyperlink>::is_local_fn(&input));
1328
1329        let input = Cow::from("https://getreu.net");
1330        assert!(!<Link as Hyperlink>::is_local_fn(&input));
1331    }
1332
1333    #[test]
1334    fn strip_local_scheme() {
1335        let mut input = Link::Text2Dest(
1336            Cow::from("xyz"),
1337            Cow::from("https://getreu.net"),
1338            Cow::from("xyz"),
1339        );
1340        let expected = input.clone();
1341        input.strip_local_scheme();
1342        assert_eq!(input, expected);
1343
1344        //
1345        let mut input = Link::Text2Dest(
1346            Cow::from("xyz"),
1347            Cow::from("tpnote:/dir/My doc.md"),
1348            Cow::from("xyz"),
1349        );
1350        let expected = Link::Text2Dest(
1351            Cow::from("xyz"),
1352            Cow::from("/dir/My doc.md"),
1353            Cow::from("xyz"),
1354        );
1355        input.strip_local_scheme();
1356        assert_eq!(input, expected);
1357    }
1358
1359    #[test]
1360    fn test_is_autolink() {
1361        let input = Link::Image(Cow::from("abc"), Cow::from("abc"));
1362        assert!(input.is_autolink());
1363
1364        //
1365        let input = Link::Text2Dest(Cow::from("abc"), Cow::from("abc"), Cow::from("xyz"));
1366        assert!(input.is_autolink());
1367
1368        //
1369        let input = Link::Image(Cow::from("abc"), Cow::from("abcd"));
1370        assert!(!input.is_autolink());
1371
1372        //
1373        let input = Link::Text2Dest(Cow::from("abc"), Cow::from("abcd"), Cow::from("xyz"));
1374        assert!(!input.is_autolink());
1375    }
1376
1377    #[test]
1378    fn test_rewrite_local_link() {
1379        let root_path = Path::new("/my/");
1380        let docdir = Path::new("/my/abs/note path/");
1381
1382        // Should panic: this is not a relative path.
1383        let mut input = take_link("<a href=\"ftp://getreu.net\">Blog</a>")
1384            .unwrap()
1385            .1
1386            .1;
1387        input
1388            .rebase_local_link(root_path, docdir, true, false)
1389            .unwrap();
1390        assert!(input.get_local_link_dest_path().is_none());
1391
1392        //
1393        let root_path = Path::new("/my/");
1394        let docdir = Path::new("/my/abs/note path/");
1395
1396        // Check relative path to image.
1397        let mut input = take_link("<img src=\"down/./down/../../t m p.jpg\" alt=\"Image\" />")
1398            .unwrap()
1399            .1
1400            .1;
1401        let expected = "<img src=\"/abs/note%20path/t%20m%20p.jpg\" \
1402            alt=\"Image\">";
1403        input
1404            .rebase_local_link(root_path, docdir, true, false)
1405            .unwrap();
1406        let outpath = input.get_local_link_src_path().unwrap();
1407        let output = input.to_html();
1408        assert_eq!(output, expected);
1409        assert_eq!(outpath, PathBuf::from("/abs/note path/t m p.jpg"));
1410
1411        // Check relative path to image. Canonicalized?
1412        let mut input = take_link("<img src=\"down/./../../t m p.jpg\" alt=\"Image\" />")
1413            .unwrap()
1414            .1
1415            .1;
1416        let expected = "<img src=\"/abs/t%20m%20p.jpg\" alt=\"Image\">";
1417        input
1418            .rebase_local_link(root_path, docdir, true, false)
1419            .unwrap();
1420        let outpath = input.get_local_link_src_path().unwrap();
1421        let output = input.to_html();
1422        assert_eq!(output, expected);
1423        assert_eq!(outpath, PathBuf::from("/abs/t m p.jpg"));
1424
1425        // Check relative path to note file.
1426        let mut input = take_link("<a href=\"./down/./../my note 1.md\">my note 1</a>")
1427            .unwrap()
1428            .1
1429            .1;
1430        let expected = "<a href=\"/abs/note%20path/my%20note%201.md\">my note 1</a>";
1431        input
1432            .rebase_local_link(root_path, docdir, true, false)
1433            .unwrap();
1434        let outpath = input.get_local_link_dest_path().unwrap();
1435        let output = input.to_html();
1436        assert_eq!(output, expected);
1437        assert_eq!(outpath, PathBuf::from("/abs/note path/my note 1.md"));
1438
1439        // Check absolute path to note file.
1440        let mut input = take_link("<a href=\"/dir/./down/../my note 1.md\">my note 1</a>")
1441            .unwrap()
1442            .1
1443            .1;
1444        let expected = "<a href=\"/dir/my%20note%201.md\">my note 1</a>";
1445        input
1446            .rebase_local_link(root_path, docdir, true, false)
1447            .unwrap();
1448        let outpath = input.get_local_link_dest_path().unwrap();
1449        let output = input.to_html();
1450        assert_eq!(output, expected);
1451        assert_eq!(outpath, PathBuf::from("/dir/my note 1.md"));
1452
1453        // Check relative path to note file. Canonicalized?
1454        let mut input = take_link("<a href=\"./down/./../dir/my note 1.md\">my note 1</a>")
1455            .unwrap()
1456            .1
1457            .1;
1458        let expected = "<a href=\"dir/my%20note%201.md\">my note 1</a>";
1459        input
1460            .rebase_local_link(root_path, docdir, false, false)
1461            .unwrap();
1462        let outpath = input.get_local_link_dest_path().unwrap();
1463        let output = input.to_html();
1464        assert_eq!(output, expected);
1465        assert_eq!(outpath, PathBuf::from("dir/my note 1.md"));
1466
1467        // Check relative link in input.
1468        let mut input = take_link("<a href=\"./down/./../dir/my note 1.md\">my note 1</a>")
1469            .unwrap()
1470            .1
1471            .1;
1472        let expected = "<a href=\"/path/dir/my%20note%201.md\">my note 1</a>";
1473        input
1474            .rebase_local_link(
1475                Path::new("/my/note/"),
1476                Path::new("/my/note/path/"),
1477                true,
1478                false,
1479            )
1480            .unwrap();
1481        let outpath = input.get_local_link_dest_path().unwrap();
1482        let output = input.to_html();
1483        assert_eq!(output, expected);
1484        assert_eq!(outpath, PathBuf::from("/path/dir/my note 1.md"));
1485
1486        // Check absolute link in input.
1487        let mut input = take_link("<a href=\"/down/./../dir/my note 1.md\">my note 1</a>")
1488            .unwrap()
1489            .1
1490            .1;
1491        let expected = "<a href=\"/dir/my%20note%201.md\">my note 1</a>";
1492        input
1493            .rebase_local_link(root_path, Path::new("/my/ignored/"), true, false)
1494            .unwrap();
1495        let outpath = input.get_local_link_dest_path().unwrap();
1496        let output = input.to_html();
1497        assert_eq!(output, expected);
1498        assert_eq!(outpath, PathBuf::from("/dir/my note 1.md"));
1499
1500        // Check absolute link in input, not in `root_path`.
1501        let mut input = take_link("<a href=\"/down/../../dir/my note 1.md\">my note 1</a>")
1502            .unwrap()
1503            .1
1504            .1;
1505        let output = input
1506            .rebase_local_link(root_path, Path::new("/my/notepath/"), true, false)
1507            .unwrap_err();
1508        assert!(matches!(output, NoteError::InvalidLocalPath { .. }));
1509
1510        // Check relative link in input, not in `root_path`.
1511        let mut input = take_link("<a href=\"../../dir/my note 1.md\">my note 1</a>")
1512            .unwrap()
1513            .1
1514            .1;
1515        let output = input
1516            .rebase_local_link(root_path, Path::new("/my/notepath/"), true, false)
1517            .unwrap_err();
1518        assert!(matches!(output, NoteError::InvalidLocalPath { .. }));
1519
1520        // Check relative link in input, with underflow.
1521        let root_path = Path::new("/");
1522        let mut input = take_link("<a href=\"../../dir/my note 1.md\">my note 1</a>")
1523            .unwrap()
1524            .1
1525            .1;
1526        let output = input
1527            .rebase_local_link(root_path, Path::new("/my/"), true, false)
1528            .unwrap_err();
1529        assert!(matches!(output, NoteError::InvalidLocalPath { .. }));
1530
1531        // Check relative link in input, not in `root_path`.
1532        let root_path = Path::new("/my");
1533        let mut input = take_link("<a href=\"../../dir/my note 1.md\">my note 1</a>")
1534            .unwrap()
1535            .1
1536            .1;
1537        let output = input
1538            .rebase_local_link(root_path, Path::new("/my/notepath"), true, false)
1539            .unwrap_err();
1540        assert!(matches!(output, NoteError::InvalidLocalPath { .. }));
1541
1542        // Test autolink.
1543        let root_path = Path::new("/my");
1544        let mut input =
1545            take_link("<a href=\"tpnote:dir/3.0-my note.md\">tpnote:dir/3.0-my note.md</a>")
1546                .unwrap()
1547                .1
1548                .1;
1549        input.strip_local_scheme();
1550        input
1551            .rebase_local_link(root_path, Path::new("/my/path"), true, false)
1552            .unwrap();
1553        input.rewrite_autolink();
1554        input.apply_format_attribute();
1555        let outpath = input.get_local_link_dest_path().unwrap();
1556        let output = input.to_html();
1557        let expected = "<a href=\"/path/dir/3.0-my%20note.md\">dir/3.0-my note.md</a>";
1558        assert_eq!(output, expected);
1559        assert_eq!(outpath, PathBuf::from("/path/dir/3.0-my note.md"));
1560
1561        // Test short autolink 1 with sort-tag only.
1562        let root_path = Path::new("/my");
1563        let mut input = take_link("<a href=\"tpnote:dir/3.0\">tpnote:dir/3.0</a>")
1564            .unwrap()
1565            .1
1566            .1;
1567        input.strip_local_scheme();
1568        input
1569            .rebase_local_link(root_path, Path::new("/my/path"), true, false)
1570            .unwrap();
1571        input.rewrite_autolink();
1572        input.apply_format_attribute();
1573        let outpath = input.get_local_link_dest_path().unwrap();
1574        let output = input.to_html();
1575        let expected = "<a href=\"/path/dir/3.0\">dir/3.0</a>";
1576        assert_eq!(output, expected);
1577        assert_eq!(outpath, PathBuf::from("/path/dir/3.0"));
1578
1579        // The link text contains inline content.
1580        let root_path = Path::new("/my");
1581        let mut input = take_link(
1582            "<a href=\
1583            \"/uri\">link <em>foo <strong>bar</strong> <code>#</code></em>\
1584            </a>",
1585        )
1586        .unwrap()
1587        .1
1588        .1;
1589        input.strip_local_scheme();
1590        input
1591            .rebase_local_link(root_path, Path::new("/my/path"), true, false)
1592            .unwrap();
1593        let outpath = input.get_local_link_dest_path().unwrap();
1594        let expected = "<a href=\"/uri\">link <em>foo <strong>bar\
1595            </strong> <code>#</code></em></a>";
1596
1597        let output = input.to_html();
1598        assert_eq!(output, expected);
1599        assert_eq!(outpath, PathBuf::from("/uri"));
1600    }
1601
1602    #[test]
1603    fn test_rewrite_autolink() {
1604        //
1605        let mut input = Link::Text2Dest(
1606            Cow::from("http://getreu.net"),
1607            Cow::from("http://getreu.net"),
1608            Cow::from("title"),
1609        );
1610        let expected = Link::Text2Dest(
1611            Cow::from("getreu.net"),
1612            Cow::from("http://getreu.net"),
1613            Cow::from("title"),
1614        );
1615        input.rewrite_autolink();
1616        let output = input;
1617        assert_eq!(output, expected);
1618
1619        //
1620        let mut input = Link::Text2Dest(
1621            Cow::from("/dir/3.0"),
1622            Cow::from("/dir/3.0-My note.md"),
1623            Cow::from("title"),
1624        );
1625        let expected = Link::Text2Dest(
1626            Cow::from("/dir/3.0"),
1627            Cow::from("/dir/3.0-My note.md"),
1628            Cow::from("title"),
1629        );
1630        input.rewrite_autolink();
1631        let output = input;
1632        assert_eq!(output, expected);
1633
1634        //
1635        let mut input = Link::Text2Dest(
1636            Cow::from("tpnote:/dir/3.0"),
1637            Cow::from("/dir/3.0-My note.md"),
1638            Cow::from("title"),
1639        );
1640        let expected = Link::Text2Dest(
1641            Cow::from("/dir/3.0"),
1642            Cow::from("/dir/3.0-My note.md"),
1643            Cow::from("title"),
1644        );
1645        input.rewrite_autolink();
1646        let output = input;
1647        assert_eq!(output, expected);
1648
1649        //
1650        let mut input = Link::Text2Dest(
1651            Cow::from("tpnote:/dir/3.0"),
1652            Cow::from("/dir/3.0-My note.md?"),
1653            Cow::from("title"),
1654        );
1655        let expected = Link::Text2Dest(
1656            Cow::from("/dir/3.0"),
1657            Cow::from("/dir/3.0-My note.md?"),
1658            Cow::from("title"),
1659        );
1660        input.rewrite_autolink();
1661        let output = input;
1662        assert_eq!(output, expected);
1663
1664        //
1665        let mut input = Link::Text2Dest(
1666            Cow::from("/dir/3.0-My note.md"),
1667            Cow::from("/dir/3.0-My note.md"),
1668            Cow::from("title"),
1669        );
1670        let expected = Link::Text2Dest(
1671            Cow::from("/dir/3.0-My note.md"),
1672            Cow::from("/dir/3.0-My note.md"),
1673            Cow::from("title"),
1674        );
1675        input.rewrite_autolink();
1676        let output = input;
1677        assert_eq!(output, expected);
1678    }
1679
1680    #[test]
1681    fn test_apply_format_attribute() {
1682        //
1683        let mut input = Link::Text2Dest(
1684            Cow::from("tpnote:/dir/3.0"),
1685            Cow::from("/dir/3.0-My note.md"),
1686            Cow::from("title"),
1687        );
1688        let expected = Link::Text2Dest(
1689            Cow::from("tpnote:/dir/3.0"),
1690            Cow::from("/dir/3.0-My note.md"),
1691            Cow::from("title"),
1692        );
1693        input.apply_format_attribute();
1694        let output = input;
1695        assert_eq!(output, expected);
1696
1697        //
1698        let mut input = Link::Text2Dest(
1699            Cow::from("does not matter"),
1700            Cow::from("/dir/3.0-My note.md?"),
1701            Cow::from("title"),
1702        );
1703        let expected = Link::Text2Dest(
1704            Cow::from("My note"),
1705            Cow::from("/dir/3.0-My note.md"),
1706            Cow::from("title"),
1707        );
1708        input.apply_format_attribute();
1709        let output = input;
1710        assert_eq!(output, expected);
1711
1712        let mut input = Link::Text2Dest(
1713            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1714            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1715            Cow::from("title"),
1716        );
1717        let expected = Link::Text2Dest(
1718            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1719            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1720            Cow::from("title"),
1721        );
1722        input.apply_format_attribute();
1723        let output = input;
1724        assert_eq!(output, expected);
1725
1726        //
1727        let mut input = Link::Text2Dest(
1728            Cow::from("does not matter"),
1729            Cow::from("/dir/3.0-My note--red_blue_green.jpg?"),
1730            Cow::from("title"),
1731        );
1732        let expected = Link::Text2Dest(
1733            Cow::from("My note--red_blue_green"),
1734            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1735            Cow::from("title"),
1736        );
1737        input.apply_format_attribute();
1738        let output = input;
1739        assert_eq!(output, expected);
1740
1741        //
1742        let mut input = Link::Text2Dest(
1743            Cow::from("does not matter"),
1744            Cow::from("/dir/3.0-My note--red_blue_green.jpg?--"),
1745            Cow::from("title"),
1746        );
1747        let expected = Link::Text2Dest(
1748            Cow::from("My note"),
1749            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1750            Cow::from("title"),
1751        );
1752        input.apply_format_attribute();
1753        let output = input;
1754        assert_eq!(output, expected);
1755
1756        //
1757        let mut input = Link::Text2Dest(
1758            Cow::from("does not matter"),
1759            Cow::from("/dir/3.0-My note--red_blue_green.jpg?_"),
1760            Cow::from("title"),
1761        );
1762        let expected = Link::Text2Dest(
1763            Cow::from("My note--red"),
1764            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1765            Cow::from("title"),
1766        );
1767        input.apply_format_attribute();
1768        let output = input;
1769        assert_eq!(output, expected);
1770
1771        //
1772        let mut input = Link::Text2Dest(
1773            Cow::from("does not matter"),
1774            Cow::from("/dir/3.0-My note--red_blue_green.jpg??"),
1775            Cow::from("title"),
1776        );
1777        let expected = Link::Text2Dest(
1778            Cow::from("3.0-My note--red_blue_green.jpg"),
1779            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1780            Cow::from("title"),
1781        );
1782        input.apply_format_attribute();
1783        let output = input;
1784        assert_eq!(output, expected);
1785
1786        //
1787        let mut input = Link::Text2Dest(
1788            Cow::from("does not matter"),
1789            Cow::from("/dir/3.0-My note--red_blue_green.jpg?#."),
1790            Cow::from("title"),
1791        );
1792        let expected = Link::Text2Dest(
1793            Cow::from("3"),
1794            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1795            Cow::from("title"),
1796        );
1797        input.apply_format_attribute();
1798        let output = input;
1799        assert_eq!(output, expected);
1800
1801        //
1802        let mut input = Link::Text2Dest(
1803            Cow::from("does not matter"),
1804            Cow::from("/dir/3.0-My note--red_blue_green.jpg??.:_"),
1805            Cow::from("title"),
1806        );
1807        let expected = Link::Text2Dest(
1808            Cow::from("0-My note--red"),
1809            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1810            Cow::from("title"),
1811        );
1812        input.apply_format_attribute();
1813        let output = input;
1814        assert_eq!(output, expected);
1815
1816        //
1817        let mut input = Link::Text2Dest(
1818            Cow::from("does not matter"),
1819            Cow::from("/dir/3.0-My note--red_blue_green.jpg?_:_"),
1820            Cow::from("title"),
1821        );
1822        let expected = Link::Text2Dest(
1823            Cow::from("blue"),
1824            Cow::from("/dir/3.0-My note--red_blue_green.jpg"),
1825            Cow::from("title"),
1826        );
1827        input.apply_format_attribute();
1828        let output = input;
1829        assert_eq!(output, expected);
1830    }
1831
1832    #[test]
1833    fn get_local_link_dest_path() {
1834        //
1835        let input = Link::Text2Dest(Cow::from("xyz"), Cow::from("/dir/3.0"), Cow::from("title"));
1836        assert_eq!(
1837            input.get_local_link_dest_path(),
1838            Some(Path::new("/dir/3.0"))
1839        );
1840
1841        //
1842        let input = Link::Text2Dest(
1843            Cow::from("xyz"),
1844            Cow::from("http://getreu.net"),
1845            Cow::from("title"),
1846        );
1847        assert_eq!(input.get_local_link_dest_path(), None);
1848
1849        //
1850        let input = Link::Text2Dest(Cow::from("xyz"), Cow::from("dir/doc.md"), Cow::from("xyz"));
1851        let expected = Path::new("dir/doc.md");
1852        let res = input.get_local_link_dest_path().unwrap();
1853        assert_eq!(res, expected);
1854
1855        //
1856        let input = Link::Text2Dest(Cow::from("xyz"), Cow::from("d#ir/doc.md"), Cow::from("xyz"));
1857        let expected = Path::new("d#ir/doc.md");
1858        let res = input.get_local_link_dest_path().unwrap();
1859        assert_eq!(res, expected);
1860
1861        //
1862        let input = Link::Text2Dest(
1863            Cow::from("xyz"),
1864            Cow::from("dir/doc.md#1"),
1865            Cow::from("xyz"),
1866        );
1867        let expected = Path::new("dir/doc.md");
1868        let res = input.get_local_link_dest_path().unwrap();
1869        assert_eq!(res, expected);
1870    }
1871
1872    #[test]
1873    fn test_split_path_and_fragment() {
1874        use crate::html::split_path_and_fragment;
1875
1876        // A `#` in a directory name is data, not a fragment: it precedes a
1877        // later separator, so it stays in the path half.
1878        assert_eq!(
1879            split_path_and_fragment("Task #7/note.md"),
1880            ("Task #7/note.md", "")
1881        );
1882
1883        // A `#` in the final segment, with no separator after it, is the
1884        // author's fragment.
1885        assert_eq!(
1886            split_path_and_fragment("note.md#anchor"),
1887            ("note.md", "#anchor")
1888        );
1889
1890        // Both at once: exactly one `#` survives as a fragment, the one
1891        // the author wrote.
1892        assert_eq!(
1893            split_path_and_fragment("Task #7/note.md#anchor"),
1894            ("Task #7/note.md", "#anchor")
1895        );
1896
1897        // No `#` at all.
1898        assert_eq!(split_path_and_fragment("dir/note.md"), ("dir/note.md", ""));
1899
1900        // A bare fragment, no path.
1901        assert_eq!(split_path_and_fragment("#1"), ("", "#1"));
1902    }
1903
1904    #[test]
1905    fn test_percent_encode_path() {
1906        use crate::html::percent_encode_path;
1907        use percent_encoding::percent_decode_str;
1908
1909        // Round-trip: decoding what we encode returns the original bytes.
1910        for segment in [
1911            "Meeting #12-x",
1912            "a?b",
1913            "100%",
1914            "report %23.md",
1915            "with space",
1916            "a+b",
1917            "a&b",
1918            "em—dash",
1919            "a↔b",
1920            "already%20encoded",
1921        ] {
1922            let path = format!("/dir/{segment}/note.md");
1923            let encoded = percent_encode_path(&path);
1924            let decoded = percent_decode_str(&encoded).decode_utf8().unwrap();
1925            assert_eq!(decoded, path, "round-trip failed for segment {segment:?}");
1926        }
1927
1928        // `#` and `?` are encoded so a browser cannot mistake them for URL
1929        // syntax.
1930        assert_eq!(
1931            percent_encode_path("/Meeting #12/note.md"),
1932            "/Meeting%20%2312/note.md"
1933        );
1934        assert_eq!(percent_encode_path("/a?b"), "/a%3Fb");
1935
1936        // `%` is encoded first (and only once): a literal `%23` in a file
1937        // name must not be reinterpreted as an encoded `#`.
1938        assert_eq!(percent_encode_path("/report %23.md"), "/report%20%2523.md");
1939        let encoded = percent_encode_path("/report %23.md");
1940        let decoded = percent_decode_str(&encoded).decode_utf8().unwrap();
1941        assert_eq!(decoded, "/report %23.md");
1942
1943        // The leading `/` and the `/` separators are never encoded.
1944        assert!(percent_encode_path("/a/b/c").starts_with('/'));
1945        assert_eq!(percent_encode_path("/a/b/c"), "/a/b/c");
1946    }
1947
1948    #[test]
1949    fn test_append_html_ext() {
1950        //
1951        let mut input = Link::Text2Dest(
1952            Cow::from("abc"),
1953            Cow::from("/dir/3.0-My note.md"),
1954            Cow::from("title"),
1955        );
1956        let expected = Link::Text2Dest(
1957            Cow::from("abc"),
1958            Cow::from("/dir/3.0-My note.md.html"),
1959            Cow::from("title"),
1960        );
1961        input.append_html_ext();
1962        let output = input;
1963        assert_eq!(output, expected);
1964    }
1965
1966    #[test]
1967    fn test_append_html_ext_with_fragment() {
1968        // The fragment must survive, reattached after the appended `.html`.
1969        let mut input = Link::Text2Dest(
1970            Cow::from("abc"),
1971            Cow::from("/dir/3.0-My note.md#ch1"),
1972            Cow::from("title"),
1973        );
1974        let expected = Link::Text2Dest(
1975            Cow::from("abc"),
1976            Cow::from("/dir/3.0-My note.md.html#ch1"),
1977            Cow::from("title"),
1978        );
1979        input.append_html_ext();
1980        let output = input;
1981        assert_eq!(output, expected);
1982    }
1983
1984    #[test]
1985    fn test_to_html() {
1986        //
1987        let input = Link::Text2Dest(
1988            Cow::from("te\\x/t"),
1989            Cow::from("de\\s/t"),
1990            Cow::from("ti\\t/le"),
1991        );
1992        let expected = "<a href=\"de/s/t\" title=\"ti\\t/le\">te\\x/t</a>";
1993        let output = input.to_html();
1994        assert_eq!(output, expected);
1995
1996        //
1997        let input = Link::Text2Dest(
1998            Cow::from("te&> xt"),
1999            Cow::from("de&> st"),
2000            Cow::from("ti&> tle"),
2001        );
2002        let expected = "<a href=\"de&amp;&gt;%20st\" title=\"ti&amp;&gt; tle\">te&> xt</a>";
2003        let output = input.to_html();
2004        assert_eq!(output, expected);
2005
2006        //
2007        let input = Link::Image(Cow::from("al&t"), Cow::from("sr&c"));
2008        let expected = "<img src=\"sr&amp;c\" alt=\"al&amp;t\">";
2009        let output = input.to_html();
2010        assert_eq!(output, expected);
2011
2012        //
2013        let input = Link::Text2Dest(Cow::from("te&> xt"), Cow::from("de&> st"), Cow::from(""));
2014        let expected = "<a href=\"de&amp;&gt;%20st\">te&> xt</a>";
2015        let output = input.to_html();
2016        assert_eq!(output, expected);
2017    }
2018
2019    #[test]
2020    fn test_rewrite_links() {
2021        use crate::config::LocalLinkKind;
2022
2023        let allowed_urls = Arc::new(RwLock::new(HashSet::new()));
2024        let input = "abc<a href=\"ftp://getreu.net\">Blog</a>\
2025            def<a href=\"https://getreu.net\">https://getreu.net</a>\
2026            ghi<img src=\"t m p.jpg\" alt=\"test 1\" />\
2027            jkl<a href=\"down/../down/my note 1.md\">my note 1</a>\
2028            mno<a href=\"http:./down/../dir/my note.md\">http:./down/../dir/my note.md</a>\
2029            pqr<a href=\"http:/down/../dir/my note.md\">\
2030            http:/down/../dir/my note.md</a>\
2031            stu<a href=\"http:/../dir/underflow/my note.md\">\
2032            not allowed dir</a>\
2033            vwx<a href=\"http:../../../not allowed dir/my note.md\">\
2034            not allowed</a>"
2035            .to_string();
2036        let expected = "abc<a href=\"ftp://getreu.net\">Blog</a>\
2037            def<a href=\"https://getreu.net\">getreu.net</a>\
2038            ghi<img src=\"/abs/note%20path/t%20m%20p.jpg\" alt=\"test 1\">\
2039            jkl<a href=\"/abs/note%20path/down/my%20note%201.md\">my note 1</a>\
2040            mno<a href=\"/abs/note%20path/dir/my%20note.md\">./down/../dir/my note.md</a>\
2041            pqr<a href=\"/dir/my%20note.md\">/down/../dir/my note.md</a>\
2042            stu<i>&lt;INVALID: /../dir/underflow/my note.md&gt;</i>\
2043            vwx<i>&lt;INVALID: ../../../not allowed dir/my note.md&gt;</i>"
2044            .to_string();
2045
2046        let root_path = Path::new("/my/");
2047        let docdir = Path::new("/my/abs/note path/");
2048        let output = rewrite_links(
2049            input,
2050            root_path,
2051            docdir,
2052            LocalLinkKind::Short,
2053            false,
2054            allowed_urls.clone(),
2055        );
2056        let url = allowed_urls.read_recursive();
2057
2058        assert!(url.contains(&PathBuf::from("/abs/note path/t m p.jpg")));
2059        assert!(url.contains(&PathBuf::from("/abs/note path/dir/my note.md")));
2060        assert!(url.contains(&PathBuf::from("/abs/note path/down/my note 1.md")));
2061        assert_eq!(output, expected);
2062    }
2063
2064    #[test]
2065    fn test_rewrite_links2() {
2066        use crate::config::LocalLinkKind;
2067
2068        let allowed_urls = Arc::new(RwLock::new(HashSet::new()));
2069        let input = "abd<a href=\"tpnote:dir/my note.md\">\
2070            <img src=\"/imagedir/favicon-32x32.png\" alt=\"logo\"></a>abd"
2071            .to_string();
2072        let expected = "abd<a href=\"/abs/note%20path/dir/my%20note.md\">\
2073            <img src=\"/imagedir/favicon-32x32.png\" alt=\"logo\"></a>abd";
2074        let root_path = Path::new("/my/");
2075        let docdir = Path::new("/my/abs/note path/");
2076        let output = rewrite_links(
2077            input,
2078            root_path,
2079            docdir,
2080            LocalLinkKind::Short,
2081            false,
2082            allowed_urls.clone(),
2083        );
2084        let url = allowed_urls.read_recursive();
2085        println!("{:?}", allowed_urls.read_recursive());
2086        assert!(url.contains(&PathBuf::from("/abs/note path/dir/my note.md")));
2087        assert_eq!(output, expected);
2088    }
2089
2090    #[test]
2091    fn test_rewrite_links3() {
2092        use crate::config::LocalLinkKind;
2093
2094        // A bare fragment (`#1`) denotes the current document: there is no
2095        // path to rebase, so it must survive every rewriting mode verbatim,
2096        // and it must not register the docdir as an "allowed" local link,
2097        // since no separate resource is referenced.
2098        let allowed_urls = Arc::new(RwLock::new(HashSet::new()));
2099        let input = "abd<a href=\"#1\"></a>abd".to_string();
2100        let expected = "abd<a href=\"#1\"></a>abd";
2101        let root_path = Path::new("/my/");
2102        let docdir = Path::new("/my/abs/note path/");
2103        let output = rewrite_links(
2104            input,
2105            root_path,
2106            docdir,
2107            LocalLinkKind::Short,
2108            false,
2109            allowed_urls.clone(),
2110        );
2111        let url = allowed_urls.read_recursive();
2112        println!("{:?}", allowed_urls.read_recursive());
2113        assert!(!url.contains(&PathBuf::from("/abs/note path/")));
2114        assert_eq!(output, expected);
2115    }
2116
2117    #[test]
2118    fn test_rewrite_links_bare_fragment_all_modes() {
2119        use crate::config::LocalLinkKind;
2120
2121        // The bug-report fixture: `[Chapter one](#ch1)` must resolve to
2122        // `#ch1` verbatim, regardless of `LocalLinkKind`.
2123        let root_path = Path::new("/my/");
2124        let docdir = Path::new("/my/abs/note path/");
2125        let input = "<a href=\"#ch1\">Chapter one</a>".to_string();
2126        let expected = "<a href=\"#ch1\">Chapter one</a>";
2127
2128        for kind in [LocalLinkKind::Off, LocalLinkKind::Short, LocalLinkKind::Long] {
2129            let allowed_urls = Arc::new(RwLock::new(HashSet::new()));
2130            let output = rewrite_links(
2131                input.clone(),
2132                root_path,
2133                docdir,
2134                kind,
2135                false,
2136                allowed_urls,
2137            );
2138            assert_eq!(output, expected, "mode {kind:?} must leave a bare fragment untouched");
2139        }
2140    }
2141
2142    /// The feature-request's own acceptance-test fixture, rendered as the
2143    /// HTML `pulldown-cmark` would already have produced (this tests
2144    /// `assign_heading_ids` in isolation, not the Markdown renderer).
2145    const HEADING_FIXTURE: &str = concat!(
2146        "<h2>Chapter one</h2>",
2147        "<h2 id=\"ch1\">Chapter one</h2>",
2148        "<h3>S9 — Check the PIN</h3>",
2149        "<h3>2. Second section</h3>",
2150        "<h3>Duplicate</h3>",
2151        "<h3>Duplicate</h3>",
2152        "<h3><em>Emphasis</em> and <code>code</code></h3>",
2153        "<h3>Ümlaut und Größe</h3>",
2154        "<h3>Trailing punctuation!</h3>",
2155    );
2156
2157    #[test]
2158    fn test_assign_heading_ids_gfm() {
2159        use crate::config::HeadingIdPolicy;
2160
2161        let expected = concat!(
2162            "<h2 id=\"chapter-one\">Chapter one</h2>",
2163            "<h2 id=\"ch1\">Chapter one</h2>",
2164            "<h3 id=\"s9--check-the-pin\">S9 — Check the PIN</h3>",
2165            "<h3 id=\"2-second-section\">2. Second section</h3>",
2166            "<h3 id=\"duplicate\">Duplicate</h3>",
2167            "<h3 id=\"duplicate-1\">Duplicate</h3>",
2168            "<h3 id=\"emphasis-and-code\"><em>Emphasis</em> and <code>code</code></h3>",
2169            "<h3 id=\"ümlaut-und-größe\">Ümlaut und Größe</h3>",
2170            "<h3 id=\"trailing-punctuation\">Trailing punctuation!</h3>",
2171        );
2172
2173        let output = assign_heading_ids(HEADING_FIXTURE.to_string(), HeadingIdPolicy::Gfm);
2174        assert_eq!(output, expected);
2175        assert!(!output.contains("{#ch1}"), "raw heading-attribute syntax must never leak");
2176    }
2177
2178    #[test]
2179    fn test_assign_heading_ids_pandoc() {
2180        use crate::config::HeadingIdPolicy;
2181
2182        // Differs from Gfm on exactly one row: the leading `2. ` is
2183        // dropped entirely, rather than keeping the digit.
2184        let expected = concat!(
2185            "<h2 id=\"chapter-one\">Chapter one</h2>",
2186            "<h2 id=\"ch1\">Chapter one</h2>",
2187            "<h3 id=\"s9--check-the-pin\">S9 — Check the PIN</h3>",
2188            "<h3 id=\"second-section\">2. Second section</h3>",
2189            "<h3 id=\"duplicate\">Duplicate</h3>",
2190            "<h3 id=\"duplicate-1\">Duplicate</h3>",
2191            "<h3 id=\"emphasis-and-code\"><em>Emphasis</em> and <code>code</code></h3>",
2192            "<h3 id=\"ümlaut-und-größe\">Ümlaut und Größe</h3>",
2193            "<h3 id=\"trailing-punctuation\">Trailing punctuation!</h3>",
2194        );
2195
2196        let output = assign_heading_ids(HEADING_FIXTURE.to_string(), HeadingIdPolicy::Pandoc);
2197        assert_eq!(output, expected);
2198    }
2199
2200    #[test]
2201    fn test_assign_heading_ids_off() {
2202        use crate::config::HeadingIdPolicy;
2203
2204        let output = assign_heading_ids(HEADING_FIXTURE.to_string(), HeadingIdPolicy::Off);
2205        assert_eq!(output, HEADING_FIXTURE, "Off must leave the HTML byte-for-byte unchanged");
2206    }
2207
2208    #[test]
2209    fn test_assign_heading_ids_avoids_colliding_with_explicit_id() {
2210        use crate::config::HeadingIdPolicy;
2211
2212        // A later auto-generated slug that would collide with an EARLIER
2213        // explicit `{#id}` must be disambiguated, not silently duplicated.
2214        let input = "<h2 id=\"duplicate\">Explicit</h2><h2>Duplicate</h2>".to_string();
2215        let expected = "<h2 id=\"duplicate\">Explicit</h2><h2 id=\"duplicate-1\">Duplicate</h2>";
2216
2217        let output = assign_heading_ids(input, HeadingIdPolicy::Gfm);
2218        assert_eq!(output, expected);
2219    }
2220
2221    /// A `#` in a directory name must not end up as a bare byte in the
2222    /// `href`: browsers read an unencoded `#` as the start of a fragment
2223    /// and never send anything after it, so the server only ever sees a
2224    /// truncated path. `rewrite_links` is the function shared by the
2225    /// viewer and `--export`, so this covers both call sites at once.
2226    #[test]
2227    fn test_rewrite_links_hash_in_dir_name() {
2228        use crate::config::LocalLinkKind;
2229
2230        let allowed_urls = Arc::new(RwLock::new(HashSet::new()));
2231        let input = "<a href=\"01-Agenda.md\">link</a>".to_string();
2232        let root_path = Path::new("/notes/");
2233        let docdir = Path::new("/notes/Meeting #12-Project kickoff/");
2234        let output = rewrite_links(
2235            input,
2236            root_path,
2237            docdir,
2238            LocalLinkKind::Short,
2239            false,
2240            allowed_urls.clone(),
2241        );
2242
2243        // The `#` that is part of the directory name is percent-encoded,
2244        // so the browser cannot mistake it for the start of a fragment.
2245        assert!(
2246            output.contains("href=\"/Meeting%20%2312-Project%20kickoff/01-Agenda.md\""),
2247            "unexpected output: {output}"
2248        );
2249        // No bare `#` remains in the href.
2250        assert!(!output.contains("Meeting #12"));
2251
2252        // Bookkeeping still holds the raw, decoded filesystem path — this
2253        // is what the viewer compares an incoming (percent-decoded)
2254        // request path against, so encoding the `href` must not encode
2255        // this side too.
2256        let url = allowed_urls.read_recursive();
2257        assert!(url.contains(&PathBuf::from(
2258            "/Meeting #12-Project kickoff/01-Agenda.md"
2259        )));
2260    }
2261
2262    #[test]
2263    fn test_is_empty_html() {
2264        // Bring new methods into scope.
2265        use crate::html::HtmlStr;
2266
2267        // Test where input is '<!DOCTYPE html>'
2268        // See: [HTML doctype declaration](https://www.w3schools.com/tags/tag_doctype.ASP)
2269        assert!(String::from("<!DOCTYPE html>").is_empty_html());
2270
2271        // This should fail:
2272        assert!(!String::from("<!DOCTYPE html>>").is_empty_html());
2273
2274        // Test where input is '<!DOCTYPE html>'
2275        // See: [HTML doctype declaration](https://www.w3schools.com/tags/tag_doctype.ASP)
2276        assert!(
2277            String::from(
2278                " <!DOCTYPE HTML PUBLIC \
2279            \"-//W3C//DTD HTML 4.01 Transitional//EN\" \
2280            \"http://www.w3.org/TR/html4/loose.dtd\">"
2281            )
2282            .is_empty_html()
2283        );
2284
2285        // Test where input is '<!DOCTYPE html>'
2286        // See: [HTML doctype declaration](https://www.w3schools.com/tags/tag_doctype.ASP)
2287        assert!(
2288            String::from(
2289                " <!DOCTYPE html PUBLIC \
2290            \"-//W3C//DTD XHTML 1.1//EN\" \
2291            \"http://www.w3.org/TR/xhtml11/DTD/xhtml11.dtd\">"
2292            )
2293            .is_empty_html()
2294        );
2295
2296        // Test where input is '<!DOCTYPE html>Some content'
2297        assert!(!String::from("<!DOCTYPE html>Some content").is_empty_html());
2298
2299        // Test where input is an empty string
2300        assert!(String::from("").is_empty_html());
2301
2302        // Test where input is not empty HTML.
2303        // Convention: we consider empty only `` or `<!DOCTYPE html>`.
2304        assert!(!String::from("<html></html>").is_empty_html());
2305
2306        // Test where input is not empty HTML with doctype
2307        // Convention: we consider empty only `` or `<!DOCTYPE html>`.
2308        assert!(!String::from("<!DOCTYPE html><html></html>").is_empty_html());
2309    }
2310
2311    #[test]
2312    fn test_has_html_start_tag() {
2313        // Bring new methods into scope.
2314        use crate::html::HtmlStr;
2315
2316        // Test where input is '<!DOCTYPE html>Some content'
2317        assert!(String::from("<!DOCTYPE html>Some content").has_html_start_tag());
2318
2319        // This fails because we require be convention `<!DOCTYPE html>` as
2320        // first tag
2321        assert!(!String::from("<html>Some content</html>").has_html_start_tag());
2322
2323        // This fails because we require be convention `<!DOCTYPE html>` as
2324        // first tag
2325        assert!(!String::from("<HTML>").has_html_start_tag());
2326
2327        // Test where input starts with spaces
2328        assert!(String::from("  <!doctype html>Some content").has_html_start_tag());
2329
2330        // Test where input is a non-HTML doctype
2331        assert!(!String::from("<!DOCTYPE other>").has_html_start_tag());
2332
2333        // Test where input is an empty string
2334        assert!(!String::from("").has_html_start_tag());
2335    }
2336
2337    #[test]
2338    fn test_is_html_unchecked() {
2339        // Bring new methods into scope.
2340        use crate::html::HtmlStr;
2341
2342        // Test with `<!DOCTYPE html>` tag
2343        let html = "<!doctype html>";
2344        assert!(html.is_html_unchecked());
2345
2346        // Test with `<!DOCTYPE html>` tag
2347        let html = "<!doctype html abc>def";
2348        assert!(html.is_html_unchecked());
2349
2350        // Test with `<!DOCTYPE html>` tag
2351        let html = "<!doctype html";
2352        assert!(!html.is_html_unchecked());
2353
2354        // Test with `<html>` tag
2355        let html = "<html><body></body></html>";
2356        assert!(html.is_html_unchecked());
2357
2358        // Test with `<html>` tag
2359        let html = "<html abc>def";
2360        assert!(html.is_html_unchecked());
2361
2362        // Test with `<html>` tag
2363        let html = "<html abc def";
2364        assert!(!html.is_html_unchecked());
2365
2366        // Test with leading whitespace
2367        let html = "   <!doctype html><html><body></body></html>";
2368        assert!(html.is_html_unchecked());
2369
2370        // Test with non-html content
2371        let html = "<!DOCTYPE xml><root></root>";
2372        assert!(!html.is_html_unchecked());
2373
2374        // Test with partial `<!DOCTYPE>` tag
2375        let html = "<!doctype>";
2376        assert!(!html.is_html_unchecked());
2377    }
2378
2379    #[test]
2380    fn test_prepend_html_start_tag() {
2381        // Bring new methods into scope.
2382        use crate::html::HtmlString;
2383
2384        // Test where input already has doctype HTML
2385        assert_eq!(
2386            String::from("<!DOCTYPE html>Some content").prepend_html_start_tag(),
2387            Ok(String::from("<!DOCTYPE html>Some content"))
2388        );
2389
2390        // Test where input already has doctype HTML
2391        assert_eq!(
2392            String::from("<!DOCTYPE html>").prepend_html_start_tag(),
2393            Ok(String::from("<!DOCTYPE html>"))
2394        );
2395
2396        // Test where input has no HTML tag
2397        assert_eq!(
2398            String::from("<html>Some content").prepend_html_start_tag(),
2399            Ok(String::from("<!DOCTYPE html><html>Some content"))
2400        );
2401
2402        // Test where input has a non-HTML doctype
2403        assert_eq!(
2404            String::from("<!DOCTYPE other>").prepend_html_start_tag(),
2405            Err(InputStreamError::NonHtmlDoctype {
2406                html: "<!DOCTYPE other>".to_string()
2407            })
2408        );
2409
2410        // Test where input has no HTML tag
2411        assert_eq!(
2412            String::from("Some content").prepend_html_start_tag(),
2413            Ok(String::from("<!DOCTYPE html>Some content"))
2414        );
2415
2416        // Test where input is an empty string
2417        assert_eq!(
2418            String::from("").prepend_html_start_tag(),
2419            Ok(String::from("<!DOCTYPE html>"))
2420        );
2421    }
2422}