Skip to main content

codoseo_crawler/
extract.rs

1//! Pulls SEO fields out of an HTML page as it streams in, without building a DOM.
2//!
3//! Encoding: a BOM wins, then the `Content-Type` charset, then `<meta charset>`
4//! (which lol_html switches to mid-stream), then UTF-8. Encodings that aren't
5//! ASCII-compatible (UTF-16) are transcoded to UTF-8 before parsing.
6//!
7//! Text comes from one document-wide handler and is routed by state that the
8//! element handlers keep (inside `<title>`, a heading, a link, a skipped element).
9//! Malformed input never panics: a parser error just ends extraction early.
10
11use std::cell::RefCell;
12use std::rc::Rc;
13
14use codoseo_core::page::{JsonLdStatus, OgTags, PageFields};
15use codoseo_core::url::normalize;
16use encoding_rs::{Decoder, Encoding, UTF_8, UTF_16BE, UTF_16LE};
17use lol_html::html_content::{Element, EndTag};
18use lol_html::{AsciiCompatibleEncoding, HandlerResult, HtmlRewriter, Settings, doc_text, element};
19use serde::Serialize;
20use url::Url;
21use xxhash_rust::xxh3::Xxh3;
22
23const MAX_TITLE_CHARS: usize = 1_000;
24const MAX_HEADING_CHARS: usize = 300;
25const MAX_ANCHOR_CHARS: usize = 200;
26const MAX_HEADINGS: usize = 50;
27const MAX_LINKS: usize = 5_000;
28/// JSON-LD blocks bigger than this are reported as too large instead of parsed.
29const MAX_JSONLD_BYTES: usize = 1024 * 1024;
30/// How far to look for `<meta charset>` before parsing, as browsers do.
31const PRESCAN_BYTES: usize = 1024;
32
33#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
34pub struct Link {
35    pub url: Url,
36    pub anchor: String,
37    pub nofollow: bool,
38}
39
40#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
41pub struct Extracted {
42    pub fields: PageFields,
43    pub links: Vec<Link>,
44}
45
46/// Extracts everything from a complete body in one call.
47pub fn extract(page_url: &Url, content_type: Option<&str>, body: &[u8]) -> Extracted {
48    let mut e = Extractor::new(page_url, content_type);
49    e.write(body);
50    e.finish()
51}
52
53type Rewriter = HtmlRewriter<'static, fn(&[u8])>;
54
55enum Stage {
56    /// Holding the first bytes until we can check for a BOM.
57    Sniffing(Vec<u8>),
58    Direct(Rewriter),
59    Transcoding(Decoder, Rewriter),
60    Stopped,
61}
62
63pub struct Extractor {
64    page_url: Url,
65    header_encoding: Option<&'static Encoding>,
66    state: Rc<RefCell<State>>,
67    stage: Stage,
68}
69
70impl Extractor {
71    pub fn new(page_url: &Url, content_type: Option<&str>) -> Extractor {
72        Extractor {
73            page_url: page_url.clone(),
74            header_encoding: content_type.and_then(charset_from_content_type),
75            state: Rc::new(RefCell::new(State::new(page_url.scheme() == "https"))),
76            stage: Stage::Sniffing(Vec::new()),
77        }
78    }
79
80    pub fn write(&mut self, chunk: &[u8]) {
81        if let Stage::Sniffing(pending) = &mut self.stage {
82            pending.extend_from_slice(chunk);
83            if pending.len() >= PRESCAN_BYTES {
84                let first = std::mem::take(pending);
85                self.begin(&first);
86            }
87            return;
88        }
89        self.feed(chunk);
90    }
91
92    pub fn finish(mut self) -> Extracted {
93        if let Stage::Sniffing(pending) = &mut self.stage {
94            let first = std::mem::take(pending);
95            self.begin(&first);
96        }
97        match std::mem::replace(&mut self.stage, Stage::Stopped) {
98            Stage::Direct(rw) => {
99                let _ = rw.end();
100            }
101            Stage::Transcoding(mut decoder, mut rw) => {
102                let tail = decode(&mut decoder, &[], true);
103                if rw.write(tail.as_bytes()).is_ok() {
104                    let _ = rw.end();
105                }
106            }
107            Stage::Sniffing(_) | Stage::Stopped => {}
108        }
109        let mut state = self.state.borrow_mut();
110        state.close_open_elements();
111        state.build(&self.page_url)
112    }
113
114    fn begin(&mut self, first: &[u8]) {
115        let (encoding, skip, adjust_on_meta) = if first.starts_with(&[0xEF, 0xBB, 0xBF]) {
116            (UTF_8, 3, false)
117        } else if first.starts_with(&[0xFF, 0xFE]) {
118            (UTF_16LE, 2, false)
119        } else if first.starts_with(&[0xFE, 0xFF]) {
120            (UTF_16BE, 2, false)
121        } else if let Some(enc) = self.header_encoding {
122            (enc, 0, false)
123        } else if let Some(enc) = prescan_meta_charset(first) {
124            (enc, 0, false)
125        } else {
126            // A meta tag past the pre-scan window can still switch encoding mid-stream.
127            (UTF_8, 0, true)
128        };
129        self.stage = match AsciiCompatibleEncoding::new(encoding) {
130            Some(ascii) => Stage::Direct(build_rewriter(&self.state, ascii, adjust_on_meta)),
131            None => {
132                let utf8 = AsciiCompatibleEncoding::new(UTF_8).expect("UTF-8 is ASCII-compatible");
133                Stage::Transcoding(
134                    encoding.new_decoder_without_bom_handling(),
135                    build_rewriter(&self.state, utf8, false),
136                )
137            }
138        };
139        self.feed(&first[skip.min(first.len())..]);
140    }
141
142    fn feed(&mut self, chunk: &[u8]) {
143        let ok = match &mut self.stage {
144            Stage::Direct(rw) => rw.write(chunk).is_ok(),
145            Stage::Transcoding(decoder, rw) => {
146                let text = decode(decoder, chunk, false);
147                rw.write(text.as_bytes()).is_ok()
148            }
149            Stage::Sniffing(_) | Stage::Stopped => true,
150        };
151        if !ok {
152            self.stage = Stage::Stopped;
153        }
154    }
155}
156
157fn decode(decoder: &mut Decoder, src: &[u8], last: bool) -> String {
158    let mut out = String::with_capacity(
159        decoder
160            .max_utf8_buffer_length(src.len())
161            .unwrap_or(src.len() * 3),
162    );
163    let _ = decoder.decode_to_string(src, &mut out, last);
164    out
165}
166
167fn charset_from_content_type(ct: &str) -> Option<&'static Encoding> {
168    ct.split(';').skip(1).find_map(|param| {
169        let (key, value) = param.split_once('=')?;
170        if !key.trim().eq_ignore_ascii_case("charset") {
171            return None;
172        }
173        Encoding::for_label(value.trim().trim_matches(['"', '\'']).as_bytes())
174    })
175}
176
177fn discard(_: &[u8]) {}
178
179fn build_rewriter(
180    state: &Rc<RefCell<State>>,
181    encoding: AsciiCompatibleEncoding,
182    adjust_on_meta: bool,
183) -> Rewriter {
184    let s = |state: &Rc<RefCell<State>>| Rc::clone(state);
185    let (
186        s_any,
187        s_svg,
188        s_title,
189        s_meta,
190        s_link,
191        s_base,
192        s_head,
193        s_a,
194        s_img,
195        s_script,
196        s_skip,
197        s_media,
198        s_text,
199    ) = (
200        s(state),
201        s(state),
202        s(state),
203        s(state),
204        s(state),
205        s(state),
206        s(state),
207        s(state),
208        s(state),
209        s(state),
210        s(state),
211        s(state),
212        s(state),
213    );
214    let settings = Settings::new()
215        .with_encoding(encoding)
216        .with_adjust_charset_on_meta_tag(adjust_on_meta)
217        .append_element_content_handler(element!("*", move |el| {
218            let mut st = s_any.borrow_mut();
219            st.words.break_word();
220            if is_block_or_break(&el.tag_name()) {
221                st.separate_open_text();
222            }
223            Ok(())
224        }))
225        .append_element_content_handler(element!("svg", move |el| {
226            s_svg.borrow_mut().svg_depth += 1;
227            let st = Rc::clone(&s_svg);
228            on_end(el, move || st.borrow_mut().svg_depth -= 1)
229        }))
230        .append_element_content_handler(element!("title", move |el| {
231            let mut st = s_title.borrow_mut();
232            if st.svg_depth > 0 {
233                return Ok(());
234            }
235            st.title_count = st.title_count.saturating_add(1);
236            st.title_buf = Some(String::new());
237            drop(st);
238            let st = Rc::clone(&s_title);
239            on_end(el, move || st.borrow_mut().close_title())
240        }))
241        .append_element_content_handler(element!("meta", move |el| {
242            s_meta.borrow_mut().on_meta(el);
243            Ok(())
244        }))
245        .append_element_content_handler(element!("link[href]", move |el| {
246            s_link.borrow_mut().on_link(el);
247            Ok(())
248        }))
249        .append_element_content_handler(element!("base[href]", move |el| {
250            let mut st = s_base.borrow_mut();
251            if st.base_raw.is_none() {
252                st.base_raw = el.get_attribute("href");
253            }
254            Ok(())
255        }))
256        .append_element_content_handler(element!("h1, h2", move |el| {
257            let level = if el.tag_name().eq_ignore_ascii_case("h1") {
258                1
259            } else {
260                2
261            };
262            let mut st = s_head.borrow_mut();
263            st.close_heading(); // an unclosed heading ends where the next one starts
264            st.heading = Some((level, String::new()));
265            drop(st);
266            let st = Rc::clone(&s_head);
267            on_end(el, move || st.borrow_mut().close_heading())
268        }))
269        .append_element_content_handler(element!("a[href]", move |el| {
270            let mut st = s_a.borrow_mut();
271            if st.links_raw.len() >= MAX_LINKS {
272                return Ok(());
273            }
274            let rel = el
275                .get_attribute("rel")
276                .unwrap_or_default()
277                .to_ascii_lowercase();
278            let nofollow = rel.split_ascii_whitespace().any(|r| r == "nofollow");
279            st.links_raw.push(RawLink {
280                href: el.get_attribute("href").unwrap_or_default(),
281                anchor: String::new(),
282                nofollow,
283            });
284            st.anchor_open = true;
285            drop(st);
286            let st = Rc::clone(&s_a);
287            on_end(el, move || st.borrow_mut().anchor_open = false)
288        }))
289        .append_element_content_handler(element!("img", move |el| {
290            let mut st = s_img.borrow_mut();
291            if !el.has_attribute("alt") {
292                st.images_missing_alt += 1;
293            }
294            st.check_mixed(el.get_attribute("src").as_deref());
295            Ok(())
296        }))
297        .append_element_content_handler(element!("script", move |el| {
298            let mut st = s_script.borrow_mut();
299            st.check_mixed(el.get_attribute("src").as_deref());
300            let is_jsonld = el
301                .get_attribute("type")
302                .is_some_and(|t| t.trim().eq_ignore_ascii_case("application/ld+json"));
303            if is_jsonld {
304                st.jsonld_buf = Some(String::new());
305            }
306            st.skip_depth += 1;
307            drop(st);
308            let st = Rc::clone(&s_script);
309            on_end(el, move || st.borrow_mut().close_script())
310        }))
311        .append_element_content_handler(element!("style, noscript, template", move |el| {
312            s_skip.borrow_mut().skip_depth += 1;
313            let st = Rc::clone(&s_skip);
314            on_end(el, move || st.borrow_mut().skip_depth -= 1)
315        }))
316        .append_element_content_handler(element!(
317            "iframe[src], video[src], audio[src], source[src], embed[src]",
318            move |el| {
319                s_media
320                    .borrow_mut()
321                    .check_mixed(el.get_attribute("src").as_deref());
322                Ok(())
323            }
324        ))
325        .append_document_content_handler(doc_text!(move |t| {
326            s_text.borrow_mut().on_text(t.as_str());
327            Ok(())
328        }));
329    HtmlRewriter::new(settings, discard as fn(&[u8]))
330}
331
332/// Runs `f` at the element's end tag, or right away if it can't have one.
333fn on_end(el: &mut Element<'_, '_>, f: impl FnOnce() + 'static) -> HandlerResult {
334    if el.can_have_content() && !el.is_self_closing() {
335        el.on_end_tag(Box::new(move |_: &mut EndTag<'_>| {
336            f();
337            Ok(())
338        }))?;
339    } else {
340        f();
341    }
342    Ok(())
343}
344
345struct RawLink {
346    href: String,
347    anchor: String,
348    nofollow: bool,
349}
350
351/// Counts words and hashes the visible text with whitespace collapsed.
352struct WordCounter {
353    count: u32,
354    in_word: bool,
355    space_pending: bool,
356    hashed_any: bool,
357    hasher: Xxh3,
358}
359
360impl WordCounter {
361    fn feed(&mut self, text: &str) {
362        let mut buf = [0u8; 4];
363        for ch in text.chars() {
364            if ch.is_whitespace() {
365                self.break_word();
366                continue;
367            }
368            if !self.in_word {
369                self.count = self.count.saturating_add(1);
370                self.in_word = true;
371                if self.space_pending && self.hashed_any {
372                    self.hasher.update(b" ");
373                }
374                self.space_pending = false;
375            }
376            self.hasher.update(ch.encode_utf8(&mut buf).as_bytes());
377            self.hashed_any = true;
378        }
379    }
380
381    fn break_word(&mut self) {
382        if self.in_word {
383            self.in_word = false;
384            self.space_pending = true;
385        }
386    }
387}
388
389struct State {
390    page_is_https: bool,
391    svg_depth: u32,
392    skip_depth: u32,
393    title: Option<String>,
394    title_count: u8,
395    title_buf: Option<String>,
396    meta_description: Option<String>,
397    meta_robots: Vec<String>,
398    canonical_raw: Option<String>,
399    hreflang_raw: Vec<(String, String)>,
400    base_raw: Option<String>,
401    h1: Vec<String>,
402    h2: Vec<String>,
403    heading: Option<(u8, String)>,
404    links_raw: Vec<RawLink>,
405    anchor_open: bool,
406    images_missing_alt: u32,
407    og: OgTags,
408    jsonld_buf: Option<String>,
409    /// The open JSON-LD block went over [`MAX_JSONLD_BYTES`]; its text is dropped.
410    jsonld_overflow: bool,
411    jsonld_valid: u16,
412    jsonld_invalid: bool,
413    jsonld_too_large: bool,
414    mixed_content: u32,
415    words: WordCounter,
416}
417
418impl State {
419    fn new(page_is_https: bool) -> State {
420        State {
421            page_is_https,
422            svg_depth: 0,
423            skip_depth: 0,
424            title: None,
425            title_count: 0,
426            title_buf: None,
427            meta_description: None,
428            meta_robots: Vec::new(),
429            canonical_raw: None,
430            hreflang_raw: Vec::new(),
431            base_raw: None,
432            h1: Vec::new(),
433            h2: Vec::new(),
434            heading: None,
435            links_raw: Vec::new(),
436            anchor_open: false,
437            images_missing_alt: 0,
438            og: OgTags::default(),
439            jsonld_buf: None,
440            jsonld_overflow: false,
441            jsonld_valid: 0,
442            jsonld_invalid: false,
443            jsonld_too_large: false,
444            mixed_content: 0,
445            words: WordCounter {
446                count: 0,
447                in_word: false,
448                space_pending: false,
449                hashed_any: false,
450                hasher: Xxh3::new(),
451            },
452        }
453    }
454
455    fn on_text(&mut self, raw: &str) {
456        if let Some(buf) = &mut self.jsonld_buf {
457            if !self.jsonld_overflow {
458                if buf.len() + raw.len() > MAX_JSONLD_BYTES {
459                    self.jsonld_overflow = true;
460                    *buf = String::new();
461                } else {
462                    buf.push_str(raw);
463                }
464            }
465            return;
466        }
467        if let Some(buf) = &mut self.title_buf {
468            push_capped(buf, raw, MAX_TITLE_CHARS * 4);
469            return;
470        }
471        if self.skip_depth > 0 {
472            return;
473        }
474        if let Some((_, buf)) = &mut self.heading {
475            push_capped(buf, raw, MAX_HEADING_CHARS * 4);
476        }
477        if self.anchor_open
478            && let Some(link) = self.links_raw.last_mut()
479        {
480            push_capped(&mut link.anchor, raw, MAX_ANCHOR_CHARS * 4);
481        }
482        self.words.feed(raw);
483    }
484
485    /// A block element or `<br>` starts: keep words apart in open heading and link text.
486    fn separate_open_text(&mut self) {
487        if let Some((_, buf)) = &mut self.heading {
488            push_capped(buf, " ", MAX_HEADING_CHARS * 4);
489        }
490        if self.anchor_open
491            && let Some(link) = self.links_raw.last_mut()
492        {
493            push_capped(&mut link.anchor, " ", MAX_ANCHOR_CHARS * 4);
494        }
495    }
496
497    fn on_meta(&mut self, el: &Element<'_, '_>) {
498        let content = || {
499            el.get_attribute("content")
500                .map(|c| clean_attribute(&c, MAX_TITLE_CHARS))
501        };
502        if let Some(name) = el.get_attribute("name") {
503            match name.trim().to_ascii_lowercase().as_str() {
504                "description" if self.meta_description.is_none() => {
505                    self.meta_description = content()
506                }
507                "robots" => self.meta_robots.extend(content()),
508                _ => {}
509            }
510        }
511        if let Some(property) = el.get_attribute("property") {
512            let slot = match property.trim().to_ascii_lowercase().as_str() {
513                "og:title" => &mut self.og.title,
514                "og:description" => &mut self.og.description,
515                "og:image" => &mut self.og.image,
516                _ => return,
517            };
518            if slot.is_none() {
519                *slot = content();
520            }
521        }
522    }
523
524    fn on_link(&mut self, el: &Element<'_, '_>) {
525        let rel = el
526            .get_attribute("rel")
527            .unwrap_or_default()
528            .to_ascii_lowercase();
529        let href = el.get_attribute("href").unwrap_or_default();
530        for r in rel.split_ascii_whitespace() {
531            match r {
532                "canonical" if self.canonical_raw.is_none() => {
533                    self.canonical_raw = Some(href.clone())
534                }
535                "alternate" => {
536                    if let Some(lang) = el.get_attribute("hreflang") {
537                        self.hreflang_raw
538                            .push((lang.trim().to_owned(), href.clone()));
539                    }
540                }
541                "stylesheet" => self.check_mixed(Some(&href)),
542                _ => {}
543            }
544        }
545    }
546
547    fn check_mixed(&mut self, src: Option<&str>) {
548        let is_http = src
549            .map(str::trim_start)
550            .and_then(|s| s.get(..5))
551            .is_some_and(|p| p.eq_ignore_ascii_case("http:"));
552        if self.page_is_https && is_http {
553            self.mixed_content += 1;
554        }
555    }
556
557    fn close_title(&mut self) {
558        if let Some(buf) = self.title_buf.take() {
559            let text = clean(&buf, MAX_TITLE_CHARS);
560            if self.title.is_none() && !text.is_empty() {
561                self.title = Some(text);
562            }
563        }
564    }
565
566    fn close_heading(&mut self) {
567        if let Some((level, buf)) = self.heading.take() {
568            let text = clean(&buf, MAX_HEADING_CHARS);
569            let list = if level == 1 {
570                &mut self.h1
571            } else {
572                &mut self.h2
573            };
574            if !text.is_empty() && list.len() < MAX_HEADINGS {
575                list.push(text);
576            }
577        }
578    }
579
580    fn close_script(&mut self) {
581        self.skip_depth = self.skip_depth.saturating_sub(1);
582        if let Some(buf) = self.jsonld_buf.take() {
583            if std::mem::take(&mut self.jsonld_overflow) {
584                self.jsonld_too_large = true;
585            } else if serde_json::from_str::<serde::de::IgnoredAny>(buf.trim()).is_ok() {
586                self.jsonld_valid = self.jsonld_valid.saturating_add(1);
587            } else {
588                self.jsonld_invalid = true;
589            }
590        }
591    }
592
593    fn close_open_elements(&mut self) {
594        self.close_title();
595        self.close_heading();
596        if self.jsonld_buf.is_some() {
597            self.close_script();
598        }
599    }
600
601    fn build(&mut self, page_url: &Url) -> Extracted {
602        let base = self
603            .base_raw
604            .as_deref()
605            .and_then(|b| normalize(page_url, &unescape_attribute(b)))
606            .unwrap_or_else(|| page_url.clone());
607        let resolve = |raw: &str| normalize(&base, &unescape_attribute(raw));
608
609        let links = self
610            .links_raw
611            .drain(..)
612            .filter_map(|l| {
613                Some(Link {
614                    url: resolve(&l.href)?,
615                    anchor: clean(&l.anchor, MAX_ANCHOR_CHARS),
616                    nofollow: l.nofollow,
617                })
618            })
619            .collect();
620
621        let jsonld = if self.jsonld_invalid {
622            JsonLdStatus::Invalid
623        } else if self.jsonld_too_large {
624            JsonLdStatus::TooLarge
625        } else if self.jsonld_valid > 0 {
626            JsonLdStatus::Valid(self.jsonld_valid)
627        } else {
628            JsonLdStatus::Absent
629        };
630
631        let fields = PageFields {
632            title: self.title.take(),
633            title_count: self.title_count,
634            meta_description: self.meta_description.take(),
635            meta_robots: (!self.meta_robots.is_empty()).then(|| self.meta_robots.join(", ")),
636            x_robots_tag: None,
637            canonical: self.canonical_raw.as_deref().and_then(resolve),
638            hreflang: self
639                .hreflang_raw
640                .drain(..)
641                .filter_map(|(lang, href)| Some((lang, resolve(&href)?)))
642                .collect(),
643            h1: std::mem::take(&mut self.h1),
644            h2: std::mem::take(&mut self.h2),
645            word_count: self.words.count,
646            content_hash: self.words.hasher.digest(),
647            images_missing_alt: self.images_missing_alt,
648            og: std::mem::take(&mut self.og),
649            jsonld,
650            mixed_content: self.mixed_content,
651        };
652        Extracted { fields, links }
653    }
654}
655
656fn push_capped(buf: &mut String, text: &str, max_bytes: usize) {
657    if buf.len() < max_bytes {
658        buf.push_str(text);
659    }
660}
661
662fn unescape(raw: &str) -> String {
663    htmlize::unescape(raw).into_owned()
664}
665
666/// Attribute values follow different rules: a legacy entity without `;` followed by
667/// `=` or a letter stays as written, so `?a=1&region=us` is not turned into `®ion`.
668fn unescape_attribute(raw: &str) -> String {
669    htmlize::unescape_attribute(raw).into_owned()
670}
671
672fn clean_attribute(raw: &str, max_chars: usize) -> String {
673    let decoded = unescape_attribute(raw);
674    let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" ");
675    match collapsed.char_indices().nth(max_chars) {
676        Some((cut, _)) => collapsed[..cut].to_owned(),
677        None => collapsed,
678    }
679}
680
681/// Decodes entities, collapses whitespace and caps the length in characters.
682fn clean(raw: &str, max_chars: usize) -> String {
683    let decoded = unescape(raw);
684    let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" ");
685    match collapsed.char_indices().nth(max_chars) {
686        Some((cut, _)) => collapsed[..cut].to_owned(),
687        None => collapsed,
688    }
689}
690
691fn is_block_or_break(tag: &str) -> bool {
692    matches!(
693        tag,
694        "br" | "p"
695            | "div"
696            | "li"
697            | "ul"
698            | "ol"
699            | "dl"
700            | "dt"
701            | "dd"
702            | "h1"
703            | "h2"
704            | "h3"
705            | "h4"
706            | "h5"
707            | "h6"
708            | "section"
709            | "article"
710            | "header"
711            | "footer"
712            | "nav"
713            | "aside"
714            | "main"
715            | "table"
716            | "tr"
717            | "td"
718            | "th"
719            | "thead"
720            | "tbody"
721            | "blockquote"
722            | "figure"
723            | "figcaption"
724            | "hr"
725            | "pre"
726            | "address"
727            | "details"
728            | "summary"
729            | "form"
730            | "fieldset"
731            | "legend"
732            | "option"
733            | "button"
734    )
735}
736
737/// The HTML spec's encoding pre-scan, simplified: the first `<meta>` in the first
738/// 1024 bytes that declares a charset (via `charset` or `http-equiv` + `content`).
739fn prescan_meta_charset(bytes: &[u8]) -> Option<&'static Encoding> {
740    let s = &bytes[..bytes.len().min(PRESCAN_BYTES)];
741    let mut i = 0;
742    while i < s.len() {
743        if s[i..].starts_with(b"<!--") {
744            i = find(s, i + 4, b"-->")? + 3;
745            continue;
746        }
747        let is_meta = s[i] == b'<'
748            && s.len() > i + 5
749            && s[i + 1..i + 5].eq_ignore_ascii_case(b"meta")
750            && matches!(s[i + 5], b' ' | b'\t' | b'\n' | b'\r' | b'\x0c' | b'/');
751        if is_meta {
752            let (attrs, end) = meta_attributes(s, i + 5);
753            if let Some(enc) = charset_from_meta(&attrs) {
754                return Some(match enc.name() {
755                    "UTF-16LE" | "UTF-16BE" => UTF_8,
756                    "x-user-defined" => encoding_rs::WINDOWS_1252,
757                    _ => enc,
758                });
759            }
760            i = end;
761            continue;
762        }
763        i += 1;
764    }
765    None
766}
767
768fn find(s: &[u8], from: usize, needle: &[u8]) -> Option<usize> {
769    s.get(from..)?
770        .windows(needle.len())
771        .position(|w| w == needle)
772        .map(|p| p + from)
773}
774
775fn meta_attributes(s: &[u8], mut i: usize) -> (Vec<(String, String)>, usize) {
776    let mut attrs = Vec::new();
777    loop {
778        while i < s.len() && (s[i].is_ascii_whitespace() || s[i] == b'/') {
779            i += 1;
780        }
781        if i >= s.len() || s[i] == b'>' {
782            return (attrs, i + 1);
783        }
784        let start = i;
785        while i < s.len() && !s[i].is_ascii_whitespace() && !matches!(s[i], b'=' | b'>' | b'/') {
786            i += 1;
787        }
788        let name = String::from_utf8_lossy(&s[start..i]).to_ascii_lowercase();
789        while i < s.len() && s[i].is_ascii_whitespace() {
790            i += 1;
791        }
792        let mut value = String::new();
793        if i < s.len() && s[i] == b'=' {
794            i += 1;
795            while i < s.len() && s[i].is_ascii_whitespace() {
796                i += 1;
797            }
798            if i < s.len() && matches!(s[i], b'"' | b'\'') {
799                let quote = s[i];
800                let vstart = i + 1;
801                i = vstart;
802                while i < s.len() && s[i] != quote {
803                    i += 1;
804                }
805                value = String::from_utf8_lossy(&s[vstart..i.min(s.len())]).into_owned();
806                i += 1;
807            } else {
808                let vstart = i;
809                while i < s.len() && !s[i].is_ascii_whitespace() && s[i] != b'>' {
810                    i += 1;
811                }
812                value = String::from_utf8_lossy(&s[vstart..i]).into_owned();
813            }
814        }
815        attrs.push((name, value));
816    }
817}
818
819fn charset_from_meta(attrs: &[(String, String)]) -> Option<&'static Encoding> {
820    let get = |name: &str| {
821        attrs
822            .iter()
823            .find(|(n, _)| n == name)
824            .map(|(_, v)| v.as_str())
825    };
826    if let Some(charset) = get("charset") {
827        return Encoding::for_label(charset.trim().as_bytes());
828    }
829    if get("http-equiv").is_some_and(|v| v.trim().eq_ignore_ascii_case("content-type")) {
830        let content = get("content")?;
831        let lower = content.to_ascii_lowercase();
832        let at = lower.find("charset")? + "charset".len();
833        let rest = content[at..].trim_start().strip_prefix('=')?.trim_start();
834        let label: String = rest
835            .trim_start_matches(['"', '\''])
836            .chars()
837            .take_while(|c| !matches!(c, ';' | '"' | '\'' | ' '))
838            .collect();
839        return Encoding::for_label(label.as_bytes());
840    }
841    None
842}