Skip to main content

codoseo_core/
page.rs

1//! One crawled URL and the fields pulled out of it.
2
3use serde::{Deserialize, Serialize};
4use url::Url;
5use xxhash_rust::xxh3::Xxh3;
6
7use crate::check::IssueBits;
8
9#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
10#[serde(rename_all = "snake_case")]
11pub enum Indexability {
12    Indexable,
13    Noindex,
14    Canonicalised,
15    Redirected,
16    ClientError,
17    ServerError,
18    BlockedByRobots,
19}
20
21/// Why a fetch produced no response.
22#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
23#[serde(rename_all = "snake_case")]
24pub enum FetchFailure {
25    Timeout,
26    Connect,
27    RedirectLoop,
28    TooManyRedirects,
29    InvalidRedirect,
30    Blocked,
31    Other,
32}
33
34#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
35pub struct OgTags {
36    pub title: Option<String>,
37    pub description: Option<String>,
38    pub image: Option<String>,
39}
40
41#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
42#[serde(rename_all = "snake_case")]
43pub enum JsonLdStatus {
44    #[default]
45    Absent,
46    /// Number of JSON-LD blocks, all of which parse as JSON.
47    Valid(u16),
48    /// At least one block does not parse as JSON.
49    Invalid,
50    /// A block is too big to check (over 1 MB).
51    TooLarge,
52}
53
54#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
55pub struct PageFields {
56    pub title: Option<String>,
57    pub title_count: u8,
58    pub meta_description: Option<String>,
59    pub meta_robots: Option<String>,
60    /// From the `X-Robots-Tag` header; filled by the crawler, not the extractor.
61    pub x_robots_tag: Option<String>,
62    pub canonical: Option<Url>,
63    pub hreflang: Vec<(String, Url)>,
64    pub h1: Vec<String>,
65    pub h2: Vec<String>,
66    pub word_count: u32,
67    pub content_hash: u64,
68    pub images_missing_alt: u32,
69    pub og: OgTags,
70    pub jsonld: JsonLdStatus,
71    pub mixed_content: u32,
72}
73
74#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
75pub struct PageRecord {
76    pub url: Url,
77    pub url_hash: u64,
78    pub status: u16,
79    /// Every redirect hop in order: the status returned and the URL that returned it.
80    pub redirect_chain: Vec<(u16, Url)>,
81    pub response_ms: u32,
82    pub size_bytes: u64,
83    pub content_type: Option<String>,
84    /// Clicks from the homepage; `None` for pages only found in sitemaps.
85    pub depth: Option<u16>,
86    pub in_sitemap: bool,
87    pub indexability: Indexability,
88    pub fields: PageFields,
89    pub inlinks: u32,
90    pub outlinks_internal: u32,
91    pub outlinks_external: u32,
92    pub issues: IssueBits,
93    /// Hash of the fields change detection compares.
94    pub key_hash: u64,
95    /// Where a redirecting URL ends up.
96    #[serde(default)]
97    pub redirect_target: Option<Url>,
98    /// Internal links on the page marked `rel=nofollow`.
99    #[serde(default)]
100    pub outlinks_nofollow: u32,
101    /// Set when there was no response (`status` is 0).
102    #[serde(default)]
103    pub error: Option<FetchFailure>,
104}
105
106impl PageRecord {
107    /// Hash of the fields change detection compares: status, title, description, first H1,
108    /// canonical, both robots directives, indexability, redirect target and sitemap
109    /// membership. Timings and counts are left out so they don't read as changes.
110    pub fn compute_key_hash(&self) -> u64 {
111        let mut h = Xxh3::new();
112        h.update(&self.status.to_le_bytes());
113        hash_str(&mut h, self.fields.title.as_deref());
114        hash_str(&mut h, self.fields.meta_description.as_deref());
115        hash_str(&mut h, self.fields.h1.first().map(String::as_str));
116        hash_str(&mut h, self.fields.canonical.as_ref().map(Url::as_str));
117        hash_str(&mut h, self.fields.meta_robots.as_deref());
118        hash_str(&mut h, self.fields.x_robots_tag.as_deref());
119        h.update(&[indexability_code(self.indexability)]);
120        hash_str(&mut h, self.redirect_target.as_ref().map(Url::as_str));
121        h.update(&[u8::from(self.in_sitemap)]);
122        h.digest()
123    }
124
125    /// A successful response that can be read as an HTML page.
126    pub fn is_html_ok(&self) -> bool {
127        if !(200..300).contains(&self.status) || self.error.is_some() {
128            return false;
129        }
130        match &self.content_type {
131            None => true,
132            Some(ct) => {
133                let ct = ct.to_ascii_lowercase();
134                ct.contains("text/html") || ct.contains("application/xhtml+xml")
135            }
136        }
137    }
138}
139
140/// Length-prefixed, so neighbouring fields can't blur into each other.
141fn hash_str(h: &mut Xxh3, value: Option<&str>) {
142    match value {
143        None => h.update(&[0]),
144        Some(s) => {
145            h.update(&[1]);
146            h.update(&(s.len() as u64).to_le_bytes());
147            h.update(s.as_bytes())
148        }
149    };
150}
151
152/// Fixed codes, so the hash doesn't change if variants are reordered.
153fn indexability_code(i: Indexability) -> u8 {
154    match i {
155        Indexability::Indexable => 0,
156        Indexability::Noindex => 1,
157        Indexability::Canonicalised => 2,
158        Indexability::Redirected => 3,
159        Indexability::ClientError => 4,
160        Indexability::ServerError => 5,
161        Indexability::BlockedByRobots => 6,
162    }
163}
164
165/// Bots whose scoped robots directives (`googlebot: noindex`) apply to us.
166const OUR_AGENTS: [&str; 2] = ["codoseobot", "googlebot"];
167
168/// Directives that carry a value after a colon; these are not agent prefixes.
169const VALUE_DIRECTIVES: [&str; 4] = [
170    "unavailable_after",
171    "max-snippet",
172    "max-image-preview",
173    "max-video-preview",
174];
175
176/// True when a robots directive list (meta robots or `X-Robots-Tag`) contains one of `words`
177/// for us. Follows Google's rules: an `agent:` prefix scopes every following comma-separated
178/// directive until the next prefix, directives before any prefix apply to all bots, and only
179/// `codoseobot` and `googlebot` scopes apply to us.
180fn has_directive(value: Option<&str>, words: [&str; 2]) -> bool {
181    let Some(value) = value else { return false };
182    let mut applies = true;
183    for token in value.to_ascii_lowercase().split(',') {
184        let directive = match token.split_once(':') {
185            Some((name, rest)) if !VALUE_DIRECTIVES.contains(&name.trim()) => {
186                applies = OUR_AGENTS.contains(&name.trim());
187                rest
188            }
189            _ => token,
190        };
191        if applies && words.contains(&directive.trim()) {
192            return true;
193        }
194    }
195    false
196}
197
198/// `noindex` or `none` in the meta robots tag or the `X-Robots-Tag` header.
199pub fn is_noindex(meta_robots: Option<&str>, x_robots_tag: Option<&str>) -> bool {
200    let words = ["noindex", "none"];
201    has_directive(meta_robots, words) || has_directive(x_robots_tag, words)
202}
203
204/// `nofollow` or `none` in the meta robots tag or the `X-Robots-Tag` header.
205pub fn is_nofollow(meta_robots: Option<&str>, x_robots_tag: Option<&str>) -> bool {
206    let words = ["nofollow", "none"];
207    has_directive(meta_robots, words) || has_directive(x_robots_tag, words)
208}
209
210/// Whether search engines would index this URL. Status 0 (no response) counts as a server error.
211pub fn indexability(
212    url: &Url,
213    status: u16,
214    fields: &PageFields,
215    robots_blocked: bool,
216) -> Indexability {
217    if robots_blocked {
218        return Indexability::BlockedByRobots;
219    }
220    match status {
221        0..=199 | 500.. => Indexability::ServerError,
222        300..=399 => Indexability::Redirected,
223        400..=499 => Indexability::ClientError,
224        200..=299 => {
225            if is_noindex(
226                fields.meta_robots.as_deref(),
227                fields.x_robots_tag.as_deref(),
228            ) {
229                Indexability::Noindex
230            } else if fields.canonical.as_ref().is_some_and(|c| c != url) {
231                Indexability::Canonicalised
232            } else {
233                Indexability::Indexable
234            }
235        }
236    }
237}