1use serde::{Deserialize, Serialize};
4use url::Url;
5use xxhash_rust::xxh3::Xxh3;
6
7use crate::check::IssueBits;
8
9#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
10#[serde(rename_all = "snake_case")]
11pub enum Indexability {
12 Indexable,
13 Noindex,
14 Canonicalised,
15 Redirected,
16 ClientError,
17 ServerError,
18 BlockedByRobots,
19}
20
21#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
23#[serde(rename_all = "snake_case")]
24pub enum FetchFailure {
25 Timeout,
26 Connect,
27 RedirectLoop,
28 TooManyRedirects,
29 InvalidRedirect,
30 Blocked,
31 Other,
32}
33
34#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
35pub struct OgTags {
36 pub title: Option<String>,
37 pub description: Option<String>,
38 pub image: Option<String>,
39}
40
41#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
42#[serde(rename_all = "snake_case")]
43pub enum JsonLdStatus {
44 #[default]
45 Absent,
46 Valid(u16),
48 Invalid,
50 TooLarge,
52}
53
54#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
55pub struct PageFields {
56 pub title: Option<String>,
57 pub title_count: u8,
58 pub meta_description: Option<String>,
59 pub meta_robots: Option<String>,
60 pub x_robots_tag: Option<String>,
62 pub canonical: Option<Url>,
63 pub hreflang: Vec<(String, Url)>,
64 pub h1: Vec<String>,
65 pub h2: Vec<String>,
66 pub word_count: u32,
67 pub content_hash: u64,
68 pub images_missing_alt: u32,
69 pub og: OgTags,
70 pub jsonld: JsonLdStatus,
71 pub mixed_content: u32,
72}
73
74#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
75pub struct PageRecord {
76 pub url: Url,
77 pub url_hash: u64,
78 pub status: u16,
79 pub redirect_chain: Vec<(u16, Url)>,
81 pub response_ms: u32,
82 pub size_bytes: u64,
83 pub content_type: Option<String>,
84 pub depth: Option<u16>,
86 pub in_sitemap: bool,
87 pub indexability: Indexability,
88 pub fields: PageFields,
89 pub inlinks: u32,
90 pub outlinks_internal: u32,
91 pub outlinks_external: u32,
92 pub issues: IssueBits,
93 pub key_hash: u64,
95 #[serde(default)]
97 pub redirect_target: Option<Url>,
98 #[serde(default)]
100 pub outlinks_nofollow: u32,
101 #[serde(default)]
103 pub error: Option<FetchFailure>,
104}
105
106impl PageRecord {
107 pub fn compute_key_hash(&self) -> u64 {
111 let mut h = Xxh3::new();
112 h.update(&self.status.to_le_bytes());
113 hash_str(&mut h, self.fields.title.as_deref());
114 hash_str(&mut h, self.fields.meta_description.as_deref());
115 hash_str(&mut h, self.fields.h1.first().map(String::as_str));
116 hash_str(&mut h, self.fields.canonical.as_ref().map(Url::as_str));
117 hash_str(&mut h, self.fields.meta_robots.as_deref());
118 hash_str(&mut h, self.fields.x_robots_tag.as_deref());
119 h.update(&[indexability_code(self.indexability)]);
120 hash_str(&mut h, self.redirect_target.as_ref().map(Url::as_str));
121 h.update(&[u8::from(self.in_sitemap)]);
122 h.digest()
123 }
124
125 pub fn is_html_ok(&self) -> bool {
127 if !(200..300).contains(&self.status) || self.error.is_some() {
128 return false;
129 }
130 match &self.content_type {
131 None => true,
132 Some(ct) => {
133 let ct = ct.to_ascii_lowercase();
134 ct.contains("text/html") || ct.contains("application/xhtml+xml")
135 }
136 }
137 }
138}
139
140fn hash_str(h: &mut Xxh3, value: Option<&str>) {
142 match value {
143 None => h.update(&[0]),
144 Some(s) => {
145 h.update(&[1]);
146 h.update(&(s.len() as u64).to_le_bytes());
147 h.update(s.as_bytes())
148 }
149 };
150}
151
152fn indexability_code(i: Indexability) -> u8 {
154 match i {
155 Indexability::Indexable => 0,
156 Indexability::Noindex => 1,
157 Indexability::Canonicalised => 2,
158 Indexability::Redirected => 3,
159 Indexability::ClientError => 4,
160 Indexability::ServerError => 5,
161 Indexability::BlockedByRobots => 6,
162 }
163}
164
165const OUR_AGENTS: [&str; 2] = ["codoseobot", "googlebot"];
167
168const VALUE_DIRECTIVES: [&str; 4] = [
170 "unavailable_after",
171 "max-snippet",
172 "max-image-preview",
173 "max-video-preview",
174];
175
176fn has_directive(value: Option<&str>, words: [&str; 2]) -> bool {
181 let Some(value) = value else { return false };
182 let mut applies = true;
183 for token in value.to_ascii_lowercase().split(',') {
184 let directive = match token.split_once(':') {
185 Some((name, rest)) if !VALUE_DIRECTIVES.contains(&name.trim()) => {
186 applies = OUR_AGENTS.contains(&name.trim());
187 rest
188 }
189 _ => token,
190 };
191 if applies && words.contains(&directive.trim()) {
192 return true;
193 }
194 }
195 false
196}
197
198pub fn is_noindex(meta_robots: Option<&str>, x_robots_tag: Option<&str>) -> bool {
200 let words = ["noindex", "none"];
201 has_directive(meta_robots, words) || has_directive(x_robots_tag, words)
202}
203
204pub fn is_nofollow(meta_robots: Option<&str>, x_robots_tag: Option<&str>) -> bool {
206 let words = ["nofollow", "none"];
207 has_directive(meta_robots, words) || has_directive(x_robots_tag, words)
208}
209
210pub fn indexability(
212 url: &Url,
213 status: u16,
214 fields: &PageFields,
215 robots_blocked: bool,
216) -> Indexability {
217 if robots_blocked {
218 return Indexability::BlockedByRobots;
219 }
220 match status {
221 0..=199 | 500.. => Indexability::ServerError,
222 300..=399 => Indexability::Redirected,
223 400..=499 => Indexability::ClientError,
224 200..=299 => {
225 if is_noindex(
226 fields.meta_robots.as_deref(),
227 fields.x_robots_tag.as_deref(),
228 ) {
229 Indexability::Noindex
230 } else if fields.canonical.as_ref().is_some_and(|c| c != url) {
231 Indexability::Canonicalised
232 } else {
233 Indexability::Indexable
234 }
235 }
236 }
237}