use serde::{Deserialize, Serialize};
use url::Url;
use xxhash_rust::xxh3::Xxh3;
use crate::check::IssueBits;
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum Indexability {
Indexable,
Noindex,
Canonicalised,
Redirected,
ClientError,
ServerError,
BlockedByRobots,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum FetchFailure {
Timeout,
Connect,
RedirectLoop,
TooManyRedirects,
InvalidRedirect,
Blocked,
Other,
}
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct OgTags {
pub title: Option<String>,
pub description: Option<String>,
pub image: Option<String>,
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum JsonLdStatus {
#[default]
Absent,
Valid(u16),
Invalid,
TooLarge,
}
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct PageFields {
pub title: Option<String>,
pub title_count: u8,
pub meta_description: Option<String>,
pub meta_robots: Option<String>,
pub x_robots_tag: Option<String>,
pub canonical: Option<Url>,
pub hreflang: Vec<(String, Url)>,
pub h1: Vec<String>,
pub h2: Vec<String>,
pub word_count: u32,
pub content_hash: u64,
pub images_missing_alt: u32,
pub og: OgTags,
pub jsonld: JsonLdStatus,
pub mixed_content: u32,
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct PageRecord {
pub url: Url,
pub url_hash: u64,
pub status: u16,
pub redirect_chain: Vec<(u16, Url)>,
pub response_ms: u32,
pub size_bytes: u64,
pub content_type: Option<String>,
pub depth: Option<u16>,
pub in_sitemap: bool,
pub indexability: Indexability,
pub fields: PageFields,
pub inlinks: u32,
pub outlinks_internal: u32,
pub outlinks_external: u32,
pub issues: IssueBits,
pub key_hash: u64,
#[serde(default)]
pub redirect_target: Option<Url>,
#[serde(default)]
pub outlinks_nofollow: u32,
#[serde(default)]
pub error: Option<FetchFailure>,
}
impl PageRecord {
pub fn compute_key_hash(&self) -> u64 {
let mut h = Xxh3::new();
h.update(&self.status.to_le_bytes());
hash_str(&mut h, self.fields.title.as_deref());
hash_str(&mut h, self.fields.meta_description.as_deref());
hash_str(&mut h, self.fields.h1.first().map(String::as_str));
hash_str(&mut h, self.fields.canonical.as_ref().map(Url::as_str));
hash_str(&mut h, self.fields.meta_robots.as_deref());
hash_str(&mut h, self.fields.x_robots_tag.as_deref());
h.update(&[indexability_code(self.indexability)]);
hash_str(&mut h, self.redirect_target.as_ref().map(Url::as_str));
h.update(&[u8::from(self.in_sitemap)]);
h.digest()
}
pub fn is_html_ok(&self) -> bool {
if !(200..300).contains(&self.status) || self.error.is_some() {
return false;
}
match &self.content_type {
None => true,
Some(ct) => {
let ct = ct.to_ascii_lowercase();
ct.contains("text/html") || ct.contains("application/xhtml+xml")
}
}
}
}
fn hash_str(h: &mut Xxh3, value: Option<&str>) {
match value {
None => h.update(&[0]),
Some(s) => {
h.update(&[1]);
h.update(&(s.len() as u64).to_le_bytes());
h.update(s.as_bytes())
}
};
}
fn indexability_code(i: Indexability) -> u8 {
match i {
Indexability::Indexable => 0,
Indexability::Noindex => 1,
Indexability::Canonicalised => 2,
Indexability::Redirected => 3,
Indexability::ClientError => 4,
Indexability::ServerError => 5,
Indexability::BlockedByRobots => 6,
}
}
const OUR_AGENTS: [&str; 2] = ["codoseobot", "googlebot"];
const VALUE_DIRECTIVES: [&str; 4] = [
"unavailable_after",
"max-snippet",
"max-image-preview",
"max-video-preview",
];
fn has_directive(value: Option<&str>, words: [&str; 2]) -> bool {
let Some(value) = value else { return false };
let mut applies = true;
for token in value.to_ascii_lowercase().split(',') {
let directive = match token.split_once(':') {
Some((name, rest)) if !VALUE_DIRECTIVES.contains(&name.trim()) => {
applies = OUR_AGENTS.contains(&name.trim());
rest
}
_ => token,
};
if applies && words.contains(&directive.trim()) {
return true;
}
}
false
}
pub fn is_noindex(meta_robots: Option<&str>, x_robots_tag: Option<&str>) -> bool {
let words = ["noindex", "none"];
has_directive(meta_robots, words) || has_directive(x_robots_tag, words)
}
pub fn is_nofollow(meta_robots: Option<&str>, x_robots_tag: Option<&str>) -> bool {
let words = ["nofollow", "none"];
has_directive(meta_robots, words) || has_directive(x_robots_tag, words)
}
pub fn indexability(
url: &Url,
status: u16,
fields: &PageFields,
robots_blocked: bool,
) -> Indexability {
if robots_blocked {
return Indexability::BlockedByRobots;
}
match status {
0..=199 | 500.. => Indexability::ServerError,
300..=399 => Indexability::Redirected,
400..=499 => Indexability::ClientError,
200..=299 => {
if is_noindex(
fields.meta_robots.as_deref(),
fields.x_robots_tag.as_deref(),
) {
Indexability::Noindex
} else if fields.canonical.as_ref().is_some_and(|c| c != url) {
Indexability::Canonicalised
} else {
Indexability::Indexable
}
}
}
}