use std::time::Duration;
use codoseo_core::crawl::{CrawlConfig, RobotsFile, SitemapSummary};
use codoseo_core::output::StopReason;
use codoseo_core::url::normalize;
use tokio::time::Instant;
use url::Url;
use crate::crawl::CrawlError;
use crate::fetch::{FetchError, FetchResult, Fetcher};
use crate::guard::check_url;
use crate::politeness::{Limiter, retry_after_of};
use crate::robots::{RobotsRules, fetch_robots};
use crate::sitemap::discover;
pub const BLOCKED_MSG: &str = "site blocked our crawler";
pub const LOGIN_MSG: &str = "site requires a login";
const SITEMAP_BUDGET: Duration = Duration::from_secs(60);
#[doc(hidden)]
pub struct Preflight {
pub origin: Url,
pub start: Option<(Url, Result<FetchResult, FetchError>)>,
pub rules: RobotsRules,
pub robots: Option<RobotsFile>,
pub sitemap_urls: Vec<Url>,
pub sitemap: SitemapSummary,
pub stop: Option<StopReason>,
}
pub(crate) async fn preflight(
cfg: &CrawlConfig,
fetcher: &Fetcher,
limiter: &Limiter,
deadline: Instant,
) -> Result<Preflight, CrawlError> {
let requested = normalize(&cfg.start_url, cfg.start_url.as_str())
.ok_or_else(|| CrawlError::InvalidStart(cfg.start_url.to_string()))?;
check_url(&requested, cfg.address_policy)
.map_err(|e| CrawlError::AddressBlocked(e.to_string()))?;
let (rules, robots) = robots_for(fetcher, limiter, &requested).await?;
let stop = robots_stop(&rules, robots.as_ref()).or_else(|| {
let path = path_and_query(&requested);
(!rules.allowed(&path)).then_some(StopReason::RobotsBlocked)
});
if stop.is_some() {
return Ok(Preflight {
origin: origin_of(&requested),
start: None,
rules,
robots,
sitemap_urls: Vec::new(),
sitemap: SitemapSummary::default(),
stop,
});
}
let result = fetch_start(fetcher, limiter, &requested).await;
if let Err(FetchError::Blocked(msg)) = &result {
return Err(CrawlError::AddressBlocked(msg.clone()));
}
let final_url = match &result {
Ok(res) => res.final_url.clone(),
Err(_) => requested.clone(),
};
let origin = origin_of(&final_url);
let (rules, robots) = if final_url.origin() == requested.origin() {
(rules, robots)
} else {
robots_for(fetcher, limiter, &final_url).await?
};
let stop = match &result {
Err(e) => Some(StopReason::Unreachable(e.to_string())),
Ok(res) => robots_stop(&rules, robots.as_ref()).or(match res.status {
401 => Some(StopReason::Blocked(LOGIN_MSG.to_owned())),
403 | 429 | 503 => Some(StopReason::Blocked(BLOCKED_MSG.to_owned())),
_ => None,
}),
};
let mut pre = Preflight {
origin,
start: Some((requested, result)),
rules,
robots,
sitemap_urls: Vec::new(),
sitemap: SitemapSummary::default(),
stop,
};
if pre.stop.is_none() {
let seeds = sitemap_seeds(&pre.origin, &pre.rules);
let deadline = deadline.min(Instant::now() + SITEMAP_BUDGET);
let found = discover(
fetcher,
Some(limiter),
&seeds,
cfg.limits.max_sitemap_urls,
deadline,
)
.await;
pre.sitemap_urls = found.urls;
pre.sitemap = found.summary;
}
Ok(pre)
}
async fn robots_for(
fetcher: &Fetcher,
limiter: &Limiter,
url: &Url,
) -> Result<(RobotsRules, Option<RobotsFile>), CrawlError> {
let fetched = {
let _permit = limiter.acquire().await;
fetch_robots(fetcher, url).await
};
match fetched {
Ok((rules, file)) => Ok((rules, Some(file))),
Err(FetchError::Blocked(msg)) => Err(CrawlError::AddressBlocked(msg)),
Err(_) => Ok((RobotsRules::from_status(404), None)),
}
}
fn robots_stop(rules: &RobotsRules, file: Option<&RobotsFile>) -> Option<StopReason> {
match file.map(|f| f.status) {
Some(429) => return Some(StopReason::Blocked(BLOCKED_MSG.to_owned())),
Some(status) if status >= 500 => {
return Some(StopReason::Unreachable(format!(
"robots.txt returned HTTP {status}"
)));
}
_ => {}
}
rules
.blocks_everything()
.then_some(StopReason::RobotsBlocked)
}
async fn fetch_start(
fetcher: &Fetcher,
limiter: &Limiter,
url: &Url,
) -> Result<FetchResult, FetchError> {
let first = {
let _permit = limiter.acquire().await;
fetcher.fetch(url).await
};
let Ok(res) = &first else { return first };
if !matches!(res.status, 429 | 503) {
return first;
}
limiter.on_response(res.status, retry_after_of(&res.headers));
let _permit = limiter.acquire().await;
fetcher.fetch(url).await
}
fn sitemap_seeds(origin: &Url, rules: &RobotsRules) -> Vec<Url> {
let mut seeds: Vec<Url> = rules
.sitemaps()
.iter()
.filter_map(|s| normalize(origin, s))
.collect();
if let Some(default) = normalize(origin, "/sitemap.xml") {
seeds.push(default);
}
seeds
}
pub(crate) fn origin_of(url: &Url) -> Url {
let mut origin = url.clone();
origin.set_path("/");
origin.set_query(None);
origin.set_fragment(None);
let _ = origin.set_username("");
let _ = origin.set_password(None);
normalize(&origin, origin.as_str()).unwrap_or(origin)
}
fn path_and_query(url: &Url) -> String {
match url.query() {
Some(q) => format!("{}?{q}", url.path()),
None => url.path().to_owned(),
}
}