Skip to main content

codoseo_crawler/
preflight.rs

1//! The first step of a crawl: settle the real start address, read robots.txt and the
2//! sitemaps, and decide the stop reasons that don't need a crawl.
3//!
4//! Order: robots.txt of the requested origin, then the start page (one retry after a
5//! 429 or 503), then robots.txt again when the start page moved to another origin.
6//! Every request goes through the limiter.
7
8use std::time::Duration;
9
10use codoseo_core::crawl::{CrawlConfig, RobotsFile, SitemapSummary};
11use codoseo_core::output::StopReason;
12use codoseo_core::url::normalize;
13use tokio::time::Instant;
14use url::Url;
15
16use crate::crawl::CrawlError;
17use crate::fetch::{FetchError, FetchResult, Fetcher};
18use crate::guard::check_url;
19use crate::politeness::{Limiter, retry_after_of};
20use crate::robots::{RobotsRules, fetch_robots};
21use crate::sitemap::discover;
22
23pub const BLOCKED_MSG: &str = "site blocked our crawler";
24pub const LOGIN_MSG: &str = "site requires a login";
25
26/// How long sitemap discovery may take, whatever the crawl's own deadline.
27const SITEMAP_BUDGET: Duration = Duration::from_secs(60);
28
29/// What the crawl learned before fetching its first page.
30#[doc(hidden)]
31pub struct Preflight {
32    /// Scheme, host and port of the settled start address, with path `/`.
33    pub origin: Url,
34    /// The requested start address and what fetching it gave; `None` when robots.txt
35    /// stopped the crawl first.
36    pub start: Option<(Url, Result<FetchResult, FetchError>)>,
37    pub rules: RobotsRules,
38    /// `None` when robots.txt could not be reached.
39    pub robots: Option<RobotsFile>,
40    pub sitemap_urls: Vec<Url>,
41    pub sitemap: SitemapSummary,
42    /// Set when the crawl should not go on.
43    pub stop: Option<StopReason>,
44}
45
46pub(crate) async fn preflight(
47    cfg: &CrawlConfig,
48    fetcher: &Fetcher,
49    limiter: &Limiter,
50    deadline: Instant,
51) -> Result<Preflight, CrawlError> {
52    let requested = normalize(&cfg.start_url, cfg.start_url.as_str())
53        .ok_or_else(|| CrawlError::InvalidStart(cfg.start_url.to_string()))?;
54    check_url(&requested, cfg.address_policy)
55        .map_err(|e| CrawlError::AddressBlocked(e.to_string()))?;
56
57    let (rules, robots) = robots_for(fetcher, limiter, &requested).await?;
58    let stop = robots_stop(&rules, robots.as_ref()).or_else(|| {
59        let path = path_and_query(&requested);
60        (!rules.allowed(&path)).then_some(StopReason::RobotsBlocked)
61    });
62    if stop.is_some() {
63        return Ok(Preflight {
64            origin: origin_of(&requested),
65            start: None,
66            rules,
67            robots,
68            sitemap_urls: Vec::new(),
69            sitemap: SitemapSummary::default(),
70            stop,
71        });
72    }
73
74    let result = fetch_start(fetcher, limiter, &requested).await;
75    if let Err(FetchError::Blocked(msg)) = &result {
76        return Err(CrawlError::AddressBlocked(msg.clone()));
77    }
78    let final_url = match &result {
79        Ok(res) => res.final_url.clone(),
80        Err(_) => requested.clone(),
81    };
82    let origin = origin_of(&final_url);
83
84    let (rules, robots) = if final_url.origin() == requested.origin() {
85        (rules, robots)
86    } else {
87        robots_for(fetcher, limiter, &final_url).await?
88    };
89
90    let stop = match &result {
91        Err(e) => Some(StopReason::Unreachable(e.to_string())),
92        Ok(res) => robots_stop(&rules, robots.as_ref()).or(match res.status {
93            401 => Some(StopReason::Blocked(LOGIN_MSG.to_owned())),
94            403 | 429 | 503 => Some(StopReason::Blocked(BLOCKED_MSG.to_owned())),
95            _ => None,
96        }),
97    };
98    let mut pre = Preflight {
99        origin,
100        start: Some((requested, result)),
101        rules,
102        robots,
103        sitemap_urls: Vec::new(),
104        sitemap: SitemapSummary::default(),
105        stop,
106    };
107    if pre.stop.is_none() {
108        let seeds = sitemap_seeds(&pre.origin, &pre.rules);
109        let deadline = deadline.min(Instant::now() + SITEMAP_BUDGET);
110        let found = discover(
111            fetcher,
112            Some(limiter),
113            &seeds,
114            cfg.limits.max_sitemap_urls,
115            deadline,
116        )
117        .await;
118        pre.sitemap_urls = found.urls;
119        pre.sitemap = found.summary;
120    }
121    Ok(pre)
122}
123
124/// Fetches robots.txt for the site `url` belongs to. A connection-level failure gives
125/// no file and allow-all rules; a refused address is an error.
126async fn robots_for(
127    fetcher: &Fetcher,
128    limiter: &Limiter,
129    url: &Url,
130) -> Result<(RobotsRules, Option<RobotsFile>), CrawlError> {
131    let fetched = {
132        let _permit = limiter.acquire().await;
133        fetch_robots(fetcher, url).await
134    };
135    match fetched {
136        Ok((rules, file)) => Ok((rules, Some(file))),
137        Err(FetchError::Blocked(msg)) => Err(CrawlError::AddressBlocked(msg)),
138        Err(_) => Ok((RobotsRules::from_status(404), None)),
139    }
140}
141
142/// The stop reasons robots.txt decides on its own, in order: 429, 5xx, a full block.
143fn robots_stop(rules: &RobotsRules, file: Option<&RobotsFile>) -> Option<StopReason> {
144    match file.map(|f| f.status) {
145        Some(429) => return Some(StopReason::Blocked(BLOCKED_MSG.to_owned())),
146        Some(status) if status >= 500 => {
147            return Some(StopReason::Unreachable(format!(
148                "robots.txt returned HTTP {status}"
149            )));
150        }
151        _ => {}
152    }
153    rules
154        .blocks_everything()
155        .then_some(StopReason::RobotsBlocked)
156}
157
158/// Fetches the start page, retrying once after a 429 or 503.
159async fn fetch_start(
160    fetcher: &Fetcher,
161    limiter: &Limiter,
162    url: &Url,
163) -> Result<FetchResult, FetchError> {
164    let first = {
165        let _permit = limiter.acquire().await;
166        fetcher.fetch(url).await
167    };
168    let Ok(res) = &first else { return first };
169    if !matches!(res.status, 429 | 503) {
170        return first;
171    }
172    limiter.on_response(res.status, retry_after_of(&res.headers));
173    let _permit = limiter.acquire().await;
174    fetcher.fetch(url).await
175}
176
177/// The `Sitemap:` lines of robots.txt plus `/sitemap.xml`.
178fn sitemap_seeds(origin: &Url, rules: &RobotsRules) -> Vec<Url> {
179    let mut seeds: Vec<Url> = rules
180        .sitemaps()
181        .iter()
182        .filter_map(|s| normalize(origin, s))
183        .collect();
184    if let Some(default) = normalize(origin, "/sitemap.xml") {
185        seeds.push(default);
186    }
187    seeds
188}
189
190/// Scheme, host and port of `url`, with path `/`.
191pub(crate) fn origin_of(url: &Url) -> Url {
192    let mut origin = url.clone();
193    origin.set_path("/");
194    origin.set_query(None);
195    origin.set_fragment(None);
196    let _ = origin.set_username("");
197    let _ = origin.set_password(None);
198    normalize(&origin, origin.as_str()).unwrap_or(origin)
199}
200
201fn path_and_query(url: &Url) -> String {
202    match url.query() {
203        Some(q) => format!("{}?{q}", url.path()),
204        None => url.path().to_owned(),
205    }
206}