1use std::time::Duration;
9
10use codoseo_core::crawl::{CrawlConfig, RobotsFile, SitemapSummary};
11use codoseo_core::output::StopReason;
12use codoseo_core::url::normalize;
13use tokio::time::Instant;
14use url::Url;
15
16use crate::crawl::CrawlError;
17use crate::fetch::{FetchError, FetchResult, Fetcher};
18use crate::guard::check_url;
19use crate::politeness::{Limiter, retry_after_of};
20use crate::robots::{RobotsRules, fetch_robots};
21use crate::sitemap::discover;
22
23pub const BLOCKED_MSG: &str = "site blocked our crawler";
24pub const LOGIN_MSG: &str = "site requires a login";
25
26const SITEMAP_BUDGET: Duration = Duration::from_secs(60);
28
29#[doc(hidden)]
31pub struct Preflight {
32 pub origin: Url,
34 pub start: Option<(Url, Result<FetchResult, FetchError>)>,
37 pub rules: RobotsRules,
38 pub robots: Option<RobotsFile>,
40 pub sitemap_urls: Vec<Url>,
41 pub sitemap: SitemapSummary,
42 pub stop: Option<StopReason>,
44}
45
46pub(crate) async fn preflight(
47 cfg: &CrawlConfig,
48 fetcher: &Fetcher,
49 limiter: &Limiter,
50 deadline: Instant,
51) -> Result<Preflight, CrawlError> {
52 let requested = normalize(&cfg.start_url, cfg.start_url.as_str())
53 .ok_or_else(|| CrawlError::InvalidStart(cfg.start_url.to_string()))?;
54 check_url(&requested, cfg.address_policy)
55 .map_err(|e| CrawlError::AddressBlocked(e.to_string()))?;
56
57 let (rules, robots) = robots_for(fetcher, limiter, &requested).await?;
58 let stop = robots_stop(&rules, robots.as_ref()).or_else(|| {
59 let path = path_and_query(&requested);
60 (!rules.allowed(&path)).then_some(StopReason::RobotsBlocked)
61 });
62 if stop.is_some() {
63 return Ok(Preflight {
64 origin: origin_of(&requested),
65 start: None,
66 rules,
67 robots,
68 sitemap_urls: Vec::new(),
69 sitemap: SitemapSummary::default(),
70 stop,
71 });
72 }
73
74 let result = fetch_start(fetcher, limiter, &requested).await;
75 if let Err(FetchError::Blocked(msg)) = &result {
76 return Err(CrawlError::AddressBlocked(msg.clone()));
77 }
78 let final_url = match &result {
79 Ok(res) => res.final_url.clone(),
80 Err(_) => requested.clone(),
81 };
82 let origin = origin_of(&final_url);
83
84 let (rules, robots) = if final_url.origin() == requested.origin() {
85 (rules, robots)
86 } else {
87 robots_for(fetcher, limiter, &final_url).await?
88 };
89
90 let stop = match &result {
91 Err(e) => Some(StopReason::Unreachable(e.to_string())),
92 Ok(res) => robots_stop(&rules, robots.as_ref()).or(match res.status {
93 401 => Some(StopReason::Blocked(LOGIN_MSG.to_owned())),
94 403 | 429 | 503 => Some(StopReason::Blocked(BLOCKED_MSG.to_owned())),
95 _ => None,
96 }),
97 };
98 let mut pre = Preflight {
99 origin,
100 start: Some((requested, result)),
101 rules,
102 robots,
103 sitemap_urls: Vec::new(),
104 sitemap: SitemapSummary::default(),
105 stop,
106 };
107 if pre.stop.is_none() {
108 let seeds = sitemap_seeds(&pre.origin, &pre.rules);
109 let deadline = deadline.min(Instant::now() + SITEMAP_BUDGET);
110 let found = discover(
111 fetcher,
112 Some(limiter),
113 &seeds,
114 cfg.limits.max_sitemap_urls,
115 deadline,
116 )
117 .await;
118 pre.sitemap_urls = found.urls;
119 pre.sitemap = found.summary;
120 }
121 Ok(pre)
122}
123
124async fn robots_for(
127 fetcher: &Fetcher,
128 limiter: &Limiter,
129 url: &Url,
130) -> Result<(RobotsRules, Option<RobotsFile>), CrawlError> {
131 let fetched = {
132 let _permit = limiter.acquire().await;
133 fetch_robots(fetcher, url).await
134 };
135 match fetched {
136 Ok((rules, file)) => Ok((rules, Some(file))),
137 Err(FetchError::Blocked(msg)) => Err(CrawlError::AddressBlocked(msg)),
138 Err(_) => Ok((RobotsRules::from_status(404), None)),
139 }
140}
141
142fn robots_stop(rules: &RobotsRules, file: Option<&RobotsFile>) -> Option<StopReason> {
144 match file.map(|f| f.status) {
145 Some(429) => return Some(StopReason::Blocked(BLOCKED_MSG.to_owned())),
146 Some(status) if status >= 500 => {
147 return Some(StopReason::Unreachable(format!(
148 "robots.txt returned HTTP {status}"
149 )));
150 }
151 _ => {}
152 }
153 rules
154 .blocks_everything()
155 .then_some(StopReason::RobotsBlocked)
156}
157
158async fn fetch_start(
160 fetcher: &Fetcher,
161 limiter: &Limiter,
162 url: &Url,
163) -> Result<FetchResult, FetchError> {
164 let first = {
165 let _permit = limiter.acquire().await;
166 fetcher.fetch(url).await
167 };
168 let Ok(res) = &first else { return first };
169 if !matches!(res.status, 429 | 503) {
170 return first;
171 }
172 limiter.on_response(res.status, retry_after_of(&res.headers));
173 let _permit = limiter.acquire().await;
174 fetcher.fetch(url).await
175}
176
177fn sitemap_seeds(origin: &Url, rules: &RobotsRules) -> Vec<Url> {
179 let mut seeds: Vec<Url> = rules
180 .sitemaps()
181 .iter()
182 .filter_map(|s| normalize(origin, s))
183 .collect();
184 if let Some(default) = normalize(origin, "/sitemap.xml") {
185 seeds.push(default);
186 }
187 seeds
188}
189
190pub(crate) fn origin_of(url: &Url) -> Url {
192 let mut origin = url.clone();
193 origin.set_path("/");
194 origin.set_query(None);
195 origin.set_fragment(None);
196 let _ = origin.set_username("");
197 let _ = origin.set_password(None);
198 normalize(&origin, origin.as_str()).unwrap_or(origin)
199}
200
201fn path_and_query(url: &Url) -> String {
202 match url.query() {
203 Some(q) => format!("{}?{q}", url.path()),
204 None => url.path().to_owned(),
205 }
206}