Skip to main content

codoseo_crawler/
robots.rs

1//! robots.txt rules for CodoSEObot, following Google's handling of groups, status
2//! codes and size.
3//!
4//! Matching is linear: rules are `*` wildcards with an optional trailing `$`, matched
5//! greedily against the path and query with plain substring search. A hostile file
6//! can't cost more than [`MAX_RULES`] rules of at most [`MAX_PATTERN_LEN`] bytes each;
7//! anything beyond that is skipped, not the whole file.
8
9use std::time::Duration;
10
11use codoseo_core::crawl::{Politeness, RobotsFile};
12use url::Url;
13use xxhash_rust::xxh3::xxh3_64;
14
15use crate::fetch::{FetchError, Fetcher};
16
17/// The product token sites use to address us in robots.txt.
18pub const ROBOTS_AGENT: &str = "CodoSEObot";
19
20/// Google ignores everything after the first 500 KiB.
21const MAX_ROBOTS_BYTES: usize = 500 * 1024;
22pub const MAX_RULES: usize = 2_000;
23pub const MAX_PATTERN_LEN: usize = 1_024;
24
25struct Rule {
26    /// Literal pieces between `*`s; the first must match at the start of the path.
27    parts: Vec<String>,
28    anchored_end: bool,
29    /// Pattern length as written, for longest-match precedence.
30    len: usize,
31    allow: bool,
32}
33
34impl Rule {
35    fn new(pattern: &str, allow: bool) -> Option<Rule> {
36        if pattern.is_empty() || pattern.len() > MAX_PATTERN_LEN {
37            return None;
38        }
39        let pattern = if pattern.starts_with('/') || pattern.starts_with('*') {
40            pattern.to_owned()
41        } else {
42            format!("/{pattern}")
43        };
44        let (body, anchored_end) = match pattern.strip_suffix('$') {
45            Some(body) => (body, true),
46            None => (pattern.as_str(), false),
47        };
48        Some(Rule {
49            parts: body.split('*').map(str::to_owned).collect(),
50            anchored_end,
51            len: pattern.len(),
52            allow,
53        })
54    }
55
56    fn matches(&self, path: &str) -> bool {
57        let (first, rest) = self
58            .parts
59            .split_first()
60            .expect("split always yields one part");
61        let Some(mut remaining) = path.strip_prefix(first.as_str()) else {
62            return false;
63        };
64        let Some((last, middle)) = rest.split_last() else {
65            return !self.anchored_end || remaining.is_empty();
66        };
67        for part in middle {
68            match remaining.find(part.as_str()) {
69                Some(at) => remaining = &remaining[at + part.len()..],
70                None => return false,
71            }
72        }
73        if self.anchored_end {
74            remaining.len() >= last.len() && remaining.ends_with(last.as_str())
75        } else {
76            remaining.contains(last.as_str())
77        }
78    }
79}
80
81enum Kind {
82    AllowAll,
83    BlockAll,
84    Rules(Vec<Rule>),
85}
86
87pub struct RobotsRules {
88    kind: Kind,
89    delay: Option<Duration>,
90    sitemaps: Vec<String>,
91}
92
93#[derive(Default)]
94struct Group {
95    agents: Vec<String>,
96    rules: Vec<(String, bool)>,
97    delay: Option<String>,
98}
99
100/// `CodoSEObot/0.1 (+https://…)` → `codoseobot`; `*` stays `*`.
101fn product_token(value: &str) -> String {
102    let value = value.trim();
103    if value.starts_with('*') {
104        return "*".to_owned();
105    }
106    value
107        .chars()
108        .take_while(|c| c.is_ascii_alphanumeric() || *c == '-' || *c == '_')
109        .collect::<String>()
110        .to_ascii_lowercase()
111}
112
113impl RobotsRules {
114    /// Parses a robots.txt body for `agent`. Lines it doesn't understand are ignored.
115    pub fn parse(body: &[u8], agent: &str) -> RobotsRules {
116        let text = String::from_utf8_lossy(&body[..body.len().min(MAX_ROBOTS_BYTES)]);
117        let ours = product_token(agent);
118        let mut groups: Vec<Group> = Vec::new();
119        let mut sitemaps = Vec::new();
120        let mut reading_agents = false;
121
122        for line in text.lines() {
123            let line = line.split('#').next().unwrap_or("");
124            let Some((key, value)) = line.split_once(':') else {
125                continue;
126            };
127            let value = value.trim();
128            match key.trim().to_ascii_lowercase().as_str() {
129                "user-agent" => {
130                    if !reading_agents {
131                        groups.push(Group::default());
132                    }
133                    reading_agents = true;
134                    if let Some(group) = groups.last_mut() {
135                        group.agents.push(product_token(value));
136                    }
137                }
138                "sitemap" => sitemaps.push(value.to_owned()),
139                key => {
140                    reading_agents = false;
141                    let Some(group) = groups.last_mut() else {
142                        continue;
143                    };
144                    match key {
145                        "allow" | "disallow" if !value.is_empty() => {
146                            group.rules.push((value.to_owned(), key == "allow"));
147                        }
148                        "crawl-delay" if group.delay.is_none() => {
149                            group.delay = Some(value.to_owned())
150                        }
151                        _ => {}
152                    }
153                }
154            }
155        }
156
157        let named: Vec<&Group> = groups.iter().filter(|g| g.agents.contains(&ours)).collect();
158        let chosen = if named.is_empty() {
159            groups
160                .iter()
161                .filter(|g| g.agents.iter().any(|a| a == "*"))
162                .collect()
163        } else {
164            named
165        };
166        let rules: Vec<Rule> = chosen
167            .iter()
168            .flat_map(|g| g.rules.iter())
169            .filter_map(|(pattern, allow)| Rule::new(pattern, *allow))
170            .take(MAX_RULES)
171            .collect();
172        let delay = chosen
173            .iter()
174            .find_map(|g| g.delay.as_deref())
175            .and_then(parse_delay);
176        RobotsRules {
177            kind: Kind::Rules(rules),
178            delay,
179            sitemaps,
180        }
181    }
182
183    /// Rules for a robots.txt that didn't return 2xx. Like Google: a 4xx (except 429)
184    /// means there are no rules; 429 and 5xx mean the whole site is off limits for now.
185    pub fn from_status(status: u16) -> RobotsRules {
186        let kind = if (400..500).contains(&status) && status != 429 {
187            Kind::AllowAll
188        } else {
189            Kind::BlockAll
190        };
191        RobotsRules {
192            kind,
193            delay: None,
194            sitemaps: Vec::new(),
195        }
196    }
197
198    pub fn from_response(status: u16, body: &[u8], agent: &str) -> RobotsRules {
199        if (200..300).contains(&status) {
200            RobotsRules::parse(body, agent)
201        } else {
202            RobotsRules::from_status(status)
203        }
204    }
205
206    /// Takes an absolute URL or a path (with optional query).
207    pub fn allowed(&self, url_or_path: &str) -> bool {
208        let rules = match &self.kind {
209            Kind::AllowAll => return true,
210            Kind::BlockAll => return false,
211            Kind::Rules(rules) => rules,
212        };
213        let owned;
214        let path = match Url::parse(url_or_path) {
215            Ok(url) => {
216                owned = match url.query() {
217                    Some(q) => format!("{}?{q}", url.path()),
218                    None => url.path().to_owned(),
219                };
220                owned.as_str()
221            }
222            Err(_) => url_or_path,
223        };
224        if path == "/robots.txt" {
225            return true;
226        }
227        let best = rules
228            .iter()
229            .filter(|r| r.matches(path))
230            .max_by_key(|r| (r.len, r.allow));
231        best.is_none_or(|r| r.allow)
232    }
233
234    /// True when the homepage itself is off limits, so there is nothing to crawl.
235    pub fn blocks_everything(&self) -> bool {
236        !self.allowed("/")
237    }
238
239    /// `Crawl-delay`, capped at 10 seconds.
240    pub fn crawl_delay(&self) -> Option<Duration> {
241        self.delay
242    }
243
244    pub fn sitemaps(&self) -> &[String] {
245        &self.sitemaps
246    }
247}
248
249fn parse_delay(value: &str) -> Option<Duration> {
250    let max = Politeness::default().max_crawl_delay;
251    let secs: f64 = value.trim().parse().ok()?;
252    if !secs.is_finite() || secs <= 0.0 {
253        return None;
254    }
255    Some(Duration::from_secs_f64(secs.min(max.as_secs_f64())))
256}
257
258/// Fetches `/robots.txt` for the site `url` belongs to. Connection-level failures
259/// are returned as errors; HTTP error statuses become rules.
260pub async fn fetch_robots(
261    fetcher: &Fetcher,
262    url: &Url,
263) -> Result<(RobotsRules, RobotsFile), FetchError> {
264    let robots_url = url
265        .join("/robots.txt")
266        .map_err(|e| FetchError::Http(format!("bad robots.txt URL: {e}")))?;
267    let res = fetcher.fetch_raw(&robots_url, MAX_ROBOTS_BYTES).await?;
268    let ok = (200..300).contains(&res.status);
269    let body = if ok {
270        res.body.unwrap_or_default()
271    } else {
272        Default::default()
273    };
274    let rules = RobotsRules::from_response(res.status, &body, ROBOTS_AGENT);
275    let file = RobotsFile {
276        status: res.status,
277        body: String::from_utf8_lossy(&body).into_owned(),
278        hash: xxh3_64(&body),
279    };
280    Ok((rules, file))
281}