codoseo_crawler/
robots.rs1use std::time::Duration;
10
11use codoseo_core::crawl::{Politeness, RobotsFile};
12use url::Url;
13use xxhash_rust::xxh3::xxh3_64;
14
15use crate::fetch::{FetchError, Fetcher};
16
17pub const ROBOTS_AGENT: &str = "CodoSEObot";
19
20const MAX_ROBOTS_BYTES: usize = 500 * 1024;
22pub const MAX_RULES: usize = 2_000;
23pub const MAX_PATTERN_LEN: usize = 1_024;
24
25struct Rule {
26 parts: Vec<String>,
28 anchored_end: bool,
29 len: usize,
31 allow: bool,
32}
33
34impl Rule {
35 fn new(pattern: &str, allow: bool) -> Option<Rule> {
36 if pattern.is_empty() || pattern.len() > MAX_PATTERN_LEN {
37 return None;
38 }
39 let pattern = if pattern.starts_with('/') || pattern.starts_with('*') {
40 pattern.to_owned()
41 } else {
42 format!("/{pattern}")
43 };
44 let (body, anchored_end) = match pattern.strip_suffix('$') {
45 Some(body) => (body, true),
46 None => (pattern.as_str(), false),
47 };
48 Some(Rule {
49 parts: body.split('*').map(str::to_owned).collect(),
50 anchored_end,
51 len: pattern.len(),
52 allow,
53 })
54 }
55
56 fn matches(&self, path: &str) -> bool {
57 let (first, rest) = self
58 .parts
59 .split_first()
60 .expect("split always yields one part");
61 let Some(mut remaining) = path.strip_prefix(first.as_str()) else {
62 return false;
63 };
64 let Some((last, middle)) = rest.split_last() else {
65 return !self.anchored_end || remaining.is_empty();
66 };
67 for part in middle {
68 match remaining.find(part.as_str()) {
69 Some(at) => remaining = &remaining[at + part.len()..],
70 None => return false,
71 }
72 }
73 if self.anchored_end {
74 remaining.len() >= last.len() && remaining.ends_with(last.as_str())
75 } else {
76 remaining.contains(last.as_str())
77 }
78 }
79}
80
81enum Kind {
82 AllowAll,
83 BlockAll,
84 Rules(Vec<Rule>),
85}
86
87pub struct RobotsRules {
88 kind: Kind,
89 delay: Option<Duration>,
90 sitemaps: Vec<String>,
91}
92
93#[derive(Default)]
94struct Group {
95 agents: Vec<String>,
96 rules: Vec<(String, bool)>,
97 delay: Option<String>,
98}
99
100fn product_token(value: &str) -> String {
102 let value = value.trim();
103 if value.starts_with('*') {
104 return "*".to_owned();
105 }
106 value
107 .chars()
108 .take_while(|c| c.is_ascii_alphanumeric() || *c == '-' || *c == '_')
109 .collect::<String>()
110 .to_ascii_lowercase()
111}
112
113impl RobotsRules {
114 pub fn parse(body: &[u8], agent: &str) -> RobotsRules {
116 let text = String::from_utf8_lossy(&body[..body.len().min(MAX_ROBOTS_BYTES)]);
117 let ours = product_token(agent);
118 let mut groups: Vec<Group> = Vec::new();
119 let mut sitemaps = Vec::new();
120 let mut reading_agents = false;
121
122 for line in text.lines() {
123 let line = line.split('#').next().unwrap_or("");
124 let Some((key, value)) = line.split_once(':') else {
125 continue;
126 };
127 let value = value.trim();
128 match key.trim().to_ascii_lowercase().as_str() {
129 "user-agent" => {
130 if !reading_agents {
131 groups.push(Group::default());
132 }
133 reading_agents = true;
134 if let Some(group) = groups.last_mut() {
135 group.agents.push(product_token(value));
136 }
137 }
138 "sitemap" => sitemaps.push(value.to_owned()),
139 key => {
140 reading_agents = false;
141 let Some(group) = groups.last_mut() else {
142 continue;
143 };
144 match key {
145 "allow" | "disallow" if !value.is_empty() => {
146 group.rules.push((value.to_owned(), key == "allow"));
147 }
148 "crawl-delay" if group.delay.is_none() => {
149 group.delay = Some(value.to_owned())
150 }
151 _ => {}
152 }
153 }
154 }
155 }
156
157 let named: Vec<&Group> = groups.iter().filter(|g| g.agents.contains(&ours)).collect();
158 let chosen = if named.is_empty() {
159 groups
160 .iter()
161 .filter(|g| g.agents.iter().any(|a| a == "*"))
162 .collect()
163 } else {
164 named
165 };
166 let rules: Vec<Rule> = chosen
167 .iter()
168 .flat_map(|g| g.rules.iter())
169 .filter_map(|(pattern, allow)| Rule::new(pattern, *allow))
170 .take(MAX_RULES)
171 .collect();
172 let delay = chosen
173 .iter()
174 .find_map(|g| g.delay.as_deref())
175 .and_then(parse_delay);
176 RobotsRules {
177 kind: Kind::Rules(rules),
178 delay,
179 sitemaps,
180 }
181 }
182
183 pub fn from_status(status: u16) -> RobotsRules {
186 let kind = if (400..500).contains(&status) && status != 429 {
187 Kind::AllowAll
188 } else {
189 Kind::BlockAll
190 };
191 RobotsRules {
192 kind,
193 delay: None,
194 sitemaps: Vec::new(),
195 }
196 }
197
198 pub fn from_response(status: u16, body: &[u8], agent: &str) -> RobotsRules {
199 if (200..300).contains(&status) {
200 RobotsRules::parse(body, agent)
201 } else {
202 RobotsRules::from_status(status)
203 }
204 }
205
206 pub fn allowed(&self, url_or_path: &str) -> bool {
208 let rules = match &self.kind {
209 Kind::AllowAll => return true,
210 Kind::BlockAll => return false,
211 Kind::Rules(rules) => rules,
212 };
213 let owned;
214 let path = match Url::parse(url_or_path) {
215 Ok(url) => {
216 owned = match url.query() {
217 Some(q) => format!("{}?{q}", url.path()),
218 None => url.path().to_owned(),
219 };
220 owned.as_str()
221 }
222 Err(_) => url_or_path,
223 };
224 if path == "/robots.txt" {
225 return true;
226 }
227 let best = rules
228 .iter()
229 .filter(|r| r.matches(path))
230 .max_by_key(|r| (r.len, r.allow));
231 best.is_none_or(|r| r.allow)
232 }
233
234 pub fn blocks_everything(&self) -> bool {
236 !self.allowed("/")
237 }
238
239 pub fn crawl_delay(&self) -> Option<Duration> {
241 self.delay
242 }
243
244 pub fn sitemaps(&self) -> &[String] {
245 &self.sitemaps
246 }
247}
248
249fn parse_delay(value: &str) -> Option<Duration> {
250 let max = Politeness::default().max_crawl_delay;
251 let secs: f64 = value.trim().parse().ok()?;
252 if !secs.is_finite() || secs <= 0.0 {
253 return None;
254 }
255 Some(Duration::from_secs_f64(secs.min(max.as_secs_f64())))
256}
257
258pub async fn fetch_robots(
261 fetcher: &Fetcher,
262 url: &Url,
263) -> Result<(RobotsRules, RobotsFile), FetchError> {
264 let robots_url = url
265 .join("/robots.txt")
266 .map_err(|e| FetchError::Http(format!("bad robots.txt URL: {e}")))?;
267 let res = fetcher.fetch_raw(&robots_url, MAX_ROBOTS_BYTES).await?;
268 let ok = (200..300).contains(&res.status);
269 let body = if ok {
270 res.body.unwrap_or_default()
271 } else {
272 Default::default()
273 };
274 let rules = RobotsRules::from_response(res.status, &body, ROBOTS_AGENT);
275 let file = RobotsFile {
276 status: res.status,
277 body: String::from_utf8_lossy(&body).into_owned(),
278 hash: xxh3_64(&body),
279 };
280 Ok((rules, file))
281}