Skip to main content

codoseo_core/
crawl.rs

1//! Crawl configuration and the small per-crawl records stored alongside pages.
2
3use std::time::Duration;
4
5use serde::{Deserialize, Serialize};
6use url::Url;
7
8pub const USER_AGENT: &str = "CodoSEObot/0.1 (+https://codoseo.com/bot)";
9
10/// Whether the crawler may connect to private and internal addresses.
11/// The cloud uses `Public`; self-hosted, the CLI and local MCP use `AllowPrivate`.
12#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
13#[serde(rename_all = "snake_case")]
14pub enum AddressPolicy {
15    Public,
16    AllowPrivate,
17}
18
19#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
20pub struct CrawlLimits {
21    pub max_pages: u32,
22    pub max_duration: Duration,
23    pub max_page_bytes: usize,
24    pub request_timeout: Duration,
25    pub max_redirects: u8,
26    pub max_sitemap_urls: u32,
27}
28
29impl Default for CrawlLimits {
30    fn default() -> Self {
31        CrawlLimits {
32            max_pages: 500,
33            max_duration: Duration::from_secs(600),
34            max_page_bytes: 5 * 1024 * 1024,
35            request_timeout: Duration::from_secs(30),
36            max_redirects: 10,
37            max_sitemap_urls: 50_000,
38        }
39    }
40}
41
42#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
43pub struct Politeness {
44    pub requests_per_sec: f32,
45    pub per_site_connections: u32,
46    pub max_crawl_delay: Duration,
47    pub max_in_flight: u32,
48    pub max_consecutive_failures: u32,
49}
50
51impl Default for Politeness {
52    fn default() -> Self {
53        Politeness {
54            requests_per_sec: 5.0,
55            per_site_connections: 2,
56            max_crawl_delay: Duration::from_secs(10),
57            max_in_flight: 64,
58            max_consecutive_failures: 20,
59        }
60    }
61}
62
63#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
64pub struct CrawlConfig {
65    pub start_url: Url,
66    pub limits: CrawlLimits,
67    pub politeness: Politeness,
68    pub address_policy: AddressPolicy,
69    pub user_agent: String,
70}
71
72/// robots.txt as fetched at the start of a crawl (kept for change detection).
73#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
74pub struct RobotsFile {
75    pub status: u16,
76    pub body: String,
77    pub hash: u64,
78}
79
80/// What we keep about a site's sitemaps (the URL list itself is not stored).
81#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
82pub struct SitemapSummary {
83    pub files: Vec<Url>,
84    pub url_count: u32,
85    pub hash: u64,
86    /// The URL cap was reached.
87    pub truncated: bool,
88    /// Sitemap files that could not be fetched or parsed.
89    pub failed_files: u32,
90    /// False when discovery stopped early (file cap or deadline), so the URL
91    /// list may be partial and must not be read as the site shrinking.
92    pub complete: bool,
93}