Skip to main content

codoseo_core/
output.rs

1//! What a crawl produces: pages, the link graph and why it stopped.
2
3use serde::{Deserialize, Serialize};
4use url::Url;
5
6use crate::crawl::{RobotsFile, SitemapSummary};
7use crate::page::PageRecord;
8
9/// One internal link between two crawled pages.
10#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
11pub struct Edge {
12    /// Index into `CrawlOutput::pages`.
13    pub from: u32,
14    /// Index into `CrawlOutput::pages`.
15    pub to: u32,
16    /// Index into `LinkGraph::anchors`.
17    pub anchor: u32,
18    pub nofollow: bool,
19}
20
21/// Links between crawled pages, with anchor text interned so repeated navigation
22/// links don't repeat their text.
23#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
24pub struct LinkGraph {
25    pub edges: Vec<Edge>,
26    pub anchors: Vec<String>,
27}
28
29impl LinkGraph {
30    pub fn anchor(&self, e: &Edge) -> &str {
31        &self.anchors[e.anchor as usize]
32    }
33}
34
35/// Why a crawl ended.
36#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
37#[serde(rename_all = "snake_case", tag = "kind", content = "reason")]
38pub enum StopReason {
39    Completed,
40    PageLimit,
41    TimeLimit,
42    Unreachable(String),
43    Blocked(String),
44    RobotsBlocked,
45}
46
47impl StopReason {
48    /// The whole site was crawled.
49    pub fn is_complete(&self) -> bool {
50        matches!(self, StopReason::Completed)
51    }
52
53    /// The crawl got going and produced pages, even if it stopped early.
54    pub fn crawl_ran(&self) -> bool {
55        matches!(
56            self,
57            StopReason::Completed | StopReason::PageLimit | StopReason::TimeLimit
58        )
59    }
60}
61
62/// Start of a crawl's `failure_reason` when the site never answered (`StopReason::Unreachable`).
63pub const UNREACHABLE_REASON_PREFIX: &str = "site unreachable";
64/// Start of a crawl's `failure_reason` when the site refused our crawler (`StopReason::Blocked`).
65pub const BLOCKED_REASON_PREFIX: &str = "site blocked our crawler";
66
67/// Whose fault a failed crawl was, read back from its stored `failure_reason`.
68#[derive(Debug, Clone, Copy, PartialEq, Eq)]
69pub enum SiteFault {
70    /// The site didn't answer.
71    Unreachable,
72    /// The site answered with errors or a challenge page.
73    Blocked,
74}
75
76impl SiteFault {
77    /// `Some` only for the reasons the crawler's stop messages produce; internal failures
78    /// (database errors, memory budget, ...) are ours, not the site's, and give `None`.
79    pub fn from_failure_reason(reason: &str) -> Option<SiteFault> {
80        if reason.starts_with(UNREACHABLE_REASON_PREFIX) {
81            Some(SiteFault::Unreachable)
82        } else if reason.starts_with(BLOCKED_REASON_PREFIX) {
83            Some(SiteFault::Blocked)
84        } else {
85            None
86        }
87    }
88}
89
90/// A snapshot of a running crawl, for progress display.
91#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
92pub struct Progress {
93    pub pages_done: u32,
94    pub queued: u32,
95    pub failures: u32,
96    pub depth: u16,
97    pub elapsed_ms: u64,
98}
99
100#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
101pub struct CrawlOutput {
102    /// The settled start address: the origin pages are internal to.
103    pub origin: Url,
104    pub pages: Vec<PageRecord>,
105    pub links: LinkGraph,
106    pub robots: Option<RobotsFile>,
107    pub sitemap: SitemapSummary,
108    pub stop: StopReason,
109    pub duration_ms: u64,
110}