Skip to main content

codoseo_core/
output.rs

1//! What a crawl produces: pages, the link graph and why it stopped.
2
3use serde::{Deserialize, Serialize};
4use url::Url;
5
6use crate::crawl::{RobotsFile, SitemapSummary};
7use crate::page::PageRecord;
8
9/// One internal link between two crawled pages.
10#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
11pub struct Edge {
12    /// Index into `CrawlOutput::pages`.
13    pub from: u32,
14    /// Index into `CrawlOutput::pages`.
15    pub to: u32,
16    /// Index into `LinkGraph::anchors`.
17    pub anchor: u32,
18    pub nofollow: bool,
19}
20
21/// Links between crawled pages, with anchor text interned so repeated navigation
22/// links don't repeat their text.
23#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
24pub struct LinkGraph {
25    pub edges: Vec<Edge>,
26    pub anchors: Vec<String>,
27}
28
29impl LinkGraph {
30    pub fn anchor(&self, e: &Edge) -> &str {
31        &self.anchors[e.anchor as usize]
32    }
33}
34
35/// Why a crawl ended.
36#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
37#[serde(rename_all = "snake_case", tag = "kind", content = "reason")]
38pub enum StopReason {
39    Completed,
40    PageLimit,
41    TimeLimit,
42    Unreachable(String),
43    Blocked(String),
44    RobotsBlocked,
45}
46
47impl StopReason {
48    /// The whole site was crawled.
49    pub fn is_complete(&self) -> bool {
50        matches!(self, StopReason::Completed)
51    }
52
53    /// The crawl got going and produced pages, even if it stopped early.
54    pub fn crawl_ran(&self) -> bool {
55        matches!(
56            self,
57            StopReason::Completed | StopReason::PageLimit | StopReason::TimeLimit
58        )
59    }
60}
61
62/// A snapshot of a running crawl, for progress display.
63#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
64pub struct Progress {
65    pub pages_done: u32,
66    pub queued: u32,
67    pub failures: u32,
68    pub depth: u16,
69    pub elapsed_ms: u64,
70}
71
72#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
73pub struct CrawlOutput {
74    /// The settled start address: the origin pages are internal to.
75    pub origin: Url,
76    pub pages: Vec<PageRecord>,
77    pub links: LinkGraph,
78    pub robots: Option<RobotsFile>,
79    pub sitemap: SitemapSummary,
80    pub stop: StopReason,
81    pub duration_ms: u64,
82}