gossan-subdomain 0.3.3

Subdomain discovery scanner for gossan (CT logs, Wayback, permutations, DNS bruteforce), part of the security research ecosystem

//! CommonCrawl subdomain source.
use gossan_core::{Config, DiscoverySource, DomainTarget, Target};
use crate::sources::{SubdomainSource, SourceRate};
use async_trait::async_trait;
use governor::DefaultDirectRateLimiter;

pub struct CommonCrawl;

#[async_trait]
impl SubdomainSource for CommonCrawl {
    fn name(&self) -> &'static str { "commoncrawl" }
    fn requires_api_key(&self) -> bool { false }
    fn api_key_name(&self) -> &'static str { "" }
    fn rate_limit(&self) -> SourceRate { SourceRate::per_second(1) }
    fn discovery_source(&self) -> DiscoverySource { DiscoverySource::CommonCrawl }

    async fn query(
        &self,
        domain: &str,
        config: &Config,
        client: &reqwest::Client,
        limiter: &DefaultDirectRateLimiter,
    ) -> anyhow::Result<Vec<Target>> {
        let mut indices: Vec<String> = vec!["CC-MAIN-2025-08".into(), "CC-MAIN-2024-51".into(), "CC-MAIN-2024-42".into()];
        
        // Try to discover latest index
        limiter.until_ready().await;
        let collinfo = client.get("https://index.commoncrawl.org/collinfo.json").send().await;
        match collinfo {
            Ok(resp) => {
                let max_size = config.max_response_size;
                match gossan_core::read_response_limited(resp, max_size).await {
                    Ok(bytes) => match serde_json::from_slice::<serde_json::Value>(&bytes) {
                        Ok(arr) => {
                            if let Some(list) = arr.as_array() {
                                indices.clear();
                                for item in list.iter().take(3) {
                                    if let Some(id) = item.get("id").and_then(|v| v.as_str()) {
                                        indices.push(id.to_string());
                                    }
                                }
                            }
                        }
                        Err(e) => tracing::warn!(error = %e, "commoncrawl collinfo JSON parse failed; using fallback indices"),
                    },
                    Err(e) => tracing::warn!(error = %e, "commoncrawl collinfo body read failed; using fallback indices"),
                }
            }
            Err(e) => tracing::warn!(error = %e, "commoncrawl collinfo fetch failed; using fallback indices"),
        }

        let mut seen = std::collections::HashSet::new();
        let domain_lower = domain.to_lowercase();
        let max_size = config.max_response_size;

        for index in &indices {
            let url = format!(
                "https://index.commoncrawl.org/{index}-index?url=*.{}&output=json&fl=url&limit=5000",
                domain
            );
            limiter.until_ready().await;
            let resp = client.get(&url).send().await?.error_for_status()?;
            let text = String::from_utf8(gossan_core::read_response_limited(resp, max_size).await?)?;
            for line in text.lines() {
                if line.is_empty() { continue; }
                match serde_json::from_str::<serde_json::Value>(line) {
                    Ok(json) => {
                    if let Some(u) = json.get("url").and_then(|v| v.as_str()) {
                        if let Ok(parsed) = url::Url::parse(u) {
                            if let Some(host) = parsed.host_str() {
                                let host = host.to_lowercase();
                                if crate::is_subdomain_of(&host, &domain_lower) {
                                    seen.insert(host);
                                }
                            }
                        }
                    }
                    }
                    Err(e) => {
                        tracing::warn!(error = %e, line = %line.chars().take(80).collect::<String>(), "commoncrawl index line JSON parse failed; skipping line");
                    }
                }
            }
        }

        Ok(seen.into_iter().map(|d| Target::Domain(DomainTarget {
            domain: d,
            source: DiscoverySource::CommonCrawl,
        })).collect())
    }
}