include!(concat!(env!("OUT_DIR"), "/bad_websites.rs"));
pub mod firewall {
use std::sync::OnceLock;
pub static GLOBAL_BAD_WEBSITES: OnceLock<&phf::Set<&'static str>> = OnceLock::new();
pub static GLOBAL_ADS_WEBSITES: OnceLock<&phf::Set<&'static str>> = OnceLock::new();
pub static GLOBAL_TRACKING_WEBSITES: OnceLock<&phf::Set<&'static str>> = OnceLock::new();
pub static GLOBAL_GAMBLING_WEBSITES: OnceLock<&phf::Set<&'static str>> = OnceLock::new();
pub static GLOBAL_NETWORKING_WEBSITES: OnceLock<&phf::Set<&'static str>> = OnceLock::new();
#[macro_export]
macro_rules! define_firewall {
($category:expr, $($site:expr),* $(,)?) => {
match $category {
"ads" => {
if $crate::firewall::GLOBAL_ADS_WEBSITES.get().is_none() {
$crate::firewall::GLOBAL_ADS_WEBSITES
.set(&phf::phf_set! { $($site),* })
.expect("Initialization already set.");
}
},
"tracking" => {
if $crate::firewall::GLOBAL_TRACKING_WEBSITES.get().is_none() {
$crate::firewall::GLOBAL_TRACKING_WEBSITES
.set(&phf::phf_set! { $($site),* })
.expect("Initialization already set.");
}
},
"gambling" => {
if $crate::firewall::GLOBAL_GAMBLING_WEBSITES.get().is_none() {
$crate::firewall::GLOBAL_GAMBLING_WEBSITES
.set(&phf::phf_set! { $($site),* })
.expect("Initialization already set.");
}
},
"networking" => {
if $crate::firewall::GLOBAL_NETWORKING_WEBSITES.get().is_none() {
$crate::firewall::GLOBAL_NETWORKING_WEBSITES
.set(&phf::phf_set! { $($site),* })
.expect("Initialization already set.");
}
},
_ => {
if $crate::firewall::GLOBAL_BAD_WEBSITES.get().is_none() {
$crate::firewall::GLOBAL_BAD_WEBSITES
.set(&phf::phf_set! { $($site),* })
.expect("Initialization already set.");
}
},
}
};
}
}
use std::sync::OnceLock;
#[cfg(feature = "dynamic")]
pub mod dynamic;
pub const CAT_BAD: u64 = 1;
pub const CAT_ADS: u64 = 2;
pub const CAT_TRACKING: u64 = 4;
pub const CAT_GAMBLING: u64 = 8;
pub const CAT_ADULT: u64 = 16;
pub const CAT_LISTED: u64 = 32;
macro_rules! dyn_cat_or {
($host:expr, $cat:expr) => {{
#[cfg(feature = "dynamic")]
{
$crate::dynamic::dynamic_has_category($host, $cat)
}
#[cfg(not(feature = "dynamic"))]
{
false
}
}};
}
macro_rules! dyn_any_or {
($host:expr) => {{
#[cfg(feature = "dynamic")]
{
$crate::dynamic::dynamic_contains($host)
}
#[cfg(not(feature = "dynamic"))]
{
false
}
}};
}
static FIREWALL_MAP: OnceLock<fst::Map<&'static [u8]>> = OnceLock::new();
#[inline]
fn firewall_map() -> &'static fst::Map<&'static [u8]> {
FIREWALL_MAP
.get_or_init(|| fst::Map::new(FIREWALL_FST_BYTES).expect("firewall fst invalid"))
}
#[inline]
pub(crate) fn is_explicit_icann_suffix(name: &str) -> bool {
const PROBE: &[u8] = b"0--spider-psl-probe";
const CAP: usize = 256;
let bytes = name.as_bytes();
let suffix = match psl::suffix(bytes) {
Some(s) => s,
None => return false,
};
if suffix.as_bytes().len() != bytes.len()
|| !suffix.is_known()
|| suffix.typ() != Some(psl::Type::Icann)
{
return false;
}
let parent = match name.find('.') {
Some(dot) => &bytes[dot..],
None => return true,
};
let len = PROBE.len() + parent.len();
if len > CAP {
return false;
}
let mut buf = [0u8; CAP];
buf[..PROBE.len()].copy_from_slice(PROBE);
buf[PROBE.len()..len].copy_from_slice(parent);
let probe = &buf[..len];
match psl::suffix(probe) {
Some(s) => s.as_bytes().len() != probe.len(),
None => true,
}
}
#[inline]
fn fst_has_category(host: &str, cat: u64) -> bool {
map_has_category(firewall_map(), host, cat)
}
#[inline]
fn map_has_category<D: AsRef<[u8]>>(map: &fst::Map<D>, host: &str, cat: u64) -> bool {
let mut h = host;
loop {
if is_explicit_icann_suffix(h) {
break;
}
if let Some(v) = map.get(h) {
if v & cat != 0 {
return true;
}
}
match h.find('.') {
Some(dot) => {
h = &h[dot + 1..];
if !h.contains('.') {
break;
}
}
None => break,
}
}
false
}
pub fn get_host_from_url(url: &str) -> Option<&str> {
let url = url
.trim_start_matches("https://")
.trim_start_matches("http://");
if let Some(pos) = url.find('/') {
Some(&url[..pos])
} else {
Some(&url)
}
}
pub fn is_bad_website_url(host: &str) -> bool {
fst_has_category(host, CAT_BAD)
|| is_website_in_custom_set(host, &firewall::GLOBAL_BAD_WEBSITES)
|| dyn_cat_or!(host, CAT_BAD)
}
pub fn is_adult_website_url(host: &str) -> bool {
fst_has_category(host, CAT_ADULT) || dyn_cat_or!(host, CAT_ADULT)
}
pub fn is_listed_website_url(host: &str) -> bool {
fst_has_category(host, CAT_LISTED) || dyn_cat_or!(host, CAT_LISTED)
}
pub fn is_ad_website_url(host: &str) -> bool {
fst_has_category(host, CAT_ADS)
|| is_website_in_custom_set(host, &firewall::GLOBAL_ADS_WEBSITES)
|| dyn_cat_or!(host, CAT_ADS)
}
pub fn is_tracking_website_url(host: &str) -> bool {
fst_has_category(host, CAT_TRACKING)
|| is_website_in_custom_set(host, &firewall::GLOBAL_TRACKING_WEBSITES)
|| dyn_cat_or!(host, CAT_TRACKING)
}
pub fn is_gambling_website_url(host: &str) -> bool {
fst_has_category(host, CAT_GAMBLING)
|| is_website_in_custom_set(host, &firewall::GLOBAL_GAMBLING_WEBSITES)
|| dyn_cat_or!(host, CAT_GAMBLING)
}
pub fn is_networking_url(host: &str) -> bool {
fst_has_category(host, CAT_BAD)
|| is_website_in_custom_set(host, &firewall::GLOBAL_BAD_WEBSITES)
|| is_website_in_custom_set(host, &firewall::GLOBAL_NETWORKING_WEBSITES)
|| dyn_cat_or!(host, CAT_BAD)
}
pub fn is_url_bad(host: &str) -> bool {
fst_contains_any(host)
|| is_website_in_custom_set(host, &firewall::GLOBAL_BAD_WEBSITES)
|| is_website_in_custom_set(host, &firewall::GLOBAL_ADS_WEBSITES)
|| is_website_in_custom_set(host, &firewall::GLOBAL_NETWORKING_WEBSITES)
|| is_website_in_custom_set(host, &firewall::GLOBAL_TRACKING_WEBSITES)
|| is_website_in_custom_set(host, &firewall::GLOBAL_GAMBLING_WEBSITES)
|| dyn_any_or!(host)
}
#[inline]
fn fst_contains_any(host: &str) -> bool {
let map = firewall_map();
let mut h = host;
loop {
if is_explicit_icann_suffix(h) {
break;
}
if map.contains_key(h) {
return true;
}
match h.find('.') {
Some(dot) => {
h = &h[dot + 1..];
if !h.contains('.') {
break;
}
}
None => break,
}
}
false
}
fn is_website_in_custom_set(
host: &str,
set: &std::sync::OnceLock<&phf::Set<&'static str>>,
) -> bool {
set.get().map(|s| s.contains(host)).unwrap_or(false)
}
pub fn is_bad_website_url_clean(host: &str) -> bool {
get_host_from_url(host)
.map(is_bad_website_url)
.unwrap_or(false)
}
pub fn is_ad_website_url_clean(host: &str) -> bool {
get_host_from_url(host)
.map(is_ad_website_url)
.unwrap_or(false)
}
pub fn is_tracking_website_url_clean(host: &str) -> bool {
get_host_from_url(host)
.map(is_tracking_website_url)
.unwrap_or(false)
}
pub fn is_networking_website_url_clean(host: &str) -> bool {
get_host_from_url(host)
.map(is_networking_url)
.unwrap_or(false)
}
pub fn is_gambling_website_url_clean(host: &str) -> bool {
get_host_from_url(host)
.map(is_gambling_website_url)
.unwrap_or(false)
}
#[cfg(feature = "ip")]
mod ip_block {
include!(concat!(env!("OUT_DIR"), "/bad_ips.rs"));
#[inline]
pub(crate) fn ranges_contain(ranges: &[(u32, u32)], ip: u32) -> bool {
match ranges.binary_search_by(|&(start, _)| start.cmp(&ip)) {
Ok(_) => true,
Err(0) => false,
Err(i) => {
let (start, end) = ranges[i - 1];
start <= ip && ip <= end
}
}
}
#[inline]
pub(crate) fn is_bad_ipv4(ip: u32) -> bool {
ranges_contain(BAD_IP_RANGES_V4, ip)
}
}
#[cfg(feature = "ip")]
pub fn is_bad_ip(ip: std::net::IpAddr) -> bool {
match ip {
std::net::IpAddr::V4(v4) => ip_block::is_bad_ipv4(u32::from(v4)),
std::net::IpAddr::V6(_) => false,
}
}
#[cfg(feature = "ip")]
pub fn is_bad_ip_str(ip: &str) -> bool {
ip.parse::<std::net::IpAddr>()
.map(is_bad_ip)
.unwrap_or(false)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_is_bad_website_url_within_set() {
let bad_website = "wingwahlau.com";
assert!(is_bad_website_url(bad_website));
}
#[test]
fn test_is_bad_website_url_not_in_set() {
let good_website = "goodwebsite.com";
assert!(!is_bad_website_url(good_website));
}
#[test]
fn test_is_bad_website_url_empty_string() {
assert!(!is_bad_website_url(""));
}
#[test]
fn test_is_bad_website_url_case_sensitivity() {
let bad_website = "10minutesto1.net";
assert!(is_bad_website_url(bad_website.to_lowercase().as_str()));
}
#[test]
fn test_is_ad_website_url() {
assert!(is_ad_website_url("admob.google.com"));
assert!(is_ad_website_url("ads.linkedin.com"));
}
#[test]
fn test_is_tracking_website_url() {
assert!(!is_tracking_website_url("2.atlasroofing.com"));
assert!(is_tracking_website_url(
"pixel.rubiconproject.net.akadns.net"
));
}
#[test]
fn test_ning_com_whitelisted() {
assert!(!is_bad_website_url("ning.com"), "ning.com should be whitelisted");
assert!(!is_bad_website_url("competitiveintelligence.ning.com"), "subdomain of ning.com should be whitelisted");
}
#[test]
fn test_adult_websites_categorized_not_hard_refused() {
for host in ["pornhub.com", "xvideos.com"] {
assert!(is_adult_website_url(host), "{host} is on an adult list");
assert!(is_url_bad(host), "{host} still matches the any-category check");
assert!(!is_bad_website_url(host), "{host} is not a threat-feed entry");
}
}
#[test]
fn test_category_lists_are_not_hard_refused() {
assert!(is_listed_website_url("www.veed.io"));
assert!(!is_bad_website_url("www.veed.io"));
assert!(!is_bad_website_url_clean("https://www.veed.io/ms-MY/peralatan"));
for host in ["more.com", "help-teller.more.com"] {
assert!(!is_bad_website_url(host), "{host}");
assert!(!is_adult_website_url(host), "{host}");
assert!(!is_url_bad(host), "{host}");
}
}
#[test]
fn test_threat_feed_entries_still_hard_refused() {
for host in [
"zoominfo.com",
"www.zoominfo.com",
"wingwahlau.com", "10minutesto1.net", "buffooncountabletreble.com",
"backspinreentryupright.com",
"sub.backspinreentryupright.com",
] {
assert!(is_bad_website_url(host), "{host} must stay a threat verdict");
}
}
#[test]
fn test_legit_websites_not_false_positive() {
assert!(!is_bad_website_url("github.com"));
assert!(!is_bad_website_url("wikipedia.org"));
assert!(!is_bad_website_url("google.com"));
}
#[test]
fn test_trustpilot_whitelisted() {
assert!(!is_bad_website_url("trustpilot.com"), "trustpilot.com should be whitelisted");
assert!(!is_bad_website_url("www.trustpilot.com"), "www.trustpilot.com should be whitelisted");
assert!(!is_bad_website_url_clean("https://www.trustpilot.com/review/grey.co"), "trustpilot review URL should not be bad");
assert!(!is_url_bad("www.trustpilot.com"), "trustpilot should not match any bad category");
}
#[test]
fn test_usask_whitelisted() {
assert!(!is_bad_website_url("usask.ca"), "usask.ca should be whitelisted");
assert!(!is_bad_website_url("admissions.usask.ca"), "admissions.usask.ca should be whitelisted");
assert!(!is_bad_website_url("medicine.usask.ca"), "medicine.usask.ca should be whitelisted");
assert!(!is_url_bad("admissions.usask.ca"), "admissions.usask.ca should not match any bad category");
assert!(!is_url_bad("medicine.usask.ca"), "medicine.usask.ca should not match any bad category");
assert!(!is_bad_website_url_clean("https://admissions.usask.ca/programs"), "usask admissions URL should not be bad");
}
#[test]
fn test_ai_vendors_whitelisted() {
for host in [
"openai.com",
"chatgpt.com",
"anthropic.com",
"claude.ai",
"huggingface.co",
"ollama.com",
"perplexity.ai",
"labs.perplexity.ai", "stability.ai",
"meta.ai",
"openrouter.ai",
] {
assert!(!is_bad_website_url(host), "{host} should be whitelisted");
assert!(!is_url_bad(host), "{host} should not match any bad category");
}
assert!(
!is_bad_website_url_clean("https://platform.openai.com/docs/api-reference"),
"openai docs URL should not be bad"
);
}
#[test]
fn test_developer_tools_whitelisted_across_categories() {
for host in [
"honeycomb.io",
"instana.io",
"honeybadger.io",
"raygun.io",
"airbrake.io",
"backtrace.io",
"canny.io",
"plausible.io",
"ipinfo.io",
"ipgeolocation.io",
] {
assert!(!is_url_bad(host), "{host} should not be refused by any category");
}
assert!(!is_ad_website_url("plausible.io"), "plausible.io is not an ad network");
assert!(!is_tracking_website_url("honeycomb.io"), "honeycomb.io is observability, not tracking");
}
#[test]
fn test_gambling_still_blocked() {
assert!(is_url_bad("calottery.com"), "gambling must remain blocked");
assert!(is_url_bad("bet9ja.com"), "gambling must remain blocked");
}
#[test]
fn test_known_bad_still_blocked() {
for host in ["buffooncountabletreble.com", "backspinreentryupright.com"] {
assert!(is_url_bad(host), "{host} must stay blocked");
}
}
#[test]
fn test_abuse_prone_infrastructure_still_blocked() {
for host in ["use-application-dns.net", "sslip.io", "packetstream.io", "short.io"] {
assert!(is_url_bad(host), "{host} must stay blocked");
}
}
#[test]
fn test_ad_and_beacon_endpoints_still_blocked() {
for host in ["1rx.io", "bidr.io", "fpjs.io", "kameleoon.io"] {
assert!(is_url_bad(host), "{host} must stay blocked");
}
}
#[test]
fn test_public_suffix_entries_do_not_block_their_zone() {
for host in [
"www.sina.com.cn",
"www.people.com.cn",
"edu.china.com.cn",
"static.cninfo.com.cn",
"www.questmobile.com.cn",
"www.mtkxjs.com.cn",
] {
assert!(!is_bad_website_url(host), "{host} must not be refused by the com.cn entry");
assert!(!is_url_bad(host), "{host} must not match any category");
}
assert!(!is_bad_website_url_clean("https://www.sina.com.cn/news"));
}
#[test]
fn test_listed_hosts_under_a_public_suffix_still_blocked() {
for host in [
"cn-oyi-okx.com.cn", "download-sougou.com.cn", "b86-telegram.com.cn", ] {
assert!(is_bad_website_url(host), "{host} must stay blocked");
let sub = format!("login.{host}");
assert!(is_bad_website_url(&sub), "{sub} must stay blocked");
}
}
#[test]
fn test_private_and_wildcard_suffix_listings_still_block() {
for host in [
"coinbase-corp.fk",
"login.coinbase-corp.fk",
"googlecom.mm",
"anything.amplifyapp.com",
] {
assert!(is_bad_website_url(host), "{host} must stay blocked");
}
assert!(is_bad_website_url("a.b.wingwahlau.com"));
}
#[test]
fn test_walk_skips_a_public_suffix_entry_even_if_present() {
let map = fst::Map::from_iter(vec![("com.cn", CAT_BAD), ("evil.com.cn", CAT_BAD)]).unwrap();
assert!(!map_has_category(&map, "www.sina.com.cn", CAT_BAD));
assert!(map_has_category(&map, "evil.com.cn", CAT_BAD));
assert!(map_has_category(&map, "a.evil.com.cn", CAT_BAD));
let map = fst::Map::from_iter(vec![("amplifyapp.com", CAT_BAD)]).unwrap();
assert!(map_has_category(&map, "x.amplifyapp.com", CAT_BAD), "private suffix entries still block");
}
#[test]
fn test_is_explicit_icann_suffix() {
for s in ["com.cn", "co.uk", "com.au", "gov.cn", "co.tz", "ad.jp", "cn"] {
assert!(is_explicit_icann_suffix(s), "{s}");
}
for s in [
"sina.com.cn",
"bbc.co.uk",
"amplifyapp.com",
"blogspot.com",
"coinbase-corp.fk",
"googlecom.mm",
"",
] {
assert!(!is_explicit_icann_suffix(s), "{s}");
}
let long = format!("{}.com.cn", "a".repeat(300));
assert!(!is_explicit_icann_suffix(&long));
}
#[test]
fn test_define_firewall_macro() {
define_firewall!("ads", "adwebsite.com", "ad1website.com");
assert!(is_ad_website_url("adwebsite.com"));
assert!(is_ad_website_url("ad1website.com"));
define_firewall!("gambling", "gamblingwebsite.com");
assert!(is_gambling_website_url("gamblingwebsite.com"));
define_firewall!("global", "anotherbadwebsite.com", "chrome:/");
assert!(is_bad_website_url("anotherbadwebsite.com"));
assert!(is_bad_website_url("chrome:/"));
}
#[cfg(feature = "ip")]
#[test]
fn test_ip_ranges_contain() {
let ranges = &[(167772160u32, 167772415u32), (3232235776u32, 3232235779u32)];
assert!(super::ip_block::ranges_contain(ranges, 167772160)); assert!(super::ip_block::ranges_contain(ranges, 167772415)); assert!(super::ip_block::ranges_contain(ranges, 167772300)); assert!(!super::ip_block::ranges_contain(ranges, 167772416)); assert!(!super::ip_block::ranges_contain(ranges, 167772159)); assert!(super::ip_block::ranges_contain(ranges, 3232235778)); assert!(!super::ip_block::ranges_contain(ranges, 0)); assert!(!super::ip_block::ranges_contain(ranges, u32::MAX)); }
#[cfg(feature = "ip")]
#[test]
fn test_is_bad_ip_str() {
assert!(!is_bad_ip_str("not-an-ip"));
assert!(!is_bad_ip_str(""));
assert!(!is_bad_ip("10.0.0.1".parse().unwrap()));
assert!(!is_bad_ip("::1".parse().unwrap()));
}
}