use crate::error::ConfigError;
use crate::selector::{Selector, SelectorSyntax};
use crate::util::Util;
use std::borrow::Cow;
use std::io::Cursor;
use std::path::Path;
use tokio::fs;
use tokio::io::{AsyncBufReadExt, AsyncRead, BufReader};
#[derive(Clone, Debug)]
pub struct Replace {
pub to_replace: String,
pub replace_with: String,
}
#[derive(Clone, Debug)]
pub struct PageLink {
pub selector: Selector,
pub if_page_contains: Option<Selector>,
}
impl From<Selector> for PageLink {
fn from(selector: Selector) -> Self {
Self {
selector,
if_page_contains: None,
}
}
}
#[derive(Clone, Debug)]
pub struct MoveInto {
pub target: Selector,
pub selector: Selector,
}
#[derive(Clone, Debug)]
pub struct WrapIn {
pub tag: String,
pub selector: Selector,
}
#[derive(Clone, Debug)]
pub struct Header {
pub name: String,
pub value: String,
}
#[derive(Clone, Debug, Default)]
pub struct ConfigEntry {
pub title: Vec<Selector>,
pub author: Vec<Selector>,
pub date: Vec<Selector>,
pub body: Vec<Selector>,
pub strip: Vec<Selector>,
pub strip_attr: Vec<Selector>,
pub post_strip_attr: Vec<Selector>,
pub dissolve: Vec<Selector>,
pub wrap_in: Vec<WrapIn>,
pub move_into: Vec<MoveInto>,
pub move_into_body: Vec<Selector>,
pub strip_id_or_class: Vec<String>,
pub strip_image_src: Vec<String>,
pub replace: Vec<Replace>,
pub header: Vec<Header>,
pub single_page_link: Vec<PageLink>,
pub next_page_link: Vec<PageLink>,
pub skip_json_ld: Option<bool>,
pub insert_detected_image: Option<bool>,
pub prune: Option<bool>,
pub autodetect_on_failure: Option<bool>,
pub native_ad_clue: Vec<Selector>,
pub src_lazy_load_attr: Option<String>,
pub autodetect_next_page: Option<bool>,
}
impl ConfigEntry {
pub async fn parse_path(
config_path: &Path,
syntax: &dyn SelectorSyntax,
) -> Result<ConfigEntry, ConfigError> {
let mut file = fs::File::open(&config_path).await?;
let buffer = BufReader::new(&mut file);
Self::parse(buffer, &config_path.to_string_lossy(), syntax).await
}
pub async fn parse_data(
data: Cow<'static, [u8]>,
name: &str,
syntax: &dyn SelectorSyntax,
) -> Result<ConfigEntry, ConfigError> {
let data = data.as_ref();
let mut cursor = Cursor::new(data);
let buffer = BufReader::new(&mut cursor);
Self::parse(buffer, name, syntax).await
}
async fn parse<R: AsyncRead + Unpin>(
buffer: BufReader<R>,
name: &str,
syntax: &dyn SelectorSyntax,
) -> Result<ConfigEntry, ConfigError> {
let mut titles: Vec<Selector> = Vec::new();
let mut authors: Vec<Selector> = Vec::new();
let mut dates: Vec<Selector> = Vec::new();
let mut bodies: Vec<Selector> = Vec::new();
let mut strips: Vec<Selector> = Vec::new();
let mut strip_attr: Vec<Selector> = Vec::new();
let mut post_strip_attr: Vec<Selector> = Vec::new();
let mut dissolve: Vec<Selector> = Vec::new();
let mut wrap_in: Vec<WrapIn> = Vec::new();
let mut move_into: Vec<MoveInto> = Vec::new();
let mut move_into_body: Vec<Selector> = Vec::new();
let mut strip_id_or_class: Vec<String> = Vec::new();
let mut strip_image_src: Vec<String> = Vec::new();
let mut replace_vec: Vec<Replace> = Vec::new();
let mut header_vec: Vec<Header> = Vec::new();
let mut next_page_link: Vec<PageLink> = Vec::new();
let mut single_page_link: Vec<PageLink> = Vec::new();
let mut skip_json_ld: Option<bool> = None;
let mut insert_detected_image: Option<bool> = None;
let mut prune: Option<bool> = None;
let mut autodetect_on_failure: Option<bool> = None;
let mut native_ad_clue: Vec<Selector> = Vec::new();
let mut src_lazy_load_attr: Option<String> = None;
let mut autodetect_next_page: Option<bool> = None;
let mut find_string: Option<String> = None;
let selector = |directive: &str, value: &str| match syntax.parse(value) {
Ok(selector) => Some(selector),
Err(error) => {
let directive = directive.trim_end_matches(':');
tracing::warn!(config = name, directive, value, %error, "Skipping invalid selector");
None
}
};
let title = "title:";
let body = "body:";
let date = "date:";
let author = "author:";
let strip = "strip:";
let strip_attr_directive = "strip_attr:";
let post_strip_attr_directive = "post_strip_attr:";
let dissolve_directive = "dissolve:";
let strip_id = "strip_id_or_class:";
let strip_img = "strip_image_src:";
let single_page = "single_page_link:";
let next_page = "next_page_link:";
let if_page_contains = "if_page_contains:";
let find = "find_string:";
let replace = "replace_string:";
let replace_single = "replace_string(";
let http_header = "http_header(";
let wrap_in_directive = "wrap_in(";
let move_into_directive = "move_into(";
let skip_json_ld_directive = "skip_json_ld:";
let insert_detected_image_directive = "insert_detected_image:";
let prune_directive = "prune:";
let autodetect_directive = "autodetect_on_failure:";
let native_ad_directive = "native_ad_clue:";
let lazy_load_directive = "src_lazy_load_attr:";
let autodetect_next_page_directive = "autodetect_next_page:";
let mut lines = buffer.lines();
while let Ok(Some(line)) = lines.next_line().await {
let line = line.trim();
if line.starts_with('#')
|| line.is_empty()
|| IGNORED_DIRECTIVES
.iter()
.any(|directive| line.starts_with(directive))
{
continue;
}
extract_selector_vec!(line, title, titles, selector);
extract_selector_vec!(line, body, bodies, selector);
extract_selector_vec!(line, date, dates, selector);
extract_selector_vec!(line, author, authors, selector);
extract_selector_vec!(line, strip, strips, selector);
extract_selector_vec!(line, strip_attr_directive, strip_attr, selector);
extract_selector_vec!(line, post_strip_attr_directive, post_strip_attr, selector);
extract_selector_vec!(line, dissolve_directive, dissolve, selector);
extract_selector_vec!(line, native_ad_directive, native_ad_clue, selector);
extract_vec_single!(line, strip_id, strip_id_or_class);
extract_vec_single!(line, strip_img, strip_image_src);
if line.starts_with(single_page) {
let value = Util::str_extract_value(single_page, line);
single_page_link.extend(selector(single_page, value).map(PageLink::from));
continue;
}
if line.starts_with(next_page) {
let value = Util::str_extract_value(next_page, line);
next_page_link.extend(selector(next_page, value).map(PageLink::from));
continue;
}
if line.starts_with(if_page_contains) {
let value = Util::str_extract_value(if_page_contains, line);
let rule = single_page_link
.last_mut()
.or_else(|| next_page_link.last_mut());
match (rule, selector(if_page_contains, value)) {
(Some(rule), Some(condition)) => rule.if_page_contains = Some(condition),
(None, _) => tracing::warn!(
config = name,
"if_page_contains without a single_page_link or next_page_link"
),
(Some(_), None) => {}
}
continue;
}
if line.starts_with(replace_single) {
let value = line.strip_prefix(replace_single).unwrap_or_default();
let Some((to_replace, replace_with)) = value.split_once("):") else {
continue;
};
if to_replace.is_empty() {
continue;
}
replace_vec.push(Replace {
to_replace: to_replace.to_string(),
replace_with: replace_with.trim_start().to_string(),
});
continue;
}
if line.starts_with(skip_json_ld_directive) {
skip_json_ld = yes_no(name, line, skip_json_ld_directive);
continue;
}
if line.starts_with(autodetect_next_page_directive) {
autodetect_next_page = yes_no(name, line, autodetect_next_page_directive);
continue;
}
if line.starts_with(lazy_load_directive) {
let value = Util::str_extract_value(lazy_load_directive, line);
if !value.is_empty()
&& value
.chars()
.all(|c| c.is_ascii_alphanumeric() || c == '-' || c == '_')
{
src_lazy_load_attr = Some(value.to_ascii_lowercase());
} else {
tracing::warn!(config = name, value, "Skipping invalid src_lazy_load_attr");
}
continue;
}
if line.starts_with(autodetect_directive) {
autodetect_on_failure = yes_no(name, line, autodetect_directive);
continue;
}
if line.starts_with(prune_directive) {
prune = yes_no(name, line, prune_directive);
continue;
}
if line.starts_with(insert_detected_image_directive) {
insert_detected_image = yes_no(name, line, insert_detected_image_directive);
continue;
}
if line.starts_with(move_into_directive) {
let value = Util::str_extract_value(move_into_directive, line);
let Some((target, value)) = value.split_once("):") else {
continue;
};
let Some(moved) = selector(move_into_directive, value.trim()) else {
continue;
};
let target = target.trim();
if target.eq_ignore_ascii_case("body") {
move_into_body.push(moved);
} else if let Some(target) = selector(move_into_directive, target) {
move_into.push(MoveInto {
target,
selector: moved,
});
}
continue;
}
if line.starts_with(wrap_in_directive) {
let value = Util::str_extract_value(wrap_in_directive, line);
let Some((tag, rest)) = value.split_once(')') else {
continue;
};
let tag = tag.trim();
let Some(value) = rest.trim_start().strip_prefix(':') else {
continue;
};
if tag.is_empty() || !tag.chars().all(|c| c.is_ascii_alphanumeric() || c == '-') {
tracing::warn!(
config = name,
tag,
"Skipping wrap_in with an invalid tag name"
);
continue;
}
if let Some(selector) = selector(wrap_in_directive, value.trim()) {
wrap_in.push(WrapIn {
tag: tag.to_ascii_lowercase(),
selector,
});
}
continue;
}
if line.starts_with(http_header) {
let value = Util::str_extract_value(http_header, line);
let Some((name, value)) = value.split_once("):") else {
continue;
};
header_vec.push(Header {
name: name.trim().to_string(),
value: value.trim_start().to_string(),
});
continue;
}
if line.starts_with(find) {
let value = Util::str_extract_value(find, line).to_string();
if let Some(unpaired) = find_string.replace(value) {
tracing::warn!(
config = name,
find_string = unpaired,
"find_string without replace_string"
);
}
continue;
}
if line.starts_with(replace) {
let replace_with = Util::str_extract_value(replace, line).to_string();
match find_string.take() {
Some(to_replace) => replace_vec.push(Replace {
to_replace,
replace_with,
}),
None => tracing::warn!(
config = name,
replace_string = replace_with,
"replace_string without find_string"
),
}
continue;
}
tracing::debug!(config = name, line, "Unknown directive");
}
if let Some(unpaired) = find_string {
tracing::warn!(
config = name,
find_string = unpaired,
"find_string without replace_string"
);
}
let config = ConfigEntry {
title: titles,
author: authors,
date: dates,
body: bodies,
strip: strips,
strip_attr,
post_strip_attr,
dissolve,
wrap_in,
move_into,
move_into_body,
strip_id_or_class,
strip_image_src,
replace: replace_vec,
header: header_vec,
single_page_link,
next_page_link,
skip_json_ld,
insert_detected_image,
prune,
autodetect_on_failure,
native_ad_clue,
src_lazy_load_attr,
autodetect_next_page,
};
Ok(config)
}
}
const IGNORED_DIRECTIVES: &[&str] = &[
"tidy:",
"parser:",
"test_url:",
"test_contains:",
"convert_double_br_tags:",
"requires_login:",
"login_uri:",
"login_username_field:",
"login_password_field:",
"login_extra_fields:",
"not_logged_in_xpath:",
"single_page_link_in_feed:",
"footnotes:",
"strip_comments:",
"skip_id_or_class:",
];
fn yes_no(config: &str, line: &str, directive: &str) -> Option<bool> {
let value = Util::str_extract_value(directive, line);
if value.eq_ignore_ascii_case("yes") || value.eq_ignore_ascii_case("true") {
Some(true)
} else if value.eq_ignore_ascii_case("no") || value.eq_ignore_ascii_case("false") {
Some(false)
} else {
let directive = directive.trim_end_matches(':');
tracing::warn!(config, directive, value, "Expected yes or no");
None
}
}
#[cfg(test)]
mod tests {
use super::ConfigEntry;
use crate::selector::XPathSyntax;
use std::borrow::Cow;
async fn parse(config: &'static str) -> ConfigEntry {
ConfigEntry::parse_data(Cow::Borrowed(config.as_bytes()), "test.txt", &XPathSyntax)
.await
.unwrap()
}
#[tokio::test]
async fn invalid_selectors_are_skipped() {
let config = parse(
"title: //h1\n\
title: contains(@class='x')\n\
author: //a[@rel='author'] | //span[@class='by']\n\
strip: //div[\n\
strip: //aside\n\
next_page_link: //a[@rel='next']\n\
next_page_link: ///a\n",
)
.await;
let strs = |selectors: &[crate::selector::Selector]| {
selectors
.iter()
.map(|s| s.as_str().to_string())
.collect::<Vec<_>>()
};
let links = |links: &[super::PageLink]| {
links
.iter()
.map(|l| l.selector.as_str().to_string())
.collect::<Vec<_>>()
};
assert_eq!(strs(&config.title), ["//h1"]);
assert_eq!(
strs(&config.author),
["//a[@rel='author'] | //span[@class='by']"]
);
assert_eq!(strs(&config.strip), ["//aside"]);
assert_eq!(links(&config.next_page_link), ["//a[@rel='next']"]);
}
#[tokio::test]
async fn if_page_contains() {
let condition = |link: &super::PageLink| {
link.if_page_contains
.as_ref()
.map(|c| c.as_str().to_string())
};
let config = parse(
"single_page_link: //a[@class='a']\n\
single_page_link: //a[@class='b']\n\
next_page_link: //a[@rel='next']\n\
if_page_contains: //div[@id='x']\n",
)
.await;
assert_eq!(condition(&config.single_page_link[0]), None);
assert_eq!(
condition(&config.single_page_link[1]).as_deref(),
Some("//div[@id='x']")
);
assert_eq!(condition(&config.next_page_link[0]), None);
let config = parse("next_page_link: //a\nif_page_contains: //nav\n").await;
assert_eq!(
condition(&config.next_page_link[0]).as_deref(),
Some("//nav")
);
}
#[tokio::test]
async fn yes_no_directives() {
assert_eq!(parse("skip_json_ld: yes\n").await.skip_json_ld, Some(true));
assert_eq!(parse("skip_json_ld:no \n").await.skip_json_ld, Some(false));
assert_eq!(parse("skip_json_ld: maybe\n").await.skip_json_ld, None);
assert_eq!(parse("prune: false\n").await.prune, Some(false));
assert_eq!(parse("prune:no\n").await.prune, Some(false));
assert_eq!(parse("title: //h1\n").await.skip_json_ld, None);
assert_eq!(
parse("insert_detected_image: no \n")
.await
.insert_detected_image,
Some(false)
);
}
#[tokio::test]
async fn move_into() {
let config = parse(
"move_into(//div[@id='intro']): //div[@id='body']\n\
move_into(body)://div[contains(@id,'photo')]\n\
move_into(//p): (//img)[1]\n",
)
.await;
let moves = config
.move_into
.iter()
.map(|m| (m.target.as_str(), m.selector.as_str()))
.collect::<Vec<_>>();
assert_eq!(
moves,
[
("//div[@id='intro']", "//div[@id='body']"),
("//p", "(//img)[1]")
]
);
assert_eq!(
config
.move_into_body
.iter()
.map(|s| s.as_str())
.collect::<Vec<_>>(),
["//div[contains(@id,'photo')]"]
);
}
#[tokio::test]
async fn wrap_in() {
let config = parse(
"wrap_in(blockquote): //div[contains(@class, 'quote')]\n\
wrap_in(i)://p[@class='bio']\n\
wrap_in(b c): //p\n\
wrap_in(): //p\n",
)
.await;
let wraps = config
.wrap_in
.iter()
.map(|w| (w.tag.as_str(), w.selector.as_str()))
.collect::<Vec<_>>();
assert_eq!(
wraps,
[
("blockquote", "//div[contains(@class, 'quote')]"),
("i", "//p[@class='bio']"),
]
);
}
#[tokio::test]
async fn find_and_replace_string() {
let config = parse(
"find_string: <p class=\"x\">\n\
replace_string: <p>\n\
find_string: foo\n\
# c\n\
title: //h1\n\
replace_string: bar\n\
replace_string: no find\n\
find_string: never replaced\n\
find_string: ü\n\
replace_string: ö\n\
find_string: unpaired at the end\n",
)
.await;
let replaces = config
.replace
.iter()
.map(|r| (r.to_replace.as_str(), r.replace_with.as_str()))
.collect::<Vec<_>>();
assert_eq!(
replaces,
[("<p class=\"x\">", "<p>"), ("foo", "bar"), ("ü", "ö")]
);
assert_eq!(config.title.len(), 1);
}
#[tokio::test]
async fn replace_string_and_http_header() {
let config = parse(
"replace_string(<noscript>): <div>\n\
replace_string(data-src=\"):src=\"\n\
replace_string( - Site name):\n\
replace_string(): x\n\
replace_string(no colon)\n\
http_header(user-agent): Mozilla/5.0 (X11)\n\
http_header(referer):https://example.com/\n",
)
.await;
let replaces = config
.replace
.iter()
.map(|r| (r.to_replace.as_str(), r.replace_with.as_str()))
.collect::<Vec<_>>();
assert_eq!(
replaces,
[
("<noscript>", "<div>"),
("data-src=\"", "src=\""),
(" - Site name", ""),
]
);
let headers = config
.header
.iter()
.map(|h| (h.name.as_str(), h.value.as_str()))
.collect::<Vec<_>>();
assert_eq!(
headers,
[
("user-agent", "Mozilla/5.0 (X11)"),
("referer", "https://example.com/"),
]
);
}
}