use std::ffi::OsStr;
use std::fs::File;
use std::io::{BufRead, BufReader, Lines};
use std::path::{Path, PathBuf};
use anyhow::{anyhow, Context, Result};
use regex::{Regex, RegexBuilder};
const GIT_SCISSORS: &str = "# ------------------------ >8 ------------------------";
lazy_static! {
static ref TOKEN_RE: Regex = RegexBuilder::new(
r"
(
\w | # letters .. or:
[ \\ + . / : = ? @ _ ~ ' - ] # all possible chars found in an URL, an email,
# or a word with an apostrophe (like doesn't)
)+
"
).ignore_whitespace(true).build().expect("syntax error in static regex");
static ref ABBREV_RE: Regex = RegexBuilder::new(
r"
^
(\p{Lu}+) # Some upper case letters
\p{Lu} # An uppercase letter
\p{Ll} # A lower case letter
"
)
.ignore_whitespace(true).build().expect("syntax error in static regex");
static ref CONSTANT_RE: Regex = RegexBuilder::new(
r"
# Only uppercase letters, except maybe a 's' at the end
^(\p{Lu}+) s ?$
"
)
.ignore_whitespace(true).build().expect("syntax error in static regex");
static ref HEXA_RE: Regex = RegexBuilder::new(
r"
# Only letter a to f and numbers, at list 5 in size
[a-f0-9]{5,}
"
).ignore_whitespace(true).build().expect("syntax error in static regex");
static ref IDENT_RE_DEFAULT: Regex = RegexBuilder::new(
r"
# A word is just a bunch of unicode characters matching
# the Alphabetic group, possibly inside space escapes like
# \n or \t, and possibly containing exactly one apostrophe
(\\[nrt])*
(
\p{Alphabetic}+ ' \p{Alphabetic}+ | (\p{Alphabetic}+)
)
(\\[nrt])*
"
).ignore_whitespace(true).build().expect("syntax error in static regex");
static ref IDENT_RE_NO_C_ESCAPE: Regex = RegexBuilder::new(
r"
# Same as IDENT_RE, without handling \n, \r or \t
\p{Alphabetic}+ ' \p{Alphabetic}+ | (\p{Alphabetic}+)
"
).ignore_whitespace(true).build().expect("syntax error in static regex");
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum ExtractMode {
Default,
NoCEscape,
}
impl ExtractMode {
fn from_path_ext(p: &Path) -> Self {
if let Some(e) = p.extension() {
if e == "tex" {
return ExtractMode::NoCEscape;
}
}
ExtractMode::Default
}
}
pub struct TokenProcessor {
path: PathBuf,
extract_mode: ExtractMode,
}
impl TokenProcessor {
pub fn new(path: &Path) -> Self {
Self {
path: path.to_path_buf(),
extract_mode: ExtractMode::from_path_ext(path),
}
}
pub fn each_token<F>(&self, mut f: F) -> Result<()>
where
F: FnMut(&str, usize, usize) -> Result<()>,
{
let source = File::open(&self.path)
.with_context(|| format!("Could not open '{}' for reading", self.path.display()))?;
let lines = RelevantLines::new(source, self.path.file_name());
for (i, line) in lines.enumerate() {
let line = line
.map_err(|e| anyhow!("When reading line from '{}': {}", self.path.display(), e))?;
let tokenizer = Tokenizer::new(&line, self.extract_mode);
for (word, pos) in tokenizer {
f(word, i + 1, pos)?
}
}
Ok(())
}
}
struct RelevantLines {
lines: Lines<BufReader<File>>,
is_git_message: bool,
}
impl RelevantLines {
fn new(source: File, filename: Option<&OsStr>) -> Self {
let is_git_message = filename == Some(OsStr::new("COMMIT_EDITMSG"));
let reader = BufReader::new(source);
let lines = reader.lines();
Self {
lines,
is_git_message,
}
}
}
impl Iterator for RelevantLines {
type Item = Result<String, std::io::Error>;
fn next(&mut self) -> Option<<Self as Iterator>::Item> {
let line_result = self.lines.next()?;
match line_result {
e @ Err(_) => Some(e),
Ok(s) if self.is_git_message && s == GIT_SCISSORS => None,
x => Some(x),
}
}
}
struct Tokenizer<'a> {
input: &'a str,
pos: usize,
extract_mode: ExtractMode,
}
impl<'a> Tokenizer<'a> {
fn new(input: &'a str, extract_mode: ExtractMode) -> Self {
Self {
input,
pos: 0,
extract_mode,
}
}
}
impl<'a> Iterator for Tokenizer<'a> {
type Item = (&'a str, usize);
fn next(&mut self) -> Option<<Self as Iterator>::Item> {
loop {
let captures = TOKEN_RE.captures(&self.input[self.pos..])?;
let token_match = captures.get(0).unwrap();
let token = token_match.as_str();
let start = token_match.range().start;
let next_word = extract_word(token, self.extract_mode);
if let Some((w, pos)) = next_word {
let res = (w, self.pos + start + pos);
self.pos += start + pos + w.len();
return Some(res);
} else {
self.pos += start + token.len();
}
}
}
}
fn extract_word(token: &str, extract_mode: ExtractMode) -> Option<(&str, usize)> {
if token == "s" {
return None;
}
if token.contains("://") {
return None;
}
if token.contains('@') {
return None;
}
if HEXA_RE.find(token).is_some() {
return None;
}
let (captures, index) = match extract_mode {
ExtractMode::NoCEscape => (IDENT_RE_NO_C_ESCAPE.captures(token), 0),
ExtractMode::Default => (IDENT_RE_DEFAULT.captures(token), 2),
};
if let Some(captures) = captures {
let ident_match = captures.get(index).unwrap();
let pos = ident_match.start();
let ident = ident_match.as_str();
return word_from_ident(ident, pos);
}
None
}
fn word_from_ident(ident: &str, pos: usize) -> Option<(&str, usize)> {
let mut iter = ident.char_indices();
let (_, first_char) = iter.next().expect("empty ident");
if first_char.is_lowercase() {
if let Some(p) = ident.find(char::is_uppercase) {
return Some((&ident[..p], pos));
}
}
if first_char.is_uppercase() {
if let Some(captures) = CONSTANT_RE.captures(ident) {
let res = captures.get(1).unwrap().as_str();
return Some((res, pos));
}
if let Some(captures) = ABBREV_RE.captures(ident) {
let res = captures.get(1).unwrap().as_str();
return Some((res, pos));
}
let (second_pos, _) = match iter.next() {
None => return Some((ident, pos)),
Some(x) => x,
};
if let Some(next_upper) = (&ident[second_pos..]).find(char::is_uppercase) {
let res = &ident[..next_upper + second_pos];
return Some((res, pos));
}
}
Some((ident, pos))
}
#[cfg(test)]
mod tests;