use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::{Arc, Mutex, MutexGuard, OnceLock};
use std::time::{Duration, Instant, SystemTime};
use crate::capture::{AnchorList, FlatKind, FlatTok, shares_previous_span};
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub(crate) struct Caps {
pub files: usize,
pub file_bytes: u64,
pub total_bytes: u64,
pub dirs: usize,
pub depth: usize,
}
impl Default for Caps {
fn default() -> Self {
Self {
files: 4096,
file_bytes: 4 << 20,
total_bytes: 32 << 20,
dirs: 4096,
depth: 32,
}
}
}
pub(crate) struct Search<'a> {
pub dir: &'a Path,
pub entry_names: &'a [&'a str],
pub caps: Caps,
}
pub(crate) struct Slice {
pub path: PathBuf,
pub text: String,
pub anchors: AnchorList,
pub line: usize,
pub column: usize,
}
impl Search<'_> {
pub(crate) fn find(
&self,
toks: &[FlatTok],
mut accept: impl FnMut(&Path) -> bool,
) -> Option<Slice> {
if toks.is_empty() {
return None;
}
let quoted = quoted_includes(toks);
let mut first: Option<Slice> = None;
let mut budget = self.caps.total_bytes;
let candidates = self.candidate_files();
for path in candidates.iter() {
let Ok(meta) = std::fs::metadata(path) else {
continue;
};
let len = meta.len();
if len > self.caps.file_bytes {
continue;
}
if len > budget {
break;
}
budget -= len;
let Some(file) = self.file(path, &meta) else {
continue;
};
let Some(sites) = file.sites.as_deref() else {
continue;
};
let text = file.text.as_str();
for wanted in [true, false] {
for site in sites {
if site.named(text, self.entry_names) != wanted {
continue;
}
let Some(found) = verify(text, site.body.clone(), toks) else {
continue;
};
if !accept(path) {
continue;
}
let (line, column) = line_column(text, found.start);
let slice = Slice {
path: path.clone(),
text: text[found.start..found.end].to_owned(),
anchors: found.anchors,
line,
column,
};
if headers_exist_beside(&slice.path, "ed) {
return Some(slice);
}
if first.is_none() {
first = Some(slice);
}
}
}
}
first
}
fn file(&self, path: &Path, meta: &std::fs::Metadata) -> Option<File> {
let stamp = (meta.modified().ok(), meta.len());
if let Some(hit) = cache().file(self.dir, path, stamp) {
return Some(hit);
}
#[cfg(test)]
counted(|counts| counts.reads += 1);
let text = std::fs::read_to_string(path).ok()?;
let file = File {
stamp,
sites: sites(&text).map(Arc::from),
text: Arc::new(text),
};
cache().remember_file(self.dir, path, &file, self.caps.total_bytes);
Some(file)
}
fn candidate_files(&self) -> Arc<[PathBuf]> {
if let Some(listed) = cache().listing(self.dir, &self.caps) {
return listed;
}
let mut out = Vec::new();
let mut dirs = 0usize;
self.walk(self.dir, 0, &mut dirs, &mut out);
out.sort();
let files: Arc<[PathBuf]> = Arc::from(out);
cache().remember_listing(self.dir, self.caps, &files);
files
}
fn walk(&self, dir: &Path, depth: usize, dirs: &mut usize, out: &mut Vec<PathBuf>) {
if depth > self.caps.depth || *dirs >= self.caps.dirs || out.len() >= self.caps.files {
return;
}
*dirs += 1;
#[cfg(test)]
counted(|counts| counts.dirs += 1);
let Ok(entries) = std::fs::read_dir(dir) else {
return;
};
let mut files = Vec::new();
let mut subdirs = Vec::new();
for entry in entries.flatten() {
let name = entry.file_name();
let Some(name) = name.to_str() else {
continue;
};
if name.starts_with('.') {
continue;
}
let Ok(kind) = entry.file_type() else {
continue;
};
if kind.is_dir() {
if name == "target" {
continue;
}
subdirs.push(entry.path());
} else if kind.is_file() && name.ends_with(".rs") {
files.push(entry.path());
}
}
files.sort();
subdirs.sort();
for file in files {
if out.len() >= self.caps.files {
return;
}
out.push(file);
}
for subdir in subdirs {
self.walk(&subdir, depth + 1, dirs, out);
}
}
}
const LISTING_WINDOW: Duration = Duration::from_secs(2);
const DIRS: usize = 8;
type Stamp = (Option<SystemTime>, u64);
#[derive(Clone)]
struct File {
stamp: Stamp,
text: Arc<String>,
sites: Option<Arc<[Site]>>,
}
struct Listing {
files: Arc<[PathBuf]>,
caps: Caps,
at: Instant,
}
struct Dir {
dir: PathBuf,
listing: Option<Listing>,
files: HashMap<PathBuf, File>,
bytes: u64,
used: Instant,
}
#[derive(Default)]
struct Cache {
dirs: Vec<Dir>,
bytes: u64,
}
static CACHE: OnceLock<Mutex<Cache>> = OnceLock::new();
fn cache() -> MutexGuard<'static, Cache> {
CACHE
.get_or_init(|| Mutex::new(Cache::default()))
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
}
impl Cache {
fn dir(&mut self, dir: &Path) -> &mut Dir {
if let Some(at) = self.dirs.iter().position(|entry| entry.dir == dir) {
let entry = &mut self.dirs[at];
entry.used = Instant::now();
return entry;
}
while self.dirs.len() >= DIRS {
self.forget_oldest();
}
self.dirs.push(Dir {
dir: dir.to_owned(),
listing: None,
files: HashMap::new(),
bytes: 0,
used: Instant::now(),
});
self.dirs.last_mut().expect("the entry just pushed")
}
fn forget_oldest(&mut self) {
let Some(at) = (0..self.dirs.len()).min_by_key(|at| self.dirs[*at].used) else {
return;
};
let gone = self.dirs.swap_remove(at);
self.bytes -= gone.bytes;
}
fn listing(&mut self, dir: &Path, caps: &Caps) -> Option<Arc<[PathBuf]>> {
let listing = self.dir(dir).listing.as_ref()?;
(listing.caps == *caps && listing.at.elapsed() < LISTING_WINDOW)
.then(|| Arc::clone(&listing.files))
}
fn remember_listing(&mut self, dir: &Path, caps: Caps, files: &Arc<[PathBuf]>) {
self.dir(dir).listing = Some(Listing {
files: Arc::clone(files),
caps,
at: Instant::now(),
});
}
fn file(&mut self, dir: &Path, path: &Path, stamp: Stamp) -> Option<File> {
let file = self.dir(dir).files.get(path)?;
(file.stamp == stamp).then(|| file.clone())
}
fn remember_file(&mut self, dir: &Path, path: &Path, file: &File, cap: u64) {
let len = file.text.len() as u64;
let entry = self.dir(dir);
let was = entry
.files
.insert(path.to_owned(), file.clone())
.map_or(0, |old| old.text.len() as u64);
entry.bytes = entry.bytes + len - was;
self.bytes = self.bytes + len - was;
if self.bytes > cap {
self.dirs.clear();
self.bytes = 0;
}
}
}
#[cfg(test)]
fn expire_listing(dir: &Path) {
let mut cache = cache();
let entry = cache.dir(dir);
entry.listing = entry.listing.take().and_then(|listing| {
Instant::now()
.checked_sub(LISTING_WINDOW)
.map(|then| Listing {
at: then,
..listing
})
});
}
#[cfg(test)]
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
struct Counts {
dirs: usize,
reads: usize,
}
#[cfg(test)]
thread_local! {
static COUNTS: std::cell::Cell<Counts> = const { std::cell::Cell::new(Counts { dirs: 0, reads: 0 }) };
}
#[cfg(test)]
fn counted(f: impl FnOnce(&mut Counts)) {
COUNTS.with(|counts| {
let mut now = counts.get();
f(&mut now);
counts.set(now);
});
}
#[cfg(test)]
impl Counts {
fn now() -> Self {
COUNTS.with(std::cell::Cell::get)
}
fn since(self) -> Self {
let now = Self::now();
Self {
dirs: now.dirs - self.dirs,
reads: now.reads - self.reads,
}
}
}
fn quoted_includes(toks: &[FlatTok]) -> Vec<String> {
if let [only] = toks
&& only.kind() == FlatKind::Literal
&& let Some(text) = crate::capture::string_literal_text(only.text())
{
return quoted_includes_in_text(&text);
}
let mut out = Vec::new();
for window in toks.windows(3) {
let [hash, name, operand] = window else {
continue;
};
if hash.kind() != FlatKind::Punct || hash.text() != "#" {
continue;
}
if name.kind() != FlatKind::Ident || !matches!(name.text(), "include" | "include_next") {
continue;
}
if operand.kind() != FlatKind::Literal {
continue;
}
if let Some(header) = crate::capture::string_literal_text(operand.text()) {
out.push(header);
}
}
out
}
fn quoted_includes_in_text(text: &str) -> Vec<String> {
let mut out = Vec::new();
for line in text.lines() {
let line = line.trim_start();
let Some(rest) = line.strip_prefix('#') else {
continue;
};
let rest = rest.trim_start();
let rest = rest
.strip_prefix("include_next")
.or_else(|| rest.strip_prefix("include"))
.map(str::trim_start);
let Some(rest) = rest else {
continue;
};
if let Some(inner) = rest.strip_prefix('"')
&& let Some(end) = inner.find('"')
{
out.push(inner[..end].to_owned());
}
}
out
}
fn headers_exist_beside(path: &Path, quoted: &[String]) -> bool {
if quoted.is_empty() {
return true;
}
let Some(dir) = path.parent() else {
return false;
};
quoted.iter().all(|header| dir.join(header).is_file())
}
fn line_column(text: &str, at: usize) -> (usize, usize) {
let before = &text[..at];
let line = before.bytes().filter(|b| *b == b'\n').count() + 1;
let start = before.rfind('\n').map_or(0, |i| i + 1);
(line, before[start..].chars().count())
}
struct Site {
body: std::ops::Range<usize>,
name: std::ops::Range<usize>,
}
impl Site {
fn named(&self, text: &str, names: &[&str]) -> bool {
names.contains(&&text[self.name.clone()])
}
}
fn sites(text: &str) -> Option<Vec<Site>> {
let bytes = text.as_bytes();
let mut out: Vec<Site> = Vec::new();
let mut open: Vec<(usize, u8, Option<std::ops::Range<usize>>)> = Vec::new();
let mut ident: Option<std::ops::Range<usize>> = None;
let mut bang: Option<std::ops::Range<usize>> = None;
let mut at = 0usize;
while at < bytes.len() {
let b = bytes[at];
if b.is_ascii_whitespace() {
at += 1;
continue;
}
if b == b'/' && bytes.get(at + 1) == Some(&b'/') {
at = bytes[at..]
.iter()
.position(|c| *c == b'\n')
.map_or(bytes.len(), |i| at + i + 1);
continue;
}
if b == b'/' && bytes.get(at + 1) == Some(&b'*') {
at = block_comment_end(bytes, at)?;
continue;
}
if b == b'"' {
at = string_end(bytes, at)?;
ident = None;
bang = None;
continue;
}
if b == b'\'' {
at = char_or_lifetime_end(text, at)?;
ident = None;
bang = None;
continue;
}
if is_ident_start(b) {
let end = ident_end(bytes, at);
let word = &text[at..end];
if let Some(after) = raw_string_end(bytes, end, word) {
at = after;
ident = None;
bang = None;
continue;
}
if matches!(word, "b" | "c") && bytes.get(end) == Some(&b'"') {
at = string_end(bytes, end)?;
ident = None;
bang = None;
continue;
}
ident = Some(at..end);
bang = None;
at = end;
continue;
}
if b == b'!' {
bang = ident.take();
at += 1;
continue;
}
if let Some(close) = closing_delimiter(b) {
open.push((at + 1, close, bang.take()));
ident = None;
at += 1;
continue;
}
if matches!(b, b')' | b']' | b'}') {
let (start, close, name) = open.pop()?;
if close != b {
return None;
}
if let Some(name) = name {
out.push(Site {
body: start..at,
name,
});
}
ident = None;
bang = None;
at += 1;
continue;
}
ident = None;
bang = None;
at += 1;
}
out.sort_by_key(|site| site.body.start);
Some(out)
}
fn closing_delimiter(b: u8) -> Option<u8> {
match b {
b'(' => Some(b')'),
b'[' => Some(b']'),
b'{' => Some(b'}'),
_ => None,
}
}
fn is_ident_start(b: u8) -> bool {
b.is_ascii_alphabetic() || b == b'_' || b >= 0x80
}
fn is_ident_continue(b: u8) -> bool {
b.is_ascii_alphanumeric() || b == b'_' || b >= 0x80
}
fn ident_end(bytes: &[u8], at: usize) -> usize {
let mut end = at + 1;
while end < bytes.len() && is_ident_continue(bytes[end]) {
end += 1;
}
end
}
fn block_comment_end(bytes: &[u8], at: usize) -> Option<usize> {
let mut depth = 0usize;
let mut i = at;
while i + 1 < bytes.len() {
match (bytes[i], bytes[i + 1]) {
(b'/', b'*') => {
depth += 1;
i += 2;
}
(b'*', b'/') => {
depth -= 1;
i += 2;
if depth == 0 {
return Some(i);
}
}
_ => i += 1,
}
}
None
}
fn string_end(bytes: &[u8], at: usize) -> Option<usize> {
let mut i = at + 1;
while i < bytes.len() {
match bytes[i] {
b'\\' => i += 2,
b'"' => return Some(i + 1),
_ => i += 1,
}
}
None
}
fn raw_string_end(bytes: &[u8], at: usize, prefix: &str) -> Option<usize> {
if !matches!(prefix, "r" | "br" | "cr") {
return None;
}
let mut i = at;
while bytes.get(i) == Some(&b'#') {
i += 1;
}
let hashes = i - at;
if bytes.get(i) != Some(&b'"') {
return None;
}
i += 1;
while i < bytes.len() {
if bytes[i] == b'"' {
let after = i + 1;
if bytes.len() >= after + hashes
&& bytes[after..after + hashes].iter().all(|b| *b == b'#')
{
return Some(after + hashes);
}
}
i += 1;
}
Some(bytes.len())
}
fn char_or_lifetime_end(text: &str, at: usize) -> Option<usize> {
let bytes = text.as_bytes();
let rest = text.get(at + 1..)?;
let mut chars = rest.char_indices();
let Some((_, first)) = chars.next() else {
return Some(bytes.len());
};
if first == '\\' {
let mut i = at + 1;
while i < bytes.len() {
match bytes[i] {
b'\\' => i += 2,
b'\'' => return Some(i + 1),
_ => i += 1,
}
}
return None;
}
if let Some((next, _)) = chars.next() {
if rest.as_bytes()[next] == b'\'' {
return Some(at + 1 + next + 1);
}
} else {
return Some(bytes.len());
}
if is_ident_start(bytes[at + 1]) {
return Some(ident_end(bytes, at + 1));
}
Some(at + 1)
}
struct Match {
start: usize,
end: usize,
anchors: AnchorList,
}
fn verify(text: &str, body: std::ops::Range<usize>, toks: &[FlatTok]) -> Option<Match> {
let bytes = text.as_bytes();
let mut at = body.start;
let mut anchors = AnchorList::with_capacity(toks.len());
let mut start = None;
let mut end = body.start;
let mut prev: Option<&FlatTok> = None;
for tok in toks {
if let Some(prev) = prev
&& shares_previous_span(prev, tok)
{
continue;
}
prev = Some(tok);
at = skip_trivia(bytes, at, body.end)?;
let spelling = tok.text();
let stop = at + spelling.len();
if stop > body.end || !text[at..body.end].starts_with(spelling) {
return None;
}
if matches!(tok.kind(), FlatKind::Ident | FlatKind::Literal)
&& bytes.get(stop).copied().is_some_and(is_ident_continue)
{
return None;
}
if start.is_none() {
start = Some(at);
}
let base = start.expect("the first token set the start");
anchors.push(((at - base) as u32, (stop - base) as u32, tok.span()));
at = stop;
end = stop;
}
let start = start?;
if skip_trivia(bytes, at, body.end)? != body.end {
return None;
}
if end.checked_sub(start)? > u32::MAX as usize {
return None;
}
Some(Match {
start,
end,
anchors,
})
}
fn skip_trivia(bytes: &[u8], mut at: usize, limit: usize) -> Option<usize> {
while at < limit {
if bytes[at].is_ascii_whitespace() {
at += 1;
continue;
}
if bytes[at] == b'/' && bytes.get(at + 1) == Some(&b'/') {
at = bytes[at..limit]
.iter()
.position(|c| *c == b'\n')
.map_or(limit, |i| at + i + 1);
continue;
}
if bytes[at] == b'/' && bytes.get(at + 1) == Some(&b'*') {
let end = block_comment_end(&bytes[..limit], at)?;
at = end;
continue;
}
break;
}
Some(at)
}
#[cfg(test)]
mod tests {
use super::*;
fn found(text: &str, names: &[&str]) -> Vec<String> {
let sites = sites(text).expect("the fixture scans");
let mut out = Vec::new();
for wanted in [true, false] {
for site in &sites {
if site.named(text, names) == wanted {
out.push(text[site.body.clone()].to_owned());
}
}
}
out
}
#[test]
fn a_site_is_an_identifier_a_bang_and_a_delimiter() {
assert_eq!(found("c99! { int x; }", &["c99"]), vec![" int x; "]);
assert_eq!(found("cinrs::c99!(int x;)", &["c99"]), vec!["int x;"]);
assert_eq!(found("c99![int x;]", &["c99"]), vec!["int x;"]);
assert_eq!(found("c99 /*x*/ ! { a }", &["c99"]), vec![" a "]);
}
#[test]
fn a_name_that_was_not_asked_for_is_only_a_bare_site() {
let text = "c99! { a }\nother! { b }";
assert_eq!(found(text, &["c99"]), vec![" a ", " b "]);
assert_eq!(found(text, &[]), vec![" a ", " b "]);
assert_eq!(
found("x99! { a }\nc99! { b }", &["c99"]),
vec![" b ", " a "]
);
}
#[test]
fn an_identifier_that_merely_ends_with_the_name_is_not_a_site() {
let text = "xc99! { a }";
assert_eq!(found(text, &["c99"]), vec![" a "]);
let sites = sites(text).expect("scans");
assert!(!sites[0].named(text, &["c99"]));
assert!(sites[0].named(text, &["xc99"]));
}
#[test]
fn a_comment_or_a_string_is_not_code() {
assert_eq!(found("// c99! { a }\nc99! { b }", &["c99"]), vec![" b "]);
assert_eq!(found("/* c99! { a } */ c99! { b }", &["c99"]), vec![" b "]);
assert_eq!(
found("/* /* c99! { a } */ */ c99! { b }", &["c99"]),
vec![" b "]
);
assert_eq!(
found(r#"let s = "c99! { a }"; c99! { b }"#, &["c99"]),
vec![" b "]
);
assert_eq!(
found(
r##"let s = r#"c99! { a } "unbalanced { "#; c99! { b }"##,
&["c99"]
),
vec![" b "]
);
assert_eq!(found(r#"let s = b"{"; c99! { b }"#, &["c99"]), vec![" b "]);
}
#[test]
fn a_lifetime_is_not_a_character_literal() {
assert_eq!(
found(
"fn f<'a>(x: &'a str) -> &'static str { x }\nc99! { a }",
&["c99"]
),
vec![" a "]
);
assert_eq!(
found("'outer: loop { break 'outer; }\nc99! { a }", &["c99"]),
vec![" a "]
);
assert_eq!(found("let c = '}'; c99! { a }", &["c99"]), vec![" a "]);
assert_eq!(found(r#"let c = '\''; c99! { a }"#, &["c99"]), vec![" a "]);
assert_eq!(found(r#"let c = '\\'; c99! { a }"#, &["c99"]), vec![" a "]);
assert_eq!(found("let c = b'{'; c99! { a }", &["c99"]), vec![" a "]);
}
#[test]
fn a_raw_identifier_is_not_a_raw_string() {
assert_eq!(found("let r#match = 1; c99! { a }", &["c99"]), vec![" a "]);
}
#[test]
fn nested_invocations_are_all_candidates() {
let text = "outer! { c99! { a } }";
assert_eq!(found(text, &["c99"]), vec![" a ", " c99! { a } "]);
}
#[test]
fn a_file_that_cannot_be_scanned_is_passed_over() {
assert!(sites("} c99! { a }").is_none());
assert!(sites("c99! ( a } ").is_none());
assert!(sites("/* c99! { a }").is_none());
}
#[test]
fn a_line_and_column_are_counted_the_way_a_span_counts_them() {
assert_eq!(line_column("abc", 0), (1, 0));
assert_eq!(line_column("ab\ncd", 4), (2, 1));
assert_eq!(line_column("// ★\nx", 8), (2, 1));
assert_eq!(line_column("★x", "★".len()), (1, 1));
}
use std::str::FromStr;
use proc_macro2::TokenStream;
use crate::capture::flat_tokens;
struct Crate {
dir: PathBuf,
}
impl Crate {
fn new(name: &str) -> Self {
let dir =
std::env::temp_dir().join(format!("cinrs-locate-{}-{name}", std::process::id()));
std::fs::remove_dir_all(&dir).ok();
std::fs::create_dir_all(&dir).expect("a temporary directory");
Self { dir }
}
fn file(&self, name: &str, text: &str) -> &Self {
let path = self.dir.join(name);
if let Some(parent) = path.parent() {
std::fs::create_dir_all(parent).expect("a temporary directory");
}
std::fs::write(path, text).expect("a writable temporary file");
self
}
fn find(&self, input: &str) -> Option<Slice> {
self.find_with(input, Caps::default())
}
fn find_with(&self, input: &str, caps: Caps) -> Option<Slice> {
let toks = flat_tokens(TokenStream::from_str(input).expect("the input lexes"));
Search {
dir: &self.dir,
entry_names: &["c99"],
caps,
}
.find(&toks, |_| true)
}
}
impl Drop for Crate {
fn drop(&mut self) {
std::fs::remove_dir_all(&self.dir).ok();
}
}
fn found_in(krate: &Crate, slice: &Slice) -> String {
slice
.path
.strip_prefix(&krate.dir)
.expect("the match is inside the crate")
.to_string_lossy()
.replace('\\', "/")
}
#[test]
fn the_invocation_is_found_among_decoys() {
let body = "int fact(int n) { return n == 0 ? 1 : n * fact(n - 1); }";
let krate = Crate::new("decoys");
krate
.file(
"src/other.rs",
"cinrs::c99! { int g(void) { return 2; } }\n",
)
.file(
"src/comment.rs",
&format!("// cinrs::c99! {{ {body} }}\n/* c99! {{ {body} }} */\n"),
)
.file(
"src/string.rs",
&format!("const S: &str = r#\"c99! {{ {body} }}\"#;\n"),
)
.file(
"src/prefix.rs",
"c99! { int fact(int n) { return n == 0 ? 1 : n }\n",
)
.file(
"src/real.rs",
&format!(
"fn f<'a>(s: &'a str) -> char {{ 'outer: loop {{ break 'outer }} '}}' }}\n\
const R: &str = r##\"c99! {{ }} \"#\"##;\n\
const C: char = '\\'';\n\
mod inner {{\n cinrs::c99! {{ {body} }}\n}}\n"
),
);
let slice = krate.find(body).expect("the invocation is found");
assert_eq!(found_in(&krate, &slice), "src/real.rs");
assert_eq!(slice.text, body);
assert_eq!((slice.line, slice.column), (5, 18));
let tokens = flat_tokens(TokenStream::from_str(body).expect("lexes")).len();
assert_eq!(slice.anchors.len(), tokens);
let mut last = 0;
for (start, end, _) in &slice.anchors {
assert!(last <= *start && start <= end && *end as usize <= slice.text.len());
last = *end;
}
}
#[test]
fn a_renamed_import_is_found_without_the_name() {
let krate = Crate::new("renamed");
krate.file(
"src/lib.rs",
"use cinrs::c99 as compile_c;\ncompile_c! { int x; }\n",
);
let slice = krate.find("int x;").expect("found by the bare search");
assert_eq!(slice.text, "int x;");
}
#[test]
fn every_delimiter_holds_a_body() {
for (open, close) in [('{', '}'), ('(', ')'), ('[', ']')] {
let krate = Crate::new(&format!("delim-{open}"));
krate.file("src/lib.rs", &format!("c99!{open} int x; {close}\n"));
assert_eq!(
krate.find("int x;").expect("found").text,
"int x;",
"{open}{close}"
);
}
}
#[test]
fn a_body_that_differs_from_the_tokens_is_not_matched() {
let krate = Crate::new("unsaved");
krate.file("src/lib.rs", "c99! { int x = 1; }\n");
assert!(krate.find("int x = 2;").is_none());
assert!(krate.find("int x = 1").is_none());
assert!(krate.find("int x = 1;;").is_none());
krate.file("src/lib.rs", "c99! { int xs; }\n");
assert!(krate.find("int x;").is_none());
krate.file("src/lib.rs", "c99! { int x = 10UL; }\n");
assert!(krate.find("int x = 10;").is_none());
assert!(krate.find("int x = 10UL;").is_some());
}
#[test]
fn comments_inside_the_body_are_skipped_and_kept() {
let krate = Crate::new("comments");
krate.file(
"src/lib.rs",
"c99! {\n int x; // a comment\n /* and\n another */ int y;\n}\n",
);
let slice = krate.find("int x; int y;").expect("found");
assert_eq!(
slice.text,
"int x; // a comment\n /* and\n another */ int y;"
);
}
#[test]
fn the_first_of_several_matches_wins_deterministically() {
let krate = Crate::new("several");
krate
.file("src/a.rs", "c99! { int x; }\n")
.file("src/b.rs", "c99! { int x; }\n")
.file("lib.rs", "c99! { int x; }\n");
assert_eq!(
found_in(&krate, &krate.find("int x;").expect("found")),
"lib.rs"
);
}
#[test]
fn the_headers_a_unit_includes_by_name_are_read_off_its_tokens() {
let toks = flat_tokens(
TokenStream::from_str("#include \"point.h\"\n#include <stdio.h>\nint x;")
.expect("lexes"),
);
assert_eq!(quoted_includes(&toks), vec!["point.h".to_owned()]);
let literal = flat_tokens(
TokenStream::from_str("r#\"\n #include \"a/b.h\"\n#include <stdio.h>\n\"#")
.expect("lexes"),
);
assert_eq!(quoted_includes(&literal), vec!["a/b.h".to_owned()]);
let plain = flat_tokens(TokenStream::from_str("int x;").expect("lexes"));
assert!(quoted_includes(&plain).is_empty());
}
#[test]
fn the_candidate_whose_directory_holds_the_header_is_preferred() {
let krate = Crate::new("prefer");
std::fs::create_dir_all(krate.dir.join("src/with")).expect("a temporary directory");
krate
.file("src/a.rs", "c99! { #include \"h.h\"\nint x; }\n")
.file("src/with/b.rs", "c99! { #include \"h.h\"\nint x; }\n")
.file("src/with/h.h", "int declared(void);\n");
let slice = krate
.find("#include \"h.h\"\nint x;")
.expect("the invocation is found");
assert_eq!(found_in(&krate, &slice), "src/with/b.rs");
std::fs::remove_file(krate.dir.join("src/with/h.h")).expect("a removable file");
let slice = krate
.find("#include \"h.h\"\nint x;")
.expect("the invocation is found");
assert_eq!(found_in(&krate, &slice), "src/a.rs");
}
#[test]
fn a_candidate_the_caller_refuses_is_passed_over() {
let krate = Crate::new("accept");
krate
.file("src/a.rs", "include_c99!(\"x.c\");\n")
.file("src/b.rs", "include_c99!(\"x.c\");\n");
let toks = flat_tokens(TokenStream::from_str("\"x.c\"").expect("lexes"));
let search = Search {
dir: &krate.dir,
entry_names: &["include_c99"],
caps: Caps::default(),
};
let slice = search
.find(&toks, |path| path.ends_with("b.rs"))
.expect("the second is accepted");
assert_eq!(found_in(&krate, &slice), "src/b.rs");
}
#[test]
fn target_and_hidden_directories_are_not_searched() {
let krate = Crate::new("skipped");
krate
.file("target/debug/build/generated.rs", "c99! { int x; }\n")
.file(".hidden/lib.rs", "c99! { int x; }\n");
assert!(krate.find("int x;").is_none());
}
#[test]
fn the_caps_stop_the_search() {
let krate = Crate::new("caps");
krate
.file(
"src/a.rs",
&format!("// {}\nc99! {{ int y; }}\n", "pad".repeat(64)),
)
.file("src/b.rs", "c99! { int x; }\n");
assert!(krate.find("int x;").is_some());
let one_file = Caps {
files: 1,
..Caps::default()
};
assert!(krate.find_with("int x;", one_file).is_none());
let few_bytes = Caps {
total_bytes: 100,
..Caps::default()
};
assert!(krate.find_with("int x;", few_bytes).is_none());
let small_files = Caps {
file_bytes: 20,
..Caps::default()
};
assert!(krate.find_with("int x;", small_files).is_some());
let no_dirs = Caps {
dirs: 0,
..Caps::default()
};
assert!(krate.find_with("int x;", no_dirs).is_none());
}
static WATCHING: Mutex<()> = Mutex::new(());
fn watching() -> MutexGuard<'static, ()> {
WATCHING
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
}
#[test]
fn a_burst_of_searches_walks_the_crate_once() {
let _watching = watching();
let krate = Crate::new("burst");
krate
.file("src/lib.rs", "c99! { int x; }\n")
.file("src/other.rs", "// nothing to find here\n");
let before = Counts::now();
assert!(krate.find("int x;").is_some());
assert_eq!(
before.since(),
Counts { dirs: 2, reads: 1 },
"the first search does the work"
);
let before = Counts::now();
for _ in 0..16 {
assert!(krate.find("int x;").is_some());
}
assert_eq!(
before.since(),
Counts::default(),
"the rest of the burst is served from memory"
);
}
#[test]
fn a_file_saved_again_is_read_again() {
let krate = Crate::new("edited");
krate.file("src/lib.rs", "c99! { int x = 1; }\n");
assert!(krate.find("int x = 1;").is_some());
krate.file("src/lib.rs", "c99! { int x = 22; }\n");
assert!(krate.find("int x = 22;").is_some());
assert!(krate.find("int x = 1;").is_none());
}
#[test]
fn a_new_file_becomes_a_candidate_when_the_crate_is_walked_again() {
let krate = Crate::new("listed");
krate.file("src/a.rs", "c99! { int x; }\n");
assert!(krate.find("int x;").is_some());
let written = Instant::now();
krate.file("src/b.rs", "c99! { int y; }\n");
let missed = krate.find("int y;").is_none();
if written.elapsed() < LISTING_WINDOW {
assert!(missed, "the listing is the one from before the file");
}
expire_listing(&krate.dir);
assert!(krate.find("int y;").is_some());
}
#[test]
fn everything_is_forgotten_when_the_ceiling_is_passed() {
let _watching = watching();
let caps = Caps {
total_bytes: 600,
..Caps::default()
};
let pad = "pad".repeat(100);
let a = Crate::new("ceiling-a");
a.file("src/lib.rs", &format!("// {pad}\nc99! {{ int x; }}\n"));
let b = Crate::new("ceiling-b");
b.file("src/lib.rs", &format!("// {pad}\nc99! {{ int y; }}\n"));
assert!(a.find_with("int x;", caps).is_some());
assert!(b.find_with("int y;", caps).is_some());
let before = Counts::now();
assert!(a.find_with("int x;", caps).is_some());
let again = before.since();
assert!(again.reads > 0, "the text was thrown away: {again:?}");
assert!(again.dirs > 0, "and so was the listing: {again:?}");
}
}