pub mod javascript;
pub mod python;
pub mod sfc;
use crate::framework::{ManifestSignals, Setting};
use crate::model::{FileFacts, Import, ModuleId, ModuleTraits};
use crate::resolve::{ModuleIndex, PathAlias};
use std::path::{Path, PathBuf};
pub struct AnalyzeInput<'a> {
pub module: ModuleId,
pub format: &'a str,
pub path: &'a str,
pub source: &'a str,
}
pub trait Analyzer: Send + Sync {
fn language(&self) -> &'static str;
fn formats(&self) -> &'static [&'static str];
fn analyze(&self, input: &AnalyzeInput<'_>) -> FileFacts;
fn resolve(&self, specifier: &str, importer: &Path, index: &ModuleIndex) -> Option<ModuleId>;
fn normalize_import(&self, _import: &mut Import, _importer: &Path, _index: &ModuleIndex) {}
fn glob_targets(
&self,
_specifier: &str,
_importer: &Path,
_index: &ModuleIndex,
) -> Vec<ModuleId> {
Vec::new()
}
fn entry_globs(&self) -> &'static [&'static str] {
&[]
}
fn test_globs(&self) -> &'static [&'static str] {
&[]
}
fn is_self_starting(&self, source: &str) -> bool {
source.starts_with("#!")
}
fn manifests(&self) -> &'static [&'static str] {
&[]
}
fn manifest_entries(&self, _directory: &Path, _manifest: &str, _text: &str) -> Vec<PathBuf> {
Vec::new()
}
fn manifest_signals(&self, _manifest: &str, _text: &str) -> ManifestSignals {
ManifestSignals::default()
}
fn config_setting(&self, _config: &str, _text: &str, _key: &str) -> Option<Setting> {
None
}
fn alias_configs(&self) -> &'static [&'static str] {
&[]
}
fn path_aliases(&self, _directory: &Path, _config: &str, _text: &str) -> Vec<PathAlias> {
Vec::new()
}
fn import_roots(&self, _modules: &[PathBuf]) -> Vec<PathBuf> {
Vec::new()
}
fn module_traits(&self, _path: &str) -> ModuleTraits {
ModuleTraits::default()
}
}
pub static ANALYZERS: &[&dyn Analyzer] = &[&javascript::JsAnalyzer, &python::PythonAnalyzer];
pub fn analyzer_for(format: &str) -> Option<&'static dyn Analyzer> {
ANALYZERS
.iter()
.copied()
.find(|a| a.formats().contains(&format))
}
pub fn supports(format: &str) -> bool {
analyzer_for(format).is_some()
}
pub fn supported_formats() -> Vec<&'static str> {
let mut formats: Vec<&'static str> = ANALYZERS
.iter()
.flat_map(|a| a.formats().iter().copied())
.collect();
formats.sort_unstable();
formats
}
pub fn supported_extensions() -> Vec<&'static str> {
let formats = supported_formats();
let mut extensions: Vec<&'static str> = cpd_tokenizer::formats::SUPPORTED_FORMATS
.iter()
.filter(|entry| formats.contains(&entry.name))
.flat_map(|entry| entry.extensions.iter().copied())
.collect();
extensions.sort_unstable();
extensions.dedup();
extensions
}
pub fn is_source_path(path: &str) -> bool {
supported_extensions()
.iter()
.any(|extension| path.ends_with(&format!(".{extension}")))
}
pub fn is_identifier_like(text: &str) -> bool {
!text.is_empty()
&& text.len() <= 100
&& text.starts_with(|c: char| c.is_alphabetic() || c == '_' || c == '$')
&& text
.chars()
.all(|c| c.is_alphanumeric() || c == '_' || c == '$')
}
#[derive(Default)]
pub(crate) struct Quotes {
open: u8,
}
impl Quotes {
pub(crate) fn step(&mut self, bytes: &[u8], at: usize) -> Option<usize> {
let byte = bytes[at];
if self.open != 0 {
match byte {
b'\\' => return Some(2),
_ if byte == self.open => self.open = 0,
_ => {}
}
return Some(1);
}
if matches!(byte, b'"' | b'\'' | b'`') {
self.open = byte;
return Some(1);
}
None
}
}
pub(crate) fn outside_strings(bytes: &[u8], from: usize) -> impl Iterator<Item = (usize, u8)> + '_ {
let (mut quotes, mut at) = (Quotes::default(), from);
std::iter::from_fn(move || {
while at < bytes.len() {
match quotes.step(bytes, at) {
Some(step) => at += step,
None => {
at += 1;
return Some((at - 1, bytes[at - 1]));
}
}
}
None
})
}
pub(crate) fn skip_while(bytes: &[u8], from: usize, keep: impl Fn(u8) -> bool) -> usize {
let mut at = from;
while at < bytes.len() && keep(bytes[at]) {
at += 1;
}
at
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn quotes_step_over_strings_and_their_escapes() {
let structure = |text: &str| {
let (bytes, mut quotes, mut at, mut seen) =
(text.as_bytes(), Quotes::default(), 0, String::new());
while at < bytes.len() {
match quotes.step(bytes, at) {
Some(step) => at += step,
None => {
seen.push(bytes[at] as char);
at += 1;
}
}
}
seen
};
assert_eq!(structure(r#"a("x,y", 'z')"#), "a(, )");
assert_eq!(structure(r#"["a\"b", c]"#), "[, c]");
assert_eq!(structure(r#"["a\\", c]"#), "[, c]");
assert_eq!(structure("`${x}` + y"), " + y");
}
#[test]
fn skip_while_stops_at_the_first_byte_it_does_not_keep() {
let bytes = b" name=1";
assert_eq!(skip_while(bytes, 0, |b| b == b' '), 2);
assert_eq!(skip_while(bytes, 2, |b| b.is_ascii_alphabetic()), 6);
assert_eq!(
skip_while(bytes, 8, |_| true),
8,
"at the end stays at the end"
);
}
#[test]
fn every_supported_format_has_exactly_one_analyzer() {
let formats = supported_formats();
let mut seen = std::collections::HashSet::new();
for f in &formats {
assert!(seen.insert(*f), "format {f} is claimed by two analyzers");
assert!(supports(f), "{f} must resolve");
}
}
#[test]
fn every_analyzer_format_is_one_the_walker_knows() {
let known = cpd_tokenizer::formats::list_formats();
for analyzer in ANALYZERS {
for format in analyzer.formats() {
assert!(
known.contains(format),
"{} claims format {format:?}, which cpd_tokenizer::formats does not define",
analyzer.language()
);
}
}
}
#[test]
fn language_ids_are_stable_lower_case_and_distinct() {
let mut ids: Vec<&str> = ANALYZERS.iter().map(|a| a.language()).collect();
for id in &ids {
assert!(
!id.is_empty() && id.chars().all(|c| c.is_ascii_lowercase()),
"{id:?} is not a lower-case ascii id"
);
}
let before = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), before, "two analyzers share a language id");
}
#[test]
fn supported_extensions_come_from_the_format_table() {
let extensions = supported_extensions();
for want in ["ts", "tsx", "js", "mjs", "py", "pyi"] {
assert!(
extensions.contains(&want),
"{want} missing from {extensions:?}"
);
}
assert!(!extensions.contains(&"java"));
assert!(is_source_path("src/a.tsx"));
assert!(is_source_path("pkg/mod.py"));
assert!(!is_source_path("README.md"));
}
#[test]
fn identifier_shaped_strings_are_told_from_prose() {
for yes in ["handleRoot", "_private", "$el", "a1", "run_phase"] {
assert!(is_identifier_like(yes), "{yes}");
}
for no in [
"",
"1abc",
"has space",
"path/to/file",
"kebab-case",
&"x".repeat(101),
] {
assert!(!is_identifier_like(no), "{no:?}");
}
}
#[test]
fn known_formats_map_to_the_right_language() {
for f in ["javascript", "typescript", "jsx", "tsx"] {
assert_eq!(analyzer_for(f).map(|a| a.language()), Some("js"), "{f}");
}
assert_eq!(analyzer_for("python").map(|a| a.language()), Some("python"));
assert_eq!(analyzer_for("ruby").map(|a| a.language()), None);
}
#[test]
fn the_trait_defaults_mean_nothing_special() {
struct Minimal;
impl Analyzer for Minimal {
fn language(&self) -> &'static str {
"minimal"
}
fn formats(&self) -> &'static [&'static str] {
&["minimal"]
}
fn analyze(&self, _: &AnalyzeInput<'_>) -> FileFacts {
FileFacts::default()
}
fn resolve(&self, _: &str, _: &Path, _: &ModuleIndex) -> Option<ModuleId> {
None
}
}
let m = Minimal;
assert!(m.entry_globs().is_empty());
assert!(m.test_globs().is_empty());
assert!(m.manifests().is_empty());
assert!(m.is_self_starting("#!/usr/bin/env thing\n"));
assert!(!m.is_self_starting("plain source\n"));
assert_eq!(m.module_traits("any/file.minimal"), ModuleTraits::default());
let index = ModuleIndex::new(vec![PathBuf::from("/p")]);
assert!(m.manifest_entries(Path::new("/p"), "x.toml", "").is_empty());
assert!(m.manifest_signals("x.toml", "").dependencies.is_empty());
assert!(m.config_setting("x.config.toml", "", "key").is_none());
assert!(
m.resolve("./x", Path::new("/p/a.minimal"), &index)
.is_none()
);
}
}