pub mod javascript;
pub mod python;
use crate::model::{FileFacts, Import, ModuleId, ModuleTraits};
use crate::resolve::{ModuleIndex, PathAlias};
use std::path::{Path, PathBuf};
pub struct AnalyzeInput<'a> {
pub module: ModuleId,
pub format: &'a str,
pub path: &'a str,
pub source: &'a str,
}
pub trait Analyzer: Send + Sync {
fn language(&self) -> &'static str;
fn formats(&self) -> &'static [&'static str];
fn analyze(&self, input: &AnalyzeInput<'_>) -> FileFacts;
fn resolve(&self, specifier: &str, importer: &Path, index: &ModuleIndex) -> Option<ModuleId>;
fn normalize_import(&self, _import: &mut Import, _importer: &Path, _index: &ModuleIndex) {}
fn entry_globs(&self) -> &'static [&'static str] {
&[]
}
fn test_globs(&self) -> &'static [&'static str] {
&[]
}
fn is_self_starting(&self, source: &str) -> bool {
source.starts_with("#!")
}
fn manifests(&self) -> &'static [&'static str] {
&[]
}
fn manifest_entries(&self, _directory: &Path, _manifest: &str, _text: &str) -> Vec<PathBuf> {
Vec::new()
}
fn alias_configs(&self) -> &'static [&'static str] {
&[]
}
fn path_aliases(&self, _directory: &Path, _config: &str, _text: &str) -> Vec<PathAlias> {
Vec::new()
}
fn import_roots(&self, _modules: &[PathBuf]) -> Vec<PathBuf> {
Vec::new()
}
fn module_traits(&self, _path: &str) -> ModuleTraits {
ModuleTraits::default()
}
}
pub static ANALYZERS: &[&dyn Analyzer] = &[&javascript::JsAnalyzer, &python::PythonAnalyzer];
pub fn analyzer_for(format: &str) -> Option<&'static dyn Analyzer> {
ANALYZERS
.iter()
.copied()
.find(|a| a.formats().contains(&format))
}
pub fn supports(format: &str) -> bool {
analyzer_for(format).is_some()
}
pub fn supported_formats() -> Vec<&'static str> {
let mut formats: Vec<&'static str> = ANALYZERS
.iter()
.flat_map(|a| a.formats().iter().copied())
.collect();
formats.sort_unstable();
formats
}
pub fn supported_extensions() -> Vec<&'static str> {
let formats = supported_formats();
let mut extensions: Vec<&'static str> = cpd_tokenizer::formats::SUPPORTED_FORMATS
.iter()
.filter(|entry| formats.contains(&entry.name))
.flat_map(|entry| entry.extensions.iter().copied())
.collect();
extensions.sort_unstable();
extensions.dedup();
extensions
}
pub fn is_source_path(path: &str) -> bool {
supported_extensions()
.iter()
.any(|extension| path.ends_with(&format!(".{extension}")))
}
pub fn language_of(format: &str) -> Option<&'static str> {
analyzer_for(format).map(|a| a.language())
}
pub fn is_identifier_like(text: &str) -> bool {
!text.is_empty()
&& text.len() <= 100
&& text.starts_with(|c: char| c.is_alphabetic() || c == '_' || c == '$')
&& text
.chars()
.all(|c| c.is_alphanumeric() || c == '_' || c == '$')
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn every_supported_format_has_exactly_one_analyzer() {
let formats = supported_formats();
let mut seen = std::collections::HashSet::new();
for f in &formats {
assert!(seen.insert(*f), "format {f} is claimed by two analyzers");
assert!(supports(f), "{f} must resolve");
}
}
#[test]
fn every_analyzer_format_is_one_the_walker_knows() {
let known = cpd_tokenizer::formats::list_formats();
for analyzer in ANALYZERS {
for format in analyzer.formats() {
assert!(
known.contains(format),
"{} claims format {format:?}, which cpd_tokenizer::formats does not define",
analyzer.language()
);
}
}
}
#[test]
fn language_ids_are_stable_lower_case_and_distinct() {
let mut ids: Vec<&str> = ANALYZERS.iter().map(|a| a.language()).collect();
for id in &ids {
assert!(
!id.is_empty() && id.chars().all(|c| c.is_ascii_lowercase()),
"{id:?} is not a lower-case ascii id"
);
}
let before = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), before, "two analyzers share a language id");
}
#[test]
fn supported_extensions_come_from_the_format_table() {
let extensions = supported_extensions();
for want in ["ts", "tsx", "js", "mjs", "py", "pyi"] {
assert!(
extensions.contains(&want),
"{want} missing from {extensions:?}"
);
}
assert!(!extensions.contains(&"java"));
assert!(is_source_path("src/a.tsx"));
assert!(is_source_path("pkg/mod.py"));
assert!(!is_source_path("README.md"));
}
#[test]
fn identifier_shaped_strings_are_told_from_prose() {
for yes in ["handleRoot", "_private", "$el", "a1", "run_phase"] {
assert!(is_identifier_like(yes), "{yes}");
}
for no in [
"",
"1abc",
"has space",
"path/to/file",
"kebab-case",
&"x".repeat(101),
] {
assert!(!is_identifier_like(no), "{no:?}");
}
}
#[test]
fn known_formats_map_to_the_right_language() {
for f in ["javascript", "typescript", "jsx", "tsx"] {
assert_eq!(language_of(f), Some("js"), "{f}");
}
assert_eq!(language_of("python"), Some("python"));
assert_eq!(language_of("ruby"), None);
}
#[test]
fn the_trait_defaults_mean_nothing_special() {
struct Minimal;
impl Analyzer for Minimal {
fn language(&self) -> &'static str {
"minimal"
}
fn formats(&self) -> &'static [&'static str] {
&["minimal"]
}
fn analyze(&self, _: &AnalyzeInput<'_>) -> FileFacts {
FileFacts::default()
}
fn resolve(&self, _: &str, _: &Path, _: &ModuleIndex) -> Option<ModuleId> {
None
}
}
let m = Minimal;
assert!(m.entry_globs().is_empty());
assert!(m.test_globs().is_empty());
assert!(m.manifests().is_empty());
assert!(m.is_self_starting("#!/usr/bin/env thing\n"));
assert!(!m.is_self_starting("plain source\n"));
assert_eq!(m.module_traits("any/file.minimal"), ModuleTraits::default());
let index = ModuleIndex::new(vec![PathBuf::from("/p")]);
assert!(m.manifest_entries(Path::new("/p"), "x.toml", "").is_empty());
assert!(
m.resolve("./x", Path::new("/p/a.minimal"), &index)
.is_none()
);
}
}