basta 0.2.0

Dead code detection — unused files, exports, symbols and imports across JavaScript, TypeScript, Vue, Svelte, Astro and Python
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
//! Language plug-ins.
//!
//! One [`Analyzer`] per language family. It is the *only* place a language
//! is allowed to be special: the walker, the graph, the confidence model, the
//! classifier, the reporters and both CLIs are language-agnostic, and they
//! learn everything they need about a language through this trait.
//!
//! An analyzer answers four questions about its language, and nothing else
//! in the crate answers any of them:
//!
//! 1. **What does one file declare, import and read?** — [`Analyzer::analyze`]
//!    turns a source file into [`FileFacts`], strictly within that file.
//!    Resolution across files happens once, in [`crate::graph`], on facts
//!    from every language at the same time.
//! 2. **Which file does a specifier name?** — [`Analyzer::resolve`], given the
//!    index of every module in the scan.
//! 3. **Where does a program in this language start?** — [`Analyzer::entry_globs`],
//!    [`Analyzer::is_self_starting`], and the manifests it can read. What a
//!    *framework* starts is data, in [`crate::framework`]; the analyzer only
//!    reads the manifest and config files that table points at. A
//!    language whose projects rename their own import paths also declares
//!    [`Analyzer::alias_configs`].
//! 4. **What can be told about a file from its path?** — [`Analyzer::module_traits`].
//!
//! The first two are required. The rest have defaults that mean "nothing
//! special", so a minimal language is one file and one line in [`ANALYZERS`].
//! See `docs/basta-extending.md` for the walk-through.

pub mod javascript;
pub mod python;
pub mod sfc;

use crate::framework::{ManifestSignals, Setting};
use crate::model::{FileFacts, Import, ModuleId, ModuleTraits};
use crate::resolve::{ModuleIndex, PathAlias};
use std::path::{Path, PathBuf};

/// Everything an [`Analyzer`] needs about the file it is given.
pub struct AnalyzeInput<'a> {
    /// Id the emitted facts must carry.
    pub module: ModuleId,
    /// jscpd format name, so one analyzer can serve several dialects.
    pub format: &'a str,
    /// Scan-root-relative display path.
    pub path: &'a str,
    pub source: &'a str,
}

/// A language plug-in.
pub trait Analyzer: Send + Sync {
    /// Stable, lower-case id that appears in reports: `js`, `python`.
    fn language(&self) -> &'static str;

    /// jscpd format names this analyzer serves. Every one must exist in
    /// `cpd_tokenizer::formats` — that table is how the walker maps file
    /// extensions to formats, so a format it does not know is never walked.
    fn formats(&self) -> &'static [&'static str];

    /// Declarations, imports and references of one file. An analyzer that
    /// cannot parse its input returns [`FileFacts::unparsed`] rather than an
    /// error: one broken file must not fail the run, but the graph has to
    /// know its references are unknown.
    fn analyze(&self, input: &AnalyzeInput<'_>) -> FileFacts;

    /// The module a specifier names, as seen from `importer`, or `None` when
    /// it names something outside the scan (a dependency, the standard
    /// library). Only `index` may be consulted for what exists; the resolver
    /// must not invent files.
    fn resolve(&self, specifier: &str, importer: &Path, index: &ModuleIndex) -> Option<ModuleId>;

    /// A chance to rewrite an import before it is resolved, for languages
    /// where the written form is ambiguous. Python's `from pkg import x`
    /// names either a symbol in `pkg/__init__.py` or the module `pkg/x.py`,
    /// and only the index can tell which.
    fn normalize_import(&self, _import: &mut Import, _importer: &Path, _index: &ModuleIndex) {}

    /// Every module a [`ImportKind::Glob`] specifier reaches.
    ///
    /// The specifier is the glob pattern the analyzer kept, relative to the
    /// importer; turning it into files is the same path arithmetic as
    /// [`Analyzer::resolve`], which is why it lives beside it rather than in
    /// the graph.
    ///
    /// [`ImportKind::Glob`]: crate::model::ImportKind::Glob
    fn glob_targets(
        &self,
        _specifier: &str,
        _importer: &Path,
        _index: &ModuleIndex,
    ) -> Vec<ModuleId> {
        Vec::new()
    }

    /// Globs, against scan-root-relative paths, of files that are entry
    /// points by convention: `src/index.ts`, `__main__.py`, a framework's
    /// route directory. A glob without a leading `**/` also matches at any
    /// depth.
    fn entry_globs(&self) -> &'static [&'static str] {
        &[]
    }

    /// Globs of files that are tests, fixtures or examples in this language
    /// (`*.test.ts`, `test_*.py`). Directory conventions shared by every
    /// language (`tests/`, `__tests__/`, `fixtures/`) are built in.
    fn test_globs(&self) -> &'static [&'static str] {
        &[]
    }

    /// Whether a file declares that it runs on its own. A shebang says so in
    /// any language; Python adds `if __name__ == "__main__"`.
    fn is_self_starting(&self, source: &str) -> bool {
        source.starts_with("#!")
    }

    /// Manifest file names this analyzer can read for entry points
    /// (`package.json`, `pyproject.toml`). Looked for in every directory
    /// that holds a scanned file, and in the scan roots.
    fn manifests(&self) -> &'static [&'static str] {
        &[]
    }

    /// Absolute paths a manifest names as entry points. `directory` holds
    /// the manifest; `text` is its contents. Paths need not exist — a
    /// manifest often names a built file, and the analyzer should return the
    /// sources it could have been built from as well.
    fn manifest_entries(&self, _directory: &Path, _manifest: &str, _text: &str) -> Vec<PathBuf> {
        Vec::new()
    }

    /// What a manifest says about the project's toolchain — the packages it
    /// depends on and the sections it carries — for the same manifests.
    ///
    /// This is how a framework is recognised in a project that keeps no
    /// config file for it: `"next"` among the dependencies, a `"jest"`
    /// section in `package.json`. See [`crate::framework`].
    fn manifest_signals(&self, _manifest: &str, _text: &str) -> ManifestSignals {
        ManifestSignals::default()
    }

    /// What a framework's config file assigns to `key`, when the file is
    /// written in this analyzer's language and the value is a literal
    /// (`srcDir: "src"`, `imports: false`). `None` for any other file, so
    /// asking every analyzer in turn finds the one that can read it.
    fn config_setting(&self, _config: &str, _text: &str, _key: &str) -> Option<Setting> {
        None
    }

    /// Config files that declare import path aliases (`tsconfig.json`).
    /// Looked for in the same directories as [`Analyzer::manifests`].
    fn alias_configs(&self) -> &'static [&'static str] {
        &[]
    }

    /// Path aliases one such config declares. `directory` holds the config;
    /// `text` is its contents. Targets must be absolute — join them to
    /// `directory` — and need not exist, since the index decides that.
    fn path_aliases(&self, _directory: &Path, _config: &str, _text: &str) -> Vec<PathAlias> {
        Vec::new()
    }

    /// Directories absolute imports may be rooted at, derived from the files
    /// in the scan. A Python `src/` layout puts the package root at `src/`,
    /// which no scan root and no importer's own package can reveal.
    /// Consulted once, after the index is built.
    fn import_roots(&self, _modules: &[PathBuf]) -> Vec<PathBuf> {
        Vec::new()
    }

    /// What the path alone says about a file. See [`ModuleTraits`].
    fn module_traits(&self, _path: &str) -> ModuleTraits {
        ModuleTraits::default()
    }
}

/// Registered analyzers, consulted in order. Add new languages here.
pub static ANALYZERS: &[&dyn Analyzer] = &[&javascript::JsAnalyzer, &python::PythonAnalyzer];

/// The analyzer serving `format`, if any.
pub fn analyzer_for(format: &str) -> Option<&'static dyn Analyzer> {
    ANALYZERS
        .iter()
        .copied()
        .find(|a| a.formats().contains(&format))
}

/// True when basta can analyze this jscpd format.
pub fn supports(format: &str) -> bool {
    analyzer_for(format).is_some()
}

/// Every format basta analyzes, sorted, for `--list` and error messages.
pub fn supported_formats() -> Vec<&'static str> {
    let mut formats: Vec<&'static str> = ANALYZERS
        .iter()
        .flat_map(|a| a.formats().iter().copied())
        .collect();
    formats.sort_unstable();
    formats
}

/// Every file extension basta analyzes, from the tokenizer's format table.
/// Used wherever a path has to be recognised as source without parsing it —
/// a `package.json` script line, a `files` entry.
pub fn supported_extensions() -> Vec<&'static str> {
    let formats = supported_formats();
    let mut extensions: Vec<&'static str> = cpd_tokenizer::formats::SUPPORTED_FORMATS
        .iter()
        .filter(|entry| formats.contains(&entry.name))
        .flat_map(|entry| entry.extensions.iter().copied())
        .collect();
    extensions.sort_unstable();
    extensions.dedup();
    extensions
}

/// True when `path` ends in an extension some analyzer serves.
pub fn is_source_path(path: &str) -> bool {
    supported_extensions()
        .iter()
        .any(|extension| path.ends_with(&format!(".{extension}")))
}

/// True for a string that could be the name of a declaration.
///
/// Used on string literals, to decide whether one is worth remembering as
/// weak evidence that a name may be looked up at runtime. `$` is accepted for
/// JavaScript's sake; it cannot appear in a Python identifier, so a Python
/// string containing one simply never matches a declaration.
///
/// The length cap keeps a data-heavy file — a fixture full of prose, a
/// generated table of strings — from filling the set with things no
/// declaration could be called.
pub fn is_identifier_like(text: &str) -> bool {
    !text.is_empty()
        && text.len() <= 100
        && text.starts_with(|c: char| c.is_alphabetic() || c == '_' || c == '$')
        && text
            .chars()
            .all(|c| c.is_alphanumeric() || c == '_' || c == '$')
}

/// Whether a byte scan is inside a quoted run, so that a bracket, a comma or a
/// `//` written inside a string is not mistaken for structure.
///
/// The hand-written scanners over markup and bundler configs all need this,
/// and each one tracking it separately is how an escaped quote came to be
/// handled three different ways.
#[derive(Default)]
pub(crate) struct Quotes {
    open: u8,
}

impl Quotes {
    /// Feed the byte at `at`. When it belongs to a string — an opening quote,
    /// anything inside, an escape, the closing quote — returns how many bytes
    /// to step over; otherwise `None`, and the byte is the caller's to read.
    pub(crate) fn step(&mut self, bytes: &[u8], at: usize) -> Option<usize> {
        let byte = bytes[at];
        if self.open != 0 {
            match byte {
                b'\\' => return Some(2),
                _ if byte == self.open => self.open = 0,
                _ => {}
            }
            return Some(1);
        }
        if matches!(byte, b'"' | b'\'' | b'`') {
            self.open = byte;
            return Some(1);
        }
        None
    }
}

/// Every byte from `from` on that lies outside a string, with its index — the
/// view a bracket-matching or comma-splitting scan wants.
pub(crate) fn outside_strings(bytes: &[u8], from: usize) -> impl Iterator<Item = (usize, u8)> + '_ {
    let (mut quotes, mut at) = (Quotes::default(), from);
    std::iter::from_fn(move || {
        while at < bytes.len() {
            match quotes.step(bytes, at) {
                Some(step) => at += step,
                None => {
                    at += 1;
                    return Some((at - 1, bytes[at - 1]));
                }
            }
        }
        None
    })
}

/// The first index at or after `from` whose byte does not satisfy `keep`.
pub(crate) fn skip_while(bytes: &[u8], from: usize, keep: impl Fn(u8) -> bool) -> usize {
    let mut at = from;
    while at < bytes.len() && keep(bytes[at]) {
        at += 1;
    }
    at
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn quotes_step_over_strings_and_their_escapes() {
        // Positions the scan stops at, i.e. bytes that are structure.
        let structure = |text: &str| {
            let (bytes, mut quotes, mut at, mut seen) =
                (text.as_bytes(), Quotes::default(), 0, String::new());
            while at < bytes.len() {
                match quotes.step(bytes, at) {
                    Some(step) => at += step,
                    None => {
                        seen.push(bytes[at] as char);
                        at += 1;
                    }
                }
            }
            seen
        };
        assert_eq!(structure(r#"a("x,y", 'z')"#), "a(, )");
        // An escaped quote does not close the string…
        assert_eq!(structure(r#"["a\"b", c]"#), "[, c]");
        // …but an escaped backslash does not escape the quote after it, which
        // a check of the previous byte alone gets wrong.
        assert_eq!(structure(r#"["a\\", c]"#), "[, c]");
        assert_eq!(structure("`${x}` + y"), " + y");
    }

    #[test]
    fn skip_while_stops_at_the_first_byte_it_does_not_keep() {
        let bytes = b"  name=1";
        assert_eq!(skip_while(bytes, 0, |b| b == b' '), 2);
        assert_eq!(skip_while(bytes, 2, |b| b.is_ascii_alphabetic()), 6);
        assert_eq!(
            skip_while(bytes, 8, |_| true),
            8,
            "at the end stays at the end"
        );
    }

    #[test]
    fn every_supported_format_has_exactly_one_analyzer() {
        let formats = supported_formats();
        let mut seen = std::collections::HashSet::new();
        for f in &formats {
            assert!(seen.insert(*f), "format {f} is claimed by two analyzers");
            assert!(supports(f), "{f} must resolve");
        }
    }

    #[test]
    fn every_analyzer_format_is_one_the_walker_knows() {
        // The walker maps extensions to formats through the tokenizer's
        // table. An analyzer claiming a format that table lacks would compile,
        // register, and never see a single file.
        let known = cpd_tokenizer::formats::list_formats();
        for analyzer in ANALYZERS {
            for format in analyzer.formats() {
                assert!(
                    known.contains(format),
                    "{} claims format {format:?}, which cpd_tokenizer::formats does not define",
                    analyzer.language()
                );
            }
        }
    }

    #[test]
    fn language_ids_are_stable_lower_case_and_distinct() {
        let mut ids: Vec<&str> = ANALYZERS.iter().map(|a| a.language()).collect();
        for id in &ids {
            assert!(
                !id.is_empty() && id.chars().all(|c| c.is_ascii_lowercase()),
                "{id:?} is not a lower-case ascii id"
            );
        }
        let before = ids.len();
        ids.sort_unstable();
        ids.dedup();
        assert_eq!(ids.len(), before, "two analyzers share a language id");
    }

    #[test]
    fn supported_extensions_come_from_the_format_table() {
        let extensions = supported_extensions();
        for want in ["ts", "tsx", "js", "mjs", "py", "pyi"] {
            assert!(
                extensions.contains(&want),
                "{want} missing from {extensions:?}"
            );
        }
        assert!(!extensions.contains(&"java"));
        assert!(is_source_path("src/a.tsx"));
        assert!(is_source_path("pkg/mod.py"));
        assert!(!is_source_path("README.md"));
    }

    #[test]
    fn identifier_shaped_strings_are_told_from_prose() {
        for yes in ["handleRoot", "_private", "$el", "a1", "run_phase"] {
            assert!(is_identifier_like(yes), "{yes}");
        }
        for no in [
            "",
            "1abc",
            "has space",
            "path/to/file",
            "kebab-case",
            &"x".repeat(101),
        ] {
            assert!(!is_identifier_like(no), "{no:?}");
        }
    }

    #[test]
    fn known_formats_map_to_the_right_language() {
        for f in ["javascript", "typescript", "jsx", "tsx"] {
            assert_eq!(analyzer_for(f).map(|a| a.language()), Some("js"), "{f}");
        }
        assert_eq!(analyzer_for("python").map(|a| a.language()), Some("python"));
        assert_eq!(analyzer_for("ruby").map(|a| a.language()), None);
    }

    #[test]
    fn the_trait_defaults_mean_nothing_special() {
        // A minimal analyzer: two required methods and the defaults.
        struct Minimal;
        impl Analyzer for Minimal {
            fn language(&self) -> &'static str {
                "minimal"
            }
            fn formats(&self) -> &'static [&'static str] {
                &["minimal"]
            }
            fn analyze(&self, _: &AnalyzeInput<'_>) -> FileFacts {
                FileFacts::default()
            }
            fn resolve(&self, _: &str, _: &Path, _: &ModuleIndex) -> Option<ModuleId> {
                None
            }
        }
        let m = Minimal;
        assert!(m.entry_globs().is_empty());
        assert!(m.test_globs().is_empty());
        assert!(m.manifests().is_empty());
        assert!(m.is_self_starting("#!/usr/bin/env thing\n"));
        assert!(!m.is_self_starting("plain source\n"));
        assert_eq!(m.module_traits("any/file.minimal"), ModuleTraits::default());
        let index = ModuleIndex::new(vec![PathBuf::from("/p")]);
        assert!(m.manifest_entries(Path::new("/p"), "x.toml", "").is_empty());
        assert!(m.manifest_signals("x.toml", "").dependencies.is_empty());
        assert!(m.config_setting("x.config.toml", "", "key").is_none());
        assert!(
            m.resolve("./x", Path::new("/p/a.minimal"), &index)
                .is_none()
        );
    }
}