Skip to main content

fallow_extract/
lib.rs

1//! Parsing and extraction engine for fallow codebase intelligence.
2//!
3//! This crate handles all file parsing: JS/TS via Oxc, Vue/Svelte SFC extraction,
4//! Astro frontmatter, MDX import/export extraction, CSS Module class name extraction,
5//! HTML asset reference extraction, and incremental caching of parse results.
6
7#![warn(missing_docs)]
8#![cfg_attr(not(test), deny(clippy::disallowed_methods))]
9#![cfg_attr(
10    test,
11    allow(
12        clippy::unwrap_used,
13        clippy::expect_used,
14        reason = "tests use unwrap and expect to keep fixture setup concise"
15    )
16)]
17
18mod asset_url;
19pub mod astro;
20pub mod cache;
21pub(crate) mod complexity;
22pub mod css;
23pub mod css_classes;
24pub mod css_in_js;
25pub mod css_metrics;
26pub mod flags;
27pub mod glimmer;
28pub(crate) mod graphql;
29pub(crate) mod html;
30pub(crate) mod iconify;
31pub mod inventory;
32pub mod mdx;
33mod module_info;
34mod parse;
35pub mod sfc;
36pub mod sfc_css;
37mod sfc_props;
38mod sfc_template;
39pub mod similar_code;
40mod source_map;
41pub mod suppress;
42/// Tailwind CSS arbitrary-value detection.
43pub mod tailwind;
44pub(crate) mod template_complexity;
45mod template_expression_scan;
46mod template_usage;
47/// Visitor utilities for AST extraction.
48pub mod visitor;
49
50use std::path::Path;
51use std::sync::atomic::{AtomicBool, Ordering};
52
53use rayon::prelude::*;
54
55use cache::CacheStore;
56use fallow_types::discover::{DiscoveredFile, FileId};
57
58pub use fallow_types::extract::{
59    AngularComponentFieldArrayTypeFact, AngularTemplateMemberAccessFact, AngularThisSpreadFact,
60    ClassHeritageInfo, ClassThisMemberAccessFact, ClassThisWholeObjectUseFact,
61    ComputedEnumKeyUseFact, DefaultImportWholeObjectUseFact, DynamicCustomElementRenderFact,
62    DynamicImportInfo, DynamicImportPattern, ExportInfo, ExportName,
63    ExportedObjectInstancePropertyFact, FactoryCallMemberAccessFact, FactoryFnMemberAccessFact,
64    FactoryFnWholeObjectFact, FactoryReturnExport, FactoryReturnObjectPropertyAccessFact,
65    FactoryReturnObjectShapeExport, FluentChainMemberAccessFact, FluentChainNewMemberAccessFact,
66    ImportInfo, ImportedName, InstanceExportBindingFact, LocalTypeDeclaration, MemberAccess,
67    MemberInfo, MemberKind, ModuleInfo, ModuleLoadMechanism, ParseResult,
68    PlaywrightFixtureAliasFact, PlaywrightFixtureDefinitionFact, PlaywrightFixtureTypeFact,
69    PlaywrightFixtureUseFact, PublicSignatureTypeReference, QualifiedClassMemberAccessFact,
70    ReExportInfo, RequireCallInfo, RequiredTypeMemberFact, SemanticFact, SourceParseDegradation,
71    SourceReadFailure, StringEnumMemberValueFact, TypeAliasSurfaceTargetFact, TypeMemberTypeEntry,
72    TypedPropertyMemberAccessFact, VisibilityTag, VitestModuleMockAction,
73    VitestModuleMockOperationFact, compute_line_offsets,
74};
75
76pub use astro::{
77    extract_astro_frontmatter, extract_astro_style_regions, extract_astro_template_regions,
78};
79pub use css::{
80    ThemeScan, ThemeTokenDef, extract_apply_tokens, extract_apply_tokens_located,
81    extract_css_module_exports, extract_css_var_reads_located, scan_theme_blocks,
82};
83pub use css_classes::{
84    MarkupClassScan, MarkupClassToken, is_edit_distance_one, is_typo_edit, scan_markup_class_tokens,
85};
86pub use css_in_js::{
87    ConsumerQuery, CssInJsObjectSheets, CssInJsToken, CssInJsTokenDef, CssInJsTokenOrigin,
88    TokenConsumerHit, css_in_js_consumer_scan, css_in_js_object_sheets, css_in_js_theme_consumers,
89    css_in_js_theme_token_defs, css_in_js_token_consumers, css_in_js_token_defs,
90    css_in_js_virtual_stylesheet, panda_style_value_consumers, panda_token_call_consumers,
91};
92pub use css_metrics::{compute_css_analytics, parse_css_color_rgb};
93pub use glimmer::{is_glimmer_file, strip_glimmer_templates};
94pub use mdx::{extract_mdx_statements, extract_mdx_statements_mapped};
95pub use sfc::{
96    SourceRegion, extract_sfc_scripts, extract_sfc_styles, extract_sfc_template_regions,
97    is_sfc_file,
98};
99pub use sfc_css::{
100    scoped_unused_classes, sfc_preprocessor_virtual_stylesheet, sfc_virtual_stylesheet,
101};
102pub use similar_code::extract_similar_code_functions;
103pub use source_map::ExtractionResult;
104pub use tailwind::{TailwindArbitraryUse, scan_tailwind_arbitrary_values};
105
106#[expect(
107    clippy::expect_used,
108    reason = "static regex patterns are hard-coded analyzer invariants covered by extraction tests"
109)]
110fn static_regex(pattern: &str) -> regex::Regex {
111    regex::Regex::new(pattern).expect("static regex pattern should compile")
112}
113
114pub use parse::parse_source_to_module;
115
116/// Leading UTF-8 byte order mark codepoint.
117///
118/// Windows editors (Notepad, older VS settings, some IDE plugins) emit a UTF-8
119/// BOM at the start of source files. fallow's contract is "UTF-8 with or
120/// without BOM; line offsets are computed against the post-BOM view; the BOM,
121/// if present on input, is preserved on output by `fallow fix`."
122const BOM_CHAR: char = '\u{FEFF}';
123// Small, cache-hot inputs are faster on one thread than through Rayon setup.
124// Larger file sets still use parallel parsing where parse work dominates.
125const PARALLEL_PARSE_FILE_THRESHOLD: usize = 32;
126
127/// Strip the leading UTF-8 BOM if present.
128///
129/// Called at every file-read entry point in this crate so the rest of the
130/// pipeline (content hash, `compute_line_offsets`, oxc parser, downstream
131/// analyses) sees a consistent post-BOM view. Mirrors the
132/// `fallow_config` layer (`config_writer.rs::BOM`) so config-shaped sources
133/// and source-code-shaped sources are processed symmetrically. See issue #475.
134#[must_use]
135fn strip_bom(source: &str) -> &str {
136    source.strip_prefix(BOM_CHAR).unwrap_or(source)
137}
138
139/// Parse all files, extracting imports and exports.
140///
141/// Small file sets use a sequential fast path to avoid parallel scheduling
142/// overhead; larger file sets use parallel extraction.
143/// Uses the cache to skip reparsing files whose content hasn't changed.
144///
145/// When `need_complexity` is true, per-function cyclomatic/cognitive complexity
146/// metrics are computed during parsing (needed by the `health` command).
147/// Pass `false` for dead-code analysis where complexity data is unused.
148pub fn parse_all_files(
149    files: &[DiscoveredFile],
150    cache: Option<&CacheStore>,
151    need_complexity: bool,
152) -> ParseResult {
153    parse_all_files_cancellable(files, cache, need_complexity, None)
154}
155
156/// Parse all files, abandoning the remaining ones once `cancellation` is set.
157///
158/// Rayon's `map`/`collect` cannot short-circuit, so cancellation makes the
159/// per-file body a no-op instead of stopping the iteration: the scheduled
160/// items still drain, but at one atomic load each. The returned
161/// [`ParseResult`] is therefore truncated whenever the token flipped, and
162/// callers must treat a set token as a failed run rather than as a project
163/// with fewer modules.
164pub fn parse_all_files_cancellable(
165    files: &[DiscoveredFile],
166    cache: Option<&CacheStore>,
167    need_complexity: bool,
168    cancellation: Option<&AtomicBool>,
169) -> ParseResult {
170    let parse_one = |file: &DiscoveredFile| {
171        if cancellation.is_some_and(|cancelled| cancelled.load(Ordering::SeqCst)) {
172            return ParseFileResult::default();
173        }
174        parse_single_file_cached(file, cache, need_complexity)
175    };
176    let results: Vec<ParseFileResult> = if files.len() <= PARALLEL_PARSE_FILE_THRESHOLD {
177        files.iter().map(parse_one).collect()
178    } else {
179        files.par_iter().map(parse_one).collect()
180    };
181
182    let mut modules = Vec::with_capacity(results.len());
183    let mut read_failures = Vec::new();
184    let mut parse_degradations = Vec::new();
185    let mut hits = 0usize;
186    let mut misses = 0usize;
187    let mut parse_cpu_nanos = 0u64;
188
189    // `results` is a positional map over `files`, so zipping recovers the path
190    // for a module without carrying one on `ModuleInfo`.
191    for (file, result) in files.iter().zip(results) {
192        hits += result.cache_hits;
193        misses += result.cache_misses;
194        parse_cpu_nanos = parse_cpu_nanos.saturating_add(result.parse_cpu_nanos);
195        if let Some(module) = result.module {
196            if module.parse_error_count > 0 {
197                parse_degradations.push(SourceParseDegradation {
198                    file_id: module.file_id,
199                    path: file.path.clone(),
200                    error_count: module.parse_error_count,
201                    panicked: module.parse_panicked,
202                });
203            }
204            modules.push(module);
205        }
206        if let Some(failure) = result.read_failure {
207            read_failures.push(failure);
208        }
209    }
210
211    if hits > 0 || misses > 0 {
212        tracing::info!(
213            cache_hits = hits,
214            cache_misses = misses,
215            "incremental cache stats"
216        );
217    }
218
219    ParseResult {
220        modules,
221        read_failures,
222        parse_degradations,
223        cache_hits: hits,
224        cache_misses: misses,
225        parse_cpu_ms: parse_cpu_nanos as f64 / 1_000_000.0,
226    }
227}
228
229#[derive(Default)]
230struct ParseFileResult {
231    module: Option<ModuleInfo>,
232    read_failure: Option<SourceReadFailure>,
233    cache_hits: usize,
234    cache_misses: usize,
235    parse_cpu_nanos: u64,
236}
237
238impl ParseFileResult {
239    fn cache_hit(module: ModuleInfo) -> Self {
240        Self {
241            module: Some(module),
242            read_failure: None,
243            cache_hits: 1,
244            cache_misses: 0,
245            parse_cpu_nanos: 0,
246        }
247    }
248
249    fn cache_miss(module: ModuleInfo, parse_cpu_nanos: u64) -> Self {
250        Self {
251            module: Some(module),
252            read_failure: None,
253            cache_hits: 0,
254            cache_misses: 1,
255            parse_cpu_nanos,
256        }
257    }
258
259    fn read_failure(file: &DiscoveredFile, error: &std::io::Error) -> Self {
260        Self {
261            module: None,
262            read_failure: Some(SourceReadFailure {
263                file_id: file.id,
264                path: file.path.clone(),
265                error: error.to_string(),
266            }),
267            cache_hits: 0,
268            cache_misses: 0,
269            parse_cpu_nanos: 0,
270        }
271    }
272}
273
274/// Parse a single file, consulting the cache first.
275///
276/// Cache validation strategy (fast path -> slow path):
277/// 1. Open the file so unreadable sources cannot use stale cached analysis
278/// 2. Read mtime + ctime + size from the open handle
279/// 3. If all three match the cached entry -> cache hit, return immediately
280/// 4. Otherwise -> read file, compute content hash
281/// 5. If content hash matches cached entry -> cache hit (file was rewritten or
282///    `touch`ed but its content is unchanged)
283/// 6. Otherwise -> cache miss, full parse
284///
285/// Step 3 requires ctime as well as mtime because mtime is writer-controlled:
286/// a same-length rewrite whose mtime is restored (`touch -r`, a codemod, a
287/// `git checkout` of an equal-length revision) leaves `(mtime, size)`
288/// unchanged, and serving the cached module for it means reporting the OLD
289/// file's unused exports with an auto-fixable `remove-export` action. A file
290/// whose ctime moved falls through to step 4 and still hits on the content
291/// hash, so the cost of the stricter gate is one read, not a reparse.
292fn parse_single_file_cached(
293    file: &DiscoveredFile,
294    cache: Option<&CacheStore>,
295    need_complexity: bool,
296) -> ParseFileResult {
297    let cached_by_path = cache.and_then(|store| store.get_by_path_only(&file.path));
298
299    if let Some(cached) = cached_by_path
300        && cached.file_size == file.size_bytes
301    {
302        let source_file = match std::fs::File::open(&file.path) {
303            Ok(source_file) => source_file,
304            Err(error) => return ParseFileResult::read_failure(file, &error),
305        };
306        if let Ok(metadata) = source_file.metadata()
307            && metadata.len() == cached.file_size
308        {
309            let fingerprint =
310                fallow_types::source_fingerprint::SourceFingerprint::from_metadata(&metadata);
311            if cached.source_fingerprint() == fingerprint
312                && fingerprint.is_trustworthy_without_content()
313                && (!need_complexity || cached.complexity_extracted)
314            {
315                return ParseFileResult::cache_hit(cache::cached_to_module_opts(
316                    cached,
317                    file.id,
318                    need_complexity,
319                ));
320            }
321        }
322    }
323
324    let raw = match std::fs::read_to_string(&file.path) {
325        Ok(raw) => raw,
326        Err(error) => return ParseFileResult::read_failure(file, &error),
327    };
328    let source = strip_bom(&raw);
329    let content_hash = xxhash_rust::xxh3::xxh3_64(source.as_bytes());
330
331    if let Some(cached) = cached_by_path
332        && cached.content_hash == content_hash
333        && (!need_complexity || cached.complexity_extracted)
334    {
335        return ParseFileResult::cache_hit(cache::cached_to_module_opts(
336            cached,
337            file.id,
338            need_complexity,
339        ));
340    }
341
342    let parse_start = std::time::Instant::now();
343    let module = parse_source_to_module(file.id, &file.path, source, content_hash, need_complexity);
344    let parse_cpu_nanos = u64::try_from(parse_start.elapsed().as_nanos()).unwrap_or(u64::MAX);
345    ParseFileResult::cache_miss(module, parse_cpu_nanos)
346}
347
348/// Parse a single file and extract module information (without complexity).
349#[must_use]
350pub fn parse_single_file(file: &DiscoveredFile) -> Option<ModuleInfo> {
351    let raw = std::fs::read_to_string(&file.path).ok()?;
352    let source = strip_bom(&raw);
353    let content_hash = xxhash_rust::xxh3::xxh3_64(source.as_bytes());
354    Some(parse_source_to_module(
355        file.id,
356        &file.path,
357        source,
358        content_hash,
359        false,
360    ))
361}
362
363/// Parse from in-memory content (for LSP, includes complexity).
364#[must_use]
365pub fn parse_from_content(file_id: FileId, path: &Path, content: &str) -> ModuleInfo {
366    let content = strip_bom(content);
367    let content_hash = xxhash_rust::xxh3::xxh3_64(content.as_bytes());
368    parse_source_to_module(file_id, path, content, content_hash, true)
369}
370
371#[cfg(all(test, not(miri)))]
372mod tests;