Skip to main content

fallow_extract/
lib.rs

1//! Parsing and extraction engine for fallow codebase intelligence.
2//!
3//! This crate handles all file parsing: JS/TS via Oxc, Vue/Svelte SFC extraction,
4//! Astro frontmatter, MDX import/export extraction, CSS Module class name extraction,
5//! HTML asset reference extraction, and incremental caching of parse results.
6
7#![warn(missing_docs)]
8#![cfg_attr(not(test), deny(clippy::disallowed_methods))]
9#![cfg_attr(
10    test,
11    allow(
12        clippy::unwrap_used,
13        clippy::expect_used,
14        reason = "tests use unwrap and expect to keep fixture setup concise"
15    )
16)]
17
18mod asset_url;
19pub mod astro;
20pub mod cache;
21pub(crate) mod complexity;
22mod component_contracts;
23pub mod css;
24pub mod css_classes;
25pub mod css_in_js;
26pub mod css_metrics;
27pub mod federation_runtime;
28pub mod flags;
29mod function_body;
30pub mod glimmer;
31pub(crate) mod graphql;
32pub(crate) mod html;
33pub(crate) mod iconify;
34pub mod inventory;
35mod jsdoc_attach;
36mod jsdoc_deprecated;
37pub mod mdx;
38mod module_info;
39pub mod og_image;
40mod parse;
41pub mod sfc;
42pub mod sfc_css;
43mod sfc_props;
44mod sfc_template;
45pub mod similar_code;
46mod source_map;
47pub mod suppress;
48/// Tailwind CSS arbitrary-value detection.
49pub mod tailwind;
50pub(crate) mod template_complexity;
51mod template_expression_scan;
52mod template_usage;
53/// Visitor utilities for AST extraction.
54pub mod visitor;
55
56use std::path::Path;
57use std::sync::atomic::{AtomicBool, Ordering};
58
59use rayon::prelude::*;
60
61use cache::CacheStore;
62use fallow_types::discover::{DiscoveredFile, FileId};
63
64pub use fallow_types::extract::{
65    AngularComponentFieldArrayTypeFact, AngularTemplateMemberAccessFact, AngularThisSpreadFact,
66    ClassHeritageInfo, ClassThisMemberAccessFact, ClassThisWholeObjectUseFact,
67    ComputedEnumKeyUseFact, DefaultImportWholeObjectUseFact, DynamicCustomElementRenderFact,
68    DynamicImportInfo, DynamicImportPattern, ExportInfo, ExportName,
69    ExportedObjectInstancePropertyFact, FactoryCallMemberAccessFact, FactoryFnMemberAccessFact,
70    FactoryFnWholeObjectFact, FactoryReturnExport, FactoryReturnObjectPropertyAccessFact,
71    FactoryReturnObjectShapeExport, FlagPatterns, FluentChainMemberAccessFact,
72    FluentChainNewMemberAccessFact, ImportInfo, ImportedName, InstanceExportBindingFact,
73    LocalTypeDeclaration, MemberAccess, MemberInfo, MemberKind, ModuleInfo, ModuleLoadMechanism,
74    ParseResult, PlaywrightFixtureAliasFact, PlaywrightFixtureDefinitionFact,
75    PlaywrightFixtureTypeFact, PlaywrightFixtureUseFact, PublicSignatureTypeReference,
76    QualifiedClassMemberAccessFact, ReExportInfo, RequireCallInfo, RequiredTypeMemberFact,
77    SemanticFact, SourceParseDegradation, SourceReadFailure, StringEnumMemberValueFact,
78    TypeAliasSurfaceTargetFact, TypeMemberTypeEntry, TypedPropertyMemberAccessFact, VisibilityTag,
79    VitestModuleMockAction, VitestModuleMockOperationFact, compute_line_offsets,
80};
81
82pub use astro::{
83    extract_astro_frontmatter, extract_astro_style_regions, extract_astro_template_regions,
84};
85pub use css::{
86    StylesheetTokens, ThemeScan, ThemeTokenDef, extract_apply_tokens, extract_apply_tokens_located,
87    extract_css_module_exports, extract_css_var_reads_located, scan_stylesheet_tokens,
88    scan_theme_blocks,
89};
90pub use css_classes::{
91    MarkupClassScan, MarkupClassToken, is_edit_distance_one, is_typo_edit, scan_markup_class_tokens,
92};
93pub use css_in_js::{
94    ConsumerQuery, CssInJsObjectSheets, CssInJsToken, CssInJsTokenDef, CssInJsTokenOrigin,
95    TokenConsumerHit, css_in_js_consumer_scan, css_in_js_object_sheets, css_in_js_theme_token_defs,
96    css_in_js_token_defs, css_in_js_virtual_stylesheet,
97};
98pub use css_metrics::{compute_css_analytics, parse_css_color_rgb};
99pub use glimmer::{is_glimmer_file, strip_glimmer_templates};
100pub use mdx::{extract_mdx_statements, extract_mdx_statements_mapped};
101pub use sfc::{
102    SourceRegion, extract_sfc_scripts, extract_sfc_styles, extract_sfc_template_regions,
103    is_sfc_file,
104};
105pub use sfc_css::{
106    scoped_unused_classes, sfc_preprocessor_virtual_stylesheet, sfc_virtual_stylesheet,
107};
108pub use similar_code::extract_similar_code_functions;
109pub use source_map::ExtractionResult;
110pub use tailwind::{TailwindArbitraryUse, scan_tailwind_arbitrary_values};
111
112#[expect(
113    clippy::expect_used,
114    reason = "static regex patterns are hard-coded analyzer invariants covered by extraction tests"
115)]
116fn static_regex(pattern: &str) -> regex::Regex {
117    regex::Regex::new(pattern).expect("static regex pattern should compile")
118}
119
120pub use parse::{parse_source_to_module, parse_source_to_module_with_flags};
121
122/// Leading UTF-8 byte order mark codepoint.
123///
124/// Windows editors (Notepad, older VS settings, some IDE plugins) emit a UTF-8
125/// BOM at the start of source files. fallow's contract is "UTF-8 with or
126/// without BOM; line offsets are computed against the post-BOM view; the BOM,
127/// if present on input, is preserved on output by `fallow fix`."
128const BOM_CHAR: char = '\u{FEFF}';
129// Small, cache-hot inputs are faster on one thread than through Rayon setup.
130// Larger file sets still use parallel parsing where parse work dominates.
131const PARALLEL_PARSE_FILE_THRESHOLD: usize = 32;
132
133/// Strip the leading UTF-8 BOM if present.
134///
135/// Called at every file-read entry point in this crate so the rest of the
136/// pipeline (content hash, `compute_line_offsets`, oxc parser, downstream
137/// analyses) sees a consistent post-BOM view. Mirrors the
138/// `fallow_config` layer (`config_writer.rs::BOM`) so config-shaped sources
139/// and source-code-shaped sources are processed symmetrically. See issue #475.
140#[must_use]
141fn strip_bom(source: &str) -> &str {
142    source.strip_prefix(BOM_CHAR).unwrap_or(source)
143}
144
145/// Parse all files, extracting imports and exports.
146///
147/// Small file sets use a sequential fast path to avoid parallel scheduling
148/// overhead; larger file sets use parallel extraction.
149/// Uses the cache to skip reparsing files whose content hasn't changed.
150///
151/// When `need_complexity` is true, per-function cyclomatic/cognitive complexity
152/// metrics are computed during parsing (needed by the `health` command).
153/// Pass `false` for dead-code analysis where complexity data is unused.
154///
155/// Flag detection uses the built-in patterns only. A caller with a resolved
156/// config uses [`parse_all_files_cancellable`] with the config's patterns,
157/// because the cache keys on them.
158pub fn parse_all_files(
159    files: &[DiscoveredFile],
160    cache: Option<&CacheStore>,
161    need_complexity: bool,
162) -> ParseResult {
163    parse_all_files_cancellable(
164        files,
165        cache,
166        need_complexity,
167        None,
168        &FlagPatterns::default(),
169    )
170}
171
172/// Parse all files, abandoning the remaining ones once `cancellation` is set.
173///
174/// Rayon's `map`/`collect` cannot short-circuit, so cancellation makes the
175/// per-file body a no-op instead of stopping the iteration: the scheduled
176/// items still drain, but at one atomic load each. The returned
177/// [`ParseResult`] is therefore truncated whenever the token flipped, and
178/// callers must treat a set token as a failed run rather than as a project
179/// with fewer modules.
180///
181/// `flag_patterns` are the user flag patterns that detection applies on top
182/// of the built-in ones. They must match the patterns the cache was keyed on.
183pub fn parse_all_files_cancellable(
184    files: &[DiscoveredFile],
185    cache: Option<&CacheStore>,
186    need_complexity: bool,
187    cancellation: Option<&AtomicBool>,
188    flag_patterns: &FlagPatterns,
189) -> ParseResult {
190    let parse_one = |file: &DiscoveredFile| {
191        if cancellation.is_some_and(|cancelled| cancelled.load(Ordering::SeqCst)) {
192            return ParseFileResult::default();
193        }
194        parse_single_file_cached(file, cache, need_complexity, flag_patterns)
195    };
196    let results: Vec<ParseFileResult> = if files.len() <= PARALLEL_PARSE_FILE_THRESHOLD {
197        files.iter().map(parse_one).collect()
198    } else {
199        files.par_iter().map(parse_one).collect()
200    };
201
202    let mut modules = Vec::with_capacity(results.len());
203    let mut read_failures = Vec::new();
204    let mut parse_degradations = Vec::new();
205    let mut hits = 0usize;
206    let mut misses = 0usize;
207    let mut parse_cpu_nanos = 0u64;
208    let mut files_read = 0u64;
209    let mut source_bytes_read = 0u64;
210    let mut css_masked_bytes = 0u64;
211
212    // `results` is a positional map over `files`, so zipping recovers the path
213    // for a module without carrying one on `ModuleInfo`.
214    for (file, result) in files.iter().zip(results) {
215        hits += result.cache_hits;
216        misses += result.cache_misses;
217        parse_cpu_nanos = parse_cpu_nanos.saturating_add(result.parse_cpu_nanos);
218        if let Some(bytes) = result.source_bytes_read {
219            files_read += 1;
220            source_bytes_read += bytes;
221        }
222        css_masked_bytes += result.css_masked_bytes;
223        if let Some(module) = result.module {
224            if module.parse_error_count > 0 {
225                parse_degradations.push(SourceParseDegradation {
226                    file_id: module.file_id,
227                    path: file.path.clone(),
228                    error_count: module.parse_error_count,
229                    panicked: module.parse_panicked,
230                });
231            }
232            modules.push(module);
233        }
234        if let Some(failure) = result.read_failure {
235            read_failures.push(failure);
236        }
237    }
238
239    if hits > 0 || misses > 0 {
240        tracing::info!(
241            cache_hits = hits,
242            cache_misses = misses,
243            "incremental cache stats"
244        );
245    }
246
247    ParseResult {
248        modules,
249        read_failures,
250        parse_degradations,
251        cache_hits: hits,
252        cache_misses: misses,
253        parse_cpu_ms: parse_cpu_nanos as f64 / 1_000_000.0,
254        files_read,
255        source_bytes_read,
256        css_masked_bytes,
257    }
258}
259
260#[derive(Default)]
261struct ParseFileResult {
262    module: Option<ModuleInfo>,
263    read_failure: Option<SourceReadFailure>,
264    cache_hits: usize,
265    cache_misses: usize,
266    parse_cpu_nanos: u64,
267    /// Source bytes read from disk for this file, or `None` when the file was
268    /// served from cache metadata without a read.
269    source_bytes_read: Option<u64>,
270    /// Source bytes that the CSS comment mask read during the parse.
271    css_masked_bytes: u64,
272}
273
274impl ParseFileResult {
275    fn cache_hit(module: ModuleInfo) -> Self {
276        Self {
277            module: Some(module),
278            read_failure: None,
279            cache_hits: 1,
280            cache_misses: 0,
281            parse_cpu_nanos: 0,
282            source_bytes_read: None,
283            css_masked_bytes: 0,
284        }
285    }
286
287    fn cache_miss(module: ModuleInfo, parse_cpu_nanos: u64) -> Self {
288        Self {
289            module: Some(module),
290            read_failure: None,
291            cache_hits: 0,
292            cache_misses: 1,
293            parse_cpu_nanos,
294            source_bytes_read: None,
295            css_masked_bytes: 0,
296        }
297    }
298
299    const fn with_source_bytes_read(mut self, bytes: usize) -> Self {
300        self.source_bytes_read = Some(bytes as u64);
301        self
302    }
303
304    fn read_failure(file: &DiscoveredFile, error: &std::io::Error) -> Self {
305        Self {
306            module: None,
307            read_failure: Some(SourceReadFailure {
308                file_id: file.id,
309                path: file.path.clone(),
310                error: error.to_string(),
311            }),
312            cache_hits: 0,
313            cache_misses: 0,
314            parse_cpu_nanos: 0,
315            source_bytes_read: None,
316            css_masked_bytes: 0,
317        }
318    }
319}
320
321/// Parse a single file, consulting the cache first.
322///
323/// Cache validation strategy (fast path -> slow path):
324/// 1. Open the file so unreadable sources cannot use stale cached analysis
325/// 2. Read mtime + ctime + size from the open handle
326/// 3. If all three match the cached entry -> cache hit, return immediately
327/// 4. Otherwise -> read file, compute content hash
328/// 5. If content hash matches cached entry -> cache hit (file was rewritten or
329///    `touch`ed but its content is unchanged)
330/// 6. Otherwise -> cache miss, full parse
331///
332/// Step 3 requires ctime as well as mtime because mtime is writer-controlled:
333/// a same-length rewrite whose mtime is restored (`touch -r`, a codemod, a
334/// `git checkout` of an equal-length revision) leaves `(mtime, size)`
335/// unchanged, and serving the cached module for it means reporting the OLD
336/// file's unused exports with an auto-fixable `remove-export` action. A file
337/// whose ctime moved falls through to step 4 and still hits on the content
338/// hash, so the cost of the stricter gate is one read, not a reparse.
339fn parse_single_file_cached(
340    file: &DiscoveredFile,
341    cache: Option<&CacheStore>,
342    need_complexity: bool,
343    flag_patterns: &FlagPatterns,
344) -> ParseFileResult {
345    let cached_by_path = cache.and_then(|store| store.get_by_path_only(&file.path));
346
347    if let Some(cached) = cached_by_path
348        && cached.file_size == file.size_bytes
349    {
350        let source_file = match std::fs::File::open(&file.path) {
351            Ok(source_file) => source_file,
352            Err(error) => return ParseFileResult::read_failure(file, &error),
353        };
354        if let Ok(metadata) = source_file.metadata()
355            && metadata.len() == cached.file_size
356        {
357            let fingerprint =
358                fallow_types::source_fingerprint::SourceFingerprint::from_metadata(&metadata);
359            if cached.source_fingerprint() == fingerprint
360                && fingerprint.is_trustworthy_without_content()
361                && (!need_complexity || cached.complexity_extracted)
362            {
363                return ParseFileResult::cache_hit(cache::cached_to_module_opts(
364                    cached,
365                    file.id,
366                    need_complexity,
367                ));
368            }
369        }
370    }
371
372    let raw = match std::fs::read_to_string(&file.path) {
373        Ok(raw) => raw,
374        Err(error) => return ParseFileResult::read_failure(file, &error),
375    };
376    let source = strip_bom(&raw);
377    let content_hash = xxhash_rust::xxh3::xxh3_64(source.as_bytes());
378
379    if let Some(cached) = cached_by_path
380        && cached.content_hash == content_hash
381        && (!need_complexity || cached.complexity_extracted)
382    {
383        return ParseFileResult::cache_hit(cache::cached_to_module_opts(
384            cached,
385            file.id,
386            need_complexity,
387        ))
388        .with_source_bytes_read(raw.len());
389    }
390
391    let parse_start = std::time::Instant::now();
392    // Drop a count that a scan outside a parse left on this thread.
393    css::take_comment_masked_bytes();
394    let module = parse_source_to_module_with_flags(
395        file.id,
396        &file.path,
397        source,
398        content_hash,
399        need_complexity,
400        flag_patterns,
401    );
402    let parse_cpu_nanos = u64::try_from(parse_start.elapsed().as_nanos()).unwrap_or(u64::MAX);
403    let mut result =
404        ParseFileResult::cache_miss(module, parse_cpu_nanos).with_source_bytes_read(raw.len());
405    result.css_masked_bytes = css::take_comment_masked_bytes();
406    result
407}
408
409/// Parse a single file and extract module information (without complexity).
410#[must_use]
411pub fn parse_single_file(file: &DiscoveredFile) -> Option<ModuleInfo> {
412    let raw = std::fs::read_to_string(&file.path).ok()?;
413    let source = strip_bom(&raw);
414    let content_hash = xxhash_rust::xxh3::xxh3_64(source.as_bytes());
415    Some(parse_source_to_module(
416        file.id,
417        &file.path,
418        source,
419        content_hash,
420        false,
421    ))
422}
423
424/// Parse from in-memory content (for LSP, includes complexity).
425#[must_use]
426pub fn parse_from_content(file_id: FileId, path: &Path, content: &str) -> ModuleInfo {
427    let content = strip_bom(content);
428    let content_hash = xxhash_rust::xxh3::xxh3_64(content.as_bytes());
429    parse_source_to_module(file_id, path, content, content_hash, true)
430}
431
432#[cfg(all(test, not(miri)))]
433mod tests;