1use std::fs;
2use std::path::Path;
3
4use anyhow::{Context, Result};
5use globset::{Glob, GlobSet, GlobSetBuilder};
6use serde_json::Value;
7
8use crate::config::{pointer_strings, pointer_u64};
9use crate::model::{InventoryFile, SkippedCounts};
10
11const NULL_BYTE_WINDOW: usize = 4096;
12
13fn ignore_set(patterns: &[String]) -> Result<GlobSet> {
14 let mut builder = GlobSetBuilder::new();
15 for pattern in patterns {
16 builder
17 .add(Glob::new(pattern).with_context(|| format!("invalid ignore glob {pattern:?}"))?);
18 if !pattern.contains('/') {
19 builder.add(
20 Glob::new(&format!("**/{pattern}"))
21 .with_context(|| format!("invalid ignore glob {pattern:?}"))?,
22 );
23 }
24 }
25 Ok(builder.build()?)
26}
27
28fn language_for_path(path: &str) -> &'static str {
29 let lower = path.to_ascii_lowercase();
30 let extension = lower.rsplit('.').next().unwrap_or_default();
31 match extension {
32 "rs" => "Rust",
33 "py" | "pyi" => "Python",
34 "js" | "mjs" | "cjs" => "JavaScript",
35 "jsx" => "JSX",
36 "ts" | "mts" | "cts" => "TypeScript",
37 "tsx" => "TSX",
38 "swift" => "Swift",
39 "kt" | "kts" => "Kotlin",
40 "java" => "Java",
41 "go" => "Go",
42 "rb" => "Ruby",
43 "php" => "PHP",
44 "c" | "h" => "C",
45 "cc" | "cpp" | "cxx" | "hpp" => "C++",
46 "cs" => "C#",
47 "sh" | "bash" | "zsh" => "Shell",
48 "md" | "mdx" => "Markdown",
49 "json" | "jsonl" => "JSON",
50 "yaml" | "yml" => "YAML",
51 "toml" => "TOML",
52 "xml" => "XML",
53 "html" | "htm" => "HTML",
54 "css" | "scss" | "sass" | "less" => "CSS",
55 "sql" => "SQL",
56 "graphql" | "gql" => "GraphQL",
57 "csv" => "CSV",
58 "tsv" => "TSV",
59 "txt" | "text" => "Plain Text",
60 "svg" => "SVG",
61 _ => "Plain Text",
62 }
63}
64
65fn classification_for_path(path: &str) -> &'static str {
66 let lower = path.to_ascii_lowercase();
67 let name = lower.rsplit('/').next().unwrap_or(&lower);
68 if lower.starts_with("tests/")
69 || lower.starts_with("test/")
70 || lower.contains("/tests/")
71 || lower.contains("/test/")
72 || name.contains(".test.")
73 || name.contains("_test.")
74 || name.starts_with("test_")
75 || lower.contains("__tests__")
76 {
77 "test"
78 } else if lower.starts_with("docs/") || lower.ends_with(".md") || lower.ends_with(".mdx") {
79 "docs"
80 } else if lower.starts_with("scripts/")
81 || lower.starts_with("tools/")
82 || lower.starts_with(".github/")
83 {
84 "tool"
85 } else if lower.starts_with("config/")
86 || matches!(
87 name,
88 "cargo.toml" | "pyproject.toml" | "package.json" | "tsconfig.json" | "wrangler.toml"
89 )
90 {
91 "config"
92 } else if lower.starts_with("src/")
93 || lower.starts_with("app/")
94 || lower.starts_with("lib/")
95 || lower.starts_with("crates/")
96 || lower.starts_with("packages/")
97 {
98 "source"
99 } else {
100 "other"
101 }
102}
103
104fn line_counts(text: &str, language: &str) -> (usize, usize, usize, usize) {
105 if text.is_empty() {
106 return (0, 0, 0, 0);
107 }
108 let lines: Vec<&str> = text.lines().collect();
109 let mut blank = 0;
110 let mut comments = 0;
111 let mut code = 0;
112 let mut in_block_comment = false;
113 for line in &lines {
114 let trimmed = line.trim();
115 if trimmed.is_empty() {
116 blank += 1;
117 continue;
118 }
119 if in_block_comment {
120 comments += 1;
121 if trimmed.contains("*/") {
122 in_block_comment = false;
123 }
124 continue;
125 }
126 let line_comment = match language {
127 "Python" | "Ruby" | "Shell" | "YAML" | "TOML" => trimmed.starts_with('#'),
128 "Markdown" => trimmed.starts_with("<!--"),
129 _ => trimmed.starts_with("//"),
130 };
131 if line_comment {
132 comments += 1;
133 } else if trimmed.starts_with("/*") {
134 comments += 1;
135 in_block_comment = !trimmed.contains("*/");
136 } else {
137 code += 1;
138 }
139 }
140 (lines.len(), code, comments, blank)
141}
142
143fn profile_for(path: &str, bytes: usize, config: &Value) -> &'static str {
144 let lower = path.to_ascii_lowercase();
145 let data_extension = matches!(
146 lower.rsplit('.').next().unwrap_or_default(),
147 "csv" | "tsv" | "parquet" | "ndjson" | "jsonl" | "sqlite" | "db" | "xml" | "json"
148 );
149 let data_path = lower.starts_with("data/")
150 || lower.contains("/data/")
151 || lower.contains("fixtures/")
152 || lower.contains("reference_data/");
153 let min_bytes = pointer_u64(config, "/health/data_context_min_bytes", 262_144) as usize;
154 if data_extension && (data_path || bytes >= min_bytes) {
155 "data_context"
156 } else {
157 "agent_context"
158 }
159}
160
161pub fn build(
162 repo_root: &Path,
163 tracked_paths: &[String],
164 config: &Value,
165) -> Result<(Vec<InventoryFile>, SkippedCounts)> {
166 let patterns = pointer_strings(config, "/inventory/ignore_globs");
167 let ignored = ignore_set(&patterns)?;
168 let mut skipped = SkippedCounts::default();
169 let mut records = Vec::new();
170 for relative_path in tracked_paths {
171 if ignored.is_match(relative_path) {
172 skipped.ignored += 1;
173 continue;
174 }
175 let absolute_path = repo_root.join(relative_path);
176 let metadata = match fs::symlink_metadata(&absolute_path) {
177 Ok(metadata) => metadata,
178 Err(error) if error.kind() == std::io::ErrorKind::NotFound => {
179 skipped.missing += 1;
180 continue;
181 }
182 Err(error) => {
183 return Err(error)
184 .with_context(|| format!("failed to inspect {}", absolute_path.display()));
185 }
186 };
187 if metadata.is_dir() {
191 skipped.ignored += 1;
192 continue;
193 }
194 let raw = if metadata.file_type().is_symlink() {
198 fs::read_link(&absolute_path)
199 .with_context(|| format!("failed to read link {}", absolute_path.display()))?
200 .to_string_lossy()
201 .into_owned()
202 .into_bytes()
203 } else {
204 fs::read(&absolute_path)
205 .with_context(|| format!("failed to read {}", absolute_path.display()))?
206 };
207 if raw[..raw.len().min(NULL_BYTE_WINDOW)].contains(&0) {
208 skipped.binary += 1;
209 continue;
210 }
211 let bytes = raw.len();
212 let Ok(text) = String::from_utf8(raw) else {
213 skipped.undecodable += 1;
214 continue;
215 };
216 let language = language_for_path(relative_path).to_string();
217 let (lines, code_lines, comment_lines, blank_lines) = line_counts(&text, &language);
218 records.push(InventoryFile {
219 path: relative_path.replace('\\', "/"),
220 absolute_path,
221 bytes,
222 lines,
223 blank_lines,
224 code_lines,
225 comment_lines,
226 language,
227 profile: profile_for(relative_path, bytes, config).to_string(),
228 classification: classification_for_path(relative_path).to_string(),
229 text,
230 });
231 }
232 records.sort_by(|left, right| left.path.cmp(&right.path));
233 Ok((records, skipped))
234}
235
236#[cfg(test)]
237mod tests {
238 use std::fs;
239 #[cfg(unix)]
240 use std::os::unix::fs::symlink;
241
242 use tempfile::tempdir;
243
244 use super::build;
245 use crate::config;
246
247 #[cfg(unix)]
248 #[test]
249 fn tracked_symlinks_are_analyzed_without_following_their_targets() {
250 let repository = tempdir().expect("repository");
251 let outside = tempdir().expect("outside");
252 let secret = outside.path().join("secret.txt");
253 fs::write(&secret, "do not read this target").expect("secret");
254 symlink(&secret, repository.path().join("linked.txt")).expect("symlink");
255
256 let (files, skipped) = build(
257 repository.path(),
258 &["linked.txt".to_string()],
259 &config::default_config(),
260 )
261 .expect("inventory");
262
263 assert_eq!(files.len(), 1);
264 assert_eq!(files[0].text, secret.to_string_lossy());
265 assert!(!files[0].text.contains("do not read this target"));
266 assert_eq!(skipped.missing, 0);
267 }
268
269 #[test]
270 fn tracked_gitlink_directories_are_skipped_instead_of_read_as_files() {
271 let repository = tempdir().expect("repository");
272 fs::create_dir_all(repository.path().join("vendor/submodule")).expect("gitlink directory");
273
274 let (files, skipped) = build(
275 repository.path(),
276 &["vendor/submodule".to_string()],
277 &config::default_config(),
278 )
279 .expect("inventory");
280
281 assert!(files.is_empty());
282 assert_eq!(skipped.ignored, 1);
283 }
284}