git-slop 0.10.1

A deterministic repository token-defragmenter for humans and AI agents.
Documentation
use std::fs;
use std::path::Path;

use anyhow::{Context, Result};
use globset::{Glob, GlobSet, GlobSetBuilder};
use serde_json::Value;

use crate::config::{pointer_strings, pointer_u64};
use crate::model::{InventoryFile, SkippedCounts};

const NULL_BYTE_WINDOW: usize = 4096;

fn decode_text(raw: Vec<u8>) -> Option<String> {
    if raw.starts_with(&[0xef, 0xbb, 0xbf]) {
        return String::from_utf8(raw[3..].to_vec()).ok();
    }
    if raw.starts_with(&[0xff, 0xfe]) || raw.starts_with(&[0xfe, 0xff]) {
        let little_endian = raw.starts_with(&[0xff, 0xfe]);
        let bytes = &raw[2..];
        if bytes.len() % 2 != 0 {
            return None;
        }
        let units = bytes.chunks_exact(2).map(|pair| {
            if little_endian {
                u16::from_le_bytes([pair[0], pair[1]])
            } else {
                u16::from_be_bytes([pair[0], pair[1]])
            }
        });
        return char::decode_utf16(units)
            .collect::<Result<String, _>>()
            .ok();
    }
    String::from_utf8(raw).ok()
}

fn looks_binary(raw: &[u8]) -> bool {
    if raw.is_empty() || raw.starts_with(&[0xff, 0xfe]) || raw.starts_with(&[0xfe, 0xff]) {
        return false;
    }
    let window = NULL_BYTE_WINDOW.min(raw.len());
    let starts = [
        0,
        raw.len().saturating_sub(window) / 2,
        raw.len().saturating_sub(window),
    ];
    let mut sampled = 0usize;
    let mut suspicious = 0usize;
    for start in starts {
        for byte in &raw[start..(start + window).min(raw.len())] {
            sampled += 1;
            if *byte == 0 || *byte < 0x09 || matches!(*byte, 0x0b | 0x0c | 0x0e..=0x1f) {
                suspicious += 1;
            }
        }
    }
    raw.iter().take(window).any(|byte| *byte == 0) || suspicious.saturating_mul(100) > sampled
}

fn ignore_set(patterns: &[String]) -> Result<GlobSet> {
    let mut builder = GlobSetBuilder::new();
    for pattern in patterns {
        builder
            .add(Glob::new(pattern).with_context(|| format!("invalid ignore glob {pattern:?}"))?);
        if !pattern.contains('/') {
            builder.add(
                Glob::new(&format!("**/{pattern}"))
                    .with_context(|| format!("invalid ignore glob {pattern:?}"))?,
            );
        }
    }
    Ok(builder.build()?)
}

fn language_for_path(path: &str) -> &'static str {
    let lower = path.to_ascii_lowercase();
    let extension = lower.rsplit('.').next().unwrap_or_default();
    match extension {
        "rs" => "Rust",
        "py" | "pyi" => "Python",
        "js" | "mjs" | "cjs" => "JavaScript",
        "jsx" => "JSX",
        "ts" | "mts" | "cts" => "TypeScript",
        "tsx" => "TSX",
        "swift" => "Swift",
        "kt" | "kts" => "Kotlin",
        "java" => "Java",
        "go" => "Go",
        "rb" => "Ruby",
        "php" => "PHP",
        "c" | "h" => "C",
        "cc" | "cpp" | "cxx" | "hpp" => "C++",
        "cs" => "C#",
        "sh" | "bash" | "zsh" => "Shell",
        "md" | "mdx" => "Markdown",
        "json" | "jsonl" => "JSON",
        "yaml" | "yml" => "YAML",
        "toml" => "TOML",
        "xml" => "XML",
        "html" | "htm" => "HTML",
        "css" | "scss" | "sass" | "less" => "CSS",
        "sql" => "SQL",
        "graphql" | "gql" => "GraphQL",
        "csv" => "CSV",
        "tsv" => "TSV",
        "txt" | "text" => "Plain Text",
        "svg" => "SVG",
        _ => "Plain Text",
    }
}

fn classification_for_path(path: &str) -> &'static str {
    let lower = path.to_ascii_lowercase();
    let name = lower.rsplit('/').next().unwrap_or(&lower);
    if lower.starts_with("tests/")
        || lower.starts_with("test/")
        || lower.contains("/tests/")
        || lower.contains("/test/")
        || name.contains(".test.")
        || name.contains("_test.")
        || name.starts_with("test_")
        || lower.contains("__tests__")
    {
        "test"
    } else if lower.starts_with("docs/") || lower.ends_with(".md") || lower.ends_with(".mdx") {
        "docs"
    } else if lower.starts_with("scripts/")
        || lower.starts_with("tools/")
        || lower.starts_with(".github/")
    {
        "tool"
    } else if lower.starts_with("config/")
        || matches!(
            name,
            "cargo.toml" | "pyproject.toml" | "package.json" | "tsconfig.json" | "wrangler.toml"
        )
    {
        "config"
    } else if lower.starts_with("src/")
        || lower.starts_with("xtask/src/")
        || lower.starts_with("app/")
        || lower.starts_with("lib/")
        || lower.starts_with("crates/")
        || lower.starts_with("packages/")
    {
        "source"
    } else {
        "other"
    }
}

fn line_counts(text: &str, language: &str) -> (usize, usize, usize, usize) {
    if text.is_empty() {
        return (0, 0, 0, 0);
    }
    let lines: Vec<&str> = text.lines().collect();
    let mut blank = 0;
    let mut comments = 0;
    let mut code = 0;
    let mut in_block_comment = false;
    for line in &lines {
        let trimmed = line.trim();
        if trimmed.is_empty() {
            blank += 1;
            continue;
        }
        if in_block_comment {
            comments += 1;
            if trimmed.contains("*/") {
                in_block_comment = false;
            }
            continue;
        }
        let line_comment = match language {
            "Python" | "Ruby" | "Shell" | "YAML" | "TOML" => trimmed.starts_with('#'),
            "Markdown" => trimmed.starts_with("<!--"),
            _ => trimmed.starts_with("//"),
        };
        if line_comment {
            comments += 1;
        } else if trimmed.starts_with("/*") {
            comments += 1;
            in_block_comment = !trimmed.contains("*/");
        } else {
            code += 1;
        }
    }
    (lines.len(), code, comments, blank)
}

fn profile_for(path: &str, bytes: usize, config: &Value) -> &'static str {
    let lower = path.to_ascii_lowercase();
    let data_extension = matches!(
        lower.rsplit('.').next().unwrap_or_default(),
        "csv" | "tsv" | "parquet" | "ndjson" | "jsonl" | "sqlite" | "db" | "xml" | "json"
    );
    let data_path = lower.starts_with("data/")
        || lower.contains("/data/")
        || lower.contains("fixtures/")
        || lower.contains("reference_data/");
    let min_bytes = pointer_u64(config, "/health/data_context_min_bytes", 262_144) as usize;
    if data_extension && (data_path || bytes >= min_bytes) {
        "data_context"
    } else {
        "agent_context"
    }
}

fn path_override(path: &str, config: &Value) -> (Option<String>, Option<String>, Option<String>) {
    let mut classification = None;
    let mut profile = None;
    let mut language = None;
    for mapping in config
        .pointer("/inventory/path_overrides")
        .and_then(Value::as_array)
        .into_iter()
        .flatten()
    {
        let Some(pattern) = mapping.get("glob").and_then(Value::as_str) else {
            continue;
        };
        if Glob::new(pattern)
            .ok()
            .is_some_and(|glob| glob.compile_matcher().is_match(path))
        {
            if let Some(value) = mapping.get("classification").and_then(Value::as_str) {
                classification = Some(value.to_string());
            }
            if let Some(value) = mapping.get("profile").and_then(Value::as_str) {
                profile = Some(value.to_string());
            }
            if let Some(value) = mapping.get("language").and_then(Value::as_str) {
                language = Some(value.to_string());
            }
        }
    }
    (classification, profile, language)
}

pub fn build(
    repo_root: &Path,
    tracked_paths: &[String],
    config: &Value,
) -> Result<(Vec<InventoryFile>, SkippedCounts)> {
    let patterns = pointer_strings(config, "/inventory/ignore_globs");
    let ignored = ignore_set(&patterns)?;
    let mut skipped = SkippedCounts::default();
    let mut records = Vec::new();
    for relative_path in tracked_paths {
        if ignored.is_match(relative_path) {
            skipped.ignored += 1;
            continue;
        }
        let absolute_path = repo_root.join(relative_path);
        let metadata = match fs::symlink_metadata(&absolute_path) {
            Ok(metadata) => metadata,
            Err(error) if error.kind() == std::io::ErrorKind::NotFound => {
                skipped.missing += 1;
                continue;
            }
            Err(error) => {
                return Err(error)
                    .with_context(|| format!("failed to inspect {}", absolute_path.display()));
            }
        };
        // Git represents a submodule as a tracked gitlink whose worktree path
        // is a directory. It is repository metadata, not a text file owned by
        // this analyzer.
        if metadata.is_dir() {
            skipped.ignored += 1;
            continue;
        }
        // Analyze the link stored by Git, never the target it happens to resolve to
        // on the current machine. Following a tracked symlink could otherwise read
        // arbitrary content outside the repository.
        let raw = if metadata.file_type().is_symlink() {
            fs::read_link(&absolute_path)
                .with_context(|| format!("failed to read link {}", absolute_path.display()))?
                .to_string_lossy()
                .into_owned()
                .into_bytes()
        } else {
            fs::read(&absolute_path)
                .with_context(|| format!("failed to read {}", absolute_path.display()))?
        };
        if looks_binary(&raw) {
            skipped.binary += 1;
            continue;
        }
        let bytes = raw.len();
        let Some(text) = decode_text(raw) else {
            skipped.undecodable += 1;
            continue;
        };
        let (classification_override, profile_override, language_override) =
            path_override(relative_path, config);
        let language = language_override.unwrap_or_else(|| language_for_path(relative_path).into());
        let (lines, code_lines, comment_lines, blank_lines) = line_counts(&text, &language);
        records.push(InventoryFile {
            path: relative_path.replace('\\', "/"),
            bytes,
            lines,
            blank_lines,
            code_lines,
            comment_lines,
            language,
            profile: profile_override
                .unwrap_or_else(|| profile_for(relative_path, bytes, config).to_string()),
            classification: classification_override
                .unwrap_or_else(|| classification_for_path(relative_path).to_string()),
            text,
        });
    }
    records.sort_by(|left, right| left.path.cmp(&right.path));
    Ok((records, skipped))
}

#[cfg(test)]
mod tests {
    use std::fs;
    #[cfg(unix)]
    use std::os::unix::fs::symlink;

    use tempfile::tempdir;

    use super::build;
    use crate::config;

    #[cfg(unix)]
    #[test]
    fn tracked_symlinks_are_analyzed_without_following_their_targets() {
        let repository = tempdir().expect("repository");
        let outside = tempdir().expect("outside");
        let secret = outside.path().join("secret.txt");
        fs::write(&secret, "do not read this target").expect("secret");
        symlink(&secret, repository.path().join("linked.txt")).expect("symlink");

        let (files, skipped) = build(
            repository.path(),
            &["linked.txt".to_string()],
            &config::default_config(),
        )
        .expect("inventory");

        assert_eq!(files.len(), 1);
        assert_eq!(files[0].text, secret.to_string_lossy());
        assert!(!files[0].text.contains("do not read this target"));
        assert_eq!(skipped.missing, 0);
    }

    #[test]
    fn tracked_gitlink_directories_are_skipped_instead_of_read_as_files() {
        let repository = tempdir().expect("repository");
        fs::create_dir_all(repository.path().join("vendor/submodule")).expect("gitlink directory");

        let (files, skipped) = build(
            repository.path(),
            &["vendor/submodule".to_string()],
            &config::default_config(),
        )
        .expect("inventory");

        assert!(files.is_empty());
        assert_eq!(skipped.ignored, 1);
    }

    #[test]
    fn utf8_bom_and_utf16_bom_text_are_decoded_instead_of_marked_binary() {
        let repository = tempdir().expect("repository");
        fs::write(repository.path().join("utf8.txt"), b"\xef\xbb\xbfhello\n").expect("utf8 bom");
        fs::write(
            repository.path().join("utf16.txt"),
            [0xff, 0xfe, b'h', 0, b'i', 0, b'\n', 0],
        )
        .expect("utf16 bom");
        let (files, skipped) = build(
            repository.path(),
            &["utf8.txt".to_string(), "utf16.txt".to_string()],
            &config::default_config(),
        )
        .expect("inventory");
        assert_eq!(files.len(), 2);
        let decoded = files
            .iter()
            .map(|file| (file.path.as_str(), file.text.as_str()))
            .collect::<std::collections::BTreeMap<_, _>>();
        assert_eq!(decoded["utf8.txt"], "hello\n");
        assert_eq!(decoded["utf16.txt"], "hi\n");
        assert_eq!(skipped.binary, 0);
        assert_eq!(skipped.undecodable, 0);
    }
}