use std::collections::{BTreeMap, HashMap};
use std::fs;
use std::path::{Component, Path, PathBuf};
use std::sync::LazyLock;
use std::time::Instant;
use anyhow::{Context, Result, bail};
use chrono::{SecondsFormat, Utc};
use regex::Regex;
use serde::{Deserialize, Serialize};
use serde_json::{Value, json};
use sha2::{Digest, Sha256};
use tiktoken_rs::{
CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
};
use unicode_normalization::UnicodeNormalization;
use unicode_segmentation::UnicodeSegmentation;
use crate::config;
use crate::estimate;
use crate::git;
use crate::health;
use crate::history;
use crate::inventory;
use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
use crate::overlays;
use crate::report;
use crate::scoring;
static CAMEL_CASE_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
static NUMBER_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
#[derive(Debug, Serialize, Deserialize)]
struct CachedTokenData {
token_count: usize,
structural_tokens: Vec<String>,
content_fingerprint: String,
}
fn token_cache_key(text: &str, tokenizer: &str, large_file_bytes: usize, mode: &str) -> String {
let mut digest = Sha256::new();
digest.update(b"git-slop-token-cache-v3\0");
digest.update(tokenizer.as_bytes());
digest.update([0]);
digest.update(large_file_bytes.to_le_bytes());
digest.update(mode.as_bytes());
digest.update([0]);
digest.update(text.as_bytes());
hex::encode(digest.finalize())
}
fn load_cached_tokens(path: &Path) -> Option<CachedTokenData> {
serde_json::from_slice(&fs::read(path).ok()?).ok()
}
fn write_cached_tokens(path: &Path, cached: &CachedTokenData) -> Result<()> {
if let Some(parent) = path.parent() {
fs::create_dir_all(parent)?;
}
let temporary = path.with_extension(format!("json.tmp-{}", std::process::id()));
fs::write(&temporary, serde_json::to_vec(cached)?)?;
fs::rename(&temporary, path)?;
Ok(())
}
fn enforce_cache_limits(root: &Path, max_entries: usize, max_bytes: u64) -> Result<(usize, u64)> {
let mut entries = fs::read_dir(root)
.into_iter()
.flatten()
.filter_map(Result::ok)
.filter_map(|entry| {
let metadata = entry.metadata().ok()?;
metadata
.is_file()
.then_some((entry.path(), metadata.modified().ok(), metadata.len()))
})
.collect::<Vec<_>>();
entries.sort_by_key(|(path, modified, _)| (*modified, path.clone()));
let mut total_bytes = entries.iter().map(|entry| entry.2).sum::<u64>();
let mut total_entries = entries.len();
for (path, _, bytes) in entries {
if total_entries <= max_entries && total_bytes <= max_bytes {
break;
}
if fs::remove_file(&path).is_ok() {
total_entries = total_entries.saturating_sub(1);
total_bytes = total_bytes.saturating_sub(bytes);
}
}
Ok((total_entries, total_bytes))
}
fn replace_quoted_strings(text: &str) -> String {
let mut result = String::with_capacity(text.len());
let mut chars = text.chars().peekable();
let mut previous = None;
while let Some(character) = chars.next() {
if !matches!(character, '\'' | '"' | '`') {
result.push(character);
previous = Some(character);
continue;
}
if character == '\''
&& previous.is_some_and(char::is_alphanumeric)
&& chars.peek().is_some_and(|next| next.is_alphanumeric())
{
result.push(character);
previous = Some(character);
continue;
}
result.push_str(" str ");
let quote = character;
let mut escaped = false;
for next in chars.by_ref() {
if escaped {
escaped = false;
continue;
}
if next == '\\' {
escaped = true;
} else if next == quote {
break;
}
}
previous = Some(' ');
}
result
}
fn structural_mode(path: &str) -> &'static str {
if matches!(
Path::new(path).extension().and_then(|value| value.to_str()),
Some("md" | "mdx" | "txt")
) {
"prose"
} else {
"code"
}
}
fn structural_content_tokens(mode: &str, text: &str) -> Vec<String> {
let normalized: String = text.nfkc().collect();
let normalized = normalized.replace(['\u{2018}', '\u{2019}'], "'");
let normalized = ACRONYM_BOUNDARY_RE.replace_all(&normalized, "$1 $2");
let normalized = CAMEL_CASE_RE.replace_all(&normalized, "$1 $2");
let normalized = normalized.replace(['-', '/'], " ");
let normalized = if mode == "prose" {
normalized
} else {
replace_quoted_strings(&normalized)
};
let normalized = NUMBER_RE.replace_all(&normalized, " 0 ");
let lower = normalized.to_lowercase();
lower
.unicode_words()
.flat_map(|word| word.split('_'))
.map(|word| {
word.split_once('\'')
.filter(|(prefix, suffix)| prefix.chars().count() == 1 && !suffix.is_empty())
.map_or(word, |(_, suffix)| suffix)
})
.filter(|item| item.chars().count() > 1)
.map(ToOwned::to_owned)
.collect()
}
fn structural_path_tokens(path: &str) -> Vec<String> {
path.replace(['-', '_', '.'], "/")
.to_ascii_lowercase()
.split('/')
.filter(|item| !item.is_empty())
.map(ToOwned::to_owned)
.collect()
}
#[cfg(test)]
fn structural_tokens(path: &str, text: &str) -> Vec<String> {
let mut tokens = structural_content_tokens(structural_mode(path), text);
tokens.extend(structural_path_tokens(path));
tokens
}
fn content_fingerprint(text: &str) -> String {
hex::encode(Sha256::digest(text.as_bytes()))
}
fn top_terms(tokens: &[String], limit: usize) -> Vec<String> {
let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
for token in tokens {
*counts.entry(token).or_default() += 1;
}
let mut ranked: Vec<(&str, usize)> = counts.into_iter().collect();
ranked.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(right.0)));
ranked
.into_iter()
.take(limit)
.map(|(term, _)| term.to_string())
.collect()
}
fn has_inline_tests(language: &str, text: &str) -> bool {
match language {
"Rust" => text.contains("#[cfg(test)]") || text.contains("#[test]"),
"Go" => text.contains("func Test") || text.contains("func Benchmark"),
"Python" => text.contains("def test_") || text.contains("class Test"),
"JavaScript" | "JSX" | "TypeScript" | "TSX" => {
text.contains("describe(") || text.contains("test(") || text.contains("it(")
}
"Swift" => text.contains("XCTestCase") || text.contains("@Test"),
_ => false,
}
}
fn configured_context_encoder(config: &Value) -> Result<CoreBPE> {
let tokenizer_name = match config.pointer("/tokenization/context_tokenizer_name") {
Some(Value::String(name)) if !name.trim().is_empty() => name.as_str(),
Some(Value::String(_)) => {
bail!("tokenization.context_tokenizer_name must not be empty")
}
Some(_) => bail!("tokenization.context_tokenizer_name must be a string"),
None => "cl100k_base",
};
let encoder = match tokenizer_name {
"cl100k_base" => cl100k_base(),
"o200k_base" => o200k_base(),
"o200k_harmony" => o200k_harmony(),
"p50k_base" => p50k_base(),
"p50k_edit" => p50k_edit(),
"r50k_base" => r50k_base(),
unsupported => {
bail!(
"unsupported tokenization.context_tokenizer_name {unsupported:?}; \
supported encodings: cl100k_base, o200k_base, o200k_harmony, \
p50k_base, p50k_edit, r50k_base"
)
}
};
encoder.with_context(|| format!("failed to initialize {tokenizer_name} tokenizer"))
}
fn action_queue(files: &[FileAnalysis]) -> Vec<Value> {
let mut files: Vec<&FileAnalysis> = files.iter().collect();
files.sort_by(|left, right| {
right
.slop_score
.total_cmp(&left.slop_score)
.then_with(|| right.tokens.cmp(&left.tokens))
.then_with(|| left.path.cmp(&right.path))
});
files
.into_iter()
.map(|file| {
let non_context_reasons = file.reason_codes.iter().any(|reason| {
!matches!(reason.as_str(), "critical_token_cost" | "high_token_cost")
});
json!({
"path": file.path,
"slop_score": file.slop_score,
"slop_band": file.slop_band,
"context_band": file.context_band,
"tokens": file.tokens,
"age_days": file.age_days,
"revisions_window": file.revisions_window,
"churn_pressure": file.churn_pressure,
"reason_codes": file.reason_codes,
"is_pure_context_hotspot": !file.reason_codes.is_empty() && !non_context_reasons
})
})
.collect()
}
pub fn run_find() -> Result<FindResult> {
let repo_root = git::resolve_repo_root()?;
run_find_in(&repo_root)
}
pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
run_find_scoped(repo_root, false, None, false)
}
pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
run_find_scoped(repo_root, allow_shallow, None, false)
}
#[derive(Debug, Clone, Default)]
pub struct FindOptions {
pub allow_shallow: bool,
pub scope: Option<String>,
pub progress: bool,
pub allow_empty_scope: bool,
}
fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
let Some(raw) = value.map(str::trim) else {
return Ok(None);
};
if raw.is_empty() || raw == "." {
return Ok(None);
}
let path = Path::new(raw);
if path.is_absolute() {
bail!("--scope must be repo-relative, received {raw:?}");
}
let mut parts = Vec::new();
for component in path.components() {
match component {
Component::Normal(part) => parts.push(
part.to_str()
.ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
),
Component::CurDir => {}
Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
bail!("--scope must not escape the repository, received {raw:?}");
}
}
}
let normalized = parts.join("/");
Ok((!normalized.is_empty()).then_some(normalized))
}
fn selected_path_digest(paths: &[String]) -> String {
let mut digest = Sha256::new();
for path in paths {
digest.update(path.as_bytes());
digest.update([0]);
}
hex::encode(digest.finalize())
}
fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
let mut digest = Sha256::new();
for path in paths {
digest.update(path.as_bytes());
digest.update([0]);
let absolute = repo_root.join(path);
let metadata = fs::symlink_metadata(&absolute)
.with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
let bytes = if metadata.file_type().is_symlink() {
fs::read_link(&absolute)
.with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
.to_string_lossy()
.into_owned()
.into_bytes()
} else if metadata.is_dir() {
b"<gitlink>".to_vec()
} else {
fs::read(&absolute)
.with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
};
digest.update(bytes);
digest.update([0]);
}
Ok(hex::encode(digest.finalize()))
}
pub fn run_find_scoped(
repo_root: &Path,
allow_shallow: bool,
scope: Option<&str>,
progress: bool,
) -> Result<FindResult> {
run_find_with_options(
repo_root,
&FindOptions {
allow_shallow,
scope: scope.map(ToOwned::to_owned),
progress,
allow_empty_scope: false,
},
)
}
pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
let allow_shallow = options.allow_shallow;
let scope = options.scope.as_deref();
let progress = options.progress;
let allow_empty_scope = options.allow_empty_scope;
let started = Instant::now();
let phase = |name: &str| {
if progress {
eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
}
};
phase("preflight");
config::ensure_runtime_gitignore(repo_root)?;
let _scan_lock = config::acquire_scan_lock(repo_root)?;
let loaded_config = config::load(repo_root)?;
let mut repo = git::repo_metadata(repo_root)?;
if repo.is_shallow && !allow_shallow {
bail!(
"repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
);
}
let all_tracked_paths = git::list_tracked_files(repo_root)?;
let scope = normalize_scope(scope)?;
if let Some(scope) = scope.as_deref() {
if !repo_root.join(scope).exists() {
bail!("--scope does not exist in the repository: {scope}");
}
}
let tracked_paths = all_tracked_paths
.iter()
.filter(|path| {
scope
.as_deref()
.is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
})
.cloned()
.collect::<Vec<_>>();
if tracked_paths.is_empty() && !allow_empty_scope {
bail!(
"{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
scope.as_deref().map_or_else(
|| "repository".to_string(),
|scope| format!("--scope {scope:?}")
)
);
}
let scope_identity = ScopeIdentity {
mode: if scope.is_some() {
"scoped"
} else {
"repository"
}
.to_string(),
path: scope.clone(),
selected_path_count: tracked_paths.len(),
selected_path_digest: selected_path_digest(&tracked_paths),
};
let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
bail!(
"analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope or raise the explicit budget",
estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
estimate.memory_budget_bytes / 1024 / 1024
);
}
let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
phase("inventory");
let encoder = configured_context_encoder(&loaded_config)?;
let mut token_counts = BTreeMap::new();
let mut line_counts = BTreeMap::new();
let mut token_data = HashMap::new();
let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
.unwrap_or("cl100k_base")
.to_string();
let large_file_bytes =
config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
let cache_root = config::cache_dir(repo_root).join("token-v3");
let mut cache_cleanup_warnings = Vec::new();
for version in ["token-v1", "token-v2"] {
let legacy_cache = config::cache_dir(repo_root).join(version);
if legacy_cache.exists() {
if let Err(error) = fs::remove_dir_all(&legacy_cache) {
cache_cleanup_warnings.push(format!(
"failed to remove legacy cache {}: {error}",
legacy_cache.display()
));
}
}
}
if let Ok(entries) = fs::read_dir(&cache_root) {
for entry in entries.filter_map(Result::ok) {
if entry.file_name().to_string_lossy().contains(".tmp-") {
if let Err(error) = fs::remove_file(entry.path()) {
cache_cleanup_warnings.push(format!(
"failed to remove abandoned cache temporary {}: {error}",
entry.path().display()
));
}
}
}
}
let mut cache_hits = 0usize;
let mut cache_misses = 0usize;
let mut structurally_skipped_large_files = 0usize;
for file in &inventory_files {
if file.bytes > large_file_bytes {
structurally_skipped_large_files += 1;
}
let mode = structural_mode(&file.path);
let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
let cache_path = cache_root.join(format!("{cache_key}.json"));
let cached = if let Some(cached) = load_cached_tokens(&cache_path) {
cache_hits += 1;
cached
} else {
cache_misses += 1;
let cached = CachedTokenData {
token_count: encoder.encode_ordinary(&file.text).len(),
structural_tokens: if file.bytes > large_file_bytes {
Vec::new()
} else {
structural_content_tokens(mode, &file.text)
},
content_fingerprint: content_fingerprint(&file.text),
};
write_cached_tokens(&cache_path, &cached)?;
cached
};
let count = cached.token_count;
let mut structural = cached.structural_tokens;
if file.bytes <= large_file_bytes {
structural.extend(structural_path_tokens(&file.path));
}
let top_term_limit =
config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
let fingerprint = cached.content_fingerprint;
token_counts.insert(file.path.clone(), count);
line_counts.insert(file.path.clone(), file.lines);
token_data.insert(
file.path.clone(),
(
structural.len(),
top_terms(&structural, top_term_limit),
structural,
fingerprint,
),
);
}
let (cache_entries, cache_bytes) = enforce_cache_limits(
&cache_root,
config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
)?;
phase("tokenization");
repo.analyzed_content_digest = Some(starting_content_digest.clone());
let analyzed_paths: Vec<String> = inventory_files
.iter()
.map(|file| file.path.clone())
.collect();
let now = Utc::now();
let (history_by_path, commits, history_diagnostics) = history::analyze_history(
repo_root,
&analyzed_paths,
&token_counts,
&line_counts,
&loaded_config,
now,
)?;
phase("history");
let mut files = Vec::with_capacity(inventory_files.len());
for file in inventory_files {
let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
let inline_tests = has_inline_tests(&file.language, &file.text);
let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
token_data
.remove(&file.path)
.unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
files.push(FileAnalysis {
path: file.path,
bytes: file.bytes,
lines: file.lines,
blank_lines: file.blank_lines,
code_lines: file.code_lines,
comment_lines: file.comment_lines,
language: file.language,
profile: file.profile,
classification: file.classification,
has_inline_tests: inline_tests,
tokens,
context_band: scoring::context_band_for_tokens(tokens, &loaded_config),
context_pressure: scoring::context_pressure_for_tokens(tokens, &loaded_config),
content_fingerprint,
structural_tokens,
structural_token_count,
top_structural_terms,
age_days: history.age_days,
revisions_window: history.revisions_window,
recency_weighted_commits: history.recency_weighted_commits,
added_window: history.added_window,
deleted_window: history.deleted_window,
churn_lines_window: history.line_churn_window,
line_churn_window: history.line_churn_window,
token_churn_window: history.token_churn_window,
relative_churn_window: history.relative_churn_window,
late_churn_spike: history.late_churn_spike,
author_count_window: history.author_count_window,
author_entropy: history.author_entropy,
top_author_share: history.top_author_share,
days_since_non_bot_edit: history.days_since_non_bot_edit,
recent_maintainer_diversity: history.recent_maintainer_diversity,
age_pressure: 0.0,
revision_norm: 0.0,
relative_churn_norm: 0.0,
churn_pressure: 0.0,
slop_score: 0.0,
slop_band: "low".to_string(),
reason_codes: Vec::new(),
costs: json!({}),
overlays: json!({}),
});
}
scoring::apply_scoring(&mut files, &loaded_config);
let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
phase("relationships");
let folders = scoring::build_folder_analyses(&files, &loaded_config);
let queue = action_queue(&files);
let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
let analyzed_revision_at = repo.head_commit_timestamp.clone();
let ending_worktree = git::worktree_state(repo_root)?;
if ending_worktree.digest != repo.worktree_state_digest {
bail!("repository changed during analysis; no mixed-snapshot report was published");
}
if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
bail!(
"selected file content changed during analysis; no mixed-snapshot report was published"
);
}
let analysis = Analysis {
repo_root: PathBuf::from(repo_root),
repo,
config: loaded_config,
generated_at,
analyzed_revision_at,
skipped,
tracked_file_count: all_tracked_paths.len(),
scope: scope_identity,
files,
folders,
organization,
action_queue: queue,
diagnostics: json!({
"analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
"estimate": estimate,
"cache_hits": cache_hits,
"cache_misses": cache_misses,
"cache_entries": cache_entries,
"cache_bytes": cache_bytes,
"cache_cleanup_warnings": cache_cleanup_warnings,
"structurally_skipped_large_files": structurally_skipped_large_files,
"analysis_status": if structurally_skipped_large_files > 0 { "degraded_large_files" } else { "complete" },
"history": history_diagnostics,
"scope": scope
}),
};
let rollup = health::build_health_rollup(&analysis);
let result = report::write_report_bundle(&analysis, &rollup)?;
phase("report writing");
if result.report.get("schema_version").and_then(Value::as_u64) != Some(4) {
bail!("internal error: report writer did not produce schema 4");
}
Ok(result)
}
#[cfg(test)]
mod tests {
use serde_json::json;
use tiktoken_rs::{cl100k_base, r50k_base};
use super::{
action_queue, configured_context_encoder, replace_quoted_strings, structural_tokens,
};
use crate::model::FileAnalysis;
use crate::scoring;
fn file(path: &str, relative_churn: f64) -> FileAnalysis {
FileAnalysis {
path: path.to_string(),
bytes: 400,
lines: 100,
blank_lines: 0,
code_lines: 100,
comment_lines: 0,
language: "Rust".to_string(),
profile: "agent_context".to_string(),
classification: "source".to_string(),
has_inline_tests: false,
tokens: 100,
context_band: "compact".to_string(),
context_pressure: 0.0,
content_fingerprint: String::new(),
structural_tokens: Vec::new(),
structural_token_count: 0,
top_structural_terms: Vec::new(),
age_days: 0,
revisions_window: 1,
recency_weighted_commits: 0.0,
added_window: 0,
deleted_window: 0,
churn_lines_window: 0,
line_churn_window: 0,
token_churn_window: 0,
relative_churn_window: relative_churn,
late_churn_spike: 0.0,
author_count_window: 0,
author_entropy: 0.0,
top_author_share: 0.0,
days_since_non_bot_edit: None,
recent_maintainer_diversity: 0,
age_pressure: 0.0,
revision_norm: 0.0,
relative_churn_norm: 0.0,
churn_pressure: 0.0,
slop_score: 0.0,
slop_band: String::new(),
reason_codes: Vec::new(),
costs: json!({}),
overlays: json!({}),
}
}
#[test]
fn structural_normalization_is_deterministic() {
let tokens = structural_tokens(
"src/my_file.rs",
"let camelCase = \"secret 123\"; // hello-world",
);
assert!(tokens.contains(&"camel".to_string()));
assert!(tokens.contains(&"case".to_string()));
assert!(tokens.contains(&"str".to_string()));
assert!(tokens.contains(&"my".to_string()));
assert_eq!(
replace_quoted_strings("'one' \"two\" `three`"),
" str str str "
);
}
#[test]
fn structural_normalization_preserves_unicode_and_apostrophe_words() {
let tokens = structural_tokens("docs/café.md", "L’équipe can’t rename HTTPServer_value");
assert!(tokens.contains(&"équipe".to_string()));
assert!(tokens.contains(&"can't".to_string()));
assert!(tokens.contains(&"http".to_string()));
assert!(tokens.contains(&"server".to_string()));
assert!(tokens.contains(&"value".to_string()));
assert!(tokens.iter().any(|token| token.contains("café")));
}
#[test]
fn configured_tokenizer_is_used_exactly_and_unknown_names_fail_closed() {
let text = "お誕生日おめでとう";
let configured = configured_context_encoder(&json!({"tokenization": {
"context_tokenizer_name": "r50k_base"
}}))
.unwrap();
assert_eq!(
configured.encode_ordinary(text).len(),
r50k_base().unwrap().encode_ordinary(text).len()
);
assert_ne!(
configured.encode_ordinary(text).len(),
cl100k_base().unwrap().encode_ordinary(text).len()
);
let result = configured_context_encoder(&json!({"tokenization": {
"context_tokenizer_name": "not-a-real-encoding"
}}));
let error = match result {
Ok(_) => panic!("unsupported tokenizer must fail closed"),
Err(error) => error,
};
assert!(
error
.to_string()
.contains("unsupported tokenization.context_tokenizer_name")
);
}
#[test]
fn action_queue_prioritizes_line_relative_churn_signal() {
let mut files = vec![file("src/quiet.rs", 0.1), file("src/volatile.rs", 2.0)];
scoring::apply_scoring(&mut files, &json!({}));
let queue = action_queue(&files);
assert_eq!(queue[0]["path"], "src/volatile.rs");
assert_eq!(queue[0]["reason_codes"][1], "high_relative_churn");
assert_eq!(queue[0]["is_pure_context_hotspot"], false);
}
}