use std::collections::{HashMap, HashSet};
use std::fs;
use std::path::{Path, PathBuf};
use std::sync::{Arc, LazyLock, Mutex};
use std::time::UNIX_EPOCH;
use globset::{GlobBuilder, GlobSet, GlobSetBuilder};
use scryer_db::{ArchitecturalDecision, Symbol};
use scryer_engine::EngineService;
use scryer_engine::search::{
Bm25Builder, Bm25Index, tokenize_identifier, tokenize_query, tokenize_text,
};
use super::adr::{ParsedAdr, parse_adr_markdown};
use super::graph::{EdgeDirection, one_hop_edges, source_file_path};
use crate::context::ProjectContextResolver;
const DOC_DIRS: [&str; 3] = ["docs/adr", "docs/learnings", "docs/plans"];
const ADR_FIELD_WEIGHTS: [f32; 6] = [3.0, 2.0, 1.0, 1.0, 1.0, 1.0];
const PATH_BOOST: f64 = 3.0;
const FILE_SCORE: f64 = 5.0;
const ANCESTOR_SCORE: f64 = 3.0;
const NEIGHBOUR_SCORE: f64 = 2.0;
const TEXT_SCORE_MAX: f64 = 3.0;
const SPECIFICITY_STEP: f64 = 0.2;
const SPECIFICITY_CAP: usize = 5;
pub(crate) const NEIGHBOUR_EDGE_TYPES: [&str; 3] = ["calls", "implements", "instantiates"];
const DOC_SNIPPET_CHARS: usize = 200;
#[derive(Debug)]
struct GlobEntry {
raw: String,
matcher: GlobSet,
literal_prefix: String,
}
impl GlobEntry {
fn specificity(&self) -> usize {
self.literal_prefix
.split('/')
.filter(|s| !s.is_empty())
.count()
}
}
#[derive(Debug, Default)]
pub struct AdrGlobs {
entries: Vec<GlobEntry>,
}
pub(crate) fn normalize_rel_path(path: &str) -> String {
let path = path.trim().replace('\\', "/");
let mut p = path.as_str();
while let Some(rest) = p.strip_prefix("./") {
p = rest;
}
p.trim_start_matches('/').trim_end_matches('/').to_string()
}
impl AdrGlobs {
pub fn new(paths: &[String]) -> Self {
let mut entries = Vec::new();
for raw in paths {
let norm = normalize_rel_path(raw);
if norm.is_empty() {
continue;
}
let meta_at = norm.find(['*', '?', '[', '{']);
let patterns = if meta_at.is_some() {
vec![norm.clone()]
} else {
vec![norm.clone(), format!("{norm}/**")]
};
let mut builder = GlobSetBuilder::new();
let mut ok = true;
for pattern in &patterns {
match GlobBuilder::new(pattern)
.literal_separator(true)
.case_insensitive(true)
.build()
{
Ok(glob) => {
builder.add(glob);
}
Err(err) => {
tracing::warn!("Ignoring invalid ADR affected_path '{raw}': {err}");
ok = false;
}
}
}
let Some(matcher) = ok.then(|| builder.build().ok()).flatten() else {
continue;
};
let literal_prefix = match meta_at {
Some(i) => norm[..i].trim_end_matches('/').to_string(),
None => norm.clone(),
};
entries.push(GlobEntry {
raw: raw.clone(),
matcher,
literal_prefix,
});
}
Self { entries }
}
pub fn is_empty(&self) -> bool {
self.entries.is_empty()
}
pub fn matches(&self, path: &str) -> Option<(&str, usize)> {
let path = normalize_rel_path(path);
self.entries
.iter()
.filter(|e| e.matcher.is_match(&path))
.map(|e| (e.raw.as_str(), e.specificity()))
.max_by_key(|(_, spec)| *spec)
}
pub fn reverse_prefix(&self, query_path: &str) -> Option<(&str, usize)> {
let q = normalize_rel_path(query_path).to_lowercase();
if q.is_empty() {
return None;
}
self.entries
.iter()
.filter(|e| {
let prefix = e.literal_prefix.to_lowercase();
prefix == q || prefix.starts_with(&format!("{q}/"))
})
.map(|e| (e.raw.as_str(), e.specificity()))
.max_by_key(|(_, spec)| *spec)
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum AdrSource {
File,
Db,
}
#[derive(Debug)]
pub struct AdrEntry {
pub adr: ParsedAdr,
pub source: AdrSource,
pub globs: AdrGlobs,
}
#[derive(Debug)]
pub struct AdrCorpus {
pub entries: Vec<AdrEntry>,
index: Bm25Index<usize>,
}
type Fingerprint = (PathBuf, u128, usize);
type CorpusCache = HashMap<u64, (Fingerprint, Arc<AdrCorpus>)>;
static ADR_CORPUS: LazyLock<Mutex<CorpusCache>> = LazyLock::new(Mutex::default);
pub(crate) fn invalidate_adr_corpus(project_id: u64) {
ADR_CORPUS
.lock()
.unwrap_or_else(|e| e.into_inner())
.remove(&project_id);
}
fn corpus_files(root: &Path) -> Vec<PathBuf> {
let mut files = Vec::new();
for dir in DOC_DIRS {
let Ok(entries) = fs::read_dir(root.join(dir)) else {
continue;
};
let mut dir_files: Vec<PathBuf> = entries
.flatten()
.map(|e| e.path())
.filter(|p| p.is_file() && p.extension().and_then(|s| s.to_str()) == Some("md"))
.collect();
dir_files.sort();
files.extend(dir_files);
}
files
}
fn fingerprint(root: &Path, files: &[PathBuf]) -> Fingerprint {
let max_mtime = files
.iter()
.filter_map(|p| fs::metadata(p).and_then(|m| m.modified()).ok())
.filter_map(|t| t.duration_since(UNIX_EPOCH).ok())
.map(|d| d.as_nanos())
.max()
.unwrap_or(0);
(root.to_path_buf(), max_mtime, files.len())
}
pub(crate) async fn load_corpus(
engine: &EngineService,
project_id: u64,
root: &Path,
) -> anyhow::Result<Arc<AdrCorpus>> {
let files = corpus_files(root);
let fp = fingerprint(root, &files);
{
let cache = ADR_CORPUS.lock().unwrap_or_else(|e| e.into_inner());
if let Some((cached_fp, corpus)) = cache.get(&project_id)
&& *cached_fp == fp
{
return Ok(Arc::clone(corpus));
}
}
let corpus = Arc::new(build_corpus(engine, project_id, root, &files).await?);
ADR_CORPUS
.lock()
.unwrap_or_else(|e| e.into_inner())
.insert(project_id, (fp, Arc::clone(&corpus)));
Ok(corpus)
}
async fn build_corpus(
engine: &EngineService,
project_id: u64,
root: &Path,
files: &[PathBuf],
) -> anyhow::Result<AdrCorpus> {
let mut adrs: Vec<(ParsedAdr, AdrSource)> = Vec::new();
let adr_dir = root.join("docs").join("adr");
let mut adr_disk_numbers = HashSet::new();
for path in files {
let Ok(content) = fs::read_to_string(path) else {
continue;
};
let parsed = parse_adr_markdown(path, &content);
if path.starts_with(&adr_dir)
&& let Some(num) = parsed.adr_number
{
adr_disk_numbers.insert(num);
}
adrs.push((parsed, AdrSource::File));
}
let mut guard = engine.db().lock().await;
let db_adrs =
ArchitecturalDecision::filter(ArchitecturalDecision::fields().project_id().eq(project_id))
.exec(&mut *guard)
.await?;
for adr in db_adrs {
if adr_dir.exists() && !adr_disk_numbers.contains(&adr.adr_number) {
let del_sql = format!(
"DELETE FROM architectural_decision WHERE project_id = {project_id} AND id = {};",
adr.id
);
let _ = toasty::sql::statement(&del_sql).exec(&mut *guard).await;
continue;
}
if adrs.iter().any(|(a, _)| a.title == adr.title) {
continue;
}
adrs.push((
ParsedAdr {
file_path: "turso::architectural_decision".to_string(),
adr_number: Some(adr.adr_number),
title: adr.title,
status: adr.status,
context: adr.context,
decision: adr.decision,
consequences: adr.consequences,
body: String::new(),
affected_paths: serde_json::from_str(&adr.affected_paths).unwrap_or_default(),
},
AdrSource::Db,
));
}
drop(guard);
let mut builder = Bm25Builder::new(&ADR_FIELD_WEIGHTS);
let mut entries = Vec::with_capacity(adrs.len());
for (idx, (adr, source)) in adrs.into_iter().enumerate() {
builder.add(
idx,
&[
tokenize_text(&adr.title),
tokenize_text(&adr.decision),
tokenize_text(&adr.context),
tokenize_text(&adr.consequences),
tokenize_text(&adr.body),
tokenize_text(&adr.affected_paths.join(" ")),
],
);
entries.push(AdrEntry {
globs: AdrGlobs::new(&adr.affected_paths),
adr,
source,
});
}
Ok(AdrCorpus {
entries,
index: builder.build(),
})
}
#[derive(Debug, Clone)]
pub struct RankedAdr {
pub idx: usize,
pub score: f64,
pub specificity: usize,
pub match_reasons: Vec<String>,
}
fn sort_ranked(ranked: &mut [RankedAdr]) {
ranked.sort_by(|a, b| {
b.score
.total_cmp(&a.score)
.then_with(|| b.specificity.cmp(&a.specificity))
.then_with(|| a.idx.cmp(&b.idx))
});
}
fn text_reason(terms: &[String]) -> String {
format!("text: {}", terms.join(", "))
}
fn query_as_path(query: &str, root: &Path) -> Option<String> {
let q = query.trim();
if q.is_empty() || q.contains(char::is_whitespace) {
return None;
}
let path = Path::new(q);
if !q.contains('/') && path.extension().is_none() {
return None;
}
let rel = path.strip_prefix(root).unwrap_or(path);
Some(normalize_rel_path(&rel.to_string_lossy()))
}
pub(crate) fn rank_query(corpus: &AdrCorpus, query: &str, root: &Path) -> Vec<RankedAdr> {
let terms = tokenize_query(query);
let mut ranked: HashMap<usize, RankedAdr> = HashMap::new();
for hit in corpus.index.search(&terms, |_| true) {
let idx = *corpus.index.doc(hit.doc_idx);
ranked.insert(
idx,
RankedAdr {
idx,
score: f64::from(hit.score),
specificity: 0,
match_reasons: vec![text_reason(&hit.matched_terms)],
},
);
}
if let Some(path) = query_as_path(query, root) {
for (idx, entry) in corpus.entries.iter().enumerate() {
let matched = entry
.globs
.matches(&path)
.map(|(p, s)| (p, s, "covers"))
.or_else(|| {
entry
.globs
.reverse_prefix(&path)
.map(|(p, s)| (p, s, "is within"))
});
if let Some((pattern, specificity, verb)) = matched {
let r = ranked.entry(idx).or_insert_with(|| RankedAdr {
idx,
score: 0.0,
specificity: 0,
match_reasons: Vec::new(),
});
r.score += PATH_BOOST;
r.specificity = specificity;
r.match_reasons
.insert(0, format!("affected_path '{pattern}' {verb} {path}"));
}
}
}
let mut ranked: Vec<RankedAdr> = ranked.into_values().filter(|r| r.score > 0.0).collect();
sort_ranked(&mut ranked);
ranked
}
#[derive(Debug, Clone, Default)]
pub struct InvariantTarget {
pub symbol: Option<String>,
pub file_path: Option<String>,
pub project: Option<String>,
}
#[derive(Debug)]
pub struct InvariantRanking {
pub target: String,
pub resolved_symbol: Option<String>,
pub file: Option<String>,
pub corpus: Arc<AdrCorpus>,
pub ranked: Vec<RankedAdr>,
}
struct SymbolContext {
qualified_name: String,
file: Option<String>,
neighbours: Vec<(String, &'static str)>,
terms: Vec<String>,
}
async fn resolve_symbol_context(
engine: &EngineService,
project_id: u64,
symbol: &str,
file_hint: Option<&str>,
) -> anyhow::Result<Option<SymbolContext>> {
let mut guard = engine.db().lock().await;
let mut candidates = Symbol::filter(
Symbol::fields()
.project_id()
.eq(project_id)
.and(Symbol::fields().name().eq(symbol)),
)
.exec(&mut *guard)
.await?;
if candidates.is_empty() {
candidates = Symbol::filter(
Symbol::fields()
.project_id()
.eq(project_id)
.and(Symbol::fields().qualified_name().eq(symbol)),
)
.exec(&mut *guard)
.await?;
}
candidates.retain(|s| s.kind != "reexport");
let mut chosen: Option<(Symbol, Option<String>)> = None;
for candidate in candidates {
let path = source_file_path(&mut guard, project_id, candidate.file_id).await?;
let hinted = file_hint.is_some_and(|h| path.as_deref() == Some(h));
if chosen.is_none() || hinted {
chosen = Some((candidate, path));
}
if hinted {
break;
}
}
let Some((sym, file)) = chosen else {
return Ok(None);
};
let mut neighbours: Vec<(String, &'static str)> = Vec::new();
for (direction, label) in [
(EdgeDirection::Inbound, "caller"),
(EdgeDirection::Outbound, "callee"),
] {
let edges = one_hop_edges(
&mut guard,
project_id,
sym.id,
direction,
Some(&NEIGHBOUR_EDGE_TYPES),
)
.await?;
for edge in edges {
let Some(neighbour) = Symbol::filter(
Symbol::fields()
.project_id()
.eq(project_id)
.and(Symbol::fields().id().eq(direction.neighbour(&edge))),
)
.first()
.exec(&mut *guard)
.await?
else {
continue;
};
let Some(path) = source_file_path(&mut guard, project_id, neighbour.file_id).await?
else {
continue;
};
if Some(&path) != file.as_ref() && !neighbours.iter().any(|(p, _)| *p == path) {
neighbours.push((path, label));
}
}
}
let parent = sym
.qualified_name
.rsplit_once("::")
.map(|(p, _)| p.split_once("::").map_or("", |(_, rest)| rest))
.unwrap_or("");
let doc: String = sym
.docstring
.as_deref()
.unwrap_or("")
.chars()
.take(DOC_SNIPPET_CHARS)
.collect();
let mut terms = Vec::new();
for t in tokenize_identifier(&sym.name)
.into_iter()
.chain(tokenize_text(parent))
.chain(tokenize_text(&doc))
{
if !terms.contains(&t) {
terms.push(t);
}
}
Ok(Some(SymbolContext {
qualified_name: sym.qualified_name,
file,
neighbours,
terms,
}))
}
fn ancestors(file: &str) -> Vec<&str> {
let mut out = Vec::new();
let mut cur = file;
while let Some((parent, _)) = cur.rsplit_once('/') {
out.push(parent);
cur = parent;
}
out
}
pub(crate) async fn rank_invariants_for_symbol(
context: &ProjectContextResolver,
engine: &EngineService,
target: InvariantTarget,
) -> anyhow::Result<InvariantRanking> {
let file_arg = target.file_path.as_deref().map(Path::new);
let (project, rel_path) = context
.resolve_project(file_arg, target.project.as_deref())
.await?;
let root = PathBuf::from(&project.root_path);
let corpus = load_corpus(engine, project.id, &root).await?;
let rel_file = rel_path
.or_else(|| {
let p = Path::new(target.file_path.as_deref()?);
Some(p.strip_prefix(&root).unwrap_or(p).to_path_buf())
})
.map(|p| normalize_rel_path(&p.to_string_lossy()));
let label = target
.symbol
.clone()
.or_else(|| target.file_path.clone())
.unwrap_or_else(|| "project".to_string());
let (resolved_symbol, file, neighbours, terms) = match &target.symbol {
Some(symbol) => {
match resolve_symbol_context(engine, project.id, symbol, rel_file.as_deref()).await? {
Some(ctx) => (
Some(ctx.qualified_name),
ctx.file.or(rel_file),
ctx.neighbours,
ctx.terms,
),
None => (None, rel_file, Vec::new(), tokenize_query(symbol)),
}
}
None => (None, rel_file, Vec::new(), Vec::new()),
};
let mut ranked: HashMap<usize, RankedAdr> = HashMap::new();
for (idx, entry) in corpus.entries.iter().enumerate() {
if entry.globs.is_empty() {
continue;
}
let mut r = RankedAdr {
idx,
score: 0.0,
specificity: 0,
match_reasons: Vec::new(),
};
let mut covers_file = false;
if let Some(file) = &file {
if let Some((pattern, spec)) = entry.globs.matches(file) {
covers_file = true;
r.score += FILE_SCORE;
r.specificity = spec;
r.match_reasons.push(format!(
"affected_path '{pattern}' covers symbol file {file}"
));
} else if let Some((dir, pattern, spec)) = ancestors(file)
.into_iter()
.find_map(|dir| entry.globs.matches(dir).map(|(p, s)| (dir, p, s)))
{
r.score += ANCESTOR_SCORE;
r.specificity = spec;
r.match_reasons
.push(format!("affected_path '{pattern}' covers {dir}/"));
}
}
r.score += SPECIFICITY_STEP * r.specificity.min(SPECIFICITY_CAP) as f64;
let neighbour_match = if covers_file {
None
} else {
neighbours.iter().find_map(|(path, label)| {
entry
.globs
.matches(path)
.map(|(pattern, _)| (path, label, pattern))
})
};
if let Some((path, label, pattern)) = neighbour_match {
r.score += NEIGHBOUR_SCORE;
r.match_reasons
.push(format!("affected_path '{pattern}' covers {label} {path}"));
}
if r.score > 0.0 {
ranked.insert(idx, r);
}
}
let hits = corpus.index.search(&terms, |_| true);
let max = hits.first().map(|h| f64::from(h.score)).unwrap_or(0.0);
if max > 0.0 {
for hit in hits {
let idx = *corpus.index.doc(hit.doc_idx);
let r = ranked.entry(idx).or_insert_with(|| RankedAdr {
idx,
score: 0.0,
specificity: 0,
match_reasons: Vec::new(),
});
r.score += TEXT_SCORE_MAX * f64::from(hit.score) / max;
r.match_reasons.push(text_reason(&hit.matched_terms));
}
}
let mut ranked: Vec<RankedAdr> = ranked.into_values().collect();
sort_ranked(&mut ranked);
Ok(InvariantRanking {
target: label,
resolved_symbol,
file,
corpus,
ranked,
})
}
#[cfg(test)]
mod tests {
use super::*;
fn globs(paths: &[&str]) -> AdrGlobs {
AdrGlobs::new(&paths.iter().map(|p| p.to_string()).collect::<Vec<_>>())
}
#[test]
fn bare_directory_matches_itself_and_descendants() {
let g = globs(&["crates/scryer-db"]);
assert!(g.matches("crates/scryer-db").is_some());
assert!(g.matches("crates/scryer-db/src/db.rs").is_some());
assert!(g.matches("crates/scryer-dbx/src/db.rs").is_none());
let g = globs(&["./crates/scryer-db/"]);
assert!(g.matches("/crates/scryer-db/src/db.rs").is_some());
}
#[test]
fn double_star_and_single_star() {
let g = globs(&["crates/scryer-engine/**"]);
assert!(
g.matches("crates/scryer-engine/src/search/bm25.rs")
.is_some()
);
assert!(g.matches("crates/scryer-mcp/src/server.rs").is_none());
let g = globs(&["crates/*/Cargo.toml"]);
assert!(g.matches("crates/scryer-db/Cargo.toml").is_some());
assert!(g.matches("crates/a/b/Cargo.toml").is_none());
}
#[test]
fn single_file_and_case_insensitivity() {
let g = globs(&["crates/scryer-mcp/src/tools/adr.rs"]);
assert!(g.matches("crates/scryer-mcp/src/tools/adr.rs").is_some());
assert!(g.matches("Crates/Scryer-MCP/src/tools/ADR.rs").is_some());
assert!(
g.matches("crates/scryer-mcp/src/tools/adr_rank.rs")
.is_none()
);
}
#[test]
fn most_specific_entry_wins_and_reverse_prefix() {
let g = globs(&["crates/**", "crates/scryer-mcp/src/daemon/**"]);
let (pattern, spec) = g.matches("crates/scryer-mcp/src/daemon/mod.rs").unwrap();
assert_eq!(pattern, "crates/scryer-mcp/src/daemon/**");
assert_eq!(spec, 4);
assert_eq!(
g.reverse_prefix("crates/scryer-mcp").map(|(p, _)| p),
Some("crates/scryer-mcp/src/daemon/**")
);
assert!(g.reverse_prefix("crates/scryer-db").is_none());
}
#[test]
fn invalid_globs_are_skipped() {
let g = globs(&["crates/[oops", "", "docs/adr"]);
assert!(g.matches("docs/adr/0001.md").is_some());
}
#[test]
fn ancestors_deepest_first() {
assert_eq!(ancestors("a/b/c.rs"), vec!["a/b", "a"]);
assert!(ancestors("c.rs").is_empty());
}
#[test]
fn path_like_queries() {
let root = Path::new("/repo");
assert_eq!(
query_as_path("/repo/crates/x/src/lib.rs", root).as_deref(),
Some("crates/x/src/lib.rs")
);
assert_eq!(
query_as_path("Cargo.toml", root).as_deref(),
Some("Cargo.toml")
);
assert!(query_as_path("logging stdout", root).is_none());
assert!(query_as_path("Storage", root).is_none());
}
}