Skip to main content

cli/
ingest_git.rs

1//! `mushroomdb ingest-git <db> <repo>`: build and maintain a graph of a git
2//! repository. First run ingests the whole history; later runs apply only the
3//! commits after the recorded `GitSync` head, so deletes and renames retract
4//! or move derived edges instead of leaving them stale.
5//!
6//! Graph shape:
7//!
8//! | Label | key (`id`) | props |
9//! |---|---|---|
10//! | `Author` | mailmap-resolved email | `name`, `name_counts` |
11//! | `Commit` | full sha | `message`, `ts`, `author_id`, `pr_id` (with `--prs`) |
12//! | `File` | path | `path`, `dir`, `ext`, `commits`, `n_commits`, `top_author_id`, `author_counts`, plus the working-tree props in [`structure`](crate::structure) |
13//! | `Symbol` | `"<path>#<qualified name>"` | see [`structure`](crate::structure) |
14//! | `PR` | `"pr:<number>"` | `number`, `title`, `url`, `merged_at`, `author_login` |
15//! | `GitSync` | `"__mushroomdb_git_sync__"` | `sha`, `synced_at`, `repo`, `recurse`, `prs`, `structure`, `docs` |
16//!
17//! Edges: user `TOUCHED` Commit→File and `MERGED_AS` PR→Commit, auto-FK
18//! `AUTHOR` Commit→Author, `TOP_AUTHOR` File→Author and `PR` Commit→PR,
19//! rule-derived `CO_CHANGED` File→File and `KNOWS` Author→File — and, from the
20//! working-tree pass, `DEFINES`, `IMPORTS`, `CALLS` and `MENTIONS`.
21//!
22//! With `--recurse-submodules` each initialised submodule is walked as its own
23//! *unit*: its file keys carry the submodule's path in the parent, and it
24//! resumes from its own `GitSync` marker.
25use crate::structure;
26use crate::CliError;
27use core_api::{
28    default_max_edges, Direction, GraphError, IngestOptions, Predicate, ResultSet, RuleDef,
29    SharedDb, Value, WriteGuard, WRITE_LOCK_WAIT,
30};
31use std::collections::{BTreeMap, BTreeSet};
32use std::path::{Path, PathBuf};
33use std::process::Command;
34
35/// Cap on the `commits` list stored per file. Bounds both node size and the
36/// cost of the jaccard overlap the `co_changed` rule runs over that list.
37pub const DEFAULT_MAX_COMMITS_PER_FILE: usize = 200;
38
39/// Applied when the user names no `--exclude` pattern of their own.
40///
41/// Defined in core-api because the `impact` MCP tool builds its default file
42/// list straight off a working tree and has to leave out exactly what this
43/// ingest leaves out; two lists would let the tool ask about files no store was
44/// ever going to hold.
45pub use core_api::repograph::DEFAULT_EXCLUDES;
46
47/// Minimum jaccard overlap of two files' `commits` lists for `CO_CHANGED`.
48const CO_CHANGE_MIN: f64 = 0.25;
49
50/// WAL bytes past the last snapshot that make writing a new one worthwhile.
51///
52/// Every open reads `wal.bin` whole and replays it frame by frame, so the tail
53/// is paid again on every hook, every MCP start and every CLI call — while a
54/// snapshot is paid once by the run that writes it. Measured on this
55/// repository, an 8.1 MB tail costs 156 ms per open against ~440 ms to
56/// snapshot it away, so a tail this size pays its snapshot back within three
57/// opens. Below the threshold the replay is cheap enough that snapshotting on
58/// every incremental run would cost more than it saves.
59pub const SNAPSHOT_WAL_BYTES: u64 = 4 * 1024 * 1024;
60
61/// Key of the singleton `GitSync` node holding the last ingested sha.
62///
63/// Node keys are a single namespace shared with `File` keys, which are repo
64/// paths — so this cannot be `"HEAD"`. A repository with a file named `HEAD`
65/// (git's own `.git/HEAD` aside, plenty of projects ship one) would otherwise
66/// have the sha written onto its `File` node, leaving no sync marker and
67/// forcing a full re-ingest on every run.
68pub(crate) const SYNC_KEY: &str = "__mushroomdb_git_sync__";
69
70/// What the CLI prints, and exits 3 on, when another process holds the store's
71/// cross-process write lock. Nothing was written, so retrying is always safe.
72pub const BUSY_MESSAGE: &str = "another mushroomdb process is writing; retry";
73
74/// Name of the `Commit.pr_id` foreign-key rule. Identical to the name the
75/// zero-config FK inference would choose, so the two never both create it.
76const PR_FK_RULE: &str = "auto_fk_commit_pr_id";
77
78#[derive(Debug, Clone, PartialEq, Eq)]
79pub struct IngestGitOpts {
80    pub repo: PathBuf,
81    /// Paths to skip. Pattern ending in `/` = path prefix, pattern starting
82    /// with `*.` = extension, otherwise = substring of the path.
83    pub exclude: Vec<String>,
84    pub max_commits_per_file: usize,
85    /// Walk every initialised submodule as its own unit.
86    pub recurse_submodules: bool,
87    /// Ask `gh` for merged pull requests and link them to their commits.
88    pub prs: bool,
89    /// Read the working tree: content hashes, `Symbol` nodes, imports and
90    /// calls. Off with `--no-structure`. Recorded on the `GitSync` node.
91    pub structure: bool,
92    /// Index Markdown bodies, headings and mentions. Off with `--no-docs`, and
93    /// inert without `structure`. Recorded on the `GitSync` node.
94    pub docs: bool,
95    /// Add the database directory to the repository's `.gitignore`.
96    pub ensure_gitignore: bool,
97}
98
99#[derive(Debug, Default, Clone, PartialEq, Eq)]
100pub struct IngestGitReport {
101    pub commits: usize,
102    pub files: usize,
103    pub authors: usize,
104    pub renamed: usize,
105    pub deleted: usize,
106    pub incremental: bool,
107    pub rules_created: Vec<String>,
108    /// Submodules walked as their own units.
109    pub submodules: usize,
110    /// Pull requests inserted by this run.
111    pub prs: usize,
112    /// Whether this run appended the database directory to `.gitignore`.
113    pub gitignore_added: bool,
114    /// What the working-tree pass saw. All zeros with `--no-structure`.
115    pub structure: crate::structure::StructureReport,
116}
117
118/// Paths the commit walk left for the working-tree pass to look at.
119#[derive(Default)]
120struct StructureWork {
121    /// Files added, modified, or renamed *into* this window's paths.
122    touched: BTreeSet<String>,
123    /// Keys that no longer name a file: renamed away, or deleted. Any file
124    /// whose `imports` or `mentions` list still holds one of these has to be
125    /// extracted again, or the edge it derived stays behind.
126    stale: BTreeSet<String>,
127}
128
129/// One git working tree walked by a run: the repository itself, or one of its
130/// submodules.
131///
132/// A submodule's paths are keys under `prefix` (its path in the parent), and it
133/// carries its own sync marker, so the two histories advance independently.
134#[derive(Debug, Clone)]
135struct RepoUnit {
136    path: PathBuf,
137    /// `""` for the repository itself, `"<displaypath>/"` for a submodule.
138    prefix: String,
139    sync_key: String,
140}
141
142#[derive(Debug)]
143enum Change {
144    Added(String),
145    Modified(String),
146    Deleted(String),
147    Renamed { from: String, to: String },
148}
149
150#[derive(Debug)]
151struct GitCommit {
152    sha: String,
153    author_name: String,
154    author_email: String,
155    ts: i64,
156    subject: String,
157    changes: Vec<Change>,
158}
159
160/// Simple, dependency-free path matcher. Documented in `docs/site/ingest-git.md`.
161///
162/// The matcher itself is `core_api::repograph::path_excluded`, next to
163/// [`DEFAULT_EXCLUDES`], because the `impact` MCP tool filters a working tree
164/// with the same patterns and must read them the same way.
165fn excluded(path: &str, patterns: &[String]) -> bool {
166    core_api::repograph::path_excluded(path, patterns)
167}
168
169fn git_output(repo: &Path, args: &[&str]) -> Result<std::process::Output, CliError> {
170    Command::new("git")
171        .arg("-C")
172        .arg(repo)
173        .args(args)
174        .output()
175        .map_err(|e| CliError(format!("cannot run git in {}: {e}", repo.display())))
176}
177
178/// `git log --reverse --name-status -M --format=<RS>%H<US>%aN<US>%aE<US>%at<US>%s <range>`
179///
180/// Returns oldest commit first. The walk ends at `head` — never at the symbolic
181/// `HEAD` — so the range is pinned to the same sha the caller will record as the
182/// sync marker. See [`head_sha`] for why that matters.
183///
184/// `%aN` and `%aE` are the mailmap-resolved name and address, so a repository
185/// with a `.mailmap` reports one identity for a contributor who has committed
186/// under several addresses. Without one they are exactly `%an`/`%ae`.
187fn read_log(repo: &Path, since: Option<&str>, head: &str) -> Result<Vec<GitCommit>, CliError> {
188    if let Some(s) = since {
189        let spec = format!("{s}^{{commit}}");
190        if !git_output(repo, &["cat-file", "-e", &spec])?
191            .status
192            .success()
193        {
194            return Err(CliError(format!(
195                "recorded sync head {s} is not in {} (history rewritten?); \
196                 ingest into a fresh database directory",
197                repo.display()
198            )));
199        }
200    }
201    let mut cmd = Command::new("git");
202    cmd.arg("-C").arg(repo).args([
203        // Without this git renders any non-ASCII byte in a path as an octal
204        // escape, so `src/café.rs` would be stored under a mangled key that no
205        // later run matches. Paths containing a tab or newline stay quoted and
206        // escaped either way — git has to, or they would break the format below.
207        "-c",
208        "core.quotePath=false",
209        "log",
210        "--reverse",
211        "--name-status",
212        "-M",
213        "--no-color",
214        // Record separator \x1e between commits, unit separator \x1f between
215        // header fields. A commit subject containing either byte splits its own
216        // record: the message is truncated at the first \x1f, and a \x1e drops
217        // the remainder of that commit's header. Accepted — the parse degrades
218        // to a skipped or shortened message, never a panic or a wrong sha.
219        "--format=%x1e%H%x1f%aN%x1f%aE%x1f%at%x1f%s",
220    ]);
221    match since {
222        Some(s) => cmd.arg(format!("{s}..{head}")),
223        None => cmd.arg(head),
224    };
225    let out = cmd
226        .output()
227        .map_err(|e| CliError(format!("cannot run git: {e}")))?;
228    if !out.status.success() {
229        return Err(CliError(format!(
230            "git log failed: {}",
231            String::from_utf8_lossy(&out.stderr).trim()
232        )));
233    }
234    let text = String::from_utf8_lossy(&out.stdout);
235    let mut commits = Vec::new();
236    for block in text.split('\x1e').filter(|b| !b.trim().is_empty()) {
237        let mut lines = block.lines();
238        let header = lines.next().unwrap_or("");
239        let f: Vec<&str> = header.split('\x1f').collect();
240        if f.len() < 5 {
241            continue;
242        }
243        let mut changes = Vec::new();
244        for l in lines {
245            let cols: Vec<&str> = l.split('\t').collect();
246            match cols.as_slice() {
247                [s, p] if s.starts_with('A') => changes.push(Change::Added((*p).to_string())),
248                [s, p] if s.starts_with('M') || s.starts_with('T') => {
249                    changes.push(Change::Modified((*p).to_string()))
250                }
251                [s, p] if s.starts_with('D') => changes.push(Change::Deleted((*p).to_string())),
252                [s, from, to] if s.starts_with('R') => changes.push(Change::Renamed {
253                    from: (*from).to_string(),
254                    to: (*to).to_string(),
255                }),
256                // A copy is a brand-new path with no prior history here.
257                [s, _from, to] if s.starts_with('C') => {
258                    changes.push(Change::Added((*to).to_string()))
259                }
260                _ => {}
261            }
262        }
263        commits.push(GitCommit {
264            sha: f[0].into(),
265            author_name: f[1].into(),
266            author_email: f[2].into(),
267            ts: f[3].parse().unwrap_or(0),
268            subject: f[4].into(),
269            changes,
270        });
271    }
272    Ok(commits)
273}
274
275/// Resolve `HEAD` to a concrete sha. `Ok(None)` means the repository has no
276/// commits yet; a path that is not a repository at all is an error.
277///
278/// Called **before** [`read_log`], and the resulting sha is both the end of the
279/// walk and the recorded sync marker. Resolving it afterwards instead would open
280/// a window: a commit landing between the walk and the `rev-parse` would push
281/// the marker past a commit that was never ingested, and every later run would
282/// skip it silently. Pinning both to one sha closes that window — a commit that
283/// lands mid-run is simply outside this range and gets picked up next time.
284///
285/// Asking git also beats taking `log.last()`. The two agree in practice, since a
286/// reachability walk emits its tip first and reversing puts it last, but that is
287/// a property of the traversal and of this module's parser rather than something
288/// the resume marker should rest on.
289fn head_sha(repo: &Path) -> Result<Option<String>, CliError> {
290    let out = git_output(repo, &["rev-parse", "--verify", "-q", "HEAD^{commit}"])?;
291    if !out.status.success() {
292        // No commits yet, or not a repository at all — only the latter is an error.
293        if !git_output(repo, &["rev-parse", "--git-dir"])?
294            .status
295            .success()
296        {
297            return Err(CliError(format!(
298                "not a git repository: {}",
299                repo.display()
300            )));
301        }
302        return Ok(None);
303    }
304    let sha = String::from_utf8_lossy(&out.stdout).trim().to_string();
305    if sha.is_empty() {
306        return Err(CliError(format!(
307            "git could not resolve HEAD in {}",
308            repo.display()
309        )));
310    }
311    Ok(Some(sha))
312}
313
314/// Absolute, symlink-free form of `p`, falling back to `p` when it cannot be
315/// resolved (a path that does not exist yet, or a permission error). The result
316/// is what `GitSync.repo` records, so a later run can find the repository again
317/// from any working directory.
318fn canonical(p: &Path) -> PathBuf {
319    std::fs::canonicalize(p).unwrap_or_else(|_| p.to_path_buf())
320}
321
322/// The submodule paths recorded in a unit's `.gitmodules`, relative to that
323/// unit.
324///
325/// These are the paths git reports as ordinary changes in `--name-status`
326/// output while being gitlinks — a commit pointer, not a file. They get no
327/// `File` node whether or not the submodule is walked, and the list is read
328/// from configuration so it is the same for an uninitialised submodule.
329fn gitlink_paths(unit: &Path) -> BTreeSet<String> {
330    let out = Command::new("git")
331        .arg("config")
332        .arg("--file")
333        .arg(unit.join(".gitmodules"))
334        .args(["--get-regexp", r"^submodule\..*\.path$"])
335        .output();
336    let Ok(out) = out else { return BTreeSet::new() };
337    if !out.status.success() {
338        return BTreeSet::new(); // no .gitmodules, so no submodules
339    }
340    String::from_utf8_lossy(&out.stdout)
341        .lines()
342        .filter_map(|l| l.split_once(' '))
343        .map(|(_, path)| path.trim().to_string())
344        .filter(|p| !p.is_empty())
345        .collect()
346}
347
348/// The repository itself, then one unit per initialised submodule when
349/// `recurse` is set.
350///
351/// `git submodule foreach` visits only initialised submodules and says nothing
352/// about the rest, which is the behaviour wanted here: a submodule that was
353/// never checked out has no working tree to walk. `$displaypath` is relative to
354/// the top-level repository, so it is exactly the key prefix its files need.
355fn repo_units(repo: &Path, recurse: bool) -> Result<Vec<RepoUnit>, CliError> {
356    let mut units = vec![RepoUnit {
357        path: repo.to_path_buf(),
358        prefix: String::new(),
359        sync_key: SYNC_KEY.to_string(),
360    }];
361    if !recurse {
362        return Ok(units);
363    }
364    let out = git_output(
365        repo,
366        &[
367            "submodule",
368            "foreach",
369            "--quiet",
370            "--recursive",
371            "echo \"$displaypath\"",
372        ],
373    )?;
374    if !out.status.success() {
375        return Ok(units);
376    }
377    let mut paths: Vec<String> = String::from_utf8_lossy(&out.stdout)
378        .lines()
379        .map(|l| l.trim_end_matches('/').trim().to_string())
380        .filter(|l| !l.is_empty())
381        .collect();
382    paths.sort();
383    paths.dedup();
384    for dp in paths {
385        let path = repo.join(&dp);
386        // `foreach` already skips the uninitialised, but a stale entry or a
387        // removed checkout would otherwise fail the whole run.
388        if !git_output(&path, &["rev-parse", "--git-dir"])?
389            .status
390            .success()
391        {
392            continue;
393        }
394        units.push(RepoUnit {
395            prefix: format!("{dp}/"),
396            sync_key: format!("{SYNC_KEY}:{dp}"),
397            path,
398        });
399    }
400    Ok(units)
401}
402
403/// Rewrite one unit's changes into repository-wide keys: drop gitlinks, then
404/// prepend the unit's prefix so a submodule's `src/lib.rs` is stored under
405/// `vendor/lib/src/lib.rs`.
406///
407/// Done before anything else looks at the log, so exclusion patterns, the
408/// stored keys and the `File` state query all speak the same path.
409fn localise(log: &mut [GitCommit], prefix: &str, gitlinks: &BTreeSet<String>) {
410    let keep = |p: &String| !gitlinks.contains(p.as_str());
411    for c in log.iter_mut() {
412        c.changes.retain(|ch| match ch {
413            Change::Added(p) | Change::Modified(p) | Change::Deleted(p) => keep(p),
414            Change::Renamed { from, to } => keep(from) && keep(to),
415        });
416        if prefix.is_empty() {
417            continue;
418        }
419        for ch in c.changes.iter_mut() {
420            match ch {
421                Change::Added(p) | Change::Modified(p) | Change::Deleted(p) => {
422                    *p = format!("{prefix}{p}")
423                }
424                Change::Renamed { from, to } => {
425                    *from = format!("{prefix}{from}");
426                    *to = format!("{prefix}{to}");
427                }
428            }
429        }
430    }
431}
432
433/// Append `<db dir>/` to the repository's `.gitignore` unless it is already
434/// listed, creating the file if it does not exist. Returns whether it was
435/// written.
436///
437/// A database kept outside the repository is left alone: the repository has no
438/// business ignoring a path it does not contain.
439fn ensure_gitignore(repo: &Path, db_dir: &Path) -> Result<bool, CliError> {
440    let Ok(rel) = canonical(db_dir)
441        .strip_prefix(repo)
442        .map(|p| p.to_path_buf())
443    else {
444        return Ok(false);
445    };
446    if rel.as_os_str().is_empty() {
447        return Ok(false);
448    }
449    let line = format!("{}/", rel.to_string_lossy().replace('\\', "/"));
450    let path = repo.join(".gitignore");
451    let current = match std::fs::read_to_string(&path) {
452        Ok(s) => s,
453        Err(e) if e.kind() == std::io::ErrorKind::NotFound => String::new(),
454        Err(e) => return Err(CliError(format!("cannot read {}: {e}", path.display()))),
455    };
456    let bare = line.trim_end_matches('/');
457    if current
458        .lines()
459        .map(|l| l.trim())
460        .any(|l| l == line || l == bare || l == format!("/{line}") || l == format!("/{bare}"))
461    {
462        return Ok(false);
463    }
464    let mut next = current;
465    if !next.is_empty() && !next.ends_with('\n') {
466        next.push('\n');
467    }
468    next.push_str(&line);
469    next.push('\n');
470    std::fs::write(&path, next)
471        .map_err(|e| CliError(format!("cannot write {}: {e}", path.display())))?;
472    Ok(true)
473}
474
475/// Separator inside an `author_counts` or `name_counts` entry. Neither an email
476/// address nor a git author name can contain a tab, so `<text>\tcount`
477/// round-trips unambiguously.
478const AUTHOR_COUNT_SEP: char = '\t';
479
480/// How many commits carried each spelling of one email's `%aN`, in the order
481/// the spellings were first seen.
482///
483/// One person routinely commits under more than one name — `Ada Lovelace` at
484/// work and `Ada M. Lovelace` from a laptop that was configured once and never
485/// again. Merging them on email is right, and every tool that prints an author
486/// then has to choose *which* of the names to show. Showing the first one
487/// walked means a display name decided by the oldest commit in the window,
488/// which on this repository labelled an identity with a name carried by 21% of
489/// its commits.
490#[derive(Default, Clone, Debug, PartialEq, Eq)]
491struct AuthorState {
492    /// `(name, commits)` in first-seen order, which is what breaks a tie.
493    names: Vec<(String, usize)>,
494}
495
496impl AuthorState {
497    /// Credit one more commit to `name`.
498    fn touch(&mut self, name: &str) {
499        self.add(name, 1);
500    }
501
502    /// Credit `n` more commits to `name`, appending it if it is new. Order is
503    /// preserved, so the entry that was seen first stays first.
504    fn add(&mut self, name: &str, n: usize) {
505        match self.names.iter_mut().find(|(held, _)| held == name) {
506            Some(entry) => entry.1 += n,
507            None => self.names.push((name.to_string(), n)),
508        }
509    }
510
511    /// Fold another window's counts into this one, keeping first-seen order.
512    fn absorb(&mut self, other: &AuthorState) {
513        for (name, n) in &other.names {
514            self.add(name, *n);
515        }
516    }
517
518    /// The name on the most commits. A tie goes to the name seen first, which
519    /// is why the strict `>` matters: it never unseats an equal incumbent.
520    fn display_name(&self) -> String {
521        let mut best: Option<&(String, usize)> = None;
522        for entry in &self.names {
523            if best.is_none_or(|(_, n)| entry.1 > *n) {
524                best = Some(entry);
525            }
526        }
527        best.map(|(name, _)| name.clone()).unwrap_or_default()
528    }
529
530    /// The distribution as an `Author.name_counts` prop: `"name\tcount"`
531    /// strings in first-seen order.
532    ///
533    /// Stored for the same reason `File.author_counts` is: an incremental run
534    /// sees only the new window, and the `name` prop alone cannot say how far
535    /// ahead the incumbent spelling is. Without it every sync would relabel the
536    /// identity from a handful of commits.
537    fn name_counts_value(&self) -> Value {
538        Value::List(
539            self.names
540                .iter()
541                .map(|(name, n)| Value::Str(format!("{name}{AUTHOR_COUNT_SEP}{n}")))
542                .collect(),
543        )
544    }
545
546    /// Inverse of [`AuthorState::name_counts_value`]. Entries that are not
547    /// `name<TAB>count` are skipped rather than failing the run.
548    fn set_name_counts(&mut self, list: &[Value]) {
549        for v in list {
550            let Value::Str(s) = v else { continue };
551            let Some((name, n)) = s.rsplit_once(AUTHOR_COUNT_SEP) else {
552                continue;
553            };
554            let Ok(n) = n.parse::<usize>() else { continue };
555            if !name.is_empty() {
556                self.add(name, n);
557            }
558        }
559    }
560}
561
562fn file_props(st: &FileState, path: &str) -> Vec<(String, Value)> {
563    let commits = &st.commits;
564    let dir = path
565        .rsplit_once('/')
566        .map(|(d, _)| d)
567        .unwrap_or("")
568        .to_string();
569    let ext = path
570        .rsplit_once('.')
571        .map(|(_, e)| e)
572        .unwrap_or("")
573        .to_string();
574    vec![
575        ("id".into(), Value::Str(path.into())),
576        ("path".into(), Value::Str(path.into())),
577        ("dir".into(), Value::Str(dir)),
578        ("ext".into(), Value::Str(ext)),
579        (
580            "commits".into(),
581            Value::List(commits.iter().map(|s| Value::Str(s.clone())).collect()),
582        ),
583        // The true total, which past `--max-commits-per-file` is larger than
584        // the `commits` list it is stored beside. It is also what
585        // `author_counts` sums to, so the two props agree at any history
586        // length.
587        ("n_commits".into(), Value::Int(st.n_commits as i64)),
588        ("top_author_id".into(), Value::Str(st.top_author())),
589        // Written on every touch so the next incremental run can rebuild the
590        // distribution instead of crediting the whole history to the incumbent.
591        ("author_counts".into(), st.author_counts_value()),
592    ]
593}
594
595/// In-memory per-file state accumulated while walking commits.
596#[derive(Default, Clone)]
597struct FileState {
598    commits: Vec<String>,
599    by_author: BTreeMap<String, usize>,
600    n_commits: usize,
601}
602
603impl FileState {
604    fn touch(&mut self, sha: &str, author: &str, cap: usize) {
605        self.commits.push(sha.to_string());
606        if cap > 0 && self.commits.len() > cap {
607            self.commits.remove(0);
608        }
609        self.n_commits += 1;
610        *self.by_author.entry(author.to_string()).or_default() += 1;
611    }
612
613    /// Most commits wins; ties break on the lexicographically smallest email
614    /// so the result is deterministic across runs.
615    fn top_author(&self) -> String {
616        self.by_author
617            .iter()
618            .max_by(|a, b| a.1.cmp(b.1).then(b.0.cmp(a.0)))
619            .map(|(a, _)| a.clone())
620            .unwrap_or_default()
621    }
622
623    /// The per-author distribution as a `File.author_counts` prop: a list of
624    /// `"email\tcount"` strings in email order.
625    ///
626    /// This is the state an incremental run needs and cannot recompute — the
627    /// walk only sees the new window, and `top_author_id` alone cannot say how
628    /// far ahead the incumbent is. Without it a challenger's commits reset on
629    /// every sync and ownership can never change.
630    fn author_counts_value(&self) -> Value {
631        Value::List(
632            self.by_author
633                .iter()
634                .map(|(email, n)| Value::Str(format!("{email}{AUTHOR_COUNT_SEP}{n}")))
635                .collect(),
636        )
637    }
638
639    /// Inverse of [`FileState::author_counts_value`]. Entries that are not
640    /// `email<TAB>count` are skipped rather than failing the run.
641    fn set_author_counts(&mut self, list: &[Value]) {
642        for v in list {
643            let Value::Str(s) = v else { continue };
644            let Some((email, n)) = s.rsplit_once(AUTHOR_COUNT_SEP) else {
645                continue;
646            };
647            let Ok(n) = n.parse::<usize>() else { continue };
648            if !email.is_empty() {
649                *self.by_author.entry(email.to_string()).or_default() += n;
650            }
651        }
652    }
653}
654
655/// One merged pull request as reported by `gh`.
656#[derive(Debug, Clone, PartialEq, Eq)]
657struct PullRequest {
658    number: i64,
659    title: String,
660    url: String,
661    merged_at: String,
662    author_login: String,
663    /// `mergeCommit.oid`, absent for a pull request merged some other way (or
664    /// whose merge commit has since been rewritten).
665    merge_sha: Option<String>,
666}
667
668fn pr_key(number: i64) -> String {
669    format!("pr:{number}")
670}
671
672/// The pull request number in a squash-merge subject: `\(#(\d+)\)$`.
673///
674/// Hand-rolled rather than pulled in as a dependency — the pattern is anchored
675/// at the end of the subject and made of two literals around a run of digits.
676fn subject_pr(subject: &str) -> Option<i64> {
677    let rest = subject.strip_suffix(')')?;
678    let at = rest.rfind("(#")?;
679    let digits = &rest[at + 2..];
680    if digits.is_empty() || !digits.bytes().all(|b| b.is_ascii_digit()) {
681        return None;
682    }
683    digits.parse().ok()
684}
685
686/// `gh pr list --state merged` in `repo`, or an empty list with one warning.
687///
688/// Every failure is a skip: `gh` may not be installed, the repository may have
689/// no GitHub remote, and the user may not be authenticated. None of that is a
690/// reason to fail an ingest that is otherwise complete.
691fn fetch_prs(repo: &Path) -> Vec<PullRequest> {
692    let out = Command::new("gh")
693        .current_dir(repo)
694        .args([
695            "pr",
696            "list",
697            "--state",
698            "merged",
699            "--limit",
700            "1000",
701            "--json",
702            "number,title,url,mergedAt,mergeCommit,author",
703        ])
704        .output();
705    let out = match out {
706        Ok(o) => o,
707        Err(_) => {
708            eprintln!("ingest-git: --prs skipped: gh is not on PATH");
709            return Vec::new();
710        }
711    };
712    if !out.status.success() {
713        let detail = String::from_utf8_lossy(&out.stderr)
714            .lines()
715            .find(|l| !l.trim().is_empty())
716            .unwrap_or("no detail")
717            .to_string();
718        eprintln!("ingest-git: --prs skipped: gh pr list failed: {detail}");
719        return Vec::new();
720    }
721    let parsed: serde_json::Value = match serde_json::from_slice(&out.stdout) {
722        Ok(v) => v,
723        Err(e) => {
724            eprintln!("ingest-git: --prs skipped: gh pr list output is not JSON: {e}");
725            return Vec::new();
726        }
727    };
728    let Some(items) = parsed.as_array() else {
729        eprintln!("ingest-git: --prs skipped: gh pr list did not return a list");
730        return Vec::new();
731    };
732    let mut prs: Vec<PullRequest> = items
733        .iter()
734        .filter_map(|v| {
735            let number = v.get("number")?.as_i64()?;
736            let str_at = |k: &str| {
737                v.get(k)
738                    .and_then(|x| x.as_str())
739                    .unwrap_or_default()
740                    .to_string()
741            };
742            Some(PullRequest {
743                number,
744                title: str_at("title"),
745                url: str_at("url"),
746                merged_at: str_at("mergedAt"),
747                author_login: v
748                    .get("author")
749                    .and_then(|a| a.get("login"))
750                    .and_then(|l| l.as_str())
751                    .unwrap_or_default()
752                    .to_string(),
753                merge_sha: v
754                    .get("mergeCommit")
755                    .and_then(|m| m.get("oid"))
756                    .and_then(|o| o.as_str())
757                    .filter(|s| !s.is_empty())
758                    .map(str::to_string),
759            })
760        })
761        .collect();
762    // Ascending by number: the insert order, and so the node order, is the
763    // same on every run whatever order gh listed them in.
764    prs.sort_by_key(|p| p.number);
765    prs.dedup_by_key(|p| p.number);
766    prs
767}
768
769/// Insert the `PR` nodes that are new, declare the `Commit.pr_id` foreign key,
770/// and index titles for search. Runs before the commits so the FK resolves.
771fn ingest_prs(
772    w: &mut WriteGuard<'_>,
773    prs: &[PullRequest],
774    ingest: &IngestOptions,
775    report: &mut IngestGitReport,
776) -> Result<(), CliError> {
777    let rows: Vec<BTreeMap<String, Value>> = prs
778        .iter()
779        .filter(|p| !w.has_node(&pr_key(p.number)))
780        .map(|p| {
781            BTreeMap::from([
782                ("id".to_string(), Value::Str(pr_key(p.number))),
783                ("number".to_string(), Value::Int(p.number)),
784                ("title".to_string(), Value::Str(p.title.clone())),
785                ("url".to_string(), Value::Str(p.url.clone())),
786                ("merged_at".to_string(), Value::Str(p.merged_at.clone())),
787                (
788                    "author_login".to_string(),
789                    Value::Str(p.author_login.clone()),
790                ),
791            ])
792        })
793        .collect();
794    if !rows.is_empty() {
795        let r = w.ingest_with_edges("PR", rows, ingest, &[])?;
796        report.prs = r.inserted;
797        report.rules_created.extend(r.rules_created);
798    }
799    // Declared here rather than left to FK inference, which only fires on a
800    // batch of `Commit` rows that already carry `pr_id` — a run that links a
801    // pull request to a commit ingested earlier would otherwise leave the
802    // property with no edge behind it.
803    if !w.rules().iter().any(|r| r.name == PR_FK_RULE) {
804        let predicate = Predicate::KeyMatch {
805            field: "pr_id".into(),
806        };
807        let max_edges = Some(default_max_edges(&predicate));
808        w.create_rule(RuleDef {
809            name: PR_FK_RULE.into(),
810            src_label: "Commit".into(),
811            dst_label: "PR".into(),
812            predicate,
813            edge_type: "PR".into(),
814            weight_prop: None,
815            max_edges,
816            approximate: false,
817            via_label: None,
818            via_edge: None,
819            via_dir: None,
820            namespace: None,
821        })?;
822        report.rules_created.push(PR_FK_RULE.to_string());
823    }
824    if !w
825        .fulltext_pairs()
826        .contains(&("PR".to_string(), "title".to_string()))
827    {
828        w.enable_fulltext("PR", "title")?;
829    }
830    Ok(())
831}
832
833/// Point every commit that carries a pull request at it: the merge commit by
834/// sha, a squash merge by its `(#N)` subject. Runs after the commits are in the
835/// graph, so a pull request merged before the last sync is linked too.
836fn link_prs(
837    w: &mut WriteGuard<'_>,
838    prs: &[PullRequest],
839    ingest: &IngestOptions,
840) -> Result<(), CliError> {
841    let by_sha: BTreeMap<&str, i64> = prs
842        .iter()
843        .filter_map(|p| p.merge_sha.as_deref().map(|s| (s, p.number)))
844        .collect();
845    let known: BTreeSet<i64> = prs.iter().map(|p| p.number).collect();
846
847    let rs = w.query(
848        "MATCH (c:Commit) RETURN c.id AS id, c.message AS message, c.pr_id AS pr_id",
849        &BTreeMap::new(),
850    )?;
851    let mut updates: Vec<(String, String)> = Vec::new();
852    let mut links: BTreeMap<i64, BTreeSet<String>> = BTreeMap::new();
853    for i in 0..rs.len() {
854        let Some(Value::Str(sha)) = rs.get(i, "id") else {
855            continue;
856        };
857        let subject = match rs.get(i, "message") {
858            Some(Value::Str(s)) => s.as_str(),
859            _ => "",
860        };
861        let Some(number) = by_sha
862            .get(sha.as_str())
863            .copied()
864            .or_else(|| subject_pr(subject).filter(|n| known.contains(n)))
865        else {
866            continue;
867        };
868        let key = pr_key(number);
869        links.entry(number).or_default().insert(sha.clone());
870        if rs.get(i, "pr_id") != Some(&Value::Str(key.clone())) {
871            updates.push((sha.clone(), key));
872        }
873    }
874    updates.sort(); // by sha: the same store state writes the same records
875    for (sha, key) in updates {
876        w.set_prop(&sha, "pr_id", Value::Str(key))?;
877    }
878
879    let mut edges: Vec<(String, String, String)> = Vec::new();
880    for (number, shas) in links {
881        let src = pr_key(number);
882        let existing: BTreeSet<String> = w
883            .neighbors(&src, "MERGED_AS", Direction::Out)
884            .unwrap_or_default()
885            .into_iter()
886            .collect();
887        for sha in shas {
888            if !existing.contains(&sha) {
889                edges.push(("MERGED_AS".to_string(), src.clone(), sha));
890            }
891        }
892    }
893    if !edges.is_empty() {
894        w.ingest_with_edges("PR", Vec::new(), ingest, &edges)?;
895    }
896    Ok(())
897}
898
899/// Cypher behind [`file_state_from`] — the cumulative state of every live
900/// `File` node belonging to one unit.
901///
902/// The prefix filter is what keeps a submodule's files out of the parent's
903/// walk and vice versa. `startsWith` is the documented Cypher form (see
904/// `docs/site/query.md`); the parent unit's empty prefix matches everything, so
905/// its own submodules' keys are dropped in [`file_state_from`].
906const FILE_STATE_QUERY: &str =
907    "MATCH (f:File) WHERE startsWith(f.id, $prefix) RETURN f.id AS id, f.commits AS commits, \
908     f.n_commits AS n, f.top_author_id AS top, f.author_counts AS author_counts";
909
910/// Rebuild the in-memory per-file state from the `File` nodes already in the
911/// graph so incremental runs keep `commits` and ownership counts cumulative.
912///
913/// `nested` holds the key prefixes of any submodules inside this unit; their
914/// files belong to their own unit's walk and are dropped here.
915fn file_state_from(rs: &ResultSet, nested: &[String]) -> BTreeMap<String, FileState> {
916    let mut files = BTreeMap::new();
917    for i in 0..rs.len() {
918        let id = match rs.get(i, "id") {
919            Some(Value::Str(s)) => s.clone(),
920            _ => continue,
921        };
922        if nested.iter().any(|p| id.starts_with(p.as_str())) {
923            continue;
924        }
925        let mut st = FileState::default();
926        if let Some(Value::List(l)) = rs.get(i, "commits") {
927            st.commits = l
928                .iter()
929                .filter_map(|v| match v {
930                    Value::Str(s) => Some(s.clone()),
931                    _ => None,
932                })
933                .collect();
934        }
935        if let Some(Value::Int(n)) = rs.get(i, "n") {
936            st.n_commits = *n as usize;
937        }
938        match rs.get(i, "author_counts") {
939            Some(Value::List(l)) => st.set_author_counts(l),
940            // A node written before `author_counts` existed (a store built by
941            // 0.4.x). Fall back to the old approximation — the whole prior
942            // history credited to the incumbent — so those stores keep
943            // working. The prop is written on the next touch, from which point
944            // ownership tracks reality; a full re-ingest repairs it at once.
945            _ => {
946                if let Some(Value::Str(t)) = rs.get(i, "top") {
947                    st.by_author.insert(t.clone(), st.n_commits.max(1));
948                }
949            }
950        }
951        files.insert(id, st);
952    }
953    files
954}
955
956/// The real per-name distribution for every author of this unit, read from the
957/// whole history rather than from the window.
958///
959/// A store built by 0.6.0 has `Author` nodes carrying a `name` and no
960/// `name_counts`, and nothing in the graph records which spelling each commit
961/// used — the display name it holds is the *first* one that version happened to
962/// see, which on a repository where one person commits under two names is the
963/// wrong one about as often as not. Guessing from it cannot recover: seeding the
964/// incumbent with the prior commit count makes the true majority wait for more
965/// new commits than the entire history it is already behind by.
966///
967/// So the log is walked in full, once, on the first run that meets such a node.
968/// The result covers every commit up to `head`, this run's window included, so
969/// the caller uses it *instead of* the window rather than on top of it.
970/// A store this version wrote never reaches here.
971fn legacy_name_counts(
972    w: &WriteGuard<'_>,
973    p: &Pending,
974) -> Result<BTreeMap<String, AuthorState>, CliError> {
975    let mut out: BTreeMap<String, AuthorState> = BTreeMap::new();
976    let stale = p.log.iter().any(|c| {
977        w.has_node(&c.author_email)
978            && w.node_ref(&c.author_email)
979                .and_then(|n| n.prop("name_counts"))
980                .is_none()
981    });
982    let Some(head) = p.head.as_deref().filter(|_| stale) else {
983        return Ok(out);
984    };
985    for c in read_log(&p.unit.path, None, head)? {
986        out.entry(c.author_email).or_default().touch(&c.author_name);
987    }
988    Ok(out)
989}
990
991/// Accumulated effect of one log window, before anything is written.
992#[derive(Default)]
993struct Walk {
994    files: BTreeMap<String, FileState>,
995    /// Email → the names its commits were authored under, and how many each.
996    authors: BTreeMap<String, AuthorState>,
997    commit_rows: Vec<BTreeMap<String, Value>>,
998    touched_edges: Vec<(String, String, String)>,
999    /// Paths whose `File` props changed in this window.
1000    dirty: BTreeSet<String>,
1001    deleted: BTreeSet<String>,
1002    /// Node renames to apply, collapsed across chained renames.
1003    renamed: Vec<(String, String)>,
1004    /// Any old path → its final path in this window, for edge retargeting.
1005    alias: BTreeMap<String, String>,
1006}
1007
1008impl Walk {
1009    fn rename(&mut self, from: &str, to: &str, node_exists: bool) {
1010        if let Some(e) = self.renamed.iter_mut().find(|(_, t)| t == from) {
1011            e.1 = to.to_string();
1012        } else if node_exists {
1013            // A node cannot be both deleted and moved; the move wins.
1014            self.deleted.remove(from);
1015            self.renamed.push((from.to_string(), to.to_string()));
1016        } else {
1017            self.deleted.insert(from.to_string());
1018        }
1019        for v in self.alias.values_mut() {
1020            if v == from {
1021                *v = to.to_string();
1022            }
1023        }
1024        self.alias.insert(from.to_string(), to.to_string());
1025    }
1026}
1027
1028/// The `GitSync` props that say *how* a unit was ingested, as opposed to how
1029/// far. Compared against the stored node so that changing a flag refreshes the
1030/// marker even when there is nothing new to walk.
1031fn marker_flag_props(unit: &RepoUnit, opts: &IngestGitOpts) -> Vec<(String, Value)> {
1032    vec![
1033        (
1034            "repo".into(),
1035            Value::Str(unit.path.to_string_lossy().into_owned()),
1036        ),
1037        ("recurse".into(), Value::Bool(opts.recurse_submodules)),
1038        ("prs".into(), Value::Bool(opts.prs)),
1039        ("structure".into(), Value::Bool(opts.structure)),
1040        ("docs".into(), Value::Bool(opts.docs)),
1041    ]
1042}
1043
1044/// One unit and everything read about it before the write pass opens.
1045struct Pending {
1046    unit: RepoUnit,
1047    /// The sha its marker resumes from, absent on a first run.
1048    since: Option<String>,
1049    /// The stored marker disagrees with this run's flags.
1050    stale: bool,
1051    /// `None` when the unit has no commits at all.
1052    head: Option<String>,
1053    log: Vec<GitCommit>,
1054    /// Key prefixes of the submodules nested inside this unit, whose files
1055    /// belong to their own walk.
1056    nested: Vec<String>,
1057}
1058
1059/// Whether `db_dir` would open faster if this run left a snapshot behind.
1060///
1061/// A full ingest always qualifies: it is the run that writes the whole history
1062/// into an empty WAL, and leaving that WAL to be replayed on every subsequent
1063/// open is the single largest fixed cost in the hook path. An incremental run
1064/// qualifies when the store has no snapshot at all — a store built by an older
1065/// release, or one whose full ingest predates this rule — or when the tail
1066/// past the last snapshot has grown beyond [`SNAPSHOT_WAL_BYTES`].
1067///
1068/// The tail *is* `wal.bin`: a snapshot replaces the live WAL with a minimal
1069/// baseline, so its length on disk measures exactly what an open has to replay.
1070///
1071/// Only a run that reached its write phase asks. A run with nothing to take
1072/// returns before entering a write scope, so a store with no snapshot and no
1073/// new commits keeps waiting for a run that has work — which in the hook path
1074/// is the next commit.
1075fn snapshot_due(db_dir: &Path, full: bool) -> bool {
1076    if full || !db_dir.join("snapshot.bin").exists() {
1077        return true;
1078    }
1079    std::fs::metadata(db_dir.join("wal.bin")).is_ok_and(|m| m.len() > SNAPSHOT_WAL_BYTES)
1080}
1081
1082/// Write a snapshot if one is due, after the run's own writes are committed.
1083///
1084/// The WAL is archived rather than dropped — see [`AUTOMATIC_SNAPSHOT`], which
1085/// every snapshot mushroomdb takes on its own shares — and every archive is
1086/// kept: [`AUTO_SNAPSHOT_RETENTION`] is `None` by default, so a store that
1087/// syncs on every commit grows an archive per 4 MiB of churn rather than
1088/// losing any of its history. `mushroomdb snapshot <db> --retention N` is how
1089/// a caller bounds that growth once they have decided the trade is worth it.
1090///
1091/// [`AUTOMATIC_SNAPSHOT`]: crate::AUTOMATIC_SNAPSHOT
1092/// [`AUTO_SNAPSHOT_RETENTION`]: crate::AUTO_SNAPSHOT_RETENTION
1093///
1094/// # Why it never fails the run
1095///
1096/// A snapshot is an optimisation for the *next* open, and the data is already
1097/// durable either way:
1098///
1099/// * `Busy` — another process holds the cross-process write lock. Skipped
1100///   without a word; the next run that qualifies takes it.
1101/// * anything else — reported on stderr, because a store that cannot be
1102///   snapshotted is worth knowing about, and the run still succeeds.
1103///
1104/// Call this only from a run that reached its write phase. A run with nothing
1105/// to do returns before entering a write scope, and taking the lock to
1106/// snapshot would turn a genuine no-op into a write.
1107fn snapshot_if_due(db: &SharedDb, db_dir: &Path, full: bool) {
1108    if !snapshot_due(db_dir, full) {
1109        return;
1110    }
1111    let taken = db
1112        .write_with_wait(WRITE_LOCK_WAIT)
1113        .and_then(|mut g| crate::snapshot_automatically(&mut g));
1114    match taken {
1115        Ok(()) | Err(GraphError::Busy { .. }) => {}
1116        Err(e) => eprintln!("snapshot skipped: {e}"),
1117    }
1118}
1119
1120pub fn run_ingest_git(db_dir: &Path, opts: &IngestGitOpts) -> Result<IngestGitReport, CliError> {
1121    // Absolute and symlink-free: it is recorded on the marker, and a later run
1122    // has no reason to share this one's working directory.
1123    let repo = canonical(&opts.repo);
1124    let units = repo_units(&repo, opts.recurse_submodules)?;
1125    let prefixes: Vec<String> = units
1126        .iter()
1127        .map(|u| u.prefix.clone())
1128        .filter(|p| !p.is_empty())
1129        .collect();
1130
1131    let db = SharedDb::open(db_dir)?;
1132    let mut report = IngestGitReport {
1133        submodules: units.len() - 1,
1134        ..Default::default()
1135    };
1136    if opts.ensure_gitignore {
1137        report.gitignore_added = ensure_gitignore(&repo, db_dir)?;
1138    }
1139
1140    let mut pending: Vec<Pending> = Vec::new();
1141    {
1142        let r = db.read();
1143        for unit in units {
1144            let nested = prefixes
1145                .iter()
1146                .filter(|p| **p != unit.prefix && p.starts_with(&unit.prefix))
1147                .cloned()
1148                .collect();
1149            let (since, stale) = match r.node_ref(&unit.sync_key) {
1150                Some(n) => (
1151                    match n.prop("sha") {
1152                        Some(Value::Str(s)) if !s.is_empty() => Some(s),
1153                        _ => None,
1154                    },
1155                    marker_flag_props(&unit, opts)
1156                        .into_iter()
1157                        .any(|(k, v)| n.prop(&k) != Some(v)),
1158                ),
1159                None => (None, false),
1160            };
1161            pending.push(Pending {
1162                unit,
1163                since,
1164                stale,
1165                head: None,
1166                log: Vec::new(),
1167                nested,
1168            });
1169        }
1170    }
1171    // The repository itself is always the first unit, and it is what "this
1172    // database has seen this repo before" means.
1173    report.incremental = pending[0].since.is_some();
1174
1175    for p in &mut pending {
1176        // Pin the end of the walk and the marker to one sha, resolved first. A
1177        // commit landing mid-run then falls outside this range instead of being
1178        // skipped by a marker that advanced past it.
1179        let Some(head) = head_sha(&p.unit.path)? else {
1180            continue; // this unit has no commits yet
1181        };
1182        let mut log = read_log(&p.unit.path, p.since.as_deref(), &head)?;
1183        localise(&mut log, &p.unit.prefix, &gitlink_paths(&p.unit.path));
1184        p.head = Some(head);
1185        p.log = log;
1186    }
1187
1188    let prs = if opts.prs {
1189        fetch_prs(&repo)
1190    } else {
1191        Vec::new()
1192    };
1193    if prs.is_empty() && pending.iter().all(|p| p.log.is_empty() && !p.stale) {
1194        // Nothing new: leave the store untouched so `commit_seq` does not move.
1195        return Ok(report);
1196    }
1197
1198    // Held for the rest of the run. Another process holding the store's
1199    // cross-process write lock is a retry, not a failure: nothing was written.
1200    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1201        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1202        other => CliError(other.to_string()),
1203    })?;
1204    let ingest = IngestOptions::default(); // key `id`, auto-FK suffix `_id`
1205
1206    // Pull requests first: their nodes are what `Commit.pr_id` resolves to.
1207    if !prs.is_empty() {
1208        ingest_prs(&mut w, &prs, &ingest, &mut report)?;
1209    }
1210
1211    let mut authors: BTreeSet<String> = BTreeSet::new();
1212    let mut work = StructureWork::default();
1213    for p in &pending {
1214        if !p.log.is_empty() {
1215            ingest_unit(
1216                &mut w,
1217                p,
1218                opts,
1219                &ingest,
1220                &mut report,
1221                &mut authors,
1222                &mut work,
1223            )?;
1224        }
1225    }
1226    report.authors = authors.len();
1227
1228    // Rules and fulltext, created after the data so each backfills once. They
1229    // span every unit, so they are declared once for the database rather than
1230    // once per repository walked.
1231    if report.commits > 0 && !w.rules().iter().any(|r| r.name == "co_changed") {
1232        let co = Predicate::Overlap {
1233            field: "commits".into(),
1234            min: CO_CHANGE_MIN,
1235        };
1236        w.create_rule(RuleDef {
1237            name: "co_changed".into(),
1238            src_label: "File".into(),
1239            dst_label: "File".into(),
1240            predicate: co.clone(),
1241            edge_type: "CO_CHANGED".into(),
1242            weight_prop: Some("score".into()),
1243            max_edges: Some(10),
1244            approximate: false,
1245            via_label: None,
1246            via_edge: None,
1247            via_dir: None,
1248            namespace: None,
1249        })?;
1250        w.create_rule(RuleDef {
1251            name: "knows".into(),
1252            src_label: "Author".into(),
1253            dst_label: "File".into(),
1254            predicate: co,
1255            edge_type: "KNOWS".into(),
1256            weight_prop: Some("score".into()),
1257            max_edges: Some(20),
1258            approximate: false,
1259            via_label: Some("File".into()),
1260            via_edge: Some("TOP_AUTHOR".into()),
1261            via_dir: Some(Direction::In),
1262            namespace: None,
1263        })?;
1264        report
1265            .rules_created
1266            .extend(["co_changed".to_string(), "knows".to_string()]);
1267        for (l, field) in [("File", "path"), ("Commit", "message"), ("Author", "name")] {
1268            if !w
1269                .fulltext_pairs()
1270                .contains(&(l.to_string(), field.to_string()))
1271            {
1272                w.enable_fulltext(l, field)?;
1273            }
1274        }
1275    }
1276
1277    // The working tree, on top of the history. It runs after the commit walk
1278    // so every `File` node it reads exists, and its rules are declared after
1279    // its props so each one backfills exactly once.
1280    if opts.structure {
1281        // A first run has nothing to be incremental against, and a run whose
1282        // flags changed (structure or docs just turned on) has to revisit
1283        // files its predecessor deliberately skipped.
1284        let full = !report.incremental || pending.iter().any(|p| p.stale);
1285        report.structure = if full {
1286            structure::refresh_all(&mut w, &repo, "", opts.docs)?
1287        } else {
1288            let mut paths = work.touched;
1289            paths.extend(structure::importers_of(&w, &work.stale)?);
1290            // The retired keys themselves, not just the files that named them.
1291            // An incremental refresh sweeps orphaned symbols only under the
1292            // paths it is handed, and a file this window deleted or renamed
1293            // away is exactly where the orphans are.
1294            paths.extend(work.stale.iter().cloned());
1295            let paths: Vec<String> = paths.into_iter().collect();
1296            structure::refresh_files(&mut w, &repo, "", &paths, opts.docs)?
1297        };
1298        report
1299            .rules_created
1300            .extend(structure::ensure_rules_and_fulltext(&mut w)?);
1301    }
1302
1303    // Every commit this run could link is in the graph by now.
1304    if !prs.is_empty() {
1305        link_prs(&mut w, &prs, &ingest)?;
1306    }
1307
1308    // The markers go last, once every phase of the run has succeeded. They say
1309    // how far a *complete* run got, so a failure anywhere above — a working-tree
1310    // batch that will not commit, a `gh` link pass that errors — leaves them
1311    // where they were and the next run re-walks the same window rather than
1312    // stepping over it. Re-walking a window that was partly applied is safe:
1313    // commits already in the graph are skipped as duplicate keys, file props
1314    // are rewritten from the recomputed state, and a rename whose node already
1315    // moved finds nothing to move.
1316    for p in &pending {
1317        write_marker(&mut w, p, opts)?;
1318    }
1319
1320    // Every write this run makes is now committed, so leave the store in the
1321    // shape that opens fastest. The guard is released first: `snapshot()` takes
1322    // the same cross-process write lock this one holds.
1323    drop(w);
1324    snapshot_if_due(&db, db_dir, !report.incremental);
1325    Ok(report)
1326}
1327
1328/// Wall-clock seconds since the Unix epoch, for [`SYNCED_AT`].
1329///
1330/// This is the one place the ingest reads a clock. A clock before the epoch
1331/// reads as `0` rather than going negative.
1332fn now_unix() -> i64 {
1333    std::time::SystemTime::now()
1334        .duration_since(std::time::UNIX_EPOCH)
1335        .map_or(0, |d| i64::try_from(d.as_secs()).unwrap_or(i64::MAX))
1336}
1337
1338/// Marker prop holding when this store last took data from the repository.
1339///
1340/// `Commit.ts` says when the work was *written*, which on a store synced to
1341/// its repository's head is indistinguishable from now — so it cannot answer
1342/// "how stale is my graph". This can. Absent on a store built before it
1343/// existed, which readers must tolerate.
1344pub const SYNCED_AT: &str = "synced_at";
1345
1346/// Record how far this unit got, and under what flags.
1347///
1348/// Props are written only where they differ, so a run that touches one unit
1349/// does not churn the markers of the others — and a run that changes nothing
1350/// writes nothing at all, [`SYNCED_AT`] included.
1351///
1352/// The stamp therefore means "when this run last changed this marker": a new
1353/// head, or a flag ingested differently from last time. A no-op re-run leaving
1354/// it alone is the point rather than a gap — a graph that had nothing to take
1355/// is not staler for having checked.
1356fn write_marker(w: &mut WriteGuard<'_>, p: &Pending, opts: &IngestGitOpts) -> Result<(), CliError> {
1357    let Some(head) = p.head.as_deref() else {
1358        return Ok(()); // no commits, so nothing to resume from
1359    };
1360    let mut props = marker_flag_props(&p.unit, opts);
1361    if !p.log.is_empty() {
1362        props.push(("sha".into(), Value::Str(head.to_string())));
1363    }
1364    let key = p.unit.sync_key.clone();
1365    if w.has_node(&key) {
1366        let mut changed = false;
1367        for (k, v) in props {
1368            let current = w.node_ref(&key).and_then(|n| n.prop(&k));
1369            if current.as_ref() != Some(&v) {
1370                w.set_prop(&key, &k, v)?;
1371                changed = true;
1372            }
1373        }
1374        if changed {
1375            w.set_prop(&key, SYNCED_AT, Value::Int(now_unix()))?;
1376        }
1377        return Ok(());
1378    }
1379    if p.log.is_empty() {
1380        return Ok(());
1381    }
1382    // It carries `id` like every other label here, so the key is readable from
1383    // Cypher.
1384    props.push(("id".into(), Value::Str(key.clone())));
1385    props.push((SYNCED_AT.into(), Value::Int(now_unix())));
1386    props.sort_by(|a, b| a.0.cmp(&b.0));
1387    w.insert_node("GitSync", &key, props)?;
1388    Ok(())
1389}
1390
1391/// Walk one unit's new commits into the graph.
1392///
1393/// Every path in `p.log` is already a repository-wide key, so this is the
1394/// single-repository algorithm unchanged: only the `File` state it starts from
1395/// and the counts it adds to are scoped to the unit.
1396fn ingest_unit(
1397    w: &mut WriteGuard<'_>,
1398    p: &Pending,
1399    opts: &IngestGitOpts,
1400    ingest: &IngestOptions,
1401    report: &mut IngestGitReport,
1402    authors: &mut BTreeSet<String>,
1403    work: &mut StructureWork,
1404) -> Result<(), CliError> {
1405    let log = &p.log;
1406    let incremental = p.since.is_some();
1407
1408    let mut walk = Walk {
1409        files: if incremental {
1410            let params =
1411                BTreeMap::from([("prefix".to_string(), Value::Str(p.unit.prefix.clone()))]);
1412            file_state_from(&w.query(FILE_STATE_QUERY, &params)?, &p.nested)
1413        } else {
1414            BTreeMap::new()
1415        },
1416        ..Default::default()
1417    };
1418
1419    for c in log {
1420        walk.authors
1421            .entry(c.author_email.clone())
1422            .or_default()
1423            .touch(&c.author_name);
1424        walk.commit_rows.push(BTreeMap::from([
1425            ("id".to_string(), Value::Str(c.sha.clone())),
1426            ("message".to_string(), Value::Str(c.subject.clone())),
1427            ("ts".to_string(), Value::Int(c.ts)),
1428            ("author_id".to_string(), Value::Str(c.author_email.clone())),
1429        ]));
1430        for ch in &c.changes {
1431            match ch {
1432                Change::Added(p) | Change::Modified(p) => {
1433                    if excluded(p, &opts.exclude) {
1434                        continue;
1435                    }
1436                    walk.deleted.remove(p);
1437                    walk.files.entry(p.clone()).or_default().touch(
1438                        &c.sha,
1439                        &c.author_email,
1440                        opts.max_commits_per_file,
1441                    );
1442                    walk.dirty.insert(p.clone());
1443                    walk.touched_edges
1444                        .push(("TOUCHED".into(), c.sha.clone(), p.clone()));
1445                }
1446                Change::Deleted(p) => {
1447                    if excluded(p, &opts.exclude) {
1448                        continue;
1449                    }
1450                    walk.files.remove(p);
1451                    walk.dirty.remove(p);
1452                    walk.deleted.insert(p.clone());
1453                }
1454                Change::Renamed { from, to } => {
1455                    if excluded(to, &opts.exclude) {
1456                        // Moved out of scope: drop the old node, keep no alias
1457                        // so its TOUCHED edges are filtered out below.
1458                        walk.files.remove(from);
1459                        walk.dirty.remove(from);
1460                        walk.deleted.insert(from.clone());
1461                        continue;
1462                    }
1463                    let mut st = walk.files.remove(from).unwrap_or_default();
1464                    st.touch(&c.sha, &c.author_email, opts.max_commits_per_file);
1465                    walk.files.insert(to.clone(), st);
1466                    walk.dirty.remove(from);
1467                    walk.dirty.insert(to.clone());
1468                    walk.deleted.remove(to);
1469                    walk.touched_edges
1470                        .push(("TOUCHED".into(), c.sha.clone(), to.clone()));
1471                    let exists = incremental && w.has_node(from);
1472                    walk.rename(from, to, exists);
1473                }
1474            }
1475        }
1476    }
1477
1478    // What the working-tree pass has to look at afterwards: the paths this
1479    // window changed, and the keys it left pointing at nothing. A key that
1480    // ended the window live again — a file moved away and back — is neither.
1481    work.touched.extend(walk.dirty.iter().cloned());
1482    for key in walk.deleted.iter().chain(walk.alias.keys()) {
1483        if !walk.files.contains_key(key) {
1484            work.stale.insert(key.clone());
1485        }
1486    }
1487
1488    // 1. Authors first: the auto-FK rules for `Commit.author_id` and
1489    //    `File.top_author_id` only infer once their targets resolve to Author.
1490    //    An email already in the graph keeps its accumulated `name_counts`, so
1491    //    the display name is the majority spelling over the whole history
1492    //    rather than over this window.
1493    let legacy = legacy_name_counts(w, p)?;
1494    let mut author_rows = Vec::new();
1495    let mut author_updates = Vec::new();
1496    for (email, seen) in &walk.authors {
1497        let mut merged = AuthorState::default();
1498        match (
1499            w.node_ref(email).and_then(|n| n.prop("name_counts")),
1500            legacy.get(email),
1501        ) {
1502            // The usual path: what the graph already counted, plus this window.
1503            (Some(Value::List(l)), _) => {
1504                merged.set_name_counts(&l);
1505                merged.absorb(seen);
1506            }
1507            // A node written before `name_counts` existed. The recovered walk
1508            // is the whole history and already contains this window, so the
1509            // window is not added to it.
1510            (_, Some(whole_history)) => merged = whole_history.clone(),
1511            // A new author, or a unit whose history was already recovered.
1512            _ => merged.absorb(seen),
1513        }
1514        let props = [
1515            ("name", Value::Str(merged.display_name())),
1516            ("name_counts", merged.name_counts_value()),
1517        ];
1518        if w.has_node(email) {
1519            for (field, want) in props {
1520                if w.node_ref(email).and_then(|n| n.prop(field)).as_ref() != Some(&want) {
1521                    author_updates.push((email.clone(), field, want));
1522                }
1523            }
1524        } else {
1525            let mut row = BTreeMap::from([("id".to_string(), Value::Str(email.clone()))]);
1526            row.extend(props.into_iter().map(|(f, v)| (f.to_string(), v)));
1527            author_rows.push(row);
1528        }
1529    }
1530    if !author_updates.is_empty() {
1531        let mut b = w.batch();
1532        for (key, field, value) in author_updates {
1533            b.set_prop(&key, field, value);
1534        }
1535        b.commit()?;
1536    }
1537    let a = w.ingest_with_edges("Author", author_rows, ingest, &[])?;
1538    report.rules_created.extend(a.rules_created);
1539    authors.extend(walk.authors.keys().cloned());
1540
1541    // 2. Deletes run first so a rename can claim a path freed in this same
1542    //    window, then renames carry each node (and its history) to its new path.
1543    //    `walk.files` is the authority on what is still live: a rename whose
1544    //    destination is not in it is a delete, not a move.
1545    for p in &walk.deleted {
1546        if w.has_node(p) {
1547            w.delete_node(p)?;
1548            report.deleted += 1;
1549        }
1550    }
1551    for (from, to) in &walk.renamed {
1552        if !w.has_node(from) || from == to {
1553            // Nothing to move, or a rename that swapped back to its own path.
1554            continue;
1555        }
1556        if !walk.files.contains_key(to) {
1557            // The destination did not survive the window — it was deleted, or
1558            // moved into an excluded path, after this rename. The node goes
1559            // with it; renaming into a dead path would strand a phantom node
1560            // that no later phase refreshes.
1561            w.delete_node(from)?;
1562            report.deleted += 1;
1563            continue;
1564        }
1565        if w.has_node(to) {
1566            // A pre-existing node already holds the destination path (deleted
1567            // earlier in this window, then claimed by this rename).
1568            w.delete_node(to)?;
1569            report.deleted += 1;
1570        }
1571        w.rename_node(from, to)?;
1572        // The key moved, so the `id` prop must move with it. Phase 3 also sets
1573        // it for every dirty path; doing it here keeps the invariant local to
1574        // the rename and independent of that filter.
1575        w.set_prop(to, "id", Value::Str(to.clone()))?;
1576        report.renamed += 1;
1577    }
1578
1579    // 3. File nodes. Existing nodes are updated in place (including `id`, which
1580    //    must follow the key after a rename); new paths go through ingest so
1581    //    the `top_author_id` auto-FK rule is inferred.
1582    //    Updates to existing nodes go in one batch rather than one WAL commit
1583    //    per property. Every frame in the WAL is replayed on every later open,
1584    //    and each one re-fires the rules watching the property it carries, so a
1585    //    run that appends a frame per property makes every subsequent open
1586    //    slower for as long as that frame lives. The op order inside the batch
1587    //    is the order the individual writes had, so the resulting state is the
1588    //    same one either form produces.
1589    let mut new_file_rows = Vec::new();
1590    let mut updates = Vec::new();
1591    let mut written = 0usize;
1592    for (path, st) in &walk.files {
1593        if incremental && !walk.dirty.contains(path) {
1594            continue;
1595        }
1596        written += 1;
1597        let props = file_props(st, path);
1598        if w.has_node(path) {
1599            updates.extend(props.into_iter().map(|(k, v)| (path.clone(), k, v)));
1600        } else {
1601            new_file_rows.push(props.into_iter().collect::<BTreeMap<_, _>>());
1602        }
1603    }
1604    if !updates.is_empty() {
1605        let mut b = w.batch();
1606        for (key, field, value) in updates {
1607            b.set_prop(&key, &field, value);
1608        }
1609        b.commit()?;
1610    }
1611    let f = w.ingest_with_edges("File", new_file_rows, ingest, &[])?;
1612    report.rules_created.extend(f.rules_created);
1613    // What this run wrote, not every file it has ever seen: an incremental run
1614    // loads the whole known file set to fold the new commits into it, and
1615    // reporting that set made a one-file run print the size of the repository.
1616    report.files += written;
1617
1618    // 4. Commits, then their TOUCHED edges. The two must be separate batches:
1619    //    a batch that both inserts nodes firing a new rule and carries a user
1620    //    edge of a not-yet-interned type writes a WAL frame that cannot be
1621    //    replayed (`Intern` records are emitted in a pre-pass, but on replay the
1622    //    rule fires — and interns its edge type — before the later `Intern`
1623    //    record is read). See the report for a reproducer.
1624    let c = w.ingest_with_edges("Commit", walk.commit_rows, ingest, &[])?;
1625    report.rules_created.extend(c.rules_created);
1626    report.commits += log.len();
1627
1628    // Edges name File keys, so files must already exist. A path renamed later
1629    // in this same window is retargeted to where its node ended up.
1630    let touched: Vec<(String, String, String)> = walk
1631        .touched_edges
1632        .into_iter()
1633        .map(|(t, sha, p)| {
1634            let p = walk.alias.get(&p).cloned().unwrap_or(p);
1635            (t, sha, p)
1636        })
1637        .filter(|(_, _, p)| walk.files.contains_key(p))
1638        .collect();
1639    if !touched.is_empty() {
1640        w.ingest_with_edges("Commit", Vec::new(), ingest, &touched)?;
1641    }
1642    Ok(())
1643}
1644
1645// ── sync and touch ──────────────────────────────────────────────────────────
1646//
1647// `ingest-git` is the command a person runs. These two are what a *hook* runs:
1648// `sync` after a commit lands, `touch` after a single file is edited. Both read
1649// the repository out of the `GitSync` marker rather than taking it as an
1650// argument, so a hook line carries only the database path and keeps working
1651// when the checkout moves.
1652
1653/// What a store with no `GitSync` node is told. Naming the fix matters: this is
1654/// the error a hook installed against the wrong database prints.
1655const NO_MARKER: &str = "store has no git sync marker; run ingest-git first";
1656
1657/// The `GitSync` props that say how this store was built, read back so a later
1658/// `sync` repeats the same run without being told any of it again.
1659///
1660/// `exclude` and `max_commits_per_file` are not on the marker, so a `sync`
1661/// applies [`DEFAULT_EXCLUDES`] and [`DEFAULT_MAX_COMMITS_PER_FILE`]. A store
1662/// first built with custom `--exclude` patterns should keep being maintained
1663/// with `ingest-git`, which takes them.
1664#[derive(Debug, Clone, PartialEq, Eq)]
1665struct SyncMarker {
1666    repo: PathBuf,
1667    recurse: bool,
1668    prs: bool,
1669    structure: bool,
1670    docs: bool,
1671}
1672
1673impl SyncMarker {
1674    fn opts(&self) -> IngestGitOpts {
1675        IngestGitOpts {
1676            repo: self.repo.clone(),
1677            exclude: DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect(),
1678            max_commits_per_file: DEFAULT_MAX_COMMITS_PER_FILE,
1679            recurse_submodules: self.recurse,
1680            prs: self.prs,
1681            structure: self.structure,
1682            docs: self.docs,
1683            ensure_gitignore: false,
1684        }
1685    }
1686}
1687
1688/// Read the marker off an already-open handle.
1689///
1690/// Deliberately not "open the store and read the marker": opening this store is
1691/// by far the most expensive thing either command does — it replays the whole
1692/// WAL — so both of them open once and read the marker through that same
1693/// handle. A separate read-only open just to learn the repository path would
1694/// double the cost of every hook invocation.
1695fn marker_of(r: &structure::Db) -> Result<SyncMarker, CliError> {
1696    let node = r
1697        .node_ref(SYNC_KEY)
1698        .ok_or_else(|| CliError(NO_MARKER.into()))?;
1699    let flag = |name: &str| matches!(node.prop(name), Some(Value::Bool(true)));
1700    match node.prop("repo") {
1701        Some(Value::Str(repo)) if !repo.is_empty() => Ok(SyncMarker {
1702            repo: PathBuf::from(repo),
1703            recurse: flag("recurse"),
1704            prs: flag("prs"),
1705            // A marker written before these two flags existed carries neither,
1706            // and a working-tree pass is what such a run did.
1707            structure: node.prop("structure") != Some(Value::Bool(false)),
1708            docs: node.prop("docs") != Some(Value::Bool(false)),
1709        }),
1710        _ => Err(CliError(NO_MARKER.into())),
1711    }
1712}
1713
1714/// Open a store that must already exist.
1715///
1716/// `SharedDb::open` runs `create_dir_all`, so without this guard a hook line
1717/// carrying a typo'd path would keep creating empty databases and reporting
1718/// that they hold no marker — the same trap [`run_recall`] guards against.
1719///
1720/// [`run_recall`]: crate::recall::run_recall
1721fn open_existing(db_dir: &Path) -> Result<SharedDb, CliError> {
1722    if !db_dir.exists() {
1723        return Err(CliError(format!(
1724            "no database directory at {}",
1725            db_dir.display()
1726        )));
1727    }
1728    Ok(SharedDb::open(db_dir)?)
1729}
1730
1731/// What one [`run_sync`] did.
1732#[derive(Debug, Default, Clone, PartialEq, Eq)]
1733pub struct SyncReport {
1734    /// The incremental history walk.
1735    pub git: IngestGitReport,
1736    /// The working-tree pass over the dirty paths, which the history walk does
1737    /// not see: an edit that has not been committed is in no commit.
1738    pub structure: crate::structure::StructureReport,
1739    /// Dirty paths handed to that pass. Higher than `structure.files_scanned`
1740    /// when some of them are new to the graph, or no longer on disk.
1741    pub dirty_refreshed: usize,
1742}
1743
1744/// Paths that differ from `HEAD` or are not tracked at all, repository-relative
1745/// and sorted.
1746///
1747/// `-z` rather than the default listing: git escapes and quotes a path holding
1748/// a tab, a newline or a non-ASCII byte, and a quoted path matches no key.
1749fn dirty_paths(repo: &Path, exclude: &[String]) -> Result<Vec<String>, CliError> {
1750    const LISTS: [&[&str]; 2] = [
1751        &["diff", "--name-only", "-z", "HEAD"],
1752        &["ls-files", "--others", "--exclude-standard", "-z"],
1753    ];
1754    let mut out = BTreeSet::new();
1755    for args in LISTS {
1756        let o = git_output(repo, args)?;
1757        if !o.status.success() {
1758            // `diff HEAD` fails in a repository with no commits yet. Nothing is
1759            // dirty relative to a head that does not exist.
1760            continue;
1761        }
1762        for path in String::from_utf8_lossy(&o.stdout).split('\0') {
1763            if path.is_empty() || excluded(path, exclude) {
1764                continue;
1765            }
1766            out.insert(path.to_string());
1767        }
1768    }
1769    Ok(out.into_iter().collect())
1770}
1771
1772/// Bring the store up to date with the repository it was built from: the
1773/// commits since the marker, then the working tree where it differs from
1774/// `HEAD`.
1775///
1776/// The second half is what a plain `ingest-git` cannot do. Its working-tree
1777/// pass only visits the paths the *commits* touched, so a file edited and not
1778/// yet committed keeps whatever the graph last recorded about it. A hook that
1779/// runs on every commit wants the uncommitted remainder refreshed too.
1780pub fn run_sync(db_dir: &Path) -> Result<SyncReport, CliError> {
1781    // One handle for the whole run. It is opened before `run_ingest_git`, which
1782    // opens its own and commits through it, but a `SharedDb` refreshes off the
1783    // WAL when a write scope is entered — so the dirty pass below sees every
1784    // commit the ingest just made without this handle being reopened.
1785    let db = open_existing(db_dir)?;
1786    let marker = marker_of(&db.read())?;
1787    let opts = marker.opts();
1788    let mut report = SyncReport {
1789        git: run_ingest_git(db_dir, &opts)?,
1790        ..Default::default()
1791    };
1792    if !opts.structure {
1793        return Ok(report);
1794    }
1795
1796    // Only the root repository's working tree. A submodule's dirty files are
1797    // its own checkout's business, and `--recurse-submodules` resumes each unit
1798    // from its own marker on the next commit there.
1799    let repo = canonical(&marker.repo);
1800    let paths = dirty_paths(&repo, &opts.exclude)?;
1801    report.dirty_refreshed = paths.len();
1802    if paths.is_empty() {
1803        // Nothing to refresh: never enter a write scope, so the run takes no
1804        // lock and `commit_seq` cannot move.
1805        return Ok(report);
1806    }
1807
1808    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1809        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1810        other => CliError(other.to_string()),
1811    })?;
1812    report.structure = structure::refresh_files(&mut w, &repo, "", &paths, opts.docs)?;
1813
1814    // The history half already snapshotted if this was a first ingest; this
1815    // covers the tail the dirty pass just appended. `full` is false because a
1816    // sync is by definition a run against a store that already exists.
1817    drop(w);
1818    snapshot_if_due(&db, db_dir, false);
1819    Ok(report)
1820}
1821
1822pub fn format_touch(r: &structure::StructureReport) -> String {
1823    format!(
1824        "touch: {} file(s), {} symbol(s), {} import(s), {} call(s), {} mention(s)\n",
1825        r.files_scanned, r.symbols, r.imports, r.calls, r.mentions
1826    )
1827}
1828
1829/// One [`SyncReport`] as a single JSON object, for `sync --json`.
1830///
1831/// `text` is exactly what [`format_sync`] prints, so a caller that has the
1832/// object never has to render the digest a second way and never has to parse
1833/// the counts back out of the prose. The MCP `sync` tool is that caller: it
1834/// runs this binary and hands both halves straight to the assistant.
1835#[must_use]
1836pub fn format_sync_json(r: &SyncReport) -> String {
1837    let g = &r.git;
1838    let s = &r.structure;
1839    let gs = &g.structure;
1840    let value = serde_json::json!({
1841        "text": format_sync(r),
1842        "commits": g.commits,
1843        "files": g.files,
1844        "authors": g.authors,
1845        "renamed": g.renamed,
1846        "deleted": g.deleted,
1847        "incremental": g.incremental,
1848        "submodules": g.submodules,
1849        "prs": g.prs,
1850        "rules_created": g.rules_created,
1851        "scanned": {
1852            "files": gs.files_scanned,
1853            "symbols": gs.symbols,
1854            "imports": gs.imports,
1855            "calls": gs.calls,
1856            "mentions": gs.mentions,
1857        },
1858        "dirty_refreshed": r.dirty_refreshed,
1859        "dirty": {
1860            "files": s.files_scanned,
1861            "symbols": s.symbols,
1862            "imports": s.imports,
1863            "calls": s.calls,
1864            "mentions": s.mentions,
1865        },
1866    });
1867    format!("{value}\n")
1868}
1869
1870pub fn format_sync(r: &SyncReport) -> String {
1871    let mut out = format_ingest_git(&r.git);
1872    let s = &r.structure;
1873    out.push_str(&format!(
1874        "  dirty {} path(s): scanned {}, {} symbol(s), {} import(s), {} call(s)\n",
1875        r.dirty_refreshed, s.files_scanned, s.symbols, s.imports, s.calls
1876    ));
1877    out
1878}
1879
1880/// Re-extract exactly the files named, and nothing else.
1881///
1882/// `files` comes from argv when a caller has the paths; otherwise they are read
1883/// out of a `PostToolUse` hook payload on stdin, the same way [`run_recall`]
1884/// reads a prompt. Anything that is not a working-tree file this store already
1885/// knows — a path outside the repository, an excluded one, one the graph has
1886/// never seen — is dropped without comment, because a hook fires on every edit
1887/// the assistant makes and most of them are none of this store's business.
1888///
1889/// [`run_recall`]: crate::recall::run_recall
1890pub fn run_touch(
1891    db_dir: &Path,
1892    files: &[PathBuf],
1893    hook_stdin: Option<&str>,
1894) -> Result<structure::StructureReport, CliError> {
1895    let named: Vec<PathBuf> = if files.is_empty() {
1896        hook_stdin.map(paths_from_payload).unwrap_or_default()
1897    } else {
1898        files.to_vec()
1899    };
1900    if named.is_empty() {
1901        return Ok(structure::StructureReport::default());
1902    }
1903
1904    let db = open_existing(db_dir)?;
1905    let marker = marker_of(&db.read())?;
1906    if !marker.structure {
1907        // The store was built with `--no-structure`, so it holds no working-tree
1908        // props at all and re-extracting one file would be the only exception.
1909        return Ok(structure::StructureReport::default());
1910    }
1911    let repo = canonical(&marker.repo);
1912    let exclude: Vec<String> = DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect();
1913    let cwd = std::env::current_dir().unwrap_or_else(|_| PathBuf::from("."));
1914    let mut paths: BTreeSet<String> = BTreeSet::new();
1915    for path in &named {
1916        let Some(rel) = repo_relative(&repo, &cwd, path) else {
1917            continue;
1918        };
1919        if !excluded(&rel, &exclude) {
1920            paths.insert(rel);
1921        }
1922    }
1923    if paths.is_empty() {
1924        // Nothing of ours changed: never enter a write scope, so no lock is
1925        // taken and a running writer is never made to wait.
1926        return Ok(structure::StructureReport::default());
1927    }
1928
1929    let paths: Vec<String> = paths.into_iter().collect();
1930    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1931        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1932        other => CliError(other.to_string()),
1933    })?;
1934    structure::refresh_files(&mut w, &repo, "", &paths, marker.docs)
1935}
1936
1937/// The file paths in a `PostToolUse` payload.
1938///
1939/// `tool_input.file_path` covers Edit and Write; `tool_input.edits[].file_path`
1940/// covers the multi-edit shape. Both are read, so a payload carrying either (or
1941/// both) is handled without knowing which tool produced it. A payload that is
1942/// not JSON, or that names no file, yields nothing — never an error.
1943fn paths_from_payload(raw: &str) -> Vec<PathBuf> {
1944    let Ok(v) = serde_json::from_str::<serde_json::Value>(raw) else {
1945        return Vec::new();
1946    };
1947    let input = &v["tool_input"];
1948    let mut out = Vec::new();
1949    let mut push = |value: &serde_json::Value| {
1950        if let Some(s) = value.as_str().map(str::trim).filter(|s| !s.is_empty()) {
1951            out.push(PathBuf::from(s));
1952        }
1953    };
1954    push(&input["file_path"]);
1955    if let Some(edits) = input["edits"].as_array() {
1956        for e in edits {
1957            push(&e["file_path"]);
1958        }
1959    }
1960    out
1961}
1962
1963/// `path` as a key under `repo`, or `None` when it is not inside it.
1964///
1965/// A hook payload carries absolute paths and a person typing the command uses
1966/// relative ones, so a relative path is taken against `cwd`. Both sides are
1967/// then resolved through symlinks before comparing: the marker records the
1968/// canonical repository path, and on macOS a checkout under `/tmp` is reached
1969/// through a symlink that never compares equal as written.
1970fn repo_relative(repo: &Path, cwd: &Path, path: &Path) -> Option<String> {
1971    let absolute = if path.is_absolute() {
1972        path.to_path_buf()
1973    } else {
1974        cwd.join(path)
1975    };
1976    let resolved = resolve_symlinks(&absolute);
1977    let rel = resolved.strip_prefix(repo).ok()?;
1978    let key: Vec<String> = rel
1979        .components()
1980        .map(|c| c.as_os_str().to_string_lossy().into_owned())
1981        .collect();
1982    let key = key.join("/");
1983    (!key.is_empty()).then_some(key)
1984}
1985
1986/// [`canonical`] that still works for a path that no longer exists: a file
1987/// deleted between the edit and the hook resolves through its parent directory.
1988fn resolve_symlinks(p: &Path) -> PathBuf {
1989    if let Ok(resolved) = std::fs::canonicalize(p) {
1990        return resolved;
1991    }
1992    match (p.parent(), p.file_name()) {
1993        (Some(dir), Some(name)) => std::fs::canonicalize(dir)
1994            .map(|d| d.join(name))
1995            .unwrap_or_else(|_| p.to_path_buf()),
1996        _ => p.to_path_buf(),
1997    }
1998}
1999
2000pub fn format_ingest_git(r: &IngestGitReport) -> String {
2001    let mut out = format!(
2002        "ingest-git: {} commit(s), {} file(s), {} author(s){}\n",
2003        r.commits,
2004        r.files,
2005        r.authors,
2006        if r.incremental { " (incremental)" } else { "" }
2007    );
2008    if r.renamed + r.deleted > 0 {
2009        out.push_str(&format!("  renamed {}  deleted {}\n", r.renamed, r.deleted));
2010    }
2011    if r.submodules + r.prs > 0 {
2012        out.push_str(&format!(
2013            "  submodules {}  pull requests {}\n",
2014            r.submodules, r.prs
2015        ));
2016    }
2017    let s = &r.structure;
2018    if s.files_scanned > 0 {
2019        out.push_str(&format!(
2020            "  scanned {} file(s): {} symbol(s), {} import(s), {} call(s), {} mention(s)\n",
2021            s.files_scanned, s.symbols, s.imports, s.calls, s.mentions
2022        ));
2023        if s.skipped_large + s.symbols_capped > 0 {
2024            out.push_str(&format!(
2025                "  hash-only {}  symbol cap hit on {}\n",
2026                s.skipped_large, s.symbols_capped
2027            ));
2028        }
2029    }
2030    if r.gitignore_added {
2031        out.push_str("  added the database directory to .gitignore\n");
2032    }
2033    if !r.rules_created.is_empty() {
2034        out.push_str(&format!("  rules: {}\n", r.rules_created.join(", ")));
2035    }
2036    out
2037}
2038
2039#[cfg(test)]
2040mod tests {
2041    use super::*;
2042
2043    #[test]
2044    fn exclude_matches_prefix_extension_and_substring() {
2045        let pats = vec![
2046            "target/".to_string(),
2047            "*.lock".into(),
2048            "node_modules".into(),
2049        ];
2050        assert!(excluded("target/debug/foo.rs", &pats));
2051        assert!(
2052            !excluded("targeted/foo.rs", &pats),
2053            "prefix needs the slash"
2054        );
2055        assert!(excluded("Cargo.lock", &pats));
2056        assert!(!excluded("Cargo.toml", &pats));
2057        assert!(excluded("ui/node_modules/x/y.js", &pats));
2058        assert!(!excluded("src/lib.rs", &pats));
2059        assert!(!excluded("anything", &[]));
2060    }
2061
2062    /// A `*.` pattern is a file-name suffix, so a compound one works. Reading
2063    /// only the last dot segment would make `*.min.js` — a default — inert,
2064    /// and generated bundles are both the largest files in a tree and the ones
2065    /// least worth parsing.
2066    #[test]
2067    fn a_compound_suffix_pattern_matches() {
2068        let defaults: Vec<String> = DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect();
2069        // Not under `dist/`, so only the suffix rule can match it.
2070        assert!(excluded("ui/build/bundle.min.js", &defaults));
2071        assert!(excluded("bundle.min.js", &defaults));
2072        assert!(
2073            !excluded("ui/src/app.js", &defaults),
2074            "an ordinary source file is not a bundle"
2075        );
2076        assert!(!excluded("ui/src/minify.js", &defaults));
2077        // The single-extension form is unchanged, and a bare suffix is not a
2078        // match: `*.lock` means something *dot* lock.
2079        assert!(excluded("Cargo.lock", &defaults));
2080        assert!(!excluded(".lock", &defaults));
2081        assert!(!excluded("src/lib.rs", &defaults));
2082    }
2083
2084    #[test]
2085    fn file_props_split_dir_and_ext() {
2086        let mut st = FileState::default();
2087        st.touch("sha1", "a@x.test", 200);
2088        let m: BTreeMap<_, _> = file_props(&st, "src/a/b.rs").into_iter().collect();
2089        assert_eq!(m["dir"], Value::Str("src/a".into()));
2090        assert_eq!(m["ext"], Value::Str("rs".into()));
2091        assert_eq!(m["n_commits"], Value::Int(1));
2092        assert_eq!(m["id"], Value::Str("src/a/b.rs".into()));
2093        assert_eq!(m["top_author_id"], Value::Str("a@x.test".into()));
2094        assert_eq!(
2095            m["author_counts"],
2096            Value::List(vec![Value::Str("a@x.test\t1".into())])
2097        );
2098        let m: BTreeMap<_, _> = file_props(&FileState::default(), "README")
2099            .into_iter()
2100            .collect();
2101        assert_eq!(m["dir"], Value::Str(String::new()));
2102        assert_eq!(m["ext"], Value::Str(String::new()));
2103    }
2104
2105    /// The prop is the whole point of the incremental fix: it must survive a
2106    /// round trip so the next run resumes the real distribution, not the
2107    /// incumbent's total.
2108    #[test]
2109    fn author_counts_round_trip_preserves_the_distribution() {
2110        let mut st = FileState::default();
2111        for _ in 0..3 {
2112            st.touch("s", "alice@x.test", 200);
2113        }
2114        for _ in 0..4 {
2115            st.touch("s", "bob@x.test", 200);
2116        }
2117        let Value::List(encoded) = st.author_counts_value() else {
2118            panic!("author_counts must be a list");
2119        };
2120        assert_eq!(
2121            encoded,
2122            vec![
2123                Value::Str("alice@x.test\t3".into()),
2124                Value::Str("bob@x.test\t4".into()),
2125            ],
2126            "email order, so the prop is stable across runs"
2127        );
2128        let mut reloaded = FileState::default();
2129        reloaded.set_author_counts(&encoded);
2130        assert_eq!(reloaded.by_author, st.by_author);
2131        assert_eq!(reloaded.top_author(), "bob@x.test");
2132    }
2133
2134    /// Malformed entries are skipped, not fatal: the run degrades to the counts
2135    /// it can read rather than refusing to sync.
2136    #[test]
2137    fn author_counts_skips_entries_it_cannot_parse() {
2138        let mut st = FileState::default();
2139        st.set_author_counts(&[
2140            Value::Str("alice@x.test\t2".into()),
2141            Value::Str("no-tab-here".into()),
2142            Value::Str("bob@x.test\tnotanumber".into()),
2143            Value::Str("\t5".into()),
2144            Value::Int(7),
2145        ]);
2146        assert_eq!(st.by_author, BTreeMap::from([("alice@x.test".into(), 2)]));
2147    }
2148
2149    #[test]
2150    fn commits_list_is_capped_and_top_author_is_deterministic() {
2151        let mut st = FileState::default();
2152        for i in 0..5 {
2153            st.touch(&format!("sha{i}"), "b@x.test", 3);
2154        }
2155        st.touch("shaX", "a@x.test", 3);
2156        assert_eq!(st.commits, vec!["sha3", "sha4", "shaX"]);
2157        assert_eq!(st.n_commits, 6);
2158        assert_eq!(st.top_author(), "b@x.test");
2159
2160        // Past the cap the stored `n_commits` is the true total, not the length
2161        // of the truncated `commits` list, and `author_counts` sums to the same
2162        // number. A file over `--max-commits-per-file` would otherwise report a
2163        // history frozen at the cap.
2164        let m: BTreeMap<_, _> = file_props(&st, "src/hot.rs").into_iter().collect();
2165        assert_eq!(m["n_commits"], Value::Int(6));
2166        let Value::List(commits) = &m["commits"] else {
2167            panic!("commits must be a list");
2168        };
2169        assert_eq!(commits.len(), 3, "the list is still capped at 3");
2170        assert!(
2171            matches!(m["n_commits"], Value::Int(n) if n as usize > commits.len()),
2172            "n_commits must exceed the capped list once the cap is passed"
2173        );
2174        assert_eq!(
2175            m["author_counts"],
2176            Value::List(vec![
2177                Value::Str("a@x.test\t1".into()),
2178                Value::Str("b@x.test\t5".into()),
2179            ]),
2180            "the per-author counts sum to n_commits, not to the capped list"
2181        );
2182
2183        let mut tie = FileState::default();
2184        tie.touch("s", "b@x.test", 10);
2185        tie.touch("s", "a@x.test", 10);
2186        assert_eq!(tie.top_author(), "a@x.test", "ties break on smallest email");
2187    }
2188}