Skip to main content

cli/
ingest_git.rs

1//! `mushroomdb ingest-git <db> <repo>`: build and maintain a graph of a git
2//! repository. First run ingests the whole history; later runs apply only the
3//! commits after the recorded `GitSync` head, so deletes and renames retract
4//! or move derived edges instead of leaving them stale.
5//!
6//! Graph shape:
7//!
8//! | Label | key (`id`) | props |
9//! |---|---|---|
10//! | `Author` | mailmap-resolved email | `name`, `name_counts` |
11//! | `Commit` | full sha | `message`, `ts`, `author_id`, `pr_id` (with `--prs`) |
12//! | `File` | path | `path`, `dir`, `ext`, `commits`, `n_commits`, `top_author_id`, `author_counts`, plus the working-tree props in [`structure`](crate::structure) |
13//! | `Symbol` | `"<path>#<qualified name>"` | see [`structure`](crate::structure) |
14//! | `PR` | `"pr:<number>"` | `number`, `title`, `url`, `merged_at`, `author_login` |
15//! | `GitSync` | `"__mushroomdb_git_sync__"` | `sha`, `synced_at`, `repo`, `recurse`, `prs`, `structure`, `docs` |
16//!
17//! Edges: user `TOUCHED` Commit→File and `MERGED_AS` PR→Commit, auto-FK
18//! `AUTHOR` Commit→Author, `TOP_AUTHOR` File→Author and `PR` Commit→PR,
19//! rule-derived `CO_CHANGED` File→File and `KNOWS` Author→File — and, from the
20//! working-tree pass, `DEFINES`, `IMPORTS`, `CALLS` and `MENTIONS`.
21//!
22//! With `--recurse-submodules` each initialised submodule is walked as its own
23//! *unit*: its file keys carry the submodule's path in the parent, and it
24//! resumes from its own `GitSync` marker.
25use crate::structure;
26use crate::CliError;
27use core_api::{
28    default_max_edges, Direction, GraphError, IngestOptions, Predicate, ResultSet, RuleDef,
29    SharedDb, Value, WriteGuard, WRITE_LOCK_WAIT,
30};
31use std::collections::{BTreeMap, BTreeSet};
32use std::path::{Path, PathBuf};
33use std::process::Command;
34
35/// Cap on the `commits` list stored per file. Bounds both node size and the
36/// cost of the jaccard overlap the `co_changed` rule runs over that list.
37pub const DEFAULT_MAX_COMMITS_PER_FILE: usize = 200;
38
39/// Applied when the user names no `--exclude` pattern of their own.
40///
41/// Defined in core-api because the `impact` MCP tool builds its default file
42/// list straight off a working tree and has to leave out exactly what this
43/// ingest leaves out; two lists would let the tool ask about files no store was
44/// ever going to hold.
45pub use core_api::repograph::DEFAULT_EXCLUDES;
46
47/// Minimum jaccard overlap of two files' `commits` lists for `CO_CHANGED`.
48const CO_CHANGE_MIN: f64 = 0.25;
49
50/// WAL bytes past the last snapshot that make writing a new one worthwhile.
51///
52/// Every open reads `wal.bin` whole and replays it frame by frame, so the tail
53/// is paid again on every hook, every MCP start and every CLI call — while a
54/// snapshot is paid once by the run that writes it. Measured on this
55/// repository, an 8.1 MB tail costs 156 ms per open against ~440 ms to
56/// snapshot it away, so a tail this size pays its snapshot back within three
57/// opens. Below the threshold the replay is cheap enough that snapshotting on
58/// every incremental run would cost more than it saves.
59pub const SNAPSHOT_WAL_BYTES: u64 = 4 * 1024 * 1024;
60
61/// Key of the singleton `GitSync` node holding the last ingested sha.
62///
63/// Node keys are a single namespace shared with `File` keys, which are repo
64/// paths — so this cannot be `"HEAD"`. A repository with a file named `HEAD`
65/// (git's own `.git/HEAD` aside, plenty of projects ship one) would otherwise
66/// have the sha written onto its `File` node, leaving no sync marker and
67/// forcing a full re-ingest on every run.
68pub(crate) const SYNC_KEY: &str = "__mushroomdb_git_sync__";
69
70/// What the CLI prints, and exits 3 on, when another process holds the store's
71/// cross-process write lock. Nothing was written, so retrying is always safe.
72pub const BUSY_MESSAGE: &str = "another mushroomdb process is writing; retry";
73
74/// Name of the `Commit.pr_id` foreign-key rule. Identical to the name the
75/// zero-config FK inference would choose, so the two never both create it.
76const PR_FK_RULE: &str = "auto_fk_commit_pr_id";
77
78#[derive(Debug, Clone, PartialEq, Eq)]
79pub struct IngestGitOpts {
80    pub repo: PathBuf,
81    /// Paths to skip. Pattern ending in `/` = path prefix, pattern starting
82    /// with `*.` = extension, otherwise = substring of the path.
83    pub exclude: Vec<String>,
84    pub max_commits_per_file: usize,
85    /// Walk every initialised submodule as its own unit.
86    pub recurse_submodules: bool,
87    /// Ask `gh` for merged pull requests and link them to their commits.
88    pub prs: bool,
89    /// Read the working tree: content hashes, `Symbol` nodes, imports and
90    /// calls. Off with `--no-structure`. Recorded on the `GitSync` node.
91    pub structure: bool,
92    /// Index Markdown bodies, headings and mentions. Off with `--no-docs`, and
93    /// inert without `structure`. Recorded on the `GitSync` node.
94    pub docs: bool,
95    /// Add the database directory to the repository's `.gitignore`.
96    pub ensure_gitignore: bool,
97}
98
99#[derive(Debug, Default, Clone, PartialEq, Eq)]
100pub struct IngestGitReport {
101    pub commits: usize,
102    pub files: usize,
103    pub authors: usize,
104    pub renamed: usize,
105    pub deleted: usize,
106    pub incremental: bool,
107    pub rules_created: Vec<String>,
108    /// Submodules walked as their own units.
109    pub submodules: usize,
110    /// Pull requests inserted by this run.
111    pub prs: usize,
112    /// Whether this run appended the database directory to `.gitignore`.
113    pub gitignore_added: bool,
114    /// What the working-tree pass saw. All zeros with `--no-structure`.
115    pub structure: crate::structure::StructureReport,
116}
117
118/// Paths the commit walk left for the working-tree pass to look at.
119#[derive(Default)]
120struct StructureWork {
121    /// Files added, modified, or renamed *into* this window's paths.
122    touched: BTreeSet<String>,
123    /// Keys that no longer name a file: renamed away, or deleted. Any file
124    /// whose `imports` or `mentions` list still holds one of these has to be
125    /// extracted again, or the edge it derived stays behind.
126    stale: BTreeSet<String>,
127}
128
129/// One git working tree walked by a run: the repository itself, or one of its
130/// submodules.
131///
132/// A submodule's paths are keys under `prefix` (its path in the parent), and it
133/// carries its own sync marker, so the two histories advance independently.
134#[derive(Debug, Clone)]
135struct RepoUnit {
136    path: PathBuf,
137    /// `""` for the repository itself, `"<displaypath>/"` for a submodule.
138    prefix: String,
139    sync_key: String,
140}
141
142#[derive(Debug)]
143enum Change {
144    Added(String),
145    Modified(String),
146    Deleted(String),
147    Renamed { from: String, to: String },
148}
149
150#[derive(Debug)]
151struct GitCommit {
152    sha: String,
153    author_name: String,
154    author_email: String,
155    ts: i64,
156    subject: String,
157    changes: Vec<Change>,
158}
159
160/// Simple, dependency-free path matcher. Documented in `docs/site/ingest-git.md`.
161///
162/// The matcher itself is `core_api::repograph::path_excluded`, next to
163/// [`DEFAULT_EXCLUDES`], because the `impact` MCP tool filters a working tree
164/// with the same patterns and must read them the same way.
165fn excluded(path: &str, patterns: &[String]) -> bool {
166    core_api::repograph::path_excluded(path, patterns)
167}
168
169fn git_output(repo: &Path, args: &[&str]) -> Result<std::process::Output, CliError> {
170    Command::new("git")
171        .arg("-C")
172        .arg(repo)
173        .args(args)
174        .output()
175        .map_err(|e| CliError(format!("cannot run git in {}: {e}", repo.display())))
176}
177
178/// `git log --reverse --name-status -M --format=<RS>%H<US>%aN<US>%aE<US>%at<US>%s <range>`
179///
180/// Returns oldest commit first. The walk ends at `head` — never at the symbolic
181/// `HEAD` — so the range is pinned to the same sha the caller will record as the
182/// sync marker. See [`head_sha`] for why that matters.
183///
184/// `%aN` and `%aE` are the mailmap-resolved name and address, so a repository
185/// with a `.mailmap` reports one identity for a contributor who has committed
186/// under several addresses. Without one they are exactly `%an`/`%ae`.
187fn read_log(repo: &Path, since: Option<&str>, head: &str) -> Result<Vec<GitCommit>, CliError> {
188    if let Some(s) = since {
189        let spec = format!("{s}^{{commit}}");
190        if !git_output(repo, &["cat-file", "-e", &spec])?
191            .status
192            .success()
193        {
194            return Err(CliError(format!(
195                "recorded sync head {s} is not in {} (history rewritten?); \
196                 ingest into a fresh database directory",
197                repo.display()
198            )));
199        }
200    }
201    let mut cmd = Command::new("git");
202    cmd.arg("-C").arg(repo).args([
203        // Without this git renders any non-ASCII byte in a path as an octal
204        // escape, so `src/café.rs` would be stored under a mangled key that no
205        // later run matches. Paths containing a tab or newline stay quoted and
206        // escaped either way — git has to, or they would break the format below.
207        "-c",
208        "core.quotePath=false",
209        "log",
210        "--reverse",
211        "--name-status",
212        "-M",
213        "--no-color",
214        // Record separator \x1e between commits, unit separator \x1f between
215        // header fields. A commit subject containing either byte splits its own
216        // record: the message is truncated at the first \x1f, and a \x1e drops
217        // the remainder of that commit's header. Accepted — the parse degrades
218        // to a skipped or shortened message, never a panic or a wrong sha.
219        "--format=%x1e%H%x1f%aN%x1f%aE%x1f%at%x1f%s",
220    ]);
221    match since {
222        Some(s) => cmd.arg(format!("{s}..{head}")),
223        None => cmd.arg(head),
224    };
225    let out = cmd
226        .output()
227        .map_err(|e| CliError(format!("cannot run git: {e}")))?;
228    if !out.status.success() {
229        return Err(CliError(format!(
230            "git log failed: {}",
231            String::from_utf8_lossy(&out.stderr).trim()
232        )));
233    }
234    let text = String::from_utf8_lossy(&out.stdout);
235    let mut commits = Vec::new();
236    for block in text.split('\x1e').filter(|b| !b.trim().is_empty()) {
237        let mut lines = block.lines();
238        let header = lines.next().unwrap_or("");
239        let f: Vec<&str> = header.split('\x1f').collect();
240        if f.len() < 5 {
241            continue;
242        }
243        let mut changes = Vec::new();
244        for l in lines {
245            let cols: Vec<&str> = l.split('\t').collect();
246            match cols.as_slice() {
247                [s, p] if s.starts_with('A') => changes.push(Change::Added((*p).to_string())),
248                [s, p] if s.starts_with('M') || s.starts_with('T') => {
249                    changes.push(Change::Modified((*p).to_string()))
250                }
251                [s, p] if s.starts_with('D') => changes.push(Change::Deleted((*p).to_string())),
252                [s, from, to] if s.starts_with('R') => changes.push(Change::Renamed {
253                    from: (*from).to_string(),
254                    to: (*to).to_string(),
255                }),
256                // A copy is a brand-new path with no prior history here.
257                [s, _from, to] if s.starts_with('C') => {
258                    changes.push(Change::Added((*to).to_string()))
259                }
260                _ => {}
261            }
262        }
263        commits.push(GitCommit {
264            sha: f[0].into(),
265            author_name: f[1].into(),
266            author_email: f[2].into(),
267            ts: f[3].parse().unwrap_or(0),
268            subject: f[4].into(),
269            changes,
270        });
271    }
272    Ok(commits)
273}
274
275/// Resolve `HEAD` to a concrete sha. `Ok(None)` means the repository has no
276/// commits yet; a path that is not a repository at all is an error.
277///
278/// Called **before** [`read_log`], and the resulting sha is both the end of the
279/// walk and the recorded sync marker. Resolving it afterwards instead would open
280/// a window: a commit landing between the walk and the `rev-parse` would push
281/// the marker past a commit that was never ingested, and every later run would
282/// skip it silently. Pinning both to one sha closes that window — a commit that
283/// lands mid-run is simply outside this range and gets picked up next time.
284///
285/// Asking git also beats taking `log.last()`. The two agree in practice, since a
286/// reachability walk emits its tip first and reversing puts it last, but that is
287/// a property of the traversal and of this module's parser rather than something
288/// the resume marker should rest on.
289fn head_sha(repo: &Path) -> Result<Option<String>, CliError> {
290    let out = git_output(repo, &["rev-parse", "--verify", "-q", "HEAD^{commit}"])?;
291    if !out.status.success() {
292        // No commits yet, or not a repository at all — only the latter is an error.
293        if !git_output(repo, &["rev-parse", "--git-dir"])?
294            .status
295            .success()
296        {
297            return Err(CliError(format!(
298                "not a git repository: {}",
299                repo.display()
300            )));
301        }
302        return Ok(None);
303    }
304    let sha = String::from_utf8_lossy(&out.stdout).trim().to_string();
305    if sha.is_empty() {
306        return Err(CliError(format!(
307            "git could not resolve HEAD in {}",
308            repo.display()
309        )));
310    }
311    Ok(Some(sha))
312}
313
314/// Absolute, symlink-free form of `p`, falling back to `p` when it cannot be
315/// resolved (a path that does not exist yet, or a permission error). The result
316/// is what `GitSync.repo` records, so a later run can find the repository again
317/// from any working directory.
318fn canonical(p: &Path) -> PathBuf {
319    std::fs::canonicalize(p).unwrap_or_else(|_| p.to_path_buf())
320}
321
322/// The submodule paths recorded in a unit's `.gitmodules`, relative to that
323/// unit.
324///
325/// These are the paths git reports as ordinary changes in `--name-status`
326/// output while being gitlinks — a commit pointer, not a file. They get no
327/// `File` node whether or not the submodule is walked, and the list is read
328/// from configuration so it is the same for an uninitialised submodule.
329fn gitlink_paths(unit: &Path) -> BTreeSet<String> {
330    let out = Command::new("git")
331        .arg("config")
332        .arg("--file")
333        .arg(unit.join(".gitmodules"))
334        .args(["--get-regexp", r"^submodule\..*\.path$"])
335        .output();
336    let Ok(out) = out else { return BTreeSet::new() };
337    if !out.status.success() {
338        return BTreeSet::new(); // no .gitmodules, so no submodules
339    }
340    String::from_utf8_lossy(&out.stdout)
341        .lines()
342        .filter_map(|l| l.split_once(' '))
343        .map(|(_, path)| path.trim().to_string())
344        .filter(|p| !p.is_empty())
345        .collect()
346}
347
348/// The repository itself, then one unit per initialised submodule when
349/// `recurse` is set.
350///
351/// `git submodule foreach` visits only initialised submodules and says nothing
352/// about the rest, which is the behaviour wanted here: a submodule that was
353/// never checked out has no working tree to walk. `$displaypath` is relative to
354/// the top-level repository, so it is exactly the key prefix its files need.
355fn repo_units(repo: &Path, recurse: bool) -> Result<Vec<RepoUnit>, CliError> {
356    let mut units = vec![RepoUnit {
357        path: repo.to_path_buf(),
358        prefix: String::new(),
359        sync_key: SYNC_KEY.to_string(),
360    }];
361    if !recurse {
362        return Ok(units);
363    }
364    let out = git_output(
365        repo,
366        &[
367            "submodule",
368            "foreach",
369            "--quiet",
370            "--recursive",
371            "echo \"$displaypath\"",
372        ],
373    )?;
374    if !out.status.success() {
375        return Ok(units);
376    }
377    let mut paths: Vec<String> = String::from_utf8_lossy(&out.stdout)
378        .lines()
379        .map(|l| l.trim_end_matches('/').trim().to_string())
380        .filter(|l| !l.is_empty())
381        .collect();
382    paths.sort();
383    paths.dedup();
384    for dp in paths {
385        let path = repo.join(&dp);
386        // `foreach` already skips the uninitialised, but a stale entry or a
387        // removed checkout would otherwise fail the whole run.
388        if !git_output(&path, &["rev-parse", "--git-dir"])?
389            .status
390            .success()
391        {
392            continue;
393        }
394        units.push(RepoUnit {
395            prefix: format!("{dp}/"),
396            sync_key: format!("{SYNC_KEY}:{dp}"),
397            path,
398        });
399    }
400    Ok(units)
401}
402
403/// Rewrite one unit's changes into repository-wide keys: drop gitlinks, then
404/// prepend the unit's prefix so a submodule's `src/lib.rs` is stored under
405/// `vendor/lib/src/lib.rs`.
406///
407/// Done before anything else looks at the log, so exclusion patterns, the
408/// stored keys and the `File` state query all speak the same path.
409fn localise(log: &mut [GitCommit], prefix: &str, gitlinks: &BTreeSet<String>) {
410    let keep = |p: &String| !gitlinks.contains(p.as_str());
411    for c in log.iter_mut() {
412        c.changes.retain(|ch| match ch {
413            Change::Added(p) | Change::Modified(p) | Change::Deleted(p) => keep(p),
414            Change::Renamed { from, to } => keep(from) && keep(to),
415        });
416        if prefix.is_empty() {
417            continue;
418        }
419        for ch in c.changes.iter_mut() {
420            match ch {
421                Change::Added(p) | Change::Modified(p) | Change::Deleted(p) => {
422                    *p = format!("{prefix}{p}")
423                }
424                Change::Renamed { from, to } => {
425                    *from = format!("{prefix}{from}");
426                    *to = format!("{prefix}{to}");
427                }
428            }
429        }
430    }
431}
432
433/// Append `<db dir>/` to the repository's `.gitignore` unless it is already
434/// listed, creating the file if it does not exist. Returns whether it was
435/// written.
436///
437/// A database kept outside the repository is left alone: the repository has no
438/// business ignoring a path it does not contain.
439fn ensure_gitignore(repo: &Path, db_dir: &Path) -> Result<bool, CliError> {
440    let Ok(rel) = canonical(db_dir)
441        .strip_prefix(repo)
442        .map(|p| p.to_path_buf())
443    else {
444        return Ok(false);
445    };
446    if rel.as_os_str().is_empty() {
447        return Ok(false);
448    }
449    let line = format!("{}/", rel.to_string_lossy().replace('\\', "/"));
450    let path = repo.join(".gitignore");
451    let current = match std::fs::read_to_string(&path) {
452        Ok(s) => s,
453        Err(e) if e.kind() == std::io::ErrorKind::NotFound => String::new(),
454        Err(e) => return Err(CliError(format!("cannot read {}: {e}", path.display()))),
455    };
456    let bare = line.trim_end_matches('/');
457    if current
458        .lines()
459        .map(|l| l.trim())
460        .any(|l| l == line || l == bare || l == format!("/{line}") || l == format!("/{bare}"))
461    {
462        return Ok(false);
463    }
464    let mut next = current;
465    if !next.is_empty() && !next.ends_with('\n') {
466        next.push('\n');
467    }
468    next.push_str(&line);
469    next.push('\n');
470    std::fs::write(&path, next)
471        .map_err(|e| CliError(format!("cannot write {}: {e}", path.display())))?;
472    Ok(true)
473}
474
475/// Separator inside an `author_counts` or `name_counts` entry. Neither an email
476/// address nor a git author name can contain a tab, so `<text>\tcount`
477/// round-trips unambiguously.
478const AUTHOR_COUNT_SEP: char = '\t';
479
480/// How many commits carried each spelling of one email's `%aN`, in the order
481/// the spellings were first seen.
482///
483/// One person routinely commits under more than one name — `Ada Lovelace` at
484/// work and `Ada M. Lovelace` from a laptop that was configured once and never
485/// again. Merging them on email is right, and every tool that prints an author
486/// then has to choose *which* of the names to show. Showing the first one
487/// walked means a display name decided by the oldest commit in the window,
488/// which on this repository labelled an identity with a name carried by 21% of
489/// its commits.
490#[derive(Default, Clone, Debug, PartialEq, Eq)]
491struct AuthorState {
492    /// `(name, commits)` in first-seen order, which is what breaks a tie.
493    names: Vec<(String, usize)>,
494}
495
496impl AuthorState {
497    /// Credit one more commit to `name`.
498    fn touch(&mut self, name: &str) {
499        self.add(name, 1);
500    }
501
502    /// Credit `n` more commits to `name`, appending it if it is new. Order is
503    /// preserved, so the entry that was seen first stays first.
504    fn add(&mut self, name: &str, n: usize) {
505        match self.names.iter_mut().find(|(held, _)| held == name) {
506            Some(entry) => entry.1 += n,
507            None => self.names.push((name.to_string(), n)),
508        }
509    }
510
511    /// Fold another window's counts into this one, keeping first-seen order.
512    fn absorb(&mut self, other: &AuthorState) {
513        for (name, n) in &other.names {
514            self.add(name, *n);
515        }
516    }
517
518    /// The name on the most commits. A tie goes to the name seen first, which
519    /// is why the strict `>` matters: it never unseats an equal incumbent.
520    fn display_name(&self) -> String {
521        let mut best: Option<&(String, usize)> = None;
522        for entry in &self.names {
523            if best.is_none_or(|(_, n)| entry.1 > *n) {
524                best = Some(entry);
525            }
526        }
527        best.map(|(name, _)| name.clone()).unwrap_or_default()
528    }
529
530    /// The distribution as an `Author.name_counts` prop: `"name\tcount"`
531    /// strings in first-seen order.
532    ///
533    /// Stored for the same reason `File.author_counts` is: an incremental run
534    /// sees only the new window, and the `name` prop alone cannot say how far
535    /// ahead the incumbent spelling is. Without it every sync would relabel the
536    /// identity from a handful of commits.
537    fn name_counts_value(&self) -> Value {
538        Value::List(
539            self.names
540                .iter()
541                .map(|(name, n)| Value::Str(format!("{name}{AUTHOR_COUNT_SEP}{n}")))
542                .collect(),
543        )
544    }
545
546    /// Inverse of [`AuthorState::name_counts_value`]. Entries that are not
547    /// `name<TAB>count` are skipped rather than failing the run.
548    fn set_name_counts(&mut self, list: &[Value]) {
549        for v in list {
550            let Value::Str(s) = v else { continue };
551            let Some((name, n)) = s.rsplit_once(AUTHOR_COUNT_SEP) else {
552                continue;
553            };
554            let Ok(n) = n.parse::<usize>() else { continue };
555            if !name.is_empty() {
556                self.add(name, n);
557            }
558        }
559    }
560}
561
562fn file_props(st: &FileState, path: &str) -> Vec<(String, Value)> {
563    let commits = &st.commits;
564    let dir = path
565        .rsplit_once('/')
566        .map(|(d, _)| d)
567        .unwrap_or("")
568        .to_string();
569    let ext = path
570        .rsplit_once('.')
571        .map(|(_, e)| e)
572        .unwrap_or("")
573        .to_string();
574    vec![
575        ("id".into(), Value::Str(path.into())),
576        ("path".into(), Value::Str(path.into())),
577        ("dir".into(), Value::Str(dir)),
578        ("ext".into(), Value::Str(ext)),
579        (
580            "commits".into(),
581            Value::List(commits.iter().map(|s| Value::Str(s.clone())).collect()),
582        ),
583        // The true total, which past `--max-commits-per-file` is larger than
584        // the `commits` list it is stored beside. It is also what
585        // `author_counts` sums to, so the two props agree at any history
586        // length.
587        ("n_commits".into(), Value::Int(st.n_commits as i64)),
588        ("top_author_id".into(), Value::Str(st.top_author())),
589        // Written on every touch so the next incremental run can rebuild the
590        // distribution instead of crediting the whole history to the incumbent.
591        ("author_counts".into(), st.author_counts_value()),
592    ]
593}
594
595/// In-memory per-file state accumulated while walking commits.
596#[derive(Default, Clone)]
597struct FileState {
598    commits: Vec<String>,
599    by_author: BTreeMap<String, usize>,
600    n_commits: usize,
601}
602
603impl FileState {
604    fn touch(&mut self, sha: &str, author: &str, cap: usize) {
605        self.commits.push(sha.to_string());
606        if cap > 0 && self.commits.len() > cap {
607            self.commits.remove(0);
608        }
609        self.n_commits += 1;
610        *self.by_author.entry(author.to_string()).or_default() += 1;
611    }
612
613    /// Most commits wins; ties break on the lexicographically smallest email
614    /// so the result is deterministic across runs.
615    fn top_author(&self) -> String {
616        self.by_author
617            .iter()
618            .max_by(|a, b| a.1.cmp(b.1).then(b.0.cmp(a.0)))
619            .map(|(a, _)| a.clone())
620            .unwrap_or_default()
621    }
622
623    /// The per-author distribution as a `File.author_counts` prop: a list of
624    /// `"email\tcount"` strings in email order.
625    ///
626    /// This is the state an incremental run needs and cannot recompute — the
627    /// walk only sees the new window, and `top_author_id` alone cannot say how
628    /// far ahead the incumbent is. Without it a challenger's commits reset on
629    /// every sync and ownership can never change.
630    fn author_counts_value(&self) -> Value {
631        Value::List(
632            self.by_author
633                .iter()
634                .map(|(email, n)| Value::Str(format!("{email}{AUTHOR_COUNT_SEP}{n}")))
635                .collect(),
636        )
637    }
638
639    /// Inverse of [`FileState::author_counts_value`]. Entries that are not
640    /// `email<TAB>count` are skipped rather than failing the run.
641    fn set_author_counts(&mut self, list: &[Value]) {
642        for v in list {
643            let Value::Str(s) = v else { continue };
644            let Some((email, n)) = s.rsplit_once(AUTHOR_COUNT_SEP) else {
645                continue;
646            };
647            let Ok(n) = n.parse::<usize>() else { continue };
648            if !email.is_empty() {
649                *self.by_author.entry(email.to_string()).or_default() += n;
650            }
651        }
652    }
653}
654
655/// One merged pull request as reported by `gh`.
656#[derive(Debug, Clone, PartialEq, Eq)]
657struct PullRequest {
658    number: i64,
659    title: String,
660    url: String,
661    merged_at: String,
662    author_login: String,
663    /// `mergeCommit.oid`, absent for a pull request merged some other way (or
664    /// whose merge commit has since been rewritten).
665    merge_sha: Option<String>,
666}
667
668fn pr_key(number: i64) -> String {
669    format!("pr:{number}")
670}
671
672/// The pull request number in a squash-merge subject: `\(#(\d+)\)$`.
673///
674/// Hand-rolled rather than pulled in as a dependency — the pattern is anchored
675/// at the end of the subject and made of two literals around a run of digits.
676fn subject_pr(subject: &str) -> Option<i64> {
677    let rest = subject.strip_suffix(')')?;
678    let at = rest.rfind("(#")?;
679    let digits = &rest[at + 2..];
680    if digits.is_empty() || !digits.bytes().all(|b| b.is_ascii_digit()) {
681        return None;
682    }
683    digits.parse().ok()
684}
685
686/// `gh pr list --state merged` in `repo`, or an empty list with one warning.
687///
688/// Every failure is a skip: `gh` may not be installed, the repository may have
689/// no GitHub remote, and the user may not be authenticated. None of that is a
690/// reason to fail an ingest that is otherwise complete.
691fn fetch_prs(repo: &Path) -> Vec<PullRequest> {
692    let out = Command::new("gh")
693        .current_dir(repo)
694        .args([
695            "pr",
696            "list",
697            "--state",
698            "merged",
699            "--limit",
700            "1000",
701            "--json",
702            "number,title,url,mergedAt,mergeCommit,author",
703        ])
704        .output();
705    let out = match out {
706        Ok(o) => o,
707        Err(_) => {
708            eprintln!("ingest-git: --prs skipped: gh is not on PATH");
709            return Vec::new();
710        }
711    };
712    if !out.status.success() {
713        let detail = String::from_utf8_lossy(&out.stderr)
714            .lines()
715            .find(|l| !l.trim().is_empty())
716            .unwrap_or("no detail")
717            .to_string();
718        eprintln!("ingest-git: --prs skipped: gh pr list failed: {detail}");
719        return Vec::new();
720    }
721    let parsed: serde_json::Value = match serde_json::from_slice(&out.stdout) {
722        Ok(v) => v,
723        Err(e) => {
724            eprintln!("ingest-git: --prs skipped: gh pr list output is not JSON: {e}");
725            return Vec::new();
726        }
727    };
728    let Some(items) = parsed.as_array() else {
729        eprintln!("ingest-git: --prs skipped: gh pr list did not return a list");
730        return Vec::new();
731    };
732    let mut prs: Vec<PullRequest> = items
733        .iter()
734        .filter_map(|v| {
735            let number = v.get("number")?.as_i64()?;
736            let str_at = |k: &str| {
737                v.get(k)
738                    .and_then(|x| x.as_str())
739                    .unwrap_or_default()
740                    .to_string()
741            };
742            Some(PullRequest {
743                number,
744                title: str_at("title"),
745                url: str_at("url"),
746                merged_at: str_at("mergedAt"),
747                author_login: v
748                    .get("author")
749                    .and_then(|a| a.get("login"))
750                    .and_then(|l| l.as_str())
751                    .unwrap_or_default()
752                    .to_string(),
753                merge_sha: v
754                    .get("mergeCommit")
755                    .and_then(|m| m.get("oid"))
756                    .and_then(|o| o.as_str())
757                    .filter(|s| !s.is_empty())
758                    .map(str::to_string),
759            })
760        })
761        .collect();
762    // Ascending by number: the insert order, and so the node order, is the
763    // same on every run whatever order gh listed them in.
764    prs.sort_by_key(|p| p.number);
765    prs.dedup_by_key(|p| p.number);
766    prs
767}
768
769/// Insert the `PR` nodes that are new, declare the `Commit.pr_id` foreign key,
770/// and index titles for search. Runs before the commits so the FK resolves.
771fn ingest_prs(
772    w: &mut WriteGuard<'_>,
773    prs: &[PullRequest],
774    ingest: &IngestOptions,
775    report: &mut IngestGitReport,
776) -> Result<(), CliError> {
777    let rows: Vec<BTreeMap<String, Value>> = prs
778        .iter()
779        .filter(|p| !w.has_node(&pr_key(p.number)))
780        .map(|p| {
781            BTreeMap::from([
782                ("id".to_string(), Value::Str(pr_key(p.number))),
783                ("number".to_string(), Value::Int(p.number)),
784                ("title".to_string(), Value::Str(p.title.clone())),
785                ("url".to_string(), Value::Str(p.url.clone())),
786                ("merged_at".to_string(), Value::Str(p.merged_at.clone())),
787                (
788                    "author_login".to_string(),
789                    Value::Str(p.author_login.clone()),
790                ),
791            ])
792        })
793        .collect();
794    if !rows.is_empty() {
795        let r = w.ingest_with_edges("PR", rows, ingest, &[])?;
796        report.prs = r.inserted;
797        report.rules_created.extend(r.rules_created);
798    }
799    // Declared here rather than left to FK inference, which only fires on a
800    // batch of `Commit` rows that already carry `pr_id` — a run that links a
801    // pull request to a commit ingested earlier would otherwise leave the
802    // property with no edge behind it.
803    if !w.rules().iter().any(|r| r.name == PR_FK_RULE) {
804        let predicate = Predicate::KeyMatch {
805            field: "pr_id".into(),
806        };
807        let max_edges = Some(default_max_edges(&predicate));
808        w.create_rule(RuleDef {
809            name: PR_FK_RULE.into(),
810            src_label: "Commit".into(),
811            dst_label: "PR".into(),
812            predicate,
813            edge_type: "PR".into(),
814            weight_prop: None,
815            max_edges,
816            approximate: false,
817            via_label: None,
818            via_edge: None,
819            via_dir: None,
820        })?;
821        report.rules_created.push(PR_FK_RULE.to_string());
822    }
823    if !w
824        .fulltext_pairs()
825        .contains(&("PR".to_string(), "title".to_string()))
826    {
827        w.enable_fulltext("PR", "title")?;
828    }
829    Ok(())
830}
831
832/// Point every commit that carries a pull request at it: the merge commit by
833/// sha, a squash merge by its `(#N)` subject. Runs after the commits are in the
834/// graph, so a pull request merged before the last sync is linked too.
835fn link_prs(
836    w: &mut WriteGuard<'_>,
837    prs: &[PullRequest],
838    ingest: &IngestOptions,
839) -> Result<(), CliError> {
840    let by_sha: BTreeMap<&str, i64> = prs
841        .iter()
842        .filter_map(|p| p.merge_sha.as_deref().map(|s| (s, p.number)))
843        .collect();
844    let known: BTreeSet<i64> = prs.iter().map(|p| p.number).collect();
845
846    let rs = w.query(
847        "MATCH (c:Commit) RETURN c.id AS id, c.message AS message, c.pr_id AS pr_id",
848        &BTreeMap::new(),
849    )?;
850    let mut updates: Vec<(String, String)> = Vec::new();
851    let mut links: BTreeMap<i64, BTreeSet<String>> = BTreeMap::new();
852    for i in 0..rs.len() {
853        let Some(Value::Str(sha)) = rs.get(i, "id") else {
854            continue;
855        };
856        let subject = match rs.get(i, "message") {
857            Some(Value::Str(s)) => s.as_str(),
858            _ => "",
859        };
860        let Some(number) = by_sha
861            .get(sha.as_str())
862            .copied()
863            .or_else(|| subject_pr(subject).filter(|n| known.contains(n)))
864        else {
865            continue;
866        };
867        let key = pr_key(number);
868        links.entry(number).or_default().insert(sha.clone());
869        if rs.get(i, "pr_id") != Some(&Value::Str(key.clone())) {
870            updates.push((sha.clone(), key));
871        }
872    }
873    updates.sort(); // by sha: the same store state writes the same records
874    for (sha, key) in updates {
875        w.set_prop(&sha, "pr_id", Value::Str(key))?;
876    }
877
878    let mut edges: Vec<(String, String, String)> = Vec::new();
879    for (number, shas) in links {
880        let src = pr_key(number);
881        let existing: BTreeSet<String> = w
882            .neighbors(&src, "MERGED_AS", Direction::Out)
883            .unwrap_or_default()
884            .into_iter()
885            .collect();
886        for sha in shas {
887            if !existing.contains(&sha) {
888                edges.push(("MERGED_AS".to_string(), src.clone(), sha));
889            }
890        }
891    }
892    if !edges.is_empty() {
893        w.ingest_with_edges("PR", Vec::new(), ingest, &edges)?;
894    }
895    Ok(())
896}
897
898/// Cypher behind [`file_state_from`] — the cumulative state of every live
899/// `File` node belonging to one unit.
900///
901/// The prefix filter is what keeps a submodule's files out of the parent's
902/// walk and vice versa. `startsWith` is the documented Cypher form (see
903/// `docs/site/query.md`); the parent unit's empty prefix matches everything, so
904/// its own submodules' keys are dropped in [`file_state_from`].
905const FILE_STATE_QUERY: &str =
906    "MATCH (f:File) WHERE startsWith(f.id, $prefix) RETURN f.id AS id, f.commits AS commits, \
907     f.n_commits AS n, f.top_author_id AS top, f.author_counts AS author_counts";
908
909/// Rebuild the in-memory per-file state from the `File` nodes already in the
910/// graph so incremental runs keep `commits` and ownership counts cumulative.
911///
912/// `nested` holds the key prefixes of any submodules inside this unit; their
913/// files belong to their own unit's walk and are dropped here.
914fn file_state_from(rs: &ResultSet, nested: &[String]) -> BTreeMap<String, FileState> {
915    let mut files = BTreeMap::new();
916    for i in 0..rs.len() {
917        let id = match rs.get(i, "id") {
918            Some(Value::Str(s)) => s.clone(),
919            _ => continue,
920        };
921        if nested.iter().any(|p| id.starts_with(p.as_str())) {
922            continue;
923        }
924        let mut st = FileState::default();
925        if let Some(Value::List(l)) = rs.get(i, "commits") {
926            st.commits = l
927                .iter()
928                .filter_map(|v| match v {
929                    Value::Str(s) => Some(s.clone()),
930                    _ => None,
931                })
932                .collect();
933        }
934        if let Some(Value::Int(n)) = rs.get(i, "n") {
935            st.n_commits = *n as usize;
936        }
937        match rs.get(i, "author_counts") {
938            Some(Value::List(l)) => st.set_author_counts(l),
939            // A node written before `author_counts` existed (a store built by
940            // 0.4.x). Fall back to the old approximation — the whole prior
941            // history credited to the incumbent — so those stores keep
942            // working. The prop is written on the next touch, from which point
943            // ownership tracks reality; a full re-ingest repairs it at once.
944            _ => {
945                if let Some(Value::Str(t)) = rs.get(i, "top") {
946                    st.by_author.insert(t.clone(), st.n_commits.max(1));
947                }
948            }
949        }
950        files.insert(id, st);
951    }
952    files
953}
954
955/// The real per-name distribution for every author of this unit, read from the
956/// whole history rather than from the window.
957///
958/// A store built by 0.6.0 has `Author` nodes carrying a `name` and no
959/// `name_counts`, and nothing in the graph records which spelling each commit
960/// used — the display name it holds is the *first* one that version happened to
961/// see, which on a repository where one person commits under two names is the
962/// wrong one about as often as not. Guessing from it cannot recover: seeding the
963/// incumbent with the prior commit count makes the true majority wait for more
964/// new commits than the entire history it is already behind by.
965///
966/// So the log is walked in full, once, on the first run that meets such a node.
967/// The result covers every commit up to `head`, this run's window included, so
968/// the caller uses it *instead of* the window rather than on top of it.
969/// A store this version wrote never reaches here.
970fn legacy_name_counts(
971    w: &WriteGuard<'_>,
972    p: &Pending,
973) -> Result<BTreeMap<String, AuthorState>, CliError> {
974    let mut out: BTreeMap<String, AuthorState> = BTreeMap::new();
975    let stale = p.log.iter().any(|c| {
976        w.has_node(&c.author_email)
977            && w.node_ref(&c.author_email)
978                .and_then(|n| n.prop("name_counts"))
979                .is_none()
980    });
981    let Some(head) = p.head.as_deref().filter(|_| stale) else {
982        return Ok(out);
983    };
984    for c in read_log(&p.unit.path, None, head)? {
985        out.entry(c.author_email).or_default().touch(&c.author_name);
986    }
987    Ok(out)
988}
989
990/// Accumulated effect of one log window, before anything is written.
991#[derive(Default)]
992struct Walk {
993    files: BTreeMap<String, FileState>,
994    /// Email → the names its commits were authored under, and how many each.
995    authors: BTreeMap<String, AuthorState>,
996    commit_rows: Vec<BTreeMap<String, Value>>,
997    touched_edges: Vec<(String, String, String)>,
998    /// Paths whose `File` props changed in this window.
999    dirty: BTreeSet<String>,
1000    deleted: BTreeSet<String>,
1001    /// Node renames to apply, collapsed across chained renames.
1002    renamed: Vec<(String, String)>,
1003    /// Any old path → its final path in this window, for edge retargeting.
1004    alias: BTreeMap<String, String>,
1005}
1006
1007impl Walk {
1008    fn rename(&mut self, from: &str, to: &str, node_exists: bool) {
1009        if let Some(e) = self.renamed.iter_mut().find(|(_, t)| t == from) {
1010            e.1 = to.to_string();
1011        } else if node_exists {
1012            // A node cannot be both deleted and moved; the move wins.
1013            self.deleted.remove(from);
1014            self.renamed.push((from.to_string(), to.to_string()));
1015        } else {
1016            self.deleted.insert(from.to_string());
1017        }
1018        for v in self.alias.values_mut() {
1019            if v == from {
1020                *v = to.to_string();
1021            }
1022        }
1023        self.alias.insert(from.to_string(), to.to_string());
1024    }
1025}
1026
1027/// The `GitSync` props that say *how* a unit was ingested, as opposed to how
1028/// far. Compared against the stored node so that changing a flag refreshes the
1029/// marker even when there is nothing new to walk.
1030fn marker_flag_props(unit: &RepoUnit, opts: &IngestGitOpts) -> Vec<(String, Value)> {
1031    vec![
1032        (
1033            "repo".into(),
1034            Value::Str(unit.path.to_string_lossy().into_owned()),
1035        ),
1036        ("recurse".into(), Value::Bool(opts.recurse_submodules)),
1037        ("prs".into(), Value::Bool(opts.prs)),
1038        ("structure".into(), Value::Bool(opts.structure)),
1039        ("docs".into(), Value::Bool(opts.docs)),
1040    ]
1041}
1042
1043/// One unit and everything read about it before the write pass opens.
1044struct Pending {
1045    unit: RepoUnit,
1046    /// The sha its marker resumes from, absent on a first run.
1047    since: Option<String>,
1048    /// The stored marker disagrees with this run's flags.
1049    stale: bool,
1050    /// `None` when the unit has no commits at all.
1051    head: Option<String>,
1052    log: Vec<GitCommit>,
1053    /// Key prefixes of the submodules nested inside this unit, whose files
1054    /// belong to their own walk.
1055    nested: Vec<String>,
1056}
1057
1058/// Whether `db_dir` would open faster if this run left a snapshot behind.
1059///
1060/// A full ingest always qualifies: it is the run that writes the whole history
1061/// into an empty WAL, and leaving that WAL to be replayed on every subsequent
1062/// open is the single largest fixed cost in the hook path. An incremental run
1063/// qualifies when the store has no snapshot at all — a store built by an older
1064/// release, or one whose full ingest predates this rule — or when the tail
1065/// past the last snapshot has grown beyond [`SNAPSHOT_WAL_BYTES`].
1066///
1067/// The tail *is* `wal.bin`: a snapshot replaces the live WAL with a minimal
1068/// baseline, so its length on disk measures exactly what an open has to replay.
1069///
1070/// Only a run that reached its write phase asks. A run with nothing to take
1071/// returns before entering a write scope, so a store with no snapshot and no
1072/// new commits keeps waiting for a run that has work — which in the hook path
1073/// is the next commit.
1074fn snapshot_due(db_dir: &Path, full: bool) -> bool {
1075    if full || !db_dir.join("snapshot.bin").exists() {
1076        return true;
1077    }
1078    std::fs::metadata(db_dir.join("wal.bin")).is_ok_and(|m| m.len() > SNAPSHOT_WAL_BYTES)
1079}
1080
1081/// Write a snapshot if one is due, after the run's own writes are committed.
1082///
1083/// The WAL is archived rather than dropped — see [`AUTOMATIC_SNAPSHOT`], which
1084/// every snapshot mushroomdb takes on its own shares — and every archive is
1085/// kept: [`AUTO_SNAPSHOT_RETENTION`] is `None` by default, so a store that
1086/// syncs on every commit grows an archive per 4 MiB of churn rather than
1087/// losing any of its history. `mushroomdb snapshot <db> --retention N` is how
1088/// a caller bounds that growth once they have decided the trade is worth it.
1089///
1090/// [`AUTOMATIC_SNAPSHOT`]: crate::AUTOMATIC_SNAPSHOT
1091/// [`AUTO_SNAPSHOT_RETENTION`]: crate::AUTO_SNAPSHOT_RETENTION
1092///
1093/// # Why it never fails the run
1094///
1095/// A snapshot is an optimisation for the *next* open, and the data is already
1096/// durable either way:
1097///
1098/// * `Busy` — another process holds the cross-process write lock. Skipped
1099///   without a word; the next run that qualifies takes it.
1100/// * anything else — reported on stderr, because a store that cannot be
1101///   snapshotted is worth knowing about, and the run still succeeds.
1102///
1103/// Call this only from a run that reached its write phase. A run with nothing
1104/// to do returns before entering a write scope, and taking the lock to
1105/// snapshot would turn a genuine no-op into a write.
1106fn snapshot_if_due(db: &SharedDb, db_dir: &Path, full: bool) {
1107    if !snapshot_due(db_dir, full) {
1108        return;
1109    }
1110    let taken = db
1111        .write_with_wait(WRITE_LOCK_WAIT)
1112        .and_then(|mut g| crate::snapshot_automatically(&mut g));
1113    match taken {
1114        Ok(()) | Err(GraphError::Busy { .. }) => {}
1115        Err(e) => eprintln!("snapshot skipped: {e}"),
1116    }
1117}
1118
1119pub fn run_ingest_git(db_dir: &Path, opts: &IngestGitOpts) -> Result<IngestGitReport, CliError> {
1120    // Absolute and symlink-free: it is recorded on the marker, and a later run
1121    // has no reason to share this one's working directory.
1122    let repo = canonical(&opts.repo);
1123    let units = repo_units(&repo, opts.recurse_submodules)?;
1124    let prefixes: Vec<String> = units
1125        .iter()
1126        .map(|u| u.prefix.clone())
1127        .filter(|p| !p.is_empty())
1128        .collect();
1129
1130    let db = SharedDb::open(db_dir)?;
1131    let mut report = IngestGitReport {
1132        submodules: units.len() - 1,
1133        ..Default::default()
1134    };
1135    if opts.ensure_gitignore {
1136        report.gitignore_added = ensure_gitignore(&repo, db_dir)?;
1137    }
1138
1139    let mut pending: Vec<Pending> = Vec::new();
1140    {
1141        let r = db.read();
1142        for unit in units {
1143            let nested = prefixes
1144                .iter()
1145                .filter(|p| **p != unit.prefix && p.starts_with(&unit.prefix))
1146                .cloned()
1147                .collect();
1148            let (since, stale) = match r.node_ref(&unit.sync_key) {
1149                Some(n) => (
1150                    match n.prop("sha") {
1151                        Some(Value::Str(s)) if !s.is_empty() => Some(s),
1152                        _ => None,
1153                    },
1154                    marker_flag_props(&unit, opts)
1155                        .into_iter()
1156                        .any(|(k, v)| n.prop(&k) != Some(v)),
1157                ),
1158                None => (None, false),
1159            };
1160            pending.push(Pending {
1161                unit,
1162                since,
1163                stale,
1164                head: None,
1165                log: Vec::new(),
1166                nested,
1167            });
1168        }
1169    }
1170    // The repository itself is always the first unit, and it is what "this
1171    // database has seen this repo before" means.
1172    report.incremental = pending[0].since.is_some();
1173
1174    for p in &mut pending {
1175        // Pin the end of the walk and the marker to one sha, resolved first. A
1176        // commit landing mid-run then falls outside this range instead of being
1177        // skipped by a marker that advanced past it.
1178        let Some(head) = head_sha(&p.unit.path)? else {
1179            continue; // this unit has no commits yet
1180        };
1181        let mut log = read_log(&p.unit.path, p.since.as_deref(), &head)?;
1182        localise(&mut log, &p.unit.prefix, &gitlink_paths(&p.unit.path));
1183        p.head = Some(head);
1184        p.log = log;
1185    }
1186
1187    let prs = if opts.prs {
1188        fetch_prs(&repo)
1189    } else {
1190        Vec::new()
1191    };
1192    if prs.is_empty() && pending.iter().all(|p| p.log.is_empty() && !p.stale) {
1193        // Nothing new: leave the store untouched so `commit_seq` does not move.
1194        return Ok(report);
1195    }
1196
1197    // Held for the rest of the run. Another process holding the store's
1198    // cross-process write lock is a retry, not a failure: nothing was written.
1199    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1200        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1201        other => CliError(other.to_string()),
1202    })?;
1203    let ingest = IngestOptions::default(); // key `id`, auto-FK suffix `_id`
1204
1205    // Pull requests first: their nodes are what `Commit.pr_id` resolves to.
1206    if !prs.is_empty() {
1207        ingest_prs(&mut w, &prs, &ingest, &mut report)?;
1208    }
1209
1210    let mut authors: BTreeSet<String> = BTreeSet::new();
1211    let mut work = StructureWork::default();
1212    for p in &pending {
1213        if !p.log.is_empty() {
1214            ingest_unit(
1215                &mut w,
1216                p,
1217                opts,
1218                &ingest,
1219                &mut report,
1220                &mut authors,
1221                &mut work,
1222            )?;
1223        }
1224    }
1225    report.authors = authors.len();
1226
1227    // Rules and fulltext, created after the data so each backfills once. They
1228    // span every unit, so they are declared once for the database rather than
1229    // once per repository walked.
1230    if report.commits > 0 && !w.rules().iter().any(|r| r.name == "co_changed") {
1231        let co = Predicate::Overlap {
1232            field: "commits".into(),
1233            min: CO_CHANGE_MIN,
1234        };
1235        w.create_rule(RuleDef {
1236            name: "co_changed".into(),
1237            src_label: "File".into(),
1238            dst_label: "File".into(),
1239            predicate: co.clone(),
1240            edge_type: "CO_CHANGED".into(),
1241            weight_prop: Some("score".into()),
1242            max_edges: Some(10),
1243            approximate: false,
1244            via_label: None,
1245            via_edge: None,
1246            via_dir: None,
1247        })?;
1248        w.create_rule(RuleDef {
1249            name: "knows".into(),
1250            src_label: "Author".into(),
1251            dst_label: "File".into(),
1252            predicate: co,
1253            edge_type: "KNOWS".into(),
1254            weight_prop: Some("score".into()),
1255            max_edges: Some(20),
1256            approximate: false,
1257            via_label: Some("File".into()),
1258            via_edge: Some("TOP_AUTHOR".into()),
1259            via_dir: Some(Direction::In),
1260        })?;
1261        report
1262            .rules_created
1263            .extend(["co_changed".to_string(), "knows".to_string()]);
1264        for (l, field) in [("File", "path"), ("Commit", "message"), ("Author", "name")] {
1265            if !w
1266                .fulltext_pairs()
1267                .contains(&(l.to_string(), field.to_string()))
1268            {
1269                w.enable_fulltext(l, field)?;
1270            }
1271        }
1272    }
1273
1274    // The working tree, on top of the history. It runs after the commit walk
1275    // so every `File` node it reads exists, and its rules are declared after
1276    // its props so each one backfills exactly once.
1277    if opts.structure {
1278        // A first run has nothing to be incremental against, and a run whose
1279        // flags changed (structure or docs just turned on) has to revisit
1280        // files its predecessor deliberately skipped.
1281        let full = !report.incremental || pending.iter().any(|p| p.stale);
1282        report.structure = if full {
1283            structure::refresh_all(&mut w, &repo, "", opts.docs)?
1284        } else {
1285            let mut paths = work.touched;
1286            paths.extend(structure::importers_of(&w, &work.stale)?);
1287            // The retired keys themselves, not just the files that named them.
1288            // An incremental refresh sweeps orphaned symbols only under the
1289            // paths it is handed, and a file this window deleted or renamed
1290            // away is exactly where the orphans are.
1291            paths.extend(work.stale.iter().cloned());
1292            let paths: Vec<String> = paths.into_iter().collect();
1293            structure::refresh_files(&mut w, &repo, "", &paths, opts.docs)?
1294        };
1295        report
1296            .rules_created
1297            .extend(structure::ensure_rules_and_fulltext(&mut w)?);
1298    }
1299
1300    // Every commit this run could link is in the graph by now.
1301    if !prs.is_empty() {
1302        link_prs(&mut w, &prs, &ingest)?;
1303    }
1304
1305    // The markers go last, once every phase of the run has succeeded. They say
1306    // how far a *complete* run got, so a failure anywhere above — a working-tree
1307    // batch that will not commit, a `gh` link pass that errors — leaves them
1308    // where they were and the next run re-walks the same window rather than
1309    // stepping over it. Re-walking a window that was partly applied is safe:
1310    // commits already in the graph are skipped as duplicate keys, file props
1311    // are rewritten from the recomputed state, and a rename whose node already
1312    // moved finds nothing to move.
1313    for p in &pending {
1314        write_marker(&mut w, p, opts)?;
1315    }
1316
1317    // Every write this run makes is now committed, so leave the store in the
1318    // shape that opens fastest. The guard is released first: `snapshot()` takes
1319    // the same cross-process write lock this one holds.
1320    drop(w);
1321    snapshot_if_due(&db, db_dir, !report.incremental);
1322    Ok(report)
1323}
1324
1325/// Wall-clock seconds since the Unix epoch, for [`SYNCED_AT`].
1326///
1327/// This is the one place the ingest reads a clock. A clock before the epoch
1328/// reads as `0` rather than going negative.
1329fn now_unix() -> i64 {
1330    std::time::SystemTime::now()
1331        .duration_since(std::time::UNIX_EPOCH)
1332        .map_or(0, |d| i64::try_from(d.as_secs()).unwrap_or(i64::MAX))
1333}
1334
1335/// Marker prop holding when this store last took data from the repository.
1336///
1337/// `Commit.ts` says when the work was *written*, which on a store synced to
1338/// its repository's head is indistinguishable from now — so it cannot answer
1339/// "how stale is my graph". This can. Absent on a store built before it
1340/// existed, which readers must tolerate.
1341pub const SYNCED_AT: &str = "synced_at";
1342
1343/// Record how far this unit got, and under what flags.
1344///
1345/// Props are written only where they differ, so a run that touches one unit
1346/// does not churn the markers of the others — and a run that changes nothing
1347/// writes nothing at all, [`SYNCED_AT`] included.
1348///
1349/// The stamp therefore means "when this run last changed this marker": a new
1350/// head, or a flag ingested differently from last time. A no-op re-run leaving
1351/// it alone is the point rather than a gap — a graph that had nothing to take
1352/// is not staler for having checked.
1353fn write_marker(w: &mut WriteGuard<'_>, p: &Pending, opts: &IngestGitOpts) -> Result<(), CliError> {
1354    let Some(head) = p.head.as_deref() else {
1355        return Ok(()); // no commits, so nothing to resume from
1356    };
1357    let mut props = marker_flag_props(&p.unit, opts);
1358    if !p.log.is_empty() {
1359        props.push(("sha".into(), Value::Str(head.to_string())));
1360    }
1361    let key = p.unit.sync_key.clone();
1362    if w.has_node(&key) {
1363        let mut changed = false;
1364        for (k, v) in props {
1365            let current = w.node_ref(&key).and_then(|n| n.prop(&k));
1366            if current.as_ref() != Some(&v) {
1367                w.set_prop(&key, &k, v)?;
1368                changed = true;
1369            }
1370        }
1371        if changed {
1372            w.set_prop(&key, SYNCED_AT, Value::Int(now_unix()))?;
1373        }
1374        return Ok(());
1375    }
1376    if p.log.is_empty() {
1377        return Ok(());
1378    }
1379    // It carries `id` like every other label here, so the key is readable from
1380    // Cypher.
1381    props.push(("id".into(), Value::Str(key.clone())));
1382    props.push((SYNCED_AT.into(), Value::Int(now_unix())));
1383    props.sort_by(|a, b| a.0.cmp(&b.0));
1384    w.insert_node("GitSync", &key, props)?;
1385    Ok(())
1386}
1387
1388/// Walk one unit's new commits into the graph.
1389///
1390/// Every path in `p.log` is already a repository-wide key, so this is the
1391/// single-repository algorithm unchanged: only the `File` state it starts from
1392/// and the counts it adds to are scoped to the unit.
1393fn ingest_unit(
1394    w: &mut WriteGuard<'_>,
1395    p: &Pending,
1396    opts: &IngestGitOpts,
1397    ingest: &IngestOptions,
1398    report: &mut IngestGitReport,
1399    authors: &mut BTreeSet<String>,
1400    work: &mut StructureWork,
1401) -> Result<(), CliError> {
1402    let log = &p.log;
1403    let incremental = p.since.is_some();
1404
1405    let mut walk = Walk {
1406        files: if incremental {
1407            let params =
1408                BTreeMap::from([("prefix".to_string(), Value::Str(p.unit.prefix.clone()))]);
1409            file_state_from(&w.query(FILE_STATE_QUERY, &params)?, &p.nested)
1410        } else {
1411            BTreeMap::new()
1412        },
1413        ..Default::default()
1414    };
1415
1416    for c in log {
1417        walk.authors
1418            .entry(c.author_email.clone())
1419            .or_default()
1420            .touch(&c.author_name);
1421        walk.commit_rows.push(BTreeMap::from([
1422            ("id".to_string(), Value::Str(c.sha.clone())),
1423            ("message".to_string(), Value::Str(c.subject.clone())),
1424            ("ts".to_string(), Value::Int(c.ts)),
1425            ("author_id".to_string(), Value::Str(c.author_email.clone())),
1426        ]));
1427        for ch in &c.changes {
1428            match ch {
1429                Change::Added(p) | Change::Modified(p) => {
1430                    if excluded(p, &opts.exclude) {
1431                        continue;
1432                    }
1433                    walk.deleted.remove(p);
1434                    walk.files.entry(p.clone()).or_default().touch(
1435                        &c.sha,
1436                        &c.author_email,
1437                        opts.max_commits_per_file,
1438                    );
1439                    walk.dirty.insert(p.clone());
1440                    walk.touched_edges
1441                        .push(("TOUCHED".into(), c.sha.clone(), p.clone()));
1442                }
1443                Change::Deleted(p) => {
1444                    if excluded(p, &opts.exclude) {
1445                        continue;
1446                    }
1447                    walk.files.remove(p);
1448                    walk.dirty.remove(p);
1449                    walk.deleted.insert(p.clone());
1450                }
1451                Change::Renamed { from, to } => {
1452                    if excluded(to, &opts.exclude) {
1453                        // Moved out of scope: drop the old node, keep no alias
1454                        // so its TOUCHED edges are filtered out below.
1455                        walk.files.remove(from);
1456                        walk.dirty.remove(from);
1457                        walk.deleted.insert(from.clone());
1458                        continue;
1459                    }
1460                    let mut st = walk.files.remove(from).unwrap_or_default();
1461                    st.touch(&c.sha, &c.author_email, opts.max_commits_per_file);
1462                    walk.files.insert(to.clone(), st);
1463                    walk.dirty.remove(from);
1464                    walk.dirty.insert(to.clone());
1465                    walk.deleted.remove(to);
1466                    walk.touched_edges
1467                        .push(("TOUCHED".into(), c.sha.clone(), to.clone()));
1468                    let exists = incremental && w.has_node(from);
1469                    walk.rename(from, to, exists);
1470                }
1471            }
1472        }
1473    }
1474
1475    // What the working-tree pass has to look at afterwards: the paths this
1476    // window changed, and the keys it left pointing at nothing. A key that
1477    // ended the window live again — a file moved away and back — is neither.
1478    work.touched.extend(walk.dirty.iter().cloned());
1479    for key in walk.deleted.iter().chain(walk.alias.keys()) {
1480        if !walk.files.contains_key(key) {
1481            work.stale.insert(key.clone());
1482        }
1483    }
1484
1485    // 1. Authors first: the auto-FK rules for `Commit.author_id` and
1486    //    `File.top_author_id` only infer once their targets resolve to Author.
1487    //    An email already in the graph keeps its accumulated `name_counts`, so
1488    //    the display name is the majority spelling over the whole history
1489    //    rather than over this window.
1490    let legacy = legacy_name_counts(w, p)?;
1491    let mut author_rows = Vec::new();
1492    let mut author_updates = Vec::new();
1493    for (email, seen) in &walk.authors {
1494        let mut merged = AuthorState::default();
1495        match (
1496            w.node_ref(email).and_then(|n| n.prop("name_counts")),
1497            legacy.get(email),
1498        ) {
1499            // The usual path: what the graph already counted, plus this window.
1500            (Some(Value::List(l)), _) => {
1501                merged.set_name_counts(&l);
1502                merged.absorb(seen);
1503            }
1504            // A node written before `name_counts` existed. The recovered walk
1505            // is the whole history and already contains this window, so the
1506            // window is not added to it.
1507            (_, Some(whole_history)) => merged = whole_history.clone(),
1508            // A new author, or a unit whose history was already recovered.
1509            _ => merged.absorb(seen),
1510        }
1511        let props = [
1512            ("name", Value::Str(merged.display_name())),
1513            ("name_counts", merged.name_counts_value()),
1514        ];
1515        if w.has_node(email) {
1516            for (field, want) in props {
1517                if w.node_ref(email).and_then(|n| n.prop(field)).as_ref() != Some(&want) {
1518                    author_updates.push((email.clone(), field, want));
1519                }
1520            }
1521        } else {
1522            let mut row = BTreeMap::from([("id".to_string(), Value::Str(email.clone()))]);
1523            row.extend(props.into_iter().map(|(f, v)| (f.to_string(), v)));
1524            author_rows.push(row);
1525        }
1526    }
1527    if !author_updates.is_empty() {
1528        let mut b = w.batch();
1529        for (key, field, value) in author_updates {
1530            b.set_prop(&key, field, value);
1531        }
1532        b.commit()?;
1533    }
1534    let a = w.ingest_with_edges("Author", author_rows, ingest, &[])?;
1535    report.rules_created.extend(a.rules_created);
1536    authors.extend(walk.authors.keys().cloned());
1537
1538    // 2. Deletes run first so a rename can claim a path freed in this same
1539    //    window, then renames carry each node (and its history) to its new path.
1540    //    `walk.files` is the authority on what is still live: a rename whose
1541    //    destination is not in it is a delete, not a move.
1542    for p in &walk.deleted {
1543        if w.has_node(p) {
1544            w.delete_node(p)?;
1545            report.deleted += 1;
1546        }
1547    }
1548    for (from, to) in &walk.renamed {
1549        if !w.has_node(from) || from == to {
1550            // Nothing to move, or a rename that swapped back to its own path.
1551            continue;
1552        }
1553        if !walk.files.contains_key(to) {
1554            // The destination did not survive the window — it was deleted, or
1555            // moved into an excluded path, after this rename. The node goes
1556            // with it; renaming into a dead path would strand a phantom node
1557            // that no later phase refreshes.
1558            w.delete_node(from)?;
1559            report.deleted += 1;
1560            continue;
1561        }
1562        if w.has_node(to) {
1563            // A pre-existing node already holds the destination path (deleted
1564            // earlier in this window, then claimed by this rename).
1565            w.delete_node(to)?;
1566            report.deleted += 1;
1567        }
1568        w.rename_node(from, to)?;
1569        // The key moved, so the `id` prop must move with it. Phase 3 also sets
1570        // it for every dirty path; doing it here keeps the invariant local to
1571        // the rename and independent of that filter.
1572        w.set_prop(to, "id", Value::Str(to.clone()))?;
1573        report.renamed += 1;
1574    }
1575
1576    // 3. File nodes. Existing nodes are updated in place (including `id`, which
1577    //    must follow the key after a rename); new paths go through ingest so
1578    //    the `top_author_id` auto-FK rule is inferred.
1579    //    Updates to existing nodes go in one batch rather than one WAL commit
1580    //    per property. Every frame in the WAL is replayed on every later open,
1581    //    and each one re-fires the rules watching the property it carries, so a
1582    //    run that appends a frame per property makes every subsequent open
1583    //    slower for as long as that frame lives. The op order inside the batch
1584    //    is the order the individual writes had, so the resulting state is the
1585    //    same one either form produces.
1586    let mut new_file_rows = Vec::new();
1587    let mut updates = Vec::new();
1588    let mut written = 0usize;
1589    for (path, st) in &walk.files {
1590        if incremental && !walk.dirty.contains(path) {
1591            continue;
1592        }
1593        written += 1;
1594        let props = file_props(st, path);
1595        if w.has_node(path) {
1596            updates.extend(props.into_iter().map(|(k, v)| (path.clone(), k, v)));
1597        } else {
1598            new_file_rows.push(props.into_iter().collect::<BTreeMap<_, _>>());
1599        }
1600    }
1601    if !updates.is_empty() {
1602        let mut b = w.batch();
1603        for (key, field, value) in updates {
1604            b.set_prop(&key, &field, value);
1605        }
1606        b.commit()?;
1607    }
1608    let f = w.ingest_with_edges("File", new_file_rows, ingest, &[])?;
1609    report.rules_created.extend(f.rules_created);
1610    // What this run wrote, not every file it has ever seen: an incremental run
1611    // loads the whole known file set to fold the new commits into it, and
1612    // reporting that set made a one-file run print the size of the repository.
1613    report.files += written;
1614
1615    // 4. Commits, then their TOUCHED edges. The two must be separate batches:
1616    //    a batch that both inserts nodes firing a new rule and carries a user
1617    //    edge of a not-yet-interned type writes a WAL frame that cannot be
1618    //    replayed (`Intern` records are emitted in a pre-pass, but on replay the
1619    //    rule fires — and interns its edge type — before the later `Intern`
1620    //    record is read). See the report for a reproducer.
1621    let c = w.ingest_with_edges("Commit", walk.commit_rows, ingest, &[])?;
1622    report.rules_created.extend(c.rules_created);
1623    report.commits += log.len();
1624
1625    // Edges name File keys, so files must already exist. A path renamed later
1626    // in this same window is retargeted to where its node ended up.
1627    let touched: Vec<(String, String, String)> = walk
1628        .touched_edges
1629        .into_iter()
1630        .map(|(t, sha, p)| {
1631            let p = walk.alias.get(&p).cloned().unwrap_or(p);
1632            (t, sha, p)
1633        })
1634        .filter(|(_, _, p)| walk.files.contains_key(p))
1635        .collect();
1636    if !touched.is_empty() {
1637        w.ingest_with_edges("Commit", Vec::new(), ingest, &touched)?;
1638    }
1639    Ok(())
1640}
1641
1642// ── sync and touch ──────────────────────────────────────────────────────────
1643//
1644// `ingest-git` is the command a person runs. These two are what a *hook* runs:
1645// `sync` after a commit lands, `touch` after a single file is edited. Both read
1646// the repository out of the `GitSync` marker rather than taking it as an
1647// argument, so a hook line carries only the database path and keeps working
1648// when the checkout moves.
1649
1650/// What a store with no `GitSync` node is told. Naming the fix matters: this is
1651/// the error a hook installed against the wrong database prints.
1652const NO_MARKER: &str = "store has no git sync marker; run ingest-git first";
1653
1654/// The `GitSync` props that say how this store was built, read back so a later
1655/// `sync` repeats the same run without being told any of it again.
1656///
1657/// `exclude` and `max_commits_per_file` are not on the marker, so a `sync`
1658/// applies [`DEFAULT_EXCLUDES`] and [`DEFAULT_MAX_COMMITS_PER_FILE`]. A store
1659/// first built with custom `--exclude` patterns should keep being maintained
1660/// with `ingest-git`, which takes them.
1661#[derive(Debug, Clone, PartialEq, Eq)]
1662struct SyncMarker {
1663    repo: PathBuf,
1664    recurse: bool,
1665    prs: bool,
1666    structure: bool,
1667    docs: bool,
1668}
1669
1670impl SyncMarker {
1671    fn opts(&self) -> IngestGitOpts {
1672        IngestGitOpts {
1673            repo: self.repo.clone(),
1674            exclude: DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect(),
1675            max_commits_per_file: DEFAULT_MAX_COMMITS_PER_FILE,
1676            recurse_submodules: self.recurse,
1677            prs: self.prs,
1678            structure: self.structure,
1679            docs: self.docs,
1680            ensure_gitignore: false,
1681        }
1682    }
1683}
1684
1685/// Read the marker off an already-open handle.
1686///
1687/// Deliberately not "open the store and read the marker": opening this store is
1688/// by far the most expensive thing either command does — it replays the whole
1689/// WAL — so both of them open once and read the marker through that same
1690/// handle. A separate read-only open just to learn the repository path would
1691/// double the cost of every hook invocation.
1692fn marker_of(r: &structure::Db) -> Result<SyncMarker, CliError> {
1693    let node = r
1694        .node_ref(SYNC_KEY)
1695        .ok_or_else(|| CliError(NO_MARKER.into()))?;
1696    let flag = |name: &str| matches!(node.prop(name), Some(Value::Bool(true)));
1697    match node.prop("repo") {
1698        Some(Value::Str(repo)) if !repo.is_empty() => Ok(SyncMarker {
1699            repo: PathBuf::from(repo),
1700            recurse: flag("recurse"),
1701            prs: flag("prs"),
1702            // A marker written before these two flags existed carries neither,
1703            // and a working-tree pass is what such a run did.
1704            structure: node.prop("structure") != Some(Value::Bool(false)),
1705            docs: node.prop("docs") != Some(Value::Bool(false)),
1706        }),
1707        _ => Err(CliError(NO_MARKER.into())),
1708    }
1709}
1710
1711/// Open a store that must already exist.
1712///
1713/// `SharedDb::open` runs `create_dir_all`, so without this guard a hook line
1714/// carrying a typo'd path would keep creating empty databases and reporting
1715/// that they hold no marker — the same trap [`run_recall`] guards against.
1716///
1717/// [`run_recall`]: crate::recall::run_recall
1718fn open_existing(db_dir: &Path) -> Result<SharedDb, CliError> {
1719    if !db_dir.exists() {
1720        return Err(CliError(format!(
1721            "no database directory at {}",
1722            db_dir.display()
1723        )));
1724    }
1725    Ok(SharedDb::open(db_dir)?)
1726}
1727
1728/// What one [`run_sync`] did.
1729#[derive(Debug, Default, Clone, PartialEq, Eq)]
1730pub struct SyncReport {
1731    /// The incremental history walk.
1732    pub git: IngestGitReport,
1733    /// The working-tree pass over the dirty paths, which the history walk does
1734    /// not see: an edit that has not been committed is in no commit.
1735    pub structure: crate::structure::StructureReport,
1736    /// Dirty paths handed to that pass. Higher than `structure.files_scanned`
1737    /// when some of them are new to the graph, or no longer on disk.
1738    pub dirty_refreshed: usize,
1739}
1740
1741/// Paths that differ from `HEAD` or are not tracked at all, repository-relative
1742/// and sorted.
1743///
1744/// `-z` rather than the default listing: git escapes and quotes a path holding
1745/// a tab, a newline or a non-ASCII byte, and a quoted path matches no key.
1746fn dirty_paths(repo: &Path, exclude: &[String]) -> Result<Vec<String>, CliError> {
1747    const LISTS: [&[&str]; 2] = [
1748        &["diff", "--name-only", "-z", "HEAD"],
1749        &["ls-files", "--others", "--exclude-standard", "-z"],
1750    ];
1751    let mut out = BTreeSet::new();
1752    for args in LISTS {
1753        let o = git_output(repo, args)?;
1754        if !o.status.success() {
1755            // `diff HEAD` fails in a repository with no commits yet. Nothing is
1756            // dirty relative to a head that does not exist.
1757            continue;
1758        }
1759        for path in String::from_utf8_lossy(&o.stdout).split('\0') {
1760            if path.is_empty() || excluded(path, exclude) {
1761                continue;
1762            }
1763            out.insert(path.to_string());
1764        }
1765    }
1766    Ok(out.into_iter().collect())
1767}
1768
1769/// Bring the store up to date with the repository it was built from: the
1770/// commits since the marker, then the working tree where it differs from
1771/// `HEAD`.
1772///
1773/// The second half is what a plain `ingest-git` cannot do. Its working-tree
1774/// pass only visits the paths the *commits* touched, so a file edited and not
1775/// yet committed keeps whatever the graph last recorded about it. A hook that
1776/// runs on every commit wants the uncommitted remainder refreshed too.
1777pub fn run_sync(db_dir: &Path) -> Result<SyncReport, CliError> {
1778    // One handle for the whole run. It is opened before `run_ingest_git`, which
1779    // opens its own and commits through it, but a `SharedDb` refreshes off the
1780    // WAL when a write scope is entered — so the dirty pass below sees every
1781    // commit the ingest just made without this handle being reopened.
1782    let db = open_existing(db_dir)?;
1783    let marker = marker_of(&db.read())?;
1784    let opts = marker.opts();
1785    let mut report = SyncReport {
1786        git: run_ingest_git(db_dir, &opts)?,
1787        ..Default::default()
1788    };
1789    if !opts.structure {
1790        return Ok(report);
1791    }
1792
1793    // Only the root repository's working tree. A submodule's dirty files are
1794    // its own checkout's business, and `--recurse-submodules` resumes each unit
1795    // from its own marker on the next commit there.
1796    let repo = canonical(&marker.repo);
1797    let paths = dirty_paths(&repo, &opts.exclude)?;
1798    report.dirty_refreshed = paths.len();
1799    if paths.is_empty() {
1800        // Nothing to refresh: never enter a write scope, so the run takes no
1801        // lock and `commit_seq` cannot move.
1802        return Ok(report);
1803    }
1804
1805    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1806        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1807        other => CliError(other.to_string()),
1808    })?;
1809    report.structure = structure::refresh_files(&mut w, &repo, "", &paths, opts.docs)?;
1810
1811    // The history half already snapshotted if this was a first ingest; this
1812    // covers the tail the dirty pass just appended. `full` is false because a
1813    // sync is by definition a run against a store that already exists.
1814    drop(w);
1815    snapshot_if_due(&db, db_dir, false);
1816    Ok(report)
1817}
1818
1819pub fn format_touch(r: &structure::StructureReport) -> String {
1820    format!(
1821        "touch: {} file(s), {} symbol(s), {} import(s), {} call(s), {} mention(s)\n",
1822        r.files_scanned, r.symbols, r.imports, r.calls, r.mentions
1823    )
1824}
1825
1826/// One [`SyncReport`] as a single JSON object, for `sync --json`.
1827///
1828/// `text` is exactly what [`format_sync`] prints, so a caller that has the
1829/// object never has to render the digest a second way and never has to parse
1830/// the counts back out of the prose. The MCP `sync` tool is that caller: it
1831/// runs this binary and hands both halves straight to the assistant.
1832#[must_use]
1833pub fn format_sync_json(r: &SyncReport) -> String {
1834    let g = &r.git;
1835    let s = &r.structure;
1836    let gs = &g.structure;
1837    let value = serde_json::json!({
1838        "text": format_sync(r),
1839        "commits": g.commits,
1840        "files": g.files,
1841        "authors": g.authors,
1842        "renamed": g.renamed,
1843        "deleted": g.deleted,
1844        "incremental": g.incremental,
1845        "submodules": g.submodules,
1846        "prs": g.prs,
1847        "rules_created": g.rules_created,
1848        "scanned": {
1849            "files": gs.files_scanned,
1850            "symbols": gs.symbols,
1851            "imports": gs.imports,
1852            "calls": gs.calls,
1853            "mentions": gs.mentions,
1854        },
1855        "dirty_refreshed": r.dirty_refreshed,
1856        "dirty": {
1857            "files": s.files_scanned,
1858            "symbols": s.symbols,
1859            "imports": s.imports,
1860            "calls": s.calls,
1861            "mentions": s.mentions,
1862        },
1863    });
1864    format!("{value}\n")
1865}
1866
1867pub fn format_sync(r: &SyncReport) -> String {
1868    let mut out = format_ingest_git(&r.git);
1869    let s = &r.structure;
1870    out.push_str(&format!(
1871        "  dirty {} path(s): scanned {}, {} symbol(s), {} import(s), {} call(s)\n",
1872        r.dirty_refreshed, s.files_scanned, s.symbols, s.imports, s.calls
1873    ));
1874    out
1875}
1876
1877/// Re-extract exactly the files named, and nothing else.
1878///
1879/// `files` comes from argv when a caller has the paths; otherwise they are read
1880/// out of a `PostToolUse` hook payload on stdin, the same way [`run_recall`]
1881/// reads a prompt. Anything that is not a working-tree file this store already
1882/// knows — a path outside the repository, an excluded one, one the graph has
1883/// never seen — is dropped without comment, because a hook fires on every edit
1884/// the assistant makes and most of them are none of this store's business.
1885///
1886/// [`run_recall`]: crate::recall::run_recall
1887pub fn run_touch(
1888    db_dir: &Path,
1889    files: &[PathBuf],
1890    hook_stdin: Option<&str>,
1891) -> Result<structure::StructureReport, CliError> {
1892    let named: Vec<PathBuf> = if files.is_empty() {
1893        hook_stdin.map(paths_from_payload).unwrap_or_default()
1894    } else {
1895        files.to_vec()
1896    };
1897    if named.is_empty() {
1898        return Ok(structure::StructureReport::default());
1899    }
1900
1901    let db = open_existing(db_dir)?;
1902    let marker = marker_of(&db.read())?;
1903    if !marker.structure {
1904        // The store was built with `--no-structure`, so it holds no working-tree
1905        // props at all and re-extracting one file would be the only exception.
1906        return Ok(structure::StructureReport::default());
1907    }
1908    let repo = canonical(&marker.repo);
1909    let exclude: Vec<String> = DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect();
1910    let cwd = std::env::current_dir().unwrap_or_else(|_| PathBuf::from("."));
1911    let mut paths: BTreeSet<String> = BTreeSet::new();
1912    for path in &named {
1913        let Some(rel) = repo_relative(&repo, &cwd, path) else {
1914            continue;
1915        };
1916        if !excluded(&rel, &exclude) {
1917            paths.insert(rel);
1918        }
1919    }
1920    if paths.is_empty() {
1921        // Nothing of ours changed: never enter a write scope, so no lock is
1922        // taken and a running writer is never made to wait.
1923        return Ok(structure::StructureReport::default());
1924    }
1925
1926    let paths: Vec<String> = paths.into_iter().collect();
1927    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1928        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1929        other => CliError(other.to_string()),
1930    })?;
1931    structure::refresh_files(&mut w, &repo, "", &paths, marker.docs)
1932}
1933
1934/// The file paths in a `PostToolUse` payload.
1935///
1936/// `tool_input.file_path` covers Edit and Write; `tool_input.edits[].file_path`
1937/// covers the multi-edit shape. Both are read, so a payload carrying either (or
1938/// both) is handled without knowing which tool produced it. A payload that is
1939/// not JSON, or that names no file, yields nothing — never an error.
1940fn paths_from_payload(raw: &str) -> Vec<PathBuf> {
1941    let Ok(v) = serde_json::from_str::<serde_json::Value>(raw) else {
1942        return Vec::new();
1943    };
1944    let input = &v["tool_input"];
1945    let mut out = Vec::new();
1946    let mut push = |value: &serde_json::Value| {
1947        if let Some(s) = value.as_str().map(str::trim).filter(|s| !s.is_empty()) {
1948            out.push(PathBuf::from(s));
1949        }
1950    };
1951    push(&input["file_path"]);
1952    if let Some(edits) = input["edits"].as_array() {
1953        for e in edits {
1954            push(&e["file_path"]);
1955        }
1956    }
1957    out
1958}
1959
1960/// `path` as a key under `repo`, or `None` when it is not inside it.
1961///
1962/// A hook payload carries absolute paths and a person typing the command uses
1963/// relative ones, so a relative path is taken against `cwd`. Both sides are
1964/// then resolved through symlinks before comparing: the marker records the
1965/// canonical repository path, and on macOS a checkout under `/tmp` is reached
1966/// through a symlink that never compares equal as written.
1967fn repo_relative(repo: &Path, cwd: &Path, path: &Path) -> Option<String> {
1968    let absolute = if path.is_absolute() {
1969        path.to_path_buf()
1970    } else {
1971        cwd.join(path)
1972    };
1973    let resolved = resolve_symlinks(&absolute);
1974    let rel = resolved.strip_prefix(repo).ok()?;
1975    let key: Vec<String> = rel
1976        .components()
1977        .map(|c| c.as_os_str().to_string_lossy().into_owned())
1978        .collect();
1979    let key = key.join("/");
1980    (!key.is_empty()).then_some(key)
1981}
1982
1983/// [`canonical`] that still works for a path that no longer exists: a file
1984/// deleted between the edit and the hook resolves through its parent directory.
1985fn resolve_symlinks(p: &Path) -> PathBuf {
1986    if let Ok(resolved) = std::fs::canonicalize(p) {
1987        return resolved;
1988    }
1989    match (p.parent(), p.file_name()) {
1990        (Some(dir), Some(name)) => std::fs::canonicalize(dir)
1991            .map(|d| d.join(name))
1992            .unwrap_or_else(|_| p.to_path_buf()),
1993        _ => p.to_path_buf(),
1994    }
1995}
1996
1997pub fn format_ingest_git(r: &IngestGitReport) -> String {
1998    let mut out = format!(
1999        "ingest-git: {} commit(s), {} file(s), {} author(s){}\n",
2000        r.commits,
2001        r.files,
2002        r.authors,
2003        if r.incremental { " (incremental)" } else { "" }
2004    );
2005    if r.renamed + r.deleted > 0 {
2006        out.push_str(&format!("  renamed {}  deleted {}\n", r.renamed, r.deleted));
2007    }
2008    if r.submodules + r.prs > 0 {
2009        out.push_str(&format!(
2010            "  submodules {}  pull requests {}\n",
2011            r.submodules, r.prs
2012        ));
2013    }
2014    let s = &r.structure;
2015    if s.files_scanned > 0 {
2016        out.push_str(&format!(
2017            "  scanned {} file(s): {} symbol(s), {} import(s), {} call(s), {} mention(s)\n",
2018            s.files_scanned, s.symbols, s.imports, s.calls, s.mentions
2019        ));
2020        if s.skipped_large + s.symbols_capped > 0 {
2021            out.push_str(&format!(
2022                "  hash-only {}  symbol cap hit on {}\n",
2023                s.skipped_large, s.symbols_capped
2024            ));
2025        }
2026    }
2027    if r.gitignore_added {
2028        out.push_str("  added the database directory to .gitignore\n");
2029    }
2030    if !r.rules_created.is_empty() {
2031        out.push_str(&format!("  rules: {}\n", r.rules_created.join(", ")));
2032    }
2033    out
2034}
2035
2036#[cfg(test)]
2037mod tests {
2038    use super::*;
2039
2040    #[test]
2041    fn exclude_matches_prefix_extension_and_substring() {
2042        let pats = vec![
2043            "target/".to_string(),
2044            "*.lock".into(),
2045            "node_modules".into(),
2046        ];
2047        assert!(excluded("target/debug/foo.rs", &pats));
2048        assert!(
2049            !excluded("targeted/foo.rs", &pats),
2050            "prefix needs the slash"
2051        );
2052        assert!(excluded("Cargo.lock", &pats));
2053        assert!(!excluded("Cargo.toml", &pats));
2054        assert!(excluded("ui/node_modules/x/y.js", &pats));
2055        assert!(!excluded("src/lib.rs", &pats));
2056        assert!(!excluded("anything", &[]));
2057    }
2058
2059    /// A `*.` pattern is a file-name suffix, so a compound one works. Reading
2060    /// only the last dot segment would make `*.min.js` — a default — inert,
2061    /// and generated bundles are both the largest files in a tree and the ones
2062    /// least worth parsing.
2063    #[test]
2064    fn a_compound_suffix_pattern_matches() {
2065        let defaults: Vec<String> = DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect();
2066        // Not under `dist/`, so only the suffix rule can match it.
2067        assert!(excluded("ui/build/bundle.min.js", &defaults));
2068        assert!(excluded("bundle.min.js", &defaults));
2069        assert!(
2070            !excluded("ui/src/app.js", &defaults),
2071            "an ordinary source file is not a bundle"
2072        );
2073        assert!(!excluded("ui/src/minify.js", &defaults));
2074        // The single-extension form is unchanged, and a bare suffix is not a
2075        // match: `*.lock` means something *dot* lock.
2076        assert!(excluded("Cargo.lock", &defaults));
2077        assert!(!excluded(".lock", &defaults));
2078        assert!(!excluded("src/lib.rs", &defaults));
2079    }
2080
2081    #[test]
2082    fn file_props_split_dir_and_ext() {
2083        let mut st = FileState::default();
2084        st.touch("sha1", "a@x.test", 200);
2085        let m: BTreeMap<_, _> = file_props(&st, "src/a/b.rs").into_iter().collect();
2086        assert_eq!(m["dir"], Value::Str("src/a".into()));
2087        assert_eq!(m["ext"], Value::Str("rs".into()));
2088        assert_eq!(m["n_commits"], Value::Int(1));
2089        assert_eq!(m["id"], Value::Str("src/a/b.rs".into()));
2090        assert_eq!(m["top_author_id"], Value::Str("a@x.test".into()));
2091        assert_eq!(
2092            m["author_counts"],
2093            Value::List(vec![Value::Str("a@x.test\t1".into())])
2094        );
2095        let m: BTreeMap<_, _> = file_props(&FileState::default(), "README")
2096            .into_iter()
2097            .collect();
2098        assert_eq!(m["dir"], Value::Str(String::new()));
2099        assert_eq!(m["ext"], Value::Str(String::new()));
2100    }
2101
2102    /// The prop is the whole point of the incremental fix: it must survive a
2103    /// round trip so the next run resumes the real distribution, not the
2104    /// incumbent's total.
2105    #[test]
2106    fn author_counts_round_trip_preserves_the_distribution() {
2107        let mut st = FileState::default();
2108        for _ in 0..3 {
2109            st.touch("s", "alice@x.test", 200);
2110        }
2111        for _ in 0..4 {
2112            st.touch("s", "bob@x.test", 200);
2113        }
2114        let Value::List(encoded) = st.author_counts_value() else {
2115            panic!("author_counts must be a list");
2116        };
2117        assert_eq!(
2118            encoded,
2119            vec![
2120                Value::Str("alice@x.test\t3".into()),
2121                Value::Str("bob@x.test\t4".into()),
2122            ],
2123            "email order, so the prop is stable across runs"
2124        );
2125        let mut reloaded = FileState::default();
2126        reloaded.set_author_counts(&encoded);
2127        assert_eq!(reloaded.by_author, st.by_author);
2128        assert_eq!(reloaded.top_author(), "bob@x.test");
2129    }
2130
2131    /// Malformed entries are skipped, not fatal: the run degrades to the counts
2132    /// it can read rather than refusing to sync.
2133    #[test]
2134    fn author_counts_skips_entries_it_cannot_parse() {
2135        let mut st = FileState::default();
2136        st.set_author_counts(&[
2137            Value::Str("alice@x.test\t2".into()),
2138            Value::Str("no-tab-here".into()),
2139            Value::Str("bob@x.test\tnotanumber".into()),
2140            Value::Str("\t5".into()),
2141            Value::Int(7),
2142        ]);
2143        assert_eq!(st.by_author, BTreeMap::from([("alice@x.test".into(), 2)]));
2144    }
2145
2146    #[test]
2147    fn commits_list_is_capped_and_top_author_is_deterministic() {
2148        let mut st = FileState::default();
2149        for i in 0..5 {
2150            st.touch(&format!("sha{i}"), "b@x.test", 3);
2151        }
2152        st.touch("shaX", "a@x.test", 3);
2153        assert_eq!(st.commits, vec!["sha3", "sha4", "shaX"]);
2154        assert_eq!(st.n_commits, 6);
2155        assert_eq!(st.top_author(), "b@x.test");
2156
2157        // Past the cap the stored `n_commits` is the true total, not the length
2158        // of the truncated `commits` list, and `author_counts` sums to the same
2159        // number. A file over `--max-commits-per-file` would otherwise report a
2160        // history frozen at the cap.
2161        let m: BTreeMap<_, _> = file_props(&st, "src/hot.rs").into_iter().collect();
2162        assert_eq!(m["n_commits"], Value::Int(6));
2163        let Value::List(commits) = &m["commits"] else {
2164            panic!("commits must be a list");
2165        };
2166        assert_eq!(commits.len(), 3, "the list is still capped at 3");
2167        assert!(
2168            matches!(m["n_commits"], Value::Int(n) if n as usize > commits.len()),
2169            "n_commits must exceed the capped list once the cap is passed"
2170        );
2171        assert_eq!(
2172            m["author_counts"],
2173            Value::List(vec![
2174                Value::Str("a@x.test\t1".into()),
2175                Value::Str("b@x.test\t5".into()),
2176            ]),
2177            "the per-author counts sum to n_commits, not to the capped list"
2178        );
2179
2180        let mut tie = FileState::default();
2181        tie.touch("s", "b@x.test", 10);
2182        tie.touch("s", "a@x.test", 10);
2183        assert_eq!(tie.top_author(), "a@x.test", "ties break on smallest email");
2184    }
2185}