Skip to main content

cli/
ingest_git.rs

1//! `mushroomdb ingest-git <db> <repo>`: build and maintain a graph of a git
2//! repository. First run ingests the whole history; later runs apply only the
3//! commits after the recorded `GitSync` head, so deletes and renames retract
4//! or move derived edges instead of leaving them stale.
5//!
6//! Graph shape:
7//!
8//! | Label | key (`id`) | props |
9//! |---|---|---|
10//! | `Author` | mailmap-resolved email | `name`, `name_counts` |
11//! | `Commit` | full sha | `message`, `ts`, `author_id`, `pr_id` (with `--prs`) |
12//! | `File` | path | `path`, `dir`, `ext`, `commits`, `n_commits`, `top_author_id`, `author_counts`, plus the working-tree props in [`structure`](crate::structure) |
13//! | `Symbol` | `"<path>#<qualified name>"` | see [`structure`](crate::structure) |
14//! | `PR` | `"pr:<number>"` | `number`, `title`, `url`, `merged_at`, `author_login` |
15//! | `GitSync` | `"__mushroomdb_git_sync__"` | `sha`, `synced_at`, `repo`, `recurse`, `prs`, `structure`, `docs` |
16//!
17//! Edges: user `TOUCHED` Commit→File and `MERGED_AS` PR→Commit, auto-FK
18//! `AUTHOR` Commit→Author, `TOP_AUTHOR` File→Author and `PR` Commit→PR,
19//! rule-derived `CO_CHANGED` File→File and `KNOWS` Author→File — and, from the
20//! working-tree pass, `DEFINES`, `IMPORTS`, `CALLS` and `MENTIONS`.
21//!
22//! With `--recurse-submodules` each initialised submodule is walked as its own
23//! *unit*: its file keys carry the submodule's path in the parent, and it
24//! resumes from its own `GitSync` marker.
25use crate::structure;
26use crate::CliError;
27use core_api::{
28    default_max_edges, Direction, GraphError, IngestOptions, Predicate, ResultSet, RuleDef,
29    SharedDb, Value, WriteGuard, WRITE_LOCK_WAIT,
30};
31use std::collections::{BTreeMap, BTreeSet};
32use std::path::{Path, PathBuf};
33use std::process::Command;
34
35/// Cap on the `commits` list stored per file. Bounds both node size and the
36/// cost of the jaccard overlap the `co_changed` rule runs over that list.
37pub const DEFAULT_MAX_COMMITS_PER_FILE: usize = 200;
38
39/// Applied when the user names no `--exclude` pattern of their own.
40///
41/// Defined in core-api because the `impact` MCP tool builds its default file
42/// list straight off a working tree and has to leave out exactly what this
43/// ingest leaves out; two lists would let the tool ask about files no store was
44/// ever going to hold.
45pub use core_api::repograph::DEFAULT_EXCLUDES;
46
47/// Minimum jaccard overlap of two files' `commits` lists for `CO_CHANGED`.
48const CO_CHANGE_MIN: f64 = 0.25;
49
50/// WAL bytes past the last snapshot that make writing a new one worthwhile.
51///
52/// Every open reads `wal.bin` whole and replays it frame by frame, so the tail
53/// is paid again on every hook, every MCP start and every CLI call — while a
54/// snapshot is paid once by the run that writes it. Measured on this
55/// repository, an 8.1 MB tail costs 156 ms per open against ~440 ms to
56/// snapshot it away, so a tail this size pays its snapshot back within three
57/// opens. Below the threshold the replay is cheap enough that snapshotting on
58/// every incremental run would cost more than it saves.
59pub const SNAPSHOT_WAL_BYTES: u64 = 4 * 1024 * 1024;
60
61/// Key of the singleton `GitSync` node holding the last ingested sha.
62///
63/// Node keys are a single namespace shared with `File` keys, which are repo
64/// paths — so this cannot be `"HEAD"`. A repository with a file named `HEAD`
65/// (git's own `.git/HEAD` aside, plenty of projects ship one) would otherwise
66/// have the sha written onto its `File` node, leaving no sync marker and
67/// forcing a full re-ingest on every run.
68pub(crate) const SYNC_KEY: &str = "__mushroomdb_git_sync__";
69
70/// What the CLI prints, and exits 3 on, when another process holds the store's
71/// cross-process write lock. Nothing was written, so retrying is always safe.
72pub const BUSY_MESSAGE: &str = "another mushroomdb process is writing; retry";
73
74/// Name of the `Commit.pr_id` foreign-key rule. Identical to the name the
75/// zero-config FK inference would choose, so the two never both create it.
76const PR_FK_RULE: &str = "auto_fk_commit_pr_id";
77
78#[derive(Debug, Clone, PartialEq, Eq)]
79pub struct IngestGitOpts {
80    pub repo: PathBuf,
81    /// Paths to skip. Pattern ending in `/` = path prefix, pattern starting
82    /// with `*.` = extension, otherwise = substring of the path.
83    pub exclude: Vec<String>,
84    pub max_commits_per_file: usize,
85    /// Walk every initialised submodule as its own unit.
86    pub recurse_submodules: bool,
87    /// Ask `gh` for merged pull requests and link them to their commits.
88    pub prs: bool,
89    /// Read the working tree: content hashes, `Symbol` nodes, imports and
90    /// calls. Off with `--no-structure`. Recorded on the `GitSync` node.
91    pub structure: bool,
92    /// Index Markdown bodies, headings and mentions. Off with `--no-docs`, and
93    /// inert without `structure`. Recorded on the `GitSync` node.
94    pub docs: bool,
95    /// Add the database directory to the repository's `.gitignore`.
96    pub ensure_gitignore: bool,
97}
98
99#[derive(Debug, Default, Clone, PartialEq, Eq)]
100pub struct IngestGitReport {
101    pub commits: usize,
102    pub files: usize,
103    pub authors: usize,
104    pub renamed: usize,
105    pub deleted: usize,
106    pub incremental: bool,
107    pub rules_created: Vec<String>,
108    /// Submodules walked as their own units.
109    pub submodules: usize,
110    /// Pull requests inserted by this run.
111    pub prs: usize,
112    /// Whether this run appended the database directory to `.gitignore`.
113    pub gitignore_added: bool,
114    /// What the working-tree pass saw. All zeros with `--no-structure`.
115    pub structure: crate::structure::StructureReport,
116}
117
118/// Paths the commit walk left for the working-tree pass to look at.
119#[derive(Default)]
120struct StructureWork {
121    /// Files added, modified, or renamed *into* this window's paths.
122    touched: BTreeSet<String>,
123    /// Keys that no longer name a file: renamed away, or deleted. Any file
124    /// whose `imports` or `mentions` list still holds one of these has to be
125    /// extracted again, or the edge it derived stays behind.
126    stale: BTreeSet<String>,
127}
128
129/// One git working tree walked by a run: the repository itself, or one of its
130/// submodules.
131///
132/// A submodule's paths are keys under `prefix` (its path in the parent), and it
133/// carries its own sync marker, so the two histories advance independently.
134#[derive(Debug, Clone)]
135struct RepoUnit {
136    path: PathBuf,
137    /// `""` for the repository itself, `"<displaypath>/"` for a submodule.
138    prefix: String,
139    sync_key: String,
140}
141
142#[derive(Debug)]
143enum Change {
144    Added(String),
145    Modified(String),
146    Deleted(String),
147    Renamed { from: String, to: String },
148}
149
150#[derive(Debug)]
151struct GitCommit {
152    sha: String,
153    author_name: String,
154    author_email: String,
155    ts: i64,
156    subject: String,
157    changes: Vec<Change>,
158}
159
160/// Simple, dependency-free path matcher. Documented in `docs/site/ingest-git.md`.
161///
162/// The matcher itself is `core_api::repograph::path_excluded`, next to
163/// [`DEFAULT_EXCLUDES`], because the `impact` MCP tool filters a working tree
164/// with the same patterns and must read them the same way.
165fn excluded(path: &str, patterns: &[String]) -> bool {
166    core_api::repograph::path_excluded(path, patterns)
167}
168
169fn git_output(repo: &Path, args: &[&str]) -> Result<std::process::Output, CliError> {
170    Command::new("git")
171        .arg("-C")
172        .arg(repo)
173        .args(args)
174        .output()
175        .map_err(|e| CliError(format!("cannot run git in {}: {e}", repo.display())))
176}
177
178/// `git log --reverse --name-status -M --format=<RS>%H<US>%aN<US>%aE<US>%at<US>%s <range>`
179///
180/// Returns oldest commit first. The walk ends at `head` — never at the symbolic
181/// `HEAD` — so the range is pinned to the same sha the caller will record as the
182/// sync marker. See [`head_sha`] for why that matters.
183///
184/// `%aN` and `%aE` are the mailmap-resolved name and address, so a repository
185/// with a `.mailmap` reports one identity for a contributor who has committed
186/// under several addresses. Without one they are exactly `%an`/`%ae`.
187fn read_log(repo: &Path, since: Option<&str>, head: &str) -> Result<Vec<GitCommit>, CliError> {
188    if let Some(s) = since {
189        let spec = format!("{s}^{{commit}}");
190        if !git_output(repo, &["cat-file", "-e", &spec])?
191            .status
192            .success()
193        {
194            return Err(CliError(format!(
195                "recorded sync head {s} is not in {} (history rewritten?); \
196                 ingest into a fresh database directory",
197                repo.display()
198            )));
199        }
200    }
201    let mut cmd = Command::new("git");
202    cmd.arg("-C").arg(repo).args([
203        // Without this git renders any non-ASCII byte in a path as an octal
204        // escape, so `src/café.rs` would be stored under a mangled key that no
205        // later run matches. Paths containing a tab or newline stay quoted and
206        // escaped either way — git has to, or they would break the format below.
207        "-c",
208        "core.quotePath=false",
209        "log",
210        "--reverse",
211        "--name-status",
212        "-M",
213        "--no-color",
214        // Record separator \x1e between commits, unit separator \x1f between
215        // header fields. A commit subject containing either byte splits its own
216        // record: the message is truncated at the first \x1f, and a \x1e drops
217        // the remainder of that commit's header. Accepted — the parse degrades
218        // to a skipped or shortened message, never a panic or a wrong sha.
219        "--format=%x1e%H%x1f%aN%x1f%aE%x1f%at%x1f%s",
220    ]);
221    match since {
222        Some(s) => cmd.arg(format!("{s}..{head}")),
223        None => cmd.arg(head),
224    };
225    let out = cmd
226        .output()
227        .map_err(|e| CliError(format!("cannot run git: {e}")))?;
228    if !out.status.success() {
229        return Err(CliError(format!(
230            "git log failed: {}",
231            String::from_utf8_lossy(&out.stderr).trim()
232        )));
233    }
234    let text = String::from_utf8_lossy(&out.stdout);
235    let mut commits = Vec::new();
236    for block in text.split('\x1e').filter(|b| !b.trim().is_empty()) {
237        let mut lines = block.lines();
238        let header = lines.next().unwrap_or("");
239        let f: Vec<&str> = header.split('\x1f').collect();
240        if f.len() < 5 {
241            continue;
242        }
243        let mut changes = Vec::new();
244        for l in lines {
245            let cols: Vec<&str> = l.split('\t').collect();
246            match cols.as_slice() {
247                [s, p] if s.starts_with('A') => changes.push(Change::Added((*p).to_string())),
248                [s, p] if s.starts_with('M') || s.starts_with('T') => {
249                    changes.push(Change::Modified((*p).to_string()))
250                }
251                [s, p] if s.starts_with('D') => changes.push(Change::Deleted((*p).to_string())),
252                [s, from, to] if s.starts_with('R') => changes.push(Change::Renamed {
253                    from: (*from).to_string(),
254                    to: (*to).to_string(),
255                }),
256                // A copy is a brand-new path with no prior history here.
257                [s, _from, to] if s.starts_with('C') => {
258                    changes.push(Change::Added((*to).to_string()))
259                }
260                _ => {}
261            }
262        }
263        commits.push(GitCommit {
264            sha: f[0].into(),
265            author_name: f[1].into(),
266            author_email: f[2].into(),
267            ts: f[3].parse().unwrap_or(0),
268            subject: f[4].into(),
269            changes,
270        });
271    }
272    Ok(commits)
273}
274
275/// Resolve `HEAD` to a concrete sha. `Ok(None)` means the repository has no
276/// commits yet; a path that is not a repository at all is an error.
277///
278/// Called **before** [`read_log`], and the resulting sha is both the end of the
279/// walk and the recorded sync marker. Resolving it afterwards instead would open
280/// a window: a commit landing between the walk and the `rev-parse` would push
281/// the marker past a commit that was never ingested, and every later run would
282/// skip it silently. Pinning both to one sha closes that window — a commit that
283/// lands mid-run is simply outside this range and gets picked up next time.
284///
285/// Asking git also beats taking `log.last()`. The two agree in practice, since a
286/// reachability walk emits its tip first and reversing puts it last, but that is
287/// a property of the traversal and of this module's parser rather than something
288/// the resume marker should rest on.
289fn head_sha(repo: &Path) -> Result<Option<String>, CliError> {
290    let out = git_output(repo, &["rev-parse", "--verify", "-q", "HEAD^{commit}"])?;
291    if !out.status.success() {
292        // No commits yet, or not a repository at all — only the latter is an error.
293        if !git_output(repo, &["rev-parse", "--git-dir"])?
294            .status
295            .success()
296        {
297            return Err(CliError(format!(
298                "not a git repository: {}",
299                repo.display()
300            )));
301        }
302        return Ok(None);
303    }
304    let sha = String::from_utf8_lossy(&out.stdout).trim().to_string();
305    if sha.is_empty() {
306        return Err(CliError(format!(
307            "git could not resolve HEAD in {}",
308            repo.display()
309        )));
310    }
311    Ok(Some(sha))
312}
313
314/// Absolute, symlink-free form of `p`, falling back to `p` when it cannot be
315/// resolved (a path that does not exist yet, or a permission error). The result
316/// is what `GitSync.repo` records, so a later run can find the repository again
317/// from any working directory.
318fn canonical(p: &Path) -> PathBuf {
319    std::fs::canonicalize(p).unwrap_or_else(|_| p.to_path_buf())
320}
321
322/// The submodule paths recorded in a unit's `.gitmodules`, relative to that
323/// unit.
324///
325/// These are the paths git reports as ordinary changes in `--name-status`
326/// output while being gitlinks — a commit pointer, not a file. They get no
327/// `File` node whether or not the submodule is walked, and the list is read
328/// from configuration so it is the same for an uninitialised submodule.
329fn gitlink_paths(unit: &Path) -> BTreeSet<String> {
330    let out = Command::new("git")
331        .arg("config")
332        .arg("--file")
333        .arg(unit.join(".gitmodules"))
334        .args(["--get-regexp", r"^submodule\..*\.path$"])
335        .output();
336    let Ok(out) = out else { return BTreeSet::new() };
337    if !out.status.success() {
338        return BTreeSet::new(); // no .gitmodules, so no submodules
339    }
340    String::from_utf8_lossy(&out.stdout)
341        .lines()
342        .filter_map(|l| l.split_once(' '))
343        .map(|(_, path)| path.trim().to_string())
344        .filter(|p| !p.is_empty())
345        .collect()
346}
347
348/// The repository itself, then one unit per initialised submodule when
349/// `recurse` is set.
350///
351/// `git submodule foreach` visits only initialised submodules and says nothing
352/// about the rest, which is the behaviour wanted here: a submodule that was
353/// never checked out has no working tree to walk. `$displaypath` is relative to
354/// the top-level repository, so it is exactly the key prefix its files need.
355fn repo_units(repo: &Path, recurse: bool) -> Result<Vec<RepoUnit>, CliError> {
356    let mut units = vec![RepoUnit {
357        path: repo.to_path_buf(),
358        prefix: String::new(),
359        sync_key: SYNC_KEY.to_string(),
360    }];
361    if !recurse {
362        return Ok(units);
363    }
364    let out = git_output(
365        repo,
366        &[
367            "submodule",
368            "foreach",
369            "--quiet",
370            "--recursive",
371            "echo \"$displaypath\"",
372        ],
373    )?;
374    if !out.status.success() {
375        return Ok(units);
376    }
377    let mut paths: Vec<String> = String::from_utf8_lossy(&out.stdout)
378        .lines()
379        .map(|l| l.trim_end_matches('/').trim().to_string())
380        .filter(|l| !l.is_empty())
381        .collect();
382    paths.sort();
383    paths.dedup();
384    for dp in paths {
385        let path = repo.join(&dp);
386        // `foreach` already skips the uninitialised, but a stale entry or a
387        // removed checkout would otherwise fail the whole run.
388        if !git_output(&path, &["rev-parse", "--git-dir"])?
389            .status
390            .success()
391        {
392            continue;
393        }
394        units.push(RepoUnit {
395            prefix: format!("{dp}/"),
396            sync_key: format!("{SYNC_KEY}:{dp}"),
397            path,
398        });
399    }
400    Ok(units)
401}
402
403/// Rewrite one unit's changes into repository-wide keys: drop gitlinks, then
404/// prepend the unit's prefix so a submodule's `src/lib.rs` is stored under
405/// `vendor/lib/src/lib.rs`.
406///
407/// Done before anything else looks at the log, so exclusion patterns, the
408/// stored keys and the `File` state query all speak the same path.
409fn localise(log: &mut [GitCommit], prefix: &str, gitlinks: &BTreeSet<String>) {
410    let keep = |p: &String| !gitlinks.contains(p.as_str());
411    for c in log.iter_mut() {
412        c.changes.retain(|ch| match ch {
413            Change::Added(p) | Change::Modified(p) | Change::Deleted(p) => keep(p),
414            Change::Renamed { from, to } => keep(from) && keep(to),
415        });
416        if prefix.is_empty() {
417            continue;
418        }
419        for ch in c.changes.iter_mut() {
420            match ch {
421                Change::Added(p) | Change::Modified(p) | Change::Deleted(p) => {
422                    *p = format!("{prefix}{p}")
423                }
424                Change::Renamed { from, to } => {
425                    *from = format!("{prefix}{from}");
426                    *to = format!("{prefix}{to}");
427                }
428            }
429        }
430    }
431}
432
433/// Append `<db dir>/` to the repository's `.gitignore` unless it is already
434/// listed, creating the file if it does not exist. Returns whether it was
435/// written.
436///
437/// A database kept outside the repository is left alone: the repository has no
438/// business ignoring a path it does not contain.
439fn ensure_gitignore(repo: &Path, db_dir: &Path) -> Result<bool, CliError> {
440    let Ok(rel) = canonical(db_dir)
441        .strip_prefix(repo)
442        .map(|p| p.to_path_buf())
443    else {
444        return Ok(false);
445    };
446    if rel.as_os_str().is_empty() {
447        return Ok(false);
448    }
449    let line = format!("{}/", rel.to_string_lossy().replace('\\', "/"));
450    let path = repo.join(".gitignore");
451    let current = match std::fs::read_to_string(&path) {
452        Ok(s) => s,
453        Err(e) if e.kind() == std::io::ErrorKind::NotFound => String::new(),
454        Err(e) => return Err(CliError(format!("cannot read {}: {e}", path.display()))),
455    };
456    let bare = line.trim_end_matches('/');
457    if current
458        .lines()
459        .map(|l| l.trim())
460        .any(|l| l == line || l == bare || l == format!("/{line}") || l == format!("/{bare}"))
461    {
462        return Ok(false);
463    }
464    let mut next = current;
465    if !next.is_empty() && !next.ends_with('\n') {
466        next.push('\n');
467    }
468    next.push_str(&line);
469    next.push('\n');
470    std::fs::write(&path, next)
471        .map_err(|e| CliError(format!("cannot write {}: {e}", path.display())))?;
472    Ok(true)
473}
474
475/// Separator inside an `author_counts` or `name_counts` entry. Neither an email
476/// address nor a git author name can contain a tab, so `<text>\tcount`
477/// round-trips unambiguously.
478const AUTHOR_COUNT_SEP: char = '\t';
479
480/// How many commits carried each spelling of one email's `%aN`, in the order
481/// the spellings were first seen.
482///
483/// One person routinely commits under more than one name — `Ada Lovelace` at
484/// work and `Ada M. Lovelace` from a laptop that was configured once and never
485/// again. Merging them on email is right, and every tool that prints an author
486/// then has to choose *which* of the names to show. Showing the first one
487/// walked means a display name decided by the oldest commit in the window,
488/// which on this repository labelled an identity with a name carried by 21% of
489/// its commits.
490#[derive(Default, Clone, Debug, PartialEq, Eq)]
491struct AuthorState {
492    /// `(name, commits)` in first-seen order, which is what breaks a tie.
493    names: Vec<(String, usize)>,
494}
495
496impl AuthorState {
497    /// Credit one more commit to `name`.
498    fn touch(&mut self, name: &str) {
499        self.add(name, 1);
500    }
501
502    /// Credit `n` more commits to `name`, appending it if it is new. Order is
503    /// preserved, so the entry that was seen first stays first.
504    fn add(&mut self, name: &str, n: usize) {
505        match self.names.iter_mut().find(|(held, _)| held == name) {
506            Some(entry) => entry.1 += n,
507            None => self.names.push((name.to_string(), n)),
508        }
509    }
510
511    /// Fold another window's counts into this one, keeping first-seen order.
512    fn absorb(&mut self, other: &AuthorState) {
513        for (name, n) in &other.names {
514            self.add(name, *n);
515        }
516    }
517
518    /// The name on the most commits. A tie goes to the name seen first, which
519    /// is why the strict `>` matters: it never unseats an equal incumbent.
520    fn display_name(&self) -> String {
521        let mut best: Option<&(String, usize)> = None;
522        for entry in &self.names {
523            if best.is_none_or(|(_, n)| entry.1 > *n) {
524                best = Some(entry);
525            }
526        }
527        best.map(|(name, _)| name.clone()).unwrap_or_default()
528    }
529
530    /// The distribution as an `Author.name_counts` prop: `"name\tcount"`
531    /// strings in first-seen order.
532    ///
533    /// Stored for the same reason `File.author_counts` is: an incremental run
534    /// sees only the new window, and the `name` prop alone cannot say how far
535    /// ahead the incumbent spelling is. Without it every sync would relabel the
536    /// identity from a handful of commits.
537    fn name_counts_value(&self) -> Value {
538        Value::List(
539            self.names
540                .iter()
541                .map(|(name, n)| Value::Str(format!("{name}{AUTHOR_COUNT_SEP}{n}")))
542                .collect(),
543        )
544    }
545
546    /// Inverse of [`AuthorState::name_counts_value`]. Entries that are not
547    /// `name<TAB>count` are skipped rather than failing the run.
548    fn set_name_counts(&mut self, list: &[Value]) {
549        for v in list {
550            let Value::Str(s) = v else { continue };
551            let Some((name, n)) = s.rsplit_once(AUTHOR_COUNT_SEP) else {
552                continue;
553            };
554            let Ok(n) = n.parse::<usize>() else { continue };
555            if !name.is_empty() {
556                self.add(name, n);
557            }
558        }
559    }
560}
561
562fn file_props(st: &FileState, path: &str) -> Vec<(String, Value)> {
563    let commits = &st.commits;
564    let dir = path
565        .rsplit_once('/')
566        .map(|(d, _)| d)
567        .unwrap_or("")
568        .to_string();
569    let ext = path
570        .rsplit_once('.')
571        .map(|(_, e)| e)
572        .unwrap_or("")
573        .to_string();
574    vec![
575        ("id".into(), Value::Str(path.into())),
576        ("path".into(), Value::Str(path.into())),
577        ("dir".into(), Value::Str(dir)),
578        ("ext".into(), Value::Str(ext)),
579        (
580            "commits".into(),
581            Value::List(commits.iter().map(|s| Value::Str(s.clone())).collect()),
582        ),
583        // The true total, which past `--max-commits-per-file` is larger than
584        // the `commits` list it is stored beside. It is also what
585        // `author_counts` sums to, so the two props agree at any history
586        // length.
587        ("n_commits".into(), Value::Int(st.n_commits as i64)),
588        ("top_author_id".into(), Value::Str(st.top_author())),
589        // Written on every touch so the next incremental run can rebuild the
590        // distribution instead of crediting the whole history to the incumbent.
591        ("author_counts".into(), st.author_counts_value()),
592    ]
593}
594
595/// In-memory per-file state accumulated while walking commits.
596#[derive(Default, Clone)]
597struct FileState {
598    commits: Vec<String>,
599    by_author: BTreeMap<String, usize>,
600    n_commits: usize,
601}
602
603impl FileState {
604    fn touch(&mut self, sha: &str, author: &str, cap: usize) {
605        self.commits.push(sha.to_string());
606        if cap > 0 && self.commits.len() > cap {
607            self.commits.remove(0);
608        }
609        self.n_commits += 1;
610        *self.by_author.entry(author.to_string()).or_default() += 1;
611    }
612
613    /// Most commits wins; ties break on the lexicographically smallest email
614    /// so the result is deterministic across runs.
615    fn top_author(&self) -> String {
616        self.by_author
617            .iter()
618            .max_by(|a, b| a.1.cmp(b.1).then(b.0.cmp(a.0)))
619            .map(|(a, _)| a.clone())
620            .unwrap_or_default()
621    }
622
623    /// The per-author distribution as a `File.author_counts` prop: a list of
624    /// `"email\tcount"` strings in email order.
625    ///
626    /// This is the state an incremental run needs and cannot recompute — the
627    /// walk only sees the new window, and `top_author_id` alone cannot say how
628    /// far ahead the incumbent is. Without it a challenger's commits reset on
629    /// every sync and ownership can never change.
630    fn author_counts_value(&self) -> Value {
631        Value::List(
632            self.by_author
633                .iter()
634                .map(|(email, n)| Value::Str(format!("{email}{AUTHOR_COUNT_SEP}{n}")))
635                .collect(),
636        )
637    }
638
639    /// Inverse of [`FileState::author_counts_value`]. Entries that are not
640    /// `email<TAB>count` are skipped rather than failing the run.
641    fn set_author_counts(&mut self, list: &[Value]) {
642        for v in list {
643            let Value::Str(s) = v else { continue };
644            let Some((email, n)) = s.rsplit_once(AUTHOR_COUNT_SEP) else {
645                continue;
646            };
647            let Ok(n) = n.parse::<usize>() else { continue };
648            if !email.is_empty() {
649                *self.by_author.entry(email.to_string()).or_default() += n;
650            }
651        }
652    }
653}
654
655/// One merged pull request as reported by `gh`.
656#[derive(Debug, Clone, PartialEq, Eq)]
657struct PullRequest {
658    number: i64,
659    title: String,
660    url: String,
661    merged_at: String,
662    author_login: String,
663    /// `mergeCommit.oid`, absent for a pull request merged some other way (or
664    /// whose merge commit has since been rewritten).
665    merge_sha: Option<String>,
666}
667
668fn pr_key(number: i64) -> String {
669    format!("pr:{number}")
670}
671
672/// The pull request number in a squash-merge subject: `\(#(\d+)\)$`.
673///
674/// Hand-rolled rather than pulled in as a dependency — the pattern is anchored
675/// at the end of the subject and made of two literals around a run of digits.
676fn subject_pr(subject: &str) -> Option<i64> {
677    let rest = subject.strip_suffix(')')?;
678    let at = rest.rfind("(#")?;
679    let digits = &rest[at + 2..];
680    if digits.is_empty() || !digits.bytes().all(|b| b.is_ascii_digit()) {
681        return None;
682    }
683    digits.parse().ok()
684}
685
686/// `gh pr list --state merged` in `repo`, or an empty list with one warning.
687///
688/// Every failure is a skip: `gh` may not be installed, the repository may have
689/// no GitHub remote, and the user may not be authenticated. None of that is a
690/// reason to fail an ingest that is otherwise complete.
691fn fetch_prs(repo: &Path) -> Vec<PullRequest> {
692    let out = Command::new("gh")
693        .current_dir(repo)
694        .args([
695            "pr",
696            "list",
697            "--state",
698            "merged",
699            "--limit",
700            "1000",
701            "--json",
702            "number,title,url,mergedAt,mergeCommit,author",
703        ])
704        .output();
705    let out = match out {
706        Ok(o) => o,
707        Err(_) => {
708            eprintln!("ingest-git: --prs skipped: gh is not on PATH");
709            return Vec::new();
710        }
711    };
712    if !out.status.success() {
713        let detail = String::from_utf8_lossy(&out.stderr)
714            .lines()
715            .find(|l| !l.trim().is_empty())
716            .unwrap_or("no detail")
717            .to_string();
718        eprintln!("ingest-git: --prs skipped: gh pr list failed: {detail}");
719        return Vec::new();
720    }
721    let parsed: serde_json::Value = match serde_json::from_slice(&out.stdout) {
722        Ok(v) => v,
723        Err(e) => {
724            eprintln!("ingest-git: --prs skipped: gh pr list output is not JSON: {e}");
725            return Vec::new();
726        }
727    };
728    let Some(items) = parsed.as_array() else {
729        eprintln!("ingest-git: --prs skipped: gh pr list did not return a list");
730        return Vec::new();
731    };
732    let mut prs: Vec<PullRequest> = items
733        .iter()
734        .filter_map(|v| {
735            let number = v.get("number")?.as_i64()?;
736            let str_at = |k: &str| {
737                v.get(k)
738                    .and_then(|x| x.as_str())
739                    .unwrap_or_default()
740                    .to_string()
741            };
742            Some(PullRequest {
743                number,
744                title: str_at("title"),
745                url: str_at("url"),
746                merged_at: str_at("mergedAt"),
747                author_login: v
748                    .get("author")
749                    .and_then(|a| a.get("login"))
750                    .and_then(|l| l.as_str())
751                    .unwrap_or_default()
752                    .to_string(),
753                merge_sha: v
754                    .get("mergeCommit")
755                    .and_then(|m| m.get("oid"))
756                    .and_then(|o| o.as_str())
757                    .filter(|s| !s.is_empty())
758                    .map(str::to_string),
759            })
760        })
761        .collect();
762    // Ascending by number: the insert order, and so the node order, is the
763    // same on every run whatever order gh listed them in.
764    prs.sort_by_key(|p| p.number);
765    prs.dedup_by_key(|p| p.number);
766    prs
767}
768
769/// Insert the `PR` nodes that are new, declare the `Commit.pr_id` foreign key,
770/// and index titles for search. Runs before the commits so the FK resolves.
771fn ingest_prs(
772    w: &mut WriteGuard<'_>,
773    prs: &[PullRequest],
774    ingest: &IngestOptions,
775    report: &mut IngestGitReport,
776) -> Result<(), CliError> {
777    let rows: Vec<BTreeMap<String, Value>> = prs
778        .iter()
779        .filter(|p| !w.has_node(&pr_key(p.number)))
780        .map(|p| {
781            BTreeMap::from([
782                ("id".to_string(), Value::Str(pr_key(p.number))),
783                ("number".to_string(), Value::Int(p.number)),
784                ("title".to_string(), Value::Str(p.title.clone())),
785                ("url".to_string(), Value::Str(p.url.clone())),
786                ("merged_at".to_string(), Value::Str(p.merged_at.clone())),
787                (
788                    "author_login".to_string(),
789                    Value::Str(p.author_login.clone()),
790                ),
791            ])
792        })
793        .collect();
794    if !rows.is_empty() {
795        let r = w.ingest_with_edges("PR", rows, ingest, &[])?;
796        report.prs = r.inserted;
797        report.rules_created.extend(r.rules_created);
798    }
799    // Declared here rather than left to FK inference, which only fires on a
800    // batch of `Commit` rows that already carry `pr_id` — a run that links a
801    // pull request to a commit ingested earlier would otherwise leave the
802    // property with no edge behind it.
803    if !w.rules().iter().any(|r| r.name == PR_FK_RULE) {
804        let predicate = Predicate::KeyMatch {
805            field: "pr_id".into(),
806        };
807        let max_edges = Some(default_max_edges(&predicate));
808        w.create_rule(RuleDef {
809            name: PR_FK_RULE.into(),
810            src_label: "Commit".into(),
811            dst_label: "PR".into(),
812            predicate,
813            edge_type: "PR".into(),
814            weight_prop: None,
815            max_edges,
816            approximate: false,
817            via_label: None,
818            via_edge: None,
819            via_dir: None,
820        })?;
821        report.rules_created.push(PR_FK_RULE.to_string());
822    }
823    if !w
824        .fulltext_pairs()
825        .contains(&("PR".to_string(), "title".to_string()))
826    {
827        w.enable_fulltext("PR", "title")?;
828    }
829    Ok(())
830}
831
832/// Point every commit that carries a pull request at it: the merge commit by
833/// sha, a squash merge by its `(#N)` subject. Runs after the commits are in the
834/// graph, so a pull request merged before the last sync is linked too.
835fn link_prs(
836    w: &mut WriteGuard<'_>,
837    prs: &[PullRequest],
838    ingest: &IngestOptions,
839) -> Result<(), CliError> {
840    let by_sha: BTreeMap<&str, i64> = prs
841        .iter()
842        .filter_map(|p| p.merge_sha.as_deref().map(|s| (s, p.number)))
843        .collect();
844    let known: BTreeSet<i64> = prs.iter().map(|p| p.number).collect();
845
846    let rs = w.query(
847        "MATCH (c:Commit) RETURN c.id AS id, c.message AS message, c.pr_id AS pr_id",
848        &BTreeMap::new(),
849    )?;
850    let mut updates: Vec<(String, String)> = Vec::new();
851    let mut links: BTreeMap<i64, BTreeSet<String>> = BTreeMap::new();
852    for i in 0..rs.len() {
853        let Some(Value::Str(sha)) = rs.get(i, "id") else {
854            continue;
855        };
856        let subject = match rs.get(i, "message") {
857            Some(Value::Str(s)) => s.as_str(),
858            _ => "",
859        };
860        let Some(number) = by_sha
861            .get(sha.as_str())
862            .copied()
863            .or_else(|| subject_pr(subject).filter(|n| known.contains(n)))
864        else {
865            continue;
866        };
867        let key = pr_key(number);
868        links.entry(number).or_default().insert(sha.clone());
869        if rs.get(i, "pr_id") != Some(&Value::Str(key.clone())) {
870            updates.push((sha.clone(), key));
871        }
872    }
873    updates.sort(); // by sha: the same store state writes the same records
874    for (sha, key) in updates {
875        w.set_prop(&sha, "pr_id", Value::Str(key))?;
876    }
877
878    let mut edges: Vec<(String, String, String)> = Vec::new();
879    for (number, shas) in links {
880        let src = pr_key(number);
881        let existing: BTreeSet<String> = w
882            .neighbors(&src, "MERGED_AS", Direction::Out)
883            .unwrap_or_default()
884            .into_iter()
885            .collect();
886        for sha in shas {
887            if !existing.contains(&sha) {
888                edges.push(("MERGED_AS".to_string(), src.clone(), sha));
889            }
890        }
891    }
892    if !edges.is_empty() {
893        w.ingest_with_edges("PR", Vec::new(), ingest, &edges)?;
894    }
895    Ok(())
896}
897
898/// Cypher behind [`file_state_from`] — the cumulative state of every live
899/// `File` node belonging to one unit.
900///
901/// The prefix filter is what keeps a submodule's files out of the parent's
902/// walk and vice versa. `startsWith` is the documented Cypher form (see
903/// `docs/site/query.md`); the parent unit's empty prefix matches everything, so
904/// its own submodules' keys are dropped in [`file_state_from`].
905const FILE_STATE_QUERY: &str =
906    "MATCH (f:File) WHERE startsWith(f.id, $prefix) RETURN f.id AS id, f.commits AS commits, \
907     f.n_commits AS n, f.top_author_id AS top, f.author_counts AS author_counts";
908
909/// Rebuild the in-memory per-file state from the `File` nodes already in the
910/// graph so incremental runs keep `commits` and ownership counts cumulative.
911///
912/// `nested` holds the key prefixes of any submodules inside this unit; their
913/// files belong to their own unit's walk and are dropped here.
914fn file_state_from(rs: &ResultSet, nested: &[String]) -> BTreeMap<String, FileState> {
915    let mut files = BTreeMap::new();
916    for i in 0..rs.len() {
917        let id = match rs.get(i, "id") {
918            Some(Value::Str(s)) => s.clone(),
919            _ => continue,
920        };
921        if nested.iter().any(|p| id.starts_with(p.as_str())) {
922            continue;
923        }
924        let mut st = FileState::default();
925        if let Some(Value::List(l)) = rs.get(i, "commits") {
926            st.commits = l
927                .iter()
928                .filter_map(|v| match v {
929                    Value::Str(s) => Some(s.clone()),
930                    _ => None,
931                })
932                .collect();
933        }
934        if let Some(Value::Int(n)) = rs.get(i, "n") {
935            st.n_commits = *n as usize;
936        }
937        match rs.get(i, "author_counts") {
938            Some(Value::List(l)) => st.set_author_counts(l),
939            // A node written before `author_counts` existed (a store built by
940            // 0.4.x). Fall back to the old approximation — the whole prior
941            // history credited to the incumbent — so those stores keep
942            // working. The prop is written on the next touch, from which point
943            // ownership tracks reality; a full re-ingest repairs it at once.
944            _ => {
945                if let Some(Value::Str(t)) = rs.get(i, "top") {
946                    st.by_author.insert(t.clone(), st.n_commits.max(1));
947                }
948            }
949        }
950        files.insert(id, st);
951    }
952    files
953}
954
955/// The real per-name distribution for every author of this unit, read from the
956/// whole history rather than from the window.
957///
958/// A store built by 0.6.0 has `Author` nodes carrying a `name` and no
959/// `name_counts`, and nothing in the graph records which spelling each commit
960/// used — the display name it holds is the *first* one that version happened to
961/// see, which on a repository where one person commits under two names is the
962/// wrong one about as often as not. Guessing from it cannot recover: seeding the
963/// incumbent with the prior commit count makes the true majority wait for more
964/// new commits than the entire history it is already behind by.
965///
966/// So the log is walked in full, once, on the first run that meets such a node.
967/// The result covers every commit up to `head`, this run's window included, so
968/// the caller uses it *instead of* the window rather than on top of it.
969/// A store this version wrote never reaches here.
970fn legacy_name_counts(
971    w: &WriteGuard<'_>,
972    p: &Pending,
973) -> Result<BTreeMap<String, AuthorState>, CliError> {
974    let mut out: BTreeMap<String, AuthorState> = BTreeMap::new();
975    let stale = p.log.iter().any(|c| {
976        w.has_node(&c.author_email)
977            && w.node_ref(&c.author_email)
978                .and_then(|n| n.prop("name_counts"))
979                .is_none()
980    });
981    let Some(head) = p.head.as_deref().filter(|_| stale) else {
982        return Ok(out);
983    };
984    for c in read_log(&p.unit.path, None, head)? {
985        out.entry(c.author_email).or_default().touch(&c.author_name);
986    }
987    Ok(out)
988}
989
990/// Accumulated effect of one log window, before anything is written.
991#[derive(Default)]
992struct Walk {
993    files: BTreeMap<String, FileState>,
994    /// Email → the names its commits were authored under, and how many each.
995    authors: BTreeMap<String, AuthorState>,
996    commit_rows: Vec<BTreeMap<String, Value>>,
997    touched_edges: Vec<(String, String, String)>,
998    /// Paths whose `File` props changed in this window.
999    dirty: BTreeSet<String>,
1000    deleted: BTreeSet<String>,
1001    /// Node renames to apply, collapsed across chained renames.
1002    renamed: Vec<(String, String)>,
1003    /// Any old path → its final path in this window, for edge retargeting.
1004    alias: BTreeMap<String, String>,
1005}
1006
1007impl Walk {
1008    fn rename(&mut self, from: &str, to: &str, node_exists: bool) {
1009        if let Some(e) = self.renamed.iter_mut().find(|(_, t)| t == from) {
1010            e.1 = to.to_string();
1011        } else if node_exists {
1012            // A node cannot be both deleted and moved; the move wins.
1013            self.deleted.remove(from);
1014            self.renamed.push((from.to_string(), to.to_string()));
1015        } else {
1016            self.deleted.insert(from.to_string());
1017        }
1018        for v in self.alias.values_mut() {
1019            if v == from {
1020                *v = to.to_string();
1021            }
1022        }
1023        self.alias.insert(from.to_string(), to.to_string());
1024    }
1025}
1026
1027/// The `GitSync` props that say *how* a unit was ingested, as opposed to how
1028/// far. Compared against the stored node so that changing a flag refreshes the
1029/// marker even when there is nothing new to walk.
1030fn marker_flag_props(unit: &RepoUnit, opts: &IngestGitOpts) -> Vec<(String, Value)> {
1031    vec![
1032        (
1033            "repo".into(),
1034            Value::Str(unit.path.to_string_lossy().into_owned()),
1035        ),
1036        ("recurse".into(), Value::Bool(opts.recurse_submodules)),
1037        ("prs".into(), Value::Bool(opts.prs)),
1038        ("structure".into(), Value::Bool(opts.structure)),
1039        ("docs".into(), Value::Bool(opts.docs)),
1040    ]
1041}
1042
1043/// One unit and everything read about it before the write pass opens.
1044struct Pending {
1045    unit: RepoUnit,
1046    /// The sha its marker resumes from, absent on a first run.
1047    since: Option<String>,
1048    /// The stored marker disagrees with this run's flags.
1049    stale: bool,
1050    /// `None` when the unit has no commits at all.
1051    head: Option<String>,
1052    log: Vec<GitCommit>,
1053    /// Key prefixes of the submodules nested inside this unit, whose files
1054    /// belong to their own walk.
1055    nested: Vec<String>,
1056}
1057
1058/// Whether `db_dir` would open faster if this run left a snapshot behind.
1059///
1060/// A full ingest always qualifies: it is the run that writes the whole history
1061/// into an empty WAL, and leaving that WAL to be replayed on every subsequent
1062/// open is the single largest fixed cost in the hook path. An incremental run
1063/// qualifies when the store has no snapshot at all — a store built by an older
1064/// release, or one whose full ingest predates this rule — or when the tail
1065/// past the last snapshot has grown beyond [`SNAPSHOT_WAL_BYTES`].
1066///
1067/// The tail *is* `wal.bin`: a snapshot replaces the live WAL with a minimal
1068/// baseline, so its length on disk measures exactly what an open has to replay.
1069///
1070/// Only a run that reached its write phase asks. A run with nothing to take
1071/// returns before entering a write scope, so a store with no snapshot and no
1072/// new commits keeps waiting for a run that has work — which in the hook path
1073/// is the next commit.
1074fn snapshot_due(db_dir: &Path, full: bool) -> bool {
1075    if full || !db_dir.join("snapshot.bin").exists() {
1076        return true;
1077    }
1078    std::fs::metadata(db_dir.join("wal.bin")).is_ok_and(|m| m.len() > SNAPSHOT_WAL_BYTES)
1079}
1080
1081/// Write a snapshot if one is due, after the run's own writes are committed.
1082///
1083/// The WAL is archived rather than dropped — see [`AUTOMATIC_SNAPSHOT`], which
1084/// every snapshot mushroomdb takes on its own shares — and the archives are
1085/// bounded by [`AUTO_SNAPSHOT_RETENTION`], so a store that syncs on every
1086/// commit does not grow an archive per 4 MiB of churn forever.
1087///
1088/// [`AUTOMATIC_SNAPSHOT`]: crate::AUTOMATIC_SNAPSHOT
1089/// [`AUTO_SNAPSHOT_RETENTION`]: crate::AUTO_SNAPSHOT_RETENTION
1090///
1091/// # Why it never fails the run
1092///
1093/// A snapshot is an optimisation for the *next* open, and the data is already
1094/// durable either way:
1095///
1096/// * `Busy` — another process holds the cross-process write lock. Skipped
1097///   without a word; the next run that qualifies takes it.
1098/// * anything else — reported on stderr, because a store that cannot be
1099///   snapshotted is worth knowing about, and the run still succeeds.
1100///
1101/// Call this only from a run that reached its write phase. A run with nothing
1102/// to do returns before entering a write scope, and taking the lock to
1103/// snapshot would turn a genuine no-op into a write.
1104fn snapshot_if_due(db: &SharedDb, db_dir: &Path, full: bool) {
1105    if !snapshot_due(db_dir, full) {
1106        return;
1107    }
1108    let taken = db
1109        .write_with_wait(WRITE_LOCK_WAIT)
1110        .and_then(|mut g| crate::snapshot_automatically(&mut g));
1111    match taken {
1112        Ok(()) | Err(GraphError::Busy { .. }) => {}
1113        Err(e) => eprintln!("snapshot skipped: {e}"),
1114    }
1115}
1116
1117pub fn run_ingest_git(db_dir: &Path, opts: &IngestGitOpts) -> Result<IngestGitReport, CliError> {
1118    // Absolute and symlink-free: it is recorded on the marker, and a later run
1119    // has no reason to share this one's working directory.
1120    let repo = canonical(&opts.repo);
1121    let units = repo_units(&repo, opts.recurse_submodules)?;
1122    let prefixes: Vec<String> = units
1123        .iter()
1124        .map(|u| u.prefix.clone())
1125        .filter(|p| !p.is_empty())
1126        .collect();
1127
1128    let db = SharedDb::open(db_dir)?;
1129    let mut report = IngestGitReport {
1130        submodules: units.len() - 1,
1131        ..Default::default()
1132    };
1133    if opts.ensure_gitignore {
1134        report.gitignore_added = ensure_gitignore(&repo, db_dir)?;
1135    }
1136
1137    let mut pending: Vec<Pending> = Vec::new();
1138    {
1139        let r = db.read();
1140        for unit in units {
1141            let nested = prefixes
1142                .iter()
1143                .filter(|p| **p != unit.prefix && p.starts_with(&unit.prefix))
1144                .cloned()
1145                .collect();
1146            let (since, stale) = match r.node_ref(&unit.sync_key) {
1147                Some(n) => (
1148                    match n.prop("sha") {
1149                        Some(Value::Str(s)) if !s.is_empty() => Some(s),
1150                        _ => None,
1151                    },
1152                    marker_flag_props(&unit, opts)
1153                        .into_iter()
1154                        .any(|(k, v)| n.prop(&k) != Some(v)),
1155                ),
1156                None => (None, false),
1157            };
1158            pending.push(Pending {
1159                unit,
1160                since,
1161                stale,
1162                head: None,
1163                log: Vec::new(),
1164                nested,
1165            });
1166        }
1167    }
1168    // The repository itself is always the first unit, and it is what "this
1169    // database has seen this repo before" means.
1170    report.incremental = pending[0].since.is_some();
1171
1172    for p in &mut pending {
1173        // Pin the end of the walk and the marker to one sha, resolved first. A
1174        // commit landing mid-run then falls outside this range instead of being
1175        // skipped by a marker that advanced past it.
1176        let Some(head) = head_sha(&p.unit.path)? else {
1177            continue; // this unit has no commits yet
1178        };
1179        let mut log = read_log(&p.unit.path, p.since.as_deref(), &head)?;
1180        localise(&mut log, &p.unit.prefix, &gitlink_paths(&p.unit.path));
1181        p.head = Some(head);
1182        p.log = log;
1183    }
1184
1185    let prs = if opts.prs {
1186        fetch_prs(&repo)
1187    } else {
1188        Vec::new()
1189    };
1190    if prs.is_empty() && pending.iter().all(|p| p.log.is_empty() && !p.stale) {
1191        // Nothing new: leave the store untouched so `commit_seq` does not move.
1192        return Ok(report);
1193    }
1194
1195    // Held for the rest of the run. Another process holding the store's
1196    // cross-process write lock is a retry, not a failure: nothing was written.
1197    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1198        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1199        other => CliError(other.to_string()),
1200    })?;
1201    let ingest = IngestOptions::default(); // key `id`, auto-FK suffix `_id`
1202
1203    // Pull requests first: their nodes are what `Commit.pr_id` resolves to.
1204    if !prs.is_empty() {
1205        ingest_prs(&mut w, &prs, &ingest, &mut report)?;
1206    }
1207
1208    let mut authors: BTreeSet<String> = BTreeSet::new();
1209    let mut work = StructureWork::default();
1210    for p in &pending {
1211        if !p.log.is_empty() {
1212            ingest_unit(
1213                &mut w,
1214                p,
1215                opts,
1216                &ingest,
1217                &mut report,
1218                &mut authors,
1219                &mut work,
1220            )?;
1221        }
1222    }
1223    report.authors = authors.len();
1224
1225    // Rules and fulltext, created after the data so each backfills once. They
1226    // span every unit, so they are declared once for the database rather than
1227    // once per repository walked.
1228    if report.commits > 0 && !w.rules().iter().any(|r| r.name == "co_changed") {
1229        let co = Predicate::Overlap {
1230            field: "commits".into(),
1231            min: CO_CHANGE_MIN,
1232        };
1233        w.create_rule(RuleDef {
1234            name: "co_changed".into(),
1235            src_label: "File".into(),
1236            dst_label: "File".into(),
1237            predicate: co.clone(),
1238            edge_type: "CO_CHANGED".into(),
1239            weight_prop: Some("score".into()),
1240            max_edges: Some(10),
1241            approximate: false,
1242            via_label: None,
1243            via_edge: None,
1244            via_dir: None,
1245        })?;
1246        w.create_rule(RuleDef {
1247            name: "knows".into(),
1248            src_label: "Author".into(),
1249            dst_label: "File".into(),
1250            predicate: co,
1251            edge_type: "KNOWS".into(),
1252            weight_prop: Some("score".into()),
1253            max_edges: Some(20),
1254            approximate: false,
1255            via_label: Some("File".into()),
1256            via_edge: Some("TOP_AUTHOR".into()),
1257            via_dir: Some(Direction::In),
1258        })?;
1259        report
1260            .rules_created
1261            .extend(["co_changed".to_string(), "knows".to_string()]);
1262        for (l, field) in [("File", "path"), ("Commit", "message"), ("Author", "name")] {
1263            if !w
1264                .fulltext_pairs()
1265                .contains(&(l.to_string(), field.to_string()))
1266            {
1267                w.enable_fulltext(l, field)?;
1268            }
1269        }
1270    }
1271
1272    // The working tree, on top of the history. It runs after the commit walk
1273    // so every `File` node it reads exists, and its rules are declared after
1274    // its props so each one backfills exactly once.
1275    if opts.structure {
1276        // A first run has nothing to be incremental against, and a run whose
1277        // flags changed (structure or docs just turned on) has to revisit
1278        // files its predecessor deliberately skipped.
1279        let full = !report.incremental || pending.iter().any(|p| p.stale);
1280        report.structure = if full {
1281            structure::refresh_all(&mut w, &repo, "", opts.docs)?
1282        } else {
1283            let mut paths = work.touched;
1284            paths.extend(structure::importers_of(&w, &work.stale)?);
1285            // The retired keys themselves, not just the files that named them.
1286            // An incremental refresh sweeps orphaned symbols only under the
1287            // paths it is handed, and a file this window deleted or renamed
1288            // away is exactly where the orphans are.
1289            paths.extend(work.stale.iter().cloned());
1290            let paths: Vec<String> = paths.into_iter().collect();
1291            structure::refresh_files(&mut w, &repo, "", &paths, opts.docs)?
1292        };
1293        report
1294            .rules_created
1295            .extend(structure::ensure_rules_and_fulltext(&mut w)?);
1296    }
1297
1298    // Every commit this run could link is in the graph by now.
1299    if !prs.is_empty() {
1300        link_prs(&mut w, &prs, &ingest)?;
1301    }
1302
1303    // The markers go last, once every phase of the run has succeeded. They say
1304    // how far a *complete* run got, so a failure anywhere above — a working-tree
1305    // batch that will not commit, a `gh` link pass that errors — leaves them
1306    // where they were and the next run re-walks the same window rather than
1307    // stepping over it. Re-walking a window that was partly applied is safe:
1308    // commits already in the graph are skipped as duplicate keys, file props
1309    // are rewritten from the recomputed state, and a rename whose node already
1310    // moved finds nothing to move.
1311    for p in &pending {
1312        write_marker(&mut w, p, opts)?;
1313    }
1314
1315    // Every write this run makes is now committed, so leave the store in the
1316    // shape that opens fastest. The guard is released first: `snapshot()` takes
1317    // the same cross-process write lock this one holds.
1318    drop(w);
1319    snapshot_if_due(&db, db_dir, !report.incremental);
1320    Ok(report)
1321}
1322
1323/// Wall-clock seconds since the Unix epoch, for [`SYNCED_AT`].
1324///
1325/// This is the one place the ingest reads a clock. A clock before the epoch
1326/// reads as `0` rather than going negative.
1327fn now_unix() -> i64 {
1328    std::time::SystemTime::now()
1329        .duration_since(std::time::UNIX_EPOCH)
1330        .map_or(0, |d| i64::try_from(d.as_secs()).unwrap_or(i64::MAX))
1331}
1332
1333/// Marker prop holding when this store last took data from the repository.
1334///
1335/// `Commit.ts` says when the work was *written*, which on a store synced to
1336/// its repository's head is indistinguishable from now — so it cannot answer
1337/// "how stale is my graph". This can. Absent on a store built before it
1338/// existed, which readers must tolerate.
1339pub const SYNCED_AT: &str = "synced_at";
1340
1341/// Record how far this unit got, and under what flags.
1342///
1343/// Props are written only where they differ, so a run that touches one unit
1344/// does not churn the markers of the others — and a run that changes nothing
1345/// writes nothing at all, [`SYNCED_AT`] included.
1346///
1347/// The stamp therefore means "when this run last changed this marker": a new
1348/// head, or a flag ingested differently from last time. A no-op re-run leaving
1349/// it alone is the point rather than a gap — a graph that had nothing to take
1350/// is not staler for having checked.
1351fn write_marker(w: &mut WriteGuard<'_>, p: &Pending, opts: &IngestGitOpts) -> Result<(), CliError> {
1352    let Some(head) = p.head.as_deref() else {
1353        return Ok(()); // no commits, so nothing to resume from
1354    };
1355    let mut props = marker_flag_props(&p.unit, opts);
1356    if !p.log.is_empty() {
1357        props.push(("sha".into(), Value::Str(head.to_string())));
1358    }
1359    let key = p.unit.sync_key.clone();
1360    if w.has_node(&key) {
1361        let mut changed = false;
1362        for (k, v) in props {
1363            let current = w.node_ref(&key).and_then(|n| n.prop(&k));
1364            if current.as_ref() != Some(&v) {
1365                w.set_prop(&key, &k, v)?;
1366                changed = true;
1367            }
1368        }
1369        if changed {
1370            w.set_prop(&key, SYNCED_AT, Value::Int(now_unix()))?;
1371        }
1372        return Ok(());
1373    }
1374    if p.log.is_empty() {
1375        return Ok(());
1376    }
1377    // It carries `id` like every other label here, so the key is readable from
1378    // Cypher.
1379    props.push(("id".into(), Value::Str(key.clone())));
1380    props.push((SYNCED_AT.into(), Value::Int(now_unix())));
1381    props.sort_by(|a, b| a.0.cmp(&b.0));
1382    w.insert_node("GitSync", &key, props)?;
1383    Ok(())
1384}
1385
1386/// Walk one unit's new commits into the graph.
1387///
1388/// Every path in `p.log` is already a repository-wide key, so this is the
1389/// single-repository algorithm unchanged: only the `File` state it starts from
1390/// and the counts it adds to are scoped to the unit.
1391fn ingest_unit(
1392    w: &mut WriteGuard<'_>,
1393    p: &Pending,
1394    opts: &IngestGitOpts,
1395    ingest: &IngestOptions,
1396    report: &mut IngestGitReport,
1397    authors: &mut BTreeSet<String>,
1398    work: &mut StructureWork,
1399) -> Result<(), CliError> {
1400    let log = &p.log;
1401    let incremental = p.since.is_some();
1402
1403    let mut walk = Walk {
1404        files: if incremental {
1405            let params =
1406                BTreeMap::from([("prefix".to_string(), Value::Str(p.unit.prefix.clone()))]);
1407            file_state_from(&w.query(FILE_STATE_QUERY, &params)?, &p.nested)
1408        } else {
1409            BTreeMap::new()
1410        },
1411        ..Default::default()
1412    };
1413
1414    for c in log {
1415        walk.authors
1416            .entry(c.author_email.clone())
1417            .or_default()
1418            .touch(&c.author_name);
1419        walk.commit_rows.push(BTreeMap::from([
1420            ("id".to_string(), Value::Str(c.sha.clone())),
1421            ("message".to_string(), Value::Str(c.subject.clone())),
1422            ("ts".to_string(), Value::Int(c.ts)),
1423            ("author_id".to_string(), Value::Str(c.author_email.clone())),
1424        ]));
1425        for ch in &c.changes {
1426            match ch {
1427                Change::Added(p) | Change::Modified(p) => {
1428                    if excluded(p, &opts.exclude) {
1429                        continue;
1430                    }
1431                    walk.deleted.remove(p);
1432                    walk.files.entry(p.clone()).or_default().touch(
1433                        &c.sha,
1434                        &c.author_email,
1435                        opts.max_commits_per_file,
1436                    );
1437                    walk.dirty.insert(p.clone());
1438                    walk.touched_edges
1439                        .push(("TOUCHED".into(), c.sha.clone(), p.clone()));
1440                }
1441                Change::Deleted(p) => {
1442                    if excluded(p, &opts.exclude) {
1443                        continue;
1444                    }
1445                    walk.files.remove(p);
1446                    walk.dirty.remove(p);
1447                    walk.deleted.insert(p.clone());
1448                }
1449                Change::Renamed { from, to } => {
1450                    if excluded(to, &opts.exclude) {
1451                        // Moved out of scope: drop the old node, keep no alias
1452                        // so its TOUCHED edges are filtered out below.
1453                        walk.files.remove(from);
1454                        walk.dirty.remove(from);
1455                        walk.deleted.insert(from.clone());
1456                        continue;
1457                    }
1458                    let mut st = walk.files.remove(from).unwrap_or_default();
1459                    st.touch(&c.sha, &c.author_email, opts.max_commits_per_file);
1460                    walk.files.insert(to.clone(), st);
1461                    walk.dirty.remove(from);
1462                    walk.dirty.insert(to.clone());
1463                    walk.deleted.remove(to);
1464                    walk.touched_edges
1465                        .push(("TOUCHED".into(), c.sha.clone(), to.clone()));
1466                    let exists = incremental && w.has_node(from);
1467                    walk.rename(from, to, exists);
1468                }
1469            }
1470        }
1471    }
1472
1473    // What the working-tree pass has to look at afterwards: the paths this
1474    // window changed, and the keys it left pointing at nothing. A key that
1475    // ended the window live again — a file moved away and back — is neither.
1476    work.touched.extend(walk.dirty.iter().cloned());
1477    for key in walk.deleted.iter().chain(walk.alias.keys()) {
1478        if !walk.files.contains_key(key) {
1479            work.stale.insert(key.clone());
1480        }
1481    }
1482
1483    // 1. Authors first: the auto-FK rules for `Commit.author_id` and
1484    //    `File.top_author_id` only infer once their targets resolve to Author.
1485    //    An email already in the graph keeps its accumulated `name_counts`, so
1486    //    the display name is the majority spelling over the whole history
1487    //    rather than over this window.
1488    let legacy = legacy_name_counts(w, p)?;
1489    let mut author_rows = Vec::new();
1490    let mut author_updates = Vec::new();
1491    for (email, seen) in &walk.authors {
1492        let mut merged = AuthorState::default();
1493        match (
1494            w.node_ref(email).and_then(|n| n.prop("name_counts")),
1495            legacy.get(email),
1496        ) {
1497            // The usual path: what the graph already counted, plus this window.
1498            (Some(Value::List(l)), _) => {
1499                merged.set_name_counts(&l);
1500                merged.absorb(seen);
1501            }
1502            // A node written before `name_counts` existed. The recovered walk
1503            // is the whole history and already contains this window, so the
1504            // window is not added to it.
1505            (_, Some(whole_history)) => merged = whole_history.clone(),
1506            // A new author, or a unit whose history was already recovered.
1507            _ => merged.absorb(seen),
1508        }
1509        let props = [
1510            ("name", Value::Str(merged.display_name())),
1511            ("name_counts", merged.name_counts_value()),
1512        ];
1513        if w.has_node(email) {
1514            for (field, want) in props {
1515                if w.node_ref(email).and_then(|n| n.prop(field)).as_ref() != Some(&want) {
1516                    author_updates.push((email.clone(), field, want));
1517                }
1518            }
1519        } else {
1520            let mut row = BTreeMap::from([("id".to_string(), Value::Str(email.clone()))]);
1521            row.extend(props.into_iter().map(|(f, v)| (f.to_string(), v)));
1522            author_rows.push(row);
1523        }
1524    }
1525    if !author_updates.is_empty() {
1526        let mut b = w.batch();
1527        for (key, field, value) in author_updates {
1528            b.set_prop(&key, field, value);
1529        }
1530        b.commit()?;
1531    }
1532    let a = w.ingest_with_edges("Author", author_rows, ingest, &[])?;
1533    report.rules_created.extend(a.rules_created);
1534    authors.extend(walk.authors.keys().cloned());
1535
1536    // 2. Deletes run first so a rename can claim a path freed in this same
1537    //    window, then renames carry each node (and its history) to its new path.
1538    //    `walk.files` is the authority on what is still live: a rename whose
1539    //    destination is not in it is a delete, not a move.
1540    for p in &walk.deleted {
1541        if w.has_node(p) {
1542            w.delete_node(p)?;
1543            report.deleted += 1;
1544        }
1545    }
1546    for (from, to) in &walk.renamed {
1547        if !w.has_node(from) || from == to {
1548            // Nothing to move, or a rename that swapped back to its own path.
1549            continue;
1550        }
1551        if !walk.files.contains_key(to) {
1552            // The destination did not survive the window — it was deleted, or
1553            // moved into an excluded path, after this rename. The node goes
1554            // with it; renaming into a dead path would strand a phantom node
1555            // that no later phase refreshes.
1556            w.delete_node(from)?;
1557            report.deleted += 1;
1558            continue;
1559        }
1560        if w.has_node(to) {
1561            // A pre-existing node already holds the destination path (deleted
1562            // earlier in this window, then claimed by this rename).
1563            w.delete_node(to)?;
1564            report.deleted += 1;
1565        }
1566        w.rename_node(from, to)?;
1567        // The key moved, so the `id` prop must move with it. Phase 3 also sets
1568        // it for every dirty path; doing it here keeps the invariant local to
1569        // the rename and independent of that filter.
1570        w.set_prop(to, "id", Value::Str(to.clone()))?;
1571        report.renamed += 1;
1572    }
1573
1574    // 3. File nodes. Existing nodes are updated in place (including `id`, which
1575    //    must follow the key after a rename); new paths go through ingest so
1576    //    the `top_author_id` auto-FK rule is inferred.
1577    //    Updates to existing nodes go in one batch rather than one WAL commit
1578    //    per property. Every frame in the WAL is replayed on every later open,
1579    //    and each one re-fires the rules watching the property it carries, so a
1580    //    run that appends a frame per property makes every subsequent open
1581    //    slower for as long as that frame lives. The op order inside the batch
1582    //    is the order the individual writes had, so the resulting state is the
1583    //    same one either form produces.
1584    let mut new_file_rows = Vec::new();
1585    let mut updates = Vec::new();
1586    let mut written = 0usize;
1587    for (path, st) in &walk.files {
1588        if incremental && !walk.dirty.contains(path) {
1589            continue;
1590        }
1591        written += 1;
1592        let props = file_props(st, path);
1593        if w.has_node(path) {
1594            updates.extend(props.into_iter().map(|(k, v)| (path.clone(), k, v)));
1595        } else {
1596            new_file_rows.push(props.into_iter().collect::<BTreeMap<_, _>>());
1597        }
1598    }
1599    if !updates.is_empty() {
1600        let mut b = w.batch();
1601        for (key, field, value) in updates {
1602            b.set_prop(&key, &field, value);
1603        }
1604        b.commit()?;
1605    }
1606    let f = w.ingest_with_edges("File", new_file_rows, ingest, &[])?;
1607    report.rules_created.extend(f.rules_created);
1608    // What this run wrote, not every file it has ever seen: an incremental run
1609    // loads the whole known file set to fold the new commits into it, and
1610    // reporting that set made a one-file run print the size of the repository.
1611    report.files += written;
1612
1613    // 4. Commits, then their TOUCHED edges. The two must be separate batches:
1614    //    a batch that both inserts nodes firing a new rule and carries a user
1615    //    edge of a not-yet-interned type writes a WAL frame that cannot be
1616    //    replayed (`Intern` records are emitted in a pre-pass, but on replay the
1617    //    rule fires — and interns its edge type — before the later `Intern`
1618    //    record is read). See the report for a reproducer.
1619    let c = w.ingest_with_edges("Commit", walk.commit_rows, ingest, &[])?;
1620    report.rules_created.extend(c.rules_created);
1621    report.commits += log.len();
1622
1623    // Edges name File keys, so files must already exist. A path renamed later
1624    // in this same window is retargeted to where its node ended up.
1625    let touched: Vec<(String, String, String)> = walk
1626        .touched_edges
1627        .into_iter()
1628        .map(|(t, sha, p)| {
1629            let p = walk.alias.get(&p).cloned().unwrap_or(p);
1630            (t, sha, p)
1631        })
1632        .filter(|(_, _, p)| walk.files.contains_key(p))
1633        .collect();
1634    if !touched.is_empty() {
1635        w.ingest_with_edges("Commit", Vec::new(), ingest, &touched)?;
1636    }
1637    Ok(())
1638}
1639
1640// ── sync and touch ──────────────────────────────────────────────────────────
1641//
1642// `ingest-git` is the command a person runs. These two are what a *hook* runs:
1643// `sync` after a commit lands, `touch` after a single file is edited. Both read
1644// the repository out of the `GitSync` marker rather than taking it as an
1645// argument, so a hook line carries only the database path and keeps working
1646// when the checkout moves.
1647
1648/// What a store with no `GitSync` node is told. Naming the fix matters: this is
1649/// the error a hook installed against the wrong database prints.
1650const NO_MARKER: &str = "store has no git sync marker; run ingest-git first";
1651
1652/// The `GitSync` props that say how this store was built, read back so a later
1653/// `sync` repeats the same run without being told any of it again.
1654///
1655/// `exclude` and `max_commits_per_file` are not on the marker, so a `sync`
1656/// applies [`DEFAULT_EXCLUDES`] and [`DEFAULT_MAX_COMMITS_PER_FILE`]. A store
1657/// first built with custom `--exclude` patterns should keep being maintained
1658/// with `ingest-git`, which takes them.
1659#[derive(Debug, Clone, PartialEq, Eq)]
1660struct SyncMarker {
1661    repo: PathBuf,
1662    recurse: bool,
1663    prs: bool,
1664    structure: bool,
1665    docs: bool,
1666}
1667
1668impl SyncMarker {
1669    fn opts(&self) -> IngestGitOpts {
1670        IngestGitOpts {
1671            repo: self.repo.clone(),
1672            exclude: DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect(),
1673            max_commits_per_file: DEFAULT_MAX_COMMITS_PER_FILE,
1674            recurse_submodules: self.recurse,
1675            prs: self.prs,
1676            structure: self.structure,
1677            docs: self.docs,
1678            ensure_gitignore: false,
1679        }
1680    }
1681}
1682
1683/// Read the marker off an already-open handle.
1684///
1685/// Deliberately not "open the store and read the marker": opening this store is
1686/// by far the most expensive thing either command does — it replays the whole
1687/// WAL — so both of them open once and read the marker through that same
1688/// handle. A separate read-only open just to learn the repository path would
1689/// double the cost of every hook invocation.
1690fn marker_of(r: &structure::Db) -> Result<SyncMarker, CliError> {
1691    let node = r
1692        .node_ref(SYNC_KEY)
1693        .ok_or_else(|| CliError(NO_MARKER.into()))?;
1694    let flag = |name: &str| matches!(node.prop(name), Some(Value::Bool(true)));
1695    match node.prop("repo") {
1696        Some(Value::Str(repo)) if !repo.is_empty() => Ok(SyncMarker {
1697            repo: PathBuf::from(repo),
1698            recurse: flag("recurse"),
1699            prs: flag("prs"),
1700            // A marker written before these two flags existed carries neither,
1701            // and a working-tree pass is what such a run did.
1702            structure: node.prop("structure") != Some(Value::Bool(false)),
1703            docs: node.prop("docs") != Some(Value::Bool(false)),
1704        }),
1705        _ => Err(CliError(NO_MARKER.into())),
1706    }
1707}
1708
1709/// Open a store that must already exist.
1710///
1711/// `SharedDb::open` runs `create_dir_all`, so without this guard a hook line
1712/// carrying a typo'd path would keep creating empty databases and reporting
1713/// that they hold no marker — the same trap [`run_recall`] guards against.
1714///
1715/// [`run_recall`]: crate::recall::run_recall
1716fn open_existing(db_dir: &Path) -> Result<SharedDb, CliError> {
1717    if !db_dir.exists() {
1718        return Err(CliError(format!(
1719            "no database directory at {}",
1720            db_dir.display()
1721        )));
1722    }
1723    Ok(SharedDb::open(db_dir)?)
1724}
1725
1726/// What one [`run_sync`] did.
1727#[derive(Debug, Default, Clone, PartialEq, Eq)]
1728pub struct SyncReport {
1729    /// The incremental history walk.
1730    pub git: IngestGitReport,
1731    /// The working-tree pass over the dirty paths, which the history walk does
1732    /// not see: an edit that has not been committed is in no commit.
1733    pub structure: crate::structure::StructureReport,
1734    /// Dirty paths handed to that pass. Higher than `structure.files_scanned`
1735    /// when some of them are new to the graph, or no longer on disk.
1736    pub dirty_refreshed: usize,
1737}
1738
1739/// Paths that differ from `HEAD` or are not tracked at all, repository-relative
1740/// and sorted.
1741///
1742/// `-z` rather than the default listing: git escapes and quotes a path holding
1743/// a tab, a newline or a non-ASCII byte, and a quoted path matches no key.
1744fn dirty_paths(repo: &Path, exclude: &[String]) -> Result<Vec<String>, CliError> {
1745    const LISTS: [&[&str]; 2] = [
1746        &["diff", "--name-only", "-z", "HEAD"],
1747        &["ls-files", "--others", "--exclude-standard", "-z"],
1748    ];
1749    let mut out = BTreeSet::new();
1750    for args in LISTS {
1751        let o = git_output(repo, args)?;
1752        if !o.status.success() {
1753            // `diff HEAD` fails in a repository with no commits yet. Nothing is
1754            // dirty relative to a head that does not exist.
1755            continue;
1756        }
1757        for path in String::from_utf8_lossy(&o.stdout).split('\0') {
1758            if path.is_empty() || excluded(path, exclude) {
1759                continue;
1760            }
1761            out.insert(path.to_string());
1762        }
1763    }
1764    Ok(out.into_iter().collect())
1765}
1766
1767/// Bring the store up to date with the repository it was built from: the
1768/// commits since the marker, then the working tree where it differs from
1769/// `HEAD`.
1770///
1771/// The second half is what a plain `ingest-git` cannot do. Its working-tree
1772/// pass only visits the paths the *commits* touched, so a file edited and not
1773/// yet committed keeps whatever the graph last recorded about it. A hook that
1774/// runs on every commit wants the uncommitted remainder refreshed too.
1775pub fn run_sync(db_dir: &Path) -> Result<SyncReport, CliError> {
1776    // One handle for the whole run. It is opened before `run_ingest_git`, which
1777    // opens its own and commits through it, but a `SharedDb` refreshes off the
1778    // WAL when a write scope is entered — so the dirty pass below sees every
1779    // commit the ingest just made without this handle being reopened.
1780    let db = open_existing(db_dir)?;
1781    let marker = marker_of(&db.read())?;
1782    let opts = marker.opts();
1783    let mut report = SyncReport {
1784        git: run_ingest_git(db_dir, &opts)?,
1785        ..Default::default()
1786    };
1787    if !opts.structure {
1788        return Ok(report);
1789    }
1790
1791    // Only the root repository's working tree. A submodule's dirty files are
1792    // its own checkout's business, and `--recurse-submodules` resumes each unit
1793    // from its own marker on the next commit there.
1794    let repo = canonical(&marker.repo);
1795    let paths = dirty_paths(&repo, &opts.exclude)?;
1796    report.dirty_refreshed = paths.len();
1797    if paths.is_empty() {
1798        // Nothing to refresh: never enter a write scope, so the run takes no
1799        // lock and `commit_seq` cannot move.
1800        return Ok(report);
1801    }
1802
1803    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1804        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1805        other => CliError(other.to_string()),
1806    })?;
1807    report.structure = structure::refresh_files(&mut w, &repo, "", &paths, opts.docs)?;
1808
1809    // The history half already snapshotted if this was a first ingest; this
1810    // covers the tail the dirty pass just appended. `full` is false because a
1811    // sync is by definition a run against a store that already exists.
1812    drop(w);
1813    snapshot_if_due(&db, db_dir, false);
1814    Ok(report)
1815}
1816
1817pub fn format_touch(r: &structure::StructureReport) -> String {
1818    format!(
1819        "touch: {} file(s), {} symbol(s), {} import(s), {} call(s), {} mention(s)\n",
1820        r.files_scanned, r.symbols, r.imports, r.calls, r.mentions
1821    )
1822}
1823
1824/// One [`SyncReport`] as a single JSON object, for `sync --json`.
1825///
1826/// `text` is exactly what [`format_sync`] prints, so a caller that has the
1827/// object never has to render the digest a second way and never has to parse
1828/// the counts back out of the prose. The MCP `sync` tool is that caller: it
1829/// runs this binary and hands both halves straight to the assistant.
1830#[must_use]
1831pub fn format_sync_json(r: &SyncReport) -> String {
1832    let g = &r.git;
1833    let s = &r.structure;
1834    let gs = &g.structure;
1835    let value = serde_json::json!({
1836        "text": format_sync(r),
1837        "commits": g.commits,
1838        "files": g.files,
1839        "authors": g.authors,
1840        "renamed": g.renamed,
1841        "deleted": g.deleted,
1842        "incremental": g.incremental,
1843        "submodules": g.submodules,
1844        "prs": g.prs,
1845        "rules_created": g.rules_created,
1846        "scanned": {
1847            "files": gs.files_scanned,
1848            "symbols": gs.symbols,
1849            "imports": gs.imports,
1850            "calls": gs.calls,
1851            "mentions": gs.mentions,
1852        },
1853        "dirty_refreshed": r.dirty_refreshed,
1854        "dirty": {
1855            "files": s.files_scanned,
1856            "symbols": s.symbols,
1857            "imports": s.imports,
1858            "calls": s.calls,
1859            "mentions": s.mentions,
1860        },
1861    });
1862    format!("{value}\n")
1863}
1864
1865pub fn format_sync(r: &SyncReport) -> String {
1866    let mut out = format_ingest_git(&r.git);
1867    let s = &r.structure;
1868    out.push_str(&format!(
1869        "  dirty {} path(s): scanned {}, {} symbol(s), {} import(s), {} call(s)\n",
1870        r.dirty_refreshed, s.files_scanned, s.symbols, s.imports, s.calls
1871    ));
1872    out
1873}
1874
1875/// Re-extract exactly the files named, and nothing else.
1876///
1877/// `files` comes from argv when a caller has the paths; otherwise they are read
1878/// out of a `PostToolUse` hook payload on stdin, the same way [`run_recall`]
1879/// reads a prompt. Anything that is not a working-tree file this store already
1880/// knows — a path outside the repository, an excluded one, one the graph has
1881/// never seen — is dropped without comment, because a hook fires on every edit
1882/// the assistant makes and most of them are none of this store's business.
1883///
1884/// [`run_recall`]: crate::recall::run_recall
1885pub fn run_touch(
1886    db_dir: &Path,
1887    files: &[PathBuf],
1888    hook_stdin: Option<&str>,
1889) -> Result<structure::StructureReport, CliError> {
1890    let named: Vec<PathBuf> = if files.is_empty() {
1891        hook_stdin.map(paths_from_payload).unwrap_or_default()
1892    } else {
1893        files.to_vec()
1894    };
1895    if named.is_empty() {
1896        return Ok(structure::StructureReport::default());
1897    }
1898
1899    let db = open_existing(db_dir)?;
1900    let marker = marker_of(&db.read())?;
1901    if !marker.structure {
1902        // The store was built with `--no-structure`, so it holds no working-tree
1903        // props at all and re-extracting one file would be the only exception.
1904        return Ok(structure::StructureReport::default());
1905    }
1906    let repo = canonical(&marker.repo);
1907    let exclude: Vec<String> = DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect();
1908    let cwd = std::env::current_dir().unwrap_or_else(|_| PathBuf::from("."));
1909    let mut paths: BTreeSet<String> = BTreeSet::new();
1910    for path in &named {
1911        let Some(rel) = repo_relative(&repo, &cwd, path) else {
1912            continue;
1913        };
1914        if !excluded(&rel, &exclude) {
1915            paths.insert(rel);
1916        }
1917    }
1918    if paths.is_empty() {
1919        // Nothing of ours changed: never enter a write scope, so no lock is
1920        // taken and a running writer is never made to wait.
1921        return Ok(structure::StructureReport::default());
1922    }
1923
1924    let paths: Vec<String> = paths.into_iter().collect();
1925    let mut w = db.write_with_wait(WRITE_LOCK_WAIT).map_err(|e| match e {
1926        GraphError::Busy { .. } => CliError(BUSY_MESSAGE.to_string()),
1927        other => CliError(other.to_string()),
1928    })?;
1929    structure::refresh_files(&mut w, &repo, "", &paths, marker.docs)
1930}
1931
1932/// The file paths in a `PostToolUse` payload.
1933///
1934/// `tool_input.file_path` covers Edit and Write; `tool_input.edits[].file_path`
1935/// covers the multi-edit shape. Both are read, so a payload carrying either (or
1936/// both) is handled without knowing which tool produced it. A payload that is
1937/// not JSON, or that names no file, yields nothing — never an error.
1938fn paths_from_payload(raw: &str) -> Vec<PathBuf> {
1939    let Ok(v) = serde_json::from_str::<serde_json::Value>(raw) else {
1940        return Vec::new();
1941    };
1942    let input = &v["tool_input"];
1943    let mut out = Vec::new();
1944    let mut push = |value: &serde_json::Value| {
1945        if let Some(s) = value.as_str().map(str::trim).filter(|s| !s.is_empty()) {
1946            out.push(PathBuf::from(s));
1947        }
1948    };
1949    push(&input["file_path"]);
1950    if let Some(edits) = input["edits"].as_array() {
1951        for e in edits {
1952            push(&e["file_path"]);
1953        }
1954    }
1955    out
1956}
1957
1958/// `path` as a key under `repo`, or `None` when it is not inside it.
1959///
1960/// A hook payload carries absolute paths and a person typing the command uses
1961/// relative ones, so a relative path is taken against `cwd`. Both sides are
1962/// then resolved through symlinks before comparing: the marker records the
1963/// canonical repository path, and on macOS a checkout under `/tmp` is reached
1964/// through a symlink that never compares equal as written.
1965fn repo_relative(repo: &Path, cwd: &Path, path: &Path) -> Option<String> {
1966    let absolute = if path.is_absolute() {
1967        path.to_path_buf()
1968    } else {
1969        cwd.join(path)
1970    };
1971    let resolved = resolve_symlinks(&absolute);
1972    let rel = resolved.strip_prefix(repo).ok()?;
1973    let key: Vec<String> = rel
1974        .components()
1975        .map(|c| c.as_os_str().to_string_lossy().into_owned())
1976        .collect();
1977    let key = key.join("/");
1978    (!key.is_empty()).then_some(key)
1979}
1980
1981/// [`canonical`] that still works for a path that no longer exists: a file
1982/// deleted between the edit and the hook resolves through its parent directory.
1983fn resolve_symlinks(p: &Path) -> PathBuf {
1984    if let Ok(resolved) = std::fs::canonicalize(p) {
1985        return resolved;
1986    }
1987    match (p.parent(), p.file_name()) {
1988        (Some(dir), Some(name)) => std::fs::canonicalize(dir)
1989            .map(|d| d.join(name))
1990            .unwrap_or_else(|_| p.to_path_buf()),
1991        _ => p.to_path_buf(),
1992    }
1993}
1994
1995pub fn format_ingest_git(r: &IngestGitReport) -> String {
1996    let mut out = format!(
1997        "ingest-git: {} commit(s), {} file(s), {} author(s){}\n",
1998        r.commits,
1999        r.files,
2000        r.authors,
2001        if r.incremental { " (incremental)" } else { "" }
2002    );
2003    if r.renamed + r.deleted > 0 {
2004        out.push_str(&format!("  renamed {}  deleted {}\n", r.renamed, r.deleted));
2005    }
2006    if r.submodules + r.prs > 0 {
2007        out.push_str(&format!(
2008            "  submodules {}  pull requests {}\n",
2009            r.submodules, r.prs
2010        ));
2011    }
2012    let s = &r.structure;
2013    if s.files_scanned > 0 {
2014        out.push_str(&format!(
2015            "  scanned {} file(s): {} symbol(s), {} import(s), {} call(s), {} mention(s)\n",
2016            s.files_scanned, s.symbols, s.imports, s.calls, s.mentions
2017        ));
2018        if s.skipped_large + s.symbols_capped > 0 {
2019            out.push_str(&format!(
2020                "  hash-only {}  symbol cap hit on {}\n",
2021                s.skipped_large, s.symbols_capped
2022            ));
2023        }
2024    }
2025    if r.gitignore_added {
2026        out.push_str("  added the database directory to .gitignore\n");
2027    }
2028    if !r.rules_created.is_empty() {
2029        out.push_str(&format!("  rules: {}\n", r.rules_created.join(", ")));
2030    }
2031    out
2032}
2033
2034#[cfg(test)]
2035mod tests {
2036    use super::*;
2037
2038    #[test]
2039    fn exclude_matches_prefix_extension_and_substring() {
2040        let pats = vec![
2041            "target/".to_string(),
2042            "*.lock".into(),
2043            "node_modules".into(),
2044        ];
2045        assert!(excluded("target/debug/foo.rs", &pats));
2046        assert!(
2047            !excluded("targeted/foo.rs", &pats),
2048            "prefix needs the slash"
2049        );
2050        assert!(excluded("Cargo.lock", &pats));
2051        assert!(!excluded("Cargo.toml", &pats));
2052        assert!(excluded("ui/node_modules/x/y.js", &pats));
2053        assert!(!excluded("src/lib.rs", &pats));
2054        assert!(!excluded("anything", &[]));
2055    }
2056
2057    /// A `*.` pattern is a file-name suffix, so a compound one works. Reading
2058    /// only the last dot segment would make `*.min.js` — a default — inert,
2059    /// and generated bundles are both the largest files in a tree and the ones
2060    /// least worth parsing.
2061    #[test]
2062    fn a_compound_suffix_pattern_matches() {
2063        let defaults: Vec<String> = DEFAULT_EXCLUDES.iter().map(|p| (*p).to_string()).collect();
2064        // Not under `dist/`, so only the suffix rule can match it.
2065        assert!(excluded("ui/build/bundle.min.js", &defaults));
2066        assert!(excluded("bundle.min.js", &defaults));
2067        assert!(
2068            !excluded("ui/src/app.js", &defaults),
2069            "an ordinary source file is not a bundle"
2070        );
2071        assert!(!excluded("ui/src/minify.js", &defaults));
2072        // The single-extension form is unchanged, and a bare suffix is not a
2073        // match: `*.lock` means something *dot* lock.
2074        assert!(excluded("Cargo.lock", &defaults));
2075        assert!(!excluded(".lock", &defaults));
2076        assert!(!excluded("src/lib.rs", &defaults));
2077    }
2078
2079    #[test]
2080    fn file_props_split_dir_and_ext() {
2081        let mut st = FileState::default();
2082        st.touch("sha1", "a@x.test", 200);
2083        let m: BTreeMap<_, _> = file_props(&st, "src/a/b.rs").into_iter().collect();
2084        assert_eq!(m["dir"], Value::Str("src/a".into()));
2085        assert_eq!(m["ext"], Value::Str("rs".into()));
2086        assert_eq!(m["n_commits"], Value::Int(1));
2087        assert_eq!(m["id"], Value::Str("src/a/b.rs".into()));
2088        assert_eq!(m["top_author_id"], Value::Str("a@x.test".into()));
2089        assert_eq!(
2090            m["author_counts"],
2091            Value::List(vec![Value::Str("a@x.test\t1".into())])
2092        );
2093        let m: BTreeMap<_, _> = file_props(&FileState::default(), "README")
2094            .into_iter()
2095            .collect();
2096        assert_eq!(m["dir"], Value::Str(String::new()));
2097        assert_eq!(m["ext"], Value::Str(String::new()));
2098    }
2099
2100    /// The prop is the whole point of the incremental fix: it must survive a
2101    /// round trip so the next run resumes the real distribution, not the
2102    /// incumbent's total.
2103    #[test]
2104    fn author_counts_round_trip_preserves_the_distribution() {
2105        let mut st = FileState::default();
2106        for _ in 0..3 {
2107            st.touch("s", "alice@x.test", 200);
2108        }
2109        for _ in 0..4 {
2110            st.touch("s", "bob@x.test", 200);
2111        }
2112        let Value::List(encoded) = st.author_counts_value() else {
2113            panic!("author_counts must be a list");
2114        };
2115        assert_eq!(
2116            encoded,
2117            vec![
2118                Value::Str("alice@x.test\t3".into()),
2119                Value::Str("bob@x.test\t4".into()),
2120            ],
2121            "email order, so the prop is stable across runs"
2122        );
2123        let mut reloaded = FileState::default();
2124        reloaded.set_author_counts(&encoded);
2125        assert_eq!(reloaded.by_author, st.by_author);
2126        assert_eq!(reloaded.top_author(), "bob@x.test");
2127    }
2128
2129    /// Malformed entries are skipped, not fatal: the run degrades to the counts
2130    /// it can read rather than refusing to sync.
2131    #[test]
2132    fn author_counts_skips_entries_it_cannot_parse() {
2133        let mut st = FileState::default();
2134        st.set_author_counts(&[
2135            Value::Str("alice@x.test\t2".into()),
2136            Value::Str("no-tab-here".into()),
2137            Value::Str("bob@x.test\tnotanumber".into()),
2138            Value::Str("\t5".into()),
2139            Value::Int(7),
2140        ]);
2141        assert_eq!(st.by_author, BTreeMap::from([("alice@x.test".into(), 2)]));
2142    }
2143
2144    #[test]
2145    fn commits_list_is_capped_and_top_author_is_deterministic() {
2146        let mut st = FileState::default();
2147        for i in 0..5 {
2148            st.touch(&format!("sha{i}"), "b@x.test", 3);
2149        }
2150        st.touch("shaX", "a@x.test", 3);
2151        assert_eq!(st.commits, vec!["sha3", "sha4", "shaX"]);
2152        assert_eq!(st.n_commits, 6);
2153        assert_eq!(st.top_author(), "b@x.test");
2154
2155        // Past the cap the stored `n_commits` is the true total, not the length
2156        // of the truncated `commits` list, and `author_counts` sums to the same
2157        // number. A file over `--max-commits-per-file` would otherwise report a
2158        // history frozen at the cap.
2159        let m: BTreeMap<_, _> = file_props(&st, "src/hot.rs").into_iter().collect();
2160        assert_eq!(m["n_commits"], Value::Int(6));
2161        let Value::List(commits) = &m["commits"] else {
2162            panic!("commits must be a list");
2163        };
2164        assert_eq!(commits.len(), 3, "the list is still capped at 3");
2165        assert!(
2166            matches!(m["n_commits"], Value::Int(n) if n as usize > commits.len()),
2167            "n_commits must exceed the capped list once the cap is passed"
2168        );
2169        assert_eq!(
2170            m["author_counts"],
2171            Value::List(vec![
2172                Value::Str("a@x.test\t1".into()),
2173                Value::Str("b@x.test\t5".into()),
2174            ]),
2175            "the per-author counts sum to n_commits, not to the capped list"
2176        );
2177
2178        let mut tie = FileState::default();
2179        tie.touch("s", "b@x.test", 10);
2180        tie.touch("s", "a@x.test", 10);
2181        assert_eq!(tie.top_author(), "a@x.test", "ties break on smallest email");
2182    }
2183}