cli/enrich.rs
1//! `mushroomdb enrich` — the optional `PostToolUse` hook body for `Grep`.
2//!
3//! Claude Code runs this after a `Grep` has returned, handing it the tool call
4//! and its result as JSON on stdin. A grep answers "where does this name
5//! appear"; the graph answers "what is it" — where it is defined, how many
6//! places call it, who owns the file. This hook appends the second answer to
7//! the first, so the names the search just surfaced arrive with their
8//! definitions rather than as a list of line numbers.
9//!
10//! Unlike [`crate::intercept`], which replaces a search, this only ever adds to
11//! one: adoption is by construction, since nothing has to be chosen.
12//!
13//! # What it looks at
14//!
15//! The pattern first — a grep for a bare name is usually a grep *for that
16//! symbol* — and then the identifiers in the result text, which is where a
17//! regex search's real subjects are. A token earns a line only if it resolves
18//! to a `Symbol` the graph holds, through the same lookup `context` uses for a
19//! bare name, so a token that is merely a word costs nothing but a map lookup.
20//!
21//! Everything else — a store that will not open, a payload that will not
22//! parse, a pattern that names nothing — is silence, like every other hook
23//! this binary writes.
24//!
25//! # Where the facts land
26//!
27//! The text goes out as `additionalContext` on a `hookSpecificOutput` object,
28//! which puts it in the turn *beside* the tool result — not inside it. Claude
29//! Code also documents `updatedToolOutput` for `PostToolUse`, which would
30//! rewrite the result itself, but the reference's list of the tools that
31//! support it could not be retrieved when this was written, and a key the host
32//! ignores is a hook that silently does nothing. Anyone reading the grep-
33//! enrichment arm's numbers should read them as "the facts arrived in the same
34//! turn, adjacent to the matches", not "the matches came back annotated".
35//!
36//! # Cost
37//!
38//! One pass over the `Symbol` nodes builds the name index ([`name_index`]),
39//! and every candidate is answered out of it. Asking the graph per candidate
40//! instead would be up to [`MAX_CANDIDATES`] full scans of the symbol table on
41//! every single `Grep`, which is the whole budget spent on names that mostly
42//! turn out to be ordinary words.
43
44use crate::hook::{cut_to, open_for_hook};
45use crate::intercept::is_identifier;
46use core_api::repograph::{context_with, sanitize, ContextOptions};
47use core_api::Value;
48use std::collections::HashMap;
49use std::fmt::Write as _;
50use std::path::Path;
51
52/// The most this hook may append to a tool result. Wider than the pre-edit
53/// budget because a grep result is already long and the facts have to be
54/// distinguishable from it, and still small enough that a search-heavy session
55/// does not pay for it twice over.
56pub const MAX_CONTEXT_BYTES: usize = 800;
57
58/// Symbols reported. Past five the reader is scanning a second search result
59/// rather than reading an answer.
60const MAX_SYMBOLS: usize = 5;
61
62/// Identifier-shaped tokens taken off a payload before the store is consulted.
63/// A grep result can run to thousands of lines, and every candidate costs a
64/// lookup; the ones that matter are near the top, because that is how a grep
65/// orders its matches.
66const MAX_CANDIDATES: usize = 40;
67
68/// The text this payload has names in: the pattern, then whatever the tool
69/// returned.
70///
71/// `tool_response` is the field Claude Code's hooks reference names for a
72/// tool's result. Its shape is the tool's own, and `Grep`'s is not something
73/// this hook should depend on, so every string found inside it is treated the
74/// same way: as text that may contain identifiers. A payload with no usable
75/// result leaves the pattern, which is on its own often the whole question.
76fn candidate_text(input: &serde_json::Value) -> Vec<String> {
77 let mut out = Vec::new();
78 if let Some(p) = input["tool_input"]["pattern"].as_str() {
79 out.push(p.to_string());
80 }
81 collect_strings(&input["tool_response"], &mut out);
82 out
83}
84
85/// Every string leaf of `v`, in order, appended to `out`. Object *keys* are
86/// deliberately not collected: they are the tool's vocabulary, not the
87/// repository's.
88fn collect_strings(v: &serde_json::Value, out: &mut Vec<String>) {
89 match v {
90 serde_json::Value::String(s) => out.push(s.clone()),
91 serde_json::Value::Array(a) => a.iter().for_each(|x| collect_strings(x, out)),
92 serde_json::Value::Object(o) => o.values().for_each(|x| collect_strings(x, out)),
93 _ => {}
94 }
95}
96
97/// Whether every `:` in `token` belongs to a `::` pair.
98///
99/// [`is_identifier`] allows a lone `:` because a search *pattern* is one token
100/// by construction and a qualified name has its colons in pairs. A grep result
101/// is not one token: its lines read `path:line:text`, and splitting them on
102/// anything but `:` leaves tokens like `rs:1:fn` that pass the identifier test
103/// and name nothing. Requiring the pairs keeps `Type::method` and drops those,
104/// without splitting on `:` and losing qualified names altogether.
105fn colons_are_paired(token: &str) -> bool {
106 let b = token.as_bytes();
107 let mut i = 0;
108 while i < b.len() {
109 if b[i] == b':' {
110 if b.get(i + 1) != Some(&b':') {
111 return false;
112 }
113 i += 2;
114 } else {
115 i += 1;
116 }
117 }
118 true
119}
120
121/// The identifier-shaped tokens in `texts`, first occurrence first, without
122/// repeats and capped at [`MAX_CANDIDATES`].
123fn candidates(texts: &[String]) -> Vec<String> {
124 let mut out: Vec<String> = Vec::new();
125 for text in texts {
126 for token in text.split(|c: char| !(c.is_ascii_alphanumeric() || c == '_' || c == ':')) {
127 // `is_identifier` also rejects a leading digit and anything under
128 // three characters, which is the same floor the redirect uses:
129 // a two-letter name is not evidence the symbol was meant.
130 if !is_identifier(token) || !colons_are_paired(token) {
131 continue;
132 }
133 if !out.iter().any(|t| t == token) {
134 out.push(token.to_string());
135 }
136 if out.len() >= MAX_CANDIDATES {
137 return out;
138 }
139 }
140 }
141 out
142}
143
144/// Every `Symbol` the store holds, by the bare `name` prop, from one scan.
145///
146/// The values are the keys sharing that name, sorted, which is exactly what
147/// [`core_api::repograph::named_symbols`] returns for a single name — this is
148/// that answer for every name at once, so a hook with dozens of candidates
149/// pays for one pass rather than dozens.
150fn name_index(db: &crate::structure::Db) -> HashMap<String, Vec<String>> {
151 let mut out: HashMap<String, Vec<String>> = HashMap::new();
152 for node in db.nodes_with_label("Symbol") {
153 if let Some(Value::Str(name)) = node.prop("name") {
154 out.entry(name).or_default().push(sanitize(node.key()));
155 }
156 }
157 for keys in out.values_mut() {
158 keys.sort();
159 }
160 out
161}
162
163/// One symbol's line: where it is, how many files call it, who owns its file.
164///
165/// Named by its key rather than by the bare name that found it, because a key
166/// is what every other command here takes as a target — the line doubles as
167/// the argument for the `explore` a reader may want next.
168fn describe(db: &crate::structure::Db, key: &str) -> String {
169 let report = context_with(db, None, key, &ContextOptions { source: false });
170 let mut out = String::new();
171 let _ = write!(out, "{key} — ");
172 match (report.file.is_empty(), report.lines) {
173 (false, Some((line, _))) => {
174 let _ = write!(out, "defined at {}:{line}", report.file);
175 }
176 (false, None) => {
177 let _ = write!(out, "defined in {}", report.file);
178 }
179 (true, _) => out.push_str("defined somewhere the graph did not record"),
180 }
181 // Caller *files*, not call sites: "twelve places in one file" and "twelve
182 // files" are different facts, and the one that predicts a change's reach
183 // is the count of files.
184 let callers = report.callers.len() + report.callers_not_shown;
185 let _ = write!(out, ", {callers} callers");
186 if let Some(owner) = &report.owner {
187 let _ = write!(out, ", owner {owner}");
188 }
189 out
190}
191
192/// The whole hook body: parse the payload, open the store, describe whatever
193/// the search named that the graph holds.
194///
195/// `None` for every failure and for every search that named nothing, because
196/// the caller's only two options are "append this" and "stay out of the way".
197#[must_use]
198pub fn run(db_dir: &Path, payload: &str) -> Option<String> {
199 let input: serde_json::Value = serde_json::from_str(payload).ok()?;
200 let tokens = candidates(&candidate_text(&input));
201 if tokens.is_empty() {
202 return None;
203 }
204 let db = open_for_hook(db_dir)?;
205
206 let by_name = name_index(&db);
207 let mut keys: Vec<&str> = Vec::new();
208 for token in &tokens {
209 // A name several symbols share is ambiguous, and picking one of them
210 // would be inventing an answer; a token earns a line only where it
211 // resolves to exactly one symbol.
212 if let Some([key]) = by_name.get(token).map(Vec::as_slice) {
213 keys.push(key);
214 }
215 if keys.len() >= MAX_SYMBOLS {
216 break;
217 }
218 }
219 if keys.is_empty() {
220 return None;
221 }
222
223 let mut out = String::from("about the symbols grep found: ");
224 for (i, key) in keys.iter().enumerate() {
225 let line = describe(&db, key);
226 // The budget is a hard cap, and a half-written symbol is worse than
227 // one fewer: the line is only added if the whole of it fits.
228 let sep = if i > 0 { "; " } else { "" };
229 if out.len() + sep.len() + line.len() > MAX_CONTEXT_BYTES {
230 break;
231 }
232 out.push_str(sep);
233 out.push_str(&line);
234 }
235 // Not one symbol fit, which takes a pathological key to manage.
236 if out.ends_with(": ") {
237 return None;
238 }
239 // The per-line check above already holds the budget; `cut_to` is the
240 // backstop that makes the cap true by construction rather than by
241 // argument.
242 Some(cut_to(out, MAX_CONTEXT_BYTES))
243}
244
245#[cfg(test)]
246mod tests {
247 use super::{candidate_text, candidates, colons_are_paired};
248
249 #[test]
250 fn qualified_names_survive_but_grep_line_prefixes_do_not() {
251 assert!(colons_are_paired("Type::method"));
252 assert!(colons_are_paired("render_map"));
253 assert!(!colons_are_paired("rs:1:fn"));
254 assert!(!colons_are_paired("a:::b"));
255 }
256
257 #[test]
258 fn the_pattern_comes_first_and_repeats_are_dropped() {
259 let payload = serde_json::json!({
260 "tool_input": {"pattern": "render_map"},
261 "tool_response": {"content": "a.rs:1:fn render_map()\nb.rs:2:render_map();"},
262 });
263 let found = candidates(&candidate_text(&payload));
264 assert_eq!(found.first().map(String::as_str), Some("render_map"));
265 assert_eq!(
266 found.iter().filter(|t| *t == "render_map").count(),
267 1,
268 "{found:?}"
269 );
270 assert!(
271 !found.iter().any(|t| t.contains(':') && !t.contains("::")),
272 "a `path:line:text` fragment is not a name: {found:?}"
273 );
274 assert!(!found.iter().any(|t| t == "1"), "{found:?}");
275 }
276
277 #[test]
278 fn a_payload_with_no_result_still_offers_the_pattern() {
279 let payload = serde_json::json!({"tool_input": {"pattern": "render_map"}});
280 assert_eq!(candidates(&candidate_text(&payload)), vec!["render_map"]);
281 assert!(candidates(&candidate_text(&serde_json::json!({}))).is_empty());
282 }
283}