Skip to main content

diffler_core/syntax/
intraline.rs

1//! Char-precise intra-line change emphasis driven by an AST diff (syndiff),
2//! the structural counterpart to the textual engine in [`crate::pairing`].
3//! Only the byte ranges that differ structurally are emphasized, so a
4//! reformatted or re-wrapped block highlights just the tokens that changed.
5
6use std::ops::Range;
7
8use syndiff::{SyntaxDiffOptions, build_tree, diff_trees};
9
10use crate::model::{FileDiff, Hunk, LineKind};
11use crate::syntax::registry::LanguageRegistry;
12use crate::syntax::{MAX_PARSE_BYTES, line_bounds, parse, split_range_by_line};
13
14/// Emphasis byte ranges per line (one inner vec per source line).
15type LineEmphasis = Vec<Vec<Range<usize>>>;
16
17/// Bounds the AST-diff graph search so a huge, heavily rewritten file cannot
18/// stall the render thread; beyond it `diff_trees` returns `None` and the
19/// caller falls back to the textual engine. Well above any normal diff.
20const GRAPH_LIMIT: usize = 250_000;
21
22impl LanguageRegistry {
23    /// Per-line emphasis byte ranges for both sides, from an AST diff of the
24    /// full old/new content. `None` (caller falls back to the textual engine)
25    /// when the language is unsupported, content is too large, parsing fails,
26    /// or the diff exceeds its graph budget.
27    fn line_emphasis(
28        &self,
29        path: &str,
30        old_src: &str,
31        new_src: &str,
32    ) -> Option<(LineEmphasis, LineEmphasis)> {
33        if old_src.len() > MAX_PARSE_BYTES || new_src.len() > MAX_PARSE_BYTES {
34            return None;
35        }
36        let entry = self.for_path(path)?;
37        // markdown's block tree is coarse (a paragraph is one opaque node); the
38        // textual word-diff emphasizes prose edits far better than an AST diff.
39        if entry.name == "markdown" {
40            return None;
41        }
42        let old_ts = parse(entry, old_src)?;
43        let new_ts = parse(entry, new_src)?;
44        let old_tree = build_tree(old_ts.walk(), old_src);
45        let new_tree = build_tree(new_ts.walk(), new_src);
46        let options = SyntaxDiffOptions {
47            graph_limit: GRAPH_LIMIT,
48        };
49        let (old_ranges, new_ranges) = diff_trees(&old_tree, &new_tree, None, None, Some(options))?;
50        Some((
51            per_line_emphasis(old_src, &old_ranges),
52            per_line_emphasis(new_src, &new_ranges),
53        ))
54    }
55
56    /// Set char-precise emphasis on `file`'s diff lines from the AST diff.
57    /// Returns `false` when the syntactic engine is unavailable, so the caller
58    /// can fall back to the textual engine.
59    pub fn syntactic_emphasis(&self, file: &mut FileDiff) -> bool {
60        let emphasis = match (file.old_text.as_deref(), file.new_text.as_deref()) {
61            (Some(old), Some(new)) => self.line_emphasis(&file.path, old, new),
62            _ => None,
63        };
64        let Some((old_emph, new_emph)) = emphasis else {
65            return false;
66        };
67        for hunk in &mut file.hunks {
68            for line in &mut hunk.lines {
69                let ranges = match (line.new_no, line.old_no) {
70                    (Some(n), _) => new_emph.get(n as usize - 1),
71                    (None, Some(o)) => old_emph.get(o as usize - 1),
72                    _ => None,
73                };
74                line.emphasis =
75                    classify_line(line.kind, &line.text, ranges.map_or(&[], Vec::as_slice));
76            }
77            refine_partial_changes(hunk);
78        }
79        true
80    }
81}
82
83/// Where the AST diff flagged a *partial* line change (some token ranges, not
84/// the whole line and not a reformat), replace the coarse token ranges with a
85/// word-level diff of the paired lines, so only the tokens that actually
86/// changed are emphasized (an edit inside a string scalar shouldn't light up the
87/// whole scalar). Emphasis means "differs from the homolog": a line with no
88/// pair (wholly new or wholly gone) renders plain, keeping off the stray
89/// fragments the AST diff leaves when it matches a token of new code against
90/// something elsewhere in the old tree.
91fn refine_partial_changes(hunk: &mut Hunk) {
92    let pairs = crate::pairing::paired_run_indices(&hunk.lines);
93    let paired: std::collections::HashSet<usize> =
94        pairs.iter().flat_map(|&(d, a)| [d, a]).collect();
95    for (index, line) in hunk.lines.iter_mut().enumerate() {
96        if matches!(line.kind, LineKind::Deleted | LineKind::Added) && !paired.contains(&index) {
97            line.emphasis = Vec::new();
98        }
99    }
100    for (del_idx, add_idx) in pairs {
101        let partial = hunk
102            .lines
103            .get(del_idx)
104            .is_some_and(|l| !l.emphasis.is_empty())
105            || hunk
106                .lines
107                .get(add_idx)
108                .is_some_and(|l| !l.emphasis.is_empty());
109        if !partial {
110            continue;
111        }
112        let (Some(old), Some(new)) = (
113            hunk.lines.get(del_idx).map(|l| l.text.clone()),
114            hunk.lines.get(add_idx).map(|l| l.text.clone()),
115        ) else {
116            continue;
117        };
118        // the same pair gate as the textual engine, so a refinement that
119        // comes back scattered or near-total drops to plain lines too
120        let (old_emph, new_emph) = crate::pairing::gated_pair_emphasis(&old, &new);
121        if let Some(line) = hunk.lines.get_mut(del_idx) {
122            line.emphasis = old_emph;
123        }
124        if let Some(line) = hunk.lines.get_mut(add_idx) {
125            line.emphasis = new_emph;
126        }
127    }
128}
129
130/// Map whole-file changed byte ranges to the raw per-line, within-line ranges.
131fn per_line_emphasis(src: &str, ranges: &[Range<usize>]) -> LineEmphasis {
132    let bounds = line_bounds(src);
133    let starts: Vec<usize> = bounds.iter().map(|&(s, _)| s).collect();
134    let mut out = vec![Vec::new(); bounds.len()];
135    for r in ranges {
136        split_range_by_line(&bounds, &starts, r, |li, rr| {
137            if let Some(v) = out.get_mut(li) {
138                v.push(rr);
139            }
140        });
141    }
142    out
143}
144
145/// Emphasis for an added/deleted `line` from its raw changed byte `ranges`.
146/// Every changed line keeps its full +/- background; emphasis only marks
147/// punctual edits: a line that changed mostly or entirely gets none, because
148/// highlighting almost everything highlights nothing.
149fn classify_line(kind: LineKind, text: &str, ranges: &[Range<usize>]) -> Vec<Range<usize>> {
150    let _ = kind;
151    let ranges = clamp(ranges, text.len());
152    if ranges.is_empty() || !crate::pairing::emphasis_is_punctual(text, &ranges) {
153        return Vec::new();
154    }
155    ranges
156}
157
158/// Clip ranges to the line's length and drop any that become empty.
159fn clamp(ranges: &[Range<usize>], len: usize) -> Vec<Range<usize>> {
160    ranges
161        .iter()
162        .filter_map(|r| {
163            let end = r.end.min(len);
164            (r.start < end).then_some(r.start..end)
165        })
166        .collect()
167}
168
169#[cfg(test)]
170mod tests {
171    use super::*;
172
173    fn line_with(src: &str, needle: &str) -> usize {
174        src.lines()
175            .position(|l| l.contains(needle))
176            .unwrap_or_else(|| panic!("no line with {needle:?}"))
177    }
178
179    #[test]
180    fn pure_reindent_is_not_emphasized() {
181        let reg = LanguageRegistry::build();
182        let old = "fn f() {\n    let x = compute();\n    use_it(x);\n}\n";
183        let new = "fn f() {\n        let x = compute();\n        use_it(x);\n}\n";
184        let (_, new_e) = reg.line_emphasis("a.rs", old, new).expect("rust parses");
185        assert!(
186            new_e.iter().all(Vec::is_empty),
187            "reindentation must produce no emphasis, got {new_e:?}"
188        );
189    }
190
191    #[test]
192    fn a_real_token_change_is_emphasized() {
193        let reg = LanguageRegistry::build();
194        let old = "fn f() {\n    let x = 1;\n}\n";
195        let new = "fn f() {\n    let x = 2;\n}\n";
196        let (_, new_e) = reg.line_emphasis("a.rs", old, new).expect("rust parses");
197        let changed = line_with(new, "let x = 2");
198        let signature = line_with(new, "fn f()");
199        assert!(!new_e[changed].is_empty(), "the changed line is emphasized");
200        assert!(
201            new_e[signature].is_empty(),
202            "the unchanged signature line is not"
203        );
204    }
205
206    #[test]
207    fn in_string_edit_is_char_precise_not_whole_token() {
208        use crate::model::{DiffLine, FileDiff, FileStatus, Hunk, HunkId, LineKind};
209        let old_line = "fn f() { let s = \"foo/bar\"; }";
210        let new_line = "fn f() { let s = \"foo/EXTRA/bar\"; }";
211        let mut file = FileDiff {
212            path: "a.rs".into(),
213            old_path: None,
214            status: FileStatus::Modified,
215            binary: false,
216            old_text: Some(format!("{old_line}\n")),
217            new_text: Some(format!("{new_line}\n")),
218            hunks: vec![Hunk {
219                id: HunkId("h".into()),
220                old_start: 1,
221                old_lines: 1,
222                new_start: 1,
223                new_lines: 1,
224                context: String::new(),
225                lines: vec![
226                    DiffLine::new(LineKind::Deleted, Some(1), None, old_line.to_owned()),
227                    DiffLine::new(LineKind::Added, None, Some(1), new_line.to_owned()),
228                ],
229            }],
230            hashes: crate::model::HashCache::default(),
231        };
232        assert!(LanguageRegistry::build().syntactic_emphasis(&mut file));
233        let added = &file.hunks[0].lines[1];
234        assert!(!added.emphasis.is_empty(), "the changed line is emphasized");
235        let covered: String = added
236            .emphasis
237            .iter()
238            .filter_map(|r| new_line.get(r.clone()))
239            .collect();
240        // only the inserted run is emphasized, not the whole "foo/EXTRA/bar" token
241        assert!(
242            covered.contains("EXTRA"),
243            "covers the insertion: {covered:?}"
244        );
245        assert!(
246            !covered.contains("foo"),
247            "the unchanged prefix is not emphasized: {covered:?}"
248        );
249    }
250
251    /// A block of wholly-new code where the AST diff matches stray tokens
252    /// (a `}`, an identifier) against the old tree and would light up
253    /// fragments inside plain added lines.
254    #[test]
255    fn wholly_new_code_never_carries_fragment_emphasis() {
256        use crate::model::{DiffLine, FileDiff, FileStatus, Hunk, HunkId, LineKind};
257        let old_src = "function keep(path: string): string {\n    return path;\n}\n";
258        let added = [
259            "function fresh(path: string): string {",
260            "    if (!path) {",
261            "        return \"missing\";",
262            "    }",
263            "    return path;",
264            "}",
265        ];
266        let new_src = format!("{old_src}\n{}\n", added.join("\n"));
267        let lines = added
268            .iter()
269            .enumerate()
270            .map(|(i, text)| {
271                DiffLine::new(
272                    LineKind::Added,
273                    None,
274                    Some(5 + i as u32),
275                    (*text).to_owned(),
276                )
277            })
278            .collect();
279        let mut file = FileDiff {
280            path: "a.ts".into(),
281            old_path: None,
282            status: FileStatus::Modified,
283            binary: false,
284            old_text: Some(old_src.to_owned()),
285            new_text: Some(new_src),
286            hunks: vec![Hunk {
287                id: HunkId("h".into()),
288                old_start: 3,
289                old_lines: 0,
290                new_start: 5,
291                new_lines: 6,
292                context: String::new(),
293                lines,
294            }],
295            hashes: crate::model::HashCache::default(),
296        };
297        assert!(LanguageRegistry::build().syntactic_emphasis(&mut file));
298        for line in &file.hunks[0].lines {
299            assert!(
300                line.emphasis.is_empty(),
301                "no pair, no emphasis: {:?} got {:?}",
302                line.text,
303                line.emphasis
304            );
305        }
306    }
307
308    #[test]
309    fn tsx_wrap_and_reindent_marks_only_real_changes() {
310        let reg = LanguageRegistry::build();
311        let old = "<Form>\n  <Button onClick={onApply}>Apply</Button>\n</Form>\n";
312        let new = "{(values) => (\n  <Form>\n    <Button onClick={() => apply(values)}>Apply</Button>\n  </Form>\n)}\n";
313        let (_, new_e) = reg.line_emphasis("a.tsx", old, new).expect("tsx parses");
314        let reindented = line_with(new, "<Form>");
315        let changed = line_with(new, "apply(values)");
316        assert!(
317            new_e[reindented].is_empty(),
318            "a reindented-but-identical line is not emphasized, got {:?}",
319            new_e[reindented]
320        );
321        assert!(
322            !new_e[changed].is_empty(),
323            "the structurally changed line is emphasized"
324        );
325    }
326
327    #[test]
328    fn unsupported_language_returns_none() {
329        let reg = LanguageRegistry::build();
330        assert!(reg.line_emphasis("a.zzz", "a\n", "b\n").is_none());
331    }
332
333    #[test]
334    fn classify_unchanged_line_gets_no_emphasis() {
335        // a reindent/move: nothing changed within the line, plain +/- bg
336        let emph = classify_line(LineKind::Added, "    <Form>", &[]);
337        assert!(emph.is_empty());
338    }
339
340    #[test]
341    fn classify_whole_line_change_keeps_background_without_emphasis() {
342        // every non-whitespace byte changed -> full +/- bg, no char emphasis
343        let text = "    let entirely_new = compute();";
344        let ranges = [4..7, 8..20, 21..22, 23..text.len()];
345        let emph = classify_line(LineKind::Added, text, &ranges);
346        assert!(
347            emph.is_empty(),
348            "no char emphasis when the whole line changed"
349        );
350    }
351
352    #[test]
353    fn classify_mostly_changed_line_drops_emphasis() {
354        // more than the punctual share changed: highlighting it all says nothing
355        let text = "    let entirely_new = compute();";
356        let ranges = [4..7, 8..20, 23..30];
357        let emph = classify_line(LineKind::Added, text, &ranges);
358        assert!(emph.is_empty(), "{emph:?}");
359    }
360
361    #[test]
362    fn classify_partial_change_keeps_emphasis() {
363        // only `2` changed in `    let x = 2;`
364        let text = "    let x = 2;";
365        let changed = 12..13;
366        let emph = classify_line(LineKind::Added, text, std::slice::from_ref(&changed));
367        assert_eq!(emph.len(), 1);
368        assert_eq!(emph[0], changed);
369    }
370}