Skip to main content

diffler_core/syntax/
intraline.rs

1//! Char-precise intra-line change emphasis driven by an AST diff (syndiff).
2//! Unlike the textual engine in [`crate::pairing`], reindentation and block
3//! wrapping are not flagged — only the byte ranges that differ structurally
4//! are emphasized — so a reformatted or re-wrapped block highlights just the
5//! tokens that actually changed.
6
7use std::ops::Range;
8
9use syndiff::{SyntaxDiffOptions, build_tree, diff_trees};
10
11use crate::model::{FileDiff, Hunk, LineKind};
12use crate::syntax::registry::LanguageRegistry;
13use crate::syntax::{MAX_PARSE_BYTES, line_bounds, parse, split_range_by_line};
14
15/// Emphasis byte ranges per line (one inner vec per source line).
16type LineEmphasis = Vec<Vec<Range<usize>>>;
17
18/// Bounds the AST-diff graph search so a huge, heavily rewritten file cannot
19/// stall the render thread; beyond it `diff_trees` returns `None` and the
20/// caller falls back to the textual engine. Well above any normal diff.
21const GRAPH_LIMIT: usize = 250_000;
22
23impl LanguageRegistry {
24    /// Per-line emphasis byte ranges for both sides, from an AST diff of the
25    /// full old/new content. `None` (caller falls back to the textual engine)
26    /// when the language is unsupported, content is too large, parsing fails,
27    /// or the diff exceeds its graph budget.
28    fn line_emphasis(
29        &self,
30        path: &str,
31        old_src: &str,
32        new_src: &str,
33    ) -> Option<(LineEmphasis, LineEmphasis)> {
34        if old_src.len() > MAX_PARSE_BYTES || new_src.len() > MAX_PARSE_BYTES {
35            return None;
36        }
37        let entry = self.for_path(path)?;
38        // markdown's block tree is coarse (a paragraph is one opaque node); the
39        // textual word-diff emphasizes prose edits far better than an AST diff.
40        if entry.name == "markdown" {
41            return None;
42        }
43        let old_ts = parse(entry, old_src)?;
44        let new_ts = parse(entry, new_src)?;
45        let old_tree = build_tree(old_ts.walk(), old_src);
46        let new_tree = build_tree(new_ts.walk(), new_src);
47        let options = SyntaxDiffOptions {
48            graph_limit: GRAPH_LIMIT,
49        };
50        let (old_ranges, new_ranges) = diff_trees(&old_tree, &new_tree, None, None, Some(options))?;
51        Some((
52            per_line_emphasis(old_src, &old_ranges),
53            per_line_emphasis(new_src, &new_ranges),
54        ))
55    }
56
57    /// Set char-precise emphasis on `file`'s diff lines from the AST diff.
58    /// Returns `false` when the syntactic engine is unavailable, so the caller
59    /// can fall back to the textual engine.
60    pub fn syntactic_emphasis(&self, file: &mut FileDiff) -> bool {
61        let emphasis = match (file.old_text.as_deref(), file.new_text.as_deref()) {
62            (Some(old), Some(new)) => self.line_emphasis(&file.path, old, new),
63            _ => None,
64        };
65        let Some((old_emph, new_emph)) = emphasis else {
66            return false;
67        };
68        for hunk in &mut file.hunks {
69            for line in &mut hunk.lines {
70                let ranges = match (line.new_no, line.old_no) {
71                    (Some(n), _) => new_emph.get(n as usize - 1),
72                    (None, Some(o)) => old_emph.get(o as usize - 1),
73                    _ => None,
74                };
75                line.emphasis =
76                    classify_line(line.kind, &line.text, ranges.map_or(&[], Vec::as_slice));
77            }
78            refine_partial_changes(hunk);
79        }
80        true
81    }
82}
83
84/// Where the AST diff flagged a *partial* line change (some token ranges, not
85/// the whole line and not a reformat), replace the coarse token ranges with a
86/// word-level diff of the paired lines, so only the tokens that actually
87/// changed are emphasized (an edit inside a string scalar shouldn't light up the
88/// whole scalar). Emphasis means "differs from the homolog": a line with no
89/// pair — wholly new or wholly gone — renders plain, never with the stray
90/// fragments the AST diff leaves when it matches a token of new code against
91/// something elsewhere in the old tree.
92fn refine_partial_changes(hunk: &mut Hunk) {
93    let pairs = crate::pairing::paired_run_indices(&hunk.lines);
94    let paired: std::collections::HashSet<usize> =
95        pairs.iter().flat_map(|&(d, a)| [d, a]).collect();
96    for (index, line) in hunk.lines.iter_mut().enumerate() {
97        if matches!(line.kind, LineKind::Deleted | LineKind::Added) && !paired.contains(&index) {
98            line.emphasis = Vec::new();
99        }
100    }
101    for (del_idx, add_idx) in pairs {
102        let partial = hunk
103            .lines
104            .get(del_idx)
105            .is_some_and(|l| !l.emphasis.is_empty())
106            || hunk
107                .lines
108                .get(add_idx)
109                .is_some_and(|l| !l.emphasis.is_empty());
110        if !partial {
111            continue;
112        }
113        let (Some(old), Some(new)) = (
114            hunk.lines.get(del_idx).map(|l| l.text.clone()),
115            hunk.lines.get(add_idx).map(|l| l.text.clone()),
116        ) else {
117            continue;
118        };
119        // the same pair gate as the textual engine, so a refinement that
120        // comes back scattered or near-total drops to plain lines too
121        let (old_emph, new_emph) = crate::pairing::gated_pair_emphasis(&old, &new);
122        if let Some(line) = hunk.lines.get_mut(del_idx) {
123            line.emphasis = old_emph;
124        }
125        if let Some(line) = hunk.lines.get_mut(add_idx) {
126            line.emphasis = new_emph;
127        }
128    }
129}
130
131/// Map whole-file changed byte ranges to the raw per-line, within-line ranges.
132fn per_line_emphasis(src: &str, ranges: &[Range<usize>]) -> LineEmphasis {
133    let bounds = line_bounds(src);
134    let starts: Vec<usize> = bounds.iter().map(|&(s, _)| s).collect();
135    let mut out = vec![Vec::new(); bounds.len()];
136    for r in ranges {
137        split_range_by_line(&bounds, &starts, r, |li, rr| {
138            if let Some(v) = out.get_mut(li) {
139                v.push(rr);
140            }
141        });
142    }
143    out
144}
145
146/// Emphasis for an added/deleted `line` from its raw changed byte `ranges`.
147/// Every changed line keeps its full +/- background; emphasis only marks
148/// punctual edits — a line that changed mostly or entirely gets none, because
149/// highlighting almost everything highlights nothing.
150fn classify_line(kind: LineKind, text: &str, ranges: &[Range<usize>]) -> Vec<Range<usize>> {
151    let _ = kind;
152    let ranges = clamp(ranges, text.len());
153    if ranges.is_empty() || !crate::pairing::emphasis_is_punctual(text, &ranges) {
154        return Vec::new();
155    }
156    ranges
157}
158
159/// Clip ranges to the line's length and drop any that become empty.
160fn clamp(ranges: &[Range<usize>], len: usize) -> Vec<Range<usize>> {
161    ranges
162        .iter()
163        .filter_map(|r| {
164            let end = r.end.min(len);
165            (r.start < end).then_some(r.start..end)
166        })
167        .collect()
168}
169
170#[cfg(test)]
171mod tests {
172    use super::*;
173
174    fn line_with(src: &str, needle: &str) -> usize {
175        src.lines()
176            .position(|l| l.contains(needle))
177            .unwrap_or_else(|| panic!("no line with {needle:?}"))
178    }
179
180    #[test]
181    fn pure_reindent_is_not_emphasized() {
182        let reg = LanguageRegistry::build();
183        let old = "fn f() {\n    let x = compute();\n    use_it(x);\n}\n";
184        let new = "fn f() {\n        let x = compute();\n        use_it(x);\n}\n";
185        let (_, new_e) = reg.line_emphasis("a.rs", old, new).expect("rust parses");
186        assert!(
187            new_e.iter().all(Vec::is_empty),
188            "reindentation must produce no emphasis, got {new_e:?}"
189        );
190    }
191
192    #[test]
193    fn a_real_token_change_is_emphasized() {
194        let reg = LanguageRegistry::build();
195        let old = "fn f() {\n    let x = 1;\n}\n";
196        let new = "fn f() {\n    let x = 2;\n}\n";
197        let (_, new_e) = reg.line_emphasis("a.rs", old, new).expect("rust parses");
198        let changed = line_with(new, "let x = 2");
199        let signature = line_with(new, "fn f()");
200        assert!(!new_e[changed].is_empty(), "the changed line is emphasized");
201        assert!(
202            new_e[signature].is_empty(),
203            "the unchanged signature line is not"
204        );
205    }
206
207    #[test]
208    fn in_string_edit_is_char_precise_not_whole_token() {
209        use crate::model::{DiffLine, FileDiff, FileStatus, Hunk, HunkId, LineKind};
210        let old_line = "fn f() { let s = \"foo/bar\"; }";
211        let new_line = "fn f() { let s = \"foo/EXTRA/bar\"; }";
212        let mut file = FileDiff {
213            path: "a.rs".into(),
214            old_path: None,
215            status: FileStatus::Modified,
216            binary: false,
217            old_text: Some(format!("{old_line}\n")),
218            new_text: Some(format!("{new_line}\n")),
219            hunks: vec![Hunk {
220                id: HunkId("h".into()),
221                old_start: 1,
222                old_lines: 1,
223                new_start: 1,
224                new_lines: 1,
225                context: String::new(),
226                lines: vec![
227                    DiffLine::new(LineKind::Deleted, Some(1), None, old_line.to_owned()),
228                    DiffLine::new(LineKind::Added, None, Some(1), new_line.to_owned()),
229                ],
230            }],
231            hashes: crate::model::HashCache::default(),
232        };
233        assert!(LanguageRegistry::build().syntactic_emphasis(&mut file));
234        let added = &file.hunks[0].lines[1];
235        assert!(!added.emphasis.is_empty(), "the changed line is emphasized");
236        let covered: String = added
237            .emphasis
238            .iter()
239            .filter_map(|r| new_line.get(r.clone()))
240            .collect();
241        // only the inserted run is emphasized, not the whole "foo/EXTRA/bar" token
242        assert!(
243            covered.contains("EXTRA"),
244            "covers the insertion: {covered:?}"
245        );
246        assert!(
247            !covered.contains("foo"),
248            "the unchanged prefix is not emphasized: {covered:?}"
249        );
250    }
251
252    /// A block of wholly-new code where the AST diff matches stray tokens
253    /// (a `}`, an identifier) against the old tree and would light up
254    /// fragments inside plain added lines.
255    #[test]
256    fn wholly_new_code_never_carries_fragment_emphasis() {
257        use crate::model::{DiffLine, FileDiff, FileStatus, Hunk, HunkId, LineKind};
258        let old_src = "function keep(path: string): string {\n    return path;\n}\n";
259        let added = [
260            "function fresh(path: string): string {",
261            "    if (!path) {",
262            "        return \"missing\";",
263            "    }",
264            "    return path;",
265            "}",
266        ];
267        let new_src = format!("{old_src}\n{}\n", added.join("\n"));
268        let lines = added
269            .iter()
270            .enumerate()
271            .map(|(i, text)| {
272                DiffLine::new(
273                    LineKind::Added,
274                    None,
275                    Some(5 + i as u32),
276                    (*text).to_owned(),
277                )
278            })
279            .collect();
280        let mut file = FileDiff {
281            path: "a.ts".into(),
282            old_path: None,
283            status: FileStatus::Modified,
284            binary: false,
285            old_text: Some(old_src.to_owned()),
286            new_text: Some(new_src),
287            hunks: vec![Hunk {
288                id: HunkId("h".into()),
289                old_start: 3,
290                old_lines: 0,
291                new_start: 5,
292                new_lines: 6,
293                context: String::new(),
294                lines,
295            }],
296            hashes: crate::model::HashCache::default(),
297        };
298        assert!(LanguageRegistry::build().syntactic_emphasis(&mut file));
299        for line in &file.hunks[0].lines {
300            assert!(
301                line.emphasis.is_empty(),
302                "no pair, no emphasis — {:?} got {:?}",
303                line.text,
304                line.emphasis
305            );
306        }
307    }
308
309    #[test]
310    fn tsx_wrap_and_reindent_marks_only_real_changes() {
311        let reg = LanguageRegistry::build();
312        let old = "<Form>\n  <Button onClick={onApply}>Apply</Button>\n</Form>\n";
313        let new = "{(values) => (\n  <Form>\n    <Button onClick={() => apply(values)}>Apply</Button>\n  </Form>\n)}\n";
314        let (_, new_e) = reg.line_emphasis("a.tsx", old, new).expect("tsx parses");
315        let reindented = line_with(new, "<Form>");
316        let changed = line_with(new, "apply(values)");
317        assert!(
318            new_e[reindented].is_empty(),
319            "a reindented-but-identical line is not emphasized, got {:?}",
320            new_e[reindented]
321        );
322        assert!(
323            !new_e[changed].is_empty(),
324            "the structurally changed line is emphasized"
325        );
326    }
327
328    #[test]
329    fn unsupported_language_returns_none() {
330        let reg = LanguageRegistry::build();
331        assert!(reg.line_emphasis("a.zzz", "a\n", "b\n").is_none());
332    }
333
334    #[test]
335    fn classify_unchanged_line_gets_no_emphasis() {
336        // a reindent/move: nothing changed within the line, plain +/- bg
337        let emph = classify_line(LineKind::Added, "    <Form>", &[]);
338        assert!(emph.is_empty());
339    }
340
341    #[test]
342    fn classify_whole_line_change_keeps_background_without_emphasis() {
343        // every non-whitespace byte changed -> full +/- bg, no char emphasis
344        let text = "    let entirely_new = compute();";
345        let ranges = [4..7, 8..20, 21..22, 23..text.len()];
346        let emph = classify_line(LineKind::Added, text, &ranges);
347        assert!(
348            emph.is_empty(),
349            "no char emphasis when the whole line changed"
350        );
351    }
352
353    #[test]
354    fn classify_mostly_changed_line_drops_emphasis() {
355        // more than the punctual share changed: highlighting it all says nothing
356        let text = "    let entirely_new = compute();";
357        let ranges = [4..7, 8..20, 23..30];
358        let emph = classify_line(LineKind::Added, text, &ranges);
359        assert!(emph.is_empty(), "{emph:?}");
360    }
361
362    #[test]
363    fn classify_partial_change_keeps_emphasis() {
364        // only `2` changed in `    let x = 2;`
365        let text = "    let x = 2;";
366        let changed = 12..13;
367        let emph = classify_line(LineKind::Added, text, std::slice::from_ref(&changed));
368        assert_eq!(emph.len(), 1);
369        assert_eq!(emph[0], changed);
370    }
371}