Skip to main content

diffler_core/syntax/
intraline.rs

1//! Char-precise intra-line change emphasis driven by an AST diff (syndiff),
2//! the structural counterpart to the textual engine in [`crate::pairing`].
3//! Only the byte ranges that differ structurally are emphasized, so a
4//! reformatted or re-wrapped block highlights just the tokens that changed.
5
6use std::ops::Range;
7
8use syndiff::{SyntaxDiffOptions, build_tree, diff_trees};
9
10use crate::model::{FileDiff, Hunk, LineKind};
11use crate::syntax::registry::LanguageRegistry;
12use crate::syntax::{MAX_PARSE_BYTES, line_bounds, parse, split_range_by_line};
13
14/// Emphasis byte ranges per line (one inner vec per source line).
15type LineEmphasis = Vec<Vec<Range<usize>>>;
16
17/// Bounds the AST-diff graph search so a huge, heavily rewritten file cannot
18/// stall the render thread; beyond it `diff_trees` returns `None` and the
19/// caller falls back to the textual engine. Well above any normal diff.
20const GRAPH_LIMIT: usize = 250_000;
21
22impl LanguageRegistry {
23    /// Per-line emphasis byte ranges for both sides, from an AST diff of the
24    /// full old/new content. `None` (caller falls back to the textual engine)
25    /// when the language is unsupported, content is too large, parsing fails,
26    /// or the diff exceeds its graph budget.
27    fn line_emphasis(
28        &self,
29        path: &str,
30        old_src: &str,
31        new_src: &str,
32    ) -> Option<(LineEmphasis, LineEmphasis)> {
33        if old_src.len() > MAX_PARSE_BYTES || new_src.len() > MAX_PARSE_BYTES {
34            return None;
35        }
36        let entry = self.for_path(path)?;
37        // markdown's block tree is coarse (a paragraph is one opaque node); the
38        // textual word-diff emphasizes prose edits far better than an AST diff.
39        if entry.name == "markdown" {
40            return None;
41        }
42        let old_ts = parse(entry, old_src)?;
43        let new_ts = parse(entry, new_src)?;
44        let old_tree = build_tree(old_ts.walk(), old_src);
45        let new_tree = build_tree(new_ts.walk(), new_src);
46        let options = SyntaxDiffOptions {
47            graph_limit: GRAPH_LIMIT,
48        };
49        let (old_ranges, new_ranges) = diff_trees(&old_tree, &new_tree, None, None, Some(options))?;
50        Some((
51            per_line_emphasis(old_src, &old_ranges),
52            per_line_emphasis(new_src, &new_ranges),
53        ))
54    }
55
56    /// Set char-precise emphasis on `file`'s diff lines from the AST diff.
57    /// `mark_reformat_only` additionally flags a paired deleted/added line the
58    /// AST diff found no structural difference on at all (a pure reformat)
59    /// for the structural algorithm's dimmed rendering. Returns `false` when
60    /// the syntactic engine is unavailable, so the caller can fall back to
61    /// the textual engine (and structural mode silently reads as a plain
62    /// histogram diff for that file).
63    pub fn syntactic_emphasis(&self, file: &mut FileDiff, mark_reformat_only: bool) -> bool {
64        let emphasis = match (file.old_text.as_deref(), file.new_text.as_deref()) {
65            (Some(old), Some(new)) => self.line_emphasis(&file.path, old, new),
66            _ => None,
67        };
68        let Some((old_emph, new_emph)) = emphasis else {
69            return false;
70        };
71        let mark_reformat_only = mark_reformat_only
72            && self
73                .for_path(&file.path)
74                .is_some_and(|entry| !entry.layout_significant);
75        for hunk in &mut file.hunks {
76            for line in &mut hunk.lines {
77                let ranges = match (line.new_no, line.old_no) {
78                    (Some(n), _) => new_emph.get(n as usize - 1),
79                    (None, Some(o)) => old_emph.get(o as usize - 1),
80                    _ => None,
81                };
82                line.emphasis =
83                    classify_line(line.kind, &line.text, ranges.map_or(&[], Vec::as_slice));
84            }
85            refine_partial_changes(hunk);
86            if mark_reformat_only {
87                mark_reformat_pairs(hunk, &old_emph, &new_emph);
88            }
89        }
90        true
91    }
92}
93
94/// Flag a paired deleted/added line as `reformat_only` when the two differ in
95/// whitespace alone and the AST diff found no token changed on either side,
96/// which keeps a whitespace edit inside a string literal a real change.
97fn mark_reformat_pairs(hunk: &mut Hunk, old_emph: &LineEmphasis, new_emph: &LineEmphasis) {
98    let unchanged = |emph: &LineEmphasis, number: Option<u32>| {
99        number
100            .and_then(|n| n.checked_sub(1))
101            .and_then(|i| emph.get(i as usize))
102            .is_some_and(Vec::is_empty)
103    };
104    let squeezed = |text: &str| text.split_whitespace().collect::<String>();
105    for (del_idx, add_idx) in crate::pairing::paired_run_indices(&hunk.lines) {
106        let (Some(del), Some(add)) = (hunk.lines.get(del_idx), hunk.lines.get(add_idx)) else {
107            continue;
108        };
109        if unchanged(old_emph, del.old_no)
110            && unchanged(new_emph, add.new_no)
111            && squeezed(&del.text) == squeezed(&add.text)
112        {
113            if let Some(line) = hunk.lines.get_mut(del_idx) {
114                line.reformat_only = true;
115            }
116            if let Some(line) = hunk.lines.get_mut(add_idx) {
117                line.reformat_only = true;
118            }
119        }
120    }
121}
122
123/// Where the AST diff flagged a *partial* line change (some token ranges, not
124/// the whole line and not a reformat), replace the coarse token ranges with a
125/// word-level diff of the paired lines, so only the tokens that actually
126/// changed are emphasized (an edit inside a string scalar shouldn't light up the
127/// whole scalar). Emphasis means "differs from the homolog": a line with no
128/// pair (wholly new or wholly gone) renders plain, keeping off the stray
129/// fragments the AST diff leaves when it matches a token of new code against
130/// something elsewhere in the old tree.
131fn refine_partial_changes(hunk: &mut Hunk) {
132    let pairs = crate::pairing::paired_run_indices(&hunk.lines);
133    let paired: std::collections::HashSet<usize> =
134        pairs.iter().flat_map(|&(d, a)| [d, a]).collect();
135    for (index, line) in hunk.lines.iter_mut().enumerate() {
136        if matches!(line.kind, LineKind::Deleted | LineKind::Added) && !paired.contains(&index) {
137            line.emphasis = Vec::new();
138        }
139    }
140    for (del_idx, add_idx) in pairs {
141        let partial = hunk
142            .lines
143            .get(del_idx)
144            .is_some_and(|l| !l.emphasis.is_empty())
145            || hunk
146                .lines
147                .get(add_idx)
148                .is_some_and(|l| !l.emphasis.is_empty());
149        if !partial {
150            continue;
151        }
152        let (Some(old), Some(new)) = (
153            hunk.lines.get(del_idx).map(|l| l.text.clone()),
154            hunk.lines.get(add_idx).map(|l| l.text.clone()),
155        ) else {
156            continue;
157        };
158        // the same pair gate as the textual engine, so a refinement that
159        // comes back scattered or near-total drops to plain lines too
160        let (old_emph, new_emph) = crate::pairing::gated_pair_emphasis(&old, &new);
161        if let Some(line) = hunk.lines.get_mut(del_idx) {
162            line.emphasis = old_emph;
163        }
164        if let Some(line) = hunk.lines.get_mut(add_idx) {
165            line.emphasis = new_emph;
166        }
167    }
168}
169
170/// Map whole-file changed byte ranges to the raw per-line, within-line ranges.
171fn per_line_emphasis(src: &str, ranges: &[Range<usize>]) -> LineEmphasis {
172    let bounds = line_bounds(src);
173    let starts: Vec<usize> = bounds.iter().map(|&(s, _)| s).collect();
174    let mut out = vec![Vec::new(); bounds.len()];
175    for r in ranges {
176        split_range_by_line(&bounds, &starts, r, |li, rr| {
177            if let Some(v) = out.get_mut(li) {
178                v.push(rr);
179            }
180        });
181    }
182    out
183}
184
185/// Emphasis for an added/deleted `line` from its raw changed byte `ranges`.
186/// Every changed line keeps its full +/- background; emphasis only marks
187/// punctual edits: a line that changed mostly or entirely gets none, because
188/// highlighting almost everything highlights nothing.
189fn classify_line(kind: LineKind, text: &str, ranges: &[Range<usize>]) -> Vec<Range<usize>> {
190    let _ = kind;
191    let ranges = clamp(ranges, text.len());
192    if ranges.is_empty() || !crate::pairing::emphasis_is_punctual(text, &ranges) {
193        return Vec::new();
194    }
195    ranges
196}
197
198/// Clip ranges to the line's length and drop any that become empty.
199fn clamp(ranges: &[Range<usize>], len: usize) -> Vec<Range<usize>> {
200    ranges
201        .iter()
202        .filter_map(|r| {
203            let end = r.end.min(len);
204            (r.start < end).then_some(r.start..end)
205        })
206        .collect()
207}
208
209#[cfg(test)]
210mod tests {
211    use super::*;
212
213    fn line_with(src: &str, needle: &str) -> usize {
214        src.lines()
215            .position(|l| l.contains(needle))
216            .unwrap_or_else(|| panic!("no line with {needle:?}"))
217    }
218
219    #[test]
220    fn pure_reindent_is_not_emphasized() {
221        let reg = LanguageRegistry::build();
222        let old = "fn f() {\n    let x = compute();\n    use_it(x);\n}\n";
223        let new = "fn f() {\n        let x = compute();\n        use_it(x);\n}\n";
224        let (_, new_e) = reg.line_emphasis("a.rs", old, new).expect("rust parses");
225        assert!(
226            new_e.iter().all(Vec::is_empty),
227            "reindentation must produce no emphasis, got {new_e:?}"
228        );
229    }
230
231    #[test]
232    fn a_real_token_change_is_emphasized() {
233        let reg = LanguageRegistry::build();
234        let old = "fn f() {\n    let x = 1;\n}\n";
235        let new = "fn f() {\n    let x = 2;\n}\n";
236        let (_, new_e) = reg.line_emphasis("a.rs", old, new).expect("rust parses");
237        let changed = line_with(new, "let x = 2");
238        let signature = line_with(new, "fn f()");
239        assert!(!new_e[changed].is_empty(), "the changed line is emphasized");
240        assert!(
241            new_e[signature].is_empty(),
242            "the unchanged signature line is not"
243        );
244    }
245
246    #[test]
247    fn in_string_edit_is_char_precise_not_whole_token() {
248        use crate::model::{DiffLine, FileDiff, FileStatus, Hunk, HunkId, LineKind};
249        let old_line = "fn f() { let s = \"foo/bar\"; }";
250        let new_line = "fn f() { let s = \"foo/EXTRA/bar\"; }";
251        let mut file = FileDiff {
252            path: "a.rs".into(),
253            old_path: None,
254            status: FileStatus::Modified,
255            binary: false,
256            old_text: Some(format!("{old_line}\n")),
257            new_text: Some(format!("{new_line}\n")),
258            hunks: vec![Hunk {
259                id: HunkId("h".into()),
260                old_start: 1,
261                old_lines: 1,
262                new_start: 1,
263                new_lines: 1,
264                context: String::new(),
265                lines: vec![
266                    DiffLine::new(LineKind::Deleted, Some(1), None, old_line.to_owned()),
267                    DiffLine::new(LineKind::Added, None, Some(1), new_line.to_owned()),
268                ],
269            }],
270            hashes: crate::model::HashCache::default(),
271            blobs: crate::model::BlobIds::default(),
272        };
273        assert!(LanguageRegistry::build().syntactic_emphasis(&mut file, false));
274        let added = &file.hunks[0].lines[1];
275        assert!(!added.emphasis.is_empty(), "the changed line is emphasized");
276        let covered: String = added
277            .emphasis
278            .iter()
279            .filter_map(|r| new_line.get(r.clone()))
280            .collect();
281        // only the inserted run is emphasized, not the whole "foo/EXTRA/bar" token
282        assert!(
283            covered.contains("EXTRA"),
284            "covers the insertion: {covered:?}"
285        );
286        assert!(
287            !covered.contains("foo"),
288            "the unchanged prefix is not emphasized: {covered:?}"
289        );
290    }
291
292    /// The changed lines the structural algorithm flags reformat-only.
293    fn reformat_flagged(path: &str, old: &str, new: &str) -> Vec<String> {
294        let mut file = crate::model::FileDiff {
295            path: path.into(),
296            old_path: None,
297            status: crate::model::FileStatus::Modified,
298            binary: false,
299            old_text: Some(old.into()),
300            new_text: Some(new.into()),
301            hunks: crate::diffalgo::histogram_hunks(old, new, path, 3, true),
302            hashes: crate::model::HashCache::default(),
303            blobs: crate::model::BlobIds::default(),
304        };
305        assert!(LanguageRegistry::build().syntactic_emphasis(&mut file, true));
306        file.hunks
307            .iter()
308            .flat_map(|h| &h.lines)
309            .filter(|l| l.reformat_only)
310            .map(|l| l.text.clone())
311            .collect()
312    }
313
314    #[test]
315    fn structural_mode_marks_a_pure_reindent_pair_reformat_only() {
316        let flagged = reformat_flagged(
317            "a.rs",
318            "fn f() {\n    let x = compute();\n}\n",
319            "fn f() {\n        let x = compute();\n}\n",
320        );
321        assert_eq!(
322            flagged,
323            ["    let x = compute();", "        let x = compute();"]
324        );
325    }
326
327    #[test]
328    fn structural_mode_leaves_a_real_change_unflagged() {
329        let flagged = reformat_flagged(
330            "a.rs",
331            "fn f() {\n    let x = 1;\n}\n",
332            "fn f() {\n    let x = 2;\n}\n",
333        );
334        assert!(flagged.is_empty(), "{flagged:?}");
335    }
336
337    #[test]
338    fn structural_mode_keeps_whitespace_inside_a_string_a_change() {
339        let flagged = reformat_flagged(
340            "a.rs",
341            "fn f() {\n    let s = \"a b\";\n}\n",
342            "fn f() {\n    let s = \"a  b\";\n}\n",
343        );
344        assert!(flagged.is_empty(), "{flagged:?}");
345    }
346
347    #[test]
348    fn structural_mode_never_flags_a_reindent_where_layout_is_syntax() {
349        let python = reformat_flagged(
350            "a.py",
351            "x = 1\nif c:\n    pass\ny = 2\n",
352            "x = 1\nif c:\n    pass\n    y = 2\n",
353        );
354        assert!(python.is_empty(), "moving into the block: {python:?}");
355        let yaml = reformat_flagged("a.yaml", "a:\n  b: 1\nc: 2\n", "a:\n  b: 1\n  c: 2\n");
356        assert!(yaml.is_empty(), "nesting a key: {yaml:?}");
357    }
358
359    /// A block of wholly-new code where the AST diff matches stray tokens
360    /// (a `}`, an identifier) against the old tree and would light up
361    /// fragments inside plain added lines.
362    #[test]
363    fn wholly_new_code_never_carries_fragment_emphasis() {
364        use crate::model::{DiffLine, FileDiff, FileStatus, Hunk, HunkId, LineKind};
365        let old_src = "function keep(path: string): string {\n    return path;\n}\n";
366        let added = [
367            "function fresh(path: string): string {",
368            "    if (!path) {",
369            "        return \"missing\";",
370            "    }",
371            "    return path;",
372            "}",
373        ];
374        let new_src = format!("{old_src}\n{}\n", added.join("\n"));
375        let lines = added
376            .iter()
377            .enumerate()
378            .map(|(i, text)| {
379                DiffLine::new(
380                    LineKind::Added,
381                    None,
382                    Some(5 + i as u32),
383                    (*text).to_owned(),
384                )
385            })
386            .collect();
387        let mut file = FileDiff {
388            path: "a.ts".into(),
389            old_path: None,
390            status: FileStatus::Modified,
391            binary: false,
392            old_text: Some(old_src.to_owned()),
393            new_text: Some(new_src),
394            hunks: vec![Hunk {
395                id: HunkId("h".into()),
396                old_start: 3,
397                old_lines: 0,
398                new_start: 5,
399                new_lines: 6,
400                context: String::new(),
401                lines,
402            }],
403            hashes: crate::model::HashCache::default(),
404            blobs: crate::model::BlobIds::default(),
405        };
406        assert!(LanguageRegistry::build().syntactic_emphasis(&mut file, false));
407        for line in &file.hunks[0].lines {
408            assert!(
409                line.emphasis.is_empty(),
410                "no pair, no emphasis: {:?} got {:?}",
411                line.text,
412                line.emphasis
413            );
414        }
415    }
416
417    #[test]
418    fn tsx_wrap_and_reindent_marks_only_real_changes() {
419        let reg = LanguageRegistry::build();
420        let old = "<Form>\n  <Button onClick={onApply}>Apply</Button>\n</Form>\n";
421        let new = "{(values) => (\n  <Form>\n    <Button onClick={() => apply(values)}>Apply</Button>\n  </Form>\n)}\n";
422        let (_, new_e) = reg.line_emphasis("a.tsx", old, new).expect("tsx parses");
423        let reindented = line_with(new, "<Form>");
424        let changed = line_with(new, "apply(values)");
425        assert!(
426            new_e[reindented].is_empty(),
427            "a reindented-but-identical line is not emphasized, got {:?}",
428            new_e[reindented]
429        );
430        assert!(
431            !new_e[changed].is_empty(),
432            "the structurally changed line is emphasized"
433        );
434    }
435
436    #[test]
437    fn unsupported_language_returns_none() {
438        let reg = LanguageRegistry::build();
439        assert!(reg.line_emphasis("a.zzz", "a\n", "b\n").is_none());
440    }
441
442    #[test]
443    fn classify_unchanged_line_gets_no_emphasis() {
444        // a reindent/move: nothing changed within the line, plain +/- bg
445        let emph = classify_line(LineKind::Added, "    <Form>", &[]);
446        assert!(emph.is_empty());
447    }
448
449    #[test]
450    fn classify_whole_line_change_keeps_background_without_emphasis() {
451        // every non-whitespace byte changed -> full +/- bg, no char emphasis
452        let text = "    let entirely_new = compute();";
453        let ranges = [4..7, 8..20, 21..22, 23..text.len()];
454        let emph = classify_line(LineKind::Added, text, &ranges);
455        assert!(
456            emph.is_empty(),
457            "no char emphasis when the whole line changed"
458        );
459    }
460
461    #[test]
462    fn classify_mostly_changed_line_drops_emphasis() {
463        // more than the punctual share changed: highlighting it all says nothing
464        let text = "    let entirely_new = compute();";
465        let ranges = [4..7, 8..20, 23..30];
466        let emph = classify_line(LineKind::Added, text, &ranges);
467        assert!(emph.is_empty(), "{emph:?}");
468    }
469
470    #[test]
471    fn classify_partial_change_keeps_emphasis() {
472        // only `2` changed in `    let x = 2;`
473        let text = "    let x = 2;";
474        let changed = 12..13;
475        let emph = classify_line(LineKind::Added, text, std::slice::from_ref(&changed));
476        assert_eq!(emph.len(), 1);
477        assert_eq!(emph[0], changed);
478    }
479}