Skip to main content

diffler_core/syntax/
intraline.rs

1//! Char-precise intra-line change emphasis driven by an AST diff (syndiff).
2//! Unlike the textual engine in [`crate::pairing`], reindentation and block
3//! wrapping are not flagged — only the byte ranges that differ structurally
4//! are emphasized — so a reformatted or re-wrapped block highlights just the
5//! tokens that actually changed.
6
7use std::ops::Range;
8
9use syndiff::{SyntaxDiffOptions, build_tree, diff_trees};
10
11use crate::model::{FileDiff, LineKind};
12use crate::syntax::registry::LanguageRegistry;
13use crate::syntax::{MAX_PARSE_BYTES, line_bounds, parse, split_range_by_line};
14
15/// Emphasis byte ranges per line (one inner vec per source line).
16type LineEmphasis = Vec<Vec<Range<usize>>>;
17
18/// Bounds the AST-diff graph search so a huge, heavily rewritten file cannot
19/// stall the render thread; beyond it `diff_trees` returns `None` and the
20/// caller falls back to the textual engine. Well above any normal diff.
21const GRAPH_LIMIT: usize = 250_000;
22
23impl LanguageRegistry {
24    /// Per-line emphasis byte ranges for both sides, from an AST diff of the
25    /// full old/new content. `None` (caller falls back to the textual engine)
26    /// when the language is unsupported, content is too large, parsing fails,
27    /// or the diff exceeds its graph budget.
28    pub fn line_emphasis(
29        &self,
30        path: &str,
31        old_src: &str,
32        new_src: &str,
33    ) -> Option<(LineEmphasis, LineEmphasis)> {
34        if old_src.len() > MAX_PARSE_BYTES || new_src.len() > MAX_PARSE_BYTES {
35            return None;
36        }
37        let entry = self.for_path(path)?;
38        let old_ts = parse(entry, old_src)?;
39        let new_ts = parse(entry, new_src)?;
40        let old_tree = build_tree(old_ts.walk(), old_src);
41        let new_tree = build_tree(new_ts.walk(), new_src);
42        let options = SyntaxDiffOptions {
43            graph_limit: GRAPH_LIMIT,
44        };
45        let (old_ranges, new_ranges) = diff_trees(&old_tree, &new_tree, None, None, Some(options))?;
46        Some((
47            per_line_emphasis(old_src, &old_ranges),
48            per_line_emphasis(new_src, &new_ranges),
49        ))
50    }
51
52    /// Set char-precise emphasis on `file`'s diff lines from the AST diff.
53    /// Returns `false` when the syntactic engine is unavailable, so the caller
54    /// can fall back to the textual engine.
55    pub fn syntactic_emphasis(&self, file: &mut FileDiff) -> bool {
56        let emphasis = match (file.old_text.as_deref(), file.new_text.as_deref()) {
57            (Some(old), Some(new)) => self.line_emphasis(&file.path, old, new),
58            _ => None,
59        };
60        let Some((old_emph, new_emph)) = emphasis else {
61            return false;
62        };
63        for hunk in &mut file.hunks {
64            for line in &mut hunk.lines {
65                let ranges = match (line.new_no, line.old_no) {
66                    (Some(n), _) => new_emph.get(n as usize - 1),
67                    (None, Some(o)) => old_emph.get(o as usize - 1),
68                    _ => None,
69                };
70                let (moved, emphasis) =
71                    classify_line(line.kind, &line.text, ranges.map_or(&[], Vec::as_slice));
72                line.moved = moved;
73                line.emphasis = emphasis;
74            }
75        }
76        true
77    }
78}
79
80/// Map whole-file changed byte ranges to the raw per-line, within-line ranges.
81fn per_line_emphasis(src: &str, ranges: &[Range<usize>]) -> LineEmphasis {
82    let bounds = line_bounds(src);
83    let starts: Vec<usize> = bounds.iter().map(|&(s, _)| s).collect();
84    let mut out = vec![Vec::new(); bounds.len()];
85    for r in ranges {
86        split_range_by_line(&bounds, &starts, r, |li, rr| {
87            if let Some(v) = out.get_mut(li) {
88                v.push(rr);
89            }
90        });
91    }
92    out
93}
94
95/// Classify an added/deleted `line` from its raw changed byte `ranges`:
96/// - no change → a reindent/move: `(moved = true, no emphasis)`, a thin rail;
97/// - the whole content changed → `(false, no emphasis)`, a full +/- background;
98/// - some tokens changed → `(false, those ranges)`, background plus emphasis.
99fn classify_line(kind: LineKind, text: &str, ranges: &[Range<usize>]) -> (bool, Vec<Range<usize>>) {
100    let ranges = clamp(ranges, text.len());
101    let changed = matches!(kind, LineKind::Added | LineKind::Deleted);
102    if ranges.is_empty() {
103        return (changed, Vec::new());
104    }
105    if whole_line_changed(text, &ranges) {
106        return (false, Vec::new());
107    }
108    (false, ranges)
109}
110
111/// True when every non-whitespace byte of the line is emphasized (gaps fall
112/// only on whitespace) — i.e. the entire content changed.
113fn whole_line_changed(text: &str, ranges: &[Range<usize>]) -> bool {
114    let mut any_content = false;
115    for (i, &b) in text.as_bytes().iter().enumerate() {
116        if b == b' ' || b == b'\t' {
117            continue;
118        }
119        any_content = true;
120        if !ranges.iter().any(|r| r.start <= i && i < r.end) {
121            return false;
122        }
123    }
124    any_content
125}
126
127/// Clip ranges to the line's length and drop any that become empty.
128fn clamp(ranges: &[Range<usize>], len: usize) -> Vec<Range<usize>> {
129    ranges
130        .iter()
131        .filter_map(|r| {
132            let end = r.end.min(len);
133            (r.start < end).then_some(r.start..end)
134        })
135        .collect()
136}
137
138#[cfg(test)]
139mod tests {
140    use super::*;
141
142    fn line_with(src: &str, needle: &str) -> usize {
143        src.lines()
144            .position(|l| l.contains(needle))
145            .unwrap_or_else(|| panic!("no line with {needle:?}"))
146    }
147
148    #[test]
149    fn pure_reindent_is_not_emphasized() {
150        let reg = LanguageRegistry::build();
151        let old = "fn f() {\n    let x = compute();\n    use_it(x);\n}\n";
152        let new = "fn f() {\n        let x = compute();\n        use_it(x);\n}\n";
153        let (_, new_e) = reg.line_emphasis("a.rs", old, new).expect("rust parses");
154        assert!(
155            new_e.iter().all(Vec::is_empty),
156            "reindentation must produce no emphasis, got {new_e:?}"
157        );
158    }
159
160    #[test]
161    fn a_real_token_change_is_emphasized() {
162        let reg = LanguageRegistry::build();
163        let old = "fn f() {\n    let x = 1;\n}\n";
164        let new = "fn f() {\n    let x = 2;\n}\n";
165        let (_, new_e) = reg.line_emphasis("a.rs", old, new).expect("rust parses");
166        let changed = line_with(new, "let x = 2");
167        let signature = line_with(new, "fn f()");
168        assert!(!new_e[changed].is_empty(), "the changed line is emphasized");
169        assert!(
170            new_e[signature].is_empty(),
171            "the unchanged signature line is not"
172        );
173    }
174
175    #[test]
176    fn tsx_wrap_and_reindent_marks_only_real_changes() {
177        let reg = LanguageRegistry::build();
178        let old = "<Form>\n  <Button onClick={onApply}>Apply</Button>\n</Form>\n";
179        let new = "{(values) => (\n  <Form>\n    <Button onClick={() => apply(values)}>Apply</Button>\n  </Form>\n)}\n";
180        let (_, new_e) = reg.line_emphasis("a.tsx", old, new).expect("tsx parses");
181        let reindented = line_with(new, "<Form>");
182        let changed = line_with(new, "apply(values)");
183        assert!(
184            new_e[reindented].is_empty(),
185            "a reindented-but-identical line is not emphasized, got {:?}",
186            new_e[reindented]
187        );
188        assert!(
189            !new_e[changed].is_empty(),
190            "the structurally changed line is emphasized"
191        );
192    }
193
194    #[test]
195    fn unsupported_language_returns_none() {
196        let reg = LanguageRegistry::build();
197        assert!(reg.line_emphasis("a.zzz", "a\n", "b\n").is_none());
198    }
199
200    #[test]
201    fn classify_reindented_line_is_a_move() {
202        // an added/deleted line with no changed bytes only moved/reindented
203        let (moved, emph) = classify_line(LineKind::Added, "    <Form>", &[]);
204        assert!(moved, "reindent/move is flagged");
205        assert!(emph.is_empty());
206    }
207
208    #[test]
209    fn classify_whole_line_change_keeps_background_without_emphasis() {
210        // every non-whitespace byte changed -> full +/- bg, no char emphasis
211        let text = "    let entirely_new = compute();";
212        let ranges = [4..7, 8..20, 21..22, 23..text.len()];
213        let (moved, emph) = classify_line(LineKind::Added, text, &ranges);
214        assert!(!moved, "a wholly-changed line is a real change, not a move");
215        assert!(
216            emph.is_empty(),
217            "no char emphasis when the whole line changed"
218        );
219    }
220
221    #[test]
222    fn classify_partial_change_keeps_emphasis() {
223        // only `2` changed in `    let x = 2;`
224        let text = "    let x = 2;";
225        let changed = 12..13;
226        let (moved, emph) = classify_line(LineKind::Added, text, std::slice::from_ref(&changed));
227        assert!(!moved);
228        assert_eq!(emph.len(), 1);
229        assert_eq!(emph[0], changed);
230    }
231
232    #[test]
233    fn classify_context_line_is_never_moved() {
234        let (moved, emph) = classify_line(LineKind::Context, "    unchanged", &[]);
235        assert!(!moved);
236        assert!(emph.is_empty());
237    }
238}