Skip to main content

diffler_core/
diff.rs

1//! Intra-line diff: byte ranges of changed regions between a paired
2//! old/new line, used for word-level emphasis on top of line diffs.
3
4use std::ops::Range;
5
6use similar::{ChangeTag, InlineChangeMode, InlineChangeOptions, TextDiff};
7
8/// Below this token-level similarity the pair reads better as plain +/-
9/// lines; the refinement falls back to unemphasized output under it.
10const MIN_INLINE_RATIO: f32 = 0.5;
11
12/// Emphasis runs separated by this many characters or fewer merge into one
13/// span: two highlights straddling a two-char gap read as noise, one reads
14/// as the edit. Merging happens before `pairing::MAX_EMPHASIS_RUNS` counts
15/// the runs — tune the two together.
16const MAX_GAP_CHARS: usize = 2;
17
18/// Byte ranges (into each input) that differ between the two lines.
19///
20/// Word-level, not char-level: lines tokenize into unicode words,
21/// punctuation, and whitespace (UAX #29), so unrelated tokens can never
22/// share a stray letter (`npm` → `bun` is a whole-token swap, not an edit
23/// around a common `n`). A semantic-cleanup pass absorbs coincidental
24/// matches and snaps boundaries to word edges, mirroring what GitHub-class
25/// diff viewers ship.
26///
27/// Returns `(old_emphasis, new_emphasis)`. Adjacent and near-adjacent
28/// ranges are merged.
29///
30/// ```
31/// use diffler_core::diff::intraline;
32///
33/// let (old, new) = intraline("if x < y:", "if x <= y:");
34/// assert!(old.is_empty());
35/// assert_eq!(new, vec![6..7]);
36/// ```
37pub fn intraline(old: &str, new: &str) -> (Vec<Range<usize>>, Vec<Range<usize>>) {
38    let diff = TextDiff::from_lines(old, new);
39    let mut options = InlineChangeOptions::new();
40    options
41        .mode(InlineChangeMode::UnicodeWords)
42        .semantic_cleanup(true)
43        .min_ratio(MIN_INLINE_RATIO);
44
45    // positions accumulate globally per side: `from_lines` splits on embedded
46    // `\r` too, and a fresh counter per segment would emit segment-local
47    // offsets where the caller expects offsets into the whole line
48    let mut old_ranges: Vec<Range<usize>> = Vec::new();
49    let mut new_ranges: Vec<Range<usize>> = Vec::new();
50    let (mut old_pos, mut new_pos) = (0usize, 0usize);
51    for op in diff.ops() {
52        for change in diff.iter_inline_changes_with_options(op, options) {
53            let piece_len: usize = change.values().iter().map(|(_, piece)| piece.len()).sum();
54            let (ranges, pos) = match change.tag() {
55                ChangeTag::Delete => (&mut old_ranges, &mut old_pos),
56                ChangeTag::Insert => (&mut new_ranges, &mut new_pos),
57                ChangeTag::Equal => {
58                    old_pos += piece_len;
59                    new_pos += piece_len;
60                    continue;
61                }
62            };
63            for &(emphasized, piece) in change.values() {
64                if emphasized {
65                    ranges.push(*pos..*pos + piece.len());
66                }
67                *pos += piece.len();
68            }
69        }
70    }
71
72    (
73        coalesce(old, drop_indent_only(old, old_ranges)),
74        coalesce(new, drop_indent_only(new, new_ranges)),
75    )
76}
77
78/// Emphasis on the line's leading indentation is a reindent artifact, not an
79/// edit — matching the AST engine (`syntax::intraline`), which never flags
80/// reformatting. Whitespace changes inside or after content stay: a trailing
81/// space or a tab→space swap is invisible without the highlight.
82fn drop_indent_only(text: &str, ranges: Vec<Range<usize>>) -> Vec<Range<usize>> {
83    ranges
84        .into_iter()
85        .filter(|r| {
86            let ws_only = text.get(r.clone()).is_some_and(|s| s.trim().is_empty());
87            let in_indent = text.get(..r.start).is_some_and(|s| s.trim().is_empty());
88            !(ws_only && in_indent)
89        })
90        .collect()
91}
92
93/// Merge runs whose gap is `MAX_GAP_CHARS` characters or fewer.
94fn coalesce(text: &str, ranges: Vec<Range<usize>>) -> Vec<Range<usize>> {
95    let mut out: Vec<Range<usize>> = Vec::new();
96    for range in ranges {
97        if let Some(last) = out.last_mut()
98            && text
99                .get(last.end..range.start)
100                .is_some_and(|gap| gap.chars().count() <= MAX_GAP_CHARS)
101        {
102            last.end = range.end;
103            continue;
104        }
105        out.push(range);
106    }
107    out
108}
109
110#[cfg(test)]
111mod tests {
112    use super::*;
113
114    #[test]
115    fn equal_lines_have_no_emphasis() {
116        let (old, new) = intraline("same line", "same line");
117        assert!(old.is_empty());
118        assert!(new.is_empty());
119    }
120
121    #[test]
122    fn ranges_are_in_bounds_and_ascending() {
123        let old = "if claims.expiry < now():";
124        let new = "if claims.expiry <= now() - LEEWAY:";
125        let (old_r, new_r) = intraline(old, new);
126        for r in &old_r {
127            assert!(r.end <= old.len());
128        }
129        let mut prev_end = 0;
130        for r in &new_r {
131            assert!(r.start >= prev_end && r.end <= new.len());
132            prev_end = r.end;
133        }
134    }
135
136    #[test]
137    fn insertion_is_emphasized_on_new_side_only() {
138        let (old, new) = intraline("session.touch()", "session.touch(now())");
139        assert!(old.is_empty());
140        let joined: String = new
141            .iter()
142            .map(|r| &"session.touch(now())"[r.clone()])
143            .collect();
144        assert_eq!(joined, "now()");
145    }
146
147    #[test]
148    fn changed_word_is_emphasized_whole_never_fragmented() {
149        // char-level LCS latches onto the shared `n` of npm/bun; word-level
150        // must swap the whole token
151        let old = "npm run lint";
152        let new = "bun run lint";
153        let (old_r, new_r) = intraline(old, new);
154        assert_eq!(old_r, vec![0..3]);
155        assert_eq!(new_r, vec![0..3]);
156    }
157
158    #[test]
159    fn prose_edit_emphasizes_only_the_changed_words() {
160        let old = "runs `better-auth migrate` against src";
161        let new = "runs `auth migrate` against src";
162        let (old_r, new_r) = intraline(old, new);
163        let joined: String = old_r.iter().map(|r| &old[r.clone()]).collect();
164        assert_eq!(joined, "better-");
165        assert!(new_r.is_empty(), "new side only lost words: {new_r:?}");
166    }
167
168    #[test]
169    fn dissimilar_lines_fall_back_to_no_emphasis() {
170        // token overlap below the ratio floor: whole-line rewrite, no confetti
171        let (old, new) = intraline(
172            "npm run migrate -w services/auth",
173            "bun run --filter '@syte-tech/auth-service' migrate",
174        );
175        assert!(old.is_empty(), "{old:?}");
176        assert!(new.is_empty(), "{new:?}");
177    }
178
179    #[test]
180    fn tiny_gaps_between_runs_merge_into_one_span() {
181        let text = "ab";
182        let merged = coalesce(text, vec![0..1, 1..2]);
183        assert_eq!(merged, vec![0..2]);
184        let text = "a--b";
185        let merged = coalesce(text, vec![0..1, 3..4]);
186        assert_eq!(merged, vec![0..4]);
187        let text = "a---b";
188        let merged = coalesce(text, vec![0..1, 4..5]);
189        assert_eq!(merged, vec![0..1, 4..5]);
190    }
191
192    #[test]
193    fn combining_characters_stay_whole() {
194        // "e\u{301}" is one grapheme inside a word token; emphasis must
195        // cover it atomically
196        let old_line = "drink cafe daily";
197        let new_line = "drink cafe\u{301} daily";
198        let (_, new) = intraline(old_line, new_line);
199        for r in &new {
200            assert!(new_line.is_char_boundary(r.start), "range splits a char");
201            assert!(new_line.is_char_boundary(r.end), "range splits a char");
202        }
203        let joined: String = new.iter().map(|r| &new_line[r.clone()]).collect();
204        assert!(joined.contains('\u{301}'), "emphasis: {new:?}");
205    }
206
207    #[test]
208    fn indentation_only_changes_carry_no_emphasis() {
209        // a re-indented line pairs with its twin; the differing indent must
210        // not render as a phantom edit block
211        let (old, new) = intraline("        openAPI(),", "            openAPI(),");
212        assert!(old.is_empty(), "{old:?}");
213        assert!(new.is_empty(), "{new:?}");
214    }
215
216    #[test]
217    fn whitespace_shift_beside_a_real_edit_keeps_the_edit() {
218        let old = "foo  bar";
219        let new = "foo bar baz";
220        let (old_r, new_r) = intraline(old, new);
221        // the shrunk mid-line gap sits after content, so it may stay marked;
222        // only leading-indent emphasis is dropped
223        assert!(old_r.iter().all(|r| r.start >= 3), "{old_r:?}");
224        let joined: String = new_r.iter().map(|r| &new[r.clone()]).collect();
225        assert!(joined.contains("baz"), "the added word survives: {new_r:?}");
226    }
227
228    #[test]
229    fn trailing_and_midline_whitespace_edits_stay_visible() {
230        // without the highlight these changes are invisible on screen
231        let (_, new_r) = intraline("foo();", "foo(); ");
232        assert_eq!(new_r, vec![6..7], "trailing space stays marked");
233        let (old_r, new_r) = intraline("foo\tbar();", "foo    bar();");
234        assert!(
235            !old_r.is_empty() && !new_r.is_empty(),
236            "tab swap stays marked"
237        );
238    }
239
240    #[test]
241    fn empty_inputs() {
242        let (old, new) = intraline("", "");
243        assert!(old.is_empty());
244        assert!(new.is_empty());
245    }
246
247    #[test]
248    fn embedded_carriage_returns_keep_offsets_global() {
249        // a lone `\r` splits the text into segments internally; emphasis
250        // offsets must still address the whole line
251        let old = "alpha\rfoo bar baz";
252        let new = "alpha\rfoo QUX baz";
253        let (old_r, new_r) = intraline(old, new);
254        let covered: String = old_r.iter().map(|r| &old[r.clone()]).collect();
255        assert_eq!(covered, "bar", "{old_r:?}");
256        let covered: String = new_r.iter().map(|r| &new[r.clone()]).collect();
257        assert_eq!(covered, "QUX", "{new_r:?}");
258
259        // multibyte text before the `\r` must not desync byte offsets
260        let old = "héé\rfoo bar baz";
261        let new = "héé\rfoo QUX baz";
262        let (old_r, new_r) = intraline(old, new);
263        for r in old_r.iter().chain(&new_r) {
264            assert!(old.is_char_boundary(r.start) && old.is_char_boundary(r.end));
265        }
266        let covered: String = new_r.iter().map(|r| &new[r.clone()]).collect();
267        assert_eq!(covered, "QUX", "{new_r:?}");
268
269        // edits in two segments emphasize each in place, in order
270        let old = "aa bb cc\rdd ee ff";
271        let new = "aa XX cc\rdd YY ff";
272        let (_, new_r) = intraline(old, new);
273        let covered: Vec<&str> = new_r.iter().map(|r| &new[r.clone()]).collect();
274        assert_eq!(covered, ["XX", "YY"], "{new_r:?}");
275    }
276}