Skip to main content

diffler_core/
diff.rs

1//! Intra-line diff: byte ranges of changed regions between a paired
2//! old/new line, used for word-level emphasis on top of line diffs.
3
4use std::ops::Range;
5
6use similar::{ChangeTag, InlineChangeMode, InlineChangeOptions, TextDiff};
7
8/// Below this token-level similarity the pair reads better as plain +/-
9/// lines; the refinement falls back to unemphasized output under it.
10const MIN_INLINE_RATIO: f32 = 0.5;
11
12/// Emphasis runs separated by this many characters or fewer merge into one
13/// span: two highlights straddling a two-char gap read as noise, one reads
14/// as the edit. Merging happens before `pairing::MAX_EMPHASIS_RUNS` counts
15/// the runs — tune the two together.
16const MAX_GAP_CHARS: usize = 2;
17
18/// Byte ranges (into each input) that differ between the two lines.
19///
20/// Word-level, not char-level: lines tokenize into unicode words,
21/// punctuation, and whitespace (UAX #29), so unrelated tokens can never
22/// share a stray letter (`npm` → `bun` is a whole-token swap, not an edit
23/// around a common `n`). A semantic-cleanup pass absorbs coincidental
24/// matches and snaps boundaries to word edges, mirroring what GitHub-class
25/// diff viewers ship.
26///
27/// Returns `(old_emphasis, new_emphasis)`. Adjacent and near-adjacent
28/// ranges are merged.
29///
30/// ```
31/// use diffler_core::diff::intraline;
32///
33/// let (old, new) = intraline("if x < y:", "if x <= y:");
34/// assert!(old.is_empty());
35/// assert_eq!(new, vec![6..7]);
36/// ```
37pub fn intraline(old: &str, new: &str) -> (Vec<Range<usize>>, Vec<Range<usize>>) {
38    let diff = TextDiff::from_lines(old, new);
39    let mut options = InlineChangeOptions::new();
40    options
41        .mode(InlineChangeMode::UnicodeWords)
42        .semantic_cleanup(true)
43        .min_ratio(MIN_INLINE_RATIO);
44
45    // positions accumulate globally per side: `from_lines` splits on embedded
46    // `\r` too, and a fresh counter per segment would emit segment-local
47    // offsets where the caller expects offsets into the whole line
48    let mut old_ranges: Vec<Range<usize>> = Vec::new();
49    let mut new_ranges: Vec<Range<usize>> = Vec::new();
50    let (mut old_pos, mut new_pos) = (0usize, 0usize);
51    for op in diff.ops() {
52        for change in diff.iter_inline_changes_with_options(op, options) {
53            let piece_len: usize = change.values().iter().map(|(_, piece)| piece.len()).sum();
54            let (ranges, pos) = match change.tag() {
55                ChangeTag::Delete => (&mut old_ranges, &mut old_pos),
56                ChangeTag::Insert => (&mut new_ranges, &mut new_pos),
57                ChangeTag::Equal => {
58                    old_pos += piece_len;
59                    new_pos += piece_len;
60                    continue;
61                }
62            };
63            for &(emphasized, piece) in change.values() {
64                if emphasized {
65                    ranges.push(*pos..*pos + piece.len());
66                }
67                *pos += piece.len();
68            }
69        }
70    }
71
72    (coalesce(old, old_ranges), coalesce(new, new_ranges))
73}
74
75/// Merge runs whose gap is `MAX_GAP_CHARS` characters or fewer.
76fn coalesce(text: &str, ranges: Vec<Range<usize>>) -> Vec<Range<usize>> {
77    let mut out: Vec<Range<usize>> = Vec::new();
78    for range in ranges {
79        if let Some(last) = out.last_mut()
80            && text
81                .get(last.end..range.start)
82                .is_some_and(|gap| gap.chars().count() <= MAX_GAP_CHARS)
83        {
84            last.end = range.end;
85            continue;
86        }
87        out.push(range);
88    }
89    out
90}
91
92#[cfg(test)]
93mod tests {
94    use super::*;
95
96    #[test]
97    fn equal_lines_have_no_emphasis() {
98        let (old, new) = intraline("same line", "same line");
99        assert!(old.is_empty());
100        assert!(new.is_empty());
101    }
102
103    #[test]
104    fn ranges_are_in_bounds_and_ascending() {
105        let old = "if claims.expiry < now():";
106        let new = "if claims.expiry <= now() - LEEWAY:";
107        let (old_r, new_r) = intraline(old, new);
108        for r in &old_r {
109            assert!(r.end <= old.len());
110        }
111        let mut prev_end = 0;
112        for r in &new_r {
113            assert!(r.start >= prev_end && r.end <= new.len());
114            prev_end = r.end;
115        }
116    }
117
118    #[test]
119    fn insertion_is_emphasized_on_new_side_only() {
120        let (old, new) = intraline("session.touch()", "session.touch(now())");
121        assert!(old.is_empty());
122        let joined: String = new
123            .iter()
124            .map(|r| &"session.touch(now())"[r.clone()])
125            .collect();
126        assert_eq!(joined, "now()");
127    }
128
129    #[test]
130    fn changed_word_is_emphasized_whole_never_fragmented() {
131        // char-level LCS latches onto the shared `n` of npm/bun; word-level
132        // must swap the whole token
133        let old = "npm run lint";
134        let new = "bun run lint";
135        let (old_r, new_r) = intraline(old, new);
136        assert_eq!(old_r, vec![0..3]);
137        assert_eq!(new_r, vec![0..3]);
138    }
139
140    #[test]
141    fn prose_edit_emphasizes_only_the_changed_words() {
142        let old = "runs `better-auth migrate` against src";
143        let new = "runs `auth migrate` against src";
144        let (old_r, new_r) = intraline(old, new);
145        let joined: String = old_r.iter().map(|r| &old[r.clone()]).collect();
146        assert_eq!(joined, "better-");
147        assert!(new_r.is_empty(), "new side only lost words: {new_r:?}");
148    }
149
150    #[test]
151    fn dissimilar_lines_fall_back_to_no_emphasis() {
152        // token overlap below the ratio floor: whole-line rewrite, no confetti
153        let (old, new) = intraline(
154            "npm run migrate -w services/auth",
155            "bun run --filter '@syte-tech/auth-service' migrate",
156        );
157        assert!(old.is_empty(), "{old:?}");
158        assert!(new.is_empty(), "{new:?}");
159    }
160
161    #[test]
162    fn tiny_gaps_between_runs_merge_into_one_span() {
163        let text = "ab";
164        let merged = coalesce(text, vec![0..1, 1..2]);
165        assert_eq!(merged, vec![0..2]);
166        let text = "a--b";
167        let merged = coalesce(text, vec![0..1, 3..4]);
168        assert_eq!(merged, vec![0..4]);
169        let text = "a---b";
170        let merged = coalesce(text, vec![0..1, 4..5]);
171        assert_eq!(merged, vec![0..1, 4..5]);
172    }
173
174    #[test]
175    fn combining_characters_stay_whole() {
176        // "e\u{301}" is one grapheme inside a word token; emphasis must
177        // cover it atomically
178        let old_line = "drink cafe daily";
179        let new_line = "drink cafe\u{301} daily";
180        let (_, new) = intraline(old_line, new_line);
181        for r in &new {
182            assert!(new_line.is_char_boundary(r.start), "range splits a char");
183            assert!(new_line.is_char_boundary(r.end), "range splits a char");
184        }
185        let joined: String = new.iter().map(|r| &new_line[r.clone()]).collect();
186        assert!(joined.contains('\u{301}'), "emphasis: {new:?}");
187    }
188
189    #[test]
190    fn empty_inputs() {
191        let (old, new) = intraline("", "");
192        assert!(old.is_empty());
193        assert!(new.is_empty());
194    }
195
196    #[test]
197    fn embedded_carriage_returns_keep_offsets_global() {
198        // a lone `\r` splits the text into segments internally; emphasis
199        // offsets must still address the whole line
200        let old = "alpha\rfoo bar baz";
201        let new = "alpha\rfoo QUX baz";
202        let (old_r, new_r) = intraline(old, new);
203        let covered: String = old_r.iter().map(|r| &old[r.clone()]).collect();
204        assert_eq!(covered, "bar", "{old_r:?}");
205        let covered: String = new_r.iter().map(|r| &new[r.clone()]).collect();
206        assert_eq!(covered, "QUX", "{new_r:?}");
207
208        // multibyte text before the `\r` must not desync byte offsets
209        let old = "héé\rfoo bar baz";
210        let new = "héé\rfoo QUX baz";
211        let (old_r, new_r) = intraline(old, new);
212        for r in old_r.iter().chain(&new_r) {
213            assert!(old.is_char_boundary(r.start) && old.is_char_boundary(r.end));
214        }
215        let covered: String = new_r.iter().map(|r| &new[r.clone()]).collect();
216        assert_eq!(covered, "QUX", "{new_r:?}");
217
218        // edits in two segments emphasize each in place, in order
219        let old = "aa bb cc\rdd ee ff";
220        let new = "aa XX cc\rdd YY ff";
221        let (_, new_r) = intraline(old, new);
222        let covered: Vec<&str> = new_r.iter().map(|r| &new[r.clone()]).collect();
223        assert_eq!(covered, ["XX", "YY"], "{new_r:?}");
224    }
225}