Skip to main content

diffler_core/
model.rs

1//! Diff model: what changed, organized as files -> hunks -> lines.
2
3use std::ops::Range;
4
5use serde::{Deserialize, Serialize};
6
7/// FNV-1a 64 as lowercase hex. Content hashes key persisted viewed marks and
8/// derived caches, so the algorithm is pinned forever (tested below).
9fn stable_hash(bytes: &[u8]) -> String {
10    let mut hash: u64 = 0xcbf2_9ce4_8422_2325;
11    for byte in bytes {
12        hash ^= u64::from(*byte);
13        hash = hash.wrapping_mul(0x0100_0000_01b3);
14    }
15    format!("{hash:016x}")
16}
17
18#[derive(Debug, Clone, Default, PartialEq, Eq)]
19pub struct DiffModel {
20    pub files: Vec<FileDiff>,
21}
22
23impl DiffModel {
24    /// Cheap content identity of the whole model (file paths + per-file
25    /// sides hashes), so callers can skip invalidating derived state when
26    /// a refresh recomputed an identical diff.
27    pub fn fingerprint(&self) -> String {
28        let mut buf = Vec::new();
29        for file in &self.files {
30            buf.extend_from_slice(file.path.as_bytes());
31            buf.push(0);
32            buf.extend_from_slice(file.sides_hash().as_bytes());
33            buf.push(b'\n');
34        }
35        stable_hash(&buf)
36    }
37
38    /// The diff line carrying number `line` on the requested side of
39    /// `file`'s hunks, if it is part of the diff.
40    pub fn find_line(&self, file: &str, line: u32, on_old_side: bool) -> Option<&DiffLine> {
41        let file = self.files.iter().find(|f| f.path == file)?;
42        file.hunks.iter().flat_map(|h| &h.lines).find(|l| {
43            let no = if on_old_side { l.old_no } else { l.new_no };
44            no == Some(line)
45        })
46    }
47}
48
49#[derive(Debug, Clone, PartialEq, Eq)]
50pub struct FileDiff {
51    pub path: String,
52    pub old_path: Option<String>,
53    pub status: FileStatus,
54    pub binary: bool,
55    /// Full contents of each side, used for whole-file syntax highlighting.
56    /// `None` for binary files and for the missing side of adds/deletes.
57    pub old_text: Option<String>,
58    pub new_text: Option<String>,
59    pub hunks: Vec<Hunk>,
60    /// Lazily memoized content hashes; texts never change after construction.
61    pub hashes: HashCache,
62}
63
64/// Memo slots for [`FileDiff::content_hash`]/[`FileDiff::sides_hash`], which
65/// the UI probes every frame: hashing full file contents per frame is the
66/// cost this avoids. Compares equal always so `FileDiff` equality is on data.
67#[derive(Debug, Clone, Default)]
68pub struct HashCache {
69    content: std::sync::OnceLock<String>,
70    sides: std::sync::OnceLock<String>,
71}
72
73impl PartialEq for HashCache {
74    fn eq(&self, _: &Self) -> bool {
75        true
76    }
77}
78
79impl Eq for HashCache {}
80
81impl FileDiff {
82    /// Content identity of the new side, used for viewed-mark invalidation.
83    pub fn content_hash(&self) -> String {
84        self.hashes
85            .content
86            .get_or_init(|| stable_hash(self.new_text.as_deref().unwrap_or("").as_bytes()))
87            .clone()
88    }
89
90    /// `(added, deleted)` line counts across the file's hunks.
91    pub fn diffstat(&self) -> (usize, usize) {
92        let mut added = 0;
93        let mut deleted = 0;
94        for line in self.hunks.iter().flat_map(|h| &h.lines) {
95            match line.kind {
96                LineKind::Added => added += 1,
97                LineKind::Deleted => deleted += 1,
98                LineKind::Context => {}
99            }
100        }
101        (added, deleted)
102    }
103
104    /// Content identity of both sides, for caches derived from old and new
105    /// text (e.g. syntax highlighting). Viewed marks key on `content_hash`
106    /// instead: they only care about the side the reviewer reads.
107    pub fn sides_hash(&self) -> String {
108        self.hashes
109            .sides
110            .get_or_init(|| {
111                let mut bytes = Vec::from(self.old_text.as_deref().unwrap_or("").as_bytes());
112                bytes.push(0);
113                bytes.extend_from_slice(self.new_text.as_deref().unwrap_or("").as_bytes());
114                stable_hash(&bytes)
115            })
116            .clone()
117    }
118}
119
120#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
121#[serde(rename_all = "snake_case")]
122pub enum FileStatus {
123    Added,
124    Modified,
125    Deleted,
126    Renamed,
127    Untracked,
128    /// A walkthrough's own file: one a stop or note anchors outside the diff,
129    /// shown at its current content with nothing to compare it against.
130    Unchanged,
131}
132
133impl FileStatus {
134    /// Single-character shape naming the status in the diff sidebar. It reads
135    /// by form alone, so a palette whose hues a reader cannot separate still
136    /// carries the status; colour reinforces it.
137    pub const fn glyph(self) -> char {
138        match self {
139            Self::Added => '+',
140            Self::Modified => '●',
141            Self::Deleted => '−',
142            Self::Renamed => '~',
143            Self::Untracked => '○',
144            Self::Unchanged => '·',
145        }
146    }
147
148    /// Neogit-style row label shown in file headers and the diff pane.
149    pub const fn label(self) -> &'static str {
150        match self {
151            Self::Added => "new file",
152            Self::Modified => "modified",
153            Self::Deleted => "deleted",
154            Self::Renamed => "renamed",
155            Self::Untracked => "untracked",
156            Self::Unchanged => "unchanged",
157        }
158    }
159}
160
161/// Stable identity for a hunk: hash of its normalized content. Survives
162/// edits elsewhere in the file; changes when the hunk's lines change.
163#[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
164pub struct HunkId(pub String);
165
166#[derive(Debug, Clone, PartialEq, Eq)]
167pub struct Hunk {
168    pub id: HunkId,
169    pub old_start: u32,
170    pub old_lines: u32,
171    pub new_start: u32,
172    pub new_lines: u32,
173    /// git's section heading: the enclosing function/section name git emits
174    /// after the second `@@` of the hunk header. Empty when git gives none
175    /// (e.g. a top-of-file hunk). Excluded from `id`, which keys only on lines.
176    pub context: String,
177    pub lines: Vec<DiffLine>,
178}
179
180impl Hunk {
181    pub fn header(&self) -> String {
182        format!(
183            "@@ -{},{} +{},{} @@",
184            self.old_start, self.old_lines, self.new_start, self.new_lines
185        )
186    }
187}
188
189#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
190#[serde(rename_all = "snake_case")]
191pub enum LineKind {
192    Context,
193    Deleted,
194    Added,
195}
196
197impl LineKind {
198    /// Unified-diff origin character (' ', '-', '+').
199    pub const fn origin(self) -> char {
200        match self {
201            Self::Context => ' ',
202            Self::Deleted => '-',
203            Self::Added => '+',
204        }
205    }
206}
207
208#[derive(Debug, Clone, PartialEq, Eq)]
209pub struct DiffLine {
210    pub kind: LineKind,
211    pub old_no: Option<u32>,
212    pub new_no: Option<u32>,
213    /// Line content without the trailing newline.
214    pub text: String,
215    /// Byte ranges within `text` to emphasize (intra-line changes).
216    pub emphasis: Vec<Range<usize>>,
217}
218
219impl DiffLine {
220    pub fn new(kind: LineKind, old_no: Option<u32>, new_no: Option<u32>, text: String) -> Self {
221        Self {
222            kind,
223            old_no,
224            new_no,
225            text,
226            emphasis: Vec::new(),
227        }
228    }
229}
230
231/// Hash the hunk's content (kinds + text) into a stable id. `occurrence` is
232/// how many earlier hunks in the same file hash to the same content: two
233/// hunks with byte-identical lines in one file get distinct ids this way,
234/// while every other hunk's id depends on its own content alone, so it
235/// survives edits elsewhere in the file. [`disambiguated_hunk_id`] is the
236/// usual way to call this, since it tracks the count for the caller.
237pub fn hunk_id(file_path: &str, lines: &[DiffLine], occurrence: usize) -> HunkId {
238    let mut buf = String::new();
239    buf.push_str(file_path);
240    buf.push('\n');
241    for line in lines {
242        let tag = match line.kind {
243            LineKind::Context => ' ',
244            LineKind::Deleted => '-',
245            LineKind::Added => '+',
246        };
247        buf.push(tag);
248        buf.push_str(&line.text);
249        buf.push('\n');
250    }
251    if occurrence > 0 {
252        buf.push_str(&occurrence.to_string());
253        buf.push('\n');
254    }
255    HunkId(stable_hash(buf.as_bytes()))
256}
257
258/// [`hunk_id`] for one hunk of a file whose other hunks are being assigned
259/// ids through the same `seen` map: each distinct content gets occurrence 0
260/// the first time and counts up from there, so two identical hunks in one
261/// file never collide.
262// every caller shares one default-hashed map for one file's hunks, so a
263// generic hasher buys nothing
264#[allow(clippy::implicit_hasher)]
265pub fn disambiguated_hunk_id(
266    file_path: &str,
267    lines: &[DiffLine],
268    seen: &mut std::collections::HashMap<HunkId, usize>,
269) -> HunkId {
270    let base = hunk_id(file_path, lines, 0);
271    let occurrence = seen.entry(base).or_insert(0);
272    let id = hunk_id(file_path, lines, *occurrence);
273    *occurrence += 1;
274    id
275}
276
277#[cfg(test)]
278mod tests {
279    use super::*;
280
281    // hashes key persisted viewed marks: the algorithm must stay stable
282    // across releases, so pin known FNV-1a 64 values
283    #[test]
284    fn stable_hash_is_fnv1a64_and_never_changes() {
285        assert_eq!(stable_hash(b""), "cbf29ce484222325");
286        assert_eq!(stable_hash(b"hello"), "a430d84680aabd0b");
287    }
288
289    #[test]
290    fn file_status_glyph_and_label_cover_all_variants() {
291        let glyphs = [
292            FileStatus::Added,
293            FileStatus::Modified,
294            FileStatus::Deleted,
295            FileStatus::Renamed,
296            FileStatus::Untracked,
297            FileStatus::Unchanged,
298        ]
299        .map(FileStatus::glyph);
300        assert_eq!(glyphs, ['+', '●', '−', '~', '○', '·']);
301        let mut distinct = glyphs.to_vec();
302        distinct.sort_unstable();
303        distinct.dedup();
304        assert_eq!(
305            distinct.len(),
306            glyphs.len(),
307            "no two statuses share a shape"
308        );
309
310        assert_eq!(FileStatus::Added.label(), "new file");
311        assert_eq!(FileStatus::Modified.label(), "modified");
312        assert_eq!(FileStatus::Deleted.label(), "deleted");
313        assert_eq!(FileStatus::Renamed.label(), "renamed");
314        assert_eq!(FileStatus::Untracked.label(), "untracked");
315        assert_eq!(FileStatus::Unchanged.label(), "unchanged");
316    }
317
318    fn line(kind: LineKind, text: &str) -> DiffLine {
319        DiffLine::new(kind, None, None, text.to_owned())
320    }
321
322    #[test]
323    fn hunk_id_is_stable() {
324        let lines = vec![line(LineKind::Deleted, "a"), line(LineKind::Added, "b")];
325        let id1 = hunk_id("src/x.rs", &lines, 0);
326        let id2 = hunk_id("src/x.rs", &lines, 0);
327        assert_eq!(id1, id2);
328    }
329
330    #[test]
331    fn hunk_id_changes_with_content() {
332        let a = vec![line(LineKind::Added, "x")];
333        let b = vec![line(LineKind::Added, "y")];
334        assert_ne!(hunk_id("f", &a, 0), hunk_id("f", &b, 0));
335    }
336
337    #[test]
338    fn hunk_id_changes_with_kind() {
339        let a = vec![line(LineKind::Added, "x")];
340        let b = vec![line(LineKind::Deleted, "x")];
341        assert_ne!(hunk_id("f", &a, 0), hunk_id("f", &b, 0));
342    }
343
344    #[test]
345    fn hunk_id_changes_with_file() {
346        let lines = vec![line(LineKind::Added, "x")];
347        assert_ne!(hunk_id("a", &lines, 0), hunk_id("b", &lines, 0));
348    }
349
350    #[test]
351    fn hunk_id_changes_with_occurrence() {
352        let lines = vec![line(LineKind::Added, "x")];
353        assert_ne!(hunk_id("f", &lines, 0), hunk_id("f", &lines, 1));
354    }
355
356    #[test]
357    fn disambiguated_hunk_id_gives_identical_hunks_distinct_ids() {
358        let lines = vec![line(LineKind::Added, "x")];
359        let mut seen = std::collections::HashMap::new();
360        let first = disambiguated_hunk_id("f", &lines, &mut seen);
361        let second = disambiguated_hunk_id("f", &lines, &mut seen);
362        assert_ne!(first, second);
363        assert_eq!(first, hunk_id("f", &lines, 0));
364        assert_eq!(second, hunk_id("f", &lines, 1));
365    }
366
367    #[test]
368    fn header_formats() {
369        let hunk = Hunk {
370            id: HunkId("h".into()),
371            old_start: 10,
372            old_lines: 7,
373            new_start: 10,
374            new_lines: 9,
375            context: String::new(),
376            lines: vec![],
377        };
378        assert_eq!(hunk.header(), "@@ -10,7 +10,9 @@");
379    }
380
381    #[test]
382    fn content_hash_changes_when_new_text_changes() {
383        let base = FileDiff {
384            path: "f.rs".into(),
385            old_path: None,
386            status: FileStatus::Modified,
387            binary: false,
388            old_text: None,
389            new_text: Some("fn main() {}".into()),
390            hunks: vec![],
391            hashes: HashCache::default(),
392        };
393        let mut changed = base.clone();
394        changed.new_text = Some("fn main() { let x = 1; }".into());
395        assert_ne!(base.content_hash(), changed.content_hash());
396    }
397
398    #[test]
399    fn content_hash_is_stable() {
400        let file = FileDiff {
401            path: "f.rs".into(),
402            old_path: None,
403            status: FileStatus::Modified,
404            binary: false,
405            old_text: None,
406            new_text: Some("same content".into()),
407            hunks: vec![],
408            hashes: HashCache::default(),
409        };
410        assert_eq!(file.content_hash(), file.content_hash());
411    }
412
413    #[test]
414    fn sides_hash_changes_when_old_text_changes() {
415        let base = FileDiff {
416            path: "f.rs".into(),
417            old_path: None,
418            status: FileStatus::Modified,
419            binary: false,
420            old_text: Some("fn main() {}".into()),
421            new_text: Some("fn main() { let x = 1; }".into()),
422            hunks: vec![],
423            hashes: HashCache::default(),
424        };
425        let mut changed = base.clone();
426        changed.old_text = Some("fn main() { unreachable!() }".into());
427        assert_eq!(
428            base.content_hash(),
429            changed.content_hash(),
430            "same new side, same content hash"
431        );
432        assert_ne!(base.sides_hash(), changed.sides_hash());
433    }
434
435    fn one_file_model(path: &str, old_text: &str, new_text: &str) -> DiffModel {
436        DiffModel {
437            files: vec![FileDiff {
438                path: path.to_owned(),
439                old_path: None,
440                status: FileStatus::Modified,
441                binary: false,
442                old_text: Some(old_text.to_owned()),
443                new_text: Some(new_text.to_owned()),
444                hunks: vec![],
445                hashes: HashCache::default(),
446            }],
447        }
448    }
449
450    #[test]
451    fn fingerprint_is_stable_for_identical_models() {
452        let a = one_file_model("f.rs", "old", "new");
453        let b = one_file_model("f.rs", "old", "new");
454        assert_eq!(a.fingerprint(), b.fingerprint());
455    }
456
457    #[test]
458    fn fingerprint_changes_with_content_path_and_file_set() {
459        let base = one_file_model("f.rs", "old", "new");
460        assert_ne!(
461            base.fingerprint(),
462            one_file_model("f.rs", "old", "newer").fingerprint(),
463            "changed side changes the fingerprint"
464        );
465        assert_ne!(
466            base.fingerprint(),
467            one_file_model("g.rs", "old", "new").fingerprint(),
468            "renamed file changes the fingerprint"
469        );
470        let mut grown = base.clone();
471        grown.files.extend(one_file_model("g.rs", "", "x").files);
472        assert_ne!(
473            base.fingerprint(),
474            grown.fingerprint(),
475            "added file changes the fingerprint"
476        );
477    }
478
479    fn model_with_lines() -> DiffModel {
480        DiffModel {
481            files: vec![FileDiff {
482                path: "f.rs".into(),
483                old_path: None,
484                status: FileStatus::Modified,
485                binary: false,
486                old_text: None,
487                new_text: None,
488                hunks: vec![Hunk {
489                    id: HunkId("h".into()),
490                    old_start: 1,
491                    old_lines: 2,
492                    new_start: 1,
493                    new_lines: 2,
494                    context: String::new(),
495                    lines: vec![
496                        DiffLine::new(LineKind::Context, Some(1), Some(1), "one".into()),
497                        DiffLine::new(LineKind::Deleted, Some(2), None, "two".into()),
498                        DiffLine::new(LineKind::Added, None, Some(2), "TWO".into()),
499                    ],
500                }],
501                hashes: HashCache::default(),
502            }],
503        }
504    }
505
506    #[test]
507    fn diffstat_counts_added_and_deleted_over_hunks() {
508        // model_with_lines: one context, one deleted, one added line
509        let model = model_with_lines();
510        assert_eq!(model.files[0].diffstat(), (1, 1));
511
512        // two hunks: 2 added + 1 deleted total, context ignored
513        let file = FileDiff {
514            path: "f.rs".into(),
515            old_path: None,
516            status: FileStatus::Modified,
517            binary: false,
518            old_text: None,
519            new_text: None,
520            hunks: vec![
521                Hunk {
522                    id: HunkId("a".into()),
523                    old_start: 1,
524                    old_lines: 1,
525                    new_start: 1,
526                    new_lines: 2,
527                    context: String::new(),
528                    lines: vec![
529                        DiffLine::new(LineKind::Context, Some(1), Some(1), "ctx".into()),
530                        DiffLine::new(LineKind::Added, None, Some(2), "add one".into()),
531                    ],
532                },
533                Hunk {
534                    id: HunkId("b".into()),
535                    old_start: 5,
536                    old_lines: 1,
537                    new_start: 6,
538                    new_lines: 1,
539                    context: String::new(),
540                    lines: vec![
541                        DiffLine::new(LineKind::Deleted, Some(5), None, "gone".into()),
542                        DiffLine::new(LineKind::Added, None, Some(6), "add two".into()),
543                    ],
544                },
545            ],
546            hashes: HashCache::default(),
547        };
548        assert_eq!(file.diffstat(), (2, 1));
549    }
550
551    #[test]
552    fn find_line_matches_the_requested_side() {
553        let model = model_with_lines();
554        let new_side = model.find_line("f.rs", 2, false).expect("new side");
555        assert_eq!(new_side.text, "TWO");
556        let old_side = model.find_line("f.rs", 2, true).expect("old side");
557        assert_eq!(old_side.text, "two");
558    }
559
560    #[test]
561    fn find_line_misses_unknown_files_and_lines() {
562        let model = model_with_lines();
563        assert!(model.find_line("nope.rs", 1, false).is_none());
564        assert!(model.find_line("f.rs", 99, false).is_none());
565    }
566
567    #[test]
568    fn content_hash_falls_back_for_none() {
569        let file = FileDiff {
570            path: "f.rs".into(),
571            old_path: None,
572            status: FileStatus::Deleted,
573            binary: false,
574            old_text: None,
575            new_text: None,
576            hunks: vec![],
577            hashes: HashCache::default(),
578        };
579        // must not panic, must return a non-empty string (git hash of empty blob)
580        let hash = file.content_hash();
581        assert!(!hash.is_empty());
582    }
583}