Skip to main content

diffler_core/
model.rs

1//! Diff model: what changed, organized as files -> hunks -> lines.
2
3use std::ops::Range;
4
5use serde::{Deserialize, Serialize};
6
7/// FNV-1a 64 as lowercase hex. Content hashes key persisted viewed marks and
8/// derived caches, so the algorithm is pinned forever (tested below).
9fn stable_hash(bytes: &[u8]) -> String {
10    let mut hash: u64 = 0xcbf2_9ce4_8422_2325;
11    for byte in bytes {
12        hash ^= u64::from(*byte);
13        hash = hash.wrapping_mul(0x0100_0000_01b3);
14    }
15    format!("{hash:016x}")
16}
17
18#[derive(Debug, Clone, Default, PartialEq, Eq)]
19pub struct DiffModel {
20    pub files: Vec<FileDiff>,
21}
22
23impl DiffModel {
24    /// Cheap content identity of the whole model (file paths + per-file
25    /// sides hashes), so callers can skip invalidating derived state when
26    /// a refresh recomputed an identical diff.
27    pub fn fingerprint(&self) -> String {
28        let mut buf = Vec::new();
29        for file in &self.files {
30            buf.extend_from_slice(file.path.as_bytes());
31            buf.push(0);
32            buf.extend_from_slice(file.sides_hash().as_bytes());
33            buf.push(b'\n');
34        }
35        stable_hash(&buf)
36    }
37
38    /// The diff line carrying number `line` on the requested side of
39    /// `file`'s hunks, if it is part of the diff.
40    pub fn find_line(&self, file: &str, line: u32, on_old_side: bool) -> Option<&DiffLine> {
41        let file = self.files.iter().find(|f| f.path == file)?;
42        file.hunks.iter().flat_map(|h| &h.lines).find(|l| {
43            let no = if on_old_side { l.old_no } else { l.new_no };
44            no == Some(line)
45        })
46    }
47}
48
49#[derive(Debug, Clone, PartialEq, Eq)]
50pub struct FileDiff {
51    pub path: String,
52    pub old_path: Option<String>,
53    pub status: FileStatus,
54    pub binary: bool,
55    /// Full contents of each side, used for whole-file syntax highlighting.
56    /// `None` for binary files and for the missing side of adds/deletes.
57    pub old_text: Option<String>,
58    pub new_text: Option<String>,
59    pub hunks: Vec<Hunk>,
60    /// Lazily memoized content hashes; texts never change after construction.
61    pub hashes: HashCache,
62}
63
64/// Memo slots for [`FileDiff::content_hash`]/[`FileDiff::sides_hash`], which
65/// the UI probes every frame: hashing full file contents per frame is the
66/// cost this avoids. Compares equal always so `FileDiff` equality is on data.
67#[derive(Debug, Clone, Default)]
68pub struct HashCache {
69    content: std::sync::OnceLock<String>,
70    sides: std::sync::OnceLock<String>,
71}
72
73impl PartialEq for HashCache {
74    fn eq(&self, _: &Self) -> bool {
75        true
76    }
77}
78
79impl Eq for HashCache {}
80
81impl FileDiff {
82    /// Content identity of the new side, used for viewed-mark invalidation.
83    pub fn content_hash(&self) -> String {
84        self.hashes
85            .content
86            .get_or_init(|| stable_hash(self.new_text.as_deref().unwrap_or("").as_bytes()))
87            .clone()
88    }
89
90    /// `(added, deleted)` line counts across the file's hunks.
91    pub fn diffstat(&self) -> (usize, usize) {
92        let mut added = 0;
93        let mut deleted = 0;
94        for line in self.hunks.iter().flat_map(|h| &h.lines) {
95            match line.kind {
96                LineKind::Added => added += 1,
97                LineKind::Deleted => deleted += 1,
98                LineKind::Context => {}
99            }
100        }
101        (added, deleted)
102    }
103
104    /// Content identity of both sides, for caches derived from old and new
105    /// text (e.g. syntax highlighting). Viewed marks key on `content_hash`
106    /// instead: they only care about the side the reviewer reads.
107    pub fn sides_hash(&self) -> String {
108        self.hashes
109            .sides
110            .get_or_init(|| {
111                let mut bytes = Vec::from(self.old_text.as_deref().unwrap_or("").as_bytes());
112                bytes.push(0);
113                bytes.extend_from_slice(self.new_text.as_deref().unwrap_or("").as_bytes());
114                stable_hash(&bytes)
115            })
116            .clone()
117    }
118}
119
120#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
121#[serde(rename_all = "snake_case")]
122pub enum FileStatus {
123    Added,
124    Modified,
125    Deleted,
126    Renamed,
127    Untracked,
128    /// A walkthrough's own file: one a stop or note anchors outside the diff,
129    /// shown at its current content with nothing to compare it against.
130    Unchanged,
131}
132
133impl FileStatus {
134    /// Single-character shape naming the status in the diff sidebar. It reads
135    /// by form alone, so a palette whose hues a reader cannot separate still
136    /// carries the status; colour reinforces it.
137    pub const fn glyph(self) -> char {
138        match self {
139            Self::Added => '+',
140            Self::Modified => '●',
141            Self::Deleted => '−',
142            Self::Renamed => '~',
143            Self::Untracked => '○',
144            Self::Unchanged => '·',
145        }
146    }
147
148    /// Neogit-style row label shown in file headers and the diff pane.
149    pub const fn label(self) -> &'static str {
150        match self {
151            Self::Added => "new file",
152            Self::Modified => "modified",
153            Self::Deleted => "deleted",
154            Self::Renamed => "renamed",
155            Self::Untracked => "untracked",
156            Self::Unchanged => "unchanged",
157        }
158    }
159}
160
161/// Stable identity for a hunk: hash of its normalized content. Survives
162/// edits elsewhere in the file; changes when the hunk's lines change.
163#[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
164pub struct HunkId(pub String);
165
166#[derive(Debug, Clone, PartialEq, Eq)]
167pub struct Hunk {
168    pub id: HunkId,
169    pub old_start: u32,
170    pub old_lines: u32,
171    pub new_start: u32,
172    pub new_lines: u32,
173    /// git's section heading: the enclosing function/section name git emits
174    /// after the second `@@` of the hunk header. Empty when git gives none
175    /// (e.g. a top-of-file hunk). Excluded from `id`, which keys only on lines.
176    pub context: String,
177    pub lines: Vec<DiffLine>,
178}
179
180impl Hunk {
181    pub fn header(&self) -> String {
182        format!(
183            "@@ -{},{} +{},{} @@",
184            self.old_start, self.old_lines, self.new_start, self.new_lines
185        )
186    }
187}
188
189#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
190#[serde(rename_all = "snake_case")]
191pub enum LineKind {
192    Context,
193    Deleted,
194    Added,
195}
196
197impl LineKind {
198    /// Unified-diff origin character (' ', '-', '+').
199    pub const fn origin(self) -> char {
200        match self {
201            Self::Context => ' ',
202            Self::Deleted => '-',
203            Self::Added => '+',
204        }
205    }
206}
207
208#[derive(Debug, Clone, PartialEq, Eq)]
209pub struct DiffLine {
210    pub kind: LineKind,
211    pub old_no: Option<u32>,
212    pub new_no: Option<u32>,
213    /// Line content without the trailing newline.
214    pub text: String,
215    /// Byte ranges within `text` to emphasize (intra-line changes).
216    pub emphasis: Vec<Range<usize>>,
217    /// True on a paired deleted/added line the structural algorithm found to
218    /// be a pure reformat (identical token structure, e.g. reindentation):
219    /// the renderer dims it, leaving red/green for an actual change.
220    pub reformat_only: bool,
221}
222
223impl DiffLine {
224    pub fn new(kind: LineKind, old_no: Option<u32>, new_no: Option<u32>, text: String) -> Self {
225        Self {
226            kind,
227            old_no,
228            new_no,
229            text,
230            emphasis: Vec::new(),
231            reformat_only: false,
232        }
233    }
234
235    /// The line number on the named side: `old_no` on the old side (where a
236    /// deletion lives), `new_no` everywhere else.
237    pub const fn number_on(&self, old_side: bool) -> Option<u32> {
238        if old_side { self.old_no } else { self.new_no }
239    }
240}
241
242/// Hash the hunk's content (kinds + text) into a stable id. `occurrence` is
243/// how many earlier hunks in the same file hash to the same content: two
244/// hunks with byte-identical lines in one file get distinct ids this way,
245/// while every other hunk's id depends on its own content alone, so it
246/// survives edits elsewhere in the file. [`disambiguated_hunk_id`] is the
247/// usual way to call this, since it tracks the count for the caller.
248pub fn hunk_id(file_path: &str, lines: &[DiffLine], occurrence: usize) -> HunkId {
249    let mut buf = String::new();
250    buf.push_str(file_path);
251    buf.push('\n');
252    for line in lines {
253        let tag = match line.kind {
254            LineKind::Context => ' ',
255            LineKind::Deleted => '-',
256            LineKind::Added => '+',
257        };
258        buf.push(tag);
259        buf.push_str(&line.text);
260        buf.push('\n');
261    }
262    if occurrence > 0 {
263        buf.push_str(&occurrence.to_string());
264        buf.push('\n');
265    }
266    HunkId(stable_hash(buf.as_bytes()))
267}
268
269/// [`hunk_id`] for one hunk of a file whose other hunks are being assigned
270/// ids through the same `seen` map: each distinct content gets occurrence 0
271/// the first time and counts up from there, so two identical hunks in one
272/// file never collide.
273// every caller shares one default-hashed map for one file's hunks, so a
274// generic hasher buys nothing
275#[allow(clippy::implicit_hasher)]
276pub fn disambiguated_hunk_id(
277    file_path: &str,
278    lines: &[DiffLine],
279    seen: &mut std::collections::HashMap<HunkId, usize>,
280) -> HunkId {
281    let base = hunk_id(file_path, lines, 0);
282    let occurrence = seen.entry(base).or_insert(0);
283    let id = hunk_id(file_path, lines, *occurrence);
284    *occurrence += 1;
285    id
286}
287
288#[cfg(test)]
289mod tests {
290    use super::*;
291
292    // hashes key persisted viewed marks: the algorithm must stay stable
293    // across releases, so pin known FNV-1a 64 values
294    #[test]
295    fn stable_hash_is_fnv1a64_and_never_changes() {
296        assert_eq!(stable_hash(b""), "cbf29ce484222325");
297        assert_eq!(stable_hash(b"hello"), "a430d84680aabd0b");
298    }
299
300    #[test]
301    fn file_status_glyph_and_label_cover_all_variants() {
302        let glyphs = [
303            FileStatus::Added,
304            FileStatus::Modified,
305            FileStatus::Deleted,
306            FileStatus::Renamed,
307            FileStatus::Untracked,
308            FileStatus::Unchanged,
309        ]
310        .map(FileStatus::glyph);
311        assert_eq!(glyphs, ['+', '●', '−', '~', '○', '·']);
312        let mut distinct = glyphs.to_vec();
313        distinct.sort_unstable();
314        distinct.dedup();
315        assert_eq!(
316            distinct.len(),
317            glyphs.len(),
318            "no two statuses share a shape"
319        );
320
321        assert_eq!(FileStatus::Added.label(), "new file");
322        assert_eq!(FileStatus::Modified.label(), "modified");
323        assert_eq!(FileStatus::Deleted.label(), "deleted");
324        assert_eq!(FileStatus::Renamed.label(), "renamed");
325        assert_eq!(FileStatus::Untracked.label(), "untracked");
326        assert_eq!(FileStatus::Unchanged.label(), "unchanged");
327    }
328
329    fn line(kind: LineKind, text: &str) -> DiffLine {
330        DiffLine::new(kind, None, None, text.to_owned())
331    }
332
333    #[test]
334    fn hunk_id_is_stable() {
335        let lines = vec![line(LineKind::Deleted, "a"), line(LineKind::Added, "b")];
336        let id1 = hunk_id("src/x.rs", &lines, 0);
337        let id2 = hunk_id("src/x.rs", &lines, 0);
338        assert_eq!(id1, id2);
339    }
340
341    #[test]
342    fn hunk_id_changes_with_content() {
343        let a = vec![line(LineKind::Added, "x")];
344        let b = vec![line(LineKind::Added, "y")];
345        assert_ne!(hunk_id("f", &a, 0), hunk_id("f", &b, 0));
346    }
347
348    #[test]
349    fn hunk_id_changes_with_kind() {
350        let a = vec![line(LineKind::Added, "x")];
351        let b = vec![line(LineKind::Deleted, "x")];
352        assert_ne!(hunk_id("f", &a, 0), hunk_id("f", &b, 0));
353    }
354
355    #[test]
356    fn hunk_id_changes_with_file() {
357        let lines = vec![line(LineKind::Added, "x")];
358        assert_ne!(hunk_id("a", &lines, 0), hunk_id("b", &lines, 0));
359    }
360
361    #[test]
362    fn hunk_id_changes_with_occurrence() {
363        let lines = vec![line(LineKind::Added, "x")];
364        assert_ne!(hunk_id("f", &lines, 0), hunk_id("f", &lines, 1));
365    }
366
367    #[test]
368    fn disambiguated_hunk_id_gives_identical_hunks_distinct_ids() {
369        let lines = vec![line(LineKind::Added, "x")];
370        let mut seen = std::collections::HashMap::new();
371        let first = disambiguated_hunk_id("f", &lines, &mut seen);
372        let second = disambiguated_hunk_id("f", &lines, &mut seen);
373        assert_ne!(first, second);
374        assert_eq!(first, hunk_id("f", &lines, 0));
375        assert_eq!(second, hunk_id("f", &lines, 1));
376    }
377
378    #[test]
379    fn header_formats() {
380        let hunk = Hunk {
381            id: HunkId("h".into()),
382            old_start: 10,
383            old_lines: 7,
384            new_start: 10,
385            new_lines: 9,
386            context: String::new(),
387            lines: vec![],
388        };
389        assert_eq!(hunk.header(), "@@ -10,7 +10,9 @@");
390    }
391
392    #[test]
393    fn content_hash_changes_when_new_text_changes() {
394        let base = FileDiff {
395            path: "f.rs".into(),
396            old_path: None,
397            status: FileStatus::Modified,
398            binary: false,
399            old_text: None,
400            new_text: Some("fn main() {}".into()),
401            hunks: vec![],
402            hashes: HashCache::default(),
403        };
404        let mut changed = base.clone();
405        changed.new_text = Some("fn main() { let x = 1; }".into());
406        assert_ne!(base.content_hash(), changed.content_hash());
407    }
408
409    #[test]
410    fn content_hash_is_stable() {
411        let file = FileDiff {
412            path: "f.rs".into(),
413            old_path: None,
414            status: FileStatus::Modified,
415            binary: false,
416            old_text: None,
417            new_text: Some("same content".into()),
418            hunks: vec![],
419            hashes: HashCache::default(),
420        };
421        assert_eq!(file.content_hash(), file.content_hash());
422    }
423
424    #[test]
425    fn sides_hash_changes_when_old_text_changes() {
426        let base = FileDiff {
427            path: "f.rs".into(),
428            old_path: None,
429            status: FileStatus::Modified,
430            binary: false,
431            old_text: Some("fn main() {}".into()),
432            new_text: Some("fn main() { let x = 1; }".into()),
433            hunks: vec![],
434            hashes: HashCache::default(),
435        };
436        let mut changed = base.clone();
437        changed.old_text = Some("fn main() { unreachable!() }".into());
438        assert_eq!(
439            base.content_hash(),
440            changed.content_hash(),
441            "same new side, same content hash"
442        );
443        assert_ne!(base.sides_hash(), changed.sides_hash());
444    }
445
446    fn one_file_model(path: &str, old_text: &str, new_text: &str) -> DiffModel {
447        DiffModel {
448            files: vec![FileDiff {
449                path: path.to_owned(),
450                old_path: None,
451                status: FileStatus::Modified,
452                binary: false,
453                old_text: Some(old_text.to_owned()),
454                new_text: Some(new_text.to_owned()),
455                hunks: vec![],
456                hashes: HashCache::default(),
457            }],
458        }
459    }
460
461    #[test]
462    fn fingerprint_is_stable_for_identical_models() {
463        let a = one_file_model("f.rs", "old", "new");
464        let b = one_file_model("f.rs", "old", "new");
465        assert_eq!(a.fingerprint(), b.fingerprint());
466    }
467
468    #[test]
469    fn fingerprint_changes_with_content_path_and_file_set() {
470        let base = one_file_model("f.rs", "old", "new");
471        assert_ne!(
472            base.fingerprint(),
473            one_file_model("f.rs", "old", "newer").fingerprint(),
474            "changed side changes the fingerprint"
475        );
476        assert_ne!(
477            base.fingerprint(),
478            one_file_model("g.rs", "old", "new").fingerprint(),
479            "renamed file changes the fingerprint"
480        );
481        let mut grown = base.clone();
482        grown.files.extend(one_file_model("g.rs", "", "x").files);
483        assert_ne!(
484            base.fingerprint(),
485            grown.fingerprint(),
486            "added file changes the fingerprint"
487        );
488    }
489
490    fn model_with_lines() -> DiffModel {
491        DiffModel {
492            files: vec![FileDiff {
493                path: "f.rs".into(),
494                old_path: None,
495                status: FileStatus::Modified,
496                binary: false,
497                old_text: None,
498                new_text: None,
499                hunks: vec![Hunk {
500                    id: HunkId("h".into()),
501                    old_start: 1,
502                    old_lines: 2,
503                    new_start: 1,
504                    new_lines: 2,
505                    context: String::new(),
506                    lines: vec![
507                        DiffLine::new(LineKind::Context, Some(1), Some(1), "one".into()),
508                        DiffLine::new(LineKind::Deleted, Some(2), None, "two".into()),
509                        DiffLine::new(LineKind::Added, None, Some(2), "TWO".into()),
510                    ],
511                }],
512                hashes: HashCache::default(),
513            }],
514        }
515    }
516
517    #[test]
518    fn diffstat_counts_added_and_deleted_over_hunks() {
519        // model_with_lines: one context, one deleted, one added line
520        let model = model_with_lines();
521        assert_eq!(model.files[0].diffstat(), (1, 1));
522
523        // two hunks: 2 added + 1 deleted total, context ignored
524        let file = FileDiff {
525            path: "f.rs".into(),
526            old_path: None,
527            status: FileStatus::Modified,
528            binary: false,
529            old_text: None,
530            new_text: None,
531            hunks: vec![
532                Hunk {
533                    id: HunkId("a".into()),
534                    old_start: 1,
535                    old_lines: 1,
536                    new_start: 1,
537                    new_lines: 2,
538                    context: String::new(),
539                    lines: vec![
540                        DiffLine::new(LineKind::Context, Some(1), Some(1), "ctx".into()),
541                        DiffLine::new(LineKind::Added, None, Some(2), "add one".into()),
542                    ],
543                },
544                Hunk {
545                    id: HunkId("b".into()),
546                    old_start: 5,
547                    old_lines: 1,
548                    new_start: 6,
549                    new_lines: 1,
550                    context: String::new(),
551                    lines: vec![
552                        DiffLine::new(LineKind::Deleted, Some(5), None, "gone".into()),
553                        DiffLine::new(LineKind::Added, None, Some(6), "add two".into()),
554                    ],
555                },
556            ],
557            hashes: HashCache::default(),
558        };
559        assert_eq!(file.diffstat(), (2, 1));
560    }
561
562    #[test]
563    fn find_line_matches_the_requested_side() {
564        let model = model_with_lines();
565        let new_side = model.find_line("f.rs", 2, false).expect("new side");
566        assert_eq!(new_side.text, "TWO");
567        let old_side = model.find_line("f.rs", 2, true).expect("old side");
568        assert_eq!(old_side.text, "two");
569    }
570
571    #[test]
572    fn find_line_misses_unknown_files_and_lines() {
573        let model = model_with_lines();
574        assert!(model.find_line("nope.rs", 1, false).is_none());
575        assert!(model.find_line("f.rs", 99, false).is_none());
576    }
577
578    #[test]
579    fn content_hash_falls_back_for_none() {
580        let file = FileDiff {
581            path: "f.rs".into(),
582            old_path: None,
583            status: FileStatus::Deleted,
584            binary: false,
585            old_text: None,
586            new_text: None,
587            hunks: vec![],
588            hashes: HashCache::default(),
589        };
590        // must not panic, must return a non-empty string (git hash of empty blob)
591        let hash = file.content_hash();
592        assert!(!hash.is_empty());
593    }
594}