Skip to main content

diffler_core/
model.rs

1//! Diff model: what changed, organized as files -> hunks -> lines.
2
3use std::ops::Range;
4
5use serde::{Deserialize, Serialize};
6
7/// FNV-1a 64 as lowercase hex. Content hashes key persisted viewed marks and
8/// derived caches, so the algorithm is pinned forever (tested below).
9fn stable_hash(bytes: &[u8]) -> String {
10    let mut hash: u64 = 0xcbf2_9ce4_8422_2325;
11    for byte in bytes {
12        hash ^= u64::from(*byte);
13        hash = hash.wrapping_mul(0x0100_0000_01b3);
14    }
15    format!("{hash:016x}")
16}
17
18#[derive(Debug, Clone, Default, PartialEq, Eq)]
19pub struct DiffModel {
20    pub files: Vec<FileDiff>,
21}
22
23impl DiffModel {
24    /// Cheap content identity of the whole model (file paths + per-file
25    /// sides hashes), so callers can skip invalidating derived state when
26    /// a refresh recomputed an identical diff.
27    pub fn fingerprint(&self) -> String {
28        let mut buf = Vec::new();
29        for file in &self.files {
30            buf.extend_from_slice(file.path.as_bytes());
31            buf.push(0);
32            buf.extend_from_slice(file.sides_hash().as_bytes());
33            buf.push(b'\n');
34        }
35        stable_hash(&buf)
36    }
37
38    /// The diff line carrying number `line` on the requested side of
39    /// `file`'s hunks, if it is part of the diff.
40    pub fn find_line(&self, file: &str, line: u32, on_old_side: bool) -> Option<&DiffLine> {
41        let file = self.files.iter().find(|f| f.path == file)?;
42        file.hunks.iter().flat_map(|h| &h.lines).find(|l| {
43            let no = if on_old_side { l.old_no } else { l.new_no };
44            no == Some(line)
45        })
46    }
47}
48
49#[derive(Debug, Clone, PartialEq, Eq)]
50pub struct FileDiff {
51    pub path: String,
52    pub old_path: Option<String>,
53    pub status: FileStatus,
54    pub binary: bool,
55    /// Full contents of each side, used for whole-file syntax highlighting.
56    /// `None` for binary files and for the missing side of adds/deletes.
57    pub old_text: Option<String>,
58    pub new_text: Option<String>,
59    pub hunks: Vec<Hunk>,
60    /// Lazily memoized content hashes; texts never change after construction.
61    pub hashes: HashCache,
62    /// Each side's git blob, for a binary file the pane reads bytes from (an
63    /// image preview).
64    pub blobs: BlobIds,
65}
66
67/// The git blob ids of a file's two sides, hex-encoded, `None` for a side
68/// that does not exist (an add, a delete). A working-tree side's id is only
69/// a hash of the file on disk, with no object behind it in the store.
70#[derive(Debug, Clone, Default, PartialEq, Eq)]
71pub struct BlobIds {
72    pub old: Option<String>,
73    pub new: Option<String>,
74}
75
76/// Memo slots for [`FileDiff::content_hash`]/[`FileDiff::sides_hash`], which
77/// the UI probes every frame: hashing full file contents per frame is the
78/// cost this avoids. Compares equal always so `FileDiff` equality is on data.
79#[derive(Debug, Clone, Default)]
80pub struct HashCache {
81    content: std::sync::OnceLock<String>,
82    sides: std::sync::OnceLock<String>,
83}
84
85impl PartialEq for HashCache {
86    fn eq(&self, _: &Self) -> bool {
87        true
88    }
89}
90
91impl Eq for HashCache {}
92
93impl FileDiff {
94    /// Content identity of the new side, used for viewed-mark invalidation.
95    /// A binary file has no text, so its new side's git blob id stands in.
96    pub fn content_hash(&self) -> String {
97        self.hashes
98            .content
99            .get_or_init(|| match (&self.new_text, &self.blobs.new) {
100                (None, Some(blob)) => blob.clone(),
101                (text, _) => stable_hash(text.as_deref().unwrap_or("").as_bytes()),
102            })
103            .clone()
104    }
105
106    /// `(added, deleted)` line counts across the file's hunks.
107    pub fn diffstat(&self) -> (usize, usize) {
108        let mut added = 0;
109        let mut deleted = 0;
110        for line in self.hunks.iter().flat_map(|h| &h.lines) {
111            match line.kind {
112                LineKind::Added => added += 1,
113                LineKind::Deleted => deleted += 1,
114                LineKind::Context => {}
115            }
116        }
117        (added, deleted)
118    }
119
120    /// Content identity of both sides, for caches derived from old and new
121    /// text (e.g. syntax highlighting). Viewed marks key on `content_hash`
122    /// instead: they only care about the side the reviewer reads.
123    pub fn sides_hash(&self) -> String {
124        self.hashes
125            .sides
126            .get_or_init(|| {
127                let mut bytes = Vec::from(self.old_text.as_deref().unwrap_or("").as_bytes());
128                bytes.push(0);
129                bytes.extend_from_slice(self.new_text.as_deref().unwrap_or("").as_bytes());
130                // a binary file carries no text, so its blob ids are its identity
131                for blob in [&self.blobs.old, &self.blobs.new] {
132                    bytes.push(0);
133                    bytes.extend_from_slice(blob.as_deref().unwrap_or("").as_bytes());
134                }
135                stable_hash(&bytes)
136            })
137            .clone()
138    }
139}
140
141#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
142#[serde(rename_all = "snake_case")]
143pub enum FileStatus {
144    Added,
145    Modified,
146    Deleted,
147    Renamed,
148    Untracked,
149    /// A walkthrough's own file: one a stop or note anchors outside the diff,
150    /// shown at its current content with nothing to compare it against.
151    Unchanged,
152}
153
154impl FileStatus {
155    /// Single-character shape naming the status in the diff sidebar. It reads
156    /// by form alone, so a palette whose hues a reader cannot separate still
157    /// carries the status; colour reinforces it.
158    pub const fn glyph(self) -> char {
159        match self {
160            Self::Added => '+',
161            Self::Modified => '●',
162            Self::Deleted => '−',
163            Self::Renamed => '~',
164            Self::Untracked => '○',
165            Self::Unchanged => '·',
166        }
167    }
168
169    /// Neogit-style row label shown in file headers and the diff pane.
170    pub const fn label(self) -> &'static str {
171        match self {
172            Self::Added => "new file",
173            Self::Modified => "modified",
174            Self::Deleted => "deleted",
175            Self::Renamed => "renamed",
176            Self::Untracked => "untracked",
177            Self::Unchanged => "unchanged",
178        }
179    }
180}
181
182/// Stable identity for a hunk: hash of its normalized content. Survives
183/// edits elsewhere in the file; changes when the hunk's lines change.
184#[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
185pub struct HunkId(pub String);
186
187#[derive(Debug, Clone, PartialEq, Eq)]
188pub struct Hunk {
189    pub id: HunkId,
190    pub old_start: u32,
191    pub old_lines: u32,
192    pub new_start: u32,
193    pub new_lines: u32,
194    /// git's section heading: the enclosing function/section name git emits
195    /// after the second `@@` of the hunk header. Empty when git gives none
196    /// (e.g. a top-of-file hunk). Excluded from `id`, which keys only on lines.
197    pub context: String,
198    pub lines: Vec<DiffLine>,
199}
200
201impl Hunk {
202    pub fn header(&self) -> String {
203        format!(
204            "@@ -{},{} +{},{} @@",
205            self.old_start, self.old_lines, self.new_start, self.new_lines
206        )
207    }
208}
209
210#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
211#[serde(rename_all = "snake_case")]
212pub enum LineKind {
213    Context,
214    Deleted,
215    Added,
216}
217
218impl LineKind {
219    /// Unified-diff origin character (' ', '-', '+').
220    pub const fn origin(self) -> char {
221        match self {
222            Self::Context => ' ',
223            Self::Deleted => '-',
224            Self::Added => '+',
225        }
226    }
227}
228
229#[derive(Debug, Clone, PartialEq, Eq)]
230pub struct DiffLine {
231    pub kind: LineKind,
232    pub old_no: Option<u32>,
233    pub new_no: Option<u32>,
234    /// Line content without the trailing newline.
235    pub text: String,
236    /// Byte ranges within `text` to emphasize (intra-line changes).
237    pub emphasis: Vec<Range<usize>>,
238    /// True on a paired deleted/added line the structural algorithm found to
239    /// be a pure reformat (identical token structure, e.g. reindentation):
240    /// the renderer dims it, leaving red/green for an actual change.
241    pub reformat_only: bool,
242}
243
244impl DiffLine {
245    pub fn new(kind: LineKind, old_no: Option<u32>, new_no: Option<u32>, text: String) -> Self {
246        Self {
247            kind,
248            old_no,
249            new_no,
250            text,
251            emphasis: Vec::new(),
252            reformat_only: false,
253        }
254    }
255
256    /// The line number on the named side: `old_no` on the old side (where a
257    /// deletion lives), `new_no` everywhere else.
258    pub const fn number_on(&self, old_side: bool) -> Option<u32> {
259        if old_side { self.old_no } else { self.new_no }
260    }
261}
262
263/// Hash the hunk's content (kinds + text) into a stable id. `occurrence` is
264/// how many earlier hunks in the same file hash to the same content: two
265/// hunks with byte-identical lines in one file get distinct ids this way,
266/// while every other hunk's id depends on its own content alone, so it
267/// survives edits elsewhere in the file. [`disambiguated_hunk_id`] is the
268/// usual way to call this, since it tracks the count for the caller.
269pub fn hunk_id(file_path: &str, lines: &[DiffLine], occurrence: usize) -> HunkId {
270    let mut buf = String::new();
271    buf.push_str(file_path);
272    buf.push('\n');
273    for line in lines {
274        let tag = match line.kind {
275            LineKind::Context => ' ',
276            LineKind::Deleted => '-',
277            LineKind::Added => '+',
278        };
279        buf.push(tag);
280        buf.push_str(&line.text);
281        buf.push('\n');
282    }
283    if occurrence > 0 {
284        buf.push_str(&occurrence.to_string());
285        buf.push('\n');
286    }
287    HunkId(stable_hash(buf.as_bytes()))
288}
289
290/// [`hunk_id`] for one hunk of a file whose other hunks are being assigned
291/// ids through the same `seen` map: each distinct content gets occurrence 0
292/// the first time and counts up from there, so two identical hunks in one
293/// file never collide.
294// every caller shares one default-hashed map for one file's hunks, so a
295// generic hasher buys nothing
296#[allow(clippy::implicit_hasher)]
297pub fn disambiguated_hunk_id(
298    file_path: &str,
299    lines: &[DiffLine],
300    seen: &mut std::collections::HashMap<HunkId, usize>,
301) -> HunkId {
302    let base = hunk_id(file_path, lines, 0);
303    let occurrence = seen.entry(base).or_insert(0);
304    let id = hunk_id(file_path, lines, *occurrence);
305    *occurrence += 1;
306    id
307}
308
309#[cfg(test)]
310mod tests {
311    use super::*;
312
313    // hashes key persisted viewed marks: the algorithm must stay stable
314    // across releases, so pin known FNV-1a 64 values
315    #[test]
316    fn stable_hash_is_fnv1a64_and_never_changes() {
317        assert_eq!(stable_hash(b""), "cbf29ce484222325");
318        assert_eq!(stable_hash(b"hello"), "a430d84680aabd0b");
319    }
320
321    #[test]
322    fn file_status_glyph_and_label_cover_all_variants() {
323        let glyphs = [
324            FileStatus::Added,
325            FileStatus::Modified,
326            FileStatus::Deleted,
327            FileStatus::Renamed,
328            FileStatus::Untracked,
329            FileStatus::Unchanged,
330        ]
331        .map(FileStatus::glyph);
332        assert_eq!(glyphs, ['+', '●', '−', '~', '○', '·']);
333        let mut distinct = glyphs.to_vec();
334        distinct.sort_unstable();
335        distinct.dedup();
336        assert_eq!(
337            distinct.len(),
338            glyphs.len(),
339            "no two statuses share a shape"
340        );
341
342        assert_eq!(FileStatus::Added.label(), "new file");
343        assert_eq!(FileStatus::Modified.label(), "modified");
344        assert_eq!(FileStatus::Deleted.label(), "deleted");
345        assert_eq!(FileStatus::Renamed.label(), "renamed");
346        assert_eq!(FileStatus::Untracked.label(), "untracked");
347        assert_eq!(FileStatus::Unchanged.label(), "unchanged");
348    }
349
350    fn line(kind: LineKind, text: &str) -> DiffLine {
351        DiffLine::new(kind, None, None, text.to_owned())
352    }
353
354    #[test]
355    fn hunk_id_is_stable() {
356        let lines = vec![line(LineKind::Deleted, "a"), line(LineKind::Added, "b")];
357        let id1 = hunk_id("src/x.rs", &lines, 0);
358        let id2 = hunk_id("src/x.rs", &lines, 0);
359        assert_eq!(id1, id2);
360    }
361
362    #[test]
363    fn hunk_id_changes_with_content() {
364        let a = vec![line(LineKind::Added, "x")];
365        let b = vec![line(LineKind::Added, "y")];
366        assert_ne!(hunk_id("f", &a, 0), hunk_id("f", &b, 0));
367    }
368
369    #[test]
370    fn hunk_id_changes_with_kind() {
371        let a = vec![line(LineKind::Added, "x")];
372        let b = vec![line(LineKind::Deleted, "x")];
373        assert_ne!(hunk_id("f", &a, 0), hunk_id("f", &b, 0));
374    }
375
376    #[test]
377    fn hunk_id_changes_with_file() {
378        let lines = vec![line(LineKind::Added, "x")];
379        assert_ne!(hunk_id("a", &lines, 0), hunk_id("b", &lines, 0));
380    }
381
382    #[test]
383    fn hunk_id_changes_with_occurrence() {
384        let lines = vec![line(LineKind::Added, "x")];
385        assert_ne!(hunk_id("f", &lines, 0), hunk_id("f", &lines, 1));
386    }
387
388    #[test]
389    fn disambiguated_hunk_id_gives_identical_hunks_distinct_ids() {
390        let lines = vec![line(LineKind::Added, "x")];
391        let mut seen = std::collections::HashMap::new();
392        let first = disambiguated_hunk_id("f", &lines, &mut seen);
393        let second = disambiguated_hunk_id("f", &lines, &mut seen);
394        assert_ne!(first, second);
395        assert_eq!(first, hunk_id("f", &lines, 0));
396        assert_eq!(second, hunk_id("f", &lines, 1));
397    }
398
399    #[test]
400    fn header_formats() {
401        let hunk = Hunk {
402            id: HunkId("h".into()),
403            old_start: 10,
404            old_lines: 7,
405            new_start: 10,
406            new_lines: 9,
407            context: String::new(),
408            lines: vec![],
409        };
410        assert_eq!(hunk.header(), "@@ -10,7 +10,9 @@");
411    }
412
413    #[test]
414    fn content_hash_changes_when_new_text_changes() {
415        let base = FileDiff {
416            path: "f.rs".into(),
417            old_path: None,
418            status: FileStatus::Modified,
419            binary: false,
420            old_text: None,
421            new_text: Some("fn main() {}".into()),
422            hunks: vec![],
423            hashes: HashCache::default(),
424            blobs: BlobIds::default(),
425        };
426        let mut changed = base.clone();
427        changed.new_text = Some("fn main() { let x = 1; }".into());
428        assert_ne!(base.content_hash(), changed.content_hash());
429    }
430
431    #[test]
432    fn content_hash_is_stable() {
433        let file = FileDiff {
434            path: "f.rs".into(),
435            old_path: None,
436            status: FileStatus::Modified,
437            binary: false,
438            old_text: None,
439            new_text: Some("same content".into()),
440            hunks: vec![],
441            hashes: HashCache::default(),
442            blobs: BlobIds::default(),
443        };
444        assert_eq!(file.content_hash(), file.content_hash());
445    }
446
447    #[test]
448    fn sides_hash_changes_when_old_text_changes() {
449        let base = FileDiff {
450            path: "f.rs".into(),
451            old_path: None,
452            status: FileStatus::Modified,
453            binary: false,
454            old_text: Some("fn main() {}".into()),
455            new_text: Some("fn main() { let x = 1; }".into()),
456            hunks: vec![],
457            hashes: HashCache::default(),
458            blobs: BlobIds::default(),
459        };
460        let mut changed = base.clone();
461        changed.old_text = Some("fn main() { unreachable!() }".into());
462        assert_eq!(
463            base.content_hash(),
464            changed.content_hash(),
465            "same new side, same content hash"
466        );
467        assert_ne!(base.sides_hash(), changed.sides_hash());
468    }
469
470    fn one_file_model(path: &str, old_text: &str, new_text: &str) -> DiffModel {
471        DiffModel {
472            files: vec![FileDiff {
473                path: path.to_owned(),
474                old_path: None,
475                status: FileStatus::Modified,
476                binary: false,
477                old_text: Some(old_text.to_owned()),
478                new_text: Some(new_text.to_owned()),
479                hunks: vec![],
480                hashes: HashCache::default(),
481                blobs: BlobIds::default(),
482            }],
483        }
484    }
485
486    #[test]
487    fn fingerprint_is_stable_for_identical_models() {
488        let a = one_file_model("f.rs", "old", "new");
489        let b = one_file_model("f.rs", "old", "new");
490        assert_eq!(a.fingerprint(), b.fingerprint());
491    }
492
493    #[test]
494    fn fingerprint_changes_with_content_path_and_file_set() {
495        let base = one_file_model("f.rs", "old", "new");
496        assert_ne!(
497            base.fingerprint(),
498            one_file_model("f.rs", "old", "newer").fingerprint(),
499            "changed side changes the fingerprint"
500        );
501        assert_ne!(
502            base.fingerprint(),
503            one_file_model("g.rs", "old", "new").fingerprint(),
504            "renamed file changes the fingerprint"
505        );
506        let mut grown = base.clone();
507        grown.files.extend(one_file_model("g.rs", "", "x").files);
508        assert_ne!(
509            base.fingerprint(),
510            grown.fingerprint(),
511            "added file changes the fingerprint"
512        );
513    }
514
515    fn model_with_lines() -> DiffModel {
516        DiffModel {
517            files: vec![FileDiff {
518                path: "f.rs".into(),
519                old_path: None,
520                status: FileStatus::Modified,
521                binary: false,
522                old_text: None,
523                new_text: None,
524                hunks: vec![Hunk {
525                    id: HunkId("h".into()),
526                    old_start: 1,
527                    old_lines: 2,
528                    new_start: 1,
529                    new_lines: 2,
530                    context: String::new(),
531                    lines: vec![
532                        DiffLine::new(LineKind::Context, Some(1), Some(1), "one".into()),
533                        DiffLine::new(LineKind::Deleted, Some(2), None, "two".into()),
534                        DiffLine::new(LineKind::Added, None, Some(2), "TWO".into()),
535                    ],
536                }],
537                hashes: HashCache::default(),
538                blobs: BlobIds::default(),
539            }],
540        }
541    }
542
543    #[test]
544    fn diffstat_counts_added_and_deleted_over_hunks() {
545        // model_with_lines: one context, one deleted, one added line
546        let model = model_with_lines();
547        assert_eq!(model.files[0].diffstat(), (1, 1));
548
549        // two hunks: 2 added + 1 deleted total, context ignored
550        let file = FileDiff {
551            path: "f.rs".into(),
552            old_path: None,
553            status: FileStatus::Modified,
554            binary: false,
555            old_text: None,
556            new_text: None,
557            hunks: vec![
558                Hunk {
559                    id: HunkId("a".into()),
560                    old_start: 1,
561                    old_lines: 1,
562                    new_start: 1,
563                    new_lines: 2,
564                    context: String::new(),
565                    lines: vec![
566                        DiffLine::new(LineKind::Context, Some(1), Some(1), "ctx".into()),
567                        DiffLine::new(LineKind::Added, None, Some(2), "add one".into()),
568                    ],
569                },
570                Hunk {
571                    id: HunkId("b".into()),
572                    old_start: 5,
573                    old_lines: 1,
574                    new_start: 6,
575                    new_lines: 1,
576                    context: String::new(),
577                    lines: vec![
578                        DiffLine::new(LineKind::Deleted, Some(5), None, "gone".into()),
579                        DiffLine::new(LineKind::Added, None, Some(6), "add two".into()),
580                    ],
581                },
582            ],
583            hashes: HashCache::default(),
584            blobs: BlobIds::default(),
585        };
586        assert_eq!(file.diffstat(), (2, 1));
587    }
588
589    #[test]
590    fn find_line_matches_the_requested_side() {
591        let model = model_with_lines();
592        let new_side = model.find_line("f.rs", 2, false).expect("new side");
593        assert_eq!(new_side.text, "TWO");
594        let old_side = model.find_line("f.rs", 2, true).expect("old side");
595        assert_eq!(old_side.text, "two");
596    }
597
598    #[test]
599    fn find_line_misses_unknown_files_and_lines() {
600        let model = model_with_lines();
601        assert!(model.find_line("nope.rs", 1, false).is_none());
602        assert!(model.find_line("f.rs", 99, false).is_none());
603    }
604
605    #[test]
606    fn content_hash_falls_back_for_none() {
607        let file = FileDiff {
608            path: "f.rs".into(),
609            old_path: None,
610            status: FileStatus::Deleted,
611            binary: false,
612            old_text: None,
613            new_text: None,
614            hunks: vec![],
615            hashes: HashCache::default(),
616            blobs: BlobIds::default(),
617        };
618        // must not panic, must return a non-empty string (git hash of empty blob)
619        let hash = file.content_hash();
620        assert!(!hash.is_empty());
621    }
622}