Skip to main content

heddle_object_model/object/
state_core.rs

1// SPDX-License-Identifier: Apache-2.0
2//! Core state type and its leaf value types (Status, StateSignature,
3//! SignatureStatus, Verification).
4
5use std::collections::BTreeMap;
6
7use chrono::{DateTime, Utc};
8use serde::{Deserialize, Serialize};
9
10use super::{Attribution, ChangeId, ContentHash, Principal, StateId};
11
12// ── Status ──────────────────────────────────────────────────────────
13
14/// Lifecycle status of a state.
15#[derive(Clone, Copy, Debug, PartialEq, Eq, Default, Serialize, Deserialize)]
16pub enum Status {
17    #[default]
18    Draft,
19    Published,
20}
21
22impl Status {
23    pub fn to_byte(&self) -> u8 {
24        match self {
25            Status::Draft => 0,
26            Status::Published => 1,
27        }
28    }
29
30    pub fn from_byte(b: u8) -> Option<Self> {
31        match b {
32            0 => Some(Status::Draft),
33            1 => Some(Status::Published),
34            _ => None,
35        }
36    }
37}
38
39#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
40pub enum ChangeLineageKind {
41    CherryPick,
42    Collapse,
43    Revert,
44    GitProjection,
45}
46
47impl ChangeLineageKind {
48    fn to_byte(self) -> u8 {
49        match self {
50            Self::CherryPick => 1,
51            Self::Collapse => 2,
52            Self::Revert => 3,
53            Self::GitProjection => 4,
54        }
55    }
56}
57
58#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
59pub struct ChangeLineage {
60    pub kind: ChangeLineageKind,
61    pub source_change: ChangeId,
62    pub source_state: StateId,
63}
64
65// ── StateSignature ──────────────────────────────────────────────────
66
67/// Signature information for a state.
68#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
69pub struct StateSignature {
70    pub algorithm: String,
71    pub public_key: String,
72    pub signature: String,
73}
74
75impl StateSignature {
76    pub fn algorithm(&self) -> &str {
77        &self.algorithm
78    }
79}
80
81/// Signature verification result.
82#[derive(Clone, Copy, Debug, PartialEq, Eq)]
83pub enum SignatureStatus {
84    Valid,
85    Legacy,
86    Invalid,
87    Unsigned,
88}
89
90impl SignatureStatus {
91    pub fn is_valid(self) -> bool {
92        self == SignatureStatus::Valid
93    }
94
95    pub fn is_unsigned(self) -> bool {
96        self == SignatureStatus::Unsigned
97    }
98
99    pub fn is_legacy(self) -> bool {
100        self == SignatureStatus::Legacy
101    }
102}
103
104// ── Verification ────────────────────────────────────────────────────
105
106/// Verification information for a state.
107#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
108pub struct Verification {
109    pub tests_passed: Option<bool>,
110    pub tests_failed: Option<u32>,
111    pub coverage_pct: Option<f32>,
112    pub coverage_delta: Option<f32>,
113    pub lint_warnings: Option<u32>,
114    #[serde(default)]
115    pub custom: BTreeMap<String, serde_json::Value>,
116}
117
118impl Verification {
119    pub fn new() -> Self {
120        Self::default()
121    }
122
123    pub fn with_tests_passed(mut self, passed: bool) -> Self {
124        self.tests_passed = Some(passed);
125        self
126    }
127
128    pub fn with_tests_failed(mut self, failed: u32) -> Self {
129        self.tests_failed = Some(failed);
130        self
131    }
132
133    pub fn is_empty(&self) -> bool {
134        self.tests_passed.is_none()
135            && self.tests_failed.is_none()
136            && self.coverage_pct.is_none()
137            && self.coverage_delta.is_none()
138            && self.lint_warnings.is_none()
139            && self.custom.is_empty()
140    }
141
142    pub(crate) fn hash_len(&self) -> usize {
143        let mut len = 0;
144        len += 1 + self.tests_passed.map(|_| 1).unwrap_or(0);
145        len += 1 + self.tests_failed.map(|_| 4).unwrap_or(0);
146        len += 1 + self.coverage_pct.map(|_| 4).unwrap_or(0);
147        len += 1 + self.coverage_delta.map(|_| 4).unwrap_or(0);
148        len += 1 + self.lint_warnings.map(|_| 4).unwrap_or(0);
149        len += 4;
150        for (key, value) in &self.custom {
151            let value_bytes = serde_json::to_vec(value).unwrap_or_default();
152            len += 4 + key.len();
153            len += 4 + value_bytes.len();
154        }
155        len
156    }
157
158    pub(crate) fn update_hasher(&self, hasher: &mut blake3::Hasher) {
159        let tests_passed = self.tests_passed.map(u8::from);
160        write_optional_u8(hasher, tests_passed);
161        write_optional_u32(hasher, self.tests_failed);
162        write_optional_f32(hasher, self.coverage_pct);
163        write_optional_f32(hasher, self.coverage_delta);
164        write_optional_u32(hasher, self.lint_warnings);
165        let custom_len = self.custom.len() as u32;
166        hasher.update(&custom_len.to_le_bytes());
167        for (key, value) in &self.custom {
168            let key_bytes = key.as_bytes();
169            let value_bytes = serde_json::to_vec(value).unwrap_or_default();
170            hasher.update(&(key_bytes.len() as u32).to_le_bytes());
171            hasher.update(key_bytes);
172            hasher.update(&(value_bytes.len() as u32).to_le_bytes());
173            hasher.update(&value_bytes);
174        }
175    }
176}
177
178fn write_optional_u8(hasher: &mut blake3::Hasher, value: Option<u8>) {
179    match value {
180        Some(v) => {
181            hasher.update(&[1]);
182            hasher.update(&[v]);
183        }
184        None => {
185            hasher.update(&[0]);
186        }
187    }
188}
189
190fn write_optional_u32(hasher: &mut blake3::Hasher, value: Option<u32>) {
191    match value {
192        Some(v) => {
193            hasher.update(&[1]);
194            hasher.update(&v.to_le_bytes());
195        }
196        None => {
197            hasher.update(&[0]);
198        }
199    }
200}
201
202fn write_optional_f32(hasher: &mut blake3::Hasher, value: Option<f32>) {
203    match value {
204        Some(v) => {
205            hasher.update(&[1]);
206            hasher.update(&v.to_le_bytes());
207        }
208        None => {
209            hasher.update(&[0]);
210        }
211    }
212}
213
214// ── State ───────────────────────────────────────────────────────────
215
216/// Sentinel stored in [`State::authored_tz_offset`] / [`State::committer_tz_offset`]
217/// for Git's `-0000` timezone token ("local zone unknown"). A seconds-east
218/// offset cannot distinguish `-0000` from `+0000`, yet the sign is part of the
219/// commit bytes and therefore its object id. `i32::MIN` is far outside any real
220/// offset (Git's `±HHMM` spans at most ±99h59m), so it never collides with one.
221pub const TZ_OFFSET_NEGATIVE_UTC: i32 = i32::MIN;
222
223/// Convert a parsed Git timezone (minutes east of UTC plus whether the token was
224/// `-0000`) to the seconds-east value stored on a [`State`], keeping the `-0000`
225/// sign as [`TZ_OFFSET_NEGATIVE_UTC`].
226pub fn git_tz_offset_seconds(minutes_east: i16, negative_utc: bool) -> i32 {
227    if negative_utc {
228        TZ_OFFSET_NEGATIVE_UTC
229    } else {
230        i32::from(minutes_east) * 60
231    }
232}
233
234/// Immutable source-history state. `state_id` is recomputed from every encoded
235/// field; mutable repository metadata lives in `StateAttachment` objects.
236#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
237pub struct State {
238    #[serde(skip)]
239    pub state_id: StateId,
240    pub change_id: ChangeId,
241    pub tree: ContentHash,
242    pub parents: Vec<StateId>,
243    pub attribution: Attribution,
244    pub intent: Option<String>,
245    pub confidence: Option<f32>,
246    pub created_at: DateTime<Utc>,
247    pub verification: Option<Verification>,
248    pub status: Status,
249    // --- tail-only optional fields below. Add new fields here, never above. ---
250    #[serde(default)]
251    pub provenance: Option<ContentHash>,
252    /// Authoring timestamp for this state, when distinct from
253    /// `created_at`.
254    ///
255    /// `created_at` is the *committer* time — when the state object
256    /// came into being in its current form. `authored_at` is the
257    /// *author* time — when someone actually wrote the change — which
258    /// survives `git rebase`, cherry-pick, squash-merge, and `git
259    /// commit --amend`. The ingest-backed `import git` path fills
260    /// this from the git author time; native heddle commits leave it
261    /// `None` and blame falls back to `created_at`.
262    ///
263    /// **Part of the state hash (#564 de-lossy step 1).** Author time
264    /// is part of a git commit's identity: two commits that differ
265    /// *only* by author timestamp are distinct git objects, so folding
266    /// it into the hash keeps them from dedup-colliding to one State in
267    /// the content-addressed store. `None` hashes as a single absence
268    /// byte, so native commits are unaffected beyond the format bump.
269    #[serde(default)]
270    pub authored_at: Option<DateTime<Utc>>,
271    // --- git-fidelity fields (#564 de-lossy step 1, #565) ---
272    //
273    // These preserve the parts of an imported git commit that Heddle's
274    // model used to drop, so a commit can be byte-reconstructed later
275    // (#566/#567) and the git mirror can be eliminated (#568). UNLIKE the
276    // W1 tail fields above, these ARE part of the content hash (see
277    // `update_hash`): two git-distinct commits that differ only in
278    // committer, timezone, verbatim message, gpgsig, or extra headers must
279    // hash differently so they can't dedup-collide in the content-addressed
280    // store. They are still tail-append + `#[serde(default)]` so legacy
281    // on-disk states keep deserializing.
282    /// The git committer identity, when distinct from the author
283    /// ([`Attribution::principal`]). Git records both an author (who wrote
284    /// the change) and a committer (who created this commit object); for
285    /// rebased / cherry-picked / amended commits the two differ. `None`
286    /// for native heddle commits and for legacy imports from before #565.
287    #[serde(default)]
288    pub committer: Option<Principal>,
289    /// Timezone offset (seconds east of UTC) of the *author* timestamp
290    /// ([`State::authored_at`] / `created_at` fallback). Git stores the
291    /// author's local offset (e.g. `+0000`, `-0700`); Heddle used to
292    /// discard it. `0` for native commits and legacy imports.
293    #[serde(default)]
294    pub authored_tz_offset: i32,
295    /// Timezone offset (seconds east of UTC) of the *committer* timestamp
296    /// (`created_at`). `0` for native commits and legacy imports.
297    #[serde(default)]
298    pub committer_tz_offset: i32,
299    /// The verbatim git commit message body (everything after the header
300    /// block), preserved exactly so reconstruction is byte-stable. Distinct
301    /// from `intent`, which is the trimmed first line surfaced in the UI.
302    /// `None` for native commits and legacy imports.
303    ///
304    /// Stored as raw bytes, NOT a `String`: a commit with a non-UTF8
305    /// `encoding` (latin-1, shift-jis, …) carries message bytes that are not
306    /// valid UTF-8 (e.g. `0xe9` for latin-1 `é`); a `String` could not
307    /// round-trip them byte-identically. (non-UTF8 author/committer identity
308    /// *names* are not yet byte-preserved — `Principal` is still `String`; see
309    /// #564.)
310    #[serde(default)]
311    pub raw_message: Option<Vec<u8>>,
312    /// The SINGLE canonical "this state's content is NOT byte-faithful to the
313    /// original git object" marker (#567). Set to `true` by lossy import
314    /// population paths whenever an unrepresentable tree entry was dropped or
315    /// converted during import, so the rebuilt tree (hence commit) no longer
316    /// hashes to the original SHA. The git-export fidelity guard reads this one
317    /// flag to decide whether reconstruct-from-state is safe, instead of
318    /// enumerating import surfaces. `false` for native heddle commits and for
319    /// lossless imports.
320    ///
321    /// Provenance metadata, NOT part of the content hash: a lossy import always
322    /// drops/converts tree entries, so its tree — and therefore the rest of the
323    /// hashed identity — already differs from a lossless import of the same
324    /// source; folding the flag in would add nothing but break every existing
325    /// content hash.
326    #[serde(default)]
327    pub git_lossy: bool,
328    /// Every git commit header beyond the ones Heddle models natively
329    /// (tree/parents/author/committer), in their original order. ORDER IS
330    /// LOAD-BEARING for #566 byte-exactness — this is a `Vec`, never a map.
331    /// Empty for native commits and legacy imports.
332    ///
333    /// `gpgsig` is just one of these headers and is kept INLINE at its
334    /// captured ordinal (not split into a separate field): when a commit's
335    /// extension headers are in non-canonical order — e.g. `x-custom`, then
336    /// `gpgsig`, then `mergetag` — splitting gpgsig out would lose its
337    /// position and break byte-identical reconstruction. The serialization
338    /// source of truth for the signature is its position here (spike §3).
339    ///
340    /// Both the header name and value are raw bytes (`Vec<u8>`), NOT
341    /// `String`s: extra-header VALUES (a `mergetag` payload is a full tag
342    /// object; custom headers; gpgsig armor) can be non-UTF8, so a
343    /// `String` would force a lossy `to_string()` that destroys those bytes.
344    /// Names are ASCII by git's spec but are bytes too so the whole tuple is
345    /// byte-exact and no conversion sneaks in.
346    #[serde(default)]
347    pub extra_headers: Vec<(Vec<u8>, Vec<u8>)>,
348    pub lineage: Vec<ChangeLineage>,
349}
350
351impl State {
352    pub fn new(tree: ContentHash, parents: Vec<StateId>, attribution: Attribution) -> Self {
353        Self::new_snapshot(tree, parents, attribution)
354    }
355
356    pub fn new_snapshot(
357        tree: ContentHash,
358        parents: Vec<StateId>,
359        attribution: Attribution,
360    ) -> Self {
361        Self::new_with_change_id(tree, parents, attribution, ChangeId::generate())
362    }
363
364    pub fn new_merge(tree: ContentHash, parents: Vec<StateId>, attribution: Attribution) -> Self {
365        Self::new_snapshot(tree, parents, attribution)
366    }
367
368    pub fn new_refresh_of(
369        tree: ContentHash,
370        parents: Vec<StateId>,
371        attribution: Attribution,
372        change_id: ChangeId,
373    ) -> Self {
374        Self::new_with_change_id(tree, parents, attribution, change_id)
375    }
376
377    pub fn new_fork_of(tree: ContentHash, parents: Vec<StateId>, attribution: Attribution) -> Self {
378        Self::new_snapshot(tree, parents, attribution)
379    }
380
381    pub fn new_collapse_of(
382        tree: ContentHash,
383        parents: Vec<StateId>,
384        attribution: Attribution,
385    ) -> Self {
386        Self::new_snapshot(tree, parents, attribution)
387    }
388
389    fn new_with_change_id(
390        tree: ContentHash,
391        parents: Vec<StateId>,
392        attribution: Attribution,
393        change_id: ChangeId,
394    ) -> Self {
395        let mut state = Self {
396            state_id: StateId::default(),
397            change_id,
398            tree,
399            parents,
400            attribution,
401            intent: None,
402            confidence: None,
403            created_at: Utc::now(),
404            verification: None,
405            provenance: None,
406            authored_at: None,
407            committer: None,
408            authored_tz_offset: 0,
409            committer_tz_offset: 0,
410            raw_message: None,
411            git_lossy: false,
412            extra_headers: Vec::new(),
413            lineage: Vec::new(),
414            status: Status::Draft,
415        };
416        state.refresh_state_id();
417        state
418    }
419
420    pub fn with_intent(mut self, intent: impl Into<String>) -> Self {
421        self.intent = Some(intent.into());
422        self.refresh_state_id();
423        self
424    }
425
426    pub fn with_confidence(mut self, confidence: f32) -> Self {
427        self.confidence = Some(confidence.clamp(0.0, 1.0));
428        self.refresh_state_id();
429        self
430    }
431
432    pub fn with_verification(mut self, verification: Verification) -> Self {
433        self.verification = Some(verification);
434        self.refresh_state_id();
435        self
436    }
437
438    pub fn with_provenance(mut self, provenance: ContentHash) -> Self {
439        self.provenance = Some(provenance);
440        self.refresh_state_id();
441        self
442    }
443
444    /// Record the authoring timestamp separately from `created_at`.
445    /// Used by the git-ingest importer to preserve the distinction
446    /// between "when the change was originally written" (authored)
447    /// and "when this commit object came into being" (committer time,
448    /// stored in `created_at` so re-imports stay deterministic).
449    /// Native heddle commits leave this `None`; blame display then
450    /// falls back to `created_at`.
451    ///
452    /// **Part of the state hash (#564 de-lossy step 1)** — see the
453    /// `authored_at` field docs and `update_hash`.
454    pub fn with_authored_at(mut self, timestamp: DateTime<Utc>) -> Self {
455        self.authored_at = Some(timestamp);
456        self.refresh_state_id();
457        self
458    }
459
460    /// Record the git committer identity (distinct from the author).
461    ///
462    /// **Part of the state hash** — see the `committer` field docs and
463    /// `update_hash`. #564 de-lossy step 1.
464    pub fn with_committer(mut self, committer: Principal) -> Self {
465        self.committer = Some(committer);
466        self.refresh_state_id();
467        self
468    }
469
470    /// Record the author/committer timezone offsets (seconds east of UTC).
471    /// **Part of the state hash.** #564 de-lossy step 1.
472    pub fn with_tz_offsets(mut self, authored: i32, committer: i32) -> Self {
473        self.authored_tz_offset = authored;
474        self.committer_tz_offset = committer;
475        self.refresh_state_id();
476        self
477    }
478
479    /// Record the verbatim git commit message body, as raw bytes (so a
480    /// non-UTF8 message round-trips byte-identically; see the `raw_message`
481    /// field docs). **Part of the state hash.** #564 de-lossy step 1.
482    pub fn with_raw_message(mut self, raw_message: impl AsRef<[u8]>) -> Self {
483        self.raw_message = Some(raw_message.as_ref().to_vec());
484        self.refresh_state_id();
485        self
486    }
487
488    /// Mark this state's content as NOT byte-faithful to the original git
489    /// object — set by the `--lossy` import/ingest paths when a tree entry was
490    /// dropped or converted. The git-export fidelity guard reads this single
491    /// signal to skip reconstruct-from-state (#567). Not part of the content
492    /// hash (see the `git_lossy` field docs).
493    pub fn with_git_lossy(mut self, git_lossy: bool) -> Self {
494        self.git_lossy = git_lossy;
495        self.refresh_state_id();
496        self
497    }
498
499    /// Record the ordered remaining git commit headers as raw bytes. ORDER
500    /// IS LOAD-BEARING (#566). **Part of the state hash.** #564 de-lossy
501    /// step 1.
502    pub fn with_extra_headers(mut self, extra_headers: Vec<(Vec<u8>, Vec<u8>)>) -> Self {
503        self.extra_headers = extra_headers;
504        self.refresh_state_id();
505        self
506    }
507
508    pub fn with_lineage(mut self, lineage: Vec<ChangeLineage>) -> Self {
509        self.lineage = lineage;
510        self.refresh_state_id();
511        self
512    }
513
514    pub fn with_status(mut self, status: Status) -> Self {
515        self.status = status;
516        self.refresh_state_id();
517        self
518    }
519
520    pub fn with_change_id(mut self, change_id: ChangeId) -> Self {
521        self.change_id = change_id;
522        self.refresh_state_id();
523        self
524    }
525
526    pub fn with_timestamp(mut self, timestamp: DateTime<Utc>) -> Self {
527        self.created_at = timestamp;
528        self.refresh_state_id();
529        self
530    }
531
532    pub fn compute_hash(&self) -> ContentHash {
533        let content_len = self.hash_len();
534        ContentHash::compute_typed_with_len("state", content_len, |hasher| {
535            self.update_hash(hasher);
536        })
537    }
538
539    pub fn hash(&mut self) -> ContentHash {
540        self.refresh_state_id();
541        self.state_id.as_content_hash()
542    }
543
544    pub fn id(&self) -> StateId {
545        StateId::from_content_hash(self.compute_hash())
546    }
547
548    /// Encode the canonical named-field msgpack representation used in packs
549    /// and object transfer. Local loose storage may wrap a different encoding,
550    /// but it must not redefine the portable object body.
551    pub fn encode_current_msgpack(&self) -> crate::error::Result<Vec<u8>> {
552        Ok(rmp_serde::to_vec_named(self)?)
553    }
554
555    /// Decode the canonical named-field msgpack representation and restore the
556    /// derived in-memory id omitted from serde.
557    pub fn decode_current_msgpack(bytes: &[u8]) -> crate::error::Result<Self> {
558        let mut state: Self = rmp_serde::from_slice(bytes)?;
559        state.refresh_state_id();
560        Ok(state)
561    }
562
563    /// Format-4 identity for agent states hashed before `thought_level` and
564    /// `parent` entered the transcript. Graph edges keep this id.
565    pub fn pre_cursor_id(&self) -> StateId {
566        StateId::from_content_hash(self.compute_pre_cursor_hash())
567    }
568
569    /// Accept the current id, or a format-4 agent id that omitted unpublished
570    /// cursor fields. Published cursor fields require the current hash.
571    pub fn accepts_stored_id(&self, stored: &StateId) -> bool {
572        if self.id() == *stored {
573            return true;
574        }
575        let unpublished_cursor = self
576            .attribution
577            .agent
578            .as_ref()
579            .is_none_or(|agent| agent.thought_level.is_none() && agent.parent.is_none());
580        unpublished_cursor && self.pre_cursor_id() == *stored
581    }
582
583    /// Content hash that produced `stored` when this state accepts that id.
584    ///
585    /// Format-4 agent states keep a pre-cursor id. Verification and re-signing
586    /// must use that hash, not the current format-5 [`Self::compute_hash`].
587    pub fn hash_for_stored_id(&self, stored: &StateId) -> ContentHash {
588        if self.id() == *stored {
589            self.compute_hash()
590        } else if self.accepts_stored_id(stored) {
591            self.compute_pre_cursor_hash()
592        } else {
593            self.compute_hash()
594        }
595    }
596
597    pub fn is_root(&self) -> bool {
598        self.parents.is_empty()
599    }
600
601    pub fn is_merge(&self) -> bool {
602        self.parents.len() > 1
603    }
604
605    pub fn is_agent_authored(&self) -> bool {
606        self.attribution.agent.is_some()
607    }
608
609    pub fn first_parent(&self) -> Option<&StateId> {
610        self.parents.first()
611    }
612
613    fn hash_len(&self) -> u64 {
614        self.hash_len_core() + self.hash_len_fidelity()
615    }
616
617    fn hash_len_pre_cursor(&self) -> u64 {
618        self.hash_len_core_pre_cursor() + self.hash_len_fidelity()
619    }
620
621    /// Hashed length of the core state fields. Mirrors [`Self::update_hash_core`].
622    fn hash_len_core(&self) -> u64 {
623        self.hash_len_core_versioned(true)
624    }
625
626    fn hash_len_core_pre_cursor(&self) -> u64 {
627        self.hash_len_core_versioned(false)
628    }
629
630    fn hash_len_core_versioned(&self, include_cursor_fields: bool) -> u64 {
631        let principal = &self.attribution.principal;
632        let mut len = 0u64;
633
634        len += 16;
635
636        len += self.tree.as_bytes().len() as u64;
637        len += 4;
638        len += (self.parents.len() * 32) as u64;
639
640        len += principal.name.len() as u64 + 1;
641        len += principal.email.len() as u64 + 1;
642
643        len += 1;
644        if let Some(agent) = &self.attribution.agent {
645            len += agent.provider.len() as u64 + 1;
646            len += agent.model.len() as u64 + 1;
647
648            len += 1;
649            if let Some(session_id) = &agent.session_id {
650                len += session_id.len() as u64 + 1;
651            }
652
653            len += 1;
654            if let Some(segment_id) = &agent.segment_id {
655                len += segment_id.len() as u64 + 1;
656            }
657
658            len += 1;
659            if let Some(policy_id) = &agent.policy_id {
660                len += policy_id.len() as u64 + 1;
661            }
662
663            if include_cursor_fields {
664                len += 1;
665                if let Some(thought_level) = &agent.thought_level {
666                    len += thought_level.len() as u64 + 1;
667                }
668
669                len += 1;
670                if let Some(parent) = &agent.parent {
671                    len += parent.len() as u64 + 1;
672                }
673            }
674        }
675
676        len += 1;
677        if let Some(intent) = &self.intent {
678            len += intent.len() as u64 + 1;
679        }
680
681        len += 1;
682        if self.confidence.is_some() {
683            len += 4;
684        }
685
686        len += 8;
687
688        len += 1;
689        if let Some(verification) = &self.verification {
690            len += verification.hash_len() as u64;
691        }
692
693        len += 1;
694        if self.provenance.is_some() {
695            len += 32;
696        }
697
698        len += 1;
699
700        len
701    }
702
703    /// Hashed length of the appended git-fidelity block (#565). Mirrors
704    /// [`Self::update_hash_fidelity`] byte-for-byte. Kept separate from
705    /// [`Self::hash_len_core`] so the migration-only pre-bump hash can omit it
706    /// exactly.
707    fn hash_len_fidelity(&self) -> u64 {
708        let mut len = 0u64;
709
710        // git-fidelity fields (#564 step 1). Must mirror `update_hash`
711        // byte-for-byte. committer: 1 tag byte + (name+NUL, email+NUL).
712        len += 1;
713        if let Some(committer) = &self.committer {
714            len += committer.name.len() as u64 + 1;
715            len += committer.email.len() as u64 + 1;
716        }
717        // both tz offsets: i32 LE, always present.
718        len += 4;
719        len += 4;
720        // authored_at (author time): 1 tag byte + (i64 LE when Some).
721        len += 1;
722        if self.authored_at.is_some() {
723            len += 8;
724        }
725        // raw_message: optional-bytes framing (1 tag + u32 len + bytes) — a
726        // length prefix, not NUL-termination, since the message can contain
727        // NUL bytes (it's byte-typed for non-UTF8 fidelity).
728        len += 1;
729        if let Some(raw_message) = &self.raw_message {
730            len += 4 + raw_message.len() as u64;
731        }
732        // extra_headers (gpgsig rides inline here at its captured position):
733        // u32 count, then per pair u32 key_len+key, u32 val_len+val.
734        len += 4;
735        for (key, value) in &self.extra_headers {
736            len += 4 + key.len() as u64;
737            len += 4 + value.len() as u64;
738        }
739        len += 4 + (self.lineage.len() as u64 * 49);
740
741        len
742    }
743
744    fn update_hash(&self, hasher: &mut blake3::Hasher) {
745        self.update_hash_core(hasher);
746        self.update_hash_fidelity(hasher);
747    }
748
749    fn compute_pre_cursor_hash(&self) -> ContentHash {
750        let content_len = self.hash_len_pre_cursor();
751        ContentHash::compute_typed_with_len("state", content_len, |hasher| {
752            self.update_hash_pre_cursor(hasher);
753        })
754    }
755
756    fn update_hash_pre_cursor(&self, hasher: &mut blake3::Hasher) {
757        self.update_hash_core_pre_cursor(hasher);
758        self.update_hash_fidelity(hasher);
759    }
760
761    /// Hash the pre-#565 fields (everything through the status byte). Mirrors
762    /// [`Self::hash_len_core`]. The migration-only pre-bump hash is exactly
763    /// this with no fidelity block appended.
764    fn update_hash_core(&self, hasher: &mut blake3::Hasher) {
765        self.update_hash_core_versioned(hasher, true);
766    }
767
768    fn update_hash_core_pre_cursor(&self, hasher: &mut blake3::Hasher) {
769        self.update_hash_core_versioned(hasher, false);
770    }
771
772    fn update_hash_core_versioned(&self, hasher: &mut blake3::Hasher, include_cursor_fields: bool) {
773        let principal = &self.attribution.principal;
774
775        hasher.update(self.change_id.as_bytes());
776
777        hasher.update(self.tree.as_bytes());
778        hasher.update(&(self.parents.len() as u32).to_le_bytes());
779        for parent in &self.parents {
780            hasher.update(parent.as_bytes());
781        }
782
783        hasher.update(&principal.name);
784        hasher.update(&[0]);
785        hasher.update(&principal.email);
786        hasher.update(&[0]);
787
788        if let Some(agent) = &self.attribution.agent {
789            hasher.update(&[1]);
790            hasher.update(agent.provider.as_bytes());
791            hasher.update(&[0]);
792            hasher.update(agent.model.as_bytes());
793            hasher.update(&[0]);
794            write_optional_string(hasher, &agent.session_id);
795            write_optional_string(hasher, &agent.segment_id);
796            write_optional_string(hasher, &agent.policy_id);
797            if include_cursor_fields {
798                write_optional_string(hasher, &agent.thought_level);
799                write_optional_string(hasher, &agent.parent);
800            }
801        } else {
802            hasher.update(&[0]);
803        }
804
805        write_optional_string(hasher, &self.intent);
806
807        if let Some(confidence) = self.confidence {
808            hasher.update(&[1]);
809            hasher.update(&confidence.to_le_bytes());
810        } else {
811            hasher.update(&[0]);
812        }
813
814        hasher.update(&self.created_at.timestamp().to_le_bytes());
815
816        if let Some(verification) = &self.verification {
817            hasher.update(&[1]);
818            verification.update_hasher(hasher);
819        } else {
820            hasher.update(&[0]);
821        }
822
823        if let Some(provenance) = self.provenance {
824            hasher.update(&[1]);
825            hasher.update(provenance.as_bytes());
826        } else {
827            hasher.update(&[0]);
828        }
829
830        hasher.update(&[self.status.to_byte()]);
831    }
832
833    /// Hash the appended git-fidelity block (#565). Mirrors
834    /// [`Self::hash_len_fidelity`]. Kept separate from
835    /// [`Self::update_hash_core`] so the migration-only pre-bump hash can omit
836    /// it exactly.
837    ///
838    /// git-fidelity fields (#564 de-lossy step 1, #565) are DELIBERATELY part
839    /// of the content hash — the opposite of the W1 tail fields. Two git
840    /// commits that differ only in committer, author/committer time, timezone,
841    /// verbatim message, or extra headers (gpgsig included) are distinct git
842    /// objects; folding these into identity prevents them from dedup-colliding
843    /// to one State in the content-addressed store. This re-hashes every
844    /// pre-#565 state (a real format bump; acceptable pre-0.3). Keep this in
845    /// sync with `hash_len_fidelity`.
846    fn update_hash_fidelity(&self, hasher: &mut blake3::Hasher) {
847        if let Some(committer) = &self.committer {
848            hasher.update(&[1]);
849            hasher.update(&committer.name);
850            hasher.update(&[0]);
851            hasher.update(&committer.email);
852            hasher.update(&[0]);
853        } else {
854            hasher.update(&[0]);
855        }
856
857        hasher.update(&self.authored_tz_offset.to_le_bytes());
858        hasher.update(&self.committer_tz_offset.to_le_bytes());
859
860        // Author time (#564): committer time is hashed above as created_at;
861        // author time is the other half of a git commit's temporal identity.
862        if let Some(authored_at) = self.authored_at {
863            hasher.update(&[1]);
864            hasher.update(&authored_at.timestamp().to_le_bytes());
865        } else {
866            hasher.update(&[0]);
867        }
868
869        write_optional_bytes(hasher, &self.raw_message);
870
871        // extra_headers (gpgsig is one of these, kept inline at its position).
872        hasher.update(&(self.extra_headers.len() as u32).to_le_bytes());
873        for (key, value) in &self.extra_headers {
874            hasher.update(&(key.len() as u32).to_le_bytes());
875            hasher.update(key);
876            hasher.update(&(value.len() as u32).to_le_bytes());
877            hasher.update(value);
878        }
879        hasher.update(&(self.lineage.len() as u32).to_le_bytes());
880        for lineage in &self.lineage {
881            hasher.update(&[lineage.kind.to_byte()]);
882            hasher.update(lineage.source_change.as_bytes());
883            hasher.update(lineage.source_state.as_bytes());
884        }
885    }
886
887    fn refresh_state_id(&mut self) {
888        self.state_id = StateId::from_content_hash(self.compute_hash());
889    }
890}
891
892/// Length-prefixed optional-bytes framing for the hash: `[1] + u32-LE len +
893/// bytes` when `Some`, a single `[0]` when `None`. Unlike
894/// [`write_optional_string`]'s NUL-terminated framing this is binary-safe —
895/// `raw_message` can contain NUL bytes, so a length prefix (not a terminator)
896/// is required to keep the hash unambiguous.
897fn write_optional_bytes(hasher: &mut blake3::Hasher, value: &Option<Vec<u8>>) {
898    match value {
899        Some(bytes) => {
900            hasher.update(&[1]);
901            hasher.update(&(bytes.len() as u32).to_le_bytes());
902            hasher.update(bytes);
903        }
904        None => {
905            hasher.update(&[0]);
906        }
907    }
908}
909
910fn write_optional_string(hasher: &mut blake3::Hasher, value: &Option<String>) {
911    match value {
912        Some(value) => {
913            hasher.update(&[1]);
914            hasher.update(value.as_bytes());
915            hasher.update(&[0]);
916        }
917        None => {
918            hasher.update(&[0]);
919        }
920    }
921}
922
923/// Parse the *extension* headers from a raw git commit object's content bytes
924/// (the bytes `git cat-file commit <sha>` prints — i.e. gix's `Commit::data`),
925/// in their exact on-the-wire order, ready to store in [`State::extra_headers`].
926///
927/// A commit's header block runs from the start of the content up to the first
928/// blank line (the header/body separator). Its leading headers are always, in
929/// fixed order, `tree`, zero-or-more `parent`, `author`, `committer`; Heddle
930/// models those natively. Every header **after** `committer` is an extension
931/// header (`encoding`, `gpgsig`, `mergetag`, or any unknown/future name) and is
932/// returned here as a `(name, value)` byte pair at its real position.
933///
934/// **This is the single source of truth for extension-header order and bytes.**
935/// Both git import paths (the CLI bridge and the ingest walker) build
936/// `extra_headers` from it. The alternative — stitching the vec back together
937/// from a decoder's *typed* accessors (gix surfaces `encoding`, and historically
938/// `gpgsig`, as fields *outside* its `extra_headers`) — silently reorders the
939/// headers git happens to model as typed fields, which breaks #566 byte-exact
940/// reconstruction. So we never consult those typed accessors for position; the
941/// raw header block is authoritative. (#564 de-lossy step 1 — close-the-class.)
942///
943/// Folded continuation lines (a value line beginning with a single space
944/// `0x20`, used by `gpgsig`/`mergetag`) are **unfolded**: each continuation
945/// contributes a `\n` plus the line with exactly one leading space stripped, so
946/// the stored value holds the value's real internal newlines with no trailing
947/// newline. The serializer (#566) re-folds by mapping every `\n` back to `\n `
948/// (spike §2). A "blank" line inside an armored value is ` \n` on the wire (one
949/// space), so it unfolds to an empty segment — never confused with the
950/// header/body separator, which is a truly empty line.
951pub fn parse_commit_extension_headers(commit_content: &[u8]) -> Vec<(Vec<u8>, Vec<u8>)> {
952    // The header block ends at the first *empty* line. Folded "blank" lines
953    // inside an armored value are ` \n` (a single space), never empty, so the
954    // first `\n\n` reliably marks the header/body boundary.
955    let header_block = match find_subslice(commit_content, b"\n\n") {
956        Some(idx) => &commit_content[..idx],
957        // No separator (malformed / header-only) — treat all of it as headers.
958        None => commit_content,
959    };
960
961    // Collect every logical header (name, unfolded value) in order; the
962    // extension headers are the ones after the `committer` line.
963    let mut headers: Vec<(Vec<u8>, Vec<u8>)> = Vec::new();
964    for line in header_block.split(|&b| b == b'\n') {
965        if line.first() == Some(&b' ') {
966            // Continuation of the current header value: restore the newline
967            // that folding replaced and strip exactly one leading space.
968            if let Some((_, value)) = headers.last_mut() {
969                value.push(b'\n');
970                value.extend_from_slice(&line[1..]);
971            }
972            // A continuation with no preceding header is malformed git; skip it
973            // rather than panic.
974            continue;
975        }
976        // New header: `name<SP>value`. A header line with no space is degenerate
977        // (git never emits one in this region) — record it with an empty value
978        // so no bytes are silently dropped.
979        let (name, value) = match line.iter().position(|&b| b == b' ') {
980            Some(sp) => (line[..sp].to_vec(), line[sp + 1..].to_vec()),
981            None => (line.to_vec(), Vec::new()),
982        };
983        headers.push((name, value));
984    }
985
986    // Extension headers are everything strictly after `committer`. git always
987    // emits exactly one committer line ahead of the extension headers; if it is
988    // somehow absent, fall back to excluding the four core names so nothing is
989    // silently dropped or mis-captured.
990    match headers.iter().position(|(name, _)| name == b"committer") {
991        Some(idx) => headers.split_off(idx + 1),
992        None => headers
993            .into_iter()
994            .filter(|(name, _)| {
995                !matches!(
996                    name.as_slice(),
997                    b"tree" | b"parent" | b"author" | b"committer"
998                )
999            })
1000            .collect(),
1001    }
1002}
1003
1004/// Index of the first occurrence of `needle` in `haystack`, or `None`.
1005fn find_subslice(haystack: &[u8], needle: &[u8]) -> Option<usize> {
1006    if needle.is_empty() || needle.len() > haystack.len() {
1007        return None;
1008    }
1009    haystack.windows(needle.len()).position(|w| w == needle)
1010}
1011
1012#[cfg(test)]
1013mod tests {
1014    use super::*;
1015    use crate::object::Principal;
1016
1017    fn sample_attribution() -> Attribution {
1018        Attribution::human(Principal::new("Alice", "alice@example.com"))
1019    }
1020
1021    #[test]
1022    fn format4_agent_states_keep_pre_cursor_id_when_cursor_fields_are_unpublished() {
1023        use crate::object::Agent;
1024
1025        let created_at = DateTime::from_timestamp(1_700_000_000, 0).expect("fixed test timestamp");
1026        let tree = ContentHash::from_bytes([11; 32]);
1027        let mut agent_state = State::new(
1028            tree,
1029            Vec::new(),
1030            Attribution::with_agent(
1031                Principal::new("Author", "author@example.com"),
1032                Agent::new("anthropic", "opus"),
1033            ),
1034        );
1035        agent_state.created_at = created_at;
1036        agent_state.state_id = agent_state.id();
1037        assert_ne!(
1038            agent_state.id(),
1039            agent_state.pre_cursor_id(),
1040            "None cursor tags must change the current id of every agent state"
1041        );
1042        assert!(
1043            agent_state.accepts_stored_id(&agent_state.pre_cursor_id()),
1044            "format-4 agent ids must still validate after the cursor hash bump"
1045        );
1046        assert!(agent_state.accepts_stored_id(&agent_state.id()));
1047
1048        let mut published = agent_state.clone();
1049        published.attribution.agent = Some(
1050            Agent::new("anthropic", "opus")
1051                .with_thought_level("high")
1052                .with_parent("agent-1"),
1053        );
1054        published.state_id = published.id();
1055        assert!(
1056            !published.accepts_stored_id(&agent_state.pre_cursor_id()),
1057            "published cursor fields must not validate against a format-4 id"
1058        );
1059        assert_ne!(published.id(), agent_state.id());
1060
1061        let mut human = State::new(
1062            tree,
1063            Vec::new(),
1064            Attribution::human(Principal::new("Author", "author@example.com")),
1065        );
1066        human.created_at = created_at;
1067        assert_eq!(
1068            human.id(),
1069            human.pre_cursor_id(),
1070            "human states never hashed the agent cursor tags"
1071        );
1072        assert_eq!(
1073            agent_state.hash_for_stored_id(&agent_state.pre_cursor_id()),
1074            agent_state.compute_pre_cursor_hash(),
1075            "accepted format-4 ids must verify against the preserved hash"
1076        );
1077        assert_eq!(
1078            agent_state.hash_for_stored_id(&agent_state.id()),
1079            agent_state.compute_hash()
1080        );
1081    }
1082
1083    #[test]
1084    fn new_snapshot_sets_fresh_logical_identity() {
1085        let state =
1086            State::new_snapshot(ContentHash::compute(b"tree"), vec![], sample_attribution());
1087        assert!(!state.change_id.is_zero());
1088        assert_eq!(state.state_id, state.id());
1089    }
1090
1091    #[test]
1092    fn new_refresh_preserves_explicit_logical_identity() {
1093        let logical_change_id = ChangeId::from_bytes([7; 16]);
1094        let state = State::new_refresh_of(
1095            ContentHash::compute(b"tree"),
1096            vec![],
1097            sample_attribution(),
1098            logical_change_id,
1099        );
1100        assert_eq!(state.change_id, logical_change_id);
1101    }
1102
1103    #[test]
1104    fn new_merge_uses_fresh_logical_identity() {
1105        let state = State::new_merge(
1106            ContentHash::compute(b"tree"),
1107            vec![StateId::from_bytes([1; 32]), StateId::from_bytes([2; 32])],
1108            sample_attribution(),
1109        );
1110        assert!(!state.change_id.is_zero());
1111        assert!(state.is_merge());
1112    }
1113
1114    #[test]
1115    fn with_change_id_invalidates_cached_hash_when_logical_identity_changes() {
1116        let mut state =
1117            State::new_snapshot(ContentHash::compute(b"tree"), vec![], sample_attribution());
1118        let original_hash = state.hash();
1119        let replacement = ChangeId::from_bytes([9; 16]);
1120
1121        let mut updated = state.with_change_id(replacement);
1122
1123        assert_eq!(updated.change_id, replacement);
1124        assert_ne!(updated.hash(), original_hash);
1125        assert_eq!(updated.hash(), updated.compute_hash());
1126    }
1127
1128    #[test]
1129    fn agent_segment_is_part_of_state_hash() {
1130        let principal = Principal::new("Alice", "alice@example.com");
1131        let attribution_a = Attribution::with_agent(
1132            principal.clone(),
1133            crate::object::Agent::new("openai", "gpt-5").with_session("sess-1", "seg-1"),
1134        );
1135        let attribution_b = Attribution::with_agent(
1136            principal,
1137            crate::object::Agent::new("openai", "gpt-5").with_session("sess-1", "seg-2"),
1138        );
1139        let tree = ContentHash::compute(b"tree");
1140        let timestamp = Utc::now();
1141        let logical_change_id = ChangeId::from_bytes([3; 16]);
1142        let state_a = State::new_snapshot(tree, vec![], attribution_a)
1143            .with_change_id(logical_change_id)
1144            .with_timestamp(timestamp);
1145        let state_b = State::new_snapshot(tree, vec![], attribution_b)
1146            .with_change_id(logical_change_id)
1147            .with_timestamp(timestamp);
1148
1149        assert_ne!(state_a.compute_hash(), state_b.compute_hash());
1150    }
1151
1152    #[test]
1153    fn agent_segment_is_included_in_state_hash_length_prefix() {
1154        let state = State::new_snapshot(
1155            ContentHash::compute(b"tree"),
1156            vec![],
1157            Attribution::with_agent(
1158                Principal::new("Alice", "alice@example.com"),
1159                crate::object::Agent::new("openai", "gpt-5").with_session("sess-1", "segment-1"),
1160            ),
1161        );
1162        let segment_len = "segment-1".len() as u64 + 2;
1163        let missing_segment_len_hash = ContentHash::compute_typed_with_len(
1164            "state",
1165            state.hash_len() - segment_len,
1166            |hasher| state.update_hash(hasher),
1167        );
1168
1169        assert_ne!(
1170            state.compute_hash(),
1171            missing_segment_len_hash,
1172            "segment_id's option tag, bytes, and terminator must affect the typed length prefix",
1173        );
1174    }
1175
1176    fn sample_state() -> State {
1177        State::new_snapshot(ContentHash::compute(b"tree"), vec![], sample_attribution())
1178    }
1179
1180    fn assert_mutator_invalidates_cached_hash(
1181        mut state: State,
1182        mutate: impl FnOnce(State) -> State,
1183    ) {
1184        let original_hash = state.hash();
1185        let mut updated = mutate(state);
1186        assert_ne!(updated.hash(), original_hash);
1187        assert_eq!(updated.hash(), updated.compute_hash());
1188    }
1189
1190    #[test]
1191    fn with_intent_invalidates_cached_hash() {
1192        assert_mutator_invalidates_cached_hash(sample_state(), |state| {
1193            state.with_intent("capture intent")
1194        });
1195    }
1196
1197    #[test]
1198    fn with_confidence_invalidates_cached_hash() {
1199        assert_mutator_invalidates_cached_hash(sample_state(), |state| state.with_confidence(0.9));
1200    }
1201
1202    #[test]
1203    fn with_verification_invalidates_cached_hash() {
1204        assert_mutator_invalidates_cached_hash(sample_state(), |state| {
1205            state.with_verification(Verification::new().with_tests_passed(true))
1206        });
1207    }
1208
1209    #[test]
1210    fn with_status_invalidates_cached_hash() {
1211        assert_mutator_invalidates_cached_hash(sample_state(), |state| {
1212            state.with_status(Status::Published)
1213        });
1214    }
1215
1216    #[test]
1217    fn with_timestamp_invalidates_cached_hash() {
1218        assert_mutator_invalidates_cached_hash(sample_state(), |state| {
1219            state.with_timestamp(Utc::now() + chrono::Duration::seconds(1))
1220        });
1221    }
1222
1223    /// The git-fidelity fields (#564 step 1) MUST be part of the hash so two
1224    /// git-distinct commits can't dedup-collide. Each field, set in
1225    /// isolation, must move the hash.
1226    #[test]
1227    fn fidelity_fields_are_part_of_state_hash() {
1228        let base = sample_state();
1229        let base_hash = base.compute_hash();
1230
1231        let with_committer = sample_state().with_change_id(base.change_id);
1232        let mut with_committer =
1233            with_committer.with_committer(Principal::new("Carol", "carol@example.com"));
1234        with_committer.created_at = base.created_at;
1235        assert_ne!(
1236            with_committer.hash(),
1237            base_hash,
1238            "committer must affect the state hash"
1239        );
1240
1241        for mutate in [
1242            |s: State| s.with_tz_offsets(3600, -7200),
1243            |s: State| s.with_authored_at(Utc::now() + chrono::Duration::seconds(1)),
1244            |s: State| s.with_raw_message("verbatim body\n"),
1245            // gpgsig now rides inline in extra_headers at its captured position.
1246            |s: State| {
1247                s.with_extra_headers(vec![(
1248                    b"gpgsig".to_vec(),
1249                    b"-----BEGIN PGP SIGNATURE-----\n".to_vec(),
1250                )])
1251            },
1252            |s: State| s.with_extra_headers(vec![(b"mergetag".to_vec(), b"x".to_vec())]),
1253        ] {
1254            let seeded = sample_state().with_change_id(base.change_id);
1255            let mut decorated = mutate(seeded);
1256            decorated.created_at = base.created_at;
1257            assert_ne!(
1258                decorated.hash(),
1259                base_hash,
1260                "fidelity field must affect the state hash"
1261            );
1262        }
1263    }
1264
1265    /// extra_headers order is load-bearing (#566): the same pairs in a
1266    /// different order must hash differently.
1267    #[test]
1268    fn extra_headers_order_affects_hash() {
1269        let base = sample_state();
1270        let one = sample_state().with_change_id(base.change_id);
1271        let mut one = one.with_extra_headers(vec![
1272            (b"a".to_vec(), b"1".to_vec()),
1273            (b"b".to_vec(), b"2".to_vec()),
1274        ]);
1275        one.created_at = base.created_at;
1276
1277        let two = sample_state().with_change_id(base.change_id);
1278        let mut two = two.with_extra_headers(vec![
1279            (b"b".to_vec(), b"2".to_vec()),
1280            (b"a".to_vec(), b"1".to_vec()),
1281        ]);
1282        two.created_at = base.created_at;
1283
1284        assert_ne!(one.hash(), two.hash());
1285    }
1286
1287    /// The fidelity fields set together produce a stable, recomputable
1288    /// hash (guards against a `hash_len`/`update_hash` divergence making
1289    /// the cached hash differ from a fresh `compute_hash`).
1290    #[test]
1291    fn fidelity_fields_hash_is_stable() {
1292        let mut state = sample_state()
1293            .with_committer(Principal::new("Dave", "dave@example.com"))
1294            .with_tz_offsets(3600, 0)
1295            .with_authored_at(Utc::now())
1296            .with_raw_message("body\n")
1297            .with_extra_headers(vec![
1298                (b"gpgsig".to_vec(), b"sig".to_vec()),
1299                (b"k".to_vec(), b"v".to_vec()),
1300            ]);
1301        assert_eq!(state.hash(), state.compute_hash());
1302    }
1303
1304    /// A non-UTF8 git message body (latin-1 `café` = `caf\xe9`) must be
1305    /// stored byte-identically. `raw_message` is `Vec<u8>`, not `String`,
1306    /// precisely so these bytes survive; the hash stays stable/recomputable
1307    /// over the raw bytes (length-prefixed framing, NUL-safe). #564 step 1.
1308    #[test]
1309    fn non_utf8_raw_message_is_byte_preserved() {
1310        let raw = b"caf\xe9\n".to_vec();
1311        assert!(
1312            String::from_utf8(raw.clone()).is_err(),
1313            "test fixture must be invalid UTF-8 to be meaningful"
1314        );
1315        let mut state = sample_state().with_raw_message(&raw);
1316        assert_eq!(
1317            state.raw_message.as_deref(),
1318            Some(raw.as_slice()),
1319            "raw bytes preserved verbatim"
1320        );
1321        // rmp serialize → deserialize (the store's on-disk codec) keeps the
1322        // bytes intact, and the hash recomputes identically afterwards.
1323        let bytes = rmp_serde::to_vec(&state).expect("serialize state");
1324        let back: State = rmp_serde::from_slice(&bytes).expect("deserialize state");
1325        assert_eq!(back.raw_message.as_deref(), Some(raw.as_slice()));
1326        let mut back = back;
1327        assert_eq!(state.hash(), back.hash());
1328        assert_eq!(back.hash(), back.compute_hash());
1329    }
1330
1331    /// A NUL byte inside the message must not be swallowed/truncated by the
1332    /// hash framing — length-prefixed `raw_message` is what makes this safe,
1333    /// where the old NUL-terminated string framing would have been ambiguous.
1334    #[test]
1335    fn raw_message_with_nul_byte_changes_hash() {
1336        let base = sample_state();
1337        let with_nul = sample_state().with_change_id(base.change_id);
1338        let mut a = with_nul.with_raw_message(b"a\x00b");
1339        a.created_at = base.created_at;
1340
1341        let other = sample_state().with_change_id(base.change_id);
1342        let mut b = other.with_raw_message(b"a\x00c");
1343        b.created_at = base.created_at;
1344
1345        assert_ne!(a.hash(), b.hash());
1346    }
1347
1348    /// Close-the-class conformance: extension headers are captured from the
1349    /// raw commit header block in their EXACT on-the-wire order, regardless of
1350    /// which ones a decoder would surface as typed fields. A commit whose
1351    /// optional headers are in non-canonical order — `x-custom`, then a folded
1352    /// `gpgsig`, then `encoding`, then a folded `mergetag` — must reproduce that
1353    /// exact ordered `(name, value)` byte sequence. This fails if any header is
1354    /// reordered, prepended, appended, or dropped. #564 de-lossy step 1.
1355    #[test]
1356    fn parse_extension_headers_preserves_noncanonical_wire_order() {
1357        // A folded `mergetag` value carries a full tag object, which itself has
1358        // an internal blank line between the tag headers and the tag message —
1359        // on the wire that blank line is folded to a single space (` `), NEVER
1360        // an empty line, so it must not be mistaken for the header/body split.
1361        // Built line-by-line (NOT a `\`-continued literal, which would eat the
1362        // load-bearing leading space on each folded continuation line).
1363        let lines: &[&[u8]] = &[
1364            b"tree 1111111111111111111111111111111111111111",
1365            b"parent 2222222222222222222222222222222222222222",
1366            b"author Alice <alice@example.com> 1700000000 +0000",
1367            b"committer Bob <bob@example.com> 1700000100 +0000",
1368            b"x-custom custom value",
1369            b"gpgsig -----BEGIN PGP SIGNATURE-----",
1370            b" sig-line-1",
1371            b" -----END PGP SIGNATURE-----",
1372            b"encoding ISO-8859-1",
1373            b"mergetag object 3333333333333333333333333333333333333333",
1374            b" type commit",
1375            b" tag sidetag",
1376            b" tagger Carol <carol@example.com> 1700000050 +0000",
1377            b" ", // folded blank line inside the tag object (one space)
1378            b" signed side tag",
1379            b"", // the real header/body separator (empty line)
1380            b"the commit message",
1381            b"",
1382        ];
1383        let content = lines.join(&b'\n');
1384
1385        let headers = parse_commit_extension_headers(&content);
1386
1387        let expected: Vec<(Vec<u8>, Vec<u8>)> = vec![
1388            (b"x-custom".to_vec(), b"custom value".to_vec()),
1389            (
1390                b"gpgsig".to_vec(),
1391                // Unfolded: internal newlines restored, NO trailing newline (the
1392                // serializer re-folds each `\n` to `\n `, spike §2).
1393                b"-----BEGIN PGP SIGNATURE-----\nsig-line-1\n-----END PGP SIGNATURE-----"
1394                    .to_vec(),
1395            ),
1396            (b"encoding".to_vec(), b"ISO-8859-1".to_vec()),
1397            (
1398                b"mergetag".to_vec(),
1399                // The folded ` \n` blank line unfolds to an empty segment, so the
1400                // tag object's header/message split survives as a real `\n\n`.
1401                b"object 3333333333333333333333333333333333333333\ntype commit\ntag sidetag\ntagger Carol <carol@example.com> 1700000050 +0000\n\nsigned side tag".to_vec(),
1402            ),
1403        ];
1404
1405        assert_eq!(headers, expected);
1406    }
1407
1408    /// A commit with no extension headers (the common case) yields an empty
1409    /// vec — `tree`/`parent`/`author`/`committer` are modelled natively and
1410    /// never leak into `extra_headers`.
1411    #[test]
1412    fn parse_extension_headers_empty_when_only_core_headers() {
1413        let content: &[u8] = b"\
1414tree 1111111111111111111111111111111111111111\n\
1415author Alice <alice@example.com> 1700000000 +0000\n\
1416committer Bob <bob@example.com> 1700000100 +0000\n\
1417\n\
1418just a message\n";
1419        assert!(parse_commit_extension_headers(content).is_empty());
1420    }
1421}