gix_object/lib.rs
1//! This crate provides types for [read-only git objects][crate::ObjectRef] backed by bytes provided in git's serialization format
2//! as well as [mutable versions][Object] of these. Both types of objects can be encoded.
3//!
4//! ## Decode Borrowed Objects
5//!
6//! ```
7//! let object = gix_object::ObjectRef::from_loose(b"blob 5\0hello", gix_hash::Kind::Sha1).unwrap();
8//! let blob = object.as_blob().unwrap();
9//!
10//! assert_eq!(blob.data, b"hello");
11//! assert_eq!(object.kind(), gix_object::Kind::Blob);
12//! ```
13//!
14//! ## Mutate And Encode Owned Objects
15//!
16//! ```
17//! use gix_object::WriteTo;
18//!
19//! let object = gix_object::ObjectRef::from_loose(b"blob 5\0hello", gix_hash::Kind::Sha1)
20//! .unwrap()
21//! .into_owned()
22//! .unwrap();
23//! let mut blob = object.into_blob();
24//! blob.data.extend_from_slice(b" world");
25//!
26//! let mut out = Vec::new();
27//! blob.write_to(&mut out).unwrap();
28//! assert_eq!(out, b"hello world");
29//! assert_eq!(blob.loose_header().as_slice(), b"blob 11\0");
30//! ```
31//! ## Feature Flags
32#![cfg_attr(
33 all(doc, feature = "document-features"),
34 doc = ::document_features::document_features!()
35)]
36#![cfg_attr(all(doc, feature = "document-features"), feature(doc_cfg))]
37#![deny(missing_docs)]
38#![forbid(unsafe_code)]
39
40use gix_error::ExnMessageResult;
41use std::borrow::Cow;
42
43use gix_error::ExnResult;
44
45/// For convenience to allow using `bstr` without adding it to own cargo manifest.
46pub use bstr;
47use bstr::{BStr, BString, ByteSlice};
48/// For convenience to allow using `gix-date` without adding it to own cargo manifest.
49pub use gix_date as date;
50use smallvec::SmallVec;
51
52///
53pub mod commit;
54mod object;
55/// Cryptographic signature discovery and, with the `signature` feature, external signing and verification.
56pub mod signature;
57///
58pub mod tag;
59///
60pub mod tree;
61
62mod blob;
63///
64pub mod data;
65
66///
67pub mod find;
68
69mod traits;
70pub use traits::{Exists, Find, FindExt, FindObjectOrHeader, Header as FindHeader, HeaderExt, Write, WriteTo};
71
72pub mod encode;
73pub(crate) mod parse;
74
75///
76pub mod kind;
77
78/// The four types of objects that git differentiates.
79#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
80#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone, Copy)]
81#[expect(missing_docs)]
82pub enum Kind {
83 Tree,
84 Blob,
85 Commit,
86 Tag,
87}
88/// A chunk of any [`data`](BlobRef::data).
89#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone, Copy)]
90#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
91pub struct BlobRef<'a> {
92 /// The bytes themselves.
93 pub data: &'a [u8],
94}
95
96/// A mutable chunk of any [`data`](Blob::data).
97#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone)]
98#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
99pub struct Blob {
100 /// The data itself.
101 pub data: Vec<u8>,
102}
103
104/// A git commit parsed using [`from_bytes()`](CommitRef::from_bytes()).
105///
106/// A commit encapsulates information about a point in time at which the state of the repository is recorded, usually after a
107/// change which is documented in the commit `message`.
108#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone)]
109#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
110pub struct CommitRef<'a> {
111 /// HEX hash of tree object we point to.
112 ///
113 /// Use [`tree()`](CommitRef::tree()) to obtain a decoded version of it.
114 #[cfg_attr(feature = "serde", serde(borrow))]
115 pub tree: &'a BStr,
116 /// HEX hash of each parent commit. Empty for first commit in repository.
117 pub parents: SmallVec<[&'a BStr; 1]>,
118 /// The raw author header value as encountered during parsing.
119 ///
120 /// Use the [`author()`](CommitRef::author()) method to obtain a parsed version of it.
121 #[cfg_attr(feature = "serde", serde(borrow))]
122 pub author: &'a BStr,
123 /// The raw committer header value as encountered during parsing.
124 ///
125 /// Use the [`committer()`](CommitRef::committer()) method to obtain a parsed version of it.
126 #[cfg_attr(feature = "serde", serde(borrow))]
127 pub committer: &'a BStr,
128 /// The name of the message encoding, otherwise [UTF-8 should be assumed](https://github.com/git/git/blob/e67fbf927dfdf13d0b21dc6ea15dc3c7ef448ea0/commit.c#L1493:L1493).
129 pub encoding: Option<&'a BStr>,
130 /// The commit message documenting the change.
131 pub message: &'a BStr,
132 /// Extra header fields, in order of them being encountered, made accessible with the iterator returned by [`extra_headers()`](CommitRef::extra_headers()).
133 pub extra_headers: Vec<(&'a BStr, Cow<'a, BStr>)>,
134}
135
136/// Like [`CommitRef`], but as `Iterator` to support (up to) entirely allocation free parsing.
137/// It's particularly useful to traverse the commit graph without ever allocating arrays for parents.
138#[derive(Copy, Clone)]
139pub struct CommitRefIter<'a> {
140 data: &'a [u8],
141 state: commit::ref_iter::State,
142 hash_kind: gix_hash::Kind,
143}
144
145/// A mutable git commit, representing an annotated state of a working tree along with a reference to its historical commits.
146#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone)]
147#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
148pub struct Commit {
149 /// The hash of recorded working tree state.
150 pub tree: gix_hash::ObjectId,
151 /// Hash of each parent commit. Empty for the first commit in repository.
152 pub parents: SmallVec<[gix_hash::ObjectId; 1]>,
153 /// Who wrote this commit.
154 pub author: gix_actor::Signature,
155 /// Who committed this commit.
156 ///
157 /// This may be different from the `author` in case the author couldn't write to the repository themselves and
158 /// is commonly encountered with contributed commits.
159 pub committer: gix_actor::Signature,
160 /// The name of the message encoding, otherwise [UTF-8 should be assumed](https://github.com/git/git/blob/e67fbf927dfdf13d0b21dc6ea15dc3c7ef448ea0/commit.c#L1493:L1493).
161 pub encoding: Option<BString>,
162 /// The commit message documenting the change.
163 pub message: BString,
164 /// Extra header fields, in order of them being encountered, made accessible with the iterator returned
165 /// by [`extra_headers()`](Commit::extra_headers()).
166 pub extra_headers: Vec<(BString, BString)>,
167}
168
169/// Represents a git tag, commonly indicating a software release.
170#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone, Copy)]
171#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
172pub struct TagRef<'a> {
173 /// The hash in hexadecimal being the object this tag points to. Use [`target()`](TagRef::target()) to obtain a byte representation.
174 #[cfg_attr(feature = "serde", serde(borrow))]
175 pub target: &'a BStr,
176 /// The kind of object that `target` points to.
177 pub target_kind: Kind,
178 /// The name of the tag, e.g. "v1.0".
179 pub name: &'a BStr,
180 /// The raw tagger header value as encountered during parsing.
181 ///
182 /// Use the [`tagger()`](TagRef::tagger()) method to obtain a parsed version of it.
183 #[cfg_attr(feature = "serde", serde(borrow))]
184 pub tagger: Option<&'a BStr>,
185 /// The message describing this release.
186 pub message: &'a BStr,
187 /// Any Git-supported in-body cryptographic signature.
188 ///
189 /// Use [`signature()`](TagRef::signature()) to also obtain its detected format.
190 pub signature: Option<&'a BStr>,
191}
192
193/// Like [`TagRef`], but as `Iterator` to support entirely allocation free parsing.
194/// It's particularly useful to dereference only the target chain.
195#[derive(Copy, Clone)]
196pub struct TagRefIter<'a> {
197 data: &'a [u8],
198 state: tag::ref_iter::State,
199 hash_kind: gix_hash::Kind,
200}
201
202/// A mutable git tag.
203#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone)]
204#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
205pub struct Tag {
206 /// The hash this tag is pointing to.
207 pub target: gix_hash::ObjectId,
208 /// The kind of object this tag is pointing to.
209 pub target_kind: Kind,
210 /// The name of the tag, e.g. "v1.0".
211 pub name: BString,
212 /// The tags author.
213 pub tagger: Option<gix_actor::Signature>,
214 /// The message describing the tag.
215 pub message: BString,
216 /// Any Git-supported in-body cryptographic signature.
217 ///
218 /// Use [`signature()`](Tag::signature()) to also obtain its detected format.
219 pub signature: Option<BString>,
220}
221
222/// Immutable objects are read-only structures referencing most data from [a byte slice](ObjectRef::from_bytes()).
223///
224/// Immutable objects are expected to be deserialized from bytes that acts as backing store, and they
225/// cannot be mutated or serialized. Instead, one will [convert](ObjectRef::into_owned()) them into their [`mutable`](Object) counterparts
226/// which support mutation and serialization.
227///
228/// An `ObjectRef` is representing [`Trees`](TreeRef), [`Blobs`](BlobRef), [`Commits`](CommitRef), or [`Tags`](TagRef).
229#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone)]
230#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
231#[expect(missing_docs)]
232pub enum ObjectRef<'a> {
233 #[cfg_attr(feature = "serde", serde(borrow))]
234 Tree(TreeRef<'a>),
235 Blob(BlobRef<'a>),
236 Commit(CommitRef<'a>),
237 Tag(TagRef<'a>),
238}
239
240/// Mutable objects with each field being separately allocated and changeable.
241///
242/// Mutable objects are Commits, Trees, Blobs and Tags that can be changed and serialized.
243///
244/// They either created using object [construction](Object) or by [deserializing existing objects](ObjectRef::from_bytes())
245/// and converting these [into mutable copies](ObjectRef::into_owned()) for adjustments.
246///
247/// An `Object` is representing [`Trees`](Tree), [`Blobs`](Blob), [`Commits`](Commit), or [`Tags`](Tag).
248#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone)]
249#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
250#[expect(missing_docs)]
251pub enum Object {
252 Tree(Tree),
253 Blob(Blob),
254 Commit(Commit),
255 Tag(Tag),
256}
257/// A directory snapshot containing files (blobs), directories (trees) and submodules (commits).
258#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone)]
259#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
260pub struct TreeRef<'a> {
261 /// The directories and files contained in this tree.
262 ///
263 /// Beware that the sort order isn't *quite* by name, so one may bisect only with a [`tree::EntryRef`] to handle ordering correctly.
264 #[cfg_attr(feature = "serde", serde(borrow))]
265 pub entries: Vec<tree::EntryRef<'a>>,
266}
267
268/// A directory snapshot containing files (blobs), directories (trees) and submodules (commits), lazily evaluated.
269#[derive(Default, PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone, Copy)]
270pub struct TreeRefIter<'a> {
271 /// The hash kind to use for parsing this tree.
272 hash_kind: gix_hash::Kind,
273 /// The directories and files contained in this tree.
274 data: &'a [u8],
275}
276
277/// A mutable Tree, containing other trees, blobs or commits.
278#[derive(Default, PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone)]
279#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
280pub struct Tree {
281 /// The directories and files contained in this tree. They must be and remain sorted by [`filename`][tree::Entry::filename].
282 ///
283 /// Beware that the sort order isn't *quite* by name, so one may bisect only with a [`tree::Entry`] to handle ordering correctly.
284 pub entries: Vec<tree::Entry>,
285}
286
287impl Tree {
288 /// Return an empty tree which serializes to a well-known hash
289 pub fn empty() -> Self {
290 Tree { entries: Vec::new() }
291 }
292}
293
294/// A borrowed object using a slice as backing buffer, or in other words a bytes buffer that knows the kind of object it represents.
295#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone, Copy)]
296pub struct Data<'a> {
297 /// kind of object
298 pub kind: Kind,
299 /// The hash kind to use for parsing this data.
300 pub object_hash: gix_hash::Kind,
301 /// decoded, decompressed data, owned by a backing store.
302 pub data: &'a [u8],
303}
304
305/// Information about an object, which includes its kind and the amount of bytes it would have when obtained.
306#[derive(PartialEq, Eq, Debug, Hash, Ord, PartialOrd, Clone, Copy)]
307pub struct Header {
308 /// The kind of object.
309 pub kind: Kind,
310 /// The object's size in bytes, or the size of the buffer when it's retrieved in full.
311 pub size: u64,
312}
313
314///
315pub mod decode {
316 mod error {
317 pub(crate) fn empty_error() -> gix_error::Message {
318 gix_error::validation("object parsing failed")
319 }
320 }
321
322 pub(crate) use error::empty_error;
323
324 use bstr::ByteSlice;
325 use gix_error::{ErrorExt, ExnMessageResult, ResultExt, validation};
326 /// Decode a loose object header, being `<kind> <size>\0`, returns
327 /// ([`kind`](super::Kind), `size`, `consumed bytes`).
328 ///
329 /// `size` is the uncompressed size of the payload in bytes.
330 /// Invalid kind or size fields include their bytes as `input` [metadata](gix_error::Exn::metadata()).
331 pub fn loose_header(input: &[u8]) -> ExnMessageResult<(super::Kind, u64, usize)> {
332 let kind_end = input
333 .find_byte(0x20)
334 .ok_or_else(|| validation("Expected '<type> <size>'").raise())?;
335 let kind = super::Kind::from_bytes(&input[..kind_end])
336 .or_raise(|| validation("The object header contained an unknown object kind."))?;
337 let size_end = input
338 .find_byte(0x0)
339 .ok_or_else(|| validation("Did not find 0 byte in header").raise())?;
340 let size_bytes = &input[kind_end + 1..size_end];
341 let size = gix_utils::btoi::to_signed(size_bytes)
342 .or_raise(|| validation("Object size in header could not be parsed").with("input", size_bytes))?;
343 Ok((kind, size, size_end + 1))
344 }
345}
346
347fn object_hasher(hash_kind: gix_hash::Kind, object_kind: Kind, object_size: u64) -> gix_hash::Hasher {
348 let mut hasher = gix_hash::hasher(hash_kind);
349 hasher.update(&encode::loose_header(object_kind, object_size));
350 hasher
351}
352
353/// A function to compute a hash of kind `object_hash` for an object of `object_kind` and its `data`.
354#[doc(alias = "hash_object", alias = "git2")]
355pub fn compute_hash(hash_kind: gix_hash::Kind, object_kind: Kind, data: &[u8]) -> ExnMessageResult<gix_hash::ObjectId> {
356 let mut hasher = object_hasher(hash_kind, object_kind, data.len() as u64);
357 hasher.update(data);
358 hasher.try_finalize()
359}
360
361/// A function to compute a hash of kind `object_hash` for an object of `object_kind` and its data read from `stream`
362/// which has to yield exactly `stream_len` bytes.
363/// Use `progress` to learn about progress in bytes processed and `should_interrupt` to be able to abort the operation
364/// if set to `true`.
365#[doc(alias = "hash_file", alias = "git2")]
366pub fn compute_stream_hash(
367 hash_kind: gix_hash::Kind,
368 object_kind: Kind,
369 stream: &mut dyn std::io::Read,
370 stream_len: u64,
371 progress: &mut dyn gix_features::progress::Progress,
372 should_interrupt: &std::sync::atomic::AtomicBool,
373) -> ExnResult<gix_hash::ObjectId> {
374 let hasher = object_hasher(hash_kind, object_kind, stream_len);
375 gix_hash::bytes_with_hasher(stream, stream_len, hasher, progress, should_interrupt)
376}