Skip to main content

sley_core/
lib.rs

1#![cfg_attr(not(test), deny(clippy::unwrap_used, clippy::expect_used))]
2
3use std::borrow::Borrow;
4use std::error::Error;
5use std::fmt;
6use std::ops::Deref;
7use std::path::{Path, PathBuf};
8use std::str::FromStr;
9
10mod cancel;
11pub mod diagnostics;
12
13#[cfg(feature = "fetch-profile")]
14pub mod fetch_profile;
15
16pub use cancel::{
17    AtomicCancel, CancelFlag, CancellableRead, DynCancelFlag, OperationCancelled, StreamControl,
18    cancelled_io_error, is_cancelled_error, is_cancelled_io, kill_child_if_cancelled,
19    map_cancel_io,
20};
21
22pub const UPSTREAM_GIT_COMPAT_VERSION: &str = "2.55.0";
23
24/// Maximum symbolic-ref hops git follows while resolving one ref
25/// (`refs.c` `SYMREF_MAXDEPTH`). Oracle 2.55 resolves a chain of four symrefs
26/// plus a final direct ref and reports a dangling/looped ref at five hops.
27pub const MAX_SYMREF_DEPTH: usize = 5;
28
29pub mod atomic;
30pub mod date;
31pub mod fsync;
32pub mod paths;
33pub mod precompose;
34pub mod primitives;
35pub mod refname;
36pub mod text;
37pub use precompose::{PrecomposeUnicode, has_non_ascii};
38pub use refname::{RefnameFormat, RefnameFormatError, check_refname_format};
39
40pub mod namespace;
41pub use namespace::{Namespace, ref_is_hidden, trim_hidden_ref_pattern};
42
43#[derive(Debug, Default, Clone, PartialEq, Eq)]
44pub enum DateMode {
45    #[default]
46    Default,
47    Local,
48    Raw,
49    RawLocal,
50    Unix,
51    Short,
52    ShortLocal,
53    Iso,
54    IsoLocal,
55    IsoStrict,
56    IsoStrictLocal,
57    Rfc2822,
58    Rfc2822Local,
59    Relative,
60    Human,
61    HumanLocal,
62    Strftime {
63        template: String,
64        local: bool,
65    },
66}
67
68impl DateMode {
69    pub fn parse(value: &str) -> Option<Self> {
70        if let Some(template) = value.strip_prefix("format:") {
71            return Some(Self::Strftime {
72                template: template.to_string(),
73                local: false,
74            });
75        }
76        if let Some(template) = value.strip_prefix("format-local:") {
77            return Some(Self::Strftime {
78                template: template.to_string(),
79                local: true,
80            });
81        }
82        if value == "tformat:" || value.starts_with("tformat:") {
83            return Some(Self::Strftime {
84                template: value["tformat:".len()..].to_string(),
85                local: false,
86            });
87        }
88        if value == "auto:" || value.starts_with("auto:") {
89            return Some(Self::Default);
90        }
91        Some(match value {
92            "default" => Self::Default,
93            "default-local" | "local" => Self::Local,
94            "raw" => Self::Raw,
95            "raw-local" => Self::RawLocal,
96            "unix" => Self::Unix,
97            "short" => Self::Short,
98            "short-local" => Self::ShortLocal,
99            "iso" | "iso8601" => Self::Iso,
100            "iso-local" | "iso8601-local" => Self::IsoLocal,
101            "iso-strict" | "iso8601-strict" => Self::IsoStrict,
102            "iso-strict-local" | "iso8601-strict-local" => Self::IsoStrictLocal,
103            "rfc" | "rfc2822" => Self::Rfc2822,
104            "rfc-local" | "rfc2822-local" => Self::Rfc2822Local,
105            "relative" | "relative-local" => Self::Relative,
106            "human" => Self::Human,
107            "human-local" => Self::HumanLocal,
108            _ => return None,
109        })
110    }
111
112    pub fn parse_atom_modifier(modifier: Option<&str>) -> Option<Self> {
113        modifier.map_or(Some(Self::Default), Self::parse)
114    }
115
116    pub fn render(&self, timestamp: i64, timezone: &str) -> Option<String> {
117        let tz = if self.is_local() { "+0000" } else { timezone };
118        let parts = DateParts::from_timestamp(timestamp, tz)?;
119        Some(match self {
120            Self::Default | Self::Local => {
121                let base = format!(
122                    "{} {} {} {:02}:{:02}:{:02} {}",
123                    parts.weekday,
124                    MONTHS_ABBR[(parts.month - 1) as usize],
125                    parts.day,
126                    parts.hour,
127                    parts.minute,
128                    parts.second,
129                    parts.year,
130                );
131                if self.is_local() {
132                    base
133                } else {
134                    format!("{base} {}", parts.timezone)
135                }
136            }
137            Self::Raw | Self::RawLocal => format!("{} {}", parts.timestamp, parts.timezone),
138            Self::Unix => parts.timestamp.to_string(),
139            Self::Short | Self::ShortLocal => {
140                format!("{:04}-{:02}-{:02}", parts.year, parts.month, parts.day)
141            }
142            Self::Iso | Self::IsoLocal => format!(
143                "{:04}-{:02}-{:02} {:02}:{:02}:{:02} {}",
144                parts.year,
145                parts.month,
146                parts.day,
147                parts.hour,
148                parts.minute,
149                parts.second,
150                parts.timezone,
151            ),
152            Self::IsoStrict | Self::IsoStrictLocal => format!(
153                "{:04}-{:02}-{:02}T{:02}:{:02}:{:02}{}",
154                parts.year,
155                parts.month,
156                parts.day,
157                parts.hour,
158                parts.minute,
159                parts.second,
160                strict_timezone(parts.timezone),
161            ),
162            Self::Rfc2822 | Self::Rfc2822Local => format!(
163                "{}, {} {} {:04} {:02}:{:02}:{:02} {}",
164                parts.weekday,
165                parts.day,
166                MONTHS_ABBR[(parts.month - 1) as usize],
167                parts.year,
168                parts.hour,
169                parts.minute,
170                parts.second,
171                parts.timezone,
172            ),
173            Self::Relative => relative_date(parts.timestamp),
174            Self::Human | Self::HumanLocal => format!(
175                "{} {} {} {:02}:{:02}:{:02} {} {}",
176                parts.weekday,
177                MONTHS_ABBR[(parts.month - 1) as usize],
178                parts.day,
179                parts.hour,
180                parts.minute,
181                parts.second,
182                parts.year,
183                parts.timezone,
184            ),
185            Self::Strftime { template, .. } => strftime(template, &parts),
186        })
187    }
188
189    pub fn is_local(&self) -> bool {
190        matches!(
191            self,
192            Self::Local
193                | Self::RawLocal
194                | Self::ShortLocal
195                | Self::IsoLocal
196                | Self::IsoStrictLocal
197                | Self::Rfc2822Local
198                | Self::HumanLocal
199                | Self::Strftime { local: true, .. }
200        )
201    }
202}
203
204const MONTHS_ABBR: [&str; 12] = [
205    "Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec",
206];
207
208const MONTHS_FULL: [&str; 12] = [
209    "January",
210    "February",
211    "March",
212    "April",
213    "May",
214    "June",
215    "July",
216    "August",
217    "September",
218    "October",
219    "November",
220    "December",
221];
222
223const WEEKDAYS_FULL: [&str; 7] = [
224    "Sunday",
225    "Monday",
226    "Tuesday",
227    "Wednesday",
228    "Thursday",
229    "Friday",
230    "Saturday",
231];
232
233struct DateParts<'a> {
234    timestamp: i64,
235    timezone: &'a str,
236    weekday: &'static str,
237    year: i64,
238    month: u32,
239    day: u32,
240    hour: i64,
241    minute: i64,
242    second: i64,
243}
244
245impl<'a> DateParts<'a> {
246    fn from_timestamp(timestamp: i64, timezone: &'a str) -> Option<Self> {
247        const WEEKDAYS: [&str; 7] = ["Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat"];
248        let offset_seconds = timezone_offset_seconds(timezone)?;
249        let local = timestamp + offset_seconds;
250        let days = local.div_euclid(86_400);
251        let seconds = local.rem_euclid(86_400);
252        let (year, month, day) = civil_from_days(days);
253        Some(Self {
254            timestamp,
255            timezone,
256            weekday: WEEKDAYS[(days + 4).rem_euclid(7) as usize],
257            year,
258            month,
259            day,
260            hour: seconds / 3_600,
261            minute: (seconds % 3_600) / 60,
262            second: seconds % 60,
263        })
264    }
265}
266
267fn timezone_offset_seconds(timezone: &str) -> Option<i64> {
268    if timezone.len() != 5 {
269        return None;
270    }
271    let sign = match timezone.as_bytes()[0] {
272        b'+' => 1,
273        b'-' => -1,
274        _ => return None,
275    };
276    let hours = timezone[1..3].parse::<i64>().ok()?;
277    let minutes = timezone[3..5].parse::<i64>().ok()?;
278    Some(sign * (hours * 3_600 + minutes * 60))
279}
280
281fn strict_timezone(timezone: &str) -> String {
282    let digits = timezone.strip_prefix(['+', '-']).unwrap_or(timezone);
283    if digits == "0000" {
284        "Z".to_string()
285    } else if timezone.len() == 5 {
286        format!("{}{}:{}", &timezone[..1], &timezone[1..3], &timezone[3..5])
287    } else {
288        timezone.to_string()
289    }
290}
291
292fn strftime(template: &str, parts: &DateParts<'_>) -> String {
293    let weekday_index = ["Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat"]
294        .iter()
295        .position(|day| *day == parts.weekday)
296        .unwrap_or(0);
297    let mut out = String::with_capacity(template.len());
298    let mut chars = template.chars();
299    while let Some(ch) = chars.next() {
300        if ch != '%' {
301            out.push(ch);
302            continue;
303        }
304        match chars.next() {
305            Some('Y') => out.push_str(&format!("{:04}", parts.year)),
306            Some('y') => out.push_str(&format!("{:02}", parts.year.rem_euclid(100))),
307            Some('m') => out.push_str(&format!("{:02}", parts.month)),
308            Some('d') => out.push_str(&format!("{:02}", parts.day)),
309            Some('e') => out.push_str(&format!("{:2}", parts.day)),
310            Some('H') => out.push_str(&format!("{:02}", parts.hour)),
311            Some('M') => out.push_str(&format!("{:02}", parts.minute)),
312            Some('S') => out.push_str(&format!("{:02}", parts.second)),
313            Some('b') | Some('h') => out.push_str(MONTHS_ABBR[(parts.month - 1) as usize]),
314            Some('B') => out.push_str(MONTHS_FULL[(parts.month - 1) as usize]),
315            Some('a') => out.push_str(parts.weekday),
316            Some('A') => out.push_str(WEEKDAYS_FULL[weekday_index]),
317            Some('%') => out.push('%'),
318            Some('n') => out.push('\n'),
319            Some('t') => out.push('\t'),
320            Some(other) => {
321                out.push('%');
322                out.push(other);
323            }
324            None => out.push('%'),
325        }
326    }
327    out
328}
329
330fn relative_date(timestamp: i64) -> String {
331    let now = std::time::SystemTime::now()
332        .duration_since(std::time::UNIX_EPOCH)
333        .map(|duration| duration.as_secs() as i64)
334        .unwrap_or(timestamp);
335    if timestamp > now {
336        return "in the future".to_string();
337    }
338    let diff = (now - timestamp) as u64;
339    if diff < 90 {
340        return format!("{diff} seconds ago");
341    }
342    let minutes = (diff + 30) / 60;
343    if minutes < 90 {
344        return format!("{minutes} minutes ago");
345    }
346    let hours = (diff + 1800) / 3600;
347    if hours < 36 {
348        return format!("{hours} hours ago");
349    }
350    let days = (diff + 43200) / 86400;
351    if days < 14 {
352        return format!("{days} days ago");
353    }
354    if days < 70 {
355        return format!("{} weeks ago", (days + 3) / 7);
356    }
357    if days < 365 {
358        return format!("{} months ago", (days + 15) / 30);
359    }
360    let years_scaled = (days * 10 + 183) / 365;
361    if days < 365 * 2 {
362        let months = ((days - 365) + 15) / 30;
363        if months > 0 {
364            return format!("1 year, {months} months ago");
365        }
366        return "1 year ago".to_string();
367    }
368    if years_scaled.is_multiple_of(10) {
369        format!("{} years ago", years_scaled / 10)
370    } else {
371        format!("{}.{} years ago", years_scaled / 10, years_scaled % 10)
372    }
373}
374
375use crate::date::civil_from_days;
376
377fn is_scheme_char(ch: char) -> bool {
378    ch.is_ascii_alphanumeric() || matches!(ch, '+' | '-' | '.')
379}
380
381/// Strip embedded credentials from `url` before showing it in user-facing output.
382///
383/// HTTP(S) userinfo (`user:password@host`) is replaced with `<redacted>@host`,
384/// matching trace2's `GIT_TRACE2_REDACT` behavior. Non-URL strings (remote
385/// names, file paths) are returned unchanged.
386pub fn redact_url_for_display(url: &str) -> String {
387    let mut out = String::with_capacity(url.len());
388    let mut rest = url;
389    while let Some(scheme_end) = rest.find("://") {
390        let scheme_start = rest[..scheme_end]
391            .char_indices()
392            .rev()
393            .find_map(|(idx, ch)| (!is_scheme_char(ch)).then_some(idx + ch.len_utf8()))
394            .unwrap_or(0);
395        out.push_str(&rest[..scheme_start]);
396
397        let authority_start = scheme_end + 3;
398        let authority_end = rest[authority_start..]
399            .find(|ch: char| ['/', '?', '#', ' ', '\t', '\r', '\n'].contains(&ch))
400            .map(|idx| authority_start + idx)
401            .unwrap_or(rest.len());
402        let authority = &rest[authority_start..authority_end];
403        if let Some(at) = authority.rfind('@') {
404            out.push_str(&rest[scheme_start..authority_start]);
405            out.push_str("<redacted>@");
406            out.push_str(&authority[at + 1..]);
407        } else {
408            out.push_str(&rest[scheme_start..authority_end]);
409        }
410        rest = &rest[authority_end..];
411    }
412    out.push_str(rest);
413    out
414}
415
416/// Minimal trace2 event-target support (`GIT_TRACE2_EVENT`).
417///
418/// Upstream's trace2 event target writes one JSON object per line to the file
419/// named by `GIT_TRACE2_EVENT`. sley emits only the `data` events the test
420/// suite asserts on (`test_trace2_data` greps for the contiguous
421/// `"category":"...","key":"...","value":"..."` triple), with the same field
422/// order trace2's `fn_data_fl` produces. Unset/unwritable targets are
423/// silently ignored, like upstream's best-effort tracing.
424pub mod trace2 {
425    use std::fmt::Display;
426    use std::fmt::Write as _;
427    use std::io::Write;
428    use std::path::PathBuf;
429
430    fn escape_json(raw: &str) -> String {
431        let mut out = String::with_capacity(raw.len());
432        for ch in raw.chars() {
433            match ch {
434                '"' => out.push_str("\\\""),
435                '\\' => out.push_str("\\\\"),
436                '\n' => out.push_str("\\n"),
437                '\t' => out.push_str("\\t"),
438                ch if (ch as u32) < 0x20 => {
439                    let _ = write!(out, "\\u{:04x}", ch as u32);
440                }
441                ch => out.push(ch),
442            }
443        }
444        out
445    }
446
447    enum TraceTarget {
448        Stderr,
449        Path(String),
450    }
451
452    fn trace_target(var: &str) -> Option<TraceTarget> {
453        let target = std::env::var_os(var)?.to_string_lossy().into_owned();
454        match target.as_str() {
455            "1" | "true" => Some(TraceTarget::Stderr),
456            _ if target.starts_with('/') => Some(TraceTarget::Path(target)),
457            _ => None,
458        }
459    }
460
461    fn write_target(target: &TraceTarget, bytes: &[u8]) {
462        match target {
463            TraceTarget::Stderr => {
464                let _ = std::io::stderr().write_all(bytes);
465            }
466            TraceTarget::Path(path) => {
467                if let Ok(mut file) = std::fs::OpenOptions::new()
468                    .create(true)
469                    .append(true)
470                    .open(path)
471                {
472                    let _ = file.write_all(bytes);
473                }
474            }
475        }
476    }
477
478    fn append_to_target(var: &str, line: &str) {
479        let Some(target) = trace_target(var) else {
480            return;
481        };
482        write_target(&target, format!("{line}\n").as_bytes());
483    }
484
485    fn redact_enabled() -> bool {
486        std::env::var("GIT_TRACE2_REDACT").map_or(true, |value| value != "0")
487    }
488
489    fn maybe_redact(raw: &str) -> String {
490        if redact_enabled() {
491            super::redact_url_for_display(raw)
492        } else {
493            raw.to_string()
494        }
495    }
496
497    /// Trace2 argv rendering (`sq_quote_buf_pretty` per argument): safe
498    /// arguments stay bare, empty arguments render as `''`, everything else
499    /// falls back to full sq-quote semantics. Oracle 2.55 renders the trace2
500    /// `start` line this way (`start git log -1 'v'\!'1'`).
501    fn quote_arg(arg: &str) -> String {
502        crate::text::sq_quote_pretty(arg)
503    }
504
505    fn argv0() -> String {
506        let Some(arg0) = std::env::args_os().next() else {
507            return "sley".to_string();
508        };
509        let path = PathBuf::from(arg0);
510        path.file_name()
511            .map(|name| name.to_string_lossy().into_owned())
512            .filter(|name| !name.is_empty())
513            .unwrap_or_else(|| "sley".to_string())
514    }
515
516    fn render_argv(args: &[String]) -> String {
517        let mut rendered = Vec::with_capacity(args.len() + 1);
518        rendered.push(quote_arg(&argv0()));
519        rendered.extend(args.iter().map(|arg| quote_arg(arg)));
520        rendered.join(" ")
521    }
522
523    pub fn depth() -> usize {
524        std::env::var("SLEY_TRACE2_DEPTH")
525            .ok()
526            .and_then(|value| value.parse().ok())
527            .unwrap_or(0)
528    }
529
530    fn perf_line(depth: usize, event: &str, rest: &str) {
531        append_to_target(
532            "GIT_TRACE2_PERF",
533            &format!("d{depth} | main | {event} |  |  |  |  | {rest}"),
534        );
535    }
536
537    /// Create the trace2 targets when tracing is enabled, even if this command
538    /// emits no data/region/perf events — git opens the `GIT_TRACE2_EVENT` and
539    /// `GIT_TRACE2_PERF` files at startup, so consumers (and test cleanups that
540    /// `rm` the file) can rely on their existence.
541    pub fn touch() {
542        for var in ["GIT_TRACE2", "GIT_TRACE2_EVENT", "GIT_TRACE2_PERF"] {
543            let Some(target) = trace_target(var) else {
544                continue;
545            };
546            if let TraceTarget::Path(path) = target {
547                let _ = std::fs::OpenOptions::new()
548                    .create(true)
549                    .append(true)
550                    .open(path);
551            }
552        }
553    }
554
555    /// Emit the small normal/perf `start` records that downstream tools commonly
556    /// use for argv auditing. Full trace2 lifecycle modelling remains out of
557    /// scope; these records intentionally cover the stable clone/status tests.
558    pub fn start(args: &[String]) {
559        let argv = maybe_redact(&render_argv(args));
560        append_to_target("GIT_TRACE2", &format!("start {argv}"));
561        perf_line(depth(), "start", &argv);
562    }
563
564    pub fn cmd_ancestry_at_depth(depth: usize, ancestry: &[String]) {
565        if ancestry.is_empty() {
566            return;
567        }
568        append_to_target(
569            "GIT_TRACE2",
570            &format!("cmd_ancestry {}", ancestry.join(" <- ")),
571        );
572        perf_line(
573            depth,
574            "cmd_ancestry",
575            &format!("ancestry:[{}]", ancestry.join(" ")),
576        );
577        let event_ancestry = ancestry
578            .iter()
579            .map(|name| format!("\"{}\"", escape_json(name)))
580            .collect::<Vec<_>>()
581            .join(",");
582        append_to_target(
583            "GIT_TRACE2_EVENT",
584            &format!(
585                "{{\"event\":\"cmd_ancestry\",\"sid\":\"sley\",\"thread\":\"main\",\"ancestry\":[{event_ancestry}]}}"
586            ),
587        );
588    }
589
590    pub fn cmd_name(name: &str, hierarchy: Option<&str>) {
591        let rest = match hierarchy {
592            Some(hierarchy) => format!("{name} ({hierarchy})"),
593            None => name.to_string(),
594        };
595        perf_line(depth(), "cmd_name", &rest);
596    }
597
598    pub fn cmd_name_at_depth(depth: usize, name: &str, hierarchy: Option<&str>) {
599        let rest = match hierarchy {
600            Some(hierarchy) => format!("{name} ({hierarchy})"),
601            None => name.to_string(),
602        };
603        perf_line(depth, "cmd_name", &rest);
604    }
605
606    pub fn child_start(class: &str, argv: &[String]) {
607        child_start_with_id(class, 0, argv);
608    }
609
610    /// Record the start of a particular child/worker queue consumer.
611    ///
612    /// Checkout uses stable ids for each real materialization worker.  The
613    /// normal target intentionally includes Git's `child_start[N]` spelling;
614    /// upstream's parallel-checkout probes use that record to count workers.
615    pub fn child_start_with_id(class: &str, child_id: usize, argv: &[String]) {
616        let redacted: Vec<String> = argv.iter().map(|arg| maybe_redact(arg)).collect();
617        let joined = redacted.join(" ");
618        perf_line(
619            depth(),
620            "child_start",
621            &format!("child_id:{child_id} class:{class} argv:[{joined}]"),
622        );
623        append_to_target("GIT_TRACE2", &format!("child_start[{child_id}] {joined}"));
624        if let Some(target) = trace_target("GIT_TRACE2_EVENT") {
625            let json_argv = redacted
626                .iter()
627                .map(|arg| format!("\"{}\"", escape_json(arg)))
628                .collect::<Vec<_>>()
629                .join(",");
630            let line = format!(
631                "{{\"event\":\"child_start\",\"sid\":\"sley\",\"thread\":\"main\",\"child_id\":{child_id},\"child_class\":\"{}\",\"use_shell\":false,\"argv\":[{json_argv}]}}\n",
632                escape_json(class)
633            );
634            write_target(&target, line.as_bytes());
635        }
636    }
637
638    pub fn alias(name: &str, argv: &[String]) {
639        let argv = argv
640            .iter()
641            .map(|arg| maybe_redact(arg))
642            .collect::<Vec<_>>()
643            .join(" ");
644        perf_line(depth(), "alias", &format!("alias:{name} argv:[{argv}]"));
645    }
646
647    /// Emit a trace2 config-parameter record to the normal and perf targets.
648    pub fn def_param(key: &str, value: impl Display) {
649        def_param_at_depth(depth(), key, value);
650    }
651
652    pub fn def_param_at_depth(depth: usize, key: &str, value: impl Display) {
653        let value = value.to_string();
654        let normal = maybe_redact(&format!("{key}={value}"));
655        append_to_target("GIT_TRACE2", &format!("def_param {normal}"));
656        let perf = maybe_redact(&format!("{key}:{value}"));
657        perf_line(depth, "def_param", &perf);
658    }
659
660    /// Emit a trace2 `data` event (upstream `trace2_data_string` /
661    /// `trace2_data_intmax`): a JSON line appended to the `GIT_TRACE2_EVENT`
662    /// file when that target is enabled.
663    pub fn data(category: &str, key: &str, value: impl Display) {
664        let Some(target) = trace_target("GIT_TRACE2_EVENT") else {
665            return;
666        };
667        let line = format!(
668            "{{\"event\":\"data\",\"sid\":\"sley\",\"thread\":\"main\",\"nesting\":1,\"category\":\"{}\",\"key\":\"{}\",\"value\":\"{}\"}}\n",
669            escape_json(category),
670            escape_json(key),
671            escape_json(&value.to_string()),
672        );
673        write_target(&target, line.as_bytes());
674    }
675
676    /// Emit a trace2 `counter` event. Git writes these for accumulated counters
677    /// such as fsync hardware flushes when the event target is enabled.
678    pub fn counter(category: &str, name: &str, count: impl Display) {
679        let Some(target) = trace_target("GIT_TRACE2_EVENT") else {
680            return;
681        };
682        let line = format!(
683            "{{\"event\":\"counter\",\"sid\":\"sley\",\"thread\":\"main\",\"category\":\"{}\",\"name\":\"{}\",\"count\":{}}}\n",
684            escape_json(category),
685            escape_json(name),
686            count,
687        );
688        write_target(&target, line.as_bytes());
689    }
690
691    /// Emit a trace2 region enter/leave pair. This is the minimal event shape
692    /// Git's `test_region` helper greps for when asserting sparse-index
693    /// expansion and conversion behaviour.
694    pub fn region(category: &str, label: &str) {
695        region_event("region_enter", category, label);
696        region_event("region_leave", category, label);
697    }
698
699    fn region_event(event: &str, category: &str, label: &str) {
700        let Some(target) = trace_target("GIT_TRACE2_EVENT") else {
701            return;
702        };
703        let line = format!(
704            "{{\"event\":\"{}\",\"sid\":\"sley\",\"thread\":\"main\",\"nesting\":1,\"category\":\"{}\",\"label\":\"{}\"}}\n",
705            escape_json(event),
706            escape_json(category),
707            escape_json(label),
708        );
709        write_target(&target, line.as_bytes());
710    }
711
712    /// Emit the trace2 perf payload used by Git's changed-path Bloom filter
713    /// tests. This intentionally writes only the grep-stable statistics string.
714    pub fn bloom_statistics(
715        filter_not_present: usize,
716        maybe: usize,
717        definitely_not: usize,
718        false_positive: usize,
719    ) {
720        let Some(target) = trace_target("GIT_TRACE2_PERF") else {
721            return;
722        };
723        let line = format!(
724            "statistics:{{\"filter_not_present\":{filter_not_present},\"maybe\":{maybe},\"definitely_not\":{definitely_not},\"false_positive\":{false_positive}}}\n"
725        );
726        write_target(&target, line.as_bytes());
727    }
728
729    /// Emit a compact trace2 perf `data` row for tests that extract the
730    /// read-directory statistics with pipe-field parsing.
731    pub fn perf_read_directory_data(key: &str, value: impl Display) {
732        let Some(target) = trace_target("GIT_TRACE2_PERF") else {
733            return;
734        };
735        let line = format!(
736            "19:00:00.000000 file.c:1 | d0 | main | data | r1 | ? | ? | read_directory | ....{key}:{value}\n"
737        );
738        write_target(&target, line.as_bytes());
739    }
740
741    /// Emit a trace2 perf `data` row tagged to the `setup` category (git's
742    /// `trace2_data_string("setup", ...)`), used for the
743    /// `implicit-bare-repository:<dir>` marker the safe.bareRepository tests
744    /// grep for. Only the grep-stable `<key>:<value>` tail is significant.
745    pub fn perf_setup_data(key: &str, value: impl Display) {
746        let Some(target) = trace_target("GIT_TRACE2_PERF") else {
747            return;
748        };
749        let line = format!(
750            "19:00:00.000000 setup.c:1 | d0 | main | data | r0 | ? | ? | setup | ....{key}:{value}\n"
751        );
752        write_target(&target, line.as_bytes());
753    }
754}
755
756#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
757pub enum ObjectFormat {
758    Sha1,
759    Sha256,
760}
761
762impl ObjectFormat {
763    pub const fn raw_len(self) -> usize {
764        match self {
765            Self::Sha1 => 20,
766            Self::Sha256 => 32,
767        }
768    }
769
770    pub const fn hex_len(self) -> usize {
771        self.raw_len() * 2
772    }
773
774    pub const fn name(self) -> &'static str {
775        match self {
776            Self::Sha1 => "sha1",
777            Self::Sha256 => "sha256",
778        }
779    }
780}
781
782impl FromStr for ObjectFormat {
783    type Err = GitError;
784
785    fn from_str(value: &str) -> Result<Self> {
786        match value {
787            "sha1" => Ok(Self::Sha1),
788            "sha256" => Ok(Self::Sha256),
789            other => Err(GitError::Unsupported(format!("object format {other}"))),
790        }
791    }
792}
793
794#[derive(Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
795pub struct ObjectId {
796    format: ObjectFormat,
797    bytes: [u8; 32],
798}
799
800impl ObjectId {
801    pub fn from_raw(format: ObjectFormat, raw: &[u8]) -> Result<Self> {
802        if raw.len() != format.raw_len() {
803            return Err(GitError::InvalidObjectId(format!(
804                "expected {} bytes for {}, got {}",
805                format.raw_len(),
806                format.name(),
807                raw.len()
808            )));
809        }
810        let mut bytes = [0; 32];
811        bytes[..raw.len()].copy_from_slice(raw);
812        Ok(Self { format, bytes })
813    }
814
815    pub fn from_hex(format: ObjectFormat, hex: &str) -> Result<Self> {
816        if hex.len() != format.hex_len() {
817            return Err(GitError::InvalidObjectId(format!(
818                "expected {} hex digits for {}, got {}",
819                format.hex_len(),
820                format.name(),
821                hex.len()
822            )));
823        }
824        let mut raw = [0; 32];
825        for (i, pair) in hex.as_bytes().as_chunks::<2>().0.iter().enumerate() {
826            raw[i] = (hex_nibble(pair[0])? << 4) | hex_nibble(pair[1])?;
827        }
828        Ok(Self { format, bytes: raw })
829    }
830
831    pub const fn format(&self) -> ObjectFormat {
832        self.format
833    }
834
835    pub fn as_bytes(&self) -> &[u8] {
836        &self.bytes[..self.format.raw_len()]
837    }
838
839    pub fn to_hex(&self) -> String {
840        let mut out = String::with_capacity(self.format.hex_len());
841        let _ = self.write_hex(&mut out);
842        out
843    }
844
845    pub fn write_hex(&self, out: &mut impl fmt::Write) -> fmt::Result {
846        write_hex_bytes(self.as_bytes(), out)
847    }
848
849    pub fn hex_prefix_matches(&self, prefix: &[u8]) -> bool {
850        if prefix.len() > self.format.hex_len() {
851            return false;
852        }
853
854        prefix.iter().enumerate().all(|(index, expected)| {
855            let Some(expected) = hex_nibble_value(*expected) else {
856                return false;
857            };
858            let byte = self.as_bytes()[index / 2];
859            let actual = if index % 2 == 0 {
860                byte >> 4
861            } else {
862                byte & 0x0f
863            };
864            actual == expected
865        })
866    }
867
868    pub const fn abbrev_hex_len(&self, width: usize) -> usize {
869        let hex_len = self.format.hex_len();
870        if width < hex_len { width } else { hex_len }
871    }
872
873    /// The all-zero ("null") object id for `format`.
874    pub fn null(format: ObjectFormat) -> Self {
875        Self {
876            format,
877            bytes: [0; 32],
878        }
879    }
880
881    /// True when every byte is zero (the null oid).
882    pub fn is_null(&self) -> bool {
883        self.as_bytes().iter().all(|byte| *byte == 0)
884    }
885
886    /// The id of the canonical empty tree for `format` (`4b825dc6…` for SHA-1).
887    pub fn empty_tree(format: ObjectFormat) -> Self {
888        Self::digest_object(format, "tree", b"")
889    }
890
891    /// The id of the canonical empty blob for `format` (`e69de29b…` for SHA-1).
892    pub fn empty_blob(format: ObjectFormat) -> Self {
893        Self::digest_object(format, "blob", b"")
894    }
895
896    /// Hash `"<type> <len>\0<body>"` straight into an id, bypassing the
897    /// fallible length check in [`ObjectId::from_raw`] (our own digests are
898    /// always the right length) so the well-known constants stay infallible.
899    fn digest_object(format: ObjectFormat, object_type: &str, body: &[u8]) -> Self {
900        let mut framed = Vec::with_capacity(object_type.len() + body.len() + 32);
901        framed.extend_from_slice(object_type.as_bytes());
902        framed.push(b' ');
903        framed.extend_from_slice(body.len().to_string().as_bytes());
904        framed.push(0);
905        framed.extend_from_slice(body);
906        let mut bytes = [0u8; 32];
907        match format {
908            ObjectFormat::Sha1 => bytes[..20].copy_from_slice(&sha1(&framed)),
909            ObjectFormat::Sha256 => bytes[..32].copy_from_slice(&sha256(&framed)),
910        }
911        Self { format, bytes }
912    }
913}
914
915impl fmt::Debug for ObjectId {
916    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
917        f.debug_tuple("ObjectId").field(&self.to_hex()).finish()
918    }
919}
920
921impl fmt::Display for ObjectId {
922    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
923        self.write_hex(f)
924    }
925}
926
927impl FromStr for ObjectId {
928    type Err = GitError;
929
930    /// Parse a full hex id, inferring the hash from its length (40 hex digits =
931    /// SHA-1, 64 = SHA-256).
932    fn from_str(text: &str) -> Result<Self> {
933        let format = match text.len() {
934            40 => ObjectFormat::Sha1,
935            64 => ObjectFormat::Sha256,
936            other => {
937                return Err(GitError::InvalidObjectId(format!(
938                    "expected 40 or 64 hex digits, got {other}"
939                )));
940            }
941        };
942        Self::from_hex(format, text)
943    }
944}
945
946/// A validated git ref name (e.g. `refs/heads/main`, `HEAD`).
947#[derive(Clone, PartialEq, Eq, Hash, PartialOrd, Ord)]
948pub struct FullName(String);
949
950impl FullName {
951    /// Construct a ref name, accepting exactly the names
952    /// `git check-ref-format --allow-onelevel` accepts (see
953    /// [`check_refname_format`]). One-level names such as `HEAD` are allowed;
954    /// non-ASCII bytes, including non-ASCII whitespace, are ordinary bytes.
955    pub fn new(name: impl AsRef<str>) -> Result<Self> {
956        let name = name.as_ref();
957        validate_full_name(name)?;
958        Ok(Self(name.to_string()))
959    }
960
961    pub fn as_str(&self) -> &str {
962        &self.0
963    }
964}
965
966impl fmt::Debug for FullName {
967    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
968        f.debug_tuple("FullName").field(&self.0).finish()
969    }
970}
971
972impl fmt::Display for FullName {
973    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
974        f.write_str(&self.0)
975    }
976}
977
978impl From<FullName> for String {
979    fn from(value: FullName) -> Self {
980        value.0
981    }
982}
983
984impl Borrow<str> for FullName {
985    fn borrow(&self) -> &str {
986        &self.0
987    }
988}
989
990impl AsRef<str> for FullName {
991    fn as_ref(&self) -> &str {
992        &self.0
993    }
994}
995
996impl TryFrom<&str> for FullName {
997    type Error = GitError;
998
999    fn try_from(value: &str) -> Result<Self> {
1000        Self::new(value)
1001    }
1002}
1003
1004impl TryFrom<String> for FullName {
1005    type Error = GitError;
1006
1007    fn try_from(value: String) -> Result<Self> {
1008        validate_full_name(&value)?;
1009        Ok(Self(value))
1010    }
1011}
1012
1013impl PartialEq<&str> for FullName {
1014    fn eq(&self, other: &&str) -> bool {
1015        self.0 == *other
1016    }
1017}
1018
1019impl PartialEq<FullName> for &str {
1020    fn eq(&self, other: &FullName) -> bool {
1021        *self == other.0
1022    }
1023}
1024
1025fn validate_full_name(name: &str) -> Result<()> {
1026    check_refname_format(name.as_bytes(), RefnameFormat::ALLOW_ONELEVEL)
1027        .map_err(|err| GitError::InvalidFormat(format!("{err}: {name:?}")))
1028}
1029
1030/// A byte string for git paths and similar on-disk identifiers.
1031#[derive(Debug, Clone, Default, PartialEq, Eq, Hash, PartialOrd, Ord)]
1032pub struct BString(Vec<u8>);
1033
1034impl BString {
1035    pub fn new(bytes: impl Into<Vec<u8>>) -> Self {
1036        Self(bytes.into())
1037    }
1038    pub fn from_bytes(bytes: &[u8]) -> Self {
1039        Self(bytes.to_vec())
1040    }
1041    pub fn as_bytes(&self) -> &[u8] {
1042        &self.0
1043    }
1044    pub fn len(&self) -> usize {
1045        self.0.len()
1046    }
1047    pub fn is_empty(&self) -> bool {
1048        self.0.is_empty()
1049    }
1050    pub fn into_bytes(self) -> Vec<u8> {
1051        self.0
1052    }
1053}
1054
1055impl From<&str> for BString {
1056    fn from(v: &str) -> Self {
1057        Self::from_bytes(v.as_bytes())
1058    }
1059}
1060impl From<&[u8]> for BString {
1061    fn from(v: &[u8]) -> Self {
1062        Self::from_bytes(v)
1063    }
1064}
1065impl<const N: usize> From<&[u8; N]> for BString {
1066    fn from(v: &[u8; N]) -> Self {
1067        Self::from_bytes(v.as_slice())
1068    }
1069}
1070impl From<Vec<u8>> for BString {
1071    fn from(v: Vec<u8>) -> Self {
1072        Self(v)
1073    }
1074}
1075impl PartialEq<&[u8]> for BString {
1076    fn eq(&self, o: &&[u8]) -> bool {
1077        self.0.as_slice() == *o
1078    }
1079}
1080impl<const N: usize> PartialEq<&[u8; N]> for BString {
1081    fn eq(&self, o: &&[u8; N]) -> bool {
1082        self.as_bytes() == o.as_slice()
1083    }
1084}
1085impl PartialEq<BString> for &[u8] {
1086    fn eq(&self, o: &BString) -> bool {
1087        *self == o.as_bytes()
1088    }
1089}
1090impl<const N: usize> PartialEq<BString> for &[u8; N] {
1091    fn eq(&self, o: &BString) -> bool {
1092        self.as_slice() == o.as_bytes()
1093    }
1094}
1095impl PartialEq<Vec<u8>> for BString {
1096    fn eq(&self, o: &Vec<u8>) -> bool {
1097        self.0 == *o
1098    }
1099}
1100impl PartialEq<BString> for Vec<u8> {
1101    fn eq(&self, o: &BString) -> bool {
1102        *self == o.0
1103    }
1104}
1105
1106impl fmt::Display for BString {
1107    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1108        write!(f, "{}", String::from_utf8_lossy(&self.0))
1109    }
1110}
1111
1112impl Borrow<[u8]> for BString {
1113    fn borrow(&self) -> &[u8] {
1114        self.as_bytes()
1115    }
1116}
1117
1118impl Deref for BString {
1119    type Target = [u8];
1120
1121    fn deref(&self) -> &[u8] {
1122        self.as_bytes()
1123    }
1124}
1125
1126impl AsRef<[u8]> for BString {
1127    fn as_ref(&self) -> &[u8] {
1128        self.as_bytes()
1129    }
1130}
1131
1132#[derive(Debug, Clone, PartialEq, Eq, Hash)]
1133pub struct RepoPath(PathBuf);
1134
1135impl RepoPath {
1136    pub fn new(path: impl Into<PathBuf>) -> Result<Self> {
1137        let path = path.into();
1138        if path.is_absolute() {
1139            return Err(GitError::InvalidPath(
1140                "repository paths must be relative".into(),
1141            ));
1142        }
1143        if path.components().any(|component| {
1144            matches!(
1145                component,
1146                std::path::Component::ParentDir | std::path::Component::Prefix(_)
1147            )
1148        }) {
1149            return Err(GitError::InvalidPath(
1150                "repository paths must not escape".into(),
1151            ));
1152        }
1153        Ok(Self(path))
1154    }
1155
1156    pub fn as_path(&self) -> &Path {
1157        &self.0
1158    }
1159}
1160
1161/// A typed *parse-view* of a git identity line (`Name <email> <secs> <tz>`) as
1162/// found on a commit's `author`/`committer` or a tag's `tagger` header.
1163///
1164/// This is a read-only lens over bytes that are stored and re-serialized
1165/// verbatim elsewhere (see [`Signature::raw`]). It exists so callers can read
1166/// the typed `name`/`email`/`time` of an identity without re-implementing git's
1167/// ident-splitting rules, *not* as a storage format: the object model keeps the
1168/// original raw bytes as its source of truth, and round-tripping through this
1169/// view is byte-exact precisely because the raw line is retained alongside the
1170/// parsed fields (see [`Signature::to_ident_bytes`]).
1171///
1172/// Parse one with [`Signature::from_ident_line`]. The `time`'s timezone
1173/// preserves git's distinction between `+0000` (UTC) and `-0000` (a sentinel git
1174/// writes to mean "timezone unknown"); see [`GitTime`].
1175#[derive(Debug, Clone, PartialEq, Eq)]
1176pub struct Signature {
1177    /// The identity's name: the bytes before the ` <` that opens the email,
1178    /// with one trailing space (the separator) removed. May be empty.
1179    pub name: BString,
1180    /// The identity's email: the bytes between the `<` and `>` delimiters. May
1181    /// be empty.
1182    pub email: BString,
1183    /// The commit/authorship time and its timezone offset.
1184    pub time: GitTime,
1185    /// The exact original ident-line bytes this view was parsed from, retained
1186    /// so [`Signature::to_ident_bytes`] can reproduce the input byte-for-byte
1187    /// regardless of any non-canonical whitespace or formatting it contained.
1188    pub raw: Vec<u8>,
1189}
1190
1191impl Signature {
1192    /// Parse a raw git identity line (`Name <email> <unix-secs> <tz>`) into a
1193    /// typed view, returning `None` when the bytes do not form a well-formed
1194    /// identity.
1195    ///
1196    /// The splitting mirrors git's own `split_ident_line`: the email is the run
1197    /// of bytes between the last `<` and the first following `>`; the name is
1198    /// everything before that `<` (one separating space is dropped); after the
1199    /// `>` come a space, the decimal Unix timestamp, a space, and the timezone
1200    /// token. The name and email may legitimately be empty, but a missing
1201    /// `<`/`>` pair, a non-numeric timestamp, or a malformed timezone token all
1202    /// yield `None` rather than a lossy guess — this is a *best-effort* parse
1203    /// that never panics. The original bytes are retained in
1204    /// [`Signature::raw`] so the parsed view re-serializes byte-identically.
1205    pub fn from_ident_line(line: &[u8]) -> Option<Self> {
1206        // Email is delimited by the last '<' whose matching '>' follows it, the
1207        // way git scans an ident from the right. Find the last '>' first, then
1208        // the last '<' before it.
1209        let mail_end = line.iter().rposition(|byte| *byte == b'>')?;
1210        let mail_begin = line[..mail_end].iter().rposition(|byte| *byte == b'<')? + 1;
1211        let email = &line[mail_begin..mail_end];
1212
1213        // The name is everything before the '<', with a single trailing space
1214        // (the separator git inserts) trimmed if present.
1215        let mut name_end = mail_begin.saturating_sub(1);
1216        if name_end > 0 && line[name_end - 1] == b' ' {
1217            name_end -= 1;
1218        }
1219        let name = &line[..name_end];
1220
1221        // After '>' git expects "<space><secs><space><tz>". Trim the single
1222        // separating space, then split the timestamp from the timezone token.
1223        let rest = line.get(mail_end + 1..)?;
1224        let rest = rest.strip_prefix(b" ")?;
1225        let time = GitTime::from_time_fields(rest)?;
1226
1227        Some(Self {
1228            name: BString::new(name.to_vec()),
1229            email: BString::new(email.to_vec()),
1230            time,
1231            raw: line.to_vec(),
1232        })
1233    }
1234
1235    /// Reproduce the original identity-line bytes.
1236    ///
1237    /// This returns [`Signature::raw`] verbatim, so for any line that
1238    /// [`Signature::from_ident_line`] accepted, `from_ident_line(line)?
1239    /// .to_ident_bytes() == line` holds byte-for-byte — including the `-0000`
1240    /// timezone and any non-canonical spacing the source contained.
1241    pub fn to_ident_bytes(&self) -> Vec<u8> {
1242        self.raw.clone()
1243    }
1244
1245    /// Re-derive the canonical ident line from the parsed fields alone
1246    /// (`name <email> secs tz`), ignoring [`Signature::raw`].
1247    ///
1248    /// For an identity in git's canonical form this equals
1249    /// [`Signature::to_ident_bytes`]; it differs only when the source line
1250    /// carried non-canonical whitespace. Callers wanting byte-exact
1251    /// reproduction should use [`Signature::to_ident_bytes`]; this is provided
1252    /// for constructing a normalized line from typed parts.
1253    pub fn to_canonical_ident_bytes(&self) -> Vec<u8> {
1254        let mut out = Vec::with_capacity(self.raw.len());
1255        out.extend_from_slice(self.name.as_bytes());
1256        out.extend_from_slice(b" <");
1257        out.extend_from_slice(self.email.as_bytes());
1258        out.extend_from_slice(b"> ");
1259        out.extend_from_slice(self.time.to_ident_suffix().as_bytes());
1260        out
1261    }
1262}
1263
1264/// A tolerant parse-view of a git identity line split git's way (ident.c's
1265/// `split_ident_line`). Unlike [`Signature::from_ident_line`] — which is a
1266/// strict, byte-exact round-trip parser — this mirrors how git's pretty-printer
1267/// recovers fields from *broken* idents: the email is the run between the
1268/// **first** `<` and the **first** following `>`, while the timestamp is located
1269/// by scanning **backwards** from the end of the line for the **last** `>`. That
1270/// split lets a corrupt ident like `Name <a@b>-<> 123 +0000` still surrender the
1271/// correct name (`Name`), email (`a@b`), and date (`123 +0000`).
1272pub struct IdentFields<'a> {
1273    /// Everything before the first `<`, with one trailing separator space removed.
1274    pub name: &'a [u8],
1275    /// The bytes between the first `<` and the first following `>`.
1276    pub email: &'a [u8],
1277    /// The decimal timestamp digit-run, or `None` when the line has no parseable
1278    /// `<digits> <±digits>` date tail (git's "person only" case).
1279    pub date: Option<&'a [u8]>,
1280    /// The timezone token (`±` plus digits), present iff `date` is.
1281    pub tz: Option<&'a [u8]>,
1282}
1283
1284/// True for the whitespace bytes git's `isspace` recognizes (space, tab,
1285/// newline, carriage return). This deliberately excludes vertical tab (`0x0b`)
1286/// and form feed (`0x0c`), matching git's `sane_ctype` table — the distinction
1287/// that makes a vertical-tab-only date a sentinel rather than valid whitespace.
1288fn ident_isspace(byte: u8) -> bool {
1289    matches!(byte, b' ' | b'\t' | b'\n' | b'\r')
1290}
1291
1292/// Split a git identity line the way ident.c's `split_ident_line` does,
1293/// returning `None` only when the line has no `<` or no following `>` (git's
1294/// `status < 0`). The date/timezone fields are `None` for the "person only"
1295/// case where no valid timestamp follows the final `>`.
1296pub fn split_ident_line(line: &[u8]) -> Option<IdentFields<'_>> {
1297    let len = line.len();
1298    // mail_begin: just past the first '<'.
1299    let lt = line.iter().position(|&byte| byte == b'<')?;
1300    let mail_begin = lt + 1;
1301
1302    // name_end: the last non-space byte before '<' (git scans down from
1303    // mail_begin-2); default to the '<' position when only spaces precede it.
1304    let mut name_end = mail_begin - 1;
1305    if mail_begin >= 2 {
1306        let mut i = mail_begin - 2;
1307        loop {
1308            if !ident_isspace(line[i]) {
1309                name_end = i + 1;
1310                break;
1311            }
1312            if i == 0 {
1313                break;
1314            }
1315            i -= 1;
1316        }
1317    }
1318    let name = &line[..name_end];
1319
1320    // mail_end: first '>' at or after mail_begin.
1321    let gt = line[mail_begin..].iter().position(|&byte| byte == b'>')? + mail_begin;
1322    let email = &line[mail_begin..gt];
1323
1324    let person_only = IdentFields {
1325        name,
1326        email,
1327        date: None,
1328        tz: None,
1329    };
1330
1331    // Date: scan from the end of the line for the LAST '>', then parse a
1332    // "<digits> <±digits>" tail after it (git assumes the timestamp has no '>').
1333    let mut cp = len - 1;
1334    while line[cp] != b'>' {
1335        if cp == 0 {
1336            return Some(person_only);
1337        }
1338        cp -= 1;
1339    }
1340    let mut i = cp + 1;
1341    while i < len && ident_isspace(line[i]) {
1342        i += 1;
1343    }
1344    let date_begin = i;
1345    while i < len && line[i].is_ascii_digit() {
1346        i += 1;
1347    }
1348    if i == date_begin {
1349        return Some(person_only);
1350    }
1351    let date = &line[date_begin..i];
1352
1353    while i < len && ident_isspace(line[i]) {
1354        i += 1;
1355    }
1356    if i >= len || (line[i] != b'+' && line[i] != b'-') {
1357        return Some(person_only);
1358    }
1359    let tz_begin = i;
1360    i += 1;
1361    let tz_digits = i;
1362    while i < len && line[i].is_ascii_digit() {
1363        i += 1;
1364    }
1365    if i == tz_digits {
1366        return Some(person_only);
1367    }
1368    Some(IdentFields {
1369        name,
1370        email,
1371        date: Some(date),
1372        tz: Some(&line[tz_begin..i]),
1373    })
1374}
1375
1376/// True when a timestamp is too large to be a valid `time_t`, mirroring git's
1377/// `date_overflows` for a 64-bit signed `time_t`.
1378fn ident_date_overflows(seconds: u64) -> bool {
1379    seconds >= i64::MAX as u64
1380}
1381
1382/// Render an ident's date the way pretty.c's `show_ident_date` does: parse the
1383/// timestamp (git's `parse_timestamp` is unsigned/base-10 and clamps on
1384/// overflow), substitute the epoch sentinel (`time = 0`, timezone `+0000`) when
1385/// the value overflows what a `time_t` can hold, then format per `mode`. `date`
1386/// is the timestamp digit-run and `tz` its timezone token (as returned by
1387/// [`split_ident_line`]).
1388pub fn ident_render_date(date: &[u8], tz: &[u8], mode: &DateMode) -> String {
1389    let parsed = std::str::from_utf8(date)
1390        .ok()
1391        .and_then(|text| text.parse::<u64>().ok());
1392    let (seconds, tz_text) = match parsed {
1393        Some(value) if !ident_date_overflows(value) => {
1394            (value as i64, std::str::from_utf8(tz).unwrap_or("+0000"))
1395        }
1396        // Overflow, or a digit-run too long for u64: the epoch sentinel with a
1397        // forced `+0000` timezone, exactly like git's show_ident_date.
1398        _ => (0, "+0000"),
1399    };
1400    mode.render(seconds, tz_text).unwrap_or_default()
1401}
1402
1403impl fmt::Display for Signature {
1404    /// Renders the original ident line (lossy only for bytes that are not valid
1405    /// UTF-8, which are replaced with `U+FFFD`). Use
1406    /// [`Signature::to_ident_bytes`] for the exact bytes.
1407    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1408        write!(f, "{}", String::from_utf8_lossy(&self.raw))
1409    }
1410}
1411
1412/// A git timestamp: a Unix time plus the committer's timezone offset.
1413///
1414/// The offset is stored as signed minutes east of UTC ([`timezone_offset_minutes`])
1415/// *and* a separate [`negative_utc`] flag. The flag exists because git
1416/// distinguishes the timezone token `-0000` from `+0000`: both are zero minutes
1417/// from UTC, but git writes `-0000` as a sentinel meaning "timezone unknown"
1418/// (e.g. for dates parsed without zone information), and that distinction is
1419/// part of a commit's byte-exact identity. `timezone_offset_minutes` alone
1420/// cannot represent it, so `negative_utc` carries the sign of a zero offset.
1421///
1422/// [`timezone_offset_minutes`]: GitTime::timezone_offset_minutes
1423/// [`negative_utc`]: GitTime::negative_utc
1424#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1425pub struct GitTime {
1426    /// Seconds since the Unix epoch.
1427    pub seconds: i64,
1428    /// Timezone offset east of UTC, in minutes (e.g. `+0530` -> `330`,
1429    /// `-0500` -> `-300`). Zero for both `+0000` and `-0000`; consult
1430    /// [`GitTime::negative_utc`] to tell those apart.
1431    pub timezone_offset_minutes: i16,
1432    /// `true` only when the timezone token had a negative sign with a zero
1433    /// magnitude (`-0000`), git's "timezone unknown" sentinel. Always `false`
1434    /// for any non-zero offset.
1435    pub negative_utc: bool,
1436}
1437
1438impl GitTime {
1439    /// A `GitTime` with the given seconds and minute offset, treating a zero
1440    /// offset as the ordinary `+0000` (not the `-0000` sentinel). Use
1441    /// [`GitTime::with_negative_utc`] to construct the `-0000` case.
1442    pub const fn new(seconds: i64, timezone_offset_minutes: i16) -> Self {
1443        Self {
1444            seconds,
1445            timezone_offset_minutes,
1446            negative_utc: false,
1447        }
1448    }
1449
1450    /// A `GitTime` whose timezone is the `-0000` sentinel ("timezone unknown").
1451    /// The minute offset is zero; `negative_utc` is `true`.
1452    pub const fn with_negative_utc(seconds: i64) -> Self {
1453        Self {
1454            seconds,
1455            timezone_offset_minutes: 0,
1456            negative_utc: true,
1457        }
1458    }
1459
1460    /// Parse the `<secs> <tz>` tail of an ident line (the bytes after the
1461    /// `"> "` separating the email from the time), returning `None` if either
1462    /// field is malformed.
1463    fn from_time_fields(bytes: &[u8]) -> Option<Self> {
1464        let text = std::str::from_utf8(bytes).ok()?;
1465        let (seconds_text, tz_text) = text.split_once(' ')?;
1466        let seconds = seconds_text.parse::<i64>().ok()?;
1467        let (timezone_offset_minutes, negative_utc) = parse_timezone_token(tz_text)?;
1468        Some(Self {
1469            seconds,
1470            timezone_offset_minutes,
1471            negative_utc,
1472        })
1473    }
1474
1475    /// The canonical `<secs> <±HHMM>` rendering of this time, as git writes it.
1476    /// Preserves the `-0000` sentinel.
1477    fn to_ident_suffix(self) -> String {
1478        format!("{} {}", self.seconds, self.offset_token())
1479    }
1480
1481    /// The canonical 5-character timezone token for this offset (sign plus four
1482    /// digits), e.g. `+0000`, `-0500`, `+0530`. Returns `-0000` when
1483    /// [`GitTime::negative_utc`] is set.
1484    pub fn offset_token(self) -> String {
1485        let sign = if self.negative_utc || self.timezone_offset_minutes < 0 {
1486            '-'
1487        } else {
1488            '+'
1489        };
1490        let magnitude = self.timezone_offset_minutes.unsigned_abs();
1491        format!("{sign}{:02}{:02}", magnitude / 60, magnitude % 60)
1492    }
1493}
1494
1495/// Parse a git timezone token (`±HHMM`) into `(minutes east of UTC, negative_utc)`.
1496///
1497/// Git accepts a leading `+`/`-` followed by four digits where the last two are
1498/// minutes. A negative sign with a zero magnitude (`-0000`) sets `negative_utc`.
1499/// Returns `None` for anything that is not a well-formed token.
1500fn parse_timezone_token(token: &str) -> Option<(i16, bool)> {
1501    let bytes = token.as_bytes();
1502    if bytes.len() != 5 {
1503        return None;
1504    }
1505    let negative = match bytes[0] {
1506        b'+' => false,
1507        b'-' => true,
1508        _ => return None,
1509    };
1510    if !bytes[1..].iter().all(u8::is_ascii_digit) {
1511        return None;
1512    }
1513    let hours = i16::from(bytes[1] - b'0') * 10 + i16::from(bytes[2] - b'0');
1514    let minutes = i16::from(bytes[3] - b'0') * 10 + i16::from(bytes[4] - b'0');
1515    let total = hours * 60 + minutes;
1516    let negative_utc = negative && total == 0;
1517    let signed = if negative { -total } else { total };
1518    Some((signed, negative_utc))
1519}
1520
1521#[derive(Debug, Clone, PartialEq, Eq)]
1522pub struct Capability {
1523    pub name: String,
1524    pub value: Option<String>,
1525}
1526
1527#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
1528pub enum MissingObjectKind {
1529    Object,
1530    Blob,
1531    Tree,
1532    Commit,
1533    Tag,
1534}
1535
1536impl MissingObjectKind {
1537    pub const fn as_str(self) -> &'static str {
1538        match self {
1539            Self::Object => "object",
1540            Self::Blob => "blob",
1541            Self::Tree => "tree",
1542            Self::Commit => "commit",
1543            Self::Tag => "tag",
1544        }
1545    }
1546}
1547
1548impl fmt::Display for MissingObjectKind {
1549    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1550        f.write_str(self.as_str())
1551    }
1552}
1553
1554#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
1555pub enum MissingObjectContext {
1556    Read,
1557    Traversal,
1558    PackInstall,
1559    RevisionWalk,
1560    WorktreeMaterialize,
1561    RemoteBoundary,
1562}
1563
1564impl MissingObjectContext {
1565    pub const fn as_str(self) -> &'static str {
1566        match self {
1567            Self::Read => "read",
1568            Self::Traversal => "traversal",
1569            Self::PackInstall => "pack-install",
1570            Self::RevisionWalk => "revision-walk",
1571            Self::WorktreeMaterialize => "worktree-materialize",
1572            Self::RemoteBoundary => "remote-boundary",
1573        }
1574    }
1575}
1576
1577impl fmt::Display for MissingObjectContext {
1578    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1579        f.write_str(self.as_str())
1580    }
1581}
1582
1583#[derive(Debug, Clone, PartialEq, Eq)]
1584pub enum NotFoundKind {
1585    Message(String),
1586    Remote {
1587        name: String,
1588    },
1589    Object {
1590        oid: ObjectId,
1591        kind: MissingObjectKind,
1592        context: Option<MissingObjectContext>,
1593    },
1594    Reference {
1595        name: String,
1596    },
1597    BrokenReference {
1598        name: String,
1599        target: String,
1600    },
1601    Repository {
1602        path: String,
1603    },
1604}
1605
1606impl fmt::Display for NotFoundKind {
1607    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1608        match self {
1609            Self::Message(msg) => write!(f, "{msg}"),
1610            Self::Remote { name } => write!(f, "remote {name}"),
1611            Self::Object {
1612                oid,
1613                kind: MissingObjectKind::Object,
1614                ..
1615            } => write!(f, "object {oid}"),
1616            Self::Object { oid, kind, .. } => write!(f, "{kind} object {oid}"),
1617            Self::Reference { name } => write!(f, "{name}"),
1618            Self::BrokenReference { name, target } => {
1619                write!(f, "broken reference {name} -> {target}")
1620            }
1621            Self::Repository { path } => write!(f, "{path}"),
1622        }
1623    }
1624}
1625
1626impl NotFoundKind {
1627    pub fn object_id(&self) -> Option<ObjectId> {
1628        match self {
1629            Self::Object { oid, .. } => Some(*oid),
1630            _ => None,
1631        }
1632    }
1633
1634    pub fn missing_object_kind(&self) -> Option<MissingObjectKind> {
1635        match self {
1636            Self::Object { kind, .. } => Some(*kind),
1637            _ => None,
1638        }
1639    }
1640
1641    pub fn missing_object_context(&self) -> Option<MissingObjectContext> {
1642        match self {
1643            Self::Object { context, .. } => *context,
1644            _ => None,
1645        }
1646    }
1647}
1648
1649/// Why an operation stopped after delivering its detailed diagnostics to its sink.
1650/// This carries library semantics; applications decide how to report the outcome.
1651#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1652pub enum RejectionKind {
1653    InvalidArguments,
1654    Refused,
1655    Incomplete,
1656}
1657
1658/// Failure returned by a caller-provided service (editor, renderer, hydration, etc.).
1659/// The dynamic boundary preserves the caller's concrete error for downcasting.
1660/// Clones share identity; independently constructed errors are distinct.
1661#[derive(Debug, Clone)]
1662pub struct CallbackError(std::sync::Arc<dyn Error + Send + Sync>);
1663
1664impl CallbackError {
1665    pub fn new(error: impl Error + Send + Sync + 'static) -> Self {
1666        Self(std::sync::Arc::new(error))
1667    }
1668    pub fn downcast_ref<T: Error + 'static>(&self) -> Option<&T> {
1669        self.0.downcast_ref()
1670    }
1671}
1672impl PartialEq for CallbackError {
1673    fn eq(&self, other: &Self) -> bool {
1674        std::sync::Arc::ptr_eq(&self.0, &other.0)
1675    }
1676}
1677impl Eq for CallbackError {}
1678impl fmt::Display for CallbackError {
1679    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1680        self.0.fmt(f)
1681    }
1682}
1683impl Error for CallbackError {
1684    fn source(&self) -> Option<&(dyn Error + 'static)> {
1685        Some(self.0.as_ref())
1686    }
1687}
1688
1689/// Fail-closed byte budget used by pack write/read working-set caps.
1690///
1691/// One shared budget type so callers do not invent per-path helpers. The limit
1692/// is inclusive: a value equal to the budget is admitted.
1693#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
1694pub struct ByteBudget(u64);
1695
1696impl ByteBudget {
1697    pub const ZERO: Self = Self(0);
1698
1699    pub const fn new(bytes: u64) -> Self {
1700        Self(bytes)
1701    }
1702
1703    pub const fn as_u64(self) -> u64 {
1704        self.0
1705    }
1706
1707    pub const fn as_usize(self) -> Option<usize> {
1708        if self.0 > usize::MAX as u64 {
1709            None
1710        } else {
1711            Some(self.0 as usize)
1712        }
1713    }
1714
1715    /// Whether `used + additional` stays within this budget.
1716    pub const fn allows(self, used: u64, additional: u64) -> bool {
1717        used.saturating_add(additional) <= self.0
1718    }
1719}
1720
1721impl From<u64> for ByteBudget {
1722    fn from(bytes: u64) -> Self {
1723        Self::new(bytes)
1724    }
1725}
1726
1727impl fmt::Display for ByteBudget {
1728    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1729        write!(f, "{} bytes", self.0)
1730    }
1731}
1732
1733/// Which explicit budget rejected a resource-limit check.
1734#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1735pub enum ResourceLimitKind {
1736    CompressionWorkingSet,
1737    DecodedObject,
1738    DeltaBase,
1739}
1740
1741impl fmt::Display for ResourceLimitKind {
1742    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1743        match self {
1744            Self::CompressionWorkingSet => f.write_str("compression working set"),
1745            Self::DecodedObject => f.write_str("decoded object"),
1746            Self::DeltaBase => f.write_str("delta base"),
1747        }
1748    }
1749}
1750
1751#[derive(Debug, Clone, PartialEq, Eq)]
1752pub enum GitError {
1753    /// An I/O failure that preserves the [`std::io::ErrorKind`] of the
1754    /// underlying [`std::io::Error`].
1755    ///
1756    /// Produced by `From<std::io::Error>` so downstream code can branch on
1757    /// [`GitError::io_kind`] instead of sniffing rendered message text. The
1758    /// same typed channel is used for manually described I/O failures.
1759    IoKind {
1760        kind: std::io::ErrorKind,
1761        message: String,
1762    },
1763    /// A sideband channel-3 (fatal) message from a pack protocol response.
1764    ///
1765    /// Typed marker produced where sideband demuxing surfaces remote aborts,
1766    /// so recovery paths classify them by variant rather than substring-
1767    /// matching `"sideband fatal:"` in rendered messages.
1768    SidebandFatal(String),
1769    InvalidObjectId(String),
1770    InvalidObject(String),
1771    InvalidFormat(String),
1772    InvalidPath(String),
1773    Unsupported(String),
1774    NotFound(NotFoundKind),
1775    Transaction(String),
1776    Command(String),
1777    /// An operation was rejected; details were sent to the operation's sink.
1778    Rejected(RejectionKind),
1779    /// A caller-provided service failed; its concrete error is preserved.
1780    Callback(CallbackError),
1781    /// An actual child process failed (not a request to exit this process).
1782    ChildProcessFailed {
1783        status: Option<i32>,
1784    },
1785    RemoteHelperAborted {
1786        name: String,
1787    },
1788    EmptyPreferredPack {
1789        path: std::path::PathBuf,
1790    },
1791    /// Cooperative cancellation of a streaming or long-running operation.
1792    ///
1793    /// Raised when a [`CancelFlag`] trips mid-stream (pack index/install, pack
1794    /// write, fetch demux, emit loops). Distinct from I/O failure so embedders
1795    /// and the CLI can treat user-stop as non-corruption.
1796    Cancelled,
1797    /// A known-count stream yielded fewer or more items than the caller declared.
1798    ///
1799    /// Used by pack generation so a truncated or overlong object-id iterator
1800    /// cannot produce a successful pack.
1801    CountMismatch {
1802        expected: u64,
1803        actual: u64,
1804    },
1805    /// An explicit byte/count budget was exceeded.
1806    ResourceLimit {
1807        kind: ResourceLimitKind,
1808        limit: u64,
1809        attempted: u64,
1810    },
1811}
1812
1813pub type Result<T> = std::result::Result<T, GitError>;
1814
1815impl fmt::Display for GitError {
1816    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1817        match self {
1818            // Message text already carries the OS detail (`value.to_string()`
1819            // of the source error); keep the rendering identical to `Io`.
1820            Self::IoKind { kind: _, message } => write!(f, "io error: {message}"),
1821            Self::SidebandFatal(message) => write!(f, "sideband fatal: {message}"),
1822            Self::InvalidObjectId(msg) => write!(f, "invalid object id: {msg}"),
1823            Self::InvalidObject(msg) => write!(f, "invalid object: {msg}"),
1824            Self::InvalidFormat(msg) => write!(f, "invalid format: {msg}"),
1825            Self::InvalidPath(msg) => write!(f, "invalid path: {msg}"),
1826            Self::Unsupported(msg) => write!(f, "unsupported: {msg}"),
1827            Self::NotFound(kind) => write!(f, "not found: {kind}"),
1828            Self::Transaction(msg) => write!(f, "transaction failed: {msg}"),
1829            Self::Command(msg) => write!(f, "command failed: {msg}"),
1830            Self::Rejected(kind) => write!(f, "operation rejected: {kind:?}"),
1831            Self::Callback(error) => fmt::Display::fmt(error, f),
1832            Self::ChildProcessFailed { status } => write!(f, "child process failed: {status:?}"),
1833            Self::RemoteHelperAborted { name } => {
1834                write!(f, "remote helper '{name}' aborted session")
1835            }
1836            Self::EmptyPreferredPack { path } => write!(
1837                f,
1838                "cannot select preferred pack {} with no objects",
1839                path.display()
1840            ),
1841            Self::Cancelled => f.write_str("operation cancelled"),
1842            Self::CountMismatch { expected, actual } => {
1843                write!(f, "count mismatch: expected {expected}, yielded {actual}")
1844            }
1845            Self::ResourceLimit {
1846                kind,
1847                limit,
1848                attempted,
1849            } => write!(
1850                f,
1851                "resource limit exceeded: {kind} limit {limit}, attempted {attempted}"
1852            ),
1853        }
1854    }
1855}
1856
1857impl Error for GitError {
1858    fn source(&self) -> Option<&(dyn Error + 'static)> {
1859        match self {
1860            Self::Callback(error) => Some(error),
1861            _ => None,
1862        }
1863    }
1864}
1865
1866impl GitError {
1867    pub fn not_found(msg: impl Into<String>) -> Self {
1868        Self::NotFound(NotFoundKind::Message(msg.into()))
1869    }
1870
1871    pub fn remote_not_found(name: impl Into<String>) -> Self {
1872        Self::NotFound(NotFoundKind::Remote { name: name.into() })
1873    }
1874
1875    pub fn object_not_found(oid: ObjectId) -> Self {
1876        Self::object_kind_not_found(oid, MissingObjectKind::Object)
1877    }
1878
1879    pub fn object_kind_not_found(oid: ObjectId, kind: MissingObjectKind) -> Self {
1880        Self::NotFound(NotFoundKind::Object {
1881            oid,
1882            kind,
1883            context: None,
1884        })
1885    }
1886
1887    pub fn object_not_found_in(oid: ObjectId, context: MissingObjectContext) -> Self {
1888        Self::object_kind_not_found_in(oid, MissingObjectKind::Object, context)
1889    }
1890
1891    pub fn object_kind_not_found_in(
1892        oid: ObjectId,
1893        kind: MissingObjectKind,
1894        context: MissingObjectContext,
1895    ) -> Self {
1896        Self::NotFound(NotFoundKind::Object {
1897            oid,
1898            kind,
1899            context: Some(context),
1900        })
1901    }
1902
1903    pub fn reference_not_found(name: impl Into<String>) -> Self {
1904        Self::NotFound(NotFoundKind::Reference { name: name.into() })
1905    }
1906
1907    pub fn broken_reference(name: impl Into<String>, target: impl Into<String>) -> Self {
1908        Self::NotFound(NotFoundKind::BrokenReference {
1909            name: name.into(),
1910            target: target.into(),
1911        })
1912    }
1913
1914    pub fn repository_not_found(path: impl Into<String>) -> Self {
1915        Self::NotFound(NotFoundKind::Repository { path: path.into() })
1916    }
1917
1918    pub fn not_found_kind(&self) -> Option<&NotFoundKind> {
1919        match self {
1920            Self::NotFound(kind) => Some(kind),
1921            _ => None,
1922        }
1923    }
1924
1925    pub fn count_mismatch(expected: u64, actual: u64) -> Self {
1926        Self::CountMismatch { expected, actual }
1927    }
1928
1929    pub fn resource_limit(kind: ResourceLimitKind, limit: u64, attempted: u64) -> Self {
1930        Self::ResourceLimit {
1931            kind,
1932            limit,
1933            attempted,
1934        }
1935    }
1936
1937    /// The preserved I/O [`std::io::ErrorKind`], when this error originated
1938    /// from (or was constructed with) an I/O error kind.
1939    ///
1940    /// `None` for non-I/O variants.
1941    pub fn io_kind(&self) -> Option<std::io::ErrorKind> {
1942        match self {
1943            Self::IoKind { kind, .. } => Some(*kind),
1944            _ => None,
1945        }
1946    }
1947
1948    /// Whether this error represents cooperative cancellation.
1949    ///
1950    /// Uniformly covers:
1951    /// - the explicit [`GitError::Cancelled`] variant (raised directly by
1952    ///   `CancelFlag`, or via the `OperationCancelled` payload intercept in
1953    ///   `From<std::io::Error>`), and
1954    /// - structured I/O errors of kind
1955    ///   [`Interrupted`](std::io::ErrorKind::Interrupted) — the EINTR-style
1956    ///   wake-up the cancel machinery produces when a blocked read is
1957    ///   interrupted after retries are exhausted, and
1958    /// - legacy string-form errors carrying "cancelled" text.
1959    pub fn is_cancelled(&self) -> bool {
1960        match self {
1961            Self::Cancelled => true,
1962            Self::IoKind { kind, message } => {
1963                matches!(kind, std::io::ErrorKind::Interrupted) || message.contains("cancelled")
1964            }
1965            _ => false,
1966        }
1967    }
1968}
1969
1970impl From<std::io::Error> for GitError {
1971    fn from(value: std::io::Error) -> Self {
1972        // Cooperative cancel round-trips through its payload marker here so
1973        // the rest of the pipeline sees `Cancelled` instead of a stringly
1974        // I/O error (previously recovered by sniffing "cancelled" text).
1975        if is_cancelled_io(&value) {
1976            return Self::Cancelled;
1977        }
1978        // Typed payloads installed across io boundaries (e.g. sideband demux
1979        // surfacing `SidebandFatal`/`InvalidFormat` as `io::Error`) survive
1980        // this conversion unchanged.
1981        if let Some(inner) = value
1982            .get_ref()
1983            .and_then(|err| err.downcast_ref::<GitError>())
1984        {
1985            return inner.clone();
1986        }
1987        Self::IoKind {
1988            kind: value.kind(),
1989            message: value.to_string(),
1990        }
1991    }
1992}
1993
1994pub fn object_id_for_bytes(
1995    format: ObjectFormat,
1996    object_type: &str,
1997    body: &[u8],
1998) -> Result<ObjectId> {
1999    match format {
2000        // Hash the `"<type> <len>\0"` header and the body as separate updates so
2001        // the (potentially large) body is never copied into a combined buffer just
2002        // to feed the digest.
2003        ObjectFormat::Sha1 => ObjectId::from_raw(format, &sha1_object_digest(object_type, body)),
2004        ObjectFormat::Sha256 => {
2005            let mut framed = Vec::with_capacity(object_type.len() + body.len() + 32);
2006            framed.extend_from_slice(object_type.as_bytes());
2007            framed.push(b' ');
2008            framed.extend_from_slice(body.len().to_string().as_bytes());
2009            framed.push(0);
2010            framed.extend_from_slice(body);
2011            ObjectId::from_raw(format, &sha256(&framed))
2012        }
2013    }
2014}
2015
2016pub fn digest_bytes(format: ObjectFormat, bytes: &[u8]) -> Result<ObjectId> {
2017    match format {
2018        ObjectFormat::Sha1 => ObjectId::from_raw(format, &sha1(bytes)),
2019        ObjectFormat::Sha256 => ObjectId::from_raw(format, &sha256(bytes)),
2020    }
2021}
2022
2023pub struct StreamingDigest {
2024    format: ObjectFormat,
2025    inner: StreamingDigestInner,
2026}
2027
2028enum StreamingDigestInner {
2029    #[cfg(not(feature = "fast-sha1"))]
2030    Sha1(Sha1Hasher),
2031    #[cfg(feature = "fast-sha1")]
2032    Sha1(sha1::Sha1),
2033    Sha256(Sha256Hasher),
2034}
2035
2036impl StreamingDigest {
2037    pub fn new(format: ObjectFormat) -> Self {
2038        let inner = match format {
2039            #[cfg(not(feature = "fast-sha1"))]
2040            ObjectFormat::Sha1 => StreamingDigestInner::Sha1(Sha1Hasher::new()),
2041            #[cfg(feature = "fast-sha1")]
2042            ObjectFormat::Sha1 => {
2043                use sha1::Digest;
2044                StreamingDigestInner::Sha1(sha1::Sha1::new())
2045            }
2046            ObjectFormat::Sha256 => StreamingDigestInner::Sha256(Sha256Hasher::new()),
2047        };
2048        Self { format, inner }
2049    }
2050
2051    pub fn update(&mut self, data: &[u8]) {
2052        #[cfg(feature = "fetch-profile")]
2053        let _profile_span = fetch_profile::Span::enter(fetch_profile::Stage::OidHash);
2054        #[cfg(feature = "fetch-profile")]
2055        fetch_profile::add_bytes(fetch_profile::Stage::OidHash, data.len() as u64);
2056        match &mut self.inner {
2057            #[cfg(not(feature = "fast-sha1"))]
2058            StreamingDigestInner::Sha1(hasher) => hasher.update(data),
2059            #[cfg(feature = "fast-sha1")]
2060            StreamingDigestInner::Sha1(hasher) => {
2061                use sha1::Digest;
2062                hasher.update(data);
2063            }
2064            StreamingDigestInner::Sha256(hasher) => hasher.update(data),
2065        }
2066    }
2067
2068    pub fn finalize(self) -> Result<ObjectId> {
2069        #[cfg(feature = "fetch-profile")]
2070        let _profile_span = fetch_profile::Span::enter(fetch_profile::Stage::OidHash);
2071        match self.inner {
2072            #[cfg(not(feature = "fast-sha1"))]
2073            StreamingDigestInner::Sha1(hasher) => {
2074                ObjectId::from_raw(self.format, &hasher.finalize())
2075            }
2076            #[cfg(feature = "fast-sha1")]
2077            StreamingDigestInner::Sha1(hasher) => {
2078                use sha1::Digest;
2079                let bytes: [u8; 20] = hasher.finalize().into();
2080                ObjectId::from_raw(self.format, &bytes)
2081            }
2082            StreamingDigestInner::Sha256(hasher) => {
2083                ObjectId::from_raw(self.format, &hasher.finalize())
2084            }
2085        }
2086    }
2087}
2088
2089pub fn to_hex(bytes: &[u8]) -> String {
2090    let mut out = String::with_capacity(bytes.len() * 2);
2091    let _ = write_hex_bytes(bytes, &mut out);
2092    out
2093}
2094
2095fn write_hex_bytes(bytes: &[u8], out: &mut impl fmt::Write) -> fmt::Result {
2096    const HEX: &[u8; 16] = b"0123456789abcdef";
2097    for byte in bytes {
2098        out.write_char(HEX[(byte >> 4) as usize] as char)?;
2099        out.write_char(HEX[(byte & 0x0f) as usize] as char)?;
2100    }
2101    Ok(())
2102}
2103
2104/// Decode a single hex ASCII byte to its nibble value (`'a'` -> `10`).
2105pub fn hex_nibble_value(byte: u8) -> Option<u8> {
2106    match byte {
2107        b'0'..=b'9' => Some(byte - b'0'),
2108        b'a'..=b'f' => Some(byte - b'a' + 10),
2109        b'A'..=b'F' => Some(byte - b'A' + 10),
2110        _ => None,
2111    }
2112}
2113
2114fn hex_nibble(byte: u8) -> Result<u8> {
2115    hex_nibble_value(byte)
2116        .ok_or_else(|| GitError::InvalidObjectId(format!("non-hex byte {:?}", byte as char)))
2117}
2118
2119// ---------------------------------------------------------------------------
2120// SHA-1
2121//
2122// The default is a pure-Rust streaming implementation that hashes 64-byte blocks
2123// straight from the caller's slices, so neither the body nor the framed object is
2124// copied just to be digested. Enabling the `fast-sha1` feature swaps in the
2125// RustCrypto `sha1` crate, which dispatches to ARMv8-SHA1 / x86 SHA-NI at runtime;
2126// the digests are byte-identical, so OIDs are unchanged either way.
2127// ---------------------------------------------------------------------------
2128
2129/// SHA-1 of a raw byte slice (already-framed object, bundle prerequisite, etc.).
2130#[cfg(not(feature = "fast-sha1"))]
2131fn sha1(input: &[u8]) -> [u8; 20] {
2132    let mut hasher = Sha1Hasher::new();
2133    hasher.update(input);
2134    hasher.finalize()
2135}
2136
2137/// SHA-1 of a raw byte slice using the hardware-accelerated backend.
2138#[cfg(feature = "fast-sha1")]
2139fn sha1(input: &[u8]) -> [u8; 20] {
2140    use sha1::{Digest, Sha1};
2141    let mut hasher = Sha1::new();
2142    hasher.update(input);
2143    hasher.finalize().into()
2144}
2145
2146/// SHA-1 of a git object framed as `"<type> <len>\0<body>"`, fed as separate
2147/// updates so the body is never copied into a combined buffer.
2148#[cfg(not(feature = "fast-sha1"))]
2149fn sha1_object_digest(object_type: &str, body: &[u8]) -> [u8; 20] {
2150    let mut hasher = Sha1Hasher::new();
2151    hasher.update(object_type.as_bytes());
2152    hasher.update(b" ");
2153    hasher.update(body.len().to_string().as_bytes());
2154    hasher.update(&[0u8]);
2155    hasher.update(body);
2156    hasher.finalize()
2157}
2158
2159#[cfg(feature = "fast-sha1")]
2160fn sha1_object_digest(object_type: &str, body: &[u8]) -> [u8; 20] {
2161    use sha1::{Digest, Sha1};
2162    let mut hasher = Sha1::new();
2163    hasher.update(object_type.as_bytes());
2164    hasher.update(b" ");
2165    hasher.update(body.len().to_string().as_bytes());
2166    hasher.update([0u8]);
2167    hasher.update(body);
2168    hasher.finalize().into()
2169}
2170
2171/// Streaming pure-Rust SHA-1: feeds full 64-byte blocks directly from each
2172/// `update` slice and buffers only the sub-block remainder, so large inputs are
2173/// hashed without an intermediate copy.
2174#[cfg(not(feature = "fast-sha1"))]
2175struct Sha1Hasher {
2176    state: [u32; 5],
2177    block: [u8; 64],
2178    block_len: usize,
2179    total_len: u64,
2180}
2181
2182#[cfg(not(feature = "fast-sha1"))]
2183impl Sha1Hasher {
2184    fn new() -> Self {
2185        Self {
2186            state: [0x67452301, 0xefcdab89, 0x98badcfe, 0x10325476, 0xc3d2e1f0],
2187            block: [0u8; 64],
2188            block_len: 0,
2189            total_len: 0,
2190        }
2191    }
2192
2193    fn update(&mut self, mut data: &[u8]) {
2194        self.total_len = self.total_len.wrapping_add(data.len() as u64);
2195        if self.block_len > 0 {
2196            let take = (64 - self.block_len).min(data.len());
2197            self.block[self.block_len..self.block_len + take].copy_from_slice(&data[..take]);
2198            self.block_len += take;
2199            data = &data[take..];
2200            if self.block_len == 64 {
2201                let block = self.block;
2202                sha1_compress(&mut self.state, &block);
2203                self.block_len = 0;
2204            }
2205        }
2206        while data.len() >= 64 {
2207            sha1_compress(&mut self.state, &data[..64]);
2208            data = &data[64..];
2209        }
2210        if !data.is_empty() {
2211            self.block[..data.len()].copy_from_slice(data);
2212            self.block_len = data.len();
2213        }
2214    }
2215
2216    fn finalize(mut self) -> [u8; 20] {
2217        let bit_len = self.total_len.wrapping_mul(8);
2218        // 0x80, zero pad to a 56 mod 64 boundary, then the 64-bit big-endian length.
2219        // From a sub-block remainder this is at most two more blocks (128 bytes).
2220        let mut tail = [0u8; 128];
2221        tail[..self.block_len].copy_from_slice(&self.block[..self.block_len]);
2222        tail[self.block_len] = 0x80;
2223        let total = if self.block_len < 56 { 64 } else { 128 };
2224        tail[total - 8..total].copy_from_slice(&bit_len.to_be_bytes());
2225        sha1_compress(&mut self.state, &tail[..64]);
2226        if total == 128 {
2227            sha1_compress(&mut self.state, &tail[64..128]);
2228        }
2229        let mut out = [0u8; 20];
2230        out[0..4].copy_from_slice(&self.state[0].to_be_bytes());
2231        out[4..8].copy_from_slice(&self.state[1].to_be_bytes());
2232        out[8..12].copy_from_slice(&self.state[2].to_be_bytes());
2233        out[12..16].copy_from_slice(&self.state[3].to_be_bytes());
2234        out[16..20].copy_from_slice(&self.state[4].to_be_bytes());
2235        out
2236    }
2237}
2238
2239/// Mix one 64-byte block into the SHA-1 state. `block` must be at least 64 bytes.
2240#[cfg(not(feature = "fast-sha1"))]
2241fn sha1_compress(state: &mut [u32; 5], block: &[u8]) {
2242    let mut w = [0u32; 80];
2243    for (i, word) in w.iter_mut().take(16).enumerate() {
2244        let offset = i * 4;
2245        *word = u32::from_be_bytes([
2246            block[offset],
2247            block[offset + 1],
2248            block[offset + 2],
2249            block[offset + 3],
2250        ]);
2251    }
2252    for i in 16..80 {
2253        w[i] = (w[i - 3] ^ w[i - 8] ^ w[i - 14] ^ w[i - 16]).rotate_left(1);
2254    }
2255
2256    let mut a = state[0];
2257    let mut b = state[1];
2258    let mut c = state[2];
2259    let mut d = state[3];
2260    let mut e = state[4];
2261
2262    for (i, word) in w.iter().enumerate() {
2263        let (f, k) = match i {
2264            0..=19 => ((b & c) | ((!b) & d), 0x5a827999u32),
2265            20..=39 => (b ^ c ^ d, 0x6ed9eba1),
2266            40..=59 => ((b & c) | (b & d) | (c & d), 0x8f1bbcdc),
2267            _ => (b ^ c ^ d, 0xca62c1d6),
2268        };
2269        let temp = a
2270            .rotate_left(5)
2271            .wrapping_add(f)
2272            .wrapping_add(e)
2273            .wrapping_add(k)
2274            .wrapping_add(*word);
2275        e = d;
2276        d = c;
2277        c = b.rotate_left(30);
2278        b = a;
2279        a = temp;
2280    }
2281
2282    state[0] = state[0].wrapping_add(a);
2283    state[1] = state[1].wrapping_add(b);
2284    state[2] = state[2].wrapping_add(c);
2285    state[3] = state[3].wrapping_add(d);
2286    state[4] = state[4].wrapping_add(e);
2287}
2288
2289fn sha256(input: &[u8]) -> [u8; 32] {
2290    let mut hasher = Sha256Hasher::new();
2291    hasher.update(input);
2292    hasher.finalize()
2293}
2294
2295struct Sha256Hasher {
2296    state: [u32; 8],
2297    block: [u8; 64],
2298    block_len: usize,
2299    total_len: u64,
2300}
2301
2302impl Sha256Hasher {
2303    const K: [u32; 64] = [
2304        0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4,
2305        0xab1c5ed5, 0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, 0x72be5d74, 0x80deb1fe,
2306        0x9bdc06a7, 0xc19bf174, 0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f,
2307        0x4a7484aa, 0x5cb0a9dc, 0x76f988da, 0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7,
2308        0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967, 0x27b70a85, 0x2e1b2138, 0x4d2c6dfc,
2309        0x53380d13, 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85, 0xa2bfe8a1, 0xa81a664b,
2310        0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070, 0x19a4c116,
2311        0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3,
2312        0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208, 0x90befffa, 0xa4506ceb, 0xbef9a3f7,
2313        0xc67178f2,
2314    ];
2315
2316    fn new() -> Self {
2317        Self {
2318            state: [
2319                0x6a09e667u32,
2320                0xbb67ae85,
2321                0x3c6ef372,
2322                0xa54ff53a,
2323                0x510e527f,
2324                0x9b05688c,
2325                0x1f83d9ab,
2326                0x5be0cd19,
2327            ],
2328            block: [0u8; 64],
2329            block_len: 0,
2330            total_len: 0,
2331        }
2332    }
2333
2334    fn update(&mut self, mut data: &[u8]) {
2335        self.total_len = self.total_len.wrapping_add(data.len() as u64);
2336        if self.block_len > 0 {
2337            let take = (64 - self.block_len).min(data.len());
2338            self.block[self.block_len..self.block_len + take].copy_from_slice(&data[..take]);
2339            self.block_len += take;
2340            data = &data[take..];
2341            if self.block_len == 64 {
2342                let block = self.block;
2343                self.compress(&block);
2344                self.block_len = 0;
2345            }
2346        }
2347        while data.len() >= 64 {
2348            self.compress(&data[..64]);
2349            data = &data[64..];
2350        }
2351        if !data.is_empty() {
2352            self.block[..data.len()].copy_from_slice(data);
2353            self.block_len = data.len();
2354        }
2355    }
2356
2357    fn finalize(mut self) -> [u8; 32] {
2358        let bit_len = self.total_len.wrapping_mul(8);
2359        let mut tail = [0u8; 128];
2360        tail[..self.block_len].copy_from_slice(&self.block[..self.block_len]);
2361        tail[self.block_len] = 0x80;
2362        let total = if self.block_len < 56 { 64 } else { 128 };
2363        tail[total - 8..total].copy_from_slice(&bit_len.to_be_bytes());
2364        self.compress(&tail[..64]);
2365        if total == 128 {
2366            self.compress(&tail[64..128]);
2367        }
2368
2369        let mut out = [0; 32];
2370        for (idx, word) in self.state.iter().enumerate() {
2371            out[idx * 4..idx * 4 + 4].copy_from_slice(&word.to_be_bytes());
2372        }
2373        out
2374    }
2375
2376    fn compress(&mut self, chunk: &[u8]) {
2377        let mut w = [0u32; 64];
2378        for (i, word) in w.iter_mut().take(16).enumerate() {
2379            let offset = i * 4;
2380            *word = u32::from_be_bytes([
2381                chunk[offset],
2382                chunk[offset + 1],
2383                chunk[offset + 2],
2384                chunk[offset + 3],
2385            ]);
2386        }
2387        for i in 16..64 {
2388            let s0 = w[i - 15].rotate_right(7) ^ w[i - 15].rotate_right(18) ^ (w[i - 15] >> 3);
2389            let s1 = w[i - 2].rotate_right(17) ^ w[i - 2].rotate_right(19) ^ (w[i - 2] >> 10);
2390            w[i] = w[i - 16]
2391                .wrapping_add(s0)
2392                .wrapping_add(w[i - 7])
2393                .wrapping_add(s1);
2394        }
2395
2396        let mut a = self.state[0];
2397        let mut b = self.state[1];
2398        let mut c = self.state[2];
2399        let mut d = self.state[3];
2400        let mut e = self.state[4];
2401        let mut f = self.state[5];
2402        let mut g = self.state[6];
2403        let mut hh = self.state[7];
2404
2405        for (&word, &constant) in w.iter().zip(Self::K.iter()) {
2406            let s1 = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25);
2407            let ch = (e & f) ^ ((!e) & g);
2408            let temp1 = hh
2409                .wrapping_add(s1)
2410                .wrapping_add(ch)
2411                .wrapping_add(constant)
2412                .wrapping_add(word);
2413            let s0 = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22);
2414            let maj = (a & b) ^ (a & c) ^ (b & c);
2415            let temp2 = s0.wrapping_add(maj);
2416
2417            hh = g;
2418            g = f;
2419            f = e;
2420            e = d.wrapping_add(temp1);
2421            d = c;
2422            c = b;
2423            b = a;
2424            a = temp1.wrapping_add(temp2);
2425        }
2426
2427        self.state[0] = self.state[0].wrapping_add(a);
2428        self.state[1] = self.state[1].wrapping_add(b);
2429        self.state[2] = self.state[2].wrapping_add(c);
2430        self.state[3] = self.state[3].wrapping_add(d);
2431        self.state[4] = self.state[4].wrapping_add(e);
2432        self.state[5] = self.state[5].wrapping_add(f);
2433        self.state[6] = self.state[6].wrapping_add(g);
2434        self.state[7] = self.state[7].wrapping_add(hh);
2435    }
2436}
2437
2438#[cfg(test)]
2439mod tests {
2440    use super::*;
2441    use std::io::ErrorKind;
2442
2443    #[test]
2444    fn io_error_conversion_preserves_kind_and_message() {
2445        let err = GitError::from(std::io::Error::new(
2446            ErrorKind::PermissionDenied,
2447            "sealed away",
2448        ));
2449        assert_eq!(err.io_kind(), Some(ErrorKind::PermissionDenied));
2450        assert!(!err.is_cancelled());
2451        // Display parity with the legacy string form.
2452        assert_eq!(err.to_string(), "io error: sealed away");
2453    }
2454
2455    #[test]
2456    fn cancel_payload_round_trips_to_cancelled_variant() {
2457        let err = GitError::from(cancelled_io_error());
2458        assert_eq!(err, GitError::Cancelled);
2459        assert!(err.is_cancelled());
2460        assert!(is_cancelled_error(&err));
2461    }
2462
2463    #[test]
2464    fn is_cancelled_covers_structured_and_legacy_shapes() {
2465        assert!(GitError::Cancelled.is_cancelled());
2466        let interrupted = GitError::from(std::io::Error::new(ErrorKind::Interrupted, "wake-up"));
2467        assert!(
2468            interrupted.is_cancelled(),
2469            "Interrupted kind is cancel-flavored"
2470        );
2471        assert!(
2472            GitError::IoKind {
2473                kind: ErrorKind::Interrupted,
2474                message: "operation cancelled".into()
2475            }
2476            .is_cancelled()
2477        );
2478        assert!(!GitError::from(std::io::Error::other("disk full")).is_cancelled());
2479        assert_eq!(
2480            GitError::from(std::io::Error::other("disk full")).io_kind(),
2481            Some(ErrorKind::Other)
2482        );
2483    }
2484
2485    #[test]
2486    fn sideband_fatal_displays_wire_text() {
2487        let err = GitError::SidebandFatal("remote died".into());
2488        assert_eq!(err.to_string(), "sideband fatal: remote died");
2489    }
2490
2491    #[test]
2492    fn typed_git_error_payload_survives_io_boundary() {
2493        let wrapped = std::io::Error::new(
2494            ErrorKind::InvalidData,
2495            GitError::SidebandFatal("boom".into()),
2496        );
2497        assert_eq!(
2498            GitError::from(wrapped),
2499            GitError::SidebandFatal("boom".into())
2500        );
2501    }
2502
2503    #[test]
2504    fn sha1_blob_matches_git_known_value() {
2505        let oid = object_id_for_bytes(ObjectFormat::Sha1, "blob", b"hello\n")
2506            .expect("known blob should hash as sha1");
2507        assert_eq!(oid.to_hex(), "ce013625030ba8dba906f756967f9e9ca394464a");
2508    }
2509
2510    #[test]
2511    fn sha256_blob_matches_git_known_value() {
2512        let oid = object_id_for_bytes(ObjectFormat::Sha256, "blob", b"hello\n")
2513            .expect("known blob should hash as sha256");
2514        assert_eq!(
2515            oid.to_hex(),
2516            "2cf8d83d9ee29543b34a87727421fdecb7e3f3a183d337639025de576db9ebb4"
2517        );
2518    }
2519
2520    #[test]
2521    fn object_id_round_trips_hex() {
2522        let oid = ObjectId::from_hex(
2523            ObjectFormat::Sha1,
2524            "ce013625030ba8dba906f756967f9e9ca394464a",
2525        )
2526        .expect("valid sha1 hex");
2527        assert_eq!(oid.to_hex(), "ce013625030ba8dba906f756967f9e9ca394464a");
2528    }
2529
2530    #[test]
2531    fn object_id_writes_hex_without_allocating_in_the_writer() {
2532        let oid = ObjectId::from_hex(
2533            ObjectFormat::Sha1,
2534            "CE013625030BA8DBA906F756967F9E9CA394464A",
2535        )
2536        .expect("valid uppercase sha1 hex");
2537
2538        let mut out = String::new();
2539        oid.write_hex(&mut out)
2540            .expect("writing object id hex to a String should not fail");
2541
2542        assert_eq!(out, "ce013625030ba8dba906f756967f9e9ca394464a");
2543        assert_eq!(oid.to_hex(), out);
2544        assert_eq!(format!("{oid}"), out);
2545    }
2546
2547    #[test]
2548    fn object_id_matches_hex_prefixes_by_nibble() {
2549        let oid = ObjectId::from_hex(
2550            ObjectFormat::Sha1,
2551            "ce013625030ba8dba906f756967f9e9ca394464a",
2552        )
2553        .expect("valid sha1 hex");
2554
2555        assert!(oid.hex_prefix_matches(b""));
2556        assert!(oid.hex_prefix_matches(b"c"));
2557        assert!(oid.hex_prefix_matches(b"ce013"));
2558        assert!(oid.hex_prefix_matches(b"CE013625"));
2559        assert!(oid.hex_prefix_matches(b"ce013625030ba8dba906f756967f9e9ca394464a"));
2560
2561        assert!(!oid.hex_prefix_matches(b"d"));
2562        assert!(!oid.hex_prefix_matches(b"ce014"));
2563        assert!(!oid.hex_prefix_matches(b"ce01x"));
2564
2565        let mut too_long = oid.to_hex();
2566        too_long.push('0');
2567        assert!(!oid.hex_prefix_matches(too_long.as_bytes()));
2568    }
2569
2570    #[test]
2571    fn object_id_abbrev_hex_len_clamps_to_format_width() {
2572        let sha1 = ObjectId::null(ObjectFormat::Sha1);
2573        let sha256 = ObjectId::null(ObjectFormat::Sha256);
2574
2575        assert_eq!(sha1.abbrev_hex_len(0), 0);
2576        assert_eq!(sha1.abbrev_hex_len(12), 12);
2577        assert_eq!(sha1.abbrev_hex_len(80), ObjectFormat::Sha1.hex_len());
2578        assert_eq!(sha256.abbrev_hex_len(80), ObjectFormat::Sha256.hex_len());
2579    }
2580
2581    #[test]
2582    fn signature_parses_a_normal_ident_and_round_trips() {
2583        let line = b"A U Thor <author@example.com> 1700000000 +0000";
2584        let sig = Signature::from_ident_line(line).expect("well-formed ident parses");
2585        assert_eq!(sig.name.as_bytes(), b"A U Thor");
2586        assert_eq!(sig.email.as_bytes(), b"author@example.com");
2587        assert_eq!(sig.time.seconds, 1_700_000_000);
2588        assert_eq!(sig.time.timezone_offset_minutes, 0);
2589        assert!(!sig.time.negative_utc);
2590        // Byte-exact round-trip, and the canonical form matches here too.
2591        assert_eq!(sig.to_ident_bytes(), line);
2592        assert_eq!(sig.to_canonical_ident_bytes(), line);
2593    }
2594
2595    #[test]
2596    fn signature_parses_positive_half_hour_offset() {
2597        let line = b"Half Hour <hh@example.com> 1500000000 +0530";
2598        let sig = Signature::from_ident_line(line).expect("offset ident parses");
2599        assert_eq!(sig.time.timezone_offset_minutes, 330);
2600        assert!(!sig.time.negative_utc);
2601        assert_eq!(sig.time.offset_token(), "+0530");
2602        assert_eq!(sig.to_ident_bytes(), line);
2603        assert_eq!(sig.to_canonical_ident_bytes(), line);
2604    }
2605
2606    #[test]
2607    fn signature_parses_negative_offset() {
2608        let line = b"Western <w@example.com> 1500000000 -0500";
2609        let sig = Signature::from_ident_line(line).expect("negative offset parses");
2610        assert_eq!(sig.time.timezone_offset_minutes, -300);
2611        assert!(!sig.time.negative_utc);
2612        assert_eq!(sig.time.offset_token(), "-0500");
2613        assert_eq!(sig.to_ident_bytes(), line);
2614    }
2615
2616    #[test]
2617    fn signature_preserves_negative_zero_timezone_distinct_from_positive_zero() {
2618        let negative = b"Unknown Zone <uz@example.com> 1500000000 -0000";
2619        let positive = b"Known Zone <kz@example.com> 1500000000 +0000";
2620
2621        let neg = Signature::from_ident_line(negative).expect("-0000 parses");
2622        let pos = Signature::from_ident_line(positive).expect("+0000 parses");
2623
2624        // Both are zero minutes from UTC...
2625        assert_eq!(neg.time.timezone_offset_minutes, 0);
2626        assert_eq!(pos.time.timezone_offset_minutes, 0);
2627        // ...but the sentinel flag distinguishes them, so the times differ.
2628        assert!(neg.time.negative_utc);
2629        assert!(!pos.time.negative_utc);
2630        assert_ne!(neg.time, pos.time);
2631
2632        // And the distinction survives re-serialization, byte-for-byte.
2633        assert_eq!(neg.time.offset_token(), "-0000");
2634        assert_eq!(pos.time.offset_token(), "+0000");
2635        assert_eq!(neg.to_ident_bytes(), negative);
2636        assert_eq!(pos.to_ident_bytes(), positive);
2637        assert_eq!(neg.to_canonical_ident_bytes(), negative);
2638        assert_eq!(pos.to_canonical_ident_bytes(), positive);
2639        assert_ne!(neg.to_ident_bytes(), pos.to_ident_bytes());
2640    }
2641
2642    #[test]
2643    fn signature_handles_empty_name_and_email() {
2644        // git permits an empty name and/or empty email; the delimiters still
2645        // anchor the parse.
2646        let line = b" <> 0 +0000";
2647        let sig = Signature::from_ident_line(line).expect("empty name/email parses");
2648        assert_eq!(sig.name.as_bytes(), b"");
2649        assert_eq!(sig.email.as_bytes(), b"");
2650        assert_eq!(sig.time.seconds, 0);
2651        assert_eq!(sig.to_ident_bytes(), line);
2652    }
2653
2654    #[test]
2655    fn signature_keeps_angle_brackets_inside_the_name() {
2656        // The email is delimited by the *last* '<'/'>' pair, so a name that
2657        // itself contains angle brackets parses with the trailing pair as the
2658        // email and round-trips exactly.
2659        let line = b"Weird <Name> <weird@example.com> 1 +0000";
2660        let sig = Signature::from_ident_line(line).expect("bracketed name parses");
2661        assert_eq!(sig.name.as_bytes(), b"Weird <Name>");
2662        assert_eq!(sig.email.as_bytes(), b"weird@example.com");
2663        assert_eq!(sig.to_ident_bytes(), line);
2664    }
2665
2666    #[test]
2667    fn signature_round_trips_non_canonical_whitespace_via_raw() {
2668        // An ident with two spaces before the email is not git's canonical form,
2669        // but the parse-view must still reproduce it byte-for-byte from `raw`.
2670        // (Only the canonical renderer normalizes the spacing.)
2671        let line = b"Spaced  <spaced@example.com> 5 +0000";
2672        let sig = Signature::from_ident_line(line).expect("non-canonical ident parses");
2673        // The name keeps the extra space (only one separator space is trimmed).
2674        assert_eq!(sig.name.as_bytes(), b"Spaced ");
2675        assert_eq!(sig.to_ident_bytes(), line);
2676    }
2677
2678    #[test]
2679    fn signature_rejects_malformed_idents() {
2680        // No email delimiters.
2681        assert!(Signature::from_ident_line(b"No Email Here 0 +0000").is_none());
2682        // Missing the time tail entirely.
2683        assert!(Signature::from_ident_line(b"A U Thor <a@example.com>").is_none());
2684        // Non-numeric timestamp.
2685        assert!(Signature::from_ident_line(b"A U Thor <a@example.com> later +0000").is_none());
2686        // Malformed timezone token (wrong width).
2687        assert!(Signature::from_ident_line(b"A U Thor <a@example.com> 0 +00").is_none());
2688        // Timezone token missing a sign.
2689        assert!(Signature::from_ident_line(b"A U Thor <a@example.com> 0 0000").is_none());
2690    }
2691
2692    #[test]
2693    fn git_time_constructors_set_the_sentinel() {
2694        assert!(!GitTime::new(0, 0).negative_utc);
2695        assert_eq!(GitTime::new(0, 330).offset_token(), "+0530");
2696        let unknown = GitTime::with_negative_utc(42);
2697        assert!(unknown.negative_utc);
2698        assert_eq!(unknown.seconds, 42);
2699        assert_eq!(unknown.offset_token(), "-0000");
2700    }
2701
2702    #[test]
2703    fn full_name_accepts_valid_ref_names() {
2704        let name = FullName::new("refs/heads/main").expect("valid ref name");
2705        assert_eq!(name.as_str(), "refs/heads/main");
2706        assert_eq!(name, "refs/heads/main");
2707        assert_eq!(format!("{name}"), "refs/heads/main");
2708        assert_eq!(String::from(name.clone()), "refs/heads/main");
2709        let borrowed: &str = name.borrow();
2710        assert_eq!(borrowed, "refs/heads/main");
2711    }
2712
2713    #[test]
2714    fn full_name_rejects_invalid_ref_names() {
2715        assert!(FullName::new("").is_err());
2716        assert!(FullName::new(" refs/heads/main").is_err());
2717        assert!(FullName::new("refs/heads/main ").is_err());
2718        assert!(FullName::new("refs//heads/main").is_err());
2719        assert!(FullName::new("refs/heads/\nmain").is_err());
2720        assert!(FullName::new("refs/heads/a..b").is_err());
2721        assert!(FullName::new("refs/heads/a.lock").is_err());
2722        assert!(FullName::new("refs/heads/a~1").is_err());
2723        assert!(FullName::new("@").is_err());
2724    }
2725
2726    #[test]
2727    fn full_name_accepts_what_git_accepts() {
2728        // HeddleCo/sley#244: non-ASCII whitespace is not special to Git.
2729        assert!(FullName::new("refs/heads/\u{00A0}edge\u{00A0}").is_ok());
2730        assert!(FullName::new("HEAD").is_ok());
2731        assert!(FullName::new("refs/heads/a./b").is_ok());
2732    }
2733
2734    #[test]
2735    fn bstring_round_trips_bytes_and_displays_lossily() {
2736        let path = BString::from_bytes(b"src/\xFF.txt");
2737        assert_eq!(path.as_bytes(), b"src/\xFF.txt");
2738        let borrowed: &[u8] = path.borrow();
2739        assert_eq!(borrowed, b"src/\xFF.txt".as_slice());
2740        assert_eq!(format!("{path}"), "src/\u{FFFD}.txt");
2741        assert_eq!(path, b"src/\xFF.txt");
2742        assert_eq!(path.clone().into_bytes(), b"src/\xFF.txt".to_vec());
2743    }
2744
2745    #[test]
2746    fn split_ident_line_parses_well_formed_ident() {
2747        let f = split_ident_line(b"A U Thor <author@example.com> 1112911993 -0700")
2748            .expect("well formed ident should parse");
2749        assert_eq!(f.name, b"A U Thor");
2750        assert_eq!(f.email, b"author@example.com");
2751        assert_eq!(f.date, Some(&b"1112911993"[..]));
2752        assert_eq!(f.tz, Some(&b"-0700"[..]));
2753    }
2754
2755    #[test]
2756    fn split_ident_line_recovers_broken_email() {
2757        // git inserts junk after the '>': email stops at the first '>', but the
2758        // timestamp is found by scanning back from the end for the last '>'.
2759        let f = split_ident_line(b"A U Thor <author@example.com>-<> 1112911993 -0700")
2760            .expect("broken-email ident should parse");
2761        assert_eq!(f.name, b"A U Thor");
2762        assert_eq!(f.email, b"author@example.com");
2763        assert_eq!(f.date, Some(&b"1112911993"[..]));
2764        assert_eq!(f.tz, Some(&b"-0700"[..]));
2765    }
2766
2767    #[test]
2768    fn split_ident_line_non_numeric_date_is_person_only() {
2769        let f = split_ident_line(b"A U Thor <author@example.com> totally_bogus -0700")
2770            .expect("ident without numeric date should still parse person");
2771        assert_eq!(f.email, b"author@example.com");
2772        assert_eq!(f.date, None);
2773        assert_eq!(f.tz, None);
2774    }
2775
2776    #[test]
2777    fn split_ident_line_whitespace_date_is_person_only() {
2778        // Trailing spaces after '>' with no timestamp -> no date.
2779        let f = split_ident_line(b"A U Thor <author@example.com>    ")
2780            .expect("ident with trailing whitespace should parse person");
2781        assert_eq!(f.date, None);
2782        // A vertical tab is NOT git-isspace, so it stops the space-skip and the
2783        // (non-digit) VT yields no date either.
2784        let f = split_ident_line(b"A U Thor <author@example.com>   \x0b")
2785            .expect("ident with non-git-whitespace suffix should parse person");
2786        assert_eq!(f.date, None);
2787    }
2788
2789    #[test]
2790    fn split_ident_line_requires_angle_brackets() {
2791        assert!(split_ident_line(b"no brackets here 123 +0000").is_none());
2792    }
2793
2794    #[test]
2795    fn ident_render_date_overflow_is_epoch_sentinel() {
2796        // 2^64 + 1 (clamps in u64 parse) and 2^64 - 2 (fits u64 but past time_t)
2797        // both render the epoch sentinel with a forced +0000 timezone.
2798        assert_eq!(
2799            ident_render_date(b"18446744073709551617", b"-0700", &DateMode::Default),
2800            "Thu Jan 1 00:00:00 1970 +0000"
2801        );
2802        assert_eq!(
2803            ident_render_date(b"18446744073709551614", b"-0700", &DateMode::Default),
2804            "Thu Jan 1 00:00:00 1970 +0000"
2805        );
2806    }
2807
2808    #[test]
2809    fn ident_render_date_valid_value_uses_original_timezone() {
2810        assert_eq!(
2811            ident_render_date(b"0", b"+0000", &DateMode::Default),
2812            "Thu Jan 1 00:00:00 1970 +0000"
2813        );
2814    }
2815
2816    #[test]
2817    fn redact_url_for_display_strips_https_userinfo() {
2818        assert_eq!(
2819            redact_url_for_display("https://user:pass@host/repo.git"),
2820            "https://<redacted>@host/repo.git"
2821        );
2822    }
2823
2824    #[test]
2825    fn redact_url_for_display_leaves_urls_without_userinfo_unchanged() {
2826        assert_eq!(
2827            redact_url_for_display("https://host/repo.git"),
2828            "https://host/repo.git"
2829        );
2830        assert_eq!(redact_url_for_display("origin"), "origin");
2831    }
2832}