Skip to main content

sley_core/
lib.rs

1#![cfg_attr(not(test), deny(clippy::unwrap_used, clippy::expect_used))]
2
3use std::borrow::Borrow;
4use std::error::Error;
5use std::fmt;
6use std::ops::Deref;
7use std::path::{Path, PathBuf};
8use std::str::FromStr;
9
10mod cancel;
11pub mod diagnostics;
12
13#[cfg(feature = "fetch-profile")]
14pub mod fetch_profile;
15
16pub use cancel::{
17    AtomicCancel, CancelFlag, CancellableRead, DynCancelFlag, OperationCancelled, StreamControl,
18    cancelled_io_error, is_cancelled_error, is_cancelled_io, kill_child_if_cancelled,
19    map_cancel_io,
20};
21
22pub const UPSTREAM_GIT_COMPAT_VERSION: &str = "2.55.0";
23
24/// Maximum symbolic-ref hops git follows while resolving one ref
25/// (`refs.c` `SYMREF_MAXDEPTH`). Oracle 2.55 resolves a chain of four symrefs
26/// plus a final direct ref and reports a dangling/looped ref at five hops.
27pub const MAX_SYMREF_DEPTH: usize = 5;
28
29pub mod atomic;
30pub mod date;
31pub mod fsync;
32pub mod path_safety;
33pub mod paths;
34pub mod precompose;
35pub mod primitives;
36pub mod refname;
37pub mod text;
38pub use precompose::{PrecomposeUnicode, has_non_ascii};
39pub use refname::{RefnameFormat, RefnameFormatError, check_refname_format};
40
41pub mod namespace;
42pub use namespace::{Namespace, ref_is_hidden, trim_hidden_ref_pattern};
43
44#[derive(Debug, Default, Clone, PartialEq, Eq)]
45pub enum DateMode {
46    #[default]
47    Default,
48    Local,
49    Raw,
50    RawLocal,
51    Unix,
52    Short,
53    ShortLocal,
54    Iso,
55    IsoLocal,
56    IsoStrict,
57    IsoStrictLocal,
58    Rfc2822,
59    Rfc2822Local,
60    Relative,
61    Human,
62    HumanLocal,
63    Strftime {
64        template: String,
65        local: bool,
66    },
67}
68
69impl DateMode {
70    pub fn parse(value: &str) -> Option<Self> {
71        if let Some(template) = value.strip_prefix("format:") {
72            return Some(Self::Strftime {
73                template: template.to_string(),
74                local: false,
75            });
76        }
77        if let Some(template) = value.strip_prefix("format-local:") {
78            return Some(Self::Strftime {
79                template: template.to_string(),
80                local: true,
81            });
82        }
83        if value == "tformat:" || value.starts_with("tformat:") {
84            return Some(Self::Strftime {
85                template: value["tformat:".len()..].to_string(),
86                local: false,
87            });
88        }
89        if value == "auto:" || value.starts_with("auto:") {
90            return Some(Self::Default);
91        }
92        Some(match value {
93            "default" => Self::Default,
94            "default-local" | "local" => Self::Local,
95            "raw" => Self::Raw,
96            "raw-local" => Self::RawLocal,
97            "unix" => Self::Unix,
98            "short" => Self::Short,
99            "short-local" => Self::ShortLocal,
100            "iso" | "iso8601" => Self::Iso,
101            "iso-local" | "iso8601-local" => Self::IsoLocal,
102            "iso-strict" | "iso8601-strict" => Self::IsoStrict,
103            "iso-strict-local" | "iso8601-strict-local" => Self::IsoStrictLocal,
104            "rfc" | "rfc2822" => Self::Rfc2822,
105            "rfc-local" | "rfc2822-local" => Self::Rfc2822Local,
106            "relative" | "relative-local" => Self::Relative,
107            "human" => Self::Human,
108            "human-local" => Self::HumanLocal,
109            _ => return None,
110        })
111    }
112
113    pub fn parse_atom_modifier(modifier: Option<&str>) -> Option<Self> {
114        modifier.map_or(Some(Self::Default), Self::parse)
115    }
116
117    pub fn render(&self, timestamp: i64, timezone: &str) -> Option<String> {
118        let tz = if self.is_local() { "+0000" } else { timezone };
119        let parts = DateParts::from_timestamp(timestamp, tz)?;
120        Some(match self {
121            Self::Default | Self::Local => {
122                let base = format!(
123                    "{} {} {} {:02}:{:02}:{:02} {}",
124                    parts.weekday,
125                    MONTHS_ABBR[(parts.month - 1) as usize],
126                    parts.day,
127                    parts.hour,
128                    parts.minute,
129                    parts.second,
130                    parts.year,
131                );
132                if self.is_local() {
133                    base
134                } else {
135                    format!("{base} {}", parts.timezone)
136                }
137            }
138            Self::Raw | Self::RawLocal => format!("{} {}", parts.timestamp, parts.timezone),
139            Self::Unix => parts.timestamp.to_string(),
140            Self::Short | Self::ShortLocal => {
141                format!("{:04}-{:02}-{:02}", parts.year, parts.month, parts.day)
142            }
143            Self::Iso | Self::IsoLocal => format!(
144                "{:04}-{:02}-{:02} {:02}:{:02}:{:02} {}",
145                parts.year,
146                parts.month,
147                parts.day,
148                parts.hour,
149                parts.minute,
150                parts.second,
151                parts.timezone,
152            ),
153            Self::IsoStrict | Self::IsoStrictLocal => format!(
154                "{:04}-{:02}-{:02}T{:02}:{:02}:{:02}{}",
155                parts.year,
156                parts.month,
157                parts.day,
158                parts.hour,
159                parts.minute,
160                parts.second,
161                strict_timezone(parts.timezone),
162            ),
163            Self::Rfc2822 | Self::Rfc2822Local => format!(
164                "{}, {} {} {:04} {:02}:{:02}:{:02} {}",
165                parts.weekday,
166                parts.day,
167                MONTHS_ABBR[(parts.month - 1) as usize],
168                parts.year,
169                parts.hour,
170                parts.minute,
171                parts.second,
172                parts.timezone,
173            ),
174            Self::Relative => relative_date(parts.timestamp),
175            Self::Human | Self::HumanLocal => format!(
176                "{} {} {} {:02}:{:02}:{:02} {} {}",
177                parts.weekday,
178                MONTHS_ABBR[(parts.month - 1) as usize],
179                parts.day,
180                parts.hour,
181                parts.minute,
182                parts.second,
183                parts.year,
184                parts.timezone,
185            ),
186            Self::Strftime { template, .. } => strftime(template, &parts),
187        })
188    }
189
190    pub fn is_local(&self) -> bool {
191        matches!(
192            self,
193            Self::Local
194                | Self::RawLocal
195                | Self::ShortLocal
196                | Self::IsoLocal
197                | Self::IsoStrictLocal
198                | Self::Rfc2822Local
199                | Self::HumanLocal
200                | Self::Strftime { local: true, .. }
201        )
202    }
203}
204
205const MONTHS_ABBR: [&str; 12] = [
206    "Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec",
207];
208
209const MONTHS_FULL: [&str; 12] = [
210    "January",
211    "February",
212    "March",
213    "April",
214    "May",
215    "June",
216    "July",
217    "August",
218    "September",
219    "October",
220    "November",
221    "December",
222];
223
224const WEEKDAYS_FULL: [&str; 7] = [
225    "Sunday",
226    "Monday",
227    "Tuesday",
228    "Wednesday",
229    "Thursday",
230    "Friday",
231    "Saturday",
232];
233
234struct DateParts<'a> {
235    timestamp: i64,
236    timezone: &'a str,
237    weekday: &'static str,
238    year: i64,
239    month: u32,
240    day: u32,
241    hour: i64,
242    minute: i64,
243    second: i64,
244}
245
246impl<'a> DateParts<'a> {
247    fn from_timestamp(timestamp: i64, timezone: &'a str) -> Option<Self> {
248        const WEEKDAYS: [&str; 7] = ["Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat"];
249        let offset_seconds = timezone_offset_seconds(timezone)?;
250        let local = timestamp + offset_seconds;
251        let days = local.div_euclid(86_400);
252        let seconds = local.rem_euclid(86_400);
253        let (year, month, day) = civil_from_days(days);
254        Some(Self {
255            timestamp,
256            timezone,
257            weekday: WEEKDAYS[(days + 4).rem_euclid(7) as usize],
258            year,
259            month,
260            day,
261            hour: seconds / 3_600,
262            minute: (seconds % 3_600) / 60,
263            second: seconds % 60,
264        })
265    }
266}
267
268fn timezone_offset_seconds(timezone: &str) -> Option<i64> {
269    if timezone.len() != 5 {
270        return None;
271    }
272    let sign = match timezone.as_bytes()[0] {
273        b'+' => 1,
274        b'-' => -1,
275        _ => return None,
276    };
277    let hours = timezone[1..3].parse::<i64>().ok()?;
278    let minutes = timezone[3..5].parse::<i64>().ok()?;
279    Some(sign * (hours * 3_600 + minutes * 60))
280}
281
282fn strict_timezone(timezone: &str) -> String {
283    let digits = timezone.strip_prefix(['+', '-']).unwrap_or(timezone);
284    if digits == "0000" {
285        "Z".to_string()
286    } else if timezone.len() == 5 {
287        format!("{}{}:{}", &timezone[..1], &timezone[1..3], &timezone[3..5])
288    } else {
289        timezone.to_string()
290    }
291}
292
293fn strftime(template: &str, parts: &DateParts<'_>) -> String {
294    let weekday_index = ["Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat"]
295        .iter()
296        .position(|day| *day == parts.weekday)
297        .unwrap_or(0);
298    let mut out = String::with_capacity(template.len());
299    let mut chars = template.chars();
300    while let Some(ch) = chars.next() {
301        if ch != '%' {
302            out.push(ch);
303            continue;
304        }
305        match chars.next() {
306            Some('Y') => out.push_str(&format!("{:04}", parts.year)),
307            Some('y') => out.push_str(&format!("{:02}", parts.year.rem_euclid(100))),
308            Some('m') => out.push_str(&format!("{:02}", parts.month)),
309            Some('d') => out.push_str(&format!("{:02}", parts.day)),
310            Some('e') => out.push_str(&format!("{:2}", parts.day)),
311            Some('H') => out.push_str(&format!("{:02}", parts.hour)),
312            Some('M') => out.push_str(&format!("{:02}", parts.minute)),
313            Some('S') => out.push_str(&format!("{:02}", parts.second)),
314            Some('b') | Some('h') => out.push_str(MONTHS_ABBR[(parts.month - 1) as usize]),
315            Some('B') => out.push_str(MONTHS_FULL[(parts.month - 1) as usize]),
316            Some('a') => out.push_str(parts.weekday),
317            Some('A') => out.push_str(WEEKDAYS_FULL[weekday_index]),
318            Some('%') => out.push('%'),
319            Some('n') => out.push('\n'),
320            Some('t') => out.push('\t'),
321            Some(other) => {
322                out.push('%');
323                out.push(other);
324            }
325            None => out.push('%'),
326        }
327    }
328    out
329}
330
331fn relative_date(timestamp: i64) -> String {
332    let now = std::time::SystemTime::now()
333        .duration_since(std::time::UNIX_EPOCH)
334        .map(|duration| duration.as_secs() as i64)
335        .unwrap_or(timestamp);
336    if timestamp > now {
337        return "in the future".to_string();
338    }
339    let diff = (now - timestamp) as u64;
340    if diff < 90 {
341        return format!("{diff} seconds ago");
342    }
343    let minutes = (diff + 30) / 60;
344    if minutes < 90 {
345        return format!("{minutes} minutes ago");
346    }
347    let hours = (diff + 1800) / 3600;
348    if hours < 36 {
349        return format!("{hours} hours ago");
350    }
351    let days = (diff + 43200) / 86400;
352    if days < 14 {
353        return format!("{days} days ago");
354    }
355    if days < 70 {
356        return format!("{} weeks ago", (days + 3) / 7);
357    }
358    if days < 365 {
359        return format!("{} months ago", (days + 15) / 30);
360    }
361    let years_scaled = (days * 10 + 183) / 365;
362    if days < 365 * 2 {
363        let months = ((days - 365) + 15) / 30;
364        if months > 0 {
365            return format!("1 year, {months} months ago");
366        }
367        return "1 year ago".to_string();
368    }
369    if years_scaled.is_multiple_of(10) {
370        format!("{} years ago", years_scaled / 10)
371    } else {
372        format!("{}.{} years ago", years_scaled / 10, years_scaled % 10)
373    }
374}
375
376use crate::date::civil_from_days;
377
378fn is_scheme_char(ch: char) -> bool {
379    ch.is_ascii_alphanumeric() || matches!(ch, '+' | '-' | '.')
380}
381
382/// Strip embedded credentials from `url` before showing it in user-facing output.
383///
384/// HTTP(S) userinfo (`user:password@host`) is replaced with `<redacted>@host`,
385/// matching trace2's `GIT_TRACE2_REDACT` behavior. Non-URL strings (remote
386/// names, file paths) are returned unchanged.
387pub fn redact_url_for_display(url: &str) -> String {
388    let mut out = String::with_capacity(url.len());
389    let mut rest = url;
390    while let Some(scheme_end) = rest.find("://") {
391        let scheme_start = rest[..scheme_end]
392            .char_indices()
393            .rev()
394            .find_map(|(idx, ch)| (!is_scheme_char(ch)).then_some(idx + ch.len_utf8()))
395            .unwrap_or(0);
396        out.push_str(&rest[..scheme_start]);
397
398        let authority_start = scheme_end + 3;
399        let authority_end = rest[authority_start..]
400            .find(|ch: char| ['/', '?', '#', ' ', '\t', '\r', '\n'].contains(&ch))
401            .map(|idx| authority_start + idx)
402            .unwrap_or(rest.len());
403        let authority = &rest[authority_start..authority_end];
404        if let Some(at) = authority.rfind('@') {
405            out.push_str(&rest[scheme_start..authority_start]);
406            out.push_str("<redacted>@");
407            out.push_str(&authority[at + 1..]);
408        } else {
409            out.push_str(&rest[scheme_start..authority_end]);
410        }
411        rest = &rest[authority_end..];
412    }
413    out.push_str(rest);
414    out
415}
416
417/// Minimal trace2 event-target support (`GIT_TRACE2_EVENT`).
418///
419/// Upstream's trace2 event target writes one JSON object per line to the file
420/// named by `GIT_TRACE2_EVENT`. sley emits only the `data` events the test
421/// suite asserts on (`test_trace2_data` greps for the contiguous
422/// `"category":"...","key":"...","value":"..."` triple), with the same field
423/// order trace2's `fn_data_fl` produces. Unset/unwritable targets are
424/// silently ignored, like upstream's best-effort tracing.
425pub mod trace2 {
426    use std::fmt::Display;
427    use std::fmt::Write as _;
428    use std::io::Write;
429    use std::path::PathBuf;
430
431    fn escape_json(raw: &str) -> String {
432        let mut out = String::with_capacity(raw.len());
433        for ch in raw.chars() {
434            match ch {
435                '"' => out.push_str("\\\""),
436                '\\' => out.push_str("\\\\"),
437                '\n' => out.push_str("\\n"),
438                '\t' => out.push_str("\\t"),
439                ch if (ch as u32) < 0x20 => {
440                    let _ = write!(out, "\\u{:04x}", ch as u32);
441                }
442                ch => out.push(ch),
443            }
444        }
445        out
446    }
447
448    enum TraceTarget {
449        Stderr,
450        Path(String),
451    }
452
453    fn trace_target(var: &str) -> Option<TraceTarget> {
454        let target = std::env::var_os(var)?.to_string_lossy().into_owned();
455        match target.as_str() {
456            "1" | "true" => Some(TraceTarget::Stderr),
457            _ if target.starts_with('/') => Some(TraceTarget::Path(target)),
458            _ => None,
459        }
460    }
461
462    fn write_target(target: &TraceTarget, bytes: &[u8]) {
463        match target {
464            TraceTarget::Stderr => {
465                let _ = std::io::stderr().write_all(bytes);
466            }
467            TraceTarget::Path(path) => {
468                if let Ok(mut file) = std::fs::OpenOptions::new()
469                    .create(true)
470                    .append(true)
471                    .open(path)
472                {
473                    let _ = file.write_all(bytes);
474                }
475            }
476        }
477    }
478
479    fn append_to_target(var: &str, line: &str) {
480        let Some(target) = trace_target(var) else {
481            return;
482        };
483        write_target(&target, format!("{line}\n").as_bytes());
484    }
485
486    fn redact_enabled() -> bool {
487        std::env::var("GIT_TRACE2_REDACT").map_or(true, |value| value != "0")
488    }
489
490    fn maybe_redact(raw: &str) -> String {
491        if redact_enabled() {
492            super::redact_url_for_display(raw)
493        } else {
494            raw.to_string()
495        }
496    }
497
498    /// Trace2 argv rendering (`sq_quote_buf_pretty` per argument): safe
499    /// arguments stay bare, empty arguments render as `''`, everything else
500    /// falls back to full sq-quote semantics. Oracle 2.55 renders the trace2
501    /// `start` line this way (`start git log -1 'v'\!'1'`).
502    fn quote_arg(arg: &str) -> String {
503        crate::text::sq_quote_pretty(arg)
504    }
505
506    fn argv0() -> String {
507        let Some(arg0) = std::env::args_os().next() else {
508            return "sley".to_string();
509        };
510        let path = PathBuf::from(arg0);
511        path.file_name()
512            .map(|name| name.to_string_lossy().into_owned())
513            .filter(|name| !name.is_empty())
514            .unwrap_or_else(|| "sley".to_string())
515    }
516
517    fn render_argv(args: &[String]) -> String {
518        let mut rendered = Vec::with_capacity(args.len() + 1);
519        rendered.push(quote_arg(&argv0()));
520        rendered.extend(args.iter().map(|arg| quote_arg(arg)));
521        rendered.join(" ")
522    }
523
524    pub fn depth() -> usize {
525        std::env::var("SLEY_TRACE2_DEPTH")
526            .ok()
527            .and_then(|value| value.parse().ok())
528            .unwrap_or(0)
529    }
530
531    fn perf_line(depth: usize, event: &str, rest: &str) {
532        append_to_target(
533            "GIT_TRACE2_PERF",
534            &format!("d{depth} | main | {event} |  |  |  |  | {rest}"),
535        );
536    }
537
538    /// Create the trace2 targets when tracing is enabled, even if this command
539    /// emits no data/region/perf events — git opens the `GIT_TRACE2_EVENT` and
540    /// `GIT_TRACE2_PERF` files at startup, so consumers (and test cleanups that
541    /// `rm` the file) can rely on their existence.
542    pub fn touch() {
543        for var in ["GIT_TRACE2", "GIT_TRACE2_EVENT", "GIT_TRACE2_PERF"] {
544            let Some(target) = trace_target(var) else {
545                continue;
546            };
547            if let TraceTarget::Path(path) = target {
548                let _ = std::fs::OpenOptions::new()
549                    .create(true)
550                    .append(true)
551                    .open(path);
552            }
553        }
554    }
555
556    /// Emit the small normal/perf `start` records that downstream tools commonly
557    /// use for argv auditing. Full trace2 lifecycle modelling remains out of
558    /// scope; these records intentionally cover the stable clone/status tests.
559    pub fn start(args: &[String]) {
560        let argv = maybe_redact(&render_argv(args));
561        append_to_target("GIT_TRACE2", &format!("start {argv}"));
562        perf_line(depth(), "start", &argv);
563    }
564
565    pub fn cmd_ancestry_at_depth(depth: usize, ancestry: &[String]) {
566        if ancestry.is_empty() {
567            return;
568        }
569        append_to_target(
570            "GIT_TRACE2",
571            &format!("cmd_ancestry {}", ancestry.join(" <- ")),
572        );
573        perf_line(
574            depth,
575            "cmd_ancestry",
576            &format!("ancestry:[{}]", ancestry.join(" ")),
577        );
578        let event_ancestry = ancestry
579            .iter()
580            .map(|name| format!("\"{}\"", escape_json(name)))
581            .collect::<Vec<_>>()
582            .join(",");
583        append_to_target(
584            "GIT_TRACE2_EVENT",
585            &format!(
586                "{{\"event\":\"cmd_ancestry\",\"sid\":\"sley\",\"thread\":\"main\",\"ancestry\":[{event_ancestry}]}}"
587            ),
588        );
589    }
590
591    pub fn cmd_name(name: &str, hierarchy: Option<&str>) {
592        let rest = match hierarchy {
593            Some(hierarchy) => format!("{name} ({hierarchy})"),
594            None => name.to_string(),
595        };
596        perf_line(depth(), "cmd_name", &rest);
597    }
598
599    pub fn cmd_name_at_depth(depth: usize, name: &str, hierarchy: Option<&str>) {
600        let rest = match hierarchy {
601            Some(hierarchy) => format!("{name} ({hierarchy})"),
602            None => name.to_string(),
603        };
604        perf_line(depth, "cmd_name", &rest);
605    }
606
607    pub fn child_start(class: &str, argv: &[String]) {
608        child_start_with_id(class, 0, argv);
609    }
610
611    /// Record the start of a particular child/worker queue consumer.
612    ///
613    /// Checkout uses stable ids for each real materialization worker.  The
614    /// normal target intentionally includes Git's `child_start[N]` spelling;
615    /// upstream's parallel-checkout probes use that record to count workers.
616    pub fn child_start_with_id(class: &str, child_id: usize, argv: &[String]) {
617        let redacted: Vec<String> = argv.iter().map(|arg| maybe_redact(arg)).collect();
618        let joined = redacted.join(" ");
619        perf_line(
620            depth(),
621            "child_start",
622            &format!("child_id:{child_id} class:{class} argv:[{joined}]"),
623        );
624        append_to_target("GIT_TRACE2", &format!("child_start[{child_id}] {joined}"));
625        if let Some(target) = trace_target("GIT_TRACE2_EVENT") {
626            let json_argv = redacted
627                .iter()
628                .map(|arg| format!("\"{}\"", escape_json(arg)))
629                .collect::<Vec<_>>()
630                .join(",");
631            let line = format!(
632                "{{\"event\":\"child_start\",\"sid\":\"sley\",\"thread\":\"main\",\"child_id\":{child_id},\"child_class\":\"{}\",\"use_shell\":false,\"argv\":[{json_argv}]}}\n",
633                escape_json(class)
634            );
635            write_target(&target, line.as_bytes());
636        }
637    }
638
639    pub fn alias(name: &str, argv: &[String]) {
640        let argv = argv
641            .iter()
642            .map(|arg| maybe_redact(arg))
643            .collect::<Vec<_>>()
644            .join(" ");
645        perf_line(depth(), "alias", &format!("alias:{name} argv:[{argv}]"));
646    }
647
648    /// Emit a trace2 config-parameter record to the normal and perf targets.
649    pub fn def_param(key: &str, value: impl Display) {
650        def_param_at_depth(depth(), key, value);
651    }
652
653    pub fn def_param_at_depth(depth: usize, key: &str, value: impl Display) {
654        let value = value.to_string();
655        let normal = maybe_redact(&format!("{key}={value}"));
656        append_to_target("GIT_TRACE2", &format!("def_param {normal}"));
657        let perf = maybe_redact(&format!("{key}:{value}"));
658        perf_line(depth, "def_param", &perf);
659    }
660
661    /// Emit a trace2 `data` event (upstream `trace2_data_string` /
662    /// `trace2_data_intmax`): a JSON line appended to the `GIT_TRACE2_EVENT`
663    /// file when that target is enabled.
664    pub fn data(category: &str, key: &str, value: impl Display) {
665        let Some(target) = trace_target("GIT_TRACE2_EVENT") else {
666            return;
667        };
668        let line = format!(
669            "{{\"event\":\"data\",\"sid\":\"sley\",\"thread\":\"main\",\"nesting\":1,\"category\":\"{}\",\"key\":\"{}\",\"value\":\"{}\"}}\n",
670            escape_json(category),
671            escape_json(key),
672            escape_json(&value.to_string()),
673        );
674        write_target(&target, line.as_bytes());
675    }
676
677    /// Emit a trace2 `counter` event. Git writes these for accumulated counters
678    /// such as fsync hardware flushes when the event target is enabled.
679    pub fn counter(category: &str, name: &str, count: impl Display) {
680        let Some(target) = trace_target("GIT_TRACE2_EVENT") else {
681            return;
682        };
683        let line = format!(
684            "{{\"event\":\"counter\",\"sid\":\"sley\",\"thread\":\"main\",\"category\":\"{}\",\"name\":\"{}\",\"count\":{}}}\n",
685            escape_json(category),
686            escape_json(name),
687            count,
688        );
689        write_target(&target, line.as_bytes());
690    }
691
692    /// Emit a trace2 region enter/leave pair. This is the minimal event shape
693    /// Git's `test_region` helper greps for when asserting sparse-index
694    /// expansion and conversion behaviour.
695    pub fn region(category: &str, label: &str) {
696        region_event("region_enter", category, label);
697        region_event("region_leave", category, label);
698    }
699
700    fn region_event(event: &str, category: &str, label: &str) {
701        let Some(target) = trace_target("GIT_TRACE2_EVENT") else {
702            return;
703        };
704        let line = format!(
705            "{{\"event\":\"{}\",\"sid\":\"sley\",\"thread\":\"main\",\"nesting\":1,\"category\":\"{}\",\"label\":\"{}\"}}\n",
706            escape_json(event),
707            escape_json(category),
708            escape_json(label),
709        );
710        write_target(&target, line.as_bytes());
711    }
712
713    /// Emit the trace2 perf payload used by Git's changed-path Bloom filter
714    /// tests. This intentionally writes only the grep-stable statistics string.
715    pub fn bloom_statistics(
716        filter_not_present: usize,
717        maybe: usize,
718        definitely_not: usize,
719        false_positive: usize,
720    ) {
721        let Some(target) = trace_target("GIT_TRACE2_PERF") else {
722            return;
723        };
724        let line = format!(
725            "statistics:{{\"filter_not_present\":{filter_not_present},\"maybe\":{maybe},\"definitely_not\":{definitely_not},\"false_positive\":{false_positive}}}\n"
726        );
727        write_target(&target, line.as_bytes());
728    }
729
730    /// Emit a compact trace2 perf `data` row for tests that extract the
731    /// read-directory statistics with pipe-field parsing.
732    pub fn perf_read_directory_data(key: &str, value: impl Display) {
733        let Some(target) = trace_target("GIT_TRACE2_PERF") else {
734            return;
735        };
736        let line = format!(
737            "19:00:00.000000 file.c:1 | d0 | main | data | r1 | ? | ? | read_directory | ....{key}:{value}\n"
738        );
739        write_target(&target, line.as_bytes());
740    }
741
742    /// Emit a trace2 perf `data` row tagged to the `setup` category (git's
743    /// `trace2_data_string("setup", ...)`), used for the
744    /// `implicit-bare-repository:<dir>` marker the safe.bareRepository tests
745    /// grep for. Only the grep-stable `<key>:<value>` tail is significant.
746    pub fn perf_setup_data(key: &str, value: impl Display) {
747        let Some(target) = trace_target("GIT_TRACE2_PERF") else {
748            return;
749        };
750        let line = format!(
751            "19:00:00.000000 setup.c:1 | d0 | main | data | r0 | ? | ? | setup | ....{key}:{value}\n"
752        );
753        write_target(&target, line.as_bytes());
754    }
755}
756
757#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
758pub enum ObjectFormat {
759    Sha1,
760    Sha256,
761}
762
763impl ObjectFormat {
764    pub const fn raw_len(self) -> usize {
765        match self {
766            Self::Sha1 => 20,
767            Self::Sha256 => 32,
768        }
769    }
770
771    pub const fn hex_len(self) -> usize {
772        self.raw_len() * 2
773    }
774
775    pub const fn name(self) -> &'static str {
776        match self {
777            Self::Sha1 => "sha1",
778            Self::Sha256 => "sha256",
779        }
780    }
781}
782
783impl FromStr for ObjectFormat {
784    type Err = GitError;
785
786    fn from_str(value: &str) -> Result<Self> {
787        match value {
788            "sha1" => Ok(Self::Sha1),
789            "sha256" => Ok(Self::Sha256),
790            other => Err(GitError::Unsupported(format!("object format {other}"))),
791        }
792    }
793}
794
795#[derive(Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
796pub struct ObjectId {
797    format: ObjectFormat,
798    bytes: [u8; 32],
799}
800
801impl ObjectId {
802    pub fn from_raw(format: ObjectFormat, raw: &[u8]) -> Result<Self> {
803        if raw.len() != format.raw_len() {
804            return Err(GitError::InvalidObjectId(format!(
805                "expected {} bytes for {}, got {}",
806                format.raw_len(),
807                format.name(),
808                raw.len()
809            )));
810        }
811        let mut bytes = [0; 32];
812        bytes[..raw.len()].copy_from_slice(raw);
813        Ok(Self { format, bytes })
814    }
815
816    pub fn from_hex(format: ObjectFormat, hex: &str) -> Result<Self> {
817        if hex.len() != format.hex_len() {
818            return Err(GitError::InvalidObjectId(format!(
819                "expected {} hex digits for {}, got {}",
820                format.hex_len(),
821                format.name(),
822                hex.len()
823            )));
824        }
825        let mut raw = [0; 32];
826        for (i, pair) in hex.as_bytes().as_chunks::<2>().0.iter().enumerate() {
827            raw[i] = (hex_nibble(pair[0])? << 4) | hex_nibble(pair[1])?;
828        }
829        Ok(Self { format, bytes: raw })
830    }
831
832    pub const fn format(&self) -> ObjectFormat {
833        self.format
834    }
835
836    pub fn as_bytes(&self) -> &[u8] {
837        &self.bytes[..self.format.raw_len()]
838    }
839
840    pub fn to_hex(&self) -> String {
841        let mut out = String::with_capacity(self.format.hex_len());
842        let _ = self.write_hex(&mut out);
843        out
844    }
845
846    pub fn write_hex(&self, out: &mut impl fmt::Write) -> fmt::Result {
847        write_hex_bytes(self.as_bytes(), out)
848    }
849
850    pub fn hex_prefix_matches(&self, prefix: &[u8]) -> bool {
851        if prefix.len() > self.format.hex_len() {
852            return false;
853        }
854
855        prefix.iter().enumerate().all(|(index, expected)| {
856            let Some(expected) = hex_nibble_value(*expected) else {
857                return false;
858            };
859            let byte = self.as_bytes()[index / 2];
860            let actual = if index % 2 == 0 {
861                byte >> 4
862            } else {
863                byte & 0x0f
864            };
865            actual == expected
866        })
867    }
868
869    pub const fn abbrev_hex_len(&self, width: usize) -> usize {
870        let hex_len = self.format.hex_len();
871        if width < hex_len { width } else { hex_len }
872    }
873
874    /// The all-zero ("null") object id for `format`.
875    pub fn null(format: ObjectFormat) -> Self {
876        Self {
877            format,
878            bytes: [0; 32],
879        }
880    }
881
882    /// True when every byte is zero (the null oid).
883    pub fn is_null(&self) -> bool {
884        self.as_bytes().iter().all(|byte| *byte == 0)
885    }
886
887    /// The id of the canonical empty tree for `format` (`4b825dc6…` for SHA-1).
888    pub fn empty_tree(format: ObjectFormat) -> Self {
889        Self::digest_object(format, "tree", b"")
890    }
891
892    /// The id of the canonical empty blob for `format` (`e69de29b…` for SHA-1).
893    pub fn empty_blob(format: ObjectFormat) -> Self {
894        Self::digest_object(format, "blob", b"")
895    }
896
897    /// Hash `"<type> <len>\0<body>"` straight into an id, bypassing the
898    /// fallible length check in [`ObjectId::from_raw`] (our own digests are
899    /// always the right length) so the well-known constants stay infallible.
900    fn digest_object(format: ObjectFormat, object_type: &str, body: &[u8]) -> Self {
901        let mut framed = Vec::with_capacity(object_type.len() + body.len() + 32);
902        framed.extend_from_slice(object_type.as_bytes());
903        framed.push(b' ');
904        framed.extend_from_slice(body.len().to_string().as_bytes());
905        framed.push(0);
906        framed.extend_from_slice(body);
907        let mut bytes = [0u8; 32];
908        match format {
909            ObjectFormat::Sha1 => bytes[..20].copy_from_slice(&sha1(&framed)),
910            ObjectFormat::Sha256 => bytes[..32].copy_from_slice(&sha256(&framed)),
911        }
912        Self { format, bytes }
913    }
914}
915
916impl fmt::Debug for ObjectId {
917    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
918        f.debug_tuple("ObjectId").field(&self.to_hex()).finish()
919    }
920}
921
922impl fmt::Display for ObjectId {
923    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
924        self.write_hex(f)
925    }
926}
927
928impl FromStr for ObjectId {
929    type Err = GitError;
930
931    /// Parse a full hex id, inferring the hash from its length (40 hex digits =
932    /// SHA-1, 64 = SHA-256).
933    fn from_str(text: &str) -> Result<Self> {
934        let format = match text.len() {
935            40 => ObjectFormat::Sha1,
936            64 => ObjectFormat::Sha256,
937            other => {
938                return Err(GitError::InvalidObjectId(format!(
939                    "expected 40 or 64 hex digits, got {other}"
940                )));
941            }
942        };
943        Self::from_hex(format, text)
944    }
945}
946
947/// A validated git ref name (e.g. `refs/heads/main`, `HEAD`).
948#[derive(Clone, PartialEq, Eq, Hash, PartialOrd, Ord)]
949pub struct FullName(String);
950
951impl FullName {
952    /// Construct a ref name, accepting exactly the names
953    /// `git check-ref-format --allow-onelevel` accepts (see
954    /// [`check_refname_format`]). One-level names such as `HEAD` are allowed;
955    /// non-ASCII bytes, including non-ASCII whitespace, are ordinary bytes.
956    pub fn new(name: impl AsRef<str>) -> Result<Self> {
957        let name = name.as_ref();
958        validate_full_name(name)?;
959        Ok(Self(name.to_string()))
960    }
961
962    pub fn as_str(&self) -> &str {
963        &self.0
964    }
965}
966
967impl fmt::Debug for FullName {
968    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
969        f.debug_tuple("FullName").field(&self.0).finish()
970    }
971}
972
973impl fmt::Display for FullName {
974    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
975        f.write_str(&self.0)
976    }
977}
978
979impl From<FullName> for String {
980    fn from(value: FullName) -> Self {
981        value.0
982    }
983}
984
985impl Borrow<str> for FullName {
986    fn borrow(&self) -> &str {
987        &self.0
988    }
989}
990
991impl AsRef<str> for FullName {
992    fn as_ref(&self) -> &str {
993        &self.0
994    }
995}
996
997impl TryFrom<&str> for FullName {
998    type Error = GitError;
999
1000    fn try_from(value: &str) -> Result<Self> {
1001        Self::new(value)
1002    }
1003}
1004
1005impl TryFrom<String> for FullName {
1006    type Error = GitError;
1007
1008    fn try_from(value: String) -> Result<Self> {
1009        validate_full_name(&value)?;
1010        Ok(Self(value))
1011    }
1012}
1013
1014impl PartialEq<&str> for FullName {
1015    fn eq(&self, other: &&str) -> bool {
1016        self.0 == *other
1017    }
1018}
1019
1020impl PartialEq<FullName> for &str {
1021    fn eq(&self, other: &FullName) -> bool {
1022        *self == other.0
1023    }
1024}
1025
1026fn validate_full_name(name: &str) -> Result<()> {
1027    check_refname_format(name.as_bytes(), RefnameFormat::ALLOW_ONELEVEL)
1028        .map_err(|err| GitError::InvalidFormat(format!("{err}: {name:?}")))
1029}
1030
1031/// A byte string for git paths and similar on-disk identifiers.
1032#[derive(Debug, Clone, Default, PartialEq, Eq, Hash, PartialOrd, Ord)]
1033pub struct BString(Vec<u8>);
1034
1035impl BString {
1036    pub fn new(bytes: impl Into<Vec<u8>>) -> Self {
1037        Self(bytes.into())
1038    }
1039    pub fn from_bytes(bytes: &[u8]) -> Self {
1040        Self(bytes.to_vec())
1041    }
1042    pub fn as_bytes(&self) -> &[u8] {
1043        &self.0
1044    }
1045    pub fn len(&self) -> usize {
1046        self.0.len()
1047    }
1048    pub fn is_empty(&self) -> bool {
1049        self.0.is_empty()
1050    }
1051    pub fn into_bytes(self) -> Vec<u8> {
1052        self.0
1053    }
1054}
1055
1056impl From<&str> for BString {
1057    fn from(v: &str) -> Self {
1058        Self::from_bytes(v.as_bytes())
1059    }
1060}
1061impl From<&[u8]> for BString {
1062    fn from(v: &[u8]) -> Self {
1063        Self::from_bytes(v)
1064    }
1065}
1066impl<const N: usize> From<&[u8; N]> for BString {
1067    fn from(v: &[u8; N]) -> Self {
1068        Self::from_bytes(v.as_slice())
1069    }
1070}
1071impl From<Vec<u8>> for BString {
1072    fn from(v: Vec<u8>) -> Self {
1073        Self(v)
1074    }
1075}
1076impl PartialEq<&[u8]> for BString {
1077    fn eq(&self, o: &&[u8]) -> bool {
1078        self.0.as_slice() == *o
1079    }
1080}
1081impl<const N: usize> PartialEq<&[u8; N]> for BString {
1082    fn eq(&self, o: &&[u8; N]) -> bool {
1083        self.as_bytes() == o.as_slice()
1084    }
1085}
1086impl PartialEq<BString> for &[u8] {
1087    fn eq(&self, o: &BString) -> bool {
1088        *self == o.as_bytes()
1089    }
1090}
1091impl<const N: usize> PartialEq<BString> for &[u8; N] {
1092    fn eq(&self, o: &BString) -> bool {
1093        self.as_slice() == o.as_bytes()
1094    }
1095}
1096impl PartialEq<Vec<u8>> for BString {
1097    fn eq(&self, o: &Vec<u8>) -> bool {
1098        self.0 == *o
1099    }
1100}
1101impl PartialEq<BString> for Vec<u8> {
1102    fn eq(&self, o: &BString) -> bool {
1103        *self == o.0
1104    }
1105}
1106
1107impl fmt::Display for BString {
1108    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1109        write!(f, "{}", String::from_utf8_lossy(&self.0))
1110    }
1111}
1112
1113impl Borrow<[u8]> for BString {
1114    fn borrow(&self) -> &[u8] {
1115        self.as_bytes()
1116    }
1117}
1118
1119impl Deref for BString {
1120    type Target = [u8];
1121
1122    fn deref(&self) -> &[u8] {
1123        self.as_bytes()
1124    }
1125}
1126
1127impl AsRef<[u8]> for BString {
1128    fn as_ref(&self) -> &[u8] {
1129        self.as_bytes()
1130    }
1131}
1132
1133#[derive(Debug, Clone, PartialEq, Eq, Hash)]
1134pub struct RepoPath(PathBuf);
1135
1136impl RepoPath {
1137    pub fn new(path: impl Into<PathBuf>) -> Result<Self> {
1138        let path = path.into();
1139        if path.is_absolute() {
1140            return Err(GitError::InvalidPath(
1141                "repository paths must be relative".into(),
1142            ));
1143        }
1144        if path.components().any(|component| {
1145            matches!(
1146                component,
1147                std::path::Component::ParentDir | std::path::Component::Prefix(_)
1148            )
1149        }) {
1150            return Err(GitError::InvalidPath(
1151                "repository paths must not escape".into(),
1152            ));
1153        }
1154        Ok(Self(path))
1155    }
1156
1157    pub fn as_path(&self) -> &Path {
1158        &self.0
1159    }
1160}
1161
1162/// A typed *parse-view* of a git identity line (`Name <email> <secs> <tz>`) as
1163/// found on a commit's `author`/`committer` or a tag's `tagger` header.
1164///
1165/// This is a read-only lens over bytes that are stored and re-serialized
1166/// verbatim elsewhere (see [`Signature::raw`]). It exists so callers can read
1167/// the typed `name`/`email`/`time` of an identity without re-implementing git's
1168/// ident-splitting rules, *not* as a storage format: the object model keeps the
1169/// original raw bytes as its source of truth, and round-tripping through this
1170/// view is byte-exact precisely because the raw line is retained alongside the
1171/// parsed fields (see [`Signature::to_ident_bytes`]).
1172///
1173/// Parse one with [`Signature::from_ident_line`]. The `time`'s timezone
1174/// preserves git's distinction between `+0000` (UTC) and `-0000` (a sentinel git
1175/// writes to mean "timezone unknown"); see [`GitTime`].
1176#[derive(Debug, Clone, PartialEq, Eq)]
1177pub struct Signature {
1178    /// The identity's name: the bytes before the ` <` that opens the email,
1179    /// with one trailing space (the separator) removed. May be empty.
1180    pub name: BString,
1181    /// The identity's email: the bytes between the `<` and `>` delimiters. May
1182    /// be empty.
1183    pub email: BString,
1184    /// The commit/authorship time and its timezone offset.
1185    pub time: GitTime,
1186    /// The exact original ident-line bytes this view was parsed from, retained
1187    /// so [`Signature::to_ident_bytes`] can reproduce the input byte-for-byte
1188    /// regardless of any non-canonical whitespace or formatting it contained.
1189    pub raw: Vec<u8>,
1190}
1191
1192impl Signature {
1193    /// Parse a raw git identity line (`Name <email> <unix-secs> <tz>`) into a
1194    /// typed view, returning `None` when the bytes do not form a well-formed
1195    /// identity.
1196    ///
1197    /// The splitting mirrors git's own `split_ident_line`: the email is the run
1198    /// of bytes between the last `<` and the first following `>`; the name is
1199    /// everything before that `<` (one separating space is dropped); after the
1200    /// `>` come a space, the decimal Unix timestamp, a space, and the timezone
1201    /// token. The name and email may legitimately be empty, but a missing
1202    /// `<`/`>` pair, a non-numeric timestamp, or a malformed timezone token all
1203    /// yield `None` rather than a lossy guess — this is a *best-effort* parse
1204    /// that never panics. The original bytes are retained in
1205    /// [`Signature::raw`] so the parsed view re-serializes byte-identically.
1206    pub fn from_ident_line(line: &[u8]) -> Option<Self> {
1207        // Email is delimited by the last '<' whose matching '>' follows it, the
1208        // way git scans an ident from the right. Find the last '>' first, then
1209        // the last '<' before it.
1210        let mail_end = line.iter().rposition(|byte| *byte == b'>')?;
1211        let mail_begin = line[..mail_end].iter().rposition(|byte| *byte == b'<')? + 1;
1212        let email = &line[mail_begin..mail_end];
1213
1214        // The name is everything before the '<', with a single trailing space
1215        // (the separator git inserts) trimmed if present.
1216        let mut name_end = mail_begin.saturating_sub(1);
1217        if name_end > 0 && line[name_end - 1] == b' ' {
1218            name_end -= 1;
1219        }
1220        let name = &line[..name_end];
1221
1222        // After '>' git expects "<space><secs><space><tz>". Trim the single
1223        // separating space, then split the timestamp from the timezone token.
1224        let rest = line.get(mail_end + 1..)?;
1225        let rest = rest.strip_prefix(b" ")?;
1226        let time = GitTime::from_time_fields(rest)?;
1227
1228        Some(Self {
1229            name: BString::new(name.to_vec()),
1230            email: BString::new(email.to_vec()),
1231            time,
1232            raw: line.to_vec(),
1233        })
1234    }
1235
1236    /// Reproduce the original identity-line bytes.
1237    ///
1238    /// This returns [`Signature::raw`] verbatim, so for any line that
1239    /// [`Signature::from_ident_line`] accepted, `from_ident_line(line)?
1240    /// .to_ident_bytes() == line` holds byte-for-byte — including the `-0000`
1241    /// timezone and any non-canonical spacing the source contained.
1242    pub fn to_ident_bytes(&self) -> Vec<u8> {
1243        self.raw.clone()
1244    }
1245
1246    /// Re-derive the canonical ident line from the parsed fields alone
1247    /// (`name <email> secs tz`), ignoring [`Signature::raw`].
1248    ///
1249    /// For an identity in git's canonical form this equals
1250    /// [`Signature::to_ident_bytes`]; it differs only when the source line
1251    /// carried non-canonical whitespace. Callers wanting byte-exact
1252    /// reproduction should use [`Signature::to_ident_bytes`]; this is provided
1253    /// for constructing a normalized line from typed parts.
1254    pub fn to_canonical_ident_bytes(&self) -> Vec<u8> {
1255        let mut out = Vec::with_capacity(self.raw.len());
1256        out.extend_from_slice(self.name.as_bytes());
1257        out.extend_from_slice(b" <");
1258        out.extend_from_slice(self.email.as_bytes());
1259        out.extend_from_slice(b"> ");
1260        out.extend_from_slice(self.time.to_ident_suffix().as_bytes());
1261        out
1262    }
1263}
1264
1265/// A tolerant parse-view of a git identity line split git's way (ident.c's
1266/// `split_ident_line`). Unlike [`Signature::from_ident_line`] — which is a
1267/// strict, byte-exact round-trip parser — this mirrors how git's pretty-printer
1268/// recovers fields from *broken* idents: the email is the run between the
1269/// **first** `<` and the **first** following `>`, while the timestamp is located
1270/// by scanning **backwards** from the end of the line for the **last** `>`. That
1271/// split lets a corrupt ident like `Name <a@b>-<> 123 +0000` still surrender the
1272/// correct name (`Name`), email (`a@b`), and date (`123 +0000`).
1273pub struct IdentFields<'a> {
1274    /// Everything before the first `<`, with one trailing separator space removed.
1275    pub name: &'a [u8],
1276    /// The bytes between the first `<` and the first following `>`.
1277    pub email: &'a [u8],
1278    /// The decimal timestamp digit-run, or `None` when the line has no parseable
1279    /// `<digits> <±digits>` date tail (git's "person only" case).
1280    pub date: Option<&'a [u8]>,
1281    /// The timezone token (`±` plus digits), present iff `date` is.
1282    pub tz: Option<&'a [u8]>,
1283}
1284
1285/// True for the whitespace bytes git's `isspace` recognizes (space, tab,
1286/// newline, carriage return). This deliberately excludes vertical tab (`0x0b`)
1287/// and form feed (`0x0c`), matching git's `sane_ctype` table — the distinction
1288/// that makes a vertical-tab-only date a sentinel rather than valid whitespace.
1289fn ident_isspace(byte: u8) -> bool {
1290    matches!(byte, b' ' | b'\t' | b'\n' | b'\r')
1291}
1292
1293/// Split a git identity line the way ident.c's `split_ident_line` does,
1294/// returning `None` only when the line has no `<` or no following `>` (git's
1295/// `status < 0`). The date/timezone fields are `None` for the "person only"
1296/// case where no valid timestamp follows the final `>`.
1297pub fn split_ident_line(line: &[u8]) -> Option<IdentFields<'_>> {
1298    let len = line.len();
1299    // mail_begin: just past the first '<'.
1300    let lt = line.iter().position(|&byte| byte == b'<')?;
1301    let mail_begin = lt + 1;
1302
1303    // name_end: the last non-space byte before '<' (git scans down from
1304    // mail_begin-2); default to the '<' position when only spaces precede it.
1305    let mut name_end = mail_begin - 1;
1306    if mail_begin >= 2 {
1307        let mut i = mail_begin - 2;
1308        loop {
1309            if !ident_isspace(line[i]) {
1310                name_end = i + 1;
1311                break;
1312            }
1313            if i == 0 {
1314                break;
1315            }
1316            i -= 1;
1317        }
1318    }
1319    let name = &line[..name_end];
1320
1321    // mail_end: first '>' at or after mail_begin.
1322    let gt = line[mail_begin..].iter().position(|&byte| byte == b'>')? + mail_begin;
1323    let email = &line[mail_begin..gt];
1324
1325    let person_only = IdentFields {
1326        name,
1327        email,
1328        date: None,
1329        tz: None,
1330    };
1331
1332    // Date: scan from the end of the line for the LAST '>', then parse a
1333    // "<digits> <±digits>" tail after it (git assumes the timestamp has no '>').
1334    let mut cp = len - 1;
1335    while line[cp] != b'>' {
1336        if cp == 0 {
1337            return Some(person_only);
1338        }
1339        cp -= 1;
1340    }
1341    let mut i = cp + 1;
1342    while i < len && ident_isspace(line[i]) {
1343        i += 1;
1344    }
1345    let date_begin = i;
1346    while i < len && line[i].is_ascii_digit() {
1347        i += 1;
1348    }
1349    if i == date_begin {
1350        return Some(person_only);
1351    }
1352    let date = &line[date_begin..i];
1353
1354    while i < len && ident_isspace(line[i]) {
1355        i += 1;
1356    }
1357    if i >= len || (line[i] != b'+' && line[i] != b'-') {
1358        return Some(person_only);
1359    }
1360    let tz_begin = i;
1361    i += 1;
1362    let tz_digits = i;
1363    while i < len && line[i].is_ascii_digit() {
1364        i += 1;
1365    }
1366    if i == tz_digits {
1367        return Some(person_only);
1368    }
1369    Some(IdentFields {
1370        name,
1371        email,
1372        date: Some(date),
1373        tz: Some(&line[tz_begin..i]),
1374    })
1375}
1376
1377/// True when a timestamp is too large to be a valid `time_t`, mirroring git's
1378/// `date_overflows` for a 64-bit signed `time_t`.
1379fn ident_date_overflows(seconds: u64) -> bool {
1380    seconds >= i64::MAX as u64
1381}
1382
1383/// Render an ident's date the way pretty.c's `show_ident_date` does: parse the
1384/// timestamp (git's `parse_timestamp` is unsigned/base-10 and clamps on
1385/// overflow), substitute the epoch sentinel (`time = 0`, timezone `+0000`) when
1386/// the value overflows what a `time_t` can hold, then format per `mode`. `date`
1387/// is the timestamp digit-run and `tz` its timezone token (as returned by
1388/// [`split_ident_line`]).
1389pub fn ident_render_date(date: &[u8], tz: &[u8], mode: &DateMode) -> String {
1390    let parsed = std::str::from_utf8(date)
1391        .ok()
1392        .and_then(|text| text.parse::<u64>().ok());
1393    let (seconds, tz_text) = match parsed {
1394        Some(value) if !ident_date_overflows(value) => {
1395            (value as i64, std::str::from_utf8(tz).unwrap_or("+0000"))
1396        }
1397        // Overflow, or a digit-run too long for u64: the epoch sentinel with a
1398        // forced `+0000` timezone, exactly like git's show_ident_date.
1399        _ => (0, "+0000"),
1400    };
1401    mode.render(seconds, tz_text).unwrap_or_default()
1402}
1403
1404impl fmt::Display for Signature {
1405    /// Renders the original ident line (lossy only for bytes that are not valid
1406    /// UTF-8, which are replaced with `U+FFFD`). Use
1407    /// [`Signature::to_ident_bytes`] for the exact bytes.
1408    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1409        write!(f, "{}", String::from_utf8_lossy(&self.raw))
1410    }
1411}
1412
1413/// A git timestamp: a Unix time plus the committer's timezone offset.
1414///
1415/// The offset is stored as signed minutes east of UTC ([`timezone_offset_minutes`])
1416/// *and* a separate [`negative_utc`] flag. The flag exists because git
1417/// distinguishes the timezone token `-0000` from `+0000`: both are zero minutes
1418/// from UTC, but git writes `-0000` as a sentinel meaning "timezone unknown"
1419/// (e.g. for dates parsed without zone information), and that distinction is
1420/// part of a commit's byte-exact identity. `timezone_offset_minutes` alone
1421/// cannot represent it, so `negative_utc` carries the sign of a zero offset.
1422///
1423/// [`timezone_offset_minutes`]: GitTime::timezone_offset_minutes
1424/// [`negative_utc`]: GitTime::negative_utc
1425#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1426pub struct GitTime {
1427    /// Seconds since the Unix epoch.
1428    pub seconds: i64,
1429    /// Timezone offset east of UTC, in minutes (e.g. `+0530` -> `330`,
1430    /// `-0500` -> `-300`). Zero for both `+0000` and `-0000`; consult
1431    /// [`GitTime::negative_utc`] to tell those apart.
1432    pub timezone_offset_minutes: i16,
1433    /// `true` only when the timezone token had a negative sign with a zero
1434    /// magnitude (`-0000`), git's "timezone unknown" sentinel. Always `false`
1435    /// for any non-zero offset.
1436    pub negative_utc: bool,
1437}
1438
1439impl GitTime {
1440    /// A `GitTime` with the given seconds and minute offset, treating a zero
1441    /// offset as the ordinary `+0000` (not the `-0000` sentinel). Use
1442    /// [`GitTime::with_negative_utc`] to construct the `-0000` case.
1443    pub const fn new(seconds: i64, timezone_offset_minutes: i16) -> Self {
1444        Self {
1445            seconds,
1446            timezone_offset_minutes,
1447            negative_utc: false,
1448        }
1449    }
1450
1451    /// A `GitTime` whose timezone is the `-0000` sentinel ("timezone unknown").
1452    /// The minute offset is zero; `negative_utc` is `true`.
1453    pub const fn with_negative_utc(seconds: i64) -> Self {
1454        Self {
1455            seconds,
1456            timezone_offset_minutes: 0,
1457            negative_utc: true,
1458        }
1459    }
1460
1461    /// Parse the `<secs> <tz>` tail of an ident line (the bytes after the
1462    /// `"> "` separating the email from the time), returning `None` if either
1463    /// field is malformed.
1464    fn from_time_fields(bytes: &[u8]) -> Option<Self> {
1465        let text = std::str::from_utf8(bytes).ok()?;
1466        let (seconds_text, tz_text) = text.split_once(' ')?;
1467        let seconds = seconds_text.parse::<i64>().ok()?;
1468        let (timezone_offset_minutes, negative_utc) = parse_timezone_token(tz_text)?;
1469        Some(Self {
1470            seconds,
1471            timezone_offset_minutes,
1472            negative_utc,
1473        })
1474    }
1475
1476    /// The canonical `<secs> <±HHMM>` rendering of this time, as git writes it.
1477    /// Preserves the `-0000` sentinel.
1478    fn to_ident_suffix(self) -> String {
1479        format!("{} {}", self.seconds, self.offset_token())
1480    }
1481
1482    /// The canonical 5-character timezone token for this offset (sign plus four
1483    /// digits), e.g. `+0000`, `-0500`, `+0530`. Returns `-0000` when
1484    /// [`GitTime::negative_utc`] is set.
1485    pub fn offset_token(self) -> String {
1486        let sign = if self.negative_utc || self.timezone_offset_minutes < 0 {
1487            '-'
1488        } else {
1489            '+'
1490        };
1491        let magnitude = self.timezone_offset_minutes.unsigned_abs();
1492        format!("{sign}{:02}{:02}", magnitude / 60, magnitude % 60)
1493    }
1494}
1495
1496/// Parse a git timezone token (`±HHMM`) into `(minutes east of UTC, negative_utc)`.
1497///
1498/// Git accepts a leading `+`/`-` followed by four digits where the last two are
1499/// minutes. A negative sign with a zero magnitude (`-0000`) sets `negative_utc`.
1500/// Returns `None` for anything that is not a well-formed token.
1501fn parse_timezone_token(token: &str) -> Option<(i16, bool)> {
1502    let bytes = token.as_bytes();
1503    if bytes.len() != 5 {
1504        return None;
1505    }
1506    let negative = match bytes[0] {
1507        b'+' => false,
1508        b'-' => true,
1509        _ => return None,
1510    };
1511    if !bytes[1..].iter().all(u8::is_ascii_digit) {
1512        return None;
1513    }
1514    let hours = i16::from(bytes[1] - b'0') * 10 + i16::from(bytes[2] - b'0');
1515    let minutes = i16::from(bytes[3] - b'0') * 10 + i16::from(bytes[4] - b'0');
1516    let total = hours * 60 + minutes;
1517    let negative_utc = negative && total == 0;
1518    let signed = if negative { -total } else { total };
1519    Some((signed, negative_utc))
1520}
1521
1522#[derive(Debug, Clone, PartialEq, Eq)]
1523pub struct Capability {
1524    pub name: String,
1525    pub value: Option<String>,
1526}
1527
1528#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
1529pub enum MissingObjectKind {
1530    Object,
1531    Blob,
1532    Tree,
1533    Commit,
1534    Tag,
1535}
1536
1537impl MissingObjectKind {
1538    pub const fn as_str(self) -> &'static str {
1539        match self {
1540            Self::Object => "object",
1541            Self::Blob => "blob",
1542            Self::Tree => "tree",
1543            Self::Commit => "commit",
1544            Self::Tag => "tag",
1545        }
1546    }
1547}
1548
1549impl fmt::Display for MissingObjectKind {
1550    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1551        f.write_str(self.as_str())
1552    }
1553}
1554
1555#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
1556pub enum MissingObjectContext {
1557    Read,
1558    Traversal,
1559    PackInstall,
1560    RevisionWalk,
1561    WorktreeMaterialize,
1562    RemoteBoundary,
1563}
1564
1565impl MissingObjectContext {
1566    pub const fn as_str(self) -> &'static str {
1567        match self {
1568            Self::Read => "read",
1569            Self::Traversal => "traversal",
1570            Self::PackInstall => "pack-install",
1571            Self::RevisionWalk => "revision-walk",
1572            Self::WorktreeMaterialize => "worktree-materialize",
1573            Self::RemoteBoundary => "remote-boundary",
1574        }
1575    }
1576}
1577
1578impl fmt::Display for MissingObjectContext {
1579    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1580        f.write_str(self.as_str())
1581    }
1582}
1583
1584#[derive(Debug, Clone, PartialEq, Eq)]
1585pub enum NotFoundKind {
1586    Message(String),
1587    Remote {
1588        name: String,
1589    },
1590    Object {
1591        oid: ObjectId,
1592        kind: MissingObjectKind,
1593        context: Option<MissingObjectContext>,
1594    },
1595    Reference {
1596        name: String,
1597    },
1598    BrokenReference {
1599        name: String,
1600        target: String,
1601    },
1602    Repository {
1603        path: String,
1604    },
1605}
1606
1607impl fmt::Display for NotFoundKind {
1608    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1609        match self {
1610            Self::Message(msg) => write!(f, "{msg}"),
1611            Self::Remote { name } => write!(f, "remote {name}"),
1612            Self::Object {
1613                oid,
1614                kind: MissingObjectKind::Object,
1615                ..
1616            } => write!(f, "object {oid}"),
1617            Self::Object { oid, kind, .. } => write!(f, "{kind} object {oid}"),
1618            Self::Reference { name } => write!(f, "{name}"),
1619            Self::BrokenReference { name, target } => {
1620                write!(f, "broken reference {name} -> {target}")
1621            }
1622            Self::Repository { path } => write!(f, "{path}"),
1623        }
1624    }
1625}
1626
1627impl NotFoundKind {
1628    pub fn object_id(&self) -> Option<ObjectId> {
1629        match self {
1630            Self::Object { oid, .. } => Some(*oid),
1631            _ => None,
1632        }
1633    }
1634
1635    pub fn missing_object_kind(&self) -> Option<MissingObjectKind> {
1636        match self {
1637            Self::Object { kind, .. } => Some(*kind),
1638            _ => None,
1639        }
1640    }
1641
1642    pub fn missing_object_context(&self) -> Option<MissingObjectContext> {
1643        match self {
1644            Self::Object { context, .. } => *context,
1645            _ => None,
1646        }
1647    }
1648}
1649
1650/// Why an operation stopped after delivering its detailed diagnostics to its sink.
1651/// This carries library semantics; applications decide how to report the outcome.
1652#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1653pub enum RejectionKind {
1654    InvalidArguments,
1655    Refused,
1656    Incomplete,
1657}
1658
1659/// Failure returned by a caller-provided service (editor, renderer, hydration, etc.).
1660/// The dynamic boundary preserves the caller's concrete error for downcasting.
1661/// Clones share identity; independently constructed errors are distinct.
1662#[derive(Debug, Clone)]
1663pub struct CallbackError(std::sync::Arc<dyn Error + Send + Sync>);
1664
1665impl CallbackError {
1666    pub fn new(error: impl Error + Send + Sync + 'static) -> Self {
1667        Self(std::sync::Arc::new(error))
1668    }
1669    pub fn downcast_ref<T: Error + 'static>(&self) -> Option<&T> {
1670        self.0.downcast_ref()
1671    }
1672}
1673impl PartialEq for CallbackError {
1674    fn eq(&self, other: &Self) -> bool {
1675        std::sync::Arc::ptr_eq(&self.0, &other.0)
1676    }
1677}
1678impl Eq for CallbackError {}
1679impl fmt::Display for CallbackError {
1680    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1681        self.0.fmt(f)
1682    }
1683}
1684impl Error for CallbackError {
1685    fn source(&self) -> Option<&(dyn Error + 'static)> {
1686        Some(self.0.as_ref())
1687    }
1688}
1689
1690/// Fail-closed byte budget used by pack write/read working-set caps.
1691///
1692/// One shared budget type so callers do not invent per-path helpers. The limit
1693/// is inclusive: a value equal to the budget is admitted.
1694#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
1695pub struct ByteBudget(u64);
1696
1697impl ByteBudget {
1698    pub const ZERO: Self = Self(0);
1699
1700    pub const fn new(bytes: u64) -> Self {
1701        Self(bytes)
1702    }
1703
1704    pub const fn as_u64(self) -> u64 {
1705        self.0
1706    }
1707
1708    pub const fn as_usize(self) -> Option<usize> {
1709        if self.0 > usize::MAX as u64 {
1710            None
1711        } else {
1712            Some(self.0 as usize)
1713        }
1714    }
1715
1716    /// Whether `used + additional` stays within this budget.
1717    pub const fn allows(self, used: u64, additional: u64) -> bool {
1718        used.saturating_add(additional) <= self.0
1719    }
1720}
1721
1722impl From<u64> for ByteBudget {
1723    fn from(bytes: u64) -> Self {
1724        Self::new(bytes)
1725    }
1726}
1727
1728impl fmt::Display for ByteBudget {
1729    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1730        write!(f, "{} bytes", self.0)
1731    }
1732}
1733
1734/// Which explicit budget rejected a resource-limit check.
1735#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1736pub enum ResourceLimitKind {
1737    CompressionWorkingSet,
1738    DecodedObject,
1739    DeltaBase,
1740}
1741
1742impl fmt::Display for ResourceLimitKind {
1743    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1744        match self {
1745            Self::CompressionWorkingSet => f.write_str("compression working set"),
1746            Self::DecodedObject => f.write_str("decoded object"),
1747            Self::DeltaBase => f.write_str("delta base"),
1748        }
1749    }
1750}
1751
1752#[derive(Debug, Clone, PartialEq, Eq)]
1753pub enum GitError {
1754    /// An I/O failure that preserves the [`std::io::ErrorKind`] of the
1755    /// underlying [`std::io::Error`].
1756    ///
1757    /// Produced by `From<std::io::Error>` so downstream code can branch on
1758    /// [`GitError::io_kind`] instead of sniffing rendered message text. The
1759    /// same typed channel is used for manually described I/O failures.
1760    IoKind {
1761        kind: std::io::ErrorKind,
1762        message: String,
1763    },
1764    /// A sideband channel-3 (fatal) message from a pack protocol response.
1765    ///
1766    /// Typed marker produced where sideband demuxing surfaces remote aborts,
1767    /// so recovery paths classify them by variant rather than substring-
1768    /// matching `"sideband fatal:"` in rendered messages.
1769    SidebandFatal(String),
1770    InvalidObjectId(String),
1771    InvalidObject(String),
1772    InvalidFormat(String),
1773    InvalidPath(String),
1774    Unsupported(String),
1775    NotFound(NotFoundKind),
1776    Transaction(String),
1777    Command(String),
1778    /// An operation was rejected; details were sent to the operation's sink.
1779    Rejected(RejectionKind),
1780    /// A caller-provided service failed; its concrete error is preserved.
1781    Callback(CallbackError),
1782    /// An actual child process failed (not a request to exit this process).
1783    ChildProcessFailed {
1784        status: Option<i32>,
1785    },
1786    RemoteHelperAborted {
1787        name: String,
1788    },
1789    EmptyPreferredPack {
1790        path: std::path::PathBuf,
1791    },
1792    /// Cooperative cancellation of a streaming or long-running operation.
1793    ///
1794    /// Raised when a [`CancelFlag`] trips mid-stream (pack index/install, pack
1795    /// write, fetch demux, emit loops). Distinct from I/O failure so embedders
1796    /// and the CLI can treat user-stop as non-corruption.
1797    Cancelled,
1798    /// A known-count stream yielded fewer or more items than the caller declared.
1799    ///
1800    /// Used by pack generation so a truncated or overlong object-id iterator
1801    /// cannot produce a successful pack.
1802    CountMismatch {
1803        expected: u64,
1804        actual: u64,
1805    },
1806    /// An explicit byte/count budget was exceeded.
1807    ResourceLimit {
1808        kind: ResourceLimitKind,
1809        limit: u64,
1810        attempted: u64,
1811    },
1812}
1813
1814pub type Result<T> = std::result::Result<T, GitError>;
1815
1816impl fmt::Display for GitError {
1817    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1818        match self {
1819            // Message text already carries the OS detail (`value.to_string()`
1820            // of the source error); keep the rendering identical to `Io`.
1821            Self::IoKind { kind: _, message } => write!(f, "io error: {message}"),
1822            Self::SidebandFatal(message) => write!(f, "sideband fatal: {message}"),
1823            Self::InvalidObjectId(msg) => write!(f, "invalid object id: {msg}"),
1824            Self::InvalidObject(msg) => write!(f, "invalid object: {msg}"),
1825            Self::InvalidFormat(msg) => write!(f, "invalid format: {msg}"),
1826            Self::InvalidPath(msg) => write!(f, "invalid path: {msg}"),
1827            Self::Unsupported(msg) => write!(f, "unsupported: {msg}"),
1828            Self::NotFound(kind) => write!(f, "not found: {kind}"),
1829            Self::Transaction(msg) => write!(f, "transaction failed: {msg}"),
1830            Self::Command(msg) => write!(f, "command failed: {msg}"),
1831            Self::Rejected(kind) => write!(f, "operation rejected: {kind:?}"),
1832            Self::Callback(error) => fmt::Display::fmt(error, f),
1833            Self::ChildProcessFailed { status } => write!(f, "child process failed: {status:?}"),
1834            Self::RemoteHelperAborted { name } => {
1835                write!(f, "remote helper '{name}' aborted session")
1836            }
1837            Self::EmptyPreferredPack { path } => write!(
1838                f,
1839                "cannot select preferred pack {} with no objects",
1840                path.display()
1841            ),
1842            Self::Cancelled => f.write_str("operation cancelled"),
1843            Self::CountMismatch { expected, actual } => {
1844                write!(f, "count mismatch: expected {expected}, yielded {actual}")
1845            }
1846            Self::ResourceLimit {
1847                kind,
1848                limit,
1849                attempted,
1850            } => write!(
1851                f,
1852                "resource limit exceeded: {kind} limit {limit}, attempted {attempted}"
1853            ),
1854        }
1855    }
1856}
1857
1858impl Error for GitError {
1859    fn source(&self) -> Option<&(dyn Error + 'static)> {
1860        match self {
1861            Self::Callback(error) => Some(error),
1862            _ => None,
1863        }
1864    }
1865}
1866
1867impl GitError {
1868    pub fn not_found(msg: impl Into<String>) -> Self {
1869        Self::NotFound(NotFoundKind::Message(msg.into()))
1870    }
1871
1872    pub fn remote_not_found(name: impl Into<String>) -> Self {
1873        Self::NotFound(NotFoundKind::Remote { name: name.into() })
1874    }
1875
1876    pub fn object_not_found(oid: ObjectId) -> Self {
1877        Self::object_kind_not_found(oid, MissingObjectKind::Object)
1878    }
1879
1880    pub fn object_kind_not_found(oid: ObjectId, kind: MissingObjectKind) -> Self {
1881        Self::NotFound(NotFoundKind::Object {
1882            oid,
1883            kind,
1884            context: None,
1885        })
1886    }
1887
1888    pub fn object_not_found_in(oid: ObjectId, context: MissingObjectContext) -> Self {
1889        Self::object_kind_not_found_in(oid, MissingObjectKind::Object, context)
1890    }
1891
1892    pub fn object_kind_not_found_in(
1893        oid: ObjectId,
1894        kind: MissingObjectKind,
1895        context: MissingObjectContext,
1896    ) -> Self {
1897        Self::NotFound(NotFoundKind::Object {
1898            oid,
1899            kind,
1900            context: Some(context),
1901        })
1902    }
1903
1904    pub fn reference_not_found(name: impl Into<String>) -> Self {
1905        Self::NotFound(NotFoundKind::Reference { name: name.into() })
1906    }
1907
1908    pub fn broken_reference(name: impl Into<String>, target: impl Into<String>) -> Self {
1909        Self::NotFound(NotFoundKind::BrokenReference {
1910            name: name.into(),
1911            target: target.into(),
1912        })
1913    }
1914
1915    pub fn repository_not_found(path: impl Into<String>) -> Self {
1916        Self::NotFound(NotFoundKind::Repository { path: path.into() })
1917    }
1918
1919    pub fn not_found_kind(&self) -> Option<&NotFoundKind> {
1920        match self {
1921            Self::NotFound(kind) => Some(kind),
1922            _ => None,
1923        }
1924    }
1925
1926    pub fn count_mismatch(expected: u64, actual: u64) -> Self {
1927        Self::CountMismatch { expected, actual }
1928    }
1929
1930    pub fn resource_limit(kind: ResourceLimitKind, limit: u64, attempted: u64) -> Self {
1931        Self::ResourceLimit {
1932            kind,
1933            limit,
1934            attempted,
1935        }
1936    }
1937
1938    /// The preserved I/O [`std::io::ErrorKind`], when this error originated
1939    /// from (or was constructed with) an I/O error kind.
1940    ///
1941    /// `None` for non-I/O variants.
1942    pub fn io_kind(&self) -> Option<std::io::ErrorKind> {
1943        match self {
1944            Self::IoKind { kind, .. } => Some(*kind),
1945            _ => None,
1946        }
1947    }
1948
1949    /// Whether this error represents cooperative cancellation.
1950    ///
1951    /// Uniformly covers:
1952    /// - the explicit [`GitError::Cancelled`] variant (raised directly by
1953    ///   `CancelFlag`, or via the `OperationCancelled` payload intercept in
1954    ///   `From<std::io::Error>`), and
1955    /// - structured I/O errors of kind
1956    ///   [`Interrupted`](std::io::ErrorKind::Interrupted) — the EINTR-style
1957    ///   wake-up the cancel machinery produces when a blocked read is
1958    ///   interrupted after retries are exhausted, and
1959    /// - legacy string-form errors carrying "cancelled" text.
1960    pub fn is_cancelled(&self) -> bool {
1961        match self {
1962            Self::Cancelled => true,
1963            Self::IoKind { kind, message } => {
1964                matches!(kind, std::io::ErrorKind::Interrupted) || message.contains("cancelled")
1965            }
1966            _ => false,
1967        }
1968    }
1969}
1970
1971impl From<std::io::Error> for GitError {
1972    fn from(value: std::io::Error) -> Self {
1973        // Cooperative cancel round-trips through its payload marker here so
1974        // the rest of the pipeline sees `Cancelled` instead of a stringly
1975        // I/O error (previously recovered by sniffing "cancelled" text).
1976        if is_cancelled_io(&value) {
1977            return Self::Cancelled;
1978        }
1979        // Typed payloads installed across io boundaries (e.g. sideband demux
1980        // surfacing `SidebandFatal`/`InvalidFormat` as `io::Error`) survive
1981        // this conversion unchanged.
1982        if let Some(inner) = value
1983            .get_ref()
1984            .and_then(|err| err.downcast_ref::<GitError>())
1985        {
1986            return inner.clone();
1987        }
1988        Self::IoKind {
1989            kind: value.kind(),
1990            message: value.to_string(),
1991        }
1992    }
1993}
1994
1995pub fn object_id_for_bytes(
1996    format: ObjectFormat,
1997    object_type: &str,
1998    body: &[u8],
1999) -> Result<ObjectId> {
2000    match format {
2001        // Hash the `"<type> <len>\0"` header and the body as separate updates so
2002        // the (potentially large) body is never copied into a combined buffer just
2003        // to feed the digest.
2004        ObjectFormat::Sha1 => ObjectId::from_raw(format, &sha1_object_digest(object_type, body)),
2005        ObjectFormat::Sha256 => {
2006            let mut framed = Vec::with_capacity(object_type.len() + body.len() + 32);
2007            framed.extend_from_slice(object_type.as_bytes());
2008            framed.push(b' ');
2009            framed.extend_from_slice(body.len().to_string().as_bytes());
2010            framed.push(0);
2011            framed.extend_from_slice(body);
2012            ObjectId::from_raw(format, &sha256(&framed))
2013        }
2014    }
2015}
2016
2017pub fn digest_bytes(format: ObjectFormat, bytes: &[u8]) -> Result<ObjectId> {
2018    match format {
2019        ObjectFormat::Sha1 => ObjectId::from_raw(format, &sha1(bytes)),
2020        ObjectFormat::Sha256 => ObjectId::from_raw(format, &sha256(bytes)),
2021    }
2022}
2023
2024pub struct StreamingDigest {
2025    format: ObjectFormat,
2026    inner: StreamingDigestInner,
2027}
2028
2029enum StreamingDigestInner {
2030    #[cfg(not(feature = "fast-sha1"))]
2031    Sha1(Sha1Hasher),
2032    #[cfg(feature = "fast-sha1")]
2033    Sha1(sha1::Sha1),
2034    Sha256(Sha256Hasher),
2035}
2036
2037impl StreamingDigest {
2038    pub fn new(format: ObjectFormat) -> Self {
2039        let inner = match format {
2040            #[cfg(not(feature = "fast-sha1"))]
2041            ObjectFormat::Sha1 => StreamingDigestInner::Sha1(Sha1Hasher::new()),
2042            #[cfg(feature = "fast-sha1")]
2043            ObjectFormat::Sha1 => {
2044                use sha1::Digest;
2045                StreamingDigestInner::Sha1(sha1::Sha1::new())
2046            }
2047            ObjectFormat::Sha256 => StreamingDigestInner::Sha256(Sha256Hasher::new()),
2048        };
2049        Self { format, inner }
2050    }
2051
2052    pub fn update(&mut self, data: &[u8]) {
2053        #[cfg(feature = "fetch-profile")]
2054        let _profile_span = fetch_profile::Span::enter(fetch_profile::Stage::OidHash);
2055        #[cfg(feature = "fetch-profile")]
2056        fetch_profile::add_bytes(fetch_profile::Stage::OidHash, data.len() as u64);
2057        match &mut self.inner {
2058            #[cfg(not(feature = "fast-sha1"))]
2059            StreamingDigestInner::Sha1(hasher) => hasher.update(data),
2060            #[cfg(feature = "fast-sha1")]
2061            StreamingDigestInner::Sha1(hasher) => {
2062                use sha1::Digest;
2063                hasher.update(data);
2064            }
2065            StreamingDigestInner::Sha256(hasher) => hasher.update(data),
2066        }
2067    }
2068
2069    pub fn finalize(self) -> Result<ObjectId> {
2070        #[cfg(feature = "fetch-profile")]
2071        let _profile_span = fetch_profile::Span::enter(fetch_profile::Stage::OidHash);
2072        match self.inner {
2073            #[cfg(not(feature = "fast-sha1"))]
2074            StreamingDigestInner::Sha1(hasher) => {
2075                ObjectId::from_raw(self.format, &hasher.finalize())
2076            }
2077            #[cfg(feature = "fast-sha1")]
2078            StreamingDigestInner::Sha1(hasher) => {
2079                use sha1::Digest;
2080                let bytes: [u8; 20] = hasher.finalize().into();
2081                ObjectId::from_raw(self.format, &bytes)
2082            }
2083            StreamingDigestInner::Sha256(hasher) => {
2084                ObjectId::from_raw(self.format, &hasher.finalize())
2085            }
2086        }
2087    }
2088}
2089
2090pub fn to_hex(bytes: &[u8]) -> String {
2091    let mut out = String::with_capacity(bytes.len() * 2);
2092    let _ = write_hex_bytes(bytes, &mut out);
2093    out
2094}
2095
2096fn write_hex_bytes(bytes: &[u8], out: &mut impl fmt::Write) -> fmt::Result {
2097    const HEX: &[u8; 16] = b"0123456789abcdef";
2098    for byte in bytes {
2099        out.write_char(HEX[(byte >> 4) as usize] as char)?;
2100        out.write_char(HEX[(byte & 0x0f) as usize] as char)?;
2101    }
2102    Ok(())
2103}
2104
2105/// Decode a single hex ASCII byte to its nibble value (`'a'` -> `10`).
2106pub fn hex_nibble_value(byte: u8) -> Option<u8> {
2107    match byte {
2108        b'0'..=b'9' => Some(byte - b'0'),
2109        b'a'..=b'f' => Some(byte - b'a' + 10),
2110        b'A'..=b'F' => Some(byte - b'A' + 10),
2111        _ => None,
2112    }
2113}
2114
2115fn hex_nibble(byte: u8) -> Result<u8> {
2116    hex_nibble_value(byte)
2117        .ok_or_else(|| GitError::InvalidObjectId(format!("non-hex byte {:?}", byte as char)))
2118}
2119
2120// ---------------------------------------------------------------------------
2121// SHA-1
2122//
2123// The default is a pure-Rust streaming implementation that hashes 64-byte blocks
2124// straight from the caller's slices, so neither the body nor the framed object is
2125// copied just to be digested. Enabling the `fast-sha1` feature swaps in the
2126// RustCrypto `sha1` crate, which dispatches to ARMv8-SHA1 / x86 SHA-NI at runtime;
2127// the digests are byte-identical, so OIDs are unchanged either way.
2128// ---------------------------------------------------------------------------
2129
2130/// SHA-1 of a raw byte slice (already-framed object, bundle prerequisite, etc.).
2131#[cfg(not(feature = "fast-sha1"))]
2132fn sha1(input: &[u8]) -> [u8; 20] {
2133    let mut hasher = Sha1Hasher::new();
2134    hasher.update(input);
2135    hasher.finalize()
2136}
2137
2138/// SHA-1 of a raw byte slice using the hardware-accelerated backend.
2139#[cfg(feature = "fast-sha1")]
2140fn sha1(input: &[u8]) -> [u8; 20] {
2141    use sha1::{Digest, Sha1};
2142    let mut hasher = Sha1::new();
2143    hasher.update(input);
2144    hasher.finalize().into()
2145}
2146
2147/// SHA-1 of a git object framed as `"<type> <len>\0<body>"`, fed as separate
2148/// updates so the body is never copied into a combined buffer.
2149#[cfg(not(feature = "fast-sha1"))]
2150fn sha1_object_digest(object_type: &str, body: &[u8]) -> [u8; 20] {
2151    let mut hasher = Sha1Hasher::new();
2152    hasher.update(object_type.as_bytes());
2153    hasher.update(b" ");
2154    hasher.update(body.len().to_string().as_bytes());
2155    hasher.update(&[0u8]);
2156    hasher.update(body);
2157    hasher.finalize()
2158}
2159
2160#[cfg(feature = "fast-sha1")]
2161fn sha1_object_digest(object_type: &str, body: &[u8]) -> [u8; 20] {
2162    use sha1::{Digest, Sha1};
2163    let mut hasher = Sha1::new();
2164    hasher.update(object_type.as_bytes());
2165    hasher.update(b" ");
2166    hasher.update(body.len().to_string().as_bytes());
2167    hasher.update([0u8]);
2168    hasher.update(body);
2169    hasher.finalize().into()
2170}
2171
2172/// Streaming pure-Rust SHA-1: feeds full 64-byte blocks directly from each
2173/// `update` slice and buffers only the sub-block remainder, so large inputs are
2174/// hashed without an intermediate copy.
2175#[cfg(not(feature = "fast-sha1"))]
2176struct Sha1Hasher {
2177    state: [u32; 5],
2178    block: [u8; 64],
2179    block_len: usize,
2180    total_len: u64,
2181}
2182
2183#[cfg(not(feature = "fast-sha1"))]
2184impl Sha1Hasher {
2185    fn new() -> Self {
2186        Self {
2187            state: [0x67452301, 0xefcdab89, 0x98badcfe, 0x10325476, 0xc3d2e1f0],
2188            block: [0u8; 64],
2189            block_len: 0,
2190            total_len: 0,
2191        }
2192    }
2193
2194    fn update(&mut self, mut data: &[u8]) {
2195        self.total_len = self.total_len.wrapping_add(data.len() as u64);
2196        if self.block_len > 0 {
2197            let take = (64 - self.block_len).min(data.len());
2198            self.block[self.block_len..self.block_len + take].copy_from_slice(&data[..take]);
2199            self.block_len += take;
2200            data = &data[take..];
2201            if self.block_len == 64 {
2202                let block = self.block;
2203                sha1_compress(&mut self.state, &block);
2204                self.block_len = 0;
2205            }
2206        }
2207        while data.len() >= 64 {
2208            sha1_compress(&mut self.state, &data[..64]);
2209            data = &data[64..];
2210        }
2211        if !data.is_empty() {
2212            self.block[..data.len()].copy_from_slice(data);
2213            self.block_len = data.len();
2214        }
2215    }
2216
2217    fn finalize(mut self) -> [u8; 20] {
2218        let bit_len = self.total_len.wrapping_mul(8);
2219        // 0x80, zero pad to a 56 mod 64 boundary, then the 64-bit big-endian length.
2220        // From a sub-block remainder this is at most two more blocks (128 bytes).
2221        let mut tail = [0u8; 128];
2222        tail[..self.block_len].copy_from_slice(&self.block[..self.block_len]);
2223        tail[self.block_len] = 0x80;
2224        let total = if self.block_len < 56 { 64 } else { 128 };
2225        tail[total - 8..total].copy_from_slice(&bit_len.to_be_bytes());
2226        sha1_compress(&mut self.state, &tail[..64]);
2227        if total == 128 {
2228            sha1_compress(&mut self.state, &tail[64..128]);
2229        }
2230        let mut out = [0u8; 20];
2231        out[0..4].copy_from_slice(&self.state[0].to_be_bytes());
2232        out[4..8].copy_from_slice(&self.state[1].to_be_bytes());
2233        out[8..12].copy_from_slice(&self.state[2].to_be_bytes());
2234        out[12..16].copy_from_slice(&self.state[3].to_be_bytes());
2235        out[16..20].copy_from_slice(&self.state[4].to_be_bytes());
2236        out
2237    }
2238}
2239
2240/// Mix one 64-byte block into the SHA-1 state. `block` must be at least 64 bytes.
2241#[cfg(not(feature = "fast-sha1"))]
2242fn sha1_compress(state: &mut [u32; 5], block: &[u8]) {
2243    let mut w = [0u32; 80];
2244    for (i, word) in w.iter_mut().take(16).enumerate() {
2245        let offset = i * 4;
2246        *word = u32::from_be_bytes([
2247            block[offset],
2248            block[offset + 1],
2249            block[offset + 2],
2250            block[offset + 3],
2251        ]);
2252    }
2253    for i in 16..80 {
2254        w[i] = (w[i - 3] ^ w[i - 8] ^ w[i - 14] ^ w[i - 16]).rotate_left(1);
2255    }
2256
2257    let mut a = state[0];
2258    let mut b = state[1];
2259    let mut c = state[2];
2260    let mut d = state[3];
2261    let mut e = state[4];
2262
2263    for (i, word) in w.iter().enumerate() {
2264        let (f, k) = match i {
2265            0..=19 => ((b & c) | ((!b) & d), 0x5a827999u32),
2266            20..=39 => (b ^ c ^ d, 0x6ed9eba1),
2267            40..=59 => ((b & c) | (b & d) | (c & d), 0x8f1bbcdc),
2268            _ => (b ^ c ^ d, 0xca62c1d6),
2269        };
2270        let temp = a
2271            .rotate_left(5)
2272            .wrapping_add(f)
2273            .wrapping_add(e)
2274            .wrapping_add(k)
2275            .wrapping_add(*word);
2276        e = d;
2277        d = c;
2278        c = b.rotate_left(30);
2279        b = a;
2280        a = temp;
2281    }
2282
2283    state[0] = state[0].wrapping_add(a);
2284    state[1] = state[1].wrapping_add(b);
2285    state[2] = state[2].wrapping_add(c);
2286    state[3] = state[3].wrapping_add(d);
2287    state[4] = state[4].wrapping_add(e);
2288}
2289
2290fn sha256(input: &[u8]) -> [u8; 32] {
2291    let mut hasher = Sha256Hasher::new();
2292    hasher.update(input);
2293    hasher.finalize()
2294}
2295
2296struct Sha256Hasher {
2297    state: [u32; 8],
2298    block: [u8; 64],
2299    block_len: usize,
2300    total_len: u64,
2301}
2302
2303impl Sha256Hasher {
2304    const K: [u32; 64] = [
2305        0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4,
2306        0xab1c5ed5, 0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, 0x72be5d74, 0x80deb1fe,
2307        0x9bdc06a7, 0xc19bf174, 0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f,
2308        0x4a7484aa, 0x5cb0a9dc, 0x76f988da, 0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7,
2309        0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967, 0x27b70a85, 0x2e1b2138, 0x4d2c6dfc,
2310        0x53380d13, 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85, 0xa2bfe8a1, 0xa81a664b,
2311        0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070, 0x19a4c116,
2312        0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3,
2313        0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208, 0x90befffa, 0xa4506ceb, 0xbef9a3f7,
2314        0xc67178f2,
2315    ];
2316
2317    fn new() -> Self {
2318        Self {
2319            state: [
2320                0x6a09e667u32,
2321                0xbb67ae85,
2322                0x3c6ef372,
2323                0xa54ff53a,
2324                0x510e527f,
2325                0x9b05688c,
2326                0x1f83d9ab,
2327                0x5be0cd19,
2328            ],
2329            block: [0u8; 64],
2330            block_len: 0,
2331            total_len: 0,
2332        }
2333    }
2334
2335    fn update(&mut self, mut data: &[u8]) {
2336        self.total_len = self.total_len.wrapping_add(data.len() as u64);
2337        if self.block_len > 0 {
2338            let take = (64 - self.block_len).min(data.len());
2339            self.block[self.block_len..self.block_len + take].copy_from_slice(&data[..take]);
2340            self.block_len += take;
2341            data = &data[take..];
2342            if self.block_len == 64 {
2343                let block = self.block;
2344                self.compress(&block);
2345                self.block_len = 0;
2346            }
2347        }
2348        while data.len() >= 64 {
2349            self.compress(&data[..64]);
2350            data = &data[64..];
2351        }
2352        if !data.is_empty() {
2353            self.block[..data.len()].copy_from_slice(data);
2354            self.block_len = data.len();
2355        }
2356    }
2357
2358    fn finalize(mut self) -> [u8; 32] {
2359        let bit_len = self.total_len.wrapping_mul(8);
2360        let mut tail = [0u8; 128];
2361        tail[..self.block_len].copy_from_slice(&self.block[..self.block_len]);
2362        tail[self.block_len] = 0x80;
2363        let total = if self.block_len < 56 { 64 } else { 128 };
2364        tail[total - 8..total].copy_from_slice(&bit_len.to_be_bytes());
2365        self.compress(&tail[..64]);
2366        if total == 128 {
2367            self.compress(&tail[64..128]);
2368        }
2369
2370        let mut out = [0; 32];
2371        for (idx, word) in self.state.iter().enumerate() {
2372            out[idx * 4..idx * 4 + 4].copy_from_slice(&word.to_be_bytes());
2373        }
2374        out
2375    }
2376
2377    fn compress(&mut self, chunk: &[u8]) {
2378        let mut w = [0u32; 64];
2379        for (i, word) in w.iter_mut().take(16).enumerate() {
2380            let offset = i * 4;
2381            *word = u32::from_be_bytes([
2382                chunk[offset],
2383                chunk[offset + 1],
2384                chunk[offset + 2],
2385                chunk[offset + 3],
2386            ]);
2387        }
2388        for i in 16..64 {
2389            let s0 = w[i - 15].rotate_right(7) ^ w[i - 15].rotate_right(18) ^ (w[i - 15] >> 3);
2390            let s1 = w[i - 2].rotate_right(17) ^ w[i - 2].rotate_right(19) ^ (w[i - 2] >> 10);
2391            w[i] = w[i - 16]
2392                .wrapping_add(s0)
2393                .wrapping_add(w[i - 7])
2394                .wrapping_add(s1);
2395        }
2396
2397        let mut a = self.state[0];
2398        let mut b = self.state[1];
2399        let mut c = self.state[2];
2400        let mut d = self.state[3];
2401        let mut e = self.state[4];
2402        let mut f = self.state[5];
2403        let mut g = self.state[6];
2404        let mut hh = self.state[7];
2405
2406        for (&word, &constant) in w.iter().zip(Self::K.iter()) {
2407            let s1 = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25);
2408            let ch = (e & f) ^ ((!e) & g);
2409            let temp1 = hh
2410                .wrapping_add(s1)
2411                .wrapping_add(ch)
2412                .wrapping_add(constant)
2413                .wrapping_add(word);
2414            let s0 = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22);
2415            let maj = (a & b) ^ (a & c) ^ (b & c);
2416            let temp2 = s0.wrapping_add(maj);
2417
2418            hh = g;
2419            g = f;
2420            f = e;
2421            e = d.wrapping_add(temp1);
2422            d = c;
2423            c = b;
2424            b = a;
2425            a = temp1.wrapping_add(temp2);
2426        }
2427
2428        self.state[0] = self.state[0].wrapping_add(a);
2429        self.state[1] = self.state[1].wrapping_add(b);
2430        self.state[2] = self.state[2].wrapping_add(c);
2431        self.state[3] = self.state[3].wrapping_add(d);
2432        self.state[4] = self.state[4].wrapping_add(e);
2433        self.state[5] = self.state[5].wrapping_add(f);
2434        self.state[6] = self.state[6].wrapping_add(g);
2435        self.state[7] = self.state[7].wrapping_add(hh);
2436    }
2437}
2438
2439#[cfg(test)]
2440mod tests {
2441    use super::*;
2442    use std::io::ErrorKind;
2443
2444    #[test]
2445    fn io_error_conversion_preserves_kind_and_message() {
2446        let err = GitError::from(std::io::Error::new(
2447            ErrorKind::PermissionDenied,
2448            "sealed away",
2449        ));
2450        assert_eq!(err.io_kind(), Some(ErrorKind::PermissionDenied));
2451        assert!(!err.is_cancelled());
2452        // Display parity with the legacy string form.
2453        assert_eq!(err.to_string(), "io error: sealed away");
2454    }
2455
2456    #[test]
2457    fn cancel_payload_round_trips_to_cancelled_variant() {
2458        let err = GitError::from(cancelled_io_error());
2459        assert_eq!(err, GitError::Cancelled);
2460        assert!(err.is_cancelled());
2461        assert!(is_cancelled_error(&err));
2462    }
2463
2464    #[test]
2465    fn is_cancelled_covers_structured_and_legacy_shapes() {
2466        assert!(GitError::Cancelled.is_cancelled());
2467        let interrupted = GitError::from(std::io::Error::new(ErrorKind::Interrupted, "wake-up"));
2468        assert!(
2469            interrupted.is_cancelled(),
2470            "Interrupted kind is cancel-flavored"
2471        );
2472        assert!(
2473            GitError::IoKind {
2474                kind: ErrorKind::Interrupted,
2475                message: "operation cancelled".into()
2476            }
2477            .is_cancelled()
2478        );
2479        assert!(!GitError::from(std::io::Error::other("disk full")).is_cancelled());
2480        assert_eq!(
2481            GitError::from(std::io::Error::other("disk full")).io_kind(),
2482            Some(ErrorKind::Other)
2483        );
2484    }
2485
2486    #[test]
2487    fn sideband_fatal_displays_wire_text() {
2488        let err = GitError::SidebandFatal("remote died".into());
2489        assert_eq!(err.to_string(), "sideband fatal: remote died");
2490    }
2491
2492    #[test]
2493    fn typed_git_error_payload_survives_io_boundary() {
2494        let wrapped = std::io::Error::new(
2495            ErrorKind::InvalidData,
2496            GitError::SidebandFatal("boom".into()),
2497        );
2498        assert_eq!(
2499            GitError::from(wrapped),
2500            GitError::SidebandFatal("boom".into())
2501        );
2502    }
2503
2504    #[test]
2505    fn sha1_blob_matches_git_known_value() {
2506        let oid = object_id_for_bytes(ObjectFormat::Sha1, "blob", b"hello\n")
2507            .expect("known blob should hash as sha1");
2508        assert_eq!(oid.to_hex(), "ce013625030ba8dba906f756967f9e9ca394464a");
2509    }
2510
2511    #[test]
2512    fn sha256_blob_matches_git_known_value() {
2513        let oid = object_id_for_bytes(ObjectFormat::Sha256, "blob", b"hello\n")
2514            .expect("known blob should hash as sha256");
2515        assert_eq!(
2516            oid.to_hex(),
2517            "2cf8d83d9ee29543b34a87727421fdecb7e3f3a183d337639025de576db9ebb4"
2518        );
2519    }
2520
2521    #[test]
2522    fn object_id_round_trips_hex() {
2523        let oid = ObjectId::from_hex(
2524            ObjectFormat::Sha1,
2525            "ce013625030ba8dba906f756967f9e9ca394464a",
2526        )
2527        .expect("valid sha1 hex");
2528        assert_eq!(oid.to_hex(), "ce013625030ba8dba906f756967f9e9ca394464a");
2529    }
2530
2531    #[test]
2532    fn object_id_writes_hex_without_allocating_in_the_writer() {
2533        let oid = ObjectId::from_hex(
2534            ObjectFormat::Sha1,
2535            "CE013625030BA8DBA906F756967F9E9CA394464A",
2536        )
2537        .expect("valid uppercase sha1 hex");
2538
2539        let mut out = String::new();
2540        oid.write_hex(&mut out)
2541            .expect("writing object id hex to a String should not fail");
2542
2543        assert_eq!(out, "ce013625030ba8dba906f756967f9e9ca394464a");
2544        assert_eq!(oid.to_hex(), out);
2545        assert_eq!(format!("{oid}"), out);
2546    }
2547
2548    #[test]
2549    fn object_id_matches_hex_prefixes_by_nibble() {
2550        let oid = ObjectId::from_hex(
2551            ObjectFormat::Sha1,
2552            "ce013625030ba8dba906f756967f9e9ca394464a",
2553        )
2554        .expect("valid sha1 hex");
2555
2556        assert!(oid.hex_prefix_matches(b""));
2557        assert!(oid.hex_prefix_matches(b"c"));
2558        assert!(oid.hex_prefix_matches(b"ce013"));
2559        assert!(oid.hex_prefix_matches(b"CE013625"));
2560        assert!(oid.hex_prefix_matches(b"ce013625030ba8dba906f756967f9e9ca394464a"));
2561
2562        assert!(!oid.hex_prefix_matches(b"d"));
2563        assert!(!oid.hex_prefix_matches(b"ce014"));
2564        assert!(!oid.hex_prefix_matches(b"ce01x"));
2565
2566        let mut too_long = oid.to_hex();
2567        too_long.push('0');
2568        assert!(!oid.hex_prefix_matches(too_long.as_bytes()));
2569    }
2570
2571    #[test]
2572    fn object_id_abbrev_hex_len_clamps_to_format_width() {
2573        let sha1 = ObjectId::null(ObjectFormat::Sha1);
2574        let sha256 = ObjectId::null(ObjectFormat::Sha256);
2575
2576        assert_eq!(sha1.abbrev_hex_len(0), 0);
2577        assert_eq!(sha1.abbrev_hex_len(12), 12);
2578        assert_eq!(sha1.abbrev_hex_len(80), ObjectFormat::Sha1.hex_len());
2579        assert_eq!(sha256.abbrev_hex_len(80), ObjectFormat::Sha256.hex_len());
2580    }
2581
2582    #[test]
2583    fn signature_parses_a_normal_ident_and_round_trips() {
2584        let line = b"A U Thor <author@example.com> 1700000000 +0000";
2585        let sig = Signature::from_ident_line(line).expect("well-formed ident parses");
2586        assert_eq!(sig.name.as_bytes(), b"A U Thor");
2587        assert_eq!(sig.email.as_bytes(), b"author@example.com");
2588        assert_eq!(sig.time.seconds, 1_700_000_000);
2589        assert_eq!(sig.time.timezone_offset_minutes, 0);
2590        assert!(!sig.time.negative_utc);
2591        // Byte-exact round-trip, and the canonical form matches here too.
2592        assert_eq!(sig.to_ident_bytes(), line);
2593        assert_eq!(sig.to_canonical_ident_bytes(), line);
2594    }
2595
2596    #[test]
2597    fn signature_parses_positive_half_hour_offset() {
2598        let line = b"Half Hour <hh@example.com> 1500000000 +0530";
2599        let sig = Signature::from_ident_line(line).expect("offset ident parses");
2600        assert_eq!(sig.time.timezone_offset_minutes, 330);
2601        assert!(!sig.time.negative_utc);
2602        assert_eq!(sig.time.offset_token(), "+0530");
2603        assert_eq!(sig.to_ident_bytes(), line);
2604        assert_eq!(sig.to_canonical_ident_bytes(), line);
2605    }
2606
2607    #[test]
2608    fn signature_parses_negative_offset() {
2609        let line = b"Western <w@example.com> 1500000000 -0500";
2610        let sig = Signature::from_ident_line(line).expect("negative offset parses");
2611        assert_eq!(sig.time.timezone_offset_minutes, -300);
2612        assert!(!sig.time.negative_utc);
2613        assert_eq!(sig.time.offset_token(), "-0500");
2614        assert_eq!(sig.to_ident_bytes(), line);
2615    }
2616
2617    #[test]
2618    fn signature_preserves_negative_zero_timezone_distinct_from_positive_zero() {
2619        let negative = b"Unknown Zone <uz@example.com> 1500000000 -0000";
2620        let positive = b"Known Zone <kz@example.com> 1500000000 +0000";
2621
2622        let neg = Signature::from_ident_line(negative).expect("-0000 parses");
2623        let pos = Signature::from_ident_line(positive).expect("+0000 parses");
2624
2625        // Both are zero minutes from UTC...
2626        assert_eq!(neg.time.timezone_offset_minutes, 0);
2627        assert_eq!(pos.time.timezone_offset_minutes, 0);
2628        // ...but the sentinel flag distinguishes them, so the times differ.
2629        assert!(neg.time.negative_utc);
2630        assert!(!pos.time.negative_utc);
2631        assert_ne!(neg.time, pos.time);
2632
2633        // And the distinction survives re-serialization, byte-for-byte.
2634        assert_eq!(neg.time.offset_token(), "-0000");
2635        assert_eq!(pos.time.offset_token(), "+0000");
2636        assert_eq!(neg.to_ident_bytes(), negative);
2637        assert_eq!(pos.to_ident_bytes(), positive);
2638        assert_eq!(neg.to_canonical_ident_bytes(), negative);
2639        assert_eq!(pos.to_canonical_ident_bytes(), positive);
2640        assert_ne!(neg.to_ident_bytes(), pos.to_ident_bytes());
2641    }
2642
2643    #[test]
2644    fn signature_handles_empty_name_and_email() {
2645        // git permits an empty name and/or empty email; the delimiters still
2646        // anchor the parse.
2647        let line = b" <> 0 +0000";
2648        let sig = Signature::from_ident_line(line).expect("empty name/email parses");
2649        assert_eq!(sig.name.as_bytes(), b"");
2650        assert_eq!(sig.email.as_bytes(), b"");
2651        assert_eq!(sig.time.seconds, 0);
2652        assert_eq!(sig.to_ident_bytes(), line);
2653    }
2654
2655    #[test]
2656    fn signature_keeps_angle_brackets_inside_the_name() {
2657        // The email is delimited by the *last* '<'/'>' pair, so a name that
2658        // itself contains angle brackets parses with the trailing pair as the
2659        // email and round-trips exactly.
2660        let line = b"Weird <Name> <weird@example.com> 1 +0000";
2661        let sig = Signature::from_ident_line(line).expect("bracketed name parses");
2662        assert_eq!(sig.name.as_bytes(), b"Weird <Name>");
2663        assert_eq!(sig.email.as_bytes(), b"weird@example.com");
2664        assert_eq!(sig.to_ident_bytes(), line);
2665    }
2666
2667    #[test]
2668    fn signature_round_trips_non_canonical_whitespace_via_raw() {
2669        // An ident with two spaces before the email is not git's canonical form,
2670        // but the parse-view must still reproduce it byte-for-byte from `raw`.
2671        // (Only the canonical renderer normalizes the spacing.)
2672        let line = b"Spaced  <spaced@example.com> 5 +0000";
2673        let sig = Signature::from_ident_line(line).expect("non-canonical ident parses");
2674        // The name keeps the extra space (only one separator space is trimmed).
2675        assert_eq!(sig.name.as_bytes(), b"Spaced ");
2676        assert_eq!(sig.to_ident_bytes(), line);
2677    }
2678
2679    #[test]
2680    fn signature_rejects_malformed_idents() {
2681        // No email delimiters.
2682        assert!(Signature::from_ident_line(b"No Email Here 0 +0000").is_none());
2683        // Missing the time tail entirely.
2684        assert!(Signature::from_ident_line(b"A U Thor <a@example.com>").is_none());
2685        // Non-numeric timestamp.
2686        assert!(Signature::from_ident_line(b"A U Thor <a@example.com> later +0000").is_none());
2687        // Malformed timezone token (wrong width).
2688        assert!(Signature::from_ident_line(b"A U Thor <a@example.com> 0 +00").is_none());
2689        // Timezone token missing a sign.
2690        assert!(Signature::from_ident_line(b"A U Thor <a@example.com> 0 0000").is_none());
2691    }
2692
2693    #[test]
2694    fn git_time_constructors_set_the_sentinel() {
2695        assert!(!GitTime::new(0, 0).negative_utc);
2696        assert_eq!(GitTime::new(0, 330).offset_token(), "+0530");
2697        let unknown = GitTime::with_negative_utc(42);
2698        assert!(unknown.negative_utc);
2699        assert_eq!(unknown.seconds, 42);
2700        assert_eq!(unknown.offset_token(), "-0000");
2701    }
2702
2703    #[test]
2704    fn full_name_accepts_valid_ref_names() {
2705        let name = FullName::new("refs/heads/main").expect("valid ref name");
2706        assert_eq!(name.as_str(), "refs/heads/main");
2707        assert_eq!(name, "refs/heads/main");
2708        assert_eq!(format!("{name}"), "refs/heads/main");
2709        assert_eq!(String::from(name.clone()), "refs/heads/main");
2710        let borrowed: &str = name.borrow();
2711        assert_eq!(borrowed, "refs/heads/main");
2712    }
2713
2714    #[test]
2715    fn full_name_rejects_invalid_ref_names() {
2716        assert!(FullName::new("").is_err());
2717        assert!(FullName::new(" refs/heads/main").is_err());
2718        assert!(FullName::new("refs/heads/main ").is_err());
2719        assert!(FullName::new("refs//heads/main").is_err());
2720        assert!(FullName::new("refs/heads/\nmain").is_err());
2721        assert!(FullName::new("refs/heads/a..b").is_err());
2722        assert!(FullName::new("refs/heads/a.lock").is_err());
2723        assert!(FullName::new("refs/heads/a~1").is_err());
2724        assert!(FullName::new("@").is_err());
2725    }
2726
2727    #[test]
2728    fn full_name_accepts_what_git_accepts() {
2729        // HeddleCo/sley#244: non-ASCII whitespace is not special to Git.
2730        assert!(FullName::new("refs/heads/\u{00A0}edge\u{00A0}").is_ok());
2731        assert!(FullName::new("HEAD").is_ok());
2732        assert!(FullName::new("refs/heads/a./b").is_ok());
2733    }
2734
2735    #[test]
2736    fn bstring_round_trips_bytes_and_displays_lossily() {
2737        let path = BString::from_bytes(b"src/\xFF.txt");
2738        assert_eq!(path.as_bytes(), b"src/\xFF.txt");
2739        let borrowed: &[u8] = path.borrow();
2740        assert_eq!(borrowed, b"src/\xFF.txt".as_slice());
2741        assert_eq!(format!("{path}"), "src/\u{FFFD}.txt");
2742        assert_eq!(path, b"src/\xFF.txt");
2743        assert_eq!(path.clone().into_bytes(), b"src/\xFF.txt".to_vec());
2744    }
2745
2746    #[test]
2747    fn split_ident_line_parses_well_formed_ident() {
2748        let f = split_ident_line(b"A U Thor <author@example.com> 1112911993 -0700")
2749            .expect("well formed ident should parse");
2750        assert_eq!(f.name, b"A U Thor");
2751        assert_eq!(f.email, b"author@example.com");
2752        assert_eq!(f.date, Some(&b"1112911993"[..]));
2753        assert_eq!(f.tz, Some(&b"-0700"[..]));
2754    }
2755
2756    #[test]
2757    fn split_ident_line_recovers_broken_email() {
2758        // git inserts junk after the '>': email stops at the first '>', but the
2759        // timestamp is found by scanning back from the end for the last '>'.
2760        let f = split_ident_line(b"A U Thor <author@example.com>-<> 1112911993 -0700")
2761            .expect("broken-email ident should parse");
2762        assert_eq!(f.name, b"A U Thor");
2763        assert_eq!(f.email, b"author@example.com");
2764        assert_eq!(f.date, Some(&b"1112911993"[..]));
2765        assert_eq!(f.tz, Some(&b"-0700"[..]));
2766    }
2767
2768    #[test]
2769    fn split_ident_line_non_numeric_date_is_person_only() {
2770        let f = split_ident_line(b"A U Thor <author@example.com> totally_bogus -0700")
2771            .expect("ident without numeric date should still parse person");
2772        assert_eq!(f.email, b"author@example.com");
2773        assert_eq!(f.date, None);
2774        assert_eq!(f.tz, None);
2775    }
2776
2777    #[test]
2778    fn split_ident_line_whitespace_date_is_person_only() {
2779        // Trailing spaces after '>' with no timestamp -> no date.
2780        let f = split_ident_line(b"A U Thor <author@example.com>    ")
2781            .expect("ident with trailing whitespace should parse person");
2782        assert_eq!(f.date, None);
2783        // A vertical tab is NOT git-isspace, so it stops the space-skip and the
2784        // (non-digit) VT yields no date either.
2785        let f = split_ident_line(b"A U Thor <author@example.com>   \x0b")
2786            .expect("ident with non-git-whitespace suffix should parse person");
2787        assert_eq!(f.date, None);
2788    }
2789
2790    #[test]
2791    fn split_ident_line_requires_angle_brackets() {
2792        assert!(split_ident_line(b"no brackets here 123 +0000").is_none());
2793    }
2794
2795    #[test]
2796    fn ident_render_date_overflow_is_epoch_sentinel() {
2797        // 2^64 + 1 (clamps in u64 parse) and 2^64 - 2 (fits u64 but past time_t)
2798        // both render the epoch sentinel with a forced +0000 timezone.
2799        assert_eq!(
2800            ident_render_date(b"18446744073709551617", b"-0700", &DateMode::Default),
2801            "Thu Jan 1 00:00:00 1970 +0000"
2802        );
2803        assert_eq!(
2804            ident_render_date(b"18446744073709551614", b"-0700", &DateMode::Default),
2805            "Thu Jan 1 00:00:00 1970 +0000"
2806        );
2807    }
2808
2809    #[test]
2810    fn ident_render_date_valid_value_uses_original_timezone() {
2811        assert_eq!(
2812            ident_render_date(b"0", b"+0000", &DateMode::Default),
2813            "Thu Jan 1 00:00:00 1970 +0000"
2814        );
2815    }
2816
2817    #[test]
2818    fn redact_url_for_display_strips_https_userinfo() {
2819        assert_eq!(
2820            redact_url_for_display("https://user:pass@host/repo.git"),
2821            "https://<redacted>@host/repo.git"
2822        );
2823    }
2824
2825    #[test]
2826    fn redact_url_for_display_leaves_urls_without_userinfo_unchanged() {
2827        assert_eq!(
2828            redact_url_for_display("https://host/repo.git"),
2829            "https://host/repo.git"
2830        );
2831        assert_eq!(redact_url_for_display("origin"), "origin");
2832    }
2833}