Skip to main content

copybook_codec/lib_api/
run_summary.rs

1// SPDX-License-Identifier: AGPL-3.0-or-later
2//! Processing run statistics and display formatting.
3
4use copybook_error::Error;
5use std::fmt;
6
7/// How many individual record failures a [`RunSummary`] retains.
8///
9/// A run over a badly mismatched file can fail on every record; keeping the
10/// first few is enough to diagnose it without holding the whole file in memory.
11/// `verify` already reports its errors under the same "first 10" convention.
12pub const MAX_CAPTURED_FAILURES: usize = 10;
13
14/// A single record that failed during a decode or encode run.
15#[derive(Debug, Clone, PartialEq)]
16pub struct RecordFailure {
17    /// 1-based index of the record within the input.
18    pub record_index: u64,
19    /// The error that caused this record to fail, with its code and context.
20    pub error: Error,
21}
22
23impl fmt::Display for RecordFailure {
24    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
25        write!(f, "record {}: {}", self.record_index, self.error)
26    }
27}
28
29/// Summary of a processing run with comprehensive statistics.
30///
31/// Captures record counts, error rates, throughput, and resource usage
32/// for a complete decode or encode operation.
33#[derive(Debug, Default, Clone, PartialEq)]
34pub struct RunSummary {
35    /// Total number of records decoded or encoded successfully.
36    pub records_processed: u64,
37    /// Number of records that encountered errors during processing.
38    pub records_with_errors: u64,
39    /// Number of non-fatal warnings generated during processing.
40    pub warnings: u64,
41    /// Wall-clock processing time in milliseconds.
42    pub processing_time_ms: u64,
43    /// Total bytes read from input.
44    ///
45    /// Fixed records count payload bytes; RDW records count header-plus-payload
46    /// physical bytes.
47    pub bytes_processed: u64,
48    /// SHA-256 fingerprint of the schema used for processing.
49    pub schema_fingerprint: String,
50    /// Processing throughput in MiB/s.
51    pub throughput_mbps: f64,
52    /// Peak memory usage in bytes, if available from the runtime.
53    pub peak_memory_bytes: Option<u64>,
54    /// Number of worker threads used for parallel processing.
55    pub threads_used: usize,
56    /// The first [`MAX_CAPTURED_FAILURES`] record failures seen during the run.
57    ///
58    /// `records_with_errors` remains the full count; this carries enough of the
59    /// detail to tell the caller *which* records failed and *why*.
60    pub failures: Vec<RecordFailure>,
61}
62
63impl RunSummary {
64    /// Create a new run summary with default values
65    #[must_use]
66    pub fn new() -> Self {
67        Self::default()
68    }
69
70    /// Create a new run summary with specified thread count
71    #[must_use]
72    pub fn with_threads(threads: usize) -> Self {
73        Self {
74            threads_used: threads,
75            ..Self::default()
76        }
77    }
78
79    /// Calculate throughput based on bytes and time
80    #[allow(clippy::cast_precision_loss)]
81    pub fn calculate_throughput(&mut self) {
82        if self.processing_time_ms > 0 {
83            let seconds = self.processing_time_ms as f64 / 1000.0;
84            let megabytes = self.bytes_processed as f64 / (1024.0 * 1024.0);
85            self.throughput_mbps = megabytes / seconds;
86        }
87    }
88
89    /// Check if processing had any errors
90    #[must_use]
91    pub const fn has_errors(&self) -> bool {
92        self.records_with_errors > 0
93    }
94
95    /// Count a failed record and retain its detail, up to [`MAX_CAPTURED_FAILURES`].
96    ///
97    /// Callers previously incremented `records_with_errors` and dropped the
98    /// `Error`, which left the caller with a count and no way to find out what
99    /// went wrong. Recording both keeps the count authoritative while making the
100    /// first failures reportable.
101    pub fn note_failure(&mut self, record_index: u64, error: &Error) {
102        self.records_with_errors += 1;
103        if self.failures.len() < MAX_CAPTURED_FAILURES {
104            self.failures.push(RecordFailure {
105                record_index,
106                error: error.clone(),
107            });
108        }
109    }
110
111    /// Number of failures that occurred beyond the retained [`Self::failures`].
112    #[must_use]
113    pub fn undisclosed_failure_count(&self) -> u64 {
114        self.records_with_errors
115            .saturating_sub(self.failures.len() as u64)
116    }
117
118    /// Check if processing had any warnings
119    #[must_use]
120    pub const fn has_warnings(&self) -> bool {
121        self.warnings > 0
122    }
123
124    /// Check if processing was successful (no errors)
125    #[must_use]
126    pub const fn is_successful(&self) -> bool {
127        !self.has_errors()
128    }
129
130    /// Get the total number of records attempted (processed + errors)
131    #[must_use]
132    pub const fn total_records(&self) -> u64 {
133        self.records_processed + self.records_with_errors
134    }
135
136    /// Get the success rate as a percentage (0.0 to 100.0)
137    #[must_use]
138    #[allow(clippy::cast_precision_loss)]
139    pub fn success_rate(&self) -> f64 {
140        let total = self.total_records();
141        if total == 0 {
142            100.0
143        } else {
144            (self.records_processed as f64 / total as f64) * 100.0
145        }
146    }
147
148    /// Get the error rate as a percentage (0.0 to 100.0)
149    #[must_use]
150    pub fn error_rate(&self) -> f64 {
151        100.0 - self.success_rate()
152    }
153
154    /// Get processing time in seconds
155    #[must_use]
156    #[allow(clippy::cast_precision_loss)]
157    pub fn processing_time_seconds(&self) -> f64 {
158        self.processing_time_ms as f64 / 1000.0
159    }
160
161    /// Get bytes processed in megabytes
162    #[must_use]
163    #[allow(clippy::cast_precision_loss)]
164    pub fn bytes_processed_mb(&self) -> f64 {
165        self.bytes_processed as f64 / (1024.0 * 1024.0)
166    }
167
168    /// Set the schema fingerprint
169    pub fn set_schema_fingerprint(&mut self, fingerprint: String) {
170        self.schema_fingerprint = fingerprint;
171    }
172
173    /// Set the peak memory usage
174    pub fn set_peak_memory_bytes(&mut self, bytes: u64) {
175        self.peak_memory_bytes = Some(bytes);
176    }
177}
178
179impl fmt::Display for RunSummary {
180    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
181        writeln!(f, "Processing Summary:")?;
182        writeln!(f, "  Records processed: {}", self.records_processed)?;
183        writeln!(f, "  Records with errors: {}", self.records_with_errors)?;
184        writeln!(f, "  Warnings: {}", self.warnings)?;
185        writeln!(f, "  Success rate: {:.1}%", self.success_rate())?;
186        writeln!(
187            f,
188            "  Processing time: {:.2}s",
189            self.processing_time_seconds()
190        )?;
191        writeln!(f, "  Bytes processed: {:.2} MB", self.bytes_processed_mb())?;
192        writeln!(f, "  Throughput: {:.2} MB/s", self.throughput_mbps)?;
193        writeln!(f, "  Threads used: {}", self.threads_used)?;
194        if let Some(peak_memory) = self.peak_memory_bytes {
195            #[allow(clippy::cast_precision_loss)]
196            let peak_mb = peak_memory as f64 / (1024.0 * 1024.0);
197            writeln!(f, "  Peak memory: {peak_mb:.2} MB")?;
198        }
199        if !self.schema_fingerprint.is_empty() {
200            writeln!(f, "  Schema fingerprint: {}", self.schema_fingerprint)?;
201        }
202        Ok(())
203    }
204}