Skip to main content

copybook_codec/
iterator.rs

1// SPDX-License-Identifier: AGPL-3.0-or-later
2//! Record iterator for streaming access to decoded records
3//!
4//! This module provides iterator-based access to records for programmatic processing,
5//! allowing users to process records one at a time without loading entire files into memory.
6//!
7//! # Overview
8//!
9//! The iterator module implements streaming record processing with bounded memory usage.
10//! It provides low-level iterator primitives for reading COBOL data files sequentially,
11//! supporting both fixed-length and RDW (Record Descriptor Word) variable-length formats.
12//!
13//! Key capabilities:
14//!
15//! 1. **Streaming iteration** ([`RecordIterator`]) - Process records one at a time
16//! 2. **Format flexibility** - Handle both fixed-length and RDW variable-length records
17//! 3. **Raw access** ([`RecordIterator::read_raw_record`]) - Access undecoded record bytes
18//! 4. **Convenience functions** ([`iter_records_from_file`], [`iter_records`]) - Simplified creation
19//!
20//! # Performance Characteristics
21//!
22//! The iterator uses buffered I/O and maintains bounded memory usage:
23//! - **Memory**: One record buffer (typically <32 KiB per record)
24//! - **Throughput**: Depends on decode complexity (DISPLAY vs COMP-3)
25//! - **Latency**: Sequential I/O optimized with `BufReader`
26//!
27//! For high-throughput parallel processing, consider using [`crate::decode_file_to_jsonl`]
28//! which provides parallel worker pools and streaming output.
29//!
30//! # Examples
31//!
32//! ## Basic Fixed-Length Record Iteration
33//!
34//! ```rust
35//! use copybook_codec::{iter_records_from_file, DecodeOptions, Codepage, RecordFormat};
36//! use copybook_core::parse_copybook;
37//!
38//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
39//! // Parse copybook schema
40//! let copybook_text = r#"
41//!     01 CUSTOMER-RECORD.
42//!        05 CUSTOMER-ID    PIC 9(5).
43//!        05 CUSTOMER-NAME  PIC X(20).
44//!        05 BALANCE        PIC S9(7)V99 COMP-3.
45//! "#;
46//! let schema = parse_copybook(copybook_text)?;
47//!
48//! // Configure decoding options
49//! let options = DecodeOptions::new()
50//!     .with_codepage(Codepage::CP037)
51//!     .with_format(RecordFormat::Fixed);
52//!
53//! // Create iterator from file
54//! # #[cfg(not(test))]
55//! let iterator = iter_records_from_file("customers.bin", &schema, &options)?;
56//!
57//! // Process records one at a time
58//! # #[cfg(not(test))]
59//! for (index, result) in iterator.enumerate() {
60//!     match result {
61//!         Ok(json_value) => {
62//!             println!("Record {}: {}", index + 1, json_value);
63//!         }
64//!         Err(error) => {
65//!             eprintln!("Error in record {}: {}", index + 1, error);
66//!             break; // Stop on first error
67//!         }
68//!     }
69//! }
70//! # Ok(())
71//! # }
72//! ```
73//!
74//! ## RDW Variable-Length Records
75//!
76//! ```rust
77//! use copybook_codec::{RecordIterator, DecodeOptions, RecordFormat};
78//! use copybook_core::parse_copybook;
79//! use std::fs::File;
80//!
81//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
82//! let copybook_text = r#"
83//!     01 TRANSACTION.
84//!        05 TRAN-ID       PIC 9(10).
85//!        05 TRAN-AMOUNT   PIC S9(9)V99 COMP-3.
86//!        05 TRAN-DESC     PIC X(100).
87//! "#;
88//! let schema = parse_copybook(copybook_text)?;
89//!
90//! let options = DecodeOptions::new()
91//!     .with_format(RecordFormat::RDW);  // RDW variable-length format
92//!
93//! # #[cfg(not(test))]
94//! let file = File::open("transactions.dat")?;
95//! # #[cfg(test)]
96//! # let file = std::io::Cursor::new(vec![]);
97//! let mut iterator = RecordIterator::new(file, &schema, &options)?;
98//!
99//! // Process with error recovery
100//! let mut processed = 0;
101//! let mut errors = 0;
102//!
103//! for (index, result) in iterator.enumerate() {
104//!     match result {
105//!         Ok(json_value) => {
106//!             processed += 1;
107//!             // Process record...
108//!         }
109//!         Err(error) => {
110//!             errors += 1;
111//!             eprintln!("Record {}: {}", index + 1, error);
112//!
113//!             if errors > 10 {
114//!                 eprintln!("Too many errors, stopping");
115//!                 break;
116//!             }
117//!         }
118//!     }
119//! }
120//!
121//! println!("Processed: {}, Errors: {}", processed, errors);
122//! # Ok(())
123//! # }
124//! ```
125//!
126//! ## Raw Record Access (No Decoding)
127//!
128//! ```rust
129//! use copybook_codec::{RecordIterator, DecodeOptions, RecordFormat};
130//! use copybook_core::parse_copybook;
131//! use std::io::Cursor;
132//!
133//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
134//! let copybook_text = "01 RECORD.\n   05 DATA PIC X(10).";
135//! let schema = parse_copybook(copybook_text)?;
136//!
137//! let options = DecodeOptions::new()
138//!     .with_format(RecordFormat::Fixed);
139//!
140//! let data = b"RECORD0001RECORD0002";
141//! let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options)?;
142//!
143//! // Read raw bytes without JSON decoding
144//! while let Some(raw_bytes) = iterator.read_raw_record()? {
145//!     println!("Raw record {}: {} bytes",
146//!              iterator.current_record_index(),
147//!              raw_bytes.len());
148//!
149//!     // Process raw bytes directly...
150//!     // (useful for binary analysis, checksums, etc.)
151//! }
152//! # Ok(())
153//! # }
154//! ```
155//!
156//! ## Collecting Records into a Vec
157//!
158//! ```rust
159//! use copybook_codec::{iter_records, DecodeOptions};
160//! use copybook_core::parse_copybook;
161//! use serde_json::Value;
162//! use std::io::Cursor;
163//!
164//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
165//! let copybook_text = "01 RECORD.\n   05 ID PIC 9(5).";
166//! let schema = parse_copybook(copybook_text)?;
167//! let options = DecodeOptions::default();
168//!
169//! let data = b"0000100002";
170//! let iterator = iter_records(Cursor::new(data), &schema, &options)?;
171//!
172//! // Collect all successful records
173//! let records: Vec<Value> = iterator
174//!     .filter_map(Result::ok)  // Skip errors
175//!     .collect();
176//!
177//! println!("Collected {} records", records.len());
178//! # Ok(())
179//! # }
180//! ```
181//!
182//! ## Using with `DecodeOptions` and Metadata
183//!
184//! ```rust
185//! use copybook_codec::{iter_records_from_file, DecodeOptions, Codepage, JsonNumberMode};
186//! use copybook_core::parse_copybook;
187//!
188//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
189//! let copybook_text = r#"
190//!     01 RECORD.
191//!        05 AMOUNT PIC S9(9)V99 COMP-3.
192//! "#;
193//! let schema = parse_copybook(copybook_text)?;
194//!
195//! // Configure with lossless numbers and metadata
196//! let options = DecodeOptions::new()
197//!     .with_codepage(Codepage::CP037)
198//!     .with_json_number_mode(JsonNumberMode::Lossless)
199//!     .with_emit_meta(true);  // Include field metadata
200//!
201//! # #[cfg(not(test))]
202//! let iterator = iter_records_from_file("data.bin", &schema, &options)?;
203//!
204//! # #[cfg(not(test))]
205//! for result in iterator {
206//!     let json_value = result?;
207//!     // JSON includes metadata: {"AMOUNT": "123.45", "_meta": {...}}
208//!     println!("{}", serde_json::to_string_pretty(&json_value)?);
209//! }
210//! # Ok(())
211//! # }
212//! ```
213
214use crate::lib_api::decode_record_with_raw_data;
215use crate::options::{DecodeOptions, RecordFormat};
216use copybook_core::{Error, ErrorCode, ErrorContext, Result, Schema};
217use copybook_rdw::{RdwHeader, VbBlockReader};
218use serde_json::Value;
219use std::io::{BufReader, Read};
220
221/// Outcome of reading a whole record-sized unit at a record boundary.
222enum BoundaryRead {
223    /// The buffer was filled.
224    Complete,
225    /// The reader was already exhausted; nothing was consumed.
226    CleanEof,
227    /// The reader ran out mid-unit after this many bytes.
228    Partial(usize),
229}
230
231/// Fill `buffer` from `reader`, distinguishing a clean end of file from a
232/// truncated final unit.
233///
234/// [`Read::read_exact`] reports [`std::io::ErrorKind::UnexpectedEof`] for both
235/// cases and consumes the bytes it did read, so it cannot tell "the file ended
236/// on a record boundary" from "the file ends with a partial record". Treating
237/// the two alike silently discards the trailing bytes.
238fn fill_at_record_boundary<R: Read>(
239    reader: &mut R,
240    buffer: &mut [u8],
241) -> std::io::Result<BoundaryRead> {
242    let mut filled = 0;
243    while filled < buffer.len() {
244        match reader.read(&mut buffer[filled..]) {
245            Ok(0) => break,
246            Ok(read) => filled += read,
247            Err(e) if e.kind() == std::io::ErrorKind::Interrupted => {}
248            Err(e) => return Err(e),
249        }
250    }
251
252    if filled == buffer.len() {
253        Ok(BoundaryRead::Complete)
254    } else if filled == 0 {
255        Ok(BoundaryRead::CleanEof)
256    } else {
257        Ok(BoundaryRead::Partial(filled))
258    }
259}
260
261/// Iterator over records in a data file, yielding decoded JSON values
262///
263/// This iterator provides streaming access to records, processing them one at a time
264/// to maintain bounded memory usage even for very large files.
265///
266/// # Examples
267///
268/// ```rust,no_run
269/// use copybook_codec::{RecordIterator, DecodeOptions};
270/// use copybook_core::parse_copybook;
271/// # use std::io::Cursor;
272///
273/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
274/// let copybook_text = "01 RECORD.\n   05 ID PIC 9(5).\n   05 NAME PIC X(20).";
275/// let mut schema = parse_copybook(copybook_text)?;
276/// schema.lrecl_fixed = Some(25);
277/// let options = DecodeOptions::default();
278/// # let record_bytes = b"00001ALICE               ";
279/// # let file = Cursor::new(&record_bytes[..]);
280/// // let file = std::fs::File::open("data.bin")?;
281///
282/// let mut iterator = RecordIterator::new(file, &schema, &options)?;
283///
284/// for (record_index, result) in iterator.enumerate() {
285///     match result {
286///         Ok(json_value) => {
287///             println!("Record {}: {}", record_index + 1, json_value);
288///         }
289///         Err(error) => {
290///             eprintln!("Error in record {}: {}", record_index + 1, error);
291///         }
292///     }
293/// }
294/// # Ok(())
295/// # }
296/// ```
297/// Byte-source framing for [`RecordIterator`]: a plain buffered stream for
298/// fixed/RDW records, or a bounded VB block reader.
299#[derive(Debug)]
300enum FramingInput<R: Read> {
301    Stream(BufReader<R>),
302    Blocks(VbBlockReader<R>),
303}
304
305pub struct RecordIterator<R: Read> {
306    /// The buffered reader or VB block reader
307    input: FramingInput<R>,
308    /// The schema for decoding records
309    schema: Schema,
310    /// Decoding options
311    options: DecodeOptions,
312    /// Current record index (1-based)
313    record_index: u64,
314    /// Whether the iterator has reached EOF
315    eof_reached: bool,
316    /// Buffer for reading record data
317    buffer: Vec<u8>,
318    /// Complete RDW frame for `RawMode::RecordRDW` envelope capture.
319    raw_data_with_header: Option<Vec<u8>>,
320}
321
322impl<R: Read> RecordIterator<R> {
323    /// Create a new record iterator
324    ///
325    /// # Arguments
326    ///
327    /// * `reader` - The input stream to read from
328    /// * `schema` - The parsed copybook schema
329    /// * `options` - Decoding options
330    ///
331    /// # Errors
332    /// Returns an error if the record format is incompatible with the schema.
333    #[inline]
334    #[must_use = "Handle the Result or propagate the error"]
335    pub fn new(reader: R, schema: &Schema, options: &DecodeOptions) -> Result<Self> {
336        let input = if options.format == RecordFormat::Vb {
337            FramingInput::Blocks(VbBlockReader::new(reader, options.strict_mode))
338        } else {
339            FramingInput::Stream(BufReader::new(reader))
340        };
341        Ok(Self {
342            input,
343            schema: schema.clone(),
344            options: options.clone(),
345            record_index: 0,
346            eof_reached: false,
347            buffer: Vec::new(),
348            raw_data_with_header: None,
349        })
350    }
351
352    /// Get the current record index (1-based)
353    ///
354    /// This returns the index of the last record that was successfully read,
355    /// or 0 if no records have been read yet.
356    #[inline]
357    #[must_use]
358    pub fn current_record_index(&self) -> u64 {
359        self.record_index
360    }
361
362    /// Check if the iterator has reached the end of the file
363    #[inline]
364    #[must_use]
365    pub fn is_eof(&self) -> bool {
366        self.eof_reached
367    }
368
369    /// Get a reference to the schema being used
370    #[inline]
371    #[must_use]
372    pub fn schema(&self) -> &Schema {
373        &self.schema
374    }
375
376    /// Get a reference to the decode options being used
377    #[inline]
378    #[must_use]
379    pub fn options(&self) -> &DecodeOptions {
380        &self.options
381    }
382
383    fn record_context_for(record_index: u64, details: String) -> ErrorContext {
384        ErrorContext {
385            record_index: Some(record_index + 1),
386            field_path: None,
387            byte_offset: None,
388            line_number: None,
389            details: Some(details),
390        }
391    }
392
393    fn rdw_header_read_error_for(record_index: u64, error: &std::io::Error) -> Error {
394        Error::new(
395            ErrorCode::CBKR201_RDW_READ_ERROR,
396            format!("Failed to read RDW header: {error}"),
397        )
398        .with_context(Self::record_context_for(
399            record_index,
400            "I/O failure while reading RDW header".to_string(),
401        ))
402    }
403
404    fn rdw_payload_read_error_for(
405        record_index: u64,
406        error: &std::io::Error,
407        length: usize,
408    ) -> Error {
409        let (code, details) = if error.kind() == std::io::ErrorKind::UnexpectedEof {
410            (
411                ErrorCode::CBKF221_RDW_UNDERFLOW,
412                format!("File ends before the declared {length}-byte RDW payload"),
413            )
414        } else {
415            (
416                ErrorCode::CBKR201_RDW_READ_ERROR,
417                "I/O failure while reading RDW payload".to_string(),
418            )
419        };
420        Error::new(code, format!("Failed to read RDW payload: {error}"))
421            .with_context(Self::record_context_for(record_index, details))
422    }
423
424    /// Read the next record without decoding it
425    ///
426    /// This method reads the raw bytes of the next record without performing
427    /// JSON decoding. Useful for applications that need access to raw record data
428    /// for binary analysis, checksums, or custom processing.
429    ///
430    /// # Returns
431    ///
432    /// * `Ok(Some(bytes))` - The raw record bytes
433    /// * `Ok(None)` - End of file reached
434    /// * `Err(error)` - An error occurred while reading
435    ///
436    /// # Errors
437    /// Returns an error if underlying I/O operations fail or the record format is invalid.
438    ///
439    /// # Examples
440    ///
441    /// ```rust
442    /// use copybook_codec::{RecordIterator, DecodeOptions, RecordFormat};
443    /// use copybook_core::parse_copybook;
444    /// use std::io::Cursor;
445    ///
446    /// # fn example() -> Result<(), Box<dyn std::error::Error>> {
447    /// let copybook_text = "01 RECORD.\n   05 DATA PIC X(8).";
448    /// let schema = parse_copybook(copybook_text)?;
449    ///
450    /// let options = DecodeOptions::new()
451    ///     .with_format(RecordFormat::Fixed);
452    ///
453    /// let data = b"RECORD01RECORD02";
454    /// let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options)?;
455    ///
456    /// // Read raw bytes
457    /// if let Some(raw_bytes) = iterator.read_raw_record()? {
458    ///     assert_eq!(raw_bytes, b"RECORD01");
459    ///     assert_eq!(iterator.current_record_index(), 1);
460    /// }
461    ///
462    /// if let Some(raw_bytes) = iterator.read_raw_record()? {
463    ///     assert_eq!(raw_bytes, b"RECORD02");
464    ///     assert_eq!(iterator.current_record_index(), 2);
465    /// }
466    ///
467    /// // End of file
468    /// assert!(iterator.read_raw_record()?.is_none());
469    /// assert!(iterator.is_eof());
470    /// # Ok(())
471    /// # }
472    /// ```
473    #[inline]
474    #[must_use = "Handle the Result or propagate the error"]
475    pub fn read_raw_record(&mut self) -> Result<Option<Vec<u8>>> {
476        if self.eof_reached {
477            return Ok(None);
478        }
479
480        self.buffer.clear();
481        self.raw_data_with_header = None;
482
483        if self.options.format == RecordFormat::Vb {
484            return self.read_vb_record();
485        }
486        let reader = match &mut self.input {
487            FramingInput::Stream(reader) => reader,
488            FramingInput::Blocks(_) => {
489                return Err(Error::new(
490                    ErrorCode::CBKI001_INVALID_STATE,
491                    "VB block reader active for fixed/RDW decode",
492                ));
493            }
494        };
495
496        let record_data = match self.options.format {
497            RecordFormat::Fixed => {
498                let lrecl = crate::file::fixed::lrecl(&self.schema)? as usize;
499                self.buffer.resize(lrecl, 0);
500
501                match fill_at_record_boundary(reader, &mut self.buffer) {
502                    Ok(BoundaryRead::Complete) => {
503                        self.record_index += 1;
504                        Some(self.buffer.clone())
505                    }
506                    Ok(BoundaryRead::CleanEof) => {
507                        self.eof_reached = true;
508                        return Ok(None);
509                    }
510                    Ok(BoundaryRead::Partial(read)) => {
511                        self.eof_reached = true;
512                        return Err(Error::new(
513                            ErrorCode::CBKR101_FIXED_RECORD_ERROR,
514                            format!("Incomplete record at end of file: expected {lrecl} bytes"),
515                        )
516                        .with_context(ErrorContext {
517                            record_index: Some(self.record_index + 1),
518                            field_path: None,
519                            byte_offset: None,
520                            line_number: None,
521                            details: Some(format!(
522                                "File ends with partial record ({read} of {lrecl} bytes)"
523                            )),
524                        }));
525                    }
526                    Err(e) => {
527                        return Err(Error::new(
528                            ErrorCode::CBKR101_FIXED_RECORD_ERROR,
529                            format!("Failed to read fixed record: {e}"),
530                        ));
531                    }
532                }
533            }
534            RecordFormat::RDW => Self::read_rdw_raw_record(
535                reader,
536                &mut self.buffer,
537                &mut self.raw_data_with_header,
538                &mut self.record_index,
539                &mut self.eof_reached,
540            )?,
541            RecordFormat::Vb => {
542                return Err(Error::new(
543                    ErrorCode::CBKI001_INVALID_STATE,
544                    "VB format handled by the block reader path",
545                ));
546            }
547        };
548
549        Ok(record_data)
550    }
551
552    /// Read one RDW-framed record payload from the buffered stream.
553    ///
554    /// A free-standing helper (rather than a `&mut self` method) so the
555    /// caller can hold the `FramingInput` borrow across the call.
556    ///
557    /// # Errors
558    /// Returns `CBKF221_RDW_UNDERFLOW` on a truncated header or payload.
559    #[inline]
560    #[must_use = "Handle the Result or propagate the error"]
561    fn read_rdw_raw_record(
562        reader: &mut std::io::BufReader<R>,
563        buffer: &mut Vec<u8>,
564        raw_data_with_header: &mut Option<Vec<u8>>,
565        record_index: &mut u64,
566        eof_reached: &mut bool,
567    ) -> Result<Option<Vec<u8>>> {
568        // Read RDW header
569        let mut rdw_header = [0u8; 4];
570        match fill_at_record_boundary(reader, &mut rdw_header) {
571            Ok(BoundaryRead::Complete) => {}
572            Ok(BoundaryRead::CleanEof) => {
573                *eof_reached = true;
574                return Ok(None);
575            }
576            Ok(BoundaryRead::Partial(read)) => {
577                *eof_reached = true;
578                return Err(Error::new(
579                    ErrorCode::CBKF221_RDW_UNDERFLOW,
580                    "Incomplete RDW header at end of file: expected 4 bytes".to_string(),
581                )
582                .with_context(Self::record_context_for(
583                    *record_index,
584                    format!("File ends with partial RDW header ({read} of 4 bytes)"),
585                )));
586            }
587            Err(error) => {
588                return Err(Self::rdw_header_read_error_for(*record_index, &error));
589            }
590        }
591
592        // Parse length (payload bytes only)
593        let length = usize::from(RdwHeader::from_bytes(rdw_header).length());
594
595        // Read payload
596        buffer.resize(length, 0);
597        match reader.read_exact(buffer) {
598            Ok(()) => {
599                let mut framed = Vec::with_capacity(4 + length);
600                framed.extend_from_slice(&rdw_header);
601                framed.extend_from_slice(buffer);
602                *raw_data_with_header = Some(framed);
603                *record_index += 1;
604                Ok(Some(buffer.clone()))
605            }
606            Err(error) => Err(Self::rdw_payload_read_error_for(
607                *record_index,
608                &error,
609                length,
610            )),
611        }
612    }
613
614    /// Read the next VB-framed record payload.
615    ///
616    /// The `raw_data_with_header` envelope carries the original record RDW
617    /// header plus payload, matching `RawMode::RecordRDW` semantics.
618    fn read_vb_record(&mut self) -> Result<Option<Vec<u8>>> {
619        let FramingInput::Blocks(blocks) = &mut self.input else {
620            return Err(Error::new(
621                ErrorCode::CBKI001_INVALID_STATE,
622                "stream reader active for VB decode",
623            ));
624        };
625        match blocks.read_record()? {
626            None => {
627                self.eof_reached = true;
628                Ok(None)
629            }
630            Some(record) => {
631                let mut raw_data_with_header =
632                    Vec::with_capacity(record.rdw.len() + record.payload.len());
633                raw_data_with_header.extend_from_slice(&record.rdw);
634                raw_data_with_header.extend_from_slice(&record.payload);
635                self.raw_data_with_header = Some(raw_data_with_header);
636                self.record_index += 1;
637                Ok(Some(record.payload))
638            }
639        }
640    }
641
642    /// Decode the next record to JSON
643    ///
644    /// This is the main method used by the Iterator implementation.
645    /// It reads and decodes the next record in one operation.
646    #[inline]
647    fn decode_next_record(&mut self) -> Result<Option<Value>> {
648        match self.read_raw_record()? {
649            Some(record_bytes) => {
650                let raw_data_with_header = self.raw_data_with_header.take();
651                let json_value = decode_record_with_raw_data(
652                    &self.schema,
653                    &record_bytes,
654                    &self.options,
655                    raw_data_with_header.as_deref(),
656                    self.record_index,
657                )?;
658                Ok(Some(json_value))
659            }
660            None => Ok(None),
661        }
662    }
663}
664
665impl<R: Read> Iterator for RecordIterator<R> {
666    type Item = Result<Value>;
667
668    #[inline]
669    fn next(&mut self) -> Option<Self::Item> {
670        if self.eof_reached {
671            return None;
672        }
673
674        match self.decode_next_record() {
675            Ok(Some(value)) => Some(Ok(value)),
676            Ok(None) => {
677                self.eof_reached = true;
678                None
679            }
680            Err(error) => {
681                // On error, we still advance the record index if we were able to read something
682                Some(Err(error))
683            }
684        }
685    }
686}
687
688/// Convenience function to create a record iterator from a file path
689///
690/// This is the most common way to create an iterator for processing COBOL data files.
691/// It handles file opening and iterator creation in a single call.
692///
693/// # Arguments
694///
695/// * `file_path` - Path to the data file
696/// * `schema` - The parsed copybook schema
697/// * `options` - Decoding options
698///
699/// # Errors
700/// Returns an error if the file cannot be opened or the iterator cannot be created.
701///
702/// # Examples
703///
704/// ## Basic Usage with Fixed-Length Records
705///
706/// ```rust,no_run
707/// use copybook_codec::{iter_records_from_file, DecodeOptions, Codepage, RecordFormat};
708/// use copybook_core::parse_copybook;
709///
710/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
711/// let copybook_text = r#"
712///     01 EMPLOYEE-RECORD.
713///        05 EMP-ID        PIC 9(6).
714///        05 EMP-NAME      PIC X(30).
715///        05 EMP-SALARY    PIC S9(7)V99 COMP-3.
716/// "#;
717/// let schema = parse_copybook(copybook_text)?;
718///
719/// let options = DecodeOptions::new()
720///     .with_codepage(Codepage::CP037)
721///     .with_format(RecordFormat::Fixed);
722///
723/// let iterator = iter_records_from_file("employees.dat", &schema, &options)?;
724///
725/// for (index, result) in iterator.enumerate() {
726///     match result {
727///         Ok(employee) => println!("Employee {}: {}", index + 1, employee),
728///         Err(e) => eprintln!("Error at record {}: {}", index + 1, e),
729///     }
730/// }
731/// # Ok(())
732/// # }
733/// ```
734///
735/// ## Processing with Error Limits
736///
737/// ```rust,no_run
738/// use copybook_codec::{iter_records_from_file, DecodeOptions};
739/// use copybook_core::parse_copybook;
740///
741/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
742/// # let schema = parse_copybook("01 R.\n   05 F PIC X(1).")?;
743/// # let options = DecodeOptions::default();
744/// let iterator = iter_records_from_file("data.bin", &schema, &options)?;
745///
746/// let mut success_count = 0;
747/// let mut error_count = 0;
748/// const MAX_ERRORS: usize = 100;
749///
750/// for result in iterator {
751///     match result {
752///         Ok(_) => success_count += 1,
753///         Err(e) => {
754///             error_count += 1;
755///             eprintln!("Error: {}", e);
756///
757///             if error_count >= MAX_ERRORS {
758///                 eprintln!("Too many errors, aborting");
759///                 break;
760///             }
761///         }
762///     }
763/// }
764///
765/// println!("Success: {}, Errors: {}", success_count, error_count);
766/// # Ok(())
767/// # }
768/// ```
769#[inline]
770#[must_use = "Handle the Result or propagate the error"]
771pub fn iter_records_from_file<P: AsRef<std::path::Path>>(
772    file_path: P,
773    schema: &Schema,
774    options: &DecodeOptions,
775) -> Result<RecordIterator<std::fs::File>> {
776    let file = std::fs::File::open(file_path).map_err(|e| {
777        Error::new(
778            ErrorCode::CBKR201_RDW_READ_ERROR,
779            format!("failed to open input file: {e}"),
780        )
781    })?;
782
783    RecordIterator::new(file, schema, options)
784}
785
786/// Convenience function to create a record iterator from any readable source
787///
788/// This function provides maximum flexibility by accepting any type that implements
789/// the `Read` trait, including files, cursors, network streams, or custom readers.
790///
791/// # Arguments
792///
793/// * `reader` - Any type implementing Read (File, Cursor, `TcpStream`, etc.)
794/// * `schema` - The parsed copybook schema
795/// * `options` - Decoding options
796///
797/// # Errors
798/// Returns an error if the iterator cannot be created.
799///
800/// # Examples
801///
802/// ## Using with In-Memory Data (Cursor)
803///
804/// ```rust
805/// use copybook_codec::{iter_records, DecodeOptions, RecordFormat};
806/// use copybook_core::parse_copybook;
807/// use std::io::Cursor;
808///
809/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
810/// let copybook_text = "01 RECORD.\n   05 ID PIC 9(3).\n   05 NAME PIC X(5).";
811/// let schema = parse_copybook(copybook_text)?;
812///
813/// let options = DecodeOptions::new()
814///     .with_format(RecordFormat::Fixed);
815///
816/// // Create iterator from in-memory data
817/// let data = b"001ALICE002BOB  003CAROL";
818/// let iterator = iter_records(Cursor::new(data), &schema, &options)?;
819///
820/// let records: Vec<_> = iterator.collect::<Result<Vec<_>, _>>()?;
821/// assert_eq!(records.len(), 3);
822/// # Ok(())
823/// # }
824/// ```
825///
826/// ## Using with File
827///
828/// ```rust,no_run
829/// use copybook_codec::{iter_records, DecodeOptions};
830/// use copybook_core::parse_copybook;
831/// use std::fs::File;
832///
833/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
834/// let schema = parse_copybook("01 RECORD.\n   05 DATA PIC X(10).")?;
835/// let options = DecodeOptions::default();
836///
837/// let file = File::open("data.bin")?;
838/// let iterator = iter_records(file, &schema, &options)?;
839///
840/// for result in iterator {
841///     let record = result?;
842///     println!("{}", record);
843/// }
844/// # Ok(())
845/// # }
846/// ```
847///
848/// ## Using with Compressed Data
849///
850/// ```text
851/// use copybook_codec::{iter_records, DecodeOptions};
852/// use copybook_core::parse_copybook;
853/// use std::fs::File;
854/// use flate2::read::GzDecoder;
855///
856/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
857/// let schema = parse_copybook("01 RECORD.\n   05 DATA PIC X(10).")?;
858/// let options = DecodeOptions::default();
859///
860/// // Read from gzipped file
861/// let file = File::open("data.bin.gz")?;
862/// let decoder = GzDecoder::new(file);
863/// let iterator = iter_records(decoder, &schema, &options)?;
864///
865/// for result in iterator {
866///     let record = result?;
867///     // Process decompressed record...
868/// }
869/// # Ok(())
870/// # }
871/// ```
872#[inline]
873#[must_use = "Handle the Result or propagate the error"]
874pub fn iter_records<R: Read>(
875    reader: R,
876    schema: &Schema,
877    options: &DecodeOptions,
878) -> Result<RecordIterator<R>> {
879    RecordIterator::new(reader, schema, options)
880}
881
882#[cfg(test)]
883#[allow(clippy::expect_used)]
884#[allow(clippy::unwrap_used)]
885#[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)]
886mod tests {
887    use super::*;
888    use crate::Codepage;
889    use copybook_core::parse_copybook;
890    use std::collections::VecDeque;
891    use std::io::{self, Cursor, Read};
892
893    #[test]
894    fn test_record_iterator_basic() {
895        let copybook_text = r"
896            01 RECORD.
897               05 ID PIC 9(3).
898               05 NAME PIC X(5).
899        ";
900
901        let schema = parse_copybook(copybook_text).unwrap();
902
903        // Create test data: two 8-byte fixed records
904        let test_data = b"001ALICE002BOB  ";
905        let cursor = Cursor::new(test_data);
906
907        let options = DecodeOptions {
908            format: RecordFormat::Fixed,
909            ..DecodeOptions::default()
910        };
911
912        let iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
913
914        // Just test that the iterator can be created successfully
915        assert_eq!(iterator.current_record_index(), 0);
916        assert!(!iterator.is_eof());
917    }
918
919    #[test]
920    fn test_record_iterator_rdw() {
921        let copybook_text = r"
922            01 RECORD.
923               05 ID PIC 9(3).
924               05 NAME PIC X(5).
925        ";
926
927        let schema = parse_copybook(copybook_text).unwrap();
928
929        // Create RDW test data:
930        // Record 1: length=8, reserved=0, data="001ALICE"
931        // Record 2: length=6, reserved=0, data="002BOB"
932        let test_data = vec![
933            0x00, 0x08, 0x00, 0x00, // RDW header: length=8, reserved=0
934            b'0', b'0', b'1', b'A', b'L', b'I', b'C', b'E', // Record 1 data
935            0x00, 0x06, 0x00, 0x00, // RDW header: length=6, reserved=0
936            b'0', b'0', b'2', b'B', b'O', b'B', // Record 2 data
937        ];
938
939        let cursor = Cursor::new(test_data);
940
941        let options = DecodeOptions {
942            format: RecordFormat::RDW,
943            ..DecodeOptions::default()
944        };
945
946        let iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
947
948        // Just test that the iterator can be created successfully
949        assert_eq!(iterator.current_record_index(), 0);
950        assert!(!iterator.is_eof());
951    }
952
953    #[test]
954    fn test_raw_record_reading() {
955        let copybook_text = r"
956            01 RECORD.
957               05 ID PIC 9(3).
958               05 NAME PIC X(5).
959        ";
960
961        let schema = parse_copybook(copybook_text).unwrap();
962
963        let test_data = b"001ALICE";
964        let cursor = Cursor::new(test_data);
965
966        let options = DecodeOptions {
967            format: RecordFormat::Fixed,
968            ..DecodeOptions::default()
969        };
970
971        let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
972
973        // Read raw record
974        let raw_record = iterator.read_raw_record().unwrap().unwrap();
975        assert_eq!(raw_record, b"001ALICE");
976        assert_eq!(iterator.current_record_index(), 1);
977
978        // End of file
979        assert!(iterator.read_raw_record().unwrap().is_none());
980    }
981
982    #[test]
983    fn test_iterator_error_handling() {
984        let copybook_text = r"
985            01 RECORD.
986               05 ID PIC 9(3).
987               05 NAME PIC X(5).
988        ";
989
990        let schema = parse_copybook(copybook_text).unwrap();
991
992        // Create incomplete record (only 4 bytes instead of 8)
993        let test_data = b"001A";
994        let cursor = Cursor::new(test_data);
995
996        let options = DecodeOptions {
997            format: RecordFormat::Fixed,
998            ..DecodeOptions::default()
999        };
1000
1001        let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1002
1003        // A truncated final record is reported, not silently dropped: returning
1004        // EOF here would discard the four trailing bytes with no signal to the
1005        // caller.
1006        let error = iterator
1007            .next()
1008            .expect("truncated data yields an item")
1009            .expect_err("truncated data is an error");
1010        assert_eq!(error.code, ErrorCode::CBKR101_FIXED_RECORD_ERROR);
1011        assert!(iterator.next().is_none());
1012    }
1013
1014    #[test]
1015    fn test_iterator_fixed_format_missing_lrecl_errors_on_next() {
1016        // A schema without a fixed record length
1017        let copybook_text = "01 SOME-GROUP. 05 SOME-FIELD PIC X(1).";
1018        let mut schema = parse_copybook(copybook_text).unwrap();
1019        schema.lrecl_fixed = None; // Ensure it's None
1020
1021        let test_data = b"";
1022        let cursor = Cursor::new(test_data);
1023
1024        let options = DecodeOptions {
1025            format: RecordFormat::Fixed,
1026            ..DecodeOptions::default()
1027        };
1028
1029        let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1030
1031        let first = iterator.next().unwrap();
1032        assert!(first.is_err());
1033        if let Err(e) = first {
1034            assert_eq!(e.code, ErrorCode::CBKI001_INVALID_STATE);
1035            assert_eq!(e.message, crate::file::fixed::FIXED_FORMAT_LRECL_MISSING);
1036        }
1037    }
1038
1039    #[test]
1040    fn test_iterator_fixed_format_zero_lrecl_errors_on_next() {
1041        let copybook_text = "01 SOME-GROUP. 05 SOME-FIELD PIC X(1).";
1042        let mut schema = parse_copybook(copybook_text).unwrap();
1043        schema.lrecl_fixed = Some(0);
1044
1045        let options = DecodeOptions {
1046            format: RecordFormat::Fixed,
1047            ..DecodeOptions::default()
1048        };
1049
1050        let mut iterator = RecordIterator::new(Cursor::new(b"DATA"), &schema, &options).unwrap();
1051
1052        let error = iterator.next().unwrap().unwrap_err();
1053        assert_eq!(error.code, ErrorCode::CBKI001_INVALID_STATE);
1054        assert_eq!(error.message, "LRECL must be greater than zero");
1055    }
1056
1057    #[test]
1058    fn test_iterator_schema_and_options_accessors() {
1059        let copybook_text = r"
1060            01 RECORD.
1061               05 ID PIC 9(3).
1062               05 NAME PIC X(5).
1063        ";
1064
1065        let mut schema = parse_copybook(copybook_text).unwrap();
1066        schema.lrecl_fixed = Some(8);
1067        let test_data = b"001ALICE";
1068        let cursor = Cursor::new(test_data);
1069
1070        let options = DecodeOptions {
1071            format: RecordFormat::Fixed,
1072            codepage: Codepage::ASCII,
1073            ..DecodeOptions::default()
1074        };
1075
1076        let iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1077
1078        // Test schema accessor
1079        assert_eq!(iterator.schema().fields[0].name, "RECORD");
1080
1081        // Test options accessor
1082        assert_eq!(iterator.options().format, RecordFormat::Fixed);
1083    }
1084
1085    #[test]
1086    fn test_iterator_multiple_fixed_records() {
1087        let copybook_text = r"
1088            01 RECORD.
1089               05 ID PIC 9(3).
1090               05 NAME PIC X(5).
1091        ";
1092
1093        let mut schema = parse_copybook(copybook_text).unwrap();
1094        schema.lrecl_fixed = Some(8);
1095
1096        // Create test data: three 8-byte fixed records
1097        let test_data = b"001ALICE002BOB  003CAROL";
1098        let cursor = Cursor::new(test_data);
1099
1100        let options = DecodeOptions {
1101            format: RecordFormat::Fixed,
1102            codepage: Codepage::ASCII,
1103            ..DecodeOptions::default()
1104        };
1105
1106        let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1107
1108        // Read all records
1109        let mut count = 0;
1110        for result in iterator.by_ref() {
1111            assert!(result.is_ok(), "Record {count} should decode successfully");
1112            count += 1;
1113        }
1114
1115        assert_eq!(count, 3);
1116        assert_eq!(iterator.current_record_index(), 3);
1117        assert!(iterator.is_eof());
1118    }
1119
1120    #[test]
1121    fn test_iterator_rdw_multiple_records() {
1122        let copybook_text = r"
1123            01 RECORD.
1124               05 ID PIC 9(3).
1125               05 NAME PIC X(5).
1126        ";
1127
1128        let schema = parse_copybook(copybook_text).unwrap();
1129
1130        // Create RDW test data with three records
1131        let test_data = vec![
1132            // Record 1
1133            0x00, 0x08, 0x00, 0x00, // RDW header: length=8
1134            b'0', b'0', b'1', b'A', b'L', b'I', b'C', b'E', // Record 2
1135            0x00, 0x06, 0x00, 0x00, // RDW header: length=6
1136            b'0', b'0', b'2', b'B', b'O', b'B', // Record 3
1137            0x00, 0x08, 0x00, 0x00, // RDW header: length=8
1138            b'0', b'0', b'3', b'C', b'A', b'R', b'O', b'L',
1139        ];
1140
1141        let cursor = Cursor::new(test_data);
1142
1143        let options = DecodeOptions {
1144            format: RecordFormat::RDW,
1145            codepage: Codepage::ASCII,
1146            ..DecodeOptions::default()
1147        };
1148
1149        let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1150
1151        // Read all records
1152        let mut count = 0;
1153        for result in iterator.by_ref() {
1154            assert!(result.is_ok(), "Record {count} should decode successfully");
1155            count += 1;
1156        }
1157
1158        assert_eq!(count, 3);
1159        assert_eq!(iterator.current_record_index(), 3);
1160        assert!(iterator.is_eof());
1161    }
1162
1163    #[test]
1164    fn test_iter_records_convenience() {
1165        let copybook_text = r"
1166            01 RECORD.
1167               05 ID PIC 9(3).
1168               05 NAME PIC X(5).
1169        ";
1170
1171        let schema = parse_copybook(copybook_text).unwrap();
1172
1173        let test_data = b"001ALICE002BOB  ";
1174        let cursor = Cursor::new(test_data);
1175
1176        let options = DecodeOptions {
1177            format: RecordFormat::Fixed,
1178            ..DecodeOptions::default()
1179        };
1180
1181        let iterator = iter_records(cursor, &schema, &options).unwrap();
1182
1183        assert_eq!(iterator.current_record_index(), 0);
1184        assert!(!iterator.is_eof());
1185    }
1186
1187    #[test]
1188    fn test_iterator_with_empty_data() {
1189        let copybook_text = r"
1190            01 RECORD.
1191               05 ID PIC 9(3).
1192               05 NAME PIC X(5).
1193        ";
1194
1195        let mut schema = parse_copybook(copybook_text).unwrap();
1196        schema.lrecl_fixed = Some(8);
1197
1198        let test_data = b"";
1199        let cursor = Cursor::new(test_data);
1200
1201        let options = DecodeOptions {
1202            format: RecordFormat::Fixed,
1203            ..DecodeOptions::default()
1204        };
1205
1206        let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1207
1208        // Should immediately return None for empty data
1209        assert!(iterator.next().is_none());
1210        assert!(iterator.is_eof());
1211        assert_eq!(iterator.current_record_index(), 0);
1212    }
1213
1214    #[test]
1215    fn test_iterator_raw_record_eof() {
1216        let copybook_text = r"
1217            01 RECORD.
1218               05 ID PIC 9(3).
1219               05 NAME PIC X(5).
1220        ";
1221
1222        let schema = parse_copybook(copybook_text).unwrap();
1223
1224        let test_data = b"001ALICE";
1225        let cursor = Cursor::new(test_data);
1226
1227        let options = DecodeOptions {
1228            format: RecordFormat::Fixed,
1229            ..DecodeOptions::default()
1230        };
1231
1232        let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1233
1234        // Read first record
1235        assert!(iterator.read_raw_record().unwrap().is_some());
1236        assert_eq!(iterator.current_record_index(), 1);
1237
1238        // Read second record (should be None)
1239        assert!(iterator.read_raw_record().unwrap().is_none());
1240        assert!(iterator.is_eof());
1241    }
1242
1243    #[test]
1244    fn test_iterator_collect_results() {
1245        let copybook_text = r"
1246            01 RECORD.
1247               05 ID PIC 9(3).
1248               05 NAME PIC X(5).
1249        ";
1250
1251        let mut schema = parse_copybook(copybook_text).unwrap();
1252        schema.lrecl_fixed = Some(8);
1253
1254        let test_data = b"001ALICE002BOB  003CAROL";
1255        let cursor = Cursor::new(test_data);
1256
1257        let options = DecodeOptions {
1258            format: RecordFormat::Fixed,
1259            codepage: Codepage::ASCII,
1260            ..DecodeOptions::default()
1261        };
1262
1263        let iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1264
1265        // Collect all results
1266        let results: Vec<Result<Value>> = iterator.collect();
1267
1268        assert_eq!(results.len(), 3);
1269        for result in results {
1270            assert!(result.is_ok());
1271        }
1272    }
1273
1274    #[test]
1275    fn test_iterator_with_decode_error() {
1276        let copybook_text = r"
1277            01 RECORD.
1278               05 ID PIC 9(3).
1279               05 NAME PIC X(5).
1280        ";
1281
1282        let mut schema = parse_copybook(copybook_text).unwrap();
1283        schema.lrecl_fixed = Some(8);
1284
1285        // Create data that will decode successfully for first record
1286        let test_data = b"001ALICE";
1287        let cursor = Cursor::new(test_data);
1288
1289        let options = DecodeOptions {
1290            format: RecordFormat::Fixed,
1291            codepage: Codepage::ASCII,
1292            ..DecodeOptions::default()
1293        };
1294
1295        let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1296
1297        // First record should decode successfully
1298        let first = iterator.next();
1299        assert!(first.is_some());
1300        assert!(first.unwrap().is_ok());
1301
1302        // Second call should return None (EOF)
1303        assert!(iterator.next().is_none());
1304    }
1305
1306    #[derive(Default)]
1307    struct FailingReader {
1308        fail: bool,
1309    }
1310
1311    impl Read for FailingReader {
1312        fn read(&mut self, _buf: &mut [u8]) -> io::Result<usize> {
1313            if self.fail {
1314                Ok(0)
1315            } else {
1316                self.fail = true;
1317                Err(io::Error::other("forced read error"))
1318            }
1319        }
1320    }
1321
1322    #[test]
1323    fn test_iterator_fixed_format_read_error_code() {
1324        let copybook_text = r"
1325            01 RECORD.
1326               05 ID PIC 9(3).
1327               05 NAME PIC X(5).
1328        ";
1329
1330        let schema = parse_copybook(copybook_text).unwrap();
1331
1332        let mut schema = schema;
1333        schema.lrecl_fixed = Some(8);
1334
1335        let mut iterator =
1336            RecordIterator::new(FailingReader::default(), &schema, &DecodeOptions::default())
1337                .unwrap();
1338
1339        let error = iterator.read_raw_record().unwrap_err();
1340        assert_eq!(error.code, ErrorCode::CBKR101_FIXED_RECORD_ERROR);
1341    }
1342
1343    enum ReadStep {
1344        Bytes(Vec<u8>),
1345        Error(io::ErrorKind),
1346        Eof,
1347    }
1348
1349    struct ScriptedReader {
1350        steps: VecDeque<ReadStep>,
1351    }
1352
1353    impl ScriptedReader {
1354        fn new(steps: impl IntoIterator<Item = ReadStep>) -> Self {
1355            Self {
1356                steps: steps.into_iter().collect(),
1357            }
1358        }
1359    }
1360
1361    impl Read for ScriptedReader {
1362        fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
1363            match self.steps.pop_front() {
1364                Some(ReadStep::Bytes(mut bytes)) => {
1365                    let read = bytes.len().min(buf.len());
1366                    buf[..read].copy_from_slice(&bytes[..read]);
1367                    if read < bytes.len() {
1368                        bytes.drain(..read);
1369                        self.steps.push_front(ReadStep::Bytes(bytes));
1370                    }
1371                    Ok(read)
1372                }
1373                Some(ReadStep::Error(kind)) => Err(io::Error::new(kind, "scripted read error")),
1374                Some(ReadStep::Eof) | None => Ok(0),
1375            }
1376        }
1377    }
1378
1379    fn rdw_iterator(reader: ScriptedReader) -> RecordIterator<ScriptedReader> {
1380        let schema = eight_byte_schema();
1381        let options = DecodeOptions::default()
1382            .with_format(RecordFormat::RDW)
1383            .with_codepage(Codepage::ASCII);
1384        RecordIterator::new(reader, &schema, &options).unwrap()
1385    }
1386
1387    fn complete_rdw_record() -> Vec<u8> {
1388        let mut record = vec![0x00, 0x08, 0x00, 0x00];
1389        record.extend_from_slice(b"RECORD01");
1390        record
1391    }
1392
1393    fn assert_rdw_error(error: Error, code: ErrorCode, record_index: u64) {
1394        assert_eq!(error.code, code);
1395        assert_eq!(
1396            error.context.and_then(|context| context.record_index),
1397            Some(record_index)
1398        );
1399    }
1400
1401    #[test]
1402    fn rdw_clean_boundary_eof_returns_none() {
1403        let mut iterator = rdw_iterator(ScriptedReader::new([ReadStep::Eof]));
1404
1405        assert!(iterator.read_raw_record().unwrap().is_none());
1406        assert!(iterator.is_eof());
1407    }
1408
1409    #[test]
1410    fn rdw_partial_next_header_keeps_underflow_and_context() {
1411        let mut iterator = rdw_iterator(ScriptedReader::new([
1412            ReadStep::Bytes(complete_rdw_record()),
1413            ReadStep::Bytes(vec![0x00, 0x08]),
1414            ReadStep::Eof,
1415        ]));
1416
1417        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1418        assert_rdw_error(
1419            iterator.read_raw_record().unwrap_err(),
1420            ErrorCode::CBKF221_RDW_UNDERFLOW,
1421            2,
1422        );
1423    }
1424
1425    #[test]
1426    fn rdw_next_header_io_error_maps_to_read_error_with_context() {
1427        let mut iterator = rdw_iterator(ScriptedReader::new([
1428            ReadStep::Bytes(complete_rdw_record()),
1429            ReadStep::Error(io::ErrorKind::Other),
1430        ]));
1431
1432        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1433        assert_rdw_error(
1434            iterator.read_raw_record().unwrap_err(),
1435            ErrorCode::CBKR201_RDW_READ_ERROR,
1436            2,
1437        );
1438    }
1439
1440    #[test]
1441    fn rdw_partial_next_payload_keeps_underflow_and_context() {
1442        let mut iterator = rdw_iterator(ScriptedReader::new([
1443            ReadStep::Bytes(complete_rdw_record()),
1444            ReadStep::Bytes(vec![0x00, 0x08, 0x00, 0x00]),
1445            ReadStep::Bytes(b"ABC".to_vec()),
1446            ReadStep::Eof,
1447        ]));
1448
1449        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1450        assert_rdw_error(
1451            iterator.read_raw_record().unwrap_err(),
1452            ErrorCode::CBKF221_RDW_UNDERFLOW,
1453            2,
1454        );
1455    }
1456
1457    #[test]
1458    fn rdw_next_payload_io_error_maps_to_read_error_with_context() {
1459        let mut iterator = rdw_iterator(ScriptedReader::new([
1460            ReadStep::Bytes(complete_rdw_record()),
1461            ReadStep::Bytes(vec![0x00, 0x08, 0x00, 0x00]),
1462            ReadStep::Bytes(b"ABC".to_vec()),
1463            ReadStep::Error(io::ErrorKind::Other),
1464        ]));
1465
1466        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1467        assert_rdw_error(
1468            iterator.read_raw_record().unwrap_err(),
1469            ErrorCode::CBKR201_RDW_READ_ERROR,
1470            2,
1471        );
1472    }
1473
1474    /// Reader that hands out one byte per call, so `read_exact` cannot fill the
1475    /// buffer in a single call and the boundary logic is exercised for real.
1476    struct DribbleReader {
1477        data: Vec<u8>,
1478        position: usize,
1479    }
1480
1481    impl Read for DribbleReader {
1482        fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
1483            if self.position >= self.data.len() || buf.is_empty() {
1484                return Ok(0);
1485            }
1486            buf[0] = self.data[self.position];
1487            self.position += 1;
1488            Ok(1)
1489        }
1490    }
1491
1492    fn eight_byte_schema() -> Schema {
1493        let mut schema = parse_copybook("01 RECORD.\n   05 NAME PIC X(8).").unwrap();
1494        schema.lrecl_fixed = Some(8);
1495        schema
1496    }
1497
1498    #[test]
1499    fn fixed_trailing_partial_record_is_reported_not_discarded() {
1500        // Twelve bytes for an 8-byte LRECL: one whole record and a 4-byte tail.
1501        let data = b"RECORD01HALF".to_vec();
1502        let schema = eight_byte_schema();
1503        let options = DecodeOptions::default().with_codepage(Codepage::ASCII);
1504        let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options).unwrap();
1505
1506        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1507
1508        let error = iterator.read_raw_record().unwrap_err();
1509        assert_eq!(error.code, ErrorCode::CBKR101_FIXED_RECORD_ERROR);
1510        let context = error.context.expect("partial record reports context");
1511        assert_eq!(context.record_index, Some(2));
1512        assert!(
1513            context.details.unwrap_or_default().contains("4 of 8 bytes"),
1514            "details should name the short count"
1515        );
1516    }
1517
1518    #[test]
1519    fn fixed_partial_record_terminates_iteration_after_one_error() {
1520        let schema = eight_byte_schema();
1521        let options = DecodeOptions::default().with_codepage(Codepage::ASCII);
1522        let mut iterator =
1523            RecordIterator::new(Cursor::new(b"RECORD01HALF".to_vec()), &schema, &options).unwrap();
1524
1525        assert!(iterator.next().unwrap().is_ok());
1526        assert!(iterator.next().unwrap().is_err());
1527        assert!(
1528            iterator.next().is_none(),
1529            "iteration must stop after the partial-record error"
1530        );
1531    }
1532
1533    #[test]
1534    fn fixed_file_ending_on_a_record_boundary_is_still_clean_eof() {
1535        let schema = eight_byte_schema();
1536        let options = DecodeOptions::default().with_codepage(Codepage::ASCII);
1537        let mut iterator =
1538            RecordIterator::new(Cursor::new(b"RECORD01".to_vec()), &schema, &options).unwrap();
1539
1540        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1541        assert!(iterator.read_raw_record().unwrap().is_none());
1542        assert!(iterator.is_eof());
1543    }
1544
1545    #[test]
1546    fn fixed_partial_record_detected_when_reads_return_one_byte_at_a_time() {
1547        // A short `read` is not end of file. The boundary helper must keep
1548        // reading until the buffer is full or the reader is genuinely dry.
1549        let schema = eight_byte_schema();
1550        let options = DecodeOptions::default().with_codepage(Codepage::ASCII);
1551        let reader = DribbleReader {
1552            data: b"RECORD01HALF".to_vec(),
1553            position: 0,
1554        };
1555        let mut iterator = RecordIterator::new(reader, &schema, &options).unwrap();
1556
1557        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1558        assert_eq!(
1559            iterator.read_raw_record().unwrap_err().code,
1560            ErrorCode::CBKR101_FIXED_RECORD_ERROR
1561        );
1562    }
1563
1564    #[test]
1565    fn rdw_trailing_partial_header_is_reported_not_discarded() {
1566        // One 8-byte RDW record followed by a 2-byte stub of a second header.
1567        let mut data = vec![0x00, 0x08, 0x00, 0x00];
1568        data.extend_from_slice(b"RECORD01");
1569        data.extend_from_slice(&[0x00, 0x08]);
1570
1571        let schema = eight_byte_schema();
1572        let options = DecodeOptions::default()
1573            .with_format(RecordFormat::RDW)
1574            .with_codepage(Codepage::ASCII);
1575        let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options).unwrap();
1576
1577        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1578
1579        let error = iterator.read_raw_record().unwrap_err();
1580        assert_eq!(error.code, ErrorCode::CBKF221_RDW_UNDERFLOW);
1581    }
1582
1583    #[test]
1584    fn rdw_file_ending_on_a_record_boundary_is_still_clean_eof() {
1585        let mut data = vec![0x00, 0x08, 0x00, 0x00];
1586        data.extend_from_slice(b"RECORD01");
1587
1588        let schema = eight_byte_schema();
1589        let options = DecodeOptions::default()
1590            .with_format(RecordFormat::RDW)
1591            .with_codepage(Codepage::ASCII);
1592        let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options).unwrap();
1593
1594        assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1595        assert!(iterator.read_raw_record().unwrap().is_none());
1596        assert!(iterator.is_eof());
1597    }
1598}