copybook_codec/iterator.rs
1// SPDX-License-Identifier: AGPL-3.0-or-later
2//! Record iterator for streaming access to decoded records
3//!
4//! This module provides iterator-based access to records for programmatic processing,
5//! allowing users to process records one at a time without loading entire files into memory.
6//!
7//! # Overview
8//!
9//! The iterator module implements streaming record processing with bounded memory usage.
10//! It provides low-level iterator primitives for reading COBOL data files sequentially,
11//! supporting both fixed-length and RDW (Record Descriptor Word) variable-length formats.
12//!
13//! Key capabilities:
14//!
15//! 1. **Streaming iteration** ([`RecordIterator`]) - Process records one at a time
16//! 2. **Format flexibility** - Handle both fixed-length and RDW variable-length records
17//! 3. **Raw access** ([`RecordIterator::read_raw_record`]) - Access undecoded record bytes
18//! 4. **Convenience functions** ([`iter_records_from_file`], [`iter_records`]) - Simplified creation
19//!
20//! # Performance Characteristics
21//!
22//! The iterator uses buffered I/O and maintains bounded memory usage:
23//! - **Memory**: One record buffer (typically <32 KiB per record)
24//! - **Throughput**: Depends on decode complexity (DISPLAY vs COMP-3)
25//! - **Latency**: Sequential I/O optimized with `BufReader`
26//!
27//! For high-throughput parallel processing, consider using [`crate::decode_file_to_jsonl`]
28//! which provides parallel worker pools and streaming output.
29//!
30//! # Examples
31//!
32//! ## Basic Fixed-Length Record Iteration
33//!
34//! ```rust
35//! use copybook_codec::{iter_records_from_file, DecodeOptions, Codepage, RecordFormat};
36//! use copybook_core::parse_copybook;
37//!
38//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
39//! // Parse copybook schema
40//! let copybook_text = r#"
41//! 01 CUSTOMER-RECORD.
42//! 05 CUSTOMER-ID PIC 9(5).
43//! 05 CUSTOMER-NAME PIC X(20).
44//! 05 BALANCE PIC S9(7)V99 COMP-3.
45//! "#;
46//! let schema = parse_copybook(copybook_text)?;
47//!
48//! // Configure decoding options
49//! let options = DecodeOptions::new()
50//! .with_codepage(Codepage::CP037)
51//! .with_format(RecordFormat::Fixed);
52//!
53//! // Create iterator from file
54//! # #[cfg(not(test))]
55//! let iterator = iter_records_from_file("customers.bin", &schema, &options)?;
56//!
57//! // Process records one at a time
58//! # #[cfg(not(test))]
59//! for (index, result) in iterator.enumerate() {
60//! match result {
61//! Ok(json_value) => {
62//! println!("Record {}: {}", index + 1, json_value);
63//! }
64//! Err(error) => {
65//! eprintln!("Error in record {}: {}", index + 1, error);
66//! break; // Stop on first error
67//! }
68//! }
69//! }
70//! # Ok(())
71//! # }
72//! ```
73//!
74//! ## RDW Variable-Length Records
75//!
76//! ```rust
77//! use copybook_codec::{RecordIterator, DecodeOptions, RecordFormat};
78//! use copybook_core::parse_copybook;
79//! use std::fs::File;
80//!
81//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
82//! let copybook_text = r#"
83//! 01 TRANSACTION.
84//! 05 TRAN-ID PIC 9(10).
85//! 05 TRAN-AMOUNT PIC S9(9)V99 COMP-3.
86//! 05 TRAN-DESC PIC X(100).
87//! "#;
88//! let schema = parse_copybook(copybook_text)?;
89//!
90//! let options = DecodeOptions::new()
91//! .with_format(RecordFormat::RDW); // RDW variable-length format
92//!
93//! # #[cfg(not(test))]
94//! let file = File::open("transactions.dat")?;
95//! # #[cfg(test)]
96//! # let file = std::io::Cursor::new(vec![]);
97//! let mut iterator = RecordIterator::new(file, &schema, &options)?;
98//!
99//! // Process with error recovery
100//! let mut processed = 0;
101//! let mut errors = 0;
102//!
103//! for (index, result) in iterator.enumerate() {
104//! match result {
105//! Ok(json_value) => {
106//! processed += 1;
107//! // Process record...
108//! }
109//! Err(error) => {
110//! errors += 1;
111//! eprintln!("Record {}: {}", index + 1, error);
112//!
113//! if errors > 10 {
114//! eprintln!("Too many errors, stopping");
115//! break;
116//! }
117//! }
118//! }
119//! }
120//!
121//! println!("Processed: {}, Errors: {}", processed, errors);
122//! # Ok(())
123//! # }
124//! ```
125//!
126//! ## Raw Record Access (No Decoding)
127//!
128//! ```rust
129//! use copybook_codec::{RecordIterator, DecodeOptions, RecordFormat};
130//! use copybook_core::parse_copybook;
131//! use std::io::Cursor;
132//!
133//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
134//! let copybook_text = "01 RECORD.\n 05 DATA PIC X(10).";
135//! let schema = parse_copybook(copybook_text)?;
136//!
137//! let options = DecodeOptions::new()
138//! .with_format(RecordFormat::Fixed);
139//!
140//! let data = b"RECORD0001RECORD0002";
141//! let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options)?;
142//!
143//! // Read raw bytes without JSON decoding
144//! while let Some(raw_bytes) = iterator.read_raw_record()? {
145//! println!("Raw record {}: {} bytes",
146//! iterator.current_record_index(),
147//! raw_bytes.len());
148//!
149//! // Process raw bytes directly...
150//! // (useful for binary analysis, checksums, etc.)
151//! }
152//! # Ok(())
153//! # }
154//! ```
155//!
156//! ## Collecting Records into a Vec
157//!
158//! ```rust
159//! use copybook_codec::{iter_records, DecodeOptions};
160//! use copybook_core::parse_copybook;
161//! use serde_json::Value;
162//! use std::io::Cursor;
163//!
164//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
165//! let copybook_text = "01 RECORD.\n 05 ID PIC 9(5).";
166//! let schema = parse_copybook(copybook_text)?;
167//! let options = DecodeOptions::default();
168//!
169//! let data = b"0000100002";
170//! let iterator = iter_records(Cursor::new(data), &schema, &options)?;
171//!
172//! // Collect all successful records
173//! let records: Vec<Value> = iterator
174//! .filter_map(Result::ok) // Skip errors
175//! .collect();
176//!
177//! println!("Collected {} records", records.len());
178//! # Ok(())
179//! # }
180//! ```
181//!
182//! ## Using with `DecodeOptions` and Metadata
183//!
184//! ```rust
185//! use copybook_codec::{iter_records_from_file, DecodeOptions, Codepage, JsonNumberMode};
186//! use copybook_core::parse_copybook;
187//!
188//! # fn example() -> Result<(), Box<dyn std::error::Error>> {
189//! let copybook_text = r#"
190//! 01 RECORD.
191//! 05 AMOUNT PIC S9(9)V99 COMP-3.
192//! "#;
193//! let schema = parse_copybook(copybook_text)?;
194//!
195//! // Configure with lossless numbers and metadata
196//! let options = DecodeOptions::new()
197//! .with_codepage(Codepage::CP037)
198//! .with_json_number_mode(JsonNumberMode::Lossless)
199//! .with_emit_meta(true); // Include field metadata
200//!
201//! # #[cfg(not(test))]
202//! let iterator = iter_records_from_file("data.bin", &schema, &options)?;
203//!
204//! # #[cfg(not(test))]
205//! for result in iterator {
206//! let json_value = result?;
207//! // JSON includes metadata: {"AMOUNT": "123.45", "_meta": {...}}
208//! println!("{}", serde_json::to_string_pretty(&json_value)?);
209//! }
210//! # Ok(())
211//! # }
212//! ```
213
214use crate::lib_api::decode_record_with_raw_data;
215use crate::options::{DecodeOptions, RecordFormat};
216use copybook_core::{Error, ErrorCode, ErrorContext, Result, Schema};
217use copybook_rdw::RdwHeader;
218use serde_json::Value;
219use std::io::{BufReader, Read};
220
221/// Outcome of reading a whole record-sized unit at a record boundary.
222enum BoundaryRead {
223 /// The buffer was filled.
224 Complete,
225 /// The reader was already exhausted; nothing was consumed.
226 CleanEof,
227 /// The reader ran out mid-unit after this many bytes.
228 Partial(usize),
229}
230
231/// Fill `buffer` from `reader`, distinguishing a clean end of file from a
232/// truncated final unit.
233///
234/// [`Read::read_exact`] reports [`std::io::ErrorKind::UnexpectedEof`] for both
235/// cases and consumes the bytes it did read, so it cannot tell "the file ended
236/// on a record boundary" from "the file ends with a partial record". Treating
237/// the two alike silently discards the trailing bytes.
238fn fill_at_record_boundary<R: Read>(
239 reader: &mut R,
240 buffer: &mut [u8],
241) -> std::io::Result<BoundaryRead> {
242 let mut filled = 0;
243 while filled < buffer.len() {
244 match reader.read(&mut buffer[filled..]) {
245 Ok(0) => break,
246 Ok(read) => filled += read,
247 Err(e) if e.kind() == std::io::ErrorKind::Interrupted => {}
248 Err(e) => return Err(e),
249 }
250 }
251
252 if filled == buffer.len() {
253 Ok(BoundaryRead::Complete)
254 } else if filled == 0 {
255 Ok(BoundaryRead::CleanEof)
256 } else {
257 Ok(BoundaryRead::Partial(filled))
258 }
259}
260
261/// Iterator over records in a data file, yielding decoded JSON values
262///
263/// This iterator provides streaming access to records, processing them one at a time
264/// to maintain bounded memory usage even for very large files.
265///
266/// # Examples
267///
268/// ```rust,no_run
269/// use copybook_codec::{RecordIterator, DecodeOptions};
270/// use copybook_core::parse_copybook;
271/// # use std::io::Cursor;
272///
273/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
274/// let copybook_text = "01 RECORD.\n 05 ID PIC 9(5).\n 05 NAME PIC X(20).";
275/// let mut schema = parse_copybook(copybook_text)?;
276/// schema.lrecl_fixed = Some(25);
277/// let options = DecodeOptions::default();
278/// # let record_bytes = b"00001ALICE ";
279/// # let file = Cursor::new(&record_bytes[..]);
280/// // let file = std::fs::File::open("data.bin")?;
281///
282/// let mut iterator = RecordIterator::new(file, &schema, &options)?;
283///
284/// for (record_index, result) in iterator.enumerate() {
285/// match result {
286/// Ok(json_value) => {
287/// println!("Record {}: {}", record_index + 1, json_value);
288/// }
289/// Err(error) => {
290/// eprintln!("Error in record {}: {}", record_index + 1, error);
291/// }
292/// }
293/// }
294/// # Ok(())
295/// # }
296/// ```
297pub struct RecordIterator<R: Read> {
298 /// The buffered reader
299 reader: BufReader<R>,
300 /// The schema for decoding records
301 schema: Schema,
302 /// Decoding options
303 options: DecodeOptions,
304 /// Current record index (1-based)
305 record_index: u64,
306 /// Whether the iterator has reached EOF
307 eof_reached: bool,
308 /// Buffer for reading record data
309 buffer: Vec<u8>,
310 /// Complete RDW frame for `RawMode::RecordRDW` envelope capture.
311 raw_data_with_header: Option<Vec<u8>>,
312}
313
314impl<R: Read> RecordIterator<R> {
315 /// Create a new record iterator
316 ///
317 /// # Arguments
318 ///
319 /// * `reader` - The input stream to read from
320 /// * `schema` - The parsed copybook schema
321 /// * `options` - Decoding options
322 ///
323 /// # Errors
324 /// Returns an error if the record format is incompatible with the schema.
325 #[inline]
326 #[must_use = "Handle the Result or propagate the error"]
327 pub fn new(reader: R, schema: &Schema, options: &DecodeOptions) -> Result<Self> {
328 Ok(Self {
329 reader: BufReader::new(reader),
330 schema: schema.clone(),
331 options: options.clone(),
332 record_index: 0,
333 eof_reached: false,
334 buffer: Vec::new(),
335 raw_data_with_header: None,
336 })
337 }
338
339 /// Get the current record index (1-based)
340 ///
341 /// This returns the index of the last record that was successfully read,
342 /// or 0 if no records have been read yet.
343 #[inline]
344 #[must_use]
345 pub fn current_record_index(&self) -> u64 {
346 self.record_index
347 }
348
349 /// Check if the iterator has reached the end of the file
350 #[inline]
351 #[must_use]
352 pub fn is_eof(&self) -> bool {
353 self.eof_reached
354 }
355
356 /// Get a reference to the schema being used
357 #[inline]
358 #[must_use]
359 pub fn schema(&self) -> &Schema {
360 &self.schema
361 }
362
363 /// Get a reference to the decode options being used
364 #[inline]
365 #[must_use]
366 pub fn options(&self) -> &DecodeOptions {
367 &self.options
368 }
369
370 fn next_record_context(&self, details: String) -> ErrorContext {
371 ErrorContext {
372 record_index: Some(self.record_index + 1),
373 field_path: None,
374 byte_offset: None,
375 line_number: None,
376 details: Some(details),
377 }
378 }
379
380 fn rdw_header_read_error(&self, error: &std::io::Error) -> Error {
381 Error::new(
382 ErrorCode::CBKR201_RDW_READ_ERROR,
383 format!("Failed to read RDW header: {error}"),
384 )
385 .with_context(self.next_record_context("I/O failure while reading RDW header".to_string()))
386 }
387
388 fn rdw_payload_read_error(&self, error: &std::io::Error, length: usize) -> Error {
389 let (code, details) = if error.kind() == std::io::ErrorKind::UnexpectedEof {
390 (
391 ErrorCode::CBKF221_RDW_UNDERFLOW,
392 format!("File ends before the declared {length}-byte RDW payload"),
393 )
394 } else {
395 (
396 ErrorCode::CBKR201_RDW_READ_ERROR,
397 "I/O failure while reading RDW payload".to_string(),
398 )
399 };
400 Error::new(code, format!("Failed to read RDW payload: {error}"))
401 .with_context(self.next_record_context(details))
402 }
403
404 /// Read the next record without decoding it
405 ///
406 /// This method reads the raw bytes of the next record without performing
407 /// JSON decoding. Useful for applications that need access to raw record data
408 /// for binary analysis, checksums, or custom processing.
409 ///
410 /// # Returns
411 ///
412 /// * `Ok(Some(bytes))` - The raw record bytes
413 /// * `Ok(None)` - End of file reached
414 /// * `Err(error)` - An error occurred while reading
415 ///
416 /// # Errors
417 /// Returns an error if underlying I/O operations fail or the record format is invalid.
418 ///
419 /// # Examples
420 ///
421 /// ```rust
422 /// use copybook_codec::{RecordIterator, DecodeOptions, RecordFormat};
423 /// use copybook_core::parse_copybook;
424 /// use std::io::Cursor;
425 ///
426 /// # fn example() -> Result<(), Box<dyn std::error::Error>> {
427 /// let copybook_text = "01 RECORD.\n 05 DATA PIC X(8).";
428 /// let schema = parse_copybook(copybook_text)?;
429 ///
430 /// let options = DecodeOptions::new()
431 /// .with_format(RecordFormat::Fixed);
432 ///
433 /// let data = b"RECORD01RECORD02";
434 /// let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options)?;
435 ///
436 /// // Read raw bytes
437 /// if let Some(raw_bytes) = iterator.read_raw_record()? {
438 /// assert_eq!(raw_bytes, b"RECORD01");
439 /// assert_eq!(iterator.current_record_index(), 1);
440 /// }
441 ///
442 /// if let Some(raw_bytes) = iterator.read_raw_record()? {
443 /// assert_eq!(raw_bytes, b"RECORD02");
444 /// assert_eq!(iterator.current_record_index(), 2);
445 /// }
446 ///
447 /// // End of file
448 /// assert!(iterator.read_raw_record()?.is_none());
449 /// assert!(iterator.is_eof());
450 /// # Ok(())
451 /// # }
452 /// ```
453 #[inline]
454 #[must_use = "Handle the Result or propagate the error"]
455 pub fn read_raw_record(&mut self) -> Result<Option<Vec<u8>>> {
456 if self.eof_reached {
457 return Ok(None);
458 }
459
460 self.buffer.clear();
461 self.raw_data_with_header = None;
462
463 let record_data = match self.options.format {
464 RecordFormat::Fixed => {
465 let lrecl = crate::file::fixed::lrecl(&self.schema)? as usize;
466 self.buffer.resize(lrecl, 0);
467
468 match fill_at_record_boundary(&mut self.reader, &mut self.buffer) {
469 Ok(BoundaryRead::Complete) => {
470 self.record_index += 1;
471 Some(self.buffer.clone())
472 }
473 Ok(BoundaryRead::CleanEof) => {
474 self.eof_reached = true;
475 return Ok(None);
476 }
477 Ok(BoundaryRead::Partial(read)) => {
478 self.eof_reached = true;
479 return Err(Error::new(
480 ErrorCode::CBKR101_FIXED_RECORD_ERROR,
481 format!("Incomplete record at end of file: expected {lrecl} bytes"),
482 )
483 .with_context(ErrorContext {
484 record_index: Some(self.record_index + 1),
485 field_path: None,
486 byte_offset: None,
487 line_number: None,
488 details: Some(format!(
489 "File ends with partial record ({read} of {lrecl} bytes)"
490 )),
491 }));
492 }
493 Err(e) => {
494 return Err(Error::new(
495 ErrorCode::CBKR101_FIXED_RECORD_ERROR,
496 format!("Failed to read fixed record: {e}"),
497 ));
498 }
499 }
500 }
501 RecordFormat::RDW => {
502 // Read RDW header
503 let mut rdw_header = [0u8; 4];
504 match fill_at_record_boundary(&mut self.reader, &mut rdw_header) {
505 Ok(BoundaryRead::Complete) => {}
506 Ok(BoundaryRead::CleanEof) => {
507 self.eof_reached = true;
508 return Ok(None);
509 }
510 Ok(BoundaryRead::Partial(read)) => {
511 self.eof_reached = true;
512 return Err(Error::new(
513 ErrorCode::CBKF221_RDW_UNDERFLOW,
514 "Incomplete RDW header at end of file: expected 4 bytes".to_string(),
515 )
516 .with_context(ErrorContext {
517 record_index: Some(self.record_index + 1),
518 field_path: None,
519 byte_offset: None,
520 line_number: None,
521 details: Some(format!(
522 "File ends with partial RDW header ({read} of 4 bytes)"
523 )),
524 }));
525 }
526 Err(e) => {
527 return Err(self.rdw_header_read_error(&e));
528 }
529 }
530
531 // Parse length (payload bytes only)
532 let length = usize::from(RdwHeader::from_bytes(rdw_header).length());
533
534 // Read payload
535 self.buffer.resize(length, 0);
536 match self.reader.read_exact(&mut self.buffer) {
537 Ok(()) => {
538 let mut raw_data_with_header = Vec::with_capacity(4 + length);
539 raw_data_with_header.extend_from_slice(&rdw_header);
540 raw_data_with_header.extend_from_slice(&self.buffer);
541 self.raw_data_with_header = Some(raw_data_with_header);
542 self.record_index += 1;
543 Some(self.buffer.clone())
544 }
545 Err(e) => {
546 return Err(self.rdw_payload_read_error(&e, length));
547 }
548 }
549 }
550 };
551
552 Ok(record_data)
553 }
554
555 /// Decode the next record to JSON
556 ///
557 /// This is the main method used by the Iterator implementation.
558 /// It reads and decodes the next record in one operation.
559 #[inline]
560 fn decode_next_record(&mut self) -> Result<Option<Value>> {
561 match self.read_raw_record()? {
562 Some(record_bytes) => {
563 let raw_data_with_header = self.raw_data_with_header.take();
564 let json_value = decode_record_with_raw_data(
565 &self.schema,
566 &record_bytes,
567 &self.options,
568 raw_data_with_header.as_deref(),
569 self.record_index,
570 )?;
571 Ok(Some(json_value))
572 }
573 None => Ok(None),
574 }
575 }
576}
577
578impl<R: Read> Iterator for RecordIterator<R> {
579 type Item = Result<Value>;
580
581 #[inline]
582 fn next(&mut self) -> Option<Self::Item> {
583 if self.eof_reached {
584 return None;
585 }
586
587 match self.decode_next_record() {
588 Ok(Some(value)) => Some(Ok(value)),
589 Ok(None) => {
590 self.eof_reached = true;
591 None
592 }
593 Err(error) => {
594 // On error, we still advance the record index if we were able to read something
595 Some(Err(error))
596 }
597 }
598 }
599}
600
601/// Convenience function to create a record iterator from a file path
602///
603/// This is the most common way to create an iterator for processing COBOL data files.
604/// It handles file opening and iterator creation in a single call.
605///
606/// # Arguments
607///
608/// * `file_path` - Path to the data file
609/// * `schema` - The parsed copybook schema
610/// * `options` - Decoding options
611///
612/// # Errors
613/// Returns an error if the file cannot be opened or the iterator cannot be created.
614///
615/// # Examples
616///
617/// ## Basic Usage with Fixed-Length Records
618///
619/// ```rust,no_run
620/// use copybook_codec::{iter_records_from_file, DecodeOptions, Codepage, RecordFormat};
621/// use copybook_core::parse_copybook;
622///
623/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
624/// let copybook_text = r#"
625/// 01 EMPLOYEE-RECORD.
626/// 05 EMP-ID PIC 9(6).
627/// 05 EMP-NAME PIC X(30).
628/// 05 EMP-SALARY PIC S9(7)V99 COMP-3.
629/// "#;
630/// let schema = parse_copybook(copybook_text)?;
631///
632/// let options = DecodeOptions::new()
633/// .with_codepage(Codepage::CP037)
634/// .with_format(RecordFormat::Fixed);
635///
636/// let iterator = iter_records_from_file("employees.dat", &schema, &options)?;
637///
638/// for (index, result) in iterator.enumerate() {
639/// match result {
640/// Ok(employee) => println!("Employee {}: {}", index + 1, employee),
641/// Err(e) => eprintln!("Error at record {}: {}", index + 1, e),
642/// }
643/// }
644/// # Ok(())
645/// # }
646/// ```
647///
648/// ## Processing with Error Limits
649///
650/// ```rust,no_run
651/// use copybook_codec::{iter_records_from_file, DecodeOptions};
652/// use copybook_core::parse_copybook;
653///
654/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
655/// # let schema = parse_copybook("01 R.\n 05 F PIC X(1).")?;
656/// # let options = DecodeOptions::default();
657/// let iterator = iter_records_from_file("data.bin", &schema, &options)?;
658///
659/// let mut success_count = 0;
660/// let mut error_count = 0;
661/// const MAX_ERRORS: usize = 100;
662///
663/// for result in iterator {
664/// match result {
665/// Ok(_) => success_count += 1,
666/// Err(e) => {
667/// error_count += 1;
668/// eprintln!("Error: {}", e);
669///
670/// if error_count >= MAX_ERRORS {
671/// eprintln!("Too many errors, aborting");
672/// break;
673/// }
674/// }
675/// }
676/// }
677///
678/// println!("Success: {}, Errors: {}", success_count, error_count);
679/// # Ok(())
680/// # }
681/// ```
682#[inline]
683#[must_use = "Handle the Result or propagate the error"]
684pub fn iter_records_from_file<P: AsRef<std::path::Path>>(
685 file_path: P,
686 schema: &Schema,
687 options: &DecodeOptions,
688) -> Result<RecordIterator<std::fs::File>> {
689 let file = std::fs::File::open(file_path).map_err(|e| {
690 Error::new(
691 ErrorCode::CBKR201_RDW_READ_ERROR,
692 format!("failed to open input file: {e}"),
693 )
694 })?;
695
696 RecordIterator::new(file, schema, options)
697}
698
699/// Convenience function to create a record iterator from any readable source
700///
701/// This function provides maximum flexibility by accepting any type that implements
702/// the `Read` trait, including files, cursors, network streams, or custom readers.
703///
704/// # Arguments
705///
706/// * `reader` - Any type implementing Read (File, Cursor, `TcpStream`, etc.)
707/// * `schema` - The parsed copybook schema
708/// * `options` - Decoding options
709///
710/// # Errors
711/// Returns an error if the iterator cannot be created.
712///
713/// # Examples
714///
715/// ## Using with In-Memory Data (Cursor)
716///
717/// ```rust
718/// use copybook_codec::{iter_records, DecodeOptions, RecordFormat};
719/// use copybook_core::parse_copybook;
720/// use std::io::Cursor;
721///
722/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
723/// let copybook_text = "01 RECORD.\n 05 ID PIC 9(3).\n 05 NAME PIC X(5).";
724/// let schema = parse_copybook(copybook_text)?;
725///
726/// let options = DecodeOptions::new()
727/// .with_format(RecordFormat::Fixed);
728///
729/// // Create iterator from in-memory data
730/// let data = b"001ALICE002BOB 003CAROL";
731/// let iterator = iter_records(Cursor::new(data), &schema, &options)?;
732///
733/// let records: Vec<_> = iterator.collect::<Result<Vec<_>, _>>()?;
734/// assert_eq!(records.len(), 3);
735/// # Ok(())
736/// # }
737/// ```
738///
739/// ## Using with File
740///
741/// ```rust,no_run
742/// use copybook_codec::{iter_records, DecodeOptions};
743/// use copybook_core::parse_copybook;
744/// use std::fs::File;
745///
746/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
747/// let schema = parse_copybook("01 RECORD.\n 05 DATA PIC X(10).")?;
748/// let options = DecodeOptions::default();
749///
750/// let file = File::open("data.bin")?;
751/// let iterator = iter_records(file, &schema, &options)?;
752///
753/// for result in iterator {
754/// let record = result?;
755/// println!("{}", record);
756/// }
757/// # Ok(())
758/// # }
759/// ```
760///
761/// ## Using with Compressed Data
762///
763/// ```text
764/// use copybook_codec::{iter_records, DecodeOptions};
765/// use copybook_core::parse_copybook;
766/// use std::fs::File;
767/// use flate2::read::GzDecoder;
768///
769/// # fn example() -> Result<(), Box<dyn std::error::Error>> {
770/// let schema = parse_copybook("01 RECORD.\n 05 DATA PIC X(10).")?;
771/// let options = DecodeOptions::default();
772///
773/// // Read from gzipped file
774/// let file = File::open("data.bin.gz")?;
775/// let decoder = GzDecoder::new(file);
776/// let iterator = iter_records(decoder, &schema, &options)?;
777///
778/// for result in iterator {
779/// let record = result?;
780/// // Process decompressed record...
781/// }
782/// # Ok(())
783/// # }
784/// ```
785#[inline]
786#[must_use = "Handle the Result or propagate the error"]
787pub fn iter_records<R: Read>(
788 reader: R,
789 schema: &Schema,
790 options: &DecodeOptions,
791) -> Result<RecordIterator<R>> {
792 RecordIterator::new(reader, schema, options)
793}
794
795#[cfg(test)]
796#[allow(clippy::expect_used)]
797#[allow(clippy::unwrap_used)]
798#[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)]
799mod tests {
800 use super::*;
801 use crate::Codepage;
802 use copybook_core::parse_copybook;
803 use std::collections::VecDeque;
804 use std::io::{self, Cursor, Read};
805
806 #[test]
807 fn test_record_iterator_basic() {
808 let copybook_text = r"
809 01 RECORD.
810 05 ID PIC 9(3).
811 05 NAME PIC X(5).
812 ";
813
814 let schema = parse_copybook(copybook_text).unwrap();
815
816 // Create test data: two 8-byte fixed records
817 let test_data = b"001ALICE002BOB ";
818 let cursor = Cursor::new(test_data);
819
820 let options = DecodeOptions {
821 format: RecordFormat::Fixed,
822 ..DecodeOptions::default()
823 };
824
825 let iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
826
827 // Just test that the iterator can be created successfully
828 assert_eq!(iterator.current_record_index(), 0);
829 assert!(!iterator.is_eof());
830 }
831
832 #[test]
833 fn test_record_iterator_rdw() {
834 let copybook_text = r"
835 01 RECORD.
836 05 ID PIC 9(3).
837 05 NAME PIC X(5).
838 ";
839
840 let schema = parse_copybook(copybook_text).unwrap();
841
842 // Create RDW test data:
843 // Record 1: length=8, reserved=0, data="001ALICE"
844 // Record 2: length=6, reserved=0, data="002BOB"
845 let test_data = vec![
846 0x00, 0x08, 0x00, 0x00, // RDW header: length=8, reserved=0
847 b'0', b'0', b'1', b'A', b'L', b'I', b'C', b'E', // Record 1 data
848 0x00, 0x06, 0x00, 0x00, // RDW header: length=6, reserved=0
849 b'0', b'0', b'2', b'B', b'O', b'B', // Record 2 data
850 ];
851
852 let cursor = Cursor::new(test_data);
853
854 let options = DecodeOptions {
855 format: RecordFormat::RDW,
856 ..DecodeOptions::default()
857 };
858
859 let iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
860
861 // Just test that the iterator can be created successfully
862 assert_eq!(iterator.current_record_index(), 0);
863 assert!(!iterator.is_eof());
864 }
865
866 #[test]
867 fn test_raw_record_reading() {
868 let copybook_text = r"
869 01 RECORD.
870 05 ID PIC 9(3).
871 05 NAME PIC X(5).
872 ";
873
874 let schema = parse_copybook(copybook_text).unwrap();
875
876 let test_data = b"001ALICE";
877 let cursor = Cursor::new(test_data);
878
879 let options = DecodeOptions {
880 format: RecordFormat::Fixed,
881 ..DecodeOptions::default()
882 };
883
884 let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
885
886 // Read raw record
887 let raw_record = iterator.read_raw_record().unwrap().unwrap();
888 assert_eq!(raw_record, b"001ALICE");
889 assert_eq!(iterator.current_record_index(), 1);
890
891 // End of file
892 assert!(iterator.read_raw_record().unwrap().is_none());
893 }
894
895 #[test]
896 fn test_iterator_error_handling() {
897 let copybook_text = r"
898 01 RECORD.
899 05 ID PIC 9(3).
900 05 NAME PIC X(5).
901 ";
902
903 let schema = parse_copybook(copybook_text).unwrap();
904
905 // Create incomplete record (only 4 bytes instead of 8)
906 let test_data = b"001A";
907 let cursor = Cursor::new(test_data);
908
909 let options = DecodeOptions {
910 format: RecordFormat::Fixed,
911 ..DecodeOptions::default()
912 };
913
914 let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
915
916 // A truncated final record is reported, not silently dropped: returning
917 // EOF here would discard the four trailing bytes with no signal to the
918 // caller.
919 let error = iterator
920 .next()
921 .expect("truncated data yields an item")
922 .expect_err("truncated data is an error");
923 assert_eq!(error.code, ErrorCode::CBKR101_FIXED_RECORD_ERROR);
924 assert!(iterator.next().is_none());
925 }
926
927 #[test]
928 fn test_iterator_fixed_format_missing_lrecl_errors_on_next() {
929 // A schema without a fixed record length
930 let copybook_text = "01 SOME-GROUP. 05 SOME-FIELD PIC X(1).";
931 let mut schema = parse_copybook(copybook_text).unwrap();
932 schema.lrecl_fixed = None; // Ensure it's None
933
934 let test_data = b"";
935 let cursor = Cursor::new(test_data);
936
937 let options = DecodeOptions {
938 format: RecordFormat::Fixed,
939 ..DecodeOptions::default()
940 };
941
942 let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
943
944 let first = iterator.next().unwrap();
945 assert!(first.is_err());
946 if let Err(e) = first {
947 assert_eq!(e.code, ErrorCode::CBKI001_INVALID_STATE);
948 assert_eq!(e.message, crate::file::fixed::FIXED_FORMAT_LRECL_MISSING);
949 }
950 }
951
952 #[test]
953 fn test_iterator_fixed_format_zero_lrecl_errors_on_next() {
954 let copybook_text = "01 SOME-GROUP. 05 SOME-FIELD PIC X(1).";
955 let mut schema = parse_copybook(copybook_text).unwrap();
956 schema.lrecl_fixed = Some(0);
957
958 let options = DecodeOptions {
959 format: RecordFormat::Fixed,
960 ..DecodeOptions::default()
961 };
962
963 let mut iterator = RecordIterator::new(Cursor::new(b"DATA"), &schema, &options).unwrap();
964
965 let error = iterator.next().unwrap().unwrap_err();
966 assert_eq!(error.code, ErrorCode::CBKI001_INVALID_STATE);
967 assert_eq!(error.message, "LRECL must be greater than zero");
968 }
969
970 #[test]
971 fn test_iterator_schema_and_options_accessors() {
972 let copybook_text = r"
973 01 RECORD.
974 05 ID PIC 9(3).
975 05 NAME PIC X(5).
976 ";
977
978 let mut schema = parse_copybook(copybook_text).unwrap();
979 schema.lrecl_fixed = Some(8);
980 let test_data = b"001ALICE";
981 let cursor = Cursor::new(test_data);
982
983 let options = DecodeOptions {
984 format: RecordFormat::Fixed,
985 codepage: Codepage::ASCII,
986 ..DecodeOptions::default()
987 };
988
989 let iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
990
991 // Test schema accessor
992 assert_eq!(iterator.schema().fields[0].name, "RECORD");
993
994 // Test options accessor
995 assert_eq!(iterator.options().format, RecordFormat::Fixed);
996 }
997
998 #[test]
999 fn test_iterator_multiple_fixed_records() {
1000 let copybook_text = r"
1001 01 RECORD.
1002 05 ID PIC 9(3).
1003 05 NAME PIC X(5).
1004 ";
1005
1006 let mut schema = parse_copybook(copybook_text).unwrap();
1007 schema.lrecl_fixed = Some(8);
1008
1009 // Create test data: three 8-byte fixed records
1010 let test_data = b"001ALICE002BOB 003CAROL";
1011 let cursor = Cursor::new(test_data);
1012
1013 let options = DecodeOptions {
1014 format: RecordFormat::Fixed,
1015 codepage: Codepage::ASCII,
1016 ..DecodeOptions::default()
1017 };
1018
1019 let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1020
1021 // Read all records
1022 let mut count = 0;
1023 for result in iterator.by_ref() {
1024 assert!(result.is_ok(), "Record {count} should decode successfully");
1025 count += 1;
1026 }
1027
1028 assert_eq!(count, 3);
1029 assert_eq!(iterator.current_record_index(), 3);
1030 assert!(iterator.is_eof());
1031 }
1032
1033 #[test]
1034 fn test_iterator_rdw_multiple_records() {
1035 let copybook_text = r"
1036 01 RECORD.
1037 05 ID PIC 9(3).
1038 05 NAME PIC X(5).
1039 ";
1040
1041 let schema = parse_copybook(copybook_text).unwrap();
1042
1043 // Create RDW test data with three records
1044 let test_data = vec![
1045 // Record 1
1046 0x00, 0x08, 0x00, 0x00, // RDW header: length=8
1047 b'0', b'0', b'1', b'A', b'L', b'I', b'C', b'E', // Record 2
1048 0x00, 0x06, 0x00, 0x00, // RDW header: length=6
1049 b'0', b'0', b'2', b'B', b'O', b'B', // Record 3
1050 0x00, 0x08, 0x00, 0x00, // RDW header: length=8
1051 b'0', b'0', b'3', b'C', b'A', b'R', b'O', b'L',
1052 ];
1053
1054 let cursor = Cursor::new(test_data);
1055
1056 let options = DecodeOptions {
1057 format: RecordFormat::RDW,
1058 codepage: Codepage::ASCII,
1059 ..DecodeOptions::default()
1060 };
1061
1062 let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1063
1064 // Read all records
1065 let mut count = 0;
1066 for result in iterator.by_ref() {
1067 assert!(result.is_ok(), "Record {count} should decode successfully");
1068 count += 1;
1069 }
1070
1071 assert_eq!(count, 3);
1072 assert_eq!(iterator.current_record_index(), 3);
1073 assert!(iterator.is_eof());
1074 }
1075
1076 #[test]
1077 fn test_iter_records_convenience() {
1078 let copybook_text = r"
1079 01 RECORD.
1080 05 ID PIC 9(3).
1081 05 NAME PIC X(5).
1082 ";
1083
1084 let schema = parse_copybook(copybook_text).unwrap();
1085
1086 let test_data = b"001ALICE002BOB ";
1087 let cursor = Cursor::new(test_data);
1088
1089 let options = DecodeOptions {
1090 format: RecordFormat::Fixed,
1091 ..DecodeOptions::default()
1092 };
1093
1094 let iterator = iter_records(cursor, &schema, &options).unwrap();
1095
1096 assert_eq!(iterator.current_record_index(), 0);
1097 assert!(!iterator.is_eof());
1098 }
1099
1100 #[test]
1101 fn test_iterator_with_empty_data() {
1102 let copybook_text = r"
1103 01 RECORD.
1104 05 ID PIC 9(3).
1105 05 NAME PIC X(5).
1106 ";
1107
1108 let mut schema = parse_copybook(copybook_text).unwrap();
1109 schema.lrecl_fixed = Some(8);
1110
1111 let test_data = b"";
1112 let cursor = Cursor::new(test_data);
1113
1114 let options = DecodeOptions {
1115 format: RecordFormat::Fixed,
1116 ..DecodeOptions::default()
1117 };
1118
1119 let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1120
1121 // Should immediately return None for empty data
1122 assert!(iterator.next().is_none());
1123 assert!(iterator.is_eof());
1124 assert_eq!(iterator.current_record_index(), 0);
1125 }
1126
1127 #[test]
1128 fn test_iterator_raw_record_eof() {
1129 let copybook_text = r"
1130 01 RECORD.
1131 05 ID PIC 9(3).
1132 05 NAME PIC X(5).
1133 ";
1134
1135 let schema = parse_copybook(copybook_text).unwrap();
1136
1137 let test_data = b"001ALICE";
1138 let cursor = Cursor::new(test_data);
1139
1140 let options = DecodeOptions {
1141 format: RecordFormat::Fixed,
1142 ..DecodeOptions::default()
1143 };
1144
1145 let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1146
1147 // Read first record
1148 assert!(iterator.read_raw_record().unwrap().is_some());
1149 assert_eq!(iterator.current_record_index(), 1);
1150
1151 // Read second record (should be None)
1152 assert!(iterator.read_raw_record().unwrap().is_none());
1153 assert!(iterator.is_eof());
1154 }
1155
1156 #[test]
1157 fn test_iterator_collect_results() {
1158 let copybook_text = r"
1159 01 RECORD.
1160 05 ID PIC 9(3).
1161 05 NAME PIC X(5).
1162 ";
1163
1164 let mut schema = parse_copybook(copybook_text).unwrap();
1165 schema.lrecl_fixed = Some(8);
1166
1167 let test_data = b"001ALICE002BOB 003CAROL";
1168 let cursor = Cursor::new(test_data);
1169
1170 let options = DecodeOptions {
1171 format: RecordFormat::Fixed,
1172 codepage: Codepage::ASCII,
1173 ..DecodeOptions::default()
1174 };
1175
1176 let iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1177
1178 // Collect all results
1179 let results: Vec<Result<Value>> = iterator.collect();
1180
1181 assert_eq!(results.len(), 3);
1182 for result in results {
1183 assert!(result.is_ok());
1184 }
1185 }
1186
1187 #[test]
1188 fn test_iterator_with_decode_error() {
1189 let copybook_text = r"
1190 01 RECORD.
1191 05 ID PIC 9(3).
1192 05 NAME PIC X(5).
1193 ";
1194
1195 let mut schema = parse_copybook(copybook_text).unwrap();
1196 schema.lrecl_fixed = Some(8);
1197
1198 // Create data that will decode successfully for first record
1199 let test_data = b"001ALICE";
1200 let cursor = Cursor::new(test_data);
1201
1202 let options = DecodeOptions {
1203 format: RecordFormat::Fixed,
1204 codepage: Codepage::ASCII,
1205 ..DecodeOptions::default()
1206 };
1207
1208 let mut iterator = RecordIterator::new(cursor, &schema, &options).unwrap();
1209
1210 // First record should decode successfully
1211 let first = iterator.next();
1212 assert!(first.is_some());
1213 assert!(first.unwrap().is_ok());
1214
1215 // Second call should return None (EOF)
1216 assert!(iterator.next().is_none());
1217 }
1218
1219 #[derive(Default)]
1220 struct FailingReader {
1221 fail: bool,
1222 }
1223
1224 impl Read for FailingReader {
1225 fn read(&mut self, _buf: &mut [u8]) -> io::Result<usize> {
1226 if self.fail {
1227 Ok(0)
1228 } else {
1229 self.fail = true;
1230 Err(io::Error::other("forced read error"))
1231 }
1232 }
1233 }
1234
1235 #[test]
1236 fn test_iterator_fixed_format_read_error_code() {
1237 let copybook_text = r"
1238 01 RECORD.
1239 05 ID PIC 9(3).
1240 05 NAME PIC X(5).
1241 ";
1242
1243 let schema = parse_copybook(copybook_text).unwrap();
1244
1245 let mut schema = schema;
1246 schema.lrecl_fixed = Some(8);
1247
1248 let mut iterator =
1249 RecordIterator::new(FailingReader::default(), &schema, &DecodeOptions::default())
1250 .unwrap();
1251
1252 let error = iterator.read_raw_record().unwrap_err();
1253 assert_eq!(error.code, ErrorCode::CBKR101_FIXED_RECORD_ERROR);
1254 }
1255
1256 enum ReadStep {
1257 Bytes(Vec<u8>),
1258 Error(io::ErrorKind),
1259 Eof,
1260 }
1261
1262 struct ScriptedReader {
1263 steps: VecDeque<ReadStep>,
1264 }
1265
1266 impl ScriptedReader {
1267 fn new(steps: impl IntoIterator<Item = ReadStep>) -> Self {
1268 Self {
1269 steps: steps.into_iter().collect(),
1270 }
1271 }
1272 }
1273
1274 impl Read for ScriptedReader {
1275 fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
1276 match self.steps.pop_front() {
1277 Some(ReadStep::Bytes(mut bytes)) => {
1278 let read = bytes.len().min(buf.len());
1279 buf[..read].copy_from_slice(&bytes[..read]);
1280 if read < bytes.len() {
1281 bytes.drain(..read);
1282 self.steps.push_front(ReadStep::Bytes(bytes));
1283 }
1284 Ok(read)
1285 }
1286 Some(ReadStep::Error(kind)) => Err(io::Error::new(kind, "scripted read error")),
1287 Some(ReadStep::Eof) | None => Ok(0),
1288 }
1289 }
1290 }
1291
1292 fn rdw_iterator(reader: ScriptedReader) -> RecordIterator<ScriptedReader> {
1293 let schema = eight_byte_schema();
1294 let options = DecodeOptions::default()
1295 .with_format(RecordFormat::RDW)
1296 .with_codepage(Codepage::ASCII);
1297 RecordIterator::new(reader, &schema, &options).unwrap()
1298 }
1299
1300 fn complete_rdw_record() -> Vec<u8> {
1301 let mut record = vec![0x00, 0x08, 0x00, 0x00];
1302 record.extend_from_slice(b"RECORD01");
1303 record
1304 }
1305
1306 fn assert_rdw_error(error: Error, code: ErrorCode, record_index: u64) {
1307 assert_eq!(error.code, code);
1308 assert_eq!(
1309 error.context.and_then(|context| context.record_index),
1310 Some(record_index)
1311 );
1312 }
1313
1314 #[test]
1315 fn rdw_clean_boundary_eof_returns_none() {
1316 let mut iterator = rdw_iterator(ScriptedReader::new([ReadStep::Eof]));
1317
1318 assert!(iterator.read_raw_record().unwrap().is_none());
1319 assert!(iterator.is_eof());
1320 }
1321
1322 #[test]
1323 fn rdw_partial_next_header_keeps_underflow_and_context() {
1324 let mut iterator = rdw_iterator(ScriptedReader::new([
1325 ReadStep::Bytes(complete_rdw_record()),
1326 ReadStep::Bytes(vec![0x00, 0x08]),
1327 ReadStep::Eof,
1328 ]));
1329
1330 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1331 assert_rdw_error(
1332 iterator.read_raw_record().unwrap_err(),
1333 ErrorCode::CBKF221_RDW_UNDERFLOW,
1334 2,
1335 );
1336 }
1337
1338 #[test]
1339 fn rdw_next_header_io_error_maps_to_read_error_with_context() {
1340 let mut iterator = rdw_iterator(ScriptedReader::new([
1341 ReadStep::Bytes(complete_rdw_record()),
1342 ReadStep::Error(io::ErrorKind::Other),
1343 ]));
1344
1345 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1346 assert_rdw_error(
1347 iterator.read_raw_record().unwrap_err(),
1348 ErrorCode::CBKR201_RDW_READ_ERROR,
1349 2,
1350 );
1351 }
1352
1353 #[test]
1354 fn rdw_partial_next_payload_keeps_underflow_and_context() {
1355 let mut iterator = rdw_iterator(ScriptedReader::new([
1356 ReadStep::Bytes(complete_rdw_record()),
1357 ReadStep::Bytes(vec![0x00, 0x08, 0x00, 0x00]),
1358 ReadStep::Bytes(b"ABC".to_vec()),
1359 ReadStep::Eof,
1360 ]));
1361
1362 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1363 assert_rdw_error(
1364 iterator.read_raw_record().unwrap_err(),
1365 ErrorCode::CBKF221_RDW_UNDERFLOW,
1366 2,
1367 );
1368 }
1369
1370 #[test]
1371 fn rdw_next_payload_io_error_maps_to_read_error_with_context() {
1372 let mut iterator = rdw_iterator(ScriptedReader::new([
1373 ReadStep::Bytes(complete_rdw_record()),
1374 ReadStep::Bytes(vec![0x00, 0x08, 0x00, 0x00]),
1375 ReadStep::Bytes(b"ABC".to_vec()),
1376 ReadStep::Error(io::ErrorKind::Other),
1377 ]));
1378
1379 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1380 assert_rdw_error(
1381 iterator.read_raw_record().unwrap_err(),
1382 ErrorCode::CBKR201_RDW_READ_ERROR,
1383 2,
1384 );
1385 }
1386
1387 /// Reader that hands out one byte per call, so `read_exact` cannot fill the
1388 /// buffer in a single call and the boundary logic is exercised for real.
1389 struct DribbleReader {
1390 data: Vec<u8>,
1391 position: usize,
1392 }
1393
1394 impl Read for DribbleReader {
1395 fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
1396 if self.position >= self.data.len() || buf.is_empty() {
1397 return Ok(0);
1398 }
1399 buf[0] = self.data[self.position];
1400 self.position += 1;
1401 Ok(1)
1402 }
1403 }
1404
1405 fn eight_byte_schema() -> Schema {
1406 let mut schema = parse_copybook("01 RECORD.\n 05 NAME PIC X(8).").unwrap();
1407 schema.lrecl_fixed = Some(8);
1408 schema
1409 }
1410
1411 #[test]
1412 fn fixed_trailing_partial_record_is_reported_not_discarded() {
1413 // Twelve bytes for an 8-byte LRECL: one whole record and a 4-byte tail.
1414 let data = b"RECORD01HALF".to_vec();
1415 let schema = eight_byte_schema();
1416 let options = DecodeOptions::default().with_codepage(Codepage::ASCII);
1417 let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options).unwrap();
1418
1419 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1420
1421 let error = iterator.read_raw_record().unwrap_err();
1422 assert_eq!(error.code, ErrorCode::CBKR101_FIXED_RECORD_ERROR);
1423 let context = error.context.expect("partial record reports context");
1424 assert_eq!(context.record_index, Some(2));
1425 assert!(
1426 context.details.unwrap_or_default().contains("4 of 8 bytes"),
1427 "details should name the short count"
1428 );
1429 }
1430
1431 #[test]
1432 fn fixed_partial_record_terminates_iteration_after_one_error() {
1433 let schema = eight_byte_schema();
1434 let options = DecodeOptions::default().with_codepage(Codepage::ASCII);
1435 let mut iterator =
1436 RecordIterator::new(Cursor::new(b"RECORD01HALF".to_vec()), &schema, &options).unwrap();
1437
1438 assert!(iterator.next().unwrap().is_ok());
1439 assert!(iterator.next().unwrap().is_err());
1440 assert!(
1441 iterator.next().is_none(),
1442 "iteration must stop after the partial-record error"
1443 );
1444 }
1445
1446 #[test]
1447 fn fixed_file_ending_on_a_record_boundary_is_still_clean_eof() {
1448 let schema = eight_byte_schema();
1449 let options = DecodeOptions::default().with_codepage(Codepage::ASCII);
1450 let mut iterator =
1451 RecordIterator::new(Cursor::new(b"RECORD01".to_vec()), &schema, &options).unwrap();
1452
1453 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1454 assert!(iterator.read_raw_record().unwrap().is_none());
1455 assert!(iterator.is_eof());
1456 }
1457
1458 #[test]
1459 fn fixed_partial_record_detected_when_reads_return_one_byte_at_a_time() {
1460 // A short `read` is not end of file. The boundary helper must keep
1461 // reading until the buffer is full or the reader is genuinely dry.
1462 let schema = eight_byte_schema();
1463 let options = DecodeOptions::default().with_codepage(Codepage::ASCII);
1464 let reader = DribbleReader {
1465 data: b"RECORD01HALF".to_vec(),
1466 position: 0,
1467 };
1468 let mut iterator = RecordIterator::new(reader, &schema, &options).unwrap();
1469
1470 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1471 assert_eq!(
1472 iterator.read_raw_record().unwrap_err().code,
1473 ErrorCode::CBKR101_FIXED_RECORD_ERROR
1474 );
1475 }
1476
1477 #[test]
1478 fn rdw_trailing_partial_header_is_reported_not_discarded() {
1479 // One 8-byte RDW record followed by a 2-byte stub of a second header.
1480 let mut data = vec![0x00, 0x08, 0x00, 0x00];
1481 data.extend_from_slice(b"RECORD01");
1482 data.extend_from_slice(&[0x00, 0x08]);
1483
1484 let schema = eight_byte_schema();
1485 let options = DecodeOptions::default()
1486 .with_format(RecordFormat::RDW)
1487 .with_codepage(Codepage::ASCII);
1488 let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options).unwrap();
1489
1490 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1491
1492 let error = iterator.read_raw_record().unwrap_err();
1493 assert_eq!(error.code, ErrorCode::CBKF221_RDW_UNDERFLOW);
1494 }
1495
1496 #[test]
1497 fn rdw_file_ending_on_a_record_boundary_is_still_clean_eof() {
1498 let mut data = vec![0x00, 0x08, 0x00, 0x00];
1499 data.extend_from_slice(b"RECORD01");
1500
1501 let schema = eight_byte_schema();
1502 let options = DecodeOptions::default()
1503 .with_format(RecordFormat::RDW)
1504 .with_codepage(Codepage::ASCII);
1505 let mut iterator = RecordIterator::new(Cursor::new(data), &schema, &options).unwrap();
1506
1507 assert_eq!(iterator.read_raw_record().unwrap().unwrap(), b"RECORD01");
1508 assert!(iterator.read_raw_record().unwrap().is_none());
1509 assert!(iterator.is_eof());
1510 }
1511}