Skip to main content

readcon_core/
parser.rs

1use crate::error::ParseError;
2use crate::helpers::symbol_to_atomic_number;
3use crate::types::{
4    AtomDatum, ConFrame, FrameHeader, PreboxHeader, SECTION_CHARGES, SECTION_DISPLACEMENTS,
5    SECTION_ENERGIES, SECTION_FORCES, SECTION_MAGMOMS, SECTION_SPINS, SECTION_SPREADS,
6    SECTION_VELOCITIES, decode_fixed_bitmask_for_spec, meta,
7};
8use serde_json::Value;
9use std::collections::BTreeMap;
10use std::iter::Peekable;
11use std::sync::Arc;
12
13/// Line source with peek for section detection (blank-line separators).
14///
15/// Implemented by the memchr cursor on the hot iterator path and by
16/// `Peekable<I>` for tests / ad-hoc callers.
17pub trait LineStream<'a> {
18    fn next_line(&mut self) -> Option<&'a str>;
19    fn peek_line(&mut self) -> Option<&'a str>;
20}
21
22impl<'a, I> LineStream<'a> for Peekable<I>
23where
24    I: Iterator<Item = &'a str>,
25{
26    #[inline]
27    fn next_line(&mut self) -> Option<&'a str> {
28        Iterator::next(self)
29    }
30    #[inline]
31    fn peek_line(&mut self) -> Option<&'a str> {
32        Peekable::peek(self).copied()
33    }
34}
35
36/// Hot-path: parse up to 5 whitespace-separated f64s into a stack buffer.
37/// Returns count of tokens actually present (before padding).
38/// Pads `out[found..max]` from `defaults` when `found < max` and `found >= min`.
39///
40/// Single-pass over the line bytes: skip ASCII whitespace, then
41/// [`fast_float2::parse_partial`] (Eisel–Lemire / SIMD-class decimal kernel)
42/// with a token-boundary check. No `SplitWhitespace`, no per-token `&str`,
43/// no heap `Vec`. Prefer this over allocating [`parse_line_of_n_f64`] on atom lines.
44#[inline]
45pub fn parse_line_of_range_f64_stack(
46    line: &str,
47    min: usize,
48    max: usize,
49    defaults: &[f64],
50    out: &mut [f64; 5],
51) -> Result<usize, ParseError> {
52    debug_assert!(max <= 5 && min <= max && defaults.len() >= max);
53    let bytes = line.as_bytes();
54    let n = bytes.len();
55    let mut i = 0usize;
56    let mut found = 0usize;
57    while found < max {
58        while i < n && bytes[i].is_ascii_whitespace() {
59            i += 1;
60        }
61        if i >= n {
62            break;
63        }
64        let (val, consumed) = fast_float2::parse_partial::<f64, _>(&bytes[i..]).map_err(|_| {
65            // Best-effort token for the error message (up to next whitespace).
66            let end = bytes[i..]
67                .iter()
68                .position(|b| b.is_ascii_whitespace())
69                .map(|k| i + k)
70                .unwrap_or(n);
71            let token = String::from_utf8_lossy(&bytes[i..end]);
72            ParseError::InvalidNumberFormat(format!("invalid float: {token}"))
73        })?;
74        let next = i + consumed;
75        // Reject partial tokens like "1.2abc" (must end at whitespace or EOS).
76        if next < n && !bytes[next].is_ascii_whitespace() {
77            let end = bytes[i..]
78                .iter()
79                .position(|b| b.is_ascii_whitespace())
80                .map(|k| i + k)
81                .unwrap_or(n);
82            let token = String::from_utf8_lossy(&bytes[i..end]);
83            return Err(ParseError::InvalidNumberFormat(format!(
84                "invalid float: {token}"
85            )));
86        }
87        out[found] = val;
88        found += 1;
89        i = next;
90    }
91    while i < n && bytes[i].is_ascii_whitespace() {
92        i += 1;
93    }
94    // Reject trailing extra tokens (same contract as the old split loop).
95    if i < n {
96        return Err(ParseError::InvalidVectorLength {
97            expected: max,
98            found: found + 1,
99        });
100    }
101    if found < min || found > max {
102        return Err(ParseError::InvalidVectorLength {
103            expected: max,
104            found,
105        });
106    }
107    while found < max {
108        out[found] = defaults[found];
109        found += 1;
110    }
111    Ok(found)
112}
113
114/// Parses a line of whitespace-separated f64 values using fast-float2.
115///
116/// This is the hot-path parser for coordinate and velocity lines. It uses
117/// `fast_float2::parse` instead of `str::parse::<f64>()` for better throughput
118/// on the numeric-heavy atom data lines. Fixed-width atom lines use
119/// [`parse_line_of_range_f64_stack`] to avoid a heap `Vec` per line.
120///
121/// # Arguments
122///
123/// * `line` - A string slice representing a single line of data.
124/// * `n` - The exact number of f64 values expected on the line.
125pub fn parse_line_of_n_f64(line: &str, n: usize) -> Result<Vec<f64>, ParseError> {
126    if n <= 5 {
127        let defaults = [0.0f64; 5];
128        let mut buf = [0.0f64; 5];
129        parse_line_of_range_f64_stack(line, n, n, &defaults, &mut buf)?;
130        return Ok(buf[..n].to_vec());
131    }
132    let mut values = Vec::with_capacity(n);
133    for token in line.split_ascii_whitespace() {
134        let val: f64 = fast_float2::parse(token)
135            .map_err(|_| ParseError::InvalidNumberFormat(format!("invalid float: {token}")))?;
136        values.push(val);
137    }
138    if values.len() == n {
139        Ok(values)
140    } else {
141        Err(ParseError::InvalidVectorLength {
142            expected: n,
143            found: values.len(),
144        })
145    }
146}
147
148/// Parses a line of whitespace-separated f64 values, accepting between `min`
149/// and `max` values (inclusive). Returns a vector of exactly `max` elements,
150/// padding with values from `defaults` when fewer than `max` are present.
151///
152/// Used for atom lines where column 5 (atom_index) is optional.
153/// Prefer [`parse_line_of_range_f64_stack`] on the atom hot path (`max <= 5`).
154pub fn parse_line_of_range_f64(
155    line: &str,
156    min: usize,
157    max: usize,
158    defaults: &[f64],
159) -> Result<Vec<f64>, ParseError> {
160    if max <= 5 {
161        let mut buf = [0.0f64; 5];
162        parse_line_of_range_f64_stack(line, min, max, defaults, &mut buf)?;
163        return Ok(buf[..max].to_vec());
164    }
165    let mut values = Vec::with_capacity(max);
166    for token in line.split_ascii_whitespace() {
167        let val: f64 = fast_float2::parse(token)
168            .map_err(|_| ParseError::InvalidNumberFormat(format!("invalid float: {token}")))?;
169        values.push(val);
170    }
171    if values.len() < min || values.len() > max {
172        return Err(ParseError::InvalidVectorLength {
173            expected: max,
174            found: values.len(),
175        });
176    }
177    while values.len() < max {
178        let idx = values.len();
179        values.push(defaults[idx]);
180    }
181    Ok(values)
182}
183
184fn metadata_json_error(message: impl Into<String>) -> ParseError {
185    ParseError::InvalidMetadataJson(message.into())
186}
187
188fn validate_metadata_number(key: &str, value: &Value) -> Result<(), ParseError> {
189    if value.as_f64().is_some() {
190        Ok(())
191    } else {
192        Err(metadata_json_error(format!(
193            "{key} must be a finite number"
194        )))
195    }
196}
197
198fn validate_metadata_integer(key: &str, value: &Value) -> Result<(), ParseError> {
199    if value.as_u64().is_some() {
200        Ok(())
201    } else {
202        Err(metadata_json_error(format!(
203            "{key} must be a non-negative integer"
204        )))
205    }
206}
207
208/// Validates a parsed metadata JSON object against the spec v2 schema.
209///
210/// Type-checks the `validate` and `sections` keys, then runs per-key
211/// schema checks for the recommended metadata keys. Used by the
212/// builder/Python `set_metadata_json` paths to fail fast on malformed
213/// input, and by the parser when the file requested strict validation
214/// via `"validate": true`.
215pub fn validate_metadata_schema(
216    json_obj: &serde_json::Map<String, Value>,
217) -> Result<(), ParseError> {
218    let strict_requested = match json_obj.get(meta::VALIDATE) {
219        Some(Value::Bool(b)) => *b,
220        Some(_) => return Err(metadata_json_error("validate must be a boolean")),
221        None => false,
222    };
223
224    match json_obj.get(meta::SECTIONS) {
225        Some(Value::Array(values)) => {
226            if values.iter().any(|entry| !entry.is_string()) {
227                return Err(metadata_json_error("sections must be an array of strings"));
228            }
229        }
230        Some(_) => return Err(metadata_json_error("sections must be an array of strings")),
231        None if strict_requested => {
232            return Err(metadata_json_error(
233                "validate=true requires a sections array, even when empty",
234            ));
235        }
236        None => {}
237    }
238
239    for (key, value) in json_obj {
240        match key.as_str() {
241            meta::CON_SPEC_VERSION | meta::SECTIONS | meta::VALIDATE => {}
242            meta::ENERGY
243            | meta::TIME
244            | meta::TIMESTEP
245            | meta::CONVERGENCE_FMAX
246            | meta::CONVERGENCE_ENERGY
247            | meta::FMAX => validate_metadata_number(key, value)?,
248            meta::FRAME_INDEX | meta::NEB_BEAD | meta::NEB_BAND => {
249                validate_metadata_integer(key, value)?
250            }
251            meta::GENERATOR if !value.is_string() => {
252                return Err(metadata_json_error("generator must be a string"));
253            }
254            meta::GENERATOR => {}
255            meta::UNITS | meta::POTENTIAL if !value.is_object() => {
256                return Err(metadata_json_error(format!("{key} must be an object")));
257            }
258            meta::POTENTIAL => {
259                if let Some(potential_type) = value.get("type")
260                    && !potential_type.is_string()
261                {
262                    return Err(metadata_json_error("potential.type must be a string"));
263                }
264            }
265            meta::UNITS => {
266                // Full v3 checks (required keys) run when con_spec_version >= 3.
267            }
268            meta::PBC => validate_pbc_metadata(value)?,
269            meta::BONDS => validate_bonds_metadata(value)?,
270            meta::LATTICE_VECTORS => validate_lattice_vectors_metadata(value)?,
271            meta::CONVERGED if !value.is_boolean() => {
272                return Err(metadata_json_error("converged must be a boolean"));
273            }
274            meta::CONVERGED => {}
275            _ => {}
276        }
277    }
278
279    Ok(())
280}
281
282fn validate_pbc_metadata(value: &Value) -> Result<(), ParseError> {
283    let Some(values) = value.as_array() else {
284        return Err(metadata_json_error("pbc must be a length-3 boolean array"));
285    };
286    if values.len() != 3 || values.iter().any(|entry| !entry.is_boolean()) {
287        return Err(metadata_json_error("pbc must be a length-3 boolean array"));
288    }
289    Ok(())
290}
291
292/// Validate optional `bonds` frame topology metadata.
293///
294/// Each element is either a length-2 non-negative integer pair `[i, j]` or an
295/// object `{"i": i, "j": j, "order"?: integer}`. Indices are 0-based into
296/// `atom_data` order (parser does not yet know atom count at metadata time;
297/// bounds are enforced when projecting to chemfiles / selection).
298fn validate_bonds_metadata(value: &Value) -> Result<(), ParseError> {
299    let Some(items) = value.as_array() else {
300        return Err(metadata_json_error("bonds must be an array"));
301    };
302    for (idx, item) in items.iter().enumerate() {
303        if let Some(pair) = item.as_array() {
304            if pair.len() != 2 {
305                return Err(metadata_json_error(format!(
306                    "bonds[{idx}] pair must have exactly two indices"
307                )));
308            }
309            for (k, entry) in pair.iter().enumerate() {
310                let Some(n) = entry.as_u64() else {
311                    return Err(metadata_json_error(format!(
312                        "bonds[{idx}][{k}] must be a non-negative integer"
313                    )));
314                };
315                if n > u32::MAX as u64 {
316                    return Err(metadata_json_error(format!(
317                        "bonds[{idx}][{k}] index exceeds u32"
318                    )));
319                }
320            }
321            continue;
322        }
323        if let Some(obj) = item.as_object() {
324            for key in ["i", "j"] {
325                let Some(entry) = obj.get(key) else {
326                    return Err(metadata_json_error(format!(
327                        "bonds[{idx}] object must include \"{key}\""
328                    )));
329                };
330                let Some(n) = entry.as_u64() else {
331                    return Err(metadata_json_error(format!(
332                        "bonds[{idx}].{key} must be a non-negative integer"
333                    )));
334                };
335                if n > u32::MAX as u64 {
336                    return Err(metadata_json_error(format!(
337                        "bonds[{idx}].{key} index exceeds u32"
338                    )));
339                }
340            }
341            if let Some(order) = obj.get("order")
342                && order.as_i64().is_none()
343            {
344                return Err(metadata_json_error(format!(
345                    "bonds[{idx}].order must be an integer when present"
346                )));
347            }
348            continue;
349        }
350        return Err(metadata_json_error(format!(
351            "bonds[{idx}] must be [i, j] or {{\"i\": i, \"j\": j, \"order\"?: ...}}"
352        )));
353    }
354    Ok(())
355}
356
357fn validate_lattice_vectors_metadata(value: &Value) -> Result<(), ParseError> {
358    let Some(rows) = value.as_array() else {
359        return Err(metadata_json_error(
360            "lattice_vectors must be a 3x3 numeric array",
361        ));
362    };
363    if rows.len() != 3 {
364        return Err(metadata_json_error(
365            "lattice_vectors must be a 3x3 numeric array",
366        ));
367    }
368    for row in rows {
369        let Some(entries) = row.as_array() else {
370            return Err(metadata_json_error(
371                "lattice_vectors must be a 3x3 numeric array",
372            ));
373        };
374        if entries.len() != 3 || entries.iter().any(|entry| entry.as_f64().is_none()) {
375            return Err(metadata_json_error(
376                "lattice_vectors must be a 3x3 numeric array",
377            ));
378        }
379    }
380    Ok(())
381}
382
383/// Parses a line of whitespace-separated values into a vector of a specific type.
384///
385/// This generic helper function takes a string slice, splits it by whitespace,
386/// and attempts to parse each substring into the target type `T`. The type `T`
387/// must implement `std::str::FromStr`.
388///
389/// # Arguments
390///
391/// * `line` - A string slice representing a single line of data.
392/// * `n` - The exact number of values expected on the line.
393///
394/// # Errors
395///
396/// * `ParseError::InvalidVectorLength` if the number of parsed values is not equal to `n`.
397/// * Propagates any error from the `parse()` method of the target type `T`.
398///
399/// # Example
400///
401/// ```
402/// use readcon_core::parser::parse_line_of_n;
403/// let line = "10.5 20.0 30.5";
404/// let values: Vec<f64> = parse_line_of_n(line, 3).unwrap();
405/// assert_eq!(values, vec![10.5, 20.0, 30.5]);
406///
407/// let result = parse_line_of_n::<i32>(line, 2);
408/// assert!(result.is_err());
409/// ```
410pub fn parse_line_of_n<T: std::str::FromStr>(line: &str, n: usize) -> Result<Vec<T>, ParseError>
411where
412    ParseError: From<<T as std::str::FromStr>::Err>,
413{
414    let values: Vec<T> = line
415        .split_whitespace()
416        .map(|s| s.parse::<T>())
417        .collect::<Result<_, _>>()?;
418
419    if values.len() == n {
420        Ok(values)
421    } else {
422        Err(ParseError::InvalidVectorLength {
423            expected: n,
424            found: values.len(),
425        })
426    }
427}
428
429/// Parses the 9-line header of a `.con` file frame from an iterator.
430///
431/// This function consumes the next 9 lines from the given line iterator to
432/// construct a `FrameHeader`. The iterator is advanced by 9 lines on success.
433///
434/// # Arguments
435///
436/// * `lines` - A mutable reference to an iterator that yields string slices.
437///
438/// # Errors
439///
440/// * `ParseError::IncompleteHeader` if the iterator has fewer than 9 lines remaining.
441/// * Propagates any errors from `parse_line_of_n` if the numeric data within
442///   the header is malformed.
443///
444/// # Panics
445///
446/// This function will panic if the intermediate vectors for box dimensions or angles,
447/// after being successfully parsed, cannot be converted into fixed-size arrays.
448/// This should not happen if `parse_line_of_n` is used correctly with `n=3`.
449pub fn parse_frame_header<'a>(
450    lines: &mut impl Iterator<Item = &'a str>,
451) -> Result<FrameHeader, ParseError> {
452    let prebox1 = lines
453        .next()
454        .ok_or(ParseError::IncompleteHeader)?
455        .to_string();
456    let prebox2_raw = lines.next().ok_or(ParseError::IncompleteHeader)?;
457
458    // Line 2: if it starts with '{', parse as JSON metadata (spec v2+).
459    // Otherwise treat as a legacy (pre-v2) file with spec_version = 1.
460    let trimmed = prebox2_raw.trim();
461    let (spec_version, metadata, sections, validate, sections_declared) =
462        if trimmed.starts_with('{') {
463            let json_val: serde_json::Value = serde_json::from_str(trimmed)
464                .map_err(|e| ParseError::InvalidMetadataJson(e.to_string()))?;
465            let json_obj = json_val.as_object().ok_or_else(|| {
466                ParseError::InvalidMetadataJson("expected a JSON object".to_string())
467            })?;
468            let ver = json_obj
469                .get(meta::CON_SPEC_VERSION)
470                .and_then(|v| v.as_u64())
471                .ok_or(ParseError::MissingSpecVersion)? as u32;
472            if ver > crate::CON_SPEC_VERSION {
473                return Err(ParseError::UnsupportedSpecVersion(ver));
474            }
475            if ver >= 3 {
476                match json_obj.get(meta::UNITS) {
477                    Some(u) => crate::units::validate_v3_units_metadata(u)
478                        .map_err(|e| ParseError::ValidationError(format!("v3 units: {e}")))?,
479                    None => {
480                        return Err(ParseError::ValidationError(
481                        "con_spec_version >= 3 requires metadata \"units\" with length and energy"
482                            .into(),
483                    ));
484                    }
485                }
486            }
487
488            // Single pass over the JSON object: collect sections, capture the
489            // validate flag, copy the rest into metadata. Folds the previous
490            // pre-extract get(validate) + re-iterate pattern into one walk.
491            let mut sections: Vec<String> = Vec::new();
492            let mut metadata = BTreeMap::new();
493            let mut sections_declared = false;
494            let mut validate = false;
495            for (k, v) in json_obj {
496                match k.as_str() {
497                    meta::CON_SPEC_VERSION => {}
498                    meta::SECTIONS => {
499                        sections_declared = true;
500                        let arr = v.as_array().ok_or_else(|| {
501                            metadata_json_error("sections must be an array of strings")
502                        })?;
503                        sections.reserve(arr.len());
504                        for entry in arr {
505                            let s = entry.as_str().ok_or_else(|| {
506                                metadata_json_error("sections must be an array of strings")
507                            })?;
508                            sections.push(s.to_string());
509                        }
510                    }
511                    meta::VALIDATE => {
512                        validate = match v {
513                            Value::Bool(b) => *b,
514                            _ => return Err(metadata_json_error("validate must be a boolean")),
515                        };
516                        metadata.insert(k.clone(), v.clone());
517                    }
518                    _ => {
519                        metadata.insert(k.clone(), v.clone());
520                    }
521                }
522            }
523
524            // Strict-mode schema check fires only when the file requested
525            // it. Hot-path parses (validate=false) skip the per-key match.
526            if validate {
527                validate_metadata_schema(json_obj)?;
528            }
529
530            (ver, metadata, sections, validate, sections_declared)
531        } else {
532            // Legacy file: no JSON metadata line.
533            (1_u32, BTreeMap::new(), Vec::new(), false, false)
534        };
535    let prebox2 = prebox2_raw.to_string();
536
537    let boxl_vec = parse_line_of_n_f64(lines.next().ok_or(ParseError::IncompleteHeader)?, 3)?;
538    let angles_vec = parse_line_of_n_f64(lines.next().ok_or(ParseError::IncompleteHeader)?, 3)?;
539    let postbox1 = lines
540        .next()
541        .ok_or(ParseError::IncompleteHeader)?
542        .to_string();
543    let postbox2 = lines
544        .next()
545        .ok_or(ParseError::IncompleteHeader)?
546        .to_string();
547    let natm_types =
548        parse_line_of_n::<usize>(lines.next().ok_or(ParseError::IncompleteHeader)?, 1)?[0];
549    let natms_per_type = parse_line_of_n::<usize>(
550        lines.next().ok_or(ParseError::IncompleteHeader)?,
551        natm_types,
552    )?;
553    let masses_per_type = parse_line_of_n_f64(
554        lines.next().ok_or(ParseError::IncompleteHeader)?,
555        natm_types,
556    )?;
557    if validate {
558        validate_header_geometry(&boxl_vec, &angles_vec, natm_types, &natms_per_type)?;
559        validate_masses(&masses_per_type)?;
560    }
561    Ok(FrameHeader {
562        prebox_header: PreboxHeader {
563            user: prebox1,
564            metadata_line: prebox2,
565        },
566        boxl: boxl_vec.try_into().unwrap(),
567        angles: angles_vec.try_into().unwrap(),
568        postbox_header: [postbox1, postbox2],
569        natm_types,
570        natms_per_type,
571        masses_per_type,
572        spec_version,
573        metadata,
574        sections,
575        strict_validation: validate,
576        sections_declared,
577    })
578}
579
580/// Parses a complete frame from a `.con` file, including its header and atomic data.
581///
582/// This function first parses the complete frame header and then uses the information within it
583/// (specifically the number of atom types and atoms per type) to parse the subsequent
584/// atom coordinate blocks.
585///
586/// # Arguments
587///
588/// * `lines` - A mutable reference to an iterator that yields string slices for the frame.
589///
590/// # Errors
591///
592/// * `ParseError::IncompleteFrame` if the iterator ends before all expected
593///   atomic data has been read.
594/// * Propagates any errors from the underlying calls to `parse_frame_header` and
595///   `parse_line_of_n`.
596///
597/// # Example
598///
599/// ```
600/// use readcon_core::parser::parse_single_frame;
601///
602/// let frame_text = r#"
603///Generated by test
604///{"con_spec_version":2}
605///10.0 10.0 10.0
606///90.0 90.0 90.0
607///POSTBOX LINE 1
608///POSTBOX LINE 2
609///2
610///1 1
611///12.011 1.008
612///C
613///Coordinates of Component 1
614///1.0 1.0 1.0 0.0 1
615///H
616///Coordinates of Component 2
617///2.0 2.0 2.0 0.0 2
618/// "#;
619///
620/// let mut lines = frame_text.trim().lines();
621/// let con_frame = parse_single_frame(&mut lines).unwrap();
622///
623/// assert_eq!(con_frame.header.natm_types, 2);
624/// assert_eq!(con_frame.atom_data.len(), 2);
625/// assert_eq!(&*con_frame.atom_data[0].symbol, "C");
626/// assert_eq!(con_frame.atom_data[1].atom_id, 2);
627/// ```
628pub fn parse_single_frame<'a>(
629    lines: &mut impl Iterator<Item = &'a str>,
630) -> Result<ConFrame, ParseError> {
631    let header = parse_frame_header(lines)?;
632    let validate = header.strict_validation;
633    let total_atoms: usize = header.natms_per_type.iter().sum();
634    let mut atom_data = Vec::with_capacity(total_atoms);
635    // SoA positions: default f64 fills a flat `Vec` then one Arc wrap (profile:
636    // per-row ArcArray mut checks were a real cost on multi-atom parse).
637    use crate::storage_dtype::{ElementKind, FloatArray2, StorageDtypes};
638    let dt = StorageDtypes::from_metadata(&header.metadata).unwrap_or_default();
639    let f64_positions = dt.positions == ElementKind::Float64;
640    let mut pos_flat = if f64_positions {
641        vec![0.0f64; total_atoms.saturating_mul(3)]
642    } else {
643        Vec::new()
644    };
645    let mut positions_other = if f64_positions {
646        None
647    } else {
648        Some(FloatArray2::zeros(dt.positions, total_atoms, 3))
649    };
650
651    let mut global_atom_idx: u64 = 0;
652    let mut atom_i = 0usize;
653    for (type_idx, num_atoms) in header.natms_per_type.iter().enumerate() {
654        // Allocate the per-component Arc<str> directly from the trimmed
655        // line; going through a String intermediate would add a second
656        // allocation and copy for no semantic gain.
657        let symbol_line = lines.next().ok_or(ParseError::IncompleteFrame)?;
658        let symbol: Arc<str> = Arc::from(symbol_line.trim());
659        let coord_label = lines.next().ok_or(ParseError::IncompleteFrame)?;
660        if validate {
661            validate_coordinate_component(type_idx, symbol.as_ref(), coord_label)?;
662        }
663        for _ in 0..*num_atoms {
664            let coord_line = lines.next().ok_or(ParseError::IncompleteFrame)?;
665            // Column 5 (atom_index) is optional; defaults to sequential index.
666            let defaults = [0.0, 0.0, 0.0, 0.0, global_atom_idx as f64];
667            let mut vals = [0.0f64; 5];
668            parse_line_of_range_f64_stack(coord_line, 4, 5, &defaults, &mut vals)?;
669            let (fixed, atom_id) = if validate {
670                parse_identity_columns(coord_line, "coordinate", 3, 4, 5, header.spec_version)?
671            } else {
672                (
673                    decode_fixed_bitmask_for_spec(vals[3] as u8, header.spec_version),
674                    vals[4] as u64,
675                )
676            };
677            let xyz = [vals[0], vals[1], vals[2]];
678            if f64_positions {
679                let o = atom_i * 3;
680                pos_flat[o] = xyz[0];
681                pos_flat[o + 1] = xyz[1];
682                pos_flat[o + 2] = xyz[2];
683            } else if let Some(ref mut pos) = positions_other {
684                pos.set_f64_row(atom_i, xyz);
685            }
686            atom_data.push(AtomDatum {
687                // This is a cheap reference-count increment, not a full string clone.
688                symbol: Arc::clone(&symbol),
689                x: xyz[0],
690                y: xyz[1],
691                z: xyz[2],
692                fixed,
693                atom_id,
694                velocity: None,
695                force: None,
696                energy: None,
697                charge: None,
698                spin: None,
699                magmom: None,
700                displacement: None,
701                spread: None,
702            });
703            global_atom_idx += 1;
704            atom_i += 1;
705        }
706    }
707    let positions = if f64_positions {
708        FloatArray2::from_f64_row_major(total_atoms, 3, pos_flat)
709    } else {
710        positions_other.expect("non-f64 positions allocated")
711    };
712    // Sections still attach to AoS; assemble uses prefilled positions (no second pos pass).
713    Ok(crate::types::con_frame_from_atom_data_with_positions(
714        header, atom_data, positions,
715    ))
716}
717
718fn validate_header_geometry(
719    boxl: &[f64],
720    angles: &[f64],
721    natm_types: usize,
722    natms_per_type: &[usize],
723) -> Result<(), ParseError> {
724    if boxl
725        .iter()
726        .any(|length| !length.is_finite() || *length <= 0.0)
727        || angles
728            .iter()
729            .any(|angle| !angle.is_finite() || *angle <= 0.0 || *angle >= 180.0)
730    {
731        return Err(ParseError::ValidationError(
732            "cell geometry must have positive lengths and angles between 0 and 180 degrees"
733                .to_string(),
734        ));
735    }
736    if natm_types == 0 || natms_per_type.contains(&0) {
737        return Err(ParseError::ValidationError(
738            "atom counts must contain at least one atom per component".to_string(),
739        ));
740    }
741    Ok(())
742}
743
744fn validate_masses(masses_per_type: &[f64]) -> Result<(), ParseError> {
745    if masses_per_type
746        .iter()
747        .any(|mass| !mass.is_finite() || *mass <= 0.0)
748    {
749        return Err(ParseError::ValidationError(
750            "component masses must be positive".to_string(),
751        ));
752    }
753    Ok(())
754}
755
756fn validate_coordinate_component(
757    type_idx: usize,
758    symbol: &str,
759    label: &str,
760) -> Result<(), ParseError> {
761    let expected_label = format!("Coordinates of Component {}", type_idx + 1);
762    if label.trim() != expected_label {
763        return Err(ParseError::ValidationError(format!(
764            "expected coordinate label {expected_label:?}, found {label:?}"
765        )));
766    }
767    if symbol != "X" && symbol_to_atomic_number(symbol) == 0 {
768        return Err(ParseError::ValidationError(format!(
769            "unknown component symbol {symbol}"
770        )));
771    }
772    Ok(())
773}
774
775/// Strict-validation parser for the per-row identity columns
776/// (fixed bitmask + atom_id) used by every section type.
777///
778/// `n_cols` is the total whitespace-separated column count expected on
779/// the row in strict mode, and `(fixed_idx, atom_id_idx)` are the
780/// 0-based positions of the fixed bitmask and atom_id columns inside
781/// that layout. Each section calls in with its own values:
782///
783/// - coordinates / velocities / forces: 5 cols, fixed=3, atom_id=4
784/// - energies: 3 cols, fixed=1, atom_id=2
785///
786/// String-based parsing on purpose: strict v2 mode rejects values that
787/// are not in the canonical integer form (e.g. `5.0` for a bitmask),
788/// which an f64 round-trip would silently accept.
789fn parse_identity_columns(
790    line: &str,
791    row_kind: &str,
792    fixed_idx: usize,
793    atom_id_idx: usize,
794    n_cols: usize,
795    spec_version: u32,
796) -> Result<([bool; 3], u64), ParseError> {
797    let columns = line.split_ascii_whitespace().collect::<Vec<_>>();
798    if columns.len() != n_cols {
799        return Err(ParseError::ValidationError(format!(
800            "{row_kind} rows require {n_cols} columns including fixed_flag and atom_id in validate mode"
801        )));
802    }
803    let fixed_flag = columns[fixed_idx].parse::<u8>().map_err(|_| {
804        ParseError::ValidationError(format!("{row_kind} fixed_flag must be an integer bitmask"))
805    })?;
806    if fixed_flag > 7 {
807        return Err(ParseError::ValidationError(format!(
808            "{row_kind} fixed_flag must be between 0 and 7"
809        )));
810    }
811    let atom_id = columns[atom_id_idx].parse::<u64>().map_err(|_| {
812        ParseError::ValidationError(format!("{row_kind} atom_id must be an integer"))
813    })?;
814    Ok((
815        decode_fixed_bitmask_for_spec(fixed_flag, spec_version),
816        atom_id,
817    ))
818}
819
820fn validate_section_component(
821    section: &str,
822    type_idx: usize,
823    atom_idx: usize,
824    symbol: &str,
825    label: &str,
826    header: &FrameHeader,
827    atom_data: &[AtomDatum],
828) -> Result<(), ParseError> {
829    let expected_label = format!("{section} of Component {}", type_idx + 1);
830    if label.trim() != expected_label {
831        return Err(ParseError::ValidationError(format!(
832            "expected section label {expected_label:?}, found {label:?}"
833        )));
834    }
835
836    if header.natms_per_type[type_idx] == 0 {
837        return Ok(());
838    }
839
840    let expected_symbol = atom_data
841        .get(atom_idx)
842        .map(|atom| atom.symbol.as_ref())
843        .ok_or_else(|| {
844            ParseError::ValidationError(format!(
845                "{section} component {} has no coordinate atom to validate against",
846                type_idx + 1
847            ))
848        })?;
849    if symbol != expected_symbol {
850        return Err(ParseError::ValidationError(format!(
851            "{section} component {} symbol mismatch: expected {expected_symbol}, found {symbol}",
852            type_idx + 1
853        )));
854    }
855
856    Ok(())
857}
858
859fn validate_section_atom_identity(
860    section: &str,
861    atom_idx: usize,
862    fixed: [bool; 3],
863    atom_id: u64,
864    atom_data: &[AtomDatum],
865) -> Result<(), ParseError> {
866    let atom = atom_data.get(atom_idx).ok_or_else(|| {
867        ParseError::ValidationError(format!(
868            "{section} row {atom_idx} has no coordinate atom to validate against"
869        ))
870    })?;
871
872    if atom.fixed != fixed {
873        return Err(ParseError::ValidationError(format!(
874            "{section} row {atom_idx} fixed mask mismatch for atom_id {}",
875            atom.atom_id
876        )));
877    }
878    if atom.atom_id != atom_id {
879        return Err(ParseError::ValidationError(format!(
880            "{section} row {atom_idx} atom_id mismatch: expected {}, found {atom_id}",
881            atom.atom_id
882        )));
883    }
884
885    Ok(())
886}
887
888/// Attempts to parse an optional velocity section following coordinate blocks.
889///
890/// In `.convel` files, after all coordinate blocks there is a blank separator line
891/// followed by per-component velocity blocks with the same structure as coordinate
892/// blocks (symbol line, "Velocities of Component N" line, then atom lines with
893/// `vx vy vz fixed atomID`).
894///
895/// This function peeks at the next line. If it is blank (or contains only whitespace),
896/// it consumes the blank line and parses velocity data into the existing `atom_data`.
897/// If the next line is not blank (or is absent), no velocities are parsed.
898///
899/// Returns `Ok(true)` if velocities were found and parsed, `Ok(false)` otherwise.
900pub fn parse_velocity_section<'a>(
901    lines: &mut impl LineStream<'a>,
902    header: &FrameHeader,
903    atom_data: &mut [AtomDatum],
904) -> Result<bool, ParseError> {
905    let validate = header.strict_validation;
906    // Peek at the next line to check for blank separator
907    match lines.peek_line() {
908        Some(line) if line.trim().is_empty() => {
909            // Consume the blank separator
910            lines.next_line();
911        }
912        _ => return Ok(false),
913    }
914
915    let mut atom_idx: usize = 0;
916    for (type_idx, &num_atoms) in header.natms_per_type.iter().enumerate() {
917        // Symbol line
918        let symbol = lines
919            .next_line()
920            .ok_or(ParseError::IncompleteVelocitySection)?
921            .trim();
922
923        // "Velocities of Component N" line
924        let comp_line = lines
925            .next_line()
926            .ok_or(ParseError::IncompleteVelocitySection)?;
927        // Validate it looks like a velocity header (optional strictness)
928        if !comp_line.contains("Velocities of Component") {
929            return Err(ParseError::IncompleteVelocitySection);
930        }
931        if validate {
932            validate_section_component(
933                "Velocities",
934                type_idx,
935                atom_idx,
936                symbol,
937                comp_line,
938                header,
939                atom_data,
940            )?;
941        }
942
943        for _ in 0..num_atoms {
944            let vel_line = lines
945                .next_line()
946                .ok_or(ParseError::IncompleteVelocitySection)?;
947            // Column 5 (atom_index) is optional in velocity lines too.
948            let defaults = [0.0, 0.0, 0.0, 0.0, atom_idx as f64];
949            let mut vals = [0.0f64; 5];
950            parse_line_of_range_f64_stack(vel_line, 4, 5, &defaults, &mut vals)?;
951            if validate {
952                let (fixed, atom_id) =
953                    parse_identity_columns(vel_line, "velocities", 3, 4, 5, header.spec_version)?;
954                validate_section_atom_identity("velocities", atom_idx, fixed, atom_id, atom_data)?;
955            }
956            if atom_idx < atom_data.len() {
957                atom_data[atom_idx].velocity = Some([vals[0], vals[1], vals[2]]);
958            }
959            atom_idx += 1;
960        }
961    }
962
963    Ok(true)
964}
965
966/// Attempts to parse a force section following coordinate (and optional velocity) blocks.
967///
968/// Force sections mirror velocity sections: a blank separator line followed by per-component
969/// force blocks (symbol line, "Forces of Component N" line, then atom lines with
970/// `fx fy fz fixed_flag atom_id`).
971///
972/// Returns `Ok(true)` if forces were found and parsed, `Ok(false)` otherwise.
973pub fn parse_force_section<'a>(
974    lines: &mut impl LineStream<'a>,
975    header: &FrameHeader,
976    atom_data: &mut [AtomDatum],
977) -> Result<bool, ParseError> {
978    let validate = header.strict_validation;
979    // Peek at the next line to check for blank separator
980    match lines.peek_line() {
981        Some(line) if line.trim().is_empty() => {
982            lines.next_line();
983        }
984        _ => return Ok(false),
985    }
986
987    let mut atom_idx: usize = 0;
988    for (type_idx, &num_atoms) in header.natms_per_type.iter().enumerate() {
989        let symbol = lines
990            .next_line()
991            .ok_or(ParseError::IncompleteForceSection)?
992            .trim();
993
994        let comp_line = lines
995            .next_line()
996            .ok_or(ParseError::IncompleteForceSection)?;
997        if !comp_line.contains("Forces of Component") {
998            return Err(ParseError::IncompleteForceSection);
999        }
1000        if validate {
1001            validate_section_component(
1002                "Forces", type_idx, atom_idx, symbol, comp_line, header, atom_data,
1003            )?;
1004        }
1005
1006        for _ in 0..num_atoms {
1007            let force_line = lines
1008                .next_line()
1009                .ok_or(ParseError::IncompleteForceSection)?;
1010            let defaults = [0.0, 0.0, 0.0, 0.0, atom_idx as f64];
1011            let mut vals = [0.0f64; 5];
1012            parse_line_of_range_f64_stack(force_line, 4, 5, &defaults, &mut vals)?;
1013            if validate {
1014                let (fixed, atom_id) =
1015                    parse_identity_columns(force_line, "forces", 3, 4, 5, header.spec_version)?;
1016                validate_section_atom_identity("forces", atom_idx, fixed, atom_id, atom_data)?;
1017            }
1018            if atom_idx < atom_data.len() {
1019                atom_data[atom_idx].force = Some([vals[0], vals[1], vals[2]]);
1020            }
1021            atom_idx += 1;
1022        }
1023    }
1024
1025    Ok(true)
1026}
1027
1028/// Attempts to parse an energies section following coordinate (and optional
1029/// velocity / force) blocks.
1030///
1031/// Energy sections mirror force sections but with one scalar per atom:
1032/// blank separator, then per-component blocks of (symbol, "Energies of
1033/// Component N", and atom lines `e fixed_flag atom_id`). The two
1034/// trailing identity columns are optional and used only for strict
1035/// validation; in non-strict mode any whitespace after the energy is
1036/// ignored.
1037///
1038/// Returns `Ok(true)` if energies were found and parsed, `Ok(false)`
1039/// otherwise.
1040pub fn parse_energy_section<'a>(
1041    lines: &mut impl LineStream<'a>,
1042    header: &FrameHeader,
1043    atom_data: &mut [AtomDatum],
1044) -> Result<bool, ParseError> {
1045    let validate = header.strict_validation;
1046    match lines.peek_line() {
1047        Some(line) if line.trim().is_empty() => {
1048            lines.next_line();
1049        }
1050        _ => return Ok(false),
1051    }
1052
1053    let mut atom_idx: usize = 0;
1054    for (type_idx, &num_atoms) in header.natms_per_type.iter().enumerate() {
1055        let symbol = lines
1056            .next_line()
1057            .ok_or(ParseError::IncompleteEnergySection)?
1058            .trim();
1059
1060        let comp_line = lines
1061            .next_line()
1062            .ok_or(ParseError::IncompleteEnergySection)?;
1063        if !comp_line.contains("Energies of Component") {
1064            return Err(ParseError::IncompleteEnergySection);
1065        }
1066        if validate {
1067            validate_section_component(
1068                "Energies", type_idx, atom_idx, symbol, comp_line, header, atom_data,
1069            )?;
1070        }
1071
1072        for _ in 0..num_atoms {
1073            let energy_line = lines
1074                .next_line()
1075                .ok_or(ParseError::IncompleteEnergySection)?;
1076            // Single energy column, plus optional fixed flag and atom_id
1077            // for round-trip identity checks.
1078            let defaults = [0.0, 0.0, atom_idx as f64];
1079            let vals = parse_line_of_range_f64(energy_line, 1, 3, &defaults)?;
1080            if validate {
1081                let (fixed, atom_id) =
1082                    parse_identity_columns(energy_line, "energies", 1, 2, 3, header.spec_version)?;
1083                validate_section_atom_identity("energies", atom_idx, fixed, atom_id, atom_data)?;
1084            }
1085            if atom_idx < atom_data.len() {
1086                atom_data[atom_idx].energy = Some(vals[0]);
1087            }
1088            atom_idx += 1;
1089        }
1090    }
1091
1092    Ok(true)
1093}
1094
1095/// Parses declared sections from a frame's header metadata.
1096///
1097/// If `header.sections` is non-empty (v2 file with `"sections"` key in JSON),
1098/// parses each declared section in order. Otherwise falls back to legacy
1099/// blank-separator velocity detection.
1100pub fn parse_declared_sections<'a>(
1101    lines: &mut impl LineStream<'a>,
1102    header: &mut FrameHeader,
1103    atom_data: &mut [AtomDatum],
1104) -> Result<usize, ParseError> {
1105    let mut applied = 0usize;
1106    if !header.sections_declared && header.sections.is_empty() {
1107        // Legacy: try velocity detection via blank separator
1108        let found = parse_velocity_section(lines, header, atom_data)?;
1109        if found {
1110            header.sections.push(SECTION_VELOCITIES.into());
1111            applied = 1;
1112        }
1113    } else {
1114        let sections = std::mem::take(&mut header.sections);
1115        for section in &sections {
1116            match section.as_str() {
1117                SECTION_VELOCITIES => {
1118                    let found = parse_velocity_section(lines, header, atom_data)?;
1119                    if !found {
1120                        return Err(ParseError::IncompleteVelocitySection);
1121                    }
1122                    applied += 1;
1123                }
1124                SECTION_FORCES => {
1125                    let found = parse_force_section(lines, header, atom_data)?;
1126                    if !found {
1127                        return Err(ParseError::IncompleteForceSection);
1128                    }
1129                    applied += 1;
1130                }
1131                SECTION_ENERGIES => {
1132                    let found = parse_energy_section(lines, header, atom_data)?;
1133                    if !found {
1134                        return Err(ParseError::IncompleteEnergySection);
1135                    }
1136                    applied += 1;
1137                }
1138                SECTION_CHARGES => {
1139                    let found = parse_charge_section(lines, header, atom_data)?;
1140                    if !found {
1141                        return Err(ParseError::IncompleteSection(SECTION_CHARGES.into()));
1142                    }
1143                    applied += 1;
1144                }
1145                SECTION_SPINS => {
1146                    let found = parse_spin_section(lines, header, atom_data)?;
1147                    if !found {
1148                        return Err(ParseError::IncompleteSection(SECTION_SPINS.into()));
1149                    }
1150                    applied += 1;
1151                }
1152                SECTION_MAGMOMS => {
1153                    let found = parse_magmom_section(lines, header, atom_data)?;
1154                    if !found {
1155                        return Err(ParseError::IncompleteSection(SECTION_MAGMOMS.into()));
1156                    }
1157                    applied += 1;
1158                }
1159                SECTION_DISPLACEMENTS => {
1160                    let found = parse_displacement_section(lines, header, atom_data)?;
1161                    if !found {
1162                        return Err(ParseError::IncompleteSection(SECTION_DISPLACEMENTS.into()));
1163                    }
1164                    applied += 1;
1165                }
1166                SECTION_SPREADS => {
1167                    let found = parse_spread_section(lines, header, atom_data)?;
1168                    if !found {
1169                        return Err(ParseError::IncompleteSection(SECTION_SPREADS.into()));
1170                    }
1171                    applied += 1;
1172                }
1173                other => return Err(ParseError::UnknownSection(other.to_string())),
1174            }
1175        }
1176        header.sections = sections;
1177    }
1178    Ok(applied)
1179}
1180
1181/// Scalar per-atom section: blank separator + per-type blocks
1182/// (`symbol`, `"X of Component N"`, lines `value fixed atom_id`).
1183fn parse_scalar_atom_section<'a>(
1184    lines: &mut impl LineStream<'a>,
1185    header: &FrameHeader,
1186    atom_data: &mut [AtomDatum],
1187    section_name: &str,
1188    component_label: &str,
1189    mut set_value: impl FnMut(&mut AtomDatum, f64),
1190) -> Result<bool, ParseError> {
1191    let validate = header.strict_validation;
1192    match lines.peek_line() {
1193        Some(line) if line.trim().is_empty() => {
1194            lines.next_line();
1195        }
1196        _ => return Ok(false),
1197    }
1198
1199    let mut atom_idx: usize = 0;
1200    for (type_idx, &num_atoms) in header.natms_per_type.iter().enumerate() {
1201        let symbol = lines
1202            .next_line()
1203            .ok_or_else(|| ParseError::IncompleteSection(section_name.into()))?
1204            .trim();
1205
1206        let comp_line = lines
1207            .next_line()
1208            .ok_or_else(|| ParseError::IncompleteSection(section_name.into()))?;
1209        if !comp_line.contains(component_label) {
1210            return Err(ParseError::IncompleteSection(section_name.into()));
1211        }
1212        if validate {
1213            validate_section_component(
1214                component_label.trim_end_matches(" of Component").trim(),
1215                type_idx,
1216                atom_idx,
1217                symbol,
1218                comp_line,
1219                header,
1220                atom_data,
1221            )?;
1222        }
1223
1224        for _ in 0..num_atoms {
1225            let data_line = lines
1226                .next_line()
1227                .ok_or_else(|| ParseError::IncompleteSection(section_name.into()))?;
1228            let defaults = [0.0, 0.0, atom_idx as f64];
1229            let vals = parse_line_of_range_f64(data_line, 1, 3, &defaults)?;
1230            if validate {
1231                let (fixed, atom_id) =
1232                    parse_identity_columns(data_line, section_name, 1, 2, 3, header.spec_version)?;
1233                validate_section_atom_identity(section_name, atom_idx, fixed, atom_id, atom_data)?;
1234            }
1235            if atom_idx < atom_data.len() {
1236                set_value(&mut atom_data[atom_idx], vals[0]);
1237            }
1238            atom_idx += 1;
1239        }
1240    }
1241    Ok(true)
1242}
1243
1244pub fn parse_charge_section<'a>(
1245    lines: &mut impl LineStream<'a>,
1246    header: &FrameHeader,
1247    atom_data: &mut [AtomDatum],
1248) -> Result<bool, ParseError> {
1249    parse_scalar_atom_section(
1250        lines,
1251        header,
1252        atom_data,
1253        SECTION_CHARGES,
1254        "Charges of Component",
1255        |a, v| a.charge = Some(v),
1256    )
1257}
1258
1259pub fn parse_spin_section<'a>(
1260    lines: &mut impl LineStream<'a>,
1261    header: &FrameHeader,
1262    atom_data: &mut [AtomDatum],
1263) -> Result<bool, ParseError> {
1264    parse_scalar_atom_section(
1265        lines,
1266        header,
1267        atom_data,
1268        SECTION_SPINS,
1269        "Spins of Component",
1270        |a, v| a.spin = Some(v),
1271    )
1272}
1273
1274/// Magmoms: 3-vector per atom, same layout as velocities/forces.
1275pub fn parse_magmom_section<'a>(
1276    lines: &mut impl LineStream<'a>,
1277    header: &FrameHeader,
1278    atom_data: &mut [AtomDatum],
1279) -> Result<bool, ParseError> {
1280    let validate = header.strict_validation;
1281    match lines.peek_line() {
1282        Some(line) if line.trim().is_empty() => {
1283            lines.next_line();
1284        }
1285        _ => return Ok(false),
1286    }
1287
1288    let mut atom_idx: usize = 0;
1289    for (type_idx, &num_atoms) in header.natms_per_type.iter().enumerate() {
1290        let symbol = lines
1291            .next_line()
1292            .ok_or_else(|| ParseError::IncompleteSection(SECTION_MAGMOMS.into()))?
1293            .trim();
1294
1295        let comp_line = lines
1296            .next_line()
1297            .ok_or_else(|| ParseError::IncompleteSection(SECTION_MAGMOMS.into()))?;
1298        if !comp_line.contains("Magmoms of Component") {
1299            return Err(ParseError::IncompleteSection(SECTION_MAGMOMS.into()));
1300        }
1301        if validate {
1302            validate_section_component(
1303                "Magmoms", type_idx, atom_idx, symbol, comp_line, header, atom_data,
1304            )?;
1305        }
1306
1307        for _ in 0..num_atoms {
1308            let mm_line = lines
1309                .next_line()
1310                .ok_or_else(|| ParseError::IncompleteSection(SECTION_MAGMOMS.into()))?;
1311            let defaults = [0.0, 0.0, 0.0, 0.0, atom_idx as f64];
1312            let mut vals = [0.0f64; 5];
1313            parse_line_of_range_f64_stack(mm_line, 4, 5, &defaults, &mut vals)?;
1314            if validate {
1315                let (fixed, atom_id) =
1316                    parse_identity_columns(mm_line, SECTION_MAGMOMS, 3, 4, 5, header.spec_version)?;
1317                validate_section_atom_identity(
1318                    SECTION_MAGMOMS,
1319                    atom_idx,
1320                    fixed,
1321                    atom_id,
1322                    atom_data,
1323                )?;
1324            }
1325            if atom_idx < atom_data.len() {
1326                atom_data[atom_idx].magmom = Some([vals[0], vals[1], vals[2]]);
1327            }
1328            atom_idx += 1;
1329        }
1330    }
1331    Ok(true)
1332}
1333
1334/// Displacements: 3-vector per atom (Angstrom), same layout as velocities/forces.
1335pub fn parse_displacement_section<'a>(
1336    lines: &mut impl LineStream<'a>,
1337    header: &FrameHeader,
1338    atom_data: &mut [AtomDatum],
1339) -> Result<bool, ParseError> {
1340    let validate = header.strict_validation;
1341    match lines.peek_line() {
1342        Some(line) if line.trim().is_empty() => {
1343            lines.next_line();
1344        }
1345        _ => return Ok(false),
1346    }
1347
1348    let mut atom_idx: usize = 0;
1349    for (type_idx, &num_atoms) in header.natms_per_type.iter().enumerate() {
1350        let symbol = lines
1351            .next_line()
1352            .ok_or_else(|| ParseError::IncompleteSection(SECTION_DISPLACEMENTS.into()))?
1353            .trim();
1354
1355        let comp_line = lines
1356            .next_line()
1357            .ok_or_else(|| ParseError::IncompleteSection(SECTION_DISPLACEMENTS.into()))?;
1358        if !comp_line.contains("Displacements of Component") {
1359            return Err(ParseError::IncompleteSection(SECTION_DISPLACEMENTS.into()));
1360        }
1361        if validate {
1362            validate_section_component(
1363                "Displacements",
1364                type_idx,
1365                atom_idx,
1366                symbol,
1367                comp_line,
1368                header,
1369                atom_data,
1370            )?;
1371        }
1372
1373        for _ in 0..num_atoms {
1374            let dsp_line = lines
1375                .next_line()
1376                .ok_or_else(|| ParseError::IncompleteSection(SECTION_DISPLACEMENTS.into()))?;
1377            let defaults = [0.0, 0.0, 0.0, 0.0, atom_idx as f64];
1378            let mut vals = [0.0f64; 5];
1379            parse_line_of_range_f64_stack(dsp_line, 4, 5, &defaults, &mut vals)?;
1380            if validate {
1381                let (fixed, atom_id) = parse_identity_columns(
1382                    dsp_line,
1383                    SECTION_DISPLACEMENTS,
1384                    3,
1385                    4,
1386                    5,
1387                    header.spec_version,
1388                )?;
1389                validate_section_atom_identity(
1390                    SECTION_DISPLACEMENTS,
1391                    atom_idx,
1392                    fixed,
1393                    atom_id,
1394                    atom_data,
1395                )?;
1396            }
1397            if atom_idx < atom_data.len() {
1398                atom_data[atom_idx].displacement = Some([vals[0], vals[1], vals[2]]);
1399            }
1400            atom_idx += 1;
1401        }
1402    }
1403    Ok(true)
1404}
1405
1406/// Spreads: per-atom root-mean-square spread `[sx, sy, sz]` (Angstrom, each
1407/// non-negative), same layout as velocities/forces.
1408pub fn parse_spread_section<'a>(
1409    lines: &mut impl LineStream<'a>,
1410    header: &FrameHeader,
1411    atom_data: &mut [AtomDatum],
1412) -> Result<bool, ParseError> {
1413    let validate = header.strict_validation;
1414    match lines.peek_line() {
1415        Some(line) if line.trim().is_empty() => {
1416            lines.next_line();
1417        }
1418        _ => return Ok(false),
1419    }
1420
1421    let mut atom_idx: usize = 0;
1422    for (type_idx, &num_atoms) in header.natms_per_type.iter().enumerate() {
1423        let symbol = lines
1424            .next_line()
1425            .ok_or_else(|| ParseError::IncompleteSection(SECTION_SPREADS.into()))?
1426            .trim();
1427
1428        let comp_line = lines
1429            .next_line()
1430            .ok_or_else(|| ParseError::IncompleteSection(SECTION_SPREADS.into()))?;
1431        if !comp_line.contains("Spreads of Component") {
1432            return Err(ParseError::IncompleteSection(SECTION_SPREADS.into()));
1433        }
1434        if validate {
1435            validate_section_component(
1436                "Spreads", type_idx, atom_idx, symbol, comp_line, header, atom_data,
1437            )?;
1438        }
1439
1440        for _ in 0..num_atoms {
1441            let spr_line = lines
1442                .next_line()
1443                .ok_or_else(|| ParseError::IncompleteSection(SECTION_SPREADS.into()))?;
1444            let defaults = [0.0, 0.0, 0.0, 0.0, atom_idx as f64];
1445            let mut vals = [0.0f64; 5];
1446            parse_line_of_range_f64_stack(spr_line, 4, 5, &defaults, &mut vals)?;
1447            if !crate::types::spread_row_ok([vals[0], vals[1], vals[2]]) {
1448                return Err(crate::types::invalid_spread(atom_idx));
1449            }
1450            if validate {
1451                let (fixed, atom_id) = parse_identity_columns(
1452                    spr_line,
1453                    SECTION_SPREADS,
1454                    3,
1455                    4,
1456                    5,
1457                    header.spec_version,
1458                )?;
1459                validate_section_atom_identity(
1460                    SECTION_SPREADS,
1461                    atom_idx,
1462                    fixed,
1463                    atom_id,
1464                    atom_data,
1465                )?;
1466            }
1467            if atom_idx < atom_data.len() {
1468                atom_data[atom_idx].spread = Some([vals[0], vals[1], vals[2]]);
1469            }
1470            atom_idx += 1;
1471        }
1472    }
1473    Ok(true)
1474}
1475
1476#[cfg(test)]
1477mod tests {
1478    use super::*;
1479    use crate::iterators::ConFrameIterator;
1480
1481    #[test]
1482    fn test_parse_line_of_n_success() {
1483        let line = "1.0 2.5 -3.0";
1484        let values = parse_line_of_n::<f64>(line, 3).unwrap();
1485        assert_eq!(values, vec![1.0, 2.5, -3.0]);
1486    }
1487
1488    #[test]
1489    fn test_parse_line_of_n_too_short() {
1490        let line = "1.0 2.5";
1491        let result = parse_line_of_n::<f64>(line, 3);
1492        assert!(result.is_err());
1493        assert!(matches!(
1494            result.unwrap_err(),
1495            ParseError::InvalidVectorLength {
1496                expected: 3,
1497                found: 2
1498            }
1499        ));
1500    }
1501
1502    #[test]
1503    fn test_parse_line_of_n_too_long() {
1504        let line = "1.0 2.5 -3.0 4.0";
1505        let result = parse_line_of_n::<f64>(line, 3);
1506        assert!(result.is_err());
1507        assert!(matches!(
1508            result.unwrap_err(),
1509            ParseError::InvalidVectorLength {
1510                expected: 3,
1511                found: 4
1512            }
1513        ));
1514    }
1515
1516    #[test]
1517    fn test_parse_line_of_n_invalid_float() {
1518        let line = "1.0 abc -3.0";
1519        let result = parse_line_of_n::<f64>(line, 3);
1520        assert!(result.is_err());
1521        assert!(matches!(
1522            result.unwrap_err(),
1523            ParseError::InvalidNumberFormat(_)
1524        ));
1525    }
1526
1527    #[test]
1528    fn test_parse_frame_header_success() {
1529        let lines = [
1530            "PREBOX1",
1531            "{\"con_spec_version\":2}",
1532            "10.0 20.0 30.0",
1533            "90.0 90.0 90.0",
1534            "POSTBOX1",
1535            "POSTBOX2",
1536            "2",
1537            "1 1",
1538            "12.011 1.008",
1539        ];
1540        let mut line_it = lines.iter().copied();
1541        match parse_frame_header(&mut line_it) {
1542            Ok(header) => {
1543                assert_eq!(header.prebox_header.user, "PREBOX1");
1544                assert_eq!(header.spec_version, 2);
1545                assert_eq!(header.boxl, [10.0, 20.0, 30.0]);
1546                assert_eq!(header.angles, [90.0, 90.0, 90.0]);
1547                assert_eq!(header.postbox_header, ["POSTBOX1", "POSTBOX2"]);
1548                assert_eq!(header.natm_types, 2);
1549                assert_eq!(header.natms_per_type, vec![1, 1]);
1550                assert_eq!(header.masses_per_type, vec![12.011, 1.008]);
1551            }
1552            Err(e) => {
1553                panic!(
1554                    "Parsing failed when it should have succeeded. Error: {:?}",
1555                    e
1556                );
1557            }
1558        }
1559    }
1560
1561    #[test]
1562    fn test_parse_frame_header_missing_line() {
1563        let lines = [
1564            "PREBOX1",
1565            "{\"con_spec_version\":2}",
1566            "10.0 20.0 30.0",
1567            "90.0 90.0 90.0",
1568            "POSTBOX1",
1569            "POSTBOX2",
1570            "2",
1571            "1 1",
1572            // Missing masses_per_type
1573        ];
1574        let mut line_it = lines.iter().copied();
1575        let result = parse_frame_header(&mut line_it);
1576        assert!(result.is_err());
1577        assert!(matches!(result.unwrap_err(), ParseError::IncompleteHeader));
1578    }
1579
1580    #[test]
1581    fn test_parse_frame_header_missing_spec_version() {
1582        let lines = vec![
1583            "PREBOX1",
1584            "{}",
1585            "10.0 20.0 30.0",
1586            "90.0 90.0 90.0",
1587            "POSTBOX1",
1588            "POSTBOX2",
1589            "2",
1590            "1 1",
1591            "12.011 1.008",
1592        ];
1593        let mut line_it = lines.iter().copied();
1594        let result = parse_frame_header(&mut line_it);
1595        assert!(result.is_err());
1596        assert!(matches!(
1597            result.unwrap_err(),
1598            ParseError::MissingSpecVersion
1599        ));
1600    }
1601
1602    #[test]
1603    fn test_parse_frame_header_legacy_no_json() {
1604        // A non-JSON line 2 is treated as a legacy (v1) file.
1605        let lines = vec![
1606            "PREBOX1",
1607            "0.0000 TIME",
1608            "10.0 20.0 30.0",
1609            "90.0 90.0 90.0",
1610            "POSTBOX1",
1611            "POSTBOX2",
1612            "2",
1613            "1 1",
1614            "12.011 1.008",
1615        ];
1616        let mut line_it = lines.iter().copied();
1617        let header = parse_frame_header(&mut line_it).unwrap();
1618        assert_eq!(header.spec_version, 1);
1619        assert!(header.metadata.is_empty());
1620    }
1621
1622    #[test]
1623    fn test_parse_frame_header_malformed_json() {
1624        // Line 2 starts with '{' but is not valid JSON -- this IS an error.
1625        let lines = vec![
1626            "PREBOX1",
1627            "{broken json",
1628            "10.0 20.0 30.0",
1629            "90.0 90.0 90.0",
1630            "POSTBOX1",
1631            "POSTBOX2",
1632            "2",
1633            "1 1",
1634            "12.011 1.008",
1635        ];
1636        let mut line_it = lines.iter().copied();
1637        let result = parse_frame_header(&mut line_it);
1638        assert!(result.is_err());
1639        assert!(matches!(
1640            result.unwrap_err(),
1641            ParseError::InvalidMetadataJson(_)
1642        ));
1643    }
1644
1645    #[test]
1646    fn test_parse_frame_header_unsupported_version() {
1647        let lines = vec![
1648            "PREBOX1",
1649            "{\"con_spec_version\":999}",
1650            "10.0 20.0 30.0",
1651            "90.0 90.0 90.0",
1652            "POSTBOX1",
1653            "POSTBOX2",
1654            "2",
1655            "1 1",
1656            "12.011 1.008",
1657        ];
1658        let mut line_it = lines.iter().copied();
1659        let result = parse_frame_header(&mut line_it);
1660        assert!(result.is_err());
1661        assert!(matches!(
1662            result.unwrap_err(),
1663            ParseError::UnsupportedSpecVersion(999)
1664        ));
1665    }
1666
1667    #[test]
1668    fn test_v3_missing_units_rejected() {
1669        let lines = vec![
1670            "PREBOX1",
1671            "{\"con_spec_version\":3}",
1672            "10.0 20.0 30.0",
1673            "90.0 90.0 90.0",
1674            "POSTBOX1",
1675            "POSTBOX2",
1676            "1",
1677            "1",
1678            "1.0",
1679        ];
1680        let mut line_it = lines.iter().copied();
1681        let err = parse_frame_header(&mut line_it).unwrap_err();
1682        let msg = err.to_string();
1683        assert!(
1684            msg.contains("units") || msg.contains("v3"),
1685            "expected units error, got {msg}"
1686        );
1687    }
1688
1689    #[test]
1690    fn test_v3_invalid_units_rejected() {
1691        let lines = vec![
1692            "PREBOX1",
1693            r#"{"con_spec_version":3,"units":{"length":"eV","energy":"angstrom"}}"#,
1694            "10.0 20.0 30.0",
1695            "90.0 90.0 90.0",
1696            "POSTBOX1",
1697            "POSTBOX2",
1698            "1",
1699            "1",
1700            "1.0",
1701        ];
1702        let mut line_it = lines.iter().copied();
1703        assert!(parse_frame_header(&mut line_it).is_err());
1704    }
1705
1706    #[test]
1707    fn test_v3_valid_units_exposes_length_energy() {
1708        let lines = vec![
1709            "PREBOX1",
1710            r#"{"con_spec_version":3,"units":{"length":"angstrom","energy":"eV","mass":"amu","time":"fs"}}"#,
1711            "10.0 20.0 30.0",
1712            "90.0 90.0 90.0",
1713            "POSTBOX1",
1714            "POSTBOX2",
1715            "1",
1716            "1",
1717            "1.0",
1718        ];
1719        let mut line_it = lines.iter().copied();
1720        let header = parse_frame_header(&mut line_it).unwrap();
1721        assert_eq!(header.spec_version, 3);
1722        assert_eq!(header.length_unit(), Some("angstrom"));
1723        assert_eq!(header.energy_unit(), Some("eV"));
1724        let f = header.conversion_factor_to("length", "nm").unwrap();
1725        assert!((f - 0.1).abs() < 1e-12);
1726    }
1727
1728    #[test]
1729    fn test_parse_frame_header_extra_metadata_preserved() {
1730        let lines = vec![
1731            "PREBOX1",
1732            "{\"con_spec_version\":2,\"generator\":\"test\"}",
1733            "10.0 20.0 30.0",
1734            "90.0 90.0 90.0",
1735            "POSTBOX1",
1736            "POSTBOX2",
1737            "2",
1738            "1 1",
1739            "12.011 1.008",
1740        ];
1741        let mut line_it = lines.iter().copied();
1742        let header = parse_frame_header(&mut line_it).unwrap();
1743        assert_eq!(header.spec_version, 2);
1744        assert_eq!(
1745            header.metadata.get("generator"),
1746            Some(&serde_json::Value::String("test".to_string()))
1747        );
1748    }
1749
1750    #[test]
1751    fn test_parse_frame_header_invalid_natms_per_type() {
1752        let lines = vec![
1753            "PREBOX1",
1754            "{\"con_spec_version\":2}",
1755            "10.0 20.0 30.0",
1756            "90.0 90.0 90.0",
1757            "POSTBOX1",
1758            "POSTBOX2",
1759            "2",
1760            "1 1 1", // 3 values, but natm_types is 2
1761            "12.011 1.008",
1762        ];
1763        let mut line_it = lines.iter().copied();
1764        let result = parse_frame_header(&mut line_it);
1765        assert!(result.is_err());
1766        assert!(matches!(
1767            result.unwrap_err(),
1768            ParseError::InvalidVectorLength {
1769                expected: 2,
1770                found: 3
1771            }
1772        ));
1773    }
1774
1775    #[test]
1776    fn test_parse_single_frame_success() {
1777        let lines = vec![
1778            "PREBOX1",
1779            "{\"con_spec_version\":2}",
1780            "10.0 20.0 30.0",
1781            "90.0 90.0 90.0",
1782            "POSTBOX1",
1783            "POSTBOX2",
1784            "2",
1785            "3 3",
1786            "12.011 1.008",
1787            "1",
1788            "Coordinates of Component 1",
1789            "0.0 0.0 0.0 0.0 1",
1790            "1.0940 0.0 0.0 0.0 2",
1791            "-0.5470 0.9499 0.0 0.0 3",
1792            "2",
1793            "Coordinates of Component 2",
1794            "5.0 5.0 5.0 0.0 4",
1795            "6.0940 5.0 5.0 0.0 5",
1796            "5.5470 5.9499 5.0 0.0 6",
1797        ];
1798        let mut line_it = lines.iter().copied();
1799        let frame = parse_single_frame(&mut line_it).unwrap();
1800
1801        assert_eq!(frame.header.natm_types, 2);
1802        assert_eq!(frame.header.natms_per_type, vec![3, 3]);
1803        assert_eq!(frame.header.masses_per_type, vec![12.011, 1.008]);
1804        assert_eq!(frame.atom_data.len(), 6);
1805        assert_eq!(&*frame.atom_data[0].symbol, "1");
1806        assert_eq!(frame.atom_data[0].atom_id, 1);
1807        assert_eq!(&*frame.atom_data[5].symbol, "2");
1808        assert_eq!(frame.atom_data[5].atom_id, 6);
1809    }
1810
1811    #[test]
1812    fn test_column_4_value_1_is_x_only_on_spec_2() {
1813        let lines = vec![
1814            "Generated by probe",
1815            "{\"con_spec_version\":2}",
1816            "20.0 20.0 20.0",
1817            "90.0 90.0 90.0",
1818            "",
1819            "",
1820            "1",
1821            "1",
1822            "63.5",
1823            "Cu",
1824            "Coordinates of component   1",
1825            "1.0 2.0 3.0 1 1",
1826        ];
1827        let mut line_it = lines.iter().copied();
1828        let frame = parse_single_frame(&mut line_it).unwrap();
1829        assert_eq!(frame.header.spec_version, 2);
1830        assert_eq!(frame.atom_data[0].fixed, [true, false, false]);
1831    }
1832
1833    #[test]
1834    fn test_column_4_value_1_is_all_fixed_on_spec_1() {
1835        let lines = vec![
1836            "Generated by probe",
1837            "0.0000 TIME",
1838            "20.0 20.0 20.0",
1839            "90.0 90.0 90.0",
1840            "",
1841            "",
1842            "1",
1843            "1",
1844            "63.5",
1845            "Cu",
1846            "Coordinates of component   1",
1847            "1.0 2.0 3.0 1 1",
1848        ];
1849        let mut line_it = lines.iter().copied();
1850        let frame = parse_single_frame(&mut line_it).unwrap();
1851        assert_eq!(frame.header.spec_version, 1);
1852        assert_eq!(frame.atom_data[0].fixed, [true, true, true]);
1853    }
1854
1855    #[test]
1856    fn test_parse_single_frame_missing_line() {
1857        // With a valid header but truncated atom data, we get IncompleteFrame.
1858        let lines = vec![
1859            "PREBOX1",
1860            "{\"con_spec_version\":2}",
1861            "10.0 20.0 30.0",
1862            "90.0 90.0 90.0",
1863            "POSTBOX1",
1864            "POSTBOX2",
1865            "2",
1866            "3 3",
1867            "12.011 1.008",
1868            "1",
1869            "Coordinates of Component 1",
1870            "0.0 0.0 0.0 0.0 1",
1871            "1.0940 0.0 0.0 0.0 2",
1872            "-0.5470 0.9499 0.0 0.0 3",
1873            // Missing Component 2 entirely
1874        ];
1875        let mut line_it = lines.iter().copied();
1876        let result = parse_single_frame(&mut line_it);
1877        assert!(result.is_err());
1878        assert!(matches!(result.unwrap_err(), ParseError::IncompleteFrame));
1879    }
1880
1881    #[test]
1882    fn test_parse_single_frame_missing_atom_index_defaults_sequential() {
1883        // Column 5 (atom_index) is optional; when absent, defaults to sequential.
1884        let lines = vec![
1885            "PREBOX1",
1886            "{\"con_spec_version\":2}",
1887            "10.0 20.0 30.0",
1888            "90.0 90.0 90.0",
1889            "POSTBOX1",
1890            "POSTBOX2",
1891            "2",
1892            "3 3",
1893            "12.011 1.008",
1894            "1",
1895            "Coordinates of Component 1",
1896            "0.0 0.0 0.0 0.0 1",
1897            "1.0940 0.0 0.0 0.0 2",
1898            "-0.5470 0.9499 0.0 0.0 3",
1899            "2",
1900            "Coordinates of Component 2",
1901            "5.0 5.0 5.0 0.0",       // No atom_index: defaults to 3
1902            "6.0940 5.0 5.0 0.0 10", // Explicit atom_index: 10
1903            "5.5470 5.9499 5.0 0.0", // No atom_index: defaults to 5
1904        ];
1905        let mut line_it = lines.iter().copied();
1906        let frame = parse_single_frame(&mut line_it).unwrap();
1907        assert_eq!(frame.atom_data.len(), 6);
1908        // First type: explicit atom_index values
1909        assert_eq!(frame.atom_data[0].atom_id, 1);
1910        assert_eq!(frame.atom_data[1].atom_id, 2);
1911        assert_eq!(frame.atom_data[2].atom_id, 3);
1912        // Second type: mixed explicit and defaulted
1913        assert_eq!(frame.atom_data[3].atom_id, 3); // defaulted (global idx 3)
1914        assert_eq!(frame.atom_data[4].atom_id, 10); // explicit
1915        assert_eq!(frame.atom_data[5].atom_id, 5); // defaulted (global idx 5)
1916    }
1917
1918    #[test]
1919    fn test_parse_single_frame_too_few_columns_fails() {
1920        // Only 3 columns (missing fixed_flag too) should still fail.
1921        let lines = vec![
1922            "PREBOX1",
1923            "{\"con_spec_version\":2}",
1924            "10.0 20.0 30.0",
1925            "90.0 90.0 90.0",
1926            "POSTBOX1",
1927            "POSTBOX2",
1928            "1",
1929            "1",
1930            "12.011",
1931            "C",
1932            "Coordinates of Component 1",
1933            "0.0 0.0 0.0", // Only 3 values
1934        ];
1935        let mut line_it = lines.iter().copied();
1936        let result = parse_single_frame(&mut line_it);
1937        assert!(result.is_err());
1938        assert!(matches!(
1939            result.unwrap_err(),
1940            ParseError::InvalidVectorLength {
1941                expected: 5,
1942                found: 3
1943            }
1944        ));
1945    }
1946
1947    #[test]
1948    fn test_parse_velocity_section_present() {
1949        let lines = vec![
1950            "PREBOX1",
1951            "{\"con_spec_version\":2}",
1952            "10.0 20.0 30.0",
1953            "90.0 90.0 90.0",
1954            "POSTBOX1",
1955            "POSTBOX2",
1956            "2",
1957            "1 1",
1958            "63.546 1.008",
1959            "Cu",
1960            "Coordinates of Component 1",
1961            "0.0 0.0 0.0 1.0 0",
1962            "H",
1963            "Coordinates of Component 2",
1964            "1.0 2.0 3.0 0.0 1",
1965            "",
1966            "Cu",
1967            "Velocities of Component 1",
1968            "0.1 0.2 0.3 1.0 0",
1969            "H",
1970            "Velocities of Component 2",
1971            "0.4 0.5 0.6 0.0 1",
1972        ];
1973        let mut line_it = lines.iter().copied().peekable();
1974        // Parse the frame first (consuming 15 lines)
1975        let mut frame =
1976            parse_single_frame(&mut line_it).expect("coordinate parsing should succeed");
1977        assert!(!frame.has_velocities());
1978
1979        // Now parse the velocity section
1980        let has_vel = parse_velocity_section(&mut line_it, &frame.header, &mut frame.atom_data)
1981            .expect("velocity parsing should succeed");
1982        assert!(has_vel);
1983        assert_eq!(frame.atom_data[0].velocity, Some([0.1, 0.2, 0.3]));
1984        assert_eq!(frame.atom_data[1].velocity, Some([0.4, 0.5, 0.6]));
1985    }
1986
1987    #[test]
1988    fn test_validate_true_accepts_matching_section_identity() {
1989        let text = r#"
1990PREBOX1
1991{"con_spec_version":2,"sections":["velocities"],"validate":true}
199210.0 20.0 30.0
199390.0 90.0 90.0
1994POSTBOX1
1995POSTBOX2
19962
19971 1
199863.546 1.008
1999Cu
2000Coordinates of Component 1
20010.0 0.0 0.0 5 0
2002H
2003Coordinates of Component 2
20041.0 2.0 3.0 0 1
2005
2006Cu
2007Velocities of Component 1
20080.1 0.2 0.3 5 0
2009H
2010Velocities of Component 2
20110.4 0.5 0.6 0 1
2012"#;
2013        let mut iter = ConFrameIterator::new(text.trim());
2014        let frame = iter.next().unwrap().unwrap();
2015
2016        assert!(frame.has_velocities());
2017        assert_eq!(
2018            frame
2019                .header
2020                .metadata
2021                .get("validate")
2022                .and_then(|v| v.as_bool()),
2023            Some(true)
2024        );
2025    }
2026
2027    #[test]
2028    fn test_validate_true_rejects_section_atom_id_mismatch() {
2029        let text = r#"
2030PREBOX1
2031{"con_spec_version":2,"sections":["velocities"],"validate":true}
203210.0 20.0 30.0
203390.0 90.0 90.0
2034POSTBOX1
2035POSTBOX2
20361
20371
203863.546
2039Cu
2040Coordinates of Component 1
20410.0 0.0 0.0 5 0
2042
2043Cu
2044Velocities of Component 1
20450.1 0.2 0.3 5 99
2046"#;
2047        let mut iter = ConFrameIterator::new(text.trim());
2048        let err = iter.next().unwrap().unwrap_err();
2049
2050        assert!(matches!(err, ParseError::ValidationError(_)));
2051        assert!(err.to_string().contains("atom_id mismatch"));
2052    }
2053
2054    #[test]
2055    fn test_validate_true_rejects_section_symbol_mismatch() {
2056        let text = r#"
2057PREBOX1
2058{"con_spec_version":2,"sections":["forces"],"validate":true}
205910.0 20.0 30.0
206090.0 90.0 90.0
2061POSTBOX1
2062POSTBOX2
20631
20641
206563.546
2066Cu
2067Coordinates of Component 1
20680.0 0.0 0.0 5 0
2069
2070H
2071Forces of Component 1
20720.1 0.2 0.3 5 0
2073"#;
2074        let mut iter = ConFrameIterator::new(text.trim());
2075        let err = iter.next().unwrap().unwrap_err();
2076
2077        assert!(matches!(err, ParseError::ValidationError(_)));
2078        assert!(err.to_string().contains("symbol mismatch"));
2079    }
2080
2081    #[test]
2082    fn test_validate_absent_allows_legacy_duplicate_identity_mismatch() {
2083        let text = r#"
2084PREBOX1
2085{"con_spec_version":2,"sections":["velocities"]}
208610.0 20.0 30.0
208790.0 90.0 90.0
2088POSTBOX1
2089POSTBOX2
20901
20911
209263.546
2093Cu
2094Coordinates of Component 1
20950.0 0.0 0.0 5 0
2096
2097Cu
2098Velocities of Component 1
20990.1 0.2 0.3 0 99
2100"#;
2101        let mut iter = ConFrameIterator::new(text.trim());
2102        let frame = iter.next().unwrap().unwrap();
2103
2104        assert!(frame.has_velocities());
2105        assert_eq!(frame.atom_data[0].atom_id, 0);
2106        assert_eq!(frame.atom_data[0].fixed, [true, false, true]);
2107    }
2108
2109    #[test]
2110    fn test_validate_must_be_boolean_when_present() {
2111        let lines = vec![
2112            "PREBOX1",
2113            "{\"con_spec_version\":2,\"validate\":\"yes\",\"sections\":[]}",
2114            "10.0 20.0 30.0",
2115            "90.0 90.0 90.0",
2116            "POSTBOX1",
2117            "POSTBOX2",
2118            "1",
2119            "1",
2120            "12.011",
2121        ];
2122        let mut line_it = lines.iter().copied();
2123        let err = parse_frame_header(&mut line_it).unwrap_err();
2124
2125        assert!(matches!(err, ParseError::InvalidMetadataJson(_)));
2126        assert!(err.to_string().contains("validate"));
2127    }
2128
2129    #[test]
2130    fn test_validate_true_requires_sections_key() {
2131        let lines = vec![
2132            "PREBOX1",
2133            "{\"con_spec_version\":2,\"validate\":true}",
2134            "10.0 20.0 30.0",
2135            "90.0 90.0 90.0",
2136            "POSTBOX1",
2137            "POSTBOX2",
2138            "1",
2139            "1",
2140            "12.011",
2141        ];
2142        let mut line_it = lines.iter().copied();
2143        let err = parse_frame_header(&mut line_it).unwrap_err();
2144
2145        assert!(matches!(err, ParseError::InvalidMetadataJson(_)));
2146        assert!(err.to_string().contains("sections"));
2147    }
2148
2149    #[test]
2150    fn test_sections_must_be_string_array_when_present() {
2151        let lines = vec![
2152            "PREBOX1",
2153            "{\"con_spec_version\":2,\"sections\":[\"velocities\",7]}",
2154            "10.0 20.0 30.0",
2155            "90.0 90.0 90.0",
2156            "POSTBOX1",
2157            "POSTBOX2",
2158            "1",
2159            "1",
2160            "12.011",
2161        ];
2162        let mut line_it = lines.iter().copied();
2163        let err = parse_frame_header(&mut line_it).unwrap_err();
2164
2165        assert!(matches!(err, ParseError::InvalidMetadataJson(_)));
2166        assert!(err.to_string().contains("sections"));
2167    }
2168
2169    #[test]
2170    fn test_validate_true_rejects_non_integer_coordinate_identity_columns() {
2171        let text = r#"
2172PREBOX1
2173{"con_spec_version":2,"sections":[],"validate":true}
217410.0 20.0 30.0
217590.0 90.0 90.0
2176POSTBOX1
2177POSTBOX2
21781
21791
218063.546
2181Cu
2182Coordinates of Component 1
21830.0 0.0 0.0 5.0 0
2184"#;
2185        let mut iter = ConFrameIterator::new(text.trim());
2186        let err = iter.next().unwrap().unwrap_err();
2187
2188        assert!(matches!(err, ParseError::ValidationError(_)));
2189        assert!(err.to_string().contains("fixed_flag"));
2190    }
2191
2192    #[test]
2193    fn test_validate_true_rejects_non_exact_coordinate_label() {
2194        let text = r#"
2195PREBOX1
2196{"con_spec_version":2,"sections":[],"validate":true}
219710.0 20.0 30.0
219890.0 90.0 90.0
2199POSTBOX1
2200POSTBOX2
22011
22021
220363.546
2204Cu
2205Coordinates Component 1
22060.0 0.0 0.0 5 0
2207"#;
2208        let mut iter = ConFrameIterator::new(text.trim());
2209        let err = iter.next().unwrap().unwrap_err();
2210
2211        assert!(matches!(err, ParseError::ValidationError(_)));
2212        assert!(err.to_string().contains("Coordinates of Component 1"));
2213    }
2214
2215    #[test]
2216    fn test_validate_true_rejects_unknown_component_symbol() {
2217        let text = r#"
2218PREBOX1
2219{"con_spec_version":2,"sections":[],"validate":true}
222010.0 20.0 30.0
222190.0 90.0 90.0
2222POSTBOX1
2223POSTBOX2
22241
22251
222663.546
2227Qq
2228Coordinates of Component 1
22290.0 0.0 0.0 0 0
2230"#;
2231        let mut iter = ConFrameIterator::new(text.trim());
2232        let err = iter.next().unwrap().unwrap_err();
2233
2234        assert!(matches!(err, ParseError::ValidationError(_)));
2235        assert!(err.to_string().contains("symbol"));
2236    }
2237
2238    #[test]
2239    fn test_declared_section_must_be_present() {
2240        let text = r#"
2241PREBOX1
2242{"con_spec_version":2,"sections":["velocities"]}
224310.0 20.0 30.0
224490.0 90.0 90.0
2245POSTBOX1
2246POSTBOX2
22471
22481
224963.546
2250Cu
2251Coordinates of Component 1
22520.0 0.0 0.0 0 0
2253"#;
2254        let mut iter = ConFrameIterator::new(text.trim());
2255        let err = iter.next().unwrap().unwrap_err();
2256
2257        assert!(matches!(err, ParseError::IncompleteVelocitySection));
2258    }
2259
2260    #[test]
2261    fn test_non_finite_cell_geometry_rejected_in_strict_mode() {
2262        let lines = vec![
2263            "PREBOX1",
2264            "{\"con_spec_version\":2,\"sections\":[],\"validate\":true}",
2265            "10.0 NaN 30.0",
2266            "90.0 90.0 90.0",
2267            "POSTBOX1",
2268            "POSTBOX2",
2269            "1",
2270            "1",
2271            "12.011",
2272        ];
2273        let mut line_it = lines.iter().copied();
2274        let err = parse_frame_header(&mut line_it).unwrap_err();
2275
2276        assert!(matches!(err, ParseError::ValidationError(_)));
2277        assert!(err.to_string().contains("cell"));
2278    }
2279
2280    #[test]
2281    fn test_validate_true_rejects_non_physical_cell_geometry() {
2282        let lines = vec![
2283            "PREBOX1",
2284            "{\"con_spec_version\":2,\"sections\":[],\"validate\":true}",
2285            "0.0 20.0 30.0",
2286            "90.0 180.0 90.0",
2287            "POSTBOX1",
2288            "POSTBOX2",
2289            "1",
2290            "1",
2291            "12.011",
2292        ];
2293        let mut line_it = lines.iter().copied();
2294        let err = parse_frame_header(&mut line_it).unwrap_err();
2295
2296        assert!(matches!(err, ParseError::ValidationError(_)));
2297        assert!(err.to_string().contains("cell"));
2298    }
2299
2300    #[test]
2301    fn test_validate_true_rejects_reserved_metadata_type_mismatch() {
2302        let lines = vec![
2303            "PREBOX1",
2304            "{\"con_spec_version\":2,\"sections\":[],\"validate\":true,\"energy\":\"low\"}",
2305            "10.0 20.0 30.0",
2306            "90.0 90.0 90.0",
2307            "POSTBOX1",
2308            "POSTBOX2",
2309            "1",
2310            "1",
2311            "12.011",
2312        ];
2313        let mut line_it = lines.iter().copied();
2314        let err = parse_frame_header(&mut line_it).unwrap_err();
2315
2316        assert!(matches!(err, ParseError::InvalidMetadataJson(_)));
2317        assert!(err.to_string().contains("energy"));
2318    }
2319
2320    #[test]
2321    fn test_validate_true_rejects_malformed_bonds() {
2322        let lines = vec![
2323            "PREBOX1",
2324            "{\"con_spec_version\":2,\"sections\":[],\"validate\":true,\"bonds\":[[0]]}",
2325            "10.0 20.0 30.0",
2326            "90.0 90.0 90.0",
2327            "POSTBOX1",
2328            "POSTBOX2",
2329            "1",
2330            "1",
2331            "12.011",
2332        ];
2333        let mut line_it = lines.iter().copied();
2334        let err = parse_frame_header(&mut line_it).unwrap_err();
2335        assert!(matches!(err, ParseError::InvalidMetadataJson(_)));
2336        assert!(err.to_string().contains("bonds"));
2337    }
2338
2339    #[test]
2340    fn test_bonds_metadata_round_trip_in_header() {
2341        use crate::types::{Bond, meta};
2342        let lines = vec![
2343            "PREBOX1",
2344            "{\"con_spec_version\":2,\"bonds\":[[0,1],{\"i\":0,\"j\":2,\"order\":1}]}",
2345            "10.0 20.0 30.0",
2346            "90.0 90.0 90.0",
2347            "POSTBOX1",
2348            "POSTBOX2",
2349            "1",
2350            "1",
2351            "12.011",
2352        ];
2353        let mut line_it = lines.iter().copied();
2354        let header = parse_frame_header(&mut line_it).expect("header");
2355        let bonds = header.bonds();
2356        assert_eq!(bonds.len(), 2);
2357        assert_eq!(bonds[0], Bond::new(0, 1));
2358        assert_eq!(bonds[1].i, 0);
2359        assert_eq!(bonds[1].j, 2);
2360        assert_eq!(bonds[1].order, Some(1));
2361        assert!(header.metadata.contains_key(meta::BONDS));
2362    }
2363
2364    #[test]
2365    fn test_parse_velocity_section_absent() {
2366        let lines = vec![
2367            "PREBOX1",
2368            "{\"con_spec_version\":2}",
2369            "10.0 20.0 30.0",
2370            "90.0 90.0 90.0",
2371            "POSTBOX1",
2372            "POSTBOX2",
2373            "1",
2374            "1",
2375            "12.011",
2376            "C",
2377            "Coordinates of Component 1",
2378            "0.0 0.0 0.0 0.0 1",
2379        ];
2380        let mut line_it = lines.iter().copied().peekable();
2381        let mut frame = parse_single_frame(&mut line_it).expect("parse should succeed");
2382        let has_vel = parse_velocity_section(&mut line_it, &frame.header, &mut frame.atom_data)
2383            .expect("should succeed with no velocities");
2384        assert!(!has_vel);
2385        assert_eq!(frame.atom_data[0].velocity, None);
2386    }
2387
2388    #[test]
2389    fn test_parse_line_of_range_f64_exact() {
2390        let vals = parse_line_of_range_f64("1.0 2.0 3.0 0.0 42", 4, 5, &[0.0; 5]).unwrap();
2391        assert_eq!(vals, vec![1.0, 2.0, 3.0, 0.0, 42.0]);
2392    }
2393
2394    #[test]
2395    fn test_byte_scan_stack_parses_realistic_coord_line() {
2396        // Typical CON atom line: x y z fixed_mask [atom_id]
2397        let line = "   0.63939999999999997    0.90449999999999997   -0.00009999999999977 1    0";
2398        let defaults = [0.0, 0.0, 0.0, 0.0, 99.0];
2399        let mut buf = [0.0f64; 5];
2400        let n = parse_line_of_range_f64_stack(line, 4, 5, &defaults, &mut buf).unwrap();
2401        assert_eq!(n, 5);
2402        assert!((buf[0] - 0.6394).abs() < 1e-4);
2403        assert!((buf[1] - 0.9045).abs() < 1e-4);
2404        assert_eq!(buf[3] as u8, 1);
2405        assert_eq!(buf[4] as u64, 0);
2406        // trailing junk must error
2407        let bad = "1.0 2.0 3.0 1 0 EXTRA";
2408        assert!(parse_line_of_range_f64_stack(bad, 4, 5, &defaults, &mut buf).is_err());
2409        // partial non-boundary token must error
2410        let glued = "1.0 2.0 3.0abc 1 0";
2411        assert!(parse_line_of_range_f64_stack(glued, 4, 5, &defaults, &mut buf).is_err());
2412    }
2413
2414    #[test]
2415    fn test_parse_line_of_range_f64_stack_matches_vec_api() {
2416        let defaults = [0.0, 0.0, 0.0, 0.0, 7.0];
2417        let line = "1.0 2.0 3.0 0.0";
2418        let mut buf = [0.0f64; 5];
2419        parse_line_of_range_f64_stack(line, 4, 5, &defaults, &mut buf).unwrap();
2420        let via_vec = parse_line_of_range_f64(line, 4, 5, &defaults).unwrap();
2421        assert_eq!(&buf[..5], via_vec.as_slice());
2422        assert_eq!(buf[4], 7.0);
2423    }
2424
2425    #[test]
2426    fn test_parse_line_of_range_f64_padded() {
2427        let defaults = [0.0, 0.0, 0.0, 0.0, 99.0];
2428        let vals = parse_line_of_range_f64("1.0 2.0 3.0 0.0", 4, 5, &defaults).unwrap();
2429        assert_eq!(vals, vec![1.0, 2.0, 3.0, 0.0, 99.0]);
2430    }
2431
2432    #[test]
2433    fn test_parse_line_of_range_f64_too_few() {
2434        let result = parse_line_of_range_f64("1.0 2.0 3.0", 4, 5, &[0.0; 5]);
2435        assert!(result.is_err());
2436    }
2437
2438    #[test]
2439    fn test_parse_line_of_range_f64_too_many() {
2440        let result = parse_line_of_range_f64("1.0 2.0 3.0 0.0 5.0 6.0", 4, 5, &[0.0; 5]);
2441        assert!(result.is_err());
2442    }
2443
2444    #[test]
2445    fn test_parse_all_four_column_lines() {
2446        // All atom lines have only 4 columns; atom_index defaults to sequential.
2447        let lines = vec![
2448            "PREBOX1",
2449            "{\"con_spec_version\":2}",
2450            "10.0 10.0 10.0",
2451            "90.0 90.0 90.0",
2452            "POSTBOX1",
2453            "POSTBOX2",
2454            "1",
2455            "3",
2456            "12.011",
2457            "C",
2458            "Coordinates of Component 1",
2459            "0.0 0.0 0.0 0",
2460            "1.0 0.0 0.0 0",
2461            "2.0 0.0 0.0 1",
2462        ];
2463        let mut line_it = lines.iter().copied();
2464        let frame = parse_single_frame(&mut line_it).unwrap();
2465        assert_eq!(frame.atom_data[0].atom_id, 0);
2466        assert_eq!(frame.atom_data[1].atom_id, 1);
2467        assert_eq!(frame.atom_data[2].atom_id, 2);
2468        assert!(frame.atom_data[2].is_fixed());
2469    }
2470}