Skip to main content

datui_lib/
sdf.rs

1//! SDF (structure-data format) compound files, read into a table: one row per record,
2//! with the molecule's name, its atom and bond counts, and every data item
3//! (`> <FIELD>`) as a column.
4//!
5//! A record is a molfile header (name, program, comment, counts line), the connection
6//! table up to `M  END`, then data items, and ends at `$$$$`. The connection table is
7//! passed over line by line and never held. The reader takes the file a piece at a
8//! time; a line, a value and the number of fields are bounded.
9
10use std::collections::HashMap;
11use std::path::Path;
12
13use color_eyre::Result;
14use polars::prelude::*;
15
16use crate::OpenOptions;
17use crate::model_files::MetaValue;
18use crate::notes::Note;
19use crate::segments::{Converted, Segments};
20use crate::text_formats::{Detail, Pieces, capped_list, count, note};
21use crate::unfinished::Writer;
22
23/// What datui does with an SDF file: see [`crate::readers`].
24pub(crate) const READER: crate::readers::Reader = crate::readers::Reader {
25    convert: Some(|input| {
26        crate::text_formats::read_one(input, |pieces| {
27            convert(input.display, input.options, input.writer, pieces)
28        })
29    }),
30    scan: crate::readers::read_into,
31    signatures: &[crate::readers::Signature {
32        says: |head, _| looks_like(head),
33        kind: crate::readers::Kind::Text,
34        trusted: crate::readers::Trusted {
35            listing: false,
36            ..crate::readers::EVERYWHERE
37        },
38    }],
39    ..crate::readers::BASE
40};
41
42/// The longest line kept; the rest of a longer one is cut off.
43pub const MAX_LINE: usize = 1 << 20;
44/// The longest value kept, its lines together; the rest is cut off.
45pub const MAX_VALUE: usize = 1 << 20;
46/// The most data fields; one past this is left out.
47pub const MAX_FIELDS: usize = 4096;
48/// The longest field name.
49pub const MAX_NAME: usize = 256;
50/// Rows held before a batch is handed over.
51pub const BATCH_ROWS: usize = 16_384;
52/// Text held before a batch is handed over, whatever its rows. Each cell counts
53/// too, empty or not: every field is a column in every row, so a file of thousands
54/// of fields would otherwise hold gigabytes of empty cells in one batch.
55pub const BATCH_TEXT: usize = 32 << 20;
56
57/// The columns every SDF table starts with.
58pub const CORE: [&str; 3] = ["name", "atoms", "bonds"];
59
60/// Whether the first bytes look like an SDF file: a counts line (`V2000` or `V3000`)
61/// on the fourth line, or `M  END` followed by a data item or `$$$$`.
62pub fn looks_like(head: &[u8]) -> bool {
63    let text = head.strip_prefix(b"\xef\xbb\xbf").unwrap_or(head);
64    let mut lines = text.split(|&b| b == b'\n');
65    // A counts line: numbers in fixed columns, ending in its version.
66    let counts = lines.nth(3).unwrap_or_default().trim_ascii();
67    let counts_line = (counts.ends_with(b"V2000") || counts.ends_with(b"V3000"))
68        && counts[..counts.len() - 5]
69            .iter()
70            .all(|b| b.is_ascii_digit() || b.is_ascii_whitespace());
71    let has = |needle: &[u8]| text.windows(needle.len()).any(|w| w == needle);
72    counts_line || (has(b"M  END") && (has(b"$$$$") || has(b"> <")))
73}
74
75/// What one data field's values were, for its type.
76#[derive(Debug, Clone, PartialEq)]
77pub struct FieldColumn {
78    pub name: String,
79    /// Records holding a value.
80    pub values: u64,
81    /// Every value is an integer.
82    pub integers: bool,
83    /// Every value is a number.
84    pub numbers: bool,
85}
86
87impl FieldColumn {
88    /// The type the column is cast to.
89    pub fn dtype(&self) -> DataType {
90        if self.values == 0 {
91            DataType::String
92        } else if self.integers {
93            DataType::Int64
94        } else if self.numbers {
95            DataType::Float64
96        } else {
97            DataType::String
98        }
99    }
100}
101
102/// What reading noticed.
103#[derive(Debug, Clone, Default, PartialEq)]
104pub struct Stats {
105    pub records: u64,
106    /// Lines cut at [`MAX_LINE`].
107    pub long_lines: u64,
108    /// Values cut at [`MAX_VALUE`].
109    pub long_values: u64,
110    /// Data items in a record that already had the field; the first is kept.
111    pub repeated: u64,
112    /// Data items of fields past [`MAX_FIELDS`], left out.
113    pub fields_dropped: u64,
114    /// Records in the V3000 format.
115    pub v3000: u64,
116    /// Whether the last record had no `$$$$`.
117    pub unterminated: bool,
118}
119
120#[derive(Debug, Clone, Copy, PartialEq, Eq)]
121enum Place {
122    /// The molfile header: line 0 is the name, 3 the counts line.
123    Header(u8),
124    /// The connection table, up to `M  END`.
125    Ctab,
126    /// Between data items.
127    Data,
128    /// A data item's value lines, up to a blank line.
129    Value,
130}
131
132/// One record being read.
133#[derive(Debug, Default)]
134struct Record {
135    name: Option<String>,
136    atoms: Option<u32>,
137    bonds: Option<u32>,
138    values: HashMap<usize, String>,
139    /// The field whose value is being read, if it is kept.
140    field: Option<usize>,
141    value: String,
142    /// The value was cut at [`MAX_VALUE`]; its other lines are passed over.
143    cut: bool,
144    /// Whether anything but blank lines was read: blank lines alone are no record.
145    started: bool,
146}
147
148/// Reads an SDF file a piece at a time.
149#[derive(Debug)]
150pub struct SdfReader {
151    buf: Vec<u8>,
152    /// The rest of a line longer than [`MAX_LINE`] is being passed over.
153    cutting: bool,
154    place: Place,
155    record: Record,
156    fields: Vec<FieldColumn>,
157    by_name: HashMap<String, usize>,
158    names: Vec<Option<String>>,
159    atoms: Vec<Option<u32>>,
160    bonds: Vec<Option<u32>>,
161    /// Per field, a value per row.
162    columns: Vec<Vec<Option<String>>>,
163    rows: usize,
164    held: usize,
165    stats: Stats,
166}
167
168impl Default for SdfReader {
169    fn default() -> Self {
170        Self::new()
171    }
172}
173
174impl SdfReader {
175    pub fn new() -> Self {
176        Self {
177            buf: Vec::new(),
178            cutting: false,
179            place: Place::Header(0),
180            record: Record::default(),
181            fields: Vec::new(),
182            by_name: HashMap::new(),
183            names: Vec::new(),
184            atoms: Vec::new(),
185            bonds: Vec::new(),
186            columns: Vec::new(),
187            rows: 0,
188            held: 0,
189            stats: Stats::default(),
190        }
191    }
192
193    pub fn stats(&self) -> &Stats {
194        &self.stats
195    }
196
197    /// The data fields, in the order they were first seen.
198    pub fn fields(&self) -> &[FieldColumn] {
199        &self.fields
200    }
201
202    /// Read `bytes`, the next piece of the file.
203    pub fn push(&mut self, bytes: &[u8]) {
204        let mut rest = bytes;
205        while let Some(end) = rest.iter().position(|&b| b == b'\n') {
206            self.take(&rest[..end]);
207            if !self.cutting {
208                let line = std::mem::take(&mut self.buf);
209                self.line(&line);
210            }
211            self.cutting = false;
212            self.buf.clear();
213            rest = &rest[end + 1..];
214        }
215        self.take(rest);
216    }
217
218    /// Hold `part` of the line being read, up to [`MAX_LINE`].
219    fn take(&mut self, part: &[u8]) {
220        if self.cutting {
221            return;
222        }
223        let room = MAX_LINE.saturating_sub(self.buf.len());
224        if part.len() > room {
225            self.buf.extend_from_slice(&part[..room]);
226            self.stats.long_lines += 1;
227            let line = std::mem::take(&mut self.buf);
228            self.line(&line);
229            self.cutting = true;
230        } else {
231            self.buf.extend_from_slice(part);
232        }
233    }
234
235    /// A full batch, once one is held: [`BATCH_ROWS`] rows, or [`BATCH_TEXT`] of text.
236    pub fn take_batch(&mut self) -> PolarsResult<Option<DataFrame>> {
237        if self.rows < BATCH_ROWS && self.held < BATCH_TEXT {
238            return Ok(None);
239        }
240        self.batch().map(Some)
241    }
242
243    /// The end of the file: the last line and record, and the rows not yet taken.
244    pub fn finish(&mut self) -> PolarsResult<DataFrame> {
245        if !self.cutting && !self.buf.is_empty() {
246            let line = std::mem::take(&mut self.buf);
247            self.line(&line);
248        }
249        self.buf.clear();
250        self.cutting = false;
251        if self.record.started {
252            self.stats.unterminated = true;
253            self.end_record();
254        }
255        self.batch()
256    }
257
258    fn batch(&mut self) -> PolarsResult<DataFrame> {
259        self.held = 0;
260        let height = std::mem::take(&mut self.rows);
261        let mut columns = vec![
262            StringChunked::from_iter_options(
263                "name".into(),
264                std::mem::take(&mut self.names).into_iter(),
265            )
266            .into_column(),
267            UInt32Chunked::from_iter_options(
268                "atoms".into(),
269                std::mem::take(&mut self.atoms).into_iter(),
270            )
271            .into_column(),
272            UInt32Chunked::from_iter_options(
273                "bonds".into(),
274                std::mem::take(&mut self.bonds).into_iter(),
275            )
276            .into_column(),
277        ];
278        for (field, values) in self.fields.iter().zip(self.columns.iter_mut()) {
279            columns.push(
280                StringChunked::from_iter_options(
281                    field.name.as_str().into(),
282                    std::mem::take(values).into_iter(),
283                )
284                .into_column(),
285            );
286        }
287        DataFrame::new(height, columns)
288    }
289
290    fn line(&mut self, raw: &[u8]) {
291        let raw = raw.strip_suffix(b"\r").unwrap_or(raw);
292        let line = String::from_utf8_lossy(raw);
293        let line = line.as_ref();
294        if line.starts_with("$$$$") {
295            self.end_value();
296            if self.record.started {
297                self.end_record();
298            } else {
299                // Blank lines and a `$$$$` hold no record.
300                self.record = Record::default();
301                self.place = Place::Header(0);
302            }
303            return;
304        }
305        if !line.trim().is_empty() {
306            self.record.started = true;
307        }
308        match self.place {
309            Place::Header(n) => {
310                // The name line may be blank; lines after it say what made the file.
311                if n == 0 {
312                    self.record.name = Some(line.trim().to_string()).filter(|s| !s.is_empty());
313                } else if n == 3 {
314                    self.counts(line);
315                }
316                self.place = if n >= 3 {
317                    Place::Ctab
318                } else {
319                    Place::Header(n + 1)
320                };
321            }
322            Place::Ctab => {
323                if line.starts_with("M  END") {
324                    self.place = Place::Data;
325                } else if let Some(counts) = line.strip_prefix("M  V30 COUNTS") {
326                    let mut numbers = counts.split_whitespace();
327                    self.record.atoms = numbers.next().and_then(|n| n.parse().ok());
328                    self.record.bonds = numbers.next().and_then(|n| n.parse().ok());
329                } else if is_item_header(line) {
330                    // A connection table cut short of `M  END`.
331                    self.place = Place::Data;
332                    self.item(line);
333                }
334            }
335            Place::Data => {
336                if is_item_header(line) {
337                    self.item(line);
338                }
339            }
340            Place::Value => {
341                if line.trim().is_empty() {
342                    self.end_value();
343                    self.place = Place::Data;
344                } else if is_item_header(line) && line.contains('<') {
345                    // A writer that leaves out the blank line.
346                    self.end_value();
347                    self.item(line);
348                } else if self.record.field.is_some() && !self.record.cut {
349                    let value = &mut self.record.value;
350                    let sep = usize::from(!value.is_empty());
351                    let room = MAX_VALUE.saturating_sub(value.len() + sep);
352                    let mut take = line.len().min(room);
353                    while !line.is_char_boundary(take) {
354                        take -= 1;
355                    }
356                    if take > 0 {
357                        if sep == 1 {
358                            value.push('\n');
359                        }
360                        value.push_str(&line[..take]);
361                    }
362                    if take < line.len() {
363                        self.stats.long_values += 1;
364                        self.record.cut = true;
365                    }
366                }
367            }
368        }
369    }
370
371    /// The counts line: atoms in columns 1-3, bonds in 4-6; V3000 gives them later.
372    fn counts(&mut self, line: &str) {
373        if line.contains("V3000") {
374            self.stats.v3000 += 1;
375            return;
376        }
377        let number = |range: std::ops::Range<usize>| {
378            line.get(range).and_then(|s| s.trim().parse::<u32>().ok())
379        };
380        self.record.atoms = number(0..3);
381        self.record.bonds = number(3..6);
382    }
383
384    /// A data item's header line: `> <FIELD>`, `>  <FIELD> (ID)`, `> 25 <FIELD>`.
385    fn item(&mut self, line: &str) {
386        self.place = Place::Value;
387        self.record.value.clear();
388        self.record.field = None;
389        self.record.cut = false;
390        let Some(name) = item_name(line) else {
391            return;
392        };
393        let index = match self.by_name.get(&name) {
394            Some(&i) => i,
395            None if self.fields.len() >= MAX_FIELDS => {
396                self.stats.fields_dropped += 1;
397                return;
398            }
399            None => {
400                let i = self.fields.len();
401                self.by_name.insert(name.clone(), i);
402                // A field may be called `name`, or what an earlier one became; its
403                // column takes a suffix, as columns of one table must differ.
404                let mut column = name.clone();
405                let mut n = 1;
406                while CORE.contains(&column.as_str())
407                    || self.fields.iter().any(|f| f.name == column)
408                {
409                    n += 1;
410                    column = format!("{name}_{n}");
411                }
412                self.fields.push(FieldColumn {
413                    name: column,
414                    values: 0,
415                    integers: true,
416                    numbers: true,
417                });
418                self.columns.push(vec![None; self.rows]);
419                i
420            }
421        };
422        if self.record.values.contains_key(&index) {
423            self.stats.repeated += 1;
424            return;
425        }
426        self.record.field = Some(index);
427    }
428
429    fn end_value(&mut self) {
430        if self.place != Place::Value {
431            return;
432        }
433        let value = std::mem::take(&mut self.record.value);
434        if let Some(field) = self.record.field.take() {
435            self.keep(field, value);
436        }
437    }
438
439    fn keep(&mut self, field: usize, value: String) {
440        let trimmed = value.trim();
441        if trimmed.is_empty() {
442            return;
443        }
444        let column = &mut self.fields[field];
445        column.values += 1;
446        column.integers &= trimmed.parse::<i64>().is_ok();
447        column.numbers &= trimmed.parse::<f64>().is_ok_and(f64::is_finite);
448        let value = if trimmed.len() == value.len() {
449            value
450        } else {
451            trimmed.to_string()
452        };
453        self.record.values.insert(field, value);
454    }
455
456    fn end_record(&mut self) {
457        let record = std::mem::take(&mut self.record);
458        self.place = Place::Header(0);
459        self.names.push(record.name);
460        self.atoms.push(record.atoms);
461        self.bonds.push(record.bonds);
462        let mut values = record.values;
463        for (i, column) in self.columns.iter_mut().enumerate() {
464            let value = values.remove(&i);
465            self.held += size_of::<Option<String>>() + value.as_ref().map_or(0, String::len);
466            column.push(value);
467        }
468        self.rows += 1;
469        self.stats.records += 1;
470    }
471}
472
473/// Whether `line` starts a data item.
474fn is_item_header(line: &str) -> bool {
475    line.starts_with('>')
476}
477
478/// The field a data item header names: what is between `<` and `>`, or else the first
479/// word after the `>` (`> DT12`).
480pub fn item_name(line: &str) -> Option<String> {
481    let rest = line.strip_prefix('>')?;
482    let name = match rest.find('<') {
483        Some(open) => {
484            let after = &rest[open + 1..];
485            let close = after.find('>')?;
486            &after[..close]
487        }
488        None => rest.split_whitespace().next()?,
489    };
490    let name = name.trim();
491    if name.is_empty() {
492        return None;
493    }
494    let mut cut = name.len().min(MAX_NAME);
495    while !name.is_char_boundary(cut) {
496        cut -= 1;
497    }
498    Some(name[..cut].to_string())
499}
500
501/// The fields as numbers where every value was one.
502fn type_fields(lf: LazyFrame, fields: &[FieldColumn]) -> LazyFrame {
503    let casts: Vec<Expr> = fields
504        .iter()
505        .filter(|f| f.dtype() != DataType::String)
506        .map(|f| col(f.name.as_str()).cast(f.dtype()))
507        .collect();
508    if casts.is_empty() {
509        lf
510    } else {
511        lf.with_columns(casts)
512    }
513}
514
515fn notes(stats: &Stats) -> Vec<Note> {
516    let mut notes = Vec::new();
517    let of_records = format!("of {}", count(stats.records, "record", "records"));
518    if stats.long_lines > 0 {
519        notes.push(note(
520            format!(
521                "{} cut at {} MiB",
522                count(stats.long_lines, "line", "lines"),
523                MAX_LINE >> 20
524            ),
525            "in the whole file".to_string(),
526        ));
527    }
528    if stats.long_values > 0 {
529        notes.push(note(
530            format!(
531                "{} cut at {} MiB",
532                count(stats.long_values, "value", "values"),
533                MAX_VALUE >> 20
534            ),
535            of_records.clone(),
536        ));
537    }
538    if stats.repeated > 0 {
539        notes.push(note(
540            format!(
541                "repeated field: {} {} first kept",
542                count(stats.repeated, "data item", "data items"),
543                crate::glyphs::get().middot
544            ),
545            of_records.clone(),
546        ));
547    }
548    if stats.fields_dropped > 0 {
549        notes.push(note(
550            format!(
551                "{} left out: past the first {MAX_FIELDS} fields",
552                count(stats.fields_dropped, "data item", "data items")
553            ),
554            of_records.clone(),
555        ));
556    }
557    if stats.unterminated {
558        notes.push(note(
559            format!(
560                "last record has no $$$$ {} read as is",
561                crate::glyphs::get().middot
562            ),
563            of_records,
564        ));
565    }
566    notes
567}
568
569/// The SDF tab of the Info panel: the counts and each field's type and coverage.
570pub fn detail(reader: &SdfReader) -> Detail {
571    let stats = reader.stats();
572    let sep = format!(" {} ", crate::glyphs::get().middot);
573    let mut head = format!(
574        "SDF{sep}{}{sep}{}",
575        count(stats.records, "record", "records"),
576        count(reader.fields().len() as u64, "field", "fields")
577    );
578    if stats.v3000 > 0 {
579        head.push_str(&sep);
580        head.push_str(&format!(
581            "{} V3000",
582            count(stats.v3000, "record", "records")
583        ));
584    }
585    let list = capped_list(
586        reader.fields().iter().map(|f| {
587            (
588                f.name.clone(),
589                MetaValue::Text(format!(
590                    "{}{sep}in {} of {}",
591                    f.dtype(),
592                    crate::numfmt::group_chrome(f.values as usize),
593                    count(stats.records, "record", "records"),
594                )),
595            )
596        }),
597        reader.fields().len(),
598    );
599    Detail {
600        tab: crate::text_formats::tab(crate::FileFormat::Sdf),
601        lines: vec![head],
602        list_title: "Fields",
603        list,
604        first: false,
605        ..Default::default()
606    }
607}
608
609/// Read an SDF file, given a piece at a time by `pieces`, into segments written through
610/// `writer`.
611pub(crate) fn convert(
612    display: &Path,
613    options: &OpenOptions,
614    writer: &Writer,
615    pieces: &mut Pieces<'_>,
616) -> Result<(Converted, Detail)> {
617    let mut reader = SdfReader::new();
618    let mut segments = Segments::new(options, writer);
619    pieces(&mut |piece| {
620        reader.push(piece);
621        if let Some(df) = reader.take_batch()? {
622            segments.write(&df)?;
623        }
624        Ok(())
625    })?;
626    let last = reader.finish()?;
627    if reader.stats().records == 0 {
628        return Err(crate::error_display::FileError::new(display, "no SDF records").into());
629    }
630    segments.write(&last)?;
631    let (lf, files) = segments.finish()?;
632    let lf = type_fields(lf, reader.fields());
633    Ok((
634        Converted {
635            lf,
636            files,
637            notes: notes(reader.stats()),
638            other_tables: Vec::new(),
639        },
640        detail(&reader),
641    ))
642}
643
644#[cfg(test)]
645mod tests {
646    use super::*;
647
648    /// A file that is not one names itself, in the one shape.
649    #[test]
650    fn errors_name_the_file() {
651        crate::readers::bad_input::each_names_its_file(
652            crate::FileFormat::Sdf,
653            &[("empty.sdf", b"", "No SDF records")],
654        );
655    }
656
657    const SAMPLE: &str = "aspirin
658  RDKit          2D
659
660  3  2  0  0  0  0  0  0  0  0999 V2000
661    0.0000    0.0000    0.0000 C   0  0  0  0  0  0  0  0  0  0  0  0
662    1.2990    0.7500    0.0000 C   0  0
663    2.5981   -0.0000    0.0000 O   0  0
664  1  2  1  0
665  2  3  2  0
666M  END
667> <ID>
668101
669
670> <LogP>  (MD-1)
6711.19
672
673> <SMILES>
674CC(=O)O
675
676$$$$
677
678  RDKit          2D
679
680  1  0  0  0  0  0  0  0  0  0999 V2000
681    0.0000    0.0000    0.0000 N   0  0
682M  END
683>  <ID>
684102
685
686> 25 <Notes>
687first line
688second line
689
690> <LogP>
691n/a
692
693$$$$
694";
695
696    fn read(text: &[u8], piece: usize) -> (DataFrame, SdfReader) {
697        let mut reader = SdfReader::new();
698        let mut frames = Vec::new();
699        for chunk in text.chunks(piece) {
700            reader.push(chunk);
701            if let Some(df) = reader.take_batch().unwrap() {
702                frames.push(df);
703            }
704        }
705        frames.push(reader.finish().unwrap());
706        let width = frames.last().unwrap().width();
707        let mut df = frames.pop().unwrap();
708        for f in frames.into_iter().rev() {
709            assert!(f.width() <= width);
710            df = f.vstack(&df).unwrap_or(df);
711        }
712        (df, reader)
713    }
714
715    fn strings(df: &DataFrame, name: &str) -> Vec<Option<String>> {
716        df.column(name)
717            .unwrap()
718            .str()
719            .unwrap()
720            .iter()
721            .map(|s| s.map(String::from))
722            .collect()
723    }
724
725    /// Empty cells count toward a batch: a record of 2,000 fields makes every later
726    /// row 2,000 cells wide, so the batch is handed over long before `BATCH_ROWS`.
727    #[test]
728    fn a_wide_file_hands_over_small_batches() {
729        let mut reader = SdfReader::new();
730        let mut wide = String::from("wide\n\n\n  0  0  0  0  0  0  0  0  0  0999 V2000\nM  END\n");
731        for i in 0..2_000 {
732            wide.push_str(&format!("> <F{i}>\nx\n\n"));
733        }
734        wide.push_str("$$$$\n");
735        reader.push(wide.as_bytes());
736        let narrow = "n\n\n\n  0  0  0  0  0  0  0  0  0  0999 V2000\nM  END\n$$$$\n";
737        let mut rows_at_batch = None;
738        for row in 1..BATCH_ROWS {
739            reader.push(narrow.as_bytes());
740            if let Some(df) = reader.take_batch().unwrap() {
741                rows_at_batch = Some((row, df.height()));
742                break;
743            }
744        }
745        let (row, height) = rows_at_batch.expect("a batch is handed over");
746        assert!(height < BATCH_ROWS / 8, "{height} rows of 2,003 columns");
747        assert_eq!(height, row + 1);
748    }
749
750    #[test]
751    fn records_read_as_rows_with_their_fields() {
752        for piece in [1, 5, 64, 4096] {
753            let (df, reader) = read(SAMPLE.as_bytes(), piece);
754            assert_eq!(df.height(), 2, "piece {piece}");
755            assert_eq!(
756                df.get_column_names(),
757                ["name", "atoms", "bonds", "ID", "LogP", "SMILES", "Notes"]
758            );
759            assert_eq!(strings(&df, "name"), [Some("aspirin".into()), None]);
760            let atoms: Vec<_> = df.column("atoms").unwrap().u32().unwrap().iter().collect();
761            assert_eq!(atoms, [Some(3), Some(1)]);
762            let bonds: Vec<_> = df.column("bonds").unwrap().u32().unwrap().iter().collect();
763            assert_eq!(bonds, [Some(2), Some(0)]);
764            assert_eq!(
765                strings(&df, "Notes"),
766                [None, Some("first line\nsecond line".into())]
767            );
768            assert_eq!(strings(&df, "SMILES"), [Some("CC(=O)O".into()), None]);
769            let fields = reader.fields();
770            assert_eq!(fields[0].dtype(), DataType::Int64);
771            assert_eq!(fields[1].dtype(), DataType::String, "n/a is not a number");
772            assert_eq!(reader.stats().records, 2);
773            assert!(!reader.stats().unterminated);
774        }
775    }
776
777    #[test]
778    fn field_types_are_cast_on_the_frame() {
779        let (df, reader) = read(b"m\n\n\n  0  0  0  0  0  0  0  0  0  0999 V2000\nM  END\n> <MW>\n180.16\n\n> <N>\n3\n\n$$$$\n", 7);
780        let lf = type_fields(df.lazy(), reader.fields());
781        let df = lf.collect().unwrap();
782        assert_eq!(df.column("MW").unwrap().dtype(), &DataType::Float64);
783        assert_eq!(df.column("N").unwrap().i64().unwrap().get(0), Some(3));
784    }
785
786    #[test]
787    fn v3000_counts_and_an_unterminated_record() {
788        let text = "x\n\n\n  0  0  0     0  0            999 V3000\nM  V30 BEGIN CTAB\nM  V30 COUNTS 12 11 0 0 0\nM  V30 END CTAB\nM  END\n> <A>\n1\n";
789        let (df, reader) = read(text.as_bytes(), 3);
790        assert_eq!(df.column("atoms").unwrap().u32().unwrap().get(0), Some(12));
791        assert_eq!(df.column("bonds").unwrap().u32().unwrap().get(0), Some(11));
792        assert!(reader.stats().unterminated);
793        assert_eq!(reader.stats().v3000, 1);
794    }
795
796    #[test]
797    fn item_names() {
798        assert_eq!(item_name("> <KEY>").as_deref(), Some("KEY"));
799        assert_eq!(item_name(">  <KEY> (ID)").as_deref(), Some("KEY"));
800        assert_eq!(item_name("> 25 <a b>").as_deref(), Some("a b"));
801        assert_eq!(item_name("> DT12 (MD-08974)").as_deref(), Some("DT12"));
802        assert_eq!(item_name("> <>"), None);
803        assert_eq!(item_name(">"), None);
804    }
805
806    #[test]
807    fn long_lines_and_values_are_cut() {
808        let mut text = b"m\n\n\n  0  0\nM  END\n> <Big>\n".to_vec();
809        text.extend(std::iter::repeat_n(b'a', MAX_LINE + 5));
810        text.extend_from_slice(b"\nmore\n\n$$$$\n");
811        let (df, reader) = read(&text, 1 << 16);
812        let big = df
813            .column("Big")
814            .unwrap()
815            .str()
816            .unwrap()
817            .get(0)
818            .unwrap()
819            .len();
820        assert!(big <= MAX_VALUE, "{big}");
821        assert_eq!(reader.stats().long_lines, 1);
822        assert_eq!(reader.stats().long_values, 1);
823    }
824
825    #[test]
826    fn sniffing() {
827        assert!(looks_like(SAMPLE.as_bytes()));
828        assert!(looks_like(b"\n\n\nM  END\n> <A>\n1\n\n$$$$\n"));
829        assert!(!looks_like(b"a,b\n1,2\n"));
830        assert!(
831            !looks_like(b"model,code\nx,1\ny,2\nz,V2000\n"),
832            "a word, not a counts line"
833        );
834    }
835
836    #[test]
837    fn a_field_named_like_a_core_column_gets_its_own() {
838        let mut reader = SdfReader::new();
839        reader.push(
840            b"aspirin\n\n\n  0  0  0  0  0  0  0  0  0  0999 V2000\nM  END\n\
841              > <name>\nASA\n\n> <name_2>\nx\n\n$$$$\n",
842        );
843        let df = reader.finish().unwrap();
844        let names: Vec<&str> = df.get_column_names().iter().map(|c| c.as_str()).collect();
845        assert_eq!(names, ["name", "atoms", "bonds", "name_2", "name_2_2"]);
846        assert_eq!(strings(&df, "name"), [Some("aspirin".into())]);
847        assert_eq!(strings(&df, "name_2"), [Some("ASA".into())]);
848    }
849}