Skip to main content

data_beans/utilities/
io_helpers.rs

1use crate::sparse_io::{COLUMN_SEP, ROW_SEP};
2use legume_numeric::matrix::common_io::*;
3use std::ops::Range;
4
5/// Parse a names file into one joined string per line.
6///
7/// Reads `name_file` as lines of whitespace-separated words, then for each line
8/// joins the words at the column indices in `name_columns` (clamped to the
9/// line's word count) with `name_sep`. Single source of truth shared by the
10/// row/column name readers below and by the zarr/hdf5 backends'
11/// `register_names_file` (the backend-specific dataset write stays per backend).
12pub fn parse_name_file(
13    name_file: &str,
14    name_columns: Range<usize>,
15    name_sep: &str,
16) -> anyhow::Result<Vec<String>> {
17    let name_data = read_lines_of_words(name_file, -1)?;
18    let names: Vec<String> = name_data
19        .lines
20        .iter()
21        .map(|line| {
22            // Clamp the upper bound to the line's word count so a large
23            // `name_columns.end` (e.g. a user-supplied word count) iterates at
24            // most `line.len()` times rather than materializing a huge range.
25            (name_columns.start..name_columns.end.min(line.len()))
26                .filter_map(|i| line.get(i))
27                .map(|s| s.to_string())
28                .collect::<Vec<_>>()
29                .join(name_sep)
30        })
31        .collect();
32    Ok(names)
33}
34
35/// Read row names from a file, joining multi-column names with ROW_SEP
36pub fn read_row_names(
37    row_file: Box<str>,
38    max_row_name_idx: usize,
39) -> anyhow::Result<Vec<Box<str>>> {
40    Ok(parse_name_file(&row_file, 0..max_row_name_idx, ROW_SEP)?
41        .into_iter()
42        .map(String::into_boxed_str)
43        .collect())
44}
45
46/// Read column names from a file, joining multi-column names with COLUMN_SEP
47pub fn read_col_names(
48    col_file: Box<str>,
49    max_column_name_idx: usize,
50) -> anyhow::Result<Vec<Box<str>>> {
51    Ok(
52        parse_name_file(&col_file, 0..max_column_name_idx, COLUMN_SEP)?
53            .into_iter()
54            .map(String::into_boxed_str)
55            .collect(),
56    )
57}
58
59/// Parse an index specification into an explicit, ordered list of indices.
60///
61/// Accepts comma-separated single indices and inclusive ranges, e.g.
62/// `0,2,5`, `1-10`, or `1-20,50-55`. Whitespace around tokens is ignored.
63/// Ranges are inclusive on both ends (`1-10` -> 1,2,...,10).
64pub fn parse_index_spec(spec: &str) -> anyhow::Result<Vec<usize>> {
65    let mut out = Vec::new();
66    for token in spec.split(',') {
67        let token = token.trim();
68        if token.is_empty() {
69            continue;
70        }
71        match token.split_once('-') {
72            Some((start, end)) => {
73                let start: usize = start
74                    .trim()
75                    .parse()
76                    .map_err(|_| anyhow::anyhow!("invalid range start in `{}`", token))?;
77                let end: usize = end
78                    .trim()
79                    .parse()
80                    .map_err(|_| anyhow::anyhow!("invalid range end in `{}`", token))?;
81                if start > end {
82                    anyhow::bail!("range start {} exceeds end {} in `{}`", start, end, token);
83                }
84                out.extend(start..=end);
85            }
86            None => {
87                let idx: usize = token
88                    .parse()
89                    .map_err(|_| anyhow::anyhow!("invalid index `{}`", token))?;
90                out.push(idx);
91            }
92        }
93    }
94    if out.is_empty() {
95        anyhow::bail!("no valid indices parsed from `{}`", spec);
96    }
97    Ok(out)
98}
99
100// Re-export constants for convenience
101pub use crate::sparse_io::{MAX_COLUMN_NAME_IDX, MAX_ROW_NAME_IDX};
102
103/// Target ~1 MB per chunk: big enough for good compression, small enough for
104/// fast random-access reads of individual columns/rows.
105const TARGET_CHUNK_BYTES: usize = 1024 * 1024;
106/// Never fewer than this many elements in a chunk.
107const MIN_CHUNK_ELEMS: usize = 8192;
108
109/// Compute the chunk size (in elements) for a 1-D array of `nelem` elements,
110/// each `elem_bytes` wide. Returns at least 1 even for empty arrays.
111pub fn chunk_elems(nelem: usize, elem_bytes: usize) -> usize {
112    (TARGET_CHUNK_BYTES / elem_bytes.max(1))
113        .max(MIN_CHUNK_ELEMS)
114        .min(nelem.max(1))
115}
116
117#[cfg(test)]
118mod tests {
119    use super::parse_index_spec;
120
121    #[test]
122    fn singles_and_ranges() {
123        assert_eq!(parse_index_spec("0,2,5").unwrap(), vec![0, 2, 5]);
124        assert_eq!(parse_index_spec("1-5").unwrap(), vec![1, 2, 3, 4, 5]);
125        assert_eq!(
126            parse_index_spec("1-3,10,20-22").unwrap(),
127            vec![1, 2, 3, 10, 20, 21, 22]
128        );
129        // single-element range and surrounding whitespace
130        assert_eq!(parse_index_spec(" 7 - 7 , 9 ").unwrap(), vec![7, 9]);
131    }
132
133    #[test]
134    fn rejects_bad_input() {
135        assert!(parse_index_spec("5-1").is_err());
136        assert!(parse_index_spec("abc").is_err());
137        assert!(parse_index_spec("").is_err());
138    }
139}