Skip to main content

legume_numeric/matrix/
common_io.rs

1/// Default `env_logger` filter when verbose mode is on. Promotes everything to
2/// info. (The k-NN backend, instant-distance, emits no `log` output, so no
3/// per-crate pin is needed here.) Override via `RUST_LOG=...`.
4pub const VERBOSE_LOG_FILTER: &str = "info";
5
6/// Default `env_logger` filter when verbose mode is off.
7pub const QUIET_LOG_FILTER: &str = "warn";
8
9use flate2::read::MultiGzDecoder;
10use rayon::prelude::*;
11use std::ffi::OsStr;
12use std::fs::File;
13use std::io::{BufRead, BufReader, BufWriter, Write};
14use std::path::Path;
15use tempfile::tempdir;
16
17#[cfg(test)]
18mod tests;
19
20/// Define a Delimiter enum to handle both &str and `Vec<char>`
21pub enum Delimiter {
22    Str(String),
23    Chars(Vec<char>),
24}
25
26impl From<&str> for Delimiter {
27    fn from(s: &str) -> Self {
28        Delimiter::Str(s.to_string())
29    }
30}
31
32impl From<Vec<char>> for Delimiter {
33    fn from(chars: Vec<char>) -> Self {
34        Delimiter::Chars(chars)
35    }
36}
37
38impl From<&[char]> for Delimiter {
39    fn from(chars: &[char]) -> Self {
40        Delimiter::Chars(chars.to_vec())
41    }
42}
43
44impl<const N: usize> From<&[char; N]> for Delimiter {
45    fn from(chars: &[char; N]) -> Self {
46        Delimiter::Chars(chars.to_vec())
47    }
48}
49
50///
51/// Read every line of the input_file into memory
52///
53/// * `input_file` - file name--either gzipped or not
54///
55pub fn read_lines(input_file_path: &str) -> anyhow::Result<Vec<Box<str>>> {
56    let buf: Box<dyn BufRead> = open_buf_reader(input_file_path)?;
57    let mut lines = vec![];
58    for x in buf.lines() {
59        lines.push(x?.into_boxed_str());
60    }
61    Ok(lines)
62}
63
64///
65/// Write every line into the output_file
66///
67/// * `lines` - vector of lines
68/// * `output_file` - file name--either gzipped or not
69///
70pub fn write_lines(lines: &Vec<Box<str>>, output_file_path: &str) -> anyhow::Result<()> {
71    write_types(lines, output_file_path)
72}
73
74///
75/// Write every line into the output_file
76///
77/// * `lines` - vector of lines
78/// * `output_file` - file name--either gzipped or not
79///
80pub fn write_types<T>(lines: &Vec<T>, output_file_path: &str) -> anyhow::Result<()>
81where
82    T: std::fmt::Display,
83{
84    let mut buf = open_buf_writer(output_file_path)?;
85    for line in lines {
86        if let Err(e) = writeln!(buf, "{}", line) {
87            if e.kind() == std::io::ErrorKind::BrokenPipe {
88                return Ok(());
89            } else {
90                return Err(anyhow::anyhow!("unexpected error: {}", e));
91            }
92        }
93    }
94    buf.flush()?;
95    Ok(())
96}
97
98pub struct ReadLinesOut<T: Send> {
99    pub lines: Vec<Vec<T>>,
100    pub header: Vec<Box<str>>,
101}
102
103///
104/// Generic function to read lines and parse them into a vector of words or types.
105///
106/// * `input_file` - file name--either gzipped or not
107/// * `hdr_line` - location of a header line (-1 = no header line)
108/// * `parse_fn` - function to parse each line into the desired type
109///
110pub fn read_lines_of_words_generic<T>(
111    input_file: &str,
112    hdr_line: i64,
113    parse_header_fn: impl Fn(&str) -> Vec<Box<str>> + Sync,
114    parse_fn: impl Fn(&str) -> Vec<T> + Sync,
115) -> anyhow::Result<ReadLinesOut<T>>
116where
117    T: Send,
118{
119    let buf_reader: Box<dyn BufRead> = open_buf_reader(input_file)?;
120
121    fn is_not_comment_line(line: &str) -> bool {
122        if line.starts_with('#') || line.starts_with('%') {
123            return false;
124        }
125        true
126    }
127
128    let lines_raw: Vec<Box<str>> = buf_reader
129        .lines()
130        .map_while(Result::ok)
131        .map(|x| x.into_boxed_str())
132        .filter(|x| is_not_comment_line(x.as_ref()))
133        .collect();
134
135    let mut header = vec![];
136
137    // Parsing takes more time, so split them into parallel jobs
138    let mut lines: Vec<(usize, Vec<T>)> = if hdr_line < 0 {
139        lines_raw
140            .iter()
141            .enumerate()
142            .par_bridge()
143            .map(|(i, s)| (i, parse_fn(s)))
144            .collect()
145    } else {
146        let n_skip = hdr_line as usize;
147        if lines_raw.len() < (n_skip + 1) {
148            return Err(anyhow::anyhow!("not enough data"));
149        }
150
151        header.extend(parse_header_fn(&lines_raw[n_skip]));
152
153        lines_raw[(n_skip + 1)..]
154            .iter()
155            .enumerate()
156            .par_bridge()
157            .map(|(i, s)| (i, parse_fn(s)))
158            .collect()
159    };
160
161    if lines.len() > 100_000 {
162        lines.par_sort_by_key(|&(i, _)| i);
163    } else {
164        lines.sort_by_key(|&(i, _)| i);
165    }
166
167    let lines = lines.into_iter().map(|(_, x)| x).collect();
168    Ok(ReadLinesOut { lines, header })
169}
170
171///
172/// Specialized function to read lines and parse them into a vector of types.
173///
174/// * `input_file` - file name--either gzipped or not
175/// * `delim` - delimiter
176/// * `hdr_line` - location of a header line (-1 = no header line)
177///
178pub fn read_lines_of_types<T>(
179    input_file: &str,
180    delim: impl Into<Delimiter>,
181    hdr_line: i64,
182) -> anyhow::Result<ReadLinesOut<T>>
183where
184    T: Send + std::str::FromStr + std::fmt::Display,
185    <T as std::str::FromStr>::Err: std::fmt::Debug,
186{
187    let delim = delim.into(); // Convert the input delimiter into the Delimiter enum
188
189    let parse_fn = move |line: &str| -> Vec<T> {
190        match &delim {
191            Delimiter::Str(s) => line
192                .split(s.as_str())
193                .map(|x| x.parse::<T>().expect("failed to parse"))
194                .collect(),
195            Delimiter::Chars(chars) => line
196                .split(chars.as_slice())
197                .map(|x| x.parse::<T>().expect("failed to parse"))
198                .collect(),
199        }
200    };
201
202    let parse_header_fn = |line: &str| -> Vec<Box<str>> {
203        line.split_whitespace()
204            .map(|x| x.to_owned().into_boxed_str())
205            .collect()
206    };
207
208    read_lines_of_words_generic(input_file, hdr_line, parse_header_fn, parse_fn)
209}
210
211///
212/// Specialized function to read lines and parse them into a vector of words.
213///
214/// * `input_file` - file name--either gzipped or not
215/// * `hdr_line` - location of a header line (-1 = no header line)
216///
217pub fn read_lines_of_words(
218    input_file: &str,
219    hdr_line: i64,
220) -> anyhow::Result<ReadLinesOut<Box<str>>> {
221    let parse_fn = |line: &str| -> Vec<Box<str>> {
222        line.split_whitespace()
223            .map(|x| x.to_owned().into_boxed_str())
224            .collect()
225    };
226
227    read_lines_of_words_generic(input_file, hdr_line, parse_fn, parse_fn)
228}
229
230///
231/// Specialized function to read lines and parse them into a vector of words.
232///
233/// * `input_file` - file name--either gzipped or not
234/// * `delim` - delimiter
235/// * `hdr_line` - location of a header line (-1 = no header line)
236///
237/// Trim a field and strip one pair of symmetric quotes.
238///
239/// Only a field quoted at BOTH ends is a quoted field. Trimming either end
240/// on its own corrupts content that merely happens to start or finish with a
241/// quote: a GTF attribute column reads `gene_id "X"; gene_name "Y"`, which
242/// ends in a quote it needs, and losing it leaves the attribute unparseable.
243/// This is the same rule the delimited tokenizer applies to every field, so
244/// any caller inspecting raw lines (header sniffing, previews) can match it.
245pub fn unquote_field(x: &str) -> &str {
246    let t = x.trim();
247    for q in ['"', '\''] {
248        if t.len() >= 2 && t.starts_with(q) && t.ends_with(q) {
249            return &t[q.len_utf8()..t.len() - q.len_utf8()];
250        }
251    }
252    t
253}
254
255pub fn read_lines_of_words_delim(
256    input_file: &str,
257    delim: impl Into<Delimiter>,
258    hdr_line: i64,
259) -> anyhow::Result<ReadLinesOut<Box<str>>> {
260    let delim = delim.into(); // Convert the input delimiter into the Delimiter enum
261
262    // Outer quotes come off here, once, rather than at each consumer.
263    //
264    // A csv writer that quotes every field yields `"x"` where the caller asked
265    // for `x`, so a name match fails and a reader that falls back to reading by
266    // POSITION then silently takes whichever columns happen to sit there. The
267    // name-list reader in this same file already unquoted; the general reader
268    // did not, which is the inconsistency this removes.
269    //
270    // This splitter cannot honour a quoted field containing the delimiter
271    // anyway, so trimming the outer quotes loses nothing it had.
272    // Every field is also trimmed. That is wider than unquoting and worth
273    // stating: it is what removes the trailing \r on CRLF input, and it applies
274    // to every delimited file the workspace reads through here.
275    //
276    // Only a field quoted at BOTH ends is a quoted field. Trimming either end
277    // on its own corrupts content that merely happens to start or finish with a
278    // quote: a GTF attribute column reads `gene_id "X"; gene_name "Y"`, which
279    // ends in a quote it needs, and losing it leaves the attribute unparseable.
280    let parse_fn = |line: &str| -> Vec<Box<str>> {
281        match &delim {
282            Delimiter::Str(s) => line
283                .split(s.as_str())
284                .map(|x| unquote_field(x).to_owned().into_boxed_str())
285                .collect(),
286            Delimiter::Chars(chars) => line
287                .split(chars.as_slice())
288                .map(|x| unquote_field(x).to_owned().into_boxed_str())
289                .collect(),
290        }
291    };
292
293    read_lines_of_words_generic(input_file, hdr_line, parse_fn, parse_fn)
294}
295
296////////////////////////////////////////////////////////////
297// Name lists (a column of feature / gene names)           //
298////////////////////////////////////////////////////////////
299
300/// Header labels recognized as the name-bearing column of a name list, matched
301/// case-insensitively.
302const NAME_LIST_HEADERS: [&str; 12] = [
303    "gene",
304    "genes",
305    "gene_name",
306    "gene_names",
307    "gene_id",
308    "gene_symbol",
309    "feature",
310    "features",
311    "feature_name",
312    "symbol",
313    "name",
314    "id",
315];
316
317/// Position of the name-bearing column in `header`, if one is labelled.
318fn name_list_column(header: &[Box<str>]) -> Option<usize> {
319    header.iter().position(|h| {
320        let h = h.trim().trim_matches('"').to_ascii_lowercase();
321        NAME_LIST_HEADERS.contains(&h.as_str())
322    })
323}
324
325/// Read a flat list of names (genes / features) from a file, keeping one column
326/// and dropping every other.
327///
328/// The format is inferred from the extension: `.parquet`, else delimited text —
329/// tab, comma, or whitespace, optionally gzipped (`.txt`, `.tsv`, `.csv`,
330/// `.tsv.gz`, …). A header whose label is gene-like (`gene`, `feature`,
331/// `symbol`, …) selects the column to read; without one the first column is
332/// used. Every other column is ignored, so a two-column `gene<TAB>celltype`
333/// marker table doubles as a plain gene list. Names are de-duplicated, keeping
334/// first-seen order.
335///
336/// Names are returned verbatim — matching them against a data vocabulary
337/// (symbol / Ensembl / case) is the caller's job.
338pub fn read_name_list(file_path: &str) -> anyhow::Result<Vec<Box<str>>> {
339    let is_parquet = Path::new(file_path)
340        .extension()
341        .and_then(OsStr::to_str)
342        .is_some_and(|e| e.eq_ignore_ascii_case("parquet"));
343
344    let names: Vec<Box<str>> = if is_parquet {
345        let header = crate::matrix::parquet::peek_parquet_field_names(file_path)?;
346        let col = name_list_column(&header).unwrap_or(0);
347        crate::matrix::parquet::read_parquet_string_column(file_path, col)?
348    } else {
349        let raw = read_lines(file_path)?;
350        let data_lines: Vec<&str> = raw
351            .iter()
352            .map(|line| line.trim())
353            .filter(|line| !line.is_empty() && !line.starts_with('#') && !line.starts_with('%'))
354            .collect();
355
356        // Sniff ONE delimiter from the first row — tab, else comma, else whitespace.
357        // Splitting on all three at once would tear a value like `T cell` into two
358        // fields and shift every column after it.
359        let delim: &[char] = match data_lines.first() {
360            Some(first) if first.contains('\t') => &['\t'],
361            Some(first) if first.contains(',') => &[','],
362            _ => &[' ', '\t'],
363        };
364
365        // Blank fields (repeated or leading delimiters) are dropped before the column
366        // is indexed, so ragged indentation does not shift it either.
367        let rows: Vec<Vec<&str>> = data_lines
368            .iter()
369            .map(|line| {
370                line.split(delim)
371                    .map(|w| w.trim().trim_matches('"'))
372                    .filter(|w| !w.is_empty())
373                    .collect::<Vec<&str>>()
374            })
375            .filter(|row| !row.is_empty())
376            .collect();
377
378        // A gene-like label in the first row means it is a header: read the column it
379        // names and skip it. Otherwise the file is headerless and data starts at row 0
380        // — treating that first row as a header would silently drop a gene.
381        let header: Vec<Box<str>> = rows
382            .first()
383            .map(|row| row.iter().map(|w| (*w).into()).collect())
384            .unwrap_or_default();
385        let (col, skip) = match name_list_column(&header) {
386            Some(col) => (col, 1),
387            None => (0, 0),
388        };
389
390        rows.iter()
391            .skip(skip)
392            .filter_map(|row| row.get(col).map(|w| (*w).into()))
393            .collect()
394    };
395
396    let mut seen: std::collections::HashSet<Box<str>> = std::collections::HashSet::new();
397    let names: Vec<Box<str>> = names
398        .into_iter()
399        .filter(|n| !n.is_empty())
400        .filter(|n| seen.insert(n.clone()))
401        .collect();
402
403    if names.is_empty() {
404        return Err(anyhow::anyhow!("no names found in {file_path}"));
405    }
406    Ok(names)
407}
408
409///
410/// Open a file for reading, and return a buffered reader
411/// * `input_file` - file name--either gzipped or not
412pub fn open_buf_reader(input_file: &str) -> anyhow::Result<Box<dyn BufRead>> {
413    // take a look at the extension
414    // return buffered reader accordingly
415    let ext = Path::new(input_file).extension().and_then(|x| x.to_str());
416    match ext {
417        Some("gz") | Some("bgz") | Some("bgzf") => {
418            let input_file = File::open(input_file)?;
419            // `MultiGzDecoder`, not `GzDecoder`: BGZF and anything else
420            // written by `bgzip` is a concatenation of gzip members, and a
421            // decoder that stops after the first one drops the rest of the
422            // file silently, with no error and no short read.
423            let decoder = MultiGzDecoder::new(input_file);
424            // A wide buffer over the inflater: a body read one line at a time
425            // would otherwise refill through it every few kilobytes.
426            Ok(Box::new(BufReader::with_capacity(1 << 20, decoder)))
427        }
428        _ => {
429            // dbg!(input_file);
430            let input_file = File::open(input_file)?;
431            Ok(Box::new(BufReader::new(input_file)))
432        }
433    }
434}
435
436////////////////////////////////////
437// First-line inspection of a table //
438////////////////////////////////////
439
440/// The first line of a delimited file, split on `delimiters`, unquoted and
441/// trimmed, empty fields dropped. Reads through gzip.
442///
443/// For error messages and for header detection. It makes no claim about
444/// whether that line is a header; it only shows what the file actually holds.
445pub fn first_line_fields(path: &str, delimiters: &[char]) -> anyhow::Result<Vec<Box<str>>> {
446    let mut first = String::new();
447    open_buf_reader(path)?.read_line(&mut first)?;
448    Ok(first
449        .trim_end_matches(['\n', '\r'])
450        .split(delimiters)
451        .map(|f| unquote_field(f).to_string().into_boxed_str())
452        .filter(|f| !f.is_empty())
453        .collect())
454}
455
456/// Is the first line a header? Decided by type, not by guessing intent: a line
457/// whose fields after column 0 fail to parse as numbers cannot be a data row.
458///
459/// Column 0 is skipped because a row-name column is non-numeric either way.
460/// Fields are unquoted with the tokenizer's own rule first, so a fully quoted
461/// numeric field (`"100.5"`) reads as numeric and a quoted headerless file does
462/// not lose its first data row to a phantom header. Reads through gzip.
463///
464/// `Some(0)` when the first line is a header; `None` when every field after
465/// column 0 is numeric, or the file cannot be read.
466pub fn detect_header_row_numeric(file_path: &str, delimiters: &[char]) -> Option<usize> {
467    let mut first = String::new();
468    open_buf_reader(file_path)
469        .ok()?
470        .read_line(&mut first)
471        .ok()?;
472    let fields: Vec<&str> = first
473        .trim_end_matches(['\n', '\r'])
474        .split(delimiters)
475        .map(unquote_field)
476        .collect();
477    let any_non_numeric_after_col0 = fields
478        .iter()
479        .skip(1)
480        .any(|t| !t.is_empty() && !is_numeric_or_missing(t));
481    // Both outcomes are logged WITH the fields, so a header swallowed as data
482    // (all-numeric sample IDs) or a data row taken as a header shows up in the
483    // log next to the evidence, not only as a wrong row count later.
484    let preview = fields
485        .iter()
486        .take(6)
487        .copied()
488        .collect::<Vec<_>>()
489        .join(", ");
490    if any_non_numeric_after_col0 {
491        log::info!("{file_path}: first line treated as a header (non-numeric): [{preview}]");
492        Some(0)
493    } else {
494        log::info!(
495            "{file_path}: first line treated as data (all numeric after column 0): [{preview}]"
496        );
497        None
498    }
499}
500
501/// A field a count row may legitimately hold: a number, or one of the missing
502/// value spellings R and friends write (`NA`, `N/A`; `NaN` already parses).
503fn is_numeric_or_missing(t: &str) -> bool {
504    t.parse::<f64>().is_ok() || matches!(t, "NA" | "N/A" | "na" | "n/a")
505}
506
507///
508/// Open a file for writing, and return a buffered writer
509/// * `output_file` - file name--either gzipped or not
510pub fn open_buf_writer(output_file: &str) -> anyhow::Result<Box<dyn std::io::Write>> {
511    // we can simply override with stdout
512    if output_file.eq_ignore_ascii_case("stdout") {
513        return Ok(Box::new(std::io::BufWriter::new(std::io::stdout())));
514    }
515
516    if output_file.eq_ignore_ascii_case("stderr") {
517        return Ok(Box::new(std::io::BufWriter::new(std::io::stderr())));
518    }
519
520    // take a look at the extension
521    let output_file = Path::new(output_file);
522    let ext = output_file.extension().and_then(|x| x.to_str());
523
524    match ext {
525        Some("gz") => {
526            let output_file = File::create(output_file)?;
527            let encoder =
528                flate2::write::GzEncoder::new(output_file, flate2::Compression::default());
529            Ok(Box::new(BufWriter::new(encoder)))
530        }
531        _ => {
532            let output_file = File::create(output_file)?;
533            Ok(Box::new(BufWriter::new(output_file)))
534        }
535    }
536}
537
538///
539/// Create a directory if needed
540/// * `file` - file name
541///
542pub fn mkdir(file: &str) -> anyhow::Result<()> {
543    let path = Path::new(file);
544    std::fs::create_dir_all(path)?;
545    Ok(())
546}
547
548/// Ensure the parent directory of an output path/prefix exists.
549///
550/// CLI subcommands typically take an `--out` value used as a *file
551/// prefix* (e.g. `results/run1` → writes `results/run1.foo.parquet`).
552/// Creates the parent directory tree (`results/`) if missing, but
553/// never creates a directory named after the prefix itself
554/// (`results/run1/`).
555///
556/// `Path::parent()` returns `Some("")` for a bare filename like
557/// `"run1"`; `create_dir_all` would error on that, so skip it.
558pub fn mkdir_parent(path: &str) -> anyhow::Result<()> {
559    if let Some(parent) = Path::new(path).parent() {
560        if !parent.as_os_str().is_empty() {
561            std::fs::create_dir_all(parent)?;
562        }
563    }
564    Ok(())
565}
566
567pub trait PathOsToStr {
568    #[allow(clippy::wrong_self_convention)]
569    fn into_boxed_str(&self) -> Box<str>;
570}
571
572impl PathOsToStr for Path {
573    fn into_boxed_str(&self) -> Box<str> {
574        self.to_str()
575            .expect("failed to convert to string")
576            .to_string()
577            .into_boxed_str()
578    }
579}
580
581impl PathOsToStr for OsStr {
582    fn into_boxed_str(&self) -> Box<str> {
583        self.to_str()
584            .expect("failed to convert to string")
585            .to_string()
586            .into_boxed_str()
587    }
588}
589
590/// more general file/directory copy function
591pub fn recursive_copy(src_path: &str, dst_path: &str) -> anyhow::Result<()> {
592    let src = Path::new(src_path);
593    let dst = Path::new(dst_path);
594
595    if src.is_dir() {
596        mkdir(dst_path)?;
597        for entry in std::fs::read_dir(src)? {
598            let entry = entry?;
599            if let (Some(src_path), Some(dst_path)) =
600                (entry.path().to_str(), dst.join(entry.file_name()).to_str())
601            {
602                let file_type = entry.file_type()?;
603                if file_type.is_dir() {
604                    recursive_copy(src_path, dst_path)?;
605                } else if file_type.is_file() {
606                    std::fs::copy(src_path, dst_path)?;
607                }
608            }
609        }
610    } else if src.is_file() {
611        if let Some(dir) = dirname(dst_path).as_deref() {
612            mkdir(dir)?;
613        }
614        std::fs::copy(src, dst)?;
615    } else if src.is_symlink() {
616        if let Ok(abs_src) = std::fs::read_link(src) {
617            if let Some(abs_src_path) = abs_src.to_str() {
618                recursive_copy(abs_src_path, dst_path)?;
619            }
620        }
621    }
622
623    Ok(())
624}
625
626/// Unzip `zip_path` into the `extract_path` If `extract_path` is
627/// `None`, just use a current directory.
628/// * Returns `extract_path`
629pub fn unzip_dir(zip_path: &str, extract_path: Option<&str>) -> anyhow::Result<Box<str>> {
630    let zip_file = std::fs::File::open(zip_path)?;
631    let mut archive = zip::ZipArchive::new(zip_file)?;
632
633    let extract_path = extract_path
634        .map(std::path::PathBuf::from)
635        .unwrap_or(std::env::current_dir()?);
636
637    for i in 0..archive.len() {
638        let mut file = archive.by_index(i)?;
639        let out_path = extract_path.join(file.name());
640        // println!("{}", out_path.to_str().unwrap());
641        if file.is_dir() {
642            std::fs::create_dir_all(&out_path)?;
643        } else {
644            if let Some(parent) = out_path.parent() {
645                std::fs::create_dir_all(parent)?;
646            }
647            let mut outfile = std::fs::File::create(&out_path)?;
648            std::io::copy(&mut file, &mut outfile)?;
649        }
650    }
651
652    Ok(extract_path.into_boxed_str())
653}
654
655/// Zip a directory into a zip archive using `Stored` compression
656/// (zarr chunks are already zstd-compressed internally). In-zip entries are
657/// prefixed with the source directory's basename.
658pub fn zip_dir(source_dir: &str, zip_path: &str) -> anyhow::Result<()> {
659    zip_dir_as(source_dir, zip_path, None)
660}
661
662/// Like [`zip_dir`] but lets the caller override the in-zip root name, so the
663/// archive can use a different prefix than the source directory's basename
664/// without renaming or moving the source on disk.
665pub fn zip_dir_as(
666    source_dir: &str,
667    zip_path: &str,
668    entry_root: Option<&str>,
669) -> anyhow::Result<()> {
670    use std::io::Write;
671    use zip::write::SimpleFileOptions;
672    use zip::ZipWriter;
673
674    let file = std::fs::File::create(zip_path)?;
675    let mut zip = ZipWriter::new(file);
676    let options = SimpleFileOptions::default().compression_method(zip::CompressionMethod::Stored);
677    let source = Path::new(source_dir);
678
679    fn collect_entries(dir: &Path, out: &mut Vec<std::path::PathBuf>) -> std::io::Result<()> {
680        for entry in std::fs::read_dir(dir)? {
681            let entry = entry?;
682            let path = entry.path();
683            out.push(path.clone());
684            // Use entry.file_type() (does not follow symlinks) to avoid
685            // infinite recursion on circular symlinks.
686            if entry.file_type()?.is_dir() {
687                collect_entries(&path, out)?;
688            }
689        }
690        Ok(())
691    }
692
693    let mut entries = vec![];
694    collect_entries(source, &mut entries)?;
695    entries.sort();
696
697    let root_name =
698        entry_root.unwrap_or_else(|| source.file_name().and_then(|s| s.to_str()).unwrap_or(""));
699
700    for path in &entries {
701        let rel_under_source = path.strip_prefix(source).unwrap_or(path);
702        let rel = if root_name.is_empty() {
703            rel_under_source.to_path_buf()
704        } else {
705            Path::new(root_name).join(rel_under_source)
706        };
707        if path.symlink_metadata()?.is_dir() {
708            zip.add_directory(format!("{}/", rel.display()), options)?;
709        } else {
710            zip.start_file(rel.display().to_string(), options)?;
711            let data = std::fs::read(path)?;
712            zip.write_all(&data)?;
713        }
714    }
715
716    zip.finish()?;
717    Ok(())
718}
719
720/// just get the directory (parent) name
721pub fn dirname(file_path: &str) -> Option<Box<str>> {
722    Path::new(file_path).parent().map(|x| x.into_boxed_str())
723}
724
725///
726/// Take the parent directory, basename, and extension of a file
727/// * `file` - file name
728///
729pub fn dir_base_ext(file_path: &str) -> anyhow::Result<(Box<str>, Box<str>, Box<str>)> {
730    let path = Path::new(file_path);
731
732    let dir = path
733        .parent()
734        .map_or(".".to_string().into_boxed_str(), |x| x.into_boxed_str());
735
736    let ext = path
737        .extension()
738        .map_or("".to_string().into_boxed_str(), |x| x.into_boxed_str());
739
740    let base = path
741        .file_stem()
742        .and_then(|x| x.to_str())
743        .map(|x| strip_data_ext(x).to_string().into_boxed_str())
744        .ok_or(anyhow::anyhow!("failed to find base here: {}", file_path))?;
745
746    Ok((dir, base, ext))
747}
748
749///
750/// Take the basename of a file
751/// * `file` - file name
752///
753pub fn basename(file: &str) -> anyhow::Result<Box<str>> {
754    let path = Path::new(file);
755    if let Some(base) = path.file_stem().and_then(|s| s.to_str()) {
756        Ok(strip_data_ext(base).to_string().into_boxed_str())
757    } else {
758        Err(anyhow::anyhow!("no file stem"))
759    }
760}
761
762/// Strip a leftover sparse-data extension (`.zarr`, `.h5`, `.h5ad`) that
763/// `Path::file_stem` leaves behind on doubly-extended paths like
764/// `sample.zarr.zip` (file_stem → `sample.zarr`).
765fn strip_data_ext(stem: &str) -> &str {
766    for sfx in [".zarr", ".h5ad", ".h5"] {
767        if let Some(s) = stem.strip_suffix(sfx) {
768            return s;
769        }
770    }
771    stem
772}
773
774///
775/// Take the extension of a file
776/// * `file` - file name
777///
778pub fn file_ext(file: &str) -> anyhow::Result<Box<str>> {
779    let path = Path::new(file);
780    if let Some(ext) = path.extension() {
781        Ok(ext.into_boxed_str())
782    } else {
783        Err(anyhow::anyhow!("failed to extract extension"))
784    }
785}
786
787///
788/// Create a temporary directory and suggest a file name
789/// * `suffix` - suffix of the file name
790///
791pub fn create_temp_dir_file(suffix: &str) -> anyhow::Result<std::path::PathBuf> {
792    let temp_dir = tempdir()?.path().to_path_buf();
793    std::fs::create_dir_all(&temp_dir)?;
794    let temp_file = tempfile::Builder::new()
795        .suffix(suffix)
796        .tempfile_in(temp_dir)?
797        .path()
798        .to_owned();
799
800    Ok(temp_file)
801}
802
803///
804/// Remove a file if it exists
805/// * `file` - file name
806///
807pub fn remove_file(file: &str) -> anyhow::Result<()> {
808    let path = Path::new(file);
809    if path.exists() {
810        if path.is_file() {
811            std::fs::remove_file(path)?;
812        } else {
813            std::fs::remove_dir_all(path)?;
814        }
815    }
816    Ok(())
817}
818
819///
820/// Remove a file if it exists
821/// * `files` - file name
822///
823pub fn remove_all_files(files: &Vec<Box<str>>) -> anyhow::Result<()> {
824    for file in files {
825        remove_file(file)?;
826    }
827    Ok(())
828}
829
830/// Extensions a data file's name is read through, stripped from the end of
831/// the name repeatedly by [`file_stem`]; dots inside the name stay.
832pub const DATA_FILE_EXTENSIONS: &[&str] = &[
833    "gz", "bgz", "bz2", "zst", "tsv", "csv", "txt", "tab", "gaf", "gmt", "obo", "bed", "vcf",
834    "parquet", "pq",
835];
836
837/// The file name minus its known extensions
838/// (`a/b/goa_human.gaf.gz` → `goa_human`, `c2.cp.v1.symbols.gmt` → `c2.cp.v1.symbols`):
839/// the name tools derive a relation or a source label from, so they agree.
840pub fn file_stem(path: &str) -> String {
841    let mut stem = std::path::Path::new(path)
842        .file_name()
843        .map(|s| s.to_string_lossy().into_owned())
844        .unwrap_or_else(|| path.to_string());
845    while let Some((base, ext)) = stem.rsplit_once('.') {
846        if base.is_empty() || !DATA_FILE_EXTENSIONS.contains(&ext.to_lowercase().as_str()) {
847            break;
848        }
849        stem.truncate(base.len());
850    }
851    stem
852}