Skip to main content

fastqc_rust/sequence/
mod.rs

1pub mod bam;
2pub mod casava;
3pub mod fast5;
4pub mod fastq;
5pub mod group;
6
7pub use bam::open_sequence_file;
8pub use group::SequenceFileGroup;
9
10use crate::utils::phred::PhredEncoding;
11
12/// A single sequence record with ID, bases, quality scores, and filter status.
13///
14/// Mirrors `Sequence.Sequence` in Java. The `sequence` field stores
15/// uppercase ASCII bases as bytes (matching Java's `toUpperCase()` in the constructor).
16/// The `quality` field stores raw ASCII quality characters as bytes.
17#[derive(Debug, Clone)]
18pub struct Sequence {
19    pub id: String,
20    /// Uppercase ASCII nucleotide bases (A, C, G, T, N).
21    pub sequence: Vec<u8>,
22    /// Raw ASCII quality characters (not yet offset-adjusted).
23    pub quality: Vec<u8>,
24    /// Whether this sequence was flagged as filtered (e.g. CASAVA filtered).
25    pub is_filtered: bool,
26    /// Colorspace representation, if applicable (SOLiD data).
27    pub colorspace: Option<Vec<u8>>,
28}
29
30impl Sequence {
31    /// Create a new Sequence, converting the base sequence to uppercase.
32    ///
33    /// The Java constructor calls `sequence.toUpperCase()` on the
34    /// sequence string, so we replicate that here.
35    pub fn new(id: String, mut sequence: Vec<u8>, quality: Vec<u8>) -> Self {
36        // uppercase conversion matches Java constructor behavior.
37        // In-place mutation avoids allocating a new Vec.
38        sequence.make_ascii_uppercase();
39        Self {
40            id,
41            sequence,
42            quality,
43            is_filtered: false,
44            colorspace: None,
45        }
46    }
47
48    /// Length of the sequence in bases.
49    pub fn len(&self) -> usize {
50        self.sequence.len()
51    }
52
53    /// Whether the sequence is empty.
54    pub fn is_empty(&self) -> bool {
55        self.sequence.is_empty()
56    }
57
58    /// Roughly how much heap this record holds, for the analysis pipeline's
59    /// memory budgeting. The `Sequence` itself is not counted — only what it
60    /// owns, which is everything that scales with read length.
61    pub fn heap_bytes(&self) -> usize {
62        self.id.len()
63            + self.sequence.len()
64            + self.quality.len()
65            + self.colorspace.as_ref().map_or(0, Vec::len)
66    }
67}
68
69/// Trait for reading sequences from various file formats.
70///
71/// Mirrors `Sequence.SequenceFile` interface.
72pub trait SequenceFile: Send {
73    /// Read the next sequence from the file, or None at EOF.
74    fn next(&mut self) -> Option<std::io::Result<Sequence>>;
75
76    /// The display name of this file (typically the filename).
77    fn name(&self) -> &str;
78
79    /// Whether this file contains colorspace data (SOLiD).
80    fn is_colorspace(&self) -> bool;
81
82    /// The quality encoding of this file, if the format specifies it.
83    ///
84    /// FASTQ carries no encoding information, so the default is `None` and
85    /// the encoding is inferred downstream from the lowest observed quality
86    /// character. BAM/SAM override this: their quality data is Phred+33 by
87    /// construction, so inference would be guessing at a known answer (and
88    /// guesses wrong when no base is below Q31).
89    fn known_phred_encoding(&self) -> Option<PhredEncoding> {
90        None
91    }
92
93    /// Estimated percentage complete (0.0 - 100.0), for progress display.
94    fn percent_complete(&self) -> f64;
95
96    /// Threads decompressing this file in the background, beside the one
97    /// calling [`next`](Self::next), so the runner can count them against the
98    /// thread budget.
99    fn background_threads(&self) -> usize {
100        0
101    }
102}