1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
use std::path::PathBuf;
use std::process;
use clap::Parser;
use fastqc_rust::config::{FastQCConfig, TemplateName};
use fastqc_rust::runner;
/// FastQC - A high throughput sequence QC analysis tool
#[derive(Parser, Debug)]
#[command(name = "fastqc", version = fastqc_rust::VERSION_BANNER, about)]
struct Cli {
/// Create report files in the specified output directory.
/// The directory must already exist.
#[arg(short, long, value_name = "DIR")]
outdir: Option<PathBuf>,
/// Force the file format. Valid formats: bam, sam, bam_mapped, sam_mapped, fastq.
#[arg(short, long, value_name = "FORMAT")]
format: Option<String>,
/// Files were generated by an Illumina CASAVA pipeline version >= 1.8.
/// Sequences flagged as filtered will be excluded from the analysis.
#[arg(long)]
casava: bool,
/// Files come from nanopore sequences and are in fast5 format.
/// In this mode you can pass in directories to process.
#[arg(long)]
nano: bool,
/// If running with --casava, don't remove reads flagged by CASAVA as poor quality.
#[arg(long)]
nofilter: bool,
/// Disable grouping of bases for reads > 50bp.
/// All reports will show data for every base in the read.
/// WARNING: Using this option with very long reads may cause excessive memory use.
#[arg(long)]
nogroup: bool,
/// Use exponential base groups in report.
#[arg(long)]
expgroup: bool,
/// Extract the zipped report file after creating it.
/// By default files are not extracted after creation.
#[arg(long)]
extract: bool,
/// Do not extract the zipped report file after creating it (default behavior).
#[arg(long)]
noextract: bool,
/// Delete the zipped output file after it has been extracted.
/// Only has an effect if --extract is also specified.
#[arg(long)]
delete: bool,
/// Specifies a non-default file which contains the list of contaminants to
/// screen overrepresented sequences against.
#[arg(short = 'c', long, value_name = "FILE")]
contaminants: Option<PathBuf>,
/// Specifies a non-default file which contains the list of adapter sequences
/// which will be explicitly searched against the library.
#[arg(short, long, value_name = "FILE")]
adapters: Option<PathBuf>,
/// Specifies a non-default file which contains a set of criteria used to
/// determine the warn/error limits for the various modules.
#[arg(short, long, value_name = "FILE")]
limits: Option<PathBuf>,
/// Total thread budget for the run [default: available CPUs, up to 6].
/// Spread across the files processed simultaneously and then within each
/// file (a reader, its gzip decoder, and analysis workers).
#[arg(short, long, value_name = "N", value_parser = parse_threads)]
threads: Option<usize>,
/// Background gzip decoders per .fastq.gz file. The first counts towards
/// --threads and keeps up with the analysis; any more do not count. 0
/// decodes on the reading thread, as -t 1 does.
#[arg(long = "decompress-threads", value_name = "N", default_value = "1")]
decompress_threads: usize,
/// Specifies the length of Kmer to look for in the Kmer content module.
/// Specified Kmer length must be between 2 and 10. Default length is 7.
#[arg(short, long, value_name = "N", default_value = "7")]
kmers: u8,
/// Suppress all progress messages on stdout and only report errors.
#[arg(short, long)]
quiet: bool,
/// Selects a directory to be used for temporary files written when
/// generating report images. Defaults to system temp dir.
#[arg(short, long, value_name = "DIR")]
dir: Option<PathBuf>,
/// Ignore reads shorter than this length (bp).
// Java uses --min_length (underscore), clap defaults to --min-length (hyphen).
// Allow both forms.
#[arg(
long = "min_length",
alias = "min-length",
value_name = "N",
default_value = "0"
)]
min_length: usize,
/// Ignore reads longer than this length (bp).
#[arg(
long = "max_length",
alias = "max-length",
value_name = "N",
default_value = "0"
)]
max_length: usize,
/// Specifies the truncation length used for duplicate detection.
/// Reads longer than this value will be truncated before checking for duplicates.
// Java uses --dup_length (underscore)
#[arg(
long = "dup_length",
alias = "dup-length",
value_name = "N",
default_value = "0"
)]
dup_length: usize,
/// Quality scores are encoded in the obsolete Phred64 format.
#[arg(long)]
phred64: bool,
/// Embed charts in the HTML report as PNG instead of SVG.
#[arg(long)]
png: bool,
/// No-op: SVG is now the default. Kept so existing commands still work.
#[arg(long, hide = true)]
svg: bool,
/// Select the HTML report template.
/// "classic" produces the original FastQC report layout.
/// "modern" uses a redesigned layout with responsive sidebar and help text.
#[arg(long, value_name = "NAME", default_value = "classic")]
template: TemplateName,
/// Input files (one or more FastQ, BAM, or SAM files).
#[arg(required = true)]
files: Vec<PathBuf>,
}
fn parse_threads(value: &str) -> Result<usize, String> {
match value.parse::<usize>() {
Ok(0) => Err("must be at least 1".to_string()),
Ok(n) => Ok(n),
Err(e) => Err(e.to_string()),
}
}
fn main() {
let cli = Cli::parse();
// Validate kmer range
if cli.kmers < 2 || cli.kmers > 10 {
eprintln!(
"Error: kmer size must be between 2 and 10, got {}",
cli.kmers
);
process::exit(1);
}
// Validate format if provided
if let Some(ref fmt) = cli.format {
match fmt.as_str() {
"bam" | "sam" | "bam_mapped" | "sam_mapped" | "fastq" => {}
_ => {
eprintln!(
"Error: unrecognized format '{}'. \
Valid formats: bam, sam, bam_mapped, sam_mapped, fastq",
fmt
);
process::exit(1);
}
}
}
if cli.min_length > 0 && cli.max_length > 0 && cli.max_length < cli.min_length {
eprintln!("Error: --max_length cannot be less than --min_length");
process::exit(1);
}
// Validate output directory exists if specified
if let Some(ref dir) = cli.outdir {
if !dir.is_dir() {
eprintln!(
"Error: output directory '{}' does not exist or is not a directory",
dir.display()
);
process::exit(1);
}
}
// Build config
let do_unzip = if cli.extract {
Some(true)
} else if cli.noextract {
Some(false)
} else {
None
};
let config = FastQCConfig {
nogroup: cli.nogroup,
expgroup: cli.expgroup,
quiet: cli.quiet,
kmer_size: cli.kmers,
threads: cli.threads,
output_dir: cli.outdir,
casava: cli.casava,
nano: cli.nano,
nofilter: cli.nofilter,
do_unzip,
delete_after_unzip: cli.delete,
sequence_format: cli.format,
contaminant_file: cli.contaminants,
adapter_file: cli.adapters,
limits_file: cli.limits,
min_length: cli.min_length,
max_length: cli.max_length,
dup_length: cli.dup_length,
phred64: cli.phred64,
png_output: cli.png,
temp_dir: cli.dir,
template: cli.template,
decompress_threads: cli.decompress_threads,
};
if let Err(exit_code) = runner::run(&config, &cli.files) {
process::exit(exit_code);
}
}