rudb_csv/dialect.rs
1//! What a CSV file's punctuation is, and working it out from the bytes.
2//!
3//! A CSV file does not say how it is written. The delimiter, the quote, the escape and whether the
4//! first line is a header are all conventions, and a reader that demands to be told them is a
5//! reader every loader script has to be rewritten for. DuckDB sniffs, so this sniffs, and every
6//! rule here was read off duckdb v1.4.1's `sniff_csv` rather than reasoned about.
7//!
8//! The candidates are DuckDB's: comma, pipe, semicolon and tab for the delimiter, and the double
9//! quote for the quote and the escape. A file with no quote character in it is reported as having
10//! no quote at all rather than as having the default one, which is visible in `sniff_csv` and is
11//! reproduced because the same field is printed back in an error message.
12
13use rudb_common::{Error, Result};
14
15/// The delimiters tried, in the order they are tried.
16///
17/// Order settles a tie and ties happen: a one column file of `a;b` has neither a comma nor a tab in
18/// it and is one column under both. Comma first is DuckDB's order and is the one that matters,
19/// since the file that arrives with no clue in it is a comma separated one often enough.
20pub const DELIMITERS: [u8; 4] = *b",|;\t";
21
22/// How a CSV file is punctuated.
23#[derive(Debug, Clone, Copy, PartialEq, Eq)]
24pub struct Dialect {
25 /// The byte between two fields.
26 pub delimiter: u8,
27 /// The byte that opens and closes a field that may hold a delimiter or a newline, if the file
28 /// has one.
29 pub quote: Option<u8>,
30 /// The byte that makes the next quote a literal one, if the file has one. When it is the quote
31 /// itself, which is what RFC 4180 says and what every writer does, a doubled quote is one
32 /// quote.
33 pub escape: Option<u8>,
34 /// Whether the first line names the columns rather than being one of them.
35 pub header: bool,
36}
37
38/// What the caller said about how the file is written, where the sniffer would otherwise decide.
39///
40/// Every field is optional and a `None` means nothing was said, which is the common case and is the
41/// one the sniffer is for. What is given is not sniffed: `read_csv('f.csv', delim=';')` does not try
42/// the four candidates and pick one, it uses the semicolon, and a file that is really comma
43/// separated then comes back as one column. That is DuckDB's behaviour and it is the useful one,
44/// since somebody who wrote the delimiter down knows something the first megabyte of the file does
45/// not say.
46///
47/// A given value also changes the block DuckDB prints under a conversion error, where a line reads
48/// `(Set By User)` rather than `(Auto-Detected)`, which is why this is carried into the reader
49/// rather than folded into a [`Dialect`] and forgotten.
50///
51/// The last three are not punctuation, and they are here because they travel the same road: the
52/// binder works them out from the call to sniff the file with and the executor works them out again
53/// from the plan to read it with, and a second struct beside this one would be a second thing for
54/// the two ends to keep in step.
55#[derive(Debug, Clone, Default, PartialEq, Eq)]
56pub struct Given {
57 /// The byte between two fields.
58 pub delimiter: Option<u8>,
59 /// The byte that opens and closes a field.
60 pub quote: Option<u8>,
61 /// The byte that makes the next quote a literal one.
62 pub escape: Option<u8>,
63 /// Whether the first line names the columns.
64 pub header: Option<bool>,
65 /// The strings that read as a null, which is `nullstr` on a call and `NULL` on a `COPY`.
66 ///
67 /// `None` is DuckDB's default, where an empty field is a null. A field is compared once its
68 /// quotes and escapes are gone, so `"NA"` is as much a null as `NA` is, and with any list here
69 /// an empty field is an empty string rather than a null. Both were measured on
70 /// `v2.0.0-dev84237`, where `allow_quoted_nulls` is on unless it is turned off, and it cannot
71 /// be turned off here.
72 pub nulls: Option<Vec<String>>,
73 /// What to call the columns, first column first, in place of what the header or the generator
74 /// called them. A list shorter than the file names the columns it reaches.
75 pub names: Option<Vec<String>>,
76 /// Whether the column types are the caller's rather than the sniffer's, which is what `COPY t
77 /// FROM` does with the table's.
78 ///
79 /// The reader converts to whatever types it is told either way. What this changes is the error
80 /// for a value that does not convert, which tells somebody who set the type to look at the data
81 /// and somebody who did not to set one.
82 pub typed: bool,
83}
84
85impl Given {
86 /// How a byte is written in the block under a conversion error, and where it came from.
87 #[must_use]
88 pub fn shown(given: Option<u8>, sniffed: Option<u8>) -> String {
89 format!("{} {}", Dialect::shown(sniffed), Self::source(given.is_some()))
90 }
91
92 /// The strings that read as a null, with DuckDB's default of the empty string written out as
93 /// `None` so that a caller who wrote `NULL ''` gets the same fast test as one who wrote nothing.
94 #[must_use]
95 pub fn null_strings(&self) -> Option<Vec<Vec<u8>>> {
96 match self.nulls.as_deref() {
97 None => None,
98 Some([only]) if only.is_empty() => None,
99 Some(nulls) => Some(nulls.iter().map(|text| text.as_bytes().to_vec()).collect()),
100 }
101 }
102
103 /// What the block calls a value the caller gave and one it worked out.
104 #[must_use]
105 pub const fn source(given: bool) -> &'static str {
106 if given { "(Set By User)" } else { "(Auto-Detected)" }
107 }
108}
109
110impl Dialect {
111 /// The dialect a file with nothing unusual in it has.
112 #[must_use]
113 pub const fn comma_separated() -> Self {
114 Self { delimiter: b',', quote: None, escape: None, header: true }
115 }
116
117 /// The quote byte, or the double quote when the file has none.
118 ///
119 /// A file with no quote in it still needs a byte to compare against while splitting, and the
120 /// one that cannot appear is the one that never appeared. Splitting on the default is what makes
121 /// a quote that turns up after the sample still read as a quote.
122 #[must_use]
123 pub const fn quote_byte(self) -> u8 {
124 match self.quote {
125 Some(quote) => quote,
126 None => b'"',
127 }
128 }
129
130 /// The escape byte, or the quote byte when the file has none.
131 #[must_use]
132 pub const fn escape_byte(self) -> u8 {
133 match self.escape {
134 Some(escape) => escape,
135 None => self.quote_byte(),
136 }
137 }
138
139 /// How a byte is written in the block DuckDB prints under a CSV error.
140 ///
141 /// A tab is `\t` there and a byte that is nothing is `(empty)`, which is why this takes the
142 /// option rather than the byte.
143 #[must_use]
144 pub fn shown(byte: Option<u8>) -> String {
145 match byte {
146 None => "(empty)".to_string(),
147 Some(b'\t') => "\\t".to_string(),
148 Some(byte) => (byte as char).to_string(),
149 }
150 }
151}
152
153/// Which delimiter splits `sample` into the most columns, consistently.
154///
155/// Consistently is the whole test. Every candidate splits every line into some number of fields,
156/// and the one to take is the candidate under which all the lines agree, because a delimiter that
157/// is really just a character inside the data will land in some lines and not others. Among the
158/// candidates that agree, the one that found the most columns wins, since a file is more likely to
159/// have three columns separated by something than one column containing it.
160///
161/// # Errors
162///
163/// When the sample has no complete line in it, which is a file of one unterminated line and is the
164/// one case where there is nothing to count.
165pub fn delimiter(sample: &[u8], quote: Option<u8>) -> Result<u8> {
166 let mut best = (1usize, DELIMITERS[0]);
167 for candidate in DELIMITERS {
168 let dialect = Dialect { delimiter: candidate, quote, escape: quote, header: false };
169 let Some(width) = consistent_width(sample, dialect) else { continue };
170 if width > best.0 {
171 best = (width, candidate);
172 }
173 }
174 if sample.iter().all(|&byte| byte != b'\n' && byte != b'\r') && sample.is_empty() {
175 return Err(Error::io("the file is empty"));
176 }
177 Ok(best.1)
178}
179
180/// The number of fields every line has under `dialect`, when they all have the same number.
181fn consistent_width(sample: &[u8], dialect: Dialect) -> Option<usize> {
182 let mut at = 0;
183 let mut fields = Vec::new();
184 let mut width = None;
185 let mut lines = 0;
186 while at < sample.len() {
187 let next = crate::scan::record(sample, at, dialect, true, &mut fields).ok()??;
188 at = next;
189 lines += 1;
190 match width {
191 None => width = Some(fields.len()),
192 Some(held) if held == fields.len() => {}
193 Some(_) => return None,
194 }
195 }
196 if lines == 0 { None } else { width }
197}
198
199/// Whether the file uses a quote character at all, and which one.
200///
201/// Only the double quote is looked for, which is what `sniff_csv` reports and what every writer
202/// emits. A field is quoted when the first byte after a delimiter or a line start is the quote, so
203/// a double quote sitting in the middle of a field does not make the file a quoted one.
204#[must_use]
205pub fn quote(sample: &[u8]) -> Option<u8> {
206 let mut at_field_start = true;
207 for &byte in sample {
208 if at_field_start && byte == b'"' {
209 return Some(b'"');
210 }
211 at_field_start = byte == b'\n' || byte == b'\r' || DELIMITERS.contains(&byte);
212 }
213 None
214}
215
216#[cfg(test)]
217mod tests {
218 use super::*;
219
220 #[test]
221 fn the_delimiter_is_the_one_every_line_agrees_on() {
222 assert_eq!(delimiter(b"a,b,c\n1,2,3\n", None).unwrap(), b',');
223 assert_eq!(delimiter(b"a|b\n1|x\n", None).unwrap(), b'|');
224 assert_eq!(delimiter(b"a;b\n1;x\n", None).unwrap(), b';');
225 assert_eq!(delimiter(b"a\tb\n1\tx\n", None).unwrap(), b'\t');
226 }
227
228 #[test]
229 fn a_character_that_lands_in_some_lines_and_not_others_is_not_the_delimiter() {
230 // The semicolon splits the first line into two and the second into one, so it is a
231 // character in the data. The comma splits both into two and is the answer.
232 let sample = b"a,b;c\n1,2\n";
233 assert_eq!(delimiter(sample, None).unwrap(), b',');
234 }
235
236 #[test]
237 fn the_delimiter_that_finds_more_columns_wins_among_the_ones_that_agree() {
238 // Every line is one field under a tab and three under a comma, and both are consistent.
239 assert_eq!(delimiter(b"a,b,c\nx,y,z\n", None).unwrap(), b',');
240 }
241
242 #[test]
243 fn a_file_with_nothing_to_split_on_is_comma_separated_and_one_column() {
244 assert_eq!(delimiter(b"a\nb\n", None).unwrap(), b',');
245 }
246
247 #[test]
248 fn a_delimiter_inside_a_quoted_field_does_not_count() {
249 let sample = b"a,b\n1,\"x,y\"\n";
250 assert_eq!(delimiter(sample, Some(b'"')).unwrap(), b',');
251 }
252
253 #[test]
254 fn a_quote_is_found_where_a_field_starts_and_not_in_the_middle_of_one() {
255 assert_eq!(quote(b"a,b\n1,\"x\"\n"), Some(b'"'));
256 assert_eq!(quote(b"a,b\n1,x\n"), None);
257 assert_eq!(quote(b"a,b\n1,he said \"hi\"\n"), None);
258 }
259
260 #[test]
261 fn the_block_duckdb_prints_writes_a_tab_as_two_characters() {
262 assert_eq!(Dialect::shown(None), "(empty)");
263 assert_eq!(Dialect::shown(Some(b'\t')), "\\t");
264 assert_eq!(Dialect::shown(Some(b',')), ",");
265 }
266}