datui-lib 0.4.1

Data Exploration in the Terminal (library)
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
//! Following an NDJSON file, journal JSON included, or one piped in: read by a scan of
//! datui's own that stops at the last newline written. A file still being written
//! usually ends partway through an object, which no JSON parser takes, and Polars'
//! NDJSON reader fails the whole read on it even when told to skip bad lines. Every
//! read of the file goes through this scan: the open's schema and summary, a count, a
//! filter, a page read from the start.

use std::fs::File;
use std::io::Cursor;
use std::num::NonZeroUsize;
use std::path::{Path, PathBuf};
use std::sync::Arc;

use polars::prelude::*;

use super::Spool;

/// The name the scan carries in a plan.
const SCAN_NAME: &str = "NDJSON";

/// The scan of a followed NDJSON file: its complete lines from the start, as many rows
/// as the bound asks for.
pub(crate) struct LinesScan {
    path: PathBuf,
    schema: SchemaRef,
    ignore_errors: bool,
    /// Standard input being copied to the file: once it has ended, a last line with
    /// no newline is a line too.
    spool: Option<Arc<Spool>>,
    /// The handle a deleted file is read through.
    held: Option<Arc<File>>,
}

/// How much of `bytes`, a file written a line at a time, is whole lines: through the
/// last newline, or all of it once nothing more is coming. Looks back from the end
/// only as far as the last line.
pub(crate) fn complete(bytes: &[u8], ended: bool) -> usize {
    if ended {
        return bytes.len();
    }
    memchr::memrchr(b'\n', bytes).map_or(0, |at| at + 1)
}

/// Where the line after the first `rows` lines that are not blank starts in `bytes`:
/// a read of that many rows stops there. Blank lines are not rows, as Polars reads them.
fn after_rows(bytes: &[u8], rows: usize) -> usize {
    let mut start = 0;
    let mut seen = 0;
    while start < bytes.len() && seen < rows {
        let end = memchr::memchr(b'\n', &bytes[start..]).map_or(bytes.len(), |at| start + at);
        if polars::io::ndjson::core::is_json_line(&bytes[start..end]) {
            seen += 1;
        }
        start = end + 1;
    }
    start.min(bytes.len())
}

/// Bytes parsed at once by a read of every row.
const RUN: usize = 64 << 20;

/// Where the run of whole lines from `start` ends: about `run` bytes on, at a line's
/// end, or at the end of `bytes`.
fn run_end(bytes: &[u8], start: usize, run: usize) -> usize {
    let at = start.saturating_add(run);
    if at >= bytes.len() {
        return bytes.len();
    }
    memchr::memchr(b'\n', &bytes[at..]).map_or(bytes.len(), |n| at + n + 1)
}

/// The file's bytes as they stand, mapped rather than read: a followed file can be
/// many times the memory. `None` when it is empty, which cannot be mapped.
fn map(file: &File) -> std::io::Result<Option<memmap2::Mmap>> {
    if file.metadata()?.len() == 0 {
        return Ok(None);
    }
    // SAFETY: the file is only ever appended to while it is followed; a truncation is
    // seen by the watcher, which reads it again from the start. Polars maps a followed
    // file the same way.
    Ok(Some(unsafe { memmap2::Mmap::map(file)? }))
}

impl LinesScan {
    /// The scan of the NDJSON file at `path`, its schema inferred from the first
    /// `infer` complete lines, or all of them.
    pub(crate) fn open(
        path: &Path,
        infer: Option<NonZeroUsize>,
        ignore_errors: bool,
        spool: Option<Arc<Spool>>,
    ) -> PolarsResult<LinesScan> {
        let ended = spool.as_ref().is_some_and(|s| s.ended().is_some());
        let file = File::open(path)?;
        let map = map(&file)?;
        let bytes = map.as_deref().unwrap_or_default();
        let bytes = &bytes[..complete(bytes, ended)];
        let bytes = bytes.strip_prefix(b"\xef\xbb\xbf").unwrap_or(bytes);
        let schema = match polars::io::ndjson::infer_schema(&mut Cursor::new(bytes), infer) {
            Err(_) if ignore_errors => {
                // A line that is not a JSON object is a row of nulls: the schema comes
                // from the lines that are.
                let mut objects = Vec::new();
                let wanted = infer.map_or(usize::MAX, NonZeroUsize::get);
                let lines = bytes
                    .split(|&b| b == b'\n')
                    .filter(|line| {
                        matches!(
                            serde_json::from_slice::<serde_json::Value>(line),
                            Ok(serde_json::Value::Object(_))
                        )
                    })
                    .take(wanted);
                for line in lines {
                    objects.extend_from_slice(line);
                    objects.push(b'\n');
                }
                polars::io::ndjson::infer_schema(&mut Cursor::new(objects), infer)?
            }
            inferred => inferred?,
        };
        Ok(LinesScan {
            path: path.to_path_buf(),
            schema: Arc::new(schema),
            ignore_errors,
            spool,
            held: None,
        })
    }

    /// The scan as a frame.
    pub(crate) fn lazy(self) -> PolarsResult<LazyFrame> {
        let schema = self.schema.clone();
        LazyFrame::anonymous_scan(
            Arc::new(self),
            ScanArgsAnonymous {
                schema: Some(schema),
                name: SCAN_NAME,
                ..Default::default()
            },
        )
    }

    /// The lines scan `scan` is, if it is one.
    pub(super) fn in_plan(scan: &polars::lazy::dsl::DslPlan) -> Option<&LinesScan> {
        use polars::lazy::dsl::{DslPlan, FileScanDsl};
        let DslPlan::Scan { scan_type, .. } = scan else {
            return None;
        };
        let FileScanDsl::Anonymous { function, .. } = &**scan_type else {
            return None;
        };
        function.as_any().downcast_ref::<LinesScan>()
    }

    /// The lines scan `scan` is, when it reads the file at `path` by its name.
    pub(super) fn of<'a>(
        scan: &'a polars::lazy::dsl::DslPlan,
        path: &str,
    ) -> Option<&'a LinesScan> {
        Self::in_plan(scan)
            .filter(|s| s.held.is_none() && super::same_file(&s.path.to_string_lossy(), path))
    }

    pub(super) fn ignore_errors(&self) -> bool {
        self.ignore_errors
    }

    pub(crate) fn schema(&self) -> &SchemaRef {
        &self.schema
    }

    /// This scan, reading through `file`: the file was deleted.
    pub(super) fn held(&self, file: &File) -> Option<LinesScan> {
        Some(LinesScan {
            held: Some(Arc::new(file.try_clone().ok()?)),
            ..self.with_schema(self.schema.clone())
        })
    }

    /// This scan, reading `schema`'s columns: the fields that arrived after the open
    /// joined to them.
    pub(crate) fn with_schema(&self, schema: SchemaRef) -> LinesScan {
        LinesScan {
            path: self.path.clone(),
            schema,
            ignore_errors: self.ignore_errors,
            spool: self.spool.clone(),
            held: self.held.clone(),
        }
    }
}

impl AnonymousScan for LinesScan {
    fn as_any(&self) -> &dyn std::any::Any {
        self
    }

    fn schema(&self, _infer_schema_length: Option<usize>) -> PolarsResult<SchemaRef> {
        Ok(self.schema.clone())
    }

    fn allows_predicate_pushdown(&self) -> bool {
        true
    }

    fn allows_projection_pushdown(&self) -> bool {
        true
    }

    fn scan(&self, mut args: AnonymousScanArgs) -> PolarsResult<DataFrame> {
        args.predicate = crate::pushdown::evaluable(args.predicate.take());
        // Asked before the file is mapped: an end heard after it could count a line the
        // map cut off.
        let ended = self.spool.as_ref().is_some_and(|s| s.ended().is_some());
        let file = match &self.held {
            Some(held) => held.try_clone()?,
            None => File::open(&self.path)?,
        };
        let columns: Schema = match args.with_columns.as_deref() {
            Some(names) => names
                .iter()
                .filter_map(|name| self.schema.get_field(name))
                .collect(),
            None => (*self.schema).clone(),
        };
        let map = map(&file)?;
        let bytes = map.as_deref().unwrap_or_default();
        let mut bytes = &bytes[..complete(bytes, ended)];
        if let Some(rows) = args.n_rows {
            bytes = &bytes[..after_rows(bytes, rows)];
        }
        if columns.is_empty() {
            // A count: the rows, with no columns to parse.
            let df = DataFrame::empty_with_height(polars::io::ndjson::count_rows(bytes));
            return match args.predicate {
                Some(predicate) => df.lazy().filter(predicate).collect(),
                None => Ok(df),
            };
        }
        let columns = Arc::new(columns);
        // A run of lines at a time, each filtered before the next is parsed: a filter
        // over a large file holds the rows it keeps, not every row's columns.
        let mut out = DataFrame::empty_with_schema(&columns);
        let mut start = 0;
        while start < bytes.len() {
            let end = run_end(bytes, start, RUN);
            let mut df = parse_run(&bytes[start..end], &columns, self.ignore_errors)?;
            if let Some(predicate) = &args.predicate {
                df = df.lazy().filter(predicate.clone()).collect()?;
            }
            out.vstack_mut(&df)?;
            start = end;
        }
        out.rechunk_mut();
        Ok(out)
    }
}

/// The rows of `run`, whole lines, read as `columns` by Polars' NDJSON line parser, as
/// a read from a mark is: its whole-file reader samples line lengths and panics on some
/// short lines. With `ignore_errors`, a line that is not JSON is a row of nulls, as the
/// watcher counts it, rather than the end of the read.
pub(super) fn parse_run(
    run: &[u8],
    columns: &Schema,
    ignore_errors: bool,
) -> PolarsResult<DataFrame> {
    let run = run.strip_prefix(b"\xef\xbb\xbf").unwrap_or(run);
    match polars::io::ndjson::core::parse_ndjson(run, None, columns, ignore_errors) {
        Err(_) if ignore_errors => {}
        read => return read,
    }
    let mut out = DataFrame::empty_with_schema(columns);
    for line in run.split(|&b| b == b'\n') {
        if !polars::io::ndjson::core::is_json_line(line) {
            continue;
        }
        let row = polars::io::ndjson::core::parse_ndjson(line, Some(1), columns, true)
            .unwrap_or_else(|_| DataFrame::full_null(columns, 1));
        out.vstack_mut(&row)?;
    }
    Ok(out)
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn only_whole_lines_are_complete_until_the_end() {
        assert_eq!(complete(b"", false), 0);
        assert_eq!(complete(b"{\"a\":", false), 0);
        assert_eq!(complete(b"{\"a\":1}\n{\"a\"", false), 8);
        assert_eq!(complete(b"{\"a\":1}\n", false), 8);
        assert_eq!(complete(b"{\"a\":1}\n{\"a\":2}", true), 15);
    }

    /// Short lines (blank, whitespace, not JSON) among the objects never fail a read:
    /// every read gives the objects, and a line that is not JSON a row of nulls.
    #[test]
    fn short_and_bad_lines_read_as_the_watcher_counts_them() {
        let dir = tempfile::tempdir().unwrap();
        let cases: [(&str, Vec<Option<i64>>); 4] = [
            ("{\"a\":1}\n\n   \n{\"a\":2}\n", vec![Some(1), Some(2)]),
            (
                "{\"a\":1}\n{\"a\":2}\ngarbage\n{\"a\":3}\n",
                vec![Some(1), Some(2), None, Some(3)],
            ),
            (
                "{\"a\":1} \n{\"a\":2}\t\n{\"a\":3}\n",
                vec![Some(1), Some(2), Some(3)],
            ),
            ("\u{feff}{\"a\":1}\n{\"a\":2}\n", vec![Some(1), Some(2)]),
        ];
        for (text, ids) in cases {
            let path = dir.path().join("short.ndjson");
            std::fs::write(&path, text).unwrap();
            let lf = LinesScan::open(&path, None, true, None)
                .unwrap()
                .lazy()
                .unwrap();
            let all = lf.clone().collect().unwrap();
            let got: Vec<Option<i64>> = all.column("a").unwrap().i64().unwrap().to_vec();
            assert_eq!(got, ids, "{text:?}");
            let head = lf.clone().slice(0, 2).collect().unwrap();
            assert_eq!(head.height(), 2, "{text:?}");
            let kept = lf.clone().filter(col("a").gt(lit(1))).collect().unwrap();
            let above = ids.iter().filter(|v| v.is_some_and(|v| v > 1)).count();
            assert_eq!(kept.height(), above, "{text:?}");
            let count = lf.clone().select([len()]).collect().unwrap();
            assert_eq!(
                count
                    .column("len")
                    .unwrap()
                    .get(0)
                    .unwrap()
                    .extract::<usize>(),
                Some(ids.len()),
                "{text:?}"
            );
        }
    }

    #[test]
    fn a_run_ends_at_a_line_s_end() {
        let bytes = b"{\"a\":1}\n{\"a\":22}\n{\"a\":3}";
        assert_eq!(run_end(bytes, 0, 3), 8);
        assert_eq!(run_end(bytes, 8, 3), 17);
        assert_eq!(run_end(bytes, 17, 3), bytes.len());
        assert_eq!(run_end(bytes, 0, 100), bytes.len());
    }

    #[test]
    fn a_read_of_some_rows_skips_blank_lines() {
        let bytes = b"{\"a\":1}\n\n{\"a\":2}\n  \n{\"a\":3}\n";
        assert_eq!(after_rows(bytes, 0), 0);
        assert_eq!(after_rows(bytes, 1), 8);
        assert_eq!(after_rows(bytes, 2), 17);
        assert_eq!(after_rows(bytes, 3), bytes.len());
        assert_eq!(after_rows(bytes, 9), bytes.len());
    }

    /// The scan reads what the file holds at each moment, through its last newline,
    /// wherever the writer has got to: never an error for an object cut off.
    #[test]
    fn a_cut_off_last_object_is_never_read() {
        let dir = tempfile::tempdir().unwrap();
        let path = dir.path().join("growing.ndjson");
        let mut text = String::new();
        for i in 0..200 {
            text.push_str(&format!("{{\"id\": {i}, \"msg\": \"line {i}\"}}\n"));
        }
        let bytes = text.as_bytes();
        let first = bytes.iter().position(|&b| b == b'\n').unwrap() + 1;
        std::fs::write(&path, &bytes[..first]).unwrap();
        let lf = LinesScan::open(&path, None, true, None)
            .unwrap()
            .lazy()
            .unwrap();
        for cut in (first..=bytes.len()).step_by(7).chain([bytes.len()]) {
            std::fs::write(&path, &bytes[..cut]).unwrap();
            let whole = bytes[..cut].iter().filter(|&&b| b == b'\n').count();
            let df = lf.clone().collect().unwrap();
            assert_eq!(df.height(), whole, "cut at {cut}");
            let count = lf.clone().select([len()]).collect().unwrap();
            assert_eq!(
                count
                    .column("len")
                    .unwrap()
                    .get(0)
                    .unwrap()
                    .extract::<usize>(),
                Some(whole),
                "count at {cut}"
            );
            let filtered = lf
                .clone()
                .filter(col("id").gt_eq(lit(0)))
                .select([col("msg")])
                .collect()
                .unwrap();
            assert_eq!(filtered.height(), whole, "filter at {cut}");
            let head = lf.clone().slice(0, 3).collect().unwrap();
            assert_eq!(head.height(), whole.min(3), "head at {cut}");
        }
    }
}