1use std::path::{Path, PathBuf};
10
11use polars::prelude::*;
12
13use crate::filter_modal::{FilterOperator, FilterStatement, LogicalOperator};
14use crate::pivot_melt_modal::PivotAggregation;
15use crate::{CompressionFormat, FileFormat, OpenOptions};
16
17pub(crate) fn py_str(s: &str) -> String {
19 let mut out = String::with_capacity(s.len() + 2);
20 out.push('"');
21 for c in s.chars() {
22 match c {
23 '\\' => out.push_str("\\\\"),
24 '"' => out.push_str("\\\""),
25 '\n' => out.push_str("\\n"),
26 '\r' => out.push_str("\\r"),
27 '\t' => out.push_str("\\t"),
28 c if c.is_control() => out.push_str(&format!("\\u{:04x}", c as u32)),
29 c => out.push(c),
30 }
31 }
32 out.push('"');
33 out
34}
35
36pub(crate) fn py_comment(text: &str) -> String {
40 let mut out = String::with_capacity(text.len() + 2);
41 out.push_str("# ");
42 for c in text.chars() {
43 match c {
44 '\n' => out.push_str("\\n"),
45 '\r' => out.push_str("\\r"),
46 '\t' => out.push('\t'),
47 c if c.is_control() || c == '\u{2028}' || c == '\u{2029}' => {
48 out.push_str(&format!("\\u{:04x}", c as u32))
49 }
50 c => out.push(c),
51 }
52 }
53 out
54}
55
56pub(crate) fn py_float(f: f64) -> String {
59 if f.is_nan() {
60 "float(\"nan\")".to_string()
61 } else if f.is_infinite() {
62 if f > 0.0 {
63 "float(\"inf\")".to_string()
64 } else {
65 "float(\"-inf\")".to_string()
66 }
67 } else {
68 format!("{f:?}")
70 }
71}
72
73pub(crate) fn py_bool(b: bool) -> &'static str {
74 if b { "True" } else { "False" }
75}
76
77pub(crate) fn py_names(names: &[String]) -> String {
79 let items: Vec<String> = names.iter().map(|n| py_str(n)).collect();
80 format!("[{}]", items.join(", "))
81}
82
83pub(crate) fn sort_call(columns: &[String], descending: &[bool]) -> String {
85 let by = match columns {
86 [one] => py_str(one),
87 _ => py_names(columns),
88 };
89 let descending = if descending.iter().all(|d| !d) {
90 String::new()
91 } else if descending.iter().all(|d| *d) {
92 "descending=True, ".to_string()
93 } else {
94 let flags: Vec<&str> = descending.iter().map(|d| py_bool(*d)).collect();
95 format!("descending=[{}], ", flags.join(", "))
96 };
97 format!(".sort({by}, {descending}nulls_last=True, maintain_order=True)")
98}
99
100#[derive(Debug, Clone, PartialEq)]
104pub enum FilterValue {
105 Typed(Scalar),
106 Str(String),
107}
108
109impl FilterValue {
110 fn lit(&self) -> Expr {
111 match self {
112 FilterValue::Typed(scalar) => lit(scalar.clone()),
113 FilterValue::Str(s) => lit(s.as_str()),
114 }
115 }
116
117 fn python(&self) -> String {
118 match self {
119 FilterValue::Typed(scalar) => crate::typed_value::python(scalar),
120 FilterValue::Str(s) => py_str(s),
121 }
122 }
123}
124
125#[derive(Debug, Clone, PartialEq)]
128pub struct SidebarFilter {
129 pub column: String,
130 pub operator: FilterOperator,
131 pub value: FilterValue,
132 pub text: String,
134 pub logical_op: LogicalOperator,
135 pub searched: Vec<(String, DataType)>,
138}
139
140impl SidebarFilter {
141 pub fn typed_in(statement: &FilterStatement, schema: &Schema, shown: &[String]) -> Self {
145 let mut filter = Self::typed(statement, schema.get(&statement.column));
146 if statement.operator.is_find() {
147 let spec = filter.find_spec();
148 let names: Vec<&String> = if statement.column == crate::filter_modal::ANY_COLUMN {
149 if statement.columns.is_empty() {
150 shown.iter().collect()
151 } else {
152 statement.columns.iter().collect()
153 }
154 } else {
155 vec![&statement.column]
156 };
157 filter.searched = names
158 .into_iter()
159 .filter_map(|name| Some((name.clone(), schema.get(name)?.clone())))
160 .filter(|(name, dtype)| crate::find::cell_matches(&spec, name, dtype).is_some())
161 .collect();
162 }
163 filter
164 }
165
166 pub fn unscriptable_columns(&self) -> Vec<String> {
169 if !self.operator.is_find() {
170 return Vec::new();
171 }
172 self.searched
173 .iter()
174 .filter(|(_, dtype)| matches!(dtype, DataType::Duration(_)))
175 .map(|(name, _)| name.clone())
176 .collect()
177 }
178
179 fn find_spec(&self) -> crate::find::FindSpec {
181 crate::find::FindSpec {
182 pattern: self.text.clone(),
183 regex: self.operator == FilterOperator::HasRegex,
184 fuzzy: self.operator == FilterOperator::HasFuzzy,
185 column: None,
186 }
187 }
188
189 pub fn typed(statement: &FilterStatement, dtype: Option<&DataType>) -> Self {
190 let text = statement.value.as_str();
191 let value = match dtype {
192 None | Some(DataType::String) => FilterValue::Str(text.to_string()),
193 Some(dtype) => crate::typed_value::parse(text, dtype)
194 .map(FilterValue::Typed)
195 .unwrap_or_else(|_| FilterValue::Str(text.to_string())),
196 };
197 Self {
198 column: statement.column.clone(),
199 operator: statement.operator,
200 value,
201 text: statement.value.clone(),
202 logical_op: statement.logical_op,
203 searched: Vec::new(),
204 }
205 }
206
207 pub fn problem(statement: &FilterStatement, dtype: Option<&DataType>) -> Option<String> {
211 if statement.operator == FilterOperator::HasRegex
213 && let Err(reason) = Self::typed(statement, dtype).find_spec().check()
214 {
215 return Some(reason);
216 }
217 if statement.operator.is_find()
220 && let Some(dtype) = dtype
221 && crate::find::cell_matches(
222 &Self::typed(statement, Some(dtype)).find_spec(),
223 &statement.column,
224 dtype,
225 )
226 .is_none()
227 {
228 return Some(format!("{}: no text to match", statement.column));
229 }
230 let compares = statement.operator.takes_value()
231 && !statement.operator.is_find()
232 && !matches!(
233 statement.operator,
234 FilterOperator::Contains | FilterOperator::NotContains
235 );
236 let dtype = dtype.filter(|_| compares)?;
237 crate::typed_value::parse(&statement.value, dtype)
238 .err()
239 .map(|why| format!("{}: {why}", statement.column))
240 }
241
242 fn expr(&self) -> Expr {
243 let column = col(&self.column);
244 let contains = || {
245 col(&self.column)
246 .str()
247 .contains_literal(lit(self.text.as_str()))
248 };
249 match self.operator {
250 FilterOperator::Eq => column.eq(self.value.lit()),
251 FilterOperator::NotEq => column.neq(self.value.lit()),
252 FilterOperator::Gt => column.gt(self.value.lit()),
253 FilterOperator::Lt => column.lt(self.value.lit()),
254 FilterOperator::GtEq => column.gt_eq(self.value.lit()),
255 FilterOperator::LtEq => column.lt_eq(self.value.lit()),
256 FilterOperator::Contains => contains(),
257 FilterOperator::NotContains => contains().not(),
258 FilterOperator::IsNull => column.is_null(),
259 FilterOperator::IsNotNull => column.is_not_null(),
260 FilterOperator::Has | FilterOperator::HasRegex | FilterOperator::HasFuzzy => {
261 let spec = self.find_spec();
262 let cells: Vec<Expr> = self
263 .searched
264 .iter()
265 .filter_map(|(name, dtype)| crate::find::cell_matches(&spec, name, dtype))
266 .collect();
267 cells
268 .into_iter()
269 .reduce(|a, b| a.or(b))
270 .unwrap_or(lit(false))
271 }
272 }
273 }
274
275 fn python(&self) -> String {
276 let column = format!("pl.col({})", py_str(&self.column));
277 let op = match self.operator {
278 FilterOperator::Eq => "==",
279 FilterOperator::NotEq => "!=",
280 FilterOperator::Gt => ">",
281 FilterOperator::Lt => "<",
282 FilterOperator::GtEq => ">=",
283 FilterOperator::LtEq => "<=",
284 FilterOperator::Contains | FilterOperator::NotContains => {
285 let not = if self.operator == FilterOperator::NotContains {
286 "~"
287 } else {
288 ""
289 };
290 return format!(
291 "{not}{column}.str.contains({}, literal=True)",
292 py_str(&self.text)
293 );
294 }
295 FilterOperator::IsNull => return format!("{column}.is_null()"),
296 FilterOperator::IsNotNull => return format!("{column}.is_not_null()"),
297 FilterOperator::Has | FilterOperator::HasRegex | FilterOperator::HasFuzzy => {
298 let spec = self.find_spec();
299 let cells: Vec<String> = self
302 .searched
303 .iter()
304 .filter(|(_, dtype)| !matches!(dtype, DataType::Duration(_)))
305 .map(|(name, _)| {
306 let column = format!("pl.col({}).cast(pl.String)", py_str(name));
307 match spec.regex_source() {
308 None => format!(
309 "{column}.str.contains({}, literal=True)",
310 py_str(&spec.pattern)
311 ),
312 Some(source) => {
313 format!("{column}.str.contains({})", py_str(&source))
314 }
315 }
316 })
317 .collect();
318 return match cells.len() {
319 0 => "pl.lit(False)".to_string(),
320 1 => format!("{}.fill_null(False)", cells[0]),
321 _ => format!("pl.any_horizontal({}).fill_null(False)", cells.join(", ")),
322 };
323 }
324 };
325 format!("{column} {op} {}", self.value.python())
326 }
327}
328
329pub fn filters_expr(filters: &[SidebarFilter]) -> Option<Expr> {
332 filters.iter().fold(None, |all, f| {
333 Some(match all {
334 None => f.expr(),
335 Some(all) => match f.logical_op {
336 LogicalOperator::And => all.and(f.expr()),
337 LogicalOperator::Or => all.or(f.expr()),
338 },
339 })
340 })
341}
342
343fn filters_python(filters: &[SidebarFilter]) -> String {
344 let mut out = String::new();
345 let mut last: Option<LogicalOperator> = None;
346 for (i, f) in filters.iter().enumerate() {
347 let term = format!("({})", f.python());
348 if i == 0 {
349 out = if filters.len() == 1 { f.python() } else { term };
351 continue;
352 }
353 if last.is_some_and(|l| l != f.logical_op) {
356 out = format!("({out})");
357 }
358 let op = match f.logical_op {
359 LogicalOperator::And => "&",
360 LogicalOperator::Or => "|",
361 };
362 out = format!("{out} {op} {term}");
363 last = Some(f.logical_op);
364 }
365 out
366}
367
368#[derive(Debug, Clone, PartialEq)]
370pub enum Step {
371 Query {
374 query: String,
375 input: SchemaRef,
376 keys: Vec<String>,
377 },
378 QueryRows {
380 query: String,
381 input: SchemaRef,
382 },
383 Sql {
386 sql: String,
387 ordered_by: Vec<String>,
388 },
389 Search {
391 patterns: Vec<String>,
392 columns: Vec<String>,
393 },
394 Filter(Vec<SidebarFilter>),
395 Sort {
396 columns: Vec<String>,
397 descending: Vec<bool>,
398 },
399 Reverse,
400 Select(Vec<String>),
401 Drop(Vec<String>),
402 Pivot {
403 index: Vec<String>,
404 on: String,
405 values: String,
406 aggregation: PivotAggregation,
407 },
408 Melt {
409 index: Vec<String>,
410 on: Vec<String>,
411 variable_name: String,
412 value_name: String,
413 },
414 Matching(Vec<(String, String)>),
417 Unreproducible(String),
419}
420
421impl Step {
422 fn python(&self) -> Vec<String> {
424 match self {
425 Step::Query { query, input, keys } => match crate::query::parse_nodes(query) {
426 Ok(mut nodes) => {
427 nodes.resolve_division(input);
428 nodes.resolve_time_zones(input);
429 nodes.python_steps(keys)
430 }
431 Err(e) => vec![py_comment(&format!("the query did not parse: {e}"))],
432 },
433 Step::QueryRows { query, input } => match crate::query::parse_nodes(query) {
434 Ok(mut nodes) => {
435 nodes.resolve_division(input);
436 nodes.resolve_time_zones(input);
437 nodes.python_filter().into_iter().collect()
438 }
439 Err(e) => vec![py_comment(&format!("the query did not parse: {e}"))],
440 },
441 Step::Sql { sql, ordered_by } => {
442 let sql = sql.trim();
443 let verbatim = sql.contains('\n')
446 && !sql.contains("\"\"\"")
447 && !sql.ends_with('"')
448 && sql
449 .chars()
450 .all(|c| c == '\n' || c == '\t' || (c != '\\' && !c.is_control()));
451 let mut lines = if verbatim {
452 vec![
453 ".sql(".to_string(),
454 format!(" \"\"\"{sql}\"\"\","),
455 " table_name=\"df\",".to_string(),
456 ")".to_string(),
457 ]
458 } else {
459 vec![format!(".sql({}, table_name=\"df\")", py_str(sql))]
460 };
461 if !ordered_by.is_empty() {
462 lines.push(sort_call(ordered_by, &vec![false; ordered_by.len()]));
463 }
464 lines
465 }
466 Step::Search { patterns, columns } => {
467 let terms: Vec<String> = patterns
468 .iter()
469 .map(|p| {
470 let any: Vec<String> = columns
471 .iter()
472 .map(|c| {
473 format!(
474 "pl.col({}).str.contains({}, strict=False)",
475 py_str(c),
476 py_str(p)
477 )
478 })
479 .collect();
480 if any.len() == 1 || patterns.len() == 1 {
481 any.join(" | ")
482 } else {
483 format!("({})", any.join(" | "))
484 }
485 })
486 .collect();
487 vec![format!(".filter({})", terms.join(" & "))]
488 }
489 Step::Filter(filters) => vec![format!(".filter({})", filters_python(filters))],
490 Step::Sort {
491 columns,
492 descending,
493 } => vec![sort_call(columns, descending)],
494 Step::Reverse => vec![".reverse()".to_string()],
495 Step::Select(columns) => vec![format!(".select({})", py_names(columns))],
496 Step::Drop(columns) => vec![format!(".drop({})", py_names(columns))],
497 Step::Pivot {
498 index,
499 on,
500 values,
501 aggregation,
502 } => {
503 let agg = match aggregation {
506 PivotAggregation::Last => "last()",
507 PivotAggregation::First => "first()",
508 PivotAggregation::Min => "min()",
509 PivotAggregation::Max => "max()",
510 PivotAggregation::Avg => "mean()",
511 PivotAggregation::Med => "median()",
512 PivotAggregation::Std => "std()",
513 PivotAggregation::Count => "len()",
514 };
515 let cell = match aggregation {
516 PivotAggregation::Count => "sum",
517 _ => "first",
518 };
519 let keys: Vec<String> = index.iter().chain([on]).cloned().collect();
520 vec![
521 format!(".group_by({}, maintain_order=True)", py_names(&keys)),
522 format!(".agg(pl.col({}).{agg})", py_str(values)),
523 ".collect()".to_string(),
524 ".pipe(".to_string(),
525 " lambda cells: cells.pivot(".to_string(),
526 format!(" on={},", py_str(on)),
527 format!(
528 " on_columns=cells[{}].unique().sort(nulls_last=True),",
529 py_str(on)
530 ),
531 format!(" index={},", py_names(index)),
532 format!(" values={},", py_str(values)),
533 format!(" aggregate_function={},", py_str(cell)),
534 " )".to_string(),
535 ")".to_string(),
536 ".lazy()".to_string(),
537 ]
538 }
539 Step::Melt {
540 index,
541 on,
542 variable_name,
543 value_name,
544 } => vec![format!(
545 ".unpivot(on={}, index={}, variable_name={}, value_name={})",
546 py_names(on),
547 py_names(index),
548 py_str(variable_name),
549 py_str(value_name)
550 )],
551 Step::Matching(keys) => {
552 let terms: Vec<String> = keys
553 .iter()
554 .map(|(key, value)| format!("{key}.eq_missing({value})"))
555 .collect();
556 let terms = if terms.len() == 1 {
557 terms
558 } else {
559 terms.into_iter().map(|t| format!("({t})")).collect()
560 };
561 vec![format!(".filter({})", terms.join(" & "))]
562 }
563 Step::Unreproducible(what) => vec![py_comment(what)],
564 }
565 }
566}
567
568#[derive(Debug, Clone, PartialEq)]
570pub enum Source {
571 Read {
575 call: String,
576 after: Vec<String>,
577 notes: Vec<String>,
578 imports: Vec<&'static str>,
580 },
581 Placeholder { what: String },
584}
585
586pub struct OpenRecord<'a> {
588 pub paths: Option<&'a [PathBuf]>,
590 pub options: &'a OpenOptions,
591 pub format: Option<FileFormat>,
595 pub read_mode: Option<crate::ReadMode>,
597 pub schema: &'a Schema,
599 pub remote_objects: Vec<String>,
601 pub s3_endpoint: Option<String>,
604 pub s3_region: Option<String>,
605 pub unsigned: bool,
607 pub read_as_text: Vec<String>,
609 pub spec: Option<String>,
611}
612
613fn is_url(path: &Path) -> bool {
614 crate::source::is_remote_url(path)
615}
616
617fn without_secrets(url: &str) -> (String, bool) {
621 let url = &*crate::source::split_source_id(url).1;
623 let Some(scheme_end) = url.find("://").map(|i| i + 3) else {
624 return (url.to_string(), false);
625 };
626 let (scheme, rest) = url.split_at(scheme_end);
627 let host_end = rest.find(['/', '?', '#']).unwrap_or(rest.len());
628 let (authority, path) = rest.split_at(host_end);
629 let azure = ["abfs://", "abfss://"]
631 .iter()
632 .any(|s| scheme.eq_ignore_ascii_case(s));
633 let host = match authority.rsplit_once('@') {
634 Some((_, host)) if !azure => host,
635 _ => authority,
636 };
637 let http = scheme.eq_ignore_ascii_case("http://") || scheme.eq_ignore_ascii_case("https://");
638 let path = match path.find(['?', '#']) {
639 Some(i) if http => &path[..i],
640 _ => path,
641 };
642 let kept = format!("{scheme}{host}{path}");
643 let cut = kept != url;
644 (kept, cut)
645}
646
647fn file_format(path: &Path, record: &OpenRecord) -> Option<FileFormat> {
650 record.format.or(record.options.format).or_else(|| {
651 FileFormat::from_path(path).or_else(|| {
652 CompressionFormat::from_extension(path)
653 .and_then(|_| path.file_stem())
654 .and_then(|stem| FileFormat::from_path(Path::new(stem)))
655 })
656 })
657}
658
659fn commonest_format<'a>(names: impl Iterator<Item = &'a str>) -> Option<FileFormat> {
661 let mut counts: Vec<(FileFormat, usize)> = Vec::new();
662 for name in names {
663 if let Some(format) = FileFormat::from_path(Path::new(name)) {
664 match counts.iter_mut().find(|(f, _)| *f == format) {
665 Some((_, n)) => *n += 1,
666 None => counts.push((format, 1)),
667 }
668 }
669 }
670 counts.into_iter().max_by_key(|(_, n)| *n).map(|(f, _)| f)
671}
672
673struct Target {
675 text: String,
678 format: FileFormat,
679 below: bool,
681 pattern: bool,
683 literal: bool,
685}
686
687fn reader_target(path: &Path, record: &OpenRecord) -> Option<Target> {
689 let text = path.to_string_lossy().to_string();
690 if is_url(path) {
691 let (text, _) = without_secrets(&text);
692 if let Some(format) = file_format(Path::new(&text), record) {
693 let pattern = crate::source::has_glob_chars(Path::new(&text));
694 return Some(Target {
695 text,
696 format,
697 below: false,
698 pattern,
699 literal: false,
700 });
701 }
702 let format = record
704 .format
705 .or(record.options.format)
706 .or_else(|| commonest_format(record.remote_objects.iter().map(String::as_str)))?;
707 let base = text.trim_end_matches('/');
708 let ext = format_extension(format)?;
709 return Some(Target {
710 text: format!("{base}/**/*.{ext}"),
711 format,
712 below: true,
713 pattern: true,
714 literal: false,
715 });
716 }
717 if path.is_dir() {
718 let entries: Vec<std::fs::DirEntry> = std::fs::read_dir(path).ok()?.flatten().collect();
719 let mut names: Vec<String> = entries
720 .iter()
721 .filter(|e| e.path().is_file())
722 .map(|e| e.file_name().to_string_lossy().to_string())
723 .collect();
724 if names
726 .iter()
727 .any(|n| FileFormat::from_path(Path::new(n)) == Some(FileFormat::Arrow))
728 {
729 names.retain(|n| !crate::discover::is_hugging_face_metadata(n));
730 }
731 let has_dirs = entries.iter().any(|e| e.path().is_dir());
732 let format = record
733 .format
734 .or(record.options.format)
735 .or_else(|| commonest_format(names.iter().map(String::as_str)));
736 let base = crate::source::escape_glob(text.trim_end_matches(['/', '\\']));
738 let (text, format, below) = match format {
741 Some(FileFormat::Parquet) | None if has_dirs => {
742 (format!("{base}/**/*.parquet"), FileFormat::Parquet, true)
743 }
744 Some(format) => (
745 format!("{base}/*.{}", format_extension(format)?),
746 format,
747 false,
748 ),
749 None => return None,
750 };
751 return Some(Target {
752 text,
753 format,
754 below,
755 pattern: true,
756 literal: false,
757 });
758 }
759 let format = file_format(path, record)?;
760 let pattern = crate::source::expands_as_glob(path);
761 Some(Target {
762 literal: !pattern && scans_by_pattern(format) && crate::source::has_glob_chars(path),
765 text,
766 format,
767 below: false,
768 pattern,
769 })
770}
771
772fn scans_by_pattern(format: FileFormat) -> bool {
774 python_of(format).is_some_and(|python| !python.eager)
775}
776
777fn hugging_face_files(paths: &[PathBuf], table: Option<&str>) -> Option<Vec<String>> {
780 let [path] = paths else {
781 return None;
782 };
783 if is_url(path) || !path.is_dir() {
784 return None;
785 }
786 let dict_split = crate::hf_splits::dataset_dict(path).and_then(|splits| {
788 let listed: Vec<&str> = splits.iter().map(String::as_str).collect();
789 crate::hf_splits::pick(&listed, table).ok()?.split
790 });
791 let dict = dict_split.is_some();
792 let table = if dict { None } else { table };
793 let path = &dict_split.map_or_else(|| path.clone(), |split| path.join(split));
794 let mut inside: Vec<PathBuf> = std::fs::read_dir(path)
795 .ok()?
796 .flatten()
797 .map(|e| e.path())
798 .filter(|p| p.is_file() && FileFormat::from_path(p) == Some(FileFormat::Arrow))
799 .collect();
800 inside.sort();
801 let cache = ["dataset_info.json", "state.json"]
802 .iter()
803 .any(|name| path.join(name).is_file());
804 if !cache && !dict {
805 return None;
806 }
807 if cache {
808 let names: Vec<&str> = inside
809 .iter()
810 .map(|f| f.file_name().and_then(|n| n.to_str()).unwrap_or_default())
811 .collect();
812 let (chosen, _) = crate::hf_splits::choose(&names, table).ok()?;
813 inside = chosen.into_iter().map(|i| inside[i].clone()).collect();
814 }
815 Some(
816 inside
817 .iter()
818 .map(|p| p.to_string_lossy().to_string())
819 .collect(),
820 )
821}
822
823fn arrow_read(inputs: &[(String, bool)], extra: Option<&str>) -> String {
827 let extra = extra.map(|e| format!(", {e}")).unwrap_or_default();
828 let names: Vec<String> = inputs.iter().map(|(name, _)| name.clone()).collect();
829 if inputs.iter().all(|(_, stream)| *stream) {
830 return match names.as_slice() {
831 [one] => format!("pl.read_ipc_stream({}{extra}).lazy()", py_str(one)),
832 many => format!(
833 "pl.concat([pl.read_ipc_stream(f{extra}) for f in {}]).lazy()",
834 py_names(many)
835 ),
836 };
837 }
838 if inputs.iter().all(|(_, stream)| !*stream) {
839 return match names.as_slice() {
840 [one] => format!("pl.scan_ipc({}{extra})", py_str(one)),
841 many => format!("pl.scan_ipc({}{extra})", py_names(many)),
842 };
843 }
844 let reads: Vec<String> = inputs
845 .iter()
846 .map(|(name, stream)| match stream {
847 true => format!("pl.read_ipc_stream({}{extra}).lazy()", py_str(name)),
848 false => format!("pl.scan_ipc({}{extra})", py_str(name)),
849 })
850 .collect();
851 format!(
852 "pl.concat([{}], how=\"diagonal_relaxed\")",
853 reads.join(", ")
854 )
855}
856
857fn format_extension(format: FileFormat) -> Option<&'static str> {
860 python_of(format)?;
861 let d = format.descriptor();
862 (d.many_files || format.separator().is_some())
863 .then(|| d.extensions.first().copied())
864 .flatten()
865}
866
867pub(crate) struct Python {
870 pub call: &'static str,
872 pub eager: bool,
874 pub glob_flag: bool,
876 pub arguments: Option<fn(&mut Call<'_>) -> Option<Source>>,
879}
880
881pub(crate) struct Call<'a> {
883 pub record: &'a OpenRecord<'a>,
884 pub paths: &'a [PathBuf],
885 pub format: FileFormat,
886 pub names: &'a [String],
888 pub below: bool,
890 pub storage: Option<String>,
892 pub args: Vec<String>,
893 pub after: Vec<String>,
894 pub skip_tail: Option<String>,
896 pub notes: Vec<String>,
897}
898
899impl Call<'_> {
900 fn storage_for(&self, names: impl IntoIterator<Item = impl AsRef<str>>) -> Option<String> {
902 names
903 .into_iter()
904 .any(|n| store_scheme(n.as_ref()).is_some())
905 .then(|| self.storage.clone())
906 .flatten()
907 }
908}
909
910fn store_scheme(name: &str) -> Option<&'static str> {
913 let (scheme, _) = name.split_once("://")?;
914 match scheme.to_ascii_lowercase().as_str() {
915 "s3" | "s3a" => Some("s3"),
916 "gs" | "gcs" => Some("gs"),
917 "az" | "adl" | "azure" | "abfs" | "abfss" => Some("azure"),
918 _ => None,
919 }
920}
921
922fn storage_options(name: &str, record: &OpenRecord, endpoint: Option<&str>) -> Option<String> {
926 let mut pairs: Vec<(&str, String)> = Vec::new();
927 match store_scheme(name)? {
928 "s3" => {
929 pairs.extend(endpoint.map(|e| ("aws_endpoint_url", e.to_string())));
930 pairs.extend(record.s3_region.clone().map(|r| ("aws_region", r)));
931 }
932 "azure" => {
933 pairs.extend(
934 crate::source::azure_parts(name).map(|(account, ..)| ("account_name", account)),
935 );
936 }
937 _ => {}
938 }
939 if record.unsigned {
940 pairs.push(("skip_signature", "true".to_string()));
941 }
942 let pairs: Vec<String> = pairs
943 .iter()
944 .map(|(k, v)| format!("{}: {}", py_str(k), py_str(v)))
945 .collect();
946 (!pairs.is_empty()).then(|| format!("storage_options={{{}}}", pairs.join(", ")))
947}
948
949pub(crate) fn ndjson_arguments(call: &mut Call<'_>) -> Option<Source> {
951 if let Some(s) = call.storage_for(call.names) {
952 call.args.push(s);
953 }
954 None
955}
956
957fn python_of(format: FileFormat) -> Option<&'static Python> {
958 crate::readers::of(format).python.as_ref()
959}
960
961pub(crate) fn parquet_arguments(call: &mut Call<'_>) -> Option<Source> {
963 if call.record.options.hive || call.below {
964 call.args.push("hive_partitioning=True".to_string());
965 }
966 if let Some(s) = call.storage_for(call.names) {
967 call.args.push(s);
968 }
969 None
970}
971
972pub(crate) fn csv_arguments(call: &mut Call<'_>) -> Option<Source> {
974 let options = call.record.options;
975 let names = call.names.join(", ");
976 let comment = options.comment_char.as_deref().filter(|c| !c.is_empty());
978 let dialect: Vec<&str> = [
979 (comment.is_some_and(|c| c.len() > 5), "--comment"),
980 (options.header_rows().is_some(), "--header-rows"),
981 (options.skip_initial_space, "--skip-initial-space"),
982 ]
983 .into_iter()
984 .filter_map(|(set, flag)| set.then_some(flag))
985 .collect();
986 if !dialect.is_empty() {
987 return Some(Source::Placeholder {
988 what: format!(
989 "{names}: datui read it with {}, which it cannot write as Python: load it here.",
990 dialect.join(", ")
991 ),
992 });
993 }
994 let separator = options
995 .delimiter
996 .or_else(|| call.format.separator())
997 .unwrap_or(b',');
998 let args = &mut call.args;
999 if separator != b',' {
1000 args.push(format!(
1001 "separator={}",
1002 py_str(&(separator as char).to_string())
1003 ));
1004 }
1005 if let Some(prefix) = comment {
1006 args.push(format!("comment_prefix={}", py_str(prefix)));
1007 }
1008 if options.has_header == Some(false) {
1009 args.push("has_header=False".to_string());
1010 }
1011 if let Some(n) = options.skip_lines {
1012 args.push(format!("skip_lines={n}"));
1013 }
1014 if let Some(n) = options.skip_rows {
1015 args.push(format!("skip_rows={n}"));
1016 }
1017 if let Some(n) = options.infer_schema_length {
1018 args.push(format!("infer_schema_length={n}"));
1019 }
1020 if options.ignore_errors {
1021 args.push("ignore_errors=True".to_string());
1022 }
1023 let bucket_prefix = call.below && call.paths.iter().any(|p| is_url(p));
1025 if options.csv_try_parse_dates() && !bucket_prefix {
1026 call.args.push("try_parse_dates=True".to_string());
1027 }
1028 if let Some(nulls) = csv_null_values(options, call.record.schema) {
1029 call.args.push(format!("null_values={nulls}"));
1030 }
1031 if !options.typing.text.is_empty() {
1033 let text: Vec<String> = options
1034 .typing
1035 .text
1036 .iter()
1037 .map(|name| format!("{}: pl.String", py_str(name)))
1038 .collect();
1039 call.args
1040 .push(format!("schema_overrides={{{}}}", text.join(", ")));
1041 }
1042 if let Some(s) = call.storage_for(call.names) {
1043 call.args.push(s);
1044 }
1045 if let Some(n) = options.skip_tail_rows.filter(|n| *n > 0) {
1046 call.skip_tail = Some(format!(".filter(pl.int_range(pl.len()) < pl.len() - {n})"));
1047 }
1048 match options.compression.or_else(|| {
1049 call.paths
1050 .first()
1051 .and_then(|p| CompressionFormat::from_extension(p))
1052 }) {
1053 Some(CompressionFormat::Bzip2 | CompressionFormat::Xz) => Some(Source::Placeholder {
1054 what: format!(
1055 "{names}: Polars cannot read bzip2 or xz; decompress it and read it with pl.scan_csv."
1056 ),
1057 }),
1058 _ => None,
1059 }
1060}
1061
1062pub(crate) fn arrow_arguments(call: &mut Call<'_>) -> Option<Source> {
1065 let options = call.record.options;
1066 let inputs: Option<Vec<(String, bool)>> = match &options.arrow_parts {
1067 Some(parts) => Some(
1068 parts
1069 .iter()
1070 .map(|part| match part {
1071 crate::ipc_stream::Part::InPlace(p) => (p, false),
1072 crate::ipc_stream::Part::Converted { source, .. } => (source, true),
1073 })
1074 .map(|(p, stream)| (without_secrets(&p.to_string_lossy()).0, stream))
1075 .collect(),
1076 ),
1077 None => hugging_face_files(call.paths, options.table.as_deref())
1078 .map(|files| files.into_iter().map(|f| (f, false)).collect()),
1079 };
1080 match inputs {
1081 Some(inputs) => {
1082 let extra = call.storage_for(inputs.iter().map(|(name, _)| name));
1083 let read = arrow_read(&inputs, extra.as_deref());
1084 let mut after = std::mem::take(&mut call.after);
1085 after.extend(call.skip_tail.take());
1086 Some(Source::Read {
1087 call: read,
1088 after,
1089 notes: std::mem::take(&mut call.notes),
1090 imports: Vec::new(),
1091 })
1092 }
1093 None => {
1094 if let Some(s) = call.storage_for(call.names) {
1095 call.args.push(s);
1096 }
1097 None
1098 }
1099 }
1100}
1101
1102pub(crate) fn excel_arguments(call: &mut Call<'_>) -> Option<Source> {
1104 if let Some(sheet) = &call.record.options.table {
1105 match sheet.parse::<usize>() {
1106 Ok(i) => call.args.push(format!("sheet_id={}", i + 1)),
1108 Err(_) => call.args.push(format!("sheet_name={}", py_str(sheet))),
1109 }
1110 }
1111 call.notes.push(
1112 "datui types a worksheet's columns itself; Polars may read some differently.".to_string(),
1113 );
1114 None
1115}
1116
1117fn sql_ident(name: &str) -> String {
1119 format!("\"{}\"", name.replace('"', "\"\""))
1120}
1121
1122fn whole(call: &mut Call<'_>, read: String, imports: Vec<&'static str>) -> Source {
1125 call.notes.extend(read_whole_note(call.record, &read));
1126 let mut after = std::mem::take(&mut call.after);
1127 after.extend(call.skip_tail.take());
1128 Source::Read {
1129 call: read,
1130 after,
1131 notes: std::mem::take(&mut call.notes),
1132 imports,
1133 }
1134}
1135
1136fn read_whole_note(record: &OpenRecord, call: &str) -> Option<String> {
1139 let name = call.split('(').next().unwrap_or(call);
1140 (record.read_mode == Some(crate::ReadMode::Lazy)).then(|| {
1141 format!(
1142 "Read: {} in datui; {name} reads the file whole into memory.",
1143 crate::ReadMode::Lazy.label()
1144 )
1145 })
1146}
1147
1148pub(crate) fn sqlite_arguments(call: &mut Call<'_>) -> Option<Source> {
1150 let [file] = call.names else {
1151 return None;
1152 };
1153 let Some(table) = call.record.options.table.as_deref() else {
1154 return Some(Source::Placeholder {
1155 what: format!("{file}: datui could not tell which table it read; load it here."),
1156 });
1157 };
1158 if call.paths.iter().any(|p| is_url(p)) {
1159 return Some(Source::Placeholder {
1160 what: format!(
1161 "{file} --table {table}: sqlite3 opens a local file; download it and read it \
1162 with pl.read_database."
1163 ),
1164 });
1165 }
1166 call.notes.push(
1167 "datui types a table's columns from their declared types; Polars infers them from \
1168 the values."
1169 .to_string(),
1170 );
1171 let query = format!("SELECT * FROM {}", sql_ident(table));
1172 let read = format!(
1173 "pl.read_database({}, sqlite3.connect({})).lazy()",
1174 py_str(&query),
1175 py_str(file)
1176 );
1177 Some(whole(call, read, vec!["import sqlite3"]))
1178}
1179
1180pub(crate) fn numpy_arguments(call: &mut Call<'_>) -> Option<Source> {
1182 let [file] = call.names else {
1183 return None;
1184 };
1185 let table = call.record.options.table.as_deref();
1186 let archive = table.is_some() || file.to_ascii_lowercase().ends_with(".npz");
1187 let array = match table {
1188 Some(name) => format!("np.load({})[{}]", py_str(file), py_str(name)),
1189 None if archive => format!("next(iter(np.load({}).values()))", py_str(file)),
1191 None => format!("np.load({})", py_str(file)),
1192 };
1193 let names: Vec<String> = call
1194 .record
1195 .schema
1196 .iter_names()
1197 .map(|n| n.to_string())
1198 .collect();
1199 let schema = if names.iter().any(|n| n.contains('.')) {
1201 call.notes.push(
1202 "datui names a nested field's columns outer.inner; Polars keeps the field as a struct."
1203 .to_string(),
1204 );
1205 String::new()
1206 } else {
1207 format!(", schema={}", py_names(&names))
1208 };
1209 let read = format!("pl.from_numpy({array}{schema}, orient=\"row\").lazy()");
1210 Some(whole(call, read, vec!["import numpy as np"]))
1211}
1212
1213fn named_with_table(names: &[String], record: &OpenRecord) -> String {
1216 let names = names.join(", ");
1217 match record.options.table.as_deref() {
1218 Some(table) => format!("{names} --table {table}"),
1219 None => names,
1220 }
1221}
1222
1223pub(crate) fn lines_arguments(call: &mut Call<'_>) -> Option<Source> {
1226 let names = call.names.join(", ");
1227 let path = match call.paths {
1228 [one]
1229 if !is_url(one)
1230 && !one.is_dir()
1231 && CompressionFormat::from_extension(one).is_none() =>
1232 {
1233 one
1234 }
1235 _ => {
1236 return Some(Source::Placeholder {
1237 what: format!("{names}: datui read it as lines; load it here."),
1238 });
1239 }
1240 };
1241 let read = format!(
1242 "pl.LazyFrame({{\"line\": open({}, encoding=\"utf-8\", errors=\"replace\", newline=\"\").read().removesuffix(\"\\n\").split(\"\\n\")}})",
1243 py_str(&path.to_string_lossy())
1244 );
1245 let mut after = vec![".with_columns(pl.col(\"line\").str.strip_suffix(\"\\r\"))".to_string()];
1246 after.append(&mut call.after);
1247 Some(Source::Read {
1248 call: read,
1249 after,
1250 notes: std::mem::take(&mut call.notes),
1251 imports: Vec::new(),
1252 })
1253}
1254
1255pub fn source(record: &OpenRecord) -> Source {
1257 let Some(paths) = record.paths else {
1258 return Source::Placeholder {
1259 what: "The data datui was handed: load it here as a DataFrame or LazyFrame."
1260 .to_string(),
1261 };
1262 };
1263 let teed;
1265 let paths = match (&record.options.tee, paths) {
1266 (Some(tee), [one]) if crate::stdin::is_stdin(one) => {
1267 teed = [tee.clone()];
1268 &teed[..]
1269 }
1270 _ => paths,
1271 };
1272 if paths.iter().any(|p| crate::stdin::is_stdin(p)) {
1273 return Source::Placeholder {
1274 what: "The data datui read from standard input: load it here.".to_string(),
1275 };
1276 }
1277 let spec = record.spec.clone().or_else(|| {
1278 let options = record.options;
1279 options
1280 .spec_name
1281 .clone()
1282 .or_else(|| options.spec_file.as_ref().map(|f| f.display().to_string()))
1283 });
1284 if let Some(spec) = spec {
1285 return Source::Placeholder {
1286 what: format!(
1287 "datui read this through the format spec {spec}, which it cannot write as \
1288 Python: load it here."
1289 ),
1290 };
1291 }
1292 let targets: Option<Vec<Target>> = paths.iter().map(|p| reader_target(p, record)).collect();
1293 let Some(targets) = targets.filter(|t| !t.is_empty()) else {
1294 let names: Vec<String> = paths
1295 .iter()
1296 .map(|p| without_secrets(&p.to_string_lossy()).0)
1297 .collect();
1298 return Source::Placeholder {
1299 what: format!(
1300 "{}: datui could not name a Polars reader for this data; load it here.",
1301 named_with_table(&names, record)
1302 ),
1303 };
1304 };
1305 let format = targets[0].format;
1306 if targets.iter().any(|t| t.format != format) {
1307 return Source::Placeholder {
1308 what: "The files are of more than one format: load them here.".to_string(),
1309 };
1310 }
1311 let below = targets.iter().any(|t| t.below);
1312 let python = python_of(format);
1313 let literal = targets.iter().any(|t| t.literal);
1316 let no_glob = literal
1317 && python.is_some_and(|python| python.glob_flag)
1318 && !targets.iter().any(|t| t.pattern);
1319 let names: Vec<String> = targets
1320 .into_iter()
1321 .map(|t| {
1322 if t.literal && !no_glob {
1323 crate::source::escape_glob(&t.text)
1324 } else {
1325 t.text
1326 }
1327 })
1328 .collect();
1329 let Some(python) = python else {
1330 return Source::Placeholder {
1331 what: format!(
1332 "{}: Polars has no reader for {} files; load it here.",
1333 named_with_table(&names, record),
1334 format.title()
1335 ),
1336 };
1337 };
1338 let target = match names.as_slice() {
1339 [one] => py_str(one),
1340 many => py_names(many),
1341 };
1342 let options = record.options;
1343 let mut args: Vec<String> = vec![target];
1344 if no_glob {
1345 args.push("glob=False".to_string());
1346 }
1347 let mut notes = Vec::new();
1348 let endpoint = record.s3_endpoint.as_deref().map(without_secrets);
1349 if paths
1350 .iter()
1351 .any(|p| is_url(p) && without_secrets(&p.to_string_lossy()).1)
1352 || endpoint.as_ref().is_some_and(|(_, cut)| *cut)
1353 {
1354 notes.push(
1355 "datui left a user, password or query string out of the URL, as it may be a \
1356 credential: add it back if the server needs it."
1357 .to_string(),
1358 );
1359 }
1360 let in_store = names.iter().find(|n| store_scheme(n).is_some());
1361 let storage = in_store
1362 .and_then(|name| storage_options(name, record, endpoint.as_ref().map(|(e, _)| e.as_str())));
1363 if let Some(name) = in_store
1365 && python.eager
1366 {
1367 notes.push(format!(
1368 "{} reads no object store: download {name} and read it from disk.",
1369 python.call
1370 ));
1371 }
1372 let mut call = Call {
1373 record,
1374 paths,
1375 format,
1376 names: &names,
1377 below,
1378 storage,
1379 args,
1380 after: options.read_python.clone(),
1382 skip_tail: None,
1383 notes,
1384 };
1385 if let Some(arguments) = python.arguments
1386 && let Some(source) = arguments(&mut call)
1387 {
1388 return source;
1389 }
1390 let Call {
1391 args,
1392 mut after,
1393 skip_tail,
1394 mut notes,
1395 ..
1396 } = call;
1397 after.extend(skip_tail);
1398 if !record.read_as_text.is_empty() {
1399 notes.push(format!(
1400 "datui read these columns as text from every file: {}.",
1401 record.read_as_text.join(", ")
1402 ));
1403 }
1404 let mut call = format!("{}({})", python.call, args.join(", "));
1405 if python.eager {
1406 notes.extend(read_whole_note(record, &call));
1407 call.push_str(".lazy()");
1408 }
1409 Source::Read {
1410 call,
1411 after,
1412 notes,
1413 imports: Vec::new(),
1414 }
1415}
1416
1417fn csv_null_values(options: &OpenOptions, schema: &Schema) -> Option<String> {
1421 let specs = options.null_values.as_ref().filter(|s| !s.is_empty())?;
1422 let mut global = Vec::new();
1423 let mut per_column: Vec<(String, String)> = Vec::new();
1424 for spec in specs {
1425 match spec.find('=') {
1426 Some(i) => per_column.push((spec[..i].to_string(), spec[i + 1..].to_string())),
1427 None => global.push(spec.clone()),
1428 }
1429 }
1430 let dict = |pairs: Vec<(String, String)>| {
1431 let items: Vec<String> = pairs
1432 .iter()
1433 .map(|(c, v)| format!("{}: {}", py_str(c), py_str(v)))
1434 .collect();
1435 format!("{{{}}}", items.join(", "))
1436 };
1437 Some(match (global.as_slice(), per_column.is_empty()) {
1438 ([one], true) => py_str(one),
1439 (_, true) => py_names(&global),
1440 ([], false) => dict(per_column),
1441 (_, false) => dict(
1442 schema
1443 .iter_names()
1444 .map(|name| {
1445 let value = per_column
1446 .iter()
1447 .rev()
1448 .find(|(c, _)| c == name.as_str())
1449 .map(|(_, v)| v.clone())
1450 .unwrap_or_else(|| global[0].clone());
1451 (name.to_string(), value)
1452 })
1453 .collect(),
1454 ),
1455 })
1456}
1457
1458pub fn py_value(value: &AnyValue) -> Option<String> {
1461 Some(match value {
1462 AnyValue::Null => "None".to_string(),
1463 AnyValue::Boolean(b) => py_bool(*b).to_string(),
1464 AnyValue::String(s) => py_str(s),
1465 AnyValue::StringOwned(s) => py_str(s),
1466 AnyValue::Int8(v) => v.to_string(),
1467 AnyValue::Int16(v) => v.to_string(),
1468 AnyValue::Int32(v) => v.to_string(),
1469 AnyValue::Int64(v) => v.to_string(),
1470 AnyValue::UInt8(v) => v.to_string(),
1471 AnyValue::UInt16(v) => v.to_string(),
1472 AnyValue::UInt32(v) => v.to_string(),
1473 AnyValue::UInt64(v) => v.to_string(),
1474 AnyValue::Float32(v) => py_float(f64::from(*v)),
1475 AnyValue::Float64(v) => py_float(*v),
1476 AnyValue::Date(days) => {
1477 let date = chrono::NaiveDate::from_ymd_opt(1970, 1, 1)?
1478 .checked_add_signed(chrono::Duration::days(i64::from(*days)))?;
1479 use chrono::Datelike;
1480 format!("pl.date({}, {}, {})", date.year(), date.month(), date.day())
1481 }
1482 _ => return None,
1483 })
1484}
1485
1486#[derive(Debug, Clone, PartialEq)]
1488pub struct Script {
1489 pub source: Source,
1490 pub steps: Vec<Step>,
1491}
1492
1493impl Script {
1494 pub fn render(&self) -> String {
1495 let mut out = String::from("import polars as pl\n");
1496 if let Source::Read { imports, .. } = &self.source {
1497 for import in imports {
1498 out.push_str(import);
1499 out.push('\n');
1500 }
1501 }
1502 out.push('\n');
1503 let (head, mut lines) = match &self.source {
1504 Source::Read {
1505 call, after, notes, ..
1506 } => {
1507 for note in notes {
1508 out.push_str(&py_comment(note));
1509 out.push('\n');
1510 }
1511 (call.clone(), after.clone())
1512 }
1513 Source::Placeholder { what } => {
1514 out.push_str(&py_comment(what));
1515 out.push_str("\ndf = ...\n\n");
1516 ("df.lazy()".to_string(), Vec::new())
1517 }
1518 };
1519 let mut stopped = false;
1522 for step in &self.steps {
1523 let calls = step.python();
1524 if stopped {
1525 for call in &calls {
1527 lines.extend(call.lines().map(|c| {
1528 if c.starts_with('#') {
1529 c.to_string()
1530 } else {
1531 format!("# {c}")
1532 }
1533 }));
1534 }
1535 } else {
1536 stopped = matches!(step, Step::Unreproducible(_));
1537 lines.extend(calls);
1538 }
1539 }
1540 if lines.is_empty() {
1541 out.push_str(&format!("df = {head}\n"));
1542 } else {
1543 out.push_str("df = (\n");
1544 out.push_str(&format!(" {head}\n"));
1545 for line in lines {
1546 out.push_str(&format!(" {line}\n"));
1547 }
1548 out.push_str(")\n");
1549 }
1550 out
1551 }
1552}
1553
1554#[cfg(test)]
1555mod tests {
1556 use super::*;
1557
1558 fn statement(column: &str, operator: FilterOperator, value: &str) -> FilterStatement {
1559 FilterStatement {
1560 columns: Vec::new(),
1561 column: column.to_string(),
1562 operator,
1563 value: value.to_string(),
1564 logical_op: LogicalOperator::And,
1565 }
1566 }
1567
1568 fn script(steps: Vec<Step>) -> String {
1569 Script {
1570 source: Source::Read {
1571 call: "pl.scan_parquet(\"sales.parquet\")".to_string(),
1572 after: Vec::new(),
1573 notes: Vec::new(),
1574 imports: Vec::new(),
1575 },
1576 steps,
1577 }
1578 .render()
1579 }
1580
1581 fn typed_frame() -> DataFrame {
1583 let tz = TimeZone::opt_try_new(Some("Europe/Paris")).unwrap();
1584 let us = |h: i64| 1_704_067_200_000_000 + h * 3_600_000_000;
1585 df!(
1586 "d" => &[Some(19723i32), Some(19724), Some(19725), None],
1587 "t" => &[Some(us(0)), Some(us(5)), Some(us(24)), None],
1588 "c" => &[Some(5 * 3_600_000_000_000i64), Some(6 * 3_600_000_000_000 + 500_000_000), Some(7 * 3_600_000_000_000), None],
1589 "du" => &[Some(1_000i64), Some(90_000), Some(3_600_000), None],
1590 "m" => &[Some("1.50"), Some("2.00"), Some("3.25"), None],
1591 "f" => &[Some(0.1f32), Some(0.2), Some(0.1), None],
1592 "x" => &[Some(0.1 + 0.2), Some(0.3), Some(1.0), None],
1593 )
1594 .unwrap()
1595 .lazy()
1596 .with_columns([
1597 col("d").cast(DataType::Date),
1598 col("t").cast(DataType::Datetime(TimeUnit::Microseconds, None)),
1599 col("t")
1600 .cast(DataType::Datetime(TimeUnit::Microseconds, tz))
1601 .alias("z"),
1602 col("c").cast(DataType::Time),
1603 col("du").cast(DataType::Duration(TimeUnit::Milliseconds)),
1604 col("m").cast(DataType::Decimal(10, 2)),
1605 ])
1606 .collect()
1607 .unwrap()
1608 }
1609
1610 #[test]
1611 fn every_operator_compares_in_the_columns_own_type() {
1612 use FilterOperator::*;
1613 let frame = typed_frame();
1614 let schema = frame.schema().clone();
1615 let rows = |column: &str, operator, value: &str| {
1616 let statement = statement(column, operator, value);
1617 assert_eq!(SidebarFilter::problem(&statement, schema.get(column)), None);
1618 let typed = SidebarFilter::typed(&statement, schema.get(column));
1619 frame
1620 .clone()
1621 .lazy()
1622 .filter(filters_expr(&[typed]).unwrap())
1623 .collect()
1624 .unwrap_or_else(|e| panic!("{column} {operator:?} {value}: {e}"))
1625 .height()
1626 };
1627 for (column, value, counts) in [
1628 ("d", "2024-01-02", [1, 2, 1, 1, 2, 2]),
1629 ("t", "2024-01-01", [1, 2, 2, 0, 3, 1]),
1631 ("t", "2024-01-01 05:00", [1, 2, 1, 1, 2, 2]),
1632 ("t", "2024-01-01T05:00:00.000", [1, 2, 1, 1, 2, 2]),
1633 ("z", "2024-01-01 06:00", [1, 2, 1, 1, 2, 2]),
1635 ("z", "2024-01-01 05:00+00:00", [1, 2, 1, 1, 2, 2]),
1636 ("c", "06:00:00.5", [1, 2, 1, 1, 2, 2]),
1637 ("du", "1m 30s", [1, 2, 1, 1, 2, 2]),
1638 ("m", "2", [1, 2, 1, 1, 2, 2]),
1639 ("m", "1.5", [1, 2, 2, 0, 3, 1]),
1640 ("x", "0.3", [1, 2, 2, 0, 3, 1]),
1642 ] {
1643 let got = [Eq, NotEq, Gt, Lt, GtEq, LtEq].map(|op| rows(column, op, value));
1644 assert_eq!(got, counts, "{column} {value}");
1645 }
1646 assert_eq!(rows("f", Eq, "0.1"), 2);
1648 assert_eq!(rows("x", IsNull, ""), 1);
1649 assert_eq!(rows("x", IsNotNull, ""), 3);
1650 }
1651
1652 #[test]
1653 fn a_value_that_does_not_read_as_the_column_says_so() {
1654 let frame = typed_frame();
1655 let schema = frame.schema();
1656 let problem = |column: &str, operator, value: &str| {
1657 SidebarFilter::problem(&statement(column, operator, value), schema.get(column))
1658 };
1659 assert_eq!(
1660 problem("d", FilterOperator::Eq, "2024-13-01").as_deref(),
1661 Some("d: \"2024-13-01\" is not a date written YYYY-MM-DD")
1662 );
1663 assert!(problem("t", FilterOperator::Gt, "soon").is_some());
1664 assert!(problem("x", FilterOperator::Lt, "abc").is_some());
1665 assert_eq!(problem("d", FilterOperator::Contains, "2024"), None);
1667 assert_eq!(problem("d", FilterOperator::IsNull, ""), None);
1668 }
1669
1670 #[test]
1671 fn typed_filters_read_back_in_python() {
1672 let frame = typed_frame();
1673 let schema = frame.schema();
1674 let python = |column: &str, operator, value: &str| {
1675 SidebarFilter::typed(&statement(column, operator, value), schema.get(column)).python()
1676 };
1677 assert_eq!(
1678 python("d", FilterOperator::Eq, "2024-01-02"),
1679 "pl.col(\"d\") == pl.date(2024, 1, 2)"
1680 );
1681 assert_eq!(
1682 python("t", FilterOperator::Gt, "2024-01-01 05:00"),
1683 "pl.col(\"t\") > pl.datetime(2024, 1, 1, 5, 0, 0, 0, time_unit=\"us\")"
1684 );
1685 assert_eq!(
1686 python("z", FilterOperator::LtEq, "2024-01-01 06:00"),
1687 "pl.col(\"z\") <= pl.datetime(2024, 1, 1, 5, 0, 0, 0, time_unit=\"us\", \
1688 time_zone=\"UTC\").dt.convert_time_zone(\"Europe/Paris\")"
1689 );
1690 assert_eq!(
1691 python("c", FilterOperator::Eq, "06:00:00.5"),
1692 "pl.col(\"c\") == pl.time(6, 0, 0, 500000)"
1693 );
1694 assert_eq!(
1695 python("du", FilterOperator::Lt, "1h"),
1696 "pl.col(\"du\") < pl.duration(milliseconds=3600000, time_unit=\"ms\")"
1697 );
1698 assert_eq!(
1699 python("m", FilterOperator::NotEq, "1.5"),
1700 "pl.col(\"m\") != pl.lit(\"1.50\").cast(pl.Decimal(10, 2))"
1701 );
1702 assert_eq!(
1703 python("x", FilterOperator::Eq, "0.3"),
1704 "pl.col(\"x\") == 0.3"
1705 );
1706 assert_eq!(
1707 python("x", FilterOperator::IsNull, ""),
1708 "pl.col(\"x\").is_null()"
1709 );
1710 }
1711
1712 #[test]
1713 fn strings_and_floats_read_back_in_python() {
1714 assert_eq!(py_str("a\"b\\c\nd"), "\"a\\\"b\\\\c\\nd\"");
1715 assert_eq!(py_str("\u{1}"), "\"\\u0001\"");
1716 assert_eq!(py_float(5.0), "5.0");
1717 assert_eq!(py_float(0.1), "0.1");
1718 assert_eq!(py_float(1e20), "1e20");
1719 assert_eq!(py_float(f64::NAN), "float(\"nan\")");
1720 }
1721
1722 #[test]
1723 fn text_in_a_comment_cannot_end_it() {
1724 assert_eq!(
1725 py_comment("a\nimport os\r\u{2028}x"),
1726 "# a\\nimport os\\r\\u2028x"
1727 );
1728 let text = script(vec![Step::Unreproducible("where k is \"\nboom()".into())]);
1729 assert!(text.contains(" # where k is \"\\nboom()\n"), "{text}");
1730 }
1731
1732 #[test]
1733 fn sql_is_triple_quoted_only_where_nothing_in_it_ends_the_string() {
1734 let sql = |sql: &str| {
1735 Step::Sql {
1736 sql: sql.into(),
1737 ordered_by: Vec::new(),
1738 }
1739 .python()
1740 };
1741 assert_eq!(
1742 sql("SELECT *\nFROM df"),
1743 vec![
1744 ".sql(",
1745 " \"\"\"SELECT *\nFROM df\"\"\",",
1746 " table_name=\"df\",",
1747 ")"
1748 ]
1749 );
1750 assert_eq!(
1751 sql("SELECT *\nFROM df ORDER BY \"a\""),
1752 vec![".sql(\"SELECT *\\nFROM df ORDER BY \\\"a\\\"\", table_name=\"df\")"]
1753 );
1754 let text = script(vec![
1756 Step::Unreproducible("drilled into a group held as lists".into()),
1757 Step::Sql {
1758 sql: "SELECT *\nFROM df".into(),
1759 ordered_by: Vec::new(),
1760 },
1761 ]);
1762 assert!(text.contains(" # FROM df\"\"\",\n"), "{text}");
1763 }
1764
1765 #[test]
1766 fn a_view_with_nothing_applied_is_the_reader() {
1767 assert_eq!(
1768 script(Vec::new()),
1769 "import polars as pl\n\ndf = pl.scan_parquet(\"sales.parquet\")\n"
1770 );
1771 }
1772
1773 #[test]
1774 fn filters_typed_by_column_then_a_multi_column_sort_then_a_projection() {
1775 let schema = Schema::from_iter([
1776 Field::new("region".into(), DataType::String),
1777 Field::new("amount".into(), DataType::Float64),
1778 Field::new("qty".into(), DataType::Int64),
1779 ]);
1780 let mut or = statement("qty", FilterOperator::GtEq, "3");
1781 or.logical_op = LogicalOperator::Or;
1782 let filters: Vec<SidebarFilter> = [
1783 statement("region", FilterOperator::Eq, "north"),
1784 statement("amount", FilterOperator::Gt, "10"),
1785 or,
1786 ]
1787 .iter()
1788 .map(|s| SidebarFilter::typed(s, schema.get(&s.column)))
1789 .collect();
1790 let text = script(vec![
1791 Step::Filter(filters),
1792 Step::Sort {
1793 columns: vec!["amount".into(), "region".into()],
1794 descending: vec![true, false],
1795 },
1796 Step::Select(vec!["order_id".into(), "customer".into(), "amount".into()]),
1797 ]);
1798 assert_eq!(
1799 text,
1800 "import polars as pl\n\n\
1801 df = (\n \
1802 pl.scan_parquet(\"sales.parquet\")\n \
1803 .filter(((pl.col(\"region\") == \"north\") & (pl.col(\"amount\") > 10.0)) | (pl.col(\"qty\") >= 3))\n \
1804 .sort([\"amount\", \"region\"], descending=[True, False], nulls_last=True, maintain_order=True)\n \
1805 .select([\"order_id\", \"customer\", \"amount\"])\n\
1806 )\n"
1807 );
1808 }
1809
1810 #[test]
1811 fn contains_filters_are_literal_and_a_number_that_does_not_parse_stays_text() {
1812 let s = SidebarFilter::typed(
1813 &statement("name", FilterOperator::NotContains, "a.b"),
1814 Some(&DataType::String),
1815 );
1816 assert_eq!(
1817 s.python(),
1818 "~pl.col(\"name\").str.contains(\"a.b\", literal=True)"
1819 );
1820 let s = SidebarFilter::typed(
1821 &statement("n", FilterOperator::Eq, "n/a"),
1822 Some(&DataType::Int64),
1823 );
1824 assert_eq!(s.value, FilterValue::Str("n/a".into()));
1825 }
1826
1827 #[test]
1828 fn one_sort_column_reads_plainly() {
1829 assert_eq!(
1830 sort_call(&["amount".into()], &[true]),
1831 ".sort(\"amount\", descending=True, nulls_last=True, maintain_order=True)"
1832 );
1833 }
1834
1835 #[test]
1836 fn steps_after_one_python_cannot_repeat_are_commented_out() {
1837 let text = script(vec![
1838 Step::Unreproducible("drilled into a group held as lists".into()),
1839 Step::Reverse,
1840 ]);
1841 assert!(
1842 text.contains(" # drilled into a group held as lists\n # .reverse()\n"),
1843 "{text}"
1844 );
1845 }
1846
1847 #[test]
1848 fn a_placeholder_source_leaves_df_to_the_user() {
1849 let text = Script {
1850 source: Source::Placeholder {
1851 what: "The data datui read from standard input: load it here.".into(),
1852 },
1853 steps: vec![Step::Reverse],
1854 }
1855 .render();
1856 assert_eq!(
1857 text,
1858 "import polars as pl\n\n\
1859 # The data datui read from standard input: load it here.\n\
1860 df = ...\n\n\
1861 df = (\n df.lazy()\n .reverse()\n)\n"
1862 );
1863 }
1864
1865 #[test]
1866 fn a_grouped_query_groups_then_orders_by_its_keys() {
1867 let input = Schema::from_iter([
1868 Field::new("dept".into(), DataType::String),
1869 Field::new("salary".into(), DataType::Float64),
1870 Field::new("id".into(), DataType::Int64),
1871 Field::new("age".into(), DataType::Int64),
1872 ]);
1873 let text = script(vec![Step::Query {
1874 query: "select avg salary, n: count id by dept where age > 30".into(),
1875 input: Arc::new(input),
1876 keys: vec!["dept".into()],
1877 }]);
1878 assert!(
1879 text.contains(
1880 " .filter(pl.col(\"age\") > 30.0)\n \
1881 .group_by(\"dept\")\n \
1882 .agg(pl.col(\"salary\").mean().alias(\"avg_salary\"), pl.col(\"id\").count().alias(\"n\"))\n \
1883 .sort(\"dept\", nulls_last=True, maintain_order=True)\n"
1884 ),
1885 "{text}"
1886 );
1887 }
1888
1889 #[test]
1890 fn csv_options_become_reader_arguments() {
1891 let mut options = OpenOptions::new();
1892 options.delimiter = Some(b';');
1893 options.has_header = Some(false);
1894 options.skip_rows = Some(2);
1895 options.null_values = Some(vec!["NA".into()]);
1896 options.skip_tail_rows = Some(1);
1897 let paths = vec![PathBuf::from("data/x.csv")];
1898 let schema = Schema::default();
1899 let record = OpenRecord {
1900 paths: Some(&paths),
1901 options: &options,
1902 schema: &schema,
1903 remote_objects: Vec::new(),
1904 s3_endpoint: None,
1905 s3_region: None,
1906 unsigned: false,
1907 format: None,
1908 read_mode: None,
1909 read_as_text: Vec::new(),
1910 spec: None,
1911 };
1912 let Source::Read { call, after, .. } = source(&record) else {
1913 panic!("a CSV has a reader");
1914 };
1915 assert_eq!(
1916 call,
1917 "pl.scan_csv(\"data/x.csv\", separator=\";\", has_header=False, skip_rows=2, \
1918 try_parse_dates=True, null_values=\"NA\")"
1919 );
1920 assert_eq!(
1921 after,
1922 vec![".filter(pl.int_range(pl.len()) < pl.len() - 1)"]
1923 );
1924 }
1925
1926 #[test]
1929 fn reads_python_cannot_repeat_leave_a_placeholder() {
1930 let schema = Schema::default();
1931 let placeholder = |paths: &[PathBuf], options: &OpenOptions, spec: Option<&str>| {
1932 let record = OpenRecord {
1933 paths: Some(paths),
1934 options,
1935 schema: &schema,
1936 remote_objects: Vec::new(),
1937 s3_endpoint: None,
1938 s3_region: None,
1939 unsigned: false,
1940 format: None,
1941 read_mode: None,
1942 read_as_text: Vec::new(),
1943 spec: spec.map(str::to_string),
1944 };
1945 match source(&record) {
1946 Source::Placeholder { what } => what,
1947 Source::Read { call, .. } => panic!("a reader was written: {call}"),
1948 }
1949 };
1950 let plain = OpenOptions::new();
1951 let what = placeholder(&[PathBuf::from("a.l2")], &plain, Some("acme.l2feed"));
1952 assert!(what.contains("acme.l2feed"), "{what}");
1953 let mut named = OpenOptions::new();
1954 named.spec_name = Some("acme.l2feed".into());
1955 placeholder(&[PathBuf::from("a.bin")], &named, None);
1956 placeholder(&[PathBuf::from("track.gpx")], &plain, None);
1957 placeholder(&[PathBuf::from("drive.nmea")], &plain, None);
1958 let csv = [PathBuf::from("log.csv")];
1959 let mut comment = OpenOptions::new();
1960 comment.comment_char = Some("######".into());
1961 assert!(placeholder(&csv, &comment, None).contains("--comment"));
1962 let mut rows = OpenOptions::new();
1963 rows.header_rows = vec![3, 2];
1964 assert!(placeholder(&csv, &rows, None).contains("--header-rows"));
1965 let mut space = OpenOptions::new();
1966 space.skip_initial_space = true;
1967 assert!(placeholder(&csv, &space, None).contains("--skip-initial-space"));
1968 }
1969
1970 fn record_for<'a>(
1971 paths: &'a [PathBuf],
1972 options: &'a OpenOptions,
1973 schema: &'a Schema,
1974 ) -> OpenRecord<'a> {
1975 OpenRecord {
1976 paths: Some(paths),
1977 options,
1978 format: None,
1979 read_mode: None,
1980 schema,
1981 remote_objects: Vec::new(),
1982 s3_endpoint: None,
1983 s3_region: None,
1984 unsigned: false,
1985 read_as_text: Vec::new(),
1986 spec: None,
1987 }
1988 }
1989
1990 fn call_of(source: Source) -> (String, Vec<String>) {
1991 match source {
1992 Source::Read { call, notes, .. } => (call, notes),
1993 Source::Placeholder { what } => panic!("a placeholder: {what}"),
1994 }
1995 }
1996
1997 #[test]
2001 fn every_store_reader_gets_its_storage_options() {
2002 let schema = Schema::default();
2003 let options = OpenOptions::new();
2004 let s3 = [PathBuf::from("s3://b/logs/a.jsonl")];
2005 let mut record = record_for(&s3, &options, &schema);
2006 record.s3_endpoint = Some("http://localhost:9000".into());
2007 record.s3_region = Some("us-east-1".into());
2008 assert_eq!(
2009 call_of(source(&record)).0,
2010 "pl.scan_ndjson(\"s3://b/logs/a.jsonl\", storage_options={\"aws_endpoint_url\": \
2011 \"http://localhost:9000\", \"aws_region\": \"us-east-1\"})"
2012 );
2013 let gcs = [PathBuf::from("gs://public/x.parquet")];
2014 let mut record = record_for(&gcs, &options, &schema);
2015 record.unsigned = true;
2016 assert_eq!(
2017 call_of(source(&record)).0,
2018 "pl.scan_parquet(\"gs://public/x.parquet\", storage_options={\"skip_signature\": \"true\"})"
2019 );
2020 let azure = [PathBuf::from(
2021 "abfss://data@acct.dfs.core.windows.net/t/x.csv",
2022 )];
2023 let (call, notes) = call_of(source(&record_for(&azure, &options, &schema)));
2024 assert!(
2025 call.starts_with("pl.scan_csv(\"abfss://data@acct.dfs.core.windows.net/t/x.csv\", ")
2026 && call.contains("storage_options={\"account_name\": \"acct\"}"),
2027 "{call}"
2028 );
2029 assert!(
2030 notes.is_empty(),
2031 "the container is no credential: {notes:?}"
2032 );
2033 let json = [PathBuf::from("s3://b/x.json")];
2034 let (call, notes) = call_of(source(&record_for(&json, &options, &schema)));
2035 assert_eq!(call, "pl.read_json(\"s3://b/x.json\").lazy()");
2036 assert!(notes[0].contains("reads no object store"), "{notes:?}");
2037 }
2038
2039 #[test]
2042 fn a_source_id_is_no_credential() {
2043 assert_eq!(
2044 without_secrets("s3://minio@bucket/x.parquet"),
2045 ("s3://bucket/x.parquet".to_string(), false)
2046 );
2047 assert_eq!(
2048 without_secrets("abfss://c@a.dfs.core.windows.net/x"),
2049 ("abfss://c@a.dfs.core.windows.net/x".to_string(), false)
2050 );
2051 }
2052
2053 #[test]
2055 fn a_teed_pipe_reads_its_file() {
2056 let schema = Schema::default();
2057 let mut options = OpenOptions::new();
2058 options.tee = Some(PathBuf::from("rec.csv"));
2059 let stdin = [PathBuf::from("-")];
2060 let mut record = record_for(&stdin, &options, &schema);
2061 record.format = Some(FileFormat::Csv);
2062 assert_eq!(
2063 call_of(source(&record)).0,
2064 "pl.scan_csv(\"rec.csv\", try_parse_dates=True)"
2065 );
2066 }
2067
2068 #[test]
2071 fn a_whole_read_of_a_lazy_table_says_so() {
2072 let schema = Schema::default();
2073 let mut options = OpenOptions::new();
2074 options.table = Some("orders".into());
2075 let db = [PathBuf::from("shop.db")];
2076 let mut record = record_for(&db, &options, &schema);
2077 record.format = Some(FileFormat::Sqlite);
2078 record.read_mode = Some(crate::ReadMode::Lazy);
2079 let (call, notes) = call_of(source(&record));
2080 assert!(call.starts_with("pl.read_database("), "{call}");
2081 assert!(
2082 notes.contains(
2083 &"Read: lazy scan in datui; pl.read_database reads the file whole into memory."
2084 .to_string()
2085 ),
2086 "{notes:?}"
2087 );
2088 let json = [PathBuf::from("a.json")];
2089 let mut record = record_for(&json, &options, &schema);
2090 record.read_mode = Some(crate::ReadMode::InMemory);
2091 assert!(call_of(source(&record)).1.is_empty());
2092 }
2093
2094 #[test]
2096 fn a_comment_character_is_the_comment_prefix() {
2097 let schema = Schema::default();
2098 let mut options = OpenOptions::new();
2099 options.comment_char = Some("#".into());
2100 let csv = [PathBuf::from("log.csv")];
2101 assert_eq!(
2102 call_of(source(&record_for(&csv, &options, &schema))).0,
2103 "pl.scan_csv(\"log.csv\", comment_prefix=\"#\", try_parse_dates=True)"
2104 );
2105 }
2106
2107 #[test]
2110 fn a_placeholder_names_the_format_read_and_the_table() {
2111 let schema = Schema::default();
2112 let paths = vec![PathBuf::from("flight.bin")];
2113 let mut options = OpenOptions::new();
2114 options.table = Some("GPS".into());
2115 let record = OpenRecord {
2116 paths: Some(&paths),
2117 options: &options,
2118 format: Some(FileFormat::Dataflash),
2119 read_mode: None,
2120 schema: &schema,
2121 remote_objects: Vec::new(),
2122 s3_endpoint: None,
2123 s3_region: None,
2124 unsigned: false,
2125 read_as_text: Vec::new(),
2126 spec: None,
2127 };
2128 let Source::Placeholder { what } = source(&record) else {
2129 panic!("Polars reads no DataFlash");
2130 };
2131 assert_eq!(
2132 what,
2133 "flight.bin --table GPS: Polars has no reader for DataFlash files; load it here."
2134 );
2135 }
2136
2137 #[test]
2138 fn credentials_in_a_url_stay_out_of_the_script() {
2139 assert_eq!(
2140 without_secrets("https://u:p@host.example/d/x.parquet?X-Amz-Signature=abc#f"),
2141 ("https://host.example/d/x.parquet".to_string(), true)
2142 );
2143 assert_eq!(
2144 without_secrets("s3://bucket/data-?.parquet"),
2145 ("s3://bucket/data-?.parquet".to_string(), false)
2146 );
2147 let options = OpenOptions::new();
2148 let schema = Schema::default();
2149 let paths = vec![PathBuf::from(
2150 "https://user:secret@host.example/d/x.csv?token=s3cr3t",
2151 )];
2152 let record = OpenRecord {
2153 paths: Some(&paths),
2154 options: &options,
2155 schema: &schema,
2156 remote_objects: Vec::new(),
2157 s3_endpoint: Some("http://key:secret@localhost:9000".into()),
2158 s3_region: None,
2159 unsigned: false,
2160 format: None,
2161 read_mode: None,
2162 read_as_text: Vec::new(),
2163 spec: None,
2164 };
2165 let text = Script {
2166 source: source(&record),
2167 steps: Vec::new(),
2168 }
2169 .render();
2170 assert!(
2171 !text.contains("secret") && !text.contains("s3cr3t"),
2172 "{text}"
2173 );
2174 assert!(text.contains("\"https://host.example/d/x.csv\""), "{text}");
2175 assert!(text.contains("# datui left a user"), "{text}");
2176 }
2177
2178 #[test]
2179 fn stdin_and_bucket_prefixes() {
2180 let options = OpenOptions::new();
2181 let schema = Schema::default();
2182 let stdin = vec![PathBuf::from("-")];
2183 let record = |paths: &'static [PathBuf]| OpenRecord {
2184 paths: Some(paths),
2185 options: &options,
2186 schema: &schema,
2187 remote_objects: vec!["s3://b/p/year=2024/a.parquet".into()],
2188 s3_endpoint: Some("http://localhost:9000".into()),
2189 s3_region: None,
2190 unsigned: false,
2191 format: None,
2192 read_mode: None,
2193 read_as_text: Vec::new(),
2194 spec: None,
2195 };
2196 let stdin: &'static [PathBuf] = Box::leak(stdin.into_boxed_slice());
2197 assert!(matches!(source(&record(stdin)), Source::Placeholder { .. }));
2198 let prefix: &'static [PathBuf] =
2199 Box::leak(vec![PathBuf::from("s3://b/p/")].into_boxed_slice());
2200 let Source::Read { call, .. } = source(&record(prefix)) else {
2201 panic!("a Parquet prefix has a reader");
2202 };
2203 assert_eq!(
2204 call,
2205 "pl.scan_parquet(\"s3://b/p/**/*.parquet\", hive_partitioning=True, \
2206 storage_options={\"aws_endpoint_url\": \"http://localhost:9000\"})"
2207 );
2208 }
2209
2210 #[test]
2213 fn a_kept_find_names_what_the_script_cannot_match() {
2214 let schema = Schema::from_iter([
2215 Field::new("name".into(), DataType::String),
2216 Field::new("took".into(), DataType::Duration(TimeUnit::Milliseconds)),
2217 ]);
2218 let statement = FilterStatement {
2219 columns: vec!["name".into(), "took".into()],
2220 column: crate::filter_modal::ANY_COLUMN.to_string(),
2221 operator: FilterOperator::Has,
2222 value: "1d".to_string(),
2223 logical_op: LogicalOperator::And,
2224 };
2225 let filter = SidebarFilter::typed_in(&statement, &schema, &[]);
2226 assert_eq!(filter.unscriptable_columns(), ["took"]);
2227 assert!(!filter.python().contains("took"), "{}", filter.python());
2228
2229 let bad = FilterStatement {
2230 columns: Vec::new(),
2231 column: "name".into(),
2232 operator: FilterOperator::HasRegex,
2233 value: "(".into(),
2234 logical_op: LogicalOperator::And,
2235 };
2236 let why = SidebarFilter::problem(&bad, Some(&DataType::String)).expect("refused");
2237 assert!(why.starts_with("Not a regex"), "{why}");
2238 }
2239
2240 #[test]
2243 fn a_kept_find_searches_the_shown_text_columns() {
2244 let schema = Schema::from_iter([
2245 Field::new("name".into(), DataType::String),
2246 Field::new("tags".into(), DataType::List(Box::new(DataType::String))),
2247 Field::new("note".into(), DataType::String),
2248 Field::new("n".into(), DataType::Int64),
2249 ]);
2250 let statement = FilterStatement {
2251 columns: Vec::new(),
2252 column: crate::filter_modal::ANY_COLUMN.to_string(),
2253 operator: FilterOperator::Has,
2254 value: "al".to_string(),
2255 logical_op: LogicalOperator::And,
2256 };
2257 let shown = ["name", "tags", "n"].map(String::from);
2258 let filter = SidebarFilter::typed_in(&statement, &schema, &shown);
2259 let names: Vec<&str> = filter.searched.iter().map(|(n, _)| n.as_str()).collect();
2260 assert_eq!(names, ["name", "n"]);
2261 let script = filter.python();
2262 assert!(
2263 !script.contains("tags") && !script.contains("note"),
2264 "{script}"
2265 );
2266 assert!(script.contains("pl.any_horizontal"), "{script}");
2267 }
2268}