use polars::prelude::*;
use std::io::Write as _;
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum BackendChoice {
#[default]
Auto,
Native,
Osc52,
}
impl BackendChoice {
pub fn parse(s: &str) -> Option<Self> {
match s.trim().to_ascii_lowercase().as_str() {
"auto" => Some(Self::Auto),
"native" => Some(Self::Native),
"osc52" => Some(Self::Osc52),
_ => None,
}
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Payload {
pub text: String,
pub html: Option<String>,
}
impl Payload {
pub fn text(text: String) -> Self {
Self { text, html: None }
}
}
pub trait Destination {
fn write(&mut self, payload: Payload) -> Result<(), String>;
fn describe(&self) -> &'static str;
fn accepts(&self) -> Accepts {
Accepts {
html: true,
base64_limit: None,
}
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct Accepts {
pub html: bool,
pub base64_limit: Option<usize>,
}
pub struct Native {
clipboard: arboard::Clipboard,
}
impl Native {
pub fn new() -> Result<Self, String> {
arboard::Clipboard::new()
.map(|clipboard| Self { clipboard })
.map_err(|e| format!("clipboard unavailable: {e}"))
}
}
impl Destination for Native {
fn write(&mut self, payload: Payload) -> Result<(), String> {
let result = match payload.html {
Some(html) => self.clipboard.set_html(html, Some(payload.text)),
None => self.clipboard.set_text(payload.text),
};
result.map_err(|e| format!("copy failed: {e}"))
}
fn describe(&self) -> &'static str {
"clipboard"
}
}
pub struct Osc52 {
pub limit: usize,
}
impl Destination for Osc52 {
fn write(&mut self, payload: Payload) -> Result<(), String> {
let sequence = osc52_sequence(&payload.text, self.limit)?;
drop(payload);
let mut out = std::io::stdout();
out.write_all(sequence.as_bytes())
.and_then(|()| out.flush())
.map_err(|e| format!("copy failed: {e}"))
}
fn describe(&self) -> &'static str {
"terminal"
}
fn accepts(&self) -> Accepts {
Accepts {
html: false,
base64_limit: Some(self.limit),
}
}
}
pub fn base64_len(bytes: usize) -> usize {
bytes.div_ceil(3).saturating_mul(4)
}
pub fn osc52_sequence(text: &str, limit: usize) -> Result<String, String> {
use base64::Engine as _;
let encoded = base64_len(text.len());
if encoded > limit {
return Err(over_osc52_limit(Some(encoded), limit));
}
let mut sequence = String::with_capacity(encoded + 8);
sequence.push_str("\x1b]52;c;");
base64::engine::general_purpose::STANDARD.encode_string(text.as_bytes(), &mut sequence);
sequence.push('\x07');
Ok(sequence)
}
pub(crate) fn over_osc52_limit(encoded: Option<usize>, limit: usize) -> String {
let size = match encoded {
Some(bytes) => format_kb(bytes),
None => format!("over {}", format_kb(limit)),
};
format!(
"the copy is {size} of base64 and the terminal path is capped at {} \
(raise [clipboard] osc52_limit, or export to a file)",
format_kb(limit),
)
}
fn format_kb(bytes: usize) -> String {
format!("{} KB", bytes.div_ceil(1024))
}
pub fn destination(
choice: BackendChoice,
osc52_limit: usize,
) -> Result<Box<dyn Destination>, String> {
match choice {
BackendChoice::Native => Native::new().map(|n| Box::new(n) as Box<dyn Destination>),
BackendChoice::Osc52 => Ok(Box::new(Osc52 { limit: osc52_limit })),
BackendChoice::Auto => Ok(match Native::new() {
Ok(native) => Box::new(native),
Err(_) => Box::new(Osc52 { limit: osc52_limit }),
}),
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum CopyFormat {
#[default]
Tsv,
Csv,
Markdown,
}
impl CopyFormat {
pub const ALL: [Self; 3] = [Self::Tsv, Self::Csv, Self::Markdown];
pub fn as_str(self) -> &'static str {
match self {
Self::Tsv => "TSV",
Self::Csv => "CSV",
Self::Markdown => "Markdown",
}
}
}
pub fn delimited(df: &DataFrame, separator: u8, header: bool) -> Result<String, String> {
let mut out = Vec::new();
let mut df = crate::nested_json::frame_as_json(df).map_err(|e| e.to_string())?;
CsvWriter::new(&mut out)
.with_separator(separator)
.include_header(header)
.finish(&mut df)
.map_err(|e| e.to_string())?;
let mut text = String::from_utf8(out).map_err(|e| e.to_string())?;
while text.ends_with('\n') || text.ends_with('\r') {
text.pop();
}
Ok(text)
}
pub fn markdown(df: &DataFrame) -> Result<String, String> {
let mut layout = MarkdownLayout::new(df);
layout.measure(df)?;
let mut out = String::with_capacity(layout.len(df.height()));
layout.write(df, &mut out)?;
Ok(out)
}
struct MarkdownLayout {
names: Vec<String>,
numeric: Vec<bool>,
widths: Vec<usize>,
}
impl MarkdownLayout {
fn new(df: &DataFrame) -> Self {
let names: Vec<String> = df
.get_column_names()
.iter()
.map(|name| markdown_escape(name))
.collect();
let widths = names.iter().map(|n| n.chars().count().max(3)).collect();
let numeric = df
.columns()
.iter()
.map(|c| c.dtype().is_primitive_numeric())
.collect();
Self {
names,
numeric,
widths,
}
}
fn measure(&mut self, df: &DataFrame) -> Result<(), String> {
for (column, width) in df.columns().iter().zip(&mut self.widths) {
let series = column.as_materialized_series();
for row in 0..df.height() {
*width = (*width).max(markdown_cell(series, row)?.chars().count());
}
}
Ok(())
}
fn len(&self, rows: usize) -> usize {
let line = self.widths.iter().sum::<usize>() + 3 * self.widths.len() + 1;
(rows + 2) * line + rows + 1
}
fn write_line<'a>(&self, out: &mut String, cells: impl Iterator<Item = &'a str>) {
out.push_str("| ");
for (i, ((cell, &width), &right)) in cells.zip(&self.widths).zip(&self.numeric).enumerate()
{
if i > 0 {
out.push_str(" | ");
}
let fill = width - cell.chars().count();
if right {
out.extend(std::iter::repeat_n(' ', fill));
out.push_str(cell);
} else {
out.push_str(cell);
out.extend(std::iter::repeat_n(' ', fill));
}
}
out.push_str(" |");
}
fn write(&self, df: &DataFrame, out: &mut String) -> Result<(), String> {
self.write_line(out, self.names.iter().map(String::as_str));
out.push_str("\n|");
for (i, (&width, &right)) in self.widths.iter().zip(&self.numeric).enumerate() {
if i > 0 {
out.push('|');
}
out.push(' ');
if right {
out.extend(std::iter::repeat_n('-', width.saturating_sub(1)));
out.push(':');
} else {
out.extend(std::iter::repeat_n('-', width));
}
out.push(' ');
}
out.push('|');
self.write_rows(df, out)
}
fn write_rows(&self, df: &DataFrame, out: &mut String) -> Result<(), String> {
let series: Vec<&Series> = df
.columns()
.iter()
.map(Column::as_materialized_series)
.collect();
let mut cells = Vec::with_capacity(series.len());
for row in 0..df.height() {
cells.clear();
for s in &series {
cells.push(markdown_cell(s, row)?);
}
out.push('\n');
self.write_line(out, cells.iter().map(String::as_str));
}
Ok(())
}
}
fn markdown_escape(s: &str) -> String {
s.replace('|', "\\|").replace(['\n', '\r'], " ")
}
fn markdown_cell(series: &Series, row: usize) -> Result<String, String> {
Ok(match series.get(row).map_err(|e| e.to_string())? {
AnyValue::Null => String::new(),
v => markdown_escape(&crate::exact::value_text(&v)),
})
}
pub fn html_table(df: &DataFrame, header: bool) -> Result<String, String> {
let escape = |s: &str| {
s.replace('&', "&")
.replace('<', "<")
.replace('>', ">")
};
let column_names = df.get_column_names_owned();
let mut out = String::from("<table>");
if header {
out.push_str("<thead><tr>");
for name in &column_names {
out.push_str(&format!("<th>{}</th>", escape(name)));
}
out.push_str("</tr></thead>");
}
out.push_str("<tbody>");
for row in 0..df.height() {
out.push_str("<tr>");
for name in &column_names {
let value = df
.column(name)
.map_err(|e| e.to_string())?
.as_materialized_series()
.get(row)
.map_err(|e| e.to_string())?;
let text = match value {
AnyValue::Null => String::new(),
v => crate::exact::value_text(&v),
};
out.push_str(&format!("<td>{}</td>", escape(&text)));
}
out.push_str("</tr>");
}
out.push_str("</tbody></table>");
Ok(out)
}
pub fn tabular_payload(
df: &DataFrame,
format: CopyFormat,
header: bool,
html: bool,
) -> Result<Payload, String> {
let df = &crate::nested_json::frame_as_cells(df).map_err(|e| e.to_string())?;
let text = match format {
CopyFormat::Tsv => delimited(df, b'\t', header)?,
CopyFormat::Csv => delimited(df, b',', header)?,
CopyFormat::Markdown => markdown(df)?,
};
let html = match format {
CopyFormat::Tsv | CopyFormat::Csv if html => Some(html_table(df, header)?),
_ => None,
};
Ok(Payload { text, html })
}
const BOUNDED_BATCH_ROWS: usize = 1024;
pub fn bounded_table_text(
lf: LazyFrame,
format: CopyFormat,
header: bool,
limit: usize,
) -> Result<(String, usize), String> {
use std::sync::{Arc, Mutex};
let polars_error = |e: PolarsError| crate::error_display::user_message_from_polars(&e);
let schema = lf.clone().collect_schema().map_err(polars_error)?;
let state = Arc::new(Mutex::new(BoundedText::new(format, header, limit)));
let sink_state = Arc::clone(&state);
let sink = lf
.sink_batches(
PlanCallback::new(move |batch: DataFrame| {
let mut text = sink_state
.lock()
.map_err(|_| PolarsError::ComputeError("copy lock failed".into()))?;
Ok(text.take(batch))
}),
true,
std::num::NonZeroUsize::new(BOUNDED_BATCH_ROWS),
)
.map_err(polars_error)?;
crate::statistics::collect_lazy(sink, true).map_err(polars_error)?;
let mut text = std::mem::replace(
&mut *state.lock().map_err(|_| "copy lock failed".to_string())?,
BoundedText::new(format, header, limit),
);
if !text.started {
text.take(DataFrame::empty_with_schema(&schema));
}
text.finish()
}
struct BoundedText {
format: CopyFormat,
header: bool,
limit: usize,
started: bool,
text: String,
rows: usize,
markdown: Option<(MarkdownLayout, Vec<DataFrame>)>,
over: bool,
error: Option<String>,
}
impl BoundedText {
fn new(format: CopyFormat, header: bool, limit: usize) -> Self {
Self {
format,
header,
limit,
started: false,
text: String::new(),
rows: 0,
markdown: None,
over: false,
error: None,
}
}
fn take(&mut self, batch: DataFrame) -> bool {
if self.over || self.error.is_some() {
return true;
}
if let Err(e) = self.try_take(batch) {
self.error = Some(e);
}
self.over || self.error.is_some()
}
fn try_take(&mut self, batch: DataFrame) -> Result<(), String> {
let first = !self.started;
self.started = true;
self.rows += batch.height();
let batch = crate::nested_json::frame_as_cells(&batch).map_err(|e| e.to_string())?;
let separator = match self.format {
CopyFormat::Tsv => b'\t',
CopyFormat::Csv => b',',
CopyFormat::Markdown => {
let (layout, rows) = self
.markdown
.get_or_insert_with(|| (MarkdownLayout::new(&batch), Vec::new()));
layout.measure(&batch)?;
self.over = base64_len(layout.len(self.rows)) > self.limit;
rows.push(batch);
return Ok(());
}
};
let mut out = Vec::new();
let mut batch = crate::nested_json::frame_as_json(&batch).map_err(|e| e.to_string())?;
CsvWriter::new(&mut out)
.with_separator(separator)
.include_header(first && self.header)
.finish(&mut batch)
.map_err(|e| e.to_string())?;
self.text
.push_str(&String::from_utf8(out).map_err(|e| e.to_string())?);
self.over = base64_len(self.text.len().saturating_sub(1)) > self.limit;
Ok(())
}
fn finish(mut self) -> Result<(String, usize), String> {
if let Some(e) = self.error {
return Err(e);
}
if self.over {
return Err(over_osc52_limit(None, self.limit));
}
if let Some((layout, frames)) = self.markdown.take() {
self.text.reserve(layout.len(self.rows));
let mut frames = frames.iter();
if let Some(first) = frames.next() {
layout.write(first, &mut self.text)?;
}
for frame in frames {
layout.write_rows(frame, &mut self.text)?;
}
}
while self.text.ends_with('\n') || self.text.ends_with('\r') {
self.text.pop();
}
let encoded = base64_len(self.text.len());
if encoded > self.limit {
return Err(over_osc52_limit(Some(encoded), self.limit));
}
Ok((self.text, self.rows))
}
}
#[cfg(test)]
mod tests {
use super::*;
fn markdown_reference(df: &DataFrame) -> Result<String, String> {
let column_names = df.get_column_names_owned();
let escape = |s: &str| s.replace('|', "\\|").replace(['\n', '\r'], " ");
let mut names: Vec<String> = Vec::with_capacity(column_names.len());
let mut cells: Vec<Vec<String>> = Vec::with_capacity(column_names.len());
let mut numeric: Vec<bool> = Vec::with_capacity(column_names.len());
for name in &column_names {
let column = df.column(name).map_err(|e| e.to_string())?;
names.push(escape(name));
numeric.push(column.dtype().is_primitive_numeric());
let series = column.as_materialized_series();
let mut body = Vec::with_capacity(df.height());
for i in 0..df.height() {
let value = series.get(i).map_err(|e| e.to_string())?;
body.push(match value {
AnyValue::Null => String::new(),
v => escape(&crate::exact::value_text(&v)),
});
}
cells.push(body);
}
let widths: Vec<usize> = names
.iter()
.zip(&cells)
.map(|(name, body)| {
body.iter()
.map(|c| c.chars().count())
.max()
.unwrap_or(0)
.max(name.chars().count())
.max(3)
})
.collect();
let pad = |s: &str, w: usize, right: bool| {
let fill = " ".repeat(w - s.chars().count());
if right {
format!("{fill}{s}")
} else {
format!("{s}{fill}")
}
};
let mut lines = Vec::with_capacity(df.height() + 2);
lines.push(format!(
"| {} |",
names
.iter()
.zip(&widths)
.zip(&numeric)
.map(|((n, &w), &num)| pad(n, w, num))
.collect::<Vec<_>>()
.join(" | ")
));
lines.push(format!(
"|{}|",
widths
.iter()
.zip(&numeric)
.map(|(&w, &num)| {
if num {
format!(" {}: ", "-".repeat(w.saturating_sub(1)))
} else {
format!(" {} ", "-".repeat(w))
}
})
.collect::<Vec<_>>()
.join("|")
));
for row in 0..df.height() {
lines.push(format!(
"| {} |",
cells
.iter()
.zip(&widths)
.zip(&numeric)
.map(|((body, &w), &num)| pad(&body[row], w, num))
.collect::<Vec<_>>()
.join(" | ")
));
}
Ok(lines.join("\n"))
}
fn tricky() -> DataFrame {
df!(
"name" => ["plain", "tab\there", "pipe|and\nnewline", "\"quoted\""],
"n" => [Some(1i64), Some(2), None, Some(4)],
)
.unwrap()
}
#[test]
fn tsv_quotes_only_what_a_paste_needs_quoted() {
let text = delimited(&tricky(), b'\t', true).unwrap();
let lines: Vec<&str> = text.split('\n').collect();
assert_eq!(lines[0], "name\tn");
assert_eq!(lines[1], "plain\t1");
assert_eq!(lines[2], "\"tab\there\"\t2");
assert!(text.contains("\"pipe|and\nnewline\"\t"));
assert!(!text.contains('∅'));
assert!(text.ends_with("\"\"\"quoted\"\"\"\t4"), "{text:?}");
}
#[test]
fn header_toggle_is_honored() {
let with = delimited(&tricky(), b',', true).unwrap();
let without = delimited(&tricky(), b',', false).unwrap();
assert!(with.starts_with("name,n"));
assert!(without.starts_with("plain,1"));
}
#[test]
fn markdown_escapes_aligns_and_keeps_nulls_empty() {
let text = markdown(&tricky()).unwrap();
let lines: Vec<&str> = text.split('\n').collect();
assert!(lines[0].starts_with("| name"));
assert!(lines[1].contains("-: |"), "{}", lines[1]);
assert!(lines[0].ends_with("| n |"), "{}", lines[0]);
assert!(lines[2].ends_with("| 1 |"), "{}", lines[2]);
assert!(text.contains("pipe\\|and newline"), "{text}");
let width = lines[0].chars().count();
assert!(lines.iter().all(|l| l.chars().count() == width), "{text}");
}
#[test]
fn html_flavor_escapes_and_rides_beside_tsv_only() {
let payload = tabular_payload(&tricky(), CopyFormat::Tsv, true, true).unwrap();
let html = payload.html.expect("tsv carries the html flavor");
assert!(html.starts_with("<table><thead>"));
assert!(html.contains("<td>\"quoted\"</td>"));
let md = tabular_payload(&tricky(), CopyFormat::Markdown, true, true).unwrap();
assert!(md.html.is_none(), "markdown is its own rich flavor");
let text_only = tabular_payload(&tricky(), CopyFormat::Csv, true, false).unwrap();
assert!(text_only.html.is_none(), "not built where it cannot go");
assert_eq!(text_only.text, delimited(&tricky(), b',', true).unwrap());
}
#[test]
fn every_format_copies_floats_exactly() {
let df = df!("x" => [1000000.125f64, -0.0]).unwrap();
for format in [CopyFormat::Tsv, CopyFormat::Csv, CopyFormat::Markdown] {
let payload = tabular_payload(&df, format, false, true).unwrap();
assert!(
payload.text.contains("1000000.125"),
"{format:?}: {payload:?}"
);
assert!(!payload.text.contains("e6"), "{format:?}: {payload:?}");
if let Some(html) = payload.html {
assert!(html.contains("<td>1000000.125</td>"), "{html}");
}
}
}
fn varied() -> DataFrame {
let mut df = df!(
"name" => ["plain", "tab\there", "pipe|and\nnewline", "\"quoted\"", "été", "日本語"],
"n" => [Some(1i64), Some(-22), None, Some(4), Some(1_000_000), Some(0)],
"x" => [Some(0.5f64), None, Some(-1.25), Some(3.0), Some(1e-9), Some(2.5)],
)
.unwrap();
let tags: ListChunked = (0..6)
.map(|i| (i % 2 == 0).then(|| Series::new("".into(), [format!("t{i}"), "a,b".into()])))
.collect();
df.with_column(tags.with_name("tags".into()).into_column())
.unwrap();
df
}
#[test]
fn markdown_writes_what_it_wrote_holding_every_cell() {
let empty = varied().head(Some(0));
for df in [tricky(), varied(), empty] {
let df = crate::nested_json::frame_as_json(&df).unwrap();
let text = markdown(&df).unwrap();
assert_eq!(text, markdown_reference(&df).unwrap());
let layout = {
let mut layout = MarkdownLayout::new(&df);
layout.measure(&df).unwrap();
layout
};
assert_eq!(layout.len(df.height()), text.chars().count());
}
}
#[test]
fn base64_is_sized_before_it_is_encoded() {
use base64::Engine as _;
for (bytes, encoded) in [
(0, 0),
(1, 4),
(2, 4),
(3, 4),
(4, 8),
(5, 8),
(6, 8),
(7, 12),
] {
assert_eq!(base64_len(bytes), encoded, "{bytes}");
let text = "a".repeat(bytes);
let real = base64::engine::general_purpose::STANDARD
.encode(&text)
.len();
assert_eq!(real, encoded);
}
assert!(osc52_sequence("ééé", 8).is_ok());
assert!(osc52_sequence("éééé", 8).is_err());
assert_eq!(
osc52_sequence("abcdef", 8).unwrap(),
"\x1b]52;c;YWJjZGVm\x07"
);
let err = osc52_sequence("abcdefg", 8).unwrap_err();
assert!(err.starts_with("the copy is 1 KB of base64"), "{err}");
}
#[test]
fn a_bounded_table_copy_writes_what_a_whole_one_would() {
for format in CopyFormat::ALL {
for header in [true, false] {
for df in [tricky(), varied(), varied().head(Some(0))] {
let whole = tabular_payload(&df, format, header, false).unwrap().text;
let (text, rows) =
bounded_table_text(df.clone().lazy(), format, header, 1 << 20).unwrap();
assert_eq!(text, whole, "{format:?} header {header}");
assert_eq!(rows, df.height());
}
}
}
}
#[test]
fn durations_copy_as_a_csv_export_writes_them() {
use crate::nested_json::tests::{duration_text, durations};
let df = durations();
let row = |i: usize, separator: &str| {
duration_text()
.iter()
.map(|(_, text)| text[i].unwrap_or(""))
.collect::<Vec<_>>()
.join(separator)
};
for format in CopyFormat::ALL {
let payload = tabular_payload(&df, format, true, true).unwrap();
let (bounded, rows) =
bounded_table_text(df.clone().lazy(), format, true, 1 << 20).unwrap();
assert_eq!(bounded, payload.text, "{format:?}");
assert_eq!(rows, df.height());
let lines: Vec<&str> = payload.text.lines().collect();
match format {
CopyFormat::Tsv | CopyFormat::Csv => {
let separator = if format == CopyFormat::Tsv { "\t" } else { "," };
assert_eq!(lines[0], ["ms", "us", "ns"].join(separator));
for (i, line) in lines[1..].iter().enumerate() {
assert_eq!(*line, row(i, separator), "{format:?} row {i}");
}
let html = payload.html.expect("html beside tsv and csv");
assert!(html.contains("<td>-PT1.5S</td>"), "{html}");
assert!(html.contains("<tr><td></td><td></td><td></td></tr>"));
}
CopyFormat::Markdown => {
assert_eq!(lines.len(), 2 + df.height());
for (i, line) in lines[2..].iter().enumerate() {
let cells: Vec<&str> =
line.trim_matches('|').split('|').map(str::trim).collect();
assert_eq!(cells.join(","), row(i, ","), "row {i}");
}
}
}
}
}
#[test]
fn dates_copy_as_each_format_wrote_them() {
use crate::nested_json::tests::calendar;
let df = calendar(false);
for format in CopyFormat::ALL {
let payload = tabular_payload(&df, format, true, true).unwrap();
let (bounded, _) =
bounded_table_text(df.clone().lazy(), format, true, 1 << 20).unwrap();
assert_eq!(bounded, payload.text, "{format:?}");
let separator = match format {
CopyFormat::Tsv => b'\t',
CopyFormat::Csv => b',',
CopyFormat::Markdown => {
assert_eq!(payload.text, markdown_reference(&df).unwrap());
continue;
}
};
let mut written = Vec::new();
CsvWriter::new(&mut written)
.with_separator(separator)
.finish(&mut df.clone())
.unwrap();
let written = String::from_utf8(written).unwrap();
assert_eq!(payload.text, written.trim_end_matches('\n'), "{format:?}");
assert_eq!(payload.html, Some(html_table(&df, true).unwrap()));
}
assert!(
markdown_reference(&df)
.unwrap()
.contains("| 1970-01-01 01:00:00.000000 +01:00 |")
);
let past = calendar(true);
let stored = "-9223372036854775807 us since 1970-01-01 UTC";
for format in CopyFormat::ALL {
let payload = tabular_payload(&past, format, true, true).unwrap();
let (bounded, _) =
bounded_table_text(past.clone().lazy(), format, true, 1 << 20).unwrap();
assert_eq!(bounded, payload.text, "{format:?}");
assert!(payload.text.contains(stored), "{format:?}");
if let Some(html) = payload.html {
assert!(html.contains(&format!("<td>{stored}</td>")), "{html}");
}
}
}
#[test]
fn a_bounded_table_copy_fails_in_the_words_a_whole_one_does() {
let lf = df!("s" => ["a"]).unwrap().lazy().select([col("missing")]);
let whole = crate::statistics::collect_lazy(lf.clone(), true).unwrap_err();
let err = bounded_table_text(lf, CopyFormat::Tsv, true, 1 << 20).unwrap_err();
assert_eq!(
err,
crate::error_display::user_message_from_polars(&whole),
"{whole}"
);
}
#[test]
fn a_bounded_table_copy_stops_at_the_first_batch_over_the_cap() {
let limit = 64 * 1024;
let rows = 100_000;
let df = df!("id" => (0..rows as i64).map(|i| i + 1_000_000_000).collect::<Vec<_>>(),
"k" => (0..rows as i64).map(|i| i % 10 + 10).collect::<Vec<_>>())
.unwrap();
for format in CopyFormat::ALL {
let mut text = BoundedText::new(format, true, limit);
let mut taken = 0;
for offset in (0..rows).step_by(BOUNDED_BATCH_ROWS) {
taken += 1;
if text.take(df.slice(offset as i64, BOUNDED_BATCH_ROWS)) {
break;
}
}
let fits = limit / 4 * 3 / 16;
assert!(
taken * BOUNDED_BATCH_ROWS <= fits + 2 * BOUNDED_BATCH_ROWS,
"{format:?}: {taken} batches"
);
assert!(text.text.len() <= limit / 4 * 3 + BOUNDED_BATCH_ROWS * 32);
let err = text.finish().unwrap_err();
assert!(err.starts_with("the copy is over 64 KB of base64"), "{err}");
let err = bounded_table_text(df.clone().lazy(), format, true, limit).unwrap_err();
assert!(err.contains("osc52_limit"), "{err}");
}
}
#[test]
fn a_capped_destination_takes_text_only() {
let osc = destination(BackendChoice::Osc52, 4096).unwrap();
assert_eq!(
osc.accepts(),
Accepts {
html: false,
base64_limit: Some(4096)
}
);
let auto = destination(BackendChoice::Auto, 4096).unwrap();
assert_eq!(auto.accepts().html, auto.describe() == "clipboard");
assert_eq!(
auto.accepts().base64_limit.is_some(),
auto.describe() == "terminal"
);
}
#[test]
fn html_escapes_markup_in_values() {
let df = df!("x" => ["<b>&"]).unwrap();
let html = html_table(&df, false).unwrap();
assert!(html.contains("<td><b>&</td>"), "{html}");
}
#[test]
fn osc52_wraps_base64_and_the_cap_names_the_config() {
let seq = osc52_sequence("hello", 1024).unwrap();
assert_eq!(seq, "\x1b]52;c;aGVsbG8=\x07");
let err = osc52_sequence("hello world, far too long", 8).unwrap_err();
assert!(err.contains("osc52_limit"), "{err}");
}
#[test]
fn backend_choice_parses_the_config_words() {
assert_eq!(BackendChoice::parse("auto"), Some(BackendChoice::Auto));
assert_eq!(BackendChoice::parse("Native"), Some(BackendChoice::Native));
assert_eq!(BackendChoice::parse("OSC52"), Some(BackendChoice::Osc52));
assert_eq!(BackendChoice::parse("wayland"), None);
}
}