use std::collections::HashMap;
use std::fs::File;
use std::io::{BufRead, BufReader};
use std::ops::Range;
use std::path::Path;
use anyhow::Result;
use polars::prelude::*;
pub const MAX_COLUMNS: usize = 1024;
pub const SHOWN_SHARE: f64 = 0.2;
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
pub enum Kind {
Null,
Bool,
Int,
Float,
Str,
Nested,
}
impl Kind {
const fn bit(self) -> u8 {
match self {
Kind::Null => 1,
Kind::Bool => 2,
Kind::Int => 4,
Kind::Float => 8,
Kind::Str => 16,
Kind::Nested => 32,
}
}
}
#[derive(Clone, Debug)]
pub struct Field {
pub name: String,
pub rows: usize,
kinds: u8,
last_row: usize,
}
impl Field {
pub fn dtype(&self) -> DataType {
let has = |kind: Kind| self.kinds & kind.bit() != 0;
if has(Kind::Nested) || has(Kind::Str) {
DataType::String
} else if has(Kind::Float) {
DataType::Float64
} else if has(Kind::Int) {
DataType::Int64
} else if has(Kind::Bool) {
DataType::Boolean
} else {
DataType::String
}
}
}
#[derive(Debug, Default)]
pub struct Fields {
fields: Vec<Field>,
pub rows: usize,
pub malformed: usize,
pub dropped: usize,
}
impl Fields {
pub fn columns(&self) -> Vec<&Field> {
let mut ordered: Vec<(usize, &Field)> = self.fields.iter().enumerate().collect();
ordered.sort_by(|(ai, a), (bi, b)| b.rows.cmp(&a.rows).then(ai.cmp(bi)));
ordered
.into_iter()
.map(|(_, field)| field)
.take(MAX_COLUMNS)
.collect()
}
pub fn schema(&self) -> SchemaRef {
let mut schema = Schema::default();
for field in self.columns() {
schema.with_column(field.name.as_str().into(), field.dtype());
}
Arc::new(schema)
}
pub fn paths(&self) -> Vec<KeyPath> {
self.columns()
.iter()
.map(|field| vec![field.name.clone()])
.collect()
}
pub fn notes(&self) -> Option<String> {
let mut said = Vec::new();
if let Some(shown) = self.shown() {
said.push(format!(
"{} of {} keys shown (:reset select for all)",
shown.len(),
self.columns().len()
));
}
if self.dropped > 0 {
said.push(format!(
"{} keys past the {MAX_COLUMNS} column cap",
self.dropped
));
}
if self.malformed > 0 {
said.push(format!(
"{} line{} not a JSON object",
self.malformed,
if self.malformed == 1 { "" } else { "s" }
));
}
(!said.is_empty()).then(|| said.join("; "))
}
pub fn shown(&self) -> Option<Vec<usize>> {
let columns = self.columns();
let floor = (self.rows as f64 * SHOWN_SHARE).ceil() as usize;
let shown: Vec<usize> = columns
.iter()
.enumerate()
.filter(|(_, field)| field.rows >= floor.max(1))
.map(|(at, _)| at)
.collect();
(shown.len() < columns.len() && !shown.is_empty()).then_some(shown)
}
}
#[derive(Default)]
pub struct Scan {
line: Vec<u8>,
at: HashMap<Vec<u8>, usize>,
found: Fields,
under: KeyPath,
carrying: usize,
}
impl Scan {
pub fn new() -> Self {
Self::default()
}
pub fn under(path: KeyPath) -> Self {
Self {
under: path,
..Self::default()
}
}
pub fn carrying(&self) -> usize {
self.carrying
}
#[inline]
pub fn byte(&mut self, byte: u8) {
if byte == b'\n' {
self.record();
} else {
self.line.push(byte);
}
}
pub fn finish(mut self) -> Fields {
if !self.line.is_empty() {
self.record();
}
self.found
}
fn record(&mut self) {
let row = self.found.rows;
self.found.rows += 1;
let line = std::mem::take(&mut self.line);
let mut fields = Vec::new();
if !fields_of(trim_cr(&line), &mut fields) {
self.found.malformed += 1;
self.line = line;
self.line.clear();
return;
}
if !self.under.is_empty() {
let inner = fields
.iter()
.find(|(name, _, _)| *name == self.under[0].as_bytes())
.and_then(|(_, kind, raw)| descend(*kind, raw, &self.under[1..]))
.filter(|(kind, _)| *kind == Kind::Nested)
.map(|(_, raw)| raw);
fields.clear();
match inner.filter(|raw| fields_of(raw, &mut fields)) {
Some(_) => self.carrying += 1,
None => {
self.line = line;
self.line.clear();
return;
}
}
}
for (name, kind, _) in fields {
match self.at.get(name) {
Some(&at) => {
let field = &mut self.found.fields[at];
field.kinds |= kind.bit();
if field.last_row != row {
field.last_row = row;
field.rows += 1;
}
}
None => {
if self.found.fields.len() >= MAX_COLUMNS {
self.found.dropped += 1;
continue;
}
self.at.insert(name.to_vec(), self.found.fields.len());
self.found.fields.push(Field {
name: String::from_utf8_lossy(name).into_owned(),
rows: 1,
kinds: kind.bit(),
last_row: row,
});
}
}
}
self.line = line;
self.line.clear();
}
}
pub const SAMPLE_RECORDS: usize = 20_480;
pub const SAMPLE_BYTES: u64 = 256 << 20;
pub struct Sample {
pub fields: Fields,
pub complete: bool,
}
pub fn sample_under(file: &Path, under: &KeyPath) -> Result<Sample> {
let mut scan = Scan::under(under.clone());
let mut reader = BufReader::with_capacity(1 << 20, File::open(file)?);
let mut read = 0u64;
let complete = loop {
let chunk = reader.fill_buf()?;
if chunk.is_empty() {
break true;
}
let taken = chunk.len();
for &byte in chunk {
scan.byte(byte);
}
read += taken as u64;
reader.consume(taken);
if scan.carrying() >= SAMPLE_RECORDS || read >= SAMPLE_BYTES {
break false;
}
};
Ok(Sample {
fields: scan.finish(),
complete,
})
}
pub fn locate(line: &[u8], path: &KeyPath) -> Option<Range<usize>> {
let mut fields = Vec::new();
if !fields_of(line, &mut fields) {
return None;
}
let key = path.first()?;
let (_, kind, raw) = fields.iter().find(|(name, _, _)| *name == key.as_bytes())?;
let (_, raw) = descend(*kind, raw, &path[1..])?;
Some(span_of(line, raw))
}
pub fn insert_point(line: &[u8]) -> Option<(usize, bool)> {
let mut fields = Vec::new();
if !fields_of(line, &mut fields) {
return None;
}
let close = line.iter().rposition(|&byte| byte == b'}')?;
Some((close, !fields.is_empty()))
}
fn span_of(outer: &[u8], inner: &[u8]) -> Range<usize> {
let at = inner.as_ptr() as usize - outer.as_ptr() as usize;
at..at + inner.len()
}
pub fn is_number(text: &str) -> bool {
matches!(number_end(text.as_bytes(), 0), Some((_, end)) if end == text.len())
}
pub fn page(
bytes: &[u8],
schema: &SchemaRef,
paths: &[KeyPath],
wanted: &[usize],
skip: usize,
take: usize,
) -> Result<DataFrame> {
let picked: Vec<(&str, &DataType, &KeyPath)> = wanted
.iter()
.filter_map(|&at| Some((schema.get_at_index(at)?, paths.get(at)?)))
.map(|((name, dtype), path)| (name.as_str(), dtype, path))
.collect();
let mut column_at: HashMap<&[u8], Vec<usize>> = HashMap::new();
for (at, (_, _, path)) in picked.iter().enumerate() {
if let Some(first) = path.first() {
column_at.entry(first.as_bytes()).or_default().push(at);
}
}
let mut lines: Vec<&[u8]> = bytes.split(|&byte| byte == b'\n').collect();
if bytes.last() == Some(&b'\n') {
lines.pop();
}
let lines = &lines[skip.min(lines.len())..];
let lines = &lines[..take.min(lines.len())];
let mut values: Vec<Vec<Option<String>>> = vec![Vec::with_capacity(lines.len()); picked.len()];
let mut fields = Vec::new();
for line in lines {
for column in values.iter_mut() {
column.push(None);
}
fields.clear();
if !fields_of(trim_cr(line), &mut fields) {
continue;
}
for (name, kind, raw) in &fields {
let Some(columns) = column_at.get(name) else {
continue;
};
for &at in columns {
let path = picked[at].2;
let Some((kind, raw)) = descend(*kind, raw, &path[1..]) else {
continue;
};
let last = values[at].len() - 1;
values[at][last] = text(kind, raw);
}
}
}
let columns: Vec<Column> = picked
.iter()
.zip(values)
.map(|((name, dtype, _), values)| column(name, dtype, values))
.collect();
let height = columns.first().map_or(0, |c| c.len());
Ok(DataFrame::new(height, columns)?)
}
pub type KeyPath = Vec<String>;
fn descend<'a>(kind: Kind, raw: &'a [u8], rest: &[String]) -> Option<(Kind, &'a [u8])> {
let Some(key) = rest.first() else {
return Some((kind, raw));
};
if kind != Kind::Nested {
return None;
}
let mut inner = Vec::new();
if !fields_of(raw, &mut inner) {
return None;
}
let (_, kind, raw) = inner.iter().find(|(name, _, _)| *name == key.as_bytes())?;
descend(*kind, raw, &rest[1..])
}
fn text(kind: Kind, raw: &[u8]) -> Option<String> {
match kind {
Kind::Null => None,
Kind::Str => Some(unescape(&String::from_utf8_lossy(
&raw[1..raw.len().saturating_sub(1).max(1)],
))),
_ => Some(String::from_utf8_lossy(raw).into_owned()),
}
}
fn column(name: &str, dtype: &DataType, values: Vec<Option<String>>) -> Column {
let name: PlSmallStr = name.into();
match dtype {
DataType::Int64 => Column::new(
name,
values
.iter()
.map(|v| v.as_ref().and_then(|v| v.parse::<i64>().ok()))
.collect::<Vec<_>>(),
),
DataType::Float64 => Column::new(
name,
values
.iter()
.map(|v| v.as_ref().and_then(|v| v.parse::<f64>().ok()))
.collect::<Vec<_>>(),
),
DataType::Boolean => Column::new(
name,
values
.iter()
.map(|v| match v.as_deref() {
Some("true") => Some(true),
Some("false") => Some(false),
_ => None,
})
.collect::<Vec<_>>(),
),
_ => Column::new(name, values),
}
}
fn trim_cr(line: &[u8]) -> &[u8] {
match line.last() {
Some(b'\r') => &line[..line.len() - 1],
_ => line,
}
}
fn fields_of<'a>(line: &'a [u8], out: &mut Vec<(&'a [u8], Kind, &'a [u8])>) -> bool {
let mut at = skip_space(line, 0);
if line.get(at) != Some(&b'{') {
return false;
}
at = skip_space(line, at + 1);
if line.get(at) == Some(&b'}') {
return skip_space(line, at + 1) == line.len();
}
loop {
at = skip_space(line, at);
let Some(key_end) = string_end(line, at) else {
return false;
};
let key = &line[at + 1..key_end - 1];
at = skip_space(line, key_end);
if line.get(at) != Some(&b':') {
return false;
}
at = skip_space(line, at + 1);
let Some((kind, end)) = value_end(line, at) else {
return false;
};
out.push((key, kind, &line[at..end]));
at = skip_space(line, end);
match line.get(at) {
Some(b',') => at += 1,
Some(b'}') => return skip_space(line, at + 1) == line.len(),
_ => return false,
}
}
}
fn skip_space(line: &[u8], mut at: usize) -> usize {
while matches!(line.get(at), Some(b' ' | b'\t' | b'\r')) {
at += 1;
}
at
}
fn string_end(line: &[u8], at: usize) -> Option<usize> {
if line.get(at) != Some(&b'"') {
return None;
}
let mut i = at + 1;
loop {
match line.get(i)? {
b'\\' => {
line.get(i + 1)?;
i += 2;
}
b'"' => return Some(i + 1),
byte if *byte < 0x20 => return None,
_ => i += 1,
}
}
}
fn value_end(line: &[u8], at: usize) -> Option<(Kind, usize)> {
match line.get(at)? {
b'"' => string_end(line, at).map(|end| (Kind::Str, end)),
b'{' | b'[' => nested_end(line, at).map(|end| (Kind::Nested, end)),
b't' => line[at..]
.starts_with(b"true")
.then_some((Kind::Bool, at + 4)),
b'f' => line[at..]
.starts_with(b"false")
.then_some((Kind::Bool, at + 5)),
b'n' => line[at..]
.starts_with(b"null")
.then_some((Kind::Null, at + 4)),
b'-' | b'0'..=b'9' => number_end(line, at),
_ => None,
}
}
fn nested_end(line: &[u8], at: usize) -> Option<usize> {
let mut depth = 0usize;
let mut i = at;
while let Some(byte) = line.get(i) {
match byte {
b'"' => {
i = string_end(line, i)?;
continue;
}
b'{' | b'[' => depth += 1,
b'}' | b']' => {
depth -= 1;
if depth == 0 {
return Some(i + 1);
}
}
_ => {}
}
i += 1;
}
None
}
fn number_end(line: &[u8], at: usize) -> Option<(Kind, usize)> {
let mut i = at;
let mut kind = Kind::Int;
if line.get(i) == Some(&b'-') {
i += 1;
}
let whole = i;
while matches!(line.get(i), Some(b'0'..=b'9')) {
i += 1;
}
if i == whole || (i - whole > 1 && line[whole] == b'0') {
return None;
}
if line.get(i) == Some(&b'.') {
i += 1;
let start = i;
while matches!(line.get(i), Some(b'0'..=b'9')) {
i += 1;
}
if i == start {
return None;
}
kind = Kind::Float;
}
if matches!(line.get(i), Some(b'e' | b'E')) {
i += 1;
if matches!(line.get(i), Some(b'+' | b'-')) {
i += 1;
}
let start = i;
while matches!(line.get(i), Some(b'0'..=b'9')) {
i += 1;
}
if i == start {
return None;
}
kind = Kind::Float;
}
Some((kind, i))
}
fn unescape(text: &str) -> String {
if !text.contains('\\') {
return text.to_string();
}
let mut out = String::with_capacity(text.len());
let mut chars = text.chars();
while let Some(c) = chars.next() {
if c != '\\' {
out.push(c);
continue;
}
match chars.next() {
Some('n') => out.push('\n'),
Some('t') => out.push('\t'),
Some('r') => out.push('\r'),
Some('b') => out.push('\u{8}'),
Some('f') => out.push('\u{c}'),
Some('u') => {
let hex: String = chars.by_ref().take(4).collect();
match u32::from_str_radix(&hex, 16).ok().and_then(char::from_u32) {
Some(c) => out.push(c),
None => {
out.push_str("\\u");
out.push_str(&hex);
}
}
}
Some(other) => out.push(other),
None => out.push('\\'),
}
}
out
}
#[cfg(test)]
mod tests {
use super::*;
fn scan(text: &str) -> Fields {
let mut scan = Scan::new();
for byte in text.bytes() {
scan.byte(byte);
}
scan.finish()
}
fn names(fields: &Fields) -> Vec<&str> {
fields
.columns()
.iter()
.map(|field| field.name.as_str())
.collect()
}
fn cell(df: &DataFrame, column: &str, row: usize) -> String {
df.column(column)
.unwrap()
.get(row)
.unwrap()
.str_value()
.to_string()
}
const LOG: &str = concat!(
r#"{"ts":"2026-09-07T10:00:00Z","level":"info","msg":"started","port":8080}"#,
"\n",
r#"{"ts":"2026-09-07T10:00:01Z","level":"error","msg":"boom","err":{"code":500,"why":"bad"}}"#,
"\n",
);
#[test]
fn the_keys_are_the_columns_most_common_first() {
let fields = scan(LOG);
assert_eq!(fields.rows, 2);
assert_eq!(names(&fields), ["ts", "level", "msg", "port", "err"]);
}
#[test]
fn a_column_is_typed_by_everything_that_was_ever_in_it() {
let fields = scan(concat!(
r#"{"n":1,"f":1,"s":1,"b":true,"j":{"a":1}}"#,
"\n",
r#"{"n":2,"f":1.5,"s":"x","b":false,"j":[1]}"#,
"\n"
));
let by_name: HashMap<&str, DataType> = fields
.columns()
.iter()
.map(|field| (field.name.as_str(), field.dtype()))
.collect();
assert_eq!(by_name["n"], DataType::Int64, "whole numbers throughout");
assert_eq!(by_name["f"], DataType::Float64, "one of them was not");
assert_eq!(by_name["s"], DataType::String, "a number and a string");
assert_eq!(by_name["b"], DataType::Boolean);
assert_eq!(
by_name["j"],
DataType::String,
"a document, kept as its own text"
);
}
#[test]
fn a_nested_value_is_carried_through_as_the_files_own_bytes() {
let fields = scan(LOG);
let schema = fields.schema();
let df = page(
LOG.as_bytes(),
&schema,
&fields.paths(),
&(0..schema.len()).collect::<Vec<_>>(),
0,
usize::MAX,
)
.unwrap();
assert_eq!(
cell(&df, "err", 1),
r#"{"code":500,"why":"bad"}"#,
"byte for byte, so `K` can open it"
);
assert!(
crate::ui::json::reindent(&cell(&df, "err", 1)).is_some(),
"and it really is a document"
);
}
#[test]
fn a_string_arrives_without_its_quotes_and_with_its_escapes_resolved() {
let line = "{\"msg\":\"one\\ntwo \\\"quoted\\\" \\u00e9\"}\n";
let fields = scan(line);
let schema = fields.schema();
let df = page(
line.as_bytes(),
&schema,
&fields.paths(),
&[0],
0,
usize::MAX,
)
.unwrap();
assert_eq!(cell(&df, "msg", 0), "one\ntwo \"quoted\" é");
}
#[test]
fn a_missing_key_is_a_null_and_not_a_shifted_row() {
let fields = scan(LOG);
let schema = fields.schema();
let df = page(
LOG.as_bytes(),
&schema,
&fields.paths(),
&(0..schema.len()).collect::<Vec<_>>(),
0,
usize::MAX,
)
.unwrap();
assert_eq!(df.height(), 2);
assert_eq!(cell(&df, "port", 0), "8080");
assert!(df.column("port").unwrap().get(1).unwrap().is_null());
}
#[test]
fn a_line_that_is_not_a_record_is_still_a_row() {
let text = concat!(r#"{"a":1}"#, "\n", "not json at all\n", r#"{"a":3}"#, "\n");
let fields = scan(text);
assert_eq!(fields.rows, 3);
assert_eq!(fields.malformed, 1);
let schema = fields.schema();
let df = page(
text.as_bytes(),
&schema,
&fields.paths(),
&[0],
0,
usize::MAX,
)
.unwrap();
assert_eq!(df.height(), 3);
assert_eq!(cell(&df, "a", 0), "1");
assert!(df.column("a").unwrap().get(1).unwrap().is_null());
assert_eq!(cell(&df, "a", 2), "3");
}
#[test]
fn a_brace_inside_a_string_does_not_end_the_record() {
let text = "{\"msg\":\"a } and a { in prose\",\"n\":1}\n";
let fields = scan(text);
assert_eq!(names(&fields), ["msg", "n"]);
}
#[test]
fn the_rare_keys_are_available_but_not_shown() {
let mut text = String::new();
for i in 0..9 {
text.push_str(&format!("{{\"a\":{i}}}\n"));
}
text.push_str("{\"a\":9,\"rare\":1}\n");
let fields = scan(&text);
assert_eq!(names(&fields), ["a", "rare"]);
assert_eq!(
fields.shown(),
Some(vec![0]),
"one in ten is below the share"
);
}
#[test]
fn nothing_is_hidden_when_every_key_is_common() {
let fields = scan(concat!(r#"{"a":1,"b":2}"#, "\n", r#"{"a":3,"b":4}"#, "\n"));
assert_eq!(fields.shown(), None, "so the view is left alone");
}
#[test]
fn a_key_repeated_in_one_record_counts_once() {
let fields = scan(concat!(r#"{"a":1,"a":2}"#, "\n", r#"{"b":1}"#, "\n"));
let counts: HashMap<&str, usize> = fields
.columns()
.iter()
.map(|field| (field.name.as_str(), field.rows))
.collect();
assert_eq!(counts["a"], 1);
}
#[test]
fn a_file_that_does_not_end_in_a_newline_still_ends_its_record() {
let fields = scan(r#"{"a":1}"#);
assert_eq!(fields.rows, 1);
assert_eq!(names(&fields), ["a"]);
}
#[test]
fn numbers_are_read_in_the_shape_json_allows() {
assert_eq!(number_end(b"01", 0), None, "no leading zero");
assert_eq!(number_end(b"1.", 0), None, "no bare point");
assert_eq!(number_end(b"1e3", 0), Some((Kind::Float, 3)));
assert_eq!(number_end(b"-0.5,", 0), Some((Kind::Float, 4)));
assert_eq!(number_end(b"42}", 0), Some((Kind::Int, 2)));
}
#[test]
fn a_page_builds_only_the_window_it_was_asked_for() {
let fields = scan(LOG);
let schema = fields.schema();
let df = page(LOG.as_bytes(), &schema, &fields.paths(), &[0], 1, 1).unwrap();
assert_eq!(df.height(), 1);
assert_eq!(cell(&df, "ts", 0), "2026-09-07T10:00:01Z");
}
#[test]
fn what_is_inside_a_column_is_found_by_looking_under_it() {
let path = std::env::temp_dir().join("plv-jsonl-sample.jsonl");
std::fs::write(&path, LOG).unwrap();
let sample = sample_under(&path, &vec!["err".to_string()]).unwrap();
assert!(sample.complete, "the whole file fits inside the bounds");
assert_eq!(
sample
.fields
.columns()
.iter()
.map(|field| field.name.as_str())
.collect::<Vec<_>>(),
["code", "why"]
);
assert_eq!(sample.fields.columns()[0].dtype(), DataType::Int64);
}
#[test]
fn a_column_can_be_read_from_inside_another() {
let fields = scan(LOG);
let mut schema = (*fields.schema()).clone();
schema.with_column("err.code".into(), DataType::Int64);
let schema = Arc::new(schema);
let mut paths = fields.paths();
paths.push(vec!["err".to_string(), "code".to_string()]);
let at = schema.len() - 1;
let df = page(LOG.as_bytes(), &schema, &paths, &[at], 0, usize::MAX).unwrap();
assert!(
df.column("err.code").unwrap().get(0).unwrap().is_null(),
"the record with no err has nothing there"
);
assert_eq!(cell(&df, "err.code", 1), "500");
}
#[test]
fn a_dotted_key_is_not_mistaken_for_a_path() {
let line = r#"{"a.b":1,"a":{"b":2}}"#.to_string() + "\n";
let fields = scan(&line);
let schema = fields.schema();
let paths = fields.paths();
let df = page(line.as_bytes(), &schema, &paths, &[0], 0, usize::MAX).unwrap();
assert_eq!(cell(&df, "a.b", 0), "1", "the key, not the route");
}
#[test]
fn a_page_only_builds_the_columns_it_was_asked_for() {
let fields = scan(LOG);
let schema = fields.schema();
let df = page(
LOG.as_bytes(),
&schema,
&fields.paths(),
&[1],
0,
usize::MAX,
)
.unwrap();
assert_eq!(df.get_column_names(), ["level"]);
assert_eq!(cell(&df, "level", 1), "error");
}
}