Skip to main content

mcd_core/
tables.rs

1//! CSV table loading and typed value coercion.
2
3use indexmap::{IndexMap, IndexSet};
4use rust_decimal::Decimal;
5use serde::{Deserialize, Serialize};
6use time::{
7    Date, OffsetDateTime, PrimitiveDateTime, Time, format_description::well_known::Rfc3339,
8};
9
10use crate::{
11    errors::{Diagnostic, McdError, Result},
12    manifest::{Manifest, TableManifestEntry},
13    package::McdPackage,
14    schema::{ColumnType, TableColumnSchema, TableSchema},
15};
16
17/// Load all manifest-declared tables in manifest order.
18pub fn load_manifest_tables(
19    package: &McdPackage,
20    manifest: &Manifest,
21) -> Result<IndexMap<String, DataTable>> {
22    let mut tables = IndexMap::new();
23    for entry in &manifest.tables {
24        let table = DataTable::from_manifest_entry(package, entry)?;
25        tables.insert(entry.id.clone(), table);
26    }
27    Ok(tables)
28}
29
30/// A loaded CSV-backed table.
31#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
32#[serde(rename_all = "camelCase")]
33pub struct DataTable {
34    /// Stable table id.
35    pub id: String,
36    /// Package CSV path.
37    pub source: String,
38    /// Parsed schema.
39    pub schema: TableSchema,
40    /// Ordered typed rows.
41    pub rows: Vec<TableRow>,
42}
43
44impl DataTable {
45    /// Load and validate a manifest-declared table.
46    pub fn from_manifest_entry(package: &McdPackage, entry: &TableManifestEntry) -> Result<Self> {
47        if !package.contains(&entry.data) {
48            return Err(McdError::from_diagnostic(
49                Diagnostic::error(
50                    "table.data.missing",
51                    format!("Declared table data file '{}' is missing.", entry.data),
52                )
53                .with_source(entry.data.clone()),
54            ));
55        }
56
57        let schema = TableSchema::from_package(package, &entry.schema)?;
58        if schema.id != entry.id {
59            return Err(McdError::from_diagnostic(
60                Diagnostic::error(
61                    "schema.table.mismatch",
62                    format!(
63                        "Table schema id '{}' does not match manifest table id '{}'.",
64                        schema.id, entry.id
65                    ),
66                )
67                .with_source(entry.schema.clone()),
68            ));
69        }
70
71        let bytes = package.read(&entry.data)?;
72        let rows = load_csv_rows(&entry.id, &entry.data, &entry.schema, bytes, &schema)?;
73
74        Ok(Self {
75            id: entry.id.clone(),
76            source: entry.data.clone(),
77            schema,
78            rows,
79        })
80    }
81}
82
83/// A typed table row, keyed by schema column name.
84#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
85pub struct TableRow {
86    /// Typed cells.
87    #[serde(flatten)]
88    pub cells: IndexMap<String, TypedValue>,
89}
90
91/// A typed table cell value.
92#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
93#[serde(tag = "type", content = "value", rename_all = "snake_case")]
94pub enum TypedValue {
95    /// Null value from an empty nullable cell.
96    Null,
97    /// String value.
98    String(String),
99    /// Integer value.
100    Integer(i64),
101    /// Decimal value, serialized as a decimal string for stability.
102    Decimal(String),
103    /// Boolean value.
104    Boolean(bool),
105    /// Date value.
106    Date(String),
107    /// Datetime value.
108    Datetime(String),
109    /// Time value.
110    Time(String),
111    /// Enum member.
112    Enum(String),
113}
114
115fn load_csv_rows(
116    table_id: &str,
117    source: &str,
118    schema_source: &str,
119    bytes: &[u8],
120    schema: &TableSchema,
121) -> Result<Vec<TableRow>> {
122    let mut reader = csv::ReaderBuilder::new()
123        .has_headers(true)
124        .flexible(false)
125        .from_reader(bytes);
126
127    let headers = reader.headers().map_err(|err| {
128        McdError::from_diagnostic(
129            Diagnostic::error(
130                "csv.header.missing",
131                format!("CSV table '{table_id}' must include a header row: {err}."),
132            )
133            .with_source(format!("{source}:1")),
134        )
135    })?;
136    if headers.is_empty() {
137        return Err(McdError::from_diagnostic(
138            Diagnostic::error(
139                "csv.header.missing",
140                format!("CSV table '{table_id}' must include a header row."),
141            )
142            .with_source(format!("{source}:1")),
143        ));
144    }
145
146    let actual_headers = headers.iter().map(ToOwned::to_owned).collect::<Vec<_>>();
147    let expected_headers = schema
148        .columns
149        .iter()
150        .map(|column| column.name.clone())
151        .collect::<Vec<_>>();
152    if actual_headers != expected_headers {
153        return Err(McdError::from_diagnostic(
154            Diagnostic::error(
155                "csv.header.mismatch",
156                format!(
157                    "CSV header does not match table schema for table '{}'. Expected [{}], got [{}].",
158                    table_id,
159                    expected_headers.join(", "),
160                    actual_headers.join(", ")
161                ),
162            )
163            .with_source(format!("{source}:1"))
164            .with_related(schema_source.to_owned()),
165        ));
166    }
167
168    let mut rows = Vec::new();
169    for (record_index, record) in reader.records().enumerate() {
170        let record = record.map_err(|err| {
171            McdError::from_diagnostic(
172                Diagnostic::error("csv.row.invalid", format!("Invalid CSV row: {err}."))
173                    .with_source(format!("{source}:{}", record_index + 2)),
174            )
175        })?;
176
177        let mut cells = IndexMap::new();
178        for (column_index, column) in schema.columns.iter().enumerate() {
179            let raw = record.get(column_index).unwrap_or_default();
180            let value = coerce_cell(raw, column, source, record_index + 2)?;
181            cells.insert(column.name.clone(), value);
182        }
183        rows.push(TableRow { cells });
184    }
185
186    validate_primary_key_rows(source, schema, &rows)?;
187
188    Ok(rows)
189}
190
191fn validate_primary_key_rows(source: &str, schema: &TableSchema, rows: &[TableRow]) -> Result<()> {
192    if schema.primary_key.is_empty() {
193        return Ok(());
194    }
195
196    let mut keys = IndexSet::new();
197    for (row_index, row) in rows.iter().enumerate() {
198        let row_number = row_index + 2;
199        let key = row_key(row, &schema.primary_key).ok_or_else(|| {
200            McdError::from_diagnostic(
201                Diagnostic::error(
202                    "csv.primary_key.null",
203                    "Primary key columns cannot contain null values.",
204                )
205                .with_source(format!("{source}:{row_number}")),
206            )
207        })?;
208        if !keys.insert(key) {
209            return Err(McdError::from_diagnostic(
210                Diagnostic::error(
211                    "csv.primary_key.duplicate",
212                    format!("Duplicate primary key value in table '{}'.", schema.id),
213                )
214                .with_source(format!("{source}:{row_number}")),
215            ));
216        }
217    }
218
219    Ok(())
220}
221
222/// Return a stable typed key for a row, or `None` if any key part is null.
223#[must_use]
224pub fn row_key(row: &TableRow, columns: &[String]) -> Option<Vec<String>> {
225    columns
226        .iter()
227        .map(|column| row.cells.get(column).and_then(typed_key_part))
228        .collect()
229}
230
231/// Return a stable typed key part for relational comparisons.
232#[must_use]
233pub fn typed_key_part(value: &TypedValue) -> Option<String> {
234    match value {
235        TypedValue::Null => None,
236        TypedValue::String(value) => Some(format!("string:{value}")),
237        TypedValue::Integer(value) => Some(format!("integer:{value}")),
238        TypedValue::Decimal(value) => Some(format!("decimal:{value}")),
239        TypedValue::Boolean(value) => Some(format!("boolean:{value}")),
240        TypedValue::Date(value) => Some(format!("date:{value}")),
241        TypedValue::Datetime(value) => Some(format!("datetime:{value}")),
242        TypedValue::Time(value) => Some(format!("time:{value}")),
243        TypedValue::Enum(value) => Some(format!("enum:{value}")),
244    }
245}
246
247/// Coerce one CSV cell into its schema type.
248pub fn coerce_cell(
249    raw: &str,
250    column: &TableColumnSchema,
251    source: &str,
252    row_number: usize,
253) -> Result<TypedValue> {
254    let value = raw.trim();
255    if value.is_empty() {
256        if column.nullable {
257            return Ok(TypedValue::Null);
258        }
259        return Err(McdError::from_diagnostic(
260            Diagnostic::error(
261                "csv.cell.empty.nonnullable",
262                format!("Column '{}' does not allow empty cells.", column.name),
263            )
264            .with_source(format!("{source}:{row_number}")),
265        ));
266    }
267
268    match column.value_type {
269        ColumnType::String => Ok(TypedValue::String(raw.to_owned())),
270        ColumnType::Integer => value
271            .parse::<i64>()
272            .map(TypedValue::Integer)
273            .map_err(|_| cell_type_error("csv.cell.integer.invalid", column, source, row_number)),
274        ColumnType::Decimal => value
275            .parse::<Decimal>()
276            .map(|decimal| TypedValue::Decimal(decimal.normalize().to_string()))
277            .map_err(|_| cell_type_error("csv.cell.decimal.invalid", column, source, row_number)),
278        ColumnType::Boolean => parse_bool(value)
279            .map(TypedValue::Boolean)
280            .ok_or_else(|| cell_type_error("csv.cell.boolean.invalid", column, source, row_number)),
281        ColumnType::Date => Date::parse(
282            value,
283            &time::macros::format_description!("[year]-[month]-[day]"),
284        )
285        .map(|date| TypedValue::Date(date.to_string()))
286        .map_err(|_| cell_type_error("csv.cell.date.invalid", column, source, row_number)),
287        ColumnType::Datetime => parse_datetime(value)
288            .map(TypedValue::Datetime)
289            .ok_or_else(|| {
290                cell_type_error("csv.cell.datetime.invalid", column, source, row_number)
291            }),
292        ColumnType::Time => Time::parse(
293            value,
294            &time::macros::format_description!("[hour]:[minute]:[second]"),
295        )
296        .or_else(|_| Time::parse(value, &time::macros::format_description!("[hour]:[minute]")))
297        .map(|time| time.to_string())
298        .map(TypedValue::Time)
299        .map_err(|_| cell_type_error("csv.cell.time.invalid", column, source, row_number)),
300        ColumnType::Enum => {
301            if column.enum_values.iter().any(|allowed| allowed == value) {
302                Ok(TypedValue::Enum(value.to_owned()))
303            } else {
304                Err(McdError::from_diagnostic(
305                    Diagnostic::error(
306                        "csv.cell.enum.invalid",
307                        format!(
308                            "Value '{}' is not a member of enum column '{}'.",
309                            value, column.name
310                        ),
311                    )
312                    .with_source(format!("{source}:{row_number}")),
313                ))
314            }
315        }
316    }
317}
318
319fn parse_bool(value: &str) -> Option<bool> {
320    match value {
321        "true" | "TRUE" | "True" => Some(true),
322        "false" | "FALSE" | "False" => Some(false),
323        _ => None,
324    }
325}
326
327fn parse_datetime(value: &str) -> Option<String> {
328    OffsetDateTime::parse(value, &Rfc3339)
329        .map(|datetime| datetime.to_string())
330        .or_else(|_| {
331            PrimitiveDateTime::parse(
332                value,
333                &time::macros::format_description!("[year]-[month]-[day]T[hour]:[minute]:[second]"),
334            )
335            .map(|datetime| datetime.to_string())
336        })
337        .ok()
338}
339
340fn cell_type_error(
341    code: &'static str,
342    column: &TableColumnSchema,
343    source: &str,
344    row_number: usize,
345) -> McdError {
346    McdError::from_diagnostic(
347        Diagnostic::error(
348            code,
349            format!(
350                "Cell in column '{}' is not a valid {:?}.",
351                column.name, column.value_type
352            ),
353        )
354        .with_source(format!("{source}:{row_number}")),
355    )
356}
357
358#[cfg(test)]
359mod tests {
360    use super::*;
361    use proptest::prelude::*;
362
363    fn column(value_type: ColumnType, nullable: bool) -> TableColumnSchema {
364        TableColumnSchema {
365            name: "value".to_owned(),
366            value_type,
367            label: None,
368            unit: None,
369            nullable,
370            enum_values: Vec::new(),
371        }
372    }
373
374    fn table_schema(primary_key: Vec<String>) -> TableSchema {
375        TableSchema {
376            id: "revenue".to_owned(),
377            primary_key,
378            foreign_keys: Vec::new(),
379            columns: vec![
380                TableColumnSchema {
381                    name: "quarter".to_owned(),
382                    value_type: ColumnType::String,
383                    label: None,
384                    unit: None,
385                    nullable: false,
386                    enum_values: Vec::new(),
387                },
388                TableColumnSchema {
389                    name: "amount".to_owned(),
390                    value_type: ColumnType::Decimal,
391                    label: None,
392                    unit: None,
393                    nullable: false,
394                    enum_values: Vec::new(),
395                },
396            ],
397        }
398    }
399
400    #[test]
401    fn coerces_decimal() {
402        let value = coerce_cell("12.3400", &column(ColumnType::Decimal, false), "t.csv", 2)
403            .expect("decimal");
404        assert_eq!(value, TypedValue::Decimal("12.34".to_owned()));
405    }
406
407    #[test]
408    fn rejects_duplicate_primary_key_rows() {
409        let schema = table_schema(vec!["quarter".to_owned()]);
410        let err = load_csv_rows(
411            "revenue",
412            "tables/revenue.csv",
413            "tables/revenue.schema.json",
414            b"quarter,amount\nQ1,1\nQ1,2\n",
415            &schema,
416        )
417        .expect_err("duplicate key should fail");
418
419        assert_eq!(
420            err.diagnostic().map(|d| d.code.as_str()),
421            Some("csv.primary_key.duplicate")
422        );
423    }
424
425    #[test]
426    fn rejects_empty_nonnullable_cell() {
427        let err =
428            coerce_cell("", &column(ColumnType::String, false), "t.csv", 2).expect_err("invalid");
429        assert_eq!(
430            err.diagnostic().map(|d| d.code.as_str()),
431            Some("csv.cell.empty.nonnullable")
432        );
433    }
434
435    #[test]
436    fn rejects_bad_date() {
437        let err = coerce_cell("2026-99-99", &column(ColumnType::Date, false), "t.csv", 2)
438            .expect_err("invalid");
439        assert_eq!(
440            err.diagnostic().map(|d| d.code.as_str()),
441            Some("csv.cell.date.invalid")
442        );
443    }
444
445    #[test]
446    fn rejects_bad_decimal_datetime_time_boolean_and_enum() {
447        let cases = [
448            (
449                "not-money",
450                TableColumnSchema {
451                    name: "value".to_owned(),
452                    value_type: ColumnType::Decimal,
453                    label: None,
454                    unit: None,
455                    nullable: false,
456                    enum_values: Vec::new(),
457                },
458                "csv.cell.decimal.invalid",
459            ),
460            (
461                "2026-05-18 12:00:00",
462                TableColumnSchema {
463                    name: "value".to_owned(),
464                    value_type: ColumnType::Datetime,
465                    label: None,
466                    unit: None,
467                    nullable: false,
468                    enum_values: Vec::new(),
469                },
470                "csv.cell.datetime.invalid",
471            ),
472            (
473                "25:00",
474                TableColumnSchema {
475                    name: "value".to_owned(),
476                    value_type: ColumnType::Time,
477                    label: None,
478                    unit: None,
479                    nullable: false,
480                    enum_values: Vec::new(),
481                },
482                "csv.cell.time.invalid",
483            ),
484            (
485                "yes",
486                TableColumnSchema {
487                    name: "value".to_owned(),
488                    value_type: ColumnType::Boolean,
489                    label: None,
490                    unit: None,
491                    nullable: false,
492                    enum_values: Vec::new(),
493                },
494                "csv.cell.boolean.invalid",
495            ),
496            (
497                "bronze",
498                TableColumnSchema {
499                    name: "value".to_owned(),
500                    value_type: ColumnType::Enum,
501                    label: None,
502                    unit: None,
503                    nullable: false,
504                    enum_values: vec!["silver".to_owned(), "gold".to_owned()],
505                },
506                "csv.cell.enum.invalid",
507            ),
508        ];
509
510        for (raw, column, code) in cases {
511            let err = coerce_cell(raw, &column, "t.csv", 2).expect_err("invalid");
512            assert_eq!(err.diagnostic().map(|d| d.code.as_str()), Some(code));
513        }
514    }
515
516    proptest! {
517        #[test]
518        fn coerces_generated_integers(raw in any::<i64>()) {
519            let value = coerce_cell(
520                &raw.to_string(),
521                &column(ColumnType::Integer, false),
522                "t.csv",
523                2,
524            )
525            .expect("integer");
526            prop_assert_eq!(value, TypedValue::Integer(raw));
527        }
528    }
529}