1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
# CSV → JSONL with learned column profiles + drift detection (#708).
#
# Every run profiles the records it writes — per column: null rate, type mix,
# distinct estimate, numeric / string summaries, top values — into a bounded
# sketch, compares it with a rolling baseline of earlier runs kept in the
# `state:` store, and reports a statistically significant change per column.
# No thresholds to write: after `min_history` runs the baseline is the norm.
#
# faucet run cli/examples/csv_to_jsonl_with_profiling.yaml # repeat a few times
# faucet profiling show cli/examples/csv_to_jsonl_with_profiling.yaml
# faucet profiling reset cli/examples/csv_to_jsonl_with_profiling.yaml --column amount
# faucet doctor cli/examples/csv_to_jsonl_with_profiling.yaml # baseline depth
#
# Needs a `state:` block (the baseline lives there). Profiling runs after
# transforms and masking, so a masked column is profiled masked.
version: 1
name: csv_to_jsonl_with_profiling
pipeline:
source:
type: file
config:
path: ./data/input.csv
sink:
type: file
config:
path: ./out/records.jsonl
state:
type: file
config:
path: ./state
transforms:
# CSV values are strings; profile `amount` as a number (an empty cell
# becomes null, which the null-rate metric then tracks).
- type: cast
config:
fields: { amount: float }
on_error: "null"
profiling:
# Columns to profile (default: every column) and `prefix*` globs to skip
# (default: the `_faucet_*` metadata columns).
# columns: [amount, region]
exclude: ["_faucet_*"]
# Runs of baseline before drift detection starts, and the rolling window.
min_history: 5
window: 20
# Numeric-metric test: zscore (sensitivity 3.0) | iqr (fence 1.5).
method: zscore
# A previously unseen value or type is reported at this share of the run.
new_value_min_share: 0.05
# Population-stability-index threshold for a shifted value distribution.
psi_threshold: 0.2
# Columns with more distinct values than this publish no example values.
categorical_max_distinct: 100
# warn (log + metric) | notify (+ a `profile_drift` notification) | fail
# (+ the run is reported failed; the data is already written).
on_drift: warn