duckfn 0.0.18

Write DuckDB extensions in plain Rust: attribute macros that turn ordinary functions into scalar/aggregate/table functions, SQL macros and nested LIST/MAP/ARRAY/STRUCT types.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
//! `#[duck_*]` 的文档参数(`description` / `comment` / `example`)与 CLI 的 CSV 导出。
//!
//! 覆盖刻意挑出来的易错点:
//!
//! - **A 普通函数**:三列齐全时逐字落在 CSV 里。
//! - **B 没有 metadata**:默认导出跳过它;`--all` 才写出来(三列留空)。
//! - **C CSV 特殊字符**:逗号、双引号、前后空格、非 ASCII、Jekyll 的 `{{ }}` / `{% raw %}`;
//!   源文本里的换行会在写出时压成空格(Markdown 表格与 `read_csv()` 都不接受裸换行)。
//! - **D overload**:多个签名挂到同一个函数集时只出一行,名字是函数集名而不是 Rust 函数名,
//!   元数据按「description/comment 取第一个非空、examples 拼接去重」合并。
//! - **E 多个 example**:CSV 的 `example` 是**一个字符串**,
//!   `generate_md.sh` 只做 `'[' || example.jekyll_format_function() || ']'`,
//!   **不会**按分隔符再拆一次(对照它原生分支的 `list_reduce(lambda x, y : x || ', ' || y)`),
//!   所以多条必须先在写出时拼好。拼的规则见 `join_examples`:每条去掉结尾分号后按 `"; "` 连接 ——
//!   真实 SQL 里逗号遍地都是(`FROM (VALUES (1, 'a'), (2, NULL)) v(i, s)`),用逗号当分隔符根本读不出
//!   一条示例在哪结束。下面用的示例都是这种带逗号的真 SQL,不是 `SELECT a, SELECT b`。
//!
//! 另外用一个可选的 DuckDB 往返测试把「这份 CSV 真的能被 `read_csv()` 读回原值」钉死 ——
//! 转义写错了只有真读一遍才知道。没装 `duckdb` 命令行时该测试打印提示后跳过。
//!
//! Documentation arguments of the `#[duck_*]` macros and the CLI's CSV export. The cases are
//! chosen for the things that actually go wrong: a fully documented function, one with no metadata,
//! CSV special characters, overloads collapsing into one row, and multiple examples. A separate
//! round-trip test (skipped when the `duckdb` CLI is not installed) proves the file still reads back
//! through `read_csv()`.
//!
//! CSV 那部分要 duckfn 的 `cli` feature:
//!
//! ```text
//! cargo test -p duckfn --features cli
//! ```

use duckfn::declared_function_descriptions;

// ============================================================================
// A 普通函数:三列齐全
// ============================================================================

/// 单条示例 + 注释:三列都有值。
///
/// A single example plus a comment: all three columns carry a value.
#[duckfn::duck_scalar_function(
    description = "Doubles an INTEGER",
    comment = "NULL in, NULL out",
    example = "SELECT docs_double_it(21)"
)]
fn docs_double_it(v: Option<i64>) -> Option<i64> {
    v.map(|x| x * 2)
}

// ============================================================================
// E 多个 example
// ============================================================================

/// 多条示例:`examples = [...]` 的顺序会被保留,导出时按 `"; "` 拼成一个字段。
///
/// 示例刻意写成常见的真 SQL(内部全是逗号),并且**第二条以分号结尾** —— 收尾分号在导出时会被去掉,
/// 免得拼出 `...; ;` 这种一眼像 bug 的东西。
///
/// Several examples: the order of `examples = [...]` is preserved and they are joined with `"; "`
/// into the single CSV field. The examples are ordinary SQL full of commas, and the second one *does*
/// end with a semicolon: that trailing semicolon is dropped on export so the result cannot end up
/// looking like `...; ;`.
#[duckfn::duck_scalar_function(
    description = "Adds two INTEGERs",
    examples = [
        "SELECT docs_add_two(i, 1) FROM (VALUES (1), (2), (3)) v(i) ORDER BY i",
        "SELECT docs_add_two(i, j) FROM (VALUES (1, 2), (3, 4)) v(i, j);"
    ]
)]
fn docs_add_two(a: i64, b: i64) -> i64 {
    a + b
}

// ============================================================================
// B 没有 metadata
// ============================================================================

/// 没写文档参数的函数同样会被收集(三个字段为空),但 `is_documented()` 为假、默认导出不写它。
///
/// A function with no documentation argument is collected too (all three fields empty), but
/// `is_documented()` is false and the default export leaves it out.
#[duckfn::duck_scalar_function]
fn docs_undocumented(v: i64) -> i64 {
    v
}

// ============================================================================
// C CSV 特殊字符
// ============================================================================

/// 描述里同时有逗号、双引号与真正的换行;注释两侧带空格;示例里再来一次逗号与双引号。
///
/// The description carries a comma, a double quote and a real newline; the comment is padded with
/// spaces; the example repeats the comma and the quote.
#[duckfn::duck_scalar_function(
    description = "Comma, \"quote\", and\na newline",
    comment = "  padded  ",
    example = "SELECT docs_special('a,b') -- say \"hi\" {{ }} {% raw %}"
)]
fn docs_special() -> i64 {
    0
}

/// 非 ASCII:CSV 是 UTF-8,读回来必须一字不差。
///
/// Non-ASCII: the CSV is UTF-8 and must read back byte for byte.
#[duckfn::duck_scalar_function(
    description = "把 INTEGER 翻倍",
    example = "SELECT docs_unicode()"
)]
fn docs_unicode() -> i64 {
    0
}

// ============================================================================
// D overload:两个签名挂到同一个函数集
// ============================================================================

/// 函数集 `docs_overloaded` 的 INTEGER 签名:提供 description。
///
/// The INTEGER signature of the `docs_overloaded` set: supplies the description.
#[duckfn::duck_scalar_function(
    overloads_name = "docs_overloaded",
    description = "Overloaded: INTEGER input",
    example = "SELECT docs_overloaded(i) FROM (VALUES (1), (2)) v(i)"
)]
fn docs_overload_int(v: i64) -> i64 {
    v
}

/// 同一个函数集的 VARCHAR 签名:提供 comment,并且和上面共享第一条 example(导出时必须去重),
/// 第二条示例以分号结尾(导出时被去掉)。
///
/// The VARCHAR signature of the same set: supplies the comment, shares the first example with the
/// other signature (which the export must deduplicate), and ends its second example with a
/// semicolon (dropped on export).
#[duckfn::duck_scalar_function(
    overloads_name = "docs_overloaded",
    comment = "Also accepts VARCHAR",
    examples = [
        "SELECT docs_overloaded(i) FROM (VALUES (1), (2)) v(i)",
        "SELECT docs_overloaded(s) FROM (VALUES ('a'), ('b')) v(s);"
    ]
)]
fn docs_overload_str(v: String) -> String {
    v
}

// ============================================================================
// 默认导出(只写有文档的):整份文件逐字节比对
// ============================================================================

/// 默认导出的完整内容,按函数名排序:
///
/// - 只在含 `,` 或 `"` 时才加引号(`csv` crate 的最小转义),字段里的 `"` 双写;
///   所以 `Adds two INTEGERs` 是裸的,而 `NULL in, NULL out` 带引号;
/// - `docs_overloaded` 只占一行,`function` 是函数集名而不是两个 Rust 函数名;
/// - `docs_special` 描述里的换行被压成空格(见 `flatten_newlines`),因此没有任何字段跨物理行;
/// - 没写文档的 `docs_undocumented` 不出现。
///
/// The full content of the default export, sorted by function name: quoting only where a field
/// contains `,` or `"`, with inner quotes doubled (so `Adds two INTEGERs` stays bare and
/// `NULL in, NULL out` is quoted); one single row for the overload set; the newline inside
/// `docs_special`'s description flattened to a space (see `flatten_newlines`), so no field spans a
/// physical line; and no `docs_undocumented`.
#[cfg(feature = "cli")]
const EXPECTED_DEFAULT_CSV: &str = concat!(
    "function,description,comment,example\n",
    "docs_add_two,Adds two INTEGERs,,\"SELECT docs_add_two(i, 1) FROM (VALUES (1), (2), (3)) v(i) ORDER BY i; SELECT docs_add_two(i, j) FROM (VALUES (1, 2), (3, 4)) v(i, j)\"\n",
    "docs_double_it,Doubles an INTEGER,\"NULL in, NULL out\",SELECT docs_double_it(21)\n",
    "docs_overloaded,Overloaded: INTEGER input,Also accepts VARCHAR,\"SELECT docs_overloaded(i) FROM (VALUES (1), (2)) v(i); SELECT docs_overloaded(s) FROM (VALUES ('a'), ('b')) v(s)\"\n",
    "docs_special,\"Comma, \"\"quote\"\", and a newline\",  padded  ,\"SELECT docs_special('a,b') -- say \"\"hi\"\" {{ }} {% raw %}\"\n",
    "docs_unicode,把 INTEGER 翻倍,,SELECT docs_unicode()\n",
);

/// 收集到的文档按函数名排序,字段与各条属性一一对应。
///
/// The collected entries are sorted by function name and every field matches the attribute that
/// produced it.
#[test]
fn declared_documentation_is_collected() {
    let rows = declared_function_descriptions();
    // 只断言本文件声明的函数,其它测试文件提交的条目不关这里的事。
    //
    // Only the functions declared here are asserted on; submissions from other test files are
    // none of this test's business.
    let find = |name: &str| {
        rows.iter()
            .find(|row| row.function == name)
            .unwrap_or_else(|| panic!("`{name}` is missing from the collected documentation"))
    };

    // A:三列齐全。
    let double_it = find("docs_double_it");
    assert_eq!(double_it.description.as_deref(), Some("Doubles an INTEGER"));
    assert_eq!(double_it.comment.as_deref(), Some("NULL in, NULL out"));
    assert_eq!(
        double_it.examples,
        vec!["SELECT docs_double_it(21)".to_string()]
    );
    assert!(double_it.is_documented());

    // E:多条示例按声明顺序保留(原样保留,连结尾的分号都还在 —— 归一化只发生在写 CSV 时);
    // 没写 comment 就是 None。
    //
    // E: several examples keep their declared order and their literal text — the trailing semicolon
    // included. Normalisation happens only when the CSV is written.
    let add_two = find("docs_add_two");
    assert_eq!(add_two.description.as_deref(), Some("Adds two INTEGERs"));
    assert_eq!(add_two.comment, None);
    assert_eq!(
        add_two.examples,
        vec![
            "SELECT docs_add_two(i, 1) FROM (VALUES (1), (2), (3)) v(i) ORDER BY i".to_string(),
            "SELECT docs_add_two(i, j) FROM (VALUES (1, 2), (3, 4)) v(i, j);".to_string()
        ]
    );

    // B:收集到了,但没内容。
    let undocumented = find("docs_undocumented");
    assert!(!undocumented.is_documented());
    assert_eq!(undocumented.description, None);
    assert_eq!(undocumented.comment, None);
    assert!(undocumented.examples.is_empty());

    // C:收集到的还是原样的字符 —— 逗号、引号、真实换行、前后空格、非 ASCII 都还在;
    // 换行只在写 CSV 时才被压平(见 `export_writes_documented_rows_only`)。
    //
    // C: what is collected is still the literal text — comma, quote, a real newline, padding and
    // non-ASCII all intact. The newline is only flattened when the CSV is written.
    let special = find("docs_special");
    assert_eq!(
        special.description.as_deref(),
        Some("Comma, \"quote\", and\na newline")
    );
    assert_eq!(special.comment.as_deref(), Some("  padded  "));
    assert_eq!(find("docs_unicode").description.as_deref(), Some("把 INTEGER 翻倍"));

    // D:两个签名合并成一条,名字是函数集名;Rust 函数名不出现在结果里。
    assert!(
        rows.iter().all(|row| row.function != "docs_overload_int"
            && row.function != "docs_overload_str"),
        "overloads must be reported under the function-set name, not the Rust function names"
    );
    let overloaded: Vec<_> = rows
        .iter()
        .filter(|row| row.function == "docs_overloaded")
        .collect();
    assert_eq!(overloaded.len(), 1, "the overload set must collapse into one row");
    let overloaded = overloaded[0];
    // description 只有 INTEGER 签名写了 → 取第一个非空就是它,与遍历顺序无关。
    // comment 只有 VARCHAR 签名写了。
    assert_eq!(
        overloaded.description.as_deref(),
        Some("Overloaded: INTEGER input")
    );
    assert_eq!(overloaded.comment.as_deref(), Some("Also accepts VARCHAR"));
    // 两个签名各给了 example,其中第一条重复 → 只保留一次。重复项被去掉后,两种遍历顺序都会得到
    // 同一个结果,所以这里可以直接逐项断言。
    //
    // Both signatures contribute examples and the first one is shared, so it must appear once. With
    // the duplicate dropped, both iteration orders give the same list, so it can be asserted item by
    // item.
    assert_eq!(
        overloaded.examples,
        vec![
            "SELECT docs_overloaded(i) FROM (VALUES (1), (2)) v(i)".to_string(),
            "SELECT docs_overloaded(s) FROM (VALUES ('a'), ('b')) v(s);".to_string()
        ]
    );
}

/// 本测试文件声明的函数名前缀。示例扩展现在与本 crate 同处一个包(`quack` feature),打开它时
/// `declared_function_descriptions()` 里会多出几十个示例函数,所以逐字节比对只针对本文件自己声明的
/// 那几行 —— 这也正是这几个测试真正关心的事。
///
/// The prefix of the functions this test file declares. The example extension now lives in the same
/// package behind the `quack` feature, so with it on `declared_function_descriptions()` returns
/// dozens of example functions as well; the byte-for-byte comparisons therefore look at this file's
/// own rows only, which is what these tests are about.
#[cfg(feature = "cli")]
const OWN_PREFIX: &str = "docs_";

/// 从整份 CSV 里挑出本测试自己声明的那几行(连同表头)。
///
/// Keeps the rows this test declares, plus the header, out of the whole CSV.
#[cfg(feature = "cli")]
fn own_csv(text: &str) -> String {
    text.lines()
        .filter(|line| line.starts_with("function,") || line.starts_with(OWN_PREFIX))
        .map(|line| format!("{line}\n"))
        .collect()
}

/// 默认导出:只写有文档的函数,表头是 community-extensions 认的四列。
///
/// The default export: only documented rows, with the header community-extensions expects.
#[cfg(feature = "cli")]
#[test]
fn export_writes_documented_rows_only() {
    let dir = temp_dir("default");
    let summary = duckfn::cli::function_descriptions::export(&dir, false).expect("export must work");
    let path = duckfn::cli::function_descriptions::output_path(&dir, false);
    let text = std::fs::read_to_string(&path).expect("the CSV must be readable");
    let _ = std::fs::remove_dir_all(&dir);

    assert_eq!(
        path.file_name().and_then(|name| name.to_str()),
        Some("function_descriptions.csv")
    );

    // 统计跟着「本 crate 实际声明的全部函数」走,不写死数字:示例扩展与运行时同处一个包。
    //
    // The counts follow every function this crate declares instead of hard-coded numbers, because
    // the example extension shares the package with the runtime.
    let declared = declared_function_descriptions();
    assert_eq!(
        summary.written,
        declared.iter().filter(|row| row.is_documented()).count()
    );
    assert_eq!(summary.without_description, 0);
    assert_eq!(
        summary.skipped,
        declared.len() - summary.written,
        "every undocumented function must be skipped"
    );

    // 本测试声明的行逐字节比对:转义、排序、合并、跳过、换行压平一次性都验了。
    //
    // Byte-for-byte comparison of the rows this test declares: escaping, ordering, merging,
    // skipping and newline flattening in one assertion.
    let own = own_csv(&text);
    assert_eq!(own, EXPECTED_DEFAULT_CSV);
    assert!(
        !text.lines().any(|line| line.starts_with("docs_undocumented")),
        "the undocumented function must not be exported by default"
    );
    // 没有字段跨物理行 —— 这正是换行被压平的结果,Markdown 表格与 `read_csv()` 都要求这样。
    //
    // No field spans a physical line, which is what the flattening buys: both the Markdown table
    // and `read_csv()` require it.
    assert_eq!(own.lines().count(), 6, "header + 5 rows, no multi-line field");
    // LF 换行(仓库约定),且文件以换行结束。
    //
    // LF line endings (the repository convention), with a trailing newline.
    assert!(!text.contains('\r'));
    assert!(text.ends_with('\n'));
}

/// `--all`:文件名带 `_all` 后缀,没写文档的函数也在里面(三列留空)。
///
/// `--all`: the file name carries the `_all` suffix and undocumented functions are included with
/// empty fields.
#[cfg(feature = "cli")]
#[test]
fn export_all_includes_undocumented_rows() {
    let dir = temp_dir("all");
    let summary = duckfn::cli::function_descriptions::export(&dir, true).expect("export must work");
    let path = duckfn::cli::function_descriptions::output_path(&dir, true);
    let text = std::fs::read_to_string(&path).expect("the CSV must be readable");
    let _ = std::fs::remove_dir_all(&dir);

    assert_eq!(
        path.file_name().and_then(|name| name.to_str()),
        Some("function_descriptions_all.csv")
    );

    // 同上:统计跟着本 crate 实际声明的全部函数走。
    //
    // As above: the counts follow every function this crate declares.
    let declared = declared_function_descriptions();
    assert_eq!(summary.written, declared.len());
    assert_eq!(
        summary.without_description,
        declared
            .iter()
            .filter(|row| row.description.is_none())
            .count()
    );
    assert_eq!(summary.skipped, 0);

    // `docs_undocumented` 排序上正好在 `docs_unicode` 前面,插进去即可。
    //
    // `docs_undocumented` sorts right before `docs_unicode`, so inserting it is enough.
    let expected = EXPECTED_DEFAULT_CSV.replace(
        "docs_unicode,",
        "docs_undocumented,,,\ndocs_unicode,",
    );
    let own = own_csv(&text);
    assert_eq!(own, expected);
    assert_eq!(own.lines().count(), 7, "header + 6 rows, no multi-line field");
}

// ============================================================================
// 往返:这份 CSV 真的能被 DuckDB 读回来
// ============================================================================

/// 把导出的 CSV 交给 `duckdb` 的 `read_csv()`,确认值原样回来、JOIN 键对得上、
/// `generate_md.sh` 那两段 SQL 的语义成立:
///
/// - 用 `function_name == other.function` 能查到行;
/// - `'[' || other.example || ']'` 给出 `[a, b]`(与原生 `list_reduce(x || ', ' || y)` 一致);
/// - 空字段被读成 `NULL`(不是空串)—— `generate_md.sh` 那边 `'[' || NULL || ']'` 就是 NULL,
///   页面留空,正是「没写示例」想要的效果。
///
/// Feeds the exported CSV to `duckdb`'s `read_csv()` and checks that values come back unchanged,
/// the JOIN key matches, and the semantics of the two `generate_md.sh` statements hold.
///
/// 没装 `duckdb` 命令行时打印提示并跳过。
///
/// Prints a notice and returns when the `duckdb` CLI is not installed.
#[cfg(feature = "cli")]
#[test]
fn csv_round_trips_through_duckdb() {
    let Some(duckdb) = duckdb_binary() else {
        eprintln!("skipping csv_round_trips_through_duckdb: no `duckdb` CLI on PATH");
        return;
    };

    let dir = temp_dir("roundtrip");
    duckfn::cli::function_descriptions::export(&dir, true).expect("export must work");
    let path = duckfn::cli::function_descriptions::output_path(&dir, true);
    let source = format!("read_csv('{}')", sql_path(&path));
    let select = |expr: &str, function: &str| {
        duckdb_scalar(
            &duckdb,
            &format!("SELECT {expr} FROM {source} WHERE function = '{function}'"),
        )
    };

    // A + C:含逗号与双引号的描述、两侧带空格的注释、非 ASCII,读回来一字不差。
    assert_eq!(
        select("description", "docs_special"),
        "Comma, \"quote\", and a newline"
    );
    assert_eq!(select("comment", "docs_special"), "  padded  ");
    assert_eq!(
        select("example", "docs_special"),
        "SELECT docs_special('a,b') -- say \"hi\" {{ }} {% raw %}"
    );
    assert_eq!(select("description", "docs_unicode"), "把 INTEGER 翻倍");

    // C:源文本里的换行没有落到 CSV 里 —— 裸换行会被 `read_csv()` 读成 `\r\n`(实测),
    // 而且会拆断 `generate_md.sh` 生成的 Markdown 表格,所以写出时就压成空格了。
    //
    // The newline in the source text never reaches the CSV: `read_csv()` reads a bare `\n` back as
    // `\r\n` (measured) and it would break the Markdown table `generate_md.sh` builds, so it is
    // flattened to a space on the way out.
    assert_eq!(select("contains(description, chr(10))", "docs_special"), "false");
    assert_eq!(select("contains(description, chr(13))", "docs_special"), "false");

    // E:多条示例拼成一个字段:按 `"; "` 连接,结尾分号被去掉,所以不会出现 `;;`;
    // `generate_md.sh` 包上方括号后就是最终渲染的形态。
    //
    // E: several examples end up in one field, joined with `"; "` and with the trailing semicolon
    // dropped (so no `;;`); wrapping in brackets is what `generate_md.sh` finally renders.
    assert_eq!(
        select("example", "docs_add_two"),
        "SELECT docs_add_two(i, 1) FROM (VALUES (1), (2), (3)) v(i) ORDER BY i; \
         SELECT docs_add_two(i, j) FROM (VALUES (1, 2), (3, 4)) v(i, j)"
    );
    assert_eq!(
        select("'[' || example || ']'", "docs_add_two"),
        "[SELECT docs_add_two(i, 1) FROM (VALUES (1), (2), (3)) v(i) ORDER BY i; \
         SELECT docs_add_two(i, j) FROM (VALUES (1, 2), (3, 4)) v(i, j)]"
    );

    // D:重载只占一行,`function` 是函数集名。
    assert_eq!(
        duckdb_scalar(
            &duckdb,
            &format!("SELECT count(*) FROM {source} WHERE function = 'docs_overloaded'"),
        ),
        "1"
    );
    assert_eq!(
        select("description", "docs_overloaded"),
        "Overloaded: INTEGER input"
    );
    // D + E:合并后的示例也读得回来 —— 共享的那条只出现一次,结尾分号被去掉。
    //
    // D + E: the merged examples read back too — the shared one appears once and the trailing
    // semicolon is gone.
    assert_eq!(
        select("example", "docs_overloaded"),
        "SELECT docs_overloaded(i) FROM (VALUES (1), (2)) v(i); \
         SELECT docs_overloaded(s) FROM (VALUES ('a'), ('b')) v(s)"
    );

    // B:`--all` 里的空字段被读成 NULL;`docs_overloaded` 的 comment 有值、example 有值。
    assert_eq!(select("comment IS NULL", "docs_undocumented"), "true");
    assert_eq!(select("comment IS NULL", "docs_double_it"), "false");

    // `function` 列就是 JOIN 键:函数名与排序都对(只看本测试声明的那几个,示例扩展与运行时同处
    // 一个包,打开 `quack` 时 CSV 里还有几十个示例函数)。
    //
    // The `function` column is the JOIN key: names and ordering both hold (scoped to this test's own
    // functions, since the example extension shares the package and adds dozens of rows when `quack`
    // is on).
    assert_eq!(
        duckdb_scalar(
            &duckdb,
            &format!(
                "SELECT string_agg(function, ',') FROM \
                 (SELECT function FROM {source} WHERE starts_with(function, 'docs_') ORDER BY 1)"
            ),
        ),
        "docs_add_two,docs_double_it,docs_overloaded,docs_special,docs_undocumented,docs_unicode"
    );

    let _ = std::fs::remove_dir_all(&dir);
}

/// 每个测试用各自的临时目录,避免互相覆盖。
///
/// Each test gets its own temporary directory so they cannot overwrite one another.
#[cfg(feature = "cli")]
fn temp_dir(label: &str) -> std::path::PathBuf {
    let dir = std::env::temp_dir().join(format!("duckfn_doc_test_{label}_{}", std::process::id()));
    let _ = std::fs::remove_dir_all(&dir);
    dir
}

/// `duckdb` 命令行:`DUCKDB` 环境变量优先(与 Justfile 的约定一致),否则 PATH 里的 `duckdb`。
///
/// The `duckdb` CLI: the `DUCKDB` environment variable wins (same convention as the Justfile),
/// otherwise `duckdb` from `PATH`.
#[cfg(feature = "cli")]
fn duckdb_binary() -> Option<String> {
    let candidate = std::env::var("DUCKDB").unwrap_or_else(|_| "duckdb".to_string());
    let ok = std::process::Command::new(&candidate)
        .arg("--version")
        .output()
        .is_ok_and(|output| output.status.success());
    ok.then_some(candidate)
}

/// 跑一条只返回单个值的 SQL,返回去掉行尾换行的原始输出。
///
/// Runs a query returning a single value and returns the raw output without the trailing newline.
/// Raw output (rather than parsing) is what makes the multi-line description comparable exactly.
#[cfg(feature = "cli")]
fn duckdb_scalar(duckdb: &str, sql: &str) -> String {
    let output = std::process::Command::new(duckdb)
        .args(["-noheader", "-list", "-c", sql])
        .output()
        .expect("running duckdb");
    assert!(
        output.status.success(),
        "duckdb failed: {}\nSQL: {sql}",
        String::from_utf8_lossy(&output.stderr)
    );
    String::from_utf8(output.stdout)
        .expect("duckdb writes UTF-8")
        .trim_end_matches(['\r', '\n'])
        .to_string()
}

/// DuckDB 的 SQL 字符串字面量里用正斜杠更省事,单引号要双写。
///
/// Forward slashes keep the path simple inside a SQL string literal, and single quotes are doubled.
#[cfg(feature = "cli")]
fn sql_path(path: &std::path::Path) -> String {
    path.display().to_string().replace('\\', "/").replace('\'', "''")
}