Skip to main content

graphforge_storage/
schemas.rs

1//! Canonical Arrow schemas for every Parquet file GraphForge reads or writes.
2//!
3//! These constants are the single source of truth for column names, types, and
4//! nullability.  All `TableProvider` implementations, relational lowering, and
5//! language bindings must reference these schemas rather than defining their own.
6//!
7//! # Dual-key identity
8//!
9//! Every node and edge row carries two identity columns:
10//!
11//! | Column | Type | Purpose |
12//! |--------|------|---------|
13//! | `*_uuid` | `FixedSizeBinary(16)` | Canonical UUIDv7 — stable, globally unique, never changes |
14//! | `*_id` | `UInt64` | Surrogate assigned at scan time — used for DataFusion joins, never in API outputs |
15//!
16//! # File layout
17//!
18//! ```text
19//! topology/nodes.parquet            → TOPOLOGY_NODES_SCHEMA
20//! topology/edges/TYPENAME.parquet   → TYPED_EDGE_SCHEMA
21//! topology/edges/_exploratory.parquet → EXPLORATORY_EDGE_SCHEMA
22//! properties/ENTITY_TYPE.parquet    → property_schema(entity_type, defs)
23//! indexes/adjacency/index_manifest.parquet → ADJACENCY_MANIFEST_SCHEMA
24//! indexes/adjacency/REL.out.csr     → ADJACENCY_CSR_SCHEMA (Arrow IPC, not Parquet)
25//! ```
26
27use std::collections::HashMap;
28use std::sync::{Arc, LazyLock};
29
30use arrow::datatypes::{DataType, Field, Fields, Schema, SchemaRef, TimeUnit};
31use graphforge_ontology::ontology::{PropertyDef, PropertyValueType};
32
33// ---------------------------------------------------------------------------
34// Helpers
35// ---------------------------------------------------------------------------
36
37/// UUID column: `FixedSizeBinary(16)`, not nullable.
38pub(crate) fn uuid_field(name: &str) -> Field {
39    Field::new(name, DataType::FixedSizeBinary(16), false)
40}
41
42/// Surrogate ID column: `UInt64`, not nullable.
43pub(crate) fn id_field(name: &str) -> Field {
44    Field::new(name, DataType::UInt64, false)
45}
46
47/// The Arrow fields of a typed Cypher `duration` value (ADR 0009): signed
48/// `Struct{months: Int64, days: Int64, seconds: Int64, nanos: Int64}`. A struct
49/// (not Arrow `Interval`) because Parquet cannot persist `Interval(MonthDayNano)`;
50/// the single source of truth shared by storage and the graphforge-rel query value.
51/// months/days/seconds are Int64 and seconds is split from nanos so billion-year
52/// `duration.between`/`inSeconds` spans fit (#920/#1011); nanos is
53/// nanoseconds-of-second, sharing the sign of seconds.
54#[must_use]
55pub fn duration_struct_fields() -> Fields {
56    Fields::from(vec![
57        Field::new("months", DataType::Int64, true),
58        Field::new("days", DataType::Int64, true),
59        Field::new("seconds", DataType::Int64, true),
60        Field::new("nanos", DataType::Int64, true),
61    ])
62}
63
64/// `Struct{epoch_day: Int64}` — a Cypher `date` typed value (ADR 0012): i64 days
65/// since the Unix epoch, spanning the full openCypher year range
66/// −999,999,999..+999,999,999. A self-describing one-field struct (a bare Int64
67/// would be indistinguishable from an integer property on decode); orders
68/// chronologically. The single source of truth shared by storage and graphforge-rel. (#1011)
69#[must_use]
70pub fn date_struct_fields() -> Fields {
71    Fields::from(vec![Field::new("epoch_day", DataType::Int64, true)])
72}
73
74/// `Struct{date: Int64, time: Time64(ns)}` — a Cypher `localdatetime` typed
75/// value (ADR 0009/0012). A two-field struct (not an epoch instant) so it spans
76/// the full openCypher year range at nanosecond precision; `date` is i64 days
77/// (#1011). The single source of truth shared by storage and the graphforge-rel value.
78#[must_use]
79pub fn localdatetime_struct_fields() -> Fields {
80    Fields::from(vec![
81        Field::new("date", DataType::Int64, true),
82        Field::new("time", DataType::Time64(TimeUnit::Nanosecond), true),
83    ])
84}
85
86/// `Struct{time: Time64(ns), offset: Int32}` — a Cypher `time` typed value: a
87/// time-of-day plus its UTC offset in seconds (ADR 0009). Shared by storage and
88/// graphforge-rel. (#920)
89#[must_use]
90pub fn time_struct_fields() -> Fields {
91    Fields::from(vec![
92        Field::new("time", DataType::Time64(TimeUnit::Nanosecond), true),
93        Field::new("offset", DataType::Int32, true),
94    ])
95}
96
97/// `Struct{date: Int64, time: Time64(ns), offset: Int32, zone: Utf8}` — a
98/// Cypher `datetime` typed value: a date+time (date = i64 days, #1011), its UTC
99/// offset in seconds, and an optional named IANA zone (null when offset-only)
100/// (ADR 0009/0012). Shared by storage and graphforge-rel.
101#[must_use]
102pub fn datetime_struct_fields() -> Fields {
103    Fields::from(vec![
104        Field::new("date", DataType::Int64, true),
105        Field::new("time", DataType::Time64(TimeUnit::Nanosecond), true),
106        Field::new("offset", DataType::Int32, true),
107        Field::new("zone", DataType::Utf8, true),
108    ])
109}
110
111/// Timestamp column: `Timestamp(Microsecond, UTC)`, not nullable.
112pub(crate) fn ts_field(name: &str) -> Field {
113    Field::new(
114        name,
115        DataType::Timestamp(TimeUnit::Microsecond, Some("UTC".into())),
116        false,
117    )
118}
119
120// ---------------------------------------------------------------------------
121// TOPOLOGY_NODES_SCHEMA
122// ---------------------------------------------------------------------------
123
124/// Schema for `topology/nodes.parquet`.
125///
126/// Stores the identity and type of every node.  Property data lives in
127/// per-entity-type files under `properties/`.
128pub static TOPOLOGY_NODES_SCHEMA: LazyLock<SchemaRef> = LazyLock::new(|| {
129    Arc::new(Schema::new(vec![
130        uuid_field("node_uuid"),
131        id_field("node_id"),
132        // Immutable primary label retained for legacy files and property-stem
133        // routing. `type_ids` is the authoritative full label set (#799).
134        Field::new("type_id", DataType::UInt32, false),
135        Field::new(
136            "type_ids",
137            DataType::List(Arc::new(Field::new("item", DataType::UInt32, false))),
138            false,
139        ),
140        ts_field("created_at"),
141        ts_field("updated_at"),
142    ]))
143});
144
145// ---------------------------------------------------------------------------
146// TYPED_EDGE_SCHEMA
147// ---------------------------------------------------------------------------
148
149/// Schema for `topology/edges/TYPENAME.parquet`.
150///
151/// One file per relation type.  Joins against `topology/nodes.parquet` use
152/// `src_id`/`dst_id` surrogates; `src_uuid`/`dst_uuid` are for API outputs.
153pub static TYPED_EDGE_SCHEMA: LazyLock<SchemaRef> = LazyLock::new(|| {
154    Arc::new(Schema::new(vec![
155        uuid_field("edge_uuid"),
156        uuid_field("src_uuid"),
157        uuid_field("dst_uuid"),
158        id_field("edge_id"),
159        id_field("src_id"),
160        id_field("dst_id"),
161        ts_field("created_at"),
162    ]))
163});
164
165// ---------------------------------------------------------------------------
166// EXPLORATORY_EDGE_SCHEMA
167// ---------------------------------------------------------------------------
168
169/// Schema for `topology/edges/_exploratory.parquet`.
170///
171/// Catch-all bucket for edges whose relation type was not declared in a formal
172/// ontology at write time.  Extends [`TYPED_EDGE_SCHEMA`] with a
173/// `rel_type_name` string column so the executor can filter by relation name.
174pub static EXPLORATORY_EDGE_SCHEMA: LazyLock<SchemaRef> = LazyLock::new(|| {
175    let mut fields: Vec<Field> = TYPED_EDGE_SCHEMA
176        .fields()
177        .iter()
178        .map(|f| f.as_ref().clone())
179        .collect();
180    fields.push(Field::new("rel_type_name", DataType::Utf8, false));
181    Arc::new(Schema::new(fields))
182});
183
184// ---------------------------------------------------------------------------
185// PROPERTY_BASE_SCHEMA
186// ---------------------------------------------------------------------------
187
188/// Minimal schema for `properties/ENTITY_TYPE.parquet` before per-type columns
189/// are added.  Contains only the join key (`node_uuid`).
190///
191/// Use [`property_schema`] to build the full per-entity-type schema.
192pub static PROPERTY_BASE_SCHEMA: LazyLock<SchemaRef> =
193    LazyLock::new(|| Arc::new(Schema::new(vec![uuid_field("node_uuid")])));
194
195// ---------------------------------------------------------------------------
196// EDGE_PROPERTY_BASE_SCHEMA
197// ---------------------------------------------------------------------------
198
199/// Minimal schema for `edge_properties/REL_TYPE.parquet` before per-relation
200/// columns are added.  Contains only the join key (`edge_uuid`).
201///
202/// Edge properties live in a dedicated `edge_properties/` directory (keyed by
203/// `edge_uuid`) so a relation type can never collide with a node label sharing
204/// the same name in `properties/`.
205pub static EDGE_PROPERTY_BASE_SCHEMA: LazyLock<SchemaRef> =
206    LazyLock::new(|| Arc::new(Schema::new(vec![uuid_field("edge_uuid")])));
207
208// ---------------------------------------------------------------------------
209// ADJACENCY_CSR_SCHEMA
210// ---------------------------------------------------------------------------
211
212/// Fields of one adjacency entry: `{edge_id: UInt64, neighbor_id: UInt64}`.
213///
214/// Shared between [`ADJACENCY_CSR_SCHEMA`] and the array builders in the
215/// [`adjacency`](crate::adjacency) module so the file schema and the arrays
216/// written into it can never drift apart.
217pub(crate) fn adjacency_entry_fields() -> Fields {
218    Fields::from(vec![id_field("edge_id"), id_field("neighbor_id")])
219}
220
221/// Schema for `indexes/adjacency/<REL_TYPE>.<dir>.csr` (Arrow IPC, ADR 0005).
222///
223/// One non-nullable column with one row per surrogate `node_id` in
224/// `0..node_count`:
225///
226/// ```text
227/// adjacency: LargeList<Struct { edge_id: UInt64, neighbor_id: UInt64 }>
228/// ```
229///
230/// This is the CSR (compressed sparse row) structure in its idiomatic Arrow
231/// encoding: the list's offsets buffer **is** the CSR offsets array (length
232/// `node_count + 1`, `Int64`), and the struct child **is** the targets array
233/// (length `edge_count`). A node with no neighbors is an empty list; an empty
234/// graph is a zero-row batch.
235pub static ADJACENCY_CSR_SCHEMA: LazyLock<SchemaRef> = LazyLock::new(|| {
236    Arc::new(Schema::new(vec![Field::new(
237        "adjacency",
238        DataType::LargeList(Arc::new(Field::new(
239            "item",
240            DataType::Struct(adjacency_entry_fields()),
241            false,
242        ))),
243        false,
244    )]))
245});
246
247// ---------------------------------------------------------------------------
248// ADJACENCY_MANIFEST_SCHEMA
249// ---------------------------------------------------------------------------
250
251/// Schema for `indexes/adjacency/index_manifest.parquet` (ADR 0005).
252///
253/// One row per CSR file. `topology_generation` records the topology counter
254/// the CSR was built from; a mismatch against the project's current counter
255/// marks the index stale. `relation_type` is a relation type name or the
256/// reserved `_all` stem for the union index.
257pub static ADJACENCY_MANIFEST_SCHEMA: LazyLock<SchemaRef> = LazyLock::new(|| {
258    Arc::new(Schema::new(vec![
259        Field::new("relation_type", DataType::Utf8, false),
260        Field::new("direction", DataType::Utf8, false),
261        Field::new("topology_generation", DataType::UInt64, false),
262        ts_field("built_at"),
263        Field::new("node_count", DataType::UInt64, false),
264        Field::new("edge_count", DataType::UInt64, false),
265    ]))
266});
267
268// ---------------------------------------------------------------------------
269// ADJACENCY_DELTA_SCHEMA
270// ---------------------------------------------------------------------------
271
272/// Schema for `indexes/adjacency/deltas/<generation>.parquet` (#765).
273///
274/// One row per edge created by the topology commit that bumped the counter to
275/// `<generation>`, in creation (ascending `edge_id`) order. The provider merges
276/// a contiguous chain of these segments onto the base CSR to serve a fresh view
277/// without a full rebuild. `rel_type_name` is the typed file stem (or the
278/// exploratory row's relation) so a per-relation overlay can filter by it; the
279/// union (`_all`) overlay takes every row.
280pub static ADJACENCY_DELTA_SCHEMA: LazyLock<SchemaRef> = LazyLock::new(|| {
281    Arc::new(Schema::new(vec![
282        Field::new("rel_type_name", DataType::Utf8, false),
283        Field::new("edge_id", DataType::UInt64, false),
284        Field::new("src_id", DataType::UInt64, false),
285        Field::new("dst_id", DataType::UInt64, false),
286    ]))
287});
288
289// ---------------------------------------------------------------------------
290// Runtime builders
291// ---------------------------------------------------------------------------
292
293/// Map a [`PropertyValueType`] to its Arrow [`DataType`].
294///
295/// `List` and `Map` are encoded as JSON in a `LargeUtf8` column because Arrow
296/// does not have a single schema for heterogeneous or self-describing values.
297#[must_use]
298pub fn property_type_to_arrow(vt: &PropertyValueType) -> DataType {
299    match vt {
300        PropertyValueType::Utf8 => DataType::Utf8,
301        PropertyValueType::Int64 => DataType::Int64,
302        PropertyValueType::Float64 => DataType::Float64,
303        PropertyValueType::Bool => DataType::Boolean,
304        PropertyValueType::Duration => DataType::Struct(duration_struct_fields()),
305        PropertyValueType::DateTime => {
306            DataType::Timestamp(TimeUnit::Microsecond, Some("UTC".into()))
307        }
308        PropertyValueType::List | PropertyValueType::Map => DataType::LargeUtf8,
309    }
310}
311
312/// Build a `properties/ENTITY_TYPE.parquet` schema for `entity_type`.
313///
314/// The schema is `PROPERTY_BASE_SCHEMA` (`node_uuid`) followed by one column
315/// per entry in `property_defs`, with schema-level metadata identifying the
316/// entity type.
317#[must_use]
318pub fn property_schema(entity_type: &str, property_defs: &[PropertyDef]) -> Schema {
319    let mut fields = vec![uuid_field("node_uuid")];
320    for def in property_defs {
321        fields.push(Field::new(
322            &def.name,
323            property_type_to_arrow(&def.value_type),
324            def.nullable,
325        ));
326    }
327    let meta: HashMap<String, String> =
328        [("graphforge.entity_type".to_owned(), entity_type.to_owned())]
329            .into_iter()
330            .collect();
331    Schema::new(fields).with_metadata(meta)
332}
333
334/// Build a query result schema with GraphForge pipeline metadata attached.
335///
336/// The metadata keys `graphforge.query_id`, `graphforge.ontology_version`, and
337/// `graphforge.ir_version` allow consumers to trace result rows back to the
338/// exact pipeline that produced them.
339///
340/// # Panics (debug builds only)
341///
342/// Panics if any field name ends with `_id`.  Surrogate ID columns (`node_id`,
343/// `src_id`, `dst_id`, `edge_id`) are execution-internal and must never appear
344/// in public API result schemas — see the storage architecture contract.
345#[must_use]
346pub fn result_schema(
347    fields: Vec<Field>,
348    query_id: &str,
349    ontology_ver: &str,
350    ir_ver: &str,
351) -> Schema {
352    debug_assert!(
353        fields.iter().all(|f| !f.name().ends_with("_id")),
354        "result_schema: surrogate '*_id' columns must not appear in public API results \
355         (offending fields: {:?})",
356        fields
357            .iter()
358            .filter(|f| f.name().ends_with("_id"))
359            .map(arrow::datatypes::Field::name)
360            .collect::<Vec<_>>()
361    );
362    let meta: HashMap<String, String> = [
363        ("graphforge.query_id".to_owned(), query_id.to_owned()),
364        (
365            "graphforge.ontology_version".to_owned(),
366            ontology_ver.to_owned(),
367        ),
368        ("graphforge.ir_version".to_owned(), ir_ver.to_owned()),
369    ]
370    .into_iter()
371    .collect();
372    Schema::new(fields).with_metadata(meta)
373}
374
375// ---------------------------------------------------------------------------
376// Tests
377// ---------------------------------------------------------------------------
378
379#[cfg(test)]
380mod tests {
381    use super::*;
382    use graphforge_ontology::ontology::PropertyValueType;
383
384    #[test]
385    fn topology_nodes_schema_field_names() {
386        let s = &*TOPOLOGY_NODES_SCHEMA;
387        assert_eq!(s.fields().len(), 6);
388        let names: Vec<&str> = s.fields().iter().map(|f| f.name().as_str()).collect();
389        assert_eq!(
390            names,
391            [
392                "node_uuid",
393                "node_id",
394                "type_id",
395                "type_ids",
396                "created_at",
397                "updated_at"
398            ]
399        );
400    }
401
402    #[test]
403    fn typed_edge_schema_field_types() {
404        let s = &*TYPED_EDGE_SCHEMA;
405        assert_eq!(s.fields().len(), 7);
406        assert_eq!(
407            s.fields()
408                .iter()
409                .map(|field| field.name().as_str())
410                .collect::<Vec<_>>(),
411            [
412                "edge_uuid",
413                "src_uuid",
414                "dst_uuid",
415                "edge_id",
416                "src_id",
417                "dst_id",
418                "created_at",
419            ]
420        );
421
422        // UUID fields are FixedSizeBinary(16)
423        for name in ["edge_uuid", "src_uuid", "dst_uuid"] {
424            let f = s.field_with_name(name).unwrap();
425            assert_eq!(
426                f.data_type(),
427                &DataType::FixedSizeBinary(16),
428                "{name} should be FixedSizeBinary(16)"
429            );
430        }
431
432        // Surrogate fields are UInt64
433        for name in ["edge_id", "src_id", "dst_id"] {
434            let f = s.field_with_name(name).unwrap();
435            assert_eq!(f.data_type(), &DataType::UInt64, "{name} should be UInt64");
436        }
437    }
438
439    #[test]
440    fn exploratory_edge_schema_has_rel_type_name() {
441        let s = &*EXPLORATORY_EDGE_SCHEMA;
442        // Must be one field longer than TYPED_EDGE_SCHEMA
443        assert_eq!(s.fields().len(), TYPED_EDGE_SCHEMA.fields().len() + 1);
444
445        let last = s.fields().last().unwrap();
446        assert_eq!(last.name(), "rel_type_name");
447        assert_eq!(last.data_type(), &DataType::Utf8);
448        assert!(!last.is_nullable());
449    }
450
451    #[test]
452    fn exploratory_edge_schema_extends_typed_edge_schema() {
453        // The first N fields of EXPLORATORY_EDGE_SCHEMA must match TYPED_EDGE_SCHEMA exactly.
454        let typed = &*TYPED_EDGE_SCHEMA;
455        let exploratory = &*EXPLORATORY_EDGE_SCHEMA;
456        for (i, typed_field) in typed.fields().iter().enumerate() {
457            assert_eq!(
458                exploratory.field(i).as_ref(),
459                typed_field.as_ref(),
460                "field {i} mismatch between TYPED and EXPLORATORY schemas"
461            );
462        }
463    }
464
465    #[test]
466    fn property_type_to_arrow_all_variants() {
467        assert_eq!(
468            property_type_to_arrow(&PropertyValueType::Utf8),
469            DataType::Utf8
470        );
471        assert_eq!(
472            property_type_to_arrow(&PropertyValueType::Int64),
473            DataType::Int64
474        );
475        assert_eq!(
476            property_type_to_arrow(&PropertyValueType::Float64),
477            DataType::Float64
478        );
479        assert_eq!(
480            property_type_to_arrow(&PropertyValueType::Bool),
481            DataType::Boolean
482        );
483        assert_eq!(
484            property_type_to_arrow(&PropertyValueType::Duration),
485            DataType::Struct(duration_struct_fields())
486        );
487        assert_eq!(
488            property_type_to_arrow(&PropertyValueType::DateTime),
489            DataType::Timestamp(TimeUnit::Microsecond, Some("UTC".into()))
490        );
491        assert_eq!(
492            property_type_to_arrow(&PropertyValueType::List),
493            DataType::LargeUtf8
494        );
495        assert_eq!(
496            property_type_to_arrow(&PropertyValueType::Map),
497            DataType::LargeUtf8
498        );
499    }
500
501    #[test]
502    fn property_schema_roundtrip() {
503        use graphforge_ontology::ontology::PropertyDef;
504        let defs = vec![
505            PropertyDef {
506                owner: "Person".into(),
507                name: "name".into(),
508                value_type: PropertyValueType::Utf8,
509                nullable: false,
510                multivalued: false,
511                default_json: None,
512            },
513            PropertyDef {
514                owner: "Person".into(),
515                name: "age".into(),
516                value_type: PropertyValueType::Int64,
517                nullable: true,
518                multivalued: false,
519                default_json: None,
520            },
521        ];
522        let schema = property_schema("Person", &defs);
523
524        // node_uuid + 2 property fields
525        assert_eq!(schema.fields().len(), 3);
526        assert_eq!(schema.field(0).name(), "node_uuid");
527        assert_eq!(schema.field(1).name(), "name");
528        assert_eq!(schema.field(1).data_type(), &DataType::Utf8);
529        assert!(!schema.field(1).is_nullable());
530        assert_eq!(schema.field(2).name(), "age");
531        assert_eq!(schema.field(2).data_type(), &DataType::Int64);
532        assert!(schema.field(2).is_nullable());
533
534        // Metadata
535        assert_eq!(
536            schema.metadata().get("graphforge.entity_type"),
537            Some(&"Person".to_owned())
538        );
539    }
540
541    #[test]
542    fn result_schema_metadata() {
543        let schema = result_schema(
544            vec![Field::new("n_name", DataType::Utf8, true)],
545            "qid-123",
546            "sha256:abc",
547            "0.1.0",
548        );
549        let meta = schema.metadata();
550        assert_eq!(meta.get("graphforge.query_id"), Some(&"qid-123".to_owned()));
551        assert_eq!(
552            meta.get("graphforge.ontology_version"),
553            Some(&"sha256:abc".to_owned())
554        );
555        assert_eq!(meta.get("graphforge.ir_version"), Some(&"0.1.0".to_owned()));
556        assert_eq!(schema.fields().len(), 1);
557    }
558
559    #[test]
560    fn adjacency_csr_schema_shape() {
561        let s = &*ADJACENCY_CSR_SCHEMA;
562        assert_eq!(s.fields().len(), 1);
563        let f = s.field(0);
564        assert_eq!(f.name(), "adjacency");
565        assert!(!f.is_nullable());
566        let DataType::LargeList(item) = f.data_type() else {
567            panic!("adjacency should be a LargeList, got {:?}", f.data_type());
568        };
569        assert!(!item.is_nullable());
570        let DataType::Struct(entry) = item.data_type() else {
571            panic!("list item should be a Struct, got {:?}", item.data_type());
572        };
573        let names: Vec<&str> = entry.iter().map(|f| f.name().as_str()).collect();
574        assert_eq!(names, ["edge_id", "neighbor_id"]);
575        for f in entry {
576            assert_eq!(f.data_type(), &DataType::UInt64);
577            assert!(!f.is_nullable());
578        }
579    }
580
581    #[test]
582    fn adjacency_manifest_schema_field_names_and_types() {
583        let s = &*ADJACENCY_MANIFEST_SCHEMA;
584        let names: Vec<&str> = s.fields().iter().map(|f| f.name().as_str()).collect();
585        assert_eq!(
586            names,
587            [
588                "relation_type",
589                "direction",
590                "topology_generation",
591                "built_at",
592                "node_count",
593                "edge_count"
594            ]
595        );
596        for name in ["relation_type", "direction"] {
597            let f = s.field_with_name(name).unwrap();
598            assert_eq!(f.data_type(), &DataType::Utf8);
599        }
600        for name in ["topology_generation", "node_count", "edge_count"] {
601            let f = s.field_with_name(name).unwrap();
602            assert_eq!(f.data_type(), &DataType::UInt64);
603        }
604        assert_eq!(
605            s.field_with_name("built_at").unwrap().data_type(),
606            &DataType::Timestamp(TimeUnit::Microsecond, Some("UTC".into()))
607        );
608        assert!(s.fields().iter().all(|f| !f.is_nullable()));
609    }
610
611    #[test]
612    fn result_schema_allows_uuid_fields() {
613        // UUID columns (*_uuid) are allowed; surrogate *_id columns are not.
614        let schema = result_schema(
615            vec![
616                Field::new("node_uuid", DataType::FixedSizeBinary(16), false),
617                Field::new("name", DataType::Utf8, true),
618            ],
619            "q1",
620            "v1",
621            "0.1.0",
622        );
623        assert_eq!(schema.fields().len(), 2);
624    }
625
626    #[cfg(debug_assertions)]
627    #[test]
628    #[should_panic(expected = "surrogate '*_id' columns must not appear")]
629    fn result_schema_rejects_surrogate_id_field() {
630        // In debug builds, passing a *_id field must panic.
631        let _ = result_schema(
632            vec![Field::new("node_id", DataType::UInt64, false)],
633            "q1",
634            "v1",
635            "0.1.0",
636        );
637    }
638}