Skip to main content

spg_storage/
lib.rs

1//! In-memory storage primitives.
2//!
3//! v0.3 is intentionally simple: a flat catalog of tables, each holding rows
4//! as `Vec<Value>` (positional, matching the table's `TableSchema`). No MVCC,
5//! no on-disk format — those land in later milestones.
6#![no_std]
7// v3.3.2 NEON path for l2_distance_sq (aarch64 only). Scoped allow:
8// `unsafe_code = "deny"` at workspace level stays in force for every
9// other crate.
10#![cfg_attr(target_arch = "aarch64", allow(unsafe_code))]
11
12extern crate alloc;
13
14pub mod bignum;
15pub mod bloom;
16mod codec;
17pub mod fts_simple;
18pub mod halfvec;
19pub mod jsonb_gin;
20mod nsw;
21pub mod persistent;
22pub mod persistent_btree;
23pub mod posting;
24pub mod quantize;
25pub mod row_header;
26pub mod row_locator;
27pub mod segment;
28pub mod snapshot;
29mod table;
30pub mod trgm;
31pub mod vacuum;
32
33pub use self::bloom::{BloomError, BloomFilter};
34// v7.31 monster tier-3 cut 3 — on-disk codec moved to `codec`; the
35// public dense-row surface keeps its `spg_storage::*` paths, and the
36// low-level write/read primitives stay crate-visible for the
37// `Catalog::serialize`/`deserialize` methods that remain in this file.
38pub(crate) use self::codec::*;
39pub use self::codec::{
40    decode_row_body_dense, decode_row_body_dense_pruned, encode_row_body_dense,
41    encode_row_body_dense_into, encode_row_body_dense_masked_into, row_body_encoded_len,
42};
43// v7.31 monster tier-3 cut 2 — HNSW algorithms moved to `nsw`; the
44// public vector-search surface keeps its `spg_storage::*` paths via
45// these re-exports, and `nsw_insert_at` stays crate-visible for the
46// `Table` insert paths in the `table` module.
47pub(crate) use self::nsw::nsw_insert_at;
48pub use self::nsw::{NswMetric, cosine_dot_norms_f32, inner_product_f32, nsw_index_on, nsw_query};
49pub use self::posting::PostingList;
50
51/// The list handed back for an absent key, so callers cannot tell an
52/// absent key from an empty posting list — the property the old
53/// `&[][..]` return had, kept.
54static EMPTY_POSTINGS: crate::posting::PostingList = crate::posting::PostingList::new();
55pub use self::row_locator::{RowLocator, RowLocatorError};
56pub use self::segment::{
57    BRIN_SIDECAR_MAGIC, BrinSummary, OwnedSegment, SEGMENT_COMPRESS_ALGO_LZSS,
58    SEGMENT_COMPRESS_ALGO_NONE, SEGMENT_MAGIC, SEGMENT_MAGIC_V2, SEGMENT_PAGE_BYTES, SegmentError,
59    SegmentMeta, SegmentReader, derive_brin_summaries, encode_segment, wrap_v2_envelope,
60    wrap_v2_envelope_with_brin,
61};
62
63use alloc::borrow::Cow;
64use alloc::boxed::Box;
65use alloc::collections::{BTreeMap, BTreeSet};
66use alloc::format;
67use alloc::string::{String, ToString};
68use alloc::sync::Arc;
69use alloc::vec::Vec;
70use core::fmt;
71
72use self::persistent::PersistentVec;
73use self::persistent_btree::PersistentBTreeMap;
74
75/// In-cell encoding for `DataType::Vector`. Mirrors
76/// `spg_sql::ast::VecEncoding` — kept here so storage stays
77/// dep-free of `spg-sql`. The engine bridges between the two
78/// at DDL-execution time.
79///
80/// `F32` is the pre-v6 default: each cell holds a raw `Vec<f32>`.
81/// `Sq8` (v6.0.1) stores `Sq8Vector { min, max, bytes: Vec<u8> }`
82/// per cell; 4× compression vs `F32` with recall@10 ≥ 0.95 on
83/// natural embeddings (Gaussian / unit-sphere corpora).
84/// `F16` (v6.0.3, DDL keyword `HALF`) stores each element as
85/// IEEE-754 binary16; 2× compression and bit-exact dequantise.
86#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
87pub enum VecEncoding {
88    #[default]
89    F32,
90    Sq8,
91    F16,
92}
93
94impl fmt::Display for VecEncoding {
95    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
96        match self {
97            Self::F32 => f.write_str("F32"),
98            Self::Sq8 => f.write_str("SQ8"),
99            Self::F16 => f.write_str("HALF"),
100        }
101    }
102}
103
104/// Runtime type tags. `Vector { dim, encoding }` / `Varchar(max)` /
105/// `Char(size)` are parameterised; the parameter travels with both
106/// the column schema and the on-wire serialised representation.
107#[derive(Debug, Clone, Copy, PartialEq, Eq)]
108pub enum DataType {
109    /// 16-bit signed. Backed by `Value::SmallInt(i16)`; arithmetic that
110    /// would overflow surfaces as a type error at INSERT time.
111    SmallInt,
112    Int,    // 32-bit signed
113    BigInt, // 64-bit signed
114    Float,  // f64 (PG double precision)
115    /// v7.38 (read01, T-float4) — `real` / `float4`: 32-bit IEEE float (PG
116    /// `real`). Backed by `Value::Real(f32)`; behaves like `Float` for most
117    /// dispatch but renders / stores at f32 precision.
118    Real,
119    Text,
120    /// `VARCHAR(n)` — same byte representation as `Text`, but INSERT
121    /// rejects values longer than `n` Unicode characters.
122    Varchar(u32),
123    /// `CHAR(n)` — same representation as `Text`, but INSERT right-pads
124    /// with U+0020 to exactly `n` Unicode characters (or rejects when
125    /// the input is already longer).
126    Char(u32),
127    Bool,
128    /// pgvector-style fixed-dimension vector. `encoding` selects
129    /// the in-cell representation (`F32` = pre-v6 raw f32 buffer;
130    /// `Sq8` = v6.0.1 8-bit scalar-quantised). The DDL grammar
131    /// surfaces encoding via the optional `USING <encoding>`
132    /// clause: `VECTOR(128) USING SQ8`.
133    Vector {
134        dim: u32,
135        encoding: VecEncoding,
136    },
137    /// `NUMERIC(precision, scale)` — exact fixed-point decimal stored as
138    /// a scaled `i128`. `precision` caps total decimal digits, `scale`
139    /// fixes digits after the decimal point. v1.12 supports up to
140    /// precision 38 (the i128-safe ceiling). `NUMERIC` and `NUMERIC(p)`
141    /// surface as `Numeric { precision: p, scale: 0 }`.
142    Numeric {
143        /// v7.39 (round 272) — widened from u8. PG's declared precision
144        /// runs to 1000; at u8 it could not even be spelled, and the
145        /// parser rejected anything past 38 (i128's width) outright.
146        precision: u16,
147        /// v7.39 (round 271) — widened alongside the value's scale.
148        /// v7.39 (round 273) — and signed: PG's DECLARED scale runs
149        /// -1000..=1000, where a negative one rounds to tens / hundreds.
150        /// A VALUE's display scale is always non-negative.
151        scale: i16,
152    },
153    /// `DATE` — calendar date with day precision, stored as `i32` days
154    /// since the Unix epoch (1970-01-01).
155    Date,
156    /// `TIMESTAMP` (a.k.a. `MySQL` `DATETIME`) — instant with microsecond
157    /// precision, stored as `i64` microseconds since the Unix epoch.
158    Timestamp,
159    /// v7.9.2 `TIMESTAMPTZ` — bit-identical to `Timestamp` on disk
160    /// (i64 microseconds, UTC by convention). Carried as a distinct
161    /// type tag so the PG-wire layer can advertise OID 1184 (PG's
162    /// `timestamp with time zone`) and `sqlx`/`pgx`/JDBC clients
163    /// decode into their TZ-aware datetime types. The internal
164    /// semantics are unchanged: SPG never stored per-row offsets,
165    /// and neither did PG — `TIMESTAMPTZ` in PG is also UTC i64.
166    Timestamptz,
167    /// v7.39 (round 291) — PG's `name`: the type its catalogs use for
168    /// identifiers. Text truncated to NAMEDATALEN-1 (63) bytes, with
169    /// its own type identity — `pg_typeof('abc'::name)` is `name`, and
170    /// `CREATE TABLE t (a name)` is legal SQL that SPG rejected.
171    Name,
172    /// v7.39 (round 640) — PG's `xid`: a transaction id. [`Value::Xid`]
173    /// has existed since round 512, so a `'5'::xid` literal already knew
174    /// what it was; this is the DECLARED half, which nothing had. Without
175    /// it `pg_typeof(NULL::xid)` answered `bigint`, `pg_type` could not
176    /// list oid 28 — leaving the 48 `pg_attribute` rows that describe
177    /// `xmin` / `xmax` pointing at a type no catalog carried — and
178    /// `CREATE TABLE t (a xid)` was refused as an unknown type.
179    ///
180    /// On disk it is the 8-byte body its BIGINT sibling writes, and it
181    /// reads back as a `Value::Xid`, so a stored column and a literal are
182    /// the same thing to everything downstream.
183    ///
184    /// What is NOT yet true of the identity: PG gives `xid` equality and
185    /// hashing and no ordering operator at all, so `min` / `max` /
186    /// `count(DISTINCT …)` / `<=` all error there and all answer here.
187    /// Measured, not assumed — and left for the operator surface rather
188    /// than claimed by this comment.
189    Xid,
190    /// v7.39 (round 640) — PG's `xid8`: the same transaction id, 64 bits
191    /// wide and monotonic. Unlike [`DataType::Xid`] it has no value of
192    /// its own; a cell is a `Value::BigInt` and only the declared type
193    /// witnesses it. That is enough for `pg_typeof`, the catalogs and
194    /// the wire OID, and not enough to refuse a bigint where PG refuses
195    /// one. `pg_current_xact_id()` returns this type on PG.
196    Xid8,
197    /// v7.39 (round 667) — PG's `oid`: an unsigned 32-bit object
198    /// identifier. Modelled exactly like [`DataType::Xid8`] above: it has
199    /// no value of its own, a cell is a `Value::BigInt`, and only the
200    /// declared type witnesses it.
201    ///
202    /// That deliberately buys less than a full value type. What it buys:
203    /// `CREATE TABLE t(o OID)` is accepted (it was rejected outright with
204    /// `type "oid" does not exist`, while the neighbouring `XID` worked),
205    /// `pg_typeof` answers `oid` rather than `bigint`, and the catalogs
206    /// report their own key columns honestly. What it does NOT buy is
207    /// refusing a bigint where PG refuses an oid — `sum(oid)` and
208    /// `avg(oid)` still answer here and error on PG, because at runtime
209    /// the cell is indistinguishable from a bigint. Round 664 tried to
210    /// close those two by name and withdrew: a guard keyed on the name
211    /// would have caught `sum(bigint)` with it.
212    ///
213    /// The cast itself was already right before this — `4294967296::oid`
214    /// and `'abc'::oid` produce PG's errors word for word, and `(-1)::oid`
215    /// wraps to 4294967295 as PG does. Only the resulting type was lost,
216    /// because `conversions.rs` mapped the target to `BigInt`.
217    Oid,
218    /// `INTERVAL` — calendar-aware span (months + microseconds). v2.11
219    /// supports INTERVAL only as a runtime intermediate (literals,
220    /// arithmetic results); on-disk encoding is rejected so this branch
221    /// can't appear in a `ColumnSchema`.
222    Interval,
223    /// v4.9: `JSON` — text-backed JSON document. We don't parse
224    /// the content (no path operators or jsonb functions yet) —
225    /// the column accepts any TEXT-compatible value and round-trips
226    /// it verbatim. PG OID 114 on the wire.
227    Json,
228    /// v7.9.0: `JSONB` — semantically identical to `Json` on
229    /// the storage side (same `Value::Json` cells, same
230    /// row codec), but advertised as PG OID 3802 on the wire
231    /// so `sqlx`-style clients that bind `jsonb` columns
232    /// decode correctly. mailrs migration blocker #3.
233    Jsonb,
234    /// v7.10.4: `BYTES` / `BYTEA` — variable-length raw binary.
235    /// Backed by `Value::Bytes(Vec<u8>)`. PG wire OID 17. Literal
236    /// forms accepted by parser/engine: PG hex form `'\xDEADBEEF'`
237    /// (case-insensitive hex pairs) and escape form
238    /// `'foo\\000bar'` (the latter decoded at coercion time when
239    /// the target column is BYTEA — TEXT columns leave the
240    /// backslash sequence verbatim).
241    Bytes,
242    /// v7.10.9: `TEXT[]` — single-dimension TEXT array. Elements
243    /// may be NULL (PG semantics). PG wire OID 1009. Literal
244    /// forms: `ARRAY['a', 'b', NULL]` and the PG external form
245    /// `'{a,b,NULL}'::TEXT[]`. Engine implements `= ANY(arr)`,
246    /// `<> ALL(arr)`, and 1-based indexing `arr[i]`. Catalog
247    /// FILE_VERSION 18+; older snapshots reject this DataType
248    /// (forward-only by design — TEXT[] columns aren't readable
249    /// on a pre-v7.10 binary).
250    TextArray,
251    /// v7.11.12: `INT[]` — single-dimension i32 array. PG wire
252    /// OID 1007 (_int4). Same `ARRAY[...]` / `'{1,2,3}'::INT[]`
253    /// literal surface as TEXT[]. Catalog FILE_VERSION 19+.
254    IntArray,
255    /// v7.11.12: `BIGINT[]` — single-dimension i64 array. PG
256    /// wire OID 1016 (_int8). Catalog FILE_VERSION 19+.
257    BigIntArray,
258    /// v7.39 (round 694) — `oid[]`. It exists for the reason
259    /// [`DataType::Oid`] does: mapping it onto `BigIntArray` answers
260    /// `pg_typeof('{1,2}'::oid[])` with `bigint[]`, which is the defect
261    /// round 667 closed for the scalar.
262    OidArray,
263    /// v7.37.5 β-P4 — `INTERVAL[]` — single-dimension array of
264    /// `IntervalSpan { months, days, micros }`. PG wire OID 1187
265    /// (`_interval`). Catalog tag 35 + per-cell body
266    /// `[u16 count][per elem: u8 null + (if non-null) 16-byte
267    /// interval body in LE PG-byte-equal field order]`.
268    /// FILE_VERSION 48+.
269    IntervalArray,
270    /// v7.37.5 γ — full PG array-of-scalar family. Catalog tags
271    /// 36..48; wire OIDs from PG `pg_type.dat`. Per-element body
272    /// uses the scalar's existing `write_value_body` shape.
273    /// FILE_VERSION 48+ (same window as β; no separate bump).
274    BoolArray, // PG `_bool`        OID 1000, tag 36
275    SmallIntArray,    // PG `_int2`        OID 1005, tag 37
276    FloatArray,       // PG `_float8`      OID 1022, tag 38
277    NumericArray,     // PG `_numeric`     OID 1231, tag 39
278    DateArray,        // PG `_date`        OID 1182, tag 40
279    TimestampArray,   // PG `_timestamp`   OID 1115, tag 41
280    TimestamptzArray, // PG `_timestamptz` OID 1185, tag 42
281    UuidArray,        // PG `_uuid`        OID 2951, tag 43
282    JsonArray,        // PG `_json`        OID 199,  tag 44
283    JsonbArray,       // PG `_jsonb`       OID 3807, tag 45
284    BytesArray,       // PG `_bytea`       OID 1001, tag 46
285    VarcharArray,     // PG `_varchar`     OID 1015, tag 47
286    CharArray,        // PG `_bpchar`      OID 1014, tag 48
287    /// v7.37.5 δ — PG 14+ multirange types. A multirange is an
288    /// ordered collection of non-overlapping ranges of the same
289    /// element kind (e.g. `int4multirange(int4range(1,5),
290    /// int4range(10,15))` → `{[1,5),[10,15)}`). The same DataType
291    /// variant covers all six builtin multiranges; `RangeKind`
292    /// pins the element type so encode/decode/display can route
293    /// off one switch (parallel to `Range(RangeKind)`).
294    /// Wire OIDs: int4multirange=4451, int8multirange=4537,
295    /// nummultirange=4536, tsmultirange=4533, tstzmultirange=4534,
296    /// datemultirange=4535. Catalog tag 49 + 1-byte RangeKind on
297    /// the dense type-tag side. FILE_VERSION 48+ (same window as
298    /// β/γ, no separate bump).
299    Multirange(RangeKind),
300    /// v7.37.5 ε — PG geometry scalar family. Mirrors PG's seven
301    /// builtin geometric types one-for-one. Body shapes (LE):
302    ///   Point   = 16 B fixed (f64 x + f64 y)            OID 600
303    ///   Lseg    = 32 B fixed (Point p1 + Point p2)      OID 601
304    ///   Path    = varlena ([u8 closed][u32 n][Point*n]) OID 602
305    ///   Box     = 32 B fixed (Point ur + Point ll)      OID 603
306    ///   Polygon = varlena ([u32 n][Point*n])            OID 604
307    ///   Line    = 24 B fixed (f64 a + f64 b + f64 c)    OID 628
308    ///   Circle  = 24 B fixed (Point center + f64 r)     OID 718
309    /// Catalog tags 50..56. FILE_VERSION 48+ (same window as β/γ/δ;
310    /// no separate bump). Geometric operators (`<->` / `@>` / `&&`
311    /// / `<<` / `>>` / `~=`) are a planner-integration follow-up,
312    /// parallel to the Range operator defer in e2e_pg_range.rs.
313    Point,
314    Lseg,
315    Path,
316    PgBox,
317    Polygon,
318    Line,
319    Circle,
320    /// v7.37.5 ζ-A — PG network address family. Body shapes (LE):
321    ///   Inet     = 18 B fixed (u8 family + u8 bits + 16 B addr)  OID 869
322    ///   Cidr     = 18 B fixed (same shape as Inet; CIDR rejects
323    ///                          host bits at parse / coerce)       OID 650
324    ///   Macaddr  = 6 B fixed                                      OID 829
325    ///   Macaddr8 = 8 B fixed (EUI-64)                             OID 774
326    /// Catalog tags 57-60. FILE_VERSION 48+. `family = 4` is IPv4
327    /// (uses the first 4 bytes of the 16-B addr slot, rest 0);
328    /// `family = 6` is IPv6 (full 16 B).
329    Inet,
330    Cidr,
331    Macaddr,
332    Macaddr8,
333    /// v7.39 (read01 pg_lsn.c) — PG `pg_lsn` (WAL location). 8 bytes,
334    /// rendered `%X/%X`. Catalog tag 66. OID 3220.
335    PgLsn,
336    /// v7.37.5 ζ-A — PG bit string. Body = `[u32 nbits][ceil(nbits/8) bytes]`,
337    /// big-endian within each byte (matches PG binary).
338    ///   Bit         OID 1560 (fixed-length, but SPG carries the
339    ///                         length per cell — column declaration
340    ///                         `BIT(n)` constrains at coerce time)
341    ///   BitVarying  OID 1562 (variable-length, declared as `VARBIT`)
342    /// Catalog tags 61-62.
343    /// v7.39 (round 281) — `BIT(n)`: a FIXED-length bit string. `0`
344    /// means the type was written without a typmod, which PG treats as
345    /// `bit(1)`. Column assignment requires the length to match
346    /// exactly; an explicit cast pads or truncates instead.
347    Bit(u32),
348    /// v7.39 (round 281) — `BIT VARYING(n)`: `n` is a MAXIMUM, and `0`
349    /// means unbounded (`varbit` with no typmod).
350    BitVarying(u32),
351    /// v7.37.5 ζ-A — PG `xml`. Body identical to TEXT (storage is
352    /// the verbatim XML string; no parse-time validation). Only
353    /// the wire OID (142) differs. Catalog tag 63.
354    Xml,
355    /// v7.37.5 ζ-A — PG `"char"` (the internal single-byte type,
356    /// distinct from `CHAR(n)` / `BPCHAR`). Body = 1 byte raw.
357    /// OID 18. Catalog tag 64.
358    Char1,
359    /// v7.37.5 ζ-A — `MONEY[]`. Body = `[u16 count][per elem: u8 null
360    /// + (non-null) i64 LE cents]`. OID 791. Catalog tag 65.
361    MoneyArray,
362    /// v7.12.0: PG `tsvector` — ordered, deduplicated set of
363    /// `(lexeme, positions, weight)` tuples. PG wire OID 3614.
364    /// Catalog FILE_VERSION 20+. Storage shape is row-codec
365    /// tag 22; the schema-agnostic `write_value` path emits tag
366    /// 18. Literal: `'foo:1 bar:2,3'::tsvector` (PG external
367    /// form). G-CRIT-3 entry — v7.12.0 only ships the type +
368    /// codec; matching `@@` lands in v7.12.2.
369    TsVector,
370    /// v7.12.0: PG `tsquery` — parse tree of lexemes joined by
371    /// `&` `|` `!` and phrase operators. PG wire OID 3615.
372    /// Catalog FILE_VERSION 20+.
373    TsQuery,
374    /// v7.17.0: PG `uuid` — 128-bit identifier stored as
375    /// `Value::Uuid([u8; 16])`. PG wire OID 2950. Canonical
376    /// text form is lowercase 8-4-4-4-12 hyphenated; input
377    /// also accepts uppercase, unhyphenated, and brace-wrapped
378    /// forms (`{xxxx…}`). Catalog FILE_VERSION 36+; tag 24 on
379    /// the dense type-tag side, tag 20 on the schema-agnostic
380    /// value side. The drop-in PG/MySQL surface for Django /
381    /// Rails / Hibernate "id UUID PRIMARY KEY DEFAULT
382    /// gen_random_uuid()" default-PK pattern.
383    Uuid,
384    /// v7.17.0 Phase 3.P0-32: PG `time` (without time zone) — i64
385    /// microseconds since 00:00:00. PG wire OID 1083. Display:
386    /// canonical zero-padded `HH:MM:SS` when fractional is zero,
387    /// `HH:MM:SS.ffffff` otherwise. Catalog FILE_VERSION 37+;
388    /// tag 25 on the dense type-tag side, tag 21 on the schema-
389    /// agnostic value side. The wall-clock-of-day half of PG's
390    /// date/time triplet (date / time / timestamp).
391    Time,
392    /// v7.17.0 Phase 3.P0-33: MySQL `YEAR` — u16 in range
393    /// 1901..=2155 plus the special zero-year sentinel 0. No
394    /// dedicated PG OID (advertised as INT4 / OID 23 on the wire
395    /// — psql renders integers, MySQL CLI renders 4-digit
396    /// zero-padded text). Display always 4 digits: `0000` for the
397    /// zero-year, `1985` / `2007` / etc otherwise. Catalog
398    /// FILE_VERSION 38+; tag 26 on the dense type-tag side, tag
399    /// 22 on the schema-agnostic value side.
400    Year,
401    /// v7.17.0 Phase 3.P0-34: PG `time with time zone` (TIMETZ) —
402    /// i64 microseconds since 00:00:00 in the local wall clock
403    /// PLUS i32 offset-from-UTC in seconds. PG wire OID 1266.
404    /// Display: `HH:MM:SS[.ffffff]±HH[:MM]` (PG `timetz_out`).
405    /// Range: offset in ±50400 seconds (±14 hours). Catalog
406    /// FILE_VERSION 39+; tag 27 on the dense type-tag side, tag
407    /// 23 on the schema-agnostic value side.
408    TimeTz,
409    /// v7.17.0 Phase 3.P0-35: PG `money` — i64 cents (locale-
410    /// independent storage). PG wire OID 790. Display: en_US
411    /// locale (`$N,NNN.CC`, negative → `-$1.23`). Input accepts
412    /// `$N.NN`, `$N,NNN.NN`, bare integer (treated as major
413    /// units), optional leading `-`. Range: full i64. Catalog
414    /// FILE_VERSION 40+; tag 28 on the dense type-tag side, tag
415    /// 24 on the schema-agnostic value side.
416    Money,
417    /// v7.17.0 Phase 3.P0-38: PG range type. The same DataType
418    /// variant covers all six builtin ranges (int4range,
419    /// int8range, numrange, tsrange, tstzrange, daterange) —
420    /// `RangeKind` pins the element type so encode / decode /
421    /// display can route off one switch. Catalog FILE_VERSION
422    /// 43+; tag 29 + a 1-byte RangeKind on the dense type-tag
423    /// side, tag 25 on the schema-agnostic value side.
424    Range(RangeKind),
425    /// v7.17.0 Phase 3.P0-39: PG `hstore` extension type — flat
426    /// `text => text` map with NULL value support. Catalog
427    /// FILE_VERSION 44+; tag 30 on the dense type-tag side, tag
428    /// 26 on the schema-agnostic value side. The contrib OID is
429    /// installation-dependent in real PG; SPG advertises it via
430    /// dynamic lookup, falling back to TEXT (OID 25) on the wire
431    /// when the installed `hstore` extension hasn't claimed an
432    /// OID yet.
433    Hstore,
434    /// v7.17.0 Phase 3.P0-40: PG `int[][]` — 2-dimensional INT
435    /// matrix. Storage: row-major Vec<Vec<Option<i32>>>. All
436    /// rows must share the same column count. Wire OID 1007
437    /// (same as INT[]; the dimension count travels in the data
438    /// header, not the OID). Catalog FILE_VERSION 45+; tag 31
439    /// on the dense type-tag side, tag 27 on the schema-agnostic
440    /// value side.
441    IntArray2D,
442    /// v7.17.0 Phase 3.P0-40: PG `bigint[][]` — 2-dimensional
443    /// BIGINT matrix. Storage / OID / tags mirror IntArray2D.
444    /// Tag 32 dense, tag 28 schema-agnostic.
445    BigIntArray2D,
446    /// v7.17.0 Phase 3.P0-40: PG `text[][]` — 2-dimensional TEXT
447    /// matrix. Storage: row-major Vec<Vec<Option<String>>>.
448    /// Tag 33 dense, tag 29 schema-agnostic.
449    TextArray2D,
450    /// v7.39 (read01 round 75) — `bool[][]`. BOOL is the ONE element type whose
451    /// ARRAY rendering differs from its scalar one (`t` vs `true`), so a
452    /// text-backed 2-D cannot be PG-faithful for it: rendering the whole array
453    /// wants `t`, and subscripting a cell to text wants `false`. Every other
454    /// element type renders the same either way, which is why this is the only
455    /// typed 2-D variant SPG needs.
456    BoolArray2D,
457}
458
459/// v7.17.0 Phase 3.P0-38 — pins the element type of a range value
460/// or column. Wire OIDs: Int4=3904, Int8=3926, Num=3906,
461/// Ts=3908, TsTz=3910, Date=3912.
462#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
463pub enum RangeKind {
464    Int4,
465    Int8,
466    Num,
467    Ts,
468    TsTz,
469    Date,
470}
471
472impl RangeKind {
473    pub const fn tag(self) -> u8 {
474        match self {
475            Self::Int4 => 0,
476            Self::Int8 => 1,
477            Self::Num => 2,
478            Self::Ts => 3,
479            Self::TsTz => 4,
480            Self::Date => 5,
481        }
482    }
483    pub const fn from_tag(t: u8) -> Option<Self> {
484        Some(match t {
485            0 => Self::Int4,
486            1 => Self::Int8,
487            2 => Self::Num,
488            3 => Self::Ts,
489            4 => Self::TsTz,
490            5 => Self::Date,
491            _ => return None,
492        })
493    }
494    pub const fn keyword(self) -> &'static str {
495        match self {
496            Self::Int4 => "INT4RANGE",
497            Self::Int8 => "INT8RANGE",
498            Self::Num => "NUMRANGE",
499            Self::Ts => "TSRANGE",
500            Self::TsTz => "TSTZRANGE",
501            Self::Date => "DATERANGE",
502        }
503    }
504}
505
506impl fmt::Display for DataType {
507    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
508        match self {
509            Self::SmallInt => f.write_str("SMALLINT"),
510            Self::Int => f.write_str("INT"),
511            Self::BigInt => f.write_str("BIGINT"),
512            Self::Xid => f.write_str("XID"),
513            Self::Xid8 => f.write_str("XID8"),
514            Self::Oid => f.write_str("OID"),
515            Self::OidArray => f.write_str("OID[]"),
516            Self::Float => f.write_str("FLOAT"),
517            Self::Real => f.write_str("REAL"),
518            Self::Text => f.write_str("TEXT"),
519            Self::Varchar(n) => write!(f, "VARCHAR({n})"),
520            Self::Char(n) => write!(f, "CHAR({n})"),
521            Self::Bool => f.write_str("BOOL"),
522            Self::Vector { dim, encoding } => match encoding {
523                VecEncoding::F32 => write!(f, "VECTOR({dim})"),
524                VecEncoding::Sq8 => write!(f, "VECTOR({dim}) USING SQ8"),
525                VecEncoding::F16 => write!(f, "VECTOR({dim}) USING HALF"),
526            },
527            Self::Numeric { precision, scale } => {
528                if *scale == 0 {
529                    write!(f, "NUMERIC({precision})")
530                } else {
531                    write!(f, "NUMERIC({precision}, {scale})")
532                }
533            }
534            Self::Date => f.write_str("DATE"),
535            Self::Timestamp => f.write_str("TIMESTAMP"),
536            Self::Timestamptz => f.write_str("TIMESTAMPTZ"),
537            Self::Name => f.write_str("NAME"),
538            Self::Interval => f.write_str("INTERVAL"),
539            Self::Json => f.write_str("JSON"),
540            Self::Jsonb => f.write_str("JSONB"),
541            Self::Bytes => f.write_str("BYTEA"),
542            Self::TextArray => f.write_str("TEXT[]"),
543            Self::IntArray => f.write_str("INT[]"),
544            Self::BigIntArray => f.write_str("BIGINT[]"),
545            Self::IntervalArray => f.write_str("INTERVAL[]"),
546            Self::BoolArray => f.write_str("BOOL[]"),
547            Self::SmallIntArray => f.write_str("SMALLINT[]"),
548            Self::FloatArray => f.write_str("FLOAT[]"),
549            Self::NumericArray => f.write_str("NUMERIC[]"),
550            Self::DateArray => f.write_str("DATE[]"),
551            Self::TimestampArray => f.write_str("TIMESTAMP[]"),
552            Self::TimestamptzArray => f.write_str("TIMESTAMPTZ[]"),
553            Self::UuidArray => f.write_str("UUID[]"),
554            Self::JsonArray => f.write_str("JSON[]"),
555            Self::JsonbArray => f.write_str("JSONB[]"),
556            Self::BytesArray => f.write_str("BYTEA[]"),
557            Self::VarcharArray => f.write_str("VARCHAR[]"),
558            Self::CharArray => f.write_str("CHAR[]"),
559            Self::Multirange(k) => f.write_str(match k {
560                RangeKind::Int4 => "INT4MULTIRANGE",
561                RangeKind::Int8 => "INT8MULTIRANGE",
562                RangeKind::Num => "NUMMULTIRANGE",
563                RangeKind::Ts => "TSMULTIRANGE",
564                RangeKind::TsTz => "TSTZMULTIRANGE",
565                RangeKind::Date => "DATEMULTIRANGE",
566            }),
567            Self::Point => f.write_str("POINT"),
568            Self::Lseg => f.write_str("LSEG"),
569            Self::Path => f.write_str("PATH"),
570            Self::PgBox => f.write_str("BOX"),
571            Self::Polygon => f.write_str("POLYGON"),
572            Self::Line => f.write_str("LINE"),
573            Self::Circle => f.write_str("CIRCLE"),
574            Self::Inet => f.write_str("INET"),
575            Self::Cidr => f.write_str("CIDR"),
576            Self::Macaddr => f.write_str("MACADDR"),
577            Self::Macaddr8 => f.write_str("MACADDR8"),
578            Self::PgLsn => f.write_str("PG_LSN"),
579            Self::Bit(0) => f.write_str("BIT"),
580            Self::Bit(n) => write!(f, "BIT({n})"),
581            Self::BitVarying(0) => f.write_str("VARBIT"),
582            Self::BitVarying(n) => write!(f, "VARBIT({n})"),
583            Self::Xml => f.write_str("XML"),
584            Self::Char1 => f.write_str("\"char\""),
585            Self::MoneyArray => f.write_str("MONEY[]"),
586            Self::TsVector => f.write_str("TSVECTOR"),
587            Self::TsQuery => f.write_str("TSQUERY"),
588            Self::Uuid => f.write_str("UUID"),
589            Self::Time => f.write_str("TIME"),
590            Self::Year => f.write_str("YEAR"),
591            Self::TimeTz => f.write_str("TIMETZ"),
592            Self::Money => f.write_str("MONEY"),
593            Self::Range(k) => f.write_str(k.keyword()),
594            Self::Hstore => f.write_str("HSTORE"),
595            Self::IntArray2D => f.write_str("INT[][]"),
596            Self::BigIntArray2D => f.write_str("BIGINT[][]"),
597            Self::TextArray2D => f.write_str("TEXT[][]"),
598            Self::BoolArray2D => f.write_str("BOOL[][]"),
599        }
600    }
601}
602
603/// v7.12.0 — one entry in a `Value::TsVector`. The lexeme is the
604/// (already-tokenised + stemmed in v7.12.1+) word; `positions` is
605/// a strictly-ascending list of 1-based positions; `weight` is the
606/// PG weight letter (A=3, B=2, C=1, D=0) — v7.12.0 defaults every
607/// lexeme to D, the v7.12.2 ranking path consumes the weight.
608#[derive(Debug, Clone, PartialEq, Eq)]
609pub struct TsLexeme {
610    pub word: String,
611    pub positions: Vec<u16>,
612    pub weight: u8,
613}
614
615/// v7.12.0 — parse tree for a PG `tsquery`. v7.12.0 ships the
616/// type + codec only; the `to_tsquery` / `plainto_tsquery` lexer
617/// lands in v7.12.1 and the `@@` evaluator in v7.12.2.
618#[derive(Debug, Clone, PartialEq, Eq)]
619pub enum TsQueryAst {
620    /// Single lexeme term. The `weight_mask` is the PG-style
621    /// bitmask of accepted weights (`A=1<<3`, `B=1<<2`, `C=1<<1`,
622    /// `D=1<<0`); `0` = any weight. v7.12.0 always sets it to 0.
623    Term {
624        word: String,
625        weight_mask: u8,
626    },
627    And(Box<TsQueryAst>, Box<TsQueryAst>),
628    Or(Box<TsQueryAst>, Box<TsQueryAst>),
629    Not(Box<TsQueryAst>),
630    /// `phrase <distance> phrase`. v7.12.0 only persists this; the
631    /// match semantics arrive in v7.12.2 alongside `@@`.
632    Phrase {
633        left: Box<TsQueryAst>,
634        right: Box<TsQueryAst>,
635        distance: u16,
636    },
637}
638
639/// A row-cell value, including SQL `NULL`. `Float` uses `f64`; NaN compares
640/// non-equal to itself (PG behaviour) — `PartialEq` is derived so callers
641/// must opt into NaN-aware comparison if they need stronger guarantees.
642///
643/// v7.37.42-arena Phase 1: parameterised on `'arena` so heap-bearing
644/// variants (Text/Json/Xml/Bytes/Vector/BitString.bytes) can borrow from
645/// a per-query bump arena (`Cow::Borrowed(&'arena ...)`). Persistent /
646/// catalog Values use `Value<'static>` (alias `ValueOwned`) with
647/// `Cow::Owned(...)`. Phase 1 keeps Range/Multirange recursive `Box<Value>`
648/// at `'static` (owned) — arena migration deferred to a later phase.
649/// Array-of-Option<String> variants (TextArray etc.) also stay owned in
650/// Phase 1; their nested shape is awkward for the simple Cow lift and the
651/// SCALARSQ hot path doesn't touch them.
652/// v7.38 (read01, T6) — the IEEE-style class of a NUMERIC value. `Finite` is the
653/// ordinary fixed-point case; the specials mirror PG's `'NaN'` / `'Infinity'` /
654/// `'-Infinity'`. Derived `PartialEq` gives `NaN == NaN` — correct for NUMERIC
655/// (unlike float's NaN ≠ NaN); the total order (`-Inf < finite < +Inf < NaN`)
656/// lives in the comparison paths, not in `Ord`.
657#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Hash)]
658pub enum NumericKind {
659    #[default]
660    Finite,
661    NaN,
662    PosInf,
663    NegInf,
664}
665
666#[derive(Debug, Clone, PartialEq)]
667#[non_exhaustive]
668pub enum Value<'arena> {
669    SmallInt(i16),
670    Int(i32),
671    BigInt(i64),
672    Float(f64),
673    /// v7.38 (read01, T-float4) — PG `real` (32-bit IEEE float).
674    Real(f32),
675    Text(Cow<'arena, str>),
676    Bool(bool),
677    Vector(Cow<'arena, [f32]>),
678    /// v6.0.1: 8-bit scalar-quantised vector cell. Lives in
679    /// columns declared `VECTOR(N) USING SQ8`. Layout per cell:
680    /// `Sq8Vector { min: f32, max: f32, bytes: Vec<u8> }` —
681    /// 4× compression vs `Vector(Vec<f32>)`. The wire layer
682    /// dequantises to `f32` on SELECT; INSERT path quantises
683    /// incoming `Vector(Vec<f32>)` cells into this variant.
684    Sq8Vector(crate::quantize::Sq8Vector),
685    /// v6.0.3: IEEE-754 binary16 vector cell. Lives in columns
686    /// declared `VECTOR(N) USING HALF`. Stores raw u16 LE bits
687    /// (2× compression vs `Vector(Vec<f32>)`). Wire / display
688    /// paths dequantise to f32 bit-exactly; INSERT path converts
689    /// incoming f32 vectors at the engine boundary.
690    HalfVector(crate::halfvec::HalfVector),
691    /// Exact fixed-point decimal. `scaled` holds the value as
692    /// `actual * 10^scale` so the storage type is always integral —
693    /// arithmetic never falls back to floating-point. v7.38 (read01, T6) —
694    /// `kind` classifies the value as finite (the common case, using
695    /// `scaled`/`scale`) or one of PG's NUMERIC specials (NaN / ±Infinity),
696    /// which ignore `scaled`/`scale` (canonicalized to 0).
697    Numeric {
698        scaled: i128,
699        /// v7.39 (round 271) — widened from u8. PG's numeric carries a
700        /// display scale up to 16383; at u8 a literal with 256 decimal
701        /// places could not be represented at all, and the conversion
702        /// aborted the query with an internal error.
703        scale: u16,
704        kind: NumericKind,
705    },
706    /// v7.38 (read01, T3) — an exact NUMERIC whose mantissa overflows `i128`
707    /// (PG's NUMERIC is unbounded). Boxed so the common finite case keeps its
708    /// small footprint; specials never take this form (they stay `Numeric`).
709    NumericBig(alloc::boxed::Box<crate::bignum::BigNumeric>),
710    /// Days since the Unix epoch (1970-01-01). Negative for earlier dates.
711    Date(i32),
712    /// Microseconds since the Unix epoch (1970-01-01T00:00:00Z).
713    Timestamp(i64),
714    /// Calendar span: `months` + `days` + `micros`. Three fields are
715    /// required for PG byte-equal: `'1 day'` ≠ `'24 hours'` (DST,
716    /// month-boundary, and the on-wire `pg_type` `interval` are all
717    /// `i64 micros + i32 days + i32 months`). v7.37.5 β widened from
718    /// `{months, micros}`; column storage lands in the same window.
719    Interval {
720        months: i32,
721        days: i32,
722        micros: i64,
723    },
724    /// v4.9 `JSON` — raw JSON text. No structural validation
725    /// happens at the storage layer; whatever the parser hands us
726    /// round-trips verbatim. Equality is byte-wise.
727    Json(Cow<'arena, str>),
728    /// v7.10.4 `BYTEA` — raw binary blob. Equality is byte-wise.
729    /// Layout matches `Text`'s length-prefixed shape (`[u32 LE
730    /// len][bytes]`) under tag 18; the engine accepts PG hex
731    /// literals (`'\xDEADBEEF'`) and escape literals at the
732    /// coercion boundary.
733    Bytes(Cow<'arena, [u8]>),
734    /// v7.10.9 `TEXT[]` — single-dimension TEXT array with
735    /// optional NULL elements. Equality is element-wise. PG's
736    /// NULL-element comparison semantics: NULL ≠ NULL inside
737    /// arrays under `=`, so `[NULL] != [NULL]` (the engine
738    /// honours this).
739    TextArray(Vec<Option<String>>),
740    /// v7.11.12 `INT[]` — single-dimension i32 array with optional
741    /// NULL elements. Codec mirrors TextArray with i32 LE per
742    /// element instead of length-prefixed UTF-8.
743    IntArray(Vec<Option<i32>>),
744    /// v7.11.12 `BIGINT[]` — single-dimension i64 array with optional
745    /// NULL elements.
746    BigIntArray(Vec<Option<i64>>),
747    /// v7.37.5 β-P4 `INTERVAL[]` — single-dimension array of
748    /// `IntervalSpan { months, days, micros }` with optional NULL
749    /// elements. PG external form quotes each non-NULL element
750    /// (`{"1 day","24:00:00",NULL}`) because interval text contains
751    /// spaces and colons. Storage codec follows the BigIntArray
752    /// shape with a 16-byte per-element body.
753    IntervalArray(Vec<Option<IntervalSpan>>),
754    /// v7.37.5 γ — single-dimension arrays of the remaining PG
755    /// scalar types. Each carries `Vec<Option<T>>` with the
756    /// scalar's natural Rust shape; element NULLs are first-class
757    /// (per PG: `{1,NULL,3}` is a 3-element array, not a 2-element
758    /// one). Codec follows the IntervalArray shape — `[u16 count]
759    /// [per elem: u8 null + (non-null) scalar body]`.
760    BoolArray(Vec<Option<bool>>),
761    SmallIntArray(Vec<Option<i16>>),
762    FloatArray(Vec<Option<f64>>),
763    /// PG `NUMERIC[]` — `(scaled: i128, scale: u16)` per element.
764    NumericArray(Vec<Option<(i128, u16)>>),
765    DateArray(Vec<Option<i32>>),
766    TimestampArray(Vec<Option<i64>>),
767    TimestamptzArray(Vec<Option<i64>>),
768    UuidArray(Vec<Option<[u8; 16]>>),
769    JsonArray(Vec<Option<String>>),
770    JsonbArray(Vec<Option<String>>),
771    BytesArray(Vec<Option<Vec<u8>>>),
772    VarcharArray(Vec<Option<String>>),
773    CharArray(Vec<Option<String>>),
774    /// v7.37.5 δ — PG 14+ multirange. `ranges` is a Vec of
775    /// non-overlapping bounds spans of the shared `kind`. PG's
776    /// canonical text form is `{[a,b),[c,d),...}` (comma-separated
777    /// ranges in braces; `{}` for the empty multirange). SPG's
778    /// constructor enforces no overlap/coalescing — for now the
779    /// engine trusts the caller (mirrors PG's `_construct_array`
780    /// pattern). Catalog tag 49 + 1-byte RangeKind on the dense
781    /// type-tag side; schema-less path is unreachable (multirange
782    /// is column-typed only).
783    Multirange {
784        kind: RangeKind,
785        ranges: Vec<RangeSpan>,
786    },
787    /// v7.37.5 ε — PG geometry scalars. Per-type Vec/struct shape;
788    /// codec body shape is described on the matching DataType
789    /// variant. PG canonical text forms:
790    ///   Point   `(x,y)`
791    ///   Lseg    `[(x1,y1),(x2,y2)]`
792    ///   Path    open `[(x,y),(x,y),...]` / closed `((x,y),(x,y),...)`
793    ///   Box     `(ux,uy),(lx,ly)` (PG normalises to upper-right + lower-left)
794    ///   Polygon `((x,y),(x,y),...)` (implicit closed)
795    ///   Line    `{a,b,c}` (Ax + By + C = 0)
796    ///   Circle  `<(x,y),r>`
797    Point(Point2D),
798    Lseg(Point2D, Point2D),
799    /// `closed = true` is `((p,p,...))`; `false` is `[(p,p,...)]`.
800    Path {
801        points: Vec<Point2D>,
802        closed: bool,
803    },
804    /// PG `box` — stored as `(upper_right, lower_left)` (PG's
805    /// normalised order). The engine accepts both endpoint
806    /// orderings at parse time and normalises here.
807    PgBox(Point2D, Point2D),
808    Polygon(Vec<Point2D>),
809    Line {
810        a: f64,
811        b: f64,
812        c: f64,
813    },
814    Circle {
815        center: Point2D,
816        radius: f64,
817    },
818    /// v7.37.5 ζ-A — PG `inet`. `family = 4` (IPv4) or `6` (IPv6).
819    /// `bits` is the netmask bit count (0..=32 for IPv4, 0..=128
820    /// for IPv6). `addr` is right-padded with zeros when family=4
821    /// (first 4 bytes are the address).
822    Inet {
823        family: u8,
824        bits: u8,
825        addr: [u8; 16],
826    },
827    /// v7.37.5 ζ-A — PG `cidr`. Same shape as Inet; CIDR's
828    /// invariant (host bits zero) is enforced at parse / coerce.
829    Cidr {
830        family: u8,
831        bits: u8,
832        addr: [u8; 16],
833    },
834    /// v7.37.5 ζ-A — PG `macaddr`. 6 bytes (XX:XX:XX:XX:XX:XX).
835    Macaddr([u8; 6]),
836    /// v7.37.5 ζ-A — PG `macaddr8`. 8 bytes (EUI-64).
837    Macaddr8([u8; 8]),
838    /// v7.39 (read01 pg_lsn.c) — PG `pg_lsn`, a 64-bit WAL location.
839    PgLsn(u64),
840    /// v7.39 (read01 ruleutils.c) — PG `regclass`: an OID-typed relation
841    /// reference that renders as the relation name. SPG carries BOTH
842    /// (the synthetic oid for catalog joins, the name for display) so
843    /// `conrelid = 't'::regclass` and `'t'::regclass::text` agree.
844    /// Eval-only (no column storage).
845    RegClass(i64, alloc::boxed::Box<str>),
846    /// v7.39 (round 342, V65) — PG `regproc`: an OID-typed FUNCTION
847    /// reference that renders as the function name. Same dual shape
848    /// [`Value::RegClass`] carries, and for the same reason: without the
849    /// oid half, `pg_proc.oid = 'f'::regproc` cannot join, and a callee
850    /// cannot tell `pg_get_functiondef('f'::regproc)` — which PG answers
851    /// — from `pg_get_functiondef('f')` — which PG rejects.
852    /// Eval-only (no column storage).
853    RegProc(i64, alloc::boxed::Box<str>),
854    /// v7.39 (round 648) — PG `regtype`: an OID-typed TYPE reference
855    /// that renders as the type name. The third of the shape
856    /// [`Value::RegClass`] and [`Value::RegProc`] carry, and the one
857    /// that was missing it: `::regtype` produced a plain `Value::Text`
858    /// holding the canonical name, so `'text'::regtype::oid` tried to
859    /// parse the NAME as a number and answered `invalid input syntax
860    /// for type oid: "text"` where PG answers 25. `pg_typeof` on one
861    /// said `text` rather than `regtype` for the same reason.
862    ///
863    /// Eval-only (no column storage).
864    RegType(i64, alloc::boxed::Box<str>),
865    /// v7.39 (round 512) — PG `xid` and `cid`, the transaction and command
866    /// ids the `xmin` / `xmax` / `cmin` / `cmax` system columns carry.
867    ///
868    /// Their own types rather than integers, because PG deliberately gives
869    /// them almost no operators: measured on PG18, `xmin + 1` is "operator
870    /// does not exist: xid + integer", `xmin > 0` likewise, `xmin::bigint`
871    /// is "cannot cast type xid to bigint", and there is no `max(xid)`.
872    /// Carrying them as BigInt would quietly allow all four.
873    ///
874    /// Eval-only (no column storage).
875    Xid(u32),
876    Cid(u32),
877    /// v7.39 (round 511) — PG `tid`, the physical row identity `ctid`
878    /// carries: a block number and a one-based offset inside it, rendered
879    /// `(block,offset)`.
880    ///
881    /// It is a real type rather than a two-field record because the idiom
882    /// that makes `ctid` worth having — `DELETE … WHERE ctid NOT IN (SELECT
883    /// min(ctid) … GROUP BY key)` — needs `min()` over it, and PG has no
884    /// `min(record)`. Ordering is by block then offset, so `(0,2) < (0,9) <
885    /// (0,10)`; a text form would order those `(0,10) < (0,2) < (0,9)` and
886    /// the dedup would keep the wrong row.
887    ///
888    /// Eval-only (no column storage).
889    Tid(u32, u32),
890    /// v7.37.5 ζ-A — PG `bit` / `bit varying`. `nbits` is the
891    /// actual bit count; `bytes` is the packed representation
892    /// (big-endian within each byte; final byte right-padded
893    /// with 0s if `nbits % 8 != 0`).
894    BitString {
895        nbits: u32,
896        bytes: Cow<'arena, [u8]>,
897    },
898    /// v7.37.5 ζ-A — PG `xml`. Stored verbatim as a string; no
899    /// parse-time validation (matches the SPG JSON convention).
900    Xml(Cow<'arena, str>),
901    /// v7.37.5 ζ-A — PG `"char"` (internal single-byte type,
902    /// distinct from CHAR(n)).
903    Char1(u8),
904    /// v7.38 (read01, T11) — PG `bpchar` / CHAR(n): blank-padded fixed-length
905    /// string. Stored space-padded to the declared width (as PG does + for wire
906    /// display); length / comparison / ::text / concat all ignore the trailing
907    /// blanks (handled at those sites).
908    BpChar(Cow<'arena, str>),
909    /// v7.37.5 ζ-A — PG `money[]`.
910    MoneyArray(Vec<Option<i64>>),
911    /// v7.12.0 `tsvector` — sorted-by-word, deduped lexeme set with
912    /// positions + weights. The engine enforces sort/dedup on
913    /// construction; consumers can rely on `lexemes.windows(2)`
914    /// being strictly ascending by `word`.
915    TsVector(Vec<TsLexeme>),
916    /// v7.12.0 `tsquery` — boolean / phrase parse tree over
917    /// lexemes. Engine builds via `to_tsquery` family.
918    TsQuery(TsQueryAst),
919    /// v7.17.0 `uuid` — 128-bit identifier. Stored as 16 bytes
920    /// (big-endian / network-byte order, same as RFC 4122).
921    /// Display normalises to canonical lowercase 8-4-4-4-12
922    /// hyphenated form. Equality is byte-wise.
923    Uuid([u8; 16]),
924    /// v7.17.0 Phase 3.P0-32 — PG `time` (without time zone) —
925    /// i64 microseconds since 00:00:00. Range 0..86_400_000_000.
926    /// Display: `HH:MM:SS` zero-padded, with optional `.ffffff`
927    /// suffix when fractional is non-zero.
928    Time(i64),
929    /// v7.17.0 Phase 3.P0-33 — MySQL `YEAR` — u16 in range
930    /// 1901..=2155 plus the special zero-year sentinel 0.
931    /// Display always 4 digits zero-padded (`0000` for the
932    /// sentinel; `1985`/`2007` otherwise).
933    Year(u16),
934    /// v7.17.0 Phase 3.P0-34 — PG `time with time zone` — i64
935    /// microseconds since 00:00:00 in the LOCAL wall clock PLUS
936    /// an i32 offset-from-UTC in seconds. PG preserves the
937    /// offset on output, so the wall-clock value is NOT shifted
938    /// to UTC at storage time. Offset range: ±50400 seconds
939    /// (±14 hours).
940    TimeTz {
941        us: i64,
942        offset_secs: i32,
943    },
944    /// v7.17.0 Phase 3.P0-35 — PG `money` — i64 cents
945    /// (locale-independent storage; the en_US locale renders on
946    /// display via `$N,NNN.CC`).
947    Money(i64),
948    /// v7.17.0 Phase 3.P0-39 — PG `hstore` value: flat
949    /// `text => text` map with NULL value support. Insertion
950    /// order preserved on input; duplicate keys take last-write-
951    /// wins at parse time.
952    Hstore(Vec<(String, Option<String>)>),
953    /// v7.17.0 Phase 3.P0-40 — 2D INT matrix (row-major).
954    IntArray2D(Vec<Vec<Option<i32>>>),
955    /// v7.17.0 Phase 3.P0-40 — 2D BIGINT matrix (row-major).
956    BigIntArray2D(Vec<Vec<Option<i64>>>),
957    /// v7.17.0 Phase 3.P0-40 — 2D TEXT matrix (row-major).
958    TextArray2D(Vec<Vec<Option<String>>>),
959    /// v7.39 (read01 round 75) — see `DataType::BoolArray2D`.
960    BoolArray2D(Vec<Vec<Option<bool>>>),
961    /// v7.17.0 Phase 3.P0-38 — PG range value. One shape covers
962    /// all six builtin range types; `kind` pins the element type
963    /// (must match the column's `DataType::Range(kind)`).
964    /// `lower` / `upper` are `None` for the unbounded sides;
965    /// `lower_inc` / `upper_inc` mirror the canonical PG
966    /// `[` / `(` / `]` / `)` bracket inclusivity. `empty=true`
967    /// supersedes all other fields (the empty range has no
968    /// bounds).
969    Range {
970        kind: RangeKind,
971        // v7.37.42-arena Phase 1: Range bounds stay owned ('static).
972        // Recursive arena lifetimes are awkward to migrate at this
973        // phase and the SCALARSQ hot path doesn't construct ranges.
974        lower: Option<alloc::boxed::Box<Value<'static>>>,
975        upper: Option<alloc::boxed::Box<Value<'static>>>,
976        lower_inc: bool,
977        upper_inc: bool,
978        empty: bool,
979    },
980    /// v7.38 (read01, T9) — a composite / record value (a `row(...)`
981    /// constructor or a whole-row reference). Fields are `(name, value)`; the
982    /// names are `f1..fN` for an anonymous `row(...)` or the source column
983    /// names for a table row. Transient — flows through row_to_json / to_json
984    /// and the composite text form `(a,b)`; not a storable column type here.
985    Composite(alloc::vec::Vec<(alloc::string::String, Value<'static>)>),
986    Null,
987}
988
989/// Owned `Value` — heap-bearing variants are `Cow::Owned`. Used everywhere
990/// a Value must outlive a query-scoped arena (catalog defaults, persistent
991/// storage, public APIs).
992pub type ValueOwned = Value<'static>;
993
994/// v7.37.5 ε — PG `point` building block. Shared by every other
995/// geometric type (lseg / path / box / polygon / circle all
996/// reduce to compositions of `Point2D`). Packed `{x: f64, y: f64}`,
997/// 16 B, on-disk LE field order matches the PG binary point
998/// format byte-for-byte (so a future binary BIND path lands
999/// without rearrangement).
1000#[derive(Debug, Clone, Copy, PartialEq)]
1001pub struct Point2D {
1002    pub x: f64,
1003    pub y: f64,
1004}
1005
1006/// v7.37.5 δ — single-range bounds without the kind tag. Used as
1007/// the element type of `Value::Multirange { kind, ranges }` so a
1008/// multirange carries one shared `RangeKind` plus N bounds-only
1009/// spans (saves 1 byte/elem vs duplicating the kind). The five
1010/// other fields mirror `Value::Range` exactly.
1011#[derive(Debug, Clone, PartialEq)]
1012pub struct RangeSpan {
1013    // v7.37.42-arena Phase 1: stays owned ('static) — same rationale as
1014    // Range bounds above.
1015    pub lower: Option<alloc::boxed::Box<Value<'static>>>,
1016    pub upper: Option<alloc::boxed::Box<Value<'static>>>,
1017    pub lower_inc: bool,
1018    pub upper_inc: bool,
1019    pub empty: bool,
1020}
1021
1022/// v7.37.5 β-P4 — element type for `Value::IntervalArray`. Mirrors
1023/// the `{months, days, micros}` shape of scalar `Value::Interval`,
1024/// broken out as a named struct so `IntervalArray`'s element type
1025/// is concrete (24 bytes, packed) instead of an enum-boxed Value.
1026/// All three dimensions are independent — `IntervalSpan { days: 1,
1027/// .. }` is distinct from `IntervalSpan { micros: 86_400_000_000,
1028/// .. }` per PG byte-equal.
1029#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1030pub struct IntervalSpan {
1031    pub months: i32,
1032    pub days: i32,
1033    pub micros: i64,
1034}
1035
1036impl<'arena> Value<'arena> {
1037    /// Type tag, or `None` for `NULL` (unknown at value level).
1038    pub fn data_type(&self) -> Option<DataType> {
1039        match self {
1040            Self::SmallInt(_) => Some(DataType::SmallInt),
1041            Self::Int(_) => Some(DataType::Int),
1042            Self::BigInt(_) => Some(DataType::BigInt),
1043            Self::Float(_) => Some(DataType::Float),
1044            Self::Real(_) => Some(DataType::Real),
1045            // `Text` covers both unbounded TEXT and bounded VARCHAR/CHAR
1046            // — the constraint lives on the column schema, not the value.
1047            Self::Text(_) => Some(DataType::Text),
1048            Self::Bool(_) => Some(DataType::Bool),
1049            Self::Vector(v) => Some(DataType::Vector {
1050                dim: u32::try_from(v.len()).expect("vector dim ≤ u32"),
1051                encoding: VecEncoding::F32,
1052            }),
1053            Self::Sq8Vector(q) => Some(DataType::Vector {
1054                dim: u32::try_from(q.bytes.len()).expect("vector dim ≤ u32"),
1055                encoding: VecEncoding::Sq8,
1056            }),
1057            Self::HalfVector(h) => Some(DataType::Vector {
1058                dim: u32::try_from(h.dim()).expect("vector dim ≤ u32"),
1059                encoding: VecEncoding::F16,
1060            }),
1061            // `Value::Numeric` doesn't carry its precision (the column
1062            // schema does); we surface precision=0 as "unknown" and let
1063            // the engine reconcile against the column type at coercion
1064            // time.
1065            // v7.39 (round 273) — a VALUE's display scale is unsigned and
1066            // never exceeds PG's 16383 ceiling, so it always fits the
1067            // signed declared-scale field this describes itself with.
1068            Self::Numeric { scale, .. } => Some(DataType::Numeric {
1069                precision: 0,
1070                scale: i16::try_from(*scale).unwrap_or(i16::MAX),
1071            }),
1072            Self::NumericBig(b) => Some(DataType::Numeric {
1073                precision: 0,
1074                scale: i16::try_from(b.scale()).unwrap_or(i16::MAX),
1075            }),
1076            Self::Date(_) => Some(DataType::Date),
1077            Self::Timestamp(_) => Some(DataType::Timestamp),
1078            Self::Interval { .. } => Some(DataType::Interval),
1079            Self::Json(_) => Some(DataType::Json),
1080            Self::Bytes(_) => Some(DataType::Bytes),
1081            Self::TextArray(_) => Some(DataType::TextArray),
1082            Self::IntArray(_) => Some(DataType::IntArray),
1083            Self::BigIntArray(_) => Some(DataType::BigIntArray),
1084            Self::IntervalArray(_) => Some(DataType::IntervalArray),
1085            Self::BoolArray(_) => Some(DataType::BoolArray),
1086            Self::SmallIntArray(_) => Some(DataType::SmallIntArray),
1087            Self::FloatArray(_) => Some(DataType::FloatArray),
1088            Self::NumericArray(_) => Some(DataType::NumericArray),
1089            Self::DateArray(_) => Some(DataType::DateArray),
1090            Self::TimestampArray(_) => Some(DataType::TimestampArray),
1091            Self::TimestamptzArray(_) => Some(DataType::TimestamptzArray),
1092            Self::UuidArray(_) => Some(DataType::UuidArray),
1093            Self::JsonArray(_) => Some(DataType::JsonArray),
1094            Self::JsonbArray(_) => Some(DataType::JsonbArray),
1095            Self::BytesArray(_) => Some(DataType::BytesArray),
1096            Self::VarcharArray(_) => Some(DataType::VarcharArray),
1097            Self::CharArray(_) => Some(DataType::CharArray),
1098            Self::Multirange { kind, .. } => Some(DataType::Multirange(*kind)),
1099            Self::Point(_) => Some(DataType::Point),
1100            Self::Lseg(_, _) => Some(DataType::Lseg),
1101            Self::Path { .. } => Some(DataType::Path),
1102            Self::PgBox(_, _) => Some(DataType::PgBox),
1103            Self::Polygon(_) => Some(DataType::Polygon),
1104            Self::Line { .. } => Some(DataType::Line),
1105            Self::Circle { .. } => Some(DataType::Circle),
1106            Self::Inet { .. } => Some(DataType::Inet),
1107            Self::Cidr { .. } => Some(DataType::Cidr),
1108            Self::Macaddr(_) => Some(DataType::Macaddr),
1109            Self::Macaddr8(_) => Some(DataType::Macaddr8),
1110            Self::PgLsn(_) => Some(DataType::PgLsn),
1111            // BitString could be either Bit or BitVarying; column
1112            // schema decides. Default to BitVarying when called
1113            // schema-less (rare; storage path is always
1114            // schema-aware so this only matters for diagnostics).
1115            Self::BitString { .. } => Some(DataType::BitVarying(0)),
1116            Self::Xml(_) => Some(DataType::Xml),
1117            Self::Char1(_) => Some(DataType::Char1),
1118            // BpChar reports its declared width from the padded length.
1119            Self::BpChar(s) => Some(DataType::Char(
1120                u32::try_from(s.chars().count()).unwrap_or(0),
1121            )),
1122            Self::MoneyArray(_) => Some(DataType::MoneyArray),
1123            Self::TsVector(_) => Some(DataType::TsVector),
1124            Self::TsQuery(_) => Some(DataType::TsQuery),
1125            Self::Uuid(_) => Some(DataType::Uuid),
1126            Self::Time(_) => Some(DataType::Time),
1127            Self::Year(_) => Some(DataType::Year),
1128            Self::TimeTz { .. } => Some(DataType::TimeTz),
1129            Self::Money(_) => Some(DataType::Money),
1130            Self::Range { kind, .. } => Some(DataType::Range(*kind)),
1131            Self::Hstore(_) => Some(DataType::Hstore),
1132            Self::IntArray2D(_) => Some(DataType::IntArray2D),
1133            Self::BigIntArray2D(_) => Some(DataType::BigIntArray2D),
1134            Self::TextArray2D(_) => Some(DataType::TextArray2D),
1135            Self::BoolArray2D(_) => Some(DataType::BoolArray2D),
1136            // v7.38 (read01, T9) — a transient composite/record has no storable
1137            // column DataType (it flows through row_to_json / to_json).
1138            Self::Composite(_) => None,
1139            // v7.39 (read01 ruleutils.c) — regclass is eval-only (dual
1140            // oid+name shape); no column storage type.
1141            // v7.39 (round 640) — `xid` became a column type, so its value
1142            // has a DataType to answer with. `cid` and `tid` are equally
1143            // legal column types on PG (measured: `CREATE TABLE t (a cid,
1144            // b tid)` is accepted), but SPG's grammar has no keyword for
1145            // them yet; they stay eval-only rather than half-declared.
1146            Self::Xid(_) => Some(DataType::Xid),
1147            Self::RegClass(..)
1148            | Self::RegProc(..)
1149            | Self::RegType(..)
1150            | Self::Tid(..)
1151            | Self::Cid(_) => None,
1152            Self::Null => None,
1153        }
1154    }
1155
1156    pub const fn is_null(&self) -> bool {
1157        matches!(self, Self::Null)
1158    }
1159
1160    /// v7.37.42-arena Phase 1: lift any `Value<'arena>` (possibly
1161    /// borrowing from a bump arena) into a fully-owned `Value<'static>`.
1162    /// Used at boundaries that must outlive the per-query arena
1163    /// (catalog write, public QueryResult emit, sqlx materialise).
1164    ///
1165    /// For the recursive Range/Multirange variants — bounds are already
1166    /// `Box<Value<'static>>` per Phase 1 design, so we just rebuild the
1167    /// outer enum at `'static`.
1168    pub fn into_owned(self) -> Value<'static> {
1169        match self {
1170            Value::SmallInt(n) => Value::SmallInt(n),
1171            Value::Int(n) => Value::Int(n),
1172            Value::BigInt(n) => Value::BigInt(n),
1173            Value::Float(f) => Value::Float(f),
1174            Value::Real(f) => Value::Real(f),
1175            Value::Text(s) => Value::Text(Cow::Owned(s.into_owned())),
1176            Value::Bool(b) => Value::Bool(b),
1177            Value::Vector(v) => Value::Vector(Cow::Owned(v.into_owned())),
1178            Value::Sq8Vector(q) => Value::Sq8Vector(q),
1179            Value::HalfVector(h) => Value::HalfVector(h),
1180            Value::Numeric {
1181                scaled,
1182                scale,
1183                kind,
1184            } => Value::Numeric {
1185                scaled,
1186                scale,
1187                kind,
1188            },
1189            Value::NumericBig(b) => Value::NumericBig(b),
1190            Value::Date(d) => Value::Date(d),
1191            Value::Timestamp(t) => Value::Timestamp(t),
1192            Value::Interval {
1193                months,
1194                days,
1195                micros,
1196            } => Value::Interval {
1197                months,
1198                days,
1199                micros,
1200            },
1201            Value::Json(s) => Value::Json(Cow::Owned(s.into_owned())),
1202            Value::Bytes(b) => Value::Bytes(Cow::Owned(b.into_owned())),
1203            Value::TextArray(v) => Value::TextArray(v),
1204            Value::IntArray(v) => Value::IntArray(v),
1205            Value::BigIntArray(v) => Value::BigIntArray(v),
1206            Value::IntervalArray(v) => Value::IntervalArray(v),
1207            Value::BoolArray(v) => Value::BoolArray(v),
1208            Value::SmallIntArray(v) => Value::SmallIntArray(v),
1209            Value::FloatArray(v) => Value::FloatArray(v),
1210            Value::NumericArray(v) => Value::NumericArray(v),
1211            Value::DateArray(v) => Value::DateArray(v),
1212            Value::TimestampArray(v) => Value::TimestampArray(v),
1213            Value::TimestamptzArray(v) => Value::TimestamptzArray(v),
1214            Value::UuidArray(v) => Value::UuidArray(v),
1215            Value::JsonArray(v) => Value::JsonArray(v),
1216            Value::JsonbArray(v) => Value::JsonbArray(v),
1217            Value::BytesArray(v) => Value::BytesArray(v),
1218            Value::VarcharArray(v) => Value::VarcharArray(v),
1219            Value::CharArray(v) => Value::CharArray(v),
1220            Value::Multirange { kind, ranges } => Value::Multirange { kind, ranges },
1221            // v7.38 (read01, T9) — Composite fields are already `Value<'static>`.
1222            Value::Composite(fields) => Value::Composite(fields),
1223            Value::RegClass(oid, name) => Value::RegClass(oid, name),
1224            Value::Tid(b, o) => Value::Tid(b, o),
1225            Value::Xid(x) => Value::Xid(x),
1226            Value::Cid(c) => Value::Cid(c),
1227            Value::RegProc(oid, name) => Value::RegProc(oid, name),
1228            Value::RegType(oid, name) => Value::RegType(oid, name),
1229            Value::Point(p) => Value::Point(p),
1230            Value::Lseg(a, b) => Value::Lseg(a, b),
1231            Value::Path { points, closed } => Value::Path { points, closed },
1232            Value::PgBox(a, b) => Value::PgBox(a, b),
1233            Value::Polygon(p) => Value::Polygon(p),
1234            Value::Line { a, b, c } => Value::Line { a, b, c },
1235            Value::Circle { center, radius } => Value::Circle { center, radius },
1236            Value::Inet { family, bits, addr } => Value::Inet { family, bits, addr },
1237            Value::Cidr { family, bits, addr } => Value::Cidr { family, bits, addr },
1238            Value::Macaddr(m) => Value::Macaddr(m),
1239            Value::Macaddr8(m) => Value::Macaddr8(m),
1240            Value::PgLsn(l) => Value::PgLsn(l),
1241            Value::BitString { nbits, bytes } => Value::BitString {
1242                nbits,
1243                bytes: Cow::Owned(bytes.into_owned()),
1244            },
1245            Value::Xml(s) => Value::Xml(Cow::Owned(s.into_owned())),
1246            Value::Char1(c) => Value::Char1(c),
1247            Value::BpChar(s) => Value::BpChar(Cow::Owned(s.into_owned())),
1248            Value::MoneyArray(v) => Value::MoneyArray(v),
1249            Value::TsVector(v) => Value::TsVector(v),
1250            Value::TsQuery(q) => Value::TsQuery(q),
1251            Value::Uuid(u) => Value::Uuid(u),
1252            Value::Time(t) => Value::Time(t),
1253            Value::Year(y) => Value::Year(y),
1254            Value::TimeTz { us, offset_secs } => Value::TimeTz { us, offset_secs },
1255            Value::Money(m) => Value::Money(m),
1256            Value::Range {
1257                kind,
1258                lower,
1259                upper,
1260                lower_inc,
1261                upper_inc,
1262                empty,
1263            } => Value::Range {
1264                kind,
1265                lower,
1266                upper,
1267                lower_inc,
1268                upper_inc,
1269                empty,
1270            },
1271            Value::Hstore(h) => Value::Hstore(h),
1272            Value::IntArray2D(a) => Value::IntArray2D(a),
1273            Value::BigIntArray2D(a) => Value::BigIntArray2D(a),
1274            Value::TextArray2D(a) => Value::TextArray2D(a),
1275            Value::BoolArray2D(a) => Value::BoolArray2D(a),
1276            Value::Null => Value::Null,
1277        }
1278    }
1279
1280    /// v7.37.42-arena Phase 4 — copy heap payloads into the supplied
1281    /// bump arena, yielding a `Value<'a>` whose Cow-variant payloads
1282    /// are arena-borrowed (or stay as small owned scalars for the
1283    /// `Copy`-able variants).
1284    ///
1285    /// Used at the catalog ↔ ephemeral boundary: a `ColumnSchema.default`
1286    /// is `Value<'static>` but INSERT-time eval may want it stamped into
1287    /// the per-statement arena alongside other arena-built scalars.
1288    ///
1289    /// Allocates only into the supplied arena; the input `&self` keeps
1290    /// its own storage. For `Copy`-able / nested-owned variants the
1291    /// implementation falls back to `clone()` (the nested heap blocks
1292    /// stay on the global allocator, which is fine — the boundary
1293    /// requirement is just "no aliasing of caller-owned strings").
1294    pub fn clone_into<'a>(&self, arena: &'a bumpalo::Bump) -> Value<'a> {
1295        match self {
1296            Value::Text(s) => Value::Text(Cow::Borrowed(arena.alloc_str(s))),
1297            Value::Json(s) => Value::Json(Cow::Borrowed(arena.alloc_str(s))),
1298            Value::Xml(s) => Value::Xml(Cow::Borrowed(arena.alloc_str(s))),
1299            Value::BpChar(s) => Value::BpChar(Cow::Borrowed(arena.alloc_str(s))),
1300            Value::Bytes(b) => {
1301                let slot = arena.alloc_slice_copy::<u8>(b);
1302                Value::Bytes(Cow::Borrowed(slot))
1303            }
1304            Value::Vector(v) => {
1305                let slot = arena.alloc_slice_copy::<f32>(v);
1306                Value::Vector(Cow::Borrowed(slot))
1307            }
1308            Value::BitString { nbits, bytes } => {
1309                let slot = arena.alloc_slice_copy::<u8>(bytes);
1310                Value::BitString {
1311                    nbits: *nbits,
1312                    bytes: Cow::Borrowed(slot),
1313                }
1314            }
1315            // Copy-able scalars + variants whose nested heap blocks are
1316            // `'static` regardless of `'arena` (TextArray, JsonArray,
1317            // Hstore, TsVector, Range bounds, …). Clone the heap block
1318            // via the standard `into_owned()` path then lift the
1319            // resulting `Value<'static>` to `Value<'a>` via the Cow
1320            // variance — `'static` covers any lifetime.
1321            other => other.clone().into_owned(),
1322        }
1323    }
1324}
1325
1326impl Value<'static> {
1327    /// v7.37.42-arena Phase 1 — owned-Text constructor. The variant now
1328    /// holds `Cow<'arena, str>`, so the previous `Value::Text(String)`
1329    /// shape no longer compiles directly. This helper preserves the
1330    /// historical ergonomics: `Value::text("foo")` or
1331    /// `Value::text(String::from("foo"))`.
1332    pub fn text<S: Into<String>>(s: S) -> Self {
1333        Value::Text(Cow::Owned(s.into()))
1334    }
1335
1336    /// v7.38 (read01, T6) — a finite NUMERIC from its fixed-point parts.
1337    pub const fn numeric(scaled: i128, scale: u16) -> Self {
1338        Value::Numeric {
1339            scaled,
1340            scale,
1341            kind: NumericKind::Finite,
1342        }
1343    }
1344
1345    /// v7.38 (read01, T6) — a special NUMERIC (NaN / ±Infinity). The fixed-point
1346    /// fields are canonicalized to 0 so equal specials compare byte-identical.
1347    pub const fn numeric_special(kind: NumericKind) -> Self {
1348        Value::Numeric {
1349            scaled: 0,
1350            scale: 0,
1351            kind,
1352        }
1353    }
1354
1355    /// v7.37.42-arena Phase 1 — owned-Json constructor (mirrors `text`).
1356    pub fn json<S: Into<String>>(s: S) -> Self {
1357        Value::Json(Cow::Owned(s.into()))
1358    }
1359
1360    /// v7.37.42-arena Phase 1 — owned-Xml constructor.
1361    pub fn xml<S: Into<String>>(s: S) -> Self {
1362        Value::Xml(Cow::Owned(s.into()))
1363    }
1364
1365    /// v7.37.42-arena Phase 1 — owned-Bytes constructor.
1366    pub fn bytes<B: Into<Vec<u8>>>(b: B) -> Self {
1367        Value::Bytes(Cow::Owned(b.into()))
1368    }
1369
1370    /// v7.37.42-arena Phase 1 — owned-Vector constructor.
1371    pub fn vector<V: Into<Vec<f32>>>(v: V) -> Self {
1372        Value::Vector(Cow::Owned(v.into()))
1373    }
1374
1375    /// v7.37.42-arena Phase 1 — owned-BitString constructor.
1376    pub fn bit_string<B: Into<Vec<u8>>>(nbits: u32, bytes: B) -> Self {
1377        Value::BitString {
1378            nbits,
1379            bytes: Cow::Owned(bytes.into()),
1380        }
1381    }
1382}
1383
1384/// One table row — values are positional and must match
1385/// `TableSchema.columns` in length and (modulo NULL) in `DataType`.
1386///
1387/// v7.37.42-arena Phase 1: parameterised on `'arena` so per-query rows
1388/// can borrow from a bump arena. The owned shape (`Row<'static>`, alias
1389/// `RowOwned`) is what catalog storage, public APIs, and tests use.
1390#[derive(Debug, Clone, PartialEq)]
1391pub struct Row<'arena> {
1392    pub values: Vec<Value<'arena>>,
1393}
1394
1395/// Owned `Row` — values are `Value<'static>`. Used everywhere a row must
1396/// outlive a query-scoped arena.
1397pub type RowOwned = Row<'static>;
1398
1399impl<'arena> Row<'arena> {
1400    pub const fn new(values: Vec<Value<'arena>>) -> Self {
1401        Self { values }
1402    }
1403
1404    pub fn len(&self) -> usize {
1405        self.values.len()
1406    }
1407
1408    pub fn is_empty(&self) -> bool {
1409        self.values.is_empty()
1410    }
1411}
1412
1413impl<'arena> Row<'arena> {
1414    /// v7.37.42-arena Phase 4 — copy every cell into the supplied bump
1415    /// arena, yielding a `Row<'a>` whose Cow-payloads are arena-borrowed.
1416    /// Boundary helper for catalog defaults → DML eval handoff and
1417    /// arena-local row scratch.
1418    pub fn clone_into<'a>(&self, arena: &'a bumpalo::Bump) -> Row<'a> {
1419        Row {
1420            values: self.values.iter().map(|v| v.clone_into(arena)).collect(),
1421        }
1422    }
1423
1424    /// v7.37.42-arena Phase 4 — lift this `Row<'arena>` to a fully-owned
1425    /// `Row<'static>` for catalog write / WAL serialisation. Equivalent
1426    /// to `Row::from_arena(self)` but consumes by value at any lifetime
1427    /// (callers can write `row.into_owned()` mirroring `Value::into_owned`).
1428    pub fn into_owned(self) -> Row<'static> {
1429        Row {
1430            values: self.values.into_iter().map(Value::into_owned).collect(),
1431        }
1432    }
1433}
1434
1435impl Row<'static> {
1436    /// v7.37.42-arena Phase 1 — lift any `Row<'arena>` (possibly arena-
1437    /// borrowed) into a fully-owned `Row<'static>`. Mirrors
1438    /// `Value::into_owned`.
1439    pub fn from_arena(row: Row<'_>) -> Self {
1440        Self {
1441            values: row.values.into_iter().map(Value::into_owned).collect(),
1442        }
1443    }
1444}
1445
1446/// Each bool is an independent, separately-persisted column attribute
1447/// (`nullable`, `auto_increment`, `is_unsigned`, `identity_always`) that the
1448/// catalog appendix reads and writes by name. Packing them into a bitflags
1449/// word would buy nothing and would put a decoding step between the on-disk
1450/// format and every reader of the schema.
1451#[allow(clippy::struct_excessive_bools)]
1452#[derive(Debug, Clone, PartialEq)]
1453pub struct ColumnSchema {
1454    pub name: String,
1455    pub ty: DataType,
1456    pub nullable: bool,
1457    /// Optional `DEFAULT` value, frozen at CREATE TABLE time. `None`
1458    /// means "no default" (so omitted columns become NULL, or error
1459    /// out when the column is NOT NULL). Literal defaults take this
1460    /// path.
1461    ///
1462    /// v7.37.42-arena Phase 1: explicitly `Value<'static>` — catalog
1463    /// defaults must outlive any per-query arena.
1464    pub default: Option<Value<'static>>,
1465    /// v7.9.21 — for DEFAULT expressions that need INSERT-time
1466    /// evaluation (e.g. `DEFAULT now()`, `DEFAULT CURRENT_TIMESTAMP`),
1467    /// the Display form of the expression. The engine re-parses
1468    /// it on each INSERT default-fill, evaluates against an empty
1469    /// row context, and coerces to the column type. mailrs G4.
1470    /// Persisted in catalog FILE_VERSION 15+; older catalogs
1471    /// deserialise with None.
1472    pub runtime_default: Option<String>,
1473    /// MySQL-style `AUTO_INCREMENT`. When set, an INSERT that leaves
1474    /// this column unbound (or sets it to NULL) gets the next integer
1475    /// computed from the column's current max + 1.
1476    /// v7.39 (round 676) — the collation NAME as written, when the column
1477    /// carried an explicit `COLLATE`.
1478    ///
1479    /// `spg_sql::Collation` cannot carry it: it is a two-variant MySQL enum
1480    /// and `from_collation_name` folds `C`, `POSIX`, `en_US` and `default`
1481    /// all into `Binary`. Without the name `pg_attribute.attcollation` can
1482    /// only ever report the type's default, which is what F36 records as
1483    /// "the declaration is taken and ignored".
1484    ///
1485    /// None means the column was written without a `COLLATE` clause and
1486    /// takes its type's collation. Persisted through the v88 appendix,
1487    /// which costs two bytes for a table that declares none.
1488    pub collation_name: Option<String>,
1489    pub auto_increment: bool,
1490    /// v7.17.0 Phase 1.4 — when the column is bound to a user-
1491    /// defined ENUM type (the parser saw an unknown type ident
1492    /// and the engine resolved it against `catalog.enum_types`),
1493    /// this carries the enum name so INSERT/UPDATE can validate
1494    /// the cell value against the enum's labels. `ty` is
1495    /// `DataType::Text` in that case. Persisted in catalog
1496    /// FILE_VERSION 29+; older catalogs deserialise with None.
1497    pub user_enum_type: Option<String>,
1498    /// v7.17.0 Phase 1.5 — when the column is bound to a user-
1499    /// defined DOMAIN (the parser saw an unknown type ident and
1500    /// the engine resolved it against `catalog.domain_types`),
1501    /// this carries the domain name. `ty` is the domain's base
1502    /// type; INSERT/UPDATE re-evaluates the domain's CHECK list
1503    /// + NOT NULL against the cell value. Persisted in catalog
1504    /// FILE_VERSION 30+; older catalogs deserialise with None.
1505    pub user_domain_type: Option<String>,
1506    /// v7.39 (read01 round 56) — when the column is bound to a user-defined
1507    /// COMPOSITE type. `ty` stays `DataType::Jsonb` (the on-disk form), but the
1508    /// engine REHYDRATES the stored JSON into a `Value::Composite` on read, so
1509    /// field access `(p).x`, `= ROW(…)`, ordering and the canonical `(2,b)`
1510    /// text form all work — they were already implemented on Value::Composite;
1511    /// what was missing was that the column never recorded WHICH composite type
1512    /// it holds (this field's doc comment existed for two releases, the field
1513    /// itself did not). Persisted in the composite-column appendix
1514    /// (FILE_VERSION 63+); older catalogs deserialise with None.
1515    pub user_composite_type: Option<String>,
1516    /// v7.39 (read01 round 59) — column-level privileges (PG
1517    /// `pg_attribute.attacl`). `GRANT SELECT (pub) ON t TO dan` lands here and
1518    /// does NOT touch the table's `relacl`. Empty = no column grant, which is
1519    /// every column until one is made.
1520    pub acl: Vec<AclItem>,
1521    /// v7.17.0 Phase 2.1 — MySQL `ON UPDATE CURRENT_TIMESTAMP`
1522    /// column attribute. When `Some(expr_src)`, an UPDATE that
1523    /// does NOT bind this column overrides the new value with
1524    /// the engine-evaluated expression (always `now()` in
1525    /// v7.17.0). Stored as Display-form source so storage
1526    /// stays free of spg-sql; the engine re-parses at UPDATE
1527    /// time. Persisted in catalog FILE_VERSION 32+; older
1528    /// catalogs deserialise with None — preserves the existing
1529    /// "silent ignore" behaviour for snapshots written before
1530    /// the upgrade.
1531    pub on_update_runtime: Option<String>,
1532    /// v7.17.0 Phase 2.5 — text collation. Pre-2.5 SPG accepted
1533    /// `COLLATE <name>` clauses but discarded the name, so a
1534    /// column declared `COLLATE "case_insensitive"` (or any
1535    /// MySQL `_ci` collation) still compared byte-wise — a
1536    /// Tier-S silent failure where `WHERE name = 'foo'` never
1537    /// matched stored `'Foo'`. This carries the parser-derived
1538    /// classification so the engine's WHERE evaluator can route
1539    /// text equality through a case-aware compare. `Binary` (the
1540    /// default) preserves the prior byte-wise behaviour. Only
1541    /// CaseInsensitive lands in the catalog appendix — Binary
1542    /// columns stay implicit, keeping snapshots compact.
1543    /// Persisted in catalog FILE_VERSION 34+; older catalogs
1544    /// deserialise every column as `Binary`.
1545    pub collation: Collation,
1546    /// v7.17.0 Phase 4.4 — MySQL `UNSIGNED` modifier flag. Drives
1547    /// engine-side INSERT / UPDATE range enforcement (rejects
1548    /// negative values on UNSIGNED int columns). Pre-4.4 the
1549    /// parser consumed and discarded the keyword silently, so
1550    /// every UNSIGNED column quietly accepted negatives — a
1551    /// Tier-A correctness drift. Sparse: only UNSIGNED columns
1552    /// land in the catalog appendix; the default `false` keeps
1553    /// snapshots compact for the common signed-int path.
1554    /// Persisted in catalog FILE_VERSION 35+; older catalogs
1555    /// deserialise every column as `is_unsigned = false`.
1556    pub is_unsigned: bool,
1557    /// v7.17.0 Phase 3.P0-36 — MySQL inline `ENUM('a','b','c')`
1558    /// value list. Distinct from `user_enum_type` (which points
1559    /// to a separately CREATE TYPE'd PG enum); this carries the
1560    /// column-local list MySQL DDL declares inline. When `Some`,
1561    /// `ty` is `DataType::Text` and INSERT/UPDATE validates the
1562    /// cell value against this list. Variant ORDER is preserved
1563    /// (MySQL uses it for `ORDER BY col`). Sparse: only ENUM
1564    /// columns land in the catalog appendix.
1565    /// Persisted in catalog FILE_VERSION 41+; older catalogs
1566    /// deserialise with None — preserves silent-drop behaviour
1567    /// for snapshots written before P0-36.
1568    pub inline_enum_variants: Option<Vec<String>>,
1569    /// v7.17.0 Phase 3.P0-37 — MySQL inline `SET('a','b','c')`
1570    /// variant list. Storage is TEXT (canonical comma-joined in
1571    /// definition order, de-duplicated). INSERT/UPDATE validates
1572    /// every comma-separated token against this list. Sparse:
1573    /// only SET columns land in the catalog appendix.
1574    /// Persisted in catalog FILE_VERSION 42+; older catalogs
1575    /// deserialise with None.
1576    pub inline_set_variants: Option<Vec<String>>,
1577    /// v7.37.7(sentori Epic 3 P1)— `GENERATED ALWAYS AS (<expr>)
1578    /// STORED` computed-column source. When `Some`, INSERT / UPDATE
1579    /// recompute the cell against the candidate row(re-parse the
1580    /// stored Display form and evaluate)and overwrite any
1581    /// user-supplied value, matching PG's stored-generated-column
1582    /// semantics. `None` (the default) preserves the regular
1583    /// "column value is whatever the caller passed" path.
1584    /// Persisted in catalog FILE_VERSION 50+; older catalogs
1585    /// deserialise with None.
1586    pub generated_stored_expr: Option<String>,
1587    /// v7.38 (read01) — `GENERATED ALWAYS AS IDENTITY`. Both identity
1588    /// flavours set `auto_increment`; this additionally marks the ALWAYS
1589    /// flavour, whose explicit INSERT value PG rejects ("cannot insert a
1590    /// non-DEFAULT value into column …") unless `OVERRIDING SYSTEM VALUE`.
1591    /// `false` (serial / `BY DEFAULT`) keeps the permissive path. In-memory
1592    /// only for now — not yet in the catalog appendix, so a reloaded table
1593    /// deserialises as `false` (the pre-existing permissive behaviour).
1594    pub identity_always: bool,
1595    /// v7.38 (read01) — the DEFAULT expression's source text, deparsed to
1596    /// PG-compatible form at CREATE TABLE time (e.g. `0`, `(3 + 4)`,
1597    /// `'hi'::text`, `now()`, `CURRENT_DATE`). Distinct from `default`
1598    /// (the coerced value the INSERT path fills) and `runtime_default`
1599    /// (the recompute-per-row Display form): those lose the source
1600    /// spelling, so `information_schema.columns.column_default` /
1601    /// `pg_attrdef` / `pg_get_expr` reported the coerced render
1602    /// (`0.00` for `numeric(10,2) DEFAULT 0`) instead of PG's `0`.
1603    /// `None` for a column with no explicit default. Persisted in catalog
1604    /// FILE_VERSION 58+; older catalogs deserialise with None.
1605    pub default_text: Option<String>,
1606    /// v7.39 (round 220) — `ALTER TABLE … ALTER COLUMN … RESTART [WITH n]`
1607    /// on an identity column. SPG's identity allocation is a max+1 scan;
1608    /// this floor lifts the next allocated value to at least `n`
1609    /// (`max(max+1, n)`) — exactly what a dump-restore RESTART needs, and
1610    /// safer than PG for a backward RESTART (no duplicate-key landmine).
1611    /// Persisted in the FILE_VERSION 73+ sparse appendix; older catalogs
1612    /// deserialise with None.
1613    pub auto_restart: Option<i64>,
1614    /// v7.39 (read01 round 78) — this column is the ONLY column of a FROM item
1615    /// that calls a function returning a BASE type, so the item's row type IS
1616    /// this column: a whole-row reference collapses to the value
1617    /// (`SELECT j FROM jsonb_array_elements('[1]') AS j` → `1`, PG). Runtime
1618    /// only — a catalogued table column is never one, and it is not persisted.
1619    pub scalar_row_source: bool,
1620    /// v7.39 (round 386, type-fidelity epic P1) — the declared MySQL narrow
1621    /// integer width (TINYINT / MEDIUMINT) whose range the storage `ty`
1622    /// (SmallInt / Int) is too wide to enforce. `None` for every other
1623    /// column. Drives the epic-P2 write-path range check. Persisted in the
1624    /// FILE_VERSION 81+ sparse appendix; older catalogs deserialise as None.
1625    pub mysql_int_width: Option<MysqlIntWidth>,
1626    /// v7.39 (round 424, type-fidelity epic) — the declared MySQL
1627    /// fractional-seconds precision of a temporal column: `DATETIME(3)` is
1628    /// `Some(3)`, a BARE `DATETIME` / `TIME` / `TIMESTAMP` is `Some(0)`
1629    /// (MySQL's default is zero — the fraction is dropped on write), and
1630    /// `None` means "not a MySQL-declared temporal column", which is every
1631    /// PG column and leaves microsecond behaviour untouched.
1632    ///
1633    /// Drives write-path truncation (toward zero) and render padding
1634    /// (exactly this many digits, `.000` when the fraction is zero).
1635    /// Persisted in the FILE_VERSION 82+ sparse appendix; older catalogs
1636    /// deserialise as None.
1637    pub mysql_fsp: Option<u8>,
1638}
1639
1640/// v7.17.0 Phase 2.5 — column-level text collation. Drives the
1641/// engine's WHERE / GROUP BY equality routing for `Value::Text`.
1642/// Only two variants are modelled in v7.17:
1643///   * `Binary`  — byte-wise comparison (the SPG default;
1644///                 matches PG `COLLATE "C"` / `pg_catalog.default`
1645///                 and MySQL `*_bin`).
1646///   * `CaseInsensitive` — ASCII case-folded comparison (like
1647///                 MySQL `*_ci` collations; PG has NO built-in
1648///                 collation of this name — round-761 audit: a
1649///                 nondeterministic ICU collation must be CREATEd
1650///                 there first). Non-ASCII bytes
1651///                 still compare byte-wise; full ICU folding is
1652///                 out of v7.17 scope.
1653/// New variants append at the end — older catalogs read missing
1654/// columns as `Binary`.
1655#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1656pub enum Collation {
1657    Binary,
1658    CaseInsensitive,
1659}
1660
1661/// v7.39 (round 386, type-fidelity epic P1) — the declared MySQL narrow
1662/// integer type for a column whose storage `DataType` cannot express it.
1663/// MySQL `TINYINT` (i8, -128..127) collapses to `DataType::SmallInt` (i16)
1664/// and `MEDIUMINT` (24-bit) to `DataType::Int` (i32) — both wider than the
1665/// declared type, so a range check against `ty` alone accepts out-of-range
1666/// values (`INSERT 128 INTO TINYINT` is stored silently where MariaDB
1667/// strict raises ERROR 1264). This annotation records the lost width so the
1668/// write path (epic P2) can enforce the real bounds. `SMALLINT` / `INT` /
1669/// `BIGINT` need no marker — their storage `DataType` is already faithful.
1670/// Sparse: only TINYINT / MEDIUMINT columns carry it; persisted in the
1671/// FILE_VERSION 81+ appendix, older catalogs deserialise as None.
1672#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1673pub enum MysqlIntWidth {
1674    /// MySQL `TINYINT` — signed -128..127, unsigned 0..255. Storage i16.
1675    Tiny,
1676    /// MySQL `SMALLINT UNSIGNED` — 0..65535. Storage widened to i32 (a
1677    /// signed SMALLINT keeps `DataType::SmallInt` and carries no marker).
1678    Small,
1679    /// MySQL `MEDIUMINT` — signed -8388608..8388607, unsigned 0..16777215.
1680    /// Storage i32.
1681    Medium,
1682    /// MySQL `INT UNSIGNED` — 0..4294967295. Storage widened to i64 (a
1683    /// signed INT keeps `DataType::Int` and carries no marker).
1684    Int,
1685    /// v7.39 (round 471, epic P4b) — MySQL `BIGINT UNSIGNED` —
1686    /// 0..18446744073709551615. i64 stops at 2^63-1, so the storage tag is
1687    /// widened to `Numeric` (i128-backed, scale 0), which already compares,
1688    /// orders, indexes and renders as an exact integer. A signed BIGINT
1689    /// keeps `DataType::BigInt` and carries no marker.
1690    Big,
1691}
1692
1693/// v7.39 (round 363, M4 P1) — MySQL's default accent- and
1694/// case-insensitive fold (`utf8mb4_uca1400_ai_ci`).
1695///
1696/// This is the primitive M4 rests on: a session on the MySQL dialect
1697/// compares, groups, sorts and de-duplicates text by its FOLDED form, so
1698/// `Foo` = `foo` = `FOO` and, because the default collation is accent-
1699/// insensitive too, `Bär` = `bar`. The later stages (read path, then the
1700/// UNIQUE / index write path) all route through here so they cannot fold
1701/// differently from one another.
1702///
1703/// The fold is more than case + strip-combining: MariaDB EXPANDS some
1704/// letters — `ß` → `ss`, `æ` → `ae`, `œ` → `oe` — which is why the result
1705/// is built as a `String` rather than mapped char-for-char. Every mapping
1706/// below was measured on MariaDB 11 (`'Bär'='bar'` is 1, `'straße'=
1707/// 'strasse'` is 1, `'a'='æ'` is 0, `'s'='ß'` is 0). Characters with no
1708/// entry keep their lower-cased self, so ASCII and unknown scripts pass
1709/// through unchanged.
1710#[must_use]
1711pub fn mysql_ci_fold(s: &str) -> String {
1712    let mut out = String::with_capacity(s.len());
1713    for ch in s.chars() {
1714        // Lower-case first (`À` → `à`, `Æ` → `æ`), then fold the base.
1715        for lc in ch.to_lowercase() {
1716            match fold_latin_base(lc) {
1717                Some(base) => out.push_str(base),
1718                None => out.push(lc),
1719            }
1720        }
1721    }
1722    out
1723}
1724
1725/// v7.39 (round 375) — the fold used to COMPARE / GROUP / de-dup text on
1726/// the MySQL dialect. Its default collation is PAD SPACE: trailing spaces
1727/// do not affect a comparison (`'a' = 'a '`, `'' = ' '`, measured on
1728/// MariaDB 11), so they are stripped before the case/accent fold. Only
1729/// literal spaces pad — a tab or other whitespace is significant — and
1730/// this is NOT used by `LIKE`, whose pattern treats a trailing space
1731/// literally.
1732pub fn mysql_compare_fold(s: &str) -> String {
1733    mysql_ci_fold(s.trim_end_matches(' '))
1734}
1735
1736/// The base letter(s) a lower-cased Latin character folds to, or `None`
1737/// when it is already a base / has no fold. Expansions (`ß` → `ss`) are
1738/// why this returns a string.
1739fn fold_latin_base(c: char) -> Option<&'static str> {
1740    Some(match c {
1741        'à' | 'á' | 'â' | 'ã' | 'ä' | 'å' | 'ā' | 'ă' | 'ą' => "a",
1742        'æ' => "ae",
1743        'ç' | 'ć' | 'č' | 'ĉ' | 'ċ' => "c",
1744        'ð' | 'ď' | 'đ' => "d",
1745        'è' | 'é' | 'ê' | 'ë' | 'ē' | 'ĕ' | 'ė' | 'ę' | 'ě' => "e",
1746        'ĝ' | 'ğ' | 'ġ' | 'ģ' => "g",
1747        'ì' | 'í' | 'î' | 'ï' | 'ĩ' | 'ī' | 'ĭ' | 'į' => "i",
1748        'ĵ' => "j",
1749        'ķ' => "k",
1750        'ł' | 'ĺ' | 'ļ' | 'ľ' => "l",
1751        'ñ' | 'ń' | 'ņ' | 'ň' => "n",
1752        'ò' | 'ó' | 'ô' | 'õ' | 'ö' | 'ø' | 'ō' | 'ŏ' | 'ő' => "o",
1753        'œ' => "oe",
1754        'ŕ' | 'ŗ' | 'ř' => "r",
1755        'ś' | 'š' | 'ŝ' | 'ş' => "s",
1756        'ß' => "ss",
1757        'ţ' | 'ť' | 'ŧ' => "t",
1758        'ù' | 'ú' | 'û' | 'ü' | 'ũ' | 'ū' | 'ŭ' | 'ů' | 'ű' | 'ų' => "u",
1759        'ý' | 'ÿ' => "y",
1760        'ź' | 'ž' | 'ż' => "z",
1761        _ => return None,
1762    })
1763}
1764
1765#[allow(clippy::derivable_impls)]
1766impl Default for Collation {
1767    fn default() -> Self {
1768        Self::Binary
1769    }
1770}
1771
1772impl Collation {
1773    /// Wire tag persisted in the FILE_VERSION 34+ catalog appendix.
1774    /// Stable: future variants append above the recognised range
1775    /// and unknown tags read back as `Binary` for forward-compat
1776    /// on rollback.
1777    pub const TAG_BINARY: u8 = 0;
1778    pub const TAG_CASE_INSENSITIVE: u8 = 1;
1779}
1780
1781/// v7.39 (RLS) — the command a policy applies to. `ALL` is the default and
1782/// covers every command; the others scope the policy to one statement kind.
1783/// Persisted as a single byte in the policy appendix (FILE_VERSION 59+).
1784#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1785pub enum PolicyCmd {
1786    All,
1787    Select,
1788    Insert,
1789    Update,
1790    Delete,
1791}
1792
1793impl PolicyCmd {
1794    /// PG `pg_policy.polcmd` single-char encoding.
1795    #[must_use]
1796    pub const fn as_pg_char(self) -> char {
1797        match self {
1798            Self::All => '*',
1799            Self::Select => 'r',
1800            Self::Insert => 'a',
1801            Self::Update => 'w',
1802            Self::Delete => 'd',
1803        }
1804    }
1805
1806    /// PG `pg_policies.cmd` word form.
1807    #[must_use]
1808    pub const fn as_pg_word(self) -> &'static str {
1809        match self {
1810            Self::All => "ALL",
1811            Self::Select => "SELECT",
1812            Self::Insert => "INSERT",
1813            Self::Update => "UPDATE",
1814            Self::Delete => "DELETE",
1815        }
1816    }
1817
1818    #[must_use]
1819    pub const fn to_wire_byte(self) -> u8 {
1820        match self {
1821            Self::All => 0,
1822            Self::Select => 1,
1823            Self::Insert => 2,
1824            Self::Update => 3,
1825            Self::Delete => 4,
1826        }
1827    }
1828
1829    #[must_use]
1830    pub const fn from_wire_byte(b: u8) -> Option<Self> {
1831        match b {
1832            0 => Some(Self::All),
1833            1 => Some(Self::Select),
1834            2 => Some(Self::Insert),
1835            3 => Some(Self::Update),
1836            4 => Some(Self::Delete),
1837            _ => None,
1838        }
1839    }
1840}
1841
1842/// v7.39 (RLS) — one `CREATE POLICY` object, stored per table. The `using_expr`
1843/// / `with_check_expr` hold the qualifying expression's `Display` form
1844/// (re-parsed and evaluated per row at enforcement time, exactly like
1845/// `TableSchema.checks`); `None` means the clause was absent. `roles` empty =
1846/// PUBLIC. Persisted in the policy appendix (FILE_VERSION 59+).
1847#[derive(Debug, Clone, PartialEq)]
1848pub struct PolicyDef {
1849    pub name: String,
1850    pub cmd: PolicyCmd,
1851    /// `true` = PERMISSIVE (default, OR-combined), `false` = RESTRICTIVE
1852    /// (AND-combined).
1853    pub permissive: bool,
1854    pub roles: Vec<String>,
1855    pub using_expr: Option<String>,
1856    pub with_check_expr: Option<String>,
1857}
1858
1859#[derive(Debug, Clone, PartialEq)]
1860pub struct TableSchema {
1861    pub name: String,
1862    pub columns: Vec<ColumnSchema>,
1863    /// v6.7.2 — per-table hot-tier byte budget override. `None`
1864    /// falls through to the global `SPG_HOT_TIER_BYTES` setting;
1865    /// `Some(n)` overrides it for this specific table. Set via
1866    /// `ALTER TABLE t SET hot_tier_bytes = X`. Persisted in
1867    /// catalog FILE_VERSION 11+.
1868    pub hot_tier_bytes: Option<u64>,
1869    /// v7.6.1 — FOREIGN KEY constraints declared on this table.
1870    /// Engine maintains this in lock-step with `spg-sql`'s parser
1871    /// AST; the storage layer carries the on-disk shape so a
1872    /// catalog snapshot round-trips without external mapping.
1873    /// Persisted in catalog FILE_VERSION 13+. Older catalogs
1874    /// deserialise with an empty vec.
1875    pub foreign_keys: Vec<ForeignKeyConstraint>,
1876    /// v7.9.19 — composite UNIQUE / PRIMARY KEY constraints
1877    /// declared at the table level. Each entry's leading column
1878    /// has a BTree index (created via the constraint), and INSERT
1879    /// path enforces the full-tuple uniqueness via a scan keyed
1880    /// by the leading column. Persisted in catalog FILE_VERSION
1881    /// 15+. Older catalogs (≤ 14) deserialise with an empty vec.
1882    pub uniqueness_constraints: Vec<UniquenessConstraint>,
1883    /// v7.39 (round 210) — `EXCLUDE` constraints declared at the table level.
1884    /// Enforced on INSERT/UPDATE by a full live-row scan re-checking each
1885    /// element's operator (no equality index can answer overlap). Persisted
1886    /// in catalog FILE_VERSION 72+; older catalogs deserialise with an empty
1887    /// vec.
1888    pub exclusion_constraints: Vec<ExclusionConstraint>,
1889    /// v7.13.0 — `CHECK (<expr>)` predicates declared on this
1890    /// table. Both column-level inline `CHECK (…)` and
1891    /// table-level `CHECK (…)` fold into this list. Each entry
1892    /// is the AST Expr's `Display` form, re-parsed on every
1893    /// INSERT/UPDATE and evaluated against the candidate row.
1894    /// A false / NULL result rejects the mutation (PG semantics).
1895    /// Persisted in catalog FILE_VERSION 23+. Older catalogs
1896    /// deserialise with an empty vec. v7.39 (read01 round 48) — each entry
1897    /// now carries the user's constraint name too (FILE_VERSION 60+).
1898    pub checks: Vec<CheckConstraint>,
1899    /// v7.37.6-B — declarative partition role(sentori Epic 2 P0).
1900    /// `None` = 普通表(后向兼容,< v49 catalog 默认 None)。
1901    /// `Some(Parent { … })` = `CREATE TABLE p (...) PARTITION BY RANGE (key_col)` 父表 —
1902    /// 父表自己 `rows` 永远空,INSERT 在引擎层路由到命中的 child。
1903    /// `Some(Range { … })` = `CREATE TABLE c PARTITION OF p FOR VALUES FROM (a) TO (b)` 范围子表。
1904    /// `Some(Default { … })` = `CREATE TABLE c PARTITION OF p DEFAULT` 兜底子表。
1905    /// 持久化于 FILE_VERSION 49+。
1906    pub partition_role: Option<PartitionRole>,
1907    /// v7.39 (RLS) — `CREATE POLICY` objects on this table, independent of the
1908    /// `row_security` flag (PG stores policies even on non-RLS tables; they
1909    /// only take effect once RLS is enabled). Persisted in the policy appendix
1910    /// (FILE_VERSION 59+). Older catalogs deserialise with an empty vec.
1911    pub policies: Vec<PolicyDef>,
1912    /// v7.39 (RLS) — `ALTER TABLE … ENABLE ROW LEVEL SECURITY`
1913    /// (PG `pg_class.relrowsecurity`). Fresh table = `false`.
1914    pub row_security: bool,
1915    /// v7.39 (RLS) — `ALTER TABLE … FORCE ROW LEVEL SECURITY`
1916    /// (PG `pg_class.relforcerowsecurity`); subjects the table owner to RLS
1917    /// too. Fresh table = `false`.
1918    pub force_row_security: bool,
1919    /// v7.39 (read01 round 57, ACL) — the role that owns this table: whoever
1920    /// ran CREATE TABLE (PG `pg_class.relowner`). The owner holds every
1921    /// privilege implicitly and is the only role that may ALTER / DROP it.
1922    /// `None` = an image written before FILE_VERSION 64, which predates roles
1923    /// entirely; those tables read back as owned by the login role.
1924    pub owner: Option<String>,
1925    /// v7.39 (read01 round 57, ACL) — explicit GRANTs on this table
1926    /// (PG `pg_class.relacl`). EMPTY means "never granted": PG leaves relacl
1927    /// NULL while only the owner's implicit privileges apply, and materialises
1928    /// the whole list — owner's default entry included — on the first GRANT.
1929    /// Once materialised it stays, even after every grant is revoked.
1930    pub acl: Vec<AclItem>,
1931}
1932
1933/// v7.39 (read01 round 57) — one PG `aclitem`: what `grantee` may do to a
1934/// table, and who granted it. Renders as `grantee=privs/grantor`, with an
1935/// EMPTY grantee meaning PUBLIC (`=r/owner`).
1936#[derive(Debug, Clone, PartialEq, Eq)]
1937pub struct AclItem {
1938    /// The role the privileges are held by. Empty string = PUBLIC.
1939    pub grantee: String,
1940    /// Bitmask over `priv_bits`: which privileges are held.
1941    pub privs: u16,
1942    /// Bitmask over `priv_bits`: which of them carry WITH GRANT OPTION
1943    /// (PG renders those with a trailing `*` — `r*`).
1944    pub grantable: u16,
1945    /// The role that ran the GRANT.
1946    pub grantor: String,
1947}
1948
1949/// v7.39 (read01 round 57) — the table-privilege bits, in PG's `aclitem`
1950/// rendering order (`arwdDxtm`). The order matters: `relacl` output is
1951/// byte-compared against PG.
1952pub mod priv_bits {
1953    pub const INSERT: u16 = 1 << 0; // a
1954    pub const SELECT: u16 = 1 << 1; // r
1955    pub const UPDATE: u16 = 1 << 2; // w
1956    pub const DELETE: u16 = 1 << 3; // d
1957    pub const TRUNCATE: u16 = 1 << 4; // D
1958    pub const REFERENCES: u16 = 1 << 5; // x
1959    pub const TRIGGER: u16 = 1 << 6; // t
1960    pub const MAINTAIN: u16 = 1 << 7; // m
1961    /// v7.39 (read01 round 60) — the non-table privileges. They share the
1962    /// bitmask because an aclitem is an aclitem whatever it hangs off; which
1963    /// bits are MEANINGFUL depends on the object (a sequence has r / w / U, a
1964    /// schema has U / C, a database has C / c / T).
1965    pub const USAGE: u16 = 1 << 8; // U
1966    pub const CREATE: u16 = 1 << 9; // C
1967    pub const CONNECT: u16 = 1 << 10; // c
1968    pub const TEMPORARY: u16 = 1 << 11; // T
1969    pub const EXECUTE: u16 = 1 << 12; // X
1970    /// Every TABLE privilege — what `GRANT ALL ON <table>` grants and what a
1971    /// table's owner holds.
1972    pub const ALL: u16 =
1973        INSERT | SELECT | UPDATE | DELETE | TRUNCATE | REFERENCES | TRIGGER | MAINTAIN;
1974    /// `GRANT ALL ON SEQUENCE` — PG renders a sequence owner's default as `rwU`.
1975    pub const ALL_SEQUENCE: u16 = SELECT | UPDATE | USAGE;
1976    /// `GRANT ALL ON SCHEMA` — `UC`.
1977    pub const ALL_SCHEMA: u16 = USAGE | CREATE;
1978    /// `GRANT ALL ON DATABASE` — `CTc`.
1979    pub const ALL_DATABASE: u16 = CREATE | CONNECT | TEMPORARY;
1980    /// `GRANT ALL ON FUNCTION` — just `X`.
1981    pub const ALL_FUNCTION: u16 = EXECUTE;
1982}
1983
1984/// v7.37.6-B — partition 三态(parent / range child / default child)。
1985#[derive(Debug, Clone, PartialEq, Eq)]
1986pub enum PartitionRole {
1987    Parent {
1988        kind: PartitionKind,
1989        /// 父表 columns 中 key 列的下标(单列 v7.37.6-B,
1990        /// `Vec` 为将来扩多列预留)。
1991        key_column_positions: Vec<usize>,
1992        /// `CREATE INDEX ON parent (…)` 的 Display-form 源串。
1993        /// child 创建时再 parse + 在 child 上 execute,这样 future
1994        /// child 也自动继承父表索引。fan-out 实施在引擎层。
1995        index_template_sources: Vec<String>,
1996    },
1997    Range {
1998        parent_name: String,
1999        /// 半开区间下界(`>=`,SQL `FROM (lower)`).
2000        lower: PartitionBound,
2001        /// 半开区间上界(`<`,SQL `TO (upper)`).
2002        upper: PartitionBound,
2003    },
2004    /// v7.37.16 (16.1) — LIST child:行属于本 child iff key ∈ values。
2005    /// `values` 在 child 创建时从 SQL `FOR VALUES IN (lit, …)` 求值;
2006    /// 跟 PG 一样,显式 NULL ∈ values 由 caller 单独处理(不在
2007    /// PartitionBound 内表达 NULL)。
2008    List {
2009        parent_name: String,
2010        values: Vec<PartitionBound>,
2011    },
2012    /// v7.39 (round 645) — PG 表继承的 CHILD:`CREATE TABLE c (…)
2013    /// INHERITS (p1, p2)`。跟分区 child 的三个本质区别(实测 PG18):
2014    ///   * 父表**自己有行**(分区父表永远空),所以父表的联合体要含自身;
2015    ///   * `INSERT INTO 父表` **不路由**到 child(分区会路由);
2016    ///   * `DROP TABLE 父表` 不带 CASCADE **报错**(分区父表连子表一起删)。
2017    /// 多父继承合法,故 `parent_names` 是 Vec;`pg_inherits.inhseqno`
2018    /// 正是父表在这个列表里的位置(1-based)。
2019    Inherits {
2020        parent_names: Vec<String>,
2021    },
2022    /// v7.37.16 (16.2) — HASH child:行属于本 child iff
2023    /// `pg_compatible_hash(key) mod modulus == remainder`。
2024    /// PG 强制 `0 ≤ remainder < modulus`;parser/DDL 层先 gate。
2025    Hash {
2026        parent_name: String,
2027        modulus: u32,
2028        remainder: u32,
2029    },
2030    Default {
2031        parent_name: String,
2032    },
2033}
2034
2035/// v7.37.6-B — 分区策略。
2036///
2037/// - `Range`:半开区间 `[lower, upper)`(v7.37.6-B 初始)
2038/// - `List` (v7.37.16):枚举集合 — 行属于 partition iff key ∈ children list
2039/// - `Hash` (v7.37.16):`hash(key) mod modulus == remainder`
2040#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2041pub enum PartitionKind {
2042    Range,
2043    List,
2044    Hash,
2045}
2046
2047/// v7.37.6-B — partition 边界 literal。
2048///
2049/// v7.37.6-B 仅 `TimestampTz`(i64 microseconds since epoch);
2050/// v7.37.16 (16.6) 加全 PG 内建可比类型,匹配 `Value` 的对应 variant
2051/// 以避免 LIST membership 比较时的类型转换。
2052///
2053/// `MinValue` / `MaxValue` 对应 SQL `MINVALUE` / `MAXVALUE`,仅
2054/// Range 策略有意义(LIST 无 minvalue/maxvalue 概念,HASH 不
2055/// 使用 PartitionBound)。
2056#[derive(Debug, Clone, PartialEq, Eq)]
2057pub enum PartitionBound {
2058    MinValue,
2059    MaxValue,
2060    TimestampTz(i64),
2061    /// v7.37.16 (16.6) — BIGINT partition key.
2062    BigInt(i64),
2063    /// v7.37.16 (16.6) — INTEGER partition key (also covers
2064    /// `SERIAL` since SPG decomposes it to INTEGER + sequence).
2065    Int(i32),
2066    /// v7.37.16 (16.6) — SMALLINT partition key.
2067    SmallInt(i16),
2068    /// v7.37.16 (16.6) — DATE partition key. Stored as days
2069    /// since the Unix epoch (matches `Value::Date`).
2070    Date(i32),
2071    /// v7.37.16 (16.6) — TEXT / VARCHAR partition key.
2072    Text(alloc::string::String),
2073}
2074
2075impl PartitionBound {
2076    /// v7.37.16 (16.6) — true iff this bound's underlying value
2077    /// equals `other`'s. Used for LIST partition membership
2078    /// checks. Returns false for `MinValue` / `MaxValue`
2079    /// (sentinels — never literal equality).
2080    #[must_use]
2081    pub fn equals_value(&self, other: &Value<'_>) -> bool {
2082        match (self, other) {
2083            (PartitionBound::TimestampTz(a), Value::Timestamp(b)) => a == b,
2084            (PartitionBound::BigInt(a), Value::BigInt(b)) => a == b,
2085            (PartitionBound::Int(a), Value::Int(b)) => a == b,
2086            (PartitionBound::SmallInt(a), Value::SmallInt(b)) => a == b,
2087            (PartitionBound::Date(a), Value::Date(b)) => a == b,
2088            (PartitionBound::Text(a), Value::Text(b)) => a.as_str() == b.as_ref(),
2089            _ => false,
2090        }
2091    }
2092}
2093
2094/// v7.9.19 — composite UNIQUE / PRIMARY KEY constraint persisted
2095/// on the table schema. The leading column always has a BTree
2096/// index (created at CREATE TABLE time); INSERT enforcement
2097/// scans that index for collisions on the full column tuple.
2098/// v7.39 (read01 round 48) — a `CHECK` constraint: the SQL name the user
2099/// gave it (via `ADD CONSTRAINT <name> CHECK (...)` or the inline
2100/// `CONSTRAINT <name> CHECK (...)` form) plus the predicate source. `None`
2101/// name = unnamed, in which case `pg_constraint` synthesises PG's
2102/// `<table>_<col>_check` form. Names are persisted in the constraint-name
2103/// appendix (FILE_VERSION 60+); older catalogs deserialise with `None`.
2104#[derive(Debug, Clone, PartialEq, Eq)]
2105pub struct CheckConstraint {
2106    pub name: Option<String>,
2107    /// The AST Expr's `Display` form, re-parsed on every INSERT/UPDATE.
2108    pub expr: String,
2109    /// v7.39 (round 652) — `false` for a constraint added `NOT VALID`: the
2110    /// rows already in the table were never scanned against it, and
2111    /// `pg_constraint.convalidated` says so. It does NOT weaken the check on
2112    /// new rows — INSERT and UPDATE enforce it either way, as in PG.
2113    /// `VALIDATE CONSTRAINT` does the deferred scan and flips it. Persisted
2114    /// by the FILE_VERSION 87 appendix; older catalogs deserialise as `true`,
2115    /// which is what every constraint they could hold actually was.
2116    pub validated: bool,
2117}
2118
2119#[derive(Debug, Clone, PartialEq, Eq)]
2120pub struct UniquenessConstraint {
2121    /// `true` when this constraint was declared as `PRIMARY KEY`
2122    /// (vs `UNIQUE`). Semantically PK implies NOT NULL on all
2123    /// referenced columns; the engine enforces that at CREATE
2124    /// TABLE time.
2125    pub is_primary_key: bool,
2126    /// Column positions on the parent table. ≥ 1 element. For
2127    /// single-column UNIQUE this is exactly one position; the
2128    /// BTree index alone enforces it.
2129    pub columns: Vec<usize>,
2130    /// v7.13.0 — `UNIQUE NULLS NOT DISTINCT` modifier
2131    /// (mailrs round-5 G10; PG 15+ surface). When `true`, two
2132    /// rows whose constrained columns are all NULL collide on
2133    /// the constraint. Default (`false`) is the SQL-standard
2134    /// `NULLS DISTINCT` behaviour where any NULL passes.
2135    /// Persisted in catalog FILE_VERSION 23+.
2136    pub nulls_not_distinct: bool,
2137    /// v7.39 (read01 round 48) — the constraint's SQL name when the user
2138    /// supplied one (`ADD CONSTRAINT <name> PRIMARY KEY/UNIQUE (...)`, or
2139    /// the inline `CONSTRAINT <name>` form). `None` = unnamed, in which
2140    /// case `pg_constraint` synthesises PG's `<table>_pkey` /
2141    /// `<table>_<col>_key` form. DROP CONSTRAINT resolves the stored name
2142    /// first and falls back to the synthesised one, so catalogs written
2143    /// before this field (< FILE_VERSION 60) keep working unchanged.
2144    pub name: Option<String>,
2145    /// v7.39 (round 711) — `[NOT] DEFERRABLE`. Round 621 taught the parser
2146    /// to CONSUME the clause on PK/UNIQUE (the FK path had stored it since
2147    /// round 288); this is the storing half. Persisted in the v89 timing
2148    /// appendix.
2149    pub deferrable: bool,
2150    /// `INITIALLY DEFERRED`: the check belongs to COMMIT, not the
2151    /// statement, unless `SET CONSTRAINTS … IMMEDIATE` pulls it in.
2152    pub initially_deferred: bool,
2153}
2154
2155/// v7.39 (round 210) — an `EXCLUDE` constraint. Forbids two distinct live
2156/// rows from satisfying, for EVERY element, `new.col <op> existing.col`
2157/// (e.g. `EXCLUDE USING gist (during WITH &&)` = no two `during` ranges
2158/// overlap). Unlike a uniqueness constraint the operator is not equality,
2159/// so enforcement is a full live-row scan re-checking the operator (a real
2160/// GiST index that answers overlap in O(log n) is a later perf phase). A
2161/// NULL in any element column exempts the row (matching PG / UNIQUE NULL
2162/// semantics). Persisted in catalog FILE_VERSION 72+.
2163#[derive(Debug, Clone, PartialEq, Eq)]
2164pub struct ExclusionConstraint {
2165    /// The constraint's SQL name. PG auto-names an unnamed EXCLUDE
2166    /// `<table>_<leading-col>_excl`; the engine synthesises that at CREATE
2167    /// TABLE time so this is always populated.
2168    pub name: String,
2169    /// Access method spelled after `USING` (`gist`, `spgist`, …), lower-cased.
2170    /// `None` = no `USING` clause. Purely cosmetic for enforcement; it round-
2171    /// trips into `pg_get_constraintdef`.
2172    pub method: Option<String>,
2173    /// One `(column-position, operator-spelling)` pair per element, in
2174    /// declaration order. The operator spelling is the wire token (`&&`,
2175    /// `=`, `@>`, `<@`, `&<`, `&>`) evaluated against each existing row.
2176    pub elements: Vec<(usize, String)>,
2177}
2178
2179/// v7.6.1 — Storage-layer mirror of `spg_sql::ast::ForeignKeyConstraint`.
2180/// The engine's CREATE TABLE path translates between the two; keeping
2181/// them separate preserves the no-deps boundary between
2182/// `spg-storage` and `spg-sql`.
2183#[derive(Debug, Clone, PartialEq, Eq)]
2184pub struct ForeignKeyConstraint {
2185    /// Optional user-supplied constraint name (`CONSTRAINT <name>`
2186    /// prefix). Used by `ALTER TABLE DROP CONSTRAINT <name>` in
2187    /// v7.6.8; ignored by enforcement.
2188    pub name: Option<String>,
2189    /// Positions of local columns in this table's column list.
2190    /// Same arity as `parent_columns`.
2191    pub local_columns: Vec<usize>,
2192    /// Referenced parent table name.
2193    pub parent_table: String,
2194    /// Positions of parent columns in the parent's column list.
2195    /// Engine resolves these at CREATE TABLE time (after the parent
2196    /// schema is known) so enforcement paths can skip the name
2197    /// lookup on every row.
2198    pub parent_columns: Vec<usize>,
2199    /// Referential action when a parent row is deleted.
2200    pub on_delete: FkAction,
2201    /// Referential action when a parent row's referenced columns
2202    /// are updated.
2203    pub on_update: FkAction,
2204    /// v7.38 (read01, T29) — `MATCH SIMPLE | FULL`. Defaults to `Simple`.
2205    pub match_type: MatchType,
2206    /// v7.39 (round 288) — `[NOT] DEFERRABLE`.
2207    pub deferrable: bool,
2208    /// `INITIALLY DEFERRED`: the check runs at COMMIT rather than at
2209    /// the statement, unless `SET CONSTRAINTS … IMMEDIATE` pulls it in.
2210    pub initially_deferred: bool,
2211}
2212
2213/// v7.38 (read01, T29) — FK MATCH type. Mirrors `spg_sql::ast::MatchType`.
2214#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2215pub enum MatchType {
2216    #[default]
2217    Simple,
2218    Full,
2219}
2220
2221impl MatchType {
2222    /// On-disk tag byte (catalog appendix, `FILE_VERSION` 55+).
2223    pub const fn tag(self) -> u8 {
2224        match self {
2225            Self::Simple => 0,
2226            Self::Full => 1,
2227        }
2228    }
2229    pub const fn from_tag(b: u8) -> Option<Self> {
2230        Some(match b {
2231            0 => Self::Simple,
2232            1 => Self::Full,
2233            _ => return None,
2234        })
2235    }
2236}
2237
2238/// v7.6.1 — referential action tag. Mirrors `spg_sql::ast::FkAction`.
2239#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2240pub enum FkAction {
2241    Restrict,
2242    Cascade,
2243    SetNull,
2244    SetDefault,
2245    NoAction,
2246}
2247
2248impl FkAction {
2249    /// On-disk tag byte (v13 catalog appendix).
2250    pub const fn tag(self) -> u8 {
2251        match self {
2252            Self::Restrict => 0,
2253            Self::Cascade => 1,
2254            Self::SetNull => 2,
2255            Self::SetDefault => 3,
2256            Self::NoAction => 4,
2257        }
2258    }
2259    pub const fn from_tag(b: u8) -> Option<Self> {
2260        Some(match b {
2261            0 => Self::Restrict,
2262            1 => Self::Cascade,
2263            2 => Self::SetNull,
2264            3 => Self::SetDefault,
2265            4 => Self::NoAction,
2266            _ => return None,
2267        })
2268    }
2269}
2270
2271impl TableSchema {
2272    pub fn column_position(&self, name: &str) -> Option<usize> {
2273        self.columns.iter().position(|c| c.name == name)
2274    }
2275}
2276
2277/// Key type accepted by secondary indices. Float / NULL / Vector values
2278/// can't participate in a B-tree index — `f64` is only `PartialOrd`, NULL
2279/// has SQL-three-valued semantics, and Vector belongs to the (future) HNSW
2280/// path. Index lookups on those columns fall back to full scan.
2281#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)]
2282pub enum IndexKey {
2283    Int(i64),
2284    Text(String),
2285    Bool(bool),
2286    /// v7.17.0 — `Value::Uuid` index key. Comparison is byte-wise
2287    /// (RFC 4122 byte order) so PRIMARY KEY UUID lookups land on
2288    /// the same fast-path as Int / Text.
2289    Uuid([u8; 16]),
2290    /// r1039 — `Value::Bytes` (bytea). PG orders bytea by plain byte
2291    /// comparison, shorter-prefix first (`'' < \x00 < \x0000 < \x01ff <
2292    /// \xff`, measured on 18.4), which is exactly `Vec<u8>`'s `Ord`.
2293    Bytes(Vec<u8>),
2294    /// r1039 — exact decimal, in the canonical form described on
2295    /// [`NumericKey`].
2296    ///
2297    /// r1040 — BOXED, and the box is load-bearing for every OTHER index.
2298    /// A `NumericKey` is 48 bytes against `Text(String)`'s 24, so inline
2299    /// it set the size of the whole enum and every B-tree node in every
2300    /// index grew with it: 32 bytes per key to 48, align 8 to 16.
2301    /// Measured through the release sweep, `SELECT pad FROM t ORDER BY
2302    /// id` over 400,000 rows — a walk of the primary key's index — went
2303    /// 39.4-40.6 ms to 42.3-44.1, in both leg orders. The indirection is
2304    /// charged to numeric keys, which are new, instead of to every index
2305    /// that existed already.
2306    Numeric(alloc::boxed::Box<NumericKey>),
2307}
2308
2309/// r1039 — an exact-decimal index key, canonical so that representation
2310/// equality IS value equality.
2311///
2312/// That property is the whole reason this is a struct rather than the
2313/// `(scaled, scale)` pair the value carries. `1.5` and `1.50` are the
2314/// same NUMERIC (PG18.4: `1.5::numeric = 1.50::numeric` is true) and
2315/// arrive here as `(15, 1)` and `(150, 2)`. A B-tree keyed on the raw
2316/// pair would file them apart, so `WHERE n = 1.5` would miss a row stored
2317/// as `1.50` — an index changing the answer, which is the one thing an
2318/// index may never do. `BigNumeric::cmp` carries the same warning and
2319/// declines to implement `Ord` for exactly this reason; a KEY cannot
2320/// decline, so it normalizes instead.
2321///
2322/// Canonical form: significant decimal digits with no leading and no
2323/// trailing zeros, most significant first, plus the decimal exponent of
2324/// the leading digit. Zero is the empty digit vector with `neg == false`
2325/// and `exp == 0`, so there is no `-0`.
2326///
2327/// Ordering is PG's, measured: `-Infinity < -1 < 0 < 1 < Infinity < NaN`,
2328/// and `NaN = NaN`.
2329#[derive(Debug, Clone, PartialEq, Eq)]
2330pub struct NumericKey {
2331    /// 0 = -Infinity, 1 = finite, 2 = +Infinity, 3 = NaN. Ordering the
2332    /// classes by this byte is what puts NaN on top, where PG keeps it.
2333    class: u8,
2334    /// Finite only, and never set for zero.
2335    neg: bool,
2336    /// Decimal exponent of the leading significant digit; 0 for zero.
2337    exp: i32,
2338    /// r1040 — the first [`HEAD_DIGITS`] significant digits, LEFT-ALIGNED
2339    /// (multiplied up so the leading digit always sits at 10^36). That
2340    /// alignment is what makes an integer comparison of two heads the same
2341    /// answer as a digit-by-digit one: `12` and `1` become 1.2e36 and
2342    /// 1.0e36, which order the way the digit strings do, where the bare
2343    /// integers 12 and 1 would not.
2344    ///
2345    /// Zero for the value zero and for every special.
2346    ///
2347    /// This started as a `Vec<u8>` of digits, which is correct and cost
2348    /// an allocation per key and a slice comparison per sort comparison.
2349    /// `ORDER BY <numeric>` builds one key per row and compares n log n
2350    /// times: 200,000 rows measured 65.4 ms against 39.6 for the f64
2351    /// projection that had been returning rows in the wrong order.
2352    head: u128,
2353    /// Significant digits past the 37th, one per byte, no trailing zeros.
2354    /// Empty for everything an `i128` mantissa can hold with room to
2355    /// spare — and an empty `Vec` does not allocate, which is the point.
2356    tail: Vec<u8>,
2357}
2358
2359/// Significant digits carried in [`NumericKey::head`]. 37 is the most
2360/// that can be left-aligned inside a `u128`: the largest such value is
2361/// 9.99…e36, and `u128::MAX` is 3.4e38.
2362const HEAD_DIGITS: u32 = 37;
2363/// `10^36` — where a left-aligned leading digit sits.
2364const HEAD_SCALE: u128 = 1_000_000_000_000_000_000_000_000_000_000_000_000;
2365
2366/// The `class` byte of [`NumericKey`], in PG's order.
2367const NUM_CLASS_NEG_INF: u8 = 0;
2368const NUM_CLASS_FINITE: u8 = 1;
2369const NUM_CLASS_POS_INF: u8 = 2;
2370const NUM_CLASS_NAN: u8 = 3;
2371
2372impl NumericKey {
2373    /// The key for a `Value::Numeric`'s three fields.
2374    ///
2375    /// Public because the ORDER BY key wants the same canonical form the
2376    /// index key uses: two sort keys that disagree about which of two
2377    /// NUMERICs is larger is the same class of defect as an index that
2378    /// disagrees with a scan, and one definition is how they stay honest.
2379    #[must_use]
2380    pub fn from_numeric(scaled: i128, scale: u16, kind: NumericKind) -> Self {
2381        match kind {
2382            NumericKind::Finite => {
2383                let mut buf = [0u8; 40];
2384                let n = digits_of_u128(scaled.unsigned_abs(), &mut buf);
2385                Self::finite(scaled < 0, &buf[..n], i32::from(scale))
2386            }
2387            NumericKind::NaN => Self::special(NUM_CLASS_NAN),
2388            NumericKind::PosInf => Self::special(NUM_CLASS_POS_INF),
2389            NumericKind::NegInf => Self::special(NUM_CLASS_NEG_INF),
2390        }
2391    }
2392
2393    /// The key for an exact integer — no scale, so no rounding.
2394    #[must_use]
2395    pub fn from_i128(n: i128) -> Self {
2396        let mut buf = [0u8; 40];
2397        let len = digits_of_u128(n.unsigned_abs(), &mut buf);
2398        Self::finite(n < 0, &buf[..len], 0)
2399    }
2400
2401    /// The key for a mantissa that overflowed `i128`. The two
2402    /// representations of one value land on one key.
2403    #[must_use]
2404    pub fn from_big(b: &crate::bignum::BigNumeric) -> Self {
2405        let (neg, limbs, scale) = b.parts();
2406        Self::finite(neg, &digits_of_limbs(limbs), i32::from(scale))
2407    }
2408
2409    /// The `f64` this key means, for the one comparison PG defines that
2410    /// way: `numeric` against `float8` demotes the numeric.
2411    ///
2412    /// Lossy by construction — that is the point, and it is why nothing
2413    /// else uses it.
2414    #[must_use]
2415    #[allow(clippy::cast_precision_loss)]
2416    pub fn to_f64(&self) -> f64 {
2417        match self.class {
2418            NUM_CLASS_NAN => return f64::NAN,
2419            NUM_CLASS_POS_INF => return f64::INFINITY,
2420            NUM_CLASS_NEG_INF => return f64::NEG_INFINITY,
2421            _ => {}
2422        }
2423        if self.head == 0 {
2424            return 0.0;
2425        }
2426        // `head` is `d.ddd… × 10^36`; the value is that leading digit and
2427        // its followers at `exp`. The tail is below f64's resolution by
2428        // construction (it starts at the 38th significant digit).
2429        let mantissa = self.head as f64 / HEAD_SCALE as f64;
2430        let out = mantissa * pow10_f64(self.exp);
2431        if self.neg { -out } else { out }
2432    }
2433
2434    /// The significant decimal digits, most significant first — the form
2435    /// the catalog codec writes, and the one `from_parts` reads back.
2436    #[must_use]
2437    pub fn digits(&self) -> Vec<u8> {
2438        let mut out = Vec::new();
2439        if self.head != 0 {
2440            let mut h = self.head;
2441            for _ in 0..HEAD_DIGITS {
2442                let d = u8::try_from(h / HEAD_SCALE).unwrap_or(0);
2443                out.push(d);
2444                h = (h % HEAD_SCALE) * 10;
2445            }
2446            while out.last() == Some(&0) {
2447                out.pop();
2448            }
2449        }
2450        out.extend_from_slice(&self.tail);
2451        out
2452    }
2453
2454    /// The wire parts, for the catalog codec.
2455    #[must_use]
2456    pub fn parts(&self) -> (u8, bool, i32) {
2457        (self.class, self.neg, self.exp)
2458    }
2459
2460    /// Rebuild from the wire parts. Returns `None` on parts that are not
2461    /// canonical, so a corrupt catalog cannot smuggle in a key whose `Eq`
2462    /// and `Ord` disagree.
2463    #[must_use]
2464    pub fn from_parts(class: u8, neg: bool, exp: i32, digits: &[u8]) -> Option<Self> {
2465        if class > NUM_CLASS_NAN || digits.iter().any(|d| *d > 9) {
2466            return None;
2467        }
2468        if class != NUM_CLASS_FINITE && (neg || exp != 0 || !digits.is_empty()) {
2469            return None;
2470        }
2471        if digits.is_empty() {
2472            if neg || exp != 0 {
2473                return None;
2474            }
2475            return Some(Self::special(class));
2476        }
2477        if digits[0] == 0 || digits[digits.len() - 1] == 0 {
2478            return None;
2479        }
2480        Some(Self {
2481            class,
2482            neg,
2483            exp,
2484            head: head_of(digits),
2485            tail: digits.iter().skip(HEAD_DIGITS as usize).copied().collect(),
2486        })
2487    }
2488
2489    /// Canonicalize `(-1)^neg · <digits as an integer> · 10^-scale`.
2490    ///
2491    /// `digits` is most-significant-first and may carry leading and
2492    /// trailing zeros; both are stripped, which is what makes `1.5` and
2493    /// `1.50` land on the same key.
2494    fn finite(neg: bool, digits: &[u8], scale: i32) -> Self {
2495        let lead = digits.iter().position(|d| *d != 0).unwrap_or(digits.len());
2496        let digits = &digits[lead..];
2497        if digits.is_empty() {
2498            return Self::special(NUM_CLASS_FINITE);
2499        }
2500        // The leading digit's exponent, taken BEFORE trailing zeros go:
2501        // dropping low-order digits does not move the leading one.
2502        let exp = i32::try_from(digits.len()).unwrap_or(i32::MAX) - 1 - scale;
2503        let mut end = digits.len();
2504        while end > 0 && digits[end - 1] == 0 {
2505            end -= 1;
2506        }
2507        let digits = &digits[..end];
2508        Self {
2509            class: NUM_CLASS_FINITE,
2510            neg,
2511            exp,
2512            head: head_of(digits),
2513            tail: digits.iter().skip(HEAD_DIGITS as usize).copied().collect(),
2514        }
2515    }
2516
2517    fn special(class: u8) -> Self {
2518        Self {
2519            class,
2520            neg: false,
2521            exp: 0,
2522            head: 0,
2523            tail: Vec::new(),
2524        }
2525    }
2526}
2527
2528/// The first [`HEAD_DIGITS`] of `digits`, left-aligned so the leading one
2529/// sits at `10^36`.
2530fn head_of(digits: &[u8]) -> u128 {
2531    let mut head: u128 = 0;
2532    let take = (HEAD_DIGITS as usize).min(digits.len());
2533    for d in &digits[..take] {
2534        head = head * 10 + u128::from(*d);
2535    }
2536    for _ in take..HEAD_DIGITS as usize {
2537        head *= 10;
2538    }
2539    head
2540}
2541
2542/// Decimal digits of `mag` into `buf`, most significant first; returns how
2543/// many were written. Zero writes none.
2544///
2545/// r1040 — split at `u64` on purpose. A `u128` divide is a called routine,
2546/// not an instruction, and this loop runs once per digit per key.
2547fn digits_of_u128(mag: u128, buf: &mut [u8; 40]) -> usize {
2548    if mag == 0 {
2549        return 0;
2550    }
2551    let mut rev = [0u8; 40];
2552    let mut n = 0usize;
2553    let mut big = mag;
2554    // Peel nineteen digits at a time — the most a `u64` holds — so the
2555    // wide divide runs at most twice.
2556    while big > u128::from(u64::MAX) {
2557        let mut chunk = u64::try_from(big % 10_000_000_000_000_000_000_u128).unwrap_or(0);
2558        big /= 10_000_000_000_000_000_000_u128;
2559        for _ in 0..19 {
2560            rev[n] = u8::try_from(chunk % 10).unwrap_or(0);
2561            chunk /= 10;
2562            n += 1;
2563        }
2564    }
2565    let mut small = u64::try_from(big).unwrap_or(0);
2566    while small > 0 {
2567        rev[n] = u8::try_from(small % 10).unwrap_or(0);
2568        small /= 10;
2569        n += 1;
2570    }
2571    for i in 0..n {
2572        buf[i] = rev[n - 1 - i];
2573    }
2574    n
2575}
2576
2577/// Decimal digits of a base-10^9 little-endian limb vector, most
2578/// significant first. Every limb but the leading one is padded to its
2579/// full nine digits — that padding is the whole point, since a limb of 5
2580/// in the middle of a number means `000000005`.
2581fn digits_of_limbs(limbs: &[u32]) -> Vec<u8> {
2582    let mut out = Vec::new();
2583    let mut buf = [0u8; 40];
2584    for (i, limb) in limbs.iter().enumerate().rev() {
2585        let n = digits_of_u128(u128::from(*limb), &mut buf);
2586        if i + 1 == limbs.len() {
2587            out.extend_from_slice(&buf[..n]);
2588        } else {
2589            out.extend(core::iter::repeat_n(0u8, 9 - n));
2590            out.extend_from_slice(&buf[..n]);
2591        }
2592    }
2593    out
2594}
2595
2596/// `10^e` as an `f64`, for any `e` a canonical key can carry.
2597#[allow(clippy::cast_precision_loss)]
2598fn pow10_f64(e: i32) -> f64 {
2599    let mut out = 1.0_f64;
2600    let mag = e.unsigned_abs();
2601    for _ in 0..mag {
2602        out *= 10.0;
2603    }
2604    if e < 0 { 1.0 / out } else { out }
2605}
2606
2607impl Ord for NumericKey {
2608    fn cmp(&self, other: &Self) -> core::cmp::Ordering {
2609        use core::cmp::Ordering;
2610        if self.class != other.class {
2611            return self.class.cmp(&other.class);
2612        }
2613        if self.class != NUM_CLASS_FINITE {
2614            // Each of the three specials is a single value, and PG holds
2615            // `'NaN'::numeric = 'NaN'::numeric` true.
2616            return Ordering::Equal;
2617        }
2618        // Zero first: it is stored with `neg == false` and `exp == 0`, so
2619        // the magnitude comparison below would put it above every value
2620        // smaller than 1 rather than between the negatives and positives.
2621        match (self.head == 0, other.head == 0) {
2622            (true, true) => return Ordering::Equal,
2623            (true, false) => {
2624                return if other.neg {
2625                    Ordering::Greater
2626                } else {
2627                    Ordering::Less
2628                };
2629            }
2630            (false, true) => {
2631                return if self.neg {
2632                    Ordering::Less
2633                } else {
2634                    Ordering::Greater
2635                };
2636            }
2637            (false, false) => {}
2638        }
2639        match (self.neg, other.neg) {
2640            (false, true) => return Ordering::Greater,
2641            (true, false) => return Ordering::Less,
2642            _ => {}
2643        }
2644        // Same sign, both non-zero: more integer digits is bigger, and at
2645        // equal exponent the left-aligned heads compare as one integer —
2646        // the alignment is what makes that the same answer as comparing
2647        // the digit strings. The tail only speaks when the first 37
2648        // significant digits are identical.
2649        let mag = self
2650            .exp
2651            .cmp(&other.exp)
2652            .then_with(|| self.head.cmp(&other.head))
2653            .then_with(|| self.tail.cmp(&other.tail));
2654        if self.neg { mag.reverse() } else { mag }
2655    }
2656}
2657
2658impl PartialOrd for NumericKey {
2659    fn partial_cmp(&self, other: &Self) -> Option<core::cmp::Ordering> {
2660        Some(self.cmp(other))
2661    }
2662}
2663
2664impl IndexKey {
2665    /// v7.37.43 (INSUBQ B-4) — inline-friendly BigInt fast path.
2666    /// `try_count_star_pk_in_subquery_fast` (and any other hot loop
2667    /// probing an integer PK) already holds an `i64`; this builds the
2668    /// `IndexKey` without going through the generic `from_value`
2669    /// dispatch tree.
2670    #[inline]
2671    pub fn from_i64(n: i64) -> Self {
2672        Self::Int(n)
2673    }
2674
2675    /// r1039 — the key a value takes when the INDEXED COLUMN is `ty`, or
2676    /// `None` when it takes none (→ the caller falls back to a scan).
2677    ///
2678    /// Every key under one index comes from one column, so they all live
2679    /// in one key SPACE. A probe built in a different space finds nothing
2680    /// — and "nothing" is indistinguishable from "no matching rows",
2681    /// which is how round 564 and r1037 both turned an index into a wrong
2682    /// answer (a TEXT key sought against a DATE-keyed and a UUID-keyed
2683    /// index).
2684    ///
2685    /// The two spaces this round adds make that trap reachable again from
2686    /// a new direction: `WHERE n = 2` on a NUMERIC column produces
2687    /// `Value::Int`, and an integer key would look in a space nothing
2688    /// lives in. So NUMERIC columns take integers by converting them
2689    /// exactly, and refuse anything they cannot convert; BYTEA columns
2690    /// take only `Value::Bytes`; and no other column may be keyed in
2691    /// either of the two new spaces.
2692    ///
2693    /// Use this wherever the key comes from a LITERAL or from another
2694    /// table's value. [`IndexKey::from_value`] stays right for building
2695    /// the index itself, where the value is the column's own.
2696    pub fn from_value_for_column(v: &Value<'_>, ty: DataType) -> Option<Self> {
2697        match ty {
2698            DataType::Numeric { .. } => match v {
2699                Value::SmallInt(n) => Some(Self::exact_int_key(i128::from(*n))),
2700                Value::Int(n) => Some(Self::exact_int_key(i128::from(*n))),
2701                Value::BigInt(n) => Some(Self::exact_int_key(i128::from(*n))),
2702                Value::Numeric { .. } | Value::NumericBig(_) => Self::from_value(v),
2703                // Float included: `2.0::float8` and `2.0::numeric` are not
2704                // the same value to a B-tree, and rounding one into the
2705                // other's space is how a seek reaches the wrong row.
2706                _ => None,
2707            },
2708            DataType::Bytes => match v {
2709                Value::Bytes(b) => Some(Self::Bytes(b.to_vec())),
2710                _ => None,
2711            },
2712            _ => match Self::from_value(v) {
2713                Some(Self::Numeric(_) | Self::Bytes(_)) => None,
2714                other => other,
2715            },
2716        }
2717    }
2718
2719    /// An integer as a NUMERIC key. Exact by construction — no scale, no
2720    /// rounding — which is why the conversion is allowed at all.
2721    fn exact_int_key(n: i128) -> Self {
2722        Self::Numeric(alloc::boxed::Box::new(NumericKey::from_i128(n)))
2723    }
2724
2725    pub fn from_value(v: &Value<'_>) -> Option<Self> {
2726        match v {
2727            // v7.37.43 (INSUBQ B-4) — BigInt hits first (the dominant
2728            // INSUBQ shape probes PK as BigInt). Tiny micro-win.
2729            Value::BigInt(n) => Some(Self::Int(*n)),
2730            Value::SmallInt(n) => Some(Self::Int(i64::from(*n))),
2731            Value::Int(n) => Some(Self::Int(i64::from(*n))),
2732            Value::Text(s) => Some(Self::Text(s.clone().into_owned())),
2733            // v7.38 (read01, T11) — bpchar keys compare blank-insensitively.
2734            Value::BpChar(s) => Some(Self::Text(s.trim_end_matches(' ').to_string())),
2735            Value::Bool(b) => Some(Self::Bool(*b)),
2736            // Date/Timestamp use their integer storage repr as the
2737            // index key — same order semantics, same comparison.
2738            Value::Date(d) => Some(Self::Int(i64::from(*d))),
2739            Value::Timestamp(t) => Some(Self::Int(*t)),
2740            // v7.17.0: UUID indexable via byte-wise ordering. Lookup
2741            // on `id = '...'::uuid` resolves through the secondary
2742            // index rather than full-scan.
2743            Value::Uuid(b) => Some(Self::Uuid(*b)),
2744            // v7.17.0 Phase 3.P0-32: TIME indexable via i64 — same
2745            // order semantics as Date/Timestamp.
2746            Value::Time(us) => Some(Self::Int(*us)),
2747            // v7.17.0 Phase 3.P0-33: YEAR indexable as i64 — u16
2748            // widens losslessly and gives the natural calendar
2749            // ordering.
2750            Value::Year(y) => Some(Self::Int(i64::from(*y))),
2751            // v7.17.0 Phase 3.P0-34: TIMETZ indexable by its
2752            // UTC-equivalent microseconds (local wall - offset).
2753            // Without normalising, two values for the same
2754            // physical instant in different zones would sort
2755            // wrong. Matches PG's TIMETZ index behaviour.
2756            Value::TimeTz { us, offset_secs } => {
2757                Some(Self::Int(us - i64::from(*offset_secs) * 1_000_000))
2758            }
2759            // v7.17.0 Phase 3.P0-35: MONEY indexable as i64 cents
2760            // (no scaling needed — natural numeric ordering).
2761            Value::Money(c) => Some(Self::Int(*c)),
2762            // v7.17.0 Phase 3.P0-38: ranges are NOT indexable in
2763            // v7.17.0 — they'd need a custom comparator (PG uses
2764            // SP-GiST for this). Skip.
2765            Value::Range { .. } => None,
2766            // v7.17.0 Phase 3.P0-39: hstore is NOT indexable in
2767            // v7.17.0 — map columns need GIN with bespoke ops.
2768            Value::Hstore(_) => None,
2769            // r1039 — exact decimals index through the canonical
2770            // [`NumericKey`], which is what makes `1.5` and `1.50` one key.
2771            Value::NumericBig(b) => Some(Self::Numeric(alloc::boxed::Box::new(NumericKey::from_big(b)))),
2772            Value::Numeric {
2773                scaled,
2774                scale,
2775                kind,
2776            } => Some(Self::Numeric(alloc::boxed::Box::new(
2777                NumericKey::from_numeric(*scaled, *scale, *kind),
2778            ))),
2779            // r1039 — bytea orders by plain byte comparison, which is
2780            // `Vec<u8>`'s own.
2781            Value::Bytes(b) => Some(Self::Bytes(b.to_vec())),
2782            // v7.17.0 Phase 3.P0-40: 2D arrays aren't indexable.
2783            Value::IntArray2D(_)
2784            | Value::BigIntArray2D(_)
2785            | Value::TextArray2D(_)
2786            | Value::BoolArray2D(_) => None,
2787            // v7.37.5 β-P4: INTERVAL[] isn't indexable (PG uses
2788            // GIN/intarray for array-contains queries; SPG plans
2789            // that as a separate axis under v7.37.8 GIN-on-jsonb).
2790            Value::IntervalArray(_) => None,
2791            // v7.37.5 γ — none of the array-of-scalar family is
2792            // B-tree indexable. Same reason as IntervalArray: PG
2793            // serves array-contains / array-overlap queries via
2794            // GIN, and SPG's GIN axis lands in v7.37.8.
2795            Value::BoolArray(_)
2796            | Value::SmallIntArray(_)
2797            | Value::FloatArray(_)
2798            | Value::NumericArray(_)
2799            | Value::DateArray(_)
2800            | Value::TimestampArray(_)
2801            | Value::TimestamptzArray(_)
2802            | Value::UuidArray(_)
2803            | Value::JsonArray(_)
2804            | Value::JsonbArray(_)
2805            | Value::BytesArray(_)
2806            | Value::VarcharArray(_)
2807            | Value::CharArray(_)
2808            // v7.37.5 δ — multirange not indexable (PG uses GiST/
2809            // SP-GiST + a custom operator class; SPG plans the same
2810            // axis under v7.37.8 with ranges).
2811            | Value::Multirange { .. }
2812            // v7.37.5 ε — geometric scalars not B-tree indexable
2813            // (PG uses GiST/SP-GiST for these too; SPG plans the
2814            // same axis under v7.37.8).
2815            | Value::Point(_)
2816            | Value::Lseg(_, _)
2817            | Value::Path { .. }
2818            | Value::PgBox(_, _)
2819            | Value::Polygon(_)
2820            | Value::Line { .. }
2821            | Value::Circle { .. }
2822            // v7.37.5 ζ-A — network / bit / xml / "char" / money[].
2823            // INET / CIDR / MACADDR / MACADDR8 could be B-tree
2824            // indexable (PG does this), but the byte-wise compare
2825            // family-blind would mis-order IPv4 vs IPv6; left as
2826            // a follow-up under v7.37.8 GIN window.
2827            | Value::Inet { .. }
2828            | Value::Cidr { .. }
2829            | Value::Macaddr(_)
2830            | Value::Macaddr8(_)
2831            | Value::PgLsn(_)
2832            | Value::BitString { .. }
2833            | Value::Xml(_)
2834            | Value::Char1(_)
2835            | Value::MoneyArray(_)
2836            | Value::Composite(_)
2837            | Value::Tid(..)
2838            | Value::Xid(_)
2839            | Value::Cid(_)
2840            | Value::RegClass(..)
2841            | Value::RegProc(..)
2842            | Value::RegType(..) => None,
2843            // Interval isn't index-eligible (and can't reach this path
2844            // through column storage anyway). Float / Real stay out
2845            // because `f64` is only `PartialOrd`.
2846            Value::Null
2847            | Value::Float(_)
2848            | Value::Vector(_)
2849            | Value::Sq8Vector(_)
2850            | Value::HalfVector(_)
2851            | Value::Interval { .. }
2852            | Value::Json(_)
2853            | Value::TextArray(_)
2854            | Value::IntArray(_)
2855            | Value::BigIntArray(_)
2856            | Value::TsVector(_)
2857            | Value::TsQuery(_)
2858            | Value::Real(_) => None,
2859        }
2860    }
2861}
2862
2863/// A single-column secondary index. v2.0 carries either a B-tree map
2864/// (the default — used for equality / range lookups on scalar columns)
2865/// or a navigable-small-world graph (used for kNN over vector
2866/// columns).
2867#[derive(Debug, Clone)]
2868pub struct Index {
2869    pub name: String,
2870    pub column_position: usize,
2871    pub kind: IndexKind,
2872    /// v6.8.0 — column positions of `INCLUDE (col1, col2, …)`
2873    /// non-key columns. Carries the planner's "this query is
2874    /// covered by the index" signal; lookup paths still resolve
2875    /// via the `RowLocator` to fetch the row body, but EXPLAIN
2876    /// surfaces the covered-scan annotation so operators can
2877    /// confirm the planner sees the coverage.
2878    ///
2879    /// Empty `Vec` = no `INCLUDE` clause (the legacy shape). v12
2880    /// catalog snapshots deserialise with an empty vec.
2881    pub included_columns: Vec<usize>,
2882    /// v6.8.1 — partial-index predicate stored as its canonical
2883    /// Display form (the engine re-parses it on the maintenance
2884    /// path). `None` = unconditional index (the legacy shape).
2885    /// Persisted as `[u8 has_pred][u16 LE len][bytes]` on the
2886    /// catalog snapshot (FILE_VERSION 12, appended after
2887    /// `included_columns`).
2888    pub partial_predicate: Option<String>,
2889    /// v6.8.2 — expression-index key, stored as the expression's
2890    /// canonical Display form. `None` = bare column-reference
2891    /// index (the legacy shape). Persisted alongside
2892    /// `partial_predicate` on the v12 catalog snapshot.
2893    pub expression: Option<String>,
2894    /// v7.39 (read01 round 52) — `CREATE UNIQUE INDEX … NULLS NOT DISTINCT`
2895    /// (PG 15+): a NULL in the key no longer exempts the row, so two
2896    /// all-NULL keys collide. Default `false` = SQL-standard NULLS DISTINCT.
2897    /// Persisted in the index appendix (FILE_VERSION 62+); older catalogs
2898    /// deserialise with `false`.
2899    pub nulls_not_distinct: bool,
2900    /// v7.39 (round 537) — the key column's ordering clause, as written.
2901    ///
2902    /// SPG's index does not scan in a direction, so this changes no
2903    /// lookup; `pg_indexes.indexdef` is a reproduction of the DDL and
2904    /// dropping the clause made `CREATE INDEX i ON t (a DESC NULLS
2905    /// LAST)` read back as `(a)` — a dump lost it and a schema diff saw
2906    /// drift every run. `nulls_first` is `None` when the statement did
2907    /// not say, in which case PG's default applies and neither word is
2908    /// rendered.
2909    pub descending: bool,
2910    pub nulls_first: Option<bool>,
2911    /// v7.39 (round 538) — an explicit `COLLATE` on the key, as written.
2912    /// SPG orders text by bytes, so it changes no comparison; PG prints
2913    /// it because a named collation and an inherited one are different
2914    /// objects even where they sort identically.
2915    pub collation: Option<String>,
2916    /// v7.9.29 — `CREATE UNIQUE INDEX …`. When true the engine
2917    /// rejects INSERTs whose key already appears in this index
2918    /// (combined with `partial_predicate` when present — only
2919    /// rows matching the predicate enter the uniqueness check).
2920    /// Catalog FILE_VERSION 16+; older snapshots deserialise
2921    /// with `false`. mailrs K1.
2922    pub is_unique: bool,
2923    /// v7.9.29 — extra (non-leading) column positions for
2924    /// multi-column indexes (`CREATE INDEX … (a, b, c)`). The
2925    /// planner today still only uses the leading
2926    /// `column_position` for index seeks, but UNIQUE INDEX
2927    /// enforcement walks the full tuple so partial-unique
2928    /// invariants like CalDAV `(calendar_id, uid,
2929    /// recurrence_id)` are enforced correctly. Catalog
2930    /// FILE_VERSION 16+; older snapshots deserialise empty.
2931    pub extra_column_positions: Vec<usize>,
2932}
2933
2934/// Default neighbor degree (M) for the NSW graph. Picked at construction
2935/// time and persisted with the index.
2936pub const NSW_DEFAULT_M: usize = 16;
2937
2938/// v5.2.2: outcome of a successful [`Catalog::freeze_oldest_to_cold`]
2939/// call. The catalog state has already been mutated by the time this
2940/// is returned (hot rows dropped + segment registered + Cold locators
2941/// flipped). The caller's only remaining concern is `segment_bytes` —
2942/// persist them to disk under `<db>.spg/segments/seg_<id>.spg` so a
2943/// future restart can reload via the v5.1 `SPG_PRELOAD_COLD_SEGMENT`
2944/// path. (v5.3's manifest will subsume this manual step.)
2945#[derive(Debug, Clone)]
2946pub struct FreezeReport {
2947    /// Id allocated by [`Catalog::load_segment_bytes`] for the new
2948    /// cold-tier segment. Stable across the call's success path.
2949    pub segment_id: u32,
2950    /// Number of rows that moved hot → cold. Equals the `max_rows`
2951    /// the caller asked for (the API is strict on the count).
2952    pub frozen_rows: usize,
2953    /// Hot-tier bytes reclaimed by the freeze — the
2954    /// [`Table::hot_bytes`] delta before vs after. Useful to feed
2955    /// back into the freezer's budget check on the next tick.
2956    pub bytes_freed: u64,
2957    /// Encoded segment bytes, byte-identical to what
2958    /// [`encode_segment`] produced. The catalog already owns a
2959    /// copy inside `cold_segments`; this hand-off lets the caller
2960    /// persist them without re-encoding.
2961    pub segment_bytes: Vec<u8>,
2962}
2963
2964/// v6.7.4 — read-only output of [`Catalog::prepare_freeze_slice`].
2965/// Carries every row body + key in a contiguous hot-row range,
2966/// already encoded and sorted by PK so the coordinator's merge
2967/// step is a k-way merge over already-sorted streams.
2968///
2969/// `Vec<FreezeSlice>` from N independent workers feeds
2970/// [`Catalog::commit_freeze_slices`], which concats + encodes the
2971/// merged segment + atomically swaps the catalog state.
2972#[derive(Debug, Clone)]
2973pub struct FreezeSlice {
2974    /// Hot-row index range this slice covered (half-open, in the
2975    /// table's `rows: PersistentVec` ordering at call time). The
2976    /// commit step uses this to compute the union range that
2977    /// gets passed to [`Table::delete_rows`].
2978    pub row_range: core::ops::Range<usize>,
2979    /// `(pk_u64, encoded_row_body, IndexKey)` triples, sorted
2980    /// ascending by `pk_u64`. Per-slice sort happens inside
2981    /// `prepare_freeze_slice`; the coordinator does only a
2982    /// k-way merge to reach the global PK ordering
2983    /// [`encode_segment`] requires.
2984    pub rows: Vec<(u64, Vec<u8>, IndexKey)>,
2985}
2986
2987/// v6.7.3 — outcome of a [`Catalog::compact_cold_segments`] call.
2988/// The catalog state has already been mutated when this is returned:
2989/// the merged segment is loaded into `cold_segments`, the source
2990/// segment slots are tombstoned (`None`), and every BTree-index
2991/// `RowLocator::Cold` that previously pointed at a source now
2992/// points at the merged segment. The caller's remaining job is to
2993/// persist `merged_segment_bytes` under
2994/// `<db>.spg/segments/seg_<merged_segment_id>.spg` and update the
2995/// in-memory `segment_id → path` map (remove the source ids, add
2996/// the merged id) so the next CHECKPOINT writes a manifest that
2997/// no longer lists the retired sources.
2998///
2999/// On a no-op (fewer than 2 candidate segments under the threshold),
3000/// `merged_segment_id` is `None` and `sources` is empty; the
3001/// catalog was not mutated.
3002#[derive(Debug, Clone)]
3003pub struct CompactReport {
3004    /// Source segment ids that were merged + tombstoned.
3005    pub sources: Vec<u32>,
3006    /// Id allocated for the merged segment. `None` on no-op.
3007    pub merged_segment_id: Option<u32>,
3008    /// Encoded merged-segment bytes (empty on no-op).
3009    pub merged_segment_bytes: Vec<u8>,
3010    /// Number of rows that landed in the merged segment.
3011    pub merged_rows: usize,
3012    /// `Σ source.num_rows − merged_rows`. Rows present in source
3013    /// segment payloads but unreferenced by any live BTree
3014    /// `Cold` locator — DELETE'd-but-still-frozen rows that
3015    /// compaction GC'd during the merge.
3016    pub deleted_rows_pruned: usize,
3017    /// `Σ source.bytes() − merged.bytes()`. Estimate of on-disk
3018    /// space the merge will reclaim once the source segment files
3019    /// are GC'd. Saturating subtract — never negative.
3020    pub bytes_reclaimed_estimate: u64,
3021}
3022
3023#[derive(Debug, Clone)]
3024pub enum IndexKind {
3025    /// v4.40: structural-sharing B-tree over `IndexKey`. Replaces the v0.8
3026    /// `BTreeMap<IndexKey, Vec<usize>>` — `Index::clone` is now an `Arc`
3027    /// bump regardless of index size, so `Catalog::clone` inside the
3028    /// v4.34 auto-commit wrap stays O(1) even for tables with secondary
3029    /// indices (the case that bottlenecked v4.39 at 1M rows in the
3030    /// sweep).
3031    ///
3032    /// v5.1: value type widened from `Vec<usize>` to `Vec<RowLocator>` so
3033    /// a single key can point to a mix of hot-tier rows (`RowLocator::Hot`,
3034    /// equivalent to the pre-v5 `usize` row index) and cold-tier rows
3035    /// (`RowLocator::Cold { segment_id, page_offset }`) once the v5.2
3036    /// freezer starts producing them. Pre-v5.2 only `Hot` entries appear
3037    /// — the on-disk encoding stays at `FILE_VERSION` 8 (raw u64 row index)
3038    /// because every locator round-trips through `RowLocator::from_legacy_v8_u64`
3039    /// without information loss. `FILE_VERSION` 9 with tagged encoding lands
3040    /// alongside the first freezer commit (v5.1 step 2b / v5.2).
3041    BTree(PersistentBTreeMap<IndexKey, crate::posting::PostingList>),
3042    /// Navigable-small-world graph for vector kNN search.
3043    Nsw(NswGraph),
3044    /// v6.7.1 — BRIN (Block Range INdex). Pure metadata: BRIN
3045    /// indexes carry NO in-memory key→locator map. The (min,
3046    /// max) summaries live in each cold-tier segment's v2
3047    /// envelope sidecar; the BRIN entry in `Table.indices` only
3048    /// records THAT a BRIN index exists on this column so the
3049    /// segment encoder + planner can opt into the summary path.
3050    Brin {
3051        /// The cell type at `column_position` at CREATE INDEX time.
3052        /// Used by the planner to type-check WHERE-clause range
3053        /// predicates against the BRIN-indexed column.
3054        column_type: DataType,
3055    },
3056    /// v7.12.3 — GIN inverted index over a `tsvector` column.
3057    ///
3058    /// Storage shape: `lexeme word → Vec<RowLocator>`. The posting
3059    /// list per word is appended in row-order, so range scans are
3060    /// O(matching rows) once the per-word lookup is done. Multi-
3061    /// term queries intersect / union posting lists.
3062    ///
3063    /// `IndexKey::from_value(TsVector)` returns `None` — GIN doesn't
3064    /// participate in `try_index_seek` (which is BTree-equality-keyed).
3065    /// The engine consults this index through `try_gin_lookup` on
3066    /// `WHERE col @@ tsquery` predicates instead.
3067    ///
3068    /// Backed by a `PersistentBTreeMap` so `Catalog::clone` (the
3069    /// per-write snapshot) stays O(1) — same structural-sharing
3070    /// invariant as BTree.
3071    Gin(PersistentBTreeMap<alloc::string::String, crate::posting::PostingList>),
3072    /// v7.15.0 — `USING gin (col gin_trgm_ops)` over a `TEXT`
3073    /// column. Posting lists map `trigram` (PG-compatible 3-byte
3074    /// shingle on the lower-cased + space-padded input) to row
3075    /// locators. The planner uses this index to accelerate
3076    /// `WHERE col LIKE '…'` / `ILIKE '…'` / `similarity(col, q) >
3077    /// t` — every literal run of length ≥ 1 in the pattern
3078    /// produces a trigram set, the engine intersects the posting
3079    /// lists, and the LIKE / similarity predicate is re-evaluated
3080    /// per candidate row to filter the over-approximation.
3081    /// Persisted via tag-4 index payload in `FILE_VERSION` 24+.
3082    GinTrgm(PersistentBTreeMap<alloc::string::String, crate::posting::PostingList>),
3083    /// v7.17.0 Phase 2.2 — MySQL `FULLTEXT KEY (col)` over a
3084    /// `TEXT` / `VARCHAR` column. Posting lists map
3085    /// `tsvector('simple') lexeme` to row locators. At insert /
3086    /// build time the engine derives the lexemes from the cell
3087    /// via the same lower-case tokenisation rule as
3088    /// `to_tsvector('simple', ...)` — the column itself stays a
3089    /// plain text type on disk (mysqldump round-trips would be
3090    /// broken otherwise). The planner uses this index to
3091    /// accelerate MySQL-shape `MATCH(col) AGAINST('term')`
3092    /// queries by mapping them onto the existing tsquery `@@`
3093    /// walker. Persisted via tag-5 index payload in
3094    /// `FILE_VERSION` 33+.
3095    GinFulltext(PersistentBTreeMap<alloc::string::String, crate::posting::PostingList>),
3096    /// v7.37.8(sentori Epic 5 P2)— `USING gin (col)` over a
3097    /// `JSON` / `JSONB` column. Posting lists map a canonical
3098    /// `(path, leaf)` token(see [`crate::jsonb_gin::extract_tokens`])
3099    /// to row locators so the planner can resolve
3100    /// `<col> @> <jsonb_literal>` to a candidate row set via
3101    /// posting-list intersection + per-row `json::contains`
3102    /// re-verification. Pre-7.37.8 the same DDL loaded as a
3103    /// BTree fallback so `pg_dump` JSONB-GIN scripts kept loading
3104    /// without query-time acceleration. Persisted via tag-6 index
3105    /// payload in `FILE_VERSION` 51+.
3106    GinJsonb(PersistentBTreeMap<alloc::string::String, crate::posting::PostingList>),
3107}
3108
3109impl IndexKind {
3110    /// v7.31 (memory campaign, C2) — bytes this index variant holds
3111    /// resident in RAM, computed by walking its OWN structure rather
3112    /// than a parametric guess made by the engine. Replaces the old
3113    /// `spg_admin::memory_stats` inline match, which charged NSW with
3114    /// a stale `m_max_0 * 8` per node (neighbour slots are `u32` = 4 B
3115    /// since v6.1.x, and most nodes never fill `m_max_0`) and lumped
3116    /// every GIN family index into a flat 1 KiB token — a gross
3117    /// undercount for the text-heavy posting lists that dominate
3118    /// mailrs' footprint. Per-entry container overhead uses the
3119    /// 3-word (24 B on 64-bit) `Vec`/`String` header as the charge.
3120    ///
3121    /// O(index entries): operator/monitoring surface (`memory_stats` /
3122    /// `spg_memory_stats`), not a query path.
3123    #[must_use]
3124    pub fn approx_resident_bytes(&self) -> u64 {
3125        const HEADER: usize = 24; // Vec/String 3-word header on 64-bit.
3126        let loc = core::mem::size_of::<RowLocator>();
3127        match self {
3128            IndexKind::BTree(map) => {
3129                let key = core::mem::size_of::<IndexKey>();
3130                map.iter()
3131                    .map(|(_, locs)| (key + HEADER + locs.len() * loc) as u64)
3132                    .sum()
3133            }
3134            IndexKind::Nsw(g) => {
3135                // `levels` is one byte per node; each layer's adjacency
3136                // is a `Vec<u32>` per node whose actual length we walk
3137                // (the dense layer-0 list dominates, but upper layers
3138                // are sparse — the old estimate ignored that).
3139                let mut b = g.levels.len() as u64;
3140                for layer in &g.layers {
3141                    for nbrs in layer.iter() {
3142                        b += (HEADER + nbrs.len() * core::mem::size_of::<u32>()) as u64;
3143                    }
3144                }
3145                b
3146            }
3147            // BRIN carries NO in-memory key→locator map (the (min,max)
3148            // summaries live in cold-segment sidecars on disk); the
3149            // resident footprint is just the column-type token.
3150            IndexKind::Brin { .. } => core::mem::size_of::<DataType>() as u64,
3151            IndexKind::Gin(map)
3152            | IndexKind::GinTrgm(map)
3153            | IndexKind::GinFulltext(map)
3154            | IndexKind::GinJsonb(map) => map
3155                .iter()
3156                .map(|(word, postings)| {
3157                    (word.len() + HEADER + HEADER + postings.len() * loc) as u64
3158                })
3159                .sum(),
3160        }
3161    }
3162}
3163
3164/// Multi-layer HNSW graph (v2.13). Each node is assigned a `top_level`;
3165/// it appears in layers `0..=top_level`. Higher layers are sparser, so
3166/// search starts from the entry at the top layer, greedy-descends to
3167/// layer 0, and beam-searches there. Layer 0 keeps a larger neighbour
3168/// budget (`m_max_0 = 2 * m` per the HNSW paper); upper layers cap at
3169/// `m`. The struct name stays `NswGraph` so external users / on-disk
3170/// callers don't have to track a rename — the algorithm changed, the
3171/// data slot didn't.
3172#[derive(Debug, Clone)]
3173pub struct NswGraph {
3174    /// Max neighbours per node on layers ≥ 1.
3175    pub m: usize,
3176    /// Max neighbours on layer 0 (the dense bottom layer). HNSW
3177    /// convention: `m_max_0 = 2 * m`.
3178    pub m_max_0: usize,
3179    /// Entry point — the node that sits on the topmost layer. Search
3180    /// always starts here.
3181    pub entry: Option<usize>,
3182    /// Top layer of the entry node (== `layers.len() - 1` when populated).
3183    pub entry_level: u8,
3184    /// `levels[i]` = top layer of node `i`. Nodes whose vector cell is
3185    /// NULL / non-Vector have `levels[i] = 0` and no neighbour entries.
3186    ///
3187    /// v5.5.0: backed by `PersistentVec` so `NswGraph::clone` (and the
3188    /// `Catalog::clone` on every group-commit write that contains it) is O(1)
3189    /// structural-sharing instead of an O(N) element copy.
3190    pub levels: PersistentVec<u8>,
3191    /// `layers[l][i]` = neighbours of node `i` at layer `l`. Inner vec
3192    /// is empty when node `i` doesn't reach layer `l`.
3193    ///
3194    /// v5.5.0: the per-node middle dimension (the O(N) one) is a
3195    /// `PersistentVec`; the outer layer dimension stays a plain `Vec`
3196    /// (layer count ≤ 8, so its clone is O(1) in practice) and the inner
3197    /// neighbour list stays a `Vec` (bounded by `m_max_0`).
3198    ///
3199    /// v6.1.x: neighbour slot widened from `usize` (8 B on 64-bit) to
3200    /// `u32` (4 B). Row indices are catalog-bounded by `u32::MAX` (4G
3201    /// rows per table); the cast at the NSW boundary asserts this. At
3202    /// 1M dim-128 SQ8, layer 0 adjacency alone shrinks by ~128 MiB
3203    /// — the largest single contribution to the v6.0.5-measured
3204    /// 624 MiB ambition gap. On-disk format already used u32 LE, so
3205    /// this is a pure in-memory layout change; no `FILE_VERSION` bump.
3206    pub layers: Vec<PersistentVec<Vec<u32>>>,
3207}
3208
3209impl NswGraph {
3210    fn new(m: usize) -> Self {
3211        Self {
3212            m,
3213            m_max_0: m.saturating_mul(2),
3214            entry: None,
3215            entry_level: 0,
3216            levels: PersistentVec::new(),
3217            layers: alloc::vec![PersistentVec::new()],
3218        }
3219    }
3220
3221    /// Max-neighbour budget for layer `l`.
3222    pub const fn cap_for_layer(&self, layer: u8) -> usize {
3223        if layer == 0 { self.m_max_0 } else { self.m }
3224    }
3225}
3226
3227/// Deterministic level assignment, seeded on the row index so the same
3228/// insert order reproduces the same topology. Distribution is roughly
3229/// HNSW-flavoured with `mL ≈ 1/ln(M) ≈ 0.36` for M=16: each 4-bit
3230/// chunk that comes up zero promotes the node one layer (so P(level ≥
3231/// L) ≈ (1/16)^L).
3232#[allow(clippy::verbose_bit_mask)] // clippy suggests trailing_zeros(); we need an explicit MAX cap and a stable distribution shape.
3233pub fn nsw_assign_level(row_idx: usize) -> u8 {
3234    const MAX_LEVEL: u8 = 7; // 7 ⇒ ~16^7 ≈ 2.7e8 expected nodes between promotions; ample.
3235    // SplitMix-style mixer — cheap and seedable.
3236    let mut x = (row_idx as u64).wrapping_mul(0x9E37_79B9_7F4A_7C15);
3237    x ^= x >> 30;
3238    x = x.wrapping_mul(0xBF58_476D_1CE4_E5B9);
3239    x ^= x >> 27;
3240    x = x.wrapping_mul(0x94D0_49BB_1331_11EB);
3241    x ^= x >> 31;
3242    // Count contiguous low-end zero nibbles (4-bit chunks). Each zero
3243    // nibble has probability 1/16, mirroring HNSW's `mL ≈ 1/ln(M)` for
3244    // M=16. `trailing_zeros / 4` would lose the ordering when x = 0, so
3245    // a plain loop with a cap is clearer.
3246    let mut level: u8 = 0;
3247    while x & 0xF == 0 && level < MAX_LEVEL {
3248        level += 1;
3249        x >>= 4;
3250    }
3251    level
3252}
3253
3254impl Index {
3255    fn new_btree(name: String, column_position: usize) -> Self {
3256        Self {
3257            name,
3258            column_position,
3259            kind: IndexKind::BTree(PersistentBTreeMap::new()),
3260            included_columns: Vec::new(),
3261            partial_predicate: None,
3262            expression: None,
3263            is_unique: false,
3264            nulls_not_distinct: false,
3265            descending: false,
3266            nulls_first: None,
3267            collation: None,
3268            extra_column_positions: Vec::new(),
3269        }
3270    }
3271
3272    fn new_nsw(name: String, column_position: usize, m: usize) -> Self {
3273        Self {
3274            name,
3275            column_position,
3276            kind: IndexKind::Nsw(NswGraph::new(m)),
3277            included_columns: Vec::new(),
3278            partial_predicate: None,
3279            expression: None,
3280            is_unique: false,
3281            nulls_not_distinct: false,
3282            descending: false,
3283            nulls_first: None,
3284            collation: None,
3285            extra_column_positions: Vec::new(),
3286        }
3287    }
3288
3289    /// v6.7.1 — BRIN index constructor. BRIN carries no in-memory
3290    /// data; the `column_type` snapshot is used by the segment
3291    /// encoder + planner for type-checking range predicates.
3292    fn new_brin(name: String, column_position: usize, column_type: DataType) -> Self {
3293        Self {
3294            name,
3295            column_position,
3296            kind: IndexKind::Brin { column_type },
3297            included_columns: Vec::new(),
3298            partial_predicate: None,
3299            expression: None,
3300            is_unique: false,
3301            nulls_not_distinct: false,
3302            descending: false,
3303            nulls_first: None,
3304            collation: None,
3305            extra_column_positions: Vec::new(),
3306        }
3307    }
3308
3309    /// v7.12.3 — GIN inverted-index constructor. Empty posting-list
3310    /// map; caller (typically [`Table::add_gin_index`] or
3311    /// [`Table::restore_gin_index`]) populates it from existing rows
3312    /// or from a deserialised snapshot.
3313    fn new_gin(name: String, column_position: usize) -> Self {
3314        Self {
3315            name,
3316            column_position,
3317            kind: IndexKind::Gin(PersistentBTreeMap::new()),
3318            included_columns: Vec::new(),
3319            partial_predicate: None,
3320            expression: None,
3321            is_unique: false,
3322            nulls_not_distinct: false,
3323            descending: false,
3324            nulls_first: None,
3325            collation: None,
3326            extra_column_positions: Vec::new(),
3327        }
3328    }
3329
3330    /// v7.15.0 — `gin_trgm_ops`-flavoured GIN constructor. Same
3331    /// shape as `new_gin` but the posting-list keys are 3-byte
3332    /// trigram shingles (`pg_trgm`-compatible) and the column
3333    /// type is `TEXT` / `VARCHAR` (not `TSVECTOR`).
3334    fn new_gin_trgm(name: String, column_position: usize) -> Self {
3335        Self {
3336            name,
3337            column_position,
3338            kind: IndexKind::GinTrgm(PersistentBTreeMap::new()),
3339            included_columns: Vec::new(),
3340            partial_predicate: None,
3341            expression: None,
3342            is_unique: false,
3343            nulls_not_distinct: false,
3344            descending: false,
3345            nulls_first: None,
3346            collation: None,
3347            extra_column_positions: Vec::new(),
3348        }
3349    }
3350
3351    /// v7.17.0 Phase 2.2 — MySQL `FULLTEXT KEY` GIN constructor.
3352    /// Same shape as `new_gin_trgm` but the posting-list keys
3353    /// are lower-cased word lexemes (`to_tsvector('simple', col)`
3354    /// equivalent) instead of trigrams, and the column type is
3355    /// `TEXT` / `VARCHAR` (not `TSVECTOR`).
3356    fn new_gin_fulltext(name: String, column_position: usize) -> Self {
3357        Self {
3358            name,
3359            column_position,
3360            kind: IndexKind::GinFulltext(PersistentBTreeMap::new()),
3361            included_columns: Vec::new(),
3362            partial_predicate: None,
3363            expression: None,
3364            is_unique: false,
3365            nulls_not_distinct: false,
3366            descending: false,
3367            nulls_first: None,
3368            collation: None,
3369            extra_column_positions: Vec::new(),
3370        }
3371    }
3372
3373    /// v7.37.8(sentori Epic 5 P2)— JSONB-GIN constructor. Same
3374    /// shape as the other GIN-family indexes; posting-list keys
3375    /// are the canonical `(path, leaf)` tokens emitted by
3376    /// `crate::jsonb_gin::extract_tokens`. Maintains posting
3377    /// lists from `Value::Json` cells(JSONB is a synonym for the
3378    /// same in-memory string-backed Value).
3379    fn new_gin_jsonb(name: String, column_position: usize) -> Self {
3380        Self {
3381            name,
3382            column_position,
3383            kind: IndexKind::GinJsonb(PersistentBTreeMap::new()),
3384            included_columns: Vec::new(),
3385            partial_predicate: None,
3386            expression: None,
3387            is_unique: false,
3388            nulls_not_distinct: false,
3389            descending: false,
3390            nulls_first: None,
3391            collation: None,
3392            extra_column_positions: Vec::new(),
3393        }
3394    }
3395
3396    /// v7.34.4 — descending-order iterator over `(IndexKey, locators)`
3397    /// pairs for a BTree index, with O(log N) descent to the rightmost
3398    /// leaf and lazy emission thereafter. Returns an empty iterator
3399    /// for non-BTree index kinds — callers handle both uniformly.
3400    /// Used by the ORDER BY `<indexed col>` DESC + LIMIT N executor
3401    /// path: walking only the first N matches off the rightmost leaf
3402    /// avoids the per-row materialisation + partial-sort cost on
3403    /// large tables (mailrs `content_worker` at 250 k rows).
3404    pub fn iter_desc(
3405        &self,
3406    ) -> alloc::boxed::Box<dyn Iterator<Item = (&IndexKey, &crate::posting::PostingList)> + '_>
3407    {
3408        match &self.kind {
3409            IndexKind::BTree(m) => alloc::boxed::Box::new(m.iter_rev()),
3410            IndexKind::Nsw(_)
3411            | IndexKind::Brin { .. }
3412            | IndexKind::Gin(_)
3413            | IndexKind::GinTrgm(_)
3414            | IndexKind::GinFulltext(_)
3415            | IndexKind::GinJsonb(_) => alloc::boxed::Box::new(core::iter::empty()),
3416        }
3417    }
3418
3419    /// v7.34.4 — ascending-order iterator over `(IndexKey, locators)`
3420    /// pairs. Mirror of `iter_desc` for ORDER BY ... ASC + LIMIT N.
3421    pub fn iter_asc(
3422        &self,
3423    ) -> alloc::boxed::Box<dyn Iterator<Item = (&IndexKey, &crate::posting::PostingList)> + '_>
3424    {
3425        match &self.kind {
3426            IndexKind::BTree(m) => alloc::boxed::Box::new(m.iter()),
3427            IndexKind::Nsw(_)
3428            | IndexKind::Brin { .. }
3429            | IndexKind::Gin(_)
3430            | IndexKind::GinTrgm(_)
3431            | IndexKind::GinFulltext(_)
3432            | IndexKind::GinJsonb(_) => alloc::boxed::Box::new(core::iter::empty()),
3433        }
3434    }
3435
3436    /// Look up the locators stored under `key` (B-tree only). Returns
3437    /// an empty slice when the key is absent or the index isn't a
3438    /// BTree — callers can treat both cases uniformly.
3439    ///
3440    /// v5.1: return type widened from `&[usize]` to `&[RowLocator]`.
3441    /// Pre-v5.2 callers can read the slice and `.as_hot().unwrap()`
3442    /// each entry (no `Cold` variants exist until the freezer lands);
3443    /// post-v5.2 callers dispatch hot vs. cold per locator.
3444    pub fn lookup_eq(&self, key: &IndexKey) -> &crate::posting::PostingList {
3445        match &self.kind {
3446            IndexKind::BTree(m) => m.get(key).map_or(&EMPTY_POSTINGS, |l| l),
3447            // BRIN / NSW / GIN / trigram-GIN / fulltext-GIN have
3448            // no IndexKey-keyed map; lookup is a no-op. GIN uses
3449            // [`Index::gin_lookup_word`] instead.
3450            IndexKind::Nsw(_)
3451            | IndexKind::Brin { .. }
3452            | IndexKind::Gin(_)
3453            | IndexKind::GinTrgm(_)
3454            | IndexKind::GinFulltext(_)
3455            | IndexKind::GinJsonb(_) => &EMPTY_POSTINGS,
3456        }
3457    }
3458
3459    /// v7.37.43 (INSUBQ B-2) — specialised lookup for integer-PK probes.
3460    /// `try_count_star_pk_in_subquery_fast` already holds an `i64` (the
3461    /// inner survivor key); skip the `IndexKey::from_value` enum-dispatch
3462    /// trip and build the key inline. ~20 ns × N_survivors saved on
3463    /// the INSUBQ hot loop.
3464    #[inline]
3465    pub fn lookup_eq_i64(&self, n: i64) -> &crate::posting::PostingList {
3466        match &self.kind {
3467            IndexKind::BTree(m) => m.get(&IndexKey::Int(n)).map_or(&EMPTY_POSTINGS, |l| l),
3468            IndexKind::Nsw(_)
3469            | IndexKind::Brin { .. }
3470            | IndexKind::Gin(_)
3471            | IndexKind::GinTrgm(_)
3472            | IndexKind::GinFulltext(_)
3473            | IndexKind::GinJsonb(_) => &EMPTY_POSTINGS,
3474        }
3475    }
3476
3477    /// v7.38 (perf, index range scan) — flatten the row locators for every key
3478    /// in `[lo, hi]` (bounds per `core::ops::Bound`) via the BTree's `O(log N +
3479    /// k)` range walk. Returns `None` once more than `cap` locators accumulate
3480    /// — a "this range isn't selective enough, seq-scan instead" signal that
3481    /// stops a wide range from materialising a near-full table's worth of rows
3482    /// through the index. BTree only (other kinds → None).
3483    pub fn lookup_range_capped(
3484        &self,
3485        lo: core::ops::Bound<&IndexKey>,
3486        hi: core::ops::Bound<&IndexKey>,
3487        cap: usize,
3488    ) -> Option<Vec<RowLocator>> {
3489        self.lookup_range_capped_by(lo, hi, cap, |_| true)
3490    }
3491
3492    /// v7.39 (round 490) — the same range walk, but the caller decides
3493    /// which locators are worth carrying, and the cap counts only those.
3494    ///
3495    /// A BTree index holds one locator per row VERSION. On a churned table
3496    /// the dead versions are still in there: round 490 measured a
3497    /// 1000-row range handing back 61 000 locators after 60
3498    /// delete-and-reinsert cycles with the background vacuum switched off.
3499    /// Every caller then dropped the dead ones — the mutation paths and the
3500    /// SELECT range path all test `is_row_visible` and `continue` — but only
3501    /// after they had been collected into a `Vec`, sorted, and walked.
3502    ///
3503    /// Handing the predicate down means the walk keeps ~1000, and the cap
3504    /// (which exists so an index walk never costs more than the scan it
3505    /// replaces) is once again measured in rows a caller will actually look
3506    /// at. Round 461 had to add the dead count to the budget to stop the
3507    /// seek being refused outright; with the filter here that compensation
3508    /// is no longer needed.
3509    pub fn lookup_range_capped_by(
3510        &self,
3511        lo: core::ops::Bound<&IndexKey>,
3512        hi: core::ops::Bound<&IndexKey>,
3513        cap: usize,
3514        keep: impl Fn(RowLocator) -> bool,
3515    ) -> Option<Vec<RowLocator>> {
3516        match &self.kind {
3517            IndexKind::BTree(m) => {
3518                let mut out: Vec<RowLocator> = Vec::new();
3519                for (_, locs) in m.range(lo, hi) {
3520                    out.extend(locs.iter().copied().filter(|l| keep(*l)));
3521                    if out.len() > cap {
3522                        return None;
3523                    }
3524                }
3525                Some(out)
3526            }
3527            IndexKind::Nsw(_)
3528            | IndexKind::Brin { .. }
3529            | IndexKind::Gin(_)
3530            | IndexKind::GinTrgm(_)
3531            | IndexKind::GinFulltext(_)
3532            | IndexKind::GinJsonb(_) => None,
3533        }
3534    }
3535
3536    /// v7.39 (round 560) — the index range as (key, locator) pairs.
3537    ///
3538    /// `lookup_range_capped_by` throws the KEY away and returns only
3539    /// locators, so a query whose projection is exactly the indexed
3540    /// column still goes to the row store for a value the walk already
3541    /// had in hand — paying per row for something the index knows.
3542    ///
3543    /// Uncapped on purpose: an index-only walk touches no row, so the
3544    /// selectivity ceiling that keeps a seek from being worse than the
3545    /// scan it replaces does not apply to it.
3546    ///
3547    /// v7.39 (round 562) — and it does not collect, either. This
3548    /// returned a `Vec<(IndexKey, RowLocator)>`: for a 100k-row range,
3549    /// 100k key clones into a `Vec::new()` that doubles its way up to
3550    /// several MB, all to be walked once and dropped. A profile of the
3551    /// server serving that query put 20% of the connection thread's CPU
3552    /// on the collect alone, with another 18% in the allocator beside
3553    /// it. The caller consumes the pairs in order and needs the key
3554    /// only by reference, so it can have the walk itself.
3555    pub fn range_keyed(
3556        &self,
3557        lo: core::ops::Bound<&IndexKey>,
3558        hi: core::ops::Bound<&IndexKey>,
3559    ) -> Option<impl Iterator<Item = (&IndexKey, RowLocator)> + '_> {
3560        match &self.kind {
3561            IndexKind::BTree(m) => Some(
3562                m.range(lo, hi)
3563                    .flat_map(|(k, locs)| locs.iter().map(move |l| (k, *l))),
3564            ),
3565            IndexKind::Nsw(_)
3566            | IndexKind::Brin { .. }
3567            | IndexKind::Gin(_)
3568            | IndexKind::GinTrgm(_)
3569            | IndexKind::GinFulltext(_)
3570            | IndexKind::GinJsonb(_) => None,
3571        }
3572    }
3573
3574    /// v7.12.3 — GIN posting-list lookup. Returns the row locators
3575    /// whose `tsvector` cell contains `word`. Empty when the word is
3576    /// absent from the index or this isn't a GIN index.
3577    pub fn gin_lookup_word(&self, word: &str) -> &crate::posting::PostingList {
3578        match &self.kind {
3579            // v7.17.0 Phase 2.2 — fulltext-GIN shares the same
3580            // lexeme-keyed posting list shape as the
3581            // tsvector-typed GIN, so the same lookup applies.
3582            IndexKind::Gin(m) | IndexKind::GinFulltext(m) => {
3583                m.get(&String::from(word)).map_or(&EMPTY_POSTINGS, |l| l)
3584            }
3585            IndexKind::BTree(_)
3586            | IndexKind::Nsw(_)
3587            | IndexKind::Brin { .. }
3588            | IndexKind::GinTrgm(_)
3589            | IndexKind::GinJsonb(_) => &EMPTY_POSTINGS,
3590        }
3591    }
3592
3593    /// v7.15.0 — trigram-GIN posting-list lookup. Returns the row
3594    /// locators whose indexed `TEXT` cell contains the trigram
3595    /// `tri`. Empty when the trigram is absent or this isn't a
3596    /// trigram-GIN index.
3597    pub fn gin_trgm_lookup(&self, tri: &str) -> &crate::posting::PostingList {
3598        match &self.kind {
3599            IndexKind::GinTrgm(m) => m.get(&String::from(tri)).map_or(&EMPTY_POSTINGS, |l| l),
3600            IndexKind::BTree(_)
3601            | IndexKind::Nsw(_)
3602            | IndexKind::Brin { .. }
3603            | IndexKind::Gin(_)
3604            | IndexKind::GinFulltext(_)
3605            | IndexKind::GinJsonb(_) => &EMPTY_POSTINGS,
3606        }
3607    }
3608
3609    /// v7.37.8(sentori Epic 5 P2)— JSONB-GIN posting-list lookup.
3610    /// Returns the row locators whose indexed JSONB cell carries
3611    /// the canonical `token`(see [`crate::jsonb_gin::extract_tokens`]).
3612    /// Empty when the token is absent or this isn't a JSONB-GIN
3613    /// index. Planners drive `<col> @> <jsonb_literal>` through here.
3614    pub fn gin_jsonb_lookup(&self, token: &str) -> &crate::posting::PostingList {
3615        match &self.kind {
3616            IndexKind::GinJsonb(m) => m.get(&String::from(token)).map_or(&EMPTY_POSTINGS, |l| l),
3617            IndexKind::BTree(_)
3618            | IndexKind::Nsw(_)
3619            | IndexKind::Brin { .. }
3620            | IndexKind::Gin(_)
3621            | IndexKind::GinTrgm(_)
3622            | IndexKind::GinFulltext(_) => &EMPTY_POSTINGS,
3623        }
3624    }
3625
3626    /// Borrow the NSW graph (if this is an NSW index). Callers that need
3627    /// the graph for a kNN search go through here.
3628    pub const fn nsw(&self) -> Option<&NswGraph> {
3629        match &self.kind {
3630            IndexKind::Nsw(g) => Some(g),
3631            IndexKind::BTree(_)
3632            | IndexKind::Brin { .. }
3633            | IndexKind::Gin(_)
3634            | IndexKind::GinTrgm(_)
3635            | IndexKind::GinFulltext(_)
3636            | IndexKind::GinJsonb(_) => None,
3637        }
3638    }
3639
3640    /// v6.7.1 — true when this index is a BRIN (block range) index.
3641    /// Used by the segment encoder to opt into BRIN sidecar emission
3642    /// at freeze time, and by the planner to opt into page-skipping
3643    /// on range predicates.
3644    pub const fn is_brin(&self) -> bool {
3645        matches!(self.kind, IndexKind::Brin { .. })
3646    }
3647
3648    /// v7.15.0 — true when this index is a trigram GIN
3649    /// (`gin_trgm_ops`-flavoured). Used by the LIKE planner to
3650    /// opt into trigram acceleration.
3651    pub const fn is_gin_trgm(&self) -> bool {
3652        matches!(self.kind, IndexKind::GinTrgm(_))
3653    }
3654
3655    /// v7.12.3 — true when this index is a GIN inverted index.
3656    /// Used by the planner to opt into posting-list acceleration on
3657    /// `WHERE col @@ tsquery` predicates.
3658    pub const fn is_gin(&self) -> bool {
3659        matches!(self.kind, IndexKind::Gin(_))
3660    }
3661
3662    /// v7.17.0 Phase 2.2 — true when this index is a fulltext
3663    /// GIN over a TEXT / VARCHAR column (MySQL `FULLTEXT KEY`
3664    /// surface). Used by the planner to opt the FULLTEXT-indexed
3665    /// column into MATCH AGAINST acceleration.
3666    pub const fn is_gin_fulltext(&self) -> bool {
3667        matches!(self.kind, IndexKind::GinFulltext(_))
3668    }
3669
3670    /// v7.37.8(sentori Epic 5 P2)— true when this index is a
3671    /// real JSONB-GIN(posting-list backed). Used by the planner
3672    /// to opt `<col> @> <jsonb_literal>` into posting-list seek.
3673    pub const fn is_gin_jsonb(&self) -> bool {
3674        matches!(self.kind, IndexKind::GinJsonb(_))
3675    }
3676}
3677
3678/// In-memory table: schema + a persistent row vector + secondary indices.
3679///
3680/// v4.39: `rows` is a [`PersistentVec`] (Bitmapped Vector Trie, 32-way) so
3681/// `Table::clone()` is `O(1)` — the whole reason for v4.39's existence is
3682/// to make `Catalog::clone()` cheap inside the v4.34 auto-commit wrap.
3683///
3684/// v5.2.1: `hot_bytes` tracks the encoded byte size of every row currently
3685/// in [`Self::rows`], summed over rows. Updated incrementally by `insert`
3686/// (+= encoded row size), `delete_rows` (-= removed rows' encoded sizes),
3687/// and `update_row` (-= old size, += new size). The value is what the
3688/// v5.2 freezer reads to decide when to demote cold rows — when the
3689/// catalog-wide sum crosses `SPG_HOT_TIER_BYTES` (default 4 GiB) the
3690/// freezer thread wakes. v5.2.1 ships measurement only; the freezer
3691/// itself lands in v5.2.2. Stored as `u64` so a single field clone in
3692/// `Catalog::clone` stays at the O(1) invariant v4.39 built.
3693/// v7.34 (crash-recovery P0 #2) — one row-level physical redo record.
3694/// Row-level redo replaces statement-based WAL replay (which re-executes
3695/// each SQL through the full engine — O(records × catalog_rows), the
3696/// superlinear recovery hang root-caused on the mailrs crash-recovery
3697/// P0). A `RowChange` is the exact storage mutation the engine applied
3698/// (`Table::insert` / `update_row` / `delete_rows`); replaying it on a
3699/// catalog restored from the matching checkpoint reproduces the state
3700/// WITHOUT re-validating uniqueness/FK/parse/plan — O(changed rows).
3701///
3702/// Positions are physical, not key-based: `serialize`/`deserialize`
3703/// preserve row order exactly (rows written + read back in `self.rows`
3704/// order) and the mutation ops are deterministic, so the same op sequence
3705/// replayed from the same checkpoint reproduces the same positions. This
3706/// matches PostgreSQL's physical redo and supports tables with no primary
3707/// key. (Caveat handled at replay integration: a post-checkpoint cold-tier
3708/// freeze shifts hot positions and must itself be logged or fenced by a
3709/// checkpoint — see `row-level-redo-design`.)
3710/// ## v7.37.15 (Epic W slice 1) — additive MVCC identity metadata
3711///
3712/// Each variant now also carries, additively, the stable
3713/// [`RowId`](row_header::RowId) of the affected row(s) and the
3714/// **writer version** (`xmin` for an insert, `xmax` for a
3715/// delete/update). This is the codec foundation for making
3716/// in-place MVCC tombstones durable across crash/upgrade recovery.
3717///
3718/// Two important properties for the durability path:
3719///
3720/// 1. **Replay resolution is UNCHANGED.** `apply_redo_run_on_table`
3721///    still resolves every change by physical `pos`/`positions`
3722///    exactly as before. The new metadata is *carried but unused*
3723///    by replay in this slice; resolving-by-`RowId` and
3724///    header-preserving replay are later slices.
3725/// 2. **Backward compatibility.** A redo payload written by
3726///    pre-Epic-W code carries no metadata; [`decode_redo_log`]
3727///    fills `rowid`/`rowids` with [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED)
3728///    (empty for `Delete`) and `writer_version` with `0`. See the
3729///    codec version gate in [`encode_redo_log`]/[`decode_redo_log`].
3730///
3731/// The `writer_version` is captured as `0` at the storage layer
3732/// (`Table::insert`/`delete_rows`/`update_row` don't have the
3733/// committing `TxId`), then **stamped with the real committing
3734/// version by the engine** after it drains the statement's changes
3735/// (Epic W slice 2 — [`RowChange::set_writer_version`], driven from
3736/// `Engine::writer_version_for_current_stmt`). All changes from one
3737/// statement share the one version. Replay still resolves by
3738/// physical position and does not read `writer_version` — that is a
3739/// later slice (header-preserving replay).
3740#[derive(Debug, Clone, PartialEq)]
3741pub enum RowChange {
3742    /// Append `row` to `table`.
3743    Insert {
3744        table: String,
3745        row: Row<'static>,
3746        /// Epic W: stable id the appended row will receive.
3747        /// [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED) when
3748        /// decoded from a pre-Epic-W redo payload.
3749        rowid: row_header::RowId,
3750        /// Epic W: writer version (`xmin`). `0` until the writing
3751        /// `TxId` is threaded to the storage layer (later slice).
3752        writer_version: u64,
3753    },
3754    /// Replace the row at physical `pos` in `table` with `new_row`.
3755    Update {
3756        table: String,
3757        pos: usize,
3758        new_row: Vec<Value<'static>>,
3759        /// Epic W: stable id of the row at `pos`.
3760        /// [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED) when
3761        /// decoded from a pre-Epic-W redo payload.
3762        rowid: row_header::RowId,
3763        /// Epic W: writer version (`xmax` of the superseded tuple).
3764        /// `0` until the writing `TxId` is threaded (later slice).
3765        writer_version: u64,
3766    },
3767    /// Remove the rows at the given physical `positions` from `table`.
3768    Delete {
3769        table: String,
3770        positions: Vec<usize>,
3771        /// Epic W: stable ids parallel to `positions` (same length,
3772        /// [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED) for an
3773        /// out-of-bounds input position). **Empty** when decoded from
3774        /// a pre-Epic-W redo payload (no metadata was recorded).
3775        rowids: Vec<row_header::RowId>,
3776        /// Epic W: writer version (`xmax`). `0` until the writing
3777        /// `TxId` is threaded to the storage layer (later slice).
3778        writer_version: u64,
3779    },
3780    /// v7.37.15 (Epic W durable-tombstone slice) — an **in-place MVCC
3781    /// delete**: the row(s) named by `rowids` are NOT physically
3782    /// removed; their header `xmax` is stamped so newer snapshots stop
3783    /// seeing them (vacuum reclaims later). This is the redo shape of
3784    /// the gate-on (`SPG_MVCC_INPLACE`) DELETE / UPDATE-old-version /
3785    /// ON-CONFLICT paths, which call [`Table::mark_row_deleted`]
3786    /// instead of `delete_rows`.
3787    ///
3788    /// Unlike `Delete`, the target is named by **stable `RowId`**, not
3789    /// physical position: a tombstone keeps the slot, so position would
3790    /// be ambiguous after later compaction, and the header-preserving
3791    /// replay must re-find the exact row the writer tombstoned. On
3792    /// replay the id is matched against the ids the same redo run
3793    /// produced (an `Insert`'s `rowid`, or the table's ids snapshotted
3794    /// at run start); an id that cannot be resolved is skipped and
3795    /// counted (see `apply_redo_run_on_table`) — this is the documented
3796    /// cross-checkpoint limitation until the V6 envelope persists ids.
3797    Tombstone {
3798        table: String,
3799        /// Stable ids of the tombstoned rows (from `self.rowids()[pos]`
3800        /// at capture). Never empty for a recorded tombstone.
3801        rowids: Vec<row_header::RowId>,
3802        /// The version stamped into each target row's header `xmax`
3803        /// (the deleting statement's writer version).
3804        xmax: u64,
3805    },
3806}
3807
3808impl RowChange {
3809    /// v7.39 (round 736) — which table this change applies to.
3810    #[must_use]
3811    pub fn table_name(&self) -> &str {
3812        match self {
3813            Self::Insert { table, .. }
3814            | Self::Update { table, .. }
3815            | Self::Delete { table, .. }
3816            | Self::Tombstone { table, .. } => table,
3817        }
3818    }
3819
3820    /// v7.37.15 (Epic W slice 2) — stamp the committing writer
3821    /// version onto this change. Every change drained from a single
3822    /// statement shares one version (the statement's `xmin`/`xmax`),
3823    /// so the engine calls this on each drained change with the value
3824    /// from [`Engine::writer_version_for_current_stmt`]. Additive
3825    /// metadata only: replay still resolves by physical position and
3826    /// does not read `writer_version` (that is a later slice).
3827    pub fn set_writer_version(&mut self, v: u64) {
3828        match self {
3829            RowChange::Insert { writer_version, .. }
3830            | RowChange::Update { writer_version, .. }
3831            | RowChange::Delete { writer_version, .. } => *writer_version = v,
3832            // A tombstone captures `xmax` directly from the deleting
3833            // statement's version at record time (via
3834            // `mark_row_deleted`), so it already equals `v`. Keep the
3835            // "one statement, one version" invariant mechanical by
3836            // asserting agreement in debug builds rather than silently
3837            // overwriting a possibly-different value.
3838            RowChange::Tombstone { xmax, .. } => {
3839                debug_assert_eq!(
3840                    *xmax, v,
3841                    "tombstone xmax must match the statement writer version"
3842                );
3843                *xmax = v;
3844            }
3845        }
3846    }
3847}
3848
3849/// v7.37.15 (Epic W slice 1) — leading marker byte of the
3850/// metadata-carrying redo layout. A **pre-Epic-W** redo payload leads
3851/// with `FILE_VERSION` (8..=52 today, rising ~1 per release); this
3852/// marker is `0xFF` and can therefore never collide with a real
3853/// `FILE_VERSION`, so [`decode_redo_log`] tells the two layouts apart
3854/// by inspecting the first byte alone. The compile-time assertion
3855/// below makes the "never collide" invariant a hard build gate: if
3856/// `FILE_VERSION` ever climbs toward `0xFF` the build breaks and forces
3857/// a redesign long before an ambiguity could ship.
3858const REDO_META_MARKER: u8 = 0xFF;
3859/// v7.37.15 (Epic W slice 1) — version of the metadata-carrying redo
3860/// layout that follows [`REDO_META_MARKER`]. Bumped when the per-change
3861/// metadata shape changes; an unknown value is a hard decode error.
3862const REDO_META_VERSION: u8 = 1;
3863
3864/// v7.37.15 (Epic W durable-tombstone slice) — process-wide count of
3865/// [`RowChange::Tombstone`] targets that `apply_redo` could NOT resolve
3866/// to a row by `RowId`. A non-zero value is expected only across a
3867/// checkpoint boundary (the table's ids are reassigned on deserialize
3868/// and the V6 envelope does not yet persist them), where a tombstone
3869/// naming a pre-checkpoint row is left visible rather than mis-applied.
3870/// Surfaced for observability; never affects correctness of the resolved
3871/// tombstones. Read via [`unresolved_tombstone_count`].
3872static UNRESOLVED_TOMBSTONES: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
3873
3874/// v7.39 (flip crash-replay P0) — observability read for the replay
3875/// tombstones that could not be resolved to a row (each one is a
3876/// resurrected delete).
3877#[must_use]
3878pub fn unresolved_tombstones() -> u64 {
3879    UNRESOLVED_TOMBSTONES.load(core::sync::atomic::Ordering::Relaxed)
3880}
3881
3882/// v7.37.15 (Epic W durable-tombstone slice) — read the process-wide
3883/// count of redo tombstones that could not be resolved to a row by
3884/// `RowId` during `apply_redo`. See [`UNRESOLVED_TOMBSTONES`].
3885#[must_use]
3886pub fn unresolved_tombstone_count() -> u64 {
3887    UNRESOLVED_TOMBSTONES.load(core::sync::atomic::Ordering::Relaxed)
3888}
3889// Provably-unambiguous old/new distinction: the pre-Epic-W layout's
3890// first byte is `FILE_VERSION`, which must stay strictly below the
3891// marker forever.
3892const _: () = assert!(FILE_VERSION < REDO_META_MARKER);
3893
3894/// v7.34 (crash-recovery P0 #2), extended v7.37.15 (Epic W slice 1) —
3895/// encode a row-level redo log to bytes for a WAL record.
3896///
3897/// ## Layout (Epic W metadata-carrying form, always emitted now)
3898///
3899/// `[u8 REDO_META_MARKER=0xFF][u8 REDO_META_VERSION][u8 FILE_VERSION]
3900/// [u32 count]` then per change `[u8 op][str table]` and, per op:
3901/// - `Insert [u32 n][value×n][u64 rowid][u64 writer_version]`
3902/// - `Update [u32 pos][u32 n][value×n][u64 rowid][u64 writer_version]`
3903/// - `Delete [u32 n][u32 pos×n][u64 rowid×n][u64 writer_version]`
3904/// - `Tombstone [u32 n][u64 rowid×n][u64 xmax]` (op byte 3; only ever
3905///   emitted under the metadata-carrying layout — the pre-Epic-W layout
3906///   had no in-place tombstone, so a legacy stream can never carry it)
3907///
3908/// Positions are physical (u32 ≤ 4 G rows). The `FILE_VERSION` byte
3909/// still rides along (now the 3rd byte) so the value codec decodes
3910/// string / BYTEA escapes exactly as before.
3911///
3912/// ## Backward compatibility
3913///
3914/// The **pre-Epic-W** layout was `[u8 FILE_VERSION][u32 count]…` with
3915/// no per-change metadata. [`decode_redo_log`] still decodes that form
3916/// (first byte < `0xFF`) byte-for-byte identically — every WAL file
3917/// written by released code replays unchanged.
3918#[must_use]
3919pub fn encode_redo_log(changes: &[RowChange]) -> Vec<u8> {
3920    let mut out = Vec::new();
3921    out.push(REDO_META_MARKER);
3922    out.push(REDO_META_VERSION);
3923    out.push(FILE_VERSION);
3924    codec::write_u32(&mut out, changes.len() as u32);
3925    let write_values = |out: &mut Vec<u8>, vals: &[Value<'static>]| {
3926        codec::write_u32(out, vals.len() as u32);
3927        for v in vals {
3928            codec::write_value(out, v);
3929        }
3930    };
3931    for change in changes {
3932        match change {
3933            RowChange::Insert {
3934                table,
3935                row,
3936                rowid,
3937                writer_version,
3938            } => {
3939                out.push(0);
3940                codec::write_str(&mut out, table);
3941                write_values(&mut out, &row.values);
3942                codec::write_u64(&mut out, rowid.0);
3943                codec::write_u64(&mut out, *writer_version);
3944            }
3945            RowChange::Update {
3946                table,
3947                pos,
3948                new_row,
3949                rowid,
3950                writer_version,
3951            } => {
3952                out.push(1);
3953                codec::write_str(&mut out, table);
3954                codec::write_u32(&mut out, *pos as u32);
3955                write_values(&mut out, new_row);
3956                codec::write_u64(&mut out, rowid.0);
3957                codec::write_u64(&mut out, *writer_version);
3958            }
3959            RowChange::Delete {
3960                table,
3961                positions,
3962                rowids,
3963                writer_version,
3964            } => {
3965                out.push(2);
3966                codec::write_str(&mut out, table);
3967                codec::write_u32(&mut out, positions.len() as u32);
3968                for p in positions {
3969                    codec::write_u32(&mut out, *p as u32);
3970                }
3971                // Epic W: one RowId per position (parallel). Capture
3972                // sites always produce `rowids.len() == positions.len()`;
3973                // this assertion pins that invariant at encode time so a
3974                // mismatch is a loud bug, not a silently short payload.
3975                debug_assert_eq!(
3976                    rowids.len(),
3977                    positions.len(),
3978                    "redo Delete: rowids must be parallel to positions"
3979                );
3980                for rid in rowids {
3981                    codec::write_u64(&mut out, rid.0);
3982                }
3983                codec::write_u64(&mut out, *writer_version);
3984            }
3985            RowChange::Tombstone {
3986                table,
3987                rowids,
3988                xmax,
3989            } => {
3990                out.push(3);
3991                codec::write_str(&mut out, table);
3992                codec::write_u32(&mut out, rowids.len() as u32);
3993                for rid in rowids {
3994                    codec::write_u64(&mut out, rid.0);
3995                }
3996                codec::write_u64(&mut out, *xmax);
3997            }
3998        }
3999    }
4000    out
4001}
4002
4003/// v7.34, extended v7.37.15 (Epic W slice 1) — decode a row-level redo
4004/// log written by [`encode_redo_log`].
4005///
4006/// Decodes **both** the Epic W metadata-carrying layout (first byte
4007/// `REDO_META_MARKER = 0xFF`) and the pre-Epic-W layout (first byte is
4008/// `FILE_VERSION`, always `< 0xFF`). For the old layout the per-change
4009/// metadata is absent, so `rowid`/`rowids` come back
4010/// [`RowId::UNASSIGNED`](row_header::RowId::UNASSIGNED) (empty for
4011/// `Delete`) and `writer_version` comes back `0`.
4012///
4013/// A truncated / corrupt buffer is a hard error — never a panic — the
4014/// embedding layer frames each record with its own length + CRC, so a
4015/// frame that decodes short is corruption, not a torn tail.
4016pub fn decode_redo_log(bytes: &[u8]) -> Result<Vec<RowChange>, StorageError> {
4017    let first = *bytes
4018        .first()
4019        .ok_or_else(|| StorageError::Corrupt("redo log: empty".into()))?;
4020    // Epic W: `0xFF` marker ⇒ metadata-carrying layout; anything else
4021    // is a pre-Epic-W `FILE_VERSION` byte (old layout, no metadata).
4022    let has_meta = first == REDO_META_MARKER;
4023    let (codec_version, header_len) = if has_meta {
4024        let meta_version = *bytes
4025            .get(1)
4026            .ok_or_else(|| StorageError::Corrupt("redo log: short header".into()))?;
4027        if meta_version != REDO_META_VERSION {
4028            return Err(StorageError::Corrupt(alloc::format!(
4029                "redo log: unknown metadata version {meta_version}"
4030            )));
4031        }
4032        let file_version = *bytes
4033            .get(2)
4034            .ok_or_else(|| StorageError::Corrupt("redo log: short header".into()))?;
4035        // header = [marker][meta_version][file_version]
4036        (file_version, 3usize)
4037    } else {
4038        // Old layout: the first byte IS the FILE_VERSION.
4039        (first, 1usize)
4040    };
4041    let mut cur = codec::Cursor::new(bytes).with_codec_version(codec_version);
4042    for _ in 0..header_len {
4043        cur.read_u8()?;
4044    }
4045    let count = cur.read_u32()? as usize;
4046    let mut read_values =
4047        |cur: &mut codec::Cursor<'_>| -> Result<Vec<Value<'static>>, StorageError> {
4048            let n = cur.read_u32()? as usize;
4049            let mut vals = Vec::with_capacity(n);
4050            for _ in 0..n {
4051                vals.push(cur.read_value()?);
4052            }
4053            Ok(vals)
4054        };
4055    let mut changes = Vec::with_capacity(count);
4056    for _ in 0..count {
4057        let op = cur.read_u8()?;
4058        let table = cur.read_str()?;
4059        let change = match op {
4060            0 => {
4061                let row = Row::new(read_values(&mut cur)?);
4062                let (rowid, writer_version) = if has_meta {
4063                    (row_header::RowId(cur.read_u64()?), cur.read_u64()?)
4064                } else {
4065                    (row_header::RowId::UNASSIGNED, 0)
4066                };
4067                RowChange::Insert {
4068                    table,
4069                    row,
4070                    rowid,
4071                    writer_version,
4072                }
4073            }
4074            1 => {
4075                let pos = cur.read_u32()? as usize;
4076                let new_row = read_values(&mut cur)?;
4077                let (rowid, writer_version) = if has_meta {
4078                    (row_header::RowId(cur.read_u64()?), cur.read_u64()?)
4079                } else {
4080                    (row_header::RowId::UNASSIGNED, 0)
4081                };
4082                RowChange::Update {
4083                    table,
4084                    pos,
4085                    new_row,
4086                    rowid,
4087                    writer_version,
4088                }
4089            }
4090            2 => {
4091                let n = cur.read_u32()? as usize;
4092                let mut positions = Vec::with_capacity(n);
4093                for _ in 0..n {
4094                    positions.push(cur.read_u32()? as usize);
4095                }
4096                let (rowids, writer_version) = if has_meta {
4097                    let mut rowids = Vec::with_capacity(n);
4098                    for _ in 0..n {
4099                        rowids.push(row_header::RowId(cur.read_u64()?));
4100                    }
4101                    (rowids, cur.read_u64()?)
4102                } else {
4103                    // Old layout carried no RowId metadata.
4104                    (Vec::new(), 0)
4105                };
4106                RowChange::Delete {
4107                    table,
4108                    positions,
4109                    rowids,
4110                    writer_version,
4111                }
4112            }
4113            // Op 3 is the Epic W in-place tombstone — it only exists in
4114            // the metadata-carrying layout. Guarding on `has_meta` means
4115            // a legacy stream that happens to contain a `3` byte here is
4116            // reported as an unknown op (corruption), never mis-decoded.
4117            3 if has_meta => {
4118                let n = cur.read_u32()? as usize;
4119                let mut rowids = Vec::with_capacity(n);
4120                for _ in 0..n {
4121                    rowids.push(row_header::RowId(cur.read_u64()?));
4122                }
4123                let xmax = cur.read_u64()?;
4124                RowChange::Tombstone {
4125                    table,
4126                    rowids,
4127                    xmax,
4128                }
4129            }
4130            other => {
4131                return Err(StorageError::Corrupt(alloc::format!(
4132                    "redo log: unknown op {other}"
4133                )));
4134            }
4135        };
4136        changes.push(change);
4137    }
4138    Ok(changes)
4139}
4140
4141/// v7.39 (pg_stat knife B) — per-table scan counters, bumped from
4142/// `&self` read paths. Clone (tx shadow catalogs clone tables) copies
4143/// the current values; the counters are volatile like PG's cumulative
4144/// stats.
4145#[derive(Debug, Default)]
4146pub struct ScanStats {
4147    pub seq_scan: core::sync::atomic::AtomicU64,
4148    pub seq_tup_read: core::sync::atomic::AtomicU64,
4149    pub idx_scan: core::sync::atomic::AtomicU64,
4150    pub idx_tup_fetch: core::sync::atomic::AtomicU64,
4151}
4152
4153impl Clone for ScanStats {
4154    fn clone(&self) -> Self {
4155        use core::sync::atomic::{AtomicU64, Ordering};
4156        Self {
4157            seq_scan: AtomicU64::new(self.seq_scan.load(Ordering::Relaxed)),
4158            seq_tup_read: AtomicU64::new(self.seq_tup_read.load(Ordering::Relaxed)),
4159            idx_scan: AtomicU64::new(self.idx_scan.load(Ordering::Relaxed)),
4160            idx_tup_fetch: AtomicU64::new(self.idx_tup_fetch.load(Ordering::Relaxed)),
4161        }
4162    }
4163}
4164
4165/// v7.39 (round 215) — the lower-bound sort key for a range value, used by
4166/// the range-exclusion index. The bound as an `i128` (unbounded lower =
4167/// `i128::MIN`, sorting first) plus an inclusivity rank (inclusive lower
4168/// sorts before exclusive at the same value, `[3` before `(3`). Returns
4169/// `None` for range kinds whose bound isn't an integer scalar (numrange's
4170/// numeric/bignum), for empty ranges, and for non-range values — the caller
4171/// then keeps the O(n) scan rather than risk an unsound order. Int4/Int8/
4172/// Date/Ts/TsTz all reduce here (tstzrange bounds are `Value::Timestamp`).
4173/// Maintenance (index build) and query (overlap probe) MUST agree on this
4174/// key, so both sides call exactly this function.
4175#[must_use]
4176pub fn range_excl_index_key(v: &Value<'_>) -> Option<(i128, u8)> {
4177    let Value::Range {
4178        lower,
4179        lower_inc,
4180        empty,
4181        ..
4182    } = v
4183    else {
4184        return None;
4185    };
4186    if *empty {
4187        return None;
4188    }
4189    let key = match lower {
4190        None => i128::MIN,
4191        Some(b) => match b.as_ref() {
4192            Value::SmallInt(n) => i128::from(*n),
4193            Value::Int(n) => i128::from(*n),
4194            Value::BigInt(n) => i128::from(*n),
4195            Value::Date(n) => i128::from(*n),
4196            Value::Timestamp(n) => i128::from(*n),
4197            _ => return None,
4198        },
4199    };
4200    Some((key, u8::from(!*lower_inc)))
4201}
4202
4203/// v7.39 (round 215) — a per-table range-exclusion index: an incrementally
4204/// maintained map from a range column's lower-bound key
4205/// ([`range_excl_index_key`]) to the physical row locators carrying that
4206/// bound. Lets EXCLUDE enforcement find the few candidate rows a new range
4207/// might overlap in O(log n) instead of scanning every row (measured O(N²),
4208/// r213). Because the stored ranges under a valid `EXCLUDE (col WITH &&)`
4209/// are pairwise disjoint, a candidate overlaps only its predecessor or the
4210/// successors whose lower bound precedes its upper — a handful of probes.
4211///
4212/// NOT persisted: rebuilt from the (persisted) exclusion constraints + rows
4213/// on catalog load, exactly like BRIN re-derives. Backed by a
4214/// `PersistentBTreeMap` so `Table::clone` (the per-write snapshot) stays
4215/// O(1). Locators to tombstoned rows are left in place and filtered by the
4216/// consumer via `is_deleted()` at query time — the established index pattern.
4217#[derive(Debug, Clone)]
4218pub struct ExclRangeIndex {
4219    /// The constrained range column's position in the table.
4220    pub column_position: usize,
4221    /// Lower-bound key → row locators. A key maps to a `Vec` because a
4222    /// tombstoned-then-reinserted bound can transiently collide; live rows
4223    /// under the constraint are disjoint so each key has one live locator.
4224    pub map: PersistentBTreeMap<(i128, u8), crate::posting::PostingList>,
4225}
4226
4227#[derive(Debug, Clone)]
4228pub struct Table {
4229    schema: TableSchema,
4230    /// v7.37.15 (Phase C.1) — stable per-catalog relation identity.
4231    /// [`RelId::UNASSIGNED`](row_header::RelId::UNASSIGNED) until
4232    /// `Catalog::create_table` (or the deserialize dense-assign pass)
4233    /// stamps a real id. Keys the Phase C.4 row-lock table and the
4234    /// Phase C.5 `RelationStore`; survives `DROP TABLE` slot shifts.
4235    rel_id: row_header::RelId,
4236    rows: PersistentVec<Row<'static>>,
4237    /// v7.37.15 (Phase A.2) — per-row MVCC visibility headers
4238    /// parallel to `rows`. `headers.len() == rows.len()` is the
4239    /// load-bearing invariant; debug builds assert it on every
4240    /// scan boundary, release builds rely on it from
4241    /// disciplined insert / delete / update paths.
4242    ///
4243    /// Pre-v7.37.15-loaded tables (every row currently in the
4244    /// fleet) start as `RowHeader::frozen()` — `is_all_visible_fast()`
4245    /// returns `true`, so the per-row visibility gate Phase B
4246    /// adds is a no-op against any snapshot.
4247    ///
4248    /// Headers are NOT yet serialised into the envelope at this
4249    /// commit — on snapshot deserialize every row gets a fresh
4250    /// `RowHeader::frozen()`. Phase D adds the visibility-map
4251    /// + segment-freeze story which makes serialisation
4252    /// meaningful; until then the on-disk story is "the catalog
4253    /// is the set of visible rows."
4254    headers: PersistentVec<row_header::RowHeader>,
4255    /// v7.37.15 (Phase C.1) — stable per-relation row identity
4256    /// parallel to `rows` / `headers`. `rowids[i]` is the never-
4257    /// reused [`RowId`](row_header::RowId) of the row physically at
4258    /// slot `i`; `rowids.len() == rows.len()` joins the same load-
4259    /// bearing lock-step invariant as `headers`. Compaction (delete
4260    /// / vacuum) rebuilds all three vecs together so the id travels
4261    /// with the row while the slot shifts.
4262    ///
4263    /// Introduced additively: allocated + kept lock-step, but index
4264    /// locators still address rows by physical slot at this commit.
4265    /// Later phases migrate the lock table (C.4), HOT chains (D),
4266    /// and the WAL (Epic W) to address by `RowId`.
4267    ///
4268    /// Not yet serialised into the envelope — on load every row is
4269    /// assigned a fresh dense id `1..=len` (see `next_rowid`), which
4270    /// is sufficient while the id is process-local bookkeeping. The
4271    /// V6 envelope (Phase C.6) will persist ids so a WAL redo can
4272    /// name a row across restart.
4273    rowids: PersistentVec<row_header::RowId>,
4274    /// v7.37.15 (Phase C.1) — per-relation monotonic allocator for
4275    /// `rowids`. Starts at 1 (0 is the `RowId::UNASSIGNED` sentinel);
4276    /// every append takes `next_rowid` then increments. Never reused
4277    /// even after the row is deleted / vacuumed, so a stale lock /
4278    /// redo reference can be detected rather than silently aliasing a
4279    /// later row that reused the slot.
4280    next_rowid: u64,
4281    /// v7.37.16 (autovacuum) — live count of tombstoned-but-present hot
4282    /// rows (`headers[i].xmax != XMAX_ALIVE`). Maintained incrementally:
4283    /// `mark_row_deleted` / `mark_rows_deleted` increment (the only
4284    /// tombstone producers), `delete_rows_no_index` recomputes over the
4285    /// survivors (it is the compaction hub every physical removal —
4286    /// including vacuum — flows through), and the v53 snapshot loader
4287    /// recounts verbatim-restored headers. Drives the engine's
4288    /// autovacuum threshold; not persisted (recomputed on load).
4289    dead_rows: u64,
4290    /// v7.39 (pg_stat knife A) — volatile per-table write counters
4291    /// backing `pg_stat_user_tables.n_tup_ins/upd/del`. Not persisted
4292    /// (PG's cumulative stats are shared-memory-volatile too — a
4293    /// restart zeroes them).
4294    stat_tup_ins: u64,
4295    stat_tup_upd: u64,
4296    stat_tup_del: u64,
4297    /// v7.39 (pg_stat knife B) — volatile scan counters
4298    /// (`seq_scan/seq_tup_read/idx_scan/idx_tup_fetch`). Atomics: the
4299    /// read paths that bump them hold only `&Table`.
4300    scan_stats: ScanStats,
4301    /// v7.39 (pg_stat knife C) — wall-clock stamps (unix µs, from the
4302    /// host ClockFn) for pg_stat_user_tables' last_autovacuum /
4303    /// last_analyze. Volatile, like PG's cumulative stats. SPG has no
4304    /// manual-VACUUM statement semantics, so last_vacuum stays NULL.
4305    last_autovacuum_us: Option<i64>,
4306    last_analyze_us: Option<i64>,
4307    indices: Vec<Index>,
4308    hot_bytes: u64,
4309    /// v6.7.0 — cached count of rows currently materialised in the
4310    /// cold tier via `RowLocator::Cold` entries across THIS table's
4311    /// indices. Populated by `ANALYZE` (walks every BTree index and
4312    /// counts Cold locators); the count survives until the next
4313    /// ANALYZE recomputes it. Surfaced via `spg_statistic.cold_row_count`
4314    /// and `spg_stat_segment.table_name`.
4315    ///
4316    /// Honest scope: this is a CACHED count, not a live one.
4317    /// Freezer / promote / DELETE don't currently update the cache
4318    /// incrementally — they invalidate it by setting the
4319    /// `cold_row_count_stale` flag, and the next ANALYZE re-walks.
4320    /// Incremental maintenance is a v6.7.x candidate if observation
4321    /// shows the ANALYZE walk cost dominates.
4322    cold_row_count: u64,
4323    /// v6.7.0 — set when the cached `cold_row_count` may be wrong
4324    /// because rows moved into / out of the cold tier since the last
4325    /// ANALYZE. The virtual-table surface reports the cached value
4326    /// regardless (operators run ANALYZE to refresh).
4327    cold_row_count_stale: bool,
4328    /// v7.34 (crash-recovery P0 #2) — row-level redo capture buffer.
4329    /// `None` (default, in-memory mode) captures nothing — zero overhead.
4330    /// `Some` (set by the engine when persistence is on, before a
4331    /// mutating call) makes `insert` / `update_row` / `delete_rows`
4332    /// record the physical [`RowChange`] they applied, which the engine
4333    /// drains after the statement and writes to the WAL in place of the
4334    /// SQL text. Transient: never serialized; a `Catalog::clone` between
4335    /// enable and drain copies it (cheap — empty in the steady state).
4336    redo_log: Option<Vec<RowChange>>,
4337    /// v7.39 (round 215) — per-`EXCLUDE`-constraint range-overlap indexes,
4338    /// one per single-`&&` constraint on an integer-keyable range column.
4339    /// Maintained incrementally on insert / update / rebuild (mirroring the
4340    /// BTree secondary indexes); NOT serialized — rebuilt from the schema's
4341    /// exclusion constraints on load. Empty for tables with no EXCLUDE
4342    /// constraint (the common case), so `Table::clone` pays nothing.
4343    excl_indexes: Vec<ExclRangeIndex>,
4344    /// v7.39 (round 493) — the snapshot floor below which a deleted row
4345    /// version is invisible to everyone, as of the statement now running.
4346    ///
4347    /// Runtime only: never serialised, and `0` (the default) prunes
4348    /// nothing, so any path that forgets to set it is merely slower, not
4349    /// wrong. The engine sets it from `vacuum_oldest_active()` — the same
4350    /// floor `vacuum` itself takes — before the statement's inserts.
4351    prune_horizon: u64,
4352}
4353
4354/// Catalog: insertion-ordered `Vec<Table>` for stable iter / serialize,
4355/// plus a `BTreeMap<String, usize>` sidecar index so `get` / `get_mut`
4356/// run in O(log n) instead of the old linear scan with per-element
4357/// string compares.
4358///
4359/// A pure `BTreeMap<String, Table>` was tried in an interim version
4360/// of v3.1.2 and regressed the single-table catalog benches by ~10%
4361/// (the per-element `BTreeMap` overhead outweighs the lookup win
4362/// when n is small). The sidecar shape preserves the insertion-order
4363/// iteration the on-disk encoding relies on and keeps `last_mut`
4364/// (used by the deserialize hot path) cheap.
4365/// v7.39 (pg_stat blks knife) — catalog-wide cold-tier read counter
4366/// backing pg_stat_database.blks_read. Row-granular (SPG has no 8 KB
4367/// page notion): one cold-segment row resolution = one "block read",
4368/// one hot row access = one "block hit" — the hit RATIO monitoring
4369/// dashboards compute keeps its meaning. Volatile like PG's stats.
4370#[derive(Debug, Default)]
4371pub struct ColdReadStats {
4372    pub cold_reads: core::sync::atomic::AtomicU64,
4373}
4374
4375impl Clone for ColdReadStats {
4376    fn clone(&self) -> Self {
4377        Self {
4378            cold_reads: core::sync::atomic::AtomicU64::new(
4379                self.cold_reads.load(core::sync::atomic::Ordering::Relaxed),
4380            ),
4381        }
4382    }
4383}
4384
4385#[derive(Debug, Clone, Default)]
4386pub struct Catalog {
4387    /// v7.39 (pg_stat blks knife) — see [`ColdReadStats`].
4388    pub cold_read_stats: ColdReadStats,
4389    tables: Vec<Table>,
4390    /// `name → tables[index]`. Kept in lock-step with `tables`.
4391    /// `create_table` is the only write path.
4392    by_name: BTreeMap<String, usize>,
4393    /// v7.39 (round 436) — the current session's temporary-table namespace.
4394    /// A temp table is stored under `<prefix><name>`, and every lookup tries
4395    /// that first: exactly PG's `pg_temp` search-path rule, and MySQL's
4396    /// "a TEMPORARY table shadows a permanent one of the same name".
4397    ///
4398    /// Process-local, never serialised: the engine sets it per session, and
4399    /// a catalog read back from disk starts with none. Kept here rather than
4400    /// at each of the ~170 engine call sites because `by_name` is private —
4401    /// this is the ONE place a table name becomes an index.
4402    temp_prefix: Option<String>,
4403    /// v7.39 (round 496) — the names of tables this catalog handle has had
4404    /// changed since the set was last cleared.
4405    ///
4406    /// Runtime only, never serialised. A transaction's shadow catalog
4407    /// clears it at BEGIN, so at COMMIT the set is exactly the tables the
4408    /// transaction changed — which is what lets a commit that cannot use
4409    /// the row-level merge install only those tables instead of the whole
4410    /// catalog, leaving another session's concurrent work in place.
4411    ///
4412    /// Recorded where the change actually happens (`get_mut`,
4413    /// `create_table`, `drop_table`) rather than from the statement
4414    /// classifier: round 494 tried classification for a correctness gate
4415    /// and it was wrong, because `SELECT lo_write(…)` reads as read-only.
4416    dirty_tables: alloc::collections::BTreeSet<String>,
4417    /// v7.37.15 (Phase C.1) — monotonic allocator for stable
4418    /// [`RelId`](row_header::RelId)s. Pre-incremented on each
4419    /// `create_table` so real ids start at 1 (0 is `UNASSIGNED`);
4420    /// never reused even after `DROP TABLE`, so a stale lock / redo
4421    /// reference is detectable. Process-local bookkeeping — not yet
4422    /// serialised; `deserialize` re-assigns dense ids on load (the
4423    /// V6 envelope, Phase C.6, will round-trip real ids).
4424    next_rel_id: u64,
4425    /// v5.1: in-memory cold-tier segments. Side-loaded via
4426    /// [`Catalog::load_segment_bytes`] — they live outside the
4427    /// catalog snapshot (caller persists them as separate files
4428    /// and re-loads on boot, until v5.3's `CatalogManifest` makes
4429    /// that wiring automatic). `RowLocator::Cold { segment_id, .. }`
4430    /// indexes this `Vec`. Cleared on `Catalog::new` / fresh
4431    /// `deserialize`.
4432    ///
4433    /// `Arc` wrap keeps `Catalog::clone` at O(N segments) bumps
4434    /// (rather than O(total segment bytes) memcpy) so the v4.42
4435    /// group-commit pre-image rollback invariant — clone is
4436    /// effectively free — survives the cold-tier addition.
4437    ///
4438    /// v6.7.3 — slots became `Option<…>` so cold-segment compaction
4439    /// can tombstone merged sources without breaking the
4440    /// `segment_id = index_into_vec` contract that on-disk
4441    /// `RowLocator::Cold { segment_id }` already serialized.
4442    /// `None` slot = the segment was retired by compaction; the
4443    /// physical file may still be on disk (next CHECKPOINT writes
4444    /// a manifest that no longer lists it, and the file becomes
4445    /// an orphan eligible for offline cleanup).
4446    cold_segments: Vec<Option<Arc<OwnedSegment>>>,
4447    /// v7.12.4 — user-defined functions (PL/pgSQL + SQL).
4448    /// Keyed by function name (PG overloading is out of scope).
4449    /// Bodies are stored as the raw source text the parser saw
4450    /// between `$$ ... $$`; the engine re-parses on each
4451    /// invocation. This keeps `spg-storage` free of `spg-sql`
4452    /// dependency — same pattern as partial-index predicates.
4453    functions: BTreeMap<String, FunctionDef>,
4454    /// v7.12.4 — triggers in insertion order. PG18-measured (round
4455    /// 753): PG fires same-event triggers in NAME order (a_trig
4456    /// before z_trig regardless of creation order); SPG fires in
4457    /// insertion order — a real divergence, ledgered as F31-B2.
4458    triggers: Vec<TriggerDef>,
4459    /// v7.39 (round 139) — query-rewrite RULEs, flat like triggers.
4460    rules: Vec<RuleDef>,
4461    /// v7.39 (round 280) — extended-statistics objects. Recorded so a
4462    /// pg_dump restores them and reflection reports them; the planner
4463    /// does not consult them yet.
4464    statistics_ext: Vec<StatisticsExtDef>,
4465    /// v7.39 (round 287) — server-side large objects, keyed by OID.
4466    /// PG stores them as 2 KB pages in `pg_largeobject`; the page split
4467    /// is a storage detail of ITS heap, so SPG holds the whole byte
4468    /// string and renders the pages on read. What must match is the
4469    /// observable surface: the OIDs, the bytes, and the page rows.
4470    large_objects: alloc::collections::BTreeMap<u32, Vec<u8>>,
4471    /// v7.17.0 — catalogued SEQUENCE objects (Phase 1.1). Each
4472    /// `nextval(name)` reaches in here, atomically increments
4473    /// `last_value` / flips `is_called`, returns the new value.
4474    /// Persisted in catalog FILE_VERSION 26+; older catalogs
4475    /// deserialise with an empty map.
4476    sequences: BTreeMap<String, SequenceDef>,
4477    /// v7.39 (read01 round 60) — the `public` schema's ACL (PG
4478    /// `pg_namespace.nspacl`). EMPTY = PG's default, which is not "nothing":
4479    /// PUBLIC holds USAGE and the owner holds USAGE + CREATE. Materialised on
4480    /// the first GRANT / REVOKE, exactly like a table's relacl.
4481    schema_acl: Vec<AclItem>,
4482    /// v7.39 (read01 round 60) — the database's ACL. EMPTY = PG's default:
4483    /// PUBLIC holds CONNECT + TEMPORARY, the owner holds all three.
4484    database_acl: Vec<AclItem>,
4485    /// v7.17.0 — catalogued VIEW objects (Phase 1.2). Each
4486    /// `SELECT FROM v` at engine exec-time looks up `v` here and
4487    /// prepends the view body as a synthetic CTE. Persisted in
4488    /// catalog FILE_VERSION 27+; older catalogs deserialise with
4489    /// an empty map.
4490    views: BTreeMap<String, ViewDef>,
4491    /// v7.17.0 — catalogued MATERIALIZED VIEW source registry
4492    /// (Phase 1.3). Maps name → SELECT source. The materialised
4493    /// rows themselves live as a regular `Table` with the same
4494    /// name; REFRESH re-parses + re-executes the source against
4495    /// the table. Persisted in catalog FILE_VERSION 28+;
4496    /// older catalogs deserialise with an empty map.
4497    materialized_views: BTreeMap<String, String>,
4498    /// v7.17.0 — catalogued user-defined ENUM types (Phase 1.4).
4499    /// Maps name → label list. Columns reference these by name
4500    /// via `ColumnSchema.user_enum_type`. Persisted in catalog
4501    /// FILE_VERSION 29+; older catalogs deserialise with an empty
4502    /// map.
4503    enum_types: BTreeMap<String, EnumDef>,
4504    /// v7.17.0 — catalogued user-defined DOMAIN types (Phase 1.5).
4505    /// Maps name → base + CHECK constraints. Columns reference
4506    /// these by name via `ColumnSchema.user_domain_type`.
4507    /// Persisted in catalog FILE_VERSION 30+; older catalogs
4508    /// deserialise with an empty map.
4509    domain_types: BTreeMap<String, DomainDef>,
4510    /// v7.39 (read01 round 50) — `COMMENT ON <kind> <obj> IS '…'` store.
4511    /// Keyed by a canonical `"<kind>:<name>"` string (`"table:t"`,
4512    /// `"column:t.c"`, `"index:i"`, `"view:v"`, …) so a new commentable
4513    /// object kind needs no schema change. `COMMENT … IS NULL` removes the
4514    /// entry. Persisted in catalog FILE_VERSION 61+; older catalogs
4515    /// deserialise with an empty map. Read back by obj_description /
4516    /// col_description and the pg_description view.
4517    comments: BTreeMap<String, String>,
4518    /// v7.39 (round 547) — PG's `pg_db_role_setting`: the GUC defaults
4519    /// `ALTER ROLE … SET` / `ALTER DATABASE … SET` record, applied when
4520    /// a session starts.
4521    ///
4522    /// Keyed exactly as PG keys it — `(database, role)` where an empty
4523    /// name is PG's oid 0, meaning "all". So `ALTER ROLE ALL SET` is
4524    /// `("", "")`, `ALTER DATABASE d SET` is `(d, "")`, `ALTER ROLE r
4525    /// SET` is `("", r)` and `ALTER ROLE r IN DATABASE d SET` is
4526    /// `(d, r)`. The value is that scope's parameter list.
4527    db_role_settings: BTreeMap<(String, String), BTreeMap<String, String>>,
4528    /// v7.39 (round 550) — replication slots, by name.
4529    ///
4530    /// A slot in PG is two things: a named record, and a reservation
4531    /// that holds WAL back. SPG keeps the record — which is what every
4532    /// setup script and monitoring query reads — and reports
4533    /// `wal_status = 'unreserved'`, PG's own word for a slot that no
4534    /// longer holds WAL. The whole family used to answer NULL and
4535    /// report success, so `pg_drop_replication_slot('nosuchslot')` said
4536    /// it worked and a setup script created nothing.
4537    ///
4538    /// Value: (plugin, slot_type). `plugin` is empty for a physical slot.
4539    replication_slots: BTreeMap<String, (String, String)>,
4540    /// v7.37.42-T2 ζ-B — catalogued user-defined COMPOSITE types
4541    /// (`CREATE TYPE name AS (field_name field_type, …)`). Columns
4542    /// reference these by name via
4543    /// `ColumnSchema.user_composite_type` (parallel to
4544    /// `user_enum_type` / `user_domain_type`). Persisted in catalog
4545    /// FILE_VERSION 52+; older catalogs deserialise with an empty
4546    /// map.
4547    composite_types: BTreeMap<String, CompositeDef>,
4548    /// v7.17.0 — schema-namespace registry (Phase 1.6). Tracks
4549    /// which schemas exist. `public`, `pg_catalog`, and
4550    /// `information_schema` are built-in and always present.
4551    /// Schema-qualified table references still strip the prefix
4552    /// at lookup time per v7.16-and-earlier — full
4553    /// schema-as-isolation is v7.18+ scope. Persisted in catalog
4554    /// FILE_VERSION 31+; older catalogs deserialise with just
4555    /// the built-ins.
4556    schemas: alloc::collections::BTreeSet<String>,
4557}
4558
4559/// v7.12.4 — catalogued user-defined function. `body` is the raw
4560/// source text between `$$ ... $$`; the engine re-parses it on
4561/// invocation. This keeps the storage codec stable when the
4562/// PL/pgSQL surface grows (no breaking-change risk on the disk
4563/// format).
4564// v7.39 (round 322, V46) — no longer `Eq`: COST / ROWS are f64, as in PG.
4565#[derive(Debug, Clone, PartialEq)]
4566pub struct FunctionDef {
4567    pub name: String,
4568    /// Display form of the argument list, e.g.
4569    /// `"(name TEXT, ts TIMESTAMP)"`. Empty `"()"` for the trigger
4570    /// function shape. Parser-side canonicalised before storage.
4571    pub args_repr: String,
4572    /// Display form of the return type, e.g. `"TRIGGER"` /
4573    /// `"INT"` / `"SETOF text"`. The engine special-cases
4574    /// `"TRIGGER"` (case-insensitive) to gate trigger-only
4575    /// semantics (NEW/OLD).
4576    pub returns: String,
4577    /// `LANGUAGE` clause, lowercased. `"plpgsql"` / `"sql"`.
4578    pub language: String,
4579    /// Source body of the function. PL/pgSQL: includes the
4580    /// surrounding `BEGIN ... END;`. SQL: includes the
4581    /// statement(s). The engine re-parses on invocation; bad
4582    /// bodies surface as a parse error at CALL time, not CREATE.
4583    pub body: String,
4584    /// v7.39 (read01 round 61) — the role that ran CREATE FUNCTION.
4585    pub owner: Option<String>,
4586    /// v7.39 (read01 round 61) — explicit GRANTs (PG `pg_proc.proacl`). EMPTY
4587    /// is NOT "nobody may call it": PG grants EXECUTE to PUBLIC by default, and
4588    /// leaves proacl NULL to say so. The list materialises on the first
4589    /// GRANT / REVOKE.
4590    pub acl: Vec<AclItem>,
4591    /// v7.39 (round 322, V46) — `IMMUTABLE` / `STRICT` / `PARALLEL SAFE` /
4592    /// `SECURITY DEFINER` / `LEAKPROOF` / `COST` / `ROWS`. `strict` is the
4593    /// only one with execution semantics today (a NULL argument yields a
4594    /// NULL result without running the body); the rest are recorded so
4595    /// `pg_get_functiondef` and `pg_proc` report what was declared.
4596    pub volatility: u8,
4597    pub strict: bool,
4598    pub security_definer: bool,
4599    pub leakproof: bool,
4600    pub parallel: u8,
4601    pub cost: Option<f64>,
4602    pub rows: Option<f64>,
4603}
4604
4605/// v7.39 (round 322, V46) — `FunctionDef.volatility` codes: PG's
4606/// `pg_proc.provolatile` letters.
4607pub const FN_VOLATILE: u8 = b'v';
4608pub const FN_IMMUTABLE: u8 = b'i';
4609pub const FN_STABLE: u8 = b's';
4610
4611/// v7.39 (round 322, V46) — `FunctionDef.parallel` codes: PG's
4612/// `pg_proc.proparallel` letters.
4613pub const FN_PARALLEL_UNSAFE: u8 = b'u';
4614pub const FN_PARALLEL_RESTRICTED: u8 = b'r';
4615pub const FN_PARALLEL_SAFE: u8 = b's';
4616
4617/// v7.39 (round 315, V19) — which catalogued function does a persisted
4618/// ACL key refer to?
4619///
4620/// The key was computed by whichever formula was current when the image
4621/// was written, and the multi-word fix changed that formula for bare
4622/// types like `double precision`. A miss therefore does NOT mean "no
4623/// such function": an older image's key would land nowhere and its owner
4624/// and grants would be dropped in silence. Exact match first, then the
4625/// pre-fix formula.
4626#[must_use]
4627pub fn resolve_stored_function_key(
4628    functions: &BTreeMap<String, FunctionDef>,
4629    stored: &str,
4630) -> Option<String> {
4631    if functions.contains_key(stored) {
4632        return Some(stored.to_string());
4633    }
4634    functions
4635        .values()
4636        .find(|f| function_signature_key_legacy(&f.name, &f.args_repr) == stored)
4637        .map(|f| function_signature_key(&f.name, &f.args_repr))
4638}
4639
4640/// v7.39 (round 344, V49) — re-exported from [`spg_sql`], which owns the
4641/// SQL type spellings. This crate carried a byte-identical copy because
4642/// the two were siblings that did not depend on each other; spg-sql is a
4643/// dependency-free leaf, so the dependency is acyclic and the publish
4644/// order already puts it first. One list, one place to keep it right.
4645pub use spg_sql::parser::is_multiword_type_phrase;
4646
4647/// v7.39 (round 315, V19) — the signature key as computed BEFORE the
4648/// multi-word fix, used only to recognise what an older image wrote.
4649///
4650/// The function catalogue recomputes its keys from the stored name and
4651/// argument text on load, so it needs no migration. The ACL block does
4652/// not: it persists the computed key as a string and matches on it. A
4653/// key that changed shape would simply fail to match, and the owner and
4654/// grants would be dropped without a word — so the loader falls back to
4655/// this when the stored key finds nothing.
4656#[must_use]
4657pub fn function_signature_key_legacy(name: &str, args_repr: &str) -> String {
4658    let inner = args_repr
4659        .trim()
4660        .trim_start_matches('(')
4661        .trim_end_matches(')');
4662    let types: Vec<String> = if inner.trim().is_empty() {
4663        Vec::new()
4664    } else {
4665        inner
4666            .split(',')
4667            .map(|part| {
4668                let mut words: Vec<&str> = part.split_whitespace().collect();
4669                if !words.is_empty()
4670                    && (words[0].eq_ignore_ascii_case("OUT")
4671                        || words[0].eq_ignore_ascii_case("INOUT"))
4672                {
4673                    words.remove(0);
4674                }
4675                let ty = if words.len() >= 2 {
4676                    words[1..].join(" ")
4677                } else {
4678                    words.first().map_or(String::new(), |w| (*w).to_string())
4679                };
4680                normalize_type_name(&ty)
4681            })
4682            .collect()
4683    };
4684    format!("{}({})", name.to_ascii_lowercase(), types.join(","))
4685}
4686
4687pub fn function_signature_key(name: &str, args_repr: &str) -> String {
4688    let types = function_arg_types(args_repr);
4689    format!("{}({})", name.to_ascii_lowercase(), types.join(","))
4690}
4691
4692/// The declared argument TYPES of a function, out of its `args_repr`
4693/// (`"(x INT, y DOUBLE PRECISION)"` → `["int", "float"]`). An entry may be a
4694/// bare type with no name (`"(INT)"`).
4695#[must_use]
4696pub fn function_arg_types(args_repr: &str) -> Vec<String> {
4697    let inner = args_repr
4698        .trim()
4699        .trim_start_matches('(')
4700        .trim_end_matches(')');
4701    if inner.trim().is_empty() {
4702        return Vec::new();
4703    }
4704    inner
4705        .split(',')
4706        .map(|part| {
4707            let mut words: Vec<&str> = part.split_whitespace().collect();
4708            // `OUT x INT` / `INOUT x INT` — the mode is not part of the type.
4709            if !words.is_empty()
4710                && (words[0].eq_ignore_ascii_case("OUT") || words[0].eq_ignore_ascii_case("INOUT"))
4711            {
4712                words.remove(0);
4713            }
4714            // v7.39 (round 315, V19) — two or more words is USUALLY
4715            // `name TYPE`, but not when the type itself is spelled in
4716            // several words. `double precision` was read as a parameter
4717            // named "double" of type "precision", so it keyed differently
4718            // from `x double precision` — the same signature written two
4719            // ways did not resolve to the same function. Decide by asking
4720            // whether the whole phrase names a type first; only then is
4721            // the leading word a parameter name.
4722            let whole = words.join(" ");
4723            let ty = if words.len() >= 2 && !is_multiword_type_phrase(&whole) {
4724                words[1..].join(" ")
4725            } else {
4726                whole
4727            };
4728            normalize_type_name(&ty)
4729        })
4730        .collect()
4731}
4732
4733/// v7.39 (read01 round 65) — the declared argument NAMES of a function (`""` for
4734/// a bare type with no name).
4735#[must_use]
4736pub fn function_arg_names(args_repr: &str) -> Vec<String> {
4737    let inner = args_repr
4738        .trim()
4739        .trim_start_matches('(')
4740        .trim_end_matches(')');
4741    if inner.trim().is_empty() {
4742        return Vec::new();
4743    }
4744    inner
4745        .split(',')
4746        .map(|part| {
4747            let mut words: Vec<&str> = part.split_whitespace().collect();
4748            if !words.is_empty()
4749                && (words[0].eq_ignore_ascii_case("OUT") || words[0].eq_ignore_ascii_case("INOUT"))
4750            {
4751                words.remove(0);
4752            }
4753            if words.len() >= 2 {
4754                words[0].to_string()
4755            } else {
4756                String::new()
4757            }
4758        })
4759        .collect()
4760}
4761
4762/// Fold PG's type aliases so a signature key is stable across spellings.
4763/// Unknown names pass through lower-cased — consistency is what the key needs.
4764#[must_use]
4765pub fn normalize_type_name(ty: &str) -> String {
4766    let t = ty.trim().to_ascii_lowercase();
4767    // Peel a precision/length modifier: `numeric(10,2)`, `varchar(64)`.
4768    let base = t.split_once('(').map_or(t.as_str(), |(h, _)| h).trim();
4769    match base {
4770        "int" | "int4" | "integer" => "int",
4771        "bigint" | "int8" => "bigint",
4772        "smallint" | "int2" => "smallint",
4773        "text" | "varchar" | "character varying" | "char" | "character" | "bpchar" => "text",
4774        "bool" | "boolean" => "bool",
4775        "float" | "float8" | "double precision" => "float",
4776        "real" | "float4" => "real",
4777        "numeric" | "decimal" => "numeric",
4778        "timestamptz" | "timestamp with time zone" => "timestamptz",
4779        "timestamp" | "timestamp without time zone" => "timestamp",
4780        other => other,
4781    }
4782    .to_string()
4783}
4784
4785/// v7.12.4 — catalogued trigger. References its function by
4786/// name; the function must exist at TRIGGER creation time
4787/// (forward references are deferred to v7.12.5+).
4788#[derive(Debug, Clone, PartialEq, Eq)]
4789pub struct TriggerDef {
4790    pub name: String,
4791    /// Watched table. Trigger is dropped when the table drops.
4792    pub table: String,
4793    /// `"BEFORE"` / `"AFTER"` / `"INSTEAD OF"`. Stored as the
4794    /// uppercased keyword so deserialised catalogs round-trip
4795    /// without canonicalisation surprises.
4796    pub timing: String,
4797    /// Each entry is one of `"INSERT"` / `"UPDATE"` / `"DELETE"`
4798    /// / `"TRUNCATE"`. `INSERT OR UPDATE` parses to two entries.
4799    pub events: Vec<String>,
4800    /// `"ROW"` / `"STATEMENT"`. v7.12.4 ships `"ROW"` only;
4801    /// `"STATEMENT"` parses and persists but the executor
4802    /// refuses it at trigger fire time.
4803    pub for_each: String,
4804    /// Name of the PL/pgSQL function to invoke.
4805    pub function: String,
4806    /// v7.13.0 — `UPDATE OF col, col, …` column-list filter
4807    /// (mailrs round-5 G7). Non-empty means the trigger fires
4808    /// only when at least one of these columns appears in the
4809    /// UPDATE's SET list. Empty = no column filter. Stored in
4810    /// catalog FILE_VERSION 23+; older catalogs deserialise with
4811    /// an empty vec.
4812    pub update_columns: Vec<String>,
4813    /// v7.16.1 — whether the trigger fires when its watched
4814    /// event occurs. Toggled by `ALTER TABLE … { ENABLE |
4815    /// DISABLE } TRIGGER …`; pg_dump --disable-triggers wraps
4816    /// every data block with a DISABLE/ENABLE pair so the
4817    /// rows already-computed in prod don't get re-rewritten.
4818    /// Defaults to `true` at CREATE TRIGGER time. Stored in
4819    /// catalog FILE_VERSION 25+; older catalogs deserialise
4820    /// with `enabled = true`.
4821    pub enabled: bool,
4822    /// v7.39 (round 138) — the deparsed `WHEN ( condition )` predicate text
4823    /// (re-parsed at fire time to filter row triggers). Empty = no WHEN.
4824    /// Persisted from FILE_VERSION 70; older catalogs read back empty.
4825    pub when_condition: String,
4826}
4827
4828/// v7.39 (round 280) — one `CREATE STATISTICS` object.
4829#[derive(Debug, Clone, PartialEq, Eq)]
4830pub struct StatisticsExtDef {
4831    pub name: String,
4832    pub table: String,
4833    /// PG's single-letter kinds: `d` ndistinct, `f` dependencies,
4834    /// `m` mcv. PG's default set is all three.
4835    pub kinds: Vec<String>,
4836    pub columns: Vec<String>,
4837}
4838
4839/// v7.39 (round 139) — a catalogued query-rewrite RULE. Stored flat like
4840/// `TriggerDef`, keyed by `(name, table)`. Command / WHEN text is deparsed SQL
4841/// re-parsed at rewrite time (the same round-trip trick as
4842/// `TriggerDef.when_condition`). Persisted from FILE_VERSION 71.
4843#[derive(Debug, Clone, PartialEq, Eq)]
4844pub struct RuleDef {
4845    pub name: String,
4846    pub table: String,
4847    /// Event keyword, uppercased: `INSERT` / `UPDATE` / `DELETE` / `SELECT`.
4848    pub event: String,
4849    /// `true` = `DO INSTEAD`, `false` = `DO ALSO`.
4850    pub instead: bool,
4851    /// Deparsed `WHERE` predicate text; empty = unconditional.
4852    pub when_condition: String,
4853    /// Deparsed DO command statements; empty = `NOTHING`.
4854    pub commands: Vec<String>,
4855}
4856
4857/// v7.17.0 — catalogued SEQUENCE. PG semantics: a counter object
4858/// returning monotonically increasing values via `nextval(name)`.
4859/// `last_value` is the most recent value handed out; `is_called`
4860/// is false until the first `nextval`/`setval`. Stored separately
4861/// from tables in the catalog.
4862#[derive(Debug, Clone, PartialEq, Eq)]
4863pub struct SequenceDef {
4864    pub name: String,
4865    /// Data type — narrows the i64 range. PG default BIGINT.
4866    pub data_type: SequenceDataType,
4867    pub start: i64,
4868    pub increment: i64,
4869    pub min_value: i64,
4870    pub max_value: i64,
4871    pub cache: i64,
4872    pub cycle: bool,
4873    /// `OWNED BY` target — `(table, column)` or NONE.
4874    pub owned_by: Option<(String, String)>,
4875    /// Most recently handed-out value. Meaningless when
4876    /// `is_called == false`; in that case the NEXT `nextval`
4877    /// will return `start`.
4878    pub last_value: i64,
4879    pub is_called: bool,
4880    /// v7.39 (read01 round 60) — the role that ran CREATE SEQUENCE. `None` = an
4881    /// image written before FILE_VERSION 66, which predates sequence owners.
4882    pub owner: Option<String>,
4883    /// v7.39 (read01 round 60) — explicit GRANTs on this sequence. A sequence's
4884    /// meaningful privileges are SELECT (`currval`), UPDATE (`setval`) and
4885    /// USAGE (`nextval`).
4886    pub acl: Vec<AclItem>,
4887}
4888
4889/// v7.17.0 — sequence integer width.
4890#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4891pub enum SequenceDataType {
4892    SmallInt,
4893    Int,
4894    BigInt,
4895}
4896
4897/// v7.17.0 Phase 1.6 — built-in schema names that every Catalog
4898/// understands without an explicit CREATE SCHEMA. Used by
4899/// [`Catalog::schema_exists`] and the engine's schema-qualified
4900/// lookup path.
4901#[must_use]
4902pub fn is_builtin_schema(name: &str) -> bool {
4903    name.eq_ignore_ascii_case("public")
4904        || name.eq_ignore_ascii_case("pg_catalog")
4905        || name.eq_ignore_ascii_case("information_schema")
4906}
4907
4908/// v7.17.0 — parse a PG-canonical UUID text representation into the
4909/// 16-byte network-order layout used by `Value::Uuid`. Accepted input
4910/// shapes (all case-insensitive):
4911///   * Canonical hyphenated 8-4-4-4-12 (`550e8400-e29b-41d4-a716-446655440000`)
4912///   * Unhyphenated 32-char hex (`550e8400e29b41d4a716446655440000`)
4913///   * Either form wrapped in `{ ... }`
4914///
4915/// Returns `None` for any malformed input (wrong length, non-hex
4916/// characters, misplaced hyphens). The caller surfaces a SQL error
4917/// at coercion time — silent acceptance of garbage would mask
4918/// application bugs and is exactly the divergence from PG that
4919/// breaks the 0-change cutover promise.
4920#[must_use]
4921pub fn parse_uuid_str(input: &str) -> Option<[u8; 16]> {
4922    let s = input.trim();
4923    // Strip surrounding braces if present.
4924    let s = if let Some(inner) = s.strip_prefix('{').and_then(|x| x.strip_suffix('}')) {
4925        inner
4926    } else {
4927        s
4928    };
4929    // Two valid shapes after braces are stripped: 32 hex chars or
4930    // the canonical 36-char hyphenated form.
4931    let hex: String = match s.len() {
4932        32 => s.to_ascii_lowercase(),
4933        36 => {
4934            // Hyphens must be exactly at positions 8, 13, 18, 23.
4935            let b = s.as_bytes();
4936            if b[8] != b'-' || b[13] != b'-' || b[18] != b'-' || b[23] != b'-' {
4937                return None;
4938            }
4939            let mut out = String::with_capacity(32);
4940            out.push_str(&s[0..8]);
4941            out.push_str(&s[9..13]);
4942            out.push_str(&s[14..18]);
4943            out.push_str(&s[19..23]);
4944            out.push_str(&s[24..36]);
4945            out.make_ascii_lowercase();
4946            out
4947        }
4948        _ => return None,
4949    };
4950    let bytes = hex.as_bytes();
4951    let mut out = [0u8; 16];
4952    for i in 0..16 {
4953        let hi = hex_nibble(bytes[i * 2])?;
4954        let lo = hex_nibble(bytes[i * 2 + 1])?;
4955        out[i] = (hi << 4) | lo;
4956    }
4957    Some(out)
4958}
4959
4960fn hex_nibble(b: u8) -> Option<u8> {
4961    match b {
4962        b'0'..=b'9' => Some(b - b'0'),
4963        b'a'..=b'f' => Some(10 + b - b'a'),
4964        b'A'..=b'F' => Some(10 + b - b'A'),
4965        _ => None,
4966    }
4967}
4968
4969/// v7.17.0 — render a `Value::Uuid` payload as the canonical
4970/// lowercase 8-4-4-4-12 hyphenated form PG `text` cast surfaces.
4971#[must_use]
4972pub fn format_uuid(b: &[u8; 16]) -> String {
4973    const HEX: &[u8; 16] = b"0123456789abcdef";
4974    let mut out = String::with_capacity(36);
4975    for (i, byte) in b.iter().enumerate() {
4976        if matches!(i, 4 | 6 | 8 | 10) {
4977            out.push('-');
4978        }
4979        out.push(HEX[(byte >> 4) as usize] as char);
4980        out.push(HEX[(byte & 0x0f) as usize] as char);
4981    }
4982    out
4983}
4984
4985/// v7.17.0 Phase 1.5 — catalogued user-defined DOMAIN. A domain
4986/// is a named CHECK-constrained alias over a built-in type;
4987/// columns bound to it inherit the base type plus the CHECK
4988/// predicates + NOT NULL + DEFAULT at INSERT/UPDATE time.
4989/// v7.37.17 (Phase E RC rebase) — the write-set one writer version left
4990/// on a table, addressed by stable [`row_header::RowId`]s so it can be
4991/// replayed onto a fresher clone of the relation whose physical slots
4992/// differ. Produced by [`Table::extract_tx_writeset`], consumed by
4993/// [`Table::replay_tx_writeset`].
4994#[derive(Debug, Clone, Default)]
4995pub struct TxWriteSet {
4996    /// INSERTs and UPDATE-new-versions (`header.xmin == v`).
4997    pub inserted: Vec<(row_header::RowId, Row<'static>)>,
4998    /// DELETE / UPDATE-old-version targets (`header.xmax == v`).
4999    pub tombstoned: Vec<row_header::RowId>,
5000}
5001
5002impl TxWriteSet {
5003    #[must_use]
5004    pub fn is_empty(&self) -> bool {
5005        self.inserted.is_empty() && self.tombstoned.is_empty()
5006    }
5007}
5008
5009/// v7.39 (round 260) — one named CHECK on a domain. PG auto-names an
5010/// unnamed one `<domain>_check`, then `_check1`, `_check2`, … (probed).
5011#[derive(Debug, Clone, PartialEq, Eq)]
5012pub struct DomainCheck {
5013    pub name: String,
5014    /// The predicate source, referencing the pseudo-column `VALUE`.
5015    pub expr: String,
5016}
5017
5018/// `default` / `checks` are stored as Display-form source so
5019/// `spg-storage` stays free of `spg-sql` dependency — same
5020/// pattern as FunctionDef / ViewDef.
5021#[derive(Debug, Clone, PartialEq, Eq)]
5022pub struct DomainDef {
5023    pub name: String,
5024    pub base_type: DataType,
5025    pub nullable: bool,
5026    pub default: Option<String>,
5027    /// v7.39 (round 260) — each CHECK carries its constraint NAME, so
5028    /// `ALTER DOMAIN … DROP CONSTRAINT <name>` can find it and the
5029    /// violation message can report the constraint that actually failed.
5030    /// PG's auto-naming for an unnamed check is `<domain>_check`, then
5031    /// `_check1`, `_check2`, … (probed).
5032    pub checks: Vec<DomainCheck>,
5033    /// v7.39 (round 258/259) — when this domain was declared over ANOTHER
5034    /// domain (`CREATE DOMAIN child AS parent CHECK (…)`), the parent's
5035    /// name. `base_type` is the ultimate scalar type either way, so
5036    /// without this the parent's constraints were invisible and a value
5037    /// violating them was silently accepted. PG checks the whole chain,
5038    /// base-first, and an `ALTER DOMAIN` on the parent takes effect for
5039    /// the child immediately (probed) — so the chain is walked at check
5040    /// time rather than copied at CREATE time. Catalog FILE_VERSION 74+.
5041    pub base_domain: Option<String>,
5042}
5043
5044/// v7.17.0 Phase 1.4 — catalogued user-defined ENUM type. The
5045/// label vector is order-preserving (PG enum ordering follows the
5046/// declared order). At INSERT/UPDATE on a column bound to this
5047/// enum, the engine looks up the value against `labels` and
5048/// rejects non-members.
5049#[derive(Debug, Clone, PartialEq, Eq)]
5050pub struct EnumDef {
5051    pub name: String,
5052    pub labels: Vec<String>,
5053}
5054
5055/// v7.37.42-T2 ζ-B — catalogued user-defined COMPOSITE type
5056/// (`CREATE TYPE name AS (field_name field_type, ...)`). Order
5057/// matters: PG composite literals are positional, and SPG mirrors
5058/// that. Stored as ordered `(name, DataType)` pairs to keep the
5059/// codec straightforward and to allow eventual `Value::Composite`
5060/// bodies to encode positionally. Persisted in catalog FILE_VERSION
5061/// 52+; older catalogs deserialise with an empty composite_types
5062/// map. Composite types can be used as a column type by spelling
5063/// the composite's name; the resolution from
5064/// `ColumnSchema.user_composite_type = Some(name)` happens at the
5065/// engine boundary (parallel to `user_enum_type` /
5066/// `user_domain_type`). The dense storage shape — JSON-text body
5067/// keyed by the composite's field list — keeps the codec free of
5068/// recursive `Value` bodies until the full Value::Composite arena
5069/// migration in a later phase.
5070#[derive(Debug, Clone, PartialEq, Eq)]
5071pub struct CompositeDef {
5072    pub name: String,
5073    /// Ordered `(field_name, field_type)` pairs. PG composite
5074    /// literals are positional, so order is part of the type's
5075    /// identity.
5076    pub fields: Vec<(String, DataType)>,
5077    /// v7.39 (round 264) — parallel to `fields`: the USER type name of
5078    /// each field when it is itself a composite (or another named user
5079    /// type). `DataType` has no room for one, so a nested composite
5080    /// field resolved to the parser's Text placeholder and the inner
5081    /// record stayed TEXT — `(x).inner.street` errored, `pg_typeof`
5082    /// said text, and `row_to_json` nested a string instead of an
5083    /// object. Same shape as `ColumnSchema.user_composite_type` and
5084    /// `DomainDef.base_domain`. Catalog FILE_VERSION 76+; an older
5085    /// catalog reads all-None, which is what it meant.
5086    pub field_user_types: Vec<Option<String>>,
5087}
5088
5089/// v7.17.0 Phase 1.2 — catalogued VIEW. The body is stored as the
5090/// raw source text the parser saw between `AS` and the statement
5091/// terminator; the engine re-parses on each invocation. Same
5092/// pattern as `FunctionDef` — keeps `spg-storage` free of
5093/// `spg-sql` dependency.
5094#[derive(Debug, Clone, PartialEq, Eq)]
5095pub struct ViewDef {
5096    pub name: String,
5097    /// Optional `(col, col, …)` rename list. Empty when the body's
5098    /// projected names are used directly.
5099    pub columns: Vec<String>,
5100    /// Raw SELECT source. Display-rendered at storage time so the
5101    /// catalog round-trips a deterministic form regardless of
5102    /// whitespace / comments in the original input. Re-parsed at
5103    /// SELECT-from-view time to materialise as a synthetic CTE.
5104    pub body: String,
5105    /// v7.39 (round 132) — `WITH CHECK OPTION`: 0 = none, 1 = LOCAL,
5106    /// 2 = CASCADED. A storage-local u8 (no dependency on the SQL AST).
5107    /// Persisted from FILE_VERSION 69; older catalogs read back as 0.
5108    pub check_option: u8,
5109}
5110
5111impl SequenceDataType {
5112    /// PG default min/max per AS clause.
5113    pub fn default_bounds(self, increment_positive: bool) -> (i64, i64) {
5114        match self {
5115            Self::SmallInt => {
5116                if increment_positive {
5117                    (1, i64::from(i16::MAX))
5118                } else {
5119                    (i64::from(i16::MIN), -1)
5120                }
5121            }
5122            Self::Int => {
5123                if increment_positive {
5124                    (1, i64::from(i32::MAX))
5125                } else {
5126                    (i64::from(i32::MIN), -1)
5127                }
5128            }
5129            Self::BigInt => {
5130                if increment_positive {
5131                    (1, i64::MAX)
5132                } else {
5133                    (i64::MIN, -1)
5134                }
5135            }
5136        }
5137    }
5138}
5139
5140impl Catalog {
5141    /// v7.37.15 (Phase D) — fleet-wide vacuum pass. Walks every
5142    /// user table and reclaims rows whose delete-commit version is
5143    /// older than `oldest_active_snapshot`. Returns an aggregated
5144    /// report with per-table breakdown so hosts can emit metrics.
5145    ///
5146    /// `dry_run = true` reports the work without doing it. Use it
5147    /// to estimate the cost before scheduling a real pass.
5148    pub fn vacuum_all(
5149        &mut self,
5150        oldest_active_snapshot: u64,
5151        dry_run: bool,
5152    ) -> vacuum::VacuumReport {
5153        let mut total = vacuum::VacuumReport::default();
5154        // Snapshot the table names so we don't hold an immutable
5155        // borrow during the get_mut loop.
5156        let names: Vec<String> = self
5157            .tables
5158            .iter()
5159            .map(|t| t.schema().name.clone())
5160            .collect();
5161        for name in names {
5162            let Some(t) = self.get_mut(&name) else {
5163                continue;
5164            };
5165            let r = t.vacuum(oldest_active_snapshot, dry_run);
5166            if r.rows_reclaimed > 0 {
5167                total.per_table.push((name, r.rows_reclaimed));
5168            }
5169            total.rows_reclaimed += r.rows_reclaimed;
5170            total.rows_examined += r.rows_examined;
5171        }
5172        total
5173    }
5174
5175    pub const fn new() -> Self {
5176        Self {
5177            cold_read_stats: ColdReadStats {
5178                cold_reads: core::sync::atomic::AtomicU64::new(0),
5179            },
5180            tables: Vec::new(),
5181            by_name: BTreeMap::new(),
5182            temp_prefix: None,
5183            dirty_tables: alloc::collections::BTreeSet::new(),
5184            next_rel_id: 0,
5185            cold_segments: Vec::new(),
5186            functions: BTreeMap::new(),
5187            triggers: Vec::new(),
5188            rules: Vec::new(),
5189            statistics_ext: Vec::new(),
5190            large_objects: alloc::collections::BTreeMap::new(),
5191            sequences: BTreeMap::new(),
5192            schema_acl: Vec::new(),
5193            database_acl: Vec::new(),
5194            views: BTreeMap::new(),
5195            materialized_views: BTreeMap::new(),
5196            enum_types: BTreeMap::new(),
5197            domain_types: BTreeMap::new(),
5198            comments: BTreeMap::new(),
5199            db_role_settings: BTreeMap::new(),
5200            replication_slots: BTreeMap::new(),
5201            composite_types: BTreeMap::new(),
5202            schemas: alloc::collections::BTreeSet::new(),
5203        }
5204    }
5205
5206    /// v7.12.4 — read-only view of catalogued user-defined
5207    /// functions. Engine callers go through here to look up the
5208    /// function body before re-parsing it for invocation.
5209    pub const fn functions(&self) -> &BTreeMap<String, FunctionDef> {
5210        &self.functions
5211    }
5212
5213    /// v7.12.4 — register a new user-defined function. With
5214    /// `or_replace = false`, errors if the name is taken. The
5215    /// engine validates the body before passing it here.
5216    pub fn create_function(
5217        &mut self,
5218        def: FunctionDef,
5219        or_replace: bool,
5220    ) -> Result<(), StorageError> {
5221        // v7.39 (read01 round 62) — functions are keyed by SIGNATURE, not by
5222        // name: `f(int)` and `f(text)` are two functions, as in PG. Keying by
5223        // name alone made a second overload an "already exists" error — so a
5224        // pg_dump carrying an overload set could not restore — and, worse, a
5225        // call to one overload silently ran the other.
5226        let key = function_signature_key(&def.name, &def.args_repr);
5227        if !or_replace && self.functions.contains_key(&key) {
5228            return Err(StorageError::Corrupt(format!(
5229                "function {:?} already exists (drop or use CREATE OR REPLACE)",
5230                def.name
5231            )));
5232        }
5233        self.functions.insert(key, def);
5234        Ok(())
5235    }
5236
5237    /// v7.39 (read01 round 62) — every overload of `name`.
5238    #[must_use]
5239    pub fn functions_named(&self, name: &str) -> Vec<&FunctionDef> {
5240        self.functions
5241            .values()
5242            .filter(|f| f.name.eq_ignore_ascii_case(name))
5243            .collect()
5244    }
5245
5246    /// v7.39 (read01 round 62) — one overload, by its signature key.
5247    #[must_use]
5248    pub fn function_by_key(&self, key: &str) -> Option<&FunctionDef> {
5249        self.functions.get(key)
5250    }
5251
5252    /// v7.39 (read01 round 62) — drop ONE overload. `true` if it was there.
5253    pub fn drop_function_by_key(&mut self, key: &str) -> bool {
5254        self.functions.remove(key).is_some()
5255    }
5256
5257    /// v7.12.4 — remove a user-defined function by name. Returns
5258    /// `true` if a function was removed, `false` if none matched.
5259    /// Caller decides whether to surface `if_exists` semantics.
5260    /// v7.39 (read01 round 62) — with no signature, PG drops the function only
5261    /// when the name is unambiguous. SPG mirrors that: this removes EVERY
5262    /// overload of `name`, and the caller (ddl.rs) refuses the ambiguous case
5263    /// before getting here.
5264    pub fn drop_function(&mut self, name: &str) -> bool {
5265        let keys: Vec<String> = self
5266            .functions
5267            .iter()
5268            .filter(|(_, f)| f.name.eq_ignore_ascii_case(name))
5269            .map(|(k, _)| k.clone())
5270            .collect();
5271        let hit = !keys.is_empty();
5272        for k in keys {
5273            self.functions.remove(&k);
5274        }
5275        hit
5276    }
5277
5278    /// v7.17.0 — read-only handle to catalogued sequences.
5279    /// v7.39 (read01 round 60) — the `public` schema's ACL (PG nspacl).
5280    #[must_use]
5281    pub fn schema_acl(&self) -> &[AclItem] {
5282        &self.schema_acl
5283    }
5284
5285    pub fn schema_acl_mut(&mut self) -> &mut Vec<AclItem> {
5286        &mut self.schema_acl
5287    }
5288
5289    /// v7.39 (read01 round 60) — the database's ACL.
5290    #[must_use]
5291    pub fn database_acl(&self) -> &[AclItem] {
5292        &self.database_acl
5293    }
5294
5295    pub fn database_acl_mut(&mut self) -> &mut Vec<AclItem> {
5296        &mut self.database_acl
5297    }
5298
5299    /// v7.39 (read01 round 60) — mutable sequence access, for GRANT.
5300    /// v7.39 (round 469) — resolves the session's temporary sequence
5301    /// first, like its read-only twin. `nextval` and `setval` reach the
5302    /// map through here, so a temporary sequence shadowing a permanent one
5303    /// advances the temporary one — measured against PG18, where the
5304    /// permanent sequence's counter is untouched while the temp exists.
5305    pub fn sequence_mut(&mut self, name: &str) -> Option<&mut SequenceDef> {
5306        let key = self.sequence_key(name);
5307        self.sequences.get_mut(&key)
5308    }
5309
5310    /// v7.39 (read01 round 61) — mutable function access, for GRANT.
5311    pub fn function_mut(&mut self, name: &str) -> Option<&mut FunctionDef> {
5312        self.functions.get_mut(name)
5313    }
5314
5315    /// Every catalogued sequence, temp ones included under their mangled
5316    /// storage names. Listing code filters these through
5317    /// [`Self::listed_name`]; anything resolving ONE name by its logical
5318    /// spelling wants [`Self::sequence`] instead.
5319    pub const fn sequences_all(&self) -> &BTreeMap<String, SequenceDef> {
5320        &self.sequences
5321    }
5322
5323    /// v7.39 (round 469) — resolve one sequence by its logical name, the
5324    /// session's temporary one winning over a permanent one of the same
5325    /// name. The same rule [`Self::resolve_index`] applies to tables.
5326    #[must_use]
5327    pub fn sequence(&self, name: &str) -> Option<&SequenceDef> {
5328        if let Some(mangled) = self.temp_name_for(name)
5329            && let Some(def) = self.sequences.get(&mangled)
5330        {
5331            return Some(def);
5332        }
5333        self.sequences.get(name)
5334    }
5335
5336    /// Does a sequence of this logical name exist for this session?
5337    #[must_use]
5338    pub fn has_sequence(&self, name: &str) -> bool {
5339        self.sequence(name).is_some()
5340    }
5341
5342    /// The storage key a sequence of this logical name resolves to — the
5343    /// session's temp mangling when it has one, else the name itself.
5344    #[must_use]
5345    pub fn sequence_key(&self, name: &str) -> String {
5346        if let Some(mangled) = self.temp_name_for(name)
5347            && self.sequences.contains_key(&mangled)
5348        {
5349            return mangled;
5350        }
5351        name.into()
5352    }
5353
5354    /// v7.17.0 — register a new SEQUENCE. Errors if `name`
5355    /// collides with an existing sequence and `if_not_exists`
5356    /// is false.
5357    pub fn create_sequence(
5358        &mut self,
5359        def: SequenceDef,
5360        if_not_exists: bool,
5361    ) -> Result<(), StorageError> {
5362        if self.sequences.contains_key(&def.name) {
5363            if if_not_exists {
5364                return Ok(());
5365            }
5366            // v7.39 (read01 round 47) — a sequence is a relation to PG (42P07).
5367            return Err(StorageError::Corrupt(format!(
5368                "relation {:?} already exists",
5369                def.name
5370            )));
5371        }
5372        self.sequences.insert(def.name.clone(), def);
5373        Ok(())
5374    }
5375
5376    /// v7.17.0 — remove a SEQUENCE by name. Returns `true` if a
5377    /// sequence was removed, `false` if none matched. Caller
5378    /// surfaces IF EXISTS semantics.
5379    /// v7.39 (read01 round 49) — `ALTER SEQUENCE old RENAME TO new`.
5380    /// Errors when `old` is missing or `new` is taken; the SequenceDef's own
5381    /// `name` field is rewritten so it stays self-describing.
5382    pub fn rename_sequence(&mut self, old: &str, new: &str) -> Result<(), StorageError> {
5383        if !self.sequences.contains_key(old) {
5384            return Err(StorageError::Corrupt(format!(
5385                "relation {old:?} does not exist"
5386            )));
5387        }
5388        if self.sequences.contains_key(new) {
5389            return Err(StorageError::Corrupt(format!(
5390                "relation {new:?} already exists"
5391            )));
5392        }
5393        if let Some(mut def) = self.sequences.remove(old) {
5394            def.name = new.to_string();
5395            self.sequences.insert(new.to_string(), def);
5396        }
5397        Ok(())
5398    }
5399
5400    pub fn drop_sequence(&mut self, name: &str) -> bool {
5401        self.sequences.remove(name).is_some()
5402    }
5403
5404    /// v7.17.0 — atomic nextval. Increments `last_value` per
5405    /// `increment`, returns the new value, sets `is_called`.
5406    /// Returns an error on CYCLE-less overflow.
5407    /// v7.39 (round 497) — the counter state of every sequence, for
5408    /// carrying across a commit install.
5409    ///
5410    /// A sequence's VALUE is not transactional in PG: `nextval` advances
5411    /// shared state that a rollback does not give back, because two
5412    /// sessions must never receive the same number. SPG keeps sequences in
5413    /// the catalog, and a transaction works on a catalog CLONE, so
5414    /// installing that clone at COMMIT would restore whatever the counter
5415    /// was at BEGIN. These two let the install put the live counters back.
5416    #[must_use]
5417    pub fn sequence_counters(&self) -> Vec<(String, i64, bool)> {
5418        self.sequences
5419            .iter()
5420            .map(|(k, d)| (k.clone(), d.last_value, d.is_called))
5421            .collect()
5422    }
5423
5424    /// Restore counters saved by [`Self::sequence_counters`], for the
5425    /// sequences that still exist. A sequence the transaction CREATED is
5426    /// absent from the saved set and keeps the value it was given.
5427    pub fn restore_sequence_counters(&mut self, saved: &[(String, i64, bool)]) {
5428        for (k, last, called) in saved {
5429            if let Some(d) = self.sequences.get_mut(k) {
5430                d.last_value = *last;
5431                d.is_called = *called;
5432            }
5433        }
5434    }
5435
5436    pub fn sequence_next_value(&mut self, name: &str) -> Result<i64, StorageError> {
5437        let key = self.sequence_key(name);
5438        let Some(seq) = self.sequences.get_mut(&key) else {
5439            return Err(StorageError::TableNotFound { name: name.into() });
5440        };
5441        // PG semantics: when !is_called (fresh sequence or
5442        // setval(_, false)), the next nextval returns the stored
5443        // `last_value`. When is_called, it advances by `increment`
5444        // and CYCLE-wraps on overflow.
5445        let candidate = if seq.is_called {
5446            let next = seq.last_value.checked_add(seq.increment).ok_or_else(|| {
5447                StorageError::Corrupt(format!("sequence {name:?} arithmetic overflow"))
5448            })?;
5449            if seq.increment > 0 {
5450                if next > seq.max_value {
5451                    if seq.cycle {
5452                        seq.min_value
5453                    } else {
5454                        // v7.39 (round 220) — PG's 2200H wording, not a
5455                        // Corrupt-classed error.
5456                        return Err(StorageError::SequenceExhausted {
5457                            name: name.into(),
5458                            limit: seq.max_value,
5459                            is_max: true,
5460                        });
5461                    }
5462                } else {
5463                    next
5464                }
5465            } else if next < seq.min_value {
5466                if seq.cycle {
5467                    seq.max_value
5468                } else {
5469                    return Err(StorageError::SequenceExhausted {
5470                        name: name.into(),
5471                        limit: seq.min_value,
5472                        is_max: false,
5473                    });
5474                }
5475            } else {
5476                next
5477            }
5478        } else {
5479            seq.last_value
5480        };
5481        seq.last_value = candidate;
5482        seq.is_called = true;
5483        Ok(candidate)
5484    }
5485
5486    /// v7.17.0 — currval. Errors if the session has never called
5487    /// nextval on this sequence (PG semantics). At the catalog
5488    /// level we approximate "session" with "is_called persisted";
5489    /// the engine session-tracking layer can wrap this for the
5490    /// strict per-session semantics later.
5491    pub fn sequence_current_value(&self, name: &str) -> Result<i64, StorageError> {
5492        let Some(seq) = self.sequences.get(name) else {
5493            return Err(StorageError::TableNotFound { name: name.into() });
5494        };
5495        if !seq.is_called {
5496            return Err(StorageError::Corrupt(format!(
5497                "currval of sequence {name:?} is not yet defined in this session"
5498            )));
5499        }
5500        Ok(seq.last_value)
5501    }
5502
5503    /// v7.17.0 — setval(name, value [, is_called]). PG returns
5504    /// `value` regardless. `is_called=true` means the NEXT
5505    /// nextval will return `value + increment`; `is_called=false`
5506    /// means the next nextval will return `value`.
5507    pub fn sequence_set_value(
5508        &mut self,
5509        name: &str,
5510        value: i64,
5511        is_called: bool,
5512    ) -> Result<i64, StorageError> {
5513        let key = self.sequence_key(name);
5514        let Some(seq) = self.sequences.get_mut(&key) else {
5515            return Err(StorageError::TableNotFound { name: name.into() });
5516        };
5517        // v7.39 (round 244) — PG refuses a value outside the sequence's
5518        // range (22003); SPG accepted it silently, leaving last_value out
5519        // of bounds.
5520        if value < seq.min_value || value > seq.max_value {
5521            return Err(StorageError::Unsupported(format!(
5522                "setval: value {value} is out of bounds for sequence \"{name}\" ({}..{})",
5523                seq.min_value, seq.max_value
5524            )));
5525        }
5526        seq.last_value = value;
5527        seq.is_called = is_called;
5528        Ok(value)
5529    }
5530
5531    /// v7.17.0 Phase 1.2 — read-only handle to catalogued views. Temp ones
5532    /// are in here under their mangled storage names; listing code filters
5533    /// through [`Self::listed_name`], and anything resolving ONE name by
5534    /// its logical spelling wants [`Self::view`].
5535    pub const fn views_all(&self) -> &BTreeMap<String, ViewDef> {
5536        &self.views
5537    }
5538
5539    /// v7.39 (round 469) — resolve one view by its logical name, the
5540    /// session's temporary one winning over a permanent one of the same
5541    /// name.
5542    #[must_use]
5543    pub fn view(&self, name: &str) -> Option<&ViewDef> {
5544        if let Some(mangled) = self.temp_name_for(name)
5545            && let Some(def) = self.views.get(&mangled)
5546        {
5547            return Some(def);
5548        }
5549        self.views.get(name)
5550    }
5551
5552    /// Does a view of this logical name exist for this session?
5553    #[must_use]
5554    pub fn has_view(&self, name: &str) -> bool {
5555        self.view(name).is_some()
5556    }
5557
5558    /// The storage key a view of this logical name resolves to.
5559    #[must_use]
5560    pub fn view_key(&self, name: &str) -> String {
5561        if let Some(mangled) = self.temp_name_for(name)
5562            && self.views.contains_key(&mangled)
5563        {
5564            return mangled;
5565        }
5566        name.into()
5567    }
5568
5569    /// v7.17.0 Phase 1.2 — install a VIEW. `or_replace=true`
5570    /// overwrites an existing entry; `if_not_exists=true` is a
5571    /// silent no-op when the name is taken. Errors if both flags
5572    /// are off and the name collides.
5573    pub fn create_view(
5574        &mut self,
5575        def: ViewDef,
5576        or_replace: bool,
5577        if_not_exists: bool,
5578    ) -> Result<(), StorageError> {
5579        if self.views.contains_key(&def.name) {
5580            if or_replace {
5581                self.views.insert(def.name.clone(), def);
5582                return Ok(());
5583            }
5584            if if_not_exists {
5585                return Ok(());
5586            }
5587            // v7.39 (read01 round 47) — a view is a relation to PG (42P07).
5588            return Err(StorageError::Corrupt(format!(
5589                "relation {:?} already exists",
5590                def.name
5591            )));
5592        }
5593        // Reject name collision with tables / sequences — same
5594        // namespace per PG.
5595        if self.by_name.contains_key(&def.name) {
5596            return Err(StorageError::Corrupt(format!(
5597                "view {:?} would shadow an existing table",
5598                def.name
5599            )));
5600        }
5601        if self.sequences.contains_key(&def.name) {
5602            return Err(StorageError::Corrupt(format!(
5603                "view {:?} would shadow an existing sequence",
5604                def.name
5605            )));
5606        }
5607        self.views.insert(def.name.clone(), def);
5608        Ok(())
5609    }
5610
5611    /// v7.17.0 Phase 1.2 — remove a view by name. Returns true if
5612    /// a view was removed.
5613    pub fn drop_view(&mut self, name: &str) -> bool {
5614        self.views.remove(name).is_some()
5615    }
5616
5617    /// v7.17.0 Phase 1.3 — read-only handle to the materialised-
5618    /// view source registry. Each entry pairs with a regular
5619    /// table of the same name that holds the cached rows.
5620    pub const fn materialized_views(&self) -> &BTreeMap<String, String> {
5621        &self.materialized_views
5622    }
5623
5624    /// v7.17.0 Phase 1.3 — register a source for a materialised
5625    /// view. Caller has already created the backing table.
5626    pub fn register_materialized_view(&mut self, name: String, body: String) {
5627        self.materialized_views.insert(name, body);
5628    }
5629
5630    /// v7.17.0 Phase 1.3 — drop the source registry entry. Returns
5631    /// true if a source was unregistered. Caller separately drops
5632    /// the backing table.
5633    pub fn drop_materialized_view_source(&mut self, name: &str) -> bool {
5634        self.materialized_views.remove(name).is_some()
5635    }
5636
5637    /// v7.17.0 Phase 1.4 — read-only handle to user-defined ENUM
5638    /// catalog.
5639    pub const fn enum_types(&self) -> &BTreeMap<String, EnumDef> {
5640        &self.enum_types
5641    }
5642
5643    /// v7.17.0 Phase 1.4 — install a new ENUM type. Errors if
5644    /// `name` collides with an existing enum (no IF NOT EXISTS
5645    /// per PG semantics for CREATE TYPE).
5646    pub fn create_enum_type(&mut self, def: EnumDef) -> Result<(), StorageError> {
5647        if self.enum_types.contains_key(&def.name) {
5648            return Err(StorageError::Corrupt(format!(
5649                "type {:?} already exists",
5650                def.name
5651            )));
5652        }
5653        self.enum_types.insert(def.name.clone(), def);
5654        Ok(())
5655    }
5656
5657    /// v7.17.0 Phase 1.4 — drop an ENUM type by name. Returns
5658    /// true if a type was removed.
5659    /// v7.37 D.55 — `ALTER TYPE … ADD VALUE`. Appends `label` to an existing
5660    /// enum's ordered label list, or inserts it before/after an existing label.
5661    /// `if_not_exists` makes a duplicate a no-op; otherwise a duplicate errors.
5662    /// Returns `Ok(true)` if a label was added, `Ok(false)` if it already existed
5663    /// (only possible under `if_not_exists`).
5664    /// v7.39 (read01 round 49) — `ALTER TYPE t RENAME VALUE 'old' TO 'new'`.
5665    /// The parser used to swallow this form as a no-op, so the rename was
5666    /// accepted and silently ignored. Renaming in place keeps the label's
5667    /// sort position, which is what PG does (enumsortorder is untouched).
5668    pub fn rename_enum_value(
5669        &mut self,
5670        type_name: &str,
5671        old: &str,
5672        new: &str,
5673    ) -> Result<(), StorageError> {
5674        let def = self
5675            .enum_types
5676            .get_mut(type_name)
5677            .ok_or_else(|| StorageError::Corrupt(format!("type {type_name:?} does not exist")))?;
5678        if def.labels.iter().any(|l| l == new) {
5679            return Err(StorageError::Corrupt(format!(
5680                "enum label {new:?} already exists"
5681            )));
5682        }
5683        let at = def.labels.iter().position(|l| l == old).ok_or_else(|| {
5684            StorageError::Corrupt(format!("{old:?} is not an existing enum label"))
5685        })?;
5686        def.labels[at] = new.to_string();
5687        Ok(())
5688    }
5689
5690    /// v7.39 (read01 round 50) — set (or, with `None`, remove) the comment on
5691    /// an object. `key` is the canonical `"<kind>:<name>"` form.
5692    pub fn set_comment(&mut self, key: &str, text: Option<&str>) {
5693        match text {
5694            Some(t) => {
5695                self.comments.insert(key.to_string(), t.to_string());
5696            }
5697            None => {
5698                self.comments.remove(key);
5699            }
5700        }
5701    }
5702
5703    /// v7.39 (read01 round 50) — the comment on an object, if any.
5704    #[must_use]
5705    pub fn comment(&self, key: &str) -> Option<&str> {
5706        self.comments.get(key).map(String::as_str)
5707    }
5708
5709    /// v7.39 (round 547) — record a GUC default for a scope. An empty
5710    /// database or role name is PG's oid 0 ("all"). `None` value
5711    /// removes just that parameter, as PG's RESET does.
5712    pub fn set_db_role_setting(
5713        &mut self,
5714        database: &str,
5715        role: &str,
5716        param: &str,
5717        value: Option<&str>,
5718    ) {
5719        let key = (database.to_string(), role.to_string());
5720        match value {
5721            Some(v) => {
5722                self.db_role_settings
5723                    .entry(key)
5724                    .or_default()
5725                    .insert(param.to_ascii_lowercase(), v.to_string());
5726            }
5727            None => {
5728                if let Some(m) = self.db_role_settings.get_mut(&key) {
5729                    m.remove(&param.to_ascii_lowercase());
5730                    if m.is_empty() {
5731                        self.db_role_settings.remove(&key);
5732                    }
5733                }
5734            }
5735        }
5736    }
5737
5738    /// v7.39 (round 550) — create a replication slot. `Err` carries
5739    /// PG's own message for a duplicate.
5740    ///
5741    /// # Errors
5742    /// When a slot of that name already exists.
5743    pub fn create_replication_slot(
5744        &mut self,
5745        name: &str,
5746        plugin: &str,
5747        slot_type: &str,
5748    ) -> Result<(), String> {
5749        if self.replication_slots.contains_key(name) {
5750            return Err(alloc::format!("replication slot \"{name}\" already exists"));
5751        }
5752        self.replication_slots.insert(
5753            name.to_string(),
5754            (plugin.to_string(), slot_type.to_string()),
5755        );
5756        Ok(())
5757    }
5758
5759    /// # Errors
5760    /// When no slot of that name exists — PG's message, and the case
5761    /// that used to report success.
5762    pub fn drop_replication_slot(&mut self, name: &str) -> Result<(), String> {
5763        if self.replication_slots.remove(name).is_none() {
5764            return Err(alloc::format!("replication slot \"{name}\" does not exist"));
5765        }
5766        Ok(())
5767    }
5768
5769    #[must_use]
5770    pub const fn replication_slots(&self) -> &BTreeMap<String, (String, String)> {
5771        &self.replication_slots
5772    }
5773
5774    /// PG's RESET ALL: drops this scope's whole entry, leaving the
5775    /// other scopes alone — measured on PG18, where `ALTER ROLE r RESET
5776    /// ALL` left the ALL, the database and the role-in-database rows.
5777    pub fn reset_db_role_settings(&mut self, database: &str, role: &str) {
5778        self.db_role_settings
5779            .remove(&(database.to_string(), role.to_string()));
5780    }
5781
5782    #[must_use]
5783    pub const fn db_role_settings(&self) -> &BTreeMap<(String, String), BTreeMap<String, String>> {
5784        &self.db_role_settings
5785    }
5786
5787    /// v7.39 (read01 round 50) — every `(key, text)` pair, for the
5788    /// pg_description view.
5789    #[must_use]
5790    pub const fn comments(&self) -> &BTreeMap<String, String> {
5791        &self.comments
5792    }
5793
5794    /// v7.39 (read01 round 50) — drop every comment whose key names `obj`
5795    /// (the object itself and, for a table, its columns). Called when the
5796    /// object is dropped so a later object of the same name doesn't inherit
5797    /// a stale comment.
5798    pub fn drop_comments_for(&mut self, kind: &str, name: &str) {
5799        let exact = alloc::format!("{kind}:{name}");
5800        let col_prefix = alloc::format!("column:{name}.");
5801        self.comments
5802            .retain(|k, _| *k != exact && !k.starts_with(&col_prefix));
5803    }
5804
5805    pub fn add_enum_value(
5806        &mut self,
5807        type_name: &str,
5808        label: &str,
5809        if_not_exists: bool,
5810        position: Option<(bool, String)>,
5811    ) -> Result<bool, StorageError> {
5812        let def = self
5813            .enum_types
5814            .get_mut(type_name)
5815            .ok_or_else(|| StorageError::Corrupt(format!("type {type_name:?} does not exist")))?;
5816        if def.labels.iter().any(|l| l == label) {
5817            if if_not_exists {
5818                return Ok(false);
5819            }
5820            // v7.39 (read01 round 49) — PG wording (42710 at the wire).
5821            return Err(StorageError::Corrupt(format!(
5822                "enum label {label:?} already exists"
5823            )));
5824        }
5825        match position {
5826            None => def.labels.push(label.to_string()),
5827            Some((is_before, anchor)) => {
5828                let at = def
5829                    .labels
5830                    .iter()
5831                    .position(|l| l == &anchor)
5832                    .ok_or_else(|| {
5833                        StorageError::Corrupt(format!(
5834                            "enum label {anchor:?} does not exist in type {type_name:?}"
5835                        ))
5836                    })?;
5837                let idx = if is_before { at } else { at + 1 };
5838                def.labels.insert(idx, label.to_string());
5839            }
5840        }
5841        Ok(true)
5842    }
5843
5844    pub fn drop_enum_type(&mut self, name: &str) -> bool {
5845        self.enum_types.remove(name).is_some()
5846    }
5847
5848    /// v7.17.0 Phase 1.5 — read-only handle to DOMAIN catalog.
5849    pub const fn domain_types(&self) -> &BTreeMap<String, DomainDef> {
5850        &self.domain_types
5851    }
5852
5853    /// v7.17.0 Phase 1.5 — install a DOMAIN. Errors on collision
5854    /// with an existing domain.
5855    pub fn create_domain_type(&mut self, def: DomainDef) -> Result<(), StorageError> {
5856        if self.domain_types.contains_key(&def.name) {
5857            return Err(StorageError::Corrupt(format!(
5858                "domain {:?} already exists",
5859                def.name
5860            )));
5861        }
5862        self.domain_types.insert(def.name.clone(), def);
5863        Ok(())
5864    }
5865
5866    /// v7.17.0 Phase 1.5 — drop a DOMAIN by name.
5867    pub fn drop_domain_type(&mut self, name: &str) -> bool {
5868        self.domain_types.remove(name).is_some()
5869    }
5870
5871    /// v7.37.42-T2 ζ-B — read-only handle to user-defined COMPOSITE
5872    /// catalog. Used by the engine to resolve
5873    /// `ColumnSchema.user_composite_type` lookups + by
5874    /// information_schema-style introspection.
5875    pub const fn composite_types(&self) -> &BTreeMap<String, CompositeDef> {
5876        &self.composite_types
5877    }
5878
5879    /// v7.37.42-T2 ζ-B — install a new COMPOSITE type. Errors if
5880    /// `name` already exists in the composite registry (PG forbids
5881    /// IF NOT EXISTS on CREATE TYPE composite; the engine surfaces
5882    /// the collision with the existing name).
5883    pub fn create_composite_type(&mut self, def: CompositeDef) -> Result<(), StorageError> {
5884        if self.composite_types.contains_key(&def.name) {
5885            return Err(StorageError::Corrupt(format!(
5886                "type {:?} already exists",
5887                def.name
5888            )));
5889        }
5890        self.composite_types.insert(def.name.clone(), def);
5891        Ok(())
5892    }
5893
5894    /// v7.37.42-T2 ζ-B — drop a COMPOSITE type by name. Returns
5895    /// true if a type was removed.
5896    pub fn drop_composite_type(&mut self, name: &str) -> bool {
5897        self.composite_types.remove(name).is_some()
5898    }
5899
5900    /// v7.17.0 Phase 1.6 — read-only handle to the user-created
5901    /// schema registry. Built-in schemas (`public`, `pg_catalog`,
5902    /// `information_schema`) are NOT included here; use
5903    /// [`schema_exists`](Self::schema_exists) for the full
5904    /// check.
5905    pub const fn user_schemas(&self) -> &alloc::collections::BTreeSet<String> {
5906        &self.schemas
5907    }
5908
5909    /// v7.17.0 Phase 1.6 — schema-name resolver. Returns true
5910    /// for built-in schemas + every user-CREATEd one. Used by
5911    /// CREATE SCHEMA collision checks and (future) by
5912    /// information_schema.schemata.
5913    pub fn schema_exists(&self, name: &str) -> bool {
5914        is_builtin_schema(name) || self.schemas.contains(name)
5915    }
5916
5917    /// v7.17.0 Phase 1.6 — register a new schema. Errors if the
5918    /// name already exists and `if_not_exists=false`. Built-in
5919    /// names cannot be redeclared.
5920    pub fn create_schema(&mut self, name: String, if_not_exists: bool) -> Result<(), StorageError> {
5921        if is_builtin_schema(&name) {
5922            if if_not_exists {
5923                return Ok(());
5924            }
5925            return Err(StorageError::Corrupt(format!(
5926                "schema {name:?} is built-in and cannot be redeclared"
5927            )));
5928        }
5929        if self.schemas.contains(&name) {
5930            if if_not_exists {
5931                return Ok(());
5932            }
5933            return Err(StorageError::Corrupt(format!(
5934                "schema {name:?} already exists"
5935            )));
5936        }
5937        self.schemas.insert(name);
5938        Ok(())
5939    }
5940
5941    /// v7.17.0 Phase 1.6 — drop a user-created schema. Returns
5942    /// true if a schema was removed. Built-in names always
5943    /// return false (cannot be dropped). Tables that previously
5944    /// used the schema as a prefix keep their bare name and stay
5945    /// queryable — this is the "prefix routing, not isolation"
5946    /// posture documented in v7.17 Phase 1.6.
5947    pub fn drop_schema(&mut self, name: &str) -> Result<bool, StorageError> {
5948        if is_builtin_schema(name) {
5949            return Err(StorageError::Corrupt(format!(
5950                "schema {name:?} is built-in and cannot be dropped"
5951            )));
5952        }
5953        Ok(self.schemas.remove(name))
5954    }
5955
5956    /// v7.17.0 — ALTER SEQUENCE option merge. Caller-provided
5957    /// updates overwrite the matching fields; unset fields keep
5958    /// their stored values. RESTART variants update last_value
5959    /// directly per PG: `RESTART` resets to current `start`;
5960    /// `RESTART WITH n` resets to `n`.
5961    #[allow(clippy::too_many_arguments)]
5962    pub fn alter_sequence(
5963        &mut self,
5964        name: &str,
5965        increment: Option<i64>,
5966        min_value: Option<i64>,
5967        max_value: Option<i64>,
5968        start: Option<i64>,
5969        restart: Option<Option<i64>>,
5970        cache: Option<i64>,
5971        cycle: Option<bool>,
5972        owned_by: Option<Option<(String, String)>>,
5973    ) -> Result<(), StorageError> {
5974        let Some(seq) = self.sequences.get_mut(name) else {
5975            return Err(StorageError::TableNotFound { name: name.into() });
5976        };
5977        if let Some(v) = increment {
5978            seq.increment = v;
5979        }
5980        if let Some(v) = min_value {
5981            seq.min_value = v;
5982        }
5983        if let Some(v) = max_value {
5984            seq.max_value = v;
5985        }
5986        if let Some(v) = start {
5987            seq.start = v;
5988        }
5989        if let Some(restart_value) = restart {
5990            seq.last_value = restart_value.unwrap_or(seq.start);
5991            seq.is_called = false;
5992        }
5993        if let Some(v) = cache {
5994            seq.cache = v;
5995        }
5996        if let Some(v) = cycle {
5997            seq.cycle = v;
5998        }
5999        if let Some(v) = owned_by {
6000            seq.owned_by = v;
6001        }
6002        Ok(())
6003    }
6004
6005    /// v7.12.4 — read-only slice of all catalogued triggers.
6006    /// Engine row-write paths filter this by (table, event,
6007    /// timing) and fire matches in slice order.
6008    pub fn triggers(&self) -> &[TriggerDef] {
6009        &self.triggers
6010    }
6011
6012    /// v7.15.0 — mutable handle to the trigger slice for
6013    /// `ALTER TABLE … RENAME COLUMN`, which rewrites every
6014    /// `update_columns` entry that referenced the renamed
6015    /// column.
6016    pub fn triggers_mut(&mut self) -> &mut Vec<TriggerDef> {
6017        &mut self.triggers
6018    }
6019
6020    /// v7.12.4 — register a new trigger. With `or_replace = false`,
6021    /// errors when a trigger with the same name already exists on
6022    /// the same table (PG scoping rule — trigger names are
6023    /// per-table, not global). Trigger function must already
6024    /// exist in the catalog at registration time.
6025    pub fn create_trigger(
6026        &mut self,
6027        def: TriggerDef,
6028        or_replace: bool,
6029    ) -> Result<(), StorageError> {
6030        // v7.39 (round 137) — a trigger may target a base table (BEFORE / AFTER)
6031        // or a view (INSTEAD OF). The engine enforces the timing↔target rule;
6032        // storage only requires the relation to exist as one or the other.
6033        if !self.by_name.contains_key(&def.table) && !self.views.contains_key(&def.table) {
6034            return Err(StorageError::TableNotFound {
6035                name: def.table.clone(),
6036            });
6037        }
6038        // v7.39 (read01 round 62) — functions are keyed by SIGNATURE now. A
6039        // trigger names its function by NAME (a trigger function takes no
6040        // arguments), so the existence check goes through the name index.
6041        if self.functions_named(&def.function).is_empty() {
6042            // v7.39 (round 710) — PG's wording: the FUNCTION is what does
6043            // not exist (`function nosuch_fn() does not exist`), and the
6044            // old message rode `Corrupt`'s on-disk banner besides.
6045            return Err(StorageError::Corrupt(format!(
6046                "function {}() does not exist",
6047                def.function
6048            )));
6049        }
6050        let dup = self
6051            .triggers
6052            .iter()
6053            .position(|t| t.name == def.name && t.table == def.table);
6054        match (dup, or_replace) {
6055            (Some(_), false) => Err(StorageError::Corrupt(format!(
6056                "trigger {:?} already exists on table {:?}",
6057                def.name, def.table
6058            ))),
6059            (Some(i), true) => {
6060                self.triggers[i] = def;
6061                Ok(())
6062            }
6063            (None, _) => {
6064                self.triggers.push(def);
6065                Ok(())
6066            }
6067        }
6068    }
6069
6070    /// v7.12.4 — remove a trigger by `(name, table)`. Returns
6071    /// `true` if one was removed.
6072    pub fn drop_trigger(&mut self, name: &str, table: &str) -> bool {
6073        let before = self.triggers.len();
6074        self.triggers
6075            .retain(|t| !(t.name == name && t.table == table));
6076        before != self.triggers.len()
6077    }
6078
6079    /// v7.39 (round 139) — the catalogued query-rewrite RULEs.
6080    pub fn rules(&self) -> &[RuleDef] {
6081        &self.rules
6082    }
6083
6084    /// v7.39 (round 280) — the catalogued extended-statistics objects.
6085    #[must_use]
6086    pub fn statistics_ext(&self) -> &[StatisticsExtDef] {
6087        &self.statistics_ext
6088    }
6089
6090    /// v7.39 (round 287) — every large object, ascending by OID.
6091    #[must_use]
6092    pub fn large_objects(&self) -> &alloc::collections::BTreeMap<u32, Vec<u8>> {
6093        &self.large_objects
6094    }
6095
6096    /// The bytes of one large object, or `None` when no such OID exists.
6097    #[must_use]
6098    pub fn large_object(&self, oid: u32) -> Option<&[u8]> {
6099        self.large_objects.get(&oid).map(Vec::as_slice)
6100    }
6101
6102    /// Create a large object. `oid` of 0 means "pick one" — PG's
6103    /// `lo_create(0)` / `lo_creat(-1)` spelling. Errors when the
6104    /// requested OID is taken.
6105    pub fn create_large_object(&mut self, oid: u32, bytes: Vec<u8>) -> Result<u32, String> {
6106        let id = if oid == 0 {
6107            self.next_large_object_oid()
6108        } else {
6109            oid
6110        };
6111        if self.large_objects.contains_key(&id) {
6112            return Err(format!("large object {id} already exists"));
6113        }
6114        self.large_objects.insert(id, bytes);
6115        Ok(id)
6116    }
6117
6118    /// Overwrite `len` bytes at `offset` (0-based), growing the object
6119    /// with zero bytes if the write starts past the end — PG's
6120    /// `lo_put` semantics.
6121    pub fn put_large_object(&mut self, oid: u32, offset: usize, data: &[u8]) -> Result<(), String> {
6122        let Some(buf) = self.large_objects.get_mut(&oid) else {
6123            return Err(format!("large object {oid} does not exist"));
6124        };
6125        let end = offset.saturating_add(data.len());
6126        if buf.len() < end {
6127            buf.resize(end, 0);
6128        }
6129        buf[offset..end].copy_from_slice(data);
6130        Ok(())
6131    }
6132
6133    /// v7.39 (round 306) — `lo_truncate`. PG's truncate sets the object
6134    /// to exactly `len` bytes in BOTH directions: it shortens, and it
6135    /// GROWS with zero fill when `len` exceeds the current size
6136    /// (measured — `lo_truncate(fd, 8)` over a 4-byte object leaves
6137    /// eight bytes, the last four zero).
6138    pub fn truncate_large_object(&mut self, oid: u32, len: usize) -> Result<(), String> {
6139        let Some(buf) = self.large_objects.get_mut(&oid) else {
6140            return Err(format!("large object {oid} does not exist"));
6141        };
6142        buf.resize(len, 0);
6143        Ok(())
6144    }
6145
6146    /// Remove a large object. `false` when the OID was not there.
6147    pub fn unlink_large_object(&mut self, oid: u32) -> bool {
6148        self.large_objects.remove(&oid).is_some()
6149    }
6150
6151    /// The next free OID in PG's user band.
6152    /// v7.39 (round 343, V40) — large objects have their own oid band.
6153    /// It used to start at 16_384, which is where user TABLES start, so
6154    /// the first large object and the first table shared an oid — and
6155    /// `pg_largeobject_metadata.oid` is joinable against `pg_class.oid`,
6156    /// so a join across them matched a row that has nothing to do with
6157    /// it. (PG cannot collide: every oid there comes off one counter.)
6158    /// An object already stored keeps the oid it was given; only new
6159    /// ones land in the band.
6160    fn next_large_object_oid(&self) -> u32 {
6161        self.large_objects
6162            .keys()
6163            .next_back()
6164            .map_or(500_000, |m| m.saturating_add(1))
6165    }
6166
6167    /// Register one. `Err(name)` when the name is taken.
6168    pub fn create_statistics_ext(&mut self, def: StatisticsExtDef) -> Result<(), String> {
6169        if self.statistics_ext.iter().any(|s| s.name == def.name) {
6170            return Err(def.name);
6171        }
6172        self.statistics_ext.push(def);
6173        Ok(())
6174    }
6175
6176    /// Drop one by name; false when absent.
6177    pub fn drop_statistics_ext(&mut self, name: &str) -> bool {
6178        let before = self.statistics_ext.len();
6179        self.statistics_ext.retain(|s| s.name != name);
6180        before != self.statistics_ext.len()
6181    }
6182
6183    /// v7.39 (round 139) — register a RULE. Its target relation (table or view)
6184    /// must exist; `or_replace` overwrites a same-(name,table) rule.
6185    pub fn create_rule(&mut self, def: RuleDef, or_replace: bool) -> Result<(), StorageError> {
6186        if !self.by_name.contains_key(&def.table) && !self.views.contains_key(&def.table) {
6187            return Err(StorageError::TableNotFound {
6188                name: def.table.clone(),
6189            });
6190        }
6191        let dup = self
6192            .rules
6193            .iter()
6194            .position(|r| r.name == def.name && r.table == def.table);
6195        match (dup, or_replace) {
6196            (Some(_), false) => Err(StorageError::Corrupt(format!(
6197                "rule {:?} for relation {:?} already exists",
6198                def.name, def.table
6199            ))),
6200            (Some(i), true) => {
6201                self.rules[i] = def;
6202                Ok(())
6203            }
6204            (None, _) => {
6205                self.rules.push(def);
6206                Ok(())
6207            }
6208        }
6209    }
6210
6211    /// v7.39 (round 139) — drop a RULE by `(name, table)`.
6212    pub fn drop_rule(&mut self, name: &str, table: &str) -> bool {
6213        let before = self.rules.len();
6214        self.rules.retain(|r| !(r.name == name && r.table == table));
6215        before != self.rules.len()
6216    }
6217
6218    pub fn create_table(&mut self, schema: TableSchema) -> Result<(), StorageError> {
6219        if self.by_name.contains_key(&schema.name) {
6220            return Err(StorageError::DuplicateTable {
6221                name: schema.name.clone(),
6222            });
6223        }
6224        let idx = self.tables.len();
6225        let name = schema.name.clone();
6226        self.tables.push(Table::new(schema));
6227        self.by_name.insert(name.clone(), idx);
6228        // v7.39 (round 496) — see `dirty_tables`.
6229        self.dirty_tables.insert(name);
6230        // v7.37.15 (Phase C.1) — stamp the new relation with a stable,
6231        // monotonic, never-reused RelId. Pre-increment so ids start at
6232        // 1 (0 = UNASSIGNED); a later DROP TABLE frees the slot but not
6233        // the id.
6234        self.next_rel_id += 1;
6235        let rid = row_header::RelId(self.next_rel_id);
6236        self.tables[idx].set_rel_id(rid);
6237        Ok(())
6238    }
6239
6240    /// v7.39 (round 436) — the session's temporary table of this name wins
6241    /// over a permanent one, as `pg_temp` does in PG's search path and as
6242    /// MySQL's TEMPORARY shadowing does. Every name → index resolution in
6243    /// this catalog goes through here.
6244    fn resolve_index(&self, name: &str) -> Option<usize> {
6245        if let Some(prefix) = &self.temp_prefix {
6246            let mut mangled = String::with_capacity(prefix.len() + name.len());
6247            mangled.push_str(prefix);
6248            mangled.push_str(name);
6249            if let Some(idx) = self.by_name.get(&mangled) {
6250                return Some(*idx);
6251            }
6252        }
6253        self.by_name.get(name).copied()
6254    }
6255
6256    /// v7.39 (round 436) — install the calling session's temp namespace.
6257    /// `None` disables temp resolution entirely (a session that never made
6258    /// one pays a single `Option` check per lookup).
6259    pub fn set_temp_prefix(&mut self, prefix: Option<String>) {
6260        self.temp_prefix = prefix;
6261    }
6262
6263    /// The mangled storage name a temp table of `name` takes in this
6264    /// session, or `None` when the session has no temp namespace.
6265    #[must_use]
6266    pub fn temp_name_for(&self, name: &str) -> Option<String> {
6267        self.temp_prefix
6268            .as_ref()
6269            .map(|p| alloc::format!("{p}{name}"))
6270    }
6271
6272    pub fn get(&self, name: &str) -> Option<&Table> {
6273        let idx = self.resolve_index(name)?;
6274        self.tables.get(idx)
6275    }
6276
6277    pub fn get_mut(&mut self, name: &str) -> Option<&mut Table> {
6278        let idx = self.resolve_index(name)?;
6279        // v7.39 (round 496) — the choke point for changing a table, so the
6280        // record is taken here. Over-approximate on purpose: a caller that
6281        // takes the handle and writes nothing merely carries that table
6282        // through a commit, which is the old behaviour.
6283        let recorded = self.tables.get(idx).map(|t| t.schema().name.clone());
6284        if let Some(n) = recorded {
6285            self.dirty_tables.insert(n);
6286        }
6287        self.tables.get_mut(idx)
6288    }
6289
6290    /// v7.39 (round 496) — the tables changed through this handle since
6291    /// [`Self::clear_dirty_tables`]. See `dirty_tables`.
6292    #[must_use]
6293    pub fn dirty_tables(&self) -> &alloc::collections::BTreeSet<String> {
6294        &self.dirty_tables
6295    }
6296
6297    /// r1059 — mark one table dirty without taking its handle. The
6298    /// rebase/merge paths replace a tx's shadow with a fresh base
6299    /// clone and must carry the tx's OWN dirty window across (the
6300    /// base's set is an ever-growing history, never cleared).
6301    pub fn mark_table_dirty(&mut self, name: &str) {
6302        self.dirty_tables.insert(name.into());
6303    }
6304
6305    /// v7.39 (round 496) — start a fresh recording window. A transaction's
6306    /// shadow calls this at BEGIN so the set means "changed by this tx".
6307    pub fn clear_dirty_tables(&mut self) {
6308        self.dirty_tables.clear();
6309    }
6310
6311    /// v7.39 (round 496) — put `table` in at `name`, replacing any table
6312    /// already there and keeping the rest of the catalog untouched.
6313    ///
6314    /// The commit-time table-granularity merge needs exactly this: take
6315    /// the latest committed catalog, then overwrite only the tables the
6316    /// transaction changed.
6317    pub fn install_table(&mut self, name: &str, table: Table) {
6318        match self.by_name.get(name).copied() {
6319            Some(idx) => self.tables[idx] = table,
6320            None => {
6321                let idx = self.tables.len();
6322                self.tables.push(table);
6323                self.by_name.insert(name.into(), idx);
6324            }
6325        }
6326        self.dirty_tables.insert(name.into());
6327    }
6328
6329    /// v7.37.42 (docker-fair SCALARSQ attack) — resolve a table name to
6330    /// its insertion-order index ONCE, so callers that need to fetch the
6331    /// same table many times (per-row PK probes in correlated scalar
6332    /// subqueries) can avoid the per-call `BTreeMap<String, usize>` string
6333    /// descent. The returned index is stable for the lifetime of the
6334    /// catalog snapshot the caller holds (same engine read guard).
6335    pub fn tables_position_of(&self, name: &str) -> Option<usize> {
6336        self.resolve_index(name)
6337    }
6338
6339    /// Direct positional fetch counterpart to [`tables_position_of`].
6340    /// `idx` must come from `tables_position_of` against the same catalog
6341    /// snapshot — out-of-range returns `None`.
6342    pub fn tables_at(&self, idx: usize) -> Option<&Table> {
6343        self.tables.get(idx)
6344    }
6345
6346    /// v7.34 (crash-recovery P0 #2) — replay a row-level redo log onto
6347    /// this catalog (the [`RowChange`] physical-redo apply primitive that
6348    /// row-level WAL recovery will use in place of statement re-execution).
6349    /// Applies each change in order via the same `Table` mutators the
6350    /// engine used — no uniqueness/FK/parse/plan: the original execution
6351    /// already validated, replay trusts and applies. Positions are
6352    /// physical and only valid when replayed from the matching checkpoint
6353    /// baseline in original order (see [`RowChange`] docs).
6354    ///
6355    /// A change naming an absent table, or whose position is out of range,
6356    /// is a corrupt/misaligned log and surfaces as an error rather than a
6357    /// silent skip.
6358    pub fn apply_redo(&mut self, changes: &[RowChange]) -> Result<(), StorageError> {
6359        // v7.37.5 (mailrs crash-recovery Ask 3) — true batched replay.
6360        // Pre-v7.37.5 each `RowChange::Delete` record ran a fresh
6361        // O(N) PersistentVec rebuild + O(N × indices × log N)
6362        // `rebuild_indices()` — 5000 records × 100k rows × 13 indices
6363        // ≈ 27 min on the mailrs prod-shape WAL.
6364        //
6365        // The strategy: group consecutive changes by table, and for
6366        // each run, compose all the row-level mutations through a
6367        // single "live" tracking vector + a per-table operation log,
6368        // then apply rows + indices ONCE at the end. The result:
6369        //  - DELETE blow-up: O(records × rows × indices × log rows)
6370        //    → O(rows × indices × log rows) — one rebuild per run.
6371        //  - Row-position semantics preserved: positions in a later
6372        //    `Delete` / `Update` record reference the layout produced
6373        //    by every earlier change; we walk the live-vector
6374        //    forward as each change is processed so positions
6375        //    translate correctly to the ORIGINAL row index space.
6376        //
6377        // For correctness, even with this batching `apply_redo`
6378        // remains in-order: a single per-table run only batches
6379        // a contiguous slice of changes targeting that table; a
6380        // mid-run change targeting a DIFFERENT table forces a
6381        // flush of the current run.
6382        let mut runs: alloc::vec::Vec<(String, alloc::vec::Vec<&RowChange>)> =
6383            alloc::vec::Vec::new();
6384        for change in changes {
6385            // v7.39 (flip crash-replay P0) — a replayed tombstone carries
6386            // the xmax the CRASHED process allocated, but this process's
6387            // version cursor restarted; without advancing it past every
6388            // replayed version, `Snapshot::visible`'s "deletion is in the
6389            // future" branch (xmax > snapshot.version) resurrects every
6390            // replayed delete. Same recovery contract as the snapshot
6391            // loader (`observe_persisted_version`, the pg_control-style
6392            // nextXid recovery).
6393            if let RowChange::Tombstone { xmax, .. } = change {
6394                row_header::observe_persisted_version(*xmax);
6395            }
6396            let table = match change {
6397                RowChange::Insert { table, .. }
6398                | RowChange::Update { table, .. }
6399                | RowChange::Delete { table, .. }
6400                | RowChange::Tombstone { table, .. } => table.clone(),
6401            };
6402            if runs.last().map(|(t, _)| t.as_str()) != Some(table.as_str()) {
6403                runs.push((table, alloc::vec::Vec::new()));
6404            }
6405            runs.last_mut().unwrap().1.push(change);
6406        }
6407        for (table_name, run) in runs {
6408            self.apply_redo_run_on_table(&table_name, &run)?;
6409        }
6410        Ok(())
6411    }
6412
6413    /// v7.37.5 — apply a contiguous slice of `RowChange`s all
6414    /// targeting the same `table_name`. Composes row mutations
6415    /// through a single live-tracking vector + a single tail
6416    /// for appended `Insert`s + a single in-place edit set for
6417    /// `Update`s, then writes the final row layout to
6418    /// `self.rows` and rebuilds indices ONCE.
6419    fn apply_redo_run_on_table(
6420        &mut self,
6421        table_name: &str,
6422        run: &[&RowChange],
6423    ) -> Result<(), StorageError> {
6424        // Look up the table once; the unchecked unwrap is safe
6425        // because the caller just resolved `table_name` for each
6426        // change.
6427        let table = self.get_mut(table_name).ok_or_else(|| {
6428            StorageError::Corrupt(alloc::format!("redo: unknown table {table_name:?}"))
6429        })?;
6430        // Live-tracking over both pre-existing rows and tail-
6431        // appended Insert rows. `live[i] = true` initially for
6432        // every existing row. Appended Inserts extend with `true`.
6433        // A `Delete` flips entries to `false` (using the position
6434        // mapping that walks live indices in order). An `Update`
6435        // edits in place — collected into an overlay map keyed by
6436        // ORIGINAL row position so later Updates win.
6437        let original_rows: alloc::vec::Vec<Row<'static>> = table.rows().iter().cloned().collect();
6438        let mut live: alloc::vec::Vec<bool> = alloc::vec![true; original_rows.len()];
6439        let mut tail: alloc::vec::Vec<Row<'static>> = alloc::vec::Vec::new();
6440        // Overlay: index into ORIGINAL row space (existing rows
6441        // 0..original_rows.len()) or into tail (offset
6442        // original_rows.len()). Map -> new values.
6443        let mut overlay: alloc::collections::BTreeMap<usize, alloc::vec::Vec<Value<'static>>> =
6444            alloc::collections::BTreeMap::new();
6445        // v7.37.15 (Epic W durable-tombstone slice) — extra bookkeeping
6446        // ONLY when this run actually carries an in-place `Tombstone`.
6447        // A tombstone keeps its row physically present but stamps `xmax`
6448        // on the header; the run finalizer `set_rows_and_rebuild_indices`
6449        // freezes every header (and reassigns ids), so we must re-stamp
6450        // in a post-pass keyed by RowId. When the run has no tombstone
6451        // (every default gate-off replay) this is all skipped and the
6452        // path below stays byte-for-byte the legacy one.
6453        let has_tomb = run.iter().any(|c| matches!(c, RowChange::Tombstone { .. }));
6454        // Ids of the pre-existing rows, snapshotted parallel to
6455        // `original_rows`, and ids of the tail rows filled from each
6456        // `Insert`'s carried `rowid`. Together they let a tombstone name
6457        // the exact row the writer stamped, independent of the ids the
6458        // finalizer will hand out. (When `!has_tomb`, both stay empty.)
6459        // v7.39 (flip crash-replay P0) — ids are tracked UNCONDITIONALLY
6460        // now: the finalizer preserves them so a later WAL record's
6461        // tombstone can still name rows this record produced.
6462        let orig_rowids: alloc::vec::Vec<row_header::RowId> =
6463            table.rowids().iter().copied().collect();
6464        // Headers snapshotted in lock-step: the finalizer preserves
6465        // them so earlier records' tombstone stamps survive.
6466        let orig_headers: alloc::vec::Vec<row_header::RowHeader> =
6467            table.headers().iter().copied().collect();
6468        let mut tail_rowids: alloc::vec::Vec<row_header::RowId> = alloc::vec::Vec::new();
6469        // (RowId, xmax) of every row this run tombstones.
6470        let mut tomb_targets: alloc::vec::Vec<(row_header::RowId, u64)> = alloc::vec::Vec::new();
6471        // Helper: given a "current" position (i.e. position in
6472        // the post-prior-deletes layout), translate to the
6473        // ABSOLUTE position in the unified live + tail space
6474        // by walking the live vector + tail. Returns None when
6475        // the position is out of range.
6476        fn translate(live: &[bool], tail_len: usize, current_pos: usize) -> Option<usize> {
6477            // Walk live[..] counting live entries until we hit
6478            // current_pos. Then if not yet matched, dip into tail.
6479            let mut seen = 0usize;
6480            for (i, &alive) in live.iter().enumerate() {
6481                if alive {
6482                    if seen == current_pos {
6483                        return Some(i);
6484                    }
6485                    seen += 1;
6486                }
6487            }
6488            // Position lives in tail. tail_len rows in the tail
6489            // are all live (we haven't deleted any tail rows in
6490            // this simplification; if we did, we'd extend `live`).
6491            let off = current_pos - seen;
6492            if off < tail_len {
6493                Some(live.len() + off)
6494            } else {
6495                None
6496            }
6497        }
6498        for change in run {
6499            match *change {
6500                RowChange::Insert { row, rowid, .. } => {
6501                    // Validate against schema before recording the
6502                    // change so a corrupt log surfaces as an error
6503                    // rather than silently mis-applying.
6504                    if row.len() != table.schema().columns.len() {
6505                        return Err(StorageError::ArityMismatch {
6506                            expected: table.schema().columns.len(),
6507                            actual: row.len(),
6508                        });
6509                    }
6510                    tail.push(row.clone());
6511                    // Keep the id lock-step with `tail` so a later
6512                    // tombstone (this run or a later WAL record) can
6513                    // find the row by the id the writer captured.
6514                    tail_rowids.push(*rowid);
6515                }
6516                RowChange::Update { pos, new_row, .. } => {
6517                    if new_row.len() != table.schema().columns.len() {
6518                        return Err(StorageError::ArityMismatch {
6519                            expected: table.schema().columns.len(),
6520                            actual: new_row.len(),
6521                        });
6522                    }
6523                    let abs = translate(&live, tail.len(), *pos).ok_or_else(|| {
6524                        StorageError::Corrupt(alloc::format!(
6525                            "redo: update_row position {pos} out of bounds in table {table_name:?}",
6526                        ))
6527                    })?;
6528                    // Tail edits are applied directly to `tail`
6529                    // (we own it); existing-row edits land in
6530                    // the overlay map keyed by original index.
6531                    if abs < live.len() {
6532                        overlay.insert(abs, new_row.clone());
6533                    } else {
6534                        tail[abs - live.len()] = Row::new(new_row.clone());
6535                    }
6536                }
6537                RowChange::Delete { positions, .. } => {
6538                    // De-dup + sort so the translate walk stays
6539                    // monotone (the second translate doesn't have
6540                    // to redo work the first one did, in principle;
6541                    // we keep it simple here and re-walk per
6542                    // position). Bounds-filter silently mirrors
6543                    // `Table::delete_rows`.
6544                    let mut sorted: alloc::vec::Vec<usize> = positions.clone();
6545                    sorted.sort_unstable();
6546                    sorted.dedup();
6547                    // Walk live[] once per Delete record to
6548                    // translate all positions in this record's
6549                    // post-prior-deletes layout to absolute
6550                    // indices. We MUST defer the live[] flip
6551                    // until after all positions are translated
6552                    // so two positions in the same record
6553                    // (e.g. [3, 7]) reference the same layout.
6554                    let mut to_flip_live: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
6555                    let mut to_flip_tail: alloc::vec::Vec<usize> = alloc::vec::Vec::new();
6556                    // Two-pointer walk: live[i] scanned monotonically,
6557                    // sorted positions consumed in order.
6558                    let mut seen = 0usize;
6559                    let mut sp = sorted.iter().peekable();
6560                    for (i, &alive) in live.iter().enumerate() {
6561                        if !alive {
6562                            continue;
6563                        }
6564                        while let Some(&&p) = sp.peek() {
6565                            if seen == p {
6566                                to_flip_live.push(i);
6567                                sp.next();
6568                            } else {
6569                                break;
6570                            }
6571                        }
6572                        if sp.peek().is_none() {
6573                            break;
6574                        }
6575                        seen += 1;
6576                    }
6577                    // Remaining positions fall into the tail.
6578                    for &p in sp {
6579                        // p >= seen and refers to the (p - seen)-th
6580                        // entry in tail. Filter out-of-bounds.
6581                        let off = p - seen;
6582                        if off < tail.len() {
6583                            to_flip_tail.push(off);
6584                        }
6585                    }
6586                    for i in to_flip_live {
6587                        live[i] = false;
6588                        // Any pending overlay edit for this
6589                        // index is moot — the row is gone.
6590                        overlay.remove(&i);
6591                    }
6592                    // Tail deletes: remove in REVERSE order so
6593                    // shifting indices stay valid.
6594                    to_flip_tail.sort_unstable();
6595                    to_flip_tail.dedup();
6596                    for off in to_flip_tail.into_iter().rev() {
6597                        tail.remove(off);
6598                        {
6599                            // Keep the id vector lock-step with `tail`.
6600                            tail_rowids.remove(off);
6601                        }
6602                        // Re-key tail-relative overlay entries that
6603                        // were past `off` — in practice tail edits
6604                        // are applied directly so the overlay map
6605                        // only holds existing-row keys; nothing to
6606                        // do here.
6607                    }
6608                }
6609                RowChange::Tombstone { rowids, xmax, .. } => {
6610                    // An in-place tombstone leaves the row physically
6611                    // present — it does not touch `live` / `tail` /
6612                    // `overlay`. Record the (id, xmax) targets; the
6613                    // post-finalizer pass re-stamps `xmax` onto the
6614                    // matching row's (otherwise-frozen) header.
6615                    for rid in rowids {
6616                        tomb_targets.push((*rid, *xmax));
6617                    }
6618                }
6619            }
6620        }
6621        // Compose the final row layout: keep existing rows where
6622        // live[i] = true, applying overlay edits in place; then
6623        // append the surviving tail.
6624        let mut new_rows: PersistentVec<Row> = PersistentVec::new();
6625        let mut new_hot_bytes: u64 = 0;
6626        let schema_snapshot = table.schema().clone();
6627        // Parallel to `new_rows` (only built when `has_tomb`): the RowId
6628        // of each row in its FINAL slot, so the post-pass can map a
6629        // tombstone target id → the slot to re-stamp `xmax` on.
6630        let mut final_rowids: alloc::vec::Vec<row_header::RowId> = alloc::vec::Vec::new();
6631        let mut final_headers: alloc::vec::Vec<row_header::RowHeader> = alloc::vec::Vec::new();
6632        for (i, row) in original_rows.into_iter().enumerate() {
6633            if !live[i] {
6634                continue;
6635            }
6636            let final_row = if let Some(new_values) = overlay.remove(&i) {
6637                Row::new(new_values)
6638            } else {
6639                row
6640            };
6641            new_hot_bytes = new_hot_bytes
6642                .saturating_add(row_body_encoded_len(&final_row, &schema_snapshot) as u64);
6643            new_rows.push_mut(final_row);
6644            final_rowids.push(
6645                orig_rowids
6646                    .get(i)
6647                    .copied()
6648                    .unwrap_or(row_header::RowId::UNASSIGNED),
6649            );
6650            final_headers.push(
6651                orig_headers
6652                    .get(i)
6653                    .copied()
6654                    .unwrap_or_else(row_header::RowHeader::frozen),
6655            );
6656        }
6657        for (off, row) in tail.into_iter().enumerate() {
6658            new_hot_bytes =
6659                new_hot_bytes.saturating_add(row_body_encoded_len(&row, &schema_snapshot) as u64);
6660            new_rows.push_mut(row);
6661            final_rowids.push(
6662                tail_rowids
6663                    .get(off)
6664                    .copied()
6665                    .unwrap_or(row_header::RowId::UNASSIGNED),
6666            );
6667            final_headers.push(row_header::RowHeader::frozen());
6668        }
6669        // v7.39 (flip crash-replay P0) — id-preserving finalizer, so a
6670        // LATER WAL record's tombstone still resolves rows this record
6671        // produced (per-statement replay used to reassign ids between
6672        // records, orphaning every cross-record tombstone target).
6673        table.set_rows_and_rebuild_indices_with_rowids(
6674            new_rows,
6675            new_hot_bytes,
6676            &final_rowids,
6677            &final_headers,
6678        );
6679        // v7.37.15 (Epic W durable-tombstone slice) — header-preserving
6680        // re-stamp. `set_rows_and_rebuild_indices` above froze every
6681        // header, so any row this run tombstoned is currently all-
6682        // visible again. Re-apply the `xmax` stamp by matching the
6683        // tombstone's target RowId against the final-slot id map. This
6684        // is what makes a gate-on DELETE durable across replay without
6685        // changing the on-disk snapshot format (headers/ids are still
6686        // NOT serialised — that is the deferred V6 coupling; see below).
6687        if has_tomb && !tomb_targets.is_empty() {
6688            let mut id_to_slot: alloc::collections::BTreeMap<row_header::RowId, usize> =
6689                alloc::collections::BTreeMap::new();
6690            for (slot, rid) in final_rowids.iter().enumerate() {
6691                if *rid != row_header::RowId::UNASSIGNED {
6692                    id_to_slot.insert(*rid, slot);
6693                }
6694            }
6695            let table = self.get_mut(table_name).ok_or_else(|| {
6696                StorageError::Corrupt(alloc::format!("redo: unknown table {table_name:?}"))
6697            })?;
6698            for (rid, xmax) in &tomb_targets {
6699                match id_to_slot.get(rid) {
6700                    Some(&slot) => {
6701                        // First-deleter-wins + bounds handled inside.
6702                        let _ = table.mark_row_deleted(slot, *xmax);
6703                    }
6704                    None => {
6705                        // The target row was not produced by THIS redo
6706                        // run and its id was not in the run-start
6707                        // snapshot — the documented cross-checkpoint
6708                        // limitation: after a checkpoint restore the
6709                        // table's ids are reassigned (not yet persisted
6710                        // in the envelope), so a tombstone naming a
6711                        // pre-checkpoint row cannot be resolved by id.
6712                        // Skipping leaves the row visible (identical to
6713                        // the pre-Epic-W non-durable behaviour); it is
6714                        // never a correctness regression, only an
6715                        // unclosed durability gap the V6 envelope slice
6716                        // closes. Counted for observability.
6717                        UNRESOLVED_TOMBSTONES.fetch_add(1, core::sync::atomic::Ordering::Relaxed);
6718                    }
6719                }
6720            }
6721        }
6722        Ok(())
6723    }
6724
6725    fn table_for_redo(&mut self, name: &str) -> Result<&mut Table, StorageError> {
6726        self.get_mut(name)
6727            .ok_or_else(|| StorageError::Corrupt(alloc::format!("redo: unknown table {name:?}")))
6728    }
6729
6730    /// v7.34 (crash-recovery P0 #2) — enable row-level redo capture on
6731    /// every table (the engine calls this before a mutating statement
6732    /// when persistence is on; idempotent, keeps any in-flight capture).
6733    pub fn enable_redo_all(&mut self) {
6734        for t in &mut self.tables {
6735            t.enable_redo();
6736        }
6737    }
6738
6739    /// v7.34 — drain the row-level redo captured across all tables, in
6740    /// table order then per-table apply order, and stop capturing. The
6741    /// engine calls this after a successful mutating statement and writes
6742    /// the returned [`RowChange`]s to the WAL in place of the SQL text.
6743    pub fn drain_redo(&mut self) -> Vec<RowChange> {
6744        let mut all = Vec::new();
6745        for t in &mut self.tables {
6746            all.extend(t.take_redo());
6747        }
6748        all
6749    }
6750
6751    pub fn table_count(&self) -> usize {
6752        self.tables.len()
6753    }
6754
6755    /// v7.14.0 — remove a table by name. Returns `true` when the
6756    /// table existed (and is now gone), `false` when it didn't.
6757    /// Used by `DROP TABLE` from pg_dump / mysqldump preambles
6758    /// where the dump re-creates schema and starts with
6759    /// `DROP TABLE IF EXISTS`.
6760    pub fn drop_table(&mut self, name: &str) -> bool {
6761        // v7.39 (round 436) — resolve through the session's temp namespace
6762        // first, exactly as a read would: MariaDB's plain `DROP TABLE tmp`
6763        // drops the TEMPORARY one and leaves a permanent namesake standing
6764        // (measured). Removing by the raw name would have dropped the
6765        // permanent table out from under every other session.
6766        let key = match self.temp_prefix.as_ref() {
6767            Some(p) => {
6768                let mangled = alloc::format!("{p}{name}");
6769                if self.by_name.contains_key(&mangled) {
6770                    mangled
6771                } else {
6772                    name.into()
6773                }
6774            }
6775            None => name.into(),
6776        };
6777        let Some(idx) = self.by_name.remove(&key) else {
6778            return false;
6779        };
6780        // v7.39 (round 496) — see `dirty_tables`. Recorded under the
6781        // RESOLVED key, which is what a commit-time merge looks up.
6782        self.dirty_tables.insert(key.clone());
6783        // swap_remove invalidates the trailing index → rebuild
6784        // by_name for affected entries.
6785        self.tables.swap_remove(idx);
6786        // Re-stamp moved table's index slot in by_name.
6787        if idx < self.tables.len() {
6788            let moved_name = self.tables[idx].schema.name.clone();
6789            self.by_name.insert(moved_name, idx);
6790        }
6791        true
6792    }
6793
6794    /// v7.16.2 — rename a table (mailrs round-10 A.5). Updates
6795    /// the schema name, the catalog name → index map, and
6796    /// rewrites every reference dangling at the table name:
6797    ///   * every FK on every OTHER table whose `parent_table`
6798    ///     pointed at the old name now points at the new
6799    ///     name, so FK enforcement keeps working
6800    ///   * every trigger watching the table updates its `table`
6801    ///     field
6802    /// Returns `Ok` on success; `Err(StorageError::TableNotFound)`
6803    /// when the old name isn't in the catalog and
6804    /// `Err(StorageError::DuplicateTable)` when the new name is
6805    /// already taken.
6806    pub fn rename_table(&mut self, old: &str, new: &str) -> Result<(), StorageError> {
6807        if old == new {
6808            return Ok(());
6809        }
6810        if self.by_name.contains_key(new) {
6811            return Err(StorageError::Corrupt(format!(
6812                "rename_table: target name {new:?} already exists"
6813            )));
6814        }
6815        let idx = self
6816            .by_name
6817            .remove(old)
6818            .ok_or_else(|| StorageError::TableNotFound { name: old.into() })?;
6819        self.tables[idx].schema.name = new.to_string();
6820        self.by_name.insert(new.to_string(), idx);
6821        for t in &mut self.tables {
6822            for fk in &mut t.schema.foreign_keys {
6823                if fk.parent_table == old {
6824                    fk.parent_table = new.to_string();
6825                }
6826            }
6827        }
6828        for trig in &mut self.triggers {
6829            if trig.table == old {
6830                trig.table = new.to_string();
6831            }
6832        }
6833        Ok(())
6834    }
6835
6836    /// v7.16.2 — rename an index by name. Walks every table
6837    /// since the index lives on its owning table; updates the
6838    /// name in place. Errors with `IndexNotFound` when no
6839    /// index matches. mailrs round-10 A.5.
6840    pub fn rename_index(&mut self, old: &str, new: &str) -> Result<(), StorageError> {
6841        if old == new {
6842            return Ok(());
6843        }
6844        // Reject the new name if it already exists anywhere.
6845        for t in &self.tables {
6846            if t.indices.iter().any(|i| i.name == new) {
6847                return Err(StorageError::Corrupt(format!(
6848                    "rename_index: target name {new:?} already exists"
6849                )));
6850            }
6851        }
6852        for t in &mut self.tables {
6853            for i in &mut t.indices {
6854                if i.name == old {
6855                    i.name = new.to_string();
6856                    return Ok(());
6857                }
6858            }
6859        }
6860        Err(StorageError::IndexNotFound { name: old.into() })
6861    }
6862
6863    /// v7.14.0 — remove a named index across the catalog.
6864    /// Returns `true` when found + dropped.
6865    pub fn drop_named_index(&mut self, name: &str) -> bool {
6866        for t in &mut self.tables {
6867            let before = t.indices.len();
6868            t.indices.retain(|i| i.name != name);
6869            if t.indices.len() != before {
6870                return true;
6871            }
6872        }
6873        false
6874    }
6875
6876    /// Borrow-free copy of every table's name in catalog order
6877    /// (= insertion order, matching the on-disk encoding).
6878    pub fn table_names(&self) -> Vec<String> {
6879        self.tables.iter().map(|t| t.schema.name.clone()).collect()
6880    }
6881
6882    /// v7.39 (round 436) — the marker every session's temporary-table
6883    /// namespace starts with. Public so the catalog synths can tell a
6884    /// temp table from an ordinary one without knowing the session id.
6885    pub const TEMP_NAME_MARKER: &'static str = "__spg_temp_";
6886
6887    /// v7.39 (round 437) — how a stored table name should appear to the
6888    /// CALLING session in a catalog listing (SHOW TABLES, pg_class,
6889    /// information_schema, …):
6890    ///   * an ordinary table → its own name
6891    ///   * this session's temporary table → its logical name, prefix stripped
6892    ///   * another session's temporary table → `None`, i.e. not listed
6893    ///
6894    /// Measured on both oracles: MariaDB 11 and PG 18 each list the calling
6895    /// session's own temporary tables and neither lists anybody else's.
6896    /// Round 436 stored temp tables under a prefix without teaching the
6897    /// listings about it, so the mangled names leaked to every client.
6898    #[must_use]
6899    pub fn listed_name<'a>(&self, stored: &'a str) -> Option<&'a str> {
6900        if !stored.starts_with(Self::TEMP_NAME_MARKER) {
6901            return Some(stored);
6902        }
6903        let prefix = self.temp_prefix.as_ref()?;
6904        stored.strip_prefix(prefix.as_str())
6905    }
6906
6907    /// The listing names of every table this session may see, in catalog
6908    /// order. See [`Catalog::listed_name`].
6909    #[must_use]
6910    pub fn visible_table_names(&self) -> Vec<String> {
6911        self.tables
6912            .iter()
6913            .filter_map(|t| self.listed_name(&t.schema.name).map(String::from))
6914            .collect()
6915    }
6916
6917    /// v5.1: register a cold-tier segment that already lives in
6918    /// memory (caller did the file read). Returns the
6919    /// `segment_id` that `RowLocator::Cold { segment_id, .. }`
6920    /// will reference — currently this is just the index into
6921    /// `cold_segments`, but treat it as an opaque token.
6922    ///
6923    /// Storage is `no_std`, so file I/O is the caller's
6924    /// responsibility — `spg-server` reads the file and forwards
6925    /// the bytes here. The bytes stay resident in the catalog
6926    /// for the life of the `Catalog`, parsed only once.
6927    pub fn load_segment_bytes(&mut self, bytes: Vec<u8>) -> Result<u32, StorageError> {
6928        let id = u32::try_from(self.cold_segments.len()).map_err(|_| {
6929            StorageError::Corrupt("cold segment count would exceed u32::MAX".into())
6930        })?;
6931        let seg = OwnedSegment::from_bytes(bytes)
6932            .map_err(|e| StorageError::Corrupt(format!("cold segment parse failed: {e}")))?;
6933        self.cold_segments.push(Some(Arc::new(seg)));
6934        Ok(id)
6935    }
6936
6937    /// v6.7.3 — register a cold-tier segment at a specific id. Used
6938    /// by the spg-server manifest-boot path so segments whose
6939    /// neighbouring ids were retired by compaction still get back
6940    /// the same `segment_id` they had pre-restart (the
6941    /// `RowLocator::Cold { segment_id }` baked into the BTree-index
6942    /// snapshot persists across restart and must continue to
6943    /// resolve).
6944    ///
6945    /// Pads the Vec with `None` slots up to `target_id` if needed.
6946    /// Errors when the target slot is already occupied (would
6947    /// stomp another segment), the parse fails, or `target_id`
6948    /// exceeds `u32::MAX`.
6949    pub fn load_segment_bytes_at(
6950        &mut self,
6951        target_id: u32,
6952        bytes: Vec<u8>,
6953    ) -> Result<(), StorageError> {
6954        let seg = OwnedSegment::from_bytes(bytes)
6955            .map_err(|e| StorageError::Corrupt(format!("cold segment parse failed: {e}")))?;
6956        let idx = target_id as usize;
6957        while self.cold_segments.len() <= idx {
6958            self.cold_segments.push(None);
6959        }
6960        if self.cold_segments[idx].is_some() {
6961            return Err(StorageError::Corrupt(format!(
6962                "load_segment_bytes_at: segment_id {target_id} already occupied"
6963            )));
6964        }
6965        self.cold_segments[idx] = Some(Arc::new(seg));
6966        Ok(())
6967    }
6968
6969    /// v6.7.3 — retire a cold-tier segment slot (compaction-driven).
6970    /// The physical file is the caller's concern (typically kept
6971    /// on disk until the next CHECKPOINT writes a manifest that
6972    /// no longer lists it); this just flips the in-memory slot
6973    /// to `None` so later cold lookups for `segment_id` resolve
6974    /// as "unknown" instead of returning a stale row.
6975    ///
6976    /// No-op when the slot is already `None`. Errors only when
6977    /// `segment_id` is out of bounds.
6978    pub fn tombstone_segment(&mut self, segment_id: u32) -> Result<(), StorageError> {
6979        let idx = segment_id as usize;
6980        if idx >= self.cold_segments.len() {
6981            return Err(StorageError::Corrupt(format!(
6982                "tombstone_segment: segment_id {segment_id} out of bounds (len={})",
6983                self.cold_segments.len()
6984            )));
6985        }
6986        self.cold_segments[idx] = None;
6987        Ok(())
6988    }
6989
6990    /// Number of *active* (non-tombstoned) cold segments.
6991    #[must_use]
6992    pub fn cold_segment_count(&self) -> usize {
6993        self.cold_segments.iter().filter(|s| s.is_some()).count()
6994    }
6995
6996    /// v7.37.42 (docker-fair SCALARSQ attack 3) — short-circuit guard
6997    /// for scan loops that conditionally walk the cold tier. Returns
6998    /// `false` when the catalog has never loaded a cold segment (or all
6999    /// segments are tombstoned), so callers can skip the per-table cold
7000    /// PK-index walk entirely on hot-only databases. O(N segments);
7001    /// typical N is small (single-digit) so the check is sub-µs.
7002    #[must_use]
7003    pub fn has_any_cold_segments(&self) -> bool {
7004        self.cold_segments.iter().any(Option::is_some)
7005    }
7006
7007    /// Slot count including tombstones (= the next id the
7008    /// no-arg `load_segment_bytes` would allocate).
7009    #[must_use]
7010    pub fn cold_segment_slot_count(&self) -> usize {
7011        self.cold_segments.len()
7012    }
7013
7014    /// v6.2.7 — list every *active* cold-tier segment id known to
7015    /// this catalog (skips compaction tombstones since v6.7.3).
7016    /// Used by EXPLAIN ANALYZE to annotate scan nodes with the
7017    /// segments they could have walked.
7018    #[must_use]
7019    pub fn cold_segment_ids_global(&self) -> Vec<u32> {
7020        self.cold_segments
7021            .iter()
7022            .enumerate()
7023            .filter_map(|(i, s)| s.as_ref().map(|_| i as u32))
7024            .collect()
7025    }
7026
7027    /// v5.2.1: sum of `Table::hot_bytes` across every table. The v5.2
7028    /// freezer compares this against `SPG_HOT_TIER_BYTES` (parsed at
7029    /// server startup; default 4 GiB) and wakes when the budget is
7030    /// crossed. Pre-freezer (v5.2.1) this is measurement-only — the
7031    /// counter exposes whether the budget is being approached without
7032    /// triggering any demotion.
7033    #[must_use]
7034    pub fn hot_tier_bytes(&self) -> u64 {
7035        self.tables
7036            .iter()
7037            .map(Table::hot_bytes)
7038            .fold(0u64, u64::saturating_add)
7039    }
7040
7041    /// v5.2.2: freeze the **first** `max_rows` rows of `table_name`'s
7042    /// hot tier into a brand-new cold-tier segment. The named `BTree`
7043    /// index supplies the per-row PK (its column must be an integer
7044    /// type — v5.2.2 only supports `IndexKey::Int` PKs, matching the
7045    /// `index_key_as_u64` constraint used by the cold-tier lookup
7046    /// path). On success returns a [`FreezeReport`] with the
7047    /// freshly-allocated segment id, the count of rows that moved,
7048    /// the encoded segment bytes (so the caller can persist them to
7049    /// disk for later reload via `SPG_PRELOAD_COLD_SEGMENT`), and the
7050    /// hot-tier byte delta that was reclaimed.
7051    ///
7052    /// **Semantics**:
7053    /// 1. The first `max_rows` rows (by hot-tier position — same as
7054    ///    insertion order under v4.39 `PersistentVec`) are read.
7055    /// 2. Rows are sorted ascending by PK and serialised into a new
7056    ///    segment via [`encode_segment`].
7057    /// 3. The hot rows are dropped via [`Table::delete_rows`]; the
7058    ///    `rebuild_indices` it triggers regenerates `Hot` locators
7059    ///    for every remaining row (their positions shift down by
7060    ///    `max_rows`). Existing `Cold` locators in this index — from
7061    ///    a previous freeze — are also rebuilt **but with empty
7062    ///    payload** since rebuild reads only `self.rows`; this
7063    ///    routine re-registers them at the end of the call so the
7064    ///    user-visible state preserves all prior cold locators.
7065    /// 4. The new segment is loaded into `self.cold_segments` via
7066    ///    [`Catalog::load_segment_bytes`] (allocating a fresh
7067    ///    `segment_id`). New `Cold` locators are registered on the
7068    ///    named index — one per frozen row.
7069    ///
7070    /// **v5.2.2 limits** (relaxed in later sub-versions):
7071    /// - INSERT-only flow: subsequent UPDATE/DELETE on a frozen row
7072    ///   returns a stale-locator error (no promote-on-write until
7073    ///   v5.2.3).
7074    /// - Single-table scope: callers iterate tables themselves.
7075    /// - All-or-nothing: returns `Err` and leaves catalog unchanged
7076    ///   if any step fails before the atomic swap point.
7077    ///
7078    /// Errors:
7079    /// - [`StorageError::Corrupt`] for missing table/index, non-`BTree`
7080    ///   index, non-integer PK column, `max_rows == 0`, or
7081    ///   `max_rows > row_count`.
7082    /// - The encoder's [`SegmentError`] surfaces as `Corrupt` (the
7083    ///   only realistic source is "a single row is larger than the
7084    ///   page size"; SPG schemas don't hit it in practice).
7085    pub fn freeze_oldest_to_cold(
7086        &mut self,
7087        table_name: &str,
7088        index_name: &str,
7089        max_rows: usize,
7090    ) -> Result<FreezeReport, StorageError> {
7091        // --- validation phase: never mutates ---------------------
7092        if max_rows == 0 {
7093            return Err(StorageError::Corrupt(
7094                "freeze_oldest_to_cold: max_rows must be > 0".into(),
7095            ));
7096        }
7097        let table = self.get(table_name).ok_or_else(|| {
7098            StorageError::Corrupt(format!(
7099                "freeze_oldest_to_cold: table {table_name:?} not found"
7100            ))
7101        })?;
7102        if max_rows > table.rows.len() {
7103            return Err(StorageError::Corrupt(format!(
7104                "freeze_oldest_to_cold: max_rows {max_rows} > row_count {}",
7105                table.rows.len()
7106            )));
7107        }
7108        let idx = table
7109            .indices
7110            .iter()
7111            .find(|i| i.name == index_name)
7112            .ok_or_else(|| {
7113                StorageError::Corrupt(format!(
7114                    "freeze_oldest_to_cold: index {index_name:?} not found on {table_name:?}"
7115                ))
7116            })?;
7117        if !matches!(idx.kind, IndexKind::BTree(_)) {
7118            return Err(StorageError::Corrupt(format!(
7119                "freeze_oldest_to_cold: index {index_name:?} is NSW; only BTree indices may freeze"
7120            )));
7121        }
7122        let column_position = idx.column_position;
7123
7124        // --- segment build phase: reads only --------------------
7125        let schema = table.schema.clone();
7126        let mut to_freeze: Vec<(u64, Vec<u8>, IndexKey)> = Vec::with_capacity(max_rows);
7127        for row_idx in 0..max_rows {
7128            let row = table.rows.get(row_idx).expect("bounds-checked above");
7129            let key = IndexKey::from_value(&row.values[column_position]).ok_or_else(|| {
7130                StorageError::Corrupt(format!(
7131                    "freeze_oldest_to_cold: row {row_idx} has NULL / non-key value in index column"
7132                ))
7133            })?;
7134            let pk_u64 = index_key_as_u64(&key).ok_or_else(|| {
7135                StorageError::Corrupt(format!(
7136                    "freeze_oldest_to_cold: index {index_name:?} column type is non-integer; \
7137                     v5.2.2 cold tier requires IndexKey::Int (Text PK lands in v5.5+)"
7138                ))
7139            })?;
7140            to_freeze.push((pk_u64, encode_row_body_dense(row, &schema), key));
7141        }
7142        // encode_segment requires ascending u64 keys. Sort by PK
7143        // before encoding; the caller's row-position order is not
7144        // necessarily PK order (e.g. workloads that insert random
7145        // PKs).
7146        to_freeze.sort_by_key(|(k, _, _)| *k);
7147        // Reject duplicate PKs — encode_segment also rejects them
7148        // (`SegmentError::UnsortedKey`), but the resulting error
7149        // message there is misleading. Surface a clearer one.
7150        for w in to_freeze.windows(2) {
7151            if w[0].0 == w[1].0 {
7152                return Err(StorageError::Corrupt(format!(
7153                    "freeze_oldest_to_cold: duplicate PK {} in freeze batch",
7154                    w[0].0
7155                )));
7156            }
7157        }
7158        // Snapshot the (key, locator) pairs that will be registered
7159        // post-swap. Cloning the IndexKey out before the move makes
7160        // the registration loop borrow-free.
7161        let post_swap_keys: Vec<IndexKey> = to_freeze.iter().map(|(_, _, k)| k.clone()).collect();
7162        // Segment encode is now infallible w.r.t. ordering. Map the
7163        // `SegmentError` into a `StorageError::Corrupt` so the
7164        // public surface stays one error type.
7165        let seg_rows: Vec<(u64, Vec<u8>)> = to_freeze
7166            .into_iter()
7167            .map(|(k, body, _)| (k, body))
7168            .collect();
7169        let frozen_rows = seg_rows.len();
7170        let (seg_bytes, _meta) = encode_segment(seg_rows.into_iter(), 0.01, SEGMENT_PAGE_BYTES)
7171            .map_err(|e| StorageError::Corrupt(format!("freeze_oldest_to_cold: encode: {e}")))?;
7172
7173        // --- atomic swap phase: mutations only past this point ---
7174        // v5.2.3 made `Table::rebuild_indices` preserve every Cold
7175        // locator across the per-table rebuild, so `delete_rows`
7176        // below no longer wipes prior-freeze cold entries. The pre-
7177        // v5.2.3 capture-then-re-register that used to live here
7178        // was removed in v5.3.1 — keeping it would double-count
7179        // every prior-frozen key's Cold locator on each subsequent
7180        // freeze.
7181        let bytes_before = self.get(table_name).expect("just validated").hot_bytes();
7182        let positions: Vec<usize> = (0..max_rows).collect();
7183        let t_mut = self
7184            .get_mut(table_name)
7185            .expect("just validated; still present");
7186        let removed = t_mut.delete_rows(&positions);
7187        debug_assert_eq!(removed, max_rows, "delete_rows count matches request");
7188        let bytes_after = t_mut.hot_bytes();
7189        let bytes_freed = bytes_before.saturating_sub(bytes_after);
7190
7191        let segment_id = self
7192            .load_segment_bytes(seg_bytes.clone())
7193            .map_err(|e| StorageError::Corrupt(format!("freeze_oldest_to_cold: load: {e}")))?;
7194        let new_cold = post_swap_keys.into_iter().map(|k| {
7195            (
7196                k,
7197                RowLocator::Cold {
7198                    segment_id,
7199                    page_offset: 0,
7200                },
7201            )
7202        });
7203        let t_mut = self.get_mut(table_name).expect("still present");
7204        t_mut.register_cold_locators(index_name, new_cold)?;
7205        // r944 — a freeze has to say that it froze something.
7206        //
7207        // `has_cold_rows_fast()` reads the cached count, and neither
7208        // freeze path touched it, so afterwards it answered "no cold
7209        // rows" while cold rows existed. That predicate gates four join
7210        // paths, and a gate that wrongly declines the cold-aware path
7211        // drops the frozen rows from the answer.
7212        //
7213        // Marking it stale rather than adding to it: stale reads as
7214        // true, which is the safe direction, and this function cannot
7215        // know the exact total (rows may already have been cold). ANALYZE
7216        // recomputes the number.
7217        t_mut.mark_cold_row_count_stale();
7218
7219        Ok(FreezeReport {
7220            segment_id,
7221            frozen_rows,
7222            bytes_freed,
7223            segment_bytes: seg_bytes,
7224        })
7225    }
7226
7227    /// v5.1: borrow the cold segment at `segment_id`. Used by the
7228    /// spg-server preload path to enumerate (key, locator) pairs
7229    /// after loading a segment, so it can call
7230    /// [`Table::register_cold_locators`] without re-parsing the
7231    /// bytes.
7232    #[must_use]
7233    pub fn cold_segment(&self, segment_id: u32) -> Option<&OwnedSegment> {
7234        self.cold_segments
7235            .get(segment_id as usize)
7236            .and_then(|s| s.as_deref())
7237    }
7238
7239    /// v5.1: resolve a single `RowLocator::Cold` to its underlying
7240    /// `Row`. Decoupled from [`Catalog::lookup_by_pk`] so callers
7241    /// iterating a multi-locator slice (e.g. the engine's index
7242    /// seek path) can dispatch per locator instead of getting back
7243    /// only the first row for a key. Returns `None` when the
7244    /// segment isn't registered, the key isn't `u64`-coercible, or
7245    /// the segment doesn't actually carry the key (bloom or page-
7246    /// index reject).
7247    pub fn resolve_cold_locator(
7248        &self,
7249        table_name: &str,
7250        segment_id: u32,
7251        key: &IndexKey,
7252    ) -> Option<Row<'static>> {
7253        let t = self.get(table_name)?;
7254        let u64_key = index_key_as_u64(key)?;
7255        let seg = self.cold_segments.get(segment_id as usize)?.as_ref()?;
7256        let payload = seg.lookup(u64_key)?;
7257        let (row, _) = decode_row_body_dense(&payload, &t.schema, seg.codec_version()).ok()?;
7258        // v7.39 (pg_stat blks knife) — one cold-tier "block read".
7259        self.cold_read_stats
7260            .cold_reads
7261            .fetch_add(1, core::sync::atomic::Ordering::Relaxed);
7262        Some(row)
7263    }
7264
7265    /// v5.1: indexed PK lookup that dispatches per locator,
7266    /// returning the first matching row from either the hot tier
7267    /// (`Table::rows`) or a registered cold segment.
7268    ///
7269    /// The cold path requires the index column to be coercible to
7270    /// a `u64` (the segment's PK type) and the segment payload to
7271    /// be a [`encode_row_body_dense`]-encoded row body for the
7272    /// same schema. v5.1 ships this for BIGINT / INT / SMALLINT
7273    /// PKs; other types fall through to hot-only behavior.
7274    ///
7275    /// Returns `None` if (a) the table or index doesn't exist,
7276    /// (b) the key isn't in the index at all, or (c) the key was
7277    /// resolved to a stale locator (Hot index out of range, Cold
7278    /// segment id unknown, segment lookup miss). Does not surface
7279    /// segment-decode errors — those would indicate corrupted
7280    /// cold-tier files and should be caught at
7281    /// [`Catalog::load_segment_bytes`] time.
7282    pub fn lookup_by_pk(&self, table: &str, index_name: &str, key: &IndexKey) -> Option<Row<'_>> {
7283        let t = self.get(table)?;
7284        let idx = t.indices.iter().find(|i| i.name == index_name)?;
7285        let locators = idx.lookup_eq(key);
7286        let cold_u64_key = index_key_as_u64(key);
7287        for loc in locators {
7288            match *loc {
7289                RowLocator::Hot(i) => {
7290                    if let Some(row) = t.rows.get(i) {
7291                        return Some(row.clone());
7292                    }
7293                }
7294                RowLocator::Cold {
7295                    segment_id,
7296                    page_offset: _,
7297                } => {
7298                    let Some(u64_key) = cold_u64_key else {
7299                        // Key type not coercible to u64 — cold tier
7300                        // only handles BIGINT/INT/SMALLINT in v5.1.
7301                        continue;
7302                    };
7303                    let Some(seg) = self
7304                        .cold_segments
7305                        .get(segment_id as usize)
7306                        .and_then(|s| s.as_deref())
7307                    else {
7308                        // v6.7.3 — `None` slot = compaction
7309                        // retired this segment; the live locator
7310                        // on a freshly-compacted index points to
7311                        // the merged segment_id, so a Cold hit
7312                        // here against a tombstone means the BTree
7313                        // entry hasn't been swapped yet (mid-
7314                        // compaction reader race) or the caller is
7315                        // looking up a stale snapshot. Skip — the
7316                        // next locator in the list, if any, is
7317                        // typically the merged segment.
7318                        continue;
7319                    };
7320                    let Some(payload) = seg.lookup(u64_key) else {
7321                        continue;
7322                    };
7323                    let (row, _) =
7324                        decode_row_body_dense(&payload, &t.schema, seg.codec_version()).ok()?;
7325                    return Some(row);
7326                }
7327            }
7328        }
7329        None
7330    }
7331
7332    /// v5.2.3: promote a frozen row back to the hot tier so an
7333    /// UPDATE / DELETE can mutate it. Reads the cold-tier row body
7334    /// (decoded from its registered segment), pushes it into
7335    /// `table.rows` via [`Table::insert`] (which also adds a fresh
7336    /// `Hot(new_idx)` locator on `index_name`), then retires the
7337    /// shadowed `Cold` locator via
7338    /// [`Table::remove_cold_locators_for_key`]. The cold-tier row
7339    /// in the segment file becomes garbage — recoverable when a
7340    /// future cold-segment compaction job lands.
7341    ///
7342    /// Returns:
7343    /// - `Ok(Some(new_hot_idx))` when the key resolved through a
7344    ///   cold locator and the promote completed. `new_hot_idx` is
7345    ///   the position the row now occupies in `table.rows`.
7346    /// - `Ok(None)` when the key has no Cold locator on the index
7347    ///   (already hot, or wasn't present at all). Callers treat this
7348    ///   as "nothing to do here, fall back to the hot-only path".
7349    ///
7350    /// Errors when the table / index doesn't exist, the index isn't
7351    /// `BTree`, the cold segment is missing / can't decode the row,
7352    /// or the inferred row body fails `Table::insert` validation.
7353    pub fn promote_cold_row(
7354        &mut self,
7355        table_name: &str,
7356        index_name: &str,
7357        key: &IndexKey,
7358    ) -> Result<Option<usize>, StorageError> {
7359        let cold_loc = self.find_cold_locator(table_name, index_name, key)?;
7360        let Some((segment_id, _page_offset)) = cold_loc else {
7361            return Ok(None);
7362        };
7363        let u64_key = index_key_as_u64(key).ok_or_else(|| {
7364            StorageError::Corrupt(
7365                "promote_cold_row: key type not coercible to u64 (cold tier requires integer PK)"
7366                    .into(),
7367            )
7368        })?;
7369        // Read the row body from the segment. Borrow the segment +
7370        // schema short-term so we can then take `&mut self` for the
7371        // hot-side insert.
7372        let schema = self
7373            .get(table_name)
7374            .ok_or_else(|| {
7375                StorageError::Corrupt(format!("promote_cold_row: table {table_name:?} not found"))
7376            })?
7377            .schema
7378            .clone();
7379        let seg = self
7380            .cold_segments
7381            .get(segment_id as usize)
7382            .and_then(|s| s.as_ref())
7383            .ok_or_else(|| {
7384                StorageError::Corrupt(format!(
7385                    "promote_cold_row: segment {segment_id} not registered on catalog"
7386                ))
7387            })?;
7388        let payload = seg.lookup(u64_key).ok_or_else(|| {
7389            StorageError::Corrupt(format!(
7390                "promote_cold_row: key {u64_key} resolves to segment {segment_id} \
7391                 but the segment's bloom/page lookup didn't return a row"
7392            ))
7393        })?;
7394        let (row, _consumed) = decode_row_body_dense(&payload, &schema, seg.codec_version())?;
7395        // Insert the promoted row into the hot tier. `Table::insert`
7396        // appends to `self.rows`, adds a `Hot(new_idx)` locator to
7397        // every BTree index covering the row's keyed columns, and
7398        // increments `hot_bytes`.
7399        let t = self
7400            .get_mut(table_name)
7401            .expect("table existed at lookup time");
7402        t.insert(row)?;
7403        let new_hot_idx =
7404            t.rows.len().checked_sub(1).ok_or_else(|| {
7405                StorageError::Corrupt("promote_cold_row: empty after insert".into())
7406            })?;
7407        // The hot insert added Hot(new_idx) alongside the still-
7408        // present Cold locator. Drop the Cold entry so future
7409        // lookups return only the fresh hot row.
7410        t.remove_cold_locators_for_key(index_name, key)?;
7411        Ok(Some(new_hot_idx))
7412    }
7413
7414    /// v5.2.3: shadow a frozen row's index entry. Used by DELETE
7415    /// when the row to remove lives in a cold-tier segment — the
7416    /// row body stays in the segment file (becoming garbage) but
7417    /// every `Cold` locator for `key` on `index_name` is removed
7418    /// so PK lookups stop returning it.
7419    ///
7420    /// Returns the number of cold locators retired (0 when the key
7421    /// has no cold entries — the DELETE fell on a hot row or a
7422    /// key that was already absent). Errors when the table /
7423    /// index doesn't exist or the index isn't `BTree`.
7424    ///
7425    /// Cold-segment compaction (which merges shadowed-heavy
7426    /// segments and reclaims their disk footprint) lands in a
7427    /// later v5.x sub-version; until then, repeated UPDATE/DELETE
7428    /// of cold rows can amplify cold-segment disk usage by up to
7429    /// 1-2× — still well under typical LSM-tree shadowing because
7430    /// SPG segments are bulk-baked, not write-merged.
7431    pub fn shadow_cold_row(
7432        &mut self,
7433        table_name: &str,
7434        index_name: &str,
7435        key: &IndexKey,
7436    ) -> Result<usize, StorageError> {
7437        let t = self.get_mut(table_name).ok_or_else(|| {
7438            StorageError::Corrupt(format!("shadow_cold_row: table {table_name:?} not found"))
7439        })?;
7440        t.remove_cold_locators_for_key(index_name, key)
7441    }
7442
7443    /// v6.7.4 — read-only slice preparation for the parallel
7444    /// freezer. Walks rows in `row_range`, builds the
7445    /// `(pk_u64, encoded_body, IndexKey)` triples that the
7446    /// coordinator's k-way merge consumes, sorts the slice by
7447    /// `pk_u64`, and returns a [`FreezeSlice`].
7448    ///
7449    /// Caller invariants:
7450    /// - `row_range.end <= table.rows.len()` (caller's job to
7451    ///   compute the partition).
7452    /// - All slices passed to `commit_freeze_slices` must cover a
7453    ///   contiguous half-open range `[0, total_max_rows)` with no
7454    ///   gaps and no overlaps. The coordinator validates this
7455    ///   invariant before committing.
7456    ///
7457    /// `&self`-only — multiple workers can run this concurrently
7458    /// against the same `Catalog` reference under the engine's
7459    /// write lock (workers don't mutate; the coordinator does).
7460    pub fn prepare_freeze_slice(
7461        &self,
7462        table_name: &str,
7463        index_name: &str,
7464        row_range: core::ops::Range<usize>,
7465    ) -> Result<FreezeSlice, StorageError> {
7466        let table = self.get(table_name).ok_or_else(|| {
7467            StorageError::Corrupt(format!(
7468                "prepare_freeze_slice: table {table_name:?} not found"
7469            ))
7470        })?;
7471        let idx = table
7472            .indices
7473            .iter()
7474            .find(|i| i.name == index_name)
7475            .ok_or_else(|| {
7476                StorageError::Corrupt(format!(
7477                    "prepare_freeze_slice: index {index_name:?} not found on {table_name:?}"
7478                ))
7479            })?;
7480        if !matches!(idx.kind, IndexKind::BTree(_)) {
7481            return Err(StorageError::Corrupt(format!(
7482                "prepare_freeze_slice: index {index_name:?} is NSW; only BTree indices may freeze"
7483            )));
7484        }
7485        if row_range.end > table.rows.len() {
7486            return Err(StorageError::Corrupt(format!(
7487                "prepare_freeze_slice: row_range end {} > row_count {}",
7488                row_range.end,
7489                table.rows.len()
7490            )));
7491        }
7492        let column_position = idx.column_position;
7493        let schema = table.schema.clone();
7494        let mut rows: Vec<(u64, Vec<u8>, IndexKey)> = Vec::with_capacity(row_range.len());
7495        for row_idx in row_range.clone() {
7496            let row = table.rows.get(row_idx).expect("bounds-checked above");
7497            let key = IndexKey::from_value(&row.values[column_position]).ok_or_else(|| {
7498                StorageError::Corrupt(format!(
7499                    "prepare_freeze_slice: row {row_idx} has NULL / non-key value in index column"
7500                ))
7501            })?;
7502            let pk_u64 = index_key_as_u64(&key).ok_or_else(|| {
7503                StorageError::Corrupt(format!(
7504                    "prepare_freeze_slice: index {index_name:?} column type is non-integer; \
7505                     v5.2.2 cold tier requires IndexKey::Int (Text PK lands in v5.5+)"
7506                ))
7507            })?;
7508            rows.push((pk_u64, encode_row_body_dense(row, &schema), key));
7509        }
7510        rows.sort_by_key(|(k, _, _)| *k);
7511        Ok(FreezeSlice { row_range, rows })
7512    }
7513
7514    /// v6.7.4 — coordinator commit step. Merges N
7515    /// [`FreezeSlice`]s into one segment via the standard
7516    /// [`encode_segment`] path, atomically swaps the catalog
7517    /// state (delete the union row range + register Cold
7518    /// locators + load the segment).
7519    ///
7520    /// Validates that the slices cover a contiguous, gap-free,
7521    /// overlap-free half-open range starting at index 0 (the
7522    /// freezer always freezes "oldest first" — same semantics as
7523    /// the single-threaded [`Catalog::freeze_oldest_to_cold`]).
7524    ///
7525    /// Empty `slices` → no-op success (returns a zero-row report
7526    /// without mutating). Total row count = `Σ slice.rows.len()`.
7527    pub fn commit_freeze_slices(
7528        &mut self,
7529        table_name: &str,
7530        index_name: &str,
7531        slices: Vec<FreezeSlice>,
7532    ) -> Result<FreezeReport, StorageError> {
7533        // --- validation phase: never mutates ---------------------
7534        let table = self.get(table_name).ok_or_else(|| {
7535            StorageError::Corrupt(format!(
7536                "commit_freeze_slices: table {table_name:?} not found"
7537            ))
7538        })?;
7539        let idx = table
7540            .indices
7541            .iter()
7542            .find(|i| i.name == index_name)
7543            .ok_or_else(|| {
7544                StorageError::Corrupt(format!(
7545                    "commit_freeze_slices: index {index_name:?} not found on {table_name:?}"
7546                ))
7547            })?;
7548        if !matches!(idx.kind, IndexKind::BTree(_)) {
7549            return Err(StorageError::Corrupt(format!(
7550                "commit_freeze_slices: index {index_name:?} is NSW; only BTree indices may freeze"
7551            )));
7552        }
7553        // Validate slice coverage: contiguous from 0, no gaps, no
7554        // overlaps. Allow the caller to pass slices in any order —
7555        // sort by row_range.start first.
7556        let mut ordered = slices;
7557        ordered.sort_by_key(|s| s.row_range.start);
7558        // Drop fully-empty slices that fell out of an uneven
7559        // partition; they carry no data but contribute to the
7560        // contiguity check, so keep them in line.
7561        let mut expected_start = 0usize;
7562        for s in &ordered {
7563            if s.row_range.start != expected_start {
7564                return Err(StorageError::Corrupt(format!(
7565                    "commit_freeze_slices: gap/overlap at row {}; expected start {}",
7566                    s.row_range.start, expected_start
7567                )));
7568            }
7569            expected_start = s.row_range.end;
7570        }
7571        let max_rows = expected_start;
7572        if max_rows > table.rows.len() {
7573            return Err(StorageError::Corrupt(format!(
7574                "commit_freeze_slices: total row range {} exceeds row_count {}",
7575                max_rows,
7576                table.rows.len()
7577            )));
7578        }
7579        if max_rows == 0 {
7580            return Ok(FreezeReport {
7581                segment_id: u32::MAX,
7582                frozen_rows: 0,
7583                bytes_freed: 0,
7584                segment_bytes: Vec::new(),
7585            });
7586        }
7587
7588        // --- segment build phase: reads only --------------------
7589        // K-way merge of already-sorted slices. Each slice's rows
7590        // are ascending by pk_u64; we keep a per-slice cursor and
7591        // pull the next-smallest head until every cursor drains.
7592        let total_rows: usize = ordered.iter().map(|s| s.rows.len()).sum();
7593        if total_rows != max_rows {
7594            return Err(StorageError::Corrupt(format!(
7595                "commit_freeze_slices: total slice rows {total_rows} ≠ row_range coverage {max_rows}"
7596            )));
7597        }
7598        let mut cursors: Vec<usize> = alloc::vec![0; ordered.len()];
7599        let mut merged: Vec<(u64, Vec<u8>, IndexKey)> = Vec::with_capacity(total_rows);
7600        loop {
7601            // Pick the slice whose head row has the smallest key
7602            // and isn't yet exhausted.
7603            let mut pick: Option<usize> = None;
7604            for (i, c) in cursors.iter().enumerate() {
7605                let slice = &ordered[i];
7606                if *c >= slice.rows.len() {
7607                    continue;
7608                }
7609                match pick {
7610                    None => pick = Some(i),
7611                    Some(j) => {
7612                        if slice.rows[*c].0 < ordered[j].rows[cursors[j]].0 {
7613                            pick = Some(i);
7614                        }
7615                    }
7616                }
7617            }
7618            let Some(i) = pick else { break };
7619            let row = ordered[i].rows[cursors[i]].clone();
7620            cursors[i] += 1;
7621            merged.push(row);
7622        }
7623        // Reject duplicate PKs — same error as the single-threaded
7624        // path so callers get a uniform surface.
7625        for w in merged.windows(2) {
7626            if w[0].0 == w[1].0 {
7627                return Err(StorageError::Corrupt(format!(
7628                    "commit_freeze_slices: duplicate PK {} across slices",
7629                    w[0].0
7630                )));
7631            }
7632        }
7633        let post_swap_keys: Vec<IndexKey> = merged.iter().map(|(_, _, k)| k.clone()).collect();
7634        let seg_rows: Vec<(u64, Vec<u8>)> =
7635            merged.into_iter().map(|(k, body, _)| (k, body)).collect();
7636        let frozen_rows = seg_rows.len();
7637        let (seg_bytes, _meta) = encode_segment(seg_rows.into_iter(), 0.01, SEGMENT_PAGE_BYTES)
7638            .map_err(|e| StorageError::Corrupt(format!("commit_freeze_slices: encode: {e}")))?;
7639
7640        // --- atomic swap phase: mutations only past this point ---
7641        let bytes_before = self.get(table_name).expect("just validated").hot_bytes();
7642        let positions: Vec<usize> = (0..max_rows).collect();
7643        let t_mut = self
7644            .get_mut(table_name)
7645            .expect("just validated; still present");
7646        let removed = t_mut.delete_rows(&positions);
7647        debug_assert_eq!(removed, max_rows, "delete_rows count matches request");
7648        let bytes_after = t_mut.hot_bytes();
7649        let bytes_freed = bytes_before.saturating_sub(bytes_after);
7650
7651        let segment_id = self
7652            .load_segment_bytes(seg_bytes.clone())
7653            .map_err(|e| StorageError::Corrupt(format!("commit_freeze_slices: load: {e}")))?;
7654        let new_cold = post_swap_keys.into_iter().map(|k| {
7655            (
7656                k,
7657                RowLocator::Cold {
7658                    segment_id,
7659                    page_offset: 0,
7660                },
7661            )
7662        });
7663        let t_mut = self.get_mut(table_name).expect("still present");
7664        t_mut.register_cold_locators(index_name, new_cold)?;
7665        // r944 — a freeze has to say that it froze something.
7666        //
7667        // `has_cold_rows_fast()` reads the cached count, and neither
7668        // freeze path touched it, so afterwards it answered "no cold
7669        // rows" while cold rows existed. That predicate gates four join
7670        // paths, and a gate that wrongly declines the cold-aware path
7671        // drops the frozen rows from the answer.
7672        //
7673        // Marking it stale rather than adding to it: stale reads as
7674        // true, which is the safe direction, and this function cannot
7675        // know the exact total (rows may already have been cold). ANALYZE
7676        // recomputes the number.
7677        t_mut.mark_cold_row_count_stale();
7678
7679        Ok(FreezeReport {
7680            segment_id,
7681            frozen_rows,
7682            bytes_freed,
7683            segment_bytes: seg_bytes,
7684        })
7685    }
7686
7687    /// v6.7.3 — compact every cold segment on `(table, index)` whose
7688    /// `OwnedSegment::bytes().len()` is below `target_segment_bytes`
7689    /// into a single larger merged segment. Rows present in source
7690    /// segment payloads but no longer referenced by any
7691    /// `RowLocator::Cold` on the index (DELETE'd + frozen rows
7692    /// retired via [`Catalog::shadow_cold_row`]) are GC'd in the
7693    /// merge.
7694    ///
7695    /// **Semantics**:
7696    /// 1. Walk the BTree index to collect every Cold locator that
7697    ///    targets a small (< threshold) segment. Each such
7698    ///    `(key, segment_id)` becomes a row in the merged segment;
7699    ///    payload is looked up from the source segment in-place.
7700    /// 2. Encode the collected rows into one new segment via
7701    ///    [`encode_segment`]; register it via
7702    ///    [`Catalog::load_segment_bytes`] (allocating a fresh
7703    ///    `merged_segment_id` at the end of `cold_segments`).
7704    /// 3. Rewrite the BTree index in one pass: every
7705    ///    `RowLocator::Cold { segment_id ∈ sources }` becomes
7706    ///    `RowLocator::Cold { segment_id = merged_id, page_offset = 0 }`.
7707    ///    Hot locators are untouched.
7708    /// 4. Tombstone every source slot via
7709    ///    [`Catalog::tombstone_segment`]. Source segment payloads
7710    ///    are no longer reachable through the catalog; the on-disk
7711    ///    files are the caller's concern.
7712    ///
7713    /// On fewer than 2 candidate segments the catalog is **not**
7714    /// mutated and a no-op report (`merged_segment_id: None`,
7715    /// `sources: []`) is returned. This is the routine case — a
7716    /// freshly-frozen table has at most 1 small segment, no merge
7717    /// possible.
7718    ///
7719    /// Atomicity: every mutating step runs after the read-only
7720    /// gather phase, so a panic before the merge encode leaves the
7721    /// catalog unchanged. The mutation block itself (load + rewrite +
7722    /// tombstone) takes only `&mut self` — callers serialise the
7723    /// engine write lock outside this function.
7724    ///
7725    /// Errors when the table / index doesn't exist, the index isn't
7726    /// `BTree`, the index column type isn't u64-coercible (cold-tier
7727    /// pre-condition), or a source segment fails its in-place
7728    /// row-body lookup (would indicate prior catalog corruption).
7729    pub fn compact_cold_segments(
7730        &mut self,
7731        table_name: &str,
7732        index_name: &str,
7733        target_segment_bytes: u64,
7734    ) -> Result<CompactReport, StorageError> {
7735        // --- validation phase ----------------------------------
7736        let t = self.get(table_name).ok_or_else(|| {
7737            StorageError::Corrupt(format!(
7738                "compact_cold_segments: table {table_name:?} not found"
7739            ))
7740        })?;
7741        let idx = t
7742            .indices
7743            .iter()
7744            .find(|i| i.name == index_name)
7745            .ok_or_else(|| {
7746                StorageError::Corrupt(format!(
7747                    "compact_cold_segments: index {index_name:?} not found on {table_name:?}"
7748                ))
7749            })?;
7750        let map = match &idx.kind {
7751            IndexKind::BTree(m) => m,
7752            IndexKind::Nsw(_)
7753            | IndexKind::Brin { .. }
7754            | IndexKind::Gin(_)
7755            | IndexKind::GinTrgm(_)
7756            | IndexKind::GinFulltext(_)
7757            | IndexKind::GinJsonb(_) => {
7758                return Err(StorageError::Corrupt(format!(
7759                    "compact_cold_segments: index {index_name:?} is not BTree; \
7760                     compaction applies only to BTree cold-tier indices"
7761                )));
7762            }
7763        };
7764
7765        // --- gather phase --------------------------------------
7766        // Step A: every segment_id this BTree index Cold-references.
7767        let mut referenced_ids: BTreeSet<u32> = BTreeSet::new();
7768        for (_key, locators) in map.iter() {
7769            for loc in locators {
7770                if let RowLocator::Cold { segment_id, .. } = loc {
7771                    referenced_ids.insert(*segment_id);
7772                }
7773            }
7774        }
7775        // Step B: keep only the small + still-active ones.
7776        let candidate_set: BTreeSet<u32> = referenced_ids
7777            .into_iter()
7778            .filter(|id| {
7779                self.cold_segments
7780                    .get(*id as usize)
7781                    .and_then(|s| s.as_deref())
7782                    .is_some_and(|s| (s.bytes().len() as u64) < target_segment_bytes)
7783            })
7784            .collect();
7785        if candidate_set.len() < 2 {
7786            return Ok(CompactReport {
7787                sources: Vec::new(),
7788                merged_segment_id: None,
7789                merged_segment_bytes: Vec::new(),
7790                merged_rows: 0,
7791                deleted_rows_pruned: 0,
7792                bytes_reclaimed_estimate: 0,
7793            });
7794        }
7795        // Step C: pre-count source rows for the deleted-pruned metric.
7796        let mut source_row_count: usize = 0;
7797        let mut source_byte_total: u64 = 0;
7798        for &id in &candidate_set {
7799            let seg = self.cold_segments[id as usize]
7800                .as_ref()
7801                .expect("candidate selected only when slot is Some");
7802            source_row_count = source_row_count.saturating_add(seg.meta().num_rows as usize);
7803            source_byte_total = source_byte_total.saturating_add(seg.bytes().len() as u64);
7804        }
7805        // Step D: collect (key, body) pairs from every live Cold
7806        // locator pointing at a candidate. dedupe by key — one
7807        // BTree key resolves to at most one cold payload (the
7808        // freezer + promote/shadow flow keeps Cold locators
7809        // unique per key).
7810        let mut collected: BTreeMap<u64, (Vec<u8>, IndexKey)> = BTreeMap::new();
7811        for (key, locators) in map.iter() {
7812            for loc in locators {
7813                let RowLocator::Cold { segment_id, .. } = loc else {
7814                    continue;
7815                };
7816                if !candidate_set.contains(segment_id) {
7817                    continue;
7818                }
7819                let u64_key = index_key_as_u64(key).ok_or_else(|| {
7820                    StorageError::Corrupt(format!(
7821                        "compact_cold_segments: index {index_name:?} has non-integer Cold key; \
7822                         cold tier requires IndexKey::Int (Text PK lands in v5.5+)"
7823                    ))
7824                })?;
7825                let seg = self.cold_segments[*segment_id as usize]
7826                    .as_ref()
7827                    .expect("candidate slot guaranteed Some above");
7828                let payload = seg.lookup(u64_key).ok_or_else(|| {
7829                    StorageError::Corrupt(format!(
7830                        "compact_cold_segments: BTree {index_name:?} points key={u64_key} \
7831                         at segment {segment_id} but the segment lookup missed"
7832                    ))
7833                })?;
7834                collected.insert(u64_key, (payload, key.clone()));
7835                break;
7836            }
7837        }
7838        let merged_rows = collected.len();
7839        let deleted_rows_pruned = source_row_count.saturating_sub(merged_rows);
7840
7841        // Step E: encode the merged segment. `BTreeMap<u64, _>`
7842        // iteration is ascending by key, which is what
7843        // `encode_segment` requires.
7844        let seg_rows: Vec<(u64, Vec<u8>)> = collected
7845            .iter()
7846            .map(|(k, (body, _))| (*k, body.clone()))
7847            .collect();
7848        let (seg_bytes, _meta) = encode_segment(seg_rows.into_iter(), 0.01, SEGMENT_PAGE_BYTES)
7849            .map_err(|e| StorageError::Corrupt(format!("compact_cold_segments: encode: {e}")))?;
7850        let merged_bytes_len = seg_bytes.len() as u64;
7851
7852        // --- atomic mutation phase ------------------------------
7853        let merged_segment_id = self
7854            .load_segment_bytes(seg_bytes.clone())
7855            .map_err(|e| StorageError::Corrupt(format!("compact_cold_segments: load: {e}")))?;
7856
7857        // Rewrite the BTree index: every Cold locator pointing at
7858        // a candidate source becomes a Cold locator pointing at
7859        // the merged segment. Use a flat collect-then-replace
7860        // pattern so we never hold a `&self` borrow across the
7861        // `&mut self` write.
7862        let entries: Vec<(IndexKey, crate::posting::PostingList)> = {
7863            let t = self
7864                .get(table_name)
7865                .expect("table existed at the start of this fn");
7866            let idx = t
7867                .indices
7868                .iter()
7869                .find(|i| i.name == index_name)
7870                .expect("index existed at the start of this fn");
7871            let IndexKind::BTree(map) = &idx.kind else {
7872                unreachable!("validated above");
7873            };
7874            map.iter().map(|(k, v)| (k.clone(), v.clone())).collect()
7875        };
7876        let t_mut = self
7877            .get_mut(table_name)
7878            .expect("table existed at the start of this fn");
7879        let idx_mut = t_mut
7880            .indices
7881            .iter_mut()
7882            .find(|i| i.name == index_name)
7883            .expect("index existed at the start of this fn");
7884        let IndexKind::BTree(map_mut) = &mut idx_mut.kind else {
7885            unreachable!("validated above");
7886        };
7887        for (key, locators) in entries {
7888            let mut new_locs = crate::posting::PostingList::new();
7889            let mut changed = false;
7890            for loc in &locators {
7891                match *loc {
7892                    RowLocator::Cold {
7893                        segment_id,
7894                        page_offset: _,
7895                    } if candidate_set.contains(&segment_id) => {
7896                        let replacement = RowLocator::Cold {
7897                            segment_id: merged_segment_id,
7898                            page_offset: 0,
7899                        };
7900                        if !new_locs.contains(replacement) {
7901                            new_locs.push(replacement);
7902                        }
7903                        changed = true;
7904                    }
7905                    other => new_locs.push(other),
7906                }
7907            }
7908            if changed {
7909                map_mut.insert_mut(key, new_locs);
7910            }
7911        }
7912
7913        // Tombstone every source slot. Last step — failures here
7914        // would leave the segment double-referenced in both
7915        // memory + manifest, but `tombstone_segment` only errors
7916        // on out-of-bounds, which we've already validated.
7917        for &id in &candidate_set {
7918            self.tombstone_segment(id)?;
7919        }
7920
7921        let bytes_reclaimed_estimate = source_byte_total.saturating_sub(merged_bytes_len);
7922        Ok(CompactReport {
7923            sources: candidate_set.into_iter().collect(),
7924            merged_segment_id: Some(merged_segment_id),
7925            merged_segment_bytes: seg_bytes,
7926            merged_rows,
7927            deleted_rows_pruned,
7928            bytes_reclaimed_estimate,
7929        })
7930    }
7931
7932    /// Internal helper: scan `(table, index)` for a `Cold` locator
7933    /// keyed by `key`. Returns `Ok(Some((segment_id, page_offset)))`
7934    /// when found, `Ok(None)` when the key has only hot entries
7935    /// or no entries at all, `Err` on the same input-validation
7936    /// errors as the public `promote_cold_row` / `shadow_cold_row`.
7937    fn find_cold_locator(
7938        &self,
7939        table_name: &str,
7940        index_name: &str,
7941        key: &IndexKey,
7942    ) -> Result<Option<(u32, u32)>, StorageError> {
7943        let t = self.get(table_name).ok_or_else(|| {
7944            StorageError::Corrupt(format!("find_cold_locator: table {table_name:?} not found"))
7945        })?;
7946        let idx = t
7947            .indices
7948            .iter()
7949            .find(|i| i.name == index_name)
7950            .ok_or_else(|| {
7951                StorageError::Corrupt(format!(
7952                    "find_cold_locator: index {index_name:?} not found on {table_name:?}"
7953                ))
7954            })?;
7955        if !matches!(idx.kind, IndexKind::BTree(_)) {
7956            return Err(StorageError::Corrupt(format!(
7957                "find_cold_locator: index {index_name:?} is NSW; promote-on-write only applies to BTree indices"
7958            )));
7959        }
7960        for loc in idx.lookup_eq(key) {
7961            if let RowLocator::Cold {
7962                segment_id,
7963                page_offset,
7964            } = *loc
7965            {
7966                return Ok(Some((segment_id, page_offset)));
7967            }
7968        }
7969        Ok(None)
7970    }
7971}
7972
7973/// Coerce an [`IndexKey`] to the `u64` that v5.1 cold-tier
7974/// segments use as their on-disk PK. Returns `None` for keys that
7975/// aren't representable as `u64` — Text PKs need a hash mapping
7976/// the segment writer baked in (deferred to v5.2+), Bool PKs are
7977/// almost never wide enough to be sharded into a cold tier.
7978fn index_key_as_u64(key: &IndexKey) -> Option<u64> {
7979    match key {
7980        // Reinterpret the i64 bit pattern as u64. Cold-tier segments
7981        // are sorted by this u64 view, so the chosen interpretation
7982        // only has to match between insert (bake_segment / freezer)
7983        // and lookup — using cast_unsigned keeps both sides honest
7984        // and silences clippy::cast_sign_loss.
7985        IndexKey::Int(n) => Some(n.cast_unsigned()),
7986        // Text / Bool / Uuid / Bytes / Numeric PKs aren't representable
7987        // as u64 and so can't participate in the u64-sorted cold-tier
7988        // segment PK layout. Same deferral story as Text — lookup falls
7989        // through the in-memory btree.
7990        IndexKey::Text(_)
7991        | IndexKey::Bool(_)
7992        | IndexKey::Uuid(_)
7993        | IndexKey::Bytes(_)
7994        | IndexKey::Numeric(_) => None,
7995    }
7996}
7997
7998#[derive(Debug, Clone, PartialEq, Eq)]
7999#[non_exhaustive]
8000pub enum StorageError {
8001    DuplicateTable {
8002        name: String,
8003    },
8004    TableNotFound {
8005        name: String,
8006    },
8007    ArityMismatch {
8008        expected: usize,
8009        actual: usize,
8010    },
8011    TypeMismatch {
8012        column: String,
8013        expected: DataType,
8014        actual: DataType,
8015        position: usize,
8016    },
8017    NullInNotNull {
8018        column: String,
8019    },
8020    /// Index with this name already exists on the table.
8021    DuplicateIndex {
8022        name: String,
8023    },
8024    /// Column referenced by an index doesn't exist on the table.
8025    ColumnNotFound {
8026        column: String,
8027    },
8028    /// On-disk format failed to parse — corrupted file, wrong magic, truncated
8029    /// payload, or unknown tag bytes.
8030    Corrupt(String),
8031    /// v6.0.4 — ALTER INDEX targeted an index name that doesn't
8032    /// exist on any table in this catalog.
8033    IndexNotFound {
8034        name: String,
8035    },
8036    /// v6.0.4 — operation requested isn't supported on this index
8037    /// kind / column type (e.g. ALTER INDEX REBUILD on a `BTree`
8038    /// index, or REBUILD WITH (encoding=…) on a non-vector column).
8039    Unsupported(String),
8040    /// v7.39 (round 220) — a CYCLE-less sequence ran past its bound.
8041    /// PG's 2200H phrasing: `nextval: reached maximum value of
8042    /// sequence "s" (n)` (`is_max: false` = the MINVALUE direction).
8043    SequenceExhausted {
8044        name: String,
8045        limit: i64,
8046        is_max: bool,
8047    },
8048}
8049
8050impl fmt::Display for StorageError {
8051    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
8052        match self {
8053            // v7.39 (read01 round 47) — PG's 42P07 wording.
8054            Self::DuplicateTable { name } => write!(f, "relation \"{name}\" already exists"),
8055            // v7.39 (read01 round 47) — PG's wording for a missing relation
8056            // (42P01). DROP TABLE says "table" and raises its own error at
8057            // the engine; every other path (SELECT / ALTER / …) says
8058            // "relation", which is what this carries.
8059            Self::TableNotFound { name } => write!(f, "relation \"{name}\" does not exist"),
8060            Self::ArityMismatch { expected, actual } => write!(
8061                f,
8062                "row arity mismatch: expected {expected} columns, got {actual}"
8063            ),
8064            Self::TypeMismatch {
8065                column,
8066                expected,
8067                actual,
8068                position,
8069            } => write!(
8070                f,
8071                "type mismatch in column {column:?} (position {position}): expected {expected}, got {actual}"
8072            ),
8073            Self::NullInNotNull { column } => {
8074                // v7.39 (SQLSTATE fidelity) — PG's 23502 phrasing (the
8075                // relation-qualified long form is added by engine call
8076                // sites that know the table name).
8077                write!(
8078                    f,
8079                    "null value in column \"{column}\" violates not-null constraint"
8080                )
8081            }
8082            // v7.39 (read01 round 47) — an index is a relation to PG (42P07).
8083            Self::DuplicateIndex { name } => write!(f, "relation \"{name}\" already exists"),
8084            // v7.39 (round 701) — PG's wording, and the same fix `EvalError::
8085            // ColumnNotFound` took in read01 round 81 with the same reason:
8086            // "column not found: x" matches none of the wire layer's `does
8087            // not exist` patterns, so a missing column reached the client as
8088            // the generic error class. The eval-side variant was changed and
8089            // the storage-side one was not, so which sentence you got
8090            // depended on which layer noticed — `CREATE INDEX ix ON t(nope)`
8091            // came out of storage and kept the old spelling.
8092            Self::ColumnNotFound { column } => write!(f, "column \"{column}\" does not exist"),
8093            Self::Corrupt(detail) => write!(f, "corrupt on-disk format: {detail}"),
8094            Self::IndexNotFound { name } => write!(f, "index \"{name}\" does not exist"),
8095            Self::Unsupported(detail) => write!(f, "unsupported: {detail}"),
8096            // v7.39 (round 220) — PG's exact 2200H wording.
8097            Self::SequenceExhausted {
8098                name,
8099                limit,
8100                is_max,
8101            } => write!(
8102                f,
8103                "nextval: reached {} value of sequence \"{name}\" ({limit})",
8104                if *is_max { "maximum" } else { "minimum" }
8105            ),
8106        }
8107    }
8108}
8109
8110impl ColumnSchema {
8111    pub fn new(name: impl Into<String>, ty: DataType, nullable: bool) -> Self {
8112        Self {
8113            name: name.into(),
8114            ty,
8115            nullable,
8116            collation_name: None,
8117            default: None,
8118            runtime_default: None,
8119            auto_increment: false,
8120            user_enum_type: None,
8121            user_domain_type: None,
8122            user_composite_type: None,
8123            acl: Vec::new(),
8124            on_update_runtime: None,
8125            collation: Collation::Binary,
8126            is_unsigned: false,
8127            inline_enum_variants: None,
8128            inline_set_variants: None,
8129            generated_stored_expr: None,
8130            identity_always: false,
8131            default_text: None,
8132            auto_restart: None,
8133            scalar_row_source: false,
8134            mysql_int_width: None,
8135            mysql_fsp: None,
8136        }
8137    }
8138
8139    /// Builder-style helper to attach a default value to an otherwise
8140    /// plain column schema. Used by the engine when CREATE TABLE
8141    /// specifies `column TYPE DEFAULT <expr>`.
8142    #[must_use]
8143    pub fn with_default(mut self, default: Value<'static>) -> Self {
8144        self.default = Some(default);
8145        self
8146    }
8147
8148    /// v7.9.21 — builder for runtime-evaluated defaults
8149    /// (`DEFAULT now()`, `DEFAULT CURRENT_TIMESTAMP`, …).
8150    /// `expr` is the Expr's `Display` form, re-parsed by the
8151    /// engine at each INSERT.
8152    #[must_use]
8153    pub fn with_runtime_default(mut self, expr: impl Into<String>) -> Self {
8154        self.runtime_default = Some(expr.into());
8155        self
8156    }
8157
8158    /// Builder-style helper to mark a column as `AUTO_INCREMENT`.
8159    #[must_use]
8160    pub const fn with_auto_increment(mut self) -> Self {
8161        self.auto_increment = true;
8162        self
8163    }
8164}
8165
8166impl TableSchema {
8167    pub fn new(name: impl Into<String>, columns: Vec<ColumnSchema>) -> Self {
8168        Self {
8169            name: name.into(),
8170            columns,
8171            hot_tier_bytes: None,
8172            foreign_keys: Vec::new(),
8173            uniqueness_constraints: Vec::new(),
8174            exclusion_constraints: Vec::new(),
8175            checks: Vec::new(),
8176            partition_role: None,
8177            policies: Vec::new(),
8178            row_security: false,
8179            force_row_security: false,
8180            owner: None,
8181            acl: Vec::new(),
8182        }
8183    }
8184}
8185
8186// =========================================================================
8187// Persistent binary format for the catalog.
8188//
8189// Layout (little-endian throughout):
8190//
8191//   [magic "SPGDB001" 8 bytes][version u8]
8192//   [table_count u32]
8193//   for each table:
8194//       [name_len u16][name bytes]
8195//       [col_count u16]
8196//       for each col:
8197//           [name_len u16][name bytes]
8198//           [type_tag u8 + optional payload]
8199//               1=Int 2=BigInt 3=Float 4=Text 5=Bool
8200//               6=Vector(u32 dim)
8201//               7=SmallInt
8202//               8=Varchar(u32 max)
8203//               9=Char(u32 size)
8204//               10=Numeric(u8 precision, u8 scale)
8205//               11=Date
8206//               12=Timestamp
8207//           [nullable u8]   0/1
8208//           [default_tag u8] 0=none 1=value (followed by [value_tag u8] + bytes)
8209//       [row_count u32]
8210//       for each row, for each col, one [value_tag u8] + value bytes:
8211//           tag 0 (Null)     → no body
8212//           tag 1 (Int)      → i32 LE
8213//           tag 2 (BigInt)   → i64 LE
8214//           tag 3 (Float)    → f64 LE
8215//           tag 4 (Text)     → u16 LE len + UTF-8 bytes
8216//           tag 5 (Bool)     → u8 0/1
8217//           tag 6 (Vector)   → u32 LE dim + dim×f32 LE
8218//           tag 7 (SmallInt) → i16 LE
8219//           tag 8 (Numeric)  → i128 LE (16 bytes) + u8 scale
8220//           tag 9 (Date)     → i32 LE (days since Unix epoch)
8221//           tag 10 (Timestamp) → i64 LE (microseconds since Unix epoch)
8222//
8223// Bumped to version 3 when NUMERIC was added; to version 4 when
8224// AUTO_INCREMENT (per-column flag) + NSW index `kind` byte landed;
8225// to version 5 when DATE / TIMESTAMP were added; to version 6 when
8226// NSW graph topology started travelling on disk (v2.7); to version 7
8227// when the NSW topology became multi-layer HNSW (v2.13); to version 8
8228// when row encoding switched to schema-driven dense layout (v3.0.2 —
8229// per-row NULL bitmap + per-column fixed-width body, no per-cell type
8230// tag).
8231// =========================================================================
8232
8233const FILE_MAGIC: &[u8; 8] = b"SPGDB001";
8234/// Current catalog snapshot format version emitted by [`Catalog::serialize`].
8235///
8236/// v9 (v5.2) extends v8 by serialising `BTree` index entries directly — every
8237/// `(IndexKey, Vec<RowLocator>)` pair travels on disk with the v5.1
8238/// `RowLocator::write_le` tag-prefixed codec. v8 `BTree` indices stored no
8239/// entries at all (the map was rebuilt from `Table::rows` on load); v9
8240/// preserves on-disk Cold locators so freezer-produced cold-tier index
8241/// entries survive a catalog snapshot round-trip. v8 readers are accepted
8242/// by version dispatch in [`Catalog::deserialize`] — every entry decodes
8243/// as `RowLocator::Hot(_)` via `add_index` rebuild, identical to v5.1
8244/// behaviour.
8245/// v6.7.2 — bumped from 10 to 11 to append per-table
8246/// `hot_tier_bytes: Option<u64>` after the per-table indices
8247/// section. v10 catalogs (v6.7.1) load with `hot_tier_bytes =
8248/// None` for every table (the deserialiser short-circuits when
8249/// version < 11). v11 snapshots written by a pre-v6.7.2 binary
8250/// fail loudly at the version check, matching the v6.1.2 /
8251/// v6.1.4 / v6.2.0 / v6.7.1 envelope-bump upgrade fences.
8252///
8253/// v6.8.0 — bumped from 11 to 12: per-index
8254/// `included_columns: Vec<u16>` appended at the tail of each
8255/// index payload. v11 (= v6.7.2) catalogs load with
8256/// `included_columns = Vec::new()` for every index — same
8257/// "older readers, append-only extension" pattern as the v6.7.2
8258/// hot_tier_bytes byte.
8259/// v7.13.0 — bumped from 22 to 23. mailrs round-5 G3 / G10.
8260/// Per-table appendix gains two new sections:
8261///   * `checks: Vec<String>` — CHECK predicate sources (Display
8262///     form of the AST Expr); re-parsed on INSERT/UPDATE to
8263///     enforce against candidate rows. Same persistence pattern
8264///     as `Index::partial_predicate`.
8265///   * Per `UniquenessConstraint`: trailing `nulls_not_distinct:
8266///     u8` flag for PG 15+ `UNIQUE NULLS NOT DISTINCT (cols)`
8267///     semantics.
8268/// v22 catalogs deserialise with empty `checks` and every UC
8269/// at `nulls_not_distinct = false`.
8270/// v24 introduces:
8271///   * Index kind tag 4 = trigram-GIN (`gin_trgm_ops`-flavoured
8272///     `USING gin` over a TEXT/VARCHAR column). Payload shape is
8273///     identical to tag-3 GIN (String → Vec<RowLocator>); the
8274///     keys are PG-compatible 3-byte trigram shingles instead of
8275///     tsvector lexemes. v23 catalogs deserialise unchanged — no
8276///     v23 writer ever emitted tag 4.
8277/// v25 introduces:
8278///   * Per `TriggerDef`: trailing `enabled: u8` flag (mailrs
8279///     round-9 A.2.b — `ALTER TABLE … { ENABLE | DISABLE }
8280///     TRIGGER …`). v24 catalogs deserialise with every trigger
8281///     `enabled = true`, matching pre-v7.16.1 behaviour.
8282/// v26 introduces (v7.17.0 Phase 1.1):
8283///   * Trailing SEQUENCE catalog block after triggers. Encoded
8284///     as `u32 count` followed by per-sequence:
8285///     `name`, `data_type: u8` (0=SmallInt,1=Int,2=BigInt),
8286///     `start i64`, `increment i64`, `min_value i64`,
8287///     `max_value i64`, `cache i64`, `cycle u8`,
8288///     `owned_by_tag u8` (0=NONE, 1=Column → `table`,`column`),
8289///     `last_value i64`, `is_called u8`. v25-and-below catalogs
8290///     deserialise with an empty sequences map.
8291/// v27 introduces (v7.17.0 Phase 1.2):
8292///   * Trailing VIEW catalog block after sequences. Encoded as
8293///     `u32 count` followed by per-view:
8294///     `name`, `column_count u16`, then column names, then
8295///     `body` long-string. v26-and-below catalogs deserialise
8296///     with an empty views map.
8297/// v28 introduces (v7.17.0 Phase 1.3):
8298///   * Trailing MATERIALIZED VIEW source registry block after
8299///     views. Encoded as `u32 count` followed by per-entry:
8300///     `name`, `body` long-string. The materialised rows live
8301///     as a regular Table of the same name (already covered by
8302///     the pre-existing tables block). v27-and-below catalogs
8303///     deserialise with an empty map.
8304/// v29 introduces (v7.17.0 Phase 1.4):
8305///   * Per-table user_enum_type appendix (after the CHECK
8306///     appendix). Layout: `u16 count` followed by per-binding
8307///     `[u16 col_pos][str enum_name]`. Only columns whose
8308///     `user_enum_type` is Some land here; the catalog stays
8309///     compact for the common no-enum case.
8310///   * Trailing ENUM types catalog block after materialized
8311///     views. Encoded as `u32 count` followed by per-entry:
8312///     `name`, `u16 label_count`, then `label_count` short
8313///     strings. v28-and-below catalogs deserialise with an
8314///     empty enum_types map and every column's
8315///     `user_enum_type = None`.
8316/// v30 introduces (v7.17.0 Phase 1.5):
8317///   * Per-table user_domain_type appendix (after the
8318///     user_enum_type appendix). Same shape as the enum one.
8319///   * Trailing DOMAIN types catalog block after the enum
8320///     block. Encoded as `u32 count` followed by per-entry:
8321///     `name`, `data_type` byte, `nullable u8`,
8322///     `default_present u8` + optional default string,
8323///     `u16 check_count` then `check_count` Display-form
8324///     CHECK strings. v29-and-below catalogs deserialise with
8325///     an empty domain_types map and `user_domain_type = None`.
8326/// v31 introduces (v7.17.0 Phase 1.6):
8327///   * Trailing user-schemas block after the DOMAIN block.
8328///     Encoded as `u32 count` followed by `count` schema-name
8329///     short strings. Built-in schemas (`public`, `pg_catalog`,
8330///     `information_schema`) are NOT serialised — they're
8331///     hardcoded in `is_builtin_schema`. v30-and-below catalogs
8332///     deserialise with an empty user-schemas set.
8333/// v32 introduces (v7.17.0 Phase 2.1):
8334///   * Per-table on_update_runtime appendix (after the
8335///     user_domain_type appendix). Layout: `u16 count` followed
8336///     by per-binding `[u16 col_pos][str expr_src]`. Only
8337///     columns whose `on_update_runtime` is Some land here;
8338///     the catalog stays compact when no MySQL-shaped table
8339///     uses the attribute. v31-and-below catalogs deserialise
8340///     with every column's `on_update_runtime = None`.
8341/// v33 introduces (v7.17.0 Phase 2.2):
8342///   * Index kind tag 5 = fulltext-GIN (MySQL `FULLTEXT KEY`
8343///     surface over a TEXT / VARCHAR column). Payload shape is
8344///     identical to tag-3 / tag-4 GIN (`String → Vec<RowLocator>`);
8345///     the keys are lower-cased word lexemes (same rule as
8346///     `to_tsvector('simple', text)`). v32 catalogs deserialise
8347///     unchanged — no v32 writer ever emitted tag 5, and FULLTEXT
8348///     KEY was silently dropped pre-v7.17 so no rebuild shim is
8349///     needed for round-tripped catalogs.
8350/// v34 introduces (v7.17.0 Phase 2.5):
8351///   * Per-table collation appendix (after the on_update_runtime
8352///     appendix). Sparse layout: only columns whose `collation`
8353///     is non-Binary land here. `u16 count` then per-binding
8354///     `[u16 col_pos][u8 collation_tag]` where the tag matches
8355///     `Collation::TAG_*`. Snapshots written by v33-and-below
8356///     readers deserialise every column with `collation =
8357///     Binary`, preserving the prior byte-wise compare
8358///     semantics. Unknown tags read back as Binary too — keeps
8359///     a forward-compat path if a future v35 adds variants
8360///     and someone rolls back to a v34 reader.
8361/// v35 introduces (v7.17.0 Phase 4.4):
8362///   * Per-table is_unsigned appendix (after the collation
8363///     appendix). Sparse layout: only `is_unsigned = true`
8364///     columns land. `u16 count` then per-binding `[u16 col_pos]`.
8365///     v34-and-below catalogs deserialise every column as
8366///     `is_unsigned = false`, preserving the prior silent-
8367///     accept behaviour for negative inserts on UNSIGNED columns.
8368/// v46 introduces (v7.23, mailrs round-14):
8369///   * Escaped short-string codec — `write_str` lengths >= 0xFFFF
8370///     emit `[u16 0xFFFF][u32 real_len]` so TEXT cells (mail bodies,
8371///     document text) above 64 KiB encode instead of panicking.
8372///     One-way upgrade: v45-and-below readers reject v46 catalogs
8373///     loudly via the version gate; v46 readers decode v45 catalogs
8374///     with the plain-u16 rules (0xFFFF is a legitimate length
8375///     there).
8376/// v47 introduces (v7.27, mailrs round-21):
8377///   * Escaped lengths for the REMAINING u16-length cell payloads —
8378///     BYTEA cells, TEXT[] elements, tsvector lexemes and tsquery
8379///     terms — the same `[u16 0xFFFF][u32 real_len]` escape v46
8380///     gave short strings. Round-14 fixed TEXT and missed these;
8381///     round-21 fired the BYTEA twin during a production migration.
8382///     One-way upgrade, same posture as v46.
8383/// v48 introduces (v7.37.5 β-P2, sentori cutover window):
8384///   * `INTERVAL` becomes a real column type. Catalog tag 34 in
8385///     `write_data_type`; per-row body is a fixed 16 bytes
8386///     (i64 micros + i32 days + i32 months, LE, PG-byte-equal
8387///     field order). The runtime-only days collapse is gone —
8388///     `'1 day'` and `'24 hours'` are stored distinctly. One-way
8389///     upgrade: v47 catalogs without INTERVAL columns deserialise
8390///     identically; v47 readers fed a v48 catalog that contains
8391///     INTERVAL hit the explicit "unknown data type tag: 34"
8392///     fence in `read_data_type`.
8393/// v49 introduces (v7.37.6-B, sentori Epic 2 P0):
8394///   * Per-table partition role appendix(declarative
8395///     `PARTITION BY RANGE` parent / range child / DEFAULT
8396///     child)。Layout, written **after** the inline_set_variants
8397///     appendix and **before** the per-table block close:
8398///       `[u8 role_tag]`
8399///         0 = `None`(普通表,后向兼容默认)
8400///         1 = `Parent`:  `[u8 kind_tag (0=Range)]`
8401///                        `[u16 key_col_count]` `(× u16 col_pos)`
8402///                        `[u16 tmpl_count]` `(× str source)`
8403///         2 = `Range`:   `[str parent_name]` `[Bound]` `[Bound]`
8404///         3 = `Default`: `[str parent_name]`
8405///     `PartitionBound` codec:
8406///       `[u8 bound_tag]` 0=MinValue 1=MaxValue 2=TimestampTz(`[i64 LE micros]`)
8407///     v48-and-below readers stop after the inline_set_variants
8408///     block — they don't see this appendix and deserialise every
8409///     table with `partition_role = None`. v49 writers always emit
8410///     `[0]` for plain tables, so the encoding stays one-byte-cheap.
8411/// v50 introduces (v7.37.7, sentori Epic 3 P1):
8412///   * Per-table `generated_stored_expr` appendix(stored generated
8413///     columns — `GENERATED ALWAYS AS (<expr>) STORED`)。Layout,
8414///     written **after** the partition_role appendix and before
8415///     the per-table block close:
8416///       `[u16 binding_count]`
8417///       `binding_count × { [u16 col_pos][str expr_source] }`
8418///     Sparse — only generated columns land here, so plain-shape
8419///     catalogs stay byte-for-byte identical save for the new
8420///     u16 zero count. v49-and-below readers stop after the
8421///     partition_role appendix; v50 readers default every column
8422///     to `generated_stored_expr = None` when this block is absent.
8423/// v51 introduces (v7.37.8, sentori Epic 5 P2):
8424///   * Per-index tag byte 6 = `GinJsonb`(real posting-list GIN
8425///     over a JSONB column). Payload shape mirrors tag-3 / 4 / 5:
8426///     `[u32 posting_list_count]` then `(str token, u32 locator_count,
8427///     locators …)` per posting list. Same `write_str` /
8428///     `RowLocator::write_le` codec as the rest of the GIN family.
8429///     v50 catalogs never wrote tag 6(the same DDL loaded as a
8430///     BTree fallback); v51 readers see tag 6 explicitly and dispatch
8431///     into `IndexKind::GinJsonb`.
8432/// v52 introduces (v7.37.42-T2 ζ-B composite + domain metasystem):
8433///   * Trailing COMPOSITE-types catalog block after the
8434///     user-schemas block. Encoded as `u32 count` followed by
8435///     per-entry: `name`, `u16 field_count`, then `field_count`
8436///     `[str field_name][data_type]` pairs (`write_data_type` is
8437///     reused). v51-and-below catalogs deserialise with an empty
8438///     composite_types map; v52 readers tolerate v51 catalogs by
8439///     stopping at the schema block (no composite block present
8440///     ⇒ empty map). Composite types are referenced by columns
8441///     via `ColumnSchema.user_composite_type`, mirroring the
8442///     `user_enum_type` / `user_domain_type` pattern. The block
8443///     lands here (not as a per-table appendix) so dropping the
8444///     composite type registers globally and DROP TYPE can find it
8445///     without a table scan.
8446/// v53 introduces (v7.37.16 Epic W — cross-checkpoint tombstone
8447///   durability):
8448///   * Trailing per-table MVCC appendix carrying, for every row,
8449///     its `RowHeader` (`xmin:u64`, `xmax:u64`, `flags:u8`) and its
8450///     stable `RowId` (`u64`), followed by the relation's
8451///     `next_rowid:u64`. Layout per table (after the v50
8452///     generated_stored_expr block, before the table loop closes):
8453///       `[u32 row_count]` (== `Table::rows().len()`, cross-check)
8454///       per row in physical order:
8455///         `[u64 xmin][u64 xmax][u8 flags][u64 rowid]`
8456///       `[u64 next_rowid]`
8457///     v52-and-below catalogs never wrote this block; their reader
8458///     stops after the last per-table appendix and
8459///     `deserialize_rows` leaves every row `RowHeader::frozen()`
8460///     with dense 1..=N ids — the exact pre-v53 contract. A v53
8461///     reader instead reconstructs headers + ids VERBATIM, so a
8462///     tombstone-redo naming a row inserted before the last
8463///     checkpoint resolves by `RowId` across the base-snapshot
8464///     boundary (closing the coupling the Epic W WAL slices deferred
8465///     to this format bump). Because the reader routes on `version`,
8466///     the block is strictly backward-compatible: old images load
8467///     byte-for-byte as before. `SPG_MVCC_INPLACE` is unaffected —
8468///     a gate-off database's rows are all frozen/alive, so
8469///     persisting + restoring their headers is observationally a
8470///     no-op.
8471/// v7.38 (read01 P5.05) — v54 appends a CRC32C over the whole preceding
8472/// image so a corrupted `base.spg` is caught on load instead of silently
8473/// deserialising garbage. Older images (v8..=53) carry no trailer and load
8474/// unchanged.
8475/// v7.39 (round 210) — v72 appends a per-table EXCLUDE-constraint appendix
8476/// (sparse: only tables carrying an EXCLUDE write it) at the very end of the
8477/// per-table block, after the column-ACL appendix. A v71 reader stops before
8478/// it and its tables read back with no exclusion constraints, which is what
8479/// they were.
8480/// v7.39 (round 220) — v73 appends a per-table identity-RESTART appendix
8481/// (sparse: [u16 count] then per entry [u16 col_pos][i64 LE floor]) after
8482/// the EXCLUDE appendix. A v72 reader stops before it; its columns read
8483/// back with no RESTART floor, losing only an un-consumed
8484/// `ALTER … RESTART WITH` across a restart.
8485/// r1039 — v90 adds index-key tags 4 (bytea) and 5 (the canonical
8486/// numeric key), so BYTEA and NUMERIC columns carry a real B-tree
8487/// instead of falling back to a scan. A v89 reader meeting either tag
8488/// reports a corrupt catalog rather than mis-reading it, which is the
8489/// same forward-compatibility story tag 3 (uuid) had at v36.
8490const FILE_VERSION: u8 = 90;
8491
8492/// v7.37 (round 833) — the codec version to decode a row that
8493/// [`encode_row_body_dense`] has just produced.
8494///
8495/// That encoder always writes the newest form, and every decoder gate is
8496/// a `codec_version >= N` feature test, so a freshly encoded row must be
8497/// read at the current version. Cold segments carry their own version in
8498/// their header and keep passing that; this is for in-process round
8499/// trips — sort runs on temp storage — where the bytes never outlive the
8500/// build that wrote them.
8501pub const CURRENT_ROW_CODEC_VERSION: u8 = FILE_VERSION;
8502/// First version that appends the trailing CRC32C integrity trailer.
8503const FILE_VERSION_CRC_TRAILER: u8 = 54;
8504/// Oldest format version [`Catalog::deserialize`] still accepts. v8 is the
8505/// v3.0.2 dense-row layout; pre-v8 catalogs require an offline migration.
8506const MIN_SUPPORTED_FILE_VERSION: u8 = 8;
8507
8508// IndexKey wire format (v9):
8509//   tag 0 = Int  → [i64 LE]
8510//   tag 1 = Text → [u16 LE len + UTF-8 bytes] (via write_str / read_str)
8511//   tag 2 = Bool → [u8 0/1]
8512const INDEX_KEY_TAG_INT: u8 = 0;
8513const INDEX_KEY_TAG_TEXT: u8 = 1;
8514const INDEX_KEY_TAG_BOOL: u8 = 2;
8515/// v7.17.0 — `IndexKey::Uuid([u8; 16])`. Body = raw 16 bytes
8516/// (RFC 4122 byte order). Persisted only in FILE_VERSION 36+
8517/// catalogs.
8518const INDEX_KEY_TAG_UUID: u8 = 3;
8519/// r1039 — `IndexKey::Bytes`. Body = [u32 LE len][raw bytes].
8520/// Persisted only in FILE_VERSION 90+ catalogs.
8521const INDEX_KEY_TAG_BYTES: u8 = 4;
8522/// r1039 — `IndexKey::Numeric`. Body = [u8 class][u8 neg][i32 LE exp]
8523/// [u32 LE digit count][one byte per decimal digit, 0..=9, MSD first].
8524/// Persisted only in FILE_VERSION 90+ catalogs.
8525const INDEX_KEY_TAG_NUMERIC: u8 = 5;
8526
8527impl Catalog {
8528    /// Serialize the whole catalog (schema + every row) into a self-contained
8529    /// byte buffer. Format is documented above the impl block.
8530    pub fn serialize(&self) -> Vec<u8> {
8531        let mut out = Vec::with_capacity(64);
8532        out.extend_from_slice(FILE_MAGIC);
8533        out.push(FILE_VERSION);
8534        write_u32(
8535            &mut out,
8536            u32::try_from(self.tables.len()).expect("≤ 4G tables"),
8537        );
8538        for t in &self.tables {
8539            write_str(&mut out, &t.schema.name);
8540            write_u16(
8541                &mut out,
8542                u16::try_from(t.schema.columns.len()).expect("≤ 65k columns/table"),
8543            );
8544            for c in &t.schema.columns {
8545                write_str(&mut out, &c.name);
8546                write_data_type(&mut out, c.ty);
8547                out.push(u8::from(c.nullable));
8548                match &c.default {
8549                    None => out.push(0),
8550                    Some(v) => {
8551                        out.push(1);
8552                        write_value(&mut out, v);
8553                    }
8554                }
8555                out.push(u8::from(c.auto_increment));
8556            }
8557            write_u32(
8558                &mut out,
8559                u32::try_from(t.rows.len()).expect("≤ 4G rows/table"),
8560            );
8561            // v3.0.2 dense row encoding (FILE_VERSION 8): per-row NULL
8562            // bitmap, then tightly-packed bodies. Identical wire format
8563            // as before — extracted into `encode_row_body_dense` so cold-
8564            // tier segments (v5.1+) can share the encoding.
8565            for row in &t.rows {
8566                out.extend_from_slice(&encode_row_body_dense(row, &t.schema));
8567            }
8568            // Index definitions. Per-index payload:
8569            //   [name][col_pos u16][kind u8]
8570            //     kind 0 = B-tree           (no params — rebuilt on load)
8571            //     kind 1 = NSW graph        (u16 M + serialized graph)
8572            // For NSW the graph topology travels on disk so startup
8573            // doesn't re-run the O(n²M) rebuild — see v2.7 notes.
8574            write_u16(
8575                &mut out,
8576                u16::try_from(t.indices.len()).expect("≤ 65k indices/table"),
8577            );
8578            for idx in &t.indices {
8579                write_str(&mut out, &idx.name);
8580                write_u16(
8581                    &mut out,
8582                    u16::try_from(idx.column_position).expect("≤ 65k columns/table"),
8583                );
8584                match &idx.kind {
8585                    IndexKind::BTree(map) => {
8586                        out.push(0);
8587                        // v9: serialise the full PB map. Each entry's
8588                        // RowLocator list travels with the tag-prefixed
8589                        // codec from `row_locator::write_le`, so freezer-
8590                        // produced Cold locators survive a snapshot
8591                        // round-trip. v8 BTree wrote nothing here and
8592                        // rebuilt from rows — v9 readers tolerate v8 by
8593                        // version dispatch in `Catalog::deserialize`.
8594                        write_u32(
8595                            &mut out,
8596                            u32::try_from(map.len()).expect("≤ 4G index entries/index"),
8597                        );
8598                        for (key, locators) in map {
8599                            write_index_key(&mut out, key);
8600                            write_u32(
8601                                &mut out,
8602                                u32::try_from(locators.len()).expect("≤ 4G locators/key"),
8603                            );
8604                            for loc in locators {
8605                                loc.write_le(&mut out);
8606                            }
8607                        }
8608                    }
8609                    IndexKind::Nsw(g) => {
8610                        out.push(1);
8611                        write_u16(&mut out, u16::try_from(g.m).expect("≤ 65k NSW neighbours"));
8612                        write_nsw_graph(&mut out, g);
8613                    }
8614                    IndexKind::Brin { column_type } => {
8615                        // v6.7.1 — tag byte 2 = BRIN. Payload is the
8616                        // column type code (1 byte mapping to the
8617                        // shared DataType numeric encoding); no
8618                        // further data — BRIN summaries live in
8619                        // cold segments, not the catalog.
8620                        out.push(2);
8621                        write_data_type(&mut out, *column_type);
8622                    }
8623                    IndexKind::Gin(map) => {
8624                        // v7.12.3 — tag byte 3 = GIN. Payload mirrors
8625                        // the BTree encoding but with String (lexeme
8626                        // word) keys instead of IndexKey. Tag-prefixed
8627                        // RowLocator codec so freezer-produced Cold
8628                        // locators survive snapshot round-trip.
8629                        // FILE_VERSION 21+; v20 catalogs never wrote a
8630                        // GIN index (the AM degraded to BTree fallback
8631                        // pre-v7.12.3), so no migration shim is needed.
8632                        out.push(3);
8633                        write_u32(
8634                            &mut out,
8635                            u32::try_from(map.len()).expect("≤ 4G GIN posting lists"),
8636                        );
8637                        for (word, locators) in map {
8638                            write_str(&mut out, word);
8639                            write_u32(
8640                                &mut out,
8641                                u32::try_from(locators.len()).expect("≤ 4G locators/posting list"),
8642                            );
8643                            for loc in locators {
8644                                loc.write_le(&mut out);
8645                            }
8646                        }
8647                    }
8648                    IndexKind::GinTrgm(map) => {
8649                        // v7.15.0 — tag byte 4 = GinTrgm
8650                        // (`gin_trgm_ops` GIN over a TEXT column).
8651                        // Payload shape is identical to tag-3 GIN —
8652                        // `String → Vec<RowLocator>` posting lists.
8653                        // The String keys are 3-byte trigrams instead
8654                        // of tsvector lexemes; the deserializer
8655                        // dispatches on the tag, not the key shape.
8656                        // FILE_VERSION 24+; v23 catalogs never wrote
8657                        // a trigram-GIN.
8658                        out.push(4);
8659                        write_u32(
8660                            &mut out,
8661                            u32::try_from(map.len()).expect("≤ 4G trigram-GIN posting lists"),
8662                        );
8663                        for (tri, locators) in map {
8664                            write_str(&mut out, tri);
8665                            write_u32(
8666                                &mut out,
8667                                u32::try_from(locators.len()).expect("≤ 4G locators/posting list"),
8668                            );
8669                            for loc in locators {
8670                                loc.write_le(&mut out);
8671                            }
8672                        }
8673                    }
8674                    IndexKind::GinFulltext(map) => {
8675                        // v7.17.0 Phase 2.2 — tag byte 5 =
8676                        // GinFulltext (MySQL `FULLTEXT KEY` GIN
8677                        // over a TEXT/VARCHAR column). Payload
8678                        // shape mirrors tag-3 / tag-4 GIN —
8679                        // `String → Vec<RowLocator>` posting
8680                        // lists keyed by lower-cased word
8681                        // lexemes. FILE_VERSION 33+; v32 catalogs
8682                        // never wrote a fulltext-GIN (FULLTEXT
8683                        // KEY was silently dropped pre-v7.17).
8684                        out.push(5);
8685                        write_u32(
8686                            &mut out,
8687                            u32::try_from(map.len()).expect("≤ 4G fulltext-GIN posting lists"),
8688                        );
8689                        for (lex, locators) in map {
8690                            write_str(&mut out, lex);
8691                            write_u32(
8692                                &mut out,
8693                                u32::try_from(locators.len()).expect("≤ 4G locators/posting list"),
8694                            );
8695                            for loc in locators {
8696                                loc.write_le(&mut out);
8697                            }
8698                        }
8699                    }
8700                    IndexKind::GinJsonb(map) => {
8701                        // v7.37.8 — tag byte 6 = GinJsonb
8702                        // (real posting-list GIN over a JSONB
8703                        // column; sentori Epic 5 P2). Payload
8704                        // shape mirrors tag-3 / 4 / 5 — keys are
8705                        // the canonical `(path, leaf)` tokens
8706                        // from `jsonb_gin::extract_tokens`.
8707                        // FILE_VERSION 51+; v50 catalogs never
8708                        // wrote a JSONB-GIN (the same DDL loaded
8709                        // as a BTree fallback).
8710                        out.push(6);
8711                        write_u32(
8712                            &mut out,
8713                            u32::try_from(map.len()).expect("≤ 4G JSONB-GIN posting lists"),
8714                        );
8715                        for (token, locators) in map {
8716                            write_str(&mut out, token);
8717                            write_u32(
8718                                &mut out,
8719                                u32::try_from(locators.len()).expect("≤ 4G locators/posting list"),
8720                            );
8721                            for loc in locators {
8722                                loc.write_le(&mut out);
8723                            }
8724                        }
8725                    }
8726                }
8727                // v6.8.0 — included_columns appendix per index.
8728                // Layout: [u16 num_included][num × u16 column_position].
8729                // v11 readers stop before this u16 (deserialise loop
8730                // gated on version >= 12); v12+ readers always
8731                // consume it. Empty Vec serialises as a bare 0u16.
8732                write_u16(
8733                    &mut out,
8734                    u16::try_from(idx.included_columns.len()).expect("≤ 65k INCLUDE columns/index"),
8735                );
8736                for col_pos in &idx.included_columns {
8737                    write_u16(
8738                        &mut out,
8739                        u16::try_from(*col_pos).expect("≤ 65k columns/table"),
8740                    );
8741                }
8742                // v6.8.1 — partial_predicate appendix per index.
8743                // Layout: [u8 has_pred][u16 LE len][bytes (if has_pred)].
8744                // Same v12 gate as included_columns.
8745                match &idx.partial_predicate {
8746                    None => out.push(0),
8747                    Some(pred) => {
8748                        out.push(1);
8749                        write_str(&mut out, pred);
8750                    }
8751                }
8752                // v6.8.2 — expression appendix. Same shape as
8753                // partial_predicate.
8754                match &idx.expression {
8755                    None => out.push(0),
8756                    Some(expr) => {
8757                        out.push(1);
8758                        write_str(&mut out, expr);
8759                    }
8760                }
8761                // v7.9.29 — is_unique appendix (FILE_VERSION 16+).
8762                // Single byte 0/1. v15-and-below readers stop before
8763                // this byte; v16 readers always consume it. mailrs K1.
8764                out.push(u8::from(idx.is_unique));
8765                // v7.9.29 — extra_column_positions appendix.
8766                // Layout: [u16 count][count × u16 column_position].
8767                write_u16(
8768                    &mut out,
8769                    u16::try_from(idx.extra_column_positions.len())
8770                        .expect("≤ 65k extra cols / index"),
8771                );
8772                for cp in &idx.extra_column_positions {
8773                    write_u16(&mut out, u16::try_from(*cp).expect("≤ 65k columns/table"));
8774                }
8775                // v7.39 (read01 round 52) — nulls_not_distinct (FILE_VERSION
8776                // 62+). Appended at the end of the per-index block so the v16
8777                // layout above is untouched; v61-and-below readers stop before
8778                // this byte and default the flag to false (NULLS DISTINCT).
8779                out.push(u8::from(idx.nulls_not_distinct));
8780                // v7.39 (round 537) — the key column's ordering clause
8781                // (FILE_VERSION 83+).
8782                out.push(u8::from(idx.descending));
8783                out.push(match idx.nulls_first {
8784                    None => 0,
8785                    Some(true) => 1,
8786                    Some(false) => 2,
8787                });
8788                // v7.39 (round 538) — the key's explicit collation
8789                // (FILE_VERSION 84+).
8790                match &idx.collation {
8791                    Some(c) => {
8792                        out.push(1);
8793                        write_str(&mut out, c);
8794                    }
8795                    None => out.push(0),
8796                }
8797            }
8798            // v6.7.2 — per-table hot_tier_bytes Option<u64>.
8799            // Layout: [u8 has_value][u64 LE value (if has_value)].
8800            // v10 readers stop before this byte (deserialise loop
8801            // gated on version >= 11); v11+ readers always
8802            // consume it.
8803            match t.schema.hot_tier_bytes {
8804                None => out.push(0),
8805                Some(n) => {
8806                    out.push(1);
8807                    out.extend_from_slice(&n.to_le_bytes());
8808                }
8809            }
8810            // v7.6.1 — FOREIGN KEY appendix (catalog FILE_VERSION 13+).
8811            // Layout: [u16 LE fk_count]
8812            //   per fk:
8813            //     [u8 has_name] [str name (if has_name)]
8814            //     [u16 LE local_arity] [u16 LE local_pos]*arity
8815            //     [str parent_table]
8816            //     [u16 LE parent_arity] [u16 LE parent_pos]*arity
8817            //     [u8 on_delete_tag] [u8 on_update_tag]
8818            // Older catalogs (v12 and below) skip this block entirely;
8819            // their reader stops before this byte.
8820            write_u16(
8821                &mut out,
8822                u16::try_from(t.schema.foreign_keys.len()).expect("≤ 65k FKs/table"),
8823            );
8824            for fk in &t.schema.foreign_keys {
8825                match &fk.name {
8826                    None => out.push(0),
8827                    Some(n) => {
8828                        out.push(1);
8829                        write_str(&mut out, n);
8830                    }
8831                }
8832                write_u16(
8833                    &mut out,
8834                    u16::try_from(fk.local_columns.len()).expect("≤ 65k FK columns"),
8835                );
8836                for &p in &fk.local_columns {
8837                    write_u16(&mut out, u16::try_from(p).expect("≤ 65k columns/table"));
8838                }
8839                write_str(&mut out, &fk.parent_table);
8840                write_u16(
8841                    &mut out,
8842                    u16::try_from(fk.parent_columns.len()).expect("≤ 65k FK parent columns"),
8843                );
8844                for &p in &fk.parent_columns {
8845                    write_u16(&mut out, u16::try_from(p).expect("≤ 65k columns/table"));
8846                }
8847                out.push(fk.on_delete.tag());
8848                out.push(fk.on_update.tag());
8849                // v7.38 (read01, T29) — MATCH type tag (FILE_VERSION 55+).
8850                out.push(fk.match_type.tag());
8851                // v7.39 (round 288) — constraint timing (FILE_VERSION 79+).
8852                // One byte, bit 0 = DEFERRABLE, bit 1 = INITIALLY DEFERRED.
8853                out.push(u8::from(fk.deferrable) | (u8::from(fk.initially_deferred) << 1));
8854            }
8855            // v7.9.19 — UniquenessConstraint appendix (catalog
8856            // FILE_VERSION 15+). Layout per table after the FK
8857            // block:
8858            //   [u16 count]
8859            //     per constraint:
8860            //       [u8 is_primary_key]
8861            //       [u16 arity][u16 col_pos]*arity
8862            // Older catalogs (v14 and below) skip this block.
8863            write_u16(
8864                &mut out,
8865                u16::try_from(t.schema.uniqueness_constraints.len())
8866                    .expect("≤ 65k uniqueness constraints/table"),
8867            );
8868            for uc in &t.schema.uniqueness_constraints {
8869                out.push(u8::from(uc.is_primary_key));
8870                write_u16(
8871                    &mut out,
8872                    u16::try_from(uc.columns.len()).expect("≤ 65k cols in uniqueness constraint"),
8873                );
8874                for &p in &uc.columns {
8875                    write_u16(&mut out, u16::try_from(p).expect("≤ 65k columns/table"));
8876                }
8877                // v7.13.0 — `nulls_not_distinct` flag
8878                // (FILE_VERSION 23+). Always written by writers at
8879                // version 23+; deserialise gates on `version >= 23`
8880                // so v22-and-below catalogs round-trip cleanly.
8881                out.push(u8::from(uc.nulls_not_distinct));
8882            }
8883            // v7.9.21 — runtime_default appendix per table.
8884            // Layout: [u16 count] then for each:
8885            //   [u16 col_pos][str expr]
8886            // Only columns whose runtime_default is Some land here;
8887            // catalog stays compact for the common literal-default
8888            // case.
8889            let mut rt_defaults: Vec<(usize, &str)> = Vec::new();
8890            for (i, c) in t.schema.columns.iter().enumerate() {
8891                if let Some(e) = &c.runtime_default {
8892                    rt_defaults.push((i, e.as_str()));
8893                }
8894            }
8895            write_u16(
8896                &mut out,
8897                u16::try_from(rt_defaults.len()).expect("≤ 65k runtime defaults/table"),
8898            );
8899            for (pos, expr) in rt_defaults {
8900                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
8901                write_str(&mut out, expr);
8902            }
8903            // v7.13.0 — CHECK constraint appendix per table.
8904            // Layout: [u16 count] then `count` Display-form
8905            // expression strings. Re-parsed on every INSERT/UPDATE
8906            // by the engine. FILE_VERSION 23+ only; v22 readers
8907            // never reach this block because the writer also moves
8908            // to v23 in lock-step.
8909            write_u16(
8910                &mut out,
8911                u16::try_from(t.schema.checks.len()).expect("≤ 65k CHECK constraints/table"),
8912            );
8913            for c in &t.schema.checks {
8914                // v7.39 (read01 round 48) — the expr stays in this v23
8915                // appendix (byte layout unchanged for old readers); the
8916                // name rides the v60 constraint-name appendix at the tail.
8917                write_str(&mut out, c.expr.as_str());
8918            }
8919            // v7.17.0 Phase 1.4 — per-table user_enum_type
8920            // appendix. Layout: [u16 count] then
8921            // [u16 col_pos][str enum_name] per binding. Only
8922            // columns whose user_enum_type is Some land here.
8923            let mut enum_bindings: Vec<(usize, &str)> = Vec::new();
8924            for (i, c) in t.schema.columns.iter().enumerate() {
8925                if let Some(e) = &c.user_enum_type {
8926                    enum_bindings.push((i, e.as_str()));
8927                }
8928            }
8929            write_u16(
8930                &mut out,
8931                u16::try_from(enum_bindings.len()).expect("≤ 65k enum-typed columns/table"),
8932            );
8933            for (pos, ename) in enum_bindings {
8934                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
8935                write_str(&mut out, ename);
8936            }
8937            // v7.17.0 Phase 1.5 — per-table user_domain_type
8938            // appendix. Same layout as the enum one. v29-and-
8939            // below readers stop after the enum appendix.
8940            let mut domain_bindings: Vec<(usize, &str)> = Vec::new();
8941            for (i, c) in t.schema.columns.iter().enumerate() {
8942                if let Some(d) = &c.user_domain_type {
8943                    domain_bindings.push((i, d.as_str()));
8944                }
8945            }
8946            write_u16(
8947                &mut out,
8948                u16::try_from(domain_bindings.len()).expect("≤ 65k domain-typed columns/table"),
8949            );
8950            for (pos, dname) in domain_bindings {
8951                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
8952                write_str(&mut out, dname);
8953            }
8954            // v7.17.0 Phase 2.1 — per-table on_update_runtime
8955            // appendix. Sparse: only ON UPDATE-bound columns.
8956            let mut on_update_bindings: Vec<(usize, &str)> = Vec::new();
8957            for (i, c) in t.schema.columns.iter().enumerate() {
8958                if let Some(e) = &c.on_update_runtime {
8959                    on_update_bindings.push((i, e.as_str()));
8960                }
8961            }
8962            write_u16(
8963                &mut out,
8964                u16::try_from(on_update_bindings.len()).expect("≤ 65k ON UPDATE columns/table"),
8965            );
8966            for (pos, expr_src) in on_update_bindings {
8967                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
8968                write_str(&mut out, expr_src);
8969            }
8970            // v7.17.0 Phase 2.5 — per-table collation appendix.
8971            // Sparse: only non-Binary columns land. Layout:
8972            // `[u16 count][u16 col_pos][u8 tag] × count`.
8973            let mut coll_bindings: Vec<(usize, u8)> = Vec::new();
8974            for (i, c) in t.schema.columns.iter().enumerate() {
8975                let tag = match c.collation {
8976                    Collation::Binary => continue,
8977                    Collation::CaseInsensitive => Collation::TAG_CASE_INSENSITIVE,
8978                };
8979                coll_bindings.push((i, tag));
8980            }
8981            write_u16(
8982                &mut out,
8983                u16::try_from(coll_bindings.len()).expect("≤ 65k collation bindings/table"),
8984            );
8985            for (pos, tag) in coll_bindings {
8986                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
8987                out.push(tag);
8988            }
8989            // v7.17.0 Phase 4.4 — per-table is_unsigned appendix.
8990            // Sparse: only UNSIGNED columns land. Layout:
8991            // `[u16 count][u16 col_pos] × count`.
8992            let mut unsigned_bindings: Vec<usize> = Vec::new();
8993            for (i, c) in t.schema.columns.iter().enumerate() {
8994                if c.is_unsigned {
8995                    unsigned_bindings.push(i);
8996                }
8997            }
8998            write_u16(
8999                &mut out,
9000                u16::try_from(unsigned_bindings.len()).expect("≤ 65k UNSIGNED columns/table"),
9001            );
9002            for pos in unsigned_bindings {
9003                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9004            }
9005            // v7.17.0 Phase 3.P0-36 — per-table inline_enum_variants
9006            // appendix. Sparse: only ENUM columns land. Layout:
9007            // `[u16 count] then per binding [u16 col_pos]
9008            // [u16 variant_count] then variant strings`.
9009            // FILE_VERSION 41+; v40 readers never reach this block.
9010            let mut enum_inline_bindings: Vec<(usize, &[String])> = Vec::new();
9011            for (i, c) in t.schema.columns.iter().enumerate() {
9012                if let Some(vs) = &c.inline_enum_variants {
9013                    enum_inline_bindings.push((i, vs.as_slice()));
9014                }
9015            }
9016            write_u16(
9017                &mut out,
9018                u16::try_from(enum_inline_bindings.len()).expect("≤ 65k inline-ENUM columns/table"),
9019            );
9020            for (pos, variants) in enum_inline_bindings {
9021                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9022                write_u16(
9023                    &mut out,
9024                    u16::try_from(variants.len()).expect("≤ 65k variants/ENUM"),
9025                );
9026                for v in variants {
9027                    write_str(&mut out, v.as_str());
9028                }
9029            }
9030            // v7.17.0 Phase 3.P0-37 — per-table inline_set_variants
9031            // appendix. Same layout as the inline ENUM block.
9032            // FILE_VERSION 42+; v41 readers never reach this block.
9033            let mut set_inline_bindings: Vec<(usize, &[String])> = Vec::new();
9034            for (i, c) in t.schema.columns.iter().enumerate() {
9035                if let Some(vs) = &c.inline_set_variants {
9036                    set_inline_bindings.push((i, vs.as_slice()));
9037                }
9038            }
9039            write_u16(
9040                &mut out,
9041                u16::try_from(set_inline_bindings.len()).expect("≤ 65k inline-SET columns/table"),
9042            );
9043            for (pos, variants) in set_inline_bindings {
9044                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9045                write_u16(
9046                    &mut out,
9047                    u16::try_from(variants.len()).expect("≤ 65k variants/SET"),
9048                );
9049                for v in variants {
9050                    write_str(&mut out, v.as_str());
9051                }
9052            }
9053            // v7.37.6-B — partition role appendix(FILE_VERSION 49+)。
9054            // Layout 详见 FILE_VERSION 49 docstring。普通表 = 单字节 0。
9055            write_partition_role(&mut out, t.schema.partition_role.as_ref());
9056            // v7.37.7 — per-table generated_stored_expr appendix
9057            // (FILE_VERSION 50+). Sparse: only columns whose
9058            // generated_stored_expr is Some land here.
9059            let mut gen_bindings: Vec<(usize, &str)> = Vec::new();
9060            for (i, c) in t.schema.columns.iter().enumerate() {
9061                if let Some(src) = &c.generated_stored_expr {
9062                    gen_bindings.push((i, src.as_str()));
9063                }
9064            }
9065            write_u16(
9066                &mut out,
9067                u16::try_from(gen_bindings.len()).expect("≤ 65k GENERATED STORED columns/table"),
9068            );
9069            for (pos, src) in gen_bindings {
9070                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9071                write_str(&mut out, src);
9072            }
9073            // v7.38 (read01) — per-table default_text appendix
9074            // (FILE_VERSION 58+). Sparse: only columns whose default_text
9075            // is Some land here. Mirrors the generated_stored_expr shape.
9076            let mut default_texts: Vec<(usize, &str)> = Vec::new();
9077            for (i, c) in t.schema.columns.iter().enumerate() {
9078                if let Some(src) = &c.default_text {
9079                    default_texts.push((i, src.as_str()));
9080                }
9081            }
9082            write_u16(
9083                &mut out,
9084                u16::try_from(default_texts.len()).expect("≤ 65k defaulted columns/table"),
9085            );
9086            for (pos, src) in default_texts {
9087                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9088                write_str(&mut out, src);
9089            }
9090            // v7.39 (RLS) — per-table policy appendix + the two RLS flags
9091            // (FILE_VERSION 59+). Written after the default_text block and
9092            // before the MVCC row appendix, so a v58 reader stops before it.
9093            // Layout: [u8 row_security][u8 force] [u16 policy_count] then per
9094            // policy: [str name][u8 cmd][u8 permissive][u16 role_count]
9095            // (role_count × str) [u8 has_using](+str)[u8 has_check](+str).
9096            out.push(u8::from(t.schema.row_security));
9097            out.push(u8::from(t.schema.force_row_security));
9098            write_u16(
9099                &mut out,
9100                u16::try_from(t.schema.policies.len()).expect("≤ 65k policies/table"),
9101            );
9102            for p in &t.schema.policies {
9103                write_str(&mut out, &p.name);
9104                out.push(p.cmd.to_wire_byte());
9105                out.push(u8::from(p.permissive));
9106                write_u16(
9107                    &mut out,
9108                    u16::try_from(p.roles.len()).expect("≤ 65k roles/policy"),
9109                );
9110                for r in &p.roles {
9111                    write_str(&mut out, r);
9112                }
9113                match &p.using_expr {
9114                    Some(s) => {
9115                        out.push(1);
9116                        write_str(&mut out, s);
9117                    }
9118                    None => out.push(0),
9119                }
9120                match &p.with_check_expr {
9121                    Some(s) => {
9122                        out.push(1);
9123                        write_str(&mut out, s);
9124                    }
9125                    None => out.push(0),
9126                }
9127            }
9128            // v7.37.16 (Epic W) — per-row MVCC header + stable RowId
9129            // appendix (FILE_VERSION 53+). Persists xmin/xmax/flags +
9130            // RowId for every row so a tombstone naming a pre-checkpoint
9131            // row survives a serialize→deserialize base restore
9132            // (cross-checkpoint tombstone durability). `headers` /
9133            // `rowids` are lock-step parallel to `rows` (invariant held
9134            // at every mutation boundary), so the count is `rows.len()`
9135            // and the zipped walk visits them in physical row order —
9136            // the same order the rows block above was written in. v52
9137            // readers never reach this block (the writer also moves to
9138            // v53 in lock-step); a v53 reader restores headers + ids
9139            // verbatim instead of freezing + dense-assigning.
9140            debug_assert_eq!(
9141                t.rows.len(),
9142                t.headers.len(),
9143                "headers must be lock-step with rows at serialize"
9144            );
9145            debug_assert_eq!(
9146                t.rows.len(),
9147                t.rowids.len(),
9148                "rowids must be lock-step with rows at serialize"
9149            );
9150            write_u32(
9151                &mut out,
9152                u32::try_from(t.rows.len()).expect("≤ 4G rows/table"),
9153            );
9154            for (h, rid) in t.headers.iter().zip(t.rowids.iter()) {
9155                out.extend_from_slice(&h.xmin.to_le_bytes());
9156                out.extend_from_slice(&h.xmax.to_le_bytes());
9157                out.push(h.flags);
9158                out.extend_from_slice(&rid.0.to_le_bytes());
9159            }
9160            out.extend_from_slice(&t.next_rowid.to_le_bytes());
9161            // v7.39 (read01 round 48) — constraint-name appendix
9162            // (FILE_VERSION 60+). Index-aligned to the CHECK and
9163            // uniqueness-constraint appendices written above, so the
9164            // existing byte layouts stay untouched and a v59 catalog still
9165            // decodes (its constraints just come back unnamed).
9166            // Layout: [u16 check_count] then per check
9167            //         [u8 has_name] ([str name] when has_name)
9168            //         [u16 uc_count] then per uc the same pair.
9169            write_u16(
9170                &mut out,
9171                u16::try_from(t.schema.checks.len()).expect("≤ 65k CHECK constraints/table"),
9172            );
9173            for c in &t.schema.checks {
9174                match &c.name {
9175                    Some(n) => {
9176                        out.push(1);
9177                        write_str(&mut out, n);
9178                    }
9179                    None => out.push(0),
9180                }
9181            }
9182            write_u16(
9183                &mut out,
9184                u16::try_from(t.schema.uniqueness_constraints.len())
9185                    .expect("≤ 65k uniqueness constraints/table"),
9186            );
9187            for uc in &t.schema.uniqueness_constraints {
9188                match &uc.name {
9189                    Some(n) => {
9190                        out.push(1);
9191                        write_str(&mut out, n);
9192                    }
9193                    None => out.push(0),
9194                }
9195            }
9196            // v7.39 (read01 round 56) — user_composite_type appendix
9197            // (FILE_VERSION 63+). Sparse, at the very end of the per-table
9198            // block: only composite-typed columns land here, so a v62 reader
9199            // stops before it and its composite columns stay plain JSON.
9200            let mut comp_bindings: Vec<(usize, &str)> = Vec::new();
9201            for (i, c) in t.schema.columns.iter().enumerate() {
9202                if let Some(n) = &c.user_composite_type {
9203                    comp_bindings.push((i, n.as_str()));
9204                }
9205            }
9206            write_u16(
9207                &mut out,
9208                u16::try_from(comp_bindings.len()).expect("≤ 65k composite-typed columns/table"),
9209            );
9210            for (pos, n) in comp_bindings {
9211                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9212                write_str(&mut out, n);
9213            }
9214            // v7.39 (read01 round 57) — owner + ACL appendix (FILE_VERSION
9215            // 64+), at the very end of the per-table block so a v63 reader
9216            // stops before it (its tables then read back owner-less, i.e.
9217            // owned by the login role, with no grants — which is exactly what
9218            // they were).
9219            match &t.schema.owner {
9220                Some(o) => {
9221                    out.push(1);
9222                    write_str(&mut out, o);
9223                }
9224                None => out.push(0),
9225            }
9226            write_u16(
9227                &mut out,
9228                u16::try_from(t.schema.acl.len()).expect("≤ 65k aclitems/table"),
9229            );
9230            for a in &t.schema.acl {
9231                write_str(&mut out, &a.grantee);
9232                write_u16(&mut out, a.privs);
9233                write_u16(&mut out, a.grantable);
9234                write_str(&mut out, &a.grantor);
9235            }
9236            // v7.39 (read01 round 59) — COLUMN acl appendix (FILE_VERSION 65+),
9237            // sparse: only columns that carry a grant land here, so a v64 reader
9238            // stops before it and its columns read back un-granted, which is
9239            // what they were.
9240            let granted: Vec<(usize, &ColumnSchema)> = t
9241                .schema
9242                .columns
9243                .iter()
9244                .enumerate()
9245                .filter(|(_, c)| !c.acl.is_empty())
9246                .collect();
9247            write_u16(
9248                &mut out,
9249                u16::try_from(granted.len()).expect("≤ 65k granted columns/table"),
9250            );
9251            for (pos, c) in granted {
9252                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9253                write_u16(
9254                    &mut out,
9255                    u16::try_from(c.acl.len()).expect("≤ 65k aclitems/column"),
9256                );
9257                for a in &c.acl {
9258                    write_str(&mut out, &a.grantee);
9259                    write_u16(&mut out, a.privs);
9260                    write_u16(&mut out, a.grantable);
9261                    write_str(&mut out, &a.grantor);
9262                }
9263            }
9264            // v7.39 (round 210) — EXCLUDE-constraint appendix (FILE_VERSION
9265            // 72+), at the very end of the per-table block so a v71 reader
9266            // stops before it and its tables read back with no exclusion
9267            // constraints. Layout: [u16 excl_count] then per constraint
9268            // [str name] [u8 has_method](+str) [u16 elem_count] then per
9269            // element [u16 col_pos][str op].
9270            write_u16(
9271                &mut out,
9272                u16::try_from(t.schema.exclusion_constraints.len())
9273                    .expect("≤ 65k exclusion constraints/table"),
9274            );
9275            for ex in &t.schema.exclusion_constraints {
9276                write_str(&mut out, &ex.name);
9277                match &ex.method {
9278                    Some(m) => {
9279                        out.push(1);
9280                        write_str(&mut out, m);
9281                    }
9282                    None => out.push(0),
9283                }
9284                write_u16(
9285                    &mut out,
9286                    u16::try_from(ex.elements.len()).expect("≤ 65k elements/exclusion"),
9287                );
9288                for (pos, op) in &ex.elements {
9289                    write_u16(&mut out, u16::try_from(*pos).expect("≤ 65k columns/table"));
9290                    write_str(&mut out, op);
9291                }
9292            }
9293            // v7.39 (round 220) — identity-RESTART appendix (FILE_VERSION
9294            // 73+), sparse: only columns carrying a RESTART floor land here.
9295            let restarts: Vec<(usize, i64)> = t
9296                .schema
9297                .columns
9298                .iter()
9299                .enumerate()
9300                .filter_map(|(i, c)| c.auto_restart.map(|n| (i, n)))
9301                .collect();
9302            write_u16(
9303                &mut out,
9304                u16::try_from(restarts.len()).expect("≤ 65k restart columns/table"),
9305            );
9306            for (pos, n) in restarts {
9307                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9308                out.extend_from_slice(&n.to_le_bytes());
9309            }
9310            // v7.39 (round 386, type-fidelity epic P1) — per-table
9311            // mysql_int_width appendix (FILE_VERSION 81+). Sparse: only
9312            // TINYINT / MEDIUMINT columns land. Layout:
9313            // `[u16 count]([u16 col_pos][u8 width_tag]) × count`
9314            // (tag 0 = Tiny, 1 = Medium). v80-and-below readers stop after
9315            // the identity-RESTART appendix, leaving every column at None.
9316            let int_widths: Vec<(usize, u8)> = t
9317                .schema
9318                .columns
9319                .iter()
9320                .enumerate()
9321                .filter_map(|(i, c)| {
9322                    c.mysql_int_width.map(|w| {
9323                        let tag = match w {
9324                            MysqlIntWidth::Tiny => 0u8,
9325                            MysqlIntWidth::Medium => 1u8,
9326                            MysqlIntWidth::Small => 2u8,
9327                            MysqlIntWidth::Int => 3u8,
9328                            MysqlIntWidth::Big => 4u8,
9329                        };
9330                        (i, tag)
9331                    })
9332                })
9333                .collect();
9334            write_u16(
9335                &mut out,
9336                u16::try_from(int_widths.len()).expect("≤ 65k narrow-int columns/table"),
9337            );
9338            for (pos, tag) in int_widths {
9339                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9340                out.push(tag);
9341            }
9342            // v7.39 (round 424, type-fidelity epic) — per-table mysql_fsp
9343            // appendix (FILE_VERSION 82+). Sparse: only MySQL-declared
9344            // temporal columns land. Layout:
9345            // `[u16 count]([u16 col_pos][u8 fsp]) × count`, fsp in 0..=6.
9346            // v81-and-below readers stop after the int-width appendix,
9347            // leaving every column at None (PG microsecond behaviour).
9348            let fsps: Vec<(usize, u8)> = t
9349                .schema
9350                .columns
9351                .iter()
9352                .enumerate()
9353                .filter_map(|(i, c)| c.mysql_fsp.map(|p| (i, p)))
9354                .collect();
9355            write_u16(
9356                &mut out,
9357                u16::try_from(fsps.len()).expect("≤ 65k temporal columns/table"),
9358            );
9359            for (pos, fsp) in fsps {
9360                write_u16(&mut out, u16::try_from(pos).expect("≤ 65k columns/table"));
9361                out.push(fsp);
9362            }
9363            // v7.39 (round 652) — CHECK-validated appendix (FILE_VERSION
9364            // 87+). Sparse the other way round from the ones above: the
9365            // common case is every constraint validated, so only the
9366            // NOT VALID ones are written, by their index into the CHECK
9367            // appendix. Layout: `[u16 count]([u16 check_idx]) × count`.
9368            let unvalidated: Vec<usize> = t
9369                .schema
9370                .checks
9371                .iter()
9372                .enumerate()
9373                .filter_map(|(i, c)| (!c.validated).then_some(i))
9374                .collect();
9375            write_u16(
9376                &mut out,
9377                u16::try_from(unvalidated.len()).expect("≤ 65k CHECK constraints/table"),
9378            );
9379            for idx in unvalidated {
9380                write_u16(&mut out, u16::try_from(idx).expect("≤ 65k CHECK/table"));
9381            }
9382            // v7.39 (round 677) — per-column collation names (FILE_VERSION
9383            // 88+). Sparse: only the columns that were written with an
9384            // explicit `COLLATE` appear, so a table that declares none pays
9385            // two bytes. Layout: `[u16 count]([u16 col_idx][str]) × count`.
9386            //
9387            // Without this the declaration survives CREATE TABLE and dies
9388            // at the next restart — measured: a column declared
9389            // `COLLATE "C"` reported attcollation 950 in the session that
9390            // created it and 100 after a reload.
9391            let collated: Vec<(usize, &str)> = t
9392                .schema
9393                .columns
9394                .iter()
9395                .enumerate()
9396                .filter_map(|(i, c)| c.collation_name.as_deref().map(|n| (i, n)))
9397                .collect();
9398            write_u16(
9399                &mut out,
9400                u16::try_from(collated.len()).expect("≤ 65k columns/table"),
9401            );
9402            for (idx, name) in collated {
9403                write_u16(&mut out, u16::try_from(idx).expect("≤ 65k columns/table"));
9404                write_str(&mut out, name);
9405            }
9406            // v7.39 (round 711) — PK/UNIQUE constraint timing (FILE_VERSION
9407            // 89+). Dense, one byte per uniqueness constraint in
9408            // declaration order, the same bit layout the FK block has
9409            // carried since round 288: bit 0 = DEFERRABLE, bit 1 =
9410            // INITIALLY DEFERRED. A v88 reader stops before it.
9411            write_u16(
9412                &mut out,
9413                u16::try_from(t.schema.uniqueness_constraints.len())
9414                    .expect("≤ 65k uniqueness constraints/table"),
9415            );
9416            for uc in &t.schema.uniqueness_constraints {
9417                out.push(u8::from(uc.deferrable) | (u8::from(uc.initially_deferred) << 1));
9418            }
9419        }
9420        // v7.12.4 — catalog-wide appendix: user-defined functions
9421        // then triggers. FILE_VERSION 22+ only. v21 and earlier
9422        // readers stop after the last table; v22 readers always
9423        // consume two `u32` counts (possibly zero).
9424        //
9425        // Function entry layout:
9426        //   [str name] [str args_repr] [str returns]
9427        //   [str language] [str body]
9428        // Trigger entry layout:
9429        //   [str name] [str table] [str timing]
9430        //   [u16 event_count] (event_count × str)
9431        //   [str for_each] [str function]
9432        write_u32(
9433            &mut out,
9434            u32::try_from(self.functions.len()).expect("≤ 4G functions"),
9435        );
9436        for fd in self.functions.values() {
9437            write_str(&mut out, &fd.name);
9438            write_str(&mut out, &fd.args_repr);
9439            write_str(&mut out, &fd.returns);
9440            write_str(&mut out, &fd.language);
9441            write_str_long(&mut out, &fd.body);
9442        }
9443        write_u32(
9444            &mut out,
9445            u32::try_from(self.triggers.len()).expect("≤ 4G triggers"),
9446        );
9447        for td in &self.triggers {
9448            write_str(&mut out, &td.name);
9449            write_str(&mut out, &td.table);
9450            write_str(&mut out, &td.timing);
9451            write_u16(
9452                &mut out,
9453                u16::try_from(td.events.len()).expect("≤ 65k events / trigger"),
9454            );
9455            for ev in &td.events {
9456                write_str(&mut out, ev);
9457            }
9458            write_str(&mut out, &td.for_each);
9459            write_str(&mut out, &td.function);
9460            // v7.13.0 — `UPDATE OF cols` filter
9461            // (FILE_VERSION 23+). v22 readers omit; v23 writers
9462            // always emit (possibly zero).
9463            write_u16(
9464                &mut out,
9465                u16::try_from(td.update_columns.len()).expect("≤ 65k cols / trigger"),
9466            );
9467            for c in &td.update_columns {
9468                write_str(&mut out, c);
9469            }
9470            // v7.16.1 — TriggerDef.enabled (FILE_VERSION 25+).
9471            out.push(u8::from(td.enabled));
9472            // v7.39 (round 138) — WHEN condition text (FILE_VERSION 70+).
9473            write_str(&mut out, &td.when_condition);
9474        }
9475        // v7.17.0 Phase 1.1 — SEQUENCE catalog block (FILE_VERSION 26+).
9476        write_u32(
9477            &mut out,
9478            u32::try_from(self.sequences.len()).expect("≤ 4G sequences"),
9479        );
9480        for seq in self.sequences.values() {
9481            write_str(&mut out, &seq.name);
9482            out.push(match seq.data_type {
9483                SequenceDataType::SmallInt => 0,
9484                SequenceDataType::Int => 1,
9485                SequenceDataType::BigInt => 2,
9486            });
9487            out.extend_from_slice(&seq.start.to_le_bytes());
9488            out.extend_from_slice(&seq.increment.to_le_bytes());
9489            out.extend_from_slice(&seq.min_value.to_le_bytes());
9490            out.extend_from_slice(&seq.max_value.to_le_bytes());
9491            out.extend_from_slice(&seq.cache.to_le_bytes());
9492            out.push(u8::from(seq.cycle));
9493            match &seq.owned_by {
9494                None => out.push(0),
9495                Some((table, column)) => {
9496                    out.push(1);
9497                    write_str(&mut out, table);
9498                    write_str(&mut out, column);
9499                }
9500            }
9501            out.extend_from_slice(&seq.last_value.to_le_bytes());
9502            out.push(u8::from(seq.is_called));
9503        }
9504        // v7.17.0 Phase 1.2 — VIEW catalog block (FILE_VERSION 27+).
9505        write_u32(
9506            &mut out,
9507            u32::try_from(self.views.len()).expect("≤ 4G views"),
9508        );
9509        for view in self.views.values() {
9510            write_str(&mut out, &view.name);
9511            write_u16(
9512                &mut out,
9513                u16::try_from(view.columns.len()).expect("≤ 65k cols / view"),
9514            );
9515            for c in &view.columns {
9516                write_str(&mut out, c);
9517            }
9518            write_str_long(&mut out, &view.body);
9519            // v7.39 (round 132, FILE_VERSION 69+) — WITH CHECK OPTION marker.
9520            out.push(view.check_option);
9521        }
9522        // v7.17.0 Phase 1.3 — MATERIALIZED VIEW source registry
9523        // (FILE_VERSION 28+). The backing rows live as a regular
9524        // table of the same name already in the tables block.
9525        write_u32(
9526            &mut out,
9527            u32::try_from(self.materialized_views.len()).expect("≤ 4G materialized views"),
9528        );
9529        for (name, body) in &self.materialized_views {
9530            write_str(&mut out, name);
9531            write_str_long(&mut out, body);
9532        }
9533        // v7.17.0 Phase 1.4 — ENUM types catalog block
9534        // (FILE_VERSION 29+).
9535        write_u32(
9536            &mut out,
9537            u32::try_from(self.enum_types.len()).expect("≤ 4G enum types"),
9538        );
9539        for e in self.enum_types.values() {
9540            write_str(&mut out, &e.name);
9541            write_u16(
9542                &mut out,
9543                u16::try_from(e.labels.len()).expect("≤ 65k labels / enum"),
9544            );
9545            for l in &e.labels {
9546                write_str(&mut out, l);
9547            }
9548        }
9549        // v7.17.0 Phase 1.5 — DOMAIN types catalog block
9550        // (FILE_VERSION 30+).
9551        write_u32(
9552            &mut out,
9553            u32::try_from(self.domain_types.len()).expect("≤ 4G domain types"),
9554        );
9555        for d in self.domain_types.values() {
9556            write_str(&mut out, &d.name);
9557            write_data_type(&mut out, d.base_type);
9558            out.push(u8::from(d.nullable));
9559            match &d.default {
9560                None => out.push(0),
9561                Some(s) => {
9562                    out.push(1);
9563                    write_str(&mut out, s);
9564                }
9565            }
9566            write_u16(
9567                &mut out,
9568                u16::try_from(d.checks.len()).expect("≤ 65k CHECKs / domain"),
9569            );
9570            for c in &d.checks {
9571                write_str(&mut out, &c.expr);
9572                // v7.39 (round 260) — the constraint name (FILE_VERSION 75+).
9573                write_str(&mut out, &c.name);
9574            }
9575            // v7.39 (round 259) — the parent domain (FILE_VERSION 74+).
9576            match &d.base_domain {
9577                None => out.push(0),
9578                Some(s) => {
9579                    out.push(1);
9580                    write_str(&mut out, s);
9581                }
9582            }
9583        }
9584        // v7.17.0 Phase 1.6 — user-schemas registry
9585        // (FILE_VERSION 31+). Built-ins are hardcoded in
9586        // `is_builtin_schema` and not persisted.
9587        write_u32(
9588            &mut out,
9589            u32::try_from(self.schemas.len()).expect("≤ 4G schemas"),
9590        );
9591        for name in &self.schemas {
9592            write_str(&mut out, name);
9593        }
9594        // v7.37.42-T2 ζ-B — COMPOSITE types catalog block
9595        // (FILE_VERSION 52+). Each entry: name, u16 field_count,
9596        // then field_count `[str field_name][data_type]` pairs.
9597        write_u32(
9598            &mut out,
9599            u32::try_from(self.composite_types.len()).expect("≤ 4G composite types"),
9600        );
9601        for c in self.composite_types.values() {
9602            write_str(&mut out, &c.name);
9603            write_u16(
9604                &mut out,
9605                u16::try_from(c.fields.len()).expect("≤ 65k fields / composite"),
9606            );
9607            for (i, (fname, fty)) in c.fields.iter().enumerate() {
9608                write_str(&mut out, fname);
9609                write_data_type(&mut out, *fty);
9610                // v7.39 (round 264) — the field's user type (v76+).
9611                match c.field_user_types.get(i).and_then(Option::as_ref) {
9612                    None => out.push(0),
9613                    Some(n) => {
9614                        out.push(1);
9615                        write_str(&mut out, n);
9616                    }
9617                }
9618            }
9619        }
9620        // v7.39 (read01 round 50) — COMMENT store (FILE_VERSION 61+).
9621        // Catalog-wide, written last (before the CRC trailer) so every older
9622        // reader stops before it. Layout: [u32 count] then [str key][str text].
9623        write_u32(
9624            &mut out,
9625            u32::try_from(self.comments.len()).expect("≤ 4G comments"),
9626        );
9627        for (k, v) in &self.comments {
9628            write_str(&mut out, k);
9629            write_str_long(&mut out, v);
9630        }
9631        // v7.39 (read01 round 60) — non-table ACLs (FILE_VERSION 66+), catalog-
9632        // wide and written last so a v65 reader stops before them. The sequence
9633        // block itself sits mid-image and cannot grow without breaking older
9634        // readers, so a sequence's owner + ACL rides here, keyed by name.
9635        let acl_out = |out: &mut Vec<u8>, acl: &[AclItem]| {
9636            write_u16(out, u16::try_from(acl.len()).expect("≤ 65k aclitems"));
9637            for a in acl {
9638                write_str(out, &a.grantee);
9639                write_u16(out, a.privs);
9640                write_u16(out, a.grantable);
9641                write_str(out, &a.grantor);
9642            }
9643        };
9644        let owned: Vec<&SequenceDef> = self
9645            .sequences
9646            .values()
9647            .filter(|s| s.owner.is_some() || !s.acl.is_empty())
9648            .collect();
9649        write_u32(
9650            &mut out,
9651            u32::try_from(owned.len()).expect("≤ 4G sequences"),
9652        );
9653        for seq in owned {
9654            write_str(&mut out, &seq.name);
9655            match &seq.owner {
9656                Some(o) => {
9657                    out.push(1);
9658                    write_str(&mut out, o);
9659                }
9660                None => out.push(0),
9661            }
9662            acl_out(&mut out, &seq.acl);
9663        }
9664        acl_out(&mut out, &self.schema_acl);
9665        acl_out(&mut out, &self.database_acl);
9666        // v7.39 (read01 round 61) — FUNCTION owner + ACL (FILE_VERSION 67+).
9667        // The function block sits mid-image like the sequence one, so this
9668        // rides the catalog-wide tail too, keyed by name.
9669        let fns: Vec<&FunctionDef> = self
9670            .functions
9671            .values()
9672            .filter(|f| f.owner.is_some() || !f.acl.is_empty())
9673            .collect();
9674        write_u32(&mut out, u32::try_from(fns.len()).expect("≤ 4G functions"));
9675        for f in fns {
9676            // v7.39 (read01 round 62) — keyed by SIGNATURE now: two overloads
9677            // have two ACLs.
9678            write_str(&mut out, &function_signature_key(&f.name, &f.args_repr));
9679            match &f.owner {
9680                Some(o) => {
9681                    out.push(1);
9682                    write_str(&mut out, o);
9683                }
9684                None => out.push(0),
9685            }
9686            acl_out(&mut out, &f.acl);
9687        }
9688        // v7.39 (round 139) — RULE catalog block (FILE_VERSION 71+), catalog-
9689        // wide and written last (right before the CRC trailer) so every older
9690        // reader stops cleanly before it. Layout: [u32 count] then per rule
9691        // [str name][str table][str event][u8 instead][str when]
9692        // [u16 cmd_count]([str cmd] × cmd_count).
9693        write_u32(
9694            &mut out,
9695            u32::try_from(self.rules.len()).expect("≤ 4G rules"),
9696        );
9697        for r in &self.rules {
9698            write_str(&mut out, &r.name);
9699            write_str(&mut out, &r.table);
9700            write_str(&mut out, &r.event);
9701            out.push(u8::from(r.instead));
9702            write_str(&mut out, &r.when_condition);
9703            write_u16(
9704                &mut out,
9705                u16::try_from(r.commands.len()).expect("≤ 65k commands / rule"),
9706            );
9707            for c in &r.commands {
9708                write_str(&mut out, c);
9709            }
9710        }
9711        // v7.39 (round 280) — extended-statistics block (FILE_VERSION
9712        // 77+), appended after the RULE block for the same reason: an
9713        // older reader stops cleanly before it. Layout: [u32 count]
9714        // then per object [str name][str table][u16 n]([str kind] × n)
9715        // [u16 m]([str column] × m).
9716        write_u32(
9717            &mut out,
9718            u32::try_from(self.statistics_ext.len()).expect("≤ 4G statistics objects"),
9719        );
9720        for st in &self.statistics_ext {
9721            write_str(&mut out, &st.name);
9722            write_str(&mut out, &st.table);
9723            write_u16(
9724                &mut out,
9725                u16::try_from(st.kinds.len()).expect("≤ 65k kinds"),
9726            );
9727            for k in &st.kinds {
9728                write_str(&mut out, k);
9729            }
9730            write_u16(
9731                &mut out,
9732                u16::try_from(st.columns.len()).expect("≤ 65k columns"),
9733            );
9734            for c in &st.columns {
9735                write_str(&mut out, c);
9736            }
9737        }
9738        // v7.39 (round 287) — large-object block (FILE_VERSION 78+),
9739        // appended after the statistics block for the same reason: an
9740        // older reader stops cleanly before it. Layout: [u32 count]
9741        // then per object [u32 oid][u32 len][len bytes].
9742        write_u32(
9743            &mut out,
9744            u32::try_from(self.large_objects.len()).expect("≤ 4G large objects"),
9745        );
9746        for (oid, bytes) in &self.large_objects {
9747            write_u32(&mut out, *oid);
9748            write_u32(
9749                &mut out,
9750                u32::try_from(bytes.len()).expect("≤ 4G per object"),
9751            );
9752            out.extend_from_slice(bytes);
9753        }
9754        // v7.39 (round 322, V46) — function-attribute block (FILE_VERSION
9755        // 80+), appended last for the same reason as every block before
9756        // it: an older reader stops cleanly ahead of it and simply sees
9757        // functions with PG's default attributes. Only functions that
9758        // declared something non-default are written. Layout: [u32 count]
9759        // then per function [str signature_key][u8 volatility][u8 flags]
9760        // [u8 parallel][f64 cost or NaN][f64 rows or NaN], where flags bit
9761        // 0 = strict, 1 = security definer, 2 = leakproof.
9762        let attr_fns: Vec<(&String, &FunctionDef)> = self
9763            .functions
9764            .iter()
9765            .filter(|(_, f)| {
9766                f.volatility != FN_VOLATILE
9767                    || f.strict
9768                    || f.security_definer
9769                    || f.leakproof
9770                    || f.parallel != FN_PARALLEL_UNSAFE
9771                    || f.cost.is_some()
9772                    || f.rows.is_some()
9773            })
9774            .collect();
9775        write_u32(
9776            &mut out,
9777            u32::try_from(attr_fns.len()).expect("≤ 4G functions"),
9778        );
9779        for (key, f) in attr_fns {
9780            write_str(&mut out, key);
9781            out.push(f.volatility);
9782            let flags = u8::from(f.strict)
9783                | (u8::from(f.security_definer) << 1)
9784                | (u8::from(f.leakproof) << 2);
9785            out.push(flags);
9786            out.push(f.parallel);
9787            out.extend_from_slice(&f.cost.unwrap_or(f64::NAN).to_le_bytes());
9788            out.extend_from_slice(&f.rows.unwrap_or(f64::NAN).to_le_bytes());
9789        }
9790        // v7.38 (read01 P5.05) — CRC32C trailer over the whole image so a
9791        // corrupted snapshot is rejected on load. FILE_VERSION is >= the
9792        // trailer version, so this always runs for freshly-written images.
9793        // v7.39 (round 547) — pg_db_role_setting (FILE_VERSION 85+),
9794        // catalog-wide and written LAST so a v84 reader stops before it.
9795        // Layout: [u32 scopes] then [str database][str role][u32 params]
9796        // then [str name][str value] per param.
9797        write_u32(
9798            &mut out,
9799            u32::try_from(self.db_role_settings.len()).expect("≤ 4G scopes"),
9800        );
9801        for ((db, role), params) in &self.db_role_settings {
9802            write_str(&mut out, db);
9803            write_str(&mut out, role);
9804            write_u32(&mut out, u32::try_from(params.len()).expect("≤ 4G params"));
9805            for (name, value) in params {
9806                write_str(&mut out, name);
9807                write_str(&mut out, value);
9808            }
9809        }
9810        // v7.39 (round 550) — replication slots (FILE_VERSION 86+),
9811        // written LAST so a v85 reader stops before them.
9812        write_u32(
9813            &mut out,
9814            u32::try_from(self.replication_slots.len()).expect("≤ 4G slots"),
9815        );
9816        for (name, (plugin, slot_type)) in &self.replication_slots {
9817            write_str(&mut out, name);
9818            write_str(&mut out, plugin);
9819            write_str(&mut out, slot_type);
9820        }
9821        let crc = spg_crypto::crc32c::crc32c(&out);
9822        write_u32(&mut out, crc);
9823        out
9824    }
9825
9826    /// Deserialize a previously-serialized catalog. Rejects bad magic, version
9827    /// mismatch, unknown tags, truncation, and trailing bytes.
9828    pub fn deserialize(buf: &[u8]) -> Result<Self, StorageError> {
9829        let mut cur = Cursor::new(buf);
9830        let magic = cur.take(8)?;
9831        if magic != FILE_MAGIC {
9832            return Err(StorageError::Corrupt(format!(
9833                "bad magic: expected SPGDB001, got {magic:?}"
9834            )));
9835        }
9836        let version = cur.read_u8()?;
9837        if !(MIN_SUPPORTED_FILE_VERSION..=FILE_VERSION).contains(&version) {
9838            return Err(StorageError::Corrupt(format!(
9839                "unsupported file version: {version} (supported: {MIN_SUPPORTED_FILE_VERSION}..={FILE_VERSION})"
9840            )));
9841        }
9842        // v7.23/v7.27 — escape decoding is version-gated (see
9843        // STR_LEN_ESCAPE / Cursor::codec_version).
9844        cur.codec_version = version;
9845        let table_count = cur.read_u32()? as usize;
9846        let mut cat = Self::new();
9847        for _ in 0..table_count {
9848            deserialize_table(&mut cur, &mut cat, version)?;
9849        }
9850        // v7.37.15 (Phase C.1) — stamp dense stable RelIds on load.
9851        // Pre-V6 envelopes carry no ids; a dense 1..=N assignment is
9852        // sufficient while RelId is process-local bookkeeping (the V6
9853        // envelope, Phase C.6, will round-trip real ids). Sets the
9854        // allocator above the loaded ids so a post-load CREATE TABLE
9855        // never collides.
9856        for (i, t) in cat.tables.iter_mut().enumerate() {
9857            t.set_rel_id(row_header::RelId((i as u64) + 1));
9858        }
9859        cat.next_rel_id = cat.tables.len() as u64;
9860        // v7.12.4 — catalog-wide function + trigger appendix.
9861        // FILE_VERSION 22+ only; v21 and earlier catalogs stop
9862        // after the last table.
9863        if version >= 22 {
9864            let fn_count = cur.read_u32()? as usize;
9865            for _ in 0..fn_count {
9866                let name = cur.read_str()?;
9867                let args_repr = cur.read_str()?;
9868                let returns = cur.read_str()?;
9869                let language = cur.read_str()?;
9870                let body = cur.read_str_long()?;
9871                let key = function_signature_key(&name, &args_repr);
9872                cat.functions.insert(
9873                    key,
9874                    FunctionDef {
9875                        name,
9876                        args_repr,
9877                        returns,
9878                        language,
9879                        body,
9880                        owner: None,
9881                        acl: Vec::new(),
9882                        volatility: FN_VOLATILE,
9883                        strict: false,
9884                        security_definer: false,
9885                        leakproof: false,
9886                        parallel: FN_PARALLEL_UNSAFE,
9887                        cost: None,
9888                        rows: None,
9889                    },
9890                );
9891            }
9892            let trg_count = cur.read_u32()? as usize;
9893            for _ in 0..trg_count {
9894                let name = cur.read_str()?;
9895                let table = cur.read_str()?;
9896                let timing = cur.read_str()?;
9897                let ev_count = cur.read_u16()? as usize;
9898                let mut events = Vec::with_capacity(ev_count);
9899                for _ in 0..ev_count {
9900                    events.push(cur.read_str()?);
9901                }
9902                let for_each = cur.read_str()?;
9903                let function = cur.read_str()?;
9904                // v7.13.0 — trailing `UPDATE OF cols` filter
9905                // (FILE_VERSION 23+ only; v22 catalogs omit and
9906                // deserialise with an empty vec).
9907                let update_columns = if version >= 23 {
9908                    let n = cur.read_u16()? as usize;
9909                    let mut cols = Vec::with_capacity(n);
9910                    for _ in 0..n {
9911                        cols.push(cur.read_str()?);
9912                    }
9913                    cols
9914                } else {
9915                    Vec::new()
9916                };
9917                // v7.16.1 — TriggerDef.enabled (FILE_VERSION 25+).
9918                // v24-and-below catalogs deserialise with `true`
9919                // — pre-v7.16.1 every trigger always fired.
9920                let enabled = if version >= 25 {
9921                    cur.read_u8()? != 0
9922                } else {
9923                    true
9924                };
9925                // v7.39 (round 138) — WHEN condition text added at FILE_VERSION
9926                // 70; older catalogs read back empty (no WHEN filter).
9927                let when_condition = if version >= 70 {
9928                    cur.read_str()?
9929                } else {
9930                    String::new()
9931                };
9932                cat.triggers.push(TriggerDef {
9933                    name,
9934                    table,
9935                    timing,
9936                    events,
9937                    for_each,
9938                    function,
9939                    update_columns,
9940                    enabled,
9941                    when_condition,
9942                });
9943            }
9944        }
9945        // v7.17.0 Phase 1.1 — SEQUENCE block (FILE_VERSION 26+).
9946        // v25-and-below catalogs omit; we leave the map empty.
9947        if version >= 26 {
9948            let seq_count = cur.read_u32()? as usize;
9949            for _ in 0..seq_count {
9950                let name = cur.read_str()?;
9951                let data_type = match cur.read_u8()? {
9952                    0 => SequenceDataType::SmallInt,
9953                    1 => SequenceDataType::Int,
9954                    2 => SequenceDataType::BigInt,
9955                    other => {
9956                        return Err(StorageError::Corrupt(format!(
9957                            "unknown SEQUENCE data-type tag {other}"
9958                        )));
9959                    }
9960                };
9961                let start = cur.read_i64()?;
9962                let increment = cur.read_i64()?;
9963                let min_value = cur.read_i64()?;
9964                let max_value = cur.read_i64()?;
9965                let cache = cur.read_i64()?;
9966                let cycle = cur.read_u8()? != 0;
9967                let owned_by = match cur.read_u8()? {
9968                    0 => None,
9969                    1 => {
9970                        let t = cur.read_str()?;
9971                        let c = cur.read_str()?;
9972                        Some((t, c))
9973                    }
9974                    other => {
9975                        return Err(StorageError::Corrupt(format!(
9976                            "unknown SEQUENCE owned-by tag {other}"
9977                        )));
9978                    }
9979                };
9980                let last_value = cur.read_i64()?;
9981                let is_called = cur.read_u8()? != 0;
9982                cat.sequences.insert(
9983                    name.clone(),
9984                    SequenceDef {
9985                        name,
9986                        data_type,
9987                        start,
9988                        increment,
9989                        min_value,
9990                        max_value,
9991                        cache,
9992                        cycle,
9993                        owned_by,
9994                        last_value,
9995                        is_called,
9996                        owner: None,
9997                        acl: Vec::new(),
9998                    },
9999                );
10000            }
10001        }
10002        // v7.17.0 Phase 1.2 — VIEW block (FILE_VERSION 27+).
10003        // v26-and-below catalogs omit; we leave the map empty.
10004        if version >= 27 {
10005            let view_count = cur.read_u32()? as usize;
10006            for _ in 0..view_count {
10007                let name = cur.read_str()?;
10008                let col_count = cur.read_u16()? as usize;
10009                let mut columns = Vec::with_capacity(col_count);
10010                for _ in 0..col_count {
10011                    columns.push(cur.read_str()?);
10012                }
10013                let body = cur.read_str_long()?;
10014                // v7.39 (round 132) — check-option marker added at FILE_VERSION
10015                // 69; older catalogs default to 0 (no check option).
10016                let check_option = if version >= 69 { cur.read_u8()? } else { 0 };
10017                cat.views.insert(
10018                    name.clone(),
10019                    ViewDef {
10020                        name,
10021                        columns,
10022                        body,
10023                        check_option,
10024                    },
10025                );
10026            }
10027        }
10028        // v7.17.0 Phase 1.3 — MATERIALIZED VIEW source registry
10029        // (FILE_VERSION 28+). v27-and-below catalogs omit.
10030        if version >= 28 {
10031            let mv_count = cur.read_u32()? as usize;
10032            for _ in 0..mv_count {
10033                let name = cur.read_str()?;
10034                let body = cur.read_str_long()?;
10035                cat.materialized_views.insert(name, body);
10036            }
10037        }
10038        // v7.17.0 Phase 1.4 — ENUM types catalog block
10039        // (FILE_VERSION 29+).
10040        if version >= 29 {
10041            let etype_count = cur.read_u32()? as usize;
10042            for _ in 0..etype_count {
10043                let name = cur.read_str()?;
10044                let label_count = cur.read_u16()? as usize;
10045                let mut labels = Vec::with_capacity(label_count);
10046                for _ in 0..label_count {
10047                    labels.push(cur.read_str()?);
10048                }
10049                cat.enum_types
10050                    .insert(name.clone(), EnumDef { name, labels });
10051            }
10052        }
10053        // v7.17.0 Phase 1.5 — DOMAIN types catalog block
10054        // (FILE_VERSION 30+).
10055        if version >= 30 {
10056            let dtype_count = cur.read_u32()? as usize;
10057            for _ in 0..dtype_count {
10058                let name = cur.read_str()?;
10059                let base_type = cur.read_data_type()?;
10060                let nullable = cur.read_u8()? != 0;
10061                let default = match cur.read_u8()? {
10062                    0 => None,
10063                    1 => Some(cur.read_str()?),
10064                    other => {
10065                        return Err(StorageError::Corrupt(format!(
10066                            "unknown DOMAIN default tag {other}"
10067                        )));
10068                    }
10069                };
10070                let check_count = cur.read_u16()? as usize;
10071                let mut checks: Vec<DomainCheck> = Vec::with_capacity(check_count);
10072                for i in 0..check_count {
10073                    let expr = cur.read_str()?;
10074                    // v7.39 (round 260) — names arrived in FILE_VERSION 75.
10075                    // An older catalog gets PG's auto-naming applied to the
10076                    // checks it stored, which is what they would have been.
10077                    let cname = if version >= 75 {
10078                        cur.read_str()?
10079                    } else if i == 0 {
10080                        alloc::format!("{name}_check")
10081                    } else {
10082                        alloc::format!("{name}_check{i}")
10083                    };
10084                    checks.push(DomainCheck { name: cname, expr });
10085                }
10086                // v7.39 (round 259) — the parent domain. Absent before
10087                // FILE_VERSION 74; an older catalog reads as a domain over
10088                // a scalar, which is what it was.
10089                let base_domain = if version >= 74 {
10090                    match cur.read_u8()? {
10091                        0 => None,
10092                        1 => Some(cur.read_str()?),
10093                        other => {
10094                            return Err(StorageError::Corrupt(alloc::format!(
10095                                "domain base_domain tag {other}"
10096                            )));
10097                        }
10098                    }
10099                } else {
10100                    None
10101                };
10102                cat.domain_types.insert(
10103                    name.clone(),
10104                    DomainDef {
10105                        name,
10106                        base_type,
10107                        nullable,
10108                        default,
10109                        checks,
10110                        base_domain,
10111                    },
10112                );
10113            }
10114        }
10115        // v7.17.0 Phase 1.6 — user-schemas registry
10116        // (FILE_VERSION 31+).
10117        if version >= 31 {
10118            let sch_count = cur.read_u32()? as usize;
10119            for _ in 0..sch_count {
10120                let name = cur.read_str()?;
10121                cat.schemas.insert(name);
10122            }
10123        }
10124        // v7.37.42-T2 ζ-B — COMPOSITE types catalog block
10125        // (FILE_VERSION 52+). v51-and-below readers stop at the
10126        // user-schemas block; v52 readers fed a v51 catalog see no
10127        // composite block and default to an empty map.
10128        if version >= 52 {
10129            let ctype_count = cur.read_u32()? as usize;
10130            for _ in 0..ctype_count {
10131                let name = cur.read_str()?;
10132                let field_count = cur.read_u16()? as usize;
10133                let mut fields = Vec::with_capacity(field_count);
10134                let mut field_user_types: Vec<Option<String>> = Vec::with_capacity(field_count);
10135                for _ in 0..field_count {
10136                    let fname = cur.read_str()?;
10137                    let fty = cur.read_data_type()?;
10138                    // v7.39 (round 264) — present from FILE_VERSION 76.
10139                    let ut = if version >= 76 {
10140                        match cur.read_u8()? {
10141                            0 => None,
10142                            1 => Some(cur.read_str()?),
10143                            other => {
10144                                return Err(StorageError::Corrupt(alloc::format!(
10145                                    "composite field user-type tag {other}"
10146                                )));
10147                            }
10148                        }
10149                    } else {
10150                        None
10151                    };
10152                    fields.push((fname, fty));
10153                    field_user_types.push(ut);
10154                }
10155                cat.composite_types.insert(
10156                    name.clone(),
10157                    CompositeDef {
10158                        name,
10159                        fields,
10160                        field_user_types,
10161                    },
10162                );
10163            }
10164        }
10165        // v7.39 (read01 round 50) — COMMENT store (FILE_VERSION 61+).
10166        if version >= 61 {
10167            let comment_count = cur.read_u32()? as usize;
10168            for _ in 0..comment_count {
10169                let key = cur.read_str()?;
10170                let text = cur.read_str_long()?;
10171                cat.comments.insert(key, text);
10172            }
10173        }
10174        // v7.39 (read01 round 60) — non-table ACLs (FILE_VERSION 66+).
10175        if version >= 66 {
10176            let read_acl = |cur: &mut Cursor| -> Result<Vec<AclItem>, StorageError> {
10177                let n = cur.read_u16()? as usize;
10178                let mut acl = Vec::with_capacity(n);
10179                for _ in 0..n {
10180                    let grantee = cur.read_str()?;
10181                    let privs = cur.read_u16()?;
10182                    let grantable = cur.read_u16()?;
10183                    let grantor = cur.read_str()?;
10184                    acl.push(AclItem {
10185                        grantee,
10186                        privs,
10187                        grantable,
10188                        grantor,
10189                    });
10190                }
10191                Ok(acl)
10192            };
10193            let seq_count = cur.read_u32()? as usize;
10194            for _ in 0..seq_count {
10195                let name = cur.read_str()?;
10196                let owner = if cur.read_u8()? == 1 {
10197                    Some(cur.read_str()?)
10198                } else {
10199                    None
10200                };
10201                let acl = read_acl(&mut cur)?;
10202                if let Some(seq) = cat.sequences.get_mut(&name) {
10203                    seq.owner = owner;
10204                    seq.acl = acl;
10205                }
10206            }
10207            cat.schema_acl = read_acl(&mut cur)?;
10208            cat.database_acl = read_acl(&mut cur)?;
10209            // v7.39 (read01 round 61) — FUNCTION owner + ACL (v67+; keyed by
10210            // signature from v68, when overloads became possible).
10211            if version >= 67 {
10212                let fn_count = cur.read_u32()? as usize;
10213                for _ in 0..fn_count {
10214                    let name = cur.read_str()?;
10215                    let owner = if cur.read_u8()? == 1 {
10216                        Some(cur.read_str()?)
10217                    } else {
10218                        None
10219                    };
10220                    let acl = read_acl(&mut cur)?;
10221                    // v7.39 (round 315, V19) — the stored key was computed
10222                    // by whichever formula was current when the image was
10223                    // written. A miss is not "no such function": before the
10224                    // multi-word fix, `f(double precision)` keyed as
10225                    // `f(precision)`, so an older image's grants would land
10226                    // nowhere and vanish silently. Fall back to matching by
10227                    // the old formula, which re-attaches them.
10228                    let target = resolve_stored_function_key(&cat.functions, &name);
10229                    if let Some(k) = target
10230                        && let Some(f) = cat.functions.get_mut(&k)
10231                    {
10232                        f.owner = owner;
10233                        f.acl = acl;
10234                    }
10235                }
10236            }
10237        }
10238        // v7.39 (round 139) — RULE catalog block (FILE_VERSION 71+), read from
10239        // the tail right before the CRC trailer. Pre-71 images stop before it.
10240        if version >= 71 {
10241            let rule_count = cur.read_u32()? as usize;
10242            for _ in 0..rule_count {
10243                let name = cur.read_str()?;
10244                let table = cur.read_str()?;
10245                let event = cur.read_str()?;
10246                let instead = cur.read_u8()? != 0;
10247                let when_condition = cur.read_str()?;
10248                let cmd_count = cur.read_u16()? as usize;
10249                let mut commands = Vec::with_capacity(cmd_count);
10250                for _ in 0..cmd_count {
10251                    commands.push(cur.read_str()?);
10252                }
10253                cat.rules.push(RuleDef {
10254                    name,
10255                    table,
10256                    event,
10257                    instead,
10258                    when_condition,
10259                    commands,
10260                });
10261            }
10262        }
10263        // v7.39 (round 280) — extended-statistics block (FILE_VERSION
10264        // 77+). Pre-77 images stop before it.
10265        if version >= 77 {
10266            let count = cur.read_u32()? as usize;
10267            for _ in 0..count {
10268                let name = cur.read_str()?;
10269                let table = cur.read_str()?;
10270                let nk = cur.read_u16()? as usize;
10271                let mut kinds = Vec::with_capacity(nk);
10272                for _ in 0..nk {
10273                    kinds.push(cur.read_str()?);
10274                }
10275                let nc = cur.read_u16()? as usize;
10276                let mut columns = Vec::with_capacity(nc);
10277                for _ in 0..nc {
10278                    columns.push(cur.read_str()?);
10279                }
10280                cat.statistics_ext.push(StatisticsExtDef {
10281                    name,
10282                    table,
10283                    kinds,
10284                    columns,
10285                });
10286            }
10287        }
10288        // v7.39 (round 287) — large-object block (FILE_VERSION 78+).
10289        // Pre-78 images stop before it.
10290        if version >= 78 {
10291            let count = cur.read_u32()? as usize;
10292            for _ in 0..count {
10293                let oid = cur.read_u32()?;
10294                let len = cur.read_u32()? as usize;
10295                let bytes = cur.read_bytes(len)?;
10296                cat.large_objects.insert(oid, bytes);
10297            }
10298        }
10299        // v7.39 (round 322, V46) — function-attribute block (FILE_VERSION
10300        // 80+). Pre-80 images stop before it and keep PG's defaults.
10301        if version >= 80 {
10302            let count = cur.read_u32()? as usize;
10303            for _ in 0..count {
10304                let key = cur.read_str()?;
10305                let volatility = cur.read_u8()?;
10306                let flags = cur.read_u8()?;
10307                let parallel = cur.read_u8()?;
10308                let cost = f64::from_le_bytes(cur.read_bytes(8)?.try_into().unwrap_or([0; 8]));
10309                let rows = f64::from_le_bytes(cur.read_bytes(8)?.try_into().unwrap_or([0; 8]));
10310                if let Some(f) = cat.functions.get_mut(&key) {
10311                    f.volatility = volatility;
10312                    f.strict = flags & 1 != 0;
10313                    f.security_definer = flags & 2 != 0;
10314                    f.leakproof = flags & 4 != 0;
10315                    f.parallel = parallel;
10316                    f.cost = (!cost.is_nan()).then_some(cost);
10317                    f.rows = (!rows.is_nan()).then_some(rows);
10318                }
10319            }
10320        }
10321        // v7.39 (round 547) — pg_db_role_setting (FILE_VERSION 85+).
10322        // Pre-85 images stop before it and carry no GUC defaults.
10323        if version >= 85 {
10324            let scopes = cur.read_u32()? as usize;
10325            for _ in 0..scopes {
10326                let db = cur.read_str()?;
10327                let role = cur.read_str()?;
10328                let params = cur.read_u32()? as usize;
10329                let mut m: BTreeMap<String, String> = BTreeMap::new();
10330                for _ in 0..params {
10331                    let name = cur.read_str()?;
10332                    let value = cur.read_str()?;
10333                    m.insert(name, value);
10334                }
10335                if !m.is_empty() {
10336                    cat.db_role_settings.insert((db, role), m);
10337                }
10338            }
10339        }
10340        // v7.39 (round 550) — replication slots (FILE_VERSION 86+).
10341        if version >= 86 {
10342            let count = cur.read_u32()? as usize;
10343            for _ in 0..count {
10344                let name = cur.read_str()?;
10345                let plugin = cur.read_str()?;
10346                let slot_type = cur.read_str()?;
10347                cat.replication_slots.insert(name, (plugin, slot_type));
10348            }
10349        }
10350        // v7.38 (read01 P5.05) — v54+ images end with a CRC32C over every
10351        // preceding byte; verify it before accepting the snapshot. Older
10352        // images have no trailer and fall through to the trailing-byte check.
10353        if version >= FILE_VERSION_CRC_TRAILER {
10354            let crc_start = cur.pos;
10355            let stored = cur.read_u32()?;
10356            let computed = spg_crypto::crc32c::crc32c(&buf[..crc_start]);
10357            if computed != stored {
10358                return Err(StorageError::Corrupt(format!(
10359                    "base snapshot CRC mismatch: computed {computed:#010x}, stored {stored:#010x}"
10360                )));
10361            }
10362        }
10363        if cur.pos < buf.len() {
10364            return Err(StorageError::Corrupt(format!(
10365                "trailing bytes: {} unread",
10366                buf.len() - cur.pos
10367            )));
10368        }
10369        Ok(cat)
10370    }
10371}
10372
10373#[cfg(test)]
10374mod tests;